feat(model-registry): added proxy discovery type for mixed Anthropic/OpenAI proxies

- Added `discovery.type: proxy` that hits `GET /v1/models` and routes each model via `supported_endpoint_types` (`anthropic` → `/v1/messages`, `openai` → `/v1/chat/completions`).
- Made provider-level `api` optional when `discovery.type` is `proxy`, since wire protocol is derived per-model.
- Increased discovery fetch timeout from 250ms to 10s to accommodate remote proxies.
- Documented proxy discovery configuration in `docs/models.md`.
This commit is contained in:
can1357
2026-05-30 03:53:36 +02:00
parent 420429df9c
commit 375d100555
3 changed files with 118 additions and 7 deletions
+28 -1
View File
@@ -103,7 +103,7 @@ providers:
### Allowed auth/discovery values
- `auth`: `apiKey` (default), `none`, or `oauth`; for `models.yml` custom models, `oauth` is accepted by schema but does not waive the `apiKey` requirement
- `discovery.type`: `ollama`, `llama.cpp`, or `lm-studio`
- `discovery.type`: `ollama`, `llama.cpp`, `lm-studio`, `openai-models-list`, or `proxy`
## Validation rules (current)
@@ -288,6 +288,33 @@ providers:
type: llama.cpp
```
### Proxy discovery (`discovery.type: proxy`)
For Anthropic+OpenAI-compatible proxies (new-api / one-api / similar)
that expose both `/v1/messages` and `/v1/chat/completions` behind the same
host. Discovery hits `GET /v1/models` (10s timeout, OpenAI-style payload) and
derives each model's `api` from the entry's `supported_endpoint_types`:
- contains `"anthropic"` -> `api: anthropic-messages` (routes via `/v1/messages`)
- contains `"openai"` -> `api: openai-completions` (routes via `/v1/chat/completions`)
- otherwise -> falls back to provider-level `api` if set, else dropped
Provider-level `api` is **optional** with `discovery.type: proxy` because the
per-model wire is auto-detected. The Anthropic SDK strips a trailing `/v1`
from `baseUrl` before appending `/v1/messages`, so a single discovery `baseUrl`
(ending in `/v1`) round-trips correctly to both wires.
```yaml
providers:
newapi-reseller:
baseUrl: https://api.example.com/v1
apiKey: xxxx
authHeader: true # injects Authorization: Bearer for openai models
disableStrictTools: true # most anthropic-fronted proxies reject `strict`
discovery:
type: proxy
```
### Extension provider registration
Extensions can register providers at runtime (`pi.registerProvider(...)`), including:
@@ -192,7 +192,7 @@ function validateProviderConfiguration(
}
}
if (mode === "models-config" && config.discovery && !config.api) {
if (mode === "models-config" && config.discovery && !config.api && config.discovery.type !== "proxy") {
throw new Error(`Provider ${providerName}: "api" is required when discovery is enabled at provider level.`);
}
@@ -1209,13 +1209,17 @@ export class ModelRegistry {
keylessProviders.add(providerName);
}
if (providerConfig.discovery && providerConfig.api) {
if (providerConfig.discovery && (providerConfig.api || providerConfig.discovery.type === "proxy")) {
const disableStrictCompat = providerConfig.disableStrictTools ? { disableStrictTools: true } : undefined;
discoverableProviders.push({
provider: providerName,
api: providerConfig.api as Api,
// Proxy discovery derives per-model api from /v1/models's
// supported_endpoint_types; the provider-level api is only a
// fallback for entries that don't advertise one.
api: (providerConfig.api ?? "openai-completions") as Api,
baseUrl: providerConfig.baseUrl,
headers: providerConfig.headers,
compat: providerConfig.compat,
compat: mergeCompat(providerConfig.compat, disableStrictCompat),
discovery: providerConfig.discovery,
optional: false,
});
@@ -1385,6 +1389,8 @@ export class ModelRegistry {
case "lm-studio":
case "openai-models-list":
return this.#discoverOpenAIModelsList(providerConfig);
case "proxy":
return this.#discoverProxyModels(providerConfig);
}
}
@@ -1711,7 +1717,7 @@ export class ModelRegistry {
const response = await fetch(modelsUrl, {
headers,
signal: AbortSignal.timeout(250),
signal: AbortSignal.timeout(10_000),
});
if (!response.ok) {
throw new Error(`HTTP ${response.status} from ${modelsUrl}`);
@@ -1746,6 +1752,84 @@ export class ModelRegistry {
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
}
/**
* Discover models from an Anthropic+OpenAI-compatible reseller proxy that
* exposes both `/v1/messages` and `/v1/chat/completions`, advertising each
* model's wire capabilities through `supported_endpoint_types` on
* `GET /v1/models` (new-api / one-api-style proxies).
*
* Routing per model:
* supported_endpoint_types: ["anthropic", ...] -> api: "anthropic-messages"
* supported_endpoint_types: ["openai"] -> api: "openai-completions"
* missing / neither -> provider-level api fallback
*
* Anthropic models share the same baseUrl; the Anthropic SDK strips a
* trailing `/v1` itself before appending `/v1/messages`, so the discovery
* URL (which ends in `/v1`) round-trips correctly.
*/
async #discoverProxyModels(providerConfig: DiscoveryProviderConfig): Promise<Model<Api>[]> {
const baseUrl = this.#normalizeOpenAIModelsListBaseUrl(providerConfig.baseUrl);
const modelsUrl = `${baseUrl}/models`;
const headers: Record<string, string> = { ...(providerConfig.headers ?? {}) };
const apiKey = await this.authStorage.getApiKey(providerConfig.provider);
if (apiKey && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== kNoAuth) {
headers.Authorization = `Bearer ${apiKey}`;
}
const response = await fetch(modelsUrl, {
headers,
signal: AbortSignal.timeout(10_000),
});
if (!response.ok) {
throw new Error(`HTTP ${response.status} from ${modelsUrl}`);
}
const payload = (await response.json()) as {
data?: Array<{ id?: string; supported_endpoint_types?: string[] }>;
};
const items = payload.data ?? [];
const discovered: Model<Api>[] = [];
for (const item of items) {
const id = item.id;
if (!id) continue;
const endpoints = item.supported_endpoint_types ?? [];
const api: Api | undefined = endpoints.includes("anthropic")
? "anthropic-messages"
: endpoints.includes("openai")
? "openai-completions"
: providerConfig.api;
if (!api) continue;
const isAnthropic = api === "anthropic-messages";
discovered.push(
enrichModelThinking({
id,
name: id,
api,
provider: providerConfig.provider,
baseUrl,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 8192,
headers,
// OpenAI-compat fields are no-ops on anthropic models; the
// Anthropic SDK ignores them. Provider-level disableStrictTools
// flows in via #applyProviderCompat for the third-party-Anthropic
// path.
compat: isAnthropic
? undefined
: {
supportsStore: false,
supportsDeveloperRole: false,
supportsReasoningEffort: false,
},
}),
);
}
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
}
#normalizeLlamaCppBaseUrl(baseUrl?: string): string {
const defaultBaseUrl = "http://127.0.0.1:8080";
const raw = baseUrl || defaultBaseUrl;
@@ -121,7 +121,7 @@ export const ModelOverrideSchema = z.object({
export type ModelOverride = z.infer<typeof ModelOverrideSchema>;
export const ProviderDiscoverySchema = z.object({
type: z.enum(["ollama", "llama.cpp", "lm-studio", "openai-models-list"]),
type: z.enum(["ollama", "llama.cpp", "lm-studio", "openai-models-list", "proxy"]),
});
export const ProviderAuthSchema = z.enum(["apiKey", "none", "oauth"]);