feat(model-registry): added proxy discovery type for mixed Anthropic/OpenAI proxies
- Added `discovery.type: proxy` that hits `GET /v1/models` and routes each model via `supported_endpoint_types` (`anthropic` → `/v1/messages`, `openai` → `/v1/chat/completions`). - Made provider-level `api` optional when `discovery.type` is `proxy`, since wire protocol is derived per-model. - Increased discovery fetch timeout from 250ms to 10s to accommodate remote proxies. - Documented proxy discovery configuration in `docs/models.md`.
This commit is contained in:
+28
-1
@@ -103,7 +103,7 @@ providers:
|
||||
### Allowed auth/discovery values
|
||||
|
||||
- `auth`: `apiKey` (default), `none`, or `oauth`; for `models.yml` custom models, `oauth` is accepted by schema but does not waive the `apiKey` requirement
|
||||
- `discovery.type`: `ollama`, `llama.cpp`, or `lm-studio`
|
||||
- `discovery.type`: `ollama`, `llama.cpp`, `lm-studio`, `openai-models-list`, or `proxy`
|
||||
|
||||
## Validation rules (current)
|
||||
|
||||
@@ -288,6 +288,33 @@ providers:
|
||||
type: llama.cpp
|
||||
```
|
||||
|
||||
### Proxy discovery (`discovery.type: proxy`)
|
||||
|
||||
For Anthropic+OpenAI-compatible proxies (new-api / one-api / similar)
|
||||
that expose both `/v1/messages` and `/v1/chat/completions` behind the same
|
||||
host. Discovery hits `GET /v1/models` (10s timeout, OpenAI-style payload) and
|
||||
derives each model's `api` from the entry's `supported_endpoint_types`:
|
||||
|
||||
- contains `"anthropic"` -> `api: anthropic-messages` (routes via `/v1/messages`)
|
||||
- contains `"openai"` -> `api: openai-completions` (routes via `/v1/chat/completions`)
|
||||
- otherwise -> falls back to provider-level `api` if set, else dropped
|
||||
|
||||
Provider-level `api` is **optional** with `discovery.type: proxy` because the
|
||||
per-model wire is auto-detected. The Anthropic SDK strips a trailing `/v1`
|
||||
from `baseUrl` before appending `/v1/messages`, so a single discovery `baseUrl`
|
||||
(ending in `/v1`) round-trips correctly to both wires.
|
||||
|
||||
```yaml
|
||||
providers:
|
||||
newapi-reseller:
|
||||
baseUrl: https://api.example.com/v1
|
||||
apiKey: xxxx
|
||||
authHeader: true # injects Authorization: Bearer for openai models
|
||||
disableStrictTools: true # most anthropic-fronted proxies reject `strict`
|
||||
discovery:
|
||||
type: proxy
|
||||
```
|
||||
|
||||
### Extension provider registration
|
||||
|
||||
Extensions can register providers at runtime (`pi.registerProvider(...)`), including:
|
||||
|
||||
@@ -192,7 +192,7 @@ function validateProviderConfiguration(
|
||||
}
|
||||
}
|
||||
|
||||
if (mode === "models-config" && config.discovery && !config.api) {
|
||||
if (mode === "models-config" && config.discovery && !config.api && config.discovery.type !== "proxy") {
|
||||
throw new Error(`Provider ${providerName}: "api" is required when discovery is enabled at provider level.`);
|
||||
}
|
||||
|
||||
@@ -1209,13 +1209,17 @@ export class ModelRegistry {
|
||||
keylessProviders.add(providerName);
|
||||
}
|
||||
|
||||
if (providerConfig.discovery && providerConfig.api) {
|
||||
if (providerConfig.discovery && (providerConfig.api || providerConfig.discovery.type === "proxy")) {
|
||||
const disableStrictCompat = providerConfig.disableStrictTools ? { disableStrictTools: true } : undefined;
|
||||
discoverableProviders.push({
|
||||
provider: providerName,
|
||||
api: providerConfig.api as Api,
|
||||
// Proxy discovery derives per-model api from /v1/models's
|
||||
// supported_endpoint_types; the provider-level api is only a
|
||||
// fallback for entries that don't advertise one.
|
||||
api: (providerConfig.api ?? "openai-completions") as Api,
|
||||
baseUrl: providerConfig.baseUrl,
|
||||
headers: providerConfig.headers,
|
||||
compat: providerConfig.compat,
|
||||
compat: mergeCompat(providerConfig.compat, disableStrictCompat),
|
||||
discovery: providerConfig.discovery,
|
||||
optional: false,
|
||||
});
|
||||
@@ -1385,6 +1389,8 @@ export class ModelRegistry {
|
||||
case "lm-studio":
|
||||
case "openai-models-list":
|
||||
return this.#discoverOpenAIModelsList(providerConfig);
|
||||
case "proxy":
|
||||
return this.#discoverProxyModels(providerConfig);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1711,7 +1717,7 @@ export class ModelRegistry {
|
||||
|
||||
const response = await fetch(modelsUrl, {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(250),
|
||||
signal: AbortSignal.timeout(10_000),
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`HTTP ${response.status} from ${modelsUrl}`);
|
||||
@@ -1746,6 +1752,84 @@ export class ModelRegistry {
|
||||
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
|
||||
}
|
||||
|
||||
/**
|
||||
* Discover models from an Anthropic+OpenAI-compatible reseller proxy that
|
||||
* exposes both `/v1/messages` and `/v1/chat/completions`, advertising each
|
||||
* model's wire capabilities through `supported_endpoint_types` on
|
||||
* `GET /v1/models` (new-api / one-api-style proxies).
|
||||
*
|
||||
* Routing per model:
|
||||
* supported_endpoint_types: ["anthropic", ...] -> api: "anthropic-messages"
|
||||
* supported_endpoint_types: ["openai"] -> api: "openai-completions"
|
||||
* missing / neither -> provider-level api fallback
|
||||
*
|
||||
* Anthropic models share the same baseUrl; the Anthropic SDK strips a
|
||||
* trailing `/v1` itself before appending `/v1/messages`, so the discovery
|
||||
* URL (which ends in `/v1`) round-trips correctly.
|
||||
*/
|
||||
async #discoverProxyModels(providerConfig: DiscoveryProviderConfig): Promise<Model<Api>[]> {
|
||||
const baseUrl = this.#normalizeOpenAIModelsListBaseUrl(providerConfig.baseUrl);
|
||||
const modelsUrl = `${baseUrl}/models`;
|
||||
|
||||
const headers: Record<string, string> = { ...(providerConfig.headers ?? {}) };
|
||||
const apiKey = await this.authStorage.getApiKey(providerConfig.provider);
|
||||
if (apiKey && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== kNoAuth) {
|
||||
headers.Authorization = `Bearer ${apiKey}`;
|
||||
}
|
||||
|
||||
const response = await fetch(modelsUrl, {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(10_000),
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`HTTP ${response.status} from ${modelsUrl}`);
|
||||
}
|
||||
const payload = (await response.json()) as {
|
||||
data?: Array<{ id?: string; supported_endpoint_types?: string[] }>;
|
||||
};
|
||||
const items = payload.data ?? [];
|
||||
const discovered: Model<Api>[] = [];
|
||||
for (const item of items) {
|
||||
const id = item.id;
|
||||
if (!id) continue;
|
||||
const endpoints = item.supported_endpoint_types ?? [];
|
||||
const api: Api | undefined = endpoints.includes("anthropic")
|
||||
? "anthropic-messages"
|
||||
: endpoints.includes("openai")
|
||||
? "openai-completions"
|
||||
: providerConfig.api;
|
||||
if (!api) continue;
|
||||
const isAnthropic = api === "anthropic-messages";
|
||||
discovered.push(
|
||||
enrichModelThinking({
|
||||
id,
|
||||
name: id,
|
||||
api,
|
||||
provider: providerConfig.provider,
|
||||
baseUrl,
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128000,
|
||||
maxTokens: 8192,
|
||||
headers,
|
||||
// OpenAI-compat fields are no-ops on anthropic models; the
|
||||
// Anthropic SDK ignores them. Provider-level disableStrictTools
|
||||
// flows in via #applyProviderCompat for the third-party-Anthropic
|
||||
// path.
|
||||
compat: isAnthropic
|
||||
? undefined
|
||||
: {
|
||||
supportsStore: false,
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: false,
|
||||
},
|
||||
}),
|
||||
);
|
||||
}
|
||||
return this.#applyProviderModelOverrides(providerConfig.provider, discovered);
|
||||
}
|
||||
|
||||
#normalizeLlamaCppBaseUrl(baseUrl?: string): string {
|
||||
const defaultBaseUrl = "http://127.0.0.1:8080";
|
||||
const raw = baseUrl || defaultBaseUrl;
|
||||
|
||||
@@ -121,7 +121,7 @@ export const ModelOverrideSchema = z.object({
|
||||
export type ModelOverride = z.infer<typeof ModelOverrideSchema>;
|
||||
|
||||
export const ProviderDiscoverySchema = z.object({
|
||||
type: z.enum(["ollama", "llama.cpp", "lm-studio", "openai-models-list"]),
|
||||
type: z.enum(["ollama", "llama.cpp", "lm-studio", "openai-models-list", "proxy"]),
|
||||
});
|
||||
|
||||
export const ProviderAuthSchema = z.enum(["apiKey", "none", "oauth"]);
|
||||
|
||||
Reference in New Issue
Block a user