fix: enable ollama cloud cache metadata

fixes #937
This commit is contained in:
can1357
2026-05-06 19:32:39 +02:00
parent f4732522cd
commit 071f15895a
5 changed files with 108 additions and 8 deletions
+1
View File
@@ -5,6 +5,7 @@
### Fixed
- Fixed VLLM model discovery to use `max_model_len` as the context window when the endpoint reports it.
- Fixed custom Ollama Cloud/local-proxy model aliases (for example `deepseek-v4-pro:cloud`) to inherit bundled cache-pricing metadata when the upstream model is known ([#937](https://github.com/can1357/oh-my-pi/issues/937)).
- Fixed local Ollama model discovery to apply `/api/show` thinking and vision capabilities in addition to native context windows ([#928](https://github.com/can1357/oh-my-pi/issues/928)).
## [14.7.0] - 2026-05-04
@@ -298,7 +298,10 @@ function getOllamaThinkingConfig(capabilities: string[] | undefined): ThinkingCo
* context and capability metadata from the response. Returns `undefined` when
* the endpoint is unavailable so callers can layer their own fallback.
*/
async function fetchOllamaShowMetadata(nativeBaseUrl: string, modelId: string): Promise<OllamaShowMetadata | undefined> {
async function fetchOllamaShowMetadata(
nativeBaseUrl: string,
modelId: string,
): Promise<OllamaShowMetadata | undefined> {
try {
const response = await fetch(`${nativeBaseUrl}/api/show`, {
method: "POST",
@@ -50,6 +50,7 @@ const TRAILING_CANONICAL_MARKERS = [
"minimal",
"xhigh",
"free",
"cloud",
"exacto",
"nitro",
"original",
@@ -733,6 +733,69 @@ function buildCustomModelOverlay(
};
}
// Custom provider entries often front a known upstream model through a local proxy.
// Use bundled metadata for missing pricing/capability fields, but keep the custom transport.
function shouldReplaceCustomReference(existing: Model<Api> | undefined, candidate: Model<Api>): boolean {
if (!existing) return true;
if (candidate.contextWindow !== existing.contextWindow) {
return candidate.contextWindow > existing.contextWindow;
}
if (candidate.maxTokens !== existing.maxTokens) {
return candidate.maxTokens > existing.maxTokens;
}
const existingHasCachePricing = existing.cost.cacheRead > 0 || existing.cost.cacheWrite > 0;
const candidateHasCachePricing = candidate.cost.cacheRead > 0 || candidate.cost.cacheWrite > 0;
if (candidateHasCachePricing !== existingHasCachePricing) {
return candidateHasCachePricing;
}
return existing.provider !== "openai" && candidate.provider === "openai";
}
function buildCustomReferenceMap(): Map<string, Model<Api>> {
const references = new Map<string, Model<Api>>();
for (const provider of getBundledProviders()) {
for (const model of getBundledModels(provider as Parameters<typeof getBundledModels>[0])) {
const candidate = model as Model<Api>;
if (shouldReplaceCustomReference(references.get(candidate.id), candidate)) {
references.set(candidate.id, candidate);
}
}
}
return references;
}
const customReferenceMap = buildCustomReferenceMap();
function getCustomReferenceCandidateIds(modelId: string): string[] {
const candidates = new Set<string>();
const queue = [modelId];
for (let index = 0; index < queue.length; index += 1) {
const candidate = queue[index]?.trim();
if (!candidate || candidates.has(candidate)) continue;
candidates.add(candidate);
for (const suffix of [":cloud", "-cloud"] as const) {
if (candidate.toLowerCase().endsWith(suffix)) {
queue.push(candidate.slice(0, -suffix.length));
}
}
const colonToDash = candidate.replace(/:/g, "-");
if (colonToDash !== candidate) {
queue.push(colonToDash);
}
}
return [...candidates];
}
function resolveCustomModelReference(modelId: string): Model<Api> | undefined {
for (const candidate of getCustomReferenceCandidateIds(modelId)) {
const reference = customReferenceMap.get(candidate);
if (reference) return reference;
}
return undefined;
}
function applyStandaloneCustomModelPolicies(model: CustomModelOverlay): CustomModelOverlay {
if (model.id !== "gpt-5.4" || model.provider === "github-copilot" || model.contextWindow !== undefined) {
return model;
@@ -742,23 +805,27 @@ function applyStandaloneCustomModelPolicies(model: CustomModelOverlay): CustomMo
function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuildOptions): Model<Api> {
const resolvedModel = options.useDefaults ? applyStandaloneCustomModelPolicies(model) : model;
const reference = options.useDefaults ? resolveCustomModelReference(resolvedModel.id) : undefined;
const cost =
resolvedModel.cost ?? (options.useDefaults ? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } : undefined);
const input = resolvedModel.input ?? (options.useDefaults ? ["text"] : undefined);
resolvedModel.cost ??
reference?.cost ??
(options.useDefaults ? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } : undefined);
const input = resolvedModel.input ?? reference?.input ?? (options.useDefaults ? ["text"] : undefined);
return enrichModelThinking({
id: resolvedModel.id,
name: resolvedModel.name ?? (options.useDefaults ? resolvedModel.id : undefined),
api: resolvedModel.api,
provider: resolvedModel.provider,
baseUrl: resolvedModel.baseUrl,
reasoning: resolvedModel.reasoning ?? (options.useDefaults ? false : undefined),
thinking: resolvedModel.thinking,
reasoning: resolvedModel.reasoning ?? reference?.reasoning ?? (options.useDefaults ? false : undefined),
thinking: resolvedModel.thinking ?? reference?.thinking,
input: input as ("text" | "image")[],
cost,
contextWindow: resolvedModel.contextWindow ?? (options.useDefaults ? 128000 : undefined),
maxTokens: resolvedModel.maxTokens ?? (options.useDefaults ? 16384 : undefined),
contextWindow:
resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : undefined),
maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : undefined),
headers: resolvedModel.headers,
compat: resolvedModel.compat,
compat: mergeCompat(reference?.compat, resolvedModel.compat),
contextPromotionTarget: resolvedModel.contextPromotionTarget,
premiumMultiplier: resolvedModel.premiumMultiplier,
isOAuth: resolvedModel.isOAuth,
@@ -226,6 +226,34 @@ describe("ModelRegistry", () => {
expect(variants.some(variant => variant.selector === "openrouter/z-ai/glm-4.7-20251222:nitro")).toBe(true);
});
test("uses bundled metadata for Ollama cloud aliases in custom local-proxy configs", () => {
writeRawModelsJson({
ollama: {
baseUrl: "http://127.0.0.1:11434/v1",
api: "openai-completions",
auth: "none",
models: [
{
id: "deepseek-v4-pro:cloud",
name: "DeepSeek V4 Pro (Ollama Cloud)",
reasoning: true,
input: ["text"],
contextWindow: 1_048_576,
maxTokens: 65_536,
},
],
},
});
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const model = registry.find("ollama", "deepseek-v4-pro:cloud");
const variants = registry.getCanonicalVariants("deepseek-v4-pro");
expect(model?.cost.cacheRead).toBeGreaterThan(0);
expect(model?.thinking?.maxLevel).toBe(Effort.XHigh);
expect(variants.some(variant => variant.selector === "ollama/deepseek-v4-pro:cloud")).toBe(true);
});
test("collapses anthropic latest aliases into the best upstream claude family id", () => {
writeRawModelsJson({
demo: providerConfig("https://demo.example.com/v1", [