Merge remote-tracking branch 'origin/farm/9bd5c817/fix-ollama-cloud-output-caps'
This commit is contained in:
@@ -18,6 +18,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama chat requests honoring `omitMaxOutputTokens`, sending `think: false` when reasoning is explicitly disabled, and preserving HTTP 400 response bodies in surfaced errors.
|
||||
- Fixed `AuthStorage.markUsageLimitReached` collapsing "every sibling is momentarily blocked" into "no sibling exists": it now returns `UsageLimitMarkResult` with the earliest sibling block expiry (`retryAtMs`), so retry layers can wait out a short-lived block (60s post-401, 5-min usage-probe) instead of adopting the provider's multi-hour retry-after. `rotateSessionCredential` and the auth-gateway adapt to the new shape.
|
||||
- Fixed Gemini streaming silently presenting truncated or blocked output as a successful `stop`: in-band `{"error":{...}}` events and `promptFeedback.blockReason` chunks were never inspected, and a stream ending without any `finishReason` kept the initialized `stop` — all three now surface as errors (both the API-key and gemini-cli/Antigravity consumers), and the `toolUse` stop-reason override no longer masks `SAFETY`/`MALFORMED_FUNCTION_CALL` finishes that arrive after a valid tool call.
|
||||
- Fixed Gemini/Bedrock error finishes reporting "An unknown error occurred": the raw finish/stop reason (`MALFORMED_FUNCTION_CALL`, `RECITATION`, `guardrail_intervened`, …) is now recorded into the surfaced error message.
|
||||
|
||||
@@ -16,7 +16,12 @@ import type {
|
||||
} from "../types";
|
||||
import { normalizeSystemPrompts } from "../utils";
|
||||
import { AssistantMessageEventStream } from "../utils/event-stream";
|
||||
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
||||
import {
|
||||
type CapturedHttpErrorResponse,
|
||||
finalizeErrorMessage,
|
||||
type RawHttpRequestDump,
|
||||
withHttpStatus,
|
||||
} from "../utils/http-inspector";
|
||||
import { parseStreamingJson } from "../utils/json-parse";
|
||||
import { toolWireSchema } from "../utils/schema/wire";
|
||||
import {
|
||||
@@ -29,6 +34,7 @@ import { transformMessages } from "./transform-messages";
|
||||
|
||||
export interface OllamaChatOptions extends StreamOptions {
|
||||
reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
disableReasoning?: boolean;
|
||||
toolChoice?: ToolChoice;
|
||||
}
|
||||
|
||||
@@ -91,7 +97,14 @@ function normalizeBaseUrl(baseUrl?: string): string {
|
||||
return trimmed.endsWith("/api") ? trimmed.slice(0, -4) : trimmed;
|
||||
}
|
||||
|
||||
function mapReasoning(reasoning: OllamaChatOptions["reasoning"]): boolean | "low" | "medium" | "high" | undefined {
|
||||
function mapReasoning(
|
||||
reasoning: OllamaChatOptions["reasoning"],
|
||||
disableReasoning: boolean | undefined,
|
||||
modelReasoning: boolean,
|
||||
): boolean | "low" | "medium" | "high" | undefined {
|
||||
if (disableReasoning && modelReasoning) {
|
||||
return false;
|
||||
}
|
||||
switch (reasoning) {
|
||||
case "minimal":
|
||||
case "low":
|
||||
@@ -258,7 +271,7 @@ function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefin
|
||||
}
|
||||
|
||||
function createChatBody(model: Model<"ollama-chat">, context: Context, options: OllamaChatOptions | undefined) {
|
||||
const think = mapReasoning(options?.reasoning);
|
||||
const think = mapReasoning(options?.reasoning, options?.disableReasoning, model.reasoning);
|
||||
const toolChoice = mapToolChoice(options?.toolChoice);
|
||||
const selectedTools = selectToolsForToolChoice(context.tools, options?.toolChoice);
|
||||
const tools = convertTools(selectedTools);
|
||||
@@ -268,11 +281,32 @@ function createChatBody(model: Model<"ollama-chat">, context: Context, options:
|
||||
...(tools ? { tools } : {}),
|
||||
...(think !== undefined ? { think } : {}),
|
||||
...(toolChoice !== undefined ? { tool_choice: toolChoice } : {}),
|
||||
...(options?.maxTokens !== undefined ? { options: { num_predict: options.maxTokens } } : {}),
|
||||
...(options?.maxTokens !== undefined && !model.omitMaxOutputTokens
|
||||
? { options: { num_predict: options.maxTokens } }
|
||||
: {}),
|
||||
stream: true,
|
||||
};
|
||||
}
|
||||
|
||||
async function captureHttpErrorResponse(response: Response): Promise<CapturedHttpErrorResponse> {
|
||||
let bodyText: string | undefined;
|
||||
let bodyJson: unknown;
|
||||
try {
|
||||
bodyText = await response.text();
|
||||
if (bodyText.trim()) {
|
||||
try {
|
||||
bodyJson = JSON.parse(bodyText) as unknown;
|
||||
} catch {}
|
||||
}
|
||||
} catch {}
|
||||
return {
|
||||
status: response.status,
|
||||
headers: response.headers,
|
||||
bodyText,
|
||||
bodyJson,
|
||||
};
|
||||
}
|
||||
|
||||
async function* iterateNdjson(stream: ReadableStream<Uint8Array>): AsyncGenerator<OllamaChatChunk> {
|
||||
const reader = stream.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
@@ -376,6 +410,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
||||
let firstTokenTime: number | undefined;
|
||||
const output = createEmptyOutput(model);
|
||||
let rawRequestDump: RawHttpRequestDump | undefined;
|
||||
let capturedErrorResponse: CapturedHttpErrorResponse | undefined;
|
||||
let activeThinkingIndex: number | undefined;
|
||||
let activeTextIndex: number | undefined;
|
||||
const activeToolIndices = new Set<number>();
|
||||
@@ -503,7 +538,8 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
||||
fetch: options.fetch,
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`HTTP ${response.status} from ${baseUrl}/api/chat`);
|
||||
capturedErrorResponse = await captureHttpErrorResponse(response);
|
||||
throw withHttpStatus(new Error(`HTTP ${response.status} from ${baseUrl}/api/chat`), response.status);
|
||||
}
|
||||
if (!response.body) {
|
||||
throw new Error("Ollama returned an empty response body");
|
||||
@@ -631,7 +667,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
||||
}
|
||||
output.stopReason = options.signal?.aborted ? "aborted" : "error";
|
||||
output.errorStatus = extractHttpStatusFromError(error);
|
||||
output.errorMessage = await finalizeErrorMessage(error, rawRequestDump);
|
||||
output.errorMessage = await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse);
|
||||
output.duration = Date.now() - startTime;
|
||||
if (firstTokenTime) {
|
||||
output.ttft = firstTokenTime - startTime;
|
||||
|
||||
@@ -976,6 +976,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
return castApi<"ollama-chat">({
|
||||
...base,
|
||||
reasoning: resolveOpenAiReasoningEffort(model, options),
|
||||
disableReasoning: options?.disableReasoning,
|
||||
toolChoice: options?.toolChoice,
|
||||
});
|
||||
|
||||
|
||||
@@ -22,4 +22,5 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Anthropic official-endpoint detection to require strict HTTPS hostname matching so non-official or lookalike URLs are no longer treated as official Anthropic hosts
|
||||
- Fixed Ollama Cloud dynamic discovery so same-id matches from other providers no longer supply context-window or max-output-token limits for discovered models.
|
||||
- Wired `@oh-my-pi/pi-catalog` into the release publish package list, tarball install smoke test, and root `bun generate-models` script.
|
||||
|
||||
@@ -91,7 +91,8 @@ export function ollamaCloudModelManagerOptions(
|
||||
): ModelManagerOptions<"ollama-chat"> {
|
||||
const apiKey = config?.apiKey;
|
||||
const baseUrl = normalizeOllamaCloudBaseUrl(config?.baseUrl);
|
||||
const resolveReference = createReferenceResolver(createBundledReferenceMap<"ollama-chat">("ollama-cloud"));
|
||||
const providerReferences = createBundledReferenceMap<"ollama-chat">("ollama-cloud");
|
||||
const resolveReference = createReferenceResolver(providerReferences);
|
||||
return {
|
||||
providerId: "ollama-cloud",
|
||||
fetchDynamicModels: async () => {
|
||||
@@ -115,6 +116,7 @@ export function ollamaCloudModelManagerOptions(
|
||||
if (!id) {
|
||||
return undefined;
|
||||
}
|
||||
const providerReference = providerReferences.get(id);
|
||||
const reference = resolveReference(id);
|
||||
let metadata: OllamaShowResponse | undefined;
|
||||
try {
|
||||
@@ -123,7 +125,8 @@ export function ollamaCloudModelManagerOptions(
|
||||
metadata = undefined;
|
||||
}
|
||||
const capabilities = metadata?.capabilities;
|
||||
const contextWindow = getContextWindow(metadata?.model_info) ?? reference?.contextWindow ?? 128000;
|
||||
const contextWindow =
|
||||
getContextWindow(metadata?.model_info) ?? providerReference?.contextWindow ?? 128000;
|
||||
const reasoning = capabilities ? capabilities.includes("thinking") : (reference?.reasoning ?? false);
|
||||
const thinking = capabilities ? getThinkingConfig(capabilities) : reference?.thinking;
|
||||
const input = capabilities
|
||||
@@ -143,7 +146,7 @@ export function ollamaCloudModelManagerOptions(
|
||||
input,
|
||||
cost: reference?.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow,
|
||||
maxTokens: reference?.maxTokens ?? Math.min(contextWindow, 8192),
|
||||
maxTokens: providerReference?.maxTokens ?? Math.min(contextWindow, 8192),
|
||||
};
|
||||
}),
|
||||
);
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
import { expect, test, vi } from "bun:test";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
|
||||
import { ollamaCloudModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/ollama";
|
||||
import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
const cloudModel: Model<"ollama-chat"> = {
|
||||
id: "deepseek-v4-flash",
|
||||
name: "DeepSeek V4 Flash",
|
||||
api: "ollama-chat",
|
||||
provider: "ollama-cloud",
|
||||
baseUrl: "https://ollama.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 8_192,
|
||||
};
|
||||
|
||||
function createNdjsonResponse(lines: unknown[]): Response {
|
||||
const body = `${lines.map(line => JSON.stringify(line)).join("\n")}\n`;
|
||||
return new Response(body, { status: 200, headers: { "Content-Type": "application/x-ndjson" } });
|
||||
}
|
||||
|
||||
test("ollama-cloud discovery does not inherit unsafe cross-provider maxTokens", async () => {
|
||||
const fetchMock: FetchImpl = vi.fn(async (input, _init) => {
|
||||
const url = String(input);
|
||||
if (url === "https://ollama.com/api/tags") {
|
||||
return new Response(JSON.stringify({ models: [{ name: "deepseek-v4-flash" }] }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
if (url === "https://ollama.com/api/show") {
|
||||
return new Response(JSON.stringify({ capabilities: ["completion"] }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
});
|
||||
|
||||
const options = ollamaCloudModelManagerOptions({ apiKey: "cloud-test-key", fetch: fetchMock });
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
const model = models?.find(candidate => candidate.id === "deepseek-v4-flash");
|
||||
|
||||
expect(model?.contextWindow).toBe(128000);
|
||||
expect(model?.maxTokens).toBe(8192);
|
||||
});
|
||||
|
||||
test("ollama-chat omits num_predict when model opts out of max output tokens", async () => {
|
||||
let requestBody: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
|
||||
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
|
||||
return createNdjsonResponse([
|
||||
{ model: "deepseek-v4-flash", message: { role: "assistant", content: "ok" }, done: false },
|
||||
{ model: "deepseek-v4-flash", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
|
||||
]);
|
||||
});
|
||||
|
||||
const model: Model<"ollama-chat"> = { ...cloudModel, omitMaxOutputTokens: true };
|
||||
await streamSimple(
|
||||
model,
|
||||
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
|
||||
{ apiKey: "cloud-test-key", fetch: fetchMock, maxTokens: 384000 },
|
||||
).result();
|
||||
|
||||
expect(requestBody).not.toHaveProperty("options");
|
||||
});
|
||||
|
||||
test("ollama-chat sends think false when reasoning is disabled", async () => {
|
||||
let requestBody: Record<string, unknown> | undefined;
|
||||
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
|
||||
requestBody = JSON.parse(String(init?.body ?? "{}")) as Record<string, unknown>;
|
||||
return createNdjsonResponse([
|
||||
{ model: "deepseek-v4-flash", message: { role: "assistant", content: "ok" }, done: false },
|
||||
{ model: "deepseek-v4-flash", done: true, done_reason: "stop", prompt_eval_count: 1, eval_count: 1 },
|
||||
]);
|
||||
});
|
||||
|
||||
await streamSimple(
|
||||
cloudModel,
|
||||
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
|
||||
{ apiKey: "cloud-test-key", fetch: fetchMock, disableReasoning: true },
|
||||
).result();
|
||||
|
||||
expect(requestBody?.think).toBe(false);
|
||||
});
|
||||
|
||||
test("ollama-chat surfaces HTTP 400 response bodies", async () => {
|
||||
const fetchMock: FetchImpl = vi.fn(
|
||||
async () =>
|
||||
new Response(
|
||||
JSON.stringify({ error: { message: "num_predict exceeds model cap", type: "invalid_request" } }),
|
||||
{
|
||||
status: 400,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
},
|
||||
),
|
||||
);
|
||||
|
||||
const response = await streamSimple(
|
||||
cloudModel,
|
||||
{ messages: [{ role: "user", content: "Reply ok", timestamp: Date.now() }] },
|
||||
{ apiKey: "cloud-test-key", fetch: fetchMock },
|
||||
).result();
|
||||
|
||||
expect(response.stopReason).toBe("error");
|
||||
expect(response.errorStatus).toBe(400);
|
||||
expect(response.errorMessage).toContain("HTTP 400 from https://ollama.com/api/chat");
|
||||
expect(response.errorMessage).toContain("num_predict exceeds model cap");
|
||||
});
|
||||
Reference in New Issue
Block a user