From 80862b79da54e452d7da61ab321a92f80ff6bdc0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 25 Jun 2026 11:57:51 +0000 Subject: [PATCH] fix(agent): handled ollama-cloud task backoff Added ollama-cloud subagent concurrency limiting, role fallback-chain inheritance, and visible empty length errors for native Ollama responses. Fixes #3464 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/ollama.ts | 20 ++ .../test/ollama-cloud-provider.test.ts | 19 ++ packages/coding-agent/CHANGELOG.md | 1 + .../src/config/settings-schema.ts | 11 + .../coding-agent/src/session/agent-session.ts | 11 +- packages/coding-agent/src/task/executor.ts | 45 +++- ...sue-3464-ollama-cloud-task-backoff.test.ts | 220 ++++++++++++++++++ 8 files changed, 329 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/test/issue-3464-ollama-cloud-task-backoff.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3071b87d3..7b4050ca9 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Ollama/Ollama Cloud native chat responses that finish with `done_reason: "length"` and no assistant content surfacing as a normal empty stop; they now become a context-window error instead of entering empty-stop retry recovery. ([#3464](https://github.com/can1357/oh-my-pi/issues/3464)) + ## [16.1.19] - 2026-06-25 ### Fixed diff --git a/packages/ai/src/providers/ollama.ts b/packages/ai/src/providers/ollama.ts index 0080eed46..4345814a6 100644 --- a/packages/ai/src/providers/ollama.ts +++ b/packages/ai/src/providers/ollama.ts @@ -437,6 +437,17 @@ function mapDoneReason(doneReason: string | undefined, output: AssistantMessage) return "stop"; } +const EMPTY_OLLAMA_LENGTH_COMPLETION_MESSAGE = + "Model returned no content: prompt filled the context window; raise Ollama num_ctx or shorten the prompt."; + +function hasVisibleAssistantContent(output: AssistantMessage): boolean { + return output.content.some(block => { + if (block.type === "text") return block.text.trim().length > 0; + if (block.type === "thinking") return block.thinking.trim().length > 0; + return block.type === "toolCall"; + }); +} + const OLLAMA_RETRY_DELAYS_MS = [2_000, 5_000, 10_000]; export const streamOllama: StreamFunction<"ollama-chat"> = ( @@ -702,6 +713,10 @@ export const streamOllama: StreamFunction<"ollama-chat"> = ( } endActiveThinkingBlock(); endActiveTextBlock(); + if (output.stopReason === "length" && !hasVisibleAssistantContent(output)) { + output.stopReason = "error"; + output.errorMessage = EMPTY_OLLAMA_LENGTH_COMPLETION_MESSAGE; + } // Tool calls always mean "execute and continue" in the OpenAI/Ollama contract. // If the turn produced tool-call blocks but reported a natural `stop`, promote // to `toolUse` so the agent loop runs them (it gates execution on the stop @@ -713,6 +728,11 @@ export const streamOllama: StreamFunction<"ollama-chat"> = ( if (firstTokenTime) { output.ttft = firstTokenTime - startTime; } + if (output.stopReason === "error") { + stream.push({ type: "error", reason: "error", error: output }); + stream.end(); + return; + } const doneReason = output.stopReason === "length" ? "length" : output.stopReason === "toolUse" ? "toolUse" : "stop"; stream.push({ type: "done", reason: doneReason, message: output }); diff --git a/packages/catalog/test/ollama-cloud-provider.test.ts b/packages/catalog/test/ollama-cloud-provider.test.ts index 35d43c212..4f640c145 100644 --- a/packages/catalog/test/ollama-cloud-provider.test.ts +++ b/packages/catalog/test/ollama-cloud-provider.test.ts @@ -267,6 +267,25 @@ describe("ollama-cloud provider support", () => { ]); }); + test("surfaces empty length completions as context-window errors", async () => { + const fetchMock: FetchImpl = vi.fn(async () => + createNdjsonResponse([ + { model: "gpt-oss:120b", done: true, done_reason: "length", prompt_eval_count: 1000, eval_count: 0 }, + ]), + ); + + const result = await stream( + cloudModel, + { + messages: [{ role: "user", content: "Large task prompt", timestamp: Date.now() }], + }, + { apiKey: "cloud-test-key", fetch: fetchMock }, + ).result(); + + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toContain("prompt filled the context window"); + }); + test("sends max for GLM-5.2 xhigh reasoning on Ollama Cloud", async () => { let requestBody: Record | undefined; const fetchMock: FetchImpl = vi.fn(async (_input, init) => { diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d5afd8188..ac211141d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,7 @@ ### Fixed +- Fixed ollama-cloud task/subagent fan-out exceeding the provider's three-request concurrency cap by adding a provider-specific subagent limiter, and let configured task/smol/advisor model roles inherit the default retry fallback chain when they do not define their own chain. ([#3464](https://github.com/can1357/oh-my-pi/issues/3464)) - Fixed `omp install ` failing extension validation in compiled-binary mode with `Cannot find module '@(scope)/pi-ai/oauth' from '/src/oauth.ts'` (and any other non-wildcard pi-* subpath import like `@oh-my-pi/pi-coding-agent/tools`). The bundled-registry override map seeded by `__buildLegacyPiPackageRootOverrides` only covered the bare package roots, so `rewriteLegacyPiImports` rewrote `@(scope)/pi-ai/oauth` to `@oh-my-pi/pi-ai/oauth`, fell through to `Bun.resolveSync` (which bunfs can't satisfy on Bun 1.3.14+), then left the original specifier alone — at which point Bun's native resolver failed because most plugins declare `@(scope)/pi-ai` as a `peerDependency` only and never materialize a real install. The new `scripts/generate-legacy-pi-bundled-registry.ts` reads every bundled pi-* package's non-wildcard `exports` field and emits both the heavy `legacy-pi-bundled-registry.ts` (static imports + map) and a light `legacy-pi-bundled-keys.ts` (statically imported by `legacy-pi-compat.ts` to seed the override map without the cascade through `legacy-pi-coding-agent-shim → ../index → export/html/...`). `scripts/build-binary.ts` now runs the generator before `bun build --compile`. ([#3442](https://github.com/can1357/oh-my-pi/issues/3442)) - Fixed `skill://` tool resolution losing loaded session skills when a tool runs outside the session-initialization module state. Internal URL resolution now prefers the caller's `session.skills` snapshot before falling back to the process-global skill list, so `read skill://` works across tool execution boundaries. ([#3436](https://github.com/can1357/oh-my-pi/issues/3436)) - Fixed `@image` mentions on OpenAI Codex Responses (chatgpt.com `gpt-5.5` and siblings) failing with `Codex error event: [OneOfParam] [input[N].content[…]] [invalid_enum_value] Invalid value: 'input_image'. Supported values are: 'input_text'.`. `convertToLlm` for `fileMention` always emitted a `developer`-role message, so the auto-attached image landed in a Responses content array that the Codex backend (and OpenAI Responses generally) only allows to carry `input_text`. #3421's previous fix only stopped the Codex Responses Lite header from going out on image-bearing turns; the full transport kept rejecting the same payload. `convertToLlm` now splits a mixed-content `fileMention` into two messages — text-only files stay on `developer` (so the auto-read context keeps instruction priority), while image-bearing files ride on `user` (the only Responses content slot that accepts `input_image`). ([#3443](https://github.com/can1357/oh-my-pi/issues/3443)) diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index fea682b9d..46b7d1e2d 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -4041,6 +4041,17 @@ export const SETTINGS_SCHEMA = { }, // Provider selection + "providers.ollama-cloud.maxConcurrency": { + type: "number", + default: 3, + ui: { + tab: "providers", + group: "Services", + label: "Ollama Cloud Max Concurrency", + description: + "Maximum concurrent Ollama Cloud subagent runs per process; 0 disables the provider-specific limit", + }, + }, "providers.webSearch": { type: "enum", values: SEARCH_PROVIDER_PREFERENCES, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 168d64b3b..8c5683b95 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -10619,7 +10619,16 @@ export class AgentSession { #getRetryFallbackChains(): RetryFallbackChains { const configuredChains = this.settings.get("retry.fallbackChains"); if (!configuredChains || typeof configuredChains !== "object") return {}; - return configuredChains as RetryFallbackChains; + const chains: RetryFallbackChains = { ...(configuredChains as RetryFallbackChains) }; + const defaultChain = chains.default; + if (Array.isArray(defaultChain)) { + for (const role of Object.keys(this.settings.getModelRoles())) { + if (role !== "default" && chains[role] === undefined) { + chains[role] = defaultChain; + } + } + } + return chains; } #validateRetryFallbackChains(): void { diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 6837f5d10..41cc0a2e4 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -50,12 +50,12 @@ import { type OutputValidator, summarizeValidationFailure, } from "../tools/output-schema-validator"; - import { type ReportFindingDetails, toReviewFinding } from "../tools/review"; import { ToolAbortError } from "../tools/tool-errors"; import type { EventBus } from "../utils/event-bus"; import { buildNamedToolChoice } from "../utils/tool-choice"; import type { WorkspaceTree } from "../workspace-tree"; +import { Semaphore } from "./parallel"; import { subprocessToolRegistry } from "./subprocess-tool-registry"; import { type AgentDefinition, @@ -194,6 +194,35 @@ function installSubagentRetryFallbackChain(args: { return role; } +const PROVIDER_MAX_CONCURRENCY_SETTINGS: Record = { + "ollama-cloud": "providers.ollama-cloud.maxConcurrency", +}; + +interface ProviderSemaphoreEntry { + limit: number; + semaphore: Semaphore; +} + +const providerSemaphores = new Map(); + +function getProviderConcurrencyLimit(settings: Settings, provider: string): number { + const settingPath = PROVIDER_MAX_CONCURRENCY_SETTINGS[provider]; + if (!settingPath) return 0; + const raw = settings.get(settingPath); + const limit = Number.isFinite(raw) ? Math.trunc(raw) : 0; + return limit > 0 ? limit : 0; +} + +function getProviderSemaphore(settings: Settings, provider: string): Semaphore | undefined { + const limit = getProviderConcurrencyLimit(settings, provider); + if (limit <= 0) return undefined; + const existing = providerSemaphores.get(provider); + if (existing?.limit === limit) return existing.semaphore; + const semaphore = new Semaphore(limit); + providerSemaphores.set(provider, { limit, semaphore }); + return semaphore; +} + function renderIrcPeerRoster(selfId: string): string { const peers = AgentRegistry.global() .list() @@ -1889,6 +1918,8 @@ export async function runSubprocess(options: ExecutorOptions): Promise; + resolve: () => void; +} + +function deferred(): Deferred { + const { promise, resolve } = Promise.withResolvers(); + return { promise, resolve }; +} + +function createSessionResult(session: AgentSession): CreateAgentSessionResult { + return { + session, + extensionsResult: { extensions: [], errors: [], runtime: {} as unknown } as LoadExtensionsResult, + setToolUIContext: () => {}, + eventBus: new EventBus(), + }; +} + +function createGateSession(onPrompt: () => Promise): MockPromptSession { + const listeners: Array<(event: AgentSessionEvent) => void> = []; + const session = { + agent: { state: { systemPrompt: ["test"] } }, + state: { messages: [] }, + extensionRunner: undefined, + sessionManager: { appendSessionInit: () => {} }, + getActiveToolNames: () => ["yield"], + setActiveToolsByName: async () => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + listeners.push(listener); + return () => {}; + }, + prompt: async (_text: string, _options?: PromptOptions) => { + await onPrompt(); + for (const listener of listeners) { + listener({ + type: "tool_execution_end", + toolCallId: "tool-yield", + toolName: "yield", + result: { content: [{ type: "text", text: "Result submitted." }], details: { status: "success" } }, + isError: false, + }); + } + }, + waitForIdle: async () => {}, + getLastAssistantMessage: () => undefined, + abort: async () => {}, + dispose: async () => {}, + emit: (event: AgentSessionEvent) => { + for (const listener of listeners) listener(event); + }, + }; + return session as unknown as MockPromptSession; +} + +function requireModel(provider: GeneratedProvider, id: string): Model { + const model = getBundledModel(provider, id); + if (!model) throw new Error(`Expected bundled model ${provider}/${id}`); + return model; +} + +const taskAgent: AgentDefinition = { + name: "task", + description: "General task agent", + systemPrompt: "test", + source: "bundled", +}; + +describe("issue #3464: ollama-cloud task backoff", () => { + let tempDir: TempDir; + let authStorage: AuthStorage; + let modelRegistry: ModelRegistry; + let session: AgentSession | undefined; + + beforeAll(async () => { + tempDir = TempDir.createSync("@omp-issue-3464-"); + authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db")); + authStorage.setRuntimeApiKey("anthropic", "anthropic-test-key"); + authStorage.setRuntimeApiKey("openai", "openai-test-key"); + authStorage.setRuntimeApiKey("ollama-cloud", "ollama-cloud-test-key"); + modelRegistry = new ModelRegistry(authStorage); + }); + + afterAll(() => { + authStorage.close(); + tempDir.removeSync(); + }); + + afterEach(async () => { + if (session) { + await session.dispose(); + session = undefined; + } + modelRegistry.clearSuppressedSelectors(); + vi.restoreAllMocks(); + }); + + it("uses the default fallback chain for a configured task role with no task chain", async () => { + const primary = requireModel("anthropic", "claude-sonnet-4-5"); + const fallback = requireModel("openai", "gpt-4o-mini"); + const requestedModels: string[] = []; + const mock = createMockModel(); + let primaryAttempts = 0; + const agent = new Agent({ + getApiKey: model => `${model.provider}-test-key`, + initialState: { model: primary, systemPrompt: ["Test"], tools: [], messages: [] }, + streamFn: (model, context, options) => { + requestedModels.push(`${model.provider}/${model.id}`); + if (model.provider === primary.provider && model.id === primary.id && primaryAttempts === 0) { + primaryAttempts += 1; + mock.push({ throw: "rate limit exceeded retry-after-ms=200" }); + } else { + mock.push({ content: [`ok:${model.provider}/${model.id}`] }); + } + return mock.stream(model, context, options); + }, + }); + const settings = Settings.isolated({ + "compaction.enabled": false, + "retry.baseDelayMs": 5, + "retry.maxRetries": 1, + "retry.fallbackChains": { default: [`${fallback.provider}/${fallback.id}`] }, + }); + settings.setModelRole("task", `${primary.provider}/${primary.id}`); + + session = new AgentSession({ agent, sessionManager: SessionManager.inMemory(), settings, modelRegistry }); + + await session.prompt("Task role should inherit the default fallback chain"); + await session.waitForIdle(); + + expect(requestedModels).toEqual([`${primary.provider}/${primary.id}`, `${fallback.provider}/${fallback.id}`]); + expect(session.model?.provider).toBe(fallback.provider); + expect(session.model?.id).toBe(fallback.id); + }); + + it("bounds concurrent subagent runs by the resolved ollama-cloud provider limit", async () => { + const cloudModel = requireModel("ollama-cloud", "gpt-oss:120b"); + const started: string[] = []; + const gates = new Map(); + const firstStarted = deferred(); + const secondStarted = deferred(); + vi.spyOn(sdkModule, "createAgentSession").mockImplementation(async options => { + const id = options?.agentId ?? "unknown"; + const gate = deferred(); + gates.set(id, gate); + return createSessionResult( + createGateSession(async () => { + started.push(id); + if (id === "CloudOne") firstStarted.resolve(); + if (id === "CloudTwo") secondStarted.resolve(); + await gate.promise; + }), + ); + }); + const settings = Settings.isolated({ + "providers.ollama-cloud.maxConcurrency": 1, + }); + + const first = runSubprocess({ + cwd: "/tmp", + agent: taskAgent, + task: "first", + index: 0, + id: "CloudOne", + modelOverride: `${cloudModel.provider}/${cloudModel.id}`, + settings, + modelRegistry, + enableLsp: false, + }); + const second = runSubprocess({ + cwd: "/tmp", + agent: taskAgent, + task: "second", + index: 1, + id: "CloudTwo", + modelOverride: `${cloudModel.provider}/${cloudModel.id}`, + settings, + modelRegistry, + enableLsp: false, + }); + + await firstStarted.promise; + expect(started).toEqual(["CloudOne"]); + expect(gates.has("CloudTwo")).toBe(false); + + gates.get("CloudOne")?.resolve(); + await first; + await secondStarted.promise; + expect(started).toEqual(["CloudOne", "CloudTwo"]); + gates.get("CloudTwo")?.resolve(); + await second; + }); +});