(
(response as { end_turn?: boolean } | undefined)?.end_turn,
shouldPromoteIncompleteToolUse,
);
+ // A completed provider-hosted web search that yielded no visible answer
+ // (no text, image, or client tool call) is progress, not a dead end:
+ // pause the turn so the agent loop re-samples with the search results
+ // instead of silently ending. Reasoning/native output items are preserved
+ // for replay. A search followed by visible output stays a normal stop.
+ if (sawCompletedWebSearchCall && output.stopReason === "stop" && !hasVisibleAssistantContent(output)) {
+ output.stopDetails = { type: "pause_turn" };
+ }
options?.onCompleted?.();
// `response.completed`/`response.incomplete`/`response.done` is the last event of a
// Responses stream. Stop pulling instead of waiting for the server to
@@ -3254,6 +3341,14 @@ type ReasoningOptions = {
export interface ApplyResponsesCompatPolicyOptions {
reasoningSummary?: "auto" | "detailed" | "concise" | null;
mapEffort?: (effort: string) => string;
+ /**
+ * Suppress native reasoning by sending `reasoning.effort: "none"` — the only
+ * disable level the Responses API defines (`"off"` is not a wire value and
+ * 400s everywhere). Gateways that reject `none` for a given model are
+ * handled by the reasoning-effort fallback retry, which clamps to the
+ * lowest level the error reports as allowed.
+ */
+ forceReasoningOff?: boolean;
}
export function applyResponsesCompatPolicy(
@@ -3262,6 +3357,10 @@ export function applyResponsesCompatPolicy
= new Set([
"preferWebsockets",
"openrouterVariant",
"loopGuard",
+ "acceptEmptyResponse",
] as const satisfies readonly (keyof SimpleStreamOptions)[]);
// ---------------------------------------------------------------------------
diff --git a/packages/ai/src/providers/register-builtins.ts b/packages/ai/src/providers/register-builtins.ts
index 1b7a59e20..d2dbfd48c 100644
--- a/packages/ai/src/providers/register-builtins.ts
+++ b/packages/ai/src/providers/register-builtins.ts
@@ -202,6 +202,11 @@ interface LazyStreamLimits {
* stream timeouts. Keep the lazy loader from racing it with generic errors.
*/
providerHandlesStreamTimeouts?: boolean;
+ /**
+ * The provider retries or fails over when no first event arrives, while the
+ * lazy wrapper continues to own steady-state idle detection.
+ */
+ providerHandlesFirstEventTimeouts?: boolean;
/**
* Apply OpenAI-family idle timeout precedence in the lazy wrapper. Used by
* local backends whose users historically tune slow prompt-processing gaps
@@ -210,16 +215,13 @@ interface LazyStreamLimits {
openAIIdleEnvFloorsFirstEvent?: boolean;
}
/**
- * Cloud Code Assist (google-gemini-cli / google-antigravity) routinely takes
- * longer than the global 100s default to emit its first SSE event when serving
- * the heavier Gemini 3.x Pro tiers at high thinking levels. Bump the first-event
- * floor to five minutes so callers stop seeing spurious "stream timed out while
- * waiting for the first event" aborts on legitimate cold reasoning starts.
- * The steady-state idle watchdog stays on the global default since the upstream
- * emits thinking tokens frequently once it gets going.
+ * Cloud Code Assist owns first-event detection because Antigravity can return
+ * successful headers and then never emit an SSE event. Keeping the watchdog in
+ * the provider lets it fail over before surfacing an error; the lazy wrapper
+ * still catches post-first-event stalls.
*/
const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
- defaultFirstEventTimeoutMs: 300_000,
+ providerHandlesFirstEventTimeouts: true,
};
const PROVIDER_HANDLED_STREAM_TIMEOUTS: LazyStreamLimits = {
@@ -241,6 +243,7 @@ function forwardStream(
(async () => {
try {
const providerHandlesStreamTimeouts = limits?.providerHandlesStreamTimeouts === true;
+ const providerHandlesFirstEventTimeouts = limits?.providerHandlesFirstEventTimeouts === true;
// Per-model catalog compat can widen the fallback watchdog for hosts
// with no keepalive events (e.g. Bedrock reasoning models that go
// quiet for minutes mid-thinking, issue #4758). Caller options and
@@ -258,12 +261,13 @@ function forwardStream(
(limits?.openAIIdleEnvFloorsFirstEvent
? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs)
: getStreamIdleTimeoutMs(idleTimeoutFallbackMs)));
- const firstItemTimeoutMs = providerHandlesStreamTimeouts
- ? 0
- : (options.streamFirstEventTimeoutMs ??
- (limits?.openAIIdleEnvFloorsFirstEvent
- ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, limits.defaultFirstEventTimeoutMs)
- : getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs)));
+ const firstItemTimeoutMs =
+ providerHandlesStreamTimeouts || providerHandlesFirstEventTimeouts
+ ? 0
+ : (options.streamFirstEventTimeoutMs ??
+ (limits?.openAIIdleEnvFloorsFirstEvent
+ ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, limits.defaultFirstEventTimeoutMs)
+ : getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs)));
// Providers with a server-driven local tool bridge (e.g. the Cursor
// exec channel) mark their stream busy while a local tool runs; the
// watchdog must not read that silence as a provider stall (#4593).
diff --git a/packages/ai/src/providers/vision-guard.ts b/packages/ai/src/providers/vision-guard.ts
index a16a39b31..5a85baf27 100644
--- a/packages/ai/src/providers/vision-guard.ts
+++ b/packages/ai/src/providers/vision-guard.ts
@@ -39,7 +39,11 @@ export function joinTextWithImagePlaceholder(text: string, omittedImages: boolea
* multimodal content arrays for. The compatible-mode endpoint also serves
* multimodal Qwen SKUs without `vl` in the id (e.g. `qwen3.7-plus`), so this
* guard only covers families verified to be text-only for issue #1859:
- * `qwen*-max` and `qwen*-coder*`.
+ * `qwen*-coder*` and `qwen*-max` up to and including `qwen3.7-max`.
+ *
+ * Qwen-Max became multimodal at `qwen3.8-max` (image input, issue #8019), so
+ * `-max` SKUs at version 3.8 or newer are excluded — otherwise the override
+ * would strip images from a genuinely vision-capable flagship (issue #8305).
*
* Used as a defensive override in `convertMessages` so a misconfigured custom
* provider (issue #1859) can't drive the request into an unrecoverable 400.
@@ -48,7 +52,15 @@ export function isDashscopeCompatibleModeTextOnlyQwen(model: Model<"openai-compl
if (!isDashscopeCompatibleModeUrl(model.baseUrl)) {
return false;
}
- const id = model.id.toLowerCase();
if (!isQwenModelId(model.id)) return false;
- return /\bqwen(?:[\d.]+)?-max\b/.test(id) || /\bqwen(?:[\d.]+)?-coder\b/.test(id);
+ const id = model.id.toLowerCase();
+ if (/\bqwen(?:[\d.]+)?-coder\b/.test(id)) return true;
+ const maxMatch = id.match(/\bqwen(?:(\d+)(?:\.(\d+))?)?-max\b/);
+ if (!maxMatch) return false;
+ // Bare `qwen-max` (no version) is the text-only 2.5-era flagship. Compare
+ // major/minor component-wise, not as a decimal float, so `qwen3.10-max`
+ // sorts after `qwen3.8-max`: text-only through 3.7, multimodal from 3.8 on.
+ const major = maxMatch[1] ? Number.parseInt(maxMatch[1], 10) : 0;
+ const minor = maxMatch[2] ? Number.parseInt(maxMatch[2], 10) : 0;
+ return major < 3 || (major === 3 && minor < 8);
}
diff --git a/packages/ai/src/registry/amazon-bedrock.ts b/packages/ai/src/registry/amazon-bedrock.ts
index 37e2c40c8..bb7a0fb58 100644
--- a/packages/ai/src/registry/amazon-bedrock.ts
+++ b/packages/ai/src/registry/amazon-bedrock.ts
@@ -5,7 +5,7 @@ export const amazonBedrockProvider = {
id: "amazon-bedrock",
name: "Amazon Bedrock",
// Amazon Bedrock accepts bearer tokens, IAM keys, profiles, ECS/IRSA credential chains.
- envKeys: resolveAwsRegistryApiKey,
+ envKeys: () => resolveAwsRegistryApiKey({ allowSkipAuth: true }),
mapSimpleOptions: options => {
const awsOptions = options.providerOptions as AwsBedrockProviderOptions | undefined;
return {
diff --git a/packages/ai/src/registry/aws.ts b/packages/ai/src/registry/aws.ts
index 3e23f88f0..e0486e1cd 100644
--- a/packages/ai/src/registry/aws.ts
+++ b/packages/ai/src/registry/aws.ts
@@ -1,5 +1,5 @@
import * as fs from "node:fs";
-import { $env } from "@oh-my-pi/pi-utils";
+import { $env, $flag } from "@oh-my-pi/pi-utils";
import { hasConfiguredAwsProfile } from "../utils/aws-profile";
import { AUTHENTICATED_SENTINEL } from "./types";
@@ -13,14 +13,21 @@ export interface AwsBedrockProviderOptions extends Readonly boolean]> = [
+ ["/sys/hypervisor/uuid", v => v.startsWith("ec2")],
+ ["/sys/devices/virtual/dmi/id/product_uuid", v => v.startsWith("ec2")],
+ ["/sys/devices/virtual/dmi/id/board_asset_tag", v => v.startsWith("ec2") || v.startsWith("i-")],
+ ["/sys/devices/virtual/dmi/id/sys_vendor", v => v.includes("amazon ec2")],
+ ["/sys/devices/virtual/dmi/id/bios_vendor", v => v.includes("amazon ec2")],
+ ];
+ for (const [candidate, matches] of checks) {
try {
const value = fs.readFileSync(candidate, "utf8").trim().toLowerCase();
- if (value.startsWith("ec2")) return true;
+ if (matches(value)) return true;
} catch {
// Missing/unreadable DMI metadata means this probe is inconclusive.
}
@@ -46,7 +53,8 @@ export function hasAwsCredentialSource(): boolean {
}
/** Registry key marker for AWS transports that resolve their own bearer/IAM credentials. */
-export function resolveAwsRegistryApiKey(): string | undefined {
+export function resolveAwsRegistryApiKey(options?: { allowSkipAuth?: boolean }): string | undefined {
+ if (options?.allowSkipAuth && $flag("AWS_BEDROCK_SKIP_AUTH")) return AUTHENTICATED_SENTINEL;
return hasAwsCredentialSource() ? AUTHENTICATED_SENTINEL : undefined;
}
diff --git a/packages/ai/src/registry/oauth/callback-server.ts b/packages/ai/src/registry/oauth/callback-server.ts
index 5b044b769..cbb45e563 100644
--- a/packages/ai/src/registry/oauth/callback-server.ts
+++ b/packages/ai/src/registry/oauth/callback-server.ts
@@ -17,6 +17,15 @@ import type { OAuthController, OAuthCredentials } from "./types";
const DEFAULT_TIMEOUT = 300_000;
const DEFAULT_HOSTNAME = "localhost";
const CALLBACK_PATH = "/callback";
+const IPV4_LOOPBACK = "127.0.0.1";
+const IPV6_LOOPBACK = "::1";
+/**
+ * How many times a random-port bind may be redrawn when the ephemeral port it
+ * landed on is already held on {@link IPV6_LOOPBACK} by that exact address.
+ * Small on purpose: each redraw picks a fresh port, so a repeat collision is
+ * vanishingly unlikely.
+ */
+const IPV6_COMPANION_ATTEMPTS = 4;
/**
* Path served by {@link OAuthCallbackFlow} that 302-redirects to the pending
* authorization URL. Kept out of {@link OAuthCallbackFlowOptions} because it
@@ -28,6 +37,27 @@ const LAUNCH_PATH = "/launch";
export type CallbackResult = { code: string; state: string };
+/**
+ * Subset of {@link Bun.Server} this flow depends on, so a `localhost` flow can
+ * hand back one listener per loopback address family while still looking like a
+ * single server to callers.
+ */
+interface CallbackServer {
+ readonly port: Bun.Server["port"];
+ stop: Bun.Server["stop"];
+}
+
+/**
+ * Whether a failed bind means "another process already holds this port".
+ * Bun surfaces `EADDRINUSE` on the error's `code` where the platform reports
+ * it, and otherwise only in the message, so both are checked.
+ */
+function isAddressInUse(error: unknown): boolean {
+ const code = (error as { code?: unknown } | null | undefined)?.code;
+ if (typeof code === "string") return code === "EADDRINUSE";
+ return error instanceof Error && /EADDRINUSE|in use/i.test(error.message);
+}
+
export interface OAuthCallbackFlowOptions {
preferredPort: number;
callbackPath?: string;
@@ -192,7 +222,7 @@ export abstract class OAuthCallbackFlow {
*/
async #startCallbackServer(
expectedState: string,
- ): Promise<{ server: Bun.Server; redirectUri: string; launchUrl: string | undefined }> {
+ ): Promise<{ server: CallbackServer; redirectUri: string; launchUrl: string | undefined }> {
try {
const server = this.#createServer(this.preferredPort, expectedState);
// `preferredPort: 0` opts into a random port — read the actual bound
@@ -233,7 +263,7 @@ export abstract class OAuthCallbackFlow {
* but every callback flow uses TCP; a missing port here indicates a
* configuration error rather than a fallback case.
*/
- #resolveServerPort(server: Bun.Server): number {
+ #resolveServerPort(server: CallbackServer): number {
const port = server.port;
if (typeof port !== "number") {
throw new AIError.ConfigurationError(
@@ -277,10 +307,68 @@ export abstract class OAuthCallbackFlow {
}
/**
- * Create HTTP server for OAuth callback.
+ * Create the HTTP listener(s) for the OAuth callback.
+ *
+ * `localhost` is not a single endpoint: it resolves to both
+ * {@link IPV4_LOOPBACK} and {@link IPV6_LOOPBACK}, and clients commonly try
+ * `::1` first. Binding only the IPv4 literal hands the authorization code to
+ * whatever holds the IPv6 loopback on the same port — a dev server on
+ * `*:3000` is the common case — which answers from its own routes while this
+ * flow waits out the full {@link DEFAULT_TIMEOUT}. Nothing detects it either:
+ * a specific-address bind coexists with another process's wildcard bind, so
+ * `Bun.serve` reports the port as free and the random-port fallback in
+ * {@link #startCallbackServer} never runs.
+ *
+ * Binding both loopback literals fixes the delivery rather than dodging it:
+ * the kernel routes a connection to the most specific matching bind, so our
+ * `::1` listener receives `localhost` traffic that would otherwise reach a
+ * process bound to the `::` wildcard. Both listeners answer the same routes,
+ * so which family the client resolves stops mattering.
+ *
+ * A genuine collision — another process on exactly this loopback address and
+ * port — still raises EADDRINUSE and reaches the caller's in-use policy. A
+ * host that cannot bind `::1` at all (IPv6 disabled, address unavailable) is
+ * not a collision: the IPv4 listener is the only reachable endpoint there, so
+ * it serves alone.
*/
- #createServer(port: number, expectedState: string): Bun.Server