d4317d3d20
This change introduces a new `openrouter` API type and extensively refactors OpenAI-family streaming providers, centralizing shared logic and improving robustness.
Key changes include:
- **Unified OpenAI-family Logic:** Consolidated core utilities, compat resolution, request shaping, and stream processing into `openai-shared.ts`, reducing duplication across `openai-completions`, `openai-responses`, and `openai-codex-responses`.
- **OpenRouter API Type:** Introduced a dedicated `openrouter` API type with dual-surface compatibility, allowing it to dispatch requests as either OpenAI Chat Completions or Responses.
- **Enhanced Provider Integration:**
- Improved Perplexity search to leverage shared OpenAI streaming transports, including API-key fallback to OpenRouter and support for Perplexity's Responses API.
- Integrated xAI-specific logic directly into the shared `stream.ts` dispatch, removing the dedicated `xai-responses` provider.
- Refined credential parsing for Google Gemini CLI and handling of Azure deployment names.
- **Robustness & Consistency:** Improved error handling for Codex, standardized output token parameter resolution, and ensured consistent application of reasoning suppression across all Chat Completions dialects.
- **New Documentation:** Added `provider-endpoint-constraints.md` to detail endpoint-specific behaviors and quirks for various providers.
- **Telemetry & Debugging:** Extended telemetry propagation to advisor calls and overflow compaction tasks. Improved debugging for Codex WebSocket failures and stream error messages.
- **Tooling & Security:** Updated browser stealth scripts to prevent detection and added a new `ts-no-inline-cast-access` TTSR rule.
71 lines
2.1 KiB
TypeScript
71 lines
2.1 KiB
TypeScript
import { describe, expect, it } from "bun:test";
|
|
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
|
import type { Context, Model, OpenAICompat } from "@oh-my-pi/pi-ai/types";
|
|
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
|
|
|
const testContext: Context = {
|
|
messages: [{ role: "user", content: "hello", timestamp: 0 }],
|
|
};
|
|
|
|
function createAbortedSignal(): AbortSignal {
|
|
const controller = new AbortController();
|
|
controller.abort();
|
|
return controller.signal;
|
|
}
|
|
|
|
function captureResponsesPayload(
|
|
model: Model<"openai-responses">,
|
|
reasoning: "minimal" | "low" | "medium" | "high" | "xhigh",
|
|
): Promise<Record<string, unknown>> {
|
|
const { promise, resolve } = Promise.withResolvers<Record<string, unknown>>();
|
|
streamOpenAIResponses(model, testContext, {
|
|
apiKey: "test-key",
|
|
signal: createAbortedSignal(),
|
|
reasoning,
|
|
reasoningSummary: "auto",
|
|
onPayload: payload => resolve(payload as Record<string, unknown>),
|
|
});
|
|
return promise;
|
|
}
|
|
|
|
function customResponsesModel(compat: OpenAICompat): Model<"openai-responses"> {
|
|
return {
|
|
id: "deepseek-v4-flash:cloud",
|
|
name: "deepseek-v4-flash:cloud",
|
|
api: "openai-responses",
|
|
provider: "custom",
|
|
baseUrl: "http://127.0.0.1:11434/v1",
|
|
reasoning: true,
|
|
input: ["text"],
|
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
contextWindow: 1_048_576,
|
|
maxTokens: 65_536,
|
|
thinking: {
|
|
mode: "effort",
|
|
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh],
|
|
effortMap: compat.reasoningEffortMap,
|
|
},
|
|
compat,
|
|
} as unknown as Model<"openai-responses">;
|
|
}
|
|
|
|
describe("issue #931 — openai-responses reasoning effort mapping", () => {
|
|
it("maps configured xhigh thinking effort before sending Responses reasoning payload", async () => {
|
|
const payload = await captureResponsesPayload(
|
|
customResponsesModel({
|
|
supportsReasoningEffort: true,
|
|
reasoningEffortMap: {
|
|
minimal: "low",
|
|
low: "low",
|
|
medium: "medium",
|
|
high: "high",
|
|
xhigh: "max",
|
|
},
|
|
}),
|
|
"xhigh",
|
|
);
|
|
|
|
expect(payload.reasoning).toEqual({ effort: "max", summary: "auto" });
|
|
});
|
|
});
|