Merge PR #7072: feat(cursor): expose 1M context windows in model discovery (@mmmeff)
# Conflicts: # packages/catalog/src/discovery/cursor.ts
This commit is contained in:
@@ -482,6 +482,40 @@ describe("Cursor request action encoding", () => {
|
||||
expect(payload.requestedModel?.maxMode).toBe(true);
|
||||
});
|
||||
|
||||
it("sends max-mode metadata with prior history when switching providers mid-conversation", async () => {
|
||||
const payload = await captureCursorPayload(
|
||||
{
|
||||
messages: [
|
||||
{ role: "user", content: "Summarize this repo.", timestamp: 0 },
|
||||
{
|
||||
role: "assistant",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: "claude-sonnet-4.5",
|
||||
content: [{ type: "text", text: "It is a monorepo." }],
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: 1,
|
||||
},
|
||||
{ role: "user", content: "continue", timestamp: 2 },
|
||||
],
|
||||
},
|
||||
cursorMaxModeModel,
|
||||
);
|
||||
|
||||
expect(payload.modelDetails?.maxMode).toBe(true);
|
||||
expect(payload.requestedModel?.maxMode).toBe(true);
|
||||
// History from the other provider is carried into the fresh Cursor conversation.
|
||||
expect(payload.conversationState?.turns.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("uses a resume action when a tool result is the final context message", async () => {
|
||||
const payload = await captureCursorPayload(toolResultContext());
|
||||
|
||||
|
||||
@@ -21,6 +21,9 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Ollama model-manager caches being reused after the configured base URL changed by scoping cache namespaces to the normalized native discovery endpoint, including reverse-proxy path prefixes ([#7087](https://github.com/can1357/oh-my-pi/issues/7087)).
|
||||
### Fixed
|
||||
|
||||
- Fixed Cursor 1M-context model discovery to expose the 1M-token context window for models Cursor labels with a "1M" display name (Claude and GPT families), natively 1M families Cursor serves unlabeled (Kimi K3, GLM 5.2+), and max-mode Claude/Gemini variants, instead of pinning every discovered model to the 200k default. Bumped the Cursor model-cache namespace so windows cached before this fix are refetched. ([#4798](https://github.com/can1357/oh-my-pi/issues/4798))
|
||||
|
||||
## [17.2.0] - 2026-07-30
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ import * as http2 from "node:http2";
|
||||
import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
|
||||
import { type } from "arktype";
|
||||
import { isKimiK3ModelId } from "../identity";
|
||||
import { bareModelId, parseGlmModel, semverGte } from "../identity/classify";
|
||||
import { getBundledModels } from "../models";
|
||||
import { toModelSpec } from "../provider-models/bundled-references";
|
||||
import type { Model, ModelSpec } from "../types";
|
||||
@@ -14,6 +15,19 @@ const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
|
||||
const DEFAULT_CONTEXT_WINDOW = 200_000;
|
||||
const DEFAULT_MAX_TOKENS = 64_000;
|
||||
|
||||
/**
|
||||
* `GetUsableModels` carries no context-window field, so the 1M ceiling is
|
||||
* recovered from the signals Cursor does send:
|
||||
* - display-name labels ("Opus 5 1M", "GPT-5.5 1M High") across families,
|
||||
* - natively 1M families Cursor serves unlabeled (Kimi K3, GLM 5.2+),
|
||||
* - the max-mode flag on Claude/Gemini ids, whose max-mode ceiling is 1M.
|
||||
*/
|
||||
const CURSOR_1M_CONTEXT_WINDOW = 1_000_000;
|
||||
const CURSOR_1M_NAME_PATTERN = /\b1m\b/i;
|
||||
const CURSOR_MAX_MODE_1M_ID_PATTERN = /claude|gemini/;
|
||||
/** Kimi's official bare K3 id (`k3`, `kimi/k3`); `k3-256k` is the 256k SKU and stays out. */
|
||||
const CURSOR_KIMI_K3_BARE_ID_PATTERN = /(^|\/)k3$/i;
|
||||
|
||||
/**
|
||||
* Model-id families whose native catalogs (anthropic, openai/openai-codex,
|
||||
* google) are multimodal. Cursor-only or text-only families (`composer-*`,
|
||||
@@ -292,6 +306,7 @@ function normalizeCursorModel(
|
||||
name,
|
||||
baseUrl: baseUrlOverride ?? reference.baseUrl,
|
||||
reasoning,
|
||||
contextWindow: resolveCursorContextWindow(details, id, reference.contextWindow),
|
||||
cursorMaxMode: details.maxMode,
|
||||
};
|
||||
}
|
||||
@@ -304,12 +319,53 @@ function normalizeCursorModel(
|
||||
reasoning,
|
||||
input: inferInputFromCursorId(id),
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: DEFAULT_CONTEXT_WINDOW,
|
||||
contextWindow: resolveCursorContextWindow(details, id, DEFAULT_CONTEXT_WINDOW),
|
||||
maxTokens: DEFAULT_MAX_TOKENS,
|
||||
cursorMaxMode: details.maxMode,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Context window for a discovered Cursor model: the 1M ceiling when any 1M
|
||||
* signal fires (never below a larger bundled reference), else the fallback.
|
||||
*/
|
||||
function resolveCursorContextWindow(
|
||||
model: CursorModelDetailsValue,
|
||||
id: string,
|
||||
fallback: number | null,
|
||||
): number | null {
|
||||
const labeled1M =
|
||||
CURSOR_1M_NAME_PATTERN.test(id) ||
|
||||
[model.displayName, model.displayNameShort, model.displayModelId, ...model.aliases].some(
|
||||
candidate => typeof candidate === "string" && CURSOR_1M_NAME_PATTERN.test(candidate),
|
||||
);
|
||||
if (labeled1M || isCursorNative1MModelId(id) || (model.maxMode && CURSOR_MAX_MODE_1M_ID_PATTERN.test(id))) {
|
||||
return Math.max(fallback ?? 0, CURSOR_1M_CONTEXT_WINDOW);
|
||||
}
|
||||
return fallback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Natively 1M-context families Cursor serves without a "1M" label: Kimi K3 and
|
||||
* GLM 5.2+ coding SKUs. The shared family parsers cover namespace forms
|
||||
* (`moonshotai/kimi-k3`, `z-ai/glm-5.2`) and future GLM versions (`glm-5.10`,
|
||||
* `glm-6`); vision and sub-1M variants stay out via the same gates as
|
||||
* `isGlm52ReasoningEffortModelId`.
|
||||
*/
|
||||
function isCursorNative1MModelId(id: string): boolean {
|
||||
if (isKimiK3ModelId(id) || CURSOR_KIMI_K3_BARE_ID_PATTERN.test(id)) {
|
||||
return true;
|
||||
}
|
||||
const glm = parseGlmModel(bareModelId(id));
|
||||
if (!glm || glm.vision) {
|
||||
return false;
|
||||
}
|
||||
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
|
||||
return false;
|
||||
}
|
||||
return semverGte(glm.version, "5.2");
|
||||
}
|
||||
|
||||
function pickModelDisplayName(model: CursorModelDetailsValue, fallbackId: string): string {
|
||||
const candidates = [model.displayName, model.displayNameShort, model.displayModelId, ...model.aliases, fallbackId];
|
||||
for (const candidate of candidates) {
|
||||
|
||||
@@ -41,7 +41,9 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
|
||||
case "ollama":
|
||||
return resolveOllamaModelCacheProviderId(providerId, options.baseUrl);
|
||||
case "cursor":
|
||||
return "cursor:max-mode-v2";
|
||||
// v3: max-mode Claude/Gemini rows cached before the 1M context-window
|
||||
// discovery fix carry a stale 200k window and must be refetched.
|
||||
return "cursor:max-mode-v3";
|
||||
case "litellm": {
|
||||
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
|
||||
return `litellm:rich-v5:${Bun.hash(baseUrl).toString(36)}`;
|
||||
|
||||
@@ -203,6 +203,124 @@ describe("fetchCursorUsableModels", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("assigns the 1M window from display-name labels across families", async () => {
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: [
|
||||
create(ModelDetailsSchema, { modelId: "claude-opus-5-high", displayName: "Opus 5 1M" }),
|
||||
create(ModelDetailsSchema, { modelId: "gpt-5.5-high", displayName: "GPT-5.5 1M High" }),
|
||||
create(ModelDetailsSchema, { modelId: "gpt-5.6-sol-medium", displayName: "GPT-5.6 Sol 1M" }),
|
||||
],
|
||||
});
|
||||
const labeledBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: labeledBaseUrl, timeoutMs: 1_000 });
|
||||
|
||||
expect(models).toEqual([
|
||||
expect.objectContaining({ id: "claude-opus-5-high", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "gpt-5.5-high", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "gpt-5.6-sol-medium", contextWindow: 1_000_000 }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("assigns the 1M window to natively 1M families Cursor serves unlabeled", async () => {
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: [
|
||||
create(ModelDetailsSchema, { modelId: "kimi-k3-max", displayName: "Kimi K3" }),
|
||||
create(ModelDetailsSchema, { modelId: "moonshotai/kimi-k3", displayName: "Kimi K3" }),
|
||||
create(ModelDetailsSchema, { modelId: "k3", displayName: "K3" }),
|
||||
create(ModelDetailsSchema, { modelId: "kimi/k3", displayName: "K3" }),
|
||||
create(ModelDetailsSchema, { modelId: "glm-5.2-max", displayName: "GLM 5.2 Max" }),
|
||||
create(ModelDetailsSchema, { modelId: "glm-5.10-high", displayName: "GLM 5.10 High" }),
|
||||
create(ModelDetailsSchema, { modelId: "glm-6-max", displayName: "GLM 6 Max" }),
|
||||
],
|
||||
});
|
||||
const nativeBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: nativeBaseUrl, timeoutMs: 1_000 });
|
||||
|
||||
expect(models).toEqual([
|
||||
expect.objectContaining({ id: "glm-5.10-high", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "glm-5.2-max", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "glm-6-max", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "k3", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "kimi-k3-max", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "kimi/k3", contextWindow: 1_000_000 }),
|
||||
expect.objectContaining({ id: "moonshotai/kimi-k3", contextWindow: 1_000_000 }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("keeps the default window below the GLM 5.2 floor and outside the coding variants", async () => {
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: [
|
||||
create(ModelDetailsSchema, { modelId: "glm-5.1-high", displayName: "GLM 5.1 High" }),
|
||||
create(ModelDetailsSchema, { modelId: "glm-5.2-flash", displayName: "GLM 5.2 Flash" }),
|
||||
create(ModelDetailsSchema, { modelId: "k3-256k", displayName: "K3-256k" }),
|
||||
],
|
||||
});
|
||||
const nativeBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: nativeBaseUrl, timeoutMs: 1_000 });
|
||||
|
||||
expect(models).toEqual([
|
||||
expect.objectContaining({ id: "glm-5.1-high", contextWindow: 200_000 }),
|
||||
expect.objectContaining({ id: "glm-5.2-flash", contextWindow: 200_000 }),
|
||||
expect.objectContaining({ id: "k3-256k", contextWindow: 200_000 }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("assigns the 1M window to unlabeled max-mode Claude models", async () => {
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: [
|
||||
create(ModelDetailsSchema, {
|
||||
modelId: "claude-opus-4-8-high-fast",
|
||||
displayName: "Opus 4.8 Fast",
|
||||
maxMode: true,
|
||||
}),
|
||||
],
|
||||
});
|
||||
const maxModeBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: maxModeBaseUrl, timeoutMs: 1_000 });
|
||||
|
||||
expect(models).toEqual([
|
||||
expect.objectContaining({ id: "claude-opus-4-8-high-fast", cursorMaxMode: true, contextWindow: 1_000_000 }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("keeps the default window for unlabeled non-max models and max-mode models outside 1M families", async () => {
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: [
|
||||
create(ModelDetailsSchema, { modelId: "cursor-composer-max", maxMode: true }),
|
||||
create(ModelDetailsSchema, { modelId: "claude-opus-4-8-high", displayName: "Opus 4.8" }),
|
||||
],
|
||||
});
|
||||
const defaultBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: defaultBaseUrl, timeoutMs: 1_000 });
|
||||
|
||||
expect(models).toEqual([
|
||||
expect.objectContaining({ id: "claude-opus-4-8-high", cursorMaxMode: false, contextWindow: 200_000 }),
|
||||
expect.objectContaining({ id: "cursor-composer-max", cursorMaxMode: true, contextWindow: 200_000 }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("raises a bundled reference window when the reference id is served with a 1M label", async () => {
|
||||
// `claude-4.5-sonnet` is a bundled cursor reference with a 200k window;
|
||||
// served with a 1M display name it must expose the 1M ceiling.
|
||||
const response = create(GetUsableModelsResponseSchema, {
|
||||
models: [create(ModelDetailsSchema, { modelId: "claude-4.5-sonnet", displayName: "Sonnet 4.5 1M" })],
|
||||
});
|
||||
const referenceBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
|
||||
|
||||
const models = await fetchCursorUsableModels({
|
||||
apiKey: "test-token",
|
||||
baseUrl: referenceBaseUrl,
|
||||
timeoutMs: 1_000,
|
||||
});
|
||||
|
||||
expect(models).toEqual([expect.objectContaining({ id: "claude-4.5-sonnet", contextWindow: 1_000_000 })]);
|
||||
});
|
||||
|
||||
it("ignores Cursor cache rows written before max-mode metadata was persisted", async () => {
|
||||
const cacheDbPath = await createTempCachePath();
|
||||
const staleSpec = cursorModelSpec("cursor-composer-max");
|
||||
|
||||
Reference in New Issue
Block a user