Merge PR #7072: feat(cursor): expose 1M context windows in model discovery (@mmmeff)

# Conflicts:
#	packages/catalog/src/discovery/cursor.ts
This commit is contained in:
can1357
2026-07-31 19:28:10 +02:00
5 changed files with 215 additions and 2 deletions
@@ -482,6 +482,40 @@ describe("Cursor request action encoding", () => {
expect(payload.requestedModel?.maxMode).toBe(true);
});
it("sends max-mode metadata with prior history when switching providers mid-conversation", async () => {
const payload = await captureCursorPayload(
{
messages: [
{ role: "user", content: "Summarize this repo.", timestamp: 0 },
{
role: "assistant",
api: "anthropic-messages",
provider: "anthropic",
model: "claude-sonnet-4.5",
content: [{ type: "text", text: "It is a monorepo." }],
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 1,
},
{ role: "user", content: "continue", timestamp: 2 },
],
},
cursorMaxModeModel,
);
expect(payload.modelDetails?.maxMode).toBe(true);
expect(payload.requestedModel?.maxMode).toBe(true);
// History from the other provider is carried into the fresh Cursor conversation.
expect(payload.conversationState?.turns.length).toBeGreaterThan(0);
});
it("uses a resume action when a tool result is the final context message", async () => {
const payload = await captureCursorPayload(toolResultContext());
+3
View File
@@ -21,6 +21,9 @@
### Fixed
- Fixed Ollama model-manager caches being reused after the configured base URL changed by scoping cache namespaces to the normalized native discovery endpoint, including reverse-proxy path prefixes ([#7087](https://github.com/can1357/oh-my-pi/issues/7087)).
### Fixed
- Fixed Cursor 1M-context model discovery to expose the 1M-token context window for models Cursor labels with a "1M" display name (Claude and GPT families), natively 1M families Cursor serves unlabeled (Kimi K3, GLM 5.2+), and max-mode Claude/Gemini variants, instead of pinning every discovered model to the 200k default. Bumped the Cursor model-cache namespace so windows cached before this fix are refetched. ([#4798](https://github.com/can1357/oh-my-pi/issues/4798))
## [17.2.0] - 2026-07-30
+57 -1
View File
@@ -2,6 +2,7 @@ import * as http2 from "node:http2";
import { create, fromBinary, toBinary } from "@bufbuild/protobuf";
import { type } from "arktype";
import { isKimiK3ModelId } from "../identity";
import { bareModelId, parseGlmModel, semverGte } from "../identity/classify";
import { getBundledModels } from "../models";
import { toModelSpec } from "../provider-models/bundled-references";
import type { Model, ModelSpec } from "../types";
@@ -14,6 +15,19 @@ const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels";
const DEFAULT_CONTEXT_WINDOW = 200_000;
const DEFAULT_MAX_TOKENS = 64_000;
/**
* `GetUsableModels` carries no context-window field, so the 1M ceiling is
* recovered from the signals Cursor does send:
* - display-name labels ("Opus 5 1M", "GPT-5.5 1M High") across families,
* - natively 1M families Cursor serves unlabeled (Kimi K3, GLM 5.2+),
* - the max-mode flag on Claude/Gemini ids, whose max-mode ceiling is 1M.
*/
const CURSOR_1M_CONTEXT_WINDOW = 1_000_000;
const CURSOR_1M_NAME_PATTERN = /\b1m\b/i;
const CURSOR_MAX_MODE_1M_ID_PATTERN = /claude|gemini/;
/** Kimi's official bare K3 id (`k3`, `kimi/k3`); `k3-256k` is the 256k SKU and stays out. */
const CURSOR_KIMI_K3_BARE_ID_PATTERN = /(^|\/)k3$/i;
/**
* Model-id families whose native catalogs (anthropic, openai/openai-codex,
* google) are multimodal. Cursor-only or text-only families (`composer-*`,
@@ -292,6 +306,7 @@ function normalizeCursorModel(
name,
baseUrl: baseUrlOverride ?? reference.baseUrl,
reasoning,
contextWindow: resolveCursorContextWindow(details, id, reference.contextWindow),
cursorMaxMode: details.maxMode,
};
}
@@ -304,12 +319,53 @@ function normalizeCursorModel(
reasoning,
input: inferInputFromCursorId(id),
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: DEFAULT_CONTEXT_WINDOW,
contextWindow: resolveCursorContextWindow(details, id, DEFAULT_CONTEXT_WINDOW),
maxTokens: DEFAULT_MAX_TOKENS,
cursorMaxMode: details.maxMode,
};
}
/**
* Context window for a discovered Cursor model: the 1M ceiling when any 1M
* signal fires (never below a larger bundled reference), else the fallback.
*/
function resolveCursorContextWindow(
model: CursorModelDetailsValue,
id: string,
fallback: number | null,
): number | null {
const labeled1M =
CURSOR_1M_NAME_PATTERN.test(id) ||
[model.displayName, model.displayNameShort, model.displayModelId, ...model.aliases].some(
candidate => typeof candidate === "string" && CURSOR_1M_NAME_PATTERN.test(candidate),
);
if (labeled1M || isCursorNative1MModelId(id) || (model.maxMode && CURSOR_MAX_MODE_1M_ID_PATTERN.test(id))) {
return Math.max(fallback ?? 0, CURSOR_1M_CONTEXT_WINDOW);
}
return fallback;
}
/**
* Natively 1M-context families Cursor serves without a "1M" label: Kimi K3 and
* GLM 5.2+ coding SKUs. The shared family parsers cover namespace forms
* (`moonshotai/kimi-k3`, `z-ai/glm-5.2`) and future GLM versions (`glm-5.10`,
* `glm-6`); vision and sub-1M variants stay out via the same gates as
* `isGlm52ReasoningEffortModelId`.
*/
function isCursorNative1MModelId(id: string): boolean {
if (isKimiK3ModelId(id) || CURSOR_KIMI_K3_BARE_ID_PATTERN.test(id)) {
return true;
}
const glm = parseGlmModel(bareModelId(id));
if (!glm || glm.vision) {
return false;
}
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
return false;
}
return semverGte(glm.version, "5.2");
}
function pickModelDisplayName(model: CursorModelDetailsValue, fallbackId: string): string {
const candidates = [model.displayName, model.displayNameShort, model.displayModelId, ...model.aliases, fallbackId];
for (const candidate of candidates) {
@@ -41,7 +41,9 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
case "ollama":
return resolveOllamaModelCacheProviderId(providerId, options.baseUrl);
case "cursor":
return "cursor:max-mode-v2";
// v3: max-mode Claude/Gemini rows cached before the 1M context-window
// discovery fix carry a stale 200k window and must be refetched.
return "cursor:max-mode-v3";
case "litellm": {
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
return `litellm:rich-v5:${Bun.hash(baseUrl).toString(36)}`;
@@ -203,6 +203,124 @@ describe("fetchCursorUsableModels", () => {
]);
});
it("assigns the 1M window from display-name labels across families", async () => {
const response = create(GetUsableModelsResponseSchema, {
models: [
create(ModelDetailsSchema, { modelId: "claude-opus-5-high", displayName: "Opus 5 1M" }),
create(ModelDetailsSchema, { modelId: "gpt-5.5-high", displayName: "GPT-5.5 1M High" }),
create(ModelDetailsSchema, { modelId: "gpt-5.6-sol-medium", displayName: "GPT-5.6 Sol 1M" }),
],
});
const labeledBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: labeledBaseUrl, timeoutMs: 1_000 });
expect(models).toEqual([
expect.objectContaining({ id: "claude-opus-5-high", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "gpt-5.5-high", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "gpt-5.6-sol-medium", contextWindow: 1_000_000 }),
]);
});
it("assigns the 1M window to natively 1M families Cursor serves unlabeled", async () => {
const response = create(GetUsableModelsResponseSchema, {
models: [
create(ModelDetailsSchema, { modelId: "kimi-k3-max", displayName: "Kimi K3" }),
create(ModelDetailsSchema, { modelId: "moonshotai/kimi-k3", displayName: "Kimi K3" }),
create(ModelDetailsSchema, { modelId: "k3", displayName: "K3" }),
create(ModelDetailsSchema, { modelId: "kimi/k3", displayName: "K3" }),
create(ModelDetailsSchema, { modelId: "glm-5.2-max", displayName: "GLM 5.2 Max" }),
create(ModelDetailsSchema, { modelId: "glm-5.10-high", displayName: "GLM 5.10 High" }),
create(ModelDetailsSchema, { modelId: "glm-6-max", displayName: "GLM 6 Max" }),
],
});
const nativeBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: nativeBaseUrl, timeoutMs: 1_000 });
expect(models).toEqual([
expect.objectContaining({ id: "glm-5.10-high", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "glm-5.2-max", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "glm-6-max", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "k3", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "kimi-k3-max", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "kimi/k3", contextWindow: 1_000_000 }),
expect.objectContaining({ id: "moonshotai/kimi-k3", contextWindow: 1_000_000 }),
]);
});
it("keeps the default window below the GLM 5.2 floor and outside the coding variants", async () => {
const response = create(GetUsableModelsResponseSchema, {
models: [
create(ModelDetailsSchema, { modelId: "glm-5.1-high", displayName: "GLM 5.1 High" }),
create(ModelDetailsSchema, { modelId: "glm-5.2-flash", displayName: "GLM 5.2 Flash" }),
create(ModelDetailsSchema, { modelId: "k3-256k", displayName: "K3-256k" }),
],
});
const nativeBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: nativeBaseUrl, timeoutMs: 1_000 });
expect(models).toEqual([
expect.objectContaining({ id: "glm-5.1-high", contextWindow: 200_000 }),
expect.objectContaining({ id: "glm-5.2-flash", contextWindow: 200_000 }),
expect.objectContaining({ id: "k3-256k", contextWindow: 200_000 }),
]);
});
it("assigns the 1M window to unlabeled max-mode Claude models", async () => {
const response = create(GetUsableModelsResponseSchema, {
models: [
create(ModelDetailsSchema, {
modelId: "claude-opus-4-8-high-fast",
displayName: "Opus 4.8 Fast",
maxMode: true,
}),
],
});
const maxModeBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: maxModeBaseUrl, timeoutMs: 1_000 });
expect(models).toEqual([
expect.objectContaining({ id: "claude-opus-4-8-high-fast", cursorMaxMode: true, contextWindow: 1_000_000 }),
]);
});
it("keeps the default window for unlabeled non-max models and max-mode models outside 1M families", async () => {
const response = create(GetUsableModelsResponseSchema, {
models: [
create(ModelDetailsSchema, { modelId: "cursor-composer-max", maxMode: true }),
create(ModelDetailsSchema, { modelId: "claude-opus-4-8-high", displayName: "Opus 4.8" }),
],
});
const defaultBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
const models = await fetchCursorUsableModels({ apiKey: "test-token", baseUrl: defaultBaseUrl, timeoutMs: 1_000 });
expect(models).toEqual([
expect.objectContaining({ id: "claude-opus-4-8-high", cursorMaxMode: false, contextWindow: 200_000 }),
expect.objectContaining({ id: "cursor-composer-max", cursorMaxMode: true, contextWindow: 200_000 }),
]);
});
it("raises a bundled reference window when the reference id is served with a 1M label", async () => {
// `claude-4.5-sonnet` is a bundled cursor reference with a 200k window;
// served with a 1M display name it must expose the 1M ceiling.
const response = create(GetUsableModelsResponseSchema, {
models: [create(ModelDetailsSchema, { modelId: "claude-4.5-sonnet", displayName: "Sonnet 4.5 1M" })],
});
const referenceBaseUrl = await startCursorDiscoveryServer(toBinary(GetUsableModelsResponseSchema, response));
const models = await fetchCursorUsableModels({
apiKey: "test-token",
baseUrl: referenceBaseUrl,
timeoutMs: 1_000,
});
expect(models).toEqual([expect.objectContaining({ id: "claude-4.5-sonnet", contextWindow: 1_000_000 })]);
});
it("ignores Cursor cache rows written before max-mode metadata was persisted", async () => {
const cacheDbPath = await createTempCachePath();
const staleSpec = cursorModelSpec("cursor-composer-max");