Merge PR #4882: fix(catalog): collapse Devin GLM-5.2 variants so free 200K model works when quota is exhausted (@oldschoola)

# Conflicts:
#	packages/catalog/src/models.json
This commit is contained in:
can1357
2026-07-20 22:50:16 +02:00
6 changed files with 173 additions and 8 deletions
+7
View File
@@ -174,6 +174,13 @@
- Updated cost and token configurations for various models across providers
- Renamed several models for consistency (e.g., MiniMax M3, Gemma 4 31B, Qwen variants)
### Added
- Added static fallback seed for Devin's `swe-1-7` model so it is bundled even when catalog generation runs without a Devin session token.
### Fixed
- Collapsed Devin's six GLM-5.2 variants into two logical entries (`glm-5-2` for 200K free, `glm-5-2-1m` for 1M paid). The 200K entry routes every thinking effort to the free `glm-5-2` wire UID — never to the quota-gated `glm-5-2-max` or `glm-5-2-none` — so GLM-5.2 works even when the weekly usage quota is exhausted.
## [16.3.12] - 2026-07-08
@@ -17,6 +17,7 @@ import { getGitLabDuoModels } from "@oh-my-pi/pi-ai/providers/gitlab-duo";
import { $env } from "@oh-my-pi/pi-utils";
import { ANTIGRAVITY_PRIMARY_ENDPOINT, fetchAntigravityDiscoveryModels } from "../src/discovery/antigravity";
import { fetchCodexModels } from "../src/discovery/codex";
import { DEVIN_STATIC_FALLBACK_MODELS } from "../src/discovery/devin";
import { buildGitLabDuoWorkflowFallbackModel } from "../src/discovery/gitlab-duo-workflow";
import { createModelManager } from "../src/model-manager";
import prevModelsJson from "../src/models.json" with { type: "json" };
@@ -541,6 +542,13 @@ async function generateModels() {
if (!authoritativeCatalogProviders.has("gitlab-duo-agent")) {
allModels.push(buildGitLabDuoWorkflowFallbackModel());
}
// Seed Devin fallback models so newly released free models (e.g. `swe-1-7`)
// are bundled even when catalog generation runs without a Devin session
// token. Devin is `dynamicModelsAuthoritative: true`, so live discovery
// replaces these at runtime when a key is present.
if (!authoritativeCatalogProviders.has("devin")) {
allModels.push(...DEVIN_STATIC_FALLBACK_MODELS);
}
// Seed Fireworks "Fast" serving-path variants (`<id>-fast`). Fast routers are
// not enumerated by the serverless control-plane list, so discovery never
// surfaces them; the seed projects each base entry into a fast variant.
+24
View File
@@ -149,3 +149,27 @@ function normalizeDevinModels(
}
return [...byId.values()].sort((a, b) => a.id.localeCompare(b.id));
}
/**
* Static fallback Devin models for catalog generation without a live API key.
*
* Devin discovery requires an authenticated session token; when catalog
* generation runs without one, these seeds ensure new free models (like
* `swe-1-7`) are still bundled. Live discovery is authoritative — when it
* succeeds, it replaces these seeds entirely (stale entries are pruned).
*/
export const DEVIN_STATIC_FALLBACK_MODELS: readonly ModelSpec<"devin-agent">[] = [
{
id: "swe-1-7",
name: "SWE-1.7",
api: "devin-agent",
provider: "devin",
baseUrl: DEVIN_DEFAULT_BASE_URL,
reasoning: true,
input: ["text"],
supportsTools: true,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 262_000,
maxTokens: DEFAULT_MAX_TOKENS,
},
];
+33
View File
@@ -532,6 +532,39 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
},
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
),
// GLM-5.2 200K — only the base wire UID `glm-5-2` is free on Devin's
// Coding Plan (verified via streamDevin: `glm-5-2-none` and `glm-5-2-max`
// both return "weekly usage quota exhausted" while `glm-5-2` streams
// successfully). Route every effort to `glm-5-2` so the collapsed entry
// is always free; include the paid 200K variants as members so they are
// hidden from the model list. The 1M-context variants stay as separate
// paid entries (collapsed below).
{
id: "glm-5-2",
name: "GLM-5.2",
members: ["glm-5-2", "glm-5-2-none", "glm-5-2-max"],
routing: {
[Effort.High]: "glm-5-2",
[Effort.XHigh]: "glm-5-2",
},
thinking: {
mode: "effort",
efforts: [Effort.High, Effort.XHigh],
requiresEffort: true,
},
},
// GLM-5.2 1M — paid variants that consume weekly quota. Collapse the
// three 1M-context variants into one entry with proper effort routing.
devinTierFamily(
"glm-5-2-1m",
"GLM-5.2 1M",
{
off: "glm-5-2-none-1m",
high: "glm-5-2-1m",
xhigh: "glm-5-2-max-1m",
},
[Effort.High, Effort.XHigh],
),
],
};
@@ -791,3 +791,75 @@ describe("antigravity discovery collapsing", () => {
expect(models?.[0]?.baseUrl).toBe(ANTIGRAVITY_PRIMARY_ENDPOINT);
});
});
describe("Devin GLM-5.2 collapse", () => {
function devinMemberSpec(id: string, overrides: Partial<ModelSpec<"devin-agent">> = {}): ModelSpec<"devin-agent"> {
return {
id,
name: id,
api: "devin-agent",
provider: "devin",
baseUrl: "https://server.codeium.com",
reasoning: true,
input: ["text"],
supportsTools: true,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 64_000,
...overrides,
};
}
it("collapses the three 200K GLM-5.2 variants into one logical entry routing all efforts to the free glm-5-2 wire UID", () => {
const out = collapseEffortVariants(
[
devinMemberSpec("glm-5-2"),
devinMemberSpec("glm-5-2-max"),
devinMemberSpec("glm-5-2-none", { reasoning: false }),
],
DEVIN_VARIANT_COLLAPSE_TABLE,
);
expect(out).toHaveLength(1);
const spec = out[0];
expect(spec?.id).toBe("glm-5-2");
expect(spec?.thinking?.effortRouting).toEqual({
high: "glm-5-2",
xhigh: "glm-5-2",
});
});
it("routes every effort to glm-5-2 (never to the quota-gated glm-5-2-max or glm-5-2-none)", () => {
const out = collapseEffortVariants(
[devinMemberSpec("glm-5-2"), devinMemberSpec("glm-5-2-max")],
DEVIN_VARIANT_COLLAPSE_TABLE,
);
const spec = out[0];
const routing = spec?.thinking?.effortRouting ?? {};
for (const wire of Object.values(routing)) {
expect(wire).toBe("glm-5-2");
}
});
it("collapses the three 1M GLM-5.2 variants into one paid entry with proper effort routing", () => {
const out = collapseEffortVariants(
[
devinMemberSpec("glm-5-2-1m", { contextWindow: 1_000_000 }),
devinMemberSpec("glm-5-2-max-1m", { contextWindow: 1_000_000 }),
devinMemberSpec("glm-5-2-none-1m", { contextWindow: 1_000_000, reasoning: false }),
],
DEVIN_VARIANT_COLLAPSE_TABLE,
);
expect(out).toHaveLength(1);
const spec = out[0];
expect(spec?.id).toBe("glm-5-2-1m");
expect(spec?.contextWindow).toBe(1_000_000);
expect(spec?.thinking?.effortRouting).toEqual({
off: "glm-5-2-none-1m",
high: "glm-5-2-1m",
xhigh: "glm-5-2-max-1m",
});
});
});
@@ -34,16 +34,37 @@ describe("AgentSession advisor descriptor thinking level", () => {
sharedDir = TempDir.createSync("@pi-advisor-devin-thinking-shared-");
authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
// Seeding a runtime API key exposes the bundled Devin catalog for
// `resolveAdvisorRoleSelection` / `getAvailable()` without any live
// network discovery.
authStorage.setRuntimeApiKey("devin", "test-key");
modelRegistry = new ModelRegistry(authStorage);
const anthropic = getBundledModel("anthropic", "claude-sonnet-4-5");
const devin = getBundledModel("devin", "glm-5-2");
if (!anthropic) throw new Error("Expected bundled anthropic/claude-sonnet-4-5 to exist");
if (!devin) throw new Error("Expected bundled devin/glm-5-2 to exist");
anthropicModel = anthropic;
// Register a synthetic `devin-agent` provider with a reasoning model
// that has NO `thinking` metadata. This is the exact catalog shape that
// triggered #4579: `reasoning: true` with no controllable effort surface.
// Using a synthetic model avoids brittleness from upstream catalog drift
// (e.g. variant-collapse adding `thinking.effortRouting` to bundled Devin
// models).
modelRegistry.registerProvider("devin-advisor-test", {
api: "devin-agent",
apiKey: "test-key",
baseUrl: "https://test.example.com",
models: [
{
id: "no-thinking",
name: "Test No-Thinking",
api: "devin-agent",
reasoning: true,
input: ["text"],
supportsTools: true,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 64_000,
},
],
});
const devin = modelRegistry.find("devin-advisor-test", "no-thinking");
if (!devin) throw new Error("Expected synthetic devin-advisor-test/no-thinking to register");
devinModel = devin;
});
@@ -88,8 +109,8 @@ describe("AgentSession advisor descriptor thinking level", () => {
it("Devin advisor with no configured thinking suffix boots without an unsupported-effort throw", () => {
// Confirm the catalog shape that triggered the bug: `reasoning: true` with
// no controllable `thinking.efforts`. If this drifts upstream the
// regression's assumptions no longer hold.
// no controllable `thinking.efforts`. The synthetic model in `beforeAll`
// guarantees this shape regardless of upstream catalog drift.
expect(devinModel.reasoning).toBe(true);
expect(devinModel.thinking).toBeUndefined();