Merge PR #4882: fix(catalog): collapse Devin GLM-5.2 variants so free 200K model works when quota is exhausted (@oldschoola)
# Conflicts: # packages/catalog/src/models.json
This commit is contained in:
@@ -174,6 +174,13 @@
|
||||
|
||||
- Updated cost and token configurations for various models across providers
|
||||
- Renamed several models for consistency (e.g., MiniMax M3, Gemma 4 31B, Qwen variants)
|
||||
### Added
|
||||
|
||||
- Added static fallback seed for Devin's `swe-1-7` model so it is bundled even when catalog generation runs without a Devin session token.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Collapsed Devin's six GLM-5.2 variants into two logical entries (`glm-5-2` for 200K free, `glm-5-2-1m` for 1M paid). The 200K entry routes every thinking effort to the free `glm-5-2` wire UID — never to the quota-gated `glm-5-2-max` or `glm-5-2-none` — so GLM-5.2 works even when the weekly usage quota is exhausted.
|
||||
|
||||
## [16.3.12] - 2026-07-08
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@ import { getGitLabDuoModels } from "@oh-my-pi/pi-ai/providers/gitlab-duo";
|
||||
import { $env } from "@oh-my-pi/pi-utils";
|
||||
import { ANTIGRAVITY_PRIMARY_ENDPOINT, fetchAntigravityDiscoveryModels } from "../src/discovery/antigravity";
|
||||
import { fetchCodexModels } from "../src/discovery/codex";
|
||||
import { DEVIN_STATIC_FALLBACK_MODELS } from "../src/discovery/devin";
|
||||
import { buildGitLabDuoWorkflowFallbackModel } from "../src/discovery/gitlab-duo-workflow";
|
||||
import { createModelManager } from "../src/model-manager";
|
||||
import prevModelsJson from "../src/models.json" with { type: "json" };
|
||||
@@ -541,6 +542,13 @@ async function generateModels() {
|
||||
if (!authoritativeCatalogProviders.has("gitlab-duo-agent")) {
|
||||
allModels.push(buildGitLabDuoWorkflowFallbackModel());
|
||||
}
|
||||
// Seed Devin fallback models so newly released free models (e.g. `swe-1-7`)
|
||||
// are bundled even when catalog generation runs without a Devin session
|
||||
// token. Devin is `dynamicModelsAuthoritative: true`, so live discovery
|
||||
// replaces these at runtime when a key is present.
|
||||
if (!authoritativeCatalogProviders.has("devin")) {
|
||||
allModels.push(...DEVIN_STATIC_FALLBACK_MODELS);
|
||||
}
|
||||
// Seed Fireworks "Fast" serving-path variants (`<id>-fast`). Fast routers are
|
||||
// not enumerated by the serverless control-plane list, so discovery never
|
||||
// surfaces them; the seed projects each base entry into a fast variant.
|
||||
|
||||
@@ -149,3 +149,27 @@ function normalizeDevinModels(
|
||||
}
|
||||
return [...byId.values()].sort((a, b) => a.id.localeCompare(b.id));
|
||||
}
|
||||
|
||||
/**
|
||||
* Static fallback Devin models for catalog generation without a live API key.
|
||||
*
|
||||
* Devin discovery requires an authenticated session token; when catalog
|
||||
* generation runs without one, these seeds ensure new free models (like
|
||||
* `swe-1-7`) are still bundled. Live discovery is authoritative — when it
|
||||
* succeeds, it replaces these seeds entirely (stale entries are pruned).
|
||||
*/
|
||||
export const DEVIN_STATIC_FALLBACK_MODELS: readonly ModelSpec<"devin-agent">[] = [
|
||||
{
|
||||
id: "swe-1-7",
|
||||
name: "SWE-1.7",
|
||||
api: "devin-agent",
|
||||
provider: "devin",
|
||||
baseUrl: DEVIN_DEFAULT_BASE_URL,
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
supportsTools: true,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 262_000,
|
||||
maxTokens: DEFAULT_MAX_TOKENS,
|
||||
},
|
||||
];
|
||||
|
||||
@@ -532,6 +532,39 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = {
|
||||
},
|
||||
[Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
),
|
||||
// GLM-5.2 200K — only the base wire UID `glm-5-2` is free on Devin's
|
||||
// Coding Plan (verified via streamDevin: `glm-5-2-none` and `glm-5-2-max`
|
||||
// both return "weekly usage quota exhausted" while `glm-5-2` streams
|
||||
// successfully). Route every effort to `glm-5-2` so the collapsed entry
|
||||
// is always free; include the paid 200K variants as members so they are
|
||||
// hidden from the model list. The 1M-context variants stay as separate
|
||||
// paid entries (collapsed below).
|
||||
{
|
||||
id: "glm-5-2",
|
||||
name: "GLM-5.2",
|
||||
members: ["glm-5-2", "glm-5-2-none", "glm-5-2-max"],
|
||||
routing: {
|
||||
[Effort.High]: "glm-5-2",
|
||||
[Effort.XHigh]: "glm-5-2",
|
||||
},
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.High, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
},
|
||||
},
|
||||
// GLM-5.2 1M — paid variants that consume weekly quota. Collapse the
|
||||
// three 1M-context variants into one entry with proper effort routing.
|
||||
devinTierFamily(
|
||||
"glm-5-2-1m",
|
||||
"GLM-5.2 1M",
|
||||
{
|
||||
off: "glm-5-2-none-1m",
|
||||
high: "glm-5-2-1m",
|
||||
xhigh: "glm-5-2-max-1m",
|
||||
},
|
||||
[Effort.High, Effort.XHigh],
|
||||
),
|
||||
],
|
||||
};
|
||||
|
||||
|
||||
@@ -791,3 +791,75 @@ describe("antigravity discovery collapsing", () => {
|
||||
expect(models?.[0]?.baseUrl).toBe(ANTIGRAVITY_PRIMARY_ENDPOINT);
|
||||
});
|
||||
});
|
||||
|
||||
describe("Devin GLM-5.2 collapse", () => {
|
||||
function devinMemberSpec(id: string, overrides: Partial<ModelSpec<"devin-agent">> = {}): ModelSpec<"devin-agent"> {
|
||||
return {
|
||||
id,
|
||||
name: id,
|
||||
api: "devin-agent",
|
||||
provider: "devin",
|
||||
baseUrl: "https://server.codeium.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
supportsTools: true,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 64_000,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
it("collapses the three 200K GLM-5.2 variants into one logical entry routing all efforts to the free glm-5-2 wire UID", () => {
|
||||
const out = collapseEffortVariants(
|
||||
[
|
||||
devinMemberSpec("glm-5-2"),
|
||||
devinMemberSpec("glm-5-2-max"),
|
||||
devinMemberSpec("glm-5-2-none", { reasoning: false }),
|
||||
],
|
||||
DEVIN_VARIANT_COLLAPSE_TABLE,
|
||||
);
|
||||
|
||||
expect(out).toHaveLength(1);
|
||||
const spec = out[0];
|
||||
expect(spec?.id).toBe("glm-5-2");
|
||||
expect(spec?.thinking?.effortRouting).toEqual({
|
||||
high: "glm-5-2",
|
||||
xhigh: "glm-5-2",
|
||||
});
|
||||
});
|
||||
|
||||
it("routes every effort to glm-5-2 (never to the quota-gated glm-5-2-max or glm-5-2-none)", () => {
|
||||
const out = collapseEffortVariants(
|
||||
[devinMemberSpec("glm-5-2"), devinMemberSpec("glm-5-2-max")],
|
||||
DEVIN_VARIANT_COLLAPSE_TABLE,
|
||||
);
|
||||
|
||||
const spec = out[0];
|
||||
const routing = spec?.thinking?.effortRouting ?? {};
|
||||
for (const wire of Object.values(routing)) {
|
||||
expect(wire).toBe("glm-5-2");
|
||||
}
|
||||
});
|
||||
|
||||
it("collapses the three 1M GLM-5.2 variants into one paid entry with proper effort routing", () => {
|
||||
const out = collapseEffortVariants(
|
||||
[
|
||||
devinMemberSpec("glm-5-2-1m", { contextWindow: 1_000_000 }),
|
||||
devinMemberSpec("glm-5-2-max-1m", { contextWindow: 1_000_000 }),
|
||||
devinMemberSpec("glm-5-2-none-1m", { contextWindow: 1_000_000, reasoning: false }),
|
||||
],
|
||||
DEVIN_VARIANT_COLLAPSE_TABLE,
|
||||
);
|
||||
|
||||
expect(out).toHaveLength(1);
|
||||
const spec = out[0];
|
||||
expect(spec?.id).toBe("glm-5-2-1m");
|
||||
expect(spec?.contextWindow).toBe(1_000_000);
|
||||
expect(spec?.thinking?.effortRouting).toEqual({
|
||||
off: "glm-5-2-none-1m",
|
||||
high: "glm-5-2-1m",
|
||||
xhigh: "glm-5-2-max-1m",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -34,16 +34,37 @@ describe("AgentSession advisor descriptor thinking level", () => {
|
||||
sharedDir = TempDir.createSync("@pi-advisor-devin-thinking-shared-");
|
||||
authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db"));
|
||||
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
||||
// Seeding a runtime API key exposes the bundled Devin catalog for
|
||||
// `resolveAdvisorRoleSelection` / `getAvailable()` without any live
|
||||
// network discovery.
|
||||
authStorage.setRuntimeApiKey("devin", "test-key");
|
||||
modelRegistry = new ModelRegistry(authStorage);
|
||||
const anthropic = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
const devin = getBundledModel("devin", "glm-5-2");
|
||||
if (!anthropic) throw new Error("Expected bundled anthropic/claude-sonnet-4-5 to exist");
|
||||
if (!devin) throw new Error("Expected bundled devin/glm-5-2 to exist");
|
||||
anthropicModel = anthropic;
|
||||
|
||||
// Register a synthetic `devin-agent` provider with a reasoning model
|
||||
// that has NO `thinking` metadata. This is the exact catalog shape that
|
||||
// triggered #4579: `reasoning: true` with no controllable effort surface.
|
||||
// Using a synthetic model avoids brittleness from upstream catalog drift
|
||||
// (e.g. variant-collapse adding `thinking.effortRouting` to bundled Devin
|
||||
// models).
|
||||
modelRegistry.registerProvider("devin-advisor-test", {
|
||||
api: "devin-agent",
|
||||
apiKey: "test-key",
|
||||
baseUrl: "https://test.example.com",
|
||||
models: [
|
||||
{
|
||||
id: "no-thinking",
|
||||
name: "Test No-Thinking",
|
||||
api: "devin-agent",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
supportsTools: true,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 64_000,
|
||||
},
|
||||
],
|
||||
});
|
||||
const devin = modelRegistry.find("devin-advisor-test", "no-thinking");
|
||||
if (!devin) throw new Error("Expected synthetic devin-advisor-test/no-thinking to register");
|
||||
devinModel = devin;
|
||||
});
|
||||
|
||||
@@ -88,8 +109,8 @@ describe("AgentSession advisor descriptor thinking level", () => {
|
||||
|
||||
it("Devin advisor with no configured thinking suffix boots without an unsupported-effort throw", () => {
|
||||
// Confirm the catalog shape that triggered the bug: `reasoning: true` with
|
||||
// no controllable `thinking.efforts`. If this drifts upstream the
|
||||
// regression's assumptions no longer hold.
|
||||
// no controllable `thinking.efforts`. The synthetic model in `beforeAll`
|
||||
// guarantees this shape regardless of upstream catalog drift.
|
||||
expect(devinModel.reasoning).toBe(true);
|
||||
expect(devinModel.thinking).toBeUndefined();
|
||||
|
||||
|
||||
Reference in New Issue
Block a user