fix(coding-agent): preserve bundled contextWindow/maxTokens when discovery returns sentinel fallbacks

When cached or freshly-discovered provider models carry UNK_CONTEXT_WINDOW
(222222) / UNK_MAX_TOKENS (8888) sentinels, #mergeResolvedModels was
replacing the bundled model wholesale — wiping out the correct values.

Switch to a field-level merge that preserves the bundled model's
contextWindow and maxTokens when the replacement only has sentinel
fallbacks. Custom models (via #mergeCustomModels) already had this
protection via ?? fallback; provider discoveries didn't.

Fixes the TUI showing 222222/8888 instead of the real context/token
limits for discovered models.
This commit is contained in:
Burke T
2026-05-13 08:31:00 -03:00
parent 7530114c01
commit fcaafda0fa
3 changed files with 60 additions and 1 deletions
+4
View File
@@ -19,6 +19,10 @@
- Changed search truncation metadata/renderer output from match/result-based limits to file-based limits (`fileLimitReached`, `perFileLimitReached`) and updated truncation labels accordingly
- Lowered `read.defaultLimit` default from `500` to `300` lines, and split the per-range context padding into asymmetric `RANGE_LEADING_CONTEXT_LINES = 1` / `RANGE_TRAILING_CONTEXT_LINES = 3` (was symmetric `RANGE_CONTEXT_LINES = 3`). Replay analysis over post-summarizer sessions (`scripts/session-stats/optimize_read_config.py`) showed that bare-path reads are over-provisioned at the median (file p50 = 220 lines) and that most follow-up reads are disjoint hops rather than adjacent extensions — so a smaller default plus narrower leading context reclaims tokens without measurably changing first-cover rate. Trailing context stays at 3 lines to keep anchor-stale recovery on narrow reads. Explicit `read.defaultLimit` overrides in settings are honoured unchanged.
### Fixed
- Fixed model contextWindow and maxTokens defaulting to `UNK_CONTEXT_WINDOW` (222222) / `UNK_MAX_TOKENS` (8888) when cached or freshly-discovered provider models replace bundled models through `ModelRegistry.#mergeResolvedModels`. The merge now preserves the bundled model's values when the replacement only has sentinel fallbacks.
## [15.0.0] - 2026-05-13
### Breaking Changes
@@ -18,6 +18,8 @@ import {
registerCustomApi,
type SimpleStreamOptions,
type ThinkingConfig,
UNK_CONTEXT_WINDOW,
UNK_MAX_TOKENS,
unregisterCustomApis,
} from "@oh-my-pi/pi-ai";
@@ -1053,7 +1055,16 @@ export class ModelRegistry {
const key = `${replacementModel.provider}\u0000${replacementModel.id}`;
const existingIndex = indexByKey.get(key);
if (existingIndex !== undefined) {
merged[existingIndex] = replacementModel;
const existing = merged[existingIndex];
merged[existingIndex] = {
...replacementModel,
contextWindow:
replacementModel.contextWindow === UNK_CONTEXT_WINDOW
? existing.contextWindow
: replacementModel.contextWindow,
maxTokens:
replacementModel.maxTokens === UNK_MAX_TOKENS ? existing.maxTokens : replacementModel.maxTokens,
};
} else {
merged.push(replacementModel);
indexByKey.set(key, merged.length - 1);
@@ -2121,4 +2121,48 @@ describe("ModelRegistry", () => {
expect(model?.isOAuth).toBeUndefined();
});
});
test("cached discovery with UNK contextWindow preserves bundled value", () => {
// Configure openai as a discoverable provider through models.json
writeRawModelsJson({
openai: {
baseUrl: "https://my-proxy.example.com/v1",
apiKey: "TEST_KEY",
api: "openai-completions",
discovery: { type: "openai-models-list" },
models: [],
},
});
// Pre-populate the cache with a model that has UNK sentinel values
// (simulating a discovery that didn't return limit.context)
writeModelCache<"openai-completions">(
"openai",
Date.now(),
[
{
id: "gpt-4o",
name: "GPT-4o",
api: "openai-completions",
provider: "openai",
baseUrl: "https://my-proxy.example.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 222_222, // UNK_CONTEXT_WINDOW
maxTokens: 8_888, // UNK_MAX_TOKENS
},
],
true,
cacheDbPath,
);
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const model = registry.find("openai", "gpt-4o");
expect(model).toBeDefined();
// The bundled gpt-4o has a correct contextWindow, not the UNK sentinel
expect(model!.contextWindow).not.toBe(222_222);
expect(model!.contextWindow).toBeGreaterThan(100_000);
expect(model!.maxTokens).not.toBe(8_888);
expect(model!.maxTokens).toBeGreaterThan(1000);
});
});