fix(coding-agent): preserve bundled contextWindow/maxTokens when discovery returns sentinel fallbacks
When cached or freshly-discovered provider models carry UNK_CONTEXT_WINDOW (222222) / UNK_MAX_TOKENS (8888) sentinels, #mergeResolvedModels was replacing the bundled model wholesale — wiping out the correct values. Switch to a field-level merge that preserves the bundled model's contextWindow and maxTokens when the replacement only has sentinel fallbacks. Custom models (via #mergeCustomModels) already had this protection via ?? fallback; provider discoveries didn't. Fixes the TUI showing 222222/8888 instead of the real context/token limits for discovered models.
This commit is contained in:
@@ -19,6 +19,10 @@
|
||||
- Changed search truncation metadata/renderer output from match/result-based limits to file-based limits (`fileLimitReached`, `perFileLimitReached`) and updated truncation labels accordingly
|
||||
- Lowered `read.defaultLimit` default from `500` to `300` lines, and split the per-range context padding into asymmetric `RANGE_LEADING_CONTEXT_LINES = 1` / `RANGE_TRAILING_CONTEXT_LINES = 3` (was symmetric `RANGE_CONTEXT_LINES = 3`). Replay analysis over post-summarizer sessions (`scripts/session-stats/optimize_read_config.py`) showed that bare-path reads are over-provisioned at the median (file p50 = 220 lines) and that most follow-up reads are disjoint hops rather than adjacent extensions — so a smaller default plus narrower leading context reclaims tokens without measurably changing first-cover rate. Trailing context stays at 3 lines to keep anchor-stale recovery on narrow reads. Explicit `read.defaultLimit` overrides in settings are honoured unchanged.
|
||||
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed model contextWindow and maxTokens defaulting to `UNK_CONTEXT_WINDOW` (222222) / `UNK_MAX_TOKENS` (8888) when cached or freshly-discovered provider models replace bundled models through `ModelRegistry.#mergeResolvedModels`. The merge now preserves the bundled model's values when the replacement only has sentinel fallbacks.
|
||||
## [15.0.0] - 2026-05-13
|
||||
### Breaking Changes
|
||||
|
||||
|
||||
@@ -18,6 +18,8 @@ import {
|
||||
registerCustomApi,
|
||||
type SimpleStreamOptions,
|
||||
type ThinkingConfig,
|
||||
UNK_CONTEXT_WINDOW,
|
||||
UNK_MAX_TOKENS,
|
||||
unregisterCustomApis,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
|
||||
@@ -1053,7 +1055,16 @@ export class ModelRegistry {
|
||||
const key = `${replacementModel.provider}\u0000${replacementModel.id}`;
|
||||
const existingIndex = indexByKey.get(key);
|
||||
if (existingIndex !== undefined) {
|
||||
merged[existingIndex] = replacementModel;
|
||||
const existing = merged[existingIndex];
|
||||
merged[existingIndex] = {
|
||||
...replacementModel,
|
||||
contextWindow:
|
||||
replacementModel.contextWindow === UNK_CONTEXT_WINDOW
|
||||
? existing.contextWindow
|
||||
: replacementModel.contextWindow,
|
||||
maxTokens:
|
||||
replacementModel.maxTokens === UNK_MAX_TOKENS ? existing.maxTokens : replacementModel.maxTokens,
|
||||
};
|
||||
} else {
|
||||
merged.push(replacementModel);
|
||||
indexByKey.set(key, merged.length - 1);
|
||||
|
||||
@@ -2121,4 +2121,48 @@ describe("ModelRegistry", () => {
|
||||
expect(model?.isOAuth).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
test("cached discovery with UNK contextWindow preserves bundled value", () => {
|
||||
// Configure openai as a discoverable provider through models.json
|
||||
writeRawModelsJson({
|
||||
openai: {
|
||||
baseUrl: "https://my-proxy.example.com/v1",
|
||||
apiKey: "TEST_KEY",
|
||||
api: "openai-completions",
|
||||
discovery: { type: "openai-models-list" },
|
||||
models: [],
|
||||
},
|
||||
});
|
||||
// Pre-populate the cache with a model that has UNK sentinel values
|
||||
// (simulating a discovery that didn't return limit.context)
|
||||
writeModelCache<"openai-completions">(
|
||||
"openai",
|
||||
Date.now(),
|
||||
[
|
||||
{
|
||||
id: "gpt-4o",
|
||||
name: "GPT-4o",
|
||||
api: "openai-completions",
|
||||
provider: "openai",
|
||||
baseUrl: "https://my-proxy.example.com/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 222_222, // UNK_CONTEXT_WINDOW
|
||||
maxTokens: 8_888, // UNK_MAX_TOKENS
|
||||
},
|
||||
],
|
||||
true,
|
||||
cacheDbPath,
|
||||
);
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath);
|
||||
const model = registry.find("openai", "gpt-4o");
|
||||
|
||||
expect(model).toBeDefined();
|
||||
// The bundled gpt-4o has a correct contextWindow, not the UNK sentinel
|
||||
expect(model!.contextWindow).not.toBe(222_222);
|
||||
expect(model!.contextWindow).toBeGreaterThan(100_000);
|
||||
expect(model!.maxTokens).not.toBe(8_888);
|
||||
expect(model!.maxTokens).toBeGreaterThan(1000);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user