diff --git a/docs/IMPLEMENTATION.md b/docs/IMPLEMENTATION.md index 92dd447d2..0bd2847a4 100644 --- a/docs/IMPLEMENTATION.md +++ b/docs/IMPLEMENTATION.md @@ -253,6 +253,8 @@ Provider and model configuration lives in JSON settings files. The global file h `models` is always an array (single- and multi-model providers are uniform). `defaultModel` (or the first entry) is used when no model is selected. With exactly one provider configured, `defaultProvider` may be omitted. + Optional `contextWindow` (positive number, tokens) overrides the models.dev / heuristic window for that provider. `loadConfig` applies it after `resolveProvider` via `setProviderContextWindowOverrides`, keyed as `:` for every model on a provider that sets the field, plus the bare model id for the resolved provider so occupancy lookups that only have `source.model` still hit. It takes precedence over models.dev metadata and family heuristics. OAuth-projected Codex/xAI providers still drop the field: the synthetic `ProviderSettings` written by the projection overwrites the settings entry and does not copy `contextWindow`, so a hand-edited value on `codex/...` or `xai/...` is ignored. API-key providers are unaffected. + Optional `tools` block to arm the outer per-tool wall-clock budget (unset leaves the watchdog unarmed): ```json @@ -463,7 +465,7 @@ Mid-run queue/steer/interrupt state is a pure state machine in `src/tui/session- - Pricing fetched from models.dev, cached, refreshed on a background interval - `faremeter` converts `inference.usage` counts into a formatted `$X.XXXX` cost -- The same models.dev payload also yields per-model context windows (`limit.context`), captured into the pricing cache (`contextWindows`) and loaded into `src/provider/context-window.ts`. `compactionThresholdFor(model)` returns ~60% of that window (falling back to per-family heuristics, then 128k) to size proactive compaction. Unknown/family-only models still get a sane default. +- The same models.dev payload also yields per-model context windows (`limit.context`), captured into the pricing cache (`contextWindows`) and loaded into `src/provider/context-window.ts`. A provider-level `contextWindow` settings override, when present, beats that metadata. `compactionThresholdFor(model)` returns ~60% of that window (falling back to per-family heuristics, then 128k) to size proactive compaction. Unknown/family-only models still get a sane default. ### Plugin system diff --git a/src/config.test.ts b/src/config.test.ts index 9fd350b49..5565c0f17 100644 --- a/src/config.test.ts +++ b/src/config.test.ts @@ -46,6 +46,7 @@ import { saveState } from "./session/state.js"; import { filterMcpServersForConnect } from "./trust/project-trust.js"; import { createExaMCPServerConfig } from "./mcp/exa.js"; import { withFileLogSink } from "../tests/helpers/file-log-sink.js"; +import { setProviderContextWindowOverrides } from "./provider/context-window.js"; const BUILTIN_EXA_MCP = createExaMCPServerConfig(); const originalFetch = globalThis.fetch; @@ -57,6 +58,7 @@ beforeEach(() => { afterEach(() => { globalThis.fetch = originalFetch; resetGoModelDiscoveryForTests(); + setProviderContextWindowOverrides(undefined); }); function assertConfigured( diff --git a/src/config/index.ts b/src/config/index.ts index 2552bdd36..fb869b425 100644 --- a/src/config/index.ts +++ b/src/config/index.ts @@ -16,6 +16,10 @@ import { validateEffort, type ReasoningEffort, } from "../provider/reasoning-effort.js"; +import { + buildProviderContextWindowOverrides, + setProviderContextWindowOverrides, +} from "../provider/context-window.js"; import { bootstrapPricingMetadata } from "../cost/pricing-metadata.js"; import { defaultPricingCachePath, @@ -924,6 +928,14 @@ export async function loadConfig( }; } + setProviderContextWindowOverrides( + buildProviderContextWindowOverrides( + settingsForResolution?.providers ?? {}, + resolved.providerName, + resolved.model, + ), + ); + // Enforce model/effort compatibility at the boundary. The modal only offers // supported levels, but a hand-edited local settings file can pair an effort // with a model that does not accept it; reject it here rather than shipping an diff --git a/src/config/settings.ts b/src/config/settings.ts index 2ada83676..01886fbca 100644 --- a/src/config/settings.ts +++ b/src/config/settings.ts @@ -39,6 +39,9 @@ export interface ProviderSettings { // provider regardless of model pricing — e.g. a prepaid coding plan or a // gateway whose models.dev prices do not apply. free?: boolean; + // Token-window override for compaction and the status-bar meter. Applied at + // config load into contextWindowFor. OAuth-projected Codex/xAI entries drop + // this field, so a hand-edited value on those providers is ignored. contextWindow?: number; // When true, this provider uses a Bifrost virtual key (sk-bf-...). // The marker causes the inference source to route through the Bifrost diff --git a/src/cost/cost-summary.test.ts b/src/cost/cost-summary.test.ts index ef11c197d..a7656fd9d 100644 --- a/src/cost/cost-summary.test.ts +++ b/src/cost/cost-summary.test.ts @@ -1,6 +1,9 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { setModelContextWindows } from "../provider/context-window.js"; +import { + setModelContextWindows, + setProviderContextWindowOverrides, +} from "../provider/context-window.js"; import { buildCostSummary, formatCostCommandOutput, @@ -9,7 +12,10 @@ import { } from "./cost-summary.js"; import type { CostSummaryInput } from "./cost-summary.js"; -afterEach(() => setModelContextWindows(undefined)); +afterEach(() => { + setModelContextWindows(undefined); + setProviderContextWindowOverrides(undefined); +}); const baseInput: CostSummaryInput = { modelId: "test-model", diff --git a/src/provider/context-window.test.ts b/src/provider/context-window.test.ts index 76b9fe50b..187b5bd5b 100644 --- a/src/provider/context-window.test.ts +++ b/src/provider/context-window.test.ts @@ -1,13 +1,20 @@ import { describe, expect, it, afterEach } from "bun:test"; +import { mkdtemp, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { loadConfig } from "../config/index.js"; import { + buildProviderContextWindowOverrides, contextWindowFor, hasContextWindowFor, setModelContextWindows, + setProviderContextWindowOverrides, } from "./context-window.js"; describe("contextWindowFor", () => { afterEach(() => { setModelContextWindows(undefined); + setProviderContextWindowOverrides(undefined); }); it("resolves a custom-provider-prefixed id against the bare model registry entry", () => { @@ -34,4 +41,126 @@ describe("contextWindowFor", () => { setModelContextWindows({ "grok-4.5": 500_000 }); expect(hasContextWindowFor("xai/thegreataxios:grok-4.5")).toBe(true); }); + + it("lets a provider override beat models.dev registry metadata", () => { + setModelContextWindows({ "fp-large": 200_000 }); + setProviderContextWindowOverrides({ "fp-large": 32_000 }); + expect(contextWindowFor("fp-large")).toBe(32_000); + expect(contextWindowFor("firepass:fp-large")).toBe(32_000); + }); + + it("lets a provider override beat the family heuristic", () => { + setProviderContextWindowOverrides({ "claude-sonnet": 8_000 }); + expect(contextWindowFor("claude-sonnet")).toBe(8_000); + }); + + it("keeps the override across a later models.dev registry replace", () => { + setProviderContextWindowOverrides({ "fp-large": 32_000 }); + setModelContextWindows({ "fp-large": 200_000 }); + expect(contextWindowFor("fp-large")).toBe(32_000); + }); + + it("reports confidence for an override even when the registry is empty", () => { + setProviderContextWindowOverrides({ "fp-large": 32_000 }); + expect(hasContextWindowFor("fp-large")).toBe(true); + expect(hasContextWindowFor("firepass:fp-large")).toBe(true); + }); +}); + +describe("buildProviderContextWindowOverrides", () => { + it("keys : for every model and the bare id only for the resolved provider", () => { + const overrides = buildProviderContextWindowOverrides( + { + firepass: { + models: ["fp-large", "fp-small"], + contextWindow: 32_000, + }, + other: { + models: ["fp-large", "other-model"], + contextWindow: 64_000, + }, + skipped: { + models: ["no-window"], + }, + }, + "firepass", + "fp-large", + ); + expect(overrides).toEqual({ + "firepass:fp-large": 32_000, + "firepass:fp-small": 32_000, + "fp-large": 32_000, + "fp-small": 32_000, + "other:fp-large": 64_000, + "other:other-model": 64_000, + }); + }); + + it("includes the resolved model even when it is not in the provider model list", () => { + const overrides = buildProviderContextWindowOverrides( + { + firepass: { models: ["fp-large"], contextWindow: 32_000 }, + }, + "firepass", + "fp-cli", + ); + expect(overrides["firepass:fp-cli"]).toBe(32_000); + expect(overrides["fp-cli"]).toBe(32_000); + }); + + it("skips non-positive windows", () => { + expect( + buildProviderContextWindowOverrides( + { + firepass: { models: ["fp-large"], contextWindow: 0 }, + other: { models: ["m"], contextWindow: -1 }, + }, + "firepass", + "fp-large", + ), + ).toEqual({}); + }); +}); + +describe("loadConfig provider contextWindow", () => { + afterEach(() => { + setModelContextWindows(undefined); + setProviderContextWindowOverrides(undefined); + }); + + it("applies providers..contextWindow after resolveProvider", async () => { + const cwd = await mkdtemp(join(tmpdir(), "ic-cw-")); + const globalPath = join(cwd, "global.json"); + await writeFile( + globalPath, + JSON.stringify({ + defaultProvider: "firepass", + providers: { + firepass: { + baseURL: "https://firepass.example/v1", + apiKey: "test-key", + models: ["fp-large", "fp-small"], + defaultModel: "fp-large", + contextWindow: 32_000, + }, + other: { + baseURL: "https://other.example/v1", + apiKey: "other-key", + models: ["fp-large"], + contextWindow: 64_000, + }, + }, + }), + ); + + setModelContextWindows({ "fp-large": 200_000 }); + await loadConfig(["--cwd", cwd, "hello"], { + globalSettingsPath: globalPath, + }); + + expect(contextWindowFor("firepass:fp-large")).toBe(32_000); + expect(contextWindowFor("fp-large")).toBe(32_000); + expect(contextWindowFor("firepass:fp-small")).toBe(32_000); + expect(contextWindowFor("other:fp-large")).toBe(64_000); + }); }); diff --git a/src/provider/context-window.ts b/src/provider/context-window.ts index 4789d687b..5bb72dbe7 100644 --- a/src/provider/context-window.ts +++ b/src/provider/context-window.ts @@ -1,6 +1,7 @@ // Approximate total context window (tokens) per model, used to render // context-window occupancy in the status bar and to size compaction. When -// models.dev metadata is loaded at startup it takes priority; otherwise we fall +// a provider settings override is applied at config load it takes priority; +// otherwise models.dev metadata loaded at startup wins; otherwise we fall // back to conservative per-family floors, and finally a common 128k window. import type { TokenUsage } from "@intx/types/runtime"; @@ -23,12 +24,59 @@ export function contextTokensFromUsage(usage: TokenUsage | undefined): number { // Exact model-id match wins over the family heuristics below. let contextWindowRegistry: Record = {}; +// Populated at config load from providers..contextWindow. Survives a +// later models.dev refresh because it lives beside the registry, not in it. +let contextWindowOverrides: Record = {}; + export function setModelContextWindows( windows: Record | undefined, ): void { contextWindowRegistry = windows ?? {}; } +export function setProviderContextWindowOverrides( + windows: Record | undefined, +): void { + contextWindowOverrides = windows ?? {}; +} + +export type ProviderContextWindowSource = { + models: readonly string[]; + contextWindow?: number; +}; + +function isPositiveWindow(window: number | undefined): window is number { + return window !== undefined && Number.isFinite(window) && window > 0; +} + +// Key `:` for every model on a provider that sets the knob. +// Bare model ids are added only for the resolved provider so occupancy +// lookups that only have `source.model` still hit, without letting another +// provider's same model id steal the bare slot. +export function buildProviderContextWindowOverrides( + providers: Record, + resolvedProviderName: string, + resolvedModel: string, +): Record { + const overrides: Record = {}; + for (const [name, provider] of Object.entries(providers)) { + const window = provider.contextWindow; + if (!isPositiveWindow(window)) continue; + const models = new Set(provider.models); + if (name === resolvedProviderName && resolvedModel.length > 0) { + models.add(resolvedModel); + } + for (const model of models) { + if (model.length === 0) continue; + overrides[`${name}:${model}`] = window; + if (name === resolvedProviderName) { + overrides[model] = window; + } + } + } + return overrides; +} + function heuristicWindow(model: string): number { const m = model.toLowerCase(); if (m.includes("gpt-6")) return 1_000_000; @@ -60,20 +108,33 @@ function lookupCandidates(model: string): string[] { return [model, bareModel, `${canonicalProvider}/${bareModel}`]; } -/** True when the registry has an entry for `model` under any known form, so a - * caller can distinguish a confident lookup from the heuristic fallback. */ +function lookupWindow( + table: Record, + model: string, +): number | undefined { + for (const candidate of lookupCandidates(model)) { + const exact = table[candidate]; + if (exact !== undefined) return exact; + } + return undefined; +} + +/** True when an override or the registry has an entry for `model` under any + * known form, so a caller can distinguish a confident lookup from the + * heuristic fallback. */ export function hasContextWindowFor(model: string): boolean { - return lookupCandidates(model).some( - (candidate) => contextWindowRegistry[candidate] !== undefined, + return ( + lookupWindow(contextWindowOverrides, model) !== undefined || + lookupWindow(contextWindowRegistry, model) !== undefined ); } export function contextWindowFor(model: string): number { - for (const candidate of lookupCandidates(model)) { - const exact = contextWindowRegistry[candidate]; - if (exact !== undefined) return exact; - } - return heuristicWindow(model); + return ( + lookupWindow(contextWindowOverrides, model) ?? + lookupWindow(contextWindowRegistry, model) ?? + heuristicWindow(model) + ); } // Fraction of the window at which proactive compaction should fire. Kept well diff --git a/tests/unit/config.test.ts b/tests/unit/config.test.ts index e67953e64..f45a26668 100644 --- a/tests/unit/config.test.ts +++ b/tests/unit/config.test.ts @@ -1,4 +1,4 @@ -import { test, expect } from "bun:test"; +import { afterEach, test, expect } from "bun:test"; import { mkdtemp, mkdir, @@ -11,8 +11,13 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import { loadConfig } from "../../src/config/index.js"; import { resetPricingMetadataRefreshForTests } from "../../src/cost/pricing-metadata.js"; +import { setProviderContextWindowOverrides } from "../../src/provider/context-window.js"; import { withMockedModuleDuring } from "../helpers/mock-module.js"; +afterEach(() => { + setProviderContextWindowOverrides(undefined); +}); + // Rejects immediately instead of touching the network. loadConfig's pricing // refresh is fire-and-forget, so a resolved run proves only that the injected // impl was reached — which is exactly the regression this file guards against: diff --git a/tests/unit/context-window.test.ts b/tests/unit/context-window.test.ts index d01047bfb..3a02d5f69 100644 --- a/tests/unit/context-window.test.ts +++ b/tests/unit/context-window.test.ts @@ -10,6 +10,7 @@ import { COMPACTION_RESUME_FRACTION, CONTEXT_METER_DANGER_FRACTION, setModelContextWindows, + setProviderContextWindowOverrides, } from "../../src/provider/context-window.js"; function usage(overrides: Partial): TokenUsage { @@ -23,7 +24,10 @@ function usage(overrides: Partial): TokenUsage { }; } -afterEach(() => setModelContextWindows(undefined)); +afterEach(() => { + setModelContextWindows(undefined); + setProviderContextWindowOverrides(undefined); +}); describe("contextWindowFor", () => { test("returns the gpt-5 family window for codex models", () => {