diff --git a/CHANGELOG.md b/CHANGELOG.md index b614604db..8493c0934 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,8 @@ - Roo Code support. Roo Code was discontinued in May 2026 (its repository is archived). ### Fixed +- **Codex long-context calls now use the published higher rates at their prompt-token threshold.** The bundled and live pricing readers retain LiteLLM's context tiers, and Codex applies them per call; below-threshold calls and other providers retain base pricing, and explicit price overrides still win. Audit totals use the same per-call calculation. Pricing and daily-cache versions advance so warm history re-derives with the corrected rates. The regenerated bundle preserves the snapshot refresh's base-rate guards and exact-key carry-forward for previously priced models. (#1076) +- Bundled pricing is refreshed: 5,896 primary entries (298 added, 158 repriced by upstream, 364 no-longer-published ids moved to the fallback table, which grows from 212 to 543). `grok-4.6` base input moves to $1.25/M; `claude-opus-4` is pinned at Anthropic list price since LiteLLM dropped it. - **OMP: side calls outside the chat transcript now count.** OMP logs `find`, `judge`, TTSR checks and cache warming as `model_usage` entries, not assistant messages, and the parser read only messages, so that spend was missing. On one real week that was 605 `find` decisions on TypeSafe Jev and a $0.048 cache warm. Each side call now counts at its reported cost, with its purpose in the tools column as `omp:find`, `omp:cache-warm` and so on, and a zero reported cost is priced from tokens. A side call never duplicates an assistant turn. OMP sessions re-parse once and settled days re-derive (daily cache v41). - **Text selection works again in the terminal dashboard.** Mouse tracking, added in 0.9.20 so the wheel scrolls the view, makes the terminal send mouse events to the application, and in most terminals that stops click and drag from selecting text, so copying a number out of the dashboard needed Shift held down. Tracking is now off by default and `m` turns it on and off while the dashboard runs. - **Cursor Enterprise and team seats connect in Plan & Usage instead of showing an unrecognized quota response.** Their `usage-summary` carries no individual meter, only a team on-demand meter, which now fills the bar under the Enterprise plan label. On other plans it is listed beside the monthly window. In the macOS app a rate limit, outage or unrecognized response no longer adds sign-in guidance, which is only shown when the session is missing or rejected. Reported in #1546, thanks @paulodearaujo. diff --git a/app/renderer/lib/types.ts b/app/renderer/lib/types.ts index d060d8155..f5f57431d 100644 --- a/app/renderer/lib/types.ts +++ b/app/renderer/lib/types.ts @@ -969,6 +969,15 @@ export type ModelCosts = { cacheReadCostPerToken: number webSearchCostPerRequest: number fastMultiplier: number + /** The vendor's long-context tier, applied once prompt tokens reach the + * threshold. Optional: absent on models without a published tier. */ + longContextTier?: { + thresholdTokens: number + inputCostPerToken: number + outputCostPerToken: number + cacheWriteCostPerToken?: number + cacheReadCostPerToken?: number + } } /** One (provider, model) audit bucket (src/audit-report.ts AuditRow): raw diff --git a/scripts/bundle-litellm.mjs b/scripts/bundle-litellm.mjs index 538aa0f78..f4336ef3f 100644 --- a/scripts/bundle-litellm.mjs +++ b/scripts/bundle-litellm.mjs @@ -54,6 +54,11 @@ const MANUAL_ENTRIES = { // exact gpt-5.6 tuple (Sol-tier: $5/$30 per million, 1.25x cache-write). 'gpt-5.6-codex': [5e-6, 3e-5, 6.25e-6, 5e-7], 'gpt-5.6-codex-max': [5e-6, 3e-5, 6.25e-6, 5e-7], + // LiteLLM dropped `claude-opus-4` upstream (a refresh moves dropped ids to + // the fallback tier), but the Cursor-style alias `claude-4-opus` resolves + // against PRIMARY rows - without this pin the bare id falls to the + // snowflake gateway row and under-prices by 3x. Anthropic list price. + 'claude-opus-4': [15e-6, 75e-6, 18.75e-6, 1.5e-6], } const snapshot = {} @@ -64,11 +69,43 @@ if (!res.ok) throw new Error(`HTTP ${res.status}`) const data = await res.json() const entries = Object.entries(data).filter(([k]) => k !== 'sample_spec') +// The plain context-length tiers only: `input_cost_per_token_above_272k_tokens` +// and siblings. Service-tier variants (`_above_272k_priority_tokens`, +// `_above_272k_flex_tokens`) and the 1-hour cache-write combination are NOT +// context thresholds and are deliberately not matched. The threshold comes +// from the key suffix (272k -> 272000) because LiteLLM carries no numeric +// threshold field (#1076). Mirrored in src/models.ts parseLiteLLMEntry. +const TIER_KEY_RE = /^(input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)_above_(\d+)k_tokens$/ + +function tierOf(entry) { + // Rates are read ONLY from the largest threshold a model carries, so a + // hypothetical entry with two tiers can never mix a smaller tier's rates + // under the bigger threshold. Values must be finite and non-negative, the + // same validation src/models.ts applies on the live path. + const byThreshold = new Map() + for (const [key, value] of Object.entries(entry)) { + const m = TIER_KEY_RE.exec(key) + if (!m || typeof value !== 'number' || !Number.isFinite(value) || value < 0) continue + const tokens = Number(m[2]) * 1000 + const rates = byThreshold.get(tokens) ?? {} + if (m[1] === 'input_cost_per_token') rates.input = value + else if (m[1] === 'output_cost_per_token') rates.output = value + else if (m[1] === 'cache_read_input_token_cost') rates.cacheRead = value + else rates.cacheWrite = value + byThreshold.set(tokens, rates) + } + if (byThreshold.size === 0) return null + const threshold = Math.max(...byThreshold.keys()) + const rates = byThreshold.get(threshold) + if (rates.input == null || rates.output == null) return null + return { threshold, input: rates.input, output: rates.output, cacheWrite: rates.cacheWrite ?? null, cacheRead: rates.cacheRead ?? null } +} + function toVal(entry) { const inp = entry.input_cost_per_token const out = entry.output_cost_per_token if (inp == null || out == null) return null - return [inp, out, entry.cache_creation_input_token_cost ?? null, entry.cache_read_input_token_cost ?? null, entry.provider_specific_entry?.fast ?? null] + return [inp, out, entry.cache_creation_input_token_cost ?? null, entry.cache_read_input_token_cost ?? null, entry.provider_specific_entry?.fast ?? null, tierOf(entry)] } // Pass 1: direct entries (no prefix) get priority @@ -80,8 +117,10 @@ for (const [name, entry] of entries) { // A tuple's completeness: how many optional rate slots (cache-write, // cache-read) carry a published value. A richer upstream row may cite it to // FILL a sparser entry's missing slots - never as a license to re-price it -// (see the fillsOnly guard in Pass 2). -const completeness = (val) => (val[2] != null ? 1 : 0) + (val[3] != null ? 1 : 0) + (val[5] != null ? 1 : 0) +// (see the fillsOnly guard in Pass 2). The tier slot (5) is deliberately not +// counted: tier presence must never decide which row wins, or a tier-bearing +// row would outrank the base-richer row main would have picked. +const completeness = (val) => (val[2] != null ? 1 : 0) + (val[3] != null ? 1 : 0) // Pass 2: prefixed entries - store full key + stripped (slot-fill-only) for (const [name, entry] of entries) { @@ -100,13 +139,18 @@ for (const [name, entry] of entries) { // changes across a refresh; only missing slots fill. The completeness-wins // version re-priced 43 input/output and 34 cache rates by swapping in a // different upstream row (grok-3 3/15 -> 1.25/2.5, mistral-large-latest - // 8/24 -> 0.5/1.5). + // 8/24 -> 0.5/1.5). Slot 5 (the tier object) stays out of the guard: it is + // built fresh per row, so a reference compare is always false and would + // veto fills main performs (it silently dropped the azure cache-read fill + // for gpt-5.4-pro-class rows); and since the replacement only fires when + // the candidate fills a missing BASE slot, the winning row's tier travels + // with its own base rates - splicing the old row's tier onto the new row's + // base would mix two different upstream rows. const fillsOnly = (cand, prev) => cand[0] === prev[0] && cand[1] === prev[1] && (prev[2] == null || cand[2] === prev[2]) && (prev[3] == null || cand[3] === prev[3]) - && (prev[5] == null || cand[5] === prev[5]) if (!existing || (completeness(val) > completeness(existing) && fillsOnly(val, existing))) snapshot[stripped] = val } @@ -236,6 +280,17 @@ const coveredByKey = (key) => for (const [k, v] of [...Object.entries(previousSnapshot), ...Object.entries(previousFallback)]) { if (coveredByKey(k)) continue if (fallback[k] !== undefined) continue + // Same hygiene the gap-fill passes enforce: a carried row must not be + // @pin or date-suffixed (a query can never arrive in those forms - the + // runtime only ever peels them off, never adds them; a vendor-prefixed + // key CAN arrive verbatim, so it stays carriable) and must not be free on + // both ends (an unpriced model falls to expected-free handling, not a $0 + // fallback row). Primary files legitimately hold such rows, so the guard + // lives here: when upstream drops one, carrying it verbatim would + // re-import exactly what tests/pricing-fallback-data.test.ts keeps out + // of this file. + if (/@/.test(k) || /-\d{8}$/.test(k)) continue + if (!validRates(v[0], v[1])) continue fallback[k] = v carried += 1 } diff --git a/src/audit-report.ts b/src/audit-report.ts index be9b850a5..b729d5ec2 100644 --- a/src/audit-report.ts +++ b/src/audit-report.ts @@ -1,5 +1,5 @@ import { behavioralCallWeight } from './behavioral-weight.js' -import { billableOutputTokens, cacheWriteCostPerToken, fallbackRawModelDisplayName, getModelCosts, getShortModelName, sanitizeModelForDisplay, type ModelCosts } from './models.js' +import { billableOutputTokens, cacheWriteCostPerToken, fallbackRawModelDisplayName, getModelCosts, getShortModelName, sanitizeModelForDisplay, tieredCostsFor, type ModelCosts } from './models.js' import { getProvider } from './providers/index.js' import { formatCost, formatTokens } from './format.js' import { renderTable, type TableColumn } from './text-table.js' @@ -57,6 +57,12 @@ export async function aggregateAudit(projects: ProjectSummary[]): Promise() @@ -75,6 +81,8 @@ export async function aggregateAudit(projects: ProjectSummary[]): Promisek_tokens`), applied + /// when a request's prompt tokens (input + cached input) reach the + /// threshold. Each rate the source published for the tier replaces its base + /// rate; a slot the source omitted keeps the base. Optional: absent on + /// models without a published tier and on tuples predating the extension. + longContextTier?: LongContextTier +} + +/** Long-context pricing tier, e.g. OpenAI's above-272k or Anthropic's + * above-200k rates. `thresholdTokens` is parsed from the source key suffix + * (272k → 272_000) because LiteLLM carries no numeric threshold field. */ +export type LongContextTier = { + thresholdTokens: number + inputCostPerToken: number + outputCostPerToken: number + cacheWriteCostPerToken?: number + cacheReadCostPerToken?: number } /// Providers whose reported `reasoningTokens` are a SUBSET of `outputTokens` @@ -59,22 +76,26 @@ type LiteLLMEntry = { provider_specific_entry?: { fast?: number } } -// [input, output, cacheWrite, cacheRead, fastMultiplier]. The trailing fast -// multiplier is carried straight from LiteLLM's provider_specific_entry.fast so -// new models pick it up automatically — no hand-maintained per-model table. -type SnapshotEntry = [number, number, number | null, number | null, (number | null)?] +// [input, output, cacheWrite, cacheRead, fastMultiplier, longContextTier?]. +// The trailing fast multiplier is carried straight from LiteLLM's +// provider_specific_entry.fast so new models pick it up automatically — no +// hand-maintained per-model table. The optional sixth slot carries the +// vendor's long-context tier; older bundles without it parse unchanged. +type SnapshotTier = { threshold: number, input: number, output: number, cacheWrite: number | null, cacheRead: number | null } +type SnapshotEntry = [number, number, number | null, number | null, (number | null)?, (SnapshotTier | null)?] const LITELLM_URL = 'https://raw.githubusercontent.com/BerriAI/litellm/main/model_prices_and_context_window.json' const CACHE_TTL_MS = 24 * 60 * 60 * 1000 -// Bump whenever a ModelCosts field changes pricing behavior (cacheWriteCostIsExplicit, -// added in #1075/#1078). A cache written under an older/missing version is treated as a -// miss instead of read verbatim, so a stale on-disk file can't reintroduce a killed bug -// for up to CACHE_TTL_MS after an upgrade. +// Bump whenever a ModelCosts field changes pricing behavior (cacheWriteCostIsExplicit +// from #1075/#1078; longContextTier from #1076). A cache written under an older/missing +// version is treated as a miss instead of read verbatim, so a stale on-disk file can't +// reintroduce a killed bug for up to CACHE_TTL_MS after an upgrade. // Also folded into getPricingGenerationKey() below: a resident/snapshot-caching // consumer needs the same "pricing behavior changed" signal this already gives // the on-disk LiteLLM cache, not just the on-disk cache itself. // 4: calculateCost bills an implicit cache-write rate as input for non-Anthropic models. -export const CACHE_SCHEMA_VERSION = 4 +// 5: longContextTier rides ModelCosts, so a cached costs object is tier-aware (#1076). +export const CACHE_SCHEMA_VERSION = 5 const WEB_SEARCH_COST = 0.01 const ONE_HOUR_CACHE_WRITE_MULTIPLIER_FROM_FIVE_MINUTE_RATE = 1.6 @@ -101,6 +122,7 @@ function buildCosts( cacheWrite: number | null | undefined, cacheRead: number | null | undefined, fast: number | null | undefined, + tier?: SnapshotTier | null, ): ModelCosts { return { inputCostPerToken: input, @@ -110,6 +132,13 @@ function buildCosts( webSearchCostPerRequest: WEB_SEARCH_COST, fastMultiplier: fast ?? 1, cacheWriteCostIsExplicit: cacheWrite !== null && cacheWrite !== undefined, + ...(tier ? { longContextTier: { + thresholdTokens: tier.threshold, + inputCostPerToken: tier.input, + outputCostPerToken: tier.output, + ...(tier.cacheWrite !== null ? { cacheWriteCostPerToken: tier.cacheWrite } : {}), + ...(tier.cacheRead !== null ? { cacheReadCostPerToken: tier.cacheRead } : {}), + } } : {}), } } // For grok-4.6, prompt tokens mean input tokens plus cached input tokens for a @@ -119,22 +148,51 @@ function buildCosts( const GROK_4_6_PROMPT_TOKEN_THRESHOLD = 200_000 const GROK_4_6_HIGH_PROMPT_COSTS = buildCosts(4e-6, 12e-6, null, 1e-6, null) +// Providers verified to pass the vendor's long-context surcharge through to +// the bill. The mechanism is data-driven (any model's tier rides its bundled +// or live rates), but APPLYING it is evidence-based: codex bills the OpenAI +// above-272k rate directly (#1076, measured on a real corpus), while a real +// Copilot session on gpt-5.6-terra with ~6M-token prompts billed at the base +// rate (tests/parser.test.ts "(c4) attributed cost tracks recomputed cost") — +// applying the tier there fabricates spend, the exact class #1075 warned +// about. Adding a provider here requires that kind of billing evidence AND +// threading its provider through every calculateCost site that prices it (the +// codex sites and the parser.ts central recompute pass it; the Claude journal +// paths and the copilot residual path do not, so a newly added provider whose +// calls flow through those sites would silently stay tierless). +export const TIERED_PRICING_PROVIDERS: ReadonlySet = new Set(['codex']) + // Swap in the vendor's high tier when a request's prompt crosses the published -// threshold. A user-set exact priceOverride still wins over the built-in tier. -// Kept as a helper so the next tiered model extends this one branch instead of -// copy-pasting the inline condition. -function tieredCostsFor(model: string, baseCosts: ModelCosts, promptTokens: number): ModelCosts { +// threshold. A user-set priceOverride wins over any tier: the override row +// is rebuilt without one, exact or aliased. The generic branch serves the +// models of TIERED_PRICING_PROVIDERS whose rates +// carry a longContextTier (OpenAI's above-272k family, Anthropic's above-200k); +// grok-4.6 predates the data plumbing and stays hardcoded. Each tier rate the +// source published replaces its base rate; omitted slots keep the base. +export function tieredCostsFor(model: string, baseCosts: ModelCosts, promptTokens: number, provider?: string): ModelCosts { if (exactPriceOverrideFor(model)) return baseCosts if (resolveCanonicalModelId(model) === 'grok-4.6' && promptTokens >= GROK_4_6_PROMPT_TOKEN_THRESHOLD) { return GROK_4_6_HIGH_PROMPT_COSTS } + const tier = provider !== undefined && TIERED_PRICING_PROVIDERS.has(provider) + ? baseCosts.longContextTier + : undefined + if (tier && promptTokens >= tier.thresholdTokens) { + return { + ...baseCosts, + inputCostPerToken: tier.inputCostPerToken, + outputCostPerToken: tier.outputCostPerToken, + ...(tier.cacheWriteCostPerToken !== undefined ? { cacheWriteCostPerToken: tier.cacheWriteCostPerToken } : {}), + ...(tier.cacheReadCostPerToken !== undefined ? { cacheReadCostPerToken: tier.cacheReadCostPerToken } : {}), + } + } return baseCosts } function tupleToCosts(raw: SnapshotEntry): ModelCosts { - const [input, output, cacheWrite, cacheRead, fast] = raw - return buildCosts(input, output, cacheWrite, cacheRead, fast) + const [input, output, cacheWrite, cacheRead, fast, tier] = raw + return buildCosts(input, output, cacheWrite, cacheRead, fast, tier) } function applyBuiltinPriceOverrides(pricing: Map): Map { @@ -218,6 +276,38 @@ function safePerTokenRate(n: number | undefined): number | null { return n } +// Plain context-length tiers only; mirrors scripts/bundle-litellm.mjs +// TIER_KEY_RE (service-tier `_priority`/`_flex` variants and the 1-hour +// combination are not context thresholds). The live path needs its own copy +// because the bundler is a standalone .mjs script. +const TIER_KEY_RE = /^(input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)_above_(\d+)k_tokens$/ + +function tierOfLiteLLMEntry(entry: LiteLLMEntry): SnapshotTier | null { + // Rates are read ONLY from the largest threshold a model carries, mirroring + // scripts/bundle-litellm.mjs tierOf, so a two-tier entry can never mix a + // smaller tier's rates under the bigger threshold. + const byThreshold = new Map>>() + for (const [key, value] of Object.entries(entry)) { + const match = TIER_KEY_RE.exec(key) + if (!match || typeof value !== 'number' || !Number.isFinite(value) || value < 0) continue + const tokens = Number(match[2]) * 1000 + const rates = byThreshold.get(tokens) ?? {} + rates[match[1]] = value + byThreshold.set(tokens, rates) + } + if (byThreshold.size === 0) return null + const threshold = Math.max(...byThreshold.keys()) + const rates = byThreshold.get(threshold)! + if (rates.input_cost_per_token === undefined || rates.output_cost_per_token === undefined) return null + return { + threshold, + input: rates.input_cost_per_token, + output: rates.output_cost_per_token, + cacheWrite: rates.cache_creation_input_token_cost ?? null, + cacheRead: rates.cache_read_input_token_cost ?? null, + } +} + export function parseLiteLLMEntry(entry: LiteLLMEntry): ModelCosts | null { // The live LiteLLM map is remote JSON; a null (or non-object) value for a // model would make the field reads below throw and abort the whole pricing @@ -232,6 +322,7 @@ export function parseLiteLLMEntry(entry: LiteLLMEntry): ModelCosts | null { safePerTokenRate(entry.cache_creation_input_token_cost), safePerTokenRate(entry.cache_read_input_token_cost), entry.provider_specific_entry?.fast, + tierOfLiteLLMEntry(entry), ) } @@ -1244,6 +1335,7 @@ export function calculateCost( webSearchRequests: number, speed: 'standard' | 'fast' = 'standard', oneHourCacheCreationTokens = 0, + provider?: string, ): number { const costs = getModelCosts(model) if (!costs) { @@ -1266,7 +1358,7 @@ export function calculateCost( const safeCacheCreation = Math.max(safe(cacheCreationTokens), safeOneHourCacheCreation) const safeFiveMinuteCacheCreation = Math.max(0, safeCacheCreation - safeOneHourCacheCreation) const promptTokens = safe(inputTokens) + safe(cacheReadTokens) - const tieredCosts = tieredCostsFor(model, costs, promptTokens) + const tieredCosts = tieredCostsFor(model, costs, promptTokens, provider) const multiplier = speed === 'fast' ? tieredCosts.fastMultiplier : 1 // Clamp negative inputs to 0. A corrupt JSONL that emits a negative token diff --git a/src/parser.ts b/src/parser.ts index c40878682..78ca33f51 100644 --- a/src/parser.ts +++ b/src/parser.ts @@ -2805,7 +2805,7 @@ function cachedCallToApiCall(call: CachedCall): ParsedApiCall { const costUSD = calculateCost( call.model, u.inputTokens, outputForCost, u.cacheCreationInputTokens, u.cacheReadInputTokens, - u.webSearchRequests, call.speed, u.cacheCreationOneHourTokens, + u.webSearchRequests, call.speed, u.cacheCreationOneHourTokens, call.provider, ) return applyLocalModelSavings({ provider: call.provider, diff --git a/src/providers/codex.ts b/src/providers/codex.ts index d2c41b6d2..f1f28c1b7 100644 --- a/src/providers/codex.ts +++ b/src/providers/codex.ts @@ -1145,7 +1145,9 @@ function createParser(source: SessionSource, seenKeys: Set, capture?: { if (seenKeys.has(dedupKey)) { pendingTools = []; pendingToolSequence = []; pendingSkills = []; pendingUserMessage = ''; pendingOutputChars = 0; pendingLocAdded = 0; pendingLocRemoved = 0; pendingEditFailed = 0; continue } seenKeys.add(dedupKey) - const costUSD = calculateCost(model, estInput, estOutput, 0, 0, 0) + // An estimated prompt can cross a long-context threshold and tier + // itself; measured corpora hold no estimated calls that cross. + const costUSD = calculateCost(model, estInput, estOutput, 0, 0, 0, 'standard', 0, 'codex') pendingTaskCalls.push({ provider: 'codex', @@ -1312,6 +1314,9 @@ function createParser(source: SessionSource, seenKeys: Set, capture?: { billedCacheWriteTokens, cachedInputTokens, 0, + 'standard', + 0, + 'codex', ) pendingTaskCalls.push({ diff --git a/tests/audit-report.test.ts b/tests/audit-report.test.ts index 27726fd59..bc7ce594d 100644 --- a/tests/audit-report.test.ts +++ b/tests/audit-report.test.ts @@ -73,6 +73,25 @@ function makeProject(calls: ParsedApiCall[]): ProjectSummary { } describe('aggregateAudit', () => { + it('tiers the audit recompute per call, never on the bucket sum (#1076 review blocker)', async () => { + // One 300k-prompt gpt-5.6 codex call (past the 272k tier) plus 200 small + // calls whose SUM would blow far past the threshold. The per-call + // recompute must tier only the big call; a bucket-sum tier would have + // repriced all 201 and opened a phantom gap vs attributedCostUSD. + const big = makeCall({ inputTokens: 300_000, outputTokens: 1_000 }, 300_000 * 8e-6 + 1_000 * 3e-5, 'gpt-5.6', 'codex') + const small: ParsedApiCall[] = Array.from({ length: 200 }, (_, i) => + makeCall({ inputTokens: 1_000, outputTokens: 10 }, 1_000 * 4e-6 + 10 * 2e-5, 'gpt-5.6', 'codex')) + const rows = await aggregateAudit([makeProject([big, ...small])]) + expect(rows).toHaveLength(1) + const row = rows[0]! + // Expected: the big call at tier rates (in 8e-6, out 3e-5) + the 200 + // small calls at base (in 4e-6, out 2e-5). + const expected = 300_000 * 8e-6 + 1_000 * 3e-5 + 200 * (1_000 * 4e-6 + 10 * 2e-5) + expect(row.cost.recomputedTotalUSD).toBeCloseTo(expected, 9) + // attributed priced the same way at parse time, so the invariant holds. + expect(row.attributedCostUSD).toBeCloseTo(expected, 9) + }) + it('keeps raw fields and exposes codeburn normalizations', async () => { const anthropicCall = makeCall({ inputTokens: 100, outputTokens: 50, reasoningTokens: 10, cacheReadInputTokens: 200 }, 0.5) const openaiCall = makeCall({ inputTokens: 100, outputTokens: 50, cachedInputTokens: 300 }, 0.5) diff --git a/tests/bundle-litellm.test.ts b/tests/bundle-litellm.test.ts new file mode 100644 index 000000000..0513b3a1f --- /dev/null +++ b/tests/bundle-litellm.test.ts @@ -0,0 +1,134 @@ +import { copyFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { spawnSync } from 'node:child_process' +import { fileURLToPath } from 'node:url' +import { expect, it } from 'vitest' + +it('adds context tiers without repricing base rows or losing exact-key carry-forward (#1478)', () => { + const dir = mkdtempSync(join(tmpdir(), 'codeburn-bundle-tiers-')) + try { + mkdirSync(join(dir, 'scripts')) + mkdirSync(join(dir, 'src/data'), { recursive: true }) + copyFileSync(fileURLToPath(new URL('../scripts/bundle-litellm.mjs', import.meta.url)), join(dir, 'scripts/bundle-litellm.mjs')) + const oldPrimary = { 'retired-model': [1e-6, 2e-6, null, null, null] } + const oldFallback = { + 'grok-latest': [3e-6, 15e-6, null, null, null], + 'qwen3.5-plus': [1e-6, 3e-6, null, null, null], + } + writeFileSync(join(dir, 'src/data/litellm-snapshot.json'), JSON.stringify(oldPrimary)) + writeFileSync(join(dir, 'src/data/pricing-fallback.json'), JSON.stringify(oldFallback)) + const row = (input: number, output: number, extra = {}) => ({ + input_cost_per_token: input, output_cost_per_token: output, ...extra, + }) + const source = { + 'grok-3': row(3e-6, 15e-6), + 'reseller/grok-3': row(1.25e-6, 2.5e-6, { cache_read_input_token_cost: 0.1e-6 }), + 'mistral-large-latest': row(8e-6, 24e-6), + 'reseller/mistral-large-latest': row(0.5e-6, 1.5e-6, { cache_read_input_token_cost: 0.05e-6 }), + 'gpt-5.6': row(4e-6, 20e-6, { + input_cost_per_token_above_128k_tokens: 6e-6, + output_cost_per_token_above_128k_tokens: 25e-6, + cache_read_input_token_cost_above_128k_tokens: 0.6e-6, + input_cost_per_token_above_272k_tokens: 8e-6, + output_cost_per_token_above_272k_tokens: 30e-6, + input_cost_per_token_above_272k_priority_tokens: 99e-6, + }), + 'cache-model': row(1e-6, 2e-6), + 'vendor/cache-model': row(1e-6, 2e-6, { cache_read_input_token_cost: 0.1e-6 }), + // A published cache rate is never overwritten: the vendor row is MORE + // complete (adds cache-write, changes cache-read) but a fill must keep + // every filled slot verbatim, so nothing swaps in. + 'cache-own': row(1e-6, 2e-6, { cache_read_input_token_cost: 0.2e-6 }), + 'vendor/cache-own': row(1e-6, 2e-6, { + cache_creation_input_token_cost: 0.5e-6, + cache_read_input_token_cost: 0.9e-6, + }), + // Negative and null tier values are filtered before the largest + // threshold is chosen, so the broken 200k/300k tiers cannot shadow + // the valid 128k one (whose cache slots still map). + 'tier-guard': row(1e-6, 2e-6, { + input_cost_per_token_above_128k_tokens: 2e-6, + output_cost_per_token_above_128k_tokens: 4e-6, + cache_creation_input_token_cost_above_128k_tokens: 0.5e-6, + cache_read_input_token_cost_above_128k_tokens: 0.05e-6, + input_cost_per_token_above_200k_tokens: -1, + output_cost_per_token_above_200k_tokens: -1, + input_cost_per_token_above_300k_tokens: null, + output_cost_per_token_above_300k_tokens: null, + }), + // A tier-bearing bare row must still gain a missing cache slot from a + // prefixed row with identical base rates: the tier object (slot 5) is + // rebuilt fresh per row, so any reference compare on it is always false + // and would veto this fill. + 'direct-tier': row(5e-6, 25e-6, { + input_cost_per_token_above_272k_tokens: 8e-6, + output_cost_per_token_above_272k_tokens: 30e-6, + }), + 'azure/direct-tier': row(5e-6, 25e-6, { + input_cost_per_token_above_272k_tokens: 8e-6, + output_cost_per_token_above_272k_tokens: 30e-6, + cache_read_input_token_cost: 3e-6, + }), + // Among two prefixed aliases of an absent bare key, the base-richer row + // wins even though the sparser one carries a tier: tier presence does + // not count toward completeness, and the winner's tier (here, none) + // travels with its own base rates. + 'ven/alias-model': row(2e-6, 6e-6, { + cache_read_input_token_cost: 0.2e-6, + input_cost_per_token_above_128k_tokens: 3e-6, + output_cost_per_token_above_128k_tokens: 9e-6, + }), + 'corp/alias-model': row(2e-6, 6e-6, { + cache_read_input_token_cost: 0.2e-6, + cache_creation_input_token_cost: 0.4e-6, + }), + // A prefix or date variant does not cover the old bare query. + 'gateway/x-ai/grok-latest': row(2e-6, 4e-6), + 'qwen/qwen3.5-plus-20260420': row(2e-6, 4e-6), + } + writeFileSync(join(dir, 'source.json'), JSON.stringify(source)) + const run = spawnSync(process.execPath, ['--input-type=module', '-e', ` + import { readFileSync } from 'node:fs'; + const source = JSON.parse(readFileSync('source.json', 'utf8')); + globalThis.fetch = async (url) => ({ ok: true, json: async () => + url.includes('raw.githubusercontent.com') ? source : url.includes('models.dev') ? {} : { data: [] } + }); + await import('./scripts/bundle-litellm.mjs'); + `], { cwd: dir, encoding: 'utf8', timeout: 10_000 }) + expect(run.status, run.stderr).toBe(0) + const snapshot = JSON.parse(readFileSync(join(dir, 'src/data/litellm-snapshot.json'), 'utf8')) + const fallback = JSON.parse(readFileSync(join(dir, 'src/data/pricing-fallback.json'), 'utf8')) + // The prefixed richer rows keep their own keys and never re-price the + // bare entries - not even by leaking a cache slot into them. + expect(snapshot['grok-3']).toEqual([3e-6, 15e-6, null, null, null, null]) + expect(snapshot['mistral-large-latest']).toEqual([8e-6, 24e-6, null, null, null, null]) + expect(snapshot['reseller/grok-3']).toEqual([1.25e-6, 2.5e-6, null, 0.1e-6, null, null]) + expect(snapshot['gateway/x-ai/grok-latest']).toEqual([2e-6, 4e-6, null, null, null, null]) + expect(snapshot['qwen/qwen3.5-plus-20260420']).toEqual([2e-6, 4e-6, null, null, null, null]) + // Fill adds a missing cache slot; it never overwrites a published one. + expect(snapshot['cache-model']).toEqual([1e-6, 2e-6, null, 0.1e-6, null, null]) + expect(snapshot['cache-own']).toEqual([1e-6, 2e-6, null, 0.2e-6, null, null]) + expect(snapshot['gpt-5.6']).toEqual([4e-6, 20e-6, null, null, null, { + threshold: 272_000, input: 8e-6, output: 30e-6, cacheWrite: null, cacheRead: null, + }]) + // Invalid tiers are filtered before the max, so the valid 128k tier wins. + expect(snapshot['tier-guard']).toEqual([1e-6, 2e-6, null, null, null, { + threshold: 128_000, input: 2e-6, output: 4e-6, cacheWrite: 0.5e-6, cacheRead: 0.05e-6, + }]) + // The bare tiered row keeps its tier and gains the prefixed row's + // cache-read slot (a structural tier compare would also hold here, but the + // guard must not depend on reference equality of slot 5). + expect(snapshot['direct-tier']).toEqual([5e-6, 25e-6, null, 3e-6, null, { + threshold: 272_000, input: 8e-6, output: 30e-6, cacheWrite: null, cacheRead: null, + }]) + // The base-richer alias (cache-write + cache-read, no tier) wins over the + // tier-bearing one; no tier is spliced onto its base rates. + expect(snapshot['alias-model']).toEqual([2e-6, 6e-6, 0.4e-6, 0.2e-6, null, null]) + // Exact-key-only carry: the fallback is exactly the previously-priced + // rows - the prefixed and dated variants did not count as their coverage. + expect(fallback).toEqual({ ...oldPrimary, ...oldFallback }) + } finally { + rmSync(dir, { recursive: true, force: true }) + } +}) diff --git a/tests/models.test.ts b/tests/models.test.ts index 4d47084a1..003772337 100644 --- a/tests/models.test.ts +++ b/tests/models.test.ts @@ -24,6 +24,8 @@ import { getFlatRateModelsConfigHash, parseLiteLLMEntry, unpricedModelHint, + cacheWriteCostPerToken, + tieredCostsFor, modelKeyMatches, } from '../src/models.js' import { getDailyCacheConfigHash } from '../src/usage-aggregator.js' @@ -128,9 +130,106 @@ describe('getModelCosts', () => { expect(calculateCost('gpt-5.6-codex-max', 1_000_000, 1_000_000, 0, 0, 0)).toBeGreaterThan(0) }) + describe('long-context tiers (#1076)', () => { + it('bundles the source long-context tiers with their real thresholds', () => { + // Straight from the regenerated snapshot: OpenAI's family tiers at 272k + // (NOT 128k - #1075 verified 128k fabricates +64% spend), Anthropic's at + // 200k, with each tier's own cache rates where the source publishes them. + const gpt = getModelCosts('gpt-5.6') + expect(gpt?.longContextTier).toEqual({ + thresholdTokens: 272_000, + inputCostPerToken: 8e-6, + outputCostPerToken: 3e-5, + cacheWriteCostPerToken: 1e-5, + cacheReadCostPerToken: 8e-7, + }) + const claude = getModelCosts('claude-sonnet-4-5') + expect(claude?.longContextTier?.thresholdTokens).toBe(200_000) + expect(claude?.longContextTier?.inputCostPerToken).toBe(6e-6) + // A model without a published tier resolves to no tier at all. + expect(getModelCosts('gpt-5.6-codex')?.longContextTier).toBeUndefined() + }) + + it('applies the tier to every token at and after the threshold, and only then', () => { + // prompt tokens = input + cached input; below the threshold the base + // rates apply, at it the tier's rates price the WHOLE request. + const below = calculateCost('gpt-5.6', 271_999, 0, 0, 0, 0, 'standard', 0, 'codex') + expect(below).toBeCloseTo(271_999 * 4e-6, 9) + const at = calculateCost('gpt-5.6', 272_000, 0, 0, 0, 0, 'standard', 0, 'codex') + expect(at).toBeCloseTo(272_000 * 8e-6, 9) + const above = calculateCost('gpt-5.6', 271_000, 0, 0, 1_000, 0, 'standard', 0, 'codex') + expect(above).toBeCloseTo(271_000 * 8e-6 + 1_000 * 8e-7, 9) + // The tier applies only where billing evidence exists for it: the same + // call through a provider not in TIERED_PRICING_PROVIDERS (or none, the + // default for every legacy caller) keeps the base rate. + const notEligible = calculateCost('gpt-5.6', 300_000, 0, 0, 0, 0) + expect(notEligible).toBeCloseTo(300_000 * 4e-6, 9) + const copilot = calculateCost('gpt-5.6', 300_000, 0, 0, 0, 0, 'standard', 0, 'copilot') + expect(copilot).toBeCloseTo(300_000 * 4e-6, 9) + }) + + it('keeps the base rate for slots the tier omits', () => { + // gpt-5.5's tier publishes cache read but no cache write: crossing the + // threshold must not invent a tier cache-write rate, nor drop the base. + // The base cache-write slot is implicit, so it bills at the input rate + // of whichever costs object is in effect — tiered here, per #1544's rule. + const base = getModelCosts('gpt-5.5')! + const tieredCosts = tieredCostsFor('gpt-5.5', base, 300_000, 'codex') + const tiered = calculateCost('gpt-5.5', 300_000, 0, 1_000, 0, 0, 'standard', 0, 'codex') + expect(tiered).toBeCloseTo(300_000 * tieredCosts.inputCostPerToken + 1_000 * cacheWriteCostPerToken('gpt-5.5', tieredCosts), 9) + expect(base.longContextTier!.cacheWriteCostPerToken).toBeUndefined() + }) + + it('still prices below-threshold requests exactly as before the extension', () => { + // Compat: the tier is inert below the threshold, so pre-extension + // pricing on any sub-threshold call is byte-for-byte unchanged. + expect(calculateCost('gpt-5.6', 100_000, 50_000, 1_000, 2_000, 0)) + .toBeCloseTo(100_000 * 4e-6 + 50_000 * 2e-5 + 1_000 * 5e-6 + 2_000 * 4e-7, 9) + }) + + it('an exact price override still beats the tier', () => { + setPriceOverrides({ 'gpt-5.6': { input: 2, output: 6 } }) + expect(calculateCost('gpt-5.6', 300_000, 0, 0, 0, 0, 'standard', 0, 'codex')).toBeCloseTo(300_000 * 2e-6, 9) + }) + + it('parses the tier from a live LiteLLM entry, plain context suffixes only', () => { + const costs = parseLiteLLMEntry({ + input_cost_per_token: 1e-6, + output_cost_per_token: 2e-5, + input_cost_per_token_above_272k_tokens: 5e-6, + output_cost_per_token_above_272k_tokens: 6e-5, + cache_read_input_token_cost_above_272k_tokens: 5e-7, + // Service-tier variants are not context thresholds and must be ignored. + input_cost_per_token_above_272k_priority_tokens: 9e-5, + input_cost_per_token_above_272k_flex_tokens: 8e-6, + } as never) + expect(costs?.longContextTier).toEqual({ + thresholdTokens: 272_000, + inputCostPerToken: 5e-6, + outputCostPerToken: 6e-5, + cacheReadCostPerToken: 5e-7, + }) + const none = parseLiteLLMEntry({ + input_cost_per_token: 1e-6, + output_cost_per_token: 2e-5, + input_cost_per_token_above_272k_priority_tokens: 9e-5, + } as never) + expect(none?.longContextTier).toBeUndefined() + }) + + it('old five-slot tuples still parse without a tier', () => { + // The compat path: bundles predating the sixth slot load unchanged. + const legacy = parseLiteLLMEntry({ input_cost_per_token: 1e-6, output_cost_per_token: 2e-5 })! + expect(legacy.longContextTier).toBeUndefined() + expect(legacy.inputCostPerToken).toBe(1e-6) + }) + }) + describe('grok-4.6 prompt tier', () => { it('uses the low tier below 200000 prompt tokens', () => { - expect(calculateCost('grok-4.6', 100_000, 10_000, 0, 99_999, 0)).toBeCloseTo(0.3099995, 12) + // Base input is 1.25e-6 since LiteLLM's 2026-09 reprice (was 2e-6); the + // tier rates above 200k are unchanged, so only this literal moved. + expect(calculateCost('grok-4.6', 100_000, 10_000, 0, 99_999, 0)).toBeCloseTo(0.2349995, 12) }) it('uses the high tier for every token at exactly 200000 prompt tokens', () => { diff --git a/tests/providers/codex.test.ts b/tests/providers/codex.test.ts index 8c72adf0a..cbc24fe17 100644 --- a/tests/providers/codex.test.ts +++ b/tests/providers/codex.test.ts @@ -1383,6 +1383,35 @@ describe('codex provider - forked session dedupe', () => { }) describe('codex auto-review pricing (#1047)', () => { + it('prices an auto-review whose prompt crosses gpt-5.5\'s above-272k tier at the tier (#1076)', async () => { + // End-to-end through the codex provider path: the gate is keyed on the + // provider string threaded from codex.ts, so a typo there would leave the + // call at base rates and fail this. 400k input puts the prompt well past + // 272,000 with no cache needed. + const filePath = await writeSession(tmpDir, '2026-04-14', 'rollout-auto-review-tier.jsonl', [ + sessionMeta({ session_id: 'sess-auto-tier', model: 'codex-auto-review' }), + userMessage('review the PR'), + tokenCount({ + timestamp: '2026-04-14T10:01:00Z', + last: { input: 400_000, output: 1_000 }, + total: { total: 401_000 }, + }), + ]) + const provider = createCodexProvider(tmpDir) + const calls: ParsedProviderCall[] = [] + for await (const call of provider.createSessionParser({ path: filePath, project: 'test', provider: 'codex' }, new Set()).parse()) { + calls.push(call) + } + expect(calls).toHaveLength(1) + // gpt-5.5 tier (bundled): input 1e-5, output 4.5e-5 - explicit arithmetic, + // not just self-consistency with calculateCost. + expect(calls[0]!.costUSD).toBeCloseTo(400_000 * 1e-5 + 1_000 * 4.5e-5, 12) + expect(calls[0]!.costUSD).toBe(calculateCost('gpt-5.5', 400_000, 1_000, 0, 0, 0, 'standard', 0, 'codex')) + // The same call without the codex provider stays at base rates (the + // refreshed bundle's gpt-5.5 base is 5e-6/3e-5) - the gate that keeps the + // real Copilot billing of tests/parser.test.ts (c4) intact. + expect(calculateCost('gpt-5.5', 400_000, 1_000, 0, 0, 0)).toBeCloseTo(400_000 * 5e-6 + 1_000 * 3e-5, 12) + }) it('parses auto-review as itself and prices it as GPT-5.5', async () => { const filePath = await writeSession(tmpDir, '2026-04-14', 'rollout-auto-review.jsonl', [ sessionMeta({ session_id: 'sess-auto', model: 'codex-auto-review' }), @@ -1400,7 +1429,7 @@ describe('codex auto-review pricing (#1047)', () => { } expect(calls).toHaveLength(1) expect(calls[0]!.model).toBe('codex-auto-review') - expect(calls[0]!.costUSD).toBe(calculateCost('gpt-5.5', 1_000_000, 1_000_000, 0, 0, 0)) + expect(calls[0]!.costUSD).toBe(calculateCost('gpt-5.5', 1_000_000, 1_000_000, 0, 0, 0, 'standard', 0, 'codex')) }) it('discards a warm v11 versioned $0 exact hit so unchanged rollouts reprice', async () => { @@ -1447,7 +1476,7 @@ describe('codex auto-review pricing (#1047)', () => { } expect(calls).toHaveLength(1) expect(calls[0]!.costUSD).toBeGreaterThan(0) - expect(calls[0]!.costUSD).toBe(calculateCost('gpt-5.5', 1_000_000, 1_000_000, 0, 0, 0)) + expect(calls[0]!.costUSD).toBe(calculateCost('gpt-5.5', 1_000_000, 1_000_000, 0, 0, 0, 'standard', 0, 'codex')) } finally { clearCodexMemCaches() if (prev === undefined) delete process.env['CODEBURN_CACHE_DIR']