From eb9b910d4e239abb694da02cddb58fae5065902e Mon Sep 17 00:00:00 2001 From: "shuwen.wu" Date: Fri, 25 Sep 2026 21:15:22 +0800 Subject: [PATCH 1/2] fix: correct OpenAI cached input cost calculation and fix Anthropic double-counting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - OpenAI (#13104): Previously charged all prompt tokens at full input rate; cached tokens were not priced at discount rate. Now subtracts cachedTokens from promptTokens and prices them at cachedInput rate (50% of standard input for all listed OpenAI models). - Anthropic: Fixed double-counting bug where cacheRead/cacheWrite costs were added ON TOP of full input cost instead of subtracting cached/cache-write tokens from uncached input. This caused overcharging by cache token count × full input rate. - Updated cost breakdown strings to show separate line items for uncached input, cached input/cache read, cache write, and output. - Added cachedInput pricing for all existing OpenAI model entries. --- core/llm/utils/calculateRequestCost.ts | 119 ++++++++++++------------- 1 file changed, 59 insertions(+), 60 deletions(-) diff --git a/core/llm/utils/calculateRequestCost.ts b/core/llm/utils/calculateRequestCost.ts index 794524d448d..676d74fb5ce 100644 --- a/core/llm/utils/calculateRequestCost.ts +++ b/core/llm/utils/calculateRequestCost.ts @@ -86,56 +86,52 @@ function calculateAnthropicCost( } if (!modelPricing) { - return null; // Unknown model + return null; } - // Calculate costs - const inputCost = (usage.promptTokens / 1_000_000) * modelPricing.input; + const cachedTokens = usage.promptTokensDetails?.cachedTokens ?? 0; + const cacheWriteTokens = usage.promptTokensDetails?.cacheWriteTokens ?? 0; + const uncachedInputTokens = Math.max( + 0, + usage.promptTokens - cachedTokens - cacheWriteTokens, + ); + + const uncachedInputCost = + (uncachedInputTokens / 1_000_000) * modelPricing.input; + const cacheWriteCost = + (cacheWriteTokens / 1_000_000) * modelPricing.cacheWrite; + const cacheReadCost = (cachedTokens / 1_000_000) * modelPricing.cacheRead; const outputCost = (usage.completionTokens / 1_000_000) * modelPricing.output; - // Build breakdown components + const totalCost = + uncachedInputCost + cacheWriteCost + cacheReadCost + outputCost; + const breakdownParts: string[] = []; - // Input tokens breakdown - if (usage.promptTokens > 0) { + if (uncachedInputTokens > 0) { breakdownParts.push( - `Input: ${usage.promptTokens.toLocaleString()} tokens × $${modelPricing.input}/MTok = $${inputCost.toFixed(6)}`, + `Input: ${uncachedInputTokens.toLocaleString()} tokens × $${modelPricing.input}/MTok = $${uncachedInputCost.toFixed(6)}`, ); } - // Output tokens breakdown - if (usage.completionTokens > 0) { + if (cacheWriteTokens > 0) { breakdownParts.push( - `Output: ${usage.completionTokens.toLocaleString()} tokens × $${modelPricing.output}/MTok = $${outputCost.toFixed(6)}`, + `Cache Write: ${cacheWriteTokens.toLocaleString()} tokens × $${modelPricing.cacheWrite}/MTok = $${cacheWriteCost.toFixed(6)}`, ); } - // Handle prompt caching costs if available - let cacheCost = 0; - if (usage.promptTokensDetails) { - const { cachedTokens, cacheWriteTokens } = usage.promptTokensDetails; - - if (cacheWriteTokens && cacheWriteTokens > 0) { - const cacheWriteCost = - (cacheWriteTokens / 1_000_000) * modelPricing.cacheWrite; - cacheCost += cacheWriteCost; - breakdownParts.push( - `Cache Write: ${cacheWriteTokens.toLocaleString()} tokens × $${modelPricing.cacheWrite}/MTok = $${cacheWriteCost.toFixed(6)}`, - ); - } - - if (cachedTokens && cachedTokens > 0) { - const cacheReadCost = (cachedTokens / 1_000_000) * modelPricing.cacheRead; - cacheCost += cacheReadCost; - breakdownParts.push( - `Cache Read: ${cachedTokens.toLocaleString()} tokens × $${modelPricing.cacheRead}/MTok = $${cacheReadCost.toFixed(6)}`, - ); - } + if (cachedTokens > 0) { + breakdownParts.push( + `Cache Read: ${cachedTokens.toLocaleString()} tokens × $${modelPricing.cacheRead}/MTok = $${cacheReadCost.toFixed(6)}`, + ); } - const totalCost = inputCost + outputCost + cacheCost; + if (usage.completionTokens > 0) { + breakdownParts.push( + `Output: ${usage.completionTokens.toLocaleString()} tokens × $${modelPricing.output}/MTok = $${outputCost.toFixed(6)}`, + ); + } - // Build final breakdown string let breakdown = `Model: ${model}\n`; breakdown += breakdownParts.join("\n"); if (breakdownParts.length > 1) { @@ -152,28 +148,21 @@ function calculateOpenAICost( model: string, usage: Usage, ): CostBreakdown | null { - // Normalize model name const normalizedModel = model.toLowerCase(); - // Define pricing per million tokens (MTok) by model family prefix - const pricing: Record = { - // GPT-4o models (most specific first) - "gpt-4o-mini": { input: 0.15, output: 0.6 }, - "gpt-4o": { input: 2.5, output: 10 }, - - // GPT-4 Turbo models - "gpt-4-turbo": { input: 10, output: 30 }, - - // GPT-3.5 Turbo models (most specific first) - "gpt-3.5-turbo-0125": { input: 0.5, output: 1.5 }, - "gpt-3.5-turbo-1106": { input: 1, output: 2 }, - "gpt-3.5-turbo": { input: 1.5, output: 2 }, - - // Base GPT-4 (fallback for other gpt-4 variants) - "gpt-4": { input: 30, output: 60 }, + const pricing: Record< + string, + { input: number; output: number; cachedInput: number } + > = { + "gpt-4o-mini": { input: 0.15, output: 0.6, cachedInput: 0.075 }, + "gpt-4o": { input: 2.5, output: 10, cachedInput: 1.25 }, + "gpt-4-turbo": { input: 10, output: 30, cachedInput: 5 }, + "gpt-3.5-turbo-0125": { input: 0.5, output: 1.5, cachedInput: 0.25 }, + "gpt-3.5-turbo-1106": { input: 1, output: 2, cachedInput: 0.5 }, + "gpt-3.5-turbo": { input: 1.5, output: 2, cachedInput: 0.75 }, + "gpt-4": { input: 30, output: 60, cachedInput: 15 }, }; - // Sort keys by length (longest first) to match most specific patterns first const sortedKeys = Object.keys(pricing).sort((a, b) => b.length - a.length); let modelPricing = null; @@ -185,19 +174,32 @@ function calculateOpenAICost( } if (!modelPricing) { - return null; // Unknown model + return null; } - // Calculate costs - const inputCost = (usage.promptTokens / 1_000_000) * modelPricing.input; + const cachedTokens = usage.promptTokensDetails?.cachedTokens ?? 0; + const uncachedInputTokens = Math.max(0, usage.promptTokens - cachedTokens); + + const uncachedInputCost = + (uncachedInputTokens / 1_000_000) * modelPricing.input; + const cachedInputCost = + (cachedTokens / 1_000_000) * modelPricing.cachedInput; const outputCost = (usage.completionTokens / 1_000_000) * modelPricing.output; - // Build breakdown components + const inputCost = uncachedInputCost + cachedInputCost; + const totalCost = inputCost + outputCost; + const breakdownParts: string[] = []; - if (usage.promptTokens > 0) { + if (uncachedInputTokens > 0) { breakdownParts.push( - `Input: ${usage.promptTokens.toLocaleString()} tokens × $${modelPricing.input}/MTok = $${inputCost.toFixed(6)}`, + `Input: ${uncachedInputTokens.toLocaleString()} tokens × $${modelPricing.input}/MTok = $${uncachedInputCost.toFixed(6)}`, + ); + } + + if (cachedTokens > 0) { + breakdownParts.push( + `Cached Input: ${cachedTokens.toLocaleString()} tokens × $${modelPricing.cachedInput}/MTok = $${cachedInputCost.toFixed(6)}`, ); } @@ -207,9 +209,6 @@ function calculateOpenAICost( ); } - const totalCost = inputCost + outputCost; - - // Build final breakdown string let breakdown = `Model: ${model}\n`; breakdown += breakdownParts.join("\n"); if (breakdownParts.length > 1) { From 0a25096cdbcba49bda866c55da11ad4dd155b910 Mon Sep 17 00:00:00 2001 From: dajiaohuang Date: Fri, 2 Oct 2026 15:52:21 +0800 Subject: [PATCH 2/2] fix: correct cache usage cost fixtures --- core/llm/utils/calculateRequestCost.ts | 3 +-- core/llm/utils/calculateRequestCost.vitest.ts | 4 ++-- 2 files changed, 3 insertions(+), 4 deletions(-) diff --git a/core/llm/utils/calculateRequestCost.ts b/core/llm/utils/calculateRequestCost.ts index 676d74fb5ce..60edbc3bd0f 100644 --- a/core/llm/utils/calculateRequestCost.ts +++ b/core/llm/utils/calculateRequestCost.ts @@ -182,8 +182,7 @@ function calculateOpenAICost( const uncachedInputCost = (uncachedInputTokens / 1_000_000) * modelPricing.input; - const cachedInputCost = - (cachedTokens / 1_000_000) * modelPricing.cachedInput; + const cachedInputCost = (cachedTokens / 1_000_000) * modelPricing.cachedInput; const outputCost = (usage.completionTokens / 1_000_000) * modelPricing.output; const inputCost = uncachedInputCost + cachedInputCost; diff --git a/core/llm/utils/calculateRequestCost.vitest.ts b/core/llm/utils/calculateRequestCost.vitest.ts index 32183ca1b49..8805da2c434 100644 --- a/core/llm/utils/calculateRequestCost.vitest.ts +++ b/core/llm/utils/calculateRequestCost.vitest.ts @@ -29,7 +29,7 @@ describe("calculateRequestCost", () => { { provider: "anthropic", model: "claude-3-5-sonnet-20241022", - promptTokens: 1000, + promptTokens: 3000, completionTokens: 500, cachedTokens: 2000, expectedCost: 0.0111, @@ -38,7 +38,7 @@ describe("calculateRequestCost", () => { { provider: "anthropic", model: "claude-3-5-sonnet-20241022", - promptTokens: 1000, + promptTokens: 1300, completionTokens: 500, cacheWriteTokens: 300, expectedCost: 0.011625,