Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 44 additions & 8 deletions core/llm/utils/calculateRequestCost.ts
Original file line number Diff line number Diff line change
Expand Up @@ -155,11 +155,18 @@ function calculateOpenAICost(
// Normalize model name
const normalizedModel = model.toLowerCase();

// Define pricing per million tokens (MTok) by model family prefix
const pricing: Record<string, { input: number; output: number }> = {
// Define pricing per million tokens (MTok) by model family prefix.
// `cachedInput` is set only for model families where OpenAI documents a
// prompt-cache rate. prompt_tokens from the API includes the cached portion,
// so the input cost is split: uncached tokens at the input rate and cached
// tokens at the cachedInput rate.
const pricing: Record<
string,
{ input: number; output: number; cachedInput?: number }
> = {
// GPT-4o models (most specific first)
"gpt-4o-mini": { input: 0.15, output: 0.6 },
"gpt-4o": { input: 2.5, output: 10 },
"gpt-4o-mini": { input: 0.15, output: 0.6, cachedInput: 0.075 },
"gpt-4o": { input: 2.5, output: 10, cachedInput: 1.25 },

// GPT-4 Turbo models
"gpt-4-turbo": { input: 10, output: 30 },
Expand Down Expand Up @@ -188,16 +195,45 @@ function calculateOpenAICost(
return null; // Unknown model
}

// Calculate costs
const inputCost = (usage.promptTokens / 1_000_000) * modelPricing.input;
// Split prompt tokens into uncached and cached portions. prompt_tokens from
// the OpenAI usage object already includes cached tokens; cached_tokens is
// reported under prompt_tokens_details.cached_tokens. cachedTokens is
// clamped to promptTokens to defend against malformed usage payloads that
// report more cached tokens than total prompt tokens. If the model has no
// documented cachedInput rate, the cached portion is not subtracted from
// the input cost — those prompt tokens are billed at the standard input
// rate (the API wouldn't have returned a cached_tokens value for them in
// the first place).
const reportedCachedTokens = usage.promptTokensDetails?.cachedTokens ?? 0;
const cachedTokens =
modelPricing.cachedInput !== undefined
? Math.min(reportedCachedTokens, usage.promptTokens)
: 0;
const uncachedPromptTokens = usage.promptTokens - cachedTokens;

const cachedInputCost =
cachedTokens > 0 && modelPricing.cachedInput !== undefined
? (cachedTokens / 1_000_000) * modelPricing.cachedInput
: 0;
const inputCost =
(uncachedPromptTokens / 1_000_000) * modelPricing.input + cachedInputCost;
const outputCost = (usage.completionTokens / 1_000_000) * modelPricing.output;

// Build breakdown components
const breakdownParts: string[] = [];

if (usage.promptTokens > 0) {
if (uncachedPromptTokens > 0) {
breakdownParts.push(
`Input: ${usage.promptTokens.toLocaleString()} tokens × $${modelPricing.input}/MTok = $${inputCost.toFixed(6)}`,
`Input: ${uncachedPromptTokens.toLocaleString()} tokens × $${modelPricing.input}/MTok = $${(
(uncachedPromptTokens / 1_000_000) *
modelPricing.input
).toFixed(6)}`,
);
}

if (cachedTokens > 0 && modelPricing.cachedInput !== undefined) {
breakdownParts.push(
`Cached Input: ${cachedTokens.toLocaleString()} tokens × $${modelPricing.cachedInput}/MTok = $${cachedInputCost.toFixed(6)}`,
);
}

Expand Down
66 changes: 66 additions & 0 deletions core/llm/utils/calculateRequestCost.vitest.ts
Original file line number Diff line number Diff line change
Expand Up @@ -163,6 +163,72 @@ describe("calculateRequestCost", () => {
description: "GPT-3.5 Turbo",
},

// OpenAI GPT-4o with prompt caching (cached_input is half of input)
{
provider: "openai",
model: "gpt-4o",
promptTokens: 1000,
completionTokens: 500,
cachedTokens: 700,
expectedCost: 0.006625,
description:
"GPT-4o with cached input tokens (uncached 300 + cached 700)",
},
{
provider: "openai",
model: "gpt-4o",
promptTokens: 1000,
completionTokens: 500,
cachedTokens: 1000,
expectedCost: 0.00625,
description: "GPT-4o fully cached input",
},
{
provider: "openai",
model: "gpt-4o",
promptTokens: 1000,
completionTokens: 500,
cachedTokens: 0,
expectedCost: 0.0075,
description: "GPT-4o with explicit zero cached tokens (no cache rows)",
},

// OpenAI GPT-4o-mini with prompt caching
{
provider: "openai",
model: "gpt-4o-mini",
promptTokens: 1000,
completionTokens: 200,
cachedTokens: 800,
expectedCost: 0.00021,
description:
"GPT-4o-mini with cached input tokens (uncached 200 + cached 800)",
},

// OpenAI GPT-4 (no documented cached rate) — cached_tokens present but ignored
{
provider: "openai",
model: "gpt-4",
promptTokens: 1000,
completionTokens: 100,
cachedTokens: 500,
expectedCost: 0.036,
description:
"GPT-4 ignores cachedTokens because no cachedInput rate is defined",
},

// Clamp cachedTokens > promptTokens (malformed payloads)
{
provider: "openai",
model: "gpt-4o",
promptTokens: 1000,
completionTokens: 0,
cachedTokens: 1500,
expectedCost: 0.00125,
description:
"GPT-4o clamps cachedTokens to promptTokens (all input treated as cached)",
},

// Edge cases
{
provider: "anthropic",
Expand Down
Loading