feat(usage): track cached tokens + correct input/output/cache cost (#2209)

Normalize every provider to one cache-inclusive convention via
canonicalizeUsage() before persist, and price cached + cache_creation as
subsets of prompt_tokens in calculateCostFromTokens() to stop
double-counting. usageRepo now delegates cost math to a single source.
Surface Cached tokens/cost across dashboard (overview, tokens, cost,
details). Merge Claude message_start cache with message_delta output so
cache counts survive. Compatible LLM nodes now allow multiple API-key
connections (key pool).

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
hodtien
2026-07-03 15:07:37 +07:00
committed by decolua
parent 960f8a0379
commit 54e3245ace
17 changed files with 558 additions and 71 deletions

View File

@@ -279,7 +279,10 @@ export function calculateCostFromTokens(tokens, pricing) {
const inputTokens = tokens.prompt_tokens || tokens.input_tokens || 0;
const cachedTokens = tokens.cached_tokens || tokens.cache_read_input_tokens || 0;
const nonCachedInput = Math.max(0, inputTokens - cachedTokens);
const cacheCreationTokens = tokens.cache_creation_input_tokens || 0;
// prompt_tokens is cache-inclusive (see canonicalizeUsage): cached + cache_creation
// are subsets, so subtract both to avoid charging them at the full input rate.
const nonCachedInput = Math.max(0, inputTokens - cachedTokens - cacheCreationTokens);
cost += nonCachedInput * (pricing.input / 1000000);
@@ -295,7 +298,6 @@ export function calculateCostFromTokens(tokens, pricing) {
cost += reasoningTokens * ((pricing.reasoning || pricing.output) / 1000000);
}
const cacheCreationTokens = tokens.cache_creation_input_tokens || 0;
if (cacheCreationTokens > 0) {
cost += cacheCreationTokens * ((pricing.cache_creation || pricing.input) / 1000000);
}