feat(usage): track cached tokens + correct input/output/cache cost (#2209)

Normalize every provider to one cache-inclusive convention via
canonicalizeUsage() before persist, and price cached + cache_creation as
subsets of prompt_tokens in calculateCostFromTokens() to stop
double-counting. usageRepo now delegates cost math to a single source.
Surface Cached tokens/cost across dashboard (overview, tokens, cost,
details). Merge Claude message_start cache with message_delta output so
cache counts survive. Compatible LLM nodes now allow multiple API-key
connections (key pool).

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
hodtien
2026-07-03 15:07:37 +07:00
committed by decolua
parent 960f8a0379
commit 54e3245ace
17 changed files with 558 additions and 71 deletions

View File

@@ -247,9 +247,24 @@ function mergeChunksToResponse(chunks, sourceFormat) {
if (messageStart?.message) {
finalChunk = messageStart.message;
// Merge usage if available
if (messageDelta?.usage) {
finalChunk.usage = messageDelta.usage;
// message_start.usage has input + cache; message_delta.usage has the
// final output_tokens. Merge so cache survives (delta omits it).
const startUsage = messageStart.message.usage;
const deltaUsage = messageDelta?.usage;
if (startUsage || deltaUsage) {
finalChunk.usage = {
...(startUsage || {}),
...(deltaUsage || {}),
...(startUsage?.cache_read_input_tokens !== undefined
? { cache_read_input_tokens: startUsage.cache_read_input_tokens }
: {}),
...(startUsage?.cache_creation_input_tokens !== undefined
? { cache_creation_input_tokens: startUsage.cache_creation_input_tokens }
: {}),
...(startUsage?.input_tokens !== undefined
? { input_tokens: startUsage.input_tokens }
: {})
};
}
}
}