feat(usage): track cached tokens + correct input/output/cache cost (#2209)
Normalize every provider to one cache-inclusive convention via canonicalizeUsage() before persist, and price cached + cache_creation as subsets of prompt_tokens in calculateCostFromTokens() to stop double-counting. usageRepo now delegates cost math to a single source. Surface Cached tokens/cost across dashboard (overview, tokens, cost, details). Merge Claude message_start cache with message_delta output so cache counts survive. Compatible LLM nodes now allow multiple API-key connections (key pool). Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -247,9 +247,24 @@ function mergeChunksToResponse(chunks, sourceFormat) {
|
||||
|
||||
if (messageStart?.message) {
|
||||
finalChunk = messageStart.message;
|
||||
// Merge usage if available
|
||||
if (messageDelta?.usage) {
|
||||
finalChunk.usage = messageDelta.usage;
|
||||
// message_start.usage has input + cache; message_delta.usage has the
|
||||
// final output_tokens. Merge so cache survives (delta omits it).
|
||||
const startUsage = messageStart.message.usage;
|
||||
const deltaUsage = messageDelta?.usage;
|
||||
if (startUsage || deltaUsage) {
|
||||
finalChunk.usage = {
|
||||
...(startUsage || {}),
|
||||
...(deltaUsage || {}),
|
||||
...(startUsage?.cache_read_input_tokens !== undefined
|
||||
? { cache_read_input_tokens: startUsage.cache_read_input_tokens }
|
||||
: {}),
|
||||
...(startUsage?.cache_creation_input_tokens !== undefined
|
||||
? { cache_creation_input_tokens: startUsage.cache_creation_input_tokens }
|
||||
: {}),
|
||||
...(startUsage?.input_tokens !== undefined
|
||||
? { input_tokens: startUsage.input_tokens }
|
||||
: {})
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user