feat(usage): track cached tokens + correct input/output/cache cost (#2209)
Normalize every provider to one cache-inclusive convention via canonicalizeUsage() before persist, and price cached + cache_creation as subsets of prompt_tokens in calculateCostFromTokens() to stop double-counting. usageRepo now delegates cost math to a single source. Surface Cached tokens/cost across dashboard (overview, tokens, cost, details). Merge Claude message_start cache with message_delta output so cache counts survive. Compatible LLM nodes now allow multiple API-key connections (key pool). Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -39,7 +39,16 @@ const USAGE_EXTRACTORS = {
|
||||
},
|
||||
kiro(raw) {
|
||||
const input = n(raw.inputTokens), output = n(raw.outputTokens);
|
||||
return { promptTokens: input, completionTokens: output, totalTokens: input + output };
|
||||
// ponytail: Amazon Q (Kiro upstream) does not expose cache fields today,
|
||||
// but pass through any cache_read/cache_creation/cached_tokens if the
|
||||
// event shape grows them later so cost tracking keeps working without
|
||||
// a second pass.
|
||||
const cached = n(raw.cache_read_input_tokens) || n(raw.cachedTokens) || n(raw.cached_tokens);
|
||||
const cacheCreation = n(raw.cache_creation_input_tokens);
|
||||
const out = { promptTokens: input, completionTokens: output, totalTokens: input + output };
|
||||
if (cached > 0) out.cachedTokens = cached;
|
||||
if (cacheCreation > 0) out.cacheCreationTokens = cacheCreation;
|
||||
return out;
|
||||
},
|
||||
ollama(raw) {
|
||||
const input = n(raw.prompt_eval_count), output = n(raw.eval_count);
|
||||
|
||||
Reference in New Issue
Block a user