feat: Enhance usage tracking across response handlers
This commit is contained in:
@@ -12,7 +12,6 @@ export function adjustMaxTokens(body) {
|
||||
// Tool calls with large content (like writing files) need more tokens
|
||||
if (body.tools && Array.isArray(body.tools) && body.tools.length > 0) {
|
||||
if (maxTokens < DEFAULT_MIN_TOKENS) {
|
||||
console.log(`[AUTO-ADJUST] max_tokens: ${maxTokens} → ${DEFAULT_MIN_TOKENS} (tool calling detected)`);
|
||||
maxTokens = DEFAULT_MIN_TOKENS;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -92,9 +92,34 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
}
|
||||
|
||||
case "message_delta": {
|
||||
// Extract usage from message_delta event (Claude native format)
|
||||
// Normalize to OpenAI format (prompt_tokens/completion_tokens) for consistent logging
|
||||
if (chunk.usage && typeof chunk.usage === "object") {
|
||||
const inputTokens = typeof chunk.usage.input_tokens === "number" ? chunk.usage.input_tokens : 0;
|
||||
const outputTokens = typeof chunk.usage.output_tokens === "number" ? chunk.usage.output_tokens : 0;
|
||||
const cacheReadTokens = typeof chunk.usage.cache_read_input_tokens === "number" ? chunk.usage.cache_read_input_tokens : 0;
|
||||
const cacheCreationTokens = typeof chunk.usage.cache_creation_input_tokens === "number" ? chunk.usage.cache_creation_input_tokens : 0;
|
||||
|
||||
// Use OpenAI format keys for consistent logging in stream.js
|
||||
state.usage = {
|
||||
prompt_tokens: inputTokens,
|
||||
completion_tokens: outputTokens,
|
||||
input_tokens: inputTokens,
|
||||
output_tokens: outputTokens
|
||||
};
|
||||
|
||||
// Store cache tokens if present
|
||||
if (cacheReadTokens > 0) {
|
||||
state.usage.cache_read_input_tokens = cacheReadTokens;
|
||||
}
|
||||
if (cacheCreationTokens > 0) {
|
||||
state.usage.cache_creation_input_tokens = cacheCreationTokens;
|
||||
}
|
||||
}
|
||||
|
||||
if (chunk.delta?.stop_reason) {
|
||||
state.finishReason = convertStopReason(chunk.delta.stop_reason);
|
||||
results.push({
|
||||
const finalChunk = {
|
||||
id: `chatcmpl-${state.messageId}`,
|
||||
object: "chat.completion.chunk",
|
||||
created: Math.floor(Date.now() / 1000),
|
||||
@@ -104,7 +129,41 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
delta: {},
|
||||
finish_reason: state.finishReason
|
||||
}]
|
||||
});
|
||||
};
|
||||
|
||||
// Include usage in final chunk if available
|
||||
if (state.usage && typeof state.usage === "object") {
|
||||
const inputTokens = state.usage.input_tokens || 0;
|
||||
const outputTokens = state.usage.output_tokens || 0;
|
||||
const cachedTokens = state.usage.cache_read_input_tokens || 0;
|
||||
const cacheCreationTokens = state.usage.cache_creation_input_tokens || 0;
|
||||
|
||||
// prompt_tokens = input_tokens + cache_read + cache_creation (all prompt-side tokens)
|
||||
// completion_tokens = output_tokens
|
||||
// total_tokens = prompt_tokens + completion_tokens
|
||||
const promptTokens = inputTokens + cachedTokens + cacheCreationTokens;
|
||||
const completionTokens = outputTokens;
|
||||
const totalTokens = promptTokens + completionTokens;
|
||||
|
||||
finalChunk.usage = {
|
||||
prompt_tokens: promptTokens,
|
||||
completion_tokens: completionTokens,
|
||||
total_tokens: totalTokens
|
||||
};
|
||||
|
||||
// Add prompt_tokens_details if cached tokens exist
|
||||
if (cachedTokens > 0 || cacheCreationTokens > 0) {
|
||||
finalChunk.usage.prompt_tokens_details = {};
|
||||
if (cachedTokens > 0) {
|
||||
finalChunk.usage.prompt_tokens_details.cached_tokens = cachedTokens;
|
||||
}
|
||||
if (cacheCreationTokens > 0) {
|
||||
finalChunk.usage.prompt_tokens_details.cache_creation_tokens = cacheCreationTokens;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
results.push(finalChunk);
|
||||
state.finishReasonSent = true;
|
||||
}
|
||||
break;
|
||||
|
||||
@@ -160,14 +160,56 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
}
|
||||
}
|
||||
|
||||
// Finish reason
|
||||
// Usage metadata - extract before finish reason so we can include it
|
||||
const usageMeta = response.usageMetadata || chunk.usageMetadata;
|
||||
if (usageMeta && typeof usageMeta === "object") {
|
||||
const cachedTokens = typeof usageMeta.cachedContentTokenCount === "number" ? usageMeta.cachedContentTokenCount : 0;
|
||||
const promptTokenCountRaw = typeof usageMeta.promptTokenCount === "number" ? usageMeta.promptTokenCount : 0;
|
||||
const thoughtsTokens = typeof usageMeta.thoughtsTokenCount === "number" ? usageMeta.thoughtsTokenCount : 0;
|
||||
let candidatesTokens = typeof usageMeta.candidatesTokenCount === "number" ? usageMeta.candidatesTokenCount : 0;
|
||||
const totalTokens = typeof usageMeta.totalTokenCount === "number" ? usageMeta.totalTokenCount : 0;
|
||||
|
||||
// prompt_tokens = promptTokenCount (includes cached tokens, matching claude-to-openai.js behavior)
|
||||
const promptTokens = promptTokenCountRaw;
|
||||
|
||||
// Fallback calculation if candidatesTokenCount is 0 but totalTokenCount exists
|
||||
if (candidatesTokens === 0 && totalTokens > 0) {
|
||||
candidatesTokens = totalTokens - promptTokenCountRaw - thoughtsTokens;
|
||||
if (candidatesTokens < 0) candidatesTokens = 0;
|
||||
}
|
||||
|
||||
// completion_tokens = candidatesTokenCount + thoughtsTokenCount (match Go code)
|
||||
const completionTokens = candidatesTokens + thoughtsTokens;
|
||||
|
||||
state.usage = {
|
||||
prompt_tokens: promptTokens,
|
||||
completion_tokens: completionTokens,
|
||||
total_tokens: totalTokens
|
||||
};
|
||||
|
||||
// Add prompt_tokens_details if cached tokens exist
|
||||
if (cachedTokens > 0) {
|
||||
state.usage.prompt_tokens_details = {
|
||||
cached_tokens: cachedTokens
|
||||
};
|
||||
}
|
||||
|
||||
// Add completion_tokens_details if reasoning tokens exist
|
||||
if (thoughtsTokens > 0) {
|
||||
state.usage.completion_tokens_details = {
|
||||
reasoning_tokens: thoughtsTokens
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Finish reason - include usage in final chunk
|
||||
if (candidate.finishReason) {
|
||||
let finishReason = candidate.finishReason.toLowerCase();
|
||||
if (finishReason === "stop" && state.toolCalls.size > 0) {
|
||||
finishReason = "tool_calls";
|
||||
}
|
||||
|
||||
results.push({
|
||||
const finalChunk = {
|
||||
id: `chatcmpl-${state.messageId}`,
|
||||
object: "chat.completion.chunk",
|
||||
created: Math.floor(Date.now() / 1000),
|
||||
@@ -177,24 +219,15 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
delta: {},
|
||||
finish_reason: finishReason
|
||||
}]
|
||||
});
|
||||
state.finishReason = finishReason;
|
||||
}
|
||||
|
||||
// Usage metadata
|
||||
const usage = response.usageMetadata || chunk.usageMetadata;
|
||||
if (usage && typeof usage === 'object') {
|
||||
const promptTokens = (usage.promptTokenCount || 0) + (usage.thoughtsTokenCount || 0);
|
||||
state.usage = {
|
||||
prompt_tokens: promptTokens,
|
||||
completion_tokens: usage.candidatesTokenCount || 0,
|
||||
total_tokens: usage.totalTokenCount || 0
|
||||
};
|
||||
if (usage.thoughtsTokenCount > 0) {
|
||||
state.usage.completion_tokens_details = {
|
||||
reasoning_tokens: usage.thoughtsTokenCount
|
||||
};
|
||||
|
||||
// Include usage in final chunk for downstream translators
|
||||
if (state.usage) {
|
||||
finalChunk.usage = state.usage;
|
||||
}
|
||||
|
||||
results.push(finalChunk);
|
||||
state.finishReason = finishReason;
|
||||
}
|
||||
|
||||
return results.length > 0 ? results : null;
|
||||
|
||||
@@ -17,9 +17,6 @@ export function convertKiroToOpenAI(chunk, state) {
|
||||
if (chunk.object === "chat.completion.chunk" && chunk.choices) {
|
||||
return chunk;
|
||||
}
|
||||
|
||||
|
||||
console.log("chunk", chunk);
|
||||
|
||||
// Handle string chunk (raw SSE data)
|
||||
let data = chunk;
|
||||
@@ -161,6 +158,11 @@ export function convertKiroToOpenAI(chunk, state) {
|
||||
}]
|
||||
};
|
||||
|
||||
// Include usage in final chunk if available
|
||||
if (state.usage && typeof state.usage === "object") {
|
||||
openaiChunk.usage = state.usage;
|
||||
}
|
||||
|
||||
return openaiChunk;
|
||||
}
|
||||
|
||||
|
||||
@@ -474,9 +474,38 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
|
||||
|
||||
// Response completed
|
||||
if (eventType === "response.completed") {
|
||||
// Extract usage from response.completed event
|
||||
const responseUsage = data.response?.usage;
|
||||
if (responseUsage && typeof responseUsage === "object") {
|
||||
const inputTokens = responseUsage.input_tokens || responseUsage.prompt_tokens || 0;
|
||||
const outputTokens = responseUsage.output_tokens || responseUsage.completion_tokens || 0;
|
||||
const cacheReadTokens = responseUsage.cache_read_input_tokens || 0;
|
||||
const cacheCreationTokens = responseUsage.cache_creation_input_tokens || 0;
|
||||
|
||||
// prompt_tokens = input_tokens + cache_read + cache_creation (all prompt-side tokens)
|
||||
const promptTokens = inputTokens + cacheReadTokens + cacheCreationTokens;
|
||||
|
||||
state.usage = {
|
||||
prompt_tokens: promptTokens,
|
||||
completion_tokens: outputTokens,
|
||||
total_tokens: promptTokens + outputTokens
|
||||
};
|
||||
|
||||
// Add prompt_tokens_details if cache tokens exist
|
||||
if (cacheReadTokens > 0 || cacheCreationTokens > 0) {
|
||||
state.usage.prompt_tokens_details = {};
|
||||
if (cacheReadTokens > 0) {
|
||||
state.usage.prompt_tokens_details.cached_tokens = cacheReadTokens;
|
||||
}
|
||||
if (cacheCreationTokens > 0) {
|
||||
state.usage.prompt_tokens_details.cache_creation_tokens = cacheCreationTokens;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!state.finishReasonSent) {
|
||||
state.finishReasonSent = true;
|
||||
return {
|
||||
const finalChunk = {
|
||||
id: state.chatId,
|
||||
object: "chat.completion.chunk",
|
||||
created: state.created,
|
||||
@@ -487,6 +516,13 @@ export function openaiResponsesToOpenAIResponse(chunk, state) {
|
||||
finish_reason: "stop"
|
||||
}]
|
||||
};
|
||||
|
||||
// Include usage in final chunk if available
|
||||
if (state.usage && typeof state.usage === "object") {
|
||||
finalChunk.usage = state.usage;
|
||||
}
|
||||
|
||||
return finalChunk;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -33,6 +33,40 @@ export function openaiToClaudeResponse(chunk, state) {
|
||||
const choice = chunk.choices[0];
|
||||
const delta = choice.delta;
|
||||
|
||||
// Track usage from OpenAI chunk if available
|
||||
if (chunk.usage && typeof chunk.usage === "object") {
|
||||
const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0;
|
||||
const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0;
|
||||
|
||||
// Extract cache tokens from prompt_tokens_details
|
||||
const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens;
|
||||
const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens;
|
||||
const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0;
|
||||
const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0;
|
||||
|
||||
// input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens
|
||||
// Because OpenAI's prompt_tokens includes all prompt-side tokens
|
||||
const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens;
|
||||
|
||||
state.usage = {
|
||||
input_tokens: inputTokens,
|
||||
output_tokens: outputTokens
|
||||
};
|
||||
|
||||
// Add cache_read_input_tokens if present
|
||||
if (cacheReadTokens > 0) {
|
||||
state.usage.cache_read_input_tokens = cacheReadTokens;
|
||||
}
|
||||
|
||||
// Add cache_creation_input_tokens if present
|
||||
if (cacheCreateTokens > 0) {
|
||||
state.usage.cache_creation_input_tokens = cacheCreateTokens;
|
||||
}
|
||||
|
||||
// Note: completion_tokens_details.reasoning_tokens is already included in output_tokens
|
||||
// No need to add separately as Claude expects total output_tokens
|
||||
}
|
||||
|
||||
// First chunk - ALWAYS send message_start first
|
||||
if (!state.messageStartSent) {
|
||||
state.messageStartSent = true;
|
||||
@@ -158,10 +192,12 @@ export function openaiToClaudeResponse(chunk, state) {
|
||||
});
|
||||
}
|
||||
|
||||
// Use tracked usage or default to 0
|
||||
const finalUsage = state.usage || { input_tokens: 0, output_tokens: 0 };
|
||||
results.push({
|
||||
type: "message_delta",
|
||||
delta: { stop_reason: convertFinishReason(choice.finish_reason) },
|
||||
usage: { output_tokens: 0 }
|
||||
usage: finalUsage
|
||||
});
|
||||
results.push({ type: "message_stop" });
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user