fix(qoder): report usage to all clients and stop inlining large attachments
- Coalesce Qoder's empty finish-in-delta frame with the later choices:[] usage frame so OpenAI and Claude clients receive prompt_tokens, completion_tokens and cache-hit tokens (the dashboard already saw them) - Upload inlined images through /api/v2/image/upload like qodercli, and stub oversized non-image files instead of stuffing 30MB+ data URIs into agent_chat_generation - Emit response.completed -> response.usage for chat-native upstreams so /v1/responses clients (Codex CLI, sub2api) no longer log 0/0/0 - Keep Claude message_delta.usage working when usage arrives without choices[0] - Escalate to the smallest advertised Qoder context tier (200K/400K/1M) when the estimated prompt no longer fits max_input_tokens - Pass apiKey for PAT connections and list hidden enable:false catalog keys from /v1/models
This commit is contained in:
@@ -5,7 +5,7 @@
|
||||
import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { buildChunk } from "../concerns/chunk.js";
|
||||
import { buildUsage } from "../concerns/usage.js";
|
||||
import { buildUsage, toResponsesUsage } from "../concerns/usage.js";
|
||||
import { fallbackToolCallId } from "../concerns/toolCall.js";
|
||||
import { reasoningDelta, extractReasoningText } from "../concerns/reasoning.js";
|
||||
import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } from "../schema/index.js";
|
||||
@@ -18,7 +18,13 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
||||
if (!chunk) {
|
||||
return flushEvents(state);
|
||||
}
|
||||
|
||||
|
||||
// Usage riding on the finish chunk (include_usage style, e.g. coalesced Qoder frames):
|
||||
// remember it so response.completed can report tokens even outside stream.js.
|
||||
if (chunk.usage && typeof chunk.usage === "object" && !state.usage) {
|
||||
state.usage = chunk.usage;
|
||||
}
|
||||
|
||||
if (!chunk.choices?.length) return [];
|
||||
|
||||
const events = [];
|
||||
@@ -368,17 +374,19 @@ function closeToolCall(state, emit, idx) {
|
||||
function sendCompleted(state, emit) {
|
||||
if (!state.completedSent) {
|
||||
state.completedSent = true;
|
||||
emit("response.completed", {
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id: state.responseId,
|
||||
object: "response",
|
||||
created_at: state.created,
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null
|
||||
}
|
||||
});
|
||||
const response = {
|
||||
id: state.responseId,
|
||||
object: "response",
|
||||
created_at: state.created,
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null
|
||||
};
|
||||
// Carry provider usage (recorded by stream.js or from the finish chunk itself) in the
|
||||
// Responses shape; proxies such as sub2api/Codex read tokens only from here.
|
||||
const usage = toResponsesUsage(state.usage);
|
||||
if (usage) response.usage = usage;
|
||||
emit("response.completed", { type: "response.completed", response });
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -67,48 +67,50 @@ function stopTextBlock(state, results) {
|
||||
state.textBlockStarted = false;
|
||||
}
|
||||
|
||||
function recordOpenAIUsage(chunk, state) {
|
||||
if (!chunk?.usage || typeof chunk.usage !== "object") return;
|
||||
|
||||
const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0;
|
||||
const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0;
|
||||
|
||||
// Extract cache tokens from prompt_tokens_details
|
||||
const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens;
|
||||
const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens;
|
||||
const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0;
|
||||
const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0;
|
||||
|
||||
// input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens
|
||||
// Because OpenAI's prompt_tokens includes all prompt-side tokens
|
||||
const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens;
|
||||
|
||||
state.usage = {
|
||||
input_tokens: inputTokens,
|
||||
output_tokens: outputTokens
|
||||
};
|
||||
|
||||
if (cacheReadTokens > 0) {
|
||||
state.usage.cache_read_input_tokens = cacheReadTokens;
|
||||
}
|
||||
if (cacheCreateTokens > 0) {
|
||||
state.usage.cache_creation_input_tokens = cacheCreateTokens;
|
||||
}
|
||||
}
|
||||
|
||||
// Convert OpenAI stream chunk to Claude format
|
||||
export function openaiToClaudeResponse(chunk, state) {
|
||||
if (!chunk || !chunk.choices?.[0]) return null;
|
||||
if (!chunk) return null;
|
||||
|
||||
// Track usage from OpenAI chunk if available
|
||||
if (chunk.usage && typeof chunk.usage === "object") {
|
||||
recordOpenAIUsage(chunk, state);
|
||||
}
|
||||
|
||||
if (!chunk.choices?.[0]) return null;
|
||||
|
||||
const results = [];
|
||||
const choice = chunk.choices[0];
|
||||
const delta = choice.delta;
|
||||
|
||||
// Track usage from OpenAI chunk if available
|
||||
if (chunk.usage && typeof chunk.usage === "object") {
|
||||
const promptTokens = typeof chunk.usage.prompt_tokens === "number" ? chunk.usage.prompt_tokens : 0;
|
||||
const outputTokens = typeof chunk.usage.completion_tokens === "number" ? chunk.usage.completion_tokens : 0;
|
||||
|
||||
// Extract cache tokens from prompt_tokens_details
|
||||
const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens;
|
||||
const cacheCreationTokens = chunk.usage.prompt_tokens_details?.cache_creation_tokens;
|
||||
const cacheReadTokens = typeof cachedTokens === "number" ? cachedTokens : 0;
|
||||
const cacheCreateTokens = typeof cacheCreationTokens === "number" ? cacheCreationTokens : 0;
|
||||
|
||||
// input_tokens = prompt_tokens - cached_tokens - cache_creation_tokens
|
||||
// Because OpenAI's prompt_tokens includes all prompt-side tokens
|
||||
const inputTokens = promptTokens - cacheReadTokens - cacheCreateTokens;
|
||||
|
||||
state.usage = {
|
||||
input_tokens: inputTokens,
|
||||
output_tokens: outputTokens
|
||||
};
|
||||
|
||||
// Add cache_read_input_tokens if present
|
||||
if (cacheReadTokens > 0) {
|
||||
state.usage.cache_read_input_tokens = cacheReadTokens;
|
||||
}
|
||||
|
||||
// Add cache_creation_input_tokens if present
|
||||
if (cacheCreateTokens > 0) {
|
||||
state.usage.cache_creation_input_tokens = cacheCreateTokens;
|
||||
}
|
||||
|
||||
// Note: completion_tokens_details.reasoning_tokens is already included in output_tokens
|
||||
// No need to add separately as Claude expects total output_tokens
|
||||
}
|
||||
|
||||
// First chunk - ALWAYS send message_start first
|
||||
if (!state.messageStartSent) {
|
||||
state.messageStartSent = true;
|
||||
@@ -221,8 +223,9 @@ export function openaiToClaudeResponse(chunk, state) {
|
||||
}
|
||||
}
|
||||
|
||||
// Finish
|
||||
if (choice.finish_reason) {
|
||||
// Finish (OpenAI puts this on the choice; Qoder often puts it on delta)
|
||||
const finishReason = choice.finish_reason || delta?.finish_reason;
|
||||
if (finishReason) {
|
||||
stopThinkingBlock(state, results);
|
||||
stopTextBlock(state, results);
|
||||
|
||||
@@ -244,13 +247,13 @@ export function openaiToClaudeResponse(chunk, state) {
|
||||
}
|
||||
|
||||
// Mark finish for later usage injection in stream.js
|
||||
state.finishReason = choice.finish_reason;
|
||||
state.finishReason = finishReason;
|
||||
|
||||
// Use tracked usage (will be estimated in stream.js if not valid)
|
||||
const finalUsage = state.usage || { input_tokens: 0, output_tokens: 0 };
|
||||
results.push({
|
||||
type: "message_delta",
|
||||
delta: { stop_reason: convertFinishReason(choice.finish_reason) },
|
||||
delta: { stop_reason: convertFinishReason(finishReason) },
|
||||
usage: finalUsage
|
||||
});
|
||||
results.push({ type: "message_stop" });
|
||||
|
||||
Reference in New Issue
Block a user