fix(qoder): report usage to all clients and stop inlining large attachments
- Coalesce Qoder's empty finish-in-delta frame with the later choices:[] usage frame so OpenAI and Claude clients receive prompt_tokens, completion_tokens and cache-hit tokens (the dashboard already saw them) - Upload inlined images through /api/v2/image/upload like qodercli, and stub oversized non-image files instead of stuffing 30MB+ data URIs into agent_chat_generation - Emit response.completed -> response.usage for chat-native upstreams so /v1/responses clients (Codex CLI, sub2api) no longer log 0/0/0 - Keep Claude message_delta.usage working when usage arrives without choices[0] - Escalate to the smallest advertised Qoder context tier (200K/400K/1M) when the estimated prompt no longer fits max_input_tokens - Pass apiKey for PAT connections and list hidden enable:false catalog keys from /v1/models
This commit is contained in:
@@ -3,6 +3,7 @@ import { FORMATS } from "../translator/formats.js";
|
||||
import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb.js";
|
||||
import { extractUsage, mergeUsage, hasValidUsage, estimateUsage, logUsage, addBufferToUsage, filterUsageForFormat, COLORS } from "./usageTracking.js";
|
||||
import { parseSSELine, hasValuableContent, fixInvalidId, formatSSE } from "./streamHelpers.js";
|
||||
import { toResponsesUsage } from "../translator/concerns/usage.js";
|
||||
import { getOpenAIResponsesEventName, isOpenAIResponsesTerminalEvent, formatIncompleteOpenAIResponsesStreamFailure } from "./responsesStreamHelpers.js";
|
||||
import { dbg, isDebugEnabled } from "./debugLog.js";
|
||||
|
||||
@@ -201,7 +202,8 @@ export function createSSEStream(options = {}) {
|
||||
|
||||
responsesTerminal = isOpenAIResponsesTerminalEvent(currentOpenAIResponsesEvent, parsed);
|
||||
|
||||
const isFinishChunk = parsed.choices?.[0]?.finish_reason;
|
||||
const isFinishChunk = parsed.choices?.[0]?.finish_reason
|
||||
|| parsed.choices?.[0]?.delta?.finish_reason;
|
||||
if (isFinishChunk && !hasValidUsage(parsed.usage)) {
|
||||
const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI);
|
||||
parsed.usage = filterUsageForFormat(estimated, FORMATS.OPENAI);
|
||||
@@ -365,6 +367,19 @@ export function createSSEStream(options = {}) {
|
||||
item.usage = filterUsageForFormat(buffered, sourceFormat);
|
||||
}
|
||||
|
||||
// Responses API clients (Codex, sub2api /v1/responses): usage lives on
|
||||
// response.completed → response.usage. Same buffer/estimate policy as above.
|
||||
const completedResponse = item.event === "response.completed" ? item.data?.response : null;
|
||||
if (completedResponse && typeof completedResponse === "object") {
|
||||
if (state.usage) {
|
||||
completedResponse.usage = toResponsesUsage(addBufferToUsage(state.usage)) ?? completedResponse.usage;
|
||||
} else if (!completedResponse.usage && totalContentLength > 0) {
|
||||
const estimated = estimateUsage(body, totalContentLength, FORMATS.OPENAI);
|
||||
completedResponse.usage = toResponsesUsage(estimated);
|
||||
state.usage = estimated;
|
||||
}
|
||||
}
|
||||
|
||||
const output = formatSSE(item, sourceFormat);
|
||||
reqLogger?.appendConvertedChunk?.(output);
|
||||
controller.enqueue(sharedEncoder.encode(output));
|
||||
|
||||
Reference in New Issue
Block a user