Merge remote-tracking branch 'origin/master' into gitea/feature/end
Resolved conflicts taking origin/master (v0.5.55) as canonical, with local features re-applied: - runtime log level (LOG_LEVEL env + dashboard Settings → Logging, applied immediately and persisted across restarts) - free/noAuth provider enable/disable toggle via providerStrategies.enabled - parallel model testing (Test All Models / Test Selected Keys)
This commit is contained in:
@@ -6,9 +6,10 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats } from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
function parseToolArguments(value) {
|
||||
if (!value) return {};
|
||||
@@ -60,11 +61,93 @@ function openAICompletionToClaudeMessage(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions non-streaming response body into the
|
||||
* OpenAI Responses API shape. Used when a Responses-format client (e.g. Codex)
|
||||
* is routed to a Chat Completions upstream and `stream:false` — the streaming
|
||||
* path already emits Responses events, but the JSON path returned a raw
|
||||
* `chat.completion` body, so tool_calls were invisible to Responses clients.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function openAICompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
// Reasoning → a reasoning item (summary text), mirroring the streaming path.
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
// Assistant text → a message item with output_text content.
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
// tool_calls → function_call/custom_tool_call items (Responses-native tool shape).
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
const status = choice.finish_reason === "tool_calls" ? "completed" : (choice.finish_reason === "stop" ? "completed" : (choice.finish_reason || "completed"));
|
||||
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status,
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate non-streaming response body from provider format → OpenAI format.
|
||||
*/
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames = null) {
|
||||
if (targetFormat === sourceFormat) return responseBody;
|
||||
// Provider responded in OpenAI Chat Completions shape but the client speaks
|
||||
// Responses API — convert so tool_calls/text surface as Responses `output`.
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return openAICompletionToResponses(responseBody, customToolNames);
|
||||
}
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
||||
return openAICompletionToClaudeMessage(responseBody);
|
||||
}
|
||||
@@ -198,7 +281,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
||||
/**
|
||||
* Handle non-streaming response from provider.
|
||||
*/
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, trackDone, appendLog }) {
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
trackDone();
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
let responseBody;
|
||||
@@ -235,12 +318,16 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
|
||||
const usage = extractUsageFromResponse(responseBody);
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames)
|
||||
: responseBody;
|
||||
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
||||
// Responses-format translation produces a `object:"response"` body with no
|
||||
// `choices`; skip the Chat-Completions-specific post-processing below for it.
|
||||
const isResponsesResponse = sourceFormat === FORMATS.OPENAI_RESPONSES && translatedResponse?.object === "response";
|
||||
|
||||
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
||||
if (translatedResponse?.choices?.[0]) {
|
||||
@@ -253,13 +340,13 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
|
||||
// Ensure OpenAI-required fields
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
}
|
||||
|
||||
// Strip Azure-specific fields
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
@@ -273,7 +360,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
if (!isClaudeMessageResponse && translatedResponse?.choices) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse && translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
@@ -296,6 +383,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
||||
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
||||
},
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save:", err.message);
|
||||
|
||||
@@ -44,13 +44,14 @@ export function extractUsageFromResponse(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
// Gemini format
|
||||
if (responseBody.usageMetadata) {
|
||||
// Gemini format. Antigravity / gemini-cli wrap the payload in { response: {...} }.
|
||||
const usageMetadata = responseBody.usageMetadata || responseBody.response?.usageMetadata;
|
||||
if (usageMetadata) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
|
||||
prompt_tokens: usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: usageMetadata.thoughtsTokenCount || 0
|
||||
};
|
||||
}
|
||||
|
||||
@@ -69,12 +70,31 @@ export function buildRequestDetail(base, overrides = {}) {
|
||||
providerRequest: base.providerRequest || null,
|
||||
providerResponse: base.providerResponse || null,
|
||||
response: base.response || {},
|
||||
pxpipe: base.pxpipe || undefined,
|
||||
status: base.status || "success",
|
||||
...overrides
|
||||
};
|
||||
}
|
||||
|
||||
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE" }) {
|
||||
// Build the "done" summary: duration, ttft, in/out tokens with cache breakdown
|
||||
export function formatDoneLine({ usage, latency }) {
|
||||
const u = usage || {};
|
||||
const inTok = u.prompt_tokens ?? u.input_tokens ?? 0;
|
||||
const outTok = u.completion_tokens ?? u.output_tokens ?? 0;
|
||||
const cacheRead = u.cache_read_input_tokens ?? u.cached_tokens ?? u.prompt_tokens_details?.cached_tokens ?? 0;
|
||||
const cacheCreate = u.cache_creation_input_tokens ?? 0;
|
||||
let inStr = `IN ${inTok}`;
|
||||
if (cacheRead || cacheCreate) {
|
||||
const parts = [];
|
||||
if (cacheRead) parts.push(`↻${cacheRead}`);
|
||||
if (cacheCreate) parts.push(`+${cacheCreate}`);
|
||||
inStr += ` (CACHE ${parts.join(" ")})`;
|
||||
}
|
||||
const ttftStr = latency?.ttft ? ` · TTFT ${latency.ttft}ms` : "";
|
||||
return `DONE ${latency?.total ?? 0}ms${ttftStr} · ${inStr} · OUT ${outTok}`;
|
||||
}
|
||||
|
||||
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE", silent = false }) {
|
||||
if (!tokens || typeof tokens !== "object") return;
|
||||
|
||||
const inTokens = tokens.input_tokens ?? tokens.prompt_tokens ?? 0;
|
||||
@@ -82,35 +102,17 @@ export function saveUsageStats({ provider, model, tokens, connectionId, apiKey,
|
||||
|
||||
if (inTokens === 0 && outTokens === 0) return;
|
||||
|
||||
// Extract cache/reasoning tokens (unified from different formats)
|
||||
const cacheRead = tokens.cache_read_input_tokens || tokens.cached_tokens || tokens.prompt_tokens_details?.cached_tokens || 0;
|
||||
const cacheCreation = tokens.cache_creation_input_tokens || 0;
|
||||
const reasoning = tokens.reasoning_tokens || 0;
|
||||
if (!silent) {
|
||||
const time = new Date().toLocaleTimeString("en-US", { hour12: false, hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
||||
const accountSuffix = connectionId ? ` | account=${connectionId.slice(0, 8)}...` : "";
|
||||
console.log(`${COLORS.green}[${time}] 📊 [${label}] ${provider.toUpperCase()} | in=${inTokens} | out=${outTokens}${accountSuffix}${COLORS.reset}`);
|
||||
}
|
||||
|
||||
const time = new Date().toLocaleTimeString("en-US", { hour12: false, hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
||||
const accountSuffix = connectionId ? ` | account=${connectionId.slice(0, 8)}...` : "";
|
||||
|
||||
let msg = `${COLORS.green}[${time}] 📊 [${label}] ${provider?.toUpperCase() || "UNKNOWN"} | in=${inTokens} | out=${outTokens}${accountSuffix}`;
|
||||
if (tokens.estimated) msg += ` ${COLORS.yellow}(estimated)${COLORS.reset}`;
|
||||
if (cacheRead) msg += ` | cache_read=${cacheRead}`;
|
||||
if (cacheCreation) msg += ` | cache_create=${cacheCreation}`;
|
||||
if (reasoning) msg += ` | reasoning=${reasoning}`;
|
||||
msg += `${COLORS.reset}`;
|
||||
console.log(msg);
|
||||
|
||||
<<<<<<< HEAD
|
||||
// Normalize to OpenAI token shape for storage (include all token types)
|
||||
const normalized = {
|
||||
=======
|
||||
// Canonicalize to one storage convention (prompt_tokens cache-inclusive) so
|
||||
// cached/cache-creation tokens survive to cost calc + stats. See canonicalizeUsage.
|
||||
const normalized = canonicalizeUsage(tokens) || {
|
||||
>>>>>>> 7f436e2792be4fa5a4d1c4d6b8e9bc85eaaa6a3d
|
||||
prompt_tokens: tokens.prompt_tokens ?? tokens.input_tokens ?? 0,
|
||||
completion_tokens: tokens.completion_tokens ?? tokens.output_tokens ?? 0,
|
||||
cache_read_input_tokens: cacheRead,
|
||||
cache_creation_input_tokens: cacheCreation,
|
||||
reasoning_tokens: reasoning,
|
||||
completion_tokens: tokens.completion_tokens ?? tokens.output_tokens ?? 0
|
||||
};
|
||||
|
||||
saveRequestUsage({
|
||||
|
||||
@@ -3,7 +3,8 @@ import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats } from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
|
||||
const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
@@ -34,21 +35,97 @@ function pickAssistantMessageForChatCompletion(output) {
|
||||
return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions JSON body into the Responses API shape.
|
||||
* Inlined here (not imported from nonStreamingHandler.js) to avoid a circular
|
||||
* import. Mirrors openAICompletionToResponses in nonStreamingHandler.js.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function chatCompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse OpenAI-style SSE text into a single chat completion JSON.
|
||||
* Used when provider forces streaming but client wants non-streaming.
|
||||
*/
|
||||
export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
const chunks = [];
|
||||
let streamError = null;
|
||||
|
||||
for (const line of String(rawSSE || "").split("\n")) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed.startsWith("data:")) continue;
|
||||
const payload = trimmed.slice(5).trim();
|
||||
if (!payload || payload === "[DONE]") continue;
|
||||
try { chunks.push(JSON.parse(payload)); } catch { /* ignore malformed lines */ }
|
||||
try {
|
||||
const chunk = JSON.parse(payload);
|
||||
if (chunk?.error) streamError = chunk.error;
|
||||
else chunks.push(chunk);
|
||||
} catch { /* ignore malformed lines */ }
|
||||
}
|
||||
|
||||
if (streamError) return { error: streamError };
|
||||
if (chunks.length === 0) return null;
|
||||
|
||||
const first = chunks[0];
|
||||
@@ -102,7 +179,7 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
* Handle case: provider forced streaming but client wants JSON.
|
||||
* Supports both Codex/Responses API SSE and standard Chat Completions SSE.
|
||||
*/
|
||||
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, trackDone, appendLog }) {
|
||||
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, targetFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, customToolNames, trackDone, appendLog, reqTag, log }) {
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
const isSSE = contentType.includes("text/event-stream") || (contentType === "" && isResponsesProvider(provider));
|
||||
if (!isSSE) return null; // not handled here
|
||||
@@ -116,7 +193,10 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
};
|
||||
|
||||
// Codex/Responses API SSE path
|
||||
const isCodexResponsesApi = isResponsesProvider(provider) || sourceFormat === FORMATS.OPENAI_RESPONSES;
|
||||
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
|
||||
// not the client format: a Responses-API client behind a chat-native forced-streaming
|
||||
// provider still receives chat SSE chunks, which must go through the standard path.
|
||||
const isCodexResponsesApi = isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
if (isCodexResponsesApi) {
|
||||
try {
|
||||
const jsonResponse = await convertResponsesStreamToJson(providerResponse.body);
|
||||
@@ -124,15 +204,21 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
|
||||
const usage = jsonResponse.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
// Same cache-inclusive total for the recorded detail, so the DB and the
|
||||
// client-facing usage can never disagree.
|
||||
const inTokensForLog = (usage.input_tokens || 0)
|
||||
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0)
|
||||
+ (usage.cache_creation_input_tokens || 0);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(jsonResponse.output);
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: { prompt_tokens: usage.input_tokens || 0, completion_tokens: usage.output_tokens || 0 },
|
||||
tokens: { prompt_tokens: inTokensForLog, completion_tokens: usage.output_tokens || 0 },
|
||||
response: { content: textContent, thinking: null, finish_reason: jsonResponse.status || "unknown" },
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {});
|
||||
@@ -142,9 +228,21 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
return { success: true, response: new Response(JSON.stringify(jsonResponse), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
}
|
||||
|
||||
// Build client-format response
|
||||
const inTokens = usage.input_tokens || 0;
|
||||
// Build client-format response.
|
||||
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
|
||||
// only input+output under-reports prompt_tokens — measured: 2012 reported
|
||||
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache
|
||||
// counters in, and keep them visible in prompt_tokens_details so a client can
|
||||
// tell a cache hit from a small prompt.
|
||||
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens || 0;
|
||||
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
|
||||
const outTokens = usage.output_tokens || 0;
|
||||
const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
|
||||
? { prompt_tokens_details: {
|
||||
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
|
||||
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
|
||||
: {};
|
||||
let finalResp;
|
||||
|
||||
// Extract tool calls from Responses API output (function_call items)
|
||||
@@ -179,7 +277,7 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
created: jsonResponse.created_at || Math.floor(Date.now() / 1000),
|
||||
model: jsonResponse.model || model,
|
||||
choices: [{ index: 0, message, finish_reason: finishReason }],
|
||||
usage: { prompt_tokens: inTokens, completion_tokens: outTokens, total_tokens: inTokens + outTokens }
|
||||
usage: { prompt_tokens: inTokens, completion_tokens: outTokens, total_tokens: inTokens + outTokens, ...cacheDetails }
|
||||
};
|
||||
}
|
||||
|
||||
@@ -195,12 +293,19 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
const sseText = await providerResponse.text();
|
||||
const parsed = parseSSEToOpenAIResponse(sseText, model);
|
||||
if (!parsed) return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Invalid SSE response for non-streaming request");
|
||||
if (parsed.error) {
|
||||
return createErrorResult(
|
||||
HTTP_STATUS.BAD_GATEWAY,
|
||||
parsed.error.message || "Upstream SSE stream failed"
|
||||
);
|
||||
}
|
||||
|
||||
if (onRequestSuccess) await onRequestSuccess();
|
||||
|
||||
const usage = parsed.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
@@ -215,6 +320,15 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(() => {});
|
||||
|
||||
// Re-attach usage explicitly. This handler already HAS the correct usage — it is
|
||||
// the same object written to the usage DB, and for a cached Claude request that DB
|
||||
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
|
||||
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
|
||||
// on both sides). Whatever drops it between assembly and serialisation, the client
|
||||
// must not be left unable to account for its own token spend: a caller cannot tell
|
||||
// a 90%-cached request from a cheap one without this.
|
||||
if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
|
||||
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
@@ -227,7 +341,17 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
}
|
||||
}
|
||||
|
||||
return { success: true, response: new Response(JSON.stringify(parsed), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
// A Responses-format client (e.g. Codex) forced this provider to stream,
|
||||
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
|
||||
// body; convert it to the Responses `output` shape so tool_calls are not
|
||||
// lost on the non-streaming return path. Inlined (not imported from
|
||||
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
|
||||
// already imports parseSSEToOpenAIResponse from this module.
|
||||
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
|
||||
? chatCompletionToResponses(parsed, customToolNames)
|
||||
: parsed;
|
||||
|
||||
return { success: true, response: new Response(JSON.stringify(finalBody), { headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" } }) };
|
||||
} catch (err) {
|
||||
console.error("[ChatCore] Chat Completions SSE→JSON failed:", err);
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Failed to convert streaming response to JSON");
|
||||
|
||||
@@ -5,7 +5,7 @@ import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildRequestDetail, extractRequestConfig } from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
|
||||
@@ -22,7 +22,7 @@ const CODEX_SOURCE_TO_TARGET = {
|
||||
/**
|
||||
* Determine which SSE transform stream to use based on provider/format.
|
||||
*/
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey }) {
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) {
|
||||
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
|
||||
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
|
||||
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
@@ -30,11 +30,11 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
|
||||
if (needsCodexTranslation) {
|
||||
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey);
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
}
|
||||
|
||||
if (needsTranslation(targetFormat, sourceFormat)) {
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey);
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
}
|
||||
|
||||
return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey);
|
||||
@@ -43,7 +43,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId }) {
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
@@ -67,7 +67,8 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
const shortMsg = sanitizedTitle
|
||||
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
if (log?.errorLine) log.errorLine(reqTag, "✗", `BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`);
|
||||
else console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
@@ -78,7 +79,7 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey });
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey });
|
||||
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
@@ -94,6 +95,7 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: "[Streaming - raw response not captured]",
|
||||
response: { content: "[Streaming in progress...]", thinking: null, type: "streaming" },
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save streaming request:", err.message);
|
||||
@@ -108,7 +110,7 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
/**
|
||||
* Build onStreamComplete callback for streaming usage tracking.
|
||||
*/
|
||||
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest }) {
|
||||
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest, pxpipe, reqTag, log }) {
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
|
||||
const onStreamComplete = (contentObj, usage, ttftAt) => {
|
||||
@@ -127,10 +129,15 @@ export function buildOnStreamComplete({ provider, model, connectionId, apiKey, r
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: safeContent,
|
||||
response: { content: safeContent, thinking: safeThinking, type: "streaming" },
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to update streaming content:", err.message);
|
||||
});
|
||||
|
||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, label: "STREAM USAGE", silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency }));
|
||||
};
|
||||
|
||||
return { onStreamComplete, streamDetailId };
|
||||
|
||||
Reference in New Issue
Block a user