Nested combos (comboA lists comboB, comboC, …) now stay one slot each: the inner combo always runs as fallback to produce a single answer. Failed hops are no longer written to Details/usage, and streaming no longer inserts a 0-token placeholder row. - chat.js: comboStack cycle detection; nested combos forced to fallback; persistUsage="success-only" for combo hops - combo.js: discardResponse() cancels unused bodies (fusion timeout / fallback) so dropped streams fire onStreamComplete; getComboModelsFromData keeps nested names and honors enabled=false - requestDetail.js: tokensForDetail() canonicalizes Claude/Gemini usage; shouldPersistRequestDetail() skips streaming-start and non-success hops - streamingHandler.js: drop the 0-token streaming placeholder write - RequestDetailsTab.js: read Gemini/Claude token names; show "streaming" status in amber - tests: add combo-nested.test.js (13 cases) - gitignore: ignore local .vitest/ artifacts Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
417 lines
18 KiB
JavaScript
417 lines
18 KiB
JavaScript
import { FORMATS } from "../../translator/formats.js";
|
|
import { needsTranslation } from "../../translator/index.js";
|
|
import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
|
|
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
|
|
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
|
|
import { createErrorResult } from "../../utils/error.js";
|
|
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
|
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
|
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine, tokensForDetail, shouldPersistRequestDetail } from "./requestDetail.js";
|
|
import { saveRequestDetail } from "@/lib/usageDb.js";
|
|
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
|
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
|
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
|
|
|
function parseToolArguments(value) {
|
|
if (!value) return {};
|
|
if (typeof value === "object") return value;
|
|
try {
|
|
return JSON.parse(value);
|
|
} catch {
|
|
return {};
|
|
}
|
|
}
|
|
|
|
function openAICompletionToClaudeMessage(responseBody) {
|
|
if (!responseBody?.choices?.[0]) return responseBody;
|
|
const choice = responseBody.choices[0];
|
|
const message = choice.message || {};
|
|
const content = [];
|
|
|
|
const reasoning = message.reasoning_content || message.provider_specific_fields?.reasoning_content || "";
|
|
if (reasoning) {
|
|
content.push({ type: "thinking", thinking: reasoning });
|
|
}
|
|
if (typeof message.content === "string" && message.content.length > 0) {
|
|
content.push({ type: "text", text: message.content });
|
|
}
|
|
for (const toolCall of message.tool_calls || []) {
|
|
const fn = toolCall.function || {};
|
|
content.push({
|
|
type: "tool_use",
|
|
id: toolCall.id || `toolu_${Date.now()}_${content.length}`,
|
|
name: fn.name || toolCall.name || "",
|
|
input: parseToolArguments(fn.arguments || toolCall.arguments),
|
|
});
|
|
}
|
|
if (content.length === 0) content.push({ type: "text", text: "" });
|
|
|
|
const usage = responseBody.usage || {};
|
|
return {
|
|
id: String(responseBody.id || `msg_${Date.now()}`).replace(/^chatcmpl-/, ""),
|
|
type: "message",
|
|
role: "assistant",
|
|
model: responseBody.model || "unknown",
|
|
content,
|
|
stop_reason: fromOpenAIFinish(choice.finish_reason, FORMATS.CLAUDE),
|
|
stop_sequence: null,
|
|
usage: {
|
|
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
|
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
|
},
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Convert an OpenAI Chat Completions non-streaming response body into the
|
|
* OpenAI Responses API shape. Used when a Responses-format client (e.g. Codex)
|
|
* is routed to a Chat Completions upstream and `stream:false` — the streaming
|
|
* path already emits Responses events, but the JSON path returned a raw
|
|
* `chat.completion` body, so tool_calls were invisible to Responses clients.
|
|
*/
|
|
function extractCustomToolInput(argumentsValue) {
|
|
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
|
try {
|
|
const parsed = JSON.parse(argumentsText);
|
|
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
|
} catch { /* raw freeform input */ }
|
|
return argumentsText;
|
|
}
|
|
|
|
function openAICompletionToResponses(responseBody, customToolNames = null) {
|
|
const choice = responseBody?.choices?.[0];
|
|
if (!choice) return responseBody;
|
|
|
|
const message = choice.message || {};
|
|
const output = [];
|
|
|
|
// Reasoning → a reasoning item (summary text), mirroring the streaming path.
|
|
const reasoning = message.reasoning_content || message.reasoning;
|
|
if (typeof reasoning === "string" && reasoning.length > 0) {
|
|
output.push({
|
|
type: RESPONSES_ITEM.REASONING,
|
|
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
|
});
|
|
}
|
|
|
|
// Assistant text → a message item with output_text content.
|
|
const text = typeof message.content === "string" ? message.content : "";
|
|
if (text.length > 0) {
|
|
output.push({
|
|
type: RESPONSES_ITEM.MESSAGE,
|
|
role: ROLE.ASSISTANT,
|
|
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
|
});
|
|
}
|
|
|
|
// tool_calls → function_call/custom_tool_call items (Responses-native tool shape).
|
|
for (const tc of message.tool_calls || []) {
|
|
const fn = tc.function || {};
|
|
const custom = customToolNames?.has(fn.name);
|
|
output.push({
|
|
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
|
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
|
call_id: tc.id || "",
|
|
name: fn.name || "",
|
|
...(custom
|
|
? { input: extractCustomToolInput(fn.arguments) }
|
|
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
|
});
|
|
}
|
|
|
|
const usage = responseBody.usage || {};
|
|
const status = choice.finish_reason === "tool_calls" ? "completed" : (choice.finish_reason === "stop" ? "completed" : (choice.finish_reason || "completed"));
|
|
|
|
return {
|
|
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
|
object: "response",
|
|
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
|
model: responseBody.model || "unknown",
|
|
status,
|
|
background: false,
|
|
error: null,
|
|
output,
|
|
usage: {
|
|
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
|
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
|
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
|
},
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Translate non-streaming response body from provider format → OpenAI format.
|
|
*/
|
|
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames = null) {
|
|
if (targetFormat === sourceFormat) return responseBody;
|
|
// Provider responded in OpenAI Chat Completions shape but the client speaks
|
|
// Responses API — convert so tool_calls/text surface as Responses `output`.
|
|
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
|
return openAICompletionToResponses(responseBody, customToolNames);
|
|
}
|
|
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
|
return openAICompletionToClaudeMessage(responseBody);
|
|
}
|
|
if (targetFormat === FORMATS.OPENAI) return responseBody;
|
|
|
|
// Gemini / Antigravity
|
|
if (targetFormat === FORMATS.GEMINI || targetFormat === FORMATS.ANTIGRAVITY || targetFormat === FORMATS.GEMINI_CLI || targetFormat === FORMATS.VERTEX) {
|
|
const response = responseBody.response || responseBody;
|
|
if (!response?.candidates?.[0]) return responseBody;
|
|
|
|
const candidate = response.candidates[0];
|
|
const content = candidate.content;
|
|
const usage = response.usageMetadata || responseBody.usageMetadata;
|
|
let textContent = "", reasoningContent = "";
|
|
const toolCalls = [];
|
|
|
|
if (content?.parts) {
|
|
for (const part of content.parts) {
|
|
if (part.thought === true && part.text) reasoningContent += part.text;
|
|
else if (part.text !== undefined) textContent += part.text;
|
|
if (part.functionCall) {
|
|
toolCalls.push({
|
|
id: `call_${part.functionCall.name}_${Date.now()}_${toolCalls.length}`,
|
|
type: "function",
|
|
function: { name: part.functionCall.name, arguments: JSON.stringify(part.functionCall.args || {}) }
|
|
});
|
|
}
|
|
// Handle inline image data (from image generation models)
|
|
const inlineData = part.inlineData || part.inline_data;
|
|
if (inlineData?.data) {
|
|
const mimeType = inlineData.mimeType || inlineData.mime_type || "image/png";
|
|
textContent += `\n\n`;
|
|
}
|
|
}
|
|
}
|
|
|
|
const message = { role: "assistant" };
|
|
if (textContent) message.content = textContent;
|
|
if (reasoningContent) message.reasoning_content = reasoningContent;
|
|
if (toolCalls.length > 0) message.tool_calls = toolCalls;
|
|
if (!message.content && !message.tool_calls) message.content = "";
|
|
|
|
let finishReason = (candidate.finishReason || "stop").toLowerCase();
|
|
if (finishReason === "stop" && toolCalls.length > 0) finishReason = "tool_calls";
|
|
|
|
const result = {
|
|
id: `chatcmpl-${response.responseId || Date.now()}`,
|
|
object: "chat.completion",
|
|
created: Math.floor(new Date(response.createTime || Date.now()).getTime() / 1000),
|
|
model: response.modelVersion || "gemini",
|
|
choices: [{ index: 0, message, finish_reason: finishReason }]
|
|
};
|
|
|
|
if (usage) {
|
|
result.usage = {
|
|
prompt_tokens: (usage.promptTokenCount || 0) + (usage.thoughtsTokenCount || 0),
|
|
completion_tokens: usage.candidatesTokenCount || 0,
|
|
total_tokens: usage.totalTokenCount || 0
|
|
};
|
|
if (usage.thoughtsTokenCount > 0) {
|
|
result.usage.completion_tokens_details = { reasoning_tokens: usage.thoughtsTokenCount };
|
|
}
|
|
}
|
|
return result;
|
|
}
|
|
|
|
// Claude
|
|
if (targetFormat === FORMATS.CLAUDE) {
|
|
// Always translate a Claude-format body to OpenAI, even if `content` is
|
|
// missing/null (e.g. M3 with max_tokens:1 spends the budget on thinking
|
|
// and returns `content: null`). Returning the raw body would leave the
|
|
// OpenAI client without a `choices` array and surface as a UI test error.
|
|
// Early return if the response is already in OpenAI format (has choices array)
|
|
// or if it has content as a non-array value (likely a different non-Claude format).
|
|
// Some providers (e.g. xiaomi-tokenplan) return OpenAI-format responses even when
|
|
// the request was translated to Claude format — the targetFormat is Claude but the
|
|
// actual response is OpenAI-native and needs no further translation.
|
|
if (responseBody.choices || (responseBody.content && !Array.isArray(responseBody.content))) return responseBody;
|
|
|
|
let textContent = "", thinkingContent = "";
|
|
const toolCalls = [];
|
|
|
|
for (const block of (responseBody.content || [])) {
|
|
if (block.type === "text") {
|
|
// Strip markdown code block markers (e.g. kimi wraps JSON in ```json...```)
|
|
const raw = block.text ?? "";
|
|
const text = raw.replace(/^\s*```\s*json\s*\n?/i, "").replace(/\n?\s*```\s*$/i, "");
|
|
textContent += text;
|
|
} else if (block.type === "thinking") thinkingContent += block.thinking || "";
|
|
else if (block.type === "tool_use") {
|
|
toolCalls.push({ id: block.id, type: "function", function: { name: block.name, arguments: JSON.stringify(block.input || {}) } });
|
|
}
|
|
}
|
|
|
|
const message = { role: "assistant" };
|
|
if (textContent) message.content = textContent;
|
|
if (thinkingContent) message.reasoning_content = thinkingContent;
|
|
if (toolCalls.length > 0) message.tool_calls = toolCalls;
|
|
if (!message.content && !message.tool_calls) message.content = "";
|
|
|
|
let finishReason = responseBody.stop_reason || "stop";
|
|
if (finishReason === "end_turn") finishReason = "stop";
|
|
if (finishReason === "tool_use") finishReason = "tool_calls";
|
|
|
|
const result = {
|
|
id: `chatcmpl-${responseBody.id || Date.now()}`,
|
|
object: "chat.completion",
|
|
created: Math.floor(Date.now() / 1000),
|
|
model: responseBody.model || "claude",
|
|
choices: [{ index: 0, message, finish_reason: finishReason }]
|
|
};
|
|
|
|
if (responseBody.usage) {
|
|
result.usage = {
|
|
prompt_tokens: responseBody.usage.input_tokens || 0,
|
|
completion_tokens: responseBody.usage.output_tokens || 0,
|
|
total_tokens: (responseBody.usage.input_tokens || 0) + (responseBody.usage.output_tokens || 0)
|
|
};
|
|
}
|
|
return result;
|
|
}
|
|
|
|
// Ollama
|
|
if (targetFormat === FORMATS.OLLAMA) {
|
|
return ollamaBodyToOpenAI(responseBody);
|
|
}
|
|
|
|
return responseBody;
|
|
}
|
|
|
|
/**
|
|
* Handle non-streaming response from provider.
|
|
*/
|
|
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log, streamErrorPatterns, persistUsage = "all" }) {
|
|
trackDone();
|
|
const contentType = providerResponse.headers.get("content-type") || "";
|
|
let responseBody;
|
|
|
|
if (contentType.includes("text/event-stream")) {
|
|
const sseText = await providerResponse.text();
|
|
const parsed = parseSSEToOpenAIResponse(sseText, model);
|
|
if (!parsed) {
|
|
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
|
|
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, "Invalid SSE response for non-streaming request");
|
|
}
|
|
responseBody = parsed;
|
|
} else {
|
|
try {
|
|
responseBody = await providerResponse.json();
|
|
} catch (err) {
|
|
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
|
|
console.error(`[ChatCore] Failed to parse JSON from ${provider}:`, err.message);
|
|
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Invalid JSON response from ${provider}`);
|
|
}
|
|
}
|
|
|
|
reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody);
|
|
if (onRequestSuccess) {
|
|
Promise.resolve()
|
|
.then(onRequestSuccess)
|
|
.catch(err => {
|
|
console.error("[ChatCore] onRequestSuccess failed:", err?.message || err);
|
|
});
|
|
}
|
|
|
|
// Decloak tool_use names once on raw Claude body, before any translation (INPUT side)
|
|
responseBody = decloakToolNames(responseBody, toolNameMap);
|
|
|
|
// Config-driven in-stream error detection: the HTTP call succeeded but the
|
|
// assembled content signals an upstream failure — treat it as an error so
|
|
// account/combo fallback and FAILED logging kick in (AGENTS.md hook #3).
|
|
const matchedPattern = matchStreamErrorPatterns(
|
|
streamErrorPatterns?.[provider],
|
|
responseBody?.choices?.[0]?.message?.content || responseBody?.content || "",
|
|
);
|
|
if (matchedPattern) {
|
|
appendLog({ status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}` });
|
|
if (log?.errorLine) {
|
|
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms (in-stream)\n Stream error pattern matched: ${matchedPattern}`);
|
|
}
|
|
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, `Stream error pattern matched: ${matchedPattern}`);
|
|
}
|
|
|
|
const usage = extractUsageFromResponse(responseBody);
|
|
appendLog({ tokens: usage, status: "200 OK" });
|
|
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
|
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
|
|
|
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
|
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames)
|
|
: responseBody;
|
|
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
|
// Responses-format translation produces a `object:"response"` body with no
|
|
// `choices`; skip the Chat-Completions-specific post-processing below for it.
|
|
const isResponsesResponse = sourceFormat === FORMATS.OPENAI_RESPONSES && translatedResponse?.object === "response";
|
|
|
|
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
|
if (translatedResponse?.choices?.[0]) {
|
|
const choice = translatedResponse.choices[0];
|
|
const msg = choice.message;
|
|
const hasToolCalls = Array.isArray(msg?.tool_calls) && msg.tool_calls.length > 0;
|
|
if (hasToolCalls && choice.finish_reason !== "tool_calls") {
|
|
choice.finish_reason = "tool_calls";
|
|
}
|
|
}
|
|
|
|
// Ensure OpenAI-required fields
|
|
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
|
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
|
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
|
}
|
|
|
|
// Strip Azure-specific fields
|
|
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
|
delete translatedResponse.prompt_filter_results;
|
|
if (translatedResponse?.choices) {
|
|
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
|
}
|
|
}
|
|
|
|
if (translatedResponse?.usage) {
|
|
translatedResponse.usage = filterUsageForFormat(addBufferToUsage(translatedResponse.usage), sourceFormat);
|
|
}
|
|
|
|
// Strip reasoning_content only when content is non-empty.
|
|
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
|
// reasoning_content is the only useful output and must be preserved.
|
|
if (!isClaudeMessageResponse && !isResponsesResponse && translatedResponse?.choices) {
|
|
for (const choice of translatedResponse.choices) {
|
|
if (choice?.message?.reasoning_content && choice.message.content) {
|
|
delete choice.message.reasoning_content;
|
|
}
|
|
}
|
|
}
|
|
|
|
reqLogger.logConvertedResponse(translatedResponse);
|
|
|
|
const totalLatency = Date.now() - requestStartTime;
|
|
if (shouldPersistRequestDetail(persistUsage, "success")) {
|
|
saveRequestDetail(buildRequestDetail({
|
|
provider, model, connectionId, apiKey,
|
|
latency: { ttft: totalLatency, total: totalLatency },
|
|
tokens: tokensForDetail(usage),
|
|
request: extractRequestConfig(body, stream),
|
|
providerRequest: finalBody || translatedBody || null,
|
|
providerResponse: responseBody || null,
|
|
response: {
|
|
content: translatedResponse?.choices?.[0]?.message?.content || translatedResponse?.content || null,
|
|
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
|
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
|
},
|
|
pxpipe,
|
|
status: "success"
|
|
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
|
console.error("[RequestDetail] Failed to save:", err.message);
|
|
});
|
|
}
|
|
|
|
return {
|
|
success: true,
|
|
response: new Response(JSON.stringify(translatedResponse), {
|
|
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" }
|
|
})
|
|
};
|
|
}
|