end
This commit is contained in:
@@ -326,8 +326,8 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
}
|
||||
|
||||
// Streaming response
|
||||
const { onStreamComplete } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, streamController, onStreamComplete });
|
||||
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId });
|
||||
}
|
||||
|
||||
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { needsTranslation } from "../../translator/index.js";
|
||||
import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
|
||||
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
|
||||
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
@@ -9,11 +10,65 @@ import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, sav
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
|
||||
function parseToolArguments(value) {
|
||||
if (!value) return {};
|
||||
if (typeof value === "object") return value;
|
||||
try {
|
||||
return JSON.parse(value);
|
||||
} catch {
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
function openAICompletionToClaudeMessage(responseBody) {
|
||||
if (!responseBody?.choices?.[0]) return responseBody;
|
||||
const choice = responseBody.choices[0];
|
||||
const message = choice.message || {};
|
||||
const content = [];
|
||||
|
||||
const reasoning = message.reasoning_content || message.provider_specific_fields?.reasoning_content || "";
|
||||
if (reasoning) {
|
||||
content.push({ type: "thinking", thinking: reasoning });
|
||||
}
|
||||
if (typeof message.content === "string" && message.content.length > 0) {
|
||||
content.push({ type: "text", text: message.content });
|
||||
}
|
||||
for (const toolCall of message.tool_calls || []) {
|
||||
const fn = toolCall.function || {};
|
||||
content.push({
|
||||
type: "tool_use",
|
||||
id: toolCall.id || `toolu_${Date.now()}_${content.length}`,
|
||||
name: fn.name || toolCall.name || "",
|
||||
input: parseToolArguments(fn.arguments || toolCall.arguments),
|
||||
});
|
||||
}
|
||||
if (content.length === 0) content.push({ type: "text", text: "" });
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
return {
|
||||
id: String(responseBody.id || `msg_${Date.now()}`).replace(/^chatcmpl-/, ""),
|
||||
type: "message",
|
||||
role: "assistant",
|
||||
model: responseBody.model || "unknown",
|
||||
content,
|
||||
stop_reason: fromOpenAIFinish(choice.finish_reason, FORMATS.CLAUDE),
|
||||
stop_sequence: null,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate non-streaming response body from provider format → OpenAI format.
|
||||
*/
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
|
||||
if (targetFormat === sourceFormat || targetFormat === FORMATS.OPENAI) return responseBody;
|
||||
if (targetFormat === sourceFormat) return responseBody;
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
||||
return openAICompletionToClaudeMessage(responseBody);
|
||||
}
|
||||
if (targetFormat === FORMATS.OPENAI) return responseBody;
|
||||
|
||||
// Gemini / Antigravity
|
||||
if (targetFormat === FORMATS.GEMINI || targetFormat === FORMATS.ANTIGRAVITY || targetFormat === FORMATS.GEMINI_CLI || targetFormat === FORMATS.VERTEX) {
|
||||
@@ -185,6 +240,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
: responseBody;
|
||||
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
||||
|
||||
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
||||
if (translatedResponse?.choices?.[0]) {
|
||||
@@ -197,13 +253,17 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
|
||||
// Ensure OpenAI-required fields
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
}
|
||||
|
||||
// Strip Azure-specific fields
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
if (!isClaudeMessageResponse) {
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
}
|
||||
}
|
||||
|
||||
if (translatedResponse?.usage) {
|
||||
@@ -213,7 +273,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
if (translatedResponse?.choices) {
|
||||
if (!isClaudeMessageResponse && translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { saveRequestUsage, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { COLORS } from "../../utils/stream.js";
|
||||
import { canonicalizeUsage } from "../../utils/usageTracking.js";
|
||||
|
||||
const OPTIONAL_PARAMS = [
|
||||
"temperature", "top_p", "top_k",
|
||||
@@ -48,7 +49,8 @@ export function extractUsageFromResponse(responseBody) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount
|
||||
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
|
||||
};
|
||||
}
|
||||
|
||||
@@ -96,8 +98,14 @@ export function saveUsageStats({ provider, model, tokens, connectionId, apiKey,
|
||||
msg += `${COLORS.reset}`;
|
||||
console.log(msg);
|
||||
|
||||
<<<<<<< HEAD
|
||||
// Normalize to OpenAI token shape for storage (include all token types)
|
||||
const normalized = {
|
||||
=======
|
||||
// Canonicalize to one storage convention (prompt_tokens cache-inclusive) so
|
||||
// cached/cache-creation tokens survive to cost calc + stats. See canonicalizeUsage.
|
||||
const normalized = canonicalizeUsage(tokens) || {
|
||||
>>>>>>> 7f436e2792be4fa5a4d1c4d6b8e9bc85eaaa6a3d
|
||||
prompt_tokens: tokens.prompt_tokens ?? tokens.input_tokens ?? 0,
|
||||
completion_tokens: tokens.completion_tokens ?? tokens.output_tokens ?? 0,
|
||||
cache_read_input_tokens: cacheRead,
|
||||
|
||||
@@ -43,7 +43,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete }) {
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
@@ -52,12 +52,30 @@ export function handleStreamingResponse({ providerResponse, provider, model, sou
|
||||
});
|
||||
}
|
||||
|
||||
// Warn when upstream returns unexpected Content-Type for a streaming response.
|
||||
// This often means the provider returned an HTML error page or plain-text error
|
||||
// that the SSE transform stream would forward as garbage to the client.
|
||||
// When upstream returns HTML/text instead of SSE (e.g. Cloudflare 5xx error
|
||||
// page), piping it through the SSE transform stream causes Next.js
|
||||
// "failed to pipe response" and crashes the chat router. Read the body,
|
||||
// pull a short human-readable message from the <title>, sanitize it, and
|
||||
// return a clean JSON error instead. The message is stripped of HTML tags
|
||||
// and clamped so untrusted upstream text never reaches the client verbatim
|
||||
// (the UI may render error.message as HTML).
|
||||
const upstreamContentType = (providerResponse.headers.get('content-type') || '').toLowerCase();
|
||||
if (upstreamContentType && !upstreamContentType.includes('text/event-stream') && !upstreamContentType.includes('application/json')) {
|
||||
console.warn('[STREAM] ' + provider + ' | ' + model + ' | unexpected Content-Type: ' + upstreamContentType);
|
||||
const bodyText = await providerResponse.text().catch(() => '');
|
||||
const titleMatch = bodyText.match(/<title>([^<]+)<\/title>/i);
|
||||
const sanitizedTitle = (titleMatch?.[1] || '').replace(/<[^>]*>/g, '').replace(/[\r\n]+/g, ' ').trim().slice(0, 160);
|
||||
const shortMsg = sanitizedTitle
|
||||
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
response: new Response(JSON.stringify({ error: { message: `[${status}]: ${shortMsg}` } }), {
|
||||
status,
|
||||
headers: { 'Content-Type': 'application/json', 'Access-Control-Allow-Origin': '*' },
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey });
|
||||
@@ -68,7 +86,6 @@ export function handleStreamingResponse({ providerResponse, provider, model, sou
|
||||
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
|
||||
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
provider, model, connectionId,
|
||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||
|
||||
Reference in New Issue
Block a user