fix(combo): keep nested combos as single units and stop 0-token detail rows
Nested combos (comboA lists comboB, comboC, …) now stay one slot each: the inner combo always runs as fallback to produce a single answer. Failed hops are no longer written to Details/usage, and streaming no longer inserts a 0-token placeholder row. - chat.js: comboStack cycle detection; nested combos forced to fallback; persistUsage="success-only" for combo hops - combo.js: discardResponse() cancels unused bodies (fusion timeout / fallback) so dropped streams fire onStreamComplete; getComboModelsFromData keeps nested names and honors enabled=false - requestDetail.js: tokensForDetail() canonicalizes Claude/Gemini usage; shouldPersistRequestDetail() skips streaming-start and non-success hops - streamingHandler.js: drop the 0-token streaming placeholder write - RequestDetailsTab.js: read Gemini/Claude token names; show "streaming" status in amber - tests: add combo-nested.test.js (13 cases) - gitignore: ignore local .vitest/ artifacts Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
This commit is contained in:
2
.gitignore
vendored
2
.gitignore
vendored
@@ -95,3 +95,5 @@ graphify-out/*
|
|||||||
|
|
||||||
# Pi subagent run artifacts
|
# Pi subagent run artifacts
|
||||||
.pi-subagents/
|
.pi-subagents/
|
||||||
|
# Vitest local artifacts
|
||||||
|
.vitest/
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
- **Providers**: wire the dead "Fetch Models" button on compatible nodes to the live upstream catalog, de-duplicating against already-added models
|
- **Providers**: wire the dead "Fetch Models" button on compatible nodes to the live upstream catalog, de-duplicating against already-added models
|
||||||
|
|
||||||
## Fixes
|
## Fixes
|
||||||
|
- **Usage**: nested combos (comboA lists comboB, comboC, …) stay one slot each — the inner combo always runs as fallback to produce a single answer, failed hops are not written to Details/usage, and streaming no longer inserts a 0-token placeholder. One user message against a nested fallback combo is one request row with real tokens; fusion of nested combos is N panel slots + judge, not every nested leaf
|
||||||
- **Models**: persist per-model capability assertions for custom and compatible providers and honor them everywhere — unsupported media is stripped on the chat path, `/v1/models` and `/api/models` report what the user asserted, and thinking translation follows it (asserting `reasoning:false` now actually strips thinking fields, `reasoning:true` emits them)
|
- **Models**: persist per-model capability assertions for custom and compatible providers and honor them everywhere — unsupported media is stripped on the chat path, `/v1/models` and `/api/models` report what the user asserted, and thinking translation follows it (asserting `reasoning:false` now actually strips thinking fields, `reasoning:true` emits them)
|
||||||
- **Models**: partial capability edits merge instead of overwriting, so toggling vision off no longer erases a stored reasoning assertion
|
- **Models**: partial capability edits merge instead of overwriting, so toggling vision off no longer erases a stored reasoning assertion
|
||||||
- **Capabilities**: keep server-injected readers (synced catalog, user-asserted capabilities) in process-wide state — Next.js compiles startup and each API route into separate bundles with their own module instances, so a boot-time install was invisible to every request handler and the models.dev catalog contributed nothing to upstream requests since 0532f00d
|
- **Capabilities**: keep server-injected readers (synced catalog, user-asserted capabilities) in process-wide state — Next.js compiles startup and each API route into separate bundles with their own module instances, so a boot-time install was invisible to every request handler and the models.dev catalog contributed nothing to upstream requests since 0532f00d
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ import { handleBypassRequest } from "../utils/bypassHandler.js";
|
|||||||
import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js";
|
import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js";
|
||||||
import { getExecutor } from "../executors/index.js";
|
import { getExecutor } from "../executors/index.js";
|
||||||
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
||||||
import { buildRequestDetail, extractRequestConfig } from "./chatCore/requestDetail.js";
|
import { buildRequestDetail, extractRequestConfig, shouldPersistRequestDetail } from "./chatCore/requestDetail.js";
|
||||||
import { handleForcedSSEToJson } from "./chatCore/sseToJsonHandler.js";
|
import { handleForcedSSEToJson } from "./chatCore/sseToJsonHandler.js";
|
||||||
import { handleNonStreamingResponse } from "./chatCore/nonStreamingHandler.js";
|
import { handleNonStreamingResponse } from "./chatCore/nonStreamingHandler.js";
|
||||||
import { handleStreamingResponse, buildOnStreamComplete } from "./chatCore/streamingHandler.js";
|
import { handleStreamingResponse, buildOnStreamComplete } from "./chatCore/streamingHandler.js";
|
||||||
@@ -59,7 +59,7 @@ export function stripContinuityFields(body) {
|
|||||||
return body;
|
return body;
|
||||||
}
|
}
|
||||||
|
|
||||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, capsOverride = null, streamErrorPatterns = null }) {
|
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, capsOverride = null, streamErrorPatterns = null, persistUsage = "all" }) {
|
||||||
const { provider, model } = modelInfo;
|
const { provider, model } = modelInfo;
|
||||||
const requestStartTime = Date.now();
|
const requestStartTime = Date.now();
|
||||||
// Stable per-session color so all lines of one CLI conversation share a tag
|
// Stable per-session color so all lines of one CLI conversation share a tag
|
||||||
@@ -377,16 +377,18 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
|||||||
reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody);
|
reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody);
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
trackPendingRequest(model, provider, connectionId, false, true);
|
trackPendingRequest(model, provider, connectionId, false, true);
|
||||||
saveRequestDetail(buildRequestDetail({
|
if (shouldPersistRequestDetail(persistUsage, "error")) {
|
||||||
provider, model, connectionId,
|
saveRequestDetail(buildRequestDetail({
|
||||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
provider, model, connectionId,
|
||||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||||
request: extractRequestConfig(body, stream),
|
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||||
providerRequest: translatedBody || null,
|
request: extractRequestConfig(body, stream),
|
||||||
response: { error: error.message || String(error), status: error.name === "AbortError" ? 499 : 502, thinking: null },
|
providerRequest: translatedBody || null,
|
||||||
pxpipe: pxpipeSummary,
|
response: { error: error.message || String(error), status: error.name === "AbortError" ? 499 : 502, thinking: null },
|
||||||
status: "error"
|
pxpipe: pxpipeSummary,
|
||||||
})).catch(() => { });
|
status: "error"
|
||||||
|
})).catch(() => { });
|
||||||
|
}
|
||||||
|
|
||||||
if (error.name === "AbortError") {
|
if (error.name === "AbortError") {
|
||||||
streamController.handleError(error);
|
streamController.handleError(error);
|
||||||
@@ -451,16 +453,18 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
|||||||
if (!providerResponse.ok) {
|
if (!providerResponse.ok) {
|
||||||
trackPendingRequest(model, provider, connectionId, false, true);
|
trackPendingRequest(model, provider, connectionId, false, true);
|
||||||
const { statusCode, message, resetsAtMs } = await parseUpstreamError(providerResponse, executor);
|
const { statusCode, message, resetsAtMs } = await parseUpstreamError(providerResponse, executor);
|
||||||
saveRequestDetail(buildRequestDetail({
|
if (shouldPersistRequestDetail(persistUsage, "error")) {
|
||||||
provider, model, connectionId,
|
saveRequestDetail(buildRequestDetail({
|
||||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
provider, model, connectionId,
|
||||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
||||||
request: extractRequestConfig(body, stream),
|
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
||||||
providerRequest: finalBody || translatedBody || null,
|
request: extractRequestConfig(body, stream),
|
||||||
response: { error: message, status: statusCode, thinking: null },
|
providerRequest: finalBody || translatedBody || null,
|
||||||
pxpipe: pxpipeSummary,
|
response: { error: message, status: statusCode, thinking: null },
|
||||||
status: "error"
|
pxpipe: pxpipeSummary,
|
||||||
})).catch(() => { });
|
status: "error"
|
||||||
|
})).catch(() => { });
|
||||||
|
}
|
||||||
|
|
||||||
const errMsg = formatProviderError(new Error(message), provider, model, statusCode);
|
const errMsg = formatProviderError(new Error(message), provider, model, statusCode);
|
||||||
if (log?.errorLine) {
|
if (log?.errorLine) {
|
||||||
@@ -472,7 +476,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
|||||||
}
|
}
|
||||||
|
|
||||||
const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers
|
const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers
|
||||||
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log, streamErrorPatterns };
|
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log, streamErrorPatterns, persistUsage };
|
||||||
|
|
||||||
// Early-peek streaming responses for configured in-stream error patterns.
|
// Early-peek streaming responses for configured in-stream error patterns.
|
||||||
// Some upstreams fail INSIDE a 200 SSE stream; without this the failure is
|
// Some upstreams fail INSIDE a 200 SSE stream; without this the failure is
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin
|
|||||||
import { createErrorResult } from "../../utils/error.js";
|
import { createErrorResult } from "../../utils/error.js";
|
||||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine, tokensForDetail, shouldPersistRequestDetail } from "./requestDetail.js";
|
||||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||||
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
|
||||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||||
@@ -282,7 +282,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
|||||||
/**
|
/**
|
||||||
* Handle non-streaming response from provider.
|
* Handle non-streaming response from provider.
|
||||||
*/
|
*/
|
||||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log, streamErrorPatterns }) {
|
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log, streamErrorPatterns, persistUsage = "all" }) {
|
||||||
trackDone();
|
trackDone();
|
||||||
const contentType = providerResponse.headers.get("content-type") || "";
|
const contentType = providerResponse.headers.get("content-type") || "";
|
||||||
let responseBody;
|
let responseBody;
|
||||||
@@ -387,23 +387,25 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
|||||||
reqLogger.logConvertedResponse(translatedResponse);
|
reqLogger.logConvertedResponse(translatedResponse);
|
||||||
|
|
||||||
const totalLatency = Date.now() - requestStartTime;
|
const totalLatency = Date.now() - requestStartTime;
|
||||||
saveRequestDetail(buildRequestDetail({
|
if (shouldPersistRequestDetail(persistUsage, "success")) {
|
||||||
provider, model, connectionId, apiKey,
|
saveRequestDetail(buildRequestDetail({
|
||||||
latency: { ttft: totalLatency, total: totalLatency },
|
provider, model, connectionId, apiKey,
|
||||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
latency: { ttft: totalLatency, total: totalLatency },
|
||||||
request: extractRequestConfig(body, stream),
|
tokens: tokensForDetail(usage),
|
||||||
providerRequest: finalBody || translatedBody || null,
|
request: extractRequestConfig(body, stream),
|
||||||
providerResponse: responseBody || null,
|
providerRequest: finalBody || translatedBody || null,
|
||||||
response: {
|
providerResponse: responseBody || null,
|
||||||
content: translatedResponse?.choices?.[0]?.message?.content || translatedResponse?.content || null,
|
response: {
|
||||||
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
content: translatedResponse?.choices?.[0]?.message?.content || translatedResponse?.content || null,
|
||||||
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
||||||
},
|
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
||||||
pxpipe,
|
},
|
||||||
status: "success"
|
pxpipe,
|
||||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
status: "success"
|
||||||
console.error("[RequestDetail] Failed to save:", err.message);
|
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
||||||
});
|
console.error("[RequestDetail] Failed to save:", err.message);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
return {
|
return {
|
||||||
success: true,
|
success: true,
|
||||||
|
|||||||
@@ -110,6 +110,29 @@ export function formatDoneLine({ usage, latency }) {
|
|||||||
return `DONE ${latency?.total ?? 0}ms${ttftStr} · ${inStr} · OUT ${outTok}`;
|
return `DONE ${latency?.total ?? 0}ms${ttftStr} · ${inStr} · OUT ${outTok}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Request-details storage convention: always prompt_tokens / completion_tokens.
|
||||||
|
// Translators often hand Claude `{input_tokens, output_tokens}` (or Gemini
|
||||||
|
// counts) to onStreamComplete; the Details tab only reads the OpenAI names,
|
||||||
|
// so an uncanonicalized object shows up as input=0 / output=0.
|
||||||
|
export function tokensForDetail(usage) {
|
||||||
|
if (!usage || typeof usage !== "object") {
|
||||||
|
return { prompt_tokens: 0, completion_tokens: 0 };
|
||||||
|
}
|
||||||
|
return canonicalizeUsage(usage) || {
|
||||||
|
prompt_tokens: usage.prompt_tokens ?? usage.input_tokens ?? 0,
|
||||||
|
completion_tokens: usage.completion_tokens ?? usage.output_tokens ?? 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Combo fallback/account hops must not inflate Details with 0-token rows.
|
||||||
|
// `streaming-start` is never persisted: the placeholder was status=success at
|
||||||
|
// tokens=0, and nested/fusion paths often abandon the stream before complete.
|
||||||
|
export function shouldPersistRequestDetail(persistUsage, kind) {
|
||||||
|
if (kind === "streaming-start") return false;
|
||||||
|
if (persistUsage === "success-only") return kind === "success";
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE", silent = false }) {
|
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE", silent = false }) {
|
||||||
if (!tokens || typeof tokens !== "object") return;
|
if (!tokens || typeof tokens !== "object") return;
|
||||||
|
|
||||||
|
|||||||
@@ -13,6 +13,8 @@ import {
|
|||||||
extractRequestConfig,
|
extractRequestConfig,
|
||||||
saveUsageStats,
|
saveUsageStats,
|
||||||
formatDoneLine,
|
formatDoneLine,
|
||||||
|
tokensForDetail,
|
||||||
|
shouldPersistRequestDetail,
|
||||||
} from "./requestDetail.js";
|
} from "./requestDetail.js";
|
||||||
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
|
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
|
||||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||||
@@ -60,7 +62,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
|||||||
/**
|
/**
|
||||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||||
*/
|
*/
|
||||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) {
|
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, pxpipe, reqTag, log, credentials }) {
|
||||||
if (onRequestSuccess) {
|
if (onRequestSuccess) {
|
||||||
Promise.resolve()
|
Promise.resolve()
|
||||||
.then(onRequestSuccess)
|
.then(onRequestSuccess)
|
||||||
@@ -145,35 +147,6 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
|||||||
stallTimeoutMs,
|
stallTimeoutMs,
|
||||||
);
|
);
|
||||||
|
|
||||||
saveRequestDetail(
|
|
||||||
buildRequestDetail(
|
|
||||||
{
|
|
||||||
provider,
|
|
||||||
model,
|
|
||||||
connectionId,
|
|
||||||
apiKey,
|
|
||||||
latency: { ttft: 0, total: Date.now() - requestStartTime },
|
|
||||||
tokens: { prompt_tokens: 0, completion_tokens: 0 },
|
|
||||||
request: extractRequestConfig(body, stream),
|
|
||||||
providerRequest: finalBody || translatedBody || null,
|
|
||||||
providerResponse: "[Streaming - raw response not captured]",
|
|
||||||
response: {
|
|
||||||
content: "[Streaming in progress...]",
|
|
||||||
thinking: null,
|
|
||||||
type: "streaming",
|
|
||||||
},
|
|
||||||
pxpipe,
|
|
||||||
status: "success",
|
|
||||||
},
|
|
||||||
{ id: streamDetailId },
|
|
||||||
),
|
|
||||||
).catch((err) => {
|
|
||||||
console.error(
|
|
||||||
"[RequestDetail] Failed to save streaming request:",
|
|
||||||
err.message,
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
return {
|
return {
|
||||||
success: true,
|
success: true,
|
||||||
response: new Response(transformedBody, { headers: SSE_HEADERS }),
|
response: new Response(transformedBody, { headers: SSE_HEADERS }),
|
||||||
@@ -198,6 +171,7 @@ export function buildOnStreamComplete({
|
|||||||
reqTag,
|
reqTag,
|
||||||
log,
|
log,
|
||||||
streamErrorPatterns,
|
streamErrorPatterns,
|
||||||
|
persistUsage = "all",
|
||||||
}) {
|
}) {
|
||||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||||
|
|
||||||
@@ -210,37 +184,39 @@ export function buildOnStreamComplete({
|
|||||||
const safeThinking = contentObj?.thinking || null;
|
const safeThinking = contentObj?.thinking || null;
|
||||||
const rawProviderText = typeof contentObj?.rawProviderText === "string" ? contentObj.rawProviderText : "";
|
const rawProviderText = typeof contentObj?.rawProviderText === "string" ? contentObj.rawProviderText : "";
|
||||||
|
|
||||||
saveRequestDetail(
|
if (shouldPersistRequestDetail(persistUsage, "success")) {
|
||||||
buildRequestDetail(
|
saveRequestDetail(
|
||||||
{
|
buildRequestDetail(
|
||||||
provider,
|
{
|
||||||
model,
|
provider,
|
||||||
connectionId,
|
model,
|
||||||
apiKey,
|
connectionId,
|
||||||
latency,
|
apiKey,
|
||||||
tokens: usage || { prompt_tokens: 0, completion_tokens: 0 },
|
latency,
|
||||||
request: extractRequestConfig(body, stream),
|
tokens: tokensForDetail(usage),
|
||||||
providerRequest: finalBody || translatedBody || null,
|
request: extractRequestConfig(body, stream),
|
||||||
providerResponse: rawProviderText || safeContent,
|
providerRequest: finalBody || translatedBody || null,
|
||||||
response: {
|
providerResponse: rawProviderText || safeContent,
|
||||||
content: safeContent,
|
response: {
|
||||||
thinking: safeThinking,
|
content: safeContent,
|
||||||
type: "streaming",
|
thinking: safeThinking,
|
||||||
|
type: "streaming",
|
||||||
|
},
|
||||||
|
pxpipe,
|
||||||
|
status: streamStatusForContent(
|
||||||
|
streamErrorPatterns?.[provider],
|
||||||
|
safeContent,
|
||||||
|
),
|
||||||
},
|
},
|
||||||
pxpipe,
|
{ id: streamDetailId },
|
||||||
status: streamStatusForContent(
|
),
|
||||||
streamErrorPatterns?.[provider],
|
).catch((err) => {
|
||||||
safeContent,
|
console.error(
|
||||||
),
|
"[RequestDetail] Failed to update streaming content:",
|
||||||
},
|
err.message,
|
||||||
{ id: streamDetailId },
|
);
|
||||||
),
|
});
|
||||||
).catch((err) => {
|
}
|
||||||
console.error(
|
|
||||||
"[RequestDetail] Failed to update streaming content:",
|
|
||||||
err.message,
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||||
saveUsageStats({
|
saveUsageStats({
|
||||||
|
|||||||
@@ -246,23 +246,39 @@ export function resetComboRotation(comboName) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get combo models from combos data
|
* Cancel an unused Response body so its SSE transform (and usage callback)
|
||||||
|
* is not left hanging. Fusion timeouts and combo fallback both drop Response
|
||||||
|
* objects without reading them; an unread stream never fires onStreamComplete,
|
||||||
|
* which is how request-details rows get stuck at input/output tokens = 0.
|
||||||
|
*/
|
||||||
|
export function discardResponse(res) {
|
||||||
|
if (!res || res.__timeout || res.__error) return;
|
||||||
|
const cancel = res.body?.cancel;
|
||||||
|
if (typeof cancel === "function") {
|
||||||
|
try { cancel.call(res.body); } catch { /* already closed / consumed */ }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get combo models from combos data.
|
||||||
|
* Nested combo names are kept as-is (comboA listing comboB yields ["comboB", ...]).
|
||||||
|
* Flattening them into leaves would explode fusion/round-robin: every nested
|
||||||
|
* leaf becomes its own panel/rotation slot. Nested combos run as one unit
|
||||||
|
* (see handleSingleModelChat: inner strategy forced to fallback).
|
||||||
|
*
|
||||||
* @param {string} modelStr - Model string to check
|
* @param {string} modelStr - Model string to check
|
||||||
* @param {Array|Object} combosData - Array of combos or object with combos
|
* @param {Array|Object} combosData - Array of combos or object with combos
|
||||||
* @returns {string[]|null} Array of models or null if not a combo
|
* @returns {string[]|null} Direct members, or null if not a combo
|
||||||
*/
|
*/
|
||||||
export function getComboModelsFromData(modelStr, combosData) {
|
export function getComboModelsFromData(modelStr, combosData) {
|
||||||
// Don't check if it's in provider/model format
|
if (typeof modelStr !== "string" || modelStr.includes("/")) return null;
|
||||||
if (modelStr.includes("/")) return null;
|
|
||||||
|
|
||||||
// Handle both array and object formats
|
|
||||||
const combos = Array.isArray(combosData) ? combosData : (combosData?.combos || []);
|
const combos = Array.isArray(combosData) ? combosData : (combosData?.combos || []);
|
||||||
|
const combo = combos.find((c) => c.name === modelStr);
|
||||||
const combo = combos.find(c => c.name === modelStr);
|
if (!combo || combo.enabled === false || !Array.isArray(combo.models) || combo.models.length === 0) {
|
||||||
if (combo && combo.models && combo.models.length > 0) {
|
return null;
|
||||||
return combo.models;
|
|
||||||
}
|
}
|
||||||
return null;
|
return combo.models;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -339,6 +355,9 @@ export async function handleComboChat({ body, models, handleSingleModel, log, co
|
|||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Falling through: this body will never be read by the client.
|
||||||
|
discardResponse(result);
|
||||||
|
|
||||||
// For transient errors (503/502/504), wait for cooldown before falling through
|
// For transient errors (503/502/504), wait for cooldown before falling through
|
||||||
// so a briefly-overloaded provider gets a chance to recover rather than being
|
// so a briefly-overloaded provider gets a chance to recover rather than being
|
||||||
// skipped immediately (fixes: combo falls through on transient 503)
|
// skipped immediately (fixes: combo falls through on transient 503)
|
||||||
@@ -509,8 +528,13 @@ function collectPanel(calls, { minPanel, stragglerGraceMs, panelHardTimeoutMs })
|
|||||||
const hardTimer = setTimeout(finish, panelHardTimeoutMs);
|
const hardTimer = setTimeout(finish, panelHardTimeoutMs);
|
||||||
calls.forEach((p, i) => {
|
calls.forEach((p, i) => {
|
||||||
Promise.resolve(p)
|
Promise.resolve(p)
|
||||||
.then((v) => { out[i] = v; })
|
.then((v) => {
|
||||||
.catch((e) => { out[i] = { __error: e }; })
|
// Quorum/timeout already moved on: cancel the late body so its
|
||||||
|
// streaming usage callback is not left at tokens=0 forever.
|
||||||
|
if (finished) discardResponse(v);
|
||||||
|
else out[i] = v;
|
||||||
|
})
|
||||||
|
.catch((e) => { if (!finished) out[i] = { __error: e }; })
|
||||||
.finally(() => {
|
.finally(() => {
|
||||||
settled++;
|
settled++;
|
||||||
if (out[i] && out[i].ok) ok++;
|
if (out[i] && out[i].ok) ok++;
|
||||||
@@ -590,7 +614,7 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
|||||||
if (!res) { log.warn("FUSION", `Panel ${model} dropped (straggler/timeout)`); continue; }
|
if (!res) { log.warn("FUSION", `Panel ${model} dropped (straggler/timeout)`); continue; }
|
||||||
if (res.__timeout) { log.warn("FUSION", `Panel ${model} timed out`); continue; }
|
if (res.__timeout) { log.warn("FUSION", `Panel ${model} timed out`); continue; }
|
||||||
if (res.__error) { log.warn("FUSION", `Panel ${model} threw`, { error: res.__error?.message || String(res.__error) }); continue; }
|
if (res.__error) { log.warn("FUSION", `Panel ${model} threw`, { error: res.__error?.message || String(res.__error) }); continue; }
|
||||||
if (!res.ok) { log.warn("FUSION", `Panel ${model} failed`, { status: res.status }); continue; }
|
if (!res.ok) { log.warn("FUSION", `Panel ${model} failed`, { status: res.status }); discardResponse(res); continue; }
|
||||||
try {
|
try {
|
||||||
const json = await res.clone().json();
|
const json = await res.clone().json();
|
||||||
const text = extractPanelText(json);
|
const text = extractPanelText(json);
|
||||||
@@ -602,6 +626,11 @@ export async function handleFusionChat({ body, models, handleSingleModel, log, c
|
|||||||
}
|
}
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
log.warn("FUSION", `Panel ${model} unparseable`, { error: e.message || String(e) });
|
log.warn("FUSION", `Panel ${model} unparseable`, { error: e.message || String(e) });
|
||||||
|
} finally {
|
||||||
|
// Panel responses are never returned to the client; cancel the original
|
||||||
|
// body (clone().json() already captured the payload) so streaming
|
||||||
|
// placeholders are not left as success/0-token rows.
|
||||||
|
discardResponse(res);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -92,7 +92,7 @@ function getCacheCreationTokens(tokens) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
function getInputTokens(tokens) {
|
function getInputTokens(tokens) {
|
||||||
const prompt = tokens?.prompt_tokens || tokens?.input_tokens || 0;
|
const prompt = tokens?.prompt_tokens || tokens?.input_tokens || tokens?.promptTokenCount || 0;
|
||||||
// Canonical storage keeps prompt cache-inclusive. Legacy Claude rows may have
|
// Canonical storage keeps prompt cache-inclusive. Legacy Claude rows may have
|
||||||
// stored prompt cache-exclusive; fall back to cache when it's larger so old
|
// stored prompt cache-exclusive; fall back to cache when it's larger so old
|
||||||
// rows don't under-report input.
|
// rows don't under-report input.
|
||||||
@@ -100,6 +100,10 @@ function getInputTokens(tokens) {
|
|||||||
return prompt < cache ? cache : prompt;
|
return prompt < cache ? cache : prompt;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function getOutputTokens(tokens) {
|
||||||
|
return tokens?.completion_tokens || tokens?.output_tokens || tokens?.candidatesTokenCount || 0;
|
||||||
|
}
|
||||||
|
|
||||||
export default function RequestDetailsTab() {
|
export default function RequestDetailsTab() {
|
||||||
const [details, setDetails] = useState([]);
|
const [details, setDetails] = useState([]);
|
||||||
const [pagination, setPagination] = useState({
|
const [pagination, setPagination] = useState({
|
||||||
@@ -427,13 +431,15 @@ export default function RequestDetailsTab() {
|
|||||||
"inline-flex items-center gap-1 rounded-full px-2 py-0.5 text-xs font-medium",
|
"inline-flex items-center gap-1 rounded-full px-2 py-0.5 text-xs font-medium",
|
||||||
detail.status === "error"
|
detail.status === "error"
|
||||||
? "bg-red-500/10 text-red-600"
|
? "bg-red-500/10 text-red-600"
|
||||||
: "bg-green-500/10 text-green-600"
|
: detail.status === "streaming"
|
||||||
|
? "bg-amber-500/10 text-amber-700 dark:text-amber-400"
|
||||||
|
: "bg-green-500/10 text-green-600"
|
||||||
)}>
|
)}>
|
||||||
<span className={cn(
|
<span className={cn(
|
||||||
"h-1.5 w-1.5 rounded-full",
|
"h-1.5 w-1.5 rounded-full",
|
||||||
detail.status === "error" ? "bg-red-500" : "bg-green-500"
|
detail.status === "error" ? "bg-red-500" : detail.status === "streaming" ? "bg-amber-500" : "bg-green-500"
|
||||||
)} />
|
)} />
|
||||||
{detail.status === "error" ? "Error" : "Success"}
|
{detail.status === "error" ? "Error" : detail.status === "streaming" ? "Streaming" : "Success"}
|
||||||
</span>
|
</span>
|
||||||
</td>
|
</td>
|
||||||
<td className="p-4 text-sm text-text-main text-right font-mono">
|
<td className="p-4 text-sm text-text-main text-right font-mono">
|
||||||
@@ -446,7 +452,7 @@ export default function RequestDetailsTab() {
|
|||||||
{getCacheCreationTokens(detail.tokens) > 0 ? getCacheCreationTokens(detail.tokens).toLocaleString() : "—"}
|
{getCacheCreationTokens(detail.tokens) > 0 ? getCacheCreationTokens(detail.tokens).toLocaleString() : "—"}
|
||||||
</td>
|
</td>
|
||||||
<td className="p-4 text-sm text-text-main text-right font-mono">
|
<td className="p-4 text-sm text-text-main text-right font-mono">
|
||||||
{detail.tokens?.completion_tokens?.toLocaleString() || 0}
|
{getOutputTokens(detail.tokens).toLocaleString()}
|
||||||
</td>
|
</td>
|
||||||
<td className="p-4 text-sm text-text-muted">
|
<td className="p-4 text-sm text-text-muted">
|
||||||
<div className="flex flex-col gap-0.5">
|
<div className="flex flex-col gap-0.5">
|
||||||
@@ -525,7 +531,9 @@ export default function RequestDetailsTab() {
|
|||||||
<span className="text-text-muted">Status:</span>{" "}
|
<span className="text-text-muted">Status:</span>{" "}
|
||||||
<span className={cn(
|
<span className={cn(
|
||||||
"font-medium",
|
"font-medium",
|
||||||
selectedDetail.status === "success" ? "text-green-600" : "text-red-600"
|
selectedDetail.status === "success" ? "text-green-600"
|
||||||
|
: selectedDetail.status === "streaming" ? "text-amber-600"
|
||||||
|
: "text-red-600"
|
||||||
)}>
|
)}>
|
||||||
{selectedDetail.status}
|
{selectedDetail.status}
|
||||||
</span>
|
</span>
|
||||||
@@ -561,7 +569,7 @@ export default function RequestDetailsTab() {
|
|||||||
<div>
|
<div>
|
||||||
<span className="text-text-muted">Output Tokens:</span>{" "}
|
<span className="text-text-muted">Output Tokens:</span>{" "}
|
||||||
<span className="text-text-main font-mono">
|
<span className="text-text-main font-mono">
|
||||||
{selectedDetail.tokens?.completion_tokens?.toLocaleString() || 0}
|
{getOutputTokens(selectedDetail.tokens).toLocaleString()}
|
||||||
</span>
|
</span>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|||||||
@@ -115,6 +115,7 @@ export async function handleChat(request, clientRawRequest = null) {
|
|||||||
const augmentedModels = augmentModelsWithCapacityAdapter(comboModels, requiredCapabilities, settings);
|
const augmentedModels = augmentModelsWithCapacityAdapter(comboModels, requiredCapabilities, settings);
|
||||||
const adapterAdded = augmentedModels.filter((m) => !comboModels.includes(m));
|
const adapterAdded = augmentedModels.filter((m) => !comboModels.includes(m));
|
||||||
|
|
||||||
|
const comboOpts = { persistUsage: "success-only", comboStack: new Set([modelStr]) };
|
||||||
if (comboStrategy === "fusion") {
|
if (comboStrategy === "fusion") {
|
||||||
log.info("CHAT", `Combo "${modelStr}" with ${comboModels.length} models (strategy: fusion)`);
|
log.info("CHAT", `Combo "${modelStr}" with ${comboModels.length} models (strategy: fusion)`);
|
||||||
return handleFusionChat({
|
return handleFusionChat({
|
||||||
@@ -126,7 +127,7 @@ export async function handleChat(request, clientRawRequest = null) {
|
|||||||
const { tools, tool_choice, ...cleanBody } = clientRawRequest.body || {};
|
const { tools, tool_choice, ...cleanBody } = clientRawRequest.body || {};
|
||||||
cleanRawReq = { ...clientRawRequest, body: cleanBody };
|
cleanRawReq = { ...clientRawRequest, body: cleanBody };
|
||||||
}
|
}
|
||||||
return handleSingleModelChat(b, m, cleanRawReq, request, apiKey);
|
return handleSingleModelChat(b, m, cleanRawReq, request, apiKey, comboOpts);
|
||||||
},
|
},
|
||||||
log,
|
log,
|
||||||
comboName: modelStr,
|
comboName: modelStr,
|
||||||
@@ -141,7 +142,7 @@ export async function handleChat(request, clientRawRequest = null) {
|
|||||||
body,
|
body,
|
||||||
models: augmentedModels,
|
models: augmentedModels,
|
||||||
handleSingleModel: withCapacityAdapterStripping(
|
handleSingleModel: withCapacityAdapterStripping(
|
||||||
(b, m) => handleSingleModelChat(b, m, clientRawRequest, request, apiKey),
|
(b, m) => handleSingleModelChat(b, m, clientRawRequest, request, apiKey, comboOpts),
|
||||||
adapterAdded
|
adapterAdded
|
||||||
),
|
),
|
||||||
log,
|
log,
|
||||||
@@ -161,7 +162,7 @@ export async function handleChat(request, clientRawRequest = null) {
|
|||||||
body,
|
body,
|
||||||
models: soloAugmented,
|
models: soloAugmented,
|
||||||
handleSingleModel: withCapacityAdapterStripping(
|
handleSingleModel: withCapacityAdapterStripping(
|
||||||
(b, m) => handleSingleModelChat(b, m, clientRawRequest, request, apiKey),
|
(b, m) => handleSingleModelChat(b, m, clientRawRequest, request, apiKey, { persistUsage: "success-only" }),
|
||||||
adapterAdded
|
adapterAdded
|
||||||
),
|
),
|
||||||
log,
|
log,
|
||||||
@@ -176,24 +177,42 @@ export async function handleChat(request, clientRawRequest = null) {
|
|||||||
/**
|
/**
|
||||||
* Handle single model chat request
|
* Handle single model chat request
|
||||||
*/
|
*/
|
||||||
async function handleSingleModelChat(body, modelStr, clientRawRequest = null, request = null, apiKey = null) {
|
async function handleSingleModelChat(body, modelStr, clientRawRequest = null, request = null, apiKey = null, opts = {}) {
|
||||||
// Dashboard per-key tests pin the account via x-connection-id (same header
|
// Dashboard per-key tests pin the account via x-connection-id (same header
|
||||||
// contract as embeddings/images/video). Pinned requests must not rotate.
|
// contract as embeddings/images/video). Pinned requests must not rotate.
|
||||||
const preferredConnectionId = request?.headers?.get("x-connection-id") || null;
|
const preferredConnectionId = request?.headers?.get("x-connection-id") || null;
|
||||||
const modelInfo = await getModelInfo(modelStr);
|
const modelInfo = await getModelInfo(modelStr);
|
||||||
|
const stack = opts.comboStack instanceof Set ? opts.comboStack : new Set();
|
||||||
|
const persistUsage = opts.persistUsage || "all";
|
||||||
|
|
||||||
// If provider is null, this might be a combo name - check and handle
|
// If provider is null, this might be a combo name - check and handle
|
||||||
if (!modelInfo.provider) {
|
if (!modelInfo.provider) {
|
||||||
|
if (stack.has(modelStr)) {
|
||||||
|
log.warn("CHAT", `Combo cycle detected: ${modelStr}`);
|
||||||
|
return errorResponse(HTTP_STATUS.BAD_REQUEST, `Combo cycle detected involving "${modelStr}"`);
|
||||||
|
}
|
||||||
const comboModels = await getComboModels(modelStr);
|
const comboModels = await getComboModels(modelStr);
|
||||||
if (comboModels) {
|
if (comboModels) {
|
||||||
const chatSettings = await getSettings();
|
const chatSettings = await getSettings();
|
||||||
// Check for combo-specific strategy first, fallback to global
|
|
||||||
const comboStrategies = chatSettings.comboStrategies || {};
|
const comboStrategies = chatSettings.comboStrategies || {};
|
||||||
const comboSpecificStrategy = comboStrategies[modelStr]?.fallbackStrategy;
|
const comboSpecificStrategy = comboStrategies[modelStr]?.fallbackStrategy;
|
||||||
const comboStrategy = comboSpecificStrategy || chatSettings.comboStrategy || "fallback";
|
// Nested combo (comboA lists comboB): always fallback. Running the inner
|
||||||
|
// combo's fusion/round-robin as its own combo multiplies requests and
|
||||||
|
// leaves abandoned streams at tokens=0 in Details.
|
||||||
|
const nested = stack.size > 0;
|
||||||
|
const comboStrategy = nested
|
||||||
|
? "fallback"
|
||||||
|
: (comboSpecificStrategy || chatSettings.comboStrategy || "fallback");
|
||||||
const requiredCapabilities = detectRequiredCapabilities(body);
|
const requiredCapabilities = detectRequiredCapabilities(body);
|
||||||
const augmentedModels = augmentModelsWithCapacityAdapter(comboModels, requiredCapabilities, chatSettings);
|
const augmentedModels = augmentModelsWithCapacityAdapter(comboModels, requiredCapabilities, chatSettings);
|
||||||
const adapterAdded = augmentedModels.filter((m) => !comboModels.includes(m));
|
const adapterAdded = augmentedModels.filter((m) => !comboModels.includes(m));
|
||||||
|
const nextOpts = {
|
||||||
|
persistUsage: "success-only",
|
||||||
|
comboStack: new Set(stack).add(modelStr),
|
||||||
|
};
|
||||||
|
if (nested) {
|
||||||
|
log.info("CHAT", `Nested combo "${modelStr}" → fallback (${comboModels.length} members)`);
|
||||||
|
}
|
||||||
|
|
||||||
if (comboStrategy === "fusion") {
|
if (comboStrategy === "fusion") {
|
||||||
log.info("CHAT", `Combo "${modelStr}" with ${comboModels.length} models (strategy: fusion)`);
|
log.info("CHAT", `Combo "${modelStr}" with ${comboModels.length} models (strategy: fusion)`);
|
||||||
@@ -206,7 +225,7 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re
|
|||||||
const { tools, tool_choice, ...cleanBody } = clientRawRequest.body || {};
|
const { tools, tool_choice, ...cleanBody } = clientRawRequest.body || {};
|
||||||
cleanRawReq = { ...clientRawRequest, body: cleanBody };
|
cleanRawReq = { ...clientRawRequest, body: cleanBody };
|
||||||
}
|
}
|
||||||
return handleSingleModelChat(b, m, cleanRawReq, request, apiKey);
|
return handleSingleModelChat(b, m, cleanRawReq, request, apiKey, nextOpts);
|
||||||
},
|
},
|
||||||
log,
|
log,
|
||||||
comboName: modelStr,
|
comboName: modelStr,
|
||||||
@@ -221,7 +240,7 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re
|
|||||||
body,
|
body,
|
||||||
models: augmentedModels,
|
models: augmentedModels,
|
||||||
handleSingleModel: withCapacityAdapterStripping(
|
handleSingleModel: withCapacityAdapterStripping(
|
||||||
(b, m) => handleSingleModelChat(b, m, clientRawRequest, request, apiKey),
|
(b, m) => handleSingleModelChat(b, m, clientRawRequest, request, apiKey, nextOpts),
|
||||||
adapterAdded
|
adapterAdded
|
||||||
),
|
),
|
||||||
log,
|
log,
|
||||||
@@ -323,6 +342,7 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re
|
|||||||
providerThinking,
|
providerThinking,
|
||||||
capsOverride,
|
capsOverride,
|
||||||
streamErrorPatterns: chatSettings.streamErrorPatterns || null,
|
streamErrorPatterns: chatSettings.streamErrorPatterns || null,
|
||||||
|
persistUsage,
|
||||||
// Detect source format by endpoint + body
|
// Detect source format by endpoint + body
|
||||||
sourceFormatOverride: request?.url ? detectFormatByEndpoint(new URL(request.url).pathname, body) : null,
|
sourceFormatOverride: request?.url ? detectFormatByEndpoint(new URL(request.url).pathname, body) : null,
|
||||||
onCredentialsRefreshed: async (newCreds) => {
|
onCredentialsRefreshed: async (newCreds) => {
|
||||||
|
|||||||
@@ -123,21 +123,16 @@ export async function getModelInfo(modelStr) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Check if model is a combo and get models list
|
* Check if model is a combo and get models list.
|
||||||
|
* Nested combo names are returned as members (not inlined); chat.js runs them
|
||||||
|
* as a fallback unit so fusion/round-robin of the parent is not exploded.
|
||||||
* @returns {Promise<string[]|null>} Array of models or null if not a combo
|
* @returns {Promise<string[]|null>} Array of models or null if not a combo
|
||||||
*/
|
*/
|
||||||
export async function getComboModels(modelStr) {
|
export async function getComboModels(modelStr) {
|
||||||
// Only check if it's not in provider/model format
|
if (typeof modelStr !== "string" || modelStr.includes("/")) return null;
|
||||||
if (modelStr.includes("/")) return null;
|
|
||||||
|
|
||||||
const combo = await getComboByName(modelStr);
|
const combo = await getComboByName(modelStr);
|
||||||
if (
|
if (!combo || combo.enabled === false || !Array.isArray(combo.models) || combo.models.length === 0) {
|
||||||
combo &&
|
return null;
|
||||||
combo.enabled !== false &&
|
|
||||||
combo.models &&
|
|
||||||
combo.models.length > 0
|
|
||||||
) {
|
|
||||||
return combo.models;
|
|
||||||
}
|
}
|
||||||
return null;
|
return combo.models;
|
||||||
}
|
}
|
||||||
|
|||||||
128
tests/unit/combo-nested.test.js
Normal file
128
tests/unit/combo-nested.test.js
Normal file
@@ -0,0 +1,128 @@
|
|||||||
|
import { describe, it, expect, vi } from "vitest";
|
||||||
|
|
||||||
|
import { getComboModelsFromData, discardResponse, handleFusionChat } from "../../open-sse/services/combo.js";
|
||||||
|
import { tokensForDetail, shouldPersistRequestDetail } from "../../open-sse/handlers/chatCore/requestDetail.js";
|
||||||
|
|
||||||
|
vi.mock("@/lib/usageDb.js", () => ({
|
||||||
|
saveRequestUsage: vi.fn(),
|
||||||
|
saveRequestDetail: vi.fn(),
|
||||||
|
}));
|
||||||
|
vi.mock("../../open-sse/utils/stream.js", () => ({
|
||||||
|
COLORS: {},
|
||||||
|
formatSSE: vi.fn(),
|
||||||
|
}));
|
||||||
|
|
||||||
|
const log = { info: () => {}, warn: () => {}, debug: () => {} };
|
||||||
|
|
||||||
|
describe("nested combo members stay units", () => {
|
||||||
|
it("keeps comboB as a member of comboA (does not flatten leaves into the parent)", () => {
|
||||||
|
const combos = [
|
||||||
|
{ name: "comboA", models: ["comboB", "comboC", "x/y"] },
|
||||||
|
{ name: "comboB", models: ["a/m1", "b/m2"] },
|
||||||
|
{ name: "comboC", models: ["c/m3"] },
|
||||||
|
];
|
||||||
|
expect(getComboModelsFromData("comboA", combos)).toEqual(["comboB", "comboC", "x/y"]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns null for a plain provider/model string", () => {
|
||||||
|
expect(getComboModelsFromData("openai/gpt-4", [{ name: "comboA", models: ["a/m1"] }])).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("still returns direct members of a top-level combo", () => {
|
||||||
|
const combos = [{ name: "solo", models: ["a/m1", "b/m2"] }];
|
||||||
|
expect(getComboModelsFromData("solo", combos)).toEqual(["a/m1", "b/m2"]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("ignores a disabled combo", () => {
|
||||||
|
const combos = [{ name: "solo", enabled: false, models: ["a/m1"] }];
|
||||||
|
expect(getComboModelsFromData("solo", combos)).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("shouldPersistRequestDetail", () => {
|
||||||
|
it("never persists the streaming-start placeholder (0-token fake success)", () => {
|
||||||
|
expect(shouldPersistRequestDetail("all", "streaming-start")).toBe(false);
|
||||||
|
expect(shouldPersistRequestDetail("success-only", "streaming-start")).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("combo hops skip failed attempts so nested fallback does not inflate Details", () => {
|
||||||
|
expect(shouldPersistRequestDetail("success-only", "error")).toBe(false);
|
||||||
|
expect(shouldPersistRequestDetail("success-only", "success")).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("direct (non-combo) requests still persist errors", () => {
|
||||||
|
expect(shouldPersistRequestDetail("all", "error")).toBe(true);
|
||||||
|
expect(shouldPersistRequestDetail("all", "success")).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("tokensForDetail", () => {
|
||||||
|
it("canonicalizes Claude input_tokens/output_tokens so the Details tab is not 0/0", () => {
|
||||||
|
expect(tokensForDetail({ input_tokens: 120, output_tokens: 40 })).toEqual(
|
||||||
|
expect.objectContaining({ prompt_tokens: 120, completion_tokens: 40 }),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("keeps OpenAI prompt/completion names", () => {
|
||||||
|
expect(tokensForDetail({ prompt_tokens: 10, completion_tokens: 3 })).toEqual(
|
||||||
|
expect.objectContaining({ prompt_tokens: 10, completion_tokens: 3 }),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns zeros for missing usage (streaming placeholder)", () => {
|
||||||
|
expect(tokensForDetail(null)).toEqual({ prompt_tokens: 0, completion_tokens: 0 });
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("discardResponse", () => {
|
||||||
|
it("cancels an unused body so a dropped stream can finish", () => {
|
||||||
|
const cancel = vi.fn();
|
||||||
|
discardResponse({ body: { cancel } });
|
||||||
|
expect(cancel).toHaveBeenCalledTimes(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("is a no-op for timeout/error sentinels and missing bodies", () => {
|
||||||
|
expect(() => discardResponse({ __timeout: true, body: { cancel: () => { throw new Error("no"); } } })).not.toThrow();
|
||||||
|
expect(() => discardResponse(null)).not.toThrow();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("fusion cancels unused panel bodies", () => {
|
||||||
|
it("cancels a late panel Response so it is not left as a 0-token streaming row", async () => {
|
||||||
|
const cancel = vi.fn();
|
||||||
|
const ok = (content) => {
|
||||||
|
const json = { choices: [{ message: { role: "assistant", content } }] };
|
||||||
|
const make = () => ({ ok: true, status: 200, clone: make, json: async () => json });
|
||||||
|
return make();
|
||||||
|
};
|
||||||
|
const slow = () => new Promise((resolve) => {
|
||||||
|
setTimeout(() => {
|
||||||
|
resolve({
|
||||||
|
ok: true,
|
||||||
|
status: 200,
|
||||||
|
body: { cancel },
|
||||||
|
clone() { return this; },
|
||||||
|
json: async () => ({ choices: [{ message: { content: "late" } }] }),
|
||||||
|
});
|
||||||
|
}, 40);
|
||||||
|
});
|
||||||
|
|
||||||
|
const handleSingleModel = vi.fn(async (_body, model) => {
|
||||||
|
if (model === "p/slow") return slow();
|
||||||
|
if (model === "p/judge") return ok("FINAL");
|
||||||
|
return ok(`ans-${model}`);
|
||||||
|
});
|
||||||
|
|
||||||
|
await handleFusionChat({
|
||||||
|
body: { messages: [{ role: "user", content: "Q" }] },
|
||||||
|
models: ["p/a", "p/b", "p/slow"],
|
||||||
|
handleSingleModel,
|
||||||
|
log,
|
||||||
|
judgeModel: "p/judge",
|
||||||
|
tuning: { minPanel: 2, stragglerGraceMs: 1, panelHardTimeoutMs: 200 },
|
||||||
|
});
|
||||||
|
|
||||||
|
await new Promise((r) => setTimeout(r, 80));
|
||||||
|
expect(cancel).toHaveBeenCalled();
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user