Nested combos (comboA lists comboB, comboC, …) now stay one slot each: the inner combo always runs as fallback to produce a single answer. Failed hops are no longer written to Details/usage, and streaming no longer inserts a 0-token placeholder row. - chat.js: comboStack cycle detection; nested combos forced to fallback; persistUsage="success-only" for combo hops - combo.js: discardResponse() cancels unused bodies (fusion timeout / fallback) so dropped streams fire onStreamComplete; getComboModelsFromData keeps nested names and honors enabled=false - requestDetail.js: tokensForDetail() canonicalizes Claude/Gemini usage; shouldPersistRequestDetail() skips streaming-start and non-success hops - streamingHandler.js: drop the 0-token streaming placeholder write - RequestDetailsTab.js: read Gemini/Claude token names; show "streaming" status in amber - tests: add combo-nested.test.js (13 cases) - gitignore: ignore local .vitest/ artifacts Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
167 lines
7.0 KiB
JavaScript
167 lines
7.0 KiB
JavaScript
import { saveRequestUsage, saveRequestDetail } from "@/lib/usageDb.js";
|
|
import { COLORS } from "../../utils/stream.js";
|
|
import { canonicalizeUsage } from "../../utils/usageTracking.js";
|
|
|
|
const OPTIONAL_PARAMS = [
|
|
"temperature", "top_p", "top_k",
|
|
"max_tokens", "max_completion_tokens",
|
|
"thinking", "reasoning", "enable_thinking",
|
|
"presence_penalty", "frequency_penalty",
|
|
"seed", "stop", "tools", "tool_choice",
|
|
"response_format", "prediction", "store", "metadata",
|
|
"n", "logprobs", "top_logprobs", "logit_bias",
|
|
"user", "parallel_tool_calls"
|
|
];
|
|
|
|
export function extractRequestConfig(body, stream) {
|
|
const config = { messages: body.messages || [], model: body.model, stream };
|
|
for (const param of OPTIONAL_PARAMS) {
|
|
if (body[param] !== undefined) config[param] = body[param];
|
|
}
|
|
return config;
|
|
}
|
|
|
|
export function extractUsageFromResponse(responseBody) {
|
|
if (!responseBody || typeof responseBody !== "object") return null;
|
|
|
|
// Claude format
|
|
// Note: OpenAI Responses usage ({input_tokens, input_tokens_details:{cached_tokens}})
|
|
// also matches this branch. Its prompt is cache-INCLUSIVE and its cache rides in
|
|
// input_tokens_details, so emit it as cached_tokens — the convention
|
|
// canonicalizeUsage() passes through without folding. Reading it here keeps
|
|
// cache accounting correct for /v1/responses and codex traffic.
|
|
if (responseBody.usage?.input_tokens !== undefined) {
|
|
return {
|
|
prompt_tokens: responseBody.usage.input_tokens || 0,
|
|
completion_tokens: responseBody.usage.output_tokens || 0,
|
|
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.input_tokens_details?.cached_tokens,
|
|
cache_read_input_tokens: responseBody.usage.cache_read_input_tokens,
|
|
cache_creation_input_tokens: responseBody.usage.cache_creation_input_tokens
|
|
};
|
|
}
|
|
|
|
// OpenAI format
|
|
if (responseBody.usage?.prompt_tokens !== undefined) {
|
|
return {
|
|
prompt_tokens: responseBody.usage.prompt_tokens || 0,
|
|
completion_tokens: responseBody.usage.completion_tokens || 0,
|
|
cached_tokens: responseBody.usage.cached_tokens ?? responseBody.usage.prompt_tokens_details?.cached_tokens,
|
|
reasoning_tokens: responseBody.usage.completion_tokens_details?.reasoning_tokens
|
|
};
|
|
}
|
|
|
|
// Gemini format. Antigravity / gemini-cli wrap the payload in { response: {...} }.
|
|
const usageMetadata = responseBody.usageMetadata || responseBody.response?.usageMetadata;
|
|
if (usageMetadata) {
|
|
return {
|
|
prompt_tokens: usageMetadata.promptTokenCount || 0,
|
|
completion_tokens: usageMetadata.candidatesTokenCount || 0,
|
|
cached_tokens: usageMetadata.cachedContentTokenCount || 0,
|
|
reasoning_tokens: usageMetadata.thoughtsTokenCount || 0
|
|
};
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
// Mask API keys before they reach the requestDetails data blob / DB column.
|
|
// Only the prefix is kept — enough to distinguish keys without leaking them.
|
|
export function maskApiKey(key) {
|
|
if (!key || typeof key !== "string") return undefined;
|
|
const trimmed = key.trim();
|
|
if (trimmed.length <= 8) return trimmed.charAt(0) + "***";
|
|
return trimmed.slice(0, 8) + "***";
|
|
}
|
|
|
|
export function buildRequestDetail(base, overrides = {}) {
|
|
return {
|
|
provider: base.provider || "unknown",
|
|
model: base.model || "unknown",
|
|
connectionId: base.connectionId || undefined,
|
|
apiKey: maskApiKey(base.apiKey),
|
|
timestamp: new Date().toISOString(),
|
|
latency: base.latency || { ttft: 0, total: 0 },
|
|
tokens: base.tokens || { prompt_tokens: 0, completion_tokens: 0 },
|
|
request: base.request,
|
|
providerRequest: base.providerRequest || null,
|
|
providerResponse: base.providerResponse || null,
|
|
response: base.response || {},
|
|
pxpipe: base.pxpipe || undefined,
|
|
status: base.status || "success",
|
|
...overrides
|
|
};
|
|
}
|
|
|
|
// Build the "done" summary: duration, ttft, in/out tokens with cache breakdown
|
|
export function formatDoneLine({ usage, latency }) {
|
|
const u = usage || {};
|
|
const inTok = u.prompt_tokens ?? u.input_tokens ?? 0;
|
|
const outTok = u.completion_tokens ?? u.output_tokens ?? 0;
|
|
const cacheRead = u.cache_read_input_tokens ?? u.cached_tokens ?? u.prompt_tokens_details?.cached_tokens ?? 0;
|
|
const cacheCreate = u.cache_creation_input_tokens ?? 0;
|
|
let inStr = `IN ${inTok}`;
|
|
if (cacheRead || cacheCreate) {
|
|
const parts = [];
|
|
if (cacheRead) parts.push(`↻${cacheRead}`);
|
|
if (cacheCreate) parts.push(`+${cacheCreate}`);
|
|
inStr += ` (CACHE ${parts.join(" ")})`;
|
|
}
|
|
const ttftStr = latency?.ttft ? ` · TTFT ${latency.ttft}ms` : "";
|
|
return `DONE ${latency?.total ?? 0}ms${ttftStr} · ${inStr} · OUT ${outTok}`;
|
|
}
|
|
|
|
// Request-details storage convention: always prompt_tokens / completion_tokens.
|
|
// Translators often hand Claude `{input_tokens, output_tokens}` (or Gemini
|
|
// counts) to onStreamComplete; the Details tab only reads the OpenAI names,
|
|
// so an uncanonicalized object shows up as input=0 / output=0.
|
|
export function tokensForDetail(usage) {
|
|
if (!usage || typeof usage !== "object") {
|
|
return { prompt_tokens: 0, completion_tokens: 0 };
|
|
}
|
|
return canonicalizeUsage(usage) || {
|
|
prompt_tokens: usage.prompt_tokens ?? usage.input_tokens ?? 0,
|
|
completion_tokens: usage.completion_tokens ?? usage.output_tokens ?? 0,
|
|
};
|
|
}
|
|
|
|
// Combo fallback/account hops must not inflate Details with 0-token rows.
|
|
// `streaming-start` is never persisted: the placeholder was status=success at
|
|
// tokens=0, and nested/fusion paths often abandon the stream before complete.
|
|
export function shouldPersistRequestDetail(persistUsage, kind) {
|
|
if (kind === "streaming-start") return false;
|
|
if (persistUsage === "success-only") return kind === "success";
|
|
return true;
|
|
}
|
|
|
|
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE", silent = false }) {
|
|
if (!tokens || typeof tokens !== "object") return;
|
|
|
|
const inTokens = tokens.input_tokens ?? tokens.prompt_tokens ?? 0;
|
|
const outTokens = tokens.output_tokens ?? tokens.completion_tokens ?? 0;
|
|
|
|
if (inTokens === 0 && outTokens === 0) return;
|
|
|
|
if (!silent) {
|
|
const time = new Date().toLocaleTimeString("en-US", { hour12: false, hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
|
const accountSuffix = connectionId ? ` | account=${connectionId.slice(0, 8)}...` : "";
|
|
console.log(`${COLORS.green}[${time}] 📊 [${label}] ${provider.toUpperCase()} | in=${inTokens} | out=${outTokens}${accountSuffix}${COLORS.reset}`);
|
|
}
|
|
|
|
// Canonicalize to one storage convention (prompt_tokens cache-inclusive) so
|
|
// cached/cache-creation tokens survive to cost calc + stats. See canonicalizeUsage.
|
|
const normalized = canonicalizeUsage(tokens) || {
|
|
prompt_tokens: tokens.prompt_tokens ?? tokens.input_tokens ?? 0,
|
|
completion_tokens: tokens.completion_tokens ?? tokens.output_tokens ?? 0
|
|
};
|
|
|
|
saveRequestUsage({
|
|
provider: provider || "unknown",
|
|
model: model || "unknown",
|
|
tokens: normalized,
|
|
timestamp: new Date().toISOString(),
|
|
connectionId: connectionId || undefined,
|
|
apiKey: apiKey || undefined,
|
|
endpoint: endpoint || null
|
|
}).catch(() => {});
|
|
}
|