Merge remote-tracking branch 'origin/master' into gitea/new_feature

# Conflicts:
#	open-sse/executors/qoder.js
#	open-sse/handlers/chatCore.js
#	open-sse/handlers/chatCore/sseToJsonHandler.js
#	open-sse/providers/registry/commandcode.js
#	src/app/(dashboard)/dashboard/combos/page.js
#	src/app/api/v1/models/route.js
#	src/lib/db/repos/usageRepo.js
#	src/shared/components/UsageStats.js
This commit is contained in:
2026-09-25 10:25:56 +07:00
154 changed files with 10246 additions and 1241 deletions

View File

@@ -11,6 +11,10 @@ export function toOpenAIFinish(reason, format) {
case CLAUDE_STOP.MAX_TOKENS: return OPENAI_FINISH.LENGTH;
case CLAUDE_STOP.TOOL_USE: return OPENAI_FINISH.TOOL_CALLS;
case CLAUDE_STOP.STOP_SEQUENCE: return OPENAI_FINISH.STOP;
// A refusal is a blocked turn, not a clean stop: with the default mapping an
// OpenAI client saw finish_reason "stop" and an empty message (9Router logged
// "succeeded", OUT 0) and could not tell it from a real answer.
case CLAUDE_STOP.REFUSAL: return OPENAI_FINISH.CONTENT_FILTER;
default: return OPENAI_FINISH.STOP;
}
case "commandcode":
@@ -55,6 +59,7 @@ export function fromOpenAIFinish(reason, format) {
case OPENAI_FINISH.STOP: return CLAUDE_STOP.END_TURN;
case OPENAI_FINISH.LENGTH: return CLAUDE_STOP.MAX_TOKENS;
case OPENAI_FINISH.TOOL_CALLS: return CLAUDE_STOP.TOOL_USE;
case OPENAI_FINISH.CONTENT_FILTER: return CLAUDE_STOP.REFUSAL;
default: return CLAUDE_STOP.END_TURN;
}
default:

View File

@@ -14,9 +14,6 @@ const STRIP_RULES = [
{ provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] },
// Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926)
{ provider: "cloudflare-ai", flattenContent: true },
// MiMo Desktop Preview models (account-service route): content must be plain string,
// rejects OpenAI content-part array. Cloud models keep their parts (mimo-v2-omni is multi-modal).
{ provider: "xiaomi-mimo", match: /preview/i, flattenContent: true },
{ provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true },
// VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's
// advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144),
@@ -24,6 +21,15 @@ const STRIP_RULES = [
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
// min() with the model ceiling still applies if a variant's own limit is lower.
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
// Strict OpenAI-compatible validators reject unknown assistant-message fields.
// Clients that talk to reasoning models (e.g. Hermes) echo the prior turn's
// reasoning back on every assistant message; Groq answers 400 and Mistral 422
// ("extra_forbidden") on it, which knocks these providers out of every
// multi-turn combo. Providers that *require* the field (DeepSeek, Kimi) are
// handled by reasoningContentInjector and are not listed here.
{ provider: "groq", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
{ provider: "mistral", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
{ provider: "cerebras", dropMessageFields: ["reasoning_content", "reasoning", "reasoning_details"] },
];
// Test a rule's match (regex or predicate) against the model id.
@@ -47,6 +53,15 @@ export function stripUnsupportedParams(provider, model, body) {
for (const key of rule.drop || []) {
if (body[key] !== undefined) delete body[key];
}
// Per-message field drop (assistant turns only — that is where clients replay reasoning).
if (Array.isArray(rule.dropMessageFields) && Array.isArray(body.messages)) {
for (const msg of body.messages) {
if (!msg || msg.role !== "assistant") continue;
for (const key of rule.dropMessageFields) {
if (msg[key] !== undefined) delete msg[key];
}
}
}
// CF Workers AI oneOf root schema only accepts content as plain string (#1926)
if (rule.flattenContent && Array.isArray(body.messages)) {
for (const msg of body.messages) {

View File

@@ -34,6 +34,8 @@ export function effortToThinkingLevel(effort) {
// Numeric budget → nearest discrete level (reverse map via thresholds).
// Returns null when budget <= 0 (no reasoning).
// Thresholds are midpoints between LEVEL_TO_BUDGET values: max (128000) is
// reachable, with the xhigh/max boundary at the 32768/128000 midpoint (80384).
export function budgetToLevel(budget) {
const b = Number(budget);
if (!b || b <= 0) return null;
@@ -41,7 +43,8 @@ export function budgetToLevel(budget) {
if (b <= 4096) return "low";
if (b <= 16384) return "medium";
if (b <= 28672) return "high";
return "xhigh";
if (b <= 80384) return "xhigh";
return "max";
}
// Gemini thinkingBudget (numeric) → OpenAI reasoning_effort (antigravity reverse map).

View File

@@ -2,6 +2,7 @@ import { FORMATS } from "./formats.js";
import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js";
import { prepareClaudeRequest } from "./formats/claude.js";
import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js";
import { restoreToolNames } from "../utils/opencodeFingerprint.js";
import { filterToOpenAIFormat } from "./formats/openai.js";
import { normalizeThinkingConfig } from "../services/provider.js";
import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js";
@@ -166,7 +167,7 @@ export function translateResponse(targetFormat, sourceFormat, chunk, state) {
// even when no format conversion is needed, so streamed tool_use blocks must
// be decloaked here or the client sees an unknown ("_ide"-suffixed) tool.
if (sourceFormat === targetFormat) {
return [decloakStreamChunk(chunk, state?.toolNameMap)];
return [restoreToolNames(decloakStreamChunk(chunk, state?.toolNameMap), state?.toolNameMap)];
}
let results = [chunk];
@@ -179,7 +180,8 @@ export function translateResponse(targetFormat, sourceFormat, chunk, state) {
const directFn = responseRegistry.get(`${targetFormat}:${sourceFormat}`);
if (directFn) {
const converted = directFn(chunk, state);
return converted ? (Array.isArray(converted) ? converted : [converted]) : [];
const directResults = converted ? (Array.isArray(converted) ? converted : [converted]) : [];
return restoreToolNames(directResults, state?.toolNameMap);
}
// Step 1: target -> openai (if target is not openai)
@@ -210,6 +212,8 @@ export function translateResponse(targetFormat, sourceFormat, chunk, state) {
}
}
results = restoreToolNames(results, state?.toolNameMap);
// Attach OpenAI intermediate results for logging
if (openaiResults && sourceFormat !== FORMATS.OPENAI && targetFormat !== FORMATS.OPENAI) {
results._openaiIntermediate = openaiResults;

View File

@@ -279,10 +279,11 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra
}
};
// Antigravity specific fields
if (isAntigravity) {
envelope.requestType = "agent";
} else {
// Antigravity specific fields.
// NOTE: the official Antigravity client omits `requestType` entirely on the
// agent (chat) path. Sending `requestType: "agent"` triggers a detail-free
// 429 RESOURCE_EXHAUSTED even with quota available.
if (!isAntigravity) {
// Keep safetySettings for Gemini CLI
envelope.request.safetySettings = geminiCLI.safetySettings;
}
@@ -305,7 +306,8 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
model: model,
userAgent: "antigravity",
requestId: `agent-${generateUUID()}`,
requestType: "agent",
// NOTE: official Antigravity client omits `requestType` on the agent (chat)
// path — see the note in wrapInCloudCodeEnvelope() above.
request: {
sessionId: toNumericSessionId(credentials?._clientSessionId) || deriveSessionId(credentials?.email || credentials?.connectionId),
contents: [],

View File

@@ -149,6 +149,13 @@ export function claudeToOpenAIResponse(chunk, state) {
if (chunk.delta?.stop_reason) {
state.finishReason = convertStopReason(chunk.delta.stop_reason);
// A refusal produces no content blocks at all. Surface Anthropic's own
// explanation as the message text so the client shows *why* the turn is
// empty instead of a blank reply.
const refusalNote = chunk.delta.stop_reason === "refusal" && chunk.delta.stop_details?.explanation;
if (refusalNote) {
results.push(createChunk(state, { content: refusalNote }));
}
const finalChunk = createChunk(state, {}, state.finishReason);
if (state.usage) {

View File

@@ -14,13 +14,47 @@ import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } fro
* Translate OpenAI chunk to Responses API events
* @returns {Array} Array of events with { event, data } structure
*/
// Upstream Chat Completions usage -> Responses API usage shape.
// Without this, /v1/responses never reports usage: Responses clients (Codex CLI)
// keep their "context used" gauge pinned at 0 and never auto-compact, so a long
// session grows until the upstream context limit rejects it (9router issue #3432).
//
// Note this is stored under state.responsesUsage, NOT state.usage: state.usage is
// owned by the stream layer, which fills it with normalizeUsage()-shaped counts
// (prompt_tokens/prompt_tokens_details) and hands it to finalizeStream() for
// logging and cost accounting. Overwriting it with this shape silently drops
// cached/reasoning tokens from those stats.
function toResponsesUsage(usage) {
if (!usage || typeof usage !== "object") return null;
const inputTokens = [usage.input_tokens, usage.prompt_tokens].find(Number.isFinite) ?? 0;
const outputTokens = [usage.output_tokens, usage.completion_tokens].find(Number.isFinite) ?? 0;
const responseUsage = {
input_tokens: inputTokens,
output_tokens: outputTokens,
total_tokens: Number.isFinite(usage.total_tokens) ? usage.total_tokens : inputTokens + outputTokens
};
const cachedTokens = [usage.input_tokens_details?.cached_tokens, usage.prompt_tokens_details?.cached_tokens].find(Number.isFinite);
const reasoningTokens = [usage.output_tokens_details?.reasoning_tokens, usage.completion_tokens_details?.reasoning_tokens].find(Number.isFinite);
if (Number.isFinite(cachedTokens)) responseUsage.input_tokens_details = { cached_tokens: cachedTokens };
if (Number.isFinite(reasoningTokens)) responseUsage.output_tokens_details = { reasoning_tokens: reasoningTokens };
return responseUsage;
}
export function openaiToOpenAIResponsesResponse(chunk, state) {
if (!chunk) {
return flushEvents(state);
}
// Capture upstream usage BEFORE the choices guard below: the last OpenAI chunk
// may carry usage together with an empty choices array, and it must not be dropped.
if (chunk.usage) {
state.responsesUsage = toResponsesUsage(chunk.usage);
}
if (!chunk.choices?.length) return [];
const events = [];
const nextSeq = () => ++state.seq;
@@ -112,7 +146,19 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
for (const i in state.msgItemAdded) closeMessage(state, emit, i);
closeReasoning(state, emit);
for (const i in state.funcCallIds) closeToolCall(state, emit, i);
sendCompleted(state, emit);
// Upstreams report usage either on the finish chunk itself or on a trailing chunk
// whose `choices` array is empty (OpenAI does the latter). Emitting
// response.completed here would freeze the payload before that trailing chunk is
// parsed, so when usage is not known yet we leave completion to flushEvents(),
// which runs once the upstream stream ends and by then has seen every chunk.
//
// That only holds on the direct openai:openai-responses route. When this converter
// runs as the second hop of a pivot (Claude/Gemini/Kiro upstream), translateResponse()
// drops the terminal null chunk before reaching us — the first hop returns null for
// it, leaving nothing to iterate — so flushEvents() is never called and deferring
// would swallow the terminal event entirely. Keep the old behaviour there.
const flushReachesUs = state.targetFormat === FORMATS.OPENAI;
if (state.responsesUsage || !flushReachesUs) sendCompleted(state, emit);
}
return events;
@@ -376,7 +422,8 @@ function sendCompleted(state, emit) {
created_at: state.created,
status: "completed",
background: false,
error: null
error: null,
...(state.responsesUsage ? { usage: state.responsesUsage } : {})
}
});
}

View File

@@ -14,6 +14,9 @@ export const CLAUDE_STOP = {
MAX_TOKENS: "max_tokens",
TOOL_USE: "tool_use",
STOP_SEQUENCE: "stop_sequence",
// Anthropic's API-level refusal (streaming classifier / ToS). Arrives in
// message_delta with zero output tokens; stop_details carries the reason.
REFUSAL: "refusal",
};
// Gemini finishReason values.