Merge origin/master (v0.5.91) into gitea/new_feature

Resolve conflicts:
- streamingHandler.js: adopt upstreamResponseHeaders while keeping 0-token detail row avoidance
- capabilities.js: preserve user-asserted caps and globalThis slots without local caching of catalogSource
- AddCustomModelModal.js & providers/[id]/page.js: wire STT transport marker with custom model edits/assertions
- models/custom/route.js & aliasRepo.js: persist custom model transport and invalidate user caps
- usageRepo.js: key byApiKey live stats by full API key and keep tail in maskApiKey
- UsageStats.js: lazy load charts dynamically
This commit is contained in:
2026-09-28 21:17:33 +07:00
123 changed files with 5882 additions and 411 deletions

View File

@@ -178,7 +178,7 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C
// makes the backend flag the request and answer 429 Quota Exhausted.
export const ANTIGRAVITY_PROMPT_REWRITES = [
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
{ from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." },
{ from: /You are Hermes(?: Agent)?(?:,\s*(?:an intelligent AI assistant|an AI assistant|an AI agent))?(?:,?\s*(?:built|created)\s+by\s+Nous Research)?\./gi, to: "You are an AI assistant." },
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.

View File

@@ -3,10 +3,13 @@ import REGISTRY from "../providers/registry/index.js";
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
import { PROVIDER_MODELS } from "../providers/index.js";
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel } from "../providers/models/helpers.js";
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel, opencodeFamilyFormats } from "../providers/models/helpers.js";
import { FORMATS } from "../translator/formats.js";
export { PROVIDER_MODELS };
// OpenCode providers sharing the endpoint-family fallback for unknown model ids
const isOpenCodeAlias = (aliasOrId) => !aliasOrId || ["oc", "opencode", "ocg", "opencode-go", "ocz", "opencode-zen"].includes(aliasOrId);
// Helper functions
export function getProviderModels(aliasOrId) {
@@ -53,20 +56,29 @@ export function findModelName(aliasOrId, modelId) {
}
export function getModelTargetFormat(aliasOrId, modelId) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go" || aliasOrId === "ocz" || aliasOrId === "opencode-zen") && isMuseSparkModel(modelId)) {
if (isOpenCodeAlias(aliasOrId) && isMuseSparkModel(modelId)) {
return FORMATS.OPENAI_RESPONSES;
}
const models = PROVIDER_MODELS[aliasOrId];
if (!models) return null;
return modelTargetFormat(findModel(models, modelId, aliasOrId));
const found = findModel(models, modelId, aliasOrId);
if (found) return modelTargetFormat(found);
// Family fallback keeps modelsFetcher/passthrough ids on their endpoint lane
if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.targetFormat || null;
return null;
}
// Declared upstream formats for a model (registry `supportedFormats`). Drives the
// per-model guard on the sourceFormat-matched transport; null when undeclared.
// Unknown OpenCode ids fall back to the family regex (chat lane by default) so
// auto-fetched models never wrongly use the sourceFormat-matched transport.
export function getModelSupportedFormats(aliasOrId, modelId) {
const models = PROVIDER_MODELS[aliasOrId];
if (!models) return null;
return modelSupportedFormats(findModel(models, modelId, aliasOrId));
const found = findModel(models, modelId, aliasOrId);
if (found) return modelSupportedFormats(found);
if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.supportedFormats || [FORMATS.OPENAI];
return null;
}
export function getModelType(aliasOrId, modelId) {

View File

@@ -7,7 +7,7 @@ import {
} from "../services/oauthCredentialManager.js";
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
import { getModelUpstreamId } from "../config/providerModels.js";
import { getModelUpstreamId, getProviderModels } from "../config/providerModels.js";
import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
import { dbg } from "../utils/debugLog.js";
@@ -25,6 +25,10 @@ const CODEX_SSE_USER_OUTPUT_PATTERNS = [
];
const CODEX_SSE_PEEK_BYTES = 256 * 1024;
const CODEX_MODEL_CAPACITY_MESSAGE = "Selected model is at capacity. Please try a different model.";
function isCodexResponsesLiteModel(model) {
const baseId = String(model || "").replace(/\([^()]+\)\s*$/, "");
return getProviderModels("cx").some((entry) => entry.id === baseId && entry.responsesLite === true);
}
// Server-generated item id prefixes that Codex /responses cannot resolve when store=false
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
@@ -43,7 +47,7 @@ const CODEX_PASSTHROUGH_TOOL_TYPES = new Set(["custom"]);
const RESPONSES_API_ALLOWLIST = new Set([
"model", "input", "instructions", "tools", "tool_choice", "stream", "store",
"reasoning", "service_tier", "include", "prompt_cache_key", "client_metadata",
"text"
"text", "parallel_tool_calls"
]);
// Convert role=system → role=developer in body.input (keeps content in cacheable prefix)
@@ -57,13 +61,14 @@ function convertSystemToDeveloperRole(body) {
}
// Strip server-generated item IDs (rs_/fc_/resp_/msg_) from input — avoids 404 with store=false
function stripStoredItemReferences(body) {
function stripStoredItemReferences(body, preserveLitePrefix = false) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (typeof item === "string" && SERVER_ID_PATTERN.test(item)) return false;
if (item && typeof item === "object" && !Array.isArray(item)) {
if (item.type === "item_reference") return false;
if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)) delete item.id;
if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)
&& !(preserveLitePrefix && item.role === "developer" && item.id.startsWith("msg_"))) delete item.id;
}
return true;
});
@@ -138,6 +143,7 @@ function resolveCacheSessionId(body, credentials) {
function normalizeReasoningEffort(model, value) {
const supportedLevels = getThinkingLevels("codex", model);
if (supportedLevels?.includes(value)) return value;
if (isCodexResponsesLiteModel(model) && (value === "none" || value === "minimal")) return "low";
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
if (value === "max" || value === "ultra") return "xhigh";
return value;
@@ -209,8 +215,11 @@ export class CodexExecutor extends BaseExecutor {
* Override headers to add codex-specific identity headers.
* transformRequest runs BEFORE buildHeaders, sets this._currentSessionId.
*/
buildHeaders(credentials, stream = true) {
buildHeaders(credentials, stream = true, _url = null, model = null) {
const headers = super.buildHeaders(credentials, stream);
if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model))) {
headers["x-openai-internal-codex-responses-lite"] = "true";
}
headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default";
// Identify client type to Codex backend (matches official codex CLI)
if (!headers["originator"]) headers["originator"] = "codex_cli_rs";
@@ -408,6 +417,8 @@ export class CodexExecutor extends BaseExecutor {
// Convert string input to array format (Codex API requires input as array)
const normalized = normalizeResponsesInput(body.input);
if (normalized) body.input = normalized;
const upstreamModel = getModelUpstreamId("cx", body.model || model);
const responsesLite = isCodexResponsesLiteModel(upstreamModel);
// Ensure input is present and non-empty (Codex API rejects empty input)
if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) {
@@ -417,7 +428,7 @@ export class CodexExecutor extends BaseExecutor {
// Keep system prompts in body.input as role=developer so they stay in the cacheable prefix
convertSystemToDeveloperRole(body);
// Strip server-generated item IDs (rs_/fc_/resp_/msg_) — Codex /responses can't resolve when store=false
stripStoredItemReferences(body);
stripStoredItemReferences(body, responsesLite);
// Flatten function tools + drop unsupported types
normalizeCodexTools(body);
@@ -425,7 +436,7 @@ export class CodexExecutor extends BaseExecutor {
body.stream = true;
// If no instructions provided, inject default Codex instructions
if (!body.instructions || body.instructions.trim() === "") {
if (!responsesLite && (!body.instructions || body.instructions.trim() === "")) {
body.instructions = CODEX_DEFAULT_INSTRUCTIONS;
}
@@ -438,7 +449,29 @@ export class CodexExecutor extends BaseExecutor {
}
// Map virtual Codex review models to the upstream Codex model before suffix parsing.
body.model = getModelUpstreamId("cx", body.model || model);
body.model = upstreamModel;
if (responsesLite) {
// Codex 0.155 carries tools and instructions as input prefix items.
const input = Array.isArray(body.input) ? body.input : [body.input];
const hasLitePrefix = input.some((item) => item?.type === "additional_tools");
if (!hasLitePrefix) {
const instructions = typeof body.instructions === "string" && body.instructions.trim()
? body.instructions : CODEX_DEFAULT_INSTRUCTIONS;
const prefix = [{ type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] }];
if (instructions) {
prefix.push({ type: "message", role: "developer", content: [{ type: "input_text", text: instructions }] });
}
input.unshift(...prefix);
}
body.input = input;
body.instructions = "";
body.tools = null;
body.tool_choice ||= "auto";
body.parallel_tool_calls = false;
} else {
delete body.parallel_tool_calls;
}
// Extract thinking level from model name suffix
// e.g., gpt-5.3-codex-high → high, gpt-5.3-codex → medium (default)
@@ -455,12 +488,13 @@ export class CodexExecutor extends BaseExecutor {
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
if (!body.reasoning) {
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || 'low');
body.reasoning = { effort, summary: "auto" };
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || (responsesLite ? 'medium' : 'low'));
body.reasoning = responsesLite ? { effort } : { effort, summary: "auto" };
} else {
body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort);
if (!body.reasoning.summary) body.reasoning.summary = "auto";
if (!responsesLite && !body.reasoning.summary) body.reasoning.summary = "auto";
}
if (responsesLite) body.reasoning.context = "all_turns";
delete body.reasoning_effort;
// Include reasoning encrypted content (required by Codex backend for reasoning models)

View File

@@ -148,7 +148,7 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
const reader = originalResponse.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
const bufferedLines = [];
const rawChunks = [];
let detectedError = null;
try {
@@ -162,16 +162,15 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
const parsed = JSON.parse(jsonStr);
if (parsed?.type === "error") {
detectedError = parsed;
} else {
bufferedLines.push(trimmed);
}
} catch {
bufferedLines.push(trimmed);
/* ignore */
}
}
break;
}
rawChunks.push(value);
buffer += decoder.decode(value, { stream: true });
const lines = buffer.split("\n");
buffer = lines.pop() || "";
@@ -182,7 +181,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
if (!trimmed) continue;
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
if (!jsonStr || jsonStr === "[DONE]") {
bufferedLines.push(trimmed);
stopLoop = true;
break;
}
@@ -191,7 +189,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
try {
event = JSON.parse(jsonStr);
} catch {
bufferedLines.push(trimmed);
continue;
}
@@ -201,8 +198,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
break;
}
bufferedLines.push(trimmed);
if (
event?.type === "text-delta" ||
event?.type === "reasoning-delta" ||
@@ -245,29 +240,18 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
);
}
const combinedStream = createReplayedStream(bufferedLines, buffer, reader);
const combinedStream = createRawReplayedStream(rawChunks, reader);
return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse);
}
function createReplayedStream(bufferedLines, remainingBuffer, reader) {
const encoder = new TextEncoder();
let replayed = false;
function createRawReplayedStream(rawChunks, reader) {
let chunkIndex = 0;
return new ReadableStream({
async pull(controller) {
if (!replayed) {
replayed = true;
let prefix = bufferedLines.join("\n");
if (prefix && remainingBuffer) {
prefix += "\n" + remainingBuffer;
} else if (remainingBuffer) {
prefix = remainingBuffer;
} else if (prefix) {
prefix += "\n";
}
if (prefix) {
controller.enqueue(encoder.encode(prefix));
}
if (chunkIndex < rawChunks.length) {
controller.enqueue(rawChunks[chunkIndex++]);
return;
}
try {

View File

@@ -1,12 +1,13 @@
import { BaseExecutor } from "./base.js";
import { PROVIDERS, PROVIDER_OAUTH } from "../config/providers.js";
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta } from "../providers/shared.js";
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta, mergeAnthropicBeta } from "../providers/shared.js";
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
import { OAUTH_ENDPOINTS, buildKimiHeaders } from "../config/appConstants.js";
import { buildClineHeaders } from "../shared/clineAuth.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
import { extractClaudeSessionIdFromUserId } from "../utils/claudeCloaking.js";
// Auth header descriptors — derived from registry transport.auth, fallback to hardcoded defaults.
const BEARER = { combined: true, header: "Authorization", scheme: "bearer" };
@@ -164,9 +165,21 @@ export class DefaultExecutor extends BaseExecutor {
// a node fronting Kimi or GLM answers on its own ids and never matches, so
// gateways that would choke on unknown beta flags are left untouched.
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
const clientBeta = credentials?.rawHeaders?.["anthropic-beta"];
if (model && (this.provider === "claude"
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
headers["Anthropic-Beta"] = selectAnthropicBeta(model, body);
headers["Anthropic-Beta"] = mergeAnthropicBeta(selectAnthropicBeta(model, body), clientBeta);
} else if (this.provider === "anthropic" && clientBeta) {
headers["Anthropic-Beta"] = mergeAnthropicBeta(headers["Anthropic-Beta"], clientBeta);
}
// Claude OAuth: align x-claude-code-session-id with metadata.user_id.session_id if missing
if (this.provider === "claude" && !headers["x-claude-code-session-id"]) {
const token = credentials?.accessToken || credentials?.apiKey || "";
if (token.includes("sk-ant-oat")) {
const sid = extractClaudeSessionIdFromUserId(body?.metadata?.user_id);
if (sid) headers["x-claude-code-session-id"] = sid;
}
}
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams

View File

@@ -1,8 +1,8 @@
import crypto from "node:crypto";
import { DefaultExecutor } from "./default.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { modelTargetFormat } from "../providers/models/schema.js";
import { getProviderModels } from "../config/providerModels.js";
import { getModelTargetFormat } from "../config/providerModels.js";
import { FORMATS } from "../translator/formats.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
@@ -41,16 +41,10 @@ function translatedSession(sessionId, clientTool) {
return `ses_${digest}`;
}
// Strip the thinking suffix "model(level)" so checks hit the base id.
function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …).
// Reading the registry keeps this in sync with config — never hardcode model ids here.
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …),
// including the family-regex fallback for passthrough ids — never hardcode model ids here.
function isResponsesModel(model) {
const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model));
return modelTargetFormat(entry) === "openai-responses";
return getModelTargetFormat("opencode-go", model) === FORMATS.OPENAI_RESPONSES;
}
// Flatten Chat Completions tool declarations into the Responses flat shape and

View File

@@ -9,6 +9,7 @@ import { createRequestLogger } from "../utils/requestLogger.js";
import { getModelTargetFormat, getModelSupportedFormats, getModelStrip, getModelUpstreamId, getModelType, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
import { PROVIDERS } from "../config/providers.js";
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
import { upstreamResponseHeaders } from "../utils/upstreamHeaders.js";
import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js";
import { handleBypassRequest } from "../utils/bypassHandler.js";
import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js";
@@ -493,7 +494,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
log.errorLine(reqTag, "✗", `ERROR ${statusCode} · ${provider}/${model} · ${Date.now() - requestStartTime}ms${urlStr}\n ${errMsg}`);
}
reqLogger.logError(new Error(message), finalBody || translatedBody);
return createErrorResult(statusCode, errMsg, resetsAtMs);
return createErrorResult(statusCode, errMsg, resetsAtMs, upstreamResponseHeaders(providerResponse.headers));
}
const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers

View File

@@ -4,6 +4,7 @@ import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
import { createErrorResult } from "../../utils/error.js";
import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js";
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js";
@@ -417,7 +418,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
return {
success: true,
response: new Response(JSON.stringify(restoreToolNames(translatedResponse, toolNameMap)), {
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" }
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*", ...upstreamResponseHeaders(providerResponse.headers) }
})
};
}

View File

@@ -20,6 +20,7 @@ import {
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
import { saveRequestDetail } from "@/lib/usageDb.js";
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js";
// Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat.
// Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI.
@@ -158,7 +159,12 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
return {
success: true,
response: new Response(transformedBody, { headers: SSE_HEADERS }),
response: new Response(transformedBody, {
headers: {
...SSE_HEADERS,
...upstreamResponseHeaders(providerResponse.headers),
},
}),
};
}

View File

@@ -0,0 +1,266 @@
import { Buffer } from "node:buffer";
// Gemini Live API realtime STT transport.
//
// The REST generateContent path (sttCore.transcribeGemini) only transcribes
// whole files inline. The Live API's `:bidiGenerateContent` WebSocket is the
// streaming counterpart: audio goes up as realtimeInput mediaChunks and the
// server pushes incremental `serverContent.inputTranscription` events back.
// This module owns the socket lifecycle only — envelope/response shaping
// stays in sttCore so the engine's single STT exit shape is preserved.
//
// Marker contract: dispatched from sttCore's format-switch when the model
// entry carries `transport: "gemini-live"` (registry) or the caller passes a
// transport string (custom models). Never keyed on a hardcoded model id here.
//
// Transport behavior:
// - Node >= 22 global WebSocket (undici). No new dependency.
// - Live API expects low-latency PCM; other containers are forwarded with
// their declared MIME unchanged (provider-side rejection is surfaced).
// - Text accumulation is append-only over inputTranscription segments and
// ends on serverContent.turnComplete (or graceful close with partial text).
// - Transcription deltas are kept per-frame (chunks[]) so sttCore can shape
// verbose_json segments without fabricating timestamps. goAway advisements
// rotate the socket once per call: setup replay + byte-offset resume.
const SETUP_TIMEOUT_MS = 10_000; // open → setupComplete
const TURN_TIMEOUT_MS = 60_000; // audio streamed → turnComplete
const MAX_TIMEOUT_MS = 300_000; // clamp ceiling for client-supplied lifecycle knobs
const CHUNK_BYTES = 16_384; // ~0.5s of 16-bit 16kHz mono PCM
const GOAWAY_RECONNECTS = 1; // socket rotations honoured per call
class GeminiLiveError extends Error {
constructor(message, status) {
super(message);
this.name = "GeminiLiveError";
this.status = status || 502;
}
}
// REST base (https://host/v1beta/models) → Live WS base
// (wss://host/ws/api/v1beta/models), then the bidiGenerateContent endpoint.
function toLiveWsUrl(baseUrl, model, token) {
const url = new URL(baseUrl);
url.protocol = "wss:";
if (!url.pathname.startsWith("/ws/")) url.pathname = `/ws/api${url.pathname}`;
const base = url.toString().replace(/\/+$/, "");
return `${base}/${encodeURIComponent(model)}:bidiGenerateContent?key=${encodeURIComponent(token || "")}`;
}
// Bind socket events supporting BOTH handler styles: addEventListener
// (browser WebSocket, undici) and onopen/onmessage property assignment
// (minimal polyfills). Whichever the implementation exposes, it works.
function bindSocket(ws, { onOpen, onMessage, onError, onClose }) {
if (typeof ws.addEventListener === "function") {
ws.addEventListener("open", onOpen);
ws.addEventListener("message", onMessage);
ws.addEventListener("error", onError);
ws.addEventListener("close", onClose);
return;
}
ws.onopen = onOpen;
ws.onmessage = onMessage;
ws.onerror = onError;
ws.onclose = onClose;
}
function parseFrame(data) {
try {
return JSON.parse(typeof data === "string" ? data : String(data));
} catch {
return null; // non-JSON frames carry no Live API semantics
}
}
function firstStringField(formData, key) {
const v = typeof formData?.get === "function" ? formData.get(key) : null;
return typeof v === "string" && v.trim() ? v.trim() : "";
}
// Lifecycle knobs the live registry entry advertises in params[]
// (setup/turn timeouts). They ride the same formData pass-through sttCore
// gives every transport — no sttCore change needed to reach this leaf.
function firstNumberField(formData, key, fallback) {
const n = Number(firstStringField(formData, key));
return Number.isFinite(n) && n > 0 ? Math.min(n, MAX_TIMEOUT_MS) : fallback;
}
/**
* Transcribe an audio File via the Gemini Live bidirectional stream.
* @returns {Promise<{text: string, chunks: string[]}>} transcript plus the raw
* incremental inputTranscription deltas (sttCore shapes verbose_json from them).
* @throws {GeminiLiveError} with .status for the sttCore error envelope.
*/
export async function transcribeGeminiLive({ cfg, file, model, token, formData, mimeType }) {
const WS = globalThis.WebSocket;
if (!WS) throw new GeminiLiveError("Gemini Live transport needs global WebSocket (Node >= 22)", 502);
const buf = Buffer.from(await file.arrayBuffer());
if (!buf.length) throw new GeminiLiveError("Empty audio file", 400);
const instruction = firstStringField(formData, "prompt") || "Transcribe the spoken audio verbatim.";
const language = firstStringField(formData, "language");
const setupTimeoutMs = firstNumberField(formData, "setup_timeout_ms", SETUP_TIMEOUT_MS);
const turnTimeoutMs = firstNumberField(formData, "turn_timeout_ms", TURN_TIMEOUT_MS);
// system_instruction (registry param) overrides the built-in transcription
// directive wholesale; prompt/language only shape the default.
const instructionOverride = firstStringField(formData, "system_instruction");
const systemText = instructionOverride
|| (language ? `${instruction} Language: ${language}.` : instruction);
const wsUrl = toLiveWsUrl(cfg.baseUrl, model, token);
return await new Promise((resolve, reject) => {
let text = "";
const chunks = []; // raw inputTranscription deltas, shaped by sttCore
let settled = false;
let timer = null;
let goAwayTimer = null;
let ws = null;
let generation = 0; // socket identity: superseded closes never settle
let sentBytes = 0; // audio prefix already handed to the live socket
let goAwayReconnects = GOAWAY_RECONNECTS;
const arm = (ms, message) => {
if (timer) clearTimeout(timer);
timer = setTimeout(() => fail(new GeminiLiveError(message, 504)), ms);
};
const shutdown = () => {
if (timer) { clearTimeout(timer); timer = null; }
if (goAwayTimer) { clearTimeout(goAwayTimer); goAwayTimer = null; }
// ws is null until the first open() dials (and stays null when the
// constructor throws) — fail() runs shutdown() on that path.
if (!ws) return;
try {
if (ws.readyState === WS.OPEN || ws.readyState === WS.CONNECTING) ws.close(1000);
} catch { /* socket already dead — outcome is already settled */ }
};
const succeed = () => {
if (settled) return;
settled = true;
shutdown();
resolve({ text, chunks });
};
const fail = (err) => {
if (settled) return;
settled = true;
shutdown();
reject(err);
};
const send = (frame) => {
if (ws.readyState !== WS.OPEN) return false;
try {
ws.send(JSON.stringify(frame));
} catch {
return false; // socket died mid-send — streamAudioAndPrompt maps this to a 502
}
return true;
};
// Streams every byte not yet sent, then the flushing text turn. After a
// goAway rotation this resumes from sentBytes — no audio re-upload.
const streamAudioAndPrompt = () => {
for (let off = sentBytes; off < buf.length; off += CHUNK_BYTES) {
const mediaChunk = buf.subarray(off, off + CHUNK_BYTES).toString("base64");
if (!send({ realtimeInput: { mediaChunks: [{ mimeType, data: mediaChunk }] } })) {
fail(new GeminiLiveError("Gemini Live socket closed while streaming audio", 502));
return;
}
sentBytes = Math.min(off + CHUNK_BYTES, buf.length);
}
// Final user turn: flushes the recognizer and yields turnComplete.
send({ clientContent: { turns: [{ parts: [{ text: systemText }] }], turnComplete: true } });
};
// goAway: the server names the instant it will force-close this socket.
// Graceful play = rotate BEFORE the deadline: retire the live socket,
// dial a fresh one, replay setup, resume audio from sentBytes — text and
// chunks survive the hop. Once the advisory budget is spent a later
// goAway is left to the close path, which settles on partial transcript.
const scheduleGoAwayReconnect = (goAway) => {
if (settled || goAwayTimer || goAwayReconnects <= 0) return;
const deadline = Date.parse(typeof goAway?.time === "string" ? goAway.time : "");
const delay = Number.isFinite(deadline)
? Math.max(0, Math.min(deadline - Date.now(), setupTimeoutMs))
: 0;
goAwayTimer = setTimeout(() => {
goAwayTimer = null;
if (settled) return;
goAwayReconnects--;
generation++;
try { ws?.close(1000); } catch { /* deadline crossed mid-flight — re-dial anyway */ }
open();
}, delay);
};
const open = () => {
const gen = ++generation;
try {
ws = new WS(wsUrl);
} catch {
fail(new GeminiLiveError("Gemini Live websocket connection failed", 502));
return;
}
bindSocket(ws, {
onOpen: () => {
if (settled || gen !== generation) return;
arm(setupTimeoutMs, "Gemini Live timed out waiting for setupComplete");
send({
setup: {
model: `models/${model}`,
generationConfig: {
responseModalities: ["TEXT"],
inputAudioTranscription: {},
},
systemInstruction: { parts: [{ text: systemText }] },
},
});
},
onMessage: (ev) => {
if (settled || gen !== generation) return;
const frame = parseFrame(ev?.data);
if (!frame) return;
if (frame.error) {
const e = frame.error;
fail(new GeminiLiveError(`Gemini Live error${e.status ? ` (${e.status})` : ""}: ${e.message || "unknown"}`, 502));
return;
}
if (frame.goAway) {
scheduleGoAwayReconnect(frame.goAway);
return;
}
const sc = frame.serverContent;
if (!sc) return;
const delta = typeof sc.inputTranscription?.text === "string" ? sc.inputTranscription.text : "";
// Trim before testing: a padding-only frame carries no transcript and
// must not make an empty run look like a partial success on close.
if (delta.trim()) {
text += delta;
chunks.push(delta);
}
if (sc.setupComplete) {
arm(turnTimeoutMs, "Gemini Live transcription timed out");
streamAudioAndPrompt();
return;
}
if (sc.turnComplete) succeed();
},
onError: () => {
if (settled || gen !== generation) return;
fail(new GeminiLiveError("Gemini Live websocket connection failed", 502));
},
onClose: (ev) => {
if (settled || gen !== generation) return;
// Partial transcript beats a hard error on graceful close; silence is one.
if (text.trim()) succeed();
else fail(new GeminiLiveError(`Gemini Live socket closed before completion${ev?.code ? ` (code ${ev.code})` : ""}`, 502));
},
});
};
open();
});
}

View File

@@ -1,5 +1,7 @@
import { Buffer } from "node:buffer";
import { createErrorResult } from "../utils/error.js";
import { transcribeGeminiLive } from "./geminiLiveStt.js";
import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js";
// Build auth headers from sttConfig + token
@@ -162,11 +164,26 @@ function jsonResponse(obj) {
};
}
// Model-level transport marker (registry models[].transport, e.g. the Gemini
// live STT entry's "gemini-live", or a custom model's stored transport).
// Dispatch reads the marker — never a hardcoded model id — so new realtime
// providers extend sttCore through data, not code.
function resolveModelTransport(provider, model) {
const key = PROVIDER_ID_TO_ALIAS[provider] || provider;
const models = PROVIDER_MODELS[key] || PROVIDER_MODELS[provider];
if (!Array.isArray(models)) return null;
const entry = models.find((m) => m && m.id === model && (m.kind || "llm") === "stt");
const marker = typeof entry?.transport === "string" ? entry.transport.trim() : "";
return marker || null;
}
/**
* STT core handler — dispatch by sttConfig.format.
* STT core handler — dispatch by model transport marker, else sttConfig.format.
* `transport` is the caller-supplied marker override (custom models resolve
* it in the app layer; built-ins fall back to the registry entry marker).
* @returns {Promise<{success, response, status?, error?}>}
*/
export async function handleSttCore({ provider, model, formData, credentials, sttConfig }) {
export async function handleSttCore({ provider, model, formData, credentials, sttConfig, transport }) {
const file = formData.get("file");
if (!file) return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: file");
@@ -186,8 +203,29 @@ export async function handleSttCore({ provider, model, formData, credentials, st
return createErrorResult(HTTP_STATUS.UNAUTHORIZED, `No credentials for STT provider: ${provider}`);
}
// Format-switch extension: an explicit caller marker wins over the registry
// marker; with neither, the provider-default sttConfig.format applies.
const marker = (typeof transport === "string" && transport.trim()) ? transport.trim() : resolveModelTransport(provider, model);
try {
switch (cfg.format) {
switch (marker || cfg.format) {
case "gemini-live": {
const live = await transcribeGeminiLive({ cfg, file, model, token, formData, mimeType: resolveAudioContentType(file) });
// response_format parity with the OpenAI-compatible transport: default
// envelope stays {text}; verbose_json adds segments mapped from the
// Live API's incremental inputTranscription deltas. Those frames carry
// NO timestamps, so segments expose {id,text} only (id = delta order,
// Whisper-compatible 0-based) — start/end/duration are deliberately
// absent rather than fabricated as zeros, which would misrepresent
// provider data to callers diffing transports.
const fmt = typeof formData?.get === "function"
? String(formData.get("response_format") ?? "").trim().toLowerCase()
: "";
if (fmt === "verbose_json") {
return jsonResponse({ text: live.text, segments: live.chunks.map((segText, id) => ({ id, text: segText })) });
}
return jsonResponse({ text: live.text });
}
case "deepgram": return await transcribeDeepgram(cfg, file, model, token, formData);
case "assemblyai": return await transcribeAssemblyAI(cfg, file, model, token);
case "nvidia-asr": return await transcribeNvidia(cfg, file, model, token);
@@ -196,6 +234,6 @@ export async function handleSttCore({ provider, model, formData, credentials, st
default: return await transcribeOpenAICompatible(cfg, file, model, token, formData);
}
} catch (err) {
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed");
return createErrorResult(err.status || HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed");
}
}

View File

@@ -430,21 +430,29 @@ const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
*
* @param {string[]} comboModels
* @param {Object|null} [comboLookup] optional map of combo name → models array for nested resolution
* @param {Function|null} [resolveCaps] optional (fullId) → caps override. The synced model
* catalog is server-only (it reads a file), so a browser-side resolution cannot see the
* limits it supplies and silently falls back to the generic patterns below. Callers that
* have the server's answer (/api/models, via useModelCaps) pass it here; it is merged over
* the local tables, so fields it does not carry (tools, pdf, audio/video, thinking*) survive.
* @param {number} [_depth] internal recursion depth guard
* @returns {object|null} full capabilities object, or null for empty input
*/
export function aggregateComboCapabilities(comboModels, comboLookup = null, _depth = 0) {
export function aggregateComboCapabilities(comboModels, comboLookup = null, resolveCaps = null, _depth = 0) {
if (!comboModels?.length || _depth > 6) return null;
const allCaps = comboModels.map((fullId) => {
// Nested combo: bare name (no slash) that exists in the lookup — recurse
if (!fullId.includes("/") && comboLookup?.[fullId]) {
return aggregateComboCapabilities(comboLookup[fullId], comboLookup, _depth + 1)
return aggregateComboCapabilities(comboLookup[fullId], comboLookup, resolveCaps, _depth + 1)
?? resolveCaps?.(fullId)
?? getCapabilitiesForModel(null, fullId);
}
const slash = fullId.indexOf("/");
const provider = slash === -1 ? null : fullId.slice(0, slash);
const model = slash === -1 ? fullId : fullId.slice(slash + 1);
return getCapabilitiesForModel(provider, model);
const local = getCapabilitiesForModel(provider, model);
const override = resolveCaps?.(fullId);
return override ? { ...local, ...override } : local;
});
const first = allCaps[0];
return {
@@ -482,7 +490,9 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
// handlers (silently: the setters still "succeed"). The slots therefore live on
// globalThis, which IS shared across server bundles in the same process.
// Same reason the browser bundle is safe: it never calls a setter, so the slots
// stay empty and every consumer below short-circuits.
// stay empty and every consumer below short-circuits. Every read goes through
// globalThis: caching it locally would keep a reader alive in other copies after
// setCatalogSource(null).
let catalogSource = null;
const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
catalog: null, // { getModalities, getLimits } — synced models.dev catalog
@@ -496,15 +506,13 @@ const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
*/
export function setCatalogSource(source) {
catalogSource = source || null;
SOURCE_SLOTS.catalog = source || null;
if (SOURCE_SLOTS) SOURCE_SLOTS.catalog = source || null;
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source || null;
}
function getCatalogSource() {
if (catalogSource) return catalogSource;
if (SOURCE_SLOTS.catalog) return (catalogSource = SOURCE_SLOTS.catalog);
if (typeof globalThis === "undefined") return null;
return (catalogSource = globalThis.__9rCatalogSource || null);
if (typeof globalThis === "undefined") return catalogSource;
return SOURCE_SLOTS?.catalog || globalThis.__9rCatalogSource || null;
}
// Capabilities the user asserted per provider+model (dashboard "Add/Edit Model"

View File

@@ -1,3 +1,5 @@
import { FORMATS } from "../../translator/formats.js";
// Codex auto-generates a "-review" variant for each llm model (review quota family)
export const CODEX_REVIEW_SUFFIX = "-review";
@@ -25,3 +27,20 @@ export function isMuseSparkModel(modelId) {
const base = clean.includes("/") ? clean.split("/").pop() : clean;
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
}
// Endpoint families for OpenCode models outside the curated registry (modelsFetcher /
// passthrough ids) — regex keeps auto-fetched models on the right endpoint:
// /responses (gpt/grok/muse-spark), /messages (minimax/qwen), /chat/completions (rest).
// Curated registry entries always win; this is the unknown-id fallback only.
const OPENCODE_FAMILIES = [
{ match: /^(grok|gpt|muse[-_]?spark)/i, supportedFormats: [FORMATS.OPENAI_RESPONSES], targetFormat: FORMATS.OPENAI_RESPONSES },
{ match: /^deepseek-v4-(pro|flash)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE, FORMATS.OPENAI_RESPONSES] },
{ match: /^(minimax|qwen)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE] },
{ match: /^claude-/i, supportedFormats: [FORMATS.CLAUDE] },
];
export function opencodeFamilyFormats(modelId) {
if (!modelId || typeof modelId !== "string") return null;
const base = modelId.replace(/\([^()]+\)\s*$/, "").trim();
return OPENCODE_FAMILIES.find((f) => f.match.test(base)) || null;
}

View File

@@ -2,8 +2,28 @@
//
// Fallback order (first match wins):
// 1. PROVIDER_PRICING[provider][model] — provider-specific override
// 2. MODEL_PRICING[model] — canonical model price (provider-agnostic)
// 3. PATTERN_PRICING — glob pattern match (e.g. "codex-*")
// 2. FREE_MODEL_NAMESPACES — upstream bills these at $0
// 3. MODEL_PRICING[model] — canonical model price (provider-agnostic)
// 4. PATTERN_PRICING — glob pattern match (e.g. "codex-*")
/**
* Namespaces upstream meters at $0. A free model must never inherit a paid
* rate: the vendor-prefix strip in getPricingForModel() would turn
* "cline-free/deepseek-v4.1-flash" into "deepseek-v4.1-flash" and match
* MODEL_PRICING, so the namespace is checked before both fallbacks.
*/
export const FREE_MODEL_NAMESPACES = ["cline-free/"];
export const ZERO_PRICING = {
input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0,
};
/** True when the model id sits in a namespace upstream bills at $0. */
export function isFreeModel(model) {
if (!model) return false;
const lower = String(model).toLowerCase();
return FREE_MODEL_NAMESPACES.some((ns) => lower.startsWith(ns));
}
/**
* Canonical model pricing — provider-agnostic.
@@ -361,10 +381,11 @@ export function matchPattern(pattern, model) {
}
/**
* Resolve pricing for a model using the 3-step fallback chain:
* Resolve pricing for a model using the 4-step fallback chain:
* 1. PROVIDER_PRICING[provider][model]
* 2. MODEL_PRICING[model]
* 3. PATTERN_PRICING (glob match)
* 2. free namespace (upstream bills $0)
* 3. MODEL_PRICING[model]
* 4. PATTERN_PRICING (glob match)
*
* @param {string} provider
* @param {string} model
@@ -378,12 +399,15 @@ export function getPricingForModel(provider, model) {
return PROVIDER_PRICING[provider][model];
}
// 2. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat")
// 2. Free namespaces bill $0 regardless of the model name behind them.
if (isFreeModel(model)) return ZERO_PRICING;
// 3. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat")
const baseModel = model.includes("/") ? model.split("/").pop() : model;
if (MODEL_PRICING[baseModel]) return MODEL_PRICING[baseModel];
if (MODEL_PRICING[model]) return MODEL_PRICING[model];
// 3. Pattern match
// 4. Pattern match
for (const { pattern, pricing } of PATTERN_PRICING) {
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
return pricing;

View File

@@ -0,0 +1,29 @@
export default {
id: "agnes",
priority: 120,
alias: "agnes",
aliases: [
"agnes-ai",
],
uiAlias: "agnes",
display: {
name: "Agnes AI",
icon: "auto_awesome",
color: "#7C3AED",
textIcon: "AG",
website: "https://agnes-ai.com",
notice: {
text: "OpenAI-compatible gateway from Agnes AI, offering free API credits on sign-up. Accepts a bearer token or an x-api-key header.",
apiKeyUrl: "https://platform.agnes-ai.com",
},
},
category: "freeTier",
authType: "apikey",
transport: {
baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions",
validateUrl: "https://apihub.agnes-ai.com/v1/models",
},
// No model ids could be verified without a key, so discovery is left to the
// live endpoint and any id is accepted through passthroughModels.
passthroughModels: true,
};

View File

@@ -0,0 +1,32 @@
export default {
id: "atria",
priority: 120,
alias: "atria",
aliases: [
"atria-asi",
],
uiAlias: "atria",
display: {
name: "Atria Dawn",
icon: "flare",
color: "#C2410C",
textIcon: "AD",
website: "https://atria-asi.ai",
notice: {
text: "OpenAI-compatible endpoint from Atria Dawn (AtomInnoLab). Currently a research preview offering a single text model, Atria-Dawn-Preview.",
apiKeyUrl: "https://api.atria-asi.ai/dashboard",
},
},
category: "apikey",
authType: "apikey",
transport: {
baseUrl: "https://api.atria-asi.ai/v1/chat/completions",
validateUrl: "https://api.atria-asi.ai/v1/models",
},
// Docs pin the model field to one case-sensitive id. Text-only for now: the
// service ships a hook that blocks image/PDF input, so no vision is claimed.
models: [
{ id: "Atria-Dawn-Preview", name: "Atria Dawn Preview" },
],
passthroughModels: true,
};

View File

@@ -0,0 +1,30 @@
export default {
id: "bai",
priority: 120,
alias: "bai",
aliases: [
"b-ai",
],
uiAlias: "bai",
display: {
name: "B.AI",
icon: "account_balance",
color: "#0369A1",
textIcon: "BA",
website: "https://b.ai",
notice: {
text: "OpenAI-compatible gateway with one of the larger catalogues here. Accepts a bearer token or an x-api-key header. Model ids are fetched live from the provider.",
apiKeyUrl: "https://b.ai",
},
},
category: "apikey",
authType: "apikey",
transport: {
baseUrl: "https://api.b.ai/v1/chat/completions",
validateUrl: "https://api.b.ai/v1/models",
},
// No ids hardcoded: the catalogue is large and rotates, so the live endpoint
// is the source of truth and any id is accepted via passthroughModels.
modelsFetcher: { url: "https://api.b.ai/v1/models", type: "openai" },
passthroughModels: true,
};

View File

@@ -54,6 +54,8 @@ export default {
oauthUrl: "https://api.anthropic.com/api/oauth/usage",
orgUrl: "https://api.anthropic.com/v1/organizations/{org_id}/usage",
settingsUrl: "https://api.anthropic.com/v1/settings",
profileUrl: "https://api.anthropic.com/api/oauth/profile",
resetUrl: "https://api.anthropic.com/api/organizations/{org_id}/reset_rate_limits",
},
},
models: [

View File

@@ -2,7 +2,8 @@ import { withCodexReviewModels } from "../models/helpers.js";
// Codex CLI version seen by OpenAI's backend — single source for the Version /
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
const CODEX_CLI_VERSION = "0.154.0";
const CODEX_CLI_VERSION = "0.155.0";
const GPT_6_LITE_THINKING_LEVELS = ["low", "medium", "high", "xhigh", "max"];
export default {
id: "codex",
@@ -42,6 +43,7 @@ export default {
headers: {
originator: "codex_cli_rs",
"User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`,
version: CODEX_CLI_VERSION,
},
usage: {
url: "https://chatgpt.com/backend-api/wham/usage",
@@ -51,6 +53,8 @@ export default {
},
models: [
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
{ id: "gpt-6-sol", name: "GPT 6.0 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-6-luna", name: "GPT 6.0 Luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },

View File

@@ -0,0 +1,35 @@
export default {
id: "dahl",
priority: 120,
alias: "dahl",
aliases: [
"dahl-inference",
],
uiAlias: "dahl",
display: {
name: "Dahl Inference",
icon: "hub",
color: "#1E40AF",
textIcon: "DH",
website: "https://dahl.global",
notice: {
text: "OpenAI-compatible Gonka inference node. Small, fixed catalogue (GLM-5.3-Flash, DeepSeek-V4-Flash, MiniMax-M2.7) at a flat per-token rate.",
apiKeyUrl: "https://dahl.global/dashboard",
},
},
category: "apikey",
authType: "apikey",
transport: {
baseUrl: "https://inference.dahl.global/v1/chat/completions",
validateUrl: "https://inference.dahl.global/v1/models",
},
// The live catalogue is public (no auth), so modelsFetcher works without a key
// and the ids below are a convenience seed rather than an exhaustive list.
models: [
{ id: "zai-org/GLM-5.3-Flash", name: "GLM-5.3 Flash" },
{ id: "deepseek-ai/DeepSeek-V4-Flash-0731", name: "DeepSeek V4 Flash 0731" },
{ id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7" },
],
modelsFetcher: { url: "https://inference.dahl.global/v1/models", type: "openai" },
passthroughModels: true,
};

View File

@@ -58,6 +58,7 @@ export default {
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash", params: ["language","prompt"], kind: "stt" },
{ id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash Lite (Cheapest)", params: ["language","prompt"], kind: "stt" },
{ id: "gemini-2.0-flash", name: "Gemini 2.0 Flash", params: ["language","prompt"], kind: "stt" },
{ id: "gemini-2.5-flash-native-audio-preview-09-17", name: "Gemini Live Transcription (Realtime)", params: ["language","prompt","system_instruction","setup_timeout_ms","turn_timeout_ms"], kind: "stt", transport: "gemini-live" },
{ id: "gemini-3.1-flash-tts-preview", name: "Gemini 3.1 Flash TTS", kind: "tts" },
{ id: "gemini-2.5-flash-preview-tts", name: "Gemini 2.5 Flash TTS", kind: "tts" },
{ id: "gemini-2.5-pro-preview-tts", name: "Gemini 2.5 Pro TTS", kind: "tts" },

View File

@@ -125,6 +125,11 @@ import p119 from "./selfhosted-embedding.js";
import p120 from "./fish-audio.js";
import p121 from "./alitp-intl.js";
import p122 from "./xquik.js";
import p125 from "./tokenharbor.js";
import p126 from "./dahl.js";
import p127 from "./atria.js";
import p129 from "./agnes.js";
import p130 from "./bai.js";
export default [
p0,
p1,
@@ -250,4 +255,9 @@ export default [
p120,
p121,
p122,
p125,
p126,
p127,
p129,
p130,
];

View File

@@ -35,37 +35,56 @@ export default {
],
// supportedFormats follow the endpoint table in https://opencode.ai/docs/go/
models: [
{ id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] },
{ id: "deepseek-flash", name: "DeepSeek Flash", supportedFormats: ["openai"] },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "glm-5", name: "GLM 5", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "kimi-k2.5", name: "Kimi K2.5", supportedFormats: ["openai"] },
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] },
{ id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", supportedFormats: ["openai"] },
{ id: "mimo-v2.6-pro", name: "MiMo V2.6 Pro", supportedFormats: ["openai"] },
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
{ id: "mimo-v2-pro", name: "MiMo V2 Pro", supportedFormats: ["openai"] },
{ id: "mimo-v2-omni", name: "MiMo V2 Omni", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
{ id: "space-bunny-free", name: "Space Bunny Free", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.5-plus", name: "Qwen 3.5 Plus", supportedFormats: ["openai", "claude"] },
{ id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] },
{ id: "hy3", name: "Hy3", supportedFormats: ["openai"] },
{ id: "hy3-preview", name: "Hy3 Preview", supportedFormats: ["openai"] },
// In /zen/go/v1/models but absent from the docs endpoint table — chat lane is the fallback guess
{ id: "omen-alpha", name: "Omen Alpha", supportedFormats: ["openai"] },
// Served by /zen/go/v1/responses only — the responses-only entry forces chatCore
// past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "grok-4.7", name: "Grok 4.7", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "grok-4.5", name: "Grok 4.5", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-6-luna", name: "GPT 6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
],
// Live catalogue; ids outside this curated list get their endpoint lane from the
// family regex in providers/models/helpers.js (opencodeFamilyFormats).
modelsFetcher: { url: "https://opencode.ai/zen/go/v1/models", type: "opencode-go" },
passthroughModels: true,
features: {
usage: true,
usageApikey: true,

View File

@@ -0,0 +1,49 @@
export default {
id: "tokenharbor",
priority: 120,
alias: "tokenharbor",
aliases: [
"th",
"thh",
],
uiAlias: "tokenharbor",
display: {
name: "Token Harbor",
icon: "anchor",
color: "#0F766E",
textIcon: "TH",
website: "https://tokenharbor.ai",
notice: {
text: "OpenAI-compatible aggregator. One API key reaches every model, billed per-token from a prepaid wallet. Model ids are bare (e.g. claude-opus-5.5, gpt-6-astra, deepseek-v4.1-flash:free) and are fetched live from the provider.",
apiKeyUrl: "https://tokenharbor.ai/dashboard",
},
},
category: "apikey",
authType: "apikey",
transport: {
// OpenAI-compatible. `format` is left at the shared "openai" default and
// `thinkingFormat` is deliberately NOT declared: Token Harbor forwards
// requests verbatim, so each model must resolve its own thinking wire
// format through providers/capabilities.js. Setting a provider-wide value
// would force one format (e.g. claude-adaptive) onto every model.
baseUrl: "https://tokenharbor.ai/v1/chat/completions",
validateUrl: "https://tokenharbor.ai/v1/models",
retry: {
429: 2,
},
},
// Curated seed; the live catalogue is fetched via modelsFetcher and any other
// id is accepted via passthroughModels. Their catalogue rotates (the :free set
// in particular), so this stays deliberately small and is only the offline
// fallback. Ids are bare — Token Harbor does not prefix them by upstream vendor.
models: [
{ id: "claude-opus-5.5", name: "Claude Opus 5.5" },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "gpt-6-astra", name: "GPT-6 Astra" },
{ id: "gpt-6-sol", name: "GPT-6 Sol" },
{ id: "deepseek-v4.1-flash:free", name: "DeepSeek V4.1 Flash (Free)" },
{ id: "grok-4.7", name: "Grok 4.7" },
],
modelsFetcher: { url: "https://tokenharbor.ai/v1/models", type: "openai" },
passthroughModels: true,
};

View File

@@ -77,6 +77,11 @@ export function selectAnthropicBeta(model = "", body = null) {
return flags.join(",");
}
export function mergeAnthropicBeta(...values) {
const flags = values.flatMap((v) => (typeof v === "string" ? v.split(",") : [])).map((f) => f.trim()).filter(Boolean);
return [...new Set(flags)].join(",");
}
// Shared baseUrls
export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";

View File

@@ -3,6 +3,7 @@
import { getCapabilitiesForModel } from "./capabilities.js";
import { matchPattern } from "./pricing.js";
import { resolveKiroEffortPath } from "../config/kiroConstants.js";
import { getProviderModels } from "../config/providerModels.js";
// Shared level sets (deduped) — verified against provider docs + wire in thinkingUnified.applyFormat.
const L = {
@@ -42,6 +43,8 @@ const PATTERN_THINKING = [
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
{ pattern: "*mimo*v2.6*", levels: ["none", "low", "medium", "high", "xhigh"] },
// mimo-v2.5-pro on opencode-go rejects reasoning_effort "max" (probed live); v2.5 accepts it.
{ pattern: "*mimo*v2.5-pro*", levels: ["none", "low", "medium", "high", "xhigh"] },
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
// (disable thinking instead). none kept for the picker = disable.
@@ -73,10 +76,14 @@ export function getThinkingLevels(provider, model) {
if (provider === "kiro" && resolveKiroEffortPath(model) === null) return null;
const caps = getCapabilitiesForModel(provider, model);
if (!caps.reasoning) return null;
const baseId = String(model || "").replace(/\([^()]+\)\s*$/, "");
const modelLevels = provider === "codex"
? getProviderModels("cx").find((entry) => entry.id === baseId)?.thinkingLevels
: null;
const hit = PATTERN_THINKING.find((entry) =>
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
);
let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
let levels = modelLevels || hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none");
return levels;
}

View File

@@ -1,6 +1,12 @@
import { buildClineHeaders } from "../shared/clineAuth.js";
const CLINEPASS_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/models";
// Cline's free tier is published here, not in /api/v1/models: the catalog
// endpoint carries no `cline-free/*` ids at all. Cline's own SDK calls this
// feed unauthenticated (sdk/packages/core/src/services/llms/cline-recommended-models.ts),
// so no Authorization header is sent — adding one would only make the request
// fail on a header the endpoint ignores.
const CLINE_RECOMMENDED_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/ai/cline/recommended-models";
const FETCH_TIMEOUT_MS = 5000;
/**
@@ -72,6 +78,40 @@ export async function resolveClinepassModels(credentials) {
return models.length ? { models } : null;
}
/**
* Fetch Cline's recommended-models feed and return only its `free[]` tier.
* Returns null on any failure — the free tier is additive, so a dead feed must
* never take the /api/v1/models catalog down with it.
* @param {{accessToken?: string, apiKey?: string}} credentials
* @returns {Promise<{id: string, name: string}[] | null>}
*/
async function fetchClineFreeTierModels() {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
try {
const response = await fetch(CLINE_RECOMMENDED_MODELS_ENDPOINT, {
method: "GET",
headers: { Accept: "application/json" },
signal: controller.signal,
});
if (!response.ok) return null;
const json = await response.json();
const free = Array.isArray(json?.free) ? json.free : [];
if (!free.length) return null;
return free
.filter((m) => typeof m?.id === "string" && m.id.trim() !== "")
.map((m) => ({ id: m.id, name: m.name || m.id }));
} catch {
return null;
} finally {
clearTimeout(timer);
}
}
/**
* Fetch Cline live model catalog from Cline's /models endpoint.
* Unlike resolveClinepassModels, this returns ALL models (including
@@ -91,5 +131,15 @@ export async function resolveClineModels(credentials) {
name: m.name || m.id,
}));
return models.length ? { models } : null;
// Free tier: /api/v1/models lists no `cline-free/*` ids, so merge the feed's
// free[] in. First writer wins on a shared id, keeping the catalog's entry
// for anything the two sources agree on.
const freeTier = await fetchClineFreeTierModels();
const byId = new Map(models.map((m) => [m.id, m]));
for (const m of freeTier || []) {
if (!byId.has(m.id)) byId.set(m.id, m);
}
const merged = Array.from(byId.values());
return merged.length ? { models: merged } : null;
}

View File

@@ -4,10 +4,10 @@
import { getGitHubUsage } from "./usage/github.js";
import { getGeminiUsage, getAntigravityUsage } from "./usage/google.js";
import { getClaudeUsage } from "./usage/claude.js";
import { getClaudeUsage, consumeClaudeResetGrant } from "./usage/claude.js";
import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits } from "./usage/codex.js";
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits, consumeClaudeResetGrant };
import { getKiroUsage } from "./usage/kiro.js";
import { getMiniMaxUsage } from "./usage/minimax.js";
import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js";

View File

@@ -3,7 +3,7 @@
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { ANTHROPIC_API_VERSION } from "../../providers/shared.js";
import { ANTHROPIC_API_VERSION, CLAUDE_CLI_VERSION } from "../../providers/shared.js";
import { U, parseResetTime } from "./shared.js";
// Claude API config (urls from registry, apiVersion is header logic kept here)
@@ -11,7 +11,11 @@ const CLAUDE_CONFIG = {
oauthUsageUrl: U("claude").oauthUrl,
usageUrl: U("claude").orgUrl,
settingsUrl: U("claude").settingsUrl,
profileUrl: U("claude").profileUrl,
resetUrl: U("claude").resetUrl,
apiVersion: ANTHROPIC_API_VERSION,
// Reset grants are gated by surface: only "(external, cli)" UA is eligible
userAgent: `claude-cli/${CLAUDE_CLI_VERSION} (external, cli)`,
};
// OAuth usage endpoint rate-limits (429); cool down per-token to stop hammering it.
@@ -64,12 +68,14 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
}
// Primary: OAuth usage endpoint (Claude Code consumer OAuth tokens)
const oauthResponse = await proxyAwareFetch(CLAUDE_CONFIG.oauthUsageUrl, {
// cedar_ember=1 adds the "limit reset" grant block (same flag Claude Code sends)
const oauthResponse = await proxyAwareFetch(`${CLAUDE_CONFIG.oauthUsageUrl}?cedar_ember=1`, {
method: "GET",
headers: {
"Authorization": `Bearer ${accessToken}`,
"anthropic-beta": "oauth-2025-04-20",
"anthropic-version": CLAUDE_CONFIG.apiVersion,
"User-Agent": CLAUDE_CONFIG.userAgent,
},
}, proxyOptions);
@@ -129,6 +135,7 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
return {
plan: "Claude Code",
extraUsage: data.extra_usage ?? null,
resetCredits: parseClaudeResetGrants(data.cedar_ember),
quotas,
};
}
@@ -146,6 +153,70 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
}
}
// Free "limit reset" grants (Anthropic program id "cedar_ember").
// Shape: { eligible, next_grant_id, grants: [{ id, resets_left, ends_at, paused, clears }] }
export function parseClaudeResetGrants(block) {
if (!block?.eligible || !Array.isArray(block.grants)) return null;
const grants = block.grants.filter((g) => g?.id && !g.paused && Number(g.resets_left) > 0);
const next = grants.find((g) => g.id === block.next_grant_id) || grants[0] || null;
return {
availableCount: grants.reduce((sum, g) => sum + Number(g.resets_left), 0),
nextGrantId: next?.id || null,
expiresAt: next?.ends_at || null,
clears: next?.clears || [],
cooldownUntil: block.cooldown_until || null,
weeklyResetsAt: block.weekly_resets_at || null,
grants: block.grants.filter((g) => g?.id).map((g) => ({
id: g.id,
label: g.label || "",
resetsLeft: Number(g.resets_left) || 0,
resetsTotal: Number(g.resets_total) || 0,
startsAt: g.starts_at || null,
endsAt: g.ends_at || null,
clears: Array.isArray(g.clears) ? g.clears : [],
paused: g.paused === true,
usableNow: g.usable_now === true,
useRequiresLimit: g.use_requires_limit !== false,
})),
};
}
// Spend one reset grant: refills the limits listed in grant.clears. Irreversible.
export async function consumeClaudeResetGrant(accessToken, grantId, proxyOptions = null) {
if (!accessToken) throw new Error("No Claude access token available. Please re-authorize the connection.");
if (!/^[a-z0-9_-]{1,40}$/.test(grantId || "")) throw new Error("Invalid reset grant id.");
const headers = {
"Authorization": `Bearer ${accessToken}`,
"anthropic-beta": "oauth-2025-04-20",
"anthropic-version": CLAUDE_CONFIG.apiVersion,
"User-Agent": CLAUDE_CONFIG.userAgent,
"Content-Type": "application/json",
};
const profileRes = await proxyAwareFetch(CLAUDE_CONFIG.profileUrl, { method: "GET", headers }, proxyOptions);
const profile = await profileRes.json().catch(() => null);
const orgId = profile?.organization?.uuid;
if (!profileRes.ok || !orgId) throw new Error(`Cannot resolve Claude organization (${profileRes.status}).`);
const res = await proxyAwareFetch(CLAUDE_CONFIG.resetUrl.replace("{org_id}", orgId), {
method: "POST",
headers,
body: JSON.stringify({ program: "cedar_ember", grant_id: grantId, request_id: crypto.randomUUID() }),
}, proxyOptions);
const data = await res.json().catch(() => null);
usageCache.delete(accessToken); // next read must show refilled limits
return {
ok: res.ok && data?.result === "reset",
status: res.status,
result: data?.result || null,
reason: data?.reason || null,
resetsLeft: data?.resets_left ?? null,
message: data?.error?.message || null,
};
}
/**
* Legacy Claude usage for API key / org admin users
*/

View File

@@ -333,6 +333,9 @@ export function createResponsesApiTransformStream(logger = null) {
// Regular text content
if (content) {
// The answer starts, so thinking is over. Upstreams that send reasoning via
// reasoning_content never emit "</think>", so close it here rather than at finish.
closeReasoning(controller);
if (!state.msgItemAdded[idx]) {
state.msgItemAdded[idx] = true;
const msgId = `msg_${state.responseId}_${idx}`;
@@ -372,6 +375,7 @@ export function createResponsesApiTransformStream(logger = null) {
// Handle tool_calls
if (delta.tool_calls) {
closeReasoning(controller);
closeMessage(controller, idx);
for (const tc of delta.tool_calls) {

View File

@@ -102,9 +102,28 @@ export function extractThinking(body) {
return null;
}
// Capture thinking intent from a body. Alias of extractThinking, named for clarity
// at the call-site where intent is snapshotted before format translation.
export const captureThinking = extractThinking;
// Capture thinking intent from a body before format translation strips it.
// Besides the effort, records whether an OpenAI-shaped client wants the thinking
// text itself: Claude returns it only with thinking.display "summarized", a field
// OpenAI has no equivalent for, so the intent cannot survive translation on its own.
export function captureThinking(body) {
const cfg = extractThinking(body);
if (!cfg || cfg.mode === "none") return cfg;
const display = openAIThinkingDisplay(body);
return display ? { ...cfg, display } : cfg;
}
function openAIThinkingDisplay(body) {
// Responses API: reasoning.summary is the explicit request for reasoning text.
if (body.reasoning && typeof body.reasoning === "object") {
const summary = body.reasoning.summary;
return typeof summary === "string" && summary && summary !== "none" ? "summarized" : undefined;
}
// Chat Completions has no summary knob. A client setting reasoning_effort is
// asking for reasoning, and reasoning_content is how it would receive it.
if (typeof body.reasoning_effort === "string") return "summarized";
return undefined;
}
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
@@ -302,9 +321,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
case "deepseek": {
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
body.thinking = { type: "enabled" };
// DeepSeek: low/medium→high, xhigh/max→max.
// DeepSeek: low/medium→high, xhigh/max→max. Some backends (mimo v2.5-pro/v2.6
// on opencode-go, probed live) 400 on "max" — clamp to high when the declared
// levels exclude it.
const level = toLevel(eff);
body.reasoning_effort = level === "xhigh" || level === "max" ? "max" : "high";
const want = level === "xhigh" || level === "max" ? "max" : "high";
body.reasoning_effort = want === "max" && supportedLevels && !supportedLevels.includes("max") ? "high" : want;
break;
}
case "kimi": {
@@ -380,7 +402,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
const supportedLevels = getThinkingLevels(provider, cleanModel);
// Anthropic's `display` (summarized | omitted) decides whether thinking text
// comes back at all; keep what the client asked for instead of resetting it.
const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined;
// An OpenAI-shaped client's ask arrives via the captured intent instead.
const display = typeof body.thinking?.display === "string" ? body.thinking.display : intent?.display;
stripAll(body);
applyFormat(fmt, body, cfg, caps, supportedLevels, display);
return body;

View File

@@ -432,7 +432,7 @@ export function cleanJSONSchemaForAntigravity(schema) {
return cleaned;
}
// Merge adjacent same-role messages, strip empty parts, ensure initial user turn
// Merge adjacent same-role messages, strip empty parts, ensure initial and terminal user turns
export function normalizeGeminiContents(contents) {
const out = [];
for (const c of contents || []) {
@@ -446,6 +446,23 @@ export function normalizeGeminiContents(contents) {
if (out.length > 0 && out[0].role !== "user") {
out.unshift({ role: "user", parts: [{ text: "..." }] });
}
if (out.length > 0 && out.at(-1).role === "model") {
const fnCalls = (out.at(-1).parts || []).filter(p => p && p.functionCall);
if (fnCalls.length > 0) {
const responses = fnCalls.map(p => {
const call = p.functionCall || {};
const fr = {
name: call.name || "tool",
response: { result: "Continue." }
};
if (call.id) fr.id = call.id;
return { functionResponse: fr };
});
out.push({ role: "user", parts: responses });
} else {
out.push({ role: "user", parts: [{ text: "Continue." }] });
}
}
return out;
}

View File

@@ -59,10 +59,6 @@ export function claudeToOpenAIResponse(chunk, state) {
}
if (block?.type === CLAUDE_BLOCK.TEXT) {
state.textBlockStarted = true;
} else if (block?.type === CLAUDE_BLOCK.THINKING) {
state.inThinkingBlock = true;
state.currentBlockIndex = chunk.index;
results.push(createChunk(state, { content: "<think>" }));
} else if (block?.type === CLAUDE_BLOCK.TOOL_USE) {
const toolCallIndex = state.toolCallIndex++;
// Restore original tool name from mapping (Claude OAuth)
@@ -89,6 +85,8 @@ export function claudeToOpenAIResponse(chunk, state) {
if (delta?.type === "text_delta" && delta.text) {
results.push(createChunk(state, { content: delta.text }));
} else if (delta?.type === "thinking_delta" && delta.thinking) {
// Thinking travels only in reasoning_content. No "<think>" markers in
// content: OpenAI-format clients render them as literal text.
results.push(createChunk(state, reasoningDelta(delta.thinking)));
} else if (delta?.type === "input_json_delta" && delta.partial_json) {
const toolCall = state.toolCalls.get(chunk.index);
@@ -112,10 +110,6 @@ export function claudeToOpenAIResponse(chunk, state) {
state.serverToolBlockIndex = -1;
break;
}
if (state.inThinkingBlock && chunk.index === state.currentBlockIndex) {
results.push(createChunk(state, { content: "</think>" }));
state.inThinkingBlock = false;
}
state.textBlockStarted = false;
state.thinkingBlockStarted = false;
break;

View File

@@ -129,12 +129,16 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
}
if (content) {
// The answer starts, so thinking is over. Upstreams that send reasoning via
// reasoning_content never emit "</think>", so close it here rather than at finish.
closeReasoning(state, emit);
emitTextContent(state, emit, idx, content);
}
}
// Handle tool_calls (empty array is truthy; require a real call)
if (delta.tool_calls && delta.tool_calls.length) {
closeReasoning(state, emit);
closeMessage(state, emit, idx);
for (const tc of delta.tool_calls) {
emitToolCall(state, emit, tc);
@@ -219,15 +223,19 @@ function closeReasoning(state, emit) {
part: { type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }
});
const item = {
id: state.reasoningId,
type: RESPONSES_ITEM.REASONING,
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
};
emit("response.output_item.done", {
type: "response.output_item.done",
output_index: state.reasoningIndex,
item: {
id: state.reasoningId,
type: RESPONSES_ITEM.REASONING,
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
}
item
});
recordCompletedOutputItem(state, state.reasoningIndex, item);
}
}
@@ -291,16 +299,20 @@ function closeMessage(state, emit, idx) {
part: { type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }
});
const item = {
id: msgId,
type: RESPONSES_ITEM.MESSAGE,
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
role: ROLE.ASSISTANT
};
emit("response.output_item.done", {
type: "response.output_item.done",
output_index: parseInt(idx),
item: {
id: msgId,
type: RESPONSES_ITEM.MESSAGE,
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
role: ROLE.ASSISTANT
}
item
});
recordCompletedOutputItem(state, parseInt(idx), item);
}
}
@@ -394,23 +406,51 @@ function closeToolCall(state, emit, idx) {
});
}
const item = {
id: `${custom ? "ctc" : "fc"}_${callId}`,
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
call_id: callId,
name: state.funcNames[idx] || ""
};
emit("response.output_item.done", {
type: "response.output_item.done",
output_index: parseInt(idx),
item: {
id: `${custom ? "ctc" : "fc"}_${callId}`,
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
call_id: callId,
name: state.funcNames[idx] || ""
}
item
});
recordCompletedOutputItem(state, parseInt(idx), item);
state.funcItemDone[idx] = true;
state.funcArgsDone[idx] = true;
}
}
// response.completed carries the finished Response object, so response.output has
// to repeat the items already delivered in response.output_item.done. Clients that
// build their final result from the terminal event (GitHub Copilot CLI, the OpenAI
// SDK "final response" helpers) otherwise treat the turn as empty even though the
// text was streamed - see issue #4307.
//
// Keyed by output_index so a repeated close overwrites rather than duplicating the
// item, and ordered by output_index so response.output matches the order the items
// were emitted in. Lazily created because stream.js can hand us a state it built
// itself rather than one from initState().
function recordCompletedOutputItem(state, outputIndex, item) {
state.completedOutputItems ??= new Map();
const index = Number.isInteger(outputIndex) ? outputIndex : Number.parseInt(outputIndex, 10) || 0;
state.completedOutputItems.set(index, item);
}
function collectCompletedOutputItems(state) {
const recorded = state.completedOutputItems;
if (!(recorded instanceof Map) || recorded.size === 0) return [];
return [...recorded.entries()]
.sort((left, right) => left[0] - right[0])
.map(([, item]) => item);
}
function sendCompleted(state, emit) {
if (!state.completedSent) {
state.completedSent = true;
@@ -423,6 +463,7 @@ function sendCompleted(state, emit) {
status: "completed",
background: false,
error: null,
output: collectCompletedOutputItems(state),
...(state.responsesUsage ? { usage: state.responsesUsage } : {})
}
});

View File

@@ -25,10 +25,25 @@ function deriveUuid(seed) {
function generateFakeUserID(sessionId, apiKey) {
const deviceId = apiKey ? createHash("sha256").update(`device:${apiKey}`).digest("hex") : randomBytes(32).toString("hex");
const accountUuid = apiKey ? deriveUuid(`account:${apiKey}`) : randomUUID();
const sessionUuid = sessionId || randomUUID();
const cleanSessionId = typeof sessionId === "string" ? sessionId.replace(/^claude:/i, "").trim() : null;
const sessionUuid = cleanSessionId || randomUUID();
return `{"device_id":"${deviceId}","account_uuid":"${accountUuid}","session_id":"${sessionUuid}"}`;
}
export function extractClaudeSessionIdFromUserId(userId) {
if (typeof userId !== "string" || !userId) return null;
if (userId[0] === "{") {
try {
const sid = JSON.parse(userId)?.session_id;
return typeof sid === "string" && sid ? sid.replace(/^claude:/i, "").trim() || null : null;
} catch {
return null;
}
}
const clean = userId.replace(/^claude:/i, "").trim();
return clean || null;
}
/**
* Cloak tools before sending to Claude provider (anti-ban):
* - Rename client tools with the CLAUDE_TOOL_SUFFIX ("_ide") in tools[] and messages[]
@@ -89,14 +104,29 @@ export function cloakClaudeTools(body) {
};
}
// Strip a trailing CLAUDE_TOOL_SUFFIX from a cloaked name as a last-resort
// fallback when the name isn't in toolNameMap (e.g. map lost across a retry/
// reconnect). Never strips decoy names — those are meant to reach the client
// unresolved so it can see "tool unavailable" instead of silently no-oping.
function stripCloakSuffix(name) {
if (typeof name !== "string" || !name.endsWith(CLAUDE_TOOL_SUFFIX)) return null;
if (CC_DEFAULT_TOOLS.has(name)) return null;
const original = name.slice(0, -CLAUDE_TOOL_SUFFIX.length);
return original.length > 0 ? original : null;
}
// Decloak tool_use names in non-streaming Claude response body (INPUT side)
export function decloakToolNames(body, toolNameMap) {
if (!toolNameMap?.size || !Array.isArray(body?.content)) return body;
if (!Array.isArray(body?.content)) return body;
const content = body.content.map(block => {
if (block?.type === "tool_use" && toolNameMap.has(block.name)) {
if (block?.type !== "tool_use") return block;
if (toolNameMap?.has(block.name)) {
return { ...block, name: toolNameMap.get(block.name) };
}
return block;
// toolNameMap missing/stale for this name — fall back to suffix stripping
// rather than forwarding an unresolvable "<tool>_ide" name to the client.
const fallback = stripCloakSuffix(block.name);
return fallback ? { ...block, name: fallback } : block;
});
return { ...body, content };
}
@@ -111,19 +141,21 @@ export function decloakToolNames(body, toolNameMap) {
* name appears exactly once per call — on the content_block_start event of
* a tool_use block; argument deltas carry no name.
*
* Unknown names (e.g. a CC decoy tool the model called anyway) pass through
* unchanged, matching the non-streaming decloak behavior.
* Falls back to stripping the literal CLAUDE_TOOL_SUFFIX when the name isn't
* in toolNameMap (map lost across a retry/reconnect), matching the
* non-streaming decloak behavior. Decoy tool names (real CC tool names) and
* anything else pass through unchanged.
*
* @param {object|null} chunk - Parsed SSE event (may be null on stream flush)
* @param {Map|null} toolNameMap - Suffixed → original name map from cloakClaudeTools()
* @returns {object|null} The chunk, with the tool_use name restored when cloaked
*/
export function decloakStreamChunk(chunk, toolNameMap) {
if (!toolNameMap?.size || !chunk || typeof chunk !== "object") return chunk;
if (!chunk || typeof chunk !== "object") return chunk;
if (chunk.type !== "content_block_start") return chunk;
const block = chunk.content_block;
if (block?.type !== "tool_use" || typeof block.name !== "string") return chunk;
const original = toolNameMap.get(block.name);
const original = toolNameMap?.get(block.name) || stripCloakSuffix(block.name);
if (!original) return chunk;
return { ...chunk, content_block: { ...block, name: original } };
}

View File

@@ -27,12 +27,13 @@ export function buildErrorBody(statusCode, message) {
* @param {string} message - Error message
* @returns {Response} HTTP Response object
*/
export function errorResponse(statusCode, message) {
export function errorResponse(statusCode, message, extraHeaders = null) {
return new Response(JSON.stringify(buildErrorBody(statusCode, message)), {
status: statusCode,
headers: {
"Content-Type": "application/json",
"Access-Control-Allow-Origin": "*"
"Access-Control-Allow-Origin": "*",
...extraHeaders
}
});
}
@@ -95,13 +96,13 @@ export async function parseUpstreamError(response, executor = null) {
* @param {number} [resetsAtMs] - Optional precise cooldown expiry (ms epoch) for provider-specific quota errors
* @returns {{ success: false, status: number, error: string, response: Response, resetsAtMs?: number }}
*/
export function createErrorResult(statusCode, message, resetsAtMs) {
export function createErrorResult(statusCode, message, resetsAtMs, extraHeaders = null) {
return {
success: false,
status: statusCode,
error: message,
resetsAtMs,
response: errorResponse(statusCode, message)
response: errorResponse(statusCode, message, extraHeaders)
};
}
@@ -113,7 +114,7 @@ export function createErrorResult(statusCode, message, resetsAtMs) {
* @param {string} retryAfterHuman - Human-readable retry info e.g. "reset after 30s"
* @returns {Response}
*/
export function unavailableResponse(statusCode, message, retryAfter, retryAfterHuman) {
export function unavailableResponse(statusCode, message, retryAfter, retryAfterHuman, extraHeaders = null) {
const retryAfterSec = Math.max(Math.ceil((new Date(retryAfter).getTime() - Date.now()) / 1000), 1);
const msg = `${message} (${retryAfterHuman})`;
return new Response(
@@ -121,8 +122,10 @@ export function unavailableResponse(statusCode, message, retryAfter, retryAfterH
{
status: statusCode,
headers: {
...extraHeaders,
"Content-Type": "application/json",
"Retry-After": String(retryAfterSec)
// Intentionally mis-cased to prevent duplicate headers
"retry-after": String(retryAfterSec)
}
}
);

View File

@@ -0,0 +1,12 @@
const FORWARDED = new Set(["retry-after", "x-should-retry"]);
const FORWARDED_PREFIX = "anthropic-ratelimit-";
export function upstreamResponseHeaders(headers) {
const out = {};
if (typeof headers?.forEach !== "function") return out;
headers.forEach((value, name) => {
const key = name.toLowerCase();
if (FORWARDED.has(key) || key.startsWith(FORWARDED_PREFIX)) out[key] = value;
});
return out;
}