Merge origin/master (v0.5.91) into gitea/new_feature
Resolve conflicts: - streamingHandler.js: adopt upstreamResponseHeaders while keeping 0-token detail row avoidance - capabilities.js: preserve user-asserted caps and globalThis slots without local caching of catalogSource - AddCustomModelModal.js & providers/[id]/page.js: wire STT transport marker with custom model edits/assertions - models/custom/route.js & aliasRepo.js: persist custom model transport and invalidate user caps - usageRepo.js: key byApiKey live stats by full API key and keep tail in maskApiKey - UsageStats.js: lazy load charts dynamically
This commit is contained in:
@@ -178,7 +178,7 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C
|
||||
// makes the backend flag the request and answer 429 Quota Exhausted.
|
||||
export const ANTIGRAVITY_PROMPT_REWRITES = [
|
||||
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
|
||||
{ from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." },
|
||||
{ from: /You are Hermes(?: Agent)?(?:,\s*(?:an intelligent AI assistant|an AI assistant|an AI agent))?(?:,?\s*(?:built|created)\s+by\s+Nous Research)?\./gi, to: "You are an AI assistant." },
|
||||
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
|
||||
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
|
||||
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.
|
||||
|
||||
@@ -3,10 +3,13 @@ import REGISTRY from "../providers/registry/index.js";
|
||||
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
|
||||
import { PROVIDER_MODELS } from "../providers/index.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import { CODEX_REVIEW_SUFFIX, isMuseSparkModel, opencodeFamilyFormats } from "../providers/models/helpers.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
export { PROVIDER_MODELS };
|
||||
|
||||
// OpenCode providers sharing the endpoint-family fallback for unknown model ids
|
||||
const isOpenCodeAlias = (aliasOrId) => !aliasOrId || ["oc", "opencode", "ocg", "opencode-go", "ocz", "opencode-zen"].includes(aliasOrId);
|
||||
|
||||
|
||||
// Helper functions
|
||||
export function getProviderModels(aliasOrId) {
|
||||
@@ -53,20 +56,29 @@ export function findModelName(aliasOrId, modelId) {
|
||||
}
|
||||
|
||||
export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go" || aliasOrId === "ocz" || aliasOrId === "opencode-zen") && isMuseSparkModel(modelId)) {
|
||||
if (isOpenCodeAlias(aliasOrId) && isMuseSparkModel(modelId)) {
|
||||
return FORMATS.OPENAI_RESPONSES;
|
||||
}
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelTargetFormat(findModel(models, modelId, aliasOrId));
|
||||
const found = findModel(models, modelId, aliasOrId);
|
||||
if (found) return modelTargetFormat(found);
|
||||
// Family fallback keeps modelsFetcher/passthrough ids on their endpoint lane
|
||||
if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.targetFormat || null;
|
||||
return null;
|
||||
}
|
||||
|
||||
// Declared upstream formats for a model (registry `supportedFormats`). Drives the
|
||||
// per-model guard on the sourceFormat-matched transport; null when undeclared.
|
||||
// Unknown OpenCode ids fall back to the family regex (chat lane by default) so
|
||||
// auto-fetched models never wrongly use the sourceFormat-matched transport.
|
||||
export function getModelSupportedFormats(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelSupportedFormats(findModel(models, modelId, aliasOrId));
|
||||
const found = findModel(models, modelId, aliasOrId);
|
||||
if (found) return modelSupportedFormats(found);
|
||||
if (isOpenCodeAlias(aliasOrId)) return opencodeFamilyFormats(modelId)?.supportedFormats || [FORMATS.OPENAI];
|
||||
return null;
|
||||
}
|
||||
|
||||
export function getModelType(aliasOrId, modelId) {
|
||||
|
||||
@@ -7,7 +7,7 @@ import {
|
||||
} from "../services/oauthCredentialManager.js";
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { getModelUpstreamId, getProviderModels } from "../config/providerModels.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
@@ -25,6 +25,10 @@ const CODEX_SSE_USER_OUTPUT_PATTERNS = [
|
||||
];
|
||||
const CODEX_SSE_PEEK_BYTES = 256 * 1024;
|
||||
const CODEX_MODEL_CAPACITY_MESSAGE = "Selected model is at capacity. Please try a different model.";
|
||||
function isCodexResponsesLiteModel(model) {
|
||||
const baseId = String(model || "").replace(/\([^()]+\)\s*$/, "");
|
||||
return getProviderModels("cx").some((entry) => entry.id === baseId && entry.responsesLite === true);
|
||||
}
|
||||
|
||||
// Server-generated item id prefixes that Codex /responses cannot resolve when store=false
|
||||
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
|
||||
@@ -43,7 +47,7 @@ const CODEX_PASSTHROUGH_TOOL_TYPES = new Set(["custom"]);
|
||||
const RESPONSES_API_ALLOWLIST = new Set([
|
||||
"model", "input", "instructions", "tools", "tool_choice", "stream", "store",
|
||||
"reasoning", "service_tier", "include", "prompt_cache_key", "client_metadata",
|
||||
"text"
|
||||
"text", "parallel_tool_calls"
|
||||
]);
|
||||
|
||||
// Convert role=system → role=developer in body.input (keeps content in cacheable prefix)
|
||||
@@ -57,13 +61,14 @@ function convertSystemToDeveloperRole(body) {
|
||||
}
|
||||
|
||||
// Strip server-generated item IDs (rs_/fc_/resp_/msg_) from input — avoids 404 with store=false
|
||||
function stripStoredItemReferences(body) {
|
||||
function stripStoredItemReferences(body, preserveLitePrefix = false) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (typeof item === "string" && SERVER_ID_PATTERN.test(item)) return false;
|
||||
if (item && typeof item === "object" && !Array.isArray(item)) {
|
||||
if (item.type === "item_reference") return false;
|
||||
if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)) delete item.id;
|
||||
if (typeof item.id === "string" && SERVER_ID_PATTERN.test(item.id)
|
||||
&& !(preserveLitePrefix && item.role === "developer" && item.id.startsWith("msg_"))) delete item.id;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
@@ -138,6 +143,7 @@ function resolveCacheSessionId(body, credentials) {
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (isCodexResponsesLiteModel(model) && (value === "none" || value === "minimal")) return "low";
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
@@ -209,8 +215,11 @@ export class CodexExecutor extends BaseExecutor {
|
||||
* Override headers to add codex-specific identity headers.
|
||||
* transformRequest runs BEFORE buildHeaders, sets this._currentSessionId.
|
||||
*/
|
||||
buildHeaders(credentials, stream = true) {
|
||||
buildHeaders(credentials, stream = true, _url = null, model = null) {
|
||||
const headers = super.buildHeaders(credentials, stream);
|
||||
if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model))) {
|
||||
headers["x-openai-internal-codex-responses-lite"] = "true";
|
||||
}
|
||||
headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default";
|
||||
// Identify client type to Codex backend (matches official codex CLI)
|
||||
if (!headers["originator"]) headers["originator"] = "codex_cli_rs";
|
||||
@@ -408,6 +417,8 @@ export class CodexExecutor extends BaseExecutor {
|
||||
// Convert string input to array format (Codex API requires input as array)
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
const upstreamModel = getModelUpstreamId("cx", body.model || model);
|
||||
const responsesLite = isCodexResponsesLiteModel(upstreamModel);
|
||||
|
||||
// Ensure input is present and non-empty (Codex API rejects empty input)
|
||||
if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) {
|
||||
@@ -417,7 +428,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
// Keep system prompts in body.input as role=developer so they stay in the cacheable prefix
|
||||
convertSystemToDeveloperRole(body);
|
||||
// Strip server-generated item IDs (rs_/fc_/resp_/msg_) — Codex /responses can't resolve when store=false
|
||||
stripStoredItemReferences(body);
|
||||
stripStoredItemReferences(body, responsesLite);
|
||||
// Flatten function tools + drop unsupported types
|
||||
normalizeCodexTools(body);
|
||||
|
||||
@@ -425,7 +436,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
body.stream = true;
|
||||
|
||||
// If no instructions provided, inject default Codex instructions
|
||||
if (!body.instructions || body.instructions.trim() === "") {
|
||||
if (!responsesLite && (!body.instructions || body.instructions.trim() === "")) {
|
||||
body.instructions = CODEX_DEFAULT_INSTRUCTIONS;
|
||||
}
|
||||
|
||||
@@ -438,7 +449,29 @@ export class CodexExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
// Map virtual Codex review models to the upstream Codex model before suffix parsing.
|
||||
body.model = getModelUpstreamId("cx", body.model || model);
|
||||
body.model = upstreamModel;
|
||||
|
||||
if (responsesLite) {
|
||||
// Codex 0.155 carries tools and instructions as input prefix items.
|
||||
const input = Array.isArray(body.input) ? body.input : [body.input];
|
||||
const hasLitePrefix = input.some((item) => item?.type === "additional_tools");
|
||||
if (!hasLitePrefix) {
|
||||
const instructions = typeof body.instructions === "string" && body.instructions.trim()
|
||||
? body.instructions : CODEX_DEFAULT_INSTRUCTIONS;
|
||||
const prefix = [{ type: "additional_tools", role: "developer", tools: Array.isArray(body.tools) ? body.tools : [] }];
|
||||
if (instructions) {
|
||||
prefix.push({ type: "message", role: "developer", content: [{ type: "input_text", text: instructions }] });
|
||||
}
|
||||
input.unshift(...prefix);
|
||||
}
|
||||
body.input = input;
|
||||
body.instructions = "";
|
||||
body.tools = null;
|
||||
body.tool_choice ||= "auto";
|
||||
body.parallel_tool_calls = false;
|
||||
} else {
|
||||
delete body.parallel_tool_calls;
|
||||
}
|
||||
|
||||
// Extract thinking level from model name suffix
|
||||
// e.g., gpt-5.3-codex-high → high, gpt-5.3-codex → medium (default)
|
||||
@@ -455,12 +488,13 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || (responsesLite ? 'medium' : 'low'));
|
||||
body.reasoning = responsesLite ? { effort } : { effort, summary: "auto" };
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
if (!responsesLite && !body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
if (responsesLite) body.reasoning.context = "all_turns";
|
||||
delete body.reasoning_effort;
|
||||
|
||||
// Include reasoning encrypted content (required by Codex backend for reasoning models)
|
||||
|
||||
@@ -148,7 +148,7 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
|
||||
const reader = originalResponse.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
const bufferedLines = [];
|
||||
const rawChunks = [];
|
||||
let detectedError = null;
|
||||
|
||||
try {
|
||||
@@ -162,16 +162,15 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
|
||||
const parsed = JSON.parse(jsonStr);
|
||||
if (parsed?.type === "error") {
|
||||
detectedError = parsed;
|
||||
} else {
|
||||
bufferedLines.push(trimmed);
|
||||
}
|
||||
} catch {
|
||||
bufferedLines.push(trimmed);
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
rawChunks.push(value);
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() || "";
|
||||
@@ -182,7 +181,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
|
||||
if (!trimmed) continue;
|
||||
const jsonStr = trimmed.startsWith("data:") ? trimmed.slice(5).trim() : trimmed;
|
||||
if (!jsonStr || jsonStr === "[DONE]") {
|
||||
bufferedLines.push(trimmed);
|
||||
stopLoop = true;
|
||||
break;
|
||||
}
|
||||
@@ -191,7 +189,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
|
||||
try {
|
||||
event = JSON.parse(jsonStr);
|
||||
} catch {
|
||||
bufferedLines.push(trimmed);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -201,8 +198,6 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
|
||||
break;
|
||||
}
|
||||
|
||||
bufferedLines.push(trimmed);
|
||||
|
||||
if (
|
||||
event?.type === "text-delta" ||
|
||||
event?.type === "reasoning-delta" ||
|
||||
@@ -245,29 +240,18 @@ export async function inspectAndWrapCommandCodeResponse(originalResponse, model)
|
||||
);
|
||||
}
|
||||
|
||||
const combinedStream = createReplayedStream(bufferedLines, buffer, reader);
|
||||
const combinedStream = createRawReplayedStream(rawChunks, reader);
|
||||
return wrapNdjsonAsOpenAISse(combinedStream, model, originalResponse);
|
||||
}
|
||||
|
||||
function createReplayedStream(bufferedLines, remainingBuffer, reader) {
|
||||
const encoder = new TextEncoder();
|
||||
let replayed = false;
|
||||
function createRawReplayedStream(rawChunks, reader) {
|
||||
let chunkIndex = 0;
|
||||
|
||||
return new ReadableStream({
|
||||
async pull(controller) {
|
||||
if (!replayed) {
|
||||
replayed = true;
|
||||
let prefix = bufferedLines.join("\n");
|
||||
if (prefix && remainingBuffer) {
|
||||
prefix += "\n" + remainingBuffer;
|
||||
} else if (remainingBuffer) {
|
||||
prefix = remainingBuffer;
|
||||
} else if (prefix) {
|
||||
prefix += "\n";
|
||||
}
|
||||
if (prefix) {
|
||||
controller.enqueue(encoder.encode(prefix));
|
||||
}
|
||||
if (chunkIndex < rawChunks.length) {
|
||||
controller.enqueue(rawChunks[chunkIndex++]);
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
|
||||
@@ -1,12 +1,13 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS, PROVIDER_OAUTH } from "../config/providers.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta } from "../providers/shared.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta, mergeAnthropicBeta } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
import { OAUTH_ENDPOINTS, buildKimiHeaders } from "../config/appConstants.js";
|
||||
import { buildClineHeaders } from "../shared/clineAuth.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
|
||||
import { extractClaudeSessionIdFromUserId } from "../utils/claudeCloaking.js";
|
||||
|
||||
// Auth header descriptors — derived from registry transport.auth, fallback to hardcoded defaults.
|
||||
const BEARER = { combined: true, header: "Authorization", scheme: "bearer" };
|
||||
@@ -164,9 +165,21 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
// a node fronting Kimi or GLM answers on its own ids and never matches, so
|
||||
// gateways that would choke on unknown beta flags are left untouched.
|
||||
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
|
||||
const clientBeta = credentials?.rawHeaders?.["anthropic-beta"];
|
||||
if (model && (this.provider === "claude"
|
||||
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model, body);
|
||||
headers["Anthropic-Beta"] = mergeAnthropicBeta(selectAnthropicBeta(model, body), clientBeta);
|
||||
} else if (this.provider === "anthropic" && clientBeta) {
|
||||
headers["Anthropic-Beta"] = mergeAnthropicBeta(headers["Anthropic-Beta"], clientBeta);
|
||||
}
|
||||
|
||||
// Claude OAuth: align x-claude-code-session-id with metadata.user_id.session_id if missing
|
||||
if (this.provider === "claude" && !headers["x-claude-code-session-id"]) {
|
||||
const token = credentials?.accessToken || credentials?.apiKey || "";
|
||||
if (token.includes("sk-ant-oat")) {
|
||||
const sid = extractClaudeSessionIdFromUserId(body?.metadata?.user_id);
|
||||
if (sid) headers["x-claude-code-session-id"] = sid;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import crypto from "node:crypto";
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { modelTargetFormat } from "../providers/models/schema.js";
|
||||
import { getProviderModels } from "../config/providerModels.js";
|
||||
import { getModelTargetFormat } from "../config/providerModels.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
@@ -41,16 +41,10 @@ function translatedSession(sessionId, clientTool) {
|
||||
return `ses_${digest}`;
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so checks hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …).
|
||||
// Reading the registry keeps this in sync with config — never hardcode model ids here.
|
||||
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …),
|
||||
// including the family-regex fallback for passthrough ids — never hardcode model ids here.
|
||||
function isResponsesModel(model) {
|
||||
const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model));
|
||||
return modelTargetFormat(entry) === "openai-responses";
|
||||
return getModelTargetFormat("opencode-go", model) === FORMATS.OPENAI_RESPONSES;
|
||||
}
|
||||
|
||||
// Flatten Chat Completions tool declarations into the Responses flat shape and
|
||||
|
||||
@@ -9,6 +9,7 @@ import { createRequestLogger } from "../utils/requestLogger.js";
|
||||
import { getModelTargetFormat, getModelSupportedFormats, getModelStrip, getModelUpstreamId, getModelType, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { upstreamResponseHeaders } from "../utils/upstreamHeaders.js";
|
||||
import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js";
|
||||
import { handleBypassRequest } from "../utils/bypassHandler.js";
|
||||
import { trackPendingRequest, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
@@ -493,7 +494,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
log.errorLine(reqTag, "✗", `ERROR ${statusCode} · ${provider}/${model} · ${Date.now() - requestStartTime}ms${urlStr}\n ${errMsg}`);
|
||||
}
|
||||
reqLogger.logError(new Error(message), finalBody || translatedBody);
|
||||
return createErrorResult(statusCode, errMsg, resetsAtMs);
|
||||
return createErrorResult(statusCode, errMsg, resetsAtMs, upstreamResponseHeaders(providerResponse.headers));
|
||||
}
|
||||
|
||||
const appendLog = () => {}; // request log derived from usageHistory; kept as no-op seam for handlers
|
||||
|
||||
@@ -4,6 +4,7 @@ import { fromOpenAIFinish } from "../../translator/concerns/finishReason.js";
|
||||
import { ollamaBodyToOpenAI } from "../../translator/response/ollama-to-openai.js";
|
||||
import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTracking.js";
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js";
|
||||
@@ -417,7 +418,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(restoreToolNames(translatedResponse, toolNameMap)), {
|
||||
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*" }
|
||||
headers: { "Content-Type": "application/json", "Access-Control-Allow-Origin": "*", ...upstreamResponseHeaders(providerResponse.headers) }
|
||||
})
|
||||
};
|
||||
}
|
||||
|
||||
@@ -20,6 +20,7 @@ import {
|
||||
import { streamStatusForContent } from "../../utils/streamErrorPatterns.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
import { upstreamResponseHeaders } from "../../utils/upstreamHeaders.js";
|
||||
|
||||
// Codex returns Responses API SSE → which client format to translate INTO, by request sourceFormat.
|
||||
// Gemini-family all map to ANTIGRAVITY decoder; unknown sources fall back to OPENAI.
|
||||
@@ -158,7 +159,12 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(transformedBody, { headers: SSE_HEADERS }),
|
||||
response: new Response(transformedBody, {
|
||||
headers: {
|
||||
...SSE_HEADERS,
|
||||
...upstreamResponseHeaders(providerResponse.headers),
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
266
open-sse/handlers/geminiLiveStt.js
Normal file
266
open-sse/handlers/geminiLiveStt.js
Normal file
@@ -0,0 +1,266 @@
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
// Gemini Live API realtime STT transport.
|
||||
//
|
||||
// The REST generateContent path (sttCore.transcribeGemini) only transcribes
|
||||
// whole files inline. The Live API's `:bidiGenerateContent` WebSocket is the
|
||||
// streaming counterpart: audio goes up as realtimeInput mediaChunks and the
|
||||
// server pushes incremental `serverContent.inputTranscription` events back.
|
||||
// This module owns the socket lifecycle only — envelope/response shaping
|
||||
// stays in sttCore so the engine's single STT exit shape is preserved.
|
||||
//
|
||||
// Marker contract: dispatched from sttCore's format-switch when the model
|
||||
// entry carries `transport: "gemini-live"` (registry) or the caller passes a
|
||||
// transport string (custom models). Never keyed on a hardcoded model id here.
|
||||
//
|
||||
// Transport behavior:
|
||||
// - Node >= 22 global WebSocket (undici). No new dependency.
|
||||
// - Live API expects low-latency PCM; other containers are forwarded with
|
||||
// their declared MIME unchanged (provider-side rejection is surfaced).
|
||||
// - Text accumulation is append-only over inputTranscription segments and
|
||||
// ends on serverContent.turnComplete (or graceful close with partial text).
|
||||
// - Transcription deltas are kept per-frame (chunks[]) so sttCore can shape
|
||||
// verbose_json segments without fabricating timestamps. goAway advisements
|
||||
// rotate the socket once per call: setup replay + byte-offset resume.
|
||||
|
||||
const SETUP_TIMEOUT_MS = 10_000; // open → setupComplete
|
||||
const TURN_TIMEOUT_MS = 60_000; // audio streamed → turnComplete
|
||||
const MAX_TIMEOUT_MS = 300_000; // clamp ceiling for client-supplied lifecycle knobs
|
||||
const CHUNK_BYTES = 16_384; // ~0.5s of 16-bit 16kHz mono PCM
|
||||
const GOAWAY_RECONNECTS = 1; // socket rotations honoured per call
|
||||
|
||||
class GeminiLiveError extends Error {
|
||||
constructor(message, status) {
|
||||
super(message);
|
||||
this.name = "GeminiLiveError";
|
||||
this.status = status || 502;
|
||||
}
|
||||
}
|
||||
|
||||
// REST base (https://host/v1beta/models) → Live WS base
|
||||
// (wss://host/ws/api/v1beta/models), then the bidiGenerateContent endpoint.
|
||||
function toLiveWsUrl(baseUrl, model, token) {
|
||||
const url = new URL(baseUrl);
|
||||
url.protocol = "wss:";
|
||||
if (!url.pathname.startsWith("/ws/")) url.pathname = `/ws/api${url.pathname}`;
|
||||
const base = url.toString().replace(/\/+$/, "");
|
||||
return `${base}/${encodeURIComponent(model)}:bidiGenerateContent?key=${encodeURIComponent(token || "")}`;
|
||||
}
|
||||
|
||||
// Bind socket events supporting BOTH handler styles: addEventListener
|
||||
// (browser WebSocket, undici) and onopen/onmessage property assignment
|
||||
// (minimal polyfills). Whichever the implementation exposes, it works.
|
||||
function bindSocket(ws, { onOpen, onMessage, onError, onClose }) {
|
||||
if (typeof ws.addEventListener === "function") {
|
||||
ws.addEventListener("open", onOpen);
|
||||
ws.addEventListener("message", onMessage);
|
||||
ws.addEventListener("error", onError);
|
||||
ws.addEventListener("close", onClose);
|
||||
return;
|
||||
}
|
||||
ws.onopen = onOpen;
|
||||
ws.onmessage = onMessage;
|
||||
ws.onerror = onError;
|
||||
ws.onclose = onClose;
|
||||
}
|
||||
|
||||
function parseFrame(data) {
|
||||
try {
|
||||
return JSON.parse(typeof data === "string" ? data : String(data));
|
||||
} catch {
|
||||
return null; // non-JSON frames carry no Live API semantics
|
||||
}
|
||||
}
|
||||
|
||||
function firstStringField(formData, key) {
|
||||
const v = typeof formData?.get === "function" ? formData.get(key) : null;
|
||||
return typeof v === "string" && v.trim() ? v.trim() : "";
|
||||
}
|
||||
|
||||
// Lifecycle knobs the live registry entry advertises in params[]
|
||||
// (setup/turn timeouts). They ride the same formData pass-through sttCore
|
||||
// gives every transport — no sttCore change needed to reach this leaf.
|
||||
function firstNumberField(formData, key, fallback) {
|
||||
const n = Number(firstStringField(formData, key));
|
||||
return Number.isFinite(n) && n > 0 ? Math.min(n, MAX_TIMEOUT_MS) : fallback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Transcribe an audio File via the Gemini Live bidirectional stream.
|
||||
* @returns {Promise<{text: string, chunks: string[]}>} transcript plus the raw
|
||||
* incremental inputTranscription deltas (sttCore shapes verbose_json from them).
|
||||
* @throws {GeminiLiveError} with .status for the sttCore error envelope.
|
||||
*/
|
||||
export async function transcribeGeminiLive({ cfg, file, model, token, formData, mimeType }) {
|
||||
const WS = globalThis.WebSocket;
|
||||
if (!WS) throw new GeminiLiveError("Gemini Live transport needs global WebSocket (Node >= 22)", 502);
|
||||
|
||||
const buf = Buffer.from(await file.arrayBuffer());
|
||||
if (!buf.length) throw new GeminiLiveError("Empty audio file", 400);
|
||||
|
||||
const instruction = firstStringField(formData, "prompt") || "Transcribe the spoken audio verbatim.";
|
||||
const language = firstStringField(formData, "language");
|
||||
const setupTimeoutMs = firstNumberField(formData, "setup_timeout_ms", SETUP_TIMEOUT_MS);
|
||||
const turnTimeoutMs = firstNumberField(formData, "turn_timeout_ms", TURN_TIMEOUT_MS);
|
||||
// system_instruction (registry param) overrides the built-in transcription
|
||||
// directive wholesale; prompt/language only shape the default.
|
||||
const instructionOverride = firstStringField(formData, "system_instruction");
|
||||
const systemText = instructionOverride
|
||||
|| (language ? `${instruction} Language: ${language}.` : instruction);
|
||||
const wsUrl = toLiveWsUrl(cfg.baseUrl, model, token);
|
||||
|
||||
return await new Promise((resolve, reject) => {
|
||||
let text = "";
|
||||
const chunks = []; // raw inputTranscription deltas, shaped by sttCore
|
||||
let settled = false;
|
||||
let timer = null;
|
||||
let goAwayTimer = null;
|
||||
let ws = null;
|
||||
let generation = 0; // socket identity: superseded closes never settle
|
||||
let sentBytes = 0; // audio prefix already handed to the live socket
|
||||
let goAwayReconnects = GOAWAY_RECONNECTS;
|
||||
|
||||
const arm = (ms, message) => {
|
||||
if (timer) clearTimeout(timer);
|
||||
timer = setTimeout(() => fail(new GeminiLiveError(message, 504)), ms);
|
||||
};
|
||||
const shutdown = () => {
|
||||
if (timer) { clearTimeout(timer); timer = null; }
|
||||
if (goAwayTimer) { clearTimeout(goAwayTimer); goAwayTimer = null; }
|
||||
// ws is null until the first open() dials (and stays null when the
|
||||
// constructor throws) — fail() runs shutdown() on that path.
|
||||
if (!ws) return;
|
||||
try {
|
||||
if (ws.readyState === WS.OPEN || ws.readyState === WS.CONNECTING) ws.close(1000);
|
||||
} catch { /* socket already dead — outcome is already settled */ }
|
||||
};
|
||||
const succeed = () => {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
shutdown();
|
||||
resolve({ text, chunks });
|
||||
};
|
||||
const fail = (err) => {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
shutdown();
|
||||
reject(err);
|
||||
};
|
||||
const send = (frame) => {
|
||||
if (ws.readyState !== WS.OPEN) return false;
|
||||
try {
|
||||
ws.send(JSON.stringify(frame));
|
||||
} catch {
|
||||
return false; // socket died mid-send — streamAudioAndPrompt maps this to a 502
|
||||
}
|
||||
return true;
|
||||
};
|
||||
|
||||
// Streams every byte not yet sent, then the flushing text turn. After a
|
||||
// goAway rotation this resumes from sentBytes — no audio re-upload.
|
||||
const streamAudioAndPrompt = () => {
|
||||
for (let off = sentBytes; off < buf.length; off += CHUNK_BYTES) {
|
||||
const mediaChunk = buf.subarray(off, off + CHUNK_BYTES).toString("base64");
|
||||
if (!send({ realtimeInput: { mediaChunks: [{ mimeType, data: mediaChunk }] } })) {
|
||||
fail(new GeminiLiveError("Gemini Live socket closed while streaming audio", 502));
|
||||
return;
|
||||
}
|
||||
sentBytes = Math.min(off + CHUNK_BYTES, buf.length);
|
||||
}
|
||||
// Final user turn: flushes the recognizer and yields turnComplete.
|
||||
send({ clientContent: { turns: [{ parts: [{ text: systemText }] }], turnComplete: true } });
|
||||
};
|
||||
|
||||
// goAway: the server names the instant it will force-close this socket.
|
||||
// Graceful play = rotate BEFORE the deadline: retire the live socket,
|
||||
// dial a fresh one, replay setup, resume audio from sentBytes — text and
|
||||
// chunks survive the hop. Once the advisory budget is spent a later
|
||||
// goAway is left to the close path, which settles on partial transcript.
|
||||
const scheduleGoAwayReconnect = (goAway) => {
|
||||
if (settled || goAwayTimer || goAwayReconnects <= 0) return;
|
||||
const deadline = Date.parse(typeof goAway?.time === "string" ? goAway.time : "");
|
||||
const delay = Number.isFinite(deadline)
|
||||
? Math.max(0, Math.min(deadline - Date.now(), setupTimeoutMs))
|
||||
: 0;
|
||||
goAwayTimer = setTimeout(() => {
|
||||
goAwayTimer = null;
|
||||
if (settled) return;
|
||||
goAwayReconnects--;
|
||||
generation++;
|
||||
try { ws?.close(1000); } catch { /* deadline crossed mid-flight — re-dial anyway */ }
|
||||
open();
|
||||
}, delay);
|
||||
};
|
||||
|
||||
const open = () => {
|
||||
const gen = ++generation;
|
||||
try {
|
||||
ws = new WS(wsUrl);
|
||||
} catch {
|
||||
fail(new GeminiLiveError("Gemini Live websocket connection failed", 502));
|
||||
return;
|
||||
}
|
||||
bindSocket(ws, {
|
||||
onOpen: () => {
|
||||
if (settled || gen !== generation) return;
|
||||
arm(setupTimeoutMs, "Gemini Live timed out waiting for setupComplete");
|
||||
send({
|
||||
setup: {
|
||||
model: `models/${model}`,
|
||||
generationConfig: {
|
||||
responseModalities: ["TEXT"],
|
||||
inputAudioTranscription: {},
|
||||
},
|
||||
systemInstruction: { parts: [{ text: systemText }] },
|
||||
},
|
||||
});
|
||||
},
|
||||
onMessage: (ev) => {
|
||||
if (settled || gen !== generation) return;
|
||||
const frame = parseFrame(ev?.data);
|
||||
if (!frame) return;
|
||||
|
||||
if (frame.error) {
|
||||
const e = frame.error;
|
||||
fail(new GeminiLiveError(`Gemini Live error${e.status ? ` (${e.status})` : ""}: ${e.message || "unknown"}`, 502));
|
||||
return;
|
||||
}
|
||||
if (frame.goAway) {
|
||||
scheduleGoAwayReconnect(frame.goAway);
|
||||
return;
|
||||
}
|
||||
|
||||
const sc = frame.serverContent;
|
||||
if (!sc) return;
|
||||
|
||||
const delta = typeof sc.inputTranscription?.text === "string" ? sc.inputTranscription.text : "";
|
||||
// Trim before testing: a padding-only frame carries no transcript and
|
||||
// must not make an empty run look like a partial success on close.
|
||||
if (delta.trim()) {
|
||||
text += delta;
|
||||
chunks.push(delta);
|
||||
}
|
||||
|
||||
if (sc.setupComplete) {
|
||||
arm(turnTimeoutMs, "Gemini Live transcription timed out");
|
||||
streamAudioAndPrompt();
|
||||
return;
|
||||
}
|
||||
if (sc.turnComplete) succeed();
|
||||
},
|
||||
onError: () => {
|
||||
if (settled || gen !== generation) return;
|
||||
fail(new GeminiLiveError("Gemini Live websocket connection failed", 502));
|
||||
},
|
||||
onClose: (ev) => {
|
||||
if (settled || gen !== generation) return;
|
||||
// Partial transcript beats a hard error on graceful close; silence is one.
|
||||
if (text.trim()) succeed();
|
||||
else fail(new GeminiLiveError(`Gemini Live socket closed before completion${ev?.code ? ` (code ${ev.code})` : ""}`, 502));
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
open();
|
||||
});
|
||||
}
|
||||
@@ -1,5 +1,7 @@
|
||||
import { Buffer } from "node:buffer";
|
||||
import { createErrorResult } from "../utils/error.js";
|
||||
import { transcribeGeminiLive } from "./geminiLiveStt.js";
|
||||
import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
|
||||
// Build auth headers from sttConfig + token
|
||||
@@ -162,11 +164,26 @@ function jsonResponse(obj) {
|
||||
};
|
||||
}
|
||||
|
||||
// Model-level transport marker (registry models[].transport, e.g. the Gemini
|
||||
// live STT entry's "gemini-live", or a custom model's stored transport).
|
||||
// Dispatch reads the marker — never a hardcoded model id — so new realtime
|
||||
// providers extend sttCore through data, not code.
|
||||
function resolveModelTransport(provider, model) {
|
||||
const key = PROVIDER_ID_TO_ALIAS[provider] || provider;
|
||||
const models = PROVIDER_MODELS[key] || PROVIDER_MODELS[provider];
|
||||
if (!Array.isArray(models)) return null;
|
||||
const entry = models.find((m) => m && m.id === model && (m.kind || "llm") === "stt");
|
||||
const marker = typeof entry?.transport === "string" ? entry.transport.trim() : "";
|
||||
return marker || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* STT core handler — dispatch by sttConfig.format.
|
||||
* STT core handler — dispatch by model transport marker, else sttConfig.format.
|
||||
* `transport` is the caller-supplied marker override (custom models resolve
|
||||
* it in the app layer; built-ins fall back to the registry entry marker).
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleSttCore({ provider, model, formData, credentials, sttConfig }) {
|
||||
export async function handleSttCore({ provider, model, formData, credentials, sttConfig, transport }) {
|
||||
const file = formData.get("file");
|
||||
if (!file) return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: file");
|
||||
|
||||
@@ -186,8 +203,29 @@ export async function handleSttCore({ provider, model, formData, credentials, st
|
||||
return createErrorResult(HTTP_STATUS.UNAUTHORIZED, `No credentials for STT provider: ${provider}`);
|
||||
}
|
||||
|
||||
// Format-switch extension: an explicit caller marker wins over the registry
|
||||
// marker; with neither, the provider-default sttConfig.format applies.
|
||||
const marker = (typeof transport === "string" && transport.trim()) ? transport.trim() : resolveModelTransport(provider, model);
|
||||
|
||||
try {
|
||||
switch (cfg.format) {
|
||||
switch (marker || cfg.format) {
|
||||
case "gemini-live": {
|
||||
const live = await transcribeGeminiLive({ cfg, file, model, token, formData, mimeType: resolveAudioContentType(file) });
|
||||
// response_format parity with the OpenAI-compatible transport: default
|
||||
// envelope stays {text}; verbose_json adds segments mapped from the
|
||||
// Live API's incremental inputTranscription deltas. Those frames carry
|
||||
// NO timestamps, so segments expose {id,text} only (id = delta order,
|
||||
// Whisper-compatible 0-based) — start/end/duration are deliberately
|
||||
// absent rather than fabricated as zeros, which would misrepresent
|
||||
// provider data to callers diffing transports.
|
||||
const fmt = typeof formData?.get === "function"
|
||||
? String(formData.get("response_format") ?? "").trim().toLowerCase()
|
||||
: "";
|
||||
if (fmt === "verbose_json") {
|
||||
return jsonResponse({ text: live.text, segments: live.chunks.map((segText, id) => ({ id, text: segText })) });
|
||||
}
|
||||
return jsonResponse({ text: live.text });
|
||||
}
|
||||
case "deepgram": return await transcribeDeepgram(cfg, file, model, token, formData);
|
||||
case "assemblyai": return await transcribeAssemblyAI(cfg, file, model, token);
|
||||
case "nvidia-asr": return await transcribeNvidia(cfg, file, model, token);
|
||||
@@ -196,6 +234,6 @@ export async function handleSttCore({ provider, model, formData, credentials, st
|
||||
default: return await transcribeOpenAICompatible(cfg, file, model, token, formData);
|
||||
}
|
||||
} catch (err) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed");
|
||||
return createErrorResult(err.status || HTTP_STATUS.BAD_GATEWAY, err.message || "STT request failed");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -430,21 +430,29 @@ const TRUST_UPSTREAM_VISION = new Set(["openrouter"]);
|
||||
*
|
||||
* @param {string[]} comboModels
|
||||
* @param {Object|null} [comboLookup] optional map of combo name → models array for nested resolution
|
||||
* @param {Function|null} [resolveCaps] optional (fullId) → caps override. The synced model
|
||||
* catalog is server-only (it reads a file), so a browser-side resolution cannot see the
|
||||
* limits it supplies and silently falls back to the generic patterns below. Callers that
|
||||
* have the server's answer (/api/models, via useModelCaps) pass it here; it is merged over
|
||||
* the local tables, so fields it does not carry (tools, pdf, audio/video, thinking*) survive.
|
||||
* @param {number} [_depth] internal recursion depth guard
|
||||
* @returns {object|null} full capabilities object, or null for empty input
|
||||
*/
|
||||
export function aggregateComboCapabilities(comboModels, comboLookup = null, _depth = 0) {
|
||||
export function aggregateComboCapabilities(comboModels, comboLookup = null, resolveCaps = null, _depth = 0) {
|
||||
if (!comboModels?.length || _depth > 6) return null;
|
||||
const allCaps = comboModels.map((fullId) => {
|
||||
// Nested combo: bare name (no slash) that exists in the lookup — recurse
|
||||
if (!fullId.includes("/") && comboLookup?.[fullId]) {
|
||||
return aggregateComboCapabilities(comboLookup[fullId], comboLookup, _depth + 1)
|
||||
return aggregateComboCapabilities(comboLookup[fullId], comboLookup, resolveCaps, _depth + 1)
|
||||
?? resolveCaps?.(fullId)
|
||||
?? getCapabilitiesForModel(null, fullId);
|
||||
}
|
||||
const slash = fullId.indexOf("/");
|
||||
const provider = slash === -1 ? null : fullId.slice(0, slash);
|
||||
const model = slash === -1 ? fullId : fullId.slice(slash + 1);
|
||||
return getCapabilitiesForModel(provider, model);
|
||||
const local = getCapabilitiesForModel(provider, model);
|
||||
const override = resolveCaps?.(fullId);
|
||||
return override ? { ...local, ...override } : local;
|
||||
});
|
||||
const first = allCaps[0];
|
||||
return {
|
||||
@@ -482,7 +490,9 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
// handlers (silently: the setters still "succeed"). The slots therefore live on
|
||||
// globalThis, which IS shared across server bundles in the same process.
|
||||
// Same reason the browser bundle is safe: it never calls a setter, so the slots
|
||||
// stay empty and every consumer below short-circuits.
|
||||
// stay empty and every consumer below short-circuits. Every read goes through
|
||||
// globalThis: caching it locally would keep a reader alive in other copies after
|
||||
// setCatalogSource(null).
|
||||
let catalogSource = null;
|
||||
const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
|
||||
catalog: null, // { getModalities, getLimits } — synced models.dev catalog
|
||||
@@ -496,15 +506,13 @@ const SOURCE_SLOTS = (globalThis.__9R_CAPABILITY_SOURCES ||= {
|
||||
*/
|
||||
export function setCatalogSource(source) {
|
||||
catalogSource = source || null;
|
||||
SOURCE_SLOTS.catalog = source || null;
|
||||
if (SOURCE_SLOTS) SOURCE_SLOTS.catalog = source || null;
|
||||
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source || null;
|
||||
}
|
||||
|
||||
function getCatalogSource() {
|
||||
if (catalogSource) return catalogSource;
|
||||
if (SOURCE_SLOTS.catalog) return (catalogSource = SOURCE_SLOTS.catalog);
|
||||
if (typeof globalThis === "undefined") return null;
|
||||
return (catalogSource = globalThis.__9rCatalogSource || null);
|
||||
if (typeof globalThis === "undefined") return catalogSource;
|
||||
return SOURCE_SLOTS?.catalog || globalThis.__9rCatalogSource || null;
|
||||
}
|
||||
|
||||
// Capabilities the user asserted per provider+model (dashboard "Add/Edit Model"
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
|
||||
// Codex auto-generates a "-review" variant for each llm model (review quota family)
|
||||
export const CODEX_REVIEW_SUFFIX = "-review";
|
||||
|
||||
@@ -25,3 +27,20 @@ export function isMuseSparkModel(modelId) {
|
||||
const base = clean.includes("/") ? clean.split("/").pop() : clean;
|
||||
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
|
||||
}
|
||||
|
||||
// Endpoint families for OpenCode models outside the curated registry (modelsFetcher /
|
||||
// passthrough ids) — regex keeps auto-fetched models on the right endpoint:
|
||||
// /responses (gpt/grok/muse-spark), /messages (minimax/qwen), /chat/completions (rest).
|
||||
// Curated registry entries always win; this is the unknown-id fallback only.
|
||||
const OPENCODE_FAMILIES = [
|
||||
{ match: /^(grok|gpt|muse[-_]?spark)/i, supportedFormats: [FORMATS.OPENAI_RESPONSES], targetFormat: FORMATS.OPENAI_RESPONSES },
|
||||
{ match: /^deepseek-v4-(pro|flash)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE, FORMATS.OPENAI_RESPONSES] },
|
||||
{ match: /^(minimax|qwen)/, supportedFormats: [FORMATS.OPENAI, FORMATS.CLAUDE] },
|
||||
{ match: /^claude-/i, supportedFormats: [FORMATS.CLAUDE] },
|
||||
];
|
||||
|
||||
export function opencodeFamilyFormats(modelId) {
|
||||
if (!modelId || typeof modelId !== "string") return null;
|
||||
const base = modelId.replace(/\([^()]+\)\s*$/, "").trim();
|
||||
return OPENCODE_FAMILIES.find((f) => f.match.test(base)) || null;
|
||||
}
|
||||
|
||||
@@ -2,8 +2,28 @@
|
||||
//
|
||||
// Fallback order (first match wins):
|
||||
// 1. PROVIDER_PRICING[provider][model] — provider-specific override
|
||||
// 2. MODEL_PRICING[model] — canonical model price (provider-agnostic)
|
||||
// 3. PATTERN_PRICING — glob pattern match (e.g. "codex-*")
|
||||
// 2. FREE_MODEL_NAMESPACES — upstream bills these at $0
|
||||
// 3. MODEL_PRICING[model] — canonical model price (provider-agnostic)
|
||||
// 4. PATTERN_PRICING — glob pattern match (e.g. "codex-*")
|
||||
|
||||
/**
|
||||
* Namespaces upstream meters at $0. A free model must never inherit a paid
|
||||
* rate: the vendor-prefix strip in getPricingForModel() would turn
|
||||
* "cline-free/deepseek-v4.1-flash" into "deepseek-v4.1-flash" and match
|
||||
* MODEL_PRICING, so the namespace is checked before both fallbacks.
|
||||
*/
|
||||
export const FREE_MODEL_NAMESPACES = ["cline-free/"];
|
||||
|
||||
export const ZERO_PRICING = {
|
||||
input: 0, output: 0, cached: 0, reasoning: 0, cache_creation: 0,
|
||||
};
|
||||
|
||||
/** True when the model id sits in a namespace upstream bills at $0. */
|
||||
export function isFreeModel(model) {
|
||||
if (!model) return false;
|
||||
const lower = String(model).toLowerCase();
|
||||
return FREE_MODEL_NAMESPACES.some((ns) => lower.startsWith(ns));
|
||||
}
|
||||
|
||||
/**
|
||||
* Canonical model pricing — provider-agnostic.
|
||||
@@ -361,10 +381,11 @@ export function matchPattern(pattern, model) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve pricing for a model using the 3-step fallback chain:
|
||||
* Resolve pricing for a model using the 4-step fallback chain:
|
||||
* 1. PROVIDER_PRICING[provider][model]
|
||||
* 2. MODEL_PRICING[model]
|
||||
* 3. PATTERN_PRICING (glob match)
|
||||
* 2. free namespace (upstream bills $0)
|
||||
* 3. MODEL_PRICING[model]
|
||||
* 4. PATTERN_PRICING (glob match)
|
||||
*
|
||||
* @param {string} provider
|
||||
* @param {string} model
|
||||
@@ -378,12 +399,15 @@ export function getPricingForModel(provider, model) {
|
||||
return PROVIDER_PRICING[provider][model];
|
||||
}
|
||||
|
||||
// 2. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat")
|
||||
// 2. Free namespaces bill $0 regardless of the model name behind them.
|
||||
if (isFreeModel(model)) return ZERO_PRICING;
|
||||
|
||||
// 3. Canonical model pricing (strip vendor prefix if needed: "deepseek/deepseek-chat" → "deepseek-chat")
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
if (MODEL_PRICING[baseModel]) return MODEL_PRICING[baseModel];
|
||||
if (MODEL_PRICING[model]) return MODEL_PRICING[model];
|
||||
|
||||
// 3. Pattern match
|
||||
// 4. Pattern match
|
||||
for (const { pattern, pricing } of PATTERN_PRICING) {
|
||||
if (matchPattern(pattern, baseModel) || matchPattern(pattern, model)) {
|
||||
return pricing;
|
||||
|
||||
29
open-sse/providers/registry/agnes.js
Normal file
29
open-sse/providers/registry/agnes.js
Normal file
@@ -0,0 +1,29 @@
|
||||
export default {
|
||||
id: "agnes",
|
||||
priority: 120,
|
||||
alias: "agnes",
|
||||
aliases: [
|
||||
"agnes-ai",
|
||||
],
|
||||
uiAlias: "agnes",
|
||||
display: {
|
||||
name: "Agnes AI",
|
||||
icon: "auto_awesome",
|
||||
color: "#7C3AED",
|
||||
textIcon: "AG",
|
||||
website: "https://agnes-ai.com",
|
||||
notice: {
|
||||
text: "OpenAI-compatible gateway from Agnes AI, offering free API credits on sign-up. Accepts a bearer token or an x-api-key header.",
|
||||
apiKeyUrl: "https://platform.agnes-ai.com",
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions",
|
||||
validateUrl: "https://apihub.agnes-ai.com/v1/models",
|
||||
},
|
||||
// No model ids could be verified without a key, so discovery is left to the
|
||||
// live endpoint and any id is accepted through passthroughModels.
|
||||
passthroughModels: true,
|
||||
};
|
||||
32
open-sse/providers/registry/atria.js
Normal file
32
open-sse/providers/registry/atria.js
Normal file
@@ -0,0 +1,32 @@
|
||||
export default {
|
||||
id: "atria",
|
||||
priority: 120,
|
||||
alias: "atria",
|
||||
aliases: [
|
||||
"atria-asi",
|
||||
],
|
||||
uiAlias: "atria",
|
||||
display: {
|
||||
name: "Atria Dawn",
|
||||
icon: "flare",
|
||||
color: "#C2410C",
|
||||
textIcon: "AD",
|
||||
website: "https://atria-asi.ai",
|
||||
notice: {
|
||||
text: "OpenAI-compatible endpoint from Atria Dawn (AtomInnoLab). Currently a research preview offering a single text model, Atria-Dawn-Preview.",
|
||||
apiKeyUrl: "https://api.atria-asi.ai/dashboard",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://api.atria-asi.ai/v1/chat/completions",
|
||||
validateUrl: "https://api.atria-asi.ai/v1/models",
|
||||
},
|
||||
// Docs pin the model field to one case-sensitive id. Text-only for now: the
|
||||
// service ships a hook that blocks image/PDF input, so no vision is claimed.
|
||||
models: [
|
||||
{ id: "Atria-Dawn-Preview", name: "Atria Dawn Preview" },
|
||||
],
|
||||
passthroughModels: true,
|
||||
};
|
||||
30
open-sse/providers/registry/bai.js
Normal file
30
open-sse/providers/registry/bai.js
Normal file
@@ -0,0 +1,30 @@
|
||||
export default {
|
||||
id: "bai",
|
||||
priority: 120,
|
||||
alias: "bai",
|
||||
aliases: [
|
||||
"b-ai",
|
||||
],
|
||||
uiAlias: "bai",
|
||||
display: {
|
||||
name: "B.AI",
|
||||
icon: "account_balance",
|
||||
color: "#0369A1",
|
||||
textIcon: "BA",
|
||||
website: "https://b.ai",
|
||||
notice: {
|
||||
text: "OpenAI-compatible gateway with one of the larger catalogues here. Accepts a bearer token or an x-api-key header. Model ids are fetched live from the provider.",
|
||||
apiKeyUrl: "https://b.ai",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://api.b.ai/v1/chat/completions",
|
||||
validateUrl: "https://api.b.ai/v1/models",
|
||||
},
|
||||
// No ids hardcoded: the catalogue is large and rotates, so the live endpoint
|
||||
// is the source of truth and any id is accepted via passthroughModels.
|
||||
modelsFetcher: { url: "https://api.b.ai/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -54,6 +54,8 @@ export default {
|
||||
oauthUrl: "https://api.anthropic.com/api/oauth/usage",
|
||||
orgUrl: "https://api.anthropic.com/v1/organizations/{org_id}/usage",
|
||||
settingsUrl: "https://api.anthropic.com/v1/settings",
|
||||
profileUrl: "https://api.anthropic.com/api/oauth/profile",
|
||||
resetUrl: "https://api.anthropic.com/api/organizations/{org_id}/reset_rate_limits",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
|
||||
@@ -2,7 +2,8 @@ import { withCodexReviewModels } from "../models/helpers.js";
|
||||
|
||||
// Codex CLI version seen by OpenAI's backend — single source for the Version /
|
||||
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
|
||||
const CODEX_CLI_VERSION = "0.154.0";
|
||||
const CODEX_CLI_VERSION = "0.155.0";
|
||||
const GPT_6_LITE_THINKING_LEVELS = ["low", "medium", "high", "xhigh", "max"];
|
||||
|
||||
export default {
|
||||
id: "codex",
|
||||
@@ -42,6 +43,7 @@ export default {
|
||||
headers: {
|
||||
originator: "codex_cli_rs",
|
||||
"User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`,
|
||||
version: CODEX_CLI_VERSION,
|
||||
},
|
||||
usage: {
|
||||
url: "https://chatgpt.com/backend-api/wham/usage",
|
||||
@@ -51,6 +53,8 @@ export default {
|
||||
},
|
||||
models: [
|
||||
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
|
||||
{ id: "gpt-6-sol", name: "GPT 6.0 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
|
||||
{ id: "gpt-6-luna", name: "GPT 6.0 Luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
|
||||
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
|
||||
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
|
||||
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
|
||||
|
||||
35
open-sse/providers/registry/dahl.js
Normal file
35
open-sse/providers/registry/dahl.js
Normal file
@@ -0,0 +1,35 @@
|
||||
export default {
|
||||
id: "dahl",
|
||||
priority: 120,
|
||||
alias: "dahl",
|
||||
aliases: [
|
||||
"dahl-inference",
|
||||
],
|
||||
uiAlias: "dahl",
|
||||
display: {
|
||||
name: "Dahl Inference",
|
||||
icon: "hub",
|
||||
color: "#1E40AF",
|
||||
textIcon: "DH",
|
||||
website: "https://dahl.global",
|
||||
notice: {
|
||||
text: "OpenAI-compatible Gonka inference node. Small, fixed catalogue (GLM-5.3-Flash, DeepSeek-V4-Flash, MiniMax-M2.7) at a flat per-token rate.",
|
||||
apiKeyUrl: "https://dahl.global/dashboard",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://inference.dahl.global/v1/chat/completions",
|
||||
validateUrl: "https://inference.dahl.global/v1/models",
|
||||
},
|
||||
// The live catalogue is public (no auth), so modelsFetcher works without a key
|
||||
// and the ids below are a convenience seed rather than an exhaustive list.
|
||||
models: [
|
||||
{ id: "zai-org/GLM-5.3-Flash", name: "GLM-5.3 Flash" },
|
||||
{ id: "deepseek-ai/DeepSeek-V4-Flash-0731", name: "DeepSeek V4 Flash 0731" },
|
||||
{ id: "MiniMaxAI/MiniMax-M2.7", name: "MiniMax M2.7" },
|
||||
],
|
||||
modelsFetcher: { url: "https://inference.dahl.global/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -58,6 +58,7 @@ export default {
|
||||
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash", params: ["language","prompt"], kind: "stt" },
|
||||
{ id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash Lite (Cheapest)", params: ["language","prompt"], kind: "stt" },
|
||||
{ id: "gemini-2.0-flash", name: "Gemini 2.0 Flash", params: ["language","prompt"], kind: "stt" },
|
||||
{ id: "gemini-2.5-flash-native-audio-preview-09-17", name: "Gemini Live Transcription (Realtime)", params: ["language","prompt","system_instruction","setup_timeout_ms","turn_timeout_ms"], kind: "stt", transport: "gemini-live" },
|
||||
{ id: "gemini-3.1-flash-tts-preview", name: "Gemini 3.1 Flash TTS", kind: "tts" },
|
||||
{ id: "gemini-2.5-flash-preview-tts", name: "Gemini 2.5 Flash TTS", kind: "tts" },
|
||||
{ id: "gemini-2.5-pro-preview-tts", name: "Gemini 2.5 Pro TTS", kind: "tts" },
|
||||
|
||||
@@ -125,6 +125,11 @@ import p119 from "./selfhosted-embedding.js";
|
||||
import p120 from "./fish-audio.js";
|
||||
import p121 from "./alitp-intl.js";
|
||||
import p122 from "./xquik.js";
|
||||
import p125 from "./tokenharbor.js";
|
||||
import p126 from "./dahl.js";
|
||||
import p127 from "./atria.js";
|
||||
import p129 from "./agnes.js";
|
||||
import p130 from "./bai.js";
|
||||
export default [
|
||||
p0,
|
||||
p1,
|
||||
@@ -250,4 +255,9 @@ export default [
|
||||
p120,
|
||||
p121,
|
||||
p122,
|
||||
p125,
|
||||
p126,
|
||||
p127,
|
||||
p129,
|
||||
p130,
|
||||
];
|
||||
|
||||
@@ -35,37 +35,56 @@ export default {
|
||||
],
|
||||
// supportedFormats follow the endpoint table in https://opencode.ai/docs/go/
|
||||
models: [
|
||||
{ id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] },
|
||||
{ id: "deepseek-flash", name: "DeepSeek Flash", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5", name: "GLM 5", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.5", name: "Kimi K2.5", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.6-pro", name: "MiMo V2.6 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2-pro", name: "MiMo V2 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2-omni", name: "MiMo V2 Omni", supportedFormats: ["openai"] },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "space-bunny-free", name: "Space Bunny Free", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.5-plus", name: "Qwen 3.5 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] },
|
||||
{ id: "hy3", name: "Hy3", supportedFormats: ["openai"] },
|
||||
{ id: "hy3-preview", name: "Hy3 Preview", supportedFormats: ["openai"] },
|
||||
// In /zen/go/v1/models but absent from the docs endpoint table — chat lane is the fallback guess
|
||||
{ id: "omen-alpha", name: "Omen Alpha", supportedFormats: ["openai"] },
|
||||
// Served by /zen/go/v1/responses only — the responses-only entry forces chatCore
|
||||
// past the sourceFormat-matched transports into translation (see chatCore guard).
|
||||
{ id: "grok-4.7", name: "Grok 4.7", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "grok-4.5", name: "Grok 4.5", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "gpt-6-luna", name: "GPT 6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
],
|
||||
// Live catalogue; ids outside this curated list get their endpoint lane from the
|
||||
// family regex in providers/models/helpers.js (opencodeFamilyFormats).
|
||||
modelsFetcher: { url: "https://opencode.ai/zen/go/v1/models", type: "opencode-go" },
|
||||
passthroughModels: true,
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
|
||||
49
open-sse/providers/registry/tokenharbor.js
Normal file
49
open-sse/providers/registry/tokenharbor.js
Normal file
@@ -0,0 +1,49 @@
|
||||
export default {
|
||||
id: "tokenharbor",
|
||||
priority: 120,
|
||||
alias: "tokenharbor",
|
||||
aliases: [
|
||||
"th",
|
||||
"thh",
|
||||
],
|
||||
uiAlias: "tokenharbor",
|
||||
display: {
|
||||
name: "Token Harbor",
|
||||
icon: "anchor",
|
||||
color: "#0F766E",
|
||||
textIcon: "TH",
|
||||
website: "https://tokenharbor.ai",
|
||||
notice: {
|
||||
text: "OpenAI-compatible aggregator. One API key reaches every model, billed per-token from a prepaid wallet. Model ids are bare (e.g. claude-opus-5.5, gpt-6-astra, deepseek-v4.1-flash:free) and are fetched live from the provider.",
|
||||
apiKeyUrl: "https://tokenharbor.ai/dashboard",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
// OpenAI-compatible. `format` is left at the shared "openai" default and
|
||||
// `thinkingFormat` is deliberately NOT declared: Token Harbor forwards
|
||||
// requests verbatim, so each model must resolve its own thinking wire
|
||||
// format through providers/capabilities.js. Setting a provider-wide value
|
||||
// would force one format (e.g. claude-adaptive) onto every model.
|
||||
baseUrl: "https://tokenharbor.ai/v1/chat/completions",
|
||||
validateUrl: "https://tokenharbor.ai/v1/models",
|
||||
retry: {
|
||||
429: 2,
|
||||
},
|
||||
},
|
||||
// Curated seed; the live catalogue is fetched via modelsFetcher and any other
|
||||
// id is accepted via passthroughModels. Their catalogue rotates (the :free set
|
||||
// in particular), so this stays deliberately small and is only the offline
|
||||
// fallback. Ids are bare — Token Harbor does not prefix them by upstream vendor.
|
||||
models: [
|
||||
{ id: "claude-opus-5.5", name: "Claude Opus 5.5" },
|
||||
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "gpt-6-astra", name: "GPT-6 Astra" },
|
||||
{ id: "gpt-6-sol", name: "GPT-6 Sol" },
|
||||
{ id: "deepseek-v4.1-flash:free", name: "DeepSeek V4.1 Flash (Free)" },
|
||||
{ id: "grok-4.7", name: "Grok 4.7" },
|
||||
],
|
||||
modelsFetcher: { url: "https://tokenharbor.ai/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -77,6 +77,11 @@ export function selectAnthropicBeta(model = "", body = null) {
|
||||
return flags.join(",");
|
||||
}
|
||||
|
||||
export function mergeAnthropicBeta(...values) {
|
||||
const flags = values.flatMap((v) => (typeof v === "string" ? v.split(",") : [])).map((f) => f.trim()).filter(Boolean);
|
||||
return [...new Set(flags)].join(",");
|
||||
}
|
||||
|
||||
// Shared baseUrls
|
||||
export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
import { getCapabilitiesForModel } from "./capabilities.js";
|
||||
import { matchPattern } from "./pricing.js";
|
||||
import { resolveKiroEffortPath } from "../config/kiroConstants.js";
|
||||
import { getProviderModels } from "../config/providerModels.js";
|
||||
|
||||
// Shared level sets (deduped) — verified against provider docs + wire in thinkingUnified.applyFormat.
|
||||
const L = {
|
||||
@@ -42,6 +43,8 @@ const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
{ pattern: "*mimo*v2.6*", levels: ["none", "low", "medium", "high", "xhigh"] },
|
||||
// mimo-v2.5-pro on opencode-go rejects reasoning_effort "max" (probed live); v2.5 accepts it.
|
||||
{ pattern: "*mimo*v2.5-pro*", levels: ["none", "low", "medium", "high", "xhigh"] },
|
||||
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
|
||||
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
|
||||
// (disable thinking instead). none kept for the picker = disable.
|
||||
@@ -73,10 +76,14 @@ export function getThinkingLevels(provider, model) {
|
||||
if (provider === "kiro" && resolveKiroEffortPath(model) === null) return null;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (!caps.reasoning) return null;
|
||||
const baseId = String(model || "").replace(/\([^()]+\)\s*$/, "");
|
||||
const modelLevels = provider === "codex"
|
||||
? getProviderModels("cx").find((entry) => entry.id === baseId)?.thinkingLevels
|
||||
: null;
|
||||
const hit = PATTERN_THINKING.find((entry) =>
|
||||
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
|
||||
);
|
||||
let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
|
||||
let levels = modelLevels || hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
|
||||
if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none");
|
||||
return levels;
|
||||
}
|
||||
|
||||
@@ -1,6 +1,12 @@
|
||||
import { buildClineHeaders } from "../shared/clineAuth.js";
|
||||
|
||||
const CLINEPASS_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/models";
|
||||
// Cline's free tier is published here, not in /api/v1/models: the catalog
|
||||
// endpoint carries no `cline-free/*` ids at all. Cline's own SDK calls this
|
||||
// feed unauthenticated (sdk/packages/core/src/services/llms/cline-recommended-models.ts),
|
||||
// so no Authorization header is sent — adding one would only make the request
|
||||
// fail on a header the endpoint ignores.
|
||||
const CLINE_RECOMMENDED_MODELS_ENDPOINT = "https://api.cline.bot/api/v1/ai/cline/recommended-models";
|
||||
const FETCH_TIMEOUT_MS = 5000;
|
||||
|
||||
/**
|
||||
@@ -72,6 +78,40 @@ export async function resolveClinepassModels(credentials) {
|
||||
return models.length ? { models } : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch Cline's recommended-models feed and return only its `free[]` tier.
|
||||
* Returns null on any failure — the free tier is additive, so a dead feed must
|
||||
* never take the /api/v1/models catalog down with it.
|
||||
* @param {{accessToken?: string, apiKey?: string}} credentials
|
||||
* @returns {Promise<{id: string, name: string}[] | null>}
|
||||
*/
|
||||
async function fetchClineFreeTierModels() {
|
||||
const controller = new AbortController();
|
||||
const timer = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
|
||||
|
||||
try {
|
||||
const response = await fetch(CLINE_RECOMMENDED_MODELS_ENDPOINT, {
|
||||
method: "GET",
|
||||
headers: { Accept: "application/json" },
|
||||
signal: controller.signal,
|
||||
});
|
||||
|
||||
if (!response.ok) return null;
|
||||
|
||||
const json = await response.json();
|
||||
const free = Array.isArray(json?.free) ? json.free : [];
|
||||
if (!free.length) return null;
|
||||
|
||||
return free
|
||||
.filter((m) => typeof m?.id === "string" && m.id.trim() !== "")
|
||||
.map((m) => ({ id: m.id, name: m.name || m.id }));
|
||||
} catch {
|
||||
return null;
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch Cline live model catalog from Cline's /models endpoint.
|
||||
* Unlike resolveClinepassModels, this returns ALL models (including
|
||||
@@ -91,5 +131,15 @@ export async function resolveClineModels(credentials) {
|
||||
name: m.name || m.id,
|
||||
}));
|
||||
|
||||
return models.length ? { models } : null;
|
||||
// Free tier: /api/v1/models lists no `cline-free/*` ids, so merge the feed's
|
||||
// free[] in. First writer wins on a shared id, keeping the catalog's entry
|
||||
// for anything the two sources agree on.
|
||||
const freeTier = await fetchClineFreeTierModels();
|
||||
const byId = new Map(models.map((m) => [m.id, m]));
|
||||
for (const m of freeTier || []) {
|
||||
if (!byId.has(m.id)) byId.set(m.id, m);
|
||||
}
|
||||
const merged = Array.from(byId.values());
|
||||
|
||||
return merged.length ? { models: merged } : null;
|
||||
}
|
||||
|
||||
@@ -4,10 +4,10 @@
|
||||
|
||||
import { getGitHubUsage } from "./usage/github.js";
|
||||
import { getGeminiUsage, getAntigravityUsage } from "./usage/google.js";
|
||||
import { getClaudeUsage } from "./usage/claude.js";
|
||||
import { getClaudeUsage, consumeClaudeResetGrant } from "./usage/claude.js";
|
||||
import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits } from "./usage/codex.js";
|
||||
|
||||
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
|
||||
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits, consumeClaudeResetGrant };
|
||||
import { getKiroUsage } from "./usage/kiro.js";
|
||||
import { getMiniMaxUsage } from "./usage/minimax.js";
|
||||
import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js";
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../../providers/shared.js";
|
||||
import { ANTHROPIC_API_VERSION, CLAUDE_CLI_VERSION } from "../../providers/shared.js";
|
||||
import { U, parseResetTime } from "./shared.js";
|
||||
|
||||
// Claude API config (urls from registry, apiVersion is header logic kept here)
|
||||
@@ -11,7 +11,11 @@ const CLAUDE_CONFIG = {
|
||||
oauthUsageUrl: U("claude").oauthUrl,
|
||||
usageUrl: U("claude").orgUrl,
|
||||
settingsUrl: U("claude").settingsUrl,
|
||||
profileUrl: U("claude").profileUrl,
|
||||
resetUrl: U("claude").resetUrl,
|
||||
apiVersion: ANTHROPIC_API_VERSION,
|
||||
// Reset grants are gated by surface: only "(external, cli)" UA is eligible
|
||||
userAgent: `claude-cli/${CLAUDE_CLI_VERSION} (external, cli)`,
|
||||
};
|
||||
|
||||
// OAuth usage endpoint rate-limits (429); cool down per-token to stop hammering it.
|
||||
@@ -64,12 +68,14 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
|
||||
}
|
||||
|
||||
// Primary: OAuth usage endpoint (Claude Code consumer OAuth tokens)
|
||||
const oauthResponse = await proxyAwareFetch(CLAUDE_CONFIG.oauthUsageUrl, {
|
||||
// cedar_ember=1 adds the "limit reset" grant block (same flag Claude Code sends)
|
||||
const oauthResponse = await proxyAwareFetch(`${CLAUDE_CONFIG.oauthUsageUrl}?cedar_ember=1`, {
|
||||
method: "GET",
|
||||
headers: {
|
||||
"Authorization": `Bearer ${accessToken}`,
|
||||
"anthropic-beta": "oauth-2025-04-20",
|
||||
"anthropic-version": CLAUDE_CONFIG.apiVersion,
|
||||
"User-Agent": CLAUDE_CONFIG.userAgent,
|
||||
},
|
||||
}, proxyOptions);
|
||||
|
||||
@@ -129,6 +135,7 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
|
||||
return {
|
||||
plan: "Claude Code",
|
||||
extraUsage: data.extra_usage ?? null,
|
||||
resetCredits: parseClaudeResetGrants(data.cedar_ember),
|
||||
quotas,
|
||||
};
|
||||
}
|
||||
@@ -146,6 +153,70 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
|
||||
}
|
||||
}
|
||||
|
||||
// Free "limit reset" grants (Anthropic program id "cedar_ember").
|
||||
// Shape: { eligible, next_grant_id, grants: [{ id, resets_left, ends_at, paused, clears }] }
|
||||
export function parseClaudeResetGrants(block) {
|
||||
if (!block?.eligible || !Array.isArray(block.grants)) return null;
|
||||
const grants = block.grants.filter((g) => g?.id && !g.paused && Number(g.resets_left) > 0);
|
||||
const next = grants.find((g) => g.id === block.next_grant_id) || grants[0] || null;
|
||||
return {
|
||||
availableCount: grants.reduce((sum, g) => sum + Number(g.resets_left), 0),
|
||||
nextGrantId: next?.id || null,
|
||||
expiresAt: next?.ends_at || null,
|
||||
clears: next?.clears || [],
|
||||
cooldownUntil: block.cooldown_until || null,
|
||||
weeklyResetsAt: block.weekly_resets_at || null,
|
||||
grants: block.grants.filter((g) => g?.id).map((g) => ({
|
||||
id: g.id,
|
||||
label: g.label || "",
|
||||
resetsLeft: Number(g.resets_left) || 0,
|
||||
resetsTotal: Number(g.resets_total) || 0,
|
||||
startsAt: g.starts_at || null,
|
||||
endsAt: g.ends_at || null,
|
||||
clears: Array.isArray(g.clears) ? g.clears : [],
|
||||
paused: g.paused === true,
|
||||
usableNow: g.usable_now === true,
|
||||
useRequiresLimit: g.use_requires_limit !== false,
|
||||
})),
|
||||
};
|
||||
}
|
||||
|
||||
// Spend one reset grant: refills the limits listed in grant.clears. Irreversible.
|
||||
export async function consumeClaudeResetGrant(accessToken, grantId, proxyOptions = null) {
|
||||
if (!accessToken) throw new Error("No Claude access token available. Please re-authorize the connection.");
|
||||
if (!/^[a-z0-9_-]{1,40}$/.test(grantId || "")) throw new Error("Invalid reset grant id.");
|
||||
|
||||
const headers = {
|
||||
"Authorization": `Bearer ${accessToken}`,
|
||||
"anthropic-beta": "oauth-2025-04-20",
|
||||
"anthropic-version": CLAUDE_CONFIG.apiVersion,
|
||||
"User-Agent": CLAUDE_CONFIG.userAgent,
|
||||
"Content-Type": "application/json",
|
||||
};
|
||||
|
||||
const profileRes = await proxyAwareFetch(CLAUDE_CONFIG.profileUrl, { method: "GET", headers }, proxyOptions);
|
||||
const profile = await profileRes.json().catch(() => null);
|
||||
const orgId = profile?.organization?.uuid;
|
||||
if (!profileRes.ok || !orgId) throw new Error(`Cannot resolve Claude organization (${profileRes.status}).`);
|
||||
|
||||
const res = await proxyAwareFetch(CLAUDE_CONFIG.resetUrl.replace("{org_id}", orgId), {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({ program: "cedar_ember", grant_id: grantId, request_id: crypto.randomUUID() }),
|
||||
}, proxyOptions);
|
||||
const data = await res.json().catch(() => null);
|
||||
|
||||
usageCache.delete(accessToken); // next read must show refilled limits
|
||||
return {
|
||||
ok: res.ok && data?.result === "reset",
|
||||
status: res.status,
|
||||
result: data?.result || null,
|
||||
reason: data?.reason || null,
|
||||
resetsLeft: data?.resets_left ?? null,
|
||||
message: data?.error?.message || null,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Legacy Claude usage for API key / org admin users
|
||||
*/
|
||||
|
||||
@@ -333,6 +333,9 @@ export function createResponsesApiTransformStream(logger = null) {
|
||||
|
||||
// Regular text content
|
||||
if (content) {
|
||||
// The answer starts, so thinking is over. Upstreams that send reasoning via
|
||||
// reasoning_content never emit "</think>", so close it here rather than at finish.
|
||||
closeReasoning(controller);
|
||||
if (!state.msgItemAdded[idx]) {
|
||||
state.msgItemAdded[idx] = true;
|
||||
const msgId = `msg_${state.responseId}_${idx}`;
|
||||
@@ -372,6 +375,7 @@ export function createResponsesApiTransformStream(logger = null) {
|
||||
|
||||
// Handle tool_calls
|
||||
if (delta.tool_calls) {
|
||||
closeReasoning(controller);
|
||||
closeMessage(controller, idx);
|
||||
|
||||
for (const tc of delta.tool_calls) {
|
||||
|
||||
@@ -102,9 +102,28 @@ export function extractThinking(body) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Capture thinking intent from a body. Alias of extractThinking, named for clarity
|
||||
// at the call-site where intent is snapshotted before format translation.
|
||||
export const captureThinking = extractThinking;
|
||||
// Capture thinking intent from a body before format translation strips it.
|
||||
// Besides the effort, records whether an OpenAI-shaped client wants the thinking
|
||||
// text itself: Claude returns it only with thinking.display "summarized", a field
|
||||
// OpenAI has no equivalent for, so the intent cannot survive translation on its own.
|
||||
export function captureThinking(body) {
|
||||
const cfg = extractThinking(body);
|
||||
if (!cfg || cfg.mode === "none") return cfg;
|
||||
const display = openAIThinkingDisplay(body);
|
||||
return display ? { ...cfg, display } : cfg;
|
||||
}
|
||||
|
||||
function openAIThinkingDisplay(body) {
|
||||
// Responses API: reasoning.summary is the explicit request for reasoning text.
|
||||
if (body.reasoning && typeof body.reasoning === "object") {
|
||||
const summary = body.reasoning.summary;
|
||||
return typeof summary === "string" && summary && summary !== "none" ? "summarized" : undefined;
|
||||
}
|
||||
// Chat Completions has no summary knob. A client setting reasoning_effort is
|
||||
// asking for reasoning, and reasoning_content is how it would receive it.
|
||||
if (typeof body.reasoning_effort === "string") return "summarized";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]);
|
||||
|
||||
@@ -302,9 +321,12 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
|
||||
case "deepseek": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
body.thinking = { type: "enabled" };
|
||||
// DeepSeek: low/medium→high, xhigh/max→max.
|
||||
// DeepSeek: low/medium→high, xhigh/max→max. Some backends (mimo v2.5-pro/v2.6
|
||||
// on opencode-go, probed live) 400 on "max" — clamp to high when the declared
|
||||
// levels exclude it.
|
||||
const level = toLevel(eff);
|
||||
body.reasoning_effort = level === "xhigh" || level === "max" ? "max" : "high";
|
||||
const want = level === "xhigh" || level === "max" ? "max" : "high";
|
||||
body.reasoning_effort = want === "max" && supportedLevels && !supportedLevels.includes("max") ? "high" : want;
|
||||
break;
|
||||
}
|
||||
case "kimi": {
|
||||
@@ -380,7 +402,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
|
||||
const supportedLevels = getThinkingLevels(provider, cleanModel);
|
||||
// Anthropic's `display` (summarized | omitted) decides whether thinking text
|
||||
// comes back at all; keep what the client asked for instead of resetting it.
|
||||
const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined;
|
||||
// An OpenAI-shaped client's ask arrives via the captured intent instead.
|
||||
const display = typeof body.thinking?.display === "string" ? body.thinking.display : intent?.display;
|
||||
stripAll(body);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels, display);
|
||||
return body;
|
||||
|
||||
@@ -432,7 +432,7 @@ export function cleanJSONSchemaForAntigravity(schema) {
|
||||
return cleaned;
|
||||
}
|
||||
|
||||
// Merge adjacent same-role messages, strip empty parts, ensure initial user turn
|
||||
// Merge adjacent same-role messages, strip empty parts, ensure initial and terminal user turns
|
||||
export function normalizeGeminiContents(contents) {
|
||||
const out = [];
|
||||
for (const c of contents || []) {
|
||||
@@ -446,6 +446,23 @@ export function normalizeGeminiContents(contents) {
|
||||
if (out.length > 0 && out[0].role !== "user") {
|
||||
out.unshift({ role: "user", parts: [{ text: "..." }] });
|
||||
}
|
||||
if (out.length > 0 && out.at(-1).role === "model") {
|
||||
const fnCalls = (out.at(-1).parts || []).filter(p => p && p.functionCall);
|
||||
if (fnCalls.length > 0) {
|
||||
const responses = fnCalls.map(p => {
|
||||
const call = p.functionCall || {};
|
||||
const fr = {
|
||||
name: call.name || "tool",
|
||||
response: { result: "Continue." }
|
||||
};
|
||||
if (call.id) fr.id = call.id;
|
||||
return { functionResponse: fr };
|
||||
});
|
||||
out.push({ role: "user", parts: responses });
|
||||
} else {
|
||||
out.push({ role: "user", parts: [{ text: "Continue." }] });
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
|
||||
@@ -59,10 +59,6 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
}
|
||||
if (block?.type === CLAUDE_BLOCK.TEXT) {
|
||||
state.textBlockStarted = true;
|
||||
} else if (block?.type === CLAUDE_BLOCK.THINKING) {
|
||||
state.inThinkingBlock = true;
|
||||
state.currentBlockIndex = chunk.index;
|
||||
results.push(createChunk(state, { content: "<think>" }));
|
||||
} else if (block?.type === CLAUDE_BLOCK.TOOL_USE) {
|
||||
const toolCallIndex = state.toolCallIndex++;
|
||||
// Restore original tool name from mapping (Claude OAuth)
|
||||
@@ -89,6 +85,8 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
if (delta?.type === "text_delta" && delta.text) {
|
||||
results.push(createChunk(state, { content: delta.text }));
|
||||
} else if (delta?.type === "thinking_delta" && delta.thinking) {
|
||||
// Thinking travels only in reasoning_content. No "<think>" markers in
|
||||
// content: OpenAI-format clients render them as literal text.
|
||||
results.push(createChunk(state, reasoningDelta(delta.thinking)));
|
||||
} else if (delta?.type === "input_json_delta" && delta.partial_json) {
|
||||
const toolCall = state.toolCalls.get(chunk.index);
|
||||
@@ -112,10 +110,6 @@ export function claudeToOpenAIResponse(chunk, state) {
|
||||
state.serverToolBlockIndex = -1;
|
||||
break;
|
||||
}
|
||||
if (state.inThinkingBlock && chunk.index === state.currentBlockIndex) {
|
||||
results.push(createChunk(state, { content: "</think>" }));
|
||||
state.inThinkingBlock = false;
|
||||
}
|
||||
state.textBlockStarted = false;
|
||||
state.thinkingBlockStarted = false;
|
||||
break;
|
||||
|
||||
@@ -129,12 +129,16 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
||||
}
|
||||
|
||||
if (content) {
|
||||
// The answer starts, so thinking is over. Upstreams that send reasoning via
|
||||
// reasoning_content never emit "</think>", so close it here rather than at finish.
|
||||
closeReasoning(state, emit);
|
||||
emitTextContent(state, emit, idx, content);
|
||||
}
|
||||
}
|
||||
|
||||
// Handle tool_calls (empty array is truthy; require a real call)
|
||||
if (delta.tool_calls && delta.tool_calls.length) {
|
||||
closeReasoning(state, emit);
|
||||
closeMessage(state, emit, idx);
|
||||
for (const tc of delta.tool_calls) {
|
||||
emitToolCall(state, emit, tc);
|
||||
@@ -219,15 +223,19 @@ function closeReasoning(state, emit) {
|
||||
part: { type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }
|
||||
});
|
||||
|
||||
const item = {
|
||||
id: state.reasoningId,
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
|
||||
};
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: state.reasoningIndex,
|
||||
item: {
|
||||
id: state.reasoningId,
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: state.reasoningBuf }]
|
||||
}
|
||||
item
|
||||
});
|
||||
|
||||
recordCompletedOutputItem(state, state.reasoningIndex, item);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -291,16 +299,20 @@ function closeMessage(state, emit, idx) {
|
||||
part: { type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }
|
||||
});
|
||||
|
||||
const item = {
|
||||
id: msgId,
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
|
||||
role: ROLE.ASSISTANT
|
||||
};
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: parseInt(idx),
|
||||
item: {
|
||||
id: msgId,
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, annotations: [], logprobs: [], text: fullText }],
|
||||
role: ROLE.ASSISTANT
|
||||
}
|
||||
item
|
||||
});
|
||||
|
||||
recordCompletedOutputItem(state, parseInt(idx), item);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -394,23 +406,51 @@ function closeToolCall(state, emit, idx) {
|
||||
});
|
||||
}
|
||||
|
||||
const item = {
|
||||
id: `${custom ? "ctc" : "fc"}_${callId}`,
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
|
||||
call_id: callId,
|
||||
name: state.funcNames[idx] || ""
|
||||
};
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: parseInt(idx),
|
||||
item: {
|
||||
id: `${custom ? "ctc" : "fc"}_${callId}`,
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
|
||||
call_id: callId,
|
||||
name: state.funcNames[idx] || ""
|
||||
}
|
||||
item
|
||||
});
|
||||
|
||||
recordCompletedOutputItem(state, parseInt(idx), item);
|
||||
|
||||
state.funcItemDone[idx] = true;
|
||||
state.funcArgsDone[idx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
// response.completed carries the finished Response object, so response.output has
|
||||
// to repeat the items already delivered in response.output_item.done. Clients that
|
||||
// build their final result from the terminal event (GitHub Copilot CLI, the OpenAI
|
||||
// SDK "final response" helpers) otherwise treat the turn as empty even though the
|
||||
// text was streamed - see issue #4307.
|
||||
//
|
||||
// Keyed by output_index so a repeated close overwrites rather than duplicating the
|
||||
// item, and ordered by output_index so response.output matches the order the items
|
||||
// were emitted in. Lazily created because stream.js can hand us a state it built
|
||||
// itself rather than one from initState().
|
||||
function recordCompletedOutputItem(state, outputIndex, item) {
|
||||
state.completedOutputItems ??= new Map();
|
||||
const index = Number.isInteger(outputIndex) ? outputIndex : Number.parseInt(outputIndex, 10) || 0;
|
||||
state.completedOutputItems.set(index, item);
|
||||
}
|
||||
|
||||
function collectCompletedOutputItems(state) {
|
||||
const recorded = state.completedOutputItems;
|
||||
if (!(recorded instanceof Map) || recorded.size === 0) return [];
|
||||
return [...recorded.entries()]
|
||||
.sort((left, right) => left[0] - right[0])
|
||||
.map(([, item]) => item);
|
||||
}
|
||||
|
||||
function sendCompleted(state, emit) {
|
||||
if (!state.completedSent) {
|
||||
state.completedSent = true;
|
||||
@@ -423,6 +463,7 @@ function sendCompleted(state, emit) {
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null,
|
||||
output: collectCompletedOutputItems(state),
|
||||
...(state.responsesUsage ? { usage: state.responsesUsage } : {})
|
||||
}
|
||||
});
|
||||
|
||||
@@ -25,10 +25,25 @@ function deriveUuid(seed) {
|
||||
function generateFakeUserID(sessionId, apiKey) {
|
||||
const deviceId = apiKey ? createHash("sha256").update(`device:${apiKey}`).digest("hex") : randomBytes(32).toString("hex");
|
||||
const accountUuid = apiKey ? deriveUuid(`account:${apiKey}`) : randomUUID();
|
||||
const sessionUuid = sessionId || randomUUID();
|
||||
const cleanSessionId = typeof sessionId === "string" ? sessionId.replace(/^claude:/i, "").trim() : null;
|
||||
const sessionUuid = cleanSessionId || randomUUID();
|
||||
return `{"device_id":"${deviceId}","account_uuid":"${accountUuid}","session_id":"${sessionUuid}"}`;
|
||||
}
|
||||
|
||||
export function extractClaudeSessionIdFromUserId(userId) {
|
||||
if (typeof userId !== "string" || !userId) return null;
|
||||
if (userId[0] === "{") {
|
||||
try {
|
||||
const sid = JSON.parse(userId)?.session_id;
|
||||
return typeof sid === "string" && sid ? sid.replace(/^claude:/i, "").trim() || null : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
const clean = userId.replace(/^claude:/i, "").trim();
|
||||
return clean || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cloak tools before sending to Claude provider (anti-ban):
|
||||
* - Rename client tools with the CLAUDE_TOOL_SUFFIX ("_ide") in tools[] and messages[]
|
||||
@@ -89,14 +104,29 @@ export function cloakClaudeTools(body) {
|
||||
};
|
||||
}
|
||||
|
||||
// Strip a trailing CLAUDE_TOOL_SUFFIX from a cloaked name as a last-resort
|
||||
// fallback when the name isn't in toolNameMap (e.g. map lost across a retry/
|
||||
// reconnect). Never strips decoy names — those are meant to reach the client
|
||||
// unresolved so it can see "tool unavailable" instead of silently no-oping.
|
||||
function stripCloakSuffix(name) {
|
||||
if (typeof name !== "string" || !name.endsWith(CLAUDE_TOOL_SUFFIX)) return null;
|
||||
if (CC_DEFAULT_TOOLS.has(name)) return null;
|
||||
const original = name.slice(0, -CLAUDE_TOOL_SUFFIX.length);
|
||||
return original.length > 0 ? original : null;
|
||||
}
|
||||
|
||||
// Decloak tool_use names in non-streaming Claude response body (INPUT side)
|
||||
export function decloakToolNames(body, toolNameMap) {
|
||||
if (!toolNameMap?.size || !Array.isArray(body?.content)) return body;
|
||||
if (!Array.isArray(body?.content)) return body;
|
||||
const content = body.content.map(block => {
|
||||
if (block?.type === "tool_use" && toolNameMap.has(block.name)) {
|
||||
if (block?.type !== "tool_use") return block;
|
||||
if (toolNameMap?.has(block.name)) {
|
||||
return { ...block, name: toolNameMap.get(block.name) };
|
||||
}
|
||||
return block;
|
||||
// toolNameMap missing/stale for this name — fall back to suffix stripping
|
||||
// rather than forwarding an unresolvable "<tool>_ide" name to the client.
|
||||
const fallback = stripCloakSuffix(block.name);
|
||||
return fallback ? { ...block, name: fallback } : block;
|
||||
});
|
||||
return { ...body, content };
|
||||
}
|
||||
@@ -111,19 +141,21 @@ export function decloakToolNames(body, toolNameMap) {
|
||||
* name appears exactly once per call — on the content_block_start event of
|
||||
* a tool_use block; argument deltas carry no name.
|
||||
*
|
||||
* Unknown names (e.g. a CC decoy tool the model called anyway) pass through
|
||||
* unchanged, matching the non-streaming decloak behavior.
|
||||
* Falls back to stripping the literal CLAUDE_TOOL_SUFFIX when the name isn't
|
||||
* in toolNameMap (map lost across a retry/reconnect), matching the
|
||||
* non-streaming decloak behavior. Decoy tool names (real CC tool names) and
|
||||
* anything else pass through unchanged.
|
||||
*
|
||||
* @param {object|null} chunk - Parsed SSE event (may be null on stream flush)
|
||||
* @param {Map|null} toolNameMap - Suffixed → original name map from cloakClaudeTools()
|
||||
* @returns {object|null} The chunk, with the tool_use name restored when cloaked
|
||||
*/
|
||||
export function decloakStreamChunk(chunk, toolNameMap) {
|
||||
if (!toolNameMap?.size || !chunk || typeof chunk !== "object") return chunk;
|
||||
if (!chunk || typeof chunk !== "object") return chunk;
|
||||
if (chunk.type !== "content_block_start") return chunk;
|
||||
const block = chunk.content_block;
|
||||
if (block?.type !== "tool_use" || typeof block.name !== "string") return chunk;
|
||||
const original = toolNameMap.get(block.name);
|
||||
const original = toolNameMap?.get(block.name) || stripCloakSuffix(block.name);
|
||||
if (!original) return chunk;
|
||||
return { ...chunk, content_block: { ...block, name: original } };
|
||||
}
|
||||
|
||||
@@ -27,12 +27,13 @@ export function buildErrorBody(statusCode, message) {
|
||||
* @param {string} message - Error message
|
||||
* @returns {Response} HTTP Response object
|
||||
*/
|
||||
export function errorResponse(statusCode, message) {
|
||||
export function errorResponse(statusCode, message, extraHeaders = null) {
|
||||
return new Response(JSON.stringify(buildErrorBody(statusCode, message)), {
|
||||
status: statusCode,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*"
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
...extraHeaders
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -95,13 +96,13 @@ export async function parseUpstreamError(response, executor = null) {
|
||||
* @param {number} [resetsAtMs] - Optional precise cooldown expiry (ms epoch) for provider-specific quota errors
|
||||
* @returns {{ success: false, status: number, error: string, response: Response, resetsAtMs?: number }}
|
||||
*/
|
||||
export function createErrorResult(statusCode, message, resetsAtMs) {
|
||||
export function createErrorResult(statusCode, message, resetsAtMs, extraHeaders = null) {
|
||||
return {
|
||||
success: false,
|
||||
status: statusCode,
|
||||
error: message,
|
||||
resetsAtMs,
|
||||
response: errorResponse(statusCode, message)
|
||||
response: errorResponse(statusCode, message, extraHeaders)
|
||||
};
|
||||
}
|
||||
|
||||
@@ -113,7 +114,7 @@ export function createErrorResult(statusCode, message, resetsAtMs) {
|
||||
* @param {string} retryAfterHuman - Human-readable retry info e.g. "reset after 30s"
|
||||
* @returns {Response}
|
||||
*/
|
||||
export function unavailableResponse(statusCode, message, retryAfter, retryAfterHuman) {
|
||||
export function unavailableResponse(statusCode, message, retryAfter, retryAfterHuman, extraHeaders = null) {
|
||||
const retryAfterSec = Math.max(Math.ceil((new Date(retryAfter).getTime() - Date.now()) / 1000), 1);
|
||||
const msg = `${message} (${retryAfterHuman})`;
|
||||
return new Response(
|
||||
@@ -121,8 +122,10 @@ export function unavailableResponse(statusCode, message, retryAfter, retryAfterH
|
||||
{
|
||||
status: statusCode,
|
||||
headers: {
|
||||
...extraHeaders,
|
||||
"Content-Type": "application/json",
|
||||
"Retry-After": String(retryAfterSec)
|
||||
// Intentionally mis-cased to prevent duplicate headers
|
||||
"retry-after": String(retryAfterSec)
|
||||
}
|
||||
}
|
||||
);
|
||||
|
||||
12
open-sse/utils/upstreamHeaders.js
Normal file
12
open-sse/utils/upstreamHeaders.js
Normal file
@@ -0,0 +1,12 @@
|
||||
const FORWARDED = new Set(["retry-after", "x-should-retry"]);
|
||||
const FORWARDED_PREFIX = "anthropic-ratelimit-";
|
||||
|
||||
export function upstreamResponseHeaders(headers) {
|
||||
const out = {};
|
||||
if (typeof headers?.forEach !== "function") return out;
|
||||
headers.forEach((value, name) => {
|
||||
const key = name.toLowerCase();
|
||||
if (FORWARDED.has(key) || key.startsWith(FORWARDED_PREFIX)) out[key] = value;
|
||||
});
|
||||
return out;
|
||||
}
|
||||
Reference in New Issue
Block a user