merge: integrate origin/master (v0.5.50) into gitea/new_feature
- Resolve conflicts in chatCore handlers: keep apiKey/streamErrorPatterns from the details-filters feature, adopt origin's stripContinuityFields, customToolNames, cache-inclusive usage accounting, and Responses-API SSE→JSON conversion - Adopt origin's provider usage handlers (codebuddy-intl, qoder creds) and modality detection (audio/video inputs) - Keep requestDetails apiKey column (schema v2) + masked key persistence Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
This commit is contained in:
@@ -156,6 +156,13 @@ export const LOAD_CODE_ASSIST_HEADERS = {
|
||||
"Client-Metadata": JSON.stringify({ ideType: IDE_TYPE.ANTIGRAVITY, platform: getPlatformEnum(), pluginType: PLUGIN_TYPE.GEMINI }),
|
||||
};
|
||||
|
||||
// Real Antigravity IDE doesn't send X-Goog-Api-Client/Client-Metadata on loadCodeAssist/onboardUser —
|
||||
// Google's backend fingerprints those and silently refuses to provision a cloudaicompanionProject.
|
||||
export const ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT,
|
||||
};
|
||||
|
||||
export const LOAD_CODE_ASSIST_METADATA = {
|
||||
ideType: IDE_TYPE.ANTIGRAVITY,
|
||||
platform: getPlatformEnum(),
|
||||
@@ -176,7 +183,6 @@ export const OAUTH_ENDPOINTS = {
|
||||
google: { token: "https://oauth2.googleapis.com/token", auth: "https://accounts.google.com/o/oauth2/auth" },
|
||||
openai: { token: PROVIDER_OAUTH["codex"]?.tokenUrl, auth: PROVIDER_OAUTH["codex"]?.authorizeUrl },
|
||||
anthropic: { token: PROVIDER_OAUTH["claude"]?.tokenUrl, auth: "https://api.anthropic.com/v1/oauth/authorize" }, // ≠ claude.authorizeUrl (claude.ai login) — keep
|
||||
qwen: { token: PROVIDER_OAUTH["qwen"]?.tokenUrl, auth: PROVIDER_OAUTH["qwen"]?.deviceCodeUrl },
|
||||
iflow: { token: PROVIDER_OAUTH["iflow"]?.tokenUrl, auth: PROVIDER_OAUTH["iflow"]?.authorizeUrl },
|
||||
github: { token: PROVIDER_OAUTH["github"]?.tokenUrl, auth: PROVIDER_OAUTH["github"]?.authorizeUrl, deviceCode: PROVIDER_OAUTH["github"]?.deviceCodeUrl },
|
||||
};
|
||||
|
||||
@@ -33,6 +33,21 @@ const GEMINI_VOICES = [
|
||||
"Vindemiatrix", "Sadachbia", "Sadaltager", "Sulafat",
|
||||
].map((id) => ({ id, name: id, type: "tts" }));
|
||||
|
||||
// Xiaomi MiMo preset voices (from https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5).
|
||||
// Voice id is passed via `audio.voice`; `mimo_default` = default (冰糖 on CN cluster, Mia elsewhere).
|
||||
// Voices are language-independent — the spoken language is a separate hint, not bound to the voice.
|
||||
const MIMO_VOICES = [
|
||||
{ id: "mimo_default", name: "mimo_default" },
|
||||
{ id: "冰糖", name: "冰糖" },
|
||||
{ id: "茉莉", name: "茉莉" },
|
||||
{ id: "苏打", name: "苏打" },
|
||||
{ id: "白桦", name: "白桦" },
|
||||
{ id: "Mia", name: "Mia" },
|
||||
{ id: "Chloe", name: "Chloe" },
|
||||
{ id: "Milo", name: "Milo" },
|
||||
{ id: "Dean", name: "Dean" },
|
||||
].map((v) => ({ type: "tts", ...v }));
|
||||
|
||||
// ── TTS Config (config-driven, single source of truth) ─────────────────────
|
||||
export const TTS_MODELS_CONFIG = {
|
||||
openai: {
|
||||
@@ -107,6 +122,14 @@ export const TTS_MODELS_CONFIG = {
|
||||
},
|
||||
allVoices: GEMINI_VOICES,
|
||||
},
|
||||
"xiaomi-mimo": {
|
||||
models: [
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", type: "tts" },
|
||||
],
|
||||
voices: {
|
||||
"mimo-v2.5-tts": MIMO_VOICES,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
// ── Helper: get voices for a specific model ────────────────────────────────
|
||||
|
||||
@@ -4,6 +4,7 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
|
||||
/**
|
||||
* BaseExecutor - Base class for provider executors
|
||||
@@ -31,7 +32,7 @@ export class BaseExecutor {
|
||||
if (this.provider?.startsWith?.("openai-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || OPENAI_COMPAT_BASE;
|
||||
const normalized = baseUrl.replace(/\/$/, "");
|
||||
const path = this.provider.includes("responses") ? "/responses" : "/chat/completions";
|
||||
const path = resolveOpenAICompatibleApiType(this.provider, credentials) === "responses" ? "/responses" : "/chat/completions";
|
||||
return `${normalized}${path}`;
|
||||
}
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
@@ -127,7 +128,7 @@ export class BaseExecutor {
|
||||
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
|
||||
const url = this.buildUrl(model, stream, urlIndex, credentials);
|
||||
const transformedBody = this.transformRequest(model, body, stream, credentials);
|
||||
const headers = this.buildHeaders(credentials, stream, url);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model);
|
||||
|
||||
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
|
||||
|
||||
|
||||
@@ -18,6 +18,35 @@ export class CodeBuddyExecutor extends DefaultExecutor {
|
||||
const transformed = super.transformRequest(model, body, stream, credentials);
|
||||
transformed.stream = true;
|
||||
|
||||
// Tencent's content filter flags CLI agent system prompts ("You are Claude
|
||||
// Code, Anthropic's official CLI...") as prompt injection / sensitive content
|
||||
// and rejects the whole request. Detect agent system prompts (length catch-all
|
||||
// + identity-marker regex) and replace them with a neutral one, while leaving
|
||||
// legitimate user system prompts untouched. content may be a string or typed
|
||||
// blocks ([{type:"text",text}]) depending on the incoming client format, so
|
||||
// flatten before matching and preserve the original shape on replacement.
|
||||
const NEUTRAL_PROMPT = "You are a helpful AI assistant that helps with software engineering tasks.";
|
||||
const AGENT_PATTERN = /you are claude code|claude.?code.+official.+cli|anthropic.+official.+cli|anxthxropic.+official.+cli|you are (?:cursor|windsurf|cline|aider|continue|copilot|cody)|you are an? (?:ai )?(?:coding |code )?agent|cc_entrypoint\s*=\s*(?:cli|vscode|jetbrains|gui)|claude.?code.+issues|give feedback.+claude.?code|you are .{0,30}(?:powerful )?ai agent|orchestration capabilities|OhMyOpenCode|<agent-identity>|<Role>|<Behavior_Instructions>/i;
|
||||
const flatten = (content) =>
|
||||
typeof content === "string"
|
||||
? content
|
||||
: Array.isArray(content)
|
||||
? content.map((b) => (b && typeof b.text === "string" ? b.text : "")).join("\n")
|
||||
: "";
|
||||
if (Array.isArray(transformed.messages)) {
|
||||
transformed.messages = transformed.messages.map((message) => {
|
||||
if (!message || message.role !== "system") return message;
|
||||
const text = flatten(message.content);
|
||||
if (!text) return message;
|
||||
if (text.length > 2000 || AGENT_PATTERN.test(text)) {
|
||||
return typeof message.content === "string"
|
||||
? { ...message, content: NEUTRAL_PROMPT }
|
||||
: { ...message, content: [{ type: "text", text: NEUTRAL_PROMPT }] };
|
||||
}
|
||||
return message;
|
||||
});
|
||||
}
|
||||
|
||||
// CodeBuddy only surfaces model reasoning when the request carries the CLI's
|
||||
// OpenAI-style params: reasoning_effort + reasoning_summary:"auto". 9router's
|
||||
// thinking pipeline sets reasoning_effort only when the client asks, and never
|
||||
|
||||
@@ -23,6 +23,20 @@ export class CodeBuddyIntlExecutor extends DefaultExecutor {
|
||||
} else if (eff) {
|
||||
transformed.reasoning_summary = "auto";
|
||||
}
|
||||
|
||||
// CodeBuddy rejects plain OpenAI shape (11101 invalid request): needs a
|
||||
// leading system prompt + user content as typed blocks, not a bare string.
|
||||
const source = Array.isArray(transformed.messages) ? transformed.messages : [];
|
||||
transformed.messages = [{ role: "system", content: "You are CodeBuddy Code." }];
|
||||
for (const message of source) {
|
||||
if (!message || typeof message !== "object" || ["system", "developer"].includes(message.role)) continue;
|
||||
if (message.role === "user" && typeof message.content === "string") {
|
||||
transformed.messages.push({ ...message, content: [{ type: "text", text: message.content }] });
|
||||
} else {
|
||||
transformed.messages.push({ ...message });
|
||||
}
|
||||
}
|
||||
|
||||
return transformed;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
@@ -124,8 +125,12 @@ function resolveCacheSessionId(body, credentials) {
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeReasoningEffort(value) {
|
||||
return value === "max" ? "xhigh" : value;
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
}
|
||||
|
||||
function findNestedMessage(value, depth = 0) {
|
||||
@@ -440,10 +445,10 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = normalizeReasoningEffort(body.reasoning_effort || modelEffort || 'low');
|
||||
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.reasoning.effort);
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS, PROVIDER_OAUTH } from "../config/providers.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE } from "../providers/shared.js";
|
||||
import { ANTHROPIC_API_VERSION, OPENAI_COMPAT_BASE, ANTHROPIC_COMPAT_BASE, selectAnthropicBeta } from "../providers/shared.js";
|
||||
import { resolveOpenAICompatibleApiType } from "../services/provider.js";
|
||||
import { OAUTH_ENDPOINTS, buildKimiHeaders } from "../config/appConstants.js";
|
||||
import { buildClineHeaders } from "../shared/clineAuth.js";
|
||||
import { getCachedClaudeHeaders } from "../utils/claudeHeaderCache.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
|
||||
@@ -42,21 +42,6 @@ const HEADER_HOOKS = {
|
||||
kimiHeaders: (h, c) => Object.assign(h, buildKimiHeaders(c?.providerSpecificData?.deviceId)),
|
||||
clineHeaders: (h, c) => Object.assign(h, buildClineHeaders(c.apiKey || c.accessToken)),
|
||||
kilocodeOrg: (h, c) => { if (c.providerSpecificData?.orgId) h["X-Kilocode-OrganizationID"] = c.providerSpecificData.orgId; },
|
||||
claudeOverlay: (h) => {
|
||||
const cached = getCachedClaudeHeaders();
|
||||
if (!cached) return;
|
||||
for (const lcKey of Object.keys(cached)) {
|
||||
const titleKey = lcKey.replace(/(^|-)([a-z])/g, (_, sep, ch) => sep + ch.toUpperCase());
|
||||
if (lcKey === "anthropic-beta") {
|
||||
const staticBetaStr = h[titleKey] || h[lcKey] || "";
|
||||
const flags = new Set(staticBetaStr.split(",").map(f => f.trim()).filter(Boolean));
|
||||
for (const f of cached[lcKey].split(",").map(f => f.trim()).filter(Boolean)) flags.add(f);
|
||||
cached[lcKey] = Array.from(flags).join(",");
|
||||
}
|
||||
if (titleKey !== lcKey && h[titleKey] !== undefined) delete h[titleKey];
|
||||
}
|
||||
Object.assign(h, cached);
|
||||
},
|
||||
};
|
||||
|
||||
// Config-driven OAuth refresh grants — derived from registry oauth.refresh.
|
||||
@@ -125,7 +110,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
if (this.provider?.startsWith?.("openai-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || OPENAI_COMPAT_BASE;
|
||||
const normalized = baseUrl.replace(/\/$/, "");
|
||||
const path = this.provider.includes("responses") ? "/responses" : "/chat/completions";
|
||||
const path = resolveOpenAICompatibleApiType(this.provider, credentials) === "responses" ? "/responses" : "/chat/completions";
|
||||
return `${normalized}${path}`;
|
||||
}
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
@@ -161,14 +146,18 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
return BEARER;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
const rt = credentials?.runtimeTransport;
|
||||
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
|
||||
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
|
||||
// Hooks run BEFORE auth so dynamic overlays (claude cached headers) can't clobber the token.
|
||||
// Hooks run BEFORE auth so dynamic overlays can't clobber the token.
|
||||
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
|
||||
applyAuth(headers, desc, credentials);
|
||||
|
||||
if (this.provider === "claude" && model) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
|
||||
}
|
||||
|
||||
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
|
||||
if (this.provider?.startsWith?.("anthropic-compatible-")) {
|
||||
const baseUrl = credentials?.providerSpecificData?.baseUrl || "";
|
||||
@@ -222,7 +211,6 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
const refreshers = {
|
||||
claude: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
codex: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
qwen: () => this.refreshWithForm(OAUTH_ENDPOINTS.qwen.token, { grant_type: "refresh_token", refresh_token: credentials.refreshToken, client_id: PROVIDERS.qwen.clientId }, proxyOptions),
|
||||
iflow: () => this.refreshIflow(credentials.refreshToken, proxyOptions),
|
||||
gemini: () => this.refreshFromGrant(credentials, proxyOptions),
|
||||
kiro: () => this.refreshKiro(credentials.refreshToken, proxyOptions),
|
||||
|
||||
@@ -9,7 +9,6 @@ import { KimchiExecutor } from "./kimchi.js";
|
||||
import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
import { QwenExecutor } from "./qwen.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
@@ -41,7 +40,6 @@ const executors = {
|
||||
cu: new CursorExecutor(), // Alias for cursor
|
||||
vertex: new VertexExecutor("vertex"),
|
||||
"vertex-partner": new VertexExecutor("vertex-partner"),
|
||||
qwen: new QwenExecutor(),
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
@@ -87,7 +85,6 @@ export { CodexExecutor } from "./codex.js";
|
||||
export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
export { DefaultExecutor } from "./default.js";
|
||||
export { QwenExecutor } from "./qwen.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
|
||||
@@ -33,13 +33,11 @@ import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
|
||||
import {
|
||||
QODER_CHAT_URL_ENCODED,
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
QODER_USERINFO_URL,
|
||||
QODER_CHAT_BASE_ALT,
|
||||
QODER_CHAT_SIG_PATH,
|
||||
QODER_MODEL_MAP,
|
||||
QODER_IDE_VERSION,
|
||||
QODER_CLIENT_TYPE,
|
||||
} from "../shared/qoder/constants.js";
|
||||
import { getQoderModelConfig, resolveQoderModels } from "../services/qoderModels.js";
|
||||
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
|
||||
|
||||
/**
|
||||
* Hoist role:"system" messages out of the messages array (Qoder rejects
|
||||
@@ -343,98 +341,18 @@ function wrapQoderSSE(response, model) {
|
||||
});
|
||||
}
|
||||
|
||||
// ── PAT (Personal Access Token) → job-token exchange ───────────────────────
|
||||
// PATs (pt-...) cannot sign COSY requests directly. Exchange them for a
|
||||
// short-lived job token (jt-...) via /api/v1/jobToken/exchange (plain JSON,
|
||||
// not COSY-signed), then resolve the userId from userinfo. Mirrors the
|
||||
// official qodercli flow. Cached per-PAT until near-expiry.
|
||||
const PAT_PREFIX = "pt-";
|
||||
const PAT_REFRESH_BUFFER_MS = 5 * 60 * 1000;
|
||||
const patJobCache = new Map();
|
||||
|
||||
export function isQoderPat(token) {
|
||||
return typeof token === "string" && token.startsWith(PAT_PREFIX);
|
||||
}
|
||||
|
||||
async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
{
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
"Cosy-Version": QODER_IDE_VERSION,
|
||||
"Cosy-ClientType": QODER_CLIENT_TYPE,
|
||||
},
|
||||
body: JSON.stringify({ personal_token: pat }),
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
throw new Error(`qoder PAT exchange failed: ${res.status} ${text.slice(0, 200)}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
if (!data.token) throw new Error("qoder PAT exchange returned no job token");
|
||||
|
||||
let expiresAt = Date.now() + 24 * 60 * 60 * 1000;
|
||||
if (data.expires_at) {
|
||||
const parsed = Date.parse(data.expires_at);
|
||||
if (!Number.isNaN(parsed)) expiresAt = parsed;
|
||||
} else if (typeof data.expires_in === "number" && data.expires_in > 0) {
|
||||
expiresAt = Date.now() + data.expires_in;
|
||||
}
|
||||
return { jobToken: data.token, jobRefreshToken: data.refresh_token || "", expiresAt };
|
||||
}
|
||||
|
||||
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null) {
|
||||
try {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_USERINFO_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${jobToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
},
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
const info = await res.json().catch(() => ({}));
|
||||
return info.id || info.userId || info.user_id || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Exchange a PAT for a job token + userId, caching until near-expiry so repeat
|
||||
* chat requests don't re-exchange. Returns { accessToken, userId }.
|
||||
*/
|
||||
async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
|
||||
const cached = patJobCache.get(pat);
|
||||
if (cached && cached.expiresAt - Date.now() > PAT_REFRESH_BUFFER_MS) {
|
||||
return cached;
|
||||
}
|
||||
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal);
|
||||
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal);
|
||||
const entry = { accessToken: jobToken, userId, expiresAt };
|
||||
patJobCache.set(pat, entry);
|
||||
return entry;
|
||||
}
|
||||
|
||||
export class QoderExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("qoder", PROVIDERS.qoder);
|
||||
}
|
||||
|
||||
buildUrl() {
|
||||
buildUrl(credentials) {
|
||||
// Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt-
|
||||
// with "Login expired" (403). Device tokens (dt-...) stay on api3.
|
||||
const raw = credentials?.apiKey || credentials?.accessToken;
|
||||
if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) {
|
||||
return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
|
||||
}
|
||||
return QODER_CHAT_URL_ENCODED;
|
||||
}
|
||||
|
||||
@@ -444,36 +362,24 @@ export class QoderExecutor extends BaseExecutor {
|
||||
// - COSY headers built from the *encoded* body bytes
|
||||
// - response stream re-wrapped from {statusCodeValue, body} to OpenAI SSE
|
||||
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
|
||||
const url = this.buildUrl();
|
||||
|
||||
// PAT (pt-...) → exchange for short-lived job token + resolve userId so
|
||||
// downstream COSY signing + catalog fetch work. Device tokens (dt-...) and
|
||||
// job tokens (jt-...) skip this and are used directly.
|
||||
const rawToken = credentials?.apiKey || credentials?.accessToken;
|
||||
if (isQoderPat(rawToken)) {
|
||||
try {
|
||||
const resolved = await resolvePatCredential(rawToken, proxyOptions, signal);
|
||||
credentials = {
|
||||
...credentials,
|
||||
accessToken: resolved.accessToken,
|
||||
apiKey: undefined,
|
||||
providerSpecificData: {
|
||||
authMethod: "pat",
|
||||
...(credentials?.providerSpecificData || {}),
|
||||
userId: resolved.userId || credentials?.providerSpecificData?.userId || "",
|
||||
machineId: credentials?.providerSpecificData?.machineId || "",
|
||||
},
|
||||
};
|
||||
credentials = await resolveQoderCredentials(credentials, proxyOptions, signal);
|
||||
} catch (err) {
|
||||
log?.error?.("QODER", `PAT exchange failed: ${err.message}`);
|
||||
const fakeResp = new Response(
|
||||
JSON.stringify({ error: { message: `qoder PAT exchange failed: ${err.message}` } }),
|
||||
{ status: 401, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
return { response: fakeResp, url, headers: {}, transformedBody: body };
|
||||
return { response: fakeResp, url: this.buildUrl(credentials), headers: {}, transformedBody: body };
|
||||
}
|
||||
}
|
||||
|
||||
const url = this.buildUrl(credentials);
|
||||
const psd = credentials?.providerSpecificData || {};
|
||||
if (!psd.userId) {
|
||||
// No user id → no way to sign. Surface a 401 so the dashboard nudges
|
||||
@@ -591,6 +497,4 @@ export const __test__ = {
|
||||
normalizeMessages,
|
||||
wrapQoderSSE,
|
||||
buildQoderRequestBody,
|
||||
isQoderPat,
|
||||
resolvePatCredential,
|
||||
};
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS } from "../config/appConstants.js";
|
||||
|
||||
/** portal.qwen.ai — static fingerprint matching stable Qwen Code release */
|
||||
const QWEN_USER_AGENT = "QwenCode/0.12.3 (linux; x64)";
|
||||
const QWEN_STAINLESS = {
|
||||
os: "Linux",
|
||||
arch: "x64",
|
||||
lang: "js",
|
||||
runtime: "node",
|
||||
runtimeVersion: "v18.19.1",
|
||||
packageVersion: "5.11.0",
|
||||
retryCount: "1"
|
||||
};
|
||||
const QWEN_DEFAULT_SYSTEM_MESSAGE = {
|
||||
role: "system",
|
||||
content: [{ type: "text", text: "", cache_control: { type: "ephemeral" } }]
|
||||
};
|
||||
|
||||
function ensureQwenSystemMessage(body) {
|
||||
if (!body || typeof body !== "object") return body;
|
||||
const next = { ...body };
|
||||
if (Array.isArray(next.messages)) {
|
||||
next.messages = [QWEN_DEFAULT_SYSTEM_MESSAGE, ...next.messages];
|
||||
} else {
|
||||
next.messages = [QWEN_DEFAULT_SYSTEM_MESSAGE];
|
||||
}
|
||||
return next;
|
||||
}
|
||||
|
||||
function isQwenThinkingActive(body) {
|
||||
const thinking = body?.thinking;
|
||||
if (thinking === true || body?.enable_thinking === true) return true;
|
||||
return typeof thinking === "object" && thinking !== null && !Array.isArray(thinking) && thinking.type === "enabled";
|
||||
}
|
||||
|
||||
// Qwen rejects tool_choice="required" or object forms when thinking is active; neutralize to "auto".
|
||||
function sanitizeQwenThinkingToolChoice(body) {
|
||||
if (!isQwenThinkingActive(body)) return body;
|
||||
const tc = body.tool_choice;
|
||||
const incompatible = tc === "required" || (typeof tc === "object" && tc !== null);
|
||||
if (!incompatible) return body;
|
||||
return { ...body, tool_choice: "auto" };
|
||||
}
|
||||
|
||||
function buildQwenUpstreamHeaders(credentials, stream = true) {
|
||||
const token = credentials?.apiKey || credentials?.accessToken || "";
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${token}`,
|
||||
"User-Agent": QWEN_USER_AGENT,
|
||||
"X-DashScope-AuthType": "qwen-oauth",
|
||||
"X-DashScope-CacheControl": "enable",
|
||||
"X-DashScope-UserAgent": QWEN_USER_AGENT,
|
||||
"X-Stainless-Arch": QWEN_STAINLESS.arch,
|
||||
"X-Stainless-Lang": QWEN_STAINLESS.lang,
|
||||
"X-Stainless-Os": QWEN_STAINLESS.os,
|
||||
"X-Stainless-Package-Version": QWEN_STAINLESS.packageVersion,
|
||||
"X-Stainless-Retry-Count": QWEN_STAINLESS.retryCount,
|
||||
"X-Stainless-Runtime": QWEN_STAINLESS.runtime,
|
||||
"X-Stainless-Runtime-Version": QWEN_STAINLESS.runtimeVersion,
|
||||
Connection: "keep-alive",
|
||||
"Accept-Language": "*",
|
||||
"Sec-Fetch-Mode": "cors"
|
||||
};
|
||||
headers.Accept = stream ? "text/event-stream" : "application/json";
|
||||
return headers;
|
||||
}
|
||||
|
||||
export class QwenExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("qwen");
|
||||
}
|
||||
|
||||
// Qwen tokens are bound to a resource_url returned at OAuth time.
|
||||
// Using portal.qwen.ai when the token is issued for another shard returns 401/403.
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
const resourceUrl = credentials?.providerSpecificData?.resourceUrl;
|
||||
const host = resourceUrl ? resourceUrl.replace(/^https?:\/\//, "").replace(/\/$/, "") : "portal.qwen.ai";
|
||||
return `https://${host}/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
return buildQwenUpstreamHeaders(credentials, stream);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
let next = body && typeof body === "object" ? { ...body } : body;
|
||||
if (stream && next?.messages && !next.stream_options && !next.thinking && !next.enable_thinking && next.stream !== false) {
|
||||
next.stream_options = { include_usage: true };
|
||||
}
|
||||
next = sanitizeQwenThinkingToolChoice(next);
|
||||
return ensureQwenSystemMessage(next);
|
||||
}
|
||||
|
||||
// Override to capture resource_url from refresh response (required for buildUrl).
|
||||
async refreshCredentials(credentials, log) {
|
||||
if (!credentials?.refreshToken) return null;
|
||||
try {
|
||||
const response = await fetch(OAUTH_ENDPOINTS.qwen.token, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json" },
|
||||
body: new URLSearchParams({
|
||||
grant_type: "refresh_token",
|
||||
refresh_token: credentials.refreshToken,
|
||||
client_id: PROVIDERS.qwen.clientId
|
||||
})
|
||||
});
|
||||
if (!response.ok) return null;
|
||||
const tokens = await response.json();
|
||||
log?.info?.("TOKEN", "qwen refreshed");
|
||||
return {
|
||||
accessToken: tokens.access_token,
|
||||
refreshToken: tokens.refresh_token || credentials.refreshToken,
|
||||
expiresIn: tokens.expires_in,
|
||||
providerSpecificData: {
|
||||
...(credentials.providerSpecificData || {}),
|
||||
...(tokens.resource_url ? { resourceUrl: tokens.resource_url } : {})
|
||||
}
|
||||
};
|
||||
} catch (error) {
|
||||
log?.error?.("TOKEN", `qwen refresh error: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export default QwenExecutor;
|
||||
@@ -4,7 +4,7 @@ import {
|
||||
resolveTransport,
|
||||
} from "../services/provider.js";
|
||||
import { translateRequest } from "../translator/index.js";
|
||||
import { stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { applyThinking, extractThinking, stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { normalizeClaudePassthrough } from "../translator/formats/claude.js";
|
||||
import { createStreamController } from "../utils/streamHandler.js";
|
||||
@@ -61,7 +61,6 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
|
||||
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
import { extractThinking } from "../translator/concerns/thinkingUnified.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
/**
|
||||
@@ -71,6 +70,26 @@ import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
* @param {object} options.credentials - Provider credentials
|
||||
* @param {string} options.sourceFormatOverride - Override detected source format (e.g. "openai-responses")
|
||||
*/
|
||||
/**
|
||||
* Remove translator-internal continuity fields from the outbound upstream
|
||||
* body. The Responses→Chat request translator stashes reasoning
|
||||
* `encrypted_content` on assistant messages so a later openai→responses
|
||||
* round-trip can restore the store=false continuity blob; that stash must
|
||||
* never reach an upstream provider. Chat-native proxies reject the unknown
|
||||
* assistant-message field and answer every turn with a literal "400" body
|
||||
* (observed with multi-turn Codex sessions via OpenAI-compatible nodes).
|
||||
*/
|
||||
export function stripContinuityFields(body) {
|
||||
if (!body || !Array.isArray(body.messages)) return body;
|
||||
for (const msg of body.messages) {
|
||||
if (msg && typeof msg === "object") {
|
||||
delete msg.encrypted_content;
|
||||
delete msg.reasoning_encrypted_content;
|
||||
}
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
export async function handleChatCore({
|
||||
body,
|
||||
modelInfo,
|
||||
@@ -138,7 +157,7 @@ export async function handleChatCore({
|
||||
// Multi-endpoint providers: pick transport matching sourceFormat → zero translation
|
||||
const runtimeTransport = resolveTransport(provider, sourceFormat);
|
||||
const targetFormat =
|
||||
modelTargetFormat || runtimeTransport?.format || getTargetFormat(provider);
|
||||
modelTargetFormat || runtimeTransport?.format || getTargetFormat(provider, credentials);
|
||||
if (runtimeTransport && credentials)
|
||||
credentials.runtimeTransport = runtimeTransport;
|
||||
const stripList = getModelStrip(alias, model);
|
||||
@@ -248,12 +267,25 @@ export async function handleChatCore({
|
||||
|
||||
let translatedBody;
|
||||
let toolNameMap;
|
||||
let customToolNames;
|
||||
if (passthrough) {
|
||||
log?.debug?.(
|
||||
"PASSTHROUGH",
|
||||
`${clientTool} → ${provider} | native lossless`,
|
||||
);
|
||||
translatedBody = { ...body, model: stripThinkingSuffix(upstreamModel) };
|
||||
if (provider === "codex") {
|
||||
const suffixThinking = {};
|
||||
applyThinking(sourceFormat, upstreamModel, suffixThinking, provider);
|
||||
if (suffixThinking.reasoning_effort) {
|
||||
const reasoning = translatedBody.reasoning;
|
||||
translatedBody.reasoning = {
|
||||
...(reasoning && typeof reasoning === "object" && !Array.isArray(reasoning) ? reasoning : {}),
|
||||
effort: suffixThinking.reasoning_effort,
|
||||
};
|
||||
delete translatedBody.reasoning_effort;
|
||||
}
|
||||
}
|
||||
// Normalize newer Cowork/CC beta shapes (adaptive thinking, mid-conversation system) the API rejects
|
||||
if (clientTool === "claude")
|
||||
normalizeClaudePassthrough(translatedBody, translatedBody.model);
|
||||
@@ -280,7 +312,10 @@ export async function handleChatCore({
|
||||
}
|
||||
toolNameMap = translatedBody._toolNameMap;
|
||||
delete translatedBody._toolNameMap;
|
||||
customToolNames = translatedBody._customToolNames;
|
||||
delete translatedBody._customToolNames;
|
||||
translatedBody.model = stripThinkingSuffix(upstreamModel);
|
||||
stripContinuityFields(translatedBody);
|
||||
}
|
||||
|
||||
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
|
||||
@@ -736,6 +771,8 @@ export async function handleChatCore({
|
||||
...sharedCtx,
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
targetFormat: providerResponseFormat,
|
||||
customToolNames,
|
||||
trackDone,
|
||||
appendLog,
|
||||
});
|
||||
@@ -754,6 +791,7 @@ export async function handleChatCore({
|
||||
targetFormat: providerResponseFormat,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
customToolNames,
|
||||
trackDone,
|
||||
appendLog,
|
||||
});
|
||||
@@ -773,6 +811,7 @@ export async function handleChatCore({
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
customToolNames,
|
||||
streamController,
|
||||
onStreamComplete,
|
||||
streamDetailId,
|
||||
|
||||
@@ -9,6 +9,7 @@ import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
function parseToolArguments(value) {
|
||||
if (!value) return {};
|
||||
@@ -60,11 +61,93 @@ function openAICompletionToClaudeMessage(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions non-streaming response body into the
|
||||
* OpenAI Responses API shape. Used when a Responses-format client (e.g. Codex)
|
||||
* is routed to a Chat Completions upstream and `stream:false` — the streaming
|
||||
* path already emits Responses events, but the JSON path returned a raw
|
||||
* `chat.completion` body, so tool_calls were invisible to Responses clients.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function openAICompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
// Reasoning → a reasoning item (summary text), mirroring the streaming path.
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
// Assistant text → a message item with output_text content.
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
// tool_calls → function_call/custom_tool_call items (Responses-native tool shape).
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
const status = choice.finish_reason === "tool_calls" ? "completed" : (choice.finish_reason === "stop" ? "completed" : (choice.finish_reason || "completed"));
|
||||
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status,
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate non-streaming response body from provider format → OpenAI format.
|
||||
*/
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat) {
|
||||
export function translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames = null) {
|
||||
if (targetFormat === sourceFormat) return responseBody;
|
||||
// Provider responded in OpenAI Chat Completions shape but the client speaks
|
||||
// Responses API — convert so tool_calls/text surface as Responses `output`.
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES) {
|
||||
return openAICompletionToResponses(responseBody, customToolNames);
|
||||
}
|
||||
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.CLAUDE) {
|
||||
return openAICompletionToClaudeMessage(responseBody);
|
||||
}
|
||||
@@ -198,7 +281,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
||||
/**
|
||||
* Handle non-streaming response from provider.
|
||||
*/
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
trackDone();
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
let responseBody;
|
||||
@@ -239,9 +322,12 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat, customToolNames)
|
||||
: responseBody;
|
||||
const isClaudeMessageResponse = sourceFormat === FORMATS.CLAUDE && translatedResponse?.type === "message";
|
||||
// Responses-format translation produces a `object:"response"` body with no
|
||||
// `choices`; skip the Chat-Completions-specific post-processing below for it.
|
||||
const isResponsesResponse = sourceFormat === FORMATS.OPENAI_RESPONSES && translatedResponse?.object === "response";
|
||||
|
||||
// Fix finish_reason for tool_calls: some providers return non-standard values (e.g. "other")
|
||||
if (translatedResponse?.choices?.[0]) {
|
||||
@@ -254,13 +340,13 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
|
||||
// Ensure OpenAI-required fields
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
if (!translatedResponse.object) translatedResponse.object = "chat.completion";
|
||||
if (!translatedResponse.created) translatedResponse.created = Math.floor(Date.now() / 1000);
|
||||
}
|
||||
|
||||
// Strip Azure-specific fields
|
||||
if (!isClaudeMessageResponse) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse) {
|
||||
delete translatedResponse.prompt_filter_results;
|
||||
if (translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) delete choice.content_filter_results;
|
||||
@@ -274,7 +360,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
// reasoning_content is the only useful output and must be preserved.
|
||||
if (!isClaudeMessageResponse && translatedResponse?.choices) {
|
||||
if (!isClaudeMessageResponse && !isResponsesResponse && translatedResponse?.choices) {
|
||||
for (const choice of translatedResponse.choices) {
|
||||
if (choice?.message?.reasoning_content && choice.message.content) {
|
||||
delete choice.message.reasoning_content;
|
||||
|
||||
@@ -4,12 +4,8 @@ import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import {
|
||||
buildRequestDetail,
|
||||
extractRequestConfig,
|
||||
saveUsageStats,
|
||||
formatDoneLine,
|
||||
} from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { ROLE, RESPONSES_ITEM } from "../../translator/schema/index.js";
|
||||
|
||||
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
|
||||
const isResponsesProvider = (p) =>
|
||||
@@ -41,6 +37,76 @@ function pickAssistantMessageForChatCompletion(output) {
|
||||
return { msgItem: last, textContent: textFromResponsesMessageItem(last) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an OpenAI Chat Completions JSON body into the Responses API shape.
|
||||
* Inlined here (not imported from nonStreamingHandler.js) to avoid a circular
|
||||
* import. Mirrors openAICompletionToResponses in nonStreamingHandler.js.
|
||||
*/
|
||||
function extractCustomToolInput(argumentsValue) {
|
||||
const argumentsText = typeof argumentsValue === "string" ? argumentsValue : JSON.stringify(argumentsValue || {});
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function chatCompletionToResponses(responseBody, customToolNames = null) {
|
||||
const choice = responseBody?.choices?.[0];
|
||||
if (!choice) return responseBody;
|
||||
|
||||
const message = choice.message || {};
|
||||
const output = [];
|
||||
|
||||
const reasoning = message.reasoning_content || message.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.REASONING,
|
||||
summary: [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: reasoning }],
|
||||
});
|
||||
}
|
||||
|
||||
const text = typeof message.content === "string" ? message.content : "";
|
||||
if (text.length > 0) {
|
||||
output.push({
|
||||
type: RESPONSES_ITEM.MESSAGE,
|
||||
role: ROLE.ASSISTANT,
|
||||
content: [{ type: RESPONSES_ITEM.OUTPUT_TEXT, text, annotations: [] }],
|
||||
});
|
||||
}
|
||||
|
||||
for (const tc of message.tool_calls || []) {
|
||||
const fn = tc.function || {};
|
||||
const custom = customToolNames?.has(fn.name);
|
||||
output.push({
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
id: `${custom ? "ctc" : "fc"}_${tc.id || ""}`,
|
||||
call_id: tc.id || "",
|
||||
name: fn.name || "",
|
||||
...(custom
|
||||
? { input: extractCustomToolInput(fn.arguments) }
|
||||
: { arguments: typeof fn.arguments === "string" ? fn.arguments : JSON.stringify(fn.arguments || {}) }),
|
||||
});
|
||||
}
|
||||
|
||||
const usage = responseBody.usage || {};
|
||||
return {
|
||||
id: `resp_${responseBody.id || ""}`.replace(/^resp_chatcmpl-/, "resp_"),
|
||||
object: "response",
|
||||
created_at: responseBody.created || Math.floor(Date.now() / 1000),
|
||||
model: responseBody.model || "unknown",
|
||||
status: "completed",
|
||||
background: false,
|
||||
error: null,
|
||||
output,
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || usage.input_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || usage.output_tokens || 0,
|
||||
total_tokens: usage.total_tokens || (usage.prompt_tokens || 0) + (usage.completion_tokens || 0),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse OpenAI-style SSE text into a single chat completion JSON.
|
||||
* Used when provider forces streaming but client wants non-streaming.
|
||||
@@ -136,6 +202,7 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
export async function handleForcedSSEToJson({
|
||||
providerResponse,
|
||||
sourceFormat,
|
||||
targetFormat,
|
||||
provider,
|
||||
model,
|
||||
body,
|
||||
@@ -147,6 +214,7 @@ export async function handleForcedSSEToJson({
|
||||
apiKey,
|
||||
clientRawRequest,
|
||||
onRequestSuccess,
|
||||
customToolNames,
|
||||
trackDone,
|
||||
appendLog,
|
||||
reqTag,
|
||||
@@ -170,8 +238,11 @@ export async function handleForcedSSEToJson({
|
||||
};
|
||||
|
||||
// Codex/Responses API SSE path
|
||||
// Branch on the UPSTREAM format (targetFormat = format we spoke to the provider in),
|
||||
// not the client format: a Responses-API client behind a chat-native forced-streaming
|
||||
// provider still receives chat SSE chunks, which must go through the standard path.
|
||||
const isCodexResponsesApi =
|
||||
isResponsesProvider(provider) || sourceFormat === FORMATS.OPENAI_RESPONSES;
|
||||
isResponsesProvider(provider) || targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
if (isCodexResponsesApi) {
|
||||
try {
|
||||
const jsonResponse = await convertResponsesStreamToJson(
|
||||
@@ -200,6 +271,11 @@ export async function handleForcedSSEToJson({
|
||||
}),
|
||||
);
|
||||
|
||||
// Same cache-inclusive total for the recorded detail, so the DB and the
|
||||
// client-facing usage can never disagree.
|
||||
const inTokensForLog = (usage.input_tokens || 0)
|
||||
+ (usage.cache_read_input_tokens || usage.cached_tokens || 0)
|
||||
+ (usage.cache_creation_input_tokens || 0);
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(
|
||||
jsonResponse.output,
|
||||
);
|
||||
@@ -209,9 +285,10 @@ export async function handleForcedSSEToJson({
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
apiKey,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: {
|
||||
prompt_tokens: usage.input_tokens || 0,
|
||||
prompt_tokens: inTokensForLog,
|
||||
completion_tokens: usage.output_tokens || 0,
|
||||
},
|
||||
response: {
|
||||
@@ -238,9 +315,22 @@ export async function handleForcedSSEToJson({
|
||||
};
|
||||
}
|
||||
|
||||
// Build client-format response
|
||||
const inTokens = usage.input_tokens || 0;
|
||||
// Build client-format response.
|
||||
// input_tokens EXCLUDES cached tokens on cache-capable upstreams, so summing
|
||||
// only input+output under-reports prompt_tokens — measured: 2012 reported
|
||||
// where the real prompt was ~5344 with 5332 served from cache. Fold the cache
|
||||
// counters in, and keep them visible in prompt_tokens_details so a client can
|
||||
// tell a cache hit from a small prompt.
|
||||
const cacheRead = usage.cache_read_input_tokens || usage.cached_tokens || 0;
|
||||
const cacheCreate = usage.cache_creation_input_tokens || 0;
|
||||
const inTokens = (usage.input_tokens || 0) + cacheRead + cacheCreate;
|
||||
const outTokens = usage.output_tokens || 0;
|
||||
const cacheDetails = (cacheRead > 0 || cacheCreate > 0)
|
||||
? {
|
||||
prompt_tokens_details: {
|
||||
...(cacheRead > 0 ? { cached_tokens: cacheRead } : {}),
|
||||
...(cacheCreate > 0 ? { cache_creation_tokens: cacheCreate } : {}) } }
|
||||
: {};
|
||||
let finalResp;
|
||||
|
||||
// Extract tool calls from Responses API output (function_call items)
|
||||
@@ -309,6 +399,7 @@ export async function handleForcedSSEToJson({
|
||||
prompt_tokens: inTokens,
|
||||
completion_tokens: outTokens,
|
||||
total_tokens: inTokens + outTokens,
|
||||
...cacheDetails,
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -384,23 +475,14 @@ export async function handleForcedSSEToJson({
|
||||
}),
|
||||
);
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(
|
||||
buildRequestDetail(
|
||||
{
|
||||
...ctx,
|
||||
latency: { ttft: totalLatency, total: totalLatency },
|
||||
tokens: usage,
|
||||
response: {
|
||||
content: parsed.choices?.[0]?.message?.content || null,
|
||||
thinking: parsed.choices?.[0]?.message?.reasoning_content || null,
|
||||
finish_reason: parsed.choices?.[0]?.finish_reason || "unknown",
|
||||
},
|
||||
status: "success",
|
||||
},
|
||||
{ endpoint: clientRawRequest?.endpoint || null },
|
||||
),
|
||||
).catch(() => {});
|
||||
// Re-attach usage explicitly. This handler already HAS the correct usage — it is
|
||||
// the same object written to the usage DB, and for a cached Claude request that DB
|
||||
// row reads cache_read_input_tokens: 11022 — yet the client was observed receiving
|
||||
// no usage field at all (verified 2026-08-04 with a fingerprinted payload matched
|
||||
// on both sides). Whatever drops it between assembly and serialisation, the client
|
||||
// must not be left unable to account for its own token spend: a caller cannot tell
|
||||
// a 90%-cached request from a cheap one without this.
|
||||
if (usage && Object.keys(usage).length > 0) parsed.usage = usage;
|
||||
|
||||
// Strip reasoning_content only when content is non-empty.
|
||||
// When content is empty (e.g. thinking models that used all tokens for reasoning),
|
||||
@@ -414,9 +496,19 @@ export async function handleForcedSSEToJson({
|
||||
}
|
||||
}
|
||||
|
||||
// A Responses-format client (e.g. Codex) forced this provider to stream,
|
||||
// but wants JSON back. parseSSEToOpenAIResponse yields a Chat Completions
|
||||
// body; convert it to the Responses `output` shape so tool_calls are not
|
||||
// lost on the non-streaming return path. Inlined (not imported from
|
||||
// nonStreamingHandler.js) to avoid a circular import: nonStreamingHandler
|
||||
// already imports parseSSEToOpenAIResponse from this module.
|
||||
const finalBody = sourceFormat === FORMATS.OPENAI_RESPONSES
|
||||
? chatCompletionToResponses(parsed, customToolNames)
|
||||
: parsed;
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(JSON.stringify(parsed), {
|
||||
response: new Response(JSON.stringify(finalBody), {
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
|
||||
@@ -38,6 +38,7 @@ function buildTransformStream({
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
customToolNames,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
@@ -68,6 +69,7 @@ function buildTransformStream({
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
customToolNames,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -83,6 +85,7 @@ function buildTransformStream({
|
||||
body,
|
||||
onStreamComplete,
|
||||
apiKey,
|
||||
customToolNames,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -118,6 +121,7 @@ export async function handleStreamingResponse({
|
||||
onRequestSuccess,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
customToolNames,
|
||||
streamController,
|
||||
onStreamComplete,
|
||||
streamDetailId,
|
||||
@@ -200,6 +204,7 @@ export async function handleStreamingResponse({
|
||||
userAgent,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
customToolNames,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
import createOpenAIEmbeddingAdapter from "./openai.js";
|
||||
import gemini from "./gemini.js";
|
||||
import openaiCompatNode from "./openaiCompatNode.js";
|
||||
import selfhostedEmbedding from "./selfhostedEmbedding.js";
|
||||
|
||||
const OPENAI_COMPAT_PROVIDERS = [
|
||||
"openai", "openrouter", "mistral", "voyage-ai", "fireworks",
|
||||
@@ -13,6 +14,12 @@ const ADAPTERS = {
|
||||
...Object.fromEntries(OPENAI_COMPAT_PROVIDERS.map((id) => [id, createOpenAIEmbeddingAdapter(id)])),
|
||||
gemini,
|
||||
google_ai_studio: gemini,
|
||||
// Self-hosted reads creds.providerSpecificData.baseUrl (one provider, many
|
||||
// servers) — but via its OWN adapter, not openaiCompatNode: that one falls back
|
||||
// to api.openai.com when no baseUrl is set, which under a provider called
|
||||
// "Self-hosted Embedding" means silently shipping the input and API key to
|
||||
// OpenAI. selfhostedEmbedding refuses instead.
|
||||
"selfhosted-embedding": selfhostedEmbedding,
|
||||
};
|
||||
|
||||
export function getEmbeddingAdapter(provider) {
|
||||
|
||||
46
open-sse/handlers/embeddingProviders/selfhostedEmbedding.js
Normal file
46
open-sse/handlers/embeddingProviders/selfhostedEmbedding.js
Normal file
@@ -0,0 +1,46 @@
|
||||
// Self-hosted embeddings — like openaiCompatNode, but the baseUrl is REQUIRED.
|
||||
//
|
||||
// openaiCompatNode falls back to https://api.openai.com/v1 when a connection
|
||||
// carries no providerSpecificData.baseUrl. For a custom NODE that default is
|
||||
// defensible: the node was created by pointing at some OpenAI-compatible URL, and
|
||||
// OpenAI is the archetype. For a provider whose entire purpose is "my own
|
||||
// server", it is actively harmful — a connection saved without a baseUrl sends
|
||||
// the INPUT TEXT and the API KEY to OpenAI, silently, under a provider named
|
||||
// "Self-hosted Embedding".
|
||||
//
|
||||
// Observed exactly that with a placeholder connection (2026-08-04):
|
||||
//
|
||||
// [selfhosted-embedding/embedding] [401]: Incorrect API key provided: abc.
|
||||
// You can find your API key at https://platform.openai.com/account/api-keys.
|
||||
//
|
||||
// The key "abc" was typed as a throwaway for a LOCAL server and left the network.
|
||||
// A self-hosted provider must never have a cloud fallback, so this one refuses
|
||||
// instead: no baseUrl means a configuration error, reported as such.
|
||||
import createOpenAIEmbeddingAdapter from "./openai.js";
|
||||
|
||||
const baseAdapter = createOpenAIEmbeddingAdapter("openai");
|
||||
|
||||
export class MissingBaseUrlError extends Error {
|
||||
constructor() {
|
||||
super(
|
||||
"Self-hosted Embedding needs an endpoint: set this connection's baseUrl to " +
|
||||
"the OpenAI base URL of your server, e.g. http://host:8080/v1 (note the /v1 — " +
|
||||
"\"/embeddings\" is appended to it). Refusing to fall back to api.openai.com, " +
|
||||
"which would send your input and API key to OpenAI."
|
||||
);
|
||||
this.name = "MissingBaseUrlError";
|
||||
this.isConfigError = true;
|
||||
}
|
||||
}
|
||||
|
||||
export default {
|
||||
...baseAdapter,
|
||||
buildUrl: (_model, creds) => {
|
||||
const rawBaseUrl = creds?.providerSpecificData?.baseUrl;
|
||||
if (!rawBaseUrl || !String(rawBaseUrl).trim()) throw new MissingBaseUrlError();
|
||||
// Accept either the OpenAI base or a full embeddings URL, so a value pasted
|
||||
// from a curl example works as well as one typed from the help text.
|
||||
const baseUrl = String(rawBaseUrl).trim().replace(/\/$/, "").replace(/\/embeddings$/, "");
|
||||
return `${baseUrl}/embeddings`;
|
||||
},
|
||||
};
|
||||
@@ -1,5 +1,5 @@
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS, FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import { getExecutor } from "../executors/index.js";
|
||||
import { refreshWithRetry } from "../services/tokenRefresh.js";
|
||||
import { getEmbeddingAdapter } from "./embeddingProviders/index.js";
|
||||
@@ -38,13 +38,24 @@ export async function handleEmbeddingsCore({
|
||||
}
|
||||
|
||||
const ctx = { input };
|
||||
const url = adapter.buildUrl(model, credentials, ctx);
|
||||
const headers = adapter.buildHeaders(credentials, ctx);
|
||||
const requestBody = adapter.buildBody(model, {
|
||||
input,
|
||||
encoding_format: body.encoding_format || "float",
|
||||
dimensions: body.dimensions,
|
||||
});
|
||||
// buildUrl/buildHeaders/buildBody were called bare. An adapter that rejects a
|
||||
// misconfigured connection — selfhosted-embedding throws when no baseUrl is set
|
||||
// rather than silently falling back to api.openai.com — would have escaped this
|
||||
// function uncaught, surfacing as a 500 or a request that never settles. A
|
||||
// configuration mistake is a 400 with the reason in it.
|
||||
let url, headers, requestBody;
|
||||
try {
|
||||
url = adapter.buildUrl(model, credentials, ctx);
|
||||
headers = adapter.buildHeaders(credentials, ctx);
|
||||
requestBody = adapter.buildBody(model, {
|
||||
input,
|
||||
encoding_format: body.encoding_format || "float",
|
||||
dimensions: body.dimensions,
|
||||
});
|
||||
} catch (error) {
|
||||
log?.debug?.("EMBEDDINGS", `Request build failed: ${error.message}`);
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}/${model}] ${error.message}`);
|
||||
}
|
||||
|
||||
log?.debug?.("EMBEDDINGS", `${provider.toUpperCase()} | ${model} | input_type=${Array.isArray(input) ? `array[${input.length}]` : "string"}`);
|
||||
|
||||
@@ -54,6 +65,9 @@ export async function handleEmbeddingsCore({
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify(requestBody),
|
||||
...(typeof AbortSignal?.timeout === "function"
|
||||
? { signal: AbortSignal.timeout(FETCH_CONNECT_TIMEOUT_MS) }
|
||||
: {}),
|
||||
});
|
||||
} catch (error) {
|
||||
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
|
||||
|
||||
@@ -170,9 +170,17 @@ export async function handleSttCore({ provider, model, formData, credentials, st
|
||||
const file = formData.get("file");
|
||||
if (!file) return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: file");
|
||||
|
||||
const cfg = sttConfig;
|
||||
let cfg = sttConfig;
|
||||
if (!cfg) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support STT`);
|
||||
|
||||
// Per-connection endpoint override. Registry entries carry a fixed baseUrl,
|
||||
// which is right for a named cloud service but useless for a self-hosted one
|
||||
// whose address only the operator knows. Opt-in: absent unless the connection
|
||||
// sets it, so cloud providers are untouched. Mirrors the custom embedding
|
||||
// providers, which already resolve baseUrl the same way.
|
||||
const overrideUrl = credentials?.providerSpecificData?.baseUrl;
|
||||
if (overrideUrl) cfg = { ...cfg, baseUrl: String(overrideUrl).replace(/\/+$/, "") };
|
||||
|
||||
const token = cfg.authType === "none" ? null : (credentials?.apiKey || credentials?.accessToken);
|
||||
if (cfg.authType !== "none" && !token) {
|
||||
return createErrorResult(HTTP_STATUS.UNAUTHORIZED, `No credentials for STT provider: ${provider}`);
|
||||
|
||||
@@ -48,16 +48,16 @@ function createTtsResponse(base64Audio, format, responseFormat) {
|
||||
*
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language }) {
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language, style }) {
|
||||
if (!input?.trim()) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: input");
|
||||
}
|
||||
|
||||
try {
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini)
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini, xiaomi-mimo)
|
||||
const adapter = getTtsAdapter(provider);
|
||||
if (adapter) {
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language });
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language, style });
|
||||
// Adapter may return a full {success, response} (legacy) or {base64, format}
|
||||
if (result.success !== undefined) return result;
|
||||
return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
|
||||
@@ -6,6 +6,8 @@ import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
||||
import openai from "./openai.js";
|
||||
import openrouter from "./openrouter.js";
|
||||
import gemini, { fetchGeminiVoices } from "./gemini.js";
|
||||
import xiaomiMimo from "./xiaomi-mimo.js";
|
||||
import selfhostedTts from "./selfhostedTts.js";
|
||||
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
@@ -18,6 +20,8 @@ const SPECIAL_ADAPTERS = {
|
||||
openai,
|
||||
openrouter,
|
||||
gemini,
|
||||
"xiaomi-mimo": xiaomiMimo,
|
||||
"selfhosted-tts": selfhostedTts,
|
||||
};
|
||||
|
||||
export function getTtsAdapter(provider) {
|
||||
|
||||
69
open-sse/handlers/ttsProviders/selfhostedTts.js
Normal file
69
open-sse/handlers/ttsProviders/selfhostedTts.js
Normal file
@@ -0,0 +1,69 @@
|
||||
// Self-hosted OpenAI-compatible TTS — POST {baseUrl}/v1/audio/speech.
|
||||
//
|
||||
// A SPECIAL_ADAPTER rather than a genericFormats handler on purpose: the generic
|
||||
// dispatcher resolves baseUrl from the static registry entry
|
||||
// (`synthesizeViaConfig` reads `cfg.baseUrl`) and never looks at the connection,
|
||||
// which is exactly the limitation this provider exists to lift.
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
const DEFAULT_BASE_URL = "http://localhost:8880";
|
||||
const DEFAULT_MODEL = "kokoro";
|
||||
const DEFAULT_VOICE = "af_heart";
|
||||
|
||||
export default {
|
||||
async synthesize(text, model, credentials, responseFormat = "mp3") {
|
||||
// Accept either providerSpecificData.baseUrl (how the custom embedding and
|
||||
// STT providers carry it) or a bare credentials.baseUrl (how the OpenAI TTS
|
||||
// adapter does), so a connection configured either way works.
|
||||
const raw = credentials?.providerSpecificData?.baseUrl || credentials?.baseUrl || DEFAULT_BASE_URL;
|
||||
// Tolerate a baseUrl given as the full endpoint or with a trailing /v1 —
|
||||
// both are natural things to paste, and silently double-appending the path
|
||||
// would 404 with nothing pointing at the cause.
|
||||
const base = String(raw)
|
||||
.replace(/\/+$/, "")
|
||||
.replace(/\/v1\/audio\/speech$/, "")
|
||||
.replace(/\/v1$/, "");
|
||||
|
||||
// The provider prefix is already stripped by getModelInfo, so `model` here is
|
||||
// "kokoro" or "kokoro/af_heart" — NOT "selfhosted-tts/...".
|
||||
//
|
||||
// A bare value is the MODEL, not the voice. The OpenAI adapter reads a bare
|
||||
// value as a voice, which is right for a service whose model is fixed
|
||||
// ("tts-1") and whose voice varies — but wrong here, where the model is the
|
||||
// variable part. Treating it as a voice sent voice="kokoro" upstream and
|
||||
// Kokoro answered 400, so `selfhosted-tts/kokoro` — the obvious way to
|
||||
// address this provider — was the one form that did not work (verified
|
||||
// against a live Kokoro through 9router, 2026-08-03).
|
||||
let ttsModel = DEFAULT_MODEL;
|
||||
let voice = DEFAULT_VOICE;
|
||||
if (model) {
|
||||
const parts = String(model).split("/").filter(Boolean);
|
||||
if (parts.length >= 2) {
|
||||
ttsModel = parts[0];
|
||||
voice = parts.slice(1).join("/");
|
||||
} else if (parts.length === 1) {
|
||||
ttsModel = parts[0];
|
||||
}
|
||||
}
|
||||
|
||||
const res = await fetch(`${base}/v1/audio/speech`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(credentials?.apiKey ? { Authorization: `Bearer ${credentials.apiKey}` } : {}),
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: ttsModel,
|
||||
voice,
|
||||
input: text,
|
||||
response_format: responseFormat,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.error?.message || `Self-hosted TTS failed: ${res.status}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: responseFormat };
|
||||
},
|
||||
};
|
||||
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
@@ -0,0 +1,65 @@
|
||||
// Xiaomi MiMo TTS — via OpenAI-compatible chat completions (non-streaming).
|
||||
// Docs: https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5
|
||||
// Message contract: target text in `role: assistant` content, style/voice
|
||||
// instructions in `role: user` content. Voice is selected via the top-level
|
||||
// `audio.voice` field (NOT embedded in the model name).
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
const DEFAULT_MODEL = "mimo-v2.5-tts";
|
||||
const DEFAULT_VOICE = "mimo_default";
|
||||
|
||||
export default {
|
||||
synthesize(text, model, credentials, responseFormat, { style, language } = {}) {
|
||||
if (!credentials?.apiKey) throw new Error("xiaomi-mimo API key required");
|
||||
return synthesizeMiMo(text, model, credentials.apiKey, style, language);
|
||||
},
|
||||
};
|
||||
|
||||
export async function synthesizeMiMo(text, model, apiKey, style, language) {
|
||||
const { modelId, voiceId } = parseModelVoice(model, DEFAULT_MODEL, DEFAULT_VOICE, [DEFAULT_MODEL]);
|
||||
|
||||
// Language and style are soft instructions → prepend as a role:user message.
|
||||
// MiMo auto-detects the spoken language of the text; the hint only nudges it
|
||||
// (e.g. "Speak in English.") and is independent of the chosen voice.
|
||||
const instructions = [];
|
||||
if (language) instructions.push(`Speak in ${language}.`);
|
||||
if (style) instructions.push(style);
|
||||
|
||||
const messages = [{ role: "assistant", content: text }];
|
||||
if (instructions.length) messages.unshift({ role: "user", content: instructions.join(" ") });
|
||||
|
||||
const res = await fetch("https://api.xiaomimimo.com/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
stream: false,
|
||||
messages,
|
||||
audio: {
|
||||
format: "wav",
|
||||
voice: voiceId || DEFAULT_VOICE,
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
const rawText = await res.text();
|
||||
let data = {};
|
||||
if (rawText) {
|
||||
try { data = JSON.parse(rawText); } catch { data = {}; }
|
||||
}
|
||||
|
||||
if (!res.ok) {
|
||||
throw new Error(data?.error?.message || rawText || `MiMo TTS error (${res.status})`);
|
||||
}
|
||||
|
||||
const audio = data?.choices?.[0]?.message?.audio?.data;
|
||||
if (!audio) throw new Error(data?.error?.message || "MiMo TTS returned no audio");
|
||||
|
||||
return {
|
||||
base64: audio,
|
||||
format: data?.choices?.[0]?.message?.audio?.format || "wav",
|
||||
};
|
||||
}
|
||||
@@ -47,7 +47,6 @@ export {
|
||||
refreshAccessToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshIflowToken,
|
||||
refreshGitHubToken,
|
||||
|
||||
@@ -279,7 +279,7 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*minimax*", caps: { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 131072 } },
|
||||
|
||||
// ── Xiaomi MiMo (vision, 1M / 262K ctx) ──────────────────────────
|
||||
{ pattern: "*mimo*v2.5*", caps: { vision: true, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*v2.5*", caps: { vision: true, audioInput: true, videoInput: true, contextWindow: 1048576, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*omni*", caps: { vision: true, audioInput: true, contextWindow: 262144, maxOutput: 131072 } },
|
||||
{ pattern: "*mimo*", caps: { vision: true, contextWindow: 262144, maxOutput: 131072 } },
|
||||
|
||||
|
||||
@@ -141,6 +141,122 @@ export const PROVIDER_PRICING = {
|
||||
gh: {
|
||||
"gpt-5.3-codex": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
},
|
||||
// TokenRouter — exact rates from https://api.tokenrouter.com/api/pricing ($1/1M tokens).
|
||||
// Ratio→USD: input = model_ratio×2, output = model_ratio×completion_ratio×2.
|
||||
// These override the canonical MODEL_PRICING/PATTERN_PRICING, whose rates often
|
||||
// differ from TokenRouter's reseller pricing.
|
||||
tokenrouter: {
|
||||
"MiniMax-M3": { input: 0.3, output: 1.2, cached: 0.06, reasoning: 1.2 },
|
||||
"anthropic/claude-fable-5": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-haiku-4.5": { input: 1.0, output: 5.0, cached: 0.1, cache_creation: 1.25, reasoning: 5.0 },
|
||||
"anthropic/claude-opus-4.5": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.6": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.7": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.7-fast": { input: 30, output: 150, cached: 3.0, reasoning: 150 },
|
||||
"anthropic/claude-opus-4.8": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-4.8-fast": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-opus-5": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"anthropic/claude-opus-5-fast": { input: 10, output: 50, cached: 1.0, cache_creation: 12.5, reasoning: 50 },
|
||||
"anthropic/claude-sonnet-4": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-4.5": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-4.6": { input: 3.0, output: 15.0, cached: 0.3, cache_creation: 3.75, reasoning: 15.0 },
|
||||
"anthropic/claude-sonnet-5": { input: 2, output: 10, cached: 0.2, reasoning: 10 },
|
||||
"claude-opus-4-8-m-aws": { input: 5.0, output: 25.0, cached: 0.5, cache_creation: 6.25, reasoning: 25.0 },
|
||||
"deepseek/deepseek-v3.2": { input: 0.26, output: 0.38, cached: 0.13, reasoning: 0.38 },
|
||||
"deepseek/deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28 },
|
||||
"deepseek/deepseek-v4-flash-0731": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28 },
|
||||
"deepseek/deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87 },
|
||||
"ex/gpt-5.4": { input: 2.5, output: 15.0, cached: 0.25, reasoning: 15.0 },
|
||||
"google/gemini-2.5-flash-image": { input: 0.3, output: 2.5, reasoning: 2.5 },
|
||||
"google/gemini-3-flash-preview": { input: 0.5, output: 3.0, cached: 0.05, cache_creation: 0.08333, reasoning: 3.0 },
|
||||
"google/gemini-3-pro-image-preview": { input: 2, output: 12, reasoning: 12 },
|
||||
"google/gemini-3.1-flash-image-preview": { input: 0.5, output: 3.0, reasoning: 3.0 },
|
||||
"google/gemini-3.1-flash-lite-image": { input: 0.25, output: 1.5, reasoning: 1.5 },
|
||||
"google/gemini-3.1-pro-preview": { input: 2, output: 12, cached: 0.2, cache_creation: 0.375, reasoning: 12 },
|
||||
"google/gemini-3.5-flash": { input: 1.5, output: 9.0, cached: 0.15, cache_creation: 0.08333, reasoning: 9.0 },
|
||||
"google/gemini-3.5-flash-lite": { input: 0.3, output: 2.5, cached: 0.03, cache_creation: 0.08333, reasoning: 2.5 },
|
||||
"google/gemini-3.6-flash": { input: 1.5, output: 7.5, cached: 0.15, cache_creation: 0.08333, reasoning: 7.5 },
|
||||
"google/gemini-embedding-2": { input: 1.0, output: 6.0, cached: 0.1, reasoning: 6.0 },
|
||||
"google/gemma-4-26b-a4b-it": { input: 0.06, output: 0.33, reasoning: 0.33 },
|
||||
"kling-3.0-turbo": { input: 2.1, output: 2.1, reasoning: 2.1 },
|
||||
"microsoft/mai-image-2.5": { input: 5.0, output: 47.0, reasoning: 47.0 },
|
||||
"minimax/minimax-m2-her": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.1": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.1-highspeed": { input: 0.6, output: 2.4, cached: 0.06, reasoning: 2.4 },
|
||||
"minimax/minimax-m2.5": { input: 0.3, output: 1.2, cached: 0.03, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.7": { input: 0.3, output: 1.2, cached: 0.06, reasoning: 1.2 },
|
||||
"minimax/minimax-m2.7-highspeed": { input: 0.6, output: 2.4, cached: 0.06, reasoning: 2.4 },
|
||||
"miromind/mirothinker-1-7-deepresearch": { input: 4, output: 25.0, reasoning: 25.0 },
|
||||
"miromind/mirothinker-1-7-deepresearch-mini": { input: 1.25, output: 10.0, reasoning: 10.0 },
|
||||
"mistralai/devstral-2512": { input: 0.4, output: 2.0, cached: 0.04, reasoning: 2.0 },
|
||||
"mistralai/mistral-medium-3-5": { input: 1.5, output: 7.5, reasoning: 7.5 },
|
||||
"mistralai/mistral-small-2603": { input: 0.15, output: 0.6, cached: 0.015, reasoning: 0.6 },
|
||||
"mistralai/voxtral-small-24b-2507": { input: 0.1, output: 0.3, cached: 0.01, reasoning: 0.3 },
|
||||
"moonshotai/kimi-k2.5": { input: 0.6, output: 3.0, cached: 0.1, reasoning: 3.0 },
|
||||
"moonshotai/kimi-k2.6": { input: 0.95, output: 4.0, cached: 0.16, reasoning: 4.0 },
|
||||
"moonshotai/kimi-k2.7-code": { input: 0.9286, output: 3.8571, cached: 0.1857, reasoning: 3.8571 },
|
||||
"moonshotai/kimi-k3": { input: 3.0, output: 15.0, cached: 0.3, reasoning: 15.0 },
|
||||
"nvidia/nemotron-3-super-120b-a12b": { input: 0.3, output: 0.9, cached: 0.1, reasoning: 0.9 },
|
||||
"openai/gpt-4o-mini": { input: 0.15, output: 0.6, cached: 0.075, reasoning: 0.6 },
|
||||
"openai/gpt-5": { input: 1.25, output: 10.0, cached: 0.125, reasoning: 10.0 },
|
||||
"openai/gpt-5-image": { input: 10, output: 40, cached: 2.5, reasoning: 40 },
|
||||
"openai/gpt-5-image-mini": { input: 2.5, output: 8.0, cached: 0.25, reasoning: 8.0 },
|
||||
"openai/gpt-5-mini": { input: 0.25, output: 2.0, cached: 0.025, reasoning: 2.0 },
|
||||
"openai/gpt-5.2": { input: 1.75, output: 14.0, cached: 0.175, reasoning: 14.0 },
|
||||
"openai/gpt-5.3-codex": { input: 1.75, output: 14.0, cached: 0.175, reasoning: 14.0 },
|
||||
"openai/gpt-5.4": { input: 2.5, output: 15.0, cached: 0.25, reasoning: 15.0 },
|
||||
"openai/gpt-5.4-image-2": { input: 8, output: 30.0, cached: 2.0, reasoning: 30.0 },
|
||||
"openai/gpt-5.4-mini": { input: 0.75, output: 4.5, cached: 0.075, reasoning: 4.5 },
|
||||
"openai/gpt-5.4-nano": { input: 0.2, output: 1.25, cached: 0.02, reasoning: 1.25 },
|
||||
"openai/gpt-5.4-pro": { input: 30, output: 180, reasoning: 180 },
|
||||
"openai/gpt-5.5": { input: 5.0, output: 30.0, cached: 0.5, reasoning: 30.0 },
|
||||
"openai/gpt-5.5-pro": { input: 30, output: 180, reasoning: 180 },
|
||||
"openai/gpt-5.6-luna": { input: 0.2, output: 1.2, cached: 0.02, cache_creation: 0.25, reasoning: 1.2 },
|
||||
"openai/gpt-5.6-sol": { input: 5.0, output: 30.0, cached: 0.5, cache_creation: 6.25, reasoning: 30.0 },
|
||||
"openai/gpt-5.6-terra": { input: 2, output: 12, cached: 0.2, cache_creation: 2.5, reasoning: 12 },
|
||||
"openai/gpt-audio": { input: 2.5, output: 10.0, reasoning: 10.0 },
|
||||
"openai/gpt-audio-mini": { input: 0.6, output: 2.4, reasoning: 2.4 },
|
||||
"openai/gpt-oss-120b": { input: 0.039, output: 0.18, reasoning: 0.18 },
|
||||
"qwen/qwen3-coder-next": { input: 0.12, output: 0.75, cached: 0.06, reasoning: 0.75 },
|
||||
"qwen/qwen3.5-122b-a10b": { input: 0.26, output: 2.08, reasoning: 2.08 },
|
||||
"qwen/qwen3.5-35b-a3b": { input: 0.1625, output: 1.3, reasoning: 1.3 },
|
||||
"qwen/qwen3.5-397b-a17b": { input: 0.39, output: 2.34, reasoning: 2.34 },
|
||||
"qwen/qwen3.5-9b": { input: 0.1, output: 0.15, reasoning: 0.15 },
|
||||
"qwen/qwen3.5-flash": { input: 0.1048, output: 0.4194, reasoning: 0.4194 },
|
||||
"qwen/qwen3.5-plus-02-15": { input: 0.26, output: 1.56, reasoning: 1.56 },
|
||||
"qwen/qwen3.6-plus": { input: 0.54, output: 3.21, reasoning: 3.21 },
|
||||
"qwen/qwen3.7-max": { input: 1.25, output: 3.75, cached: 0.25, reasoning: 3.75 },
|
||||
"qwen/qwen3.7-plus": { input: 0.4, output: 1.6, cached: 0.08, reasoning: 1.6 },
|
||||
"qwen/qwen3.8-max": { input: 2, output: 6, cached: 0.25, cache_creation: 2.5, reasoning: 6 },
|
||||
"qwen3.5-omni-plus": { input: 1.0, output: 5.7143, reasoning: 5.7143 },
|
||||
"qwen3.6-flash": { input: 0.171, output: 1.029, cached: 0.017, cache_creation: 0.214, reasoning: 1.029 },
|
||||
"sakana/fugu-ultra": { input: 5.0, output: 30.0, cached: 0.5, reasoning: 30.0 },
|
||||
"seed-2-0-code-preview-260328": { input: 1.0, output: 6.0, cached: 0.2, cache_creation: 0.008333, reasoning: 6.0 },
|
||||
"seed-2-0-lite-260428": { input: 0.5, output: 4.0, cached: 0.1, cache_creation: 0.008333, reasoning: 4.0 },
|
||||
"seed-2-0-mini-260428": { input: 0.2, output: 0.8, cached: 0.04, cache_creation: 0.00833, reasoning: 0.8 },
|
||||
"seed-2-0-pro-260328": { input: 1.0, output: 6.0, cached: 0.2, cache_creation: 0.008333, reasoning: 6.0 },
|
||||
"stepfun/step-3.5-flash": { input: 0.1, output: 0.3, cached: 0.02, reasoning: 0.3 },
|
||||
"stepfun/step-3.7-flash": { input: 0.2, output: 1.15, cached: 0.04, reasoning: 1.15 },
|
||||
"tencent/hy3-preview": { input: 0.066, output: 0.26, cached: 0.029, reasoning: 0.26 },
|
||||
"x-ai/grok-4.1-fast": { input: 0.2, output: 0.5, cached: 0.05, reasoning: 0.5 },
|
||||
"x-ai/grok-4.20-beta": { input: 2, output: 6, cached: 0.2, reasoning: 6 },
|
||||
"x-ai/grok-4.3": { input: 1.25, output: 2.5, cached: 0.2, reasoning: 2.5 },
|
||||
"x-ai/grok-4.5": { input: 2, output: 6, cached: 0.5, reasoning: 6 },
|
||||
"x-ai/grok-build-0.1": { input: 1.0, output: 2.0, cached: 0.2, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2-flash": { input: 0.1, output: 0.3, cached: 0.01, reasoning: 0.3 },
|
||||
"xiaomi/mimo-v2-omni": { input: 0.4, output: 2.0, cached: 0.08, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2-pro": { input: 1.0, output: 3.0, cached: 0.2, reasoning: 3.0 },
|
||||
"xiaomi/mimo-v2.5": { input: 0.4, output: 2.0, cached: 0.08, reasoning: 2.0 },
|
||||
"xiaomi/mimo-v2.5-pro": { input: 1.0, output: 3.0, cached: 0.2, reasoning: 3.0 },
|
||||
"z-ai/glm-4.5-air": { input: 0.13, output: 0.85, cached: 0.025, reasoning: 0.85 },
|
||||
"z-ai/glm-4.6": { input: 0.6, output: 2.2, cached: 0.11, reasoning: 2.2 },
|
||||
"z-ai/glm-4.6v": { input: 0.3, output: 0.9, reasoning: 0.9 },
|
||||
"z-ai/glm-4.7": { input: 0.6, output: 2.2, cached: 0.11, reasoning: 2.2 },
|
||||
"z-ai/glm-5": { input: 1.0, output: 3.2, cached: 0.2, reasoning: 3.2 },
|
||||
"z-ai/glm-5-turbo": { input: 1.2, output: 4.0, cached: 0.24, reasoning: 4.0 },
|
||||
"z-ai/glm-5.1": { input: 1.05, output: 3.5, cached: 0.525, reasoning: 3.5 },
|
||||
"z-ai/glm-5.2": { input: 1.4, output: 4.4, cached: 0.26, reasoning: 4.4 },
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -76,8 +76,7 @@ export default {
|
||||
apiVersion: "v1internal",
|
||||
loadCodeAssistEndpoint: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
|
||||
onboardUserEndpoint: "https://cloudcode-pa.googleapis.com/v1internal:onboardUser",
|
||||
loadCodeAssistUserAgent: "google-api-nodejs-client/9.15.1",
|
||||
loadCodeAssistApiClient: "google-cloud-sdk vscode_cloudshelleditor/0.1",
|
||||
loadCodeAssistUserAgent: ANTIGRAVITY_IDE_USER_AGENT,
|
||||
refreshLeadMs: 300000,
|
||||
},
|
||||
features: {
|
||||
|
||||
@@ -49,9 +49,6 @@ export default {
|
||||
header: "Authorization",
|
||||
scheme: "bearer",
|
||||
},
|
||||
hooks: [
|
||||
"claudeOverlay",
|
||||
],
|
||||
},
|
||||
usage: {
|
||||
oauthUrl: "https://api.anthropic.com/api/oauth/usage",
|
||||
|
||||
@@ -19,6 +19,8 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authType: "apikey",
|
||||
authModes: ["apikey"],
|
||||
hasProviderSpecificData: true,
|
||||
transport: {
|
||||
baseUrl: "https://api.cloudflare.com/client/v4/accounts/{accountId}/ai/v1/chat/completions",
|
||||
|
||||
@@ -38,6 +38,10 @@ export default {
|
||||
header: "Authorization",
|
||||
scheme: "bearer",
|
||||
},
|
||||
// Intl billing endpoint mirrors CN shape (data.Response.Data.Accounts[]).
|
||||
usage: {
|
||||
url: "https://www.codebuddy.ai/v2/billing/meter/get-user-resource",
|
||||
},
|
||||
},
|
||||
// Same model lineup exposed by the CN gateway — intl backend is the same catalog.
|
||||
models: [
|
||||
|
||||
@@ -75,7 +75,6 @@ import p72 from "./perplexity.js";
|
||||
import p73 from "./perplexity-agent.js";
|
||||
import p74 from "./playht.js";
|
||||
import p75 from "./qoder.js";
|
||||
import p76 from "./qwen.js";
|
||||
import p77 from "./recraft.js";
|
||||
import p78 from "./runwayml.js";
|
||||
import p79 from "./sdwebui.js";
|
||||
@@ -116,6 +115,10 @@ import p113 from "./morph.js";
|
||||
// import p114 from "./devin-cli.js";
|
||||
// import p104 from "./windsurf.js";
|
||||
import p115 from "./poolside.js";
|
||||
import p116 from "./tokenrouter.js";
|
||||
import p117 from "./selfhosted-stt.js";
|
||||
import p118 from "./selfhosted-tts.js";
|
||||
import p119 from "./selfhosted-embedding.js";
|
||||
|
||||
export default [
|
||||
p0,
|
||||
@@ -194,7 +197,6 @@ export default [
|
||||
p73,
|
||||
p74,
|
||||
p75,
|
||||
p76,
|
||||
p77,
|
||||
p78,
|
||||
p79,
|
||||
@@ -233,4 +235,8 @@ export default [
|
||||
// p114, // devin-cli — hidden, spawns local agent with shell/fs access
|
||||
// p104, // windsurf — hidden, no tool calling
|
||||
p115,
|
||||
p116,
|
||||
p117,
|
||||
p118,
|
||||
p119,
|
||||
];
|
||||
|
||||
@@ -15,6 +15,8 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authType: "apikey",
|
||||
authModes: ["apikey"],
|
||||
transport: {
|
||||
baseUrl: "https://ollama.com/api/chat",
|
||||
validateUrl: "https://ollama.com/api/tags",
|
||||
@@ -32,5 +34,6 @@ export default {
|
||||
serviceKinds: ["llm"],
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -52,5 +52,7 @@ export default {
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
// PAT (apikey) connections also carry quota usage (via job-token exchange).
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -1,33 +0,0 @@
|
||||
export default {
|
||||
id: "qwen",
|
||||
hidden: true,
|
||||
priority: 130,
|
||||
alias: "qw",
|
||||
display: {
|
||||
name: "Qwen Code",
|
||||
icon: "psychology",
|
||||
color: "#10B981",
|
||||
website: "https://chat.qwen.ai",
|
||||
notice: {
|
||||
signupUrl: "https://chat.qwen.ai",
|
||||
},
|
||||
},
|
||||
category: "oauth",
|
||||
transport: {
|
||||
baseUrl: "https://portal.qwen.ai/v1/chat/completions",
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3-coder-plus", name: "Qwen3 Coder Plus" },
|
||||
{ id: "qwen3-coder-flash", name: "Qwen3 Coder Flash" },
|
||||
{ id: "vision-model", name: "Qwen3 Vision Model" },
|
||||
{ id: "coder-model", name: "Qwen3.6 Coder Model" },
|
||||
],
|
||||
oauth: {
|
||||
clientId: "f0304373b74a44d2b584a3fb70ca9e56",
|
||||
deviceCodeUrl: "https://chat.qwen.ai/api/v1/oauth2/device/code",
|
||||
tokenUrl: "https://chat.qwen.ai/api/v1/oauth2/token",
|
||||
scope: "openid profile email model.completion",
|
||||
codeChallengeMethod: "S256",
|
||||
refreshLeadMs: 1200000,
|
||||
},
|
||||
};
|
||||
73
open-sse/providers/registry/selfhosted-embedding.js
Normal file
73
open-sse/providers/registry/selfhosted-embedding.js
Normal file
@@ -0,0 +1,73 @@
|
||||
// Self-hosted, OpenAI-compatible embeddings (llama.cpp / llama-server, vLLM,
|
||||
// Infinity, text-embeddings-inference, ...) — the embeddings counterpart of
|
||||
// selfhosted-stt and selfhosted-tts.
|
||||
//
|
||||
// Routing a self-hosted embeddings server already WORKS today, via a custom
|
||||
// provider node: getEmbeddingAdapter() matches `openai-compatible-*` and
|
||||
// `custom-embedding-*` and returns openaiCompatNode, whose buildUrl reads
|
||||
// creds.providerSpecificData.baseUrl. What is missing is a first-class provider,
|
||||
// and the gap is visible rather than functional:
|
||||
//
|
||||
// /v1/embeddings on such a node -> 200, correct vectors
|
||||
// the Embedding page in the dashboard -> the node is not listed at all
|
||||
//
|
||||
// The page renders getProvidersByKind("embedding") plus provider nodes filtered
|
||||
// to `type === "custom-embedding"`. A node created as `openai-compatible` — the
|
||||
// natural choice when ONE endpoint serves chat and embeddings behind the same
|
||||
// front door — satisfies neither, so a working self-hosted embeddings endpoint is
|
||||
// invisible on the page whose job is to show embeddings providers. Diagnosed on a
|
||||
// deployment serving Qwen3-Embedding-8B at 4096 dimensions through exactly that
|
||||
// shape (2026-08-04).
|
||||
//
|
||||
// Declaring it as a provider with serviceKinds: ["embedding"] puts it on the page
|
||||
// beside Voyage, Jina and the rest, and keeps the per-connection baseUrl that
|
||||
// makes self-hosting possible at all.
|
||||
//
|
||||
// authType is "apikey" rather than "none" for the same reason as the STT and TTS
|
||||
// entries: it is what gives the connection a credentials record, and
|
||||
// providerSpecificData.baseUrl lives there. Local servers ignore the key itself;
|
||||
// any non-empty value works.
|
||||
export default {
|
||||
id: "selfhosted-embedding",
|
||||
priority: 50,
|
||||
hasFree: true,
|
||||
alias: "selfhosted-embedding",
|
||||
display: {
|
||||
name: "Self-hosted Embedding",
|
||||
icon: "cloud",
|
||||
color: "#ffffffff",
|
||||
textIcon: "SE",
|
||||
website: "https://github.com/ggml-org/llama.cpp",
|
||||
},
|
||||
category: "apikey",
|
||||
auth: {
|
||||
apiKey: {
|
||||
// Note the /v1: the adapter appends "/embeddings" to whatever it is given,
|
||||
// so a bare http://host:8080 resolves to http://host:8080/embeddings and
|
||||
// misses the OpenAI route entirely. Give it the OpenAI base, the same value
|
||||
// an OpenAI client would use. A trailing /embeddings is tolerated.
|
||||
text: "Set providerSpecificData.baseUrl to the OpenAI base URL, e.g. http://host:8080/v1 — /embeddings is appended. The API key is not checked by local servers; any value works.",
|
||||
},
|
||||
},
|
||||
// A self-hosted server serves whatever model it was started with, so the id
|
||||
// here is a placeholder for the UI: the request passes `model` straight
|
||||
// through, and llama-server ignores an unknown value rather than rejecting it.
|
||||
// Dimensions are deliberately NOT declared — they are a property of the loaded
|
||||
// weights, and asserting a number here would be a guess that silently
|
||||
// contradicts the server.
|
||||
models: [
|
||||
{ id: "embedding", name: "Self-hosted embedding model", kind: "embedding" },
|
||||
],
|
||||
serviceKinds: ["embedding"],
|
||||
embeddingConfig: {
|
||||
// Declared for shape-consistency with the other embedding providers, and
|
||||
// read by the UI — but NOT by the request path. openaiCompatNode resolves the
|
||||
// URL purely from creds.providerSpecificData.baseUrl (falling back to
|
||||
// api.openai.com), so unlike a fixed cloud provider this baseUrl never
|
||||
// reaches the wire. Stated plainly because a reader would otherwise
|
||||
// reasonably assume it is the default endpoint.
|
||||
baseUrl: "http://localhost:8080/v1/embeddings",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
},
|
||||
};
|
||||
48
open-sse/providers/registry/selfhosted-stt.js
Normal file
48
open-sse/providers/registry/selfhosted-stt.js
Normal file
@@ -0,0 +1,48 @@
|
||||
// Self-hosted, OpenAI-compatible speech-to-text (whisper.cpp, faster-whisper,
|
||||
// Speaches, vLLM-served Whisper, ...).
|
||||
//
|
||||
// Every other STT provider here is a named cloud service with a fixed endpoint.
|
||||
// This one exists so a locally-served /v1/audio/transcriptions can be used at
|
||||
// all: set the connection's providerSpecificData.baseUrl to the full URL of the
|
||||
// endpoint, exactly as the custom embedding providers already work.
|
||||
//
|
||||
// sttCore dispatches on `format`; anything that is not one of the five named
|
||||
// cloud shapes falls through to transcribeOpenAICompatible, which POSTs the
|
||||
// standard multipart body (file, model, and optional language / prompt /
|
||||
// response_format / temperature). That is precisely what whisper.cpp's OpenAI
|
||||
// endpoint accepts.
|
||||
//
|
||||
// authType is "apikey" rather than "none" so the connection carries a
|
||||
// credentials record — which is where providerSpecificData.baseUrl lives. Local
|
||||
// servers ignore the key itself; any non-empty value works.
|
||||
export default {
|
||||
id: "selfhosted-stt",
|
||||
priority: 50,
|
||||
hasFree: true,
|
||||
alias: "selfhosted-stt",
|
||||
display: {
|
||||
name: "Self-hosted STT",
|
||||
icon: "cloud",
|
||||
color: "#ffffffff",
|
||||
textIcon: "ST",
|
||||
website: "https://github.com/ggml-org/whisper.cpp",
|
||||
},
|
||||
category: "apikey",
|
||||
auth: {
|
||||
apiKey: {
|
||||
text: "Set providerSpecificData.baseUrl to the full transcriptions URL, e.g. http://host:8080/v1/audio/transcriptions. The API key is not checked by local servers; any value works.",
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "whisper-1", name: "Whisper (self-hosted)", params: ["language", "response_format", "temperature", "prompt"], kind: "stt" },
|
||||
],
|
||||
serviceKinds: ["stt"],
|
||||
sttConfig: {
|
||||
// Overridden per connection by providerSpecificData.baseUrl; this default
|
||||
// only makes the provider usable out of the box on a same-host deployment.
|
||||
baseUrl: "http://localhost:8080/v1/audio/transcriptions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "openai",
|
||||
},
|
||||
};
|
||||
44
open-sse/providers/registry/selfhosted-tts.js
Normal file
44
open-sse/providers/registry/selfhosted-tts.js
Normal file
@@ -0,0 +1,44 @@
|
||||
// Self-hosted, OpenAI-compatible text-to-speech (Kokoro-FastAPI, openedai-speech,
|
||||
// vLLM-served TTS, ...) — the TTS counterpart of selfhosted-stt.
|
||||
//
|
||||
// Every other self-hostable TTS provider here (coqui, tortoise) carries a FIXED
|
||||
// localhost baseUrl in its registry entry and `authType: "none"`, and the generic
|
||||
// dispatcher reads `ttsConfig.baseUrl` from that entry rather than from the
|
||||
// connection. So there was no way to point TTS at a server on another host.
|
||||
//
|
||||
// `authType: "apikey"` is what makes the override possible at all: it gives the
|
||||
// connection a credentials record, which is where providerSpecificData.baseUrl
|
||||
// lives. Local servers ignore the key; any non-empty value works.
|
||||
export default {
|
||||
id: "selfhosted-tts",
|
||||
priority: 50,
|
||||
hasFree: true,
|
||||
alias: "selfhosted-tts",
|
||||
display: {
|
||||
name: "Self-hosted TTS",
|
||||
icon: "cloud",
|
||||
color: "#ffffffff",
|
||||
textIcon: "TT",
|
||||
website: "https://github.com/remsky/Kokoro-FastAPI",
|
||||
},
|
||||
category: "apikey",
|
||||
auth: {
|
||||
apiKey: {
|
||||
text: "Set providerSpecificData.baseUrl to the server root, e.g. http://host:8080 — /v1/audio/speech is appended. The API key is not checked by local servers; any value works.",
|
||||
},
|
||||
},
|
||||
// Voice is selected as "<model>/<voice>", the same convention the OpenAI TTS
|
||||
// adapter uses, so existing clients need no special casing.
|
||||
models: [
|
||||
{ id: "kokoro", name: "Kokoro (self-hosted)", params: ["voice", "response_format", "speed"], kind: "tts" },
|
||||
],
|
||||
serviceKinds: ["tts"],
|
||||
ttsConfig: {
|
||||
// Overridden per connection by providerSpecificData.baseUrl; this default
|
||||
// only makes the provider usable on a same-host deployment.
|
||||
baseUrl: "http://localhost:8880",
|
||||
defaultModel: "kokoro",
|
||||
authType: "apikey",
|
||||
format: "openai-speech",
|
||||
},
|
||||
};
|
||||
162
open-sse/providers/registry/tokenrouter.js
Normal file
162
open-sse/providers/registry/tokenrouter.js
Normal file
@@ -0,0 +1,162 @@
|
||||
export default {
|
||||
id: "tokenrouter",
|
||||
alias: "tokenrouter",
|
||||
aliases: ["tr"],
|
||||
uiAlias: "tokenrouter",
|
||||
display: {
|
||||
name: "TokenRouter",
|
||||
icon: "hub",
|
||||
color: "#0EA5E9",
|
||||
textIcon: "TR",
|
||||
website: "https://www.tokenrouter.com",
|
||||
notice: {
|
||||
text: "OpenAI-compatible gateway. 300+ models (OpenAI, Claude, Gemini, Qwen, DeepSeek, Kimi, GLM, dsb).",
|
||||
apiKeyUrl: "https://www.tokenrouter.com",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
thinkingConfig: {
|
||||
options: ["low", "medium", "high", "xhigh", "max"],
|
||||
defaultMode: "high",
|
||||
},
|
||||
transport: {
|
||||
baseUrl: "https://api.tokenrouter.com/v1/chat/completions",
|
||||
validateUrl: "https://api.tokenrouter.com/v1/models",
|
||||
thinkingFormat: "tokenrouter",
|
||||
},
|
||||
// Seed snapshot from live /v1/models (120 entries). Latest catalogue is
|
||||
// fetched via modelsFetcher; other ids still accepted via passthroughModels.
|
||||
models: [
|
||||
{ id: "MiniMax-Hailuo-2.3", name: "Minimax Hailuo 2.3", kind: "video" },
|
||||
{ id: "MiniMax-M3", name: "Minimax M3" },
|
||||
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5" },
|
||||
{ id: "anthropic/claude-haiku-4.5", name: "Claude Haiku 4.5" },
|
||||
{ id: "anthropic/claude-opus-4.5", name: "Claude Opus 4.5" },
|
||||
{ id: "anthropic/claude-opus-4.6", name: "Claude Opus 4.6" },
|
||||
{ id: "anthropic/claude-opus-4.7", name: "Claude Opus 4.7" },
|
||||
{ id: "anthropic/claude-opus-4.7-fast", name: "Claude Opus 4.7 Fast" },
|
||||
{ id: "anthropic/claude-opus-4.8", name: "Claude Opus 4.8" },
|
||||
{ id: "anthropic/claude-opus-4.8-fast", name: "Claude Opus 4.8 Fast" },
|
||||
{ id: "anthropic/claude-opus-5", name: "Claude Opus 5" },
|
||||
{ id: "anthropic/claude-opus-5-fast", name: "Claude Opus 5 Fast" },
|
||||
{ id: "anthropic/claude-sonnet-4", name: "Claude Sonnet 4" },
|
||||
{ id: "anthropic/claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
|
||||
{ id: "anthropic/claude-sonnet-4.6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "bytedance-seed/seedream-4.5", name: "Seedream 4.5", kind: "image" },
|
||||
{ id: "bytedance-seed/seedream-5.0-lite", name: "Seedream 5.0 Lite", kind: "image" },
|
||||
{ id: "bytedance-seed/seedream-5.0-pro", name: "Seedream 5.0 Pro", kind: "image" },
|
||||
{ id: "claude-haiku-4-5", name: "Claude Haiku 4 5" },
|
||||
{ id: "claude-opus-4-8-m-aws", name: "Claude Opus 4 8 M Aws" },
|
||||
{ id: "deepseek/deepseek-v3.2", name: "Deepseek V3.2" },
|
||||
{ id: "deepseek/deepseek-v4-flash", name: "Deepseek V4 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-flash-0731", name: "Deepseek V4 Flash 0731" },
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "Deepseek V4 Pro" },
|
||||
{ id: "ex/gpt-5.4", name: "Gpt 5.4" },
|
||||
{ id: "google/gemini-2.5-flash-image", name: "Gemini 2.5 Flash Image" },
|
||||
{ id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
|
||||
{ id: "google/gemini-3-pro-image-preview", name: "Gemini 3 Pro Image Preview" },
|
||||
{ id: "google/gemini-3.1-flash-image-preview", name: "Gemini 3.1 Flash Image Preview" },
|
||||
{ id: "google/gemini-3.1-flash-lite-image", name: "Gemini 3.1 Flash Lite Image" },
|
||||
{ id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
|
||||
{ id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash" },
|
||||
{ id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
|
||||
{ id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "google/gemini-embedding-2", name: "Gemini Embedding 2" },
|
||||
{ id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B It" },
|
||||
{ id: "happyhorse-1.0-t2v", name: "Happyhorse 1.0 T2V", kind: "video" },
|
||||
{ id: "kling-3.0-turbo", name: "Kling 3.0 Turbo", kind: "video" },
|
||||
{ id: "kling-v2-6", name: "Kling V2 6", kind: "video" },
|
||||
{ id: "kling-v3", name: "Kling V3", kind: "video" },
|
||||
{ id: "kling-v3-omni", name: "Kling V3 Omni", kind: "video" },
|
||||
{ id: "microsoft/mai-image-2.5", name: "Mai Image 2.5" },
|
||||
{ id: "minimax/minimax-m2-her", name: "Minimax M2 Her" },
|
||||
{ id: "minimax/minimax-m2.1", name: "Minimax M2.1" },
|
||||
{ id: "minimax/minimax-m2.1-highspeed", name: "Minimax M2.1 Highspeed" },
|
||||
{ id: "minimax/minimax-m2.5", name: "Minimax M2.5" },
|
||||
{ id: "minimax/minimax-m2.7", name: "Minimax M2.7" },
|
||||
{ id: "minimax/minimax-m2.7-highspeed", name: "Minimax M2.7 Highspeed" },
|
||||
{ id: "miromind/mirothinker-1-7-deepresearch", name: "Mirothinker 1 7 Deepresearch" },
|
||||
{ id: "miromind/mirothinker-1-7-deepresearch-mini", name: "Mirothinker 1 7 Deepresearch Mini" },
|
||||
{ id: "mistralai/devstral-2512", name: "Devstral 2512" },
|
||||
{ id: "mistralai/mistral-medium-3-5", name: "Mistral Medium 3 5" },
|
||||
{ id: "mistralai/mistral-small-2603", name: "Mistral Small 2603" },
|
||||
{ id: "mistralai/voxtral-small-24b-2507", name: "Voxtral Small 24B 2507" },
|
||||
{ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5" },
|
||||
{ id: "moonshotai/kimi-k2.6", name: "Kimi K2.6" },
|
||||
{ id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "moonshotai/kimi-k3", name: "Kimi K3" },
|
||||
{ id: "moonshotai/kimi-k3-free", name: "Kimi K3 Free" },
|
||||
{ id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", name: "Nemotron 3 Nano Omni 30B A3B Reasoning:Free" },
|
||||
{ id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B" },
|
||||
{ id: "openai/gpt-4o-mini", name: "Gpt 4O Mini" },
|
||||
{ id: "openai/gpt-5", name: "Gpt 5" },
|
||||
{ id: "openai/gpt-5-image", name: "Gpt 5 Image" },
|
||||
{ id: "openai/gpt-5-image-mini", name: "Gpt 5 Image Mini" },
|
||||
{ id: "openai/gpt-5-mini", name: "Gpt 5 Mini" },
|
||||
{ id: "openai/gpt-5.2", name: "Gpt 5.2" },
|
||||
{ id: "openai/gpt-5.4", name: "Gpt 5.4" },
|
||||
{ id: "openai/gpt-5.4-image-2", name: "Gpt 5.4 Image 2", kind: "image" },
|
||||
{ id: "openai/gpt-5.4-mini", name: "Gpt 5.4 Mini" },
|
||||
{ id: "openai/gpt-5.4-nano", name: "Gpt 5.4 Nano" },
|
||||
{ id: "openai/gpt-5.4-pro", name: "Gpt 5.4 Pro" },
|
||||
{ id: "openai/gpt-5.5", name: "Gpt 5.5" },
|
||||
{ id: "openai/gpt-5.5-pro", name: "Gpt 5.5 Pro" },
|
||||
{ id: "openai/gpt-5.6-luna", name: "Gpt 5.6 Luna" },
|
||||
{ id: "openai/gpt-5.6-sol", name: "Gpt 5.6 Sol" },
|
||||
{ id: "openai/gpt-5.6-terra", name: "Gpt 5.6 Terra" },
|
||||
{ id: "openai/gpt-audio", name: "Gpt Audio", kind: "audio" },
|
||||
{ id: "openai/gpt-audio-mini", name: "Gpt Audio Mini", kind: "audio" },
|
||||
{ id: "openai/gpt-oss-120b", name: "Gpt Oss 120B" },
|
||||
{ id: "qwen/qwen3-coder-next", name: "Qwen3 Coder Next" },
|
||||
{ id: "qwen/qwen3.5-122b-a10b", name: "Qwen3.5 122B A10B" },
|
||||
{ id: "qwen/qwen3.5-35b-a3b", name: "Qwen3.5 35B A3B" },
|
||||
{ id: "qwen/qwen3.5-397b-a17b", name: "Qwen3.5 397B A17B" },
|
||||
{ id: "qwen/qwen3.5-9b", name: "Qwen3.5 9B" },
|
||||
{ id: "qwen/qwen3.5-flash", name: "Qwen3.5 Flash" },
|
||||
{ id: "qwen/qwen3.5-plus-02-15", name: "Qwen3.5 Plus 02 15" },
|
||||
{ id: "qwen/qwen3.6-plus", name: "Qwen3.6 Plus" },
|
||||
{ id: "qwen/qwen3.7-max", name: "Qwen3.7 Max" },
|
||||
{ id: "qwen/qwen3.7-plus", name: "Qwen3.7 Plus" },
|
||||
{ id: "qwen/qwen3.8-max", name: "Qwen3.8 Max" },
|
||||
{ id: "qwen3.5-omni-plus", name: "Qwen3.5 Omni Plus" },
|
||||
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
|
||||
{ id: "sakana/fugu-ultra", name: "Fugu Ultra" },
|
||||
{ id: "seed-2-0-code-preview-260328", name: "Seed 2 0 Code Preview 260328" },
|
||||
{ id: "seed-2-0-lite-260428", name: "Seed 2 0 Lite 260428" },
|
||||
{ id: "seed-2-0-mini-260428", name: "Seed 2 0 Mini 260428" },
|
||||
{ id: "seed-2-0-pro-260328", name: "Seed 2 0 Pro 260328" },
|
||||
{ id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash" },
|
||||
{ id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash" },
|
||||
{ id: "tencent/hy3-preview", name: "Hy3 Preview" },
|
||||
{ id: "x-ai/grok-4.1-fast", name: "Grok 4.1 Fast" },
|
||||
{ id: "x-ai/grok-4.20-beta", name: "Grok 4.20 Beta" },
|
||||
{ id: "x-ai/grok-4.3", name: "Grok 4.3" },
|
||||
{ id: "x-ai/grok-4.5", name: "Grok 4.5" },
|
||||
{ id: "x-ai/grok-build-0.1", name: "Grok Build 0.1" },
|
||||
{ id: "xiaomi/mimo-v2-flash", name: "Mimo V2 Flash" },
|
||||
{ id: "xiaomi/mimo-v2-omni", name: "Mimo V2 Omni" },
|
||||
{ id: "xiaomi/mimo-v2-pro", name: "Mimo V2 Pro" },
|
||||
{ id: "xiaomi/mimo-v2.5", name: "Mimo V2.5" },
|
||||
{ id: "xiaomi/mimo-v2.5-pro", name: "Mimo V2.5 Pro" },
|
||||
{ id: "z-ai/glm-4.5-air", name: "Glm 4.5 Air" },
|
||||
{ id: "z-ai/glm-4.6", name: "Glm 4.6" },
|
||||
{ id: "z-ai/glm-4.6v", name: "Glm 4.6V" },
|
||||
{ id: "z-ai/glm-4.7", name: "Glm 4.7" },
|
||||
{ id: "z-ai/glm-5", name: "Glm 5" },
|
||||
{ id: "z-ai/glm-5-turbo", name: "Glm 5 Turbo" },
|
||||
{ id: "z-ai/glm-5.1", name: "Glm 5.1" },
|
||||
{ id: "z-ai/glm-5.2", name: "Glm 5.2" },
|
||||
],
|
||||
serviceKinds: ["llm", "embedding", "image"],
|
||||
embeddingConfig: {
|
||||
baseUrl: "https://api.tokenrouter.com/v1/embeddings",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
},
|
||||
imageConfig: {
|
||||
baseUrl: "https://api.tokenrouter.com/v1/images/generations",
|
||||
},
|
||||
modelsFetcher: { url: "https://api.tokenrouter.com/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -15,10 +15,11 @@ export default {
|
||||
textIcon: "XM",
|
||||
website: "https://xiaomimimo.com",
|
||||
notice: {
|
||||
apiKeyUrl: "https://xiaomimimo.com",
|
||||
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
serviceKinds: ["llm", "tts"],
|
||||
transport: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
validateUrl: "https://api.xiaomimimo.com/v1/models",
|
||||
@@ -42,5 +43,12 @@ export default {
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
|
||||
{ id: "mimo-v2-flash", name: "MiMo V2 Flash" },
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", kind: "tts" },
|
||||
],
|
||||
ttsConfig: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "xiaomi-mimo-tts",
|
||||
},
|
||||
};
|
||||
|
||||
@@ -47,6 +47,26 @@ export const CLAUDE_CLI_SPOOF_HEADERS = {
|
||||
"X-Stainless-Timeout": "600"
|
||||
};
|
||||
|
||||
const ANTHROPIC_BETA_BASE = [
|
||||
"claude-code-20250219",
|
||||
"oauth-2025-04-20",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"structured-outputs-2025-12-15",
|
||||
"fast-mode-2026-02-01",
|
||||
"redact-thinking-2026-02-12",
|
||||
"token-efficient-tools-2026-03-28",
|
||||
];
|
||||
const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"];
|
||||
|
||||
// Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them.
|
||||
export function selectAnthropicBeta(model = "") {
|
||||
const flags = [...ANTHROPIC_BETA_BASE];
|
||||
if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT);
|
||||
return flags.join(",");
|
||||
}
|
||||
|
||||
// Shared baseUrls
|
||||
export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
|
||||
|
||||
|
||||
@@ -31,10 +31,13 @@ const FORMAT_LEVELS = {
|
||||
step: L.base,
|
||||
};
|
||||
|
||||
const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
||||
|
||||
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
|
||||
const PATTERN_THINKING = [
|
||||
// gpt-5.6-sol accepts max (maps to xhigh on wire); live probe rejected ultra.
|
||||
{ pattern: "*gpt-5.6-sol*", levels: ["none", "minimal", "low", "medium", "high", "xhigh", "max"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
];
|
||||
|
||||
@@ -43,7 +46,9 @@ export function getThinkingLevels(provider, model) {
|
||||
if (provider === "kiro" && resolveKiroEffortPath(model) === null) return null;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (!caps.reasoning) return null;
|
||||
const hit = PATTERN_THINKING.find((p) => matchPattern(p.pattern, model));
|
||||
const hit = PATTERN_THINKING.find((entry) =>
|
||||
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
|
||||
);
|
||||
let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
|
||||
if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none");
|
||||
return levels;
|
||||
|
||||
@@ -25,9 +25,17 @@ function messagePayload(body) {
|
||||
|
||||
function captureSizeSnapshot(body) {
|
||||
const messages = messagePayload(body);
|
||||
const toolHistory = messages?.filter((message) =>
|
||||
message?.role === "tool"
|
||||
|| message?.role === "function"
|
||||
|| message?.tool_calls?.length
|
||||
|| message?.content?.some?.((part) => part?.type === "tool_use" || part?.type === "tool_result")
|
||||
) || [];
|
||||
return {
|
||||
bodyBytes: jsonBytes(body),
|
||||
messageBytes: messages ? jsonBytes(messages) : 0,
|
||||
toolSchemaBytes: jsonBytes(body?.tools || []),
|
||||
toolHistoryBytes: jsonBytes(toolHistory),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -336,7 +344,10 @@ export function formatHeadroomSizeLog(diagnostics) {
|
||||
const before = diagnostics?.before;
|
||||
const after = diagnostics?.after;
|
||||
if (!before || !after) return "";
|
||||
return `body=${before.bodyBytes}B→${after.bodyBytes}B messages=${before.messageBytes}B→${after.messageBytes}B`;
|
||||
const effective = before.bodyBytes > 0
|
||||
? (((before.bodyBytes - after.bodyBytes) / before.bodyBytes) * 100).toFixed(1)
|
||||
: "0.0";
|
||||
return `body=${before.bodyBytes}B→${after.bodyBytes}B messages=${before.messageBytes}B→${after.messageBytes}B tools=${before.toolSchemaBytes || 0}B→${after.toolSchemaBytes || 0}B toolHistory=${before.toolHistoryBytes || 0}B→${after.toolHistoryBytes || 0}B effective=${effective}%`;
|
||||
}
|
||||
|
||||
export function isHeadroomPhantomSavings(stats, diagnostics, minShrinkRatio = 0.05) {
|
||||
|
||||
173
open-sse/services/capacityAdapter.js
Normal file
173
open-sse/services/capacityAdapter.js
Normal file
@@ -0,0 +1,173 @@
|
||||
/**
|
||||
* Capacity Adapter — global fallback pools of models per input-modality capability
|
||||
* (vision / pdf / audioInput / videoInput).
|
||||
*
|
||||
* The pool models are appended as extra fallback candidates behind whatever models
|
||||
* were already going to be tried (a combo's members, or a single target model).
|
||||
* combo.js's existing reorderByCapabilities then floats a capable pool model to the
|
||||
* front only when none of the original models can handle the request — so this
|
||||
* never overrides a combo that already has a member covering the capability.
|
||||
*/
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
|
||||
const CAPABILITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
const HARD_CAPS = new Set(CAPABILITY_KEYS);
|
||||
const DEFAULT_FALLBACK_MODEL = "oc/mimo-v2.5-free";
|
||||
|
||||
// Normalize a capability entry to { enabled, roundRobin, models }. Backward-compat:
|
||||
// accept the legacy array form [{model, enabled}] (treated as enabled, fallback).
|
||||
function normalizeCapEntry(entry) {
|
||||
if (Array.isArray(entry)) {
|
||||
return { enabled: true, roundRobin: false, models: entry.map((e) => e?.model || e).filter(Boolean) };
|
||||
}
|
||||
if (entry && typeof entry === "object") {
|
||||
return {
|
||||
enabled: entry.enabled !== false,
|
||||
roundRobin: !!entry.roundRobin,
|
||||
models: Array.isArray(entry.models) ? entry.models.filter(Boolean) : [],
|
||||
};
|
||||
}
|
||||
return { enabled: false, roundRobin: false, models: [] };
|
||||
}
|
||||
|
||||
// Resolve one capability's full config. Enabled pools with no models fall back
|
||||
// to DEFAULT_FALLBACK_MODEL so the toggle is never a no-op.
|
||||
export function getCapacityAdapterConfig(cap, settings) {
|
||||
const entry = normalizeCapEntry(settings?.capacityAdapter?.[cap]);
|
||||
if (entry.enabled && entry.models.length === 0) {
|
||||
return { ...entry, models: [DEFAULT_FALLBACK_MODEL] };
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
// Flatten enabled models across all capability pools, in priority order, deduped.
|
||||
export function getCapacityAdapterModels(settings) {
|
||||
const seen = new Set();
|
||||
const models = [];
|
||||
for (const cap of CAPABILITY_KEYS) {
|
||||
const { enabled, models: pool } = getCapacityAdapterConfig(cap, settings);
|
||||
if (!enabled) continue;
|
||||
for (const m of pool) {
|
||||
if (!seen.has(m)) {
|
||||
seen.add(m);
|
||||
models.push(m);
|
||||
}
|
||||
}
|
||||
}
|
||||
return models;
|
||||
}
|
||||
|
||||
// Strategy for a capability: "round-robin" when enabled+roundRobin, else "fallback".
|
||||
export function getCapacityAdapterStrategy(cap, settings) {
|
||||
const { enabled, roundRobin } = getCapacityAdapterConfig(cap, settings);
|
||||
return enabled && roundRobin ? "round-robin" : "fallback";
|
||||
}
|
||||
|
||||
// Strategy from the request's required capabilities: picks the first capability
|
||||
// whose adapter pool is enabled and can satisfy a hard requirement.
|
||||
export function getActiveAdapterStrategy(requiredCapabilities, settings) {
|
||||
const hard = [...(requiredCapabilities || [])].filter((c) => HARD_CAPS.has(c));
|
||||
for (const cap of hard) {
|
||||
const { enabled, models } = getCapacityAdapterConfig(cap, settings);
|
||||
if (!enabled || models.length === 0) continue;
|
||||
return getCapacityAdapterStrategy(cap, settings);
|
||||
}
|
||||
return "fallback";
|
||||
}
|
||||
|
||||
function modelSatisfies(modelStr, requiredHard) {
|
||||
const slash = modelStr.indexOf("/");
|
||||
const provider = slash > 0 ? modelStr.slice(0, slash) : "";
|
||||
const model = slash > 0 ? modelStr.slice(slash + 1) : modelStr;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
return requiredHard.every((c) => caps[c] === true);
|
||||
}
|
||||
|
||||
// Prepend capacity-adapter models as priority candidates when NONE of the
|
||||
// original models (combo members, or the single target model) can satisfy the
|
||||
// request's required capabilities. Adapter models go FIRST (priority); the
|
||||
// original models follow as fallback. Leaves `models` untouched when the
|
||||
// original list already covers it (combo.js's reorderByCapabilities handles
|
||||
// that case via autoSwitch).
|
||||
export function augmentModelsWithCapacityAdapter(models, requiredCapabilities, settings) {
|
||||
const hard = [...(requiredCapabilities || [])].filter((c) => HARD_CAPS.has(c));
|
||||
if (hard.length === 0 || !Array.isArray(models) || models.length === 0) return models;
|
||||
if (models.some((m) => modelSatisfies(m, hard))) return models;
|
||||
|
||||
const pool = getCapacityAdapterModels(settings).filter((m) => !models.includes(m) && modelSatisfies(m, hard));
|
||||
if (pool.length === 0) return models;
|
||||
return [...pool, ...models];
|
||||
}
|
||||
|
||||
const CHARS_PER_TOKEN = 4; // rough estimate; avoids pulling in a tokenizer dependency
|
||||
const HEAD_KEEP = 6; // messages after system kept verbatim before dropping the middle
|
||||
|
||||
function blockLength(content) {
|
||||
if (typeof content === "string") return content.length;
|
||||
if (Array.isArray(content)) {
|
||||
return content.reduce((sum, b) => sum + (typeof b?.text === "string" ? b.text.length : 50), 0);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Trim history to fit a (possibly smaller) context window by dropping the MIDDLE.
|
||||
// Preserves: all system/instruction messages (head), and the trailing user run
|
||||
// carrying the media the switch happened for (tail). Older middle turns between
|
||||
// the head instructions and the current turn are dropped first.
|
||||
export function stripHistoryForContext(body, contextWindow) {
|
||||
const key = Array.isArray(body.messages) ? "messages"
|
||||
: Array.isArray(body.input) ? "input"
|
||||
: Array.isArray(body.contents) ? "contents"
|
||||
: null;
|
||||
if (!key) return body;
|
||||
const arr = body[key];
|
||||
if (!arr || arr.length === 0) return body;
|
||||
|
||||
const isSystem = (r) => r === "system" || r === "developer";
|
||||
const systemMsgs = arr.filter((m) => isSystem(m?.role));
|
||||
const rest = arr.filter((m) => !isSystem(m?.role));
|
||||
if (rest.length === 0) return body;
|
||||
|
||||
const isAssistant = (r) => r === "assistant" || r === "model";
|
||||
let i = rest.length - 1;
|
||||
while (i >= 0 && !isAssistant(rest[i]?.role)) i--;
|
||||
const tail = rest.slice(i + 1); // current user turn (has media) — always kept
|
||||
const older = rest.slice(0, i + 1); // everything before it
|
||||
if (older.length === 0) return body;
|
||||
|
||||
const contentOf = (m) => m.content ?? m.parts;
|
||||
// Cap at 80% of the adapter model's context window — leaves room for the response.
|
||||
const budgetChars = (contextWindow || 200000) * 0.8 * CHARS_PER_TOKEN;
|
||||
|
||||
// Prefer keeping the first HEAD_KEEP messages (initial instructions/context) verbatim;
|
||||
// only trim further if even that exceeds the adapter model's context window.
|
||||
const headKept = older.slice(0, HEAD_KEEP);
|
||||
let total = systemMsgs.concat(headKept, tail).reduce((s, m) => s + blockLength(contentOf(m)), 0);
|
||||
|
||||
// If head + tail overflow, drop head turns from the end (closest to middle) first.
|
||||
let head = headKept;
|
||||
while (total > budgetChars && head.length > 0) {
|
||||
const dropped = head.pop();
|
||||
total -= blockLength(contentOf(dropped));
|
||||
}
|
||||
|
||||
if (head.length === older.length) return body;
|
||||
return { ...body, [key]: [...systemMsgs, ...head, ...tail] };
|
||||
}
|
||||
|
||||
// Wrap a handleSingleModel callback so calls to a capacity-adapter model strip
|
||||
// history to fit its context window first. No-op passthrough when the pool is empty.
|
||||
export function withCapacityAdapterStripping(handleSingleModel, adapterModels) {
|
||||
const adapterSet = new Set(adapterModels);
|
||||
if (adapterSet.size === 0) return handleSingleModel;
|
||||
return (body, modelStr, ...rest) => {
|
||||
if (adapterSet.has(modelStr)) {
|
||||
const slash = modelStr.indexOf("/");
|
||||
const provider = slash > 0 ? modelStr.slice(0, slash) : "";
|
||||
const model = slash > 0 ? modelStr.slice(slash + 1) : modelStr;
|
||||
const { contextWindow } = getCapabilitiesForModel(provider, model);
|
||||
body = stripHistoryForContext(body, contextWindow);
|
||||
}
|
||||
return handleSingleModel(body, modelStr, ...rest);
|
||||
};
|
||||
}
|
||||
@@ -126,19 +126,33 @@ export function detectRequiredCapabilities(body) {
|
||||
const required = new Set();
|
||||
if (!body || typeof body !== "object") return required;
|
||||
|
||||
const scanBlock = (b) => {
|
||||
if (!b || typeof b !== "object") return;
|
||||
const t = b.type;
|
||||
if (t === "image_url" || t === "image" || t === "input_image")
|
||||
required.add("vision");
|
||||
if (t === "file" || t === "document" || t === "input_file")
|
||||
required.add("pdf");
|
||||
// gemini parts: inlineData/fileData carry a mime
|
||||
const mime = b.inlineData?.mimeType || b.fileData?.mimeType;
|
||||
if (typeof mime === "string" && mime.startsWith("image/"))
|
||||
required.add("vision");
|
||||
if (mime === "application/pdf") required.add("pdf");
|
||||
};
|
||||
const addByMime = (mime) => {
|
||||
if (typeof mime !== "string") return;
|
||||
if (mime.startsWith("image/")) required.add("vision");
|
||||
else if (mime === "application/pdf") required.add("pdf");
|
||||
else if (mime.startsWith("audio/")) required.add("audioInput");
|
||||
else if (mime.startsWith("video/")) required.add("videoInput");
|
||||
};
|
||||
|
||||
const scanBlock = (b) => {
|
||||
if (!b || typeof b !== "object") return;
|
||||
const t = b.type;
|
||||
if (t === "image_url" || t === "image" || t === "input_image") required.add("vision");
|
||||
if (t === "input_audio" || t === "audio_url" || t === "audio") required.add("audioInput");
|
||||
if (t === "input_video" || t === "video_url" || t === "video") required.add("videoInput");
|
||||
if (t === "file" || t === "document" || t === "input_file") {
|
||||
// Infer modality from embedded mime when available; fall back to pdf for generic files.
|
||||
let fmime = null;
|
||||
if (b.input_audio?.format) fmime = `audio/${b.input_audio.format}`;
|
||||
else if (b.file?.file_data) fmime = String(b.file.file_data).match(/^data:([^;,]+)/)?.[1];
|
||||
else if (b.source?.media_type) fmime = b.source.media_type;
|
||||
else if (b.source?.data) fmime = String(b.source.data).match(/^data:([^;,]+)/)?.[1];
|
||||
if (fmime) addByMime(fmime);
|
||||
else required.add("pdf");
|
||||
}
|
||||
// gemini parts: inlineData/fileData carry a mime
|
||||
addByMime(b.inlineData?.mimeType || b.fileData?.mimeType);
|
||||
};
|
||||
|
||||
const scanContent = (content) => {
|
||||
if (Array.isArray(content)) for (const b of content) scanBlock(b);
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
* This significantly reduces the risk of being flagged by Google's anti-abuse systems.
|
||||
*/
|
||||
|
||||
import { CLOUD_CODE_API, LOAD_CODE_ASSIST_HEADERS, LOAD_CODE_ASSIST_METADATA } from "../config/appConstants.js";
|
||||
import { CLOUD_CODE_API, LOAD_CODE_ASSIST_HEADERS, ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS, LOAD_CODE_ASSIST_METADATA } from "../config/appConstants.js";
|
||||
|
||||
// ─── Cache ────────────────────────────────────────────────────────────────────
|
||||
// connectionId -> { projectId: string, fetchedAt: number }
|
||||
@@ -157,9 +157,10 @@ export function removeConnection(connectionId) {
|
||||
*/
|
||||
async function fetchProjectId(accessToken, signal, provider) {
|
||||
const endpoints = CLOUD_CODE_API[provider] || CLOUD_CODE_API["gemini-cli"];
|
||||
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
|
||||
const response = await fetch(endpoints.loadCodeAssist, {
|
||||
method: "POST",
|
||||
headers: { ...LOAD_CODE_ASSIST_HEADERS, "Authorization": `Bearer ${accessToken}` },
|
||||
headers: { ...headers, "Authorization": `Bearer ${accessToken}` },
|
||||
body: JSON.stringify({ metadata: LOAD_CODE_ASSIST_METADATA }),
|
||||
signal
|
||||
});
|
||||
@@ -186,7 +187,7 @@ async function fetchProjectId(accessToken, signal, provider) {
|
||||
}
|
||||
}
|
||||
|
||||
return onboardUser(accessToken, tierID, signal, endpoints);
|
||||
return onboardUser(accessToken, tierID, signal, endpoints, provider);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -197,10 +198,11 @@ async function fetchProjectId(accessToken, signal, provider) {
|
||||
* @param {AbortSignal} externalSignal – propagated from the connection's AbortController
|
||||
* @returns {Promise<string|null>}
|
||||
*/
|
||||
async function onboardUser(accessToken, tierID, externalSignal, endpoints) {
|
||||
async function onboardUser(accessToken, tierID, externalSignal, endpoints, provider) {
|
||||
console.log(`[ProjectId] Onboarding user with tier: ${tierID}`);
|
||||
|
||||
const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA };
|
||||
const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS;
|
||||
const MAX_ATTEMPTS = 5;
|
||||
|
||||
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
|
||||
@@ -216,7 +218,7 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints) {
|
||||
try {
|
||||
const response = await fetch(endpoints.onboardUser, {
|
||||
method: "POST",
|
||||
headers: { ...LOAD_CODE_ASSIST_HEADERS, "Authorization": `Bearer ${accessToken}` },
|
||||
headers: { ...headers, "Authorization": `Bearer ${accessToken}` },
|
||||
body: JSON.stringify(reqBody),
|
||||
signal: localCtrl.signal
|
||||
});
|
||||
|
||||
@@ -19,9 +19,15 @@ function isAnthropicCompatible(provider) {
|
||||
return typeof provider === "string" && provider.startsWith(ANTHROPIC_COMPATIBLE_PREFIX);
|
||||
}
|
||||
|
||||
function getOpenAICompatibleType(provider) {
|
||||
if (!isOpenAICompatible(provider)) return "chat";
|
||||
return provider.includes("responses") ? "responses" : "chat";
|
||||
// Resolve the API type (chat vs responses) for an openai-compatible node.
|
||||
// The stored apiType on the connection's providerSpecificData (kept in sync with
|
||||
// the node on create/update) is authoritative. Falls back to the node ID
|
||||
// substring for legacy nodes created before apiType was persisted — their IDs
|
||||
// embed the type: openai-compatible-<chat|responses>-<uuid>.
|
||||
export function resolveOpenAICompatibleApiType(provider, credentials = null) {
|
||||
const stored = credentials?.providerSpecificData?.apiType;
|
||||
if (stored === "chat" || stored === "responses") return stored;
|
||||
return typeof provider === "string" && provider.includes("responses") ? "responses" : "chat";
|
||||
}
|
||||
|
||||
// Detect request format from body structure
|
||||
@@ -105,9 +111,9 @@ export function detectFormat(body) {
|
||||
}
|
||||
|
||||
// Get provider config (internal — no external runtime consumer)
|
||||
function getProviderConfig(provider) {
|
||||
function getProviderConfig(provider, credentials = null) {
|
||||
if (isOpenAICompatible(provider)) {
|
||||
const apiType = getOpenAICompatibleType(provider);
|
||||
const apiType = resolveOpenAICompatibleApiType(provider, credentials);
|
||||
return {
|
||||
...PROVIDERS.openai,
|
||||
format: apiType === "responses" ? "openai-responses" : "openai",
|
||||
@@ -125,14 +131,14 @@ function getProviderConfig(provider) {
|
||||
}
|
||||
|
||||
// Get target format for provider
|
||||
export function getTargetFormat(provider) {
|
||||
export function getTargetFormat(provider, credentials = null) {
|
||||
if (isOpenAICompatible(provider)) {
|
||||
return getOpenAICompatibleType(provider) === "responses" ? "openai-responses" : "openai";
|
||||
return resolveOpenAICompatibleApiType(provider, credentials) === "responses" ? "openai-responses" : "openai";
|
||||
}
|
||||
if (isAnthropicCompatible(provider)) {
|
||||
return "claude";
|
||||
}
|
||||
const config = getProviderConfig(provider);
|
||||
const config = getProviderConfig(provider, credentials);
|
||||
return config.format || "openai";
|
||||
}
|
||||
|
||||
|
||||
@@ -10,6 +10,12 @@
|
||||
*
|
||||
* On any error the live cache stays empty and chatExecuteCall surfaces the
|
||||
* problem to the user as "model config not yet fetched, retry shortly".
|
||||
*
|
||||
* PAT (Personal Access Token, pt-...) connections: a PAT cannot sign COSY
|
||||
* requests directly, so we exchange it for a short-lived job token (jt-...)
|
||||
* via openapi.qoder.sh/api/v1/jobToken/exchange (plain JSON POST), then use
|
||||
* that job token for signing. Job-token traffic must hit api2.qoder.sh —
|
||||
* api3 rejects jt- with "Login expired" (403).
|
||||
*/
|
||||
|
||||
import { createHash } from "crypto";
|
||||
@@ -18,11 +24,30 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { buildCosyHeaders } from "../shared/qoder/cosy.js";
|
||||
import {
|
||||
QODER_MODEL_LIST_URL,
|
||||
QODER_CHAT_BASE_ALT,
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
QODER_USERINFO_URL,
|
||||
QODER_IDE_VERSION,
|
||||
QODER_CLIENT_TYPE,
|
||||
} from "../shared/qoder/constants.js";
|
||||
|
||||
const FETCH_TIMEOUT_MS = 15_000;
|
||||
const CACHE_TTL_MS = 60 * 60 * 1000; // 1h, same as the Kiro catalog
|
||||
|
||||
const PAT_PREFIX = "pt-";
|
||||
|
||||
// PAT → job-token cache: a job token is short-lived (24h), so we keep it per
|
||||
// PAT and re-exchange once it is within 5 minutes of expiry.
|
||||
const PAT_REFRESH_BUFFER_MS = 5 * 60 * 1000;
|
||||
const PAT_DEFAULT_TTL_MS = 24 * 60 * 60 * 1000;
|
||||
|
||||
export function isQoderPat(token) {
|
||||
return typeof token === "string" && token.startsWith(PAT_PREFIX);
|
||||
}
|
||||
|
||||
/** @type {Map<string, { accessToken: string, userId: string, expiresAt: number }>} */
|
||||
const patJobCache = new Map();
|
||||
|
||||
/** @type {Map<string, { expiresAt: number, models: any[], rawConfigs: Map<string, object>, fetched: boolean }>} */
|
||||
const catalogCache = new Map();
|
||||
|
||||
@@ -34,6 +59,109 @@ const catalogCache = new Map();
|
||||
*/
|
||||
const inflight = new Map();
|
||||
|
||||
/**
|
||||
* Exchange a Qoder PAT (pt-...) for a short-lived job token (jt-...).
|
||||
* This endpoint is plain JSON POST — NOT COSY-signed.
|
||||
*/
|
||||
async function exchangeJobToken(pat, proxyOptions = null, signal = null) {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_JOB_TOKEN_EXCHANGE_URL,
|
||||
{
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
"Cosy-Version": QODER_IDE_VERSION,
|
||||
"Cosy-ClientType": QODER_CLIENT_TYPE,
|
||||
},
|
||||
body: JSON.stringify({ personal_token: pat }),
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
throw new Error(`qoder PAT exchange failed: ${res.status} ${text.slice(0, 200)}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
if (!data.token) throw new Error("qoder PAT exchange returned no job token");
|
||||
|
||||
let expiresAt = Date.now() + PAT_DEFAULT_TTL_MS;
|
||||
if (data.expires_at) {
|
||||
const parsed = Date.parse(data.expires_at);
|
||||
if (!Number.isNaN(parsed)) expiresAt = parsed;
|
||||
} else if (typeof data.expires_in === "number" && data.expires_in > 0) {
|
||||
expiresAt = Date.now() + data.expires_in;
|
||||
}
|
||||
return { jobToken: data.token, jobRefreshToken: data.refresh_token || "", expiresAt };
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the Qoder userId for a job token (needed for COSY signing).
|
||||
* Returns "" on any failure — callers fall back to the stored userId.
|
||||
*/
|
||||
async function fetchUserIdForJobToken(jobToken, proxyOptions = null, signal = null) {
|
||||
try {
|
||||
const res = await proxyAwareFetch(
|
||||
QODER_USERINFO_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${jobToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": "qodercli/1.0.0",
|
||||
},
|
||||
signal,
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
if (!res.ok) return "";
|
||||
const data = await res.json().catch(() => ({}));
|
||||
return data.id || data.userId || data.user_id || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a PAT to a job-token credential, cached per-PAT.
|
||||
*/
|
||||
async function resolvePatCredential(pat, proxyOptions = null, signal = null) {
|
||||
const cached = patJobCache.get(pat);
|
||||
if (cached && cached.expiresAt - Date.now() > PAT_REFRESH_BUFFER_MS) return cached;
|
||||
|
||||
const { jobToken, expiresAt } = await exchangeJobToken(pat, proxyOptions, signal);
|
||||
const userId = await fetchUserIdForJobToken(jobToken, proxyOptions, signal);
|
||||
const resolved = { accessToken: jobToken, userId, expiresAt };
|
||||
patJobCache.set(pat, resolved);
|
||||
return resolved;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve connection credentials to COSY-signable form:
|
||||
* - PAT (pt-...) connections → exchanged to a job token (jt-...) + userId
|
||||
* - everything else → passed through unchanged
|
||||
*/
|
||||
export async function resolveQoderCredentials(credentials, proxyOptions = null, signal = null) {
|
||||
const raw = credentials?.apiKey || credentials?.accessToken;
|
||||
if (isQoderPat(raw)) {
|
||||
const resolved = await resolvePatCredential(raw, proxyOptions, signal);
|
||||
return {
|
||||
...credentials,
|
||||
accessToken: resolved.accessToken,
|
||||
apiKey: undefined,
|
||||
providerSpecificData: {
|
||||
authMethod: "pat",
|
||||
...(credentials?.providerSpecificData || {}),
|
||||
userId: resolved.userId || credentials?.providerSpecificData?.userId || "",
|
||||
machineId: credentials?.providerSpecificData?.machineId || "",
|
||||
},
|
||||
};
|
||||
}
|
||||
return credentials;
|
||||
}
|
||||
|
||||
/**
|
||||
* Stable cache key per credential (so different login sessions for the same
|
||||
* account share an entry).
|
||||
@@ -68,10 +196,16 @@ async function fetchQoderCatalogRaw(credentials, signal, proxyOptions = null) {
|
||||
const creds = cosyCredsFromConnection(credentials);
|
||||
if (!creds.userId || !creds.authToken) return null;
|
||||
|
||||
// Job-token traffic is rejected by api3 ("Login expired" 403) — the
|
||||
// official qodercli serves it from api2 instead.
|
||||
const modelListUrl = String(creds.authToken).startsWith("jt-")
|
||||
? `${QODER_CHAT_BASE_ALT}/algo/api/v2/model/list`
|
||||
: QODER_MODEL_LIST_URL;
|
||||
|
||||
const headers = {
|
||||
Accept: "application/json",
|
||||
"Accept-Encoding": "identity",
|
||||
...buildCosyHeaders(Buffer.alloc(0), QODER_MODEL_LIST_URL, creds),
|
||||
...buildCosyHeaders(Buffer.alloc(0), modelListUrl, creds),
|
||||
};
|
||||
|
||||
const controller = new AbortController();
|
||||
@@ -92,7 +226,7 @@ async function fetchQoderCatalogRaw(credentials, signal, proxyOptions = null) {
|
||||
}
|
||||
}
|
||||
response = await proxyAwareFetch(
|
||||
QODER_MODEL_LIST_URL,
|
||||
modelListUrl,
|
||||
{
|
||||
method: "GET",
|
||||
headers,
|
||||
@@ -159,11 +293,16 @@ export async function getQoderModelConfig(credentials, modelKey, options = {}) {
|
||||
* one upstream request per credential.
|
||||
*/
|
||||
export async function resolveQoderModels(credentials, options = {}) {
|
||||
if (!credentials?.accessToken) return null;
|
||||
const psd = credentials.providerSpecificData || {};
|
||||
if (!psd.userId) return null;
|
||||
let resolved;
|
||||
try {
|
||||
resolved = await resolveQoderCredentials(credentials, options.proxyOptions, options.signal);
|
||||
} catch (error) {
|
||||
options.log?.warn?.("QODER", `PAT exchange failed: ${error.message}`);
|
||||
return null;
|
||||
}
|
||||
if (!resolved?.accessToken || !(resolved.providerSpecificData || {}).userId) return null;
|
||||
|
||||
const key = cacheKey(credentials);
|
||||
const key = cacheKey(resolved);
|
||||
const now = Date.now();
|
||||
if (!options.forceRefresh) {
|
||||
const cached = catalogCache.get(key);
|
||||
@@ -180,7 +319,7 @@ export async function resolveQoderModels(credentials, options = {}) {
|
||||
}
|
||||
|
||||
const fetchPromise = (async () => {
|
||||
const fetched = await fetchQoderCatalogRaw(credentials, options.signal, options.proxyOptions);
|
||||
const fetched = await fetchQoderCatalogRaw(resolved, options.signal, options.proxyOptions);
|
||||
if (!fetched) return null;
|
||||
const entry = {
|
||||
expiresAt: Date.now() + CACHE_TTL_MS,
|
||||
|
||||
@@ -6,7 +6,6 @@ import {
|
||||
refreshKimiToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshKiroToken,
|
||||
refreshIflowToken,
|
||||
@@ -26,7 +25,6 @@ export {
|
||||
refreshKimiToken,
|
||||
refreshClaudeOAuthToken,
|
||||
refreshGoogleToken,
|
||||
refreshQwenToken,
|
||||
refreshCodexToken,
|
||||
refreshKiroToken,
|
||||
refreshIflowToken,
|
||||
@@ -137,7 +135,6 @@ const REFRESH_HANDLERS = {
|
||||
antigravity: (c, log) => refreshGoogleToken(c.refreshToken, PROVIDERS.antigravity.clientId, PROVIDERS.antigravity.clientSecret, log),
|
||||
claude: (c, log) => refreshClaudeOAuthToken(c.refreshToken, log),
|
||||
codex: (c, log) => refreshCodexToken(c.refreshToken, log),
|
||||
qwen: (c, log) => refreshQwenToken(c.refreshToken, log),
|
||||
iflow: (c, log) => refreshIflowToken(c.refreshToken, log),
|
||||
github: (c, log) => refreshGitHubToken(c.refreshToken, log),
|
||||
kiro: (c, log) => refreshKiroToken(c.refreshToken, c.providerSpecificData, log),
|
||||
@@ -205,7 +202,6 @@ export function formatProviderCredentials(provider, credentials, log) {
|
||||
};
|
||||
|
||||
case "codex":
|
||||
case "qwen":
|
||||
case "iflow":
|
||||
case "openai":
|
||||
case "openrouter":
|
||||
|
||||
@@ -40,11 +40,6 @@ const REFRESH_PROFILES = {
|
||||
url: () => OAUTH_ENDPOINTS.anthropic.token,
|
||||
dedupKey: "claude",
|
||||
},
|
||||
qwen: {
|
||||
url: () => OAUTH_ENDPOINTS.qwen.token,
|
||||
dedupKey: "qwen",
|
||||
parse: (tokens) => tokens.resource_url ? { providerSpecificData: { resourceUrl: tokens.resource_url } } : {},
|
||||
},
|
||||
iflow: {
|
||||
url: () => OAUTH_ENDPOINTS.iflow.token,
|
||||
dedupKey: "iflow",
|
||||
@@ -191,11 +186,6 @@ export async function refreshGoogleToken(refreshToken, clientId, clientSecret, l
|
||||
}, log);
|
||||
}
|
||||
|
||||
// Qwen: form body + clientId, surfaces resource_url. Delegate to refreshAccessToken("qwen", ...).
|
||||
export async function refreshQwenToken(refreshToken, log) {
|
||||
return refreshAccessToken("qwen", refreshToken, {}, log);
|
||||
}
|
||||
|
||||
export function classifyOAuthRefreshError(errorText = "", status = 0) {
|
||||
let parsed = null;
|
||||
try {
|
||||
|
||||
@@ -10,14 +10,14 @@ import { getCodexUsage, consumeCodexRateLimitResetCredit, getCodexRateLimitReset
|
||||
export { consumeCodexRateLimitResetCredit, getCodexRateLimitResetCredits };
|
||||
import { getKiroUsage } from "./usage/kiro.js";
|
||||
import { getMiniMaxUsage } from "./usage/minimax.js";
|
||||
import { getCodeBuddyCnUsage } from "./usage/codebuddy-cn.js";
|
||||
import { getCodeBuddyCnUsage, getCodeBuddyIntlUsage } from "./usage/codebuddy-cn.js";
|
||||
import { getXaiUsage } from "./usage/xai.js";
|
||||
import { getGrokCliUsage } from "./usage/grok-cli.js";
|
||||
import { getKimiUsage } from "./usage/kimi.js";
|
||||
import { getDeepseekUsage } from "./usage/deepseek.js";
|
||||
import { getCommandCodeUsage } from "./usage/commandcode.js";
|
||||
import { resolveQoderCredentials } from "./qoderModels.js";
|
||||
import {
|
||||
getQwenUsage,
|
||||
getIflowUsage,
|
||||
getOllamaUsage,
|
||||
getGlmUsage,
|
||||
@@ -38,10 +38,14 @@ const USAGE_HANDLERS = {
|
||||
claude: (c) => getClaudeUsage(c.accessToken, c.proxyOptions),
|
||||
codex: (c) => getCodexUsage(c.accessToken, c.proxyOptions),
|
||||
kiro: (c) => getKiroUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
qoder: (c) => getQoderUsage(c.accessToken, c.proxyOptions),
|
||||
qwen: (c) => getQwenUsage(c.accessToken, c.providerSpecificData),
|
||||
qoder: async (c) => {
|
||||
// PAT (pt-...) connections must be exchanged to a job token before the
|
||||
// quota endpoint accepts them.
|
||||
const resolved = await resolveQoderCredentials(c, c.proxyOptions).catch(() => null);
|
||||
return getQoderUsage(resolved?.accessToken || c.accessToken, c.proxyOptions);
|
||||
},
|
||||
iflow: (c) => getIflowUsage(c.accessToken),
|
||||
ollama: (c) => getOllamaUsage(c.accessToken),
|
||||
ollama: (c) => getOllamaUsage(c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
glm: (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
"glm-cn": (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
minimax: (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
|
||||
@@ -49,6 +53,7 @@ const USAGE_HANDLERS = {
|
||||
"vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions),
|
||||
"codebuddy-cn": (c) => getCodeBuddyCnUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
xai: (c) => getXaiUsage(c.accessToken, c.proxyOptions),
|
||||
"codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData),
|
||||
deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions),
|
||||
|
||||
@@ -43,17 +43,17 @@ function refillCadence(acc) {
|
||||
return "Monthly";
|
||||
}
|
||||
|
||||
export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
async function getCodeBuddyUsage(providerId, accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
const token = accessToken || apiKey;
|
||||
if (!token) {
|
||||
return { message: "CodeBuddy CN credential not available." };
|
||||
return { message: `CodeBuddy (${providerId}) credential not available.` };
|
||||
}
|
||||
|
||||
try {
|
||||
const response = await proxyAwareFetch(U(PROVIDER_ID).url, {
|
||||
const response = await proxyAwareFetch(U(providerId).url, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
...(PROVIDERS[PROVIDER_ID]?.headers || {}),
|
||||
...(PROVIDERS[providerId]?.headers || {}),
|
||||
Authorization: `Bearer ${token}`,
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
@@ -129,10 +129,18 @@ export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificD
|
||||
});
|
||||
|
||||
const basePkg = refills[0] || accounts[0] || {};
|
||||
const plan = basePkg.PackageName || basePkg.SubProductName || "CodeBuddy CN";
|
||||
const plan = basePkg.PackageName || basePkg.SubProductName || "CodeBuddy";
|
||||
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: `CodeBuddy CN error: ${error.message}` };
|
||||
return { message: `CodeBuddy (${providerId}) error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
export async function getCodeBuddyCnUsage(accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
return getCodeBuddyUsage(PROVIDER_ID, accessToken, apiKey, providerSpecificData, proxyOptions);
|
||||
}
|
||||
|
||||
export async function getCodeBuddyIntlUsage(accessToken, apiKey, providerSpecificData, proxyOptions = null) {
|
||||
return getCodeBuddyUsage("codebuddy-intl", accessToken, apiKey, providerSpecificData, proxyOptions);
|
||||
}
|
||||
|
||||
@@ -161,7 +161,9 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
if (data.models) {
|
||||
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
|
||||
const importantModels = [
|
||||
'gemini-3-flash-agent',
|
||||
'gemini-3.6-flash-high',
|
||||
'gemini-3.6-flash-medium',
|
||||
'gemini-3.6-flash-low',
|
||||
'gemini-3.5-flash-low',
|
||||
'gemini-3.5-flash-extra-low',
|
||||
'gemini-pro-agent',
|
||||
@@ -169,10 +171,8 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
'claude-sonnet-4-6',
|
||||
'claude-opus-4-6-thinking',
|
||||
'gpt-oss-120b-medium',
|
||||
'gemini-3-flash',
|
||||
// Image generation models
|
||||
'gemini-3.1-flash-image',
|
||||
'gemini-3-pro-image',
|
||||
];
|
||||
|
||||
for (const [modelKey, info] of Object.entries(data.models)) {
|
||||
|
||||
@@ -91,6 +91,24 @@ function resolvePlan(user, config) {
|
||||
return "Grok Build";
|
||||
}
|
||||
|
||||
// Display only; upstream remains authoritative for access and quota enforcement.
|
||||
function planFromAccessToken(accessToken) {
|
||||
try {
|
||||
const payload = JSON.parse(Buffer.from(accessToken.split(".")[1], "base64url"));
|
||||
return {
|
||||
0: "Free",
|
||||
1: "SuperGrok",
|
||||
2: "X Basic",
|
||||
3: "X Premium",
|
||||
4: "X Premium Plus",
|
||||
5: "SuperGrok Heavy",
|
||||
6: "SuperGrok Lite",
|
||||
}[payload.tier] || "";
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
function makeQuota({ used, total, resetAt, unlimited = false }) {
|
||||
const safeTotal = Math.max(0, toFiniteNumber(total, 0));
|
||||
const safeUsed = Math.max(0, toFiniteNumber(used, 0));
|
||||
@@ -371,6 +389,7 @@ export async function getGrokCliUsage(accessToken, providerSpecificData = null,
|
||||
}
|
||||
|
||||
const parsed = parseGrokCliBilling(billing, user);
|
||||
parsed.plan = planFromAccessToken(accessToken) || parsed.plan;
|
||||
|
||||
if (!parsed.quotas || Object.keys(parsed.quotas).length === 0) {
|
||||
// Paid SuperGrok often returns cap=0 over REST but exposes the shared
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* Misc usage handlers (Qwen, iFlow, Ollama, GLM, Vercel AI Gateway, Qoder)
|
||||
* Misc usage handlers (iFlow, Ollama, GLM, Vercel AI Gateway, Qoder)
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
@@ -15,23 +15,6 @@ const GLM_QUOTA_URLS = {
|
||||
// Returns { balance: "95.50", total_used: "4.50" } (USD as decimal strings).
|
||||
const VERCEL_AI_GATEWAY_CREDITS_URL = U("vercel-ai-gateway").url;
|
||||
|
||||
/**
|
||||
* Qwen Usage
|
||||
*/
|
||||
export async function getQwenUsage(accessToken, providerSpecificData) {
|
||||
try {
|
||||
const resourceUrl = providerSpecificData?.resourceUrl;
|
||||
if (!resourceUrl) {
|
||||
return { message: "Qwen connected. No resource URL available." };
|
||||
}
|
||||
|
||||
// Qwen may have usage endpoint at resource URL
|
||||
return { message: "Qwen connected. Usage tracked per request." };
|
||||
} catch (error) {
|
||||
return { message: "Unable to fetch Qwen usage." };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* iFlow Usage
|
||||
*/
|
||||
@@ -46,23 +29,86 @@ export async function getIflowUsage(accessToken) {
|
||||
|
||||
/**
|
||||
* Ollama Cloud Usage
|
||||
* Ollama Cloud uses an API key from ollama.com/settings/keys
|
||||
* and has no public usage API — free tier has light usage limits (resets every 5h & 7d).
|
||||
* This returns an informational message with the plan details.
|
||||
* GET https://ollama.com/api/usage — session (5h) + weekly (7d) `usage` is a 0..1
|
||||
* ratio (1.0 = limit reached, e.g. weekly 100% used). No reset timestamp exposed.
|
||||
* POST https://ollama.com/api/me — plan label (fail-open).
|
||||
* Auth: Authorization: Bearer <apiKey>
|
||||
*/
|
||||
export async function getOllamaUsage(accessToken, providerSpecificData) {
|
||||
export async function getOllamaUsage(apiKey, providerSpecificData, proxyOptions = null) {
|
||||
if (!apiKey) {
|
||||
return { message: "Ollama Cloud API key not available." };
|
||||
}
|
||||
|
||||
try {
|
||||
// Ollama Cloud does not expose a public quota/usage API.
|
||||
// The provider is configured as noAuth with a notice explaining limits.
|
||||
// We return a graceful message so the UI shows a friendly state instead of an error.
|
||||
const plan = providerSpecificData?.plan || "Free";
|
||||
return {
|
||||
plan,
|
||||
message: "Ollama Cloud uses a free tier with light usage limits (resets every 5h & 7d). For detailed usage tracking, visit ollama.com/settings/keys.",
|
||||
quotas: [],
|
||||
};
|
||||
const response = await proxyAwareFetch("https://ollama.com/api/usage", {
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
Accept: "application/json",
|
||||
},
|
||||
}, proxyOptions);
|
||||
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
return { message: "Ollama Cloud API key invalid or expired." };
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
return { message: `Ollama Cloud usage API error (${response.status}).` };
|
||||
}
|
||||
|
||||
let data;
|
||||
try {
|
||||
data = await response.json();
|
||||
} catch {
|
||||
return { message: "Ollama Cloud usage response was not JSON." };
|
||||
}
|
||||
|
||||
// Best-effort plan label from /api/me
|
||||
const me = await proxyAwareFetch("https://ollama.com/api/me", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
Accept: "application/json",
|
||||
"Content-Length": "0",
|
||||
},
|
||||
}, proxyOptions).then((r) => (r.ok ? r.json() : null)).catch(() => null);
|
||||
|
||||
const planRaw = typeof me?.Plan === "string" ? me.Plan : "";
|
||||
const plan = planRaw
|
||||
? planRaw.charAt(0).toUpperCase() + planRaw.slice(1).toLowerCase()
|
||||
: "Ollama Cloud";
|
||||
|
||||
const limits = data?.limits && typeof data.limits === "object" ? data.limits : {};
|
||||
|
||||
// Ollama `usage` is a 0..1 ratio (1.0 = limit reached). Convert to a 0..100
|
||||
// bar. Do NOT set absolute `remaining` — QuotaTable reads remainingPercentage.
|
||||
function ratioQuota(usageRatio, resetAt = null) {
|
||||
const ratio = Math.max(0, Math.min(1, Number(usageRatio) || 0));
|
||||
const usedPct = Math.round(ratio * 100);
|
||||
return { used: usedPct, total: 100, remainingPercentage: 100 - usedPct, resetAt, unlimited: false };
|
||||
}
|
||||
|
||||
const sessionRaw = limits.session?.usage;
|
||||
const weeklyRaw = limits.weekly?.usage;
|
||||
const sessionNum = Number(sessionRaw);
|
||||
const weeklyNum = Number(weeklyRaw);
|
||||
const hasSession = sessionRaw !== undefined && sessionRaw !== null && !Number.isNaN(sessionNum);
|
||||
const hasWeekly = weeklyRaw !== undefined && weeklyRaw !== null && !Number.isNaN(weeklyNum);
|
||||
|
||||
if (!hasSession && !hasWeekly) {
|
||||
return {
|
||||
plan,
|
||||
message: "Ollama Cloud connected. No usage limits reported.",
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
const quotas = {};
|
||||
if (hasSession) quotas["Session (5h)"] = ratioQuota(sessionNum);
|
||||
if (hasWeekly) quotas["Weekly (7d)"] = ratioQuota(weeklyNum);
|
||||
|
||||
return { plan, quotas };
|
||||
} catch (error) {
|
||||
return { message: "Unable to fetch Ollama Cloud usage." };
|
||||
return { message: `Ollama Cloud error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,9 @@
|
||||
export const QODER_OPENAPI_BASE = "https://openapi.qoder.sh";
|
||||
export const QODER_CENTER_BASE = "https://center.qoder.sh";
|
||||
export const QODER_CHAT_BASE = "https://api3.qoder.sh";
|
||||
// Job-token (jt-...) traffic is rejected by api3 with "Login expired" (403);
|
||||
// the official qodercli serves it from api2 instead.
|
||||
export const QODER_CHAT_BASE_ALT = "https://api2.qoder.sh";
|
||||
|
||||
export const QODER_LOGIN_URL = "https://qoder.com/device/selectAccounts";
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
// never hardcoded per-model here. See .docs/thinking/plan.md MATRIX VI-A.
|
||||
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
import { getThinkingLevels } from "../../providers/thinkingLevels.js";
|
||||
import { PROVIDERS } from "../../providers/index.js";
|
||||
import { LEVEL_TO_BUDGET, budgetToLevel, effortToBudget, effortToThinkingLevel } from "./thinking.js";
|
||||
|
||||
@@ -37,6 +38,7 @@ export function parseSuffix(model) {
|
||||
const raw = m[2].trim().toLowerCase();
|
||||
if (raw === "none" || raw === "off") return { cleanModel, override: { mode: "none" } };
|
||||
if (raw === "auto") return { cleanModel, override: { mode: "auto" } };
|
||||
if (raw === "ultra") return { cleanModel, override: { mode: "level", level: raw } };
|
||||
if (/^\d+$/.test(raw)) return { cleanModel, override: { mode: "budget", budget: Number(raw) } };
|
||||
if (LEVEL_TO_BUDGET[raw] !== undefined) return { cleanModel, override: { mode: "level", level: raw } };
|
||||
return { cleanModel, override: null };
|
||||
@@ -134,6 +136,13 @@ function toLevel(cfg) {
|
||||
return null;
|
||||
}
|
||||
|
||||
function normalizeOpenAILevel(level, supportedLevels) {
|
||||
if (level !== "max" && level !== "ultra") return level;
|
||||
if (supportedLevels?.includes(level)) return level;
|
||||
if (level === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
return "xhigh";
|
||||
}
|
||||
|
||||
function toGeminiThinkingLevel(cfg) {
|
||||
const raw = cfg.mode === "auto" ? "high" : (toLevel(cfg) || "high");
|
||||
return effortToThinkingLevel(raw);
|
||||
@@ -213,7 +222,7 @@ function stripAll(body) {
|
||||
}
|
||||
|
||||
// Apply unified thinking config to body in the resolved provider-native format.
|
||||
function applyFormat(fmt, body, cfg, caps) {
|
||||
function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
const none = cfg.mode === "none";
|
||||
const canDisable = caps.thinkingCanDisable !== false;
|
||||
// Model cannot disable thinking → clamp "none" to minimal effort instead.
|
||||
@@ -223,8 +232,7 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "openai": {
|
||||
if (none && canDisable) { body.reasoning_effort = "none"; break; }
|
||||
const level = toLevel(eff);
|
||||
// OpenAI reasoning_effort enum caps at "xhigh" (no "max"); clamp Claude Code's "max".
|
||||
if (level) body.reasoning_effort = level === "max" ? "xhigh" : level;
|
||||
if (level) body.reasoning_effort = normalizeOpenAILevel(level, supportedLevels);
|
||||
break;
|
||||
}
|
||||
case "claude-adaptive": {
|
||||
@@ -302,6 +310,15 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
if (level) body.reasoning_effort = level === "xhigh" || level === "max" ? "high" : level;
|
||||
break;
|
||||
}
|
||||
case "tokenrouter": {
|
||||
// TokenRouter's reasoning_effort enum is low/medium/high/xhigh/max — it rejects
|
||||
// "none"/"auto" with a 400 and supports "max" natively (no clamp like openai).
|
||||
// "none" → omit the field so the upstream default applies; pass levels through.
|
||||
if (none || eff.mode === "auto") break;
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level;
|
||||
break;
|
||||
}
|
||||
case "kiro":
|
||||
// Kiro thinking handled via system-tag injection in openai-to-kiro.js; no body field here.
|
||||
break;
|
||||
@@ -329,7 +346,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
|
||||
if (!cfg) return body;
|
||||
|
||||
const fmt = resolveFormat(targetFormat, cleanModel, provider);
|
||||
const supportedLevels = getThinkingLevels(provider, cleanModel);
|
||||
stripAll(body);
|
||||
applyFormat(fmt, body, cfg, caps);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels);
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -16,7 +16,9 @@ export function hasValidContent(msg) {
|
||||
return msg.content.some(block =>
|
||||
(block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
|
||||
block.type === CLAUDE_BLOCK.TOOL_USE ||
|
||||
block.type === CLAUDE_BLOCK.TOOL_RESULT
|
||||
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
|
||||
block.type === CLAUDE_BLOCK.IMAGE ||
|
||||
block.type === CLAUDE_BLOCK.DOCUMENT
|
||||
);
|
||||
}
|
||||
return false;
|
||||
|
||||
@@ -7,7 +7,13 @@ import { OPENAI_BLOCK } from "../schema/index.js";
|
||||
export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
|
||||
// Basic constraints (not supported by Gemini API)
|
||||
"minLength", "maxLength", "exclusiveMinimum", "exclusiveMaximum",
|
||||
"minItems", "maxItems", "format",
|
||||
"minItems", "maxItems", "format", "multipleOf",
|
||||
// Array keywords the Gemini schema proto has no field for. Agent tool
|
||||
// schemas set these routinely, and one occurrence rejects the whole request
|
||||
// with "Unknown name ...: Cannot find field".
|
||||
"uniqueItems", "contains",
|
||||
// 2020-12 keywords with no Gemini equivalent
|
||||
"unevaluatedProperties", "unevaluatedItems", "contentSchema",
|
||||
// Claude rejects these in VALIDATED mode
|
||||
"default", "examples",
|
||||
// JSON Schema meta keywords
|
||||
|
||||
@@ -258,8 +258,10 @@ export function initState(sourceFormat) {
|
||||
funcArgsBuf: {},
|
||||
funcNames: {},
|
||||
funcCallIds: {},
|
||||
funcItemAdded: {},
|
||||
funcArgsDone: {},
|
||||
funcItemDone: {},
|
||||
customToolNames: new Set(),
|
||||
completedSent: false
|
||||
};
|
||||
}
|
||||
|
||||
@@ -32,6 +32,8 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
let pendingToolResults = [];
|
||||
let pendingReasoning = "";
|
||||
let pendingReasoningEncrypted = "";
|
||||
const additionalTools = [];
|
||||
const customToolNames = new Set();
|
||||
|
||||
const inputItems = normalizeResponsesInput(body.input);
|
||||
if (!inputItems) return body;
|
||||
@@ -96,7 +98,7 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
}
|
||||
result.messages.push(msg);
|
||||
}
|
||||
else if (itemType === RESPONSES_ITEM.FUNCTION_CALL) {
|
||||
else if (itemType === RESPONSES_ITEM.FUNCTION_CALL || itemType === RESPONSES_ITEM.CUSTOM_TOOL_CALL) {
|
||||
// Start or append to assistant message with tool_calls
|
||||
if (!currentAssistantMsg) {
|
||||
currentAssistantMsg = {
|
||||
@@ -108,16 +110,20 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
}
|
||||
// Skip items with empty/missing name — Codex/OpenAI reject nameless tool calls (#444)
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") continue;
|
||||
if (itemType === RESPONSES_ITEM.CUSTOM_TOOL_CALL) customToolNames.add(item.name);
|
||||
const toolInput = itemType === RESPONSES_ITEM.CUSTOM_TOOL_CALL
|
||||
? { input: typeof item.input === "string" ? item.input : JSON.stringify(item.input ?? "") }
|
||||
: item.arguments;
|
||||
currentAssistantMsg.tool_calls.push({
|
||||
id: item.call_id,
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: {
|
||||
name: item.name,
|
||||
arguments: item.arguments
|
||||
arguments: typeof toolInput === "string" ? toolInput : JSON.stringify(toolInput ?? {})
|
||||
}
|
||||
});
|
||||
}
|
||||
else if (itemType === RESPONSES_ITEM.FUNCTION_CALL_OUTPUT) {
|
||||
else if (itemType === RESPONSES_ITEM.FUNCTION_CALL_OUTPUT || itemType === RESPONSES_ITEM.CUSTOM_TOOL_CALL_OUTPUT) {
|
||||
// Flush assistant message first if exists
|
||||
if (currentAssistantMsg) {
|
||||
result.messages.push(currentAssistantMsg);
|
||||
@@ -137,6 +143,9 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
content: typeof item.output === "string" ? item.output : JSON.stringify(item.output)
|
||||
});
|
||||
}
|
||||
else if (itemType === RESPONSES_ITEM.ADDITIONAL_TOOLS) {
|
||||
if (Array.isArray(item.tools)) additionalTools.push(...item.tools);
|
||||
}
|
||||
else if (itemType === RESPONSES_ITEM.REASONING) {
|
||||
// Buffer reasoning text; attached to next assistant message/function_call.
|
||||
// Also stash encrypted_content so a later openai→responses hop can restore
|
||||
@@ -166,15 +175,45 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
// explicit `name` field and cannot be represented as Chat Completions function declarations.
|
||||
// Filter them out to avoid sending nameless functionDeclarations to downstream providers
|
||||
// such as Gemini, which strictly validates function names.
|
||||
if (body.tools && Array.isArray(body.tools)) {
|
||||
result.tools = body.tools
|
||||
const responseTools = [
|
||||
...(Array.isArray(body.tools) ? body.tools : []),
|
||||
...additionalTools,
|
||||
];
|
||||
if (responseTools.length > 0) {
|
||||
result.tools = responseTools
|
||||
.map(tool => {
|
||||
// Already in Chat Completions format: { type: "function", function: { name, ... } }
|
||||
if (tool.function) return tool;
|
||||
// Responses API function tool: { type: "function", name, description, parameters }
|
||||
// Only convert when a non-empty name is present; skip hosted tools without one.
|
||||
// Responses API function/custom tool: { type, name, description, parameters|format }.
|
||||
// Chat Completions has no freeform custom-tool declaration, so expose custom
|
||||
// tools as functions with one raw `input` string while retaining their names
|
||||
// in translator-only metadata for the response conversion.
|
||||
const name = tool.name;
|
||||
if (!name || typeof name !== "string" || name.trim() === "") return null;
|
||||
if (tool.type === "custom") {
|
||||
customToolNames.add(name);
|
||||
const formatHint = [tool.format?.syntax, tool.format?.definition].filter(Boolean).join("\n");
|
||||
return {
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: {
|
||||
name,
|
||||
description: [String(tool.description || ""), formatHint].filter(Boolean).join("\n\n"),
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: {
|
||||
input: {
|
||||
type: "string",
|
||||
description: "Raw freeform input for this custom tool"
|
||||
}
|
||||
},
|
||||
required: ["input"],
|
||||
additionalProperties: false
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
// Responses API function tool: { type: "function", name, description, parameters }
|
||||
// Only convert when a non-empty name is present; skip hosted tools without one.
|
||||
return {
|
||||
type: OPENAI_BLOCK.FUNCTION,
|
||||
function: {
|
||||
@@ -187,6 +226,7 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
})
|
||||
.filter(Boolean);
|
||||
}
|
||||
if (customToolNames.size > 0) result._customToolNames = [...customToolNames];
|
||||
|
||||
// Cleanup Responses API specific fields
|
||||
// Map Responses-only max_output_tokens to Chat max_tokens (avoid leaking unknown field upstream)
|
||||
|
||||
@@ -258,24 +258,43 @@ function closeMessage(state, emit, idx) {
|
||||
}
|
||||
}
|
||||
|
||||
function isCustomTool(state, name) {
|
||||
return !!name && state.customToolNames?.has(name);
|
||||
}
|
||||
|
||||
function extractCustomToolInput(argumentsText) {
|
||||
if (typeof argumentsText !== "string") return "";
|
||||
try {
|
||||
const parsed = JSON.parse(argumentsText);
|
||||
if (parsed && typeof parsed === "object" && typeof parsed.input === "string") return parsed.input;
|
||||
} catch { /* incomplete or raw freeform input */ }
|
||||
return argumentsText;
|
||||
}
|
||||
|
||||
function emitToolCall(state, emit, tc) {
|
||||
const tcIdx = tc.index ?? 0;
|
||||
const newCallId = tc.id;
|
||||
const funcName = tc.function?.name;
|
||||
|
||||
if (funcName) state.funcNames[tcIdx] = funcName;
|
||||
if (newCallId) state.funcCallIds[tcIdx] = newCallId;
|
||||
|
||||
// Some compatible providers split the call id and function name across
|
||||
// chunks. Wait for both before deciding whether this is a custom tool;
|
||||
// otherwise an `exec` call can be irreversibly announced as function_call.
|
||||
const callId = state.funcCallIds[tcIdx];
|
||||
if (!state.funcItemAdded[tcIdx] && callId && state.funcNames[tcIdx]) {
|
||||
state.funcItemAdded[tcIdx] = true;
|
||||
const custom = isCustomTool(state, state.funcNames[tcIdx]);
|
||||
|
||||
if (!state.funcCallIds[tcIdx] && newCallId) {
|
||||
state.funcCallIds[tcIdx] = newCallId;
|
||||
|
||||
emit("response.output_item.added", {
|
||||
type: "response.output_item.added",
|
||||
output_index: tcIdx,
|
||||
item: {
|
||||
id: `fc_${newCallId}`,
|
||||
type: RESPONSES_ITEM.FUNCTION_CALL,
|
||||
arguments: "",
|
||||
call_id: newCallId,
|
||||
id: `${custom ? "ctc" : "fc"}_${callId}`,
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
...(custom ? { input: "" } : { arguments: "" }),
|
||||
call_id: callId,
|
||||
name: state.funcNames[tcIdx] || ""
|
||||
}
|
||||
});
|
||||
@@ -285,7 +304,7 @@ function emitToolCall(state, emit, tc) {
|
||||
|
||||
if (tc.function?.arguments) {
|
||||
const refCallId = state.funcCallIds[tcIdx] || newCallId;
|
||||
if (refCallId) {
|
||||
if (state.funcItemAdded[tcIdx] && refCallId && !isCustomTool(state, state.funcNames[tcIdx])) {
|
||||
emit("response.function_call_arguments.delta", {
|
||||
type: "response.function_call_arguments.delta",
|
||||
item_id: `fc_${refCallId}`,
|
||||
@@ -293,6 +312,9 @@ function emitToolCall(state, emit, tc) {
|
||||
delta: tc.function.arguments
|
||||
});
|
||||
}
|
||||
// Custom input is emitted once at close, after the Chat JSON wrapper can be
|
||||
// parsed and unwrapped. Streaming the raw JSON fragments would expose
|
||||
// {"input":"..."} instead of the freeform program Codex expects.
|
||||
state.funcArgsBuf[tcIdx] += tc.function.arguments;
|
||||
}
|
||||
}
|
||||
@@ -301,21 +323,38 @@ function closeToolCall(state, emit, idx) {
|
||||
const callId = state.funcCallIds[idx];
|
||||
if (callId && !state.funcItemDone[idx]) {
|
||||
const args = state.funcArgsBuf[idx] || "{}";
|
||||
|
||||
emit("response.function_call_arguments.done", {
|
||||
type: "response.function_call_arguments.done",
|
||||
item_id: `fc_${callId}`,
|
||||
output_index: parseInt(idx),
|
||||
arguments: args
|
||||
});
|
||||
const custom = isCustomTool(state, state.funcNames[idx]);
|
||||
|
||||
if (custom) {
|
||||
const input = extractCustomToolInput(args);
|
||||
emit("response.custom_tool_call_input.delta", {
|
||||
type: "response.custom_tool_call_input.delta",
|
||||
item_id: `ctc_${callId}`,
|
||||
output_index: parseInt(idx),
|
||||
delta: input
|
||||
});
|
||||
emit("response.custom_tool_call_input.done", {
|
||||
type: "response.custom_tool_call_input.done",
|
||||
item_id: `ctc_${callId}`,
|
||||
output_index: parseInt(idx),
|
||||
input
|
||||
});
|
||||
} else {
|
||||
emit("response.function_call_arguments.done", {
|
||||
type: "response.function_call_arguments.done",
|
||||
item_id: `fc_${callId}`,
|
||||
output_index: parseInt(idx),
|
||||
arguments: args
|
||||
});
|
||||
}
|
||||
|
||||
emit("response.output_item.done", {
|
||||
type: "response.output_item.done",
|
||||
output_index: parseInt(idx),
|
||||
item: {
|
||||
id: `fc_${callId}`,
|
||||
type: RESPONSES_ITEM.FUNCTION_CALL,
|
||||
arguments: args,
|
||||
id: `${custom ? "ctc" : "fc"}_${callId}`,
|
||||
type: custom ? RESPONSES_ITEM.CUSTOM_TOOL_CALL : RESPONSES_ITEM.FUNCTION_CALL,
|
||||
...(custom ? { input: extractCustomToolInput(args) } : { arguments: args }),
|
||||
call_id: callId,
|
||||
name: state.funcNames[idx] || ""
|
||||
}
|
||||
|
||||
@@ -27,6 +27,9 @@ export const RESPONSES_ITEM = {
|
||||
MESSAGE: "message",
|
||||
FUNCTION_CALL: "function_call",
|
||||
FUNCTION_CALL_OUTPUT: "function_call_output",
|
||||
CUSTOM_TOOL_CALL: "custom_tool_call",
|
||||
CUSTOM_TOOL_CALL_OUTPUT: "custom_tool_call_output",
|
||||
ADDITIONAL_TOOLS: "additional_tools",
|
||||
REASONING: "reasoning",
|
||||
OUTPUT_TEXT: "output_text",
|
||||
INPUT_TEXT: "input_text",
|
||||
|
||||
@@ -1,70 +0,0 @@
|
||||
/**
|
||||
* Singleton cache for real Claude Code client headers.
|
||||
* Captures headers from authentic Claude Code requests and makes them available
|
||||
* for forwarding to api.anthropic.com, replacing static hardcoded values.
|
||||
*/
|
||||
|
||||
const CLAUDE_IDENTITY_HEADERS = [
|
||||
"user-agent",
|
||||
"anthropic-beta",
|
||||
"anthropic-version",
|
||||
"anthropic-dangerous-direct-browser-access",
|
||||
"x-app",
|
||||
"x-stainless-helper-method",
|
||||
"x-stainless-retry-count",
|
||||
"x-stainless-runtime-version",
|
||||
"x-stainless-package-version",
|
||||
"x-stainless-runtime",
|
||||
"x-stainless-lang",
|
||||
"x-stainless-arch",
|
||||
"x-stainless-os",
|
||||
"x-stainless-timeout",
|
||||
"x-claude-code-session-id",
|
||||
"package-version",
|
||||
"runtime-version",
|
||||
"os",
|
||||
"arch",
|
||||
];
|
||||
|
||||
let cachedHeaders = null;
|
||||
|
||||
/**
|
||||
* Detect if request headers look like a real Claude Code client.
|
||||
* @param {object} headers - Lowercase header key/value object
|
||||
*/
|
||||
function isClaudeCodeClient(headers) {
|
||||
const ua = (headers["user-agent"] || "").toLowerCase();
|
||||
const xApp = (headers["x-app"] || "").toLowerCase();
|
||||
return ua.includes("claude-cli") || ua.includes("claude-code") || xApp === "cli";
|
||||
}
|
||||
|
||||
/**
|
||||
* Store Claude Code identity headers if this looks like a real client request.
|
||||
* Called at the entry point before any translation/forwarding.
|
||||
* @param {object} headers - Lowercase header key/value object (from request.headers.entries())
|
||||
*/
|
||||
export function cacheClaudeHeaders(headers) {
|
||||
if (!headers || typeof headers !== "object") return;
|
||||
if (!isClaudeCodeClient(headers)) return;
|
||||
|
||||
const captured = {};
|
||||
for (const key of CLAUDE_IDENTITY_HEADERS) {
|
||||
if (headers[key] !== undefined && headers[key] !== null) {
|
||||
captured[key] = headers[key];
|
||||
}
|
||||
}
|
||||
|
||||
if (Object.keys(captured).length > 0) {
|
||||
cachedHeaders = captured;
|
||||
console.log(`[ClaudeHeaders] Cached ${Object.keys(captured).length} identity headers from Claude Code client`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the most recently cached Claude Code identity headers.
|
||||
* Returns null if no authentic client request has been seen yet (cold start).
|
||||
* @returns {object|null}
|
||||
*/
|
||||
export function getCachedClaudeHeaders() {
|
||||
return cachedHeaders;
|
||||
}
|
||||
@@ -22,6 +22,7 @@ export function detectClientTool(headers = {}, body = {}) {
|
||||
const xApp = (headers["x-app"] || "").toLowerCase();
|
||||
const openaiIntent = (headers["openai-intent"] || "").toLowerCase();
|
||||
const initiator = (headers["x-initiator"] || headers["X-Initiator"] || "").toLowerCase();
|
||||
const originator = (headers["originator"] || "").toLowerCase();
|
||||
|
||||
// Antigravity: detected via body field (not header)
|
||||
if (body.userAgent === "antigravity") return "antigravity";
|
||||
@@ -37,8 +38,10 @@ export function detectClientTool(headers = {}, body = {}) {
|
||||
// Gemini CLI
|
||||
if (ua.includes("gemini-cli")) return "gemini-cli";
|
||||
|
||||
// Codex CLI
|
||||
if (ua.includes("codex-cli")) return "codex";
|
||||
// Codex CLI/Desktop — codex-tui is the current Rust CLI, codex-cli/codex_cli_rs legacy;
|
||||
// Codex Desktop identifies via UA "Codex Desktop" or originator "codex_work_desktop"
|
||||
if (ua.includes("codex-tui") || ua.includes("codex-cli") || ua.includes("codex_cli_rs") ||
|
||||
ua.includes("codex desktop") || originator.startsWith("codex_")) return "codex";
|
||||
|
||||
// DeepSeek TUI
|
||||
if (ua.includes("deepseek-tui")) return "deepseek-tui";
|
||||
|
||||
@@ -44,6 +44,7 @@ export function createSSEStream(options = {}) {
|
||||
provider = null,
|
||||
reqLogger = null,
|
||||
toolNameMap = null,
|
||||
customToolNames = null,
|
||||
model = null,
|
||||
connectionId = null,
|
||||
body = null,
|
||||
@@ -57,7 +58,9 @@ export function createSSEStream(options = {}) {
|
||||
// Per-stream decoder with stream:true to correctly handle multi-byte chars split across chunks
|
||||
const decoder = new TextDecoder("utf-8", { fatal: false });
|
||||
|
||||
const state = mode === STREAM_MODE.TRANSLATE ? { ...initState(sourceFormat), provider, toolNameMap, model } : null;
|
||||
const state = mode === STREAM_MODE.TRANSLATE
|
||||
? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model }
|
||||
: null;
|
||||
|
||||
let totalContentLength = 0;
|
||||
let accumulatedContent = "";
|
||||
@@ -464,7 +467,7 @@ export function createSSEStream(options = {}) {
|
||||
});
|
||||
}
|
||||
|
||||
export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null) {
|
||||
export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null) {
|
||||
return createSSEStream({
|
||||
mode: STREAM_MODE.TRANSLATE,
|
||||
targetFormat,
|
||||
@@ -472,6 +475,7 @@ export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, p
|
||||
provider,
|
||||
reqLogger,
|
||||
toolNameMap,
|
||||
customToolNames,
|
||||
model,
|
||||
connectionId,
|
||||
body,
|
||||
|
||||
Reference in New Issue
Block a user