Merge origin/master (v0.5.75) into gitea/new_feature

Resolve conflicts:
- package.json / cli/package.json: take 0.5.75
- .gitignore: union both sides (upstream 9router-*/temp files + local state dirs)
- CHANGELOG.md: keep both blocks, v0.5.75 above v0.5.70
- nonStreamingHandler.js: merge imports (unwrapClineEnvelope +
  tokensForDetail/shouldPersistRequestDetail); drop dead appendRequestLog
- providers/[id]/page.js: union useState blocks (compatible-model states
  + importingClineModels)

Co-authored-by: CommandCodeBot <noreply@commandcode.ai>
This commit is contained in:
2026-09-17 14:14:43 +07:00
113 changed files with 6634 additions and 482 deletions

View File

@@ -7,6 +7,9 @@ import { createRequire } from "module";
export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion;
export const GEMINI_CLI_API_CLIENT = PROVIDERS["gemini-cli"]?.apiClient;
// === Codex CLI === derive từ registry codex.transport
export const CODEX_CLI_VERSION = PROVIDERS["codex"]?.cliVersion;
// Map Node arch to Gemini CLI arch string (x64/x86/arm64/...)
function geminiCLIArch() {
const a = arch();

View File

@@ -5,7 +5,7 @@ import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX,
import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
import { cleanJSONSchemaForAntigravity, normalizeGeminiContents } from "../translator/formats/gemini.js";
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
@@ -193,7 +193,7 @@ export class AntigravityExecutor extends BaseExecutor {
// ─── Standard (non-image) request ───
// Fix contents for Claude models via Antigravity
const contents = body.request?.contents?.map(c => {
const rawContents = (body.request?.contents || []).map(c => {
let role = c.role;
// functionResponse must be role "user" for Claude models
if (c.parts?.some(p => p.functionResponse)) {
@@ -226,15 +226,13 @@ export class AntigravityExecutor extends BaseExecutor {
return p;
});
const partsChanged = parts?.length !== c.parts?.length || modifiedParts?.some((p, idx) => p !== c.parts[idx]);
if (role !== c.role || partsChanged) {
return {
...c, role,
parts: modifiedParts || parts,
};
}
return c;
return {
...c,
role,
parts: modifiedParts || parts || [],
};
});
const contents = normalizeGeminiContents(rawContents);
// Sanitize tool schemas and function names before sending to Antigravity.
let tools = body.request?.tools;

View File

@@ -12,6 +12,7 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
import { dbg } from "../utils/debugLog.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { stripCodexUnsupportedPatterns } from "../utils/codexToolSchema.js";
// SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts.
const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
@@ -72,6 +73,9 @@ function stripStoredItemReferences(body) {
function normalizeCodexTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
// Codex's schema validator has no Unicode property escapes; a `pattern`
// carrying `\p{...}` 400s the whole request on every account (#3922).
const patternStats = { removed: 0 };
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const type = typeof tool.type === "string" ? tool.type : "";
@@ -80,6 +84,9 @@ function normalizeCodexTools(body) {
for (const st of tool.tools) {
const n = typeof st?.name === "string" ? st.name.trim().slice(0, 128) : "";
if (n) validNames.add(n);
if (st?.parameters && typeof st.parameters === "object") {
st.parameters = stripCodexUnsupportedPatterns(st.parameters, patternStats);
}
}
}
return true;
@@ -101,10 +108,13 @@ function normalizeCodexTools(body) {
tool.type = "function";
tool.name = name.slice(0, 128);
if (description) tool.description = description;
tool.parameters = parameters;
tool.parameters = stripCodexUnsupportedPatterns(parameters, patternStats);
validNames.add(name);
return true;
});
if (patternStats.removed > 0) {
dbg("CODEX", `stripped ${patternStats.removed} unsupported tool schema pattern(s)`);
}
// Drop tool_choice if it references an unknown function name
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {

View File

@@ -17,6 +17,7 @@ import { PerplexityWebExecutor } from "./perplexity-web.js";
import { OllamaLocalExecutor } from "./ollama-local.js";
import { CommandCodeExecutor } from "./commandcode.js";
import { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
import { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
import { MimoFreeExecutor } from "./mimo-free.js";
import { CodeBuddyExecutor } from "./codebuddy-cn.js";
import { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
@@ -50,6 +51,7 @@ const executors = {
"ollama-local": new OllamaLocalExecutor(),
commandcode: new CommandCodeExecutor(),
"xiaomi-tokenplan": new XiaomiTokenplanExecutor(),
"xiaomi-mimo": new XiaomiMimoExecutor(),
"mimo-free": new MimoFreeExecutor(),
mmf: new MimoFreeExecutor(), // Alias for mimo-free
"codebuddy-cn": new CodeBuddyExecutor(),
@@ -93,6 +95,7 @@ export { PerplexityWebExecutor } from "./perplexity-web.js";
export { OllamaLocalExecutor } from "./ollama-local.js";
export { CommandCodeExecutor } from "./commandcode.js";
export { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
export { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
export { MimoFreeExecutor } from "./mimo-free.js";
export { CodeBuddyExecutor } from "./codebuddy-cn.js";
export { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";

View File

@@ -127,12 +127,18 @@ async function readResponsePrefix(response, signal, maxBytes, timeoutMs) {
return decoder.decode(concatChunks(chunks, totalBytes));
}
// The instruction goes into the current user turn, never into a top-level
// `systemPrompt`: kiro.dev answers any body carrying that field with
// 400 REQUEST_BODY_INVALID, so writing it here turned every repair retry into
// a hard failure.
function appendRepairInstruction(body, kind) {
const repaired = structuredClone(body || {});
const instruction = REPAIR_INSTRUCTIONS[kind] || "Retry the previous incomplete Kiro response.";
repaired.systemPrompt = repaired.systemPrompt
? `${repaired.systemPrompt}\n\n${instruction}`
: instruction;
const msg = repaired?.conversationState?.currentMessage?.userInputMessage;
if (msg) {
const content = typeof msg.content === "string" ? msg.content : "";
msg.content = content ? `${content}\n\n${instruction}` : instruction;
}
return repaired;
}
@@ -259,6 +265,19 @@ export class KiroExecutor extends BaseExecutor {
}
}
// CLIRO parity for the Amazon surfaces: the Kiro runtime accepts the
// SSO bearer header + agent-mode marker. Without these the deprecated
// path gateway answers REQUEST_BODY_INVALID for modern payloads.
if (credentials?.accessToken) {
headers["x-amz-sso-bearer"] = credentials.accessToken;
}
headers["x-amzn-kiro-agent-mode"] = "spec";
headers["x-amzn-codewhisperer-machine-id"] = "kiro-desktop";
const profileArn = credentials?.providerSpecificData?.profileArn;
if (profileArn) {
headers["x-amzn-codewhisperer-profile-arn"] = profileArn;
}
return headers;
}
@@ -285,9 +304,13 @@ export class KiroExecutor extends BaseExecutor {
// 403 "bearer token invalid", so they must hit the CodeWhisperer
// *.amazonaws.com surface, and in the region the token was minted in
// (the baseUrls are hardcoded us-east-1).
const isCodeWhispererSurface =
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
if (!isCodeWhispererSurface) return baseUrls;
// Kiro deprecated the legacy path-style GenerateAssistantResponse on
// runtime.*.kiro.dev (IDE 1.0.228+ moved to POST / + x-amz-target). The
// path gateway now answers valid modern payloads with 400
// REQUEST_BODY_INVALID, and 400 is terminal in BaseExecutor, so kiro.dev
// must never be the first surface for any auth method. Amazon surfaces
// reject foreign tokens with 401/403, which DO fall through, so trying
// q/codewhisperer first is safe for every auth method (CLIRO parity).
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
const regionalize = (u) =>
@@ -297,20 +320,17 @@ export class KiroExecutor extends BaseExecutor {
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
if (authMethod === "api_key") {
const q = amazon.filter((u) => u.includes("://q."));
const remaining = amazon.filter((u) => !u.includes("://q."));
return q.length > 0
? [...q, ...remaining, ...others]
: [...amazon, ...others];
}
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
const q = amazon.filter((u) => u.includes("://q."));
const remaining = amazon.filter((u) => !u.includes("://q."));
return q.length > 0
? [...q, ...remaining, ...others]
: [...amazon, ...others];
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
const baseUrls = this.getOrderedBaseUrls(credentials);
return baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
const url = baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
return url;
}
// Retry only endpoint/auth-surface failures. Payload-invalid HTTP 400 must be

View File

@@ -32,14 +32,16 @@ import { SSE_DONE } from "../utils/sseConstants.js";
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
import { resolveProviderTimeoutMs } from "../services/providerTimeout.js";
import {
QODER_CHAT_URL_ENCODED,
QODER_CHAT_BASE_ALT,
QODER_CHAT_SIG_PATH,
QODER_MODEL_MAP,
QODER_CONTEXT_TIER_ENV,
qoderInferenceBase,
} from "../shared/qoder/constants.js";
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
import { encodeDataUri } from "../translator/concerns/image.js";
import { createQoderSseCoalescer } from "../shared/qoder/sse.js";
import { rewriteQoderMessageAttachments } from "../shared/qoder/attachments.js";
import { resolveQoderContextTier, applyQoderContextTier } from "../shared/qoder/contextTier.js";
/**
* Hoist role:"system" messages out of the messages array (Qoder rejects
@@ -71,15 +73,16 @@ function normalizeMessages(messages) {
*
* Text-only content is flattened to a plain string (Qoder's historical
* shape). When images are present the content stays an array and image
* blocks are kept as OpenAI-style `image_url` parts — verified against the
* upstream: it accepts both http(s) URLs and inline base64 data: URIs
* directly, no pre-upload to the /image/upload OSS flow required (that is
* a qodercli client-side choice, not a protocol requirement). The legacy
* blocks are kept as OpenAI-style `image_url` parts. Native qodercli
* uploads inlined bytes to `/api/v2/image/upload` first and then sends
* the OSS URL — `buildQoderRequestBody` does that rewrite before this
* runs. Tiny leftover data URIs are still accepted. The legacy
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
* qodercli leaves them null too.
*
* Claude-style `{type:"image", source:{...}}` blocks are converted to
* `image_url` so claude-format clients also round-trip.
* `image_url`. File/document blocks that survived rewrite become short
* stubs so 30MB PDFs never land in agent_chat_generation.
*/
function normalizeContent(content) {
if (typeof content === "string") return content;
@@ -89,10 +92,24 @@ function normalizeContent(content) {
const blocks = [];
const textParts = [];
let hasImage = false;
const pushText = (text) => {
if (!text) return;
if (hasImage || blocks.length) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
else textParts.push(text);
};
const imageUrlOf = (item) => {
if (typeof item.image_url === "string" && item.image_url) return item.image_url;
if (typeof item.image_url?.url === "string" && item.image_url.url) return item.image_url.url;
return null;
};
for (const item of content) {
if (!item || typeof item !== "object") continue;
if (item.type === OPENAI_BLOCK.IMAGE_URL && typeof item.image_url?.url === "string" && item.image_url.url) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: item.image_url.url } });
const imageUrl = item.type === OPENAI_BLOCK.IMAGE_URL ? imageUrlOf(item) : null;
if (imageUrl) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: imageUrl } });
hasImage = true;
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
// Claude base64/url image → OpenAI image_url equivalent.
@@ -104,13 +121,14 @@ function normalizeContent(content) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
hasImage = true;
}
} else if (item.type === OPENAI_BLOCK.FILE) {
const name = item.file?.filename || item.file?.name || "file";
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
} else if (item.type === CLAUDE_BLOCK.DOCUMENT) {
const name = item.title || "document";
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
} else if (typeof item.text === "string" && item.text) {
if (hasImage || blocks.length) {
// Keep ordering faithful once images are in play.
blocks.push({ type: OPENAI_BLOCK.TEXT, text: item.text });
} else {
textParts.push(item.text);
}
pushText(item.text);
}
}
@@ -190,7 +208,7 @@ function truncate(s, n) {
/**
* Map the OpenAI-style request body into the exact shape Qoder expects.
*/
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }) {
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null }) {
const qoderKey = String(model || "").replace(/^qoder\//, "");
// Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP.
@@ -209,7 +227,30 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
modelConfig = { ...retried, key: qoderKey };
}
const { messages, systemText } = normalizeMessages(body.messages || []);
const incoming = Array.isArray(body.messages)
? body.messages.map((m) => {
if (!m || typeof m !== "object") return m;
return {
...m,
content: Array.isArray(m.content)
? m.content.map((b) => (b && typeof b === "object" ? { ...b } : b))
: m.content,
};
})
: [];
try {
await rewriteQoderMessageAttachments(incoming, {
credentials,
log,
proxyOptions,
signal,
uploadFn,
});
} catch (err) {
log?.warn?.("QODER", `attachment rewrite failed: ${err.message}`);
}
const { messages, systemText } = normalizeMessages(incoming);
const tools = body.tools;
const isReasoning = !!modelConfig.is_reasoning;
const maxOutputTokens = Number(modelConfig.max_output_tokens) || 0;
@@ -228,7 +269,21 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
const sessionId = stableHash("qoder-session", psd.userId, qoderKey);
const recordId = stableChatRecordId(qoderKey, messages, tools, maxTokens);
return {
// Context-window tier (200K/400K/1M): the IDE picks one from model_config.context_config;
// qodercli-style requests default to the smallest. Escalate when the prompt no longer fits.
const tierChoice = resolveQoderContextTier(
modelConfig,
{ system: systemText, messages, tools },
{ preference: process.env[QODER_CONTEXT_TIER_ENV] },
);
if (tierChoice) {
log?.info?.(
"QODER",
`context tier ${tierChoice.tier.name} (${tierChoice.tier.tokenCount} tokens, ${tierChoice.reason}) for ~${tierChoice.estimatedTokens} prompt tokens`,
);
}
const built = {
qoderKey,
payload: {
request_id: uuidv4(),
@@ -276,6 +331,8 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
},
modelConfig,
};
if (tierChoice) applyQoderContextTier(built.payload, tierChoice.tier);
return built;
}
/**
@@ -339,6 +396,11 @@ async function peekFirstQoderFrame(reader, decoder) {
* response.text() which hangs until the socket closes — so on terminal
* events we cancel the upstream reader and close our stream immediately.
*
* Usage: Qoder puts finish_reason on `delta` and sends token counts on a
* later `choices: []` frame. Downstream OpenAI/Claude clients only read
* usage from the finish chunk, so we coalesce those two frames (see
* createQoderSseCoalescer) before forwarding.
*
* NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl).
* If detected, return 403 response so chatCore marks connection unavailable
* and triggers combo fallback instead of leaking error text into chat.
@@ -365,6 +427,11 @@ async function wrapQoderSSE(response, model) {
const upstreamDrained = peek.upstreamDone === true;
const encoder = new TextEncoder();
let doneEmitted = false;
const coalescer = createQoderSseCoalescer({ model, encoder, sseDone: SSE_DONE });
const syncDone = () => {
if (coalescer.doneEmitted) doneEmitted = true;
};
// Process one already-extracted SSE line (no trailing newline).
const processLine = (line, controller) => {
@@ -375,15 +442,17 @@ async function wrapQoderSSE(response, model) {
const data = trimmed.slice(5).trimStart();
if (data === "[DONE]") {
controller.enqueue(encoder.encode(SSE_DONE));
doneEmitted = true;
coalescer.flush(controller);
syncDone();
return;
}
let envelope;
try { envelope = JSON.parse(data); } catch { return; }
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
const inner = typeof envelope.body === "string" ? envelope.body : "";
const inner = typeof envelope.body === "string"
? envelope.body
: envelope.body != null ? JSON.stringify(envelope.body) : "";
if (statusVal !== 200) {
const msg = inner || `upstream status ${statusVal}`;
const errChunk = JSON.stringify({
@@ -399,14 +468,8 @@ async function wrapQoderSSE(response, model) {
return;
}
if (!inner) return;
if (inner === "[DONE]") {
controller.enqueue(encoder.encode(SSE_DONE));
doneEmitted = true;
return;
}
// Strip embedded newlines so the SSE frame stays a single event.
const sanitized = inner.replace(/\r?\n/g, "");
controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`));
coalescer.handleInner(inner, controller);
syncDone();
};
const stream = new ReadableStream({
@@ -465,7 +528,7 @@ async function wrapQoderSSE(response, model) {
} finally {
if (!doneEmitted) {
try {
controller.enqueue(encoder.encode(SSE_DONE));
coalescer.flush(controller);
doneEmitted = true;
} catch { /* already closed */ }
}
@@ -494,13 +557,7 @@ export class QoderExecutor extends BaseExecutor {
}
buildUrl(credentials) {
// Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt-
// with "Login expired" (403). Device tokens (dt-...) stay on api3.
const raw = credentials?.apiKey || credentials?.accessToken;
if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) {
return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
}
return QODER_CHAT_URL_ENCODED;
return `${qoderInferenceBase(credentials)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
}
// Override execute entirely — Qoder needs:

View File

@@ -0,0 +1,99 @@
import { DefaultExecutor } from "./default.js";
import { getMimoAccountCookie, invalidateMimoAccountCookieCache, MIMO_API_BASE, MIMO_API_UA } from "../shared/mimoAccount.js";
// Desktop-exclusive Preview models. These are served by the account service's
// /api/route proxy, authorized by the Xiaomi account session (NOT the sk- key).
// See shared/mimoAccount.js for the session handshake.
const PREVIEW_MODELS = new Set(["mimo-x-pro-preview", "mimo-x-flash-preview"]);
// Session cookie resolved in execute() (async) and read back by buildHeaders()
// (sync — BaseExecutor.execute does not await it). Carried on the per-request
// credentials object, same as runtimeTransport.
const COOKIE_KEY = "__mimoAccountCookie";
// Upstream calls may hand us either the bare id or a `provider/model` ref.
function bareModel(model) {
const s = String(model || "");
const i = s.indexOf("/");
return i >= 0 ? s.slice(i + 1) : s;
}
export class XiaomiMimoExecutor extends DefaultExecutor {
constructor() {
super("xiaomi-mimo");
}
static isPreviewModel(model) {
return PREVIEW_MODELS.has(bareModel(model));
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Preview models live on the account-service route, which is not one of the
// declared transports — resolve it before the default runtimeTransport path.
if (XiaomiMimoExecutor.isPreviewModel(model)) {
return `${MIMO_API_BASE}/api/route/chat/completions`;
}
// Cloud API models keep default handling, so a Claude-format client reaches
// the /anthropic/v1/messages transport.
return super.buildUrl(model, stream, urlIndex, credentials);
}
buildHeaders(credentials, stream = true, url, model) {
if (XiaomiMimoExecutor.isPreviewModel(model) && credentials?.[COOKIE_KEY]) {
// Preview models authenticate with the account-session cookie, not the key.
return {
"Content-Type": "application/json",
Accept: stream ? "text/event-stream" : "application/json",
"User-Agent": MIMO_API_UA,
Cookie: credentials[COOKIE_KEY],
};
}
return super.buildHeaders(credentials, stream, url, model);
}
transformRequest(model, body, stream, credentials) {
// super runs stripUnsupportedParams, which flattens Preview content-part
// arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js).
const out = super.transformRequest(model, body, stream, credentials);
// Preview models: thinking/params get defaults only — never override what the
// caller set explicitly. (body.model is already `xiaomi/<id>` via upstreamModelId.)
if (XiaomiMimoExecutor.isPreviewModel(model)) {
if (out.thinking == null) out.thinking = { type: "enabled" };
if (out.temperature == null) out.temperature = 1.0;
if (out.top_p == null) out.top_p = 0.95;
if (!out.max_tokens) out.max_tokens = 4096;
}
return out;
}
async execute(args) {
const { model, credentials, proxyOptions = null } = args;
if (!XiaomiMimoExecutor.isPreviewModel(model)) return super.execute(args);
const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions);
if (!cookie) {
throw new Error(
"Xiaomi MiMo account session unavailable. Sign in to MiMo Desktop once so its passToken is present, then retry.",
);
}
credentials[COOKIE_KEY] = cookie;
const result = await super.execute(args);
// A cached session can expire early — drop it and retry once with a fresh one.
if (result.response.status === 401) {
invalidateMimoAccountCookieCache();
const fresh = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions).catch(() => null);
if (fresh) {
credentials[COOKIE_KEY] = fresh;
return super.execute(args);
}
}
return result;
}
}
export const __test__ = { PREVIEW_MODELS, bareModel, COOKIE_KEY };
export default XiaomiMimoExecutor;

View File

@@ -28,7 +28,7 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
import { getCapabilitiesForModel } from "../providers/capabilities.js";
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
import { defaultClaudeToolType, shouldDefaultClaudeToolType } from "../translator/concerns/toolCall.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { maybeRejectEarlyStreamError } from "../utils/streamErrorPeek.js";
@@ -246,7 +246,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax)
// reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing.
if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) {
// Provider-scoped via quirks (shouldDefaultClaudeToolType): only gateways that declare
// requireClaudeToolType get the explicit type. Applying it unconditionally breaks
// Claude-format endpoints that only accept the legacy typeless tool shape — DeepSeek's
// Anthropic-compatible endpoint 400s with "unknown variant `custom`" (#3905).
if (shouldDefaultClaudeToolType(provider, finalFormat, translatedBody.tools, PROVIDERS)) {
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
}

View File

@@ -6,6 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin
import { createErrorResult } from "../../utils/error.js";
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js";
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine, tokensForDetail, shouldPersistRequestDetail } from "./requestDetail.js";
import { saveRequestDetail } from "@/lib/usageDb.js";
import { matchStreamErrorPatterns } from "../../utils/streamErrorPatterns.js";
@@ -305,6 +306,11 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
}
}
// Unwrap before any consumer reads choices/usage so non-stream clients get a
// bare OpenAI body and usage tracking sees data.usage. No-op unless the
// provider opts in via transport.quirks.clineEnvelope.
responseBody = unwrapClineEnvelope(responseBody, provider);
reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody);
if (onRequestSuccess) {
Promise.resolve()

View File

@@ -2,13 +2,21 @@
import { randomUUID } from "node:crypto";
import { nowSec } from "./_base.js";
import { PROVIDERS } from "../../config/providers.js";
import { CODEX_CLI_VERSION } from "../../config/appConstants.js";
const CODEX_RESPONSES_URL = PROVIDERS["codex"].baseUrl;
const CODEX_USER_AGENT = "codex_cli_rs/0.136.0";
const CODEX_VERSION = "0.136.0";
const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`;
const CODEX_ORIGINATOR = "codex_cli_rs";
const CODEX_MODEL_SUFFIX = "-image";
const CODEX_REF_DETAIL = "high";
const CODEX_IMAGES_MAIN_MODEL = "gpt-5.5";
const CODEX_TOOL_IMAGE_MODELS = new Set([
"gpt-image-1.5",
"gpt-image-2",
"gpt-image-2.5",
"gpt-image-2.5-flare",
"gpt-image-2.5-sunburst",
]);
function decodeAccountId(idToken) {
try {
@@ -27,6 +35,13 @@ function stripImageSuffix(model) {
return model.endsWith(CODEX_MODEL_SUFFIX) ? model.slice(0, -CODEX_MODEL_SUFFIX.length) : model;
}
function resolveCodexImageModels(model) {
if (CODEX_TOOL_IMAGE_MODELS.has(model)) {
return { responsesModel: CODEX_IMAGES_MAIN_MODEL, toolModel: model };
}
return { responsesModel: stripImageSuffix(model), toolModel: null };
}
function toDataUrl(input) {
if (!input || typeof input !== "string") return null;
if (/^data:image\//i.test(input) || /^https?:\/\//i.test(input)) return input;
@@ -157,7 +172,7 @@ export default {
"originator": CODEX_ORIGINATOR,
"session_id": randomUUID(),
"user-agent": CODEX_USER_AGENT,
"version": CODEX_VERSION,
"version": CODEX_CLI_VERSION,
"x-client-request-id": randomUUID(),
};
},
@@ -167,21 +182,26 @@ export default {
const single = toDataUrl(body.image);
if (single) refs.push(single);
const detail = body.image_detail || CODEX_REF_DETAIL;
const { responsesModel, toolModel } = resolveCodexImageModels(model);
const imgTool = { type: "image_generation", output_format: (body.output_format || "png").toLowerCase() };
if (toolModel) {
imgTool.action = refs.length > 0 ? "edit" : "generate";
imgTool.model = toolModel;
}
if (body.size && body.size !== "") imgTool.size = body.size;
if (body.quality && body.quality !== "") imgTool.quality = body.quality;
if (body.background && body.background !== "") imgTool.background = body.background;
return {
model: stripImageSuffix(model),
model: responsesModel,
instructions: "",
input: [{ type: "message", role: "user", content: buildContent(body.prompt, refs, detail) }],
tools: [imgTool],
tool_choice: "auto",
tool_choice: toolModel ? { type: "image_generation" } : "auto",
parallel_tool_calls: false,
prompt_cache_key: randomUUID(),
stream: true,
store: false,
reasoning: null,
reasoning: toolModel ? { effort: "medium", summary: "auto" } : null,
};
},
// Custom: codex parses SSE → either pipe to client or collect b64

View File

@@ -2,6 +2,7 @@ import { createErrorResult } from "../utils/error.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { refreshTokenByProvider } from "../services/tokenRefresh.js";
import { PROVIDER_MEDIA } from "../providers/index.js";
import { getVideoAdapter } from "./videoProviders/index.js";
// Upstream fetch deadline for video job submission/polling (the job itself is
// async upstream — this only bounds the HTTP round-trip, not video rendering).
@@ -94,21 +95,49 @@ export async function handleVideoProxyCore({
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`);
}
const method = requestId ? "GET" : "POST";
const url = buildUpstreamUrl(config, action, requestId);
const adapter = getVideoAdapter(provider);
const fetchSignal = combineSignals(signal, timeoutMs);
const doFetch = (token) =>
fetch(url, {
// Default (xAI shape) request plan; adapters override URL/method/headers/body.
const defaultPlan = () => {
const method = requestId ? "GET" : "POST";
return {
method,
headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }),
url: buildUpstreamUrl(config, action, requestId),
headers: buildHeaders({
token: credentials?.accessToken || credentials?.apiKey,
contentType: method === "POST" ? contentType : null,
idempotencyKey: method === "POST" ? idempotencyKey : null,
}),
body: method === "POST" ? rawBody : undefined,
signal: fetchSignal,
});
};
};
// Rebuilt per attempt so the auth retry below picks up the refreshed token.
const doFetch = async () => {
const plan = adapter
? await adapter.buildRequest({
config, action, requestId, rawBody, contentType, idempotencyKey, credentials, log,
token: credentials?.accessToken || credentials?.apiKey,
})
: defaultPlan();
if (plan.error) return { planError: plan.error };
return {
response: await fetch(plan.url, {
method: plan.method,
headers: plan.headers,
body: plan.body,
signal: fetchSignal,
}),
};
};
const method = requestId ? "GET" : "POST";
let upstream;
try {
upstream = await doFetch(credentials?.accessToken || credentials?.apiKey);
const first = await doFetch();
if (first.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${first.planError}`);
upstream = first.response;
} catch (error) {
if (error?.name === "AbortError" || error?.name === "TimeoutError") {
return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`);
@@ -136,7 +165,9 @@ export async function handleVideoProxyCore({
await upstream.body?.cancel?.();
} catch { /* noop */ }
try {
upstream = await doFetch(credentials.accessToken || credentials.apiKey);
const retry = await doFetch();
if (retry.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${retry.planError}`);
upstream = retry.response;
} catch (error) {
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials));
}
@@ -152,13 +183,25 @@ export async function handleVideoProxyCore({
return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`);
}
// Success: pass the upstream JSON through untouched (request_id / status / video.url).
// Success: pass the upstream JSON through untouched (request_id / status / video.url),
// unless the adapter maps a provider-native shape onto it (Vertex operations).
let outBody = bodyText;
let outType = upstream.headers.get("content-type") || "application/json";
if (adapter?.transformResponse) {
try {
outBody = JSON.stringify(adapter.transformResponse(JSON.parse(bodyText)));
outType = "application/json";
} catch {
// Non-JSON or unexpected shape — fall back to the raw upstream body.
}
}
return {
success: true,
response: new Response(bodyText, {
response: new Response(outBody, {
status: upstream.status,
headers: {
"Content-Type": upstream.headers.get("content-type") || "application/json",
"Content-Type": outType,
"Access-Control-Allow-Origin": "*",
},
}),

View File

@@ -0,0 +1,13 @@
// Video provider adapters.
//
// Default (no adapter) = xAI shape: raw body forwarded to {baseUrl}/{action},
// polled at {baseUrl}/{id}, upstream JSON passed through verbatim.
// A provider only needs an adapter when its wire format differs from that.
import openrouter from "./openrouter.js";
import vertex from "./vertex.js";
const ADAPTERS = { openrouter, vertex };
export function getVideoAdapter(provider) {
return ADAPTERS[provider] || null;
}

View File

@@ -0,0 +1,39 @@
// OpenRouter video jobs — https://openrouter.ai/docs/api/api-reference/videos
//
// Same async shape as xAI (POST → { id, status }, GET → status/unsigned_urls),
// two differences only: creation POSTs to the collection root (no `/generations`
// suffix) and the account headers come from the registry entry.
// Response bodies are passed through verbatim.
// ponytail: generations only — OpenRouter has no edits/extensions endpoint today.
const SUPPORTED_ACTIONS = new Set(["generations"]);
function headers(config, token) {
return {
Accept: "application/json",
...(config.headers || {}),
...(token ? { Authorization: `Bearer ${token}` } : {}),
};
}
export default {
buildRequest({ config, action, requestId, rawBody, contentType, token }) {
const base = config.baseUrl.replace(/\/$/, "");
if (requestId) {
return { method: "GET", url: `${base}/${encodeURIComponent(requestId)}`, headers: headers(config, token) };
}
if (!SUPPORTED_ACTIONS.has(action)) {
return { error: `OpenRouter video supports 'generations' only (got '${action}')` };
}
if (contentType && !contentType.includes("application/json")) {
return { error: "OpenRouter video requires an application/json body" };
}
return {
method: "POST",
url: base,
headers: { ...headers(config, token), "Content-Type": "application/json" },
body: rawBody,
};
},
};

View File

@@ -0,0 +1,159 @@
// Vertex AI (Veo) video jobs.
//
// Vertex does NOT speak the OpenAI-ish /v1/videos shape, so unlike OpenRouter
// this adapter translates both directions:
// create → POST {model}:predictLongRunning { instances[], parameters{} } → { name }
// poll → POST {model}:fetchPredictOperation { operationName } → { done, response }
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
//
// The operation name is a resource path (contains "/"), so it is base64url-encoded
// into the job id returned to the client — GET /v1/videos/{id} stays a flat path.
import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefresh.js";
const DEFAULT_LOCATION = "us-central1";
const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url");
// Operation name shape: projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}.
// Anchored and single-segment-per-field so a decoded path can never carry `..` or a
// host-changing prefix into the request URL.
const OPERATION_NAME_RE = /^projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/models\/[^/]+\/operations\/[^/]+$/;
function modelPathOf(operationName) {
return operationName.slice(0, operationName.indexOf("/operations/"));
}
function decodeJobId(id) {
const raw = String(id ?? "");
// Buffer.from(x, "base64url") silently drops invalid characters instead of
// throwing, so only ids that re-encode byte-for-byte are accepted.
if (!raw || raw.length > 1024 || !/^[A-Za-z0-9_-]+$/.test(raw)) return null;
const decoded = Buffer.from(raw, "base64url").toString("utf8");
if (Buffer.from(decoded, "utf8").toString("base64url") !== raw) return null;
return OPERATION_NAME_RE.test(decoded) ? decoded : null;
}
async function resolveAuth(credentials, log) {
const saJson = parseVertexSaJson(credentials?.apiKey);
const projectId =
saJson?.project_id ||
credentials?.projectId ||
credentials?.providerSpecificData?.projectId;
const location = credentials?.providerSpecificData?.location || DEFAULT_LOCATION;
if (!projectId) {
return { error: "Vertex video requires a project_id — use Service Account JSON or set providerSpecificData.projectId" };
}
let token = credentials?.accessToken;
if (saJson) {
const minted = await refreshVertexToken(saJson, log);
if (!minted?.accessToken) return { error: "Vertex video: failed to mint access token from service account JSON" };
token = minted.accessToken;
}
if (!token) return { error: "Vertex video requires Service Account JSON or an OAuth access token (raw API keys are not supported)" };
return { token, projectId, location };
}
/** OpenAI-ish video body → Vertex predictLongRunning body. */
function toVertexBody(body) {
const instance = { prompt: body.prompt };
// Image-to-video: accept the Vertex-native shape or a bare data URL / base64 string.
const image = body.image ?? body.image_url;
if (image && typeof image === "object") {
instance.image = image;
} else if (typeof image === "string") {
const match = image.match(/^data:([^;]+);base64,(.*)$/s);
instance.image = match
? { bytesBase64Encoded: match[2], mimeType: match[1] }
: { gcsUri: image };
}
if (body.video && typeof body.video === "object") instance.video = body.video;
const parameters = {};
if (body.n != null) parameters.sampleCount = Number(body.n);
if (body.duration != null) parameters.durationSeconds = Number(body.duration);
if (body.aspect_ratio) parameters.aspectRatio = body.aspect_ratio;
if (body.resolution) parameters.resolution = body.resolution;
if (body.seed != null) parameters.seed = body.seed;
if (body.negative_prompt) parameters.negativePrompt = body.negative_prompt;
// Without storageUri Vertex returns inline base64 bytes; a GCS bucket keeps
// the poll response small and is what production callers want.
if (body.storage_uri) parameters.storageUri = body.storage_uri;
if (body.generate_audio != null) parameters.generateAudio = !!body.generate_audio;
return { instances: [instance], ...(Object.keys(parameters).length ? { parameters } : {}) };
}
/** Vertex operation → the async-job shape 9Router clients already poll for. */
function fromVertexOperation(json) {
if (!json?.name) return json;
const id = encodeJobId(json.name);
if (json.error) {
return { id, request_id: id, status: "failed", error: json.error };
}
if (!json.done) {
return { id, request_id: id, status: "pending" };
}
const samples =
json.response?.videos ||
json.response?.generateVideoResponse?.generatedSamples ||
[];
const videos = samples.map((s) => ({
url: s.gcsUri || s.video?.uri || s.uri || null,
b64_json: s.bytesBase64Encoded || s.video?.bytesBase64Encoded || null,
mime_type: s.mimeType || s.video?.mimeType || "video/mp4",
}));
return { id, request_id: id, status: "completed", video: videos[0] || null, videos };
}
export default {
async buildRequest({ config, action, requestId, rawBody, contentType, credentials, log }) {
if (contentType && !contentType.includes("application/json")) {
return { error: "Vertex video requires an application/json body" };
}
const auth = await resolveAuth(credentials, log);
if (auth.error) return { error: auth.error };
const { token, projectId, location } = auth;
const base = (config.baseUrl || "https://aiplatform.googleapis.com").replace(/\/$/, "");
const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` };
if (requestId) {
const operationName = decodeJobId(requestId);
if (!operationName) return { error: "Invalid Vertex video job id" };
return {
method: "POST",
url: `${base}/v1/${modelPathOf(operationName)}:fetchPredictOperation`,
headers,
body: JSON.stringify({ operationName }),
};
}
if (action !== "generations") {
// ponytail: Veo extend/edit go through generations with `video`/`image` in the body.
return { error: `Vertex video supports 'generations' only (got '${action}')` };
}
let body;
try {
body = JSON.parse(typeof rawBody === "string" ? rawBody : rawBody.toString("utf8"));
} catch {
return { error: "Invalid JSON body" };
}
if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" };
// Plain model id only — a path segment carrying "/" or ".." would rewrite the URL.
if (!/^[A-Za-z0-9._-]+$/.test(body.model)) return { error: "Invalid Vertex video model id" };
if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" };
return {
method: "POST",
url: `${base}/v1/projects/${projectId}/locations/${location}/publishers/google/models/${body.model}:predictLongRunning`,
headers,
body: JSON.stringify(toVertexBody(body)),
};
},
transformResponse: fromVertexOperation,
};

View File

@@ -209,7 +209,10 @@ export const PROVIDER_CAPABILITIES = {
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
// deepseek-v4.1-flash replaces v4-flash (dropped from the server list;
// the old endpoint still answers 200 but the published list is the
// contract). maxOutput 128000 per the server's product-config payload.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
@@ -219,9 +222,9 @@ export const PROVIDER_CAPABILITIES = {
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
// max_output_tokens arrives as 0 for every model, so outputs are
// best-guess from the real model family. Vision tags below follow the
// upstream is_vl flag per explicit request, even though the executor
// currently sends image_urls:null (image pass-through over the agent_chat
// SSE protocol is unverified). reasoning:true on all of them — every model can
// upstream is_vl flag. The executor uploads inlined images to
// /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null
// (same as qodercli). reasoning:true on all of them — every model can
// reason; the upstream is_reasoning flag only drives model_config selection.
// thinkingFormat keeps the true-model family for documentation/UI, but
// thinkingCanDisable:false everywhere: the executor only forwards

View File

@@ -37,6 +37,7 @@ export default {
},
usage: {
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`,
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
tokenUrl: "https://oauth2.googleapis.com/token",
},

View File

@@ -20,6 +20,8 @@ export default {
authModes: [
"apikey",
],
passthroughModels: true,
modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" },
transport: {
baseUrl: "https://api.airforce/v1/chat/completions",
validateUrl: "https://api.airforce/v1/models",
@@ -27,10 +29,11 @@ export default {
"HTTP-Referer": "https://endpoint-proxy.local",
"X-Title": "Endpoint Proxy",
},
forceStream: true,
},
models: [
{ id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 },
{ id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 },
{ id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 },
{ id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 },
{ id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 },
],
};

View File

@@ -14,12 +14,16 @@ export default {
},
},
category: "oauth",
authModes: ["oauth"],
hasOAuth: true,
transport: {
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
headers: {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
quirks: { clineEnvelope: true },
tokenUrl: "https://api.cline.bot/api/v1/auth/token",
refreshUrl: "https://api.cline.bot/api/v1/auth/refresh",
auth: {

View File

@@ -14,7 +14,10 @@ export default {
},
},
category: "oauth",
authModes: ["oauth", "apikey"],
// ClinePass authenticates with a plain API key from app.cline.bot/settings/api-keys
// (category "apikey"). The OAuth extension flow used by Cline does not issue
// tokens that the ClinePass API consumer endpoint accepts (HTTP 401) — see #2333.
authModes: ["apikey", "oauth"],
hasOAuth: true,
transport: {
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
@@ -22,6 +25,8 @@ export default {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
quirks: { clineEnvelope: true },
auth: {
combined: true,
header: "Authorization",

View File

@@ -58,7 +58,9 @@ export default {
// (endpoint returns 11102 "model service info not found"), plus
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
// deepseek-v3-2-volc (absent from the server list, though still answering
// 200) and hy3-x (paid tier, not used here).
// 200) and hy3-x (paid tier, not used here). deepseek-v4-flash removed
// 2026-09: replaced server-side by deepseek-v4.1-flash (same low/high/
// xhigh efforts; endpoint still answers 200 but the list is the contract).
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
{ id: "hy3", name: "Hy3" },
{ id: "hy4-preview", name: "Hy4-Preview" },
@@ -66,7 +68,7 @@ export default {
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
{ id: "kimi-k3-1", name: "Kimi-K3" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
],
oauth: {
baseUrl: "https://copilot.tencent.com",

View File

@@ -1,5 +1,9 @@
import { withCodexReviewModels } from "../models/helpers.js";
// Codex CLI version seen by OpenAI's backend — single source for the Version /
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
const CODEX_CLI_VERSION = "0.154.0";
export default {
id: "codex",
priority: 30,
@@ -34,9 +38,10 @@ export default {
baseUrl: "https://chatgpt.com/backend-api/codex/responses",
format: "openai-responses",
forceStream: true,
cliVersion: CODEX_CLI_VERSION,
headers: {
originator: "codex_cli_rs",
"User-Agent": "codex_cli_rs/0.136.0",
"User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`,
},
usage: {
url: "https://chatgpt.com/backend-api/wham/usage",
@@ -60,6 +65,11 @@ export default {
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2", name: "GPT Image 2", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-1.5", name: "GPT Image 1.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },

View File

@@ -25,6 +25,21 @@ export default {
reasoningInject: {
scope: "all",
},
quirks: {
// DeepSeek's Anthropic-compatible endpoint
// (https://api.deepseek.com/anthropic/v1/messages) accepts ONLY the
// built-in web_search_* tools and rejects client-defined `custom` tools
// (MCP / Read / Bash / etc.) with HTTP 400
// "tools[0]: unknown variant `custom`, expected
// `web_search_20250305` or `web_search_20260209`".
//
// Declaring this whitelist makes prepareClaudeRequest() forward only
// web_search_* tools and strip everything else before sending, so MCP /
// function tools are dropped instead of failing the whole request.
// DeepSeek's OpenAI-compatible transport is unaffected (targetFormat
// there is "openai", not "claude", so prepareClaudeRequest is not run).
claudeSupportedToolTypes: ["web_search_20250305", "web_search_20260209"],
},
},
// Multi-endpoint: pick the transport matching client sourceFormat to skip translation.
transports: [

View File

@@ -123,7 +123,6 @@ import p119 from "./selfhosted-embedding.js";
import p120 from "./fish-audio.js";
import p121 from "./alitp-intl.js";
import p122 from "./xquik.js";
export default [
p0,
p1,

View File

@@ -22,6 +22,7 @@ export default {
headers: { ...CLAUDE_API_HEADERS },
quirks: {
dropOutputConfig: true,
requireClaudeToolType: true,
},
reasoningInject: {
scope: "all",

View File

@@ -22,6 +22,7 @@ export default {
headers: { ...CLAUDE_API_HEADERS },
quirks: {
dropOutputConfig: true,
requireClaudeToolType: true,
},
reasoningInject: {
scope: "all",

View File

@@ -57,6 +57,9 @@ export default {
{ id: "whisper-1", name: "Whisper 1", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-4o-transcribe", name: "GPT-4o Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-4o-mini-transcribe", name: "GPT-4o Mini Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-1", name: "GPT Image 1", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "dall-e-3", name: "DALL-E 3", params: ["size","quality","style","response_format"], kind: "image" },
{ id: "dall-e-2", name: "DALL-E 2", params: ["n","size","response_format"], kind: "image" },

View File

@@ -33,25 +33,36 @@ export default {
{ format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
],
// supportedFormats follow the endpoint table in https://opencode.ai/docs/go/
models: [
{ id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] },
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
// Muse Spark is served by /zen/go/v1/responses only — responses-only entry forces
// chatCore past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] },
{ id: "hy3", name: "Hy3", supportedFormats: ["openai"] },
// Served by /zen/go/v1/responses only — the responses-only entry forces chatCore
// past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
],

View File

@@ -40,8 +40,11 @@ export default {
{ id: "openai/gpt-image-1", name: "GPT Image 1 (via OpenRouter)", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "google/imagen-3.0-generate-002", name: "Imagen 3 (via OpenRouter)", params: ["n","size"], kind: "image" },
{ id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell (via OpenRouter)", params: ["n","size"], kind: "image" },
{ id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
],
serviceKinds: ["llm","embedding","tts","imageToText"],
serviceKinds: ["llm","embedding","tts","imageToText","video"],
ttsConfig: {
baseUrl: "https://openrouter.ai/api/v1/chat/completions",
defaultModel: "openai/gpt-4o-mini-tts",
@@ -57,6 +60,12 @@ export default {
baseUrl: "https://openrouter.ai/api/v1/images/generations",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
// Async video jobs (POST /videos → { id, status }, GET /videos/{id} polls).
// Docs: https://openrouter.ai/docs/api/api-reference/videos
videoConfig: {
baseUrl: "https://openrouter.ai/api/v1/videos",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
modelsFetcher: { url: "https://openrouter.ai/api/v1/models", type: "openrouter-free" },
passthroughModels: true,
};

View File

@@ -27,6 +27,13 @@ export default {
{ id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" },
{ id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" },
{ id: "veo-3.1-generate-preview", name: "Veo 3.1 (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
{ id: "veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
{ id: "veo-3.0-generate-001", name: "Veo 3", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
{ id: "veo-2.0-generate-001", name: "Veo 2", params: ["duration","aspect_ratio","negative_prompt","seed","storage_uri"], kind: "video" },
],
serviceKinds: ["llm","imageToText"],
serviceKinds: ["llm","imageToText","video"],
// Veo via predictLongRunning + fetchPredictOperation (adapter: handlers/videoProviders/vertex.js).
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
videoConfig: { baseUrl: "https://aiplatform.googleapis.com" },
};

View File

@@ -1,11 +1,19 @@
import { CLAUDE_API_HEADERS } from "../shared.js";
// Dual auth (same pattern as kimi):
// - API key (sk-...) → cloud API on api.xiaomimimo.com
// - Desktop account/OAuth → same cloud host, plus the Desktop-exclusive Preview
// models served by the account-service route on mimo-server-cn.xiaomimimo.com
// (authorized by a Xiaomi account session cookie, not the key).
// Endpoint is picked per model in the executor, same as opencode-go's /responses split.
export default {
id: "xiaomi-mimo",
priority: 290,
alias: "xiaomi-mimo",
aliases: [
"mimo",
"mimo-desktop",
"xmd",
],
uiAlias: "mimo",
display: {
@@ -16,9 +24,12 @@ export default {
website: "https://xiaomimimo.com",
notice: {
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
signupUrl: "https://mimo.xiaomimimo.com/desktop/invite/",
},
},
category: "apikey",
category: "oauth",
authModes: ["oauth", "apikey"],
hasOAuth: true,
serviceKinds: ["llm", "tts"],
transport: {
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
@@ -39,6 +50,11 @@ export default {
},
],
models: [
// Desktop-exclusive — served by the account-service route, which only accepts
// OpenAI format, so supportedFormats pins them to the openai transport.
{ id: "mimo-x-pro-preview", name: "MiMo-X-Pro-Preview", upstreamModelId: "xiaomi/mimo-x-pro-preview", supportedFormats: ["openai"] },
{ id: "mimo-x-flash-preview", name: "MiMo-X-Flash-Preview", upstreamModelId: "xiaomi/mimo-x-flash-preview", supportedFormats: ["openai"] },
// Cloud API models (api.xiaomimimo.com/v1)
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
{ id: "mimo-v2.5", name: "MiMo V2.5" },
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
@@ -51,4 +67,18 @@ export default {
authHeader: "bearer",
format: "xiaomi-mimo-tts",
},
features: {
usage: true,
usageApikey: true,
},
// Custom OAuth — non-standard ECDH encrypted-callback flow.
// Handled by the Xiaomi MiMo OAuth service, not the generic PKCE pipeline.
oauth: {
custom: true,
authorizeUrl: "https://platform.xiaomimimo.com/authorize",
// The callback carries ?u=<ECDH-encrypted payload> instead of ?code=.
// Decryption yields { uid, sk, url }.
callbackParam: "u",
kn: "mimocode",
},
};

View File

@@ -13,7 +13,7 @@ export function injectSystemPrompt(body, format, prompt) {
if (!body || !prompt) return;
if (typeof body !== "object") return;
// Kiro wire shape is unique (conversationState/systemPrompt) — handle directly.
// Kiro wire shape is unique (conversationState) — handle directly.
if (isKiroBody(body) || format === FORMATS.KIRO) {
injectKiroSystem(body, prompt);
return;
@@ -61,10 +61,13 @@ export function injectSystemPrompt(body, format, prompt) {
function isKiroBody(body) {
if (!body || typeof body !== "object") return false;
if (typeof body.systemPrompt !== "string") return false;
const cs = body.conversationState;
if (!cs || typeof cs !== "object") return false;
return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object");
// A top-level `systemPrompt` used to be the marker, but the Kiro translator no
// longer emits it (kiro.dev rejects the field), so gate on the turn shape.
const historyTurn = Array.isArray(cs.history)
&& cs.history.some(it => it && (it.userInputMessage || it.assistantResponseMessage));
return historyTurn || !!(cs.currentMessage && cs.currentMessage.userInputMessage);
}
// Exact idempotency: prompt present as its own SEP-delimited segment (or the
@@ -258,80 +261,33 @@ function injectGeminiSystem(body, prompt) {
}
// ---- Kiro ----
// Updates top-level systemPrompt and only the mirrored leading prefix of the
// first user history turn, else current user. next = old + SEP + prompt.
// Replace old leading prefix only; preserve time context and user tail.
// The prompt is appended to the first user turn's content — the same place the
// Kiro translator already mirrors the system text via its contentPrefix.
//
// A top-level `systemPrompt` is deliberately NOT written: the kiro.dev gateway
// answers any body carrying that field with
// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"}
// The translator stopped emitting it in v0.5.59, but this injector kept adding
// it back, so every kr/ model failed whenever an RTK prompt (caveman, ponytail)
// was active.
function injectKiroSystem(body, prompt) {
try {
let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : "";
// Repair path: a previous partial write left systemPrompt updated but user
// content still mirroring the pre-write prefix. Re-derive the effective old
// prefix from content so this pass converges instead of early-returning.
const cs0 = body.conversationState;
let firstUser0 = cs0 && Array.isArray(cs0.history)
? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null)
: null;
if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage;
if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) {
const c0 = firstUser0.content;
if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) {
// systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored
oldPrompt = "";
}
}
if (oldPrompt && hasPrompt(oldPrompt, prompt)) return;
const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt;
// Atomicity: write user content first, then systemPrompt only if content
// write succeeded (or was a no-op). If systemPrompt write then fails, the
// repair heuristic above re-derives from content on retry — no permanent
// half-applied state.
const cs = body.conversationState;
let targetMsg = null;
try {
const hist = Array.isArray(cs?.history) ? cs.history : null;
if (hist) {
for (const item of hist) {
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
}
}
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
targetMsg = cs.currentMessage.userInputMessage;
}
} catch (_) { targetMsg = null; }
let sysWritten = false;
try { body.systemPrompt = next; sysWritten = true; } catch (_) {}
const applyContent = () => {
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
if (oldPrompt === "") {
// Empty old prompt: prepend unless already at head (exact, not substring)
if (content.startsWith(prompt) || content.startsWith(next)) return;
const newContent = content ? `${next}${SEP}${content}` : next;
try { targetMsg.content = newContent; } catch (_) {}
return;
}
if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone
if (content.startsWith(next)) return; // already applied → idempotent
const tail = content.slice(oldPrompt.length);
try { targetMsg.content = `${next}${tail}`; } catch (_) {}
};
try {
if (targetMsg) applyContent();
} catch (_) {}
if (sysWritten && targetMsg) {
// verify convergence: content should now start with next (or be un-mirrored)
let ok = false;
try {
const c = targetMsg.content;
ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt);
} catch (_) {}
if (!ok) {
try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback
const hist = Array.isArray(cs?.history) ? cs.history : null;
if (hist) {
for (const item of hist) {
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
}
}
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
targetMsg = cs.currentMessage.userInputMessage;
}
if (!targetMsg) return;
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
const next = dedupStringAppend(content, prompt);
if (next === content) return; // already injected — idempotent across retries
try { targetMsg.content = next; } catch (_) { /* frozen/proxy fail-open */ }
} catch (_) {}
}

View File

@@ -19,12 +19,10 @@ function buildModelListHeaders(token, isApiKey) {
}
/**
* Fetch ClinePass live model catalog from Cline's /models endpoint.
*
* @param {object} credentials - Connection credentials ({ accessToken, apiKey })
* @returns {Promise<{ models: { id: string, name: string }[] } | null>}
* Internal: fetch the raw model list from Cline's /models endpoint.
* Returns the parsed array or null on any failure.
*/
export async function resolveClinepassModels(credentials) {
async function fetchClineRawModels(credentials) {
const isApiKey = Boolean(credentials?.apiKey);
const token = isApiKey ? credentials.apiKey : credentials?.accessToken;
if (!token) return null;
@@ -45,19 +43,53 @@ export async function resolveClinepassModels(credentials) {
const json = await response.json();
const rawList = Array.isArray(json) ? json : json?.data;
if (!Array.isArray(rawList)) return null;
const models = rawList
.filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/"))
.map((m) => ({
id: m.id,
name: m.name || m.id,
}));
return models.length ? { models } : null;
return Array.isArray(rawList) ? rawList : null;
} catch {
return null;
} finally {
clearTimeout(timer);
}
}
/**
* Fetch ClinePass live model catalog from Cline's /models endpoint.
* Returns only models with the cline-pass/ prefix.
*
* @param {object} credentials - Connection credentials ({ accessToken, apiKey })
* @returns {Promise<{ models: { id: string, name: string }[] } | null>}
*/
export async function resolveClinepassModels(credentials) {
const rawList = await fetchClineRawModels(credentials);
if (!rawList) return null;
const models = rawList
.filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/"))
.map((m) => ({
id: m.id,
name: m.name || m.id,
}));
return models.length ? { models } : null;
}
/**
* Fetch Cline live model catalog from Cline's /models endpoint.
* Unlike resolveClinepassModels, this returns ALL models (including
* free-tier models like z-ai/glm-5.3-flash) without the cline-pass/ prefix filter.
*
* @param {object} credentials - Connection credentials ({ accessToken, apiKey })
* @returns {Promise<{ models: { id: string, name: string }[] } | null>}
*/
export async function resolveClineModels(credentials) {
const rawList = await fetchClineRawModels(credentials);
if (!rawList) return null;
const models = rawList
.filter((m) => typeof m?.id === "string" && m.id.trim() !== "")
.map((m) => ({
id: m.id,
name: m.name || m.id,
}));
return models.length ? { models } : null;
}

View File

@@ -343,6 +343,30 @@ export async function resolveQoderModels(credentials, options = {}) {
}
}
/**
* Every model key the chat endpoint accepts for this credential: the IDE-visible
* models first, then catalog entries flagged `enable:false` (hidden in the IDE
* picker, e.g. by an account policy, but still served by agent_chat_generation —
* see fetchQoderCatalogRaw). /v1/models uses this so the advertised list matches
* what the router will actually route instead of collapsing to one or two keys.
*/
export function routableQoderModels(catalog) {
if (!catalog) return [];
const out = [];
const seen = new Set();
for (const m of catalog.models || []) {
if (!m?.id || seen.has(m.id)) continue;
seen.add(m.id);
out.push({ id: m.id, name: m.name || m.id, hidden: false });
}
for (const [key, cfg] of catalog.rawConfigs || []) {
if (!key || seen.has(key)) continue;
seen.add(key);
out.push({ id: key, name: cfg?.display_name || key, hidden: true });
}
return out;
}
export function invalidateQoderCatalog(credentials) {
if (!credentials) return;
catalogCache.delete(cacheKey(credentials));

View File

@@ -148,6 +148,8 @@ const REFRESH_HANDLERS = {
"codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log),
trae: (c, log) => refreshTraeToken(c.refreshToken, c, log),
cline: (c, log) => refreshClineToken(c.refreshToken, log),
// ClinePass shares Cline's WorkOS auth endpoints, so the same refresh works.
clinepass: (c, log) => refreshClineToken(c.refreshToken, log),
zed: () => refreshZedToken(),
windsurf: (c, log) => refreshWindsurfToken(c, log),
// Kimi Code OAuth (merged into id `kimi`); legacy id still routes here

View File

@@ -19,6 +19,7 @@ import { getCommandCodeUsage } from "./usage/commandcode.js";
import { getOpenCodeGoUsage } from "./usage/opencode-go.js";
import { getGroqUsage } from "./usage/groq.js";
import { getZedUsage } from "./usage/zed.js";
import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js";
import { resolveQoderCredentials } from "./qoderModels.js";
import { getGlmUsage } from "./usage/glm.js";
import {
@@ -64,6 +65,7 @@ const USAGE_HANDLERS = {
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions),
zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
"xiaomi-mimo": (c) => getXiaomiMimoUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
};
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {

View File

@@ -0,0 +1,150 @@
/**
* Antigravity weekly quota — best-effort retrieval from retrieveUserQuotaSummary.
* Failure never breaks existing per-model quota display.
*/
import { U, parseResetTime, fetchWithTimeout } from "./shared.js";
import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION } from "../../providers/shared.js";
// — Weekly quota summary config ——————————————————————————————
const WEEKLY_CONFIG = {
...U("antigravity"),
userAgent: ANTIGRAVITY_IDE_USER_AGENT,
};
// — Cache: TTL + in-flight dedup per project ———————————————
const WEEKLY_CACHE_TTL_MS = 180_000; // 3 minutes
const weeklyCache = new Map(); // cacheKey -> { result, expiresAt } | { promise }
function cacheKey(accessToken, projectId) {
return `${accessToken}::${projectId || ""}`;
}
// Exported for tests only
export function _clearWeeklyCache() {
weeklyCache.clear();
}
// — Group-name to stable key mapping ——————————————————————
const GROUP_MATCHERS = [
{ pattern: /gemini/i, key: "gemini_weekly", displayName: "Gemini (Weekly)" },
{ pattern: /claude|gpt/i, key: "claude_gpt_weekly", displayName: "Claude & GPT (Weekly)" },
];
/**
* Parse a retrieveUserQuotaSummary response into normalized weekly quotas.
* Pure function — safe to unit-test without network.
*
* @param {Object|null} data Raw JSON response
* @returns {Object} e.g. { gemini_weekly: { used, total, ... }, claude_gpt_weekly: { ... } }
*/
export function parseWeeklyQuotaSummary(data) {
if (!data || typeof data !== "object") return {};
// Groups may live at data.groups or data.quotaSummary.groups
const groups = Array.isArray(data.groups)
? data.groups
: Array.isArray(data.quotaSummary?.groups)
? data.quotaSummary.groups
: null;
if (!groups) return {};
const result = {};
for (const group of groups) {
if (!group || typeof group !== "object") continue;
const displayName = group.displayName || "";
const buckets = Array.isArray(group.buckets) ? group.buckets : [];
for (const bucket of buckets) {
if (!bucket || typeof bucket !== "object") continue;
// Identify weekly buckets by checking bucketId + displayName for "weekly"
const bucketText = `${bucket.bucketId || ""} ${bucket.displayName || ""}`.toLowerCase();
if (!bucketText.includes("weekly")) continue;
// Skip disabled buckets
if (bucket.disabled === true) continue;
const remainingFraction = Number(bucket.remainingFraction);
if (!Number.isFinite(remainingFraction)) continue;
// Match group to a known family
for (const matcher of GROUP_MATCHERS) {
if (matcher.pattern.test(displayName)) {
const total = 1000;
const remaining = Math.round(total * remainingFraction);
const used = Math.max(0, total - remaining);
result[matcher.key] = {
used,
total,
resetAt: parseResetTime(bucket.resetTime),
remainingPercentage: remainingFraction * 100,
unlimited: false,
displayName: matcher.displayName,
};
break; // first matching bucket per family wins
}
}
}
}
return result;
}
/**
* Fetch weekly quota summary — cached, deduped, never throws.
*/
export async function fetchAntigravityWeeklyQuota(accessToken, projectId, proxyOptions = null) {
const key = cacheKey(accessToken, projectId);
// Serve in-flight or cached
const hit = weeklyCache.get(key);
if (hit?.promise) return hit.promise;
if (hit && hit.expiresAt > Date.now()) return hit.result;
const promise = (async () => {
try {
const url = WEEKLY_CONFIG.quotaSummaryApiUrl;
if (!url) return {};
const response = await fetchWithTimeout(url, {
method: "POST",
headers: {
"Authorization": `Bearer ${accessToken}`,
"User-Agent": WEEKLY_CONFIG.userAgent,
"Content-Type": "application/json",
"X-Client-Name": "antigravity",
"X-Client-Version": ANTIGRAVITY_IDE_VERSION,
},
body: JSON.stringify({
...(projectId ? { project: projectId } : {}),
}),
}, 10000, proxyOptions);
if (!response.ok) return {};
const data = await response.json();
return parseWeeklyQuotaSummary(data);
} catch {
return {};
}
})();
weeklyCache.set(key, { promise });
try {
const result = await promise;
if (result && Object.keys(result).length > 0) {
weeklyCache.set(key, { result, expiresAt: Date.now() + WEEKLY_CACHE_TTL_MS });
} else {
weeklyCache.delete(key);
}
return result;
} catch {
weeklyCache.delete(key);
return {};
}
}

View File

@@ -102,32 +102,28 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
quotas["weekly (7d)"] = createQuotaObject(data.seven_day);
}
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus, seven_day_fable)
const MODEL_DISPLAY_NAMES = {
fable_5_1: "fable",
fable_5: "fable",
};
// Parse model-specific weekly windows (e.g. seven_day_sonnet, seven_day_opus)
for (const [key, value] of Object.entries(data)) {
if (key.startsWith("seven_day_") && key !== "seven_day" && hasUtilization(value)) {
const rawName = key.replace("seven_day_", "");
const modelName = MODEL_DISPLAY_NAMES[rawName] || rawName;
const modelName = key.replace("seven_day_", "");
quotas[`weekly ${modelName} (7d)`] = createQuotaObject(value);
} else if ((key === "fable" || key === "fable_5" || key === "fable_5_1") && hasUtilization(value)) {
quotas["weekly fable (7d)"] = createQuotaObject(value);
}
}
// Fallback: surface Fable quota row if weekly window exists but Fable was not returned yet
if (!quotas["weekly fable (7d)"] && hasUtilization(data.seven_day)) {
quotas["weekly fable (7d)"] = {
used: 0,
total: 100,
remaining: 100,
remainingPercentage: 100,
resetAt: parseResetTime(data.seven_day.resets_at),
unlimited: false,
};
// Model-scoped weekly limits (e.g. Fable) arrive in limits[], not as
// seven_day_* keys: { kind: "weekly_scoped", percent, resets_at,
// scope: { model: { display_name: "Fable" } } }. No limits entry means
// the account has no such window — omit the row, never fabricate one.
if (Array.isArray(data.limits)) {
for (const limit of data.limits) {
if (limit?.kind !== "weekly_scoped") continue;
const modelName = String(limit?.scope?.model?.display_name || "").trim().toLowerCase();
if (!modelName || typeof limit.percent !== "number") continue;
quotas[`weekly ${modelName} (7d)`] = createQuotaObject({
utilization: Math.max(0, Math.min(100, limit.percent)),
resets_at: limit.resets_at,
});
}
}
return {

View File

@@ -5,6 +5,7 @@
import { CLIENT_METADATA } from "../../config/appConstants.js";
import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION, ANTIGRAVITY_OAUTH_CLIENT } from "../../providers/shared.js";
import { U, parseResetTime, normalizeCloudCodeProjectId, fetchWithTimeout } from "./shared.js";
import { fetchAntigravityWeeklyQuota } from "./antigravity-weekly.js";
// Antigravity API config (from Quotio) — urls from registry, oauth client + dynamic UA kept here
const ANTIGRAVITY_CONFIG = {
@@ -157,8 +158,15 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
const data = await response.json();
const quotas = {};
// Parse model quotas (inspired by vscode-antigravity-cockpit)
if (data.models) {
// Detect tier: free-tier accounts only have weekly quotas (no separate 5h window).
// On free-tier, fetchAvailableModels returns misleading per-model quota info
// (missing remainingFraction defaults to 0, or reflects the weekly limit not a 5h window).
const paidTierId = subscriptionInfo?.paidTier?.id;
const isFreeTier = !paidTierId || paidTierId === "free-tier";
// Parse model quotas only for paid-tier accounts.
// Free-tier accounts skip this — their only meaningful quota is the weekly limit.
if (!isFreeTier && data.models) {
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
const importantModels = [
'gemini-3.8-flash-high',
@@ -212,6 +220,56 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
}
}
// Best-effort weekly quota overlay — never blocks or breaks per-model results
try {
const weeklyQuotas = await fetchAntigravityWeeklyQuota(
accessToken,
projectId,
proxyOptions
);
// Reconcile weekly quota against model family status:
// If every model in a family is locked/exhausted (remainingPercentage === 0)
// until a future reset time, the weekly limit cannot be 100% available.
// On Google's Free Starter tier, retrieveUserQuotaSummary buggily reports
// remainingFraction: 1 even after the starter quota is depleted and all models 429.
const entries = Object.entries(quotas);
const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image"));
const claudeModels = entries.filter(([k]) => k.startsWith("claude-"));
if (weeklyQuotas.gemini_weekly && geminiModels.length > 0) {
const allGeminiExhausted = geminiModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0);
if (allGeminiExhausted && weeklyQuotas.gemini_weekly.remainingPercentage > 0) {
const maxResetAt = geminiModels.reduce((max, [, q]) =>
!max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null
);
weeklyQuotas.gemini_weekly.used = weeklyQuotas.gemini_weekly.total;
weeklyQuotas.gemini_weekly.remainingPercentage = 0;
if (maxResetAt) {
weeklyQuotas.gemini_weekly.resetAt = maxResetAt;
}
}
}
if (weeklyQuotas.claude_gpt_weekly && claudeModels.length > 0) {
const allClaudeExhausted = claudeModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0);
if (allClaudeExhausted && weeklyQuotas.claude_gpt_weekly.remainingPercentage > 0) {
const maxResetAt = claudeModels.reduce((max, [, q]) =>
!max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null
);
weeklyQuotas.claude_gpt_weekly.used = weeklyQuotas.claude_gpt_weekly.total;
weeklyQuotas.claude_gpt_weekly.remainingPercentage = 0;
if (maxResetAt) {
weeklyQuotas.claude_gpt_weekly.resetAt = maxResetAt;
}
}
}
Object.assign(quotas, weeklyQuotas);
} catch {
// Silently ignore — weekly is best-effort
}
return {
plan: subscriptionInfo?.currentTier?.name || "Unknown",
quotas,

View File

@@ -0,0 +1,125 @@
/**
* Xiaomi MiMo usage — weekly quota from the Xiaomi account session.
*
* Primary path: GET {mimo-server}/api/user/usage authorized by the account-session
* cookie (see shared/mimoAccount.js). Response: { code: 0, data: { percent (remaining
* %), resetDate, resetAt } }.
*
* Fallback: the sk- API key cannot read the quota, so when no account session is
* available we surface a graceful message instead of failing.
*/
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { getMimoAccountUsage } from "../../shared/mimoAccount.js";
const USAGE_URL = "https://aistudio.xiaomimimo.com/open-apis/v1/user/usage";
/**
* @param {string|null|undefined} accessToken - sk- API key
* @param {object|null} providerSpecificData - may contain mimoPassToken, uid, etc.
* @param {object|null} proxyOptions
*/
export async function getXiaomiMimoUsage(accessToken = null, providerSpecificData = null, proxyOptions = null) {
// Preferred path: the weekly quota comes from the account service session
// (mimo-server /api/user/usage), which the sk- key cannot reach. The session is
// derived from MiMo Desktop's persisted passToken via the SSO/sts handshake.
const account = await getMimoAccountUsage(providerSpecificData, proxyOptions);
if (typeof account.percent === "number" && Number.isFinite(account.percent)) {
const remaining = Math.max(0, Math.min(100, Math.round(account.percent)));
const used = 100 - remaining;
let resetAt = null;
if (typeof account.resetAt === "number" && account.resetAt > 0) {
resetAt = new Date(account.resetAt * 1000).toISOString();
} else if (typeof account.resetDate === "string") {
const parsed = new Date(`${account.resetDate}T00:00:00Z`);
if (!Number.isNaN(parsed.getTime())) resetAt = parsed.toISOString();
}
return {
plan: "Xiaomi MiMo Desktop",
quotas: {
Weekly: { used, total: 100, remainingPercentage: remaining, resetAt, unlimited: false },
},
};
}
// Fallback: no account session available (Desktop never logged in, or its cookie
// store is locked). The sk- key cannot read the quota, so surface a clear message.
const key = accessToken || providerSpecificData?.apiKey;
if (!key || typeof key !== "string" || !key.trim()) {
return { message: "Xiaomi MiMo Desktop not connected. Add credentials to view usage." };
}
try {
const response = await proxyAwareFetch(
USAGE_URL,
{
method: "GET",
headers: {
Authorization: `Bearer ${key.trim()}`,
"X-Mimo-Source": "mimocode-cli",
Accept: "application/json",
},
signal: AbortSignal.timeout(10000),
},
proxyOptions,
);
if (response.status === 401) {
return {
plan: "Xiaomi MiMo Desktop",
message: "Weekly quota requires Xiaomi account session. API key alone is insufficient.",
};
}
if (!response.ok) {
return {
plan: "Xiaomi MiMo Desktop",
message: `Usage API error (${response.status})`,
};
}
const data = await response.json().catch(() => null);
if (!data || data.code !== 0 || !data.data) {
return {
plan: "Xiaomi MiMo Desktop",
message: "Usage endpoint returned unexpected response.",
};
}
const { percent, resetDate } = data.data;
if (typeof percent !== "number" || !Number.isFinite(percent)) {
return {
plan: "Xiaomi MiMo Desktop",
message: "Usage data missing percent field.",
};
}
// percent = remaining percentage (e.g. 94 means 94% remaining)
const remaining = Math.max(0, Math.min(100, Math.round(percent)));
const used = 100 - remaining;
// Parse resetDate — expected format "2026-09-16"
let resetAt = null;
if (resetDate && typeof resetDate === "string") {
const parsed = new Date(`${resetDate}T00:00:00Z`);
if (!Number.isNaN(parsed.getTime())) {
resetAt = parsed.toISOString();
}
}
return {
plan: "Xiaomi MiMo Desktop",
quotas: {
Weekly: {
used,
total: 100,
remainingPercentage: remaining,
resetAt,
unlimited: false,
},
},
};
} catch (error) {
return { message: `Xiaomi MiMo Desktop usage error: ${error.message}` };
}
}

View File

@@ -6,7 +6,14 @@ export function getClineAccessToken(token) {
if (typeof token !== "string") return "";
const trimmed = token.trim();
if (!trimmed) return "";
return trimmed.startsWith("workos:") ? trimmed : `workos:${trimmed}`;
if (trimmed.toLowerCase().startsWith("workos:")) return trimmed;
// Cline OAuth access tokens are WorkOS JWTs (base64url `eyJ…` header).
// ClinePass API keys (category "apikey", e.g. `clp_…`) are NOT JWTs and must
// be sent verbatim — prefixing them with `workos:` makes the Cline API reject
// the request with HTTP 401 ("Please make sure you're using the latest
// version of Cline and re-authenticate your Cline account.").
const isWorkOsJwt = /^eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+/.test(trimmed);
return isWorkOsJwt ? `workos:${trimmed}` : trimmed;
}
export function getClineAuthorizationHeader(token) {

View File

@@ -0,0 +1,19 @@
import { PROVIDERS } from "../providers/index.js";
/**
* Unwrap Cline's non-stream envelope: {"success":true,"data":{...choices...}}.
*
* Scoped to providers opting in via `transport.quirks.clineEnvelope` so no other
* provider's body is ever rewritten. The error envelope ({"success":false,...})
* never matches and passes through untouched.
*
* @param {object} body - Parsed upstream response body
* @param {string} provider - Provider id or alias
* @returns {object} The inner `data` object, or `body` unchanged
*/
export function unwrapClineEnvelope(body, provider) {
if (!provider || !PROVIDERS[provider]?.quirks?.clineEnvelope) return body;
const { success, data } = body || {};
if (success !== true || !data || typeof data !== "object" || Array.isArray(data)) return body;
return data;
}

View File

@@ -0,0 +1,264 @@
import fs from "node:fs";
import os from "node:os";
import path from "node:path";
import crypto from "node:crypto";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
/**
* Xiaomi MiMo account-session helpers (used for weekly quota).
*
* The weekly quota endpoint lives on the account service domain and is authorized
* by an account session cookie, NOT the sk- API key. Acquiring that cookie mirrors
* MiMo Desktop: a passToken (persisted in Desktop's cookie store) is exchanged via
* the passportapi SSO, then authorized for the `mimopc` service, and finally stamped
* by the mimo-server /api/sts callback into a `serviceToken` cookie.
*
* Flow (verified against MiMo Desktop traffic):
* 1. GET {api}/api/user/xiaomi/me -> 302 to account SSO (sid=mimopc)
* 2. GET account /pass/serviceLogin?sid=passportapi&_json=true -> nonce/ssecurity
* 3. GET {location}&clientSign=... -> account-level serviceToken
* 4. GET account /pass/serviceLogin?sid=mimopc&callback=<sts>&_json=true
* 5. GET {api}/api/sts?...&ticket... -> Set-Cookie: serviceToken (mimopc scope)
*/
const API_BASE = "https://mimo-server-cn.xiaomimimo.com";
const ACCOUNT_HOST = "account.xiaomi.com";
const API_UA =
"miNative PC/Normal Windows_NT/10.0.19045 SDKV/1.0.0 DEVT/PC DEVS/Windows APP/miaccount_desktop APPV/0.1.0";
const SSO_UA = "MiClaw/1.0";
const COOKIE_TTL_MS = 30 * 60 * 1000;
// Per-account session caches (keyed by passToken hash) so multiple Xiaomi
// accounts / connections can rotate without clobbering each other.
const _cache = new Map(); // key -> { cookie, at }
const _inflight = new Map(); // key -> Promise<cookie|null>
function desktopCookiePath() {
const home = os.homedir();
if (process.platform === "win32") {
return path.join(home, "AppData", "Roaming", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies");
}
if (process.platform === "darwin") {
return path.join(home, "Library", "Application Support", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies");
}
return path.join(home, ".config", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies");
}
/**
* Read the persisted Xiaomi account cookies from MiMo Desktop's Electron profile.
* The Chromium cookie DB is held with an exclusive lock while Desktop runs, so we
* copy it first and bail (return null) if that fails.
* @returns {Promise<Record<string,string>|null>}
*/
async function readDesktopAccountCookies() {
const src = desktopCookiePath();
if (!fs.existsSync(src)) return null;
const tmp = path.join(os.tmpdir(), `9r-mimo-cookies-${process.pid}-${crypto.randomBytes(4).toString("hex")}.db`);
try {
fs.copyFileSync(src, tmp);
} catch {
return null; // locked by a running Desktop
}
try {
const { DatabaseSync } = await import("node:sqlite");
const db = new DatabaseSync(tmp, { readOnly: true });
const rows = db.prepare("SELECT name, value FROM cookies WHERE host_key = ?").all("." + ACCOUNT_HOST);
db.close();
const jar = Object.fromEntries(rows.map((r) => [r.name, r.value]));
return jar.passToken ? jar : null;
} catch {
return null;
} finally {
try {
fs.unlinkSync(tmp);
} catch {
/* ignore */
}
}
}
/**
* Read just the passToken + identity cookies from Desktop's profile.
* Exported so the connect flow can persist a per-account passToken into the
* connection's providerSpecificData — this is what enables multi-account rotation.
* @returns {Promise<{passToken:string, userId:string|null, cUserId:string|null}|null>}
*/
export async function readDesktopPassToken() {
try {
const jar = await readDesktopAccountCookies();
if (!jar?.passToken) return null;
return { passToken: jar.passToken, userId: jar.userId || null, cUserId: jar.cUserId || null };
} catch {
return null;
}
}
function signatureClientSign(nonce, ssecurity) {
const input = `nonce=${nonce}` + (ssecurity && ssecurity.trim() ? `&${ssecurity}` : "");
return encodeURIComponent(crypto.createHash("sha1").update(input).digest("base64"));
}
function absorbSetCookie(jar, res) {
for (const c of res.headers.getSetCookie?.() || []) {
const m = /^([^=]+)=([^;]*)/.exec(c.trim());
if (m && m[2]) jar[m[1]] = m[2];
}
}
function cookieHeader(jar) {
return Object.entries(jar)
.filter(([, v]) => v)
.map(([k, v]) => `${k}=${v}`)
.join("; ");
}
/**
* Exchange a passToken for a mimo-server service session cookie.
* @returns {Promise<string|null>} Cookie header value, or null on failure.
*/
async function acquireServiceCookie(passJar, proxyOptions) {
const jar = { ...passJar };
const ck = () => cookieHeader(jar);
// 1. Unauthenticated API call -> 302 carrying the sts callback (sid=mimopc)
const r1 = await proxyAwareFetch(
`${API_BASE}/api/user/xiaomi/me`,
{ redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } },
proxyOptions,
);
const redirect = r1.headers.get("location");
if (!redirect) return null;
const stsCallback = new URL(redirect).searchParams.get("callback");
if (!stsCallback) return null;
// 2. passportapi SSO phase 1 -> nonce + ssecurity
const sso1 = await proxyAwareFetch(
`https://${ACCOUNT_HOST}/pass/serviceLogin?sid=passportapi&_json=true`,
{ headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } },
proxyOptions,
);
const j1 = JSON.parse((await sso1.text()).replace(/^&&&START&&&/, ""));
const nonce = j1.nonce || (j1.location ? new URL(j1.location).searchParams.get("nonce") : null);
if (!nonce || !j1.location) return null;
// 3. passportapi SSO phase 2 -> account-level serviceToken
const sso2 = await proxyAwareFetch(
`${j1.location}&clientSign=${signatureClientSign(nonce, j1.ssecurity)}`,
{ redirect: "manual", headers: { Cookie: ck(), "User-Agent": SSO_UA } },
proxyOptions,
);
absorbSetCookie(jar, sso2);
// 4. mimopc SSO -> sts callback carrying a ticket
const sso3 = await proxyAwareFetch(
`https://${ACCOUNT_HOST}/pass/serviceLogin?sid=mimopc&callback=${encodeURIComponent(stsCallback)}&_json=true`,
{ headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } },
proxyOptions,
);
const j3 = JSON.parse((await sso3.text()).replace(/^&&&START&&&/, ""));
absorbSetCookie(jar, sso3);
if (!j3?.location || !/\/api\/sts/.test(j3.location)) return null;
// 5. sts callback -> Set-Cookie: serviceToken (mimopc scope)
const sts = await proxyAwareFetch(
j3.location,
{ redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } },
proxyOptions,
);
absorbSetCookie(jar, sts);
const needed = ["serviceToken", "mimopc_ph", "mimopc_slh", "userId"];
if (!jar.serviceToken) return null;
const out = {};
for (const k of needed) if (jar[k]) out[k] = jar[k];
return cookieHeader(out);
}
/**
* Get (and cache) the mimo-server account cookie.
* @param {object|null} providerSpecificData - may carry `mimoPassToken` override
*/
async function getServiceCookie(providerSpecificData, proxyOptions) {
const passJar = providerSpecificData?.mimoPassToken
? { passToken: providerSpecificData.mimoPassToken, userId: providerSpecificData.mimoUserId, cUserId: providerSpecificData.mimoCUserId }
: await readDesktopAccountCookies();
if (!passJar) return { cookie: null, reason: "no-pass-token" };
// One cached session per passToken — accounts/connections rotate independently.
const key = crypto.createHash("sha256").update(passJar.passToken).digest("hex");
const cached = _cache.get(key);
if (cached && Date.now() - cached.at < COOKIE_TTL_MS) {
return { cookie: cached.cookie };
}
// De-dupe concurrent handshakes for the same account: a burst of requests must
// not each run the full 5-step SSO chain.
const inflight = _inflight.get(key);
if (inflight) {
const cookie = await inflight;
return cookie ? { cookie } : { cookie: null, reason: "sso-failed" };
}
const promise = (async () => {
try {
return await acquireServiceCookie(passJar, proxyOptions);
} catch {
return null; // network/parse failure — callers degrade, never throw
} finally {
_inflight.delete(key);
}
})();
_inflight.set(key, promise);
const cookie = await promise;
if (!cookie) return { cookie: null, reason: "sso-failed" };
_cache.set(key, { cookie, at: Date.now() });
return { cookie };
}
/** Drop cached sessions so the next call re-runs the handshake (e.g. after a 401). */
export function invalidateMimoAccountCookieCache() {
_cache.clear();
}
/** mimo-server account API base + the User-Agent its backend expects. */
export const MIMO_API_BASE = API_BASE;
export const MIMO_API_UA = API_UA;
/**
* Resolve the mimo-server account-session cookie, for upstream /api/route/* calls.
* @returns {Promise<string|null>} Cookie header value, or null when unavailable.
*/
export async function getMimoAccountCookie(providerSpecificData = null, proxyOptions = null) {
try {
const { cookie } = await getServiceCookie(providerSpecificData, proxyOptions);
return cookie;
} catch {
return null;
}
}
/**
* Fetch the weekly quota from the account service.
* @returns {Promise<{percent?:number, resetDate?:string, resetAt?:number, error?:string}>}
*/
export async function getMimoAccountUsage(providerSpecificData = null, proxyOptions = null) {
const { cookie, reason } = await getServiceCookie(providerSpecificData, proxyOptions);
if (!cookie) {
return { error: reason === "no-pass-token" ? "no-session" : "session-failed" };
}
try {
const res = await proxyAwareFetch(
`${API_BASE}/api/user/usage`,
{ headers: { "User-Agent": API_UA, Cookie: cookie, Accept: "application/json" }, signal: AbortSignal.timeout(10000) },
proxyOptions,
);
if (!res.ok) return { error: `http-${res.status}` };
const data = await res.json().catch(() => null);
if (!data || data.code !== 0 || !data.data) return { error: "bad-response" };
return { percent: data.data.percent, resetDate: data.data.resetDate, resetAt: data.data.resetAt };
} catch (e) {
return { error: e.message };
}
}

View File

@@ -0,0 +1,341 @@
/**
* Native qodercli does NOT stuff image/PDF bytes into agent_chat_generation.
* It PUTs them to /algo/api/v2/image/upload (COSY-signed multipart) and then
* sends the returned OSS URL. Agents like Claude Code send OpenAI/Claude
* data-URIs instead, which 9router previously forwarded verbatim — 10MB
* images become 30MB+ JSON and upstream 413s even though the model window
* is ~200k tokens.
*
* This module:
* 1. Uploads inlined images to Qoder's file API (cached by sha256).
* 2. Replaces huge non-image file blocks with a short stub.
* 3. Caps leftover data-URIs so the chat JSON stays small.
*/
import { createHash } from "crypto";
import { v4 as uuidv4 } from "uuid";
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
import { parseDataUri } from "../../translator/concerns/image.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../../translator/schema/blocks.js";
import { MAX_IMAGE_BYTES } from "../../config/mediaConfig.js";
import { buildCosyHeaders } from "./cosy.js";
import {
QODER_IMAGE_UPLOAD_SIG_PATH,
QODER_INLINE_FALLBACK_MAX_BYTES,
QODER_MAX_PAYLOAD_BYTES,
qoderInferenceBase,
} from "./constants.js";
const IMAGE_MIME_RE = /^image\//i;
const DATA_URI_RE = /data:[^;]+;base64,[A-Za-z0-9+/=\s]+/g;
function mimeExt(mime) {
const m = String(mime || "").toLowerCase();
if (m.includes("png")) return "png";
if (m.includes("jpeg") || m.includes("jpg")) return "jpg";
if (m.includes("gif")) return "gif";
if (m.includes("webp")) return "webp";
if (m.includes("bmp")) return "bmp";
if (m.includes("pdf")) return "pdf";
return "bin";
}
function decodedBytes(b64) {
if (typeof b64 !== "string" || !b64) return 0;
const compact = b64.replace(/\s/g, "");
return Math.floor(compact.length * 3 / 4);
}
function stubText({ name, mime, bytes, reason }) {
const label = name || mime || "attachment";
const size = bytes ? `, ${bytes} bytes` : "";
return `[file omitted: ${label}${size} — ${reason}]`;
}
export function buildMultipartFile(buffer, { fieldName = "file", fileName, mediaType } = {}) {
const boundary = `----9routerQoder${Date.now().toString(16)}${Math.random().toString(16).slice(2)}`;
const filename = fileName || `upload.${mimeExt(mediaType)}`;
const head = Buffer.from(
`--${boundary}\r\nContent-Disposition: form-data; name="${fieldName}"; filename="${filename}"\r\nContent-Type: ${mediaType || "application/octet-stream"}\r\n\r\n`,
);
const tail = Buffer.from(`\r\n--${boundary}--\r\n`);
const body = Buffer.concat([head, buffer, tail]);
return { boundary, body };
}
function extractUrlFromUploadResponse(json) {
if (!json || typeof json !== "object") return null;
const result = json.result && typeof json.result === "object" ? json.result : json;
const arrays = [result.imageUrls, result.image_urls, json.imageUrls, json.image_urls];
for (const arr of arrays) {
if (Array.isArray(arr) && typeof arr[0] === "string" && arr[0]) return arr[0];
}
const keys = ["imageUrl", "image_url", "url", "ossUrl", "oss_url", "originalUrl", "originUrl", "link", "image"];
for (const key of keys) {
const v = result[key] ?? json[key];
if (typeof v === "string" && v) return v;
}
if (typeof json.body === "string") {
try { return extractUrlFromUploadResponse(JSON.parse(json.body)); } catch { /* ignore */ }
}
return null;
}
async function defaultUploadImage({ buffer, mediaType, credentials, proxyOptions, signal }) {
const requestId = uuidv4();
const url = `${qoderInferenceBase(credentials)}${`/algo${QODER_IMAGE_UPLOAD_SIG_PATH}`}?request_id=${requestId}`;
const { boundary, body } = buildMultipartFile(buffer, {
fileName: `image.${mimeExt(mediaType)}`,
mediaType: mediaType || "application/octet-stream",
});
const psd = credentials?.providerSpecificData || {};
const cosyHeaders = buildCosyHeaders(body, url, {
userId: psd.userId,
authToken: credentials.accessToken,
name: credentials.displayName || "",
email: credentials.email || "",
machineId: psd.machineId || "",
});
const headers = {
...cosyHeaders,
Accept: "application/json",
"Content-Type": `multipart/form-data; boundary=${boundary}`,
"Content-Length": String(body.length),
"AI-CLIENT-TIMESTAMP": String(Math.floor(Date.now() / 1000)),
"Accept-Encoding": "identity",
};
const res = await proxyAwareFetch(
url,
{ method: "PUT", headers, body, signal },
proxyOptions,
);
if (!res.ok) {
const text = await res.text().catch(() => "");
throw new Error(`HTTP ${res.status}${text ? `: ${text.slice(0, 180)}` : ""}`);
}
const json = await res.json().catch(() => null);
const uploaded = extractUrlFromUploadResponse(json);
if (!uploaded) throw new Error("upload response missing url");
return uploaded;
}
async function uploadImageData({ base64, mediaType, credentials, proxyOptions, signal, log, uploadFn, cache }) {
const compact = String(base64 || "").replace(/\s/g, "");
if (!compact) return null;
const bytes = decodedBytes(compact);
if (bytes > MAX_IMAGE_BYTES) {
log?.warn?.("QODER", `image ${bytes} bytes exceeds upload cap, stubbing`);
return { stub: true, bytes, mime: mediaType };
}
let buffer;
try {
buffer = Buffer.from(compact, "base64");
} catch {
return { stub: true, bytes, mime: mediaType };
}
const digest = createHash("sha256").update(buffer).digest("hex");
if (cache?.has(digest)) return { url: cache.get(digest), bytes, mime: mediaType };
const doUpload = uploadFn || defaultUploadImage;
try {
const url = await doUpload({ buffer, mediaType, credentials, proxyOptions, signal });
if (typeof url === "string" && url) {
cache?.set(digest, url);
return { url, bytes, mime: mediaType };
}
} catch (err) {
log?.warn?.("QODER", `image upload failed (${err.message}); ${bytes <= QODER_INLINE_FALLBACK_MAX_BYTES ? "keeping inline" : "stubbing"}`);
}
if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return { keep: true, bytes, mime: mediaType };
return { stub: true, bytes, mime: mediaType };
}
function imageUrlBlock(url) {
return { type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } };
}
async function rewriteBlock(block, ctx) {
if (!block || typeof block !== "object") return block;
if (block.type === OPENAI_BLOCK.IMAGE_URL) {
const raw = typeof block.image_url === "string" ? block.image_url : block.image_url?.url;
if (typeof raw !== "string" || !raw) return null;
if (raw.startsWith("http://") || raw.startsWith("https://")) return imageUrlBlock(raw);
const parsed = parseDataUri(raw);
if (!parsed) return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "attachment", reason: "unreadable data URI" }) };
if (!IMAGE_MIME_RE.test(parsed.mimeType)) {
return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "file", mime: parsed.mimeType, bytes: decodedBytes(parsed.base64), reason: "non-image bytes are not inlined into Qoder context" }) };
}
const up = await uploadImageData({ ...ctx, base64: parsed.base64, mediaType: parsed.mimeType });
if (up?.url) return imageUrlBlock(up.url);
if (up?.keep) return imageUrlBlock(raw);
return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", mime: parsed.mimeType, bytes: up?.bytes, reason: "upload failed; not inlined" }) };
}
if (block.type === OPENAI_BLOCK.IMAGE || block.type === CLAUDE_BLOCK.IMAGE) {
const src = block.source || {};
if (src.type === "url" && typeof src.url === "string") return imageUrlBlock(src.url);
if (src.type === "base64" && src.data) {
const mime = src.media_type || "image/png";
const up = await uploadImageData({ ...ctx, base64: src.data, mediaType: mime });
if (up?.url) return imageUrlBlock(up.url);
if (up?.keep) return imageUrlBlock(`data:${mime};base64,${src.data}`);
return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", mime, bytes: up?.bytes, reason: "upload failed; not inlined" }) };
}
}
if (block.type === OPENAI_BLOCK.FILE && block.file) {
const file = block.file;
const name = file.filename || file.name || "file";
const dataUri = typeof file.file_data === "string" ? file.file_data : null;
const parsed = dataUri ? parseDataUri(dataUri) : null;
const b64 = parsed?.base64 || (typeof file.file_data === "string" && !file.file_data.startsWith("data:") ? file.file_data : null);
const mime = parsed?.mimeType || file.format || "application/octet-stream";
if (b64 && IMAGE_MIME_RE.test(mime)) {
const up = await uploadImageData({ ...ctx, base64: b64, mediaType: mime });
if (up?.url) return imageUrlBlock(up.url);
}
return { type: OPENAI_BLOCK.TEXT, text: stubText({ name, mime, bytes: decodedBytes(b64 || ""), reason: "Qoder reads documents via its file API, not inlined bytes" }) };
}
if (block.type === CLAUDE_BLOCK.DOCUMENT && block.source) {
const src = block.source;
const name = block.title || "document";
if (src.type === "base64" && src.data) {
const mime = src.media_type || "application/pdf";
if (IMAGE_MIME_RE.test(mime)) {
const up = await uploadImageData({ ...ctx, base64: src.data, mediaType: mime });
if (up?.url) return imageUrlBlock(up.url);
}
return { type: OPENAI_BLOCK.TEXT, text: stubText({ name, mime, bytes: decodedBytes(src.data), reason: "Qoder reads documents via its file API, not inlined bytes" }) };
}
}
if (typeof block.text === "string" && block.text.includes("data:") && block.text.length > 8192) {
const next = block.text.replace(DATA_URI_RE, (m) => {
const parsed = parseDataUri(m.trim());
const bytes = parsed ? decodedBytes(parsed.base64) : m.length;
if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return m;
return stubText({ mime: parsed?.mimeType, bytes, reason: "inlined data URI stripped from Qoder context" });
});
return { ...block, text: next };
}
return block;
}
async function rewriteContent(content, ctx) {
if (typeof content === "string") {
if (content.includes("data:") && content.length > 8192) {
return content.replace(DATA_URI_RE, (m) => {
const parsed = parseDataUri(m.trim());
const bytes = parsed ? decodedBytes(parsed.base64) : m.length;
if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return m;
return stubText({ mime: parsed?.mimeType, bytes, reason: "inlined data URI stripped from Qoder context" });
});
}
return content;
}
if (!Array.isArray(content)) return content;
const out = [];
for (const block of content) {
const next = await rewriteBlock(block, ctx);
if (next == null) continue;
out.push(next);
}
return out.length ? out : "";
}
function payloadBytes(messages) {
try {
return Buffer.byteLength(JSON.stringify(messages), "utf8");
} catch {
return 0;
}
}
function stripRemainingDataUris(messages) {
for (const msg of messages || []) {
if (typeof msg?.content === "string" && msg.content.includes("data:")) {
msg.content = msg.content.replace(DATA_URI_RE, (m) =>
stubText({ bytes: m.length, reason: "payload over Qoder size budget" }),
);
} else if (Array.isArray(msg?.content)) {
msg.content = msg.content.map((block) => {
if (block?.type === OPENAI_BLOCK.IMAGE_URL) {
const raw = typeof block.image_url === "string" ? block.image_url : block.image_url?.url;
if (typeof raw === "string" && raw.startsWith("data:")) {
return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", reason: "payload over Qoder size budget" }) };
}
}
if (typeof block?.text === "string" && block.text.includes("data:")) {
return { ...block, text: block.text.replace(DATA_URI_RE, (m) =>
stubText({ bytes: m.length, reason: "payload over Qoder size budget" }),
) };
}
return block;
});
}
}
}
/**
* Rewrite OpenAI-shaped messages in place: upload images, stub huge files.
* @returns {Promise<{imageUrls: string[], uploaded: number, stubbed: number}>}
*/
export async function rewriteQoderMessageAttachments(messages, {
credentials,
log,
proxyOptions = null,
signal = null,
uploadFn = null,
} = {}) {
const stats = { imageUrls: [], uploaded: 0, stubbed: 0 };
if (!Array.isArray(messages) || messages.length === 0) return stats;
const ctx = { credentials, log, proxyOptions, signal, uploadFn, cache: new Map() };
for (const msg of messages) {
if (!msg || typeof msg !== "object") continue;
if (Array.isArray(msg.images)) {
// Ollama-style sidecar; fold into content so normalizeMessages can see them.
const extras = msg.images.map((url) => imageUrlBlock(String(url)));
msg.content = Array.isArray(msg.content)
? [...msg.content, ...extras]
: [{ type: OPENAI_BLOCK.TEXT, text: typeof msg.content === "string" ? msg.content : "" }, ...extras];
delete msg.images;
}
msg.content = await rewriteContent(msg.content, ctx);
}
// Collect surviving http(s) image URLs for callers that want image_urls.
for (const msg of messages) {
if (!Array.isArray(msg?.content)) continue;
for (const block of msg.content) {
const url = block?.type === OPENAI_BLOCK.IMAGE_URL
? (typeof block.image_url === "string" ? block.image_url : block.image_url?.url)
: null;
if (typeof url === "string" && /^https?:\/\//i.test(url)) stats.imageUrls.push(url);
if (block?.type === OPENAI_BLOCK.TEXT && typeof block.text === "string" && block.text.startsWith("[file omitted:")) stats.stubbed += 1;
}
}
stats.uploaded = stats.imageUrls.length;
if (payloadBytes(messages) > QODER_MAX_PAYLOAD_BYTES) {
log?.warn?.("QODER", `request still ${payloadBytes(messages)} bytes after rewrite; stripping leftover data URIs`);
stripRemainingDataUris(messages);
}
return stats;
}
/** Test helper kept for callers; upload memo is now per-request. */
export function clearQoderUploadCache() {}
export const __test__ = {
extractUrlFromUploadResponse,
decodedBytes,
stubText,
payloadBytes,
};

View File

@@ -33,6 +33,39 @@ export const QODER_CHAT_SIG_PATH = "/api/v2/service/pro/sse/agent_chat_generatio
export const QODER_CHAT_URL = `${QODER_CHAT_BASE}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common`;
export const QODER_CHAT_URL_ENCODED = `${QODER_CHAT_URL}&Encode=1`;
export const QODER_MODEL_LIST_URL = `${QODER_CHAT_BASE}/algo/api/v2/model/list`;
// Official qodercli uploads images here (COSY-signed PUT multipart, field "file")
// instead of inlining base64 into agent_chat_generation.
export const QODER_IMAGE_UPLOAD_SIG_PATH = "/api/v2/image/upload";
// Drop remaining inlined binaries if the Qoder JSON body would still exceed this.
// 30MB+ payloads are what blow past Claude-Code's ~200k context on the wire.
export const QODER_MAX_PAYLOAD_BYTES = 6 * 1024 * 1024;
// If OSS upload fails, keep tiny data-URIs; anything larger is stubbed.
export const QODER_INLINE_FALLBACK_MAX_BYTES = 512 * 1024;
// Context-window tier selection (see shared/qoder/contextTier.js). The IDE exposes the
// model's context_config tiers (200K/400K/1M); we auto-escalate when the estimated prompt
// (+ headroom, tokenizer variance) no longer fits the current max_input_tokens.
export const QODER_CONTEXT_TIER_HEADROOM = 0.15;
export const QODER_CONTEXT_TIER_ENV = "QODER_CONTEXT_TIER";
export const QODER_CONTEXT_TIER_MODES = Object.freeze({ AUTO: "auto", MAX: "max", DEFAULT: "default" });
/**
* Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt- with
* "Login expired" (403). Device tokens (dt-...) stay on api3. PATs (pt-...)
* are exchanged for jt- before this is consulted.
*/
export function qoderInferenceBase(credentials) {
const raw = credentials?.apiKey || credentials?.accessToken;
if (
typeof raw === "string" &&
!raw.startsWith("pt-") &&
(raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))
) {
return QODER_CHAT_BASE_ALT;
}
return QODER_CHAT_BASE;
}
// COSY header constants. These are not arbitrary — the upstream signature
// validation matches them against the values used at signing time.

View File

@@ -0,0 +1,160 @@
/**
* Qoder context-window tiers.
*
* Each Qoder model_config ships a `context_config` list (e.g. 200K / 400K / 1M for
* qmodel_38max) while `max_input_tokens` only carries the tier the IDE currently has
* selected (~180K by default). The Qoder IDE lets the user switch tiers from the model
* picker; a qodercli-style client (which is what 9router impersonates) has no picker,
* so a long Claude-Code / Codex session that grew past the default tier is rejected
* upstream even though the model itself supports 1M.
*
* This module emulates the IDE: estimate the prompt size, pick the smallest advertised
* tier that fits (never below the model's current default), and mirror the choice into
* the same three places the IDE writes:
* parameters.context_length
* chat_context.extra.ideModelConfigOverride.max_input_tokens
* model_config.max_input_tokens
*
* Override with QODER_CONTEXT_TIER = auto (default) | max | default | <tier name, e.g. 1M>.
* Pure functions, no I/O — the executor wires them into buildQoderRequestBody.
*/
import { QODER_CONTEXT_TIER_HEADROOM, QODER_CONTEXT_TIER_MODES } from "./constants.js";
const UNIT = { K: 1_000, M: 1_000_000 };
/** "200K" | "1M" | "204800" | 204800 → integer token count (0 when unparseable). */
export function parseTierTokenCount(value) {
if (typeof value === "number") return Number.isFinite(value) && value > 0 ? Math.floor(value) : 0;
if (typeof value !== "string") return 0;
const m = value.trim().toUpperCase().match(/^(\d+(?:\.\d+)?)\s*([KM])?$/);
if (!m) return 0;
const n = Number(m[1]) * (UNIT[m[2]] || 1);
return Number.isFinite(n) && n > 0 ? Math.floor(n) : 0;
}
function tierName(entry, tokenCount) {
const raw = entry.name ?? entry.label ?? entry.display_name ?? entry.displayName ?? entry.key ?? entry.id;
if (typeof raw === "string" && raw.trim()) return raw.trim();
if (tokenCount >= UNIT.M && tokenCount % UNIT.M === 0) return `${tokenCount / UNIT.M}M`;
if (tokenCount >= UNIT.K && tokenCount % UNIT.K === 0) return `${tokenCount / UNIT.K}K`;
return String(tokenCount);
}
/**
* Normalize a model_config into sorted tiers: [{ name, tokenCount, isDefault }] ascending.
* Accepts snake_case and camelCase shapes; returns [] when the model has no tiers.
*/
export function getQoderContextTiers(modelConfig) {
const list = modelConfig?.context_config ?? modelConfig?.contextConfig;
if (!Array.isArray(list)) return [];
const byCount = new Map();
for (const entry of list) {
if (!entry || typeof entry !== "object") continue;
const tokenCount = parseTierTokenCount(
entry.tokenCount ?? entry.token_count ?? entry.max_input_tokens ?? entry.maxInputTokens ?? entry.contextLength ?? entry.context_length,
);
if (!tokenCount) continue;
const isDefault = entry.isDefault === true || entry.is_default === true || entry.default === true;
const prev = byCount.get(tokenCount);
byCount.set(tokenCount, {
name: tierName(entry, tokenCount),
tokenCount,
isDefault: (prev?.isDefault || false) || isDefault,
});
}
return [...byCount.values()].sort((a, b) => a.tokenCount - b.tokenCount);
}
const CJK_RE = /[\u1100-\u11ff\u2e80-\u9fff\uac00-\ud7af\uf900-\ufaff\uff00-\uffef]/g;
/**
* Rough prompt-size estimate in tokens. CJK characters count ~1 token each, everything
* else ~4 chars/token — the plain chars/4 rule underestimates Chinese/Japanese by up to
* 4x, which is exactly when a tier decision matters.
*/
export function estimateQoderPromptTokens({ system, messages, tools } = {}) {
let text = "";
try {
text = JSON.stringify({ system: system || "", messages: messages || [], tools: tools || [] }) || "";
} catch {
return 0;
}
const cjk = (text.match(CJK_RE) || []).length;
return Math.ceil(cjk + (text.length - cjk) / 4);
}
function normalizeMode(preference) {
const p = String(preference ?? "").trim();
return p ? p : QODER_CONTEXT_TIER_MODES.AUTO;
}
function findNamedTier(tiers, name) {
const wanted = name.replace(/\s+/g, "").toUpperCase();
const asCount = parseTierTokenCount(wanted);
return tiers.find((t) => t.name.replace(/\s+/g, "").toUpperCase() === wanted || (asCount && t.tokenCount === asCount)) || null;
}
/**
* Decide which tier a request should run under.
*
* @param {object} modelConfig raw Qoder model_config (has context_config + max_input_tokens)
* @param {{system?: string, messages?: any[], tools?: any[]}} prompt what will be sent
* @param {{preference?: string, headroom?: number}} [options]
* @returns {{ tier: {name, tokenCount, isDefault}, estimatedTokens: number, reason: string } | null}
* null → leave the payload exactly as before (no tiers, or the default already fits).
*/
export function resolveQoderContextTier(modelConfig, prompt, options = {}) {
const tiers = getQoderContextTiers(modelConfig);
if (!tiers.length) return null;
const mode = normalizeMode(options.preference);
const largest = tiers[tiers.length - 1];
const defaultTier = tiers.find((t) => t.isDefault) || tiers[0];
const estimatedTokens = estimateQoderPromptTokens(prompt);
const headroom = typeof options.headroom === "number" ? options.headroom : QODER_CONTEXT_TIER_HEADROOM;
const need = Math.ceil(estimatedTokens * (1 + headroom));
if (mode.toLowerCase() === QODER_CONTEXT_TIER_MODES.MAX) {
return { tier: largest, estimatedTokens, reason: "forced:max" };
}
if (mode.toLowerCase() === QODER_CONTEXT_TIER_MODES.DEFAULT) {
return { tier: defaultTier, estimatedTokens, reason: "forced:default" };
}
if (mode.toLowerCase() !== QODER_CONTEXT_TIER_MODES.AUTO) {
const named = findNamedTier(tiers, mode);
if (named) return { tier: named, estimatedTokens, reason: `forced:${named.name}` };
// Unknown tier name → fall through to auto rather than silently breaking requests.
}
// auto: keep the upstream default (current behaviour) while the prompt fits in it.
const currentMax = parseTierTokenCount(modelConfig?.max_input_tokens ?? modelConfig?.maxInputTokens);
const currentLimit = currentMax || defaultTier.tokenCount;
if (need <= currentLimit) return null;
const fits = tiers.find((t) => t.tokenCount >= need && t.tokenCount > currentLimit);
const tier = fits || largest;
if (tier.tokenCount <= currentLimit) return null; // nothing bigger to escalate to
return { tier, estimatedTokens, reason: fits ? "auto:fits" : "auto:largest" };
}
/**
* Write the chosen tier into a Qoder chat payload (mutates + returns it).
* Mirrors the IDE: parameters.context_length, ideModelConfigOverride, model_config.
*/
export function applyQoderContextTier(payload, tier) {
if (!payload || !tier?.tokenCount) return payload;
payload.parameters = { ...(payload.parameters || {}), context_length: tier.tokenCount };
payload.chat_context = payload.chat_context || {};
payload.chat_context.extra = {
...(payload.chat_context.extra || {}),
ideModelConfigOverride: {
...(payload.chat_context.extra?.ideModelConfigOverride || {}),
max_input_tokens: tier.tokenCount,
},
};
if (payload.model_config && typeof payload.model_config === "object") {
payload.model_config = { ...payload.model_config, max_input_tokens: tier.tokenCount };
}
return payload;
}

View File

@@ -0,0 +1,208 @@
/**
* Qoder SSE is OpenAI-shaped inside `{statusCodeValue, body}` envelopes, but
* usage arrives on a later `choices: []` frame — after finish_reason, which
* itself often lives on `delta.finish_reason` rather than the choice.
*
* Downstream (Claude translator, OpenAI clients, Claude Code) look for usage
* on the finish chunk or drop `choices: []` entirely. 9router's own dashboard
* still sees tokens because extractUsage runs on every forwarded frame.
*
* Coalesce: hold empty finish + usage-only frames, then emit one OpenAI
* include_usage-style chunk: `{choices:[{delta:{}, finish_reason}], usage}`.
*/
function num(v) {
const n = Number(v);
return Number.isFinite(n) ? n : null;
}
/**
* Normalize Qoder/OpenAI usage into the shape stream.js + Claude translation
* already understand (prompt_tokens + prompt_tokens_details.cached_tokens).
*/
export function canonicalizeQoderUsage(usage) {
if (!usage || typeof usage !== "object" || Array.isArray(usage)) return null;
const prompt = num(usage.prompt_tokens ?? usage.input_tokens);
const completion = num(usage.completion_tokens ?? usage.output_tokens);
if (prompt == null && completion == null) return null;
const details = (usage.prompt_tokens_details && typeof usage.prompt_tokens_details === "object")
? { ...usage.prompt_tokens_details }
: {};
const cached = num(
details.cached_tokens ??
usage.cached_tokens ??
usage.prompt_cache_hit_tokens ??
usage.cache_read_input_tokens,
);
const cacheCreation = num(
details.cache_creation_tokens ??
usage.cache_creation_input_tokens,
);
const promptTokens = prompt || 0;
const completionTokens = completion || 0;
const out = {
prompt_tokens: promptTokens,
completion_tokens: completionTokens,
total_tokens: num(usage.total_tokens) ?? (promptTokens + completionTokens),
};
if (cached != null) {
out.cached_tokens = cached;
details.cached_tokens = cached;
}
if (cacheCreation != null) {
details.cache_creation_tokens = cacheCreation;
}
if (Object.keys(details).length) out.prompt_tokens_details = details;
if (usage.completion_tokens_details && typeof usage.completion_tokens_details === "object") {
out.completion_tokens_details = usage.completion_tokens_details;
}
const reasoning = num(usage.reasoning_tokens ?? usage.completion_tokens_details?.reasoning_tokens);
if (reasoning != null) out.reasoning_tokens = reasoning;
return out;
}
function finishReasonOf(parsed) {
const choice = parsed?.choices?.[0];
return choice?.finish_reason || choice?.delta?.finish_reason || parsed?.finish_reason || null;
}
function hasValuableDelta(parsed) {
const delta = parsed?.choices?.[0]?.delta;
if (!delta || typeof delta !== "object") return false;
if (typeof delta.content === "string" && delta.content.length > 0) return true;
if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0) return true;
if (Array.isArray(delta.tool_calls) && delta.tool_calls.length > 0) return true;
if (delta.role) return true;
return false;
}
function parseInner(inner) {
if (inner == null || inner === "") return { raw: false, parsed: null };
if (inner === "[DONE]") return { done: true };
if (typeof inner !== "string") {
if (typeof inner === "object") return { parsed: inner };
return { raw: true, text: String(inner) };
}
try {
return { parsed: JSON.parse(inner) };
} catch {
return { raw: true, text: inner };
}
}
/**
* @param {object} opts
* @param {string} opts.model
* @param {TextEncoder} opts.encoder
* @param {string} opts.sseDone "data: [DONE]\\n\\n"
*/
export function createQoderSseCoalescer({ model, encoder, sseDone }) {
let pendingFinish = null;
let pendingUsage = null;
let lastMeta = { id: null, created: null, model };
let doneEmitted = false;
let finishAlreadyForwarded = false;
const emitJson = (controller, obj) => {
const sanitized = JSON.stringify(obj).replace(/\r?\n/g, "");
controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`));
};
const emitRaw = (controller, text) => {
controller.enqueue(encoder.encode(`data: ${String(text).replace(/\r?\n/g, "")}\n\n`));
};
const emitDone = (controller) => {
if (doneEmitted) return;
controller.enqueue(encoder.encode(sseDone));
doneEmitted = true;
};
const emitTerminal = (controller) => {
if (!pendingFinish && !pendingUsage) return;
emitJson(controller, {
id: lastMeta.id || `qoder-${Date.now()}`,
object: "chat.completion.chunk",
created: lastMeta.created || Math.floor(Date.now() / 1000),
model: lastMeta.model || model,
choices: [{ index: 0, delta: {}, finish_reason: pendingFinish || "stop" }],
...(pendingUsage ? { usage: pendingUsage } : {}),
});
pendingFinish = null;
pendingUsage = null;
};
const flush = (controller) => {
if (doneEmitted) return;
if (pendingUsage || (pendingFinish && !finishAlreadyForwarded)) {
emitTerminal(controller);
}
emitDone(controller);
};
const handleInner = (inner, controller) => {
if (doneEmitted) return { terminal: true };
const parsedInner = parseInner(inner);
if (parsedInner.done) {
flush(controller);
return { terminal: true };
}
if (parsedInner.raw) {
emitRaw(controller, parsedInner.text);
return {};
}
const parsed = parsedInner.parsed;
if (!parsed || typeof parsed !== "object") return {};
if (typeof parsed.id === "string" && parsed.id) lastMeta.id = parsed.id;
if (typeof parsed.created === "number") lastMeta.created = parsed.created;
if (typeof parsed.model === "string" && parsed.model) lastMeta.model = parsed.model;
const usage = canonicalizeQoderUsage(parsed.usage);
if (usage) pendingUsage = usage;
const finish = finishReasonOf(parsed);
if (hasValuableDelta(parsed)) {
// Stream content as-is (preserves upstream JSON for tests/clients).
emitRaw(controller, typeof inner === "string" ? inner : JSON.stringify(parsed));
if (finish) {
finishAlreadyForwarded = true;
// Keep finish around only if we still need a usage trailer.
pendingFinish = pendingUsage ? finish : null;
}
if (pendingFinish && pendingUsage) {
emitTerminal(controller);
emitDone(controller);
return { terminal: true };
}
return {};
}
if (finish) pendingFinish = finish;
// Empty finish and/or usage-only: emit as soon as we have both (Qoder
// order is finish then usage). Don't wait for the later [DONE]/keepalive.
if ((pendingFinish || finishAlreadyForwarded) && pendingUsage) {
if (!pendingFinish) pendingFinish = "stop";
emitTerminal(controller);
emitDone(controller);
return { terminal: true };
}
return {};
};
return {
handleInner,
flush,
get doneEmitted() {
return doneEmitted;
},
};
}

View File

@@ -14,6 +14,9 @@ const STRIP_RULES = [
{ provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] },
// Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926)
{ provider: "cloudflare-ai", flattenContent: true },
// MiMo Desktop Preview models (account-service route): content must be plain string,
// rejects OpenAI content-part array. Cloud models keep their parts (mimo-v2-omni is multi-modal).
{ provider: "xiaomi-mimo", match: /preview/i, flattenContent: true },
{ provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true },
// VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's
// advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144),

View File

@@ -1,5 +1,7 @@
// Tool call helper functions for translator
import { FORMATS } from "../formats.js";
// Anthropic tool_use.id must match: ^[a-zA-Z0-9_-]+$
const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/;
@@ -165,3 +167,16 @@ export function defaultClaudeToolType(tools) {
return tools.map(tool => tool?.type ? tool : { ...tool, type: "custom" });
}
// Whether Claude-format tools need explicit `type` defaulting before dispatch.
// Only gateways that declare the `requireClaudeToolType` quirk (MiniMax) reject typeless
// tools. Applying the default globally breaks Claude-format endpoints that only accept the
// legacy typeless tool shape — DeepSeek's Anthropic-compatible endpoint answers HTTP 400
// "unknown variant `custom`" and every Claude Code request routed there fails (#3905).
export function shouldDefaultClaudeToolType(provider, finalFormat, tools, PROVIDERS) {
return (
finalFormat === FORMATS.CLAUDE
&& Array.isArray(tools)
&& PROVIDERS?.[provider]?.quirks?.requireClaudeToolType === true
);
}

View File

@@ -27,6 +27,14 @@ export function lastCacheableToolIndex(tools) {
// Check if message has valid non-empty content
export function hasValidContent(msg) {
if (typeof msg.content === "string" && msg.content.trim()) return true;
if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) {
const block = msg.content;
return !!((block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
block.type === CLAUDE_BLOCK.TOOL_USE ||
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
block.type === CLAUDE_BLOCK.IMAGE ||
block.type === CLAUDE_BLOCK.DOCUMENT);
}
if (Array.isArray(msg.content)) {
return msg.content.some(block =>
(block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
@@ -38,6 +46,60 @@ export function hasValidContent(msg) {
}
return false;
}
// Content may arrive as a single content block object (spec allows string | array;
// some clients send the bare object). Wrap it as a one-block array and strip any
// client-placed cache_control: a bare-object marker must never survive
// normalization, on any path, guard or no guard.
function normalizeMessageContent(msg) {
const c = msg?.content;
if (c && typeof c === "object" && !Array.isArray(c)) {
delete c.cache_control;
msg.content = [c];
}
return msg;
}
// Total blocks carrying cache_control across system, tools, and messages — the
// upstream Messages API allows at most 4 markers per request.
function countCacheControlBlocks(body) {
let n = 0;
if (Array.isArray(body?.system)) for (const b of body.system) if (b?.cache_control) n++;
if (Array.isArray(body?.tools)) for (const t of body.tools) if (t?.cache_control) n++;
if (Array.isArray(body?.messages)) {
for (const m of body.messages) {
if (Array.isArray(m?.content)) {
for (const b of m.content) if (b?.cache_control) n++;
} else if (m?.content && typeof m.content === "object" && m.content.cache_control) n++;
}
}
return n;
}
// Trim every marker past the 4-marker budget. The head anchors (last system
// block, last cacheable tool) are held; the remaining slots go to the tail-most
// of the other markers in document order. A plain "keep the last 4 in document
// order" rule would drop the head anchors first — they lead document order, yet
// they are exactly what re-anchoring exists to pin.
function capCacheControlBlocks(body) {
const isHead = (b) => {
const sys = Array.isArray(body?.system) ? body.system : [];
if (sys.length && sys[sys.length - 1] === b) return true;
const tools = Array.isArray(body?.tools) ? body.tools : [];
const lastTool = lastCacheableToolIndex(tools);
return lastTool >= 0 && tools[lastTool] === b;
};
const marked = [];
if (Array.isArray(body?.system)) for (const b of body.system) if (b?.cache_control) marked.push(b);
if (Array.isArray(body?.tools)) for (const t of body.tools) if (t?.cache_control) marked.push(t);
if (Array.isArray(body?.messages)) {
for (const m of body.messages) {
if (Array.isArray(m?.content)) for (const b of m.content) if (b?.cache_control) marked.push(b);
}
}
const head = marked.filter(isHead);
const rest = marked.filter(b => !isHead(b));
const keep = Math.max(0, 4 - head.length);
for (const b of rest.slice(0, Math.max(0, rest.length - keep))) delete b.cache_control;
}
// Fix tool_use/tool_result ordering for Claude API
// 1. Assistant message with tool_use: remove text AFTER tool_use (Claude doesn't allow)
@@ -136,8 +198,9 @@ function hasForeignServerToolUseId(block) {
// Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject:
// 1. thinking.type "adaptive" → unsupported on Haiku
// 2. output_config.effort → unsupported on Haiku
// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed
// 4. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright
// 3. bare content-block objects (content: {block} instead of [{block}]) → wrapped first
// 4. role "system" messages (mid-conversation-system beta) → only top-level system is allowed
// 5. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright
export function normalizeClaudePassthrough(body, model = "") {
if (!body || typeof body !== "object") return body;
@@ -152,7 +215,15 @@ export function normalizeClaudePassthrough(body, model = "") {
if (Object.keys(body.output_config).length === 0) delete body.output_config;
}
// 2. Fold mid-conversation system messages into the neighbouring turn.
// 3. Wrap bare content-block objects as one-element arrays before folding.
// Some clients send content: {block} instead of content: [{block}]; the
// mid-conversation-system fold below assumes the array shape, so it must
// run first — a bare-object neighbor would otherwise be zeroed to [].
if (Array.isArray(body.messages)) {
for (const msg of body.messages) normalizeMessageContent(msg);
}
// 4. Fold mid-conversation system messages into the neighbouring turn.
// Hoisting them into body.system would insert volatile content (token counters,
// reminders) ahead of the whole conversation and invalidate the prefix cache on
// every request. Folding in place keeps the cached prefix stable.
@@ -186,7 +257,7 @@ export function normalizeClaudePassthrough(body, model = "") {
body.messages = messages;
}
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// 5. Drop thinking blocks whose signature is not Claude's (combo mixes models,
// so foreign signatures leak into history and Anthropic rejects them).
const thinkingEnabled = body.thinking?.type === "enabled";
const droppedServerToolUseIds = new Set();
@@ -233,7 +304,7 @@ export function normalizeClaudePassthrough(body, model = "") {
}
}
// 5. Drop empty text blocks and any message left with no content at all.
// 6. Drop empty text blocks and any message left with no content at all.
// Anthropic rejects `messages.N.content` blocks with empty text (400
// "text content blocks must be non-empty"); a message whose blocks were all
// stripped above must be dropped, not padded with an empty placeholder.
@@ -271,7 +342,22 @@ function markLastCacheableBlock(msg) {
// (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail.
export function anchorClaudeCache(body) {
if (!body || typeof body !== "object") return body;
if (Array.isArray(body.messages)) {
for (const msg of body.messages) normalizeMessageContent(msg);
}
// Invalid markers first, whatever the budget: Anthropic rejects a tool that
// carries BOTH defer_loading and cache_control (#3567). The re-anchor path
// below strips them anyway; the over-budget early return used to forward
// them untouched.
if (Array.isArray(body.tools)) {
for (const t of body.tools) {
if (t?.defer_loading === true) delete t.cache_control;
}
}
// Head anchors first, before any budget guard: the 1h TTL on system/tools is
// the point of re-anchoring, and skipping it because the client spent its
// budget would silently downgrade a cache hit to the 5m default.
if (Array.isArray(body.system)) {
const last = body.system.length - 1;
body.system.forEach((block, i) => {
@@ -289,6 +375,15 @@ export function anchorClaudeCache(body) {
});
}
// Budget guard AFTER the head anchors: with the last system block and last
// tool pinned, at most 2 slots remain. At >= 4 markers the client has spent
// the rest of the budget and every remaining marker is itself a valid
// breakpoint — re-anchoring the tail could only exceed 4, so trim instead.
if (countCacheControlBlocks(body) >= 4) {
capCacheControlBlocks(body);
return body;
}
if (Array.isArray(body.messages)) {
let anchored = null;
for (let i = body.messages.length - 1; i >= 0; i--) {
@@ -368,6 +463,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
// Pass 1: remove cache_control + filter empty messages
for (let i = 0; i < len; i++) {
const msg = body.messages[i];
normalizeMessageContent(msg);
// Remove cache_control from content blocks
if (Array.isArray(msg.content)) {
@@ -461,8 +557,21 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
// Strip built-in tools (e.g. web_search_20250305) and normalize to Anthropic-native shape
// (drop `type` field, fold `function.{name,description,parameters}`) for non-Anthropic providers
if (provider !== "claude") {
// Provider-specific whitelist of Anthropic tool `type` values that the
// upstream actually accepts. When the provider declares it
// (e.g. DeepSeek — only web_search_*), keep only listed types; otherwise
// keep the prior behaviour of dropping every non-function tool, which is
// correct for OpenAI-compatible targets reached through this Claude-format
// pass (their tools get normalized below to function-style).
const supportedTypes = PROVIDERS[provider]?.quirks?.claudeSupportedToolTypes;
const hasWhitelist = Array.isArray(supportedTypes);
body.tools = body.tools
.filter(tool => !tool.type || tool.type === "function")
.filter(tool => {
const t = tool?.type;
if (!t || t === "function") return true;
if (hasWhitelist) return supportedTypes.includes(t);
return false;
})
.map(tool => {
if (tool.function) {
return {
@@ -471,6 +580,13 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
input_schema: tool.function.parameters,
};
}
// When the provider declared a supportedToolTypes whitelist, keep
// the surviving tools' `type` field intact — the upstream
// Anthropic-compatible endpoint (e.g. DeepSeek) requires it to
// route built-ins like web_search_* correctly. Without a
// whitelist, preserve prior behaviour and strip `type` so the
// tool is normalized to plain Anthropic shape.
if (hasWhitelist) return tool;
const { type, ...rest } = tool;
return rest;
});

View File

@@ -432,3 +432,21 @@ export function cleanJSONSchemaForAntigravity(schema) {
return cleaned;
}
// Merge adjacent same-role messages, strip empty parts, ensure initial user turn
export function normalizeGeminiContents(contents) {
const out = [];
for (const c of contents || []) {
if (!c?.role || !Array.isArray(c.parts)) continue;
const parts = c.parts.filter(p => p && Object.keys(p).length > 0);
if (parts.length === 0) continue;
const last = out.at(-1);
if (last?.role === c.role) last.parts.push(...parts);
else out.push({ ...c, parts: [...parts] });
}
if (out.length > 0 && out[0].role !== "user") {
out.unshift({ role: "user", parts: [{ text: "..." }] });
}
return out;
}

View File

@@ -242,9 +242,9 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
? (credentials?.providerSpecificData?.profileArn || "")
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));
// Kiro CLI/KAS sends system prompt as top-level `systemPrompt`. Keep a
// content fallback too because the CodeWhisperer surface does not always
// enforce top-level systemPrompt for direct calls.
// The system prompt travels inside the first user turn's content (contentPrefix):
// the CodeWhisperer surface rejects a top-level `systemPrompt` with
// 400 REQUEST_BODY_INVALID, so the value below is only a replay cache key.
const timestamp = new Date().toISOString();
const systemPromptParts = [];
if (thinkingBudget !== null && !usesNativeGptEffort) {
@@ -316,14 +316,11 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
conversationState: {
chatTriggerType: "MANUAL",
conversationId,
agentContinuationId: continuationId,
agentTaskType: "vibe",
currentMessage: {
userInputMessage,
},
history: canonical.history,
},
agentMode: "vibe",
};
if (profileArn) payload.profileArn = profileArn;

View File

@@ -142,6 +142,13 @@ function systemReminderText(content) {
// Convert single Claude message - returns single message or array of messages
function convertClaudeMessage(msg) {
// Some clients send content as a single block object; normalize to the
// one-element array every branch below (the system-reminder fold included)
// expects. Must run BEFORE the role branch: systemReminderText only reads
// arrays and strings, so a bare-object system turn was dropped outright.
if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) {
msg.content = [msg.content];
}
// Mid-conversation system message -> user (per Anthropic placement rules)
if (msg.role === ROLE.SYSTEM) {
const text = systemReminderText(msg.content);

View File

@@ -15,7 +15,8 @@ import {
generateRequestId,
generateSessionId,
generateProjectId,
cleanJSONSchemaForAntigravity
cleanJSONSchemaForAntigravity,
normalizeGeminiContents
} from "../formats/gemini.js";
import { deriveSessionId, toNumericSessionId } from "../../utils/sessionManager.js";
import { ROLE, GEMINI_ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js";
@@ -35,17 +36,6 @@ function sanitizeGeminiFunctionName(name) {
return sanitized.substring(0, 64);
}
function normalizeGeminiContents(contents) {
const out = [];
for (const c of contents || []) {
if (!c?.role || !Array.isArray(c.parts) || c.parts.length === 0) continue;
const last = out.at(-1);
if (last?.role === c.role) last.parts.push(...c.parts);
else out.push({ ...c, parts: [...c.parts] });
}
return out;
}
// Core: Convert OpenAI request to Gemini format (base for all variants)
function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) {
const result = {
@@ -163,12 +153,14 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
}
// Check if there are actual tool responses in the next messages
const hasActualResponses = toolCallIds.some(fid => toolResponses[fid]);
const isIntermediate = i < body.messages.length - 1;
const hasActualResponses = toolCallIds.some(fid => toolResponses[fid] !== undefined);
if (hasActualResponses) {
if (hasActualResponses || isIntermediate) {
const toolParts = [];
for (const fid of toolCallIds) {
if (!toolResponses[fid]) continue;
let resp = toolResponses[fid];
if (resp === undefined) resp = "";
let name = tcID2Name[fid];
if (!name) {
@@ -180,7 +172,6 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG
}
}
let resp = toolResponses[fid];
let parsedResp = tryParseJSON(resp);
if (parsedResp === null) {
parsedResp = { result: resp };

View File

@@ -340,9 +340,9 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
const timestamp = new Date().toISOString();
// Kiro CLI/KAS sends these as top-level systemPrompt. Keep a content fallback
// too because the CodeWhisperer surface does not always enforce top-level
// systemPrompt for direct calls.
// The system prompt travels inside the first user turn's content (contentPrefix):
// the CodeWhisperer surface rejects a top-level `systemPrompt` with
// 400 REQUEST_BODY_INVALID, so the value below is only a replay cache key.
const systemPromptParts = [];
if (thinkingBudget !== null && !usesNativeGptEffort) {
systemPromptParts.push(buildThinkingSystemPrefix(thinkingBudget));
@@ -397,8 +397,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
conversationState: {
chatTriggerType: "MANUAL",
conversationId,
agentContinuationId: continuationId,
agentTaskType: "vibe",
currentMessage: {
userInputMessage: {
content: replayCurrent.content || "",
@@ -414,7 +412,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
},
history: canonical.history
},
agentMode: "vibe",
};
if (profileArn) {

View File

@@ -0,0 +1,80 @@
// Codex-specific tool JSON Schema compatibility.
//
// `https://chatgpt.com/backend-api/codex/responses` validates every function
// tool's `parameters` with a regex engine that does not implement Unicode
// property escapes. A `pattern` such as
//
// "^(?!__.*__$)[^\\p{Cc}\\p{Cf}\\p{Zl}\\p{Zp}\"\\\\./\\[\\]]{1,200}$"
//
// is a perfectly valid ECMAScript `u`-mode regex, but Codex answers
//
// 400 Invalid schema for function 'Artifact': '^\p{Cc}...' is not a 'regex'
// param: tools[0].parameters
//
// The request is deterministically malformed for this provider, so every
// account fails identically and the combo pays a full failover before landing
// somewhere that accepts it (#3922).
//
// Scope guardrail (#3667): this is NOT a global schema sanitizer. Providers
// that do support `\p{...}` keep the constraint untouched — the strip runs only
// on the Codex dispatch path, and only on `pattern` strings that actually
// contain a property escape. Everything else in the schema (including valid
// patterns) passes through byte-identical.
// `\p{...}` / `\P{...}` with an odd number of preceding backslashes — an even
// count means the backslash itself is escaped, so `\\p{Cc}` is a literal "p".
const UNICODE_PROPERTY_ESCAPE = /(^|[^\\])(\\\\)*\\[pP]\{/;
export function hasUnicodePropertyEscape(pattern) {
return typeof pattern === "string" && UNICODE_PROPERTY_ESCAPE.test(pattern);
}
// Copy-on-write walk: returns the original reference when nothing changed, so
// untouched schemas keep object identity and callers can cheaply detect a no-op.
// `properties` is special-cased because its keys are arbitrary property *names*
// (which may themselves be "pattern" or "properties") and must never be read as
// schema keywords; every other key recurses as an ordinary schema node.
function stripNode(node, stats) {
if (Array.isArray(node)) {
let changed = false;
const next = node.map((item) => {
const cleaned = stripNode(item, stats);
if (cleaned !== item) changed = true;
return cleaned;
});
return changed ? next : node;
}
if (!node || typeof node !== "object") return node;
let changed = false;
const next = {};
for (const [key, value] of Object.entries(node)) {
if (key === "pattern" && hasUnicodePropertyEscape(value)) {
stats.removed++;
changed = true;
continue;
}
if (key === "properties" && value && typeof value === "object" && !Array.isArray(value)) {
let propsChanged = false;
const props = {};
for (const [propName, propSchema] of Object.entries(value)) {
const cleaned = stripNode(propSchema, stats);
if (cleaned !== propSchema) propsChanged = true;
props[propName] = cleaned;
}
if (propsChanged) changed = true;
next[key] = propsChanged ? props : value;
continue;
}
const cleaned = stripNode(value, stats);
if (cleaned !== value) changed = true;
next[key] = cleaned;
}
return changed ? next : node;
}
// Remove only the `pattern` constraints Codex's validator rejects.
// Returns the same reference when the schema is already compatible.
export function stripCodexUnsupportedPatterns(schema, stats = { removed: 0 }) {
return stripNode(schema, stats);
}