Merge origin/master (v0.5.95) into gitea/new_feature

This commit is contained in:
luulam committed 2026-10-04 20:36:44 +07:00
commit 3658e6eeae
137 files changed
+6673 -355

No files matched your search

+31
View File
@@ -1,6 +1,37 @@
# v0.5.95 (2026-10-01)
## Features
- **Providers**: add Meta Muse provider with OAuth login and model catalog; add v1m System One provider
- **GLM**: add Z.ai OAuth login to GLM Coding (dual-auth)
- **Codex**: add GPT-6.1 Sol; expose 1M context variants for GPT-6 and GPT-5.6; add gpt-daybreak/reserve models and route bare `gpt-5.x`/`gpt-6.x` slugs to codex
- **Claude**: add Claude Sonnet 5.5 (plus `claude-opus-5.5` models in the Kiro registry)
- **CLI**: add `connect` command for remote 9Router servers
- **Providers**: per-provider custom header overrides from the registry
- **Agnes**: seed the 2.5/3.0 model ids in the registry
- **Usage**: sync `?provider=` URL param with provider filter for bookmarkable deep links (#4395)
- **Dashboard**: drop NEW badges in sidebar, mark 9Remote as HOT
## Fixes
- **Claude**: preserve intentional prefill from non-messages[] source formats; keep a trailing user turn so cleanup never yields assistant prefill
- **Claude**: cache a tool loop's final tool results with the 4th breakpoint
- **Claude**: resolve Sonnet 5.x to adaptive thinking so no forged thinking placeholders are sent; inject unsigned thinking placeholders for opencode-go DeepSeek `/messages` (#4436)
- **Thinking**: add `xhigh` to claude-adaptive thinking levels
- **Claude**: keep a user turn whose only block is `container_upload`
- **Capabilities**: publish real GPT-6/GPT-5.4+ context windows and combo token limits
- **Responses**: wait for real usage before emitting `response.completed`, bounded by a 3s watchdog
- **Codex**: stop refresh-token reuse that logs accounts out on auto-ping; preserve hosted web search on GPT-6 Sol/Luna; remove ghost models
- **Grok CLI**: send Grok CLI 1.0.44 so proxy stops returning HTTP 426
- **Proxy**: auto-fallback to insecure TLS on self-signed cert errors; hold strictProxy when no proxy resolves
- **Translator**: strip `errorMessage` and other non-standard schema keywords from Gemini tool schemas; dedupe same-name tools for DeepSeek models (#3333)
- **Codebuddy**: parse the 6004 rate limit error and extract `resetsAtMs`; forward `recurring` for codebuddy-intl quota packs (#4422)
- **CLI Tools**: replace `sk_9router` placeholder with first active dashboard API key
- **Dashboard**: exclude hidden providers from usage stats provider list
- **Capabilities**: add deepseek-v4-1-flash vision alias; add zed to live catalog providers
# v0.5.91 (2026-09-26)
## Features
- **Web Search & Fetch**: add TinyFish Search and Fetch with one API-key connection, normalized results, and official provider icon
- **Providers**: add Token Harbor provider and four OpenAI-compatible aggregator providers (dahl, atria, agnes, bai)
- **Claude**: forward `x-claude-code-session-id` on OAuth requests; merge client `anthropic-beta` flags and forward rate-limit headers; return thinking text to OpenAI-format clients
- **Codex**: add GPT-6 Sol and Luna support
+18
View File
@@ -90,6 +90,24 @@ That's it! Start coding with FREE AI models.
---
## 🔌 Connect to a Remote 9Router
Already running 9Router on another machine (e.g. a team server on your LAN)? Point this machine's CLI tools at it — no local server is started:
```bash
npx 9router connect http://<server-host>:20128 # pick tools interactively
npx 9router connect http://<server-host>:20128 --tools claude,codex # or choose up front
npx 9router connect --reset --tools claude,codex # undo
```
It logs in with the dashboard password (hidden prompt), reuses or creates an API key named `cli-<hostname>`, and writes each tool's config (backing up the original once as `*.bak-9router`).
Supported: `claude`, `codex`, `opencode`, `droid`, `crush`, `kilo`, `cline`, or `all`. Other options: `--model`, `--opus/--sonnet/--haiku/--fable`, `--api-key`, `--key-name`, `--print-env`. See `9router connect --help`.
> ⚠️ Over plain `http://` the password and API key are sent unencrypted — use a trusted LAN/VPN or put HTTPS in front. The API key is stored in each tool's config file.
---
## 🛠️ Supported CLI Tools
Claude-Code • OpenClaw • Codex • OpenCode • Cursor • Antigravity • Cline • Continue • Droid • Roo • Copilot • Kilo Code • Gemini CLI • Qwen Code • iFlow • Crush • Crusher • Aider
+15
View File
@@ -80,6 +80,19 @@ if (args[0] === "xai" && args[1] === "video") {
return;
}
// `9router connect <url>` configures local CLI tools against a remote server —
// no local server, no runtime deps. Usable via `npx 9router connect …`.
if (args[0] === "connect") {
const { run } = require("./src/cli/commands/connect");
run(args.slice(1))
.then((code) => process.exit(code))
.catch((err) => {
console.error(`❌ ${err?.message || err}`);
process.exit(1);
});
return;
}
// Self-heal SQLite runtime deps (sql.js + better-sqlite3) into ~/.9router/runtime
// so the server can resolve them via NODE_PATH. Best-effort — sql.js is required,
// better-sqlite3 is optional. Logs to stderr only on failure.
@@ -154,6 +167,8 @@ Options:
-v, --version Show version
Commands:
connect <server-url> Configure Claude Code for a remote 9router server
(npx 9router connect http://host:20128 — no install needed)
xai video --prompt "..." --output video.mp4
Generate a Grok Imagine video via the running gateway
(see: ${APP_NAME} xai video --help)
+4
View File
@@ -3,6 +3,10 @@
// Postinstall: warm-up SQLite deps into ~/.9router/runtime so the first
// `9router` start doesn't need network. Failure here is non-fatal —
// cli.js will retry at runtime if anything is missing.
// `npx 9router …` (npm_command=exec) is typically a one-shot `connect` — skip
// the runtime warm-up; cli.js self-heals it if the server is started later.
if (process.env.npm_command === "exec") process.exit(0);
const { ensureSqliteRuntime } = require("./sqliteRuntime");
const { ensureTrayRuntime } = require("./trayRuntime");
+3 -2
View File
@@ -1,6 +1,6 @@
{
"name": "9router",
"version": "0.5.91",
"version": "0.5.95",
"description": "9Router CLI - Start and manage 9Router server",
"bin": {
"9router": "./cli.js"
@@ -27,7 +27,8 @@
"node-forge": "^1.3.3",
"node-machine-id": "^1.1.12",
"react": "19.2.1",
"react-dom": "19.2.1"
"react-dom": "19.2.1",
"confbox": "^0.2.4"
},
"comment_sqlite": "sql.js + better-sqlite3 are NOT bundled here. They are installed into ~/.9router/runtime/node_modules by hooks/postinstall.js (and re-checked at runtime by cli.js). This avoids Windows EBUSY errors when updating the global CLI, since native .node files no longer live under the locked install dir.",
"comment_systray": "systray2 is NOT bundled here. It is lazy-installed into ~/.9router/runtime/node_modules by hooks/postinstall.js on macOS/Linux only. Windows uses PowerShell NotifyIcon (zero binary). This avoids shipping unsigned Go binaries that trigger antivirus false positives (Kaspersky). We use the systray2 fork because the legacy systray@1.0.5 ships a 2017 x86_64 binary that fails to load on modern macOS dyld. Neither package ships an arm64 macOS binary, so on Apple Silicon hooks/trayRuntime.js overlays our own arm64 build from the tray-binaries GitHub release; without it the tray requires Rosetta 2.",
+269
View File
@@ -0,0 +1,269 @@
/**
* `9router connect <server-url>` — point local CLI tools (Claude Code, Codex, …) at a
* REMOTE 9router server. Nothing runs locally: we log in with the dashboard
* password, reuse/create an API key for this machine, then write the tool's
* settings files (see connectTools.js). Works via `npx 9router connect …` with no global install.
*
* The password and API key are never printed (key is masked).
*/
const os = require("os");
const { TOOL_IDS, CLAUDE_MODELS, resolveTools } = require("./connectTools");
const DEFAULT_MODEL = "cc/claude-sonnet-5";
const HELP = `
Usage: 9router connect <server-url> [options]
Configure CLI tools on THIS machine to use a remote 9router server.
No local server is started. Run without installing:
npx 9router connect http://<server-host>:20128
npx 9router connect http://<server-host>:20128 --tools claude,codex,opencode
Options:
--tools <list> Comma-separated tools to configure (prompted if omitted
in a terminal; default: claude). Supported:
${TOOL_IDS.join(", ")}, all
--password <pw> Dashboard password (or env NINE_ROUTER_PASSWORD;
prompted if omitted — preferred, keeps it out of shell history)
--key-name <name> API key name to reuse/create (default: cli-<hostname>)
--api-key <key> Use this API key, skip login + key lookup
--model <model> Model for non-Claude tools (default: ${DEFAULT_MODEL})
--fable|--opus|--sonnet|--haiku <model>
Override Claude Code model mapping
--print-env Also print OpenAI-compatible env vars for other CLIs
--reset Remove 9router settings from the selected tools and exit
-h, --help Show this help
`;
function parseArgs(argv) {
const opts = {
password: process.env.NINE_ROUTER_PASSWORD || null,
keyName: `cli-${os.hostname()}`.slice(0, 64),
apiKey: null,
models: {},
};
for (let i = 0; i < argv.length; i++) {
const a = argv[i];
const next = () => {
const v = argv[++i];
if (v === undefined) throw new Error(`Missing value for ${a}`);
return v;
};
if (a === "--password") opts.password = next();
else if (a === "--key-name") opts.keyName = next();
else if (a === "--api-key") opts.apiKey = next();
else if (a === "--tools") opts.tools = next().split(",");
else if (a === "--model") opts.model = next();
else if (a === "--print-env") opts.printEnv = true;
else if (a === "--reset") opts.reset = true;
else if (a === "-h" || a === "--help") opts.help = true;
else if (a.startsWith("--") && CLAUDE_MODELS.some((m) => `--${m.flag}` === a)) opts.models[a.slice(2)] = next();
else if (!a.startsWith("-") && !opts.url) opts.url = a;
else throw new Error(`Unknown option: ${a}`);
}
return opts;
}
function normalizeServerUrl(input) {
let raw = String(input || "").trim();
if (!/^https?:\/\//i.test(raw)) raw = `http://${raw}`;
const u = new URL(raw);
// Accept pasted dashboard/API URLs: keep only origin.
return u.origin;
}
function maskKey(key) {
if (!key || key.length < 12) return "****";
return `${key.slice(0, 6)}…${key.slice(-4)}`;
}
// Enquirer rejects with an empty value on Ctrl+C / Esc.
class Cancelled extends Error {}
const prompt = (p) => p.run().catch((err) => { throw err || new Cancelled("Cancelled"); });
async function promptPassword() {
if (!process.stdin.isTTY) throw new Error("Password required: pass --password or set NINE_ROUTER_PASSWORD");
const { Password } = require("enquirer");
return prompt(new Password({ message: "9router dashboard password" }));
}
async function request(url, { method = "GET", body, cookie, apiKey } = {}) {
const headers = { Accept: "application/json" };
if (body) headers["Content-Type"] = "application/json";
if (cookie) headers.Cookie = cookie;
if (apiKey) headers.Authorization = `Bearer ${apiKey}`;
let res;
try {
res = await fetch(url, { method, headers, body: body ? JSON.stringify(body) : undefined, redirect: "manual" });
} catch (err) {
throw new Error(`Cannot reach ${new URL(url).origin}: ${err.cause?.code || err.message}`);
}
let data = null;
try { data = await res.json(); } catch { /* non-JSON */ }
// Server-controlled strings get printed later — strip control chars (terminal escape injection).
if (typeof data?.error === "string") data.error = data.error.replace(/[\x00-\x1f\x7f]/g, "");
return { status: res.status, headers: res.headers, data };
}
function extractAuthCookie(headers) {
const list = typeof headers.getSetCookie === "function" ? headers.getSetCookie() : [headers.get("set-cookie") || ""];
for (const c of list) {
const m = /(?:^|,\s*)auth_token=([^;]+)/.exec(c);
if (m) return `auth_token=${m[1]}`;
}
return null;
}
async function login(server, password) {
const res = await request(`${server}/api/auth/login`, { method: "POST", body: { password } });
if (res.status === 200 && res.data?.success) {
const cookie = extractAuthCookie(res.headers);
if (!cookie) throw new Error("Login succeeded but server returned no session cookie");
return cookie;
}
throw new Error(`Login failed (${res.status}): ${res.data?.error || "unknown error"}`);
}
async function getOrCreateApiKey(server, cookie, keyName) {
const list = await request(`${server}/api/keys`, { cookie });
if (list.status === 401) throw new Error("Unauthorized listing API keys — wrong password or session rejected");
if (list.status !== 200) throw new Error(`Failed to list API keys (${list.status}): ${list.data?.error || ""}`);
const keys = (list.data?.keys || []).filter((k) => k.isActive !== false);
const existing = keys.find((k) => k.name === keyName);
if (existing) return { key: existing.key, created: false };
const created = await request(`${server}/api/keys`, { method: "POST", cookie, body: { name: keyName } });
if (created.status !== 201 || !created.data?.key) {
throw new Error(`Failed to create API key (${created.status}): ${created.data?.error || ""}`);
}
return { key: created.data.key, created: true };
}
async function listModels(server, apiKey) {
const res = await request(`${server}/v1/models`, { apiKey });
if (res.status === 401) throw new Error("API key rejected by server (/v1/models returned 401)");
if (res.status !== 200) return null;
return new Set((res.data?.data || []).map((m) => m.id));
}
async function promptTools() {
const { MultiSelect } = require("enquirer");
const { TOOLS } = require("./connectTools");
return prompt(new MultiSelect({
message: "Select CLI tools to configure (space to toggle, enter to confirm)",
choices: TOOLS.map((t) => ({ name: t.id, message: t.name, hint: t.paths()[0], enabled: t.id === "claude" })),
validate: (v) => v.length > 0 || "Select at least one tool",
}));
}
async function selectTools(opts) {
if (opts.tools) return resolveTools(opts.tools);
if (process.stdin.isTTY) return resolveTools(await promptTools());
return resolveTools(["claude"]);
}
async function run(argv) {
try {
return await runConnect(argv);
} catch (err) {
if (err instanceof Cancelled) {
console.log("Cancelled.");
return 130;
}
throw err;
}
}
async function runConnect(argv) {
const opts = parseArgs(argv);
if (opts.help) {
console.log(HELP);
return 0;
}
if (!opts.reset && !opts.url) {
console.log(HELP);
return 1;
}
// Pick tools before any network call: bad names fail fast, and cancelling
// the picker never leaves a freshly created key on the server.
const tools = await selectTools(opts);
if (opts.reset) {
let failed = 0;
for (const t of tools) {
try {
const files = await t.reset();
console.log(files.length ? `✅ ${t.name}: removed 9router settings (${files.join(", ")})` : `• ${t.name}: nothing to reset`);
} catch (err) {
failed++;
console.log(`❌ ${t.name}: ${err.message}`);
}
}
return failed ? 1 : 0;
}
const server = normalizeServerUrl(opts.url);
const isRemoteHttp = server.startsWith("http://") && !/^http:\/\/(localhost|127\.0\.0\.1|\[::1\])(:|$)/.test(server);
if (isRemoteHttp) {
console.log("\x1b[33m⚠ Plain HTTP: password and API key travel unencrypted. Use only on a trusted LAN/VPN.\x1b[0m");
}
let apiKey = opts.apiKey;
if (apiKey) {
console.log("• Using provided API key");
} else {
const password = opts.password ?? (await promptPassword());
console.log(`• Logging in to ${server}`);
const cookie = await login(server, password);
const result = await getOrCreateApiKey(server, cookie, opts.keyName);
apiKey = result.key;
console.log(`• ${result.created ? "Created" : "Reusing"} API key "${opts.keyName}" (${maskKey(apiKey)})`);
}
const available = await listModels(server, apiKey);
const warnMissing = (label, model, flag) => {
if (available && !available.has(model)) {
console.log(`\x1b[33m⚠ ${label}: "${model}" not listed by server — override with ${flag} <model>\x1b[0m`);
}
};
const claudeModels = {};
if (tools.some((t) => t.id === "claude")) {
for (const m of CLAUDE_MODELS) {
claudeModels[m.envKey] = opts.models[m.flag] || m.defaultValue;
warnMissing(`claude ${m.flag}`, claudeModels[m.envKey], `--${m.flag}`);
}
}
const model = opts.model || DEFAULT_MODEL;
if (tools.some((t) => t.id !== "claude")) warnMissing("model", model, "--model");
const ctx = { baseUrl: server, apiKey, model, claudeModels };
let failed = 0;
for (const t of tools) {
try {
const files = await t.apply(ctx);
console.log(`✅ ${t.name} → ${files.join(", ")}`);
} catch (err) {
failed++;
console.log(`❌ ${t.name}: ${err.message}`);
}
}
console.log(` Base URL: ${server}/v1`);
if (tools.some((t) => t.id === "claude")) {
for (const m of CLAUDE_MODELS) console.log(` ${m.envKey}=${claudeModels[m.envKey]}`);
}
if (tools.some((t) => t.id !== "claude")) console.log(` Model (other tools): ${model}`);
console.log(` Restart the tools to apply. Undo: npx 9router connect --reset --tools ${tools.map((t) => t.id).join(",")}`);
if (opts.printEnv) {
console.log("\nOpenAI-compatible CLIs (codex, opencode, aider, …):");
console.log(` OPENAI_BASE_URL=${server}/v1`);
console.log(` OPENAI_API_KEY=${apiKey}`);
}
return failed ? 1 : 0;
}
module.exports = { run, __test__: { parseArgs, normalizeServerUrl, extractAuthCookie, maskKey, Cancelled } };
+338
View File
@@ -0,0 +1,338 @@
/**
* Per-tool config writers for `9router connect`. File layout and keys mirror
* the dashboard routes in src/app/api/cli-tools/<tool>-settings/route.js so a
* tool configured here looks identical to one applied from the dashboard.
*
* Each tool: { id, name, paths(), apply(ctx) → string[] written, reset() → string[] touched }
* ctx: { baseUrl (origin, no /v1), apiKey, model, claudeModels: { envKey: model } }
*/
const fs = require("fs");
const path = require("path");
const os = require("os");
const home = () => os.homedir();
const v1 = (baseUrl) => (baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`);
// Drop trailing commas (JSONC) outside string literals, so values like "a,}" survive.
function stripTrailingCommas(text) {
return text.replace(/("(?:\\.|[^"\\])*")|,(\s*[}\]])/g, (m, str, tail) => str ?? tail);
}
function readJson(file) {
try {
return JSON.parse(stripTrailingCommas(fs.readFileSync(file, "utf8")));
} catch (err) {
if (err.code === "ENOENT") return null;
throw new Error(`Cannot parse ${file}: ${err.message}`);
}
}
// Files hold the API key → owner-only (0600) on POSIX; no-op on Windows.
const SECRET_MODE = 0o600;
// Rewrite an existing file on reset (no backup, existing mode kept).
function rewriteFile(file, content) {
fs.writeFileSync(file, content, { mode: SECRET_MODE });
}
// One-time backup of the user's pre-9router file, then write.
function writeFile(file, content) {
fs.mkdirSync(path.dirname(file), { recursive: true });
const backup = `${file}.bak-9router`;
if (fs.existsSync(file) && !fs.existsSync(backup)) {
fs.copyFileSync(file, backup);
fs.chmodSync(backup, SECRET_MODE);
}
fs.writeFileSync(file, content, { mode: SECRET_MODE });
fs.chmodSync(file, SECRET_MODE); // mode above only applies when creating
}
const writeJson = (file, data) => writeFile(file, JSON.stringify(data, null, 2));
// confbox is ESM-only; loaded lazily so non-codex runs don't need it.
async function toml() {
return import("confbox");
}
// ── Claude Code ─────────────────────────────────────────────────────────────
const CLAUDE_MODELS = [
{ flag: "fable", envKey: "ANTHROPIC_DEFAULT_FABLE_MODEL", defaultValue: "cc/claude-fable-5" },
{ flag: "opus", envKey: "ANTHROPIC_DEFAULT_OPUS_MODEL", defaultValue: "cc/claude-opus-5" },
{ flag: "sonnet", envKey: "ANTHROPIC_DEFAULT_SONNET_MODEL", defaultValue: "cc/claude-sonnet-5" },
{ flag: "haiku", envKey: "ANTHROPIC_DEFAULT_HAIKU_MODEL", defaultValue: "cc/claude-haiku-4-5-20251001" },
];
const CLAUDE_RESET_KEYS = ["ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN", ...CLAUDE_MODELS.map((m) => m.envKey)];
const claudePath = () => path.join(home(), ".claude", "settings.json");
const claude = {
id: "claude",
name: "Claude Code",
paths: () => [claudePath()],
async apply({ baseUrl, apiKey, claudeModels }) {
const file = claudePath();
const cur = readJson(file) || {};
cur.hasCompletedOnboarding = true;
cur.env = { ...(cur.env || {}), ANTHROPIC_BASE_URL: v1(baseUrl), ANTHROPIC_AUTH_TOKEN: apiKey, ...claudeModels };
writeJson(file, cur);
return [file];
},
async reset() {
const file = claudePath();
const cur = readJson(file);
if (!cur) return [];
if (cur.env) {
CLAUDE_RESET_KEYS.forEach((k) => delete cur.env[k]);
if (Object.keys(cur.env).length === 0) delete cur.env;
}
rewriteFile(file, JSON.stringify(cur, null, 2));
return [file];
},
};
// ── OpenAI Codex CLI ────────────────────────────────────────────────────────
const codexPath = () => path.join(home(), ".codex", "config.toml");
const codex = {
id: "codex",
name: "OpenAI Codex CLI",
paths: () => [codexPath()],
async apply({ baseUrl, apiKey, model }) {
const { parseTOML, stringifyTOML } = await toml();
const file = codexPath();
let cfg = {};
try { cfg = parseTOML(fs.readFileSync(file, "utf8")) || {}; } catch (err) { if (err.code !== "ENOENT") throw err; }
cfg.model = model;
cfg.model_provider = "9router";
cfg.model_providers = cfg.model_providers || {};
// Custom providers ignore auth.json — key must travel as a static header.
cfg.model_providers["9router"] = {
name: "9Router",
base_url: v1(baseUrl),
wire_api: "responses",
http_headers: { Authorization: `Bearer ${apiKey}` },
};
cfg.agents = cfg.agents || {};
delete cfg.agents.subagent;
cfg.agents.default_subagent_model = model;
writeFile(file, stringifyTOML(cfg));
return [file];
},
async reset() {
const { parseTOML, stringifyTOML } = await toml();
const file = codexPath();
let cfg;
try { cfg = parseTOML(fs.readFileSync(file, "utf8")) || {}; } catch (err) { if (err.code === "ENOENT") return []; throw err; }
if (cfg.model_provider === "9router") { delete cfg.model; delete cfg.model_provider; }
if (cfg.model_providers) delete cfg.model_providers["9router"];
if (cfg.agents) { delete cfg.agents.default_subagent_model; delete cfg.agents.subagent; }
for (const k of ["model_providers", "agents"]) {
if (cfg[k] && Object.keys(cfg[k]).length === 0) delete cfg[k];
}
rewriteFile(file, stringifyTOML(cfg));
return [file];
},
};
// ── OpenCode ────────────────────────────────────────────────────────────────
const opencodePath = () => path.join(home(), ".config", "opencode", "opencode.json");
const opencode = {
id: "opencode",
name: "OpenCode",
paths: () => [opencodePath()],
async apply({ baseUrl, apiKey, model }) {
const file = opencodePath();
const cfg = readJson(file) || {};
cfg.provider = cfg.provider || {};
const p = cfg.provider["9router"] || { npm: "@ai-sdk/openai-compatible", options: {}, models: {} };
p.options = { ...p.options, baseURL: v1(baseUrl), apiKey };
p.models = p.models || {};
p.models[model] = { name: model, modalities: { input: ["text", "image"], output: ["text"] } };
cfg.provider["9router"] = p;
cfg.model = `9router/${model}`;
cfg.agent = cfg.agent || {};
cfg.agent.explorer = {
description: "Fast explorer subagent for codebase exploration",
mode: "subagent",
model: `9router/${model}`,
};
writeJson(file, cfg);
return [file];
},
async reset() {
const file = opencodePath();
const cfg = readJson(file);
if (!cfg) return [];
if (cfg.provider) delete cfg.provider["9router"];
if (cfg.model?.startsWith("9router/")) delete cfg.model;
if (cfg.agent?.explorer?.model?.startsWith("9router/")) {
delete cfg.agent.explorer;
if (Object.keys(cfg.agent).length === 0) delete cfg.agent;
}
rewriteFile(file, JSON.stringify(cfg, null, 2));
return [file];
},
};
// ── Factory Droid ───────────────────────────────────────────────────────────
const droidPath = () => path.join(home(), ".factory", "settings.json");
const isDroid9r = (m) => m.id?.startsWith("custom:9Router");
const droid = {
id: "droid",
name: "Factory Droid",
paths: () => [droidPath()],
async apply({ baseUrl, apiKey, model }) {
const file = droidPath();
const cfg = readJson(file) || {};
const others = (cfg.customModels || []).filter((m) => !isDroid9r(m));
cfg.customModels = [
{
model,
id: "custom:9Router-0",
index: 0,
baseUrl: v1(baseUrl),
apiKey,
displayName: model,
maxOutputTokens: 131072,
noImageSupport: false,
provider: "openai",
},
...others,
];
cfg.customModels.forEach((m, i) => { m.index = i; });
writeJson(file, cfg);
return [file];
},
async reset() {
const file = droidPath();
const cfg = readJson(file);
if (!cfg) return [];
if (cfg.customModels) {
cfg.customModels = cfg.customModels.filter((m) => !isDroid9r(m));
if (cfg.customModels.length === 0) delete cfg.customModels;
}
rewriteFile(file, JSON.stringify(cfg, null, 2));
return [file];
},
};
// ── Crush ───────────────────────────────────────────────────────────────────
const crushPath = () => path.join(process.env.XDG_CONFIG_HOME || path.join(home(), ".config"), "crush", "crush.json");
const crush = {
id: "crush",
name: "Crush",
paths: () => [crushPath()],
async apply({ baseUrl, apiKey, model }) {
const file = crushPath();
const cfg = readJson(file) || {};
cfg.providers = cfg.providers || {};
cfg.providers["9router"] = {
type: "openai-compat",
base_url: v1(baseUrl),
api_key: apiKey,
models: [{ id: model, name: model, context_window: 128000 }],
};
writeJson(file, cfg);
return [file];
},
async reset() {
const file = crushPath();
const cfg = readJson(file);
if (!cfg?.providers?.["9router"]) return [];
delete cfg.providers["9router"];
if (Object.keys(cfg.providers).length === 0) delete cfg.providers;
rewriteFile(file, JSON.stringify(cfg, null, 2));
return [file];
},
};
// ── Kilo Code (CLI auth only; VS Code settings left to the dashboard) ───────
const kiloPath = () => path.join(home(), ".local", "share", "kilo", "auth.json");
const kilo = {
id: "kilo",
name: "Kilo Code CLI",
paths: () => [kiloPath()],
async apply({ baseUrl, apiKey, model }) {
const file = kiloPath();
const auth = readJson(file) || {};
auth["openai-compatible"] = { type: "api-key", apiKey, baseUrl: v1(baseUrl), model };
writeJson(file, auth);
return [file];
},
async reset() {
const file = kiloPath();
const auth = readJson(file);
if (!auth) return [];
delete auth["openai-compatible"];
delete auth["9router"];
rewriteFile(file, JSON.stringify(auth, null, 2));
return [file];
},
};
// ── Cline CLI ───────────────────────────────────────────────────────────────
const clineDir = () => path.join(home(), ".cline", "data");
const clineState = () => path.join(clineDir(), "globalState.json");
const clineSecrets = () => path.join(clineDir(), "secrets.json");
const cline = {
id: "cline",
name: "Cline CLI",
paths: () => [clineState(), clineSecrets()],
async apply({ baseUrl, apiKey, model }) {
const state = readJson(clineState()) || {};
state.actModeApiProvider = "openai";
state.planModeApiProvider = "openai";
state.openAiBaseUrl = baseUrl; // Cline expects base WITHOUT /v1
state.openAiModelId = model;
state.planModeOpenAiModelId = model;
writeJson(clineState(), state);
const secrets = readJson(clineSecrets()) || {};
secrets.openAiApiKey = apiKey;
writeJson(clineSecrets(), secrets);
return [clineState(), clineSecrets()];
},
async reset() {
const state = readJson(clineState());
if (!state) return [];
if (state.actModeApiProvider === "openai") {
delete state.openAiBaseUrl;
delete state.openAiModelId;
delete state.planModeOpenAiModelId;
state.actModeApiProvider = "cline";
state.planModeApiProvider = "cline";
}
rewriteFile(clineState(), JSON.stringify(state, null, 2));
const touched = [clineState()];
const secrets = readJson(clineSecrets());
if (secrets) {
delete secrets.openAiApiKey;
rewriteFile(clineSecrets(), JSON.stringify(secrets, null, 2));
touched.push(clineSecrets());
}
return touched;
},
};
const TOOLS = [claude, codex, opencode, droid, crush, kilo, cline];
const TOOL_IDS = TOOLS.map((t) => t.id);
const TOOL_ALIASES = { "claude-code": "claude", "claudecode": "claude", "factory": "droid", "kilocode": "kilo" };
function resolveTools(list) {
const ids = new Set();
for (const raw of list) {
const id = String(raw).trim().toLowerCase();
if (!id) continue;
if (id === "all") { TOOL_IDS.forEach((t) => ids.add(t)); continue; }
const real = TOOL_ALIASES[id] || id;
if (!TOOL_IDS.includes(real)) throw new Error(`Unknown tool "${raw}". Supported: ${TOOL_IDS.join(", ")}, all`);
ids.add(real);
}
return TOOLS.filter((t) => ids.has(t.id));
}
module.exports = { TOOLS, TOOL_IDS, CLAUDE_MODELS, resolveTools, __test__: { stripTrailingCommas } };
+3 -2
View File
@@ -138,11 +138,12 @@ const OAUTH_PROVIDERS = {
iflow: { id: "iflow", alias: "if", name: "iFlow AI" },
qwen: { id: "qwen", alias: "qw", name: "Qwen Code" },
kiro: { id: "kiro", alias: "kr", name: "Kiro AI" },
glm: { id: "glm", alias: "glm", name: "Zai GLM Coding" },
};
const APIKEY_PROVIDERS = {
openrouter: { id: "openrouter", name: "OpenRouter" },
glm: { id: "glm", name: "GLM Coding" },
glm: { id: "glm", name: "Zai GLM Coding" },
minimax: { id: "minimax", name: "Minimax Coding" },
kimi: { id: "kimi", name: "Kimi" },
openai: { id: "openai", name: "OpenAI" },
@@ -399,7 +400,7 @@ async function showConnectionActions(connection, providerId, breadcrumb = []) {
* @param {string} authType - "oauth" or "apikey"
*/
// Providers that use Device Code Flow (terminal-based polling)
const DEVICE_CODE_PROVIDERS = ["github", "qwen", "kiro"];
const DEVICE_CODE_PROVIDERS = ["github", "qwen", "kiro", "glm"];
/**
* Handle adding new connection - auto-detect flow type
+1 -1
View File
@@ -21,7 +21,7 @@ const PROVIDER_ALIAS_NAMES = {
oc: "OpenCode Free",
opencode: "OpenCode Free",
openrouter: "OpenRouter",
glm: "GLM Coding",
glm: "Zai GLM Coding",
kimi: "Kimi Coding",
minimax: "Minimax Coding",
openai: "OpenAI",
+1
View File
@@ -58,6 +58,7 @@ const COOLDOWN = {
*/
export const ERROR_RULES = [
// --- Text-based rules (checked first, order = priority) ---
{ provider: "codex", text: "model is not supported when using codex with a chatgpt account", cooldownMs: MAX_RATE_LIMIT_COOLDOWN_MS },
{ text: "no credentials", cooldownMs: COOLDOWN.long },
{ text: "request not allowed", cooldownMs: COOLDOWN.short },
{ text: "improperly formed request", cooldownMs: COOLDOWN.long },
+4 -1
View File
@@ -1,8 +1,11 @@
export const GROK_CLI_VERSION = "0.2.99";
// cli-chat-proxy rejects older identities with HTTP 426. Keep this on a
// current @xai-official/grok release (1.0.44 as of 2026-10-01; minimum 1.0.13).
export const GROK_CLI_VERSION = "1.0.44";
export const GROK_CLI_MODEL = "grok-build";
export const GROK_CLI_BASE_URL = "https://cli-chat-proxy.grok.com/v1";
export const GROK_CLI_CLIENT_IDENTIFIER = "grok-shell";
export const GROK_CLI_USER_AGENT = `grok-shell/${GROK_CLI_VERSION} (linux; x86_64)`;
export const GROK_CLI_PAGER_USER_AGENT = `grok-pager/${GROK_CLI_VERSION} grok-shell/${GROK_CLI_VERSION} (linux; x86_64)`;
export function supportsGrokCliReasoningEffort(model) {
// ponytail: unknown models omit effort until live metadata reaches dispatch.
+24 -10
View File
@@ -171,7 +171,22 @@ export function resolveKiroThinkingBudget(body, headers, model) {
return null;
}
export function extractKiroEffortLevel(body) {
function parseClaudeVersion(model) {
if (typeof model !== "string") return null;
const normalized = model.toLowerCase().replace(/-/g, ".");
const match = normalized.match(/(?:^|[/.])claude(?:[/.][a-z]+)*[/.](\d+)(?:[/.](\d+))?(?:[/.]|$)/);
if (!match) return null;
return { major: Number(match[1]), minor: match[2] === undefined ? null : Number(match[2]) };
}
// Kiro effort tiers per model (kiro.dev docs + live additionalModelRequestFieldsSchema):
// 4.6 Claude models cap at low|medium|high|max; 4.7+ add xhigh. Unknown models stay conservative.
function kiroModelLacksXhigh(model) {
const v = parseClaudeVersion(model);
return !v || (v.major === 4 && v.minor !== null && v.minor <= 6);
}
export function extractKiroEffortLevel(body, model) {
const effort =
body?.output_config?.effort ??
body?.reasoning_effort ??
@@ -179,7 +194,8 @@ export function extractKiroEffortLevel(body) {
if (typeof effort !== "string") return null;
const normalized = effort.toLowerCase();
if (normalized === "none" || normalized === "off" || normalized === "disabled") return null;
if (normalized === "xhigh" || normalized === "max") return "high";
if (normalized === "xhigh") return kiroModelLacksXhigh(model) ? "high" : "xhigh";
if (normalized === "max") return "max";
if (["low", "medium", "high"].includes(normalized)) return normalized;
return null;
}
@@ -199,10 +215,10 @@ function extractKiroGptEffortLevel(body) {
return null;
}
export function buildKiroAdditionalModelRequestFields(body, effortPath = "output_config") {
export function buildKiroAdditionalModelRequestFields(body, effortPath = "output_config", model) {
const effort = effortPath === "reasoning"
? extractKiroGptEffortLevel(body)
: extractKiroEffortLevel(body);
: extractKiroEffortLevel(body, model);
if (!effort) return undefined;
if (effortPath === "reasoning") {
// Mirrors Kiro CLI/KAS buildEffortRequestFields("reasoning") for GPT.
@@ -222,11 +238,9 @@ export function resolveKiroEffortPath(model) {
return "reasoning";
}
if (!normalized.includes("claude")) return null;
const match = normalized.match(/(?:^|[/.])claude(?:[/.][a-z]+)*[/.](\d+)(?:[/.](\d+))?(?:[/.]|$)/);
if (!match) return null;
const [, majorText, minorText] = match;
const major = Number(majorText);
const minor = minorText === undefined ? null : Number(minorText);
const v = parseClaudeVersion(model);
if (!v) return null;
const { major, minor } = v;
const dateSuffixMinor = minor !== null && minor >= 1000;
// Kiro rejected additionalModelRequestFields on legacy 4.5 models in live smoke.
// Default future Claude/Kiro models to supported so new model releases do not
@@ -248,7 +262,7 @@ export function usesKiroNativeGptEffort(body, model) {
export function buildKiroAdditionalModelRequestFieldsForModel(body, model) {
const effortPath = resolveKiroEffortPath(model);
if (!effortPath) return undefined;
return buildKiroAdditionalModelRequestFields(body, effortPath);
return buildKiroAdditionalModelRequestFields(body, effortPath, model);
}
/**
+3 -1
View File
@@ -98,7 +98,7 @@ export class BaseExecutor {
return { status: response.status, message: bodyText || `HTTP ${response.status}` };
}
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
async execute({ model, body, stream, credentials, signal, log, proxyOptions = null, providerOverrides = null }) {
const fallbackCount = this.getFallbackCount();
let lastError = null;
let lastStatus = 0;
@@ -129,6 +129,8 @@ export class BaseExecutor {
const url = this.buildUrl(model, stream, urlIndex, credentials);
const transformedBody = this.transformRequest(model, body, stream, credentials);
const headers = this.buildHeaders(credentials, stream, url, model, transformedBody);
// User per-provider override wins over registry headers (blocked names filtered at the API)
if (providerOverrides?.headers) Object.assign(headers, providerOverrides.headers);
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
+27
View File
@@ -64,6 +64,33 @@ export class CodeBuddyExecutor extends DefaultExecutor {
// filter and return an error (#2071).
return transformed;
}
parseError(response, bodyText) {
if (bodyText) {
try {
const data = JSON.parse(bodyText);
const msg = data?.msg || data?.message || data?.error?.message || "";
if (data?.code === 6004 || /超出频率限制|frequency limit|限额/i.test(msg)) {
let resetsAtMs = null;
const match = msg.match(/(\d{4}-\d{2}-\d{2})\s+(\d{2}:\d{2}:\d{2})(?:\s*UTC\+?([0-9:]+))?/i);
if (match) {
const dp = match[1];
const tp = match[2];
const tz = match[3]
? (match[3].includes(":") ? (match[3].startsWith("+") ? match[3] : `+${match[3]}`) : `+${match[3].padStart(2, "0")}:00`)
: "+08:00";
const dt = new Date(`${dp}T${tp}${tz}`);
if (!isNaN(dt.getTime())) resetsAtMs = dt.getTime();
}
return {
status: 429,
message: msg || "CodeBuddy frequency limit (6004)",
resetsAtMs,
};
}
} catch {}
}
return super.parseError(response, bodyText);
}
}
export default CodeBuddyExecutor;
+27
View File
@@ -39,6 +39,33 @@ export class CodeBuddyIntlExecutor extends DefaultExecutor {
return transformed;
}
parseError(response, bodyText) {
if (bodyText) {
try {
const data = JSON.parse(bodyText);
const msg = data?.msg || data?.message || data?.error?.message || "";
if (data?.code === 6004 || /超出频率限制|frequency limit|限额/i.test(msg)) {
let resetsAtMs = null;
const match = msg.match(/(\d{4}-\d{2}-\d{2})\s+(\d{2}:\d{2}:\d{2})(?:\s*UTC\+?([0-9:]+))?/i);
if (match) {
const dp = match[1];
const tp = match[2];
const tz = match[3]
? (match[3].includes(":") ? (match[3].startsWith("+") ? match[3] : `+${match[3]}`) : `+${match[3].padStart(2, "0")}:00`)
: "+08:00";
const dt = new Date(`${dp}T${tp}${tz}`);
if (!isNaN(dt.getTime())) resetsAtMs = dt.getTime();
}
return {
status: 429,
message: msg || "CodeBuddy frequency limit (6004)",
resetsAtMs,
};
}
} catch {}
}
return super.parseError(response, bodyText);
}
}
export default CodeBuddyIntlExecutor;
+30 -4
View File
@@ -215,9 +215,9 @@ export class CodexExecutor extends BaseExecutor {
* Override headers to add codex-specific identity headers.
* transformRequest runs BEFORE buildHeaders, sets this._currentSessionId.
*/
buildHeaders(credentials, stream = true, _url = null, model = null) {
buildHeaders(credentials, stream = true, _url = null, model = null, body = null) {
const headers = super.buildHeaders(credentials, stream);
if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model))) {
if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model)) && !body?.tools?.some?.(tool => tool?.type === "web_search")) {
headers["x-openai-internal-codex-responses-lite"] = "true";
}
headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default";
@@ -418,7 +418,33 @@ export class CodexExecutor extends BaseExecutor {
const normalized = normalizeResponsesInput(body.input);
if (normalized) body.input = normalized;
const upstreamModel = getModelUpstreamId("cx", body.model || model);
const responsesLite = isCodexResponsesLiteModel(upstreamModel);
// Register hosted search before choosing transport; Lite cannot execute it.
const autoWebSearch = body._autoCodexWebSearch === true;
delete body._autoCodexWebSearch;
if (autoWebSearch && !body.tools?.some?.(tool => tool?.type === "web_search")) {
body.tools = [...(Array.isArray(body.tools) ? body.tools : []), { type: "web_search" }];
}
// Hosted search cannot run from a Lite input prefix. When switching to
// regular Responses, move all prefixed tools without duplicating definitions.
let convertedLitePrefix = false;
if (isCodexResponsesLiteModel(upstreamModel) && Array.isArray(body.input)
&& (body.tools?.some?.(tool => tool?.type === "web_search")
|| body.input.some(item => item?.type === "additional_tools" && item.tools?.some?.(tool => tool?.type === "web_search")))) {
const tools = Array.isArray(body.tools) ? [...body.tools] : [];
const seen = new Set(tools.map(tool => `${tool?.type}:${tool?.name || tool?.function?.name || ""}`));
for (const item of body.input) {
if (item?.type !== "additional_tools" || !Array.isArray(item.tools)) continue;
for (const tool of item.tools) {
const name = `${tool?.type}:${tool?.name || tool?.function?.name || ""}`;
if (!seen.has(name)) { tools.push(tool); seen.add(name); }
}
}
body.tools = tools;
convertedLitePrefix = body.input.some(item => item?.type === "additional_tools");
body.input = body.input.filter(item => item?.type !== "additional_tools");
}
const responsesLite = isCodexResponsesLiteModel(upstreamModel)
&& !body.tools?.some?.(tool => tool?.type === "web_search");
// Ensure input is present and non-empty (Codex API rejects empty input)
if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) {
@@ -436,7 +462,7 @@ export class CodexExecutor extends BaseExecutor {
body.stream = true;
// If no instructions provided, inject default Codex instructions
if (!responsesLite && (!body.instructions || body.instructions.trim() === "")) {
if (!responsesLite && !convertedLitePrefix && (!body.instructions || body.instructions.trim() === "")) {
body.instructions = CODEX_DEFAULT_INSTRUCTIONS;
}
+3
View File
@@ -41,6 +41,9 @@ function applyAuth(headers, desc, credentials) {
const HEADER_HOOKS = {
// Stable device_id from OAuth connection (CLIProxyAPI KimiTokenStorage.DeviceID)
kimiHeaders: (h, c) => Object.assign(h, buildKimiHeaders(c?.providerSpecificData?.deviceId)),
// Muse: x-api-version only on subscription (minted key) requests — plain
// Model API keys already work without it
museHeaders: (h, c) => { if (c?.accessToken && !c?.apiKey) h["x-api-version"] = "1.0.0"; },
clineHeaders: (h, c) => Object.assign(h, buildClineHeaders(c.apiKey || c.accessToken)),
kilocodeOrg: (h, c) => { if (c.providerSpecificData?.orgId) h["X-Kilocode-OrganizationID"] = c.providerSpecificData.orgId; },
};
+7 -4
View File
@@ -61,7 +61,7 @@ export function stripContinuityFields(body) {
return body;
}
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, capsOverride = null, streamErrorPatterns = null, persistUsage = "all" }) {
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, providerOverrides, capsOverride = null, streamErrorPatterns = null, persistUsage = "all" }) {
const { provider, model } = modelInfo;
const requestStartTime = Date.now();
// Stable per-session color so all lines of one CLI conversation share a tag
@@ -219,9 +219,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
stripContinuityFields(translatedBody);
}
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
if (clientTool === "claude" && Array.isArray(translatedBody.tools)) {
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools);
// Tool normalization: MCP-equivalent built-in dedup (Claude clients) + same-name
// dedup for DeepSeek models (upstream rejects duplicate tool names on all endpoints).
if (Array.isArray(translatedBody.tools)) {
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools, { clientTool, model });
if (stripped.length > 0) {
translatedBody.tools = deduped;
log?.debug?.("TOOLDEDUP", `stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`);
@@ -386,6 +387,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
signal: streamController.signal,
log,
proxyOptions,
providerOverrides,
});
providerResponse = result.response;
providerUrl = result.url;
@@ -455,6 +457,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
signal: streamController.signal,
log,
proxyOptions,
providerOverrides,
});
if (retryResult.response.ok) {
providerResponse = retryResult.response;
+31 -1
View File
@@ -1,4 +1,4 @@
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama
// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama, tinyfish
// Returns normalized shape across all providers
const DEFAULT_TIMEOUT_MS = 15000;
@@ -129,6 +129,9 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
baseUrl: providerConfig?.baseUrl,
});
}
if (provider === "tinyfish") {
return await runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl: providerConfig?.baseUrl });
}
return { success: false, status: 400, error: `Unsupported provider: ${provider}` };
} catch (err) {
log?.("fetch handler error:", err?.message || err);
@@ -136,6 +139,33 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr
}
}
async function runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl }) {
if (!["markdown", "html"].includes(fmt)) return { success: false, status: 400, error: `Unsupported TinyFish format: ${fmt}` };
const upstreamStart = Date.now();
const r = await tryFetch(baseUrl, {
method: "POST",
headers: { "content-type": "application/json", "X-API-Key": apiKey },
body: JSON.stringify({ urls: [url], format: fmt }),
}, timeoutMs);
if (!r.ok) return { success: false, status: r.timeout ? 504 : 502, error: r.error };
const upstreamMs = Date.now() - upstreamStart;
const { json } = await readJsonOrText(r.res);
if (!r.res.ok) return { success: false, status: r.res.status, error: json?.error?.message || `TinyFish error: ${r.res.status}` };
const failure = json?.errors?.[0];
if (failure) return { success: false, status: failure.status || 502, error: `TinyFish fetch failed: ${failure.error || "unknown error"}` };
const page = json?.results?.[0];
if (!page || typeof page.text !== "string") return { success: false, status: 502, error: "TinyFish returned no extractable content" };
const text = truncate(page.text, maxCharacters);
return {
success: true,
data: {
...buildData({ provider: "tinyfish", url, title: page.title, format: fmt, text, links: page.links,
costUsd: costPerQuery, responseMs: Date.now() - startedAt, upstreamMs }),
metadata: { author: page.author || null, published_at: page.published_date || null, language: page.language || null },
},
};
}
async function runFirecrawl({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt }) {
const upstreamStart = Date.now();
const r = await tryFetch("https://api.firecrawl.dev/v1/scrape", {
+24
View File
@@ -374,6 +374,29 @@ function buildXquikRequest(config, params) {
};
}
function buildTinyfishRequest(config, params) {
if (params.searchType && !["web", "news", "research_paper"].includes(params.searchType)) {
throw new Error("Unsupported TinyFish search type");
}
const qp = new URLSearchParams({ query: params.query });
if (params.searchType && params.searchType !== "web") qp.set("domain_type", params.searchType);
if (params.country) qp.set("location", params.country);
if (params.language) qp.set("language", params.language);
const { includes, excludes } = parseDomainFilter(params.domainFilter);
if (includes.length) qp.set("include_domains", includes.join(","));
if (excludes.length) qp.set("exclude_domains", excludes.join(","));
if (Number.isInteger(params.offset) && params.offset > 0) {
if (params.offset >= 110) throw new Error("TinyFish search offset exceeds available pages");
if (params.offset % 10 + params.maxResults > 10) throw new Error("TinyFish search offset and max_results must fit within one page");
qp.set("page", String(Math.floor(params.offset / 10)));
}
return {
// Keep API-key endpoint fixed; client baseUrl overrides must not receive the key.
url: `${config.baseUrl}?${qp}`,
init: { method: "GET", headers: { Accept: "application/json", "X-API-Key": params.token } },
};
}
// ── Ollama Cloud web_search ──────────────────────────────────────────────
// POST https://ollama.com/api/web_search { query, max_results }
// Response: { results: [{ title, url, content, published_at? }] }
@@ -436,6 +459,7 @@ const BUILDERS = {
"youcom": buildYouComRequest,
"searxng": buildSearxngRequest,
"xquik": buildXquikRequest,
"tinyfish": buildTinyfishRequest,
"ollama-search": buildOllamaSearchRequest,
"glm": buildGlmSearchRequest,
};
+4 -1
View File
@@ -110,7 +110,10 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential
}
const data = await resp.json();
const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType);
const results = normalized.results.slice(0, params.maxResults);
// TinyFish uses fixed 10-result pages; offset within a page is applied locally.
const pageOffset = provider.id === "tinyfish" && Number.isInteger(params.offset) && params.offset > 0
? params.offset % 10 : 0;
const results = normalized.results.slice(pageOffset, pageOffset + params.maxResults);
const duration = Date.now() - startTime;
const usage = {
queries_used: 1,
+14
View File
@@ -107,6 +107,19 @@ function normalizeTavily(data, _query, _searchType) {
return { results, totalResults: results.length };
}
function normalizeTinyfish(data) {
const now = new Date().toISOString();
const items = Array.isArray(data?.results) ? data.results : [];
return {
results: items.map((item, idx) => makeResult("tinyfish", {
title: item.title, url: item.url, snippet: item.snippet,
published_at: item.date, source_type: item.publisher || null,
author: Array.isArray(item.authors) ? item.authors.join(", ") : null,
}, idx, now)),
totalResults: data?.total_results ?? null,
};
}
function normalizeGooglePse(data, _query, _searchType) {
const now = new Date().toISOString();
const items = Array.isArray(data.items) ? data.items : [];
@@ -294,6 +307,7 @@ const NORMALIZERS = {
"youcom": normalizeYouCom,
"searxng": normalizeSearxng,
"xquik": normalizeXquik,
"tinyfish": normalizeTinyfish,
"ollama-search": normalizeOllamaSearch,
"glm": normalizeGlmSearch,
};
+3 -2
View File
@@ -19,7 +19,8 @@ export async function handleSystemoneCore({
}) {
const { provider, model } = modelInfo;
const cfg = PROVIDER_MEDIA[provider]?.systemoneConfig;
if (!cfg?.baseUrl) {
const targetUrl = credentials?.providerSpecificData?.baseUrl || cfg?.baseUrl;
if (!targetUrl) {
return createErrorResult(
HTTP_STATUS.BAD_REQUEST,
`Provider '${provider}' does not support System One.`
@@ -49,7 +50,7 @@ export async function handleSystemoneCore({
let providerResponse;
try {
providerResponse = await fetch(cfg.baseUrl, {
providerResponse = await fetch(targetUrl, {
method: "POST",
headers,
body: JSON.stringify(requestBody),
+62 -8
View File
@@ -6,15 +6,18 @@
// 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic
// 4. DEFAULT_CAPABILITIES — safe floor (always returned)
//
// Two extra layers then refine the result, and neither can override the hand
// written tables above (steps 1-2 short-circuit before they are consulted):
// Two extra layers then refine the result:
// • the synced catalog — modalities keyed by model, limits keyed by provider
// + model, refreshed from models.dev in the background. It reads a file, so
// the server installs it via setCatalogSource(); this module stays free of
// node:fs because the dashboard bundles it into the browser too.
// • visionPatterns.js — name-based vision detection, last resort so a model
// nobody has catalogued yet still accepts images.
// Both only ever turn a capability ON.
// Modalities only ever turn a capability ON. Limits from the catalog overlay
// the canonical exact entry (step 2) so a gateway-specific models.dev delta
// (Copilot's 32k Claude output, etc.) actually publishes. Step 1 still
// short-circuits: a hand-written PROVIDER_CAPABILITIES truncation is the
// gateway's own number and must not be overwritten.
//
// ── HOW TO ADD / UPDATE A MODEL ──────────────────────────────────────
// Authoritative data source: https://models.dev/api.json (145 providers, 4000+
@@ -83,8 +86,13 @@ export function capabilitiesFromServiceKind(kind) {
* otherwise mis-match. Only declare deltas vs DEFAULT.
*/
export const MODEL_CAPABILITIES = {
// Claude Fable 5.1, Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
// Claude Fable 5.1, Opus 5.5/5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern)
"claude-fable-5-1": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 },
// Claude Opus 5.5 — experimental preview on Kiro (rateMultiplier: 2.0, 1M context) (#4410)
"claude-opus-5.5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5.5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5.5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5.5-thinking-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
@@ -99,6 +107,8 @@ export const MODEL_CAPABILITIES = {
"claude-opus-4-8-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-4.6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-4-6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
// Sonnet 5.5 rejects thinking.type "disabled" (use "between_tools") and forced tool_choice (any/tool).
"claude-sonnet-5-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000, thinkingOffType: "between_tools", forcedToolChoice: false },
"claude-sonnet-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
"claude-sonnet-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 },
@@ -125,7 +135,10 @@ export const MODEL_CAPABILITIES = {
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
// short-circuits the pattern table, so a vision-only delta would drop them.
// Some providers (e.g. Kenari) expose this model under the hyphenated ID
// "deepseek-v4-1-flash" (dash instead of dot); add it as an alias (#4293).
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
"deepseek-v4-1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
@@ -153,6 +166,12 @@ const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true,
// (lower than OpenAI API's 1.05M). Sol differs from Terra/Luna. #2720
const CODEX_GPT_56_SOL_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 372000, maxOutput: 128000 };
const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 };
// Devin CLI's registry declares a 200k context window for these GPT variants.
// Keep the GPT feature/output fields because provider overrides short-circuit
// the generic pattern rather than merging with it.
const DEVIN_CLI_GPT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 };
/**
* Provider-specific capability overrides. Keyed by provider alias/id.
@@ -175,6 +194,14 @@ export const PROVIDER_CAPABILITIES = {
},
"codex": {
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-6-sol": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-6-luna": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-6-astra[1m]": CODEX_EXTENDED_CAPS,
"gpt-6-sol[1m]": CODEX_EXTENDED_CAPS,
"gpt-6-luna[1m]": CODEX_EXTENDED_CAPS,
"gpt-5.6-sol[1m]": CODEX_EXTENDED_CAPS,
"gpt-5.6-terra[1m]": CODEX_EXTENDED_CAPS,
"gpt-5.6-luna[1m]": CODEX_EXTENDED_CAPS,
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
@@ -196,6 +223,15 @@ export const PROVIDER_CAPABILITIES = {
"gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
"gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
},
"devin-cli": {
"gpt-5.4-high": DEVIN_CLI_GPT_CAPS,
"gpt-5.4-medium": DEVIN_CLI_GPT_CAPS,
"gpt-5.4-low": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-xhigh": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-high": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-medium": DEVIN_CLI_GPT_CAPS,
"gpt-5.5-low": DEVIN_CLI_GPT_CAPS,
},
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
@@ -258,6 +294,9 @@ export const PROVIDER_CAPABILITIES = {
// Qoder CN serves the identical model catalog from the CN gateway, so it shares
// the intl Qoder capability table verbatim (vision/reasoning/contextWindow).
PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"];
PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex;
PROVIDER_CAPABILITIES.dv = PROVIDER_CAPABILITIES["devin-cli"];
PROVIDER_CAPABILITIES.devin = PROVIDER_CAPABILITIES["devin-cli"];
/**
* Pattern fallback — glob (* = wildcard), matched case-insensitively and
@@ -268,6 +307,7 @@ PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"];
export const PATTERN_CAPABILITIES = [
// ── Claude (4.6+ = adaptive thinking; older/haiku = budget) ──────
{ pattern: "*claude*opus-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 } },
{ pattern: "*claude*sonnet-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 } },
{ pattern: "*claude*opus-4.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive" } },
{ pattern: "*claude*opus-4.7*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive" } },
{ pattern: "*claude*opus-4.8*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive" } },
@@ -294,11 +334,24 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
// 1.05M is the API window for the whole gpt-6 family (astra, luna, sol alike).
// A gateway that truncates lower records its own number in
// PROVIDER_CAPABILITIES, which wins over this pattern — Kiro at 272k, Codex
// OAuth at 272k/372k (see CODEX_GPT_56_* above). This entry used to carry
// Kiro's 272k, so every other provider's gpt-6 models inherited one gateway's
// limit and were published at 3.9x under their real window.
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
// gpt-5.4 is where the 1.05M window starts, but the mini and nano tiers stayed
// at 400k — first match wins, so those two have to be listed ahead of it.
{ pattern: "*gpt-5.4-mini*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
{ pattern: "*gpt-5.4-nano*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
{ pattern: "*gpt-5.4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
{ pattern: "*gpt-5.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
{ pattern: "*gpt-5.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } },
{ pattern: "*gpt-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
{ pattern: "*gpt-4o*", caps: { vision: true, search: true, contextWindow: 128000, maxOutput: 16384 } },
{ pattern: "*gpt-4.1*", caps: { vision: true, contextWindow: 1000000, maxOutput: 32768 } },
@@ -645,9 +698,10 @@ export function getCapabilitiesForModel(provider, model) {
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
}
// 2. Canonical exact
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
// 2. Canonical exact, then catalog overlay so provider-scoped models.dev
// deltas still apply. Step 1 above still short-circuits.
if (MODEL_CAPABILITIES[baseModel]) return refine(MODEL_CAPABILITIES[baseModel], provider, model);
if (MODEL_CAPABILITIES[model]) return refine(MODEL_CAPABILITIES[model], provider, model);
// 3. Pattern match (first match wins), refined by catalog + name heuristic
for (const { pattern, caps } of PATTERN_CAPABILITIES) {
+9
View File
@@ -28,6 +28,15 @@ export function isMuseSparkModel(modelId) {
return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base);
}
// "model(level)" is a 9router thinking override; strip before matching.
// Accepts both bare ids ("deepseek-v4-pro(max)") and provider-prefixed ones.
export function isDeepSeekModel(modelId) {
if (!modelId || typeof modelId !== "string") return false;
const clean = modelId.replace(/\([^()]+\)\s*$/, "").trim();
const base = clean.includes("/") ? clean.split("/").pop() : clean;
return /^deepseek-/i.test(base);
}
// Endpoint families for OpenCode models outside the curated registry (modelsFetcher /
// passthrough ids) — regex keeps auto-fetched models on the right endpoint:
// /responses (gpt/grok/muse-spark), /messages (minimax/qwen), /chat/completions (rest).
+17 -1
View File
@@ -49,6 +49,8 @@ export const MODEL_PRICING = {
"claude-opus-4-5-thinking": { input: 5.00, output: 25.00, cached: 0.50, reasoning: 37.50, cache_creation: 5.00 },
"claude-opus-4-6-thinking": { input: 5.00, output: 25.00, cached: 0.50, reasoning: 37.50, cache_creation: 5.00 },
"claude-fable-5": { input: 10.00, output: 50.00, cached: 1.00, reasoning: 50.00, cache_creation: 12.50 },
"claude-sonnet-5-5": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 },
"claude-sonnet-5": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 },
// === OpenAI / GPT ===
"gpt-3.5-turbo": { input: 0.50, output: 1.50, cached: 0.25, reasoning: 2.25, cache_creation: 0.50 },
@@ -73,7 +75,12 @@ export const MODEL_PRICING = {
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
// OpenAI Standard short-context pricing (developers.openai.com/api/docs/pricing).
// Long-context pricing is higher, but this table currently stores one rate per model.
"gpt-6-astra": { input: 10.00, output: 50.00, cached: 1.00, reasoning: 50.00, cache_creation: 12.50 },
"gpt-6.1-sol": { input: 2.00, output: 10.00, cached: 0.10, reasoning: 10.00, cache_creation: 2.50 },
"gpt-6-sol": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 },
"gpt-6-luna": { input: 0.10, output: 0.50, cached: 0.01, reasoning: 0.50, cache_creation: 0.125 },
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
@@ -152,6 +159,15 @@ export const MODEL_PRICING = {
// === Grok ===
"grok-code-fast-1": { input: 0.50, output: 2.00, cached: 0.25, reasoning: 3.00, cache_creation: 0.50 },
// === Muse (Meta Model API) ===
// Rates from https://dev.meta.ai/docs/pricing-rate-limits (contributor tier:
// cheaper, Meta may train on the data).
"muse-spark-1.3": { input: 1.25, output: 4.25, cached: 0.15, reasoning: 4.25, cache_creation: 0 },
"muse-spark-1.2": { input: 1.25, output: 4.25, cached: 0.15, reasoning: 4.25, cache_creation: 0 },
"muse-spark-1.1": { input: 1.25, output: 4.25, cached: 0.15, reasoning: 4.25, cache_creation: 0 },
"muse-spark-1.3-contributor": { input: 0.10, output: 0.20, cached: 0.002, reasoning: 0.20, cache_creation: 0 },
"muse-spark-1.2-contributor": { input: 0.10, output: 0.20, cached: 0.002, reasoning: 0.20, cache_creation: 0 },
// === OpenRouter fallback ===
"auto": { input: 2.00, output: 8.00, cached: 1.00, reasoning: 12.00, cache_creation: 2.00 },
+11 -2
View File
@@ -23,7 +23,16 @@ export default {
baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions",
validateUrl: "https://apihub.agnes-ai.com/v1/models",
},
// No model ids could be verified without a key, so discovery is left to the
// live endpoint and any id is accepted through passthroughModels.
// Seeds so the dashboard has something to show before a key is saved and
// /v1/models answers with a usable list. The live catalogue at
// apihub.agnes-ai.com/v1/models requires a token (401 "Token not provided"),
// so these ids are a curated starting set rather than a verified dump —
// passthroughModels below still accepts any id the account actually has.
models: [
{ id: "agnes-2.5-flash", name: "Agnes 2.5 Flash" },
{ id: "agnes-2.5-pro", name: "Agnes 2.5 Pro" },
{ id: "agnes-2.5-pro-beta", name: "Agnes 2.5 Pro Beta" },
{ id: "agnes-3.0-flash", name: "Agnes 3.0 Flash" },
],
passthroughModels: true,
};
+1
View File
@@ -63,6 +63,7 @@ export default {
{ id: "claude-opus-5", name: "Claude Opus 5" },
{ id: "claude-fable-5-1", name: "Claude Fable 5.1" },
{ id: "claude-fable-5", name: "Claude Fable 5" },
{ id: "claude-sonnet-5-5", name: "Claude Sonnet 5.5" },
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
{ id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" },
],
+17 -9
View File
@@ -2,7 +2,7 @@ import { withCodexReviewModels } from "../models/helpers.js";
// Codex CLI version seen by OpenAI's backend — single source for the Version /
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
const CODEX_CLI_VERSION = "0.155.0";
const CODEX_CLI_VERSION = "0.159.0";
const GPT_6_LITE_THINKING_LEVELS = ["low", "medium", "high", "xhigh", "max"];
export default {
@@ -52,23 +52,29 @@ export default {
},
},
models: [
{ id: "gpt-6.1-sol", name: "GPT 6.1 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
{ id: "gpt-6-astra[1m]", name: "GPT 6.0 Astra (extended context)", upstreamModelId: "gpt-6-astra" },
{ id: "gpt-6-sol", name: "GPT 6.0 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-6-sol[1m]", name: "GPT 6.0 Sol (extended context)", upstreamModelId: "gpt-6-sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-6-luna", name: "GPT 6.0 Luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-6-luna[1m]", name: "GPT 6.0 Luna (extended context)", upstreamModelId: "gpt-6-luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
{ id: "gpt-5.6-sol[1m]", name: "GPT 5.6 Sol (extended context)", upstreamModelId: "gpt-5.6-sol" },
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
{ id: "gpt-5.6-terra[1m]", name: "GPT 5.6 Terra (extended context)", upstreamModelId: "gpt-5.6-terra" },
{ id: "gpt-5.6-terra-review", name: "GPT 5.6 Terra Review", upstreamModelId: "gpt-5.6-terra", quotaFamily: "review" },
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna" },
{ id: "gpt-5.6-luna[1m]", name: "GPT 5.6 Luna (extended context)", upstreamModelId: "gpt-5.6-luna" },
{ id: "gpt-5.6-luna-review", name: "GPT 5.6 Luna Review", upstreamModelId: "gpt-5.6-luna", quotaFamily: "review" },
{ id: "gpt-5.5", name: "GPT 5.5" },
{ id: "gpt-5.5-review", name: "GPT 5.5 Review", upstreamModelId: "gpt-5.5", quotaFamily: "review" },
{ id: "gpt-5.4", name: "GPT 5.4" },
{ id: "gpt-5.4-review", name: "GPT 5.4 Review", upstreamModelId: "gpt-5.4", quotaFamily: "review" },
{ id: "gpt-5.4-mini", name: "GPT 5.4 Mini" },
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
// gpt-5.4 / gpt-5.4-mini / gpt-5.3-codex-spark removed: absent from backend-api/codex/models
// for ChatGPT Plus/Pro accounts and return HTTP 400 "model is not supported" (#4202).
// gpt-daybreak-blue-latest and gpt-reserve added: confirmed live via backend-api/codex/models (#4202).
{ id: "gpt-daybreak-blue-latest", name: "GPT Daybreak Blue" },
{ id: "gpt-reserve", name: "GPT Reserve" },
// Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived
// from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398).
{ id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" },
@@ -81,7 +87,7 @@ export default {
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
// gpt-5.4-image removed alongside gpt-5.4 (both are dead on the backend) (#4202).
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
],
serviceKinds: ["llm","image"],
@@ -98,7 +104,9 @@ export default {
codex_cli_simplified_flow: "true",
originator: "codex_cli_rs",
},
refreshLeadMs: 432000000,
// Access tokens live ~1h; a 5d lead rotated the refresh token on EVERY call —
// reuse of a rotated token revokes the whole OpenAI session (account logout).
refreshLeadMs: 600000,
refresh: {
encoding: "form",
scope: "openid profile email offline_access",
+20 -2
View File
@@ -5,16 +5,34 @@ export default {
priority: 140,
alias: "glm",
display: {
name: "GLM Coding",
name: "Zai GLM Coding",
icon: "code",
color: "#2563EB",
textIcon: "GL",
website: "https://open.bigmodel.cn",
notice: {
apiKeyUrl: "https://open.bigmodel.cn/usercenter/apikeys",
signupUrl: "https://chat.z.ai",
},
},
category: "apikey",
category: "oauth",
// Dual-auth like kimi: paste an API key, or OAuth-login with the Z.ai
// account to auto-mint a coding-plan key.
authModes: ["oauth", "apikey"],
hasOAuth: true,
// OAuth = ZCode CLI polling flow (apps/zcode-cli cli-oauth.ts) — no PKCE, no
// local callback: init mints a one-off poll token, the browser authorize_url
// is server-generated, and poll/ready carries the tokens. The Z.AI OAuth
// token is exchanged for a business JWT, then a long-lived coding-plan API
// key (no refresh grant — re-login on expiry, same as the official CLI).
oauth: {
providerId: "zai",
cliInitUrl: "https://zcode.z.ai/api/v1/oauth/cli/init",
cliPollUrl: "https://zcode.z.ai/api/v1/oauth/cli/poll",
businessLoginUrl: "https://api.z.ai/api/auth/z/login",
apiBaseUrl: "https://api.z.ai",
planApiKeyName: "zcode-api-key",
},
transport: {
baseUrl: "https://api.z.ai/api/anthropic/v1/messages",
format: "claude",
+1 -1
View File
@@ -1,7 +1,7 @@
/**
* Grok CLI / Grok Build (cli-chat-proxy.grok.com)
*
* Source of truth: wire capture of official @xai-official/grok 0.2.99
* Source of truth: wire capture of official @xai-official/grok 1.0.44
* talking to https://cli-chat-proxy.grok.com (OpenAI Responses API).
*
* Distinct from:
+6
View File
@@ -130,6 +130,9 @@ import p126 from "./dahl.js";
import p127 from "./atria.js";
import p129 from "./agnes.js";
import p130 from "./bai.js";
import p131 from "./tinyfish.js";
import p132 from "./v1m.js";
import p133 from "./muse.js";
export default [
p0,
p1,
@@ -260,4 +263,7 @@ export default [
p127,
p129,
p130,
p131,
p132,
p133,
];
+7 -1
View File
@@ -41,7 +41,13 @@ export default {
},
},
models: [
// Opus (added per kiro.dev/changelog/models and kiro.dev/docs/models)
// Opus 5.5 — experimental preview, 1M context, 2x credits (#4410)
// Announced 2026-09-22; confirmed in kiro.dev session UI.
{ id: "claude-opus-5.5", name: "Claude Opus 5.5" },
{ id: "claude-opus-5.5-thinking", name: "Claude Opus 5.5 (Thinking)" },
{ id: "claude-opus-5.5-agentic", name: "Claude Opus 5.5 (Agentic)" },
{ id: "claude-opus-5.5-thinking-agentic", name: "Claude Opus 5.5 (Thinking + Agentic)" },
// Opus 5
{ id: "claude-opus-5", name: "Claude Opus 5" },
{ id: "claude-opus-5-thinking", name: "Claude Opus 5 (Thinking)" },
{ id: "claude-opus-5-agentic", name: "Claude Opus 5 (Agentic)" },
+67
View File
@@ -0,0 +1,67 @@
// Muse (Meta Model API) — dual auth (same pattern as kimi):
// oauth = Muse Code subscription (Meta account device code, mints an LLM|… key)
// apikey = pay-as-you-go Model API key from dev.meta.ai
// Transport is shared; oauth accounts get x-api-version via the museHeaders hook.
// Meta issues no refresh token for the subscription flow → re-login on 401.
export default {
id: "muse",
priority: 120,
alias: "muse",
aliases: [
"muse-ai",
"meta-model-api",
"muse-code",
"muse-subscription",
],
uiAlias: "muse",
display: {
name: "Muse (Meta Model API)",
icon: "auto_awesome",
color: "#0866FF",
textIcon: "MU",
website: "https://muse.ai",
notice: {
text: "Sign in with your Meta account (Muse Code subscription) or paste a Model API key from dev.meta.ai. Subscription keys are minted per account; Meta may train on contributor-tier data.",
apiKeyUrl: "https://dev.meta.ai",
signupUrl: "https://muse.ai",
},
},
category: "oauth",
authModes: ["oauth", "apikey"],
hasOAuth: true,
transport: {
baseUrl: "https://api.meta.ai/v1/chat/completions",
validateUrl: "https://api.meta.ai/v1/models",
modelsUrl: "https://api.meta.ai/v1/models",
auth: { combined: true, header: "Authorization", scheme: "bearer", hooks: ["museHeaders"] },
},
// Multi-endpoint: Meta accepts Chat Completions and Responses wire formats on
// the same key (https://dev.meta.ai/docs/protocols). Muse Spark reasoning
// (incl. encrypted_content replay) only round-trips on Responses, so models
// pin targetFormat there.
transports: [
{
format: "openai",
baseUrl: "https://api.meta.ai/v1/chat/completions",
auth: { combined: true, header: "Authorization", scheme: "bearer", hooks: ["museHeaders"] },
},
{
format: "openai-responses",
baseUrl: "https://api.meta.ai/v1/responses",
auth: { combined: true, header: "Authorization", scheme: "bearer", hooks: ["museHeaders"] },
},
],
models: [
{ id: "muse-spark-1.3", name: "Muse Spark 1.3", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2", name: "Muse Spark 1.2", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.1", name: "Muse Spark 1.1", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
],
passthroughModels: true,
oauth: {
clientId: "1031625952748946",
deviceCodeUrl: "https://auth.meta.com/oidc/device/authorization/",
tokenUrl: "https://auth.meta.com/oidc/device/token/",
},
};
+36
View File
@@ -0,0 +1,36 @@
export default {
id: "tinyfish",
alias: "tinyfish",
display: {
name: "TinyFish",
color: "#FF6700",
textIcon: "TF",
website: "https://www.tinyfish.ai/",
notice: { apiKeyUrl: "https://agent.tinyfish.ai/api-keys" },
},
category: "apikey",
authType: "apikey",
serviceKinds: ["webSearch", "webFetch"],
searchConfig: {
baseUrl: "https://api.search.tinyfish.ai",
validateUrl: "https://api.search.tinyfish.ai/usage?limit=1",
method: "GET",
authType: "apikey",
authHeader: "x-api-key",
costPerQuery: 0,
searchTypes: ["web", "news", "research_paper"],
defaultMaxResults: 5,
maxMaxResults: 10,
timeoutMs: 10000,
},
fetchConfig: {
baseUrl: "https://api.fetch.tinyfish.ai",
method: "POST",
authType: "apikey",
authHeader: "x-api-key",
costPerQuery: 0,
formats: ["markdown", "html"],
maxCharacters: 100000,
timeoutMs: 150000,
},
};
+31
View File
@@ -0,0 +1,31 @@
export default {
id: "v1m",
priority: 45,
alias: "v1m",
aliases: ["systemone", "jev"],
uiAlias: "v1m",
display: {
name: "v1m (System One)",
icon: "psychology",
color: "#6366F1",
textIcon: "V1",
website: "https://v1m.ir",
notice: {
text: "v1m System One calibrated decision engine. Fast probabilistic evaluations over state and questions.",
apiKeyUrl: "https://v1m.ir",
},
},
category: "apikey",
authType: "apikey",
hasProviderSpecificData: true,
models: [
{ id: "rev-latest", name: "v1m Rev Latest (Calibrated)", kind: "systemone" },
{ id: "v1m-decision-engine", name: "v1m Decision Engine", kind: "systemone" },
],
serviceKinds: ["systemone"],
systemoneConfig: {
baseUrl: "https://v1m.ir/v1/systemone",
authType: "apikey",
authHeader: "bearer",
},
};
+8 -3
View File
@@ -10,8 +10,8 @@ const L = {
base: ["none", "low", "medium", "high"], // qwen, step, hunyuan, gemini-budget
onOff: ["none", "thinking"], // zai (binary), minimax (adaptive)
openai: ["none", "minimal", "low", "medium", "high", "xhigh"], // GPT-5.x / o-series (no "max")
levelMax: ["none", "low", "medium", "high", "max"], // claude-adaptive, kimi
budgetX: ["none", "low", "medium", "high", "xhigh", "max"], // claude-budget
levelMax: ["none", "low", "medium", "high", "max"], // kimi
budgetX: ["none", "low", "medium", "high", "xhigh", "max"], // claude-budget, claude-adaptive
gemini: ["minimal", "low", "medium", "high"], // gemini-3 thinkingLevel (no disable)
hiMax: ["none", "high", "max"], // deepseek (low/med→high, xhigh→max)
};
@@ -19,7 +19,7 @@ const L = {
// thinkingFormat → valid selectable levels (source of truth for UI options).
const FORMAT_LEVELS = {
openai: L.openai,
"claude-adaptive": L.levelMax,
"claude-adaptive": L.budgetX,
"claude-budget": L.budgetX,
"gemini-level": L.gemini,
"gemini-budget": L.base,
@@ -35,8 +35,13 @@ const FORMAT_LEVELS = {
const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
// Opus/Sonnet 4.6 lack xhigh (Anthropic + Kiro docs) — keep the 4-level+max set.
const CLAUDE_NO_XHIGH = ["none", "low", "medium", "high", "max"];
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
const PATTERN_THINKING = [
{ pattern: "*claude*4.6*", levels: CLAUDE_NO_XHIGH },
{ pattern: "*claude*4-6*", levels: CLAUDE_NO_XHIGH },
{ provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS },
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
+3 -2
View File
@@ -20,18 +20,19 @@ export function getQuotaCooldown(backoffLevel = 0) {
* @param {number} backoffLevel - Current backoff level for exponential backoff
* @returns {{ shouldFallback: boolean, cooldownMs: number, newBackoffLevel?: number }}
*/
export function checkFallbackError(status, errorText, backoffLevel = 0) {
export function checkFallbackError(status, errorText, backoffLevel = 0, provider = null) {
const lowerError = errorText
? (typeof errorText === "string" ? errorText : JSON.stringify(errorText)).toLowerCase()
: "";
for (const rule of ERROR_RULES) {
if (rule.provider && rule.provider !== provider) continue;
// Request-scoped rule: the request body itself is at fault — no cooldown,
// no account lock. Caller must stop rotating and surface the error.
if (rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
return { shouldFallback: false, cooldownMs: 0 };
}
// Text-based rule: match substring in error message
if (rule.text && !rule.requestScoped && lowerError && lowerError.includes(rule.text)) {
if (rule.backoff) {
+12 -1
View File
@@ -124,8 +124,19 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) {
// Config-driven prefix → provider inference (first match wins, fallback "openai").
const MODEL_PREFIX_PROVIDERS = [
// Codex CLI sends this bare virtual model for auto-review — keep it on OAuth Codex (#1398).
// Codex CLI sends this bare virtual model for auto-review - keep it on OAuth Codex (#1398).
[/^codex-auto-review$/, "codex"],
// Codex-only GPT model slugs: present in backend-api/codex/models but not on the
// OpenAI API. Without these rules a bare model id (e.g. "gpt-5.6-terra" from the
// Codex CLI /model picker) resolves to provider "openai", which 404s for users that
// only have a Codex OAuth account and no OpenAI API key (#4405).
// Ranges covered: gpt-5.x, gpt-6.x, gpt-daybreak-*, gpt-reserve* — all are
// Codex-backend models. Plain "gpt-4*" / "gpt-3.5*" / "gpt-4o*" fall through to
// the generic gpt-* → openai rule below.
[/^gpt-[56]\./, "codex"],
[/^gpt-6-/, "codex"],
[/^gpt-daybreak-/, "codex"],
[/^gpt-reserve/, "codex"],
[/^claude-/, "anthropic"],
[/^gemini-/, "gemini"],
[/^gpt-/, "openai"],
+3 -2
View File
@@ -47,8 +47,9 @@ const USAGE_HANDLERS = {
"qoder-cn": (c) => getQoderUsageFor(c),
iflow: (c) => getIflowUsage(c.accessToken),
ollama: (c) => getOllamaUsage(c.apiKey, c.providerSpecificData, c.proxyOptions),
glm: (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
"glm-cn": (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions),
// OAuth connections store the coding-plan key on accessToken (no apiKey)
glm: (c) => getGlmUsage(c.apiKey || c.accessToken, c.provider, c.proxyOptions),
"glm-cn": (c) => getGlmUsage(c.apiKey || c.accessToken, c.provider, c.proxyOptions),
minimax: (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
"minimax-cn": (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions),
"vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions),
+38 -29
View File
@@ -12,37 +12,11 @@ const GLM_QUOTA_URLS = {
};
/**
* GLM Coding Plan usage (international + China regions)
* Parse the GLM quota API response — shared by pasted API keys and
* OAuth-minted coding-plan keys (both hit the same monitor endpoint).
* Supports both TOKENS_LIMIT and CREDIT_LIMIT and dynamic intervals (e.g. session 5h, weekly 7d).
*/
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
if (!apiKey) {
return { message: "GLM API key not available." };
}
const region = provider === "glm-cn" ? "china" : "international";
const quotaUrl = GLM_QUOTA_URLS[region];
try {
const response = await proxyAwareFetch(
quotaUrl,
{
headers: {
Authorization: `Bearer ${apiKey}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (!response.ok) {
if (response.status === 401) {
return { message: "GLM API key invalid or expired." };
}
return { message: `GLM quota API error (${response.status}).` };
}
const json = await response.json();
export function parseGlmQuotaResponse(json) {
const data = json?.data && typeof json.data === "object" ? json.data : {};
const limits = Array.isArray(data.limits) ? data.limits : [];
const quotas = {};
@@ -82,6 +56,41 @@ export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
: "Unknown";
return { plan, quotas };
}
/**
* GLM Coding Plan usage (international + China regions)
*/
export async function getGlmUsage(apiKey, provider, proxyOptions = null) {
if (!apiKey) {
return { message: "GLM API key not available." };
}
const region = provider === "glm-cn" ? "china" : "international";
const quotaUrl = GLM_QUOTA_URLS[region];
try {
const response = await proxyAwareFetch(
quotaUrl,
{
headers: {
Authorization: `Bearer ${apiKey}`,
Accept: "application/json",
},
},
proxyOptions,
);
if (!response.ok) {
if (response.status === 401) {
return { message: "GLM API key invalid or expired." };
}
return { message: `GLM quota API error (${response.status}).` };
}
const json = await response.json();
const { plan, quotas } = parseGlmQuotaResponse(json);
return { plan, quotas };
} catch (error) {
return { message: `GLM error: ${error.message}` };
}
@@ -271,7 +271,9 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) {
if (canDisable) body.thinking = { type: "adaptive", ...(display ? { display } : {}) };
else delete body.thinking;
const level = toLevel(eff);
body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level };
// xhigh is model-gated (Opus/Sonnet 4.6 reject it) — clamp when not advertised.
body.output_config = { effort: level === "auto" ? "high"
: level === "xhigh" && !supportedLevels?.includes("xhigh") ? "high" : level };
break;
}
case "claude-budget": {
+94 -21
View File
@@ -7,6 +7,7 @@ import { resolveSessionId } from "../../utils/sessionManager.js";
import { isValidClaudeSignature } from "../../utils/claudeSignature.js";
import { PROVIDERS } from "../../providers/index.js";
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
import { isDeepSeekModel } from "../../providers/models/helpers.js";
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
const CACHE_CONTROL_5M = { type: "ephemeral" };
@@ -25,24 +26,32 @@ export function lastCacheableToolIndex(tools) {
}
// Check if message has valid non-empty content
// A block type outside this list makes the whole message count as empty and be
// dropped by prepareClaudeRequest — so anything the caller can legitimately
// send alone must be listed. container_upload (Files API) is one of those:
// a user turn whose only block is a file reference is valid Anthropic input
// (#4316), and dropping it forwarded `messages: []` to the provider.
const CONTENTFUL_BLOCKS = new Set([
CLAUDE_BLOCK.TOOL_USE,
CLAUDE_BLOCK.TOOL_RESULT,
CLAUDE_BLOCK.IMAGE,
CLAUDE_BLOCK.DOCUMENT,
CLAUDE_BLOCK.CONTAINER_UPLOAD,
]);
function isContentfulBlock(block) {
if (!block) return false;
if (block.type === CLAUDE_BLOCK.TEXT) return !!block.text?.trim();
return CONTENTFUL_BLOCKS.has(block.type);
}
export function hasValidContent(msg) {
if (typeof msg.content === "string" && msg.content.trim()) return true;
if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) {
const block = msg.content;
return !!((block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
block.type === CLAUDE_BLOCK.TOOL_USE ||
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
block.type === CLAUDE_BLOCK.IMAGE ||
block.type === CLAUDE_BLOCK.DOCUMENT);
return isContentfulBlock(msg.content);
}
if (Array.isArray(msg.content)) {
return msg.content.some(block =>
(block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
block.type === CLAUDE_BLOCK.TOOL_USE ||
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
block.type === CLAUDE_BLOCK.IMAGE ||
block.type === CLAUDE_BLOCK.DOCUMENT
);
return msg.content.some(isContentfulBlock);
}
return false;
}
@@ -167,7 +176,7 @@ function handlesThinkingBlocks(provider) {
return provider === "claude" || provider?.startsWith("anthropic-compatible") || provider === "deepseek";
}
function buildThinkingPlaceholder(provider) {
function buildThinkingPlaceholder(provider, unsigned = false) {
const block = {
type: CLAUDE_BLOCK.THINKING,
thinking: ".",
@@ -175,7 +184,9 @@ function buildThinkingPlaceholder(provider) {
// DeepSeek's Anthropic-compatible endpoint requires a thinking block in
// thinking mode, but it does not need Anthropic's signed-thinking fallback.
if (provider !== "deepseek") {
// The same applies to DeepSeek models served through other providers'
// Claude transports (opencode-go /messages).
if (provider !== "deepseek" && !unsigned) {
block.signature = DEFAULT_THINKING_CLAUDE_SIGNATURE;
}
@@ -215,6 +226,8 @@ export function normalizeClaudePassthrough(body, model = "") {
if (Object.keys(body.output_config).length === 0) delete body.output_config;
}
const originalLastRole = Array.isArray(body.messages) ? body.messages[body.messages.length - 1]?.role : undefined;
// 3. Wrap bare content-block objects as one-element arrays before folding.
// Some clients send content: {block} instead of content: [{block}]; the
// mid-conversation-system fold below assumes the array shape, so it must
@@ -316,11 +329,25 @@ export function normalizeClaudePassthrough(body, model = "") {
!(block?.type === CLAUDE_BLOCK.TEXT && !String(block.text ?? "").trim()));
return msg.content.length > 0;
});
body.messages = ensureTrailingUserTurn(body.messages, originalLastRole);
}
return body;
}
// Newer Claude models reject a body that ends on an assistant turn ("does not
// support assistant message prefill"). Cleanup passes delete messages left empty,
// so a trailing user turn that was empty (or held only dropped blocks) silently
// turns the previous assistant turn into the last one. Restore a user turn only
// when the client did not itself end on assistant (real prefill is its choice).
const TRAILING_USER_PLACEHOLDER = "Continue.";
export function ensureTrailingUserTurn(messages, originalLastRole) {
if (!Array.isArray(messages) || originalLastRole === ROLE.ASSISTANT) return messages;
if (messages[messages.length - 1]?.role !== ROLE.ASSISTANT) return messages;
return [...messages, { role: ROLE.USER, content: [{ type: CLAUDE_BLOCK.TEXT, text: TRAILING_USER_PLACEHOLDER }] }];
}
// Put a 5m breakpoint on the last cache-eligible block of a message.
// thinking/redacted_thinking blocks do not accept cache_control.
function markLastCacheableBlock(msg) {
@@ -335,6 +362,21 @@ function markLastCacheableBlock(msg) {
return false;
}
// In an agent's tool loop, a request ends with the results of the last
// assistant turn's tool calls -- after that turn's breakpoint. They go at the
// full input price, and the next request (which appends to them) writes them
// into the cache. When the 4-marker budget has room, a 5m breakpoint on that
// final user turn caches them now, and the next request reads them.
function markFinalToolResults(body) {
const messages = body?.messages;
const last = Array.isArray(messages) ? messages[messages.length - 1] : null;
if (last?.role !== ROLE.USER || !Array.isArray(last.content)) return false;
if (!last.content.some((block) => block?.type === CLAUDE_BLOCK.TOOL_RESULT)) return false;
if (last.content.some((block) => block?.cache_control)) return false;
if (countCacheControlBlocks(body) >= 4) return false;
return markLastCacheableBlock(last);
}
// Re-anchor cache breakpoints on a Claude passthrough body (same policy as
// prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m.
// The client's own markers point at pre-normalization offsets, so they are dropped.
@@ -404,6 +446,9 @@ export function anchorClaudeCache(body) {
anchored = markLastCacheableBlock(body.messages[i]);
}
}
// ...and a tool loop's final tool results, so the next step reads them.
markFinalToolResults(body);
}
return body;
@@ -443,12 +488,27 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
delete body.output_config;
}
// Models whose API rejects thinking "disabled" and forced tool use with a 400
// (Sonnet 5.5). Runs on every Claude-bound body, so OpenAI clients, native
// passthrough and the provider-level "off" override are all covered.
const modelCaps = getCapabilitiesForModel(provider, body.model);
if (modelCaps.thinkingOffType && body.thinking?.type === "disabled") {
body.thinking = { type: modelCaps.thinkingOffType };
// between_tools only accepts effort up to high.
const effort = body.output_config?.effort;
if (effort === "xhigh" || effort === "max") body.output_config.effort = "high";
}
if (modelCaps.forcedToolChoice === false && (body.tool_choice?.type === "any" || body.tool_choice?.type === "tool")) {
const { disable_parallel_tool_use } = body.tool_choice;
body.tool_choice = { type: "auto", ...(disable_parallel_tool_use !== undefined ? { disable_parallel_tool_use } : {}) };
}
// Clamp max_tokens to the model's real output ceiling. Models whose caps
// declare a higher maxOutput (e.g. Opus 4.8 / Sonnet 4.6 = 128000) are allowed
// up to it, so max-effort thinking gets full budget; others fall back to the
// conservative 64000 default.
if (body.max_tokens) {
const ceiling = getCapabilitiesForModel(provider, body.model).maxOutput || DEFAULT_MAX_TOKENS;
const ceiling = modelCaps.maxOutput || DEFAULT_MAX_TOKENS;
if (body.max_tokens > ceiling) body.max_tokens = ceiling;
// Reconcile against thinking budget. applyThinking (thinkingUnified.js) runs
@@ -480,6 +540,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
// 2. Messages: process in optimized passes
if (body.messages && Array.isArray(body.messages)) {
const len = body.messages.length;
const originalLastRole = body.messages[len - 1]?.role;
let filtered = [];
// Pass 1: remove cache_control + filter empty messages
@@ -504,6 +565,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
// Pass 1.5: Fix tool_use/tool_result ordering
// Each tool_use must have tool_result in the NEXT message (not same message with other content)
filtered = fixToolUseOrdering(filtered);
filtered = ensureTrailingUserTurn(filtered, originalLastRole);
body.messages = filtered;
@@ -512,6 +574,14 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
const lastMessageIsUser = lastMessage?.role === "user";
const thinkingEnabled = body.thinking?.type === "enabled" && lastMessageIsUser;
// DeepSeek models also arrive behind OpenCode Go's /messages transport.
// They carry the same thinking pass-back constraint as the official
// DeepSeek provider (verified live 2026-08-15, PR #3332 discussion), so
// they get the identical keep/placeholder handling below.
const deepSeekServed =
provider === "deepseek" ||
(provider === "opencode-go" && isDeepSeekModel(body?.model));
// Pass 2 (reverse): add cache_control to last assistant + handle thinking for Anthropic
let lastAssistantProcessed = false;
for (let i = filtered.length - 1; i >= 0; i--) {
@@ -532,15 +602,15 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
}
// Handle thinking blocks for Anthropic-compatible endpoints.
if (handlesThinkingBlocks(provider)) {
if (handlesThinkingBlocks(provider) || deepSeekServed) {
let hasToolUse = false;
let hasKeptThinking = false;
// Claude native: preserve valid signatures, drop invalid blocks.
// anthropic-compatible: replace with default (safe fallback for lenient upstreams).
// DeepSeek: keep existing thinking as-is; add an unsigned placeholder only if missing.
// DeepSeek (official + opencode-go models): keep existing thinking as-is;
// add an unsigned placeholder only if missing.
const isClaudeNative = provider === "claude";
const isDeepSeek = provider === "deepseek";
const kept = [];
for (const block of msg.content) {
const isThinking = block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING;
@@ -550,7 +620,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
hasKeptThinking = true;
kept.push(block);
}
} else if (isDeepSeek) {
} else if (deepSeekServed) {
hasKeptThinking = true;
kept.push(block);
} else {
@@ -567,7 +637,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
// Add thinking block if thinking enabled + has tool_use but no thinking
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
msg.content.unshift(buildThinkingPlaceholder(provider));
msg.content.unshift(buildThinkingPlaceholder(provider, deepSeekServed));
}
}
}
@@ -638,6 +708,9 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
body = hoistToolResultImages(body);
}
// A tool loop's final tool results: cached now, so the next step reads them.
markFinalToolResults(body);
// Apply cloaking for OAuth tokens (billing header + fake user ID)
// session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency
if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) {
+8 -1
View File
@@ -32,7 +32,14 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [
"title", "optional", "deprecated", "if", "then", "else", "contentMediaType", "contentEncoding",
// UI/Styling properties (from Cursor tools - NOT JSON Schema standard)
"cornerRadius", "fillColor", "fontFamily", "fontSize", "fontWeight",
"gap", "padding", "strokeColor", "strokeThickness", "textColor"
"gap", "padding", "strokeColor", "strokeThickness", "textColor",
// Non-standard annotation/error keywords used by some MCP tool schemas (#4283).
// Gemini's schema proto has no field for these and rejects the whole request with
// "Unknown name X: Cannot find field" if any nested schema node carries them.
"errorMessage", "errorMessages", "x-errorMessage", "x-errorMessages",
"markdownDescription", "x-intellij-html-description",
"x-taplo-info", "x-taplo", "doNotSuggest", "suggestSortText",
"minProperties", "maxProperties"
];
// Default safety settings
+24 -1
View File
@@ -1,6 +1,6 @@
import { FORMATS } from "./formats.js";
import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js";
import { prepareClaudeRequest } from "./formats/claude.js";
import { prepareClaudeRequest, ensureTrailingUserTurn } from "./formats/claude.js";
import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js";
import { restoreToolNames } from "../utils/opencodeFingerprint.js";
import { filterToOpenAIFormat } from "./formats/openai.js";
@@ -9,6 +9,7 @@ import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js";
import { captureSessionId } from "../utils/sessionManager.js";
import { AntigravityExecutor } from "../executors/antigravity.js";
import { PROVIDERS } from "../providers/index.js";
import { ROLE, GEMINI_ROLE } from "./schema/roles.js";
// Registry for translators. Lazy-init guards against circular-import order:
// translator modules call register() (side-effect) before this module's body runs.
@@ -49,10 +50,26 @@ function stripContentTypes(body, stripList = []) {
}
}
// Role the client's conversation actually ended on, in the source format's own
// shape — not every source uses messages[] (Gemini/Antigravity: contents[],
// Responses/Codex: input[]). Only an explicit trailing model/assistant turn is
// real prefill and must reach ensureTrailingUserTurn as ROLE.ASSISTANT; every
// other tail (including no role, e.g. a function output) stays undefined so
// the emptied-turn fix still applies.
function detectClientLastRole(body) {
if (Array.isArray(body?.messages)) return body.messages[body.messages.length - 1]?.role;
const items = Array.isArray(body?.contents) ? body.contents : Array.isArray(body?.input) ? body.input : null;
if (!items) return undefined;
const role = items[items.length - 1]?.role;
return role === ROLE.ASSISTANT || role === GEMINI_ROLE.MODEL ? ROLE.ASSISTANT : undefined;
}
// Translate request: source -> openai -> target
export function translateRequest(sourceFormat, targetFormat, model, body, stream = true, credentials = null, provider = null, reqLogger = null, stripList = [], connectionId = null, clientTool = null) {
ensureInitialized();
let result = body;
// Role the client actually ended on, before any translator drops an emptied turn.
const clientLastRole = detectClientLastRole(body);
// Strip explicit content types (opt-in via strip[] in PROVIDER_MODELS entry)
stripContentTypes(result, stripList);
@@ -132,6 +149,7 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
if (targetFormat === FORMATS.CLAUDE) {
const apiKey = credentials?.accessToken || credentials?.apiKey || null;
result = prepareClaudeRequest(result, provider, apiKey, connectionId, credentials?.rawHeaders, clientSessionId);
if (Array.isArray(result?.messages)) result.messages = ensureTrailingUserTurn(result.messages, clientLastRole);
}
// Claude cloaking: rename client tools with CLAUDE_TOOL_SUFFIX (anti-ban)
@@ -269,6 +287,11 @@ export function initState(sourceFormat) {
funcArgsDone: {},
funcItemDone: {},
customToolNames: new Set(),
// Chat Completions usage for response.completed. Not state.usage: other translators in
// the same pipeline overwrite that in their own shapes.
responsesUsage: null,
// finish_reason arrived before usage; response.completed waits for the usage chunk.
completionPending: false,
completedSent: false
};
}
@@ -27,17 +27,22 @@ import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } fro
function toResponsesUsage(usage) {
if (!usage || typeof usage !== "object") return null;
const inputTokens = [usage.input_tokens, usage.prompt_tokens].find(Number.isFinite) ?? 0;
const outputTokens = [usage.output_tokens, usage.completion_tokens].find(Number.isFinite) ?? 0;
const inputTokens = [usage.input_tokens, usage.prompt_tokens].find(Number.isInteger);
const outputTokens = [usage.output_tokens, usage.completion_tokens].find(Number.isInteger);
// Some upstreams attach zeroed placeholders to every chunk. Wait for real counts
// so response.completed cannot freeze the placeholder before the usage trailer.
if (inputTokens === undefined || outputTokens === undefined || inputTokens + outputTokens <= 0) {
return null;
}
const responseUsage = {
input_tokens: inputTokens,
output_tokens: outputTokens,
total_tokens: Number.isFinite(usage.total_tokens) ? usage.total_tokens : inputTokens + outputTokens
total_tokens: inputTokens + outputTokens
};
const cachedTokens = [usage.input_tokens_details?.cached_tokens, usage.prompt_tokens_details?.cached_tokens].find(Number.isFinite);
const reasoningTokens = [usage.output_tokens_details?.reasoning_tokens, usage.completion_tokens_details?.reasoning_tokens].find(Number.isFinite);
if (Number.isFinite(cachedTokens)) responseUsage.input_tokens_details = { cached_tokens: cachedTokens };
if (Number.isFinite(reasoningTokens)) responseUsage.output_tokens_details = { reasoning_tokens: reasoningTokens };
const cachedTokens = [usage.input_tokens_details?.cached_tokens, usage.prompt_tokens_details?.cached_tokens].find(Number.isInteger);
const reasoningTokens = [usage.output_tokens_details?.reasoning_tokens, usage.completion_tokens_details?.reasoning_tokens].find(Number.isInteger);
if (Number.isInteger(cachedTokens)) responseUsage.input_tokens_details = { cached_tokens: cachedTokens };
if (Number.isInteger(reasoningTokens)) responseUsage.output_tokens_details = { reasoning_tokens: reasoningTokens };
return responseUsage;
}
@@ -47,13 +52,14 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
return flushEvents(state);
}
// Capture upstream usage BEFORE the choices guard below: the last OpenAI chunk
// may carry usage together with an empty choices array, and it must not be dropped.
if (chunk.usage) {
state.responsesUsage = toResponsesUsage(chunk.usage);
}
// Capture usage before the choices guard: OpenAI may send it in a trailer
// whose choices array is empty.
const responseUsage = toResponsesUsage(chunk.usage);
if (responseUsage) state.responsesUsage = responseUsage;
if (!chunk.choices?.length) return [];
if (!chunk.choices?.length) {
return state.completionPending && state.responsesUsage ? flushEvents(state) : [];
}
const events = [];
const nextSeq = () => ++state.seq;
@@ -163,6 +169,7 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
// would swallow the terminal event entirely. Keep the old behaviour there.
const flushReachesUs = state.targetFormat === FORMATS.OPENAI;
if (state.responsesUsage || !flushReachesUs) sendCompleted(state, emit);
else state.completionPending = true;
}
return events;
+1
View File
@@ -18,6 +18,7 @@ export const CLAUDE_BLOCK = {
DOCUMENT: "document",
TOOL_USE: "tool_use",
TOOL_RESULT: "tool_result",
CONTAINER_UPLOAD: "container_upload",
THINKING: "thinking",
REDACTED_THINKING: "redacted_thinking",
SERVER_TOOL_USE: "server_tool_use",
+67 -13
View File
@@ -16,6 +16,20 @@ async function tryGotScrapingFetch() { return null; }
// DNS cache — use Map to avoid prototype pollution via malformed hostnames
const DNS_CACHE = new Map();
const TLS_CERT_ERRORS = new Set([
"SELF_SIGNED_CERT_IN_CHAIN",
"DEPTH_ZERO_SELF_SIGNED_CERT",
"UNABLE_TO_VERIFY_LEAF_SIGNATURE",
"UNABLE_TO_GET_ISSUER_CERT",
"UNABLE_TO_GET_ISSUER_CERT_LOCALLY",
"CERT_HAS_EXPIRED",
"ERR_TLS_CERT_ALTNAME_INVALID",
]);
function isTlsCertError(err) {
const code = err?.cause?.code || err?.code;
return TLS_CERT_ERRORS.has(code);
}
const MITM_BYPASS_HOSTS = [
"cloudcode-pa.googleapis.com",
"daily-cloudcode-pa.googleapis.com",
@@ -133,20 +147,44 @@ function resolveConnectionProxyUrl(targetUrl, proxyOptions) {
/**
* Create proxy dispatcher lazily (undici-compatible)
*/
async function getDispatcher(proxyUrl) {
async function getDispatcher(proxyUrl, insecure = false) {
const normalized = normalizeProxyUrl(proxyUrl);
if (!normalized) return null;
if (!normalized && !insecure) return null;
if (!proxyDispatchers.has(normalized)) {
const key = `${normalized || "direct"}::${insecure ? "insecure" : "secure"}`;
if (!proxyDispatchers.has(key)) {
// Evict oldest entry if max size reached
if (proxyDispatchers.size >= MEMORY_CONFIG.proxyDispatchersMaxSize) {
proxyDispatchers.delete(proxyDispatchers.keys().next().value);
}
const { ProxyAgent } = await import("undici");
proxyDispatchers.set(normalized, new ProxyAgent({ uri: normalized }));
const { Agent, ProxyAgent } = await import("undici");
const connect = insecure ? { rejectUnauthorized: false } : undefined;
const dispatcher = normalized
? new ProxyAgent({ uri: normalized, ...(insecure ? { requestTls: connect } : {}) })
: new Agent({ connect });
proxyDispatchers.set(key, dispatcher);
}
return proxyDispatchers.get(normalized);
return proxyDispatchers.get(key);
}
async function fetchWithTlsFallback(url, options, proxyUrl) {
try {
const dispatcher = proxyUrl ? await getDispatcher(proxyUrl) : undefined;
return await originalFetch(url, dispatcher ? { ...options, dispatcher } : options);
} catch (err) {
const isStrictSsl = process.env.STRICT_SSL === "true" || process.env.STRICT_SSL === "1";
if (!isStrictSsl && isTlsCertError(err)) {
if (options.body && typeof options.body.getReader === "function" && options.body.locked) {
throw err;
}
// ponytail: in-memory insecure agent fallback for self-signed MITM corporate/antivirus certs
console.warn(`[ProxyFetch] TLS cert verification failed (${err.cause?.code || err.code}), retrying with insecure TLS: ${url}`);
const insecureDispatcher = await getDispatcher(proxyUrl, true);
return await originalFetch(url, { ...options, dispatcher: insecureDispatcher });
}
throw err;
}
}
/**
@@ -235,8 +273,7 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) {
if (proxyUrl) {
// Proxy resolves DNS externally (not affected by /etc/hosts) — use proxy directly
try {
const dispatcher = await getDispatcher(proxyUrl);
return await originalFetch(url, { ...options, dispatcher });
return await fetchWithTlsFallback(url, options, proxyUrl);
} catch (proxyError) {
if (proxyOptions?.strictProxy === true) {
throw new Error(`[ProxyFetch] Proxy required but failed (strictProxy=true): ${proxyError.message}`);
@@ -256,20 +293,37 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) {
if (proxyUrl) {
try {
const dispatcher = await getDispatcher(proxyUrl);
return await originalFetch(url, { ...options, dispatcher });
return await fetchWithTlsFallback(url, options, proxyUrl);
} catch (proxyError) {
// If strictProxy is enabled, fail hard instead of falling back to direct
if (proxyOptions?.strictProxy === true) {
throw new Error(`[ProxyFetch] Proxy required but failed (strictProxy=true): ${proxyError.message}`);
}
console.warn(`[ProxyFetch] Proxy failed, falling back to direct: ${proxyError.message}`);
return originalFetch(url, options);
return fetchWithTlsFallback(url, options, null);
}
}
// got-scraping disabled — use native fetch directly
return originalFetch(url, options);
// Strict mode means "never leave over the direct IP". Reaching here with a
// proxy configured but unresolved is exactly that case — an inactive or
// empty pool, or every proxy removed — so refuse instead of silently
// exposing the real address (#4333). The catch blocks above only cover a
// proxy that was actually tried.
//
// Gate on a proxy being *intended*: callers like the Qoder executor set
// strictProxy to mean "do not replay this request directly if the proxy
// fails" (a replayed COSY signature returns 403), not "a proxy is required".
// With nothing configured they must keep working.
const proxyIntended = proxyOptions?.proxyPoolId
|| proxyOptions?.enabled === true
|| proxyOptions?.connectionProxyEnabled === true
|| !!normalizeString(proxyOptions?.url ?? proxyOptions?.connectionProxyUrl);
if (proxyOptions?.strictProxy === true && proxyIntended) {
throw new Error("[ProxyFetch] Proxy required but none resolved (strictProxy=true)");
}
// (Re-enable per-host by wrapping with tryGotScrapingFetch when needed)
return fetchWithTlsFallback(url, options, null);
}
/**
+42
View File
@@ -22,6 +22,11 @@ const STREAM_MODE = {
PASSTHROUGH: "passthrough" // No translation, normalize output, extract usage
};
// Upper bound on the deferred response.completed wait: a chat->responses stream
// that saw finish_reason without usage must not hold the client's terminal event
// forever when the upstream stalls with no usage trailer and no [DONE].
const PENDING_COMPLETION_FLUSH_MS = 3000;
/**
* Create unified SSE transform stream
* @param {object} options
@@ -88,9 +93,12 @@ export function createSSEStream(options = {}) {
let openAIResponsesDoneSent = false;
let streamDoneSent = false; // track duplicate [DONE] across transform + flush
let finalized = false;
let completionFlushTimer = null;
// Usage/logging tail, callable from transform() as well as flush(): a client that
// closes right after the terminal event cancels the reader, and flush() never runs.
const finalizeStream = () => {
if (completionFlushTimer) { clearTimeout(completionFlushTimer); completionFlushTimer = null; }
if (finalized) return;
finalized = true;
@@ -183,6 +191,20 @@ export function createSSEStream(options = {}) {
}
// Emit the deferred response.completed now — at [DONE], or when the watchdog
// below gives up on a usage trailer that never arrives.
const flushPendingCompletion = (controller) => {
const completed = translateResponse(targetFormat, sourceFormat, null, state);
for (const item of completed || []) {
if (item === null || item === undefined) continue;
const output = formatSSE(item, sourceFormat);
reqLogger?.appendConvertedChunk?.(output);
controller.enqueue(sharedEncoder.encode(output));
sseEmittedCount++;
}
finalizeStream();
};
return new TransformStream({
transform(chunk, controller) {
if (!ttftAt) ttftAt = Date.now();
@@ -329,6 +351,14 @@ export function createSSEStream(options = {}) {
// For Ollama: done=true is the final chunk with finish_reason/usage, must translate
// For other formats: done=true is the [DONE] sentinel, skip
if (parsed && parsed.done && targetFormat !== FORMATS.OLLAMA) {
// A direct Chat-to-Responses translation can defer response.completed
// while waiting for a usage trailer. [DONE] ends that opportunity even
// if the upstream keeps the HTTP connection open, so finish now.
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES &&
state.completionPending && !state.completedSent) {
flushPendingCompletion(controller);
}
// Synthesize response.failed if the Responses stream never sent a terminal event
if (keepsOpenAIResponsesFormat && !openAIResponsesTerminalSeen) {
const failedOutput = formatIncompleteOpenAIResponsesStreamFailure();
@@ -414,6 +444,18 @@ export function createSSEStream(options = {}) {
sseEmittedCount++;
}
}
// The completion deferral can outlive the upstream: a broken chat upstream
// may stall after finish_reason with no usage trailer and no [DONE], holding
// the connection open. Bound the wait so the client still gets a terminal event.
if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES &&
state?.completionPending && !state?.completedSent && !completionFlushTimer) {
completionFlushTimer = setTimeout(() => {
completionFlushTimer = null;
if (state?.completedSent) return;
try { flushPendingCompletion(controller); } catch { /* controller already closed */ }
}, PENDING_COMPLETION_FLUSH_MS);
}
}
},
+39 -6
View File
@@ -1,7 +1,13 @@
/**
* Strip built-in/duplicate tools when equivalent MCP tools are present.
* Goal: reduce tool definitions token bloat for Claude clients.
* Tool normalization before dispatch:
* - MCP-equivalent built-in tool dedup (Claude clients only, reduces token bloat).
* - Exact same-name tool dedup for DeepSeek models — the DeepSeek upstream rejects
* duplicate tool names with 400 "Tool names must be unique" on every endpoint
* (verified live 2026-08-15 against api.deepseek.com, opencode.go and a LiteLLM
* gateway; GLM/MiniMax/Kimi upstreams accept duplicates). First definition wins,
* tool_choice and message-history references are by name/id so nothing breaks.
*/
import { isDeepSeekModel } from "../providers/models/helpers.js";
const DEDUP_RULES = [
{
@@ -30,10 +36,21 @@ function matches(name, pattern) {
return pattern instanceof RegExp ? pattern.test(name) : false;
}
function dedupeTools(tools) {
/**
* @param {Array} tools - translated tools array
* @param {Object} [opts]
* @param {string|null} [opts.clientTool] - detected client ("claude" | "codex" | ...)
* @param {string|null} [opts.model] - model id, may carry a (level) thinking suffix
* @returns {{ tools: Array, stripped: Array<string> }}
*/
function dedupeTools(tools, opts = {}) {
if (!Array.isArray(tools) || tools.length === 0) return { tools, stripped: [] };
const names = tools.map(getToolName);
const toStrip = new Set();
const toDrop = new Set(); // indices of duplicate same-name tools
// MCP-based built-in dedup: Claude clients only (existing behavior).
if (opts.clientTool === "claude") {
for (const rule of DEDUP_RULES) {
const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p)));
if (!hasTrigger) continue;
@@ -41,9 +58,25 @@ function dedupeTools(tools) {
if (rule.strip.some((p) => matches(n, p))) toStrip.add(n);
}
}
if (toStrip.size === 0) return { tools, stripped: [] };
const out = tools.filter((t) => !toStrip.has(getToolName(t)));
return { tools: out, stripped: Array.from(toStrip) };
}
// Exact-name dedup: DeepSeek upstream rejects duplicate tool names. Applies to
// every client × provider that serves a deepseek-* model (official API, Console Go,
// LiteLLM gateways); non-DeepSeek models are untouched.
if (isDeepSeekModel(opts.model)) {
const seen = new Set();
for (let i = 0; i < tools.length; i++) {
const n = getToolName(tools[i]);
if (!n) continue;
if (seen.has(n)) toDrop.add(i);
else seen.add(n);
}
}
if (toStrip.size === 0 && toDrop.size === 0) return { tools, stripped: [] };
const out = tools.filter((t, i) => !toDrop.has(i) && !toStrip.has(getToolName(t)));
const stripped = Array.from(toDrop).map((i) => getToolName(tools[i])).concat(Array.from(toStrip));
return { tools: out, stripped };
}
export { dedupeTools };
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "9router-app",
"version": "0.5.91",
"version": "0.5.95",
"description": "9Router web dashboard",
"private": true,
"scripts": {
Binary file not shown.

After

Width:  |  Height:  |  Size: 5.8 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 7.3 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 2.0 KiB

@@ -0,0 +1,181 @@
"use client";
import { useCallback, useEffect, useState } from "react";
import PropTypes from "prop-types";
import { Card, Badge } from "@/shared/components";
import { useNotificationStore } from "@/store/notificationStore";
// Mirrors the server-side gate in /api/providers/[id]/overrides — client check is UX only
const BLOCKED_HEADERS = ["host", "content-length", "content-type", "connection", "transfer-encoding", "authorization", "cookie"];
const HEADER_NAME_RE = /^[A-Za-z0-9-]+$/;
export default function CustomConfigCard({ providerId }) {
const notify = useNotificationStore();
const [expanded, setExpanded] = useState(false);
const [rows, setRows] = useState([{ name: "", value: "" }]);
const [builtin, setBuiltin] = useState({});
const [hasOverride, setHasOverride] = useState(false);
const [saving, setSaving] = useState(false);
useEffect(() => {
let cancelled = false;
fetch(`/api/providers/${providerId}/overrides`, { cache: "no-store" })
.then((r) => (r.ok ? r.json() : null))
.then((data) => {
if (cancelled || !data) return;
// Effective set = registry built-ins with user overrides layered on top
const builtinHeaders = data.builtinHeaders || {};
const effective = { ...builtinHeaders, ...(data.headers || {}) };
const headerRows = Object.entries(effective).map(([name, value]) => ({ name, value }));
setBuiltin(builtinHeaders);
setRows(headerRows.length ? headerRows : [{ name: "", value: "" }]);
setHasOverride(Object.keys(data.headers || {}).length > 0);
})
.catch(() => {});
return () => { cancelled = true; };
}, [providerId]);
const setRow = (i, field, value) => {
setRows((prev) => prev.map((r, idx) => (idx === i ? { ...r, [field]: value } : r)));
};
const save = useCallback(async () => {
// Diff-on-save: only rows differing from the registry default become overrides,
// so a code-side registry bump still wins for everything the user left alone.
const headers = {};
for (const r of rows.filter((r) => r.name.trim())) {
const name = r.name.trim();
if (!HEADER_NAME_RE.test(name)) {
notify.error(`Invalid header name: ${name}`);
return;
}
if (BLOCKED_HEADERS.includes(name.toLowerCase())) {
notify.error(`Header ${name} cannot be overridden`);
return;
}
if (name in headers) {
notify.error(`Duplicate header name: ${name}`);
return;
}
if (r.value !== builtin[name]) headers[name] = r.value;
}
setSaving(true);
try {
const res = await fetch(`/api/providers/${providerId}/overrides`, {
method: "PUT",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ headers }),
});
if (!res.ok) {
const err = await res.json().catch(() => ({}));
notify.error(err.error || "Failed to save");
return;
}
setHasOverride(Object.keys(headers).length > 0);
notify.success("Custom headers saved");
} finally {
setSaving(false);
}
}, [rows, builtin, providerId, notify]);
const resetToBuiltin = () => {
setRows(Object.entries(builtin).map(([name, value]) => ({ name, value })));
};
// Only render when there is something to customize: registry headers or existing overrides
if (Object.keys(builtin).length === 0 && !hasOverride) return null;
return (
<Card padding="xs">
<button
type="button"
onClick={() => setExpanded((v) => !v)}
className="flex w-full items-center justify-between text-left"
>
<div className="flex items-center gap-2">
<span className="material-symbols-outlined text-primary text-[20px]">tune</span>
<span className="text-sm font-semibold">Custom Headers</span>
{hasOverride && (
<Badge variant="success" size="sm">Active</Badge>
)}
</div>
<span className="material-symbols-outlined text-text-muted">
{expanded ? "expand_less" : "expand_more"}
</span>
</button>
{expanded && (
<div className="mt-3 border-t border-border pt-3">
<div className="flex flex-col gap-2">
{rows.map((row, i) => {
const overridden = row.name.trim() in builtin && row.value !== builtin[row.name.trim()];
return (
<div key={i} className="flex items-center gap-2">
<input
value={row.name}
onChange={(e) => setRow(i, "name", e.target.value)}
placeholder="Header-Name"
spellCheck={false}
className="w-44 rounded-md border border-border bg-background px-2 py-1.5 text-sm focus:border-primary focus:outline-none"
/>
<input
value={row.value}
onChange={(e) => setRow(i, "value", e.target.value)}
placeholder="Value"
spellCheck={false}
title={overridden ? "Overridden" : row.name.trim() in builtin ? "Registry default" : ""}
className={`min-w-0 flex-1 rounded-md border bg-background px-2 py-1.5 text-sm focus:border-primary focus:outline-none ${
overridden ? "border-amber-400/60" : "border-border"
}`}
/>
<button
type="button"
title="Remove header"
onClick={() => setRows((prev) => (prev.length > 1 ? prev.filter((_, idx) => idx !== i) : [{ name: "", value: "" }]))}
className="shrink-0 text-text-muted hover:text-red-500"
>
<span className="material-symbols-outlined text-[18px]">delete</span>
</button>
</div>
);
})}
</div>
<div className="mt-3 flex items-center justify-between gap-2">
<button
type="button"
onClick={() => setRows((prev) => [...prev, { name: "", value: "" }])}
className="flex items-center gap-1 text-xs text-primary hover:underline"
>
<span className="material-symbols-outlined text-[16px]">add</span>
Add header
</button>
<div className="flex gap-2">
<button
type="button"
disabled={saving}
onClick={resetToBuiltin}
className="rounded-md border border-border px-3 py-1.5 text-xs hover:bg-black/[0.03] dark:hover:bg-white/[0.03] disabled:opacity-50"
>
Reset
</button>
<button
type="button"
disabled={saving}
onClick={save}
className="rounded-md bg-primary px-3 py-1.5 text-xs font-medium text-white hover:opacity-90 disabled:opacity-50"
>
{saving ? "Saving..." : "Save"}
</button>
</div>
</div>
</div>
)}
</Card>
);
}
CustomConfigCard.propTypes = {
providerId: PropTypes.string.isRequired,
};
@@ -22,6 +22,7 @@ import EditCompatibleNodeModal from "./EditCompatibleNodeModal";
import AddCustomModelModal from "./AddCustomModelModal";
import BulkImportCodexModal from "./BulkImportCodexModal";
import BulkImportGrokCliModal from "./BulkImportGrokCliModal";
import CustomConfigCard from "./CustomConfigCard";
const ONE_BY_ONE_DELAY_MS = 1000;
@@ -2271,6 +2272,9 @@ export default function ProviderDetailPage() {
</Card>
)}
{/* Per-provider user overrides (custom headers / connect timeout) */}
<CustomConfigCard providerId={providerId} />
{/* Models */}
<Card>
<div className="mb-4 flex flex-col gap-2 sm:flex-row sm:items-center sm:justify-between">
@@ -1,6 +1,7 @@
"use client";
import { useState, useEffect, useCallback, useRef, useMemo } from "react";
import { useSearchParams, useRouter, usePathname } from "next/navigation";
import ProviderIcon from "@/shared/components/ProviderIcon";
import QuotaTable from "./QuotaTable";
import Toggle from "@/shared/components/Toggle";
@@ -152,6 +153,9 @@ function formatTimeRemaining(value) {
export default function ProviderLimits() {
const { copied, copy } = useCopyToClipboard();
const searchParams = useSearchParams();
const router = useRouter();
const pathname = usePathname();
const [connections, setConnections] = useState([]);
const [quotaData, setQuotaData] = useState({});
const [loading, setLoading] = useState({});
@@ -171,7 +175,26 @@ export default function ProviderLimits() {
const [showEditModal, setShowEditModal] = useState(false);
const [selectedConnection, setSelectedConnection] = useState(null);
const [proxyPools, setProxyPools] = useState([]);
const [providerFilter, setProviderFilter] = useState("all");
// Initialize providerFilter from URL ?provider= param so the page
// can be bookmarked / deep-linked to a specific provider (#4217).
const [providerFilter, _setProviderFilter] = useState(
() => searchParams?.get("provider") || "all"
);
// Wrapper: keeps URL in sync with the selected provider so the view can be
// bookmarked. Replaces the URL without adding to browser history.
const setProviderFilter = useCallback((value) => {
_setProviderFilter(value);
try {
const params = new URLSearchParams(searchParams?.toString() || "");
if (value === "all") {
params.delete("provider");
} else {
params.set("provider", value);
}
const newUrl = params.toString() ? `${pathname}?${params.toString()}` : pathname;
router.replace(newUrl, { scroll: false });
} catch { /* non-fatal: URL sync is best-effort */ }
}, [pathname, router, searchParams]);
const [providerOptions, setProviderOptions] = useState([]);
const [accountFilter, setAccountFilter] = useState("active");
const [quotaSortMode, setQuotaSortMode] = useState("default");
@@ -619,7 +619,8 @@ export function parseQuotaData(provider, data) {
break;
case "codebuddy-cn":
// CodeBuddy CN mixes recurring refill packs ("Monthly"/"Weekly"/...)
case "codebuddy-intl":
// CodeBuddy CN/Intl mix recurring refill packs ("Monthly"/"Weekly"/...)
// with one-shot bonus packs ("Bonus Pack N"). Forward `recurring`
// so the UI can show "Expires in" for bonus packs (whose resetAt is
// a hard expiry, not a refresh) instead of "Reset in".
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -97,7 +98,7 @@ export async function POST(request) {
existing.openai = {
base_url: normalizedBaseUrl,
api_key: apiKey || "sk_9router",
api_key: await resolveCliApiKey(apiKey),
model: model || "provider/model-id",
};
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -81,7 +82,7 @@ export async function POST(request) {
} catch { /* No existing config */ }
const endpointUrl = `${baseUrl}/chat/completions#models.ai.azure.com`;
const keyToUse = apiKey || "sk_9router";
const keyToUse = await resolveCliApiKey(apiKey);
const newEntry = {
name: "9Router",
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -108,7 +109,7 @@ export async function POST(request) {
existing.providers["9router"] = {
type: "openai-compat",
base_url: normalizedBaseUrl,
api_key: apiKey || "sk_9router",
api_key: await resolveCliApiKey(apiKey),
models: [
{
id: modelId,
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import { exec } from "child_process";
import { promisify } from "util";
import fs from "fs/promises";
@@ -132,7 +133,7 @@ export async function POST(request) {
const dir = getDeepSeekDir();
await fs.mkdir(dir, { recursive: true });
const newConfig = build9RouterConfig(baseUrl, apiKey || "sk_9router", model);
const newConfig = build9RouterConfig(baseUrl, await resolveCliApiKey(apiKey), model);
await fs.writeFile(getDeepSeekConfigPath(), newConfig);
return NextResponse.json({
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -96,7 +97,7 @@ export async function POST(request) {
const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`;
existing.openai = {
api_key: apiKey || "sk_9router",
api_key: await resolveCliApiKey(apiKey),
base_url: normalizedBaseUrl,
model: model || "provider/model-id",
};
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import { exec } from "child_process";
import { promisify } from "util";
import fs from "fs/promises";
@@ -108,7 +109,7 @@ export async function POST(request) {
const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`;
const toml = applyGrokBuildConfig(await readConfigToml(), {
baseUrl: normalizedBaseUrl,
apiKey: apiKey || "sk_9router",
apiKey: await resolveCliApiKey(apiKey),
model: selectedModel,
contextWindow: normalizeContextWindow(contextWindow, selectedModel),
subagentModels: normalizeSubagentModels(subagentModels),
+6 -3
View File
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -51,7 +52,7 @@ const has9RouterInYml = (content) => {
// Build standard 9Router provider block for models.yml
const buildOmpProviderYaml = (baseUrl, apiKey) => {
const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`;
const key = apiKey || "sk_9router";
const key = apiKey || "";
return ` ${PROVIDER_ID}:
baseUrl: ${normalizedBaseUrl}
apiKey: ${key}
@@ -100,10 +101,12 @@ export async function POST(request) {
return NextResponse.json({ error: { message: "baseUrl is required" } }, { status: 400 });
}
const resolvedKey = await resolveCliApiKey(apiKey);
await fs.mkdir(getOmpDir(), { recursive: true });
let ymlContent = await readModelsYml();
const providerBlock = buildOmpProviderYaml(baseUrl, apiKey);
const providerBlock = buildOmpProviderYaml(baseUrl, resolvedKey);
// Remove existing 9router provider if present
const regex = new RegExp(`\\s*${PROVIDER_ID}:[\\s\\S]*?(?=\\n\\s*\\w+:|$)`, "g");
@@ -137,7 +140,7 @@ export async function POST(request) {
).run(
PROVIDER_ID,
"api_key",
JSON.stringify({ apiKey: apiKey || "sk_9router", baseUrl }),
JSON.stringify({ apiKey: resolvedKey, baseUrl }),
Math.floor(Date.now() / 1000),
Math.floor(Date.now() / 1000)
);
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import { exec } from "child_process";
import { promisify } from "util";
import fs from "fs/promises";
@@ -112,7 +113,7 @@ export async function POST(request) {
} catch { /* No existing config */ }
const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`;
const keyToUse = apiKey || "sk_9router";
const keyToUse = await resolveCliApiKey(apiKey);
const effectiveSubagentModel = subagentModel || modelsArray[0];
// Ensure provider object
+4 -2
View File
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -145,9 +146,10 @@ export async function POST(request) {
}
existing.providers["9router"] = {
...existingProvider,
baseUrl: normalizedBaseUrl,
apiKey: apiKey || "sk_9router",
api: "openai-completions",
apiKey: apiKey || existingProvider.apiKey || await resolveCliApiKey(null),
api: existingProvider.api || "openai-completions",
models: modelList,
};
+36
View File
@@ -0,0 +1,36 @@
/**
* Resolves the API key to write into a CLI tool config.
*
* CLI tool cards send an empty string when no key is explicitly selected
* (e.g. the existing config already has a provider block but the frontend
* can't read the stored Authorization header back). The routes previously
* fell back to the literal placeholder "sk_9router", which causes 401
* "Invalid API key" for any deployment with requireApiKey=true (#4399).
*
* Resolution order:
* 1. The key supplied by the caller (non-empty string).
* 2. The first active key in the dashboard's apiKeys table.
* 3. Empty string — the route writes no Authorization header value,
* which is fine for requireApiKey=false deployments.
*
* The placeholder "sk_9router" is NEVER written; it was never a real key.
*/
import { getApiKeys } from "@/lib/db";
/**
* @param {string|null|undefined} callerKey Key sent by the frontend.
* @returns {Promise<string>}
*/
export async function resolveCliApiKey(callerKey) {
if (callerKey && callerKey.trim() && callerKey.trim() !== "sk_9router") {
return callerKey.trim();
}
try {
const keys = await getApiKeys();
const active = keys.find((k) => k.isActive);
return active?.key || "";
} catch {
return "";
}
}
@@ -1,6 +1,7 @@
"use server";
import { NextResponse } from "next/server";
import { resolveCliApiKey } from "../resolveApiKey.js";
import fs from "fs/promises";
import path from "path";
import os from "os";
@@ -96,7 +97,7 @@ export async function POST(request) {
const updated = {
...existing,
baseUrl: normalizedBaseUrl,
apiKey: apiKey || "sk_9router",
apiKey: await resolveCliApiKey(apiKey),
model: model || existing.model || "provider/model-id",
_managedBy: "9router",
};
@@ -265,6 +265,8 @@ export async function GET(request, { params }) {
"qoder",
"qoder-cn",
"grok-cli",
"muse",
"glm",
];
let deviceData;
if (noPkceDeviceProviders.includes(provider)) {
@@ -498,7 +500,7 @@ export async function POST(request, { params }) {
}
// Providers that don't use PKCE for device code
const noPkceProviders = ["github", "kimi", "kimi-coding", "kilocode", "codebuddy-cn", "codebuddy-intl"];
const noPkceProviders = ["github", "kimi", "kimi-coding", "kilocode", "codebuddy-cn", "codebuddy-intl", "glm"];
let result;
if (noPkceProviders.includes(provider)) {
// kimi needs extraData._kimiDeviceId for stable X-Msh-Device-Id (CLIProxyAPI parity)
@@ -552,6 +554,8 @@ export async function POST(request, { params }) {
error: result.error,
errorDescription: result.errorDescription,
pending: isPending,
// fatal: unrecoverable (e.g. post-exchange failure) — client must stop polling and show it
...(result.fatal ? { fatal: true } : {}),
});
}
+11 -6
View File
@@ -13,15 +13,12 @@ import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy";
import { resolveCursorModels } from "open-sse/services/cursorModels.js";
import { resolveZedModels } from "open-sse/shared/zedAuth.js";
import { resolveClineModels, resolveClinepassModels } from "open-sse/services/clinepassModels.js";
import codexProvider from "open-sse/providers/registry/codex.js";
const GEMINI_CLI_MODELS_URL = "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels";
// The /codex/models endpoint gates each entry by minimal_client_version against this
// value, and codex CLI's own manifest (openai/codex codex-rs/models-manager/models.json)
// already requires 0.144.0 for its newest models, so a stale client_version here comes
// back 200 with those entries quietly missing instead of erroring.
const CODEX_CLIENT_VERSION = "0.144.6";
const CODEX_MODELS_URL = `https://chatgpt.com/backend-api/codex/models?client_version=${CODEX_CLIENT_VERSION}`;
// Model discovery must identify as the same Codex CLI version as inference.
const CODEX_MODELS_URL = `https://chatgpt.com/backend-api/codex/models?client_version=${codexProvider.transport.cliVersion}`;
const parseOpenAIStyleModels = (data) => {
if (Array.isArray(data)) return data;
@@ -172,6 +169,14 @@ function buildQoderModelsResolver(providerId) {
// Provider models endpoints configuration
const PROVIDER_MODELS_CONFIG = {
"muse": {
url: "https://api.meta.ai/v1/models",
method: "GET",
headers: { "Content-Type": "application/json", "x-api-version": "1.0.0" },
authHeader: "Authorization",
authPrefix: "Bearer ",
parseResponse: (data) => data.data || [],
},
claude: {
url: "https://api.anthropic.com/v1/models",
method: "GET",
@@ -0,0 +1,110 @@
import { NextResponse } from "next/server";
import { getSettings, updateSettings } from "@/lib/localDb";
import { PROVIDERS } from "open-sse/config/providers.js";
import { resolveProviderAlias } from "open-sse/services/model.js";
export const dynamic = "force-dynamic";
// Validation at the trust boundary — the UI also validates, but this is the gate.
const MAX_HEADERS = 20;
const MAX_HEADER_VALUE_LENGTH = 8192;
// RFC 7230 token subset: letters, digits, hyphen (no spaces, no unicode)
const HEADER_NAME_RE = /^[A-Za-z0-9-]+$/;
// Request-structure / auth headers a user override must never touch
const BLOCKED_HEADERS = new Set([
"host",
"content-length",
"content-type",
"connection",
"transfer-encoding",
"authorization",
"cookie",
]);
/**
* Validate + normalize an override payload. Returns { override } or { error }.
* An override with no headers is normalized to null (= delete).
*/
function normalizeOverride({ headers }) {
const out = {};
if (headers !== undefined && headers !== null) {
if (typeof headers !== "object" || Array.isArray(headers)) {
return { error: "headers must be an object" };
}
const entries = Object.entries(headers).filter(([, v]) => v !== "" && v != null);
if (entries.length > MAX_HEADERS) {
return { error: `Too many headers (max ${MAX_HEADERS})` };
}
const clean = {};
for (const [name, value] of entries) {
if (!HEADER_NAME_RE.test(name)) {
return { error: `Invalid header name: ${name}` };
}
if (typeof value !== "string" || /[\r\n]/.test(value)) {
return { error: `Invalid value for header ${name}` };
}
if (value.length > MAX_HEADER_VALUE_LENGTH) {
return { error: `Header ${name} value too long (max ${MAX_HEADER_VALUE_LENGTH})` };
}
if (BLOCKED_HEADERS.has(name.toLowerCase())) {
return { error: `Header ${name} cannot be overridden` };
}
clean[name] = value;
}
if (Object.keys(clean).length) out.headers = clean;
}
return { override: Object.keys(out).length ? out : null };
}
async function readOverrides() {
const settings = await getSettings();
return settings.providerOverrides || {};
}
/**
* GET /api/providers/[id]/overrides — user override for this provider
*/
export async function GET(request, { params }) {
try {
const { id } = await params;
// URL may use an alias (gcli, cc…) — key everything by canonical registry id
const canonical = resolveProviderAlias(id);
const override = (await readOverrides())[canonical] || {};
// Built-in headers come straight from the registry transport — single source of
// truth, so the UI pre-fills exactly what this provider sends upstream.
return NextResponse.json({
headers: override.headers || {},
builtinHeaders: PROVIDERS[canonical]?.headers || {},
});
} catch (error) {
console.log("Error getting provider overrides:", error);
return NextResponse.json({ error: "Failed to get overrides" }, { status: 500 });
}
}
/**
* PUT /api/providers/[id]/overrides — body: { headers: {name: value} }
* Empty payload clears the override.
*/
export async function PUT(request, { params }) {
try {
const { id } = await params;
const canonical = resolveProviderAlias(id);
const body = await request.json().catch(() => ({}));
const { override, error } = normalizeOverride(body);
if (error) {
return NextResponse.json({ error }, { status: 400 });
}
const current = await readOverrides();
const next = { ...current };
if (override) next[canonical] = override;
else delete next[canonical];
await updateSettings({ providerOverrides: next });
return NextResponse.json({ headers: override?.headers || {} });
} catch (error) {
console.log("Error saving provider overrides:", error);
return NextResponse.json({ error: "Failed to save overrides" }, { status: 500 });
}
}
+17 -3
View File
@@ -5,6 +5,7 @@ import { isOpenAICompatibleProvider, isAnthropicCompatibleProvider } from "@/sha
import { getDefaultModel } from "open-sse/config/providerModels.js";
import { resolveOllamaLocalHost, PROVIDERS } from "open-sse/config/providers.js";
import { CODEX_CLI_VERSION } from "open-sse/config/appConstants.js";
import { GROK_CLI_PAGER_USER_AGENT, GROK_CLI_VERSION } from "open-sse/config/grokCli.js";
import {
refreshProviderCredentials,
shouldRefreshCredentials,
@@ -100,6 +101,9 @@ const OAUTH_TEST_CONFIG = {
authPrefix: "Bearer ",
},
"codebuddy-cn": { tokenExists: true },
// codebuddy-intl uses the same JWT token structure as codebuddy-cn
// (access + refresh token pair, ~1-year expiry) — same test strategy (#4232).
"codebuddy-intl": { tokenExists: true },
kimchi: {
url: KIMCHI_CONFIG.validationUrl || "https://api.cast.ai/v1/llm/openai/supported-providers",
method: "GET",
@@ -120,10 +124,10 @@ const OAUTH_TEST_CONFIG = {
extraHeaders: {
Accept: "application/json",
...(PROVIDERS["grok-cli"]?.headers || {
"User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)",
"User-Agent": GROK_CLI_PAGER_USER_AGENT,
"x-xai-token-auth": "xai-grok-cli",
"x-grok-client-identifier": "grok-pager",
"x-grok-client-version": "0.2.93",
"x-grok-client-version": GROK_CLI_VERSION,
}),
},
refreshable: true,
@@ -134,6 +138,15 @@ const OAUTH_TEST_CONFIG = {
402: "Connected, but Grok Build credits are exhausted (spending limit). Add credits or upgrade SuperGrok.",
},
},
// Muse Code subscription — probe /v1/models with the minted LLM|… key
"muse": {
url: "https://api.meta.ai/v1/models",
method: "GET",
authHeader: "Authorization",
authPrefix: "Bearer ",
extraHeaders: { "x-api-version": "1.0.0" },
refreshable: false,
},
};
/**
@@ -700,7 +713,8 @@ async function testApiKeyConnection(connection, effectiveProxy = null) {
case "dahl":
case "atria":
case "agnes":
case "bai": {
case "bai":
case "muse": {
const cfg = PROVIDERS[connection.provider];
const res = await fetchWithConnectionProxy(cfg.validateUrl, { headers: { Authorization: `Bearer ${connection.apiKey}` } }, effectiveProxy);
return { valid: res.ok, error: res.ok ? null : "Invalid API key" };
+11
View File
@@ -3,6 +3,7 @@ import "open-sse/index.js";
import { getProviderConnectionById, updateProviderConnection } from "@/lib/localDb";
import { getUsageForProvider } from "open-sse/services/usage.js";
import { isUnrecoverableRefreshError } from "open-sse/services/tokenRefresh.js";
import { getExecutor } from "open-sse/executors/index.js";
import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy";
import { USAGE_APIKEY_PROVIDERS } from "@/shared/constants/providers";
@@ -21,6 +22,11 @@ function isAuthExpiredMessage(usage) {
* @returns Promise<{ connection, refreshed: boolean }>
*/
export async function refreshAndUpdateCredentials(connection, force = false, proxyOptions = null) {
// Re-read latest tokens: OpenAI rotates the refresh token on every refresh, and
// refreshing with a stale snapshot (reuse) revokes the whole session → account logout.
const latest = connection.id ? await getProviderConnectionById(connection.id) : null;
if (latest) connection = latest;
const executor = getExecutor(connection.provider);
// Build credentials object from connection
@@ -47,6 +53,11 @@ export async function refreshAndUpdateCredentials(connection, force = false, pro
// Use executor's refreshCredentials method (with optional proxy)
const refreshResult = await executor.refreshCredentials(credentials, console, proxyOptions);
// Refresh token reused/invalidated — token family is revoked; do not continue with the dead token.
if (refreshResult && isUnrecoverableRefreshError(refreshResult)) {
throw new Error("Refresh token invalid or reused. Please re-authorize the connection.");
}
if (!refreshResult) {
// Refresh failed but we still have an accessToken — try with existing token
if (connection.accessToken) {
+70 -5
View File
@@ -1,5 +1,6 @@
import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS, getModelKind } from "@/shared/constants/models";
import {
ALIAS_TO_ID,
AI_PROVIDERS,
getProviderAlias,
isAnthropicCompatibleProvider,
@@ -41,6 +42,22 @@ async function resolveQoderLiveModels(conn, provider) {
return { models: models.map((m) => ({ id: m.id, name: m.name })) };
}
// Combo seats use UI aliases; the model registry also has transport aliases.
// Capability overrides and catalog limits are keyed by provider id.
const ALIAS_TO_PROVIDER_ID = {
...Object.fromEntries(
Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id])
),
...ALIAS_TO_ID,
};
function comboSeatCapabilities(seat) {
const slash = seat.indexOf("/");
if (slash <= 0) return null;
const alias = seat.slice(0, slash);
return getCapabilitiesForModel(ALIAS_TO_PROVIDER_ID[alias] || alias, seat.slice(slash + 1));
}
// Per-provider live model resolvers. Each receives a connection record and
// returns { models: [{ id, name? }, ...] } | null on failure.
// Adding a provider here makes /v1/models prefer the live catalog for it.
@@ -255,6 +272,47 @@ function comboMatchesKinds(combo, kindFilter) {
return kindFilter.includes(kind);
}
// Nested combo names are valid seats — the model selector exposes them and
// chat routing resolves them recursively — but a no-slash seat is otherwise
// treated as a literal model and publishes the 200k floor. Expand nested
// names (cycle-guarded) so the published window is the true min across the
// whole chain.
function comboSeatLimits(combo, combosByName, visiting = new Set()) {
const name = typeof combo?.name === "string" ? combo.name : null;
if (name) {
if (visiting.has(name)) return { contextWindow: undefined, maxOutput: undefined };
visiting.add(name);
}
let contextWindow = Infinity;
let maxOutput = Infinity;
try {
for (const seat of Array.isArray(combo?.models) ? combo.models : []) {
if (typeof seat !== "string") continue;
const slash = seat.indexOf("/");
if (slash <= 0) {
const nested = combosByName.get(seat);
if (nested) {
const nestedLimits = comboSeatLimits(nested, combosByName, visiting);
if (Number.isFinite(nestedLimits.contextWindow)) contextWindow = Math.min(contextWindow, nestedLimits.contextWindow);
if (Number.isFinite(nestedLimits.maxOutput)) maxOutput = Math.min(maxOutput, nestedLimits.maxOutput);
continue;
}
}
const caps = comboSeatCapabilities(seat) || getCapabilitiesForModel(null, seat);
if (Number.isFinite(caps?.contextWindow)) contextWindow = Math.min(contextWindow, caps.contextWindow);
if (Number.isFinite(caps?.maxOutput)) maxOutput = Math.min(maxOutput, caps.maxOutput);
}
} finally {
if (name) visiting.delete(name);
}
return {
contextWindow: Number.isFinite(contextWindow) ? contextWindow : undefined,
maxOutput: Number.isFinite(maxOutput) ? maxOutput : undefined,
};
}
/**
* Build OpenAI-format models list filtered by service kinds.
* @param {string[]} kindFilter - List of service kinds to include (e.g. ["llm"], ["webSearch","webFetch"]).
@@ -311,6 +369,9 @@ export async function buildModelsList(kindFilter, options = {}) {
}
const models = [];
const combosByName = new Map(
combos.filter((c) => typeof c?.name === "string").map((c) => [c.name, c]),
);
// Lookup map so aggregateComboCapabilities can recursively resolve nested combos
const comboByName = Object.fromEntries(combos.map((c) => [c.name, c.models]));
@@ -326,19 +387,23 @@ export async function buildModelsList(kindFilter, options = {}) {
if (combo.kind === "webSearch" || combo.kind === "webFetch") {
entry.kind = combo.kind;
} else {
const comboCaps = aggregateComboCapabilities(combo.models, comboByName);
const comboCaps = aggregateComboCapabilities(combo.models, comboByName, comboSeatCapabilities);
if (comboCaps) entry.capabilities = comboCaps;
// Any seat can serve the request, so the only window a combo can promise is
// its smallest. Combo entries were the only models on this endpoint that
// published no limits at all, which leaves a client to guess from the name —
// and it guesses high (see the snake_case note on the per-provider path).
const { contextWindow, maxOutput } = comboSeatLimits(combo, combosByName);
if (Number.isFinite(contextWindow)) entry.context_length = contextWindow;
if (Number.isFinite(maxOutput)) entry.max_completion_tokens = maxOutput;
}
models.push(entry);
}
if (connections.length === 0) {
// DB unavailable -> return static models, filtered by per-model kind
const aliasToProviderId = Object.fromEntries(
Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id])
);
for (const [alias, providerModels] of Object.entries(PROVIDER_MODELS)) {
const providerId = aliasToProviderId[alias] || alias;
const providerId = ALIAS_TO_PROVIDER_ID[alias] || alias;
if (!providerMatchesKinds(providerId, kindFilter)) continue;
for (const model of providerModels) {
if (!kindFilter.includes(modelKind(model))) continue;
+2
View File
@@ -65,6 +65,8 @@ const DEFAULT_SETTINGS = {
pxpipeAutoInstall: true,
pxpipeMinChars: 25000,
pxpipeTimeoutMs: 15000,
// Per-provider user header overrides applied at dispatch: { [providerId]: { headers: {..} } }
providerOverrides: {},
};
async function readRaw() {
+1
View File
@@ -24,6 +24,7 @@ const LIMIT_TOLERANCE = 0.1;
// while building rather than on every lookup. Providers absent here keep whatever
// the local pattern table resolves; names that already match need no entry.
export const PROVIDER_ALIASES = {
"github": "github-copilot",
"glm": "zai",
"glm-cn": "zhipuai",
"claude": "anthropic",
+12
View File
@@ -77,6 +77,12 @@ export async function resolveConnectionProxyConfig(
const legacy = normalizeLegacyProxy(providerSpecificData);
// A strict pool must keep its guarantee even when the pool itself is not
// usable (inactive, or saved without a url). Otherwise the unusable-pool
// path below reports strictProxy:false and the request silently leaves
// over the direct IP — the leak strict mode exists to prevent (#4333).
let poolStrictProxy = false;
/**
* -----------------------------
* Proxy Pool Resolution
@@ -93,6 +99,8 @@ export async function resolveConnectionProxyConfig(
proxyPool.isActive === true &&
proxyUrl;
poolStrictProxy = proxyPool?.strictProxy === true;
if (isValidPool) {
/**
* Vercel/Cloudflare relay proxies use base URL rewriting
@@ -148,6 +156,8 @@ export async function resolveConnectionProxyConfig(
proxyPoolId: proxyPoolId || null,
proxyPool: null,
strictProxy: poolStrictProxy,
...legacy,
};
}
@@ -163,6 +173,8 @@ export async function resolveConnectionProxyConfig(
proxyPoolId: proxyPoolId || null,
proxyPool: null,
strictProxy: poolStrictProxy,
...legacy,
};
} catch (error) {
+12
View File
@@ -127,6 +127,10 @@ export const KIMCHI_CONFIG = { ...PROVIDER_OAUTH["kimchi"] };
// Endpoint: cli-chat-proxy.grok.com — same client_id as xai, different flow + scopes
export const GROK_CLI_CONFIG = { ...PROVIDER_OAUTH["grok-cli"] };
// Muse — subscription device code flow to auth.meta.com, no refresh
// (Meta rejects refresh_token grants; the minted Model API key never expires).
export const MUSE_CONFIG = { ...PROVIDER_OAUTH["muse"] };
// Trae (ByteDance marscode) OAuth — authorization_code flow with local callback.
// 1) POST GetLoginGuidance {loginTraceID} → {Result.LoginHost}
// 2) Browser opens ${loginHost}/authorization?client_id=...&login_trace_id=...&auth_callback_url=${cb}
@@ -201,6 +205,13 @@ export const WINDSURF_CONFIG = {
oauthTimeoutMs: 600_000,
};
// GLM Coding (Z.ai) OAuth — ZCode CLI polling flow (NOT PKCE): init mints a
// one-off poll token, the browser opens the server-generated authorize_url,
// poll/ready returns the tokens. The Z.AI OAuth token is then exchanged for a
// platform business JWT and finally a long-lived coding-plan API key (no
// refresh grant).
export const GLM_OAUTH_CONFIG = { ...PROVIDER_OAUTH["glm"] };
// Zed hosted LLM aggregator — RSA keypair native-app auth (NOT OAuth).
// Client generates ephemeral RSA-2048 keypair; user signs in at zed.dev/native_app_signin;
// Zed redirects to local callback with access_token RSA-encrypted against our public key.
@@ -241,5 +252,6 @@ export const PROVIDERS = {
GROK_CLI: "grok-cli",
TRAE: "trae",
WINDSURF: "windsurf",
GLM: "glm",
ZED: "zed",
};
+305
View File
@@ -0,0 +1,305 @@
import crypto from "crypto";
import { GLM_OAUTH_CONFIG } from "../constants/oauth.js";
// Zai GLM Coding OAuth — CLI polling flow (mirrors the official
// ZCode CLI, apps/zcode-cli packages/adapters/src/auth/cli-oauth.ts +
// coding-plan-api-key.ts). No PKCE and no local callback server:
//
// 1) POST {cliInitUrl} Authorization: Bearer <pollToken> {"provider":"zai"}
// → { code: 0, data: { authorize_url, flow_id, poll_interval_sec, expires_at } }
// 2) Browser opens authorize_url; user signs in with the Z.ai account
// 3) GET {cliPollUrl}/<flow_id> Authorization: Bearer <pollToken>
// → { data: { status: "pending" } } until
// { data: { status: "ready", token, user, accessToken, refreshToken? } }
// 4) accessToken (Z.AI OAuth token) → POST {businessLoginUrl} {"token": ...}
// → { data: { access_token } } (platform business JWT)
// 5) Business JWT → coding-plan API key via getCustomerInfo → api_keys
// list/create("zcode-api-key") → copy → "apiKey.secretKey"
//
// The coding-plan API key is the long-lived model credential; the ZAI OAuth
// provider has no refresh_token grant, so expiry means re-login (same as the
// official CLI). zcode JWT + business token ride along in providerSpecificData
// for quota/usage and debugging.
const glm = {
config: GLM_OAUTH_CONFIG,
flowType: "device_code",
requestDeviceCode: async (config) => {
const pollToken = crypto.randomBytes(32).toString("hex");
const response = await fetch(config.cliInitUrl, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${pollToken}`,
},
body: JSON.stringify({ provider: config.providerId || "zai" }),
});
if (!response.ok) {
const error = await response.text();
throw new Error(`ZCode OAuth init failed: ${error}`);
}
const payload = await response.json();
if (!isSuccessCode(payload.code) || !payload.data) {
throw new Error(payload.msg || "ZCode OAuth init returned no data");
}
const data = payload.data;
if (!data.flow_id || !data.authorize_url) {
throw new Error("ZCode OAuth init response missing flow_id/authorize_url");
}
return {
device_code: data.flow_id,
verification_uri: data.authorize_url,
// expires_at is upstream-absolute; surface a relative deadline for the UI
expires_in: relativeSeconds(data.expires_at) ?? 300,
interval: data.poll_interval_sec || 3,
_zcodePollToken: pollToken,
};
},
pollToken: async (config, deviceCode, _codeVerifier, extraData) => {
const pollToken = extraData?._zcodePollToken;
if (!pollToken) {
return {
ok: true,
data: {
error: "access_denied",
error_description: "Missing ZCode poll token — restart the login flow",
},
};
}
const response = await fetch(`${config.cliPollUrl}/${encodeURIComponent(deviceCode)}`, {
headers: { Authorization: `Bearer ${pollToken}` },
});
if (!response.ok) {
return {
ok: true,
data: {
error: "access_denied",
error_description: `ZCode poll failed (HTTP ${response.status})`,
},
};
}
const payload = await response.json();
if (!isSuccessCode(payload.code)) {
return {
ok: true,
data: { error: "access_denied", error_description: payload.msg || "ZCode poll failed" },
};
}
const data = payload.data || {};
if (data.status === "pending") {
return { ok: true, data: { error: "authorization_pending" } };
}
if (data.status === "failed") {
return {
ok: true,
data: {
error: "access_denied",
error_description: "ZCode authorization failed or was cancelled",
},
};
}
if (data.status !== "ready") {
return {
ok: true,
data: { error: "authorization_pending", error_description: `Unknown status: ${data.status}` },
};
}
// ready payload nests the ZAI OAuth tokens under data[providerId] (see
// apps/zcode-cli cli-oauth.ts parseReadyData): { status:"ready", token,
// user, zai: { access_token, refresh_token? } }. Fall back to top-level
// fields for resilience against payload drift.
const providerData = data[config.providerId] || data[data.providerId] || {};
const zaiAccessToken =
providerData.access_token ||
providerData.accessToken ||
data.accessToken ||
data.access_token;
if (!zaiAccessToken) {
return {
ok: true,
data: {
error: "access_denied",
error_description: "ZCode poll response missing access token",
},
};
}
// ready.accessToken is the Z.AI OAuth token — derive the coding-plan API key
const { planApiKey, businessToken } = await resolveCodingPlanApiKey(config, zaiAccessToken);
return {
ok: true,
data: {
access_token: planApiKey,
_zcodeJwtToken: data.token || "",
_zaiBusinessToken: businessToken,
_zaiRefreshToken:
providerData.refresh_token || providerData.refreshToken || data.refresh_token || data.refreshToken || "",
_zcodeUser: data.user || {},
},
};
},
mapTokens: (tokens) => {
const user = tokens._zcodeUser || {};
const displayName = user.name || user.email || null;
return {
accessToken: tokens.access_token,
refreshToken: null,
email: user.email || null,
...(displayName ? { displayName } : {}),
providerSpecificData: {
authMethod: "cli_poll",
username: user.name || undefined,
userId: user.user_id || undefined,
zcodeJwtToken: tokens._zcodeJwtToken || undefined,
zaiBusinessToken: tokens._zaiBusinessToken || undefined,
...(tokens._zaiRefreshToken ? { zaiRefreshToken: tokens._zaiRefreshToken } : {}),
},
};
},
};
// Business JWT → coding-plan API key ("apiKey.secretKey"). Mirrors ZCode CLI
// coding-plan-api-key.ts: getCustomerInfo → default org/project → api_keys
// list/create("zcode-api-key") → copy → secretKey.
async function resolveCodingPlanApiKey(config, zaiAccessToken) {
const businessToken = await exchangeBusinessToken(config, zaiAccessToken);
const authHeaders = {
Authorization: `Bearer ${businessToken}`,
"Content-Type": "application/json",
};
const customerInfo = await fetchBusinessJson(
`${config.apiBaseUrl}/api/biz/customer/getCustomerInfo`,
{ headers: authHeaders },
"customer info"
);
const location = pickOrgAndProject(customerInfo);
if (!location) {
throw new Error("Unable to resolve Z.ai organization and project for the coding plan");
}
const listUrl =
`${config.apiBaseUrl}/api/biz/v1/organization/${location.organizationId}` +
`/projects/${location.projectId}/api_keys`;
const keys = (await fetchBusinessJson(listUrl, { headers: authHeaders }, "api keys")) || [];
let keyEntry = Array.isArray(keys)
? keys.find((item) => item?.name === config.planApiKeyName)
: null;
if (!keyEntry) {
keyEntry = await fetchBusinessJson(
listUrl,
{
method: "POST",
headers: authHeaders,
body: JSON.stringify({ name: config.planApiKeyName }),
},
"api key create"
);
}
const apiKey = keyEntry?.apiKey?.trim();
if (!apiKey) {
throw new Error("Z.ai api_keys response is missing apiKey");
}
const secret = await fetchBusinessJson(
`${listUrl}/copy/${encodeURIComponent(apiKey)}`,
{ headers: authHeaders },
"api key copy"
);
const secretKey = secret?.secretKey?.trim();
if (!secretKey) {
throw new Error("Z.ai api key copy response is missing secretKey");
}
return { planApiKey: `${apiKey}.${secretKey}`, businessToken };
}
// POST {businessLoginUrl} {"token": <zai oauth token>} → { data: { access_token } }
async function exchangeBusinessToken(config, zaiAccessToken) {
const payload = await fetchBusinessJson(
config.businessLoginUrl,
{
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ token: zaiAccessToken }),
},
"Z.ai business login"
);
const token = payload?.access_token?.trim() || payload?.accessToken?.trim();
if (!token) {
throw new Error("Z.ai business login response is missing access_token");
}
return token;
}
// Business endpoints answer {code, msg, data}; code 0/200 (or absent) = success.
// data is returned directly (null when missing).
async function fetchBusinessJson(url, options, label) {
const response = await fetch(url, options);
const text = await response.text();
if (!response.ok) {
throw new Error(`Z.ai ${label} request failed (HTTP ${response.status}): ${text.slice(0, 200)}`);
}
let payload;
try {
payload = JSON.parse(text);
} catch {
throw new Error(`Z.ai ${label} response is not valid JSON`);
}
if (!isSuccessCode(payload?.code) || payload?.success === false) {
throw new Error(payload?.msg || `Z.ai ${label} returned business error ${payload?.code}`);
}
return payload?.data ?? payload ?? null;
}
// Prefer the org named "默认机构"/"default" and the non-team project named
// "默认项目"/"default" (projectType "2" = team), falling back to the first entries.
function pickOrgAndProject(customerInfo) {
const organizations = Array.isArray(customerInfo?.organizations)
? customerInfo.organizations
: [];
const personalOrgs = organizations
.map((organization) => ({
organization,
projects: (organization?.projects || []).filter(
(project) => String(project?.projectType ?? "").trim() !== "2"
),
}))
.filter(({ organization, projects }) =>
Boolean(organization?.organizationId && projects.length)
);
if (!personalOrgs.length) return null;
const org =
personalOrgs.find(({ organization }) => isDefaultName(organization.organizationName)) ||
personalOrgs[0];
const project =
org.projects.find((item) => isDefaultName(item?.projectName)) || org.projects[0];
if (!org.organization?.organizationId || !project?.projectId) return null;
return { organizationId: org.organization.organizationId, projectId: project.projectId };
}
function isDefaultName(name) {
const normalized = String(name || "").trim().toLowerCase();
return normalized.includes("默认机构") || normalized.includes("默认项目") || normalized === "default";
}
function isSuccessCode(code) {
return code === undefined || code === null || code === 0 || code === 200 || code === "0" || code === "200";
}
// Absolute epoch (s or ms) → seconds from now; null when absent/invalid.
function relativeSeconds(expiresAt) {
const raw = Number(expiresAt);
if (!Number.isFinite(raw) || raw <= 0) return null;
const ms = raw > 1e12 ? raw : raw * 1000;
const seconds = Math.floor((ms - Date.now()) / 1000);
return seconds > 0 ? seconds : null;
}
export default glm;
+5 -4
View File
@@ -1,3 +1,4 @@
import { GROK_CLI_PAGER_USER_AGENT, GROK_CLI_VERSION } from "open-sse/config/grokCli.js";
import { GROK_CLI_CONFIG } from "../constants/oauth.js";
import { decodeXaiIdTokenEmail, extractEmailFromAccessToken } from "../providerHelpers.js";
@@ -18,7 +19,7 @@ const grokCli = {
headers: {
"Content-Type": "application/x-www-form-urlencoded",
Accept: "application/json",
"User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)",
"User-Agent": GROK_CLI_PAGER_USER_AGENT,
},
body,
});
@@ -36,7 +37,7 @@ const grokCli = {
headers: {
"Content-Type": "application/x-www-form-urlencoded",
Accept: "application/json",
"User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)",
"User-Agent": GROK_CLI_PAGER_USER_AGENT,
},
body: new URLSearchParams({
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
@@ -69,9 +70,9 @@ const grokCli = {
headers: {
Authorization: `Bearer ${tokens.access_token}`,
Accept: "application/json",
"User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)",
"User-Agent": GROK_CLI_PAGER_USER_AGENT,
"x-xai-token-auth": "xai-grok-cli",
"x-grok-client-version": "0.2.93",
"x-grok-client-version": GROK_CLI_VERSION,
},
});
if (res.ok) return { user: await res.json() };
+12
View File
@@ -8,6 +8,7 @@ import claude from "./claude.js";
import codex from "./codex.js";
import xai from "./xai.js";
import grokCli from "./grok-cli.js";
import muse from "./muse.js";
import geminiCli from "./gemini-cli.js";
import antigravity from "./antigravity.js";
import iflow from "./iflow.js";
@@ -27,6 +28,7 @@ import kimchi from "./kimchi.js";
import trae from "./trae.js";
import windsurf from "./windsurf.js";
import zed from "./zed.js";
import glm from "./glm.js";
// Provider configurations
const PROVIDERS = {
@@ -34,6 +36,7 @@ const PROVIDERS = {
codex,
xai,
"grok-cli": grokCli,
muse,
"gemini-cli": geminiCli,
antigravity,
iflow,
@@ -53,6 +56,7 @@ const PROVIDERS = {
trae,
windsurf,
zed,
glm,
};
export { PROVIDERS };
@@ -174,7 +178,15 @@ export async function pollForToken(providerName, deviceCode, codeVerifier, extra
// Call postExchange to get additional data (copilotToken, userInfo, etc.)
let extra = null;
if (provider.postExchange) {
try {
extra = await provider.postExchange(result.data);
} catch (err) {
// The grant succeeded but post-login exchange failed (e.g. Muse key
// mint). The device code is one-shot, so re-polling can never
// recover — surface as fatal so the client stops and shows the error.
console.warn(`[oauth] ${providerName} postExchange failed:`, err?.message || err);
return { success: false, error: "exchange_failed", errorDescription: err.message, fatal: true };
}
}
const tokens = provider.mapTokens(result.data, extra);
// Kiro IDC/Builder-ID tokens lack profileArn; resolve it to avoid 403
+132
View File
@@ -0,0 +1,132 @@
import { MUSE_CONFIG } from "../constants/oauth.js";
// Muse Code subscription — Meta account device code flow to auth.meta.com,
// then mint the Model API key (LLM|…) the chat transport actually uses.
const MUSE_KEY_URL = "https://api.meta.ai/muse-code/key";
const API_VERSION = "1.0.0";
const muse = {
config: MUSE_CONFIG,
flowType: "device_code",
requestDeviceCode: async (config) => {
const response = await fetch(config.deviceCodeUrl, {
method: "POST",
headers: {
"Content-Type": "application/x-www-form-urlencoded",
Accept: "application/json",
"x-api-version": API_VERSION,
},
body: new URLSearchParams({ client_id: config.clientId }),
});
if (!response.ok) {
const error = await response.text();
throw new Error(`Muse Code device code request failed: ${error}`);
}
return await response.json();
},
pollToken: async (config, deviceCode) => {
const response = await fetch(config.tokenUrl, {
method: "POST",
headers: {
"Content-Type": "application/x-www-form-urlencoded",
Accept: "application/json",
"x-api-version": API_VERSION,
},
body: new URLSearchParams({
grant_type: "urn:ietf:params:oauth:grant-type:device_code",
device_code: deviceCode,
client_id: config.clientId,
}),
});
let data;
try {
data = await response.json();
} catch {
const text = await response.text();
data = { error: "invalid_response", error_description: text };
}
const pending =
data?.error === "authorization_pending" ||
data?.error === "slow_down";
return { ok: response.ok || pending, data };
},
postExchange: async (tokens) => {
// Mint the subscription API key; onboard:true enrolls the account on first
// login. The endpoint is aggressively rate-limited (429) and the device code
// is one-shot, so retry transient failures here instead of failing the login.
let response;
for (let attempt = 0; ; attempt++) {
response = await fetch(MUSE_KEY_URL, {
method: "POST",
headers: {
Accept: "application/json",
Authorization: `Bearer ${tokens.access_token}`,
"Content-Type": "application/json",
"x-api-version": API_VERSION,
},
body: JSON.stringify({ onboard: true }),
redirect: "error",
});
const transient = response.status === 429 || response.status >= 500;
if (!transient || attempt >= 2) break;
await new Promise((r) => setTimeout(r, 5000 * (attempt + 1)));
}
const text = await response.text();
if (!response.ok) {
// Meta's error envelope (e.g. 403 code 4705002) carries the fix-it URL
let msg = `${response.status} ${text.slice(0, 200)}`;
try {
const err = JSON.parse(text);
if (err?.title || err?.detail) {
msg = [err.title, err.detail].filter(Boolean).join(": ");
if (err.action_url) msg += ` — ${err.action_url}`;
}
} catch { /* non-JSON error body */ }
throw new Error(`Muse Code key mint failed: ${msg}`);
}
let payload;
try {
payload = JSON.parse(text);
} catch {
throw new Error("Muse Code key mint returned invalid JSON");
}
if (payload.is_subs_active === false) {
throw new Error("Muse Code subscription is inactive — activate it on muse.ai first");
}
const actionUrl = payload.action_url || payload.require_payment_action_url;
if (!payload.api_key && (payload.require_payment || actionUrl)) {
throw new Error(`Muse Code subscription required${actionUrl ? `: ${actionUrl}` : ""}`);
}
if (!payload.api_key) {
throw new Error("Muse Code key response is missing api_key");
}
return { key: payload };
},
mapTokens: (tokens, extra) => {
const payload = extra?.key || {};
// Chat requests carry the minted Model API key, not the Meta account token.
// No expiry/refresh from Meta — a dead key means re-login.
return {
accessToken: payload.api_key,
refreshToken: null,
expiresIn: null,
email: payload.user_email?.trim().toLowerCase() || undefined,
providerSpecificData: {
authMethod: "device_code",
// Kept so a future re-mint can run without another device login
oauthAccessToken: tokens.access_token,
subscriptionTier: payload.subs_tier_name || null,
},
};
},
};
export default muse;
+6 -2
View File
@@ -22,7 +22,9 @@ const NO_AUTH_PROVIDER_IDS = Object.keys(FREE_PROVIDERS).filter(id => FREE_PROVI
// Providers with per-account live catalogs via /api/providers/[id]/models.
// Static registry stays as fallback when live fetch fails or is empty.
const LIVE_CATALOG_PROVIDERS = ["cursor", "cline", "clinepass"];
// zed added in #4244: its backend customResolver already returns live models
// but the frontend omitted it, hiding Zed entirely from the Combo picker.
const LIVE_CATALOG_PROVIDERS = ["cursor", "cline", "clinepass", "zed"];
// Fetch a provider's account-scoped catalog for every active connection and merge
// the results. Entries collapse by model id on purpose: two connections of the
@@ -113,10 +115,12 @@ export default function ModelSelectModal({
const cursorConnectionIds = liveConnectionIdsByProvider.cursor;
const clineConnectionIds = liveConnectionIdsByProvider.cline;
const clinepassConnectionIds = liveConnectionIdsByProvider.clinepass;
const zedConnectionIds = liveConnectionIdsByProvider.zed;
const cursorModels = useLiveProviderModels(isOpen, cursorConnectionIds, "Cursor");
const clineModels = useLiveProviderModels(isOpen, clineConnectionIds, "Cline");
const clinepassModels = useLiveProviderModels(isOpen, clinepassConnectionIds, "ClinePass");
const zedModels = useLiveProviderModels(isOpen, zedConnectionIds, "Zed");
const fetchCombos = async () => {
try {
@@ -349,7 +353,7 @@ export default function ModelSelectModal({
hasModels: mergedModels.length > 0,
};
} else {
const liveModels = providerId === "cursor" ? cursorModels : providerId === "cline" ? clineModels : providerId === "clinepass" ? clinepassModels : [];
const liveModels = providerId === "cursor" ? cursorModels : providerId === "cline" ? clineModels : providerId === "clinepass" ? clinepassModels : providerId === "zed" ? zedModels : [];
const hardcodedModels = liveModels.length > 0
? liveModels
: getModelsByProviderId(providerId);
+7 -1
View File
@@ -175,7 +175,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess,
return;
}
if (data.error === "expired_token" || data.error === "access_denied") {
if (data.error === "expired_token" || data.error === "access_denied" || data.fatal) {
throw new Error(data.errorDescription || data.error);
}
@@ -291,6 +291,8 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess,
"qoder",
"qoder-cn",
"grok-cli",
"muse",
"glm",
];
if (deviceCodeProviders.includes(provider)) {
setIsDeviceCode(true);
@@ -333,6 +335,8 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess,
}
: (provider === "kimi" || provider === "kimi-coding")
? { _kimiDeviceId: data._kimiDeviceId }
: provider === "glm"
? { _zcodePollToken: data._zcodePollToken }
: null;
startPolling(
data.device_code,
@@ -923,6 +927,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess,
</Button>
</div>
</div>
{deviceData.user_code && (
<div className="bg-primary/10 p-4 rounded-lg">
<p className="text-xs text-text-muted mb-1">Your Code</p>
<div className="flex items-center justify-center gap-2">
@@ -935,6 +940,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess,
/>
</div>
</div>
)}
</div>
{polling && (
<div className="flex items-center justify-center gap-2 text-sm text-text-muted">
+2 -8
View File
@@ -203,9 +203,6 @@ export default function Sidebar({ onClose }) {
>
<span className="material-symbols-outlined text-[18px]">perm_media</span>
<span className="text-[13px] font-medium flex-1 text-left">Media Providers</span>
{MEDIA_PROVIDER_KINDS.some((k) => VISIBLE_MEDIA_KINDS.includes(k.id) && k.isNew) && (
<span className="text-[10px] font-semibold px-1.5 py-0.5 rounded-[3px] bg-green-500/15 text-green-400">NEW</span>
)}
<span className="material-symbols-outlined text-[14px] transition-transform" style={{ transform: mediaOpen ? "rotate(180deg)" : "rotate(0deg)" }}>
expand_more
</span>
@@ -226,9 +223,6 @@ export default function Sidebar({ onClose }) {
>
<span className="material-symbols-outlined text-[16px]">{kind.icon}</span>
<span className="text-sm">{kind.label}</span>
{kind.isNew && (
<span className="ml-auto text-[10px] font-semibold px-1.5 py-0.5 rounded-[3px] bg-green-500/15 text-green-400">NEW</span>
)}
</Link>
))}
<Link
@@ -312,8 +306,8 @@ export default function Sidebar({ onClose }) {
computer
</span>
<span className="text-[13px] font-medium">9Remote</span>
<span className="ml-auto text-[10px] font-semibold px-1.5 py-0.5 rounded-[3px] bg-green-500/15 text-green-400">
NEW
<span className="ml-auto text-[10px] font-semibold px-1.5 py-0.5 rounded-[3px] bg-orange-500/15 text-orange-400">
HOT
</span>
</button>
+2 -1
View File
@@ -368,6 +368,7 @@ export default function UsageStats({
.filter((c) => {
if (c.isActive === false) return false;
if (!isLLMProvider(c.provider)) return false;
if (AI_PROVIDERS[c.provider]?.hidden) return false;
if (seen.has(c.provider)) return false;
seen.add(c.provider);
return true;
@@ -377,7 +378,7 @@ export default function UsageStats({
nodeName: nodeNameMap[c.provider] || null,
}));
const noAuthProviders = Object.values(FREE_PROVIDERS)
.filter((p) => p.noAuth && !seen.has(p.id) && isLLMProvider(p.id))
.filter((p) => p.noAuth && !p.hidden && !seen.has(p.id) && isLLMProvider(p.id))
.map((p) => ({ provider: p.id, name: p.name }));
setProviders([...unique, ...noAuthProviders]);
})
+1 -1
View File
@@ -1,6 +1,6 @@
// Provider definitions
import REGISTRY from "open-sse/providers/registry/index.js";
import { RISK_NOTICE } from "@/shared/constants/providersDisplay";
import { RISK_NOTICE } from "@/shared/constants/providersDisplay.js";
const MEDIA_ENTRY_KEYS = [
"serviceKinds", "ttsConfig", "sttConfig", "embeddingConfig",
+10 -3
View File
@@ -172,13 +172,15 @@ export async function handleChat(request, clientRawRequest = null) {
});
}
return handleSingleModelChat(body, modelStr, clientRawRequest, request, apiKey);
return handleSingleModelChat(body, modelStr, clientRawRequest, request, apiKey, contextMarker ? `${modelStr.slice(modelStr.indexOf("/") + 1)}[${contextMarker}]` : null);
}
/**
* Handle single model chat request
*/
async function handleSingleModelChat(body, modelStr, clientRawRequest = null, request = null, apiKey = null, opts = {}) {
async function handleSingleModelChat(body, modelStr, clientRawRequest = null, request = null, apiKey = null, optsOrRequestedModel = {}) {
const opts = (optsOrRequestedModel && typeof optsOrRequestedModel === "object") ? optsOrRequestedModel : {};
const requestedModel = typeof optsOrRequestedModel === "string" ? optsOrRequestedModel : (opts.requestedModel || null);
// Dashboard per-key tests pin the account via x-connection-id (same header
// contract as embeddings/images/video). Pinned requests must not rotate.
const preferredConnectionId = request?.headers?.get("x-connection-id") || null;
@@ -268,7 +270,10 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re
let lastHeaders = null;
while (true) {
const credentials = await getProviderCredentials(provider, excludeConnectionIds, model, { preferredConnectionId });
const credentials = await getProviderCredentials(provider, excludeConnectionIds, model, {
preferredConnectionId,
requestedModel: requestedModel || model,
});
// All accounts unavailable
if (!credentials || credentials.allRateLimited) {
@@ -342,6 +347,8 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re
pxpipeTransform: chatSettings.pxpipeEnabled ? await getPxpipeTransform() : null,
onPxpipeEvent: appendPxpipeEvent,
providerThinking,
// Per-provider user overrides (custom headers / connect timeout) from settings
providerOverrides: (chatSettings.providerOverrides || {})[provider] || null,
capsOverride,
streamErrorPatterns: chatSettings.streamErrorPatterns || null,
persistUsage,
+4 -1
View File
@@ -31,6 +31,7 @@ export async function getProviderCredentials(provider, excludeConnectionIds = nu
? excludeConnectionIds
: (excludeConnectionIds ? new Set([excludeConnectionIds]) : new Set());
const preferredConnectionId = options?.preferredConnectionId || null;
const requestedModel = options?.requestedModel || model;
// Acquire mutex to prevent race conditions
const currentMutex = selectionMutex;
let resolveMutex;
@@ -94,6 +95,8 @@ export async function getProviderCredentials(provider, excludeConnectionIds = nu
const availableConnections = connections.filter(c => {
if (excludeSet.has(c.id)) return false;
if (isModelLockActive(c, model)) return false;
const enabled = c.providerSpecificData?.enabledModels;
if (providerId === "codex" && Array.isArray(enabled) && enabled.length && requestedModel && !enabled.includes(requestedModel)) return false;
// Antigravity: skip if live quota exhausted for this model
if (isAntigravity && model && antigravityQuotaCache) {
const quota = antigravityQuotaCache.get(c.id)?.[model];
@@ -274,7 +277,7 @@ export async function markAccountUnavailable(connectionId, status, errorText, pr
: Math.min(resetsAtMs - Date.now(), MAX_RATE_LIMIT_COOLDOWN_MS);
newBackoffLevel = 0;
} else {
({ shouldFallback, cooldownMs, newBackoffLevel } = checkFallbackError(status, errorText, backoffLevel));
({ shouldFallback, cooldownMs, newBackoffLevel } = checkFallbackError(status, errorText, backoffLevel, resolveProviderId(provider)));
}
// Request-scoped error (context overflow etc.): the same body fails on every
// credential — do not lock this account or rotate to the next one.
+20 -1
View File
@@ -1,6 +1,6 @@
// Re-export from open-sse with local logger
import * as log from "../utils/logger.js";
import { updateProviderConnection } from "../../lib/localDb.js";
import { getProviderConnectionById, updateProviderConnection } from "../../lib/localDb.js";
import {
getProjectIdForConnection,
invalidateProjectId,
@@ -227,6 +227,25 @@ export async function checkAndRefreshToken(provider, credentials, options = {})
creds.connectionId = creds.id;
}
// Adopt latest DB tokens: OpenAI rotates the refresh token on every refresh, and
// refreshing with a stale snapshot (reuse) revokes the whole session → account logout.
if (creds.connectionId) {
const latest = await getProviderConnectionById(creds.connectionId).catch(() => null);
const latestRefreshMs = Date.parse(latest?.lastRefreshAt || "");
const credsRefreshMs = Date.parse(creds.lastRefreshAt || "");
const dbIsNewer = Number.isFinite(latestRefreshMs)
&& (!Number.isFinite(credsRefreshMs) || latestRefreshMs > credsRefreshMs);
if (dbIsNewer && latest.refreshToken && latest.refreshToken !== creds.refreshToken) {
creds = {
...creds,
refreshToken: latest.refreshToken,
accessToken: latest.accessToken || creds.accessToken,
expiresAt: latest.expiresAt || latest.tokenExpiresAt || creds.expiresAt,
lastRefreshAt: latest.lastRefreshAt || creds.lastRefreshAt,
};
}
}
const force = options?.force === true;
// ── 1. Regular access-token expiry ────────────────────────────────────────
+10 -1
View File
@@ -116,7 +116,12 @@
"devin": "devin",
"devin-cli": "devin-cli",
"morph": "morph",
"morphllm": "morph"
"morphllm": "morph",
"muse": "muse",
"muse-ai": "muse",
"meta-model-api": "muse",
"muse-code": "muse",
"muse-subscription": "muse"
},
"idToAlias": {
"agnes": "agnes",
@@ -176,6 +181,7 @@
"mistral": "mistral",
"mmf": "mmf",
"morph": "morph",
"muse": "muse",
"nanobanana": "nanobanana",
"nebius": "nebius",
"nvidia": "nvidia",
@@ -211,6 +217,7 @@
"modelKeys": [
"af",
"ag",
"agnes",
"alicode",
"alicode-intl",
"alims-intl",
@@ -270,6 +277,7 @@
"mistral",
"mmf",
"morph",
"muse",
"nanobanana",
"nebius",
"nvidia",
@@ -302,6 +310,7 @@
"together",
"tokenharbor",
"tokenrouter",
"v1m",
"venice",
"vertex",
"vertex-partner",
+57 -12
View File
@@ -121,7 +121,9 @@
"usage": {
"oauthUrl": "https://api.anthropic.com/api/oauth/usage",
"orgUrl": "https://api.anthropic.com/v1/organizations/{org_id}/usage",
"settingsUrl": "https://api.anthropic.com/v1/settings"
"settingsUrl": "https://api.anthropic.com/v1/settings",
"profileUrl": "https://api.anthropic.com/api/oauth/profile",
"resetUrl": "https://api.anthropic.com/api/organizations/{org_id}/reset_rate_limits"
},
"clientId": "9d1c250a-e61b-44d9-88ed-5944d1962f5e",
"tokenUrl": "https://api.anthropic.com/v1/oauth/token"
@@ -199,10 +201,11 @@
"baseUrl": "https://chatgpt.com/backend-api/codex/responses",
"format": "openai-responses",
"forceStream": true,
"cliVersion": "0.154.0",
"cliVersion": "0.155.0",
"headers": {
"originator": "codex_cli_rs",
"User-Agent": "codex_cli_rs/0.154.0"
"User-Agent": "codex_cli_rs/0.155.0",
"version": "0.155.0"
},
"usage": {
"url": "https://chatgpt.com/backend-api/wham/usage",
@@ -414,13 +417,13 @@
"modelsUrl": "https://cli-chat-proxy.grok.com/v1/models",
"userUrl": "https://cli-chat-proxy.grok.com/v1/user",
"billingUrl": "https://cli-chat-proxy.grok.com/v1/billing",
"clientVersion": "0.2.99",
"clientVersion": "1.0.44",
"clientIdentifier": "grok-shell",
"tokenAuth": "xai-grok-cli",
"headers": {
"User-Agent": "grok-shell/0.2.99 (linux; x86_64)",
"User-Agent": "grok-shell/1.0.44 (linux; x86_64)",
"x-grok-client-identifier": "grok-shell",
"x-grok-client-version": "0.2.99"
"x-grok-client-version": "1.0.44"
},
"usage": {
"url": "https://cli-chat-proxy.grok.com/v1/billing?format=credits",
@@ -1086,6 +1089,14 @@
},
"format": "openai"
},
"tokenharbor": {
"baseUrl": "https://tokenharbor.ai/v1/chat/completions",
"validateUrl": "https://tokenharbor.ai/v1/models",
"retry": {
"429": 2
},
"format": "openai"
},
"dahl": {
"baseUrl": "https://inference.dahl.global/v1/chat/completions",
"validateUrl": "https://inference.dahl.global/v1/models",
@@ -1106,12 +1117,46 @@
"validateUrl": "https://api.b.ai/v1/models",
"format": "openai"
},
"tokenharbor": {
"baseUrl": "https://tokenharbor.ai/v1/chat/completions",
"validateUrl": "https://tokenharbor.ai/v1/models",
"retry": {
"429": 2
"muse": {
"baseUrl": "https://api.meta.ai/v1/chat/completions",
"validateUrl": "https://api.meta.ai/v1/models",
"modelsUrl": "https://api.meta.ai/v1/models",
"auth": {
"combined": true,
"header": "Authorization",
"scheme": "bearer",
"hooks": [
"museHeaders"
]
},
"format": "openai"
"format": "openai",
"clientId": "1031625952748946",
"tokenUrl": "https://auth.meta.com/oidc/device/token/",
"transports": [
{
"format": "openai",
"baseUrl": "https://api.meta.ai/v1/chat/completions",
"auth": {
"combined": true,
"header": "Authorization",
"scheme": "bearer",
"hooks": [
"museHeaders"
]
}
},
{
"format": "openai-responses",
"baseUrl": "https://api.meta.ai/v1/responses",
"auth": {
"combined": true,
"header": "Authorization",
"scheme": "bearer",
"hooks": [
"museHeaders"
]
}
}
]
}
}
+2
View File
@@ -23,6 +23,8 @@ const ALIAS_TOKENS = [
"af","airforce","api-airforce","llm7","llm-7","samba","sambanova","bm","bluesminds",
"bzl","bazaarlink","kgw","kilo-gateway","hunyuan","tencent","qianfan","baidu","ernie",
"dv","devin","devin-cli","morph","morphllm",
"muse","muse-ai","meta-model-api",
"muse-code","muse-subscription",
];
// Sort idToAlias by key — runtime accesses by key, order is irrelevant (content-based)
@@ -40,6 +40,9 @@ exports[`GOLDEN request: OpenAI → Claude > full body (system/image/tool/tool_r
{
"content": [
{
"cache_control": {
"type": "ephemeral",
},
"content": "sunny",
"tool_use_id": "call_1",
"type": "tool_result",
@@ -0,0 +1,238 @@
// REAL matrix: DeepSeek OFFICIAL Responses API — direct behavior vs 9router translation.
//
// Context: DeepSeek recently shipped an OpenAI-Responses-compatible endpoint
// (https://api.deepseek.com/responses) but the 9router registry does not declare it —
// responses-format requests are translated to /chat/completions. This suite answers:
//
// 1. DIRECT: does the official /responses endpoint demand reasoning pass-back
// (like official /chat/completions does: "reasoning_content must be passed back")?
// 2. 9ROUTER: when a responses-format client hits 9router, what does the translation
// to /chat/completions do with reasoning items — and does multi-turn survive the
// pass-back requirement on the chat endpoint?
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/deepseek-official-responses.real.test.js
//
// Reads the DeepSeek API key from the local 9router DB (connection id
// a91b07f2-878a-45b0-beb5-56981409ab0c). Uses handleChatCore for the 9router half.
import { describe, it, expect } from "vitest";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
import { openaiResponsesToOpenAIRequest } from "../../../open-sse/translator/request/openai-responses.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "deepseek";
const MODEL = "deepseek-reasoner";
const TIMEOUT_MS = 120000;
const CRED_ISSUE = [401, 402, 403, 429];
// API key from the local DB connection (no dashboard/DB writes — read-only).
function readApiKey() {
const Database = require("better-sqlite3");
const path = require("path");
const dbPath = path.join(process.env.APPDATA, "9router", "db", "data.sqlite");
const db = new Database(dbPath, { readonly: true });
const rows = db.prepare(
"SELECT data FROM providerConnections WHERE provider = ? AND isActive = 0 ORDER BY updatedAt DESC LIMIT 1"
).all("deepseek");
db.close();
if (!rows.length) return null;
try {
const data = JSON.parse(rows[0].data);
return data.apiKey || null;
} catch { return null; }
}
// ---- DIRECT half: raw fetch to the official /responses endpoint ----
async function directResponses(body) {
const res = await fetch("https://api.deepseek.com/responses", {
method: "POST",
headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" },
body: JSON.stringify(body),
});
const text = await res.text();
let json = null;
try { json = JSON.parse(text); } catch { /* keep null */ }
return { status: res.status, json, raw: text };
}
const TURN1_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17+26, then reply with ONLY the number." }] };
const TURN2_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer? Reply with just the number." }] };
async function directTurn1() {
const out = await directResponses({
model: MODEL, stream: false, max_output_tokens: 512,
reasoning: { effort: "high" },
input: [TURN1_USER],
});
return out;
}
describe.skipIf(!RUN_REAL)("DIRECT: DeepSeek official /responses endpoint", () => {
it("has a DeepSeek API key in the local DB", () => {
expect(process.env.DS_KEY && process.env.DS_KEY.startsWith("sk-")).toBe(true);
});
it("single-turn works and returns a Responses-shape payload", async () => {
const out = await directResponses({ model: MODEL, stream: false, input: [TURN1_USER] });
console.log(`[direct single] status=${out.status}`);
expect(out.status).toBe(200);
expect(out.json?.object).toBe("response");
expect(out.json?.output?.some((o) => o.type === "message")).toBe(true);
});
it("turn1 with reasoning returns a reasoning item", async () => {
const out = await directTurn1();
expect(out.status).toBe(200);
const types = (out.json?.output || []).map((o) => o.type);
console.log(`[direct turn1] output types=${types.join(",")} reasoning_tokens=${out.json?.usage?.output_tokens_details?.reasoning_tokens}`);
expect(types).toContain("reasoning");
});
it("turn2 WITHOUT reasoning item is accepted (no pass-back requirement)", async () => {
const t1 = await directTurn1();
const outMsg = (t1.json?.output || []).find((o) => o.type === "message");
const out = await directResponses({
model: MODEL, stream: false, max_output_tokens: 256,
input: [TURN1_USER, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER],
});
console.log(`[direct turn2 no-reasoning] status=${out.status}`);
expect(out.status).toBe(200);
});
it("turn2 WITH reasoning item is accepted", async () => {
const t1 = await directTurn1();
const reas = (t1.json?.output || []).find((o) => o.type === "reasoning");
const outMsg = (t1.json?.output || []).find((o) => o.type === "message");
const out = await directResponses({
model: MODEL, stream: false, max_output_tokens: 256,
input: [TURN1_USER, reas, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER],
});
console.log(`[direct turn2 with-reasoning] status=${out.status}`);
expect(out.status).toBe(200);
});
it("streaming single-turn returns Responses SSE events", async () => {
const res = await fetch("https://api.deepseek.com/responses", {
method: "POST",
headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" },
body: JSON.stringify({ model: MODEL, stream: true, max_output_tokens: 256, input: [TURN1_USER] }),
});
expect(res.status).toBe(200);
const text = await res.text();
const events = (text.match(/event: ([a-z_.]+)/g) || []).map((e) => e.slice(7));
const hasCreated = events.includes("response.created");
const hasCompleted = events.includes("response.completed");
console.log(`[direct stream] status=200 events=${events.join(",")}`);
expect(hasCreated).toBe(true);
expect(hasCompleted).toBe(true);
}, TIMEOUT_MS);
});
// ---- UNIT half: what the responses→chat translator does with reasoning ----
describe("UNIT: responses→chat translator reasoning handling", () => {
it("attaches reasoning item text as reasoning_content on the assistant message", () => {
const body = {
input: [
TURN1_USER,
{ type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
TURN2_USER,
],
};
const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null);
const assistant = result.messages.find((m) => m.role === "assistant");
expect(assistant.reasoning_content).toContain("43");
expect(result.messages.length).toBe(3);
});
it("leaves assistant messages bare when no reasoning item is present", () => {
const body = {
input: [TURN1_USER, { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, TURN2_USER],
};
const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null);
const assistant = result.messages.find((m) => m.role === "assistant");
expect(assistant.reasoning_content).toBeUndefined();
});
});
// ---- 9ROUTER half: handleChatCore with responses sourceFormat ----
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function via9router(body) {
const credentials = {
apiKey: process.env.DS_KEY,
connectionId: "a91b07f2-878a-45b0-beb5-56981409ab0c",
providerSpecificData: { connectionProxyEnabled: false, connectionProxyUrl: "", connectionNoProxy: "" },
};
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${MODEL}` },
modelInfo: { provider: PROVIDER, model: MODEL },
credentials,
connectionId: credentials.connectionId,
sourceFormatOverride: "openai-responses",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
describe.skipIf(!RUN_REAL)(`9ROUTER: responses-format → ${PROVIDER} translation`, () => {
it("single-turn responses request succeeds via /chat/completions", async () => {
const out = await via9router({
stream: true, max_output_tokens: 128, instructions: "You are concise.",
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
});
if (out.skip) return expect(true).toBe(true);
// 9router routes the request to /chat/completions but re-encodes the stream
// back to the client's source format (Responses SSE shape).
const isResponsesShape = /event: response\.|"type"\s*:\s*"response|"type":"response/.test(out.raw || "");
const isChatShape = /chat\.completion\.chunk|"delta"/.test(out.raw || "");
console.log(`[9r single] status=${out.status} bytes=${out.raw?.length} responsesShape=${isResponsesShape} chatShape=${isChatShape}`);
expect(out.ok).toBe(true);
expect(isResponsesShape || isChatShape).toBe(true);
}, TIMEOUT_MS);
it("multi-turn WITH reasoning item in history succeeds (translator attaches reasoning_content)", async () => {
const out = await via9router({
stream: true, max_output_tokens: 128,
input: [
TURN1_USER,
{ type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
TURN2_USER,
],
});
if (out.skip) return expect(true).toBe(true);
console.log(`[9r multi with-reasoning] status=${out.status} bytes=${out.raw?.length}`);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("multi-turn WITHOUT reasoning item in history (diagnostic: chat endpoint pass-back)", async () => {
const out = await via9router({
stream: true, max_output_tokens: 128,
input: [
TURN1_USER,
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
TURN2_USER,
],
});
if (out.skip) return expect(true).toBe(true);
console.log(`[9r multi no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// Diagnostic: the official chat endpoint requires reasoning_content pass-back;
// a 400 here proves the translation path needs the reasoning item to survive.
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,125 @@
// REAL endpoint matrix for opencode-go DeepSeek models.
//
// Verifies, against the live upstream (https://opencode.ai/zen/go), that each client
// request format actually WORKS for DeepSeek on the endpoint 9router routes it to:
//
// Claude (/v1/messages) → must return a Claude-shape SSE
// Codex (/v1/responses) → must return an OpenAI Responses-shape SSE
// OpenAI (/v1/chat/completions) → must return an OpenAI chat-shape SSE
//
// This is the live counterpart of tests/unit/opencode-go-transport-routing.test.js
// (which proves the routing decision offline). A cell here fails when the upstream
// rejects the routed endpoint+model combination — exactly the 400 that #3332 reported
// for Claude→deepseek-v4-flash on /messages.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-deepseek.real.test.js
//
// Requires an active opencode-go credential in the local DB (add via 9router dashboard).
// Skips (pass) only on credential/quota/plan rejections, mirroring the other .real tests.
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const TIMEOUT_MS = 90000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
// One request. Returns { raw } | "skip"; throws on a real upstream rejection.
async function runChat(model, body, sourceFormat) {
const credentials = await getProviderCredentials(PROVIDER, new Set(), model);
if (!credentials || credentials.allRateLimited) return "skip";
const refreshed = await checkAndRefreshToken(PROVIDER, credentials);
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${model}` },
modelInfo: { provider: PROVIDER, model },
credentials: refreshed,
connectionId: credentials.connectionId,
sourceFormatOverride: sourceFormat,
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status)) return "skip";
if (status >= 500 || status === 406) return "skip";
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return "skip";
throw new Error(`${PROVIDER}/${model} ${sourceFormat} [${result.status}]: ${result.error}`);
}
return { raw: await drainSSE(result.response) };
}
// SSE markers: response is re-encoded back to the client's source format.
const SSE_MARKER = {
openai: /chat\.completion\.chunk|"delta"|\[DONE\]/,
"openai-responses": /response\.|"type"\s*:\s*"response|\[DONE\]/,
claude: /event:\s*\w|"type"\s*:\s*"(message_start|content_block|message_delta)"/,
};
const MAX_TOKENS = 128;
const BODIES = {
claude: () => ({
stream: true,
max_tokens: MAX_TOKENS,
system: [{ type: "text", text: "You are concise." }],
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}),
openai: () => ({
stream: true,
max_tokens: MAX_TOKENS,
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}),
"openai-responses": () => ({
stream: true,
max_output_tokens: MAX_TOKENS,
instructions: "You are concise.",
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
}),
};
// Cells: (model, format). `(max)` reproduces the 9router thinking override sent by
// Claude Code — it must land on the same endpoint as the bare id.
const CELLS = [
["deepseek-v4-flash", "claude"],
["deepseek-v4-flash(max)", "claude"],
["deepseek-v4-flash", "openai-responses"],
["deepseek-v4-flash", "openai"],
["deepseek-v4-pro", "claude"],
["deepseek-v4-pro", "openai-responses"],
// Control: MiniMax keeps /messages for Claude clients.
["minimax-m3", "claude"],
];
describe.skipIf(!RUN_REAL)(`REAL opencode-go DeepSeek endpoint matrix`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash");
expect(creds && !creds.allRateLimited).toBe(true);
});
for (const [model, fmt] of CELLS) {
it(`${fmt}-format client → ${model} returns ${fmt}-shape SSE`, async () => {
const out = await runChat(model, BODIES[fmt](), fmt);
if (out === "skip") {
console.warn(`[skip] ${PROVIDER}/${model} ${fmt}: credential/quota/capability`);
return expect(true).toBe(true);
}
expect(out.raw.length, `${model} ${fmt}: empty SSE`).toBeGreaterThan(0);
expect(SSE_MARKER[fmt].test(out.raw), `${model} ${fmt}: wrong SSE shape (routed endpoint rejected?)`).toBe(true);
}, TIMEOUT_MS);
}
});
@@ -0,0 +1,221 @@
// REAL multi-turn thinking pass-back matrix for opencode-go DeepSeek.
//
// Question under test: does the /responses endpoint have the same "cc problem"
// as /messages — i.e. DeepSeek rejects a follow-up turn whose assistant history
// lacks reasoning content, because 9router does not inject a placeholder on that
// path (injectReasoningContent only rewrites body.messages, not the responses
// `input` array)?
//
// Method: turn 1 asks a thinking question non-streamed (so the upstream's own
// output JSON is easy to inspect), then turn 2 replays the assistant turn
// (reasoning + output) in the exact shape the client would, and records whether
// the upstream accepts it.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-passthrough.real.test.js
//
// Cells:
// - openai-responses: assistant turn WITHOUT reasoning item (plain client replay)
// - openai-responses: assistant turn WITH reasoning item (Codex-style store=false replay)
// - openai: control — known-good (injectReasoningContent covers chat path)
// - claude: control — known-broken (handlesThinkingBlocks excludes opencode-go)
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const MODEL = "deepseek-v4-flash";
const TIMEOUT_MS = 120000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function credentials() {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
// Fire one request; returns { ok, status, raw } — never throws on upstream 4xx.
async function send(body, sourceFormat, creds) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${MODEL}` },
modelInfo: { provider: PROVIDER, model: MODEL },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: sourceFormat,
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
// Non-stream responses-format turn 1 — returns the full JSON response object.
async function responsesTurn1(creds, retries = 2) {
const body = {
stream: false,
max_output_tokens: 1024,
reasoning: { effort: "high" },
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }],
};
for (let i = 0; i <= retries; i++) {
const out = await send(body, "openai-responses", creds);
if (out.ok) {
try { return { ok: true, json: JSON.parse(out.raw) }; } catch { return { ok: true, json: null, raw: out.raw }; }
}
if (!out.skip && out.status) return out; // real upstream rejection, don't retry
if (i < retries) await new Promise((r) => setTimeout(r, 2000)); // transient 429 → back off
}
return { ok: false, skip: true };
}
describe.skipIf(!RUN_REAL)(`REAL opencode-go thinking pass-back (${PROVIDER}/${MODEL})`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
expect(creds && !creds.allRateLimited).toBe(true);
});
it("openai-responses: follow-up with NO reasoning item in assistant history", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const t1 = await responsesTurn1(creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const outputMsg = t1.json?.output?.find?.((o) => o.type === "message");
const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || "";
const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning");
console.log(`[turn1] output_text=${JSON.stringify(text.slice(0, 60))} reasoning_item=${!!reasoningItem}`);
const body = {
stream: false,
max_output_tokens: 128,
input: [
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] },
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
],
};
const out = await send(body, "openai-responses", creds);
console.log(`[responses no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// Diagnostic only — a 400 here is the "cc problem" on the responses path.
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
it("openai-responses: follow-up WITH reasoning item (Codex-style replay)", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const t1 = await responsesTurn1(creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const outputMsg = t1.json?.output?.find?.((o) => o.type === "message");
const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || "";
const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning");
console.log(`[turn1] reasoning item present=${!!reasoningItem}`);
const body = {
stream: false,
max_output_tokens: 128,
input: [
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] },
...(reasoningItem ? [reasoningItem] : []),
{ type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] },
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
],
};
const out = await send(body, "openai-responses", creds);
console.log(`[responses with-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
it("openai-responses: streaming 2-turn conversation with thinking enabled", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const turn1Body = {
stream: true,
max_output_tokens: 1024,
reasoning: { effort: "high" },
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }],
};
const t1 = await send(turn1Body, "openai-responses", creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
console.log(`[responses stream turn1] bytes=${t1.raw.length} marker=${/response\.|"type":"response/.test(t1.raw)}`);
// Client replays the assistant turn WITHOUT any reasoning content (9router does
// not inject reasoning_content on the input[] path).
const turn2Body = {
stream: true,
max_output_tokens: 128,
input: [
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "108" }] },
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
],
};
const t2 = await send(turn2Body, "openai-responses", creds);
console.log(`[responses stream turn2] status=${t2.status} bytes=${t2.raw?.length} marker=${/response\.|"type":"response/.test(t2.raw || "")}`);
expect(t2.ok).toBe(true);
}, TIMEOUT_MS);
it("openai (control): follow-up with reasoning_content in assistant history", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const body = {
stream: false,
max_tokens: 1024,
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: "42", reasoning_content: "17 + 26 = 43. Wait, 17+26 = 43? 17+20=37, 37+6=43. Answer: 43." },
{ role: "user", content: "What was your final answer?" },
],
};
const out = await send(body, "openai", creds);
console.log(`[openai control] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
// Control: injectReasoningContent covers the chat path; a 400 here is a real bug.
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("claude (control): follow-up with plain text assistant turn (known-broken on master)", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const body = {
stream: false,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: [{ type: "text", text: "43" }] },
{ role: "user", content: "What was your final answer?" },
],
};
let out;
try {
out = await send(body, "claude", creds);
} catch (e) {
out = { ok: false, status: "threw", raw: String(e?.message || e) };
}
console.log(`[claude control] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// Known-broken on master (handlesThinkingBlocks excludes opencode-go): we record,
// not assert — the pass-back fix should flip this to ok.
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,189 @@
// REAL: opencode.go /messages thinking-placeholder acceptance + real-thinking pass-back.
//
// Decides the form of the thinking-injection follow-up:
//
// B-cell-1 /messages, thinking enabled + tool_use, assistant turn carries an
// UNSIGNED thinking placeholder {type:"thinking", thinking:"."}
// B-cell-2 /messages, same, assistant turn carries a SIGNED thinking placeholder
// (DEFAULT_THINKING_CLAUDE_SIGNATURE) — the exact shape `prepareClaudeRequest`
// would inject today if opencode-go were added to `handlesThinkingBlocks`
// (claude.js routes non-deepseek providers through the signed branch).
// B-cell-3 /messages, same, assistant turn WITHOUT any thinking block — the known
// 400 repro; sanity check that the cells above actually exercise the
// pass-back validation.
// C-cell-4 REAL multi-turn: turn 1 asks a thinking question on /messages and
// receives DeepSeek's own thinking block (no signature, as emitted by
// 9router's response translator openai-to-claude.js:138-156); turn 2
// replays that unsigned thinking block verbatim. Proves the direct path
// accepts real unsigned thinking — the natural endpoint state for C.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-placeholder.real.test.js
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
import { DEFAULT_THINKING_CLAUDE_SIGNATURE } from "../../../open-sse/config/defaultThinkingSignature.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const MODEL = "deepseek-v4-flash";
const TIMEOUT_MS = 90000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function prepare() {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
async function runChat(body, creds, model = MODEL) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${model}` },
modelInfo: { provider: PROVIDER, model },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "claude",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
const TOOL = { name: "get_weather", description: "Get weather", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
function toolTurnBody(assistantContent) {
return {
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
tools: [TOOL],
messages: [
{ role: "user", content: "Weather in Paris?" },
{ role: "assistant", content: assistantContent },
{ role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: '{"temp":"20C"}' }] },
{ role: "user", content: "Summarize in one short sentence." },
],
};
}
describe.skipIf(!RUN_REAL)(`REAL thinking placeholder acceptance (${PROVIDER}/${MODEL})`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
expect(creds && !creds.allRateLimited).toBe(true);
});
it("B-cell-1: unsigned thinking placeholder accepted on /messages", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
const out = await runChat(toolTurnBody([
{ type: "thinking", thinking: "." },
{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } },
]), creds);
console.log(`[B1 unsigned] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
expect(out.skip).not.toBe(true);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("B-cell-2: SIGNED thinking placeholder (prepareClaudeRequest shape) on /messages", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
const out = await runChat(toolTurnBody([
{ type: "thinking", thinking: ".", signature: DEFAULT_THINKING_CLAUDE_SIGNATURE },
{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } },
]), creds);
console.log(`[B2 signed] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
// This is the exact shape a naive handlesThinkingBlocks addition would inject.
// A 400 here means the follow-up MUST route opencode-go through the unsigned branch.
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("B-cell-3: REAL turn1 thinking, turn2 replay WITHOUT the thinking block (pass-back failure)", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
// Turn 1: real thinking output on /messages (no tools) — establishes a
// thinking-bearing turn in history.
const t1 = await runChat({
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
}, creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
// Turn 2: replay the assistant turn WITHOUT the thinking block — the shape a
// client whose history lost the thinking (or a gateway that dropped it) sends.
const out = await runChat({
stream: true,
max_tokens: 256,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: [{ type: "text", text: "43" }] },
{ role: "user", content: "What was your final answer? Reply with just the number." },
],
}, creds);
console.log(`[B3 real-thinking missing] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// NOTE (2026-08-16): direct raw upstream rejects this (500), but through the
// gateway it passes because `injectReasoningContent` (MODEL_RULES /deepseek/i,
// executor transformRequest) injects `reasoning_content: " "` on the assistant
// message before dispatch, and the /messages shim honors that field. This cell
// is therefore recorded as evidence of the mechanism, not asserted as a bug.
console.warn(`[B3] direct 500 vs gateway-200: shim honors reasoning_content field`);
expect(out.skip).not.toBe(true);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("C-cell-4: real thinking block from upstream replayed verbatim on /messages", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
// Turn 1: thinking enabled, no tools — capture DeepSeek's own thinking block.
const t1 = await runChat({
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
}, creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const thinkingText = (t1.raw.match(/thinking_delta[^\n]*\n[^\n]*"thinking":\s*"([^"]+)/s) || [])[1] || "";
console.log(`[turn1] thinking_delta_len=${thinkingText.length} marker=${/content_block_delta/.test(t1.raw)}`);
const noThinkingBlocks = !/type":"thinking"/.test(t1.raw);
// Turn 2: replay the assistant turn with the upstream's real (unsigned) thinking
// block — the exact conversation state a Claude Code client would have.
const out = await runChat({
stream: true,
max_tokens: 256,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: [
{ type: "thinking", thinking: thinkingText || "17 + 26 = 43" },
{ type: "text", text: "43" },
] },
{ role: "user", content: "What was your final answer? Reply with just the number." },
],
}, creds);
console.log(`[turn2 real-thinking replay] status=${out.status} noThinkingBlocks=${noThinkingBlocks}`);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,237 @@
// REAL: full tool-use conversation sessions + thinking semantics + non-streaming
// paths for opencode-go DeepSeek.
//
// Covers the blind spots of the basic endpoint matrix (which only used plain-text
// bodies): the real 2-turn tool loop that #3332 originally reported 400 for,
// whether the `(max)` thinking suffix actually produces thinking output, the
// non-streaming code paths, and the chat-only-model fallback route.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-tool-session.real.test.js
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const TIMEOUT_MS = 120000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
const WEATHER_TOOL = { name: "get_weather", description: "Get weather for a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
const TIME_TOOL = { name: "get_time", description: "Get current time in a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function prepare(model) {
const creds = await getProviderCredentials(PROVIDER, new Set(), model);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
async function runChat(body, creds, model) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${model}` },
modelInfo: { provider: PROVIDER, model },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "claude",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
// Parse Claude-shape SSE blocks: [{type:"tool_use",id,name,input}, ...]
function extractToolUses(raw) {
const blocks = [];
for (const chunk of raw.split("\n\n")) {
const line = chunk.split("\n").find((l) => l.startsWith("data: "));
if (!line) continue;
try {
const d = JSON.parse(line.slice(6));
if (d.type === "content_block_start" && d.content_block?.type === "tool_use") {
blocks.push({ id: d.content_block.id, name: d.content_block.name, input: d.content_block.input });
}
} catch { /* skip malformed */ }
}
return blocks;
}
const THINKING_BODY = {
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
};
describe.skipIf(!RUN_REAL)(`REAL tool sessions + semantics (${PROVIDER})`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash");
expect(creds && !creds.allRateLimited).toBe(true);
});
it("2-turn tool loop via /messages with thinking enabled (the original 400 shape)", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
// Turn 1: real model turn that should call the tool.
const t1 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL],
messages: [{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }],
}, creds, model);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const toolUses = extractToolUses(t1.raw);
console.log(`[loop turn1] status=200 tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`);
if (toolUses.length === 0) { console.warn("[skip] model did not call a tool on turn1"); return expect(true).toBe(true); }
// Turn 2: replay the assistant tool_use (no thinking block — client-side real
// history may or may not carry it; gateway reasoning_content covers pass-back)
// and return the tool result. This is the exact conversation shape that 400'd
// before (and which the endpoint matrix never exercised).
const t2 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL],
messages: [
{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." },
{ role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) },
{ role: "user", content: [
...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: '{"temp":"20C"}' })),
{ type: "text", text: "Summarize in one short sentence." },
] },
],
}, creds, model);
console.log(`[loop turn2] status=${t2.status} bytes=${t2.raw?.length}`);
expect(t2.skip).not.toBe(true);
expect(t2.ok).toBe(true);
}, TIMEOUT_MS);
it("parallel tool_use turn replayed with all results (thinking enabled)", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const t1 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL, TIME_TOOL],
messages: [{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }],
}, creds, model);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const toolUses = extractToolUses(t1.raw);
console.log(`[parallel turn1] tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`);
if (toolUses.length === 0) { console.warn("[skip] model did not call tools on turn1"); return expect(true).toBe(true); }
const t2 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL, TIME_TOOL],
messages: [
{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." },
{ role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) },
{ role: "user", content: [
...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: t.name === "get_weather" ? '{"temp":"20C"}' : '{"time":"14:30"}' })),
{ type: "text", text: "Summarize in one short sentence." },
] },
],
}, creds, model);
console.log(`[parallel turn2] status=${t2.status} bytes=${t2.raw?.length}`);
expect(t2.skip).not.toBe(true);
expect(t2.ok).toBe(true);
}, TIMEOUT_MS);
for (const model of ["deepseek-v4-flash(max)", "deepseek-v4-pro(max)"]) {
it(`(max) suffix on ${model} produces thinking output on /messages`, async () => {
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const out = await runChat({
stream: true,
max_tokens: 1024,
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
}, creds, model);
if (out.skip) return expect(true).toBe(true);
const hasThinkingDelta = /thinking_delta/.test(out.raw || "");
const hasText = /content_block_delta.*text/.test(out.raw || "") || /"text":"/.test(out.raw || "");
console.log(`[${model}] status=${out.status} thinking_delta=${hasThinkingDelta} text=${hasText} bytes=${out.raw?.length}`);
expect(out.ok).toBe(true);
expect(hasThinkingDelta, `(max) should enable thinking on ${model}`).toBe(true);
}, TIMEOUT_MS);
}
it("non-streaming claude-format request (JSON path) succeeds", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const out = await runChat({
stream: false,
max_tokens: 128,
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}, creds, model);
if (out.skip) return expect(true).toBe(true);
const isJson = out.raw?.trim()?.startsWith("{");
const hasText = /"text"/.test(out.raw || "");
console.log(`[nonstream claude] status=${out.status} json=${isJson} hasText=${hasText} raw=${out.raw?.slice?.(0, 120)}`);
expect(out.ok).toBe(true);
expect(isJson).toBe(true);
}, TIMEOUT_MS);
it("non-streaming openai-responses-format request succeeds", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const result = await handleChatCore({
body: {
model: `${PROVIDER}/${model}`, stream: false, max_output_tokens: 128,
instructions: "You are concise.",
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
},
modelInfo: { provider: PROVIDER, model },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "openai-responses",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || status >= 500 || status === 406) return expect(true).toBe(true);
throw new Error(`[nonstream responses] ${status}: ${result.error}`);
}
const raw = await drainSSE(result.response);
const isJson = raw?.trim()?.startsWith("{");
const hasResponsesShape = /"output"|"object":"response"/.test(raw || "");
console.log(`[nonstream responses] status=200 json=${isJson} responsesShape=${hasResponsesShape} raw=${raw?.slice?.(0, 150)}`);
expect(isJson).toBe(true);
expect(hasResponsesShape).toBe(true);
}, TIMEOUT_MS);
it("chat-only glm-5.2(max) falls back to /chat/completions for a claude-format client", async () => {
const model = "glm-5.2(max)";
const creds = await prepare("glm-5.2");
if (!creds) return expect(true).toBe(true);
const out = await runChat({
stream: true,
max_tokens: 128,
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}, creds, model);
if (out.skip) { console.warn("[skip] glm-5.2 rejected/absent upstream"); return expect(true).toBe(true); }
// Guard blocks /messages; the request is translated to chat and lands on
// /chat/completions, re-encoded to the client's claude format.
const hasClaudeShape = /event:\s*\w|"type"\s*:\s*"(message_start|content_block_delta|message_stop)"/.test(out.raw || "");
console.log(`[glm fallback] status=${out.status} claudeShape=${hasClaudeShape} bytes=${out.raw?.length}`);
expect(out.ok).toBe(true);
expect(hasClaudeShape).toBe(true);
}, TIMEOUT_MS);
});
Loaded 100 of 137 files, more files were not shown because too many files have changed in this diff. Show more