From 57c04f00fadb15809087f194fc9fb5b11c16331c Mon Sep 17 00:00:00 2001 From: decolua Date: Sun, 27 Sep 2026 00:52:50 +0700 Subject: [PATCH 01/41] test(providers): remove tests that write to the real user DB provider-priority-insert-cost.test.js imported src/lib/db without DATA_DIR isolation, so every run seeded seed-* connections into the developer live ~/.9router database (~787 rows after repeated runs). Co-Authored-By: Claude Code --- .../provider-priority-insert-cost.test.js | 138 ------------------ 1 file changed, 138 deletions(-) delete mode 100644 tests/unit/provider-priority-insert-cost.test.js diff --git a/tests/unit/provider-priority-insert-cost.test.js b/tests/unit/provider-priority-insert-cost.test.js deleted file mode 100644 index bd2f2ef6..00000000 --- a/tests/unit/provider-priority-insert-cost.test.js +++ /dev/null @@ -1,138 +0,0 @@ -import { describe, expect, it } from "vitest"; - -import { - createProviderConnection, - getProviderConnections, - deleteProviderConnection, - updateProviderConnection, -} from "../../src/lib/db/index.js"; - -// #4311: POST /api/providers was O(pool) per insert. Inside one transaction it -// read the whole pool AND renumbered every row's priority, so a 5k-key import -// was O(n*m) — ~25M statements at a 5k pool — and every parallel writer -// serialized on the same transaction. On top of that, an apikey name collision -// silently overwrote the stored key with no 409. -// -// The test DB persists across tests in a file, so each case uses its own -// provider alias; priorities are per-provider. - -async function seed(provider, n) { - for (let i = 0; i < n; i++) { - await createProviderConnection({ - provider, - authType: "apikey", - name: `seed-${i}`, - apiKey: `k${i}`, - }); - } -} - -describe("provider insert is O(1) in pool size (#4311)", () => { - it("assigns sequential priorities without a renumber pass", async () => { - const P = `openai-compatible-seq-${Date.now()}`; - await seed(P, 3); - const list = await getProviderConnections({ provider: P }); - expect(list.map((c) => c.name)).toEqual(["seed-0", "seed-1", "seed-2"]); - expect(list.map((c) => c.priority)).toEqual([1, 2, 3]); - }); - - it("keeps a large pool in insertion order", async () => { - const P = `openai-compatible-ord-${Date.now()}`; - await seed(P, 60); - const list = await getProviderConnections({ provider: P }); - expect(list).toHaveLength(60); - // The bug showed up as reordering once the pool grew past a few rows. - expect(list[0].name).toBe("seed-0"); - expect(list[59].name).toBe("seed-59"); - for (let i = 1; i < list.length; i++) { - expect(list[i].priority).toBeGreaterThan(list[i - 1].priority); - } - }); - - it("still renumbers on delete, so gaps do not accumulate", async () => { - const P = `openai-compatible-del-${Date.now()}`; - await seed(P, 4); - const before = await getProviderConnections({ provider: P }); - await deleteProviderConnection(before[0].id); - const after = await getProviderConnections({ provider: P }); - expect(after.map((c) => c.priority)).toEqual([1, 2, 3]); - }); - - it("still renumbers on an explicit priority update", async () => { - // Unique alias per run: the DB persists across runs, so a fixed alias - // would accumulate rows and make this assertion depend on test order. - const P = `openai-compatible-upd-${Date.now()}`; - await seed(P, 4); - await new Promise((r) => setTimeout(r, 10)); - const list = await getProviderConnections({ provider: P }); - // Move the last one to the front. - await updateProviderConnection(list[3].id, { priority: 1 }); - const after = await getProviderConnections({ provider: P }); - expect(after[0].name).toBe("seed-3"); - }); -}); - -describe("name collision no longer destroys a key silently (#4311)", () => { - // Seeded once: these cases each mutate the SAME row, so a per-test seed - // would make the later assertions depend on earlier ones. - const P = `openai-compatible-clash-${Date.now()}`; - const original = (async () => { - await seed(P, 1); - return (await getProviderConnections({ provider: P }))[0]; - })(); - - it("throws a typed conflict instead of overwriting, when overwrite is refused", async () => { - const orig = await original; - await expect( - createProviderConnection({ - provider: P, - authType: "apikey", - name: orig.name, - apiKey: "REPLACEMENT-KEY", - allowOverwrite: false, - }) - ).rejects.toMatchObject({ code: "PROVIDER_NAME_CONFLICT", existingId: orig.id }); - - // The stored key must be untouched. - const after = (await getProviderConnections({ provider: P }))[0]; - expect(after.apiKey).toBe(orig.apiKey); - }); - - it("still overwrites when the caller opts in", async () => { - const orig = await original; - const updated = await createProviderConnection({ - provider: P, - authType: "apikey", - name: orig.name, - apiKey: "REPLACEMENT-KEY", - allowOverwrite: true, - }); - expect(updated.id).toBe(orig.id); - const after = (await getProviderConnections({ provider: P }))[0]; - expect(after.apiKey).toBe("REPLACEMENT-KEY"); - }); - - it("defaults to the previous overwrite behaviour for existing callers", async () => { - // Every other call site in the repo (oauth routes, bulk import) omits the - // flag, so they must keep working exactly as before. - const orig = await original; - const updated = await createProviderConnection({ - provider: P, - authType: "apikey", - name: orig.name, - apiKey: "LEGACY-PATH-KEY", - }); - expect(updated.id).toBe(orig.id); - }); - - it("does not collide across different providers", async () => { - const orig = await original; - const other = await createProviderConnection({ - provider: "openai-compatible-other", - authType: "apikey", - name: orig.name, - apiKey: "other-key", - }); - expect(other.id).not.toBe(orig.id); - }); -}); From e706e4f26adaae5240069205232db3569f6a103d Mon Sep 17 00:00:00 2001 From: ZIRAN456 <75793045+ZIRAN456@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:35:58 +0700 Subject: [PATCH 02/41] feat(codebuddy): parse 6004 rate limit error and extract resetsAtMs --- open-sse/executors/codebuddy-cn.js | 27 +++++++++++++ open-sse/executors/codebuddy-intl.js | 27 +++++++++++++ tests/unit/codebuddy-parse-error.test.js | 50 ++++++++++++++++++++++++ 3 files changed, 104 insertions(+) create mode 100644 tests/unit/codebuddy-parse-error.test.js diff --git a/open-sse/executors/codebuddy-cn.js b/open-sse/executors/codebuddy-cn.js index 76ec52ff..072d055c 100644 --- a/open-sse/executors/codebuddy-cn.js +++ b/open-sse/executors/codebuddy-cn.js @@ -64,6 +64,33 @@ export class CodeBuddyExecutor extends DefaultExecutor { // filter and return an error (#2071). return transformed; } + parseError(response, bodyText) { + if (bodyText) { + try { + const data = JSON.parse(bodyText); + const msg = data?.msg || data?.message || data?.error?.message || ""; + if (data?.code === 6004 || /超出频率限制|frequency limit|限额/i.test(msg)) { + let resetsAtMs = null; + const match = msg.match(/(\d{4}-\d{2}-\d{2})\s+(\d{2}:\d{2}:\d{2})(?:\s*UTC\+?([0-9:]+))?/i); + if (match) { + const dp = match[1]; + const tp = match[2]; + const tz = match[3] + ? (match[3].includes(":") ? (match[3].startsWith("+") ? match[3] : `+${match[3]}`) : `+${match[3].padStart(2, "0")}:00`) + : "+08:00"; + const dt = new Date(`${dp}T${tp}${tz}`); + if (!isNaN(dt.getTime())) resetsAtMs = dt.getTime(); + } + return { + status: 429, + message: msg || "CodeBuddy frequency limit (6004)", + resetsAtMs, + }; + } + } catch {} + } + return super.parseError(response, bodyText); + } } export default CodeBuddyExecutor; diff --git a/open-sse/executors/codebuddy-intl.js b/open-sse/executors/codebuddy-intl.js index bb99ff47..61a58997 100644 --- a/open-sse/executors/codebuddy-intl.js +++ b/open-sse/executors/codebuddy-intl.js @@ -39,6 +39,33 @@ export class CodeBuddyIntlExecutor extends DefaultExecutor { return transformed; } + parseError(response, bodyText) { + if (bodyText) { + try { + const data = JSON.parse(bodyText); + const msg = data?.msg || data?.message || data?.error?.message || ""; + if (data?.code === 6004 || /超出频率限制|frequency limit|限额/i.test(msg)) { + let resetsAtMs = null; + const match = msg.match(/(\d{4}-\d{2}-\d{2})\s+(\d{2}:\d{2}:\d{2})(?:\s*UTC\+?([0-9:]+))?/i); + if (match) { + const dp = match[1]; + const tp = match[2]; + const tz = match[3] + ? (match[3].includes(":") ? (match[3].startsWith("+") ? match[3] : `+${match[3]}`) : `+${match[3].padStart(2, "0")}:00`) + : "+08:00"; + const dt = new Date(`${dp}T${tp}${tz}`); + if (!isNaN(dt.getTime())) resetsAtMs = dt.getTime(); + } + return { + status: 429, + message: msg || "CodeBuddy frequency limit (6004)", + resetsAtMs, + }; + } + } catch {} + } + return super.parseError(response, bodyText); + } } export default CodeBuddyIntlExecutor; diff --git a/tests/unit/codebuddy-parse-error.test.js b/tests/unit/codebuddy-parse-error.test.js new file mode 100644 index 00000000..2d07a06b --- /dev/null +++ b/tests/unit/codebuddy-parse-error.test.js @@ -0,0 +1,50 @@ +import { describe, it, expect } from "vitest"; +import { CodeBuddyExecutor } from "../../open-sse/executors/codebuddy-cn.js"; +import { CodeBuddyIntlExecutor } from "../../open-sse/executors/codebuddy-intl.js"; + +describe("CodeBuddy parseError", () => { + it("parses code 6004 frequency limit with timestamp into 429 and resetsAtMs on codebuddy-cn", () => { + const executor = new CodeBuddyExecutor(); + const bodyText = JSON.stringify({ + code: 6004, + message: "当前模型超出频率限制,请于 2026-09-28 14:30:00 后重试", + }); + + const parsed = executor.parseError({ status: 200 }, bodyText); + expect(parsed.status).toBe(429); + expect(parsed.message).toContain("超出频率限制"); + expect(parsed.resetsAtMs).toBeTruthy(); + expect(typeof parsed.resetsAtMs).toBe("number"); + + // 验证时区解析 (默认 UTC+8) + const expected = new Date("2026-09-28T14:30:00+08:00").getTime(); + expect(parsed.resetsAtMs).toBe(expected); + }); + + it("parses frequency limit text match without code 6004 on codebuddy-intl", () => { + const executor = new CodeBuddyIntlExecutor(); + const bodyText = JSON.stringify({ + code: 11000, + msg: "frequency limit exceeded, please retry after 2026-09-28 12:00:00 UTC+0", + }); + + const parsed = executor.parseError({ status: 400 }, bodyText); + expect(parsed.status).toBe(429); + expect(parsed.message).toContain("frequency limit"); + const expected = new Date("2026-09-28T12:00:00+00:00").getTime(); + expect(parsed.resetsAtMs).toBe(expected); + }); + + it("falls back to super.parseError for unrelated errors", () => { + const executor = new CodeBuddyExecutor(); + const bodyText = JSON.stringify({ + code: 11101, + message: "Non-stream chat request is currently not supported", + }); + + const parsed = executor.parseError({ status: 400 }, bodyText); + expect(parsed.status).toBe(400); + expect(parsed.message).toBe(bodyText); + expect(parsed.resetsAtMs).toBeUndefined(); + }); +}); From dcc6a2055e3839228a6300eaa5395643f2b75aef Mon Sep 17 00:00:00 2001 From: decolua Date: Mon, 28 Sep 2026 12:37:25 +0700 Subject: [PATCH 03/41] fix(dashboard): exclude hidden providers from usage stats provider list --- src/shared/components/UsageStats.js | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/shared/components/UsageStats.js b/src/shared/components/UsageStats.js index c4007812..3db6a8eb 100644 --- a/src/shared/components/UsageStats.js +++ b/src/shared/components/UsageStats.js @@ -239,6 +239,7 @@ export default function UsageStats({ period: periodProp, setPeriod: setPeriodPro const unique = (d?.connections || []).filter((c) => { if (c.isActive === false) return false; if (!isLLMProvider(c.provider)) return false; + if (AI_PROVIDERS[c.provider]?.hidden) return false; if (seen.has(c.provider)) return false; seen.add(c.provider); return true; @@ -247,7 +248,7 @@ export default function UsageStats({ period: periodProp, setPeriod: setPeriodPro nodeName: nodeNameMap[c.provider] || null, })); const noAuthProviders = Object.values(FREE_PROVIDERS) - .filter((p) => p.noAuth && !seen.has(p.id) && isLLMProvider(p.id)) + .filter((p) => p.noAuth && !p.hidden && !seen.has(p.id) && isLLMProvider(p.id)) .map((p) => ({ provider: p.id, name: p.name })); setProviders([...unique, ...noAuthProviders]); }) From 60e890d7f18b5a82f3690cbeb92f243e17d49f04 Mon Sep 17 00:00:00 2001 From: Emirhan Date: Mon, 28 Sep 2026 12:39:11 +0700 Subject: [PATCH 04/41] feat(web): add TinyFish search and fetch provider --- CHANGELOG.md | 1 + open-sse/handlers/fetch/index.js | 32 +++++++- open-sse/handlers/search/callers.js | 24 ++++++ open-sse/handlers/search/index.js | 5 +- open-sse/handlers/search/normalizers.js | 14 ++++ open-sse/providers/registry/index.js | 2 + open-sse/providers/registry/tinyfish.js | 36 +++++++++ public/providers/tinyfish.png | Bin 0 -> 7463 bytes tests/unit/tinyfish-web-provider.test.js | 91 +++++++++++++++++++++++ 9 files changed, 203 insertions(+), 2 deletions(-) create mode 100644 open-sse/providers/registry/tinyfish.js create mode 100644 public/providers/tinyfish.png create mode 100644 tests/unit/tinyfish-web-provider.test.js diff --git a/CHANGELOG.md b/CHANGELOG.md index 6cb89240..b6d8cd61 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,7 @@ # v0.5.91 (2026-09-26) ## Features +- **Web Search & Fetch**: add TinyFish Search and Fetch with one API-key connection, normalized results, and official provider icon - **Providers**: add Token Harbor provider and four OpenAI-compatible aggregator providers (dahl, atria, agnes, bai) - **Claude**: forward `x-claude-code-session-id` on OAuth requests; merge client `anthropic-beta` flags and forward rate-limit headers; return thinking text to OpenAI-format clients - **Codex**: add GPT-6 Sol and Luna support diff --git a/open-sse/handlers/fetch/index.js b/open-sse/handlers/fetch/index.js index 9bad7e18..364caa5d 100644 --- a/open-sse/handlers/fetch/index.js +++ b/open-sse/handlers/fetch/index.js @@ -1,4 +1,4 @@ -// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama +// Web Fetch handler — dispatches to firecrawl, jina-reader, tavily, exa, ollama, tinyfish // Returns normalized shape across all providers const DEFAULT_TIMEOUT_MS = 15000; @@ -129,6 +129,9 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr baseUrl: providerConfig?.baseUrl, }); } + if (provider === "tinyfish") { + return await runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl: providerConfig?.baseUrl }); + } return { success: false, status: 400, error: `Unsupported provider: ${provider}` }; } catch (err) { log?.("fetch handler error:", err?.message || err); @@ -136,6 +139,33 @@ export async function handleFetchCore({ url, format, maxCharacters, provider, pr } } +async function runTinyfish({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt, baseUrl }) { + if (!["markdown", "html"].includes(fmt)) return { success: false, status: 400, error: `Unsupported TinyFish format: ${fmt}` }; + const upstreamStart = Date.now(); + const r = await tryFetch(baseUrl, { + method: "POST", + headers: { "content-type": "application/json", "X-API-Key": apiKey }, + body: JSON.stringify({ urls: [url], format: fmt }), + }, timeoutMs); + if (!r.ok) return { success: false, status: r.timeout ? 504 : 502, error: r.error }; + const upstreamMs = Date.now() - upstreamStart; + const { json } = await readJsonOrText(r.res); + if (!r.res.ok) return { success: false, status: r.res.status, error: json?.error?.message || `TinyFish error: ${r.res.status}` }; + const failure = json?.errors?.[0]; + if (failure) return { success: false, status: failure.status || 502, error: `TinyFish fetch failed: ${failure.error || "unknown error"}` }; + const page = json?.results?.[0]; + if (!page || typeof page.text !== "string") return { success: false, status: 502, error: "TinyFish returned no extractable content" }; + const text = truncate(page.text, maxCharacters); + return { + success: true, + data: { + ...buildData({ provider: "tinyfish", url, title: page.title, format: fmt, text, links: page.links, + costUsd: costPerQuery, responseMs: Date.now() - startedAt, upstreamMs }), + metadata: { author: page.author || null, published_at: page.published_date || null, language: page.language || null }, + }, + }; +} + async function runFirecrawl({ url, fmt, timeoutMs, apiKey, maxCharacters, costPerQuery, startedAt }) { const upstreamStart = Date.now(); const r = await tryFetch("https://api.firecrawl.dev/v1/scrape", { diff --git a/open-sse/handlers/search/callers.js b/open-sse/handlers/search/callers.js index 5e3b3c09..56d08e4f 100644 --- a/open-sse/handlers/search/callers.js +++ b/open-sse/handlers/search/callers.js @@ -374,6 +374,29 @@ function buildXquikRequest(config, params) { }; } +function buildTinyfishRequest(config, params) { + if (params.searchType && !["web", "news", "research_paper"].includes(params.searchType)) { + throw new Error("Unsupported TinyFish search type"); + } + const qp = new URLSearchParams({ query: params.query }); + if (params.searchType && params.searchType !== "web") qp.set("domain_type", params.searchType); + if (params.country) qp.set("location", params.country); + if (params.language) qp.set("language", params.language); + const { includes, excludes } = parseDomainFilter(params.domainFilter); + if (includes.length) qp.set("include_domains", includes.join(",")); + if (excludes.length) qp.set("exclude_domains", excludes.join(",")); + if (Number.isInteger(params.offset) && params.offset > 0) { + if (params.offset >= 110) throw new Error("TinyFish search offset exceeds available pages"); + if (params.offset % 10 + params.maxResults > 10) throw new Error("TinyFish search offset and max_results must fit within one page"); + qp.set("page", String(Math.floor(params.offset / 10))); + } + return { + // Keep API-key endpoint fixed; client baseUrl overrides must not receive the key. + url: `${config.baseUrl}?${qp}`, + init: { method: "GET", headers: { Accept: "application/json", "X-API-Key": params.token } }, + }; +} + // ── Ollama Cloud web_search ────────────────────────────────────────────── // POST https://ollama.com/api/web_search { query, max_results } // Response: { results: [{ title, url, content, published_at? }] } @@ -436,6 +459,7 @@ const BUILDERS = { "youcom": buildYouComRequest, "searxng": buildSearxngRequest, "xquik": buildXquikRequest, + "tinyfish": buildTinyfishRequest, "ollama-search": buildOllamaSearchRequest, "glm": buildGlmSearchRequest, }; diff --git a/open-sse/handlers/search/index.js b/open-sse/handlers/search/index.js index 70c3c749..fa6d146d 100644 --- a/open-sse/handlers/search/index.js +++ b/open-sse/handlers/search/index.js @@ -110,7 +110,10 @@ async function tryDedicatedProvider({ provider, providerConfig, body, credential } const data = await resp.json(); const normalized = normalizeSearchResponse(provider.id, data, params.query, params.searchType); - const results = normalized.results.slice(0, params.maxResults); + // TinyFish uses fixed 10-result pages; offset within a page is applied locally. + const pageOffset = provider.id === "tinyfish" && Number.isInteger(params.offset) && params.offset > 0 + ? params.offset % 10 : 0; + const results = normalized.results.slice(pageOffset, pageOffset + params.maxResults); const duration = Date.now() - startTime; const usage = { queries_used: 1, diff --git a/open-sse/handlers/search/normalizers.js b/open-sse/handlers/search/normalizers.js index 3b415ae5..5ef7f36f 100644 --- a/open-sse/handlers/search/normalizers.js +++ b/open-sse/handlers/search/normalizers.js @@ -107,6 +107,19 @@ function normalizeTavily(data, _query, _searchType) { return { results, totalResults: results.length }; } +function normalizeTinyfish(data) { + const now = new Date().toISOString(); + const items = Array.isArray(data?.results) ? data.results : []; + return { + results: items.map((item, idx) => makeResult("tinyfish", { + title: item.title, url: item.url, snippet: item.snippet, + published_at: item.date, source_type: item.publisher || null, + author: Array.isArray(item.authors) ? item.authors.join(", ") : null, + }, idx, now)), + totalResults: data?.total_results ?? null, + }; +} + function normalizeGooglePse(data, _query, _searchType) { const now = new Date().toISOString(); const items = Array.isArray(data.items) ? data.items : []; @@ -294,6 +307,7 @@ const NORMALIZERS = { "youcom": normalizeYouCom, "searxng": normalizeSearxng, "xquik": normalizeXquik, + "tinyfish": normalizeTinyfish, "ollama-search": normalizeOllamaSearch, "glm": normalizeGlmSearch, }; diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 07b71d2c..6c9d1f41 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -130,6 +130,7 @@ import p126 from "./dahl.js"; import p127 from "./atria.js"; import p129 from "./agnes.js"; import p130 from "./bai.js"; +import p131 from "./tinyfish.js"; export default [ p0, p1, @@ -260,4 +261,5 @@ export default [ p127, p129, p130, + p131, ]; diff --git a/open-sse/providers/registry/tinyfish.js b/open-sse/providers/registry/tinyfish.js new file mode 100644 index 00000000..c6875251 --- /dev/null +++ b/open-sse/providers/registry/tinyfish.js @@ -0,0 +1,36 @@ +export default { + id: "tinyfish", + alias: "tinyfish", + display: { + name: "TinyFish", + color: "#FF6700", + textIcon: "TF", + website: "https://www.tinyfish.ai/", + notice: { apiKeyUrl: "https://agent.tinyfish.ai/api-keys" }, + }, + category: "apikey", + authType: "apikey", + serviceKinds: ["webSearch", "webFetch"], + searchConfig: { + baseUrl: "https://api.search.tinyfish.ai", + validateUrl: "https://api.search.tinyfish.ai/usage?limit=1", + method: "GET", + authType: "apikey", + authHeader: "x-api-key", + costPerQuery: 0, + searchTypes: ["web", "news", "research_paper"], + defaultMaxResults: 5, + maxMaxResults: 10, + timeoutMs: 10000, + }, + fetchConfig: { + baseUrl: "https://api.fetch.tinyfish.ai", + method: "POST", + authType: "apikey", + authHeader: "x-api-key", + costPerQuery: 0, + formats: ["markdown", "html"], + maxCharacters: 100000, + timeoutMs: 150000, + }, +}; diff --git a/public/providers/tinyfish.png b/public/providers/tinyfish.png new file mode 100644 index 0000000000000000000000000000000000000000..9768f27a6e76364115f24cec0b6e05b09c00957d GIT binary patch literal 7463 zcmV+?9oXWDP)Q}zxj1L$oPExW88d#rIluQ^Ywfk(wbtJ6yVkqj^{%~-4f(%Zh---F zh^fRQ#5u%Qh&K@xGF(*o~x zeiZfZyA{`$D2&MP5R#Y>&9EVHIq@6ft;8T=o(1#4^6cFK1$l3!i3s92@7&;4eDYFl zyxjZ&yw%|m{5rf77LV+Lp}m^nsYmWYk;0K86-pv6F)@qSl^8?JlaSX1pwTPDC9*Gd z&lcEbQ01?m3!FQ{9N?Dgi=tiQ`|xeQmRLezJoo6`2n{i-P-M7`M4u9WA>K*MbCA~s zaF!f9PDEN7uxNiWjvqS(B7=Q#IZ!B>ofMy+xxgOEOnm+B(MV)iem?0 z%=-@5#3n9i(+mU$+0dQ~^?0ri)$WicGN~VDIE8o@G1r7(W9NAZ8t`UhwcO0uCsR^5 z;Nvfn!WU?L*tih*jeD}B^Cp#ML!D~3VDcwz@OaHTU>Iie+OToTz$F!v_$mfhQ2>6Q zO84>~Qd#w7}pu~%*NZ9cw&deN*)2B`&{_F*uO-w>5s-fX;Hb>1W zYy9A>Ye)NFoyYZDKN4_`NBZwia-Z?gIeAK2>)Ff82d0#!EFh> zK-b*1)4&X>{(*l2y;cKVmILjmg<6l_i`KoqLz}lJpu_u<(fzZj_(DC=%MLA z{h2@;GQ1c4rGqxZG*_qn1I*qJ%=zO8W-Q-|$qP1N)bzFZbnIe$HE{(J&wJ>xhU|9$ zkz!|fl^`z(AfC@m8;79-G^)9NBAj9=NZTcZcYBm7ec~>l7MWi0C*Ym+!1`mr;d8(l z@>A=BqdKInXb3^|f|PI;FiuWMMM^4l71Iu2FOYS#hu2>brJ6L))C@0k2KY!h({L7A zO<*04SB4JQPitp8Rm&xg_BGsu{bwyuk0bp#uCJ?(0x8UQkeX2pHOf83fMz!V-ROAe zR}J{EW*7#vs)7%@Jc_qFHo&W`YNP$X?!#jbR={1iUWe#{;jpjAW)ljEFz80)rULS= z>byaVbdCrjPyxutE(_)Vcw;gR8#AVYV3>U%cFJetYEicS=xuh@(uAAc!Bj0fC=eF` z69u?yc#1aD6yDwX=D_U7fiYYoKCfv*|K=6ZyKiT7f8}ZPYF{7yJ2u2;-JifP?vUlO zm^-v1){O0rpFV1Xwog<;i5TtE*r*-2)SG*dFL=q`@hszP2Y&+4O>}3kdx`UjLx{tP zGl;Rol|Aj;Pb#_GQS0d zaB~z;vOo}`iWNu5BW++bXb%+Q;0@u_;GJotYflgn7J}$P5#}bX{0+tN6g{b9-gy?g zr}x3n1KQznwql0B8W>Wwz@6h`h`I zq0D)ZD7S%T$dyR5T78;JUI-HEv9t8?q$t}R# zIcvwsr30RKt`@{)EPA`U605*%d97ZUla!uJc+dYiGr}V-ePSd$V zI-3A9ftCZvkX=O;8?S_ zTeesM^nCU~+kvf(RwUe#%ptWEu53C9#`Q~ta@2bXC%y~oWe9&d$z8FS%Y3gv1s;f zZ&%EhXH~I1CljHS05r|arow_rc8QyoQ#8Jha(kI@5=h~$hovFGLAbSSG0Y$OGMd+` z?6e$M0eX``>w=g}$_$`sqn)QrTk~l`SxdTJqrTR4Kxj~gI+ru3^HJv+KvvFGV7{~C z(2yWhuY99<>Qz`CFlYs*t*(Wg#KYQ8nvt;(hUtvuD3dW3KMDlgT z|A1Ipr#dSnnasL@|89aNwJY+<1y3>cCNcd&F6PU=?#MDzscI9|`)-&f9z%Wk9jI=cTe=4Q@8EQ?UjWDX`rp5Mqy0Ee8lV~@w<+$Ng*j!}=q z1Q@%1hq35KU?+V6njIuATPhl(-f9Usx_Bv2``wsmNOXcs)o*!0BOS7!AbmfOx8Iin zkjsH()=-c-s?tLM1z}B?_0sbflCfj|Fd^9U&P%0*o)5cEO6gx&I0@8Ni-FV?D+puS+%2mA7l=dr7_x@5N^B-oqzRx6{|S zqs3<^*W@EiS@36izmZ8NA!u>_iDwLYZf5EU^cqYZrX>(Y)?g)obh`0 zauN(3d}t=<=~!^%busAo(j(@%EH4Fe_jYDNyK23!6U;Qql82+n{HJfxRbZG__70%z z?ST=eqdI&n9s|cO!`ma~VaBqpNcD8axv1ayWtg5K3RVxKfY+`>QC zuqql>cj;)|nWknx)-xA(jeMz{rZixf7Z?PE09EP&hShDu>raOmfWW1Vei_cXx0_GT*d_!Z}g<-2=#fqc`4|9~fN6 zsS;fxFxqrfzpmM7j_PFkt2I&2@Pus0Pb)Bb#yV#k6VDP``eC7>1`BpE&OiocFOPvnu4`ZczO;C51H^JWR9%D*^Dl z39DF}8vvNEfz0HQAFq7_AJo`FUs~K{xBltm!HPEF?7ZfT~$e!;ArHZtUI4(0p z+7rC~d|iZcufRg_R(l-9yz~xMg<|{)R+AtY0pD#0bl`1X_zF^Jwym8BFy6g&JKmpn z5Y7o+ib|+gRiH@b2Qd&90mPK_)vITIx~t&G7`!uh9M-K{hr)#mdz7ZqdYd?Jqdo5u z?>*_XIr2-DDuuAHFnhiuyvS3u(+RHakt5eLs(5or$UKGpN5dMHtDKbp)Wq)$ju1`w z`xM>4uHAu|qmu^3Q94WhI;_r3yk(6YwV-hil2n#`(5}q@N*&ZNM^qvaW zL$oa5oClv8K}>rEe*#dYZy4BLxMbo_ibRLtOjZ+R9{e=_s}BJh&!krm4;I(`@ubWW@HZ~U$Wq?K25`ICT9hvm%5 zIutZ*?5x$l#Es11mDoCaAeJp%oS}`IN*96Sz{ORjH>}5h&>9_^RyV)Ik=M7+i}hUP zR*BLjqMS28WThE+E+))yLLdSVtv--5X93Met&d+%L{l~B!21B z6Qrf$-tsYYpFd1BXdl&1k~U9Nb0$6Fdn%N{pq@=IiLQosJ2ga9fpCv2(r1xb|foY(Hg+|Aj>fgfH?~oLyPeV*OhW| z5FL6rJ9>x=4?+Kz8)7z3xAX?Y4|E(ByL`Bk9bK5rce`ax?V7uU000coNklxOtql&rIfgkV+k%jlG0&G*`a06lzty2{DBtvaw}*Qx`6# zAocR@Q@j(}3)6*x3P9Dg3c7TeL>lF~QgIF8!6{Zh1i^9R&;VF;?W`V#dzkHEfI?nDE)fw+?iJjikJT$Q-R_Fgei0q_V4G$;|G zX5_(qhlYdaDf!%`+bX3RDyFAK1SIa;JP??Rnw$zm~&mCp__M1kB-Ut-)V-m{;bl#x`7aRnZ1mHtrAHLzJ>|Q;OZ<0Vt zzxidO&xt51Ycx(D1Ald6PKgM=hN}wcMG8pH;(ey(W;&{WZj;Td*6!6UKV>I9xEvT`xRuq%b^Kh zKbx-txBO?BCfgYjZsLYiz8xkl;owWKh_GPXSE(#?j_a?w zm_DLU3!ZJePPr_r>^^h?!>6otr;!UM9T(h-0e=Ee&vGW2we+b^t6R*ZWpIuoUq_fu z733s7M^)tfq{x{Ua7?@0I6^0sWgVZ zl*3F5&|!YxVRPI^{~siuPqYMo0?>($%K?6W3819rgk_V!DH;G9c!una>=-r4lXGEy z*X8pq3vU&brx|dTwyn}vfY+Cd#)$v4Fu$Caf5_QXcxW(vM1|4q*;<%A_$A2dF%19A z*O~Lri#KEL%I(gk^!mgHyim-@H&_Wko39G2>auql)B54HU-*rxdavhUl1Cj0;ptz2 zMYN1EC4^*3cB_u5J(|nXg$~c>4ebd1@l&T3HSk*N+8F$LQ!E_O34?k*34I}h0Dt=q zpTcL~FG2h{jxIo^+7eb>MDz<*0*GL3YJ8YOK!o*O%7`TvpwEi@F;>2$j`Xjt$fS9% zS0W*c$fEy1i}dwNueJ}9;Z2}dN;)>H4jrZhT3>X@$*Jh_$^Wo%J4cf;01dt2Ov(ac zWDVl0FA9x(7t2dT0L6*Eh?OtVPo=qTabl(^sXI00Bg?44EMPqCghZ|zYQrz4av|A3 z)q2TtZS&iVlR)}CEF>82s>oU9FP5*Ub-(eLzj}xBOj#_@_j5B@S7}DpU?qT1R@I?0 zBS3F7__KcWt?maKOmbfV*$w%WxC+=wWD#j5^jimL_iWF#4vj$WgR<<^M4&WDIntuP|F1lXi^t zDH9_=qkky(i;d?1Tjjbkb1j04+}Uqg4kWqkZ42V;FFHvt>2rO@;m6qy&U(H=^otMAHuWNQd?+H1& z;x-*UnfCn#FAKeal>m}if!isDghTWK$i8RN1J09uKT`JX7IXDc4Ubau;yA-siGLE^ zL5oxmW(*^m<^yBMZPAm4&T|WZS^EL)9h9$r^fJK9%YfQE8tSkl4L>g4j(MwHrk}`M zD2qf+wp5J{S8mo5ai`AW(=m%o7xiDeo#V)CPt4aFaGKE?Y-VPlft3I>zq#|VH+7g- z-=vf8K?+6xO44-&;N!hfF}6KhknZ=U7i9zWVR;9l1jUG&g_QFT0?#jm`3D^H4ggP3 zJ?YOl^gj_=Ya${-@owiw(U~TsSt(uc^^{d;`uZ3g4A6(4uDa_|qh8Lt2Ct07d!rWM zV4Q1AiiUTIR!sa%ncOcTs7yB4cVy1Imp9#&Ru`T1eL~ z6$*5a-+=iV8D<+ydxP`LoIq?&Pf9~ z?nBS3TgxMFej|J9p5^h@q9OliRjswnfd}(7x z7l+w+N-~nppU38%M=*5KN@$U-OI<%}#WuQ@^++a_0u30wEvNy*os7BhQht`u=BF05 zFkT=H+0D>D$i~+fFtT!JSY~=;aGc{^cB7B`JcJkigcSU57KJR5^mk)}3M4f84Kz~uJZw^>y4RQOCcwmodNvW}|WZN#P?j0vnu=YSACT@<$ z2TM<)>Esi*VfZmr7#W8ibB<&AjuSX?A^}O4O-uEY#^4lj3b6`t6fuFAE#OZ8&Q4;P z&|+Gjaa&>+qE>$8J;t>DWHGC_ zu#j_6e`O}>*7V9$CHpE(0;*31WMOIJH=XN?tUGa7bY4);L;ZzkSSGVs1CYt=gFpqK z9<`9}U}8E^H}xH3J`OZPs8D&8JDGuAi(*8zU)r}js`eWIRtr408($RM=cr^i5B%L zyT9n_D)UI_kY+30Fv}<5_rpn$x$IG%xeM*mKgnE`tW)PD0fg`|9wEA*ba)D?MsW3n zg_b_cp}u?y4C?*ZGfC*N@`~O=(q`%{-hf6vqnwAlBmk|jk?xONvA_j{t(*er5Jrra z^G%L8Ewr-bH!gtj^A||ny$L%{B%8bCl}7<>8r|CoFfHc)kF?qpT1cD}*$8<_0Ahg0 zgUkJK@q8kTl~ZBN91df__b?WZhcSON4E<@yhIu%A>H-FG|J-0Eu+inr$P$M*cTByn zDVL=NvQjB82_S(FK;x021H{m&djWIDeU_&9k=?+7ZAgpThrP!V(RVGNKW&vssly#d zHAV74DGn5Mh$2&y+vGs2WDi|l5`aek7(SL_Mlk<~wKe@tb?E(+!%tpPIpnHN2Fgza zKB57i)+Q=uWC@fi1_6*mO*TK1R~Pb<0QBV2!K;*TqDH?C`G@n2 z;LAUOAL;tmhl0v?(l}zNuxU)o2eqDYMz4@&CGeF2pVlgc7zGYgry7YDpruz|dMSrv3D~7x(06mzztqYit0Q6^1ZJ3tY zkP}d626K3wsRMA*=Q+sB0&p@*sw9yYS|6k~WMaB3CFWn#a4iVnnr5)So6i@3KMv(^ zdA entry.id === "tinyfish"); +const url = "https://example.com/article"; + +afterEach(() => vi.unstubAllGlobals()); + +describe("TinyFish Search and Fetch", () => { + it("registers both kinds on one API-key provider", () => { + expect(provider.serviceKinds).toEqual(["webSearch", "webFetch"]); + expect(getProvidersByKind("webSearch").some((p) => p.id === "tinyfish")).toBe(true); + expect(getProvidersByKind("webFetch").some((p) => p.id === "tinyfish")).toBe(true); + expect(provider.searchConfig.validateUrl).toContain("/usage?limit=1"); + }); + + it("builds GET search with header auth and maps results", () => { + const request = buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, { + query: "latest release", searchType: "news", token: "secret", maxResults: 5, + country: "TR", language: "tr", domainFilter: ["example.com", "-other.com"], offset: 5, + }); + const parsed = new URL(request.url); + expect(Object.fromEntries(parsed.searchParams)).toEqual({ + query: "latest release", domain_type: "news", location: "TR", language: "tr", + include_domains: "example.com", exclude_domains: "other.com", page: "0", + }); + expect(request.url).not.toContain("secret"); + expect(request.init.headers["X-API-Key"]).toBe("secret"); + expect(buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, { + query: "release", searchType: "web", token: "secret", maxResults: 5, + providerOptions: { baseUrl: "https://example.org/collect" }, + }).url.startsWith(provider.searchConfig.baseUrl)).toBe(true); + const normalized = normalizeSearchResponse("tinyfish", { + total_results: 1, results: [{ title: "Article", url, snippet: "Preview", date: "2026-09-01" }], + }, "latest release", "news"); + expect(normalized.totalResults).toBe(1); + expect(normalized.results[0]).toMatchObject({ title: "Article", url, snippet: "Preview", published_at: "2026-09-01" }); + }); + + it("applies offset within TinyFish fixed-size result pages", async () => { + const entries = Array.from({ length: 10 }, (_, i) => ({ title: `Article ${i}`, url: `${url}/${i}`, snippet: "Preview" })); + vi.stubGlobal("fetch", vi.fn(async () => new Response(JSON.stringify({ results: entries, total_results: 10 }), { + headers: { "content-type": "application/json" }, + }))); + const result = await handleSearchCore({ + body: { query: "release", max_results: 2, offset: 3 }, + provider: { id: "tinyfish" }, providerConfig: provider.searchConfig, + credentials: { apiKey: "secret" }, + }); + expect(result.success).toBe(true); + expect(result.data.results.map((entry) => entry.title)).toEqual(["Article 3", "Article 4"]); + expect(new URL(vi.mocked(fetch).mock.calls[0][0]).searchParams.get("page")).toBe("0"); + }); + + it("rejects offsets that would silently omit requested results", () => { + const params = { query: "release", searchType: "web", token: "secret", maxResults: 5, offset: 8 }; + expect(() => buildSearchRequest({ id: "tinyfish", ...provider.searchConfig }, params)) + .toThrow("TinyFish search offset and max_results must fit within one page"); + }); + + it("maps fetched content and treats HTTP 200 per-URL errors as failures", async () => { + const fetchMock = vi.fn(async () => new Response(JSON.stringify({ + results: [{ url, title: "Article", text: "# Article", language: "en" }], errors: [], + }), { headers: { "content-type": "application/json" } })); + vi.stubGlobal("fetch", fetchMock); + const input = { url, provider: "tinyfish", providerConfig: provider.fetchConfig, credentials: { apiKey: "secret" } }; + const result = await handleFetchCore(input); + expect(result.data).toMatchObject({ provider: "tinyfish", title: "Article", content: { text: "# Article" }, metadata: { language: "en" } }); + expect(fetchMock.mock.calls[0][0]).toBe(provider.fetchConfig.baseUrl); + expect(fetchMock.mock.calls[0][1].headers["X-API-Key"]).toBe("secret"); + expect(JSON.parse(fetchMock.mock.calls[0][1].body)).toEqual({ urls: [url], format: "markdown" }); + + fetchMock.mockResolvedValueOnce(new Response(JSON.stringify({ results: [], errors: [{ url, error: "bot_blocked" }] }), { + headers: { "content-type": "application/json" }, + })); + expect(await handleFetchCore(input)).toMatchObject({ success: false, status: 502, error: "TinyFish fetch failed: bot_blocked" }); + + expect(await handleFetchCore({ ...input, format: "text" })).toMatchObject({ success: false, status: 400 }); + expect(fetchMock).toHaveBeenCalledTimes(2); + + fetchMock.mockResolvedValueOnce(new Response(JSON.stringify({ error: { message: "Invalid API key" } }), { + status: 401, headers: { "content-type": "application/json" }, + })); + expect(await handleFetchCore(input)).toMatchObject({ success: false, status: 401 }); + }); +}); From 8f9ff44f270cec369c3546290300d565e89bca0f Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 12:42:24 +0700 Subject: [PATCH 05/41] fix(codex): remove ghost models, add gpt-daybreak/reserve, route gpt-5.x/6.x bare slugs to codex (#4418) --- open-sse/providers/registry/codex.js | 13 +- open-sse/services/model.js | 13 +- .../unit/codex-registry-model-routing.test.js | 135 ++++++++++++++++++ 3 files changed, 153 insertions(+), 8 deletions(-) create mode 100644 tests/unit/codex-registry-model-routing.test.js diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index c7e192e9..bfa30e3e 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -63,12 +63,11 @@ export default { { id: "gpt-5.6-luna-review", name: "GPT 5.6 Luna Review", upstreamModelId: "gpt-5.6-luna", quotaFamily: "review" }, { id: "gpt-5.5", name: "GPT 5.5" }, { id: "gpt-5.5-review", name: "GPT 5.5 Review", upstreamModelId: "gpt-5.5", quotaFamily: "review" }, - { id: "gpt-5.4", name: "GPT 5.4" }, - { id: "gpt-5.4-review", name: "GPT 5.4 Review", upstreamModelId: "gpt-5.4", quotaFamily: "review" }, - { id: "gpt-5.4-mini", name: "GPT 5.4 Mini" }, - { id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" }, - { id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" }, - { id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" }, + // gpt-5.4 / gpt-5.4-mini / gpt-5.3-codex-spark removed: absent from backend-api/codex/models + // for ChatGPT Plus/Pro accounts and return HTTP 400 "model is not supported" (#4202). + // gpt-daybreak-blue-latest and gpt-reserve added: confirmed live via backend-api/codex/models (#4202). + { id: "gpt-daybreak-blue-latest", name: "GPT Daybreak Blue" }, + { id: "gpt-reserve", name: "GPT Reserve" }, // Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived // from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398). { id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" }, @@ -81,7 +80,7 @@ export default { { id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, - { id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + // gpt-5.4-image removed alongside gpt-5.4 (both are dead on the backend) (#4202). { id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, ], serviceKinds: ["llm","image"], diff --git a/open-sse/services/model.js b/open-sse/services/model.js index 15c50195..7715b586 100644 --- a/open-sse/services/model.js +++ b/open-sse/services/model.js @@ -124,8 +124,19 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) { // Config-driven prefix → provider inference (first match wins, fallback "openai"). const MODEL_PREFIX_PROVIDERS = [ - // Codex CLI sends this bare virtual model for auto-review — keep it on OAuth Codex (#1398). + // Codex CLI sends this bare virtual model for auto-review - keep it on OAuth Codex (#1398). [/^codex-auto-review$/, "codex"], + // Codex-only GPT model slugs: present in backend-api/codex/models but not on the + // OpenAI API. Without these rules a bare model id (e.g. "gpt-5.6-terra" from the + // Codex CLI /model picker) resolves to provider "openai", which 404s for users that + // only have a Codex OAuth account and no OpenAI API key (#4405). + // Ranges covered: gpt-5.x, gpt-6.x, gpt-daybreak-*, gpt-reserve* — all are + // Codex-backend models. Plain "gpt-4*" / "gpt-3.5*" / "gpt-4o*" fall through to + // the generic gpt-* → openai rule below. + [/^gpt-[56]\./, "codex"], + [/^gpt-6-/, "codex"], + [/^gpt-daybreak-/, "codex"], + [/^gpt-reserve/, "codex"], [/^claude-/, "anthropic"], [/^gemini-/, "gemini"], [/^gpt-/, "openai"], diff --git a/tests/unit/codex-registry-model-routing.test.js b/tests/unit/codex-registry-model-routing.test.js new file mode 100644 index 00000000..aaa52371 --- /dev/null +++ b/tests/unit/codex-registry-model-routing.test.js @@ -0,0 +1,135 @@ +/** + * Tests for #4202 and #4405 + * + * #4202 — Codex registry has ghost models (always HTTP 400) and is missing + * gpt-daybreak-blue-latest / gpt-reserve. + * Fix: remove gpt-5.4/mini/spark entries + gpt-5.4-image; add gpt-daybreak-blue-latest and gpt-reserve. + * + * #4405 — Bare Codex model slugs (e.g. gpt-5.6-terra from the CLI /model picker) + * routed to provider "openai" instead of "codex", causing 404 for users without + * an OpenAI API key connection. + * Fix: add codex-specific gpt-5.x / gpt-6.x / gpt-daybreak-* / gpt-reserve* rules + * to MODEL_PREFIX_PROVIDERS before the generic gpt-* → openai rule. + */ + +import { describe, it, expect } from "vitest"; +import fs from "fs"; +import path from "path"; + +// ── #4202 Registry checks ──────────────────────────────────────────────────── + +const codexSrc = fs.readFileSync( + path.resolve("../open-sse/providers/registry/codex.js"), + "utf-8" +); + +describe("Codex registry — ghost models removed (#4202)", () => { + it("gpt-5.4 is removed", () => { + // Match only a standalone entry, not inside gpt-5.4-mini or gpt-5.45 etc. + expect(codexSrc).not.toMatch(/id:\s*"gpt-5\.4"/); + }); + + it("gpt-5.4-mini is removed", () => { + expect(codexSrc).not.toMatch(/id:\s*"gpt-5\.4-mini"/); + }); + + it("gpt-5.3-codex-spark is removed", () => { + expect(codexSrc).not.toMatch(/id:\s*"gpt-5\.3-codex-spark"/); + }); + + it("gpt-5.4-image is removed", () => { + expect(codexSrc).not.toMatch(/id:\s*"gpt-5\.4-image"/); + }); +}); + +describe("Codex registry — new models added (#4202)", () => { + it("gpt-daybreak-blue-latest is present", () => { + expect(codexSrc).toContain('"gpt-daybreak-blue-latest"'); + }); + + it("gpt-reserve is present", () => { + expect(codexSrc).toContain('"gpt-reserve"'); + }); +}); + +describe("Codex registry — still-live models kept (#4202 regression guard)", () => { + it("gpt-5.5 still present", () => { + expect(codexSrc).toContain('"gpt-5.5"'); + }); + + it("gpt-5.6-terra still present", () => { + expect(codexSrc).toContain('"gpt-5.6-terra"'); + }); + + it("gpt-6-astra still present", () => { + expect(codexSrc).toContain('"gpt-6-astra"'); + }); +}); + +// ── #4405 Model prefix routing ─────────────────────────────────────────────── +// Replicate MODEL_PREFIX_PROVIDERS logic from open-sse/services/model.js + +const MODEL_PREFIX_PROVIDERS = [ + [/^codex-auto-review$/, "codex"], + [/^gpt-[56]\./, "codex"], + [/^gpt-6-/, "codex"], + [/^gpt-daybreak-/, "codex"], + [/^gpt-reserve/, "codex"], + [/^claude-/, "anthropic"], + [/^gemini-/, "gemini"], + [/^gpt-/, "openai"], + [/^o[134]/, "openai"], + [/^deepseek-/, "openrouter"], +]; + +function inferProvider(modelName) { + if (!modelName) return "openai"; + const m = modelName.toLowerCase(); + return MODEL_PREFIX_PROVIDERS.find(([re]) => re.test(m))?.[1] || "openai"; +} + +describe("inferProviderFromModelName — Codex gpt-* models route to codex (#4405)", () => { + it("gpt-5.6-terra → codex", () => { + expect(inferProvider("gpt-5.6-terra")).toBe("codex"); + }); + + it("gpt-5.6-sol → codex", () => { + expect(inferProvider("gpt-5.6-sol")).toBe("codex"); + }); + + it("gpt-5.5 → codex", () => { + expect(inferProvider("gpt-5.5")).toBe("codex"); + }); + + it("gpt-6-astra → codex", () => { + expect(inferProvider("gpt-6-astra")).toBe("codex"); + }); + + it("gpt-daybreak-blue-latest → codex", () => { + expect(inferProvider("gpt-daybreak-blue-latest")).toBe("codex"); + }); + + it("gpt-reserve → codex", () => { + expect(inferProvider("gpt-reserve")).toBe("codex"); + }); + + it("gpt-4o → openai (standard model not affected)", () => { + expect(inferProvider("gpt-4o")).toBe("openai"); + }); + + it("gpt-4-turbo → openai (standard model not affected)", () => { + expect(inferProvider("gpt-4-turbo")).toBe("openai"); + }); + + it("gpt-3.5-turbo → openai (standard model not affected)", () => { + expect(inferProvider("gpt-3.5-turbo")).toBe("openai"); + }); + + it("claude-opus-5 → anthropic (regression guard)", () => { + expect(inferProvider("claude-opus-5")).toBe("anthropic"); + }); + + it("codex-auto-review → codex (regression guard)", () => { + expect(inferProvider("codex-auto-review")).toBe("codex"); + }); +}); \ No newline at end of file From 7f5bd15518f6599dc236596dc8c991bea9347f7b Mon Sep 17 00:00:00 2001 From: KiMelody Date: Mon, 28 Sep 2026 12:44:16 +0700 Subject: [PATCH 06/41] fix(tools): dedupe same-name tools for DeepSeek models (#3333) DeepSeek upstream rejects duplicate tool names with 400 'Tool names must be unique' on every endpoint (api.deepseek.com, opencode.go, LiteLLM). dedupeTools now takes {clientTool, model}: MCP-equivalent built-in dedup stays Claude-only, exact same-name dedup applies to any provider serving a deepseek-* model (first definition wins; tool_choice and history references are by name/id so nothing breaks). Adds an offline endpoint routing matrix and live upstream evidence tests. --- open-sse/handlers/chatCore.js | 7 +- open-sse/utils/toolDeduper.js | 60 ++++- .../deepseek-official-responses.real.test.js | 238 ++++++++++++++++++ .../real/opencode-go-deepseek.real.test.js | 125 +++++++++ ...ncode-go-thinking-passthrough.real.test.js | 221 ++++++++++++++++ ...ncode-go-thinking-placeholder.real.test.js | 189 ++++++++++++++ .../opencode-go-tool-session.real.test.js | 237 +++++++++++++++++ .../opencode-zen-free-responses.real.test.js | 196 +++++++++++++++ .../opencode-go-transport-routing.test.js | 175 +++++++++++++ tests/unit/tool-deduper.test.js | 100 ++++++++ 10 files changed, 1534 insertions(+), 14 deletions(-) create mode 100644 tests/translator/real/deepseek-official-responses.real.test.js create mode 100644 tests/translator/real/opencode-go-deepseek.real.test.js create mode 100644 tests/translator/real/opencode-go-thinking-passthrough.real.test.js create mode 100644 tests/translator/real/opencode-go-thinking-placeholder.real.test.js create mode 100644 tests/translator/real/opencode-go-tool-session.real.test.js create mode 100644 tests/translator/real/opencode-zen-free-responses.real.test.js create mode 100644 tests/unit/opencode-go-transport-routing.test.js create mode 100644 tests/unit/tool-deduper.test.js diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index a6230fc4..8755659d 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -216,9 +216,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred stripContinuityFields(translatedBody); } - // Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only). - if (clientTool === "claude" && Array.isArray(translatedBody.tools)) { - const { tools: deduped, stripped } = dedupeTools(translatedBody.tools); + // Tool normalization: MCP-equivalent built-in dedup (Claude clients) + same-name + // dedup for DeepSeek models (upstream rejects duplicate tool names on all endpoints). + if (Array.isArray(translatedBody.tools)) { + const { tools: deduped, stripped } = dedupeTools(translatedBody.tools, { clientTool, model }); if (stripped.length > 0) { translatedBody.tools = deduped; log?.debug?.("TOOLDEDUP", `stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`); diff --git a/open-sse/utils/toolDeduper.js b/open-sse/utils/toolDeduper.js index 20c47403..ee6b5aa6 100644 --- a/open-sse/utils/toolDeduper.js +++ b/open-sse/utils/toolDeduper.js @@ -1,6 +1,11 @@ /** - * Strip built-in/duplicate tools when equivalent MCP tools are present. - * Goal: reduce tool definitions token bloat for Claude clients. + * Tool normalization before dispatch: + * - MCP-equivalent built-in tool dedup (Claude clients only, reduces token bloat). + * - Exact same-name tool dedup for DeepSeek models — the DeepSeek upstream rejects + * duplicate tool names with 400 "Tool names must be unique" on every endpoint + * (verified live 2026-08-15 against api.deepseek.com, opencode.go and a LiteLLM + * gateway; GLM/MiniMax/Kimi upstreams accept duplicates). First definition wins, + * tool_choice and message-history references are by name/id so nothing breaks. */ const DEDUP_RULES = [ @@ -30,20 +35,53 @@ function matches(name, pattern) { return pattern instanceof RegExp ? pattern.test(name) : false; } -function dedupeTools(tools) { +// "model(level)" is a 9router thinking override; strip before matching. +function isDeepSeekModel(model) { + if (typeof model !== "string") return false; + return /^deepseek-/.test(model.replace(/\([^()]+\)\s*$/, "").trim()); +} + +/** + * @param {Array} tools - translated tools array + * @param {Object} [opts] + * @param {string|null} [opts.clientTool] - detected client ("claude" | "codex" | ...) + * @param {string|null} [opts.model] - model id, may carry a (level) thinking suffix + * @returns {{ tools: Array, stripped: Array }} + */ +function dedupeTools(tools, opts = {}) { if (!Array.isArray(tools) || tools.length === 0) return { tools, stripped: [] }; const names = tools.map(getToolName); const toStrip = new Set(); - for (const rule of DEDUP_RULES) { - const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p))); - if (!hasTrigger) continue; - for (const n of names) { - if (rule.strip.some((p) => matches(n, p))) toStrip.add(n); + const toDrop = new Set(); // indices of duplicate same-name tools + + // MCP-based built-in dedup: Claude clients only (existing behavior). + if (opts.clientTool === "claude") { + for (const rule of DEDUP_RULES) { + const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p))); + if (!hasTrigger) continue; + for (const n of names) { + if (rule.strip.some((p) => matches(n, p))) toStrip.add(n); + } } } - if (toStrip.size === 0) return { tools, stripped: [] }; - const out = tools.filter((t) => !toStrip.has(getToolName(t))); - return { tools: out, stripped: Array.from(toStrip) }; + + // Exact-name dedup: DeepSeek upstream rejects duplicate tool names. Applies to + // every client × provider that serves a deepseek-* model (official API, Console Go, + // LiteLLM gateways); non-DeepSeek models are untouched. + if (isDeepSeekModel(opts.model)) { + const seen = new Set(); + for (let i = 0; i < tools.length; i++) { + const n = getToolName(tools[i]); + if (!n) continue; + if (seen.has(n)) toDrop.add(i); + else seen.add(n); + } + } + + if (toStrip.size === 0 && toDrop.size === 0) return { tools, stripped: [] }; + const out = tools.filter((t, i) => !toDrop.has(i) && !toStrip.has(getToolName(t))); + const stripped = Array.from(toDrop).map((i) => getToolName(tools[i])).concat(Array.from(toStrip)); + return { tools: out, stripped }; } export { dedupeTools }; diff --git a/tests/translator/real/deepseek-official-responses.real.test.js b/tests/translator/real/deepseek-official-responses.real.test.js new file mode 100644 index 00000000..32b20987 --- /dev/null +++ b/tests/translator/real/deepseek-official-responses.real.test.js @@ -0,0 +1,238 @@ +// REAL matrix: DeepSeek OFFICIAL Responses API — direct behavior vs 9router translation. +// +// Context: DeepSeek recently shipped an OpenAI-Responses-compatible endpoint +// (https://api.deepseek.com/responses) but the 9router registry does not declare it — +// responses-format requests are translated to /chat/completions. This suite answers: +// +// 1. DIRECT: does the official /responses endpoint demand reasoning pass-back +// (like official /chat/completions does: "reasoning_content must be passed back")? +// 2. 9ROUTER: when a responses-format client hits 9router, what does the translation +// to /chat/completions do with reasoning items — and does multi-turn survive the +// pass-back requirement on the chat endpoint? +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/deepseek-official-responses.real.test.js +// +// Reads the DeepSeek API key from the local 9router DB (connection id +// a91b07f2-878a-45b0-beb5-56981409ab0c). Uses handleChatCore for the 9router half. +import { describe, it, expect } from "vitest"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; +import { openaiResponsesToOpenAIRequest } from "../../../open-sse/translator/request/openai-responses.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "deepseek"; +const MODEL = "deepseek-reasoner"; +const TIMEOUT_MS = 120000; +const CRED_ISSUE = [401, 402, 403, 429]; + +// API key from the local DB connection (no dashboard/DB writes — read-only). +function readApiKey() { + const Database = require("better-sqlite3"); + const path = require("path"); + const dbPath = path.join(process.env.APPDATA, "9router", "db", "data.sqlite"); + const db = new Database(dbPath, { readonly: true }); + const rows = db.prepare( + "SELECT data FROM providerConnections WHERE provider = ? AND isActive = 0 ORDER BY updatedAt DESC LIMIT 1" + ).all("deepseek"); + db.close(); + if (!rows.length) return null; + try { + const data = JSON.parse(rows[0].data); + return data.apiKey || null; + } catch { return null; } +} + +// ---- DIRECT half: raw fetch to the official /responses endpoint ---- +async function directResponses(body) { + const res = await fetch("https://api.deepseek.com/responses", { + method: "POST", + headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" }, + body: JSON.stringify(body), + }); + const text = await res.text(); + let json = null; + try { json = JSON.parse(text); } catch { /* keep null */ } + return { status: res.status, json, raw: text }; +} + +const TURN1_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17+26, then reply with ONLY the number." }] }; +const TURN2_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer? Reply with just the number." }] }; + +async function directTurn1() { + const out = await directResponses({ + model: MODEL, stream: false, max_output_tokens: 512, + reasoning: { effort: "high" }, + input: [TURN1_USER], + }); + return out; +} + +describe.skipIf(!RUN_REAL)("DIRECT: DeepSeek official /responses endpoint", () => { + it("has a DeepSeek API key in the local DB", () => { + expect(process.env.DS_KEY && process.env.DS_KEY.startsWith("sk-")).toBe(true); + }); + + it("single-turn works and returns a Responses-shape payload", async () => { + const out = await directResponses({ model: MODEL, stream: false, input: [TURN1_USER] }); + console.log(`[direct single] status=${out.status}`); + expect(out.status).toBe(200); + expect(out.json?.object).toBe("response"); + expect(out.json?.output?.some((o) => o.type === "message")).toBe(true); + }); + + it("turn1 with reasoning returns a reasoning item", async () => { + const out = await directTurn1(); + expect(out.status).toBe(200); + const types = (out.json?.output || []).map((o) => o.type); + console.log(`[direct turn1] output types=${types.join(",")} reasoning_tokens=${out.json?.usage?.output_tokens_details?.reasoning_tokens}`); + expect(types).toContain("reasoning"); + }); + + it("turn2 WITHOUT reasoning item is accepted (no pass-back requirement)", async () => { + const t1 = await directTurn1(); + const outMsg = (t1.json?.output || []).find((o) => o.type === "message"); + const out = await directResponses({ + model: MODEL, stream: false, max_output_tokens: 256, + input: [TURN1_USER, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER], + }); + console.log(`[direct turn2 no-reasoning] status=${out.status}`); + expect(out.status).toBe(200); + }); + + it("turn2 WITH reasoning item is accepted", async () => { + const t1 = await directTurn1(); + const reas = (t1.json?.output || []).find((o) => o.type === "reasoning"); + const outMsg = (t1.json?.output || []).find((o) => o.type === "message"); + const out = await directResponses({ + model: MODEL, stream: false, max_output_tokens: 256, + input: [TURN1_USER, reas, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER], + }); + console.log(`[direct turn2 with-reasoning] status=${out.status}`); + expect(out.status).toBe(200); + }); + + it("streaming single-turn returns Responses SSE events", async () => { + const res = await fetch("https://api.deepseek.com/responses", { + method: "POST", + headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" }, + body: JSON.stringify({ model: MODEL, stream: true, max_output_tokens: 256, input: [TURN1_USER] }), + }); + expect(res.status).toBe(200); + const text = await res.text(); + const events = (text.match(/event: ([a-z_.]+)/g) || []).map((e) => e.slice(7)); + const hasCreated = events.includes("response.created"); + const hasCompleted = events.includes("response.completed"); + console.log(`[direct stream] status=200 events=${events.join(",")}`); + expect(hasCreated).toBe(true); + expect(hasCompleted).toBe(true); + }, TIMEOUT_MS); +}); + +// ---- UNIT half: what the responses→chat translator does with reasoning ---- +describe("UNIT: responses→chat translator reasoning handling", () => { + it("attaches reasoning item text as reasoning_content on the assistant message", () => { + const body = { + input: [ + TURN1_USER, + { type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, + TURN2_USER, + ], + }; + const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null); + const assistant = result.messages.find((m) => m.role === "assistant"); + expect(assistant.reasoning_content).toContain("43"); + expect(result.messages.length).toBe(3); + }); + + it("leaves assistant messages bare when no reasoning item is present", () => { + const body = { + input: [TURN1_USER, { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, TURN2_USER], + }; + const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null); + const assistant = result.messages.find((m) => m.role === "assistant"); + expect(assistant.reasoning_content).toBeUndefined(); + }); +}); + +// ---- 9ROUTER half: handleChatCore with responses sourceFormat ---- +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function via9router(body) { + const credentials = { + apiKey: process.env.DS_KEY, + connectionId: "a91b07f2-878a-45b0-beb5-56981409ab0c", + providerSpecificData: { connectionProxyEnabled: false, connectionProxyUrl: "", connectionNoProxy: "" }, + }; + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${MODEL}` }, + modelInfo: { provider: PROVIDER, model: MODEL }, + credentials, + connectionId: credentials.connectionId, + sourceFormatOverride: "openai-responses", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +describe.skipIf(!RUN_REAL)(`9ROUTER: responses-format → ${PROVIDER} translation`, () => { + it("single-turn responses request succeeds via /chat/completions", async () => { + const out = await via9router({ + stream: true, max_output_tokens: 128, instructions: "You are concise.", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }], + }); + if (out.skip) return expect(true).toBe(true); + // 9router routes the request to /chat/completions but re-encodes the stream + // back to the client's source format (Responses SSE shape). + const isResponsesShape = /event: response\.|"type"\s*:\s*"response|"type":"response/.test(out.raw || ""); + const isChatShape = /chat\.completion\.chunk|"delta"/.test(out.raw || ""); + console.log(`[9r single] status=${out.status} bytes=${out.raw?.length} responsesShape=${isResponsesShape} chatShape=${isChatShape}`); + expect(out.ok).toBe(true); + expect(isResponsesShape || isChatShape).toBe(true); + }, TIMEOUT_MS); + + it("multi-turn WITH reasoning item in history succeeds (translator attaches reasoning_content)", async () => { + const out = await via9router({ + stream: true, max_output_tokens: 128, + input: [ + TURN1_USER, + { type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, + TURN2_USER, + ], + }); + if (out.skip) return expect(true).toBe(true); + console.log(`[9r multi with-reasoning] status=${out.status} bytes=${out.raw?.length}`); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("multi-turn WITHOUT reasoning item in history (diagnostic: chat endpoint pass-back)", async () => { + const out = await via9router({ + stream: true, max_output_tokens: 128, + input: [ + TURN1_USER, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, + TURN2_USER, + ], + }); + if (out.skip) return expect(true).toBe(true); + console.log(`[9r multi no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // Diagnostic: the official chat endpoint requires reasoning_content pass-back; + // a 400 here proves the translation path needs the reasoning item to survive. + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-go-deepseek.real.test.js b/tests/translator/real/opencode-go-deepseek.real.test.js new file mode 100644 index 00000000..06db975f --- /dev/null +++ b/tests/translator/real/opencode-go-deepseek.real.test.js @@ -0,0 +1,125 @@ +// REAL endpoint matrix for opencode-go DeepSeek models. +// +// Verifies, against the live upstream (https://opencode.ai/zen/go), that each client +// request format actually WORKS for DeepSeek on the endpoint 9router routes it to: +// +// Claude (/v1/messages) → must return a Claude-shape SSE +// Codex (/v1/responses) → must return an OpenAI Responses-shape SSE +// OpenAI (/v1/chat/completions) → must return an OpenAI chat-shape SSE +// +// This is the live counterpart of tests/unit/opencode-go-transport-routing.test.js +// (which proves the routing decision offline). A cell here fails when the upstream +// rejects the routed endpoint+model combination — exactly the 400 that #3332 reported +// for Claude→deepseek-v4-flash on /messages. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-deepseek.real.test.js +// +// Requires an active opencode-go credential in the local DB (add via 9router dashboard). +// Skips (pass) only on credential/quota/plan rejections, mirroring the other .real tests. +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const TIMEOUT_MS = 90000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +// One request. Returns { raw } | "skip"; throws on a real upstream rejection. +async function runChat(model, body, sourceFormat) { + const credentials = await getProviderCredentials(PROVIDER, new Set(), model); + if (!credentials || credentials.allRateLimited) return "skip"; + const refreshed = await checkAndRefreshToken(PROVIDER, credentials); + + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${model}` }, + modelInfo: { provider: PROVIDER, model }, + credentials: refreshed, + connectionId: credentials.connectionId, + sourceFormatOverride: sourceFormat, + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status)) return "skip"; + if (status >= 500 || status === 406) return "skip"; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return "skip"; + throw new Error(`${PROVIDER}/${model} ${sourceFormat} [${result.status}]: ${result.error}`); + } + return { raw: await drainSSE(result.response) }; +} + +// SSE markers: response is re-encoded back to the client's source format. +const SSE_MARKER = { + openai: /chat\.completion\.chunk|"delta"|\[DONE\]/, + "openai-responses": /response\.|"type"\s*:\s*"response|\[DONE\]/, + claude: /event:\s*\w|"type"\s*:\s*"(message_start|content_block|message_delta)"/, +}; + +const MAX_TOKENS = 128; + +const BODIES = { + claude: () => ({ + stream: true, + max_tokens: MAX_TOKENS, + system: [{ type: "text", text: "You are concise." }], + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }), + openai: () => ({ + stream: true, + max_tokens: MAX_TOKENS, + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }), + "openai-responses": () => ({ + stream: true, + max_output_tokens: MAX_TOKENS, + instructions: "You are concise.", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }], + }), +}; + +// Cells: (model, format). `(max)` reproduces the 9router thinking override sent by +// Claude Code — it must land on the same endpoint as the bare id. +const CELLS = [ + ["deepseek-v4-flash", "claude"], + ["deepseek-v4-flash(max)", "claude"], + ["deepseek-v4-flash", "openai-responses"], + ["deepseek-v4-flash", "openai"], + ["deepseek-v4-pro", "claude"], + ["deepseek-v4-pro", "openai-responses"], + // Control: MiniMax keeps /messages for Claude clients. + ["minimax-m3", "claude"], +]; + +describe.skipIf(!RUN_REAL)(`REAL opencode-go DeepSeek endpoint matrix`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash"); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + for (const [model, fmt] of CELLS) { + it(`${fmt}-format client → ${model} returns ${fmt}-shape SSE`, async () => { + const out = await runChat(model, BODIES[fmt](), fmt); + if (out === "skip") { + console.warn(`[skip] ${PROVIDER}/${model} ${fmt}: credential/quota/capability`); + return expect(true).toBe(true); + } + expect(out.raw.length, `${model} ${fmt}: empty SSE`).toBeGreaterThan(0); + expect(SSE_MARKER[fmt].test(out.raw), `${model} ${fmt}: wrong SSE shape (routed endpoint rejected?)`).toBe(true); + }, TIMEOUT_MS); + } +}); diff --git a/tests/translator/real/opencode-go-thinking-passthrough.real.test.js b/tests/translator/real/opencode-go-thinking-passthrough.real.test.js new file mode 100644 index 00000000..608e26a6 --- /dev/null +++ b/tests/translator/real/opencode-go-thinking-passthrough.real.test.js @@ -0,0 +1,221 @@ +// REAL multi-turn thinking pass-back matrix for opencode-go DeepSeek. +// +// Question under test: does the /responses endpoint have the same "cc problem" +// as /messages — i.e. DeepSeek rejects a follow-up turn whose assistant history +// lacks reasoning content, because 9router does not inject a placeholder on that +// path (injectReasoningContent only rewrites body.messages, not the responses +// `input` array)? +// +// Method: turn 1 asks a thinking question non-streamed (so the upstream's own +// output JSON is easy to inspect), then turn 2 replays the assistant turn +// (reasoning + output) in the exact shape the client would, and records whether +// the upstream accepts it. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-passthrough.real.test.js +// +// Cells: +// - openai-responses: assistant turn WITHOUT reasoning item (plain client replay) +// - openai-responses: assistant turn WITH reasoning item (Codex-style store=false replay) +// - openai: control — known-good (injectReasoningContent covers chat path) +// - claude: control — known-broken (handlesThinkingBlocks excludes opencode-go) +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const MODEL = "deepseek-v4-flash"; +const TIMEOUT_MS = 120000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function credentials() { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +// Fire one request; returns { ok, status, raw } — never throws on upstream 4xx. +async function send(body, sourceFormat, creds) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${MODEL}` }, + modelInfo: { provider: PROVIDER, model: MODEL }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: sourceFormat, + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +// Non-stream responses-format turn 1 — returns the full JSON response object. +async function responsesTurn1(creds, retries = 2) { + const body = { + stream: false, + max_output_tokens: 1024, + reasoning: { effort: "high" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }], + }; + for (let i = 0; i <= retries; i++) { + const out = await send(body, "openai-responses", creds); + if (out.ok) { + try { return { ok: true, json: JSON.parse(out.raw) }; } catch { return { ok: true, json: null, raw: out.raw }; } + } + if (!out.skip && out.status) return out; // real upstream rejection, don't retry + if (i < retries) await new Promise((r) => setTimeout(r, 2000)); // transient 429 → back off + } + return { ok: false, skip: true }; +} + +describe.skipIf(!RUN_REAL)(`REAL opencode-go thinking pass-back (${PROVIDER}/${MODEL})`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + it("openai-responses: follow-up with NO reasoning item in assistant history", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const t1 = await responsesTurn1(creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const outputMsg = t1.json?.output?.find?.((o) => o.type === "message"); + const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || ""; + const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning"); + console.log(`[turn1] output_text=${JSON.stringify(text.slice(0, 60))} reasoning_item=${!!reasoningItem}`); + + const body = { + stream: false, + max_output_tokens: 128, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] }, + ], + }; + const out = await send(body, "openai-responses", creds); + console.log(`[responses no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // Diagnostic only — a 400 here is the "cc problem" on the responses path. + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); + + it("openai-responses: follow-up WITH reasoning item (Codex-style replay)", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const t1 = await responsesTurn1(creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const outputMsg = t1.json?.output?.find?.((o) => o.type === "message"); + const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || ""; + const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning"); + console.log(`[turn1] reasoning item present=${!!reasoningItem}`); + + const body = { + stream: false, + max_output_tokens: 128, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }, + ...(reasoningItem ? [reasoningItem] : []), + { type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] }, + ], + }; + const out = await send(body, "openai-responses", creds); + console.log(`[responses with-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); + + it("openai-responses: streaming 2-turn conversation with thinking enabled", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const turn1Body = { + stream: true, + max_output_tokens: 1024, + reasoning: { effort: "high" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }], + }; + const t1 = await send(turn1Body, "openai-responses", creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + console.log(`[responses stream turn1] bytes=${t1.raw.length} marker=${/response\.|"type":"response/.test(t1.raw)}`); + + // Client replays the assistant turn WITHOUT any reasoning content (9router does + // not inject reasoning_content on the input[] path). + const turn2Body = { + stream: true, + max_output_tokens: 128, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "108" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] }, + ], + }; + const t2 = await send(turn2Body, "openai-responses", creds); + console.log(`[responses stream turn2] status=${t2.status} bytes=${t2.raw?.length} marker=${/response\.|"type":"response/.test(t2.raw || "")}`); + expect(t2.ok).toBe(true); + }, TIMEOUT_MS); + + it("openai (control): follow-up with reasoning_content in assistant history", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const body = { + stream: false, + max_tokens: 1024, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: "42", reasoning_content: "17 + 26 = 43. Wait, 17+26 = 43? 17+20=37, 37+6=43. Answer: 43." }, + { role: "user", content: "What was your final answer?" }, + ], + }; + const out = await send(body, "openai", creds); + console.log(`[openai control] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`); + // Control: injectReasoningContent covers the chat path; a 400 here is a real bug. + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("claude (control): follow-up with plain text assistant turn (known-broken on master)", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const body = { + stream: false, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: [{ type: "text", text: "43" }] }, + { role: "user", content: "What was your final answer?" }, + ], + }; + let out; + try { + out = await send(body, "claude", creds); + } catch (e) { + out = { ok: false, status: "threw", raw: String(e?.message || e) }; + } + console.log(`[claude control] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // Known-broken on master (handlesThinkingBlocks excludes opencode-go): we record, + // not assert — the pass-back fix should flip this to ok. + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-go-thinking-placeholder.real.test.js b/tests/translator/real/opencode-go-thinking-placeholder.real.test.js new file mode 100644 index 00000000..aada5099 --- /dev/null +++ b/tests/translator/real/opencode-go-thinking-placeholder.real.test.js @@ -0,0 +1,189 @@ +// REAL: opencode.go /messages thinking-placeholder acceptance + real-thinking pass-back. +// +// Decides the form of the thinking-injection follow-up: +// +// B-cell-1 /messages, thinking enabled + tool_use, assistant turn carries an +// UNSIGNED thinking placeholder {type:"thinking", thinking:"."} +// B-cell-2 /messages, same, assistant turn carries a SIGNED thinking placeholder +// (DEFAULT_THINKING_CLAUDE_SIGNATURE) — the exact shape `prepareClaudeRequest` +// would inject today if opencode-go were added to `handlesThinkingBlocks` +// (claude.js routes non-deepseek providers through the signed branch). +// B-cell-3 /messages, same, assistant turn WITHOUT any thinking block — the known +// 400 repro; sanity check that the cells above actually exercise the +// pass-back validation. +// C-cell-4 REAL multi-turn: turn 1 asks a thinking question on /messages and +// receives DeepSeek's own thinking block (no signature, as emitted by +// 9router's response translator openai-to-claude.js:138-156); turn 2 +// replays that unsigned thinking block verbatim. Proves the direct path +// accepts real unsigned thinking — the natural endpoint state for C. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-placeholder.real.test.js +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; +import { DEFAULT_THINKING_CLAUDE_SIGNATURE } from "../../../open-sse/config/defaultThinkingSignature.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const MODEL = "deepseek-v4-flash"; +const TIMEOUT_MS = 90000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function prepare() { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +async function runChat(body, creds, model = MODEL) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${model}` }, + modelInfo: { provider: PROVIDER, model }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "claude", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +const TOOL = { name: "get_weather", description: "Get weather", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } }; + +function toolTurnBody(assistantContent) { + return { + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + tools: [TOOL], + messages: [ + { role: "user", content: "Weather in Paris?" }, + { role: "assistant", content: assistantContent }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: '{"temp":"20C"}' }] }, + { role: "user", content: "Summarize in one short sentence." }, + ], + }; +} + +describe.skipIf(!RUN_REAL)(`REAL thinking placeholder acceptance (${PROVIDER}/${MODEL})`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + it("B-cell-1: unsigned thinking placeholder accepted on /messages", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + const out = await runChat(toolTurnBody([ + { type: "thinking", thinking: "." }, + { type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } }, + ]), creds); + console.log(`[B1 unsigned] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`); + expect(out.skip).not.toBe(true); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("B-cell-2: SIGNED thinking placeholder (prepareClaudeRequest shape) on /messages", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + const out = await runChat(toolTurnBody([ + { type: "thinking", thinking: ".", signature: DEFAULT_THINKING_CLAUDE_SIGNATURE }, + { type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } }, + ]), creds); + console.log(`[B2 signed] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`); + // This is the exact shape a naive handlesThinkingBlocks addition would inject. + // A 400 here means the follow-up MUST route opencode-go through the unsigned branch. + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("B-cell-3: REAL turn1 thinking, turn2 replay WITHOUT the thinking block (pass-back failure)", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + + // Turn 1: real thinking output on /messages (no tools) — establishes a + // thinking-bearing turn in history. + const t1 = await runChat({ + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }], + }, creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + + // Turn 2: replay the assistant turn WITHOUT the thinking block — the shape a + // client whose history lost the thinking (or a gateway that dropped it) sends. + const out = await runChat({ + stream: true, + max_tokens: 256, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: [{ type: "text", text: "43" }] }, + { role: "user", content: "What was your final answer? Reply with just the number." }, + ], + }, creds); + console.log(`[B3 real-thinking missing] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // NOTE (2026-08-16): direct raw upstream rejects this (500), but through the + // gateway it passes because `injectReasoningContent` (MODEL_RULES /deepseek/i, + // executor transformRequest) injects `reasoning_content: " "` on the assistant + // message before dispatch, and the /messages shim honors that field. This cell + // is therefore recorded as evidence of the mechanism, not asserted as a bug. + console.warn(`[B3] direct 500 vs gateway-200: shim honors reasoning_content field`); + expect(out.skip).not.toBe(true); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("C-cell-4: real thinking block from upstream replayed verbatim on /messages", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + + // Turn 1: thinking enabled, no tools — capture DeepSeek's own thinking block. + const t1 = await runChat({ + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }], + }, creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const thinkingText = (t1.raw.match(/thinking_delta[^\n]*\n[^\n]*"thinking":\s*"([^"]+)/s) || [])[1] || ""; + console.log(`[turn1] thinking_delta_len=${thinkingText.length} marker=${/content_block_delta/.test(t1.raw)}`); + const noThinkingBlocks = !/type":"thinking"/.test(t1.raw); + + // Turn 2: replay the assistant turn with the upstream's real (unsigned) thinking + // block — the exact conversation state a Claude Code client would have. + const out = await runChat({ + stream: true, + max_tokens: 256, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: [ + { type: "thinking", thinking: thinkingText || "17 + 26 = 43" }, + { type: "text", text: "43" }, + ] }, + { role: "user", content: "What was your final answer? Reply with just the number." }, + ], + }, creds); + console.log(`[turn2 real-thinking replay] status=${out.status} noThinkingBlocks=${noThinkingBlocks}`); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-go-tool-session.real.test.js b/tests/translator/real/opencode-go-tool-session.real.test.js new file mode 100644 index 00000000..23bb29e9 --- /dev/null +++ b/tests/translator/real/opencode-go-tool-session.real.test.js @@ -0,0 +1,237 @@ +// REAL: full tool-use conversation sessions + thinking semantics + non-streaming +// paths for opencode-go DeepSeek. +// +// Covers the blind spots of the basic endpoint matrix (which only used plain-text +// bodies): the real 2-turn tool loop that #3332 originally reported 400 for, +// whether the `(max)` thinking suffix actually produces thinking output, the +// non-streaming code paths, and the chat-only-model fallback route. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-tool-session.real.test.js +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const TIMEOUT_MS = 120000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +const WEATHER_TOOL = { name: "get_weather", description: "Get weather for a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } }; +const TIME_TOOL = { name: "get_time", description: "Get current time in a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } }; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function prepare(model) { + const creds = await getProviderCredentials(PROVIDER, new Set(), model); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +async function runChat(body, creds, model) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${model}` }, + modelInfo: { provider: PROVIDER, model }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "claude", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +// Parse Claude-shape SSE blocks: [{type:"tool_use",id,name,input}, ...] +function extractToolUses(raw) { + const blocks = []; + for (const chunk of raw.split("\n\n")) { + const line = chunk.split("\n").find((l) => l.startsWith("data: ")); + if (!line) continue; + try { + const d = JSON.parse(line.slice(6)); + if (d.type === "content_block_start" && d.content_block?.type === "tool_use") { + blocks.push({ id: d.content_block.id, name: d.content_block.name, input: d.content_block.input }); + } + } catch { /* skip malformed */ } + } + return blocks; +} + +const THINKING_BODY = { + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, +}; + +describe.skipIf(!RUN_REAL)(`REAL tool sessions + semantics (${PROVIDER})`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash"); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + it("2-turn tool loop via /messages with thinking enabled (the original 400 shape)", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + + // Turn 1: real model turn that should call the tool. + const t1 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL], + messages: [{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }], + }, creds, model); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const toolUses = extractToolUses(t1.raw); + console.log(`[loop turn1] status=200 tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`); + if (toolUses.length === 0) { console.warn("[skip] model did not call a tool on turn1"); return expect(true).toBe(true); } + + // Turn 2: replay the assistant tool_use (no thinking block — client-side real + // history may or may not carry it; gateway reasoning_content covers pass-back) + // and return the tool result. This is the exact conversation shape that 400'd + // before (and which the endpoint matrix never exercised). + const t2 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL], + messages: [ + { role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }, + { role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) }, + { role: "user", content: [ + ...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: '{"temp":"20C"}' })), + { type: "text", text: "Summarize in one short sentence." }, + ] }, + ], + }, creds, model); + console.log(`[loop turn2] status=${t2.status} bytes=${t2.raw?.length}`); + expect(t2.skip).not.toBe(true); + expect(t2.ok).toBe(true); + }, TIMEOUT_MS); + + it("parallel tool_use turn replayed with all results (thinking enabled)", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + + const t1 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL, TIME_TOOL], + messages: [{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }], + }, creds, model); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const toolUses = extractToolUses(t1.raw); + console.log(`[parallel turn1] tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`); + if (toolUses.length === 0) { console.warn("[skip] model did not call tools on turn1"); return expect(true).toBe(true); } + + const t2 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL, TIME_TOOL], + messages: [ + { role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }, + { role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) }, + { role: "user", content: [ + ...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: t.name === "get_weather" ? '{"temp":"20C"}' : '{"time":"14:30"}' })), + { type: "text", text: "Summarize in one short sentence." }, + ] }, + ], + }, creds, model); + console.log(`[parallel turn2] status=${t2.status} bytes=${t2.raw?.length}`); + expect(t2.skip).not.toBe(true); + expect(t2.ok).toBe(true); + }, TIMEOUT_MS); + + for (const model of ["deepseek-v4-flash(max)", "deepseek-v4-pro(max)"]) { + it(`(max) suffix on ${model} produces thinking output on /messages`, async () => { + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + const out = await runChat({ + stream: true, + max_tokens: 1024, + messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }], + }, creds, model); + if (out.skip) return expect(true).toBe(true); + const hasThinkingDelta = /thinking_delta/.test(out.raw || ""); + const hasText = /content_block_delta.*text/.test(out.raw || "") || /"text":"/.test(out.raw || ""); + console.log(`[${model}] status=${out.status} thinking_delta=${hasThinkingDelta} text=${hasText} bytes=${out.raw?.length}`); + expect(out.ok).toBe(true); + expect(hasThinkingDelta, `(max) should enable thinking on ${model}`).toBe(true); + }, TIMEOUT_MS); + } + + it("non-streaming claude-format request (JSON path) succeeds", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + const out = await runChat({ + stream: false, + max_tokens: 128, + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }, creds, model); + if (out.skip) return expect(true).toBe(true); + const isJson = out.raw?.trim()?.startsWith("{"); + const hasText = /"text"/.test(out.raw || ""); + console.log(`[nonstream claude] status=${out.status} json=${isJson} hasText=${hasText} raw=${out.raw?.slice?.(0, 120)}`); + expect(out.ok).toBe(true); + expect(isJson).toBe(true); + }, TIMEOUT_MS); + + it("non-streaming openai-responses-format request succeeds", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + const result = await handleChatCore({ + body: { + model: `${PROVIDER}/${model}`, stream: false, max_output_tokens: 128, + instructions: "You are concise.", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }], + }, + modelInfo: { provider: PROVIDER, model }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "openai-responses", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || status >= 500 || status === 406) return expect(true).toBe(true); + throw new Error(`[nonstream responses] ${status}: ${result.error}`); + } + const raw = await drainSSE(result.response); + const isJson = raw?.trim()?.startsWith("{"); + const hasResponsesShape = /"output"|"object":"response"/.test(raw || ""); + console.log(`[nonstream responses] status=200 json=${isJson} responsesShape=${hasResponsesShape} raw=${raw?.slice?.(0, 150)}`); + expect(isJson).toBe(true); + expect(hasResponsesShape).toBe(true); + }, TIMEOUT_MS); + + it("chat-only glm-5.2(max) falls back to /chat/completions for a claude-format client", async () => { + const model = "glm-5.2(max)"; + const creds = await prepare("glm-5.2"); + if (!creds) return expect(true).toBe(true); + const out = await runChat({ + stream: true, + max_tokens: 128, + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }, creds, model); + if (out.skip) { console.warn("[skip] glm-5.2 rejected/absent upstream"); return expect(true).toBe(true); } + // Guard blocks /messages; the request is translated to chat and lands on + // /chat/completions, re-encoded to the client's claude format. + const hasClaudeShape = /event:\s*\w|"type"\s*:\s*"(message_start|content_block_delta|message_stop)"/.test(out.raw || ""); + console.log(`[glm fallback] status=${out.status} claudeShape=${hasClaudeShape} bytes=${out.raw?.length}`); + expect(out.ok).toBe(true); + expect(hasClaudeShape).toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-zen-free-responses.real.test.js b/tests/translator/real/opencode-zen-free-responses.real.test.js new file mode 100644 index 00000000..7de01d8e --- /dev/null +++ b/tests/translator/real/opencode-zen-free-responses.real.test.js @@ -0,0 +1,196 @@ +// REAL: zero-cost responses-path smoke on OpenCode Zen's free tier. +// +// After the OpenCode Go subscription lapsed, the DeepSeek go-lane evidence +// matrix could no longer be re-run. Zen's free tier keeps a responses-native +// model (muse-spark-1.3-contributor-free) reachable at no cost, so the +// responses 直通 path through 9router stays continuously testable: transport +// selection, executor free-tier fingerprint, SSE lifecycle and input replay. +// DeepSeek-specific semantics (thinking pass-back) still require the go lane +// or the official API and stay in the opencode-go / deepseek real suites. +// +// The free tier only accepts requests carrying the client fingerprint +// (opencode UA + session header + stream:true + bash/glob/grep/read quartet); +// the opencode-zen executor injects all of it, which is part of what this +// smoke verifies. +// +// OPENCODE_ZEN_KEY=oc_sk... RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-zen-free-responses.real.test.js +// +// (OPENCODE_ZEN_KEY overrides credential lookup; otherwise an opencode-zen +// connection must be configured in 9router.) +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const ENV_KEY = process.env.OPENCODE_ZEN_KEY || ""; +const PROVIDER = "opencode-zen"; +const MODEL = "muse-spark-1.3-contributor-free"; +const TIMEOUT_MS = 120000; + +async function prepare() { + if (ENV_KEY) return { accessToken: ENV_KEY }; + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +async function runResponses(body, creds) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${MODEL}` }, + modelInfo: { provider: PROVIDER, model: MODEL }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "openai-responses", + }); + if (!result.success) { + return { + ok: false, + status: Number(result.status) || "n/a", + raw: String(result.error || ""), + }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +function sseEvents(raw) { + const events = []; + for (const chunk of raw.split("\n\n")) { + const line = chunk.split("\n").find((l) => l.startsWith("data: ")); + if (!line) continue; + try { + events.push(JSON.parse(line.slice(6))); + } catch { + /* skip malformed */ + } + } + return events; +} + +function userInput(text) { + return { + type: "message", + role: "user", + content: [{ type: "input_text", text }], + }; +} + +const WEATHER_TOOL = { + type: "function", + name: "get_weather", + description: "Get weather for a city", + parameters: { + type: "object", + properties: { city: { type: "string" } }, + required: ["city"], + }, +}; + +describe.skipIf(!RUN_REAL)(`REAL zen free responses 直通 (${MODEL})`, () => { + it( + "streams a full Responses SSE lifecycle through the executor fingerprint", + async () => { + const creds = await prepare(); + if (!creds) return; + + const res = await runResponses( + { + input: [userInput("Say OK and nothing else.")], + // 256 headroom: the default reasoning effort burns most of a small + // cap and the stream ends response.incomplete, not completed. + max_output_tokens: 256, + stream: true, + }, + creds, + ); + expect(res.ok).toBe(true); + + const types = sseEvents(res.raw).map((e) => e.type); + expect(types).toContain("response.created"); + expect(types).toContain("response.completed"); + }, + TIMEOUT_MS, + ); + + it( + "accepts custom tools alongside the fingerprint quartet", + async () => { + const creds = await prepare(); + if (!creds) return; + + const res = await runResponses( + { + input: [userInput("What is the weather in Paris? Use the tool.")], + max_output_tokens: 256, + stream: true, + tools: [WEATHER_TOOL], + }, + creds, + ); + expect(res.ok).toBe(true); + expect(sseEvents(res.raw).map((e) => e.type)).toContain( + "response.completed", + ); + }, + TIMEOUT_MS, + ); + + it( + "replays turn-1 output items (incl. reasoning) as input for turn 2", + async () => { + const creds = await prepare(); + if (!creds) return; + + const turn1 = await runResponses( + { + input: [ + userInput("Think briefly, then reply with the single word OK."), + ], + max_output_tokens: 256, + stream: true, + }, + creds, + ); + expect(turn1.ok).toBe(true); + const completed = sseEvents(turn1.raw).find( + (e) => e.type === "response.completed", + ); + const output = completed?.response?.output; + expect(Array.isArray(output)).toBe(true); + expect(output.length).toBeGreaterThan(0); + + // Replay every output item verbatim (reasoning items included) — the + // input-array pass-back path responses clients depend on. + const turn2 = await runResponses( + { + input: [ + userInput("Think briefly, then reply with the single word OK."), + ...output, + userInput("Now reply with the single word DONE."), + ], + max_output_tokens: 256, + stream: true, + }, + creds, + ); + expect(turn2.ok).toBe(true); + expect(sseEvents(turn2.raw).map((e) => e.type)).toContain( + "response.completed", + ); + }, + TIMEOUT_MS, + ); +}); diff --git a/tests/unit/opencode-go-transport-routing.test.js b/tests/unit/opencode-go-transport-routing.test.js new file mode 100644 index 00000000..6a4d47be --- /dev/null +++ b/tests/unit/opencode-go-transport-routing.test.js @@ -0,0 +1,175 @@ +// Offline routing matrix for opencode-go models. +// +// Drives the REAL handleChatCore guard + targetFormat resolution (open-sse/handlers/chatCore.js:86-94) +// end-to-end; only the executor's HTTP response is mocked. The assertion target is +// credentials.runtimeTransport — the exact field DefaultExecutor.buildUrl/buildHeaders read +// (open-sse/executors/default.js:106,150) to pick the endpoint and auth scheme — so a wrong +// guard decision shows up as the wrong baseUrl here, same as it would on the wire. +// +// Cells: +// - deepseek × {openai, claude, openai-responses} × {bare, (max)} — the endpoint matrix +// under dispute in #3278/#3332. Bare and suffixed cells must resolve identically. +// - glm/kimi (chat-only) + (max) — regression cells: with the thinking suffix, the guard +// is bypassed on master (suffix isn't stripped before the registry lookup) and these get +// routed to /messages, which the upstream does not serve for them. +// - minimax + (max) + claude — suffix must NOT block a genuinely declared format. +import { describe, it, expect, vi, beforeEach } from "vitest"; + +const { executeMock } = vi.hoisted(() => ({ + executeMock: vi.fn(), +})); + +vi.mock("../../open-sse/executors/index.js", () => ({ + getExecutor: () => ({ + noAuth: true, + execute: executeMock, + }), +})); + +vi.mock("../../open-sse/utils/requestLogger.js", () => ({ + createRequestLogger: async () => ({ + logClientRawRequest: vi.fn(), + logRawRequest: vi.fn(), + logTargetRequest: vi.fn(), + logProviderResponse: vi.fn(), + logConvertedResponse: vi.fn(), + logError: vi.fn(), + }), +})); + +vi.mock("../../open-sse/utils/stream.js", () => ({ + COLORS: { red: "", reset: "" }, + createPassthroughStreamWithLogger: vi.fn(() => new TransformStream()), +})); + +vi.mock("uuid", () => ({ + v4: () => "00000000-0000-4000-8000-000000000000", +})); + +vi.mock("@/lib/usageDb.js", () => ({ + trackPendingRequest: vi.fn(), + appendRequestLog: vi.fn(async () => {}), + saveRequestDetail: vi.fn(async () => {}), + saveRequestUsage: vi.fn(async () => {}), +})); + +// image.js imports Agent from "undici" (not installed in some dev envs); the +// prefetch path is irrelevant to routing assertions. +vi.mock("../../open-sse/translator/concerns/image.js", () => ({ + encodeDataUri: (mimeType, base64) => `data:${mimeType};base64,${base64}`, + parseDataUri: (url) => { + const m = /^data:([^;]+);base64,(.*)$/.exec(url); + return m ? { mimeType: m[1], base64: m[2] } : null; + }, + fetchImageAsBase64: async () => null, +})); + +const { handleChatCore } = await import("../../open-sse/handlers/chatCore.js"); + +const BASE = "https://opencode.ai/zen/go/v1"; +const ENDPOINTS = { + openai: `${BASE}/chat/completions`, + claude: `${BASE}/messages`, + "openai-responses": `${BASE}/responses`, +}; + +// Minimal non-stream provider JSON per target format — the mocked executor's response. +const RESPONSE_BY_FORMAT = { + claude: { + id: "msg_1", type: "message", role: "assistant", model: "test", + content: [{ type: "text", text: "ok" }], + stop_reason: "end_turn", stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 1 }, + }, + openai: { + id: "chatcmpl-1", object: "chat.completion", model: "test", + choices: [{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }, + "openai-responses": { + id: "resp_1", object: "response", created_at: 0, status: "completed", model: "test", + output: [{ + type: "message", id: "msg_1", role: "assistant", status: "completed", + content: [{ type: "output_text", text: "ok", annotations: [] }], + }], + }, +}; + +async function route(model, sourceFormat) { + executeMock.mockResolvedValueOnce({ + response: new Response(JSON.stringify(RESPONSE_BY_FORMAT[sourceFormat === "openai" ? "openai" : sourceFormat] || RESPONSE_BY_FORMAT.openai), { + status: 200, + headers: { "content-type": "application/json" }, + }), + url: ENDPOINTS[sourceFormat] || ENDPOINTS.openai, + headers: {}, + transformedBody: null, + }); + + const credentials = { apiKey: "test-key", providerSpecificData: {} }; + const result = await handleChatCore({ + body: { + model: `opencode-go/${model}`, + stream: false, + max_tokens: 16, + messages: [{ role: "user", content: "hi" }], + }, + modelInfo: { provider: "opencode-go", model }, + credentials, + connectionId: "ocg-route-test", + sourceFormatOverride: sourceFormat, + log: { debug: vi.fn(), info: vi.fn(), warn: vi.fn() }, + }); + + const { credentials: creds } = executeMock.mock.calls.at(-1)[0]; + return { result, runtimeTransport: creds.runtimeTransport ?? null }; +} + +describe("opencode-go DeepSeek endpoint matrix (via real handleChatCore)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + for (const model of ["deepseek-v4-flash", "deepseek-v4-pro"]) { + for (const suffix of ["", "(max)"]) { + const id = model + suffix; + for (const [fmt, expectedUrl] of Object.entries(ENDPOINTS)) { + it(`routes ${id} + ${fmt}-format client to ${expectedUrl}`, async () => { + const { result, runtimeTransport } = await route(id, fmt); + expect(result.success).toBe(true); + expect(runtimeTransport?.baseUrl).toBe(expectedUrl); + }); + } + } + } +}); + +describe("opencode-go thinking-suffix guard (regression)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it("does NOT route chat-only glm-5.2(max) to /messages on a claude-format request", async () => { + const { result, runtimeTransport } = await route("glm-5.2(max)", "claude"); + expect(result.success).toBe(true); + expect(runtimeTransport).toBeNull(); // guard must block; falls back to chat/completions + }); + + it("does NOT route chat-only kimi-k2.6(max) to /responses on a responses-format request", async () => { + const { result, runtimeTransport } = await route("kimi-k2.6(max)", "openai-responses"); + expect(result.success).toBe(true); + expect(runtimeTransport).toBeNull(); + }); + + it("still routes minimax-m3(max) + claude-format client to /messages", async () => { + const { result, runtimeTransport } = await route("minimax-m3(max)", "claude"); + expect(result.success).toBe(true); + expect(runtimeTransport?.baseUrl).toBe(ENDPOINTS.claude); + }); + + it("does NOT route minimax-m3(max) (no responses support) to /responses", async () => { + const { result, runtimeTransport } = await route("minimax-m3(max)", "openai-responses"); + expect(result.success).toBe(true); + expect(runtimeTransport).toBeNull(); + }); +}); diff --git a/tests/unit/tool-deduper.test.js b/tests/unit/tool-deduper.test.js new file mode 100644 index 00000000..58714b00 --- /dev/null +++ b/tests/unit/tool-deduper.test.js @@ -0,0 +1,100 @@ +import { describe, it, expect } from "vitest"; +import { dedupeTools } from "../../open-sse/utils/toolDeduper.js"; + +const BASH = (name = "Bash", desc = "Run a shell command") => ({ + name, + description: desc, + input_schema: { type: "object", properties: { command: { type: "string" } }, required: ["command"] }, +}); + +const FUNC_SHAPE = (name = "Bash") => ({ + type: "function", + function: { name, description: "Run a shell command", parameters: { type: "object", properties: { command: { type: "string" } } } }, +}); + +const MCP_EXA = { name: "mcp__exa__web_search_exa", description: "search" }; + +describe("toolDeduper — MCP-equivalent built-in rules (existing behavior)", () => { + it("claude client + Exa MCP → drops built-in WebSearch/WebFetch", () => { + const { tools, stripped } = dedupeTools( + [MCP_EXA, { name: "WebSearch", description: "web" }, { name: "WebFetch", description: "web" }, BASH()], + { clientTool: "claude" } + ); + expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]); + expect(stripped.sort()).toEqual(["WebFetch", "WebSearch"]); + }); + + it("non-claude client → MCP built-in rules do NOT run (behavior preserved)", () => { + const { tools, stripped } = dedupeTools( + [MCP_EXA, { name: "WebSearch", description: "web" }], + { clientTool: "codex" } + ); + expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "WebSearch"]); + expect(stripped).toEqual([]); + }); + + it("legacy call without opts still applies MCP rules for claude callers (back-compat shape)", () => { + // chatCore always passes opts now, but old direct callers keep prior behavior: + // without clientTool, MCP rules stay dormant (they were claude-gated anyway). + const { tools, stripped } = dedupeTools([MCP_EXA, { name: "WebSearch", description: "web" }]); + expect(tools).toHaveLength(2); + expect(stripped).toEqual([]); + }); +}); + +describe("toolDeduper — DeepSeek same-name dedup (new)", () => { + it("deepseek model + duplicate tool names → keeps first definition", () => { + const first = BASH(); + const dup = BASH("Bash", "duplicate description"); + const { tools, stripped } = dedupeTools([first, dup], { model: "deepseek-v4-flash" }); + expect(tools).toEqual([first]); // first wins, including its description + expect(stripped).toEqual(["Bash"]); + }); + + it("deepseek model + (max) thinking suffix → still dedups (suffix stripped before match)", () => { + const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "dup")], { model: "deepseek-v4-flash(max)" }); + expect(tools).toHaveLength(1); + expect(stripped).toEqual(["Bash"]); + }); + + it("deepseek + 3 same-name tools → keeps first, drops both duplicates", () => { + const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "d1"), BASH("Bash", "d2")], { model: "deepseek-v4-pro" }); + expect(tools).toHaveLength(1); + expect(stripped).toEqual(["Bash", "Bash"]); + }); + + it("deepseek + OpenAI function-shape tools → dedups by function.name", () => { + const { tools } = dedupeTools([FUNC_SHAPE("Bash"), FUNC_SHAPE("Bash")], { model: "deepseek-v4-flash" }); + expect(tools).toHaveLength(1); + expect(tools[0].function.name).toBe("Bash"); + }); + + it("non-DeepSeek model + duplicate tool names → untouched (GLM/MiniMax/Kimi accept them)", () => { + const tools = [BASH(), BASH("Bash", "dup")]; + const { tools: out, stripped } = dedupeTools(tools, { model: "glm-5.2" }); + expect(out).toBe(tools); + expect(stripped).toEqual([]); + }); + + it("no model declared → same-name dedup does NOT run (safe default)", () => { + const tools = [BASH(), BASH("Bash", "dup")]; + const { tools: out, stripped } = dedupeTools(tools, {}); + expect(out).toBe(tools); + expect(stripped).toEqual([]); + }); + + it("deepseek + distinct names → nothing stripped", () => { + const { tools, stripped } = dedupeTools([BASH("Bash"), BASH("ReadFile")], { model: "deepseek-v4-flash" }); + expect(tools).toHaveLength(2); + expect(stripped).toEqual([]); + }); + + it("claude client + deepseek + MCP trigger → both rules apply (union stripped)", () => { + const { tools, stripped } = dedupeTools( + [MCP_EXA, { name: "WebSearch", description: "web" }, BASH(), BASH("Bash", "dup")], + { clientTool: "claude", model: "deepseek-v4-flash" } + ); + expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]); + expect(stripped.sort()).toEqual(["Bash", "WebSearch"]); + }); +}); From 08b21fea065e7bb4421404c5a8d0be9748f5c016 Mon Sep 17 00:00:00 2001 From: KiMelody Date: Mon, 28 Sep 2026 12:43:52 +0700 Subject: [PATCH 07/41] fix(claude): inject unsigned thinking placeholders for opencode-go DeepSeek /messages (#4436) DeepSeek models behind OpenCode Go's /messages transport carry the same thinking pass-back constraint as the official DeepSeek provider - 400 'The content[].thinking in the thinking mode must be passed back'. prepareClaudeRequest now gates model-based: opencode-go + isDeepSeekModel(body.model) reuses the official semantics - keep existing thinking verbatim, inject an unsigned placeholder on tool_use turns missing one while thinking is enabled. isDeepSeekModel moves to providers/models/helpers.js (shared with toolDeduper). --- open-sse/providers/models/helpers.js | 9 ++ open-sse/translator/formats/claude.js | 25 +++-- open-sse/utils/toolDeduper.js | 7 +- ...ode-go-deepseek-thinking-injection.test.js | 102 ++++++++++++++++++ 4 files changed, 130 insertions(+), 13 deletions(-) create mode 100644 tests/unit/opencode-go-deepseek-thinking-injection.test.js diff --git a/open-sse/providers/models/helpers.js b/open-sse/providers/models/helpers.js index cf8076bd..1a71bc12 100644 --- a/open-sse/providers/models/helpers.js +++ b/open-sse/providers/models/helpers.js @@ -28,6 +28,15 @@ export function isMuseSparkModel(modelId) { return /^muse[-_]?spark(?:$|[-_:.\s])/i.test(base); } +// "model(level)" is a 9router thinking override; strip before matching. +// Accepts both bare ids ("deepseek-v4-pro(max)") and provider-prefixed ones. +export function isDeepSeekModel(modelId) { + if (!modelId || typeof modelId !== "string") return false; + const clean = modelId.replace(/\([^()]+\)\s*$/, "").trim(); + const base = clean.includes("/") ? clean.split("/").pop() : clean; + return /^deepseek-/i.test(base); +} + // Endpoint families for OpenCode models outside the curated registry (modelsFetcher / // passthrough ids) — regex keeps auto-fetched models on the right endpoint: // /responses (gpt/grok/muse-spark), /messages (minimax/qwen), /chat/completions (rest). diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index 14c9fc10..e52bef9d 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -7,6 +7,7 @@ import { resolveSessionId } from "../../utils/sessionManager.js"; import { isValidClaudeSignature } from "../../utils/claudeSignature.js"; import { PROVIDERS } from "../../providers/index.js"; import { getCapabilitiesForModel } from "../../providers/capabilities.js"; +import { isDeepSeekModel } from "../../providers/models/helpers.js"; import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js"; const CACHE_CONTROL_5M = { type: "ephemeral" }; @@ -167,7 +168,7 @@ function handlesThinkingBlocks(provider) { return provider === "claude" || provider?.startsWith("anthropic-compatible") || provider === "deepseek"; } -function buildThinkingPlaceholder(provider) { +function buildThinkingPlaceholder(provider, unsigned = false) { const block = { type: CLAUDE_BLOCK.THINKING, thinking: ".", @@ -175,7 +176,9 @@ function buildThinkingPlaceholder(provider) { // DeepSeek's Anthropic-compatible endpoint requires a thinking block in // thinking mode, but it does not need Anthropic's signed-thinking fallback. - if (provider !== "deepseek") { + // The same applies to DeepSeek models served through other providers' + // Claude transports (opencode-go /messages). + if (provider !== "deepseek" && !unsigned) { block.signature = DEFAULT_THINKING_CLAUDE_SIGNATURE; } @@ -512,6 +515,14 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne const lastMessageIsUser = lastMessage?.role === "user"; const thinkingEnabled = body.thinking?.type === "enabled" && lastMessageIsUser; + // DeepSeek models also arrive behind OpenCode Go's /messages transport. + // They carry the same thinking pass-back constraint as the official + // DeepSeek provider (verified live 2026-08-15, PR #3332 discussion), so + // they get the identical keep/placeholder handling below. + const deepSeekServed = + provider === "deepseek" || + (provider === "opencode-go" && isDeepSeekModel(body?.model)); + // Pass 2 (reverse): add cache_control to last assistant + handle thinking for Anthropic let lastAssistantProcessed = false; for (let i = filtered.length - 1; i >= 0; i--) { @@ -532,15 +543,15 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne } // Handle thinking blocks for Anthropic-compatible endpoints. - if (handlesThinkingBlocks(provider)) { + if (handlesThinkingBlocks(provider) || deepSeekServed) { let hasToolUse = false; let hasKeptThinking = false; // Claude native: preserve valid signatures, drop invalid blocks. // anthropic-compatible: replace with default (safe fallback for lenient upstreams). - // DeepSeek: keep existing thinking as-is; add an unsigned placeholder only if missing. + // DeepSeek (official + opencode-go models): keep existing thinking as-is; + // add an unsigned placeholder only if missing. const isClaudeNative = provider === "claude"; - const isDeepSeek = provider === "deepseek"; const kept = []; for (const block of msg.content) { const isThinking = block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING; @@ -550,7 +561,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne hasKeptThinking = true; kept.push(block); } - } else if (isDeepSeek) { + } else if (deepSeekServed) { hasKeptThinking = true; kept.push(block); } else { @@ -567,7 +578,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // Add thinking block if thinking enabled + has tool_use but no thinking if (thinkingEnabled && !hasKeptThinking && hasToolUse) { - msg.content.unshift(buildThinkingPlaceholder(provider)); + msg.content.unshift(buildThinkingPlaceholder(provider, deepSeekServed)); } } } diff --git a/open-sse/utils/toolDeduper.js b/open-sse/utils/toolDeduper.js index ee6b5aa6..e17d37e4 100644 --- a/open-sse/utils/toolDeduper.js +++ b/open-sse/utils/toolDeduper.js @@ -7,6 +7,7 @@ * gateway; GLM/MiniMax/Kimi upstreams accept duplicates). First definition wins, * tool_choice and message-history references are by name/id so nothing breaks. */ +import { isDeepSeekModel } from "../providers/models/helpers.js"; const DEDUP_RULES = [ { @@ -35,12 +36,6 @@ function matches(name, pattern) { return pattern instanceof RegExp ? pattern.test(name) : false; } -// "model(level)" is a 9router thinking override; strip before matching. -function isDeepSeekModel(model) { - if (typeof model !== "string") return false; - return /^deepseek-/.test(model.replace(/\([^()]+\)\s*$/, "").trim()); -} - /** * @param {Array} tools - translated tools array * @param {Object} [opts] diff --git a/tests/unit/opencode-go-deepseek-thinking-injection.test.js b/tests/unit/opencode-go-deepseek-thinking-injection.test.js new file mode 100644 index 00000000..0b79815e --- /dev/null +++ b/tests/unit/opencode-go-deepseek-thinking-injection.test.js @@ -0,0 +1,102 @@ +/** + * opencode-go DeepSeek models on the claude→claude /messages passthrough need + * the same thinking-block handling as the official deepseek provider (#3332): + * keep existing thinking blocks verbatim, and inject an UNSIGNED placeholder + * on tool_use turns that carry none while thinking is enabled — upstream 400s + * with "The content[].thinking in the thinking mode must be passed back to the + * API" otherwise. The gate is model-based because opencode-go also serves + * non-DeepSeek models over /messages (minimax, qwen) that must stay untouched. + * + * Signed placeholders were also accepted live (2026-08-15, opencode.go /messages), + * but unsigned mirrors the official deepseek provider behavior exactly. + */ +import { describe, it, expect } from "vitest"; +import { prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; + +function makeBody(model) { + return { + model, + max_tokens: 2048, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "toolu_1", + name: "get_weather", + input: { city: "Paris" }, + }, + ], + }, + { + role: "user", + content: [ + { type: "tool_result", tool_use_id: "toolu_1", content: "18C" }, + ], + }, + ], + }; +} + +function firstBlock(body) { + return body.messages[0].content[0]; +} + +describe("prepareClaudeRequest — opencode-go DeepSeek thinking pass-back", () => { + it("injects an unsigned thinking placeholder on tool_use turns missing one", () => { + const out = prepareClaudeRequest( + makeBody("opencode-go/deepseek-v4-pro(max)"), + "opencode-go", + ); + + const block = firstBlock(out); + expect(block.type).toBe("thinking"); + expect(block.signature).toBeUndefined(); + expect(out.messages[0].content).toHaveLength(2); // placeholder + tool_use + }); + + it("keeps an existing thinking block verbatim (no re-sign, no duplicate)", () => { + const body = makeBody("opencode-go/deepseek-v4-flash"); + const realThinking = { + type: "thinking", + thinking: "actual reasoning", + signature: "sig_from_upstream", + }; + body.messages[0].content.unshift(realThinking); + + const out = prepareClaudeRequest(body, "opencode-go"); + + const thinking = out.messages[0].content.filter( + (b) => b.type === "thinking", + ); + expect(thinking).toHaveLength(1); + expect(thinking[0]).toEqual(realThinking); + }); + + it("leaves non-DeepSeek opencode-go models untouched (minimax rides /messages too)", () => { + const out = prepareClaudeRequest( + makeBody("opencode-go/minimax-m3"), + "opencode-go", + ); + + expect(out.messages[0].content).toHaveLength(1); // tool_use only + expect(firstBlock(out).type).toBe("tool_use"); + }); + + it("official deepseek provider keeps injecting unsigned placeholders (regression)", () => { + const out = prepareClaudeRequest(makeBody("deepseek-v4-pro"), "deepseek"); + + const block = firstBlock(out); + expect(block.type).toBe("thinking"); + expect(block.signature).toBeUndefined(); + }); + + it.todo( + "inject a reasoning placeholder into Responses `input` items for DeepSeek models — " + + "injectReasoningContent only rewrites body.messages, so responses-format clients " + + "replaying reasoning-bearing sessions on the openai-responses transport are " + + "uncovered (#3332 restore gate: Codex-shaped payload must stay 200)", + ); +}); From 06eda8b081cd94bb512658df231efd124a9b66f7 Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 12:52:37 +0700 Subject: [PATCH 08/41] fix(cli-tools): replace sk_9router placeholder with first active dashboard API key --- .../api/cli-tools/codewhale-settings/route.js | 3 +- .../api/cli-tools/copilot-settings/route.js | 3 +- src/app/api/cli-tools/crush-settings/route.js | 3 +- .../cli-tools/deepseek-tui-settings/route.js | 3 +- src/app/api/cli-tools/forge-settings/route.js | 3 +- .../cli-tools/grok-build-settings/route.js | 3 +- src/app/api/cli-tools/omp-settings/route.js | 9 +- .../api/cli-tools/opencode-settings/route.js | 3 +- src/app/api/cli-tools/pi-settings/route.js | 6 +- src/app/api/cli-tools/resolveApiKey.js | 36 +++++++ src/app/api/cli-tools/smelt-settings/route.js | 3 +- .../unit/cli-tools-api-key-resolution.test.js | 97 +++++++++++++++++++ 12 files changed, 159 insertions(+), 13 deletions(-) create mode 100644 src/app/api/cli-tools/resolveApiKey.js create mode 100644 tests/unit/cli-tools-api-key-resolution.test.js diff --git a/src/app/api/cli-tools/codewhale-settings/route.js b/src/app/api/cli-tools/codewhale-settings/route.js index f561dc06..0d85708e 100644 --- a/src/app/api/cli-tools/codewhale-settings/route.js +++ b/src/app/api/cli-tools/codewhale-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -97,7 +98,7 @@ export async function POST(request) { existing.openai = { base_url: normalizedBaseUrl, - api_key: apiKey || "sk_9router", + api_key: await resolveCliApiKey(apiKey), model: model || "provider/model-id", }; diff --git a/src/app/api/cli-tools/copilot-settings/route.js b/src/app/api/cli-tools/copilot-settings/route.js index 3c0bd669..74303960 100644 --- a/src/app/api/cli-tools/copilot-settings/route.js +++ b/src/app/api/cli-tools/copilot-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -81,7 +82,7 @@ export async function POST(request) { } catch { /* No existing config */ } const endpointUrl = `${baseUrl}/chat/completions#models.ai.azure.com`; - const keyToUse = apiKey || "sk_9router"; + const keyToUse = await resolveCliApiKey(apiKey); const newEntry = { name: "9Router", diff --git a/src/app/api/cli-tools/crush-settings/route.js b/src/app/api/cli-tools/crush-settings/route.js index dd96d948..72b1141e 100644 --- a/src/app/api/cli-tools/crush-settings/route.js +++ b/src/app/api/cli-tools/crush-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -108,7 +109,7 @@ export async function POST(request) { existing.providers["9router"] = { type: "openai-compat", base_url: normalizedBaseUrl, - api_key: apiKey || "sk_9router", + api_key: await resolveCliApiKey(apiKey), models: [ { id: modelId, diff --git a/src/app/api/cli-tools/deepseek-tui-settings/route.js b/src/app/api/cli-tools/deepseek-tui-settings/route.js index 0edf74da..9fcf0edf 100644 --- a/src/app/api/cli-tools/deepseek-tui-settings/route.js +++ b/src/app/api/cli-tools/deepseek-tui-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import { exec } from "child_process"; import { promisify } from "util"; import fs from "fs/promises"; @@ -132,7 +133,7 @@ export async function POST(request) { const dir = getDeepSeekDir(); await fs.mkdir(dir, { recursive: true }); - const newConfig = build9RouterConfig(baseUrl, apiKey || "sk_9router", model); + const newConfig = build9RouterConfig(baseUrl, await resolveCliApiKey(apiKey), model); await fs.writeFile(getDeepSeekConfigPath(), newConfig); return NextResponse.json({ diff --git a/src/app/api/cli-tools/forge-settings/route.js b/src/app/api/cli-tools/forge-settings/route.js index 2f7c45dc..ca35412e 100644 --- a/src/app/api/cli-tools/forge-settings/route.js +++ b/src/app/api/cli-tools/forge-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -96,7 +97,7 @@ export async function POST(request) { const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`; existing.openai = { - api_key: apiKey || "sk_9router", + api_key: await resolveCliApiKey(apiKey), base_url: normalizedBaseUrl, model: model || "provider/model-id", }; diff --git a/src/app/api/cli-tools/grok-build-settings/route.js b/src/app/api/cli-tools/grok-build-settings/route.js index 02299a3b..4d7da8c1 100644 --- a/src/app/api/cli-tools/grok-build-settings/route.js +++ b/src/app/api/cli-tools/grok-build-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import { exec } from "child_process"; import { promisify } from "util"; import fs from "fs/promises"; @@ -108,7 +109,7 @@ export async function POST(request) { const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`; const toml = applyGrokBuildConfig(await readConfigToml(), { baseUrl: normalizedBaseUrl, - apiKey: apiKey || "sk_9router", + apiKey: await resolveCliApiKey(apiKey), model: selectedModel, contextWindow: normalizeContextWindow(contextWindow, selectedModel), subagentModels: normalizeSubagentModels(subagentModels), diff --git a/src/app/api/cli-tools/omp-settings/route.js b/src/app/api/cli-tools/omp-settings/route.js index 52df53e1..ff7a4350 100644 --- a/src/app/api/cli-tools/omp-settings/route.js +++ b/src/app/api/cli-tools/omp-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -51,7 +52,7 @@ const has9RouterInYml = (content) => { // Build standard 9Router provider block for models.yml const buildOmpProviderYaml = (baseUrl, apiKey) => { const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`; - const key = apiKey || "sk_9router"; + const key = apiKey || ""; return ` ${PROVIDER_ID}: baseUrl: ${normalizedBaseUrl} apiKey: ${key} @@ -100,10 +101,12 @@ export async function POST(request) { return NextResponse.json({ error: { message: "baseUrl is required" } }, { status: 400 }); } + const resolvedKey = await resolveCliApiKey(apiKey); + await fs.mkdir(getOmpDir(), { recursive: true }); let ymlContent = await readModelsYml(); - const providerBlock = buildOmpProviderYaml(baseUrl, apiKey); + const providerBlock = buildOmpProviderYaml(baseUrl, resolvedKey); // Remove existing 9router provider if present const regex = new RegExp(`\\s*${PROVIDER_ID}:[\\s\\S]*?(?=\\n\\s*\\w+:|$)`, "g"); @@ -137,7 +140,7 @@ export async function POST(request) { ).run( PROVIDER_ID, "api_key", - JSON.stringify({ apiKey: apiKey || "sk_9router", baseUrl }), + JSON.stringify({ apiKey: resolvedKey, baseUrl }), Math.floor(Date.now() / 1000), Math.floor(Date.now() / 1000) ); diff --git a/src/app/api/cli-tools/opencode-settings/route.js b/src/app/api/cli-tools/opencode-settings/route.js index 03819c66..b77c0263 100644 --- a/src/app/api/cli-tools/opencode-settings/route.js +++ b/src/app/api/cli-tools/opencode-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import { exec } from "child_process"; import { promisify } from "util"; import fs from "fs/promises"; @@ -112,7 +113,7 @@ export async function POST(request) { } catch { /* No existing config */ } const normalizedBaseUrl = baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`; - const keyToUse = apiKey || "sk_9router"; + const keyToUse = await resolveCliApiKey(apiKey); const effectiveSubagentModel = subagentModel || modelsArray[0]; // Ensure provider object diff --git a/src/app/api/cli-tools/pi-settings/route.js b/src/app/api/cli-tools/pi-settings/route.js index ee5c507b..812c5b0f 100644 --- a/src/app/api/cli-tools/pi-settings/route.js +++ b/src/app/api/cli-tools/pi-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -145,9 +146,10 @@ export async function POST(request) { } existing.providers["9router"] = { + ...existingProvider, baseUrl: normalizedBaseUrl, - apiKey: apiKey || "sk_9router", - api: "openai-completions", + apiKey: apiKey || existingProvider.apiKey || await resolveCliApiKey(null), + api: existingProvider.api || "openai-completions", models: modelList, }; diff --git a/src/app/api/cli-tools/resolveApiKey.js b/src/app/api/cli-tools/resolveApiKey.js new file mode 100644 index 00000000..1567f3ce --- /dev/null +++ b/src/app/api/cli-tools/resolveApiKey.js @@ -0,0 +1,36 @@ +/** + * Resolves the API key to write into a CLI tool config. + * + * CLI tool cards send an empty string when no key is explicitly selected + * (e.g. the existing config already has a provider block but the frontend + * can't read the stored Authorization header back). The routes previously + * fell back to the literal placeholder "sk_9router", which causes 401 + * "Invalid API key" for any deployment with requireApiKey=true (#4399). + * + * Resolution order: + * 1. The key supplied by the caller (non-empty string). + * 2. The first active key in the dashboard's apiKeys table. + * 3. Empty string — the route writes no Authorization header value, + * which is fine for requireApiKey=false deployments. + * + * The placeholder "sk_9router" is NEVER written; it was never a real key. + */ + +import { getApiKeys } from "@/lib/db"; + +/** + * @param {string|null|undefined} callerKey Key sent by the frontend. + * @returns {Promise} + */ +export async function resolveCliApiKey(callerKey) { + if (callerKey && callerKey.trim() && callerKey.trim() !== "sk_9router") { + return callerKey.trim(); + } + try { + const keys = await getApiKeys(); + const active = keys.find((k) => k.isActive); + return active?.key || ""; + } catch { + return ""; + } +} \ No newline at end of file diff --git a/src/app/api/cli-tools/smelt-settings/route.js b/src/app/api/cli-tools/smelt-settings/route.js index 6592d0d9..bc7fa143 100644 --- a/src/app/api/cli-tools/smelt-settings/route.js +++ b/src/app/api/cli-tools/smelt-settings/route.js @@ -1,6 +1,7 @@ "use server"; import { NextResponse } from "next/server"; +import { resolveCliApiKey } from "../resolveApiKey.js"; import fs from "fs/promises"; import path from "path"; import os from "os"; @@ -96,7 +97,7 @@ export async function POST(request) { const updated = { ...existing, baseUrl: normalizedBaseUrl, - apiKey: apiKey || "sk_9router", + apiKey: await resolveCliApiKey(apiKey), model: model || existing.model || "provider/model-id", _managedBy: "9router", }; diff --git a/tests/unit/cli-tools-api-key-resolution.test.js b/tests/unit/cli-tools-api-key-resolution.test.js new file mode 100644 index 00000000..44f3ebe3 --- /dev/null +++ b/tests/unit/cli-tools-api-key-resolution.test.js @@ -0,0 +1,97 @@ +/** + * Tests for #4399 — CLI Tools Apply writes placeholder "sk_9router" key. + * + * When requireApiKey=true, the written "sk_9router" caused 401 on every + * CLI tool request. The fix: replace the literal fallback with + * resolveCliApiKey(), which reads the first active key from the DB + * (or returns "" if none exist — never the placeholder). + */ + +import { describe, it, expect } from "vitest"; +import fs from "fs"; +import path from "path"; + +// Test the resolveCliApiKey logic in isolation. +// The actual module reads from SQLite; we replicate the decision logic here. + +function resolveCliApiKeyLogic(callerKey, activeKeys = []) { + if (callerKey && callerKey.trim() && callerKey.trim() !== "sk_9router") { + return callerKey.trim(); + } + const active = activeKeys.find((k) => k.isActive); + return active?.key || ""; +} + +describe("resolveCliApiKey logic (#4399)", () => { + it("returns the caller key when non-empty and not the placeholder", () => { + expect(resolveCliApiKeyLogic("sk-real-key-123", [])).toBe("sk-real-key-123"); + }); + + it("falls back to first active DB key when caller key is empty", () => { + const keys = [{ key: "sk-db-key", isActive: true }]; + expect(resolveCliApiKeyLogic("", keys)).toBe("sk-db-key"); + }); + + it("falls back to first active DB key when caller key is null", () => { + const keys = [{ key: "sk-db-key", isActive: true }]; + expect(resolveCliApiKeyLogic(null, keys)).toBe("sk-db-key"); + }); + + it("falls back to first active DB key when caller key is undefined", () => { + const keys = [{ key: "sk-db-key", isActive: true }]; + expect(resolveCliApiKeyLogic(undefined, keys)).toBe("sk-db-key"); + }); + + it("falls back to first active DB key when caller is the placeholder itself", () => { + const keys = [{ key: "sk-db-key", isActive: true }]; + expect(resolveCliApiKeyLogic("sk_9router", keys)).toBe("sk-db-key"); + }); + + it("skips inactive keys and picks the first active one", () => { + const keys = [ + { key: "sk-inactive", isActive: false }, + { key: "sk-active", isActive: true }, + ]; + expect(resolveCliApiKeyLogic("", keys)).toBe("sk-active"); + }); + + it("returns empty string when no active DB key exists and caller is empty", () => { + const keys = [{ key: "sk-inactive", isActive: false }]; + expect(resolveCliApiKeyLogic("", keys)).toBe(""); + }); + + it("returns empty string when DB is empty and caller is empty", () => { + expect(resolveCliApiKeyLogic("", [])).toBe(""); + }); + + it("trims whitespace from caller key", () => { + expect(resolveCliApiKeyLogic(" sk-real ", [])).toBe("sk-real"); + }); + + it("never returns sk_9router", () => { + expect(resolveCliApiKeyLogic("sk_9router", [])).not.toBe("sk_9router"); + expect(resolveCliApiKeyLogic("", [])).not.toBe("sk_9router"); + }); +}); + +describe("resolveApiKey.js source checks (#4399)", () => { + const src = fs.readFileSync( + new URL("../../src/app/api/cli-tools/resolveApiKey.js", import.meta.url), + "utf-8" + ); + + it("resolveApiKey.js exports resolveCliApiKey", () => { + expect(src).toContain("resolveCliApiKey"); + }); + + it("resolveApiKey.js imports getApiKeys from DB", () => { + expect(src).toContain("getApiKeys"); + }); + + it("resolveApiKey.js does not return sk_9router as a fallback value", () => { + // The helper may reference "sk_9router" to guard against it, but must + // never use it as a return / fallback value (e.g. `return "sk_9router"`). + expect(src).not.toMatch(/return\s+"sk_9router"/); + expect(src).not.toMatch(/\|\|\s*"sk_9router"/); + }); +}); \ No newline at end of file From 7bf93178140c8792e3b351415f9b2f68fa07d102 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Mon, 28 Sep 2026 12:53:29 +0700 Subject: [PATCH 09/41] fix(codex): preserve hosted web search on GPT-6 Sol/Luna --- open-sse/executors/codex.js | 34 +++++++++-- tests/unit/codex-gpt6-lite.test.js | 96 ++++++++++++++++++++++++++++++ 2 files changed, 126 insertions(+), 4 deletions(-) diff --git a/open-sse/executors/codex.js b/open-sse/executors/codex.js index 8b8c5f79..e10e51eb 100644 --- a/open-sse/executors/codex.js +++ b/open-sse/executors/codex.js @@ -215,9 +215,9 @@ export class CodexExecutor extends BaseExecutor { * Override headers to add codex-specific identity headers. * transformRequest runs BEFORE buildHeaders, sets this._currentSessionId. */ - buildHeaders(credentials, stream = true, _url = null, model = null) { + buildHeaders(credentials, stream = true, _url = null, model = null, body = null) { const headers = super.buildHeaders(credentials, stream); - if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model))) { + if (isCodexResponsesLiteModel(model && getModelUpstreamId("cx", model)) && !body?.tools?.some?.(tool => tool?.type === "web_search")) { headers["x-openai-internal-codex-responses-lite"] = "true"; } headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default"; @@ -418,7 +418,33 @@ export class CodexExecutor extends BaseExecutor { const normalized = normalizeResponsesInput(body.input); if (normalized) body.input = normalized; const upstreamModel = getModelUpstreamId("cx", body.model || model); - const responsesLite = isCodexResponsesLiteModel(upstreamModel); + // Register hosted search before choosing transport; Lite cannot execute it. + const autoWebSearch = body._autoCodexWebSearch === true; + delete body._autoCodexWebSearch; + if (autoWebSearch && !body.tools?.some?.(tool => tool?.type === "web_search")) { + body.tools = [...(Array.isArray(body.tools) ? body.tools : []), { type: "web_search" }]; + } + // Hosted search cannot run from a Lite input prefix. When switching to + // regular Responses, move all prefixed tools without duplicating definitions. + let convertedLitePrefix = false; + if (isCodexResponsesLiteModel(upstreamModel) && Array.isArray(body.input) + && (body.tools?.some?.(tool => tool?.type === "web_search") + || body.input.some(item => item?.type === "additional_tools" && item.tools?.some?.(tool => tool?.type === "web_search")))) { + const tools = Array.isArray(body.tools) ? [...body.tools] : []; + const seen = new Set(tools.map(tool => `${tool?.type}:${tool?.name || tool?.function?.name || ""}`)); + for (const item of body.input) { + if (item?.type !== "additional_tools" || !Array.isArray(item.tools)) continue; + for (const tool of item.tools) { + const name = `${tool?.type}:${tool?.name || tool?.function?.name || ""}`; + if (!seen.has(name)) { tools.push(tool); seen.add(name); } + } + } + body.tools = tools; + convertedLitePrefix = body.input.some(item => item?.type === "additional_tools"); + body.input = body.input.filter(item => item?.type !== "additional_tools"); + } + const responsesLite = isCodexResponsesLiteModel(upstreamModel) + && !body.tools?.some?.(tool => tool?.type === "web_search"); // Ensure input is present and non-empty (Codex API rejects empty input) if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) { @@ -436,7 +462,7 @@ export class CodexExecutor extends BaseExecutor { body.stream = true; // If no instructions provided, inject default Codex instructions - if (!responsesLite && (!body.instructions || body.instructions.trim() === "")) { + if (!responsesLite && !convertedLitePrefix && (!body.instructions || body.instructions.trim() === "")) { body.instructions = CODEX_DEFAULT_INSTRUCTIONS; } diff --git a/tests/unit/codex-gpt6-lite.test.js b/tests/unit/codex-gpt6-lite.test.js index 035c8329..aaf8d3f2 100644 --- a/tests/unit/codex-gpt6-lite.test.js +++ b/tests/unit/codex-gpt6-lite.test.js @@ -44,6 +44,102 @@ describe("Codex GPT-6 Sol/Luna transport", () => { expect(body.reasoning).toEqual({ effort: "high", context: "all_turns" }); }); + it.each(["gpt-6-sol", "gpt-6-luna"])("keeps hosted web_search available on %s", async (model) => { + const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ + ok: true, status: 200, headers: new Map(), + }); + await new CodexExecutor().execute({ + model, + body: { + model, input: "Search the web", tools: [ + { type: "function", name: "run", parameters: { type: "object", properties: {} } }, + { type: "web_search" }, + ], tool_choice: "none", + }, + stream: true, credentials, + }); + const [, options] = fetchMock.mock.calls[0]; + const body = JSON.parse(options.body); + expect(options.headers["x-openai-internal-codex-responses-lite"]).toBeUndefined(); + expect(body.tools).toEqual([ + { type: "function", name: "run", parameters: { type: "object", properties: {} } }, + { type: "web_search" }, + ]); + expect(body.input.some(item => item.type === "additional_tools")).toBe(false); + expect(body.tool_choice).toBe("none"); + }); + + it.each(["gpt-6-sol", "gpt-6-luna"])("registers auto-injected hosted search on %s", async (model) => { + const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ + ok: true, status: 200, headers: new Map(), + }); + await new CodexExecutor().execute({ + model, + body: { model, input: "Search the web", _autoCodexWebSearch: true }, + stream: true, credentials, + }); + const [, options] = fetchMock.mock.calls[0]; + const body = JSON.parse(options.body); + expect(options.headers["x-openai-internal-codex-responses-lite"]).toBeUndefined(); + expect(body.tools).toEqual([{ type: "web_search" }]); + expect(body.input.some(item => item.type === "additional_tools")).toBe(false); + }); + + it("moves hosted search out of a native Lite prefix", async () => { + const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ + ok: true, status: 200, headers: new Map(), + }); + const tool = { type: "function", name: "run", parameters: { type: "object", properties: {} } }; + await new CodexExecutor().execute({ + model: "gpt-6-sol", + body: { model: "gpt-6-sol", input: [ + { type: "additional_tools", role: "developer", tools: [tool, { type: "web_search" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "search" }] }, + ], tools: null, tool_choice: "none" }, + stream: true, credentials, + }); + const [, options] = fetchMock.mock.calls[0]; + const body = JSON.parse(options.body); + expect(options.headers["x-openai-internal-codex-responses-lite"]).toBeUndefined(); + expect(body.tools).toEqual([tool, { type: "web_search" }]); + expect(body.input.some(item => item.type === "additional_tools")).toBe(false); + expect(body.tool_choice).toBe("none"); + }); + + it("preserves native Lite developer instructions when switching for hosted search", async () => { + const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ ok: true, status: 200, headers: new Map() }); + const instruction = { type: "message", role: "developer", content: [{ type: "input_text", text: "Only answer in French" }] }; + await new CodexExecutor().execute({ + model: "gpt-6-sol", body: { model: "gpt-6-sol", input: [ + { type: "additional_tools", role: "developer", tools: [{ type: "web_search" }] }, + instruction, + { type: "message", role: "user", content: [{ type: "input_text", text: "search" }] }, + ], instructions: "", tools: null }, stream: true, credentials, + }); + const [, options] = fetchMock.mock.calls[0]; + const body = JSON.parse(options.body); + expect(body.input).toContainEqual(instruction); + expect(body.instructions).toBe(""); + expect(body.tools).toEqual([{ type: "web_search" }]); + }); + + it("does not duplicate tools when hosted search appears in both tool locations", async () => { + const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ ok: true, status: 200, headers: new Map() }); + const tool = { type: "function", name: "run", parameters: { type: "object", properties: {} } }; + await new CodexExecutor().execute({ + model: "gpt-6-sol", body: { model: "gpt-6-sol", input: [ + { type: "additional_tools", role: "developer", tools: [tool, { type: "web_search" }] }, + { type: "additional_tools", role: "developer", tools: [tool] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "search" }] }, + ], tools: [tool, { type: "web_search" }] }, stream: true, credentials, + }); + const [, options] = fetchMock.mock.calls[0]; + const body = JSON.parse(options.body); + expect(options.headers["x-openai-internal-codex-responses-lite"]).toBeUndefined(); + expect(body.tools).toEqual([tool, { type: "web_search" }]); + expect(body.input.some(item => item.type === "additional_tools")).toBe(false); + }); + it("converts an ordinary Responses request to the Lite shape", () => { const executor = new CodexExecutor(); const tool = { type: "function", name: "run", parameters: { type: "object", properties: {} } }; From 1fd208e2b1c5d60ffa9baf09f62585fa4c14f125 Mon Sep 17 00:00:00 2001 From: Clayton Tavares Date: Mon, 28 Sep 2026 12:55:05 +0700 Subject: [PATCH 10/41] fix(usage): forward `recurring` for codebuddy-intl quota packs (#4422) parseQuotaData had a dedicated case for codebuddy-cn that forwards the recurring field, but none for codebuddy-intl - even though getCodeBuddyIntlUsage (open-sse/services/usage/codebuddy-cn.js) returns the exact same quota shape, bonus packs included. Falling through to the default case dropped recurring, so one-shot bonus packs (recurring:false, resetAt = hard expiry) rendered as recurring and showed 'Reset in' instead of 'Expires in'. Share the codebuddy-cn case with codebuddy-intl. Fixes #4362 --- .../usage/components/ProviderLimits/utils.js | 3 +- .../codebuddy-intl-quota-recurring.test.js | 32 +++++++++++++++++++ 2 files changed, 34 insertions(+), 1 deletion(-) create mode 100644 tests/unit/codebuddy-intl-quota-recurring.test.js diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js index 769e6e82..9cb7ad71 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js @@ -619,7 +619,8 @@ export function parseQuotaData(provider, data) { break; case "codebuddy-cn": - // CodeBuddy CN mixes recurring refill packs ("Monthly"/"Weekly"/...) + case "codebuddy-intl": + // CodeBuddy CN/Intl mix recurring refill packs ("Monthly"/"Weekly"/...) // with one-shot bonus packs ("Bonus Pack N"). Forward `recurring` // so the UI can show "Expires in" for bonus packs (whose resetAt is // a hard expiry, not a refresh) instead of "Reset in". diff --git a/tests/unit/codebuddy-intl-quota-recurring.test.js b/tests/unit/codebuddy-intl-quota-recurring.test.js new file mode 100644 index 00000000..475ed16b --- /dev/null +++ b/tests/unit/codebuddy-intl-quota-recurring.test.js @@ -0,0 +1,32 @@ +// #4362: parseQuotaData had a dedicated `case "codebuddy-cn"` that forwards +// `recurring`, but no case for "codebuddy-intl" — which returns the exact same +// quota shape (open-sse/services/usage/codebuddy-cn.js → getCodeBuddyIntlUsage). +// Falling through to `default` dropped `recurring`, so one-shot bonus packs +// (recurring:false, resetAt = hard expiry) rendered as "Reset in" instead of +// "Expires in". +import { describe, expect, it } from "vitest"; + +import { parseQuotaData } from "../../src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; + +const quotas = { + Monthly: { used: 100, total: 1000, resetAt: "2026-11-01T00:00:00Z", recurring: true }, + "Bonus Pack 1": { used: 5, total: 50, resetAt: "2026-10-15T00:00:00Z", recurring: false }, +}; + +function byName(parsed, name) { + return parsed.find((q) => q.name === name); +} + +describe("parseQuotaData forwards `recurring` for codebuddy-intl (#4362)", () => { + it("marks a bonus pack as non-recurring, like codebuddy-cn does", () => { + const intl = parseQuotaData("codebuddy-intl", { quotas }); + expect(byName(intl, "Bonus Pack 1").recurring).toBe(false); + expect(byName(intl, "Monthly").recurring).toBe(true); + }); + + it("matches codebuddy-cn exactly for the same payload", () => { + const cn = parseQuotaData("codebuddy-cn", { quotas }); + const intl = parseQuotaData("codebuddy-intl", { quotas }); + expect(intl).toEqual(cn); + }); +}); From e78b766a2095c8cebca185d936b73e3b36250637 Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 13:00:01 +0700 Subject: [PATCH 11/41] feat(kiro): add claude-opus-5.5 models to registry and capabilities --- open-sse/providers/capabilities.js | 7 ++- open-sse/providers/registry/kiro.js | 8 ++- tests/unit/kiro-claude-opus-5-5.test.js | 77 +++++++++++++++++++++++++ 3 files changed, 90 insertions(+), 2 deletions(-) create mode 100644 tests/unit/kiro-claude-opus-5-5.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index e3bb58ac..79aee774 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -83,8 +83,13 @@ export function capabilitiesFromServiceKind(kind) { * otherwise mis-match. Only declare deltas vs DEFAULT. */ export const MODEL_CAPABILITIES = { - // Claude Fable 5.1, Opus 5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern) + // Claude Fable 5.1, Opus 5.5/5, 4.6/4.7/4.8, and Kiro Sonnet 5 have 1M context + adaptive thinking (override generic claude pattern) "claude-fable-5-1": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, + // Claude Opus 5.5 — experimental preview on Kiro (rateMultiplier: 2.0, 1M context) (#4410) + "claude-opus-5.5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, + "claude-opus-5.5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, + "claude-opus-5.5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, + "claude-opus-5.5-thinking-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-opus-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-opus-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-opus-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, diff --git a/open-sse/providers/registry/kiro.js b/open-sse/providers/registry/kiro.js index f506b8d4..224cf801 100644 --- a/open-sse/providers/registry/kiro.js +++ b/open-sse/providers/registry/kiro.js @@ -41,7 +41,13 @@ export default { }, }, models: [ - // Opus (added per kiro.dev/changelog/models and kiro.dev/docs/models) + // Opus 5.5 — experimental preview, 1M context, 2x credits (#4410) + // Announced 2026-09-22; confirmed in kiro.dev session UI. + { id: "claude-opus-5.5", name: "Claude Opus 5.5" }, + { id: "claude-opus-5.5-thinking", name: "Claude Opus 5.5 (Thinking)" }, + { id: "claude-opus-5.5-agentic", name: "Claude Opus 5.5 (Agentic)" }, + { id: "claude-opus-5.5-thinking-agentic", name: "Claude Opus 5.5 (Thinking + Agentic)" }, + // Opus 5 { id: "claude-opus-5", name: "Claude Opus 5" }, { id: "claude-opus-5-thinking", name: "Claude Opus 5 (Thinking)" }, { id: "claude-opus-5-agentic", name: "Claude Opus 5 (Agentic)" }, diff --git a/tests/unit/kiro-claude-opus-5-5.test.js b/tests/unit/kiro-claude-opus-5-5.test.js new file mode 100644 index 00000000..91603295 --- /dev/null +++ b/tests/unit/kiro-claude-opus-5-5.test.js @@ -0,0 +1,77 @@ +/** + * Tests for #4410 — Kiro registry missing claude-opus-5.5 models. + * + * Kiro added Claude Opus 5.5 as an experimental preview on 2026-09-22. + * The model was missing from open-sse/providers/registry/kiro.js and + * open-sse/providers/capabilities.js. + * + * Fix: + * - Add claude-opus-5.5 / -thinking / -agentic / -thinking-agentic to kiro.js + * - Add capability entries (vision, reasoning, search, 1M context, adaptive thinking) + */ + +import { describe, it, expect } from "vitest"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import fs from "fs"; +import path from "path"; + +const kiroSrc = fs.readFileSync( + new URL("../../open-sse/providers/registry/kiro.js", import.meta.url), + "utf-8" +); + +describe("Kiro registry — claude-opus-5.5 (#4410)", () => { + it("includes claude-opus-5.5 in kiro registry", () => { + expect(kiroSrc).toContain('"claude-opus-5.5"'); + }); + + it("includes claude-opus-5.5-thinking in kiro registry", () => { + expect(kiroSrc).toContain('"claude-opus-5.5-thinking"'); + }); + + it("includes claude-opus-5.5-agentic in kiro registry", () => { + expect(kiroSrc).toContain('"claude-opus-5.5-agentic"'); + }); + + it("includes claude-opus-5.5-thinking-agentic in kiro registry", () => { + expect(kiroSrc).toContain('"claude-opus-5.5-thinking-agentic"'); + }); + + it("claude-opus-5 still present (regression guard)", () => { + expect(kiroSrc).toContain('"claude-opus-5"'); + }); +}); + +describe("capabilities — claude-opus-5.5 (#4410)", () => { + it("claude-opus-5.5 has vision:true", () => { + const caps = getCapabilitiesForModel("kiro", "claude-opus-5.5"); + expect(caps.vision).toBe(true); + }); + + it("claude-opus-5.5 has reasoning:true", () => { + const caps = getCapabilitiesForModel("kiro", "claude-opus-5.5"); + expect(caps.reasoning).toBe(true); + }); + + it("claude-opus-5.5 has contextWindow 1M", () => { + const caps = getCapabilitiesForModel("kiro", "claude-opus-5.5"); + expect(caps.contextWindow).toBe(1000000); + }); + + it("claude-opus-5.5 has maxOutput 128000", () => { + const caps = getCapabilitiesForModel("kiro", "claude-opus-5.5"); + expect(caps.maxOutput).toBe(128000); + }); + + it("claude-opus-5.5-thinking has same capabilities", () => { + const caps = getCapabilitiesForModel("kiro", "claude-opus-5.5-thinking"); + expect(caps.vision).toBe(true); + expect(caps.reasoning).toBe(true); + expect(caps.contextWindow).toBe(1000000); + }); + + it("claude-opus-5 still has 1M context (regression guard)", () => { + const caps = getCapabilitiesForModel("kiro", "claude-opus-5"); + expect(caps.contextWindow).toBe(1000000); + }); +}); \ No newline at end of file From 0a879c5cd34ace9b2c3cabc6884fb6e3502dce5c Mon Sep 17 00:00:00 2001 From: Rick Sanchez <101627789+m4tinbeigi-official@users.noreply.github.com> Date: Mon, 28 Sep 2026 13:00:28 +0700 Subject: [PATCH 12/41] feat: add v1m System One provider --- open-sse/handlers/systemoneCore.js | 5 ++-- open-sse/providers/registry/index.js | 2 ++ open-sse/providers/registry/v1m.js | 31 ++++++++++++++++++++++ public/providers/v1m.png | Bin 0 -> 2006 bytes src/shared/constants/providers.js | 2 +- tests/unit/v1m-systemone-provider.test.js | 29 ++++++++++++++++++++ 6 files changed, 66 insertions(+), 3 deletions(-) create mode 100644 open-sse/providers/registry/v1m.js create mode 100644 public/providers/v1m.png create mode 100644 tests/unit/v1m-systemone-provider.test.js diff --git a/open-sse/handlers/systemoneCore.js b/open-sse/handlers/systemoneCore.js index 82da570d..4fcb3004 100644 --- a/open-sse/handlers/systemoneCore.js +++ b/open-sse/handlers/systemoneCore.js @@ -19,7 +19,8 @@ export async function handleSystemoneCore({ }) { const { provider, model } = modelInfo; const cfg = PROVIDER_MEDIA[provider]?.systemoneConfig; - if (!cfg?.baseUrl) { + const targetUrl = credentials?.providerSpecificData?.baseUrl || cfg?.baseUrl; + if (!targetUrl) { return createErrorResult( HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support System One.` @@ -49,7 +50,7 @@ export async function handleSystemoneCore({ let providerResponse; try { - providerResponse = await fetch(cfg.baseUrl, { + providerResponse = await fetch(targetUrl, { method: "POST", headers, body: JSON.stringify(requestBody), diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 6c9d1f41..17d1fea9 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -131,6 +131,7 @@ import p127 from "./atria.js"; import p129 from "./agnes.js"; import p130 from "./bai.js"; import p131 from "./tinyfish.js"; +import p132 from "./v1m.js"; export default [ p0, p1, @@ -262,4 +263,5 @@ export default [ p129, p130, p131, + p132, ]; diff --git a/open-sse/providers/registry/v1m.js b/open-sse/providers/registry/v1m.js new file mode 100644 index 00000000..9028785c --- /dev/null +++ b/open-sse/providers/registry/v1m.js @@ -0,0 +1,31 @@ +export default { + id: "v1m", + priority: 45, + alias: "v1m", + aliases: ["systemone", "jev"], + uiAlias: "v1m", + display: { + name: "v1m (System One)", + icon: "psychology", + color: "#6366F1", + textIcon: "V1", + website: "https://v1m.ir", + notice: { + text: "v1m System One calibrated decision engine. Fast probabilistic evaluations over state and questions.", + apiKeyUrl: "https://v1m.ir", + }, + }, + category: "apikey", + authType: "apikey", + hasProviderSpecificData: true, + models: [ + { id: "rev-latest", name: "v1m Rev Latest (Calibrated)", kind: "systemone" }, + { id: "v1m-decision-engine", name: "v1m Decision Engine", kind: "systemone" }, + ], + serviceKinds: ["systemone"], + systemoneConfig: { + baseUrl: "https://v1m.ir/v1/systemone", + authType: "apikey", + authHeader: "bearer", + }, +}; diff --git a/public/providers/v1m.png b/public/providers/v1m.png new file mode 100644 index 0000000000000000000000000000000000000000..5bfda74270af951583136ade6d387bc5797da9d7 GIT binary patch literal 2006 zcmaKtdpy&NAI86%ZOz?EORUIk<{rt(b*pt;LX$3Q7IVKuR&uE(Oy%fixt2@OLWMOE zNo-5*Mp5W6j))4GhGfRkAHU!0_s8#?KR(ar{d%7NpC^Uj<{&SlDgyw3yptpT(E3mK zb#U0aHrn}g0KkTPCp_+OY~D=%V1#mjGPk>{E4rq`m4qcY*D{OduV&-1$`<;;zPlPU zcIYQEZtFR|S+Z10X;2~1dQ_yAHFvr*amo+2$Q&}R$Lk>*i*$yZ>jZZM79))6$U(uK zuFh|rFayJ}=9P_#aRwpAMCDvG;iLqNMM+>C4w90itHD}uC30E>8gIg2m zvk}wHEm?r_3Wb7pyP1pC93UbSx>hMZ7G7D9g1U_0Ih8pAYVdb z7~ISNG189!R^p$Uo;DaMjh4f|__ZNx-QNe+NQPi<;(*6%G-j$v05Hw~T6g@%SCCFX zZ3Omc17{rnZ=gx-WTxUNqT-&nvz%osP_hF0+KYo1lcwYdC-~1sSln`;f(Tdr>IhVU zZJxcr-%7yhafjMXU%oUZO$*=<@~76G9;GtUR(HC3it>dBI*tPUqh%k#OP70nE4cHK z5`{-MH6-cn{f#_rD~n82#wr1J0$b@7x*mAfI|s@+=EID8PU#RzWxrHRk1!%am6Up* zJ=;*NM*cGJ_%WV1tda;7y&`yRr#osxQt+_l<(@b1^XXa+>C`WrDogP<|FM4WYrM$r ztl;fapEG>=Taw1k*U!v06@K-ZF`oI;7QStpt^+F}?0l*6_V}=}VlwCM6-7H$!YY3; zrg2%t<=&~VlJ<7Ur&89@OHn78>`Bp%&oiz!R_)UR#;&LwK77g5oE#89{CPgmRT{Ob zB|SpRHO+h&>`L*t<{uk2r|JITTpc?+%dii&Ho=!W8;!JFswLJxhE&(9S;{xuz4Jp> zxLW;NXO7Nw<3n-JBk3VR{?pg#rOuS8tmoe&M&Q&rUm2Fy0kxh;zL_|sh}BHwy)y4e3JiEMfBN}^PlgZ1(8VFI`xlyFtfz; zoz~;`3wZ<8mKM33->u}?AssV$Pky`-xyLSSjxdIz456?E^a%s#O=z$gqTu|ahXdhJ=no)2dv=b&v0mpZb6SYZ()<%Psc(k=J4t#O>_=Wh7edTf}$pQLUxY_g&BDsB!-=O)P zfjQfsHPfk=3p%oUsbA7|6T+b13U&Gfq@zZvtij5B$;JC6XlR}N)^`wZSA8gZ4XZh| zXpchehIsRZRM=v?WmqgN<%XXpBa=IElzF2}j#yN|TOxC08EpV%y5f_@lIB{4;~0;l_%E~O zcC+1DX+0g|Nw&$E5Q{KSrxE#_PTx`JWl+Y9eu$v;!JIE+HvH=8UjppN((mTNs^dgpM>Jj>VW~T+S>2t{@6p_RM0e@ zW9zNXeKATBL=r>m;|5E%tmzwjS8T^hPWy}~D%@NPIvcsO)dED!_szzK26ihnR(+q2 z;2m8Z_~^*U;oJfggs z9%99nXMnPrzfM&(O~1o9)yhu9Xs=n5P@sXd&vz=H?%?MNP`(EqcoajPh$rRy9bU~J zlPY`L>pf|%8g+853}Y2{h3O$%3ZUKEosGJW!R9yBeSqoyks!^r8d{8QGrjkuVPQ{+ zrd)4v$jFOFckXxd53}uV7ww#3CpU;Wr;~2j-2#T-Z!VI|h5+|$&~Ez@4sE`-2%&3( zz@~j$00&Ni%!6c@e(qW*S?)*0C$Wn>M&JoTNAc6S8FryKNCb)Qd!Ld3=N9|SCvA+1 z(%kj#?d!AZkV@`Vv}VqnZlw#3I{yK?+w`%CROo=Izc+`o!9KRd12j-V zIyKZeJiKAFOMNz0AE>K3$u}E9WbhP+=T^o>1Mh8D4cYiX&W2YD*dB-3U?gIx#yHY4 z(|x5fVs;=d);NA2+-KXrb?!fw@qYuv8Q1F@R|@~9&HeO&1rW^_d8~bX74(4sP(v&7 z;4j9kvt8?EhhJ;}1SC?@O)h4c0ZvLTWv}^Xq&1}Jm*2;_D<+^k!2~267X8uRT8aT# zZ9YXq3B)BMdK2o+7ksV0g7nmuq|m3SLy$B`Q`48Zn5dO&&U;6xQ?Et-tEqUQLXA1A wM5;iWej)ZqsH_#pWiK)Lya8PU4u1J5eiJfXYzt#`uJ1MAwBHTS+DoDT4XX8`Q2+n{ literal 0 HcmV?d00001 diff --git a/src/shared/constants/providers.js b/src/shared/constants/providers.js index 1e5f30f0..fa289469 100644 --- a/src/shared/constants/providers.js +++ b/src/shared/constants/providers.js @@ -1,6 +1,6 @@ // Provider definitions import REGISTRY from "open-sse/providers/registry/index.js"; -import { RISK_NOTICE } from "@/shared/constants/providersDisplay"; +import { RISK_NOTICE } from "@/shared/constants/providersDisplay.js"; const MEDIA_ENTRY_KEYS = [ "serviceKinds", "ttsConfig", "sttConfig", "embeddingConfig", diff --git a/tests/unit/v1m-systemone-provider.test.js b/tests/unit/v1m-systemone-provider.test.js new file mode 100644 index 00000000..39a44d46 --- /dev/null +++ b/tests/unit/v1m-systemone-provider.test.js @@ -0,0 +1,29 @@ +import { describe, expect, it } from "vitest"; +import REGISTRY from "../../open-sse/providers/registry/index.js"; +import { PROVIDER_MEDIA } from "../../open-sse/providers/index.js"; +import { AI_PROVIDERS, getProvidersByKind } from "@/shared/constants/providers"; + +describe("v1m System One provider", () => { + const entry = REGISTRY.find((e) => e.id === "v1m"); + + it("is registered as a System One apikey provider", () => { + expect(entry).toBeDefined(); + expect(entry.category).toBe("apikey"); + expect(entry.serviceKinds).toEqual(["systemone"]); + expect(PROVIDER_MEDIA["v1m"]?.systemoneConfig?.baseUrl).toBe("https://v1m.ir/v1/systemone"); + }); + + it("appears in getProvidersByKind('systemone')", () => { + const list = getProvidersByKind("systemone"); + const found = list.find((p) => p.id === "v1m"); + expect(found).toBeDefined(); + expect(found.alias).toBe("v1m"); + expect(found.systemoneConfig?.baseUrl).toBe("https://v1m.ir/v1/systemone"); + }); + + it("exposes calibrated models", () => { + const ids = (entry.models || []).map((m) => m.id); + expect(ids).toContain("rev-latest"); + expect(ids).toContain("v1m-decision-engine"); + }); +}); From 45d42b8066eb9c4b5179b3f1549c8dff62727e11 Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 13:06:30 +0700 Subject: [PATCH 13/41] fix(test): add codebuddy-intl to OAUTH_TEST_CONFIG with tokenExists strategy --- src/app/api/providers/[id]/test/testUtils.js | 3 ++ tests/unit/codebuddy-intl-test-config.test.js | 39 +++++++++++++++++++ 2 files changed, 42 insertions(+) create mode 100644 tests/unit/codebuddy-intl-test-config.test.js diff --git a/src/app/api/providers/[id]/test/testUtils.js b/src/app/api/providers/[id]/test/testUtils.js index 92873ccc..ecdac3e9 100644 --- a/src/app/api/providers/[id]/test/testUtils.js +++ b/src/app/api/providers/[id]/test/testUtils.js @@ -100,6 +100,9 @@ const OAUTH_TEST_CONFIG = { authPrefix: "Bearer ", }, "codebuddy-cn": { tokenExists: true }, + // codebuddy-intl uses the same JWT token structure as codebuddy-cn + // (access + refresh token pair, ~1-year expiry) — same test strategy (#4232). + "codebuddy-intl": { tokenExists: true }, kimchi: { url: KIMCHI_CONFIG.validationUrl || "https://api.cast.ai/v1/llm/openai/supported-providers", method: "GET", diff --git a/tests/unit/codebuddy-intl-test-config.test.js b/tests/unit/codebuddy-intl-test-config.test.js new file mode 100644 index 00000000..0865e7c0 --- /dev/null +++ b/tests/unit/codebuddy-intl-test-config.test.js @@ -0,0 +1,39 @@ +/** + * Regression test for #4232 + * + * codebuddy-intl was missing from OAUTH_TEST_CONFIG. The test route handler + * returned {"valid":false,"error":"Provider test not supported"} for every + * codebuddy-intl account, regardless of token validity. + * + * Fix: add "codebuddy-intl": { tokenExists: true } alongside "codebuddy-cn". + * Both providers use the same JWT token structure (eyJ…, ~1-year expiry, + * access + refresh token pair) so the same test strategy applies. + */ + +import { describe, it, expect } from "vitest"; +import fs from "fs"; +import path from "path"; + +// Read the source file and verify the config entry is present. +// testUtils.js is a Next.js server file so we inspect the source text +// rather than importing it (avoids next/server bootstrap requirements). + +const src = fs.readFileSync( + path.resolve("../src/app/api/providers/[id]/test/testUtils.js"), + "utf-8" +); + +describe("OAUTH_TEST_CONFIG — codebuddy-intl (#4232)", () => { + it('contains "codebuddy-intl" entry', () => { + expect(src).toContain('"codebuddy-intl"'); + }); + + it('"codebuddy-intl" has tokenExists: true', () => { + // Match the specific entry: "codebuddy-intl": { tokenExists: true } + expect(src).toMatch(/"codebuddy-intl"\s*:\s*\{\s*tokenExists\s*:\s*true\s*\}/); + }); + + it('"codebuddy-cn" still has tokenExists: true (regression guard)', () => { + expect(src).toMatch(/"codebuddy-cn"\s*:\s*\{\s*tokenExists\s*:\s*true\s*\}/); + }); +}); \ No newline at end of file From 8a4f4d9d2cf552f250cb3a37b525cfef9546c251 Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 13:09:15 +0700 Subject: [PATCH 14/41] fix(capabilities): add deepseek-v4-1-flash vision alias; fix(modal): add zed to live catalog providers --- open-sse/providers/capabilities.js | 3 + src/shared/components/ModelSelectModal.js | 8 +- .../kenari-deepseek-vision-zed-modal.test.js | 80 +++++++++++++++++++ 3 files changed, 89 insertions(+), 2 deletions(-) create mode 100644 tests/unit/kenari-deepseek-vision-zed-modal.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 79aee774..5b52516d 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -130,7 +130,10 @@ export const MODEL_CAPABILITIES = { // DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose // 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry // short-circuits the pattern table, so a vision-only delta would drop them. + // Some providers (e.g. Kenari) expose this model under the hyphenated ID + // "deepseek-v4-1-flash" (dash instead of dot); add it as an alias (#4293). "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + "deepseek-v4-1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, "deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 }, // Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases diff --git a/src/shared/components/ModelSelectModal.js b/src/shared/components/ModelSelectModal.js index 20e17d95..af07f028 100644 --- a/src/shared/components/ModelSelectModal.js +++ b/src/shared/components/ModelSelectModal.js @@ -22,7 +22,9 @@ const NO_AUTH_PROVIDER_IDS = Object.keys(FREE_PROVIDERS).filter(id => FREE_PROVI // Providers with per-account live catalogs via /api/providers/[id]/models. // Static registry stays as fallback when live fetch fails or is empty. -const LIVE_CATALOG_PROVIDERS = ["cursor", "cline", "clinepass"]; +// zed added in #4244: its backend customResolver already returns live models +// but the frontend omitted it, hiding Zed entirely from the Combo picker. +const LIVE_CATALOG_PROVIDERS = ["cursor", "cline", "clinepass", "zed"]; // Fetch a provider's account-scoped catalog for every active connection and merge // the results. Entries collapse by model id on purpose: two connections of the @@ -112,10 +114,12 @@ export default function ModelSelectModal({ const cursorConnectionIds = liveConnectionIdsByProvider.cursor; const clineConnectionIds = liveConnectionIdsByProvider.cline; const clinepassConnectionIds = liveConnectionIdsByProvider.clinepass; + const zedConnectionIds = liveConnectionIdsByProvider.zed; const cursorModels = useLiveProviderModels(isOpen, cursorConnectionIds, "Cursor"); const clineModels = useLiveProviderModels(isOpen, clineConnectionIds, "Cline"); const clinepassModels = useLiveProviderModels(isOpen, clinepassConnectionIds, "ClinePass"); + const zedModels = useLiveProviderModels(isOpen, zedConnectionIds, "Zed"); const fetchCombos = async () => { try { @@ -348,7 +352,7 @@ export default function ModelSelectModal({ hasModels: mergedModels.length > 0, }; } else { - const liveModels = providerId === "cursor" ? cursorModels : providerId === "cline" ? clineModels : providerId === "clinepass" ? clinepassModels : []; + const liveModels = providerId === "cursor" ? cursorModels : providerId === "cline" ? clineModels : providerId === "clinepass" ? clinepassModels : providerId === "zed" ? zedModels : []; const hardcodedModels = liveModels.length > 0 ? liveModels : getModelsByProviderId(providerId); diff --git a/tests/unit/kenari-deepseek-vision-zed-modal.test.js b/tests/unit/kenari-deepseek-vision-zed-modal.test.js new file mode 100644 index 00000000..af65b8f3 --- /dev/null +++ b/tests/unit/kenari-deepseek-vision-zed-modal.test.js @@ -0,0 +1,80 @@ +/** + * Tests for two independent frontend/capabilities fixes: + * + * #4293 — deepseek-v4-1-flash (hyphen) has vision=false + * Kenari exposes the model under the ID "deepseek-v4-1-flash" (dash instead + * of dot). The capabilities table only had "deepseek-v4.1-flash" (dot), so + * the hyphenated variant fell through to the generic *deepseek* pattern which + * has vision:false. Fix: add "deepseek-v4-1-flash" as an alias with the same + * vision:true entry. + * + * #4244 — Zed missing from SelectModelModal LIVE_CATALOG_PROVIDERS + * The hardcoded list ["cursor","cline","clinepass"] omitted "zed", so the + * Zed provider card was never fetched and never shown in the Combo picker. + * Fix: add "zed" to the list (and wire up the corresponding state/hook/ternary). + */ + +import { describe, it, expect, beforeAll } from "vitest"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; + +// ── #4293 deepseek-v4-1-flash vision capability ─────────────────────────── + +describe("deepseek-v4-1-flash capabilities (#4293)", () => { + it("reports vision:true for the hyphenated deepseek-v4-1-flash id (Kenari variant)", () => { + // Any provider that uses the hyphenated id should get vision:true + const caps = getCapabilitiesForModel("kenari", "deepseek-v4-1-flash"); + expect(caps.vision).toBe(true); + }); + + it("still reports vision:true for the dotted deepseek-v4.1-flash id", () => { + const caps = getCapabilitiesForModel("ollama", "deepseek-v4.1-flash"); + expect(caps.vision).toBe(true); + }); + + it("still reports reasoning:true for deepseek-v4-1-flash", () => { + const caps = getCapabilitiesForModel("kenari", "deepseek-v4-1-flash"); + expect(caps.reasoning).toBe(true); + }); + + it("reports the correct contextWindow (1M) for deepseek-v4-1-flash", () => { + const caps = getCapabilitiesForModel("kenari", "deepseek-v4-1-flash"); + expect(caps.contextWindow).toBe(1000000); + }); +}); + +// ── #4244 LIVE_CATALOG_PROVIDERS includes zed ──────────────────────────── +// ModelSelectModal.js is JSX so we cannot import it in Vitest without a +// JSX transform. Read the source text and verify the constant definition +// directly — this is reliable and does not require a full React setup. + +import fs from "fs"; +import path from "path"; + +describe("ModelSelectModal LIVE_CATALOG_PROVIDERS includes zed (#4244)", () => { + let src; + beforeAll(() => { + const fileUrl = new URL("../../src/shared/components/ModelSelectModal.js", import.meta.url); + src = fs.readFileSync(fileUrl, "utf-8"); + }); + + it("includes zed in LIVE_CATALOG_PROVIDERS", () => { + // Match the const definition line and verify zed is present + const match = src.match(/const LIVE_CATALOG_PROVIDERS\s*=\s*\[([^\]]+)\]/); + expect(match).toBeTruthy(); + const list = match[1]; + expect(list).toContain('"zed"'); + }); + + it("still includes cursor, cline, clinepass", () => { + const match = src.match(/const LIVE_CATALOG_PROVIDERS\s*=\s*\[([^\]]+)\]/); + const list = match[1]; + expect(list).toContain('"cursor"'); + expect(list).toContain('"cline"'); + expect(list).toContain('"clinepass"'); + }); + + it("zedModels hook call is present in the file", () => { + expect(src).toContain("zedModels"); + expect(src).toContain("zedConnectionIds"); + }); +}); \ No newline at end of file From 24664f2c5a2a6b047c862d52106e063ff2b4eb37 Mon Sep 17 00:00:00 2001 From: claytontavaresdan Date: Mon, 28 Sep 2026 14:12:15 +0700 Subject: [PATCH 15/41] fix(proxy): hold strictProxy when no proxy resolves --- open-sse/utils/proxyFetch.js | 18 +++++ src/lib/network/connectionProxy.js | 12 +++ tests/unit/strict-proxy-enforcement.test.js | 87 +++++++++++++++++++++ 3 files changed, 117 insertions(+) create mode 100644 tests/unit/strict-proxy-enforcement.test.js diff --git a/open-sse/utils/proxyFetch.js b/open-sse/utils/proxyFetch.js index b5403fde..d7705c77 100644 --- a/open-sse/utils/proxyFetch.js +++ b/open-sse/utils/proxyFetch.js @@ -351,6 +351,24 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) { } } + // Strict mode means "never leave over the direct IP". Reaching here with a + // proxy configured but unresolved is exactly that case — an inactive or + // empty pool, or every proxy removed — so refuse instead of silently + // exposing the real address (#4333). The catch blocks above only cover a + // proxy that was actually tried. + // + // Gate on a proxy being *intended*: callers like the Qoder executor set + // strictProxy to mean "do not replay this request directly if the proxy + // fails" (a replayed COSY signature returns 403), not "a proxy is required". + // With nothing configured they must keep working. + const proxyIntended = proxyOptions?.proxyPoolId + || proxyOptions?.enabled === true + || proxyOptions?.connectionProxyEnabled === true + || !!normalizeString(proxyOptions?.url ?? proxyOptions?.connectionProxyUrl); + if (proxyOptions?.strictProxy === true && proxyIntended) { + throw new Error("[ProxyFetch] Proxy required but none resolved (strictProxy=true)"); + } + // got-scraping disabled — use native fetch directly // (Re-enable per-host by wrapping with tryGotScrapingFetch when needed) return originalFetch(url, options); diff --git a/src/lib/network/connectionProxy.js b/src/lib/network/connectionProxy.js index 9ecd2535..9d8846ba 100644 --- a/src/lib/network/connectionProxy.js +++ b/src/lib/network/connectionProxy.js @@ -77,6 +77,12 @@ export async function resolveConnectionProxyConfig( const legacy = normalizeLegacyProxy(providerSpecificData); + // A strict pool must keep its guarantee even when the pool itself is not + // usable (inactive, or saved without a url). Otherwise the unusable-pool + // path below reports strictProxy:false and the request silently leaves + // over the direct IP — the leak strict mode exists to prevent (#4333). + let poolStrictProxy = false; + /** * ----------------------------- * Proxy Pool Resolution @@ -93,6 +99,8 @@ export async function resolveConnectionProxyConfig( proxyPool.isActive === true && proxyUrl; + poolStrictProxy = proxyPool?.strictProxy === true; + if (isValidPool) { /** * Vercel/Cloudflare relay proxies use base URL rewriting @@ -148,6 +156,8 @@ export async function resolveConnectionProxyConfig( proxyPoolId: proxyPoolId || null, proxyPool: null, + strictProxy: poolStrictProxy, + ...legacy, }; } @@ -163,6 +173,8 @@ export async function resolveConnectionProxyConfig( proxyPoolId: proxyPoolId || null, proxyPool: null, + strictProxy: poolStrictProxy, + ...legacy, }; } catch (error) { diff --git a/tests/unit/strict-proxy-enforcement.test.js b/tests/unit/strict-proxy-enforcement.test.js new file mode 100644 index 00000000..81f46d03 --- /dev/null +++ b/tests/unit/strict-proxy-enforcement.test.js @@ -0,0 +1,87 @@ +// #4333: "Strict Proxy" did not hold. With a strict pool assigned and every +// proxy in it dead, requests still went out over the direct IP — the exact +// leak the setting exists to prevent. +// +// Two halves, one per layer: +// +// 1. resolveConnectionProxyConfig drops strictProxy whenever the pool is not +// usable (inactive, or saved with an empty proxyUrl). isValidPool gates the +// only two returns that carry strictProxy, so an unusable strict pool falls +// through to the legacy/none branches, which report strictProxy:false. +// +// 2. proxyAwareFetch only honours strictProxy inside the catch of a proxy +// attempt. When no proxy URL resolves there is nothing to try, so it +// reaches the trailing `return originalFetch(url, options)` and connects +// directly. +import { describe, expect, it, vi } from "vitest"; + +vi.mock("@/models", () => ({ + getProxyPoolById: vi.fn(), +})); + +const { getProxyPoolById } = await import("@/models"); +const { resolveConnectionProxyConfig } = await import("../../src/lib/network/connectionProxy.js"); +const { proxyAwareFetch } = await import("../../open-sse/utils/proxyFetch.js"); + +describe("strict pool keeps strictProxy when the pool is unusable (#4333)", () => { + it("keeps strictProxy for an inactive strict pool", async () => { + getProxyPoolById.mockResolvedValue({ + id: "p1", isActive: false, proxyUrl: "http://127.0.0.1:7890", strictProxy: true, + }); + const cfg = await resolveConnectionProxyConfig({ proxyPoolId: "p1" }); + expect(cfg.strictProxy).toBe(true); + }); + + it("keeps strictProxy for a strict pool saved without a proxy url", async () => { + getProxyPoolById.mockResolvedValue({ + id: "p2", isActive: true, proxyUrl: "", strictProxy: true, + }); + const cfg = await resolveConnectionProxyConfig({ proxyPoolId: "p2" }); + expect(cfg.strictProxy).toBe(true); + }); + + it("still reports strictProxy:false for a non-strict pool", async () => { + getProxyPoolById.mockResolvedValue({ + id: "p3", isActive: false, proxyUrl: "http://127.0.0.1:7890", strictProxy: false, + }); + const cfg = await resolveConnectionProxyConfig({ proxyPoolId: "p3" }); + expect(cfg.strictProxy).toBe(false); + }); + + it("still reports strictProxy:false when no pool is assigned", async () => { + const cfg = await resolveConnectionProxyConfig({}); + expect(cfg.strictProxy).toBe(false); + }); +}); + +describe("strictProxy refuses a direct connection (#4333)", () => { + it("throws when a pool is assigned but no proxy url resolved", async () => { + await expect( + proxyAwareFetch("https://api.example.com/v1/chat", {}, { proxyPoolId: "p1", strictProxy: true }), + ).rejects.toThrow(/strictProxy/); + }); + + it("throws when the pool is enabled but carries an empty url", async () => { + await expect( + proxyAwareFetch("https://api.example.com/v1/chat", {}, { enabled: true, url: "", strictProxy: true }), + ).rejects.toThrow(/strictProxy/); + }); + + it("does not block a caller that sets strictProxy with no proxy configured", async () => { + // The Qoder executor passes strictProxy:true to mean "do not replay this + // request directly if the proxy fails" — a replayed COSY signature gets a + // 403. With nothing configured it must still reach the network. + await expect( + proxyAwareFetch("https://nonexistent.invalid/v1/chat", {}, { strictProxy: true }), + ).rejects.not.toThrow(/strictProxy/); + }); + + it("does not block a request when strictProxy is off", async () => { + // No proxy, not strict: the call is allowed to reach the network layer. + // It fails on DNS here, which is fine — what matters is that the refusal + // is NOT the strictProxy guard. + await expect( + proxyAwareFetch("https://nonexistent.invalid/v1/chat", {}, { strictProxy: false }), + ).rejects.not.toThrow(/strictProxy/); + }); +}); From 4f274c7f2a8b315a2b99bdcfbd0abb6ca3a40c18 Mon Sep 17 00:00:00 2001 From: Clayton Tavares Date: Mon, 28 Sep 2026 14:12:37 +0700 Subject: [PATCH 16/41] fix(translator/claude): keep a user turn whose only block is container_upload --- open-sse/translator/formats/claude.js | 34 ++++++++----- open-sse/translator/schema/blocks.js | 1 + tests/unit/claude-container-upload.test.js | 55 ++++++++++++++++++++++ 3 files changed, 77 insertions(+), 13 deletions(-) create mode 100644 tests/unit/claude-container-upload.test.js diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index e52bef9d..a9bbd19d 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -26,24 +26,32 @@ export function lastCacheableToolIndex(tools) { } // Check if message has valid non-empty content +// A block type outside this list makes the whole message count as empty and be +// dropped by prepareClaudeRequest — so anything the caller can legitimately +// send alone must be listed. container_upload (Files API) is one of those: +// a user turn whose only block is a file reference is valid Anthropic input +// (#4316), and dropping it forwarded `messages: []` to the provider. +const CONTENTFUL_BLOCKS = new Set([ + CLAUDE_BLOCK.TOOL_USE, + CLAUDE_BLOCK.TOOL_RESULT, + CLAUDE_BLOCK.IMAGE, + CLAUDE_BLOCK.DOCUMENT, + CLAUDE_BLOCK.CONTAINER_UPLOAD, +]); + +function isContentfulBlock(block) { + if (!block) return false; + if (block.type === CLAUDE_BLOCK.TEXT) return !!block.text?.trim(); + return CONTENTFUL_BLOCKS.has(block.type); +} + export function hasValidContent(msg) { if (typeof msg.content === "string" && msg.content.trim()) return true; if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) { - const block = msg.content; - return !!((block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) || - block.type === CLAUDE_BLOCK.TOOL_USE || - block.type === CLAUDE_BLOCK.TOOL_RESULT || - block.type === CLAUDE_BLOCK.IMAGE || - block.type === CLAUDE_BLOCK.DOCUMENT); + return isContentfulBlock(msg.content); } if (Array.isArray(msg.content)) { - return msg.content.some(block => - (block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) || - block.type === CLAUDE_BLOCK.TOOL_USE || - block.type === CLAUDE_BLOCK.TOOL_RESULT || - block.type === CLAUDE_BLOCK.IMAGE || - block.type === CLAUDE_BLOCK.DOCUMENT - ); + return msg.content.some(isContentfulBlock); } return false; } diff --git a/open-sse/translator/schema/blocks.js b/open-sse/translator/schema/blocks.js index 91c71c4a..d2bb783c 100644 --- a/open-sse/translator/schema/blocks.js +++ b/open-sse/translator/schema/blocks.js @@ -18,6 +18,7 @@ export const CLAUDE_BLOCK = { DOCUMENT: "document", TOOL_USE: "tool_use", TOOL_RESULT: "tool_result", + CONTAINER_UPLOAD: "container_upload", THINKING: "thinking", REDACTED_THINKING: "redacted_thinking", SERVER_TOOL_USE: "server_tool_use", diff --git a/tests/unit/claude-container-upload.test.js b/tests/unit/claude-container-upload.test.js new file mode 100644 index 00000000..0b74dc2f --- /dev/null +++ b/tests/unit/claude-container-upload.test.js @@ -0,0 +1,55 @@ +// #4316: a user message whose only content block is `container_upload` +// (Anthropic Files API) was dropped whole, so the provider received +// `messages: []` and the request still returned 200 with no indication that +// the user turn had vanished. +// +// hasValidContent() enumerated the block types that count as content, and any +// type outside that list made the message look empty — prepareClaudeRequest +// then filtered it out. container_upload is valid Anthropic input on its own, +// and on the Claude→Claude route no translation runs at all, so the block +// should reach the provider untouched. +import { describe, expect, it } from "vitest"; + +import { hasValidContent, prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; + +const uploadBlock = { type: "container_upload", file_id: "file_abc123" }; + +describe("container_upload keeps the user turn alive (#4316)", () => { + it("counts a lone container_upload block as content", () => { + expect(hasValidContent({ role: "user", content: [uploadBlock] })).toBe(true); + }); + + it("counts a bare container_upload object as content", () => { + expect(hasValidContent({ role: "user", content: uploadBlock })).toBe(true); + }); + + it("does not forward messages: [] for a container_upload-only request", () => { + const body = { + model: "claude-sonnet-4-5", + max_tokens: 64, + messages: [{ role: "user", content: [uploadBlock] }], + }; + const prepared = prepareClaudeRequest(body); + expect(prepared.messages).toHaveLength(1); + expect(prepared.messages[0].role).toBe("user"); + expect(prepared.messages[0].content).toContainEqual(expect.objectContaining({ + type: "container_upload", + file_id: "file_abc123", + })); + }); + + it("still drops a genuinely empty message", () => { + expect(hasValidContent({ role: "user", content: [] })).toBe(false); + expect(hasValidContent({ role: "user", content: [{ type: "text", text: " " }] })).toBe(false); + }); + + it("keeps a container_upload alongside text", () => { + const body = { + model: "claude-sonnet-4-5", + max_tokens: 64, + messages: [{ role: "user", content: [uploadBlock, { type: "text", text: "summarise this" }] }], + }; + const prepared = prepareClaudeRequest(body); + expect(prepared.messages).toHaveLength(1); + }); +}); From aafe3002265ed05a50c0246f46068cc78c43cd03 Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 14:26:45 +0700 Subject: [PATCH 17/41] fix(translator): strip errorMessage and other non-standard schema keywords from Gemini tool schemas --- open-sse/translator/formats/gemini.js | 9 +- .../unit/gemini-unknown-schema-fields.test.js | 105 ++++++++++++++++++ 2 files changed, 113 insertions(+), 1 deletion(-) create mode 100644 tests/unit/gemini-unknown-schema-fields.test.js diff --git a/open-sse/translator/formats/gemini.js b/open-sse/translator/formats/gemini.js index 412fadcc..75a429b8 100644 --- a/open-sse/translator/formats/gemini.js +++ b/open-sse/translator/formats/gemini.js @@ -32,7 +32,14 @@ export const UNSUPPORTED_SCHEMA_CONSTRAINTS = [ "title", "optional", "deprecated", "if", "then", "else", "contentMediaType", "contentEncoding", // UI/Styling properties (from Cursor tools - NOT JSON Schema standard) "cornerRadius", "fillColor", "fontFamily", "fontSize", "fontWeight", - "gap", "padding", "strokeColor", "strokeThickness", "textColor" + "gap", "padding", "strokeColor", "strokeThickness", "textColor", + // Non-standard annotation/error keywords used by some MCP tool schemas (#4283). + // Gemini's schema proto has no field for these and rejects the whole request with + // "Unknown name X: Cannot find field" if any nested schema node carries them. + "errorMessage", "errorMessages", "x-errorMessage", "x-errorMessages", + "markdownDescription", "x-intellij-html-description", + "x-taplo-info", "x-taplo", "doNotSuggest", "suggestSortText", + "minProperties", "maxProperties" ]; // Default safety settings diff --git a/tests/unit/gemini-unknown-schema-fields.test.js b/tests/unit/gemini-unknown-schema-fields.test.js new file mode 100644 index 00000000..b28be4d5 --- /dev/null +++ b/tests/unit/gemini-unknown-schema-fields.test.js @@ -0,0 +1,105 @@ +/** + * Regression test for #4283 + * + * Gemini Antigravity rejects tool schemas that contain unknown JSON Schema + * keywords with: "Unknown name X: Cannot find field." + * + * The UNSUPPORTED_SCHEMA_CONSTRAINTS list in gemini.js did not include + * "errorMessage" (and similar non-standard annotation keywords used by some + * MCP tool schemas), causing 400 INVALID_ARGUMENT errors on tools with error + * documentation fields. + * + * Fix: add errorMessage, errorMessages, markdownDescription, and other + * non-standard annotation keywords to the strip list. + */ + +import { describe, it, expect } from "vitest"; +import { cleanJSONSchemaForAntigravity, UNSUPPORTED_SCHEMA_CONSTRAINTS } from "../../open-sse/translator/formats/gemini.js"; + +describe("UNSUPPORTED_SCHEMA_CONSTRAINTS includes non-standard annotation keywords (#4283)", () => { + it("includes errorMessage", () => { + expect(UNSUPPORTED_SCHEMA_CONSTRAINTS).toContain("errorMessage"); + }); + it("includes errorMessages", () => { + expect(UNSUPPORTED_SCHEMA_CONSTRAINTS).toContain("errorMessages"); + }); + it("includes markdownDescription", () => { + expect(UNSUPPORTED_SCHEMA_CONSTRAINTS).toContain("markdownDescription"); + }); + it("includes minProperties", () => { + expect(UNSUPPORTED_SCHEMA_CONSTRAINTS).toContain("minProperties"); + }); + it("includes maxProperties", () => { + expect(UNSUPPORTED_SCHEMA_CONSTRAINTS).toContain("maxProperties"); + }); +}); + +describe("cleanJSONSchemaForAntigravity strips errorMessage recursively (#4283)", () => { + it("strips top-level errorMessage", () => { + const schema = { + type: "object", + properties: { + code: { type: "integer" } + }, + errorMessage: "Invalid input" + }; + const result = cleanJSONSchemaForAntigravity(structuredClone(schema)); + expect(result).not.toHaveProperty("errorMessage"); + }); + + it("strips errorMessage nested inside array items", () => { + const schema = { + type: "object", + properties: { + tags: { + type: "array", + items: { + type: "string", + errorMessage: "Must be a non-empty string" + } + } + } + }; + const result = cleanJSONSchemaForAntigravity(structuredClone(schema)); + expect(result.properties.tags.items).not.toHaveProperty("errorMessage"); + }); + + it("strips markdownDescription from nested property", () => { + const schema = { + type: "object", + properties: { + name: { + type: "string", + markdownDescription: "The **name** of the resource" + } + } + }; + const result = cleanJSONSchemaForAntigravity(structuredClone(schema)); + expect(result.properties.name).not.toHaveProperty("markdownDescription"); + }); + + it("strips minProperties / maxProperties", () => { + const schema = { + type: "object", + minProperties: 1, + maxProperties: 10, + properties: { x: { type: "string" } } + }; + const result = cleanJSONSchemaForAntigravity(structuredClone(schema)); + expect(result).not.toHaveProperty("minProperties"); + expect(result).not.toHaveProperty("maxProperties"); + }); + + it("leaves other valid fields intact", () => { + const schema = { + type: "object", + description: "A valid tool", + properties: { + n: { type: "number", description: "A number" } + } + }; + const result = cleanJSONSchemaForAntigravity(structuredClone(schema)); + expect(result.description).toBe("A valid tool"); + expect(result.properties.n.description).toBe("A number"); + }); +}); \ No newline at end of file From b58bd80406a24030acac6312fce519534c61947a Mon Sep 17 00:00:00 2001 From: decolua Date: Mon, 28 Sep 2026 18:51:31 +0700 Subject: [PATCH 18/41] fix(proxy): auto-fallback to insecure TLS on self-signed cert errors Co-Authored-By: Claude Code --- open-sse/utils/proxyFetch.js | 60 ++++++++++++++++++++++++++++-------- 1 file changed, 48 insertions(+), 12 deletions(-) diff --git a/open-sse/utils/proxyFetch.js b/open-sse/utils/proxyFetch.js index d7705c77..af819cc5 100644 --- a/open-sse/utils/proxyFetch.js +++ b/open-sse/utils/proxyFetch.js @@ -99,6 +99,20 @@ async function tryGotScrapingFetch(url, options) { // DNS cache — use Map to avoid prototype pollution via malformed hostnames const DNS_CACHE = new Map(); +const TLS_CERT_ERRORS = new Set([ + "SELF_SIGNED_CERT_IN_CHAIN", + "DEPTH_ZERO_SELF_SIGNED_CERT", + "UNABLE_TO_VERIFY_LEAF_SIGNATURE", + "UNABLE_TO_GET_ISSUER_CERT", + "UNABLE_TO_GET_ISSUER_CERT_LOCALLY", + "CERT_HAS_EXPIRED", + "ERR_TLS_CERT_ALTNAME_INVALID", +]); + +function isTlsCertError(err) { + const code = err?.cause?.code || err?.code; + return TLS_CERT_ERRORS.has(code); +} const MITM_BYPASS_HOSTS = [ "cloudcode-pa.googleapis.com", "daily-cloudcode-pa.googleapis.com", @@ -216,20 +230,44 @@ function resolveConnectionProxyUrl(targetUrl, proxyOptions) { /** * Create proxy dispatcher lazily (undici-compatible) */ -async function getDispatcher(proxyUrl) { +async function getDispatcher(proxyUrl, insecure = false) { const normalized = normalizeProxyUrl(proxyUrl); - if (!normalized) return null; + if (!normalized && !insecure) return null; - if (!proxyDispatchers.has(normalized)) { + const key = `${normalized || "direct"}::${insecure ? "insecure" : "secure"}`; + if (!proxyDispatchers.has(key)) { // Evict oldest entry if max size reached if (proxyDispatchers.size >= MEMORY_CONFIG.proxyDispatchersMaxSize) { proxyDispatchers.delete(proxyDispatchers.keys().next().value); } - const { ProxyAgent } = await import("undici"); - proxyDispatchers.set(normalized, new ProxyAgent({ uri: normalized })); + const { Agent, ProxyAgent } = await import("undici"); + const connect = insecure ? { rejectUnauthorized: false } : undefined; + const dispatcher = normalized + ? new ProxyAgent({ uri: normalized, ...(insecure ? { requestTls: connect } : {}) }) + : new Agent({ connect }); + proxyDispatchers.set(key, dispatcher); } - return proxyDispatchers.get(normalized); + return proxyDispatchers.get(key); +} + +async function fetchWithTlsFallback(url, options, proxyUrl) { + try { + const dispatcher = proxyUrl ? await getDispatcher(proxyUrl) : undefined; + return await originalFetch(url, dispatcher ? { ...options, dispatcher } : options); + } catch (err) { + const isStrictSsl = process.env.STRICT_SSL === "true" || process.env.STRICT_SSL === "1"; + if (!isStrictSsl && isTlsCertError(err)) { + if (options.body && typeof options.body.getReader === "function" && options.body.locked) { + throw err; + } + // ponytail: in-memory insecure agent fallback for self-signed MITM corporate/antivirus certs + console.warn(`[ProxyFetch] TLS cert verification failed (${err.cause?.code || err.code}), retrying with insecure TLS: ${url}`); + const insecureDispatcher = await getDispatcher(proxyUrl, true); + return await originalFetch(url, { ...options, dispatcher: insecureDispatcher }); + } + throw err; + } } /** @@ -318,8 +356,7 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) { if (proxyUrl) { // Proxy resolves DNS externally (not affected by /etc/hosts) — use proxy directly try { - const dispatcher = await getDispatcher(proxyUrl); - return await originalFetch(url, { ...options, dispatcher }); + return await fetchWithTlsFallback(url, options, proxyUrl); } catch (proxyError) { if (proxyOptions?.strictProxy === true) { throw new Error(`[ProxyFetch] Proxy required but failed (strictProxy=true): ${proxyError.message}`); @@ -339,15 +376,14 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) { if (proxyUrl) { try { - const dispatcher = await getDispatcher(proxyUrl); - return await originalFetch(url, { ...options, dispatcher }); + return await fetchWithTlsFallback(url, options, proxyUrl); } catch (proxyError) { // If strictProxy is enabled, fail hard instead of falling back to direct if (proxyOptions?.strictProxy === true) { throw new Error(`[ProxyFetch] Proxy required but failed (strictProxy=true): ${proxyError.message}`); } console.warn(`[ProxyFetch] Proxy failed, falling back to direct: ${proxyError.message}`); - return originalFetch(url, options); + return fetchWithTlsFallback(url, options, null); } } @@ -371,7 +407,7 @@ export async function proxyAwareFetch(url, options = {}, proxyOptions = null) { // got-scraping disabled — use native fetch directly // (Re-enable per-host by wrapping with tryGotScrapingFetch when needed) - return originalFetch(url, options); + return fetchWithTlsFallback(url, options, null); } /** From 2afa8dafd19a8c93568bd5d6b275e1c093e20698 Mon Sep 17 00:00:00 2001 From: decolua Date: Mon, 28 Sep 2026 18:56:07 +0700 Subject: [PATCH 19/41] feat(dashboard): drop NEW badges in sidebar, mark 9Remote as HOT Remove NEW tags from the Media Providers accordion and its sub-items; relabel the 9Remote entry tag NEW -> HOT with an orange accent. Co-Authored-By: Claude Code --- src/shared/components/Sidebar.js | 10 ++-------- 1 file changed, 2 insertions(+), 8 deletions(-) diff --git a/src/shared/components/Sidebar.js b/src/shared/components/Sidebar.js index b3aa65a7..edff69e3 100644 --- a/src/shared/components/Sidebar.js +++ b/src/shared/components/Sidebar.js @@ -203,9 +203,6 @@ export default function Sidebar({ onClose }) { > perm_media Media Providers - {MEDIA_PROVIDER_KINDS.some((k) => VISIBLE_MEDIA_KINDS.includes(k.id) && k.isNew) && ( - NEW - )} expand_more @@ -226,9 +223,6 @@ export default function Sidebar({ onClose }) { > {kind.icon} {kind.label} - {kind.isNew && ( - NEW - )} ))} 9Remote - - NEW + + HOT From 92c7bdd5bc1ff79fc226b35a38b391a6535a318f Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 19:07:28 +0700 Subject: [PATCH 20/41] fix(codex): add GPT-6 Sol/Luna capabilities and official pricing Add explicit capability entries for gpt-6-sol and gpt-6-luna (vision, reasoning, search, 272k context, 128k max output) and update GPT-6 pricing to OpenAI's official standard short-context rates. --- open-sse/providers/capabilities.js | 2 ++ open-sse/providers/pricing.js | 6 +++++- tests/unit/codex-gpt6-lite.test.js | 10 ++++++++++ 3 files changed, 17 insertions(+), 1 deletion(-) diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 5b52516d..deda5cfb 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -183,6 +183,8 @@ export const PROVIDER_CAPABILITIES = { }, "codex": { "gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, + "gpt-6-sol": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, + "gpt-6-luna": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, "gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS, diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index 0cdd6844..09d896a5 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -73,7 +73,11 @@ export const MODEL_PRICING = { "gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 }, "gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 }, "gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, - "gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, + // OpenAI Standard short-context pricing (developers.openai.com/api/docs/pricing). + // Long-context pricing is higher, but this table currently stores one rate per model. + "gpt-6-astra": { input: 10.00, output: 50.00, cached: 1.00, reasoning: 50.00, cache_creation: 12.50 }, + "gpt-6-sol": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 }, + "gpt-6-luna": { input: 0.10, output: 0.50, cached: 0.01, reasoning: 0.50, cache_creation: 0.125 }, "o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 }, "o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 }, diff --git a/tests/unit/codex-gpt6-lite.test.js b/tests/unit/codex-gpt6-lite.test.js index aaf8d3f2..25182652 100644 --- a/tests/unit/codex-gpt6-lite.test.js +++ b/tests/unit/codex-gpt6-lite.test.js @@ -3,6 +3,7 @@ import { afterEach, describe, expect, it, vi } from "vitest"; import { CodexExecutor } from "../../open-sse/executors/codex.js"; import { getModelsByProviderId } from "../../open-sse/config/providerModels.js"; import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { getPricingForModel } from "../../open-sse/providers/pricing.js"; import { getThinkingLevels } from "../../open-sse/providers/thinkingLevels.js"; import * as proxyFetchModule from "../../open-sse/utils/proxyFetch.js"; @@ -17,12 +18,21 @@ describe("Codex GPT-6 Sol/Luna transport", () => { expect(getCapabilitiesForModel("codex", model)).toMatchObject({ vision: true, reasoning: true, + search: true, thinkingFormat: "openai", + contextWindow: 272000, + maxOutput: 128000, }); expect(getThinkingLevels("codex", model)).toEqual(["low", "medium", "high", "xhigh", "max"]); expect(getThinkingLevels("codex", `${model}(high)`)).toEqual(entry.thinkingLevels); }); + it("uses official OpenAI Standard pricing for GPT-6", () => { + expect(getPricingForModel("codex", "gpt-6-astra")).toMatchObject({ input: 10, cached: 1, cache_creation: 12.5, output: 50 }); + expect(getPricingForModel("codex", "gpt-6-sol")).toMatchObject({ input: 2, cached: 0.2, cache_creation: 2.5, output: 10 }); + expect(getPricingForModel("codex", "gpt-6-luna")).toMatchObject({ input: 0.1, cached: 0.01, cache_creation: 0.125, output: 0.5 }); + }); + it("keeps a native Responses Lite request intact", () => { const executor = new CodexExecutor(); const input = [ From 65826d95f2f32cfe6411b2242435dd1e9bd413dd Mon Sep 17 00:00:00 2001 From: semihisikman Date: Mon, 28 Sep 2026 19:07:35 +0700 Subject: [PATCH 21/41] feat(quota): sync ?provider= URL param with provider filter for bookmarkable deep links (#4395) --- .../usage/components/ProviderLimits/index.js | 25 ++++++++- tests/unit/quota-tracker-url-param.test.js | 53 +++++++++++++++++++ 2 files changed, 77 insertions(+), 1 deletion(-) create mode 100644 tests/unit/quota-tracker-url-param.test.js diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js index 0090b7c6..45785c6f 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js @@ -1,6 +1,7 @@ "use client"; import { useState, useEffect, useCallback, useRef, useMemo } from "react"; +import { useSearchParams, useRouter, usePathname } from "next/navigation"; import ProviderIcon from "@/shared/components/ProviderIcon"; import QuotaTable from "./QuotaTable"; import Toggle from "@/shared/components/Toggle"; @@ -152,6 +153,9 @@ function formatTimeRemaining(value) { export default function ProviderLimits() { const { copied, copy } = useCopyToClipboard(); + const searchParams = useSearchParams(); + const router = useRouter(); + const pathname = usePathname(); const [connections, setConnections] = useState([]); const [quotaData, setQuotaData] = useState({}); const [loading, setLoading] = useState({}); @@ -171,7 +175,26 @@ export default function ProviderLimits() { const [showEditModal, setShowEditModal] = useState(false); const [selectedConnection, setSelectedConnection] = useState(null); const [proxyPools, setProxyPools] = useState([]); - const [providerFilter, setProviderFilter] = useState("all"); + // Initialize providerFilter from URL ?provider= param so the page + // can be bookmarked / deep-linked to a specific provider (#4217). + const [providerFilter, _setProviderFilter] = useState( + () => searchParams?.get("provider") || "all" + ); + // Wrapper: keeps URL in sync with the selected provider so the view can be + // bookmarked. Replaces the URL without adding to browser history. + const setProviderFilter = useCallback((value) => { + _setProviderFilter(value); + try { + const params = new URLSearchParams(searchParams?.toString() || ""); + if (value === "all") { + params.delete("provider"); + } else { + params.set("provider", value); + } + const newUrl = params.toString() ? `${pathname}?${params.toString()}` : pathname; + router.replace(newUrl, { scroll: false }); + } catch { /* non-fatal: URL sync is best-effort */ } + }, [pathname, router, searchParams]); const [providerOptions, setProviderOptions] = useState([]); const [accountFilter, setAccountFilter] = useState("all"); const [quotaSortMode, setQuotaSortMode] = useState("default"); diff --git a/tests/unit/quota-tracker-url-param.test.js b/tests/unit/quota-tracker-url-param.test.js new file mode 100644 index 00000000..9baa5605 --- /dev/null +++ b/tests/unit/quota-tracker-url-param.test.js @@ -0,0 +1,53 @@ +/** + * Tests for #4217 — Quota Tracker URL ?provider= parameter support. + * + * Before this fix, visiting /dashboard/quota?provider=codex ignored the query + * parameter and always defaulted to "All providers". The dropdown selection + * updated only local state and did not update the URL. + * + * Fix: ProviderLimits initializes providerFilter from useSearchParams and + * wraps setProviderFilter to call router.replace when the filter changes. + * + * Because ProviderLimits is a JSX file we cannot import it in Vitest without + * a full Next.js setup. We verify the source text instead. + */ + +import { describe, it, expect } from "vitest"; +import fs from "fs"; +import path from "path"; + +const src = fs.readFileSync( + path.resolve("../src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js"), + "utf-8" +); + +describe("ProviderLimits — URL ?provider= sync (#4217)", () => { + it("imports useSearchParams from next/navigation", () => { + expect(src).toContain("useSearchParams"); + expect(src).toContain("next/navigation"); + }); + + it("imports useRouter from next/navigation", () => { + expect(src).toContain("useRouter"); + }); + + it("imports usePathname from next/navigation", () => { + expect(src).toContain("usePathname"); + }); + + it("initializes providerFilter from searchParams.get('provider')", () => { + expect(src).toMatch(/searchParams.*get.*provider/); + }); + + it("calls router.replace when provider filter changes", () => { + expect(src).toContain("router.replace"); + }); + + it("removes ?provider from URL when 'all' is selected", () => { + expect(src).toContain("params.delete"); + }); + + it("sets ?provider in URL when a specific provider is selected", () => { + expect(src).toContain('params.set("provider"'); + }); +}); \ No newline at end of file From 9f41ee754bbb367058d15a27669dab279f485ddc Mon Sep 17 00:00:00 2001 From: Tasa <93864300+tasarren@users.noreply.github.com> Date: Mon, 28 Sep 2026 19:09:12 +0700 Subject: [PATCH 22/41] feat(codex): expose 1M context variants for GPT-6 and GPT-5.6 Add [1m] extended-context variants for gpt-6-astra/sol/luna and gpt-5.6-sol/terra/luna mapping to their base upstream model IDs with an 872k-token context window, share the codex capability table with the cx alias, align model discovery with the inference client version, respect per-connection enabledModels during account selection, and fall back to another account when Codex reports a model unsupported for a ChatGPT account. --- open-sse/config/errorConfig.js | 1 + open-sse/providers/capabilities.js | 8 ++++++++ open-sse/providers/registry/codex.js | 6 ++++++ open-sse/services/accountFallback.js | 3 ++- src/app/api/providers/[id]/models/route.js | 9 +++------ src/sse/handlers/chat.js | 6 +++--- src/sse/services/auth.js | 5 ++++- tests/unit/codex-extended-context.mjs | 19 +++++++++++++++++++ 8 files changed, 46 insertions(+), 11 deletions(-) create mode 100644 tests/unit/codex-extended-context.mjs diff --git a/open-sse/config/errorConfig.js b/open-sse/config/errorConfig.js index 71491a4d..0197da75 100644 --- a/open-sse/config/errorConfig.js +++ b/open-sse/config/errorConfig.js @@ -58,6 +58,7 @@ const COOLDOWN = { */ export const ERROR_RULES = [ // --- Text-based rules (checked first, order = priority) --- + { provider: "codex", text: "model is not supported when using codex with a chatgpt account", cooldownMs: MAX_RATE_LIMIT_COOLDOWN_MS }, { text: "no credentials", cooldownMs: COOLDOWN.long }, { text: "request not allowed", cooldownMs: COOLDOWN.short }, { text: "improperly formed request", cooldownMs: COOLDOWN.long }, diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index deda5cfb..1e4240c8 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -161,6 +161,7 @@ const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, // (lower than OpenAI API's 1.05M). Sol differs from Terra/Luna. #2720 const CODEX_GPT_56_SOL_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 372000, maxOutput: 128000 }; const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; +const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 }; /** * Provider-specific capability overrides. Keyed by provider alias/id. @@ -185,6 +186,12 @@ export const PROVIDER_CAPABILITIES = { "gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, "gpt-6-sol": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, "gpt-6-luna": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, + "gpt-6-astra[1m]": CODEX_EXTENDED_CAPS, + "gpt-6-sol[1m]": CODEX_EXTENDED_CAPS, + "gpt-6-luna[1m]": CODEX_EXTENDED_CAPS, + "gpt-5.6-sol[1m]": CODEX_EXTENDED_CAPS, + "gpt-5.6-terra[1m]": CODEX_EXTENDED_CAPS, + "gpt-5.6-luna[1m]": CODEX_EXTENDED_CAPS, "gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS, @@ -268,6 +275,7 @@ export const PROVIDER_CAPABILITIES = { // Qoder CN serves the identical model catalog from the CN gateway, so it shares // the intl Qoder capability table verbatim (vision/reasoning/contextWindow). PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"]; +PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex; /** * Pattern fallback — glob (* = wildcard), matched case-insensitively and diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index bfa30e3e..c468aefc 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -53,13 +53,19 @@ export default { }, models: [ { id: "gpt-6-astra", name: "GPT 6.0 Astra" }, + { id: "gpt-6-astra[1m]", name: "GPT 6.0 Astra (extended context)", upstreamModelId: "gpt-6-astra" }, { id: "gpt-6-sol", name: "GPT 6.0 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, + { id: "gpt-6-sol[1m]", name: "GPT 6.0 Sol (extended context)", upstreamModelId: "gpt-6-sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, { id: "gpt-6-luna", name: "GPT 6.0 Luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, + { id: "gpt-6-luna[1m]", name: "GPT 6.0 Luna (extended context)", upstreamModelId: "gpt-6-luna", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, + { id: "gpt-5.6-sol[1m]", name: "GPT 5.6 Sol (extended context)", upstreamModelId: "gpt-5.6-sol" }, { id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, + { id: "gpt-5.6-terra[1m]", name: "GPT 5.6 Terra (extended context)", upstreamModelId: "gpt-5.6-terra" }, { id: "gpt-5.6-terra-review", name: "GPT 5.6 Terra Review", upstreamModelId: "gpt-5.6-terra", quotaFamily: "review" }, { id: "gpt-5.6-luna", name: "GPT 5.6 Luna" }, + { id: "gpt-5.6-luna[1m]", name: "GPT 5.6 Luna (extended context)", upstreamModelId: "gpt-5.6-luna" }, { id: "gpt-5.6-luna-review", name: "GPT 5.6 Luna Review", upstreamModelId: "gpt-5.6-luna", quotaFamily: "review" }, { id: "gpt-5.5", name: "GPT 5.5" }, { id: "gpt-5.5-review", name: "GPT 5.5 Review", upstreamModelId: "gpt-5.5", quotaFamily: "review" }, diff --git a/open-sse/services/accountFallback.js b/open-sse/services/accountFallback.js index 766b9981..bc329e55 100644 --- a/open-sse/services/accountFallback.js +++ b/open-sse/services/accountFallback.js @@ -20,12 +20,13 @@ export function getQuotaCooldown(backoffLevel = 0) { * @param {number} backoffLevel - Current backoff level for exponential backoff * @returns {{ shouldFallback: boolean, cooldownMs: number, newBackoffLevel?: number }} */ -export function checkFallbackError(status, errorText, backoffLevel = 0) { +export function checkFallbackError(status, errorText, backoffLevel = 0, provider = null) { const lowerError = errorText ? (typeof errorText === "string" ? errorText : JSON.stringify(errorText)).toLowerCase() : ""; for (const rule of ERROR_RULES) { + if (rule.provider && rule.provider !== provider) continue; // Text-based rule: match substring in error message if (rule.text && lowerError && lowerError.includes(rule.text)) { if (rule.backoff) { diff --git a/src/app/api/providers/[id]/models/route.js b/src/app/api/providers/[id]/models/route.js index a2e09501..4e990bca 100644 --- a/src/app/api/providers/[id]/models/route.js +++ b/src/app/api/providers/[id]/models/route.js @@ -13,15 +13,12 @@ import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy"; import { resolveCursorModels } from "open-sse/services/cursorModels.js"; import { resolveZedModels } from "open-sse/shared/zedAuth.js"; import { resolveClineModels, resolveClinepassModels } from "open-sse/services/clinepassModels.js"; +import codexProvider from "open-sse/providers/registry/codex.js"; const GEMINI_CLI_MODELS_URL = "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels"; -// The /codex/models endpoint gates each entry by minimal_client_version against this -// value, and codex CLI's own manifest (openai/codex codex-rs/models-manager/models.json) -// already requires 0.144.0 for its newest models, so a stale client_version here comes -// back 200 with those entries quietly missing instead of erroring. -const CODEX_CLIENT_VERSION = "0.144.6"; -const CODEX_MODELS_URL = `https://chatgpt.com/backend-api/codex/models?client_version=${CODEX_CLIENT_VERSION}`; +// Model discovery must identify as the same Codex CLI version as inference. +const CODEX_MODELS_URL = `https://chatgpt.com/backend-api/codex/models?client_version=${codexProvider.transport.cliVersion}`; const parseOpenAIStyleModels = (data) => { if (Array.isArray(data)) return data; diff --git a/src/sse/handlers/chat.js b/src/sse/handlers/chat.js index 172549ea..f7070f9a 100644 --- a/src/sse/handlers/chat.js +++ b/src/sse/handlers/chat.js @@ -158,13 +158,13 @@ export async function handleChat(request, clientRawRequest = null) { }); } - return handleSingleModelChat(body, modelStr, clientRawRequest, request, apiKey); + return handleSingleModelChat(body, modelStr, clientRawRequest, request, apiKey, contextMarker ? `${modelStr.slice(modelStr.indexOf("/") + 1)}[${contextMarker}]` : null); } /** * Handle single model chat request */ -async function handleSingleModelChat(body, modelStr, clientRawRequest = null, request = null, apiKey = null) { +async function handleSingleModelChat(body, modelStr, clientRawRequest = null, request = null, apiKey = null, requestedModel = null) { const modelInfo = await getModelInfo(modelStr); // If provider is null, this might be a combo name - check and handle @@ -233,7 +233,7 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re let lastHeaders = null; while (true) { - const credentials = await getProviderCredentials(provider, excludeConnectionIds, model); + const credentials = await getProviderCredentials(provider, excludeConnectionIds, model, { requestedModel: requestedModel || model }); // All accounts unavailable if (!credentials || credentials.allRateLimited) { diff --git a/src/sse/services/auth.js b/src/sse/services/auth.js index eff83e7c..cbe0a737 100644 --- a/src/sse/services/auth.js +++ b/src/sse/services/auth.js @@ -31,6 +31,7 @@ export async function getProviderCredentials(provider, excludeConnectionIds = nu ? excludeConnectionIds : (excludeConnectionIds ? new Set([excludeConnectionIds]) : new Set()); const preferredConnectionId = options?.preferredConnectionId || null; + const requestedModel = options?.requestedModel || model; // Acquire mutex to prevent race conditions const currentMutex = selectionMutex; let resolveMutex; @@ -85,6 +86,8 @@ export async function getProviderCredentials(provider, excludeConnectionIds = nu const availableConnections = connections.filter(c => { if (excludeSet.has(c.id)) return false; if (isModelLockActive(c, model)) return false; + const enabled = c.providerSpecificData?.enabledModels; + if (providerId === "codex" && Array.isArray(enabled) && enabled.length && requestedModel && !enabled.includes(requestedModel)) return false; // Antigravity: skip if live quota exhausted for this model if (isAntigravity && model && antigravityQuotaCache) { const quota = antigravityQuotaCache.get(c.id)?.[model]; @@ -259,7 +262,7 @@ export async function markAccountUnavailable(connectionId, status, errorText, pr : Math.min(resetsAtMs - Date.now(), MAX_RATE_LIMIT_COOLDOWN_MS); newBackoffLevel = 0; } else { - ({ shouldFallback, cooldownMs, newBackoffLevel } = checkFallbackError(status, errorText, backoffLevel)); + ({ shouldFallback, cooldownMs, newBackoffLevel } = checkFallbackError(status, errorText, backoffLevel, resolveProviderId(provider))); } if (!shouldFallback) return { shouldFallback: false, cooldownMs: 0 }; diff --git a/tests/unit/codex-extended-context.mjs b/tests/unit/codex-extended-context.mjs new file mode 100644 index 00000000..9af970db --- /dev/null +++ b/tests/unit/codex-extended-context.mjs @@ -0,0 +1,19 @@ +import assert from "node:assert/strict"; +import codex from "../../open-sse/providers/registry/codex.js"; +import { getModelUpstreamId } from "../../open-sse/config/providerModels.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { stripModelContextMarker } from "../../open-sse/utils/modelMarkers.js"; +import { checkFallbackError } from "../../open-sse/services/accountFallback.js"; + +for (const id of ["gpt-6-astra", "gpt-6-sol", "gpt-6-luna", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"]) { + const extended = `${id}[1m]`; + assert.equal(codex.models.find((model) => model.id === extended)?.upstreamModelId, id); + assert.equal(getModelUpstreamId("cx", extended), id); + assert.equal(getCapabilitiesForModel("codex", extended).contextWindow, 872000); + assert.equal(getCapabilitiesForModel("cx", extended).contextWindow, 872000); + assert.deepEqual(stripModelContextMarker(`cx/${extended}`), { model: `cx/${id}`, contextMarker: "1m" }); +} +const unsupported = "The 'gpt-6-sol' model is not supported when using Codex with a ChatGPT account."; +assert.equal(checkFallbackError(400, unsupported, 0, "codex").shouldFallback, true); +assert.equal(checkFallbackError(400, "Invalid JSON body", 0, "codex").shouldFallback, false); +console.log("Codex extended models and account fallback OK"); From 0bc7f86e4b4e9a20434373383fe7d78cf1db157c Mon Sep 17 00:00:00 2001 From: decolua Date: Mon, 28 Sep 2026 19:30:14 +0700 Subject: [PATCH 23/41] fix(codex): stop refresh-token reuse that logs accounts out on auto-ping MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit OpenAI rotates the refresh token on every refresh and revokes the whole session on reuse. A 5-day refreshLeadMs (access tokens live ~1h) rotated the token on every call, and three refresh writers (usage poll, auto-ping tick, 5-min background refresher) each held stale snapshots — auto-ping firing at reset time reliably triggered reuse and logged the account out. - registry: refreshLeadMs 5d -> 10min (refresh only near actual expiry) - refreshAndUpdateCredentials: re-read connection from DB before refresh; throw on unrecoverable refresh instead of continuing with a dead token - checkAndRefreshToken: adopt newer DB tokens before refreshing Co-Authored-By: Claude Code --- open-sse/providers/registry/codex.js | 4 +++- src/app/api/usage/[connectionId]/route.js | 11 +++++++++++ src/sse/services/tokenRefresh.js | 21 ++++++++++++++++++++- 3 files changed, 34 insertions(+), 2 deletions(-) diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index c468aefc..f01153db 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -103,7 +103,9 @@ export default { codex_cli_simplified_flow: "true", originator: "codex_cli_rs", }, - refreshLeadMs: 432000000, + // Access tokens live ~1h; a 5d lead rotated the refresh token on EVERY call — + // reuse of a rotated token revokes the whole OpenAI session (account logout). + refreshLeadMs: 600000, refresh: { encoding: "form", scope: "openid profile email offline_access", diff --git a/src/app/api/usage/[connectionId]/route.js b/src/app/api/usage/[connectionId]/route.js index 84dd7674..3ca9840f 100644 --- a/src/app/api/usage/[connectionId]/route.js +++ b/src/app/api/usage/[connectionId]/route.js @@ -3,6 +3,7 @@ import "open-sse/index.js"; import { getProviderConnectionById, updateProviderConnection } from "@/lib/localDb"; import { getUsageForProvider } from "open-sse/services/usage.js"; +import { isUnrecoverableRefreshError } from "open-sse/services/tokenRefresh.js"; import { getExecutor } from "open-sse/executors/index.js"; import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy"; import { USAGE_APIKEY_PROVIDERS } from "@/shared/constants/providers"; @@ -21,6 +22,11 @@ function isAuthExpiredMessage(usage) { * @returns Promise<{ connection, refreshed: boolean }> */ export async function refreshAndUpdateCredentials(connection, force = false, proxyOptions = null) { + // Re-read latest tokens: OpenAI rotates the refresh token on every refresh, and + // refreshing with a stale snapshot (reuse) revokes the whole session → account logout. + const latest = connection.id ? await getProviderConnectionById(connection.id) : null; + if (latest) connection = latest; + const executor = getExecutor(connection.provider); // Build credentials object from connection @@ -47,6 +53,11 @@ export async function refreshAndUpdateCredentials(connection, force = false, pro // Use executor's refreshCredentials method (with optional proxy) const refreshResult = await executor.refreshCredentials(credentials, console, proxyOptions); + // Refresh token reused/invalidated — token family is revoked; do not continue with the dead token. + if (refreshResult && isUnrecoverableRefreshError(refreshResult)) { + throw new Error("Refresh token invalid or reused. Please re-authorize the connection."); + } + if (!refreshResult) { // Refresh failed but we still have an accessToken — try with existing token if (connection.accessToken) { diff --git a/src/sse/services/tokenRefresh.js b/src/sse/services/tokenRefresh.js index 58a6f870..b253df5c 100644 --- a/src/sse/services/tokenRefresh.js +++ b/src/sse/services/tokenRefresh.js @@ -1,6 +1,6 @@ // Re-export from open-sse with local logger import * as log from "../utils/logger.js"; -import { updateProviderConnection } from "../../lib/localDb.js"; +import { getProviderConnectionById, updateProviderConnection } from "../../lib/localDb.js"; import { getProjectIdForConnection, invalidateProjectId, @@ -227,6 +227,25 @@ export async function checkAndRefreshToken(provider, credentials, options = {}) creds.connectionId = creds.id; } + // Adopt latest DB tokens: OpenAI rotates the refresh token on every refresh, and + // refreshing with a stale snapshot (reuse) revokes the whole session → account logout. + if (creds.connectionId) { + const latest = await getProviderConnectionById(creds.connectionId).catch(() => null); + const latestRefreshMs = Date.parse(latest?.lastRefreshAt || ""); + const credsRefreshMs = Date.parse(creds.lastRefreshAt || ""); + const dbIsNewer = Number.isFinite(latestRefreshMs) + && (!Number.isFinite(credsRefreshMs) || latestRefreshMs > credsRefreshMs); + if (dbIsNewer && latest.refreshToken && latest.refreshToken !== creds.refreshToken) { + creds = { + ...creds, + refreshToken: latest.refreshToken, + accessToken: latest.accessToken || creds.accessToken, + expiresAt: latest.expiresAt || latest.tokenExpiresAt || creds.expiresAt, + lastRefreshAt: latest.lastRefreshAt || creds.lastRefreshAt, + }; + } + } + const force = options?.force === true; // ── 1. Regular access-token expiry ──────────────────────────────────────── From 2310ad4b14cd0fab45142cf776684c037d073651 Mon Sep 17 00:00:00 2001 From: Nikan Wystaf Date: Mon, 28 Sep 2026 19:31:05 +0700 Subject: [PATCH 24/41] feat(agnes): seed the 2.5/3.0 model ids in the registry --- open-sse/providers/registry/agnes.js | 13 +++++++-- tests/unit/agnes-model-seeds.test.js | 40 ++++++++++++++++++++++++++++ 2 files changed, 51 insertions(+), 2 deletions(-) create mode 100644 tests/unit/agnes-model-seeds.test.js diff --git a/open-sse/providers/registry/agnes.js b/open-sse/providers/registry/agnes.js index 4cf2a88e..633d67c6 100644 --- a/open-sse/providers/registry/agnes.js +++ b/open-sse/providers/registry/agnes.js @@ -23,7 +23,16 @@ export default { baseUrl: "https://apihub.agnes-ai.com/v1/chat/completions", validateUrl: "https://apihub.agnes-ai.com/v1/models", }, - // No model ids could be verified without a key, so discovery is left to the - // live endpoint and any id is accepted through passthroughModels. + // Seeds so the dashboard has something to show before a key is saved and + // /v1/models answers with a usable list. The live catalogue at + // apihub.agnes-ai.com/v1/models requires a token (401 "Token not provided"), + // so these ids are a curated starting set rather than a verified dump — + // passthroughModels below still accepts any id the account actually has. + models: [ + { id: "agnes-2.5-flash", name: "Agnes 2.5 Flash" }, + { id: "agnes-2.5-pro", name: "Agnes 2.5 Pro" }, + { id: "agnes-2.5-pro-beta", name: "Agnes 2.5 Pro Beta" }, + { id: "agnes-3.0-flash", name: "Agnes 3.0 Flash" }, + ], passthroughModels: true, }; diff --git a/tests/unit/agnes-model-seeds.test.js b/tests/unit/agnes-model-seeds.test.js new file mode 100644 index 00000000..8a8a0e21 --- /dev/null +++ b/tests/unit/agnes-model-seeds.test.js @@ -0,0 +1,40 @@ +import { describe, expect, it } from "vitest"; + +import agnes from "../../open-sse/providers/registry/agnes.js"; + +// Seeds for Agnes AI. The live catalogue (apihub.agnes-ai.com/v1/models) +// answers 401 without a token, so these ids are a curated starting set and +// passthroughModels still accepts anything the account actually has. + +describe("agnes registry model seeds", () => { + it("declares the four 2.5/3.0 models", () => { + expect(agnes.models.map((m) => m.id)).toEqual([ + "agnes-2.5-flash", + "agnes-2.5-pro", + "agnes-2.5-pro-beta", + "agnes-3.0-flash", + ]); + }); + + it("gives every model a display name", () => { + for (const m of agnes.models) { + expect(typeof m.name, m.id).toBe("string"); + expect(m.name.length, m.id).toBeGreaterThan(0); + } + }); + + it("keeps passthroughModels so newer ids still work", () => { + expect(agnes.passthroughModels).toBe(true); + }); + + it("does not claim a modelsFetcher for an endpoint that needs a token", () => { + // /v1/models returns 401 "Token not provided" without auth, so a public + // modelsFetcher would just surface an error instead of a list. + expect(agnes.modelsFetcher).toBeUndefined(); + }); + + it("leaves transport untouched", () => { + expect(agnes.transport.baseUrl).toBe("https://apihub.agnes-ai.com/v1/chat/completions"); + expect(agnes.transport.validateUrl).toBe("https://apihub.agnes-ai.com/v1/models"); + }); +}); From 288098070906f02d3ce808855514e96c9697a447 Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 1 Oct 2026 09:56:27 +0700 Subject: [PATCH 25/41] feat(muse): add Meta Muse provider with OAuth login and model catalog --- open-sse/executors/default.js | 3 + open-sse/providers/pricing.js | 9 ++ open-sse/providers/registry/index.js | 2 + open-sse/providers/registry/muse.js | 67 +++++++++ public/providers/muse.png | Bin 0 -> 5928 bytes .../api/oauth/[provider]/[action]/route.js | 5 +- src/app/api/providers/[id]/models/route.js | 8 ++ src/app/api/providers/[id]/test/testUtils.js | 12 +- src/lib/oauth/constants/oauth.js | 4 + src/lib/oauth/providers/index.js | 12 +- src/lib/oauth/providers/muse.js | 132 ++++++++++++++++++ src/shared/components/OAuthModal.js | 3 +- tests/__baseline__/alias-baseline.json | 11 +- tests/__baseline__/providers-baseline.json | 63 +++++++-- tests/__baseline__/verify-alias.mjs | 2 + 15 files changed, 319 insertions(+), 14 deletions(-) create mode 100644 open-sse/providers/registry/muse.js create mode 100644 public/providers/muse.png create mode 100644 src/lib/oauth/providers/muse.js diff --git a/open-sse/executors/default.js b/open-sse/executors/default.js index e5c59741..dbd42cfc 100644 --- a/open-sse/executors/default.js +++ b/open-sse/executors/default.js @@ -41,6 +41,9 @@ function applyAuth(headers, desc, credentials) { const HEADER_HOOKS = { // Stable device_id from OAuth connection (CLIProxyAPI KimiTokenStorage.DeviceID) kimiHeaders: (h, c) => Object.assign(h, buildKimiHeaders(c?.providerSpecificData?.deviceId)), + // Muse: x-api-version only on subscription (minted key) requests — plain + // Model API keys already work without it + museHeaders: (h, c) => { if (c?.accessToken && !c?.apiKey) h["x-api-version"] = "1.0.0"; }, clineHeaders: (h, c) => Object.assign(h, buildClineHeaders(c.apiKey || c.accessToken)), kilocodeOrg: (h, c) => { if (c.providerSpecificData?.orgId) h["X-Kilocode-OrganizationID"] = c.providerSpecificData.orgId; }, }; diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index 09d896a5..a57e6bda 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -156,6 +156,15 @@ export const MODEL_PRICING = { // === Grok === "grok-code-fast-1": { input: 0.50, output: 2.00, cached: 0.25, reasoning: 3.00, cache_creation: 0.50 }, + // === Muse (Meta Model API) === + // Rates from https://dev.meta.ai/docs/pricing-rate-limits (contributor tier: + // cheaper, Meta may train on the data). + "muse-spark-1.3": { input: 1.25, output: 4.25, cached: 0.15, reasoning: 4.25, cache_creation: 0 }, + "muse-spark-1.2": { input: 1.25, output: 4.25, cached: 0.15, reasoning: 4.25, cache_creation: 0 }, + "muse-spark-1.1": { input: 1.25, output: 4.25, cached: 0.15, reasoning: 4.25, cache_creation: 0 }, + "muse-spark-1.3-contributor": { input: 0.10, output: 0.20, cached: 0.002, reasoning: 0.20, cache_creation: 0 }, + "muse-spark-1.2-contributor": { input: 0.10, output: 0.20, cached: 0.002, reasoning: 0.20, cache_creation: 0 }, + // === OpenRouter fallback === "auto": { input: 2.00, output: 8.00, cached: 1.00, reasoning: 12.00, cache_creation: 2.00 }, diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 17d1fea9..6c317ee8 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -132,6 +132,7 @@ import p129 from "./agnes.js"; import p130 from "./bai.js"; import p131 from "./tinyfish.js"; import p132 from "./v1m.js"; +import p133 from "./muse.js"; export default [ p0, p1, @@ -264,4 +265,5 @@ export default [ p130, p131, p132, + p133, ]; diff --git a/open-sse/providers/registry/muse.js b/open-sse/providers/registry/muse.js new file mode 100644 index 00000000..4903980e --- /dev/null +++ b/open-sse/providers/registry/muse.js @@ -0,0 +1,67 @@ +// Muse (Meta Model API) — dual auth (same pattern as kimi): +// oauth = Muse Code subscription (Meta account device code, mints an LLM|… key) +// apikey = pay-as-you-go Model API key from dev.meta.ai +// Transport is shared; oauth accounts get x-api-version via the museHeaders hook. +// Meta issues no refresh token for the subscription flow → re-login on 401. +export default { + id: "muse", + priority: 120, + alias: "muse", + aliases: [ + "muse-ai", + "meta-model-api", + "muse-code", + "muse-subscription", + ], + uiAlias: "muse", + display: { + name: "Muse (Meta Model API)", + icon: "auto_awesome", + color: "#0866FF", + textIcon: "MU", + website: "https://muse.ai", + notice: { + text: "Sign in with your Meta account (Muse Code subscription) or paste a Model API key from dev.meta.ai. Subscription keys are minted per account; Meta may train on contributor-tier data.", + apiKeyUrl: "https://dev.meta.ai", + signupUrl: "https://muse.ai", + }, + }, + category: "oauth", + authModes: ["oauth", "apikey"], + hasOAuth: true, + transport: { + baseUrl: "https://api.meta.ai/v1/chat/completions", + validateUrl: "https://api.meta.ai/v1/models", + modelsUrl: "https://api.meta.ai/v1/models", + auth: { combined: true, header: "Authorization", scheme: "bearer", hooks: ["museHeaders"] }, + }, + // Multi-endpoint: Meta accepts Chat Completions and Responses wire formats on + // the same key (https://dev.meta.ai/docs/protocols). Muse Spark reasoning + // (incl. encrypted_content replay) only round-trips on Responses, so models + // pin targetFormat there. + transports: [ + { + format: "openai", + baseUrl: "https://api.meta.ai/v1/chat/completions", + auth: { combined: true, header: "Authorization", scheme: "bearer", hooks: ["museHeaders"] }, + }, + { + format: "openai-responses", + baseUrl: "https://api.meta.ai/v1/responses", + auth: { combined: true, header: "Authorization", scheme: "bearer", hooks: ["museHeaders"] }, + }, + ], + models: [ + { id: "muse-spark-1.3", name: "Muse Spark 1.3", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "muse-spark-1.2", name: "Muse Spark 1.2", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "muse-spark-1.1", name: "Muse Spark 1.1", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + ], + passthroughModels: true, + oauth: { + clientId: "1031625952748946", + deviceCodeUrl: "https://auth.meta.com/oidc/device/authorization/", + tokenUrl: "https://auth.meta.com/oidc/device/token/", + }, +}; diff --git a/public/providers/muse.png b/public/providers/muse.png new file mode 100644 index 0000000000000000000000000000000000000000..560e064eb21876f0298a8b122af2d79169158626 GIT binary patch literal 5928 zcma)A_cz-Q)c(X)GgYIASVh(LwT0NDgxFM#s#$x~s2WAB5PQ{Dqo@(PloYjVme!~h zvn{bpW4wLezu^7hjvwwl=bm%#dCqg6cteC1%`Mhj007YFXsa9jixK~86lDK=pOXD3 z05IWo)KyG^^7jf}WSE`?3@Q@>@zm7R29ay`i7BeM_*|!(_0#kI`=UFiOH`L*DgOSI zE)nR}VI-ZQWGS~GkzjFjxj2$R<97P@=8dU=iqY?Gv$cE`f+S=qZE><)&Rc~)l=qij zOb~qVccb<$>PY|OzZ}2V!(Z&ZJoCB<+YSsPzAN!`D1}SyHTC}^xuHr?j-PY&_+pq7 zk@soigg)jH8szqSUzsxdW;e4jcki zTp!*x+3y{)Lz_*sF88K0bYCPNj7;{?9^~p@y_5w}y5C^WyX@M^ zM4_IeyDWtoPc!PX&hXoN5#;B?rTL|crztUFMLVF0&xkIL9d?`SYupW1lbYDuQs2I+YlP< z;3}oB1Pqik>Mjtbx__^h=igy4>>@AgjHt ziQo&2aF-U_ib&H_w=e7ndlNh${D>^2DUABWfa~jGtH;2wrB=-rx%c6gIK>}%6uw3h zm0oDf5NpgnD+jM(MU3!praz-QEw?S=pIJ9Fz0?)hFt^(=7g<4TofLodc4|JYiyyp? z`pLf9_Q*^1Jan(5`-JV_>DEza^*XutR8~*l!DsAIS$=b90XSs1xH>q0eTct%{rFL3 zq>Zf~RC(;Yd<8L|o&$xCj-JNVCI~5OZ*?cDJ~mZcCe_{owp$8fe)h%P;n8Y|x3y$) zVuuaAo+m!9QO4Zz3;alz>K7?qB^TL1I&=Ea>|>K?NhxHtOE}$dpSXH&>*{u9cETcMFD6eo zb!kb=iBwxe14o>!_e;SceumS-7z!M*;I1sZW-GUoU}MMAOFlwU5Ntpz z9&Rumk~(^(r=#IT(hFHl5dj5<)4aT7o1a{IZDp)lA2Jl%Qu1xj0u zq%!C=?HK5(t&oVPw*7svcYHWzN8;*fD8o2~n9FZPR9F3negk0Nxi8w8-7!J-1>jQ} zX*C62P4jSeXqU|}jxnd?zs(9&PglU^|A3MdFx*|gl9PATE$aWvP`nNZd-Q>s%qhMP zM-RU}HwciS@r^Q%>6U3!i;o9`{O_woDV2NR(|g2I4?!krG(*kRpBh;8L{dwl9A`h$ zC&nib-#L+ftFY9Mxctpn*>tflyG@eJ^aV7$42@$F1LHv=^VdlAiq|st3KL-Q_mus$ zOHp)ohnfP+GRxm3*)~KAe%>oQyvm=OEW>jp$YFy$oFi>a`0Q_J7M|134q7O$RW47r z>8rmDEPZM;Ki-E!M9;K((iJXmnoj|W)_Ow+GX5!gzK7eq)~#)n53^-seEfjy{pq^d zCR#%jFkp3S1J^b~i|f%T2gF z#}@}=i4eOt#isE_fb`Q(nu+xm70KLZI=R(aYSJr)uJ!?8&Ejy?I%o5XeWc}G%x z&w9KMnuOWwAwWFppDp@{cG8PL%IU0SR@y1ta+2(MVKEW>Jn_ZiUJxw8Ov)lAld)Wf z8gg&+PXlpPXlqrwdW1JFJGq8~G?%h8vOKY?60>Js@&32bMaAkrv~%N_nzI-PEIgs& zV0m+OyI92eRi`ci5jP{AWc*I*k6G3(-=;pvP##kG=VS+n;Q z(*HKB!#tEIy9V2Bw$4%26je^J` z2an@=fTh8YM71DRulNv@G)3aFofP+phHH7t645hU!f_jfA@jhMz<#1!serwwfk<#0vatU3%2ja?&yXP9h95YuDJGPATyDJP+xJN|9Qpj?{yEBof ztCrN18s=rAYcYoxuq*&$ZMw_uds}a!0Q282@Bn*8{NbVt>s1b}eZ;~b?$1u(@CV;( zaIxtz%j12nRq?Q~xJ^Ez<=OygqyMPmd@*IDYeg^;5;)_@qW|WE^N(10hfMijG4dvS zntbW|9=H+E0+pY*Q?wdWrLbv7`sek0B=5p+B)d~6;BcW85k=@pup9Mxg~F#o61_q} zd-EJX#xkj5z?}VRllA1;-HhYN;W#FESgaNL;QX#Aui)uUO8iF45$th3{?N2jHnjPO zHE6$EQyyWB6Ocj3nm(WT^<9kvp$ey+{D^DGPE%M<0-9)P;R;Ly>D=g{l=oPt)tT-t zNaqhWSAWo>78(BKJX9tgZPnTFtQcq7>8Cn`!-bJu);g63fA~g3`;hO4#QezX`;g76 zWQJ@mYCjtWp8V>TB?4FMtq%nMdYc|?L#D1xf{J#1@~_B}ANjCY;^qF5nv1 zs%6j?bjJXHnZ)jOb1RK4Zt6+NH;x6=mgnktm;u~C_)dJla|s-AY35n@_?ZfYxk7!s zv3*E{!|BS=WNwES%d2NwhK7=c$z;owH%A|FNRbDHdAy=i+wIOf4A?l5f%fELl!au>Q_gBzEA93n?53}lv5SI2}--wx71grBK0zCv%|Xs24A2xicvk z1KoRvk_)wo_?a0OGPP2`!AbcdWbRQ~byWA@WOR+5Uy~^it+{aVe`IJMcXY+$?XN|= zXjG{e43P^%*I?J;QMN(~A-P=|TAG}`&o~DSbGyG2)6zFQw{ul!2|6^t&eYG6mYZ{eoN^O9$rh&nt09V{BgdoB_ z=yAlR;4j2<3W@Vk76Qk4YDKU1$@zD`NBhQ=y$mb#B+2P0ryK;53$!in>Gq`1<3)`> zq;yHj9Yd!-LqBg=o7aikO5k~K9QMUwprPY-#HtOv=1kK9xB74G zDqR}insgBrY#?4`N^7q~hjc&p{#4k=RbQK?A4c~@LVlIzve52=i5H?w^u zrZig1?t*k(1H+(QQa@yDB1D zsg*p5T;3{SZo|M2saSBKFD!efuKKWF?UjFY<3y=EZ|cKdaC__1NTRz(_h=q*od$y3 zp53UWp}l=&h4vrG9l_>M$%R>Fifz+ZN28!8+oa$rN{%NNuA+qWD(F58wtB5w&0BS4 z{>fXJZ#XFarL98ow135KVEQA%V74$mTx0RMI#nEj&`DpFO#5Fq+pMOLY=YZ?t%9%x z&%d{BOaTk3^C*fO;{V9*Fjt7$Z4sQ%F{Vy)i$^M#onu^W#@YAn*9DMugdy>`G6CJC&}XEt7F! zJ>;!DwiGbC(D@kuI*Lj=8e!YjU<=}-+hxn_oVFG;CRxc3SsTB8NcgPAE}Ap#7x7VH zFw!Hz-)bPxB*$`M0Swx!rK?+#ik?=d*Pb|kl0Doq1x*=Afvb@ZFX>}@&qQD4Lq`-Y zZ3>-S%lt}(T;<^vz)7ZirgmsW-&gSE!(XO=;QkeJ~0=gzu7alEyQ;SsfrmT z|0duReK#ZsUR{VHZaDEA4Au#%;UGRX4R!Lb%pqA|zzV8$z3el#zl@OT3onXGW71@L z@S%bpibXQ6>GtTJV&nGk(7!*RVN>rUD0nYAM?dCP?`zaK>qOx{JlvIw5bLN0aZskn ziA)j=>Bq}4j_qDE7V!3yADD4Ce|DW1;3@56{}$p4^i;$@xgE5()(2fp6KJr(VbTrpj+Gd9^h^L>FCoU zjzL`QiUz{XK3)SWWp)6YN z6a+)c1p58RIr2-y$U@$Abi$DLcDI)6rMK0_F|?ND{D(UGSB&r}>&&@^r7`EAKIzZYEg+6V?L}&LeI~$_vimlT^AhH2f`|5a53yxOYu#7LS6C zi5pc+99XISuF6G_H27v7p8bci4K?zlE7)a=DIvs?jY}M*YaH7QVpK%$ZP_>X;Qg|; zs-C*4$&vWlZ~PRNemhI$Eh7G=e=>-f?E>mNI5a&#LOk$G%MF2|HghOW{g zs{rh!+s@_-qaW+i7847~m^K}%IHo!qIT~$@ae57%y7wnpBwEN#5}u22xTAY)C0C~* z=384;wNG`5_|wuJ4uaIlNSfb7v$87ZDh=kX6=pLH%S}zR;Q7FkPj(YeAeVNtibBJg z%3qQ*iQD7)eGF$ShyNTFCEbZ^W-ktRTx?41Kfe`Xx~@K&oqndV_d1E7J|EajCEny_ zRdv4-q9l{gNlvkDDSb^_+ncTUptN6I{v)tgxJR!eJ9Z9)#?F?7rQfQ2wejF@jab;B zv#V)*5hQPMGqqgW+z9>ToaiBe|IzgF>kMun)(rtz;wQ6M7c3M=x;ce zU!y;lBIDwDX4H-Lipne+{+BEJQIy+eeKGIILq*K5j`E2DSL8!ir3e8xraNkKykggy zKNYOfP}cWeSp^0d@cB$qYz>9`V7%>rjy5gNwUckuC3L%?>I*FXN_p-wT*+Azy_Mm5 zx;K=>nFyRl2cZv5k!BJb*kCET^XRpGd(jeKzj)RD^c(!fmr6^ix6Vw?OVdFNJ3akW zhaB@9DF&Fg^?{_#tZOsScVNHgVH^%wNGpDI)uGJkfa(buDxDnswS~*9=7;2>s2wpZ=MkA^TSGg(Xp3w3|EnGU f)0exR9rhT;ctfj;zDoU5zyTc%gnFH--K+lruBTcP literal 0 HcmV?d00001 diff --git a/src/app/api/oauth/[provider]/[action]/route.js b/src/app/api/oauth/[provider]/[action]/route.js index a40dd49f..b31a0d16 100644 --- a/src/app/api/oauth/[provider]/[action]/route.js +++ b/src/app/api/oauth/[provider]/[action]/route.js @@ -265,6 +265,7 @@ export async function GET(request, { params }) { "qoder", "qoder-cn", "grok-cli", + "muse", ]; let deviceData; if (noPkceDeviceProviders.includes(provider)) { @@ -546,12 +547,14 @@ export async function POST(request, { params }) { // Still pending or error - don't create connection for pending states const isPending = result.pending || result.error === "authorization_pending" || result.error === "slow_down"; - + return NextResponse.json({ success: false, error: result.error, errorDescription: result.errorDescription, pending: isPending, + // fatal: unrecoverable (e.g. post-exchange failure) — client must stop polling and show it + ...(result.fatal ? { fatal: true } : {}), }); } diff --git a/src/app/api/providers/[id]/models/route.js b/src/app/api/providers/[id]/models/route.js index 4e990bca..37ebce4d 100644 --- a/src/app/api/providers/[id]/models/route.js +++ b/src/app/api/providers/[id]/models/route.js @@ -169,6 +169,14 @@ function buildQoderModelsResolver(providerId) { // Provider models endpoints configuration const PROVIDER_MODELS_CONFIG = { + "muse": { + url: "https://api.meta.ai/v1/models", + method: "GET", + headers: { "Content-Type": "application/json", "x-api-version": "1.0.0" }, + authHeader: "Authorization", + authPrefix: "Bearer ", + parseResponse: (data) => data.data || [], + }, claude: { url: "https://api.anthropic.com/v1/models", method: "GET", diff --git a/src/app/api/providers/[id]/test/testUtils.js b/src/app/api/providers/[id]/test/testUtils.js index ecdac3e9..7f9eedeb 100644 --- a/src/app/api/providers/[id]/test/testUtils.js +++ b/src/app/api/providers/[id]/test/testUtils.js @@ -137,6 +137,15 @@ const OAUTH_TEST_CONFIG = { 402: "Connected, but Grok Build credits are exhausted (spending limit). Add credits or upgrade SuperGrok.", }, }, + // Muse Code subscription — probe /v1/models with the minted LLM|… key + "muse": { + url: "https://api.meta.ai/v1/models", + method: "GET", + authHeader: "Authorization", + authPrefix: "Bearer ", + extraHeaders: { "x-api-version": "1.0.0" }, + refreshable: false, + }, }; /** @@ -703,7 +712,8 @@ async function testApiKeyConnection(connection, effectiveProxy = null) { case "dahl": case "atria": case "agnes": - case "bai": { + case "bai": + case "muse": { const cfg = PROVIDERS[connection.provider]; const res = await fetchWithConnectionProxy(cfg.validateUrl, { headers: { Authorization: `Bearer ${connection.apiKey}` } }, effectiveProxy); return { valid: res.ok, error: res.ok ? null : "Invalid API key" }; diff --git a/src/lib/oauth/constants/oauth.js b/src/lib/oauth/constants/oauth.js index cd35d331..9d9d2a1f 100644 --- a/src/lib/oauth/constants/oauth.js +++ b/src/lib/oauth/constants/oauth.js @@ -127,6 +127,10 @@ export const KIMCHI_CONFIG = { ...PROVIDER_OAUTH["kimchi"] }; // Endpoint: cli-chat-proxy.grok.com — same client_id as xai, different flow + scopes export const GROK_CLI_CONFIG = { ...PROVIDER_OAUTH["grok-cli"] }; +// Muse — subscription device code flow to auth.meta.com, no refresh +// (Meta rejects refresh_token grants; the minted Model API key never expires). +export const MUSE_CONFIG = { ...PROVIDER_OAUTH["muse"] }; + // Trae (ByteDance marscode) OAuth — authorization_code flow with local callback. // 1) POST GetLoginGuidance {loginTraceID} → {Result.LoginHost} // 2) Browser opens ${loginHost}/authorization?client_id=...&login_trace_id=...&auth_callback_url=${cb} diff --git a/src/lib/oauth/providers/index.js b/src/lib/oauth/providers/index.js index 86ed5338..80063950 100644 --- a/src/lib/oauth/providers/index.js +++ b/src/lib/oauth/providers/index.js @@ -8,6 +8,7 @@ import claude from "./claude.js"; import codex from "./codex.js"; import xai from "./xai.js"; import grokCli from "./grok-cli.js"; +import muse from "./muse.js"; import geminiCli from "./gemini-cli.js"; import antigravity from "./antigravity.js"; import iflow from "./iflow.js"; @@ -34,6 +35,7 @@ const PROVIDERS = { codex, xai, "grok-cli": grokCli, + muse, "gemini-cli": geminiCli, antigravity, iflow, @@ -174,7 +176,15 @@ export async function pollForToken(providerName, deviceCode, codeVerifier, extra // Call postExchange to get additional data (copilotToken, userInfo, etc.) let extra = null; if (provider.postExchange) { - extra = await provider.postExchange(result.data); + try { + extra = await provider.postExchange(result.data); + } catch (err) { + // The grant succeeded but post-login exchange failed (e.g. Muse key + // mint). The device code is one-shot, so re-polling can never + // recover — surface as fatal so the client stops and shows the error. + console.warn(`[oauth] ${providerName} postExchange failed:`, err?.message || err); + return { success: false, error: "exchange_failed", errorDescription: err.message, fatal: true }; + } } const tokens = provider.mapTokens(result.data, extra); // Kiro IDC/Builder-ID tokens lack profileArn; resolve it to avoid 403 diff --git a/src/lib/oauth/providers/muse.js b/src/lib/oauth/providers/muse.js new file mode 100644 index 00000000..2d7f2c40 --- /dev/null +++ b/src/lib/oauth/providers/muse.js @@ -0,0 +1,132 @@ +import { MUSE_CONFIG } from "../constants/oauth.js"; + +// Muse Code subscription — Meta account device code flow to auth.meta.com, +// then mint the Model API key (LLM|…) the chat transport actually uses. +const MUSE_KEY_URL = "https://api.meta.ai/muse-code/key"; +const API_VERSION = "1.0.0"; + +const muse = { + config: MUSE_CONFIG, + flowType: "device_code", + requestDeviceCode: async (config) => { + const response = await fetch(config.deviceCodeUrl, { + method: "POST", + headers: { + "Content-Type": "application/x-www-form-urlencoded", + Accept: "application/json", + "x-api-version": API_VERSION, + }, + body: new URLSearchParams({ client_id: config.clientId }), + }); + + if (!response.ok) { + const error = await response.text(); + throw new Error(`Muse Code device code request failed: ${error}`); + } + + return await response.json(); + }, + pollToken: async (config, deviceCode) => { + const response = await fetch(config.tokenUrl, { + method: "POST", + headers: { + "Content-Type": "application/x-www-form-urlencoded", + Accept: "application/json", + "x-api-version": API_VERSION, + }, + body: new URLSearchParams({ + grant_type: "urn:ietf:params:oauth:grant-type:device_code", + device_code: deviceCode, + client_id: config.clientId, + }), + }); + + let data; + try { + data = await response.json(); + } catch { + const text = await response.text(); + data = { error: "invalid_response", error_description: text }; + } + + const pending = + data?.error === "authorization_pending" || + data?.error === "slow_down"; + return { ok: response.ok || pending, data }; + }, + postExchange: async (tokens) => { + // Mint the subscription API key; onboard:true enrolls the account on first + // login. The endpoint is aggressively rate-limited (429) and the device code + // is one-shot, so retry transient failures here instead of failing the login. + let response; + for (let attempt = 0; ; attempt++) { + response = await fetch(MUSE_KEY_URL, { + method: "POST", + headers: { + Accept: "application/json", + Authorization: `Bearer ${tokens.access_token}`, + "Content-Type": "application/json", + "x-api-version": API_VERSION, + }, + body: JSON.stringify({ onboard: true }), + redirect: "error", + }); + const transient = response.status === 429 || response.status >= 500; + if (!transient || attempt >= 2) break; + await new Promise((r) => setTimeout(r, 5000 * (attempt + 1))); + } + + const text = await response.text(); + if (!response.ok) { + // Meta's error envelope (e.g. 403 code 4705002) carries the fix-it URL + let msg = `${response.status} ${text.slice(0, 200)}`; + try { + const err = JSON.parse(text); + if (err?.title || err?.detail) { + msg = [err.title, err.detail].filter(Boolean).join(": "); + if (err.action_url) msg += ` — ${err.action_url}`; + } + } catch { /* non-JSON error body */ } + throw new Error(`Muse Code key mint failed: ${msg}`); + } + + let payload; + try { + payload = JSON.parse(text); + } catch { + throw new Error("Muse Code key mint returned invalid JSON"); + } + + if (payload.is_subs_active === false) { + throw new Error("Muse Code subscription is inactive — activate it on muse.ai first"); + } + const actionUrl = payload.action_url || payload.require_payment_action_url; + if (!payload.api_key && (payload.require_payment || actionUrl)) { + throw new Error(`Muse Code subscription required${actionUrl ? `: ${actionUrl}` : ""}`); + } + if (!payload.api_key) { + throw new Error("Muse Code key response is missing api_key"); + } + + return { key: payload }; + }, + mapTokens: (tokens, extra) => { + const payload = extra?.key || {}; + // Chat requests carry the minted Model API key, not the Meta account token. + // No expiry/refresh from Meta — a dead key means re-login. + return { + accessToken: payload.api_key, + refreshToken: null, + expiresIn: null, + email: payload.user_email?.trim().toLowerCase() || undefined, + providerSpecificData: { + authMethod: "device_code", + // Kept so a future re-mint can run without another device login + oauthAccessToken: tokens.access_token, + subscriptionTier: payload.subs_tier_name || null, + }, + }; + }, +}; + +export default muse; diff --git a/src/shared/components/OAuthModal.js b/src/shared/components/OAuthModal.js index 7a76a020..6b73bc8b 100644 --- a/src/shared/components/OAuthModal.js +++ b/src/shared/components/OAuthModal.js @@ -175,7 +175,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, return; } - if (data.error === "expired_token" || data.error === "access_denied") { + if (data.error === "expired_token" || data.error === "access_denied" || data.fatal) { throw new Error(data.errorDescription || data.error); } @@ -291,6 +291,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, "qoder", "qoder-cn", "grok-cli", + "muse", ]; if (deviceCodeProviders.includes(provider)) { setIsDeviceCode(true); diff --git a/tests/__baseline__/alias-baseline.json b/tests/__baseline__/alias-baseline.json index 80d3af8a..65062bb8 100644 --- a/tests/__baseline__/alias-baseline.json +++ b/tests/__baseline__/alias-baseline.json @@ -116,7 +116,12 @@ "devin": "devin", "devin-cli": "devin-cli", "morph": "morph", - "morphllm": "morph" + "morphllm": "morph", + "muse": "muse", + "muse-ai": "muse", + "meta-model-api": "muse", + "muse-code": "muse", + "muse-subscription": "muse" }, "idToAlias": { "agnes": "agnes", @@ -176,6 +181,7 @@ "mistral": "mistral", "mmf": "mmf", "morph": "morph", + "muse": "muse", "nanobanana": "nanobanana", "nebius": "nebius", "nvidia": "nvidia", @@ -211,6 +217,7 @@ "modelKeys": [ "af", "ag", + "agnes", "alicode", "alicode-intl", "alims-intl", @@ -270,6 +277,7 @@ "mistral", "mmf", "morph", + "muse", "nanobanana", "nebius", "nvidia", @@ -302,6 +310,7 @@ "together", "tokenharbor", "tokenrouter", + "v1m", "venice", "vertex", "vertex-partner", diff --git a/tests/__baseline__/providers-baseline.json b/tests/__baseline__/providers-baseline.json index 8c12ef73..d8fb0e92 100644 --- a/tests/__baseline__/providers-baseline.json +++ b/tests/__baseline__/providers-baseline.json @@ -121,7 +121,9 @@ "usage": { "oauthUrl": "https://api.anthropic.com/api/oauth/usage", "orgUrl": "https://api.anthropic.com/v1/organizations/{org_id}/usage", - "settingsUrl": "https://api.anthropic.com/v1/settings" + "settingsUrl": "https://api.anthropic.com/v1/settings", + "profileUrl": "https://api.anthropic.com/api/oauth/profile", + "resetUrl": "https://api.anthropic.com/api/organizations/{org_id}/reset_rate_limits" }, "clientId": "9d1c250a-e61b-44d9-88ed-5944d1962f5e", "tokenUrl": "https://api.anthropic.com/v1/oauth/token" @@ -199,10 +201,11 @@ "baseUrl": "https://chatgpt.com/backend-api/codex/responses", "format": "openai-responses", "forceStream": true, - "cliVersion": "0.154.0", + "cliVersion": "0.155.0", "headers": { "originator": "codex_cli_rs", - "User-Agent": "codex_cli_rs/0.154.0" + "User-Agent": "codex_cli_rs/0.155.0", + "version": "0.155.0" }, "usage": { "url": "https://chatgpt.com/backend-api/wham/usage", @@ -1086,6 +1089,14 @@ }, "format": "openai" }, + "tokenharbor": { + "baseUrl": "https://tokenharbor.ai/v1/chat/completions", + "validateUrl": "https://tokenharbor.ai/v1/models", + "retry": { + "429": 2 + }, + "format": "openai" + }, "dahl": { "baseUrl": "https://inference.dahl.global/v1/chat/completions", "validateUrl": "https://inference.dahl.global/v1/models", @@ -1106,12 +1117,46 @@ "validateUrl": "https://api.b.ai/v1/models", "format": "openai" }, - "tokenharbor": { - "baseUrl": "https://tokenharbor.ai/v1/chat/completions", - "validateUrl": "https://tokenharbor.ai/v1/models", - "retry": { - "429": 2 + "muse": { + "baseUrl": "https://api.meta.ai/v1/chat/completions", + "validateUrl": "https://api.meta.ai/v1/models", + "modelsUrl": "https://api.meta.ai/v1/models", + "auth": { + "combined": true, + "header": "Authorization", + "scheme": "bearer", + "hooks": [ + "museHeaders" + ] }, - "format": "openai" + "format": "openai", + "clientId": "1031625952748946", + "tokenUrl": "https://auth.meta.com/oidc/device/token/", + "transports": [ + { + "format": "openai", + "baseUrl": "https://api.meta.ai/v1/chat/completions", + "auth": { + "combined": true, + "header": "Authorization", + "scheme": "bearer", + "hooks": [ + "museHeaders" + ] + } + }, + { + "format": "openai-responses", + "baseUrl": "https://api.meta.ai/v1/responses", + "auth": { + "combined": true, + "header": "Authorization", + "scheme": "bearer", + "hooks": [ + "museHeaders" + ] + } + } + ] } } \ No newline at end of file diff --git a/tests/__baseline__/verify-alias.mjs b/tests/__baseline__/verify-alias.mjs index a3fc1f91..b8c44cba 100644 --- a/tests/__baseline__/verify-alias.mjs +++ b/tests/__baseline__/verify-alias.mjs @@ -23,6 +23,8 @@ const ALIAS_TOKENS = [ "af","airforce","api-airforce","llm7","llm-7","samba","sambanova","bm","bluesminds", "bzl","bazaarlink","kgw","kilo-gateway","hunyuan","tencent","qianfan","baidu","ernie", "dv","devin","devin-cli","morph","morphllm", + "muse","muse-ai","meta-model-api", + "muse-code","muse-subscription", ]; // Sort idToAlias by key — runtime accesses by key, order is irrelevant (content-based) From b3cf3fdef0d66772fb4099e6f15fe83979a8b06f Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 1 Oct 2026 09:56:40 +0700 Subject: [PATCH 26/41] feat(providers): per-provider custom header overrides from the registry Custom Headers card on the provider detail page pre-fills the registry transport headers; only rows the user changes are stored (settings. providerOverrides) and merged last at dispatch, so code-side registry bumps keep winning for untouched rows. Header name/value validation and a blocked list (Host, Authorization, Cookie, ...) gate the API route. Co-Authored-By: Claude Code --- open-sse/executors/base.js | 4 +- open-sse/handlers/chatCore.js | 4 +- .../providers/[id]/CustomConfigCard.js | 181 ++++++++++++++++++ .../dashboard/providers/[id]/page.js | 4 + src/app/api/providers/[id]/overrides/route.js | 110 +++++++++++ src/lib/db/repos/settingsRepo.js | 2 + src/sse/handlers/chat.js | 2 + 7 files changed, 305 insertions(+), 2 deletions(-) create mode 100644 src/app/(dashboard)/dashboard/providers/[id]/CustomConfigCard.js create mode 100644 src/app/api/providers/[id]/overrides/route.js diff --git a/open-sse/executors/base.js b/open-sse/executors/base.js index 18a62229..71e26deb 100644 --- a/open-sse/executors/base.js +++ b/open-sse/executors/base.js @@ -97,7 +97,7 @@ export class BaseExecutor { return { status: response.status, message: bodyText || `HTTP ${response.status}` }; } - async execute({ model, body, stream, credentials, signal, log, proxyOptions = null }) { + async execute({ model, body, stream, credentials, signal, log, proxyOptions = null, providerOverrides = null }) { const fallbackCount = this.getFallbackCount(); let lastError = null; let lastStatus = 0; @@ -128,6 +128,8 @@ export class BaseExecutor { const url = this.buildUrl(model, stream, urlIndex, credentials); const transformedBody = this.transformRequest(model, body, stream, credentials); const headers = this.buildHeaders(credentials, stream, url, model, transformedBody); + // User per-provider override wins over registry headers (blocked names filtered at the API) + if (providerOverrides?.headers) Object.assign(headers, providerOverrides.headers); if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0; diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index 8755659d..60d3d815 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -60,7 +60,7 @@ export function stripContinuityFields(body) { return body; } -export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking }) { +export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, headroomTimeoutMs, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking, providerOverrides }) { const { provider, model } = modelInfo; const requestStartTime = Date.now(); // Stable per-session color so all lines of one CLI conversation share a tag @@ -383,6 +383,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred signal: streamController.signal, log, proxyOptions, + providerOverrides, }); providerResponse = result.response; providerUrl = result.url; @@ -451,6 +452,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred signal: streamController.signal, log, proxyOptions, + providerOverrides, }); if (retryResult.response.ok) { providerResponse = retryResult.response; diff --git a/src/app/(dashboard)/dashboard/providers/[id]/CustomConfigCard.js b/src/app/(dashboard)/dashboard/providers/[id]/CustomConfigCard.js new file mode 100644 index 00000000..119dd1be --- /dev/null +++ b/src/app/(dashboard)/dashboard/providers/[id]/CustomConfigCard.js @@ -0,0 +1,181 @@ +"use client"; + +import { useCallback, useEffect, useState } from "react"; +import PropTypes from "prop-types"; +import { Card, Badge } from "@/shared/components"; +import { useNotificationStore } from "@/store/notificationStore"; + +// Mirrors the server-side gate in /api/providers/[id]/overrides — client check is UX only +const BLOCKED_HEADERS = ["host", "content-length", "content-type", "connection", "transfer-encoding", "authorization", "cookie"]; +const HEADER_NAME_RE = /^[A-Za-z0-9-]+$/; + +export default function CustomConfigCard({ providerId }) { + const notify = useNotificationStore(); + const [expanded, setExpanded] = useState(false); + const [rows, setRows] = useState([{ name: "", value: "" }]); + const [builtin, setBuiltin] = useState({}); + const [hasOverride, setHasOverride] = useState(false); + const [saving, setSaving] = useState(false); + + useEffect(() => { + let cancelled = false; + fetch(`/api/providers/${providerId}/overrides`, { cache: "no-store" }) + .then((r) => (r.ok ? r.json() : null)) + .then((data) => { + if (cancelled || !data) return; + // Effective set = registry built-ins with user overrides layered on top + const builtinHeaders = data.builtinHeaders || {}; + const effective = { ...builtinHeaders, ...(data.headers || {}) }; + const headerRows = Object.entries(effective).map(([name, value]) => ({ name, value })); + setBuiltin(builtinHeaders); + setRows(headerRows.length ? headerRows : [{ name: "", value: "" }]); + setHasOverride(Object.keys(data.headers || {}).length > 0); + }) + .catch(() => {}); + return () => { cancelled = true; }; + }, [providerId]); + + const setRow = (i, field, value) => { + setRows((prev) => prev.map((r, idx) => (idx === i ? { ...r, [field]: value } : r))); + }; + + const save = useCallback(async () => { + // Diff-on-save: only rows differing from the registry default become overrides, + // so a code-side registry bump still wins for everything the user left alone. + const headers = {}; + for (const r of rows.filter((r) => r.name.trim())) { + const name = r.name.trim(); + if (!HEADER_NAME_RE.test(name)) { + notify.error(`Invalid header name: ${name}`); + return; + } + if (BLOCKED_HEADERS.includes(name.toLowerCase())) { + notify.error(`Header ${name} cannot be overridden`); + return; + } + if (name in headers) { + notify.error(`Duplicate header name: ${name}`); + return; + } + if (r.value !== builtin[name]) headers[name] = r.value; + } + + setSaving(true); + try { + const res = await fetch(`/api/providers/${providerId}/overrides`, { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ headers }), + }); + if (!res.ok) { + const err = await res.json().catch(() => ({})); + notify.error(err.error || "Failed to save"); + return; + } + setHasOverride(Object.keys(headers).length > 0); + notify.success("Custom headers saved"); + } finally { + setSaving(false); + } + }, [rows, builtin, providerId, notify]); + + const resetToBuiltin = () => { + setRows(Object.entries(builtin).map(([name, value]) => ({ name, value }))); + }; + + // Only render when there is something to customize: registry headers or existing overrides + if (Object.keys(builtin).length === 0 && !hasOverride) return null; + + return ( + + + + {expanded && ( +
+
+ {rows.map((row, i) => { + const overridden = row.name.trim() in builtin && row.value !== builtin[row.name.trim()]; + return ( +
+ setRow(i, "name", e.target.value)} + placeholder="Header-Name" + spellCheck={false} + className="w-44 rounded-md border border-border bg-background px-2 py-1.5 text-sm focus:border-primary focus:outline-none" + /> + setRow(i, "value", e.target.value)} + placeholder="Value" + spellCheck={false} + title={overridden ? "Overridden" : row.name.trim() in builtin ? "Registry default" : ""} + className={`min-w-0 flex-1 rounded-md border bg-background px-2 py-1.5 text-sm focus:border-primary focus:outline-none ${ + overridden ? "border-amber-400/60" : "border-border" + }`} + /> + +
+ ); + })} +
+ +
+ +
+ + +
+
+
+ )} +
+ ); +} + +CustomConfigCard.propTypes = { + providerId: PropTypes.string.isRequired, +}; diff --git a/src/app/(dashboard)/dashboard/providers/[id]/page.js b/src/app/(dashboard)/dashboard/providers/[id]/page.js index 61b416ef..c861e900 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/page.js +++ b/src/app/(dashboard)/dashboard/providers/[id]/page.js @@ -23,6 +23,7 @@ import EditCompatibleNodeModal from "./EditCompatibleNodeModal"; import AddCustomModelModal from "./AddCustomModelModal"; import BulkImportCodexModal from "./BulkImportCodexModal"; import BulkImportGrokCliModal from "./BulkImportGrokCliModal"; +import CustomConfigCard from "./CustomConfigCard"; const ONE_BY_ONE_DELAY_MS = 1000; @@ -1743,6 +1744,9 @@ export default function ProviderDetailPage() { )} + {/* Per-provider user overrides (custom headers / connect timeout) */} + + {/* Models */}
diff --git a/src/app/api/providers/[id]/overrides/route.js b/src/app/api/providers/[id]/overrides/route.js new file mode 100644 index 00000000..f767da85 --- /dev/null +++ b/src/app/api/providers/[id]/overrides/route.js @@ -0,0 +1,110 @@ +import { NextResponse } from "next/server"; +import { getSettings, updateSettings } from "@/lib/localDb"; +import { PROVIDERS } from "open-sse/config/providers.js"; +import { resolveProviderAlias } from "open-sse/services/model.js"; + +export const dynamic = "force-dynamic"; + +// Validation at the trust boundary — the UI also validates, but this is the gate. +const MAX_HEADERS = 20; +const MAX_HEADER_VALUE_LENGTH = 8192; +// RFC 7230 token subset: letters, digits, hyphen (no spaces, no unicode) +const HEADER_NAME_RE = /^[A-Za-z0-9-]+$/; +// Request-structure / auth headers a user override must never touch +const BLOCKED_HEADERS = new Set([ + "host", + "content-length", + "content-type", + "connection", + "transfer-encoding", + "authorization", + "cookie", +]); + +/** + * Validate + normalize an override payload. Returns { override } or { error }. + * An override with no headers is normalized to null (= delete). + */ +function normalizeOverride({ headers }) { + const out = {}; + + if (headers !== undefined && headers !== null) { + if (typeof headers !== "object" || Array.isArray(headers)) { + return { error: "headers must be an object" }; + } + const entries = Object.entries(headers).filter(([, v]) => v !== "" && v != null); + if (entries.length > MAX_HEADERS) { + return { error: `Too many headers (max ${MAX_HEADERS})` }; + } + const clean = {}; + for (const [name, value] of entries) { + if (!HEADER_NAME_RE.test(name)) { + return { error: `Invalid header name: ${name}` }; + } + if (typeof value !== "string" || /[\r\n]/.test(value)) { + return { error: `Invalid value for header ${name}` }; + } + if (value.length > MAX_HEADER_VALUE_LENGTH) { + return { error: `Header ${name} value too long (max ${MAX_HEADER_VALUE_LENGTH})` }; + } + if (BLOCKED_HEADERS.has(name.toLowerCase())) { + return { error: `Header ${name} cannot be overridden` }; + } + clean[name] = value; + } + if (Object.keys(clean).length) out.headers = clean; + } + + return { override: Object.keys(out).length ? out : null }; +} + +async function readOverrides() { + const settings = await getSettings(); + return settings.providerOverrides || {}; +} + +/** + * GET /api/providers/[id]/overrides — user override for this provider + */ +export async function GET(request, { params }) { + try { + const { id } = await params; + // URL may use an alias (gcli, cc…) — key everything by canonical registry id + const canonical = resolveProviderAlias(id); + const override = (await readOverrides())[canonical] || {}; + // Built-in headers come straight from the registry transport — single source of + // truth, so the UI pre-fills exactly what this provider sends upstream. + return NextResponse.json({ + headers: override.headers || {}, + builtinHeaders: PROVIDERS[canonical]?.headers || {}, + }); + } catch (error) { + console.log("Error getting provider overrides:", error); + return NextResponse.json({ error: "Failed to get overrides" }, { status: 500 }); + } +} + +/** + * PUT /api/providers/[id]/overrides — body: { headers: {name: value} } + * Empty payload clears the override. + */ +export async function PUT(request, { params }) { + try { + const { id } = await params; + const canonical = resolveProviderAlias(id); + const body = await request.json().catch(() => ({})); + const { override, error } = normalizeOverride(body); + if (error) { + return NextResponse.json({ error }, { status: 400 }); + } + const current = await readOverrides(); + const next = { ...current }; + if (override) next[canonical] = override; + else delete next[canonical]; + await updateSettings({ providerOverrides: next }); + return NextResponse.json({ headers: override?.headers || {} }); + } catch (error) { + console.log("Error saving provider overrides:", error); + return NextResponse.json({ error: "Failed to save overrides" }, { status: 500 }); + } +} diff --git a/src/lib/db/repos/settingsRepo.js b/src/lib/db/repos/settingsRepo.js index 8b8023f0..60afee27 100644 --- a/src/lib/db/repos/settingsRepo.js +++ b/src/lib/db/repos/settingsRepo.js @@ -62,6 +62,8 @@ const DEFAULT_SETTINGS = { pxpipeAutoInstall: true, pxpipeMinChars: 25000, pxpipeTimeoutMs: 15000, + // Per-provider user header overrides applied at dispatch: { [providerId]: { headers: {..} } } + providerOverrides: {}, }; async function readRaw() { diff --git a/src/sse/handlers/chat.js b/src/sse/handlers/chat.js index f7070f9a..bbd89242 100644 --- a/src/sse/handlers/chat.js +++ b/src/sse/handlers/chat.js @@ -293,6 +293,8 @@ async function handleSingleModelChat(body, modelStr, clientRawRequest = null, re pxpipeTransform: chatSettings.pxpipeEnabled ? await getPxpipeTransform() : null, onPxpipeEvent: appendPxpipeEvent, providerThinking, + // Per-provider user overrides (custom headers / connect timeout) from settings + providerOverrides: (chatSettings.providerOverrides || {})[provider] || null, // Detect source format by endpoint + body sourceFormatOverride: request?.url ? detectFormatByEndpoint(new URL(request.url).pathname, body) : null, onCredentialsRefreshed: async (newCreds) => { From 6b9dc54dd2685c97bf05964497b169af94d72e72 Mon Sep 17 00:00:00 2001 From: Noval Arya Saputra <120592179+budipratama10@users.noreply.github.com> Date: Thu, 1 Oct 2026 09:56:13 +0700 Subject: [PATCH 27/41] fix(grok-cli): send Grok CLI 1.0.44 so proxy stops returning HTTP 426 --- open-sse/config/grokCli.js | 5 ++++- open-sse/providers/registry/grok-cli.js | 2 +- src/app/api/providers/[id]/test/testUtils.js | 5 +++-- src/lib/oauth/providers/grok-cli.js | 9 +++++---- tests/__baseline__/providers-baseline.json | 6 +++--- tests/unit/grok-cli-executor.test.js | 2 +- tests/unit/grok-cli-models.test.js | 2 +- tests/unit/grok-cli-usage.test.js | 2 +- 8 files changed, 19 insertions(+), 14 deletions(-) diff --git a/open-sse/config/grokCli.js b/open-sse/config/grokCli.js index f2e024e4..3197116e 100644 --- a/open-sse/config/grokCli.js +++ b/open-sse/config/grokCli.js @@ -1,8 +1,11 @@ -export const GROK_CLI_VERSION = "0.2.99"; +// cli-chat-proxy rejects older identities with HTTP 426. Keep this on a +// current @xai-official/grok release (1.0.44 as of 2026-10-01; minimum 1.0.13). +export const GROK_CLI_VERSION = "1.0.44"; export const GROK_CLI_MODEL = "grok-build"; export const GROK_CLI_BASE_URL = "https://cli-chat-proxy.grok.com/v1"; export const GROK_CLI_CLIENT_IDENTIFIER = "grok-shell"; export const GROK_CLI_USER_AGENT = `grok-shell/${GROK_CLI_VERSION} (linux; x86_64)`; +export const GROK_CLI_PAGER_USER_AGENT = `grok-pager/${GROK_CLI_VERSION} grok-shell/${GROK_CLI_VERSION} (linux; x86_64)`; export function supportsGrokCliReasoningEffort(model) { // ponytail: unknown models omit effort until live metadata reaches dispatch. diff --git a/open-sse/providers/registry/grok-cli.js b/open-sse/providers/registry/grok-cli.js index 403c1abe..ea87e000 100644 --- a/open-sse/providers/registry/grok-cli.js +++ b/open-sse/providers/registry/grok-cli.js @@ -1,7 +1,7 @@ /** * Grok CLI / Grok Build (cli-chat-proxy.grok.com) * - * Source of truth: wire capture of official @xai-official/grok 0.2.99 + * Source of truth: wire capture of official @xai-official/grok 1.0.44 * talking to https://cli-chat-proxy.grok.com (OpenAI Responses API). * * Distinct from: diff --git a/src/app/api/providers/[id]/test/testUtils.js b/src/app/api/providers/[id]/test/testUtils.js index 7f9eedeb..27847c2c 100644 --- a/src/app/api/providers/[id]/test/testUtils.js +++ b/src/app/api/providers/[id]/test/testUtils.js @@ -5,6 +5,7 @@ import { isOpenAICompatibleProvider, isAnthropicCompatibleProvider } from "@/sha import { getDefaultModel } from "open-sse/config/providerModels.js"; import { resolveOllamaLocalHost, PROVIDERS } from "open-sse/config/providers.js"; import { CODEX_CLI_VERSION } from "open-sse/config/appConstants.js"; +import { GROK_CLI_PAGER_USER_AGENT, GROK_CLI_VERSION } from "open-sse/config/grokCli.js"; import { refreshProviderCredentials, shouldRefreshCredentials, @@ -123,10 +124,10 @@ const OAUTH_TEST_CONFIG = { extraHeaders: { Accept: "application/json", ...(PROVIDERS["grok-cli"]?.headers || { - "User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)", + "User-Agent": GROK_CLI_PAGER_USER_AGENT, "x-xai-token-auth": "xai-grok-cli", "x-grok-client-identifier": "grok-pager", - "x-grok-client-version": "0.2.93", + "x-grok-client-version": GROK_CLI_VERSION, }), }, refreshable: true, diff --git a/src/lib/oauth/providers/grok-cli.js b/src/lib/oauth/providers/grok-cli.js index 2bd853b2..fcc79fe8 100644 --- a/src/lib/oauth/providers/grok-cli.js +++ b/src/lib/oauth/providers/grok-cli.js @@ -1,3 +1,4 @@ +import { GROK_CLI_PAGER_USER_AGENT, GROK_CLI_VERSION } from "open-sse/config/grokCli.js"; import { GROK_CLI_CONFIG } from "../constants/oauth.js"; import { decodeXaiIdTokenEmail, extractEmailFromAccessToken } from "../providerHelpers.js"; @@ -18,7 +19,7 @@ const grokCli = { headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json", - "User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)", + "User-Agent": GROK_CLI_PAGER_USER_AGENT, }, body, }); @@ -36,7 +37,7 @@ const grokCli = { headers: { "Content-Type": "application/x-www-form-urlencoded", Accept: "application/json", - "User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)", + "User-Agent": GROK_CLI_PAGER_USER_AGENT, }, body: new URLSearchParams({ grant_type: "urn:ietf:params:oauth:grant-type:device_code", @@ -69,9 +70,9 @@ const grokCli = { headers: { Authorization: `Bearer ${tokens.access_token}`, Accept: "application/json", - "User-Agent": "grok-pager/0.2.93 grok-shell/0.2.93 (linux; x86_64)", + "User-Agent": GROK_CLI_PAGER_USER_AGENT, "x-xai-token-auth": "xai-grok-cli", - "x-grok-client-version": "0.2.93", + "x-grok-client-version": GROK_CLI_VERSION, }, }); if (res.ok) return { user: await res.json() }; diff --git a/tests/__baseline__/providers-baseline.json b/tests/__baseline__/providers-baseline.json index d8fb0e92..06a65846 100644 --- a/tests/__baseline__/providers-baseline.json +++ b/tests/__baseline__/providers-baseline.json @@ -417,13 +417,13 @@ "modelsUrl": "https://cli-chat-proxy.grok.com/v1/models", "userUrl": "https://cli-chat-proxy.grok.com/v1/user", "billingUrl": "https://cli-chat-proxy.grok.com/v1/billing", - "clientVersion": "0.2.99", + "clientVersion": "1.0.44", "clientIdentifier": "grok-shell", "tokenAuth": "xai-grok-cli", "headers": { - "User-Agent": "grok-shell/0.2.99 (linux; x86_64)", + "User-Agent": "grok-shell/1.0.44 (linux; x86_64)", "x-grok-client-identifier": "grok-shell", - "x-grok-client-version": "0.2.99" + "x-grok-client-version": "1.0.44" }, "usage": { "url": "https://cli-chat-proxy.grok.com/v1/billing?format=credits", diff --git a/tests/unit/grok-cli-executor.test.js b/tests/unit/grok-cli-executor.test.js index 519bf038..a21d819d 100644 --- a/tests/unit/grok-cli-executor.test.js +++ b/tests/unit/grok-cli-executor.test.js @@ -98,7 +98,7 @@ describe("GrokCliExecutor", () => { expect(headers.Accept).toBe("text/event-stream"); expect(headers["x-xai-token-auth"]).toBeUndefined(); expect(headers["x-grok-client-identifier"]).toBe("grok-shell"); - expect(headers["x-grok-client-version"]).toBe("0.2.99"); + expect(headers["x-grok-client-version"]).toBe("1.0.44"); expect(headers["x-grok-session-id"]).toBe("sess-abc"); expect(headers["x-grok-conv-id"]).toBe("sess-abc"); expect(headers["x-grok-req-id"]).toBe("req-xyz"); diff --git a/tests/unit/grok-cli-models.test.js b/tests/unit/grok-cli-models.test.js index 3abd3159..a1673e30 100644 --- a/tests/unit/grok-cli-models.test.js +++ b/tests/unit/grok-cli-models.test.js @@ -75,6 +75,6 @@ describe("Grok CLI live models", () => { expect(fetchFn).toHaveBeenCalledTimes(2); expect(fetchFn.mock.calls[0][2]).toBe(proxyOptions); expect(fetchFn.mock.calls[1][1].headers.Authorization).toBe("Bearer new-token"); - expect(fetchFn.mock.calls[1][1].headers["x-grok-client-version"]).toBe("0.2.99"); + expect(fetchFn.mock.calls[1][1].headers["x-grok-client-version"]).toBe("1.0.44"); }); }); diff --git a/tests/unit/grok-cli-usage.test.js b/tests/unit/grok-cli-usage.test.js index 4017ea03..d4aca719 100644 --- a/tests/unit/grok-cli-usage.test.js +++ b/tests/unit/grok-cli-usage.test.js @@ -272,7 +272,7 @@ describe("getUsageForProvider(grok-cli)", () => { expect(billingCall[0]).toContain("/v1/billing"); expect(billingCall[1].headers.Authorization).toBe("Bearer test-token"); expect(billingCall[1].headers["x-xai-token-auth"]).toBe("xai-grok-cli"); - expect(billingCall[1].headers["x-grok-client-version"]).toBe("0.2.99"); + expect(billingCall[1].headers["x-grok-client-version"]).toBe("1.0.44"); expect(billingCall[1].headers["x-grok-client-identifier"]).toBe("grok-shell"); expect(billingCall[1].headers["x-userid"]).toBe( "d84768dd-224d-4052-ba49-0d336fa9160c", From 068ce87d206313861fde0a4c27adb7e10371511c Mon Sep 17 00:00:00 2001 From: Feavy Date: Thu, 1 Oct 2026 10:02:41 +0700 Subject: [PATCH 28/41] feat(glm): add Z.ai OAuth login to GLM Coding (dual-auth) OAuth login via the ZCode CLI polling flow: init -> browser authorize -> poll ready -> business JWT -> auto-mint a coding-plan API key stored on accessToken. Pasted-apikey mode is unchanged; quota works for both. No refresh grant upstream, so expiry means re-login. Display renamed to "Zai GLM Coding". --- cli/src/cli/menus/providers.js | 5 +- cli/src/cli/utils/modelSelector.js | 2 +- open-sse/providers/registry/glm.js | 22 +- open-sse/services/usage.js | 5 +- open-sse/services/usage/glm.js | 87 ++--- .../api/oauth/[provider]/[action]/route.js | 3 +- src/lib/oauth/constants/oauth.js | 8 + src/lib/oauth/providers/glm.js | 305 ++++++++++++++++++ src/lib/oauth/providers/index.js | 2 + src/shared/components/OAuthModal.js | 27 +- tests/unit/glm-oauth.test.js | 248 ++++++++++++++ 11 files changed, 656 insertions(+), 58 deletions(-) create mode 100644 src/lib/oauth/providers/glm.js create mode 100644 tests/unit/glm-oauth.test.js diff --git a/cli/src/cli/menus/providers.js b/cli/src/cli/menus/providers.js index 42fbed6f..56c51db4 100644 --- a/cli/src/cli/menus/providers.js +++ b/cli/src/cli/menus/providers.js @@ -138,11 +138,12 @@ const OAUTH_PROVIDERS = { iflow: { id: "iflow", alias: "if", name: "iFlow AI" }, qwen: { id: "qwen", alias: "qw", name: "Qwen Code" }, kiro: { id: "kiro", alias: "kr", name: "Kiro AI" }, + glm: { id: "glm", alias: "glm", name: "Zai GLM Coding" }, }; const APIKEY_PROVIDERS = { openrouter: { id: "openrouter", name: "OpenRouter" }, - glm: { id: "glm", name: "GLM Coding" }, + glm: { id: "glm", name: "Zai GLM Coding" }, minimax: { id: "minimax", name: "Minimax Coding" }, kimi: { id: "kimi", name: "Kimi" }, openai: { id: "openai", name: "OpenAI" }, @@ -399,7 +400,7 @@ async function showConnectionActions(connection, providerId, breadcrumb = []) { * @param {string} authType - "oauth" or "apikey" */ // Providers that use Device Code Flow (terminal-based polling) -const DEVICE_CODE_PROVIDERS = ["github", "qwen", "kiro"]; +const DEVICE_CODE_PROVIDERS = ["github", "qwen", "kiro", "glm"]; /** * Handle adding new connection - auto-detect flow type diff --git a/cli/src/cli/utils/modelSelector.js b/cli/src/cli/utils/modelSelector.js index cec99deb..eef4de4f 100644 --- a/cli/src/cli/utils/modelSelector.js +++ b/cli/src/cli/utils/modelSelector.js @@ -21,7 +21,7 @@ const PROVIDER_ALIAS_NAMES = { oc: "OpenCode Free", opencode: "OpenCode Free", openrouter: "OpenRouter", - glm: "GLM Coding", + glm: "Zai GLM Coding", kimi: "Kimi Coding", minimax: "Minimax Coding", openai: "OpenAI", diff --git a/open-sse/providers/registry/glm.js b/open-sse/providers/registry/glm.js index 88f4c563..b473ae71 100644 --- a/open-sse/providers/registry/glm.js +++ b/open-sse/providers/registry/glm.js @@ -5,16 +5,34 @@ export default { priority: 140, alias: "glm", display: { - name: "GLM Coding", + name: "Zai GLM Coding", icon: "code", color: "#2563EB", textIcon: "GL", website: "https://open.bigmodel.cn", notice: { apiKeyUrl: "https://open.bigmodel.cn/usercenter/apikeys", + signupUrl: "https://chat.z.ai", }, }, - category: "apikey", + category: "oauth", + // Dual-auth like kimi: paste an API key, or OAuth-login with the Z.ai + // account to auto-mint a coding-plan key. + authModes: ["oauth", "apikey"], + hasOAuth: true, + // OAuth = ZCode CLI polling flow (apps/zcode-cli cli-oauth.ts) — no PKCE, no + // local callback: init mints a one-off poll token, the browser authorize_url + // is server-generated, and poll/ready carries the tokens. The Z.AI OAuth + // token is exchanged for a business JWT, then a long-lived coding-plan API + // key (no refresh grant — re-login on expiry, same as the official CLI). + oauth: { + providerId: "zai", + cliInitUrl: "https://zcode.z.ai/api/v1/oauth/cli/init", + cliPollUrl: "https://zcode.z.ai/api/v1/oauth/cli/poll", + businessLoginUrl: "https://api.z.ai/api/auth/z/login", + apiBaseUrl: "https://api.z.ai", + planApiKeyName: "zcode-api-key", + }, transport: { baseUrl: "https://api.z.ai/api/anthropic/v1/messages", format: "claude", diff --git a/open-sse/services/usage.js b/open-sse/services/usage.js index 3ce46cfa..71b53cc9 100644 --- a/open-sse/services/usage.js +++ b/open-sse/services/usage.js @@ -46,8 +46,9 @@ const USAGE_HANDLERS = { "qoder-cn": (c) => getQoderUsageFor(c), iflow: (c) => getIflowUsage(c.accessToken), ollama: (c) => getOllamaUsage(c.apiKey, c.providerSpecificData, c.proxyOptions), - glm: (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions), - "glm-cn": (c) => getGlmUsage(c.apiKey, c.provider, c.proxyOptions), + // OAuth connections store the coding-plan key on accessToken (no apiKey) + glm: (c) => getGlmUsage(c.apiKey || c.accessToken, c.provider, c.proxyOptions), + "glm-cn": (c) => getGlmUsage(c.apiKey || c.accessToken, c.provider, c.proxyOptions), minimax: (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions), "minimax-cn": (c) => getMiniMaxUsage(c.apiKey, c.provider, c.proxyOptions), "vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions), diff --git a/open-sse/services/usage/glm.js b/open-sse/services/usage/glm.js index f4064af6..92296490 100644 --- a/open-sse/services/usage/glm.js +++ b/open-sse/services/usage/glm.js @@ -12,9 +12,55 @@ const GLM_QUOTA_URLS = { }; /** - * GLM Coding Plan usage (international + China regions) + * Parse the GLM quota API response — shared by pasted API keys and + * OAuth-minted coding-plan keys (both hit the same monitor endpoint). * Supports both TOKENS_LIMIT and CREDIT_LIMIT and dynamic intervals (e.g. session 5h, weekly 7d). */ +export function parseGlmQuotaResponse(json) { + const data = json?.data && typeof json.data === "object" ? json.data : {}; + const limits = Array.isArray(data.limits) ? data.limits : []; + const quotas = {}; + + for (const limit of limits) { + // 1. Accept both TOKENS_LIMIT and CREDIT_LIMIT from GLM API + if (!limit || (limit.type !== "TOKENS_LIMIT" && limit.type !== "CREDIT_LIMIT")) continue; + const usedPercent = Number(limit.percentage) || 0; + const resetMs = Number(limit.nextResetTime) || 0; + const remaining = Math.max(0, 100 - usedPercent); + + // 2. Map key dynamically based on type and period (unit) to avoid overwriting + let key = "session"; + if (limit.unit === 3) { + key = `Session (${limit.number}h)`; + } else if (limit.unit === 6) { + key = "Weekly (7d)"; + } else if (limit.type === "TOKENS_LIMIT") { + key = "Tokens"; + } else { + key = `Limit (${limit.number})`; + } + + quotas[key] = { + used: usedPercent, + total: 100, + remaining, + remainingPercentage: remaining, + resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null, + unlimited: false, + }; + } + + const levelRaw = typeof data.level === "string" ? data.level : ""; + const plan = levelRaw + ? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase() + : "Unknown"; + + return { plan, quotas }; +} + +/** + * GLM Coding Plan usage (international + China regions) + */ export async function getGlmUsage(apiKey, provider, proxyOptions = null) { if (!apiKey) { return { message: "GLM API key not available." }; @@ -43,44 +89,7 @@ export async function getGlmUsage(apiKey, provider, proxyOptions = null) { } const json = await response.json(); - const data = json?.data && typeof json.data === "object" ? json.data : {}; - const limits = Array.isArray(data.limits) ? data.limits : []; - const quotas = {}; - - for (const limit of limits) { - // 1. Accept both TOKENS_LIMIT and CREDIT_LIMIT from GLM API - if (!limit || (limit.type !== "TOKENS_LIMIT" && limit.type !== "CREDIT_LIMIT")) continue; - const usedPercent = Number(limit.percentage) || 0; - const resetMs = Number(limit.nextResetTime) || 0; - const remaining = Math.max(0, 100 - usedPercent); - - // 2. Map key dynamically based on type and period (unit) to avoid overwriting - let key = "session"; - if (limit.unit === 3) { - key = `Session (${limit.number}h)`; - } else if (limit.unit === 6) { - key = "Weekly (7d)"; - } else if (limit.type === "TOKENS_LIMIT") { - key = "Tokens"; - } else { - key = `Limit (${limit.number})`; - } - - quotas[key] = { - used: usedPercent, - total: 100, - remaining, - remainingPercentage: remaining, - resetAt: resetMs > 0 ? new Date(resetMs).toISOString() : null, - unlimited: false, - }; - } - - const levelRaw = typeof data.level === "string" ? data.level : ""; - const plan = levelRaw - ? levelRaw.charAt(0).toUpperCase() + levelRaw.slice(1).toLowerCase() - : "Unknown"; - + const { plan, quotas } = parseGlmQuotaResponse(json); return { plan, quotas }; } catch (error) { return { message: `GLM error: ${error.message}` }; diff --git a/src/app/api/oauth/[provider]/[action]/route.js b/src/app/api/oauth/[provider]/[action]/route.js index b31a0d16..e363833e 100644 --- a/src/app/api/oauth/[provider]/[action]/route.js +++ b/src/app/api/oauth/[provider]/[action]/route.js @@ -266,6 +266,7 @@ export async function GET(request, { params }) { "qoder-cn", "grok-cli", "muse", + "glm", ]; let deviceData; if (noPkceDeviceProviders.includes(provider)) { @@ -499,7 +500,7 @@ export async function POST(request, { params }) { } // Providers that don't use PKCE for device code - const noPkceProviders = ["github", "kimi", "kimi-coding", "kilocode", "codebuddy-cn", "codebuddy-intl"]; + const noPkceProviders = ["github", "kimi", "kimi-coding", "kilocode", "codebuddy-cn", "codebuddy-intl", "glm"]; let result; if (noPkceProviders.includes(provider)) { // kimi needs extraData._kimiDeviceId for stable X-Msh-Device-Id (CLIProxyAPI parity) diff --git a/src/lib/oauth/constants/oauth.js b/src/lib/oauth/constants/oauth.js index 9d9d2a1f..46553b35 100644 --- a/src/lib/oauth/constants/oauth.js +++ b/src/lib/oauth/constants/oauth.js @@ -205,6 +205,13 @@ export const WINDSURF_CONFIG = { oauthTimeoutMs: 600_000, }; +// GLM Coding (Z.ai) OAuth — ZCode CLI polling flow (NOT PKCE): init mints a +// one-off poll token, the browser opens the server-generated authorize_url, +// poll/ready returns the tokens. The Z.AI OAuth token is then exchanged for a +// platform business JWT and finally a long-lived coding-plan API key (no +// refresh grant). +export const GLM_OAUTH_CONFIG = { ...PROVIDER_OAUTH["glm"] }; + // Zed hosted LLM aggregator — RSA keypair native-app auth (NOT OAuth). // Client generates ephemeral RSA-2048 keypair; user signs in at zed.dev/native_app_signin; // Zed redirects to local callback with access_token RSA-encrypted against our public key. @@ -245,5 +252,6 @@ export const PROVIDERS = { GROK_CLI: "grok-cli", TRAE: "trae", WINDSURF: "windsurf", + GLM: "glm", ZED: "zed", }; diff --git a/src/lib/oauth/providers/glm.js b/src/lib/oauth/providers/glm.js new file mode 100644 index 00000000..bac45637 --- /dev/null +++ b/src/lib/oauth/providers/glm.js @@ -0,0 +1,305 @@ +import crypto from "crypto"; +import { GLM_OAUTH_CONFIG } from "../constants/oauth.js"; + +// Zai GLM Coding OAuth — CLI polling flow (mirrors the official +// ZCode CLI, apps/zcode-cli packages/adapters/src/auth/cli-oauth.ts + +// coding-plan-api-key.ts). No PKCE and no local callback server: +// +// 1) POST {cliInitUrl} Authorization: Bearer {"provider":"zai"} +// → { code: 0, data: { authorize_url, flow_id, poll_interval_sec, expires_at } } +// 2) Browser opens authorize_url; user signs in with the Z.ai account +// 3) GET {cliPollUrl}/ Authorization: Bearer +// → { data: { status: "pending" } } until +// { data: { status: "ready", token, user, accessToken, refreshToken? } } +// 4) accessToken (Z.AI OAuth token) → POST {businessLoginUrl} {"token": ...} +// → { data: { access_token } } (platform business JWT) +// 5) Business JWT → coding-plan API key via getCustomerInfo → api_keys +// list/create("zcode-api-key") → copy → "apiKey.secretKey" +// +// The coding-plan API key is the long-lived model credential; the ZAI OAuth +// provider has no refresh_token grant, so expiry means re-login (same as the +// official CLI). zcode JWT + business token ride along in providerSpecificData +// for quota/usage and debugging. +const glm = { + config: GLM_OAUTH_CONFIG, + flowType: "device_code", + requestDeviceCode: async (config) => { + const pollToken = crypto.randomBytes(32).toString("hex"); + const response = await fetch(config.cliInitUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${pollToken}`, + }, + body: JSON.stringify({ provider: config.providerId || "zai" }), + }); + if (!response.ok) { + const error = await response.text(); + throw new Error(`ZCode OAuth init failed: ${error}`); + } + const payload = await response.json(); + if (!isSuccessCode(payload.code) || !payload.data) { + throw new Error(payload.msg || "ZCode OAuth init returned no data"); + } + const data = payload.data; + if (!data.flow_id || !data.authorize_url) { + throw new Error("ZCode OAuth init response missing flow_id/authorize_url"); + } + return { + device_code: data.flow_id, + verification_uri: data.authorize_url, + // expires_at is upstream-absolute; surface a relative deadline for the UI + expires_in: relativeSeconds(data.expires_at) ?? 300, + interval: data.poll_interval_sec || 3, + _zcodePollToken: pollToken, + }; + }, + pollToken: async (config, deviceCode, _codeVerifier, extraData) => { + const pollToken = extraData?._zcodePollToken; + if (!pollToken) { + return { + ok: true, + data: { + error: "access_denied", + error_description: "Missing ZCode poll token — restart the login flow", + }, + }; + } + + const response = await fetch(`${config.cliPollUrl}/${encodeURIComponent(deviceCode)}`, { + headers: { Authorization: `Bearer ${pollToken}` }, + }); + if (!response.ok) { + return { + ok: true, + data: { + error: "access_denied", + error_description: `ZCode poll failed (HTTP ${response.status})`, + }, + }; + } + + const payload = await response.json(); + if (!isSuccessCode(payload.code)) { + return { + ok: true, + data: { error: "access_denied", error_description: payload.msg || "ZCode poll failed" }, + }; + } + + const data = payload.data || {}; + if (data.status === "pending") { + return { ok: true, data: { error: "authorization_pending" } }; + } + if (data.status === "failed") { + return { + ok: true, + data: { + error: "access_denied", + error_description: "ZCode authorization failed or was cancelled", + }, + }; + } + if (data.status !== "ready") { + return { + ok: true, + data: { error: "authorization_pending", error_description: `Unknown status: ${data.status}` }, + }; + } + + // ready payload nests the ZAI OAuth tokens under data[providerId] (see + // apps/zcode-cli cli-oauth.ts parseReadyData): { status:"ready", token, + // user, zai: { access_token, refresh_token? } }. Fall back to top-level + // fields for resilience against payload drift. + const providerData = data[config.providerId] || data[data.providerId] || {}; + const zaiAccessToken = + providerData.access_token || + providerData.accessToken || + data.accessToken || + data.access_token; + if (!zaiAccessToken) { + return { + ok: true, + data: { + error: "access_denied", + error_description: "ZCode poll response missing access token", + }, + }; + } + + // ready.accessToken is the Z.AI OAuth token — derive the coding-plan API key + const { planApiKey, businessToken } = await resolveCodingPlanApiKey(config, zaiAccessToken); + + return { + ok: true, + data: { + access_token: planApiKey, + _zcodeJwtToken: data.token || "", + _zaiBusinessToken: businessToken, + _zaiRefreshToken: + providerData.refresh_token || providerData.refreshToken || data.refresh_token || data.refreshToken || "", + _zcodeUser: data.user || {}, + }, + }; + }, + mapTokens: (tokens) => { + const user = tokens._zcodeUser || {}; + const displayName = user.name || user.email || null; + return { + accessToken: tokens.access_token, + refreshToken: null, + email: user.email || null, + ...(displayName ? { displayName } : {}), + providerSpecificData: { + authMethod: "cli_poll", + username: user.name || undefined, + userId: user.user_id || undefined, + zcodeJwtToken: tokens._zcodeJwtToken || undefined, + zaiBusinessToken: tokens._zaiBusinessToken || undefined, + ...(tokens._zaiRefreshToken ? { zaiRefreshToken: tokens._zaiRefreshToken } : {}), + }, + }; + }, +}; + +// Business JWT → coding-plan API key ("apiKey.secretKey"). Mirrors ZCode CLI +// coding-plan-api-key.ts: getCustomerInfo → default org/project → api_keys +// list/create("zcode-api-key") → copy → secretKey. +async function resolveCodingPlanApiKey(config, zaiAccessToken) { + const businessToken = await exchangeBusinessToken(config, zaiAccessToken); + const authHeaders = { + Authorization: `Bearer ${businessToken}`, + "Content-Type": "application/json", + }; + + const customerInfo = await fetchBusinessJson( + `${config.apiBaseUrl}/api/biz/customer/getCustomerInfo`, + { headers: authHeaders }, + "customer info" + ); + const location = pickOrgAndProject(customerInfo); + if (!location) { + throw new Error("Unable to resolve Z.ai organization and project for the coding plan"); + } + + const listUrl = + `${config.apiBaseUrl}/api/biz/v1/organization/${location.organizationId}` + + `/projects/${location.projectId}/api_keys`; + const keys = (await fetchBusinessJson(listUrl, { headers: authHeaders }, "api keys")) || []; + let keyEntry = Array.isArray(keys) + ? keys.find((item) => item?.name === config.planApiKeyName) + : null; + if (!keyEntry) { + keyEntry = await fetchBusinessJson( + listUrl, + { + method: "POST", + headers: authHeaders, + body: JSON.stringify({ name: config.planApiKeyName }), + }, + "api key create" + ); + } + + const apiKey = keyEntry?.apiKey?.trim(); + if (!apiKey) { + throw new Error("Z.ai api_keys response is missing apiKey"); + } + + const secret = await fetchBusinessJson( + `${listUrl}/copy/${encodeURIComponent(apiKey)}`, + { headers: authHeaders }, + "api key copy" + ); + const secretKey = secret?.secretKey?.trim(); + if (!secretKey) { + throw new Error("Z.ai api key copy response is missing secretKey"); + } + + return { planApiKey: `${apiKey}.${secretKey}`, businessToken }; +} + +// POST {businessLoginUrl} {"token": } → { data: { access_token } } +async function exchangeBusinessToken(config, zaiAccessToken) { + const payload = await fetchBusinessJson( + config.businessLoginUrl, + { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ token: zaiAccessToken }), + }, + "Z.ai business login" + ); + const token = payload?.access_token?.trim() || payload?.accessToken?.trim(); + if (!token) { + throw new Error("Z.ai business login response is missing access_token"); + } + return token; +} + +// Business endpoints answer {code, msg, data}; code 0/200 (or absent) = success. +// data is returned directly (null when missing). +async function fetchBusinessJson(url, options, label) { + const response = await fetch(url, options); + const text = await response.text(); + if (!response.ok) { + throw new Error(`Z.ai ${label} request failed (HTTP ${response.status}): ${text.slice(0, 200)}`); + } + let payload; + try { + payload = JSON.parse(text); + } catch { + throw new Error(`Z.ai ${label} response is not valid JSON`); + } + if (!isSuccessCode(payload?.code) || payload?.success === false) { + throw new Error(payload?.msg || `Z.ai ${label} returned business error ${payload?.code}`); + } + return payload?.data ?? payload ?? null; +} + +// Prefer the org named "默认机构"/"default" and the non-team project named +// "默认项目"/"default" (projectType "2" = team), falling back to the first entries. +function pickOrgAndProject(customerInfo) { + const organizations = Array.isArray(customerInfo?.organizations) + ? customerInfo.organizations + : []; + const personalOrgs = organizations + .map((organization) => ({ + organization, + projects: (organization?.projects || []).filter( + (project) => String(project?.projectType ?? "").trim() !== "2" + ), + })) + .filter(({ organization, projects }) => + Boolean(organization?.organizationId && projects.length) + ); + if (!personalOrgs.length) return null; + + const org = + personalOrgs.find(({ organization }) => isDefaultName(organization.organizationName)) || + personalOrgs[0]; + const project = + org.projects.find((item) => isDefaultName(item?.projectName)) || org.projects[0]; + if (!org.organization?.organizationId || !project?.projectId) return null; + return { organizationId: org.organization.organizationId, projectId: project.projectId }; +} + +function isDefaultName(name) { + const normalized = String(name || "").trim().toLowerCase(); + return normalized.includes("默认机构") || normalized.includes("默认项目") || normalized === "default"; +} + +function isSuccessCode(code) { + return code === undefined || code === null || code === 0 || code === 200 || code === "0" || code === "200"; +} + +// Absolute epoch (s or ms) → seconds from now; null when absent/invalid. +function relativeSeconds(expiresAt) { + const raw = Number(expiresAt); + if (!Number.isFinite(raw) || raw <= 0) return null; + const ms = raw > 1e12 ? raw : raw * 1000; + const seconds = Math.floor((ms - Date.now()) / 1000); + return seconds > 0 ? seconds : null; +} + +export default glm; diff --git a/src/lib/oauth/providers/index.js b/src/lib/oauth/providers/index.js index 80063950..8a27363d 100644 --- a/src/lib/oauth/providers/index.js +++ b/src/lib/oauth/providers/index.js @@ -28,6 +28,7 @@ import kimchi from "./kimchi.js"; import trae from "./trae.js"; import windsurf from "./windsurf.js"; import zed from "./zed.js"; +import glm from "./glm.js"; // Provider configurations const PROVIDERS = { @@ -55,6 +56,7 @@ const PROVIDERS = { trae, windsurf, zed, + glm, }; export { PROVIDERS }; diff --git a/src/shared/components/OAuthModal.js b/src/shared/components/OAuthModal.js index 6b73bc8b..95d50382 100644 --- a/src/shared/components/OAuthModal.js +++ b/src/shared/components/OAuthModal.js @@ -292,6 +292,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, "qoder-cn", "grok-cli", "muse", + "glm", ]; if (deviceCodeProviders.includes(provider)) { setIsDeviceCode(true); @@ -334,6 +335,8 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, } : (provider === "kimi" || provider === "kimi-coding") ? { _kimiDeviceId: data._kimiDeviceId } + : provider === "glm" + ? { _zcodePollToken: data._zcodePollToken } : null; startPolling( data.device_code, @@ -924,18 +927,20 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess,
-
-

Your Code

-
-

{deviceData.user_code}

-
- + )} {polling && (
diff --git a/tests/unit/glm-oauth.test.js b/tests/unit/glm-oauth.test.js new file mode 100644 index 00000000..5499b63e --- /dev/null +++ b/tests/unit/glm-oauth.test.js @@ -0,0 +1,248 @@ +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +// proxyAwareFetch captures globalThis.fetch at import time — mock the module +// (like kimi-usage.test.js) instead of stubbing global fetch for usage tests. +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), + default: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import glmOauthProvider from "../../src/lib/oauth/providers/glm.js"; +import { getProvider } from "../../src/lib/oauth/providers"; +import { PROVIDERS as TRANSPORTS, PROVIDER_OAUTH } from "../../open-sse/providers/index.js"; +import { USAGE_SUPPORTED_PROVIDERS } from "../../src/shared/constants/providers.js"; +import { getUsageForProvider } from "../../open-sse/services/usage.js"; +import { DefaultExecutor } from "../../open-sse/executors/default.js"; + +const ANTHROPIC_URL = "https://api.z.ai/api/anthropic/v1/messages"; +const PLAN_KEY = "key123.secret456"; + +function jsonResponse(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +describe("glm registry entry (dual-auth)", () => { + it("is an oauth+apikey provider with usage enabled", () => { + expect(USAGE_SUPPORTED_PROVIDERS).toContain("glm"); + expect(PROVIDER_OAUTH.glm).toBeDefined(); + }); + + it("keeps the direct api.z.ai anthropic transport (no separate gateway)", () => { + expect(TRANSPORTS.glm.baseUrl).toBe(ANTHROPIC_URL); + expect(TRANSPORTS.glm.auth.header).toBe("x-api-key"); + // no zcode gateway/hook leftovers + expect(TRANSPORTS.zcode).toBeUndefined(); + expect(PROVIDER_OAUTH.zcode).toBeUndefined(); + }); + + it("declares the ZCode CLI polling OAuth endpoints (no refresh grant)", () => { + expect(PROVIDER_OAUTH.glm.cliInitUrl).toBe("https://zcode.z.ai/api/v1/oauth/cli/init"); + expect(PROVIDER_OAUTH.glm.cliPollUrl).toBe("https://zcode.z.ai/api/v1/oauth/cli/poll"); + expect(PROVIDER_OAUTH.glm.businessLoginUrl).toBe("https://api.z.ai/api/auth/z/login"); + expect(PROVIDER_OAUTH.glm.refresh).toBeUndefined(); + }); + + it("is wired into the generic OAuth provider registry", () => { + expect(getProvider("glm")).toBe(glmOauthProvider); + }); +}); + +describe("glm OAuth flow (ZCode CLI poll protocol)", () => { + let calls; + + beforeEach(() => { + calls = []; + vi.stubGlobal( + "fetch", + vi.fn(async (url, init = {}) => { + const entry = { url: String(url), method: init.method || "GET", init }; + calls.push(entry); + const u = new URL(url); + + if (u.href === "https://zcode.z.ai/api/v1/oauth/cli/init") { + return jsonResponse({ + code: 0, + data: { + authorize_url: "https://chat.z.ai/api/oauth/authorize?x=1", + flow_id: "flow-123", + poll_interval_sec: 2, + expires_at: Math.floor(Date.now() / 1000) + 300, + }, + }); + } + if (u.pathname.startsWith("/api/v1/oauth/cli/poll/")) { + if (globalThis.__glmPollState === "pending") { + return jsonResponse({ code: 0, data: { status: "pending" } }); + } + return jsonResponse({ + code: 0, + data: { + status: "ready", + token: "zcode-jwt", + providerId: "zai", + user: { user_id: "u-1", name: "Feavy", email: "feavy@example.com" }, + zai: { access_token: "zai-oauth-token", refresh_token: "zai-refresh-token" }, + }, + }); + } + if (u.href === "https://api.z.ai/api/auth/z/login") { + return jsonResponse({ code: 200, data: { access_token: "zai-business-jwt" } }); + } + if (u.href === "https://api.z.ai/api/biz/customer/getCustomerInfo") { + return jsonResponse({ + code: 200, + data: { + organizations: [ + { + organizationId: "org-1", + organizationName: "默认机构", + projects: [{ projectId: "p-1", projectName: "默认项目", projectType: "1" }], + }, + ], + }, + }); + } + if (u.pathname.endsWith("/api_keys") && entry.method === "GET") { + return jsonResponse({ code: 200, data: [] }); + } + if (u.pathname.endsWith("/api_keys")) { + return jsonResponse({ code: 200, data: { apiKey: "key123", name: "zcode-api-key" } }); + } + if (u.pathname.endsWith("/copy/key123")) { + return jsonResponse({ code: 200, data: { secretKey: "secret456" } }); + } + return jsonResponse({ code: 500, msg: `unexpected ${url}` }, 500); + }), + ); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + delete globalThis.__glmPollState; + }); + + it("init returns the server-generated authorize URL + flow id", async () => { + const device = await glmOauthProvider.requestDeviceCode(glmOauthProvider.config); + expect(device.device_code).toBe("flow-123"); + expect(device.verification_uri).toBe("https://chat.z.ai/api/oauth/authorize?x=1"); + expect(device._zcodePollToken).toEqual(expect.any(String)); + expect(calls[0].init.headers.Authorization).toMatch(/^Bearer /); + expect(JSON.parse(calls[0].init.body)).toEqual({ provider: "zai" }); + }); + + it("maps pending poll state to authorization_pending", async () => { + globalThis.__glmPollState = "pending"; + const result = await glmOauthProvider.pollToken(glmOauthProvider.config, "flow-123", null, { + _zcodePollToken: "t", + }); + expect(result).toEqual({ ok: true, data: { error: "authorization_pending" } }); + }); + + it("derives the coding-plan API key from the ready state", async () => { + const result = await glmOauthProvider.pollToken(glmOauthProvider.config, "flow-123", null, { + _zcodePollToken: "t", + }); + + expect(result.ok).toBe(true); + expect(result.data.access_token).toBe(PLAN_KEY); + expect(result.data._zcodeJwtToken).toBe("zcode-jwt"); + + // derivation chain: business login → customer info → api_keys create → copy + const urls = calls.map((c) => c.url); + expect(urls).toContain("https://api.z.ai/api/auth/z/login"); + expect(urls).toContain("https://api.z.ai/api/biz/customer/getCustomerInfo"); + expect(urls).toContain("https://api.z.ai/api/biz/v1/organization/org-1/projects/p-1/api_keys"); + expect(urls).toContain( + "https://api.z.ai/api/biz/v1/organization/org-1/projects/p-1/api_keys/copy/key123", + ); + }); + + it("mapTokens stores the plan key + account identity (no refresh token)", () => { + const tokens = glmOauthProvider.mapTokens({ + access_token: PLAN_KEY, + _zcodeJwtToken: "zcode-jwt", + _zaiBusinessToken: "zai-business-jwt", + _zaiRefreshToken: "zai-refresh-token", + _zcodeUser: { user_id: "u-1", name: "Feavy", email: "feavy@example.com" }, + }); + + expect(tokens.accessToken).toBe(PLAN_KEY); + expect(tokens.refreshToken).toBeNull(); + expect(tokens.email).toBe("feavy@example.com"); + expect(tokens.displayName).toBe("Feavy"); + expect(tokens.providerSpecificData).toMatchObject({ + authMethod: "cli_poll", + username: "Feavy", + userId: "u-1", + zcodeJwtToken: "zcode-jwt", + zaiBusinessToken: "zai-business-jwt", + zaiRefreshToken: "zai-refresh-token", + }); + }); + + it("fails cleanly without the poll token (restart required)", async () => { + const result = await glmOauthProvider.pollToken(glmOauthProvider.config, "flow-123", null, {}); + expect(result.data.error).toBe("access_denied"); + }); +}); + +describe("glm executor + usage (dual-auth credentials)", () => { + it("sends the plan key as x-api-key (no gateway hook)", () => { + const executor = new DefaultExecutor("glm"); + const creds = { accessToken: PLAN_KEY, refreshToken: null }; + + const headers = executor.buildHeaders(creds, true, ANTHROPIC_URL, "glm-5.3", {}); + expect(headers["x-api-key"]).toBe(PLAN_KEY); + expect(headers["Authorization"]).toBeUndefined(); + }); + + it("never schedules a token refresh (no refresh grant upstream)", async () => { + const executor = new DefaultExecutor("glm"); + const result = await executor.refreshCredentials( + { accessToken: PLAN_KEY, refreshToken: null }, + console, + ); + expect(result).toBeNull(); + }); + + it("fetches quota with the OAuth-minted key stored on accessToken", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ + data: { + level: "PRO", + limits: [ + { type: "TOKENS_LIMIT", percentage: 35.5, number: 5, unit: 3, nextResetTime: Date.now() + 3600_000 }, + ], + }, + }), + ); + + const usage = await getUsageForProvider( + { provider: "glm", accessToken: PLAN_KEY, apiKey: null, providerSpecificData: {} }, + null, + ); + expect(usage.plan).toBe("Pro"); + expect(usage.quotas["Session (5h)"].used).toBe(35.5); + expect(proxyAwareFetch).toHaveBeenCalledWith( + "https://api.z.ai/api/monitor/usage/quota/limit", + expect.objectContaining({ headers: expect.objectContaining({ Authorization: `Bearer ${PLAN_KEY}` }) }), + null, + ); + }); + + it("still fetches quota for pasted apikey connections", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ data: { level: "PRO", limits: [] } }), + ); + + const usage = await getUsageForProvider( + { provider: "glm", accessToken: null, apiKey: PLAN_KEY, providerSpecificData: {} }, + null, + ); + expect(usage.plan).toBe("Pro"); + }); +}); From 7894f3d36a3f03f35c7d283639617ec5a0d4a9ed Mon Sep 17 00:00:00 2001 From: Alecto1b Date: Thu, 1 Oct 2026 09:38:45 +0700 Subject: [PATCH 29/41] fix(thinking): add xhigh to claude-adaptive thinking levels Expose xhigh in the level picker for claude-adaptive models (Opus 4.7+, Sonnet 5, Opus 5/5.5, Fable) and make the wire actually send it instead of silently clamping to high: - thinkingLevels: claude-adaptive now uses the budgetX set; Opus/Sonnet 4.6 keep low..max (both Anthropic and Kiro reject xhigh there) - thinkingUnified: pass xhigh through when the model advertises it, clamp to high otherwise - kiroConstants: pass xhigh/max through per Kiro docs + live additionalModelRequestFieldsSchema tiers; 4.6 models clamp xhigh to high (max stays valid). Fixes max being a silent no-op on Kiro. Kimi stays on levelMax. --- open-sse/config/kiroConstants.js | 34 +++++++++++++------ open-sse/providers/thinkingLevels.js | 11 ++++-- .../translator/concerns/thinkingUnified.js | 4 ++- tests/unit/thinking-levels-kiro.test.js | 21 ++++++++++++ 4 files changed, 56 insertions(+), 14 deletions(-) diff --git a/open-sse/config/kiroConstants.js b/open-sse/config/kiroConstants.js index e6408da1..f24cce04 100644 --- a/open-sse/config/kiroConstants.js +++ b/open-sse/config/kiroConstants.js @@ -171,7 +171,22 @@ export function resolveKiroThinkingBudget(body, headers, model) { return null; } -export function extractKiroEffortLevel(body) { +function parseClaudeVersion(model) { + if (typeof model !== "string") return null; + const normalized = model.toLowerCase().replace(/-/g, "."); + const match = normalized.match(/(?:^|[/.])claude(?:[/.][a-z]+)*[/.](\d+)(?:[/.](\d+))?(?:[/.]|$)/); + if (!match) return null; + return { major: Number(match[1]), minor: match[2] === undefined ? null : Number(match[2]) }; +} + +// Kiro effort tiers per model (kiro.dev docs + live additionalModelRequestFieldsSchema): +// 4.6 Claude models cap at low|medium|high|max; 4.7+ add xhigh. Unknown models stay conservative. +function kiroModelLacksXhigh(model) { + const v = parseClaudeVersion(model); + return !v || (v.major === 4 && v.minor !== null && v.minor <= 6); +} + +export function extractKiroEffortLevel(body, model) { const effort = body?.output_config?.effort ?? body?.reasoning_effort ?? @@ -179,7 +194,8 @@ export function extractKiroEffortLevel(body) { if (typeof effort !== "string") return null; const normalized = effort.toLowerCase(); if (normalized === "none" || normalized === "off" || normalized === "disabled") return null; - if (normalized === "xhigh" || normalized === "max") return "high"; + if (normalized === "xhigh") return kiroModelLacksXhigh(model) ? "high" : "xhigh"; + if (normalized === "max") return "max"; if (["low", "medium", "high"].includes(normalized)) return normalized; return null; } @@ -199,10 +215,10 @@ function extractKiroGptEffortLevel(body) { return null; } -export function buildKiroAdditionalModelRequestFields(body, effortPath = "output_config") { +export function buildKiroAdditionalModelRequestFields(body, effortPath = "output_config", model) { const effort = effortPath === "reasoning" ? extractKiroGptEffortLevel(body) - : extractKiroEffortLevel(body); + : extractKiroEffortLevel(body, model); if (!effort) return undefined; if (effortPath === "reasoning") { // Mirrors Kiro CLI/KAS buildEffortRequestFields("reasoning") for GPT. @@ -222,11 +238,9 @@ export function resolveKiroEffortPath(model) { return "reasoning"; } if (!normalized.includes("claude")) return null; - const match = normalized.match(/(?:^|[/.])claude(?:[/.][a-z]+)*[/.](\d+)(?:[/.](\d+))?(?:[/.]|$)/); - if (!match) return null; - const [, majorText, minorText] = match; - const major = Number(majorText); - const minor = minorText === undefined ? null : Number(minorText); + const v = parseClaudeVersion(model); + if (!v) return null; + const { major, minor } = v; const dateSuffixMinor = minor !== null && minor >= 1000; // Kiro rejected additionalModelRequestFields on legacy 4.5 models in live smoke. // Default future Claude/Kiro models to supported so new model releases do not @@ -248,7 +262,7 @@ export function usesKiroNativeGptEffort(body, model) { export function buildKiroAdditionalModelRequestFieldsForModel(body, model) { const effortPath = resolveKiroEffortPath(model); if (!effortPath) return undefined; - return buildKiroAdditionalModelRequestFields(body, effortPath); + return buildKiroAdditionalModelRequestFields(body, effortPath, model); } /** diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index 0b020100..5ecf2854 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -10,8 +10,8 @@ const L = { base: ["none", "low", "medium", "high"], // qwen, step, hunyuan, gemini-budget onOff: ["none", "thinking"], // zai (binary), minimax (adaptive) openai: ["none", "minimal", "low", "medium", "high", "xhigh"], // GPT-5.x / o-series (no "max") - levelMax: ["none", "low", "medium", "high", "max"], // claude-adaptive, kimi - budgetX: ["none", "low", "medium", "high", "xhigh", "max"], // claude-budget + levelMax: ["none", "low", "medium", "high", "max"], // kimi + budgetX: ["none", "low", "medium", "high", "xhigh", "max"], // claude-budget, claude-adaptive gemini: ["minimal", "low", "medium", "high"], // gemini-3 thinkingLevel (no disable) hiMax: ["none", "high", "max"], // deepseek (low/med→high, xhigh→max) }; @@ -19,7 +19,7 @@ const L = { // thinkingFormat → valid selectable levels (source of truth for UI options). const FORMAT_LEVELS = { openai: L.openai, - "claude-adaptive": L.levelMax, + "claude-adaptive": L.budgetX, "claude-budget": L.budgetX, "gemini-level": L.gemini, "gemini-budget": L.base, @@ -35,8 +35,13 @@ const FORMAT_LEVELS = { const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"]; +// Opus/Sonnet 4.6 lack xhigh (Anthropic + Kiro docs) — keep the 4-level+max set. +const CLAUDE_NO_XHIGH = ["none", "low", "medium", "high", "max"]; + // Model-name pattern overrides (glob, first match wins) — more precise than format default. const PATTERN_THINKING = [ + { pattern: "*claude*4.6*", levels: CLAUDE_NO_XHIGH }, + { pattern: "*claude*4-6*", levels: CLAUDE_NO_XHIGH }, { provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS }, { provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, diff --git a/open-sse/translator/concerns/thinkingUnified.js b/open-sse/translator/concerns/thinkingUnified.js index 5ce73b32..4bcacd3b 100644 --- a/open-sse/translator/concerns/thinkingUnified.js +++ b/open-sse/translator/concerns/thinkingUnified.js @@ -271,7 +271,9 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels, display) { if (canDisable) body.thinking = { type: "adaptive", ...(display ? { display } : {}) }; else delete body.thinking; const level = toLevel(eff); - body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level }; + // xhigh is model-gated (Opus/Sonnet 4.6 reject it) — clamp when not advertised. + body.output_config = { effort: level === "auto" ? "high" + : level === "xhigh" && !supportedLevels?.includes("xhigh") ? "high" : level }; break; } case "claude-budget": { diff --git a/tests/unit/thinking-levels-kiro.test.js b/tests/unit/thinking-levels-kiro.test.js index edb3b7aa..b1ff2acb 100644 --- a/tests/unit/thinking-levels-kiro.test.js +++ b/tests/unit/thinking-levels-kiro.test.js @@ -1,5 +1,7 @@ import { describe, it, expect } from "vitest"; import { getThinkingLevels } from "../../open-sse/providers/thinkingLevels.js"; +import { buildKiroAdditionalModelRequestFieldsForModel } from "../../open-sse/config/kiroConstants.js"; +import { applyThinking } from "../../open-sse/translator/concerns/thinkingUnified.js"; describe("getThinkingLevels for Kiro", () => { it("does not advertise native intensity for legacy Kiro models", () => { @@ -9,6 +11,25 @@ describe("getThinkingLevels for Kiro", () => { it("advertises native levels for supported Kiro models", () => { expect(getThinkingLevels("kiro", "claude-sonnet-5")).toContain("high"); + expect(getThinkingLevels("kiro", "claude-sonnet-5")).toContain("xhigh"); + expect(getThinkingLevels("kiro", "claude-sonnet-5")).toContain("max"); expect(getThinkingLevels("kiro", "gpt-5.6-sol")).toContain("xhigh"); }); + + it("omits xhigh on 4.6 models (upstream rejects it there)", () => { + for (const model of ["claude-opus-4.6", "claude-opus-4-6", "claude-sonnet-4.6"]) { + expect(getThinkingLevels("kiro", model)).not.toContain("xhigh"); + expect(getThinkingLevels("kiro", model)).toContain("max"); + } + }); + + it("passes xhigh/max through on the wire for 4.7+, clamps xhigh on 4.6", () => { + const xhigh = { output_config: { effort: "xhigh" } }; + expect(buildKiroAdditionalModelRequestFieldsForModel(xhigh, "claude-sonnet-5")?.output_config?.effort).toBe("xhigh"); + expect(buildKiroAdditionalModelRequestFieldsForModel({ output_config: { effort: "max" } }, "claude-opus-4.6")?.output_config?.effort).toBe("max"); + expect(buildKiroAdditionalModelRequestFieldsForModel(xhigh, "claude-opus-4.6")?.output_config?.effort).toBe("high"); + // Anthropic-wire path: suffix override sends real xhigh on 4.7+, high on 4.6. + expect(applyThinking("claude", "claude-opus-5.5(xhigh)", { messages: [] }, "claude").output_config?.effort).toBe("xhigh"); + expect(applyThinking("claude", "claude-opus-4.6(xhigh)", { messages: [] }, "claude").output_config?.effort).toBe("high"); + }); }); From ccd0677dc1fd4a31cc75013c417f53b6b0bdeff6 Mon Sep 17 00:00:00 2001 From: Amirsalar Sojoudi Date: Thu, 1 Oct 2026 10:06:58 +0700 Subject: [PATCH 30/41] fix(claude): resolve Sonnet 5.x to adaptive thinking so no forged thinking placeholders are sent --- open-sse/providers/capabilities.js | 1 + tests/unit/capabilities-sonnet-5-5.test.js | 21 +++++++++++++++++++++ 2 files changed, 22 insertions(+) create mode 100644 tests/unit/capabilities-sonnet-5-5.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 1e4240c8..39d116ee 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -286,6 +286,7 @@ PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex; export const PATTERN_CAPABILITIES = [ // ── Claude (4.6+ = adaptive thinking; older/haiku = budget) ────── { pattern: "*claude*opus-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 } }, + { pattern: "*claude*sonnet-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 } }, { pattern: "*claude*opus-4.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive" } }, { pattern: "*claude*opus-4.7*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive" } }, { pattern: "*claude*opus-4.8*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive" } }, diff --git a/tests/unit/capabilities-sonnet-5-5.test.js b/tests/unit/capabilities-sonnet-5-5.test.js new file mode 100644 index 00000000..824bc309 --- /dev/null +++ b/tests/unit/capabilities-sonnet-5-5.test.js @@ -0,0 +1,21 @@ +import { describe, expect, it } from "vitest"; + +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; + +// claude-sonnet-5-5 has no exact entry and matched only the generic *claude*sonnet* +// pattern (claude-budget). That sent thinking.type "enabled" and made the translator +// forge signed thinking placeholders on every tool_use turn; Sonnet 5.5 answered +// large Codex conversations with stop_reason "refusal". It must resolve like the +// rest of the 5.x family: adaptive thinking, 1M context. +describe("Claude Sonnet 5.5 capabilities", () => { + for (const model of ["claude-sonnet-5-5", "claude-sonnet-5.5", "anthropic/claude-sonnet-5-5"]) { + it(`${model} resolves to adaptive thinking + 1M context`, () => { + expect(getCapabilitiesForModel("claude", model)).toMatchObject({ + thinkingFormat: "claude-adaptive", + contextWindow: 1000000, + maxOutput: 128000, + reasoning: true, + }); + }); + } +}); From 49ba54b2baf9190472181272a7efa8edeba66cf4 Mon Sep 17 00:00:00 2001 From: MrBeanDev Date: Thu, 1 Oct 2026 10:07:10 +0700 Subject: [PATCH 31/41] feat(claude): add Claude Sonnet 5.5 - registry + exact capability entry (adaptive thinking, 1M context, 128k output) - explicit Sonnet 5 pricing for claude-sonnet-5-5 and claude-sonnet-5 ($2/$10 per 1M) - Sonnet 5.5 API restrictions: thinking "disabled" rewritten to "between_tools" (effort clamped to high), forced tool_choice any/tool rewritten to auto --- open-sse/providers/capabilities.js | 2 + open-sse/providers/pricing.js | 2 + open-sse/providers/registry/claude.js | 1 + open-sse/translator/formats/claude.js | 17 +++++- tests/unit/claude-sonnet-5-5.test.js | 74 +++++++++++++++++++++++++++ 5 files changed, 95 insertions(+), 1 deletion(-) create mode 100644 tests/unit/claude-sonnet-5-5.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 39d116ee..68e4fa0a 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -104,6 +104,8 @@ export const MODEL_CAPABILITIES = { "claude-opus-4-8-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-sonnet-4.6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-sonnet-4-6": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, + // Sonnet 5.5 rejects thinking.type "disabled" (use "between_tools") and forced tool_choice (any/tool). + "claude-sonnet-5-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000, thinkingOffType: "between_tools", forcedToolChoice: false }, "claude-sonnet-5": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-sonnet-5-thinking": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, "claude-sonnet-5-agentic": { vision: true, reasoning: true, search: true, thinkingFormat: "claude-adaptive", contextWindow: 1000000, maxOutput: 128000 }, diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index a57e6bda..4b62fa86 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -49,6 +49,8 @@ export const MODEL_PRICING = { "claude-opus-4-5-thinking": { input: 5.00, output: 25.00, cached: 0.50, reasoning: 37.50, cache_creation: 5.00 }, "claude-opus-4-6-thinking": { input: 5.00, output: 25.00, cached: 0.50, reasoning: 37.50, cache_creation: 5.00 }, "claude-fable-5": { input: 10.00, output: 50.00, cached: 1.00, reasoning: 50.00, cache_creation: 12.50 }, + "claude-sonnet-5-5": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 }, + "claude-sonnet-5": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 }, // === OpenAI / GPT === "gpt-3.5-turbo": { input: 0.50, output: 1.50, cached: 0.25, reasoning: 2.25, cache_creation: 0.50 }, diff --git a/open-sse/providers/registry/claude.js b/open-sse/providers/registry/claude.js index 3677ffb5..bbf9f1fb 100644 --- a/open-sse/providers/registry/claude.js +++ b/open-sse/providers/registry/claude.js @@ -63,6 +63,7 @@ export default { { id: "claude-opus-5", name: "Claude Opus 5" }, { id: "claude-fable-5-1", name: "Claude Fable 5.1" }, { id: "claude-fable-5", name: "Claude Fable 5" }, + { id: "claude-sonnet-5-5", name: "Claude Sonnet 5.5" }, { id: "claude-sonnet-5", name: "Claude Sonnet 5" }, { id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" }, ], diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index a9bbd19d..8fa44ccf 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -454,12 +454,27 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne delete body.output_config; } + // Models whose API rejects thinking "disabled" and forced tool use with a 400 + // (Sonnet 5.5). Runs on every Claude-bound body, so OpenAI clients, native + // passthrough and the provider-level "off" override are all covered. + const modelCaps = getCapabilitiesForModel(provider, body.model); + if (modelCaps.thinkingOffType && body.thinking?.type === "disabled") { + body.thinking = { type: modelCaps.thinkingOffType }; + // between_tools only accepts effort up to high. + const effort = body.output_config?.effort; + if (effort === "xhigh" || effort === "max") body.output_config.effort = "high"; + } + if (modelCaps.forcedToolChoice === false && (body.tool_choice?.type === "any" || body.tool_choice?.type === "tool")) { + const { disable_parallel_tool_use } = body.tool_choice; + body.tool_choice = { type: "auto", ...(disable_parallel_tool_use !== undefined ? { disable_parallel_tool_use } : {}) }; + } + // Clamp max_tokens to the model's real output ceiling. Models whose caps // declare a higher maxOutput (e.g. Opus 4.8 / Sonnet 4.6 = 128000) are allowed // up to it, so max-effort thinking gets full budget; others fall back to the // conservative 64000 default. if (body.max_tokens) { - const ceiling = getCapabilitiesForModel(provider, body.model).maxOutput || DEFAULT_MAX_TOKENS; + const ceiling = modelCaps.maxOutput || DEFAULT_MAX_TOKENS; if (body.max_tokens > ceiling) body.max_tokens = ceiling; // Reconcile against thinking budget. applyThinking (thinkingUnified.js) runs diff --git a/tests/unit/claude-sonnet-5-5.test.js b/tests/unit/claude-sonnet-5-5.test.js new file mode 100644 index 00000000..527759ab --- /dev/null +++ b/tests/unit/claude-sonnet-5-5.test.js @@ -0,0 +1,74 @@ +import { describe, expect, it } from "vitest"; + +import { getModelsByProviderId } from "../../open-sse/config/providerModels.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { getPricingForModel } from "../../open-sse/providers/pricing.js"; +import { prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; +import { translateRequest } from "../../open-sse/translator/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import "../translator/registerAll.js"; + +// Sonnet 5.5 keeps Sonnet 5's API price ($2 / $10 per 1M) and the 5.x +// adaptive-thinking family. Without explicit rows both fell through to the +// generic claude-sonnet-* pattern: $3 / $15 and budget thinking. +describe("Claude Sonnet 5.5", () => { + it("is listed for the claude provider", () => { + expect(getModelsByProviderId("claude").some((model) => model.id === "claude-sonnet-5-5")).toBe(true); + }); + + it("resolves to adaptive thinking with a 1M context", () => { + expect(getCapabilitiesForModel("claude", "claude-sonnet-5-5")).toMatchObject({ + reasoning: true, + thinkingFormat: "claude-adaptive", + contextWindow: 1000000, + maxOutput: 128000, + }); + }); + + it.each(["claude-sonnet-5-5", "claude-sonnet-5"])("prices %s at Sonnet 5 rates", (model) => { + expect(getPricingForModel("claude", model)).toEqual({ input: 2, output: 10, cached: 0.2, reasoning: 10, cache_creation: 2.5 }); + }); +}); + +// Sonnet 5.5 returns 400 for thinking.type "disabled" and for forced tool use. +describe("Claude Sonnet 5.5 request shape", () => { + const prepare = (body) => prepareClaudeRequest({ max_tokens: 1024, messages: [{ role: "user", content: "hi" }], ...body }, "claude"); + + it("turns thinking off with between_tools, clamping effort to high", () => { + const body = prepare({ model: "claude-sonnet-5-5", thinking: { type: "disabled" }, output_config: { effort: "max" } }); + expect(body.thinking).toEqual({ type: "between_tools" }); + expect(body.output_config.effort).toBe("high"); + }); + + it("maps forced tool_choice to auto", () => { + expect(prepare({ model: "claude-sonnet-5-5", tool_choice: { type: "any" } }).tool_choice).toEqual({ type: "auto" }); + expect(prepare({ model: "claude-sonnet-5-5", tool_choice: { type: "tool", name: "run", disable_parallel_tool_use: true } }).tool_choice) + .toEqual({ type: "auto", disable_parallel_tool_use: true }); + }); + + it("leaves other models untouched", () => { + const body = prepare({ model: "claude-sonnet-5", thinking: { type: "disabled" }, tool_choice: { type: "any" } }); + expect(body.thinking).toEqual({ type: "disabled" }); + expect(body.tool_choice).toEqual({ type: "any" }); + }); + + it("covers the native Claude passthrough path end to end", () => { + const out = translateRequest(FORMATS.CLAUDE, FORMATS.CLAUDE, "claude-sonnet-5-5", { + model: "claude-sonnet-5-5", max_tokens: 1000, thinking: { type: "disabled" }, tool_choice: { type: "any" }, + tools: [{ name: "run", input_schema: { type: "object", properties: {} } }], + messages: [{ role: "user", content: "hi" }], + }, true, null, "claude"); + expect(out.thinking).toEqual({ type: "between_tools" }); + expect(out.tool_choice).toEqual({ type: "auto" }); + }); + + it("covers the OpenAI-client path end to end", () => { + const out = translateRequest(FORMATS.OPENAI, FORMATS.CLAUDE, "claude-sonnet-5-5", { + model: "claude-sonnet-5-5", reasoning_effort: "none", tool_choice: "required", + tools: [{ type: "function", function: { name: "run", parameters: { type: "object", properties: {} } } }], + messages: [{ role: "user", content: "hi" }], + }, true, null, "claude"); + expect(out.thinking).toEqual({ type: "between_tools" }); + expect(out.tool_choice.type).toBe("auto"); + }); +}); From ca6e8407f105ca6d7e26633f0cac87e8ba6299ae Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Thu, 1 Oct 2026 10:09:48 +0700 Subject: [PATCH 32/41] fix(codex): refresh CLI identity for GPT-6.1 Sol MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump CODEX_CLI_VERSION 0.155.0 → 0.159.0 (npm stable) so OpenAI stops rejecting gpt-6.1-sol on connected ChatGPT accounts. Update transport header assertions and assert the User-Agent identity. Fixes #4471 --- open-sse/providers/registry/codex.js | 2 +- tests/unit/codex-gpt6-lite.test.js | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index f01153db..15a38784 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -2,7 +2,7 @@ import { withCodexReviewModels } from "../models/helpers.js"; // Codex CLI version seen by OpenAI's backend — single source for the Version / // User-Agent identity headers. Bump when the installed codex CLI is upgraded. -const CODEX_CLI_VERSION = "0.155.0"; +const CODEX_CLI_VERSION = "0.159.0"; const GPT_6_LITE_THINKING_LEVELS = ["low", "medium", "high", "xhigh", "max"]; export default { diff --git a/tests/unit/codex-gpt6-lite.test.js b/tests/unit/codex-gpt6-lite.test.js index 25182652..77cb8ed7 100644 --- a/tests/unit/codex-gpt6-lite.test.js +++ b/tests/unit/codex-gpt6-lite.test.js @@ -190,7 +190,8 @@ describe("Codex GPT-6 Sol/Luna transport", () => { const body = JSON.parse(options.body); expect(url).toBe("https://chatgpt.com/backend-api/codex/responses"); expect(options.headers["x-openai-internal-codex-responses-lite"]).toBe("true"); - expect(options.headers.version).toBe("0.155.0"); + expect(options.headers.version).toBe("0.159.0"); + expect(options.headers["User-Agent"]).toBe("codex_cli_rs/0.159.0"); expect(body.model).toBe("gpt-6-luna"); expect(body.instructions).toBe(""); expect(body.input[0].type).toBe("additional_tools"); From 31704db73e77875dd3509203819275e0e1f60ae7 Mon Sep 17 00:00:00 2001 From: B1nh M1nh <43268322+b1nhm1nh@users.noreply.github.com> Date: Thu, 1 Oct 2026 10:13:37 +0700 Subject: [PATCH 33/41] feat(cli): add connect command for remote 9router servers --- cli/README.md | 18 ++ cli/cli.js | 15 ++ cli/hooks/postinstall.js | 4 + cli/package.json | 3 +- cli/src/cli/commands/connect.js | 269 +++++++++++++++++++++ cli/src/cli/commands/connectTools.js | 338 +++++++++++++++++++++++++++ tests/unit/cli-connect.test.js | 169 ++++++++++++++ 7 files changed, 815 insertions(+), 1 deletion(-) create mode 100644 cli/src/cli/commands/connect.js create mode 100644 cli/src/cli/commands/connectTools.js create mode 100644 tests/unit/cli-connect.test.js diff --git a/cli/README.md b/cli/README.md index 19ad230f..60d9be55 100644 --- a/cli/README.md +++ b/cli/README.md @@ -90,6 +90,24 @@ That's it! Start coding with FREE AI models. --- +## 🔌 Connect to a Remote 9Router + +Already running 9Router on another machine (e.g. a team server on your LAN)? Point this machine's CLI tools at it — no local server is started: + +```bash +npx 9router connect http://:20128 # pick tools interactively +npx 9router connect http://:20128 --tools claude,codex # or choose up front +npx 9router connect --reset --tools claude,codex # undo +``` + +It logs in with the dashboard password (hidden prompt), reuses or creates an API key named `cli-`, and writes each tool's config (backing up the original once as `*.bak-9router`). + +Supported: `claude`, `codex`, `opencode`, `droid`, `crush`, `kilo`, `cline`, or `all`. Other options: `--model`, `--opus/--sonnet/--haiku/--fable`, `--api-key`, `--key-name`, `--print-env`. See `9router connect --help`. + +> ⚠️ Over plain `http://` the password and API key are sent unencrypted — use a trusted LAN/VPN or put HTTPS in front. The API key is stored in each tool's config file. + +--- + ## 🛠️ Supported CLI Tools Claude-Code • OpenClaw • Codex • OpenCode • Cursor • Antigravity • Cline • Continue • Droid • Roo • Copilot • Kilo Code • Gemini CLI • Qwen Code • iFlow • Crush • Crusher • Aider diff --git a/cli/cli.js b/cli/cli.js index 703039dd..f3032f9b 100755 --- a/cli/cli.js +++ b/cli/cli.js @@ -80,6 +80,19 @@ if (args[0] === "xai" && args[1] === "video") { return; } +// `9router connect ` configures local CLI tools against a remote server — +// no local server, no runtime deps. Usable via `npx 9router connect …`. +if (args[0] === "connect") { + const { run } = require("./src/cli/commands/connect"); + run(args.slice(1)) + .then((code) => process.exit(code)) + .catch((err) => { + console.error(`❌ ${err?.message || err}`); + process.exit(1); + }); + return; +} + // Self-heal SQLite runtime deps (sql.js + better-sqlite3) into ~/.9router/runtime // so the server can resolve them via NODE_PATH. Best-effort — sql.js is required, // better-sqlite3 is optional. Logs to stderr only on failure. @@ -154,6 +167,8 @@ Options: -v, --version Show version Commands: + connect Configure Claude Code for a remote 9router server + (npx 9router connect http://host:20128 — no install needed) xai video --prompt "..." --output video.mp4 Generate a Grok Imagine video via the running gateway (see: ${APP_NAME} xai video --help) diff --git a/cli/hooks/postinstall.js b/cli/hooks/postinstall.js index 3a59332f..728b9406 100644 --- a/cli/hooks/postinstall.js +++ b/cli/hooks/postinstall.js @@ -3,6 +3,10 @@ // Postinstall: warm-up SQLite deps into ~/.9router/runtime so the first // `9router` start doesn't need network. Failure here is non-fatal — // cli.js will retry at runtime if anything is missing. +// `npx 9router …` (npm_command=exec) is typically a one-shot `connect` — skip +// the runtime warm-up; cli.js self-heals it if the server is started later. +if (process.env.npm_command === "exec") process.exit(0); + const { ensureSqliteRuntime } = require("./sqliteRuntime"); const { ensureTrayRuntime } = require("./trayRuntime"); diff --git a/cli/package.json b/cli/package.json index 1a67862a..86755891 100644 --- a/cli/package.json +++ b/cli/package.json @@ -27,7 +27,8 @@ "node-forge": "^1.3.3", "node-machine-id": "^1.1.12", "react": "19.2.1", - "react-dom": "19.2.1" + "react-dom": "19.2.1", + "confbox": "^0.2.4" }, "comment_sqlite": "sql.js + better-sqlite3 are NOT bundled here. They are installed into ~/.9router/runtime/node_modules by hooks/postinstall.js (and re-checked at runtime by cli.js). This avoids Windows EBUSY errors when updating the global CLI, since native .node files no longer live under the locked install dir.", "comment_systray": "systray2 is NOT bundled here. It is lazy-installed into ~/.9router/runtime/node_modules by hooks/postinstall.js on macOS/Linux only. Windows uses PowerShell NotifyIcon (zero binary). This avoids shipping unsigned Go binaries that trigger antivirus false positives (Kaspersky). We use the systray2 fork because the legacy systray@1.0.5 ships a 2017 x86_64 binary that fails to load on modern macOS dyld. Neither package ships an arm64 macOS binary, so on Apple Silicon hooks/trayRuntime.js overlays our own arm64 build from the tray-binaries GitHub release; without it the tray requires Rosetta 2.", diff --git a/cli/src/cli/commands/connect.js b/cli/src/cli/commands/connect.js new file mode 100644 index 00000000..700d3cb9 --- /dev/null +++ b/cli/src/cli/commands/connect.js @@ -0,0 +1,269 @@ +/** + * `9router connect ` — point local CLI tools (Claude Code, Codex, …) at a + * REMOTE 9router server. Nothing runs locally: we log in with the dashboard + * password, reuse/create an API key for this machine, then write the tool's + * settings files (see connectTools.js). Works via `npx 9router connect …` with no global install. + * + * The password and API key are never printed (key is masked). + */ + +const os = require("os"); +const { TOOL_IDS, CLAUDE_MODELS, resolveTools } = require("./connectTools"); + +const DEFAULT_MODEL = "cc/claude-sonnet-5"; + +const HELP = ` +Usage: 9router connect [options] + +Configure CLI tools on THIS machine to use a remote 9router server. +No local server is started. Run without installing: + + npx 9router connect http://:20128 + npx 9router connect http://:20128 --tools claude,codex,opencode + +Options: + --tools Comma-separated tools to configure (prompted if omitted + in a terminal; default: claude). Supported: + ${TOOL_IDS.join(", ")}, all + --password Dashboard password (or env NINE_ROUTER_PASSWORD; + prompted if omitted — preferred, keeps it out of shell history) + --key-name API key name to reuse/create (default: cli-) + --api-key Use this API key, skip login + key lookup + --model Model for non-Claude tools (default: ${DEFAULT_MODEL}) + --fable|--opus|--sonnet|--haiku + Override Claude Code model mapping + --print-env Also print OpenAI-compatible env vars for other CLIs + --reset Remove 9router settings from the selected tools and exit + -h, --help Show this help +`; + +function parseArgs(argv) { + const opts = { + password: process.env.NINE_ROUTER_PASSWORD || null, + keyName: `cli-${os.hostname()}`.slice(0, 64), + apiKey: null, + models: {}, + }; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + const next = () => { + const v = argv[++i]; + if (v === undefined) throw new Error(`Missing value for ${a}`); + return v; + }; + if (a === "--password") opts.password = next(); + else if (a === "--key-name") opts.keyName = next(); + else if (a === "--api-key") opts.apiKey = next(); + else if (a === "--tools") opts.tools = next().split(","); + else if (a === "--model") opts.model = next(); + else if (a === "--print-env") opts.printEnv = true; + else if (a === "--reset") opts.reset = true; + else if (a === "-h" || a === "--help") opts.help = true; + else if (a.startsWith("--") && CLAUDE_MODELS.some((m) => `--${m.flag}` === a)) opts.models[a.slice(2)] = next(); + else if (!a.startsWith("-") && !opts.url) opts.url = a; + else throw new Error(`Unknown option: ${a}`); + } + return opts; +} + +function normalizeServerUrl(input) { + let raw = String(input || "").trim(); + if (!/^https?:\/\//i.test(raw)) raw = `http://${raw}`; + const u = new URL(raw); + // Accept pasted dashboard/API URLs: keep only origin. + return u.origin; +} + +function maskKey(key) { + if (!key || key.length < 12) return "****"; + return `${key.slice(0, 6)}…${key.slice(-4)}`; +} + +// Enquirer rejects with an empty value on Ctrl+C / Esc. +class Cancelled extends Error {} +const prompt = (p) => p.run().catch((err) => { throw err || new Cancelled("Cancelled"); }); + +async function promptPassword() { + if (!process.stdin.isTTY) throw new Error("Password required: pass --password or set NINE_ROUTER_PASSWORD"); + const { Password } = require("enquirer"); + return prompt(new Password({ message: "9router dashboard password" })); +} + +async function request(url, { method = "GET", body, cookie, apiKey } = {}) { + const headers = { Accept: "application/json" }; + if (body) headers["Content-Type"] = "application/json"; + if (cookie) headers.Cookie = cookie; + if (apiKey) headers.Authorization = `Bearer ${apiKey}`; + let res; + try { + res = await fetch(url, { method, headers, body: body ? JSON.stringify(body) : undefined, redirect: "manual" }); + } catch (err) { + throw new Error(`Cannot reach ${new URL(url).origin}: ${err.cause?.code || err.message}`); + } + let data = null; + try { data = await res.json(); } catch { /* non-JSON */ } + // Server-controlled strings get printed later — strip control chars (terminal escape injection). + if (typeof data?.error === "string") data.error = data.error.replace(/[\x00-\x1f\x7f]/g, ""); + return { status: res.status, headers: res.headers, data }; +} + +function extractAuthCookie(headers) { + const list = typeof headers.getSetCookie === "function" ? headers.getSetCookie() : [headers.get("set-cookie") || ""]; + for (const c of list) { + const m = /(?:^|,\s*)auth_token=([^;]+)/.exec(c); + if (m) return `auth_token=${m[1]}`; + } + return null; +} + +async function login(server, password) { + const res = await request(`${server}/api/auth/login`, { method: "POST", body: { password } }); + if (res.status === 200 && res.data?.success) { + const cookie = extractAuthCookie(res.headers); + if (!cookie) throw new Error("Login succeeded but server returned no session cookie"); + return cookie; + } + throw new Error(`Login failed (${res.status}): ${res.data?.error || "unknown error"}`); +} + +async function getOrCreateApiKey(server, cookie, keyName) { + const list = await request(`${server}/api/keys`, { cookie }); + if (list.status === 401) throw new Error("Unauthorized listing API keys — wrong password or session rejected"); + if (list.status !== 200) throw new Error(`Failed to list API keys (${list.status}): ${list.data?.error || ""}`); + const keys = (list.data?.keys || []).filter((k) => k.isActive !== false); + const existing = keys.find((k) => k.name === keyName); + if (existing) return { key: existing.key, created: false }; + + const created = await request(`${server}/api/keys`, { method: "POST", cookie, body: { name: keyName } }); + if (created.status !== 201 || !created.data?.key) { + throw new Error(`Failed to create API key (${created.status}): ${created.data?.error || ""}`); + } + return { key: created.data.key, created: true }; +} + +async function listModels(server, apiKey) { + const res = await request(`${server}/v1/models`, { apiKey }); + if (res.status === 401) throw new Error("API key rejected by server (/v1/models returned 401)"); + if (res.status !== 200) return null; + return new Set((res.data?.data || []).map((m) => m.id)); +} + +async function promptTools() { + const { MultiSelect } = require("enquirer"); + const { TOOLS } = require("./connectTools"); + return prompt(new MultiSelect({ + message: "Select CLI tools to configure (space to toggle, enter to confirm)", + choices: TOOLS.map((t) => ({ name: t.id, message: t.name, hint: t.paths()[0], enabled: t.id === "claude" })), + validate: (v) => v.length > 0 || "Select at least one tool", + })); +} + +async function selectTools(opts) { + if (opts.tools) return resolveTools(opts.tools); + if (process.stdin.isTTY) return resolveTools(await promptTools()); + return resolveTools(["claude"]); +} + +async function run(argv) { + try { + return await runConnect(argv); + } catch (err) { + if (err instanceof Cancelled) { + console.log("Cancelled."); + return 130; + } + throw err; + } +} + +async function runConnect(argv) { + const opts = parseArgs(argv); + if (opts.help) { + console.log(HELP); + return 0; + } + if (!opts.reset && !opts.url) { + console.log(HELP); + return 1; + } + + // Pick tools before any network call: bad names fail fast, and cancelling + // the picker never leaves a freshly created key on the server. + const tools = await selectTools(opts); + + if (opts.reset) { + let failed = 0; + for (const t of tools) { + try { + const files = await t.reset(); + console.log(files.length ? `✅ ${t.name}: removed 9router settings (${files.join(", ")})` : `• ${t.name}: nothing to reset`); + } catch (err) { + failed++; + console.log(`❌ ${t.name}: ${err.message}`); + } + } + return failed ? 1 : 0; + } + + const server = normalizeServerUrl(opts.url); + const isRemoteHttp = server.startsWith("http://") && !/^http:\/\/(localhost|127\.0\.0\.1|\[::1\])(:|$)/.test(server); + if (isRemoteHttp) { + console.log("\x1b[33m⚠ Plain HTTP: password and API key travel unencrypted. Use only on a trusted LAN/VPN.\x1b[0m"); + } + + let apiKey = opts.apiKey; + if (apiKey) { + console.log("• Using provided API key"); + } else { + const password = opts.password ?? (await promptPassword()); + console.log(`• Logging in to ${server}`); + const cookie = await login(server, password); + const result = await getOrCreateApiKey(server, cookie, opts.keyName); + apiKey = result.key; + console.log(`• ${result.created ? "Created" : "Reusing"} API key "${opts.keyName}" (${maskKey(apiKey)})`); + } + + const available = await listModels(server, apiKey); + const warnMissing = (label, model, flag) => { + if (available && !available.has(model)) { + console.log(`\x1b[33m⚠ ${label}: "${model}" not listed by server — override with ${flag} \x1b[0m`); + } + }; + + const claudeModels = {}; + if (tools.some((t) => t.id === "claude")) { + for (const m of CLAUDE_MODELS) { + claudeModels[m.envKey] = opts.models[m.flag] || m.defaultValue; + warnMissing(`claude ${m.flag}`, claudeModels[m.envKey], `--${m.flag}`); + } + } + const model = opts.model || DEFAULT_MODEL; + if (tools.some((t) => t.id !== "claude")) warnMissing("model", model, "--model"); + + const ctx = { baseUrl: server, apiKey, model, claudeModels }; + let failed = 0; + for (const t of tools) { + try { + const files = await t.apply(ctx); + console.log(`✅ ${t.name} → ${files.join(", ")}`); + } catch (err) { + failed++; + console.log(`❌ ${t.name}: ${err.message}`); + } + } + console.log(` Base URL: ${server}/v1`); + if (tools.some((t) => t.id === "claude")) { + for (const m of CLAUDE_MODELS) console.log(` ${m.envKey}=${claudeModels[m.envKey]}`); + } + if (tools.some((t) => t.id !== "claude")) console.log(` Model (other tools): ${model}`); + console.log(` Restart the tools to apply. Undo: npx 9router connect --reset --tools ${tools.map((t) => t.id).join(",")}`); + + if (opts.printEnv) { + console.log("\nOpenAI-compatible CLIs (codex, opencode, aider, …):"); + console.log(` OPENAI_BASE_URL=${server}/v1`); + console.log(` OPENAI_API_KEY=${apiKey}`); + } + return failed ? 1 : 0; +} + +module.exports = { run, __test__: { parseArgs, normalizeServerUrl, extractAuthCookie, maskKey, Cancelled } }; diff --git a/cli/src/cli/commands/connectTools.js b/cli/src/cli/commands/connectTools.js new file mode 100644 index 00000000..aa7cb6a0 --- /dev/null +++ b/cli/src/cli/commands/connectTools.js @@ -0,0 +1,338 @@ +/** + * Per-tool config writers for `9router connect`. File layout and keys mirror + * the dashboard routes in src/app/api/cli-tools/-settings/route.js so a + * tool configured here looks identical to one applied from the dashboard. + * + * Each tool: { id, name, paths(), apply(ctx) → string[] written, reset() → string[] touched } + * ctx: { baseUrl (origin, no /v1), apiKey, model, claudeModels: { envKey: model } } + */ + +const fs = require("fs"); +const path = require("path"); +const os = require("os"); + +const home = () => os.homedir(); +const v1 = (baseUrl) => (baseUrl.endsWith("/v1") ? baseUrl : `${baseUrl}/v1`); + +// Drop trailing commas (JSONC) outside string literals, so values like "a,}" survive. +function stripTrailingCommas(text) { + return text.replace(/("(?:\\.|[^"\\])*")|,(\s*[}\]])/g, (m, str, tail) => str ?? tail); +} + +function readJson(file) { + try { + return JSON.parse(stripTrailingCommas(fs.readFileSync(file, "utf8"))); + } catch (err) { + if (err.code === "ENOENT") return null; + throw new Error(`Cannot parse ${file}: ${err.message}`); + } +} + +// Files hold the API key → owner-only (0600) on POSIX; no-op on Windows. +const SECRET_MODE = 0o600; + +// Rewrite an existing file on reset (no backup, existing mode kept). +function rewriteFile(file, content) { + fs.writeFileSync(file, content, { mode: SECRET_MODE }); +} + +// One-time backup of the user's pre-9router file, then write. +function writeFile(file, content) { + fs.mkdirSync(path.dirname(file), { recursive: true }); + const backup = `${file}.bak-9router`; + if (fs.existsSync(file) && !fs.existsSync(backup)) { + fs.copyFileSync(file, backup); + fs.chmodSync(backup, SECRET_MODE); + } + fs.writeFileSync(file, content, { mode: SECRET_MODE }); + fs.chmodSync(file, SECRET_MODE); // mode above only applies when creating +} + +const writeJson = (file, data) => writeFile(file, JSON.stringify(data, null, 2)); + +// confbox is ESM-only; loaded lazily so non-codex runs don't need it. +async function toml() { + return import("confbox"); +} + +// ── Claude Code ───────────────────────────────────────────────────────────── +const CLAUDE_MODELS = [ + { flag: "fable", envKey: "ANTHROPIC_DEFAULT_FABLE_MODEL", defaultValue: "cc/claude-fable-5" }, + { flag: "opus", envKey: "ANTHROPIC_DEFAULT_OPUS_MODEL", defaultValue: "cc/claude-opus-5" }, + { flag: "sonnet", envKey: "ANTHROPIC_DEFAULT_SONNET_MODEL", defaultValue: "cc/claude-sonnet-5" }, + { flag: "haiku", envKey: "ANTHROPIC_DEFAULT_HAIKU_MODEL", defaultValue: "cc/claude-haiku-4-5-20251001" }, +]; +const CLAUDE_RESET_KEYS = ["ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN", ...CLAUDE_MODELS.map((m) => m.envKey)]; +const claudePath = () => path.join(home(), ".claude", "settings.json"); + +const claude = { + id: "claude", + name: "Claude Code", + paths: () => [claudePath()], + async apply({ baseUrl, apiKey, claudeModels }) { + const file = claudePath(); + const cur = readJson(file) || {}; + cur.hasCompletedOnboarding = true; + cur.env = { ...(cur.env || {}), ANTHROPIC_BASE_URL: v1(baseUrl), ANTHROPIC_AUTH_TOKEN: apiKey, ...claudeModels }; + writeJson(file, cur); + return [file]; + }, + async reset() { + const file = claudePath(); + const cur = readJson(file); + if (!cur) return []; + if (cur.env) { + CLAUDE_RESET_KEYS.forEach((k) => delete cur.env[k]); + if (Object.keys(cur.env).length === 0) delete cur.env; + } + rewriteFile(file, JSON.stringify(cur, null, 2)); + return [file]; + }, +}; + +// ── OpenAI Codex CLI ──────────────────────────────────────────────────────── +const codexPath = () => path.join(home(), ".codex", "config.toml"); + +const codex = { + id: "codex", + name: "OpenAI Codex CLI", + paths: () => [codexPath()], + async apply({ baseUrl, apiKey, model }) { + const { parseTOML, stringifyTOML } = await toml(); + const file = codexPath(); + let cfg = {}; + try { cfg = parseTOML(fs.readFileSync(file, "utf8")) || {}; } catch (err) { if (err.code !== "ENOENT") throw err; } + cfg.model = model; + cfg.model_provider = "9router"; + cfg.model_providers = cfg.model_providers || {}; + // Custom providers ignore auth.json — key must travel as a static header. + cfg.model_providers["9router"] = { + name: "9Router", + base_url: v1(baseUrl), + wire_api: "responses", + http_headers: { Authorization: `Bearer ${apiKey}` }, + }; + cfg.agents = cfg.agents || {}; + delete cfg.agents.subagent; + cfg.agents.default_subagent_model = model; + writeFile(file, stringifyTOML(cfg)); + return [file]; + }, + async reset() { + const { parseTOML, stringifyTOML } = await toml(); + const file = codexPath(); + let cfg; + try { cfg = parseTOML(fs.readFileSync(file, "utf8")) || {}; } catch (err) { if (err.code === "ENOENT") return []; throw err; } + if (cfg.model_provider === "9router") { delete cfg.model; delete cfg.model_provider; } + if (cfg.model_providers) delete cfg.model_providers["9router"]; + if (cfg.agents) { delete cfg.agents.default_subagent_model; delete cfg.agents.subagent; } + for (const k of ["model_providers", "agents"]) { + if (cfg[k] && Object.keys(cfg[k]).length === 0) delete cfg[k]; + } + rewriteFile(file, stringifyTOML(cfg)); + return [file]; + }, +}; + +// ── OpenCode ──────────────────────────────────────────────────────────────── +const opencodePath = () => path.join(home(), ".config", "opencode", "opencode.json"); + +const opencode = { + id: "opencode", + name: "OpenCode", + paths: () => [opencodePath()], + async apply({ baseUrl, apiKey, model }) { + const file = opencodePath(); + const cfg = readJson(file) || {}; + cfg.provider = cfg.provider || {}; + const p = cfg.provider["9router"] || { npm: "@ai-sdk/openai-compatible", options: {}, models: {} }; + p.options = { ...p.options, baseURL: v1(baseUrl), apiKey }; + p.models = p.models || {}; + p.models[model] = { name: model, modalities: { input: ["text", "image"], output: ["text"] } }; + cfg.provider["9router"] = p; + cfg.model = `9router/${model}`; + cfg.agent = cfg.agent || {}; + cfg.agent.explorer = { + description: "Fast explorer subagent for codebase exploration", + mode: "subagent", + model: `9router/${model}`, + }; + writeJson(file, cfg); + return [file]; + }, + async reset() { + const file = opencodePath(); + const cfg = readJson(file); + if (!cfg) return []; + if (cfg.provider) delete cfg.provider["9router"]; + if (cfg.model?.startsWith("9router/")) delete cfg.model; + if (cfg.agent?.explorer?.model?.startsWith("9router/")) { + delete cfg.agent.explorer; + if (Object.keys(cfg.agent).length === 0) delete cfg.agent; + } + rewriteFile(file, JSON.stringify(cfg, null, 2)); + return [file]; + }, +}; + +// ── Factory Droid ─────────────────────────────────────────────────────────── +const droidPath = () => path.join(home(), ".factory", "settings.json"); +const isDroid9r = (m) => m.id?.startsWith("custom:9Router"); + +const droid = { + id: "droid", + name: "Factory Droid", + paths: () => [droidPath()], + async apply({ baseUrl, apiKey, model }) { + const file = droidPath(); + const cfg = readJson(file) || {}; + const others = (cfg.customModels || []).filter((m) => !isDroid9r(m)); + cfg.customModels = [ + { + model, + id: "custom:9Router-0", + index: 0, + baseUrl: v1(baseUrl), + apiKey, + displayName: model, + maxOutputTokens: 131072, + noImageSupport: false, + provider: "openai", + }, + ...others, + ]; + cfg.customModels.forEach((m, i) => { m.index = i; }); + writeJson(file, cfg); + return [file]; + }, + async reset() { + const file = droidPath(); + const cfg = readJson(file); + if (!cfg) return []; + if (cfg.customModels) { + cfg.customModels = cfg.customModels.filter((m) => !isDroid9r(m)); + if (cfg.customModels.length === 0) delete cfg.customModels; + } + rewriteFile(file, JSON.stringify(cfg, null, 2)); + return [file]; + }, +}; + +// ── Crush ─────────────────────────────────────────────────────────────────── +const crushPath = () => path.join(process.env.XDG_CONFIG_HOME || path.join(home(), ".config"), "crush", "crush.json"); + +const crush = { + id: "crush", + name: "Crush", + paths: () => [crushPath()], + async apply({ baseUrl, apiKey, model }) { + const file = crushPath(); + const cfg = readJson(file) || {}; + cfg.providers = cfg.providers || {}; + cfg.providers["9router"] = { + type: "openai-compat", + base_url: v1(baseUrl), + api_key: apiKey, + models: [{ id: model, name: model, context_window: 128000 }], + }; + writeJson(file, cfg); + return [file]; + }, + async reset() { + const file = crushPath(); + const cfg = readJson(file); + if (!cfg?.providers?.["9router"]) return []; + delete cfg.providers["9router"]; + if (Object.keys(cfg.providers).length === 0) delete cfg.providers; + rewriteFile(file, JSON.stringify(cfg, null, 2)); + return [file]; + }, +}; + +// ── Kilo Code (CLI auth only; VS Code settings left to the dashboard) ─────── +const kiloPath = () => path.join(home(), ".local", "share", "kilo", "auth.json"); + +const kilo = { + id: "kilo", + name: "Kilo Code CLI", + paths: () => [kiloPath()], + async apply({ baseUrl, apiKey, model }) { + const file = kiloPath(); + const auth = readJson(file) || {}; + auth["openai-compatible"] = { type: "api-key", apiKey, baseUrl: v1(baseUrl), model }; + writeJson(file, auth); + return [file]; + }, + async reset() { + const file = kiloPath(); + const auth = readJson(file); + if (!auth) return []; + delete auth["openai-compatible"]; + delete auth["9router"]; + rewriteFile(file, JSON.stringify(auth, null, 2)); + return [file]; + }, +}; + +// ── Cline CLI ─────────────────────────────────────────────────────────────── +const clineDir = () => path.join(home(), ".cline", "data"); +const clineState = () => path.join(clineDir(), "globalState.json"); +const clineSecrets = () => path.join(clineDir(), "secrets.json"); + +const cline = { + id: "cline", + name: "Cline CLI", + paths: () => [clineState(), clineSecrets()], + async apply({ baseUrl, apiKey, model }) { + const state = readJson(clineState()) || {}; + state.actModeApiProvider = "openai"; + state.planModeApiProvider = "openai"; + state.openAiBaseUrl = baseUrl; // Cline expects base WITHOUT /v1 + state.openAiModelId = model; + state.planModeOpenAiModelId = model; + writeJson(clineState(), state); + const secrets = readJson(clineSecrets()) || {}; + secrets.openAiApiKey = apiKey; + writeJson(clineSecrets(), secrets); + return [clineState(), clineSecrets()]; + }, + async reset() { + const state = readJson(clineState()); + if (!state) return []; + if (state.actModeApiProvider === "openai") { + delete state.openAiBaseUrl; + delete state.openAiModelId; + delete state.planModeOpenAiModelId; + state.actModeApiProvider = "cline"; + state.planModeApiProvider = "cline"; + } + rewriteFile(clineState(), JSON.stringify(state, null, 2)); + const touched = [clineState()]; + const secrets = readJson(clineSecrets()); + if (secrets) { + delete secrets.openAiApiKey; + rewriteFile(clineSecrets(), JSON.stringify(secrets, null, 2)); + touched.push(clineSecrets()); + } + return touched; + }, +}; + +const TOOLS = [claude, codex, opencode, droid, crush, kilo, cline]; +const TOOL_IDS = TOOLS.map((t) => t.id); +const TOOL_ALIASES = { "claude-code": "claude", "claudecode": "claude", "factory": "droid", "kilocode": "kilo" }; + +function resolveTools(list) { + const ids = new Set(); + for (const raw of list) { + const id = String(raw).trim().toLowerCase(); + if (!id) continue; + if (id === "all") { TOOL_IDS.forEach((t) => ids.add(t)); continue; } + const real = TOOL_ALIASES[id] || id; + if (!TOOL_IDS.includes(real)) throw new Error(`Unknown tool "${raw}". Supported: ${TOOL_IDS.join(", ")}, all`); + ids.add(real); + } + return TOOLS.filter((t) => ids.has(t.id)); +} + +module.exports = { TOOLS, TOOL_IDS, CLAUDE_MODELS, resolveTools, __test__: { stripTrailingCommas } }; diff --git a/tests/unit/cli-connect.test.js b/tests/unit/cli-connect.test.js new file mode 100644 index 00000000..c20643a6 --- /dev/null +++ b/tests/unit/cli-connect.test.js @@ -0,0 +1,169 @@ +// `9router connect` — arg parsing, URL/cookie helpers and per-tool config writers. +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import { createRequire } from "node:module"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const require = createRequire(import.meta.url); +const connect = require("../../cli/src/cli/commands/connect.js"); +const tools = require("../../cli/src/cli/commands/connectTools.js"); +const { parseArgs, normalizeServerUrl, extractAuthCookie, maskKey } = connect.__test__; +const { stripTrailingCommas } = tools.__test__; + +const CTX = { + baseUrl: "http://gw.test:20128", + apiKey: "sk-unit-test-key-0000", + model: "cc/claude-opus-5", + claudeModels: { ANTHROPIC_DEFAULT_OPUS_MODEL: "cc/claude-opus-5" }, +}; +const tool = (id) => tools.TOOLS.find((t) => t.id === id); +const readJson = (p) => JSON.parse(fs.readFileSync(p, "utf8")); + +describe("connect helpers", () => { + it("parseArgs reads url, tools, model and claude tier overrides", () => { + const o = parseArgs(["http://h:1", "--tools", "claude,codex", "--model", "m1", "--opus", "o1", "--password", "p"]); + expect(o.url).toBe("http://h:1"); + expect(o.tools).toEqual(["claude", "codex"]); + expect(o.model).toBe("m1"); + expect(o.models).toEqual({ opus: "o1" }); + expect(o.password).toBe("p"); + }); + + it("parseArgs rejects unknown options and missing values", () => { + expect(() => parseArgs(["--nope"])).toThrow(/Unknown option/); + expect(() => parseArgs(["--tools"])).toThrow(/Missing value/); + }); + + it("normalizeServerUrl keeps only the origin and defaults to http", () => { + expect(normalizeServerUrl("http://h:20128/dashboard/cli-tools")).toBe("http://h:20128"); + expect(normalizeServerUrl("h:20128")).toBe("http://h:20128"); + expect(normalizeServerUrl("https://h/v1/")).toBe("https://h"); + }); + + it("extractAuthCookie picks auth_token from Set-Cookie", () => { + const h = new Headers(); + h.append("set-cookie", "other=1; Path=/"); + h.append("set-cookie", "auth_token=abc.def.ghi; Path=/; HttpOnly"); + expect(extractAuthCookie(h)).toBe("auth_token=abc.def.ghi"); + expect(extractAuthCookie(new Headers())).toBeNull(); + }); + + it("maskKey never reveals the middle of the key", () => { + expect(maskKey("sk-1234567890abcdef")).toBe("sk-123…cdef"); + expect(maskKey("short")).toBe("****"); + }); + + it("stripTrailingCommas leaves commas inside strings alone", () => { + expect(JSON.parse(stripTrailingCommas('{"a":"x,}","b":[1,2,],}'))).toEqual({ a: "x,}", b: [1, 2] }); + expect(JSON.parse(stripTrailingCommas('{"a":"q\\",}",}'))).toEqual({ a: 'q",}' }); + }); + + it("resolveTools handles aliases, all, and unknown names", () => { + expect(tools.resolveTools(["claude-code", "factory"]).map((t) => t.id)).toEqual(["claude", "droid"]); + expect(tools.resolveTools(["all"]).map((t) => t.id)).toEqual(tools.TOOL_IDS); + expect(() => tools.resolveTools(["bogus"])).toThrow(/Unknown tool/); + }); +}); + +describe("connect tool writers", () => { + let home; + beforeEach(() => { + home = fs.mkdtempSync(path.join(os.tmpdir(), "9r-connect-")); + vi.spyOn(os, "homedir").mockReturnValue(home); + vi.stubEnv("XDG_CONFIG_HOME", ""); + }); + afterEach(() => { + vi.restoreAllMocks(); + vi.unstubAllEnvs(); + fs.rmSync(home, { recursive: true, force: true }); + }); + + it("every tool applies then resets cleanly on an empty home", async () => { + for (const t of tools.TOOLS) { + const written = await t.apply(CTX); + expect(written.length).toBeGreaterThan(0); + // Key may live in just one of the files (cline: secrets.json). + expect(written.some((f) => fs.readFileSync(f, "utf8").includes(CTX.apiKey))).toBe(true); + if (process.platform !== "win32") { + for (const f of written) expect(fs.statSync(f).mode & 0o777).toBe(0o600); + } + await t.reset(); + for (const f of written) expect(fs.readFileSync(f, "utf8")).not.toContain(CTX.apiKey); + } + }); + + it("claude merges env and keeps unrelated settings; reset removes only 9router keys", async () => { + const f = path.join(home, ".claude", "settings.json"); + fs.mkdirSync(path.dirname(f), { recursive: true }); + fs.writeFileSync(f, JSON.stringify({ theme: "dark", env: { KEEP: "1" } })); + await tool("claude").apply(CTX); + const cfg = readJson(f); + expect(cfg.theme).toBe("dark"); + expect(cfg.env).toMatchObject({ KEEP: "1", ANTHROPIC_BASE_URL: "http://gw.test:20128/v1", ANTHROPIC_AUTH_TOKEN: CTX.apiKey }); + expect(fs.existsSync(`${f}.bak-9router`)).toBe(true); + await tool("claude").reset(); + expect(readJson(f)).toEqual({ theme: "dark", hasCompletedOnboarding: true, env: { KEEP: "1" } }); + }); + + it("codex keeps other TOML tables and drops empty ones on reset", async () => { + const f = path.join(home, ".codex", "config.toml"); + fs.mkdirSync(path.dirname(f), { recursive: true }); + fs.writeFileSync(f, '[mcp_servers.x]\ncommand = "foo"\n'); + await tool("codex").apply(CTX); + const text = fs.readFileSync(f, "utf8"); + expect(text).toContain('model_provider = "9router"'); + expect(text).toContain("[mcp_servers.x]"); + await tool("codex").reset(); + const after = fs.readFileSync(f, "utf8"); + expect(after).toContain("[mcp_servers.x]"); + expect(after).not.toMatch(/9router|model_providers|\[agents\]/); + }); + + it("droid keeps user models and puts 9router first", async () => { + const f = path.join(home, ".factory", "settings.json"); + fs.mkdirSync(path.dirname(f), { recursive: true }); + fs.writeFileSync(f, JSON.stringify({ customModels: [{ id: "mine", model: "m" }] })); + await tool("droid").apply(CTX); + expect(readJson(f).customModels.map((m) => m.id)).toEqual(["custom:9Router-0", "mine"]); + await tool("droid").reset(); + expect(readJson(f).customModels.map((m) => m.id)).toEqual(["mine"]); + }); + + it("cline uses base URL without /v1 and reset reports both files", async () => { + await tool("cline").apply(CTX); + const state = readJson(path.join(home, ".cline", "data", "globalState.json")); + expect(state.openAiBaseUrl).toBe("http://gw.test:20128"); + expect((await tool("cline").reset()).length).toBe(2); + }); +}); + +describe("connect run()", () => { + let home; + beforeEach(() => { + home = fs.mkdtempSync(path.join(os.tmpdir(), "9r-connect-run-")); + vi.spyOn(os, "homedir").mockReturnValue(home); + vi.spyOn(console, "log").mockImplementation(() => {}); + }); + afterEach(() => { + vi.restoreAllMocks(); + fs.rmSync(home, { recursive: true, force: true }); + }); + + it("unknown tool fails before any network call", async () => { + const fetchSpy = vi.spyOn(globalThis, "fetch"); + await expect(connect.run(["http://gw.test", "--tools", "bogus", "--password", "x"])).rejects.toThrow(/Unknown tool/); + expect(fetchSpy).not.toHaveBeenCalled(); + }); + + it("reset keeps going when one tool fails and returns 1", async () => { + const f = path.join(home, ".codex", "config.toml"); + fs.mkdirSync(path.dirname(f), { recursive: true }); + fs.writeFileSync(f, "this is = = not toml [[["); + const code = await connect.run(["--reset", "--tools", "codex,claude"]); + expect(code).toBe(1); + const lines = console.log.mock.calls.map((c) => c[0]); + expect(lines.some((l) => l.startsWith("❌ OpenAI Codex CLI"))).toBe(true); + expect(lines.some((l) => l.includes("Claude Code"))).toBe(true); + }); +}); From dec820b97e21d1fe46d80448c66a531110afead1 Mon Sep 17 00:00:00 2001 From: MrBeanDev Date: Thu, 1 Oct 2026 10:12:19 +0700 Subject: [PATCH 34/41] feat(codex): add GPT-6.1 Sol - registry entry with responsesLite + low..max thinking levels (matches gpt-6-sol) - pricing: $2 input / $0.10 cached / $2.50 cache write / $10 output per 1M --- open-sse/providers/pricing.js | 1 + open-sse/providers/registry/codex.js | 1 + tests/unit/codex-gpt6-lite.test.js | 10 +++++++++- 3 files changed, 11 insertions(+), 1 deletion(-) diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index 4b62fa86..45df9d22 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -78,6 +78,7 @@ export const MODEL_PRICING = { // OpenAI Standard short-context pricing (developers.openai.com/api/docs/pricing). // Long-context pricing is higher, but this table currently stores one rate per model. "gpt-6-astra": { input: 10.00, output: 50.00, cached: 1.00, reasoning: 50.00, cache_creation: 12.50 }, + "gpt-6.1-sol": { input: 2.00, output: 10.00, cached: 0.10, reasoning: 10.00, cache_creation: 2.50 }, "gpt-6-sol": { input: 2.00, output: 10.00, cached: 0.20, reasoning: 10.00, cache_creation: 2.50 }, "gpt-6-luna": { input: 0.10, output: 0.50, cached: 0.01, reasoning: 0.50, cache_creation: 0.125 }, "o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 }, diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 15a38784..abc5ec26 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -52,6 +52,7 @@ export default { }, }, models: [ + { id: "gpt-6.1-sol", name: "GPT 6.1 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, { id: "gpt-6-astra", name: "GPT 6.0 Astra" }, { id: "gpt-6-astra[1m]", name: "GPT 6.0 Astra (extended context)", upstreamModelId: "gpt-6-astra" }, { id: "gpt-6-sol", name: "GPT 6.0 Sol", responsesLite: true, thinkingLevels: GPT_6_LITE_THINKING_LEVELS }, diff --git a/tests/unit/codex-gpt6-lite.test.js b/tests/unit/codex-gpt6-lite.test.js index 77cb8ed7..534244b0 100644 --- a/tests/unit/codex-gpt6-lite.test.js +++ b/tests/unit/codex-gpt6-lite.test.js @@ -11,7 +11,7 @@ const credentials = { connectionId: "fixture", accessToken: "fixture-token" }; afterEach(() => vi.restoreAllMocks()); describe("Codex GPT-6 Sol/Luna transport", () => { - it.each(["gpt-6-sol", "gpt-6-luna"])("lists %s with Codex capabilities", (model) => { + it.each(["gpt-6.1-sol", "gpt-6-sol", "gpt-6-luna"])("lists %s with Codex capabilities", (model) => { const entry = getModelsByProviderId("codex").find((item) => item.id === model); expect(entry?.responsesLite).toBe(true); expect(entry?.thinkingLevels).toEqual(["low", "medium", "high", "xhigh", "max"]); @@ -175,6 +175,14 @@ describe("Codex GPT-6 Sol/Luna transport", () => { expect(body.reasoning.context).toBe("all_turns"); }); + it("maps GPT-6.1 Sol's Codex-only ultra effort to max", () => { + const body = new CodexExecutor().transformRequest("gpt-6.1-sol", { + model: "gpt-6.1-sol", input: "hello", reasoning: { effort: "ultra" }, + }, true, credentials); + + expect(body.reasoning).toEqual({ effort: "max", context: "all_turns" }); + }); + it("sends the Lite shape and header in the actual outbound request", async () => { const fetchMock = vi.spyOn(proxyFetchModule, "proxyAwareFetch").mockResolvedValue({ ok: true, status: 200, headers: new Map(), From 49c761cd675bd55a460545b74fb15f45ed95596e Mon Sep 17 00:00:00 2001 From: Daniel Imaino <35339571+dimaino@users.noreply.github.com> Date: Thu, 1 Oct 2026 10:18:19 +0700 Subject: [PATCH 35/41] fix(claude): cache a tool loop's final tool results with the 4th breakpoint A tool loop's request ends with the results of the last assistant turn's tool calls -- after that turn's breakpoint, so they are billed as uncached input and written to cache only by the next request. When the final user turn carries tool_result blocks and the 4-marker budget has room, markFinalToolResults puts a 5m breakpoint on its last cacheable block, in both prepareClaudeRequest and anchorClaudeCache. Never exceeds 4 markers; requests ending with a typed user message are unchanged. --- open-sse/translator/formats/claude.js | 21 +++++ .../__snapshots__/golden-request.test.js.snap | 3 + .../claude-cache-final-tool-results.test.js | 80 +++++++++++++++++++ 3 files changed, 104 insertions(+) create mode 100644 tests/unit/claude-cache-final-tool-results.test.js diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index 8fa44ccf..2795ecaa 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -346,6 +346,21 @@ function markLastCacheableBlock(msg) { return false; } +// In an agent's tool loop, a request ends with the results of the last +// assistant turn's tool calls -- after that turn's breakpoint. They go at the +// full input price, and the next request (which appends to them) writes them +// into the cache. When the 4-marker budget has room, a 5m breakpoint on that +// final user turn caches them now, and the next request reads them. +function markFinalToolResults(body) { + const messages = body?.messages; + const last = Array.isArray(messages) ? messages[messages.length - 1] : null; + if (last?.role !== ROLE.USER || !Array.isArray(last.content)) return false; + if (!last.content.some((block) => block?.type === CLAUDE_BLOCK.TOOL_RESULT)) return false; + if (last.content.some((block) => block?.cache_control)) return false; + if (countCacheControlBlocks(body) >= 4) return false; + return markLastCacheableBlock(last); +} + // Re-anchor cache breakpoints on a Claude passthrough body (same policy as // prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m. // The client's own markers point at pre-normalization offsets, so they are dropped. @@ -415,6 +430,9 @@ export function anchorClaudeCache(body) { anchored = markLastCacheableBlock(body.messages[i]); } } + + // ...and a tool loop's final tool results, so the next step reads them. + markFinalToolResults(body); } return body; @@ -672,6 +690,9 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne body = hoistToolResultImages(body); } + // A tool loop's final tool results: cached now, so the next step reads them. + markFinalToolResults(body); + // Apply cloaking for OAuth tokens (billing header + fake user ID) // session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) { diff --git a/tests/translator/__snapshots__/golden-request.test.js.snap b/tests/translator/__snapshots__/golden-request.test.js.snap index db2c7378..8bcab641 100644 --- a/tests/translator/__snapshots__/golden-request.test.js.snap +++ b/tests/translator/__snapshots__/golden-request.test.js.snap @@ -40,6 +40,9 @@ exports[`GOLDEN request: OpenAI → Claude > full body (system/image/tool/tool_r { "content": [ { + "cache_control": { + "type": "ephemeral", + }, "content": "sunny", "tool_use_id": "call_1", "type": "tool_result", diff --git a/tests/unit/claude-cache-final-tool-results.test.js b/tests/unit/claude-cache-final-tool-results.test.js new file mode 100644 index 00000000..4140acee --- /dev/null +++ b/tests/unit/claude-cache-final-tool-results.test.js @@ -0,0 +1,80 @@ +// A tool loop's request ends with the last assistant turn's tool results, after that turn's +// breakpoint: without a 4th breakpoint on them they go at the full input price, and are written +// into the cache only by the next request, which appends to them. +import { describe, it, expect } from "vitest"; +import { anchorClaudeCache, prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; + +const CC = { type: "ephemeral" }; +const text = (t, extra = {}) => ({ type: "text", text: t, ...extra }); +const tool = (name, extra = {}) => ({ name, description: "d", input_schema: { type: "object", properties: {} }, ...extra }); +const use = (id) => ({ role: "assistant", content: [text("Reading."), { type: "tool_use", id, name: "read_file", input: { path: "a" } }] }); +const result = (id, content = "file") => ({ type: "tool_result", tool_use_id: id, content }); + +function markers(body) { + const out = []; + (body.system || []).forEach((b, i) => b?.cache_control && out.push(`system[${i}]`)); + (body.tools || []).forEach((t, i) => t?.cache_control && out.push(`tools[${i}]`)); + (body.messages || []).forEach((m, i) => Array.isArray(m?.content) && m.content.forEach((b, j) => b?.cache_control && out.push(`messages[${i}].${j}`))); + return out; +} + +const loop = () => ({ + model: "claude-sonnet-4-5", + max_tokens: 1024, + system: [text("You are an agent.")], + tools: [tool("read_file"), tool("run_command")], + messages: [ + { role: "user", content: [text("Fix the bug.")] }, + use("t1"), + { role: "user", content: [result("t1")] }, + use("t2"), + { role: "user", content: [result("t2", "a"), result("t2b", "b")] }, + ], +}); + +describe("prepareClaudeRequest: a tool loop's final tool results", () => { + it("get the 4th breakpoint, after the last assistant turn's", () => { + const out = prepareClaudeRequest(loop(), "claude"); + expect(markers(out)).toEqual(["system[0]", "tools[1]", "messages[3].1", "messages[4].1"]); + expect(out.messages[4].content[1].cache_control).toEqual({ type: "ephemeral" }); + }); + + it("leave a request that ends with a typed message as it was", () => { + const body = loop(); + body.messages.push({ role: "assistant", content: [text("Done.")] }, { role: "user", content: [text("Thanks, and the tests?")] }); + const out = prepareClaudeRequest(body, "claude"); + expect(markers(out)).toEqual(["system[0]", "tools[1]", "messages[5].0"]); + }); + + it("need no tools array to be marked", () => { + const body = loop(); + delete body.tools; + const out = prepareClaudeRequest(body, "claude"); + expect(markers(out)).toEqual(["system[0]", "messages[3].1", "messages[4].1"]); + }); + + it("never take the request past four markers", () => { + const body = loop(); + body.system = [text("a", { cache_control: CC }), text("b", { cache_control: CC })]; + body.tools = body.tools.map((t) => ({ ...t, cache_control: CC })); + body.messages.forEach((m) => m.content.forEach((b) => (b.cache_control = CC))); + const out = prepareClaudeRequest(body, "claude"); + expect(markers(out).length).toBeLessThanOrEqual(4); + expect(out.messages[4].content.at(-1).cache_control).toBeTruthy(); + }); +}); + +describe("anchorClaudeCache: a passthrough tool loop's final tool results", () => { + it("are re-anchored with the last assistant turn", () => { + const out = anchorClaudeCache(loop()); + expect(markers(out)).toEqual(["system[0]", "tools[1]", "messages[3].1", "messages[4].1"]); + }); + + it("keep a client's own full budget as it is", () => { + const body = loop(); + body.messages[0].content[0].cache_control = CC; + body.messages[2].content[0].cache_control = CC; + const out = anchorClaudeCache(body); + expect(markers(out).length).toBeLessThanOrEqual(4); + }); +}); From 89ffac5a2c8108449c432afd1ad6531578fa81cd Mon Sep 17 00:00:00 2001 From: Amirsalar Sojoudi Date: Thu, 1 Oct 2026 10:23:40 +0700 Subject: [PATCH 36/41] fix(capabilities): publish real GPT-6/GPT-5.4+ context windows and combo token limits --- open-sse/providers/capabilities.js | 47 +++++++-- src/app/api/v1/models/route.js | 75 +++++++++++++- src/lib/modelCatalog/sync.js | 1 + tests/unit/gpt-6-context-window.test.js | 112 +++++++++++++++++++++ tests/unit/v1-models-combo-context.test.js | 83 +++++++++++++++ 5 files changed, 306 insertions(+), 12 deletions(-) create mode 100644 tests/unit/gpt-6-context-window.test.js create mode 100644 tests/unit/v1-models-combo-context.test.js diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index 68e4fa0a..833635dc 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -6,15 +6,18 @@ // 3. PATTERN_CAPABILITIES — glob match, ordered specific -> generic // 4. DEFAULT_CAPABILITIES — safe floor (always returned) // -// Two extra layers then refine the result, and neither can override the hand -// written tables above (steps 1-2 short-circuit before they are consulted): +// Two extra layers then refine the result: // • the synced catalog — modalities keyed by model, limits keyed by provider // + model, refreshed from models.dev in the background. It reads a file, so // the server installs it via setCatalogSource(); this module stays free of // node:fs because the dashboard bundles it into the browser too. // • visionPatterns.js — name-based vision detection, last resort so a model // nobody has catalogued yet still accepts images. -// Both only ever turn a capability ON. +// Modalities only ever turn a capability ON. Limits from the catalog overlay +// the canonical exact entry (step 2) so a gateway-specific models.dev delta +// (Copilot's 32k Claude output, etc.) actually publishes. Step 1 still +// short-circuits: a hand-written PROVIDER_CAPABILITIES truncation is the +// gateway's own number and must not be overwritten. // // ── HOW TO ADD / UPDATE A MODEL ────────────────────────────────────── // Authoritative data source: https://models.dev/api.json (145 providers, 4000+ @@ -165,6 +168,11 @@ const CODEX_GPT_56_SOL_CAPS = { vision: true, reasoning: true, search: true, th const CODEX_GPT_56_DEFAULT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; const CODEX_EXTENDED_CAPS = { ...CODEX_GPT_56_DEFAULT_CAPS, contextWindow: 872000 }; +// Devin CLI's registry declares a 200k context window for these GPT variants. +// Keep the GPT feature/output fields because provider overrides short-circuit +// the generic pattern rather than merging with it. +const DEVIN_CLI_GPT_CAPS = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 128000 }; + /** * Provider-specific capability overrides. Keyed by provider alias/id. */ @@ -215,6 +223,15 @@ export const PROVIDER_CAPABILITIES = { "gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES, "gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES, }, + "devin-cli": { + "gpt-5.4-high": DEVIN_CLI_GPT_CAPS, + "gpt-5.4-medium": DEVIN_CLI_GPT_CAPS, + "gpt-5.4-low": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-xhigh": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-high": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-medium": DEVIN_CLI_GPT_CAPS, + "gpt-5.5-low": DEVIN_CLI_GPT_CAPS, + }, // CodeBuddy.cn — authoritative per-model metadata from the gateway's model // config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision= // supportsImages). Every model reasons via OpenAI-style reasoning_effort @@ -278,6 +295,8 @@ export const PROVIDER_CAPABILITIES = { // the intl Qoder capability table verbatim (vision/reasoning/contextWindow). PROVIDER_CAPABILITIES["qoder-cn"] = PROVIDER_CAPABILITIES["qoder"]; PROVIDER_CAPABILITIES.cx = PROVIDER_CAPABILITIES.codex; +PROVIDER_CAPABILITIES.dv = PROVIDER_CAPABILITIES["devin-cli"]; +PROVIDER_CAPABILITIES.devin = PROVIDER_CAPABILITIES["devin-cli"]; /** * Pattern fallback — glob (* = wildcard), matched case-insensitively and @@ -315,11 +334,24 @@ export const PATTERN_CAPABILITIES = [ { pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } }, // ── OpenAI GPT-6.x (vision + thinking + web search) ────────────── - { pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } }, + // 1.05M is the API window for the whole gpt-6 family (astra, luna, sol alike). + // A gateway that truncates lower records its own number in + // PROVIDER_CAPABILITIES, which wins over this pattern — Kiro at 272k, Codex + // OAuth at 272k/372k (see CODEX_GPT_56_* above). This entry used to carry + // Kiro's 272k, so every other provider's gpt-6 models inherited one gateway's + // limit and were published at 3.9x under their real window. + { pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, // ── OpenAI GPT-5.x (vision + thinking + web search) ────────────── { pattern: "*gpt-5*image*", caps: { imageOutput: true } }, { pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, + // gpt-5.4 is where the 1.05M window starts, but the mini and nano tiers stayed + // at 400k — first match wins, so those two have to be listed ahead of it. + { pattern: "*gpt-5.4-mini*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, + { pattern: "*gpt-5.4-nano*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, + { pattern: "*gpt-5.4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, + { pattern: "*gpt-5.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, + { pattern: "*gpt-5.6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 1050000, maxOutput: 128000 } }, { pattern: "*gpt-5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, { pattern: "*gpt-4o*", caps: { vision: true, search: true, contextWindow: 128000, maxOutput: 16384 } }, { pattern: "*gpt-4.1*", caps: { vision: true, contextWindow: 1000000, maxOutput: 32768 } }, @@ -618,9 +650,10 @@ export function getCapabilitiesForModel(provider, model) { if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] }; } - // 2. Canonical exact - if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] }; - if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] }; + // 2. Canonical exact, then catalog overlay so provider-scoped models.dev + // deltas still apply. Step 1 above still short-circuits. + if (MODEL_CAPABILITIES[baseModel]) return refine(MODEL_CAPABILITIES[baseModel], provider, model); + if (MODEL_CAPABILITIES[model]) return refine(MODEL_CAPABILITIES[model], provider, model); // 3. Pattern match (first match wins), refined by catalog + name heuristic for (const { pattern, caps } of PATTERN_CAPABILITIES) { diff --git a/src/app/api/v1/models/route.js b/src/app/api/v1/models/route.js index 389f141b..0b1fe342 100644 --- a/src/app/api/v1/models/route.js +++ b/src/app/api/v1/models/route.js @@ -1,5 +1,6 @@ import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS, getModelKind } from "@/shared/constants/models"; import { + ALIAS_TO_ID, AI_PROVIDERS, getProviderAlias, isAnthropicCompatibleProvider, @@ -40,6 +41,22 @@ async function resolveQoderLiveModels(conn, provider) { return { models: models.map((m) => ({ id: m.id, name: m.name })) }; } +// Combo seats use UI aliases; the model registry also has transport aliases. +// Capability overrides and catalog limits are keyed by provider id. +const ALIAS_TO_PROVIDER_ID = { + ...Object.fromEntries( + Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id]) + ), + ...ALIAS_TO_ID, +}; + +function comboSeatCapabilities(seat) { + const slash = seat.indexOf("/"); + if (slash <= 0) return null; + const alias = seat.slice(0, slash); + return getCapabilitiesForModel(ALIAS_TO_PROVIDER_ID[alias] || alias, seat.slice(slash + 1)); +} + // Per-provider live model resolvers. Each receives a connection record and // returns { models: [{ id, name? }, ...] } | null on failure. // Adding a provider here makes /v1/models prefer the live catalog for it. @@ -254,6 +271,47 @@ function comboMatchesKinds(combo, kindFilter) { return kindFilter.includes(kind); } +// Nested combo names are valid seats — the model selector exposes them and +// chat routing resolves them recursively — but a no-slash seat is otherwise +// treated as a literal model and publishes the 200k floor. Expand nested +// names (cycle-guarded) so the published window is the true min across the +// whole chain. +function comboSeatLimits(combo, combosByName, visiting = new Set()) { + const name = typeof combo?.name === "string" ? combo.name : null; + if (name) { + if (visiting.has(name)) return { contextWindow: undefined, maxOutput: undefined }; + visiting.add(name); + } + + let contextWindow = Infinity; + let maxOutput = Infinity; + try { + for (const seat of Array.isArray(combo?.models) ? combo.models : []) { + if (typeof seat !== "string") continue; + const slash = seat.indexOf("/"); + if (slash <= 0) { + const nested = combosByName.get(seat); + if (nested) { + const nestedLimits = comboSeatLimits(nested, combosByName, visiting); + if (Number.isFinite(nestedLimits.contextWindow)) contextWindow = Math.min(contextWindow, nestedLimits.contextWindow); + if (Number.isFinite(nestedLimits.maxOutput)) maxOutput = Math.min(maxOutput, nestedLimits.maxOutput); + continue; + } + } + const caps = comboSeatCapabilities(seat) || getCapabilitiesForModel(null, seat); + if (Number.isFinite(caps?.contextWindow)) contextWindow = Math.min(contextWindow, caps.contextWindow); + if (Number.isFinite(caps?.maxOutput)) maxOutput = Math.min(maxOutput, caps.maxOutput); + } + } finally { + if (name) visiting.delete(name); + } + + return { + contextWindow: Number.isFinite(contextWindow) ? contextWindow : undefined, + maxOutput: Number.isFinite(maxOutput) ? maxOutput : undefined, + }; +} + /** * Build OpenAI-format models list filtered by service kinds. * @param {string[]} kindFilter - List of service kinds to include (e.g. ["llm"], ["webSearch","webFetch"]). @@ -308,6 +366,9 @@ export async function buildModelsList(kindFilter, options = {}) { } const models = []; + const combosByName = new Map( + combos.filter((c) => typeof c?.name === "string").map((c) => [c.name, c]), + ); // Lookup map so aggregateComboCapabilities can recursively resolve nested combos const comboByName = Object.fromEntries(combos.map((c) => [c.name, c.models])); @@ -323,19 +384,23 @@ export async function buildModelsList(kindFilter, options = {}) { if (combo.kind === "webSearch" || combo.kind === "webFetch") { entry.kind = combo.kind; } else { - const comboCaps = aggregateComboCapabilities(combo.models, comboByName); + const comboCaps = aggregateComboCapabilities(combo.models, comboByName, comboSeatCapabilities); if (comboCaps) entry.capabilities = comboCaps; + // Any seat can serve the request, so the only window a combo can promise is + // its smallest. Combo entries were the only models on this endpoint that + // published no limits at all, which leaves a client to guess from the name — + // and it guesses high (see the snake_case note on the per-provider path). + const { contextWindow, maxOutput } = comboSeatLimits(combo, combosByName); + if (Number.isFinite(contextWindow)) entry.context_length = contextWindow; + if (Number.isFinite(maxOutput)) entry.max_completion_tokens = maxOutput; } models.push(entry); } if (connections.length === 0) { // DB unavailable -> return static models, filtered by per-model kind - const aliasToProviderId = Object.fromEntries( - Object.entries(PROVIDER_ID_TO_ALIAS).map(([id, alias]) => [alias, id]) - ); for (const [alias, providerModels] of Object.entries(PROVIDER_MODELS)) { - const providerId = aliasToProviderId[alias] || alias; + const providerId = ALIAS_TO_PROVIDER_ID[alias] || alias; if (!providerMatchesKinds(providerId, kindFilter)) continue; for (const model of providerModels) { if (!kindFilter.includes(modelKind(model))) continue; diff --git a/src/lib/modelCatalog/sync.js b/src/lib/modelCatalog/sync.js index 973c18b3..54f72d5a 100644 --- a/src/lib/modelCatalog/sync.js +++ b/src/lib/modelCatalog/sync.js @@ -24,6 +24,7 @@ const LIMIT_TOLERANCE = 0.1; // while building rather than on every lookup. Providers absent here keep whatever // the local pattern table resolves; names that already match need no entry. export const PROVIDER_ALIASES = { + "github": "github-copilot", "glm": "zai", "glm-cn": "zhipuai", "claude": "anthropic", diff --git a/tests/unit/gpt-6-context-window.test.js b/tests/unit/gpt-6-context-window.test.js new file mode 100644 index 00000000..e3f394cd --- /dev/null +++ b/tests/unit/gpt-6-context-window.test.js @@ -0,0 +1,112 @@ +import { describe, expect, it } from "vitest"; + +import { getCapabilitiesForModel, setCatalogSource } from "../../open-sse/providers/capabilities.js"; +import { PROVIDER_ALIASES, build } from "../../src/lib/modelCatalog/sync.js"; + +// The gpt-6 family's API window is 1.05M. The pattern table published 272,000 for +// it — Kiro's own truncation, copied into the global glob — so every other +// provider's gpt-6 models were advertised at 3.9x under their real window, and a +// client reading context_length compacted (or refused) far too early. +const API_WINDOW = 1050000; +const LEGACY_GPT5_WINDOW = 400000; + +describe("gpt-6 / gpt-5.4+ context windows", () => { + it("reports the 1.05M API window for gpt-6 models on ordinary providers", () => { + for (const [provider, model] of [ + ["github", "gpt-6-luna"], + ["azure", "gpt-6-luna"], + ["openai", "gpt-6-luna"], + ["github", "gpt-6-sol"], + ["openai", "gpt-6-astra"], + ]) { + expect(getCapabilitiesForModel(provider, model).contextWindow, `${provider}/${model}`).toBe(API_WINDOW); + } + }); + + // These two gateways really do truncate below the API, and their numbers live in + // PROVIDER_CAPABILITIES, which outranks the pattern. Correcting the pattern must + // leave them alone — that split is the whole point of the layering. + it("leaves the gateways that truncate lower on their own numbers", () => { + expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna").contextWindow).toBe(272000); + expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna-thinking-agentic").contextWindow).toBe(272000); + expect(getCapabilitiesForModel("codex", "gpt-5.6-luna").contextWindow).toBe(272000); + expect(getCapabilitiesForModel("codex", "gpt-5.6-sol").contextWindow).toBe(372000); + expect(getCapabilitiesForModel("codex", "gpt-6-astra").contextWindow).toBe(272000); + }); + + it("keeps Devin CLI's seven GPT-5.4/5.5 variants at the gateway's 200k limit", () => { + for (const model of [ + "gpt-5.4-high", "gpt-5.4-medium", "gpt-5.4-low", + "gpt-5.5-xhigh", "gpt-5.5-high", "gpt-5.5-medium", "gpt-5.5-low", + ]) { + expect(getCapabilitiesForModel("devin-cli", model), model).toMatchObject({ + contextWindow: 200000, + maxOutput: 128000, + vision: true, + reasoning: true, + thinkingFormat: "openai", + }); + } + expect(getCapabilitiesForModel("dv", "gpt-5.5-high").contextWindow).toBe(200000); + expect(getCapabilitiesForModel("devin", "gpt-5.5-high").contextWindow).toBe(200000); + }); + + // gpt-5.4 is where the 1.05M window starts and the mini/nano tiers are the + // exception that stayed at 400k. Pattern resolution is first-match-wins, so this + // is really a guard on the ORDER of the entries: move the tier patterns above + // the mini/nano ones and both tiers silently report 1.05M. + it("splits the 1.05M tiers from the 400k ones", () => { + for (const model of [ + "gpt-5.4", "gpt-5.4-pro", "gpt-5.5", "gpt-5.5-pro", "gpt-5.6", "gpt-5.6-luna", "gpt-5.6-terra", + ]) { + expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(API_WINDOW); + } + for (const model of ["gpt-5.4-mini", "gpt-5.4-nano", "gpt-5", "gpt-5.1", "gpt-5.2", "gpt-5.3-codex"]) { + expect(getCapabilitiesForModel("openai", model).contextWindow, model).toBe(LEGACY_GPT5_WINDOW); + } + }); +}); + +// Copilot is "github" locally and "github-copilot" upstream. Without that mapping +// build() resolved no upstream provider for it and skipped every Copilot model, so +// the daily models.dev sync could never correct a stale hand-written number — which +// is how the gpt-6 window stayed 3.9x wrong without anything noticing. +describe("models.dev sync reaches GitHub Copilot", () => { + it("maps the local github id onto the upstream github-copilot id", () => { + expect(PROVIDER_ALIASES.github).toBe("github-copilot"); + }); + + it("records a Copilot limit that disagrees with the local tables", () => { + // Copilot caps Claude output at 32k where the local floor assumes 64k. + const upstream = { + "github-copilot": { + models: { "claude-sonnet-4.6": { limit: { context: 200000, output: 32000 } } }, + }, + }; + const entries = [ + { provider: "github", model: "claude-sonnet-4.6", current: { contextWindow: 200000, maxOutput: 64000 } }, + ]; + + // Context agrees, so only the output delta is recorded — and it is filed under + // the local id, which is what the reader looks up. + expect(build(upstream, entries).providers.github).toEqual({ + "claude-sonnet-4.6": { maxOutput: 32000 }, + }); + }); + + it("applies those Copilot deltas to an exact MODEL_CAPABILITIES id", () => { + // claude-sonnet-4.6 is canonical-exact (128k). Without refine() on that + // path the 32k Copilot delta from build() would never be read. + setCatalogSource({ + getModalities: () => null, + getLimits: (provider, model) => + provider === "github" && model === "claude-sonnet-4.6" ? { maxOutput: 32000 } : null, + }); + try { + expect(getCapabilitiesForModel("github", "claude-sonnet-4.6").maxOutput).toBe(32000); + expect(getCapabilitiesForModel("claude", "claude-sonnet-4.6").maxOutput).toBe(128000); + } finally { + setCatalogSource(null); + } + }); +}); diff --git a/tests/unit/v1-models-combo-context.test.js b/tests/unit/v1-models-combo-context.test.js new file mode 100644 index 00000000..9f52bed2 --- /dev/null +++ b/tests/unit/v1-models-combo-context.test.js @@ -0,0 +1,83 @@ +import { describe, expect, it, vi } from "vitest"; +import { setCatalogSource } from "../../open-sse/providers/capabilities.js"; + +const db = vi.hoisted(() => ({ + getProviderConnections: vi.fn(), + getCombos: vi.fn(), + getCustomModels: vi.fn(async () => []), + getModelAliases: vi.fn(async () => ({})), +})); + +vi.mock("@/lib/localDb", () => db); +vi.mock("@/lib/disabledModelsDb", () => ({ + getDisabledModels: vi.fn(async () => ({})), +})); + +const { buildModelsList } = await import("../../src/app/api/v1/models/route.js"); + +const syncedLimits = { contextWindow: 180000, maxOutput: 16000 }; + +async function modelsWithCombo(providerId, modelId, combos) { + db.getProviderConnections.mockResolvedValue([{ + id: 1, + provider: providerId, + isActive: true, + providerSpecificData: { enabledModels: [modelId] }, + }]); + db.getCombos.mockResolvedValue(combos); + setCatalogSource({ + getModalities: () => null, + getLimits: (provider, model) => + provider === providerId && model === modelId ? syncedLimits : null, + }); + try { + return await buildModelsList(["llm"]); + } finally { + setCatalogSource(null); + } +} + +describe("/v1/models combo limits", () => { + it.each([ + ["ocg", "opencode-go", "mimo-v2.5"], + ["xmtp", "xiaomi-tokenplan", "mimo-v2.5"], + ["ps", "poolside", "custom-model"], + ["ds", "deepseek", "deepseek-chat"], + ])("uses the real provider for a %s UI-alias seat", async (uiAlias, providerId, modelId) => { + const combo = { name: "ui-alias-combo", models: [`${uiAlias}/${modelId}`] }; + const models = await modelsWithCombo(providerId, modelId, [combo]); + const published = models.find((model) => model.id === combo.name); + + expect(published).toMatchObject({ + context_length: 180000, + max_completion_tokens: 16000, + capabilities: { contextWindow: 180000, maxOutput: 16000 }, + }); + }); + + it("carries provider-scoped limits through a nested combo", async () => { + const models = await modelsWithCombo("opencode-go", "mimo-v2.5", [ + { name: "inner-combo", models: ["ocg/mimo-v2.5"] }, + { name: "outer-combo", models: ["inner-combo"] }, + ]); + const outer = models.find((model) => model.id === "outer-combo"); + + expect(outer).toMatchObject({ + context_length: 180000, + max_completion_tokens: 16000, + capabilities: { contextWindow: 180000, maxOutput: 16000 }, + }); + }); + + it("publishes Devin CLI's 200k limit for a dv combo seat", async () => { + const models = await modelsWithCombo("devin-cli", "gpt-5.5-high", [ + { name: "devin-combo", models: ["dv/gpt-5.5-high"] }, + ]); + const combo = models.find((model) => model.id === "devin-combo"); + + expect(combo).toMatchObject({ + context_length: 200000, + capabilities: { contextWindow: 200000 }, + }); + }); +}); From 75834e96fffa082ea07d03356a02225f47bb3016 Mon Sep 17 00:00:00 2001 From: Amirsalar Sojoudi Date: Thu, 1 Oct 2026 10:26:54 +0700 Subject: [PATCH 37/41] fix(claude): keep a trailing user turn so cleanup never yields assistant prefill Newer Claude models reject a body that ends on an assistant turn. The empty-message cleanups in prepareClaudeRequest and normalizeClaudePassthrough protect a trailing assistant but not a trailing user turn, so an emptied last user turn silently made the previous assistant turn the last one. ensureTrailingUserTurn appends a minimal user turn ("Continue.") only when the client did not itself end on assistant, in both cleanup paths and in translateRequest against the role the client actually sent. --- open-sse/translator/formats/claude.js | 18 +++++ open-sse/translator/index.js | 5 +- tests/unit/claude-trailing-user-turn.test.js | 75 ++++++++++++++++++++ 3 files changed, 97 insertions(+), 1 deletion(-) create mode 100644 tests/unit/claude-trailing-user-turn.test.js diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index 2795ecaa..96ef0ee6 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -226,6 +226,8 @@ export function normalizeClaudePassthrough(body, model = "") { if (Object.keys(body.output_config).length === 0) delete body.output_config; } + const originalLastRole = Array.isArray(body.messages) ? body.messages[body.messages.length - 1]?.role : undefined; + // 3. Wrap bare content-block objects as one-element arrays before folding. // Some clients send content: {block} instead of content: [{block}]; the // mid-conversation-system fold below assumes the array shape, so it must @@ -327,11 +329,25 @@ export function normalizeClaudePassthrough(body, model = "") { !(block?.type === CLAUDE_BLOCK.TEXT && !String(block.text ?? "").trim())); return msg.content.length > 0; }); + body.messages = ensureTrailingUserTurn(body.messages, originalLastRole); } return body; } +// Newer Claude models reject a body that ends on an assistant turn ("does not +// support assistant message prefill"). Cleanup passes delete messages left empty, +// so a trailing user turn that was empty (or held only dropped blocks) silently +// turns the previous assistant turn into the last one. Restore a user turn only +// when the client did not itself end on assistant (real prefill is its choice). +const TRAILING_USER_PLACEHOLDER = "Continue."; + +export function ensureTrailingUserTurn(messages, originalLastRole) { + if (!Array.isArray(messages) || originalLastRole === ROLE.ASSISTANT) return messages; + if (messages[messages.length - 1]?.role !== ROLE.ASSISTANT) return messages; + return [...messages, { role: ROLE.USER, content: [{ type: CLAUDE_BLOCK.TEXT, text: TRAILING_USER_PLACEHOLDER }] }]; +} + // Put a 5m breakpoint on the last cache-eligible block of a message. // thinking/redacted_thinking blocks do not accept cache_control. function markLastCacheableBlock(msg) { @@ -524,6 +540,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // 2. Messages: process in optimized passes if (body.messages && Array.isArray(body.messages)) { const len = body.messages.length; + const originalLastRole = body.messages[len - 1]?.role; let filtered = []; // Pass 1: remove cache_control + filter empty messages @@ -548,6 +565,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // Pass 1.5: Fix tool_use/tool_result ordering // Each tool_use must have tool_result in the NEXT message (not same message with other content) filtered = fixToolUseOrdering(filtered); + filtered = ensureTrailingUserTurn(filtered, originalLastRole); body.messages = filtered; diff --git a/open-sse/translator/index.js b/open-sse/translator/index.js index 8a1add5f..9f5953da 100644 --- a/open-sse/translator/index.js +++ b/open-sse/translator/index.js @@ -1,6 +1,6 @@ import { FORMATS } from "./formats.js"; import { ensureToolCallIds, fixMissingToolResponses } from "./concerns/toolCall.js"; -import { prepareClaudeRequest } from "./formats/claude.js"; +import { prepareClaudeRequest, ensureTrailingUserTurn } from "./formats/claude.js"; import { cloakClaudeTools, decloakStreamChunk } from "../utils/claudeCloaking.js"; import { restoreToolNames } from "../utils/opencodeFingerprint.js"; import { filterToOpenAIFormat } from "./formats/openai.js"; @@ -53,6 +53,8 @@ function stripContentTypes(body, stripList = []) { export function translateRequest(sourceFormat, targetFormat, model, body, stream = true, credentials = null, provider = null, reqLogger = null, stripList = [], connectionId = null, clientTool = null) { ensureInitialized(); let result = body; + // Role the client actually ended on, before any translator drops an emptied turn. + const clientLastRole = Array.isArray(body?.messages) ? body.messages[body.messages.length - 1]?.role : undefined; // Strip explicit content types (opt-in via strip[] in PROVIDER_MODELS entry) stripContentTypes(result, stripList); @@ -132,6 +134,7 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream if (targetFormat === FORMATS.CLAUDE) { const apiKey = credentials?.accessToken || credentials?.apiKey || null; result = prepareClaudeRequest(result, provider, apiKey, connectionId, credentials?.rawHeaders, clientSessionId); + if (Array.isArray(result?.messages)) result.messages = ensureTrailingUserTurn(result.messages, clientLastRole); } // Claude cloaking: rename client tools with CLAUDE_TOOL_SUFFIX (anti-ban) diff --git a/tests/unit/claude-trailing-user-turn.test.js b/tests/unit/claude-trailing-user-turn.test.js new file mode 100644 index 00000000..d197648a --- /dev/null +++ b/tests/unit/claude-trailing-user-turn.test.js @@ -0,0 +1,75 @@ +// Anthropic rejects a body ending on an assistant turn ("This model does not support +// assistant message prefill"). Cleanup passes delete emptied messages, so an emptied +// trailing user turn used to leave the previous assistant turn last. +import { describe, it, expect } from "vitest"; +import { normalizeClaudePassthrough, prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; +import { translateRequest } from "../../open-sse/translator/index.js"; + +const roles = (body) => body.messages.map((m) => m.role); +const history = (last) => [ + { role: "user", content: "hi" }, + { role: "assistant", content: [{ type: "text", text: "hello" }] }, + last, +]; + +const emptyLastTurns = { + "empty string": { role: "user", content: "" }, + "blank text block": { role: "user", content: [{ type: "text", text: " " }] }, + "empty content array": { role: "user", content: [] }, + "unsupported block only": { role: "user", content: [{ type: "search_result", source: "x", title: "t", content: [] }] }, +}; + +describe("trailing user turn survives empty-message cleanup", () => { + for (const [name, last] of Object.entries(emptyLastTurns)) { + it(`prepareClaudeRequest: ${name}`, () => { + const out = prepareClaudeRequest({ model: "claude-opus-4-5", max_tokens: 100, messages: history(last) }, "claude"); + expect(roles(out)).toEqual(["user", "assistant", "user"]); + }); + it(`normalizeClaudePassthrough: ${name}`, () => { + const out = normalizeClaudePassthrough({ model: "claude-opus-4-5", messages: history(last) }, "claude-opus-4-5"); + expect(roles(out)).toEqual(["user", "assistant", "user"]); + }); + } + + it("passthrough: tool_result of a dropped foreign server_tool_use no longer empties the last turn into prefill", () => { + const out = normalizeClaudePassthrough({ + model: "claude-opus-4-5", + messages: [ + { role: "user", content: "analyze" }, + { role: "assistant", content: [{ type: "server_tool_use", id: "call_abc", name: "analyze_image", input: {} }, { type: "text", text: "done" }] }, + { role: "user", content: [{ type: "web_search_tool_result", tool_use_id: "call_abc", content: [] }] }, + ], + }, "claude-opus-4-5"); + expect(roles(out)).toEqual(["user", "assistant", "user"]); + }); + + it("full pipeline: OpenAI client with an empty last user message", () => { + const out = translateRequest("openai", "claude", "claude-opus-4-5", { + model: "x", max_tokens: 100, + messages: [{ role: "user", content: "hi" }, { role: "assistant", content: "yo" }, { role: "user", content: "" }], + }, true, null, "claude"); + expect(out.messages.at(-1).role).toBe("user"); + }); + + it("full pipeline: Claude client with a blank last user block", () => { + const out = translateRequest("claude", "claude", "claude-opus-4-5", { + model: "x", max_tokens: 100, messages: history({ role: "user", content: [{ type: "text", text: "" }] }), + }, true, null, "claude"); + expect(out.messages.at(-1).role).toBe("user"); + }); + + it("leaves intentional client prefill (last turn is assistant) untouched", () => { + const body = { model: "claude-opus-4-5", max_tokens: 100, messages: [ + { role: "user", content: "hi" }, + { role: "assistant", content: [{ type: "text", text: "Sure:" }] }, + ] }; + expect(roles(prepareClaudeRequest(structuredClone(body), "claude"))).toEqual(["user", "assistant"]); + expect(roles(normalizeClaudePassthrough(structuredClone(body), "claude-opus-4-5"))).toEqual(["user", "assistant"]); + }); + + it("does not append anything when the last user turn has content", () => { + const out = prepareClaudeRequest({ model: "claude-opus-4-5", max_tokens: 100, messages: history({ role: "user", content: "next" }) }, "claude"); + expect(roles(out)).toEqual(["user", "assistant", "user"]); + expect(out.messages.at(-1).content[0].text).toBe("next"); + }); +}); From 5e9bd464f269f63badd302078b191b85dcc7c70b Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 1 Oct 2026 10:28:20 +0700 Subject: [PATCH 38/41] fix(claude): preserve intentional prefill from non-messages[] source formats MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit translateRequest detected the client's terminal role only from messages[], so a Gemini contents[] or Responses input[] body ending on an explicit model/assistant turn (real prefill) read as undefined and got the "Continue." restoration appended, clobbering the prefill. Map the trailing role per source shape; only model/assistant counts as prefill — every other tail stays undefined so the emptied-turn fix still applies. Co-Authored-By: Claude Code --- open-sse/translator/index.js | 17 +++++++- ...-trailing-user-turn-source-formats.test.js | 42 +++++++++++++++++++ 2 files changed, 58 insertions(+), 1 deletion(-) create mode 100644 tests/unit/claude-trailing-user-turn-source-formats.test.js diff --git a/open-sse/translator/index.js b/open-sse/translator/index.js index 9f5953da..292ffa36 100644 --- a/open-sse/translator/index.js +++ b/open-sse/translator/index.js @@ -9,6 +9,7 @@ import { applyThinking, captureThinking } from "./concerns/thinkingUnified.js"; import { captureSessionId } from "../utils/sessionManager.js"; import { AntigravityExecutor } from "../executors/antigravity.js"; import { PROVIDERS } from "../providers/index.js"; +import { ROLE, GEMINI_ROLE } from "./schema/roles.js"; // Registry for translators. Lazy-init guards against circular-import order: // translator modules call register() (side-effect) before this module's body runs. @@ -49,12 +50,26 @@ function stripContentTypes(body, stripList = []) { } } +// Role the client's conversation actually ended on, in the source format's own +// shape — not every source uses messages[] (Gemini/Antigravity: contents[], +// Responses/Codex: input[]). Only an explicit trailing model/assistant turn is +// real prefill and must reach ensureTrailingUserTurn as ROLE.ASSISTANT; every +// other tail (including no role, e.g. a function output) stays undefined so +// the emptied-turn fix still applies. +function detectClientLastRole(body) { + if (Array.isArray(body?.messages)) return body.messages[body.messages.length - 1]?.role; + const items = Array.isArray(body?.contents) ? body.contents : Array.isArray(body?.input) ? body.input : null; + if (!items) return undefined; + const role = items[items.length - 1]?.role; + return role === ROLE.ASSISTANT || role === GEMINI_ROLE.MODEL ? ROLE.ASSISTANT : undefined; +} + // Translate request: source -> openai -> target export function translateRequest(sourceFormat, targetFormat, model, body, stream = true, credentials = null, provider = null, reqLogger = null, stripList = [], connectionId = null, clientTool = null) { ensureInitialized(); let result = body; // Role the client actually ended on, before any translator drops an emptied turn. - const clientLastRole = Array.isArray(body?.messages) ? body.messages[body.messages.length - 1]?.role : undefined; + const clientLastRole = detectClientLastRole(body); // Strip explicit content types (opt-in via strip[] in PROVIDER_MODELS entry) stripContentTypes(result, stripList); diff --git a/tests/unit/claude-trailing-user-turn-source-formats.test.js b/tests/unit/claude-trailing-user-turn-source-formats.test.js new file mode 100644 index 00000000..84d8a31a --- /dev/null +++ b/tests/unit/claude-trailing-user-turn-source-formats.test.js @@ -0,0 +1,42 @@ +// Non-messages[] sources (Gemini contents[], Responses input[]) carry the client's +// terminal role in their own shape. An explicit trailing model/assistant turn is real +// prefill and must survive translation to Claude; an emptied trailing user turn must +// still get the "Continue." restoration. +import { describe, it, expect } from "vitest"; +import { translateRequest } from "../../open-sse/translator/index.js"; + +const roles = (body) => body.messages.map((m) => m.role); + +describe("trailing user turn: non-messages[] source formats", () => { + it("keeps a Gemini trailing model turn (real prefill)", () => { + const out = translateRequest("gemini", "claude", "claude-sonnet-4-5", { + contents: [ + { role: "user", parts: [{ text: "hi" }] }, + { role: "model", parts: [{ text: "The answer is" }] }, + ], + }, false); + expect(roles(out)).toEqual(["user", "assistant"]); + }); + + it("keeps a Responses trailing assistant message (real prefill)", () => { + const out = translateRequest("openai-responses", "claude", "claude-sonnet-4-5", { + model: "claude-sonnet-4-5", + input: [ + { role: "user", content: "hi" }, + { role: "assistant", content: "The answer is" }, + ], + }, false); + expect(roles(out)).toEqual(["user", "assistant"]); + }); + + it("still restores a user turn when a Gemini trailing user turn is emptied", () => { + const out = translateRequest("gemini", "claude", "claude-sonnet-4-5", { + contents: [ + { role: "user", parts: [{ text: "hi" }] }, + { role: "model", parts: [{ text: "hello" }] }, + { role: "user", parts: [] }, + ], + }, false); + expect(roles(out)).toEqual(["user", "assistant", "user"]); + }); +}); From 7111db3598bce6acc5028a40d04953a6fcfea9f7 Mon Sep 17 00:00:00 2001 From: Amirsalar Sojoudi Date: Thu, 1 Oct 2026 10:40:33 +0700 Subject: [PATCH 39/41] fix(responses): wait for real usage before emitting response.completed --- open-sse/translator/index.js | 5 + .../translator/response/openai-responses.js | 33 +-- open-sse/utils/stream.js | 16 ++ .../openai-responses-completed-usage.test.js | 219 ++++++++++++++++++ .../unit/openai-responses-usage-pivot.test.js | 116 ++++++++++ 5 files changed, 376 insertions(+), 13 deletions(-) create mode 100644 tests/unit/openai-responses-completed-usage.test.js create mode 100644 tests/unit/openai-responses-usage-pivot.test.js diff --git a/open-sse/translator/index.js b/open-sse/translator/index.js index 292ffa36..cfccca84 100644 --- a/open-sse/translator/index.js +++ b/open-sse/translator/index.js @@ -287,6 +287,11 @@ export function initState(sourceFormat) { funcArgsDone: {}, funcItemDone: {}, customToolNames: new Set(), + // Chat Completions usage for response.completed. Not state.usage: other translators in + // the same pipeline overwrite that in their own shapes. + responsesUsage: null, + // finish_reason arrived before usage; response.completed waits for the usage chunk. + completionPending: false, completedSent: false }; } diff --git a/open-sse/translator/response/openai-responses.js b/open-sse/translator/response/openai-responses.js index 6a865d1a..1499a2a5 100644 --- a/open-sse/translator/response/openai-responses.js +++ b/open-sse/translator/response/openai-responses.js @@ -27,17 +27,22 @@ import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM, OPENAI_FINISH, MODEL_FALLBACK } fro function toResponsesUsage(usage) { if (!usage || typeof usage !== "object") return null; - const inputTokens = [usage.input_tokens, usage.prompt_tokens].find(Number.isFinite) ?? 0; - const outputTokens = [usage.output_tokens, usage.completion_tokens].find(Number.isFinite) ?? 0; + const inputTokens = [usage.input_tokens, usage.prompt_tokens].find(Number.isInteger); + const outputTokens = [usage.output_tokens, usage.completion_tokens].find(Number.isInteger); + // Some upstreams attach zeroed placeholders to every chunk. Wait for real counts + // so response.completed cannot freeze the placeholder before the usage trailer. + if (inputTokens === undefined || outputTokens === undefined || inputTokens + outputTokens <= 0) { + return null; + } const responseUsage = { input_tokens: inputTokens, output_tokens: outputTokens, - total_tokens: Number.isFinite(usage.total_tokens) ? usage.total_tokens : inputTokens + outputTokens + total_tokens: inputTokens + outputTokens }; - const cachedTokens = [usage.input_tokens_details?.cached_tokens, usage.prompt_tokens_details?.cached_tokens].find(Number.isFinite); - const reasoningTokens = [usage.output_tokens_details?.reasoning_tokens, usage.completion_tokens_details?.reasoning_tokens].find(Number.isFinite); - if (Number.isFinite(cachedTokens)) responseUsage.input_tokens_details = { cached_tokens: cachedTokens }; - if (Number.isFinite(reasoningTokens)) responseUsage.output_tokens_details = { reasoning_tokens: reasoningTokens }; + const cachedTokens = [usage.input_tokens_details?.cached_tokens, usage.prompt_tokens_details?.cached_tokens].find(Number.isInteger); + const reasoningTokens = [usage.output_tokens_details?.reasoning_tokens, usage.completion_tokens_details?.reasoning_tokens].find(Number.isInteger); + if (Number.isInteger(cachedTokens)) responseUsage.input_tokens_details = { cached_tokens: cachedTokens }; + if (Number.isInteger(reasoningTokens)) responseUsage.output_tokens_details = { reasoning_tokens: reasoningTokens }; return responseUsage; } @@ -47,13 +52,14 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { return flushEvents(state); } - // Capture upstream usage BEFORE the choices guard below: the last OpenAI chunk - // may carry usage together with an empty choices array, and it must not be dropped. - if (chunk.usage) { - state.responsesUsage = toResponsesUsage(chunk.usage); - } + // Capture usage before the choices guard: OpenAI may send it in a trailer + // whose choices array is empty. + const responseUsage = toResponsesUsage(chunk.usage); + if (responseUsage) state.responsesUsage = responseUsage; - if (!chunk.choices?.length) return []; + if (!chunk.choices?.length) { + return state.completionPending && state.responsesUsage ? flushEvents(state) : []; + } const events = []; const nextSeq = () => ++state.seq; @@ -163,6 +169,7 @@ export function openaiToOpenAIResponsesResponse(chunk, state) { // would swallow the terminal event entirely. Keep the old behaviour there. const flushReachesUs = state.targetFormat === FORMATS.OPENAI; if (state.responsesUsage || !flushReachesUs) sendCompleted(state, emit); + else state.completionPending = true; } return events; diff --git a/open-sse/utils/stream.js b/open-sse/utils/stream.js index cc0c6e30..91a041f4 100644 --- a/open-sse/utils/stream.js +++ b/open-sse/utils/stream.js @@ -266,6 +266,22 @@ export function createSSEStream(options = {}) { // For Ollama: done=true is the final chunk with finish_reason/usage, must translate // For other formats: done=true is the [DONE] sentinel, skip if (parsed && parsed.done && targetFormat !== FORMATS.OLLAMA) { + // A direct Chat-to-Responses translation can defer response.completed + // while waiting for a usage trailer. [DONE] ends that opportunity even + // if the upstream keeps the HTTP connection open, so finish now. + if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES && + state.completionPending && !state.completedSent) { + const completed = translateResponse(targetFormat, sourceFormat, null, state); + for (const item of completed || []) { + if (item === null || item === undefined) continue; + const output = formatSSE(item, sourceFormat); + reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(sharedEncoder.encode(output)); + sseEmittedCount++; + } + finalizeStream(); + } + // Synthesize response.failed if the Responses stream never sent a terminal event if (keepsOpenAIResponsesFormat && !openAIResponsesTerminalSeen) { const failedOutput = formatIncompleteOpenAIResponsesStreamFailure(); diff --git a/tests/unit/openai-responses-completed-usage.test.js b/tests/unit/openai-responses-completed-usage.test.js new file mode 100644 index 00000000..f8f8005c --- /dev/null +++ b/tests/unit/openai-responses-completed-usage.test.js @@ -0,0 +1,219 @@ +import { describe, expect, it } from "vitest"; + +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { createSSETransformStreamWithLogger } from "../../open-sse/utils/stream.js"; + +// Codex compacts its history only from the token usage reported on `response.completed` +// (sess.get_total_token_usage). Without it Codex never compacts and eventually sends a +// prompt larger than the model's context window. + +async function runTransform(targetFormat, lines) { + const encoder = new TextEncoder(); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(lines.join("\n"))); + controller.close(); + }, + }); + + const output = stream.pipeThrough( + createSSETransformStreamWithLogger(targetFormat, FORMATS.OPENAI_RESPONSES, "test", null, null, "test-model"), + ); + + const reader = output.getReader(); + const decoder = new TextDecoder(); + let text = ""; + while (true) { + const { value, done } = await reader.read(); + if (done) break; + text += decoder.decode(value, { stream: true }); + } + return text + decoder.decode(); +} + +// Parse the client-facing SSE into [{ event, data }], skipping the [DONE] sentinel. +function parseEvents(text) { + return text + .split("\n\n") + .map((block) => { + const event = block.match(/^event: (.+)$/m)?.[1]; + const data = block.match(/^data: (.+)$/m)?.[1]; + if (!event || !data || data === "[DONE]") return null; + return { event, data: JSON.parse(data) }; + }) + .filter(Boolean); +} + +const sse = (data) => [`data: ${JSON.stringify(data)}`, ""]; +const claudeSse = (data) => [`event: ${data.type}`, `data: ${JSON.stringify(data)}`, ""]; + +// Mirrors ResponseCompletedUsage in codex-rs/codex-api/src/sse/responses.rs. The three totals +// are required i64s and the details are optional, but each detail field is a required i64 +// when present. Any deviation fails Codex's parse of the whole event, killing the turn. +function expectCodexUsageShape(usage) { + for (const field of ["input_tokens", "output_tokens", "total_tokens"]) { + expect(Number.isInteger(usage[field]), field).toBe(true); + } + if (usage.input_tokens_details) { + expect(Number.isInteger(usage.input_tokens_details.cached_tokens)).toBe(true); + } + if (usage.output_tokens_details) { + expect(Number.isInteger(usage.output_tokens_details.reasoning_tokens)).toBe(true); + } +} + +describe("Responses response.completed reports token usage", () => { + it("reports Claude upstream usage to a Codex client, counting cached prompt tokens", async () => { + const events = parseEvents(await runTransform(FORMATS.CLAUDE, [ + ...claudeSse({ + type: "message_start", + message: { + id: "msg_1", type: "message", role: "assistant", model: "claude-opus-5", content: [], + usage: { input_tokens: 1000, cache_read_input_tokens: 200, cache_creation_input_tokens: 0, output_tokens: 1 }, + }, + }), + ...claudeSse({ type: "content_block_start", index: 0, content_block: { type: "text", text: "" } }), + ...claudeSse({ type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "Hello" } }), + ...claudeSse({ type: "content_block_stop", index: 0 }), + ...claudeSse({ type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 50 } }), + ...claudeSse({ type: "message_stop" }), + ])); + + const completed = events.filter((e) => e.event === "response.completed"); + expect(completed).toHaveLength(1); + + const usage = completed[0].data.response.usage; + expectCodexUsageShape(usage); + expect(usage).toMatchObject({ + input_tokens: 1200, + output_tokens: 50, + total_tokens: 1250, + input_tokens_details: { cached_tokens: 200 }, + }); + }); + + it("waits for a trailing usage-only chunk instead of completing on finish_reason", async () => { + const events = parseEvents(await runTransform(FORMATS.OPENAI, [ + ...sse({ id: "c1", object: "chat.completion.chunk", choices: [{ index: 0, delta: { role: "assistant", content: "Hi" } }] }), + ...sse({ id: "c1", object: "chat.completion.chunk", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }), + ...sse({ + id: "c1", object: "chat.completion.chunk", choices: [], + usage: { + prompt_tokens: 300, completion_tokens: 20, total_tokens: 320, + prompt_tokens_details: { cached_tokens: 100 }, + completion_tokens_details: { reasoning_tokens: 5 }, + }, + }), + "data: [DONE]", + "", + ])); + + const completed = events.filter((e) => e.event === "response.completed"); + expect(completed).toHaveLength(1); + // Terminal event last: Codex stops reading at response.completed. + expect(events.at(-1).event).toBe("response.completed"); + + const usage = completed[0].data.response.usage; + expectCodexUsageShape(usage); + expect(usage).toEqual({ + input_tokens: 300, + output_tokens: 20, + total_tokens: 320, + input_tokens_details: { cached_tokens: 100 }, + output_tokens_details: { reasoning_tokens: 5 }, + }); + }); + + it("ignores zeroed placeholder usage on every chunk and reports the real trailing counts", async () => { + const placeholder = { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 }; + const events = parseEvents(await runTransform(FORMATS.OPENAI, [ + ...sse({ id: "c3", object: "chat.completion.chunk", usage: placeholder, choices: [{ index: 0, delta: { role: "assistant", content: "Hi" } }] }), + ...sse({ id: "c3", object: "chat.completion.chunk", usage: placeholder, choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }), + ...sse({ id: "c3", object: "chat.completion.chunk", choices: [], usage: { prompt_tokens: 300, completion_tokens: 20, total_tokens: 320 } }), + "data: [DONE]", + "", + ])); + + const completed = events.filter((e) => e.event === "response.completed"); + expect(completed).toHaveLength(1); + expect(completed[0].data.response.usage).toEqual({ input_tokens: 300, output_tokens: 20, total_tokens: 320 }); + }); + + it.each([0, 999])("derives the total when upstream reports inconsistent total_tokens=%i", async (totalTokens) => { + const events = parseEvents(await runTransform(FORMATS.OPENAI, [ + ...sse({ id: "c4", object: "chat.completion.chunk", choices: [{ index: 0, delta: { role: "assistant", content: "Hi" } }] }), + ...sse({ + id: "c4", object: "chat.completion.chunk", + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + usage: { prompt_tokens: 300, completion_tokens: 20, total_tokens: totalTokens }, + }), + "data: [DONE]", + "", + ])); + + const completed = events.filter((e) => e.event === "response.completed"); + expect(completed).toHaveLength(1); + expect(completed[0].data.response.usage).toEqual({ + input_tokens: 300, + output_tokens: 20, + total_tokens: 320, + }); + }); + + it("completes at [DONE] while the upstream connection remains open", async () => { + let upstream; + const input = new ReadableStream({ + start(controller) { + upstream = controller; + }, + }); + const reader = input.pipeThrough( + createSSETransformStreamWithLogger(FORMATS.OPENAI, FORMATS.OPENAI_RESPONSES, "test"), + ).getReader(); + const decoder = new TextDecoder(); + const frames = [ + ...sse({ id: "c5", object: "chat.completion.chunk", choices: [{ index: 0, delta: { role: "assistant", content: "Hi" } }] }), + ...sse({ id: "c5", object: "chat.completion.chunk", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }), + "data: [DONE]", + "", + ]; + upstream.enqueue(new TextEncoder().encode(frames.join("\n") + "\n")); + + let timer; + try { + let output = ""; + const readUntilCompleted = async () => { + while (!output.includes('"type":"response.completed"')) { + const { value, done } = await reader.read(); + if (done) throw new Error("stream ended before response.completed"); + output += decoder.decode(value, { stream: true }); + } + }; + await Promise.race([ + readUntilCompleted(), + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error("response.completed waited for transport EOF")), 500); + }), + ]); + expect(parseEvents(output).filter((e) => e.event === "response.completed")).toHaveLength(1); + } finally { + clearTimeout(timer); + upstream.close(); + await reader.cancel(); + } + }); + + it("still completes exactly once, without inventing usage, when the upstream reports none", async () => { + const events = parseEvents(await runTransform(FORMATS.OPENAI, [ + ...sse({ id: "c2", object: "chat.completion.chunk", choices: [{ index: 0, delta: { role: "assistant", content: "Hi" } }] }), + ...sse({ id: "c2", object: "chat.completion.chunk", choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }), + "data: [DONE]", + "", + ])); + + const completed = events.filter((e) => e.event === "response.completed"); + expect(completed).toHaveLength(1); + expect(events.at(-1).event).toBe("response.completed"); + expect(completed[0].data.response).not.toHaveProperty("usage"); + }); +}); diff --git a/tests/unit/openai-responses-usage-pivot.test.js b/tests/unit/openai-responses-usage-pivot.test.js new file mode 100644 index 00000000..c9f96af7 --- /dev/null +++ b/tests/unit/openai-responses-usage-pivot.test.js @@ -0,0 +1,116 @@ +import { describe, expect, it } from "vitest"; + +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { createSSETransformStreamWithLogger } from "../../open-sse/utils/stream.js"; + +/** + * Usage must survive the PIVOT, not just the direct openai:openai-responses route. + * + * Codex talks the Responses API, so routing it at a Claude connection runs + * claude -> openai -> openai-responses. The converter that attaches usage to + * response.completed is the second hop, and it only ever sees the intermediate + * OpenAI chunk — so whether Codex learns its context size depends on the first + * hop putting usage on that intermediate chunk. + * + * Signature is (targetFormat, sourceFormat, ...) — targetFormat is what the + * UPSTREAM speaks, sourceFormat is what the CLIENT speaks. + */ +async function runTransform(chunks, targetFormat, provider) { + const encoder = new TextEncoder(); + const input = chunks.map((c) => `data: ${JSON.stringify(c)}\n\n`).join(""); + + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(input)); + controller.close(); + }, + }); + + const output = stream.pipeThrough( + createSSETransformStreamWithLogger( + targetFormat, + FORMATS.OPENAI_RESPONSES, + provider, + null, + null, + "claude-sonnet-5", + ), + ); + + const reader = output.getReader(); + const decoder = new TextDecoder(); + let text = ""; + + while (true) { + const { value, done } = await reader.read(); + if (done) break; + text += decoder.decode(value, { stream: true }); + } + + text += decoder.decode(); + return text; +} + +function completedResponse(output) { + const lines = output + .split("\n") + .filter((l) => l.startsWith("data: ") && l.includes('"type":"response.completed"')); + expect(lines.length, "expected exactly one response.completed").toBe(1); + return JSON.parse(lines[0].slice(6)).response; +} + +// Anthropic splits the counts across two events: message_start carries the whole +// prompt side (input + both cache buckets), message_delta carries only the output +// side. Neither event alone is the total, which is why the claude converter merges +// them into state before emitting the intermediate chunk. +const CLAUDE_CHUNKS_WITH_USAGE = [ + { + type: "message_start", + message: { + id: "msg_01CfUtmFqMv3Gc5s66ehaTK", + model: "claude-sonnet-5", + usage: { + input_tokens: 1500, + cache_read_input_tokens: 12000, + cache_creation_input_tokens: 300, + output_tokens: 1, + }, + }, + }, + { type: "content_block_start", index: 0, content_block: { type: "text", text: "" } }, + { type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "hi" } }, + { type: "content_block_stop", index: 0 }, + { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 42 } }, + { type: "message_stop" }, +]; + +describe("OpenAI Responses usage across the pivot", () => { + // The reported failure: a Codex session on a Claude connection grew unbounded + // (101 -> 503 -> 631 messages) until Anthropic rejected it with + // "prompt is too long: 1676806 tokens > 1000000 maximum", because every + // token_count event Codex recorded had info: null. + it("reports claude usage on response.completed so Codex can auto-compact", async () => { + const output = await runTransform(CLAUDE_CHUNKS_WITH_USAGE, FORMATS.CLAUDE, "claude"); + + // prompt side = input + cache_read + cache_creation = 1500 + 12000 + 300. + expect(completedResponse(output).usage).toEqual({ + input_tokens: 13800, + output_tokens: 42, + total_tokens: 13842, + input_tokens_details: { cached_tokens: 12000 }, + }); + }); + + // Codex deserializes usage into a struct whose three top-level counts are all + // required, so dropping any one of them discards the whole object and leaves the + // context gauge empty — the same end state as reporting nothing. + it("always reports all three top-level counts", async () => { + const output = await runTransform(CLAUDE_CHUNKS_WITH_USAGE, FORMATS.CLAUDE, "claude"); + const usage = completedResponse(output).usage; + + for (const field of ["input_tokens", "output_tokens", "total_tokens"]) { + expect(usage, `missing ${field}`).toHaveProperty(field); + expect(Number.isFinite(usage[field]), `${field} must be a number`).toBe(true); + } + }); +}); From fbcaa2828c0775503899c22df5e6bb04790758ec Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 1 Oct 2026 10:42:12 +0700 Subject: [PATCH 40/41] fix(responses): bound the deferred completion wait with a 3s watchdog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A chat->responses stream defers response.completed waiting for a usage trailer (PR #4476). A broken upstream that stalls after finish_reason — no trailer, no [DONE], connection held open — made that wait unbounded. Flush the pending completion after 3s instead. Normal paths (usage trailer, [DONE], connection close) flush immediately and clear the watchdog, so only a stalled stream ever pays the delay. Co-Authored-By: Claude Code --- open-sse/utils/stream.js | 43 +++++++-- ...enai-responses-completion-watchdog.test.js | 87 +++++++++++++++++++ 2 files changed, 121 insertions(+), 9 deletions(-) create mode 100644 tests/unit/openai-responses-completion-watchdog.test.js diff --git a/open-sse/utils/stream.js b/open-sse/utils/stream.js index 91a041f4..790f59be 100644 --- a/open-sse/utils/stream.js +++ b/open-sse/utils/stream.js @@ -22,6 +22,11 @@ const STREAM_MODE = { PASSTHROUGH: "passthrough" // No translation, normalize output, extract usage }; +// Upper bound on the deferred response.completed wait: a chat->responses stream +// that saw finish_reason without usage must not hold the client's terminal event +// forever when the upstream stalls with no usage trailer and no [DONE]. +const PENDING_COMPLETION_FLUSH_MS = 3000; + /** * Create unified SSE transform stream * @param {object} options @@ -83,10 +88,12 @@ export function createSSEStream(options = {}) { let openAIResponsesDoneSent = false; let streamDoneSent = false; // track duplicate [DONE] across transform + flush let finalized = false; + let completionFlushTimer = null; // Usage/logging tail, callable from transform() as well as flush(): a client that // closes right after the terminal event cancels the reader, and flush() never runs. const finalizeStream = () => { + if (completionFlushTimer) { clearTimeout(completionFlushTimer); completionFlushTimer = null; } if (finalized) return; finalized = true; @@ -112,6 +119,20 @@ export function createSSEStream(options = {}) { } }; + // Emit the deferred response.completed now — at [DONE], or when the watchdog + // below gives up on a usage trailer that never arrives. + const flushPendingCompletion = (controller) => { + const completed = translateResponse(targetFormat, sourceFormat, null, state); + for (const item of completed || []) { + if (item === null || item === undefined) continue; + const output = formatSSE(item, sourceFormat); + reqLogger?.appendConvertedChunk?.(output); + controller.enqueue(sharedEncoder.encode(output)); + sseEmittedCount++; + } + finalizeStream(); + }; + return new TransformStream({ transform(chunk, controller) { if (!ttftAt) ttftAt = Date.now(); @@ -271,15 +292,7 @@ export function createSSEStream(options = {}) { // if the upstream keeps the HTTP connection open, so finish now. if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES && state.completionPending && !state.completedSent) { - const completed = translateResponse(targetFormat, sourceFormat, null, state); - for (const item of completed || []) { - if (item === null || item === undefined) continue; - const output = formatSSE(item, sourceFormat); - reqLogger?.appendConvertedChunk?.(output); - controller.enqueue(sharedEncoder.encode(output)); - sseEmittedCount++; - } - finalizeStream(); + flushPendingCompletion(controller); } // Synthesize response.failed if the Responses stream never sent a terminal event @@ -393,6 +406,18 @@ export function createSSEStream(options = {}) { sseEmittedCount++; } } + + // The completion deferral can outlive the upstream: a broken chat upstream + // may stall after finish_reason with no usage trailer and no [DONE], holding + // the connection open. Bound the wait so the client still gets a terminal event. + if (targetFormat === FORMATS.OPENAI && sourceFormat === FORMATS.OPENAI_RESPONSES && + state?.completionPending && !state?.completedSent && !completionFlushTimer) { + completionFlushTimer = setTimeout(() => { + completionFlushTimer = null; + if (state?.completedSent) return; + try { flushPendingCompletion(controller); } catch { /* controller already closed */ } + }, PENDING_COMPLETION_FLUSH_MS); + } } }, diff --git a/tests/unit/openai-responses-completion-watchdog.test.js b/tests/unit/openai-responses-completion-watchdog.test.js new file mode 100644 index 00000000..2874fe4f --- /dev/null +++ b/tests/unit/openai-responses-completion-watchdog.test.js @@ -0,0 +1,87 @@ +import { describe, expect, it, vi } from "vitest"; + +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { createSSETransformStreamWithLogger } from "../../open-sse/utils/stream.js"; + +// A chat->responses stream defers response.completed when finish_reason arrives +// without usage (PR #4476). If the upstream then stalls — no usage trailer, no +// [DONE], connection held open — that deferral must not wait forever: the +// watchdog flushes the terminal event after PENDING_COMPLETION_FLUSH_MS. +const encoder = new TextEncoder(); + +const FINISH_CHUNK = { + id: "chatcmpl-1", + choices: [{ index: 0, delta: { content: "hi" }, finish_reason: "stop" }], +}; + +const USAGE_TRAILER = { id: "chatcmpl-1", choices: [], usage: { prompt_tokens: 120, completion_tokens: 30 } }; + +function completedResponses(text) { + return text + .split("\n") + .filter((l) => l.startsWith("data: ") && l.includes('"type":"response.completed"')) + .map((l) => JSON.parse(l.slice(6)).response); +} + +async function readAll(reader) { + const decoder = new TextDecoder(); + let text = ""; + while (true) { + const { value, done } = await reader.read(); + if (done) break; + text += decoder.decode(value, { stream: true }); + } + return text + decoder.decode(); +} + +async function pipe() { + let source; + const input = new ReadableStream({ start(c) { source = c; } }); + const output = input.pipeThrough( + createSSETransformStreamWithLogger(FORMATS.OPENAI, FORMATS.OPENAI_RESPONSES, "test", null, null, "gpt-test"), + ); + return { source, reader: output.getReader() }; +} + +describe("pending response.completed watchdog", () => { + it("flushes the deferred completion when the upstream stalls after finish_reason", async () => { + vi.useFakeTimers(); + try { + const { source, reader } = await pipe(); + source.enqueue(encoder.encode(`data: ${JSON.stringify(FINISH_CHUNK)}\n\n`)); + await vi.advanceTimersByTimeAsync(20); + + // No trailer, no [DONE] — only the watchdog can close this out. + await vi.advanceTimersByTimeAsync(3000); + source.close(); + + const completed = completedResponses(await readAll(reader)); + expect(completed.length, "exactly one response.completed").toBe(1); + expect(completed[0].status).toBe("completed"); + expect(completed[0].usage, "no usage was ever reported").toBeUndefined(); + } finally { + vi.useRealTimers(); + } + }); + + it("does not fire after the real usage trailer already completed the stream", async () => { + vi.useFakeTimers(); + try { + const { source, reader } = await pipe(); + source.enqueue(encoder.encode(`data: ${JSON.stringify(FINISH_CHUNK)}\n\n`)); + await vi.advanceTimersByTimeAsync(20); + source.enqueue(encoder.encode(`data: ${JSON.stringify(USAGE_TRAILER)}\n\n`)); + await vi.advanceTimersByTimeAsync(20); + + // Well past the watchdog window: nothing more may be emitted. + await vi.advanceTimersByTimeAsync(10000); + source.close(); + + const completed = completedResponses(await readAll(reader)); + expect(completed.length, "exactly one response.completed").toBe(1); + expect(completed[0].usage).toMatchObject({ input_tokens: 120, output_tokens: 30 }); + } finally { + vi.useRealTimers(); + } + }); +}); From a99cf57239ff778b61e434c2786009d5ed1c412c Mon Sep 17 00:00:00 2001 From: decolua Date: Thu, 1 Oct 2026 10:45:29 +0700 Subject: [PATCH 41/41] # v0.5.95 (2026-10-01) ## Features - **Providers**: add Meta Muse provider with OAuth login and model catalog; add v1m System One provider - **GLM**: add Z.ai OAuth login to GLM Coding (dual-auth) - **Codex**: add GPT-6.1 Sol; expose 1M context variants for GPT-6 and GPT-5.6; add gpt-daybreak/reserve models and route bare `gpt-5.x`/`gpt-6.x` slugs to codex - **Claude**: add Claude Sonnet 5.5 (plus `claude-opus-5.5` models in the Kiro registry) - **CLI**: add `connect` command for remote 9Router servers - **Providers**: per-provider custom header overrides from the registry - **Agnes**: seed the 2.5/3.0 model ids in the registry - **Usage**: sync `?provider=` URL param with provider filter for bookmarkable deep links (#4395) - **Dashboard**: drop NEW badges in sidebar, mark 9Remote as HOT ## Fixes - **Claude**: preserve intentional prefill from non-messages[] source formats; keep a trailing user turn so cleanup never yields assistant prefill - **Claude**: cache a tool loop's final tool results with the 4th breakpoint - **Claude**: resolve Sonnet 5.x to adaptive thinking so no forged thinking placeholders are sent; inject unsigned thinking placeholders for opencode-go DeepSeek `/messages` (#4436) - **Thinking**: add `xhigh` to claude-adaptive thinking levels - **Claude**: keep a user turn whose only block is `container_upload` - **Capabilities**: publish real GPT-6/GPT-5.4+ context windows and combo token limits - **Responses**: wait for real usage before emitting `response.completed`, bounded by a 3s watchdog - **Codex**: stop refresh-token reuse that logs accounts out on auto-ping; preserve hosted web search on GPT-6 Sol/Luna; remove ghost models - **Grok CLI**: send Grok CLI 1.0.44 so proxy stops returning HTTP 426 - **Proxy**: auto-fallback to insecure TLS on self-signed cert errors; hold strictProxy when no proxy resolves - **Translator**: strip `errorMessage` and other non-standard schema keywords from Gemini tool schemas; dedupe same-name tools for DeepSeek models (#3333) - **Codebuddy**: parse the 6004 rate limit error and extract `resetsAtMs`; forward `recurring` for codebuddy-intl quota packs (#4422) - **CLI Tools**: replace `sk_9router` placeholder with first active dashboard API key - **Dashboard**: exclude hidden providers from usage stats provider list - **Capabilities**: add deepseek-v4-1-flash vision alias; add zed to live catalog providers --- CHANGELOG.md | 30 ++++++++++++++++++++++++++++++ cli/package.json | 2 +- package.json | 2 +- 3 files changed, 32 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b6d8cd61..11afb966 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,33 @@ +# v0.5.95 (2026-10-01) + +## Features +- **Providers**: add Meta Muse provider with OAuth login and model catalog; add v1m System One provider +- **GLM**: add Z.ai OAuth login to GLM Coding (dual-auth) +- **Codex**: add GPT-6.1 Sol; expose 1M context variants for GPT-6 and GPT-5.6; add gpt-daybreak/reserve models and route bare `gpt-5.x`/`gpt-6.x` slugs to codex +- **Claude**: add Claude Sonnet 5.5 (plus `claude-opus-5.5` models in the Kiro registry) +- **CLI**: add `connect` command for remote 9Router servers +- **Providers**: per-provider custom header overrides from the registry +- **Agnes**: seed the 2.5/3.0 model ids in the registry +- **Usage**: sync `?provider=` URL param with provider filter for bookmarkable deep links (#4395) +- **Dashboard**: drop NEW badges in sidebar, mark 9Remote as HOT + +## Fixes +- **Claude**: preserve intentional prefill from non-messages[] source formats; keep a trailing user turn so cleanup never yields assistant prefill +- **Claude**: cache a tool loop's final tool results with the 4th breakpoint +- **Claude**: resolve Sonnet 5.x to adaptive thinking so no forged thinking placeholders are sent; inject unsigned thinking placeholders for opencode-go DeepSeek `/messages` (#4436) +- **Thinking**: add `xhigh` to claude-adaptive thinking levels +- **Claude**: keep a user turn whose only block is `container_upload` +- **Capabilities**: publish real GPT-6/GPT-5.4+ context windows and combo token limits +- **Responses**: wait for real usage before emitting `response.completed`, bounded by a 3s watchdog +- **Codex**: stop refresh-token reuse that logs accounts out on auto-ping; preserve hosted web search on GPT-6 Sol/Luna; remove ghost models +- **Grok CLI**: send Grok CLI 1.0.44 so proxy stops returning HTTP 426 +- **Proxy**: auto-fallback to insecure TLS on self-signed cert errors; hold strictProxy when no proxy resolves +- **Translator**: strip `errorMessage` and other non-standard schema keywords from Gemini tool schemas; dedupe same-name tools for DeepSeek models (#3333) +- **Codebuddy**: parse the 6004 rate limit error and extract `resetsAtMs`; forward `recurring` for codebuddy-intl quota packs (#4422) +- **CLI Tools**: replace `sk_9router` placeholder with first active dashboard API key +- **Dashboard**: exclude hidden providers from usage stats provider list +- **Capabilities**: add deepseek-v4-1-flash vision alias; add zed to live catalog providers + # v0.5.91 (2026-09-26) ## Features diff --git a/cli/package.json b/cli/package.json index 86755891..646cf879 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.91", + "version": "0.5.95", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" diff --git a/package.json b/package.json index cb0de5b3..a4240c30 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "9router-app", - "version": "0.5.91", + "version": "0.5.95", "description": "9Router web dashboard", "private": true, "scripts": {