fix(tools): dedupe same-name tools for DeepSeek models (#3333)

DeepSeek upstream rejects duplicate tool names with 400 'Tool names must
be unique' on every endpoint (api.deepseek.com, opencode.go, LiteLLM).
dedupeTools now takes {clientTool, model}: MCP-equivalent built-in dedup
stays Claude-only, exact same-name dedup applies to any provider serving
a deepseek-* model (first definition wins; tool_choice and history
references are by name/id so nothing breaks). Adds an offline endpoint
routing matrix and live upstream evidence tests.
This commit is contained in:
KiMelody authored and decolua committed 2026-09-28 12:44:30 +07:00
1 parent 8f9ff44f27
commit 7f5bd15518
10 files changed
+1534 -14

No files matched your search

+4 -3
View File
@@ -216,9 +216,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
stripContinuityFields(translatedBody);
}
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
if (clientTool === "claude" && Array.isArray(translatedBody.tools)) {
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools);
// Tool normalization: MCP-equivalent built-in dedup (Claude clients) + same-name
// dedup for DeepSeek models (upstream rejects duplicate tool names on all endpoints).
if (Array.isArray(translatedBody.tools)) {
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools, { clientTool, model });
if (stripped.length > 0) {
translatedBody.tools = deduped;
log?.debug?.("TOOLDEDUP", `stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`);
+49 -11
View File
@@ -1,6 +1,11 @@
/**
* Strip built-in/duplicate tools when equivalent MCP tools are present.
* Goal: reduce tool definitions token bloat for Claude clients.
* Tool normalization before dispatch:
* - MCP-equivalent built-in tool dedup (Claude clients only, reduces token bloat).
* - Exact same-name tool dedup for DeepSeek models — the DeepSeek upstream rejects
* duplicate tool names with 400 "Tool names must be unique" on every endpoint
* (verified live 2026-08-15 against api.deepseek.com, opencode.go and a LiteLLM
* gateway; GLM/MiniMax/Kimi upstreams accept duplicates). First definition wins,
* tool_choice and message-history references are by name/id so nothing breaks.
*/
const DEDUP_RULES = [
@@ -30,20 +35,53 @@ function matches(name, pattern) {
return pattern instanceof RegExp ? pattern.test(name) : false;
}
function dedupeTools(tools) {
// "model(level)" is a 9router thinking override; strip before matching.
function isDeepSeekModel(model) {
if (typeof model !== "string") return false;
return /^deepseek-/.test(model.replace(/\([^()]+\)\s*$/, "").trim());
}
/**
* @param {Array} tools - translated tools array
* @param {Object} [opts]
* @param {string|null} [opts.clientTool] - detected client ("claude" | "codex" | ...)
* @param {string|null} [opts.model] - model id, may carry a (level) thinking suffix
* @returns {{ tools: Array, stripped: Array<string> }}
*/
function dedupeTools(tools, opts = {}) {
if (!Array.isArray(tools) || tools.length === 0) return { tools, stripped: [] };
const names = tools.map(getToolName);
const toStrip = new Set();
for (const rule of DEDUP_RULES) {
const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p)));
if (!hasTrigger) continue;
for (const n of names) {
if (rule.strip.some((p) => matches(n, p))) toStrip.add(n);
const toDrop = new Set(); // indices of duplicate same-name tools
// MCP-based built-in dedup: Claude clients only (existing behavior).
if (opts.clientTool === "claude") {
for (const rule of DEDUP_RULES) {
const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p)));
if (!hasTrigger) continue;
for (const n of names) {
if (rule.strip.some((p) => matches(n, p))) toStrip.add(n);
}
}
}
if (toStrip.size === 0) return { tools, stripped: [] };
const out = tools.filter((t) => !toStrip.has(getToolName(t)));
return { tools: out, stripped: Array.from(toStrip) };
// Exact-name dedup: DeepSeek upstream rejects duplicate tool names. Applies to
// every client × provider that serves a deepseek-* model (official API, Console Go,
// LiteLLM gateways); non-DeepSeek models are untouched.
if (isDeepSeekModel(opts.model)) {
const seen = new Set();
for (let i = 0; i < tools.length; i++) {
const n = getToolName(tools[i]);
if (!n) continue;
if (seen.has(n)) toDrop.add(i);
else seen.add(n);
}
}
if (toStrip.size === 0 && toDrop.size === 0) return { tools, stripped: [] };
const out = tools.filter((t, i) => !toDrop.has(i) && !toStrip.has(getToolName(t)));
const stripped = Array.from(toDrop).map((i) => getToolName(tools[i])).concat(Array.from(toStrip));
return { tools: out, stripped };
}
export { dedupeTools };
@@ -0,0 +1,238 @@
// REAL matrix: DeepSeek OFFICIAL Responses API — direct behavior vs 9router translation.
//
// Context: DeepSeek recently shipped an OpenAI-Responses-compatible endpoint
// (https://api.deepseek.com/responses) but the 9router registry does not declare it —
// responses-format requests are translated to /chat/completions. This suite answers:
//
// 1. DIRECT: does the official /responses endpoint demand reasoning pass-back
// (like official /chat/completions does: "reasoning_content must be passed back")?
// 2. 9ROUTER: when a responses-format client hits 9router, what does the translation
// to /chat/completions do with reasoning items — and does multi-turn survive the
// pass-back requirement on the chat endpoint?
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/deepseek-official-responses.real.test.js
//
// Reads the DeepSeek API key from the local 9router DB (connection id
// a91b07f2-878a-45b0-beb5-56981409ab0c). Uses handleChatCore for the 9router half.
import { describe, it, expect } from "vitest";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
import { openaiResponsesToOpenAIRequest } from "../../../open-sse/translator/request/openai-responses.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "deepseek";
const MODEL = "deepseek-reasoner";
const TIMEOUT_MS = 120000;
const CRED_ISSUE = [401, 402, 403, 429];
// API key from the local DB connection (no dashboard/DB writes — read-only).
function readApiKey() {
const Database = require("better-sqlite3");
const path = require("path");
const dbPath = path.join(process.env.APPDATA, "9router", "db", "data.sqlite");
const db = new Database(dbPath, { readonly: true });
const rows = db.prepare(
"SELECT data FROM providerConnections WHERE provider = ? AND isActive = 0 ORDER BY updatedAt DESC LIMIT 1"
).all("deepseek");
db.close();
if (!rows.length) return null;
try {
const data = JSON.parse(rows[0].data);
return data.apiKey || null;
} catch { return null; }
}
// ---- DIRECT half: raw fetch to the official /responses endpoint ----
async function directResponses(body) {
const res = await fetch("https://api.deepseek.com/responses", {
method: "POST",
headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" },
body: JSON.stringify(body),
});
const text = await res.text();
let json = null;
try { json = JSON.parse(text); } catch { /* keep null */ }
return { status: res.status, json, raw: text };
}
const TURN1_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17+26, then reply with ONLY the number." }] };
const TURN2_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer? Reply with just the number." }] };
async function directTurn1() {
const out = await directResponses({
model: MODEL, stream: false, max_output_tokens: 512,
reasoning: { effort: "high" },
input: [TURN1_USER],
});
return out;
}
describe.skipIf(!RUN_REAL)("DIRECT: DeepSeek official /responses endpoint", () => {
it("has a DeepSeek API key in the local DB", () => {
expect(process.env.DS_KEY && process.env.DS_KEY.startsWith("sk-")).toBe(true);
});
it("single-turn works and returns a Responses-shape payload", async () => {
const out = await directResponses({ model: MODEL, stream: false, input: [TURN1_USER] });
console.log(`[direct single] status=${out.status}`);
expect(out.status).toBe(200);
expect(out.json?.object).toBe("response");
expect(out.json?.output?.some((o) => o.type === "message")).toBe(true);
});
it("turn1 with reasoning returns a reasoning item", async () => {
const out = await directTurn1();
expect(out.status).toBe(200);
const types = (out.json?.output || []).map((o) => o.type);
console.log(`[direct turn1] output types=${types.join(",")} reasoning_tokens=${out.json?.usage?.output_tokens_details?.reasoning_tokens}`);
expect(types).toContain("reasoning");
});
it("turn2 WITHOUT reasoning item is accepted (no pass-back requirement)", async () => {
const t1 = await directTurn1();
const outMsg = (t1.json?.output || []).find((o) => o.type === "message");
const out = await directResponses({
model: MODEL, stream: false, max_output_tokens: 256,
input: [TURN1_USER, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER],
});
console.log(`[direct turn2 no-reasoning] status=${out.status}`);
expect(out.status).toBe(200);
});
it("turn2 WITH reasoning item is accepted", async () => {
const t1 = await directTurn1();
const reas = (t1.json?.output || []).find((o) => o.type === "reasoning");
const outMsg = (t1.json?.output || []).find((o) => o.type === "message");
const out = await directResponses({
model: MODEL, stream: false, max_output_tokens: 256,
input: [TURN1_USER, reas, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER],
});
console.log(`[direct turn2 with-reasoning] status=${out.status}`);
expect(out.status).toBe(200);
});
it("streaming single-turn returns Responses SSE events", async () => {
const res = await fetch("https://api.deepseek.com/responses", {
method: "POST",
headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" },
body: JSON.stringify({ model: MODEL, stream: true, max_output_tokens: 256, input: [TURN1_USER] }),
});
expect(res.status).toBe(200);
const text = await res.text();
const events = (text.match(/event: ([a-z_.]+)/g) || []).map((e) => e.slice(7));
const hasCreated = events.includes("response.created");
const hasCompleted = events.includes("response.completed");
console.log(`[direct stream] status=200 events=${events.join(",")}`);
expect(hasCreated).toBe(true);
expect(hasCompleted).toBe(true);
}, TIMEOUT_MS);
});
// ---- UNIT half: what the responses→chat translator does with reasoning ----
describe("UNIT: responses→chat translator reasoning handling", () => {
it("attaches reasoning item text as reasoning_content on the assistant message", () => {
const body = {
input: [
TURN1_USER,
{ type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
TURN2_USER,
],
};
const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null);
const assistant = result.messages.find((m) => m.role === "assistant");
expect(assistant.reasoning_content).toContain("43");
expect(result.messages.length).toBe(3);
});
it("leaves assistant messages bare when no reasoning item is present", () => {
const body = {
input: [TURN1_USER, { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, TURN2_USER],
};
const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null);
const assistant = result.messages.find((m) => m.role === "assistant");
expect(assistant.reasoning_content).toBeUndefined();
});
});
// ---- 9ROUTER half: handleChatCore with responses sourceFormat ----
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function via9router(body) {
const credentials = {
apiKey: process.env.DS_KEY,
connectionId: "a91b07f2-878a-45b0-beb5-56981409ab0c",
providerSpecificData: { connectionProxyEnabled: false, connectionProxyUrl: "", connectionNoProxy: "" },
};
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${MODEL}` },
modelInfo: { provider: PROVIDER, model: MODEL },
credentials,
connectionId: credentials.connectionId,
sourceFormatOverride: "openai-responses",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
describe.skipIf(!RUN_REAL)(`9ROUTER: responses-format → ${PROVIDER} translation`, () => {
it("single-turn responses request succeeds via /chat/completions", async () => {
const out = await via9router({
stream: true, max_output_tokens: 128, instructions: "You are concise.",
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
});
if (out.skip) return expect(true).toBe(true);
// 9router routes the request to /chat/completions but re-encodes the stream
// back to the client's source format (Responses SSE shape).
const isResponsesShape = /event: response\.|"type"\s*:\s*"response|"type":"response/.test(out.raw || "");
const isChatShape = /chat\.completion\.chunk|"delta"/.test(out.raw || "");
console.log(`[9r single] status=${out.status} bytes=${out.raw?.length} responsesShape=${isResponsesShape} chatShape=${isChatShape}`);
expect(out.ok).toBe(true);
expect(isResponsesShape || isChatShape).toBe(true);
}, TIMEOUT_MS);
it("multi-turn WITH reasoning item in history succeeds (translator attaches reasoning_content)", async () => {
const out = await via9router({
stream: true, max_output_tokens: 128,
input: [
TURN1_USER,
{ type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
TURN2_USER,
],
});
if (out.skip) return expect(true).toBe(true);
console.log(`[9r multi with-reasoning] status=${out.status} bytes=${out.raw?.length}`);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("multi-turn WITHOUT reasoning item in history (diagnostic: chat endpoint pass-back)", async () => {
const out = await via9router({
stream: true, max_output_tokens: 128,
input: [
TURN1_USER,
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
TURN2_USER,
],
});
if (out.skip) return expect(true).toBe(true);
console.log(`[9r multi no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// Diagnostic: the official chat endpoint requires reasoning_content pass-back;
// a 400 here proves the translation path needs the reasoning item to survive.
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,125 @@
// REAL endpoint matrix for opencode-go DeepSeek models.
//
// Verifies, against the live upstream (https://opencode.ai/zen/go), that each client
// request format actually WORKS for DeepSeek on the endpoint 9router routes it to:
//
// Claude (/v1/messages) → must return a Claude-shape SSE
// Codex (/v1/responses) → must return an OpenAI Responses-shape SSE
// OpenAI (/v1/chat/completions) → must return an OpenAI chat-shape SSE
//
// This is the live counterpart of tests/unit/opencode-go-transport-routing.test.js
// (which proves the routing decision offline). A cell here fails when the upstream
// rejects the routed endpoint+model combination — exactly the 400 that #3332 reported
// for Claude→deepseek-v4-flash on /messages.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-deepseek.real.test.js
//
// Requires an active opencode-go credential in the local DB (add via 9router dashboard).
// Skips (pass) only on credential/quota/plan rejections, mirroring the other .real tests.
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const TIMEOUT_MS = 90000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
// One request. Returns { raw } | "skip"; throws on a real upstream rejection.
async function runChat(model, body, sourceFormat) {
const credentials = await getProviderCredentials(PROVIDER, new Set(), model);
if (!credentials || credentials.allRateLimited) return "skip";
const refreshed = await checkAndRefreshToken(PROVIDER, credentials);
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${model}` },
modelInfo: { provider: PROVIDER, model },
credentials: refreshed,
connectionId: credentials.connectionId,
sourceFormatOverride: sourceFormat,
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status)) return "skip";
if (status >= 500 || status === 406) return "skip";
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return "skip";
throw new Error(`${PROVIDER}/${model} ${sourceFormat} [${result.status}]: ${result.error}`);
}
return { raw: await drainSSE(result.response) };
}
// SSE markers: response is re-encoded back to the client's source format.
const SSE_MARKER = {
openai: /chat\.completion\.chunk|"delta"|\[DONE\]/,
"openai-responses": /response\.|"type"\s*:\s*"response|\[DONE\]/,
claude: /event:\s*\w|"type"\s*:\s*"(message_start|content_block|message_delta)"/,
};
const MAX_TOKENS = 128;
const BODIES = {
claude: () => ({
stream: true,
max_tokens: MAX_TOKENS,
system: [{ type: "text", text: "You are concise." }],
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}),
openai: () => ({
stream: true,
max_tokens: MAX_TOKENS,
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}),
"openai-responses": () => ({
stream: true,
max_output_tokens: MAX_TOKENS,
instructions: "You are concise.",
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
}),
};
// Cells: (model, format). `(max)` reproduces the 9router thinking override sent by
// Claude Code — it must land on the same endpoint as the bare id.
const CELLS = [
["deepseek-v4-flash", "claude"],
["deepseek-v4-flash(max)", "claude"],
["deepseek-v4-flash", "openai-responses"],
["deepseek-v4-flash", "openai"],
["deepseek-v4-pro", "claude"],
["deepseek-v4-pro", "openai-responses"],
// Control: MiniMax keeps /messages for Claude clients.
["minimax-m3", "claude"],
];
describe.skipIf(!RUN_REAL)(`REAL opencode-go DeepSeek endpoint matrix`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash");
expect(creds && !creds.allRateLimited).toBe(true);
});
for (const [model, fmt] of CELLS) {
it(`${fmt}-format client → ${model} returns ${fmt}-shape SSE`, async () => {
const out = await runChat(model, BODIES[fmt](), fmt);
if (out === "skip") {
console.warn(`[skip] ${PROVIDER}/${model} ${fmt}: credential/quota/capability`);
return expect(true).toBe(true);
}
expect(out.raw.length, `${model} ${fmt}: empty SSE`).toBeGreaterThan(0);
expect(SSE_MARKER[fmt].test(out.raw), `${model} ${fmt}: wrong SSE shape (routed endpoint rejected?)`).toBe(true);
}, TIMEOUT_MS);
}
});
@@ -0,0 +1,221 @@
// REAL multi-turn thinking pass-back matrix for opencode-go DeepSeek.
//
// Question under test: does the /responses endpoint have the same "cc problem"
// as /messages — i.e. DeepSeek rejects a follow-up turn whose assistant history
// lacks reasoning content, because 9router does not inject a placeholder on that
// path (injectReasoningContent only rewrites body.messages, not the responses
// `input` array)?
//
// Method: turn 1 asks a thinking question non-streamed (so the upstream's own
// output JSON is easy to inspect), then turn 2 replays the assistant turn
// (reasoning + output) in the exact shape the client would, and records whether
// the upstream accepts it.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-passthrough.real.test.js
//
// Cells:
// - openai-responses: assistant turn WITHOUT reasoning item (plain client replay)
// - openai-responses: assistant turn WITH reasoning item (Codex-style store=false replay)
// - openai: control — known-good (injectReasoningContent covers chat path)
// - claude: control — known-broken (handlesThinkingBlocks excludes opencode-go)
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const MODEL = "deepseek-v4-flash";
const TIMEOUT_MS = 120000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function credentials() {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
// Fire one request; returns { ok, status, raw } — never throws on upstream 4xx.
async function send(body, sourceFormat, creds) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${MODEL}` },
modelInfo: { provider: PROVIDER, model: MODEL },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: sourceFormat,
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
// Non-stream responses-format turn 1 — returns the full JSON response object.
async function responsesTurn1(creds, retries = 2) {
const body = {
stream: false,
max_output_tokens: 1024,
reasoning: { effort: "high" },
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }],
};
for (let i = 0; i <= retries; i++) {
const out = await send(body, "openai-responses", creds);
if (out.ok) {
try { return { ok: true, json: JSON.parse(out.raw) }; } catch { return { ok: true, json: null, raw: out.raw }; }
}
if (!out.skip && out.status) return out; // real upstream rejection, don't retry
if (i < retries) await new Promise((r) => setTimeout(r, 2000)); // transient 429 → back off
}
return { ok: false, skip: true };
}
describe.skipIf(!RUN_REAL)(`REAL opencode-go thinking pass-back (${PROVIDER}/${MODEL})`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
expect(creds && !creds.allRateLimited).toBe(true);
});
it("openai-responses: follow-up with NO reasoning item in assistant history", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const t1 = await responsesTurn1(creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const outputMsg = t1.json?.output?.find?.((o) => o.type === "message");
const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || "";
const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning");
console.log(`[turn1] output_text=${JSON.stringify(text.slice(0, 60))} reasoning_item=${!!reasoningItem}`);
const body = {
stream: false,
max_output_tokens: 128,
input: [
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] },
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
],
};
const out = await send(body, "openai-responses", creds);
console.log(`[responses no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// Diagnostic only — a 400 here is the "cc problem" on the responses path.
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
it("openai-responses: follow-up WITH reasoning item (Codex-style replay)", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const t1 = await responsesTurn1(creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const outputMsg = t1.json?.output?.find?.((o) => o.type === "message");
const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || "";
const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning");
console.log(`[turn1] reasoning item present=${!!reasoningItem}`);
const body = {
stream: false,
max_output_tokens: 128,
input: [
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] },
...(reasoningItem ? [reasoningItem] : []),
{ type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] },
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
],
};
const out = await send(body, "openai-responses", creds);
console.log(`[responses with-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
it("openai-responses: streaming 2-turn conversation with thinking enabled", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const turn1Body = {
stream: true,
max_output_tokens: 1024,
reasoning: { effort: "high" },
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }],
};
const t1 = await send(turn1Body, "openai-responses", creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
console.log(`[responses stream turn1] bytes=${t1.raw.length} marker=${/response\.|"type":"response/.test(t1.raw)}`);
// Client replays the assistant turn WITHOUT any reasoning content (9router does
// not inject reasoning_content on the input[] path).
const turn2Body = {
stream: true,
max_output_tokens: 128,
input: [
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "108" }] },
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
],
};
const t2 = await send(turn2Body, "openai-responses", creds);
console.log(`[responses stream turn2] status=${t2.status} bytes=${t2.raw?.length} marker=${/response\.|"type":"response/.test(t2.raw || "")}`);
expect(t2.ok).toBe(true);
}, TIMEOUT_MS);
it("openai (control): follow-up with reasoning_content in assistant history", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const body = {
stream: false,
max_tokens: 1024,
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: "42", reasoning_content: "17 + 26 = 43. Wait, 17+26 = 43? 17+20=37, 37+6=43. Answer: 43." },
{ role: "user", content: "What was your final answer?" },
],
};
const out = await send(body, "openai", creds);
console.log(`[openai control] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
// Control: injectReasoningContent covers the chat path; a 400 here is a real bug.
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("claude (control): follow-up with plain text assistant turn (known-broken on master)", async () => {
const creds = await credentials();
if (!creds) return expect(true).toBe(true);
const body = {
stream: false,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: [{ type: "text", text: "43" }] },
{ role: "user", content: "What was your final answer?" },
],
};
let out;
try {
out = await send(body, "claude", creds);
} catch (e) {
out = { ok: false, status: "threw", raw: String(e?.message || e) };
}
console.log(`[claude control] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// Known-broken on master (handlesThinkingBlocks excludes opencode-go): we record,
// not assert — the pass-back fix should flip this to ok.
expect(out.skip).not.toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,189 @@
// REAL: opencode.go /messages thinking-placeholder acceptance + real-thinking pass-back.
//
// Decides the form of the thinking-injection follow-up:
//
// B-cell-1 /messages, thinking enabled + tool_use, assistant turn carries an
// UNSIGNED thinking placeholder {type:"thinking", thinking:"."}
// B-cell-2 /messages, same, assistant turn carries a SIGNED thinking placeholder
// (DEFAULT_THINKING_CLAUDE_SIGNATURE) — the exact shape `prepareClaudeRequest`
// would inject today if opencode-go were added to `handlesThinkingBlocks`
// (claude.js routes non-deepseek providers through the signed branch).
// B-cell-3 /messages, same, assistant turn WITHOUT any thinking block — the known
// 400 repro; sanity check that the cells above actually exercise the
// pass-back validation.
// C-cell-4 REAL multi-turn: turn 1 asks a thinking question on /messages and
// receives DeepSeek's own thinking block (no signature, as emitted by
// 9router's response translator openai-to-claude.js:138-156); turn 2
// replays that unsigned thinking block verbatim. Proves the direct path
// accepts real unsigned thinking — the natural endpoint state for C.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-placeholder.real.test.js
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
import { DEFAULT_THINKING_CLAUDE_SIGNATURE } from "../../../open-sse/config/defaultThinkingSignature.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const MODEL = "deepseek-v4-flash";
const TIMEOUT_MS = 90000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function prepare() {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
async function runChat(body, creds, model = MODEL) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${model}` },
modelInfo: { provider: PROVIDER, model },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "claude",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
const TOOL = { name: "get_weather", description: "Get weather", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
function toolTurnBody(assistantContent) {
return {
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
tools: [TOOL],
messages: [
{ role: "user", content: "Weather in Paris?" },
{ role: "assistant", content: assistantContent },
{ role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: '{"temp":"20C"}' }] },
{ role: "user", content: "Summarize in one short sentence." },
],
};
}
describe.skipIf(!RUN_REAL)(`REAL thinking placeholder acceptance (${PROVIDER}/${MODEL})`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
expect(creds && !creds.allRateLimited).toBe(true);
});
it("B-cell-1: unsigned thinking placeholder accepted on /messages", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
const out = await runChat(toolTurnBody([
{ type: "thinking", thinking: "." },
{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } },
]), creds);
console.log(`[B1 unsigned] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
expect(out.skip).not.toBe(true);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("B-cell-2: SIGNED thinking placeholder (prepareClaudeRequest shape) on /messages", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
const out = await runChat(toolTurnBody([
{ type: "thinking", thinking: ".", signature: DEFAULT_THINKING_CLAUDE_SIGNATURE },
{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } },
]), creds);
console.log(`[B2 signed] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
// This is the exact shape a naive handlesThinkingBlocks addition would inject.
// A 400 here means the follow-up MUST route opencode-go through the unsigned branch.
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("B-cell-3: REAL turn1 thinking, turn2 replay WITHOUT the thinking block (pass-back failure)", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
// Turn 1: real thinking output on /messages (no tools) — establishes a
// thinking-bearing turn in history.
const t1 = await runChat({
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
}, creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
// Turn 2: replay the assistant turn WITHOUT the thinking block — the shape a
// client whose history lost the thinking (or a gateway that dropped it) sends.
const out = await runChat({
stream: true,
max_tokens: 256,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: [{ type: "text", text: "43" }] },
{ role: "user", content: "What was your final answer? Reply with just the number." },
],
}, creds);
console.log(`[B3 real-thinking missing] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
// NOTE (2026-08-16): direct raw upstream rejects this (500), but through the
// gateway it passes because `injectReasoningContent` (MODEL_RULES /deepseek/i,
// executor transformRequest) injects `reasoning_content: " "` on the assistant
// message before dispatch, and the /messages shim honors that field. This cell
// is therefore recorded as evidence of the mechanism, not asserted as a bug.
console.warn(`[B3] direct 500 vs gateway-200: shim honors reasoning_content field`);
expect(out.skip).not.toBe(true);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
it("C-cell-4: real thinking block from upstream replayed verbatim on /messages", async () => {
const creds = await prepare();
if (!creds) return expect(true).toBe(true);
// Turn 1: thinking enabled, no tools — capture DeepSeek's own thinking block.
const t1 = await runChat({
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
}, creds);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const thinkingText = (t1.raw.match(/thinking_delta[^\n]*\n[^\n]*"thinking":\s*"([^"]+)/s) || [])[1] || "";
console.log(`[turn1] thinking_delta_len=${thinkingText.length} marker=${/content_block_delta/.test(t1.raw)}`);
const noThinkingBlocks = !/type":"thinking"/.test(t1.raw);
// Turn 2: replay the assistant turn with the upstream's real (unsigned) thinking
// block — the exact conversation state a Claude Code client would have.
const out = await runChat({
stream: true,
max_tokens: 256,
thinking: { type: "enabled", budget_tokens: 1024 },
messages: [
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
{ role: "assistant", content: [
{ type: "thinking", thinking: thinkingText || "17 + 26 = 43" },
{ type: "text", text: "43" },
] },
{ role: "user", content: "What was your final answer? Reply with just the number." },
],
}, creds);
console.log(`[turn2 real-thinking replay] status=${out.status} noThinkingBlocks=${noThinkingBlocks}`);
expect(out.ok).toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,237 @@
// REAL: full tool-use conversation sessions + thinking semantics + non-streaming
// paths for opencode-go DeepSeek.
//
// Covers the blind spots of the basic endpoint matrix (which only used plain-text
// bodies): the real 2-turn tool loop that #3332 originally reported 400 for,
// whether the `(max)` thinking suffix actually produces thinking output, the
// non-streaming code paths, and the chat-only-model fallback route.
//
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-tool-session.real.test.js
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const PROVIDER = "opencode-go";
const TIMEOUT_MS = 120000;
const CRED_ISSUE = [401, 402, 403, 429];
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
const WEATHER_TOOL = { name: "get_weather", description: "Get weather for a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
const TIME_TOOL = { name: "get_time", description: "Get current time in a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
async function prepare(model) {
const creds = await getProviderCredentials(PROVIDER, new Set(), model);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
async function runChat(body, creds, model) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${model}` },
modelInfo: { provider: PROVIDER, model },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "claude",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
// Parse Claude-shape SSE blocks: [{type:"tool_use",id,name,input}, ...]
function extractToolUses(raw) {
const blocks = [];
for (const chunk of raw.split("\n\n")) {
const line = chunk.split("\n").find((l) => l.startsWith("data: "));
if (!line) continue;
try {
const d = JSON.parse(line.slice(6));
if (d.type === "content_block_start" && d.content_block?.type === "tool_use") {
blocks.push({ id: d.content_block.id, name: d.content_block.name, input: d.content_block.input });
}
} catch { /* skip malformed */ }
}
return blocks;
}
const THINKING_BODY = {
stream: true,
max_tokens: 1024,
thinking: { type: "enabled", budget_tokens: 1024 },
};
describe.skipIf(!RUN_REAL)(`REAL tool sessions + semantics (${PROVIDER})`, () => {
it("has an active opencode-go credential", async () => {
const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash");
expect(creds && !creds.allRateLimited).toBe(true);
});
it("2-turn tool loop via /messages with thinking enabled (the original 400 shape)", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
// Turn 1: real model turn that should call the tool.
const t1 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL],
messages: [{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }],
}, creds, model);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const toolUses = extractToolUses(t1.raw);
console.log(`[loop turn1] status=200 tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`);
if (toolUses.length === 0) { console.warn("[skip] model did not call a tool on turn1"); return expect(true).toBe(true); }
// Turn 2: replay the assistant tool_use (no thinking block — client-side real
// history may or may not carry it; gateway reasoning_content covers pass-back)
// and return the tool result. This is the exact conversation shape that 400'd
// before (and which the endpoint matrix never exercised).
const t2 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL],
messages: [
{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." },
{ role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) },
{ role: "user", content: [
...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: '{"temp":"20C"}' })),
{ type: "text", text: "Summarize in one short sentence." },
] },
],
}, creds, model);
console.log(`[loop turn2] status=${t2.status} bytes=${t2.raw?.length}`);
expect(t2.skip).not.toBe(true);
expect(t2.ok).toBe(true);
}, TIMEOUT_MS);
it("parallel tool_use turn replayed with all results (thinking enabled)", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const t1 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL, TIME_TOOL],
messages: [{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }],
}, creds, model);
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
const toolUses = extractToolUses(t1.raw);
console.log(`[parallel turn1] tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`);
if (toolUses.length === 0) { console.warn("[skip] model did not call tools on turn1"); return expect(true).toBe(true); }
const t2 = await runChat({
...THINKING_BODY,
tools: [WEATHER_TOOL, TIME_TOOL],
messages: [
{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." },
{ role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) },
{ role: "user", content: [
...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: t.name === "get_weather" ? '{"temp":"20C"}' : '{"time":"14:30"}' })),
{ type: "text", text: "Summarize in one short sentence." },
] },
],
}, creds, model);
console.log(`[parallel turn2] status=${t2.status} bytes=${t2.raw?.length}`);
expect(t2.skip).not.toBe(true);
expect(t2.ok).toBe(true);
}, TIMEOUT_MS);
for (const model of ["deepseek-v4-flash(max)", "deepseek-v4-pro(max)"]) {
it(`(max) suffix on ${model} produces thinking output on /messages`, async () => {
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const out = await runChat({
stream: true,
max_tokens: 1024,
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
}, creds, model);
if (out.skip) return expect(true).toBe(true);
const hasThinkingDelta = /thinking_delta/.test(out.raw || "");
const hasText = /content_block_delta.*text/.test(out.raw || "") || /"text":"/.test(out.raw || "");
console.log(`[${model}] status=${out.status} thinking_delta=${hasThinkingDelta} text=${hasText} bytes=${out.raw?.length}`);
expect(out.ok).toBe(true);
expect(hasThinkingDelta, `(max) should enable thinking on ${model}`).toBe(true);
}, TIMEOUT_MS);
}
it("non-streaming claude-format request (JSON path) succeeds", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const out = await runChat({
stream: false,
max_tokens: 128,
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}, creds, model);
if (out.skip) return expect(true).toBe(true);
const isJson = out.raw?.trim()?.startsWith("{");
const hasText = /"text"/.test(out.raw || "");
console.log(`[nonstream claude] status=${out.status} json=${isJson} hasText=${hasText} raw=${out.raw?.slice?.(0, 120)}`);
expect(out.ok).toBe(true);
expect(isJson).toBe(true);
}, TIMEOUT_MS);
it("non-streaming openai-responses-format request succeeds", async () => {
const model = "deepseek-v4-flash";
const creds = await prepare(model);
if (!creds) return expect(true).toBe(true);
const result = await handleChatCore({
body: {
model: `${PROVIDER}/${model}`, stream: false, max_output_tokens: 128,
instructions: "You are concise.",
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
},
modelInfo: { provider: PROVIDER, model },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "openai-responses",
});
if (!result.success) {
const status = Number(result.status);
if (CRED_ISSUE.includes(status) || status >= 500 || status === 406) return expect(true).toBe(true);
throw new Error(`[nonstream responses] ${status}: ${result.error}`);
}
const raw = await drainSSE(result.response);
const isJson = raw?.trim()?.startsWith("{");
const hasResponsesShape = /"output"|"object":"response"/.test(raw || "");
console.log(`[nonstream responses] status=200 json=${isJson} responsesShape=${hasResponsesShape} raw=${raw?.slice?.(0, 150)}`);
expect(isJson).toBe(true);
expect(hasResponsesShape).toBe(true);
}, TIMEOUT_MS);
it("chat-only glm-5.2(max) falls back to /chat/completions for a claude-format client", async () => {
const model = "glm-5.2(max)";
const creds = await prepare("glm-5.2");
if (!creds) return expect(true).toBe(true);
const out = await runChat({
stream: true,
max_tokens: 128,
messages: [{ role: "user", content: "Reply with the single word: hi" }],
}, creds, model);
if (out.skip) { console.warn("[skip] glm-5.2 rejected/absent upstream"); return expect(true).toBe(true); }
// Guard blocks /messages; the request is translated to chat and lands on
// /chat/completions, re-encoded to the client's claude format.
const hasClaudeShape = /event:\s*\w|"type"\s*:\s*"(message_start|content_block_delta|message_stop)"/.test(out.raw || "");
console.log(`[glm fallback] status=${out.status} claudeShape=${hasClaudeShape} bytes=${out.raw?.length}`);
expect(out.ok).toBe(true);
expect(hasClaudeShape).toBe(true);
}, TIMEOUT_MS);
});
@@ -0,0 +1,196 @@
// REAL: zero-cost responses-path smoke on OpenCode Zen's free tier.
//
// After the OpenCode Go subscription lapsed, the DeepSeek go-lane evidence
// matrix could no longer be re-run. Zen's free tier keeps a responses-native
// model (muse-spark-1.3-contributor-free) reachable at no cost, so the
// responses 直通 path through 9router stays continuously testable: transport
// selection, executor free-tier fingerprint, SSE lifecycle and input replay.
// DeepSeek-specific semantics (thinking pass-back) still require the go lane
// or the official API and stay in the opencode-go / deepseek real suites.
//
// The free tier only accepts requests carrying the client fingerprint
// (opencode UA + session header + stream:true + bash/glob/grep/read quartet);
// the opencode-zen executor injects all of it, which is part of what this
// smoke verifies.
//
// OPENCODE_ZEN_KEY=oc_sk... RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-zen-free-responses.real.test.js
//
// (OPENCODE_ZEN_KEY overrides credential lookup; otherwise an opencode-zen
// connection must be configured in 9router.)
import { describe, it, expect } from "vitest";
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
const RUN_REAL = process.env.RUN_REAL === "1";
const ENV_KEY = process.env.OPENCODE_ZEN_KEY || "";
const PROVIDER = "opencode-zen";
const MODEL = "muse-spark-1.3-contributor-free";
const TIMEOUT_MS = 120000;
async function prepare() {
if (ENV_KEY) return { accessToken: ENV_KEY };
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
if (!creds || creds.allRateLimited) return null;
return checkAndRefreshToken(PROVIDER, creds);
}
async function runResponses(body, creds) {
const result = await handleChatCore({
body: { ...body, model: `${PROVIDER}/${MODEL}` },
modelInfo: { provider: PROVIDER, model: MODEL },
credentials: creds,
connectionId: creds.connectionId,
sourceFormatOverride: "openai-responses",
});
if (!result.success) {
return {
ok: false,
status: Number(result.status) || "n/a",
raw: String(result.error || ""),
};
}
return { ok: true, status: 200, raw: await drainSSE(result.response) };
}
async function drainSSE(response) {
if (!response?.body) return "";
const reader = response.body.getReader();
const decoder = new TextDecoder();
let out = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
out += decoder.decode(value, { stream: true });
}
return out;
}
function sseEvents(raw) {
const events = [];
for (const chunk of raw.split("\n\n")) {
const line = chunk.split("\n").find((l) => l.startsWith("data: "));
if (!line) continue;
try {
events.push(JSON.parse(line.slice(6)));
} catch {
/* skip malformed */
}
}
return events;
}
function userInput(text) {
return {
type: "message",
role: "user",
content: [{ type: "input_text", text }],
};
}
const WEATHER_TOOL = {
type: "function",
name: "get_weather",
description: "Get weather for a city",
parameters: {
type: "object",
properties: { city: { type: "string" } },
required: ["city"],
},
};
describe.skipIf(!RUN_REAL)(`REAL zen free responses 直通 (${MODEL})`, () => {
it(
"streams a full Responses SSE lifecycle through the executor fingerprint",
async () => {
const creds = await prepare();
if (!creds) return;
const res = await runResponses(
{
input: [userInput("Say OK and nothing else.")],
// 256 headroom: the default reasoning effort burns most of a small
// cap and the stream ends response.incomplete, not completed.
max_output_tokens: 256,
stream: true,
},
creds,
);
expect(res.ok).toBe(true);
const types = sseEvents(res.raw).map((e) => e.type);
expect(types).toContain("response.created");
expect(types).toContain("response.completed");
},
TIMEOUT_MS,
);
it(
"accepts custom tools alongside the fingerprint quartet",
async () => {
const creds = await prepare();
if (!creds) return;
const res = await runResponses(
{
input: [userInput("What is the weather in Paris? Use the tool.")],
max_output_tokens: 256,
stream: true,
tools: [WEATHER_TOOL],
},
creds,
);
expect(res.ok).toBe(true);
expect(sseEvents(res.raw).map((e) => e.type)).toContain(
"response.completed",
);
},
TIMEOUT_MS,
);
it(
"replays turn-1 output items (incl. reasoning) as input for turn 2",
async () => {
const creds = await prepare();
if (!creds) return;
const turn1 = await runResponses(
{
input: [
userInput("Think briefly, then reply with the single word OK."),
],
max_output_tokens: 256,
stream: true,
},
creds,
);
expect(turn1.ok).toBe(true);
const completed = sseEvents(turn1.raw).find(
(e) => e.type === "response.completed",
);
const output = completed?.response?.output;
expect(Array.isArray(output)).toBe(true);
expect(output.length).toBeGreaterThan(0);
// Replay every output item verbatim (reasoning items included) — the
// input-array pass-back path responses clients depend on.
const turn2 = await runResponses(
{
input: [
userInput("Think briefly, then reply with the single word OK."),
...output,
userInput("Now reply with the single word DONE."),
],
max_output_tokens: 256,
stream: true,
},
creds,
);
expect(turn2.ok).toBe(true);
expect(sseEvents(turn2.raw).map((e) => e.type)).toContain(
"response.completed",
);
},
TIMEOUT_MS,
);
});
@@ -0,0 +1,175 @@
// Offline routing matrix for opencode-go models.
//
// Drives the REAL handleChatCore guard + targetFormat resolution (open-sse/handlers/chatCore.js:86-94)
// end-to-end; only the executor's HTTP response is mocked. The assertion target is
// credentials.runtimeTransport — the exact field DefaultExecutor.buildUrl/buildHeaders read
// (open-sse/executors/default.js:106,150) to pick the endpoint and auth scheme — so a wrong
// guard decision shows up as the wrong baseUrl here, same as it would on the wire.
//
// Cells:
// - deepseek × {openai, claude, openai-responses} × {bare, (max)} — the endpoint matrix
// under dispute in #3278/#3332. Bare and suffixed cells must resolve identically.
// - glm/kimi (chat-only) + (max) — regression cells: with the thinking suffix, the guard
// is bypassed on master (suffix isn't stripped before the registry lookup) and these get
// routed to /messages, which the upstream does not serve for them.
// - minimax + (max) + claude — suffix must NOT block a genuinely declared format.
import { describe, it, expect, vi, beforeEach } from "vitest";
const { executeMock } = vi.hoisted(() => ({
executeMock: vi.fn(),
}));
vi.mock("../../open-sse/executors/index.js", () => ({
getExecutor: () => ({
noAuth: true,
execute: executeMock,
}),
}));
vi.mock("../../open-sse/utils/requestLogger.js", () => ({
createRequestLogger: async () => ({
logClientRawRequest: vi.fn(),
logRawRequest: vi.fn(),
logTargetRequest: vi.fn(),
logProviderResponse: vi.fn(),
logConvertedResponse: vi.fn(),
logError: vi.fn(),
}),
}));
vi.mock("../../open-sse/utils/stream.js", () => ({
COLORS: { red: "", reset: "" },
createPassthroughStreamWithLogger: vi.fn(() => new TransformStream()),
}));
vi.mock("uuid", () => ({
v4: () => "00000000-0000-4000-8000-000000000000",
}));
vi.mock("@/lib/usageDb.js", () => ({
trackPendingRequest: vi.fn(),
appendRequestLog: vi.fn(async () => {}),
saveRequestDetail: vi.fn(async () => {}),
saveRequestUsage: vi.fn(async () => {}),
}));
// image.js imports Agent from "undici" (not installed in some dev envs); the
// prefetch path is irrelevant to routing assertions.
vi.mock("../../open-sse/translator/concerns/image.js", () => ({
encodeDataUri: (mimeType, base64) => `data:${mimeType};base64,${base64}`,
parseDataUri: (url) => {
const m = /^data:([^;]+);base64,(.*)$/.exec(url);
return m ? { mimeType: m[1], base64: m[2] } : null;
},
fetchImageAsBase64: async () => null,
}));
const { handleChatCore } = await import("../../open-sse/handlers/chatCore.js");
const BASE = "https://opencode.ai/zen/go/v1";
const ENDPOINTS = {
openai: `${BASE}/chat/completions`,
claude: `${BASE}/messages`,
"openai-responses": `${BASE}/responses`,
};
// Minimal non-stream provider JSON per target format — the mocked executor's response.
const RESPONSE_BY_FORMAT = {
claude: {
id: "msg_1", type: "message", role: "assistant", model: "test",
content: [{ type: "text", text: "ok" }],
stop_reason: "end_turn", stop_sequence: null,
usage: { input_tokens: 1, output_tokens: 1 },
},
openai: {
id: "chatcmpl-1", object: "chat.completion", model: "test",
choices: [{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }],
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
},
"openai-responses": {
id: "resp_1", object: "response", created_at: 0, status: "completed", model: "test",
output: [{
type: "message", id: "msg_1", role: "assistant", status: "completed",
content: [{ type: "output_text", text: "ok", annotations: [] }],
}],
},
};
async function route(model, sourceFormat) {
executeMock.mockResolvedValueOnce({
response: new Response(JSON.stringify(RESPONSE_BY_FORMAT[sourceFormat === "openai" ? "openai" : sourceFormat] || RESPONSE_BY_FORMAT.openai), {
status: 200,
headers: { "content-type": "application/json" },
}),
url: ENDPOINTS[sourceFormat] || ENDPOINTS.openai,
headers: {},
transformedBody: null,
});
const credentials = { apiKey: "test-key", providerSpecificData: {} };
const result = await handleChatCore({
body: {
model: `opencode-go/${model}`,
stream: false,
max_tokens: 16,
messages: [{ role: "user", content: "hi" }],
},
modelInfo: { provider: "opencode-go", model },
credentials,
connectionId: "ocg-route-test",
sourceFormatOverride: sourceFormat,
log: { debug: vi.fn(), info: vi.fn(), warn: vi.fn() },
});
const { credentials: creds } = executeMock.mock.calls.at(-1)[0];
return { result, runtimeTransport: creds.runtimeTransport ?? null };
}
describe("opencode-go DeepSeek endpoint matrix (via real handleChatCore)", () => {
beforeEach(() => {
vi.clearAllMocks();
});
for (const model of ["deepseek-v4-flash", "deepseek-v4-pro"]) {
for (const suffix of ["", "(max)"]) {
const id = model + suffix;
for (const [fmt, expectedUrl] of Object.entries(ENDPOINTS)) {
it(`routes ${id} + ${fmt}-format client to ${expectedUrl}`, async () => {
const { result, runtimeTransport } = await route(id, fmt);
expect(result.success).toBe(true);
expect(runtimeTransport?.baseUrl).toBe(expectedUrl);
});
}
}
}
});
describe("opencode-go thinking-suffix guard (regression)", () => {
beforeEach(() => {
vi.clearAllMocks();
});
it("does NOT route chat-only glm-5.2(max) to /messages on a claude-format request", async () => {
const { result, runtimeTransport } = await route("glm-5.2(max)", "claude");
expect(result.success).toBe(true);
expect(runtimeTransport).toBeNull(); // guard must block; falls back to chat/completions
});
it("does NOT route chat-only kimi-k2.6(max) to /responses on a responses-format request", async () => {
const { result, runtimeTransport } = await route("kimi-k2.6(max)", "openai-responses");
expect(result.success).toBe(true);
expect(runtimeTransport).toBeNull();
});
it("still routes minimax-m3(max) + claude-format client to /messages", async () => {
const { result, runtimeTransport } = await route("minimax-m3(max)", "claude");
expect(result.success).toBe(true);
expect(runtimeTransport?.baseUrl).toBe(ENDPOINTS.claude);
});
it("does NOT route minimax-m3(max) (no responses support) to /responses", async () => {
const { result, runtimeTransport } = await route("minimax-m3(max)", "openai-responses");
expect(result.success).toBe(true);
expect(runtimeTransport).toBeNull();
});
});
+100
View File
@@ -0,0 +1,100 @@
import { describe, it, expect } from "vitest";
import { dedupeTools } from "../../open-sse/utils/toolDeduper.js";
const BASH = (name = "Bash", desc = "Run a shell command") => ({
name,
description: desc,
input_schema: { type: "object", properties: { command: { type: "string" } }, required: ["command"] },
});
const FUNC_SHAPE = (name = "Bash") => ({
type: "function",
function: { name, description: "Run a shell command", parameters: { type: "object", properties: { command: { type: "string" } } } },
});
const MCP_EXA = { name: "mcp__exa__web_search_exa", description: "search" };
describe("toolDeduper — MCP-equivalent built-in rules (existing behavior)", () => {
it("claude client + Exa MCP → drops built-in WebSearch/WebFetch", () => {
const { tools, stripped } = dedupeTools(
[MCP_EXA, { name: "WebSearch", description: "web" }, { name: "WebFetch", description: "web" }, BASH()],
{ clientTool: "claude" }
);
expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]);
expect(stripped.sort()).toEqual(["WebFetch", "WebSearch"]);
});
it("non-claude client → MCP built-in rules do NOT run (behavior preserved)", () => {
const { tools, stripped } = dedupeTools(
[MCP_EXA, { name: "WebSearch", description: "web" }],
{ clientTool: "codex" }
);
expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "WebSearch"]);
expect(stripped).toEqual([]);
});
it("legacy call without opts still applies MCP rules for claude callers (back-compat shape)", () => {
// chatCore always passes opts now, but old direct callers keep prior behavior:
// without clientTool, MCP rules stay dormant (they were claude-gated anyway).
const { tools, stripped } = dedupeTools([MCP_EXA, { name: "WebSearch", description: "web" }]);
expect(tools).toHaveLength(2);
expect(stripped).toEqual([]);
});
});
describe("toolDeduper — DeepSeek same-name dedup (new)", () => {
it("deepseek model + duplicate tool names → keeps first definition", () => {
const first = BASH();
const dup = BASH("Bash", "duplicate description");
const { tools, stripped } = dedupeTools([first, dup], { model: "deepseek-v4-flash" });
expect(tools).toEqual([first]); // first wins, including its description
expect(stripped).toEqual(["Bash"]);
});
it("deepseek model + (max) thinking suffix → still dedups (suffix stripped before match)", () => {
const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "dup")], { model: "deepseek-v4-flash(max)" });
expect(tools).toHaveLength(1);
expect(stripped).toEqual(["Bash"]);
});
it("deepseek + 3 same-name tools → keeps first, drops both duplicates", () => {
const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "d1"), BASH("Bash", "d2")], { model: "deepseek-v4-pro" });
expect(tools).toHaveLength(1);
expect(stripped).toEqual(["Bash", "Bash"]);
});
it("deepseek + OpenAI function-shape tools → dedups by function.name", () => {
const { tools } = dedupeTools([FUNC_SHAPE("Bash"), FUNC_SHAPE("Bash")], { model: "deepseek-v4-flash" });
expect(tools).toHaveLength(1);
expect(tools[0].function.name).toBe("Bash");
});
it("non-DeepSeek model + duplicate tool names → untouched (GLM/MiniMax/Kimi accept them)", () => {
const tools = [BASH(), BASH("Bash", "dup")];
const { tools: out, stripped } = dedupeTools(tools, { model: "glm-5.2" });
expect(out).toBe(tools);
expect(stripped).toEqual([]);
});
it("no model declared → same-name dedup does NOT run (safe default)", () => {
const tools = [BASH(), BASH("Bash", "dup")];
const { tools: out, stripped } = dedupeTools(tools, {});
expect(out).toBe(tools);
expect(stripped).toEqual([]);
});
it("deepseek + distinct names → nothing stripped", () => {
const { tools, stripped } = dedupeTools([BASH("Bash"), BASH("ReadFile")], { model: "deepseek-v4-flash" });
expect(tools).toHaveLength(2);
expect(stripped).toEqual([]);
});
it("claude client + deepseek + MCP trigger → both rules apply (union stripped)", () => {
const { tools, stripped } = dedupeTools(
[MCP_EXA, { name: "WebSearch", description: "web" }, BASH(), BASH("Bash", "dup")],
{ clientTool: "claude", model: "deepseek-v4-flash" }
);
expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]);
expect(stripped.sort()).toEqual(["Bash", "WebSearch"]);
});
});