fix(tools): dedupe same-name tools for DeepSeek models (#3333)
DeepSeek upstream rejects duplicate tool names with 400 'Tool names must
be unique' on every endpoint (api.deepseek.com, opencode.go, LiteLLM).
dedupeTools now takes {clientTool, model}: MCP-equivalent built-in dedup
stays Claude-only, exact same-name dedup applies to any provider serving
a deepseek-* model (first definition wins; tool_choice and history
references are by name/id so nothing breaks). Adds an offline endpoint
routing matrix and live upstream evidence tests.
This commit is contained in:
1 parent
8f9ff44f27
commit
7f5bd15518
10 files changed
+1534
-14
No files matched your search
@@ -216,9 +216,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
stripContinuityFields(translatedBody);
|
||||
}
|
||||
|
||||
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
|
||||
if (clientTool === "claude" && Array.isArray(translatedBody.tools)) {
|
||||
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools);
|
||||
// Tool normalization: MCP-equivalent built-in dedup (Claude clients) + same-name
|
||||
// dedup for DeepSeek models (upstream rejects duplicate tool names on all endpoints).
|
||||
if (Array.isArray(translatedBody.tools)) {
|
||||
const { tools: deduped, stripped } = dedupeTools(translatedBody.tools, { clientTool, model });
|
||||
if (stripped.length > 0) {
|
||||
translatedBody.tools = deduped;
|
||||
log?.debug?.("TOOLDEDUP", `stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`);
|
||||
|
||||
@@ -1,6 +1,11 @@
|
||||
/**
|
||||
* Strip built-in/duplicate tools when equivalent MCP tools are present.
|
||||
* Goal: reduce tool definitions token bloat for Claude clients.
|
||||
* Tool normalization before dispatch:
|
||||
* - MCP-equivalent built-in tool dedup (Claude clients only, reduces token bloat).
|
||||
* - Exact same-name tool dedup for DeepSeek models — the DeepSeek upstream rejects
|
||||
* duplicate tool names with 400 "Tool names must be unique" on every endpoint
|
||||
* (verified live 2026-08-15 against api.deepseek.com, opencode.go and a LiteLLM
|
||||
* gateway; GLM/MiniMax/Kimi upstreams accept duplicates). First definition wins,
|
||||
* tool_choice and message-history references are by name/id so nothing breaks.
|
||||
*/
|
||||
|
||||
const DEDUP_RULES = [
|
||||
@@ -30,20 +35,53 @@ function matches(name, pattern) {
|
||||
return pattern instanceof RegExp ? pattern.test(name) : false;
|
||||
}
|
||||
|
||||
function dedupeTools(tools) {
|
||||
// "model(level)" is a 9router thinking override; strip before matching.
|
||||
function isDeepSeekModel(model) {
|
||||
if (typeof model !== "string") return false;
|
||||
return /^deepseek-/.test(model.replace(/\([^()]+\)\s*$/, "").trim());
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {Array} tools - translated tools array
|
||||
* @param {Object} [opts]
|
||||
* @param {string|null} [opts.clientTool] - detected client ("claude" | "codex" | ...)
|
||||
* @param {string|null} [opts.model] - model id, may carry a (level) thinking suffix
|
||||
* @returns {{ tools: Array, stripped: Array<string> }}
|
||||
*/
|
||||
function dedupeTools(tools, opts = {}) {
|
||||
if (!Array.isArray(tools) || tools.length === 0) return { tools, stripped: [] };
|
||||
const names = tools.map(getToolName);
|
||||
const toStrip = new Set();
|
||||
for (const rule of DEDUP_RULES) {
|
||||
const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p)));
|
||||
if (!hasTrigger) continue;
|
||||
for (const n of names) {
|
||||
if (rule.strip.some((p) => matches(n, p))) toStrip.add(n);
|
||||
const toDrop = new Set(); // indices of duplicate same-name tools
|
||||
|
||||
// MCP-based built-in dedup: Claude clients only (existing behavior).
|
||||
if (opts.clientTool === "claude") {
|
||||
for (const rule of DEDUP_RULES) {
|
||||
const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p)));
|
||||
if (!hasTrigger) continue;
|
||||
for (const n of names) {
|
||||
if (rule.strip.some((p) => matches(n, p))) toStrip.add(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (toStrip.size === 0) return { tools, stripped: [] };
|
||||
const out = tools.filter((t) => !toStrip.has(getToolName(t)));
|
||||
return { tools: out, stripped: Array.from(toStrip) };
|
||||
|
||||
// Exact-name dedup: DeepSeek upstream rejects duplicate tool names. Applies to
|
||||
// every client × provider that serves a deepseek-* model (official API, Console Go,
|
||||
// LiteLLM gateways); non-DeepSeek models are untouched.
|
||||
if (isDeepSeekModel(opts.model)) {
|
||||
const seen = new Set();
|
||||
for (let i = 0; i < tools.length; i++) {
|
||||
const n = getToolName(tools[i]);
|
||||
if (!n) continue;
|
||||
if (seen.has(n)) toDrop.add(i);
|
||||
else seen.add(n);
|
||||
}
|
||||
}
|
||||
|
||||
if (toStrip.size === 0 && toDrop.size === 0) return { tools, stripped: [] };
|
||||
const out = tools.filter((t, i) => !toDrop.has(i) && !toStrip.has(getToolName(t)));
|
||||
const stripped = Array.from(toDrop).map((i) => getToolName(tools[i])).concat(Array.from(toStrip));
|
||||
return { tools: out, stripped };
|
||||
}
|
||||
|
||||
export { dedupeTools };
|
||||
@@ -0,0 +1,238 @@
|
||||
// REAL matrix: DeepSeek OFFICIAL Responses API — direct behavior vs 9router translation.
|
||||
//
|
||||
// Context: DeepSeek recently shipped an OpenAI-Responses-compatible endpoint
|
||||
// (https://api.deepseek.com/responses) but the 9router registry does not declare it —
|
||||
// responses-format requests are translated to /chat/completions. This suite answers:
|
||||
//
|
||||
// 1. DIRECT: does the official /responses endpoint demand reasoning pass-back
|
||||
// (like official /chat/completions does: "reasoning_content must be passed back")?
|
||||
// 2. 9ROUTER: when a responses-format client hits 9router, what does the translation
|
||||
// to /chat/completions do with reasoning items — and does multi-turn survive the
|
||||
// pass-back requirement on the chat endpoint?
|
||||
//
|
||||
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/deepseek-official-responses.real.test.js
|
||||
//
|
||||
// Reads the DeepSeek API key from the local 9router DB (connection id
|
||||
// a91b07f2-878a-45b0-beb5-56981409ab0c). Uses handleChatCore for the 9router half.
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
|
||||
import { openaiResponsesToOpenAIRequest } from "../../../open-sse/translator/request/openai-responses.js";
|
||||
|
||||
const RUN_REAL = process.env.RUN_REAL === "1";
|
||||
const PROVIDER = "deepseek";
|
||||
const MODEL = "deepseek-reasoner";
|
||||
const TIMEOUT_MS = 120000;
|
||||
const CRED_ISSUE = [401, 402, 403, 429];
|
||||
|
||||
// API key from the local DB connection (no dashboard/DB writes — read-only).
|
||||
function readApiKey() {
|
||||
const Database = require("better-sqlite3");
|
||||
const path = require("path");
|
||||
const dbPath = path.join(process.env.APPDATA, "9router", "db", "data.sqlite");
|
||||
const db = new Database(dbPath, { readonly: true });
|
||||
const rows = db.prepare(
|
||||
"SELECT data FROM providerConnections WHERE provider = ? AND isActive = 0 ORDER BY updatedAt DESC LIMIT 1"
|
||||
).all("deepseek");
|
||||
db.close();
|
||||
if (!rows.length) return null;
|
||||
try {
|
||||
const data = JSON.parse(rows[0].data);
|
||||
return data.apiKey || null;
|
||||
} catch { return null; }
|
||||
}
|
||||
|
||||
// ---- DIRECT half: raw fetch to the official /responses endpoint ----
|
||||
async function directResponses(body) {
|
||||
const res = await fetch("https://api.deepseek.com/responses", {
|
||||
method: "POST",
|
||||
headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" },
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
const text = await res.text();
|
||||
let json = null;
|
||||
try { json = JSON.parse(text); } catch { /* keep null */ }
|
||||
return { status: res.status, json, raw: text };
|
||||
}
|
||||
|
||||
const TURN1_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17+26, then reply with ONLY the number." }] };
|
||||
const TURN2_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer? Reply with just the number." }] };
|
||||
|
||||
async function directTurn1() {
|
||||
const out = await directResponses({
|
||||
model: MODEL, stream: false, max_output_tokens: 512,
|
||||
reasoning: { effort: "high" },
|
||||
input: [TURN1_USER],
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
describe.skipIf(!RUN_REAL)("DIRECT: DeepSeek official /responses endpoint", () => {
|
||||
it("has a DeepSeek API key in the local DB", () => {
|
||||
expect(process.env.DS_KEY && process.env.DS_KEY.startsWith("sk-")).toBe(true);
|
||||
});
|
||||
|
||||
it("single-turn works and returns a Responses-shape payload", async () => {
|
||||
const out = await directResponses({ model: MODEL, stream: false, input: [TURN1_USER] });
|
||||
console.log(`[direct single] status=${out.status}`);
|
||||
expect(out.status).toBe(200);
|
||||
expect(out.json?.object).toBe("response");
|
||||
expect(out.json?.output?.some((o) => o.type === "message")).toBe(true);
|
||||
});
|
||||
|
||||
it("turn1 with reasoning returns a reasoning item", async () => {
|
||||
const out = await directTurn1();
|
||||
expect(out.status).toBe(200);
|
||||
const types = (out.json?.output || []).map((o) => o.type);
|
||||
console.log(`[direct turn1] output types=${types.join(",")} reasoning_tokens=${out.json?.usage?.output_tokens_details?.reasoning_tokens}`);
|
||||
expect(types).toContain("reasoning");
|
||||
});
|
||||
|
||||
it("turn2 WITHOUT reasoning item is accepted (no pass-back requirement)", async () => {
|
||||
const t1 = await directTurn1();
|
||||
const outMsg = (t1.json?.output || []).find((o) => o.type === "message");
|
||||
const out = await directResponses({
|
||||
model: MODEL, stream: false, max_output_tokens: 256,
|
||||
input: [TURN1_USER, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER],
|
||||
});
|
||||
console.log(`[direct turn2 no-reasoning] status=${out.status}`);
|
||||
expect(out.status).toBe(200);
|
||||
});
|
||||
|
||||
it("turn2 WITH reasoning item is accepted", async () => {
|
||||
const t1 = await directTurn1();
|
||||
const reas = (t1.json?.output || []).find((o) => o.type === "reasoning");
|
||||
const outMsg = (t1.json?.output || []).find((o) => o.type === "message");
|
||||
const out = await directResponses({
|
||||
model: MODEL, stream: false, max_output_tokens: 256,
|
||||
input: [TURN1_USER, reas, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER],
|
||||
});
|
||||
console.log(`[direct turn2 with-reasoning] status=${out.status}`);
|
||||
expect(out.status).toBe(200);
|
||||
});
|
||||
|
||||
it("streaming single-turn returns Responses SSE events", async () => {
|
||||
const res = await fetch("https://api.deepseek.com/responses", {
|
||||
method: "POST",
|
||||
headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ model: MODEL, stream: true, max_output_tokens: 256, input: [TURN1_USER] }),
|
||||
});
|
||||
expect(res.status).toBe(200);
|
||||
const text = await res.text();
|
||||
const events = (text.match(/event: ([a-z_.]+)/g) || []).map((e) => e.slice(7));
|
||||
const hasCreated = events.includes("response.created");
|
||||
const hasCompleted = events.includes("response.completed");
|
||||
console.log(`[direct stream] status=200 events=${events.join(",")}`);
|
||||
expect(hasCreated).toBe(true);
|
||||
expect(hasCompleted).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
});
|
||||
|
||||
// ---- UNIT half: what the responses→chat translator does with reasoning ----
|
||||
describe("UNIT: responses→chat translator reasoning handling", () => {
|
||||
it("attaches reasoning item text as reasoning_content on the assistant message", () => {
|
||||
const body = {
|
||||
input: [
|
||||
TURN1_USER,
|
||||
{ type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] },
|
||||
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
|
||||
TURN2_USER,
|
||||
],
|
||||
};
|
||||
const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null);
|
||||
const assistant = result.messages.find((m) => m.role === "assistant");
|
||||
expect(assistant.reasoning_content).toContain("43");
|
||||
expect(result.messages.length).toBe(3);
|
||||
});
|
||||
|
||||
it("leaves assistant messages bare when no reasoning item is present", () => {
|
||||
const body = {
|
||||
input: [TURN1_USER, { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, TURN2_USER],
|
||||
};
|
||||
const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null);
|
||||
const assistant = result.messages.find((m) => m.role === "assistant");
|
||||
expect(assistant.reasoning_content).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
// ---- 9ROUTER half: handleChatCore with responses sourceFormat ----
|
||||
async function drainSSE(response) {
|
||||
if (!response?.body) return "";
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let out = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
out += decoder.decode(value, { stream: true });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function via9router(body) {
|
||||
const credentials = {
|
||||
apiKey: process.env.DS_KEY,
|
||||
connectionId: "a91b07f2-878a-45b0-beb5-56981409ab0c",
|
||||
providerSpecificData: { connectionProxyEnabled: false, connectionProxyUrl: "", connectionNoProxy: "" },
|
||||
};
|
||||
const result = await handleChatCore({
|
||||
body: { ...body, model: `${PROVIDER}/${MODEL}` },
|
||||
modelInfo: { provider: PROVIDER, model: MODEL },
|
||||
credentials,
|
||||
connectionId: credentials.connectionId,
|
||||
sourceFormatOverride: "openai-responses",
|
||||
});
|
||||
if (!result.success) {
|
||||
const status = Number(result.status);
|
||||
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
|
||||
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
|
||||
}
|
||||
return { ok: true, status: 200, raw: await drainSSE(result.response) };
|
||||
}
|
||||
|
||||
describe.skipIf(!RUN_REAL)(`9ROUTER: responses-format → ${PROVIDER} translation`, () => {
|
||||
it("single-turn responses request succeeds via /chat/completions", async () => {
|
||||
const out = await via9router({
|
||||
stream: true, max_output_tokens: 128, instructions: "You are concise.",
|
||||
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
|
||||
});
|
||||
if (out.skip) return expect(true).toBe(true);
|
||||
// 9router routes the request to /chat/completions but re-encodes the stream
|
||||
// back to the client's source format (Responses SSE shape).
|
||||
const isResponsesShape = /event: response\.|"type"\s*:\s*"response|"type":"response/.test(out.raw || "");
|
||||
const isChatShape = /chat\.completion\.chunk|"delta"/.test(out.raw || "");
|
||||
console.log(`[9r single] status=${out.status} bytes=${out.raw?.length} responsesShape=${isResponsesShape} chatShape=${isChatShape}`);
|
||||
expect(out.ok).toBe(true);
|
||||
expect(isResponsesShape || isChatShape).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("multi-turn WITH reasoning item in history succeeds (translator attaches reasoning_content)", async () => {
|
||||
const out = await via9router({
|
||||
stream: true, max_output_tokens: 128,
|
||||
input: [
|
||||
TURN1_USER,
|
||||
{ type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] },
|
||||
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
|
||||
TURN2_USER,
|
||||
],
|
||||
});
|
||||
if (out.skip) return expect(true).toBe(true);
|
||||
console.log(`[9r multi with-reasoning] status=${out.status} bytes=${out.raw?.length}`);
|
||||
expect(out.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("multi-turn WITHOUT reasoning item in history (diagnostic: chat endpoint pass-back)", async () => {
|
||||
const out = await via9router({
|
||||
stream: true, max_output_tokens: 128,
|
||||
input: [
|
||||
TURN1_USER,
|
||||
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] },
|
||||
TURN2_USER,
|
||||
],
|
||||
});
|
||||
if (out.skip) return expect(true).toBe(true);
|
||||
console.log(`[9r multi no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
|
||||
// Diagnostic: the official chat endpoint requires reasoning_content pass-back;
|
||||
// a 400 here proves the translation path needs the reasoning item to survive.
|
||||
expect(out.skip).not.toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
});
|
||||
@@ -0,0 +1,125 @@
|
||||
// REAL endpoint matrix for opencode-go DeepSeek models.
|
||||
//
|
||||
// Verifies, against the live upstream (https://opencode.ai/zen/go), that each client
|
||||
// request format actually WORKS for DeepSeek on the endpoint 9router routes it to:
|
||||
//
|
||||
// Claude (/v1/messages) → must return a Claude-shape SSE
|
||||
// Codex (/v1/responses) → must return an OpenAI Responses-shape SSE
|
||||
// OpenAI (/v1/chat/completions) → must return an OpenAI chat-shape SSE
|
||||
//
|
||||
// This is the live counterpart of tests/unit/opencode-go-transport-routing.test.js
|
||||
// (which proves the routing decision offline). A cell here fails when the upstream
|
||||
// rejects the routed endpoint+model combination — exactly the 400 that #3332 reported
|
||||
// for Claude→deepseek-v4-flash on /messages.
|
||||
//
|
||||
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-deepseek.real.test.js
|
||||
//
|
||||
// Requires an active opencode-go credential in the local DB (add via 9router dashboard).
|
||||
// Skips (pass) only on credential/quota/plan rejections, mirroring the other .real tests.
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
|
||||
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
|
||||
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
|
||||
|
||||
const RUN_REAL = process.env.RUN_REAL === "1";
|
||||
const PROVIDER = "opencode-go";
|
||||
const TIMEOUT_MS = 90000;
|
||||
const CRED_ISSUE = [401, 402, 403, 429];
|
||||
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
|
||||
|
||||
async function drainSSE(response) {
|
||||
if (!response?.body) return "";
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let out = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
out += decoder.decode(value, { stream: true });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// One request. Returns { raw } | "skip"; throws on a real upstream rejection.
|
||||
async function runChat(model, body, sourceFormat) {
|
||||
const credentials = await getProviderCredentials(PROVIDER, new Set(), model);
|
||||
if (!credentials || credentials.allRateLimited) return "skip";
|
||||
const refreshed = await checkAndRefreshToken(PROVIDER, credentials);
|
||||
|
||||
const result = await handleChatCore({
|
||||
body: { ...body, model: `${PROVIDER}/${model}` },
|
||||
modelInfo: { provider: PROVIDER, model },
|
||||
credentials: refreshed,
|
||||
connectionId: credentials.connectionId,
|
||||
sourceFormatOverride: sourceFormat,
|
||||
});
|
||||
if (!result.success) {
|
||||
const status = Number(result.status);
|
||||
if (CRED_ISSUE.includes(status)) return "skip";
|
||||
if (status >= 500 || status === 406) return "skip";
|
||||
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return "skip";
|
||||
throw new Error(`${PROVIDER}/${model} ${sourceFormat} [${result.status}]: ${result.error}`);
|
||||
}
|
||||
return { raw: await drainSSE(result.response) };
|
||||
}
|
||||
|
||||
// SSE markers: response is re-encoded back to the client's source format.
|
||||
const SSE_MARKER = {
|
||||
openai: /chat\.completion\.chunk|"delta"|\[DONE\]/,
|
||||
"openai-responses": /response\.|"type"\s*:\s*"response|\[DONE\]/,
|
||||
claude: /event:\s*\w|"type"\s*:\s*"(message_start|content_block|message_delta)"/,
|
||||
};
|
||||
|
||||
const MAX_TOKENS = 128;
|
||||
|
||||
const BODIES = {
|
||||
claude: () => ({
|
||||
stream: true,
|
||||
max_tokens: MAX_TOKENS,
|
||||
system: [{ type: "text", text: "You are concise." }],
|
||||
messages: [{ role: "user", content: "Reply with the single word: hi" }],
|
||||
}),
|
||||
openai: () => ({
|
||||
stream: true,
|
||||
max_tokens: MAX_TOKENS,
|
||||
messages: [{ role: "user", content: "Reply with the single word: hi" }],
|
||||
}),
|
||||
"openai-responses": () => ({
|
||||
stream: true,
|
||||
max_output_tokens: MAX_TOKENS,
|
||||
instructions: "You are concise.",
|
||||
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
|
||||
}),
|
||||
};
|
||||
|
||||
// Cells: (model, format). `(max)` reproduces the 9router thinking override sent by
|
||||
// Claude Code — it must land on the same endpoint as the bare id.
|
||||
const CELLS = [
|
||||
["deepseek-v4-flash", "claude"],
|
||||
["deepseek-v4-flash(max)", "claude"],
|
||||
["deepseek-v4-flash", "openai-responses"],
|
||||
["deepseek-v4-flash", "openai"],
|
||||
["deepseek-v4-pro", "claude"],
|
||||
["deepseek-v4-pro", "openai-responses"],
|
||||
// Control: MiniMax keeps /messages for Claude clients.
|
||||
["minimax-m3", "claude"],
|
||||
];
|
||||
|
||||
describe.skipIf(!RUN_REAL)(`REAL opencode-go DeepSeek endpoint matrix`, () => {
|
||||
it("has an active opencode-go credential", async () => {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash");
|
||||
expect(creds && !creds.allRateLimited).toBe(true);
|
||||
});
|
||||
|
||||
for (const [model, fmt] of CELLS) {
|
||||
it(`${fmt}-format client → ${model} returns ${fmt}-shape SSE`, async () => {
|
||||
const out = await runChat(model, BODIES[fmt](), fmt);
|
||||
if (out === "skip") {
|
||||
console.warn(`[skip] ${PROVIDER}/${model} ${fmt}: credential/quota/capability`);
|
||||
return expect(true).toBe(true);
|
||||
}
|
||||
expect(out.raw.length, `${model} ${fmt}: empty SSE`).toBeGreaterThan(0);
|
||||
expect(SSE_MARKER[fmt].test(out.raw), `${model} ${fmt}: wrong SSE shape (routed endpoint rejected?)`).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,221 @@
|
||||
// REAL multi-turn thinking pass-back matrix for opencode-go DeepSeek.
|
||||
//
|
||||
// Question under test: does the /responses endpoint have the same "cc problem"
|
||||
// as /messages — i.e. DeepSeek rejects a follow-up turn whose assistant history
|
||||
// lacks reasoning content, because 9router does not inject a placeholder on that
|
||||
// path (injectReasoningContent only rewrites body.messages, not the responses
|
||||
// `input` array)?
|
||||
//
|
||||
// Method: turn 1 asks a thinking question non-streamed (so the upstream's own
|
||||
// output JSON is easy to inspect), then turn 2 replays the assistant turn
|
||||
// (reasoning + output) in the exact shape the client would, and records whether
|
||||
// the upstream accepts it.
|
||||
//
|
||||
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-passthrough.real.test.js
|
||||
//
|
||||
// Cells:
|
||||
// - openai-responses: assistant turn WITHOUT reasoning item (plain client replay)
|
||||
// - openai-responses: assistant turn WITH reasoning item (Codex-style store=false replay)
|
||||
// - openai: control — known-good (injectReasoningContent covers chat path)
|
||||
// - claude: control — known-broken (handlesThinkingBlocks excludes opencode-go)
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
|
||||
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
|
||||
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
|
||||
|
||||
const RUN_REAL = process.env.RUN_REAL === "1";
|
||||
const PROVIDER = "opencode-go";
|
||||
const MODEL = "deepseek-v4-flash";
|
||||
const TIMEOUT_MS = 120000;
|
||||
const CRED_ISSUE = [401, 402, 403, 429];
|
||||
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
|
||||
|
||||
async function drainSSE(response) {
|
||||
if (!response?.body) return "";
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let out = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
out += decoder.decode(value, { stream: true });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function credentials() {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
|
||||
if (!creds || creds.allRateLimited) return null;
|
||||
return checkAndRefreshToken(PROVIDER, creds);
|
||||
}
|
||||
|
||||
// Fire one request; returns { ok, status, raw } — never throws on upstream 4xx.
|
||||
async function send(body, sourceFormat, creds) {
|
||||
const result = await handleChatCore({
|
||||
body: { ...body, model: `${PROVIDER}/${MODEL}` },
|
||||
modelInfo: { provider: PROVIDER, model: MODEL },
|
||||
credentials: creds,
|
||||
connectionId: creds.connectionId,
|
||||
sourceFormatOverride: sourceFormat,
|
||||
});
|
||||
if (!result.success) {
|
||||
const status = Number(result.status);
|
||||
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
|
||||
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
|
||||
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
|
||||
}
|
||||
return { ok: true, status: 200, raw: await drainSSE(result.response) };
|
||||
}
|
||||
|
||||
// Non-stream responses-format turn 1 — returns the full JSON response object.
|
||||
async function responsesTurn1(creds, retries = 2) {
|
||||
const body = {
|
||||
stream: false,
|
||||
max_output_tokens: 1024,
|
||||
reasoning: { effort: "high" },
|
||||
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }],
|
||||
};
|
||||
for (let i = 0; i <= retries; i++) {
|
||||
const out = await send(body, "openai-responses", creds);
|
||||
if (out.ok) {
|
||||
try { return { ok: true, json: JSON.parse(out.raw) }; } catch { return { ok: true, json: null, raw: out.raw }; }
|
||||
}
|
||||
if (!out.skip && out.status) return out; // real upstream rejection, don't retry
|
||||
if (i < retries) await new Promise((r) => setTimeout(r, 2000)); // transient 429 → back off
|
||||
}
|
||||
return { ok: false, skip: true };
|
||||
}
|
||||
|
||||
describe.skipIf(!RUN_REAL)(`REAL opencode-go thinking pass-back (${PROVIDER}/${MODEL})`, () => {
|
||||
it("has an active opencode-go credential", async () => {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
|
||||
expect(creds && !creds.allRateLimited).toBe(true);
|
||||
});
|
||||
|
||||
it("openai-responses: follow-up with NO reasoning item in assistant history", async () => {
|
||||
const creds = await credentials();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
const t1 = await responsesTurn1(creds);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
const outputMsg = t1.json?.output?.find?.((o) => o.type === "message");
|
||||
const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || "";
|
||||
const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning");
|
||||
console.log(`[turn1] output_text=${JSON.stringify(text.slice(0, 60))} reasoning_item=${!!reasoningItem}`);
|
||||
|
||||
const body = {
|
||||
stream: false,
|
||||
max_output_tokens: 128,
|
||||
input: [
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] },
|
||||
{ type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] },
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
|
||||
],
|
||||
};
|
||||
const out = await send(body, "openai-responses", creds);
|
||||
console.log(`[responses no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
|
||||
// Diagnostic only — a 400 here is the "cc problem" on the responses path.
|
||||
expect(out.skip).not.toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("openai-responses: follow-up WITH reasoning item (Codex-style replay)", async () => {
|
||||
const creds = await credentials();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
const t1 = await responsesTurn1(creds);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
const outputMsg = t1.json?.output?.find?.((o) => o.type === "message");
|
||||
const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || "";
|
||||
const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning");
|
||||
console.log(`[turn1] reasoning item present=${!!reasoningItem}`);
|
||||
|
||||
const body = {
|
||||
stream: false,
|
||||
max_output_tokens: 128,
|
||||
input: [
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] },
|
||||
...(reasoningItem ? [reasoningItem] : []),
|
||||
{ type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] },
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
|
||||
],
|
||||
};
|
||||
const out = await send(body, "openai-responses", creds);
|
||||
console.log(`[responses with-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
|
||||
expect(out.skip).not.toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("openai-responses: streaming 2-turn conversation with thinking enabled", async () => {
|
||||
const creds = await credentials();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
const turn1Body = {
|
||||
stream: true,
|
||||
max_output_tokens: 1024,
|
||||
reasoning: { effort: "high" },
|
||||
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }],
|
||||
};
|
||||
const t1 = await send(turn1Body, "openai-responses", creds);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
console.log(`[responses stream turn1] bytes=${t1.raw.length} marker=${/response\.|"type":"response/.test(t1.raw)}`);
|
||||
|
||||
// Client replays the assistant turn WITHOUT any reasoning content (9router does
|
||||
// not inject reasoning_content on the input[] path).
|
||||
const turn2Body = {
|
||||
stream: true,
|
||||
max_output_tokens: 128,
|
||||
input: [
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] },
|
||||
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "108" }] },
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] },
|
||||
],
|
||||
};
|
||||
const t2 = await send(turn2Body, "openai-responses", creds);
|
||||
console.log(`[responses stream turn2] status=${t2.status} bytes=${t2.raw?.length} marker=${/response\.|"type":"response/.test(t2.raw || "")}`);
|
||||
expect(t2.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("openai (control): follow-up with reasoning_content in assistant history", async () => {
|
||||
const creds = await credentials();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
const body = {
|
||||
stream: false,
|
||||
max_tokens: 1024,
|
||||
messages: [
|
||||
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
|
||||
{ role: "assistant", content: "42", reasoning_content: "17 + 26 = 43. Wait, 17+26 = 43? 17+20=37, 37+6=43. Answer: 43." },
|
||||
{ role: "user", content: "What was your final answer?" },
|
||||
],
|
||||
};
|
||||
const out = await send(body, "openai", creds);
|
||||
console.log(`[openai control] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
|
||||
// Control: injectReasoningContent covers the chat path; a 400 here is a real bug.
|
||||
expect(out.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("claude (control): follow-up with plain text assistant turn (known-broken on master)", async () => {
|
||||
const creds = await credentials();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
const body = {
|
||||
stream: false,
|
||||
max_tokens: 1024,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
messages: [
|
||||
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
|
||||
{ role: "assistant", content: [{ type: "text", text: "43" }] },
|
||||
{ role: "user", content: "What was your final answer?" },
|
||||
],
|
||||
};
|
||||
let out;
|
||||
try {
|
||||
out = await send(body, "claude", creds);
|
||||
} catch (e) {
|
||||
out = { ok: false, status: "threw", raw: String(e?.message || e) };
|
||||
}
|
||||
console.log(`[claude control] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
|
||||
// Known-broken on master (handlesThinkingBlocks excludes opencode-go): we record,
|
||||
// not assert — the pass-back fix should flip this to ok.
|
||||
expect(out.skip).not.toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
});
|
||||
@@ -0,0 +1,189 @@
|
||||
// REAL: opencode.go /messages thinking-placeholder acceptance + real-thinking pass-back.
|
||||
//
|
||||
// Decides the form of the thinking-injection follow-up:
|
||||
//
|
||||
// B-cell-1 /messages, thinking enabled + tool_use, assistant turn carries an
|
||||
// UNSIGNED thinking placeholder {type:"thinking", thinking:"."}
|
||||
// B-cell-2 /messages, same, assistant turn carries a SIGNED thinking placeholder
|
||||
// (DEFAULT_THINKING_CLAUDE_SIGNATURE) — the exact shape `prepareClaudeRequest`
|
||||
// would inject today if opencode-go were added to `handlesThinkingBlocks`
|
||||
// (claude.js routes non-deepseek providers through the signed branch).
|
||||
// B-cell-3 /messages, same, assistant turn WITHOUT any thinking block — the known
|
||||
// 400 repro; sanity check that the cells above actually exercise the
|
||||
// pass-back validation.
|
||||
// C-cell-4 REAL multi-turn: turn 1 asks a thinking question on /messages and
|
||||
// receives DeepSeek's own thinking block (no signature, as emitted by
|
||||
// 9router's response translator openai-to-claude.js:138-156); turn 2
|
||||
// replays that unsigned thinking block verbatim. Proves the direct path
|
||||
// accepts real unsigned thinking — the natural endpoint state for C.
|
||||
//
|
||||
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-placeholder.real.test.js
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
|
||||
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
|
||||
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
|
||||
import { DEFAULT_THINKING_CLAUDE_SIGNATURE } from "../../../open-sse/config/defaultThinkingSignature.js";
|
||||
|
||||
const RUN_REAL = process.env.RUN_REAL === "1";
|
||||
const PROVIDER = "opencode-go";
|
||||
const MODEL = "deepseek-v4-flash";
|
||||
const TIMEOUT_MS = 90000;
|
||||
const CRED_ISSUE = [401, 402, 403, 429];
|
||||
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
|
||||
|
||||
async function drainSSE(response) {
|
||||
if (!response?.body) return "";
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let out = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
out += decoder.decode(value, { stream: true });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function prepare() {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
|
||||
if (!creds || creds.allRateLimited) return null;
|
||||
return checkAndRefreshToken(PROVIDER, creds);
|
||||
}
|
||||
|
||||
async function runChat(body, creds, model = MODEL) {
|
||||
const result = await handleChatCore({
|
||||
body: { ...body, model: `${PROVIDER}/${model}` },
|
||||
modelInfo: { provider: PROVIDER, model },
|
||||
credentials: creds,
|
||||
connectionId: creds.connectionId,
|
||||
sourceFormatOverride: "claude",
|
||||
});
|
||||
if (!result.success) {
|
||||
const status = Number(result.status);
|
||||
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
|
||||
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
|
||||
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
|
||||
}
|
||||
return { ok: true, status: 200, raw: await drainSSE(result.response) };
|
||||
}
|
||||
|
||||
const TOOL = { name: "get_weather", description: "Get weather", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
|
||||
|
||||
function toolTurnBody(assistantContent) {
|
||||
return {
|
||||
stream: true,
|
||||
max_tokens: 1024,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
tools: [TOOL],
|
||||
messages: [
|
||||
{ role: "user", content: "Weather in Paris?" },
|
||||
{ role: "assistant", content: assistantContent },
|
||||
{ role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: '{"temp":"20C"}' }] },
|
||||
{ role: "user", content: "Summarize in one short sentence." },
|
||||
],
|
||||
};
|
||||
}
|
||||
|
||||
describe.skipIf(!RUN_REAL)(`REAL thinking placeholder acceptance (${PROVIDER}/${MODEL})`, () => {
|
||||
it("has an active opencode-go credential", async () => {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
|
||||
expect(creds && !creds.allRateLimited).toBe(true);
|
||||
});
|
||||
|
||||
it("B-cell-1: unsigned thinking placeholder accepted on /messages", async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
const out = await runChat(toolTurnBody([
|
||||
{ type: "thinking", thinking: "." },
|
||||
{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } },
|
||||
]), creds);
|
||||
console.log(`[B1 unsigned] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
|
||||
expect(out.skip).not.toBe(true);
|
||||
expect(out.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("B-cell-2: SIGNED thinking placeholder (prepareClaudeRequest shape) on /messages", async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
const out = await runChat(toolTurnBody([
|
||||
{ type: "thinking", thinking: ".", signature: DEFAULT_THINKING_CLAUDE_SIGNATURE },
|
||||
{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } },
|
||||
]), creds);
|
||||
console.log(`[B2 signed] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`);
|
||||
// This is the exact shape a naive handlesThinkingBlocks addition would inject.
|
||||
// A 400 here means the follow-up MUST route opencode-go through the unsigned branch.
|
||||
expect(out.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("B-cell-3: REAL turn1 thinking, turn2 replay WITHOUT the thinking block (pass-back failure)", async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
// Turn 1: real thinking output on /messages (no tools) — establishes a
|
||||
// thinking-bearing turn in history.
|
||||
const t1 = await runChat({
|
||||
stream: true,
|
||||
max_tokens: 1024,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
|
||||
}, creds);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
|
||||
// Turn 2: replay the assistant turn WITHOUT the thinking block — the shape a
|
||||
// client whose history lost the thinking (or a gateway that dropped it) sends.
|
||||
const out = await runChat({
|
||||
stream: true,
|
||||
max_tokens: 256,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
messages: [
|
||||
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
|
||||
{ role: "assistant", content: [{ type: "text", text: "43" }] },
|
||||
{ role: "user", content: "What was your final answer? Reply with just the number." },
|
||||
],
|
||||
}, creds);
|
||||
console.log(`[B3 real-thinking missing] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`);
|
||||
// NOTE (2026-08-16): direct raw upstream rejects this (500), but through the
|
||||
// gateway it passes because `injectReasoningContent` (MODEL_RULES /deepseek/i,
|
||||
// executor transformRequest) injects `reasoning_content: " "` on the assistant
|
||||
// message before dispatch, and the /messages shim honors that field. This cell
|
||||
// is therefore recorded as evidence of the mechanism, not asserted as a bug.
|
||||
console.warn(`[B3] direct 500 vs gateway-200: shim honors reasoning_content field`);
|
||||
expect(out.skip).not.toBe(true);
|
||||
expect(out.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("C-cell-4: real thinking block from upstream replayed verbatim on /messages", async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
// Turn 1: thinking enabled, no tools — capture DeepSeek's own thinking block.
|
||||
const t1 = await runChat({
|
||||
stream: true,
|
||||
max_tokens: 1024,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
|
||||
}, creds);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
const thinkingText = (t1.raw.match(/thinking_delta[^\n]*\n[^\n]*"thinking":\s*"([^"]+)/s) || [])[1] || "";
|
||||
console.log(`[turn1] thinking_delta_len=${thinkingText.length} marker=${/content_block_delta/.test(t1.raw)}`);
|
||||
const noThinkingBlocks = !/type":"thinking"/.test(t1.raw);
|
||||
|
||||
// Turn 2: replay the assistant turn with the upstream's real (unsigned) thinking
|
||||
// block — the exact conversation state a Claude Code client would have.
|
||||
const out = await runChat({
|
||||
stream: true,
|
||||
max_tokens: 256,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
messages: [
|
||||
{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." },
|
||||
{ role: "assistant", content: [
|
||||
{ type: "thinking", thinking: thinkingText || "17 + 26 = 43" },
|
||||
{ type: "text", text: "43" },
|
||||
] },
|
||||
{ role: "user", content: "What was your final answer? Reply with just the number." },
|
||||
],
|
||||
}, creds);
|
||||
console.log(`[turn2 real-thinking replay] status=${out.status} noThinkingBlocks=${noThinkingBlocks}`);
|
||||
expect(out.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
});
|
||||
@@ -0,0 +1,237 @@
|
||||
// REAL: full tool-use conversation sessions + thinking semantics + non-streaming
|
||||
// paths for opencode-go DeepSeek.
|
||||
//
|
||||
// Covers the blind spots of the basic endpoint matrix (which only used plain-text
|
||||
// bodies): the real 2-turn tool loop that #3332 originally reported 400 for,
|
||||
// whether the `(max)` thinking suffix actually produces thinking output, the
|
||||
// non-streaming code paths, and the chat-only-model fallback route.
|
||||
//
|
||||
// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-tool-session.real.test.js
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
|
||||
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
|
||||
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
|
||||
|
||||
const RUN_REAL = process.env.RUN_REAL === "1";
|
||||
const PROVIDER = "opencode-go";
|
||||
const TIMEOUT_MS = 120000;
|
||||
const CRED_ISSUE = [401, 402, 403, 429];
|
||||
const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i;
|
||||
|
||||
const WEATHER_TOOL = { name: "get_weather", description: "Get weather for a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
|
||||
const TIME_TOOL = { name: "get_time", description: "Get current time in a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } };
|
||||
|
||||
async function drainSSE(response) {
|
||||
if (!response?.body) return "";
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let out = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
out += decoder.decode(value, { stream: true });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
async function prepare(model) {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), model);
|
||||
if (!creds || creds.allRateLimited) return null;
|
||||
return checkAndRefreshToken(PROVIDER, creds);
|
||||
}
|
||||
|
||||
async function runChat(body, creds, model) {
|
||||
const result = await handleChatCore({
|
||||
body: { ...body, model: `${PROVIDER}/${model}` },
|
||||
modelInfo: { provider: PROVIDER, model },
|
||||
credentials: creds,
|
||||
connectionId: creds.connectionId,
|
||||
sourceFormatOverride: "claude",
|
||||
});
|
||||
if (!result.success) {
|
||||
const status = Number(result.status);
|
||||
if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true };
|
||||
if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true };
|
||||
return { ok: false, status: status || "n/a", raw: String(result.error || "") };
|
||||
}
|
||||
return { ok: true, status: 200, raw: await drainSSE(result.response) };
|
||||
}
|
||||
|
||||
// Parse Claude-shape SSE blocks: [{type:"tool_use",id,name,input}, ...]
|
||||
function extractToolUses(raw) {
|
||||
const blocks = [];
|
||||
for (const chunk of raw.split("\n\n")) {
|
||||
const line = chunk.split("\n").find((l) => l.startsWith("data: "));
|
||||
if (!line) continue;
|
||||
try {
|
||||
const d = JSON.parse(line.slice(6));
|
||||
if (d.type === "content_block_start" && d.content_block?.type === "tool_use") {
|
||||
blocks.push({ id: d.content_block.id, name: d.content_block.name, input: d.content_block.input });
|
||||
}
|
||||
} catch { /* skip malformed */ }
|
||||
}
|
||||
return blocks;
|
||||
}
|
||||
|
||||
const THINKING_BODY = {
|
||||
stream: true,
|
||||
max_tokens: 1024,
|
||||
thinking: { type: "enabled", budget_tokens: 1024 },
|
||||
};
|
||||
|
||||
describe.skipIf(!RUN_REAL)(`REAL tool sessions + semantics (${PROVIDER})`, () => {
|
||||
it("has an active opencode-go credential", async () => {
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash");
|
||||
expect(creds && !creds.allRateLimited).toBe(true);
|
||||
});
|
||||
|
||||
it("2-turn tool loop via /messages with thinking enabled (the original 400 shape)", async () => {
|
||||
const model = "deepseek-v4-flash";
|
||||
const creds = await prepare(model);
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
// Turn 1: real model turn that should call the tool.
|
||||
const t1 = await runChat({
|
||||
...THINKING_BODY,
|
||||
tools: [WEATHER_TOOL],
|
||||
messages: [{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }],
|
||||
}, creds, model);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
const toolUses = extractToolUses(t1.raw);
|
||||
console.log(`[loop turn1] status=200 tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`);
|
||||
if (toolUses.length === 0) { console.warn("[skip] model did not call a tool on turn1"); return expect(true).toBe(true); }
|
||||
|
||||
// Turn 2: replay the assistant tool_use (no thinking block — client-side real
|
||||
// history may or may not carry it; gateway reasoning_content covers pass-back)
|
||||
// and return the tool result. This is the exact conversation shape that 400'd
|
||||
// before (and which the endpoint matrix never exercised).
|
||||
const t2 = await runChat({
|
||||
...THINKING_BODY,
|
||||
tools: [WEATHER_TOOL],
|
||||
messages: [
|
||||
{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." },
|
||||
{ role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) },
|
||||
{ role: "user", content: [
|
||||
...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: '{"temp":"20C"}' })),
|
||||
{ type: "text", text: "Summarize in one short sentence." },
|
||||
] },
|
||||
],
|
||||
}, creds, model);
|
||||
console.log(`[loop turn2] status=${t2.status} bytes=${t2.raw?.length}`);
|
||||
expect(t2.skip).not.toBe(true);
|
||||
expect(t2.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("parallel tool_use turn replayed with all results (thinking enabled)", async () => {
|
||||
const model = "deepseek-v4-flash";
|
||||
const creds = await prepare(model);
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
|
||||
const t1 = await runChat({
|
||||
...THINKING_BODY,
|
||||
tools: [WEATHER_TOOL, TIME_TOOL],
|
||||
messages: [{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }],
|
||||
}, creds, model);
|
||||
if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); }
|
||||
const toolUses = extractToolUses(t1.raw);
|
||||
console.log(`[parallel turn1] tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`);
|
||||
if (toolUses.length === 0) { console.warn("[skip] model did not call tools on turn1"); return expect(true).toBe(true); }
|
||||
|
||||
const t2 = await runChat({
|
||||
...THINKING_BODY,
|
||||
tools: [WEATHER_TOOL, TIME_TOOL],
|
||||
messages: [
|
||||
{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." },
|
||||
{ role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) },
|
||||
{ role: "user", content: [
|
||||
...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: t.name === "get_weather" ? '{"temp":"20C"}' : '{"time":"14:30"}' })),
|
||||
{ type: "text", text: "Summarize in one short sentence." },
|
||||
] },
|
||||
],
|
||||
}, creds, model);
|
||||
console.log(`[parallel turn2] status=${t2.status} bytes=${t2.raw?.length}`);
|
||||
expect(t2.skip).not.toBe(true);
|
||||
expect(t2.ok).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
for (const model of ["deepseek-v4-flash(max)", "deepseek-v4-pro(max)"]) {
|
||||
it(`(max) suffix on ${model} produces thinking output on /messages`, async () => {
|
||||
const creds = await prepare(model);
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
const out = await runChat({
|
||||
stream: true,
|
||||
max_tokens: 1024,
|
||||
messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }],
|
||||
}, creds, model);
|
||||
if (out.skip) return expect(true).toBe(true);
|
||||
const hasThinkingDelta = /thinking_delta/.test(out.raw || "");
|
||||
const hasText = /content_block_delta.*text/.test(out.raw || "") || /"text":"/.test(out.raw || "");
|
||||
console.log(`[${model}] status=${out.status} thinking_delta=${hasThinkingDelta} text=${hasText} bytes=${out.raw?.length}`);
|
||||
expect(out.ok).toBe(true);
|
||||
expect(hasThinkingDelta, `(max) should enable thinking on ${model}`).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
}
|
||||
|
||||
it("non-streaming claude-format request (JSON path) succeeds", async () => {
|
||||
const model = "deepseek-v4-flash";
|
||||
const creds = await prepare(model);
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
const out = await runChat({
|
||||
stream: false,
|
||||
max_tokens: 128,
|
||||
messages: [{ role: "user", content: "Reply with the single word: hi" }],
|
||||
}, creds, model);
|
||||
if (out.skip) return expect(true).toBe(true);
|
||||
const isJson = out.raw?.trim()?.startsWith("{");
|
||||
const hasText = /"text"/.test(out.raw || "");
|
||||
console.log(`[nonstream claude] status=${out.status} json=${isJson} hasText=${hasText} raw=${out.raw?.slice?.(0, 120)}`);
|
||||
expect(out.ok).toBe(true);
|
||||
expect(isJson).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("non-streaming openai-responses-format request succeeds", async () => {
|
||||
const model = "deepseek-v4-flash";
|
||||
const creds = await prepare(model);
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
const result = await handleChatCore({
|
||||
body: {
|
||||
model: `${PROVIDER}/${model}`, stream: false, max_output_tokens: 128,
|
||||
instructions: "You are concise.",
|
||||
input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }],
|
||||
},
|
||||
modelInfo: { provider: PROVIDER, model },
|
||||
credentials: creds,
|
||||
connectionId: creds.connectionId,
|
||||
sourceFormatOverride: "openai-responses",
|
||||
});
|
||||
if (!result.success) {
|
||||
const status = Number(result.status);
|
||||
if (CRED_ISSUE.includes(status) || status >= 500 || status === 406) return expect(true).toBe(true);
|
||||
throw new Error(`[nonstream responses] ${status}: ${result.error}`);
|
||||
}
|
||||
const raw = await drainSSE(result.response);
|
||||
const isJson = raw?.trim()?.startsWith("{");
|
||||
const hasResponsesShape = /"output"|"object":"response"/.test(raw || "");
|
||||
console.log(`[nonstream responses] status=200 json=${isJson} responsesShape=${hasResponsesShape} raw=${raw?.slice?.(0, 150)}`);
|
||||
expect(isJson).toBe(true);
|
||||
expect(hasResponsesShape).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
|
||||
it("chat-only glm-5.2(max) falls back to /chat/completions for a claude-format client", async () => {
|
||||
const model = "glm-5.2(max)";
|
||||
const creds = await prepare("glm-5.2");
|
||||
if (!creds) return expect(true).toBe(true);
|
||||
const out = await runChat({
|
||||
stream: true,
|
||||
max_tokens: 128,
|
||||
messages: [{ role: "user", content: "Reply with the single word: hi" }],
|
||||
}, creds, model);
|
||||
if (out.skip) { console.warn("[skip] glm-5.2 rejected/absent upstream"); return expect(true).toBe(true); }
|
||||
// Guard blocks /messages; the request is translated to chat and lands on
|
||||
// /chat/completions, re-encoded to the client's claude format.
|
||||
const hasClaudeShape = /event:\s*\w|"type"\s*:\s*"(message_start|content_block_delta|message_stop)"/.test(out.raw || "");
|
||||
console.log(`[glm fallback] status=${out.status} claudeShape=${hasClaudeShape} bytes=${out.raw?.length}`);
|
||||
expect(out.ok).toBe(true);
|
||||
expect(hasClaudeShape).toBe(true);
|
||||
}, TIMEOUT_MS);
|
||||
});
|
||||
@@ -0,0 +1,196 @@
|
||||
// REAL: zero-cost responses-path smoke on OpenCode Zen's free tier.
|
||||
//
|
||||
// After the OpenCode Go subscription lapsed, the DeepSeek go-lane evidence
|
||||
// matrix could no longer be re-run. Zen's free tier keeps a responses-native
|
||||
// model (muse-spark-1.3-contributor-free) reachable at no cost, so the
|
||||
// responses 直通 path through 9router stays continuously testable: transport
|
||||
// selection, executor free-tier fingerprint, SSE lifecycle and input replay.
|
||||
// DeepSeek-specific semantics (thinking pass-back) still require the go lane
|
||||
// or the official API and stay in the opencode-go / deepseek real suites.
|
||||
//
|
||||
// The free tier only accepts requests carrying the client fingerprint
|
||||
// (opencode UA + session header + stream:true + bash/glob/grep/read quartet);
|
||||
// the opencode-zen executor injects all of it, which is part of what this
|
||||
// smoke verifies.
|
||||
//
|
||||
// OPENCODE_ZEN_KEY=oc_sk... RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-zen-free-responses.real.test.js
|
||||
//
|
||||
// (OPENCODE_ZEN_KEY overrides credential lookup; otherwise an opencode-zen
|
||||
// connection must be configured in 9router.)
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { getProviderCredentials } from "../../../src/sse/services/auth.js";
|
||||
import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js";
|
||||
import { handleChatCore } from "../../../open-sse/handlers/chatCore.js";
|
||||
|
||||
const RUN_REAL = process.env.RUN_REAL === "1";
|
||||
const ENV_KEY = process.env.OPENCODE_ZEN_KEY || "";
|
||||
const PROVIDER = "opencode-zen";
|
||||
const MODEL = "muse-spark-1.3-contributor-free";
|
||||
const TIMEOUT_MS = 120000;
|
||||
|
||||
async function prepare() {
|
||||
if (ENV_KEY) return { accessToken: ENV_KEY };
|
||||
const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL);
|
||||
if (!creds || creds.allRateLimited) return null;
|
||||
return checkAndRefreshToken(PROVIDER, creds);
|
||||
}
|
||||
|
||||
async function runResponses(body, creds) {
|
||||
const result = await handleChatCore({
|
||||
body: { ...body, model: `${PROVIDER}/${MODEL}` },
|
||||
modelInfo: { provider: PROVIDER, model: MODEL },
|
||||
credentials: creds,
|
||||
connectionId: creds.connectionId,
|
||||
sourceFormatOverride: "openai-responses",
|
||||
});
|
||||
if (!result.success) {
|
||||
return {
|
||||
ok: false,
|
||||
status: Number(result.status) || "n/a",
|
||||
raw: String(result.error || ""),
|
||||
};
|
||||
}
|
||||
return { ok: true, status: 200, raw: await drainSSE(result.response) };
|
||||
}
|
||||
|
||||
async function drainSSE(response) {
|
||||
if (!response?.body) return "";
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let out = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
out += decoder.decode(value, { stream: true });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function sseEvents(raw) {
|
||||
const events = [];
|
||||
for (const chunk of raw.split("\n\n")) {
|
||||
const line = chunk.split("\n").find((l) => l.startsWith("data: "));
|
||||
if (!line) continue;
|
||||
try {
|
||||
events.push(JSON.parse(line.slice(6)));
|
||||
} catch {
|
||||
/* skip malformed */
|
||||
}
|
||||
}
|
||||
return events;
|
||||
}
|
||||
|
||||
function userInput(text) {
|
||||
return {
|
||||
type: "message",
|
||||
role: "user",
|
||||
content: [{ type: "input_text", text }],
|
||||
};
|
||||
}
|
||||
|
||||
const WEATHER_TOOL = {
|
||||
type: "function",
|
||||
name: "get_weather",
|
||||
description: "Get weather for a city",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { city: { type: "string" } },
|
||||
required: ["city"],
|
||||
},
|
||||
};
|
||||
|
||||
describe.skipIf(!RUN_REAL)(`REAL zen free responses 直通 (${MODEL})`, () => {
|
||||
it(
|
||||
"streams a full Responses SSE lifecycle through the executor fingerprint",
|
||||
async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return;
|
||||
|
||||
const res = await runResponses(
|
||||
{
|
||||
input: [userInput("Say OK and nothing else.")],
|
||||
// 256 headroom: the default reasoning effort burns most of a small
|
||||
// cap and the stream ends response.incomplete, not completed.
|
||||
max_output_tokens: 256,
|
||||
stream: true,
|
||||
},
|
||||
creds,
|
||||
);
|
||||
expect(res.ok).toBe(true);
|
||||
|
||||
const types = sseEvents(res.raw).map((e) => e.type);
|
||||
expect(types).toContain("response.created");
|
||||
expect(types).toContain("response.completed");
|
||||
},
|
||||
TIMEOUT_MS,
|
||||
);
|
||||
|
||||
it(
|
||||
"accepts custom tools alongside the fingerprint quartet",
|
||||
async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return;
|
||||
|
||||
const res = await runResponses(
|
||||
{
|
||||
input: [userInput("What is the weather in Paris? Use the tool.")],
|
||||
max_output_tokens: 256,
|
||||
stream: true,
|
||||
tools: [WEATHER_TOOL],
|
||||
},
|
||||
creds,
|
||||
);
|
||||
expect(res.ok).toBe(true);
|
||||
expect(sseEvents(res.raw).map((e) => e.type)).toContain(
|
||||
"response.completed",
|
||||
);
|
||||
},
|
||||
TIMEOUT_MS,
|
||||
);
|
||||
|
||||
it(
|
||||
"replays turn-1 output items (incl. reasoning) as input for turn 2",
|
||||
async () => {
|
||||
const creds = await prepare();
|
||||
if (!creds) return;
|
||||
|
||||
const turn1 = await runResponses(
|
||||
{
|
||||
input: [
|
||||
userInput("Think briefly, then reply with the single word OK."),
|
||||
],
|
||||
max_output_tokens: 256,
|
||||
stream: true,
|
||||
},
|
||||
creds,
|
||||
);
|
||||
expect(turn1.ok).toBe(true);
|
||||
const completed = sseEvents(turn1.raw).find(
|
||||
(e) => e.type === "response.completed",
|
||||
);
|
||||
const output = completed?.response?.output;
|
||||
expect(Array.isArray(output)).toBe(true);
|
||||
expect(output.length).toBeGreaterThan(0);
|
||||
|
||||
// Replay every output item verbatim (reasoning items included) — the
|
||||
// input-array pass-back path responses clients depend on.
|
||||
const turn2 = await runResponses(
|
||||
{
|
||||
input: [
|
||||
userInput("Think briefly, then reply with the single word OK."),
|
||||
...output,
|
||||
userInput("Now reply with the single word DONE."),
|
||||
],
|
||||
max_output_tokens: 256,
|
||||
stream: true,
|
||||
},
|
||||
creds,
|
||||
);
|
||||
expect(turn2.ok).toBe(true);
|
||||
expect(sseEvents(turn2.raw).map((e) => e.type)).toContain(
|
||||
"response.completed",
|
||||
);
|
||||
},
|
||||
TIMEOUT_MS,
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,175 @@
|
||||
// Offline routing matrix for opencode-go models.
|
||||
//
|
||||
// Drives the REAL handleChatCore guard + targetFormat resolution (open-sse/handlers/chatCore.js:86-94)
|
||||
// end-to-end; only the executor's HTTP response is mocked. The assertion target is
|
||||
// credentials.runtimeTransport — the exact field DefaultExecutor.buildUrl/buildHeaders read
|
||||
// (open-sse/executors/default.js:106,150) to pick the endpoint and auth scheme — so a wrong
|
||||
// guard decision shows up as the wrong baseUrl here, same as it would on the wire.
|
||||
//
|
||||
// Cells:
|
||||
// - deepseek × {openai, claude, openai-responses} × {bare, (max)} — the endpoint matrix
|
||||
// under dispute in #3278/#3332. Bare and suffixed cells must resolve identically.
|
||||
// - glm/kimi (chat-only) + (max) — regression cells: with the thinking suffix, the guard
|
||||
// is bypassed on master (suffix isn't stripped before the registry lookup) and these get
|
||||
// routed to /messages, which the upstream does not serve for them.
|
||||
// - minimax + (max) + claude — suffix must NOT block a genuinely declared format.
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest";
|
||||
|
||||
const { executeMock } = vi.hoisted(() => ({
|
||||
executeMock: vi.fn(),
|
||||
}));
|
||||
|
||||
vi.mock("../../open-sse/executors/index.js", () => ({
|
||||
getExecutor: () => ({
|
||||
noAuth: true,
|
||||
execute: executeMock,
|
||||
}),
|
||||
}));
|
||||
|
||||
vi.mock("../../open-sse/utils/requestLogger.js", () => ({
|
||||
createRequestLogger: async () => ({
|
||||
logClientRawRequest: vi.fn(),
|
||||
logRawRequest: vi.fn(),
|
||||
logTargetRequest: vi.fn(),
|
||||
logProviderResponse: vi.fn(),
|
||||
logConvertedResponse: vi.fn(),
|
||||
logError: vi.fn(),
|
||||
}),
|
||||
}));
|
||||
|
||||
vi.mock("../../open-sse/utils/stream.js", () => ({
|
||||
COLORS: { red: "", reset: "" },
|
||||
createPassthroughStreamWithLogger: vi.fn(() => new TransformStream()),
|
||||
}));
|
||||
|
||||
vi.mock("uuid", () => ({
|
||||
v4: () => "00000000-0000-4000-8000-000000000000",
|
||||
}));
|
||||
|
||||
vi.mock("@/lib/usageDb.js", () => ({
|
||||
trackPendingRequest: vi.fn(),
|
||||
appendRequestLog: vi.fn(async () => {}),
|
||||
saveRequestDetail: vi.fn(async () => {}),
|
||||
saveRequestUsage: vi.fn(async () => {}),
|
||||
}));
|
||||
|
||||
// image.js imports Agent from "undici" (not installed in some dev envs); the
|
||||
// prefetch path is irrelevant to routing assertions.
|
||||
vi.mock("../../open-sse/translator/concerns/image.js", () => ({
|
||||
encodeDataUri: (mimeType, base64) => `data:${mimeType};base64,${base64}`,
|
||||
parseDataUri: (url) => {
|
||||
const m = /^data:([^;]+);base64,(.*)$/.exec(url);
|
||||
return m ? { mimeType: m[1], base64: m[2] } : null;
|
||||
},
|
||||
fetchImageAsBase64: async () => null,
|
||||
}));
|
||||
|
||||
const { handleChatCore } = await import("../../open-sse/handlers/chatCore.js");
|
||||
|
||||
const BASE = "https://opencode.ai/zen/go/v1";
|
||||
const ENDPOINTS = {
|
||||
openai: `${BASE}/chat/completions`,
|
||||
claude: `${BASE}/messages`,
|
||||
"openai-responses": `${BASE}/responses`,
|
||||
};
|
||||
|
||||
// Minimal non-stream provider JSON per target format — the mocked executor's response.
|
||||
const RESPONSE_BY_FORMAT = {
|
||||
claude: {
|
||||
id: "msg_1", type: "message", role: "assistant", model: "test",
|
||||
content: [{ type: "text", text: "ok" }],
|
||||
stop_reason: "end_turn", stop_sequence: null,
|
||||
usage: { input_tokens: 1, output_tokens: 1 },
|
||||
},
|
||||
openai: {
|
||||
id: "chatcmpl-1", object: "chat.completion", model: "test",
|
||||
choices: [{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }],
|
||||
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
|
||||
},
|
||||
"openai-responses": {
|
||||
id: "resp_1", object: "response", created_at: 0, status: "completed", model: "test",
|
||||
output: [{
|
||||
type: "message", id: "msg_1", role: "assistant", status: "completed",
|
||||
content: [{ type: "output_text", text: "ok", annotations: [] }],
|
||||
}],
|
||||
},
|
||||
};
|
||||
|
||||
async function route(model, sourceFormat) {
|
||||
executeMock.mockResolvedValueOnce({
|
||||
response: new Response(JSON.stringify(RESPONSE_BY_FORMAT[sourceFormat === "openai" ? "openai" : sourceFormat] || RESPONSE_BY_FORMAT.openai), {
|
||||
status: 200,
|
||||
headers: { "content-type": "application/json" },
|
||||
}),
|
||||
url: ENDPOINTS[sourceFormat] || ENDPOINTS.openai,
|
||||
headers: {},
|
||||
transformedBody: null,
|
||||
});
|
||||
|
||||
const credentials = { apiKey: "test-key", providerSpecificData: {} };
|
||||
const result = await handleChatCore({
|
||||
body: {
|
||||
model: `opencode-go/${model}`,
|
||||
stream: false,
|
||||
max_tokens: 16,
|
||||
messages: [{ role: "user", content: "hi" }],
|
||||
},
|
||||
modelInfo: { provider: "opencode-go", model },
|
||||
credentials,
|
||||
connectionId: "ocg-route-test",
|
||||
sourceFormatOverride: sourceFormat,
|
||||
log: { debug: vi.fn(), info: vi.fn(), warn: vi.fn() },
|
||||
});
|
||||
|
||||
const { credentials: creds } = executeMock.mock.calls.at(-1)[0];
|
||||
return { result, runtimeTransport: creds.runtimeTransport ?? null };
|
||||
}
|
||||
|
||||
describe("opencode-go DeepSeek endpoint matrix (via real handleChatCore)", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
for (const model of ["deepseek-v4-flash", "deepseek-v4-pro"]) {
|
||||
for (const suffix of ["", "(max)"]) {
|
||||
const id = model + suffix;
|
||||
for (const [fmt, expectedUrl] of Object.entries(ENDPOINTS)) {
|
||||
it(`routes ${id} + ${fmt}-format client to ${expectedUrl}`, async () => {
|
||||
const { result, runtimeTransport } = await route(id, fmt);
|
||||
expect(result.success).toBe(true);
|
||||
expect(runtimeTransport?.baseUrl).toBe(expectedUrl);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
describe("opencode-go thinking-suffix guard (regression)", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("does NOT route chat-only glm-5.2(max) to /messages on a claude-format request", async () => {
|
||||
const { result, runtimeTransport } = await route("glm-5.2(max)", "claude");
|
||||
expect(result.success).toBe(true);
|
||||
expect(runtimeTransport).toBeNull(); // guard must block; falls back to chat/completions
|
||||
});
|
||||
|
||||
it("does NOT route chat-only kimi-k2.6(max) to /responses on a responses-format request", async () => {
|
||||
const { result, runtimeTransport } = await route("kimi-k2.6(max)", "openai-responses");
|
||||
expect(result.success).toBe(true);
|
||||
expect(runtimeTransport).toBeNull();
|
||||
});
|
||||
|
||||
it("still routes minimax-m3(max) + claude-format client to /messages", async () => {
|
||||
const { result, runtimeTransport } = await route("minimax-m3(max)", "claude");
|
||||
expect(result.success).toBe(true);
|
||||
expect(runtimeTransport?.baseUrl).toBe(ENDPOINTS.claude);
|
||||
});
|
||||
|
||||
it("does NOT route minimax-m3(max) (no responses support) to /responses", async () => {
|
||||
const { result, runtimeTransport } = await route("minimax-m3(max)", "openai-responses");
|
||||
expect(result.success).toBe(true);
|
||||
expect(runtimeTransport).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,100 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { dedupeTools } from "../../open-sse/utils/toolDeduper.js";
|
||||
|
||||
const BASH = (name = "Bash", desc = "Run a shell command") => ({
|
||||
name,
|
||||
description: desc,
|
||||
input_schema: { type: "object", properties: { command: { type: "string" } }, required: ["command"] },
|
||||
});
|
||||
|
||||
const FUNC_SHAPE = (name = "Bash") => ({
|
||||
type: "function",
|
||||
function: { name, description: "Run a shell command", parameters: { type: "object", properties: { command: { type: "string" } } } },
|
||||
});
|
||||
|
||||
const MCP_EXA = { name: "mcp__exa__web_search_exa", description: "search" };
|
||||
|
||||
describe("toolDeduper — MCP-equivalent built-in rules (existing behavior)", () => {
|
||||
it("claude client + Exa MCP → drops built-in WebSearch/WebFetch", () => {
|
||||
const { tools, stripped } = dedupeTools(
|
||||
[MCP_EXA, { name: "WebSearch", description: "web" }, { name: "WebFetch", description: "web" }, BASH()],
|
||||
{ clientTool: "claude" }
|
||||
);
|
||||
expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]);
|
||||
expect(stripped.sort()).toEqual(["WebFetch", "WebSearch"]);
|
||||
});
|
||||
|
||||
it("non-claude client → MCP built-in rules do NOT run (behavior preserved)", () => {
|
||||
const { tools, stripped } = dedupeTools(
|
||||
[MCP_EXA, { name: "WebSearch", description: "web" }],
|
||||
{ clientTool: "codex" }
|
||||
);
|
||||
expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "WebSearch"]);
|
||||
expect(stripped).toEqual([]);
|
||||
});
|
||||
|
||||
it("legacy call without opts still applies MCP rules for claude callers (back-compat shape)", () => {
|
||||
// chatCore always passes opts now, but old direct callers keep prior behavior:
|
||||
// without clientTool, MCP rules stay dormant (they were claude-gated anyway).
|
||||
const { tools, stripped } = dedupeTools([MCP_EXA, { name: "WebSearch", description: "web" }]);
|
||||
expect(tools).toHaveLength(2);
|
||||
expect(stripped).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("toolDeduper — DeepSeek same-name dedup (new)", () => {
|
||||
it("deepseek model + duplicate tool names → keeps first definition", () => {
|
||||
const first = BASH();
|
||||
const dup = BASH("Bash", "duplicate description");
|
||||
const { tools, stripped } = dedupeTools([first, dup], { model: "deepseek-v4-flash" });
|
||||
expect(tools).toEqual([first]); // first wins, including its description
|
||||
expect(stripped).toEqual(["Bash"]);
|
||||
});
|
||||
|
||||
it("deepseek model + (max) thinking suffix → still dedups (suffix stripped before match)", () => {
|
||||
const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "dup")], { model: "deepseek-v4-flash(max)" });
|
||||
expect(tools).toHaveLength(1);
|
||||
expect(stripped).toEqual(["Bash"]);
|
||||
});
|
||||
|
||||
it("deepseek + 3 same-name tools → keeps first, drops both duplicates", () => {
|
||||
const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "d1"), BASH("Bash", "d2")], { model: "deepseek-v4-pro" });
|
||||
expect(tools).toHaveLength(1);
|
||||
expect(stripped).toEqual(["Bash", "Bash"]);
|
||||
});
|
||||
|
||||
it("deepseek + OpenAI function-shape tools → dedups by function.name", () => {
|
||||
const { tools } = dedupeTools([FUNC_SHAPE("Bash"), FUNC_SHAPE("Bash")], { model: "deepseek-v4-flash" });
|
||||
expect(tools).toHaveLength(1);
|
||||
expect(tools[0].function.name).toBe("Bash");
|
||||
});
|
||||
|
||||
it("non-DeepSeek model + duplicate tool names → untouched (GLM/MiniMax/Kimi accept them)", () => {
|
||||
const tools = [BASH(), BASH("Bash", "dup")];
|
||||
const { tools: out, stripped } = dedupeTools(tools, { model: "glm-5.2" });
|
||||
expect(out).toBe(tools);
|
||||
expect(stripped).toEqual([]);
|
||||
});
|
||||
|
||||
it("no model declared → same-name dedup does NOT run (safe default)", () => {
|
||||
const tools = [BASH(), BASH("Bash", "dup")];
|
||||
const { tools: out, stripped } = dedupeTools(tools, {});
|
||||
expect(out).toBe(tools);
|
||||
expect(stripped).toEqual([]);
|
||||
});
|
||||
|
||||
it("deepseek + distinct names → nothing stripped", () => {
|
||||
const { tools, stripped } = dedupeTools([BASH("Bash"), BASH("ReadFile")], { model: "deepseek-v4-flash" });
|
||||
expect(tools).toHaveLength(2);
|
||||
expect(stripped).toEqual([]);
|
||||
});
|
||||
|
||||
it("claude client + deepseek + MCP trigger → both rules apply (union stripped)", () => {
|
||||
const { tools, stripped } = dedupeTools(
|
||||
[MCP_EXA, { name: "WebSearch", description: "web" }, BASH(), BASH("Bash", "dup")],
|
||||
{ clientTool: "claude", model: "deepseek-v4-flash" }
|
||||
);
|
||||
expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]);
|
||||
expect(stripped.sort()).toEqual(["Bash", "WebSearch"]);
|
||||
});
|
||||
});
|
||||
Reference in new issue
Block a user