diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index a6230fc4..8755659d 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -216,9 +216,10 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred stripContinuityFields(translatedBody); } - // Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only). - if (clientTool === "claude" && Array.isArray(translatedBody.tools)) { - const { tools: deduped, stripped } = dedupeTools(translatedBody.tools); + // Tool normalization: MCP-equivalent built-in dedup (Claude clients) + same-name + // dedup for DeepSeek models (upstream rejects duplicate tool names on all endpoints). + if (Array.isArray(translatedBody.tools)) { + const { tools: deduped, stripped } = dedupeTools(translatedBody.tools, { clientTool, model }); if (stripped.length > 0) { translatedBody.tools = deduped; log?.debug?.("TOOLDEDUP", `stripped ${stripped.length}: ${stripped.slice(0, 3).join(", ")}${stripped.length > 3 ? "..." : ""}`); diff --git a/open-sse/utils/toolDeduper.js b/open-sse/utils/toolDeduper.js index 20c47403..ee6b5aa6 100644 --- a/open-sse/utils/toolDeduper.js +++ b/open-sse/utils/toolDeduper.js @@ -1,6 +1,11 @@ /** - * Strip built-in/duplicate tools when equivalent MCP tools are present. - * Goal: reduce tool definitions token bloat for Claude clients. + * Tool normalization before dispatch: + * - MCP-equivalent built-in tool dedup (Claude clients only, reduces token bloat). + * - Exact same-name tool dedup for DeepSeek models — the DeepSeek upstream rejects + * duplicate tool names with 400 "Tool names must be unique" on every endpoint + * (verified live 2026-08-15 against api.deepseek.com, opencode.go and a LiteLLM + * gateway; GLM/MiniMax/Kimi upstreams accept duplicates). First definition wins, + * tool_choice and message-history references are by name/id so nothing breaks. */ const DEDUP_RULES = [ @@ -30,20 +35,53 @@ function matches(name, pattern) { return pattern instanceof RegExp ? pattern.test(name) : false; } -function dedupeTools(tools) { +// "model(level)" is a 9router thinking override; strip before matching. +function isDeepSeekModel(model) { + if (typeof model !== "string") return false; + return /^deepseek-/.test(model.replace(/\([^()]+\)\s*$/, "").trim()); +} + +/** + * @param {Array} tools - translated tools array + * @param {Object} [opts] + * @param {string|null} [opts.clientTool] - detected client ("claude" | "codex" | ...) + * @param {string|null} [opts.model] - model id, may carry a (level) thinking suffix + * @returns {{ tools: Array, stripped: Array }} + */ +function dedupeTools(tools, opts = {}) { if (!Array.isArray(tools) || tools.length === 0) return { tools, stripped: [] }; const names = tools.map(getToolName); const toStrip = new Set(); - for (const rule of DEDUP_RULES) { - const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p))); - if (!hasTrigger) continue; - for (const n of names) { - if (rule.strip.some((p) => matches(n, p))) toStrip.add(n); + const toDrop = new Set(); // indices of duplicate same-name tools + + // MCP-based built-in dedup: Claude clients only (existing behavior). + if (opts.clientTool === "claude") { + for (const rule of DEDUP_RULES) { + const hasTrigger = names.some((n) => rule.triggers.some((p) => matches(n, p))); + if (!hasTrigger) continue; + for (const n of names) { + if (rule.strip.some((p) => matches(n, p))) toStrip.add(n); + } } } - if (toStrip.size === 0) return { tools, stripped: [] }; - const out = tools.filter((t) => !toStrip.has(getToolName(t))); - return { tools: out, stripped: Array.from(toStrip) }; + + // Exact-name dedup: DeepSeek upstream rejects duplicate tool names. Applies to + // every client × provider that serves a deepseek-* model (official API, Console Go, + // LiteLLM gateways); non-DeepSeek models are untouched. + if (isDeepSeekModel(opts.model)) { + const seen = new Set(); + for (let i = 0; i < tools.length; i++) { + const n = getToolName(tools[i]); + if (!n) continue; + if (seen.has(n)) toDrop.add(i); + else seen.add(n); + } + } + + if (toStrip.size === 0 && toDrop.size === 0) return { tools, stripped: [] }; + const out = tools.filter((t, i) => !toDrop.has(i) && !toStrip.has(getToolName(t))); + const stripped = Array.from(toDrop).map((i) => getToolName(tools[i])).concat(Array.from(toStrip)); + return { tools: out, stripped }; } export { dedupeTools }; diff --git a/tests/translator/real/deepseek-official-responses.real.test.js b/tests/translator/real/deepseek-official-responses.real.test.js new file mode 100644 index 00000000..32b20987 --- /dev/null +++ b/tests/translator/real/deepseek-official-responses.real.test.js @@ -0,0 +1,238 @@ +// REAL matrix: DeepSeek OFFICIAL Responses API — direct behavior vs 9router translation. +// +// Context: DeepSeek recently shipped an OpenAI-Responses-compatible endpoint +// (https://api.deepseek.com/responses) but the 9router registry does not declare it — +// responses-format requests are translated to /chat/completions. This suite answers: +// +// 1. DIRECT: does the official /responses endpoint demand reasoning pass-back +// (like official /chat/completions does: "reasoning_content must be passed back")? +// 2. 9ROUTER: when a responses-format client hits 9router, what does the translation +// to /chat/completions do with reasoning items — and does multi-turn survive the +// pass-back requirement on the chat endpoint? +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/deepseek-official-responses.real.test.js +// +// Reads the DeepSeek API key from the local 9router DB (connection id +// a91b07f2-878a-45b0-beb5-56981409ab0c). Uses handleChatCore for the 9router half. +import { describe, it, expect } from "vitest"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; +import { openaiResponsesToOpenAIRequest } from "../../../open-sse/translator/request/openai-responses.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "deepseek"; +const MODEL = "deepseek-reasoner"; +const TIMEOUT_MS = 120000; +const CRED_ISSUE = [401, 402, 403, 429]; + +// API key from the local DB connection (no dashboard/DB writes — read-only). +function readApiKey() { + const Database = require("better-sqlite3"); + const path = require("path"); + const dbPath = path.join(process.env.APPDATA, "9router", "db", "data.sqlite"); + const db = new Database(dbPath, { readonly: true }); + const rows = db.prepare( + "SELECT data FROM providerConnections WHERE provider = ? AND isActive = 0 ORDER BY updatedAt DESC LIMIT 1" + ).all("deepseek"); + db.close(); + if (!rows.length) return null; + try { + const data = JSON.parse(rows[0].data); + return data.apiKey || null; + } catch { return null; } +} + +// ---- DIRECT half: raw fetch to the official /responses endpoint ---- +async function directResponses(body) { + const res = await fetch("https://api.deepseek.com/responses", { + method: "POST", + headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" }, + body: JSON.stringify(body), + }); + const text = await res.text(); + let json = null; + try { json = JSON.parse(text); } catch { /* keep null */ } + return { status: res.status, json, raw: text }; +} + +const TURN1_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17+26, then reply with ONLY the number." }] }; +const TURN2_USER = { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer? Reply with just the number." }] }; + +async function directTurn1() { + const out = await directResponses({ + model: MODEL, stream: false, max_output_tokens: 512, + reasoning: { effort: "high" }, + input: [TURN1_USER], + }); + return out; +} + +describe.skipIf(!RUN_REAL)("DIRECT: DeepSeek official /responses endpoint", () => { + it("has a DeepSeek API key in the local DB", () => { + expect(process.env.DS_KEY && process.env.DS_KEY.startsWith("sk-")).toBe(true); + }); + + it("single-turn works and returns a Responses-shape payload", async () => { + const out = await directResponses({ model: MODEL, stream: false, input: [TURN1_USER] }); + console.log(`[direct single] status=${out.status}`); + expect(out.status).toBe(200); + expect(out.json?.object).toBe("response"); + expect(out.json?.output?.some((o) => o.type === "message")).toBe(true); + }); + + it("turn1 with reasoning returns a reasoning item", async () => { + const out = await directTurn1(); + expect(out.status).toBe(200); + const types = (out.json?.output || []).map((o) => o.type); + console.log(`[direct turn1] output types=${types.join(",")} reasoning_tokens=${out.json?.usage?.output_tokens_details?.reasoning_tokens}`); + expect(types).toContain("reasoning"); + }); + + it("turn2 WITHOUT reasoning item is accepted (no pass-back requirement)", async () => { + const t1 = await directTurn1(); + const outMsg = (t1.json?.output || []).find((o) => o.type === "message"); + const out = await directResponses({ + model: MODEL, stream: false, max_output_tokens: 256, + input: [TURN1_USER, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER], + }); + console.log(`[direct turn2 no-reasoning] status=${out.status}`); + expect(out.status).toBe(200); + }); + + it("turn2 WITH reasoning item is accepted", async () => { + const t1 = await directTurn1(); + const reas = (t1.json?.output || []).find((o) => o.type === "reasoning"); + const outMsg = (t1.json?.output || []).find((o) => o.type === "message"); + const out = await directResponses({ + model: MODEL, stream: false, max_output_tokens: 256, + input: [TURN1_USER, reas, { type: "message", role: "assistant", content: outMsg?.content }, TURN2_USER], + }); + console.log(`[direct turn2 with-reasoning] status=${out.status}`); + expect(out.status).toBe(200); + }); + + it("streaming single-turn returns Responses SSE events", async () => { + const res = await fetch("https://api.deepseek.com/responses", { + method: "POST", + headers: { "Authorization": `Bearer ${process.env.DS_KEY}`, "Content-Type": "application/json" }, + body: JSON.stringify({ model: MODEL, stream: true, max_output_tokens: 256, input: [TURN1_USER] }), + }); + expect(res.status).toBe(200); + const text = await res.text(); + const events = (text.match(/event: ([a-z_.]+)/g) || []).map((e) => e.slice(7)); + const hasCreated = events.includes("response.created"); + const hasCompleted = events.includes("response.completed"); + console.log(`[direct stream] status=200 events=${events.join(",")}`); + expect(hasCreated).toBe(true); + expect(hasCompleted).toBe(true); + }, TIMEOUT_MS); +}); + +// ---- UNIT half: what the responses→chat translator does with reasoning ---- +describe("UNIT: responses→chat translator reasoning handling", () => { + it("attaches reasoning item text as reasoning_content on the assistant message", () => { + const body = { + input: [ + TURN1_USER, + { type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, + TURN2_USER, + ], + }; + const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null); + const assistant = result.messages.find((m) => m.role === "assistant"); + expect(assistant.reasoning_content).toContain("43"); + expect(result.messages.length).toBe(3); + }); + + it("leaves assistant messages bare when no reasoning item is present", () => { + const body = { + input: [TURN1_USER, { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, TURN2_USER], + }; + const result = openaiResponsesToOpenAIRequest(MODEL, body, false, null); + const assistant = result.messages.find((m) => m.role === "assistant"); + expect(assistant.reasoning_content).toBeUndefined(); + }); +}); + +// ---- 9ROUTER half: handleChatCore with responses sourceFormat ---- +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function via9router(body) { + const credentials = { + apiKey: process.env.DS_KEY, + connectionId: "a91b07f2-878a-45b0-beb5-56981409ab0c", + providerSpecificData: { connectionProxyEnabled: false, connectionProxyUrl: "", connectionNoProxy: "" }, + }; + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${MODEL}` }, + modelInfo: { provider: PROVIDER, model: MODEL }, + credentials, + connectionId: credentials.connectionId, + sourceFormatOverride: "openai-responses", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +describe.skipIf(!RUN_REAL)(`9ROUTER: responses-format → ${PROVIDER} translation`, () => { + it("single-turn responses request succeeds via /chat/completions", async () => { + const out = await via9router({ + stream: true, max_output_tokens: 128, instructions: "You are concise.", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }], + }); + if (out.skip) return expect(true).toBe(true); + // 9router routes the request to /chat/completions but re-encodes the stream + // back to the client's source format (Responses SSE shape). + const isResponsesShape = /event: response\.|"type"\s*:\s*"response|"type":"response/.test(out.raw || ""); + const isChatShape = /chat\.completion\.chunk|"delta"/.test(out.raw || ""); + console.log(`[9r single] status=${out.status} bytes=${out.raw?.length} responsesShape=${isResponsesShape} chatShape=${isChatShape}`); + expect(out.ok).toBe(true); + expect(isResponsesShape || isChatShape).toBe(true); + }, TIMEOUT_MS); + + it("multi-turn WITH reasoning item in history succeeds (translator attaches reasoning_content)", async () => { + const out = await via9router({ + stream: true, max_output_tokens: 128, + input: [ + TURN1_USER, + { type: "reasoning", id: "rs_1", content: [{ type: "reasoning_text", text: "17 + 26 = 43" }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, + TURN2_USER, + ], + }); + if (out.skip) return expect(true).toBe(true); + console.log(`[9r multi with-reasoning] status=${out.status} bytes=${out.raw?.length}`); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("multi-turn WITHOUT reasoning item in history (diagnostic: chat endpoint pass-back)", async () => { + const out = await via9router({ + stream: true, max_output_tokens: 128, + input: [ + TURN1_USER, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "43" }] }, + TURN2_USER, + ], + }); + if (out.skip) return expect(true).toBe(true); + console.log(`[9r multi no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // Diagnostic: the official chat endpoint requires reasoning_content pass-back; + // a 400 here proves the translation path needs the reasoning item to survive. + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-go-deepseek.real.test.js b/tests/translator/real/opencode-go-deepseek.real.test.js new file mode 100644 index 00000000..06db975f --- /dev/null +++ b/tests/translator/real/opencode-go-deepseek.real.test.js @@ -0,0 +1,125 @@ +// REAL endpoint matrix for opencode-go DeepSeek models. +// +// Verifies, against the live upstream (https://opencode.ai/zen/go), that each client +// request format actually WORKS for DeepSeek on the endpoint 9router routes it to: +// +// Claude (/v1/messages) → must return a Claude-shape SSE +// Codex (/v1/responses) → must return an OpenAI Responses-shape SSE +// OpenAI (/v1/chat/completions) → must return an OpenAI chat-shape SSE +// +// This is the live counterpart of tests/unit/opencode-go-transport-routing.test.js +// (which proves the routing decision offline). A cell here fails when the upstream +// rejects the routed endpoint+model combination — exactly the 400 that #3332 reported +// for Claude→deepseek-v4-flash on /messages. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-deepseek.real.test.js +// +// Requires an active opencode-go credential in the local DB (add via 9router dashboard). +// Skips (pass) only on credential/quota/plan rejections, mirroring the other .real tests. +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const TIMEOUT_MS = 90000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +// One request. Returns { raw } | "skip"; throws on a real upstream rejection. +async function runChat(model, body, sourceFormat) { + const credentials = await getProviderCredentials(PROVIDER, new Set(), model); + if (!credentials || credentials.allRateLimited) return "skip"; + const refreshed = await checkAndRefreshToken(PROVIDER, credentials); + + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${model}` }, + modelInfo: { provider: PROVIDER, model }, + credentials: refreshed, + connectionId: credentials.connectionId, + sourceFormatOverride: sourceFormat, + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status)) return "skip"; + if (status >= 500 || status === 406) return "skip"; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return "skip"; + throw new Error(`${PROVIDER}/${model} ${sourceFormat} [${result.status}]: ${result.error}`); + } + return { raw: await drainSSE(result.response) }; +} + +// SSE markers: response is re-encoded back to the client's source format. +const SSE_MARKER = { + openai: /chat\.completion\.chunk|"delta"|\[DONE\]/, + "openai-responses": /response\.|"type"\s*:\s*"response|\[DONE\]/, + claude: /event:\s*\w|"type"\s*:\s*"(message_start|content_block|message_delta)"/, +}; + +const MAX_TOKENS = 128; + +const BODIES = { + claude: () => ({ + stream: true, + max_tokens: MAX_TOKENS, + system: [{ type: "text", text: "You are concise." }], + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }), + openai: () => ({ + stream: true, + max_tokens: MAX_TOKENS, + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }), + "openai-responses": () => ({ + stream: true, + max_output_tokens: MAX_TOKENS, + instructions: "You are concise.", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }], + }), +}; + +// Cells: (model, format). `(max)` reproduces the 9router thinking override sent by +// Claude Code — it must land on the same endpoint as the bare id. +const CELLS = [ + ["deepseek-v4-flash", "claude"], + ["deepseek-v4-flash(max)", "claude"], + ["deepseek-v4-flash", "openai-responses"], + ["deepseek-v4-flash", "openai"], + ["deepseek-v4-pro", "claude"], + ["deepseek-v4-pro", "openai-responses"], + // Control: MiniMax keeps /messages for Claude clients. + ["minimax-m3", "claude"], +]; + +describe.skipIf(!RUN_REAL)(`REAL opencode-go DeepSeek endpoint matrix`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash"); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + for (const [model, fmt] of CELLS) { + it(`${fmt}-format client → ${model} returns ${fmt}-shape SSE`, async () => { + const out = await runChat(model, BODIES[fmt](), fmt); + if (out === "skip") { + console.warn(`[skip] ${PROVIDER}/${model} ${fmt}: credential/quota/capability`); + return expect(true).toBe(true); + } + expect(out.raw.length, `${model} ${fmt}: empty SSE`).toBeGreaterThan(0); + expect(SSE_MARKER[fmt].test(out.raw), `${model} ${fmt}: wrong SSE shape (routed endpoint rejected?)`).toBe(true); + }, TIMEOUT_MS); + } +}); diff --git a/tests/translator/real/opencode-go-thinking-passthrough.real.test.js b/tests/translator/real/opencode-go-thinking-passthrough.real.test.js new file mode 100644 index 00000000..608e26a6 --- /dev/null +++ b/tests/translator/real/opencode-go-thinking-passthrough.real.test.js @@ -0,0 +1,221 @@ +// REAL multi-turn thinking pass-back matrix for opencode-go DeepSeek. +// +// Question under test: does the /responses endpoint have the same "cc problem" +// as /messages — i.e. DeepSeek rejects a follow-up turn whose assistant history +// lacks reasoning content, because 9router does not inject a placeholder on that +// path (injectReasoningContent only rewrites body.messages, not the responses +// `input` array)? +// +// Method: turn 1 asks a thinking question non-streamed (so the upstream's own +// output JSON is easy to inspect), then turn 2 replays the assistant turn +// (reasoning + output) in the exact shape the client would, and records whether +// the upstream accepts it. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-passthrough.real.test.js +// +// Cells: +// - openai-responses: assistant turn WITHOUT reasoning item (plain client replay) +// - openai-responses: assistant turn WITH reasoning item (Codex-style store=false replay) +// - openai: control — known-good (injectReasoningContent covers chat path) +// - claude: control — known-broken (handlesThinkingBlocks excludes opencode-go) +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const MODEL = "deepseek-v4-flash"; +const TIMEOUT_MS = 120000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function credentials() { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +// Fire one request; returns { ok, status, raw } — never throws on upstream 4xx. +async function send(body, sourceFormat, creds) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${MODEL}` }, + modelInfo: { provider: PROVIDER, model: MODEL }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: sourceFormat, + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +// Non-stream responses-format turn 1 — returns the full JSON response object. +async function responsesTurn1(creds, retries = 2) { + const body = { + stream: false, + max_output_tokens: 1024, + reasoning: { effort: "high" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }], + }; + for (let i = 0; i <= retries; i++) { + const out = await send(body, "openai-responses", creds); + if (out.ok) { + try { return { ok: true, json: JSON.parse(out.raw) }; } catch { return { ok: true, json: null, raw: out.raw }; } + } + if (!out.skip && out.status) return out; // real upstream rejection, don't retry + if (i < retries) await new Promise((r) => setTimeout(r, 2000)); // transient 429 → back off + } + return { ok: false, skip: true }; +} + +describe.skipIf(!RUN_REAL)(`REAL opencode-go thinking pass-back (${PROVIDER}/${MODEL})`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + it("openai-responses: follow-up with NO reasoning item in assistant history", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const t1 = await responsesTurn1(creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const outputMsg = t1.json?.output?.find?.((o) => o.type === "message"); + const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || ""; + const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning"); + console.log(`[turn1] output_text=${JSON.stringify(text.slice(0, 60))} reasoning_item=${!!reasoningItem}`); + + const body = { + stream: false, + max_output_tokens: 128, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] }, + ], + }; + const out = await send(body, "openai-responses", creds); + console.log(`[responses no-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // Diagnostic only — a 400 here is the "cc problem" on the responses path. + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); + + it("openai-responses: follow-up WITH reasoning item (Codex-style replay)", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const t1 = await responsesTurn1(creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const outputMsg = t1.json?.output?.find?.((o) => o.type === "message"); + const text = outputMsg?.content?.map?.((c) => c.text).filter(Boolean).join("") || ""; + const reasoningItem = t1.json?.output?.find?.((o) => o.type === "reasoning"); + console.log(`[turn1] reasoning item present=${!!reasoningItem}`); + + const body = { + stream: false, + max_output_tokens: 128, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 17 + 26, then reply with ONLY the number." }] }, + ...(reasoningItem ? [reasoningItem] : []), + { type: "message", role: "assistant", content: [{ type: "output_text", text: text || "42" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] }, + ], + }; + const out = await send(body, "openai-responses", creds); + console.log(`[responses with-reasoning] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); + + it("openai-responses: streaming 2-turn conversation with thinking enabled", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const turn1Body = { + stream: true, + max_output_tokens: 1024, + reasoning: { effort: "high" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }], + }; + const t1 = await send(turn1Body, "openai-responses", creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + console.log(`[responses stream turn1] bytes=${t1.raw.length} marker=${/response\.|"type":"response/.test(t1.raw)}`); + + // Client replays the assistant turn WITHOUT any reasoning content (9router does + // not inject reasoning_content on the input[] path). + const turn2Body = { + stream: true, + max_output_tokens: 128, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "Think step by step about 12 * 9, then reply with ONLY the number." }] }, + { type: "message", role: "assistant", content: [{ type: "output_text", text: "108" }] }, + { type: "message", role: "user", content: [{ type: "input_text", text: "What was your final answer?" }] }, + ], + }; + const t2 = await send(turn2Body, "openai-responses", creds); + console.log(`[responses stream turn2] status=${t2.status} bytes=${t2.raw?.length} marker=${/response\.|"type":"response/.test(t2.raw || "")}`); + expect(t2.ok).toBe(true); + }, TIMEOUT_MS); + + it("openai (control): follow-up with reasoning_content in assistant history", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const body = { + stream: false, + max_tokens: 1024, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: "42", reasoning_content: "17 + 26 = 43. Wait, 17+26 = 43? 17+20=37, 37+6=43. Answer: 43." }, + { role: "user", content: "What was your final answer?" }, + ], + }; + const out = await send(body, "openai", creds); + console.log(`[openai control] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`); + // Control: injectReasoningContent covers the chat path; a 400 here is a real bug. + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("claude (control): follow-up with plain text assistant turn (known-broken on master)", async () => { + const creds = await credentials(); + if (!creds) return expect(true).toBe(true); + + const body = { + stream: false, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: [{ type: "text", text: "43" }] }, + { role: "user", content: "What was your final answer?" }, + ], + }; + let out; + try { + out = await send(body, "claude", creds); + } catch (e) { + out = { ok: false, status: "threw", raw: String(e?.message || e) }; + } + console.log(`[claude control] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // Known-broken on master (handlesThinkingBlocks excludes opencode-go): we record, + // not assert — the pass-back fix should flip this to ok. + expect(out.skip).not.toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-go-thinking-placeholder.real.test.js b/tests/translator/real/opencode-go-thinking-placeholder.real.test.js new file mode 100644 index 00000000..aada5099 --- /dev/null +++ b/tests/translator/real/opencode-go-thinking-placeholder.real.test.js @@ -0,0 +1,189 @@ +// REAL: opencode.go /messages thinking-placeholder acceptance + real-thinking pass-back. +// +// Decides the form of the thinking-injection follow-up: +// +// B-cell-1 /messages, thinking enabled + tool_use, assistant turn carries an +// UNSIGNED thinking placeholder {type:"thinking", thinking:"."} +// B-cell-2 /messages, same, assistant turn carries a SIGNED thinking placeholder +// (DEFAULT_THINKING_CLAUDE_SIGNATURE) — the exact shape `prepareClaudeRequest` +// would inject today if opencode-go were added to `handlesThinkingBlocks` +// (claude.js routes non-deepseek providers through the signed branch). +// B-cell-3 /messages, same, assistant turn WITHOUT any thinking block — the known +// 400 repro; sanity check that the cells above actually exercise the +// pass-back validation. +// C-cell-4 REAL multi-turn: turn 1 asks a thinking question on /messages and +// receives DeepSeek's own thinking block (no signature, as emitted by +// 9router's response translator openai-to-claude.js:138-156); turn 2 +// replays that unsigned thinking block verbatim. Proves the direct path +// accepts real unsigned thinking — the natural endpoint state for C. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-thinking-placeholder.real.test.js +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; +import { DEFAULT_THINKING_CLAUDE_SIGNATURE } from "../../../open-sse/config/defaultThinkingSignature.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const MODEL = "deepseek-v4-flash"; +const TIMEOUT_MS = 90000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function prepare() { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +async function runChat(body, creds, model = MODEL) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${model}` }, + modelInfo: { provider: PROVIDER, model }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "claude", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +const TOOL = { name: "get_weather", description: "Get weather", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } }; + +function toolTurnBody(assistantContent) { + return { + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + tools: [TOOL], + messages: [ + { role: "user", content: "Weather in Paris?" }, + { role: "assistant", content: assistantContent }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: '{"temp":"20C"}' }] }, + { role: "user", content: "Summarize in one short sentence." }, + ], + }; +} + +describe.skipIf(!RUN_REAL)(`REAL thinking placeholder acceptance (${PROVIDER}/${MODEL})`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + it("B-cell-1: unsigned thinking placeholder accepted on /messages", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + const out = await runChat(toolTurnBody([ + { type: "thinking", thinking: "." }, + { type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } }, + ]), creds); + console.log(`[B1 unsigned] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`); + expect(out.skip).not.toBe(true); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("B-cell-2: SIGNED thinking placeholder (prepareClaudeRequest shape) on /messages", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + const out = await runChat(toolTurnBody([ + { type: "thinking", thinking: ".", signature: DEFAULT_THINKING_CLAUDE_SIGNATURE }, + { type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Paris" } }, + ]), creds); + console.log(`[B2 signed] status=${out.status} raw=${out.raw?.slice?.(0, 150)}`); + // This is the exact shape a naive handlesThinkingBlocks addition would inject. + // A 400 here means the follow-up MUST route opencode-go through the unsigned branch. + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("B-cell-3: REAL turn1 thinking, turn2 replay WITHOUT the thinking block (pass-back failure)", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + + // Turn 1: real thinking output on /messages (no tools) — establishes a + // thinking-bearing turn in history. + const t1 = await runChat({ + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }], + }, creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + + // Turn 2: replay the assistant turn WITHOUT the thinking block — the shape a + // client whose history lost the thinking (or a gateway that dropped it) sends. + const out = await runChat({ + stream: true, + max_tokens: 256, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: [{ type: "text", text: "43" }] }, + { role: "user", content: "What was your final answer? Reply with just the number." }, + ], + }, creds); + console.log(`[B3 real-thinking missing] status=${out.status} raw=${out.raw?.slice?.(0, 200)}`); + // NOTE (2026-08-16): direct raw upstream rejects this (500), but through the + // gateway it passes because `injectReasoningContent` (MODEL_RULES /deepseek/i, + // executor transformRequest) injects `reasoning_content: " "` on the assistant + // message before dispatch, and the /messages shim honors that field. This cell + // is therefore recorded as evidence of the mechanism, not asserted as a bug. + console.warn(`[B3] direct 500 vs gateway-200: shim honors reasoning_content field`); + expect(out.skip).not.toBe(true); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); + + it("C-cell-4: real thinking block from upstream replayed verbatim on /messages", async () => { + const creds = await prepare(); + if (!creds) return expect(true).toBe(true); + + // Turn 1: thinking enabled, no tools — capture DeepSeek's own thinking block. + const t1 = await runChat({ + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }], + }, creds); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const thinkingText = (t1.raw.match(/thinking_delta[^\n]*\n[^\n]*"thinking":\s*"([^"]+)/s) || [])[1] || ""; + console.log(`[turn1] thinking_delta_len=${thinkingText.length} marker=${/content_block_delta/.test(t1.raw)}`); + const noThinkingBlocks = !/type":"thinking"/.test(t1.raw); + + // Turn 2: replay the assistant turn with the upstream's real (unsigned) thinking + // block — the exact conversation state a Claude Code client would have. + const out = await runChat({ + stream: true, + max_tokens: 256, + thinking: { type: "enabled", budget_tokens: 1024 }, + messages: [ + { role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }, + { role: "assistant", content: [ + { type: "thinking", thinking: thinkingText || "17 + 26 = 43" }, + { type: "text", text: "43" }, + ] }, + { role: "user", content: "What was your final answer? Reply with just the number." }, + ], + }, creds); + console.log(`[turn2 real-thinking replay] status=${out.status} noThinkingBlocks=${noThinkingBlocks}`); + expect(out.ok).toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-go-tool-session.real.test.js b/tests/translator/real/opencode-go-tool-session.real.test.js new file mode 100644 index 00000000..23bb29e9 --- /dev/null +++ b/tests/translator/real/opencode-go-tool-session.real.test.js @@ -0,0 +1,237 @@ +// REAL: full tool-use conversation sessions + thinking semantics + non-streaming +// paths for opencode-go DeepSeek. +// +// Covers the blind spots of the basic endpoint matrix (which only used plain-text +// bodies): the real 2-turn tool loop that #3332 originally reported 400 for, +// whether the `(max)` thinking suffix actually produces thinking output, the +// non-streaming code paths, and the chat-only-model fallback route. +// +// RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-go-tool-session.real.test.js +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const PROVIDER = "opencode-go"; +const TIMEOUT_MS = 120000; +const CRED_ISSUE = [401, 402, 403, 429]; +const SKIP_MSG_RE = /image|multimodal|vision|modality|unsupported|not support|reasoning_effort|deprecated|temperature|subscription|valid.*plan|embedding|quota|insufficient|model not found|context length|organization policy|disallowed|allowedmodels|failed_precondition/i; + +const WEATHER_TOOL = { name: "get_weather", description: "Get weather for a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } }; +const TIME_TOOL = { name: "get_time", description: "Get current time in a city", input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] } }; + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +async function prepare(model) { + const creds = await getProviderCredentials(PROVIDER, new Set(), model); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +async function runChat(body, creds, model) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${model}` }, + modelInfo: { provider: PROVIDER, model }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "claude", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || (status >= 500) || status === 406) return { skip: true }; + if (status === 400 && SKIP_MSG_RE.test(String(result.error || ""))) return { skip: true }; + return { ok: false, status: status || "n/a", raw: String(result.error || "") }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +// Parse Claude-shape SSE blocks: [{type:"tool_use",id,name,input}, ...] +function extractToolUses(raw) { + const blocks = []; + for (const chunk of raw.split("\n\n")) { + const line = chunk.split("\n").find((l) => l.startsWith("data: ")); + if (!line) continue; + try { + const d = JSON.parse(line.slice(6)); + if (d.type === "content_block_start" && d.content_block?.type === "tool_use") { + blocks.push({ id: d.content_block.id, name: d.content_block.name, input: d.content_block.input }); + } + } catch { /* skip malformed */ } + } + return blocks; +} + +const THINKING_BODY = { + stream: true, + max_tokens: 1024, + thinking: { type: "enabled", budget_tokens: 1024 }, +}; + +describe.skipIf(!RUN_REAL)(`REAL tool sessions + semantics (${PROVIDER})`, () => { + it("has an active opencode-go credential", async () => { + const creds = await getProviderCredentials(PROVIDER, new Set(), "deepseek-v4-flash"); + expect(creds && !creds.allRateLimited).toBe(true); + }); + + it("2-turn tool loop via /messages with thinking enabled (the original 400 shape)", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + + // Turn 1: real model turn that should call the tool. + const t1 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL], + messages: [{ role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }], + }, creds, model); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const toolUses = extractToolUses(t1.raw); + console.log(`[loop turn1] status=200 tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`); + if (toolUses.length === 0) { console.warn("[skip] model did not call a tool on turn1"); return expect(true).toBe(true); } + + // Turn 2: replay the assistant tool_use (no thinking block — client-side real + // history may or may not carry it; gateway reasoning_content covers pass-back) + // and return the tool result. This is the exact conversation shape that 400'd + // before (and which the endpoint matrix never exercised). + const t2 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL], + messages: [ + { role: "user", content: "Weather in Paris? Call the get_weather tool and then stop." }, + { role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) }, + { role: "user", content: [ + ...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: '{"temp":"20C"}' })), + { type: "text", text: "Summarize in one short sentence." }, + ] }, + ], + }, creds, model); + console.log(`[loop turn2] status=${t2.status} bytes=${t2.raw?.length}`); + expect(t2.skip).not.toBe(true); + expect(t2.ok).toBe(true); + }, TIMEOUT_MS); + + it("parallel tool_use turn replayed with all results (thinking enabled)", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + + const t1 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL, TIME_TOOL], + messages: [{ role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }], + }, creds, model); + if (!t1.ok) { console.warn(`[skip] turn1 failed ${t1.status}: ${t1.raw}`); return expect(true).toBe(true); } + const toolUses = extractToolUses(t1.raw); + console.log(`[parallel turn1] tool_uses=${toolUses.length} ${toolUses.map((t) => t.name).join(",")}`); + if (toolUses.length === 0) { console.warn("[skip] model did not call tools on turn1"); return expect(true).toBe(true); } + + const t2 = await runChat({ + ...THINKING_BODY, + tools: [WEATHER_TOOL, TIME_TOOL], + messages: [ + { role: "user", content: "Call get_weather for Paris and get_time for Tokyo, both in parallel, then stop." }, + { role: "assistant", content: toolUses.map((t) => ({ type: "tool_use", id: t.id, name: t.name, input: t.input })) }, + { role: "user", content: [ + ...toolUses.map((t) => ({ type: "tool_result", tool_use_id: t.id, content: t.name === "get_weather" ? '{"temp":"20C"}' : '{"time":"14:30"}' })), + { type: "text", text: "Summarize in one short sentence." }, + ] }, + ], + }, creds, model); + console.log(`[parallel turn2] status=${t2.status} bytes=${t2.raw?.length}`); + expect(t2.skip).not.toBe(true); + expect(t2.ok).toBe(true); + }, TIMEOUT_MS); + + for (const model of ["deepseek-v4-flash(max)", "deepseek-v4-pro(max)"]) { + it(`(max) suffix on ${model} produces thinking output on /messages`, async () => { + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + const out = await runChat({ + stream: true, + max_tokens: 1024, + messages: [{ role: "user", content: "Think step by step about 17 + 26, then reply with ONLY the number." }], + }, creds, model); + if (out.skip) return expect(true).toBe(true); + const hasThinkingDelta = /thinking_delta/.test(out.raw || ""); + const hasText = /content_block_delta.*text/.test(out.raw || "") || /"text":"/.test(out.raw || ""); + console.log(`[${model}] status=${out.status} thinking_delta=${hasThinkingDelta} text=${hasText} bytes=${out.raw?.length}`); + expect(out.ok).toBe(true); + expect(hasThinkingDelta, `(max) should enable thinking on ${model}`).toBe(true); + }, TIMEOUT_MS); + } + + it("non-streaming claude-format request (JSON path) succeeds", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + const out = await runChat({ + stream: false, + max_tokens: 128, + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }, creds, model); + if (out.skip) return expect(true).toBe(true); + const isJson = out.raw?.trim()?.startsWith("{"); + const hasText = /"text"/.test(out.raw || ""); + console.log(`[nonstream claude] status=${out.status} json=${isJson} hasText=${hasText} raw=${out.raw?.slice?.(0, 120)}`); + expect(out.ok).toBe(true); + expect(isJson).toBe(true); + }, TIMEOUT_MS); + + it("non-streaming openai-responses-format request succeeds", async () => { + const model = "deepseek-v4-flash"; + const creds = await prepare(model); + if (!creds) return expect(true).toBe(true); + const result = await handleChatCore({ + body: { + model: `${PROVIDER}/${model}`, stream: false, max_output_tokens: 128, + instructions: "You are concise.", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Reply with the single word: hi" }] }], + }, + modelInfo: { provider: PROVIDER, model }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "openai-responses", + }); + if (!result.success) { + const status = Number(result.status); + if (CRED_ISSUE.includes(status) || status >= 500 || status === 406) return expect(true).toBe(true); + throw new Error(`[nonstream responses] ${status}: ${result.error}`); + } + const raw = await drainSSE(result.response); + const isJson = raw?.trim()?.startsWith("{"); + const hasResponsesShape = /"output"|"object":"response"/.test(raw || ""); + console.log(`[nonstream responses] status=200 json=${isJson} responsesShape=${hasResponsesShape} raw=${raw?.slice?.(0, 150)}`); + expect(isJson).toBe(true); + expect(hasResponsesShape).toBe(true); + }, TIMEOUT_MS); + + it("chat-only glm-5.2(max) falls back to /chat/completions for a claude-format client", async () => { + const model = "glm-5.2(max)"; + const creds = await prepare("glm-5.2"); + if (!creds) return expect(true).toBe(true); + const out = await runChat({ + stream: true, + max_tokens: 128, + messages: [{ role: "user", content: "Reply with the single word: hi" }], + }, creds, model); + if (out.skip) { console.warn("[skip] glm-5.2 rejected/absent upstream"); return expect(true).toBe(true); } + // Guard blocks /messages; the request is translated to chat and lands on + // /chat/completions, re-encoded to the client's claude format. + const hasClaudeShape = /event:\s*\w|"type"\s*:\s*"(message_start|content_block_delta|message_stop)"/.test(out.raw || ""); + console.log(`[glm fallback] status=${out.status} claudeShape=${hasClaudeShape} bytes=${out.raw?.length}`); + expect(out.ok).toBe(true); + expect(hasClaudeShape).toBe(true); + }, TIMEOUT_MS); +}); diff --git a/tests/translator/real/opencode-zen-free-responses.real.test.js b/tests/translator/real/opencode-zen-free-responses.real.test.js new file mode 100644 index 00000000..7de01d8e --- /dev/null +++ b/tests/translator/real/opencode-zen-free-responses.real.test.js @@ -0,0 +1,196 @@ +// REAL: zero-cost responses-path smoke on OpenCode Zen's free tier. +// +// After the OpenCode Go subscription lapsed, the DeepSeek go-lane evidence +// matrix could no longer be re-run. Zen's free tier keeps a responses-native +// model (muse-spark-1.3-contributor-free) reachable at no cost, so the +// responses 直通 path through 9router stays continuously testable: transport +// selection, executor free-tier fingerprint, SSE lifecycle and input replay. +// DeepSeek-specific semantics (thinking pass-back) still require the go lane +// or the official API and stay in the opencode-go / deepseek real suites. +// +// The free tier only accepts requests carrying the client fingerprint +// (opencode UA + session header + stream:true + bash/glob/grep/read quartet); +// the opencode-zen executor injects all of it, which is part of what this +// smoke verifies. +// +// OPENCODE_ZEN_KEY=oc_sk... RUN_REAL=1 npx vitest run --config tests/vitest.config.js tests/translator/real/opencode-zen-free-responses.real.test.js +// +// (OPENCODE_ZEN_KEY overrides credential lookup; otherwise an opencode-zen +// connection must be configured in 9router.) +import { describe, it, expect } from "vitest"; +import { getProviderCredentials } from "../../../src/sse/services/auth.js"; +import { checkAndRefreshToken } from "../../../src/sse/services/tokenRefresh.js"; +import { handleChatCore } from "../../../open-sse/handlers/chatCore.js"; + +const RUN_REAL = process.env.RUN_REAL === "1"; +const ENV_KEY = process.env.OPENCODE_ZEN_KEY || ""; +const PROVIDER = "opencode-zen"; +const MODEL = "muse-spark-1.3-contributor-free"; +const TIMEOUT_MS = 120000; + +async function prepare() { + if (ENV_KEY) return { accessToken: ENV_KEY }; + const creds = await getProviderCredentials(PROVIDER, new Set(), MODEL); + if (!creds || creds.allRateLimited) return null; + return checkAndRefreshToken(PROVIDER, creds); +} + +async function runResponses(body, creds) { + const result = await handleChatCore({ + body: { ...body, model: `${PROVIDER}/${MODEL}` }, + modelInfo: { provider: PROVIDER, model: MODEL }, + credentials: creds, + connectionId: creds.connectionId, + sourceFormatOverride: "openai-responses", + }); + if (!result.success) { + return { + ok: false, + status: Number(result.status) || "n/a", + raw: String(result.error || ""), + }; + } + return { ok: true, status: 200, raw: await drainSSE(result.response) }; +} + +async function drainSSE(response) { + if (!response?.body) return ""; + const reader = response.body.getReader(); + const decoder = new TextDecoder(); + let out = ""; + while (true) { + const { done, value } = await reader.read(); + if (done) break; + out += decoder.decode(value, { stream: true }); + } + return out; +} + +function sseEvents(raw) { + const events = []; + for (const chunk of raw.split("\n\n")) { + const line = chunk.split("\n").find((l) => l.startsWith("data: ")); + if (!line) continue; + try { + events.push(JSON.parse(line.slice(6))); + } catch { + /* skip malformed */ + } + } + return events; +} + +function userInput(text) { + return { + type: "message", + role: "user", + content: [{ type: "input_text", text }], + }; +} + +const WEATHER_TOOL = { + type: "function", + name: "get_weather", + description: "Get weather for a city", + parameters: { + type: "object", + properties: { city: { type: "string" } }, + required: ["city"], + }, +}; + +describe.skipIf(!RUN_REAL)(`REAL zen free responses 直通 (${MODEL})`, () => { + it( + "streams a full Responses SSE lifecycle through the executor fingerprint", + async () => { + const creds = await prepare(); + if (!creds) return; + + const res = await runResponses( + { + input: [userInput("Say OK and nothing else.")], + // 256 headroom: the default reasoning effort burns most of a small + // cap and the stream ends response.incomplete, not completed. + max_output_tokens: 256, + stream: true, + }, + creds, + ); + expect(res.ok).toBe(true); + + const types = sseEvents(res.raw).map((e) => e.type); + expect(types).toContain("response.created"); + expect(types).toContain("response.completed"); + }, + TIMEOUT_MS, + ); + + it( + "accepts custom tools alongside the fingerprint quartet", + async () => { + const creds = await prepare(); + if (!creds) return; + + const res = await runResponses( + { + input: [userInput("What is the weather in Paris? Use the tool.")], + max_output_tokens: 256, + stream: true, + tools: [WEATHER_TOOL], + }, + creds, + ); + expect(res.ok).toBe(true); + expect(sseEvents(res.raw).map((e) => e.type)).toContain( + "response.completed", + ); + }, + TIMEOUT_MS, + ); + + it( + "replays turn-1 output items (incl. reasoning) as input for turn 2", + async () => { + const creds = await prepare(); + if (!creds) return; + + const turn1 = await runResponses( + { + input: [ + userInput("Think briefly, then reply with the single word OK."), + ], + max_output_tokens: 256, + stream: true, + }, + creds, + ); + expect(turn1.ok).toBe(true); + const completed = sseEvents(turn1.raw).find( + (e) => e.type === "response.completed", + ); + const output = completed?.response?.output; + expect(Array.isArray(output)).toBe(true); + expect(output.length).toBeGreaterThan(0); + + // Replay every output item verbatim (reasoning items included) — the + // input-array pass-back path responses clients depend on. + const turn2 = await runResponses( + { + input: [ + userInput("Think briefly, then reply with the single word OK."), + ...output, + userInput("Now reply with the single word DONE."), + ], + max_output_tokens: 256, + stream: true, + }, + creds, + ); + expect(turn2.ok).toBe(true); + expect(sseEvents(turn2.raw).map((e) => e.type)).toContain( + "response.completed", + ); + }, + TIMEOUT_MS, + ); +}); diff --git a/tests/unit/opencode-go-transport-routing.test.js b/tests/unit/opencode-go-transport-routing.test.js new file mode 100644 index 00000000..6a4d47be --- /dev/null +++ b/tests/unit/opencode-go-transport-routing.test.js @@ -0,0 +1,175 @@ +// Offline routing matrix for opencode-go models. +// +// Drives the REAL handleChatCore guard + targetFormat resolution (open-sse/handlers/chatCore.js:86-94) +// end-to-end; only the executor's HTTP response is mocked. The assertion target is +// credentials.runtimeTransport — the exact field DefaultExecutor.buildUrl/buildHeaders read +// (open-sse/executors/default.js:106,150) to pick the endpoint and auth scheme — so a wrong +// guard decision shows up as the wrong baseUrl here, same as it would on the wire. +// +// Cells: +// - deepseek × {openai, claude, openai-responses} × {bare, (max)} — the endpoint matrix +// under dispute in #3278/#3332. Bare and suffixed cells must resolve identically. +// - glm/kimi (chat-only) + (max) — regression cells: with the thinking suffix, the guard +// is bypassed on master (suffix isn't stripped before the registry lookup) and these get +// routed to /messages, which the upstream does not serve for them. +// - minimax + (max) + claude — suffix must NOT block a genuinely declared format. +import { describe, it, expect, vi, beforeEach } from "vitest"; + +const { executeMock } = vi.hoisted(() => ({ + executeMock: vi.fn(), +})); + +vi.mock("../../open-sse/executors/index.js", () => ({ + getExecutor: () => ({ + noAuth: true, + execute: executeMock, + }), +})); + +vi.mock("../../open-sse/utils/requestLogger.js", () => ({ + createRequestLogger: async () => ({ + logClientRawRequest: vi.fn(), + logRawRequest: vi.fn(), + logTargetRequest: vi.fn(), + logProviderResponse: vi.fn(), + logConvertedResponse: vi.fn(), + logError: vi.fn(), + }), +})); + +vi.mock("../../open-sse/utils/stream.js", () => ({ + COLORS: { red: "", reset: "" }, + createPassthroughStreamWithLogger: vi.fn(() => new TransformStream()), +})); + +vi.mock("uuid", () => ({ + v4: () => "00000000-0000-4000-8000-000000000000", +})); + +vi.mock("@/lib/usageDb.js", () => ({ + trackPendingRequest: vi.fn(), + appendRequestLog: vi.fn(async () => {}), + saveRequestDetail: vi.fn(async () => {}), + saveRequestUsage: vi.fn(async () => {}), +})); + +// image.js imports Agent from "undici" (not installed in some dev envs); the +// prefetch path is irrelevant to routing assertions. +vi.mock("../../open-sse/translator/concerns/image.js", () => ({ + encodeDataUri: (mimeType, base64) => `data:${mimeType};base64,${base64}`, + parseDataUri: (url) => { + const m = /^data:([^;]+);base64,(.*)$/.exec(url); + return m ? { mimeType: m[1], base64: m[2] } : null; + }, + fetchImageAsBase64: async () => null, +})); + +const { handleChatCore } = await import("../../open-sse/handlers/chatCore.js"); + +const BASE = "https://opencode.ai/zen/go/v1"; +const ENDPOINTS = { + openai: `${BASE}/chat/completions`, + claude: `${BASE}/messages`, + "openai-responses": `${BASE}/responses`, +}; + +// Minimal non-stream provider JSON per target format — the mocked executor's response. +const RESPONSE_BY_FORMAT = { + claude: { + id: "msg_1", type: "message", role: "assistant", model: "test", + content: [{ type: "text", text: "ok" }], + stop_reason: "end_turn", stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 1 }, + }, + openai: { + id: "chatcmpl-1", object: "chat.completion", model: "test", + choices: [{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }, + "openai-responses": { + id: "resp_1", object: "response", created_at: 0, status: "completed", model: "test", + output: [{ + type: "message", id: "msg_1", role: "assistant", status: "completed", + content: [{ type: "output_text", text: "ok", annotations: [] }], + }], + }, +}; + +async function route(model, sourceFormat) { + executeMock.mockResolvedValueOnce({ + response: new Response(JSON.stringify(RESPONSE_BY_FORMAT[sourceFormat === "openai" ? "openai" : sourceFormat] || RESPONSE_BY_FORMAT.openai), { + status: 200, + headers: { "content-type": "application/json" }, + }), + url: ENDPOINTS[sourceFormat] || ENDPOINTS.openai, + headers: {}, + transformedBody: null, + }); + + const credentials = { apiKey: "test-key", providerSpecificData: {} }; + const result = await handleChatCore({ + body: { + model: `opencode-go/${model}`, + stream: false, + max_tokens: 16, + messages: [{ role: "user", content: "hi" }], + }, + modelInfo: { provider: "opencode-go", model }, + credentials, + connectionId: "ocg-route-test", + sourceFormatOverride: sourceFormat, + log: { debug: vi.fn(), info: vi.fn(), warn: vi.fn() }, + }); + + const { credentials: creds } = executeMock.mock.calls.at(-1)[0]; + return { result, runtimeTransport: creds.runtimeTransport ?? null }; +} + +describe("opencode-go DeepSeek endpoint matrix (via real handleChatCore)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + for (const model of ["deepseek-v4-flash", "deepseek-v4-pro"]) { + for (const suffix of ["", "(max)"]) { + const id = model + suffix; + for (const [fmt, expectedUrl] of Object.entries(ENDPOINTS)) { + it(`routes ${id} + ${fmt}-format client to ${expectedUrl}`, async () => { + const { result, runtimeTransport } = await route(id, fmt); + expect(result.success).toBe(true); + expect(runtimeTransport?.baseUrl).toBe(expectedUrl); + }); + } + } + } +}); + +describe("opencode-go thinking-suffix guard (regression)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it("does NOT route chat-only glm-5.2(max) to /messages on a claude-format request", async () => { + const { result, runtimeTransport } = await route("glm-5.2(max)", "claude"); + expect(result.success).toBe(true); + expect(runtimeTransport).toBeNull(); // guard must block; falls back to chat/completions + }); + + it("does NOT route chat-only kimi-k2.6(max) to /responses on a responses-format request", async () => { + const { result, runtimeTransport } = await route("kimi-k2.6(max)", "openai-responses"); + expect(result.success).toBe(true); + expect(runtimeTransport).toBeNull(); + }); + + it("still routes minimax-m3(max) + claude-format client to /messages", async () => { + const { result, runtimeTransport } = await route("minimax-m3(max)", "claude"); + expect(result.success).toBe(true); + expect(runtimeTransport?.baseUrl).toBe(ENDPOINTS.claude); + }); + + it("does NOT route minimax-m3(max) (no responses support) to /responses", async () => { + const { result, runtimeTransport } = await route("minimax-m3(max)", "openai-responses"); + expect(result.success).toBe(true); + expect(runtimeTransport).toBeNull(); + }); +}); diff --git a/tests/unit/tool-deduper.test.js b/tests/unit/tool-deduper.test.js new file mode 100644 index 00000000..58714b00 --- /dev/null +++ b/tests/unit/tool-deduper.test.js @@ -0,0 +1,100 @@ +import { describe, it, expect } from "vitest"; +import { dedupeTools } from "../../open-sse/utils/toolDeduper.js"; + +const BASH = (name = "Bash", desc = "Run a shell command") => ({ + name, + description: desc, + input_schema: { type: "object", properties: { command: { type: "string" } }, required: ["command"] }, +}); + +const FUNC_SHAPE = (name = "Bash") => ({ + type: "function", + function: { name, description: "Run a shell command", parameters: { type: "object", properties: { command: { type: "string" } } } }, +}); + +const MCP_EXA = { name: "mcp__exa__web_search_exa", description: "search" }; + +describe("toolDeduper — MCP-equivalent built-in rules (existing behavior)", () => { + it("claude client + Exa MCP → drops built-in WebSearch/WebFetch", () => { + const { tools, stripped } = dedupeTools( + [MCP_EXA, { name: "WebSearch", description: "web" }, { name: "WebFetch", description: "web" }, BASH()], + { clientTool: "claude" } + ); + expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]); + expect(stripped.sort()).toEqual(["WebFetch", "WebSearch"]); + }); + + it("non-claude client → MCP built-in rules do NOT run (behavior preserved)", () => { + const { tools, stripped } = dedupeTools( + [MCP_EXA, { name: "WebSearch", description: "web" }], + { clientTool: "codex" } + ); + expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "WebSearch"]); + expect(stripped).toEqual([]); + }); + + it("legacy call without opts still applies MCP rules for claude callers (back-compat shape)", () => { + // chatCore always passes opts now, but old direct callers keep prior behavior: + // without clientTool, MCP rules stay dormant (they were claude-gated anyway). + const { tools, stripped } = dedupeTools([MCP_EXA, { name: "WebSearch", description: "web" }]); + expect(tools).toHaveLength(2); + expect(stripped).toEqual([]); + }); +}); + +describe("toolDeduper — DeepSeek same-name dedup (new)", () => { + it("deepseek model + duplicate tool names → keeps first definition", () => { + const first = BASH(); + const dup = BASH("Bash", "duplicate description"); + const { tools, stripped } = dedupeTools([first, dup], { model: "deepseek-v4-flash" }); + expect(tools).toEqual([first]); // first wins, including its description + expect(stripped).toEqual(["Bash"]); + }); + + it("deepseek model + (max) thinking suffix → still dedups (suffix stripped before match)", () => { + const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "dup")], { model: "deepseek-v4-flash(max)" }); + expect(tools).toHaveLength(1); + expect(stripped).toEqual(["Bash"]); + }); + + it("deepseek + 3 same-name tools → keeps first, drops both duplicates", () => { + const { tools, stripped } = dedupeTools([BASH(), BASH("Bash", "d1"), BASH("Bash", "d2")], { model: "deepseek-v4-pro" }); + expect(tools).toHaveLength(1); + expect(stripped).toEqual(["Bash", "Bash"]); + }); + + it("deepseek + OpenAI function-shape tools → dedups by function.name", () => { + const { tools } = dedupeTools([FUNC_SHAPE("Bash"), FUNC_SHAPE("Bash")], { model: "deepseek-v4-flash" }); + expect(tools).toHaveLength(1); + expect(tools[0].function.name).toBe("Bash"); + }); + + it("non-DeepSeek model + duplicate tool names → untouched (GLM/MiniMax/Kimi accept them)", () => { + const tools = [BASH(), BASH("Bash", "dup")]; + const { tools: out, stripped } = dedupeTools(tools, { model: "glm-5.2" }); + expect(out).toBe(tools); + expect(stripped).toEqual([]); + }); + + it("no model declared → same-name dedup does NOT run (safe default)", () => { + const tools = [BASH(), BASH("Bash", "dup")]; + const { tools: out, stripped } = dedupeTools(tools, {}); + expect(out).toBe(tools); + expect(stripped).toEqual([]); + }); + + it("deepseek + distinct names → nothing stripped", () => { + const { tools, stripped } = dedupeTools([BASH("Bash"), BASH("ReadFile")], { model: "deepseek-v4-flash" }); + expect(tools).toHaveLength(2); + expect(stripped).toEqual([]); + }); + + it("claude client + deepseek + MCP trigger → both rules apply (union stripped)", () => { + const { tools, stripped } = dedupeTools( + [MCP_EXA, { name: "WebSearch", description: "web" }, BASH(), BASH("Bash", "dup")], + { clientTool: "claude", model: "deepseek-v4-flash" } + ); + expect(tools.map((t) => t.name)).toEqual(["mcp__exa__web_search_exa", "Bash"]); + expect(stripped.sort()).toEqual(["Bash", "WebSearch"]); + }); +});