fix(opencode): route Muse Spark through the Responses API
muse-spark-1.2-contributor-free returned HTTP 500 on /zen/v1/chat/completions.
The model is only served by /zen/v1/responses, so route it there via a per-model
targetFormat and normalize the Chat fields the Responses API rejects
(max_tokens -> max_output_tokens, reasoning_effort -> reasoning{effort,summary}),
clamping max/ultra down to the highest effort the model accepts (xhigh).
Routing stays per-model: the other free models (big-pickle, hy3-free, mimo,
nemotron, laguna) are not served by /responses and keep /chat/completions.
This commit is contained in:
@@ -1,11 +1,13 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
const OPENCODE_UA = "opencode";
|
||||
const MESSAGES_MODELS = new Set();
|
||||
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
|
||||
const RESPONSES_MODELS = new Set(["muse-spark-1.2-contributor-free"]);
|
||||
|
||||
function generateRequestId() {
|
||||
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
@@ -15,19 +17,47 @@ function generateSessionId() {
|
||||
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
}
|
||||
|
||||
// Normalize any resolved id into opencode's ses_ format (stable per-conversation)
|
||||
function toOpencodeSession(id) {
|
||||
const stripped = String(id || "").replace(/^ses_/, "").replace(/-/g, "");
|
||||
return stripped ? `ses_${stripped}` : null;
|
||||
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
function isResponsesModel(model) {
|
||||
return RESPONSES_MODELS.has(baseModelId(model));
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials) {
|
||||
return toOpencodeSession(resolveSessionId({
|
||||
headers: credentials?.rawHeaders,
|
||||
const headers = credentials?.rawHeaders || {};
|
||||
return resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
}));
|
||||
generate: generateSessionId,
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeOpencodeReasoning(model, body) {
|
||||
const current = body.reasoning;
|
||||
const currentReasoning = current && typeof current === "object" && !Array.isArray(current)
|
||||
? current
|
||||
: null;
|
||||
const requestedEffort = typeof body.reasoning_effort === "string"
|
||||
? body.reasoning_effort
|
||||
: currentReasoning?.effort;
|
||||
if (typeof requestedEffort !== "string") return;
|
||||
|
||||
const cleanModel = baseModelId(model || body.model);
|
||||
const supportedLevels = getThinkingLevels("opencode", cleanModel);
|
||||
let effort = requestedEffort.toLowerCase().trim();
|
||||
if ((effort === "max" || effort === "ultra") && supportedLevels?.length && !supportedLevels.includes(effort)) {
|
||||
if (effort === "ultra" && supportedLevels.includes("max")) effort = "max";
|
||||
else if (supportedLevels.includes("xhigh")) effort = "xhigh";
|
||||
}
|
||||
|
||||
body.reasoning = { ...currentReasoning, effort };
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
delete body.reasoning_effort;
|
||||
}
|
||||
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
@@ -38,13 +68,24 @@ export class OpenCodeExecutor extends BaseExecutor {
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
this._currentSessionId = resolveOpencodeSession(body, credentials);
|
||||
if (isResponsesModel(model)) {
|
||||
// Responses API names the output cap max_output_tokens and takes thinking
|
||||
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
|
||||
if (body.max_output_tokens === undefined) {
|
||||
if (body.max_completion_tokens !== undefined) body.max_output_tokens = body.max_completion_tokens;
|
||||
else if (body.max_tokens !== undefined) body.max_output_tokens = body.max_tokens;
|
||||
}
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
normalizeOpencodeReasoning(model, body);
|
||||
}
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
return MESSAGES_MODELS.has(model)
|
||||
? `${base}/zen/v1/messages`
|
||||
return isResponsesModel(model)
|
||||
? `${base}/zen/v1/responses`
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
|
||||
@@ -125,6 +125,8 @@ export const MODEL_CAPABILITIES = {
|
||||
"kimi-for-coding-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
"kimi-k2.7-code": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
"kimi-k2.7-code-highspeed": { vision: true, videoInput: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 262144, maxOutput: 65536 },
|
||||
// OpenCode Free Muse Spark — OpenAI Responses reasoning supports up to xhigh.
|
||||
"muse-spark-1.2-contributor-free": { reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
};
|
||||
|
||||
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
|
||||
@@ -19,7 +19,11 @@ export default {
|
||||
},
|
||||
noAuth: true,
|
||||
},
|
||||
models: [],
|
||||
models: [
|
||||
// Only this model is served by /zen/v1/responses; the rest stay on
|
||||
// /chat/completions, so the format is declared per-model, not per-provider.
|
||||
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
|
||||
],
|
||||
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
|
||||
@@ -300,7 +300,16 @@ function buildReasoningInputItem(msg) {
|
||||
*/
|
||||
export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) {
|
||||
// Body already in Responses API format (e.g. Cursor CLI calling /chat/completions with input[])
|
||||
if (body.input) return { ...body, model, stream: true };
|
||||
if (body.input) {
|
||||
const out = { ...body, model, stream: true };
|
||||
if (out.max_output_tokens === undefined) {
|
||||
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
|
||||
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
|
||||
}
|
||||
delete out.max_tokens;
|
||||
delete out.max_completion_tokens;
|
||||
return out;
|
||||
}
|
||||
|
||||
const result = {
|
||||
model,
|
||||
@@ -416,7 +425,13 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
|
||||
|
||||
// Pass through other relevant fields
|
||||
if (body.temperature !== undefined) result.temperature = body.temperature;
|
||||
if (body.max_tokens !== undefined) result.max_tokens = body.max_tokens;
|
||||
if (body.max_output_tokens !== undefined) {
|
||||
result.max_output_tokens = body.max_output_tokens;
|
||||
} else if (body.max_completion_tokens !== undefined) {
|
||||
result.max_output_tokens = body.max_completion_tokens;
|
||||
} else if (body.max_tokens !== undefined) {
|
||||
result.max_output_tokens = body.max_tokens;
|
||||
}
|
||||
if (body.top_p !== undefined) result.top_p = body.top_p;
|
||||
if (body.reasoning !== undefined) result.reasoning = body.reasoning;
|
||||
if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" };
|
||||
|
||||
Reference in New Issue
Block a user