Merge remote-tracking branch 'origin/master' into gitea/new_feature
# Conflicts: # open-sse/handlers/chatCore.js # open-sse/services/combo.js # src/app/(dashboard)/dashboard/profile/page.js # src/app/api/v1/models/route.js # src/lib/db/repos/settingsRepo.js
This commit is contained in:
@@ -2,7 +2,7 @@ import { PROVIDERS } from "./providers.js";
|
||||
import REGISTRY from "../providers/registry/index.js";
|
||||
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
|
||||
import { PROVIDER_MODELS } from "../providers/index.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, modelSupportedFormats, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js";
|
||||
export { PROVIDER_MODELS };
|
||||
|
||||
@@ -54,6 +54,14 @@ export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
return modelTargetFormat(findModel(models, modelId, aliasOrId));
|
||||
}
|
||||
|
||||
// Declared upstream formats for a model (registry `supportedFormats`). Drives the
|
||||
// per-model guard on the sourceFormat-matched transport; null when undeclared.
|
||||
export function getModelSupportedFormats(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelSupportedFormats(findModel(models, modelId, aliasOrId));
|
||||
}
|
||||
|
||||
export function getModelType(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
|
||||
@@ -245,6 +245,18 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
// Strip tools/toolConfig (handled separately) and blacklisted fields that Google rejects
|
||||
const { tools: _originalTools, toolConfig: _originalToolConfig, ...requestWithoutTools } = body.request || {};
|
||||
stripBlacklisted(requestWithoutTools);
|
||||
|
||||
// Rewrite competitive system prompts (e.g. Zed IDE's Claude prompt) to prevent Antigravity from
|
||||
// flagging the request and immediately blocking it with a 429 Quota Exhausted response.
|
||||
if (requestWithoutTools.systemInstruction?.parts) {
|
||||
const oldText = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
|
||||
for (const part of requestWithoutTools.systemInstruction.parts) {
|
||||
if (typeof part.text === "string" && part.text.includes(oldText)) {
|
||||
part.text = part.text.split(oldText).join("");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const generationConfig = { ...(requestWithoutTools.generationConfig || {}) };
|
||||
if (generationConfig.maxOutputTokens > MAX_ANTIGRAVITY_OUTPUT_TOKENS) {
|
||||
generationConfig.maxOutputTokens = MAX_ANTIGRAVITY_OUTPUT_TOKENS;
|
||||
|
||||
@@ -10,7 +10,6 @@ import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
@@ -41,7 +40,6 @@ const executors = {
|
||||
vertex: new VertexExecutor("vertex"),
|
||||
"vertex-partner": new VertexExecutor("vertex-partner"),
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
@@ -86,7 +84,6 @@ export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
export { DefaultExecutor } from "./default.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
|
||||
@@ -144,6 +144,12 @@ function normalizeStopReason(value) {
|
||||
return reason || null;
|
||||
}
|
||||
|
||||
// Of the reasons stopDisposition() folds into "terminal_incomplete", only these
|
||||
// mean "usable as far as it got, then the budget ran out" -- the case
|
||||
// finish_reason "length" exists for. cancelled / pause_turn are abandoned turns
|
||||
// whose partial content must stay private, so they are deliberately absent.
|
||||
const KIRO_TRUNCATION_STOP_REASONS = new Set(["model_context_window_exceeded", "max_tokens"]);
|
||||
|
||||
function stopDisposition(stopReason, hasToolCalls) {
|
||||
if (["malformed_model_output", "invalid_model_output"].includes(stopReason)) return "retryable_protocol_failure";
|
||||
if (["cancelled", "pause_turn", "model_context_window_exceeded"].includes(stopReason)) return "terminal_incomplete";
|
||||
@@ -711,14 +717,25 @@ export class KiroExecutor extends BaseExecutor {
|
||||
};
|
||||
const emitTools = (controller) => {
|
||||
for (const tool of state.tools.values()) {
|
||||
const input = parsedToolInput(tool);
|
||||
if (tool.name === "tool_call") {
|
||||
if (typeof input.name !== "string" || !input.name.trim()) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool name");
|
||||
}
|
||||
if (!Object.prototype.hasOwnProperty.call(input, "arguments")) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool arguments");
|
||||
// Validate per tool, not per turn: one unusable fragment used to throw out
|
||||
// of emitTools and take every other complete tool call in the same turn
|
||||
// with it, which the client saw as a turn that answered nothing.
|
||||
let input;
|
||||
try {
|
||||
input = parsedToolInput(tool);
|
||||
if (tool.name === "tool_call") {
|
||||
if (typeof input.name !== "string" || !input.name.trim()) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool name");
|
||||
}
|
||||
if (!Object.prototype.hasOwnProperty.call(input, "arguments")) {
|
||||
throw new Error("Invalid Kiro tool_call payload: missing nested MCP tool arguments");
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
state.droppedTools = (state.droppedTools || 0) + 1;
|
||||
state.toolValidationError ||= error.message;
|
||||
console.error(`[Kiro] dropping unusable tool call ${tool.id} (${tool.name}): ${error.message}`);
|
||||
continue;
|
||||
}
|
||||
const index = state.toolCounter++;
|
||||
emitDelta(controller, {
|
||||
@@ -729,14 +746,26 @@ export class KiroExecutor extends BaseExecutor {
|
||||
function: { name: tool.name, arguments: "" }
|
||||
}]
|
||||
});
|
||||
const serializedInput = JSON.stringify(input);
|
||||
emitDelta(controller, {
|
||||
tool_calls: [{ index, function: { arguments: JSON.stringify(input) } }]
|
||||
tool_calls: [{ index, function: { arguments: serializedInput } }]
|
||||
});
|
||||
// Tool arguments are billed output like any other completion bytes. They
|
||||
// were never added to totalContentLength, so the /4 estimator in finish()
|
||||
// reported OUT 0 -- or the Math.max floor of 1 -- for every turn whose
|
||||
// entire answer was a tool call.
|
||||
state.totalContentLength += tool.name.length + serializedInput.length;
|
||||
state.hasToolCalls = true;
|
||||
}
|
||||
state.tools.clear();
|
||||
state.bufferedToolBytes = 0;
|
||||
if (state.stopReason === "tool_use" && !state.hasToolCalls) {
|
||||
// A declared tool turn that emitted no usable call is only fatal when the
|
||||
// turn produced nothing else. Throwing unconditionally here escaped
|
||||
// emitTools() with provenance "invalid_tool_call", which the integrity gate
|
||||
// re-derived into a repair retry -- discarding text the client had already
|
||||
// been promised.
|
||||
if (state.stopReason === "tool_use" && !state.hasToolCalls &&
|
||||
!state.hasText && !state.hasReasoning && !state.hasCode) {
|
||||
throw new Error("Kiro tool_use stop reason did not include a complete tool call");
|
||||
}
|
||||
};
|
||||
@@ -796,7 +825,6 @@ export class KiroExecutor extends BaseExecutor {
|
||||
emitDelta(controller, { content: event.payload.content });
|
||||
} else if (eventType === "toolUseEvent") {
|
||||
state.sawToolUse = true;
|
||||
if (state.toolValidationError) return true;
|
||||
const values = Array.isArray(event.payload) ? event.payload : [event.payload];
|
||||
if (!values[0]) throw new Error("Kiro toolUseEvent is empty");
|
||||
for (const value of values) {
|
||||
@@ -924,9 +952,10 @@ export class KiroExecutor extends BaseExecutor {
|
||||
} catch (error) {
|
||||
const bufferExceeded = error.code === "KIRO_BUFFER_EXCEEDED";
|
||||
if (!bufferExceeded) {
|
||||
// Keep whatever is already buffered: the rejected fragment belongs to
|
||||
// one tool, and clearing the map dropped the complete calls too.
|
||||
state.toolValidationError ||= error.message;
|
||||
state.tools.clear();
|
||||
state.bufferedToolBytes = 0;
|
||||
console.error(`[Kiro] tool fragment rejected, keeping ${state.tools.size} buffered tool(s): ${error.message}`);
|
||||
continue;
|
||||
}
|
||||
fail(
|
||||
@@ -958,7 +987,16 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
state.transportState = "clean_eof";
|
||||
const declaredDisposition = stopDisposition(state.stopReason, state.sawToolUse);
|
||||
if (["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(declaredDisposition)) {
|
||||
// model_context_window_exceeded / max_tokens map to terminal_incomplete. When
|
||||
// they arrive after the model already streamed content, fail() threw away a
|
||||
// complete-enough answer; a truncated turn is what finish_reason "length" is
|
||||
// for. chunkIndex > 0 means at least one delta already reached the client.
|
||||
const declaredTruncatedAfterOutput = declaredDisposition === "terminal_incomplete" &&
|
||||
KIRO_TRUNCATION_STOP_REASONS.has(state.stopReason) && state.chunkIndex > 0;
|
||||
if (declaredTruncatedAfterOutput) {
|
||||
console.error(`[Kiro] truncated after ${state.chunkIndex} chunk(s) (stop_reason=${state.stopReason}); keeping output`);
|
||||
}
|
||||
if (!declaredTruncatedAfterOutput && ["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(declaredDisposition)) {
|
||||
const code = declaredDisposition === "retryable_protocol_failure"
|
||||
? "kiro_retryable_protocol_failure"
|
||||
: declaredDisposition === "terminal_refusal"
|
||||
@@ -975,16 +1013,6 @@ export class KiroExecutor extends BaseExecutor {
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (state.toolValidationError) {
|
||||
fail(
|
||||
controller,
|
||||
"invalid_tool_call",
|
||||
"invalid_kiro_tool_call",
|
||||
state.toolValidationError,
|
||||
{ transport_state: state.transportState, stop_disposition: "retryable_protocol_failure" }
|
||||
);
|
||||
return;
|
||||
}
|
||||
try {
|
||||
emitTools(controller);
|
||||
} catch (error) {
|
||||
@@ -997,6 +1025,22 @@ export class KiroExecutor extends BaseExecutor {
|
||||
);
|
||||
return;
|
||||
}
|
||||
// Fail only when the turn has nothing usable left. emitTools() validates
|
||||
// per tool and drops just the unusable ones, so this has to run AFTER it:
|
||||
// before, the rejected tool was still buffered and tools.size was never 0.
|
||||
// A turn that also produced text keeps that text -- the dropped call is
|
||||
// logged, not fatal.
|
||||
if (state.toolValidationError && !state.hasToolCalls &&
|
||||
!state.hasText && !state.hasReasoning && !state.hasCode) {
|
||||
fail(
|
||||
controller,
|
||||
"invalid_tool_call",
|
||||
"invalid_kiro_tool_call",
|
||||
state.toolValidationError,
|
||||
{ transport_state: state.transportState, stop_disposition: "retryable_protocol_failure" }
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
const hasOutput = state.hasText || state.hasReasoning || state.hasCode || state.hasToolCalls;
|
||||
if (!hasOutput && !state.explicitStop) {
|
||||
@@ -1011,7 +1055,13 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
const disposition = stopDisposition(state.stopReason, state.hasToolCalls);
|
||||
if (["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(disposition)) {
|
||||
// Same reasoning as declaredTruncatedAfterOutput above.
|
||||
const truncatedAfterOutput = disposition === "terminal_incomplete" &&
|
||||
KIRO_TRUNCATION_STOP_REASONS.has(state.stopReason) && state.chunkIndex > 0;
|
||||
if (truncatedAfterOutput) {
|
||||
console.error(`[Kiro] truncated after ${state.chunkIndex} chunk(s) (stop_reason=${state.stopReason}); closing as length`);
|
||||
}
|
||||
if (!truncatedAfterOutput && ["retryable_protocol_failure", "terminal_incomplete", "terminal_refusal", "unknown_failure"].includes(disposition)) {
|
||||
const code = disposition === "retryable_protocol_failure"
|
||||
? "kiro_retryable_protocol_failure"
|
||||
: disposition === "terminal_refusal"
|
||||
@@ -1041,18 +1091,24 @@ export class KiroExecutor extends BaseExecutor {
|
||||
total_tokens: prompt + completion
|
||||
};
|
||||
}
|
||||
const finishReason = state.hasToolCalls
|
||||
? "tool_calls"
|
||||
: disposition === "length"
|
||||
? "length"
|
||||
: "stop";
|
||||
const finishReason = truncatedAfterOutput
|
||||
? "length"
|
||||
: state.hasToolCalls
|
||||
? "tool_calls"
|
||||
: disposition === "length"
|
||||
? "length"
|
||||
: "stop";
|
||||
controller.enqueue(sseChunk({}, finishReason, state.usage));
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
state.finished = true;
|
||||
options.onTerminalState?.(diagnostics({
|
||||
terminal_provenance: state.terminalProvenance || "clean_eventstream_eof",
|
||||
transport_state: state.transportState,
|
||||
stop_disposition: disposition
|
||||
// Report what this exit actually did, not the raw disposition. The
|
||||
// integrity gate re-derives its verdict from stop_disposition, so
|
||||
// reporting "terminal_incomplete" for a turn we deliberately kept made
|
||||
// it discard the very bytes we just released to the client.
|
||||
stop_disposition: truncatedAfterOutput ? "length" : disposition
|
||||
}));
|
||||
};
|
||||
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
|
||||
// Models that use /zen/go/v1/messages (Anthropic/Claude format + x-api-key auth)
|
||||
const MESSAGES_FORMAT_MODELS = new Set([
|
||||
"minimax-m3",
|
||||
"minimax-m2.7",
|
||||
"minimax-m2.5",
|
||||
"qwen3.7-max",
|
||||
"qwen3.7-plus",
|
||||
"qwen3.6-plus",
|
||||
]);
|
||||
|
||||
const BASE = "https://opencode.ai/zen/go/v1";
|
||||
|
||||
export class OpenCodeGoExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode-go", PROVIDERS["opencode-go"]);
|
||||
}
|
||||
|
||||
// buildUrl runs before buildHeaders in BaseExecutor.execute, cache model here
|
||||
buildUrl(model) {
|
||||
this._lastModel = model;
|
||||
return MESSAGES_FORMAT_MODELS.has(model)
|
||||
? `${BASE}/messages`
|
||||
: `${BASE}/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const key = credentials?.apiKey || credentials?.accessToken;
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
|
||||
if (MESSAGES_FORMAT_MODELS.has(this._lastModel)) {
|
||||
headers["x-api-key"] = key;
|
||||
headers["anthropic-version"] = ANTHROPIC_API_VERSION;
|
||||
} else {
|
||||
headers["Authorization"] = `Bearer ${key}`;
|
||||
}
|
||||
|
||||
if (stream) headers["Accept"] = "text/event-stream";
|
||||
return headers;
|
||||
}
|
||||
|
||||
transformRequest(model, body) {
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
}
|
||||
@@ -1,16 +1,43 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
// Models that use /zen/v1/messages (claude format)
|
||||
const OPENCODE_UA = "opencode";
|
||||
const MESSAGES_MODELS = new Set();
|
||||
|
||||
function generateRequestId() {
|
||||
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
}
|
||||
|
||||
function generateSessionId() {
|
||||
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
}
|
||||
|
||||
// Normalize any resolved id into opencode's ses_ format (stable per-conversation)
|
||||
function toOpencodeSession(id) {
|
||||
const stripped = String(id || "").replace(/^ses_/, "").replace(/-/g, "");
|
||||
return stripped ? `ses_${stripped}` : null;
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials) {
|
||||
return toOpencodeSession(resolveSessionId({
|
||||
headers: credentials?.rawHeaders,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
}));
|
||||
}
|
||||
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode", PROVIDERS.opencode);
|
||||
this._currentSessionId = null;
|
||||
}
|
||||
|
||||
transformRequest(model, body) {
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
this._currentSessionId = resolveOpencodeSession(body, credentials);
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
@@ -21,12 +48,23 @@ export class OpenCodeExecutor extends BaseExecutor {
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders() {
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const lower = {};
|
||||
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
|
||||
|
||||
const downstreamUa = lower["user-agent"] || "";
|
||||
const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode");
|
||||
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer public",
|
||||
"x-opencode-client": "desktop",
|
||||
"Accept": "text/event-stream"
|
||||
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
|
||||
"x-opencode-client": lower["x-opencode-client"] || "desktop",
|
||||
"x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(),
|
||||
"x-opencode-request": lower["x-opencode-request"] || generateRequestId(),
|
||||
"x-opencode-project": lower["x-opencode-project"] || "global",
|
||||
"Accept": stream ? "text/event-stream" : "*/*",
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
@@ -216,6 +216,52 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a qoder error message indicates a billing/quota block.
|
||||
* Signatures: code 112 (quota exhausted), code 10605 (queue throttle), pricingUrl field.
|
||||
*/
|
||||
function isBillingBlock(inner) {
|
||||
if (!inner || typeof inner !== "string") return false;
|
||||
const lowerMsg = inner.toLowerCase();
|
||||
// Match: {"code":"112",...}, {"code":"10605",...}, or pricingUrl field
|
||||
return /\"code\"\s*:\s*\"(112|10605)\"/.test(inner) || lowerMsg.includes("pricingurl");
|
||||
}
|
||||
|
||||
/**
|
||||
* Peek the first SSE frame to detect billing errors before piping.
|
||||
* Returns { isBilling, statusVal, message, consumed } — `consumed` is every
|
||||
* byte read so far (including the peeked line) so the caller can re-process
|
||||
* it and nothing is dropped from the stream.
|
||||
*/
|
||||
async function peekFirstQoderFrame(reader, decoder) {
|
||||
let consumed = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) return { isBilling: false, consumed, upstreamDone: true };
|
||||
|
||||
consumed += decoder.decode(value, { stream: true });
|
||||
const nl = consumed.indexOf("\n");
|
||||
if (nl === -1) continue; // need a full line first
|
||||
|
||||
const line = consumed.slice(0, nl).replace(/\r$/, "").trim();
|
||||
if (!line.startsWith("data:")) continue;
|
||||
|
||||
const data = line.slice(5).trimStart();
|
||||
if (data === "[DONE]") return { isBilling: false, consumed };
|
||||
|
||||
let envelope;
|
||||
try { envelope = JSON.parse(data); } catch { return { isBilling: false, consumed }; }
|
||||
|
||||
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
|
||||
const inner = typeof envelope.body === "string" ? envelope.body : "";
|
||||
|
||||
if (statusVal !== 200 && isBillingBlock(inner)) {
|
||||
return { isBilling: true, statusVal, message: inner || `qoder billing block (${statusVal})` };
|
||||
}
|
||||
return { isBilling: false, consumed };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap the upstream's `{statusCodeValue, body}` SSE envelope into plain
|
||||
* OpenAI SSE chunks the rest of the chatCore pipeline understands.
|
||||
@@ -230,16 +276,34 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
* [DONE]/error frame (agent keepalive). Non-streaming clients drain via
|
||||
* response.text() which hangs until the socket closes — so on terminal
|
||||
* events we cancel the upstream reader and close our stream immediately.
|
||||
*
|
||||
* NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl).
|
||||
* If detected, return 403 response so chatCore marks connection unavailable
|
||||
* and triggers combo fallback instead of leaking error text into chat.
|
||||
*/
|
||||
function wrapQoderSSE(response, model) {
|
||||
async function wrapQoderSSE(response, model) {
|
||||
if (!response.ok || !response.body) return response;
|
||||
|
||||
const decoder = new TextDecoder();
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
let doneEmitted = false;
|
||||
const reader = response.body.getReader();
|
||||
|
||||
// Peek first frame to detect billing block
|
||||
const peek = await peekFirstQoderFrame(reader, decoder);
|
||||
if (peek?.isBilling) {
|
||||
// Billing block detected — return 403 so chatCore fails this connection
|
||||
await reader.cancel().catch(() => {});
|
||||
return new Response(
|
||||
JSON.stringify({ error: { message: peek.message, code: peek.statusVal } }),
|
||||
{ status: 403, headers: { "Content-Type": "application/json" } }
|
||||
);
|
||||
}
|
||||
|
||||
// Normal flow: re-process every byte the peek consumed, then continue.
|
||||
let buffer = peek.consumed || "";
|
||||
const upstreamDrained = peek.upstreamDone === true;
|
||||
const encoder = new TextEncoder();
|
||||
let doneEmitted = false;
|
||||
|
||||
// Process one already-extracted SSE line (no trailing newline).
|
||||
const processLine = (line, controller) => {
|
||||
const trimmed = line.replace(/\r$/, "").trim();
|
||||
@@ -288,7 +352,28 @@ function wrapQoderSSE(response, model) {
|
||||
// enqueueing would never be re-invoked, hanging consumers like .text().
|
||||
async start(controller) {
|
||||
try {
|
||||
while (!doneEmitted) {
|
||||
// Drain whatever the peek already pulled off the socket first.
|
||||
let nlSeed;
|
||||
while ((nlSeed = buffer.indexOf("\n")) !== -1) {
|
||||
const line = buffer.slice(0, nlSeed);
|
||||
buffer = buffer.slice(nlSeed + 1);
|
||||
processLine(line, controller);
|
||||
if (doneEmitted) {
|
||||
await reader.cancel().catch(() => {});
|
||||
controller.close();
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (upstreamDrained) {
|
||||
// Peek hit end-of-stream: flush any trailing partial line.
|
||||
buffer += decoder.decode();
|
||||
if (buffer.length > 0) {
|
||||
processLine(buffer, controller);
|
||||
buffer = "";
|
||||
}
|
||||
}
|
||||
|
||||
while (!doneEmitted && !upstreamDrained) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) {
|
||||
buffer += decoder.decode();
|
||||
@@ -473,7 +558,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
return { response, url, headers, transformedBody: payload };
|
||||
}
|
||||
|
||||
const wrapped = wrapQoderSSE(response, `qoder/${qoderKey}`);
|
||||
const wrapped = await wrapQoderSSE(response, `qoder/${qoderKey}`);
|
||||
return { response: wrapped, url, headers, transformedBody: payload };
|
||||
}
|
||||
|
||||
@@ -497,4 +582,5 @@ export const __test__ = {
|
||||
normalizeMessages,
|
||||
wrapQoderSSE,
|
||||
buildQoderRequestBody,
|
||||
isBillingBlock,
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -44,13 +44,14 @@ export function extractUsageFromResponse(responseBody) {
|
||||
};
|
||||
}
|
||||
|
||||
// Gemini format
|
||||
if (responseBody.usageMetadata) {
|
||||
// Gemini format. Antigravity / gemini-cli wrap the payload in { response: {...} }.
|
||||
const usageMetadata = responseBody.usageMetadata || responseBody.response?.usageMetadata;
|
||||
if (usageMetadata) {
|
||||
return {
|
||||
prompt_tokens: responseBody.usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: responseBody.usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: responseBody.usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: responseBody.usageMetadata.thoughtsTokenCount || 0
|
||||
prompt_tokens: usageMetadata.promptTokenCount || 0,
|
||||
completion_tokens: usageMetadata.candidatesTokenCount || 0,
|
||||
cached_tokens: usageMetadata.cachedContentTokenCount || 0,
|
||||
reasoning_tokens: usageMetadata.thoughtsTokenCount || 0
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -29,6 +29,8 @@
|
||||
* @property {Record<string,unknown>} [providerSpecificData]
|
||||
*/
|
||||
|
||||
import { assertPublicUrl } from "../../../src/shared/utils/ssrfGuard.js";
|
||||
|
||||
// ── Helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
@@ -63,12 +65,31 @@ export function getProviderSetting(params, key) {
|
||||
|
||||
/**
|
||||
* Resolve base URL with optional override from providerOptions.baseUrl.
|
||||
*
|
||||
* The override is client-controlled and therefore SSRF-hardened: only public
|
||||
* http(s) URLs are accepted (internal/private/loopback/metadata addresses are
|
||||
* rejected via assertPublicUrl). The provider's own configured baseUrl is
|
||||
* trusted as-is (admin-controlled).
|
||||
*
|
||||
* @param {SearchProviderConfig} config
|
||||
* @param {SearchRequestParams} params
|
||||
* @returns {string}
|
||||
*/
|
||||
export function resolveBaseUrl(config, params) {
|
||||
const override = getProviderSetting(params, "baseUrl");
|
||||
if (override) {
|
||||
// SSRF guard: client-supplied base URLs must be public http(s) only.
|
||||
let parsed;
|
||||
try {
|
||||
parsed = new URL(override);
|
||||
} catch {
|
||||
throw new Error(`Invalid baseUrl: ${override}`);
|
||||
}
|
||||
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") {
|
||||
throw new Error(`Invalid baseUrl protocol: ${parsed.protocol}`);
|
||||
}
|
||||
assertPublicUrl(override);
|
||||
}
|
||||
return (override || config.baseUrl).replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
|
||||
@@ -51,6 +51,25 @@ async function huggingface({ baseUrl, apiKey, text, modelId }) {
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Fish Audio: model travels in an HTTP header, the voice is a reference_id, returns binary
|
||||
async function fishAudio({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
"model": modelId || "s2.1-pro-free",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
format: "mp3",
|
||||
...(voiceId ? { reference_id: voiceId } : {}),
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Inworld: Basic auth, JSON { audioContent }
|
||||
async function inworld({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
@@ -166,4 +185,5 @@ export const FORMAT_HANDLERS = {
|
||||
tortoise,
|
||||
openai: openaiCompat,
|
||||
"minimax-tts": minimaxTts,
|
||||
"fish-audio": fishAudio,
|
||||
};
|
||||
|
||||
@@ -205,6 +205,7 @@ export const PATTERN_CAPABILITIES = [
|
||||
|
||||
// ── Gemini (all 2.0+ multimodal + google_search grounding, 1M ctx) ─
|
||||
{ pattern: "*gemini*image*", caps: { vision: true, imageOutput: true, contextWindow: 1048576 } },
|
||||
{ pattern: "*gemini-3.7*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-3*pro*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65535 } },
|
||||
{ pattern: "*gemini-3*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-level", thinkingCanDisable: false, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
{ pattern: "*gemini-2.5*", caps: { vision: true, audioInput: true, videoInput: true, reasoning: true, search: true, thinkingFormat: "gemini-budget", thinkingRange: { min: 0, max: 24576 }, contextWindow: 1048576, maxOutput: 65536 } },
|
||||
|
||||
@@ -38,3 +38,11 @@ export function modelStrip(model) {
|
||||
export function modelTargetFormat(model) {
|
||||
return model?.targetFormat || MODEL_DEFAULTS.targetFormat;
|
||||
}
|
||||
|
||||
// Per-model declared upstream formats (e.g. ["openai", "claude"]). Guards the
|
||||
// sourceFormat-matched transport for multi-endpoint providers whose models differ
|
||||
// in endpoint support (opencode-go: kimi/glm only do /chat/completions, minimax/qwen
|
||||
// also do /messages, deepseek also does /responses).
|
||||
export function modelSupportedFormats(model) {
|
||||
return model?.supportedFormats || null;
|
||||
}
|
||||
|
||||
@@ -57,6 +57,10 @@ export const MODEL_PRICING = {
|
||||
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
|
||||
// === Gemini ===
|
||||
"gemini-3.7-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.7-flash-low": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash-high": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
"gemini-3.6-flash-medium": { input: 1.50, output: 7.50, cached: 0.15, reasoning: 11.25, cache_creation: 1.875 },
|
||||
|
||||
35
open-sse/providers/registry/alitp-intl.js
Normal file
35
open-sse/providers/registry/alitp-intl.js
Normal file
@@ -0,0 +1,35 @@
|
||||
// Token Plan — credit subscription keys on token-plan.<region>.maas.aliyuncs.com.
|
||||
// Fourth Alibaba key type: Coding Plan (alicode/alicode-intl) and Model Studio
|
||||
// (alims-intl) both reject these keys, and they reject Model Studio keys back.
|
||||
// Singapore is the only region that serves the plan; eu-central-1 answers
|
||||
// IllegalEndpoint. The Anthropic surface (/apps/anthropic/v1/messages) is not
|
||||
// authorized for this plan, so OpenAI-compatible mode is the only transport.
|
||||
export default {
|
||||
id: "alitp-intl",
|
||||
priority: 11,
|
||||
alias: "alitp-intl",
|
||||
display: {
|
||||
name: "Alibaba Token Plan",
|
||||
icon: "cloud",
|
||||
color: "#FF6A00",
|
||||
textIcon: "ATP",
|
||||
website: "https://www.alibabacloud.com/campaign/ai-landing-page-token",
|
||||
notice: {
|
||||
apiKeyUrl: "https://modelstudio.console.alibabacloud.com/?apiKey=1",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1/chat/completions",
|
||||
headers: {},
|
||||
quirks: { preserveCacheControl: true },
|
||||
},
|
||||
models: [
|
||||
{ id: "qwen3.8-max-preview", name: "Qwen3.8 Max Preview" },
|
||||
{ id: "qwen3.7-max", name: "Qwen3.7 Max" },
|
||||
{ id: "qwen3.7-plus", name: "Qwen3.7 Plus" },
|
||||
{ id: "qwen3.6-flash", name: "Qwen3.6 Flash" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
],
|
||||
};
|
||||
@@ -45,6 +45,9 @@ export default {
|
||||
clientSecret: "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",
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.7-flash-high", name: "Gemini 3.7 Flash (High)", upstreamModelId: "gemini-3.7-flash-tiered(high)" },
|
||||
{ id: "gemini-3.7-flash-medium", name: "Gemini 3.7 Flash (Medium)", upstreamModelId: "gemini-3.7-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.7-flash-low", name: "Gemini 3.7 Flash (Low)", upstreamModelId: "gemini-3.7-flash-tiered(low)" },
|
||||
{ id: "gemini-3.6-flash-high", name: "Gemini 3.6 Flash (High)", upstreamModelId: "gemini-3.6-flash-tiered(high)" },
|
||||
{ id: "gemini-3.6-flash-medium", name: "Gemini 3.6 Flash (Medium)", upstreamModelId: "gemini-3.6-flash-tiered(medium)" },
|
||||
{ id: "gemini-3.6-flash-low", name: "Gemini 3.6 Flash (Low)", upstreamModelId: "gemini-3.6-flash-tiered(low)" },
|
||||
|
||||
31
open-sse/providers/registry/fish-audio.js
Normal file
31
open-sse/providers/registry/fish-audio.js
Normal file
@@ -0,0 +1,31 @@
|
||||
// Fish Audio TTS — the model id travels in an HTTP `model` header rather than the
|
||||
// JSON body, and the voice is a reference_id (a cloned or preset voice model).
|
||||
export default {
|
||||
id: "fish-audio",
|
||||
alias: "fish",
|
||||
display: {
|
||||
name: "Fish Audio",
|
||||
icon: "record_voice_over",
|
||||
color: "#1E9BF0",
|
||||
textIcon: "FA",
|
||||
website: "https://fish.audio",
|
||||
notice: {
|
||||
apiKeyUrl: "https://fish.audio/app/api-keys/",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
serviceKinds: ["tts"],
|
||||
ttsConfig: {
|
||||
baseUrl: "https://api.fish.audio/v1/tts",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "fish-audio",
|
||||
models: [
|
||||
{ id: "s2.1-pro-free", name: "S2.1 Pro Free" },
|
||||
{ id: "s2.1-pro", name: "S2.1 Pro" },
|
||||
{ id: "s2-pro", name: "S2 Pro" },
|
||||
{ id: "s1", name: "S1" },
|
||||
],
|
||||
},
|
||||
};
|
||||
@@ -36,6 +36,7 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" },
|
||||
{ id: "gemini-3.6-flash", name: "Gemini 3.6 Flash" },
|
||||
{ id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite" },
|
||||
{ id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
|
||||
|
||||
@@ -21,6 +21,7 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "glm-5.3", name: "GLM 5.3" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
|
||||
@@ -45,6 +45,7 @@ export default {
|
||||
},
|
||||
],
|
||||
models: [
|
||||
{ id: "glm-5.3", name: "GLM 5.3" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
|
||||
@@ -119,6 +119,8 @@ import p116 from "./tokenrouter.js";
|
||||
import p117 from "./selfhosted-stt.js";
|
||||
import p118 from "./selfhosted-tts.js";
|
||||
import p119 from "./selfhosted-embedding.js";
|
||||
import p120 from "./fish-audio.js";
|
||||
import p121 from "./alitp-intl.js";
|
||||
|
||||
export default [
|
||||
p0,
|
||||
@@ -239,4 +241,6 @@ export default [
|
||||
p117,
|
||||
p118,
|
||||
p119,
|
||||
p120,
|
||||
p121,
|
||||
];
|
||||
|
||||
@@ -14,7 +14,7 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "freeTier",
|
||||
authModes: ["oauth"],
|
||||
authModes: ["oauth", "apikey"],
|
||||
hasOAuth: true,
|
||||
transport: {
|
||||
baseUrl: "https://llm.kimchi.dev/openai/v1/chat/completions",
|
||||
|
||||
@@ -22,20 +22,28 @@ export default {
|
||||
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
|
||||
headers: {},
|
||||
},
|
||||
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
|
||||
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
|
||||
// opencode-go models differ in endpoint support.
|
||||
transports: [
|
||||
{ format: "openai", baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
|
||||
{ format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
|
||||
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
|
||||
],
|
||||
models: [
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", targetFormat: "claude" },
|
||||
{ id: "minimax-m2.7", name: "MiniMax M2.7", targetFormat: "claude" },
|
||||
{ id: "minimax-m2.5", name: "MiniMax M2.5", targetFormat: "claude" },
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", targetFormat: "claude" },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", targetFormat: "claude" },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", targetFormat: "claude" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
|
||||
],
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -35,7 +35,7 @@ const USAGE_HANDLERS = {
|
||||
github: (c) => getGitHubUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
"gemini-cli": (c) => getGeminiUsage(c.accessToken, c.providerDataWithProjectId, c.proxyOptions),
|
||||
antigravity: (c) => getAntigravityUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
claude: (c) => getClaudeUsage(c.accessToken, c.proxyOptions),
|
||||
claude: (c) => getClaudeUsage(c.accessToken, c.proxyOptions, { force: c.force }),
|
||||
codex: (c) => getCodexUsage(c.accessToken, c.proxyOptions),
|
||||
kiro: (c) => getKiroUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
qoder: async (c) => {
|
||||
@@ -60,7 +60,7 @@ const USAGE_HANDLERS = {
|
||||
commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions),
|
||||
};
|
||||
|
||||
export async function getUsageForProvider(connection, proxyOptions = null) {
|
||||
export async function getUsageForProvider(connection, proxyOptions = null, options = {}) {
|
||||
const { provider, accessToken, apiKey, providerSpecificData, projectId } = connection;
|
||||
const providerDataWithProjectId = {
|
||||
...(providerSpecificData || {}),
|
||||
@@ -69,5 +69,13 @@ export async function getUsageForProvider(connection, proxyOptions = null) {
|
||||
|
||||
const handler = USAGE_HANDLERS[provider];
|
||||
if (!handler) return { message: `Usage API not implemented for ${provider}` };
|
||||
return await handler({ provider, accessToken, apiKey, providerSpecificData, providerDataWithProjectId, proxyOptions });
|
||||
return await handler({
|
||||
provider,
|
||||
accessToken,
|
||||
apiKey,
|
||||
providerSpecificData,
|
||||
providerDataWithProjectId,
|
||||
proxyOptions,
|
||||
force: options.force === true,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -19,7 +19,43 @@ const CLAUDE_CONFIG = {
|
||||
const OAUTH_429_COOLDOWN_MS = 180000;
|
||||
const oauthCooldown = new Map();
|
||||
|
||||
export async function getClaudeUsage(accessToken, proxyOptions = null) {
|
||||
// Dedup + short TTL cache per access token. Many tabs / many accounts / auto-refresh
|
||||
// all funnel through here; without this each call hits Anthropic and triggers 429.
|
||||
const USAGE_CACHE_TTL_MS = 300000;
|
||||
const usageCache = new Map(); // token -> { promise } | { result, expiresAt }
|
||||
|
||||
export async function getClaudeUsage(accessToken, proxyOptions = null, options = {}) {
|
||||
const force = options?.force === true;
|
||||
|
||||
// Serve in-flight or fresh cached result (skip on manual force)
|
||||
if (!force && accessToken) {
|
||||
const hit = usageCache.get(accessToken);
|
||||
if (hit?.promise) return hit.promise;
|
||||
if (hit && hit.expiresAt > Date.now()) return hit.result;
|
||||
}
|
||||
|
||||
const stale = (!force && accessToken && usageCache.get(accessToken)?.result) || null;
|
||||
|
||||
const promise = (async () => {
|
||||
const result = await fetchClaudeUsageRaw(accessToken, proxyOptions);
|
||||
// Only cache real quota data, not soft-failure {message: ...} payloads
|
||||
if (accessToken && result?.quotas) {
|
||||
usageCache.set(accessToken, {
|
||||
result,
|
||||
expiresAt: Date.now() + USAGE_CACHE_TTL_MS,
|
||||
});
|
||||
return result;
|
||||
}
|
||||
// Soft failure (429/error): prefer the last good read over a transient error
|
||||
if (stale) return stale;
|
||||
return result;
|
||||
})();
|
||||
|
||||
if (accessToken) usageCache.set(accessToken, { promise });
|
||||
return promise;
|
||||
}
|
||||
|
||||
async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) {
|
||||
try {
|
||||
// Skip OAuth usage call while this token is cooling down from a recent 429
|
||||
const cooldownUntil = oauthCooldown.get(accessToken);
|
||||
|
||||
@@ -161,6 +161,9 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
if (data.models) {
|
||||
// Filter only recommended/important models (must match PROVIDER_MODELS ag ids)
|
||||
const importantModels = [
|
||||
'gemini-3.7-flash-high',
|
||||
'gemini-3.7-flash-medium',
|
||||
'gemini-3.7-flash-low',
|
||||
'gemini-3.6-flash-high',
|
||||
'gemini-3.6-flash-medium',
|
||||
'gemini-3.6-flash-low',
|
||||
|
||||
@@ -62,6 +62,19 @@ function stripOpenAI(body, caps) {
|
||||
if (!Array.isArray(body.messages)) return;
|
||||
const last = body.messages.length - 1;
|
||||
body.messages.forEach((msg, i) => {
|
||||
if (caps.vision === false) {
|
||||
if (Array.isArray(msg.images)) delete msg.images;
|
||||
if (Array.isArray(msg.experimental_attachments)) {
|
||||
msg.experimental_attachments = msg.experimental_attachments.filter(
|
||||
(a) => !(a?.contentType?.startsWith("image/") || (typeof a?.url === "string" && a.url.startsWith("data:image/")))
|
||||
);
|
||||
}
|
||||
if (Array.isArray(msg.attachments)) {
|
||||
msg.attachments = msg.attachments.filter(
|
||||
(a) => !(a?.contentType?.startsWith("image/") || (typeof a?.url === "string" && a.url.startsWith("data:image/")))
|
||||
);
|
||||
}
|
||||
}
|
||||
if (!Array.isArray(msg.content)) return;
|
||||
const removed = new Set();
|
||||
msg.content = filterBlocks(msg.content, capForOpenAIBlock, caps, removed, i === last);
|
||||
|
||||
@@ -9,6 +9,9 @@ import { PROVIDERS } from "../../providers/index.js";
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
|
||||
|
||||
const CACHE_CONTROL_5M = { type: "ephemeral" };
|
||||
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
|
||||
|
||||
// Check if message has valid non-empty content
|
||||
export function hasValidContent(msg) {
|
||||
if (typeof msg.content === "string" && msg.content.trim()) return true;
|
||||
@@ -124,32 +127,38 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
if (Object.keys(body.output_config).length === 0) delete body.output_config;
|
||||
}
|
||||
|
||||
// 2. Hoist mid-conversation system messages into the top-level system field
|
||||
// 2. Fold mid-conversation system messages into the neighbouring turn.
|
||||
// Hoisting them into body.system would insert volatile content (token counters,
|
||||
// reminders) ahead of the whole conversation and invalidate the prefix cache on
|
||||
// every request. Folding in place keeps the cached prefix stable.
|
||||
if (Array.isArray(body.messages)) {
|
||||
const systemBlocks = [];
|
||||
const messages = [];
|
||||
for (const msg of body.messages) {
|
||||
if (msg.role === ROLE.SYSTEM) {
|
||||
const text = typeof msg.content === "string"
|
||||
? msg.content
|
||||
: Array.isArray(msg.content)
|
||||
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
|
||||
: "";
|
||||
if (text.trim()) systemBlocks.push({ type: CLAUDE_BLOCK.TEXT, text });
|
||||
if (msg.role !== ROLE.SYSTEM) {
|
||||
messages.push(msg);
|
||||
continue;
|
||||
}
|
||||
messages.push(msg);
|
||||
}
|
||||
const text = typeof msg.content === "string"
|
||||
? msg.content
|
||||
: Array.isArray(msg.content)
|
||||
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
|
||||
: "";
|
||||
if (!text.trim()) continue;
|
||||
|
||||
if (systemBlocks.length > 0) {
|
||||
const existing = Array.isArray(body.system)
|
||||
? body.system
|
||||
: typeof body.system === "string" && body.system.trim()
|
||||
? [{ type: "text", text: body.system }]
|
||||
: [];
|
||||
body.system = [...existing, ...systemBlocks];
|
||||
body.messages = messages;
|
||||
// Copy-on-write: the caller's body is reused across account-fallback
|
||||
// attempts, so folding must never mutate the original message.
|
||||
const block = { type: CLAUDE_BLOCK.TEXT, text };
|
||||
const prev = messages[messages.length - 1];
|
||||
if (prev?.role === ROLE.USER) {
|
||||
const content = typeof prev.content === "string"
|
||||
? [{ type: CLAUDE_BLOCK.TEXT, text: prev.content }]
|
||||
: Array.isArray(prev.content) ? [...prev.content] : [];
|
||||
messages[messages.length - 1] = { ...prev, content: [...content, block] };
|
||||
continue;
|
||||
}
|
||||
messages.push({ role: ROLE.USER, content: [block] });
|
||||
}
|
||||
body.messages = messages;
|
||||
}
|
||||
|
||||
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
|
||||
@@ -182,6 +191,70 @@ export function normalizeClaudePassthrough(body, model = "") {
|
||||
return body;
|
||||
}
|
||||
|
||||
// Put a 5m breakpoint on the last cache-eligible block of a message.
|
||||
// thinking/redacted_thinking blocks do not accept cache_control.
|
||||
function markLastCacheableBlock(msg) {
|
||||
if (!Array.isArray(msg?.content)) return false;
|
||||
for (let i = msg.content.length - 1; i >= 0; i--) {
|
||||
const block = msg.content[i];
|
||||
if (typeof block !== "object" || block === null) continue;
|
||||
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
|
||||
block.cache_control = { ...CACHE_CONTROL_5M };
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Re-anchor cache breakpoints on a Claude passthrough body (same policy as
|
||||
// prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m.
|
||||
// The client's own markers point at pre-normalization offsets, so they are dropped.
|
||||
// Must run LAST, after every step that can reshape system/tools/messages
|
||||
// (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail.
|
||||
export function anchorClaudeCache(body) {
|
||||
if (!body || typeof body !== "object") return body;
|
||||
|
||||
if (Array.isArray(body.system)) {
|
||||
const last = body.system.length - 1;
|
||||
body.system.forEach((block, i) => {
|
||||
if (typeof block !== "object" || block === null) return;
|
||||
if (i === last) block.cache_control = { ...CACHE_CONTROL_1H };
|
||||
else delete block.cache_control;
|
||||
});
|
||||
}
|
||||
|
||||
if (Array.isArray(body.tools)) {
|
||||
const last = body.tools.length - 1;
|
||||
body.tools.forEach((tool, i) => {
|
||||
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
|
||||
else delete tool.cache_control;
|
||||
});
|
||||
}
|
||||
|
||||
if (Array.isArray(body.messages)) {
|
||||
let anchored = null;
|
||||
for (let i = body.messages.length - 1; i >= 0; i--) {
|
||||
const msg = body.messages[i];
|
||||
if (!Array.isArray(msg.content)) continue;
|
||||
for (const block of msg.content) delete block.cache_control;
|
||||
|
||||
// Prefer the last assistant turn: it ends a completed exchange, so the
|
||||
// prefix up to it stays byte-stable across the following requests.
|
||||
if (anchored || msg.role !== ROLE.ASSISTANT) continue;
|
||||
anchored = markLastCacheableBlock(msg);
|
||||
}
|
||||
|
||||
// First turn of a conversation has no assistant yet — anchor the final
|
||||
// message instead, so the opening prompt is cached rather than paid twice.
|
||||
if (!anchored) {
|
||||
for (let i = body.messages.length - 1; i >= 0 && !anchored; i--) {
|
||||
anchored = markLastCacheableBlock(body.messages[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
// Prepare request for Claude format endpoints
|
||||
// - Cleanup cache_control
|
||||
// - Filter empty messages
|
||||
|
||||
@@ -287,6 +287,18 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
|
||||
toolSpecs,
|
||||
nameMap,
|
||||
});
|
||||
// canonicalizeKiroConversation() already ran its second-chance repair (flatten
|
||||
// every structured tool turn to text, then re-validate). A body that is STILL
|
||||
// invalid here cannot be made shippable, and Kiro answers it with
|
||||
// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"}.
|
||||
// Fail locally instead: chatCore turns a falsy return into a 400 without
|
||||
// spending an upstream call or a per-account cooldown. The taxonomy
|
||||
// (role:N | pair:N | id:N | spec:N | orphan:0 | current) names the offending
|
||||
// turn so the shape can be diagnosed from the log alone.
|
||||
if (!canonical.valid) {
|
||||
console.error(`[Kiro] refusing invalid conversation (claude → kiro): ${(canonical.errors || []).join(", ") || "unknown"} | turns=${(canonical.history || []).length + 1}`);
|
||||
return null;
|
||||
}
|
||||
const replayCurrent = canonical.currentMessage.userInputMessage;
|
||||
const userInputMessage = {
|
||||
content: replayCurrent.content || "",
|
||||
|
||||
@@ -421,6 +421,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
|
||||
if (body.reasoning !== undefined) result.reasoning = body.reasoning;
|
||||
if (body.reasoning_effort !== undefined) result.reasoning = { effort: body.reasoning_effort, summary: "auto" };
|
||||
if (body.service_tier !== undefined) result.service_tier = body.service_tier;
|
||||
if (body.prompt_cache_key !== undefined) result.prompt_cache_key = body.prompt_cache_key;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -379,6 +379,18 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
|
||||
toolSpecs,
|
||||
nameMap,
|
||||
});
|
||||
// canonicalizeKiroConversation() already ran its second-chance repair (flatten
|
||||
// every structured tool turn to text, then re-validate). A body that is STILL
|
||||
// invalid here cannot be made shippable, and Kiro answers it with
|
||||
// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"}.
|
||||
// Fail locally instead: chatCore turns a falsy return into a 400 without
|
||||
// spending an upstream call or a per-account cooldown. The taxonomy
|
||||
// (role:N | pair:N | id:N | spec:N | orphan:0 | current) names the offending
|
||||
// turn so the shape can be diagnosed from the log alone.
|
||||
if (!canonical.valid) {
|
||||
console.error(`[Kiro] refusing invalid conversation (openai → kiro): ${(canonical.errors || []).join(", ") || "unknown"} | turns=${(canonical.history || []).length + 1}`);
|
||||
return null;
|
||||
}
|
||||
const replayCurrent = canonical.currentMessage.userInputMessage;
|
||||
|
||||
const payload = {
|
||||
|
||||
@@ -75,6 +75,15 @@ export function kiroToClaudeResponse(chunk, state) {
|
||||
? data.usage.completion_tokens
|
||||
: 0;
|
||||
state.usage = { input_tokens: promptTokens, output_tokens: outputTokens };
|
||||
// Claude clients read cache_read/cache_creation to price a turn and to size
|
||||
// their prompt cache. Both spellings are accepted because the Kiro executor
|
||||
// emits the Chat shape and passthrough responses use the nested details form.
|
||||
const cacheRead = data.usage.cache_read_input_tokens
|
||||
?? data.usage.prompt_tokens_details?.cached_tokens;
|
||||
const cacheCreation = data.usage.cache_creation_input_tokens
|
||||
?? data.usage.prompt_tokens_details?.cache_creation_tokens;
|
||||
if (typeof cacheRead === "number") state.usage.cache_read_input_tokens = cacheRead;
|
||||
if (typeof cacheCreation === "number") state.usage.cache_creation_input_tokens = cacheCreation;
|
||||
}
|
||||
|
||||
// First chunk → emit message_start.
|
||||
@@ -254,6 +263,13 @@ export function kiroToClaudeNonStreaming(data) {
|
||||
usage: {
|
||||
input_tokens: usage.prompt_tokens || 0,
|
||||
output_tokens: usage.completion_tokens || 0,
|
||||
// Same cache preservation as the streaming path above.
|
||||
...(typeof (usage.cache_read_input_tokens ?? usage.prompt_tokens_details?.cached_tokens) === "number"
|
||||
? { cache_read_input_tokens: usage.cache_read_input_tokens ?? usage.prompt_tokens_details.cached_tokens }
|
||||
: {}),
|
||||
...(typeof (usage.cache_creation_input_tokens ?? usage.prompt_tokens_details?.cache_creation_tokens) === "number"
|
||||
? { cache_creation_input_tokens: usage.cache_creation_input_tokens ?? usage.prompt_tokens_details.cache_creation_tokens }
|
||||
: {}),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
@@ -99,8 +99,8 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
||||
}
|
||||
}
|
||||
|
||||
// Handle tool_calls
|
||||
if (delta.tool_calls) {
|
||||
// Handle tool_calls (empty array is truthy; require a real call)
|
||||
if (delta.tool_calls && delta.tool_calls.length) {
|
||||
closeMessage(state, emit, idx);
|
||||
for (const tc of delta.tool_calls) {
|
||||
emitToolCall(state, emit, tc);
|
||||
|
||||
Reference in New Issue
Block a user