merge origin/master into gitea/new_feature
Bring local branch up to v0.5.35 while keeping xAI image/edit, SuperGrok quota tracking, per-provider timeouts, and pinned model-test actions.
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
import { platform, arch } from "os";
|
||||
import { PROVIDERS, PROVIDER_OAUTH } from "./providers.js";
|
||||
import { ANTIGRAVITY_IDE_USER_AGENT } from "../providers/shared.js";
|
||||
|
||||
// === Gemini CLI === derive từ registry gemini-cli.transport
|
||||
export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion;
|
||||
@@ -59,7 +60,7 @@ export function getPlatformEnum() {
|
||||
}
|
||||
|
||||
export function getPlatformUserAgent() {
|
||||
return `antigravity/1.104.0 ${platform()}/${arch()}`;
|
||||
return ANTIGRAVITY_IDE_USER_AGENT;
|
||||
}
|
||||
|
||||
export const CLIENT_METADATA = {
|
||||
@@ -129,7 +130,7 @@ export const AG_DEFAULT_TOOLS = new Set([
|
||||
|
||||
// Antigravity chat/stream headers
|
||||
export const ANTIGRAVITY_HEADERS = {
|
||||
"User-Agent": `antigravity/1.107.0 ${platform()}/${arch()}`
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT
|
||||
};
|
||||
|
||||
// Cloud Code Assist API
|
||||
|
||||
10
open-sse/config/grokCli.js
Normal file
10
open-sse/config/grokCli.js
Normal file
@@ -0,0 +1,10 @@
|
||||
export const GROK_CLI_VERSION = "0.2.99";
|
||||
export const GROK_CLI_MODEL = "grok-build";
|
||||
export const GROK_CLI_BASE_URL = "https://cli-chat-proxy.grok.com/v1";
|
||||
export const GROK_CLI_CLIENT_IDENTIFIER = "grok-shell";
|
||||
export const GROK_CLI_USER_AGENT = `grok-shell/${GROK_CLI_VERSION} (linux; x86_64)`;
|
||||
|
||||
export function supportsGrokCliReasoningEffort(model) {
|
||||
// ponytail: unknown models omit effort until live metadata reaches dispatch.
|
||||
return /^grok-4\.5(?:$|-)/.test(String(model || ""));
|
||||
}
|
||||
@@ -131,6 +131,50 @@ export function resolveKiroThinkingBudget(body, headers, model) {
|
||||
return null;
|
||||
}
|
||||
|
||||
export function extractKiroEffortLevel(body) {
|
||||
const effort =
|
||||
body?.output_config?.effort ??
|
||||
body?.reasoning_effort ??
|
||||
(typeof body?.reasoning === "object" ? body.reasoning?.effort : null);
|
||||
if (typeof effort !== "string") return null;
|
||||
const normalized = effort.toLowerCase();
|
||||
if (normalized === "none" || normalized === "off" || normalized === "disabled") return null;
|
||||
if (normalized === "xhigh" || normalized === "max") return "high";
|
||||
if (["low", "medium", "high"].includes(normalized)) return normalized;
|
||||
return null;
|
||||
}
|
||||
|
||||
export function buildKiroAdditionalModelRequestFields(body) {
|
||||
const effort = extractKiroEffortLevel(body);
|
||||
if (!effort) return undefined;
|
||||
// Mirrors Kiro CLI/KAS buildEffortRequestFields("output_config").
|
||||
return {
|
||||
thinking: { type: "adaptive", display: "summarized" },
|
||||
output_config: { effort },
|
||||
};
|
||||
}
|
||||
|
||||
export function supportsKiroAdditionalModelRequestFields(model) {
|
||||
if (typeof model !== "string") return false;
|
||||
const normalized = model.toLowerCase().replace(/-/g, ".");
|
||||
if (!normalized.includes("claude")) return false;
|
||||
const match = normalized.match(/(?:^|[/.])claude(?:[/.][a-z]+)*[/.](\d+)(?:[/.](\d+))?(?:[/.]|$)/);
|
||||
if (!match) return false;
|
||||
const [, majorText, minorText] = match;
|
||||
const major = Number(majorText);
|
||||
const minor = minorText === undefined ? null : Number(minorText);
|
||||
const dateSuffixMinor = minor !== null && minor >= 1000;
|
||||
// Kiro rejected additionalModelRequestFields on legacy 4.5 models in live smoke.
|
||||
// Default future Claude/Kiro models to supported so new model releases do not
|
||||
// need a code allowlist update.
|
||||
return !(major < 4 || (major === 4 && (minor === null || minor <= 5 || dateSuffixMinor)));
|
||||
}
|
||||
|
||||
export function buildKiroAdditionalModelRequestFieldsForModel(body, model) {
|
||||
if (!supportsKiroAdditionalModelRequestFields(model)) return undefined;
|
||||
return buildKiroAdditionalModelRequestFields(body);
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect whether an inbound request is asking for reasoning / thinking output.
|
||||
* Thin wrapper over resolveKiroThinkingBudget (single source of truth).
|
||||
|
||||
@@ -2,7 +2,7 @@ import { PROVIDERS } from "./providers.js";
|
||||
import REGISTRY from "../providers/registry/index.js";
|
||||
// PROVIDER_MODELS now built from providers/registry (transport + models co-located)
|
||||
import { PROVIDER_MODELS } from "../providers/index.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat } from "../providers/models/schema.js";
|
||||
import { modelQuotaFamily, modelStrip, modelTargetFormat, normalizeModelId } from "../providers/models/schema.js";
|
||||
import { CODEX_REVIEW_SUFFIX } from "../providers/models/helpers.js";
|
||||
|
||||
export { PROVIDER_MODELS };
|
||||
@@ -18,46 +18,69 @@ export function getDefaultModel(aliasOrId) {
|
||||
return models?.[0]?.id || null;
|
||||
}
|
||||
|
||||
// Providers whose registry uses dots in version numbers (e.g. "claude-sonnet-4.5").
|
||||
// For these, we tolerate clients sending dashes ("claude-sonnet-4-5") by normalizing
|
||||
// digit-hyphen-digit to digit-dot-digit before lookup. Other providers are left untouched.
|
||||
const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]);
|
||||
|
||||
// Find a registry entry by id. For Kiro models, tolerates dash/dot version separators
|
||||
// ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only.
|
||||
function findModel(models, modelId, aliasOrId) {
|
||||
if (!models) return undefined;
|
||||
const found = models.find(m => m.id === modelId);
|
||||
if (found) return found;
|
||||
if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined;
|
||||
const normalized = normalizeModelId(modelId);
|
||||
if (normalized === modelId) return undefined;
|
||||
return models.find(m => m.id === normalized);
|
||||
}
|
||||
|
||||
export function isValidModel(aliasOrId, modelId, passthroughProviders = new Set()) {
|
||||
if (passthroughProviders.has(aliasOrId)) return true;
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return false;
|
||||
return models.some(m => m.id === modelId);
|
||||
return !!findModel(models, modelId, aliasOrId);
|
||||
}
|
||||
|
||||
export function findModelName(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return modelId;
|
||||
const found = models.find(m => m.id === modelId);
|
||||
const found = findModel(models, modelId, aliasOrId);
|
||||
return found?.name || modelId;
|
||||
}
|
||||
|
||||
export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
return modelTargetFormat(models.find(m => m.id === modelId));
|
||||
return modelTargetFormat(findModel(models, modelId, aliasOrId));
|
||||
}
|
||||
|
||||
export function getModelType(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
if (!models) return null;
|
||||
const found = models.find(m => m.id === modelId);
|
||||
const found = findModel(models, modelId, aliasOrId);
|
||||
return found?.kind || found?.type || null;
|
||||
}
|
||||
|
||||
export function getModelUpstreamId(aliasOrId, modelId) {
|
||||
// Split off thinking suffix "(level)" so lookup hits the base id; re-append it to
|
||||
// the result so downstream applyThinking still sees the suffix (body.model is stripped separately).
|
||||
const sufMatch = typeof modelId === "string" ? modelId.match(/\([^()]+\)\s*$/) : null;
|
||||
const suffix = sufMatch ? sufMatch[0] : "";
|
||||
const baseId = suffix ? modelId.slice(0, sufMatch.index).trim() : modelId;
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
const found = models?.find(m => m.id === modelId);
|
||||
if (found?.upstreamModelId) return found.upstreamModelId;
|
||||
if (aliasOrId === "cx" && typeof modelId === "string" && modelId.endsWith(CODEX_REVIEW_SUFFIX)) {
|
||||
return modelId.slice(0, -CODEX_REVIEW_SUFFIX.length);
|
||||
const found = findModel(models, baseId, aliasOrId);
|
||||
if (found?.upstreamModelId) return found.upstreamModelId + suffix;
|
||||
if (found?.id) return found.id + suffix;
|
||||
if (aliasOrId === "cx" && typeof baseId === "string" && baseId.endsWith(CODEX_REVIEW_SUFFIX)) {
|
||||
return baseId.slice(0, -CODEX_REVIEW_SUFFIX.length) + suffix;
|
||||
}
|
||||
return modelId;
|
||||
return baseId + suffix;
|
||||
}
|
||||
|
||||
export function getModelQuotaFamily(aliasOrId, modelId) {
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
return modelQuotaFamily(models?.find(m => m.id === modelId));
|
||||
return modelQuotaFamily(findModel(models, modelId, aliasOrId));
|
||||
}
|
||||
|
||||
// OAuth short aliases — derived from registry `alias` (single source). everything else: alias = id.
|
||||
@@ -79,5 +102,5 @@ export function getModelsByProviderId(providerId) {
|
||||
// Get strip list for a model entry (explicit opt-in only)
|
||||
// Returns array of content types to strip, e.g. ["image", "audio"]
|
||||
export function getModelStrip(alias, modelId) {
|
||||
return modelStrip(PROVIDER_MODELS[alias]?.find(m => m.id === modelId));
|
||||
return modelStrip(findModel(PROVIDER_MODELS[alias], modelId, alias));
|
||||
}
|
||||
|
||||
@@ -39,6 +39,15 @@ function envMs(name, def) {
|
||||
return Number.isFinite(n) && n > 0 ? n : def;
|
||||
}
|
||||
|
||||
function envUrl(name, def) {
|
||||
const raw = process.env[name]?.trim();
|
||||
return raw || def;
|
||||
}
|
||||
|
||||
// SearXNG endpoint used by the unauthenticated web-search provider.
|
||||
// Configure this for a separate Docker service or remote SearXNG instance.
|
||||
export const SEARXNG_URL = envUrl("SEARXNG_URL", "http://localhost:8888/search");
|
||||
|
||||
// Inter-chunk stall timeout (once tokens are flowing). Generous headroom so
|
||||
// slow reasoning models aren't aborted mid-stream. Env: STREAM_STALL_TIMEOUT_MS.
|
||||
export const STREAM_STALL_TIMEOUT_MS = envMs("STREAM_STALL_TIMEOUT_MS", 360 * 1000);
|
||||
@@ -56,6 +65,8 @@ export const GEMINI_NATIVE_TTS_FETCH_TIMEOUT_MS = envMs("GEMINI_NATIVE_TTS_FETCH
|
||||
export const DEFAULT_MAX_TOKENS = 64000;
|
||||
export const DEFAULT_MIN_TOKENS = 32000;
|
||||
|
||||
export const TOKEN_SAVER_HEADER = "x-9router-token-saver";
|
||||
|
||||
// Retry config for 429 responses (legacy - kept for backward compatibility)
|
||||
export const RETRY_CONFIG = {
|
||||
maxAttempts: 2,
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, INTERNAL_REQUEST_HEADER, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
@@ -18,7 +18,8 @@ function sanitizeFunctionName(name) {
|
||||
|
||||
const MAX_RETRY_AFTER_MS = 10000;
|
||||
const ANTIGRAVITY_TRANSIENT_RETRY_MAX_MS = 15000;
|
||||
const MAX_ANTIGRAVITY_OUTPUT_TOKENS = 16384;
|
||||
const MAX_ANTIGRAVITY_OUTPUT_TOKENS = 64000;
|
||||
const ANTIGRAVITY_IDE_REQUEST_ID_RE = /^agent\/[^/]+\/\d+\/[^/]+\/\d+$/;
|
||||
|
||||
const ANTIGRAVITY_TRANSIENT_ERROR_PATTERNS = [
|
||||
/high\s+traffic/i,
|
||||
@@ -87,6 +88,27 @@ function parseImageConfig(model) {
|
||||
return config;
|
||||
}
|
||||
|
||||
function uuidFromSeed(seed) {
|
||||
const bytes = crypto.createHash("sha256").update(String(seed || "antigravity")).digest().subarray(0, 16);
|
||||
bytes[6] = (bytes[6] & 0x0f) | 0x50;
|
||||
bytes[8] = (bytes[8] & 0x3f) | 0x80;
|
||||
const hex = bytes.toString("hex");
|
||||
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}`;
|
||||
}
|
||||
|
||||
function buildIdeRequestId({ body, request, credentials, model, requestType }) {
|
||||
if (ANTIGRAVITY_IDE_REQUEST_ID_RE.test(body?.requestId || "")) {
|
||||
return body.requestId;
|
||||
}
|
||||
|
||||
const sessionId = request?.sessionId || body?.request?.sessionId || credentials?._clientSessionId || credentials?.connectionId || credentials?.email || "anonymous";
|
||||
const conversationId = uuidFromSeed(`antigravity:conversation:${sessionId}`);
|
||||
const trajectoryId = uuidFromSeed(`antigravity:trajectory:${sessionId}:${model}:${requestType}`);
|
||||
const contentCount = Array.isArray(request?.contents) ? request.contents.length : 1;
|
||||
const step = Math.max(1, contentCount * 2 - 1);
|
||||
return `agent/${conversationId}/${Date.now()}/${trajectoryId}/${step}`;
|
||||
}
|
||||
|
||||
export class AntigravityExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("antigravity", PROVIDERS.antigravity);
|
||||
@@ -104,14 +126,10 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
// sessionId comes from transformRequest output; base.execute runs transformRequest before
|
||||
// buildHeaders, so we read it from instance state cached there (fallback: explicit arg).
|
||||
buildHeaders(credentials, stream = true, sessionId = null) {
|
||||
const sid = sessionId || this._lastSessionId;
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${credentials.accessToken}`,
|
||||
"User-Agent": this.config.headers?.["User-Agent"] || ANTIGRAVITY_HEADERS["User-Agent"],
|
||||
[INTERNAL_REQUEST_HEADER.name]: INTERNAL_REQUEST_HEADER.value,
|
||||
...(sid && { "X-Machine-Session-Id": sid }),
|
||||
"Accept": stream ? "text/event-stream" : "application/json"
|
||||
};
|
||||
}
|
||||
|
||||
@@ -142,25 +160,26 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
});
|
||||
|
||||
this._lastSessionId = sessionId;
|
||||
const request = {
|
||||
contents,
|
||||
generationConfig: {
|
||||
temperature: 1.0,
|
||||
topP: 0.95,
|
||||
topK: 40,
|
||||
maxOutputTokens: 8192,
|
||||
imageConfig,
|
||||
},
|
||||
sessionId,
|
||||
// No tools, no systemInstruction, no safetySettings for image gen
|
||||
};
|
||||
|
||||
return {
|
||||
project: projectId,
|
||||
model: cleanModel,
|
||||
userAgent: "antigravity",
|
||||
requestType: "image_gen",
|
||||
requestId: `agent-${crypto.randomUUID()}`,
|
||||
request: {
|
||||
contents,
|
||||
generationConfig: {
|
||||
temperature: 1.0,
|
||||
topP: 0.95,
|
||||
topK: 40,
|
||||
maxOutputTokens: 8192,
|
||||
imageConfig,
|
||||
},
|
||||
sessionId,
|
||||
// No tools, no systemInstruction, no safetySettings for image gen
|
||||
},
|
||||
requestId: buildIdeRequestId({ body, request, credentials, model: cleanModel, requestType: "image_gen" }),
|
||||
request,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -248,7 +267,7 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
model: model,
|
||||
userAgent: "antigravity",
|
||||
requestType: "agent",
|
||||
requestId: `agent-${crypto.randomUUID()}`,
|
||||
requestId: buildIdeRequestId({ body, request: transformedRequest, credentials, model, requestType: "agent" }),
|
||||
request: transformedRequest
|
||||
};
|
||||
}
|
||||
|
||||
@@ -8,13 +8,21 @@ import {
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
// SSE error patterns inside 200-OK body that should trigger retry as if 503
|
||||
const CODEX_SSE_OVERLOADED_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
|
||||
const CODEX_SSE_PEEK_BYTES = 4096;
|
||||
// SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts.
|
||||
const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
|
||||
const CODEX_SSE_ACCOUNT_FALLBACK_PATTERNS = ["selected model is at capacity", "model_at_capacity"];
|
||||
const CODEX_SSE_USER_OUTPUT_PATTERNS = [
|
||||
"event: response.output_text.delta",
|
||||
"event: response.function_call_arguments.delta",
|
||||
'"type":"response.output_text.delta"',
|
||||
'"type":"response.function_call_arguments.delta"',
|
||||
];
|
||||
const CODEX_SSE_PEEK_BYTES = 256 * 1024;
|
||||
const CODEX_MODEL_CAPACITY_MESSAGE = "Selected model is at capacity. Please try a different model.";
|
||||
|
||||
// Server-generated item id prefixes that Codex /responses cannot resolve when store=false
|
||||
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
|
||||
@@ -116,6 +124,62 @@ function resolveCacheSessionId(body, credentials) {
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeReasoningEffort(value) {
|
||||
return value === "max" ? "xhigh" : value;
|
||||
}
|
||||
|
||||
function findNestedMessage(value, depth = 0) {
|
||||
if (!value || depth > 6 || typeof value === "string") return null;
|
||||
if (Array.isArray(value)) {
|
||||
for (const item of value) {
|
||||
const found = findNestedMessage(item, depth + 1);
|
||||
if (found) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
if (typeof value !== "object") return null;
|
||||
if (typeof value.message === "string" && value.message.trim()) return value.message;
|
||||
if (typeof value.error?.message === "string" && value.error.message.trim()) return value.error.message;
|
||||
if (typeof value.response?.error?.message === "string" && value.response.error.message.trim()) return value.response.error.message;
|
||||
for (const child of Object.values(value)) {
|
||||
const found = findNestedMessage(child, depth + 1);
|
||||
if (found) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractSseErrorMessage(text, fallback) {
|
||||
const exact = text?.match(/Selected model is at capacity\. Please try a different model\./i)?.[0];
|
||||
if (exact) return exact;
|
||||
|
||||
for (const line of String(text || "").split(/\r?\n/)) {
|
||||
if (!line.startsWith("data:")) continue;
|
||||
const data = line.slice(5).trim();
|
||||
if (!data || data === "[DONE]") continue;
|
||||
try {
|
||||
const message = findNestedMessage(JSON.parse(data));
|
||||
if (message) return message;
|
||||
} catch {
|
||||
// Ignore non-JSON SSE data lines.
|
||||
}
|
||||
}
|
||||
|
||||
return fallback || CODEX_MODEL_CAPACITY_MESSAGE;
|
||||
}
|
||||
|
||||
function codexSseErrorResponse(status, message) {
|
||||
return new Response(JSON.stringify({
|
||||
error: {
|
||||
message,
|
||||
type: status >= 500 ? "server_error" : "invalid_request_error",
|
||||
code: status === HTTP_STATUS.SERVICE_UNAVAILABLE ? "service_unavailable" : "upstream_error",
|
||||
}
|
||||
}), {
|
||||
status,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Codex Executor - handles OpenAI Codex API (Responses API format)
|
||||
* Automatically injects default instructions if missing
|
||||
@@ -135,10 +199,17 @@ export class CodexExecutor extends BaseExecutor {
|
||||
headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default";
|
||||
// Identify client type to Codex backend (matches official codex CLI)
|
||||
if (!headers["originator"]) headers["originator"] = "codex_cli_rs";
|
||||
// Workspace binding header — improves account scope + cache affinity
|
||||
const workspaceId = credentials?.providerSpecificData?.workspaceId;
|
||||
if (typeof workspaceId === "string" && workspaceId && !headers["chatgpt-account-id"]) {
|
||||
headers["chatgpt-account-id"] = workspaceId;
|
||||
// Account/workspace binding header — required when multiple Codex accounts
|
||||
// are configured. OAuth import stores ChatGPT account ID as chatgptAccountId;
|
||||
// older/custom rows may use workspaceId/accountId. Prefer explicit workspaceId
|
||||
// but fall back to chatgptAccountId so requests don't cross-bind to the wrong
|
||||
// OpenAI account and surface as token_invalid after adding another account.
|
||||
const accountId =
|
||||
credentials?.providerSpecificData?.workspaceId ||
|
||||
credentials?.providerSpecificData?.chatgptAccountId ||
|
||||
credentials?.providerSpecificData?.accountId;
|
||||
if (typeof accountId === "string" && accountId && !headers["ChatGPT-Account-ID"]) {
|
||||
headers["ChatGPT-Account-ID"] = accountId;
|
||||
}
|
||||
return headers;
|
||||
}
|
||||
@@ -198,7 +269,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
let attempt = 0;
|
||||
while (true) {
|
||||
const result = await super.execute(args);
|
||||
const peek = await this._peekSseOverloaded(result.response);
|
||||
const peek = await this._peekSseTransientError(result.response);
|
||||
if (!peek.matched) {
|
||||
// Replace body with re-assembled stream (prefix bytes already read + rest)
|
||||
if (peek.replacementBody) {
|
||||
@@ -210,48 +281,57 @@ export class CodexExecutor extends BaseExecutor {
|
||||
}
|
||||
return result;
|
||||
}
|
||||
if (peek.accountFallback) {
|
||||
args.log?.warn?.("RETRY", `CODEX | SSE account fallback "${peek.message}"`);
|
||||
result.response = codexSseErrorResponse(HTTP_STATUS.SERVICE_UNAVAILABLE, peek.message || CODEX_MODEL_CAPACITY_MESSAGE);
|
||||
return result;
|
||||
}
|
||||
if (attempt >= attempts) {
|
||||
args.log?.warn?.("RETRY", `CODEX | SSE overloaded "${peek.matched}" — retries exhausted (${attempt}/${attempts})`);
|
||||
// Out of retries → return with replacement body so client gets the error
|
||||
if (peek.replacementBody) {
|
||||
result.response = new Response(peek.replacementBody, {
|
||||
status: result.response.status,
|
||||
statusText: result.response.statusText,
|
||||
headers: result.response.headers,
|
||||
});
|
||||
}
|
||||
result.response = codexSseErrorResponse(HTTP_STATUS.SERVICE_UNAVAILABLE, peek.message || peek.matched);
|
||||
return result;
|
||||
}
|
||||
attempt++;
|
||||
args.log?.debug?.("RETRY", `CODEX | SSE "${peek.matched}" retry ${attempt}/${attempts} after ${delayMs / 1000}s`);
|
||||
dbg("CODEX", `SSE overloaded "${peek.matched}" → retry ${attempt}/${attempts} in ${delayMs}ms`);
|
||||
try { await result.response.body?.cancel?.(); } catch { /* noop */ }
|
||||
await new Promise(r => setTimeout(r, delayMs));
|
||||
}
|
||||
}
|
||||
|
||||
// Peek first N bytes of SSE body to detect upstream "overloaded" errors.
|
||||
// Returns { matched: string|null, replacementBody: ReadableStream|null }.
|
||||
// Caller MUST use replacementBody (original body has been read).
|
||||
async _peekSseOverloaded(response) {
|
||||
if (!response || !response.ok || !response.body) return { matched: null, replacementBody: null };
|
||||
// Peek first N bytes of SSE body to detect upstream transient errors.
|
||||
// Returns { matched: string|null, message: string|null, accountFallback: boolean, replacementBody: ReadableStream|null }.
|
||||
// Caller must use replacementBody when no error matched (original body has been read).
|
||||
async _peekSseTransientError(response) {
|
||||
if (!response || !response.ok || !response.body) return { matched: null, message: null, accountFallback: false, replacementBody: null };
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
const chunks = [];
|
||||
let text = "";
|
||||
let matched = null;
|
||||
let accountFallback = false;
|
||||
try {
|
||||
while (text.length < CODEX_SSE_PEEK_BYTES) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
chunks.push(value);
|
||||
text += decoder.decode(value, { stream: true });
|
||||
const hit = CODEX_SSE_OVERLOADED_PATTERNS.find(p => text.includes(p));
|
||||
if (hit) { matched = hit; break; }
|
||||
const lowerText = text.toLowerCase();
|
||||
const accountHit = CODEX_SSE_ACCOUNT_FALLBACK_PATTERNS.find(p => lowerText.includes(p));
|
||||
if (accountHit) { matched = accountHit; accountFallback = true; break; }
|
||||
const retryHit = CODEX_SSE_RETRY_PATTERNS.find(p => lowerText.includes(p));
|
||||
if (retryHit) { matched = retryHit; break; }
|
||||
if (CODEX_SSE_USER_OUTPUT_PATTERNS.some(p => lowerText.includes(p))) break;
|
||||
}
|
||||
} catch (e) {
|
||||
dbg("CODEX", `peek read error: ${e.message}`);
|
||||
}
|
||||
|
||||
if (matched) {
|
||||
try { await reader.cancel(); } catch { /* noop */ }
|
||||
try { reader.releaseLock(); } catch { /* noop */ }
|
||||
return { matched, message: extractSseErrorMessage(text, matched), accountFallback, replacementBody: null };
|
||||
}
|
||||
|
||||
reader.releaseLock();
|
||||
|
||||
// Re-assemble stream: prefix chunks + remaining upstream body
|
||||
@@ -273,7 +353,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
try { upstreamReader?.cancel(reason); } catch { /* noop */ }
|
||||
},
|
||||
});
|
||||
return { matched, replacementBody };
|
||||
return { matched: null, message: null, accountFallback: false, replacementBody };
|
||||
}
|
||||
|
||||
// Parse Codex usage_limit_reached to extract precise resetsAtMs; fallback to default otherwise
|
||||
@@ -347,7 +427,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Extract thinking level from model name suffix
|
||||
// e.g., gpt-5.3-codex-high → high, gpt-5.3-codex → medium (default)
|
||||
const effortLevels = ['none', 'low', 'medium', 'high', 'xhigh'];
|
||||
const effortLevels = ['none', 'minimal', 'low', 'medium', 'high', 'xhigh'];
|
||||
let modelEffort = null;
|
||||
for (const level of effortLevels) {
|
||||
if (body.model.endsWith(`-${level}`)) {
|
||||
@@ -360,10 +440,11 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = body.reasoning_effort || modelEffort || 'low';
|
||||
const effort = normalizeReasoningEffort(body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
} else if (!body.reasoning.summary) {
|
||||
body.reasoning.summary = "auto";
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
@@ -391,6 +472,9 @@ export class CodexExecutor extends BaseExecutor {
|
||||
delete body.safety_identifier; // Droid CLI sends this but Codex doesn't support it
|
||||
delete body.previous_response_id; // store=false → backend can't resolve previous resp; avoid 404
|
||||
|
||||
if (body.service_tier === "fast") body.service_tier = "priority";
|
||||
if (body.service_tier && body.service_tier !== "priority") delete body.service_tier;
|
||||
|
||||
// Final allowlist filter — strip any unknown field that could trigger upstream "routing_unsupported"
|
||||
for (const k of Object.keys(body)) {
|
||||
if (!RESPONSES_API_ALLOWLIST.has(k)) delete body[k];
|
||||
|
||||
@@ -4,11 +4,13 @@ import { OAUTH_ENDPOINTS, GITHUB_COPILOT } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { openaiToOpenAIResponsesRequest } from "../translator/request/openai-responses.js";
|
||||
import { openaiResponsesToOpenAIResponse } from "../translator/response/openai-responses.js";
|
||||
import { initState } from "../translator/index.js";
|
||||
import { initState, translateRequest, translateResponse } from "../translator/index.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { parseSSELine, formatSSE } from "../utils/streamHelpers.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
import crypto from "crypto";
|
||||
|
||||
export class GithubExecutor extends BaseExecutor {
|
||||
@@ -17,6 +19,16 @@ export class GithubExecutor extends BaseExecutor {
|
||||
this.knownCodexModels = new Set();
|
||||
}
|
||||
|
||||
// Claude models get routed to Copilot's Anthropic-native /v1/messages shim (see
|
||||
// executeWithMessagesEndpoint below) — the only Copilot endpoint that surfaces
|
||||
// prompt-cache token counts. gpt/gemini/grok models stay on /chat/completions
|
||||
// (or /responses). Name-pattern check, not a registry field: Copilot's live model
|
||||
// catalog (services/copilotModels.js) regularly exposes claude-* variants ahead
|
||||
// of the static registry (registry/github.js).
|
||||
isClaudeModel(model) {
|
||||
return /claude/i.test(model || "");
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0) {
|
||||
return this.config.baseUrl;
|
||||
}
|
||||
@@ -35,47 +47,20 @@ export class GithubExecutor extends BaseExecutor {
|
||||
"x-request-id": crypto.randomUUID?.() || `${Date.now()}-${Math.random().toString(36).slice(2)}`,
|
||||
"x-vscode-user-agent-library-version": "electron-fetch",
|
||||
"X-Initiator": "user",
|
||||
// Harmless no-op on /chat/completions and /responses; required by /v1/messages.
|
||||
"anthropic-version": ANTHROPIC_API_VERSION,
|
||||
"Accept": stream ? "text/event-stream" : "application/json"
|
||||
};
|
||||
}
|
||||
|
||||
// Sanitize messages for GitHub Copilot /chat/completions endpoint.
|
||||
// Sanitize messages for GitHub Copilot /chat/completions endpoint (gpt/gemini/grok models —
|
||||
// claude models never reach this, see execute() below).
|
||||
// The endpoint only accepts 'text' and 'image_url' content part types.
|
||||
// Tool-related content (tool_use, tool_result, thinking) must be serialized as text.
|
||||
sanitizeMessagesForChatCompletions(body) {
|
||||
if (!body?.messages) return body;
|
||||
|
||||
const sanitized = { ...body };
|
||||
|
||||
// Handle response_format for Claude models via GitHub
|
||||
// GitHub's internal translation doesn't respect response_format, so we inject it as a system prompt
|
||||
// AND prepend a reminder to the last user message for maximum effectiveness
|
||||
if (body.response_format && body.model?.includes('claude')) {
|
||||
const responseFormat = body.response_format;
|
||||
let systemInstruction = '';
|
||||
if (responseFormat.type === 'json_schema' && responseFormat.json_schema?.schema) {
|
||||
systemInstruction = 'CRITICAL: You must ONLY output raw JSON. Never use markdown code blocks. Never use backticks. Never wrap JSON in triple backticks. Output ONLY the raw JSON object.';
|
||||
} else if (responseFormat.type === 'json_object') {
|
||||
systemInstruction = 'CRITICAL: You must ONLY output raw JSON. Never use markdown code blocks. Never use backticks.';
|
||||
}
|
||||
if (systemInstruction) {
|
||||
// Add to system message
|
||||
const systemIdx = body.messages.findIndex(m => m.role === 'system');
|
||||
if (systemIdx >= 0) {
|
||||
body.messages[systemIdx].content = systemInstruction + '\n\n' + body.messages[systemIdx].content;
|
||||
} else {
|
||||
body.messages.unshift({ role: 'system', content: systemInstruction });
|
||||
}
|
||||
|
||||
// Also prepend to the last user message as a reminder
|
||||
const lastUserIdx = body.messages.map((m, i) => m.role === 'user' ? i : -1).filter(i => i >= 0).pop();
|
||||
if (lastUserIdx >= 0) {
|
||||
const userMsg = body.messages[lastUserIdx];
|
||||
const userContent = typeof userMsg.content === 'string' ? userMsg.content : JSON.stringify(userMsg.content);
|
||||
userMsg.content = 'Respond with ONLY raw JSON (no markdown, no backticks, no code blocks): ' + userContent;
|
||||
}
|
||||
}
|
||||
}
|
||||
sanitized.messages = body.messages.map(msg => {
|
||||
// assistant messages with only tool_calls have content: null — leave as-is
|
||||
if (!msg.content) return msg;
|
||||
@@ -138,6 +123,15 @@ export class GithubExecutor extends BaseExecutor {
|
||||
async execute(options) {
|
||||
const { model, log } = options;
|
||||
|
||||
// Claude models: route to Copilot's Anthropic-native /v1/messages shim — the only
|
||||
// Copilot endpoint that surfaces prompt-cache token counts for Claude. Detected by
|
||||
// model NAME (not a registry field): Copilot's live model catalog regularly exposes
|
||||
// claude-* variants the static registry hasn't caught up with yet (see registry/github.js).
|
||||
if (this.isClaudeModel(model)) {
|
||||
log?.debug("GITHUB", `Using /v1/messages route for ${model}`);
|
||||
return this.executeWithMessagesEndpoint(options);
|
||||
}
|
||||
|
||||
// Only use /responses for models that are explicitly known to need it (e.g. gpt codex models)
|
||||
// and that the /responses endpoint actually serves (excludes Gemini/Claude, see #1062).
|
||||
if (this.knownCodexModels.has(model) && this.supportsResponsesEndpoint(model)) {
|
||||
@@ -145,8 +139,8 @@ export class GithubExecutor extends BaseExecutor {
|
||||
return this.executeWithResponsesEndpoint(options);
|
||||
}
|
||||
|
||||
// Sanitize messages before sending to /chat/completions
|
||||
// This handles Claude models on GitHub Copilot which reject non-text/image_url content types
|
||||
// Sanitize messages before sending to /chat/completions (gpt/gemini/grok — the
|
||||
// endpoint rejects non-text/image_url content parts).
|
||||
const sanitizedOptions = {
|
||||
...options,
|
||||
body: this.sanitizeMessagesForChatCompletions(options.body)
|
||||
@@ -251,6 +245,101 @@ export class GithubExecutor extends BaseExecutor {
|
||||
};
|
||||
}
|
||||
|
||||
// Claude models arrive here OpenAI-shape (chatCore.js targets "openai" for github —
|
||||
// see the note in execute() above), so we translate to Anthropic-native ourselves.
|
||||
// This is what makes prepareClaudeRequest() (translator/formats/claude.js) inject
|
||||
// cache_control — /chat/completions never gets there, so it never sees cache tokens.
|
||||
async executeWithMessagesEndpoint({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
|
||||
const url = this.config.messagesUrl;
|
||||
const headers = this.buildHeaders(credentials, stream);
|
||||
|
||||
// Force stream:true upstream regardless of client preference, same as
|
||||
// executeWithResponsesEndpoint below — chatCore.js's non-streaming handler already
|
||||
// knows how to buffer an SSE response into a single JSON reply when the client
|
||||
// asked for stream:false.
|
||||
const transformedBody = translateRequest(FORMATS.OPENAI, FORMATS.CLAUDE, model, body, true, credentials, "github");
|
||||
// _toolNameMap is internal bookkeeping (see openai-to-claude.js) — chatCore.js
|
||||
// normally strips it before dispatch and threads it into the response state to
|
||||
// restore original tool names; we must do the same here, or Anthropic's strict
|
||||
// schema rejects the extra field with a 400.
|
||||
const toolNameMap = transformedBody._toolNameMap;
|
||||
delete transformedBody._toolNameMap;
|
||||
|
||||
log?.debug("GITHUB", "Sending translated request to /v1/messages");
|
||||
|
||||
const response = await proxyAwareFetch(url, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify(transformedBody),
|
||||
signal
|
||||
}, proxyOptions);
|
||||
|
||||
if (!response.ok) {
|
||||
return { response, url, headers, transformedBody };
|
||||
}
|
||||
|
||||
const state = initState(FORMATS.CLAUDE);
|
||||
state.model = model;
|
||||
if (toolNameMap) state.toolNameMap = toolNameMap;
|
||||
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
|
||||
const emitAll = (controller, chunks) => {
|
||||
for (const c of chunks) {
|
||||
controller.enqueue(new TextEncoder().encode(formatSSE(c, "openai")));
|
||||
}
|
||||
};
|
||||
|
||||
const transformStream = new TransformStream({
|
||||
async transform(chunk, controller) {
|
||||
buffer += decoder.decode(chunk, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
|
||||
buffer = lines.pop() || "";
|
||||
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
|
||||
const parsed = parseSSELine(trimmed);
|
||||
if (!parsed) continue;
|
||||
|
||||
if (parsed.done && stream === true) {
|
||||
controller.enqueue(new TextEncoder().encode(SSE_DONE));
|
||||
continue;
|
||||
}
|
||||
|
||||
emitAll(controller, translateResponse(FORMATS.CLAUDE, FORMATS.OPENAI, parsed, state));
|
||||
}
|
||||
},
|
||||
flush(controller) {
|
||||
if (buffer.trim()) {
|
||||
const parsed = parseSSELine(buffer.trim());
|
||||
if (parsed && !parsed.done) {
|
||||
emitAll(controller, translateResponse(FORMATS.CLAUDE, FORMATS.OPENAI, parsed, state));
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
if (!response.body) {
|
||||
return { response: new Response("", { status: response.status, headers: response.headers }), url, headers, transformedBody };
|
||||
}
|
||||
const convertedStream = response.body.pipeThrough(transformStream);
|
||||
|
||||
return {
|
||||
response: new Response(convertedStream, {
|
||||
status: response.status,
|
||||
statusText: response.statusText,
|
||||
headers: response.headers
|
||||
}),
|
||||
url,
|
||||
headers,
|
||||
transformedBody
|
||||
};
|
||||
}
|
||||
|
||||
async refreshCopilotToken(githubAccessToken, log, proxyOptions = null) {
|
||||
try {
|
||||
const response = await proxyAwareFetch("https://api.github.com/copilot_internal/v2/token", {
|
||||
|
||||
552
open-sse/executors/grok-cli.js
Normal file
552
open-sse/executors/grok-cli.js
Normal file
@@ -0,0 +1,552 @@
|
||||
import crypto from "node:crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import {
|
||||
refreshProviderCredentials,
|
||||
shouldRefreshCredentials,
|
||||
} from "../services/oauthCredentialManager.js";
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import {
|
||||
GROK_CLI_CLIENT_IDENTIFIER,
|
||||
GROK_CLI_VERSION,
|
||||
supportsGrokCliReasoningEffort,
|
||||
} from "../config/grokCli.js";
|
||||
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { getConsistentMachineId } from "../shared/machineId.js";
|
||||
|
||||
// Server-generated item id prefixes that /responses cannot resolve when store=false
|
||||
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
|
||||
|
||||
// Hosted tool types executed server-side by Grok CLI backend
|
||||
const HOSTED_TOOL_TYPES = new Set([
|
||||
"web_search",
|
||||
"x_search",
|
||||
"web_search_preview",
|
||||
"file_search",
|
||||
"image_generation",
|
||||
"code_interpreter",
|
||||
"mcp",
|
||||
"local_shell",
|
||||
]);
|
||||
|
||||
// Fields accepted by cli-chat-proxy Responses API (mirrors Codex allowlist + Grok extras)
|
||||
const RESPONSES_API_ALLOWLIST = new Set([
|
||||
"model",
|
||||
"input",
|
||||
"instructions",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"stream",
|
||||
"store",
|
||||
"reasoning",
|
||||
"include",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"max_output_tokens",
|
||||
"parallel_tool_calls",
|
||||
"text",
|
||||
"metadata",
|
||||
"prompt_cache_key",
|
||||
]);
|
||||
|
||||
const EFFORT_LEVELS = ["low", "medium", "high", "xhigh"];
|
||||
const GROK_CLI_TURN_STORE_MAX = 5000;
|
||||
const GROK_CLI_NATIVE_ITEM_ID = /^(?:rs|msg|fc)_[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
|
||||
const GROK_CLI_FREEFORM_TOOL_PARAMETERS = {
|
||||
type: "object",
|
||||
properties: { input: { type: "string" } },
|
||||
required: ["input"],
|
||||
};
|
||||
|
||||
// Per-session last turn index so multi-turn headers never go backwards within this process
|
||||
const sessionTurnStore = new Map();
|
||||
let requestTurnStore = new WeakMap();
|
||||
|
||||
/**
|
||||
* Count user turns in a Responses `input` array.
|
||||
* Official CLI sets x-grok-turn-idx to the 1-based conversation turn (≈ user messages).
|
||||
* HAR: first chat turn → "1".
|
||||
*/
|
||||
export function countGrokCliUserTurns(input) {
|
||||
if (!Array.isArray(input)) return 1;
|
||||
let n = 0;
|
||||
for (const item of input) {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) continue;
|
||||
const type = typeof item.type === "string" ? item.type : "";
|
||||
// Responses message items (type omitted or "message") with role user
|
||||
if (item.role === "user" && (!type || type === "message")) n += 1;
|
||||
}
|
||||
return Math.max(1, n);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve monotonic turn index for a session.
|
||||
* Prefers user-message count from the payload (full history clients), but never
|
||||
* decreases vs the last index observed for the same sessionId in this process.
|
||||
*/
|
||||
export function resolveGrokCliTurnIdx(sessionId, input, requestKey = null) {
|
||||
const fromInput = countGrokCliUserTurns(input);
|
||||
if (!sessionId) return fromInput;
|
||||
|
||||
if (requestKey && requestTurnStore.has(requestKey)) {
|
||||
return requestTurnStore.get(requestKey);
|
||||
}
|
||||
|
||||
const now = Date.now();
|
||||
const existing = sessionTurnStore.get(sessionId);
|
||||
const prev = existing && now - existing.lastUsed <= MEMORY_CONFIG.sessionTtlMs
|
||||
? existing.turn
|
||||
: 0;
|
||||
if (existing) sessionTurnStore.delete(sessionId);
|
||||
|
||||
// A new delta-style request advances the turn; retries reuse requestKey.
|
||||
const turn = prev > 0 ? Math.max(fromInput, prev + (requestKey ? 1 : 0)) : fromInput;
|
||||
while (sessionTurnStore.size >= GROK_CLI_TURN_STORE_MAX) {
|
||||
sessionTurnStore.delete(sessionTurnStore.keys().next().value);
|
||||
}
|
||||
sessionTurnStore.set(sessionId, { turn, lastUsed: now });
|
||||
if (requestKey) requestTurnStore.set(requestKey, turn);
|
||||
return turn;
|
||||
}
|
||||
|
||||
/** Test helper — clear in-memory turn counters */
|
||||
export function _resetGrokCliTurnStore() {
|
||||
sessionTurnStore.clear();
|
||||
requestTurnStore = new WeakMap();
|
||||
}
|
||||
|
||||
export function _getGrokCliTurnStoreSize() {
|
||||
return sessionTurnStore.size;
|
||||
}
|
||||
|
||||
export function normalizeGrokCliEffort(value) {
|
||||
const effort = typeof value === "string" ? value.trim().toLowerCase() : "";
|
||||
if (effort === "max") return "xhigh";
|
||||
if (EFFORT_LEVELS.includes(effort)) return effort;
|
||||
return "high";
|
||||
}
|
||||
|
||||
export { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
||||
|
||||
export function resolveGrokCliSessionId(credentials, body) {
|
||||
// ponytail: clients without stable thread metadata share one connection session;
|
||||
// split further when their wire format exposes a durable conversation id.
|
||||
const explicitSessionBody = {
|
||||
prompt_cache_key: body?.prompt_cache_key,
|
||||
session_id: body?.session_id,
|
||||
conversation_id: body?.conversation_id,
|
||||
metadata: body?.metadata,
|
||||
};
|
||||
return resolveSessionId({
|
||||
headers: credentials?.rawHeaders,
|
||||
body: explicitSessionBody,
|
||||
connectionId: credentials?.connectionId || credentials?.id,
|
||||
workspaceId: credentials?.providerSpecificData?.workspaceId,
|
||||
scope: "grok-cli",
|
||||
});
|
||||
}
|
||||
|
||||
function stringifyGrokCliToolOutput(output) {
|
||||
if (typeof output === "string") return output;
|
||||
if (output === undefined) return "";
|
||||
return JSON.stringify(output);
|
||||
}
|
||||
|
||||
function isNativeGrokCliItemId(id) {
|
||||
return typeof id === "string" && GROK_CLI_NATIVE_ITEM_ID.test(id);
|
||||
}
|
||||
|
||||
function normalizeGrokCliInputItem(item) {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return item;
|
||||
const { internal_chat_message_metadata_passthrough: _metadata, ...clean } = item;
|
||||
|
||||
if (item.type === "reasoning") {
|
||||
if (!isNativeGrokCliItemId(item.id) || typeof item.encrypted_content !== "string") return null;
|
||||
return clean;
|
||||
}
|
||||
|
||||
if (item.type === "custom_tool_call") {
|
||||
const callId = item.call_id || item.id;
|
||||
const name = typeof item.name === "string" ? item.name.trim() : "";
|
||||
if (!callId || !name) return null;
|
||||
return {
|
||||
type: "function_call",
|
||||
call_id: callId,
|
||||
name,
|
||||
arguments: JSON.stringify({ input: stringifyGrokCliToolOutput(item.input ?? item.arguments) }),
|
||||
};
|
||||
}
|
||||
|
||||
if (item.type === "custom_tool_call_output" || item.type === "function_call_output") {
|
||||
const callId = item.call_id || item.id;
|
||||
if (!callId) return null;
|
||||
return {
|
||||
type: "function_call_output",
|
||||
call_id: callId,
|
||||
output: stringifyGrokCliToolOutput(item.output),
|
||||
};
|
||||
}
|
||||
|
||||
if (item.type === "function_call") {
|
||||
const callId = item.call_id || item.id;
|
||||
const name = typeof item.name === "string" ? item.name.trim() : "";
|
||||
if (!callId || !name) return null;
|
||||
return {
|
||||
type: "function_call",
|
||||
...(isNativeGrokCliItemId(item.id) ? { id: item.id } : {}),
|
||||
call_id: callId,
|
||||
name,
|
||||
arguments: typeof item.arguments === "string" ? item.arguments : JSON.stringify(item.arguments ?? {}),
|
||||
...(typeof item.status === "string" ? { status: item.status } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
return clean;
|
||||
}
|
||||
|
||||
export function normalizeGrokCliInput(body) {
|
||||
if (!Array.isArray(body?.input)) return body;
|
||||
const normalized = body.input.map(normalizeGrokCliInputItem).filter(Boolean);
|
||||
const callIds = new Set(
|
||||
normalized
|
||||
.filter((item) => item?.type === "function_call" && item.call_id)
|
||||
.map((item) => item.call_id)
|
||||
);
|
||||
body.input = normalized.filter(
|
||||
(item) => item?.type !== "function_call_output" || callIds.has(item.call_id)
|
||||
);
|
||||
return body;
|
||||
}
|
||||
|
||||
function stripStoredItemReferences(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (typeof item === "string" && SERVER_ID_PATTERN.test(item)) return false;
|
||||
if (item && typeof item === "object" && !Array.isArray(item)) {
|
||||
if (item.type === "item_reference") return false;
|
||||
if (
|
||||
typeof item.id === "string" &&
|
||||
SERVER_ID_PATTERN.test(item.id) &&
|
||||
!isNativeGrokCliItemId(item.id)
|
||||
) delete item.id;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Flatten Chat Completions tool shape → Responses flat format.
|
||||
* Keep hosted tools (web_search / x_search) passthrough.
|
||||
*/
|
||||
function normalizeGrokCliTools(body) {
|
||||
if (!Array.isArray(body.tools) || body.tools.length === 0) {
|
||||
delete body.tools;
|
||||
delete body.tool_choice;
|
||||
return;
|
||||
}
|
||||
const validNames = new Set();
|
||||
const hostedTypes = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const type = typeof tool.type === "string" ? tool.type : "";
|
||||
|
||||
if (type !== "function") {
|
||||
// Hosted tools: { type: "web_search" } / { type: "x_search" }
|
||||
if (HOSTED_TOOL_TYPES.has(type)) {
|
||||
hostedTypes.add(type);
|
||||
return true;
|
||||
}
|
||||
// Nested function shape without type
|
||||
if (!type && tool.function) {
|
||||
// fall through to function flatten below
|
||||
} else if (!type || typeof tool.name === "string") {
|
||||
// treat as bare function if name present
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
const isFunction =
|
||||
type === "function" || type === "" || tool.function || typeof tool.name === "string";
|
||||
if (!isFunction || HOSTED_TOOL_TYPES.has(type)) {
|
||||
return HOSTED_TOOL_TYPES.has(type);
|
||||
}
|
||||
|
||||
const fn =
|
||||
tool.function && typeof tool.function === "object" && !Array.isArray(tool.function)
|
||||
? tool.function
|
||||
: null;
|
||||
const rawName =
|
||||
typeof tool.name === "string" ? tool.name : typeof fn?.name === "string" ? fn.name : "";
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
|
||||
const description =
|
||||
typeof tool.description === "string"
|
||||
? tool.description
|
||||
: typeof fn?.description === "string"
|
||||
? fn.description
|
||||
: "";
|
||||
const parameters = type === "custom"
|
||||
? GROK_CLI_FREEFORM_TOOL_PARAMETERS
|
||||
: tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)
|
||||
? tool.parameters
|
||||
: fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters)
|
||||
? fn.parameters
|
||||
: { type: "object", properties: {} };
|
||||
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, 128);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
|
||||
if (body.tools.length === 0) {
|
||||
delete body.tools;
|
||||
delete body.tool_choice;
|
||||
return;
|
||||
}
|
||||
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
const choiceType = typeof body.tool_choice.type === "string" ? body.tool_choice.type : "";
|
||||
if (choiceType === "function" || choiceType === "custom") {
|
||||
const rawName = body.tool_choice.name ?? body.tool_choice.function?.name;
|
||||
const name = typeof rawName === "string" ? rawName.trim().slice(0, 128) : "";
|
||||
if (!name || !validNames.has(name)) delete body.tool_choice;
|
||||
else body.tool_choice = { type: "function", name };
|
||||
} else if (!hostedTypes.has(choiceType)) {
|
||||
delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function resolveEffortFromModel(modelId) {
|
||||
if (!modelId || typeof modelId !== "string") return null;
|
||||
for (const level of EFFORT_LEVELS) {
|
||||
if (modelId.endsWith(`-${level}`)) return level;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Grok CLI Executor — OpenAI Responses API on cli-chat-proxy.grok.com
|
||||
* Auth: OAuth device-code access token (xai-grok-cli).
|
||||
*/
|
||||
export class GrokCliExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("grok-cli", PROVIDERS["grok-cli"]);
|
||||
this._currentSessionId = null;
|
||||
this._currentReqId = null;
|
||||
this._currentTurnIdx = 1;
|
||||
this._agentId = null;
|
||||
}
|
||||
|
||||
buildUrl() {
|
||||
return this.config.baseUrl;
|
||||
}
|
||||
|
||||
async refreshCredentials(credentials, log, proxyOptions = null) {
|
||||
if (!credentials?.refreshToken) return null;
|
||||
return refreshProviderCredentials("grok-cli", credentials, log, proxyOptions);
|
||||
}
|
||||
|
||||
needsRefresh(credentials) {
|
||||
return shouldRefreshCredentials("grok-cli", credentials);
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const headers = super.buildHeaders(credentials, stream);
|
||||
|
||||
// Static fingerprint from registry
|
||||
const staticHeaders = this.config.headers || {};
|
||||
for (const [k, v] of Object.entries(staticHeaders)) {
|
||||
if (v != null && headers[k] === undefined) headers[k] = v;
|
||||
}
|
||||
|
||||
headers["x-grok-client-identifier"] =
|
||||
this.config.clientIdentifier || headers["x-grok-client-identifier"] || GROK_CLI_CLIENT_IDENTIFIER;
|
||||
headers["x-grok-client-version"] =
|
||||
this.config.clientVersion || headers["x-grok-client-version"] || GROK_CLI_VERSION;
|
||||
|
||||
const sessionId = this._currentSessionId || credentials?.connectionId || crypto.randomUUID();
|
||||
const reqId = this._currentReqId || crypto.randomUUID();
|
||||
headers["x-grok-session-id"] = sessionId;
|
||||
// CLI uses the same id for conv + session on chat turns
|
||||
headers["x-grok-conv-id"] = sessionId;
|
||||
headers["x-grok-req-id"] = reqId;
|
||||
headers["x-grok-turn-idx"] = String(this._currentTurnIdx || 1);
|
||||
|
||||
if (this._agentId) headers["x-grok-agent-id"] = this._agentId;
|
||||
|
||||
// Surface model override (CLI always sets this)
|
||||
if (this._currentModel) headers["x-grok-model-override"] = this._currentModel;
|
||||
|
||||
// Identity: mapTokens stores email top-level AND in providerSpecificData;
|
||||
// fall back either way so OAuth connections always fingerprint like the CLI.
|
||||
const psd = credentials?.providerSpecificData || {};
|
||||
const email = psd.email || credentials?.email;
|
||||
const userId = psd.userId || credentials?.userId || credentials?.providerUserId;
|
||||
if (email) headers["x-email"] = email;
|
||||
if (userId) headers["x-userid"] = userId;
|
||||
|
||||
return headers;
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
// 402 personal-team-blocked:spending-limit → surface as payment/quota for fallback
|
||||
if (response.status === 402 && bodyText) {
|
||||
try {
|
||||
const json = JSON.parse(bodyText);
|
||||
const code = json?.code || "";
|
||||
const msg = json?.error || json?.message || bodyText;
|
||||
return {
|
||||
status: 402,
|
||||
message: typeof msg === "string" ? msg : bodyText,
|
||||
code: typeof code === "string" ? code : undefined,
|
||||
};
|
||||
} catch {
|
||||
/* fall through */
|
||||
}
|
||||
}
|
||||
return super.parseError(response, bodyText);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
// Session / request ids for headers — stable per client conversation when possible
|
||||
const requestKey = body;
|
||||
this._currentSessionId = resolveGrokCliSessionId(credentials, body);
|
||||
this._currentReqId = crypto.randomUUID();
|
||||
this._agentId =
|
||||
credentials?.providerSpecificData?.deviceId ||
|
||||
credentials?.providerSpecificData?.agentId ||
|
||||
null;
|
||||
|
||||
// Normalize Responses input
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
|
||||
// Chat Completions clients arrive with messages[] — translator should have
|
||||
// converted already, but guard empty input.
|
||||
if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) {
|
||||
if (Array.isArray(body.messages) && body.messages.length > 0) {
|
||||
// Soft fallback: map messages → input messages (string content only)
|
||||
body.input = body.messages.map((m) => ({
|
||||
type: "message",
|
||||
role: m.role || "user",
|
||||
content: typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? ""),
|
||||
}));
|
||||
delete body.messages;
|
||||
} else {
|
||||
body.input = [{ type: "message", role: "user", content: "..." }];
|
||||
}
|
||||
}
|
||||
|
||||
// Keep role:"system" as-is — official grok-pager HAR sends system, not developer
|
||||
// (Codex converts system→developer; Grok CLI does not).
|
||||
normalizeGrokCliInput(body);
|
||||
stripStoredItemReferences(body);
|
||||
normalizeGrokCliTools(body);
|
||||
|
||||
// Turn index after input is finalized (user-message count, monotonic per session)
|
||||
this._currentTurnIdx = resolveGrokCliTurnIdx(this._currentSessionId, body.input, requestKey);
|
||||
|
||||
body.stream = true;
|
||||
body.store = false;
|
||||
|
||||
// Resolve upstream model id (strip effort suffix virtual models)
|
||||
let modelEffort = resolveEffortFromModel(body.model || model);
|
||||
let resolvedModel = body.model || model;
|
||||
if (modelEffort) {
|
||||
resolvedModel = resolvedModel.replace(new RegExp(`-${modelEffort}$`), "");
|
||||
}
|
||||
resolvedModel = getModelUpstreamId("gcli", resolvedModel) || resolvedModel;
|
||||
// Also try provider id key
|
||||
if (resolvedModel === (body.model || model)) {
|
||||
resolvedModel = getModelUpstreamId("grok-cli", resolvedModel) || resolvedModel;
|
||||
}
|
||||
body.model = resolvedModel;
|
||||
this._currentModel = resolvedModel;
|
||||
|
||||
// Reasoning effort priority: explicit > reasoning_effort > model suffix > default high.
|
||||
// grok-build and Composer reject reasoningEffort but still accept summary/encrypted continuity.
|
||||
const supportsReasoningEffort = supportsGrokCliReasoningEffort(resolvedModel);
|
||||
if (!body.reasoning || typeof body.reasoning !== "object") {
|
||||
body.reasoning = { summary: "concise" };
|
||||
if (supportsReasoningEffort) {
|
||||
body.reasoning.effort = normalizeGrokCliEffort(body.reasoning_effort || modelEffort);
|
||||
}
|
||||
} else {
|
||||
if (supportsReasoningEffort) {
|
||||
body.reasoning.effort = normalizeGrokCliEffort(
|
||||
body.reasoning.effort || body.reasoning_effort || modelEffort,
|
||||
);
|
||||
} else {
|
||||
delete body.reasoning.effort;
|
||||
}
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "concise";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
// Encrypted reasoning for multi-turn continuity (CLI always requests this)
|
||||
if (body.reasoning && body.reasoning.effort !== "none") {
|
||||
const include = Array.isArray(body.include) ? body.include : [];
|
||||
if (!include.includes("reasoning.encrypted_content")) {
|
||||
include.push("reasoning.encrypted_content");
|
||||
}
|
||||
body.include = include;
|
||||
}
|
||||
|
||||
// Drop Chat Completions leftovers that Responses rejects
|
||||
delete body.messages;
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
delete body.n;
|
||||
delete body.seed;
|
||||
delete body.logprobs;
|
||||
delete body.top_logprobs;
|
||||
delete body.frequency_penalty;
|
||||
delete body.presence_penalty;
|
||||
delete body.logit_bias;
|
||||
delete body.user;
|
||||
delete body.stream_options;
|
||||
delete body.prompt_cache_retention;
|
||||
delete body.safety_identifier;
|
||||
delete body.previous_response_id; // store=false → cannot resolve
|
||||
|
||||
for (const k of Object.keys(body)) {
|
||||
if (!RESPONSES_API_ALLOWLIST.has(k)) delete body[k];
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
// Lazy-resolve stable agent id once per process if connection has none
|
||||
if (!this._agentId && !args.credentials?.providerSpecificData?.deviceId) {
|
||||
try {
|
||||
const mid = await getConsistentMachineId("grok-cli-agent");
|
||||
// Format as UUID-ish for header aesthetics
|
||||
this._agentId = [
|
||||
mid.slice(0, 8),
|
||||
mid.slice(8, 12),
|
||||
"5" + mid.slice(13, 16),
|
||||
"a" + mid.slice(17, 20),
|
||||
mid.slice(0, 12).padEnd(12, "0"),
|
||||
].join("-");
|
||||
} catch {
|
||||
this._agentId = crypto.randomUUID();
|
||||
}
|
||||
} else if (args.credentials?.providerSpecificData?.deviceId) {
|
||||
this._agentId = args.credentials.providerSpecificData.deviceId;
|
||||
}
|
||||
|
||||
return super.execute(args);
|
||||
}
|
||||
}
|
||||
|
||||
export default GrokCliExecutor;
|
||||
@@ -13,6 +13,7 @@ import { QwenExecutor } from "./qwen.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
import { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
import { CommandCodeExecutor } from "./commandcode.js";
|
||||
@@ -39,6 +40,9 @@ const executors = {
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
gb: new GrokCliExecutor(), // Alias (Grok Build)
|
||||
"perplexity-web": new PerplexityWebExecutor(),
|
||||
"ollama-local": new OllamaLocalExecutor(),
|
||||
commandcode: new CommandCodeExecutor(),
|
||||
@@ -77,6 +81,7 @@ export { QwenExecutor } from "./qwen.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
export { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
export { CommandCodeExecutor } from "./commandcode.js";
|
||||
|
||||
@@ -1,18 +1,19 @@
|
||||
import { detectFormat, getTargetFormat, resolveTransport } from "../services/provider.js";
|
||||
import { translateRequest } from "../translator/index.js";
|
||||
import { stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { normalizeClaudePassthrough } from "../translator/formats/claude.js";
|
||||
import { COLORS } from "../utils/stream.js";
|
||||
import { createStreamController } from "../utils/streamHandler.js";
|
||||
import { refreshWithRetry } from "../services/tokenRefresh.js";
|
||||
import { createRequestLogger } from "../utils/requestLogger.js";
|
||||
import { getModelTargetFormat, getModelStrip, getModelUpstreamId, getModelType, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { createErrorResult, parseUpstreamError, formatProviderError } from "../utils/error.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS, TOKEN_SAVER_HEADER } from "../config/runtimeConfig.js";
|
||||
import { handleBypassRequest } from "../utils/bypassHandler.js";
|
||||
import { trackPendingRequest, appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { getExecutor } from "../executors/index.js";
|
||||
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
||||
import { buildRequestDetail, extractRequestConfig } from "./chatCore/requestDetail.js";
|
||||
import { handleForcedSSEToJson } from "./chatCore/sseToJsonHandler.js";
|
||||
import { handleNonStreamingResponse } from "./chatCore/nonStreamingHandler.js";
|
||||
@@ -23,9 +24,12 @@ import { injectCaveman } from "../rtk/caveman.js";
|
||||
import { injectPonytail } from "../rtk/ponytail.js";
|
||||
import { compressMessages, formatRtkLog } from "../rtk/index.js";
|
||||
import { compressWithHeadroom, formatHeadroomLog, formatHeadroomSizeLog, isHeadroomPhantomSavings } from "../rtk/headroom.js";
|
||||
import { compressWithPxpipe } from "../rtk/pxpipe.js";
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
|
||||
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
import { extractThinking } from "../translator/concerns/thinkingUnified.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
/**
|
||||
* Core chat handler - shared between SSE and Worker
|
||||
@@ -34,9 +38,18 @@ import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
* @param {object} options.credentials - Provider credentials
|
||||
* @param {string} options.sourceFormatOverride - Override detected source format (e.g. "openai-responses")
|
||||
*/
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, sourceFormatOverride, providerThinking }) {
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, headroomEnabled, headroomUrl, headroomCompressUserMessages, cavemanEnabled, cavemanLevel, ponytailEnabled, ponytailLevel, pxpipeEnabled, pxpipeMinChars, pxpipeTimeoutMs, pxpipeTransform, onPxpipeEvent, sourceFormatOverride, providerThinking }) {
|
||||
const { provider, model } = modelInfo;
|
||||
const requestStartTime = Date.now();
|
||||
// Stable per-session color so all lines of one CLI conversation share a tag
|
||||
const sessionSeed = (() => {
|
||||
try {
|
||||
return resolveSessionId({ headers: clientRawRequest?.headers, body, connectionId, scope: provider });
|
||||
} catch {
|
||||
return connectionId || "";
|
||||
}
|
||||
})();
|
||||
const reqTag = log?.tagForSession ? log.tagForSession(sessionSeed) : (log?.nextTag ? log.nextTag() : "");
|
||||
|
||||
const sourceFormat = sourceFormatOverride || detectFormat(body);
|
||||
|
||||
@@ -123,9 +136,9 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
let toolNameMap;
|
||||
if (passthrough) {
|
||||
log?.debug?.("PASSTHROUGH", `${clientTool} → ${provider} | native lossless`);
|
||||
translatedBody = { ...body, model: upstreamModel };
|
||||
translatedBody = { ...body, model: stripThinkingSuffix(upstreamModel) };
|
||||
// Normalize newer Cowork/CC beta shapes (adaptive thinking, mid-conversation system) the API rejects
|
||||
if (clientTool === "claude") normalizeClaudePassthrough(translatedBody, upstreamModel);
|
||||
if (clientTool === "claude") normalizeClaudePassthrough(translatedBody, translatedBody.model);
|
||||
} else {
|
||||
translatedBody = translateRequest(sourceFormat, targetFormat, upstreamModel, body, stream, credentials, provider, reqLogger, stripList, connectionId, clientTool);
|
||||
if (!translatedBody) {
|
||||
@@ -134,7 +147,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
}
|
||||
toolNameMap = translatedBody._toolNameMap;
|
||||
delete translatedBody._toolNameMap;
|
||||
translatedBody.model = upstreamModel;
|
||||
translatedBody.model = stripThinkingSuffix(upstreamModel);
|
||||
}
|
||||
|
||||
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
|
||||
@@ -150,41 +163,83 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
// Covers both passthrough (source shape) and translated (target shape) flows
|
||||
const finalFormat = passthrough ? sourceFormat : targetFormat;
|
||||
|
||||
// Request line: one correlated summary (fmt + thinking + counts + account)
|
||||
if (log?.line) {
|
||||
const clientModel = clientRawRequest?.body?.model || `${provider}/${model}`;
|
||||
const msgN = translatedBody.messages?.length || translatedBody.input?.length || translatedBody.contents?.length || body.messages?.length || body.input?.length || 0;
|
||||
const toolN = translatedBody.tools?.length || body.tools?.length || 0;
|
||||
const fmtStr = passthrough ? `FMT: ${sourceFormat} (passthrough)` : `FMT: ${sourceFormat}→${targetFormat}`;
|
||||
const showThinking = provider !== "grok-cli" || supportsGrokCliReasoningEffort(model);
|
||||
const think = showThinking ? log.fmtThink?.(extractThinking(translatedBody)) : null;
|
||||
const acc = credentials?.connectionName || credentials?.connectionId?.slice(0, 8) || "-";
|
||||
const parts = [
|
||||
`POST ${clientModel} → ${provider}/${model}`,
|
||||
fmtStr,
|
||||
stream ? "STREAM" : "JSON",
|
||||
`${msgN} MSG`,
|
||||
];
|
||||
if (toolN) parts.push(`${toolN} TOOL`);
|
||||
if (think) parts.push(`THINK:${think}`);
|
||||
parts.push(`ACC:${acc}`);
|
||||
log.line(reqTag, "▶", parts.join(" · "));
|
||||
}
|
||||
|
||||
// TTS models don't support tool messages/function calling
|
||||
if (getModelType(alias, model) === "tts" && translatedBody.messages) {
|
||||
translatedBody.messages = translatedBody.messages.filter(msg => msg.role !== "tool");
|
||||
delete translatedBody.tools;
|
||||
}
|
||||
|
||||
// Per-request opt-out: client can bypass all token savers via header
|
||||
const tokenSaverEnabled = clientRawRequest?.headers?.[TOKEN_SAVER_HEADER]?.toLowerCase() !== "off";
|
||||
|
||||
// RTK: compress tool_result content
|
||||
const rtkStats = compressMessages(translatedBody, rtkEnabled);
|
||||
const rtkStats = compressMessages(translatedBody, tokenSaverEnabled && rtkEnabled);
|
||||
const rtkLine = formatRtkLog(rtkStats);
|
||||
if (rtkLine) console.log(rtkLine);
|
||||
|
||||
// Headroom: optional external proxy compression; fail open if proxy is absent.
|
||||
const headroomDiagnostics = {};
|
||||
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, diagnostics: headroomDiagnostics });
|
||||
const headroomStats = await compressWithHeadroom(translatedBody, { enabled: tokenSaverEnabled && headroomEnabled, url: headroomUrl, model: upstreamModel, format: finalFormat, compressUserMessages: headroomCompressUserMessages, diagnostics: headroomDiagnostics });
|
||||
const headroomLine = formatHeadroomLog(headroomStats);
|
||||
const headroomSizeLine = formatHeadroomSizeLog(headroomDiagnostics);
|
||||
if (headroomLine) {
|
||||
log?.info?.("HEADROOM", `${headroomLine}${headroomSizeLine ? ` | ${headroomSizeLine}` : ""}`);
|
||||
if (isHeadroomPhantomSavings(headroomStats, headroomDiagnostics)) {
|
||||
log?.warn?.("HEADROOM", `reported token delta, but outbound JSON shrank <5%; provider may bill near-original payload | ${headroomSizeLine}`);
|
||||
log?.warn?.("HEADROOM", `reported token delta, but outbound JSON shrank <5%; provider may bill near-original payload | ${formatHeadroomSizeLog(headroomDiagnostics)}`);
|
||||
}
|
||||
} else if (headroomEnabled) log?.warn?.("HEADROOM", `skipped: ${headroomDiagnostics.reason || "compression unavailable"}${headroomDiagnostics.endpoint ? ` (${headroomDiagnostics.endpoint})` : ""}`);
|
||||
} else if (tokenSaverEnabled && headroomEnabled) log?.warn?.("HEADROOM", `skipped: ${headroomDiagnostics.reason || "compression unavailable"}${headroomDiagnostics.endpoint ? ` (${headroomDiagnostics.endpoint})` : ""}`);
|
||||
|
||||
// Token-saver flags accumulator for the single "⚙" log line below.
|
||||
const xf = [];
|
||||
|
||||
// Caveman: inject terse-style system prompt
|
||||
if (cavemanEnabled && cavemanLevel) {
|
||||
if (tokenSaverEnabled && cavemanEnabled && cavemanLevel) {
|
||||
injectCaveman(translatedBody, finalFormat, cavemanLevel);
|
||||
log?.debug?.("CAVEMAN", `${cavemanLevel} | ${finalFormat}`);
|
||||
xf.push(`CAVEMAN:${cavemanLevel}`);
|
||||
}
|
||||
|
||||
// Ponytail: inject lazy-senior-dev system prompt
|
||||
if (ponytailEnabled && ponytailLevel) {
|
||||
if (tokenSaverEnabled && ponytailEnabled && ponytailLevel) {
|
||||
injectPonytail(translatedBody, finalFormat, ponytailLevel);
|
||||
log?.debug?.("PONYTAIL", `${ponytailLevel} | ${finalFormat}`);
|
||||
xf.push(`PONYTAIL:${ponytailLevel}`);
|
||||
}
|
||||
|
||||
// PXPIPE: image bulky context (Claude-format bodies only), last saver before dispatch
|
||||
let pxpipeSummary = null;
|
||||
if (pxpipeEnabled) {
|
||||
const pxpipeResult = await compressWithPxpipe(translatedBody, {
|
||||
enabled: true, format: finalFormat, model: upstreamModel,
|
||||
minChars: pxpipeMinChars, timeoutMs: pxpipeTimeoutMs, transform: pxpipeTransform,
|
||||
});
|
||||
pxpipeSummary = pxpipeResult.summary;
|
||||
if (pxpipeResult.body) translatedBody = pxpipeResult.body;
|
||||
if (pxpipeSummary?.applied) xf.push(`PXPIPE:${pxpipeSummary.imageCount}img`);
|
||||
try { onPxpipeEvent?.({ provider, model, ...pxpipeSummary }); } catch { /* stats must not break requests */ }
|
||||
}
|
||||
|
||||
if (xf.length && log?.line) log.line(reqTag, "⚙", xf.join(" · "));
|
||||
|
||||
const executor = getExecutor(provider);
|
||||
trackPendingRequest(model, provider, connectionId, true);
|
||||
appendRequestLog({ model, provider, connectionId, status: "PENDING" }).catch(() => { });
|
||||
@@ -198,7 +253,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
if (onDisconnect) onDisconnect(reason);
|
||||
},
|
||||
onError: () => trackPendingRequest(model, provider, connectionId, false),
|
||||
log, provider, model
|
||||
log, provider, model, reqTag
|
||||
});
|
||||
|
||||
const proxyOptions = {
|
||||
@@ -253,6 +308,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: translatedBody || null,
|
||||
response: { error: error.message || String(error), status: error.name === "AbortError" ? 499 : 502, thinking: null },
|
||||
pxpipe: pxpipeSummary,
|
||||
status: "error"
|
||||
})).catch(() => { });
|
||||
|
||||
@@ -261,7 +317,9 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
return createErrorResult(499, "Request aborted");
|
||||
}
|
||||
const errMsg = formatProviderError(error, provider, model, HTTP_STATUS.BAD_GATEWAY);
|
||||
console.log(`${COLORS.red}[ERROR] ${errMsg}${COLORS.reset}`);
|
||||
if (log?.errorLine) {
|
||||
log.errorLine(reqTag, "✗", `ERROR 502 · ${provider}/${model} · ${Date.now() - requestStartTime}ms\n ${errMsg}${error.stack ? `\n ${error.stack}` : ""}`);
|
||||
}
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, errMsg);
|
||||
}
|
||||
|
||||
@@ -270,7 +328,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
try {
|
||||
const newCredentials = await refreshWithRetry(() => executor.refreshCredentials(credentials, log), 3, log);
|
||||
if (newCredentials?.accessToken || newCredentials?.copilotToken) {
|
||||
log?.info?.("TOKEN", `${provider.toUpperCase()} | refreshed`);
|
||||
if (log?.line) log.line(reqTag, "🔑", `TOKEN REFRESHED · ${provider}/${model}`);
|
||||
Object.assign(credentials, newCredentials);
|
||||
if (onCredentialsRefreshed) {
|
||||
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
|
||||
@@ -299,16 +357,20 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
request: extractRequestConfig(body, stream),
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
response: { error: message, status: statusCode, thinking: null },
|
||||
pxpipe: pxpipeSummary,
|
||||
status: "error"
|
||||
})).catch(() => { });
|
||||
|
||||
const errMsg = formatProviderError(new Error(message), provider, model, statusCode);
|
||||
console.log(`${COLORS.red}[ERROR] ${errMsg}${COLORS.reset}`);
|
||||
if (log?.errorLine) {
|
||||
const urlStr = providerUrl ? `\n URL: ${providerUrl}` : "";
|
||||
log.errorLine(reqTag, "✗", `ERROR ${statusCode} · ${provider}/${model} · ${Date.now() - requestStartTime}ms${urlStr}\n ${errMsg}`);
|
||||
}
|
||||
reqLogger.logError(new Error(message), finalBody || translatedBody);
|
||||
return createErrorResult(statusCode, errMsg, resetsAtMs);
|
||||
}
|
||||
|
||||
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess };
|
||||
const sharedCtx = { provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, pxpipe: pxpipeSummary, reqTag, log };
|
||||
const appendLog = (extra) => appendRequestLog({ model, provider, connectionId, ...extra }).catch(() => { });
|
||||
const trackDone = () => trackPendingRequest(model, provider, connectionId, false);
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats } from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
|
||||
@@ -198,7 +198,7 @@ export function translateNonStreamingResponse(responseBody, targetFormat, source
|
||||
/**
|
||||
* Handle non-streaming response from provider.
|
||||
*/
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, trackDone, appendLog }) {
|
||||
export async function handleNonStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, trackDone, appendLog, pxpipe, reqTag, log }) {
|
||||
trackDone();
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
let responseBody;
|
||||
@@ -235,7 +235,8 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
|
||||
const usage = extractUsageFromResponse(responseBody);
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const translatedResponse = needsTranslation(targetFormat, sourceFormat)
|
||||
? translateNonStreamingResponse(responseBody, targetFormat, sourceFormat)
|
||||
@@ -296,6 +297,7 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
thinking: translatedResponse?.choices?.[0]?.message?.reasoning_content || translatedResponse?.reasoning_content || null,
|
||||
finish_reason: translatedResponse?.choices?.[0]?.finish_reason || "unknown"
|
||||
},
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { endpoint: clientRawRequest?.endpoint || null })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save:", err.message);
|
||||
|
||||
@@ -69,12 +69,31 @@ export function buildRequestDetail(base, overrides = {}) {
|
||||
providerRequest: base.providerRequest || null,
|
||||
providerResponse: base.providerResponse || null,
|
||||
response: base.response || {},
|
||||
pxpipe: base.pxpipe || undefined,
|
||||
status: base.status || "success",
|
||||
...overrides
|
||||
};
|
||||
}
|
||||
|
||||
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE" }) {
|
||||
// Build the "done" summary: duration, ttft, in/out tokens with cache breakdown
|
||||
export function formatDoneLine({ usage, latency }) {
|
||||
const u = usage || {};
|
||||
const inTok = u.prompt_tokens ?? u.input_tokens ?? 0;
|
||||
const outTok = u.completion_tokens ?? u.output_tokens ?? 0;
|
||||
const cacheRead = u.cache_read_input_tokens ?? u.cached_tokens ?? u.prompt_tokens_details?.cached_tokens ?? 0;
|
||||
const cacheCreate = u.cache_creation_input_tokens ?? 0;
|
||||
let inStr = `IN ${inTok}`;
|
||||
if (cacheRead || cacheCreate) {
|
||||
const parts = [];
|
||||
if (cacheRead) parts.push(`↻${cacheRead}`);
|
||||
if (cacheCreate) parts.push(`+${cacheCreate}`);
|
||||
inStr += ` (CACHE ${parts.join(" ")})`;
|
||||
}
|
||||
const ttftStr = latency?.ttft ? ` · TTFT ${latency.ttft}ms` : "";
|
||||
return `DONE ${latency?.total ?? 0}ms${ttftStr} · ${inStr} · OUT ${outTok}`;
|
||||
}
|
||||
|
||||
export function saveUsageStats({ provider, model, tokens, connectionId, apiKey, endpoint, label = "USAGE", silent = false }) {
|
||||
if (!tokens || typeof tokens !== "object") return;
|
||||
|
||||
const inTokens = tokens.input_tokens ?? tokens.prompt_tokens ?? 0;
|
||||
@@ -82,9 +101,11 @@ export function saveUsageStats({ provider, model, tokens, connectionId, apiKey,
|
||||
|
||||
if (inTokens === 0 && outTokens === 0) return;
|
||||
|
||||
const time = new Date().toLocaleTimeString("en-US", { hour12: false, hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
||||
const accountSuffix = connectionId ? ` | account=${connectionId.slice(0, 8)}...` : "";
|
||||
console.log(`${COLORS.green}[${time}] 📊 [${label}] ${provider.toUpperCase()} | in=${inTokens} | out=${outTokens}${accountSuffix}${COLORS.reset}`);
|
||||
if (!silent) {
|
||||
const time = new Date().toLocaleTimeString("en-US", { hour12: false, hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
||||
const accountSuffix = connectionId ? ` | account=${connectionId.slice(0, 8)}...` : "";
|
||||
console.log(`${COLORS.green}[${time}] 📊 [${label}] ${provider.toUpperCase()} | in=${inTokens} | out=${outTokens}${accountSuffix}${COLORS.reset}`);
|
||||
}
|
||||
|
||||
// Canonicalize to one storage convention (prompt_tokens cache-inclusive) so
|
||||
// cached/cache-creation tokens survive to cost calc + stats. See canonicalizeUsage.
|
||||
|
||||
@@ -3,7 +3,7 @@ import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { FORMATS } from "../../translator/formats.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats } from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
|
||||
// Responses-API providers (e.g. codex) may emit SSE without content-type + use Responses output shape
|
||||
const isResponsesProvider = (p) => PROVIDERS[p]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
@@ -102,7 +102,7 @@ export function parseSSEToOpenAIResponse(rawSSE, fallbackModel) {
|
||||
* Handle case: provider forced streaming but client wants JSON.
|
||||
* Supports both Codex/Responses API SSE and standard Chat Completions SSE.
|
||||
*/
|
||||
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, trackDone, appendLog }) {
|
||||
export async function handleForcedSSEToJson({ providerResponse, sourceFormat, provider, model, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, trackDone, appendLog, reqTag, log }) {
|
||||
const contentType = providerResponse.headers.get("content-type") || "";
|
||||
const isSSE = contentType.includes("text/event-stream") || (contentType === "" && isResponsesProvider(provider));
|
||||
if (!isSSE) return null; // not handled here
|
||||
@@ -124,7 +124,8 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
|
||||
const usage = jsonResponse.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const { msgItem, textContent } = pickAssistantMessageForChatCompletion(jsonResponse.output);
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
@@ -200,7 +201,8 @@ export async function handleForcedSSEToJson({ providerResponse, sourceFormat, pr
|
||||
|
||||
const usage = parsed.usage || {};
|
||||
appendLog({ tokens: usage, status: "200 OK" });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint });
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency: { total: Date.now() - requestStartTime } }));
|
||||
|
||||
const totalLatency = Date.now() - requestStartTime;
|
||||
saveRequestDetail(buildRequestDetail({
|
||||
|
||||
@@ -5,7 +5,7 @@ import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats } from "./requestDetail.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
|
||||
@@ -43,7 +43,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId }) {
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
@@ -67,7 +67,8 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
const shortMsg = sanitizedTitle
|
||||
|| (bodyText.length < 200 ? bodyText.replace(/<[^>]*>/g, '').trim().slice(0, 160) : `Upstream returned non-SSE response (${upstreamContentType})`);
|
||||
const status = providerResponse.status || 502;
|
||||
console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
if (log?.errorLine) log.errorLine(reqTag, "✗", `BLOCKED ${status} · ${provider}/${model} · non-SSE (${upstreamContentType})\n ${shortMsg}`);
|
||||
else console.warn(`[STREAM] ${provider} | ${model} | blocked pipe: ${shortMsg} [${status}]`);
|
||||
streamController?.handleError?.(new Error(`upstream non-SSE: ${status}`));
|
||||
return {
|
||||
success: false,
|
||||
@@ -94,6 +95,7 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: "[Streaming - raw response not captured]",
|
||||
response: { content: "[Streaming in progress...]", thinking: null, type: "streaming" },
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to save streaming request:", err.message);
|
||||
@@ -108,7 +110,7 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
/**
|
||||
* Build onStreamComplete callback for streaming usage tracking.
|
||||
*/
|
||||
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest }) {
|
||||
export function buildOnStreamComplete({ provider, model, connectionId, apiKey, requestStartTime, body, stream, finalBody, translatedBody, clientRawRequest, pxpipe, reqTag, log }) {
|
||||
const streamDetailId = `${Date.now()}-${Math.random().toString(36).slice(2, 11)}`;
|
||||
|
||||
const onStreamComplete = (contentObj, usage, ttftAt) => {
|
||||
@@ -127,12 +129,15 @@ export function buildOnStreamComplete({ provider, model, connectionId, apiKey, r
|
||||
providerRequest: finalBody || translatedBody || null,
|
||||
providerResponse: safeContent,
|
||||
response: { content: safeContent, thinking: safeThinking, type: "streaming" },
|
||||
pxpipe,
|
||||
status: "success"
|
||||
}, { id: streamDetailId })).catch(err => {
|
||||
console.error("[RequestDetail] Failed to update streaming content:", err.message);
|
||||
});
|
||||
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, label: "STREAM USAGE" });
|
||||
// Persist stream usage to DB (no console line; the "📊 done" line below is authoritative)
|
||||
saveUsageStats({ provider, model, tokens: usage, connectionId, apiKey, endpoint: clientRawRequest?.endpoint, label: "STREAM USAGE", silent: true });
|
||||
if (log?.line) log.line(reqTag, "📊", formatDoneLine({ usage, latency }));
|
||||
};
|
||||
|
||||
return { onStreamComplete, streamDetailId };
|
||||
|
||||
@@ -273,6 +273,53 @@ const CHAT_SEARCH_CONFIG = {
|
||||
const tokens = data?.usage?.total_tokens || 0;
|
||||
return { text, citations, tokens };
|
||||
}
|
||||
},
|
||||
|
||||
"perplexity-agent": {
|
||||
endpoint: () => searchEndpoint("perplexity-agent"),
|
||||
buildBody: (query, model) => ({
|
||||
model,
|
||||
input: query,
|
||||
tools: [{ type: "web_search" }]
|
||||
}),
|
||||
buildHeaders: (token) => ({
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${token}`
|
||||
}),
|
||||
extractAnswer: (data) => {
|
||||
const output = Array.isArray(data?.output) ? data.output : [];
|
||||
let text = "";
|
||||
const citations = [];
|
||||
for (const item of output) {
|
||||
const parts = Array.isArray(item?.content) ? item.content : [];
|
||||
for (const p of parts) {
|
||||
if (typeof p?.text === "string") text += p.text;
|
||||
const anns = Array.isArray(p?.annotations) ? p.annotations : [];
|
||||
for (const a of anns) {
|
||||
const c = normalizeCitation(a?.url ? a : a?.url_citation);
|
||||
if (c) citations.push(c);
|
||||
}
|
||||
}
|
||||
const results = Array.isArray(item?.results) ? item.results : [];
|
||||
for (const r of results) {
|
||||
const url = r?.url || r?.link;
|
||||
if (!url) continue;
|
||||
citations.push({
|
||||
url,
|
||||
title: r?.title || "",
|
||||
snippet: r?.snippet || ""
|
||||
});
|
||||
}
|
||||
}
|
||||
if (!citations.length && Array.isArray(data?.citations)) {
|
||||
for (const c of data.citations) {
|
||||
const n = normalizeCitation(c);
|
||||
if (n) citations.push(n);
|
||||
}
|
||||
}
|
||||
const tokens = data?.usage?.total_tokens || 0;
|
||||
return { text, citations, tokens };
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
166
open-sse/handlers/videoCore.js
Normal file
166
open-sse/handlers/videoCore.js
Normal file
@@ -0,0 +1,166 @@
|
||||
import { createErrorResult } from "../utils/error.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { refreshTokenByProvider } from "../services/tokenRefresh.js";
|
||||
import { PROVIDER_MEDIA } from "../providers/index.js";
|
||||
|
||||
// Upstream fetch deadline for video job submission/polling (the job itself is
|
||||
// async upstream — this only bounds the HTTP round-trip, not video rendering).
|
||||
const VIDEO_FETCH_TIMEOUT_MS = Number(process.env.VIDEO_FETCH_TIMEOUT_MS || 120000);
|
||||
|
||||
// POST /videos/* creates a billable upstream job. A network error after the
|
||||
// request left the socket may still have created the job, so creation is NEVER
|
||||
// auto-retried (the only re-send is the auth retry after a 401/403 refresh,
|
||||
// which upstream rejects before job creation).
|
||||
export const VIDEO_ACTIONS = new Set(["generations", "edits", "extensions"]);
|
||||
|
||||
export function getVideoConfig(provider) {
|
||||
return PROVIDER_MEDIA[provider]?.videoConfig || null;
|
||||
}
|
||||
|
||||
/** Strip bearer tokens / obvious secrets from text destined for clients or logs. */
|
||||
export function sanitizeSecrets(text, credentials = null) {
|
||||
if (!text) return text;
|
||||
let out = String(text).replace(/Bearer\s+[A-Za-z0-9._~+/=-]{8,}/gi, "Bearer [redacted]");
|
||||
for (const key of ["accessToken", "refreshToken", "apiKey"]) {
|
||||
const secret = credentials?.[key];
|
||||
if (typeof secret === "string" && secret.length >= 8) {
|
||||
out = out.split(secret).join("[redacted]");
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function buildUpstreamUrl(config, action, requestId) {
|
||||
const base = config.baseUrl.replace(/\/$/, "");
|
||||
return requestId ? `${base}/${encodeURIComponent(requestId)}` : `${base}/${action}`;
|
||||
}
|
||||
|
||||
function buildHeaders({ token, contentType, idempotencyKey }) {
|
||||
const headers = { Accept: "application/json" };
|
||||
if (token) headers.Authorization = `Bearer ${token}`;
|
||||
if (contentType) headers["Content-Type"] = contentType;
|
||||
if (idempotencyKey) headers["Idempotency-Key"] = idempotencyKey;
|
||||
return headers;
|
||||
}
|
||||
|
||||
function combineSignals(signal, timeoutMs) {
|
||||
const timeoutSignal = typeof AbortSignal?.timeout === "function" ? AbortSignal.timeout(timeoutMs) : null;
|
||||
if (signal && timeoutSignal && typeof AbortSignal.any === "function") {
|
||||
return AbortSignal.any([signal, timeoutSignal]);
|
||||
}
|
||||
return signal || timeoutSignal || undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Transparent proxy for async video jobs (xAI Grok Imagine shape).
|
||||
*
|
||||
* - Forwards the raw body byte-for-byte (JSON or multipart) — no reshaping.
|
||||
* - Passes upstream JSON (request_id, status, video.url, error) back verbatim.
|
||||
* - 401/403 with a refresh token: refresh ONCE, retry ONCE. No other retry.
|
||||
* - Upstream error text is sanitized before it reaches the client.
|
||||
*
|
||||
* @param {object} options
|
||||
* @param {string} options.provider - Provider id (must have registry videoConfig)
|
||||
* @param {"generations"|"edits"|"extensions"|null} options.action - Creation action (POST)
|
||||
* @param {string|null} [options.requestId] - Poll target (GET /videos/{id})
|
||||
* @param {Buffer|string|null} [options.rawBody] - Exact body to forward
|
||||
* @param {string|null} [options.contentType] - Original Content-Type header
|
||||
* @param {string|null} [options.idempotencyKey] - Forwarded Idempotency-Key
|
||||
* @param {object} options.credentials - { accessToken?, apiKey?, refreshToken?, authType? }
|
||||
* @param {AbortSignal} [options.signal] - Client cancellation signal
|
||||
* @param {number} [options.timeoutMs]
|
||||
* @param {object} [options.log]
|
||||
* @param {function} [options.onCredentialsRefreshed]
|
||||
* @returns {Promise<{ success: boolean, response: Response, status?: number, error?: string }>}
|
||||
*/
|
||||
export async function handleVideoProxyCore({
|
||||
provider,
|
||||
action = null,
|
||||
requestId = null,
|
||||
rawBody = null,
|
||||
contentType = null,
|
||||
idempotencyKey = null,
|
||||
credentials,
|
||||
signal,
|
||||
timeoutMs = VIDEO_FETCH_TIMEOUT_MS,
|
||||
log,
|
||||
onCredentialsRefreshed,
|
||||
}) {
|
||||
const config = getVideoConfig(provider);
|
||||
if (!config) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support video generation`);
|
||||
}
|
||||
if (!requestId && !VIDEO_ACTIONS.has(action)) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`);
|
||||
}
|
||||
|
||||
const method = requestId ? "GET" : "POST";
|
||||
const url = buildUpstreamUrl(config, action, requestId);
|
||||
const fetchSignal = combineSignals(signal, timeoutMs);
|
||||
|
||||
const doFetch = (token) =>
|
||||
fetch(url, {
|
||||
method,
|
||||
headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }),
|
||||
body: method === "POST" ? rawBody : undefined,
|
||||
signal: fetchSignal,
|
||||
});
|
||||
|
||||
let upstream;
|
||||
try {
|
||||
upstream = await doFetch(credentials?.accessToken || credentials?.apiKey);
|
||||
} catch (error) {
|
||||
if (error?.name === "AbortError" || error?.name === "TimeoutError") {
|
||||
return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`);
|
||||
}
|
||||
// Never re-send a creation POST on network error — the job may already exist upstream.
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video upstream fetch failed: ${error.message}`, credentials));
|
||||
}
|
||||
|
||||
// 401/403 → refresh once → retry once (OAuth accounts only; API keys can't refresh)
|
||||
if (
|
||||
(upstream.status === HTTP_STATUS.UNAUTHORIZED || upstream.status === HTTP_STATUS.FORBIDDEN) &&
|
||||
credentials?.refreshToken
|
||||
) {
|
||||
let refreshed = null;
|
||||
try {
|
||||
refreshed = await refreshTokenByProvider(provider, credentials, log);
|
||||
} catch (error) {
|
||||
log?.warn?.("TOKEN", `${provider} | video refresh error: ${sanitizeSecrets(error.message, credentials)}`);
|
||||
}
|
||||
if (refreshed?.accessToken) {
|
||||
log?.info?.("TOKEN", `${provider.toUpperCase()} | refreshed for video ${method}`);
|
||||
Object.assign(credentials, refreshed);
|
||||
if (onCredentialsRefreshed) await onCredentialsRefreshed(refreshed);
|
||||
try {
|
||||
await upstream.body?.cancel?.();
|
||||
} catch { /* noop */ }
|
||||
try {
|
||||
upstream = await doFetch(credentials.accessToken || credentials.apiKey);
|
||||
} catch (error) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials));
|
||||
}
|
||||
} else {
|
||||
log?.warn?.("TOKEN", `${provider.toUpperCase()} | video refresh failed — account needs re-auth`);
|
||||
}
|
||||
}
|
||||
|
||||
const bodyText = await upstream.text().catch(() => "");
|
||||
|
||||
if (!upstream.ok) {
|
||||
const message = sanitizeSecrets(bodyText || `HTTP ${upstream.status}`, credentials);
|
||||
return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`);
|
||||
}
|
||||
|
||||
// Success: pass the upstream JSON through untouched (request_id / status / video.url).
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(bodyText, {
|
||||
status: upstream.status,
|
||||
headers: {
|
||||
"Content-Type": upstream.headers.get("content-type") || "application/json",
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
@@ -98,6 +98,8 @@ export const MODEL_CAPABILITIES = {
|
||||
"coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
};
|
||||
|
||||
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
|
||||
/**
|
||||
* Provider-specific capability overrides. Keyed by provider alias/id.
|
||||
*/
|
||||
@@ -111,6 +113,20 @@ export const PROVIDER_CAPABILITIES = {
|
||||
"deepseek-ai/deepseek-v4-pro": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
},
|
||||
"kiro": {
|
||||
"gpt-5.6-sol": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-terra": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-luna": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-sol-thinking": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-terra-thinking": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-luna-thinking": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-sol-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-terra-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-luna-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-sol-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-terra-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
"gpt-5.6-luna-thinking-agentic": KIRO_GPT_5_6_CAPABILITIES,
|
||||
},
|
||||
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
|
||||
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
|
||||
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
|
||||
@@ -186,6 +202,8 @@ export const PATTERN_CAPABILITIES = [
|
||||
// ── Grok (vision + Live Search) ──────────────────────────────────
|
||||
{ pattern: "*grok*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*grok-code*", caps: { reasoning: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
// Grok 4.5 (Grok CLI / Grok Build): 500k context per cli-chat-proxy /v1/models
|
||||
{ pattern: "*grok-4.5*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 500000, maxOutput: 64000 } },
|
||||
{ pattern: "*grok-4*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
{ pattern: "*grok-3*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 131072 } },
|
||||
{ pattern: "*grok*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 256000 } },
|
||||
@@ -269,13 +287,17 @@ export const PATTERN_CAPABILITIES = [
|
||||
export function getCapabilitiesForModel(provider, model) {
|
||||
if (!model) return { ...DEFAULT_CAPABILITIES };
|
||||
|
||||
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
|
||||
// 1. Provider-specific override
|
||||
if (provider && PROVIDER_CAPABILITIES[provider]?.[model]) {
|
||||
return { ...DEFAULT_CAPABILITIES, ...PROVIDER_CAPABILITIES[provider][model] };
|
||||
if (provider) {
|
||||
const providerCaps = PROVIDER_CAPABILITIES[provider];
|
||||
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
}
|
||||
|
||||
// 2. Canonical exact (strip vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7")
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
// 2. Canonical exact
|
||||
if (MODEL_CAPABILITIES[baseModel]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[baseModel] };
|
||||
if (MODEL_CAPABILITIES[model]) return { ...DEFAULT_CAPABILITIES, ...MODEL_CAPABILITIES[model] };
|
||||
|
||||
|
||||
@@ -1,5 +1,14 @@
|
||||
import { deriveModelName } from "./namePatterns.js";
|
||||
|
||||
// Normalize version separators in a model id: hyphen between two digits becomes a dot.
|
||||
// Registry ids use dots for versions ("claude-sonnet-4.5") but clients (CLIs, aliases)
|
||||
// often send them with dashes ("claude-sonnet-4-5"). Only digit-digit hyphens are
|
||||
// touched, so word/suffix hyphens stay intact ("-thinking", "-agentic", "qwen3-coder-next").
|
||||
export function normalizeModelId(modelId) {
|
||||
if (typeof modelId !== "string") return modelId;
|
||||
return modelId.replace(/(\d)-(\d)/g, "$1.$2");
|
||||
}
|
||||
|
||||
// Model defaults centralized (was scattered as `m.kind || "llm"`, `quotaFamily || "normal"`, etc.)
|
||||
export const MODEL_DEFAULTS = {
|
||||
kind: "llm",
|
||||
|
||||
@@ -28,6 +28,7 @@ export const MODEL_PRICING = {
|
||||
"claude-sonnet-4.6": { input: 3.00, output: 15.00, cached: 0.30, reasoning: 22.50, cache_creation: 3.00 },
|
||||
"claude-opus-4-5-thinking": { input: 5.00, output: 25.00, cached: 0.50, reasoning: 37.50, cache_creation: 5.00 },
|
||||
"claude-opus-4-6-thinking": { input: 5.00, output: 25.00, cached: 0.50, reasoning: 37.50, cache_creation: 5.00 },
|
||||
"claude-fable-5": { input: 10.00, output: 50.00, cached: 1.00, reasoning: 50.00, cache_creation: 12.50 },
|
||||
|
||||
// === OpenAI / GPT ===
|
||||
"gpt-3.5-turbo": { input: 0.50, output: 1.50, cached: 0.25, reasoning: 2.25, cache_creation: 0.50 },
|
||||
@@ -36,22 +37,22 @@ export const MODEL_PRICING = {
|
||||
"gpt-4o": { input: 2.50, output: 10.00, cached: 1.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-4o-mini": { input: 0.15, output: 0.60, cached: 0.075, reasoning: 0.90, cache_creation: 0.15 },
|
||||
"gpt-4.1": { input: 2.50, output: 10.00, cached: 1.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-5": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
"gpt-5-mini": { input: 0.75, output: 3.00, cached: 0.375, reasoning: 4.50, cache_creation: 0.75 },
|
||||
"gpt-5-codex": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
"gpt-5.1": { input: 4.00, output: 16.00, cached: 2.00, reasoning: 24.00, cache_creation: 4.00 },
|
||||
"gpt-5.1-codex": { input: 4.00, output: 16.00, cached: 2.00, reasoning: 24.00, cache_creation: 4.00 },
|
||||
"gpt-5": { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 },
|
||||
"gpt-5-mini": { input: 0.25, output: 2.00, cached: 0.125, reasoning: 2.00, cache_creation: 0.25 },
|
||||
"gpt-5-codex": { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 },
|
||||
"gpt-5.1": { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 },
|
||||
"gpt-5.1-codex": { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 },
|
||||
"gpt-5.1-codex-mini": { input: 1.50, output: 6.00, cached: 0.75, reasoning: 9.00, cache_creation: 1.50 },
|
||||
"gpt-5.1-codex-mini-high": { input: 2.00, output: 8.00, cached: 1.00, reasoning: 12.00, cache_creation: 2.00 },
|
||||
"gpt-5.1-codex-max": { input: 8.00, output: 32.00, cached: 4.00, reasoning: 48.00, cache_creation: 8.00 },
|
||||
"gpt-5.2": { input: 5.00, output: 20.00, cached: 2.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"gpt-5.2-codex": { input: 5.00, output: 20.00, cached: 2.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"gpt-5.3-codex": { input: 6.00, output: 24.00, cached: 3.00, reasoning: 36.00, cache_creation: 6.00 },
|
||||
"gpt-5.3-codex-xhigh": { input: 10.00, output: 40.00, cached: 5.00, reasoning: 60.00, cache_creation: 10.00 },
|
||||
"gpt-5.3-codex-high": { input: 8.00, output: 32.00, cached: 4.00, reasoning: 48.00, cache_creation: 8.00 },
|
||||
"gpt-5.3-codex-low": { input: 4.00, output: 16.00, cached: 2.00, reasoning: 24.00, cache_creation: 4.00 },
|
||||
"gpt-5.3-codex-none": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
"gpt-5.2": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
"gpt-5.2-codex": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
"gpt-5.3-codex": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
"gpt-5.3-codex-spark": { input: 3.00, output: 12.00, cached: 0.30, reasoning: 12.00, cache_creation: 3.00 },
|
||||
"gpt-5.6": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
|
||||
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
|
||||
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
|
||||
@@ -122,7 +123,7 @@ export const MODEL_PRICING = {
|
||||
* Keyed by provider alias (cc, cx, gc, gh, ...) or provider id (openai, anthropic, ...).
|
||||
*/
|
||||
export const PROVIDER_PRICING = {
|
||||
// GitHub Copilot (gh) — gpt-5.3-codex has different rate than canonical
|
||||
// GitHub Copilot (gh) — explicit override, matches canonical gpt-5.3-codex rate
|
||||
gh: {
|
||||
"gpt-5.3-codex": { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 },
|
||||
},
|
||||
@@ -140,11 +141,11 @@ export const PATTERN_PRICING = [
|
||||
{ pattern: "*-codex-max", pricing: { input: 8.00, output: 32.00, cached: 4.00, reasoning: 48.00, cache_creation: 8.00 } },
|
||||
{ pattern: "*-codex-mini-*", pricing: { input: 1.50, output: 6.00, cached: 0.75, reasoning: 9.00, cache_creation: 1.50 } },
|
||||
{ pattern: "*-codex-mini", pricing: { input: 1.50, output: 6.00, cached: 0.75, reasoning: 9.00, cache_creation: 1.50 } },
|
||||
{ pattern: "*-codex-low", pricing: { input: 4.00, output: 16.00, cached: 2.00, reasoning: 24.00, cache_creation: 4.00 } },
|
||||
{ pattern: "*-codex-none", pricing: { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 } },
|
||||
{ pattern: "*-codex-low", pricing: { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 } },
|
||||
{ pattern: "*-codex-none", pricing: { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 } },
|
||||
{ pattern: "*-codex-spark", pricing: { input: 3.00, output: 12.00, cached: 0.30, reasoning: 12.00, cache_creation: 3.00 } },
|
||||
{ pattern: "codex-*", pricing: { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 } },
|
||||
{ pattern: "*-codex", pricing: { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 } },
|
||||
{ pattern: "codex-*", pricing: { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 } },
|
||||
{ pattern: "*-codex", pricing: { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 } },
|
||||
|
||||
// --- Claude ---
|
||||
{ pattern: "claude-opus-*", pricing: { input: 5.00, output: 25.00, cached: 0.50, reasoning: 25.00, cache_creation: 6.25 } },
|
||||
@@ -161,11 +162,12 @@ export const PATTERN_PRICING = [
|
||||
{ pattern: "gemini-*", pricing: { input: 0.50, output: 3.00, cached: 0.03, reasoning: 4.50, cache_creation: 0.50 } },
|
||||
|
||||
// --- GPT (specific first, generic last) ---
|
||||
{ pattern: "gpt-5.3-*", pricing: { input: 6.00, output: 24.00, cached: 3.00, reasoning: 36.00, cache_creation: 6.00 } },
|
||||
{ pattern: "gpt-5.2-*", pricing: { input: 5.00, output: 20.00, cached: 2.50, reasoning: 30.00, cache_creation: 5.00 } },
|
||||
{ pattern: "gpt-5.1-*", pricing: { input: 4.00, output: 16.00, cached: 2.00, reasoning: 24.00, cache_creation: 4.00 } },
|
||||
{ pattern: "gpt-5-*", pricing: { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 } },
|
||||
{ pattern: "gpt-5*", pricing: { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 } },
|
||||
{ pattern: "gpt-5.6-*", pricing: { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 } },
|
||||
{ pattern: "gpt-5.3-*", pricing: { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 } },
|
||||
{ pattern: "gpt-5.2-*", pricing: { input: 1.75, output: 14.00, cached: 0.175, reasoning: 14.00, cache_creation: 1.75 } },
|
||||
{ pattern: "gpt-5.1-*", pricing: { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 } },
|
||||
{ pattern: "gpt-5-*", pricing: { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 } },
|
||||
{ pattern: "gpt-5*", pricing: { input: 1.25, output: 10.00, cached: 0.625, reasoning: 10.00, cache_creation: 1.25 } },
|
||||
{ pattern: "gpt-4o-*", pricing: { input: 0.15, output: 0.60, cached: 0.075, reasoning: 0.90, cache_creation: 0.15 } },
|
||||
{ pattern: "gpt-4o", pricing: { input: 2.50, output: 10.00, cached: 1.25, reasoning: 15.00, cache_creation: 2.50 } },
|
||||
{ pattern: "gpt-4*", pricing: { input: 2.50, output: 10.00, cached: 1.25, reasoning: 15.00, cache_creation: 2.50 } },
|
||||
|
||||
@@ -14,7 +14,7 @@ export default {
|
||||
},
|
||||
category: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://coding-intl.dashscope.aliyuncs.com/v1/chat/completions",
|
||||
baseUrl: "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions",
|
||||
headers: {},
|
||||
quirks: { preserveCacheControl: true },
|
||||
},
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
import { CLAUDE_API_HEADERS } from "../shared.js";
|
||||
|
||||
export default {
|
||||
id: "anthropic",
|
||||
priority: 30,
|
||||
@@ -19,7 +17,7 @@ export default {
|
||||
baseUrl: "https://api.anthropic.com/v1/messages",
|
||||
format: "claude",
|
||||
headers: {
|
||||
"Anthropic-Version": "2023-06-01",
|
||||
"anthropic-version": "2023-06-01",
|
||||
"Anthropic-Beta": "claude-code-20250219,interleaved-thinking-2025-05-14",
|
||||
},
|
||||
},
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
import { platform, arch } from "os";
|
||||
import { ANTIGRAVITY_OAUTH_CLIENT } from "../shared.js";
|
||||
import { ANTIGRAVITY_IDE_BASE_URL, ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_OAUTH_CLIENT } from "../shared.js";
|
||||
|
||||
export default {
|
||||
id: "antigravity",
|
||||
@@ -20,13 +19,10 @@ export default {
|
||||
category: "oauth",
|
||||
serviceKinds: ["llm", "image"],
|
||||
transport: {
|
||||
baseUrls: [
|
||||
"https://daily-cloudcode-pa.googleapis.com",
|
||||
"https://daily-cloudcode-pa.sandbox.googleapis.com",
|
||||
],
|
||||
baseUrls: [ANTIGRAVITY_IDE_BASE_URL],
|
||||
format: "antigravity",
|
||||
headers: {
|
||||
"User-Agent": "antigravity/1.107.0 darwin/arm64",
|
||||
"User-Agent": ANTIGRAVITY_IDE_USER_AGENT,
|
||||
},
|
||||
retry: {
|
||||
"429": {
|
||||
|
||||
@@ -60,12 +60,10 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "claude-fable-5", name: "Claude Fable 5" },
|
||||
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "claude-opus-4-8", name: "Claude Opus 4.8" },
|
||||
{ id: "claude-opus-4-7", name: "Claude Opus 4.7" },
|
||||
{ id: "claude-opus-4-6", name: "Claude Opus 4.6" },
|
||||
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "claude-opus-4-5-20251101", name: "Claude 4.5 Opus" },
|
||||
{ id: "claude-sonnet-4-5-20250929", name: "Claude 4.5 Sonnet" },
|
||||
{ id: "claude-haiku-4-5-20251001", name: "Claude 4.5 Haiku" },
|
||||
],
|
||||
oauth: {
|
||||
|
||||
@@ -45,22 +45,18 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
|
||||
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
|
||||
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
|
||||
{ id: "gpt-5.6-terra-review", name: "GPT 5.6 Terra Review", upstreamModelId: "gpt-5.6-terra", quotaFamily: "review" },
|
||||
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna" },
|
||||
{ id: "gpt-5.6-luna-review", name: "GPT 5.6 Luna Review", upstreamModelId: "gpt-5.6-luna", quotaFamily: "review" },
|
||||
{ id: "gpt-5.5", name: "GPT 5.5" },
|
||||
{ id: "gpt-5.5-review", name: "GPT 5.5 Review", upstreamModelId: "gpt-5.5", quotaFamily: "review" },
|
||||
{ id: "gpt-5.4", name: "GPT 5.4" },
|
||||
{ id: "gpt-5.4-review", name: "GPT 5.4 Review", upstreamModelId: "gpt-5.4", quotaFamily: "review" },
|
||||
{ id: "gpt-5.4-mini", name: "GPT 5.4 Mini" },
|
||||
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex", name: "GPT 5.3 Codex" },
|
||||
{ id: "gpt-5.3-codex-review", name: "GPT 5.3 Codex Review", upstreamModelId: "gpt-5.3-codex", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-xhigh", name: "GPT 5.3 Codex (xHigh)" },
|
||||
{ id: "gpt-5.3-codex-xhigh-review", name: "GPT 5.3 Codex (xHigh) Review", upstreamModelId: "gpt-5.3-codex-xhigh", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-high", name: "GPT 5.3 Codex (High)" },
|
||||
{ id: "gpt-5.3-codex-high-review", name: "GPT 5.3 Codex (High) Review", upstreamModelId: "gpt-5.3-codex-high", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-low", name: "GPT 5.3 Codex (Low)" },
|
||||
{ id: "gpt-5.3-codex-low-review", name: "GPT 5.3 Codex (Low) Review", upstreamModelId: "gpt-5.3-codex-low", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-none", name: "GPT 5.3 Codex (None)" },
|
||||
{ id: "gpt-5.3-codex-none-review", name: "GPT 5.3 Codex (None) Review", upstreamModelId: "gpt-5.3-codex-none", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
|
||||
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
|
||||
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
|
||||
34
open-sse/providers/registry/featherless.js
Normal file
34
open-sse/providers/registry/featherless.js
Normal file
@@ -0,0 +1,34 @@
|
||||
export default {
|
||||
id: "featherless",
|
||||
priority: 65,
|
||||
alias: "featherless",
|
||||
aliases: [
|
||||
"fl",
|
||||
],
|
||||
uiAlias: "fl",
|
||||
display: {
|
||||
name: "Featherless",
|
||||
icon: "flutter_dash",
|
||||
color: "#111827",
|
||||
textIcon: "FL",
|
||||
website: "https://featherless.ai",
|
||||
notice: {
|
||||
apiKeyUrl: "https://featherless.ai/account/api-keys",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://api.featherless.ai/v1/chat/completions",
|
||||
validateUrl: "https://api.featherless.ai/v1/models",
|
||||
},
|
||||
models: [
|
||||
{ id: "deepseek-ai/DeepSeek-V4-Pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek-ai/DeepSeek-V4-Flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "zai-org/GLM-5.2", name: "GLM 5.2" },
|
||||
{ id: "zai-org/GLM-5.1", name: "GLM 5.1" },
|
||||
{ id: "moonshotai/Kimi-K2.7-Code", name: "Kimi K2.7 Code" },
|
||||
{ id: "moonshotai/Kimi-K2.6", name: "Kimi K2.6" },
|
||||
{ id: "moonshotai/Kimi-K2.5", name: "Kimi K2.5" },
|
||||
],
|
||||
};
|
||||
@@ -18,6 +18,7 @@ export default {
|
||||
transport: {
|
||||
baseUrl: "https://api.githubcopilot.com/chat/completions",
|
||||
responsesUrl: "https://api.githubcopilot.com/responses",
|
||||
messagesUrl: "https://api.githubcopilot.com/v1/messages",
|
||||
headers: {
|
||||
"copilot-integration-id": "vscode-chat",
|
||||
"editor-version": "vscode/1.110.0",
|
||||
@@ -46,6 +47,14 @@ export default {
|
||||
{ id: "gpt-5.3-codex", name: "GPT-5.3 Codex" },
|
||||
{ id: "gpt-5.4", name: "GPT-5.4" },
|
||||
{ id: "gpt-5.4-mini", name: "GPT-5.4 Mini" },
|
||||
// Note: routing to Copilot's Anthropic-native /v1/messages shim (see
|
||||
// executors/github.js) is decided by model-NAME pattern at request time, not by
|
||||
// a static targetFormat field here — Copilot's live model catalog (see
|
||||
// services/copilotModels.js) regularly exposes claude-* models this static list
|
||||
// hasn't caught up with yet (e.g. claude-opus-4.8), and a static per-entry
|
||||
// targetFormat would silently miss those while also double-translating requests
|
||||
// for models that ARE listed here (chatCore.js would pre-translate to Claude
|
||||
// shape, then the executor would translate again). Keep these as plain entries.
|
||||
{ id: "claude-haiku-4.5", name: "Claude Haiku 4.5" },
|
||||
{ id: "claude-opus-4.5", name: "Claude Opus 4.5" },
|
||||
{ id: "claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
|
||||
|
||||
96
open-sse/providers/registry/grok-cli.js
Normal file
96
open-sse/providers/registry/grok-cli.js
Normal file
@@ -0,0 +1,96 @@
|
||||
/**
|
||||
* Grok CLI / Grok Build (cli-chat-proxy.grok.com)
|
||||
*
|
||||
* Source of truth: wire capture of official @xai-official/grok 0.2.99
|
||||
* talking to https://cli-chat-proxy.grok.com (OpenAI Responses API).
|
||||
*
|
||||
* Distinct from:
|
||||
* - `xai` → api.x.ai (API key / xAI API OAuth PKCE)
|
||||
* - `grok-web` → grok.com web SSO cookie
|
||||
*/
|
||||
import {
|
||||
GROK_CLI_BASE_URL,
|
||||
GROK_CLI_CLIENT_IDENTIFIER,
|
||||
GROK_CLI_MODEL,
|
||||
GROK_CLI_USER_AGENT,
|
||||
GROK_CLI_VERSION,
|
||||
} from "../../config/grokCli.js";
|
||||
|
||||
export default {
|
||||
id: "grok-cli",
|
||||
priority: 275,
|
||||
alias: "gcli",
|
||||
aliases: ["grok-build", "gb"],
|
||||
uiAlias: "gcli",
|
||||
display: {
|
||||
name: "Grok CLI (Grok Build)",
|
||||
icon: "auto_awesome",
|
||||
color: "#1DA1F2",
|
||||
textIcon: "GC",
|
||||
website: "https://x.ai",
|
||||
notice: {
|
||||
text: "Sign in with your xAI / Grok account via device code. Uses Grok Build subscription credits (cli-chat-proxy.grok.com).",
|
||||
signupUrl: "https://grok.com/supergrok",
|
||||
},
|
||||
},
|
||||
category: "oauth",
|
||||
authModes: ["oauth"],
|
||||
hasOAuth: true,
|
||||
thinkingConfig: {
|
||||
options: ["low", "medium", "high", "xhigh"],
|
||||
defaultMode: "high",
|
||||
},
|
||||
transport: {
|
||||
baseUrl: `${GROK_CLI_BASE_URL}/responses`,
|
||||
format: "openai-responses",
|
||||
forceStream: true,
|
||||
modelsUrl: `${GROK_CLI_BASE_URL}/models`,
|
||||
userUrl: `${GROK_CLI_BASE_URL}/user`,
|
||||
billingUrl: `${GROK_CLI_BASE_URL}/billing`,
|
||||
clientVersion: GROK_CLI_VERSION,
|
||||
clientIdentifier: GROK_CLI_CLIENT_IDENTIFIER,
|
||||
tokenAuth: "xai-grok-cli",
|
||||
headers: {
|
||||
"User-Agent": GROK_CLI_USER_AGENT,
|
||||
"x-grok-client-identifier": GROK_CLI_CLIENT_IDENTIFIER,
|
||||
"x-grok-client-version": GROK_CLI_VERSION,
|
||||
},
|
||||
// Quota tracker: official CLI polls billing?format=credits + user?include=subscription
|
||||
usage: {
|
||||
url: `${GROK_CLI_BASE_URL}/billing?format=credits`,
|
||||
userUrl: `${GROK_CLI_BASE_URL}/user?include=subscription`,
|
||||
},
|
||||
retry: {
|
||||
429: { attempts: 2, delayMs: 2000 },
|
||||
502: { attempts: 2, delayMs: 1500 },
|
||||
503: { attempts: 2, delayMs: 1500 },
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{
|
||||
id: GROK_CLI_MODEL,
|
||||
name: "Grok Build",
|
||||
contextLength: 500000,
|
||||
maxOutputTokens: 64000,
|
||||
},
|
||||
{ id: "grok-4.5", name: "Grok 4.5" },
|
||||
{ id: "grok-4.5-high", name: "Grok 4.5 (High)", upstreamModelId: "grok-4.5" },
|
||||
{ id: "grok-4.5-medium", name: "Grok 4.5 (Medium)", upstreamModelId: "grok-4.5" },
|
||||
{ id: "grok-4.5-low", name: "Grok 4.5 (Low)", upstreamModelId: "grok-4.5" },
|
||||
],
|
||||
features: {
|
||||
usage: true,
|
||||
},
|
||||
oauth: {
|
||||
// Same public client_id as Grok CLI / existing xai OAuth
|
||||
clientId: "b1a00492-073a-47ea-816f-4c329264a828",
|
||||
deviceCodeUrl: "https://auth.x.ai/oauth2/device/code",
|
||||
tokenUrl: "https://auth.x.ai/oauth2/token",
|
||||
refreshUrl: "https://auth.x.ai/oauth2/token",
|
||||
// HAR scope includes conversations read/write beyond the api-only xai scope
|
||||
scope:
|
||||
"openid profile email offline_access grok-cli:access api:access conversations:read conversations:write",
|
||||
referrer: "grok-build",
|
||||
refreshLeadMs: 5 * 60 * 1000,
|
||||
},
|
||||
};
|
||||
@@ -1,4 +1,4 @@
|
||||
// Auto-generated: static imports of all registry entries
|
||||
// Auto-generated: static imports for all registry entries
|
||||
import p0 from "./alicode-intl.js";
|
||||
import p1 from "./alicode.js";
|
||||
import p2 from "./anthropic.js";
|
||||
@@ -30,72 +30,75 @@ import p27 from "./edge-tts.js";
|
||||
import p28 from "./elevenlabs.js";
|
||||
import p29 from "./exa.js";
|
||||
import p30 from "./fal-ai.js";
|
||||
import p31 from "./firecrawl.js";
|
||||
import p32 from "./fireworks.js";
|
||||
import p33 from "./gemini-cli.js";
|
||||
import p34 from "./gemini.js";
|
||||
import p35 from "./github.js";
|
||||
import p36 from "./gitlab.js";
|
||||
import p37 from "./glm-cn.js";
|
||||
import p38 from "./glm.js";
|
||||
import p39 from "./google-pse.js";
|
||||
import p40 from "./google-tts.js";
|
||||
import p41 from "./grok-web.js";
|
||||
import p42 from "./groq.js";
|
||||
import p43 from "./huggingface.js";
|
||||
import p44 from "./hyperbolic.js";
|
||||
import p45 from "./iflow.js";
|
||||
import p46 from "./inworld.js";
|
||||
import p47 from "./jina-ai.js";
|
||||
import p48 from "./jina-reader.js";
|
||||
import p49 from "./kilocode.js";
|
||||
import p50 from "./kimchi.js";
|
||||
import p51 from "./kimi-coding.js";
|
||||
import p52 from "./kimi.js";
|
||||
import p53 from "./kiro.js";
|
||||
import p54 from "./linkup.js";
|
||||
import p55 from "./local-device.js";
|
||||
import p56 from "./mimo-free.js";
|
||||
import p57 from "./minimax-cn.js";
|
||||
import p58 from "./minimax.js";
|
||||
import p59 from "./mistral.js";
|
||||
import p60 from "./mmf.js";
|
||||
import p61 from "./nanobanana.js";
|
||||
import p62 from "./nebius.js";
|
||||
import p63 from "./nvidia.js";
|
||||
import p64 from "./ollama-local.js";
|
||||
import p65 from "./ollama.js";
|
||||
import p66 from "./openai.js";
|
||||
import p67 from "./opencode-go.js";
|
||||
import p68 from "./opencode.js";
|
||||
import p69 from "./openrouter.js";
|
||||
import p70 from "./perplexity-web.js";
|
||||
import p71 from "./perplexity.js";
|
||||
import p72 from "./playht.js";
|
||||
import p73 from "./qoder.js";
|
||||
import p74 from "./qwen.js";
|
||||
import p75 from "./recraft.js";
|
||||
import p76 from "./runwayml.js";
|
||||
import p77 from "./sdwebui.js";
|
||||
import p78 from "./searchapi.js";
|
||||
import p79 from "./searxng.js";
|
||||
import p80 from "./serper.js";
|
||||
import p81 from "./siliconflow.js";
|
||||
import p82 from "./stability-ai.js";
|
||||
import p83 from "./tavily.js";
|
||||
import p84 from "./together.js";
|
||||
import p85 from "./topaz.js";
|
||||
import p86 from "./tortoise.js";
|
||||
import p87 from "./venice.js";
|
||||
import p88 from "./vercel-ai-gateway.js";
|
||||
import p89 from "./vertex-partner.js";
|
||||
import p90 from "./vertex.js";
|
||||
import p91 from "./volcengine-ark.js";
|
||||
import p92 from "./voyage-ai.js";
|
||||
import p93 from "./xai.js";
|
||||
import p94 from "./xiaomi-mimo.js";
|
||||
import p95 from "./xiaomi-tokenplan.js";
|
||||
import p96 from "./youcom.js";
|
||||
import p31 from "./featherless.js";
|
||||
import p32 from "./firecrawl.js";
|
||||
import p33 from "./fireworks.js";
|
||||
import p34 from "./gemini-cli.js";
|
||||
import p35 from "./gemini.js";
|
||||
import p36 from "./github.js";
|
||||
import p37 from "./gitlab.js";
|
||||
import p38 from "./glm-cn.js";
|
||||
import p39 from "./glm.js";
|
||||
import p40 from "./google-pse.js";
|
||||
import p41 from "./google-tts.js";
|
||||
import p42 from "./grok-cli.js";
|
||||
import p43 from "./grok-web.js";
|
||||
import p44 from "./groq.js";
|
||||
import p45 from "./huggingface.js";
|
||||
import p46 from "./hyperbolic.js";
|
||||
import p47 from "./iflow.js";
|
||||
import p48 from "./inworld.js";
|
||||
import p49 from "./jina-ai.js";
|
||||
import p50 from "./jina-reader.js";
|
||||
import p51 from "./kilocode.js";
|
||||
import p52 from "./kimchi.js";
|
||||
import p53 from "./kimi-coding.js";
|
||||
import p54 from "./kimi.js";
|
||||
import p55 from "./kiro.js";
|
||||
import p56 from "./linkup.js";
|
||||
import p57 from "./local-device.js";
|
||||
import p58 from "./mimo-free.js";
|
||||
import p59 from "./minimax-cn.js";
|
||||
import p60 from "./minimax.js";
|
||||
import p61 from "./mistral.js";
|
||||
import p62 from "./mmf.js";
|
||||
import p63 from "./nanobanana.js";
|
||||
import p64 from "./nebius.js";
|
||||
import p65 from "./nvidia.js";
|
||||
import p66 from "./ollama-local.js";
|
||||
import p67 from "./ollama.js";
|
||||
import p68 from "./openai.js";
|
||||
import p69 from "./opencode-go.js";
|
||||
import p70 from "./opencode.js";
|
||||
import p71 from "./openrouter.js";
|
||||
import p72 from "./perplexity-web.js";
|
||||
import p73 from "./perplexity.js";
|
||||
import p74 from "./perplexity-agent.js";
|
||||
import p75 from "./playht.js";
|
||||
import p76 from "./qoder.js";
|
||||
import p77 from "./qwen.js";
|
||||
import p78 from "./recraft.js";
|
||||
import p79 from "./runwayml.js";
|
||||
import p80 from "./sdwebui.js";
|
||||
import p81 from "./searchapi.js";
|
||||
import p82 from "./searxng.js";
|
||||
import p83 from "./serper.js";
|
||||
import p84 from "./siliconflow.js";
|
||||
import p85 from "./stability-ai.js";
|
||||
import p86 from "./tavily.js";
|
||||
import p87 from "./together.js";
|
||||
import p88 from "./topaz.js";
|
||||
import p89 from "./tortoise.js";
|
||||
import p90 from "./venice.js";
|
||||
import p91 from "./vercel-ai-gateway.js";
|
||||
import p92 from "./vertex-partner.js";
|
||||
import p93 from "./vertex.js";
|
||||
import p94 from "./volcengine-ark.js";
|
||||
import p95 from "./voyage-ai.js";
|
||||
import p96 from "./xai.js";
|
||||
import p97 from "./xiaomi-mimo.js";
|
||||
import p98 from "./xiaomi-tokenplan.js";
|
||||
import p99 from "./youcom.js";
|
||||
|
||||
export default [
|
||||
p0,
|
||||
@@ -194,5 +197,8 @@ export default [
|
||||
p93,
|
||||
p94,
|
||||
p95,
|
||||
p96
|
||||
p96,
|
||||
p97,
|
||||
p98,
|
||||
p99
|
||||
];
|
||||
|
||||
@@ -42,22 +42,53 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
// Opus (added per kiro.dev/changelog/models and kiro.dev/docs/models)
|
||||
{ id: "claude-opus-4.8", name: "Claude Opus 4.8" },
|
||||
{ id: "claude-opus-4.8-thinking", name: "Claude Opus 4.8 (Thinking)" },
|
||||
{ id: "claude-opus-4.8-agentic", name: "Claude Opus 4.8 (Agentic)" },
|
||||
{ id: "claude-opus-4.8-thinking-agentic", name: "Claude Opus 4.8 (Thinking + Agentic)" },
|
||||
{ id: "claude-opus-4.7", name: "Claude Opus 4.7" },
|
||||
{ id: "claude-opus-4.7-thinking", name: "Claude Opus 4.7 (Thinking)" },
|
||||
{ id: "claude-opus-4.7-agentic", name: "Claude Opus 4.7 (Agentic)" },
|
||||
{ id: "claude-opus-4.7-thinking-agentic", name: "Claude Opus 4.7 (Thinking + Agentic)" },
|
||||
{ id: "claude-opus-4.5", name: "Claude Opus 4.5" },
|
||||
{ id: "claude-opus-4.5-thinking", name: "Claude Opus 4.5 (Thinking)" },
|
||||
{ id: "claude-opus-4.5-agentic", name: "Claude Opus 4.5 (Agentic)" },
|
||||
{ id: "claude-opus-4.5-thinking-agentic", name: "Claude Opus 4.5 (Thinking + Agentic)" },
|
||||
// Sonnet
|
||||
{ id: "claude-sonnet-5", name: "Claude Sonnet 5" },
|
||||
{ id: "claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
|
||||
// Haiku
|
||||
{ id: "claude-haiku-4.5", name: "Claude Haiku 4.5" },
|
||||
// Non-Anthropic
|
||||
{ id: "deepseek-3.2", name: "DeepSeek 3.2", strip: ["image","audio"] },
|
||||
{ id: "qwen3-coder-next", name: "Qwen3 Coder Next", strip: ["image","audio"] },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
{ id: "MiniMax-M2.5", name: "MiniMax M2.5" },
|
||||
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol", contextLength: 272000, rateMultiplier: 2.4, upstreamModelId: "gpt-5.6-sol", description: "Experimental preview of OpenAI GPT 5.6 Sol with 272k context window" },
|
||||
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra", contextLength: 272000, rateMultiplier: 1.2, upstreamModelId: "gpt-5.6-terra", description: "Experimental preview of OpenAI GPT 5.6 Terra with 272k context window" },
|
||||
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", contextLength: 272000, rateMultiplier: 0.6, upstreamModelId: "gpt-5.6-luna", description: "Experimental preview of OpenAI GPT 5.6 Luna with 272k context window" },
|
||||
// Thinking variants
|
||||
{ id: "claude-sonnet-5-thinking", name: "Claude Sonnet 5 (Thinking)" },
|
||||
{ id: "claude-sonnet-4.5-thinking", name: "Claude Sonnet 4.5 (Thinking)" },
|
||||
{ id: "claude-haiku-4.5-thinking", name: "Claude Haiku 4.5 (Thinking)" },
|
||||
{ id: "gpt-5.6-sol-thinking", name: "GPT 5.6 Sol (Thinking)", contextLength: 272000, rateMultiplier: 2.4, upstreamModelId: "gpt-5.6-sol", description: "Experimental preview of OpenAI GPT 5.6 Sol with 272k context window" },
|
||||
{ id: "gpt-5.6-terra-thinking", name: "GPT 5.6 Terra (Thinking)", contextLength: 272000, rateMultiplier: 1.2, upstreamModelId: "gpt-5.6-terra", description: "Experimental preview of OpenAI GPT 5.6 Terra with 272k context window" },
|
||||
{ id: "gpt-5.6-luna-thinking", name: "GPT 5.6 Luna (Thinking)", contextLength: 272000, rateMultiplier: 0.6, upstreamModelId: "gpt-5.6-luna", description: "Experimental preview of OpenAI GPT 5.6 Luna with 272k context window" },
|
||||
// Agentic variants
|
||||
{ id: "claude-sonnet-5-agentic", name: "Claude Sonnet 5 (Agentic)" },
|
||||
{ id: "claude-sonnet-4.5-agentic", name: "Claude Sonnet 4.5 (Agentic)" },
|
||||
{ id: "claude-haiku-4.5-agentic", name: "Claude Haiku 4.5 (Agentic)" },
|
||||
{ id: "gpt-5.6-sol-agentic", name: "GPT 5.6 Sol (Agentic)", contextLength: 272000, rateMultiplier: 2.4, upstreamModelId: "gpt-5.6-sol", description: "Experimental preview of OpenAI GPT 5.6 Sol with 272k context window" },
|
||||
{ id: "gpt-5.6-terra-agentic", name: "GPT 5.6 Terra (Agentic)", contextLength: 272000, rateMultiplier: 1.2, upstreamModelId: "gpt-5.6-terra", description: "Experimental preview of OpenAI GPT 5.6 Terra with 272k context window" },
|
||||
{ id: "gpt-5.6-luna-agentic", name: "GPT 5.6 Luna (Agentic)", contextLength: 272000, rateMultiplier: 0.6, upstreamModelId: "gpt-5.6-luna", description: "Experimental preview of OpenAI GPT 5.6 Luna with 272k context window" },
|
||||
// Thinking + Agentic variants
|
||||
{ id: "claude-sonnet-5-thinking-agentic", name: "Claude Sonnet 5 (Thinking + Agentic)" },
|
||||
{ id: "claude-sonnet-4.5-thinking-agentic", name: "Claude Sonnet 4.5 (Thinking + Agentic)" },
|
||||
{ id: "claude-haiku-4.5-thinking-agentic", name: "Claude Haiku 4.5 (Thinking + Agentic)" },
|
||||
{ id: "gpt-5.6-sol-thinking-agentic", name: "GPT 5.6 Sol (Thinking + Agentic)", contextLength: 272000, rateMultiplier: 2.4, upstreamModelId: "gpt-5.6-sol", description: "Experimental preview of OpenAI GPT 5.6 Sol with 272k context window" },
|
||||
{ id: "gpt-5.6-terra-thinking-agentic", name: "GPT 5.6 Terra (Thinking + Agentic)", contextLength: 272000, rateMultiplier: 1.2, upstreamModelId: "gpt-5.6-terra", description: "Experimental preview of OpenAI GPT 5.6 Terra with 272k context window" },
|
||||
{ id: "gpt-5.6-luna-thinking-agentic", name: "GPT 5.6 Luna (Thinking + Agentic)", contextLength: 272000, rateMultiplier: 0.6, upstreamModelId: "gpt-5.6-luna", description: "Experimental preview of OpenAI GPT 5.6 Luna with 272k context window" },
|
||||
],
|
||||
oauth: {
|
||||
ssoOidcEndpoint: "https://oidc.us-east-1.amazonaws.com",
|
||||
|
||||
49
open-sse/providers/registry/perplexity-agent.js
Normal file
49
open-sse/providers/registry/perplexity-agent.js
Normal file
@@ -0,0 +1,49 @@
|
||||
export default {
|
||||
id: "perplexity-agent",
|
||||
priority: 181,
|
||||
alias: "perplexity-agent",
|
||||
aliases: [
|
||||
"pplx-agent",
|
||||
"pplx-responses",
|
||||
],
|
||||
uiAlias: "pa",
|
||||
display: {
|
||||
name: "Perplexity Agent",
|
||||
icon: "travel_explore",
|
||||
color: "#20808D",
|
||||
textIcon: "PA",
|
||||
website: "https://www.perplexity.ai",
|
||||
notice: {
|
||||
text: "Perplexity Agent API exposes GPT, Claude, Gemini, Grok, GLM, Kimi, and Sonar models through one OpenAI-compatible Responses API.",
|
||||
apiKeyUrl: "https://www.perplexity.ai/settings/api",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
authType: "apikey",
|
||||
transport: {
|
||||
baseUrl: "https://api.perplexity.ai/v1/responses",
|
||||
validateUrl: "https://api.perplexity.ai/v1/models",
|
||||
format: "openai-responses",
|
||||
},
|
||||
models: [
|
||||
{ id: "perplexity/sonar", name: "Perplexity Sonar" },
|
||||
{ id: "openai/gpt-5.5", name: "GPT-5.5" },
|
||||
{ id: "openai/gpt-5.4", name: "GPT-5.4" },
|
||||
{ id: "openai/gpt-5.4-mini", name: "GPT-5.4 Mini" },
|
||||
{ id: "anthropic/claude-sonnet-4-6", name: "Claude Sonnet 4.6" },
|
||||
{ id: "anthropic/claude-opus-4-8", name: "Claude Opus 4.8" },
|
||||
{ id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro" },
|
||||
{ id: "xai/grok-4.20-reasoning", name: "Grok 4.20 Reasoning" },
|
||||
{ id: "perplexity/glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "perplexity/kimi-k2.7-code", name: "Kimi K2.7 Code" },
|
||||
{ id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B" },
|
||||
],
|
||||
serviceKinds: ["llm", "webSearch"],
|
||||
searchViaChat: {
|
||||
defaultModel: "perplexity/sonar",
|
||||
endpoint: "https://api.perplexity.ai/v1/responses",
|
||||
pricingUrl: "https://docs.perplexity.ai/docs/agent-api/models",
|
||||
},
|
||||
modelsFetcher: { url: "https://api.perplexity.ai/v1/models", type: "openai" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
@@ -1,3 +1,5 @@
|
||||
import { SEARXNG_URL } from "../../config/runtimeConfig.js";
|
||||
|
||||
export default {
|
||||
id: "searxng",
|
||||
alias: "searxng",
|
||||
@@ -15,7 +17,7 @@ export default {
|
||||
],
|
||||
noAuth: true,
|
||||
searchConfig: {
|
||||
baseUrl: "http://localhost:8888/search",
|
||||
baseUrl: SEARXNG_URL,
|
||||
method: "GET",
|
||||
authType: "none",
|
||||
authHeader: "none",
|
||||
|
||||
@@ -54,13 +54,22 @@ export default {
|
||||
params: ["n", "aspect_ratio", "resolution", "response_format", "size"],
|
||||
kind: "image",
|
||||
},
|
||||
{
|
||||
id: "grok-imagine-video",
|
||||
name: "Grok Imagine Video",
|
||||
params: ["duration", "aspect_ratio", "resolution"],
|
||||
kind: "video",
|
||||
},
|
||||
],
|
||||
serviceKinds: ["llm","imageToText","webSearch","image"],
|
||||
serviceKinds: ["llm", "imageToText", "webSearch", "image", "video"],
|
||||
imageConfig: {
|
||||
baseUrl: "https://api.x.ai/v1/images/generations",
|
||||
editsUrl: "https://api.x.ai/v1/images/edits",
|
||||
bodyFields: ["model", "prompt", "n", "response_format", "aspect_ratio", "resolution", "image", "images"],
|
||||
},
|
||||
// Async video jobs (POST returns { request_id }, GET polls until done/failed).
|
||||
// Docs: https://docs.x.ai/developers/rest-api-reference/inference/videos
|
||||
videoConfig: { baseUrl: "https://api.x.ai/v1/videos" },
|
||||
searchViaChat: {
|
||||
defaultModel: "grok-4.20-reasoning",
|
||||
endpoint: "https://api.x.ai/v1/responses",
|
||||
|
||||
@@ -54,6 +54,13 @@ export const KIMI_CODING_BASE_URL = "https://api.kimi.com/coding/v1/messages";
|
||||
export const OPENAI_COMPAT_BASE = "https://api.openai.com/v1";
|
||||
export const ANTHROPIC_COMPAT_BASE = "https://api.anthropic.com/v1";
|
||||
|
||||
// Official Antigravity IDE Desktop 2.1.1 fingerprint captured from macOS arm64.
|
||||
// Keep this static even when 9router runs on Linux: the provider profile is
|
||||
// intentionally matching the IDE client, not the server host.
|
||||
export const ANTIGRAVITY_IDE_VERSION = "2.1.1";
|
||||
export const ANTIGRAVITY_IDE_BASE_URL = "https://cloudcode-pa.googleapis.com";
|
||||
export const ANTIGRAVITY_IDE_USER_AGENT = `antigravity/ide/${ANTIGRAVITY_IDE_VERSION} darwin/arm64`;
|
||||
|
||||
// Antigravity OAuth client credentials (public CLI client — duplicated in usage.js + src/lib/oauth)
|
||||
export const ANTIGRAVITY_OAUTH_CLIENT = {
|
||||
clientId: "1071006060591-tmhssin2h21lcre235vtolojh4g403ep.apps.googleusercontent.com",
|
||||
|
||||
48
open-sse/providers/thinkingLevels.js
Normal file
48
open-sse/providers/thinkingLevels.js
Normal file
@@ -0,0 +1,48 @@
|
||||
// Resolve valid thinking levels per model — drives UI level picker (suffix "model(level)").
|
||||
// Reuses capabilities.js (thinkingFormat/canDisable) so this file only maps format→levels (DRY).
|
||||
import { getCapabilitiesForModel } from "./capabilities.js";
|
||||
import { matchPattern } from "./pricing.js";
|
||||
|
||||
// Shared level sets (deduped) — verified against provider docs + wire in thinkingUnified.applyFormat.
|
||||
const L = {
|
||||
base: ["none", "low", "medium", "high"], // qwen, step, hunyuan, gemini-budget
|
||||
onOff: ["none", "thinking"], // zai (binary), minimax (adaptive)
|
||||
openai: ["none", "minimal", "low", "medium", "high", "xhigh"], // GPT-5.x / o-series (no "max")
|
||||
levelMax: ["none", "low", "medium", "high", "max"], // claude-adaptive, kimi
|
||||
budgetX: ["none", "low", "medium", "high", "xhigh", "max"], // claude-budget
|
||||
gemini: ["minimal", "low", "medium", "high"], // gemini-3 thinkingLevel (no disable)
|
||||
hiMax: ["none", "high", "max"], // deepseek (low/med→high, xhigh→max)
|
||||
};
|
||||
|
||||
// thinkingFormat → valid selectable levels (source of truth for UI options).
|
||||
const FORMAT_LEVELS = {
|
||||
openai: L.openai,
|
||||
"claude-adaptive": L.levelMax,
|
||||
"claude-budget": L.budgetX,
|
||||
"gemini-level": L.gemini,
|
||||
"gemini-budget": L.base,
|
||||
zai: L.onOff,
|
||||
qwen: L.base,
|
||||
kimi: L.levelMax,
|
||||
deepseek: L.hiMax,
|
||||
minimax: L.onOff,
|
||||
hunyuan: L.base,
|
||||
step: L.base,
|
||||
};
|
||||
|
||||
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
|
||||
const PATTERN_THINKING = [
|
||||
// gpt-5.6-sol accepts max (maps to xhigh on wire); live probe rejected ultra.
|
||||
{ pattern: "*gpt-5.6-sol*", levels: ["none", "minimal", "low", "medium", "high", "xhigh", "max"] },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
];
|
||||
|
||||
// Returns valid thinking levels for a model, or null when the model has no reasoning.
|
||||
export function getThinkingLevels(provider, model) {
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (!caps.reasoning) return null;
|
||||
const hit = PATTERN_THINKING.find((p) => matchPattern(p.pattern, model));
|
||||
let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
|
||||
if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none");
|
||||
return levels;
|
||||
}
|
||||
@@ -1,9 +1,10 @@
|
||||
// Port of auto_detect_filter (rtk/src/cmds/system/pipe_cmd.rs:132-188) + JS extras
|
||||
// Order: git-diff → git-status → build-output → grep → find → tree → ls → search-list
|
||||
// → read-numbered → dedup-log → smart-truncate → null
|
||||
// Detection order: git-log → git-diff → git-status → build-output → grep → find → tree → ls → search-list
|
||||
// → read-numbered → dedup-log → smart-truncate → null
|
||||
import { DETECT_WINDOW, READ_NUMBERED_MIN_HIT_RATIO, SMART_TRUNCATE_MIN_LINES } from "./constants.js";
|
||||
import { gitDiff } from "./filters/gitDiff.js";
|
||||
import { gitStatus } from "./filters/gitStatus.js";
|
||||
import { gitLog } from "./filters/gitLog.js";
|
||||
import { buildOutput } from "./filters/buildOutput.js";
|
||||
import { grep } from "./filters/grep.js";
|
||||
import { find } from "./filters/find.js";
|
||||
@@ -17,6 +18,7 @@ import { searchList, SEARCH_LIST_HEADER_RE } from "./filters/searchList.js";
|
||||
const RE_GIT_DIFF = /^diff --git /m;
|
||||
const RE_GIT_DIFF_HUNK = /^@@ /m;
|
||||
const RE_GIT_STATUS = /^On branch |^nothing to commit|^Changes (not |to be )|^Untracked files:/m;
|
||||
const RE_GIT_LOG = /^[*|/\\ ]*commit [0-9a-f]{7,40}$/m;
|
||||
const RE_PORCELAIN = /^[ MADRCU?!][ MADRCU?!] \S/m;
|
||||
const RE_BUILD_OUTPUT = /^(npm (warn|error|ERR!)|yarn (warn|error)|\s*Compiling\s+\S+|\s*Downloading\s+\S+|added \d+ package|\[ERROR\]|BUILD (SUCCESS|FAILED)|\s*Finished\s+|Successfully (installed|built)|ERROR:)/im;
|
||||
const RE_TREE_GLYPH = /[├└]──|│ /;
|
||||
@@ -27,6 +29,7 @@ export function autoDetectFilter(text) {
|
||||
// Rust: floor_char_boundary to avoid UTF-8 split — JS .slice() by char is safe
|
||||
const head = text.length > DETECT_WINDOW ? text.slice(0, DETECT_WINDOW) : text;
|
||||
|
||||
if (RE_GIT_LOG.test(head)) return gitLog;
|
||||
if (RE_GIT_DIFF.test(head) || RE_GIT_DIFF_HUNK.test(head)) return gitDiff;
|
||||
if (RE_GIT_STATUS.test(head)) return gitStatus;
|
||||
|
||||
@@ -81,6 +84,11 @@ function isGrepLine(line) {
|
||||
function isPathLike(line) {
|
||||
const t = line.trim();
|
||||
if (t.length === 0) return false;
|
||||
// A drive-letter prefix (e.g. "C:\Users\me" or "C:/Users/me") marks a
|
||||
// Windows absolute path, so treat the whole line as path-like. Trailing
|
||||
// colons (e.g. "C:\path\file.js:10") are tolerated, matching grep-style
|
||||
// suffixes on Windows dumps.
|
||||
if (/^[A-Za-z]:[\\/]/.test(t)) return true;
|
||||
if (t.includes(":")) return false;
|
||||
return t.startsWith(".") || t.startsWith("/") || t.includes("/");
|
||||
}
|
||||
|
||||
@@ -18,6 +18,14 @@ const SHARED_AUTO_CLARITY = "Auto-Clarity: drop caveman for security warnings, i
|
||||
|
||||
const SHARED_PERSISTENCE = "ACTIVE EVERY RESPONSE. No revert after many turns. No filler drift. Still active if unsure.";
|
||||
|
||||
const SHARED_NO_INVENTED_ABBREV = "No invented abbreviations. Standard well-known tech acronyms (DB, API, HTTP, URL, JSON, ID, OS, CPU) OK. Names of code symbols, function names, API names, error strings: keep verbatim.";
|
||||
|
||||
const SHARED_PRESERVE_LANGUAGE = "Preserve the user's dominant language. User wrote Vietnamese, reply Vietnamese. User wrote English, reply English. Wenyan/classical-Chinese levels override this language-preservation rule. Code identifiers, error strings, file paths, commands: keep in their original form regardless of language.";
|
||||
|
||||
const SHARED_NO_SELF_REFERENCE = 'No self-reference. Do not name or announce the style (no "caveman mode", no "me caveman think", no "compressed mode active"). Just respond.';
|
||||
|
||||
const SHARED_NO_DECORATION = 'No decorative emoji. No narrating tool calls ("I will now search", "I used X to find Y"). No status phrases ("Sure!", "Of course!", "I\'d be happy to"). No causal arrow shorthand ("A -> B -> fails"). State the thing, the action, the reason. Then next step.';
|
||||
|
||||
export const CAVEMAN_PROMPTS = {
|
||||
[CAVEMAN_LEVELS.LITE]: [
|
||||
"Respond tersely. Keep grammar and full sentences but drop filler, hedging and pleasantries (just/really/basically/sure/of course/I'd be happy to).",
|
||||
@@ -26,6 +34,10 @@ export const CAVEMAN_PROMPTS = {
|
||||
SHARED_BOUNDARIES,
|
||||
SHARED_AUTO_CLARITY,
|
||||
SHARED_PERSISTENCE,
|
||||
SHARED_NO_INVENTED_ABBREV,
|
||||
SHARED_PRESERVE_LANGUAGE,
|
||||
SHARED_NO_SELF_REFERENCE,
|
||||
SHARED_NO_DECORATION,
|
||||
].join(" "),
|
||||
|
||||
[CAVEMAN_LEVELS.FULL]: [
|
||||
@@ -36,16 +48,24 @@ export const CAVEMAN_PROMPTS = {
|
||||
SHARED_BOUNDARIES,
|
||||
SHARED_AUTO_CLARITY,
|
||||
SHARED_PERSISTENCE,
|
||||
SHARED_NO_INVENTED_ABBREV,
|
||||
SHARED_PRESERVE_LANGUAGE,
|
||||
SHARED_NO_SELF_REFERENCE,
|
||||
SHARED_NO_DECORATION,
|
||||
].join(" "),
|
||||
|
||||
[CAVEMAN_LEVELS.ULTRA]: [
|
||||
"Respond ultra-terse. Maximum compression. Telegraphic.",
|
||||
"Abbreviate (DB/auth/config/req/res/fn/impl), strip conjunctions, use arrows for causality (X → Y). One word when one word enough.",
|
||||
"Pattern: [thing] → [result]. [fix].",
|
||||
"Strip conjunctions. One word when one word enough.",
|
||||
"Pattern: [thing] [action] [reason]. [next step].",
|
||||
SHARED_EXAMPLES,
|
||||
SHARED_BOUNDARIES,
|
||||
SHARED_AUTO_CLARITY,
|
||||
SHARED_PERSISTENCE,
|
||||
SHARED_NO_INVENTED_ABBREV,
|
||||
SHARED_PRESERVE_LANGUAGE,
|
||||
SHARED_NO_SELF_REFERENCE,
|
||||
SHARED_NO_DECORATION,
|
||||
].join(" "),
|
||||
|
||||
[CAVEMAN_LEVELS.WENYAN_LITE]: [
|
||||
@@ -55,6 +75,10 @@ export const CAVEMAN_PROMPTS = {
|
||||
SHARED_BOUNDARIES,
|
||||
SHARED_AUTO_CLARITY,
|
||||
SHARED_PERSISTENCE,
|
||||
SHARED_NO_INVENTED_ABBREV,
|
||||
SHARED_PRESERVE_LANGUAGE,
|
||||
SHARED_NO_SELF_REFERENCE,
|
||||
SHARED_NO_DECORATION,
|
||||
].join(" "),
|
||||
|
||||
[CAVEMAN_LEVELS.WENYAN]: [
|
||||
@@ -65,6 +89,10 @@ export const CAVEMAN_PROMPTS = {
|
||||
SHARED_BOUNDARIES,
|
||||
SHARED_AUTO_CLARITY,
|
||||
SHARED_PERSISTENCE,
|
||||
SHARED_NO_INVENTED_ABBREV,
|
||||
SHARED_PRESERVE_LANGUAGE,
|
||||
SHARED_NO_SELF_REFERENCE,
|
||||
SHARED_NO_DECORATION,
|
||||
].join(" "),
|
||||
|
||||
[CAVEMAN_LEVELS.WENYAN_ULTRA]: [
|
||||
@@ -74,5 +102,9 @@ export const CAVEMAN_PROMPTS = {
|
||||
SHARED_BOUNDARIES,
|
||||
SHARED_AUTO_CLARITY,
|
||||
SHARED_PERSISTENCE,
|
||||
SHARED_NO_INVENTED_ABBREV,
|
||||
SHARED_PRESERVE_LANGUAGE,
|
||||
SHARED_NO_SELF_REFERENCE,
|
||||
SHARED_NO_DECORATION,
|
||||
].join(" "),
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@ export const MIN_COMPRESS_SIZE = 500; // bytes; skip tiny blobs
|
||||
export const DETECT_WINDOW = 1024; // autodetect peeks first N chars
|
||||
export const GIT_DIFF_HUNK_MAX_LINES = 100; // per-hunk line cap
|
||||
export const GIT_DIFF_CONTEXT_KEEP = 3; // context lines around changes
|
||||
export const GIT_LOG_MAX_LINES = 200; // gitLog line cap
|
||||
export const DEDUP_LINE_MAX = 2000; // dedupLog truncation cap
|
||||
|
||||
// Rust pipe_cmd.rs parity caps
|
||||
|
||||
@@ -9,16 +9,17 @@ export function find(input) {
|
||||
const byDir = new Map();
|
||||
|
||||
for (const path of lines) {
|
||||
const lastSlash = path.lastIndexOf("/");
|
||||
// Accept both Unix ("/a/b") and Windows ("C:\a\b") separators
|
||||
const lastSep = Math.max(path.lastIndexOf("/"), path.lastIndexOf("\\"));
|
||||
let dir;
|
||||
let basename;
|
||||
if (lastSlash === -1) {
|
||||
if (lastSep === -1) {
|
||||
dir = ".";
|
||||
basename = path;
|
||||
} else {
|
||||
// Rust: PathBuf::from(path).parent().display() + file_name().display()
|
||||
dir = path.slice(0, lastSlash) || "/";
|
||||
basename = path.slice(lastSlash + 1);
|
||||
dir = path.slice(0, lastSep) || "/";
|
||||
basename = path.slice(lastSep + 1);
|
||||
}
|
||||
if (!byDir.has(dir)) byDir.set(dir, []);
|
||||
byDir.get(dir).push(basename);
|
||||
@@ -31,7 +32,8 @@ export function find(input) {
|
||||
const showDirs = dirs.slice(0, FIND_TOTAL_DIR_MAX);
|
||||
for (const dir of showDirs) {
|
||||
const files = byDir.get(dir);
|
||||
out += `${dir}/ (${files.length})\n`;
|
||||
const dirLabel = dir.replace(/\\/g, "/");
|
||||
out += `${dirLabel}/ (${files.length})\n`;
|
||||
const showFiles = files.slice(0, FIND_PER_DIR_MAX);
|
||||
for (const f of showFiles) out += ` ${f}\n`;
|
||||
if (files.length > FIND_PER_DIR_MAX) {
|
||||
|
||||
99
open-sse/rtk/filters/gitLog.js
Normal file
99
open-sse/rtk/filters/gitLog.js
Normal file
@@ -0,0 +1,99 @@
|
||||
// JS-native git-log filter
|
||||
// Compresses `git log` output: keeps commit headers, subjects, Author/Date;
|
||||
// drops body padding, decoration, embedded diff lines.
|
||||
import { GIT_LOG_MAX_LINES } from "../constants.js";
|
||||
|
||||
export function gitLog(text, maxLines = GIT_LOG_MAX_LINES) {
|
||||
if (!text) return "";
|
||||
|
||||
const input = String(text);
|
||||
const lines = input.split("\n");
|
||||
const out = [];
|
||||
let skipped = 0;
|
||||
let inCommit = false;
|
||||
let subjectSeen = false;
|
||||
|
||||
function pushLine(l) {
|
||||
if (out.length < maxLines) {
|
||||
out.push(l);
|
||||
return true;
|
||||
}
|
||||
skipped++;
|
||||
return false;
|
||||
}
|
||||
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const raw = lines[i];
|
||||
const line = raw.trimEnd();
|
||||
const trimmed = line.trim();
|
||||
|
||||
// commit <sha> header — starts new commit entry
|
||||
// Also matched with leading graph decoration (`* commit abc1234...` — --graph without --oneline)
|
||||
if (/^commit [0-9a-f]{7,40}$/i.test(trimmed) || /^[*|/\\ ]+commit [0-9a-f]{7,40}/i.test(trimmed)) {
|
||||
inCommit = true;
|
||||
subjectSeen = false;
|
||||
pushLine(line);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (inCommit) {
|
||||
// Author / Date — keep as-is (already column 0 in raw, or graph-prefix stripped by commit-header match)
|
||||
if (/^[*|/\\ ]*(Author|Date):/i.test(trimmed)) {
|
||||
pushLine(trimmed);
|
||||
continue;
|
||||
}
|
||||
// blank — skip
|
||||
if (trimmed === "") continue;
|
||||
// indented subject (4 spaces, optionally preceded by graph decoration) — first one is subject
|
||||
if (!subjectSeen && /^[*|/\\ ]* \S/.test(line)) {
|
||||
pushLine(" Subject: " + trimmed);
|
||||
subjectSeen = true;
|
||||
continue;
|
||||
}
|
||||
// stat summary: "N file(s) changed, N insertions(+), N deletions(-)"
|
||||
if (/^\d+ file\w* changed/.test(trimmed)) {
|
||||
pushLine(" " + trimmed);
|
||||
continue;
|
||||
}
|
||||
// embedded diff header — one-line marker
|
||||
if (/^diff --git /.test(trimmed)) {
|
||||
pushLine(" ... diff body omitted");
|
||||
continue;
|
||||
}
|
||||
// everything else in commit body — drop
|
||||
continue;
|
||||
}
|
||||
|
||||
// Not in a commit block (--oneline / --graph modes):
|
||||
|
||||
// Graph decoration + sha + subject: "*|/\\ <sha7> <subject>"
|
||||
const graphMatch = trimmed.match(/^[*|/\\ ]+([0-9a-f]{7,40}\s+.+)/i);
|
||||
if (graphMatch) {
|
||||
pushLine(graphMatch[1]);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Plain oneline: "<sha7> <subject>"
|
||||
if (/^[0-9a-f]{7,40}\s+/.test(trimmed)) {
|
||||
pushLine(trimmed);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Pure graph decoration (no sha) — drop
|
||||
if (/^[*|/\\ ]+$/.test(trimmed) && /[*|/\\]/.test(trimmed)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// catch-all pass-through
|
||||
pushLine(trimmed);
|
||||
}
|
||||
|
||||
if (skipped > 0) out.push(`... (${skipped} more lines)`);
|
||||
|
||||
const result = out.join("\n");
|
||||
if (!result && input) return input;
|
||||
if (result.length > input.length) return input;
|
||||
return result;
|
||||
}
|
||||
|
||||
gitLog.filterName = "git-log";
|
||||
@@ -18,6 +18,8 @@ function jsonBytes(value) {
|
||||
function messagePayload(body) {
|
||||
if (Array.isArray(body?.messages)) return body.messages;
|
||||
if (Array.isArray(body?.input)) return body.input;
|
||||
const kiro = collectKiroHeadroomMessages(body);
|
||||
if (kiro) return kiro.messages;
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -81,6 +83,121 @@ function hasUnsafeResponsesInputForCompression(body) {
|
||||
});
|
||||
}
|
||||
|
||||
function collectKiroHeadroomMessages(body) {
|
||||
const state = body?.conversationState;
|
||||
if (!state || typeof state !== "object") return null;
|
||||
|
||||
const messages = [];
|
||||
const targets = [];
|
||||
|
||||
const addTextTarget = (role, text, target, extra = {}) => {
|
||||
if (typeof text !== "string") return;
|
||||
messages.push({ role, content: text, ...extra });
|
||||
targets.push(target);
|
||||
};
|
||||
|
||||
const toToolCalls = (toolUses) => {
|
||||
if (!Array.isArray(toolUses) || toolUses.length === 0) return undefined;
|
||||
const calls = toolUses.map((toolUse) => ({
|
||||
id: toolUse?.toolUseId,
|
||||
type: "function",
|
||||
function: {
|
||||
name: toolUse?.name || "",
|
||||
arguments: JSON.stringify(toolUse?.input || {}),
|
||||
},
|
||||
})).filter((call) => call.id || call.function.name);
|
||||
return calls.length > 0 ? calls : undefined;
|
||||
};
|
||||
|
||||
const visit = (item) => {
|
||||
const user = item?.userInputMessage;
|
||||
if (user) {
|
||||
addTextTarget("system", user.systemInstruction, { object: user, key: "systemInstruction" });
|
||||
addTextTarget("user", user.content, { object: user, key: "content" });
|
||||
|
||||
const toolResults = user.userInputMessageContext?.toolResults;
|
||||
if (Array.isArray(toolResults)) {
|
||||
for (const toolResult of toolResults) {
|
||||
const content = toolResult?.content;
|
||||
if (!Array.isArray(content)) continue;
|
||||
for (const part of content) {
|
||||
addTextTarget(
|
||||
"tool",
|
||||
part?.text,
|
||||
{ object: part, key: "text" },
|
||||
toolResult?.toolUseId ? { tool_call_id: toolResult.toolUseId } : {}
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const assistant = item?.assistantResponseMessage;
|
||||
if (assistant) {
|
||||
const toolCalls = toToolCalls(assistant.toolUses);
|
||||
addTextTarget(
|
||||
"assistant",
|
||||
assistant.content,
|
||||
{ object: assistant, key: "content" },
|
||||
toolCalls ? { tool_calls: toolCalls } : {}
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
if (Array.isArray(state.history)) {
|
||||
for (const item of state.history) visit(item);
|
||||
}
|
||||
if (state.currentMessage) visit(state.currentMessage);
|
||||
|
||||
return messages.length > 0 ? { messages, targets } : null;
|
||||
}
|
||||
|
||||
function textFromHeadroomMessage(message) {
|
||||
const content = message?.content;
|
||||
if (typeof content === "string") return content;
|
||||
if (!Array.isArray(content)) return null;
|
||||
|
||||
const parts = [];
|
||||
for (const part of content) {
|
||||
if (typeof part === "string") {
|
||||
parts.push(part);
|
||||
} else if (typeof part?.text === "string") {
|
||||
parts.push(part.text);
|
||||
}
|
||||
}
|
||||
return parts.length > 0 ? parts.join("\n") : null;
|
||||
}
|
||||
|
||||
function applyKiroHeadroomMessages(projection, compressedMessages, diagnostics) {
|
||||
if (!Array.isArray(compressedMessages) || compressedMessages.length !== projection.messages.length) {
|
||||
setDiagnostic(diagnostics, "proxy response did not match Kiro message count");
|
||||
return false;
|
||||
}
|
||||
|
||||
const updates = [];
|
||||
for (let i = 0; i < projection.messages.length; i++) {
|
||||
const expected = projection.messages[i];
|
||||
const actual = compressedMessages[i];
|
||||
if (!actual || actual.role !== expected.role) {
|
||||
setDiagnostic(diagnostics, "proxy response did not preserve Kiro message order");
|
||||
return false;
|
||||
}
|
||||
|
||||
const text = textFromHeadroomMessage(actual);
|
||||
if (text === null) {
|
||||
setDiagnostic(diagnostics, "proxy response missing Kiro text content");
|
||||
return false;
|
||||
}
|
||||
updates.push({ target: projection.targets[i], text });
|
||||
}
|
||||
|
||||
for (const update of updates) {
|
||||
update.target.object[update.target.key] = update.text;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// POST messages to Headroom /v1/compress; returns compressed messages + stats or null.
|
||||
async function callCompress(url, messages, model, timeoutMs, compressUserMessages, diagnostics) {
|
||||
const endpoint = buildCompressEndpoint(url);
|
||||
@@ -171,6 +288,22 @@ export async function compressWithHeadroom(body, { enabled, url, model, format,
|
||||
return data;
|
||||
}
|
||||
|
||||
// Kiro shape: conversationState.history/currentMessage are projected to
|
||||
// OpenAI messages for the proxy, then copied back into the original Kiro
|
||||
// fields. Keep the provider payload shape intact for Kiro's executor.
|
||||
if (format === "kiro") {
|
||||
const projection = collectKiroHeadroomMessages(body);
|
||||
if (!projection) {
|
||||
setDiagnostic(diagnostics, "Kiro request did not project to messages[]");
|
||||
return null;
|
||||
}
|
||||
const data = await callCompress(url, projection.messages, model, timeoutMs, compressUserMessages, diagnostics || {});
|
||||
if (!data) return null;
|
||||
if (!applyKiroHeadroomMessages(projection, data.messages, diagnostics)) return null;
|
||||
if (diagnostics) diagnostics.after = captureSizeSnapshot(body);
|
||||
return data;
|
||||
}
|
||||
|
||||
// OpenAI shape: messages/input go straight to the proxy.
|
||||
const key = Array.isArray(body.messages) ? "messages"
|
||||
: Array.isArray(body.input) ? "input"
|
||||
|
||||
104
open-sse/rtk/pxpipe.js
Normal file
104
open-sse/rtk/pxpipe.js
Normal file
@@ -0,0 +1,104 @@
|
||||
// PXPIPE: render bulky Claude-format context as dense PNGs via pxpipe-proxy's
|
||||
// library API (transformAnthropicMessages). Fail-open like every token saver:
|
||||
// any error/timeout returns { body: null, summary } and leaves the request untouched.
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 15000;
|
||||
const DEFAULT_MIN_CHARS = 25000;
|
||||
// pxpipe's own profitability gate assumes ~4 chars/token; reuse it for the
|
||||
// estimated before/after numbers surfaced in stats (marked "estimated" in UI).
|
||||
const EST_CHARS_PER_TOKEN = 4;
|
||||
|
||||
function bodyChars(body) {
|
||||
try {
|
||||
return JSON.stringify(body)?.length || 0;
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
function estTokens(chars) {
|
||||
return Math.round(chars / EST_CHARS_PER_TOKEN);
|
||||
}
|
||||
|
||||
function skipped(reason, extra = {}) {
|
||||
return { body: null, summary: { applied: false, reason, ...extra } };
|
||||
}
|
||||
|
||||
// Transform a Claude-format request body through pxpipe. Returns
|
||||
// { body: <new body object> | null, summary } — body is null when nothing changed.
|
||||
// opts.transform is injected by the host (src side) so open-sse stays free of
|
||||
// filesystem/install concerns and remains usable standalone.
|
||||
export async function compressWithPxpipe(body, { enabled, format, model, minChars, timeoutMs, transform } = {}) {
|
||||
if (!enabled) return skipped("disabled");
|
||||
if (typeof transform !== "function") return skipped("not_installed");
|
||||
if (!body) return skipped("missing_body");
|
||||
if (format !== FORMATS.CLAUDE) return skipped("unsupported_format", { detail: format });
|
||||
|
||||
const startedAt = Date.now();
|
||||
const originalChars = bodyChars(body);
|
||||
const threshold = Number(minChars) > 0 ? Number(minChars) : DEFAULT_MIN_CHARS;
|
||||
if (originalChars < threshold) {
|
||||
return skipped("below_threshold", { originalChars, threshold });
|
||||
}
|
||||
|
||||
try {
|
||||
const encoded = new TextEncoder().encode(JSON.stringify(body));
|
||||
const budget = Number(timeoutMs) > 0 ? Number(timeoutMs) : DEFAULT_TIMEOUT_MS;
|
||||
// transformAnthropicMessages is local CPU work and can't be aborted; race a
|
||||
// timer and discard the result if it loses (input body is never mutated).
|
||||
const result = await Promise.race([
|
||||
transform({
|
||||
body: encoded,
|
||||
model,
|
||||
options: { minCompressChars: threshold },
|
||||
}),
|
||||
new Promise((resolve) => setTimeout(() => resolve(null), budget)),
|
||||
]);
|
||||
if (!result) return skipped("timeout", { originalChars, durationMs: Date.now() - startedAt });
|
||||
if (!result.applied) {
|
||||
return skipped(result.reason || "passthrough", {
|
||||
detail: result.detail,
|
||||
originalChars,
|
||||
durationMs: Date.now() - startedAt,
|
||||
});
|
||||
}
|
||||
|
||||
const newBody = JSON.parse(new TextDecoder().decode(result.body));
|
||||
const compressedBodyChars = bodyChars(newBody);
|
||||
const info = result.info || {};
|
||||
const imagedChars = info.compressedChars || 0;
|
||||
// The transformed body is BIGGER in bytes (base64 PNGs) but cheaper in tokens:
|
||||
// images bill by pixels (Anthropic: pixels/750), not by encoded length. So the
|
||||
// after-estimate is remaining-text tokens + image tokens — never chars/4 of the
|
||||
// new body. Provider-billed usage recorded per request stays the ground truth.
|
||||
const imageTokensEst = info.imageTokens
|
||||
|| (info.imagePixels ? Math.round(info.imagePixels / 750) : (info.imageCount || 0) * 4761);
|
||||
const summary = {
|
||||
applied: true,
|
||||
reason: "applied",
|
||||
originalChars,
|
||||
compressedBodyChars,
|
||||
imagedChars,
|
||||
imageCount: info.imageCount || 0,
|
||||
imageBytes: info.imageBytes || 0,
|
||||
tokensBeforeEst: info.baselineTokens || estTokens(originalChars),
|
||||
tokensAfterEst: estTokens(Math.max(0, originalChars - imagedChars)) + imageTokensEst,
|
||||
durationMs: Date.now() - startedAt,
|
||||
cacheOwnsControl: result.cache?.ownsCacheControl === true,
|
||||
};
|
||||
summary.tokensSavedEst = Math.max(0, summary.tokensBeforeEst - summary.tokensAfterEst);
|
||||
summary.savedPct = summary.tokensBeforeEst > 0
|
||||
? +((summary.tokensSavedEst / summary.tokensBeforeEst) * 100).toFixed(2)
|
||||
: 0;
|
||||
return { body: newBody, summary };
|
||||
} catch (e) {
|
||||
return skipped("transform_error", { detail: e?.message || String(e), originalChars, durationMs: Date.now() - startedAt });
|
||||
}
|
||||
}
|
||||
|
||||
export function formatPxpipeLog(summary) {
|
||||
if (!summary) return null;
|
||||
if (!summary.applied) return null;
|
||||
return `imaged ${summary.imagedChars}ch → ${summary.imageCount} image(s) | est ${summary.tokensBeforeEst}→${summary.tokensAfterEst} tokens (-${summary.savedPct}%) | ${summary.durationMs}ms`;
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
import { FILTERS } from "./constants.js";
|
||||
import { gitDiff } from "./filters/gitDiff.js";
|
||||
import { gitStatus } from "./filters/gitStatus.js";
|
||||
import { gitLog } from "./filters/gitLog.js";
|
||||
import { grep } from "./filters/grep.js";
|
||||
import { find } from "./filters/find.js";
|
||||
import { dedupLog } from "./filters/dedupLog.js";
|
||||
@@ -13,6 +14,7 @@ import { searchList } from "./filters/searchList.js";
|
||||
const REGISTRY = {
|
||||
[FILTERS.GIT_DIFF]: gitDiff,
|
||||
[FILTERS.GIT_STATUS]: gitStatus,
|
||||
[FILTERS.GIT_LOG]: gitLog,
|
||||
[FILTERS.GREP]: grep,
|
||||
[FILTERS.FIND]: find,
|
||||
[FILTERS.DEDUP_LOG]: dedupLog,
|
||||
|
||||
127
open-sse/services/grokCliModels.js
Normal file
127
open-sse/services/grokCliModels.js
Normal file
@@ -0,0 +1,127 @@
|
||||
import {
|
||||
GROK_CLI_BASE_URL,
|
||||
GROK_CLI_CLIENT_IDENTIFIER,
|
||||
GROK_CLI_MODEL,
|
||||
GROK_CLI_USER_AGENT,
|
||||
GROK_CLI_VERSION,
|
||||
} from "../config/grokCli.js";
|
||||
import { refreshProviderCredentials } from "./oauthCredentialManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
|
||||
const MODELS_URL = `${GROK_CLI_BASE_URL}/models`;
|
||||
|
||||
function modelEntries(data) {
|
||||
const value = Array.isArray(data) ? data : data?.data ?? data?.models ?? data?.results ?? [];
|
||||
if (Array.isArray(value)) return value.map((item) => [null, item]);
|
||||
if (value && typeof value === "object") return Object.entries(value);
|
||||
return [];
|
||||
}
|
||||
|
||||
export function parseGrokCliModels(data) {
|
||||
const seen = new Set();
|
||||
const models = [];
|
||||
|
||||
for (const [key, raw] of modelEntries(data)) {
|
||||
const item = typeof raw === "string" ? { id: raw } : raw;
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) continue;
|
||||
const id = String(
|
||||
item.id ?? item.model_id ?? item.modelId ?? item.model ?? item.slug ?? key ?? item.name ?? "",
|
||||
).trim();
|
||||
if (!id || seen.has(id)) continue;
|
||||
seen.add(id);
|
||||
|
||||
const model = {
|
||||
...item,
|
||||
id,
|
||||
name: item.display_name ?? item.displayName ?? item.name ?? id,
|
||||
};
|
||||
const contextLength = Number(
|
||||
item.context_length ?? item.contextLength ?? item.context_window ?? item.contextWindow,
|
||||
);
|
||||
const maxOutputTokens = Number(item.max_output_tokens ?? item.maxOutputTokens);
|
||||
if (Number.isFinite(contextLength) && contextLength > 0) model.contextLength = contextLength;
|
||||
if (Number.isFinite(maxOutputTokens) && maxOutputTokens > 0) {
|
||||
model.maxOutputTokens = maxOutputTokens;
|
||||
}
|
||||
if (id === GROK_CLI_MODEL) {
|
||||
model.contextLength ||= 500000;
|
||||
model.maxOutputTokens ||= 64000;
|
||||
}
|
||||
models.push(model);
|
||||
}
|
||||
|
||||
return models;
|
||||
}
|
||||
|
||||
function buildHeaders(accessToken, providerSpecificData = {}) {
|
||||
const headers = {
|
||||
Authorization: `Bearer ${accessToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": GROK_CLI_USER_AGENT,
|
||||
"x-xai-token-auth": "xai-grok-cli",
|
||||
"x-grok-client-version": GROK_CLI_VERSION,
|
||||
"x-grok-client-identifier": GROK_CLI_CLIENT_IDENTIFIER,
|
||||
"x-grok-client-mode": "headless",
|
||||
};
|
||||
const email = providerSpecificData?.email;
|
||||
const userId = providerSpecificData?.userId || providerSpecificData?.principalId;
|
||||
if (email) headers["x-email"] = email;
|
||||
if (userId) headers["x-userid"] = userId;
|
||||
return headers;
|
||||
}
|
||||
|
||||
export async function resolveGrokCliModels(credentials, options = {}) {
|
||||
const {
|
||||
fetchFn = proxyAwareFetch,
|
||||
log = console,
|
||||
proxyOptions = null,
|
||||
onCredentialsRefreshed,
|
||||
} = options;
|
||||
let accessToken = credentials?.accessToken;
|
||||
if (!accessToken) return { models: [], warning: "Grok CLI access token is missing." };
|
||||
|
||||
const request = (token) => fetchFn(
|
||||
MODELS_URL,
|
||||
{
|
||||
method: "GET",
|
||||
headers: buildHeaders(token, credentials?.providerSpecificData),
|
||||
},
|
||||
proxyOptions,
|
||||
);
|
||||
|
||||
try {
|
||||
let response = await request(accessToken);
|
||||
if ((response.status === 401 || response.status === 403) && credentials?.refreshToken) {
|
||||
const refreshed = await refreshProviderCredentials(
|
||||
"grok-cli",
|
||||
credentials,
|
||||
log,
|
||||
proxyOptions,
|
||||
);
|
||||
if (refreshed?.accessToken) {
|
||||
accessToken = refreshed.accessToken;
|
||||
try {
|
||||
await onCredentialsRefreshed?.(refreshed);
|
||||
} catch (error) {
|
||||
log?.warn?.("Grok CLI credential persistence failed", error);
|
||||
}
|
||||
response = await request(accessToken);
|
||||
}
|
||||
}
|
||||
|
||||
if (!response.ok) {
|
||||
const detail = await response.text().catch(() => "");
|
||||
return {
|
||||
models: [],
|
||||
warning: `Grok CLI model discovery failed (${response.status})${detail ? `: ${detail.slice(0, 160)}` : ""}`,
|
||||
};
|
||||
}
|
||||
|
||||
const models = parseGrokCliModels(await response.json());
|
||||
return models.length
|
||||
? { models }
|
||||
: { models: [], warning: "Grok CLI returned no selectable models." };
|
||||
} catch (error) {
|
||||
return { models: [], warning: `Grok CLI model discovery failed: ${error.message}` };
|
||||
}
|
||||
}
|
||||
@@ -17,6 +17,10 @@ for (const entry of REGISTRY) {
|
||||
for (const a of entry.aliases || []) ALIAS_TO_PROVIDER_ID[a] = entry.id;
|
||||
}
|
||||
|
||||
const BUILTIN_MODEL_ALIASES = {
|
||||
"grok-build": "gcli/grok-build",
|
||||
};
|
||||
|
||||
/**
|
||||
* Resolve provider alias to provider ID
|
||||
*/
|
||||
@@ -104,7 +108,9 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) {
|
||||
: aliasesOrGetter;
|
||||
|
||||
// Resolve alias
|
||||
const resolved = resolveModelAliasFromMap(parsed.model, aliases);
|
||||
const resolved =
|
||||
resolveModelAliasFromMap(parsed.model, aliases) ||
|
||||
resolveModelAliasFromMap(parsed.model, BUILTIN_MODEL_ALIASES);
|
||||
if (resolved) {
|
||||
return resolved;
|
||||
}
|
||||
|
||||
@@ -129,6 +129,9 @@ const REFRESH_HANDLERS = {
|
||||
github: (c, log) => refreshGitHubToken(c.refreshToken, log),
|
||||
kiro: (c, log) => refreshKiroToken(c.refreshToken, c.providerSpecificData, log),
|
||||
xai: (c, log) => refreshXaiToken(c.refreshToken, log),
|
||||
// Grok CLI shares xAI OAuth client + token endpoint (device-code tokens refresh the same way)
|
||||
"grok-cli": (c, log) => refreshXaiToken(c.refreshToken, log),
|
||||
gcli: (c, log) => refreshXaiToken(c.refreshToken, log),
|
||||
"codebuddy-cn": (c, log) => refreshCodebuddyToken(c.refreshToken, log),
|
||||
vertex: vertexRefreshHandler,
|
||||
"vertex-partner": vertexRefreshHandler
|
||||
@@ -187,6 +190,7 @@ export function formatProviderCredentials(provider, credentials, log) {
|
||||
case "openai":
|
||||
case "openrouter":
|
||||
case "xai":
|
||||
case "grok-cli":
|
||||
return {
|
||||
apiKey: credentials.apiKey,
|
||||
accessToken: credentials.accessToken
|
||||
|
||||
@@ -12,6 +12,7 @@ import { getKiroUsage } from "./usage/kiro.js";
|
||||
import { getMiniMaxUsage } from "./usage/minimax.js";
|
||||
import { getCodeBuddyCnUsage } from "./usage/codebuddy-cn.js";
|
||||
import { getXaiUsage } from "./usage/xai.js";
|
||||
import { getGrokCliUsage } from "./usage/grok-cli.js";
|
||||
import {
|
||||
getQwenUsage,
|
||||
getIflowUsage,
|
||||
@@ -45,6 +46,7 @@ const USAGE_HANDLERS = {
|
||||
"vercel-ai-gateway": (c) => getVercelAiGatewayUsage(c.apiKey, c.proxyOptions),
|
||||
"codebuddy-cn": (c) => getCodeBuddyCnUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions),
|
||||
xai: (c) => getXaiUsage(c.accessToken, c.proxyOptions),
|
||||
"grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions),
|
||||
};
|
||||
|
||||
export async function getUsageForProvider(connection, proxyOptions = null) {
|
||||
|
||||
@@ -2,15 +2,15 @@
|
||||
* Google usage handlers (Gemini CLI + Antigravity)
|
||||
*/
|
||||
|
||||
import { CLIENT_METADATA, getPlatformUserAgent } from "../../config/appConstants.js";
|
||||
import { ANTIGRAVITY_OAUTH_CLIENT } from "../../providers/shared.js";
|
||||
import { CLIENT_METADATA } from "../../config/appConstants.js";
|
||||
import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION, ANTIGRAVITY_OAUTH_CLIENT } from "../../providers/shared.js";
|
||||
import { U, parseResetTime, normalizeCloudCodeProjectId, fetchWithTimeout } from "./shared.js";
|
||||
|
||||
// Antigravity API config (from Quotio) — urls from registry, oauth client + dynamic UA kept here
|
||||
const ANTIGRAVITY_CONFIG = {
|
||||
...U("antigravity"),
|
||||
...ANTIGRAVITY_OAUTH_CLIENT,
|
||||
userAgent: getPlatformUserAgent(),
|
||||
userAgent: ANTIGRAVITY_IDE_USER_AGENT,
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -129,8 +129,7 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro
|
||||
"User-Agent": ANTIGRAVITY_CONFIG.userAgent,
|
||||
"Content-Type": "application/json",
|
||||
"X-Client-Name": "antigravity",
|
||||
"X-Client-Version": "1.107.0",
|
||||
"x-request-source": "local", // MITM bypass
|
||||
"X-Client-Version": ANTIGRAVITY_IDE_VERSION,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
...(projectId ? { project: projectId } : {})
|
||||
@@ -229,7 +228,6 @@ async function getAntigravitySubscriptionInfo(accessToken, proxyOptions = null)
|
||||
"Authorization": `Bearer ${accessToken}`,
|
||||
"User-Agent": ANTIGRAVITY_CONFIG.userAgent,
|
||||
"Content-Type": "application/json",
|
||||
"x-request-source": "local", // MITM bypass
|
||||
},
|
||||
body: JSON.stringify({ metadata: CLIENT_METADATA, mode: 1 }),
|
||||
}, 10000, proxyOptions);
|
||||
|
||||
328
open-sse/services/usage/grok-cli.js
Normal file
328
open-sse/services/usage/grok-cli.js
Normal file
@@ -0,0 +1,328 @@
|
||||
/**
|
||||
* Grok CLI / Grok Build usage handler
|
||||
*
|
||||
* Source of truth: official grok-shell/grok-pager traffic to cli-chat-proxy.grok.com
|
||||
* GET /v1/billing?format=credits
|
||||
* GET /v1/user?include=subscription
|
||||
*
|
||||
* Observed billing shape (protobuf-json style `{ val: number }`):
|
||||
* {
|
||||
* config: {
|
||||
* currentPeriod: { type: "USAGE_PERIOD_TYPE_WEEKLY", start, end },
|
||||
* onDemandCap: { val },
|
||||
* onDemandUsed: { val },
|
||||
* prepaidBalance: { val },
|
||||
* isUnifiedBillingUser: true,
|
||||
* billingPeriodStart, billingPeriodEnd
|
||||
* }
|
||||
* }
|
||||
*
|
||||
* Exhausted free/promo accounts return cap=0/used=0/prepaid=0 and chat 402s with
|
||||
* personal-team-blocked:spending-limit. Paid/sub accounts surface non-zero cap
|
||||
* or prepaidBalance; richer credit fields are parsed opportunistically if present.
|
||||
*/
|
||||
|
||||
import { proxyAwareFetch } from "../../utils/proxyFetch.js";
|
||||
import { U, parseResetTime, toFiniteNumber } from "./shared.js";
|
||||
import {
|
||||
GROK_CLI_CLIENT_IDENTIFIER,
|
||||
GROK_CLI_USER_AGENT,
|
||||
GROK_CLI_VERSION,
|
||||
} from "../../config/grokCli.js";
|
||||
|
||||
const USAGE = U("grok-cli");
|
||||
const BILLING_URL = USAGE.url || "https://cli-chat-proxy.grok.com/v1/billing?format=credits";
|
||||
const USER_URL = USAGE.userUrl || "https://cli-chat-proxy.grok.com/v1/user?include=subscription";
|
||||
|
||||
/** Unwrap protobuf-json `{ val: n }` or plain numbers/strings. */
|
||||
function unwrapVal(value, fallback = 0) {
|
||||
if (value == null) return fallback;
|
||||
if (typeof value === "object" && !Array.isArray(value) && "val" in value) {
|
||||
return toFiniteNumber(value.val, fallback);
|
||||
}
|
||||
return toFiniteNumber(value, fallback);
|
||||
}
|
||||
|
||||
function buildGrokCliHeaders(accessToken, providerSpecificData = {}) {
|
||||
const psd = providerSpecificData || {};
|
||||
const headers = {
|
||||
Authorization: `Bearer ${accessToken}`,
|
||||
Accept: "application/json",
|
||||
"User-Agent": GROK_CLI_USER_AGENT,
|
||||
"x-xai-token-auth": "xai-grok-cli",
|
||||
"x-grok-client-identifier": GROK_CLI_CLIENT_IDENTIFIER,
|
||||
"x-grok-client-version": GROK_CLI_VERSION,
|
||||
"x-grok-client-mode": "headless",
|
||||
};
|
||||
const email = psd.email;
|
||||
const userId = psd.userId || psd.principalId;
|
||||
if (email) headers["x-email"] = email;
|
||||
if (userId) headers["x-userid"] = userId;
|
||||
return headers;
|
||||
}
|
||||
|
||||
function subscriptionTier(user, config) {
|
||||
const rawTier =
|
||||
user?.subscriptionTier ??
|
||||
user?.subscription_tier ??
|
||||
user?.subscription?.tier ??
|
||||
config?.subscriptionTier ??
|
||||
config?.subscription_tier;
|
||||
return typeof rawTier === "string" ? rawTier.trim() : "";
|
||||
}
|
||||
|
||||
function resolvePlan(user, config) {
|
||||
const tier = subscriptionTier(user, config);
|
||||
if (tier) {
|
||||
return tier
|
||||
.replace(/[_-]+/g, " ")
|
||||
.replace(/\b\w/g, (c) => c.toUpperCase());
|
||||
}
|
||||
if (user?.hasGrokCodeAccess === true) return "Grok Code";
|
||||
if (config?.isUnifiedBillingUser === true) return "Grok Build";
|
||||
return "Grok Build";
|
||||
}
|
||||
|
||||
function makeQuota({ used, total, resetAt, unlimited = false }) {
|
||||
const safeTotal = Math.max(0, toFiniteNumber(total, 0));
|
||||
const safeUsed = Math.max(0, toFiniteNumber(used, 0));
|
||||
// Do NOT set absolute `remaining` — QuotaTable's getRemainingPercentage treats
|
||||
// `remaining` as a 0–100 percentage (same trap as Qoder credits).
|
||||
if (unlimited || safeTotal === 0) {
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: 0,
|
||||
remainingPercentage: unlimited ? 100 : 0,
|
||||
resetAt: resetAt || null,
|
||||
unlimited: true,
|
||||
};
|
||||
}
|
||||
const remaining = Math.max(0, safeTotal - safeUsed);
|
||||
const remainingPercentage = (remaining / safeTotal) * 100;
|
||||
return {
|
||||
used: safeUsed,
|
||||
total: safeTotal,
|
||||
remainingPercentage,
|
||||
resetAt: resetAt || null,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Map billing JSON → normalized quotas object for the dashboard.
|
||||
* Returns { quotas, periodEnd, exhaustedHint } or empty quotas when nothing usable.
|
||||
*/
|
||||
export function parseGrokCliBilling(billing, user = null) {
|
||||
const root = billing && typeof billing === "object" ? billing : {};
|
||||
const config =
|
||||
root.config && typeof root.config === "object" && !Array.isArray(root.config)
|
||||
? root.config
|
||||
: root;
|
||||
|
||||
const periodEnd =
|
||||
parseResetTime(config.billingPeriodEnd) ||
|
||||
parseResetTime(config.billing_period_end) ||
|
||||
parseResetTime(config.currentPeriod?.end) ||
|
||||
parseResetTime(config.resetAt || config.resetsAt || config.periodEnd) ||
|
||||
parseResetTime(root.billingPeriodEnd) ||
|
||||
parseResetTime(root.billing_period_end) ||
|
||||
parseResetTime(root.resetAt || root.resetsAt || root.periodEnd) ||
|
||||
null;
|
||||
|
||||
const quotas = {};
|
||||
const tier = subscriptionTier(user, config);
|
||||
const subscriptionAccess = Boolean(tier) && !/^(free|none|null)$/i.test(tier);
|
||||
|
||||
// Current Grok Build responses expose included monthly usage at top level.
|
||||
const monthlyLimit = unwrapVal(
|
||||
config.monthlyLimit ?? config.monthly_limit ?? root.monthlyLimit ?? root.monthly_limit,
|
||||
NaN,
|
||||
);
|
||||
const includedUsed = unwrapVal(
|
||||
config.includedUsed ?? config.included_used ?? root.includedUsed ?? root.included_used,
|
||||
NaN,
|
||||
);
|
||||
const totalUsed = unwrapVal(
|
||||
config.totalUsed ?? config.total_used ?? root.totalUsed ?? root.total_used,
|
||||
NaN,
|
||||
);
|
||||
if (Number.isFinite(monthlyLimit) && monthlyLimit > 0) {
|
||||
quotas["Monthly included"] = makeQuota({
|
||||
used: Number.isFinite(includedUsed)
|
||||
? includedUsed
|
||||
: Number.isFinite(totalUsed)
|
||||
? totalUsed
|
||||
: 0,
|
||||
total: monthlyLimit,
|
||||
resetAt: periodEnd,
|
||||
});
|
||||
}
|
||||
|
||||
// Primary: on-demand spending window (subscription / promo credits)
|
||||
const onDemandCap = unwrapVal(config.onDemandCap ?? root.onDemandCap, NaN);
|
||||
const onDemandUsed = unwrapVal(config.onDemandUsed ?? root.onDemandUsed, NaN);
|
||||
if (Number.isFinite(onDemandCap) && onDemandCap > 0) {
|
||||
const used = Number.isFinite(onDemandUsed) ? Math.max(0, onDemandUsed) : 0;
|
||||
quotas["On-demand"] = makeQuota({
|
||||
used,
|
||||
total: onDemandCap,
|
||||
resetAt: periodEnd,
|
||||
});
|
||||
} else if (
|
||||
!subscriptionAccess &&
|
||||
Number.isFinite(onDemandCap) &&
|
||||
onDemandCap === 0 &&
|
||||
Number.isFinite(onDemandUsed)
|
||||
) {
|
||||
// Cap 0 is the exhausted free/promo state (chat returns 402 spending-limit).
|
||||
// UI treats total===0 as unlimited, so use a synthetic 1/1 depleted row.
|
||||
quotas["On-demand"] = {
|
||||
used: 1,
|
||||
total: 1,
|
||||
remainingPercentage: 0,
|
||||
resetAt: periodEnd,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
// Prepaid top-up balance (remaining credits; no fixed allotment known)
|
||||
const prepaid = unwrapVal(config.prepaidBalance ?? root.prepaidBalance, NaN);
|
||||
if (Number.isFinite(prepaid) && prepaid > 0) {
|
||||
// Show full bar against the current balance (0 spent of this remaining pot).
|
||||
quotas["Prepaid"] = {
|
||||
used: 0,
|
||||
total: prepaid,
|
||||
remainingPercentage: 100,
|
||||
resetAt: null,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
|
||||
// Opportunistic richer credit envelopes (future / other account types)
|
||||
const creditBags = [
|
||||
root.credits,
|
||||
root.creditBalance,
|
||||
root.usage,
|
||||
config.credits,
|
||||
config.includedCredits,
|
||||
config.subscriptionCredits,
|
||||
].filter((bag) => bag && typeof bag === "object" && !Array.isArray(bag));
|
||||
|
||||
for (const bag of creditBags) {
|
||||
const total = unwrapVal(
|
||||
bag.total ?? bag.limit ?? bag.cap ?? bag.allocation ?? bag.amount,
|
||||
NaN,
|
||||
);
|
||||
const used = unwrapVal(bag.used ?? bag.spent ?? bag.consumed, NaN);
|
||||
const remaining = unwrapVal(bag.remaining ?? bag.balance ?? bag.left, NaN);
|
||||
if (Number.isFinite(total) && total > 0) {
|
||||
const resolvedUsed = Number.isFinite(used)
|
||||
? used
|
||||
: Number.isFinite(remaining)
|
||||
? Math.max(0, total - remaining)
|
||||
: 0;
|
||||
if (!quotas.Credits) {
|
||||
quotas.Credits = makeQuota({
|
||||
used: resolvedUsed,
|
||||
total,
|
||||
resetAt: parseResetTime(bag.resetAt || bag.resetsAt || bag.end) || periodEnd,
|
||||
});
|
||||
}
|
||||
} else if (Number.isFinite(remaining) && remaining >= 0 && !quotas.Credits) {
|
||||
quotas.Credits = {
|
||||
used: 0,
|
||||
total: remaining > 0 ? remaining : 1,
|
||||
remainingPercentage: remaining > 0 ? 100 : 0,
|
||||
resetAt: periodEnd,
|
||||
unlimited: false,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Exhausted when every finite quota bar is at 0% remaining
|
||||
const exhausted =
|
||||
Object.keys(quotas).length > 0 &&
|
||||
Object.values(quotas).every(
|
||||
(q) => q.unlimited !== true && (q.remainingPercentage ?? 100) <= 0,
|
||||
);
|
||||
|
||||
return {
|
||||
plan: resolvePlan(user, config),
|
||||
quotas,
|
||||
periodEnd,
|
||||
exhausted,
|
||||
subscriptionAccess,
|
||||
rawConfig: config,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} accessToken
|
||||
* @param {object|null} providerSpecificData
|
||||
* @param {object|null} proxyOptions
|
||||
*/
|
||||
export async function getGrokCliUsage(accessToken, providerSpecificData = null, proxyOptions = null) {
|
||||
if (!accessToken) {
|
||||
return { message: "Grok CLI access token not available." };
|
||||
}
|
||||
|
||||
const headers = buildGrokCliHeaders(accessToken, providerSpecificData);
|
||||
|
||||
try {
|
||||
// Fetch billing + user profile in parallel (same pattern as official CLI startup)
|
||||
const [billingRes, userRes] = await Promise.all([
|
||||
proxyAwareFetch(
|
||||
BILLING_URL,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
),
|
||||
proxyAwareFetch(
|
||||
USER_URL,
|
||||
{ method: "GET", headers },
|
||||
proxyOptions,
|
||||
).catch(() => null),
|
||||
]);
|
||||
|
||||
if (billingRes.status === 401 || billingRes.status === 403) {
|
||||
return { message: "Grok CLI authentication expired. Please re-authorize." };
|
||||
}
|
||||
|
||||
if (!billingRes.ok) {
|
||||
const errText = await billingRes.text().catch(() => "");
|
||||
const trimmed = errText ? `: ${errText.slice(0, 200)}` : "";
|
||||
return { message: `Grok CLI billing API error (${billingRes.status})${trimmed}` };
|
||||
}
|
||||
|
||||
const billing = await billingRes.json().catch(() => null);
|
||||
if (!billing || typeof billing !== "object") {
|
||||
return { message: "Grok CLI billing response was not JSON." };
|
||||
}
|
||||
|
||||
let user = null;
|
||||
if (userRes?.ok) {
|
||||
user = await userRes.json().catch(() => null);
|
||||
}
|
||||
|
||||
const parsed = parseGrokCliBilling(billing, user);
|
||||
|
||||
if (!parsed.quotas || Object.keys(parsed.quotas).length === 0) {
|
||||
return {
|
||||
plan: parsed.plan,
|
||||
message: parsed.subscriptionAccess
|
||||
? "Subscription access is active; Grok does not expose a numeric included quota."
|
||||
: "Grok Build connected, but no credit allotment was returned. Free promo may be exhausted.",
|
||||
quotas: {},
|
||||
};
|
||||
}
|
||||
|
||||
// Dashboard hides QuotaTable whenever `message` is set, so only attach a
|
||||
// message when there are no quota rows to render. Depleted accounts keep
|
||||
// the 0% On-demand bar without a blocking message.
|
||||
return {
|
||||
plan: parsed.plan,
|
||||
quotas: parsed.quotas,
|
||||
};
|
||||
} catch (error) {
|
||||
return { message: `Grok CLI usage error: ${error.message}` };
|
||||
}
|
||||
}
|
||||
@@ -1,17 +1,26 @@
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
|
||||
// Strip request params a given provider/model rejects upstream (e.g. HTTP 400).
|
||||
// Config-driven: add a rule instead of scattering `delete body.x` across executors.
|
||||
|
||||
// Each rule: optional provider, regex match on model, list of params to drop.
|
||||
// A param is removed only when it is present (!== undefined).
|
||||
const STRIP_RULES = [
|
||||
// claude-opus-4 series: temperature is deprecated (Anthropic 400). #1748
|
||||
{ match: /claude-opus-4/i, drop: ["temperature"] },
|
||||
// All Claude models: temperature deprecated/rejected upstream (Anthropic 400). #1748
|
||||
{ match: /claude/i, drop: ["temperature"] },
|
||||
// GitHub Copilot gpt-5.4: temperature unsupported.
|
||||
{ provider: "github", match: /gpt-5\.4/i, drop: ["temperature"] },
|
||||
// GitHub Copilot Claude (except opus/sonnet 4.6): thinking + reasoning_effort rejected. #713
|
||||
{ provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] },
|
||||
// Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926)
|
||||
{ provider: "cloudflare-ai", flattenContent: true },
|
||||
{ provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true },
|
||||
// VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's
|
||||
// advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144),
|
||||
// so clampToModelMaxOutput alone leaves it uncapped and the request 400s with
|
||||
// "integer above maximum value, expected <= 32768". Pin an explicit endpoint cap;
|
||||
// min() with the model ceiling still applies if a variant's own limit is lower.
|
||||
{ provider: "volcengine-ark", match: /kimi/i, maxOutputCap: 32768, clampToModelMaxOutput: true },
|
||||
];
|
||||
|
||||
// Test a rule's match (regex or predicate) against the model id.
|
||||
@@ -20,6 +29,12 @@ function matches(rule, model) {
|
||||
return typeof rule.match === "function" ? rule.match(model) : rule.match.test(model);
|
||||
}
|
||||
|
||||
function clampNumber(body, key, ceiling) {
|
||||
if (typeof body[key] === "number" && Number.isFinite(body[key]) && body[key] > ceiling) {
|
||||
body[key] = ceiling;
|
||||
}
|
||||
}
|
||||
|
||||
// Remove unsupported params from body in place; returns body.
|
||||
export function stripUnsupportedParams(provider, model, body) {
|
||||
if (!model || !body || typeof body !== "object") return body;
|
||||
@@ -39,6 +54,22 @@ export function stripUnsupportedParams(provider, model, body) {
|
||||
}
|
||||
}
|
||||
}
|
||||
if (rule.clampToModelMaxOutput || Number.isFinite(rule.maxOutputCap)) {
|
||||
const modelCeiling = getCapabilitiesForModel(provider, model).maxOutput;
|
||||
const candidates = [];
|
||||
if (rule.clampToModelMaxOutput && Number.isFinite(modelCeiling) && modelCeiling > 0) {
|
||||
candidates.push(modelCeiling);
|
||||
}
|
||||
if (Number.isFinite(rule.maxOutputCap) && rule.maxOutputCap > 0) {
|
||||
candidates.push(rule.maxOutputCap);
|
||||
}
|
||||
if (candidates.length > 0) {
|
||||
const ceiling = Math.min(...candidates);
|
||||
clampNumber(body, "max_tokens", ceiling);
|
||||
clampNumber(body, "max_completion_tokens", ceiling);
|
||||
clampNumber(body, "max_output_tokens", ceiling);
|
||||
}
|
||||
}
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
@@ -20,6 +20,13 @@ const FORMAT_TO_NATIVE = {
|
||||
kiro: "kiro",
|
||||
};
|
||||
|
||||
// Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent).
|
||||
export function stripThinkingSuffix(model) {
|
||||
if (typeof model !== "string") return model;
|
||||
const m = model.match(/^(.*)\([^()]+\)\s*$/);
|
||||
return m ? m[1].trim() : model;
|
||||
}
|
||||
|
||||
// Parse model-name suffix "model(value)" → { cleanModel, override }.
|
||||
// value: level name (high) | number (8192) | auto | none. null override when absent.
|
||||
export function parseSuffix(model) {
|
||||
@@ -132,18 +139,66 @@ function toGeminiThinkingLevel(cfg) {
|
||||
return effortToThinkingLevel(raw);
|
||||
}
|
||||
|
||||
function toKimiReasoningEffort(cfg) {
|
||||
const level = toLevel(cfg);
|
||||
if (level === "auto") return "high";
|
||||
if (level === "minimal") return "low";
|
||||
if (level === "xhigh") return "max";
|
||||
if (["low", "medium", "high", "max"].includes(level)) return level;
|
||||
return null;
|
||||
}
|
||||
|
||||
const GEMINI_LEVEL_OUTPUT_FLOOR = {
|
||||
minimal: 4096,
|
||||
low: 8192,
|
||||
medium: 16384,
|
||||
high: 65535,
|
||||
};
|
||||
|
||||
function geminiBudgetOutputFloor(budget) {
|
||||
if (budget === -1) return 32768;
|
||||
if (!Number.isFinite(budget)) return 32768;
|
||||
if (budget <= 1024) return 8192;
|
||||
if (budget <= 8192) return 16384;
|
||||
if (budget <= 24576) return 32768;
|
||||
return 65535;
|
||||
}
|
||||
|
||||
function geminiLevelOutputFloor(level) {
|
||||
return GEMINI_LEVEL_OUTPUT_FLOOR[level] || GEMINI_LEVEL_OUTPUT_FLOOR.high;
|
||||
}
|
||||
|
||||
// Gemini nests thinkingConfig under generationConfig. gemini-cli / antigravity wrap
|
||||
// the whole request in a { request: { generationConfig } } envelope — target the
|
||||
// envelope's generationConfig when present, else the top-level one.
|
||||
function getGeminiGenerationConfig(body) {
|
||||
if (body.request && typeof body.request === "object") {
|
||||
if (!body.request.generationConfig || typeof body.request.generationConfig !== "object") {
|
||||
body.request.generationConfig = {};
|
||||
}
|
||||
return body.request.generationConfig;
|
||||
}
|
||||
if (!body.generationConfig || typeof body.generationConfig !== "object") {
|
||||
body.generationConfig = {};
|
||||
}
|
||||
return body.generationConfig;
|
||||
}
|
||||
|
||||
function setGeminiThinking(body, tc) {
|
||||
const gc = body.request?.generationConfig
|
||||
? body.request.generationConfig
|
||||
: (body.generationConfig && typeof body.generationConfig === "object"
|
||||
? body.generationConfig
|
||||
: (body.generationConfig = {}));
|
||||
const gc = getGeminiGenerationConfig(body);
|
||||
gc.thinkingConfig = tc;
|
||||
}
|
||||
|
||||
function ensureGeminiOutputFloor(body, floor, caps) {
|
||||
const cap = Number.isFinite(caps?.maxOutput) ? caps.maxOutput : floor;
|
||||
const target = Math.min(floor, cap);
|
||||
const gc = getGeminiGenerationConfig(body);
|
||||
const current = Number(gc.maxOutputTokens);
|
||||
if (!Number.isFinite(current) || current < target) {
|
||||
gc.maxOutputTokens = target;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip every known thinking field from a body (used before re-applying / when unsupported).
|
||||
function stripAll(body) {
|
||||
delete body.thinking;
|
||||
@@ -168,11 +223,18 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "openai": {
|
||||
if (none && canDisable) { body.reasoning_effort = "none"; break; }
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level;
|
||||
// OpenAI reasoning_effort enum caps at "xhigh" (no "max"); clamp Claude Code's "max".
|
||||
if (level) body.reasoning_effort = level === "max" ? "xhigh" : level;
|
||||
break;
|
||||
}
|
||||
case "claude-adaptive": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
// output_config.effort alone does NOT turn thinking on: Anthropic requires
|
||||
// an explicit thinking:{type:"adaptive"} on Opus 4.6/4.7/4.8 and Sonnet 4.6
|
||||
// ("thinking is off unless you explicitly set it"), and Anthropic-compatible
|
||||
// shims (e.g. GitHub Copilot /v1/messages) default thinking off even for
|
||||
// Sonnet 5. Send both fields — the documented adaptive-thinking shape.
|
||||
body.thinking = { type: "adaptive" };
|
||||
const level = toLevel(eff);
|
||||
body.output_config = { effort: level === "xhigh" ? "high" : level };
|
||||
break;
|
||||
@@ -186,12 +248,14 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "gemini-level": {
|
||||
const level = none ? "minimal" : toGeminiThinkingLevel(eff);
|
||||
setGeminiThinking(body, { thinkingLevel: level, includeThoughts: level !== "minimal" });
|
||||
ensureGeminiOutputFloor(body, geminiLevelOutputFloor(level), caps);
|
||||
break;
|
||||
}
|
||||
case "gemini-budget": {
|
||||
if (none && canDisable) { setGeminiThinking(body, { thinkingBudget: 0, includeThoughts: false }); break; }
|
||||
const budget = toBudget(eff, caps.thinkingRange);
|
||||
setGeminiThinking(body, { thinkingBudget: budget ?? -1, includeThoughts: true });
|
||||
ensureGeminiOutputFloor(body, geminiBudgetOutputFloor(budget ?? -1), caps);
|
||||
break;
|
||||
}
|
||||
case "zai": {
|
||||
@@ -217,8 +281,8 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
}
|
||||
case "kimi": {
|
||||
if (none && canDisable) { body.thinking = { type: "disabled" }; break; }
|
||||
const level = toLevel(eff);
|
||||
if (level) body.reasoning_effort = level === "max" ? "high" : level;
|
||||
const effort = toKimiReasoningEffort(eff);
|
||||
if (effort) body.reasoning_effort = effort;
|
||||
break;
|
||||
}
|
||||
case "minimax": {
|
||||
|
||||
@@ -192,10 +192,27 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne
|
||||
delete body.output_config;
|
||||
}
|
||||
|
||||
// Clamp max_tokens to the model output ceiling (never above DEFAULT_MAX_TOKENS)
|
||||
// Clamp max_tokens to the model's real output ceiling. Models whose caps
|
||||
// declare a higher maxOutput (e.g. Opus 4.8 / Sonnet 4.6 = 128000) are allowed
|
||||
// up to it, so max-effort thinking gets full budget; others fall back to the
|
||||
// conservative 64000 default.
|
||||
if (body.max_tokens) {
|
||||
const ceiling = Math.min(getCapabilitiesForModel(provider, body.model).maxOutput, DEFAULT_MAX_TOKENS);
|
||||
const ceiling = getCapabilitiesForModel(provider, body.model).maxOutput || DEFAULT_MAX_TOKENS;
|
||||
if (body.max_tokens > ceiling) body.max_tokens = ceiling;
|
||||
|
||||
// Reconcile against thinking budget. applyThinking (thinkingUnified.js) runs
|
||||
// AFTER adjustMaxTokens capped max_tokens, and the claude-budget format maps
|
||||
// max effort → budget_tokens 128000 — larger than the clamped max_tokens.
|
||||
// Anthropic requires max_tokens strictly greater than budget_tokens (else 400).
|
||||
// Prefer raising max_tokens to preserve the requested thinking depth; if the
|
||||
// budget alone meets/exceeds the ceiling, cap output and shrink the budget so
|
||||
// some tokens remain for the answer.
|
||||
if (body.thinking?.type === "enabled" && body.thinking.budget_tokens && body.thinking.budget_tokens >= body.max_tokens) {
|
||||
body.max_tokens = Math.min(body.thinking.budget_tokens + 1024, ceiling);
|
||||
if (body.thinking.budget_tokens >= body.max_tokens) {
|
||||
body.thinking.budget_tokens = Math.max(1024, body.max_tokens - 1024);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 1. System: remove all cache_control, add only to last block with ttl 1h
|
||||
|
||||
@@ -3,9 +3,13 @@ import { DEFAULT_MAX_TOKENS, DEFAULT_MIN_TOKENS } from "../../config/runtimeConf
|
||||
/**
|
||||
* Adjust max_tokens based on request context
|
||||
* @param {object} body - Request body
|
||||
* @param {number} [ceiling=DEFAULT_MAX_TOKENS] - Upper bound for max_tokens.
|
||||
* Callers with model context (e.g. openai-to-claude) pass the model's real
|
||||
* maxOutput so high-output models (Opus 4.8 = 128000) aren't pre-clamped to
|
||||
* the conservative 64000 default before the model-aware step sees them.
|
||||
* @returns {number} Adjusted max_tokens
|
||||
*/
|
||||
export function adjustMaxTokens(body) {
|
||||
export function adjustMaxTokens(body, ceiling = DEFAULT_MAX_TOKENS) {
|
||||
let maxTokens = body.max_tokens || DEFAULT_MAX_TOKENS;
|
||||
|
||||
// Auto-increase for tool calling to prevent truncated arguments (min never above max)
|
||||
@@ -16,14 +20,14 @@ export function adjustMaxTokens(body) {
|
||||
}
|
||||
|
||||
// Ensure max_tokens > thinking.budget_tokens (Claude API requirement)
|
||||
// Claude API requires strictly greater, so add buffer instead of using DEFAULT_MAX_TOKENS
|
||||
// which could equal budget_tokens when budget_tokens >= 64000
|
||||
// Claude API requires strictly greater, so add buffer instead of using the
|
||||
// ceiling which could equal budget_tokens when budget_tokens >= ceiling
|
||||
if (body.thinking?.budget_tokens && maxTokens <= body.thinking.budget_tokens) {
|
||||
maxTokens = body.thinking.budget_tokens + 1024;
|
||||
}
|
||||
|
||||
// Never exceed the global ceiling
|
||||
if (maxTokens > DEFAULT_MAX_TOKENS) maxTokens = DEFAULT_MAX_TOKENS;
|
||||
// Never exceed the ceiling
|
||||
if (maxTokens > ceiling) maxTokens = ceiling;
|
||||
|
||||
return maxTokens;
|
||||
}
|
||||
|
||||
@@ -103,8 +103,16 @@ export function translateRequest(sourceFormat, targetFormat, model, body, stream
|
||||
}
|
||||
}
|
||||
|
||||
// Normalize thinking to the target provider-native format (config-driven, capability-aware)
|
||||
applyThinking(targetFormat, model, result, provider, thinkingIntent);
|
||||
// Normalize thinking to the target provider-native format (config-driven, capability-aware).
|
||||
// Kiro's GenerateAssistantResponse request does not accept the generic top-level
|
||||
// `thinking` field; its translators map thinking intent to KAS-compatible
|
||||
// systemPrompt/additionalModelRequestFields instead.
|
||||
const kiroThinkingMappedByTranslator =
|
||||
targetFormat === FORMATS.KIRO &&
|
||||
(sourceFormat === FORMATS.OPENAI || sourceFormat === FORMATS.CLAUDE);
|
||||
if (!kiroThinkingMappedByTranslator) {
|
||||
applyThinking(targetFormat, model, result, provider, thinkingIntent);
|
||||
}
|
||||
|
||||
// Always normalize to clean OpenAI format when target is OpenAI
|
||||
// This handles hybrid requests (e.g., OpenAI messages + Claude tools)
|
||||
|
||||
@@ -24,13 +24,15 @@
|
||||
*/
|
||||
import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { v4 as uuidv4 } from "uuid";
|
||||
import { applyKiroSessionReplay } from "../../utils/kiroSessionReplay.js";
|
||||
import { resolveContinuationId, resolveSessionIdentity } from "../../utils/sessionManager.js";
|
||||
import {
|
||||
resolveKiroModel,
|
||||
resolveKiroThinkingBudget,
|
||||
buildThinkingSystemPrefix,
|
||||
KIRO_AGENTIC_SYSTEM_PROMPT,
|
||||
resolveDefaultProfileArn,
|
||||
buildKiroAdditionalModelRequestFieldsForModel,
|
||||
} from "../../config/kiroConstants.js";
|
||||
import { DEFAULT_IMAGE_MIME } from "../schema/index.js";
|
||||
import { ROLE, CLAUDE_BLOCK } from "../schema/index.js";
|
||||
@@ -363,6 +365,18 @@ function reconcileOrphanedToolResults(history, currentMessage) {
|
||||
}
|
||||
}
|
||||
|
||||
function extractClaudeSystemText(system) {
|
||||
if (!system) return "";
|
||||
if (typeof system === "string") return system;
|
||||
if (Array.isArray(system)) {
|
||||
return system.map((s) => {
|
||||
if (typeof s === "string") return s;
|
||||
return s?.text || "";
|
||||
}).filter(Boolean).join("\n");
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a Kiro payload directly from a Claude Messages API request body.
|
||||
*/
|
||||
@@ -402,50 +416,75 @@ export function claudeToKiroRequest(model, body, stream, credentials) {
|
||||
? (credentials?.providerSpecificData?.profileArn || "")
|
||||
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));
|
||||
|
||||
let finalContent = currentMessage?.userInputMessage?.content || "";
|
||||
|
||||
// System prompt → prepend to the user content.
|
||||
if (body.system) {
|
||||
let systemText = "";
|
||||
if (typeof body.system === "string") {
|
||||
systemText = body.system;
|
||||
} else if (Array.isArray(body.system)) {
|
||||
systemText = body.system.map((s) => s.text || "").join("\n");
|
||||
}
|
||||
if (systemText) finalContent = `${systemText}\n\n${finalContent}`;
|
||||
}
|
||||
|
||||
// Prefix order: thinking_mode tag, timestamp marker, then agentic prompt.
|
||||
// Kiro CLI/KAS sends system prompt as top-level `systemPrompt`. Keep a
|
||||
// content fallback too because the CodeWhisperer surface does not always
|
||||
// enforce top-level systemPrompt for direct calls.
|
||||
const timestamp = new Date().toISOString();
|
||||
const prefixParts = [];
|
||||
if (thinkingBudget !== null) prefixParts.push(buildThinkingSystemPrefix(thinkingBudget));
|
||||
prefixParts.push(`[Context: Current time is ${timestamp}]`);
|
||||
if (agentic) prefixParts.push(KIRO_AGENTIC_SYSTEM_PROMPT);
|
||||
finalContent = `${prefixParts.join("\n\n")}\n\n${finalContent}`;
|
||||
const systemPromptParts = [];
|
||||
if (thinkingBudget !== null) systemPromptParts.push(buildThinkingSystemPrefix(thinkingBudget));
|
||||
if (agentic) systemPromptParts.push(KIRO_AGENTIC_SYSTEM_PROMPT);
|
||||
const systemInstruction = extractClaudeSystemText(body.system);
|
||||
if (systemInstruction) systemPromptParts.push(systemInstruction);
|
||||
const systemPrompt = systemPromptParts.filter(Boolean).join("\n\n");
|
||||
const currentTimeContext = `[Context: Current time is ${timestamp}]`;
|
||||
const contentPrefix = [systemPrompt, currentTimeContext].filter(Boolean).join("\n\n");
|
||||
|
||||
const sessionIdentity = resolveSessionIdentity({
|
||||
headers: credentials?.rawHeaders,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "kiro",
|
||||
});
|
||||
const conversationId = sessionIdentity.sessionId;
|
||||
const continuationId = resolveContinuationId({
|
||||
sessionId: conversationId,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "kiro",
|
||||
ephemeral: sessionIdentity.ephemeral,
|
||||
});
|
||||
const replay = applyKiroSessionReplay({
|
||||
conversationId,
|
||||
connectionId: credentials?.connectionId,
|
||||
modelId: upstreamModel,
|
||||
systemPrompt,
|
||||
contentPrefix,
|
||||
currentContentPrefix: currentTimeContext,
|
||||
history,
|
||||
currentMessage,
|
||||
});
|
||||
const replayCurrent = replay.currentMessage?.userInputMessage || {};
|
||||
const userInputMessage = {
|
||||
content: replayCurrent.content || "",
|
||||
modelId: upstreamModel,
|
||||
origin: "AI_EDITOR",
|
||||
...(replayCurrent.userInputMessageContext && {
|
||||
userInputMessageContext: replayCurrent.userInputMessageContext,
|
||||
}),
|
||||
...(replayCurrent.images && {
|
||||
images: replayCurrent.images,
|
||||
}),
|
||||
};
|
||||
|
||||
const payload = {
|
||||
conversationState: {
|
||||
chatTriggerType: "MANUAL",
|
||||
conversationId: uuidv4(),
|
||||
conversationId,
|
||||
agentContinuationId: continuationId,
|
||||
agentTaskType: "vibe",
|
||||
currentMessage: {
|
||||
userInputMessage: {
|
||||
content: finalContent,
|
||||
modelId: upstreamModel,
|
||||
origin: "AI_EDITOR",
|
||||
...(currentMessage?.userInputMessage?.userInputMessageContext && {
|
||||
userInputMessageContext:
|
||||
currentMessage.userInputMessage.userInputMessageContext,
|
||||
}),
|
||||
...(currentMessage?.userInputMessage?.images && {
|
||||
images: currentMessage.userInputMessage.images,
|
||||
}),
|
||||
},
|
||||
userInputMessage,
|
||||
},
|
||||
history,
|
||||
history: replay.history,
|
||||
},
|
||||
agentMode: "vibe",
|
||||
};
|
||||
|
||||
if (profileArn) payload.profileArn = profileArn;
|
||||
if (systemPrompt) payload.systemPrompt = systemPrompt;
|
||||
const additionalModelRequestFields = buildKiroAdditionalModelRequestFieldsForModel(body, upstreamModel);
|
||||
if (additionalModelRequestFields) {
|
||||
payload.additionalModelRequestFields = additionalModelRequestFields;
|
||||
}
|
||||
|
||||
if (maxTokens || temperature !== undefined || topP !== undefined) {
|
||||
payload.inferenceConfig = {};
|
||||
|
||||
@@ -129,14 +129,15 @@ function fixMissingToolResponsesOpenAI(messages) {
|
||||
}
|
||||
}
|
||||
|
||||
// Wrap mid-conversation system text so it ends as a user turn (avoids Anthropic prefill 400)
|
||||
// Wrap mid-conversation system text so it ends as a user turn (avoids Anthropic prefill 400).
|
||||
// Uses <instructions> tags that Claude models treat as authoritative directives.
|
||||
function systemReminderText(content) {
|
||||
const parts = Array.isArray(content)
|
||||
? content.filter(c => c?.type === CLAUDE_BLOCK.TEXT).map(c => c.text || "")
|
||||
: [typeof content === "string" ? content : ""];
|
||||
const text = parts.filter(Boolean).join("\n");
|
||||
if (!text.trim()) return "";
|
||||
return `<system-reminder>\n${text}\n</system-reminder>`;
|
||||
return `<instructions>\n${text}\n</instructions>`;
|
||||
}
|
||||
|
||||
// Convert single Claude message - returns single message or array of messages
|
||||
|
||||
@@ -31,11 +31,12 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
let currentAssistantMsg = null;
|
||||
let pendingToolResults = [];
|
||||
let pendingReasoning = "";
|
||||
let pendingReasoningEncrypted = "";
|
||||
|
||||
const inputItems = normalizeResponsesInput(body.input);
|
||||
if (!inputItems) return body;
|
||||
|
||||
// Extract reasoning text from summary[].text or encrypted_content fallback
|
||||
// Extract reasoning text from summary[].text (encrypted_content is continuity-only)
|
||||
const extractReasoningText = (item) => {
|
||||
if (Array.isArray(item.summary)) {
|
||||
const txt = item.summary.map(s => s?.text || "").filter(Boolean).join("\n");
|
||||
@@ -48,6 +49,13 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
return "";
|
||||
};
|
||||
|
||||
const attachPendingReasoning = (msg) => {
|
||||
if (pendingReasoning) msg.reasoning_content = pendingReasoning;
|
||||
if (pendingReasoningEncrypted) msg.encrypted_content = pendingReasoningEncrypted;
|
||||
pendingReasoning = "";
|
||||
pendingReasoningEncrypted = "";
|
||||
};
|
||||
|
||||
for (const item of inputItems) {
|
||||
// Determine item type - Droid CLI sends role-based items without 'type' field
|
||||
// Fallback: if no type but has role property, treat as message
|
||||
@@ -80,11 +88,12 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
})
|
||||
: item.content;
|
||||
const msg = { role: item.role, content };
|
||||
// Attach buffered reasoning to assistant turn (required by xiaomi-mimo thinking mode)
|
||||
if (item.role === ROLE.ASSISTANT && pendingReasoning) {
|
||||
msg.reasoning_content = pendingReasoning;
|
||||
// Attach buffered reasoning to assistant turn (required by xiaomi-mimo + store=false continuity)
|
||||
if (item.role === ROLE.ASSISTANT) attachPendingReasoning(msg);
|
||||
else {
|
||||
pendingReasoning = "";
|
||||
pendingReasoningEncrypted = "";
|
||||
}
|
||||
pendingReasoning = "";
|
||||
result.messages.push(msg);
|
||||
}
|
||||
else if (itemType === RESPONSES_ITEM.FUNCTION_CALL) {
|
||||
@@ -95,10 +104,7 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
content: null,
|
||||
tool_calls: []
|
||||
};
|
||||
if (pendingReasoning) {
|
||||
currentAssistantMsg.reasoning_content = pendingReasoning;
|
||||
pendingReasoning = "";
|
||||
}
|
||||
attachPendingReasoning(currentAssistantMsg);
|
||||
}
|
||||
// Skip items with empty/missing name — Codex/OpenAI reject nameless tool calls (#444)
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") continue;
|
||||
@@ -132,9 +138,15 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
});
|
||||
}
|
||||
else if (itemType === RESPONSES_ITEM.REASONING) {
|
||||
// Buffer reasoning text; attached to next assistant message/function_call
|
||||
// Buffer reasoning text; attached to next assistant message/function_call.
|
||||
// Also stash encrypted_content so a later openai→responses hop can restore
|
||||
// the store=false continuity blob (Grok CLI / Codex multi-turn).
|
||||
const txt = extractReasoningText(item);
|
||||
if (txt) pendingReasoning = pendingReasoning ? `${pendingReasoning}\n${txt}` : txt;
|
||||
if (typeof item.encrypted_content === "string" && item.encrypted_content) {
|
||||
// Prefer attaching to the next assistant message we create
|
||||
pendingReasoningEncrypted = item.encrypted_content;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -189,6 +201,7 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials)
|
||||
delete result.prompt_cache_key;
|
||||
delete result.store;
|
||||
delete result.reasoning;
|
||||
delete result.client_metadata;
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -202,6 +215,43 @@ function normalizeToolParameters(params) {
|
||||
return params;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a Responses `reasoning` input item from Chat Completions assistant fields.
|
||||
* Preserves encrypted blobs needed by store=false multi-turn (Grok CLI / Codex).
|
||||
* Returns null when the message has nothing useful to re-send.
|
||||
*/
|
||||
function buildReasoningInputItem(msg) {
|
||||
if (!msg || typeof msg !== "object") return null;
|
||||
|
||||
const encrypted =
|
||||
(typeof msg.encrypted_content === "string" && msg.encrypted_content) ||
|
||||
(typeof msg.reasoning_encrypted_content === "string" && msg.reasoning_encrypted_content) ||
|
||||
(typeof msg.reasoning?.encrypted_content === "string" && msg.reasoning.encrypted_content) ||
|
||||
"";
|
||||
|
||||
let summaryText = "";
|
||||
if (typeof msg.reasoning_content === "string" && msg.reasoning_content.trim()) {
|
||||
summaryText = msg.reasoning_content;
|
||||
} else if (typeof msg.reasoning === "string" && msg.reasoning.trim()) {
|
||||
summaryText = msg.reasoning;
|
||||
} else if (Array.isArray(msg.reasoning_details)) {
|
||||
summaryText = msg.reasoning_details
|
||||
.map((d) => (typeof d?.text === "string" ? d.text : typeof d?.content === "string" ? d.content : ""))
|
||||
.filter(Boolean)
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
if (!encrypted && !summaryText) return null;
|
||||
|
||||
const item = { type: RESPONSES_ITEM.REASONING };
|
||||
if (summaryText) {
|
||||
item.summary = [{ type: RESPONSES_ITEM.SUMMARY_TEXT, text: summaryText }];
|
||||
}
|
||||
// encrypted_content is the continuity token for store=false backends
|
||||
if (encrypted) item.encrypted_content = encrypted;
|
||||
return item;
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert OpenAI Chat Completions to OpenAI Responses API format
|
||||
*/
|
||||
@@ -221,17 +271,26 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials)
|
||||
const messages = body.messages || [];
|
||||
|
||||
for (const msg of messages) {
|
||||
if (msg.role === ROLE.SYSTEM) {
|
||||
// Use first system message as instructions
|
||||
if (msg.role === ROLE.SYSTEM || msg.role === ROLE.DEVELOPER) {
|
||||
// Use the first instruction-bearing message as instructions.
|
||||
// OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt.
|
||||
if (!hasSystemMessage) {
|
||||
result.instructions = typeof msg.content === "string" ? msg.content : "";
|
||||
hasSystemMessage = true;
|
||||
}
|
||||
continue; // Skip system messages in input
|
||||
continue; // Skip instruction messages in input
|
||||
}
|
||||
|
||||
// Convert user/assistant messages to input items
|
||||
if (msg.role === ROLE.USER || msg.role === ROLE.ASSISTANT) {
|
||||
// Multi-turn continuity for store=false Responses backends (Codex / Grok CLI):
|
||||
// re-emit a reasoning item before the assistant message when the chat-format
|
||||
// history carried reasoning text and/or encrypted_content from a prior turn.
|
||||
if (msg.role === ROLE.ASSISTANT) {
|
||||
const reasoningItem = buildReasoningInputItem(msg);
|
||||
if (reasoningItem) result.input.push(reasoningItem);
|
||||
}
|
||||
|
||||
const contentType = msg.role === ROLE.USER ? RESPONSES_ITEM.INPUT_TEXT : RESPONSES_ITEM.OUTPUT_TEXT;
|
||||
const content = typeof msg.content === "string"
|
||||
? [{ type: contentType, text: msg.content }]
|
||||
|
||||
@@ -6,6 +6,7 @@ import { safeParseJSON } from "../concerns/json.js";
|
||||
import { parseDataUri } from "../concerns/image.js";
|
||||
import { extractTextContent } from "../formats/gemini.js";
|
||||
import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js";
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
|
||||
// Empty prefix matches real Claude Code behavior (no tool name prefix).
|
||||
// Previously "proxy_" was used but this is a detectable fingerprint difference.
|
||||
@@ -15,9 +16,13 @@ const CLAUDE_OAUTH_TOOL_PREFIX = "";
|
||||
export function openaiToClaudeRequest(model, body, stream) {
|
||||
// Tool name mapping for Claude OAuth (capitalizedName → originalName)
|
||||
const toolNameMap = new Map();
|
||||
// Cap max_tokens at the model's real output ceiling (e.g. Opus 4.8 = 128000),
|
||||
// not the conservative 64000 default — otherwise a high-output model is
|
||||
// pre-clamped here before prepareClaudeRequest's model-aware step runs.
|
||||
const modelCeiling = getCapabilitiesForModel(null, model).maxOutput || undefined;
|
||||
const result = {
|
||||
model: model,
|
||||
max_tokens: adjustMaxTokens(body),
|
||||
max_tokens: adjustMaxTokens(body, modelCeiling),
|
||||
stream: stream
|
||||
};
|
||||
|
||||
@@ -148,7 +153,15 @@ Respond ONLY with the JSON object, no other text.`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const toolData = toolType === OPENAI_BLOCK.FUNCTION && tool.function ? tool.function : tool;
|
||||
// Function-shaped tools arrive in two flavors from real clients:
|
||||
// (a) openai-spec: { type: "function", function: { name, ... } }
|
||||
// (b) legacy/loose: { function: { name, ... } } (no parent `type`)
|
||||
// Both must yield toolData.name = "echo". Treat the bare-function shape
|
||||
// as a function tool too — Anthropic-compatible gateways (notably
|
||||
// MiniMax M3 at api.minimaxi.com) reject payloads where this branch
|
||||
// falls through with `toolData.name === undefined`, returning their
|
||||
// upstream code (2013) "invalid tool type". See #2435.
|
||||
const toolData = tool.function ?? tool;
|
||||
const originalName = toolData.name;
|
||||
|
||||
// Claude OAuth requires prefixed tool names to avoid conflicts
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js";
|
||||
import { ANTIGRAVITY_DEFAULT_SYSTEM } from "../../config/appConstants.js";
|
||||
import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js";
|
||||
function generateUUID() {
|
||||
return crypto.randomUUID();
|
||||
@@ -282,31 +281,17 @@ function wrapInCloudCodeEnvelope(model, geminiCLI, credentials = null, isAntigra
|
||||
// Antigravity specific fields
|
||||
if (isAntigravity) {
|
||||
envelope.requestType = "agent";
|
||||
|
||||
// Inject required default system prompt for Antigravity
|
||||
// Inject required default system prompt for Antigravity (double injection)
|
||||
const systemParts = [
|
||||
{ text: ANTIGRAVITY_DEFAULT_SYSTEM },
|
||||
{ text: `Please ignore the following [ignore]${ANTIGRAVITY_DEFAULT_SYSTEM}[/ignore]` }
|
||||
];
|
||||
|
||||
if (envelope.request.systemInstruction?.parts) {
|
||||
envelope.request.systemInstruction.parts.unshift(...systemParts);
|
||||
} else {
|
||||
envelope.request.systemInstruction = { role: GEMINI_ROLE.USER, parts: systemParts };
|
||||
}
|
||||
|
||||
// Add toolConfig for Antigravity
|
||||
if (geminiCLI.tools?.length > 0) {
|
||||
envelope.request.toolConfig = {
|
||||
functionCallingConfig: { mode: "VALIDATED" }
|
||||
};
|
||||
}
|
||||
} else {
|
||||
// Keep safetySettings for Gemini CLI
|
||||
envelope.request.safetySettings = geminiCLI.safetySettings;
|
||||
}
|
||||
|
||||
if (geminiCLI.tools?.length > 0) {
|
||||
envelope.request.toolConfig = {
|
||||
functionCallingConfig: { mode: "VALIDATED" }
|
||||
};
|
||||
}
|
||||
|
||||
return envelope;
|
||||
}
|
||||
|
||||
@@ -414,12 +399,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
|
||||
}
|
||||
}
|
||||
|
||||
// Add system instruction (Antigravity default - double injection + user system prompt)
|
||||
const systemParts = [
|
||||
{ text: ANTIGRAVITY_DEFAULT_SYSTEM },
|
||||
{ text: `Please ignore the following [ignore]${ANTIGRAVITY_DEFAULT_SYSTEM}[/ignore]` }
|
||||
];
|
||||
|
||||
const systemParts = [];
|
||||
// Merge user system prompt from claudeRequest
|
||||
if (claudeRequest.system) {
|
||||
if (Array.isArray(claudeRequest.system)) {
|
||||
@@ -431,10 +411,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
|
||||
}
|
||||
}
|
||||
|
||||
// Merge existing systemInstruction parts (from contents conversion)
|
||||
if (envelope.request.systemInstruction?.parts) {
|
||||
envelope.request.systemInstruction.parts.unshift(...systemParts);
|
||||
} else {
|
||||
if (systemParts.length > 0) {
|
||||
envelope.request.systemInstruction = { role: GEMINI_ROLE.USER, parts: systemParts };
|
||||
}
|
||||
|
||||
@@ -463,4 +440,3 @@ export function openaiToAntigravityRequest(model, body, stream, credentials = nu
|
||||
register(FORMATS.OPENAI, FORMATS.GEMINI, openaiToGeminiRequest, null);
|
||||
register(FORMATS.OPENAI, FORMATS.GEMINI_CLI, (model, body, stream, credentials) => wrapInCloudCodeEnvelope(model, openaiToGeminiCLIRequest(model, body, stream), credentials), null);
|
||||
register(FORMATS.OPENAI, FORMATS.ANTIGRAVITY, openaiToAntigravityRequest, null);
|
||||
|
||||
|
||||
@@ -5,13 +5,15 @@
|
||||
import { register } from "../index.js";
|
||||
import { FORMATS } from "../formats.js";
|
||||
import { v4 as uuidv4 } from "uuid";
|
||||
import { resolveSessionId } from "../../utils/sessionManager.js";
|
||||
import { applyKiroSessionReplay } from "../../utils/kiroSessionReplay.js";
|
||||
import { resolveContinuationId, resolveSessionIdentity } from "../../utils/sessionManager.js";
|
||||
import {
|
||||
resolveKiroModel,
|
||||
resolveKiroThinkingBudget,
|
||||
buildThinkingSystemPrefix,
|
||||
KIRO_AGENTIC_SYSTEM_PROMPT,
|
||||
resolveDefaultProfileArn
|
||||
resolveDefaultProfileArn,
|
||||
buildKiroAdditionalModelRequestFieldsForModel
|
||||
} from "../../config/kiroConstants.js";
|
||||
import { parseDataUri } from "../concerns/image.js";
|
||||
import { DEFAULT_IMAGE_MIME } from "../schema/index.js";
|
||||
@@ -270,6 +272,7 @@ function convertMessages(messages, tools, model) {
|
||||
let role = msg.role;
|
||||
|
||||
// Normalize: system/tool -> user
|
||||
const wasSystem = role === ROLE.SYSTEM;
|
||||
if (role === ROLE.SYSTEM || role === ROLE.TOOL) {
|
||||
role = ROLE.USER;
|
||||
}
|
||||
@@ -338,7 +341,10 @@ function convertMessages(messages, tools, model) {
|
||||
content: [{ text: toolContent }]
|
||||
});
|
||||
} else if (content) {
|
||||
pendingUserContent.push(content);
|
||||
// <instructions> tags: Claude models treat these as authoritative directives.
|
||||
pendingUserContent.push(
|
||||
wasSystem ? `<instructions>\n${content}\n</instructions>` : content
|
||||
);
|
||||
}
|
||||
} else if (role === ROLE.ASSISTANT) {
|
||||
// Extract text content and tool uses
|
||||
@@ -542,47 +548,74 @@ export function openaiToKiroRequest(model, body, stream, credentials) {
|
||||
? (credentials?.providerSpecificData?.profileArn || "")
|
||||
: (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod));
|
||||
|
||||
let finalContent = currentMessage?.userInputMessage?.content || "";
|
||||
|
||||
const timestamp = new Date().toISOString();
|
||||
|
||||
// Build the system-prompt prefix that goes ABOVE the user message body.
|
||||
// Order: thinking_mode tag first (so Kiro sees it before any user text),
|
||||
// then context/timestamp marker, then optional agentic chunked-write prompt.
|
||||
const prefixParts = [];
|
||||
// Kiro CLI/KAS sends these as top-level systemPrompt. Keep a content fallback
|
||||
// too because the CodeWhisperer surface does not always enforce top-level
|
||||
// systemPrompt for direct calls.
|
||||
const systemPromptParts = [];
|
||||
if (thinkingBudget !== null) {
|
||||
prefixParts.push(buildThinkingSystemPrefix(thinkingBudget));
|
||||
systemPromptParts.push(buildThinkingSystemPrefix(thinkingBudget));
|
||||
}
|
||||
prefixParts.push(`[Context: Current time is ${timestamp}]`);
|
||||
if (agentic) {
|
||||
prefixParts.push(KIRO_AGENTIC_SYSTEM_PROMPT);
|
||||
systemPromptParts.push(KIRO_AGENTIC_SYSTEM_PROMPT);
|
||||
}
|
||||
finalContent = `${prefixParts.join("\n\n")}\n\n${finalContent}`;
|
||||
const systemPrompt = systemPromptParts.filter(Boolean).join("\n\n");
|
||||
const currentTimeContext = `[Context: Current time is ${timestamp}]`;
|
||||
const contentPrefix = [systemPrompt, currentTimeContext].filter(Boolean).join("\n\n");
|
||||
|
||||
const sessionIdentity = resolveSessionIdentity({ headers: credentials?.rawHeaders, body, connectionId: credentials?.connectionId, scope: "kiro" });
|
||||
const conversationId = sessionIdentity.sessionId;
|
||||
const continuationId = resolveContinuationId({
|
||||
sessionId: conversationId,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "kiro",
|
||||
ephemeral: sessionIdentity.ephemeral,
|
||||
});
|
||||
const replay = applyKiroSessionReplay({
|
||||
conversationId,
|
||||
connectionId: credentials?.connectionId,
|
||||
modelId: upstreamModel,
|
||||
systemPrompt,
|
||||
contentPrefix,
|
||||
currentContentPrefix: currentTimeContext,
|
||||
history,
|
||||
currentMessage,
|
||||
});
|
||||
const replayCurrent = replay.currentMessage?.userInputMessage || {};
|
||||
|
||||
const payload = {
|
||||
conversationState: {
|
||||
chatTriggerType: "MANUAL",
|
||||
conversationId: resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.connectionId, scope: "kiro" }),
|
||||
conversationId,
|
||||
agentContinuationId: continuationId,
|
||||
agentTaskType: "vibe",
|
||||
currentMessage: {
|
||||
userInputMessage: {
|
||||
content: finalContent,
|
||||
content: replayCurrent.content || "",
|
||||
modelId: upstreamModel,
|
||||
origin: "AI_EDITOR",
|
||||
...(currentMessage?.userInputMessage?.images?.length > 0 && {
|
||||
images: currentMessage.userInputMessage.images
|
||||
...(replayCurrent.images?.length > 0 && {
|
||||
images: replayCurrent.images
|
||||
}),
|
||||
...(currentMessage?.userInputMessage?.userInputMessageContext && {
|
||||
userInputMessageContext: currentMessage.userInputMessage.userInputMessageContext
|
||||
...(replayCurrent.userInputMessageContext && {
|
||||
userInputMessageContext: replayCurrent.userInputMessageContext
|
||||
})
|
||||
}
|
||||
},
|
||||
history: history
|
||||
}
|
||||
history: replay.history
|
||||
},
|
||||
agentMode: "vibe",
|
||||
};
|
||||
|
||||
if (profileArn) {
|
||||
payload.profileArn = profileArn;
|
||||
}
|
||||
if (systemPrompt) payload.systemPrompt = systemPrompt;
|
||||
const additionalModelRequestFields = buildKiroAdditionalModelRequestFieldsForModel(body, upstreamModel);
|
||||
if (additionalModelRequestFields) {
|
||||
payload.additionalModelRequestFields = additionalModelRequestFields;
|
||||
}
|
||||
|
||||
if (maxTokens || temperature !== undefined || topP !== undefined) {
|
||||
payload.inferenceConfig = {};
|
||||
|
||||
125
open-sse/utils/kiroSessionReplay.js
Normal file
125
open-sse/utils/kiroSessionReplay.js
Normal file
@@ -0,0 +1,125 @@
|
||||
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
|
||||
const sessionStartStore = new Map();
|
||||
const MAX_SESSION_STARTS = 5000;
|
||||
|
||||
function clone(value) {
|
||||
return value == null ? value : JSON.parse(JSON.stringify(value));
|
||||
}
|
||||
|
||||
function sessionKey(connectionId, conversationId) {
|
||||
return `${connectionId || ""}:${conversationId || ""}`;
|
||||
}
|
||||
|
||||
function ensureUserMessageModelId(message, modelId) {
|
||||
if (message?.userInputMessage && !message.userInputMessage.modelId && modelId) {
|
||||
message.userInputMessage.modelId = modelId;
|
||||
}
|
||||
return message;
|
||||
}
|
||||
|
||||
function ensureHistoryModelIds(history, modelId) {
|
||||
for (const item of history || []) {
|
||||
ensureUserMessageModelId(item, modelId);
|
||||
}
|
||||
return history;
|
||||
}
|
||||
|
||||
function prefixUserMessage(message, contentPrefix, modelId) {
|
||||
const out = clone(message) || { userInputMessage: { content: "" } };
|
||||
if (!out.userInputMessage) out.userInputMessage = { content: "" };
|
||||
ensureUserMessageModelId(out, modelId);
|
||||
if (contentPrefix) {
|
||||
const content = out.userInputMessage.content || "";
|
||||
out.userInputMessage.content = content
|
||||
? `${contentPrefix}\n\n${content}`
|
||||
: contentPrefix;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function findFirstUserIndex(history) {
|
||||
return history.findIndex((item) => item?.userInputMessage);
|
||||
}
|
||||
|
||||
function rememberSessionStart(key, entry) {
|
||||
if (sessionStartStore.size >= MAX_SESSION_STARTS) {
|
||||
sessionStartStore.delete(sessionStartStore.keys().next().value);
|
||||
}
|
||||
sessionStartStore.set(key, { ...entry, lastUsed: Date.now() });
|
||||
}
|
||||
|
||||
/**
|
||||
* Preserve Kiro cacheability by freezing the first user message (`msg0`) for a
|
||||
* session, replaying that exact message as the first history user on later
|
||||
* turns, and injecting volatile current-time context only into the current turn.
|
||||
*/
|
||||
export function applyKiroSessionReplay({
|
||||
conversationId,
|
||||
connectionId,
|
||||
modelId,
|
||||
systemPrompt = "",
|
||||
contentPrefix = "",
|
||||
currentContentPrefix = "",
|
||||
history = [],
|
||||
currentMessage,
|
||||
} = {}) {
|
||||
const key = sessionKey(connectionId, conversationId);
|
||||
const existing = conversationId ? sessionStartStore.get(key) : null;
|
||||
const baseHistory = clone(history) || [];
|
||||
const baseCurrent = clone(currentMessage) || { userInputMessage: { content: "" } };
|
||||
|
||||
if (existing && existing.modelId === modelId && existing.systemPrompt === systemPrompt) {
|
||||
existing.lastUsed = Date.now();
|
||||
const firstUserIndex = findFirstUserIndex(baseHistory);
|
||||
const sessionStart = ensureUserMessageModelId(clone(existing.sessionStart), modelId);
|
||||
if (firstUserIndex >= 0) {
|
||||
baseHistory[firstUserIndex] = sessionStart;
|
||||
} else {
|
||||
baseHistory.unshift(sessionStart);
|
||||
}
|
||||
return {
|
||||
history: ensureHistoryModelIds(baseHistory, modelId),
|
||||
currentMessage: prefixUserMessage(baseCurrent, currentContentPrefix, modelId),
|
||||
replayed: true,
|
||||
};
|
||||
}
|
||||
|
||||
const firstUserIndex = findFirstUserIndex(baseHistory);
|
||||
let sessionStart;
|
||||
let nextCurrent = ensureUserMessageModelId(baseCurrent, modelId);
|
||||
if (firstUserIndex >= 0) {
|
||||
sessionStart = prefixUserMessage(baseHistory[firstUserIndex], contentPrefix, modelId);
|
||||
baseHistory[firstUserIndex] = clone(sessionStart);
|
||||
nextCurrent = prefixUserMessage(baseCurrent, currentContentPrefix, modelId);
|
||||
} else {
|
||||
sessionStart = prefixUserMessage(baseCurrent, contentPrefix, modelId);
|
||||
nextCurrent = clone(sessionStart);
|
||||
}
|
||||
|
||||
if (conversationId) {
|
||||
rememberSessionStart(key, {
|
||||
sessionStart: clone(sessionStart),
|
||||
modelId,
|
||||
systemPrompt,
|
||||
});
|
||||
}
|
||||
|
||||
return {
|
||||
history: ensureHistoryModelIds(baseHistory, modelId),
|
||||
currentMessage: nextCurrent,
|
||||
replayed: false,
|
||||
};
|
||||
}
|
||||
|
||||
export function clearKiroSessionReplayStore() {
|
||||
sessionStartStore.clear();
|
||||
}
|
||||
|
||||
const cleanup = setInterval(() => {
|
||||
const now = Date.now();
|
||||
for (const [key, entry] of sessionStartStore) {
|
||||
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) sessionStartStore.delete(key);
|
||||
}
|
||||
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
|
||||
if (cleanup.unref) cleanup.unref();
|
||||
@@ -13,6 +13,7 @@ import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
|
||||
// Runtime storage: Key = connectionId, Value = { sessionId, lastUsed }
|
||||
const runtimeSessionStore = new Map();
|
||||
const continuationStore = new Map();
|
||||
|
||||
// Periodically evict entries that haven't been used within TTL
|
||||
const cleanupInterval = setInterval(() => {
|
||||
@@ -80,6 +81,7 @@ export function generateBinaryStyleId() {
|
||||
export function clearSessionStore() {
|
||||
runtimeSessionStore.clear();
|
||||
assistantSessionStore.clear();
|
||||
continuationStore.clear();
|
||||
}
|
||||
|
||||
// Conversation-stable session store: Key = hash(scope+assistant text), Value = { sessionId, lastUsed }
|
||||
@@ -87,9 +89,10 @@ const assistantSessionStore = new Map();
|
||||
const ASSISTANT_MIN_LEN = 50;
|
||||
const ASSISTANT_CAP_LEN = 50;
|
||||
const MAX_ASSISTANT_SESSIONS = 5000;
|
||||
const MAX_CONTINUATION_SESSIONS = 5000;
|
||||
|
||||
// Client headers/body fields that carry an upstream session id (priority order)
|
||||
const SESSION_HEADER_KEYS = ["x-session-id", "session-id", "session_id", "x-amp-thread-id", "x-client-request-id"];
|
||||
const SESSION_HEADER_KEYS = ["x-session-id", "session-id", "session_id", "x-amp-thread-id"];
|
||||
const CLAUDE_CODE_SESSION_RE = /_session_([a-f0-9-]+)$/;
|
||||
|
||||
function sha16(text) {
|
||||
@@ -131,7 +134,7 @@ function extractAntigravitySession(body) {
|
||||
return m ? normalizeSessionId(m[1]) : null;
|
||||
}
|
||||
|
||||
function extractClientSessionId(headers, body) {
|
||||
function extractClientSessionId(headers, body, scope = "") {
|
||||
const claude = extractClaudeCodeSession(body?.metadata?.user_id);
|
||||
if (claude) return `claude:${claude}`;
|
||||
const antigravity = extractAntigravitySession(body);
|
||||
@@ -140,18 +143,25 @@ function extractClientSessionId(headers, body) {
|
||||
const v = headerValue(headers, key);
|
||||
if (v) return v;
|
||||
}
|
||||
const requestId = scope === "kiro" ? null : headerValue(headers, "x-client-request-id");
|
||||
if (requestId) return requestId;
|
||||
const fromBody =
|
||||
normalizeSessionId(body?.prompt_cache_key) ||
|
||||
normalizeSessionId(body?.session_id) ||
|
||||
normalizeSessionId(body?.conversation_id) ||
|
||||
normalizeSessionId(body?.metadata?.user_id);
|
||||
(scope === "kiro" ? null : normalizeSessionId(body?.metadata?.user_id));
|
||||
return fromBody || null;
|
||||
}
|
||||
|
||||
function requestMessages(body) {
|
||||
if (Array.isArray(body?.messages)) return body.messages;
|
||||
if (Array.isArray(body?.input)) return body.input;
|
||||
return [];
|
||||
}
|
||||
|
||||
// Accumulate assistant text from OpenAI/Responses-style input/messages (cap-limited)
|
||||
function accumulateAssistantText(body) {
|
||||
const items = Array.isArray(body?.input) ? body.input
|
||||
: Array.isArray(body?.messages) ? body.messages : null;
|
||||
const items = requestMessages(body);
|
||||
if (!items) return "";
|
||||
let text = "";
|
||||
for (const item of items) {
|
||||
@@ -193,16 +203,39 @@ function assistantTextSessionId(scope, body) {
|
||||
* @param {string} [opts.connectionId] - Connection identifier (fallback scope)
|
||||
* @param {string} [opts.workspaceId] - Provider workspace id (account-wide fallback)
|
||||
* @param {string} [opts.scope] - Provider scope to isolate cache keys across providers
|
||||
* @returns {string} A stable session id
|
||||
* @returns {{sessionId: string, ephemeral: boolean}} A session id plus whether it is one-shot
|
||||
*/
|
||||
export function resolveSessionId({ headers, body, connectionId, workspaceId, scope = "" } = {}) {
|
||||
const client = extractClientSessionId(headers, body);
|
||||
if (client) return client;
|
||||
const fromAssistant = assistantTextSessionId(`${scope}:${connectionId || ""}`, body);
|
||||
if (fromAssistant) return fromAssistant;
|
||||
export function resolveSessionIdentity({ headers, body, connectionId, workspaceId, scope = "" } = {}) {
|
||||
const client = extractClientSessionId(headers, body, scope);
|
||||
if (client) return { sessionId: client, ephemeral: false };
|
||||
const fromAssistant = scope === "kiro" ? null : assistantTextSessionId(`${scope}:${connectionId || ""}`, body);
|
||||
if (fromAssistant) return { sessionId: fromAssistant, ephemeral: false };
|
||||
const ws = normalizeSessionId(workspaceId);
|
||||
if (ws) return ws;
|
||||
return deriveSessionId(connectionId);
|
||||
if (ws) return { sessionId: ws, ephemeral: false };
|
||||
if (scope === "kiro") return { sessionId: generateBinaryStyleId(), ephemeral: true };
|
||||
return { sessionId: deriveSessionId(connectionId), ephemeral: false };
|
||||
}
|
||||
|
||||
export function resolveSessionId(opts = {}) {
|
||||
return resolveSessionIdentity(opts).sessionId;
|
||||
}
|
||||
|
||||
export function resolveContinuationId({ sessionId, connectionId, scope = "", ephemeral = false } = {}) {
|
||||
if (ephemeral) return crypto.randomUUID();
|
||||
const key = `${scope}:${connectionId || ""}:${sessionId || ""}`;
|
||||
const existing = continuationStore.get(key);
|
||||
if (existing) {
|
||||
existing.lastUsed = Date.now();
|
||||
continuationStore.delete(key);
|
||||
continuationStore.set(key, existing);
|
||||
return existing.continuationId;
|
||||
}
|
||||
const continuationId = crypto.randomUUID();
|
||||
if (continuationStore.size >= MAX_CONTINUATION_SESSIONS) {
|
||||
continuationStore.delete(continuationStore.keys().next().value);
|
||||
}
|
||||
continuationStore.set(key, { continuationId, lastUsed: Date.now() });
|
||||
return continuationId;
|
||||
}
|
||||
|
||||
// Capture session id from request body + credentials (envelope still intact here)
|
||||
@@ -227,5 +260,8 @@ const assistantCleanup = setInterval(() => {
|
||||
for (const [key, entry] of assistantSessionStore) {
|
||||
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) assistantSessionStore.delete(key);
|
||||
}
|
||||
for (const [key, entry] of continuationStore) {
|
||||
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) continuationStore.delete(key);
|
||||
}
|
||||
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
|
||||
if (assistantCleanup.unref) assistantCleanup.unref();
|
||||
|
||||
@@ -15,16 +15,19 @@ function getTimeString() {
|
||||
* @param {string} options.provider - Provider name
|
||||
* @param {string} options.model - Model name
|
||||
*/
|
||||
export function createStreamController({ onDisconnect, onError, log, provider, model } = {}) {
|
||||
export function createStreamController({ onDisconnect, onError, log, provider, model, reqTag = "" } = {}) {
|
||||
const abortController = new AbortController();
|
||||
const startTime = Date.now();
|
||||
let disconnected = false;
|
||||
let abortTimeout = null;
|
||||
|
||||
const logStream = (status) => {
|
||||
// Only abnormal terminations are logged; normal completion is covered by "📊 done".
|
||||
// isError uses errorLine (always shown, ignores LOG_LEVEL) so failures survive quiet levels.
|
||||
const logStream = (symbol, status, isError = false) => {
|
||||
const duration = Date.now() - startTime;
|
||||
const p = provider?.toUpperCase() || "UNKNOWN";
|
||||
console.log(`[${getTimeString()}] 🌊 [STREAM] ${p} | ${model || "unknown"} | ${duration}ms | ${status}`);
|
||||
const emit = isError ? log?.errorLine : log?.line;
|
||||
if (emit) emit(reqTag, symbol, `${status} · ${provider}/${model} · ${duration}ms`);
|
||||
else console.log(`[${getTimeString()}] ${symbol} ${provider}/${model} · ${status} · ${duration}ms`);
|
||||
};
|
||||
|
||||
return {
|
||||
@@ -38,7 +41,7 @@ export function createStreamController({ onDisconnect, onError, log, provider, m
|
||||
if (disconnected) return;
|
||||
disconnected = true;
|
||||
|
||||
logStream(`disconnect: ${reason}`);
|
||||
logStream("⚡", `DISCONNECT: ${reason}`);
|
||||
dbg("CTRL", `${provider}/${model} | disconnect=${reason} | dur=${Date.now() - startTime}ms`);
|
||||
|
||||
// Delay abort to allow cleanup
|
||||
@@ -49,13 +52,11 @@ export function createStreamController({ onDisconnect, onError, log, provider, m
|
||||
onDisconnect?.({ reason, duration: Date.now() - startTime });
|
||||
},
|
||||
|
||||
// Call when stream completes normally
|
||||
// Call when stream completes normally (no line here — "📊 done" is authoritative)
|
||||
handleComplete: () => {
|
||||
if (disconnected) return;
|
||||
disconnected = true;
|
||||
|
||||
logStream("complete");
|
||||
|
||||
if (abortTimeout) {
|
||||
clearTimeout(abortTimeout);
|
||||
abortTimeout = null;
|
||||
@@ -73,11 +74,11 @@ export function createStreamController({ onDisconnect, onError, log, provider, m
|
||||
}
|
||||
|
||||
if (error.name === "AbortError") {
|
||||
logStream("aborted");
|
||||
logStream("⚡", "ABORTED");
|
||||
return;
|
||||
}
|
||||
|
||||
logStream(`error: ${error.message}`);
|
||||
logStream("✗", `ERROR: ${error.message}${error.stack ? `\n ${error.stack}` : ""}`, true);
|
||||
onError?.(error);
|
||||
},
|
||||
|
||||
|
||||
@@ -4,6 +4,9 @@
|
||||
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
|
||||
// Legacy per-chunk usage console line; off by default (superseded by "📊 done")
|
||||
const DEBUG_USAGE = process.env.LOG_USAGE_VERBOSE === "1";
|
||||
|
||||
// ANSI color codes
|
||||
export const COLORS = {
|
||||
reset: "\x1b[0m",
|
||||
@@ -401,6 +404,10 @@ export function estimateUsage(body, contentLength, targetFormat = FORMATS.OPENAI
|
||||
export function logUsage(provider, usage, model = null, connectionId = null, apiKey = null) {
|
||||
if (!usage || typeof usage !== "object") return;
|
||||
|
||||
// Console output moved to the unified "📊 done" line (streamingHandler). Kept as
|
||||
// a no-op hook so callers stay unchanged; usage persistence happens via saveUsageStats.
|
||||
if (!DEBUG_USAGE) return;
|
||||
|
||||
const p = provider?.toUpperCase() || "UNKNOWN";
|
||||
|
||||
// Support both formats:
|
||||
|
||||
Reference in New Issue
Block a user