merge origin/master into gitea/new_feature
Bring local branch up to v0.5.35 while keeping xAI image/edit, SuperGrok quota tracking, per-provider timeouts, and pinned model-test actions.
This commit is contained in:
@@ -1,7 +1,7 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, INTERNAL_REQUEST_HEADER, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
@@ -18,7 +18,8 @@ function sanitizeFunctionName(name) {
|
||||
|
||||
const MAX_RETRY_AFTER_MS = 10000;
|
||||
const ANTIGRAVITY_TRANSIENT_RETRY_MAX_MS = 15000;
|
||||
const MAX_ANTIGRAVITY_OUTPUT_TOKENS = 16384;
|
||||
const MAX_ANTIGRAVITY_OUTPUT_TOKENS = 64000;
|
||||
const ANTIGRAVITY_IDE_REQUEST_ID_RE = /^agent\/[^/]+\/\d+\/[^/]+\/\d+$/;
|
||||
|
||||
const ANTIGRAVITY_TRANSIENT_ERROR_PATTERNS = [
|
||||
/high\s+traffic/i,
|
||||
@@ -87,6 +88,27 @@ function parseImageConfig(model) {
|
||||
return config;
|
||||
}
|
||||
|
||||
function uuidFromSeed(seed) {
|
||||
const bytes = crypto.createHash("sha256").update(String(seed || "antigravity")).digest().subarray(0, 16);
|
||||
bytes[6] = (bytes[6] & 0x0f) | 0x50;
|
||||
bytes[8] = (bytes[8] & 0x3f) | 0x80;
|
||||
const hex = bytes.toString("hex");
|
||||
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}`;
|
||||
}
|
||||
|
||||
function buildIdeRequestId({ body, request, credentials, model, requestType }) {
|
||||
if (ANTIGRAVITY_IDE_REQUEST_ID_RE.test(body?.requestId || "")) {
|
||||
return body.requestId;
|
||||
}
|
||||
|
||||
const sessionId = request?.sessionId || body?.request?.sessionId || credentials?._clientSessionId || credentials?.connectionId || credentials?.email || "anonymous";
|
||||
const conversationId = uuidFromSeed(`antigravity:conversation:${sessionId}`);
|
||||
const trajectoryId = uuidFromSeed(`antigravity:trajectory:${sessionId}:${model}:${requestType}`);
|
||||
const contentCount = Array.isArray(request?.contents) ? request.contents.length : 1;
|
||||
const step = Math.max(1, contentCount * 2 - 1);
|
||||
return `agent/${conversationId}/${Date.now()}/${trajectoryId}/${step}`;
|
||||
}
|
||||
|
||||
export class AntigravityExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("antigravity", PROVIDERS.antigravity);
|
||||
@@ -104,14 +126,10 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
// sessionId comes from transformRequest output; base.execute runs transformRequest before
|
||||
// buildHeaders, so we read it from instance state cached there (fallback: explicit arg).
|
||||
buildHeaders(credentials, stream = true, sessionId = null) {
|
||||
const sid = sessionId || this._lastSessionId;
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${credentials.accessToken}`,
|
||||
"User-Agent": this.config.headers?.["User-Agent"] || ANTIGRAVITY_HEADERS["User-Agent"],
|
||||
[INTERNAL_REQUEST_HEADER.name]: INTERNAL_REQUEST_HEADER.value,
|
||||
...(sid && { "X-Machine-Session-Id": sid }),
|
||||
"Accept": stream ? "text/event-stream" : "application/json"
|
||||
};
|
||||
}
|
||||
|
||||
@@ -142,25 +160,26 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
});
|
||||
|
||||
this._lastSessionId = sessionId;
|
||||
const request = {
|
||||
contents,
|
||||
generationConfig: {
|
||||
temperature: 1.0,
|
||||
topP: 0.95,
|
||||
topK: 40,
|
||||
maxOutputTokens: 8192,
|
||||
imageConfig,
|
||||
},
|
||||
sessionId,
|
||||
// No tools, no systemInstruction, no safetySettings for image gen
|
||||
};
|
||||
|
||||
return {
|
||||
project: projectId,
|
||||
model: cleanModel,
|
||||
userAgent: "antigravity",
|
||||
requestType: "image_gen",
|
||||
requestId: `agent-${crypto.randomUUID()}`,
|
||||
request: {
|
||||
contents,
|
||||
generationConfig: {
|
||||
temperature: 1.0,
|
||||
topP: 0.95,
|
||||
topK: 40,
|
||||
maxOutputTokens: 8192,
|
||||
imageConfig,
|
||||
},
|
||||
sessionId,
|
||||
// No tools, no systemInstruction, no safetySettings for image gen
|
||||
},
|
||||
requestId: buildIdeRequestId({ body, request, credentials, model: cleanModel, requestType: "image_gen" }),
|
||||
request,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -248,7 +267,7 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
model: model,
|
||||
userAgent: "antigravity",
|
||||
requestType: "agent",
|
||||
requestId: `agent-${crypto.randomUUID()}`,
|
||||
requestId: buildIdeRequestId({ body, request: transformedRequest, credentials, model, requestType: "agent" }),
|
||||
request: transformedRequest
|
||||
};
|
||||
}
|
||||
|
||||
@@ -8,13 +8,21 @@ import {
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
// SSE error patterns inside 200-OK body that should trigger retry as if 503
|
||||
const CODEX_SSE_OVERLOADED_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
|
||||
const CODEX_SSE_PEEK_BYTES = 4096;
|
||||
// SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts.
|
||||
const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
|
||||
const CODEX_SSE_ACCOUNT_FALLBACK_PATTERNS = ["selected model is at capacity", "model_at_capacity"];
|
||||
const CODEX_SSE_USER_OUTPUT_PATTERNS = [
|
||||
"event: response.output_text.delta",
|
||||
"event: response.function_call_arguments.delta",
|
||||
'"type":"response.output_text.delta"',
|
||||
'"type":"response.function_call_arguments.delta"',
|
||||
];
|
||||
const CODEX_SSE_PEEK_BYTES = 256 * 1024;
|
||||
const CODEX_MODEL_CAPACITY_MESSAGE = "Selected model is at capacity. Please try a different model.";
|
||||
|
||||
// Server-generated item id prefixes that Codex /responses cannot resolve when store=false
|
||||
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
|
||||
@@ -116,6 +124,62 @@ function resolveCacheSessionId(body, credentials) {
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeReasoningEffort(value) {
|
||||
return value === "max" ? "xhigh" : value;
|
||||
}
|
||||
|
||||
function findNestedMessage(value, depth = 0) {
|
||||
if (!value || depth > 6 || typeof value === "string") return null;
|
||||
if (Array.isArray(value)) {
|
||||
for (const item of value) {
|
||||
const found = findNestedMessage(item, depth + 1);
|
||||
if (found) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
if (typeof value !== "object") return null;
|
||||
if (typeof value.message === "string" && value.message.trim()) return value.message;
|
||||
if (typeof value.error?.message === "string" && value.error.message.trim()) return value.error.message;
|
||||
if (typeof value.response?.error?.message === "string" && value.response.error.message.trim()) return value.response.error.message;
|
||||
for (const child of Object.values(value)) {
|
||||
const found = findNestedMessage(child, depth + 1);
|
||||
if (found) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractSseErrorMessage(text, fallback) {
|
||||
const exact = text?.match(/Selected model is at capacity\. Please try a different model\./i)?.[0];
|
||||
if (exact) return exact;
|
||||
|
||||
for (const line of String(text || "").split(/\r?\n/)) {
|
||||
if (!line.startsWith("data:")) continue;
|
||||
const data = line.slice(5).trim();
|
||||
if (!data || data === "[DONE]") continue;
|
||||
try {
|
||||
const message = findNestedMessage(JSON.parse(data));
|
||||
if (message) return message;
|
||||
} catch {
|
||||
// Ignore non-JSON SSE data lines.
|
||||
}
|
||||
}
|
||||
|
||||
return fallback || CODEX_MODEL_CAPACITY_MESSAGE;
|
||||
}
|
||||
|
||||
function codexSseErrorResponse(status, message) {
|
||||
return new Response(JSON.stringify({
|
||||
error: {
|
||||
message,
|
||||
type: status >= 500 ? "server_error" : "invalid_request_error",
|
||||
code: status === HTTP_STATUS.SERVICE_UNAVAILABLE ? "service_unavailable" : "upstream_error",
|
||||
}
|
||||
}), {
|
||||
status,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Codex Executor - handles OpenAI Codex API (Responses API format)
|
||||
* Automatically injects default instructions if missing
|
||||
@@ -135,10 +199,17 @@ export class CodexExecutor extends BaseExecutor {
|
||||
headers["session_id"] = this._currentSessionId || credentials?.connectionId || "default";
|
||||
// Identify client type to Codex backend (matches official codex CLI)
|
||||
if (!headers["originator"]) headers["originator"] = "codex_cli_rs";
|
||||
// Workspace binding header — improves account scope + cache affinity
|
||||
const workspaceId = credentials?.providerSpecificData?.workspaceId;
|
||||
if (typeof workspaceId === "string" && workspaceId && !headers["chatgpt-account-id"]) {
|
||||
headers["chatgpt-account-id"] = workspaceId;
|
||||
// Account/workspace binding header — required when multiple Codex accounts
|
||||
// are configured. OAuth import stores ChatGPT account ID as chatgptAccountId;
|
||||
// older/custom rows may use workspaceId/accountId. Prefer explicit workspaceId
|
||||
// but fall back to chatgptAccountId so requests don't cross-bind to the wrong
|
||||
// OpenAI account and surface as token_invalid after adding another account.
|
||||
const accountId =
|
||||
credentials?.providerSpecificData?.workspaceId ||
|
||||
credentials?.providerSpecificData?.chatgptAccountId ||
|
||||
credentials?.providerSpecificData?.accountId;
|
||||
if (typeof accountId === "string" && accountId && !headers["ChatGPT-Account-ID"]) {
|
||||
headers["ChatGPT-Account-ID"] = accountId;
|
||||
}
|
||||
return headers;
|
||||
}
|
||||
@@ -198,7 +269,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
let attempt = 0;
|
||||
while (true) {
|
||||
const result = await super.execute(args);
|
||||
const peek = await this._peekSseOverloaded(result.response);
|
||||
const peek = await this._peekSseTransientError(result.response);
|
||||
if (!peek.matched) {
|
||||
// Replace body with re-assembled stream (prefix bytes already read + rest)
|
||||
if (peek.replacementBody) {
|
||||
@@ -210,48 +281,57 @@ export class CodexExecutor extends BaseExecutor {
|
||||
}
|
||||
return result;
|
||||
}
|
||||
if (peek.accountFallback) {
|
||||
args.log?.warn?.("RETRY", `CODEX | SSE account fallback "${peek.message}"`);
|
||||
result.response = codexSseErrorResponse(HTTP_STATUS.SERVICE_UNAVAILABLE, peek.message || CODEX_MODEL_CAPACITY_MESSAGE);
|
||||
return result;
|
||||
}
|
||||
if (attempt >= attempts) {
|
||||
args.log?.warn?.("RETRY", `CODEX | SSE overloaded "${peek.matched}" — retries exhausted (${attempt}/${attempts})`);
|
||||
// Out of retries → return with replacement body so client gets the error
|
||||
if (peek.replacementBody) {
|
||||
result.response = new Response(peek.replacementBody, {
|
||||
status: result.response.status,
|
||||
statusText: result.response.statusText,
|
||||
headers: result.response.headers,
|
||||
});
|
||||
}
|
||||
result.response = codexSseErrorResponse(HTTP_STATUS.SERVICE_UNAVAILABLE, peek.message || peek.matched);
|
||||
return result;
|
||||
}
|
||||
attempt++;
|
||||
args.log?.debug?.("RETRY", `CODEX | SSE "${peek.matched}" retry ${attempt}/${attempts} after ${delayMs / 1000}s`);
|
||||
dbg("CODEX", `SSE overloaded "${peek.matched}" → retry ${attempt}/${attempts} in ${delayMs}ms`);
|
||||
try { await result.response.body?.cancel?.(); } catch { /* noop */ }
|
||||
await new Promise(r => setTimeout(r, delayMs));
|
||||
}
|
||||
}
|
||||
|
||||
// Peek first N bytes of SSE body to detect upstream "overloaded" errors.
|
||||
// Returns { matched: string|null, replacementBody: ReadableStream|null }.
|
||||
// Caller MUST use replacementBody (original body has been read).
|
||||
async _peekSseOverloaded(response) {
|
||||
if (!response || !response.ok || !response.body) return { matched: null, replacementBody: null };
|
||||
// Peek first N bytes of SSE body to detect upstream transient errors.
|
||||
// Returns { matched: string|null, message: string|null, accountFallback: boolean, replacementBody: ReadableStream|null }.
|
||||
// Caller must use replacementBody when no error matched (original body has been read).
|
||||
async _peekSseTransientError(response) {
|
||||
if (!response || !response.ok || !response.body) return { matched: null, message: null, accountFallback: false, replacementBody: null };
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
const chunks = [];
|
||||
let text = "";
|
||||
let matched = null;
|
||||
let accountFallback = false;
|
||||
try {
|
||||
while (text.length < CODEX_SSE_PEEK_BYTES) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
chunks.push(value);
|
||||
text += decoder.decode(value, { stream: true });
|
||||
const hit = CODEX_SSE_OVERLOADED_PATTERNS.find(p => text.includes(p));
|
||||
if (hit) { matched = hit; break; }
|
||||
const lowerText = text.toLowerCase();
|
||||
const accountHit = CODEX_SSE_ACCOUNT_FALLBACK_PATTERNS.find(p => lowerText.includes(p));
|
||||
if (accountHit) { matched = accountHit; accountFallback = true; break; }
|
||||
const retryHit = CODEX_SSE_RETRY_PATTERNS.find(p => lowerText.includes(p));
|
||||
if (retryHit) { matched = retryHit; break; }
|
||||
if (CODEX_SSE_USER_OUTPUT_PATTERNS.some(p => lowerText.includes(p))) break;
|
||||
}
|
||||
} catch (e) {
|
||||
dbg("CODEX", `peek read error: ${e.message}`);
|
||||
}
|
||||
|
||||
if (matched) {
|
||||
try { await reader.cancel(); } catch { /* noop */ }
|
||||
try { reader.releaseLock(); } catch { /* noop */ }
|
||||
return { matched, message: extractSseErrorMessage(text, matched), accountFallback, replacementBody: null };
|
||||
}
|
||||
|
||||
reader.releaseLock();
|
||||
|
||||
// Re-assemble stream: prefix chunks + remaining upstream body
|
||||
@@ -273,7 +353,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
try { upstreamReader?.cancel(reason); } catch { /* noop */ }
|
||||
},
|
||||
});
|
||||
return { matched, replacementBody };
|
||||
return { matched: null, message: null, accountFallback: false, replacementBody };
|
||||
}
|
||||
|
||||
// Parse Codex usage_limit_reached to extract precise resetsAtMs; fallback to default otherwise
|
||||
@@ -347,7 +427,7 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Extract thinking level from model name suffix
|
||||
// e.g., gpt-5.3-codex-high → high, gpt-5.3-codex → medium (default)
|
||||
const effortLevels = ['none', 'low', 'medium', 'high', 'xhigh'];
|
||||
const effortLevels = ['none', 'minimal', 'low', 'medium', 'high', 'xhigh'];
|
||||
let modelEffort = null;
|
||||
for (const level of effortLevels) {
|
||||
if (body.model.endsWith(`-${level}`)) {
|
||||
@@ -360,10 +440,11 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = body.reasoning_effort || modelEffort || 'low';
|
||||
const effort = normalizeReasoningEffort(body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
} else if (!body.reasoning.summary) {
|
||||
body.reasoning.summary = "auto";
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
@@ -391,6 +472,9 @@ export class CodexExecutor extends BaseExecutor {
|
||||
delete body.safety_identifier; // Droid CLI sends this but Codex doesn't support it
|
||||
delete body.previous_response_id; // store=false → backend can't resolve previous resp; avoid 404
|
||||
|
||||
if (body.service_tier === "fast") body.service_tier = "priority";
|
||||
if (body.service_tier && body.service_tier !== "priority") delete body.service_tier;
|
||||
|
||||
// Final allowlist filter — strip any unknown field that could trigger upstream "routing_unsupported"
|
||||
for (const k of Object.keys(body)) {
|
||||
if (!RESPONSES_API_ALLOWLIST.has(k)) delete body[k];
|
||||
|
||||
@@ -4,11 +4,13 @@ import { OAUTH_ENDPOINTS, GITHUB_COPILOT } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { openaiToOpenAIResponsesRequest } from "../translator/request/openai-responses.js";
|
||||
import { openaiResponsesToOpenAIResponse } from "../translator/response/openai-responses.js";
|
||||
import { initState } from "../translator/index.js";
|
||||
import { initState, translateRequest, translateResponse } from "../translator/index.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { parseSSELine, formatSSE } from "../utils/streamHelpers.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { stripUnsupportedParams } from "../translator/concerns/paramSupport.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
import crypto from "crypto";
|
||||
|
||||
export class GithubExecutor extends BaseExecutor {
|
||||
@@ -17,6 +19,16 @@ export class GithubExecutor extends BaseExecutor {
|
||||
this.knownCodexModels = new Set();
|
||||
}
|
||||
|
||||
// Claude models get routed to Copilot's Anthropic-native /v1/messages shim (see
|
||||
// executeWithMessagesEndpoint below) — the only Copilot endpoint that surfaces
|
||||
// prompt-cache token counts. gpt/gemini/grok models stay on /chat/completions
|
||||
// (or /responses). Name-pattern check, not a registry field: Copilot's live model
|
||||
// catalog (services/copilotModels.js) regularly exposes claude-* variants ahead
|
||||
// of the static registry (registry/github.js).
|
||||
isClaudeModel(model) {
|
||||
return /claude/i.test(model || "");
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0) {
|
||||
return this.config.baseUrl;
|
||||
}
|
||||
@@ -35,47 +47,20 @@ export class GithubExecutor extends BaseExecutor {
|
||||
"x-request-id": crypto.randomUUID?.() || `${Date.now()}-${Math.random().toString(36).slice(2)}`,
|
||||
"x-vscode-user-agent-library-version": "electron-fetch",
|
||||
"X-Initiator": "user",
|
||||
// Harmless no-op on /chat/completions and /responses; required by /v1/messages.
|
||||
"anthropic-version": ANTHROPIC_API_VERSION,
|
||||
"Accept": stream ? "text/event-stream" : "application/json"
|
||||
};
|
||||
}
|
||||
|
||||
// Sanitize messages for GitHub Copilot /chat/completions endpoint.
|
||||
// Sanitize messages for GitHub Copilot /chat/completions endpoint (gpt/gemini/grok models —
|
||||
// claude models never reach this, see execute() below).
|
||||
// The endpoint only accepts 'text' and 'image_url' content part types.
|
||||
// Tool-related content (tool_use, tool_result, thinking) must be serialized as text.
|
||||
sanitizeMessagesForChatCompletions(body) {
|
||||
if (!body?.messages) return body;
|
||||
|
||||
const sanitized = { ...body };
|
||||
|
||||
// Handle response_format for Claude models via GitHub
|
||||
// GitHub's internal translation doesn't respect response_format, so we inject it as a system prompt
|
||||
// AND prepend a reminder to the last user message for maximum effectiveness
|
||||
if (body.response_format && body.model?.includes('claude')) {
|
||||
const responseFormat = body.response_format;
|
||||
let systemInstruction = '';
|
||||
if (responseFormat.type === 'json_schema' && responseFormat.json_schema?.schema) {
|
||||
systemInstruction = 'CRITICAL: You must ONLY output raw JSON. Never use markdown code blocks. Never use backticks. Never wrap JSON in triple backticks. Output ONLY the raw JSON object.';
|
||||
} else if (responseFormat.type === 'json_object') {
|
||||
systemInstruction = 'CRITICAL: You must ONLY output raw JSON. Never use markdown code blocks. Never use backticks.';
|
||||
}
|
||||
if (systemInstruction) {
|
||||
// Add to system message
|
||||
const systemIdx = body.messages.findIndex(m => m.role === 'system');
|
||||
if (systemIdx >= 0) {
|
||||
body.messages[systemIdx].content = systemInstruction + '\n\n' + body.messages[systemIdx].content;
|
||||
} else {
|
||||
body.messages.unshift({ role: 'system', content: systemInstruction });
|
||||
}
|
||||
|
||||
// Also prepend to the last user message as a reminder
|
||||
const lastUserIdx = body.messages.map((m, i) => m.role === 'user' ? i : -1).filter(i => i >= 0).pop();
|
||||
if (lastUserIdx >= 0) {
|
||||
const userMsg = body.messages[lastUserIdx];
|
||||
const userContent = typeof userMsg.content === 'string' ? userMsg.content : JSON.stringify(userMsg.content);
|
||||
userMsg.content = 'Respond with ONLY raw JSON (no markdown, no backticks, no code blocks): ' + userContent;
|
||||
}
|
||||
}
|
||||
}
|
||||
sanitized.messages = body.messages.map(msg => {
|
||||
// assistant messages with only tool_calls have content: null — leave as-is
|
||||
if (!msg.content) return msg;
|
||||
@@ -138,6 +123,15 @@ export class GithubExecutor extends BaseExecutor {
|
||||
async execute(options) {
|
||||
const { model, log } = options;
|
||||
|
||||
// Claude models: route to Copilot's Anthropic-native /v1/messages shim — the only
|
||||
// Copilot endpoint that surfaces prompt-cache token counts for Claude. Detected by
|
||||
// model NAME (not a registry field): Copilot's live model catalog regularly exposes
|
||||
// claude-* variants the static registry hasn't caught up with yet (see registry/github.js).
|
||||
if (this.isClaudeModel(model)) {
|
||||
log?.debug("GITHUB", `Using /v1/messages route for ${model}`);
|
||||
return this.executeWithMessagesEndpoint(options);
|
||||
}
|
||||
|
||||
// Only use /responses for models that are explicitly known to need it (e.g. gpt codex models)
|
||||
// and that the /responses endpoint actually serves (excludes Gemini/Claude, see #1062).
|
||||
if (this.knownCodexModels.has(model) && this.supportsResponsesEndpoint(model)) {
|
||||
@@ -145,8 +139,8 @@ export class GithubExecutor extends BaseExecutor {
|
||||
return this.executeWithResponsesEndpoint(options);
|
||||
}
|
||||
|
||||
// Sanitize messages before sending to /chat/completions
|
||||
// This handles Claude models on GitHub Copilot which reject non-text/image_url content types
|
||||
// Sanitize messages before sending to /chat/completions (gpt/gemini/grok — the
|
||||
// endpoint rejects non-text/image_url content parts).
|
||||
const sanitizedOptions = {
|
||||
...options,
|
||||
body: this.sanitizeMessagesForChatCompletions(options.body)
|
||||
@@ -251,6 +245,101 @@ export class GithubExecutor extends BaseExecutor {
|
||||
};
|
||||
}
|
||||
|
||||
// Claude models arrive here OpenAI-shape (chatCore.js targets "openai" for github —
|
||||
// see the note in execute() above), so we translate to Anthropic-native ourselves.
|
||||
// This is what makes prepareClaudeRequest() (translator/formats/claude.js) inject
|
||||
// cache_control — /chat/completions never gets there, so it never sees cache tokens.
|
||||
async executeWithMessagesEndpoint({ model, body, stream, credentials, signal, log, proxyOptions = null }) {
|
||||
const url = this.config.messagesUrl;
|
||||
const headers = this.buildHeaders(credentials, stream);
|
||||
|
||||
// Force stream:true upstream regardless of client preference, same as
|
||||
// executeWithResponsesEndpoint below — chatCore.js's non-streaming handler already
|
||||
// knows how to buffer an SSE response into a single JSON reply when the client
|
||||
// asked for stream:false.
|
||||
const transformedBody = translateRequest(FORMATS.OPENAI, FORMATS.CLAUDE, model, body, true, credentials, "github");
|
||||
// _toolNameMap is internal bookkeeping (see openai-to-claude.js) — chatCore.js
|
||||
// normally strips it before dispatch and threads it into the response state to
|
||||
// restore original tool names; we must do the same here, or Anthropic's strict
|
||||
// schema rejects the extra field with a 400.
|
||||
const toolNameMap = transformedBody._toolNameMap;
|
||||
delete transformedBody._toolNameMap;
|
||||
|
||||
log?.debug("GITHUB", "Sending translated request to /v1/messages");
|
||||
|
||||
const response = await proxyAwareFetch(url, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify(transformedBody),
|
||||
signal
|
||||
}, proxyOptions);
|
||||
|
||||
if (!response.ok) {
|
||||
return { response, url, headers, transformedBody };
|
||||
}
|
||||
|
||||
const state = initState(FORMATS.CLAUDE);
|
||||
state.model = model;
|
||||
if (toolNameMap) state.toolNameMap = toolNameMap;
|
||||
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
|
||||
const emitAll = (controller, chunks) => {
|
||||
for (const c of chunks) {
|
||||
controller.enqueue(new TextEncoder().encode(formatSSE(c, "openai")));
|
||||
}
|
||||
};
|
||||
|
||||
const transformStream = new TransformStream({
|
||||
async transform(chunk, controller) {
|
||||
buffer += decoder.decode(chunk, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
|
||||
buffer = lines.pop() || "";
|
||||
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) continue;
|
||||
|
||||
const parsed = parseSSELine(trimmed);
|
||||
if (!parsed) continue;
|
||||
|
||||
if (parsed.done && stream === true) {
|
||||
controller.enqueue(new TextEncoder().encode(SSE_DONE));
|
||||
continue;
|
||||
}
|
||||
|
||||
emitAll(controller, translateResponse(FORMATS.CLAUDE, FORMATS.OPENAI, parsed, state));
|
||||
}
|
||||
},
|
||||
flush(controller) {
|
||||
if (buffer.trim()) {
|
||||
const parsed = parseSSELine(buffer.trim());
|
||||
if (parsed && !parsed.done) {
|
||||
emitAll(controller, translateResponse(FORMATS.CLAUDE, FORMATS.OPENAI, parsed, state));
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
if (!response.body) {
|
||||
return { response: new Response("", { status: response.status, headers: response.headers }), url, headers, transformedBody };
|
||||
}
|
||||
const convertedStream = response.body.pipeThrough(transformStream);
|
||||
|
||||
return {
|
||||
response: new Response(convertedStream, {
|
||||
status: response.status,
|
||||
statusText: response.statusText,
|
||||
headers: response.headers
|
||||
}),
|
||||
url,
|
||||
headers,
|
||||
transformedBody
|
||||
};
|
||||
}
|
||||
|
||||
async refreshCopilotToken(githubAccessToken, log, proxyOptions = null) {
|
||||
try {
|
||||
const response = await proxyAwareFetch("https://api.github.com/copilot_internal/v2/token", {
|
||||
|
||||
552
open-sse/executors/grok-cli.js
Normal file
552
open-sse/executors/grok-cli.js
Normal file
@@ -0,0 +1,552 @@
|
||||
import crypto from "node:crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import {
|
||||
refreshProviderCredentials,
|
||||
shouldRefreshCredentials,
|
||||
} from "../services/oauthCredentialManager.js";
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import {
|
||||
GROK_CLI_CLIENT_IDENTIFIER,
|
||||
GROK_CLI_VERSION,
|
||||
supportsGrokCliReasoningEffort,
|
||||
} from "../config/grokCli.js";
|
||||
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { getConsistentMachineId } from "../shared/machineId.js";
|
||||
|
||||
// Server-generated item id prefixes that /responses cannot resolve when store=false
|
||||
const SERVER_ID_PATTERN = /^(rs|fc|resp|msg)_/;
|
||||
|
||||
// Hosted tool types executed server-side by Grok CLI backend
|
||||
const HOSTED_TOOL_TYPES = new Set([
|
||||
"web_search",
|
||||
"x_search",
|
||||
"web_search_preview",
|
||||
"file_search",
|
||||
"image_generation",
|
||||
"code_interpreter",
|
||||
"mcp",
|
||||
"local_shell",
|
||||
]);
|
||||
|
||||
// Fields accepted by cli-chat-proxy Responses API (mirrors Codex allowlist + Grok extras)
|
||||
const RESPONSES_API_ALLOWLIST = new Set([
|
||||
"model",
|
||||
"input",
|
||||
"instructions",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"stream",
|
||||
"store",
|
||||
"reasoning",
|
||||
"include",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"max_output_tokens",
|
||||
"parallel_tool_calls",
|
||||
"text",
|
||||
"metadata",
|
||||
"prompt_cache_key",
|
||||
]);
|
||||
|
||||
const EFFORT_LEVELS = ["low", "medium", "high", "xhigh"];
|
||||
const GROK_CLI_TURN_STORE_MAX = 5000;
|
||||
const GROK_CLI_NATIVE_ITEM_ID = /^(?:rs|msg|fc)_[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
|
||||
const GROK_CLI_FREEFORM_TOOL_PARAMETERS = {
|
||||
type: "object",
|
||||
properties: { input: { type: "string" } },
|
||||
required: ["input"],
|
||||
};
|
||||
|
||||
// Per-session last turn index so multi-turn headers never go backwards within this process
|
||||
const sessionTurnStore = new Map();
|
||||
let requestTurnStore = new WeakMap();
|
||||
|
||||
/**
|
||||
* Count user turns in a Responses `input` array.
|
||||
* Official CLI sets x-grok-turn-idx to the 1-based conversation turn (≈ user messages).
|
||||
* HAR: first chat turn → "1".
|
||||
*/
|
||||
export function countGrokCliUserTurns(input) {
|
||||
if (!Array.isArray(input)) return 1;
|
||||
let n = 0;
|
||||
for (const item of input) {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) continue;
|
||||
const type = typeof item.type === "string" ? item.type : "";
|
||||
// Responses message items (type omitted or "message") with role user
|
||||
if (item.role === "user" && (!type || type === "message")) n += 1;
|
||||
}
|
||||
return Math.max(1, n);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve monotonic turn index for a session.
|
||||
* Prefers user-message count from the payload (full history clients), but never
|
||||
* decreases vs the last index observed for the same sessionId in this process.
|
||||
*/
|
||||
export function resolveGrokCliTurnIdx(sessionId, input, requestKey = null) {
|
||||
const fromInput = countGrokCliUserTurns(input);
|
||||
if (!sessionId) return fromInput;
|
||||
|
||||
if (requestKey && requestTurnStore.has(requestKey)) {
|
||||
return requestTurnStore.get(requestKey);
|
||||
}
|
||||
|
||||
const now = Date.now();
|
||||
const existing = sessionTurnStore.get(sessionId);
|
||||
const prev = existing && now - existing.lastUsed <= MEMORY_CONFIG.sessionTtlMs
|
||||
? existing.turn
|
||||
: 0;
|
||||
if (existing) sessionTurnStore.delete(sessionId);
|
||||
|
||||
// A new delta-style request advances the turn; retries reuse requestKey.
|
||||
const turn = prev > 0 ? Math.max(fromInput, prev + (requestKey ? 1 : 0)) : fromInput;
|
||||
while (sessionTurnStore.size >= GROK_CLI_TURN_STORE_MAX) {
|
||||
sessionTurnStore.delete(sessionTurnStore.keys().next().value);
|
||||
}
|
||||
sessionTurnStore.set(sessionId, { turn, lastUsed: now });
|
||||
if (requestKey) requestTurnStore.set(requestKey, turn);
|
||||
return turn;
|
||||
}
|
||||
|
||||
/** Test helper — clear in-memory turn counters */
|
||||
export function _resetGrokCliTurnStore() {
|
||||
sessionTurnStore.clear();
|
||||
requestTurnStore = new WeakMap();
|
||||
}
|
||||
|
||||
export function _getGrokCliTurnStoreSize() {
|
||||
return sessionTurnStore.size;
|
||||
}
|
||||
|
||||
export function normalizeGrokCliEffort(value) {
|
||||
const effort = typeof value === "string" ? value.trim().toLowerCase() : "";
|
||||
if (effort === "max") return "xhigh";
|
||||
if (EFFORT_LEVELS.includes(effort)) return effort;
|
||||
return "high";
|
||||
}
|
||||
|
||||
export { supportsGrokCliReasoningEffort } from "../config/grokCli.js";
|
||||
|
||||
export function resolveGrokCliSessionId(credentials, body) {
|
||||
// ponytail: clients without stable thread metadata share one connection session;
|
||||
// split further when their wire format exposes a durable conversation id.
|
||||
const explicitSessionBody = {
|
||||
prompt_cache_key: body?.prompt_cache_key,
|
||||
session_id: body?.session_id,
|
||||
conversation_id: body?.conversation_id,
|
||||
metadata: body?.metadata,
|
||||
};
|
||||
return resolveSessionId({
|
||||
headers: credentials?.rawHeaders,
|
||||
body: explicitSessionBody,
|
||||
connectionId: credentials?.connectionId || credentials?.id,
|
||||
workspaceId: credentials?.providerSpecificData?.workspaceId,
|
||||
scope: "grok-cli",
|
||||
});
|
||||
}
|
||||
|
||||
function stringifyGrokCliToolOutput(output) {
|
||||
if (typeof output === "string") return output;
|
||||
if (output === undefined) return "";
|
||||
return JSON.stringify(output);
|
||||
}
|
||||
|
||||
function isNativeGrokCliItemId(id) {
|
||||
return typeof id === "string" && GROK_CLI_NATIVE_ITEM_ID.test(id);
|
||||
}
|
||||
|
||||
function normalizeGrokCliInputItem(item) {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return item;
|
||||
const { internal_chat_message_metadata_passthrough: _metadata, ...clean } = item;
|
||||
|
||||
if (item.type === "reasoning") {
|
||||
if (!isNativeGrokCliItemId(item.id) || typeof item.encrypted_content !== "string") return null;
|
||||
return clean;
|
||||
}
|
||||
|
||||
if (item.type === "custom_tool_call") {
|
||||
const callId = item.call_id || item.id;
|
||||
const name = typeof item.name === "string" ? item.name.trim() : "";
|
||||
if (!callId || !name) return null;
|
||||
return {
|
||||
type: "function_call",
|
||||
call_id: callId,
|
||||
name,
|
||||
arguments: JSON.stringify({ input: stringifyGrokCliToolOutput(item.input ?? item.arguments) }),
|
||||
};
|
||||
}
|
||||
|
||||
if (item.type === "custom_tool_call_output" || item.type === "function_call_output") {
|
||||
const callId = item.call_id || item.id;
|
||||
if (!callId) return null;
|
||||
return {
|
||||
type: "function_call_output",
|
||||
call_id: callId,
|
||||
output: stringifyGrokCliToolOutput(item.output),
|
||||
};
|
||||
}
|
||||
|
||||
if (item.type === "function_call") {
|
||||
const callId = item.call_id || item.id;
|
||||
const name = typeof item.name === "string" ? item.name.trim() : "";
|
||||
if (!callId || !name) return null;
|
||||
return {
|
||||
type: "function_call",
|
||||
...(isNativeGrokCliItemId(item.id) ? { id: item.id } : {}),
|
||||
call_id: callId,
|
||||
name,
|
||||
arguments: typeof item.arguments === "string" ? item.arguments : JSON.stringify(item.arguments ?? {}),
|
||||
...(typeof item.status === "string" ? { status: item.status } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
return clean;
|
||||
}
|
||||
|
||||
export function normalizeGrokCliInput(body) {
|
||||
if (!Array.isArray(body?.input)) return body;
|
||||
const normalized = body.input.map(normalizeGrokCliInputItem).filter(Boolean);
|
||||
const callIds = new Set(
|
||||
normalized
|
||||
.filter((item) => item?.type === "function_call" && item.call_id)
|
||||
.map((item) => item.call_id)
|
||||
);
|
||||
body.input = normalized.filter(
|
||||
(item) => item?.type !== "function_call_output" || callIds.has(item.call_id)
|
||||
);
|
||||
return body;
|
||||
}
|
||||
|
||||
function stripStoredItemReferences(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (typeof item === "string" && SERVER_ID_PATTERN.test(item)) return false;
|
||||
if (item && typeof item === "object" && !Array.isArray(item)) {
|
||||
if (item.type === "item_reference") return false;
|
||||
if (
|
||||
typeof item.id === "string" &&
|
||||
SERVER_ID_PATTERN.test(item.id) &&
|
||||
!isNativeGrokCliItemId(item.id)
|
||||
) delete item.id;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Flatten Chat Completions tool shape → Responses flat format.
|
||||
* Keep hosted tools (web_search / x_search) passthrough.
|
||||
*/
|
||||
function normalizeGrokCliTools(body) {
|
||||
if (!Array.isArray(body.tools) || body.tools.length === 0) {
|
||||
delete body.tools;
|
||||
delete body.tool_choice;
|
||||
return;
|
||||
}
|
||||
const validNames = new Set();
|
||||
const hostedTypes = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const type = typeof tool.type === "string" ? tool.type : "";
|
||||
|
||||
if (type !== "function") {
|
||||
// Hosted tools: { type: "web_search" } / { type: "x_search" }
|
||||
if (HOSTED_TOOL_TYPES.has(type)) {
|
||||
hostedTypes.add(type);
|
||||
return true;
|
||||
}
|
||||
// Nested function shape without type
|
||||
if (!type && tool.function) {
|
||||
// fall through to function flatten below
|
||||
} else if (!type || typeof tool.name === "string") {
|
||||
// treat as bare function if name present
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
const isFunction =
|
||||
type === "function" || type === "" || tool.function || typeof tool.name === "string";
|
||||
if (!isFunction || HOSTED_TOOL_TYPES.has(type)) {
|
||||
return HOSTED_TOOL_TYPES.has(type);
|
||||
}
|
||||
|
||||
const fn =
|
||||
tool.function && typeof tool.function === "object" && !Array.isArray(tool.function)
|
||||
? tool.function
|
||||
: null;
|
||||
const rawName =
|
||||
typeof tool.name === "string" ? tool.name : typeof fn?.name === "string" ? fn.name : "";
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
|
||||
const description =
|
||||
typeof tool.description === "string"
|
||||
? tool.description
|
||||
: typeof fn?.description === "string"
|
||||
? fn.description
|
||||
: "";
|
||||
const parameters = type === "custom"
|
||||
? GROK_CLI_FREEFORM_TOOL_PARAMETERS
|
||||
: tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)
|
||||
? tool.parameters
|
||||
: fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters)
|
||||
? fn.parameters
|
||||
: { type: "object", properties: {} };
|
||||
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, 128);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
|
||||
if (body.tools.length === 0) {
|
||||
delete body.tools;
|
||||
delete body.tool_choice;
|
||||
return;
|
||||
}
|
||||
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
const choiceType = typeof body.tool_choice.type === "string" ? body.tool_choice.type : "";
|
||||
if (choiceType === "function" || choiceType === "custom") {
|
||||
const rawName = body.tool_choice.name ?? body.tool_choice.function?.name;
|
||||
const name = typeof rawName === "string" ? rawName.trim().slice(0, 128) : "";
|
||||
if (!name || !validNames.has(name)) delete body.tool_choice;
|
||||
else body.tool_choice = { type: "function", name };
|
||||
} else if (!hostedTypes.has(choiceType)) {
|
||||
delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function resolveEffortFromModel(modelId) {
|
||||
if (!modelId || typeof modelId !== "string") return null;
|
||||
for (const level of EFFORT_LEVELS) {
|
||||
if (modelId.endsWith(`-${level}`)) return level;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Grok CLI Executor — OpenAI Responses API on cli-chat-proxy.grok.com
|
||||
* Auth: OAuth device-code access token (xai-grok-cli).
|
||||
*/
|
||||
export class GrokCliExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("grok-cli", PROVIDERS["grok-cli"]);
|
||||
this._currentSessionId = null;
|
||||
this._currentReqId = null;
|
||||
this._currentTurnIdx = 1;
|
||||
this._agentId = null;
|
||||
}
|
||||
|
||||
buildUrl() {
|
||||
return this.config.baseUrl;
|
||||
}
|
||||
|
||||
async refreshCredentials(credentials, log, proxyOptions = null) {
|
||||
if (!credentials?.refreshToken) return null;
|
||||
return refreshProviderCredentials("grok-cli", credentials, log, proxyOptions);
|
||||
}
|
||||
|
||||
needsRefresh(credentials) {
|
||||
return shouldRefreshCredentials("grok-cli", credentials);
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
const headers = super.buildHeaders(credentials, stream);
|
||||
|
||||
// Static fingerprint from registry
|
||||
const staticHeaders = this.config.headers || {};
|
||||
for (const [k, v] of Object.entries(staticHeaders)) {
|
||||
if (v != null && headers[k] === undefined) headers[k] = v;
|
||||
}
|
||||
|
||||
headers["x-grok-client-identifier"] =
|
||||
this.config.clientIdentifier || headers["x-grok-client-identifier"] || GROK_CLI_CLIENT_IDENTIFIER;
|
||||
headers["x-grok-client-version"] =
|
||||
this.config.clientVersion || headers["x-grok-client-version"] || GROK_CLI_VERSION;
|
||||
|
||||
const sessionId = this._currentSessionId || credentials?.connectionId || crypto.randomUUID();
|
||||
const reqId = this._currentReqId || crypto.randomUUID();
|
||||
headers["x-grok-session-id"] = sessionId;
|
||||
// CLI uses the same id for conv + session on chat turns
|
||||
headers["x-grok-conv-id"] = sessionId;
|
||||
headers["x-grok-req-id"] = reqId;
|
||||
headers["x-grok-turn-idx"] = String(this._currentTurnIdx || 1);
|
||||
|
||||
if (this._agentId) headers["x-grok-agent-id"] = this._agentId;
|
||||
|
||||
// Surface model override (CLI always sets this)
|
||||
if (this._currentModel) headers["x-grok-model-override"] = this._currentModel;
|
||||
|
||||
// Identity: mapTokens stores email top-level AND in providerSpecificData;
|
||||
// fall back either way so OAuth connections always fingerprint like the CLI.
|
||||
const psd = credentials?.providerSpecificData || {};
|
||||
const email = psd.email || credentials?.email;
|
||||
const userId = psd.userId || credentials?.userId || credentials?.providerUserId;
|
||||
if (email) headers["x-email"] = email;
|
||||
if (userId) headers["x-userid"] = userId;
|
||||
|
||||
return headers;
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
// 402 personal-team-blocked:spending-limit → surface as payment/quota for fallback
|
||||
if (response.status === 402 && bodyText) {
|
||||
try {
|
||||
const json = JSON.parse(bodyText);
|
||||
const code = json?.code || "";
|
||||
const msg = json?.error || json?.message || bodyText;
|
||||
return {
|
||||
status: 402,
|
||||
message: typeof msg === "string" ? msg : bodyText,
|
||||
code: typeof code === "string" ? code : undefined,
|
||||
};
|
||||
} catch {
|
||||
/* fall through */
|
||||
}
|
||||
}
|
||||
return super.parseError(response, bodyText);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
// Session / request ids for headers — stable per client conversation when possible
|
||||
const requestKey = body;
|
||||
this._currentSessionId = resolveGrokCliSessionId(credentials, body);
|
||||
this._currentReqId = crypto.randomUUID();
|
||||
this._agentId =
|
||||
credentials?.providerSpecificData?.deviceId ||
|
||||
credentials?.providerSpecificData?.agentId ||
|
||||
null;
|
||||
|
||||
// Normalize Responses input
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
|
||||
// Chat Completions clients arrive with messages[] — translator should have
|
||||
// converted already, but guard empty input.
|
||||
if (!body.input || (Array.isArray(body.input) && body.input.length === 0)) {
|
||||
if (Array.isArray(body.messages) && body.messages.length > 0) {
|
||||
// Soft fallback: map messages → input messages (string content only)
|
||||
body.input = body.messages.map((m) => ({
|
||||
type: "message",
|
||||
role: m.role || "user",
|
||||
content: typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? ""),
|
||||
}));
|
||||
delete body.messages;
|
||||
} else {
|
||||
body.input = [{ type: "message", role: "user", content: "..." }];
|
||||
}
|
||||
}
|
||||
|
||||
// Keep role:"system" as-is — official grok-pager HAR sends system, not developer
|
||||
// (Codex converts system→developer; Grok CLI does not).
|
||||
normalizeGrokCliInput(body);
|
||||
stripStoredItemReferences(body);
|
||||
normalizeGrokCliTools(body);
|
||||
|
||||
// Turn index after input is finalized (user-message count, monotonic per session)
|
||||
this._currentTurnIdx = resolveGrokCliTurnIdx(this._currentSessionId, body.input, requestKey);
|
||||
|
||||
body.stream = true;
|
||||
body.store = false;
|
||||
|
||||
// Resolve upstream model id (strip effort suffix virtual models)
|
||||
let modelEffort = resolveEffortFromModel(body.model || model);
|
||||
let resolvedModel = body.model || model;
|
||||
if (modelEffort) {
|
||||
resolvedModel = resolvedModel.replace(new RegExp(`-${modelEffort}$`), "");
|
||||
}
|
||||
resolvedModel = getModelUpstreamId("gcli", resolvedModel) || resolvedModel;
|
||||
// Also try provider id key
|
||||
if (resolvedModel === (body.model || model)) {
|
||||
resolvedModel = getModelUpstreamId("grok-cli", resolvedModel) || resolvedModel;
|
||||
}
|
||||
body.model = resolvedModel;
|
||||
this._currentModel = resolvedModel;
|
||||
|
||||
// Reasoning effort priority: explicit > reasoning_effort > model suffix > default high.
|
||||
// grok-build and Composer reject reasoningEffort but still accept summary/encrypted continuity.
|
||||
const supportsReasoningEffort = supportsGrokCliReasoningEffort(resolvedModel);
|
||||
if (!body.reasoning || typeof body.reasoning !== "object") {
|
||||
body.reasoning = { summary: "concise" };
|
||||
if (supportsReasoningEffort) {
|
||||
body.reasoning.effort = normalizeGrokCliEffort(body.reasoning_effort || modelEffort);
|
||||
}
|
||||
} else {
|
||||
if (supportsReasoningEffort) {
|
||||
body.reasoning.effort = normalizeGrokCliEffort(
|
||||
body.reasoning.effort || body.reasoning_effort || modelEffort,
|
||||
);
|
||||
} else {
|
||||
delete body.reasoning.effort;
|
||||
}
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "concise";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
// Encrypted reasoning for multi-turn continuity (CLI always requests this)
|
||||
if (body.reasoning && body.reasoning.effort !== "none") {
|
||||
const include = Array.isArray(body.include) ? body.include : [];
|
||||
if (!include.includes("reasoning.encrypted_content")) {
|
||||
include.push("reasoning.encrypted_content");
|
||||
}
|
||||
body.include = include;
|
||||
}
|
||||
|
||||
// Drop Chat Completions leftovers that Responses rejects
|
||||
delete body.messages;
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
delete body.n;
|
||||
delete body.seed;
|
||||
delete body.logprobs;
|
||||
delete body.top_logprobs;
|
||||
delete body.frequency_penalty;
|
||||
delete body.presence_penalty;
|
||||
delete body.logit_bias;
|
||||
delete body.user;
|
||||
delete body.stream_options;
|
||||
delete body.prompt_cache_retention;
|
||||
delete body.safety_identifier;
|
||||
delete body.previous_response_id; // store=false → cannot resolve
|
||||
|
||||
for (const k of Object.keys(body)) {
|
||||
if (!RESPONSES_API_ALLOWLIST.has(k)) delete body[k];
|
||||
}
|
||||
|
||||
return body;
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
// Lazy-resolve stable agent id once per process if connection has none
|
||||
if (!this._agentId && !args.credentials?.providerSpecificData?.deviceId) {
|
||||
try {
|
||||
const mid = await getConsistentMachineId("grok-cli-agent");
|
||||
// Format as UUID-ish for header aesthetics
|
||||
this._agentId = [
|
||||
mid.slice(0, 8),
|
||||
mid.slice(8, 12),
|
||||
"5" + mid.slice(13, 16),
|
||||
"a" + mid.slice(17, 20),
|
||||
mid.slice(0, 12).padEnd(12, "0"),
|
||||
].join("-");
|
||||
} catch {
|
||||
this._agentId = crypto.randomUUID();
|
||||
}
|
||||
} else if (args.credentials?.providerSpecificData?.deviceId) {
|
||||
this._agentId = args.credentials.providerSpecificData.deviceId;
|
||||
}
|
||||
|
||||
return super.execute(args);
|
||||
}
|
||||
}
|
||||
|
||||
export default GrokCliExecutor;
|
||||
@@ -13,6 +13,7 @@ import { QwenExecutor } from "./qwen.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
import { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
import { CommandCodeExecutor } from "./commandcode.js";
|
||||
@@ -39,6 +40,9 @@ const executors = {
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
gb: new GrokCliExecutor(), // Alias (Grok Build)
|
||||
"perplexity-web": new PerplexityWebExecutor(),
|
||||
"ollama-local": new OllamaLocalExecutor(),
|
||||
commandcode: new CommandCodeExecutor(),
|
||||
@@ -77,6 +81,7 @@ export { QwenExecutor } from "./qwen.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
export { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
export { CommandCodeExecutor } from "./commandcode.js";
|
||||
|
||||
Reference in New Issue
Block a user