- Cap maximum cooldown for rate limit handling in account unavailability and single-model chat flows
- Dynamic custom model fetching for model selection
This commit is contained in:
@@ -38,6 +38,9 @@ export const BACKOFF_CONFIG = {
|
||||
// Default cooldown for transient/unknown errors
|
||||
export const TRANSIENT_COOLDOWN_MS = 30 * 1000;
|
||||
|
||||
// Hard cap for provider-reported rate limit cooldown (e.g. codex resets_at can be 5-6h)
|
||||
export const MAX_RATE_LIMIT_COOLDOWN_MS = 30 * 60 * 1000;
|
||||
|
||||
// Cooldown durations (ms)
|
||||
const COOLDOWN = {
|
||||
long: 2 * 60 * 1000,
|
||||
|
||||
@@ -219,7 +219,6 @@ export const PROVIDER_MODELS = {
|
||||
// Gemini 3.1 series
|
||||
{ id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
|
||||
{ id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" },
|
||||
{ id: "gemini-3.1-flash-image-preview", name: "Gemini 3.1 Flash Image Preview" },
|
||||
// Gemini 3 series
|
||||
{ id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
|
||||
// Gemini 2.5 series
|
||||
@@ -342,6 +341,7 @@ export const PROVIDER_MODELS = {
|
||||
{ id: "DeepSeek-V3.2", name: "DeepSeek-V3.2" },
|
||||
],
|
||||
deepseek: [
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
|
||||
{ id: "deepseek-reasoner", name: "DeepSeek V3.2 Reasoner" },
|
||||
],
|
||||
|
||||
@@ -24,7 +24,7 @@ import { detectClientTool, isNativePassthrough } from "../utils/clientDetector.j
|
||||
* @param {object} options.credentials - Provider credentials
|
||||
* @param {string} options.sourceFormatOverride - Override detected source format (e.g. "openai-responses")
|
||||
*/
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, sourceFormatOverride, providerThinking }) {
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, rtkEnabled, sourceFormatOverride, providerThinking }) {
|
||||
const { provider, model } = modelInfo;
|
||||
const requestStartTime = Date.now();
|
||||
|
||||
@@ -82,7 +82,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
log?.debug?.("PASSTHROUGH", `${clientTool} → ${provider} | native lossless`);
|
||||
translatedBody = { ...body, model };
|
||||
} else {
|
||||
translatedBody = translateRequest(sourceFormat, targetFormat, model, body, stream, credentials, provider, reqLogger, stripList, connectionId);
|
||||
translatedBody = translateRequest(sourceFormat, targetFormat, model, body, stream, credentials, provider, reqLogger, stripList, connectionId, rtkEnabled);
|
||||
if (!translatedBody) {
|
||||
trackPendingRequest(model, provider, connectionId, false, true);
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Failed to translate request for ${sourceFormat} → ${targetFormat}`);
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
// Synchronous RTK toggle cache. Updated by /api/settings PATCH handler
|
||||
// and initialized from DB on server boot.
|
||||
let enabled = false;
|
||||
|
||||
export function setRtkEnabled(value) {
|
||||
enabled = Boolean(value);
|
||||
}
|
||||
|
||||
export function isRtkEnabled() {
|
||||
return enabled;
|
||||
}
|
||||
@@ -3,21 +3,38 @@
|
||||
import { RAW_CAP, MIN_COMPRESS_SIZE } from "./constants.js";
|
||||
import { autoDetectFilter } from "./autodetect.js";
|
||||
import { safeApply } from "./applyFilter.js";
|
||||
import { isRtkEnabled } from "./flag.js";
|
||||
|
||||
export { isRtkEnabled, setRtkEnabled } from "./flag.js";
|
||||
|
||||
// Compress tool_result content in-place. Returns stats or null if disabled/failed.
|
||||
export function compressMessages(body) {
|
||||
if (!isRtkEnabled()) return null;
|
||||
if (!body || !Array.isArray(body.messages)) return null;
|
||||
export function compressMessages(body, enabled) {
|
||||
if (!enabled) return null;
|
||||
if (!body) return null;
|
||||
// Support both OpenAI/Claude "messages" and OpenAI Responses "input"
|
||||
const items = Array.isArray(body.messages) ? body.messages
|
||||
: Array.isArray(body.input) ? body.input
|
||||
: null;
|
||||
if (!items) return null;
|
||||
|
||||
const stats = { bytesBefore: 0, bytesAfter: 0, hits: [] };
|
||||
try {
|
||||
for (let i = 0; i < body.messages.length; i++) {
|
||||
const msg = body.messages[i];
|
||||
for (let i = 0; i < items.length; i++) {
|
||||
const msg = items[i];
|
||||
if (!msg) continue;
|
||||
|
||||
// Shape 4: OpenAI Responses — top-level { type:"function_call_output", output: string | [{type:"input_text", text}] }
|
||||
if (msg.type === "function_call_output") {
|
||||
if (typeof msg.output === "string") {
|
||||
msg.output = compressText(msg.output, stats, "openai-responses-string");
|
||||
} else if (Array.isArray(msg.output)) {
|
||||
for (let k = 0; k < msg.output.length; k++) {
|
||||
const part = msg.output[k];
|
||||
if (part && part.type === "input_text" && typeof part.text === "string") {
|
||||
part.text = compressText(part.text, stats, "openai-responses-array");
|
||||
}
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Shape 1: OpenAI tool message — { role:"tool", content: "string" }
|
||||
if (msg.role === "tool" && typeof msg.content === "string") {
|
||||
msg.content = compressText(msg.content, stats, "openai-tool");
|
||||
|
||||
@@ -39,6 +39,15 @@ export function getRotatedModels(models, comboName, strategy) {
|
||||
return rotatedModels;
|
||||
}
|
||||
|
||||
/**
|
||||
* Reset in-memory rotation state when combo/settings change
|
||||
* @param {string} [comboName] - Combo name to reset; omit to clear all
|
||||
*/
|
||||
export function resetComboRotation(comboName) {
|
||||
if (comboName) comboRotationState.delete(comboName);
|
||||
else comboRotationState.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Get combo models from combos data
|
||||
* @param {string} modelStr - Model string to check
|
||||
|
||||
@@ -71,12 +71,12 @@ function stripContentTypes(body, stripList = []) {
|
||||
}
|
||||
|
||||
// Translate request: source -> openai -> target
|
||||
export function translateRequest(sourceFormat, targetFormat, model, body, stream = true, credentials = null, provider = null, reqLogger = null, stripList = [], connectionId = null) {
|
||||
export function translateRequest(sourceFormat, targetFormat, model, body, stream = true, credentials = null, provider = null, reqLogger = null, stripList = [], connectionId = null, rtkEnabled = false) {
|
||||
ensureInitialized();
|
||||
let result = body;
|
||||
|
||||
// RTK: compress tool_result content before any translation (shape-agnostic)
|
||||
const rtkStats = compressMessages(result);
|
||||
const rtkStats = compressMessages(result, rtkEnabled);
|
||||
if (rtkStats) {
|
||||
const line = formatRtkLog(rtkStats);
|
||||
if (line) console.log(line);
|
||||
|
||||
Reference in New Issue
Block a user