feat(codex): support GPT-5.6 Max and Ultra overrides
Add "ultra" reasoning level for Codex GPT-5.6 Sol and Terra, and expose Max for Luna (Luna falls back Ultra to Max since it is not supported upstream). Scoped to cx/ routes only; Kiro and generic OpenAI routing unchanged.
This commit is contained in:
@@ -8,6 +8,7 @@ import {
|
||||
import { normalizeResponsesInput } from "../translator/formats/responsesApi.js";
|
||||
import { fetchImageAsBase64 } from "../translator/concerns/image.js";
|
||||
import { getModelUpstreamId } from "../config/providerModels.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
@@ -124,8 +125,12 @@ function resolveCacheSessionId(body, credentials) {
|
||||
});
|
||||
}
|
||||
|
||||
function normalizeReasoningEffort(value) {
|
||||
return value === "max" ? "xhigh" : value;
|
||||
function normalizeReasoningEffort(model, value) {
|
||||
const supportedLevels = getThinkingLevels("codex", model);
|
||||
if (supportedLevels?.includes(value)) return value;
|
||||
if (value === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
if (value === "max" || value === "ultra") return "xhigh";
|
||||
return value;
|
||||
}
|
||||
|
||||
function findNestedMessage(value, depth = 0) {
|
||||
@@ -440,10 +445,10 @@ export class CodexExecutor extends BaseExecutor {
|
||||
|
||||
// Priority: explicit reasoning.effort > reasoning_effort param > model suffix > default (medium)
|
||||
if (!body.reasoning) {
|
||||
const effort = normalizeReasoningEffort(body.reasoning_effort || modelEffort || 'low');
|
||||
const effort = normalizeReasoningEffort(body.model, body.reasoning_effort || modelEffort || 'low');
|
||||
body.reasoning = { effort, summary: "auto" };
|
||||
} else {
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.reasoning.effort);
|
||||
body.reasoning.effort = normalizeReasoningEffort(body.model, body.reasoning.effort);
|
||||
if (!body.reasoning.summary) body.reasoning.summary = "auto";
|
||||
}
|
||||
delete body.reasoning_effort;
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { detectFormat, getTargetFormat, resolveTransport } from "../services/provider.js";
|
||||
import { translateRequest } from "../translator/index.js";
|
||||
import { stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { applyThinking, extractThinking, stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js";
|
||||
import { FORMATS } from "../translator/formats.js";
|
||||
import { normalizeClaudePassthrough } from "../translator/formats/claude.js";
|
||||
import { createStreamController } from "../utils/streamHandler.js";
|
||||
@@ -28,7 +28,6 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
|
||||
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
import { extractThinking } from "../translator/concerns/thinkingUnified.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
|
||||
/**
|
||||
@@ -137,6 +136,18 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
if (passthrough) {
|
||||
log?.debug?.("PASSTHROUGH", `${clientTool} → ${provider} | native lossless`);
|
||||
translatedBody = { ...body, model: stripThinkingSuffix(upstreamModel) };
|
||||
if (provider === "codex") {
|
||||
const suffixThinking = {};
|
||||
applyThinking(sourceFormat, upstreamModel, suffixThinking, provider);
|
||||
if (suffixThinking.reasoning_effort) {
|
||||
const reasoning = translatedBody.reasoning;
|
||||
translatedBody.reasoning = {
|
||||
...(reasoning && typeof reasoning === "object" && !Array.isArray(reasoning) ? reasoning : {}),
|
||||
effort: suffixThinking.reasoning_effort,
|
||||
};
|
||||
delete translatedBody.reasoning_effort;
|
||||
}
|
||||
}
|
||||
// Normalize newer Cowork/CC beta shapes (adaptive thinking, mid-conversation system) the API rejects
|
||||
if (clientTool === "claude") normalizeClaudePassthrough(translatedBody, translatedBody.model);
|
||||
} else {
|
||||
|
||||
@@ -31,10 +31,13 @@ const FORMAT_LEVELS = {
|
||||
step: L.base,
|
||||
};
|
||||
|
||||
const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
||||
|
||||
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
|
||||
const PATTERN_THINKING = [
|
||||
// gpt-5.6-sol accepts max (maps to xhigh on wire); live probe rejected ultra.
|
||||
{ pattern: "*gpt-5.6-sol*", levels: ["none", "minimal", "low", "medium", "high", "xhigh", "max"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
];
|
||||
|
||||
@@ -43,7 +46,9 @@ export function getThinkingLevels(provider, model) {
|
||||
if (provider === "kiro" && resolveKiroEffortPath(model) === null) return null;
|
||||
const caps = getCapabilitiesForModel(provider, model);
|
||||
if (!caps.reasoning) return null;
|
||||
const hit = PATTERN_THINKING.find((p) => matchPattern(p.pattern, model));
|
||||
const hit = PATTERN_THINKING.find((entry) =>
|
||||
(!entry.provider || entry.provider === provider) && matchPattern(entry.pattern, model)
|
||||
);
|
||||
let levels = hit?.levels || FORMAT_LEVELS[caps.thinkingFormat] || L.base;
|
||||
if (caps.thinkingCanDisable === false) levels = levels.filter((l) => l !== "none");
|
||||
return levels;
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
// never hardcoded per-model here. See .docs/thinking/plan.md MATRIX VI-A.
|
||||
|
||||
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
||||
import { getThinkingLevels } from "../../providers/thinkingLevels.js";
|
||||
import { PROVIDERS } from "../../providers/index.js";
|
||||
import { LEVEL_TO_BUDGET, budgetToLevel, effortToBudget, effortToThinkingLevel } from "./thinking.js";
|
||||
|
||||
@@ -37,6 +38,7 @@ export function parseSuffix(model) {
|
||||
const raw = m[2].trim().toLowerCase();
|
||||
if (raw === "none" || raw === "off") return { cleanModel, override: { mode: "none" } };
|
||||
if (raw === "auto") return { cleanModel, override: { mode: "auto" } };
|
||||
if (raw === "ultra") return { cleanModel, override: { mode: "level", level: raw } };
|
||||
if (/^\d+$/.test(raw)) return { cleanModel, override: { mode: "budget", budget: Number(raw) } };
|
||||
if (LEVEL_TO_BUDGET[raw] !== undefined) return { cleanModel, override: { mode: "level", level: raw } };
|
||||
return { cleanModel, override: null };
|
||||
@@ -134,6 +136,13 @@ function toLevel(cfg) {
|
||||
return null;
|
||||
}
|
||||
|
||||
function normalizeOpenAILevel(level, supportedLevels) {
|
||||
if (level !== "max" && level !== "ultra") return level;
|
||||
if (supportedLevels?.includes(level)) return level;
|
||||
if (level === "ultra" && supportedLevels?.includes("max")) return "max";
|
||||
return "xhigh";
|
||||
}
|
||||
|
||||
function toGeminiThinkingLevel(cfg) {
|
||||
const raw = cfg.mode === "auto" ? "high" : (toLevel(cfg) || "high");
|
||||
return effortToThinkingLevel(raw);
|
||||
@@ -213,7 +222,7 @@ function stripAll(body) {
|
||||
}
|
||||
|
||||
// Apply unified thinking config to body in the resolved provider-native format.
|
||||
function applyFormat(fmt, body, cfg, caps) {
|
||||
function applyFormat(fmt, body, cfg, caps, supportedLevels) {
|
||||
const none = cfg.mode === "none";
|
||||
const canDisable = caps.thinkingCanDisable !== false;
|
||||
// Model cannot disable thinking → clamp "none" to minimal effort instead.
|
||||
@@ -223,8 +232,7 @@ function applyFormat(fmt, body, cfg, caps) {
|
||||
case "openai": {
|
||||
if (none && canDisable) { body.reasoning_effort = "none"; break; }
|
||||
const level = toLevel(eff);
|
||||
// OpenAI reasoning_effort enum caps at "xhigh" (no "max"); clamp Claude Code's "max".
|
||||
if (level) body.reasoning_effort = level === "max" ? "xhigh" : level;
|
||||
if (level) body.reasoning_effort = normalizeOpenAILevel(level, supportedLevels);
|
||||
break;
|
||||
}
|
||||
case "claude-adaptive": {
|
||||
@@ -329,7 +337,8 @@ export function applyThinking(targetFormat, model, body, provider = null, intent
|
||||
if (!cfg) return body;
|
||||
|
||||
const fmt = resolveFormat(targetFormat, cleanModel, provider);
|
||||
const supportedLevels = getThinkingLevels(provider, cleanModel);
|
||||
stripAll(body);
|
||||
applyFormat(fmt, body, cfg, caps);
|
||||
applyFormat(fmt, body, cfg, caps, supportedLevels);
|
||||
return body;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user