Enhance provider models and chat handling with new thinking configurations
This commit is contained in:
@@ -221,7 +221,16 @@ export const PROVIDER_MODELS = {
|
||||
{ id: "text-embedding-005", name: "Text Embedding 005", type: "embedding" },
|
||||
{ id: "text-embedding-004", name: "Text Embedding 004 (Legacy)", type: "embedding" },
|
||||
],
|
||||
openrouter: [],
|
||||
openrouter: [
|
||||
// Embedding models
|
||||
{ id: "openai/text-embedding-3-large", name: "OpenAI Text Embedding 3 Large", type: "embedding" },
|
||||
{ id: "openai/text-embedding-3-small", name: "OpenAI Text Embedding 3 Small", type: "embedding" },
|
||||
{ id: "openai/text-embedding-ada-002", name: "OpenAI Text Embedding Ada 002", type: "embedding" },
|
||||
{ id: "qwen/qwen3-embedding-8b", name: "Qwen3 Embedding 8B", type: "embedding" },
|
||||
{ id: "perplexity/pplx-embed-v1-4b", name: "Perplexity Embed V1 4B", type: "embedding" },
|
||||
{ id: "perplexity/pplx-embed-v1-0.6b", name: "Perplexity Embed V1 0.6B", type: "embedding" },
|
||||
{ id: "nvidia/llama-nemotron-embed-vl-1b-v2:free", name: "NVIDIA Nemotron Embed VL 1B V2 (Free)", type: "embedding" },
|
||||
],
|
||||
glm: [
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
|
||||
@@ -24,7 +24,7 @@ import { detectClientTool, isNativePassthrough } from "../utils/clientDetector.j
|
||||
* @param {object} options.credentials - Provider credentials
|
||||
* @param {string} options.sourceFormatOverride - Override detected source format (e.g. "openai-responses")
|
||||
*/
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, sourceFormatOverride }) {
|
||||
export async function handleChatCore({ body, modelInfo, credentials, log, onCredentialsRefreshed, onRequestSuccess, onDisconnect, clientRawRequest, connectionId, userAgent, apiKey, ccFilterNaming, sourceFormatOverride, providerThinking }) {
|
||||
const { provider, model } = modelInfo;
|
||||
const requestStartTime = Date.now();
|
||||
|
||||
@@ -39,6 +39,20 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
const targetFormat = modelTargetFormat || getTargetFormat(provider);
|
||||
const stripList = getModelStrip(alias, model);
|
||||
|
||||
// Inject provider-level thinking config override (only if client hasn't set)
|
||||
// on/off → extended type (body.thinking), none/low/medium/high → effort type (body.reasoning_effort)
|
||||
if (providerThinking?.mode && providerThinking.mode !== "auto") {
|
||||
const mode = providerThinking.mode;
|
||||
if (mode === "on" && !body.thinking) {
|
||||
console.log("Injecting provider-level thinking config override: on");
|
||||
body = { ...body, thinking: { type: "enabled", budget_tokens: 10000 } };
|
||||
} else if (mode === "off" && !body.thinking) {
|
||||
body = { ...body, thinking: { type: "disabled" } };
|
||||
} else if (!body.reasoning_effort) {
|
||||
body = { ...body, reasoning_effort: mode };
|
||||
}
|
||||
}
|
||||
|
||||
const clientRequestedStreaming = body.stream === true || sourceFormat === FORMATS.ANTIGRAVITY || sourceFormat === FORMATS.GEMINI || sourceFormat === FORMATS.GEMINI_CLI;
|
||||
const providerRequiresStreaming = provider === "openai" || provider === "codex";
|
||||
let stream = providerRequiresStreaming ? true : (body.stream !== false);
|
||||
|
||||
Reference in New Issue
Block a user