Merge remote-tracking branch 'origin/master' into gitea/feature/end
Resolved conflicts taking origin/master (v0.5.55) as canonical, with local features re-applied: - runtime log level (LOG_LEVEL env + dashboard Settings → Logging, applied immediately and persisted across restarts) - free/noAuth provider enable/disable toggle via providerStrategies.enabled - parallel model testing (Test All Models / Test Selected Keys)
This commit is contained in:
@@ -51,6 +51,25 @@ async function huggingface({ baseUrl, apiKey, text, modelId }) {
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Fish Audio: model travels in an HTTP header, the voice is a reference_id, returns binary
|
||||
async function fishAudio({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
"model": modelId || "s2.1-pro-free",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
format: "mp3",
|
||||
...(voiceId ? { reference_id: voiceId } : {}),
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Inworld: Basic auth, JSON { audioContent }
|
||||
async function inworld({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
@@ -166,4 +185,5 @@ export const FORMAT_HANDLERS = {
|
||||
tortoise,
|
||||
openai: openaiCompat,
|
||||
"minimax-tts": minimaxTts,
|
||||
"fish-audio": fishAudio,
|
||||
};
|
||||
|
||||
@@ -6,6 +6,8 @@ import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
||||
import openai from "./openai.js";
|
||||
import openrouter from "./openrouter.js";
|
||||
import gemini, { fetchGeminiVoices } from "./gemini.js";
|
||||
import xiaomiMimo from "./xiaomi-mimo.js";
|
||||
import selfhostedTts from "./selfhostedTts.js";
|
||||
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
@@ -18,6 +20,8 @@ const SPECIAL_ADAPTERS = {
|
||||
openai,
|
||||
openrouter,
|
||||
gemini,
|
||||
"xiaomi-mimo": xiaomiMimo,
|
||||
"selfhosted-tts": selfhostedTts,
|
||||
};
|
||||
|
||||
export function getTtsAdapter(provider) {
|
||||
|
||||
69
open-sse/handlers/ttsProviders/selfhostedTts.js
Normal file
69
open-sse/handlers/ttsProviders/selfhostedTts.js
Normal file
@@ -0,0 +1,69 @@
|
||||
// Self-hosted OpenAI-compatible TTS — POST {baseUrl}/v1/audio/speech.
|
||||
//
|
||||
// A SPECIAL_ADAPTER rather than a genericFormats handler on purpose: the generic
|
||||
// dispatcher resolves baseUrl from the static registry entry
|
||||
// (`synthesizeViaConfig` reads `cfg.baseUrl`) and never looks at the connection,
|
||||
// which is exactly the limitation this provider exists to lift.
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
const DEFAULT_BASE_URL = "http://localhost:8880";
|
||||
const DEFAULT_MODEL = "kokoro";
|
||||
const DEFAULT_VOICE = "af_heart";
|
||||
|
||||
export default {
|
||||
async synthesize(text, model, credentials, responseFormat = "mp3") {
|
||||
// Accept either providerSpecificData.baseUrl (how the custom embedding and
|
||||
// STT providers carry it) or a bare credentials.baseUrl (how the OpenAI TTS
|
||||
// adapter does), so a connection configured either way works.
|
||||
const raw = credentials?.providerSpecificData?.baseUrl || credentials?.baseUrl || DEFAULT_BASE_URL;
|
||||
// Tolerate a baseUrl given as the full endpoint or with a trailing /v1 —
|
||||
// both are natural things to paste, and silently double-appending the path
|
||||
// would 404 with nothing pointing at the cause.
|
||||
const base = String(raw)
|
||||
.replace(/\/+$/, "")
|
||||
.replace(/\/v1\/audio\/speech$/, "")
|
||||
.replace(/\/v1$/, "");
|
||||
|
||||
// The provider prefix is already stripped by getModelInfo, so `model` here is
|
||||
// "kokoro" or "kokoro/af_heart" — NOT "selfhosted-tts/...".
|
||||
//
|
||||
// A bare value is the MODEL, not the voice. The OpenAI adapter reads a bare
|
||||
// value as a voice, which is right for a service whose model is fixed
|
||||
// ("tts-1") and whose voice varies — but wrong here, where the model is the
|
||||
// variable part. Treating it as a voice sent voice="kokoro" upstream and
|
||||
// Kokoro answered 400, so `selfhosted-tts/kokoro` — the obvious way to
|
||||
// address this provider — was the one form that did not work (verified
|
||||
// against a live Kokoro through 9router, 2026-08-03).
|
||||
let ttsModel = DEFAULT_MODEL;
|
||||
let voice = DEFAULT_VOICE;
|
||||
if (model) {
|
||||
const parts = String(model).split("/").filter(Boolean);
|
||||
if (parts.length >= 2) {
|
||||
ttsModel = parts[0];
|
||||
voice = parts.slice(1).join("/");
|
||||
} else if (parts.length === 1) {
|
||||
ttsModel = parts[0];
|
||||
}
|
||||
}
|
||||
|
||||
const res = await fetch(`${base}/v1/audio/speech`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(credentials?.apiKey ? { Authorization: `Bearer ${credentials.apiKey}` } : {}),
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: ttsModel,
|
||||
voice,
|
||||
input: text,
|
||||
response_format: responseFormat,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.error?.message || `Self-hosted TTS failed: ${res.status}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: responseFormat };
|
||||
},
|
||||
};
|
||||
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
@@ -0,0 +1,65 @@
|
||||
// Xiaomi MiMo TTS — via OpenAI-compatible chat completions (non-streaming).
|
||||
// Docs: https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5
|
||||
// Message contract: target text in `role: assistant` content, style/voice
|
||||
// instructions in `role: user` content. Voice is selected via the top-level
|
||||
// `audio.voice` field (NOT embedded in the model name).
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
const DEFAULT_MODEL = "mimo-v2.5-tts";
|
||||
const DEFAULT_VOICE = "mimo_default";
|
||||
|
||||
export default {
|
||||
synthesize(text, model, credentials, responseFormat, { style, language } = {}) {
|
||||
if (!credentials?.apiKey) throw new Error("xiaomi-mimo API key required");
|
||||
return synthesizeMiMo(text, model, credentials.apiKey, style, language);
|
||||
},
|
||||
};
|
||||
|
||||
export async function synthesizeMiMo(text, model, apiKey, style, language) {
|
||||
const { modelId, voiceId } = parseModelVoice(model, DEFAULT_MODEL, DEFAULT_VOICE, [DEFAULT_MODEL]);
|
||||
|
||||
// Language and style are soft instructions → prepend as a role:user message.
|
||||
// MiMo auto-detects the spoken language of the text; the hint only nudges it
|
||||
// (e.g. "Speak in English.") and is independent of the chosen voice.
|
||||
const instructions = [];
|
||||
if (language) instructions.push(`Speak in ${language}.`);
|
||||
if (style) instructions.push(style);
|
||||
|
||||
const messages = [{ role: "assistant", content: text }];
|
||||
if (instructions.length) messages.unshift({ role: "user", content: instructions.join(" ") });
|
||||
|
||||
const res = await fetch("https://api.xiaomimimo.com/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
stream: false,
|
||||
messages,
|
||||
audio: {
|
||||
format: "wav",
|
||||
voice: voiceId || DEFAULT_VOICE,
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
const rawText = await res.text();
|
||||
let data = {};
|
||||
if (rawText) {
|
||||
try { data = JSON.parse(rawText); } catch { data = {}; }
|
||||
}
|
||||
|
||||
if (!res.ok) {
|
||||
throw new Error(data?.error?.message || rawText || `MiMo TTS error (${res.status})`);
|
||||
}
|
||||
|
||||
const audio = data?.choices?.[0]?.message?.audio?.data;
|
||||
if (!audio) throw new Error(data?.error?.message || "MiMo TTS returned no audio");
|
||||
|
||||
return {
|
||||
base64: audio,
|
||||
format: data?.choices?.[0]?.message?.audio?.format || "wav",
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user