feat(tts): add Xiaomi MiMo text-to-speech support
Adds mimo-v2.5-tts as a Media Provider TTS through the existing OpenAI-compatible chat-completions endpoint. Voice is selected via the top-level audio.voice field, and an optional style/language hint is threaded through tts.js -> ttsCore.js -> the new adapter.
This commit is contained in:
@@ -33,6 +33,21 @@ const GEMINI_VOICES = [
|
||||
"Vindemiatrix", "Sadachbia", "Sadaltager", "Sulafat",
|
||||
].map((id) => ({ id, name: id, type: "tts" }));
|
||||
|
||||
// Xiaomi MiMo preset voices (from https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5).
|
||||
// Voice id is passed via `audio.voice`; `mimo_default` = default (冰糖 on CN cluster, Mia elsewhere).
|
||||
// Voices are language-independent — the spoken language is a separate hint, not bound to the voice.
|
||||
const MIMO_VOICES = [
|
||||
{ id: "mimo_default", name: "mimo_default" },
|
||||
{ id: "冰糖", name: "冰糖" },
|
||||
{ id: "茉莉", name: "茉莉" },
|
||||
{ id: "苏打", name: "苏打" },
|
||||
{ id: "白桦", name: "白桦" },
|
||||
{ id: "Mia", name: "Mia" },
|
||||
{ id: "Chloe", name: "Chloe" },
|
||||
{ id: "Milo", name: "Milo" },
|
||||
{ id: "Dean", name: "Dean" },
|
||||
].map((v) => ({ type: "tts", ...v }));
|
||||
|
||||
// ── TTS Config (config-driven, single source of truth) ─────────────────────
|
||||
export const TTS_MODELS_CONFIG = {
|
||||
openai: {
|
||||
@@ -107,6 +122,14 @@ export const TTS_MODELS_CONFIG = {
|
||||
},
|
||||
allVoices: GEMINI_VOICES,
|
||||
},
|
||||
"xiaomi-mimo": {
|
||||
models: [
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", type: "tts" },
|
||||
],
|
||||
voices: {
|
||||
"mimo-v2.5-tts": MIMO_VOICES,
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
// ── Helper: get voices for a specific model ────────────────────────────────
|
||||
|
||||
@@ -48,16 +48,16 @@ function createTtsResponse(base64Audio, format, responseFormat) {
|
||||
*
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language }) {
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language, style }) {
|
||||
if (!input?.trim()) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: input");
|
||||
}
|
||||
|
||||
try {
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini)
|
||||
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini, xiaomi-mimo)
|
||||
const adapter = getTtsAdapter(provider);
|
||||
if (adapter) {
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language });
|
||||
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language, style });
|
||||
// Adapter may return a full {success, response} (legacy) or {base64, format}
|
||||
if (result.success !== undefined) return result;
|
||||
return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
|
||||
@@ -6,6 +6,7 @@ import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
||||
import openai from "./openai.js";
|
||||
import openrouter from "./openrouter.js";
|
||||
import gemini, { fetchGeminiVoices } from "./gemini.js";
|
||||
import xiaomiMimo from "./xiaomi-mimo.js";
|
||||
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
@@ -18,6 +19,7 @@ const SPECIAL_ADAPTERS = {
|
||||
openai,
|
||||
openrouter,
|
||||
gemini,
|
||||
"xiaomi-mimo": xiaomiMimo,
|
||||
};
|
||||
|
||||
export function getTtsAdapter(provider) {
|
||||
|
||||
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
65
open-sse/handlers/ttsProviders/xiaomi-mimo.js
Normal file
@@ -0,0 +1,65 @@
|
||||
// Xiaomi MiMo TTS — via OpenAI-compatible chat completions (non-streaming).
|
||||
// Docs: https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5
|
||||
// Message contract: target text in `role: assistant` content, style/voice
|
||||
// instructions in `role: user` content. Voice is selected via the top-level
|
||||
// `audio.voice` field (NOT embedded in the model name).
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
const DEFAULT_MODEL = "mimo-v2.5-tts";
|
||||
const DEFAULT_VOICE = "mimo_default";
|
||||
|
||||
export default {
|
||||
synthesize(text, model, credentials, responseFormat, { style, language } = {}) {
|
||||
if (!credentials?.apiKey) throw new Error("xiaomi-mimo API key required");
|
||||
return synthesizeMiMo(text, model, credentials.apiKey, style, language);
|
||||
},
|
||||
};
|
||||
|
||||
export async function synthesizeMiMo(text, model, apiKey, style, language) {
|
||||
const { modelId, voiceId } = parseModelVoice(model, DEFAULT_MODEL, DEFAULT_VOICE, [DEFAULT_MODEL]);
|
||||
|
||||
// Language and style are soft instructions → prepend as a role:user message.
|
||||
// MiMo auto-detects the spoken language of the text; the hint only nudges it
|
||||
// (e.g. "Speak in English.") and is independent of the chosen voice.
|
||||
const instructions = [];
|
||||
if (language) instructions.push(`Speak in ${language}.`);
|
||||
if (style) instructions.push(style);
|
||||
|
||||
const messages = [{ role: "assistant", content: text }];
|
||||
if (instructions.length) messages.unshift({ role: "user", content: instructions.join(" ") });
|
||||
|
||||
const res = await fetch("https://api.xiaomimimo.com/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
stream: false,
|
||||
messages,
|
||||
audio: {
|
||||
format: "wav",
|
||||
voice: voiceId || DEFAULT_VOICE,
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
const rawText = await res.text();
|
||||
let data = {};
|
||||
if (rawText) {
|
||||
try { data = JSON.parse(rawText); } catch { data = {}; }
|
||||
}
|
||||
|
||||
if (!res.ok) {
|
||||
throw new Error(data?.error?.message || rawText || `MiMo TTS error (${res.status})`);
|
||||
}
|
||||
|
||||
const audio = data?.choices?.[0]?.message?.audio?.data;
|
||||
if (!audio) throw new Error(data?.error?.message || "MiMo TTS returned no audio");
|
||||
|
||||
return {
|
||||
base64: audio,
|
||||
format: data?.choices?.[0]?.message?.audio?.format || "wav",
|
||||
};
|
||||
}
|
||||
@@ -15,10 +15,11 @@ export default {
|
||||
textIcon: "XM",
|
||||
website: "https://xiaomimimo.com",
|
||||
notice: {
|
||||
apiKeyUrl: "https://xiaomimimo.com",
|
||||
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
serviceKinds: ["llm", "tts"],
|
||||
transport: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
validateUrl: "https://api.xiaomimimo.com/v1/models",
|
||||
@@ -42,5 +43,12 @@ export default {
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
|
||||
{ id: "mimo-v2-flash", name: "MiMo V2 Flash" },
|
||||
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", kind: "tts" },
|
||||
],
|
||||
ttsConfig: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
format: "xiaomi-mimo-tts",
|
||||
},
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user