feat(tts): add Xiaomi MiMo text-to-speech support

Adds mimo-v2.5-tts as a Media Provider TTS through the existing
OpenAI-compatible chat-completions endpoint. Voice is selected via the
top-level audio.voice field, and an optional style/language hint is
threaded through tts.js -> ttsCore.js -> the new adapter.
This commit is contained in:
MiQieR
2026-08-05 11:45:27 +07:00
committed by decolua
parent d0751bcff7
commit c570fe33ae
44 changed files with 396 additions and 20 deletions

View File

@@ -33,6 +33,21 @@ const GEMINI_VOICES = [
"Vindemiatrix", "Sadachbia", "Sadaltager", "Sulafat",
].map((id) => ({ id, name: id, type: "tts" }));
// Xiaomi MiMo preset voices (from https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5).
// Voice id is passed via `audio.voice`; `mimo_default` = default (冰糖 on CN cluster, Mia elsewhere).
// Voices are language-independent — the spoken language is a separate hint, not bound to the voice.
const MIMO_VOICES = [
{ id: "mimo_default", name: "mimo_default" },
{ id: "冰糖", name: "冰糖" },
{ id: "茉莉", name: "茉莉" },
{ id: "苏打", name: "苏打" },
{ id: "白桦", name: "白桦" },
{ id: "Mia", name: "Mia" },
{ id: "Chloe", name: "Chloe" },
{ id: "Milo", name: "Milo" },
{ id: "Dean", name: "Dean" },
].map((v) => ({ type: "tts", ...v }));
// ── TTS Config (config-driven, single source of truth) ─────────────────────
export const TTS_MODELS_CONFIG = {
openai: {
@@ -107,6 +122,14 @@ export const TTS_MODELS_CONFIG = {
},
allVoices: GEMINI_VOICES,
},
"xiaomi-mimo": {
models: [
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", type: "tts" },
],
voices: {
"mimo-v2.5-tts": MIMO_VOICES,
},
},
};
// ── Helper: get voices for a specific model ────────────────────────────────

View File

@@ -48,16 +48,16 @@ function createTtsResponse(base64Audio, format, responseFormat) {
*
* @returns {Promise<{success, response, status?, error?}>}
*/
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language }) {
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3", language, style }) {
if (!input?.trim()) {
return createErrorResult(HTTP_STATUS.BAD_REQUEST, "Missing required field: input");
}
try {
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini)
// Special-case adapters (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter, gemini, xiaomi-mimo)
const adapter = getTtsAdapter(provider);
if (adapter) {
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language });
const result = await adapter.synthesize(input.trim(), model, credentials, responseFormat, { language, style });
// Adapter may return a full {success, response} (legacy) or {base64, format}
if (result.success !== undefined) return result;
return createTtsResponse(result.base64, result.format, responseFormat);

View File

@@ -6,6 +6,7 @@ import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
import openai from "./openai.js";
import openrouter from "./openrouter.js";
import gemini, { fetchGeminiVoices } from "./gemini.js";
import xiaomiMimo from "./xiaomi-mimo.js";
import { FORMAT_HANDLERS } from "./genericFormats.js";
import { parseModelVoice } from "./_base.js";
@@ -18,6 +19,7 @@ const SPECIAL_ADAPTERS = {
openai,
openrouter,
gemini,
"xiaomi-mimo": xiaomiMimo,
};
export function getTtsAdapter(provider) {

View File

@@ -0,0 +1,65 @@
// Xiaomi MiMo TTS — via OpenAI-compatible chat completions (non-streaming).
// Docs: https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5
// Message contract: target text in `role: assistant` content, style/voice
// instructions in `role: user` content. Voice is selected via the top-level
// `audio.voice` field (NOT embedded in the model name).
import { parseModelVoice } from "./_base.js";
const DEFAULT_MODEL = "mimo-v2.5-tts";
const DEFAULT_VOICE = "mimo_default";
export default {
synthesize(text, model, credentials, responseFormat, { style, language } = {}) {
if (!credentials?.apiKey) throw new Error("xiaomi-mimo API key required");
return synthesizeMiMo(text, model, credentials.apiKey, style, language);
},
};
export async function synthesizeMiMo(text, model, apiKey, style, language) {
const { modelId, voiceId } = parseModelVoice(model, DEFAULT_MODEL, DEFAULT_VOICE, [DEFAULT_MODEL]);
// Language and style are soft instructions → prepend as a role:user message.
// MiMo auto-detects the spoken language of the text; the hint only nudges it
// (e.g. "Speak in English.") and is independent of the chosen voice.
const instructions = [];
if (language) instructions.push(`Speak in ${language}.`);
if (style) instructions.push(style);
const messages = [{ role: "assistant", content: text }];
if (instructions.length) messages.unshift({ role: "user", content: instructions.join(" ") });
const res = await fetch("https://api.xiaomimimo.com/v1/chat/completions", {
method: "POST",
headers: {
"Content-Type": "application/json",
"Authorization": `Bearer ${apiKey}`,
},
body: JSON.stringify({
model: modelId,
stream: false,
messages,
audio: {
format: "wav",
voice: voiceId || DEFAULT_VOICE,
},
}),
});
const rawText = await res.text();
let data = {};
if (rawText) {
try { data = JSON.parse(rawText); } catch { data = {}; }
}
if (!res.ok) {
throw new Error(data?.error?.message || rawText || `MiMo TTS error (${res.status})`);
}
const audio = data?.choices?.[0]?.message?.audio?.data;
if (!audio) throw new Error(data?.error?.message || "MiMo TTS returned no audio");
return {
base64: audio,
format: data?.choices?.[0]?.message?.audio?.format || "wav",
};
}

View File

@@ -15,10 +15,11 @@ export default {
textIcon: "XM",
website: "https://xiaomimimo.com",
notice: {
apiKeyUrl: "https://xiaomimimo.com",
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
},
},
category: "apikey",
serviceKinds: ["llm", "tts"],
transport: {
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
validateUrl: "https://api.xiaomimimo.com/v1/models",
@@ -42,5 +43,12 @@ export default {
{ id: "mimo-v2.5", name: "MiMo V2.5" },
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
{ id: "mimo-v2-flash", name: "MiMo V2 Flash" },
{ id: "mimo-v2.5-tts", name: "MiMo V2.5 TTS", kind: "tts" },
],
ttsConfig: {
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
authType: "apikey",
authHeader: "bearer",
format: "xiaomi-mimo-tts",
},
};