Feat : Skills
This commit is contained in:
39
open-sse/handlers/ttsProviders/_base.js
Normal file
39
open-sse/handlers/ttsProviders/_base.js
Normal file
@@ -0,0 +1,39 @@
|
||||
// Shared TTS helpers
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
export const UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/146.0.0.0 Safari/537.36";
|
||||
|
||||
// Convert upstream Response (binary audio) to { base64, format }
|
||||
export async function responseToBase64(res, defaultFormat = "mp3") {
|
||||
const buf = await res.arrayBuffer();
|
||||
if (buf.byteLength < 100) throw new Error("Upstream returned empty audio");
|
||||
const ctype = res.headers.get("content-type") || "";
|
||||
let format = defaultFormat;
|
||||
if (ctype.includes("wav")) format = "wav";
|
||||
else if (ctype.includes("mpeg") || ctype.includes("mp3")) format = "mp3";
|
||||
else if (ctype.includes("ogg")) format = "ogg";
|
||||
return { base64: Buffer.from(buf).toString("base64"), format };
|
||||
}
|
||||
|
||||
export async function throwUpstreamError(res) {
|
||||
const text = await res.text().catch(() => "");
|
||||
let msg = `Upstream error (${res.status})`;
|
||||
try {
|
||||
const parsed = JSON.parse(text);
|
||||
msg = parsed?.error?.message || parsed?.message || parsed?.detail?.message || (typeof parsed?.detail === "string" ? parsed.detail : null) || text || msg;
|
||||
} catch { msg = text || msg; }
|
||||
throw new Error(msg);
|
||||
}
|
||||
|
||||
// Parse `model` string as "modelId/voiceId" — match against known model list (longest prefix wins)
|
||||
export function parseModelVoice(model, defaultModel = "", defaultVoice = "", knownModels = []) {
|
||||
if (!model) return { modelId: defaultModel, voiceId: defaultVoice };
|
||||
const known = knownModels.map((m) => m.id || m).filter(Boolean).sort((a, b) => b.length - a.length);
|
||||
for (const id of known) {
|
||||
if (model === id) return { modelId: id, voiceId: defaultVoice };
|
||||
if (model.startsWith(`${id}/`)) return { modelId: id, voiceId: model.slice(id.length + 1) };
|
||||
}
|
||||
const idx = model.lastIndexOf("/");
|
||||
if (idx > 0) return { modelId: model.slice(0, idx), voiceId: model.slice(idx + 1) };
|
||||
return { modelId: defaultModel || model, voiceId: defaultVoice || model };
|
||||
}
|
||||
89
open-sse/handlers/ttsProviders/edgeTts.js
Normal file
89
open-sse/handlers/ttsProviders/edgeTts.js
Normal file
@@ -0,0 +1,89 @@
|
||||
// Microsoft Edge / Bing TTS (no auth) — via Bing translator endpoint
|
||||
import { Buffer } from "node:buffer";
|
||||
import { UA } from "./_base.js";
|
||||
|
||||
const REFRESH_MS = 5 * 60 * 1000; // token TTL ~1h, refresh early
|
||||
const VOICES_TTL = 24 * 60 * 60 * 1000;
|
||||
|
||||
const cache = { token: null, tokenTime: 0 };
|
||||
let _voicesCache = null;
|
||||
let _voicesCacheTime = 0;
|
||||
|
||||
async function getToken() {
|
||||
const now = Date.now();
|
||||
if (cache.token && now - cache.tokenTime < REFRESH_MS) return cache.token;
|
||||
const res = await fetch("https://www.bing.com/translator", {
|
||||
headers: { "User-Agent": UA, "Accept-Language": "vi,en-US;q=0.9,en;q=0.8" },
|
||||
});
|
||||
if (!res.ok) throw new Error(`Bing translator fetch failed: ${res.status}`);
|
||||
const rawCookies = res.headers.getSetCookie?.() || [];
|
||||
const cookie = rawCookies.map((c) => c.split(";")[0]).join("; ");
|
||||
const html = await res.text();
|
||||
const match = html.match(/params_AbusePreventionHelper\s*=\s*\[([^,]+),([^,]+),/);
|
||||
if (!match) throw new Error("Failed to parse Bing token");
|
||||
cache.token = { key: match[1], token: match[2].replace(/"/g, ""), cookie };
|
||||
cache.tokenTime = now;
|
||||
return cache.token;
|
||||
}
|
||||
|
||||
async function ttsRequest(text, voiceId, token) {
|
||||
const parts = voiceId.split("-");
|
||||
const xmlLang = parts.slice(0, 2).join("-");
|
||||
const gender = voiceId.toLowerCase().includes("male") ? "Male" : "Female";
|
||||
const ssml = `<speak version='1.0' xml:lang='${xmlLang}'><voice xml:lang='${xmlLang}' xml:gender='${gender}' name='${voiceId}'><prosody rate='0.00%'>${text}</prosody></voice></speak>`;
|
||||
const body = new URLSearchParams();
|
||||
body.append("ssml", ssml);
|
||||
body.append("token", token.token);
|
||||
body.append("key", token.key);
|
||||
return fetch("https://www.bing.com/tfettts?isVertical=1&&IG=1&IID=translator.5023&SFX=1", {
|
||||
method: "POST",
|
||||
body: body.toString(),
|
||||
headers: {
|
||||
"Content-Type": "application/x-www-form-urlencoded",
|
||||
"Accept": "*/*",
|
||||
"Origin": "https://www.bing.com",
|
||||
"Referer": "https://www.bing.com/translator",
|
||||
"User-Agent": UA,
|
||||
...(token.cookie ? { "Cookie": token.cookie } : {}),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export async function fetchEdgeTtsVoices() {
|
||||
const now = Date.now();
|
||||
if (_voicesCache && now - _voicesCacheTime < VOICES_TTL) return _voicesCache;
|
||||
const res = await fetch(
|
||||
"https://speech.platform.bing.com/consumer/speech/synthesize/readaloud/voices/list?trustedclienttoken=6A5AA1D4EAFF4E9FB37E23D68491D6F4",
|
||||
{ headers: { "User-Agent": UA } }
|
||||
);
|
||||
if (!res.ok) throw new Error(`Edge TTS voices fetch failed: ${res.status}`);
|
||||
const voices = await res.json();
|
||||
_voicesCache = voices;
|
||||
_voicesCacheTime = now;
|
||||
return voices;
|
||||
}
|
||||
|
||||
export default {
|
||||
noAuth: true,
|
||||
async synthesize(text, model) {
|
||||
const voiceId = model || "vi-VN-HoaiMyNeural";
|
||||
let token = await getToken();
|
||||
let res = await ttsRequest(text, voiceId, token);
|
||||
|
||||
// 429/403: invalidate cache and retry once
|
||||
if (res.status === 429 || res.status === 403) {
|
||||
cache.token = null;
|
||||
cache.tokenTime = 0;
|
||||
token = await getToken();
|
||||
res = await ttsRequest(text, voiceId, token);
|
||||
}
|
||||
|
||||
if (!res.ok) {
|
||||
const body = await res.text().catch(() => "");
|
||||
throw new Error(`Bing TTS failed: ${res.status}${body ? " - " + body : ""}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
if (buf.byteLength < 1024) throw new Error("Bing TTS returned empty audio");
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: "mp3" };
|
||||
},
|
||||
};
|
||||
48
open-sse/handlers/ttsProviders/elevenlabs.js
Normal file
48
open-sse/handlers/ttsProviders/elevenlabs.js
Normal file
@@ -0,0 +1,48 @@
|
||||
// ElevenLabs TTS — voice id with optional model_id prefix
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
const VOICES_TTL = 24 * 60 * 60 * 1000;
|
||||
const _voicesCache = new Map(); // by API key
|
||||
|
||||
export async function fetchElevenLabsVoices(apiKey) {
|
||||
if (!apiKey) throw new Error("ElevenLabs API key required");
|
||||
const now = Date.now();
|
||||
const cached = _voicesCache.get(apiKey);
|
||||
if (cached && now - cached.time < VOICES_TTL) return cached.voices;
|
||||
|
||||
const res = await fetch("https://api.elevenlabs.io/v1/voices", {
|
||||
headers: { "xi-api-key": apiKey, "Content-Type": "application/json" },
|
||||
});
|
||||
if (!res.ok) throw new Error(`ElevenLabs voices fetch failed: ${res.status}`);
|
||||
const data = await res.json();
|
||||
// Normalize: derive lang from labels for grouping
|
||||
const voices = (data.voices || []).map((v) => ({ ...v, lang: v.labels?.language || "en" }));
|
||||
_voicesCache.set(apiKey, { voices, time: now });
|
||||
return voices;
|
||||
}
|
||||
|
||||
export default {
|
||||
async synthesize(text, model, credentials) {
|
||||
if (!credentials?.apiKey) throw new Error("ElevenLabs API key required");
|
||||
let modelId = "eleven_flash_v2_5";
|
||||
let voiceId = model;
|
||||
if (model && model.includes("/")) [modelId, voiceId] = model.split("/");
|
||||
|
||||
const res = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`, {
|
||||
method: "POST",
|
||||
headers: { "xi-api-key": credentials.apiKey, "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
model_id: modelId,
|
||||
voice_settings: { stability: 0.5, similarity_boost: 0.75 },
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.detail?.message || `ElevenLabs TTS failed: ${res.status}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
if (buf.byteLength < 1024) throw new Error("ElevenLabs TTS returned empty audio");
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: "mp3" };
|
||||
},
|
||||
};
|
||||
167
open-sse/handlers/ttsProviders/genericFormats.js
Normal file
167
open-sse/handlers/ttsProviders/genericFormats.js
Normal file
@@ -0,0 +1,167 @@
|
||||
// Generic config-driven TTS handlers — dispatched by ttsConfig.format.
|
||||
// Each handler accepts { baseUrl, apiKey, text, modelId, voiceId } and returns { base64, format }.
|
||||
import { responseToBase64, throwUpstreamError } from "./_base.js";
|
||||
|
||||
// Hyperbolic: POST { text } → { audio: base64 }
|
||||
async function hyperbolic({ baseUrl, apiKey, text }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${apiKey}` },
|
||||
body: JSON.stringify({ text }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
const data = await res.json();
|
||||
return { base64: data.audio, format: "mp3" };
|
||||
}
|
||||
|
||||
// Deepgram: model via query, Token auth, returns binary
|
||||
async function deepgram({ baseUrl, apiKey, text, modelId }) {
|
||||
const url = new URL(baseUrl);
|
||||
url.searchParams.set("model", modelId || "aura-asteria-en");
|
||||
const res = await fetch(url.toString(), {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Token ${apiKey}` },
|
||||
body: JSON.stringify({ text }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Nvidia NIM: POST { input: { text }, voice, model } → binary
|
||||
async function nvidia({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${apiKey}` },
|
||||
body: JSON.stringify({ input: { text }, voice: voiceId || "default", model: modelId }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// HuggingFace: POST {baseUrl}/{modelId} { inputs: text } → binary
|
||||
async function huggingface({ baseUrl, apiKey, text, modelId }) {
|
||||
if (!modelId || modelId.includes("..")) throw new Error("Invalid HuggingFace model ID");
|
||||
const res = await fetch(`${baseUrl}/${modelId}`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${apiKey}` },
|
||||
body: JSON.stringify({ inputs: text }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Inworld: Basic auth, JSON { audioContent }
|
||||
async function inworld({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Basic ${apiKey}` },
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
voiceId: voiceId || "Alex",
|
||||
modelId: modelId || "inworld-tts-1.5-mini",
|
||||
audioConfig: { audioEncoding: "MP3" },
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
const data = await res.json();
|
||||
if (!data.audioContent) throw new Error("Inworld TTS returned no audio");
|
||||
return { base64: data.audioContent, format: "mp3" };
|
||||
}
|
||||
|
||||
// Cartesia: X-API-Key header
|
||||
async function cartesia({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"X-API-Key": apiKey,
|
||||
"Cartesia-Version": "2024-06-10",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model_id: modelId || "sonic-2",
|
||||
transcript: text,
|
||||
...(voiceId ? { voice: { mode: "id", id: voiceId } } : {}),
|
||||
output_format: { container: "mp3", bit_rate: 128000, sample_rate: 44100 },
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// PlayHT: token format "userId:apiKey", voice = s3 URL
|
||||
async function playht({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const [userId, key] = (apiKey || ":").split(":");
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Accept": "audio/mpeg",
|
||||
"X-USER-ID": userId || "",
|
||||
"Authorization": `Bearer ${key || apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
voice: voiceId || "s3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json",
|
||||
voice_engine: modelId || "PlayDialog",
|
||||
output_format: "mp3",
|
||||
speed: 1,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Coqui (local, noAuth): POST { text, speaker_id } → WAV
|
||||
async function coqui({ baseUrl, text, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ text, ...(voiceId ? { speaker_id: voiceId } : {}) }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Tortoise (local, noAuth)
|
||||
async function tortoise({ baseUrl, text, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ text, voice: voiceId || "random" }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// OpenAI-compatible upstream (qwen3-tts, etc.)
|
||||
async function openaiCompat({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
input: text,
|
||||
voice: voiceId || "alloy",
|
||||
response_format: "mp3",
|
||||
speed: 1.0,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// format → handler dispatcher
|
||||
export const FORMAT_HANDLERS = {
|
||||
hyperbolic,
|
||||
deepgram,
|
||||
"nvidia-tts": nvidia,
|
||||
"huggingface-tts": huggingface,
|
||||
inworld,
|
||||
cartesia,
|
||||
playht,
|
||||
coqui,
|
||||
tortoise,
|
||||
openai: openaiCompat,
|
||||
};
|
||||
54
open-sse/handlers/ttsProviders/googleTts.js
Normal file
54
open-sse/handlers/ttsProviders/googleTts.js
Normal file
@@ -0,0 +1,54 @@
|
||||
// Google Translate TTS (no auth) — scrape token + batchexecute RPC
|
||||
import { UA } from "./_base.js";
|
||||
|
||||
const REFRESH_MS = 11 * 60 * 1000;
|
||||
const cache = { token: null, tokenTime: 0 };
|
||||
let _idx = 0;
|
||||
|
||||
async function getToken() {
|
||||
const now = Date.now();
|
||||
if (cache.token && now - cache.tokenTime < REFRESH_MS) return cache.token;
|
||||
const res = await fetch("https://translate.google.com/", { headers: { "User-Agent": UA } });
|
||||
if (!res.ok) throw new Error(`Google translate fetch failed: ${res.status}`);
|
||||
const html = await res.text();
|
||||
const fSid = html.match(/"FdrFJe":"(.*?)"/)?.[1];
|
||||
const bl = html.match(/"cfb2h":"(.*?)"/)?.[1];
|
||||
if (!fSid || !bl) throw new Error("Failed to parse Google token");
|
||||
cache.token = { "f.sid": fSid, bl };
|
||||
cache.tokenTime = now;
|
||||
return cache.token;
|
||||
}
|
||||
|
||||
export default {
|
||||
noAuth: true,
|
||||
async synthesize(text, model) {
|
||||
const lang = model || "en";
|
||||
const token = await getToken();
|
||||
const cleanText = text.replace(/[@^*()\\/\-_+=><"'\u201c\u201d\u3010\u3011]/g, " ").replaceAll(", ", ". ");
|
||||
const rpcId = "jQ1olc";
|
||||
const reqId = (++_idx * 100000) + Math.floor(1000 + Math.random() * 9000);
|
||||
const query = new URLSearchParams({
|
||||
rpcids: rpcId,
|
||||
"f.sid": token["f.sid"],
|
||||
bl: token.bl,
|
||||
hl: lang,
|
||||
"soc-app": 1, "soc-platform": 1, "soc-device": 1,
|
||||
_reqid: reqId,
|
||||
rt: "c",
|
||||
});
|
||||
const payload = [cleanText, lang, null, "undefined", [0]];
|
||||
const body = new URLSearchParams();
|
||||
body.append("f.req", JSON.stringify([[[rpcId, JSON.stringify(payload), null, "generic"]]]));
|
||||
const res = await fetch(`https://translate.google.com/_/TranslateWebserverUi/data/batchexecute?${query}`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/x-www-form-urlencoded", "Referer": "https://translate.google.com/" },
|
||||
body: body.toString(),
|
||||
});
|
||||
if (!res.ok) throw new Error(`Google TTS failed: ${res.status}`);
|
||||
const data = await res.text();
|
||||
const split = JSON.parse(data.split("\n")[3]);
|
||||
const base64 = JSON.parse(split[0][2])[0];
|
||||
if (!base64 || base64.length < 100) throw new Error("Google TTS returned empty audio");
|
||||
return { base64, format: "mp3" };
|
||||
},
|
||||
};
|
||||
47
open-sse/handlers/ttsProviders/index.js
Normal file
47
open-sse/handlers/ttsProviders/index.js
Normal file
@@ -0,0 +1,47 @@
|
||||
// TTS provider registry
|
||||
import googleTts from "./googleTts.js";
|
||||
import edgeTts, { fetchEdgeTtsVoices } from "./edgeTts.js";
|
||||
import localDevice, { fetchLocalDeviceVoices } from "./localDevice.js";
|
||||
import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
||||
import openai from "./openai.js";
|
||||
import openrouter from "./openrouter.js";
|
||||
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
||||
import { parseModelVoice } from "./_base.js";
|
||||
|
||||
// Special providers with custom synthesize() logic
|
||||
const SPECIAL_ADAPTERS = {
|
||||
"google-tts": googleTts,
|
||||
"edge-tts": edgeTts,
|
||||
"local-device": localDevice,
|
||||
elevenlabs,
|
||||
openai,
|
||||
openrouter,
|
||||
};
|
||||
|
||||
export function getTtsAdapter(provider) {
|
||||
return SPECIAL_ADAPTERS[provider] || null;
|
||||
}
|
||||
|
||||
// Generic config-driven dispatcher (uses ttsConfig.format)
|
||||
export async function synthesizeViaConfig(provider, text, model, credentials) {
|
||||
const { AI_PROVIDERS } = await import("@/shared/constants/providers");
|
||||
const cfg = AI_PROVIDERS[provider]?.ttsConfig;
|
||||
if (!cfg) return null;
|
||||
const handler = FORMAT_HANDLERS[cfg.format];
|
||||
if (!handler) return null;
|
||||
const apiKey = credentials?.apiKey;
|
||||
if (cfg.authType !== "none" && !apiKey) throw new Error(`${provider} API key required`);
|
||||
const defaultModel = cfg.models?.[0]?.id || "";
|
||||
const { modelId, voiceId } = parseModelVoice(model, defaultModel, "", cfg.models || []);
|
||||
return handler({ baseUrl: cfg.baseUrl, apiKey, text, modelId, voiceId });
|
||||
}
|
||||
|
||||
// Voice fetchers (used by /api/media-providers/tts/voices route)
|
||||
export const VOICE_FETCHERS = {
|
||||
"edge-tts": fetchEdgeTtsVoices,
|
||||
"local-device": fetchLocalDeviceVoices,
|
||||
elevenlabs: fetchElevenLabsVoices,
|
||||
};
|
||||
|
||||
// Re-export for backward compat
|
||||
export { fetchEdgeTtsVoices, fetchLocalDeviceVoices, fetchElevenLabsVoices };
|
||||
87
open-sse/handlers/ttsProviders/localDevice.js
Normal file
87
open-sse/handlers/ttsProviders/localDevice.js
Normal file
@@ -0,0 +1,87 @@
|
||||
// Local device TTS — macOS `say` + Windows SAPI + ffmpeg
|
||||
import { execFile } from "node:child_process";
|
||||
import { promisify } from "node:util";
|
||||
import { mkdtemp, readFile, rm } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
let _voicesCache = null;
|
||||
|
||||
async function fetchVoicesMac() {
|
||||
const { stdout } = await execFileAsync("say", ["-v", "?"]);
|
||||
const voices = [];
|
||||
for (const line of stdout.split("\n")) {
|
||||
const m = line.match(/^([^\s].*?)\s{2,}([a-z]{2}_[A-Z]{2})/);
|
||||
if (!m) continue;
|
||||
const name = m[1].trim();
|
||||
const locale = m[2].trim();
|
||||
const lang = locale.split("_")[0];
|
||||
const country = locale.split("_")[1];
|
||||
voices.push({ id: name, name, locale, lang, country, gender: "" });
|
||||
}
|
||||
return voices;
|
||||
}
|
||||
|
||||
async function fetchVoicesWin() {
|
||||
const script = [
|
||||
"Add-Type -AssemblyName System.Speech;",
|
||||
"$s = New-Object System.Speech.Synthesis.SpeechSynthesizer;",
|
||||
"$s.GetInstalledVoices() | ForEach-Object { $v = $_.VoiceInfo;",
|
||||
"[PSCustomObject]@{ Name=$v.Name; Culture=$v.Culture.Name; Gender=$v.Gender } }",
|
||||
"| ConvertTo-Json -Compress",
|
||||
].join(" ");
|
||||
const { stdout } = await execFileAsync(
|
||||
"powershell.exe",
|
||||
["-NoProfile", "-NonInteractive", "-WindowStyle", "Hidden", "-Command", script],
|
||||
{ windowsHide: true }
|
||||
);
|
||||
const raw = JSON.parse(stdout.trim() || "[]");
|
||||
const list = Array.isArray(raw) ? raw : [raw];
|
||||
return list.map((v) => {
|
||||
const culture = v.Culture || "en-US";
|
||||
const [lang, country = ""] = culture.split("-");
|
||||
const genderMap = { 1: "Male", 2: "Female", Male: "Male", Female: "Female" };
|
||||
return {
|
||||
id: v.Name, name: v.Name,
|
||||
locale: culture.replace("-", "_"),
|
||||
lang, country,
|
||||
gender: genderMap[v.Gender] || "",
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
export async function fetchLocalDeviceVoices() {
|
||||
if (_voicesCache) return _voicesCache;
|
||||
try {
|
||||
const voices = process.platform === "win32" ? await fetchVoicesWin() : await fetchVoicesMac();
|
||||
_voicesCache = voices;
|
||||
return voices;
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
async function synthesizeMacOrWin(text, voiceId) {
|
||||
const dir = await mkdtemp(join(tmpdir(), "tts-"));
|
||||
const aiffPath = join(dir, "out.aiff");
|
||||
const mp3Path = join(dir, "out.mp3");
|
||||
try {
|
||||
const args = voiceId ? ["-v", voiceId, "-o", aiffPath, text] : ["-o", aiffPath, text];
|
||||
await execFileAsync("say", args);
|
||||
await execFileAsync("ffmpeg", ["-y", "-i", aiffPath, "-codec:a", "libmp3lame", "-qscale:a", "4", mp3Path]);
|
||||
const buf = await readFile(mp3Path);
|
||||
return buf.toString("base64");
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
export default {
|
||||
noAuth: true,
|
||||
async synthesize(text, model) {
|
||||
const base64 = await synthesizeMacOrWin(text, model);
|
||||
return { base64, format: "mp3" };
|
||||
},
|
||||
};
|
||||
30
open-sse/handlers/ttsProviders/openai.js
Normal file
30
open-sse/handlers/ttsProviders/openai.js
Normal file
@@ -0,0 +1,30 @@
|
||||
// OpenAI TTS — model format: "tts-model/voice"
|
||||
import { Buffer } from "node:buffer";
|
||||
|
||||
export default {
|
||||
async synthesize(text, model, credentials) {
|
||||
if (!credentials?.apiKey) throw new Error("No OpenAI API key configured");
|
||||
|
||||
let ttsModel = "gpt-4o-mini-tts";
|
||||
let voice = "alloy";
|
||||
if (model && model.includes("/")) {
|
||||
const parts = model.split("/");
|
||||
if (parts.length === 2) [ttsModel, voice] = parts;
|
||||
} else if (model) {
|
||||
voice = model;
|
||||
}
|
||||
|
||||
const baseUrl = (credentials.baseUrl || "https://api.openai.com").replace(/\/+$/, "");
|
||||
const res = await fetch(`${baseUrl}/v1/audio/speech`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${credentials.apiKey}` },
|
||||
body: JSON.stringify({ model: ttsModel, voice, input: text }),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.error?.message || `OpenAI TTS failed: ${res.status}`);
|
||||
}
|
||||
const buf = await res.arrayBuffer();
|
||||
return { base64: Buffer.from(buf).toString("base64"), format: "mp3" };
|
||||
},
|
||||
};
|
||||
70
open-sse/handlers/ttsProviders/openrouter.js
Normal file
70
open-sse/handlers/ttsProviders/openrouter.js
Normal file
@@ -0,0 +1,70 @@
|
||||
// OpenRouter TTS — via chat completions + audio modality (SSE stream)
|
||||
export default {
|
||||
async synthesize(text, model, credentials) {
|
||||
if (!credentials?.apiKey) throw new Error("No OpenRouter API key configured");
|
||||
|
||||
// model format: "tts-model/voice" e.g. "openai/gpt-4o-mini-tts/alloy"
|
||||
let ttsModel = "openai/gpt-4o-mini-tts";
|
||||
let voice = "alloy";
|
||||
if (model && model.includes("/")) {
|
||||
const lastSlash = model.lastIndexOf("/");
|
||||
const maybVoice = model.slice(lastSlash + 1);
|
||||
const maybeModel = model.slice(0, lastSlash);
|
||||
if (maybeModel.includes("/")) {
|
||||
ttsModel = maybeModel;
|
||||
voice = maybVoice;
|
||||
} else {
|
||||
voice = model;
|
||||
}
|
||||
} else if (model) {
|
||||
voice = model;
|
||||
}
|
||||
|
||||
const res = await fetch("https://openrouter.ai/api/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": `Bearer ${credentials.apiKey}`,
|
||||
"HTTP-Referer": "https://endpoint-proxy.local",
|
||||
"X-Title": "Endpoint Proxy",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: ttsModel,
|
||||
modalities: ["text", "audio"],
|
||||
audio: { voice, format: "wav" },
|
||||
stream: true,
|
||||
messages: [{ role: "user", content: text }],
|
||||
}),
|
||||
});
|
||||
|
||||
if (!res.ok) {
|
||||
const err = await res.json().catch(() => ({}));
|
||||
throw new Error(err?.error?.message || `OpenRouter TTS failed: ${res.status}`);
|
||||
}
|
||||
|
||||
// Parse SSE stream, accumulate base64 audio chunks
|
||||
const chunks = [];
|
||||
const reader = res.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = "";
|
||||
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop();
|
||||
for (const line of lines) {
|
||||
if (!line.startsWith("data: ") || line === "data: [DONE]") continue;
|
||||
try {
|
||||
const json = JSON.parse(line.slice(6));
|
||||
const audioData = json.choices?.[0]?.delta?.audio?.data;
|
||||
if (audioData) chunks.push(audioData);
|
||||
} catch {}
|
||||
}
|
||||
}
|
||||
|
||||
if (chunks.length === 0) throw new Error("OpenRouter TTS returned no audio data");
|
||||
return { base64: chunks.join(""), format: "wav" };
|
||||
},
|
||||
};
|
||||
Reference in New Issue
Block a user