diff --git a/open-sse/handlers/ttsProviders/genericFormats.js b/open-sse/handlers/ttsProviders/genericFormats.js index 2f27f9ea..a5c0cda3 100644 --- a/open-sse/handlers/ttsProviders/genericFormats.js +++ b/open-sse/handlers/ttsProviders/genericFormats.js @@ -51,6 +51,25 @@ async function huggingface({ baseUrl, apiKey, text, modelId }) { return responseToBase64(res, "wav"); } +// Fish Audio: model travels in an HTTP header, the voice is a reference_id, returns binary +async function fishAudio({ baseUrl, apiKey, text, modelId, voiceId }) { + const res = await fetch(baseUrl, { + method: "POST", + headers: { + "Content-Type": "application/json", + "Authorization": `Bearer ${apiKey}`, + "model": modelId || "s2.1-pro-free", + }, + body: JSON.stringify({ + text, + format: "mp3", + ...(voiceId ? { reference_id: voiceId } : {}), + }), + }); + if (!res.ok) await throwUpstreamError(res); + return responseToBase64(res, "mp3"); +} + // Inworld: Basic auth, JSON { audioContent } async function inworld({ baseUrl, apiKey, text, modelId, voiceId }) { const res = await fetch(baseUrl, { @@ -166,4 +185,5 @@ export const FORMAT_HANDLERS = { tortoise, openai: openaiCompat, "minimax-tts": minimaxTts, + "fish-audio": fishAudio, }; diff --git a/open-sse/providers/registry/fish-audio.js b/open-sse/providers/registry/fish-audio.js new file mode 100644 index 00000000..f016a247 --- /dev/null +++ b/open-sse/providers/registry/fish-audio.js @@ -0,0 +1,31 @@ +// Fish Audio TTS — the model id travels in an HTTP `model` header rather than the +// JSON body, and the voice is a reference_id (a cloned or preset voice model). +export default { + id: "fish-audio", + alias: "fish", + display: { + name: "Fish Audio", + icon: "record_voice_over", + color: "#1E9BF0", + textIcon: "FA", + website: "https://fish.audio", + notice: { + apiKeyUrl: "https://fish.audio/app/api-keys/", + }, + }, + category: "apikey", + authType: "apikey", + serviceKinds: ["tts"], + ttsConfig: { + baseUrl: "https://api.fish.audio/v1/tts", + authType: "apikey", + authHeader: "bearer", + format: "fish-audio", + models: [ + { id: "s2.1-pro-free", name: "S2.1 Pro Free" }, + { id: "s2.1-pro", name: "S2.1 Pro" }, + { id: "s2-pro", name: "S2 Pro" }, + { id: "s1", name: "S1" }, + ], + }, +}; diff --git a/open-sse/providers/registry/index.js b/open-sse/providers/registry/index.js index 467d1c4d..3adffde5 100644 --- a/open-sse/providers/registry/index.js +++ b/open-sse/providers/registry/index.js @@ -119,6 +119,7 @@ import p116 from "./tokenrouter.js"; import p117 from "./selfhosted-stt.js"; import p118 from "./selfhosted-tts.js"; import p119 from "./selfhosted-embedding.js"; +import p120 from "./fish-audio.js"; export default [ p0, @@ -239,4 +240,5 @@ export default [ p117, p118, p119, + p120, ]; diff --git a/tests/unit/fish-audio-tts.test.js b/tests/unit/fish-audio-tts.test.js new file mode 100644 index 00000000..8a1cd6a2 --- /dev/null +++ b/tests/unit/fish-audio-tts.test.js @@ -0,0 +1,109 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +import REGISTRY from "../../open-sse/providers/registry/index.js"; +import { PROVIDER_MEDIA } from "../../open-sse/providers/index.js"; +import { FORMAT_HANDLERS } from "../../open-sse/handlers/ttsProviders/genericFormats.js"; +import { AI_PROVIDERS } from "@/shared/constants/providers"; + +const AUDIO = new Uint8Array(256).fill(7); + +function okResponse() { + return { + ok: true, + status: 200, + headers: new Headers({ "content-type": "audio/mpeg" }), + arrayBuffer: async () => AUDIO.buffer, + }; +} + +describe("Fish Audio TTS provider", () => { + const entry = REGISTRY.find((e) => e.id === "fish-audio"); + + it("is registered as a TTS-only apikey provider", () => { + expect(entry).toBeDefined(); + expect(entry.category).toBe("apikey"); + expect(entry.serviceKinds).toEqual(["tts"]); + expect(PROVIDER_MEDIA["fish-audio"]?.ttsConfig?.baseUrl).toBe("https://api.fish.audio/v1/tts"); + }); + + it("is visible to the generic dispatcher, which reads AI_PROVIDERS", () => { + // synthesizeViaConfig() looks the provider up here, not in PROVIDER_MEDIA. + expect(AI_PROVIDERS["fish-audio"]?.ttsConfig?.format).toBe("fish-audio"); + expect(typeof FORMAT_HANDLERS["fish-audio"]).toBe("function"); + }); + + it("exposes the four documented models", () => { + const ids = (entry.ttsConfig.models || []).map((m) => m.id); + expect(ids).toEqual(["s2.1-pro-free", "s2.1-pro", "s2-pro", "s1"]); + }); + + it("keeps registry ids and aliases unique", () => { + const ids = REGISTRY.map((e) => e.id); + expect(new Set(ids).size).toBe(ids.length); + const aliases = REGISTRY.map((e) => e.alias).filter(Boolean); + expect(new Set(aliases).size).toBe(aliases.length); + }); +}); + +describe("Fish Audio TTS request shape", () => { + const handler = FORMAT_HANDLERS["fish-audio"]; + let fetchMock; + + beforeEach(() => { + fetchMock = vi.fn(async () => okResponse()); + global.fetch = fetchMock; + }); + + const callArgs = () => { + const [url, init] = fetchMock.mock.calls.at(-1); + return { url, init, body: JSON.parse(init.body) }; + }; + + it("sends the model as an HTTP header, not in the body", async () => { + await handler({ + baseUrl: "https://api.fish.audio/v1/tts", + apiKey: "sk-test", + text: "xin chào", + modelId: "s1", + voiceId: "", + }); + + const { url, init, body } = callArgs(); + expect(url).toBe("https://api.fish.audio/v1/tts"); + expect(init.headers.model).toBe("s1"); + expect(init.headers.Authorization).toBe("Bearer sk-test"); + expect(body).toEqual({ text: "xin chào", format: "mp3" }); + }); + + it("maps the voice onto reference_id, and omits it when unset", async () => { + await handler({ baseUrl: "u", apiKey: "k", text: "t", modelId: "s1", voiceId: "voice-abc" }); + expect(callArgs().body.reference_id).toBe("voice-abc"); + + await handler({ baseUrl: "u", apiKey: "k", text: "t", modelId: "s1", voiceId: "" }); + expect(callArgs().body).not.toHaveProperty("reference_id"); + }); + + it("defaults to the free model when none is given", async () => { + await handler({ baseUrl: "u", apiKey: "k", text: "t", modelId: "", voiceId: "" }); + expect(callArgs().init.headers.model).toBe("s2.1-pro-free"); + }); + + it("returns base64 audio with its format", async () => { + const out = await handler({ baseUrl: "u", apiKey: "k", text: "t", modelId: "s1", voiceId: "" }); + expect(out.format).toBe("mp3"); + expect(typeof out.base64).toBe("string"); + expect(out.base64.length).toBeGreaterThan(0); + }); + + it("surfaces the upstream error message", async () => { + global.fetch = vi.fn(async () => ({ + ok: false, + status: 402, + text: async () => JSON.stringify({ message: "Insufficient credit" }), + })); + + await expect( + handler({ baseUrl: "u", apiKey: "k", text: "t", modelId: "s1", voiceId: "" }), + ).rejects.toThrow("Insufficient credit"); + }); +});