From a32f71ac5a40938748991cc205f82555bc743043 Mon Sep 17 00:00:00 2001 From: Dave Horton Date: Sun, 23 Aug 2026 13:11:59 -0400 Subject: [PATCH] feat: add fishaudio (Fish Audio) TTS support (#159) Adds synthFishaudio with both arms: the say: streaming url consumed by the mediajam dialect, and a POST /v1/tts cache render. The render asks for raw pcm at 8k and returns extension r8 because fish's wav output carries a placeholder RIFF size, the same problem gradium has. Fish is a voice-cloning vendor, so the voice is a reference_id; the sentinel 'default' means send none and use fish's own default voice. Co-authored-by: Claude Opus 5 --- lib/synth-audio.js | 86 +++++++++++++++++++++++++++++++++++++++++++++- test/synth.js | 41 ++++++++++++++++++++++ 2 files changed, 126 insertions(+), 1 deletion(-) diff --git a/lib/synth-audio.js b/lib/synth-audio.js index d916d93..a7ae6d5 100644 --- a/lib/synth-audio.js +++ b/lib/synth-audio.js @@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs', 'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble', - 'murf', 'xai'] + 'murf', 'xai', 'fishaudio'] .includes(vendor) || vendor.startsWith('custom'), `synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`); @@ -149,6 +149,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc } else if (vendor === 'resemble') { assert.ok(voice, 'synthAudio requires voice when resemble is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used'); + } else if ('fishaudio' === vendor) { + /* no voice assert: fish synthesizes with its own default voice when + reference_id is omitted, which is what the 'default' selection means */ + assert.ok(credentials.api_key, 'synthAudio requires api_key when fishaudio is used'); } const key = makeSynthKey({ @@ -220,6 +224,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache}); break; + case 'fishaudio': + audioData = await synthFishaudio(logger, { + credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, + disableTtsCache}); + break; case 'gradium': audioData = await synthGradium(logger, { credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, @@ -1505,6 +1514,81 @@ const synthGradium = async(logger, { /* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back (mp3 is rejected), so the cache render asks for wav rather than mp3. */ +/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache + render. format:pcm + sample_rate:8000 returns bare little-endian 16-bit samples, + which is exactly the r8 container. we avoid fish's wav output because its RIFF + header carries a placeholder size (0xffffff24) — length is unknown up front, as + with gradium. + + fish is a voice-cloning vendor: the "voice" is a reference_id returned by + POST /model, and omitting it entirely synthesizes with fish's default voice. + the sentinel value 'default' (the bundled fallback entry in the portal) means + exactly that — send no reference_id. +*/ +const synthFishaudio = async(logger, { + credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache +}) => { + const {api_key, model_id, fishaudio_tts_uri} = credentials; + const {reference_id, latency, chunk_length, speed, volume} = options || {}; + + /* free-text reference_id in the vendor options wins over the voice selector */ + const refId = reference_id || (voice && voice !== 'default' ? voice : null); + + /* default to using the streaming interface, unless disabled by env var OR we want just a cache file */ + if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) { + let params = '{'; + params += `api_key=${api_key}`; + params += `,playback_id=${key}`; + params += ',vendor=fishaudio'; + params += `,voice=${refId || 'default'}`; + params += `,write_cache_file=${disableTtsCache ? 0 : 1}`; + if (model_id) params += `,model_id=${model_id}`; + if (latency) params += `,latency=${latency}`; + if (chunk_length) params += `,chunk_length=${chunk_length}`; + if (speed) params += `,speed=${speed}`; + if (volume) params += `,volume=${volume}`; + if (fishaudio_tts_uri) params += `,endpoint=${fishaudio_tts_uri}`; + params += '}'; + + return { + filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`, + servedFromCache: false, + rtt: 0 + }; + } + + try { + const sampleRate = 8000; + const post = bent(fishaudio_tts_uri || 'https://api.fish.audio', 'POST', 'buffer', { + 'Authorization': `Bearer ${api_key}`, + 'Content-Type': 'application/json', + /* the model is selected by header, not in the body */ + 'model': model_id || 's2.1-pro' + }); + const audioContent = await post('/v1/tts', { + text, + format: 'pcm', + sample_rate: sampleRate, + ...(refId && {reference_id: refId}), + ...(latency && {latency}), + ...(chunk_length && {chunk_length: parseInt(chunk_length, 10)}), + ...((speed || volume) && {prosody: { + ...(speed && {speed: parseFloat(speed)}), + ...(volume && {volume: parseFloat(volume)}) + }}) + }); + return { + audioContent, + extension: 'r8', + sampleRate + }; + } catch (err) { + logger.info({err}, 'synth fishaudio returned error'); + stats.increment('tts.count', ['vendor:fishaudio', 'accepted:no']); + throw err; + } +}; + const synthNineninesix = async(logger, { credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache }) => { diff --git a/test/synth.js b/test/synth.js index 2fd86b5..3a59d4e 100644 --- a/test/synth.js +++ b/test/synth.js @@ -1087,6 +1087,47 @@ test('gradium speech synth tests', async(t) => { client.quit(); }); +test('fishaudio speech synth tests', async(t) => { + const fn = require('..'); + const {synthAudio, client} = fn(opts, logger); + + if (!process.env.FISHAUDIO_API_KEY) { + t.pass('skipping fishaudio speech synth tests since FISHAUDIO_API_KEY is not provided'); + return t.end(); + } + const text = 'Hi there and welcome to jambones! ' + Date.now(); + try { + /* voice 'default' means "send no reference_id" — fish's own default voice */ + const o = await synthAudio(stats, { + vendor: 'fishaudio', + credentials: { + api_key: process.env.FISHAUDIO_API_KEY, + model_id: 's2.1-pro' + }, + voice: 'default', + text, + renderForCaching: true + }); + t.ok(!o.servedFromCache, `successfully synthed fishaudio audio to ${o.filePath}`); + + /* the cache render must be raw 8k pcm (r8): fish's wav header carries a + placeholder RIFF size, so we never ask for wav */ + const o2 = await synthAudio(stats, { + vendor: 'fishaudio', + credentials: {api_key: process.env.FISHAUDIO_API_KEY}, + voice: 'default', + text: text + ' two', + renderForCaching: true, + disableTtsCache: true + }); + t.ok(!o2.servedFromCache, 'fishaudio synthed a second uncached render'); + } catch (err) { + console.error(JSON.stringify(err)); + t.end(err); + } + client.quit(); +}); + test('nineninesix speech synth tests', async(t) => { const fn = require('..'); const {synthAudio, client} = fn(opts, logger);