mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
feat: add fishaudio (Fish Audio) TTS support (#159)
Adds synthFishaudio with both arms: the say: streaming url consumed by the mediajam dialect, and a POST /v1/tts cache render. The render asks for raw pcm at 8k and returns extension r8 because fish's wav output carries a placeholder RIFF size, the same problem gradium has. Fish is a voice-cloning vendor, so the voice is a reference_id; the sentinel 'default' means send none and use fish's own default voice. Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
c8850998b5
commit
a32f71ac5a
+85
-1
@@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
|
|
||||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
|
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
|
||||||
'murf', 'xai']
|
'murf', 'xai', 'fishaudio']
|
||||||
.includes(vendor) ||
|
.includes(vendor) ||
|
||||||
vendor.startsWith('custom'),
|
vendor.startsWith('custom'),
|
||||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||||
@@ -149,6 +149,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
} else if (vendor === 'resemble') {
|
} else if (vendor === 'resemble') {
|
||||||
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
||||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
||||||
|
} else if ('fishaudio' === vendor) {
|
||||||
|
/* no voice assert: fish synthesizes with its own default voice when
|
||||||
|
reference_id is omitted, which is what the 'default' selection means */
|
||||||
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when fishaudio is used');
|
||||||
}
|
}
|
||||||
|
|
||||||
const key = makeSynthKey({
|
const key = makeSynthKey({
|
||||||
@@ -220,6 +224,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
disableTtsCache});
|
disableTtsCache});
|
||||||
break;
|
break;
|
||||||
|
case 'fishaudio':
|
||||||
|
audioData = await synthFishaudio(logger, {
|
||||||
|
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
|
disableTtsCache});
|
||||||
|
break;
|
||||||
case 'gradium':
|
case 'gradium':
|
||||||
audioData = await synthGradium(logger, {
|
audioData = await synthGradium(logger, {
|
||||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
@@ -1505,6 +1514,81 @@ const synthGradium = async(logger, {
|
|||||||
|
|
||||||
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
||||||
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
||||||
|
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
|
||||||
|
render. format:pcm + sample_rate:8000 returns bare little-endian 16-bit samples,
|
||||||
|
which is exactly the r8 container. we avoid fish's wav output because its RIFF
|
||||||
|
header carries a placeholder size (0xffffff24) — length is unknown up front, as
|
||||||
|
with gradium.
|
||||||
|
|
||||||
|
fish is a voice-cloning vendor: the "voice" is a reference_id returned by
|
||||||
|
POST /model, and omitting it entirely synthesizes with fish's default voice.
|
||||||
|
the sentinel value 'default' (the bundled fallback entry in the portal) means
|
||||||
|
exactly that — send no reference_id.
|
||||||
|
*/
|
||||||
|
const synthFishaudio = async(logger, {
|
||||||
|
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||||
|
}) => {
|
||||||
|
const {api_key, model_id, fishaudio_tts_uri} = credentials;
|
||||||
|
const {reference_id, latency, chunk_length, speed, volume} = options || {};
|
||||||
|
|
||||||
|
/* free-text reference_id in the vendor options wins over the voice selector */
|
||||||
|
const refId = reference_id || (voice && voice !== 'default' ? voice : null);
|
||||||
|
|
||||||
|
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||||
|
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '{';
|
||||||
|
params += `api_key=${api_key}`;
|
||||||
|
params += `,playback_id=${key}`;
|
||||||
|
params += ',vendor=fishaudio';
|
||||||
|
params += `,voice=${refId || 'default'}`;
|
||||||
|
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||||
|
if (model_id) params += `,model_id=${model_id}`;
|
||||||
|
if (latency) params += `,latency=${latency}`;
|
||||||
|
if (chunk_length) params += `,chunk_length=${chunk_length}`;
|
||||||
|
if (speed) params += `,speed=${speed}`;
|
||||||
|
if (volume) params += `,volume=${volume}`;
|
||||||
|
if (fishaudio_tts_uri) params += `,endpoint=${fishaudio_tts_uri}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const sampleRate = 8000;
|
||||||
|
const post = bent(fishaudio_tts_uri || 'https://api.fish.audio', 'POST', 'buffer', {
|
||||||
|
'Authorization': `Bearer ${api_key}`,
|
||||||
|
'Content-Type': 'application/json',
|
||||||
|
/* the model is selected by header, not in the body */
|
||||||
|
'model': model_id || 's2.1-pro'
|
||||||
|
});
|
||||||
|
const audioContent = await post('/v1/tts', {
|
||||||
|
text,
|
||||||
|
format: 'pcm',
|
||||||
|
sample_rate: sampleRate,
|
||||||
|
...(refId && {reference_id: refId}),
|
||||||
|
...(latency && {latency}),
|
||||||
|
...(chunk_length && {chunk_length: parseInt(chunk_length, 10)}),
|
||||||
|
...((speed || volume) && {prosody: {
|
||||||
|
...(speed && {speed: parseFloat(speed)}),
|
||||||
|
...(volume && {volume: parseFloat(volume)})
|
||||||
|
}})
|
||||||
|
});
|
||||||
|
return {
|
||||||
|
audioContent,
|
||||||
|
extension: 'r8',
|
||||||
|
sampleRate
|
||||||
|
};
|
||||||
|
} catch (err) {
|
||||||
|
logger.info({err}, 'synth fishaudio returned error');
|
||||||
|
stats.increment('tts.count', ['vendor:fishaudio', 'accepted:no']);
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
const synthNineninesix = async(logger, {
|
const synthNineninesix = async(logger, {
|
||||||
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||||
}) => {
|
}) => {
|
||||||
|
|||||||
@@ -1087,6 +1087,47 @@ test('gradium speech synth tests', async(t) => {
|
|||||||
client.quit();
|
client.quit();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('fishaudio speech synth tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.FISHAUDIO_API_KEY) {
|
||||||
|
t.pass('skipping fishaudio speech synth tests since FISHAUDIO_API_KEY is not provided');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||||
|
try {
|
||||||
|
/* voice 'default' means "send no reference_id" — fish's own default voice */
|
||||||
|
const o = await synthAudio(stats, {
|
||||||
|
vendor: 'fishaudio',
|
||||||
|
credentials: {
|
||||||
|
api_key: process.env.FISHAUDIO_API_KEY,
|
||||||
|
model_id: 's2.1-pro'
|
||||||
|
},
|
||||||
|
voice: 'default',
|
||||||
|
text,
|
||||||
|
renderForCaching: true
|
||||||
|
});
|
||||||
|
t.ok(!o.servedFromCache, `successfully synthed fishaudio audio to ${o.filePath}`);
|
||||||
|
|
||||||
|
/* the cache render must be raw 8k pcm (r8): fish's wav header carries a
|
||||||
|
placeholder RIFF size, so we never ask for wav */
|
||||||
|
const o2 = await synthAudio(stats, {
|
||||||
|
vendor: 'fishaudio',
|
||||||
|
credentials: {api_key: process.env.FISHAUDIO_API_KEY},
|
||||||
|
voice: 'default',
|
||||||
|
text: text + ' two',
|
||||||
|
renderForCaching: true,
|
||||||
|
disableTtsCache: true
|
||||||
|
});
|
||||||
|
t.ok(!o2.servedFromCache, 'fishaudio synthed a second uncached render');
|
||||||
|
} catch (err) {
|
||||||
|
console.error(JSON.stringify(err));
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
test('nineninesix speech synth tests', async(t) => {
|
test('nineninesix speech synth tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {synthAudio, client} = fn(opts, logger);
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|||||||
Reference in New Issue
Block a user