Compare commits

...
5 Commits
Author SHA1 Message Date
Dave Horton f4c3c7dc8b 1.0.18 2026-08-23 13:12:37 -04:00
Dave HortonandClaude Opus 5 a32f71ac5a feat: add fishaudio (Fish Audio) TTS support (#159)
Adds synthFishaudio with both arms: the say: streaming url consumed by the
mediajam dialect, and a POST /v1/tts cache render. The render asks for raw pcm
at 8k and returns extension r8 because fish's wav output carries a placeholder
RIFF size, the same problem gradium has.

Fish is a voice-cloning vendor, so the voice is a reference_id; the sentinel
'default' means send none and use fish's own default voice.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-23 13:11:59 -04:00
Dave HortonandClaude Opus 5 c8850998b5 fix(tts): use a current nineninesix voice id in the synth test (#158)
The vendor replaced its voice catalog and now rejects unknown ids, so
the env-gated test failed whenever NINENINESIX_API_KEY was set.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-16 12:19:59 -04:00
Dave Horton e8b2009d29 1.0.17 2026-08-10 16:45:32 -04:00
Dave HortonandClaude Opus 5 27e07bb551 fix(tts): forward gradium json_config to the streaming path (#157)
synthGradium destructured json_config from options but only forwarded
pronunciation_id into the say: param block, so the setting could never
reach mediajam's streaming dialect — only the cache-render POST. The
say: parser is brace-aware, so a nested json object survives intact.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-10 16:45:07 -04:00
4 changed files with 134 additions and 5 deletions
+89 -1
View File
@@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
'murf', 'xai']
'murf', 'xai', 'fishaudio']
.includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
@@ -149,6 +149,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
} else if (vendor === 'resemble') {
assert.ok(voice, 'synthAudio requires voice when resemble is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
} else if ('fishaudio' === vendor) {
/* no voice assert: fish synthesizes with its own default voice when
reference_id is omitted, which is what the 'default' selection means */
assert.ok(credentials.api_key, 'synthAudio requires api_key when fishaudio is used');
}
const key = makeSynthKey({
@@ -220,6 +224,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'fishaudio':
audioData = await synthFishaudio(logger, {
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'gradium':
audioData = await synthGradium(logger, {
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
@@ -1463,6 +1472,10 @@ const synthGradium = async(logger, {
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (model_id) params += `,model_id=${model_id}`;
if (pronunciation_id) params += `,pronunciation_id=${pronunciation_id}`;
/* the say: param parser is brace-aware, so a nested json object survives intact */
if (json_config) {
params += `,json_config=${typeof json_config === 'string' ? json_config : JSON.stringify(json_config)}`;
}
params += '}';
return {
@@ -1501,6 +1514,81 @@ const synthGradium = async(logger, {
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
render. format:pcm + sample_rate:8000 returns bare little-endian 16-bit samples,
which is exactly the r8 container. we avoid fish's wav output because its RIFF
header carries a placeholder size (0xffffff24) — length is unknown up front, as
with gradium.
fish is a voice-cloning vendor: the "voice" is a reference_id returned by
POST /model, and omitting it entirely synthesizes with fish's default voice.
the sentinel value 'default' (the bundled fallback entry in the portal) means
exactly that — send no reference_id.
*/
const synthFishaudio = async(logger, {
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, model_id, fishaudio_tts_uri} = credentials;
const {reference_id, latency, chunk_length, speed, volume} = options || {};
/* free-text reference_id in the vendor options wins over the voice selector */
const refId = reference_id || (voice && voice !== 'default' ? voice : null);
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=fishaudio';
params += `,voice=${refId || 'default'}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (model_id) params += `,model_id=${model_id}`;
if (latency) params += `,latency=${latency}`;
if (chunk_length) params += `,chunk_length=${chunk_length}`;
if (speed) params += `,speed=${speed}`;
if (volume) params += `,volume=${volume}`;
if (fishaudio_tts_uri) params += `,endpoint=${fishaudio_tts_uri}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const sampleRate = 8000;
const post = bent(fishaudio_tts_uri || 'https://api.fish.audio', 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Content-Type': 'application/json',
/* the model is selected by header, not in the body */
'model': model_id || 's2.1-pro'
});
const audioContent = await post('/v1/tts', {
text,
format: 'pcm',
sample_rate: sampleRate,
...(refId && {reference_id: refId}),
...(latency && {latency}),
...(chunk_length && {chunk_length: parseInt(chunk_length, 10)}),
...((speed || volume) && {prosody: {
...(speed && {speed: parseFloat(speed)}),
...(volume && {volume: parseFloat(volume)})
}})
});
return {
audioContent,
extension: 'r8',
sampleRate
};
} catch (err) {
logger.info({err}, 'synth fishaudio returned error');
stats.increment('tts.count', ['vendor:fishaudio', 'accepted:no']);
throw err;
}
};
const synthNineninesix = async(logger, {
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.16",
"version": "1.0.18",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "1.0.16",
"version": "1.0.18",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.16",
"version": "1.0.18",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
+42 -1
View File
@@ -1087,6 +1087,47 @@ test('gradium speech synth tests', async(t) => {
client.quit();
});
test('fishaudio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.FISHAUDIO_API_KEY) {
t.pass('skipping fishaudio speech synth tests since FISHAUDIO_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones! ' + Date.now();
try {
/* voice 'default' means "send no reference_id" — fish's own default voice */
const o = await synthAudio(stats, {
vendor: 'fishaudio',
credentials: {
api_key: process.env.FISHAUDIO_API_KEY,
model_id: 's2.1-pro'
},
voice: 'default',
text,
renderForCaching: true
});
t.ok(!o.servedFromCache, `successfully synthed fishaudio audio to ${o.filePath}`);
/* the cache render must be raw 8k pcm (r8): fish's wav header carries a
placeholder RIFF size, so we never ask for wav */
const o2 = await synthAudio(stats, {
vendor: 'fishaudio',
credentials: {api_key: process.env.FISHAUDIO_API_KEY},
voice: 'default',
text: text + ' two',
renderForCaching: true,
disableTtsCache: true
});
t.ok(!o2.servedFromCache, 'fishaudio synthed a second uncached render');
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('nineninesix speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -1104,7 +1145,7 @@ test('nineninesix speech synth tests', async(t) => {
model_id: 'gepard-1.0'
},
language: 'en',
voice: '3ad7a827-7fd1-4954-bf35-47d4cc33d9ed',
voice: '775e4dfd-819c-4325-92ec-250f487ef7e3',
text,
renderForCaching: true
});