feat(tts): add kugelaudio as a TTS vendor (#163)

Streaming returns a say: url for the mediajam dialect; the cache render POSTs
/v1/tts/generate for raw 8 kHz PCM (r8). Language is reduced to its primary
subtag and dropped when KugelAudio does not support it.

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Dave Horton
2026-09-29 14:21:41 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 89fe0b87e9
commit fa705694e1
2 changed files with 122 additions and 2 deletions
+92 -2
View File
@@ -80,8 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
'murf', 'xai', 'fishaudio']
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld',
'resemble', 'murf', 'xai', 'fishaudio']
.includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
@@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
} else if ('gradium' === vendor) {
assert.ok(voice, 'synthAudio requires voice when gradium is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
} else if ('kugelaudio' === vendor) {
assert.ok(voice, 'synthAudio requires voice when kugelaudio is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used');
} else if ('nineninesix' === vendor) {
assert.ok(voice, 'synthAudio requires voice when nineninesix is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when nineninesix is used');
@@ -234,6 +237,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'kugelaudio':
audioData = await synthKugelaudio(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'nineninesix':
audioData = await synthNineninesix(logger, {
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
@@ -1512,6 +1520,88 @@ const synthGradium = async(logger, {
}
};
/* kugelaudio — json websocket (/ws/tts/stream) for streaming, and POST /v1/tts/generate
for the cache render. the POST streams back bare little-endian 16-bit samples at the
requested sample_rate, which is exactly the r8 container at 8000.
voices are numeric ids (or public handles). language is an ISO 639-1 code that drives
text normalization; jambonz carries BCP-47, so only the primary subtag is sent. the
api rejects codes outside its list, so an unsupported or unset language is omitted
and the voice's own language applies. options.api_uri pins a region
(e.g. api.eu.kugelaudio.com).
*/
const KUGELAUDIO_LANGUAGES = ['ar', 'bg', 'bn', 'cs', 'da', 'de', 'el', 'en', 'es', 'fa', 'fi', 'fr', 'he', 'hi',
'hr', 'hu', 'id', 'it', 'ja', 'ko', 'ms', 'nl', 'no', 'pl', 'pt', 'ro', 'ru', 'sk', 'sl', 'sr', 'sv', 'ta', 'th',
'tr', 'uk', 'ur', 'vi', 'yue', 'zh'];
const synthKugelaudio = async(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache
}) => {
const {api_key, model_id} = credentials;
const {api_uri, speed, cfg_scale, temperature, normalize, project_id, dictionary_ids} = options || {};
const isSet = (v) => v !== null && v !== undefined;
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=kugelaudio';
params += `,voice=${voice}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += `,model_id=${model_id || 'kugel-3'}`;
if (language) params += `,language=${language}`;
if (api_uri) params += `,api_uri=${api_uri}`;
if (isSet(speed)) params += `,speed=${speed}`;
if (isSet(cfg_scale)) params += `,cfg_scale=${cfg_scale}`;
if (isSet(temperature)) params += `,temperature=${temperature}`;
if (isSet(normalize)) params += `,normalize=${normalize}`;
if (isSet(project_id)) params += `,project_id=${project_id}`;
/* the say: param parser is bracket-aware, so the json array survives intact */
if (Array.isArray(dictionary_ids)) params += `,dictionary_ids=${JSON.stringify(dictionary_ids)}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const sampleRate = 8000;
const host = (api_uri || 'api.kugelaudio.com').replace(/^[a-z]+:\/\//, '').replace(/\/$/, '');
const post = bent(`https://${host}`, 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Content-Type': 'application/json; charset=utf-8'
});
const voiceId = /^\d+$/.test(`${voice}`) ? Number(voice) : voice;
const lang = language && language.split('-')[0].toLowerCase();
const audioContent = await post('/v1/tts/generate', {
text,
voice_id: voiceId,
model_id: model_id || 'kugel-3',
sample_rate: sampleRate,
...(KUGELAUDIO_LANGUAGES.includes(lang) && {language: lang}),
...(isSet(speed) && {speed: Number(speed)}),
...(isSet(cfg_scale) && {cfg_scale: Number(cfg_scale)}),
...(isSet(temperature) && {temperature: Number(temperature)}),
...(isSet(normalize) && {normalize: normalize === true || normalize === 'true'}),
...(isSet(project_id) && {project_id: Number(project_id)}),
...(Array.isArray(dictionary_ids) && {dictionary_ids})
});
return {
audioContent,
extension: 'r8',
sampleRate
};
} catch (err) {
logger.info({err}, 'synth kugelaudio returned error');
stats.increment('tts.count', ['vendor:kugelaudio', 'accepted:no']);
throw err;
}
};
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
+30
View File
@@ -1090,6 +1090,36 @@ test('gradium speech synth tests', async(t) => {
client.quit();
});
test('kugelaudio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.KUGELAUDIO_API_KEY) {
t.pass('skipping kugelaudio speech synth tests since KUGELAUDIO_API_KEY is not provided');
return t.end();
}
const text = 'Guten Tag und willkommen bei jambonz! Ihre Bestellung kostet 12,99 Euro. ' + Date.now();
try {
const opts = await synthAudio(stats, {
vendor: 'kugelaudio',
credentials: {
api_key: process.env.KUGELAUDIO_API_KEY,
model_id: 'kugel-3'
},
language: 'de-DE',
voice: '1930',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthed kugelaudio audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('fishaudio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);