diff --git a/lib/synth-audio.js b/lib/synth-audio.js index a7ae6d5..9bb1496 100644 --- a/lib/synth-audio.js +++ b/lib/synth-audio.js @@ -80,8 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc logger = logger || noopLogger; assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs', - 'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble', - 'murf', 'xai', 'fishaudio'] + 'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld', + 'resemble', 'murf', 'xai', 'fishaudio'] .includes(vendor) || vendor.startsWith('custom'), `synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`); @@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc } else if ('gradium' === vendor) { assert.ok(voice, 'synthAudio requires voice when gradium is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used'); + } else if ('kugelaudio' === vendor) { + assert.ok(voice, 'synthAudio requires voice when kugelaudio is used'); + assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used'); } else if ('nineninesix' === vendor) { assert.ok(voice, 'synthAudio requires voice when nineninesix is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when nineninesix is used'); @@ -234,6 +237,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache}); break; + case 'kugelaudio': + audioData = await synthKugelaudio(logger, { + credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, + disableTtsCache}); + break; case 'nineninesix': audioData = await synthNineninesix(logger, { credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, @@ -1512,6 +1520,88 @@ const synthGradium = async(logger, { } }; +/* kugelaudio — json websocket (/ws/tts/stream) for streaming, and POST /v1/tts/generate + for the cache render. the POST streams back bare little-endian 16-bit samples at the + requested sample_rate, which is exactly the r8 container at 8000. + + voices are numeric ids (or public handles). language is an ISO 639-1 code that drives + text normalization; jambonz carries BCP-47, so only the primary subtag is sent. the + api rejects codes outside its list, so an unsupported or unset language is omitted + and the voice's own language applies. options.api_uri pins a region + (e.g. api.eu.kugelaudio.com). +*/ +const KUGELAUDIO_LANGUAGES = ['ar', 'bg', 'bn', 'cs', 'da', 'de', 'el', 'en', 'es', 'fa', 'fi', 'fr', 'he', 'hi', + 'hr', 'hu', 'id', 'it', 'ja', 'ko', 'ms', 'nl', 'no', 'pl', 'pt', 'ro', 'ru', 'sk', 'sl', 'sr', 'sv', 'ta', 'th', + 'tr', 'uk', 'ur', 'vi', 'yue', 'zh']; +const synthKugelaudio = async(logger, { + credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, + disableTtsCache +}) => { + const {api_key, model_id} = credentials; + const {api_uri, speed, cfg_scale, temperature, normalize, project_id, dictionary_ids} = options || {}; + const isSet = (v) => v !== null && v !== undefined; + + /* default to using the streaming interface, unless disabled by env var OR we want just a cache file */ + if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) { + let params = '{'; + params += `api_key=${api_key}`; + params += `,playback_id=${key}`; + params += ',vendor=kugelaudio'; + params += `,voice=${voice}`; + params += `,write_cache_file=${disableTtsCache ? 0 : 1}`; + params += `,model_id=${model_id || 'kugel-3'}`; + if (language) params += `,language=${language}`; + if (api_uri) params += `,api_uri=${api_uri}`; + if (isSet(speed)) params += `,speed=${speed}`; + if (isSet(cfg_scale)) params += `,cfg_scale=${cfg_scale}`; + if (isSet(temperature)) params += `,temperature=${temperature}`; + if (isSet(normalize)) params += `,normalize=${normalize}`; + if (isSet(project_id)) params += `,project_id=${project_id}`; + /* the say: param parser is bracket-aware, so the json array survives intact */ + if (Array.isArray(dictionary_ids)) params += `,dictionary_ids=${JSON.stringify(dictionary_ids)}`; + params += '}'; + + return { + filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`, + servedFromCache: false, + rtt: 0 + }; + } + + try { + const sampleRate = 8000; + const host = (api_uri || 'api.kugelaudio.com').replace(/^[a-z]+:\/\//, '').replace(/\/$/, ''); + const post = bent(`https://${host}`, 'POST', 'buffer', { + 'Authorization': `Bearer ${api_key}`, + 'Content-Type': 'application/json; charset=utf-8' + }); + const voiceId = /^\d+$/.test(`${voice}`) ? Number(voice) : voice; + const lang = language && language.split('-')[0].toLowerCase(); + const audioContent = await post('/v1/tts/generate', { + text, + voice_id: voiceId, + model_id: model_id || 'kugel-3', + sample_rate: sampleRate, + ...(KUGELAUDIO_LANGUAGES.includes(lang) && {language: lang}), + ...(isSet(speed) && {speed: Number(speed)}), + ...(isSet(cfg_scale) && {cfg_scale: Number(cfg_scale)}), + ...(isSet(temperature) && {temperature: Number(temperature)}), + ...(isSet(normalize) && {normalize: normalize === true || normalize === 'true'}), + ...(isSet(project_id) && {project_id: Number(project_id)}), + ...(Array.isArray(dictionary_ids) && {dictionary_ids}) + }); + return { + audioContent, + extension: 'r8', + sampleRate + }; + } catch (err) { + logger.info({err}, 'synth kugelaudio returned error'); + stats.increment('tts.count', ['vendor:kugelaudio', 'accepted:no']); + throw err; + } +}; + /* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back (mp3 is rejected), so the cache render asks for wav rather than mp3. */ /* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache diff --git a/test/synth.js b/test/synth.js index c2d1ef6..81cffde 100644 --- a/test/synth.js +++ b/test/synth.js @@ -1090,6 +1090,36 @@ test('gradium speech synth tests', async(t) => { client.quit(); }); +test('kugelaudio speech synth tests', async(t) => { + const fn = require('..'); + const {synthAudio, client} = fn(opts, logger); + + if (!process.env.KUGELAUDIO_API_KEY) { + t.pass('skipping kugelaudio speech synth tests since KUGELAUDIO_API_KEY is not provided'); + return t.end(); + } + const text = 'Guten Tag und willkommen bei jambonz! Ihre Bestellung kostet 12,99 Euro. ' + Date.now(); + try { + const opts = await synthAudio(stats, { + vendor: 'kugelaudio', + credentials: { + api_key: process.env.KUGELAUDIO_API_KEY, + model_id: 'kugel-3' + }, + language: 'de-DE', + voice: '1930', + text, + renderForCaching: true + }); + t.ok(!opts.servedFromCache, `successfully synthed kugelaudio audio to ${opts.filePath}`); + + } catch (err) { + console.error(JSON.stringify(err)); + t.end(err); + } + client.quit(); +}); + test('fishaudio speech synth tests', async(t) => { const fn = require('..'); const {synthAudio, client} = fn(opts, logger);