From 6c67f6faa0292d5cfb828d9891fee5331edc5904 Mon Sep 17 00:00:00 2001 From: Hoan Luu Huu <110280845+xquanluu@users.noreply.github.com> Date: Thu, 1 Oct 2026 10:01:18 +0700 Subject: [PATCH] feat(tts): add speechify as a TTS vendor (#165) - streaming returns a say: url for the mediajam starter - cache render POSTs /v1/audio/stream for raw 24 kHz PCM (r24) - credential options merge under the verb's options - no model: simba-3.2 for English, simba-3.0 otherwise Co-authored-by: Claude Opus 5.5 --- lib/synth-audio.js | 88 +++++++++++++++++++++++++++++++++++++++++++++- test/synth.js | 29 +++++++++++++++ 2 files changed, 116 insertions(+), 1 deletion(-) diff --git a/lib/synth-audio.js b/lib/synth-audio.js index 9bb1496..42eb3bf 100644 --- a/lib/synth-audio.js +++ b/lib/synth-audio.js @@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs', 'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld', - 'resemble', 'murf', 'xai', 'fishaudio'] + 'resemble', 'murf', 'xai', 'fishaudio', 'speechify'] .includes(vendor) || vendor.startsWith('custom'), `synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`); @@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc } else if ('gradium' === vendor) { assert.ok(voice, 'synthAudio requires voice when gradium is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used'); + } else if ('speechify' === vendor) { + assert.ok(voice, 'synthAudio requires voice when speechify is used'); + assert.ok(credentials.api_key, 'synthAudio requires api_key when speechify is used'); } else if ('kugelaudio' === vendor) { assert.ok(voice, 'synthAudio requires voice when kugelaudio is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used'); @@ -242,6 +245,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache}); break; + case 'speechify': + audioData = await synthSpeechify(logger, { + credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, + disableTtsCache}); + break; case 'nineninesix': audioData = await synthNineninesix(logger, { credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, @@ -1602,6 +1610,84 @@ const synthKugelaudio = async(logger, { } }; +/* simba-3.2 is English only; with no model set, other languages get simba-3.0 */ +const speechifyModel = (model_id, language) => { + if (model_id) return model_id; + return !language || /^en/i.test(language) ? 'simba-3.2' : 'simba-3.0'; +}; +const synthSpeechify = async(logger, { + credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, + disableTtsCache +}) => { + const {api_key, model_id} = credentials; + let credOptions = credentials.options || {}; + if (typeof credOptions === 'string') { + try { + credOptions = JSON.parse(credOptions); + } catch { + credOptions = {}; + } + } + const {api_uri, loudness_normalization, text_normalization} = {...credOptions, ...options}; + const isSet = (v) => v !== null && v !== undefined; + const model = speechifyModel(options?.model_id || model_id, language); + + /* default to using the streaming interface, unless disabled by env var OR we want just a cache file */ + if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) { + let params = '{'; + params += `api_key=${api_key}`; + params += `,playback_id=${key}`; + params += ',vendor=speechify'; + params += `,voice=${voice}`; + params += `,write_cache_file=${disableTtsCache ? 0 : 1}`; + params += `,model_id=${model}`; + if (language) params += `,language=${language}`; + if (api_uri) params += `,api_uri=${api_uri}`; + if (isSet(loudness_normalization)) params += `,loudness_normalization=${loudness_normalization}`; + if (isSet(text_normalization)) params += `,text_normalization=${text_normalization}`; + params += '}'; + + return { + filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`, + servedFromCache: false, + rtt: 0 + }; + } + + try { + /* other pcm rates are mislabelled 24 kHz on workspaces pinned before 2026-09-30 */ + const sampleRate = 24000; + const host = (api_uri || 'api.speechify.ai').replace(/^[a-z]+:\/\//, '').replace(/\/$/, ''); + const post = bent(`https://${host}`, 'POST', 'buffer', { + 'Authorization': `Bearer ${api_key}`, + 'Content-Type': 'application/json', + 'Speechify-Caller': 'jambonz' + }); + const toBool = (v) => v === true || v === 'true'; + const speechOptions = { + ...(isSet(loudness_normalization) && {loudness_normalization: toBool(loudness_normalization)}), + ...(isSet(text_normalization) && {text_normalization: toBool(text_normalization)}) + }; + const audioContent = await post('/v1/audio/stream', { + input: text, + voice_id: voice, + model, + output_format: `pcm_${sampleRate}`, + ...(language && {language}), + ...(Object.keys(speechOptions).length && {options: speechOptions}) + }); + return { + audioContent, + extension: 'r24', + sampleRate + }; + } catch (err) { + logger.info({err}, 'synth speechify returned error'); + stats.increment('tts.count', ['vendor:speechify', 'accepted:no']); + throw err; + } +}; + /* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back (mp3 is rejected), so the cache render asks for wav rather than mp3. */ /* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache diff --git a/test/synth.js b/test/synth.js index 81cffde..00e3fb1 100644 --- a/test/synth.js +++ b/test/synth.js @@ -1120,6 +1120,35 @@ test('kugelaudio speech synth tests', async(t) => { client.quit(); }); +test('speechify speech synth tests', async(t) => { + const fn = require('..'); + const {synthAudio, client} = fn(opts, logger); + + if (!process.env.SPEECHIFY_API_KEY) { + t.pass('skipping speechify speech synth tests since SPEECHIFY_API_KEY is not provided'); + return t.end(); + } + const text = 'Hi there and welcome to jambonz! ' + Date.now(); + try { + const opts = await synthAudio(stats, { + vendor: 'speechify', + credentials: { + api_key: process.env.SPEECHIFY_API_KEY + }, + language: 'en-US', + voice: 'geffen_32', + text, + renderForCaching: true + }); + t.ok(!opts.servedFromCache, `successfully synthed speechify audio to ${opts.filePath}`); + + } catch (err) { + console.error(JSON.stringify(err)); + t.end(err); + } + client.quit(); +}); + test('fishaudio speech synth tests', async(t) => { const fn = require('..'); const {synthAudio, client} = fn(opts, logger);