Compare commits

...
Author SHA1 Message Date
Dave Horton 9694714a7f 1.0.10 2026-07-28 20:59:05 -04:00
Hoan Luu Huu 6a6de9b7a1 support google tts configuraiton (#149) 2026-07-28 20:58:26 -04:00
Dave Horton 04d1fb548b 1.0.9 2026-07-21 06:52:05 -04:00
Hoan Luu Huu bbf0167b40 support deepgramflux tts (#148) 2026-07-21 06:49:38 -04:00
Dave Horton 4cfa286242 1.0.8 2026-07-05 17:42:09 -04:00
Hoan Luu Huu 42ac63adc6 support xaiTTS (#147) 2026-07-05 17:41:36 -04:00
Dave Horton 12a121672d 1.0.7 2026-07-03 07:12:20 -04:00
Hoan Luu HuuandClaude Opus 4.8 7b94a5a969 feat(murf): add Murf.ai TTS support to synthAudio (#146)
* feat(murf): add Murf.ai TTS support to synthAudio

Add synthMurf() following the rimelabs/cartesia pattern:
- streaming path returns a say:{vendor=murf,...} filePath consumed by the
  FreeSWITCH mod_murf_tts module
- non-streaming path calls POST /v1/speech/stream (api-key header) and returns
  WAV audio for cache rendering

Register murf in the supported-vendor assert list and the synth switch.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(murf): drop Accept: audio/basic header (caused 406 Not Acceptable)

Murf's /v1/speech/stream rejects an unmatched Accept header with 406; the
response container is chosen by the `format` body field instead. Verified a
WAV request now returns 200 (valid RIFF/WAVE).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-03 07:11:39 -04:00
4 changed files with 431 additions and 4 deletions
+240 -1
View File
@@ -80,7 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
'whisper', 'deepgram', 'rimelabs', 'cartesia', 'inworld', 'resemble'].includes(vendor) ||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'inworld', 'resemble', 'murf', 'xai']
.includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
if ('google' === vendor) {
@@ -124,9 +125,19 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
if (!credentials.deepgram_tts_uri) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
}
} else if ('deepgramflux' === vendor) {
// Deepgram Flux TTS (/v2/speak); the flux model rides on `voice`/`model`
if (!credentials.deepgram_tts_uri) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgramflux is used');
}
} else if ('xai' === vendor) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when xai is used');
} else if ('cartesia' === vendor) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when cartesia is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when cartesia is used');
} else if ('murf' === vendor) {
assert.ok(voice, 'synthAudio requires voice when murf is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when murf is used');
} else if (vendor === 'resemble') {
assert.ok(voice, 'synthAudio requires voice when resemble is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
@@ -211,6 +222,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'murf':
audioData = await synthMurf(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'whisper':
audioData = await synthWhisper(logger, {
credentials, stats, voice, key, text, instructions, renderForCaching, disableTtsStreaming,
@@ -220,6 +236,15 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
audioData = await synthDeepgram(logger, {credentials, stats, model, key, text,
renderForCaching, disableTtsStreaming, disableTtsCache});
break;
case 'deepgramflux':
audioData = await synthDeepgramFlux(logger, {credentials, stats, model: model || voice, key, text,
renderForCaching, disableTtsStreaming, disableTtsCache});
break;
case 'xai':
audioData = await synthXai(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'resemble':
audioData = await synthResemble(logger, {
credentials, stats, voice, key, text, options, renderForCaching, disableTtsStreaming, disableTtsCache});
@@ -358,6 +383,32 @@ const synthPolly = async(createHash, retrieveHash, logger,
}
};
/* google AudioConfig settings we support, as [google camelCase name, freeswitch param name] */
const GOOGLE_AUDIO_SETTINGS = [
['speakingRate', 'speaking_rate'],
['pitch', 'pitch'],
['volumeGainDb', 'volume_gain_db']
];
/**
* Extract google AudioConfig settings from the synthesizer options. They may be supplied
* nested under an audioConfig property (mirroring google's AudioConfig object) or flat at
* the top level, and either google's camelCase or snake_case names are accepted.
* @see https://cloud.google.com/text-to-speech/docs/reference/rest/v1/text/synthesize#AudioConfig
* @returns object keyed by google's camelCase names, holding only valid numeric settings
*/
const googleAudioConfig = (options) => {
const provided = {...options, ...(options?.audioConfig || {})};
const audioConfig = {};
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
const value = provided[name] ?? provided[snakeName];
/* note: 0 is meaningful for pitch and volumeGainDb, so check for absence explicitly */
if (value === undefined || value === null || value === '') continue;
const num = Number(value);
if (Number.isFinite(num)) audioConfig[name] = num;
}
return audioConfig;
};
const synthGoogle = async(logger, {
credentials, stats, language, voice, gender, key, text, model, options, instructions,
@@ -393,6 +444,15 @@ const synthGoogle = async(logger, {
// comma is used to separate parameters in freeswitch tts module
const prompt = options?.prompt || instructions;
if (prompt) params += `,prompt=${prompt.replace(/\n/g, ' ').replace(/,/g, ';')}`;
/**
* AudioConfig settings. Note google only honors these in some api modes:
* tts applies all of them, live (HD voices) applies speakingRate only,
* and gemini ignores them entirely (use prompt instead for style control).
*/
const audioSettings = googleAudioConfig(options);
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
if (name in audioSettings) params += `,${snakeName}=${audioSettings[name]}`;
}
params += '}';
return {
@@ -461,6 +521,9 @@ const synthGoogle = async(logger, {
sampleRate = 8000;
}
/* gemini voices do not support the AudioConfig settings; they use prompt for style control */
if (!isGemini) Object.assign(audioConfig, googleAudioConfig(options));
const opts = { input, voice: voiceParams, audioConfig };
try {
@@ -969,6 +1032,72 @@ const synthRimelabs = async(logger, {
throw err;
}
};
const synthMurf = async(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, model_id, api_uri, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
/* param keys here must match mod_murf_tts's text_param handler */
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=murf';
params += `,voice=${voice}`;
if (model_id) params += `,model_id=${model_id}`;
if (language) params += `,language=${language}`;
if (api_uri) params += `,api_uri=${api_uri}`;
if (opts.style) params += `,style=${opts.style}`;
if (opts.rate !== undefined && opts.rate !== null) params += `,rate=${opts.rate}`;
if (opts.pitch !== undefined && opts.pitch !== null) params += `,pitch=${opts.pitch}`;
if (opts.variation !== undefined && opts.variation !== null) params += `,variation=${opts.variation}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const sampleRate = 8000;
/* no Accept header: murf returns 406 if it doesn't match; the response
container is selected by the `format` field in the body instead */
const post = bent(api_uri || 'https://global.api.murf.ai', 'POST', 'buffer', {
'api-key': api_key,
'Content-Type': 'application/json'
});
/* murf REST schema is documented loosely; field names follow the SDK params
(voice_id/model/format/sample_rate) plus the websocket voice fields. */
const audioContent = await post('/v1/speech/stream', {
text,
voice_id: voice,
...(model_id && {model: model_id}),
...(language && {locale: language}),
...(opts.style && {style: opts.style}),
...(opts.rate !== undefined && opts.rate !== null && {rate: opts.rate}),
...(opts.pitch !== undefined && opts.pitch !== null && {pitch: opts.pitch}),
...(opts.variation !== undefined && opts.variation !== null && {variation: opts.variation}),
format: 'WAV',
sample_rate: sampleRate,
channel_type: 'MONO'
});
return {
audioContent,
extension: 'wav',
sampleRate
};
} catch (err) {
logger.info({err}, 'synth murf returned error');
stats.increment('tts.count', ['vendor:murf', 'accepted:no']);
throw err;
}
};
const synthWhisper = async(logger, {credentials, stats, voice, key, text, instructions,
renderForCaching, disableTtsStreaming, disableTtsCache}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
@@ -1059,6 +1188,116 @@ const synthDeepgram = async(logger, {credentials, stats, model, key, text, rende
}
};
// Deepgram Flux TTS — the conversation-native model served from /v2/speak.
// Streaming rides the mediajam deepgramflux dialect via a say: filePath; the
// batch/cache path POSTs to /v2/speak (mp3 is batch-only for Flux).
const synthDeepgramFlux = async(logger, {credentials, stats, model, key, text, renderForCaching,
disableTtsStreaming, disableTtsCache}) => {
const {api_key, deepgram_tts_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=deepgramflux';
params += `,voice=${model}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent(deepgram_tts_uri || 'https://api.deepgram.com', 'POST', 'buffer', {
// on-premise deepgram does not require to have api_key
...(api_key && {'Authorization': `Token ${api_key}`}),
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const audioContent = await post(`/v2/speak?model=${model}`, {
text
});
return {
audioContent,
extension: 'mp3',
sampleRate: 8000
};
} catch (err) {
logger.info({err}, 'synth Deepgram Flux returned error');
stats.increment('tts.count', ['vendor:deepgramflux', 'accepted:no']);
throw err;
}
};
const synthXai = async(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, api_uri, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
const speed = opts.speed;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=xai';
if (voice) params += `,voice=${voice}`;
if (language) params += `,language=${language}`;
if (speed !== null && speed !== undefined) params += `,speed=${speed}`;
if (opts.optimize_streaming_latency != null) {
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency}`;
}
if (opts.text_normalization != null) params += `,text_normalization=${opts.text_normalization}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (api_uri) params += `,endpoint=${api_uri}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent(`https://${api_uri || 'api.x.ai'}`, 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Content-Type': 'application/json'
});
const audioContent = await post('/v1/tts', {
text,
language: language || 'auto',
...(voice && {voice_id: voice}),
...(speed !== null && speed !== undefined && {speed}),
...(opts.optimize_streaming_latency != null && {optimize_streaming_latency: opts.optimize_streaming_latency}),
...(opts.text_normalization != null && {text_normalization: opts.text_normalization}),
output_format: {
codec: 'wav',
sample_rate: 8000
}
});
return {
audioContent,
extension: 'wav',
sampleRate: 8000
};
} catch (err) {
// xAI errors are JSON {code, error} - read the body so the surfaced message isn't 'undefined'
if (err.name === 'StatusError' && typeof err.text === 'function') {
try {
const body = await err.text();
if (body) err.message = body;
} catch (readErr) {
logger.info({readErr}, 'synth xai: failed to read error response body');
}
}
logger.info({err}, 'synth xai returned error');
stats.increment('tts.count', ['vendor:xai', 'accepted:no']);
throw err;
}
};
const synthCartesia = async(logger, {
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.6",
"version": "1.0.10",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "1.0.6",
"version": "1.0.10",
"license": "MIT",
"dependencies": {
"23": "^0.0.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.6",
"version": "1.0.10",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
+188
View File
@@ -442,6 +442,95 @@ test('Google TTS streaming tests (!JAMBONES_DISABLE_TTS_STREAMING)', async(t) =>
});
t.ok(result.filePath.includes('api_mode=tts'), 'options.apiMode=tts overrides HD voice default');
/* AudioConfig settings (speakingRate, pitch, volumeGainDb) */
const googleCreds = {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
};
// Test 9: AudioConfig nested under options.audioConfig, as in google's API docs
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'Testing nested audioConfig settings.',
options: { audioConfig: { speakingRate: 1.4, pitch: -2.5, volumeGainDb: 6 } },
disableTtsCache: true
});
t.ok(result.filePath.includes(',speaking_rate=1.4'), 'nested audioConfig sets speaking_rate');
t.ok(result.filePath.includes(',pitch=-2.5'), 'nested audioConfig sets pitch');
t.ok(result.filePath.includes(',volume_gain_db=6'), 'nested audioConfig sets volume_gain_db');
// Test 10: AudioConfig flat at the top level of options, in camelCase or snake_case
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'Testing flat audioConfig settings.',
options: { speaking_rate: '0.8', volumeGainDb: 3 },
disableTtsCache: true
});
t.ok(result.filePath.includes(',speaking_rate=0.8'), 'flat snake_case speaking_rate is honored');
t.ok(result.filePath.includes(',volume_gain_db=3'), 'flat camelCase volumeGainDb is honored');
t.ok(!result.filePath.includes(',pitch='), 'unspecified audioConfig setting is omitted');
// Test 11: zero is a meaningful value for pitch and volumeGainDb, not an absent one
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'Testing zero audioConfig settings.',
options: { audioConfig: { pitch: 0, volumeGainDb: 0 } },
disableTtsCache: true
});
t.ok(result.filePath.includes(',pitch=0'), 'pitch=0 is passed through rather than dropped');
t.ok(result.filePath.includes(',volume_gain_db=0'), 'volume_gain_db=0 is passed through rather than dropped');
// Test 12: non-numeric values are ignored, so they cannot corrupt the freeswitch param string
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'Testing invalid audioConfig settings.',
options: { audioConfig: { speakingRate: 'fast,evil=1', pitch: null, volumeGainDb: '' } },
disableTtsCache: true
});
t.ok(!result.filePath.includes('speaking_rate'), 'non-numeric speakingRate is ignored');
t.ok(!result.filePath.includes('evil=1'), 'non-numeric value cannot inject extra params');
t.ok(!result.filePath.includes('pitch='), 'null pitch is ignored');
t.ok(!result.filePath.includes('volume_gain_db'), 'empty volumeGainDb is ignored');
// Test 13: no audioConfig supplied leaves the param string untouched
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'Testing absent audioConfig settings.',
disableTtsCache: true
});
t.ok(!/speaking_rate|pitch=|volume_gain_db/.test(result.filePath),
'no audioConfig params are added when none are supplied');
// Test 14: HD voice (api_mode=live) also carries speaking_rate, the one setting google streams support
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Chirp3-HD-Charon',
text: 'Testing audioConfig on an HD voice.',
options: { audioConfig: { speakingRate: 1.25 } },
disableTtsCache: true
});
t.ok(result.filePath.includes('api_mode=live'), 'HD voice with audioConfig still uses api_mode=live');
t.ok(result.filePath.includes(',speaking_rate=1.25'), 'HD voice streaming path carries speaking_rate');
} catch (err) {
console.error(err);
t.end(err);
@@ -526,6 +615,42 @@ test('Google TTS non-streaming tests (JAMBONES_DISABLE_TTS_STREAMING=true)', asy
t.ok(!result.filePath.startsWith('say:'), 'Gemini TTS does NOT return streaming say: path when disabled');
t.ok(result.filePath.endsWith('.mp3'), 'Gemini TTS returns mp3 file path');
const googleCreds = {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
};
/**
* Test 4: AudioConfig settings are accepted by the synthesize API.
* Google rejects out-of-range values with a 400, so a successful render also confirms
* the settings reached the request rather than being silently dropped.
*/
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'This is a test of audioConfig with streaming disabled.',
options: { audioConfig: { speakingRate: 1.4, pitch: -2.5, volumeGainDb: 6 } },
disableTtsCache: true
});
t.ok(result.filePath.endsWith('.mp3'), 'standard voice renders mp3 with audioConfig settings applied');
/* Test 5: gemini voices ignore the AudioConfig settings rather than failing on them */
result = await synthAudio(stats, {
vendor: 'google',
credentials: googleCreds,
language: 'en-US',
voice: 'Kore',
model: geminiModel,
text: 'This is a test of audioConfig on Gemini TTS.',
options: { audioConfig: { speakingRate: 1.4, pitch: -2.5, volumeGainDb: 6 } },
disableTtsCache: true
});
t.ok(result.filePath.endsWith('.mp3'), 'gemini voice renders mp3 with audioConfig settings skipped');
} catch (err) {
console.error(err);
t.end(err);
@@ -1118,6 +1243,69 @@ test('Deepgram speech synth tests', async(t) => {
client.quit();
})
test('Deepgram Flux speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.DEEPGRAM_API_KEY) {
t.pass('skipping Deepgram Flux speech synth tests since DEEPGRAM_API_KEY');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
const opts = await synthAudio(stats, {
vendor: 'deepgramflux',
credentials: {
api_key: process.env.DEEPGRAM_API_KEY
},
model: process.env.DEEPGRAM_FLUX_MODEL || 'flux-alexis-en',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized deepgramflux audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('xai speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.XAI_API_KEY) {
t.pass('skipping xai speech synth tests - no XAI_API_KEY');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
const opts = await synthAudio(stats, {
vendor: 'xai',
credentials: {
api_key: process.env.XAI_API_KEY,
options: JSON.stringify({
voice: process.env.XAI_VOICE || 'eve',
speed: 1.0,
optimize_streaming_latency: 1,
text_normalization: true
})
},
language: 'en',
voice: process.env.XAI_VOICE || 'eve',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache && opts.filePath, `successfully synthesized xai audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('TTS Cache tests', async(t) => {
const fn = require('..');
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);