mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
04d1fb548b | ||
|
|
bbf0167b40 | ||
|
|
4cfa286242 | ||
|
|
42ac63adc6 | ||
|
|
12a121672d | ||
|
|
7b94a5a969 |
+202
-1
@@ -80,7 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
logger = logger || noopLogger;
|
logger = logger || noopLogger;
|
||||||
|
|
||||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||||
'whisper', 'deepgram', 'rimelabs', 'cartesia', 'inworld', 'resemble'].includes(vendor) ||
|
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'inworld', 'resemble', 'murf', 'xai']
|
||||||
|
.includes(vendor) ||
|
||||||
vendor.startsWith('custom'),
|
vendor.startsWith('custom'),
|
||||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||||
if ('google' === vendor) {
|
if ('google' === vendor) {
|
||||||
@@ -124,9 +125,19 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
if (!credentials.deepgram_tts_uri) {
|
if (!credentials.deepgram_tts_uri) {
|
||||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
|
||||||
}
|
}
|
||||||
|
} else if ('deepgramflux' === vendor) {
|
||||||
|
// Deepgram Flux TTS (/v2/speak); the flux model rides on `voice`/`model`
|
||||||
|
if (!credentials.deepgram_tts_uri) {
|
||||||
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgramflux is used');
|
||||||
|
}
|
||||||
|
} else if ('xai' === vendor) {
|
||||||
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when xai is used');
|
||||||
} else if ('cartesia' === vendor) {
|
} else if ('cartesia' === vendor) {
|
||||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when cartesia is used');
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when cartesia is used');
|
||||||
assert.ok(credentials.model_id, 'synthAudio requires model_id when cartesia is used');
|
assert.ok(credentials.model_id, 'synthAudio requires model_id when cartesia is used');
|
||||||
|
} else if ('murf' === vendor) {
|
||||||
|
assert.ok(voice, 'synthAudio requires voice when murf is used');
|
||||||
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when murf is used');
|
||||||
} else if (vendor === 'resemble') {
|
} else if (vendor === 'resemble') {
|
||||||
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
||||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
||||||
@@ -211,6 +222,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
disableTtsCache});
|
disableTtsCache});
|
||||||
break;
|
break;
|
||||||
|
case 'murf':
|
||||||
|
audioData = await synthMurf(logger, {
|
||||||
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
|
disableTtsCache});
|
||||||
|
break;
|
||||||
case 'whisper':
|
case 'whisper':
|
||||||
audioData = await synthWhisper(logger, {
|
audioData = await synthWhisper(logger, {
|
||||||
credentials, stats, voice, key, text, instructions, renderForCaching, disableTtsStreaming,
|
credentials, stats, voice, key, text, instructions, renderForCaching, disableTtsStreaming,
|
||||||
@@ -220,6 +236,15 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
audioData = await synthDeepgram(logger, {credentials, stats, model, key, text,
|
audioData = await synthDeepgram(logger, {credentials, stats, model, key, text,
|
||||||
renderForCaching, disableTtsStreaming, disableTtsCache});
|
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||||
break;
|
break;
|
||||||
|
case 'deepgramflux':
|
||||||
|
audioData = await synthDeepgramFlux(logger, {credentials, stats, model: model || voice, key, text,
|
||||||
|
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||||
|
break;
|
||||||
|
case 'xai':
|
||||||
|
audioData = await synthXai(logger, {
|
||||||
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
|
disableTtsCache});
|
||||||
|
break;
|
||||||
case 'resemble':
|
case 'resemble':
|
||||||
audioData = await synthResemble(logger, {
|
audioData = await synthResemble(logger, {
|
||||||
credentials, stats, voice, key, text, options, renderForCaching, disableTtsStreaming, disableTtsCache});
|
credentials, stats, voice, key, text, options, renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||||
@@ -969,6 +994,72 @@ const synthRimelabs = async(logger, {
|
|||||||
throw err;
|
throw err;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
const synthMurf = async(logger, {
|
||||||
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||||
|
}) => {
|
||||||
|
const {api_key, model_id, api_uri, options: credOpts} = credentials;
|
||||||
|
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||||
|
|
||||||
|
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||||
|
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
/* param keys here must match mod_murf_tts's text_param handler */
|
||||||
|
let params = '{';
|
||||||
|
params += `api_key=${api_key}`;
|
||||||
|
params += `,playback_id=${key}`;
|
||||||
|
params += ',vendor=murf';
|
||||||
|
params += `,voice=${voice}`;
|
||||||
|
if (model_id) params += `,model_id=${model_id}`;
|
||||||
|
if (language) params += `,language=${language}`;
|
||||||
|
if (api_uri) params += `,api_uri=${api_uri}`;
|
||||||
|
if (opts.style) params += `,style=${opts.style}`;
|
||||||
|
if (opts.rate !== undefined && opts.rate !== null) params += `,rate=${opts.rate}`;
|
||||||
|
if (opts.pitch !== undefined && opts.pitch !== null) params += `,pitch=${opts.pitch}`;
|
||||||
|
if (opts.variation !== undefined && opts.variation !== null) params += `,variation=${opts.variation}`;
|
||||||
|
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const sampleRate = 8000;
|
||||||
|
/* no Accept header: murf returns 406 if it doesn't match; the response
|
||||||
|
container is selected by the `format` field in the body instead */
|
||||||
|
const post = bent(api_uri || 'https://global.api.murf.ai', 'POST', 'buffer', {
|
||||||
|
'api-key': api_key,
|
||||||
|
'Content-Type': 'application/json'
|
||||||
|
});
|
||||||
|
/* murf REST schema is documented loosely; field names follow the SDK params
|
||||||
|
(voice_id/model/format/sample_rate) plus the websocket voice fields. */
|
||||||
|
const audioContent = await post('/v1/speech/stream', {
|
||||||
|
text,
|
||||||
|
voice_id: voice,
|
||||||
|
...(model_id && {model: model_id}),
|
||||||
|
...(language && {locale: language}),
|
||||||
|
...(opts.style && {style: opts.style}),
|
||||||
|
...(opts.rate !== undefined && opts.rate !== null && {rate: opts.rate}),
|
||||||
|
...(opts.pitch !== undefined && opts.pitch !== null && {pitch: opts.pitch}),
|
||||||
|
...(opts.variation !== undefined && opts.variation !== null && {variation: opts.variation}),
|
||||||
|
format: 'WAV',
|
||||||
|
sample_rate: sampleRate,
|
||||||
|
channel_type: 'MONO'
|
||||||
|
});
|
||||||
|
return {
|
||||||
|
audioContent,
|
||||||
|
extension: 'wav',
|
||||||
|
sampleRate
|
||||||
|
};
|
||||||
|
} catch (err) {
|
||||||
|
logger.info({err}, 'synth murf returned error');
|
||||||
|
stats.increment('tts.count', ['vendor:murf', 'accepted:no']);
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
|
};
|
||||||
const synthWhisper = async(logger, {credentials, stats, voice, key, text, instructions,
|
const synthWhisper = async(logger, {credentials, stats, voice, key, text, instructions,
|
||||||
renderForCaching, disableTtsStreaming, disableTtsCache}) => {
|
renderForCaching, disableTtsStreaming, disableTtsCache}) => {
|
||||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||||
@@ -1059,6 +1150,116 @@ const synthDeepgram = async(logger, {credentials, stats, model, key, text, rende
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Deepgram Flux TTS — the conversation-native model served from /v2/speak.
|
||||||
|
// Streaming rides the mediajam deepgramflux dialect via a say: filePath; the
|
||||||
|
// batch/cache path POSTs to /v2/speak (mp3 is batch-only for Flux).
|
||||||
|
const synthDeepgramFlux = async(logger, {credentials, stats, model, key, text, renderForCaching,
|
||||||
|
disableTtsStreaming, disableTtsCache}) => {
|
||||||
|
const {api_key, deepgram_tts_uri} = credentials;
|
||||||
|
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '{';
|
||||||
|
params += `api_key=${api_key}`;
|
||||||
|
params += `,playback_id=${key}`;
|
||||||
|
params += ',vendor=deepgramflux';
|
||||||
|
params += `,voice=${model}`;
|
||||||
|
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||||
|
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const post = bent(deepgram_tts_uri || 'https://api.deepgram.com', 'POST', 'buffer', {
|
||||||
|
// on-premise deepgram does not require to have api_key
|
||||||
|
...(api_key && {'Authorization': `Token ${api_key}`}),
|
||||||
|
'Accept': 'audio/mpeg',
|
||||||
|
'Content-Type': 'application/json'
|
||||||
|
});
|
||||||
|
const audioContent = await post(`/v2/speak?model=${model}`, {
|
||||||
|
text
|
||||||
|
});
|
||||||
|
return {
|
||||||
|
audioContent,
|
||||||
|
extension: 'mp3',
|
||||||
|
sampleRate: 8000
|
||||||
|
};
|
||||||
|
} catch (err) {
|
||||||
|
logger.info({err}, 'synth Deepgram Flux returned error');
|
||||||
|
stats.increment('tts.count', ['vendor:deepgramflux', 'accepted:no']);
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const synthXai = async(logger, {
|
||||||
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||||
|
}) => {
|
||||||
|
const {api_key, api_uri, options: credOpts} = credentials;
|
||||||
|
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||||
|
const speed = opts.speed;
|
||||||
|
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '{';
|
||||||
|
params += `api_key=${api_key}`;
|
||||||
|
params += `,playback_id=${key}`;
|
||||||
|
params += ',vendor=xai';
|
||||||
|
if (voice) params += `,voice=${voice}`;
|
||||||
|
if (language) params += `,language=${language}`;
|
||||||
|
if (speed !== null && speed !== undefined) params += `,speed=${speed}`;
|
||||||
|
if (opts.optimize_streaming_latency != null) {
|
||||||
|
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency}`;
|
||||||
|
}
|
||||||
|
if (opts.text_normalization != null) params += `,text_normalization=${opts.text_normalization}`;
|
||||||
|
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||||
|
if (api_uri) params += `,endpoint=${api_uri}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const post = bent(`https://${api_uri || 'api.x.ai'}`, 'POST', 'buffer', {
|
||||||
|
'Authorization': `Bearer ${api_key}`,
|
||||||
|
'Content-Type': 'application/json'
|
||||||
|
});
|
||||||
|
const audioContent = await post('/v1/tts', {
|
||||||
|
text,
|
||||||
|
language: language || 'auto',
|
||||||
|
...(voice && {voice_id: voice}),
|
||||||
|
...(speed !== null && speed !== undefined && {speed}),
|
||||||
|
...(opts.optimize_streaming_latency != null && {optimize_streaming_latency: opts.optimize_streaming_latency}),
|
||||||
|
...(opts.text_normalization != null && {text_normalization: opts.text_normalization}),
|
||||||
|
output_format: {
|
||||||
|
codec: 'wav',
|
||||||
|
sample_rate: 8000
|
||||||
|
}
|
||||||
|
});
|
||||||
|
return {
|
||||||
|
audioContent,
|
||||||
|
extension: 'wav',
|
||||||
|
sampleRate: 8000
|
||||||
|
};
|
||||||
|
} catch (err) {
|
||||||
|
// xAI errors are JSON {code, error} - read the body so the surfaced message isn't 'undefined'
|
||||||
|
if (err.name === 'StatusError' && typeof err.text === 'function') {
|
||||||
|
try {
|
||||||
|
const body = await err.text();
|
||||||
|
if (body) err.message = body;
|
||||||
|
} catch (readErr) {
|
||||||
|
logger.info({readErr}, 'synth xai: failed to read error response body');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
logger.info({err}, 'synth xai returned error');
|
||||||
|
stats.increment('tts.count', ['vendor:xai', 'accepted:no']);
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
const synthCartesia = async(logger, {
|
const synthCartesia = async(logger, {
|
||||||
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||||
}) => {
|
}) => {
|
||||||
|
|||||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "1.0.6",
|
"version": "1.0.9",
|
||||||
"lockfileVersion": 2,
|
"lockfileVersion": 2,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "1.0.6",
|
"version": "1.0.9",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"23": "^0.0.0",
|
"23": "^0.0.0",
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "1.0.6",
|
"version": "1.0.9",
|
||||||
"description": "TTS-related speech utilities for jambonz",
|
"description": "TTS-related speech utilities for jambonz",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"author": "Dave Horton",
|
"author": "Dave Horton",
|
||||||
|
|||||||
@@ -1118,6 +1118,69 @@ test('Deepgram speech synth tests', async(t) => {
|
|||||||
client.quit();
|
client.quit();
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test('Deepgram Flux speech synth tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.DEEPGRAM_API_KEY) {
|
||||||
|
t.pass('skipping Deepgram Flux speech synth tests since DEEPGRAM_API_KEY');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
const text = 'Hi there and welcome to jambones!';
|
||||||
|
try {
|
||||||
|
const opts = await synthAudio(stats, {
|
||||||
|
vendor: 'deepgramflux',
|
||||||
|
credentials: {
|
||||||
|
api_key: process.env.DEEPGRAM_API_KEY
|
||||||
|
},
|
||||||
|
model: process.env.DEEPGRAM_FLUX_MODEL || 'flux-alexis-en',
|
||||||
|
text,
|
||||||
|
renderForCaching: true
|
||||||
|
});
|
||||||
|
t.ok(!opts.servedFromCache, `successfully synthesized deepgramflux audio to ${opts.filePath}`);
|
||||||
|
|
||||||
|
} catch (err) {
|
||||||
|
console.error(JSON.stringify(err));
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('xai speech synth tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.XAI_API_KEY) {
|
||||||
|
t.pass('skipping xai speech synth tests - no XAI_API_KEY');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
const text = 'Hi there and welcome to jambones!';
|
||||||
|
try {
|
||||||
|
const opts = await synthAudio(stats, {
|
||||||
|
vendor: 'xai',
|
||||||
|
credentials: {
|
||||||
|
api_key: process.env.XAI_API_KEY,
|
||||||
|
options: JSON.stringify({
|
||||||
|
voice: process.env.XAI_VOICE || 'eve',
|
||||||
|
speed: 1.0,
|
||||||
|
optimize_streaming_latency: 1,
|
||||||
|
text_normalization: true
|
||||||
|
})
|
||||||
|
},
|
||||||
|
language: 'en',
|
||||||
|
voice: process.env.XAI_VOICE || 'eve',
|
||||||
|
text,
|
||||||
|
renderForCaching: true
|
||||||
|
});
|
||||||
|
t.ok(!opts.servedFromCache && opts.filePath, `successfully synthesized xai audio to ${opts.filePath}`);
|
||||||
|
|
||||||
|
} catch (err) {
|
||||||
|
console.error(JSON.stringify(err));
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
test('TTS Cache tests', async(t) => {
|
test('TTS Cache tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);
|
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);
|
||||||
|
|||||||
Reference in New Issue
Block a user