mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bf229d0ab0 | ||
|
|
f08fedb8ca | ||
|
|
fbed59e5de | ||
|
|
4f1685a365 | ||
|
|
2701af102a | ||
|
|
7f939b96d2 | ||
|
|
4d58ca6daf | ||
|
|
16dd7a2805 | ||
|
|
8f3e930004 | ||
|
|
3f4c444d82 | ||
|
|
fd7d8b8bcd | ||
|
|
46f833c7fa | ||
|
|
4eabfbe4b7 | ||
|
|
f3ab2baa6a | ||
|
|
fb412e2ddf | ||
|
|
f06f96a6f0 | ||
|
|
2988e800b1 | ||
|
|
dbfabeaddf | ||
|
|
c3188e40bb | ||
|
|
d0dfd07204 | ||
|
|
04a2466f54 | ||
|
|
0f9a9edc4d | ||
|
|
31a54f595b | ||
|
|
3560a6d4d9 | ||
|
|
be8053db4f | ||
|
|
4ffae38a3f | ||
|
|
9e74760c39 | ||
|
|
ced1a0ef0d | ||
|
|
1609d0b205 | ||
|
|
ef8ada2793 | ||
|
|
444ad2522f | ||
|
|
1caea60803 | ||
|
|
97c3588cfd | ||
|
|
da3aa5aadb | ||
|
|
2fe89f132c | ||
|
|
4bca840ba2 | ||
|
|
f858ccb781 | ||
|
|
3cf9894b44 | ||
|
|
436b15d648 | ||
|
|
c3b7ea4cd1 | ||
|
|
4b5430d61d | ||
|
|
9fc8fe8341 | ||
|
|
b31e40b8a5 |
@@ -12,6 +12,7 @@ module.exports = (opts, logger) => {
|
||||
client,
|
||||
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
|
||||
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
|
||||
addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger),
|
||||
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
|
||||
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
|
||||
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
const fs = require('fs/promises');
|
||||
const {noopLogger, makeSynthKey} = require('./utils');
|
||||
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
|
||||
|
||||
async function addFileToCache(client, logger, path,
|
||||
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
|
||||
let key;
|
||||
logger = logger || noopLogger;
|
||||
|
||||
try {
|
||||
key = makeSynthKey({
|
||||
account_sid,
|
||||
vendor,
|
||||
language: language || '',
|
||||
voice: voice || deploymentId,
|
||||
engine,
|
||||
text,
|
||||
});
|
||||
const audioBuffer = await fs.readFile(path);
|
||||
await client.setex(key, EXPIRES, audioBuffer.toString('base64'));
|
||||
} catch (err) {
|
||||
logger.error(err, 'addFileToCache: Error');
|
||||
return;
|
||||
}
|
||||
|
||||
logger.debug(`addFileToCache: added ${path} to cache with key ${key}`);
|
||||
return key;
|
||||
}
|
||||
|
||||
module.exports = addFileToCache;
|
||||
+79
-34
@@ -77,7 +77,7 @@ const trimTrailingSilence = (buffer) => {
|
||||
*/
|
||||
async function synthAudio(client, logger, stats, { account_sid,
|
||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
||||
disableTtsCache, renderForCaching, options
|
||||
disableTtsCache, renderForCaching, disableTtsStreaming, options
|
||||
}) {
|
||||
let audioBuffer;
|
||||
let servedFromCache = false;
|
||||
@@ -143,6 +143,10 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
(
|
||||
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
||||
['microsoft', 'azure'].includes(vendor)
|
||||
) ||
|
||||
(
|
||||
!process.env.JAMBONES_DISABLE_TTS_STREAMING &&
|
||||
vendor === 'elevenlabs'
|
||||
)
|
||||
) {
|
||||
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
||||
@@ -179,7 +183,9 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
case 'azure':
|
||||
case 'microsoft':
|
||||
vendorLabel = 'microsoft';
|
||||
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId, filePath});
|
||||
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId,
|
||||
filePath, renderForCaching, disableTtsStreaming});
|
||||
if (audioBuffer?.filePath) return audioBuffer;
|
||||
break;
|
||||
case 'nuance':
|
||||
model = model || 'enhanced';
|
||||
@@ -196,17 +202,14 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
break;
|
||||
case 'elevenlabs':
|
||||
audioBuffer = await synthElevenlabs(logger, {
|
||||
credentials, options, stats, language, voice, text, renderForCaching, filePath
|
||||
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
|
||||
});
|
||||
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
|
||||
return audioBuffer;
|
||||
}
|
||||
else {
|
||||
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
|
||||
}
|
||||
if (audioBuffer?.filePath) return audioBuffer;
|
||||
break;
|
||||
case 'whisper':
|
||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
|
||||
audioBuffer = await synthWhisper(logger, {
|
||||
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
|
||||
if (audioBuffer?.filePath) return audioBuffer;
|
||||
break;
|
||||
case 'deepgram':
|
||||
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
|
||||
@@ -380,10 +383,47 @@ const synthMicrosoft = async(logger, {
|
||||
language,
|
||||
voice,
|
||||
text,
|
||||
filePath
|
||||
filePath,
|
||||
renderForCaching,
|
||||
disableTtsStreaming
|
||||
}) => {
|
||||
try {
|
||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
|
||||
// let clean up the text
|
||||
let content = text;
|
||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||
/**
|
||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||
* Otherwise sending plain text we get "Voice does not match"
|
||||
*/
|
||||
content = `<speak>${text}</speak>`;
|
||||
}
|
||||
|
||||
if (content.startsWith('<speak>')) {
|
||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
|
||||
// eslint-disable-next-line max-len
|
||||
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
|
||||
logger.info({content}, 'synthMicrosoft');
|
||||
}
|
||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '';
|
||||
params += `{api_key=${apiKey}`;
|
||||
params += `,language=${language}`;
|
||||
params += ',vendor=microsoft';
|
||||
params += `,voice=${voice}`;
|
||||
params += ',write_cache_file=1';
|
||||
if (region) params += `,region=${region}`;
|
||||
if (custom_tts_endpoint) params += `,endpointId=${custom_tts_endpoint}`;
|
||||
if (process.env.JAMBONES_HTTP_PROXY_IP) params += `,http_proxy_ip=${process.env.JAMBONES_HTTP_PROXY_IP}`;
|
||||
if (process.env.JAMBONES_HTTP_PROXY_PORT) params += `,http_proxy_port=${process.env.JAMBONES_HTTP_PROXY_PORT}`;
|
||||
params += '}';
|
||||
return {
|
||||
filePath: `say:${params}${content.replace(/\n/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
if (use_custom_tts && custom_tts_endpoint_url) {
|
||||
return await _synthOnPremMicrosoft(logger, {
|
||||
credentials,
|
||||
@@ -395,20 +435,12 @@ const synthMicrosoft = async(logger, {
|
||||
});
|
||||
}
|
||||
const trimSilence = filePath.endsWith('.r8');
|
||||
let content = text;
|
||||
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
||||
speechConfig.speechSynthesisLanguage = language;
|
||||
speechConfig.speechSynthesisVoiceName = voice;
|
||||
if (use_custom_tts && custom_tts_endpoint) {
|
||||
speechConfig.endpointId = custom_tts_endpoint;
|
||||
}
|
||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||
/**
|
||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||
* Otherwise sending plain text we get "Voice does not match"
|
||||
*/
|
||||
content = `<speak>${text}</speak>`;
|
||||
}
|
||||
speechConfig.speechSynthesisOutputFormat = trimSilence ?
|
||||
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
|
||||
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||
@@ -420,14 +452,6 @@ const synthMicrosoft = async(logger, {
|
||||
}
|
||||
const synthesizer = new SpeechSynthesizer(speechConfig);
|
||||
|
||||
if (content.startsWith('<speak>')) {
|
||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
|
||||
// eslint-disable-next-line max-len
|
||||
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
|
||||
logger.info({content}, 'synthMicrosoft');
|
||||
}
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
const speakAsync = content.startsWith('<speak') ?
|
||||
synthesizer.speakSsmlAsync.bind(synthesizer) :
|
||||
@@ -603,14 +627,18 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
|
||||
}
|
||||
};
|
||||
|
||||
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
|
||||
const synthElevenlabs = async(logger, {
|
||||
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
|
||||
}) => {
|
||||
const {api_key, model_id, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
|
||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '';
|
||||
params += `{api_key=${api_key}`;
|
||||
params += ',vendor=elevenlabs';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
|
||||
params += ',write_cache_file=1';
|
||||
@@ -621,7 +649,7 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
@@ -652,8 +680,25 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
||||
}
|
||||
};
|
||||
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
||||
const {api_key, model_id, baseURL, timeout} = credentials;
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
|
||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '';
|
||||
params += `{api_key=${api_key}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
params += ',vendor=whisper';
|
||||
params += `,voice=${voice}`;
|
||||
params += ',write_cache_file=1';
|
||||
if (speed) params += `,speed=${speed}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
try {
|
||||
const openai = new OpenAI.OpenAI({
|
||||
apiKey: api_key,
|
||||
@@ -678,7 +723,7 @@ const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
||||
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
|
||||
const {api_key} = credentials;
|
||||
try {
|
||||
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', {
|
||||
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
|
||||
'Authorization': `Token ${api_key}`,
|
||||
'Accept': 'audio/mpeg',
|
||||
'Content-Type': 'application/json'
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.36",
|
||||
"version": "0.0.46",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.36",
|
||||
"version": "0.0.46",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.496.0",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.36",
|
||||
"version": "0.0.46",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
|
||||
+66
-2
@@ -20,7 +20,7 @@ const stats = {
|
||||
|
||||
test('Google speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
const {synthAudio, addFileToCache, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
|
||||
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
|
||||
@@ -58,6 +58,14 @@ test('Google speech synth tests', async(t) => {
|
||||
});
|
||||
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
|
||||
|
||||
const success = await addFileToCache(opts.filePath, {
|
||||
vendor: 'google',
|
||||
language: 'en-GB',
|
||||
gender: 'FEMALE',
|
||||
text: 'This is a test. This is only a test'
|
||||
});
|
||||
t.ok(success, `successfully added ${opts.filePath} to cache`);
|
||||
|
||||
opts = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: {
|
||||
@@ -180,6 +188,7 @@ test('Azure speech synth tests', async(t) => {
|
||||
language: 'en-US',
|
||||
voice: 'en-US-ChristopherNeural',
|
||||
text: longText,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
||||
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
|
||||
@@ -195,6 +204,7 @@ test('Azure speech synth tests', async(t) => {
|
||||
language: 'en-US',
|
||||
voice: 'en-US-ChristopherNeural',
|
||||
text: longText,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
|
||||
} catch (err) {
|
||||
@@ -204,6 +214,58 @@ test('Azure speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('Azure SSML tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.MICROSOFT_API_KEY || !process.env.MICROSOFT_REGION) {
|
||||
t.pass('skipping Microsoft speech synth tests since MICROSOFT_API_KEY or MICROSOFT_REGION not provided');
|
||||
return t.end();
|
||||
}
|
||||
try {
|
||||
const text = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="en-US">
|
||||
<voice name="en-US-JennyMultilingualNeural">
|
||||
<mstts:express-as style="cheerful" styledegree="2">That'd be just amazing!
|
||||
</mstts:express-as>
|
||||
</voice>
|
||||
</speak>`;
|
||||
|
||||
let opts = await synthAudio(stats, {
|
||||
vendor: 'microsoft',
|
||||
credentials: {
|
||||
api_key: process.env.MICROSOFT_API_KEY,
|
||||
region: process.env.MICROSOFT_REGION,
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'en-US-ChristopherNeural',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
||||
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
|
||||
t.pass('successfully used proxy to reach microsoft tts service');
|
||||
}
|
||||
|
||||
opts = await synthAudio(stats, {
|
||||
vendor: 'microsoft',
|
||||
credentials: {
|
||||
api_key: process.env.MICROSOFT_API_KEY,
|
||||
region: process.env.MICROSOFT_REGION,
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'en-US-ChristopherNeural',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
|
||||
test('Azure custom voice speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
@@ -225,6 +287,7 @@ test('Azure custom voice speech synth tests', async(t) => {
|
||||
language: 'en-US',
|
||||
voice: process.env.MICROSOFT_CUSTOM_VOICE,
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
||||
|
||||
@@ -239,6 +302,7 @@ test('Azure custom voice speech synth tests', async(t) => {
|
||||
language: 'en-US',
|
||||
voice: process.env.MICROSOFT_CUSTOM_VOICE,
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`);
|
||||
} catch (err) {
|
||||
@@ -529,7 +593,7 @@ test('Deepgram speech synth tests', async(t) => {
|
||||
credentials: {
|
||||
api_key: process.env.DEEPGRAM_API_KEY
|
||||
},
|
||||
model: 'alpha-asteria-en-v2',
|
||||
model: 'aura-asteria-en',
|
||||
text,
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
||||
|
||||
Reference in New Issue
Block a user