mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4eabfbe4b7 | ||
|
|
dbfabeaddf | ||
|
|
d0dfd07204 | ||
|
|
04a2466f54 | ||
|
|
0f9a9edc4d | ||
|
|
ced1a0ef0d | ||
|
|
1609d0b205 | ||
|
|
ef8ada2793 | ||
|
|
444ad2522f | ||
|
|
1caea60803 | ||
|
|
97c3588cfd | ||
|
|
da3aa5aadb | ||
|
|
2fe89f132c | ||
|
|
4bca840ba2 | ||
|
|
f858ccb781 | ||
|
|
3cf9894b44 | ||
|
|
436b15d648 | ||
|
|
c3b7ea4cd1 | ||
|
|
4b5430d61d | ||
|
|
9fc8fe8341 | ||
|
|
b31e40b8a5 | ||
|
|
62e1c69f69 | ||
|
|
bc68b672ac | ||
|
|
da1e279128 | ||
|
|
60bcfe07d7 | ||
|
|
d343c81088 | ||
|
|
706f6d5808 | ||
|
|
1b1a0f19d0 | ||
|
|
119ac50f7f | ||
|
|
cb479f04d5 | ||
|
|
7e21e0b666 | ||
|
|
dabdb5b584 |
@@ -12,6 +12,7 @@ module.exports = (opts, logger) => {
|
|||||||
client,
|
client,
|
||||||
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
|
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
|
||||||
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
|
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
|
||||||
|
addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger),
|
||||||
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
|
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
|
||||||
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
|
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
|
||||||
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
|
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
|
||||||
|
|||||||
@@ -0,0 +1,30 @@
|
|||||||
|
const fs = require('fs/promises');
|
||||||
|
const {noopLogger, makeSynthKey} = require('./utils');
|
||||||
|
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
|
||||||
|
|
||||||
|
async function addFileToCache(client, logger, path,
|
||||||
|
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
|
||||||
|
let key;
|
||||||
|
logger = logger || noopLogger;
|
||||||
|
|
||||||
|
try {
|
||||||
|
key = makeSynthKey({
|
||||||
|
account_sid,
|
||||||
|
vendor,
|
||||||
|
language: language || '',
|
||||||
|
voice: voice || deploymentId,
|
||||||
|
engine,
|
||||||
|
text,
|
||||||
|
});
|
||||||
|
const audioBuffer = await fs.readFile(path);
|
||||||
|
await client.setex(key, EXPIRES, audioBuffer.toString('base64'));
|
||||||
|
} catch (err) {
|
||||||
|
logger.error(err, 'addFileToCache: Error');
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
logger.debug(`addFileToCache: added ${path} to cache with key ${key}`);
|
||||||
|
return key;
|
||||||
|
}
|
||||||
|
|
||||||
|
module.exports = addFileToCache;
|
||||||
@@ -97,7 +97,7 @@ const getAwsVoices = async(_client, logger, credentials) => {
|
|||||||
secretAccessKey
|
secretAccessKey
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
const command = new DescribeVoicesCommand({LanguageCode: 'en-US'});
|
const command = new DescribeVoicesCommand({});
|
||||||
const response = await client.send(command);
|
const response = await client.send(command);
|
||||||
return response;
|
return response;
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
|
|||||||
+56
-7
@@ -76,7 +76,8 @@ const trimTrailingSilence = (buffer) => {
|
|||||||
* the synthesized audio, and a variable indicating whether it was served from cache
|
* the synthesized audio, and a variable indicating whether it was served from cache
|
||||||
*/
|
*/
|
||||||
async function synthAudio(client, logger, stats, { account_sid,
|
async function synthAudio(client, logger, stats, { account_sid,
|
||||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache, options
|
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
||||||
|
disableTtsCache, renderForCaching, disableTtsStreaming, options
|
||||||
}) {
|
}) {
|
||||||
let audioBuffer;
|
let audioBuffer;
|
||||||
let servedFromCache = false;
|
let servedFromCache = false;
|
||||||
@@ -142,6 +143,10 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
(
|
(
|
||||||
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
||||||
['microsoft', 'azure'].includes(vendor)
|
['microsoft', 'azure'].includes(vendor)
|
||||||
|
) ||
|
||||||
|
(
|
||||||
|
!process.env.JAMBONES_DISABLE_TTS_STREAMING &&
|
||||||
|
vendor === 'elevenlabs'
|
||||||
)
|
)
|
||||||
) {
|
) {
|
||||||
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
||||||
@@ -194,10 +199,15 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
||||||
break;
|
break;
|
||||||
case 'elevenlabs':
|
case 'elevenlabs':
|
||||||
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
|
audioBuffer = await synthElevenlabs(logger, {
|
||||||
|
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
|
||||||
|
});
|
||||||
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
break;
|
break;
|
||||||
case 'whisper':
|
case 'whisper':
|
||||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
|
audioBuffer = await synthWhisper(logger, {
|
||||||
|
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
|
||||||
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
break;
|
break;
|
||||||
case 'deepgram':
|
case 'deepgram':
|
||||||
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
|
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
|
||||||
@@ -594,9 +604,32 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text}) => {
|
const synthElevenlabs = async(logger, {
|
||||||
|
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
|
||||||
|
}) => {
|
||||||
const {api_key, model_id, options: credOpts} = credentials;
|
const {api_key, model_id, options: credOpts} = credentials;
|
||||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||||
|
|
||||||
|
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||||
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '';
|
||||||
|
params += `{api_key=${api_key}`;
|
||||||
|
params += `,model_id=${model_id}`;
|
||||||
|
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
|
||||||
|
params += ',write_cache_file=1';
|
||||||
|
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
|
||||||
|
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
|
||||||
|
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
|
||||||
|
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
const optimize_streaming_latency = opts.optimize_streaming_latency ?
|
const optimize_streaming_latency = opts.optimize_streaming_latency ?
|
||||||
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
|
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
|
||||||
try {
|
try {
|
||||||
@@ -622,8 +655,24 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
|
||||||
const {api_key, model_id, baseURL, timeout} = credentials;
|
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||||
|
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||||
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '';
|
||||||
|
params += `{api_key=${api_key}`;
|
||||||
|
params += `,model_id=${model_id}`;
|
||||||
|
params += `,voice=${voice}`;
|
||||||
|
params += ',write_cache_file=1';
|
||||||
|
if (speed) params += `,speed=${speed}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
const openai = new OpenAI.OpenAI({
|
const openai = new OpenAI.OpenAI({
|
||||||
apiKey: api_key,
|
apiKey: api_key,
|
||||||
@@ -648,7 +697,7 @@ const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
|||||||
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
|
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
|
||||||
const {api_key} = credentials;
|
const {api_key} = credentials;
|
||||||
try {
|
try {
|
||||||
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', {
|
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
|
||||||
'Authorization': `Token ${api_key}`,
|
'Authorization': `Token ${api_key}`,
|
||||||
'Accept': 'audio/mpeg',
|
'Accept': 'audio/mpeg',
|
||||||
'Content-Type': 'application/json'
|
'Content-Type': 'application/json'
|
||||||
|
|||||||
Generated
+1619
-2122
File diff suppressed because it is too large
Load Diff
+13
-13
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "0.0.33",
|
"version": "0.0.43",
|
||||||
"description": "TTS-related speech utilities for jambonz",
|
"description": "TTS-related speech utilities for jambonz",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"author": "Dave Horton",
|
"author": "Dave Horton",
|
||||||
@@ -24,26 +24,26 @@
|
|||||||
},
|
},
|
||||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@aws-sdk/client-polly": "^3.359.0",
|
"@aws-sdk/client-polly": "^3.496.0",
|
||||||
"@aws-sdk/client-sts": "^3.458.0",
|
"@aws-sdk/client-sts": "^3.496.0",
|
||||||
"@google-cloud/text-to-speech": "^4.2.1",
|
"@google-cloud/text-to-speech": "^5.0.2",
|
||||||
"@grpc/grpc-js": "^1.8.13",
|
"@grpc/grpc-js": "^1.9.14",
|
||||||
"@jambonz/realtimedb-helpers": "^0.8.7",
|
"@jambonz/realtimedb-helpers": "^0.8.7",
|
||||||
"bent": "^7.3.12",
|
"bent": "^7.3.12",
|
||||||
"debug": "^4.3.4",
|
"debug": "^4.3.4",
|
||||||
"form-urlencoded": "^6.1.0",
|
"form-urlencoded": "^6.1.4",
|
||||||
"google-protobuf": "^3.21.2",
|
"google-protobuf": "^3.21.2",
|
||||||
"ibm-watson": "^8.0.0",
|
"ibm-watson": "^8.0.0",
|
||||||
"microsoft-cognitiveservices-speech-sdk": "1.32.0",
|
"microsoft-cognitiveservices-speech-sdk": "1.34.0",
|
||||||
"openai": "^4.16.2",
|
"openai": "^4.25.0",
|
||||||
"undici": "^5.21.0"
|
"undici": "^6.4.0"
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
"config": "^3.3.9",
|
"config": "^3.3.10",
|
||||||
"eslint": "^8.33.0",
|
"eslint": "^8.56.0",
|
||||||
"eslint-plugin-promise": "^6.1.1",
|
"eslint-plugin-promise": "^6.1.1",
|
||||||
"nyc": "^15.1.0",
|
"nyc": "^15.1.0",
|
||||||
"pino": "^7.2.0",
|
"pino": "^8.17.0",
|
||||||
"tape": "^5.1.1"
|
"tape": "^5.7.3"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+10
-2
@@ -20,7 +20,7 @@ const stats = {
|
|||||||
|
|
||||||
test('Google speech synth tests', async(t) => {
|
test('Google speech synth tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {synthAudio, client} = fn(opts, logger);
|
const {synthAudio, addFileToCache, client} = fn(opts, logger);
|
||||||
|
|
||||||
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
|
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
|
||||||
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
|
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
|
||||||
@@ -58,6 +58,14 @@ test('Google speech synth tests', async(t) => {
|
|||||||
});
|
});
|
||||||
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
|
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
|
||||||
|
|
||||||
|
const success = await addFileToCache(opts.filePath, {
|
||||||
|
vendor: 'google',
|
||||||
|
language: 'en-GB',
|
||||||
|
gender: 'FEMALE',
|
||||||
|
text: 'This is a test. This is only a test'
|
||||||
|
});
|
||||||
|
t.ok(success, `successfully added ${opts.filePath} to cache`);
|
||||||
|
|
||||||
opts = await synthAudio(stats, {
|
opts = await synthAudio(stats, {
|
||||||
vendor: 'google',
|
vendor: 'google',
|
||||||
credentials: {
|
credentials: {
|
||||||
@@ -529,7 +537,7 @@ test('Deepgram speech synth tests', async(t) => {
|
|||||||
credentials: {
|
credentials: {
|
||||||
api_key: process.env.DEEPGRAM_API_KEY
|
api_key: process.env.DEEPGRAM_API_KEY
|
||||||
},
|
},
|
||||||
model: 'alpha-asteria-en-v2',
|
model: 'aura-asteria-en',
|
||||||
text,
|
text,
|
||||||
});
|
});
|
||||||
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
||||||
|
|||||||
Reference in New Issue
Block a user