Compare commits

..
19 Commits
Author SHA1 Message Date
Dave Horton 436b15d648 0.0.38 2024-01-26 11:09:24 -05:00
Dave Horton c3b7ea4cd1 fix prev commit 2024-01-26 11:08:06 -05:00
Dave Horton 4b5430d61d 0.0.37 2024-01-26 09:54:51 -05:00
Dave Horton 9fc8fe8341 Merge pull request #56 from jambonz/feat/cache-streaming
add function to add an audio file generated externally to cache
2024-01-26 09:54:24 -05:00
Dave Horton b31e40b8a5 add function to add an audio file generated externally to cache 2024-01-26 09:52:32 -05:00
Dave Horton 62e1c69f69 0.0.36 2024-01-25 13:14:31 -05:00
Dave Horton bc68b672ac Merge pull request #55 from jambonz/tts-streaming-cache-attribute
add tts param to indicate caching
2024-01-25 13:14:00 -05:00
Dave Horton da1e279128 add tts param to indicate caching 2024-01-25 13:07:49 -05:00
Dave Horton 60bcfe07d7 0.0.35 2024-01-25 09:40:14 -05:00
Dave Horton d343c81088 Merge pull request #54 from jambonz/feat/fix_getVoice_aws
fix get aws voice should not be limit in en-US
2024-01-25 09:39:36 -05:00
Quan HL 706f6d5808 fix get aws voice should not be limit in en-US 2024-01-25 21:37:38 +07:00
Dave Horton 1b1a0f19d0 0.0.34 2024-01-22 15:18:21 -05:00
Dave Horton 119ac50f7f Merge pull request #53 from jambonz/elevenlabs-streaming
Elevenlabs streaming
2024-01-22 15:17:33 -05:00
Dave Horton cb479f04d5 add param to synthAuydio to indicate whether audio is being generated specifically for caching purposes 2024-01-22 08:10:00 -05:00
Dave Horton 7e21e0b666 changes to support elevenlabs tts streaming 2024-01-21 21:36:53 -05:00
Dave Horton dabdb5b584 if streaming env is set prepare to use streaming tts 2024-01-20 14:15:13 -05:00
Dave Horton f36ba027d0 0.0.33 2023-12-25 22:13:29 -05:00
Dave Horton 08aae32975 Merge pull request #51 from jambonz/feat/deepgram
support deepgram
2023-12-25 22:10:50 -05:00
Quan HL 4cfc730d92 support deepgram 2023-12-26 09:26:31 +07:00
7 changed files with 1762 additions and 2143 deletions
+1
View File
@@ -12,6 +12,7 @@ module.exports = (opts, logger) => {
client,
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger),
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
+30
View File
@@ -0,0 +1,30 @@
const fs = require('fs/promises');
const {noopLogger, makeSynthKey} = require('./utils');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
async function addFileToCache(client, logger, path,
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
let key;
logger = logger || noopLogger;
try {
key = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text,
});
const audioBuffer = await fs.readFile(path);
await client.setex(key, EXPIRES, audioBuffer.toString('base64'));
} catch (err) {
logger.error(err, 'addFileToCache: Error');
return;
}
logger.debug(`addFileToCache: added ${path} to cache with key ${key}`);
return key;
}
module.exports = addFileToCache;
+1 -1
View File
@@ -97,7 +97,7 @@ const getAwsVoices = async(_client, logger, credentials) => {
secretAccessKey
}
});
const command = new DescribeVoicesCommand({LanguageCode: 'en-US'});
const command = new DescribeVoicesCommand({});
const response = await client.send(command);
return response;
} catch (err) {
+62 -6
View File
@@ -76,7 +76,8 @@ const trimTrailingSilence = (buffer) => {
* the synthesized audio, and a variable indicating whether it was served from cache
*/
async function synthAudio(client, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache, options
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -84,7 +85,7 @@ async function synthAudio(client, logger, stats, { account_sid,
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper'].includes(vendor) ||
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper', 'deepgram'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
if ('google' === vendor) {
@@ -142,6 +143,10 @@ async function synthAudio(client, logger, stats, { account_sid,
(
process.env.JAMBONES_TTS_TRIM_SILENCE &&
['microsoft', 'azure'].includes(vendor)
) ||
(
process.env.JAMBONES_ELEVENLABS_STREAMING &&
vendor === 'elevenlabs'
)
) {
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
@@ -194,11 +199,22 @@ async function synthAudio(client, logger, stats, { account_sid,
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
audioBuffer = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, filePath
});
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
return audioBuffer;
}
else {
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
}
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
break;
case 'deepgram':
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
@@ -591,9 +607,30 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text}) => {
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
params += ',write_cache_file=1';
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
const optimize_streaming_latency = opts.optimize_streaming_latency ?
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
try {
@@ -640,8 +677,27 @@ const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
stats.increment('tts.count', ['vendor:openai', 'accepted:no']);
throw err;
}
}
;
};
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
const {api_key} = credentials;
try {
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(`/v1/speak?model=${model}`, {
text
});
return mp3;
} catch (err) {
logger.info({err}, 'synth Deepgram returned error');
stats.increment('tts.count', ['vendor:deepgram', 'accepted:no']);
throw err;
}
};
const getFileExtFromMime = (mime) => {
switch (mime) {
case 'audio/wav':
+1619 -2122
View File
File diff suppressed because it is too large Load Diff
+13 -13
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.32",
"version": "0.0.38",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -24,26 +24,26 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.359.0",
"@aws-sdk/client-sts": "^3.458.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^5.0.2",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
"debug": "^4.3.4",
"form-urlencoded": "^6.1.0",
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.32.0",
"openai": "^4.16.2",
"undici": "^5.21.0"
"microsoft-cognitiveservices-speech-sdk": "1.34.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
},
"devDependencies": {
"config": "^3.3.9",
"eslint": "^8.33.0",
"config": "^3.3.10",
"eslint": "^8.56.0",
"eslint-plugin-promise": "^6.1.1",
"nyc": "^15.1.0",
"pino": "^7.2.0",
"tape": "^5.1.1"
"pino": "^8.17.0",
"tape": "^5.7.3"
}
}
+36 -1
View File
@@ -20,7 +20,7 @@ const stats = {
test('Google speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
const {synthAudio, addFileToCache, client} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
@@ -58,6 +58,14 @@ test('Google speech synth tests', async(t) => {
});
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
const success = await addFileToCache(opts.filePath, {
vendor: 'google',
language: 'en-GB',
gender: 'FEMALE',
text: 'This is a test. This is only a test'
});
t.ok(success, `successfully added ${opts.filePath} to cache`);
opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
@@ -514,6 +522,33 @@ test('whisper speech synth tests', async(t) => {
client.quit();
})
test('Deepgram speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.DEEPGRAM_API_KEY) {
t.pass('skipping Deepgram speech synth tests since DEEPGRAM_API_KEY');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'deepgram',
credentials: {
api_key: process.env.DEEPGRAM_API_KEY
},
model: 'alpha-asteria-en-v2',
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('TTS Cache tests', async(t) => {
const fn = require('..');
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);