Compare commits

..
33 Commits
Author SHA1 Message Date
Dave Horton 9827f7405d 0.0.28 2023-11-29 14:17:58 -05:00
Dave Horton 906a6b52b8 fix issue with conflicting EC2 permissions calling AWS STS 2023-11-29 14:17:46 -05:00
Dave Horton 1f0b9ff539 0.0.27 2023-11-27 09:45:39 -05:00
Dave Horton b6f8368357 Merge pull request #46 from jambonz/feat/aws-sts-auth-token
Feat/aws sts auth token
2023-11-27 09:44:38 -05:00
Dave Horton 6bc487b3d1 logging 2023-11-27 09:37:49 -05:00
Dave Horton 9699abe0a3 dowgrade azure to 1.32.0 per: https://github.com/microsoft/cognitive-services-speech-sdk-js/issues/752 2023-11-27 09:32:03 -05:00
Dave Horton 5e1e7b17c6 remove console logging in tests 2023-11-27 09:27:18 -05:00
Dave Horton 1f52cd4f08 added function to get an AWS security token using STS 2023-11-27 09:26:41 -05:00
Dave Horton 7df4e2f4c7 0.0.26 2023-11-14 08:48:14 -05:00
Dave Horton e99f7c5087 Merge pull request #43 from jambonz/feat/realtimedb
use realtimedb-helper for initiate redis connection
2023-11-10 07:49:28 -05:00
Quan HL c7d981c23a wip 2023-11-10 10:37:22 +07:00
Quan HL 95c54b9b12 use realtimedb-helper for initiate redis connection 2023-11-10 10:34:50 +07:00
Dave Horton 48192aeba1 Merge pull request #42 from jambonz/gh-actions
update gihub actions to test openai and elevenlabs
2023-11-09 08:44:14 -05:00
Dave Horton 8f931cd8a5 update gihub actions to test openai and elevenlabs 2023-11-09 08:42:32 -05:00
Dave Horton 144baafe94 0.0.25 2023-11-09 08:32:07 -05:00
Dave Horton ed3e513419 fix google speech test 2023-11-09 08:31:59 -05:00
Dave Horton 8d93fdc42a Merge pull request #41 from jambonz/feat/openai
support whisper tts
2023-11-09 08:30:24 -05:00
Quan HL ea523a7a1d wip 2023-11-09 12:59:20 +07:00
Quan HL 750ed97312 update review 2023-11-09 09:19:47 +07:00
Quan HL 73baa81177 update review 2023-11-09 09:15:54 +07:00
Quan HL f86234a769 support openai 2023-11-09 07:10:07 +07:00
Dave Horton d564f24e6c 0.0.24 2023-10-30 19:54:56 -04:00
Dave Horton 464d8462d9 Merge pull request #39 from jambonz/feat/google_custom_voice_01
fix google custom voice
2023-10-30 19:54:44 -04:00
Quan HL 6853f0e342 fix google custom voice 2023-10-31 06:24:52 +07:00
Dave Horton 507045dcba 0.0.23 2023-10-29 22:07:02 -04:00
Dave Horton 08758bbbff Merge pull request #38 from jambonz/feat/google_custom_voice
feat support google custom voice
2023-10-29 22:06:41 -04:00
Hoan Luu Huu a3aa1169b8 feat support google custom voice 2023-10-30 01:55:50 +00:00
Hoan Luu Huu 7cae19a4e5 feat support google custom voice 2023-10-30 01:54:02 +00:00
Hoan Luu Huu 8c4d5a7cee feat support google custom voice 2023-10-30 01:52:48 +00:00
Dave Horton eb2c39072b 0.0.22 2023-10-14 13:18:08 +02:00
Dave Horton e5932ffc18 Merge pull request #34 from jambonz/feat/elevenlabs
add elevenlabs
2023-10-14 07:17:14 -04:00
Quan HL 0a98c6a376 fix review comment 2023-10-14 18:13:04 +07:00
Quan HL ea153e9833 add elevenlabs 2023-10-12 14:22:15 +07:00
11 changed files with 3638 additions and 2429 deletions
+5 -1
View File
@@ -24,4 +24,8 @@ jobs:
IBM_TTS_API_KEY: ${{ secrets.IBM_TTS_API_KEY }}
IBM_TTS_REGION: ${{ secrets.IBM_TTS_REGION }}
MICROSOFT_API_KEY: ${{ secrets.MICROSOFT_API_KEY }}
MICROSOFT_REGION: ${{ secrets.MICROSOFT_REGION }}
MICROSOFT_REGION: ${{ secrets.MICROSOFT_REGION }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ELEVENLABS_API_KEY: ${{ secrets.ELEVENLABS_API_KEY }}
ELEVENLABS_VOICE_ID: ${{ secrets.ELEVENLABS_VOICE_ID }}
ELEVENLABS_MODEL_ID: ${{ secrets.ELEVENLABS_MODEL_ID }}
+2 -1
View File
@@ -1,2 +1,3 @@
# speech-utils
TTS-related speech utilities for jambonz
TTS-related speech utilities for jambonz.
+8 -26
View File
@@ -1,33 +1,14 @@
const {noopLogger} = require('./lib/utils');
const Redis = require('ioredis');
module.exports = (opts, logger) => {
logger = logger || noopLogger;
const connectionOpts = {...opts};
// Support legacy app
if (process.env.JAMBONES_REDIS_USERNAME && process.env.JAMBONES_REDIS_PASSWORD) {
if (Array.isArray(connectionOpts)) {
for (const o of opts) {
o.username = process.env.JAMBONES_REDIS_USERNAME;
o.password = process.env.JAMBONES_REDIS_PASSWORD;
}
} else {
connectionOpts.username = process.env.JAMBONES_REDIS_USERNAME;
connectionOpts.password = process.env.JAMBONES_REDIS_PASSWORD;
}
}
const client = new Redis(connectionOpts);
['ready', 'connect', 'reconnecting', 'error', 'end', 'warning']
.forEach((event) => {
client.on(event, (...args) => {
if ('error' === event) {
if (process.env.NODE_ENV === 'test' && args[0]?.code === 'ECONNREFUSED') return;
logger.error({...args}, '@jambonz/realtimedb-helpers - redis error');
}
else logger.debug({args}, `redis event ${event}`);
});
});
let client = opts.redis_client;
const {
client: redisClient,
createHash,
retrieveHash
} = require('@jambonz/realtimedb-helpers')(opts, logger);
client = opts.redis_client || redisClient;
return {
client,
@@ -36,6 +17,7 @@ module.exports = (opts, logger) => {
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
getAwsAuthToken: require('./lib/get-aws-sts-token').bind(null, logger, createHash, retrieveHash),
getTtsVoices: require('./lib/get-tts-voices').bind(null, client, logger),
};
};
+45
View File
@@ -0,0 +1,45 @@
const { STSClient, GetSessionTokenCommand } = require('@aws-sdk/client-sts');
const {makeAwsKey, noopLogger} = require('./utils');
const debug = require('debug')('jambonz:speech-utils');
const EXPIRY = 3600;
async function getAwsAuthToken(
logger,
createHash, retrieveHash,
awsAccessKeyId, awsSecretAccessKey, awsRegion) {
logger = logger || noopLogger;
try {
const key = makeAwsKey(awsAccessKeyId);
const obj = await retrieveHash(key);
if (obj) return {...obj, servedFromCache: true};
/* access token not found in cache, so generate it using STS */
const stsClient = new STSClient({
region: awsRegion,
credentials: {
accessKeyId: awsAccessKeyId,
secretAccessKey: awsSecretAccessKey,
}
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
const data = await stsClient.send(command);
const credentials = {
accessKeyId: data.Credentials.AccessKeyId,
secretAccessKey: data.Credentials.SecretAccessKey,
sessionToken: data.Credentials.SessionToken
};
/* expire 10 minutes before the hour, so we don't lose the use of it during a call */
createHash(key, credentials, EXPIRY - 600)
.catch((err) => logger.error(err, `Error saving hash for key ${key}`));
return {...credentials, servedFromCache: false};
} catch (err) {
debug(err, 'getAwsAuthToken: Error retrieving AWS auth token');
logger.error(err, 'getAwsAuthToken: Error retrieving AWS auth token');
throw err;
}
}
module.exports = getAwsAuthToken;
+66 -2
View File
@@ -38,6 +38,7 @@ const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
const debug = require('debug')('jambonz:realtimedb-helpers');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp';
const OpenAI = require('openai');
const trimTrailingSilence = (buffer) => {
@@ -82,7 +83,8 @@ async function synthAudio(client, logger, stats, { account_sid,
let rtt;
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm'].includes(vendor) ||
assert.ok(['google', 'aws', 'polly', 'microsoft',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
if ('google' === vendor) {
@@ -116,6 +118,14 @@ async function synthAudio(client, logger, stats, { account_sid,
language = 'en-US'; // WellSaid only supports English atm
assert.ok(voice, 'synthAudio requires voice when wellsaid is used');
assert.ok(!text.startsWith('<speak'), 'wellsaid does not support SSML tags');
} else if ('elevenlabs' === vendor) {
assert.ok(voice, 'synthAudio requires voice when elevenlabs is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when elevenlabs is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when elevenlabs is used');
} else if ('whisper' === vendor) {
assert.ok(voice, 'synthAudio requires voice when whisper is used');
assert.ok(credentials.model_id, 'synthAudio requires model when whisper is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when whisper is used');
} else if (vendor.startsWith('custom')) {
assert.ok(credentials.custom_tts_url, `synthAudio requires custom_tts_url in credentials when ${vendor} is used`);
}
@@ -183,6 +193,12 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'wellsaid':
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
@@ -253,7 +269,8 @@ const synthGoogle = async(logger, {credentials, stats, language, voice, gender,
const client = new ttsGoogle.TextToSpeechClient(credentials);
const opts = {
voice: {
name: voice,
...(typeof voice === 'string' && {name: voice}),
...(typeof voice === 'object' && {customVoice: voice}),
languageCode: language,
ssmlGender: gender || 'SSML_VOICE_GENDER_UNSPECIFIED'
},
@@ -568,6 +585,53 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, stats, language, voice, text}) => {
const {api_key, model_id} = credentials;
try {
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
'xi-api-key': api_key,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(`/v1/text-to-speech/${voice}`, {
text,
model_id,
voice_settings: {
stability: 0.5,
similarity_boost: 0.5
}
});
return mp3;
} catch (err) {
logger.info({err}, 'synth Elevenlabs returned error');
stats.increment('tts.count', ['vendor:elevenlabs', 'accepted:no']);
throw err;
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
const {api_key, model_id, baseURL, timeout} = credentials;
try {
const openai = new OpenAI.OpenAI({
apiKey: api_key,
timeout: timeout || 5000,
...(baseURL && {baseURL})
});
const mp3 = await openai.audio.speech.create({
model: model_id,
voice,
input: text,
response_format: 'mp3'
});
return Buffer.from(await mp3.arrayBuffer());
} catch (err) {
logger.info({err}, 'synth whisper returned error');
stats.increment('tts.count', ['vendor:openai', 'accepted:no']);
throw err;
}
}
;
const getFileExtFromMime = (mime) => {
switch (mime) {
case 'audio/wav':
+7
View File
@@ -43,6 +43,12 @@ function makeIbmKey(apiKey) {
return `ibm:${hash.digest('hex')}`;
}
function makeAwsKey(awsAccessKeyId) {
const hash = crypto.createHash('sha1');
hash.update(awsAccessKeyId);
return `aws:${hash.digest('hex')}`;
}
function makeNuanceKey(clientId, secret, scope) {
const hash = crypto.createHash('sha1');
hash.update(`${clientId}:${secret}:${scope}`);
@@ -110,6 +116,7 @@ module.exports = {
makeSynthKey,
makeNuanceKey,
makeIbmKey,
makeAwsKey,
getNuanceAccessToken,
createNuanceClient,
createKryptonClient,
+3362 -2393
View File
File diff suppressed because it is too large Load Diff
+5 -3
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.21",
"version": "0.0.28",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -25,15 +25,17 @@
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.359.0",
"@aws-sdk/client-sts": "^3.458.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
"debug": "^4.3.4",
"form-urlencoded": "^6.1.0",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
"ioredis": "^5.3.2",
"microsoft-cognitiveservices-speech-sdk": "1.32.0",
"openai": "^4.16.2",
"undici": "^5.21.0"
},
"devDependencies": {
+40
View File
@@ -0,0 +1,40 @@
const test = require('tape').test ;
const config = require('config');
const opts = config.get('redis');
const logger = require('pino')({level: 'error'});
process.on('unhandledRejection', (reason, p) => {
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
});
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
test('AWS - create and cache auth token', async(t) => {
const fn = require('..');
const {client, getAwsAuthToken} = fn(opts, logger);
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
t.pass('skipping AWS auth token tests since no AWS credentials provided');
t.end();
client.quit();
return;
}
try {
let obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
//console.log({obj}, 'received auth token from AWS');
t.ok(obj.sessionToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
await sleep(250);
obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
//console.log({obj}, 'received auth token from AWS - second request');
t.ok(obj.sessionToken && obj.servedFromCache, 'successfully received access token from cache');
await client.flushall();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
+3
View File
@@ -1,4 +1,7 @@
require('./docker_start');
require('./synth');
require('./list-voices');
require('./aws');
require('./ibm');
require('./nuance');
require('./docker_stop');
+95 -3
View File
@@ -38,7 +38,7 @@ test('Google speech synth tests', async(t) => {
},
},
language: 'en-GB',
gender: 'MALE',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
salt: 'foo.bar',
});
@@ -53,7 +53,7 @@ test('Google speech synth tests', async(t) => {
},
},
language: 'en-GB',
gender: 'MALE',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
});
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
@@ -68,7 +68,7 @@ test('Google speech synth tests', async(t) => {
},
disableTtsCache: true,
language: 'en-GB',
gender: 'MALE',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized google audio regardless of current cache to ${opts.filePath}`);
@@ -79,6 +79,40 @@ test('Google speech synth tests', async(t) => {
client.quit();
});
test('Google speech Custom voice synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_CUSTOM_VOICE_FILE && !process.env.GCP_CUSTOM_VOICE_JSON_KEY || !process.env.GCP_CUSTOM_VOICE_MODEL) {
t.pass('skipping google speech synth tests since neither GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_MODEL is not provided');
return t.end();
}
try {
const str = process.env.GCP_CUSTOM_VOICE_JSON_KEY || fs.readFileSync(process.env.GCP_CUSTOM_VOICE_FILE);
const creds = JSON.parse(str);
let opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-AU',
text: 'This is a test. This is only a test',
voice: {
reportedUsage:"REALTIME",
model: process.env.GCP_CUSTOM_VOICE_MODEL
}
});
t.ok(!opts.servedFromCache, `successfully synthesized google custom voice audio to ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('AWS speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -411,6 +445,64 @@ test('Custom Vendor speech synth tests', async(t) => {
client.quit();
});
test('Elevenlabs speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.ELEVENLABS_API_KEY || !process.env.ELEVENLABS_VOICE_ID || !process.env.ELEVENLABS_MODEL_ID) {
t.pass('skipping ElevenLabs speech synth tests since ELEVENLABS_API_KEY or ELEVENLABS_VOICE_ID or ELEVENLABS_MODEL_ID not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'elevenlabs',
credentials: {
api_key: process.env.ELEVENLABS_API_KEY,
model_id: process.env.ELEVENLABS_MODEL_ID
},
language: 'en-US',
voice: process.env.ELEVENLABS_VOICE_ID,
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('whisper speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.OPENAI_API_KEY) {
t.pass('skipping OPENAI speech synth tests since OPENAI_API_KEY not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'whisper',
credentials: {
api_key: process.env.OPENAI_API_KEY,
model_id: 'tts-1'
},
language: 'en-US',
voice: 'alloy',
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('TTS Cache tests', async(t) => {
const fn = require('..');
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);