Compare commits

...
7 Commits
Author SHA1 Message Date
Dave Horton 1ab7cb20e6 bump version 2023-03-24 14:47:35 -04:00
Dave Horton ba1052e629 Merge pull request #15 from jambonz/feature/get-tts-voices
add functions to retrieve voices for google and tts
2023-03-24 14:46:05 -04:00
Dave Horton 58270ad87f add functions to retrieve voices for google and tts 2023-03-24 14:43:51 -04:00
Dave Horton b172caee49 add support for nuance tts on-prem 2023-03-24 14:09:58 -04:00
Dave Horton 58374ca0fa Merge pull request #14 from jambonz/feature/nuance-onprem
add support for Nuance TTS on-prem (vs hosted)
2023-03-24 14:07:19 -04:00
Dave Horton 5a7e0d37f4 add support for Nuance TTS on-prem (vs hosted) 2023-03-24 14:05:18 -04:00
Dave Horton 6fa68bc712 more changes trying to get AWS V3 sdk properly configured 2023-03-21 08:20:29 -04:00
9 changed files with 353 additions and 24 deletions
+42 -6
View File
@@ -1,9 +1,11 @@
const assert = require('assert'); const assert = require('assert');
const {noopLogger, createNuanceClient} = require('./utils'); const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token'); const getNuanceAccessToken = require('./get-nuance-access-token');
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb'); const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1'); const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth'); const { IamAuthenticator } = require('ibm-watson/auth');
const ttsGoogle = require('@google-cloud/text-to-speech');
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
const getIbmVoices = async(client, logger, credentials) => { const getIbmVoices = async(client, logger, credentials) => {
const {tts_region, tts_api_key} = credentials; const {tts_region, tts_api_key} = credentials;
@@ -21,15 +23,20 @@ const getIbmVoices = async(client, logger, credentials) => {
}; };
const getNuanceVoices = async(client, logger, credentials) => { const getNuanceVoices = async(client, logger, credentials) => {
const {client_id: clientId, secret: secret} = credentials; const {client_id: clientId, secret: secret, nuance_tts_uri} = credentials;
return new Promise(async(resolve, reject) => { return new Promise(async(resolve, reject) => {
/* get a nuance access token */ /* get a nuance access token */
let token, nuanceClient; let token, nuanceClient;
try { try {
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts'); if (nuance_tts_uri) {
token = access_token.access_token; nuanceClient = await createKryptonClient(nuance_tts_uri);
nuanceClient = await createNuanceClient(token); }
else {
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts');
token = access_token.access_token;
nuanceClient = await createNuanceClient(token);
}
} catch (err) { } catch (err) {
logger.error({err}, 'getTtsVoices: error retrieving access token'); logger.error({err}, 'getTtsVoices: error retrieving access token');
return reject(err); return reject(err);
@@ -75,6 +82,30 @@ const getNuanceVoices = async(client, logger, credentials) => {
}); });
}; };
const getGoogleVoices = async(_client, logger, credentials) => {
const client = new ttsGoogle.TextToSpeechClient({credentials});
return await client.listVoices();
};
const getAwsVoices = async(_client, logger, credentials) => {
try {
const {region, accessKeyId, secretAccessKey} = credentials;
const client = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
const command = new DescribeVoicesCommand({LanguageCode: 'en-US'});
const response = await client.send(command);
return response;
} catch (err) {
logger.info({err}, 'testMicrosoftTts - failed to list voices for region ${region}');
throw err;
}
};
/** /**
* Synthesize speech to an mp3 file, and also cache the generated speech * Synthesize speech to an mp3 file, and also cache the generated speech
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying * in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
@@ -94,7 +125,7 @@ const getNuanceVoices = async(client, logger, credentials) => {
async function getTtsVoices(client, logger, {vendor, credentials}) { async function getTtsVoices(client, logger, {vendor, credentials}) {
logger = logger || noopLogger; logger = logger || noopLogger;
assert.ok(['nuance', 'ibm'].includes(vendor), assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly'].includes(vendor),
`getTtsVoices not supported for vendor ${vendor}`); `getTtsVoices not supported for vendor ${vendor}`);
switch (vendor) { switch (vendor) {
@@ -102,6 +133,11 @@ async function getTtsVoices(client, logger, {vendor, credentials}) {
return getNuanceVoices(client, logger, credentials); return getNuanceVoices(client, logger, credentials);
case 'ibm': case 'ibm':
return getIbmVoices(client, logger, credentials); return getIbmVoices(client, logger, credentials);
case 'google':
return getGoogleVoices(client, logger, credentials);
case 'aws':
case 'polly':
return getAwsVoices(client, logger, credentials);
default: default:
break; break;
} }
+23 -8
View File
@@ -16,7 +16,7 @@ const {
CancellationDetails, CancellationDetails,
SpeechSynthesisOutputFormat SpeechSynthesisOutputFormat
} = sdk; } = sdk;
const {makeSynthKey, createNuanceClient, noopLogger, createRivaClient} = require('./utils'); const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token'); const getNuanceAccessToken = require('./get-nuance-access-token');
const { const {
SynthesisRequest, SynthesisRequest,
@@ -75,8 +75,10 @@ async function synthAudio(client, logger, stats, { account_sid,
} }
else if ('nuance' === vendor) { else if ('nuance' === vendor) {
assert.ok(voice, 'synthAudio requires voice when nuance is used'); assert.ok(voice, 'synthAudio requires voice when nuance is used');
assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used'); if (!credentials.nuance_tts_uri) {
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used'); assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used');
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used');
}
} }
else if ('nvidia' === vendor) { else if ('nvidia' === vendor) {
assert.ok(voice, 'synthAudio requires voice when nvidia is used'); assert.ok(voice, 'synthAudio requires voice when nvidia is used');
@@ -184,7 +186,14 @@ async function synthAudio(client, logger, stats, { account_sid,
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => { const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => {
try { try {
const polly = new PollyClient({...credentials}); const {region, accessKeyId, secretAccessKey} = credentials;
const polly = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
const opts = { const opts = {
Engine: engine, Engine: engine,
OutputFormat: 'mp3', OutputFormat: 'mp3',
@@ -364,10 +373,16 @@ const synthWellSaid = async(logger, {credentials, stats, language, voice, gender
}; };
const synthNuance = async(client, logger, {credentials, stats, voice, model, text}) => { const synthNuance = async(client, logger, {credentials, stats, voice, model, text}) => {
/* get a nuance access token */ let nuanceClient;
const {client_id, secret} = credentials; const {client_id, secret, nuance_tts_uri} = credentials;
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts'); if (nuance_tts_uri) {
const nuanceClient = await createNuanceClient(access_token); nuanceClient = await createKryptonClient(nuance_tts_uri);
}
else {
/* get a nuance access token */
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts');
nuanceClient = await createNuanceClient(access_token);
}
const v = new Voice(); const v = new Voice();
const p = new AudioParameters(); const p = new AudioParameters();
+6
View File
@@ -77,6 +77,11 @@ const getNuanceAccessToken = async(clientId, secret, scope = 'asr tts') => {
return json.access_token; return json.access_token;
}; };
const createKryptonClient = async(uri) => {
const client = new SynthesizerClient(uri, grpc.credentials.createInsecure());
return client;
};
const createNuanceClient = async(access_token) => { const createNuanceClient = async(access_token) => {
//if (nuanceClientMap.has(access_token)) return nuanceClientMap.get(access_token); //if (nuanceClientMap.has(access_token)) return nuanceClientMap.get(access_token);
@@ -108,6 +113,7 @@ module.exports = {
makeIbmKey, makeIbmKey,
getNuanceAccessToken, getNuanceAccessToken,
createNuanceClient, createNuanceClient,
createKryptonClient,
createRivaClient, createRivaClient,
makeBasicAuthHeader, makeBasicAuthHeader,
NUANCE_AUTH_ENDPOINT, NUANCE_AUTH_ENDPOINT,
+2 -2
View File
@@ -1,12 +1,12 @@
{ {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.5", "version": "0.0.8",
"lockfileVersion": 2, "lockfileVersion": 2,
"requires": true, "requires": true,
"packages": { "packages": {
"": { "": {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.5", "version": "0.0.8",
"license": "MIT", "license": "MIT",
"dependencies": { "dependencies": {
"@aws-sdk/client-polly": "^3.276.0", "@aws-sdk/client-polly": "^3.276.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.5", "version": "0.0.8",
"description": "TTS-related speech utilities for jambonz", "description": "TTS-related speech utilities for jambonz",
"main": "index.js", "main": "index.js",
"author": "Dave Horton", "author": "Dave Horton",
+1 -2
View File
@@ -1,5 +1,4 @@
require('./docker_start'); require('./docker_start');
require('./synth'); require('./synth');
require('./nuance'); require('./list-voices');
require('./ibm');
require('./docker_stop'); require('./docker_stop');
+205
View File
@@ -0,0 +1,205 @@
const test = require('tape').test ;
const config = require('config');
const opts = config.get('redis');
const fs = require('fs');
const logger = require('pino')({level: 'error'});
process.on('unhandledRejection', (reason, p) => {
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
});
const stats = {
increment: () => {},
histogram: () => {}
};
test('IBM - create access key', async(t) => {
const fn = require('..');
const {client, getIbmAccessToken} = fn(opts, logger);
if (!process.env.IBM_API_KEY ) {
t.pass('skipping IBM test since no IBM api_key provided');
t.end();
client.quit();
return;
}
try {
let obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM');
t.ok(obj.access_token && !obj.servedFromCache, 'successfull received access token from IBM');
obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM - second request');
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('IBM - retrieve tts voices test', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.IBM_TTS_API_KEY || !process.env.IBM_TTS_REGION) {
t.pass('skipping IBM test since no IBM api_key and/or region provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'ibm',
credentials: {
tts_api_key: process.env.IBM_TTS_API_KEY,
tts_region: process.env.IBM_TTS_REGION
}
};
const obj = await getTtsVoices(opts);
const {voices} = obj.result;
//console.log(JSON.stringify(voices));
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nuance hosted tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
t.pass('skipping Nuance hosted test since no Nuance client_id and secret provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'nuance',
credentials: {
client_id: process.env.NUANCE_CLIENT_ID,
secret: process.env.NUANCE_SECRET
}
};
let voices = await getTtsVoices(opts);
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nuance on-prem tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.NUANCE_TTS_URI ) {
t.pass('skipping Nuance on-prem test since no Nuance uri provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'nuance',
credentials: {
nuance_tts_uri: process.env.NUANCE_TTS_URI
}
};
let voices = await getTtsVoices(opts);
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Google tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
return t.end();
}
try {
const str = process.env.GCP_JSON_KEY || fs.readFileSync(process.env.GCP_FILE);
const credentials = JSON.parse(str);
const opts = {
vendor: 'google',
credentials
};
let result = await getTtsVoices(opts);
t.ok(result[0].voices.length > 0, `GetVoices: successfully retrieved ${result[0].voices.length} voices from Google`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('AWS tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
t.pass('skipping AWS speech synth tests since AWS_ACCESS_KEY_ID, AWS_SECRET_ACCESS_KEY, or AWS_REGION not provided');
return t.end();
}
try {
const opts = {
vendor: 'aws',
credentials: {
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
region: process.env.AWS_REGION,
}
};
let result = await getTtsVoices(opts);
t.ok(result?.Voices?.length > 0, `GetVoices: successfully retrieved ${result.Voices.length} voices from AWS`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
+35 -4
View File
@@ -12,12 +12,12 @@ const stats = {
histogram: () => {} histogram: () => {}
}; };
test('Nuance tests', async(t) => { test('Nuance hosted tests', async(t) => {
const fn = require('..'); const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger); const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) { if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
t.pass('skipping Nuance test since no Nuance client_id and secret provided'); t.pass('skipping Nuance hosted test since no Nuance client_id and secret provided');
t.end(); t.end();
client.quit(); client.quit();
return; return;
@@ -31,8 +31,39 @@ test('Nuance tests', async(t) => {
} }
}; };
let voices = await getTtsVoices(opts); let voices = await getTtsVoices(opts);
//console.log(`received ${voices.length} voices from Nuance`); t.ok(voices.length > 0 && voices[0].language,
//console.log(JSON.stringify(voices)); `GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nuance on-prem tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.NUANCE_TTS_URI ) {
t.pass('skipping Nuance on-prem test since no Nuance uri provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'nuance',
credentials: {
nuance_tts_uri: process.env.NUANCE_TTS_URI
}
};
let voices = await getTtsVoices(opts);
t.ok(voices.length > 0 && voices[0].language, t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`); `GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
+38 -1
View File
@@ -212,7 +212,7 @@ test('Azure custom voice speech synth tests', async(t) => {
client.quit(); client.quit();
}); });
test('Nuance speech synth tests', async(t) => { test('Nuance hosted speech synth tests', async(t) => {
const fn = require('..'); const fn = require('..');
const {synthAudio, client} = fn(opts, logger); const {synthAudio, client} = fn(opts, logger);
@@ -251,6 +251,43 @@ test('Nuance speech synth tests', async(t) => {
client.quit(); client.quit();
}); });
test('Nuance on-prem speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.NUANCE_TTS_URI) {
t.pass('skipping Nuance on prem speech synth tests since NUANCE_TTS_URI not provided');
return t.end();
}
try {
let opts = await synthAudio(stats, {
vendor: 'nuance',
credentials: {
nuance_tts_uri: process.env.NUANCE_TTS_URI
},
language: 'en-US',
voice: 'Evan',
text: 'This is a test of on-prem. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized nuance audio to ${opts.filePath}`);
opts = await synthAudio(stats, {
vendor: 'nuance',
credentials: {
nuance_tts_uri: process.env.NUANCE_TTS_URI
},
language: 'en-US',
voice: 'Evan',
text: 'This is a test of on-prem. This is only a test',
});
t.ok(opts.servedFromCache, `successfully retrieved nuance audio from cache ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nvidia speech synth tests', async(t) => { test('Nvidia speech synth tests', async(t) => {
const fn = require('..'); const fn = require('..');
const {synthAudio, client} = fn(opts, logger); const {synthAudio, client} = fn(opts, logger);