Compare commits

...
9 Commits
Author SHA1 Message Date
Dave Horton 11746d3f22 bump version and minor changes 2023-03-31 20:05:07 -04:00
Dave Horton 3fcfbd10a1 Merge pull request #17 from jambonz/fix/imterim_audio_cut
fix: use synthesized audio data directly from microsoft sdk
2023-03-31 20:03:38 -04:00
Quan HL df8acbed0e fix: audioData is getter 2023-04-01 06:58:55 +07:00
Quan HL 4296ed7256 fix: user synthesized audio data directly from microsoft sdk 2023-04-01 06:37:33 +07:00
Quan HL f4b271c7b3 fix: user synthesized audio data directly from microsoft sdk 2023-04-01 06:36:15 +07:00
Dave Horton d5c71de27d update to latest speech packages 2023-03-31 15:56:18 -04:00
Dave Horton 1ab7cb20e6 bump version 2023-03-24 14:47:35 -04:00
Dave Horton ba1052e629 Merge pull request #15 from jambonz/feature/get-tts-voices
add functions to retrieve voices for google and tts
2023-03-24 14:46:05 -04:00
Dave Horton 58270ad87f add functions to retrieve voices for google and tts 2023-03-24 14:43:51 -04:00
6 changed files with 1345 additions and 1410 deletions
+32 -1
View File
@@ -4,6 +4,8 @@ const getNuanceAccessToken = require('./get-nuance-access-token');
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const ttsGoogle = require('@google-cloud/text-to-speech');
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
const getIbmVoices = async(client, logger, credentials) => {
const {tts_region, tts_api_key} = credentials;
@@ -80,6 +82,30 @@ const getNuanceVoices = async(client, logger, credentials) => {
});
};
const getGoogleVoices = async(_client, logger, credentials) => {
const client = new ttsGoogle.TextToSpeechClient({credentials});
return await client.listVoices();
};
const getAwsVoices = async(_client, logger, credentials) => {
try {
const {region, accessKeyId, secretAccessKey} = credentials;
const client = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
const command = new DescribeVoicesCommand({LanguageCode: 'en-US'});
const response = await client.send(command);
return response;
} catch (err) {
logger.info({err}, 'testMicrosoftTts - failed to list voices for region ${region}');
throw err;
}
};
/**
* Synthesize speech to an mp3 file, and also cache the generated speech
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
@@ -99,7 +125,7 @@ const getNuanceVoices = async(client, logger, credentials) => {
async function getTtsVoices(client, logger, {vendor, credentials}) {
logger = logger || noopLogger;
assert.ok(['nuance', 'ibm'].includes(vendor),
assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly'].includes(vendor),
`getTtsVoices not supported for vendor ${vendor}`);
switch (vendor) {
@@ -107,6 +133,11 @@ async function getTtsVoices(client, logger, {vendor, credentials}) {
return getNuanceVoices(client, logger, credentials);
case 'ibm':
return getIbmVoices(client, logger, credentials);
case 'google':
return getGoogleVoices(client, logger, credentials);
case 'aws':
case 'polly':
return getAwsVoices(client, logger, credentials);
default:
break;
}
+2 -9
View File
@@ -2,14 +2,12 @@ const assert = require('assert');
const fs = require('fs');
const bent = require('bent');
const ttsGoogle = require('@google-cloud/text-to-speech');
//const Polly = require('aws-sdk/clients/polly');
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
const sdk = require('microsoft-cognitiveservices-speech-sdk');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const {
AudioConfig,
ResultReason,
SpeechConfig,
SpeechSynthesizer,
@@ -302,8 +300,7 @@ const synthMicrosoft = async(logger, {
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
const config = AudioConfig.fromAudioFileOutput(filePath);
const synthesizer = new SpeechSynthesizer(speechConfig, config);
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
@@ -329,11 +326,8 @@ const synthMicrosoft = async(logger, {
break;
case ResultReason.SynthesizingAudioCompleted:
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
resolve(result.audioData);
synthesizer.close();
fs.readFile(filePath, (err, data) => {
if (err) return reject(err);
resolve(data);
});
break;
default:
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
@@ -444,7 +438,6 @@ const synthNvidia = async(client, logger, {credentials, stats, language, voice,
request.setText(text);
return new Promise((resolve, reject) => {
console.log(`language ${language} voice ${voice} model ${model} text ${text}`);
rivaClient.synthesize(request, (err, response) => {
if (err) {
console.error(err);
+1095 -1388
View File
File diff suppressed because it is too large Load Diff
+10 -10
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.7",
"version": "0.0.10",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -24,18 +24,18 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.276.0",
"@google-cloud/text-to-speech": "^4.2.0",
"@grpc/grpc-js": "^1.8.7",
"@jambonz/realtimedb-helpers": "^0.6.3",
"aws-sdk": "^2.1310.0",
"@aws-sdk/client-polly": "^3.303.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"@jambonz/promisify-redis": "^0.0.6",
"bent": "^7.3.12",
"debug": "^4.3.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^7.1.2",
"form-urlencoded": "^6.1.0",
"microsoft-cognitiveservices-speech-sdk": "^1.25.0",
"undici": "^5.19.1"
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
"redis": "^3.1.2",
"undici": "^5.21.0"
},
"devDependencies": {
"config": "^3.3.9",
+1 -2
View File
@@ -1,5 +1,4 @@
require('./docker_start');
require('./synth');
require('./nuance');
require('./ibm');
require('./list-voices');
require('./docker_stop');
+205
View File
@@ -0,0 +1,205 @@
const test = require('tape').test ;
const config = require('config');
const opts = config.get('redis');
const fs = require('fs');
const logger = require('pino')({level: 'error'});
process.on('unhandledRejection', (reason, p) => {
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
});
const stats = {
increment: () => {},
histogram: () => {}
};
test('IBM - create access key', async(t) => {
const fn = require('..');
const {client, getIbmAccessToken} = fn(opts, logger);
if (!process.env.IBM_API_KEY ) {
t.pass('skipping IBM test since no IBM api_key provided');
t.end();
client.quit();
return;
}
try {
let obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM');
t.ok(obj.access_token && !obj.servedFromCache, 'successfull received access token from IBM');
obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM - second request');
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('IBM - retrieve tts voices test', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.IBM_TTS_API_KEY || !process.env.IBM_TTS_REGION) {
t.pass('skipping IBM test since no IBM api_key and/or region provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'ibm',
credentials: {
tts_api_key: process.env.IBM_TTS_API_KEY,
tts_region: process.env.IBM_TTS_REGION
}
};
const obj = await getTtsVoices(opts);
const {voices} = obj.result;
//console.log(JSON.stringify(voices));
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nuance hosted tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
t.pass('skipping Nuance hosted test since no Nuance client_id and secret provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'nuance',
credentials: {
client_id: process.env.NUANCE_CLIENT_ID,
secret: process.env.NUANCE_SECRET
}
};
let voices = await getTtsVoices(opts);
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nuance on-prem tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.NUANCE_TTS_URI ) {
t.pass('skipping Nuance on-prem test since no Nuance uri provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'nuance',
credentials: {
nuance_tts_uri: process.env.NUANCE_TTS_URI
}
};
let voices = await getTtsVoices(opts);
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Google tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
return t.end();
}
try {
const str = process.env.GCP_JSON_KEY || fs.readFileSync(process.env.GCP_FILE);
const credentials = JSON.parse(str);
const opts = {
vendor: 'google',
credentials
};
let result = await getTtsVoices(opts);
t.ok(result[0].voices.length > 0, `GetVoices: successfully retrieved ${result[0].voices.length} voices from Google`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('AWS tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
t.pass('skipping AWS speech synth tests since AWS_ACCESS_KEY_ID, AWS_SECRET_ACCESS_KEY, or AWS_REGION not provided');
return t.end();
}
try {
const opts = {
vendor: 'aws',
credentials: {
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
region: process.env.AWS_REGION,
}
};
let result = await getTtsVoices(opts);
t.ok(result?.Voices?.length > 0, `GetVoices: successfully retrieved ${result.Voices.length} voices from AWS`);
await client.flushallAsync();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});