mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
11746d3f22 | ||
|
|
3fcfbd10a1 | ||
|
|
df8acbed0e | ||
|
|
4296ed7256 | ||
|
|
f4b271c7b3 | ||
|
|
d5c71de27d | ||
|
|
1ab7cb20e6 | ||
|
|
ba1052e629 | ||
|
|
58270ad87f | ||
|
|
b172caee49 | ||
|
|
58374ca0fa | ||
|
|
5a7e0d37f4 | ||
|
|
6fa68bc712 | ||
|
|
d914b26cac | ||
|
|
7c7be6bbb1 | ||
|
|
46127bb763 | ||
|
|
17305325ff | ||
|
|
0e883cf82a | ||
|
|
2f3a766713 | ||
|
|
b429645ec2 | ||
|
|
0becf33bab | ||
|
|
3d3875741c |
+42
-6
@@ -1,9 +1,11 @@
|
|||||||
const assert = require('assert');
|
const assert = require('assert');
|
||||||
const {noopLogger, createNuanceClient} = require('./utils');
|
const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils');
|
||||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||||
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
|
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
|
||||||
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
||||||
const { IamAuthenticator } = require('ibm-watson/auth');
|
const { IamAuthenticator } = require('ibm-watson/auth');
|
||||||
|
const ttsGoogle = require('@google-cloud/text-to-speech');
|
||||||
|
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
|
||||||
|
|
||||||
const getIbmVoices = async(client, logger, credentials) => {
|
const getIbmVoices = async(client, logger, credentials) => {
|
||||||
const {tts_region, tts_api_key} = credentials;
|
const {tts_region, tts_api_key} = credentials;
|
||||||
@@ -21,15 +23,20 @@ const getIbmVoices = async(client, logger, credentials) => {
|
|||||||
};
|
};
|
||||||
|
|
||||||
const getNuanceVoices = async(client, logger, credentials) => {
|
const getNuanceVoices = async(client, logger, credentials) => {
|
||||||
const {client_id: clientId, secret: secret} = credentials;
|
const {client_id: clientId, secret: secret, nuance_tts_uri} = credentials;
|
||||||
|
|
||||||
return new Promise(async(resolve, reject) => {
|
return new Promise(async(resolve, reject) => {
|
||||||
/* get a nuance access token */
|
/* get a nuance access token */
|
||||||
let token, nuanceClient;
|
let token, nuanceClient;
|
||||||
try {
|
try {
|
||||||
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts');
|
if (nuance_tts_uri) {
|
||||||
token = access_token.access_token;
|
nuanceClient = await createKryptonClient(nuance_tts_uri);
|
||||||
nuanceClient = await createNuanceClient(token);
|
}
|
||||||
|
else {
|
||||||
|
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts');
|
||||||
|
token = access_token.access_token;
|
||||||
|
nuanceClient = await createNuanceClient(token);
|
||||||
|
}
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
logger.error({err}, 'getTtsVoices: error retrieving access token');
|
logger.error({err}, 'getTtsVoices: error retrieving access token');
|
||||||
return reject(err);
|
return reject(err);
|
||||||
@@ -75,6 +82,30 @@ const getNuanceVoices = async(client, logger, credentials) => {
|
|||||||
});
|
});
|
||||||
};
|
};
|
||||||
|
|
||||||
|
const getGoogleVoices = async(_client, logger, credentials) => {
|
||||||
|
const client = new ttsGoogle.TextToSpeechClient({credentials});
|
||||||
|
return await client.listVoices();
|
||||||
|
};
|
||||||
|
|
||||||
|
const getAwsVoices = async(_client, logger, credentials) => {
|
||||||
|
try {
|
||||||
|
const {region, accessKeyId, secretAccessKey} = credentials;
|
||||||
|
const client = new PollyClient({
|
||||||
|
region,
|
||||||
|
credentials: {
|
||||||
|
accessKeyId,
|
||||||
|
secretAccessKey
|
||||||
|
}
|
||||||
|
});
|
||||||
|
const command = new DescribeVoicesCommand({LanguageCode: 'en-US'});
|
||||||
|
const response = await client.send(command);
|
||||||
|
return response;
|
||||||
|
} catch (err) {
|
||||||
|
logger.info({err}, 'testMicrosoftTts - failed to list voices for region ${region}');
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Synthesize speech to an mp3 file, and also cache the generated speech
|
* Synthesize speech to an mp3 file, and also cache the generated speech
|
||||||
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
|
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
|
||||||
@@ -94,7 +125,7 @@ const getNuanceVoices = async(client, logger, credentials) => {
|
|||||||
async function getTtsVoices(client, logger, {vendor, credentials}) {
|
async function getTtsVoices(client, logger, {vendor, credentials}) {
|
||||||
logger = logger || noopLogger;
|
logger = logger || noopLogger;
|
||||||
|
|
||||||
assert.ok(['nuance', 'ibm'].includes(vendor),
|
assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly'].includes(vendor),
|
||||||
`getTtsVoices not supported for vendor ${vendor}`);
|
`getTtsVoices not supported for vendor ${vendor}`);
|
||||||
|
|
||||||
switch (vendor) {
|
switch (vendor) {
|
||||||
@@ -102,6 +133,11 @@ async function getTtsVoices(client, logger, {vendor, credentials}) {
|
|||||||
return getNuanceVoices(client, logger, credentials);
|
return getNuanceVoices(client, logger, credentials);
|
||||||
case 'ibm':
|
case 'ibm':
|
||||||
return getIbmVoices(client, logger, credentials);
|
return getIbmVoices(client, logger, credentials);
|
||||||
|
case 'google':
|
||||||
|
return getGoogleVoices(client, logger, credentials);
|
||||||
|
case 'aws':
|
||||||
|
case 'polly':
|
||||||
|
return getAwsVoices(client, logger, credentials);
|
||||||
default:
|
default:
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|||||||
+61
-26
@@ -2,21 +2,19 @@ const assert = require('assert');
|
|||||||
const fs = require('fs');
|
const fs = require('fs');
|
||||||
const bent = require('bent');
|
const bent = require('bent');
|
||||||
const ttsGoogle = require('@google-cloud/text-to-speech');
|
const ttsGoogle = require('@google-cloud/text-to-speech');
|
||||||
//const Polly = require('aws-sdk/clients/polly');
|
|
||||||
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
|
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
|
||||||
|
|
||||||
const sdk = require('microsoft-cognitiveservices-speech-sdk');
|
const sdk = require('microsoft-cognitiveservices-speech-sdk');
|
||||||
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
||||||
const { IamAuthenticator } = require('ibm-watson/auth');
|
const { IamAuthenticator } = require('ibm-watson/auth');
|
||||||
const {
|
const {
|
||||||
AudioConfig,
|
|
||||||
ResultReason,
|
ResultReason,
|
||||||
SpeechConfig,
|
SpeechConfig,
|
||||||
SpeechSynthesizer,
|
SpeechSynthesizer,
|
||||||
CancellationDetails,
|
CancellationDetails,
|
||||||
SpeechSynthesisOutputFormat
|
SpeechSynthesisOutputFormat
|
||||||
} = sdk;
|
} = sdk;
|
||||||
const {makeSynthKey, createNuanceClient, noopLogger, createRivaClient} = require('./utils');
|
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
|
||||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||||
const {
|
const {
|
||||||
SynthesisRequest,
|
SynthesisRequest,
|
||||||
@@ -32,7 +30,7 @@ const {
|
|||||||
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
|
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
|
||||||
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
|
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
|
||||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||||
const EXPIRES = process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 3600 * 24; // cache tts for 24 hours
|
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 24 * 60) * 60; // cache tts for 24 hours
|
||||||
const TMP_FOLDER = '/tmp';
|
const TMP_FOLDER = '/tmp';
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -75,8 +73,10 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
}
|
}
|
||||||
else if ('nuance' === vendor) {
|
else if ('nuance' === vendor) {
|
||||||
assert.ok(voice, 'synthAudio requires voice when nuance is used');
|
assert.ok(voice, 'synthAudio requires voice when nuance is used');
|
||||||
assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used');
|
if (!credentials.nuance_tts_uri) {
|
||||||
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used');
|
assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used');
|
||||||
|
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used');
|
||||||
|
}
|
||||||
}
|
}
|
||||||
else if ('nvidia' === vendor) {
|
else if ('nvidia' === vendor) {
|
||||||
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
|
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
|
||||||
@@ -155,7 +155,8 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
||||||
break;
|
break;
|
||||||
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
|
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
|
||||||
audioBuffer = await synthCustomVendor(logger, {credentials, stats, language, voice, text});
|
({ audioBuffer, filePath } = await synthCustomVendor(logger,
|
||||||
|
{credentials, stats, language, voice, text, filePath}));
|
||||||
break;
|
break;
|
||||||
default:
|
default:
|
||||||
assert(`synthAudio: unsupported speech vendor ${vendor}`);
|
assert(`synthAudio: unsupported speech vendor ${vendor}`);
|
||||||
@@ -183,7 +184,14 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
|
|
||||||
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => {
|
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => {
|
||||||
try {
|
try {
|
||||||
const polly = new PollyClient(credentials);
|
const {region, accessKeyId, secretAccessKey} = credentials;
|
||||||
|
const polly = new PollyClient({
|
||||||
|
region,
|
||||||
|
credentials: {
|
||||||
|
accessKeyId,
|
||||||
|
secretAccessKey
|
||||||
|
}
|
||||||
|
});
|
||||||
const opts = {
|
const opts = {
|
||||||
Engine: engine,
|
Engine: engine,
|
||||||
OutputFormat: 'mp3',
|
OutputFormat: 'mp3',
|
||||||
@@ -292,8 +300,7 @@ const synthMicrosoft = async(logger, {
|
|||||||
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
||||||
}
|
}
|
||||||
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||||
const config = AudioConfig.fromAudioFileOutput(filePath);
|
const synthesizer = new SpeechSynthesizer(speechConfig);
|
||||||
const synthesizer = new SpeechSynthesizer(speechConfig, config);
|
|
||||||
|
|
||||||
if (content.startsWith('<speak>')) {
|
if (content.startsWith('<speak>')) {
|
||||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||||
@@ -319,11 +326,8 @@ const synthMicrosoft = async(logger, {
|
|||||||
break;
|
break;
|
||||||
case ResultReason.SynthesizingAudioCompleted:
|
case ResultReason.SynthesizingAudioCompleted:
|
||||||
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
||||||
|
resolve(result.audioData);
|
||||||
synthesizer.close();
|
synthesizer.close();
|
||||||
fs.readFile(filePath, (err, data) => {
|
|
||||||
if (err) return reject(err);
|
|
||||||
resolve(data);
|
|
||||||
});
|
|
||||||
break;
|
break;
|
||||||
default:
|
default:
|
||||||
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
|
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
|
||||||
@@ -363,10 +367,16 @@ const synthWellSaid = async(logger, {credentials, stats, language, voice, gender
|
|||||||
};
|
};
|
||||||
|
|
||||||
const synthNuance = async(client, logger, {credentials, stats, voice, model, text}) => {
|
const synthNuance = async(client, logger, {credentials, stats, voice, model, text}) => {
|
||||||
/* get a nuance access token */
|
let nuanceClient;
|
||||||
const {client_id, secret} = credentials;
|
const {client_id, secret, nuance_tts_uri} = credentials;
|
||||||
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts');
|
if (nuance_tts_uri) {
|
||||||
const nuanceClient = await createNuanceClient(access_token);
|
nuanceClient = await createKryptonClient(nuance_tts_uri);
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
/* get a nuance access token */
|
||||||
|
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts');
|
||||||
|
nuanceClient = await createNuanceClient(access_token);
|
||||||
|
}
|
||||||
|
|
||||||
const v = new Voice();
|
const v = new Voice();
|
||||||
const p = new AudioParameters();
|
const p = new AudioParameters();
|
||||||
@@ -428,7 +438,6 @@ const synthNvidia = async(client, logger, {credentials, stats, language, voice,
|
|||||||
request.setText(text);
|
request.setText(text);
|
||||||
|
|
||||||
return new Promise((resolve, reject) => {
|
return new Promise((resolve, reject) => {
|
||||||
console.log(`language ${language} voice ${voice} model ${model} text ${text}`);
|
|
||||||
rivaClient.synthesize(request, (err, response) => {
|
rivaClient.synthesize(request, (err, response) => {
|
||||||
if (err) {
|
if (err) {
|
||||||
console.error(err);
|
console.error(err);
|
||||||
@@ -440,30 +449,56 @@ const synthNvidia = async(client, logger, {credentials, stats, language, voice,
|
|||||||
};
|
};
|
||||||
|
|
||||||
|
|
||||||
// CustomVendor accept only mp3
|
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text, filePath}) => {
|
||||||
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text}) => {
|
|
||||||
const {vendor, auth_token, custom_tts_url} = credentials;
|
const {vendor, auth_token, custom_tts_url} = credentials;
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const post = bent('POST', 'buffer', {
|
const post = bent('POST', {
|
||||||
'Authorization': `Bearer ${auth_token}`,
|
'Authorization': `Bearer ${auth_token}`,
|
||||||
'Accept': 'audio/mpeg',
|
|
||||||
'Content-Type': 'application/json'
|
'Content-Type': 'application/json'
|
||||||
});
|
});
|
||||||
|
|
||||||
const mp3 = await post(custom_tts_url, {
|
const response = await post(custom_tts_url, {
|
||||||
language,
|
language,
|
||||||
format: 'audio/mpeg',
|
|
||||||
voice,
|
voice,
|
||||||
type: text.startsWith('<speak>') ? 'ssml' : 'text',
|
type: text.startsWith('<speak>') ? 'ssml' : 'text',
|
||||||
text
|
text
|
||||||
});
|
});
|
||||||
|
|
||||||
return mp3;
|
const regex = /\.[^\.]*$/g;
|
||||||
|
const mime = response.headers['content-type'];
|
||||||
|
const buffer = await response.arrayBuffer();
|
||||||
|
return {
|
||||||
|
audioBuffer: buffer,
|
||||||
|
filePath: filePath.replace(regex, getFileExtFromMime(mime))
|
||||||
|
};
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
logger.info({err}, `Vendor ${vendor} returned error`);
|
logger.info({err}, `Vendor ${vendor} returned error`);
|
||||||
throw err;
|
throw err;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
const getFileExtFromMime = (mime) => {
|
||||||
|
switch (mime) {
|
||||||
|
case 'audio/wav':
|
||||||
|
case 'audio/x-wav':
|
||||||
|
return '.wav';
|
||||||
|
case /audio\/l16.*rate=8000/.test(mime) ? mime : 'cant match value':
|
||||||
|
return '.r8';
|
||||||
|
case /audio\/l16.*rate=16000/.test(mime) ? mime : 'cant match value':
|
||||||
|
return '.r16';
|
||||||
|
case /audio\/l16.*rate=24000/.test(mime) ? mime : 'cant match value':
|
||||||
|
return '.r24';
|
||||||
|
case /audio\/l16.*rate=32000/.test(mime) ? mime : 'cant match value':
|
||||||
|
return '.r32';
|
||||||
|
case /audio\/l16.*rate=48000/.test(mime) ? mime : 'cant match value':
|
||||||
|
return '.r48';
|
||||||
|
case 'audio/mpeg':
|
||||||
|
case 'audio/mp3':
|
||||||
|
return '.mp3';
|
||||||
|
default:
|
||||||
|
return '.wav';
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
module.exports = synthAudio;
|
module.exports = synthAudio;
|
||||||
|
|||||||
@@ -77,6 +77,11 @@ const getNuanceAccessToken = async(clientId, secret, scope = 'asr tts') => {
|
|||||||
return json.access_token;
|
return json.access_token;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
const createKryptonClient = async(uri) => {
|
||||||
|
const client = new SynthesizerClient(uri, grpc.credentials.createInsecure());
|
||||||
|
return client;
|
||||||
|
};
|
||||||
|
|
||||||
const createNuanceClient = async(access_token) => {
|
const createNuanceClient = async(access_token) => {
|
||||||
|
|
||||||
//if (nuanceClientMap.has(access_token)) return nuanceClientMap.get(access_token);
|
//if (nuanceClientMap.has(access_token)) return nuanceClientMap.get(access_token);
|
||||||
@@ -108,6 +113,7 @@ module.exports = {
|
|||||||
makeIbmKey,
|
makeIbmKey,
|
||||||
getNuanceAccessToken,
|
getNuanceAccessToken,
|
||||||
createNuanceClient,
|
createNuanceClient,
|
||||||
|
createKryptonClient,
|
||||||
createRivaClient,
|
createRivaClient,
|
||||||
makeBasicAuthHeader,
|
makeBasicAuthHeader,
|
||||||
NUANCE_AUTH_ENDPOINT,
|
NUANCE_AUTH_ENDPOINT,
|
||||||
|
|||||||
Generated
+1095
-1388
File diff suppressed because it is too large
Load Diff
+10
-10
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "0.0.2",
|
"version": "0.0.10",
|
||||||
"description": "TTS-related speech utilities for jambonz",
|
"description": "TTS-related speech utilities for jambonz",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"author": "Dave Horton",
|
"author": "Dave Horton",
|
||||||
@@ -24,18 +24,18 @@
|
|||||||
},
|
},
|
||||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@aws-sdk/client-polly": "^3.276.0",
|
"@aws-sdk/client-polly": "^3.303.0",
|
||||||
"@google-cloud/text-to-speech": "^4.2.0",
|
"@google-cloud/text-to-speech": "^4.2.1",
|
||||||
"@grpc/grpc-js": "^1.8.7",
|
"@grpc/grpc-js": "^1.8.13",
|
||||||
"@jambonz/realtimedb-helpers": "^0.6.3",
|
"@jambonz/promisify-redis": "^0.0.6",
|
||||||
"aws-sdk": "^2.1310.0",
|
|
||||||
"bent": "^7.3.12",
|
"bent": "^7.3.12",
|
||||||
"debug": "^4.3.4",
|
"debug": "^4.3.4",
|
||||||
"google-protobuf": "^3.21.2",
|
|
||||||
"ibm-watson": "^7.1.2",
|
|
||||||
"form-urlencoded": "^6.1.0",
|
"form-urlencoded": "^6.1.0",
|
||||||
"microsoft-cognitiveservices-speech-sdk": "^1.25.0",
|
"google-protobuf": "^3.21.2",
|
||||||
"undici": "^5.19.1"
|
"ibm-watson": "^8.0.0",
|
||||||
|
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
|
||||||
|
"redis": "^3.1.2",
|
||||||
|
"undici": "^5.21.0"
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
"config": "^3.3.9",
|
"config": "^3.3.9",
|
||||||
|
|||||||
+1
-2
@@ -1,5 +1,4 @@
|
|||||||
require('./docker_start');
|
require('./docker_start');
|
||||||
require('./synth');
|
require('./synth');
|
||||||
require('./nuance');
|
require('./list-voices');
|
||||||
require('./ibm');
|
|
||||||
require('./docker_stop');
|
require('./docker_stop');
|
||||||
|
|||||||
@@ -0,0 +1,205 @@
|
|||||||
|
const test = require('tape').test ;
|
||||||
|
const config = require('config');
|
||||||
|
const opts = config.get('redis');
|
||||||
|
const fs = require('fs');
|
||||||
|
const logger = require('pino')({level: 'error'});
|
||||||
|
process.on('unhandledRejection', (reason, p) => {
|
||||||
|
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
|
||||||
|
});
|
||||||
|
|
||||||
|
const stats = {
|
||||||
|
increment: () => {},
|
||||||
|
histogram: () => {}
|
||||||
|
};
|
||||||
|
|
||||||
|
test('IBM - create access key', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getIbmAccessToken} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.IBM_API_KEY ) {
|
||||||
|
t.pass('skipping IBM test since no IBM api_key provided');
|
||||||
|
t.end();
|
||||||
|
client.quit();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
let obj = await getIbmAccessToken(process.env.IBM_API_KEY);
|
||||||
|
//console.log({obj}, 'received access token from IBM');
|
||||||
|
t.ok(obj.access_token && !obj.servedFromCache, 'successfull received access token from IBM');
|
||||||
|
|
||||||
|
obj = await getIbmAccessToken(process.env.IBM_API_KEY);
|
||||||
|
//console.log({obj}, 'received access token from IBM - second request');
|
||||||
|
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
t.end();
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('IBM - retrieve tts voices test', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.IBM_TTS_API_KEY || !process.env.IBM_TTS_REGION) {
|
||||||
|
t.pass('skipping IBM test since no IBM api_key and/or region provided');
|
||||||
|
t.end();
|
||||||
|
client.quit();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const opts = {
|
||||||
|
vendor: 'ibm',
|
||||||
|
credentials: {
|
||||||
|
tts_api_key: process.env.IBM_TTS_API_KEY,
|
||||||
|
tts_region: process.env.IBM_TTS_REGION
|
||||||
|
}
|
||||||
|
};
|
||||||
|
const obj = await getTtsVoices(opts);
|
||||||
|
const {voices} = obj.result;
|
||||||
|
//console.log(JSON.stringify(voices));
|
||||||
|
t.ok(voices.length > 0 && voices[0].language,
|
||||||
|
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
|
||||||
|
t.end();
|
||||||
|
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('Nuance hosted tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
|
||||||
|
t.pass('skipping Nuance hosted test since no Nuance client_id and secret provided');
|
||||||
|
t.end();
|
||||||
|
client.quit();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const opts = {
|
||||||
|
vendor: 'nuance',
|
||||||
|
credentials: {
|
||||||
|
client_id: process.env.NUANCE_CLIENT_ID,
|
||||||
|
secret: process.env.NUANCE_SECRET
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let voices = await getTtsVoices(opts);
|
||||||
|
t.ok(voices.length > 0 && voices[0].language,
|
||||||
|
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
|
||||||
|
t.end();
|
||||||
|
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('Nuance on-prem tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.NUANCE_TTS_URI ) {
|
||||||
|
t.pass('skipping Nuance on-prem test since no Nuance uri provided');
|
||||||
|
t.end();
|
||||||
|
client.quit();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const opts = {
|
||||||
|
vendor: 'nuance',
|
||||||
|
credentials: {
|
||||||
|
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let voices = await getTtsVoices(opts);
|
||||||
|
t.ok(voices.length > 0 && voices[0].language,
|
||||||
|
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
|
||||||
|
t.end();
|
||||||
|
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('Google tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
|
||||||
|
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const str = process.env.GCP_JSON_KEY || fs.readFileSync(process.env.GCP_FILE);
|
||||||
|
const credentials = JSON.parse(str);
|
||||||
|
const opts = {
|
||||||
|
vendor: 'google',
|
||||||
|
credentials
|
||||||
|
};
|
||||||
|
let result = await getTtsVoices(opts);
|
||||||
|
t.ok(result[0].voices.length > 0, `GetVoices: successfully retrieved ${result[0].voices.length} voices from Google`);
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
|
||||||
|
t.end();
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('AWS tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
|
||||||
|
t.pass('skipping AWS speech synth tests since AWS_ACCESS_KEY_ID, AWS_SECRET_ACCESS_KEY, or AWS_REGION not provided');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const opts = {
|
||||||
|
vendor: 'aws',
|
||||||
|
credentials: {
|
||||||
|
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
|
||||||
|
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
|
||||||
|
region: process.env.AWS_REGION,
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let result = await getTtsVoices(opts);
|
||||||
|
t.ok(result?.Voices?.length > 0, `GetVoices: successfully retrieved ${result.Voices.length} voices from AWS`);
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
|
||||||
|
t.end();
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
+35
-4
@@ -12,12 +12,12 @@ const stats = {
|
|||||||
histogram: () => {}
|
histogram: () => {}
|
||||||
};
|
};
|
||||||
|
|
||||||
test('Nuance tests', async(t) => {
|
test('Nuance hosted tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {client, getTtsVoices} = fn(opts, logger);
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
|
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
|
||||||
t.pass('skipping Nuance test since no Nuance client_id and secret provided');
|
t.pass('skipping Nuance hosted test since no Nuance client_id and secret provided');
|
||||||
t.end();
|
t.end();
|
||||||
client.quit();
|
client.quit();
|
||||||
return;
|
return;
|
||||||
@@ -31,8 +31,39 @@ test('Nuance tests', async(t) => {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
let voices = await getTtsVoices(opts);
|
let voices = await getTtsVoices(opts);
|
||||||
//console.log(`received ${voices.length} voices from Nuance`);
|
t.ok(voices.length > 0 && voices[0].language,
|
||||||
//console.log(JSON.stringify(voices));
|
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||||
|
|
||||||
|
await client.flushallAsync();
|
||||||
|
|
||||||
|
t.end();
|
||||||
|
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('Nuance on-prem tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {client, getTtsVoices} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.NUANCE_TTS_URI ) {
|
||||||
|
t.pass('skipping Nuance on-prem test since no Nuance uri provided');
|
||||||
|
t.end();
|
||||||
|
client.quit();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const opts = {
|
||||||
|
vendor: 'nuance',
|
||||||
|
credentials: {
|
||||||
|
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let voices = await getTtsVoices(opts);
|
||||||
t.ok(voices.length > 0 && voices[0].language,
|
t.ok(voices.length > 0 && voices[0].language,
|
||||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||||
|
|
||||||
|
|||||||
+38
-2
@@ -212,7 +212,7 @@ test('Azure custom voice speech synth tests', async(t) => {
|
|||||||
client.quit();
|
client.quit();
|
||||||
});
|
});
|
||||||
|
|
||||||
test('Nuance speech synth tests', async(t) => {
|
test('Nuance hosted speech synth tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {synthAudio, client} = fn(opts, logger);
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
@@ -251,6 +251,43 @@ test('Nuance speech synth tests', async(t) => {
|
|||||||
client.quit();
|
client.quit();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('Nuance on-prem speech synth tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.NUANCE_TTS_URI) {
|
||||||
|
t.pass('skipping Nuance on prem speech synth tests since NUANCE_TTS_URI not provided');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
let opts = await synthAudio(stats, {
|
||||||
|
vendor: 'nuance',
|
||||||
|
credentials: {
|
||||||
|
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||||
|
},
|
||||||
|
language: 'en-US',
|
||||||
|
voice: 'Evan',
|
||||||
|
text: 'This is a test of on-prem. This is only a test',
|
||||||
|
});
|
||||||
|
t.ok(!opts.servedFromCache, `successfully synthesized nuance audio to ${opts.filePath}`);
|
||||||
|
|
||||||
|
opts = await synthAudio(stats, {
|
||||||
|
vendor: 'nuance',
|
||||||
|
credentials: {
|
||||||
|
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||||
|
},
|
||||||
|
language: 'en-US',
|
||||||
|
voice: 'Evan',
|
||||||
|
text: 'This is a test of on-prem. This is only a test',
|
||||||
|
});
|
||||||
|
t.ok(opts.servedFromCache, `successfully retrieved nuance audio from cache ${opts.filePath}`);
|
||||||
|
} catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
test('Nvidia speech synth tests', async(t) => {
|
test('Nvidia speech synth tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {synthAudio, client} = fn(opts, logger);
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
@@ -348,7 +385,6 @@ test('Custom Vendor speech synth tests', async(t) => {
|
|||||||
let obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew`);
|
let obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew`);
|
||||||
t.ok(obj.headers.Authorization == 'Bearer some_jwt_token', 'Custom Vendor Authentication Header is correct');
|
t.ok(obj.headers.Authorization == 'Bearer some_jwt_token', 'Custom Vendor Authentication Header is correct');
|
||||||
t.ok(obj.body.language == 'en-US', 'Custom Vendor Language is correct');
|
t.ok(obj.body.language == 'en-US', 'Custom Vendor Language is correct');
|
||||||
t.ok(obj.body.format == 'audio/mpeg', 'Custom Vendor format is correct');
|
|
||||||
t.ok(obj.body.voice == 'English-US.Female-1', 'Custom Vendor voice is correct');
|
t.ok(obj.body.voice == 'English-US.Female-1', 'Custom Vendor voice is correct');
|
||||||
t.ok(obj.body.type == 'text', 'Custom Vendor type is correct');
|
t.ok(obj.body.type == 'text', 'Custom Vendor type is correct');
|
||||||
t.ok(obj.body.text == 'This is a test. This is only a test', 'Custom Vendor text is correct');
|
t.ok(obj.body.text == 'This is a test. This is only a test', 'Custom Vendor text is correct');
|
||||||
|
|||||||
Reference in New Issue
Block a user