mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-04 07:43:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
eb2c39072b | ||
|
|
e5932ffc18 | ||
|
|
0a98c6a376 | ||
|
|
ea153e9833 | ||
|
|
b5daeff047 | ||
|
|
da02926c9a | ||
|
|
da3cdbb7aa | ||
|
|
625f147137 | ||
|
|
2e5687978e | ||
|
|
897481d34c | ||
|
|
bd5282e681 | ||
|
|
95a1384f02 | ||
|
|
35deeecf70 | ||
|
|
9d2ac3273f | ||
|
|
b1049aad7f | ||
|
|
40f51e7509 | ||
|
|
a0e2fe167c | ||
|
|
1fa853faa3 | ||
|
|
95e8d942b8 | ||
|
|
7a91876cd7 | ||
|
|
d07344ba3b | ||
|
|
44d8af2a96 | ||
|
|
b530db9a62 | ||
|
|
4c166c8eb4 | ||
|
|
8246dbea21 | ||
|
|
0084f6a468 | ||
|
|
98f679f43a | ||
|
|
7e7841b5ff | ||
|
|
75ce537db1 | ||
|
|
830be783b8 | ||
|
|
38c3219425 | ||
|
|
e37b96a9c2 | ||
|
|
c184fbae26 | ||
|
|
66d33ebd60 | ||
|
|
98e1bf62f9 | ||
|
|
63edbc2883 | ||
|
|
fb75de6af5 | ||
|
|
617f7af4af | ||
|
|
6c5c8e734f | ||
|
|
4349cd7e40 | ||
|
|
b6058ca242 | ||
|
|
68cbd63bbd | ||
|
|
521560e276 | ||
|
|
d606141f57 | ||
|
|
0d58954537 | ||
|
|
7c0eafded3 | ||
|
|
ab7d145288 | ||
|
|
11746d3f22 | ||
|
|
3fcfbd10a1 | ||
|
|
df8acbed0e | ||
|
|
4296ed7256 | ||
|
|
f4b271c7b3 | ||
|
|
d5c71de27d |
+3
-1
@@ -8,6 +8,8 @@
|
||||
},
|
||||
"redis-auth": {
|
||||
"host": "127.0.0.1",
|
||||
"port": 3380
|
||||
"port": 3380,
|
||||
"username": "daveh",
|
||||
"password": "foobarbazzle"
|
||||
}
|
||||
}
|
||||
@@ -1,15 +1,23 @@
|
||||
const {noopLogger} = require('./lib/utils');
|
||||
const promisify = require('@jambonz/promisify-redis');
|
||||
const redis = promisify(require('redis'));
|
||||
const Redis = require('ioredis');
|
||||
|
||||
module.exports = (opts, logger) => {
|
||||
const {host = '127.0.0.1', port = 6379, tls = false} = opts;
|
||||
logger = logger || noopLogger;
|
||||
const connectionOpts = {...opts};
|
||||
// Support legacy app
|
||||
if (process.env.JAMBONES_REDIS_USERNAME && process.env.JAMBONES_REDIS_PASSWORD) {
|
||||
if (Array.isArray(connectionOpts)) {
|
||||
for (const o of opts) {
|
||||
o.username = process.env.JAMBONES_REDIS_USERNAME;
|
||||
o.password = process.env.JAMBONES_REDIS_PASSWORD;
|
||||
}
|
||||
} else {
|
||||
connectionOpts.username = process.env.JAMBONES_REDIS_USERNAME;
|
||||
connectionOpts.password = process.env.JAMBONES_REDIS_PASSWORD;
|
||||
}
|
||||
}
|
||||
|
||||
const url = process.env.JAMBONES_REDIS_USERNAME && process.env.JAMBONES_REDIS_PASSWORD ?
|
||||
`${process.env.JAMBONES_REDIS_USERNAME}:${process.env.JAMBONES_REDIS_PASSWORD}@${host}:${port}` :
|
||||
`${host}:${port}`;
|
||||
const client = redis.createClient(tls ? `rediss://${url}` : `redis://${url}`);
|
||||
const client = new Redis(connectionOpts);
|
||||
['ready', 'connect', 'reconnecting', 'error', 'end', 'warning']
|
||||
.forEach((event) => {
|
||||
client.on(event, (...args) => {
|
||||
@@ -23,6 +31,7 @@ module.exports = (opts, logger) => {
|
||||
|
||||
return {
|
||||
client,
|
||||
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
|
||||
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
|
||||
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
|
||||
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
|
||||
|
||||
@@ -9,7 +9,7 @@ async function getIbmAccessToken(client, logger, apiKey) {
|
||||
logger = logger || noopLogger;
|
||||
try {
|
||||
const key = makeIbmKey(apiKey);
|
||||
const access_token = await client.getAsync(key);
|
||||
const access_token = await client.get(key);
|
||||
if (access_token) return {access_token, servedFromCache: true};
|
||||
|
||||
/* access token not found in cache, so fetch it from Ibm */
|
||||
|
||||
@@ -9,7 +9,7 @@ async function getNuanceAccessToken(client, logger, clientId, secret, scope) {
|
||||
logger = logger || noopLogger;
|
||||
try {
|
||||
const key = makeNuanceKey(clientId, secret, scope);
|
||||
const access_token = await client.getAsync(key);
|
||||
const access_token = await client.get(key);
|
||||
if (access_token) return {access_token, servedFromCache: true};
|
||||
|
||||
/* access token not found in cache, so fetch it from Nuance */
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
async function getTtsSize(client, logger, pattern = null) {
|
||||
let keys;
|
||||
if (pattern) {
|
||||
keys = await client.keys(pattern);
|
||||
} else {
|
||||
keys = await client.keys('tts:*');
|
||||
}
|
||||
return keys.length;
|
||||
}
|
||||
|
||||
module.exports = getTtsSize;
|
||||
@@ -19,12 +19,12 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
|
||||
|
||||
try {
|
||||
if (all) {
|
||||
const keys = await client.keysAsync('tts:*');
|
||||
purgedCount = await client.delAsync(keys);
|
||||
const keys = await client.keys('tts:*');
|
||||
purgedCount = await client.del(keys);
|
||||
|
||||
} else if (account_sid && !vendor && !language && !voice && !engine && !text) {
|
||||
const keys = await client.keysAsync(`tts:${account_sid}:*`);
|
||||
purgedCount = await client.delAsync(keys);
|
||||
const keys = await client.keys(`tts:${account_sid}:*`);
|
||||
purgedCount = await client.del(keys);
|
||||
}
|
||||
else {
|
||||
const key = makeSynthKey({
|
||||
@@ -35,7 +35,7 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
|
||||
engine,
|
||||
text,
|
||||
});
|
||||
purgedCount = await client.delAsync(key);
|
||||
purgedCount = await client.del(key);
|
||||
if (purgedCount === 0) error = 'Specified item not found';
|
||||
}
|
||||
|
||||
|
||||
+144
-34
@@ -2,21 +2,25 @@ const assert = require('assert');
|
||||
const fs = require('fs');
|
||||
const bent = require('bent');
|
||||
const ttsGoogle = require('@google-cloud/text-to-speech');
|
||||
//const Polly = require('aws-sdk/clients/polly');
|
||||
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
|
||||
|
||||
const sdk = require('microsoft-cognitiveservices-speech-sdk');
|
||||
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
||||
const { IamAuthenticator } = require('ibm-watson/auth');
|
||||
const {
|
||||
AudioConfig,
|
||||
ResultReason,
|
||||
SpeechConfig,
|
||||
SpeechSynthesizer,
|
||||
CancellationDetails,
|
||||
SpeechSynthesisOutputFormat
|
||||
} = sdk;
|
||||
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
|
||||
const {
|
||||
makeSynthKey,
|
||||
createNuanceClient,
|
||||
createKryptonClient,
|
||||
createRivaClient,
|
||||
noopLogger
|
||||
} = require('./utils');
|
||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||
const {
|
||||
SynthesisRequest,
|
||||
@@ -32,9 +36,27 @@ const {
|
||||
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
|
||||
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
|
||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 24 * 60) * 60; // cache tts for 24 hours
|
||||
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
|
||||
const TMP_FOLDER = '/tmp';
|
||||
|
||||
|
||||
const trimTrailingSilence = (buffer) => {
|
||||
assert.ok(buffer instanceof Buffer, 'trimTrailingSilence - argument is not a Buffer');
|
||||
|
||||
let offset = buffer.length;
|
||||
while (offset > 0) {
|
||||
// Get 16-bit value from the buffer (read in reverse)
|
||||
const value = buffer.readUInt16BE(offset - 2);
|
||||
if (value !== 0) {
|
||||
break;
|
||||
}
|
||||
offset -= 2;
|
||||
}
|
||||
|
||||
// Trim the silence from the end
|
||||
return offset === buffer.length ? buffer : buffer.subarray(0, offset);
|
||||
};
|
||||
|
||||
/**
|
||||
* Synthesize speech to an mp3 file, and also cache the generated speech
|
||||
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
|
||||
@@ -60,7 +82,8 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
let rtt;
|
||||
logger = logger || noopLogger;
|
||||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm'].includes(vendor) ||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft',
|
||||
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs'].includes(vendor) ||
|
||||
vendor.startsWith('custom'),
|
||||
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
|
||||
if ('google' === vendor) {
|
||||
@@ -83,7 +106,7 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
else if ('nvidia' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
|
||||
assert.ok(language, 'synthAudio requires language when nvidia is used');
|
||||
assert.ok(credentials.riva_uri, 'synthAudio requires riva_uri in credentials when nuance is used');
|
||||
assert.ok(credentials.riva_server_uri, 'synthAudio requires riva_server_uri in credentials when nvidia is used');
|
||||
}
|
||||
else if ('ibm' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when ibm is used');
|
||||
@@ -106,14 +129,19 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
text
|
||||
});
|
||||
let filePath;
|
||||
if (['nuance', 'nvidia'].includes(vendor)) {
|
||||
if (['nuance', 'nvidia'].includes(vendor) ||
|
||||
(
|
||||
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
||||
['microsoft', 'azure'].includes(vendor)
|
||||
)
|
||||
) {
|
||||
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
||||
}
|
||||
else filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.mp3`;
|
||||
debug(`synth key is ${key}`);
|
||||
let cached;
|
||||
if (!disableTtsCache) {
|
||||
cached = await client.getAsync(key);
|
||||
cached = await client.get(key);
|
||||
}
|
||||
if (cached) {
|
||||
// found in cache - extend the expiry and use it
|
||||
@@ -121,7 +149,7 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
servedFromCache = true;
|
||||
stats.increment('tts.cache.requests', ['found:yes']);
|
||||
audioBuffer = Buffer.from(cached, 'base64');
|
||||
client.expireAsync(key, EXPIRES).catch((err) => logger.info(err, 'Error setting expires'));
|
||||
client.expire(key, EXPIRES).catch((err) => logger.info(err, 'Error setting expires'));
|
||||
}
|
||||
if (!cached) {
|
||||
// not found in cache - go get it from speech vendor and add to cache
|
||||
@@ -156,6 +184,9 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
case 'wellsaid':
|
||||
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
||||
break;
|
||||
case 'elevenlabs':
|
||||
audioBuffer = await synthElevenlabs(logger, {credentials, stats, language, voice, text, filePath});
|
||||
break;
|
||||
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
|
||||
({ audioBuffer, filePath } = await synthCustomVendor(logger,
|
||||
{credentials, stats, language, voice, text, filePath}));
|
||||
@@ -170,10 +201,8 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
debug(`tts rtt time for ${text.length} chars on ${vendorLabel}: ${rtt}`);
|
||||
logger.info(`tts rtt time for ${text.length} chars on ${vendorLabel}: ${rtt}`);
|
||||
|
||||
client.setexAsync(key, EXPIRES, audioBuffer.toString('base64'))
|
||||
client.setex(key, EXPIRES, audioBuffer.toString('base64'))
|
||||
.catch((err) => logger.error(err, `error calling setex on key ${key}`));
|
||||
|
||||
if (['microsoft'].includes(vendor)) return {filePath, servedFromCache, rtt};
|
||||
}
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
@@ -278,6 +307,48 @@ const synthIbm = async(logger, {credentials, stats, voice, text}) => {
|
||||
}
|
||||
};
|
||||
|
||||
async function _synthOnPremMicrosoft(logger, {
|
||||
credentials,
|
||||
stats,
|
||||
language,
|
||||
voice,
|
||||
text,
|
||||
filePath
|
||||
}) {
|
||||
const {use_custom_tts, custom_tts_endpoint_url} = credentials;
|
||||
let content = text;
|
||||
|
||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||
/**
|
||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||
* Otherwise sending plain text we get "Voice does not match"
|
||||
*/
|
||||
content = `<speak>${text}</speak>`;
|
||||
}
|
||||
|
||||
if (content.startsWith('<speak>')) {
|
||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
|
||||
// eslint-disable-next-line max-len
|
||||
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
|
||||
logger.info({content}, 'synthMicrosoft');
|
||||
}
|
||||
|
||||
try {
|
||||
const trimSilence = filePath.endsWith('.r8');
|
||||
const post = bent('POST', 'buffer', {
|
||||
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
|
||||
'Content-Type': 'application/ssml+xml',
|
||||
'User-Agent': 'Jambonz'
|
||||
});
|
||||
const mp3 = await post(custom_tts_endpoint_url, content);
|
||||
return mp3;
|
||||
} catch (err) {
|
||||
logger.info({err}, '_synthMicrosoftByHttp returned error');
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
const synthMicrosoft = async(logger, {
|
||||
credentials,
|
||||
stats,
|
||||
@@ -287,23 +358,36 @@ const synthMicrosoft = async(logger, {
|
||||
filePath
|
||||
}) => {
|
||||
try {
|
||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
|
||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
|
||||
if (use_custom_tts && custom_tts_endpoint_url) {
|
||||
return await _synthOnPremMicrosoft(logger, {
|
||||
credentials,
|
||||
stats,
|
||||
language,
|
||||
voice,
|
||||
text,
|
||||
filePath
|
||||
});
|
||||
}
|
||||
const trimSilence = filePath.endsWith('.r8');
|
||||
let content = text;
|
||||
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
||||
speechConfig.speechSynthesisLanguage = language;
|
||||
speechConfig.speechSynthesisVoiceName = voice;
|
||||
if (use_custom_tts && custom_tts_endpoint) {
|
||||
speechConfig.endpointId = custom_tts_endpoint;
|
||||
|
||||
}
|
||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||
/**
|
||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||
* Otherwise sending plain text we get "Voice does not match"
|
||||
*/
|
||||
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
||||
content = `<speak>${text}</speak>`;
|
||||
}
|
||||
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||
const config = AudioConfig.fromAudioFileOutput(filePath);
|
||||
const synthesizer = new SpeechSynthesizer(speechConfig, config);
|
||||
speechConfig.speechSynthesisOutputFormat = trimSilence ?
|
||||
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
|
||||
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||
const synthesizer = new SpeechSynthesizer(speechConfig);
|
||||
|
||||
if (content.startsWith('<speak>')) {
|
||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||
@@ -328,12 +412,11 @@ const synthMicrosoft = async(logger, {
|
||||
reject(cancellation.errorDetails);
|
||||
break;
|
||||
case ResultReason.SynthesizingAudioCompleted:
|
||||
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
||||
let buffer = Buffer.from(result.audioData);
|
||||
if (trimSilence) buffer = trimTrailingSilence(buffer);
|
||||
resolve(buffer);
|
||||
synthesizer.close();
|
||||
fs.readFile(filePath, (err, data) => {
|
||||
if (err) return reject(err);
|
||||
resolve(data);
|
||||
});
|
||||
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
||||
break;
|
||||
default:
|
||||
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
|
||||
@@ -433,21 +516,25 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
|
||||
};
|
||||
|
||||
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
|
||||
const {riva_uri} = credentials;
|
||||
const rivaClient = await createRivaClient(riva_uri);
|
||||
|
||||
const request = new SynthesizeSpeechRequest();
|
||||
request.setVoiceName(voice);
|
||||
request.setLanguageCode(language);
|
||||
request.setSampleRateHz(8000);
|
||||
request.setEncoding(AudioEncoding.LINEAR_PCM);
|
||||
request.setText(text);
|
||||
const {riva_server_uri} = credentials;
|
||||
let rivaClient, request;
|
||||
try {
|
||||
rivaClient = await createRivaClient(riva_server_uri);
|
||||
request = new SynthesizeSpeechRequest();
|
||||
request.setVoiceName(voice);
|
||||
request.setLanguageCode(language);
|
||||
request.setSampleRateHz(8000);
|
||||
request.setEncoding(AudioEncoding.LINEAR_PCM);
|
||||
request.setText(text);
|
||||
} catch (err) {
|
||||
logger.info({err}, 'error creating riva client');
|
||||
return Promise.reject(err);
|
||||
}
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
console.log(`language ${language} voice ${voice} model ${model} text ${text}`);
|
||||
rivaClient.synthesize(request, (err, response) => {
|
||||
if (err) {
|
||||
console.error(err);
|
||||
logger.info({err, voice, language}, 'error synthesizing speech using Nvidia');
|
||||
return reject(err);
|
||||
}
|
||||
resolve(Buffer.from(response.getAudio()));
|
||||
@@ -485,6 +572,29 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
|
||||
}
|
||||
};
|
||||
|
||||
const synthElevenlabs = async(logger, {credentials, stats, language, voice, text}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
try {
|
||||
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
|
||||
'xi-api-key': api_key,
|
||||
'Accept': 'audio/mpeg',
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const mp3 = await post(`/v1/text-to-speech/${voice}`, {
|
||||
text,
|
||||
model_id,
|
||||
voice_settings: {
|
||||
stability: 0.5,
|
||||
similarity_boost: 0.5
|
||||
}
|
||||
});
|
||||
return mp3;
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synthEvenlabs returned error');
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const getFileExtFromMime = (mime) => {
|
||||
switch (mime) {
|
||||
case 'audio/wav':
|
||||
|
||||
@@ -106,7 +106,6 @@ const createRivaClient = async(rivaUri) => {
|
||||
return client;
|
||||
};
|
||||
|
||||
|
||||
module.exports = {
|
||||
makeSynthKey,
|
||||
makeNuanceKey,
|
||||
|
||||
Generated
+1764
-2223
File diff suppressed because it is too large
Load Diff
+10
-11
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.8",
|
||||
"version": "0.0.22",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -9,7 +9,7 @@
|
||||
"test": "test"
|
||||
},
|
||||
"scripts": {
|
||||
"test": "NODE_ENV=test JAMBONES_REDIS_USERNAME=daveh JAMBONES_REDIS_PASSWORD=foobarbazzle node test/ ",
|
||||
"test": "NODE_ENV=test node test/ ",
|
||||
"coverage": "nyc --reporter html --report-dir ./coverage npm run test",
|
||||
"jslint": "eslint index.js lib",
|
||||
"build": "./build_stubs.sh"
|
||||
@@ -24,18 +24,17 @@
|
||||
},
|
||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.276.0",
|
||||
"@google-cloud/text-to-speech": "^4.2.0",
|
||||
"@grpc/grpc-js": "^1.8.7",
|
||||
"@jambonz/realtimedb-helpers": "^0.6.3",
|
||||
"aws-sdk": "^2.1310.0",
|
||||
"@aws-sdk/client-polly": "^3.359.0",
|
||||
"@google-cloud/text-to-speech": "^4.2.1",
|
||||
"@grpc/grpc-js": "^1.8.13",
|
||||
"bent": "^7.3.12",
|
||||
"debug": "^4.3.4",
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^7.1.2",
|
||||
"form-urlencoded": "^6.1.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.25.0",
|
||||
"undici": "^5.19.1"
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^8.0.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
|
||||
"ioredis": "^5.3.2",
|
||||
"undici": "^5.21.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"config": "^3.3.9",
|
||||
|
||||
+2
-2
@@ -31,7 +31,7 @@ test('IBM - create access key', async(t) => {
|
||||
//console.log({obj}, 'received access token from IBM - second request');
|
||||
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
t.end();
|
||||
}
|
||||
catch (err) {
|
||||
@@ -65,7 +65,7 @@ test('IBM - retrieve tts voices test', async(t) => {
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
|
||||
|
||||
+6
-6
@@ -31,7 +31,7 @@ test('IBM - create access key', async(t) => {
|
||||
//console.log({obj}, 'received access token from IBM - second request');
|
||||
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
t.end();
|
||||
}
|
||||
catch (err) {
|
||||
@@ -65,7 +65,7 @@ test('IBM - retrieve tts voices test', async(t) => {
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
|
||||
@@ -99,7 +99,7 @@ test('Nuance hosted tests', async(t) => {
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
|
||||
@@ -132,7 +132,7 @@ test('Nuance on-prem tests', async(t) => {
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
|
||||
@@ -162,7 +162,7 @@ test('Google tests', async(t) => {
|
||||
let result = await getTtsVoices(opts);
|
||||
t.ok(result[0].voices.length > 0, `GetVoices: successfully retrieved ${result[0].voices.length} voices from Google`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
}
|
||||
@@ -193,7 +193,7 @@ test('AWS tests', async(t) => {
|
||||
let result = await getTtsVoices(opts);
|
||||
t.ok(result?.Voices?.length > 0, `GetVoices: successfully retrieved ${result.Voices.length} voices from AWS`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
}
|
||||
|
||||
+2
-2
@@ -34,7 +34,7 @@ test('Nuance hosted tests', async(t) => {
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
|
||||
@@ -67,7 +67,7 @@ test('Nuance on-prem tests', async(t) => {
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||
|
||||
await client.flushallAsync();
|
||||
await client.flushall();
|
||||
|
||||
t.end();
|
||||
|
||||
|
||||
+44
-12
@@ -300,7 +300,7 @@ test('Nvidia speech synth tests', async(t) => {
|
||||
let opts = await synthAudio(stats, {
|
||||
vendor: 'nvidia',
|
||||
credentials: {
|
||||
riva_uri: process.env.RIVA_URI,
|
||||
riva_server_uri: process.env.RIVA_URI,
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'English-US.Female-1',
|
||||
@@ -311,7 +311,7 @@ test('Nvidia speech synth tests', async(t) => {
|
||||
opts = await synthAudio(stats, {
|
||||
vendor: 'nvidia',
|
||||
credentials: {
|
||||
riva_uri: process.env.RIVA_URI,
|
||||
riva_server_uri: process.env.RIVA_URI,
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'English-US.Female-1',
|
||||
@@ -411,20 +411,52 @@ test('Custom Vendor speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('Elevenlabs speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.ELEVENLABS_API_KEY || !process.env.ELEVENLABS_VOICE_ID || !process.env.ELEVENLABS_MODEL_ID) {
|
||||
t.pass('skipping IBM Watson speech synth tests since IBM_TTS_API_KEY or IBM_TTS_API_KEY not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones!';
|
||||
try {
|
||||
let opts = await synthAudio(stats, {
|
||||
vendor: 'elevenlabs',
|
||||
credentials: {
|
||||
api_key: process.env.ELEVENLABS_API_KEY,
|
||||
model_id: process.env.ELEVENLABS_MODEL_ID
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: process.env.ELEVENLABS_VOICE_ID,
|
||||
text,
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized eleven audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
})
|
||||
|
||||
test('TTS Cache tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {purgeTtsCache, client} = fn(opts, logger);
|
||||
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);
|
||||
|
||||
try {
|
||||
// save some random tts keys to cache
|
||||
const minRecords = 8;
|
||||
for (const i in Array(minRecords).fill(0)) {
|
||||
await client.setAsync(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
}
|
||||
const count = await getTtsSize();
|
||||
t.ok(count >= minRecords, 'getTtsSize worked.');
|
||||
|
||||
const {purgedCount} = await purgeTtsCache();
|
||||
t.ok(purgedCount >= minRecords, `successfully purged at least ${minRecords} tts records from cache`);
|
||||
|
||||
const cached = (await client.keysAsync('tts:*')).length;
|
||||
const cached = (await client.keys('tts:*')).length;
|
||||
t.equal(cached, 0, `successfully purged all tts records from cache`);
|
||||
|
||||
} catch (err) {
|
||||
@@ -435,11 +467,11 @@ test('TTS Cache tests', async(t) => {
|
||||
try {
|
||||
// save some random tts keys to cache
|
||||
for (const i in Array(10).fill(0)) {
|
||||
await client.setAsync(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
}
|
||||
// save a specific key to tts cache
|
||||
const opts = {vendor: 'aws', language: 'en-US', voice: 'MALE', engine: 'Engine', text: 'Hello World!'};
|
||||
await client.setAsync(makeSynthKey(opts), opts.text);
|
||||
await client.set(makeSynthKey(opts), opts.text);
|
||||
|
||||
const {purgedCount} = await purgeTtsCache({all: false, ...opts});
|
||||
t.ok(purgedCount === 1, `successfully purged one specific tts record from cache`);
|
||||
@@ -455,7 +487,7 @@ test('TTS Cache tests', async(t) => {
|
||||
t.ok(error, `error returned when specified key was not found`);
|
||||
|
||||
// make sure other tts keys are still there
|
||||
const cached = (await client.keysAsync('tts:*')).length;
|
||||
const cached = (await client.keys('tts:*')).length;
|
||||
t.ok(cached >= 1, `successfully kept all non-specified tts records in cache`);
|
||||
|
||||
} catch (err) {
|
||||
@@ -471,21 +503,21 @@ test('TTS Cache tests', async(t) => {
|
||||
const account_sid = "12412512_cabc_5aff"
|
||||
const account_sid2 = "22412512_cabc_5aff"
|
||||
for (const i in Array(minRecords).fill(0)) {
|
||||
await client.setAsync(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
}
|
||||
for (const i in Array(minRecords).fill(0)) {
|
||||
await client.setAsync(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
}
|
||||
const {purgedCount} = await purgeTtsCache({account_sid});
|
||||
t.equal(purgedCount, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid}`);
|
||||
|
||||
let cached = (await client.keysAsync('tts:*')).length;
|
||||
let cached = (await client.keys('tts:*')).length;
|
||||
t.equal(cached, minRecords, `successfully purged all tts records from cache for account_sid:${account_sid}`);
|
||||
|
||||
const {purgedCount: purgedCount2} = await purgeTtsCache({account_sid: account_sid2});
|
||||
t.equal(purgedCount2, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid2}`);
|
||||
|
||||
cached = (await client.keysAsync('tts:*')).length;
|
||||
cached = (await client.keys('tts:*')).length;
|
||||
t.equal(cached, 0, `successfully purged all tts records from cache`);
|
||||
|
||||
} catch (err) {
|
||||
|
||||
Reference in New Issue
Block a user