Compare commits

..
73 Commits
Author SHA1 Message Date
Dave Horton 7df4e2f4c7 0.0.26 2023-11-14 08:48:14 -05:00
Dave Horton e99f7c5087 Merge pull request #43 from jambonz/feat/realtimedb
use realtimedb-helper for initiate redis connection
2023-11-10 07:49:28 -05:00
Quan HL c7d981c23a wip 2023-11-10 10:37:22 +07:00
Quan HL 95c54b9b12 use realtimedb-helper for initiate redis connection 2023-11-10 10:34:50 +07:00
Dave Horton 48192aeba1 Merge pull request #42 from jambonz/gh-actions
update gihub actions to test openai and elevenlabs
2023-11-09 08:44:14 -05:00
Dave Horton 8f931cd8a5 update gihub actions to test openai and elevenlabs 2023-11-09 08:42:32 -05:00
Dave Horton 144baafe94 0.0.25 2023-11-09 08:32:07 -05:00
Dave Horton ed3e513419 fix google speech test 2023-11-09 08:31:59 -05:00
Dave Horton 8d93fdc42a Merge pull request #41 from jambonz/feat/openai
support whisper tts
2023-11-09 08:30:24 -05:00
Quan HL ea523a7a1d wip 2023-11-09 12:59:20 +07:00
Quan HL 750ed97312 update review 2023-11-09 09:19:47 +07:00
Quan HL 73baa81177 update review 2023-11-09 09:15:54 +07:00
Quan HL f86234a769 support openai 2023-11-09 07:10:07 +07:00
Dave Horton d564f24e6c 0.0.24 2023-10-30 19:54:56 -04:00
Dave Horton 464d8462d9 Merge pull request #39 from jambonz/feat/google_custom_voice_01
fix google custom voice
2023-10-30 19:54:44 -04:00
Quan HL 6853f0e342 fix google custom voice 2023-10-31 06:24:52 +07:00
Dave Horton 507045dcba 0.0.23 2023-10-29 22:07:02 -04:00
Dave Horton 08758bbbff Merge pull request #38 from jambonz/feat/google_custom_voice
feat support google custom voice
2023-10-29 22:06:41 -04:00
Hoan Luu Huu a3aa1169b8 feat support google custom voice 2023-10-30 01:55:50 +00:00
Hoan Luu Huu 7cae19a4e5 feat support google custom voice 2023-10-30 01:54:02 +00:00
Hoan Luu Huu 8c4d5a7cee feat support google custom voice 2023-10-30 01:52:48 +00:00
Dave Horton eb2c39072b 0.0.22 2023-10-14 13:18:08 +02:00
Dave Horton e5932ffc18 Merge pull request #34 from jambonz/feat/elevenlabs
add elevenlabs
2023-10-14 07:17:14 -04:00
Quan HL 0a98c6a376 fix review comment 2023-10-14 18:13:04 +07:00
Quan HL ea153e9833 add elevenlabs 2023-10-12 14:22:15 +07:00
Dave Horton b5daeff047 0.0.21 2023-09-08 07:58:11 -04:00
Dave Horton da02926c9a Merge pull request #31 from jambonz/fix/onprem-azure
fix raw audio downloaded from onprem azure
2023-09-08 07:57:36 -04:00
Quan HL da3cdbb7aa fix raw audio downloaded from onprem azure 2023-09-08 15:42:57 +07:00
Dave Horton 625f147137 0.0.20 2023-08-30 21:08:52 -04:00
Dave Horton 2e5687978e 0.0.19 2023-08-30 21:08:41 -04:00
Dave Horton 897481d34c Merge pull request #29 from jambonz/feat/azure_fromhost
support self hosted microsoft
2023-08-30 21:07:44 -04:00
Quan HL bd5282e681 fix 2023-08-28 20:05:55 +07:00
Quan HL 95a1384f02 wip 2023-08-25 15:57:43 +07:00
Quan HL 35deeecf70 wip 2023-08-25 15:57:23 +07:00
Quan HL 9d2ac3273f wip 2023-08-25 13:53:09 +07:00
Quan HL b1049aad7f wip 2023-08-25 13:49:04 +07:00
Quan HL 40f51e7509 wip 2023-08-11 18:25:57 +07:00
Quan HL a0e2fe167c wip 2023-08-11 16:04:11 +07:00
Quan HL 1fa853faa3 wip 2023-08-11 13:30:11 +07:00
Quan HL 95e8d942b8 fix jslint 2023-08-09 18:00:17 +07:00
Quan HL 7a91876cd7 support self hosted microsoft 2023-08-09 17:56:38 +07:00
Dave Horton d07344ba3b Merge pull request #28 from jambonz/revert/ssml-silence-trim
revert change to _not_ trim silence when azure ssml is used
2023-07-25 12:34:22 -04:00
Dave Horton 44d8af2a96 revert change to _not_ trim silence when azure ssml is used 2023-07-25 11:08:06 -04:00
Dave Horton b530db9a62 0.0.18 2023-07-25 07:40:33 -04:00
Dave Horton 4c166c8eb4 synth_audio: dont trim silence for Azure when using SSML 2023-07-25 07:40:28 -04:00
Dave Horton 8246dbea21 0.0.17 2023-07-19 10:10:39 -04:00
Dave Horton 0084f6a468 update deps 2023-07-19 10:09:31 -04:00
Dave Horton 98f679f43a 0.0.16 2023-07-19 10:04:17 -04:00
Dave Horton 7e7841b5ff Merge pull request #23 from jambonz/feature/trim-silence
trim trailing silence from azure tts when JAMBONES_TTS_TRIM_SILENCE i…
2023-07-19 10:03:42 -04:00
Dave Horton 75ce537db1 linting 2023-07-19 10:02:10 -04:00
Dave Horton 830be783b8 trim trailing silence from azure tts when JAMBONES_TTS_TRIM_SILENCE is set 2023-07-19 10:00:31 -04:00
Dave Horton 38c3219425 Merge pull request #21 from jambonz/snyk-fix-255777535fcb48b70c964213f7dd9fe8
[Snyk] Security upgrade @aws-sdk/client-polly from 3.303.0 to 3.347.1
2023-06-07 13:09:27 -04:00
snyk-bot e37b96a9c2 fix: package.json & package-lock.json to reduce vulnerabilities
The following vulnerabilities are fixed with an upgrade:
- https://snyk.io/vuln/SNYK-JS-FASTXMLPARSER-5668858
2023-06-07 15:35:04 +00:00
Dave Horton c184fbae26 0.0.15 2023-06-03 09:15:48 -04:00
Dave Horton 66d33ebd60 change default tts cache duration to 4 hours 2023-06-03 09:15:45 -04:00
Dave Horton 98e1bf62f9 0.0.14 2023-05-31 11:15:38 -04:00
Dave Horton 63edbc2883 Merge pull request #20 from jambonz/feat/clear_tts
feat: ioredis and getsize of tts cache
2023-05-31 11:13:48 -04:00
Quan HL fb75de6af5 feat: ioredis and getsize of tts cache 2023-05-31 21:56:50 +07:00
Quan HL 617f7af4af feat: ioredis and getsize of tts cache 2023-05-31 21:53:18 +07:00
Dave Horton 6c5c8e734f 0.0.13 2023-05-10 07:39:21 -04:00
Dave Horton 4349cd7e40 Merge pull request #19 from jambonz/fix/nvidia
fixes for riva tts
2023-05-10 07:38:58 -04:00
Dave Horton b6058ca242 fix nvidia test 2023-05-10 07:36:25 -04:00
Dave Horton 68cbd63bbd minor logging 2023-05-09 13:59:38 -04:00
Dave Horton 521560e276 fixes for riva tts 2023-05-09 13:57:41 -04:00
Dave Horton d606141f57 fix issue in prev commit for microsoft 2023-04-01 13:19:27 -04:00
Dave Horton 0d58954537 bump version 2023-04-01 10:48:05 -04:00
Dave Horton 7c0eafded3 Merge pull request #18 from jambonz/fix/microsoft-buffer
fix: microsft retrun arrayBuffer not buffer, convert it now to buffer
2023-04-01 10:46:40 -04:00
Quan HL ab7d145288 fix: microsft retrun arrayBuffer not buffer, convert it now to buffer 2023-04-01 13:39:33 +07:00
Dave Horton 11746d3f22 bump version and minor changes 2023-03-31 20:05:07 -04:00
Dave Horton 3fcfbd10a1 Merge pull request #17 from jambonz/fix/imterim_audio_cut
fix: use synthesized audio data directly from microsoft sdk
2023-03-31 20:03:38 -04:00
Quan HL df8acbed0e fix: audioData is getter 2023-04-01 06:58:55 +07:00
Quan HL 4296ed7256 fix: user synthesized audio data directly from microsoft sdk 2023-04-01 06:37:33 +07:00
Quan HL f4b271c7b3 fix: user synthesized audio data directly from microsoft sdk 2023-04-01 06:36:15 +07:00
15 changed files with 2216 additions and 1839 deletions
+5 -1
View File
@@ -24,4 +24,8 @@ jobs:
IBM_TTS_API_KEY: ${{ secrets.IBM_TTS_API_KEY }}
IBM_TTS_REGION: ${{ secrets.IBM_TTS_REGION }}
MICROSOFT_API_KEY: ${{ secrets.MICROSOFT_API_KEY }}
MICROSOFT_REGION: ${{ secrets.MICROSOFT_REGION }}
MICROSOFT_REGION: ${{ secrets.MICROSOFT_REGION }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ELEVENLABS_API_KEY: ${{ secrets.ELEVENLABS_API_KEY }}
ELEVENLABS_VOICE_ID: ${{ secrets.ELEVENLABS_VOICE_ID }}
ELEVENLABS_MODEL_ID: ${{ secrets.ELEVENLABS_MODEL_ID }}
+3 -1
View File
@@ -8,6 +8,8 @@
},
"redis-auth": {
"host": "127.0.0.1",
"port": 3380
"port": 3380,
"username": "daveh",
"password": "foobarbazzle"
}
}
+6 -18
View File
@@ -1,28 +1,16 @@
const {noopLogger} = require('./lib/utils');
const promisify = require('@jambonz/promisify-redis');
const redis = promisify(require('redis'));
module.exports = (opts, logger) => {
const {host = '127.0.0.1', port = 6379, tls = false} = opts;
logger = logger || noopLogger;
const url = process.env.JAMBONES_REDIS_USERNAME && process.env.JAMBONES_REDIS_PASSWORD ?
`${process.env.JAMBONES_REDIS_USERNAME}:${process.env.JAMBONES_REDIS_PASSWORD}@${host}:${port}` :
`${host}:${port}`;
const client = redis.createClient(tls ? `rediss://${url}` : `redis://${url}`);
['ready', 'connect', 'reconnecting', 'error', 'end', 'warning']
.forEach((event) => {
client.on(event, (...args) => {
if ('error' === event) {
if (process.env.NODE_ENV === 'test' && args[0]?.code === 'ECONNREFUSED') return;
logger.error({...args}, '@jambonz/realtimedb-helpers - redis error');
}
else logger.debug({args}, `redis event ${event}`);
});
});
let client = opts.redis_client;
if (!client) {
const {client: redisClient} = require('@jambonz/realtimedb-helpers')(opts, logger);
client = redisClient;
}
return {
client,
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
+1 -1
View File
@@ -9,7 +9,7 @@ async function getIbmAccessToken(client, logger, apiKey) {
logger = logger || noopLogger;
try {
const key = makeIbmKey(apiKey);
const access_token = await client.getAsync(key);
const access_token = await client.get(key);
if (access_token) return {access_token, servedFromCache: true};
/* access token not found in cache, so fetch it from Ibm */
+1 -1
View File
@@ -9,7 +9,7 @@ async function getNuanceAccessToken(client, logger, clientId, secret, scope) {
logger = logger || noopLogger;
try {
const key = makeNuanceKey(clientId, secret, scope);
const access_token = await client.getAsync(key);
const access_token = await client.get(key);
if (access_token) return {access_token, servedFromCache: true};
/* access token not found in cache, so fetch it from Nuance */
+11
View File
@@ -0,0 +1,11 @@
async function getTtsSize(client, logger, pattern = null) {
let keys;
if (pattern) {
keys = await client.keys(pattern);
} else {
keys = await client.keys('tts:*');
}
return keys.length;
}
module.exports = getTtsSize;
+5 -5
View File
@@ -19,12 +19,12 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
try {
if (all) {
const keys = await client.keysAsync('tts:*');
purgedCount = await client.delAsync(keys);
const keys = await client.keys('tts:*');
purgedCount = await client.del(keys);
} else if (account_sid && !vendor && !language && !voice && !engine && !text) {
const keys = await client.keysAsync(`tts:${account_sid}:*`);
purgedCount = await client.delAsync(keys);
const keys = await client.keys(`tts:${account_sid}:*`);
purgedCount = await client.del(keys);
}
else {
const key = makeSynthKey({
@@ -35,7 +35,7 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
engine,
text,
});
purgedCount = await client.delAsync(key);
purgedCount = await client.del(key);
if (purgedCount === 0) error = 'Specified item not found';
}
+182 -34
View File
@@ -2,21 +2,25 @@ const assert = require('assert');
const fs = require('fs');
const bent = require('bent');
const ttsGoogle = require('@google-cloud/text-to-speech');
//const Polly = require('aws-sdk/clients/polly');
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
const sdk = require('microsoft-cognitiveservices-speech-sdk');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const {
AudioConfig,
ResultReason,
SpeechConfig,
SpeechSynthesizer,
CancellationDetails,
SpeechSynthesisOutputFormat
} = sdk;
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
const {
makeSynthKey,
createNuanceClient,
createKryptonClient,
createRivaClient,
noopLogger
} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const {
SynthesisRequest,
@@ -32,8 +36,27 @@ const {
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
const debug = require('debug')('jambonz:realtimedb-helpers');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 24 * 60) * 60; // cache tts for 24 hours
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp';
const OpenAI = require('openai');
const trimTrailingSilence = (buffer) => {
assert.ok(buffer instanceof Buffer, 'trimTrailingSilence - argument is not a Buffer');
let offset = buffer.length;
while (offset > 0) {
// Get 16-bit value from the buffer (read in reverse)
const value = buffer.readUInt16BE(offset - 2);
if (value !== 0) {
break;
}
offset -= 2;
}
// Trim the silence from the end
return offset === buffer.length ? buffer : buffer.subarray(0, offset);
};
/**
* Synthesize speech to an mp3 file, and also cache the generated speech
@@ -60,7 +83,8 @@ async function synthAudio(client, logger, stats, { account_sid,
let rtt;
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm'].includes(vendor) ||
assert.ok(['google', 'aws', 'polly', 'microsoft',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
if ('google' === vendor) {
@@ -83,7 +107,7 @@ async function synthAudio(client, logger, stats, { account_sid,
else if ('nvidia' === vendor) {
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
assert.ok(language, 'synthAudio requires language when nvidia is used');
assert.ok(credentials.riva_uri, 'synthAudio requires riva_uri in credentials when nuance is used');
assert.ok(credentials.riva_server_uri, 'synthAudio requires riva_server_uri in credentials when nvidia is used');
}
else if ('ibm' === vendor) {
assert.ok(voice, 'synthAudio requires voice when ibm is used');
@@ -94,6 +118,14 @@ async function synthAudio(client, logger, stats, { account_sid,
language = 'en-US'; // WellSaid only supports English atm
assert.ok(voice, 'synthAudio requires voice when wellsaid is used');
assert.ok(!text.startsWith('<speak'), 'wellsaid does not support SSML tags');
} else if ('elevenlabs' === vendor) {
assert.ok(voice, 'synthAudio requires voice when elevenlabs is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when elevenlabs is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when elevenlabs is used');
} else if ('whisper' === vendor) {
assert.ok(voice, 'synthAudio requires voice when whisper is used');
assert.ok(credentials.model_id, 'synthAudio requires model when whisper is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when whisper is used');
} else if (vendor.startsWith('custom')) {
assert.ok(credentials.custom_tts_url, `synthAudio requires custom_tts_url in credentials when ${vendor} is used`);
}
@@ -106,14 +138,19 @@ async function synthAudio(client, logger, stats, { account_sid,
text
});
let filePath;
if (['nuance', 'nvidia'].includes(vendor)) {
if (['nuance', 'nvidia'].includes(vendor) ||
(
process.env.JAMBONES_TTS_TRIM_SILENCE &&
['microsoft', 'azure'].includes(vendor)
)
) {
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
}
else filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.mp3`;
debug(`synth key is ${key}`);
let cached;
if (!disableTtsCache) {
cached = await client.getAsync(key);
cached = await client.get(key);
}
if (cached) {
// found in cache - extend the expiry and use it
@@ -121,7 +158,7 @@ async function synthAudio(client, logger, stats, { account_sid,
servedFromCache = true;
stats.increment('tts.cache.requests', ['found:yes']);
audioBuffer = Buffer.from(cached, 'base64');
client.expireAsync(key, EXPIRES).catch((err) => logger.info(err, 'Error setting expires'));
client.expire(key, EXPIRES).catch((err) => logger.info(err, 'Error setting expires'));
}
if (!cached) {
// not found in cache - go get it from speech vendor and add to cache
@@ -156,6 +193,12 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'wellsaid':
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
@@ -170,10 +213,8 @@ async function synthAudio(client, logger, stats, { account_sid,
debug(`tts rtt time for ${text.length} chars on ${vendorLabel}: ${rtt}`);
logger.info(`tts rtt time for ${text.length} chars on ${vendorLabel}: ${rtt}`);
client.setexAsync(key, EXPIRES, audioBuffer.toString('base64'))
client.setex(key, EXPIRES, audioBuffer.toString('base64'))
.catch((err) => logger.error(err, `error calling setex on key ${key}`));
if (['microsoft'].includes(vendor)) return {filePath, servedFromCache, rtt};
}
return new Promise((resolve, reject) => {
@@ -228,7 +269,8 @@ const synthGoogle = async(logger, {credentials, stats, language, voice, gender,
const client = new ttsGoogle.TextToSpeechClient(credentials);
const opts = {
voice: {
name: voice,
...(typeof voice === 'string' && {name: voice}),
...(typeof voice === 'object' && {customVoice: voice}),
languageCode: language,
ssmlGender: gender || 'SSML_VOICE_GENDER_UNSPECIFIED'
},
@@ -278,6 +320,48 @@ const synthIbm = async(logger, {credentials, stats, voice, text}) => {
}
};
async function _synthOnPremMicrosoft(logger, {
credentials,
stats,
language,
voice,
text,
filePath
}) {
const {use_custom_tts, custom_tts_endpoint_url} = credentials;
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
try {
const trimSilence = filePath.endsWith('.r8');
const post = bent('POST', 'buffer', {
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
'Content-Type': 'application/ssml+xml',
'User-Agent': 'Jambonz'
});
const mp3 = await post(custom_tts_endpoint_url, content);
return mp3;
} catch (err) {
logger.info({err}, '_synthMicrosoftByHttp returned error');
throw err;
}
}
const synthMicrosoft = async(logger, {
credentials,
stats,
@@ -287,23 +371,36 @@ const synthMicrosoft = async(logger, {
filePath
}) => {
try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
if (use_custom_tts && custom_tts_endpoint_url) {
return await _synthOnPremMicrosoft(logger, {
credentials,
stats,
language,
voice,
text,
filePath
});
}
const trimSilence = filePath.endsWith('.r8');
let content = text;
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
speechConfig.speechSynthesisLanguage = language;
speechConfig.speechSynthesisVoiceName = voice;
if (use_custom_tts && custom_tts_endpoint) {
speechConfig.endpointId = custom_tts_endpoint;
}
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
const config = AudioConfig.fromAudioFileOutput(filePath);
const synthesizer = new SpeechSynthesizer(speechConfig, config);
speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
@@ -328,12 +425,11 @@ const synthMicrosoft = async(logger, {
reject(cancellation.errorDetails);
break;
case ResultReason.SynthesizingAudioCompleted:
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
let buffer = Buffer.from(result.audioData);
if (trimSilence) buffer = trimTrailingSilence(buffer);
resolve(buffer);
synthesizer.close();
fs.readFile(filePath, (err, data) => {
if (err) return reject(err);
resolve(data);
});
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
break;
default:
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
@@ -433,20 +529,25 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
};
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
const {riva_uri} = credentials;
const rivaClient = await createRivaClient(riva_uri);
const request = new SynthesizeSpeechRequest();
request.setVoiceName(voice);
request.setLanguageCode(language);
request.setSampleRateHz(8000);
request.setEncoding(AudioEncoding.LINEAR_PCM);
request.setText(text);
const {riva_server_uri} = credentials;
let rivaClient, request;
try {
rivaClient = await createRivaClient(riva_server_uri);
request = new SynthesizeSpeechRequest();
request.setVoiceName(voice);
request.setLanguageCode(language);
request.setSampleRateHz(8000);
request.setEncoding(AudioEncoding.LINEAR_PCM);
request.setText(text);
} catch (err) {
logger.info({err}, 'error creating riva client');
return Promise.reject(err);
}
return new Promise((resolve, reject) => {
rivaClient.synthesize(request, (err, response) => {
if (err) {
console.error(err);
logger.info({err, voice, language}, 'error synthesizing speech using Nvidia');
return reject(err);
}
resolve(Buffer.from(response.getAudio()));
@@ -484,6 +585,53 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, stats, language, voice, text}) => {
const {api_key, model_id} = credentials;
try {
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
'xi-api-key': api_key,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(`/v1/text-to-speech/${voice}`, {
text,
model_id,
voice_settings: {
stability: 0.5,
similarity_boost: 0.5
}
});
return mp3;
} catch (err) {
logger.info({err}, 'synth Elevenlabs returned error');
stats.increment('tts.count', ['vendor:elevenlabs', 'accepted:no']);
throw err;
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
const {api_key, model_id, baseURL, timeout} = credentials;
try {
const openai = new OpenAI.OpenAI({
apiKey: api_key,
timeout: timeout || 5000,
...(baseURL && {baseURL})
});
const mp3 = await openai.audio.speech.create({
model: model_id,
voice,
input: text,
response_format: 'mp3'
});
return Buffer.from(await mp3.arrayBuffer());
} catch (err) {
logger.info({err}, 'synth whisper returned error');
stats.increment('tts.count', ['vendor:openai', 'accepted:no']);
throw err;
}
}
;
const getFileExtFromMime = (mime) => {
switch (mime) {
case 'audio/wav':
-1
View File
@@ -106,7 +106,6 @@ const createRivaClient = async(rivaUri) => {
return client;
};
module.exports = {
makeSynthKey,
makeNuanceKey,
+1876 -1746
View File
File diff suppressed because it is too large Load Diff
+6 -6
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.9",
"version": "0.0.26",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -9,7 +9,7 @@
"test": "test"
},
"scripts": {
"test": "NODE_ENV=test JAMBONES_REDIS_USERNAME=daveh JAMBONES_REDIS_PASSWORD=foobarbazzle node test/ ",
"test": "NODE_ENV=test node test/ ",
"coverage": "nyc --reporter html --report-dir ./coverage npm run test",
"jslint": "eslint index.js lib",
"build": "./build_stubs.sh"
@@ -24,17 +24,17 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.303.0",
"@aws-sdk/client-polly": "^3.359.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"@jambonz/promisify-redis": "^0.0.6",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
"debug": "^4.3.4",
"form-urlencoded": "^6.1.0",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
"redis": "^3.1.2",
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
"openai": "^4.16.2",
"undici": "^5.21.0"
},
"devDependencies": {
+2 -2
View File
@@ -31,7 +31,7 @@ test('IBM - create access key', async(t) => {
//console.log({obj}, 'received access token from IBM - second request');
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
await client.flushallAsync();
await client.flushall();
t.end();
}
catch (err) {
@@ -65,7 +65,7 @@ test('IBM - retrieve tts voices test', async(t) => {
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
await client.flushallAsync();
await client.flushall();
t.end();
+6 -6
View File
@@ -31,7 +31,7 @@ test('IBM - create access key', async(t) => {
//console.log({obj}, 'received access token from IBM - second request');
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
await client.flushallAsync();
await client.flushall();
t.end();
}
catch (err) {
@@ -65,7 +65,7 @@ test('IBM - retrieve tts voices test', async(t) => {
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
await client.flushallAsync();
await client.flushall();
t.end();
@@ -99,7 +99,7 @@ test('Nuance hosted tests', async(t) => {
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
await client.flushall();
t.end();
@@ -132,7 +132,7 @@ test('Nuance on-prem tests', async(t) => {
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
await client.flushall();
t.end();
@@ -162,7 +162,7 @@ test('Google tests', async(t) => {
let result = await getTtsVoices(opts);
t.ok(result[0].voices.length > 0, `GetVoices: successfully retrieved ${result[0].voices.length} voices from Google`);
await client.flushallAsync();
await client.flushall();
t.end();
}
@@ -193,7 +193,7 @@ test('AWS tests', async(t) => {
let result = await getTtsVoices(opts);
t.ok(result?.Voices?.length > 0, `GetVoices: successfully retrieved ${result.Voices.length} voices from AWS`);
await client.flushallAsync();
await client.flushall();
t.end();
}
+2 -2
View File
@@ -34,7 +34,7 @@ test('Nuance hosted tests', async(t) => {
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
await client.flushall();
t.end();
@@ -67,7 +67,7 @@ test('Nuance on-prem tests', async(t) => {
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
await client.flushallAsync();
await client.flushall();
t.end();
+110 -15
View File
@@ -38,7 +38,7 @@ test('Google speech synth tests', async(t) => {
},
},
language: 'en-GB',
gender: 'MALE',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
salt: 'foo.bar',
});
@@ -53,7 +53,7 @@ test('Google speech synth tests', async(t) => {
},
},
language: 'en-GB',
gender: 'MALE',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
});
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
@@ -68,7 +68,7 @@ test('Google speech synth tests', async(t) => {
},
disableTtsCache: true,
language: 'en-GB',
gender: 'MALE',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized google audio regardless of current cache to ${opts.filePath}`);
@@ -79,6 +79,40 @@ test('Google speech synth tests', async(t) => {
client.quit();
});
test('Google speech Custom voice synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_CUSTOM_VOICE_FILE && !process.env.GCP_CUSTOM_VOICE_JSON_KEY || !process.env.GCP_CUSTOM_VOICE_MODEL) {
t.pass('skipping google speech synth tests since neither GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_MODEL is not provided');
return t.end();
}
try {
const str = process.env.GCP_CUSTOM_VOICE_JSON_KEY || fs.readFileSync(process.env.GCP_CUSTOM_VOICE_FILE);
const creds = JSON.parse(str);
let opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-AU',
text: 'This is a test. This is only a test',
voice: {
reportedUsage:"REALTIME",
model: process.env.GCP_CUSTOM_VOICE_MODEL
}
});
t.ok(!opts.servedFromCache, `successfully synthesized google custom voice audio to ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('AWS speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -300,7 +334,7 @@ test('Nvidia speech synth tests', async(t) => {
let opts = await synthAudio(stats, {
vendor: 'nvidia',
credentials: {
riva_uri: process.env.RIVA_URI,
riva_server_uri: process.env.RIVA_URI,
},
language: 'en-US',
voice: 'English-US.Female-1',
@@ -311,7 +345,7 @@ test('Nvidia speech synth tests', async(t) => {
opts = await synthAudio(stats, {
vendor: 'nvidia',
credentials: {
riva_uri: process.env.RIVA_URI,
riva_server_uri: process.env.RIVA_URI,
},
language: 'en-US',
voice: 'English-US.Female-1',
@@ -411,20 +445,81 @@ test('Custom Vendor speech synth tests', async(t) => {
client.quit();
});
test('Elevenlabs speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.ELEVENLABS_API_KEY || !process.env.ELEVENLABS_VOICE_ID || !process.env.ELEVENLABS_MODEL_ID) {
t.pass('skipping ElevenLabs speech synth tests since ELEVENLABS_API_KEY or ELEVENLABS_VOICE_ID or ELEVENLABS_MODEL_ID not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'elevenlabs',
credentials: {
api_key: process.env.ELEVENLABS_API_KEY,
model_id: process.env.ELEVENLABS_MODEL_ID
},
language: 'en-US',
voice: process.env.ELEVENLABS_VOICE_ID,
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('whisper speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.OPENAI_API_KEY) {
t.pass('skipping OPENAI speech synth tests since OPENAI_API_KEY not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'whisper',
credentials: {
api_key: process.env.OPENAI_API_KEY,
model_id: 'tts-1'
},
language: 'en-US',
voice: 'alloy',
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('TTS Cache tests', async(t) => {
const fn = require('..');
const {purgeTtsCache, client} = fn(opts, logger);
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);
try {
// save some random tts keys to cache
const minRecords = 8;
for (const i in Array(minRecords).fill(0)) {
await client.setAsync(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
}
const count = await getTtsSize();
t.ok(count >= minRecords, 'getTtsSize worked.');
const {purgedCount} = await purgeTtsCache();
t.ok(purgedCount >= minRecords, `successfully purged at least ${minRecords} tts records from cache`);
const cached = (await client.keysAsync('tts:*')).length;
const cached = (await client.keys('tts:*')).length;
t.equal(cached, 0, `successfully purged all tts records from cache`);
} catch (err) {
@@ -435,11 +530,11 @@ test('TTS Cache tests', async(t) => {
try {
// save some random tts keys to cache
for (const i in Array(10).fill(0)) {
await client.setAsync(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
}
// save a specific key to tts cache
const opts = {vendor: 'aws', language: 'en-US', voice: 'MALE', engine: 'Engine', text: 'Hello World!'};
await client.setAsync(makeSynthKey(opts), opts.text);
await client.set(makeSynthKey(opts), opts.text);
const {purgedCount} = await purgeTtsCache({all: false, ...opts});
t.ok(purgedCount === 1, `successfully purged one specific tts record from cache`);
@@ -455,7 +550,7 @@ test('TTS Cache tests', async(t) => {
t.ok(error, `error returned when specified key was not found`);
// make sure other tts keys are still there
const cached = (await client.keysAsync('tts:*')).length;
const cached = (await client.keys('tts:*')).length;
t.ok(cached >= 1, `successfully kept all non-specified tts records in cache`);
} catch (err) {
@@ -471,21 +566,21 @@ test('TTS Cache tests', async(t) => {
const account_sid = "12412512_cabc_5aff"
const account_sid2 = "22412512_cabc_5aff"
for (const i in Array(minRecords).fill(0)) {
await client.setAsync(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
}
for (const i in Array(minRecords).fill(0)) {
await client.setAsync(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
}
const {purgedCount} = await purgeTtsCache({account_sid});
t.equal(purgedCount, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid}`);
let cached = (await client.keysAsync('tts:*')).length;
let cached = (await client.keys('tts:*')).length;
t.equal(cached, minRecords, `successfully purged all tts records from cache for account_sid:${account_sid}`);
const {purgedCount: purgedCount2} = await purgeTtsCache({account_sid: account_sid2});
t.equal(purgedCount2, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid2}`);
cached = (await client.keysAsync('tts:*')).length;
cached = (await client.keys('tts:*')).length;
t.equal(cached, 0, `successfully purged all tts records from cache`);
} catch (err) {