Compare commits

...
64 Commits
Author SHA1 Message Date
Dave Horton 531fa21f88 0.1.2 2024-05-28 12:55:00 -04:00
Dave Horton 904495d819 Merge pull request #73 from Catharsis68/feat/tts-cache-improvement
Improve handling of TTS cache by adding the file extension to the cac…
2024-05-28 12:54:31 -04:00
Markus Frindt f13fc84853 merge latest main into feature branch 2024-05-28 18:44:41 +02:00
Dave Horton e099bbb58f 0.1.1 2024-05-28 09:43:24 -04:00
Dave Horton e60a2d2ba3 Merge pull request #72 from jambonz/feat/verbio_speech
add verbio tts/stt
2024-05-28 09:42:28 -04:00
Markus Frindt 39d54050cc simplify return for streaming responses 2024-05-28 15:23:53 +02:00
Markus Frindt 7618d334db add namespace for custom provider 2024-05-27 11:04:35 +02:00
Markus Frindt 71f20178d3 add test case 2024-05-24 14:09:07 +02:00
Markus Frindt cb6ab2479f Improve handling of TTS cache by adding the file extension to the cache key 2024-05-24 14:03:31 +02:00
Quan HL 2212be341b add synthesize verbio 2024-05-20 18:11:01 +07:00
Quan HL 5d2d921f31 add verbio tts/stt 2024-05-20 17:24:06 +07:00
Dave Horton acb2d0c7ce Merge pull request #71 from jambonz/feat/azure_private_endpoint
support tts stream private endpoint
2024-05-14 06:56:51 -04:00
Quan HL 90d6048f52 fix private azure link with credential 2024-05-14 14:01:25 +07:00
Quan HL e5985620c0 support tts stream private endpoint 2024-05-05 14:28:11 +07:00
Dave Horton eb57f4d290 bump version 2024-05-02 07:46:47 -04:00
Dave Horton f0a1ab139c Merge pull request #69 from jambonz/feat/aws_polly_rolearn
support AWS Polly RoleArn credential
2024-05-02 07:44:38 -04:00
Quan HL 79289a7249 wip 2024-05-02 15:51:45 +07:00
Quan HL 5998eebdca wip 2024-05-02 15:48:20 +07:00
Quan HL e6a7017b55 wip 2024-05-02 14:20:24 +07:00
Quan HL c8571af129 wip 2024-04-30 15:47:14 +07:00
Quan HL 6144a9c164 wip 2024-04-30 15:45:26 +07:00
Quan HL 68a7f2b0d4 wip 2024-04-30 15:45:10 +07:00
Quan HL 0d1cd37097 wip 2024-04-30 11:20:21 +07:00
Quan HL 3de8e5ff57 accept aws polly without credential 2024-04-22 20:02:59 +07:00
Quan HL b4aad7991b update get aws voices 2024-04-22 16:19:04 +07:00
Quan HL 8a5c5c1966 support mod_google_tts 2024-04-19 16:06:53 +07:00
Quan HL 08a56d1a40 support AWS Polly RoleArn credential 2024-04-19 15:30:22 +07:00
Dave Horton ba61f20334 0.0.51 2024-04-12 07:19:56 -04:00
Dave Horton 10316d786e Merge pull request #67 from jambonz/feat/mod_rimelabs_tts
support mod_rimelabs_tts
2024-04-12 07:10:32 -04:00
Quan HL 38cdc106cc wip 2024-04-12 17:55:55 +07:00
Quan HL 410b99ef24 support mod_rimelabs_tts 2024-04-12 15:57:02 +07:00
Dave Horton 51db63f992 Merge pull request #66 from jambonz/gh-actions
add PlayHT to CI test
2024-04-08 09:49:48 -04:00
Dave Horton c86dceadde add PlayHT to CI test 2024-04-08 09:47:04 -04:00
Dave Horton f56f98f40f 0.0.50 2024-04-08 09:44:17 -04:00
Dave Horton 12a36593aa Merge pull request #65 from jambonz/feat/mod_playht_tts
support mod_playht_tts
2024-04-08 09:43:00 -04:00
Quan HL 8382491477 wip 2024-04-08 20:31:56 +07:00
Quan HL 545d559e27 wip 2024-04-08 19:14:58 +07:00
Quan HL cc8963802f wip 2024-04-08 17:33:21 +07:00
Quan HL 282d87f922 wip 2024-04-08 17:24:15 +07:00
Quan HL 6a288c1db0 support mod_playht_tts 2024-04-08 17:20:32 +07:00
Dave Horton 8650328e64 0.0.49 2024-04-07 12:15:02 -04:00
Dave Horton 85fa6da5b5 update to azure speech sdk 1.36.0 2024-04-07 12:14:55 -04:00
Dave Horton e54e913fdd 0.0.48 2024-04-04 15:52:40 -04:00
Dave Horton b088c0d7d9 Merge pull request #63 from jambonz/feat/mod_deepgram_tts
Feat/mod deepgram tts
2024-04-04 15:52:10 -04:00
Hoan Luu Huu f154a40692 Merge branch 'main' into feat/mod_deepgram_tts 2024-04-04 18:58:38 +07:00
Quan HL 0471b94ebe wip 2024-04-04 10:12:24 +07:00
Dave Horton 8279891dff 0.0.47 2024-04-03 13:33:24 -04:00
Dave Horton f546ca998d cache files for azure tts streaming are r8 2024-04-03 13:32:32 -04:00
Dave Horton bf229d0ab0 0.0.46 2024-04-03 13:22:20 -04:00
Dave Horton f08fedb8ca enable caching from azure tts streaming 2024-04-03 13:17:36 -04:00
Quan HL f3cc38089c mod_deepgra_tts 2024-04-03 20:46:52 +07:00
Dave Horton fbed59e5de 0.0.45 2024-04-02 15:10:54 -04:00
Dave Horton 4f1685a365 Merge pull request #59 from jambonz/feat/azure_tts
support azure streaming
2024-03-30 09:21:07 -04:00
Quan HL 2701af102a wip 2024-03-30 17:49:14 +07:00
Quan HL 7f939b96d2 wip 2024-03-30 17:38:00 +07:00
Quan HL 4d58ca6daf wip 2024-03-30 17:34:49 +07:00
Hoan Luu Huu 16dd7a2805 Merge branch 'main' into feat/azure_tts 2024-03-30 17:04:01 +07:00
Quan HL f06f96a6f0 wip 2024-03-10 06:41:46 +07:00
Hoan Luu Huu 2988e800b1 Merge branch 'main' into feat/azure_tts 2024-03-10 06:39:00 +07:00
Quan HL 31a54f595b wip 2024-02-26 15:42:09 +07:00
Quan HL 3560a6d4d9 wip 2024-02-26 14:01:53 +07:00
Quan HL be8053db4f wip 2024-02-26 13:54:34 +07:00
Quan HL 4ffae38a3f wip 2024-02-26 13:49:37 +07:00
Quan HL 9e74760c39 support azure streaming 2024-02-26 13:33:22 +07:00
19 changed files with 758 additions and 155 deletions
+4 -2
View File
@@ -8,8 +8,8 @@ jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v3
- uses: actions/setup-node@v3
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: lts/*
- run: npm install
@@ -32,5 +32,7 @@ jobs:
ELEVENLABS_API_KEY: ${{ secrets.ELEVENLABS_API_KEY }}
ELEVENLABS_VOICE_ID: ${{ secrets.ELEVENLABS_VOICE_ID }}
ELEVENLABS_MODEL_ID: ${{ secrets.ELEVENLABS_MODEL_ID }}
PLAYHT_USER_ID: ${{ secrets.PLAYHT_USER_ID }}
PLAYHT_API_KEY: ${{ secrets.PLAYHT_API_KEY }}
JAMBONES_HTTP_PROXY_IP: 127.0.0.1
JAMBONES_HTTP_PROXY_PORT: 3128
+2
View File
@@ -39,3 +39,5 @@ node_modules
examples/*
.vscode
.env
+3 -2
View File
@@ -13,10 +13,11 @@ module.exports = (opts, logger) => {
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger),
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
synthAudio: require('./lib/synth-audio').bind(null, client, createHash, retrieveHash, logger),
getVerbioAccessToken: require('./lib/get-verbio-token').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
getAwsAuthToken: require('./lib/get-aws-sts-token').bind(null, logger, createHash, retrieveHash),
getTtsVoices: require('./lib/get-tts-voices').bind(null, client, logger),
getTtsVoices: require('./lib/get-tts-voices').bind(null, client, createHash, retrieveHash, logger),
};
};
+2 -1
View File
@@ -1,6 +1,7 @@
const fs = require('fs/promises');
const {noopLogger, makeSynthKey} = require('./utils');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
const {JAMBONES_TTS_CACHE_DURATION_MINS} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
async function addFileToCache(client, logger, path,
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
+20
View File
@@ -0,0 +1,20 @@
const JAMBONES_TTS_TRIM_SILENCE = process.env.JAMBONES_TTS_TRIM_SILENCE;
const JAMBONES_DISABLE_TTS_STREAMING = process.env.JAMBONES_DISABLE_TTS_STREAMING;
const JAMBONES_HTTP_PROXY_IP = process.env.JAMBONES_HTTP_PROXY_IP;
const JAMBONES_HTTP_PROXY_PORT = process.env.JAMBONES_HTTP_PROXY_PORT;
const JAMBONES_TTS_CACHE_DURATION_MINS = (parseInt(process.env.JAMBONES_TTS_CACHE_DURATION_MINS) || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp';
const HTTP_TIMEOUT = 5000;
module.exports = {
JAMBONES_TTS_TRIM_SILENCE,
JAMBONES_DISABLE_TTS_STREAMING,
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
TMP_FOLDER,
HTTP_TIMEOUT
};
+24 -15
View File
@@ -1,32 +1,41 @@
const { STSClient, GetSessionTokenCommand } = require('@aws-sdk/client-sts');
const { STSClient, GetSessionTokenCommand, AssumeRoleCommand } = require('@aws-sdk/client-sts');
const {makeAwsKey, noopLogger} = require('./utils');
const debug = require('debug')('jambonz:speech-utils');
const EXPIRY = 3600;
async function getAwsAuthToken(
logger,
createHash, retrieveHash,
awsAccessKeyId, awsSecretAccessKey, awsRegion) {
logger, createHash, retrieveHash,
awsAccessKeyId, awsSecretAccessKey, awsRegion, roleArn = null) {
logger = logger || noopLogger;
try {
const key = makeAwsKey(awsAccessKeyId);
const key = makeAwsKey(roleArn || awsAccessKeyId);
const obj = await retrieveHash(key);
if (obj) return {...obj, servedFromCache: true};
/* access token not found in cache, so generate it using STS */
const stsClient = new STSClient({
region: awsRegion,
credentials: {
accessKeyId: awsAccessKeyId,
secretAccessKey: awsSecretAccessKey,
}
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
const data = await stsClient.send(command);
let data;
if (roleArn) {
const stsClient = new STSClient({ region: awsRegion});
const roleToAssume = { RoleArn: roleArn, RoleSessionName: 'Jambonz_Speech', DurationSeconds: EXPIRY};
const command = new AssumeRoleCommand(roleToAssume);
data = await stsClient.send(command);
} else {
/* access token not found in cache, so generate it using STS */
const stsClient = new STSClient({
region: awsRegion,
credentials: {
accessKeyId: awsAccessKeyId,
secretAccessKey: awsSecretAccessKey,
}
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
data = await stsClient.send(command);
}
const credentials = {
accessKeyId: data.Credentials.AccessKeyId,
secretAccessKey: data.Credentials.SecretAccessKey,
sessionToken: data.Credentials.SessionToken,
securityToken: data.Credentials.SessionToken
};
+1 -1
View File
@@ -2,8 +2,8 @@ const formurlencoded = require('form-urlencoded');
const {Pool} = require('undici');
const pool = new Pool('https://iam.cloud.ibm.com');
const {makeIbmKey, noopLogger} = require('./utils');
const { HTTP_TIMEOUT } = require('./config');
const debug = require('debug')('jambonz:realtimedb-helpers');
const HTTP_TIMEOUT = 5000;
async function getIbmAccessToken(client, logger, apiKey) {
logger = logger || noopLogger;
+1 -1
View File
@@ -2,8 +2,8 @@ const formurlencoded = require('form-urlencoded');
const {Pool} = require('undici');
const pool = new Pool('https://auth.crt.nuance.com');
const {makeNuanceKey, makeBasicAuthHeader, noopLogger} = require('./utils');
const { HTTP_TIMEOUT } = require('./config');
const debug = require('debug')('jambonz:realtimedb-helpers');
const HTTP_TIMEOUT = 5000;
async function getNuanceAccessToken(client, logger, clientId, secret, scope) {
logger = logger || noopLogger;
+49 -12
View File
@@ -1,11 +1,16 @@
const assert = require('assert');
const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const ttsGoogle = require('@google-cloud/text-to-speech');
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
const getAwsAuthToken = require('./get-aws-sts-token');
const {Pool} = require('undici');
const { HTTP_TIMEOUT } = require('./config');
const verbioVoicePool = new Pool('https://us.rest.speechcenter.verbio.com');
const getIbmVoices = async(client, logger, credentials) => {
const {tts_region, tts_api_key} = credentials;
@@ -87,16 +92,26 @@ const getGoogleVoices = async(_client, logger, credentials) => {
return await client.listVoices();
};
const getAwsVoices = async(_client, logger, credentials) => {
const getAwsVoices = async(_client, createHash, retrieveHash, logger, credentials) => {
try {
const {region, accessKeyId, secretAccessKey} = credentials;
const client = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
const {region, accessKeyId, secretAccessKey, roleArn} = credentials;
let client = null;
if (accessKeyId && secretAccessKey) {
client = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
} else if (roleArn) {
client = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
});
} else {
client = new PollyClient({region});
}
const command = new DescribeVoicesCommand({});
const response = await client.send(command);
return response;
@@ -106,6 +121,26 @@ const getAwsVoices = async(_client, logger, credentials) => {
}
};
const getVerbioVoices = async(client, logger, credentials) => {
try {
const access_token = await getVerbioAccessToken(client, logger, credentials);
const { body} = await verbioVoicePool.request({
path: '/api/v1/voices',
method: 'GET',
headers: {
'Authorization': `Bearer ${access_token.access_token}`,
'User-Agent': 'jambonz'
},
timeout: HTTP_TIMEOUT,
followRedirects: false
});
return await body.json();
} catch (err) {
logger.info({err}, 'getVerbioVoices - failed to list voices for Verbio');
throw err;
}
};
/**
* Synthesize speech to an mp3 file, and also cache the generated speech
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
@@ -122,10 +157,10 @@ const getAwsVoices = async(_client, logger, credentials) => {
* @returns object containing filepath to an mp3 file in the /tmp folder containing
* the synthesized audio, and a variable indicating whether it was served from cache
*/
async function getTtsVoices(client, logger, {vendor, credentials}) {
async function getTtsVoices(client, createHash, retrieveHash, logger, {vendor, credentials}) {
logger = logger || noopLogger;
assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly'].includes(vendor),
assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly', 'verbio'].includes(vendor),
`getTtsVoices not supported for vendor ${vendor}`);
switch (vendor) {
@@ -137,7 +172,9 @@ async function getTtsVoices(client, logger, {vendor, credentials}) {
return getGoogleVoices(client, logger, credentials);
case 'aws':
case 'polly':
return getAwsVoices(client, logger, credentials);
return getAwsVoices(client, createHash, retrieveHash, logger, credentials);
case 'verbio':
return getVerbioVoices(client, logger, credentials);
default:
break;
}
+51
View File
@@ -0,0 +1,51 @@
const {Pool} = require('undici');
const { noopLogger, makeVerbioKey } = require('./utils');
const { HTTP_TIMEOUT } = require('./config');
const pool = new Pool('https://auth.speechcenter.verbio.com:444');
const debug = require('debug')('jambonz:realtimedb-helpers');
async function getVerbioAccessToken(client, logger, credentials) {
logger = logger || noopLogger;
const { client_id, client_secret } = credentials;
try {
const key = makeVerbioKey(client_id);
const access_token = await client.get(key);
if (access_token) {
return {access_token, servedFromCache: true};
}
const payload = {
client_id,
client_secret
};
const {statusCode, headers, body} = await pool.request({
path: '/api/v1/token',
method: 'POST',
headers: {
'Content-Type': 'application/json',
'User-Agent': 'jambonz'
},
body: JSON.stringify(payload),
timeout: HTTP_TIMEOUT,
followRedirects: false
});
if (200 !== statusCode) {
logger.debug({statusCode, headers, body: await body.text()}, 'error fetching access token from Verbio');
const err = new Error();
err.statusCode = statusCode;
throw err;
}
const json = await body.json();
const expiry = Math.floor(json.expiration_time - Date.now() / 1000 - 30);
await client.set(key, json.access_token, 'EX', expiry);
return {...json, servedFromCache: false};
} catch (err) {
debug(err, `getVerbioAccessToken: Error retrieving Verbio access token for client_id ${client_id}`);
logger.error(err, `getVerbioAccessToken: Error retrieving Verbio access token for client_id ${client_id}`);
throw err;
}
}
module.exports = getVerbioAccessToken;
+275 -59
View File
@@ -19,9 +19,11 @@ const {
createNuanceClient,
createKryptonClient,
createRivaClient,
noopLogger
noopLogger,
makeFilePath
} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
const {
SynthesisRequest,
Voice,
@@ -36,9 +38,15 @@ const {
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
const debug = require('debug')('jambonz:realtimedb-helpers');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp';
const {
JAMBONES_DISABLE_TTS_STREAMING,
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
const OpenAI = require('openai');
const getAwsAuthToken = require('./get-aws-sts-token');
const trimTrailingSilence = (buffer) => {
@@ -75,7 +83,7 @@ const trimTrailingSilence = (buffer) => {
* @returns object containing filepath to an mp3 file in the /tmp folder containing
* the synthesized audio, and a variable indicating whether it was served from cache
*/
async function synthAudio(client, logger, stats, { account_sid,
async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, disableTtsStreaming, options
}) {
@@ -84,10 +92,10 @@ async function synthAudio(client, logger, stats, { account_sid,
let rtt;
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper', 'deepgram'].includes(vendor) ||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs',
'whisper', 'deepgram', 'playht', 'rimelabs', 'verbio'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid ..etc, not ${vendor}`);
if ('google' === vendor) {
assert.ok(language, 'synthAudio requires language when google is used');
}
@@ -123,12 +131,25 @@ async function synthAudio(client, logger, stats, { account_sid,
assert.ok(voice, 'synthAudio requires voice when elevenlabs is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when elevenlabs is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when elevenlabs is used');
} else if ('playht' === vendor) {
assert.ok(voice, 'synthAudio requires voice when playht is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when playht is used');
assert.ok(credentials.user_id, 'synthAudio requires user_id when playht is used');
assert.ok(credentials.voice_engine, 'synthAudio requires voice_engine when playht is used');
} else if ('rimelabs' === vendor) {
assert.ok(voice, 'synthAudio requires voice when rimelabs is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when rimelabs is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when rimelabs is used');
} else if ('whisper' === vendor) {
assert.ok(voice, 'synthAudio requires voice when whisper is used');
assert.ok(credentials.model_id, 'synthAudio requires model when whisper is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when whisper is used');
} else if (vendor.startsWith('custom')) {
assert.ok(credentials.custom_tts_url, `synthAudio requires custom_tts_url in credentials when ${vendor} is used`);
} else if ('verbio' === vendor) {
assert.ok(voice, 'synthAudio requires voice when verbio is used');
assert.ok(credentials.client_id, 'synthAudio requires client_id when verbio is used');
assert.ok(credentials.client_secret, 'synthAudio requires client_secret when verbio is used');
}
const key = makeSynthKey({
account_sid,
@@ -139,19 +160,7 @@ async function synthAudio(client, logger, stats, { account_sid,
text
});
let filePath;
if (['nuance', 'nvidia'].includes(vendor) ||
(
process.env.JAMBONES_TTS_TRIM_SILENCE &&
['microsoft', 'azure'].includes(vendor)
) ||
(
!process.env.JAMBONES_DISABLE_TTS_STREAMING &&
vendor === 'elevenlabs'
)
) {
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
}
else filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.mp3`;
filePath = makeFilePath(vendor, key, salt);
debug(`synth key is ${key}`);
let cached;
if (!disableTtsCache) {
@@ -178,12 +187,14 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'aws':
case 'polly':
vendorLabel = 'aws';
audioBuffer = await synthPolly(logger, {credentials, stats, language, voice, text, engine});
audioBuffer = await synthPolly(createHash, retrieveHash, logger,
{credentials, stats, language, voice, text, engine});
break;
case 'azure':
case 'microsoft':
vendorLabel = 'microsoft';
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId, filePath});
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId,
filePath, renderForCaching, disableTtsStreaming});
break;
case 'nuance':
model = model || 'enhanced';
@@ -201,16 +212,30 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
if (audioBuffer?.filePath) return audioBuffer;
});
break;
case 'playht':
audioBuffer = await synthPlayHT(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
break;
case 'rimelabs':
audioBuffer = await synthRimelabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
break;
case 'verbio':
audioBuffer = await synthVerbio(client, logger, {
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
if (audioBuffer?.filePath) return audioBuffer;
break;
case 'deepgram':
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text,
renderForCaching, disableTtsStreaming});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
@@ -219,6 +244,7 @@ async function synthAudio(client, logger, stats, { account_sid,
default:
assert(`synthAudio: unsupported speech vendor ${vendor}`);
}
if('filePath' in audioBuffer) return audioBuffer;
const diff = process.hrtime(startAt);
const time = diff[0] * 1e3 + diff[1] * 1e-6;
rtt = time.toFixed(0);
@@ -238,16 +264,28 @@ async function synthAudio(client, logger, stats, { account_sid,
});
}
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => {
const synthPolly = async(createHash, retrieveHash, logger,
{credentials, stats, language, voice, engine, text}) => {
try {
const {region, accessKeyId, secretAccessKey} = credentials;
const polly = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
const {region, accessKeyId, secretAccessKey, roleArn} = credentials;
let polly;
if (accessKeyId && secretAccessKey) {
polly = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
} else if (roleArn) {
polly = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
});
} else {
// AWS RoleArn assigned to Instance profile
polly = new PollyClient({region});
}
const opts = {
Engine: engine,
OutputFormat: 'mp3',
@@ -341,7 +379,7 @@ async function _synthOnPremMicrosoft(logger, {
text,
filePath
}) {
const {use_custom_tts, custom_tts_endpoint_url} = credentials;
const {use_custom_tts, custom_tts_endpoint_url, api_key} = credentials;
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
@@ -365,7 +403,8 @@ async function _synthOnPremMicrosoft(logger, {
const post = bent('POST', 'buffer', {
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
'Content-Type': 'application/ssml+xml',
'User-Agent': 'Jambonz'
'User-Agent': 'Jambonz',
...(api_key && {'Ocp-Apim-Subscription-Key': api_key})
});
const mp3 = await post(custom_tts_endpoint_url, content);
return mp3;
@@ -381,10 +420,48 @@ const synthMicrosoft = async(logger, {
language,
voice,
text,
filePath
filePath,
renderForCaching,
disableTtsStreaming
}) => {
try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
// let clean up the text
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${apiKey}`;
params += `,language=${language}`;
params += ',vendor=microsoft';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (region) params += `,region=${region}`;
if (custom_tts_endpoint) params += `,endpointId=${custom_tts_endpoint}`;
if (custom_tts_endpoint_url) params += `,endpoint=${custom_tts_endpoint_url}`;
if (JAMBONES_HTTP_PROXY_IP) params += `,http_proxy_ip=${JAMBONES_HTTP_PROXY_IP}`;
if (JAMBONES_HTTP_PROXY_PORT) params += `,http_proxy_port=${JAMBONES_HTTP_PROXY_PORT}`;
params += '}';
return {
filePath: `say:${params}${content.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
if (use_custom_tts && custom_tts_endpoint_url) {
return await _synthOnPremMicrosoft(logger, {
credentials,
@@ -396,39 +473,23 @@ const synthMicrosoft = async(logger, {
});
}
const trimSilence = filePath.endsWith('.r8');
let content = text;
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
speechConfig.speechSynthesisLanguage = language;
speechConfig.speechSynthesisVoiceName = voice;
if (use_custom_tts && custom_tts_endpoint) {
speechConfig.endpointId = custom_tts_endpoint;
}
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
if (JAMBONES_HTTP_PROXY_IP && JAMBONES_HTTP_PROXY_PORT) {
logger.debug(
`synthMicrosoft: using proxy ${process.env.JAMBONES_HTTP_PROXY_IP}:${process.env.JAMBONES_HTTP_PROXY_PORT}`);
speechConfig.setProxy(process.env.JAMBONES_HTTP_PROXY_IP, process.env.JAMBONES_HTTP_PROXY_PORT);
`synthMicrosoft: using proxy ${JAMBONES_HTTP_PROXY_IP}:${JAMBONES_HTTP_PROXY_PORT}`);
speechConfig.setProxy(JAMBONES_HTTP_PROXY_IP, JAMBONES_HTTP_PROXY_PORT);
}
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
return new Promise((resolve, reject) => {
const speakAsync = content.startsWith('<speak') ?
synthesizer.speakSsmlAsync.bind(synthesizer) :
@@ -611,7 +672,7 @@ const synthElevenlabs = async(logger, {
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=elevenlabs';
@@ -657,10 +718,151 @@ const synthElevenlabs = async(logger, {
}
};
const synthPlayHT = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, user_id, voice_engine, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,user_id=${user_id}`;
params += ',vendor=playht';
params += `,voice=${voice}`;
params += `,voice_engine=${voice_engine}`;
params += ',write_cache_file=1';
if (opts.quality) params += `,quality=${opts.quality}`;
if (opts.speed) params += `,speed=${opts.speed}`;
if (opts.seed) params += `,style=${opts.seed}`;
if (opts.temperature) params += `,temperature=${opts.temperature}`;
if (opts.emotion) params += `,emotion=${opts.emotion}`;
if (opts.voice_guidance) params += `,voice_guidance=${opts.voice_guidance}`;
if (opts.style_guidance) params += `,style_guidance=${opts.style_guidance}`;
if (opts.text_guidance) params += `,text_guidance=${opts.text_guidance}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://api.play.ht', 'POST', 'buffer', {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post('/api/v2/tts/stream', {
text,
voice,
voice_engine,
output_format: 'mp3',
sample_rate: 8000,
...opts
});
return mp3;
} catch (err) {
logger.info({err}, 'synth PlayHT returned error');
stats.increment('tts.count', ['vendor:playht', 'accepted:no']);
throw err;
}
};
const synthRimelabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += ',vendor=rimelabs';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
if (opts.reduceLatency) params += `,reduce_latency=${opts.reduceLatency}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://users.rime.ai', 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Accept': 'audio/mp3',
'Content-Type': 'application/json'
});
const mp3 = await post('/v1/rime-tts', {
speaker: voice,
text,
modelId: model_id,
samplingRate: 8000,
...opts
});
return mp3;
} catch (err) {
logger.info({err}, 'synth rimelabs returned error');
stats.increment('tts.count', ['vendor:rimelabs', 'accepted:no']);
throw err;
}
};
const synthVerbio = async(client, logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
//https://doc.speechcenter.verbio.com/#tag/Text-To-Speech-REST-API
if (text.length > 2000) {
throw new Error('Verbio cannot synthesize for the text length larger than 2000 characters');
}
const token = await getVerbioAccessToken(client, logger, credentials);
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{access_token=${token.access_token}`;
params += ',vendor=verbio';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://us.rest.speechcenter.verbio.com', 'POST', 'buffer', {
'Authorization': `Bearer ${token.access_token}`,
'User-Agent': 'jambonz',
'Content-Type': 'application/json'
});
const r8 = await post('/api/v1/synthesize', {
voice_id: voice,
output_sample_rate: '8k',
output_encoding: 'pcm16',
text
});
return r8;
} catch (err) {
logger.info({err}, 'synth Verbio returned error');
stats.increment('tts.count', ['vendor:verbio', 'accepted:no']);
throw err;
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
@@ -697,8 +899,22 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
}
};
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
const {api_key} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=deepgram';
params += `,voice=${model}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
+48 -3
View File
@@ -3,10 +3,10 @@ const {SynthesizerClient} = require('../stubs/nuance/synthesizer_grpc_pb');
const {RivaSpeechSynthesisClient} = require('../stubs/riva/proto/riva_tts_grpc_pb');
const {Pool} = require('undici');
const pool = new Pool('https://auth.crt.nuance.com');
const HTTP_TIMEOUT = 5000;
const NUANCE_AUTH_ENDPOINT = 'tts.api.nuance.com:443';
const grpc = require('@grpc/grpc-js');
const formurlencoded = require('form-urlencoded');
const { JAMBONES_DISABLE_TTS_STREAMING, JAMBONES_TTS_TRIM_SILENCE, TMP_FOLDER, HTTP_TIMEOUT } = require('./config');
const debug = require('debug')('jambonz:realtimedb-helpers');
/**
@@ -19,7 +19,44 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
function makeSynthKey({account_sid = '', vendor, language, voice, engine = '', text}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
return `tts${account_sid ? (':' + account_sid) : ''}:${hash.digest('hex')}`;
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension(vendor);
const key = `tts${accountKey}:${namespace}:${hexHashKey}`;
return key;
}
function makeFilePath(vendor, key, salt = '') {
const extension = getFileExtension(vendor);
return `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt}`)}.${extension}`;
}
function getFileExtension(vendor) {
const mp3Extension = 'mp3';
const r8Extension = 'r8';
switch (vendor) {
case 'azure':
case 'microsoft':
if (!JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
return r8Extension;
} else {
return mp3Extension;
}
case 'deepgram':
case 'elevenlabs':
case 'rimlabs':
if (!JAMBONES_DISABLE_TTS_STREAMING) {
return r8Extension;
} else {
return mp3Extension;
}
case 'nuance':
case 'nvidia':
return r8Extension;
default:
return mp3Extension;
}
}
const noopLogger = {
@@ -49,6 +86,12 @@ function makeAwsKey(awsAccessKeyId) {
return `aws:${hash.digest('hex')}`;
}
function makeVerbioKey(client_id) {
const hash = crypto.createHash('sha1');
hash.update(client_id);
return `verbio:${hash.digest('hex')}`;
}
function makeNuanceKey(clientId, secret, scope) {
const hash = crypto.createHash('sha1');
hash.update(`${clientId}:${secret}:${scope}`);
@@ -117,11 +160,13 @@ module.exports = {
makeNuanceKey,
makeIbmKey,
makeAwsKey,
makeVerbioKey,
getNuanceAccessToken,
createNuanceClient,
createKryptonClient,
createRivaClient,
makeBasicAuthHeader,
NUANCE_AUTH_ENDPOINT,
noopLogger
noopLogger,
makeFilePath
};
+84 -46
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.44",
"version": "0.1.2",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.0.44",
"version": "0.1.2",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
@@ -19,7 +19,7 @@
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.34.0",
"microsoft-cognitiveservices-speech-sdk": "1.36.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
},
@@ -1124,14 +1124,6 @@
"node": "^12.22.0 || ^14.17.0 || >=16.0.0"
}
},
"node_modules/@fastify/busboy": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/@fastify/busboy/-/busboy-2.1.0.tgz",
"integrity": "sha512-+KpH+QxZU7O4675t3mnkQKcZZg56u+K/Ct2K+N2AZYNVK8kyeo/bI18tI8aPm3tvNNRyTWfj6s5tnGNlcbQRsA==",
"engines": {
"node": ">=14"
}
},
"node_modules/@google-cloud/text-to-speech": {
"version": "5.0.2",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz",
@@ -3125,13 +3117,14 @@
}
},
"node_modules/es5-ext": {
"version": "0.10.62",
"resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.62.tgz",
"integrity": "sha512-BHLqn0klhEpnOKSrzn/Xsz2UIW8j+cGmo9JLzr8BiUapV8hPL9+FliFqjwr9ngW7jWdnxv6eO+/LqyhJVqgrjA==",
"version": "0.10.64",
"resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.64.tgz",
"integrity": "sha512-p2snDhiLaXe6dahss1LddxqEm+SkuDvV8dnIQG0MWjyHpcMNfXKPE+/Cc0y+PhxJX3A4xGNeFCj5oc0BUh6deg==",
"hasInstallScript": true,
"dependencies": {
"es6-iterator": "^2.0.3",
"es6-symbol": "^3.1.3",
"esniff": "^2.0.1",
"next-tick": "^1.1.0"
},
"engines": {
@@ -3278,6 +3271,25 @@
"url": "https://opencollective.com/eslint"
}
},
"node_modules/esniff": {
"version": "2.0.1",
"resolved": "https://registry.npmjs.org/esniff/-/esniff-2.0.1.tgz",
"integrity": "sha512-kTUIGKQ/mDPFoJ0oVfcmyJn4iBDRptjNVIzwIFR7tqWXdVI9xfA2RMwY/gbSpJG3lkdWNEjLap/NqVHZiJsdfg==",
"dependencies": {
"d": "^1.0.1",
"es5-ext": "^0.10.62",
"event-emitter": "^0.3.5",
"type": "^2.7.2"
},
"engines": {
"node": ">=0.10"
}
},
"node_modules/esniff/node_modules/type": {
"version": "2.7.2",
"resolved": "https://registry.npmjs.org/type/-/type-2.7.2.tgz",
"integrity": "sha512-dzlvlNlt6AXU7EBSfpAscydQ7gXB+pPGsPnfJnZpiNJBDj7IaJzQlBZYGdEi4R9HmPdBv2XmWJ6YUtoTa7lmCw=="
},
"node_modules/espree": {
"version": "9.6.1",
"resolved": "https://registry.npmjs.org/espree/-/espree-9.6.1.tgz",
@@ -3350,6 +3362,15 @@
"node": ">=0.10.0"
}
},
"node_modules/event-emitter": {
"version": "0.3.5",
"resolved": "https://registry.npmjs.org/event-emitter/-/event-emitter-0.3.5.tgz",
"integrity": "sha512-D9rRn9y7kLPnJ+hMq7S/nhvoKwwvVJahBi2BPmx3bvbsEdK3W9ii8cBSGjP+72/LnM4n6fo3+dkCX5FeTQruXA==",
"dependencies": {
"d": "1",
"es5-ext": "~0.10.14"
}
},
"node_modules/event-target-shim": {
"version": "5.0.1",
"resolved": "https://registry.npmjs.org/event-target-shim/-/event-target-shim-5.0.1.tgz",
@@ -3551,9 +3572,9 @@
"dev": true
},
"node_modules/follow-redirects": {
"version": "1.15.5",
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.5.tgz",
"integrity": "sha512-vSFWUON1B+yAw1VN4xMfxgn5fTUiaOzAJCKBwIIgT/+7CuGy9+r+5gITvP62j3RmaD5Ph65UaERdOSRGUzZtgw==",
"version": "1.15.6",
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.6.tgz",
"integrity": "sha512-wWN62YITEaOpSK584EZXJafH1AGpO8RVgElfkuXbTOrPX4fIfOyEpW/CsiNd8JdYrAoOvafRTOEnvsO++qCqFA==",
"funding": [
{
"type": "individual",
@@ -5084,9 +5105,9 @@
}
},
"node_modules/microsoft-cognitiveservices-speech-sdk": {
"version": "1.34.0",
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.34.0.tgz",
"integrity": "sha512-WAR0YqouRzVux2kI+f5wTPC6NyJgIVC1g65d79dJ9I32WPJs2kK+eb/BMB6mhSdCjackO5FsrW7JLaQ/vB1heQ==",
"version": "1.36.0",
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.36.0.tgz",
"integrity": "sha512-wPxuEXgjLdqMMIrdBtl8jquGahLV19LQE0ie8MI/PcBcNLG5buVzwS2rQEyHMsRGx+C/4OdBo1ROdNIUzCm4Lg==",
"dependencies": {
"@types/webrtc": "^0.0.37",
"agent-base": "^6.0.1",
@@ -6851,12 +6872,9 @@
}
},
"node_modules/undici": {
"version": "6.4.0",
"resolved": "https://registry.npmjs.org/undici/-/undici-6.4.0.tgz",
"integrity": "sha512-wYaKgftNqf6Je7JQ51YzkEkEevzOgM7at5JytKO7BjaURQpERW8edQSMrr2xb+Yv4U8Yg47J24+lc9+NbeXMFA==",
"dependencies": {
"@fastify/busboy": "^2.0.0"
},
"version": "6.11.1",
"resolved": "https://registry.npmjs.org/undici/-/undici-6.11.1.tgz",
"integrity": "sha512-KyhzaLJnV1qa3BSHdj4AZ2ndqI0QWPxYzaIOio0WzcEJB9gvuysprJSLtpvc2D9mhR9jPDUk7xlJlZbH2KR5iw==",
"engines": {
"node": ">=18.0"
}
@@ -8092,11 +8110,6 @@
"integrity": "sha512-gMsVel9D7f2HLkBma9VbtzZRehRogVRfbr++f06nL2vnCGCNlzOD+/MUov/F4p8myyAHspEhVobgjpX64q5m6A==",
"dev": true
},
"@fastify/busboy": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/@fastify/busboy/-/busboy-2.1.0.tgz",
"integrity": "sha512-+KpH+QxZU7O4675t3mnkQKcZZg56u+K/Ct2K+N2AZYNVK8kyeo/bI18tI8aPm3tvNNRyTWfj6s5tnGNlcbQRsA=="
},
"@google-cloud/text-to-speech": {
"version": "5.0.2",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz",
@@ -9652,12 +9665,13 @@
}
},
"es5-ext": {
"version": "0.10.62",
"resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.62.tgz",
"integrity": "sha512-BHLqn0klhEpnOKSrzn/Xsz2UIW8j+cGmo9JLzr8BiUapV8hPL9+FliFqjwr9ngW7jWdnxv6eO+/LqyhJVqgrjA==",
"version": "0.10.64",
"resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.64.tgz",
"integrity": "sha512-p2snDhiLaXe6dahss1LddxqEm+SkuDvV8dnIQG0MWjyHpcMNfXKPE+/Cc0y+PhxJX3A4xGNeFCj5oc0BUh6deg==",
"requires": {
"es6-iterator": "^2.0.3",
"es6-symbol": "^3.1.3",
"esniff": "^2.0.1",
"next-tick": "^1.1.0"
}
},
@@ -9766,6 +9780,24 @@
"integrity": "sha512-wpc+LXeiyiisxPlEkUzU6svyS1frIO3Mgxj1fdy7Pm8Ygzguax2N3Fa/D/ag1WqbOprdI+uY6wMUl8/a2G+iag==",
"dev": true
},
"esniff": {
"version": "2.0.1",
"resolved": "https://registry.npmjs.org/esniff/-/esniff-2.0.1.tgz",
"integrity": "sha512-kTUIGKQ/mDPFoJ0oVfcmyJn4iBDRptjNVIzwIFR7tqWXdVI9xfA2RMwY/gbSpJG3lkdWNEjLap/NqVHZiJsdfg==",
"requires": {
"d": "^1.0.1",
"es5-ext": "^0.10.62",
"event-emitter": "^0.3.5",
"type": "^2.7.2"
},
"dependencies": {
"type": {
"version": "2.7.2",
"resolved": "https://registry.npmjs.org/type/-/type-2.7.2.tgz",
"integrity": "sha512-dzlvlNlt6AXU7EBSfpAscydQ7gXB+pPGsPnfJnZpiNJBDj7IaJzQlBZYGdEi4R9HmPdBv2XmWJ6YUtoTa7lmCw=="
}
}
},
"espree": {
"version": "9.6.1",
"resolved": "https://registry.npmjs.org/espree/-/espree-9.6.1.tgz",
@@ -9813,6 +9845,15 @@
"integrity": "sha512-kVscqXk4OCp68SZ0dkgEKVi6/8ij300KBWTJq32P/dYeWTSwK41WyTxalN1eRmA5Z9UU/LX9D7FWSmV9SAYx6g==",
"dev": true
},
"event-emitter": {
"version": "0.3.5",
"resolved": "https://registry.npmjs.org/event-emitter/-/event-emitter-0.3.5.tgz",
"integrity": "sha512-D9rRn9y7kLPnJ+hMq7S/nhvoKwwvVJahBi2BPmx3bvbsEdK3W9ii8cBSGjP+72/LnM4n6fo3+dkCX5FeTQruXA==",
"requires": {
"d": "1",
"es5-ext": "~0.10.14"
}
},
"event-target-shim": {
"version": "5.0.1",
"resolved": "https://registry.npmjs.org/event-target-shim/-/event-target-shim-5.0.1.tgz",
@@ -9964,9 +10005,9 @@
"dev": true
},
"follow-redirects": {
"version": "1.15.5",
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.5.tgz",
"integrity": "sha512-vSFWUON1B+yAw1VN4xMfxgn5fTUiaOzAJCKBwIIgT/+7CuGy9+r+5gITvP62j3RmaD5Ph65UaERdOSRGUzZtgw=="
"version": "1.15.6",
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.6.tgz",
"integrity": "sha512-wWN62YITEaOpSK584EZXJafH1AGpO8RVgElfkuXbTOrPX4fIfOyEpW/CsiNd8JdYrAoOvafRTOEnvsO++qCqFA=="
},
"for-each": {
"version": "0.3.3",
@@ -11108,9 +11149,9 @@
}
},
"microsoft-cognitiveservices-speech-sdk": {
"version": "1.34.0",
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.34.0.tgz",
"integrity": "sha512-WAR0YqouRzVux2kI+f5wTPC6NyJgIVC1g65d79dJ9I32WPJs2kK+eb/BMB6mhSdCjackO5FsrW7JLaQ/vB1heQ==",
"version": "1.36.0",
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.36.0.tgz",
"integrity": "sha512-wPxuEXgjLdqMMIrdBtl8jquGahLV19LQE0ie8MI/PcBcNLG5buVzwS2rQEyHMsRGx+C/4OdBo1ROdNIUzCm4Lg==",
"requires": {
"@types/webrtc": "^0.0.37",
"agent-base": "^6.0.1",
@@ -12405,12 +12446,9 @@
}
},
"undici": {
"version": "6.4.0",
"resolved": "https://registry.npmjs.org/undici/-/undici-6.4.0.tgz",
"integrity": "sha512-wYaKgftNqf6Je7JQ51YzkEkEevzOgM7at5JytKO7BjaURQpERW8edQSMrr2xb+Yv4U8Yg47J24+lc9+NbeXMFA==",
"requires": {
"@fastify/busboy": "^2.0.0"
}
"version": "6.11.1",
"resolved": "https://registry.npmjs.org/undici/-/undici-6.11.1.tgz",
"integrity": "sha512-KyhzaLJnV1qa3BSHdj4AZ2ndqI0QWPxYzaIOio0WzcEJB9gvuysprJSLtpvc2D9mhR9jPDUk7xlJlZbH2KR5iw=="
},
"undici-types": {
"version": "5.26.5",
+3 -2
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.44",
"version": "0.1.2",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -12,6 +12,7 @@
"test": "NODE_ENV=test node test/ ",
"coverage": "nyc --reporter html --report-dir ./coverage npm run test",
"jslint": "eslint index.js lib",
"jslint:fix": "eslint --fix '**/*.js'",
"build": "./build_stubs.sh"
},
"repository": {
@@ -34,7 +35,7 @@
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.34.0",
"microsoft-cognitiveservices-speech-sdk": "1.36.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
},
+3 -3
View File
@@ -62,9 +62,9 @@ function deserialize_nuance_tts_v1_UnarySynthesisResponse(buffer_arg) {
//
// The Synthesizer service offers these functionalities:
// - GetVoices: Queries the list of available voices, with filters to reduce the search space.
// - Synthesize: Synthesizes audio from input text and parameters, and returns an audio stream.
// - UnarySynthesize: Synthesizes audio from input text and parameters, and returns a single audio response.
// - GetVoices: Queries the list of available voices, with filters to reduce the search space.
// - Synthesize: Synthesizes audio from input text and parameters, and returns an audio stream.
// - UnarySynthesize: Synthesizes audio from input text and parameters, and returns a single audio response.
var SynthesizerService = exports.SynthesizerService = {
getVoices: {
path: '/nuance.tts.v1.Synthesizer/GetVoices',
+1 -1
View File
@@ -1 +1 @@
// GENERATED CODE -- NO SERVICES IN PROTO
// GENERATED CODE -- NO SERVICES IN PROTO
+5 -5
View File
@@ -57,7 +57,7 @@ function deserialize_nvidia_riva_tts_SynthesizeSpeechResponse(buffer_arg) {
var RivaSpeechSynthesisService = exports.RivaSpeechSynthesisService = {
// Used to request text-to-speech from the service. Submit a request containing the
// desired text and configuration, and receive audio bytes in the requested format.
synthesize: {
synthesize: {
path: '/nvidia.riva.tts.RivaSpeechSynthesis/Synthesize',
requestStream: false,
responseStream: false,
@@ -69,9 +69,9 @@ synthesize: {
responseDeserialize: deserialize_nvidia_riva_tts_SynthesizeSpeechResponse,
},
// Used to request text-to-speech returned via stream as it becomes available.
// Submit a SynthesizeSpeechRequest with desired text and configuration,
// and receive stream of bytes in the requested format.
synthesizeOnline: {
// Submit a SynthesizeSpeechRequest with desired text and configuration,
// and receive stream of bytes in the requested format.
synthesizeOnline: {
path: '/nvidia.riva.tts.RivaSpeechSynthesis/SynthesizeOnline',
requestStream: false,
responseStream: true,
@@ -83,7 +83,7 @@ synthesizeOnline: {
responseDeserialize: deserialize_nvidia_riva_tts_SynthesizeSpeechResponse,
},
// Enables clients to request the configuration of the current Synthesize service, or a specific model within the service.
getRivaSynthesisConfig: {
getRivaSynthesisConfig: {
path: '/nvidia.riva.tts.RivaSpeechSynthesis/GetRivaSynthesisConfig',
requestStream: false,
responseStream: false,
+26
View File
@@ -12,6 +12,32 @@ const stats = {
histogram: () => {}
};
test('Verbio - get Access key and voices', async(t) => {
const fn = require('..');
const {client, getTtsVoices, getVerbioAccessToken} = fn(opts, logger);
if (!process.env.VERBIO_CLIENT_ID || !process.env.VERBIO_CLIENT_SECRET) {
t.pass('skipping Verbio test since no Verbio Keys provided');
t.end();
client.quit();
return;
}
try {
const credentials = {
client_id: process.env.VERBIO_CLIENT_ID,
client_secret: process.env.VERBIO_CLIENT_SECRET
};
let obj = await getVerbioAccessToken(credentials);
t.ok(obj.access_token , 'successfully received access token not from cache');
const voices = await getTtsVoices({vendor: 'verbio', credentials});
t.ok(voices && voices.length != 0, 'successfully received verbio voices');
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('IBM - create access key', async(t) => {
const fn = require('..');
const {client, getIbmAccessToken} = fn(opts, logger);
+156 -2
View File
@@ -162,6 +162,33 @@ test('AWS speech synth tests', async(t) => {
client.quit();
});
test('AWS speech synth tests by RoleArn', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.AWS_ROLE_ARN || !process.env.AWS_REGION) {
t.pass('skipping AWS speech synth tests by RoleArn since AWS_ROLE_ARN or AWS_REGION not provided');
return t.end();
}
try {
let opts = await synthAudio(stats, {
vendor: 'aws',
credentials: {
roleArn: process.env.AWS_ROLE_ARN,
region: process.env.AWS_REGION,
},
language: 'en-US',
voice: 'Joey',
text: 'This is a test. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized aws by roleArn audio to ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Azure speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -188,6 +215,7 @@ test('Azure speech synth tests', async(t) => {
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text: longText,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
@@ -203,6 +231,7 @@ test('Azure speech synth tests', async(t) => {
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text: longText,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
} catch (err) {
@@ -237,6 +266,7 @@ test('Azure SSML tests', async(t) => {
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
@@ -252,6 +282,7 @@ test('Azure SSML tests', async(t) => {
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
} catch (err) {
@@ -283,6 +314,7 @@ test('Azure custom voice speech synth tests', async(t) => {
language: 'en-US',
voice: process.env.MICROSOFT_CUSTOM_VOICE,
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
@@ -297,6 +329,7 @@ test('Azure custom voice speech synth tests', async(t) => {
language: 'en-US',
voice: process.env.MICROSOFT_CUSTOM_VOICE,
text,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`);
} catch (err) {
@@ -543,6 +576,81 @@ test('Elevenlabs speech synth tests', async(t) => {
client.quit();
})
test('PlayHT speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.PLAYHT_API_KEY || !process.env.PLAYHT_USER_ID) {
t.pass('skipping PlayHT speech synth tests since PLAYHT_API_KEY or PLAYHT_USER_ID is/are not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'playht',
credentials: {
api_key: process.env.PLAYHT_API_KEY,
user_id: process.env.PLAYHT_USER_ID,
voice_engine: 'PlayHT2.0-turbo',
options: JSON.stringify({
quality: "medium",
speed: 1,
seed: 1,
temperature: 1,
emotion: "female_happy",
voice_guidance: 3,
style_guidance: 20,
text_guidance: 1,
})
},
language: 'en-US',
voice: 's3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully playht eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('rimelabs speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.RIMELABS_API_KEY) {
t.pass('skipping rimelabs speech synth tests since RIMELABS_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
model_id: 'mist',
options: JSON.stringify({
speedAlpha: 1.0,
reduceLatency: false
})
},
language: 'en-US',
voice: 'amber',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized rimelabs audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('whisper speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -562,6 +670,40 @@ test('whisper speech synth tests', async(t) => {
language: 'en-US',
voice: 'alloy',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('Verbio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.VERBIO_CLIENT_ID || !process.env.VERBIO_CLIENT_SECRET) {
t.pass('skipping Verbio Synthesize test since no Verbio Keys provided');
t.end();
client.quit();
return;
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'verbio',
credentials: {
client_id: process.env.VERBIO_CLIENT_ID,
client_secret: process.env.VERBIO_CLIENT_SECRET
},
language: 'en-US',
voice: 'tommy_en-us',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`);
@@ -589,6 +731,7 @@ test('Deepgram speech synth tests', async(t) => {
},
model: 'aura-asteria-en',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
@@ -646,8 +789,19 @@ test('TTS Cache tests', async(t) => {
t.ok(error, `error returned when specified key was not found`);
// make sure other tts keys are still there
const cached = (await client.keys('tts:*')).length;
t.ok(cached >= 1, `successfully kept all non-specified tts records in cache`);
const cached = await client.keys('tts:*')
t.ok(cached.length >= 1, `successfully kept all non-specified tts records in cache`);
// retrieve keys from cache and check the key contains the file extension
let key = cached[0];
t.ok(key.includes('mp3'), `tts cache extension shoult be part of the key and equal mp3`);
process.env.VG_TRIM_TTS_SILENCE = 'true';
await client.set(makeSynthKey({ vendor: 'azure' }), 'value');
const r8Keys = await client.keys('tts:r8*');
key = r8Keys[0];
t.ok(key.includes('r8'), `tts cache extension shoult be part of the key and equal r8`);
} catch (err) {
console.error(JSON.stringify(err));