Compare commits

..
69 Commits
Author SHA1 Message Date
Dave Horton e099bbb58f 0.1.1 2024-05-28 09:43:24 -04:00
Dave Horton e60a2d2ba3 Merge pull request #72 from jambonz/feat/verbio_speech
add verbio tts/stt
2024-05-28 09:42:28 -04:00
Quan HL 2212be341b add synthesize verbio 2024-05-20 18:11:01 +07:00
Quan HL 5d2d921f31 add verbio tts/stt 2024-05-20 17:24:06 +07:00
Dave Horton acb2d0c7ce Merge pull request #71 from jambonz/feat/azure_private_endpoint
support tts stream private endpoint
2024-05-14 06:56:51 -04:00
Quan HL 90d6048f52 fix private azure link with credential 2024-05-14 14:01:25 +07:00
Quan HL e5985620c0 support tts stream private endpoint 2024-05-05 14:28:11 +07:00
Dave Horton eb57f4d290 bump version 2024-05-02 07:46:47 -04:00
Dave Horton f0a1ab139c Merge pull request #69 from jambonz/feat/aws_polly_rolearn
support AWS Polly RoleArn credential
2024-05-02 07:44:38 -04:00
Quan HL 79289a7249 wip 2024-05-02 15:51:45 +07:00
Quan HL 5998eebdca wip 2024-05-02 15:48:20 +07:00
Quan HL e6a7017b55 wip 2024-05-02 14:20:24 +07:00
Quan HL c8571af129 wip 2024-04-30 15:47:14 +07:00
Quan HL 6144a9c164 wip 2024-04-30 15:45:26 +07:00
Quan HL 68a7f2b0d4 wip 2024-04-30 15:45:10 +07:00
Quan HL 0d1cd37097 wip 2024-04-30 11:20:21 +07:00
Quan HL 3de8e5ff57 accept aws polly without credential 2024-04-22 20:02:59 +07:00
Quan HL b4aad7991b update get aws voices 2024-04-22 16:19:04 +07:00
Quan HL 8a5c5c1966 support mod_google_tts 2024-04-19 16:06:53 +07:00
Quan HL 08a56d1a40 support AWS Polly RoleArn credential 2024-04-19 15:30:22 +07:00
Dave Horton ba61f20334 0.0.51 2024-04-12 07:19:56 -04:00
Dave Horton 10316d786e Merge pull request #67 from jambonz/feat/mod_rimelabs_tts
support mod_rimelabs_tts
2024-04-12 07:10:32 -04:00
Quan HL 38cdc106cc wip 2024-04-12 17:55:55 +07:00
Quan HL 410b99ef24 support mod_rimelabs_tts 2024-04-12 15:57:02 +07:00
Dave Horton 51db63f992 Merge pull request #66 from jambonz/gh-actions
add PlayHT to CI test
2024-04-08 09:49:48 -04:00
Dave Horton c86dceadde add PlayHT to CI test 2024-04-08 09:47:04 -04:00
Dave Horton f56f98f40f 0.0.50 2024-04-08 09:44:17 -04:00
Dave Horton 12a36593aa Merge pull request #65 from jambonz/feat/mod_playht_tts
support mod_playht_tts
2024-04-08 09:43:00 -04:00
Quan HL 8382491477 wip 2024-04-08 20:31:56 +07:00
Quan HL 545d559e27 wip 2024-04-08 19:14:58 +07:00
Quan HL cc8963802f wip 2024-04-08 17:33:21 +07:00
Quan HL 282d87f922 wip 2024-04-08 17:24:15 +07:00
Quan HL 6a288c1db0 support mod_playht_tts 2024-04-08 17:20:32 +07:00
Dave Horton 8650328e64 0.0.49 2024-04-07 12:15:02 -04:00
Dave Horton 85fa6da5b5 update to azure speech sdk 1.36.0 2024-04-07 12:14:55 -04:00
Dave Horton e54e913fdd 0.0.48 2024-04-04 15:52:40 -04:00
Dave Horton b088c0d7d9 Merge pull request #63 from jambonz/feat/mod_deepgram_tts
Feat/mod deepgram tts
2024-04-04 15:52:10 -04:00
Hoan Luu Huu f154a40692 Merge branch 'main' into feat/mod_deepgram_tts 2024-04-04 18:58:38 +07:00
Quan HL 0471b94ebe wip 2024-04-04 10:12:24 +07:00
Dave Horton 8279891dff 0.0.47 2024-04-03 13:33:24 -04:00
Dave Horton f546ca998d cache files for azure tts streaming are r8 2024-04-03 13:32:32 -04:00
Dave Horton bf229d0ab0 0.0.46 2024-04-03 13:22:20 -04:00
Dave Horton f08fedb8ca enable caching from azure tts streaming 2024-04-03 13:17:36 -04:00
Quan HL f3cc38089c mod_deepgra_tts 2024-04-03 20:46:52 +07:00
Dave Horton fbed59e5de 0.0.45 2024-04-02 15:10:54 -04:00
Dave Horton 4f1685a365 Merge pull request #59 from jambonz/feat/azure_tts
support azure streaming
2024-03-30 09:21:07 -04:00
Quan HL 2701af102a wip 2024-03-30 17:49:14 +07:00
Quan HL 7f939b96d2 wip 2024-03-30 17:38:00 +07:00
Quan HL 4d58ca6daf wip 2024-03-30 17:34:49 +07:00
Hoan Luu Huu 16dd7a2805 Merge branch 'main' into feat/azure_tts 2024-03-30 17:04:01 +07:00
Dave Horton 8f3e930004 0.0.44 2024-03-20 19:43:26 -04:00
Dave Horton 3f4c444d82 add azure SSML tests 2024-03-20 19:43:18 -04:00
Hoan Luu Huu fd7d8b8bcd Merge pull request #62 from jambonz/feat/mod_dub
say command for freeswitch module to include vendor and voice
2024-03-20 13:40:51 +07:00
Hoan Luu Huu 46f833c7fa Merge branch 'main' into feat/mod_dub 2024-03-12 18:10:21 +07:00
Dave Horton 4eabfbe4b7 0.0.43 2024-03-11 09:25:14 -04:00
Quan HL f3ab2baa6a wip 2024-03-10 07:34:55 +07:00
Hoan Luu Huu fb412e2ddf Merge branch 'main' into feat/mod_dub 2024-03-10 06:43:09 +07:00
Quan HL f06f96a6f0 wip 2024-03-10 06:41:46 +07:00
Hoan Luu Huu 2988e800b1 Merge branch 'main' into feat/azure_tts 2024-03-10 06:39:00 +07:00
Dave Horton dbfabeaddf Merge pull request #61 from jambonz/fix/duplicate-calls
remove seemingly redundant code, reintroduce param to force bypass of…
2024-03-09 18:25:15 -05:00
Quan HL c3188e40bb support mod_dub 2024-03-09 16:59:00 +07:00
Dave Horton d0dfd07204 remove seemingly redundant code, reintroduce param to force bypass of tts streaming 2024-03-07 13:44:40 -05:00
Dave Horton 04a2466f54 Merge pull request #60 from jambonz/fix/deepgram_tts
update deepgram tts endpoint
2024-03-05 09:12:58 -05:00
Quan HL 0f9a9edc4d update deepgram tts endpoint 2024-03-05 20:55:19 +07:00
Quan HL 31a54f595b wip 2024-02-26 15:42:09 +07:00
Quan HL 3560a6d4d9 wip 2024-02-26 14:01:53 +07:00
Quan HL be8053db4f wip 2024-02-26 13:54:34 +07:00
Quan HL 4ffae38a3f wip 2024-02-26 13:49:37 +07:00
Quan HL 9e74760c39 support azure streaming 2024-02-26 13:33:22 +07:00
14 changed files with 734 additions and 142 deletions
+4 -2
View File
@@ -8,8 +8,8 @@ jobs:
build: build:
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- uses: actions/checkout@v3 - uses: actions/checkout@v4
- uses: actions/setup-node@v3 - uses: actions/setup-node@v4
with: with:
node-version: lts/* node-version: lts/*
- run: npm install - run: npm install
@@ -32,5 +32,7 @@ jobs:
ELEVENLABS_API_KEY: ${{ secrets.ELEVENLABS_API_KEY }} ELEVENLABS_API_KEY: ${{ secrets.ELEVENLABS_API_KEY }}
ELEVENLABS_VOICE_ID: ${{ secrets.ELEVENLABS_VOICE_ID }} ELEVENLABS_VOICE_ID: ${{ secrets.ELEVENLABS_VOICE_ID }}
ELEVENLABS_MODEL_ID: ${{ secrets.ELEVENLABS_MODEL_ID }} ELEVENLABS_MODEL_ID: ${{ secrets.ELEVENLABS_MODEL_ID }}
PLAYHT_USER_ID: ${{ secrets.PLAYHT_USER_ID }}
PLAYHT_API_KEY: ${{ secrets.PLAYHT_API_KEY }}
JAMBONES_HTTP_PROXY_IP: 127.0.0.1 JAMBONES_HTTP_PROXY_IP: 127.0.0.1
JAMBONES_HTTP_PROXY_PORT: 3128 JAMBONES_HTTP_PROXY_PORT: 3128
+3 -2
View File
@@ -13,10 +13,11 @@ module.exports = (opts, logger) => {
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger), getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger), purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger), addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger),
synthAudio: require('./lib/synth-audio').bind(null, client, logger), synthAudio: require('./lib/synth-audio').bind(null, client, createHash, retrieveHash, logger),
getVerbioAccessToken: require('./lib/get-verbio-token').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger), getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger), getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
getAwsAuthToken: require('./lib/get-aws-sts-token').bind(null, logger, createHash, retrieveHash), getAwsAuthToken: require('./lib/get-aws-sts-token').bind(null, logger, createHash, retrieveHash),
getTtsVoices: require('./lib/get-tts-voices').bind(null, client, logger), getTtsVoices: require('./lib/get-tts-voices').bind(null, client, createHash, retrieveHash, logger),
}; };
}; };
+3
View File
@@ -0,0 +1,3 @@
module.exports = {
HTTP_TIMEOUT: 5000
};
+24 -15
View File
@@ -1,32 +1,41 @@
const { STSClient, GetSessionTokenCommand } = require('@aws-sdk/client-sts'); const { STSClient, GetSessionTokenCommand, AssumeRoleCommand } = require('@aws-sdk/client-sts');
const {makeAwsKey, noopLogger} = require('./utils'); const {makeAwsKey, noopLogger} = require('./utils');
const debug = require('debug')('jambonz:speech-utils'); const debug = require('debug')('jambonz:speech-utils');
const EXPIRY = 3600; const EXPIRY = 3600;
async function getAwsAuthToken( async function getAwsAuthToken(
logger, logger, createHash, retrieveHash,
createHash, retrieveHash, awsAccessKeyId, awsSecretAccessKey, awsRegion, roleArn = null) {
awsAccessKeyId, awsSecretAccessKey, awsRegion) {
logger = logger || noopLogger; logger = logger || noopLogger;
try { try {
const key = makeAwsKey(awsAccessKeyId); const key = makeAwsKey(roleArn || awsAccessKeyId);
const obj = await retrieveHash(key); const obj = await retrieveHash(key);
if (obj) return {...obj, servedFromCache: true}; if (obj) return {...obj, servedFromCache: true};
/* access token not found in cache, so generate it using STS */ let data;
const stsClient = new STSClient({ if (roleArn) {
region: awsRegion, const stsClient = new STSClient({ region: awsRegion});
credentials: { const roleToAssume = { RoleArn: roleArn, RoleSessionName: 'Jambonz_Speech', DurationSeconds: EXPIRY};
accessKeyId: awsAccessKeyId, const command = new AssumeRoleCommand(roleToAssume);
secretAccessKey: awsSecretAccessKey,
} data = await stsClient.send(command);
}); } else {
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY}); /* access token not found in cache, so generate it using STS */
const data = await stsClient.send(command); const stsClient = new STSClient({
region: awsRegion,
credentials: {
accessKeyId: awsAccessKeyId,
secretAccessKey: awsSecretAccessKey,
}
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
data = await stsClient.send(command);
}
const credentials = { const credentials = {
accessKeyId: data.Credentials.AccessKeyId, accessKeyId: data.Credentials.AccessKeyId,
secretAccessKey: data.Credentials.SecretAccessKey, secretAccessKey: data.Credentials.SecretAccessKey,
sessionToken: data.Credentials.SessionToken,
securityToken: data.Credentials.SessionToken securityToken: data.Credentials.SessionToken
}; };
+1 -1
View File
@@ -2,8 +2,8 @@ const formurlencoded = require('form-urlencoded');
const {Pool} = require('undici'); const {Pool} = require('undici');
const pool = new Pool('https://iam.cloud.ibm.com'); const pool = new Pool('https://iam.cloud.ibm.com');
const {makeIbmKey, noopLogger} = require('./utils'); const {makeIbmKey, noopLogger} = require('./utils');
const { HTTP_TIMEOUT } = require('./constants');
const debug = require('debug')('jambonz:realtimedb-helpers'); const debug = require('debug')('jambonz:realtimedb-helpers');
const HTTP_TIMEOUT = 5000;
async function getIbmAccessToken(client, logger, apiKey) { async function getIbmAccessToken(client, logger, apiKey) {
logger = logger || noopLogger; logger = logger || noopLogger;
+1 -1
View File
@@ -2,8 +2,8 @@ const formurlencoded = require('form-urlencoded');
const {Pool} = require('undici'); const {Pool} = require('undici');
const pool = new Pool('https://auth.crt.nuance.com'); const pool = new Pool('https://auth.crt.nuance.com');
const {makeNuanceKey, makeBasicAuthHeader, noopLogger} = require('./utils'); const {makeNuanceKey, makeBasicAuthHeader, noopLogger} = require('./utils');
const { HTTP_TIMEOUT } = require('./constants');
const debug = require('debug')('jambonz:realtimedb-helpers'); const debug = require('debug')('jambonz:realtimedb-helpers');
const HTTP_TIMEOUT = 5000;
async function getNuanceAccessToken(client, logger, clientId, secret, scope) { async function getNuanceAccessToken(client, logger, clientId, secret, scope) {
logger = logger || noopLogger; logger = logger || noopLogger;
+49 -12
View File
@@ -1,11 +1,16 @@
const assert = require('assert'); const assert = require('assert');
const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils'); const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token'); const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb'); const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1'); const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth'); const { IamAuthenticator } = require('ibm-watson/auth');
const ttsGoogle = require('@google-cloud/text-to-speech'); const ttsGoogle = require('@google-cloud/text-to-speech');
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly'); const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
const getAwsAuthToken = require('./get-aws-sts-token');
const {Pool} = require('undici');
const { HTTP_TIMEOUT } = require('./constants');
const verbioVoicePool = new Pool('https://us.rest.speechcenter.verbio.com');
const getIbmVoices = async(client, logger, credentials) => { const getIbmVoices = async(client, logger, credentials) => {
const {tts_region, tts_api_key} = credentials; const {tts_region, tts_api_key} = credentials;
@@ -87,16 +92,26 @@ const getGoogleVoices = async(_client, logger, credentials) => {
return await client.listVoices(); return await client.listVoices();
}; };
const getAwsVoices = async(_client, logger, credentials) => { const getAwsVoices = async(_client, createHash, retrieveHash, logger, credentials) => {
try { try {
const {region, accessKeyId, secretAccessKey} = credentials; const {region, accessKeyId, secretAccessKey, roleArn} = credentials;
const client = new PollyClient({ let client = null;
region, if (accessKeyId && secretAccessKey) {
credentials: { client = new PollyClient({
accessKeyId, region,
secretAccessKey credentials: {
} accessKeyId,
}); secretAccessKey
}
});
} else if (roleArn) {
client = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
});
} else {
client = new PollyClient({region});
}
const command = new DescribeVoicesCommand({}); const command = new DescribeVoicesCommand({});
const response = await client.send(command); const response = await client.send(command);
return response; return response;
@@ -106,6 +121,26 @@ const getAwsVoices = async(_client, logger, credentials) => {
} }
}; };
const getVerbioVoices = async(client, logger, credentials) => {
try {
const access_token = await getVerbioAccessToken(client, logger, credentials);
const { body} = await verbioVoicePool.request({
path: '/api/v1/voices',
method: 'GET',
headers: {
'Authorization': `Bearer ${access_token.access_token}`,
'User-Agent': 'jambonz'
},
timeout: HTTP_TIMEOUT,
followRedirects: false
});
return await body.json();
} catch (err) {
logger.info({err}, 'getVerbioVoices - failed to list voices for Verbio');
throw err;
}
};
/** /**
* Synthesize speech to an mp3 file, and also cache the generated speech * Synthesize speech to an mp3 file, and also cache the generated speech
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying * in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
@@ -122,10 +157,10 @@ const getAwsVoices = async(_client, logger, credentials) => {
* @returns object containing filepath to an mp3 file in the /tmp folder containing * @returns object containing filepath to an mp3 file in the /tmp folder containing
* the synthesized audio, and a variable indicating whether it was served from cache * the synthesized audio, and a variable indicating whether it was served from cache
*/ */
async function getTtsVoices(client, logger, {vendor, credentials}) { async function getTtsVoices(client, createHash, retrieveHash, logger, {vendor, credentials}) {
logger = logger || noopLogger; logger = logger || noopLogger;
assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly'].includes(vendor), assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly', 'verbio'].includes(vendor),
`getTtsVoices not supported for vendor ${vendor}`); `getTtsVoices not supported for vendor ${vendor}`);
switch (vendor) { switch (vendor) {
@@ -137,7 +172,9 @@ async function getTtsVoices(client, logger, {vendor, credentials}) {
return getGoogleVoices(client, logger, credentials); return getGoogleVoices(client, logger, credentials);
case 'aws': case 'aws':
case 'polly': case 'polly':
return getAwsVoices(client, logger, credentials); return getAwsVoices(client, createHash, retrieveHash, logger, credentials);
case 'verbio':
return getVerbioVoices(client, logger, credentials);
default: default:
break; break;
} }
+51
View File
@@ -0,0 +1,51 @@
const {Pool} = require('undici');
const { noopLogger, makeVerbioKey } = require('./utils');
const { HTTP_TIMEOUT } = require('./constants');
const pool = new Pool('https://auth.speechcenter.verbio.com:444');
const debug = require('debug')('jambonz:realtimedb-helpers');
async function getVerbioAccessToken(client, logger, credentials) {
logger = logger || noopLogger;
const { client_id, client_secret } = credentials;
try {
const key = makeVerbioKey(client_id);
const access_token = await client.get(key);
if (access_token) {
return {access_token, servedFromCache: true};
}
const payload = {
client_id,
client_secret
};
const {statusCode, headers, body} = await pool.request({
path: '/api/v1/token',
method: 'POST',
headers: {
'Content-Type': 'application/json',
'User-Agent': 'jambonz'
},
body: JSON.stringify(payload),
timeout: HTTP_TIMEOUT,
followRedirects: false
});
if (200 !== statusCode) {
logger.debug({statusCode, headers, body: await body.text()}, 'error fetching access token from Verbio');
const err = new Error();
err.statusCode = statusCode;
throw err;
}
const json = await body.json();
const expiry = Math.floor(json.expiration_time - Date.now() / 1000 - 30);
await client.set(key, json.access_token, 'EX', expiry);
return {...json, servedFromCache: false};
} catch (err) {
debug(err, `getVerbioAccessToken: Error retrieving Verbio access token for client_id ${client_id}`);
logger.error(err, `getVerbioAccessToken: Error retrieving Verbio access token for client_id ${client_id}`);
throw err;
}
}
module.exports = getVerbioAccessToken;
+284 -59
View File
@@ -22,6 +22,7 @@ const {
noopLogger noopLogger
} = require('./utils'); } = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token'); const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
const { const {
SynthesisRequest, SynthesisRequest,
Voice, Voice,
@@ -39,6 +40,7 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp'; const TMP_FOLDER = '/tmp';
const OpenAI = require('openai'); const OpenAI = require('openai');
const getAwsAuthToken = require('./get-aws-sts-token');
const trimTrailingSilence = (buffer) => { const trimTrailingSilence = (buffer) => {
@@ -75,19 +77,19 @@ const trimTrailingSilence = (buffer) => {
* @returns object containing filepath to an mp3 file in the /tmp folder containing * @returns object containing filepath to an mp3 file in the /tmp folder containing
* the synthesized audio, and a variable indicating whether it was served from cache * the synthesized audio, and a variable indicating whether it was served from cache
*/ */
async function synthAudio(client, logger, stats, { account_sid, async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, options disableTtsCache, renderForCaching, disableTtsStreaming, options
}) { }) {
let audioBuffer; let audioBuffer;
let servedFromCache = false; let servedFromCache = false;
let rtt; let rtt;
logger = logger || noopLogger; logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper', 'deepgram'].includes(vendor) || 'whisper', 'deepgram', 'playht', 'rimelabs', 'verbio'].includes(vendor) ||
vendor.startsWith('custom'), vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`); `synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid ..etc, not ${vendor}`);
if ('google' === vendor) { if ('google' === vendor) {
assert.ok(language, 'synthAudio requires language when google is used'); assert.ok(language, 'synthAudio requires language when google is used');
} }
@@ -123,12 +125,25 @@ async function synthAudio(client, logger, stats, { account_sid,
assert.ok(voice, 'synthAudio requires voice when elevenlabs is used'); assert.ok(voice, 'synthAudio requires voice when elevenlabs is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when elevenlabs is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when elevenlabs is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when elevenlabs is used'); assert.ok(credentials.model_id, 'synthAudio requires model_id when elevenlabs is used');
} else if ('playht' === vendor) {
assert.ok(voice, 'synthAudio requires voice when playht is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when playht is used');
assert.ok(credentials.user_id, 'synthAudio requires user_id when playht is used');
assert.ok(credentials.voice_engine, 'synthAudio requires voice_engine when playht is used');
} else if ('rimelabs' === vendor) {
assert.ok(voice, 'synthAudio requires voice when rimelabs is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when rimelabs is used');
assert.ok(credentials.model_id, 'synthAudio requires model_id when rimelabs is used');
} else if ('whisper' === vendor) { } else if ('whisper' === vendor) {
assert.ok(voice, 'synthAudio requires voice when whisper is used'); assert.ok(voice, 'synthAudio requires voice when whisper is used');
assert.ok(credentials.model_id, 'synthAudio requires model when whisper is used'); assert.ok(credentials.model_id, 'synthAudio requires model when whisper is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when whisper is used'); assert.ok(credentials.api_key, 'synthAudio requires api_key when whisper is used');
} else if (vendor.startsWith('custom')) { } else if (vendor.startsWith('custom')) {
assert.ok(credentials.custom_tts_url, `synthAudio requires custom_tts_url in credentials when ${vendor} is used`); assert.ok(credentials.custom_tts_url, `synthAudio requires custom_tts_url in credentials when ${vendor} is used`);
} else if ('verbio' === vendor) {
assert.ok(voice, 'synthAudio requires voice when verbio is used');
assert.ok(credentials.client_id, 'synthAudio requires client_id when verbio is used');
assert.ok(credentials.client_secret, 'synthAudio requires client_secret when verbio is used');
} }
const key = makeSynthKey({ const key = makeSynthKey({
account_sid, account_sid,
@@ -139,14 +154,14 @@ async function synthAudio(client, logger, stats, { account_sid,
text text
}); });
let filePath; let filePath;
if (['nuance', 'nvidia'].includes(vendor) || if (['nuance', 'nvidia', 'verbio'].includes(vendor) ||
( (
process.env.JAMBONES_TTS_TRIM_SILENCE && (process.env.JAMBONES_TTS_TRIM_SILENCE || !process.env.JAMBONES_DISABLE_TTS_STREAMING) &&
['microsoft', 'azure'].includes(vendor) ['microsoft', 'azure'].includes(vendor)
) || ) ||
( (
!process.env.JAMBONES_DISABLE_TTS_STREAMING && !process.env.JAMBONES_DISABLE_TTS_STREAMING &&
vendor === 'elevenlabs' ['elevenlabs', 'deepgram', 'rimelabs'].includes(vendor)
) )
) { ) {
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`; filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
@@ -178,12 +193,15 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'aws': case 'aws':
case 'polly': case 'polly':
vendorLabel = 'aws'; vendorLabel = 'aws';
audioBuffer = await synthPolly(logger, {credentials, stats, language, voice, text, engine}); audioBuffer = await synthPolly(createHash, retrieveHash, logger,
{credentials, stats, language, voice, text, engine});
break; break;
case 'azure': case 'azure':
case 'microsoft': case 'microsoft':
vendorLabel = 'microsoft'; vendorLabel = 'microsoft';
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId, filePath}); audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId,
filePath, renderForCaching, disableTtsStreaming});
if (audioBuffer?.filePath) return audioBuffer;
break; break;
case 'nuance': case 'nuance':
model = model || 'enhanced'; model = model || 'enhanced';
@@ -200,24 +218,36 @@ async function synthAudio(client, logger, stats, { account_sid,
break; break;
case 'elevenlabs': case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, { audioBuffer = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, filePath credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
}); });
if (typeof audioBuffer === 'object' && audioBuffer.filePath) { if (audioBuffer?.filePath) return audioBuffer;
return audioBuffer; break;
} case 'playht':
else { audioBuffer = await synthPlayHT(logger, {
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath}); credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
} });
if (audioBuffer?.filePath) return audioBuffer;
break;
case 'rimelabs':
audioBuffer = await synthRimelabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
if (audioBuffer?.filePath) return audioBuffer;
break; break;
case 'whisper': case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text, renderForCaching}); audioBuffer = await synthWhisper(logger, {
if (typeof audioBuffer === 'object' && audioBuffer.filePath) { credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
return audioBuffer; if (audioBuffer?.filePath) return audioBuffer;
} break;
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text}); case 'verbio':
audioBuffer = await synthVerbio(client, logger, {
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
if (audioBuffer?.filePath) return audioBuffer;
break; break;
case 'deepgram': case 'deepgram':
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text}); audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text,
renderForCaching, disableTtsStreaming});
if (audioBuffer?.filePath) return audioBuffer;
break; break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value': case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger, ({ audioBuffer, filePath } = await synthCustomVendor(logger,
@@ -245,16 +275,28 @@ async function synthAudio(client, logger, stats, { account_sid,
}); });
} }
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => { const synthPolly = async(createHash, retrieveHash, logger,
{credentials, stats, language, voice, engine, text}) => {
try { try {
const {region, accessKeyId, secretAccessKey} = credentials; const {region, accessKeyId, secretAccessKey, roleArn} = credentials;
const polly = new PollyClient({ let polly;
region, if (accessKeyId && secretAccessKey) {
credentials: { polly = new PollyClient({
accessKeyId, region,
secretAccessKey credentials: {
} accessKeyId,
}); secretAccessKey
}
});
} else if (roleArn) {
polly = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
});
} else {
// AWS RoleArn assigned to Instance profile
polly = new PollyClient({region});
}
const opts = { const opts = {
Engine: engine, Engine: engine,
OutputFormat: 'mp3', OutputFormat: 'mp3',
@@ -348,7 +390,7 @@ async function _synthOnPremMicrosoft(logger, {
text, text,
filePath filePath
}) { }) {
const {use_custom_tts, custom_tts_endpoint_url} = credentials; const {use_custom_tts, custom_tts_endpoint_url, api_key} = credentials;
let content = text; let content = text;
if (use_custom_tts && !content.startsWith('<speak')) { if (use_custom_tts && !content.startsWith('<speak')) {
@@ -372,7 +414,8 @@ async function _synthOnPremMicrosoft(logger, {
const post = bent('POST', 'buffer', { const post = bent('POST', 'buffer', {
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3', 'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
'Content-Type': 'application/ssml+xml', 'Content-Type': 'application/ssml+xml',
'User-Agent': 'Jambonz' 'User-Agent': 'Jambonz',
...(api_key && {'Ocp-Apim-Subscription-Key': api_key})
}); });
const mp3 = await post(custom_tts_endpoint_url, content); const mp3 = await post(custom_tts_endpoint_url, content);
return mp3; return mp3;
@@ -388,10 +431,48 @@ const synthMicrosoft = async(logger, {
language, language,
voice, voice,
text, text,
filePath filePath,
renderForCaching,
disableTtsStreaming
}) => { }) => {
try { try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials; const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
// let clean up the text
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${apiKey}`;
params += `,language=${language}`;
params += ',vendor=microsoft';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (region) params += `,region=${region}`;
if (custom_tts_endpoint) params += `,endpointId=${custom_tts_endpoint}`;
if (custom_tts_endpoint_url) params += `,endpoint=${custom_tts_endpoint_url}`;
if (process.env.JAMBONES_HTTP_PROXY_IP) params += `,http_proxy_ip=${process.env.JAMBONES_HTTP_PROXY_IP}`;
if (process.env.JAMBONES_HTTP_PROXY_PORT) params += `,http_proxy_port=${process.env.JAMBONES_HTTP_PROXY_PORT}`;
params += '}';
return {
filePath: `say:${params}${content.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
if (use_custom_tts && custom_tts_endpoint_url) { if (use_custom_tts && custom_tts_endpoint_url) {
return await _synthOnPremMicrosoft(logger, { return await _synthOnPremMicrosoft(logger, {
credentials, credentials,
@@ -403,20 +484,12 @@ const synthMicrosoft = async(logger, {
}); });
} }
const trimSilence = filePath.endsWith('.r8'); const trimSilence = filePath.endsWith('.r8');
let content = text;
const speechConfig = SpeechConfig.fromSubscription(apiKey, region); const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
speechConfig.speechSynthesisLanguage = language; speechConfig.speechSynthesisLanguage = language;
speechConfig.speechSynthesisVoiceName = voice; speechConfig.speechSynthesisVoiceName = voice;
if (use_custom_tts && custom_tts_endpoint) { if (use_custom_tts && custom_tts_endpoint) {
speechConfig.endpointId = custom_tts_endpoint; speechConfig.endpointId = custom_tts_endpoint;
} }
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = trimSilence ? speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm : SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3; SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
@@ -428,14 +501,6 @@ const synthMicrosoft = async(logger, {
} }
const synthesizer = new SpeechSynthesizer(speechConfig); const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
return new Promise((resolve, reject) => { return new Promise((resolve, reject) => {
const speakAsync = content.startsWith('<speak') ? const speakAsync = content.startsWith('<speak') ?
synthesizer.speakSsmlAsync.bind(synthesizer) : synthesizer.speakSsmlAsync.bind(synthesizer) :
@@ -611,14 +676,18 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
} }
}; };
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => { const synthElevenlabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, model_id, options: credOpts} = credentials; const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}'); const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */ /* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) { if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = ''; let params = '';
params += `{api_key=${api_key}`; params += `{api_key=${api_key}`;
params += ',vendor=elevenlabs';
params += `,voice=${voice}`;
params += `,model_id=${model_id}`; params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`; params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
params += ',write_cache_file=1'; params += ',write_cache_file=1';
@@ -660,13 +729,155 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
} }
}; };
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching}) => { const synthPlayHT = async(logger, {
const {api_key, model_id, baseURL, timeout, speed} = credentials; credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */ }) => {
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) { const {api_key, user_id, voice_engine, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,user_id=${user_id}`;
params += ',vendor=playht';
params += `,voice=${voice}`;
params += `,voice_engine=${voice_engine}`;
params += ',write_cache_file=1';
if (opts.quality) params += `,quality=${opts.quality}`;
if (opts.speed) params += `,speed=${opts.speed}`;
if (opts.seed) params += `,style=${opts.seed}`;
if (opts.temperature) params += `,temperature=${opts.temperature}`;
if (opts.emotion) params += `,emotion=${opts.emotion}`;
if (opts.voice_guidance) params += `,voice_guidance=${opts.voice_guidance}`;
if (opts.style_guidance) params += `,style_guidance=${opts.style_guidance}`;
if (opts.text_guidance) params += `,text_guidance=${opts.text_guidance}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://api.play.ht', 'POST', 'buffer', {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post('/api/v2/tts/stream', {
text,
voice,
voice_engine,
output_format: 'mp3',
sample_rate: 8000,
...opts
});
return mp3;
} catch (err) {
logger.info({err}, 'synth PlayHT returned error');
stats.increment('tts.count', ['vendor:playht', 'accepted:no']);
throw err;
}
};
const synthRimelabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = ''; let params = '';
params += `{api_key=${api_key}`; params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`; params += `,model_id=${model_id}`;
params += ',vendor=rimelabs';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
if (opts.reduceLatency) params += `,reduce_latency=${opts.reduceLatency}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://users.rime.ai', 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Accept': 'audio/mp3',
'Content-Type': 'application/json'
});
const mp3 = await post('/v1/rime-tts', {
speaker: voice,
text,
modelId: model_id,
samplingRate: 8000,
...opts
});
return mp3;
} catch (err) {
logger.info({err}, 'synth rimelabs returned error');
stats.increment('tts.count', ['vendor:rimelabs', 'accepted:no']);
throw err;
}
};
const synthVerbio = async(client, logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
//https://doc.speechcenter.verbio.com/#tag/Text-To-Speech-REST-API
if (text.length > 2000) {
throw new Error('Verbio cannot synthesize for the text length larger than 2000 characters');
}
const token = await getVerbioAccessToken(client, logger, credentials);
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{access_token=${token.access_token}`;
params += ',vendor=verbio';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('https://us.rest.speechcenter.verbio.com', 'POST', 'buffer', {
'Authorization': `Bearer ${token.access_token}`,
'User-Agent': 'jambonz',
'Content-Type': 'application/json'
});
const r8 = await post('/api/v1/synthesize', {
voice_id: voice,
output_sample_rate: '8k',
output_encoding: 'pcm16',
text
});
return r8;
} catch (err) {
logger.info({err}, 'synth Verbio returned error');
stats.increment('tts.count', ['vendor:verbio', 'accepted:no']);
throw err;
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += ',vendor=whisper';
params += `,voice=${voice}`; params += `,voice=${voice}`;
params += ',write_cache_file=1'; params += ',write_cache_file=1';
if (speed) params += `,speed=${speed}`; if (speed) params += `,speed=${speed}`;
@@ -699,10 +910,24 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
} }
}; };
const synthDeepgram = async(logger, {credentials, stats, model, text}) => { const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
const {api_key} = credentials; const {api_key} = credentials;
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=deepgram';
params += `,voice=${model}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try { try {
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', { const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`, 'Authorization': `Token ${api_key}`,
'Accept': 'audio/mpeg', 'Accept': 'audio/mpeg',
'Content-Type': 'application/json' 'Content-Type': 'application/json'
+8 -1
View File
@@ -3,10 +3,10 @@ const {SynthesizerClient} = require('../stubs/nuance/synthesizer_grpc_pb');
const {RivaSpeechSynthesisClient} = require('../stubs/riva/proto/riva_tts_grpc_pb'); const {RivaSpeechSynthesisClient} = require('../stubs/riva/proto/riva_tts_grpc_pb');
const {Pool} = require('undici'); const {Pool} = require('undici');
const pool = new Pool('https://auth.crt.nuance.com'); const pool = new Pool('https://auth.crt.nuance.com');
const HTTP_TIMEOUT = 5000;
const NUANCE_AUTH_ENDPOINT = 'tts.api.nuance.com:443'; const NUANCE_AUTH_ENDPOINT = 'tts.api.nuance.com:443';
const grpc = require('@grpc/grpc-js'); const grpc = require('@grpc/grpc-js');
const formurlencoded = require('form-urlencoded'); const formurlencoded = require('form-urlencoded');
const { HTTP_TIMEOUT } = require('./constants');
const debug = require('debug')('jambonz:realtimedb-helpers'); const debug = require('debug')('jambonz:realtimedb-helpers');
/** /**
@@ -49,6 +49,12 @@ function makeAwsKey(awsAccessKeyId) {
return `aws:${hash.digest('hex')}`; return `aws:${hash.digest('hex')}`;
} }
function makeVerbioKey(client_id) {
const hash = crypto.createHash('sha1');
hash.update(client_id);
return `verbio:${hash.digest('hex')}`;
}
function makeNuanceKey(clientId, secret, scope) { function makeNuanceKey(clientId, secret, scope) {
const hash = crypto.createHash('sha1'); const hash = crypto.createHash('sha1');
hash.update(`${clientId}:${secret}:${scope}`); hash.update(`${clientId}:${secret}:${scope}`);
@@ -117,6 +123,7 @@ module.exports = {
makeNuanceKey, makeNuanceKey,
makeIbmKey, makeIbmKey,
makeAwsKey, makeAwsKey,
makeVerbioKey,
getNuanceAccessToken, getNuanceAccessToken,
createNuanceClient, createNuanceClient,
createKryptonClient, createKryptonClient,
+84 -46
View File
@@ -1,12 +1,12 @@
{ {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.42", "version": "0.1.1",
"lockfileVersion": 2, "lockfileVersion": 2,
"requires": true, "requires": true,
"packages": { "packages": {
"": { "": {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.42", "version": "0.1.1",
"license": "MIT", "license": "MIT",
"dependencies": { "dependencies": {
"@aws-sdk/client-polly": "^3.496.0", "@aws-sdk/client-polly": "^3.496.0",
@@ -19,7 +19,7 @@
"form-urlencoded": "^6.1.4", "form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2", "google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0", "ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.34.0", "microsoft-cognitiveservices-speech-sdk": "1.36.0",
"openai": "^4.25.0", "openai": "^4.25.0",
"undici": "^6.4.0" "undici": "^6.4.0"
}, },
@@ -1124,14 +1124,6 @@
"node": "^12.22.0 || ^14.17.0 || >=16.0.0" "node": "^12.22.0 || ^14.17.0 || >=16.0.0"
} }
}, },
"node_modules/@fastify/busboy": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/@fastify/busboy/-/busboy-2.1.0.tgz",
"integrity": "sha512-+KpH+QxZU7O4675t3mnkQKcZZg56u+K/Ct2K+N2AZYNVK8kyeo/bI18tI8aPm3tvNNRyTWfj6s5tnGNlcbQRsA==",
"engines": {
"node": ">=14"
}
},
"node_modules/@google-cloud/text-to-speech": { "node_modules/@google-cloud/text-to-speech": {
"version": "5.0.2", "version": "5.0.2",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz", "resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz",
@@ -3125,13 +3117,14 @@
} }
}, },
"node_modules/es5-ext": { "node_modules/es5-ext": {
"version": "0.10.62", "version": "0.10.64",
"resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.62.tgz", "resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.64.tgz",
"integrity": "sha512-BHLqn0klhEpnOKSrzn/Xsz2UIW8j+cGmo9JLzr8BiUapV8hPL9+FliFqjwr9ngW7jWdnxv6eO+/LqyhJVqgrjA==", "integrity": "sha512-p2snDhiLaXe6dahss1LddxqEm+SkuDvV8dnIQG0MWjyHpcMNfXKPE+/Cc0y+PhxJX3A4xGNeFCj5oc0BUh6deg==",
"hasInstallScript": true, "hasInstallScript": true,
"dependencies": { "dependencies": {
"es6-iterator": "^2.0.3", "es6-iterator": "^2.0.3",
"es6-symbol": "^3.1.3", "es6-symbol": "^3.1.3",
"esniff": "^2.0.1",
"next-tick": "^1.1.0" "next-tick": "^1.1.0"
}, },
"engines": { "engines": {
@@ -3278,6 +3271,25 @@
"url": "https://opencollective.com/eslint" "url": "https://opencollective.com/eslint"
} }
}, },
"node_modules/esniff": {
"version": "2.0.1",
"resolved": "https://registry.npmjs.org/esniff/-/esniff-2.0.1.tgz",
"integrity": "sha512-kTUIGKQ/mDPFoJ0oVfcmyJn4iBDRptjNVIzwIFR7tqWXdVI9xfA2RMwY/gbSpJG3lkdWNEjLap/NqVHZiJsdfg==",
"dependencies": {
"d": "^1.0.1",
"es5-ext": "^0.10.62",
"event-emitter": "^0.3.5",
"type": "^2.7.2"
},
"engines": {
"node": ">=0.10"
}
},
"node_modules/esniff/node_modules/type": {
"version": "2.7.2",
"resolved": "https://registry.npmjs.org/type/-/type-2.7.2.tgz",
"integrity": "sha512-dzlvlNlt6AXU7EBSfpAscydQ7gXB+pPGsPnfJnZpiNJBDj7IaJzQlBZYGdEi4R9HmPdBv2XmWJ6YUtoTa7lmCw=="
},
"node_modules/espree": { "node_modules/espree": {
"version": "9.6.1", "version": "9.6.1",
"resolved": "https://registry.npmjs.org/espree/-/espree-9.6.1.tgz", "resolved": "https://registry.npmjs.org/espree/-/espree-9.6.1.tgz",
@@ -3350,6 +3362,15 @@
"node": ">=0.10.0" "node": ">=0.10.0"
} }
}, },
"node_modules/event-emitter": {
"version": "0.3.5",
"resolved": "https://registry.npmjs.org/event-emitter/-/event-emitter-0.3.5.tgz",
"integrity": "sha512-D9rRn9y7kLPnJ+hMq7S/nhvoKwwvVJahBi2BPmx3bvbsEdK3W9ii8cBSGjP+72/LnM4n6fo3+dkCX5FeTQruXA==",
"dependencies": {
"d": "1",
"es5-ext": "~0.10.14"
}
},
"node_modules/event-target-shim": { "node_modules/event-target-shim": {
"version": "5.0.1", "version": "5.0.1",
"resolved": "https://registry.npmjs.org/event-target-shim/-/event-target-shim-5.0.1.tgz", "resolved": "https://registry.npmjs.org/event-target-shim/-/event-target-shim-5.0.1.tgz",
@@ -3551,9 +3572,9 @@
"dev": true "dev": true
}, },
"node_modules/follow-redirects": { "node_modules/follow-redirects": {
"version": "1.15.5", "version": "1.15.6",
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.5.tgz", "resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.6.tgz",
"integrity": "sha512-vSFWUON1B+yAw1VN4xMfxgn5fTUiaOzAJCKBwIIgT/+7CuGy9+r+5gITvP62j3RmaD5Ph65UaERdOSRGUzZtgw==", "integrity": "sha512-wWN62YITEaOpSK584EZXJafH1AGpO8RVgElfkuXbTOrPX4fIfOyEpW/CsiNd8JdYrAoOvafRTOEnvsO++qCqFA==",
"funding": [ "funding": [
{ {
"type": "individual", "type": "individual",
@@ -5084,9 +5105,9 @@
} }
}, },
"node_modules/microsoft-cognitiveservices-speech-sdk": { "node_modules/microsoft-cognitiveservices-speech-sdk": {
"version": "1.34.0", "version": "1.36.0",
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.34.0.tgz", "resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.36.0.tgz",
"integrity": "sha512-WAR0YqouRzVux2kI+f5wTPC6NyJgIVC1g65d79dJ9I32WPJs2kK+eb/BMB6mhSdCjackO5FsrW7JLaQ/vB1heQ==", "integrity": "sha512-wPxuEXgjLdqMMIrdBtl8jquGahLV19LQE0ie8MI/PcBcNLG5buVzwS2rQEyHMsRGx+C/4OdBo1ROdNIUzCm4Lg==",
"dependencies": { "dependencies": {
"@types/webrtc": "^0.0.37", "@types/webrtc": "^0.0.37",
"agent-base": "^6.0.1", "agent-base": "^6.0.1",
@@ -6851,12 +6872,9 @@
} }
}, },
"node_modules/undici": { "node_modules/undici": {
"version": "6.4.0", "version": "6.11.1",
"resolved": "https://registry.npmjs.org/undici/-/undici-6.4.0.tgz", "resolved": "https://registry.npmjs.org/undici/-/undici-6.11.1.tgz",
"integrity": "sha512-wYaKgftNqf6Je7JQ51YzkEkEevzOgM7at5JytKO7BjaURQpERW8edQSMrr2xb+Yv4U8Yg47J24+lc9+NbeXMFA==", "integrity": "sha512-KyhzaLJnV1qa3BSHdj4AZ2ndqI0QWPxYzaIOio0WzcEJB9gvuysprJSLtpvc2D9mhR9jPDUk7xlJlZbH2KR5iw==",
"dependencies": {
"@fastify/busboy": "^2.0.0"
},
"engines": { "engines": {
"node": ">=18.0" "node": ">=18.0"
} }
@@ -8092,11 +8110,6 @@
"integrity": "sha512-gMsVel9D7f2HLkBma9VbtzZRehRogVRfbr++f06nL2vnCGCNlzOD+/MUov/F4p8myyAHspEhVobgjpX64q5m6A==", "integrity": "sha512-gMsVel9D7f2HLkBma9VbtzZRehRogVRfbr++f06nL2vnCGCNlzOD+/MUov/F4p8myyAHspEhVobgjpX64q5m6A==",
"dev": true "dev": true
}, },
"@fastify/busboy": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/@fastify/busboy/-/busboy-2.1.0.tgz",
"integrity": "sha512-+KpH+QxZU7O4675t3mnkQKcZZg56u+K/Ct2K+N2AZYNVK8kyeo/bI18tI8aPm3tvNNRyTWfj6s5tnGNlcbQRsA=="
},
"@google-cloud/text-to-speech": { "@google-cloud/text-to-speech": {
"version": "5.0.2", "version": "5.0.2",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz", "resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz",
@@ -9652,12 +9665,13 @@
} }
}, },
"es5-ext": { "es5-ext": {
"version": "0.10.62", "version": "0.10.64",
"resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.62.tgz", "resolved": "https://registry.npmjs.org/es5-ext/-/es5-ext-0.10.64.tgz",
"integrity": "sha512-BHLqn0klhEpnOKSrzn/Xsz2UIW8j+cGmo9JLzr8BiUapV8hPL9+FliFqjwr9ngW7jWdnxv6eO+/LqyhJVqgrjA==", "integrity": "sha512-p2snDhiLaXe6dahss1LddxqEm+SkuDvV8dnIQG0MWjyHpcMNfXKPE+/Cc0y+PhxJX3A4xGNeFCj5oc0BUh6deg==",
"requires": { "requires": {
"es6-iterator": "^2.0.3", "es6-iterator": "^2.0.3",
"es6-symbol": "^3.1.3", "es6-symbol": "^3.1.3",
"esniff": "^2.0.1",
"next-tick": "^1.1.0" "next-tick": "^1.1.0"
} }
}, },
@@ -9766,6 +9780,24 @@
"integrity": "sha512-wpc+LXeiyiisxPlEkUzU6svyS1frIO3Mgxj1fdy7Pm8Ygzguax2N3Fa/D/ag1WqbOprdI+uY6wMUl8/a2G+iag==", "integrity": "sha512-wpc+LXeiyiisxPlEkUzU6svyS1frIO3Mgxj1fdy7Pm8Ygzguax2N3Fa/D/ag1WqbOprdI+uY6wMUl8/a2G+iag==",
"dev": true "dev": true
}, },
"esniff": {
"version": "2.0.1",
"resolved": "https://registry.npmjs.org/esniff/-/esniff-2.0.1.tgz",
"integrity": "sha512-kTUIGKQ/mDPFoJ0oVfcmyJn4iBDRptjNVIzwIFR7tqWXdVI9xfA2RMwY/gbSpJG3lkdWNEjLap/NqVHZiJsdfg==",
"requires": {
"d": "^1.0.1",
"es5-ext": "^0.10.62",
"event-emitter": "^0.3.5",
"type": "^2.7.2"
},
"dependencies": {
"type": {
"version": "2.7.2",
"resolved": "https://registry.npmjs.org/type/-/type-2.7.2.tgz",
"integrity": "sha512-dzlvlNlt6AXU7EBSfpAscydQ7gXB+pPGsPnfJnZpiNJBDj7IaJzQlBZYGdEi4R9HmPdBv2XmWJ6YUtoTa7lmCw=="
}
}
},
"espree": { "espree": {
"version": "9.6.1", "version": "9.6.1",
"resolved": "https://registry.npmjs.org/espree/-/espree-9.6.1.tgz", "resolved": "https://registry.npmjs.org/espree/-/espree-9.6.1.tgz",
@@ -9813,6 +9845,15 @@
"integrity": "sha512-kVscqXk4OCp68SZ0dkgEKVi6/8ij300KBWTJq32P/dYeWTSwK41WyTxalN1eRmA5Z9UU/LX9D7FWSmV9SAYx6g==", "integrity": "sha512-kVscqXk4OCp68SZ0dkgEKVi6/8ij300KBWTJq32P/dYeWTSwK41WyTxalN1eRmA5Z9UU/LX9D7FWSmV9SAYx6g==",
"dev": true "dev": true
}, },
"event-emitter": {
"version": "0.3.5",
"resolved": "https://registry.npmjs.org/event-emitter/-/event-emitter-0.3.5.tgz",
"integrity": "sha512-D9rRn9y7kLPnJ+hMq7S/nhvoKwwvVJahBi2BPmx3bvbsEdK3W9ii8cBSGjP+72/LnM4n6fo3+dkCX5FeTQruXA==",
"requires": {
"d": "1",
"es5-ext": "~0.10.14"
}
},
"event-target-shim": { "event-target-shim": {
"version": "5.0.1", "version": "5.0.1",
"resolved": "https://registry.npmjs.org/event-target-shim/-/event-target-shim-5.0.1.tgz", "resolved": "https://registry.npmjs.org/event-target-shim/-/event-target-shim-5.0.1.tgz",
@@ -9964,9 +10005,9 @@
"dev": true "dev": true
}, },
"follow-redirects": { "follow-redirects": {
"version": "1.15.5", "version": "1.15.6",
"resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.5.tgz", "resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.6.tgz",
"integrity": "sha512-vSFWUON1B+yAw1VN4xMfxgn5fTUiaOzAJCKBwIIgT/+7CuGy9+r+5gITvP62j3RmaD5Ph65UaERdOSRGUzZtgw==" "integrity": "sha512-wWN62YITEaOpSK584EZXJafH1AGpO8RVgElfkuXbTOrPX4fIfOyEpW/CsiNd8JdYrAoOvafRTOEnvsO++qCqFA=="
}, },
"for-each": { "for-each": {
"version": "0.3.3", "version": "0.3.3",
@@ -11108,9 +11149,9 @@
} }
}, },
"microsoft-cognitiveservices-speech-sdk": { "microsoft-cognitiveservices-speech-sdk": {
"version": "1.34.0", "version": "1.36.0",
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.34.0.tgz", "resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.36.0.tgz",
"integrity": "sha512-WAR0YqouRzVux2kI+f5wTPC6NyJgIVC1g65d79dJ9I32WPJs2kK+eb/BMB6mhSdCjackO5FsrW7JLaQ/vB1heQ==", "integrity": "sha512-wPxuEXgjLdqMMIrdBtl8jquGahLV19LQE0ie8MI/PcBcNLG5buVzwS2rQEyHMsRGx+C/4OdBo1ROdNIUzCm4Lg==",
"requires": { "requires": {
"@types/webrtc": "^0.0.37", "@types/webrtc": "^0.0.37",
"agent-base": "^6.0.1", "agent-base": "^6.0.1",
@@ -12405,12 +12446,9 @@
} }
}, },
"undici": { "undici": {
"version": "6.4.0", "version": "6.11.1",
"resolved": "https://registry.npmjs.org/undici/-/undici-6.4.0.tgz", "resolved": "https://registry.npmjs.org/undici/-/undici-6.11.1.tgz",
"integrity": "sha512-wYaKgftNqf6Je7JQ51YzkEkEevzOgM7at5JytKO7BjaURQpERW8edQSMrr2xb+Yv4U8Yg47J24+lc9+NbeXMFA==", "integrity": "sha512-KyhzaLJnV1qa3BSHdj4AZ2ndqI0QWPxYzaIOio0WzcEJB9gvuysprJSLtpvc2D9mhR9jPDUk7xlJlZbH2KR5iw=="
"requires": {
"@fastify/busboy": "^2.0.0"
}
}, },
"undici-types": { "undici-types": {
"version": "5.26.5", "version": "5.26.5",
+2 -2
View File
@@ -1,6 +1,6 @@
{ {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.42", "version": "0.1.1",
"description": "TTS-related speech utilities for jambonz", "description": "TTS-related speech utilities for jambonz",
"main": "index.js", "main": "index.js",
"author": "Dave Horton", "author": "Dave Horton",
@@ -34,7 +34,7 @@
"form-urlencoded": "^6.1.4", "form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2", "google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0", "ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.34.0", "microsoft-cognitiveservices-speech-sdk": "1.36.0",
"openai": "^4.25.0", "openai": "^4.25.0",
"undici": "^6.4.0" "undici": "^6.4.0"
}, },
+26
View File
@@ -12,6 +12,32 @@ const stats = {
histogram: () => {} histogram: () => {}
}; };
test('Verbio - get Access key and voices', async(t) => {
const fn = require('..');
const {client, getTtsVoices, getVerbioAccessToken} = fn(opts, logger);
if (!process.env.VERBIO_CLIENT_ID || !process.env.VERBIO_CLIENT_SECRET) {
t.pass('skipping Verbio test since no Verbio Keys provided');
t.end();
client.quit();
return;
}
try {
const credentials = {
client_id: process.env.VERBIO_CLIENT_ID,
client_secret: process.env.VERBIO_CLIENT_SECRET
};
let obj = await getVerbioAccessToken(credentials);
t.ok(obj.access_token , 'successfully received access token not from cache');
const voices = await getTtsVoices({vendor: 'verbio', credentials});
t.ok(voices && voices.length != 0, 'successfully received verbio voices');
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('IBM - create access key', async(t) => { test('IBM - create access key', async(t) => {
const fn = require('..'); const fn = require('..');
const {client, getIbmAccessToken} = fn(opts, logger); const {client, getIbmAccessToken} = fn(opts, logger);
+194 -1
View File
@@ -162,6 +162,33 @@ test('AWS speech synth tests', async(t) => {
client.quit(); client.quit();
}); });
test('AWS speech synth tests by RoleArn', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.AWS_ROLE_ARN || !process.env.AWS_REGION) {
t.pass('skipping AWS speech synth tests by RoleArn since AWS_ROLE_ARN or AWS_REGION not provided');
return t.end();
}
try {
let opts = await synthAudio(stats, {
vendor: 'aws',
credentials: {
roleArn: process.env.AWS_ROLE_ARN,
region: process.env.AWS_REGION,
},
language: 'en-US',
voice: 'Joey',
text: 'This is a test. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized aws by roleArn audio to ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Azure speech synth tests', async(t) => { test('Azure speech synth tests', async(t) => {
const fn = require('..'); const fn = require('..');
const {synthAudio, client} = fn(opts, logger); const {synthAudio, client} = fn(opts, logger);
@@ -188,6 +215,7 @@ test('Azure speech synth tests', async(t) => {
language: 'en-US', language: 'en-US',
voice: 'en-US-ChristopherNeural', voice: 'en-US-ChristopherNeural',
text: longText, text: longText,
renderForCaching: true
}); });
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`); t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) { if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
@@ -203,6 +231,7 @@ test('Azure speech synth tests', async(t) => {
language: 'en-US', language: 'en-US',
voice: 'en-US-ChristopherNeural', voice: 'en-US-ChristopherNeural',
text: longText, text: longText,
renderForCaching: true
}); });
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`); t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
} catch (err) { } catch (err) {
@@ -212,6 +241,58 @@ test('Azure speech synth tests', async(t) => {
client.quit(); client.quit();
}); });
test('Azure SSML tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.MICROSOFT_API_KEY || !process.env.MICROSOFT_REGION) {
t.pass('skipping Microsoft speech synth tests since MICROSOFT_API_KEY or MICROSOFT_REGION not provided');
return t.end();
}
try {
const text = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="en-US">
<voice name="en-US-JennyMultilingualNeural">
<mstts:express-as style="cheerful" styledegree="2">That'd be just amazing!
</mstts:express-as>
</voice>
</speak>`;
let opts = await synthAudio(stats, {
vendor: 'microsoft',
credentials: {
api_key: process.env.MICROSOFT_API_KEY,
region: process.env.MICROSOFT_REGION,
},
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
t.pass('successfully used proxy to reach microsoft tts service');
}
opts = await synthAudio(stats, {
vendor: 'microsoft',
credentials: {
api_key: process.env.MICROSOFT_API_KEY,
region: process.env.MICROSOFT_REGION,
},
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Azure custom voice speech synth tests', async(t) => { test('Azure custom voice speech synth tests', async(t) => {
const fn = require('..'); const fn = require('..');
const {synthAudio, client} = fn(opts, logger); const {synthAudio, client} = fn(opts, logger);
@@ -233,6 +314,7 @@ test('Azure custom voice speech synth tests', async(t) => {
language: 'en-US', language: 'en-US',
voice: process.env.MICROSOFT_CUSTOM_VOICE, voice: process.env.MICROSOFT_CUSTOM_VOICE,
text, text,
renderForCaching: true
}); });
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`); t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
@@ -247,6 +329,7 @@ test('Azure custom voice speech synth tests', async(t) => {
language: 'en-US', language: 'en-US',
voice: process.env.MICROSOFT_CUSTOM_VOICE, voice: process.env.MICROSOFT_CUSTOM_VOICE,
text, text,
renderForCaching: true
}); });
t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`); t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`);
} catch (err) { } catch (err) {
@@ -493,6 +576,81 @@ test('Elevenlabs speech synth tests', async(t) => {
client.quit(); client.quit();
}) })
test('PlayHT speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.PLAYHT_API_KEY || !process.env.PLAYHT_USER_ID) {
t.pass('skipping PlayHT speech synth tests since PLAYHT_API_KEY or PLAYHT_USER_ID is/are not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'playht',
credentials: {
api_key: process.env.PLAYHT_API_KEY,
user_id: process.env.PLAYHT_USER_ID,
voice_engine: 'PlayHT2.0-turbo',
options: JSON.stringify({
quality: "medium",
speed: 1,
seed: 1,
temperature: 1,
emotion: "female_happy",
voice_guidance: 3,
style_guidance: 20,
text_guidance: 1,
})
},
language: 'en-US',
voice: 's3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully playht eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('rimelabs speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.RIMELABS_API_KEY) {
t.pass('skipping rimelabs speech synth tests since RIMELABS_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
model_id: 'mist',
options: JSON.stringify({
speedAlpha: 1.0,
reduceLatency: false
})
},
language: 'en-US',
voice: 'amber',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized rimelabs audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('whisper speech synth tests', async(t) => { test('whisper speech synth tests', async(t) => {
const fn = require('..'); const fn = require('..');
const {synthAudio, client} = fn(opts, logger); const {synthAudio, client} = fn(opts, logger);
@@ -512,6 +670,40 @@ test('whisper speech synth tests', async(t) => {
language: 'en-US', language: 'en-US',
voice: 'alloy', voice: 'alloy',
text, text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('Verbio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.VERBIO_CLIENT_ID || !process.env.VERBIO_CLIENT_SECRET) {
t.pass('skipping Verbio Synthesize test since no Verbio Keys provided');
t.end();
client.quit();
return;
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'verbio',
credentials: {
client_id: process.env.VERBIO_CLIENT_ID,
client_secret: process.env.VERBIO_CLIENT_SECRET
},
language: 'en-US',
voice: 'tommy_en-us',
text,
renderForCaching: true
}); });
t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`); t.ok(!opts.servedFromCache, `successfully synthesized whisper audio to ${opts.filePath}`);
@@ -537,8 +729,9 @@ test('Deepgram speech synth tests', async(t) => {
credentials: { credentials: {
api_key: process.env.DEEPGRAM_API_KEY api_key: process.env.DEEPGRAM_API_KEY
}, },
model: 'alpha-asteria-en-v2', model: 'aura-asteria-en',
text, text,
renderForCaching: true
}); });
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`); t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);