Compare commits

...
107 Commits
Author SHA1 Message Date
Dave Horton 5463e9f56e 0.2.19 2025-08-17 09:35:07 -04:00
Dave Horton 2eb36af650 Merge pull request #124 from jambonz/fix/aws_tts
support mod_aws_tts with engine parameter
2025-08-17 09:23:58 -04:00
Quan HL b186cbc4f2 support mod_aws_tts with engine parameter 2025-08-17 06:44:40 +07:00
Dave Horton b1b9c182a9 0.2.18 2025-08-14 08:17:02 -04:00
Dave Horton 15c77626fc Merge pull request #123 from jambonz/feat/mod_aws_tts
support mod_aws_tts
2025-08-14 08:16:32 -04:00
Quan HL 76126ec0b3 wip 2025-08-14 18:49:29 +07:00
Quan HL 55431ee511 wip 2025-08-14 17:19:46 +07:00
Quan HL 1c76f74b4a wip 2025-08-14 17:13:05 +07:00
Quan HL 62ad8abb8e wip 2025-08-14 16:49:47 +07:00
Quan HL 92734aaedb wip 2025-08-14 16:38:36 +07:00
Quan HL fe4ccfe7d7 support mod_aws_tts 2025-08-14 16:34:40 +07:00
Dave Horton ab05976032 0.2.17 2025-08-13 07:41:24 -04:00
Dave Horton babd8abc51 Merge pull request #122 from jambonz/feat/resemble_tts_01
support resemble tts
2025-08-13 07:33:12 -04:00
Quan HL 91f85e16b2 support resemble tts 2025-08-13 14:00:24 +07:00
Dave Horton 8a0c5e3dbd tag 2025-08-10 20:11:52 -04:00
Dave Horton d40e012138 Merge pull request #120 from jambonz/feat/resemble_tts
support resemble ai
2025-08-10 19:12:26 -04:00
Quan HL 49b23f5240 support resemble ai 2025-08-10 15:29:01 +07:00
Dave Horton 043642ea5f 0.2.15 2025-07-13 10:27:02 -04:00
Dave Horton 91ea9820f9 Merge pull request #119 from vdharashive/main
TMP_FOLDER configurable
2025-07-10 07:32:16 -04:00
Vinod Dharashive fad0f57b13 TMP_FOLDER configurable
https://github.com/jambonz/speech-utils/issues/118
2025-07-10 15:01:08 +05:30
Dave Horton 2608e149a8 0.2.14 2025-07-09 13:43:38 -04:00
Dave Horton d21289acd6 sec fix 2025-07-09 13:43:12 -04:00
Dave Horton a7b19e1d8a Merge pull request #117 from sathishkumarpa-Kore/PLAT-41939_2
Fix security vulnerability by upgrading ibm-watson to latest version
2025-07-09 13:42:01 -04:00
sathishkumarpa-Kore 09c7aa9a5f Fix security vulnerability by upgrading ibm-watson 2025-07-03 18:22:34 +05:30
Dave Horton f311c21327 0.2.13 2025-06-26 08:02:24 -04:00
Dave Horton 535d0191da Merge pull request #115 from jambonz/feat/inworld_tts
support inworld tts
2025-06-26 08:01:23 -04:00
Quan HL c00d4f9be4 support inworld tts 2025-06-26 17:34:24 +07:00
Dave Horton db135ee5ad 0.2.12 2025-06-11 11:08:34 +02:00
Dave Horton 5eac4c2ad8 Merge pull request #114 from vasudevanubrolu/feat/893-azure-ssml
Feat/893 azure ssml
2025-06-11 11:05:48 +02:00
vasudevanubrolu e23a1a6d09 feat/893 azure ssml add namespace check 2025-06-04 12:53:37 +05:30
vasudevanubrolu 3a78300a08 feat/893 azure ssml only change free text 2025-06-04 11:18:11 +05:30
vasudevanubrolu 1d7390e7ae feat/893 azure ssml fix only on condition 2025-05-30 14:55:22 +05:30
vasudevanubrolu 3c0940f657 feat/893 azure ssml lang syntax fix 2025-05-30 14:55:22 +05:30
Dave Horton 6fb6195b16 0.2.11 2025-05-27 10:09:01 -04:00
Dave Horton 2ff8587601 Merge pull request #113 from vasudevanubrolu/feat/893-azure-ssml
Feat/893 azure ssml
2025-05-27 10:08:48 -04:00
vasudevanubrolu 925bd26a70 feat/893 add add lang tag for accent to be picked 2025-05-27 13:41:07 +05:30
vasudevanubrolu ddea485f5f feat/893 azure ssml lang for on prem 2025-05-26 15:13:45 +05:30
vasudevanubrolu 49de25feb8 feat/893 ssml config 2025-05-26 12:10:22 +05:30
vasudevanubrolu 08b55b8d79 feat/893 support default azure ssml config 2025-05-26 12:10:22 +05:30
vasudevanubrolu 59bca302b9 feat/893 azure ssml based on env config 2025-05-26 12:10:22 +05:30
Dave Horton 0d98f73c43 0.2.10 2025-05-13 09:57:01 -04:00
Dave Horton 36670e0080 Merge pull request #111 from vasudevanubrolu/feat/864-playht-onprem
feat/864 playht on prem
2025-05-13 09:50:46 -04:00
vasudevanubrolu 61672f9868 feat/864 playht on prem pr changes 2025-05-13 18:47:26 +05:30
vasudevan-Kore ddeca8eb99 feat/864 playht on prem 2025-05-13 18:47:26 +05:30
Dave Horton 6e8271b2f6 0.2.9 2025-05-13 07:47:03 -04:00
Dave Horton 74a4938eb6 Merge pull request #112 from jambonz/feat/whisper_instructions
support openai whisper instructions
2025-05-13 07:46:34 -04:00
Quan HL cb50f603cd wip 2025-05-13 18:02:11 +07:00
Quan HL 9a524f00bc wip 2025-05-13 17:58:57 +07:00
Quan HL 545df0b770 wip 2025-05-13 17:55:41 +07:00
Quan HL 9409405769 support openai whisper instructions 2025-05-13 15:23:59 +07:00
Dave Horton 7dc3bbdb01 0.2.8 2025-05-08 09:31:40 -04:00
Dave Horton f8f9de2645 Merge pull request #110 from jambonz/feat/rimlabs_arcana
support rimelabs arcana
2025-05-06 09:22:35 -04:00
Quan HL 467d7ede26 support rimelabs arcana 2025-05-06 16:16:18 +07:00
Dave Horton fc211ab2e7 0.2.7 2025-04-28 19:31:59 -04:00
Dave Horton 536f8aab31 Merge pull request #109 from jambonz/feat/riva_tts
support riva tts stream
2025-04-28 19:31:30 -04:00
Hoan Luu Huu 69b3fdffbe Merge branch 'main' into feat/riva_tts 2025-04-28 08:47:44 +07:00
Dave Horton 53fe72d89e 0.2.6 2025-04-23 07:12:54 -04:00
Dave Horton e79c15c5da update version 2025-04-23 07:12:25 -04:00
Hoan Luu Huu a4a427e174 Merge branch 'main' into feat/riva_tts 2025-04-23 18:11:19 +07:00
Dave Horton b697bc5268 Merge pull request #108 from jambonz/feat/ell_tts_new_params
elevenlabs tts speed and pronunciation_dictionary_locators
2025-04-23 07:08:49 -04:00
Quan HL d35d7f0aec support riva tts stream 2025-04-23 17:54:21 +07:00
Quan HL ed1c564fa2 wip 2025-04-04 16:15:45 +07:00
Quan HL 7189d471c1 elevenlabs tts speed and pronunciation_dictionary_locators 2025-04-04 15:44:29 +07:00
Dave Horton 04080cc5ec 0.2.4 2025-03-19 21:48:31 -04:00
Dave Horton 3e5ab4af27 Merge pull request #107 from jambonz/update-deps
update undici
2025-03-19 21:48:02 -04:00
Dave Horton f06dddd2f7 update undici 2025-03-19 21:46:21 -04:00
Dave Horton 7b4a71f55d 0.2.3 2025-02-07 07:20:12 -05:00
Dave Horton d552b65618 Merge pull request #106 from jambonz/feat/rimelabs_voices
rimelabs support multiple model and languages
2025-02-07 07:19:47 -05:00
Quan HL f701b50244 rimelabs support multiple model and languages 2025-02-07 14:20:52 +07:00
Dave Horton 199e502fbe 0.2.2 2025-02-03 08:14:15 -05:00
Dave Horton 6769779189 Merge pull request #105 from jambonz/fix/gh_1059
tts key should include model
2025-02-03 08:13:54 -05:00
Quan HL 4b43c3986c wip 2025-02-02 19:03:09 +07:00
Quan HL e1292772e6 tts key should include model 2025-02-02 18:54:16 +07:00
Dave Horton 37fb045431 0.2.1 2024-12-18 22:16:43 -05:00
Dave Horton c328df67a2 major semver bump 2024-12-18 22:15:28 -05:00
Dave Horton 877cba7065 Merge pull request #103 from jambonz/feat/refactor_synth
remove audio extension from audio key
2024-12-18 22:13:52 -05:00
Quan HL 35a178b468 remove audio extension from audio key 2024-12-18 16:58:54 +07:00
Quan HL 7327190471 remove audio extension from audio key 2024-12-18 16:46:24 +07:00
Dave Horton e5b5e3d0c6 0.1.24 2024-12-16 07:27:18 -05:00
Dave Horton 3760be088b update deps 2024-12-16 07:26:49 -05:00
Dave Horton 5cb52eb8ec Merge pull request #101 from jambonz/feat/tts_cartesia
support cartesia tts
2024-12-16 07:25:29 -05:00
Quan HL 8a390a8edf support cartesia tts 2024-12-16 15:58:18 +07:00
Dave Horton d96ab5cdf3 Merge pull request #99 from jambonz/feat/support_aws_instance_profile
support aws instance profile to get key
2024-11-29 21:58:22 -05:00
Quan HL 3bf74671c6 support aws instance profile to get key 2024-11-30 08:40:26 +07:00
Dave Horton a47ef6d7c4 0.1.22 2024-11-04 07:38:23 -05:00
Dave Horton 84089fa528 Merge pull request #98 from jambonz/fix/freshdesk_411
Fix custom tts vendor cached file can not be played
2024-11-04 07:37:49 -05:00
Quan HL 72be44eea2 adding testcase 2024-11-04 15:53:11 +07:00
Quan HL 05d6c4b32d fixed custom vendor cache audio stores file extension 2024-11-04 15:43:44 +07:00
Dave Horton 0c7e15d0a2 0.1.21 2024-10-31 09:48:18 -04:00
Dave Horton 63efecf9d9 Merge pull request #97 from jambonz/feat/google_voice_cloning
support google voice cloning
2024-10-31 09:47:12 -04:00
Quan HL 153ac3f1a4 fix review comment 2024-10-31 20:30:35 +07:00
Quan HL 115faa9f89 support google voice cloning 2024-10-31 20:23:11 +07:00
Dave Horton f183852961 0.1.20 2024-10-18 12:23:26 -04:00
Dave Horton 34c3e01729 Merge pull request #95 from jambonz/fix/rimelabs
fix rimelabs typo issue on getFileExtension function
2024-10-18 12:22:52 -04:00
Quan HL 9b2b16199e fix rimelabs typo issue on getFileExtension function 2024-10-18 22:44:17 +07:00
Dave Horton 9112c5f0ea 0.1.19 2024-10-16 07:23:40 -04:00
Dave Horton 50783dfd0a Merge pull request #94 from jambonz/fix/playht30_lang
add language to playht3.0
2024-10-16 07:23:04 -04:00
Quan HL ca0ef76fe1 add language to playht3.0 2024-10-16 08:14:38 +07:00
Dave Horton 7c91c537e4 0.1.18 2024-10-11 07:33:37 -04:00
Dave Horton 9747526664 Merge pull request #93 from jambonz/fix/playht_3.0
fixed playht3.0 cannot be played if credential is cached
2024-10-11 07:32:57 -04:00
Quan HL 31a0c7b02c fixed playht3.0 cannot be played if credential is cached 2024-10-11 12:05:38 +07:00
Dave Horton c18fbacd1b update playht3 2024-10-09 13:29:01 -04:00
Dave Horton b0fee6bbf1 Merge pull request #92 from jambonz/feat/playht30
support playht3.0
2024-10-09 13:26:53 -04:00
Quan HL f6cead6e92 add top_p and repetition_penalty to playht3.0 2024-10-03 19:24:23 +07:00
Quan HL 05fc96edc0 wip 2024-09-27 18:24:03 +07:00
Quan HL 6794a0b3be support playht3.0 2024-09-27 12:25:47 +07:00
Quan HL 1a04fd736c support playht3.0 2024-09-27 12:08:41 +07:00
9 changed files with 1586 additions and 1414 deletions
+32 -2
View File
@@ -3,8 +3,29 @@ const {noopLogger, makeSynthKey} = require('./utils');
const {JAMBONES_TTS_CACHE_DURATION_MINS} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
function getExtensionAndSampleRate(path) {
const match = path.match(/\.([^.]*)$/);
if (!match) {
//default should be wav file.
return ['wav', 8000];
}
const extension = match[1];
const sampleRateMap = {
r8: 8000,
r16: 16000,
r24: 24000,
r44: 44100,
r48: 48000,
r96: 96000,
};
const sampleRate = sampleRateMap[extension] || 8000;
return [extension, sampleRate];
}
async function addFileToCache(client, logger, path,
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
{account_sid, vendor, language, voice, deploymentId, engine, model, text, instructions}) {
let key;
logger = logger || noopLogger;
@@ -15,10 +36,19 @@ async function addFileToCache(client, logger, path,
language: language || '',
voice: voice || deploymentId,
engine,
model,
text,
instructions
});
const [extension, sampleRate] = getExtensionAndSampleRate(path);
const audioBuffer = await fs.readFile(path);
await client.setex(key, EXPIRES, audioBuffer.toString('base64'));
await client.setex(key, EXPIRES, JSON.stringify(
{
audioContent: audioBuffer.toString('base64'),
extension,
sampleRate
}
));
} catch (err) {
logger.error(err, 'addFileToCache: Error');
return;
+4 -2
View File
@@ -2,13 +2,14 @@ const JAMBONES_TTS_TRIM_SILENCE = process.env.JAMBONES_TTS_TRIM_SILENCE;
const JAMBONES_DISABLE_TTS_STREAMING = process.env.JAMBONES_DISABLE_TTS_STREAMING;
const JAMBONES_DISABLE_AZURE_TTS_STREAMING = process.env.JAMBONES_DISABLE_AZURE_TTS_STREAMING;
const JAMBONES_EAGERLY_PRE_CACHE_AUDIO = process.env.JAMBONES_EAGERLY_PRE_CACHE_AUDIO;
const JAMBONES_AZURE_ENABLE_SSML = process.env.JAMBONES_AZURE_ENABLE_SSML;
const JAMBONES_HTTP_PROXY_IP = process.env.JAMBONES_HTTP_PROXY_IP;
const JAMBONES_HTTP_PROXY_PORT = process.env.JAMBONES_HTTP_PROXY_PORT;
const JAMBONES_TTS_CACHE_DURATION_MINS =
(parseInt(process.env.JAMBONES_TTS_CACHE_DURATION_MINS) || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp';
const TMP_FOLDER = process.env.JAMBONES_TMP_FOLDER || '/tmp';
const HTTP_TIMEOUT = 5000;
@@ -21,5 +22,6 @@ module.exports = {
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
TMP_FOLDER,
HTTP_TIMEOUT
HTTP_TIMEOUT,
JAMBONES_AZURE_ENABLE_SSML
};
+31 -8
View File
@@ -7,22 +7,23 @@ const CACHE_EXPIRY = process.env.AWS_STS_SESSION_RESET_EXPIRY || (EXPIRY - 600);
async function getAwsAuthToken(
logger, createHash, retrieveHash,
{accessKeyId, secretAccessKey, region, roleArn}) {
{speech_credential_sid, accessKeyId, secretAccessKey, region, roleArn}) {
logger = logger || noopLogger;
try {
const key = makeAwsKey(roleArn || accessKeyId);
// if incase instance profile is used, speech_credential_sid will be used as key to lookup cache
const key = makeAwsKey(roleArn || accessKeyId || speech_credential_sid);
const obj = await retrieveHash(key);
if (obj) return {...obj, servedFromCache: true};
/* access token not found in cache, so generate it using STS */
let data;
let expiry = CACHE_EXPIRY;
if (roleArn) {
const stsClient = new STSClient({ region });
const roleToAssume = { RoleArn: roleArn, RoleSessionName: 'Jambonz_Speech', DurationSeconds: EXPIRY};
const command = new AssumeRoleCommand(roleToAssume);
data = await stsClient.send(command);
} else {
/* access token not found in cache, so generate it using STS */
} else if (accessKeyId) {
const stsClient = new STSClient({
region,
credentials: {
@@ -32,6 +33,26 @@ async function getAwsAuthToken(
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
data = await stsClient.send(command);
} else {
// instance profile is used.
const stsClient = new STSClient({ region });
const cred = await stsClient.config.credentials();
// method in the AWS SDK automatically fetches credentials using the default credential
// provider chain. If the credentials come from an instance profile or an environment
// variable, their expiration is controlled by AWS and not explicitly by our code.
if (cred && cred.expiration) {
const currentTime = new Date();
const expiryTime = new Date(cred.expiration);
const remainingTimeInSeconds = Math.round((expiryTime - currentTime) / 1000);
expiry = remainingTimeInSeconds;
}
data = {
Credentials: {
AccessKeyId: cred.accessKeyId,
SecretAccessKey: cred.secretAccessKey,
SessionToken: cred.sessionToken
}
};
}
const credentials = {
@@ -40,9 +61,11 @@ async function getAwsAuthToken(
sessionToken: data.Credentials.SessionToken,
securityToken: data.Credentials.SessionToken
};
createHash(key, credentials, CACHE_EXPIRY)
.catch((err) => logger.error(err, `Error saving hash for key ${key}`));
// Only cache if expiry is good
if (expiry > 0) {
createHash(key, credentials, expiry)
.catch((err) => logger.error(err, `Error saving hash for key ${key}`));
}
return {...credentials, servedFromCache: false};
} catch (err) {
+3 -1
View File
@@ -12,7 +12,7 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
* @returns {object} result - {error, purgedCount}
*/
async function purgeTtsCache(client, logger, {all, account_sid, vendor,
language, voice, deploymentId, engine, text} = {all: true}) {
language, voice, deploymentId, engine, model, text, instructions} = {all: true}) {
logger = logger || noopLogger;
let purgedCount = 0, error;
@@ -33,7 +33,9 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
language: language || '',
voice: voice || deploymentId,
engine,
model,
text,
instructions
});
purgedCount = await client.del(key);
if (purgedCount === 0) error = 'Specified item not found';
+548 -138
View File
File diff suppressed because it is too large Load Diff
+19 -45
View File
@@ -6,7 +6,7 @@ const pool = new Pool('https://auth.crt.nuance.com');
const NUANCE_AUTH_ENDPOINT = 'tts.api.nuance.com:443';
const grpc = require('@grpc/grpc-js');
const formurlencoded = require('form-urlencoded');
const { JAMBONES_DISABLE_TTS_STREAMING, JAMBONES_TTS_TRIM_SILENCE, TMP_FOLDER, HTTP_TIMEOUT } = require('./config');
const { TMP_FOLDER, HTTP_TIMEOUT } = require('./config');
const debug = require('debug')('jambonz:realtimedb-helpers');
/**
@@ -17,59 +17,27 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
//const nuanceClientMap = new Map();
function makeSynthKey({
account_sid = '', vendor, language, voice, engine = '', text,
renderForCaching = false}) {
account_sid = '',
vendor,
language,
voice,
engine = '',
model = '',
text,
instructions = '',
}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
hash.update(`${language}:${vendor}:${voice}:${engine}:${model}:${text}:${instructions}`);
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension({vendor, renderForCaching});
const key = `tts${accountKey}:${namespace}:${hexHashKey}`;
const key = `tts${accountKey}:${hexHashKey}`;
return key;
}
function makeFilePath({vendor, key, salt = '', renderForCaching = false}) {
const extension = getFileExtension({vendor, renderForCaching});
function makeFilePath({key, salt = '', extension}) {
return `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt}`)}.${extension}`;
}
function getFileExtension({vendor, renderForCaching = false}) {
const mp3Extension = 'mp3';
const r8Extension = 'r8';
switch (vendor) {
case 'azure':
case 'microsoft':
if (!renderForCaching && !JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
return r8Extension;
} else {
return mp3Extension;
}
case 'deepgram':
case 'elevenlabs':
case 'rimlabs':
case 'playht':
if (renderForCaching || JAMBONES_DISABLE_TTS_STREAMING) {
return mp3Extension;
} else {
return r8Extension;
}
case 'nuance':
case 'nvidia':
case 'verbio':
return r8Extension;
default:
// If vendor is custom
if (vendor.startsWith('custom')) {
if (renderForCaching || JAMBONES_DISABLE_TTS_STREAMING) {
return mp3Extension;
} else {
return r8Extension;
}
}
return mp3Extension;
}
}
const noopLogger = {
info: () => {},
@@ -98,6 +66,11 @@ function makeAwsKey(awsAccessKeyId) {
return `aws:${hash.digest('hex')}`;
}
function makePlayhtKey(apiKey) {
const hash = crypto.createHash('sha1');
hash.update(apiKey);
return `playht:${hash.digest('hex')}`;
}
function makeVerbioKey(client_id) {
const hash = crypto.createHash('sha1');
hash.update(client_id);
@@ -171,6 +144,7 @@ module.exports = {
makeSynthKey,
makeNuanceKey,
makeIbmKey,
makePlayhtKey,
makeAwsKey,
makeVerbioKey,
getNuanceAccessToken,
+717 -1178
View File
File diff suppressed because it is too large Load Diff
+7 -5
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.1.16",
"version": "0.2.19",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -26,19 +26,21 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"23": "^0.0.0",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^5.0.2",
"@cartesia/cartesia-js": "^2.1.0",
"@google-cloud/text-to-speech": "^5.5.0",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
"debug": "^4.3.4",
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"ibm-watson": "^11.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
"openai": "^4.98.0",
"undici": "^7.5.0"
},
"devDependencies": {
"config": "^3.3.11",
+225 -35
View File
@@ -5,7 +5,7 @@ const fs = require('fs');
const {makeSynthKey} = require('../lib/utils');
const logger = require('pino')();
const bent = require('bent');
const getJSON = bent('json')
const getJSON = bent('json');
process.on('unhandledRejection', (reason, p) => {
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
@@ -91,14 +91,17 @@ test('Google speech Custom voice synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_CUSTOM_VOICE_FILE && !process.env.GCP_CUSTOM_VOICE_JSON_KEY || !process.env.GCP_CUSTOM_VOICE_MODEL) {
t.pass('skipping google speech synth tests since neither GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_MODEL is not provided');
if (!process.env.GCP_CUSTOM_VOICE_FILE &&
!process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
!process.env.GCP_CUSTOM_VOICE_MODEL) {
t.pass(`skipping google speech synth tests since neither
GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_MODEL is not provided`);
return t.end();
}
try {
const str = process.env.GCP_CUSTOM_VOICE_JSON_KEY || fs.readFileSync(process.env.GCP_CUSTOM_VOICE_FILE);
const creds = JSON.parse(str);
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
@@ -109,7 +112,7 @@ test('Google speech Custom voice synth tests', async(t) => {
language: 'en-AU',
text: 'This is a test. This is only a test',
voice: {
reportedUsage:"REALTIME",
reportedUsage: 'REALTIME',
model: process.env.GCP_CUSTOM_VOICE_MODEL
}
});
@@ -121,6 +124,48 @@ test('Google speech Custom voice synth tests', async(t) => {
client.quit();
});
test('Google speech voice cloning synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_CUSTOM_VOICE_FILE &&
!process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
!process.env.GCP_VOICE_CLONING_FILE &&
!process.env.GCP_VOICE_CLONING_JSON_KEY) {
t.pass(`skipping google speech synth tests since neither
GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided,
GCP_VOICE_CLONING_FILE nor GCP_VOICE_CLONING_JSON_KEY is not provided`);
return t.end();
}
try {
const googleKey = process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
fs.readFileSync(process.env.GCP_CUSTOM_VOICE_FILE);
const voice_cloning_key = process.env.GCP_VOICE_CLONING_JSON_KEY ||
fs.readFileSync(process.env.GCP_VOICE_CLONING_FILE).toString();
const creds = JSON.parse(googleKey);
const opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
project_id: creds.project_id
},
},
language: 'en-US',
text: 'This is a test. This is only a test. This is a test. This is only a test. This is a test. This is only a test',
voice: {
voice_cloning_key
}
});
t.ok(!opts.servedFromCache, `successfully synthesized google voice cloning audio to ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('AWS speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -140,6 +185,7 @@ test('AWS speech synth tests', async(t) => {
language: 'en-US',
voice: 'Joey',
text: 'This is a test. This is only a test',
renderForCaching: true,
});
t.ok(!opts.servedFromCache, `successfully synthesized aws audio to ${opts.filePath}`);
@@ -153,6 +199,7 @@ test('AWS speech synth tests', async(t) => {
language: 'en-US',
voice: 'Joey',
text: 'This is a test. This is only a test',
renderForCaching: true,
});
t.ok(opts.servedFromCache, `successfully retrieved aws audio from cache ${opts.filePath}`);
} catch (err) {
@@ -509,6 +556,7 @@ test('Custom Vendor speech synth tests', async(t) => {
text: 'This is a test. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized custom vendor audio to ${opts.filePath}`);
t.ok(opts.filePath.endsWith('wav'), 'audio is cached as wav file');
let obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew`);
t.ok(obj.headers.Authorization == 'Bearer some_jwt_token', 'Custom Vendor Authentication Header is correct');
t.ok(obj.body.language == 'en-US', 'Custom Vendor Language is correct');
@@ -516,6 +564,21 @@ test('Custom Vendor speech synth tests', async(t) => {
t.ok(obj.body.type == 'text', 'Custom Vendor type is correct');
t.ok(obj.body.text == 'This is a test. This is only a test', 'Custom Vendor text is correct');
// Checking if cache is stored with wav format
opts = await synthAudio(stats, {
vendor: 'custom:somethingnew',
credentials: {
use_for_tts: 1,
custom_tts_url: "http://127.0.0.1:3100/somethingnew",
auth_token: 'some_jwt_token'
},
language: 'en-US',
voice: 'English-US.Female-1',
text: 'This is a test. This is only a test',
});
t.ok(opts.servedFromCache, `successfully get custom vendor cached audio to ${opts.filePath}`);
t.ok(opts.filePath.endsWith('wav'), 'audio is cached as wav file');
opts = await synthAudio(stats, {
vendor: 'custom:somethingnew2',
credentials: {
@@ -574,9 +637,9 @@ test('Elevenlabs speech synth tests', async(t) => {
t.end(err);
}
client.quit();
})
});
test('PlayHT speech synth tests', async(t) => {
const testPlayHT = async(t, voice_engine) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -584,32 +647,74 @@ test('PlayHT speech synth tests', async(t) => {
t.pass('skipping PlayHT speech synth tests since PLAYHT_API_KEY or PLAYHT_USER_ID is/are not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
const text = 'Hi there and welcome to jambones! ' + Date.now();
try {
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'playht',
credentials: {
api_key: process.env.PLAYHT_API_KEY,
user_id: process.env.PLAYHT_USER_ID,
voice_engine: 'PlayHT2.0-turbo',
voice_engine,
options: JSON.stringify({
quality: "medium",
quality: 'medium',
speed: 1,
seed: 1,
temperature: 1,
emotion: "female_happy",
emotion: 'female_happy',
voice_guidance: 3,
style_guidance: 20,
text_guidance: 1,
})
},
language: 'en-US',
language: 'english',
voice: 's3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully playht eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
};
test('PlayHT speech synth tests', async(t) => {
await testPlayHT(t, 'PlayHT2.0-turbo');
});
test('PlayHT3.0 speech synth tests', async(t) => {
await testPlayHT(t, 'Play3.0');
});
test('Cartesia speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.CARTESIA_API_KEY) {
t.pass('skipping Cartesia speech synth tests since CARTESIA_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones! ' + Date.now();
try {
const opts = await synthAudio(stats, {
vendor: 'cartesia',
credentials: {
api_key: process.env.CARTESIA_API_KEY,
model_id: 'sonic-english',
options: JSON.stringify({
speed: 1,
emotion: 'female_happy',
})
},
language: 'en',
voice: '694f9389-aac1-45b6-b726-9d9369183238',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully playht eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
@@ -617,7 +722,66 @@ test('PlayHT speech synth tests', async(t) => {
client.quit();
});
test('rimelabs speech synth tests', async(t) => {
test('inworld speech synth', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.INWORLD_API_KEY) {
t.pass('skipping inworld speech synth tests since INWORLD_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
const opts = await synthAudio(stats, {
vendor: 'inworld',
credentials: {
api_key: process.env.INWORLD_API_KEY,
model_id: 'inworld-tts-1'
},
language: 'en',
voice: 'Ashley',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized inworld audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('resemble speech synth', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.RESEMBLE_API_KEY) {
t.pass('skipping resemble speech synth tests since RESEMBLE_API_KEY is not provided');
return t.end();
}
const text = '<speak prompt="Speak in an excited, upbeat tone">Hello from Resemble!</speak>';
try {
const opts = await synthAudio(stats, {
vendor: 'resemble',
credentials: {
api_key: process.env.RESEMBLE_API_KEY,
},
language: 'en',
voice: '3f5fb9f1',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized resemble audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('rimelabs speech synth tests mist', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -627,7 +791,7 @@ test('rimelabs speech synth tests', async(t) => {
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
@@ -637,7 +801,7 @@ test('rimelabs speech synth tests', async(t) => {
reduceLatency: false
})
},
language: 'en-US',
language: 'eng',
voice: 'amber',
text,
renderForCaching: true
@@ -651,6 +815,40 @@ test('rimelabs speech synth tests', async(t) => {
client.quit();
});
test('rimelabs speech synth tests mistv2', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.RIMELABS_API_KEY) {
t.pass('skipping rimelabs speech synth tests since RIMELABS_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
const opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
model_id: 'mistv2',
options: JSON.stringify({
speedAlpha: 1.0,
reduceLatency: false
})
},
language: 'spa',
voice: 'pablo',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized rimelabs mistv2 audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('whisper speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -750,11 +948,12 @@ test('TTS Cache tests', async(t) => {
// save some random tts keys to cache
const minRecords = 8;
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, model: i, text: i,
instructions: i}), i);
}
const count = await getTtsSize();
t.ok(count >= minRecords, 'getTtsSize worked.');
const {purgedCount} = await purgeTtsCache();
t.ok(purgedCount >= minRecords, `successfully purged at least ${minRecords} tts records from cache`);
@@ -769,7 +968,7 @@ test('TTS Cache tests', async(t) => {
try {
// save some random tts keys to cache
for (const i in Array(10).fill(0)) {
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
}
// save a specific key to tts cache
const opts = {vendor: 'aws', language: 'en-US', voice: 'MALE', engine: 'Engine', text: 'Hello World!'};
@@ -785,24 +984,15 @@ test('TTS Cache tests', async(t) => {
language: 'non-existing',
voice: 'non-existing',
});
t.ok(purgedCountWhenErrored === 0, `purged no records when specified key was not found`);
t.ok(error, `error returned when specified key was not found`);
t.ok(purgedCountWhenErrored === 0, 'purged no records when specified key was not found');
t.ok(error, 'error returned when specified key was not found');
// make sure other tts keys are still there
const cached = await client.keys('tts:*')
t.ok(cached.length >= 1, `successfully kept all non-specified tts records in cache`);
const cached = await client.keys('tts:*');
t.ok(cached.length >= 1, 'successfully kept all non-specified tts records in cache');
// retrieve keys from cache and check the key contains the file extension
let key = cached[0];
t.ok(key.includes('mp3'), `tts cache extension shoult be part of the key and equal mp3`);
process.env.VG_TRIM_TTS_SILENCE = 'true';
process.env.VG_TRIM_TTS_SILENCE = 'true';
await client.set(makeSynthKey({ vendor: 'azure' }), 'value');
const r8Keys = await client.keys('tts:r8*');
key = r8Keys[0];
t.ok(key.includes('r8'), `tts cache extension shoult be part of the key and equal r8`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
@@ -816,10 +1006,10 @@ test('TTS Cache tests', async(t) => {
const account_sid = "12412512_cabc_5aff"
const account_sid2 = "22412512_cabc_5aff"
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
}
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
}
const {purgedCount} = await purgeTtsCache({account_sid});
t.equal(purgedCount, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid}`);