Compare commits

..
59 Commits
Author SHA1 Message Date
Dave Horton a47ef6d7c4 0.1.22 2024-11-04 07:38:23 -05:00
Dave Horton 84089fa528 Merge pull request #98 from jambonz/fix/freshdesk_411
Fix custom tts vendor cached file can not be played
2024-11-04 07:37:49 -05:00
Quan HL 72be44eea2 adding testcase 2024-11-04 15:53:11 +07:00
Quan HL 05d6c4b32d fixed custom vendor cache audio stores file extension 2024-11-04 15:43:44 +07:00
Dave Horton 0c7e15d0a2 0.1.21 2024-10-31 09:48:18 -04:00
Dave Horton 63efecf9d9 Merge pull request #97 from jambonz/feat/google_voice_cloning
support google voice cloning
2024-10-31 09:47:12 -04:00
Quan HL 153ac3f1a4 fix review comment 2024-10-31 20:30:35 +07:00
Quan HL 115faa9f89 support google voice cloning 2024-10-31 20:23:11 +07:00
Dave Horton f183852961 0.1.20 2024-10-18 12:23:26 -04:00
Dave Horton 34c3e01729 Merge pull request #95 from jambonz/fix/rimelabs
fix rimelabs typo issue on getFileExtension function
2024-10-18 12:22:52 -04:00
Quan HL 9b2b16199e fix rimelabs typo issue on getFileExtension function 2024-10-18 22:44:17 +07:00
Dave Horton 9112c5f0ea 0.1.19 2024-10-16 07:23:40 -04:00
Dave Horton 50783dfd0a Merge pull request #94 from jambonz/fix/playht30_lang
add language to playht3.0
2024-10-16 07:23:04 -04:00
Quan HL ca0ef76fe1 add language to playht3.0 2024-10-16 08:14:38 +07:00
Dave Horton 7c91c537e4 0.1.18 2024-10-11 07:33:37 -04:00
Dave Horton 9747526664 Merge pull request #93 from jambonz/fix/playht_3.0
fixed playht3.0 cannot be played if credential is cached
2024-10-11 07:32:57 -04:00
Quan HL 31a0c7b02c fixed playht3.0 cannot be played if credential is cached 2024-10-11 12:05:38 +07:00
Dave Horton c18fbacd1b update playht3 2024-10-09 13:29:01 -04:00
Dave Horton b0fee6bbf1 Merge pull request #92 from jambonz/feat/playht30
support playht3.0
2024-10-09 13:26:53 -04:00
Quan HL f6cead6e92 add top_p and repetition_penalty to playht3.0 2024-10-03 19:24:23 +07:00
Quan HL 05fc96edc0 wip 2024-09-27 18:24:03 +07:00
Quan HL 6794a0b3be support playht3.0 2024-09-27 12:25:47 +07:00
Quan HL 1a04fd736c support playht3.0 2024-09-27 12:08:41 +07:00
Dave Horton 1846203807 0.1.16 2024-09-16 15:53:14 -04:00
Dave Horton 75be8658c1 Merge pull request #91 from jambonz/fix/diff_playht_voice_quality
fix playht has stream and cached audio differrent quality
2024-09-16 15:52:42 -04:00
Quan HL 91a5eebbaf fixed review comment 2024-09-16 18:46:10 +07:00
Quan HL c96f1e86ee fix review comment 2024-09-16 18:15:45 +07:00
Quan HL 8016c0886a fix playht has stream and cached audio differrent quality 2024-09-16 09:01:45 +07:00
Dave Horton 8f216e64d8 0.1.15 2024-08-12 09:30:09 -04:00
Dave Horton e1f4486e01 bump version 2024-08-12 09:27:12 -04:00
Dave Horton b0d6272974 Merge pull request #84 from jambonz/feat/precache_audio_with_tts_stream
support precache audio with tts stream enabled
2024-08-12 09:26:00 -04:00
Quan HL ef23b0807a add comment 2024-08-12 20:16:24 +07:00
Quan HL ab7e25243d improve on check precache 2024-08-12 20:10:43 +07:00
Quan HL bf0ea14423 install docker 2024-08-12 18:40:47 +07:00
Quan HL 305dabd84b wip 2024-08-12 18:35:48 +07:00
Quan HL b6a3fa5081 support precache audio with tts stream enabled 2024-08-12 18:29:01 +07:00
Dave Horton 73feadc4c4 0.1.13 2024-08-06 11:01:09 -04:00
Dave Horton aad0f4d62c Merge pull request #82 from jambonz/feat/deepgram_tts_endpoint
deepgram tts support endpoint for on-premise
2024-08-06 11:00:38 -04:00
Hoan Luu Huu 602b0cc60e Merge branch 'main' into feat/deepgram_tts_endpoint 2024-07-31 14:03:41 +07:00
Dave Horton 8511e762c8 version update 2024-07-30 07:29:25 -04:00
Dave Horton b75d3068c9 Merge pull request #81 from jambonz/feat/gh_fs_832
allow configure STS session expiry
2024-07-30 07:14:06 -04:00
Quan HL 461178e726 wip 2024-07-29 20:56:55 +07:00
Quan HL e0e4d47340 wip 2024-07-29 20:53:55 +07:00
Quan HL a595faa378 deepgram tts support endpoint for on-premise 2024-07-29 19:45:13 +07:00
Quan HL 7fd1e1a3c3 allow configure STS session expiry 2024-07-29 18:18:37 +07:00
Dave Horton 7f6a3d349c 0.1.11 2024-06-14 07:37:19 -04:00
Dave Horton 50429ff535 bump version 2024-06-14 07:36:53 -04:00
Dave Horton 3bf0ef8ea3 Merge pull request #80 from jambonz/fix/aws_arnrole
fix aws arnrole
2024-06-14 07:34:22 -04:00
Quan HL e9a5e83e36 wip 2024-06-14 15:04:19 +07:00
Quan HL 86a64ac091 wip 2024-06-14 15:00:56 +07:00
Quan HL 97e06b3ab3 wip 2024-06-14 10:23:22 +07:00
Quan HL 09e833d910 wip 2024-06-14 10:19:55 +07:00
Quan HL 8c4e12e54f wip 2024-06-14 10:18:41 +07:00
Quan HL 2642bd71a4 wip 2024-06-14 10:16:57 +07:00
Quan HL c4feac916f fix aws arnrole 2024-06-14 10:15:23 +07:00
Dave Horton 2ec56f564e 0.1.9 2024-06-06 12:31:13 -04:00
Dave Horton 9399ddcb58 Merge pull request #78 from jambonz/env/disable-ms-streaming
Env/disable ms streaming
2024-06-06 12:30:48 -04:00
Dave Horton aebf4eda30 lint 2024-06-06 12:25:53 -04:00
Dave Horton 6b0bdfdf2f add env JAMBONES_DISABLE_AZURE_TTS_STREAMING to disable Microsoft TTS streaming 2024-06-06 12:25:08 -04:00
11 changed files with 1881 additions and 1456 deletions
+5
View File
@@ -13,6 +13,11 @@ jobs:
with:
node-version: lts/*
- run: npm install
- name: Install Docker Compose
run: |
sudo curl -L "https://github.com/docker/compose/releases/download/1.29.2/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
sudo chmod +x /usr/local/bin/docker-compose
docker-compose --version
- run: npm run jslint
- run: sudo apt update && sudo apt install -y squid
- run: sudo cp test/squid.conf /etc/squid/squid.conf
+1 -1
View File
@@ -1,3 +1,3 @@
npm audit
#npm audit
npm run jslint:fix || true
npm test
+4
View File
@@ -1,5 +1,7 @@
const JAMBONES_TTS_TRIM_SILENCE = process.env.JAMBONES_TTS_TRIM_SILENCE;
const JAMBONES_DISABLE_TTS_STREAMING = process.env.JAMBONES_DISABLE_TTS_STREAMING;
const JAMBONES_DISABLE_AZURE_TTS_STREAMING = process.env.JAMBONES_DISABLE_AZURE_TTS_STREAMING;
const JAMBONES_EAGERLY_PRE_CACHE_AUDIO = process.env.JAMBONES_EAGERLY_PRE_CACHE_AUDIO;
const JAMBONES_HTTP_PROXY_IP = process.env.JAMBONES_HTTP_PROXY_IP;
const JAMBONES_HTTP_PROXY_PORT = process.env.JAMBONES_HTTP_PROXY_PORT;
@@ -13,9 +15,11 @@ const HTTP_TIMEOUT = 5000;
module.exports = {
JAMBONES_TTS_TRIM_SILENCE,
JAMBONES_DISABLE_TTS_STREAMING,
JAMBONES_DISABLE_AZURE_TTS_STREAMING,
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
TMP_FOLDER,
HTTP_TIMEOUT
};
+10 -9
View File
@@ -1,20 +1,22 @@
const { STSClient, GetSessionTokenCommand, AssumeRoleCommand } = require('@aws-sdk/client-sts');
const {makeAwsKey, noopLogger} = require('./utils');
const debug = require('debug')('jambonz:speech-utils');
const EXPIRY = 3600;
const EXPIRY = process.env.AWS_STS_SESSION_DURATION || 3600;
// by default reset aws session before expiry time 10 mins
const CACHE_EXPIRY = process.env.AWS_STS_SESSION_RESET_EXPIRY || (EXPIRY - 600);
async function getAwsAuthToken(
logger, createHash, retrieveHash,
awsAccessKeyId, awsSecretAccessKey, awsRegion, roleArn = null) {
{accessKeyId, secretAccessKey, region, roleArn}) {
logger = logger || noopLogger;
try {
const key = makeAwsKey(roleArn || awsAccessKeyId);
const key = makeAwsKey(roleArn || accessKeyId);
const obj = await retrieveHash(key);
if (obj) return {...obj, servedFromCache: true};
let data;
if (roleArn) {
const stsClient = new STSClient({ region: awsRegion});
const stsClient = new STSClient({ region });
const roleToAssume = { RoleArn: roleArn, RoleSessionName: 'Jambonz_Speech', DurationSeconds: EXPIRY};
const command = new AssumeRoleCommand(roleToAssume);
@@ -22,10 +24,10 @@ async function getAwsAuthToken(
} else {
/* access token not found in cache, so generate it using STS */
const stsClient = new STSClient({
region: awsRegion,
region,
credentials: {
accessKeyId: awsAccessKeyId,
secretAccessKey: awsSecretAccessKey,
accessKeyId,
secretAccessKey,
}
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
@@ -39,8 +41,7 @@ async function getAwsAuthToken(
securityToken: data.Credentials.SessionToken
};
/* expire 10 minutes before the hour, so we don't lose the use of it during a call */
createHash(key, credentials, EXPIRY - 600)
createHash(key, credentials, CACHE_EXPIRY)
.catch((err) => logger.error(err, `Error saving hash for key ${key}`));
return {...credentials, servedFromCache: false};
+6 -1
View File
@@ -107,7 +107,12 @@ const getAwsVoices = async(_client, createHash, retrieveHash, logger, credential
} else if (roleArn) {
client = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
credentials: await getAwsAuthToken(
logger, createHash, retrieveHash,
{
region,
roleArn
}),
});
} else {
client = new PollyClient({region});
+153 -20
View File
@@ -20,7 +20,8 @@ const {
createKryptonClient,
createRivaClient,
noopLogger,
makeFilePath
makeFilePath,
makePlayhtKey
} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
@@ -40,9 +41,11 @@ const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
const debug = require('debug')('jambonz:realtimedb-helpers');
const {
JAMBONES_DISABLE_TTS_STREAMING,
JAMBONES_DISABLE_AZURE_TTS_STREAMING,
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
const OpenAI = require('openai');
@@ -85,7 +88,7 @@ const trimTrailingSilence = (buffer) => {
*/
async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, disableTtsStreaming, options
disableTtsCache, renderForCaching = false, disableTtsStreaming, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -150,28 +153,67 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
assert.ok(voice, 'synthAudio requires voice when verbio is used');
assert.ok(credentials.client_id, 'synthAudio requires client_id when verbio is used');
assert.ok(credentials.client_secret, 'synthAudio requires client_secret when verbio is used');
} else if ('deepgram' === vendor) {
if (!credentials.deepgram_tts_uri) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
}
}
const key = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text
text,
renderForCaching
});
let filePath;
filePath = makeFilePath(vendor, key, salt);
// used only for custom vendor
let fileExtension;
filePath = makeFilePath({vendor, voice, key, salt, renderForCaching});
debug(`synth key is ${key}`);
let cached;
if (!disableTtsCache) {
cached = await client.get(key);
/**
* If we are using tts streaming and also precaching audio, audio could have been cached by streaming (r8)
* or here in speech-utils due to precaching (mp3), so we need to check for both keys.
*/
if (!cached && JAMBONES_EAGERLY_PRE_CACHE_AUDIO) {
const preCachekey = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text,
renderForCaching: true
});
cached = await client.get(preCachekey);
if (cached) {
// Precache audio is available update filpath with precache file extension.
filePath = makeFilePath({vendor, voice, key, salt, renderForCaching: true});
}
}
}
if (cached) {
// found in cache - extend the expiry and use it
debug('result WAS found in cache');
servedFromCache = true;
stats.increment('tts.cache.requests', ['found:yes']);
audioBuffer = Buffer.from(cached, 'base64');
if (vendor.startsWith('custom')) {
// custom vendors support multiple mime types such as: mp3, wav, r8, r16 ...etc,
// mime type/file extension is available when http response has header Content-type.
// In cache, file extension is store together with audiBuffer in a json.
// Normal cache audio will be base64 string
const payload = JSON.parse(cached);
filePath = filePath.replace(/\.[^\.]*$/g, payload.fileExtension);
audioBuffer = Buffer.from(payload.audioBuffer, 'base64');
} else {
audioBuffer = Buffer.from(cached, 'base64');
}
client.expire(key, EXPIRES).catch((err) => logger.info(err, 'Error setting expires'));
}
if (!cached) {
@@ -215,7 +257,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
});
break;
case 'playht':
audioBuffer = await synthPlayHT(logger, {
audioBuffer = await synthPlayHT(client, logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
break;
@@ -238,7 +280,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
renderForCaching, disableTtsStreaming});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
({ audioBuffer, filePath, fileExtension } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
break;
default:
@@ -252,7 +294,14 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
debug(`tts rtt time for ${text.length} chars on ${vendorLabel}: ${rtt}`);
logger.info(`tts rtt time for ${text.length} chars on ${vendorLabel}: ${rtt}`);
client.setex(key, EXPIRES, audioBuffer.toString('base64'))
const base64Audio = audioBuffer.toString('base64');
const cacheContent = vendor.startsWith('custom') ?
JSON.stringify({
audioBuffer: base64Audio,
fileExtension
}) : base64Audio;
client.setex(key, EXPIRES, cacheContent)
.catch((err) => logger.error(err, `error calling setex on key ${key}`));
}
@@ -280,7 +329,12 @@ const synthPolly = async(createHash, retrieveHash, logger,
} else if (roleArn) {
polly = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
credentials: await getAwsAuthToken(
logger, createHash, retrieveHash,
{
region,
roleArn
}),
});
} else {
// AWS RoleArn assigned to Instance profile
@@ -318,6 +372,44 @@ const synthPolly = async(createHash, retrieveHash, logger,
const synthGoogle = async(logger, {credentials, stats, language, voice, gender, text}) => {
const client = new ttsGoogle.TextToSpeechClient(credentials);
// If google custom voice cloning is used.
// At this time 31 Oct 2024, google node sdk has not support voice cloning yet.
if (typeof voice === 'object' && voice.voice_cloning_key) {
try {
const accessToken = await client.auth.getAccessToken();
const projectId = await client.getProjectId();
const post = bent('https://texttospeech.googleapis.com', 'POST', 'json', {
'Authorization': `Bearer ${accessToken}`,
'x-goog-user-project': projectId,
'Content-Type': 'application/json; charset=utf-8'
});
const payload = {
input: {
text
},
voice: {
language_code: language,
voice_clone: {
voice_cloning_key: voice.voice_cloning_key
}
},
audioConfig: {
// Cloning voice at this time still in v1 beta version, and it support LINEAR16 in Wav format, 24.000Hz
audioEncoding: 'LINEAR16',
sample_rate_hertz: 24000
}
};
const wav = await post('/v1beta1/text:synthesize', payload);
return Buffer.from(wav.audioContent, 'base64');
} catch (err) {
logger.info({err: await err.text()}, 'synthGoogle returned error');
throw err;
}
}
const opts = {
voice: {
...(typeof voice === 'string' && {name: voice}),
@@ -443,7 +535,8 @@ const synthMicrosoft = async(logger, {
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
if (!JAMBONES_DISABLE_TTS_STREAMING && !JAMBONES_DISABLE_AZURE_TTS_STREAMING &&
!renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${apiKey}`;
params += `,language=${language}`;
@@ -655,9 +748,11 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
const regex = /\.[^\.]*$/g;
const mime = response.headers['content-type'];
const buffer = await response.arrayBuffer();
const fileExtension = getFileExtFromMime(mime);
return {
audioBuffer: buffer,
filePath: filePath.replace(regex, getFileExtFromMime(mime))
filePath: filePath.replace(regex, fileExtension),
fileExtension
};
} catch (err) {
logger.info({err}, `Vendor ${vendor} returned error`);
@@ -720,12 +815,40 @@ const synthElevenlabs = async(logger, {
}
};
const synthPlayHT = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
const synthPlayHT = async(client, logger, {
credentials, options, stats, voice, language, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, user_id, voice_engine, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
let synthesizeUrl = 'https://api.play.ht/api/v2/tts/stream';
// If model is play3.0, the synthesizeUrl is got from authentication endpoint
if (voice_engine === 'Play3.0') {
try {
const post = bent('https://api.play.ht', 'POST', 'json', 201, {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
'Accept': 'application/json'
});
const key = makePlayhtKey(api_key);
const url = await client.get(key);
if (!url) {
const {inference_address, expires_at_ms} = await post('/api/v3/auth');
synthesizeUrl = inference_address;
const expiry = Math.floor((expires_at_ms - Date.now()) / 1000 - 30);
await client.set(key, inference_address, 'EX', expiry);
} else {
// Use cached URL
synthesizeUrl = url;
}
} catch (err) {
logger.info({err}, 'synth PlayHT returned error for authentication version 3.0');
stats.increment('tts.count', ['vendor:playht', 'accepted:no']);
throw err;
}
}
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
@@ -734,7 +857,9 @@ const synthPlayHT = async(logger, {
params += ',vendor=playht';
params += `,voice=${voice}`;
params += `,voice_engine=${voice_engine}`;
params += `,synthesize_url=${synthesizeUrl}`;
params += ',write_cache_file=1';
params += `,language=${language}`;
if (opts.quality) params += `,quality=${opts.quality}`;
if (opts.speed) params += `,speed=${opts.speed}`;
if (opts.seed) params += `,style=${opts.seed}`;
@@ -743,6 +868,8 @@ const synthPlayHT = async(logger, {
if (opts.voice_guidance) params += `,voice_guidance=${opts.voice_guidance}`;
if (opts.style_guidance) params += `,style_guidance=${opts.style_guidance}`;
if (opts.text_guidance) params += `,text_guidance=${opts.text_guidance}`;
if (opts.top_p) params += `,top_p=${opts.top_p}`;
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
params += '}';
return {
@@ -753,14 +880,18 @@ const synthPlayHT = async(logger, {
}
try {
const post = bent('https://api.play.ht', 'POST', 'buffer', {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
const post = bent('POST', 'buffer', {
...(voice_engine !== 'Play3.0' && {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
}),
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post('/api/v2/tts/stream', {
const mp3 = await post(synthesizeUrl, {
text,
...(voice_engine === 'Play3.0' && { language }),
voice,
voice_engine,
output_format: 'mp3',
@@ -902,13 +1033,14 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
};
const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
const {api_key} = credentials;
const {api_key, deepgram_tts_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=deepgram';
params += `,voice=${model}`;
params += ',write_cache_file=1';
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
params += '}';
return {
@@ -918,8 +1050,9 @@ const synthDeepgram = async(logger, {credentials, stats, model, text, renderForC
};
}
try {
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
const post = bent(deepgram_tts_uri || 'https://api.deepgram.com', 'POST', 'buffer', {
// on-premise deepgram does not require to have api_key
...(api_key && {'Authorization': `Token ${api_key}`}),
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
+35 -10
View File
@@ -16,46 +16,65 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
*/
//const nuanceClientMap = new Map();
function makeSynthKey({account_sid = '', vendor, language, voice, engine = '', text}) {
function makeSynthKey({
account_sid = '', vendor, language, voice, engine = '', text,
renderForCaching = false}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension(vendor);
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension({vendor, voice, renderForCaching});
const key = `tts${accountKey}:${namespace}:${hexHashKey}`;
return key;
}
function makeFilePath(vendor, key, salt = '') {
const extension = getFileExtension(vendor);
function makeFilePath({vendor, voice, key, salt = '', renderForCaching = false}) {
const extension = getFileExtension({vendor, renderForCaching, voice});
return `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt}`)}.${extension}`;
}
function getFileExtension(vendor) {
function getFileExtension({vendor, voice, renderForCaching = false}) {
const mp3Extension = 'mp3';
const r8Extension = 'r8';
const wavExtension = 'wav';
switch (vendor) {
case 'azure':
case 'microsoft':
if (!JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
if (!renderForCaching && !JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
return r8Extension;
} else {
return mp3Extension;
}
case 'deepgram':
case 'elevenlabs':
case 'rimlabs':
if (!JAMBONES_DISABLE_TTS_STREAMING) {
return r8Extension;
} else {
case 'rimelabs':
case 'playht':
if (renderForCaching || JAMBONES_DISABLE_TTS_STREAMING) {
return mp3Extension;
} else {
return r8Extension;
}
case 'nuance':
case 'nvidia':
case 'verbio':
return r8Extension;
case 'google':
// google voice cloning just support wav.
if (typeof voice === 'object' && voice.voice_cloning_key) {
return wavExtension;
} else {
return mp3Extension;
}
default:
// If vendor is custom
if (vendor.startsWith('custom')) {
if (renderForCaching || JAMBONES_DISABLE_TTS_STREAMING) {
return mp3Extension;
} else {
return r8Extension;
}
}
return mp3Extension;
}
}
@@ -87,6 +106,11 @@ function makeAwsKey(awsAccessKeyId) {
return `aws:${hash.digest('hex')}`;
}
function makePlayhtKey(apiKey) {
const hash = crypto.createHash('sha1');
hash.update(apiKey);
return `playht:${hash.digest('hex')}`;
}
function makeVerbioKey(client_id) {
const hash = crypto.createHash('sha1');
hash.update(client_id);
@@ -160,6 +184,7 @@ module.exports = {
makeSynthKey,
makeNuanceKey,
makeIbmKey,
makePlayhtKey,
makeAwsKey,
makeVerbioKey,
getNuanceAccessToken,
+1573 -1398
View File
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.1.8",
"version": "0.1.22",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -28,7 +28,7 @@
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^5.0.2",
"@google-cloud/text-to-speech": "^5.5.0",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
@@ -36,7 +36,7 @@
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.36.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
},
+10 -2
View File
@@ -19,12 +19,20 @@ test('AWS - create and cache auth token', async(t) => {
return;
}
try {
let obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
let obj = await getAwsAuthToken({
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
region: process.env.AWS_REGION
});
//console.log({obj}, 'received auth token from AWS');
t.ok(obj.securityToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
await sleep(250);
obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
obj = await getAwsAuthToken({
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
region: process.env.AWS_REGION
});
//console.log({obj}, 'received auth token from AWS - second request');
t.ok(obj.securityToken && obj.servedFromCache, 'successfully received access token from cache');
+81 -12
View File
@@ -91,14 +91,17 @@ test('Google speech Custom voice synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_CUSTOM_VOICE_FILE && !process.env.GCP_CUSTOM_VOICE_JSON_KEY || !process.env.GCP_CUSTOM_VOICE_MODEL) {
t.pass('skipping google speech synth tests since neither GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_MODEL is not provided');
if (!process.env.GCP_CUSTOM_VOICE_FILE &&
!process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
!process.env.GCP_CUSTOM_VOICE_MODEL) {
t.pass(`skipping google speech synth tests since neither
GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_MODEL is not provided`);
return t.end();
}
try {
const str = process.env.GCP_CUSTOM_VOICE_JSON_KEY || fs.readFileSync(process.env.GCP_CUSTOM_VOICE_FILE);
const creds = JSON.parse(str);
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
@@ -109,7 +112,7 @@ test('Google speech Custom voice synth tests', async(t) => {
language: 'en-AU',
text: 'This is a test. This is only a test',
voice: {
reportedUsage:"REALTIME",
reportedUsage: 'REALTIME',
model: process.env.GCP_CUSTOM_VOICE_MODEL
}
});
@@ -121,6 +124,48 @@ test('Google speech Custom voice synth tests', async(t) => {
client.quit();
});
test('Google speech voice cloning synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_CUSTOM_VOICE_FILE &&
!process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
!process.env.GCP_VOICE_CLONING_FILE &&
!process.env.GCP_VOICE_CLONING_JSON_KEY) {
t.pass(`skipping google speech synth tests since neither
GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided,
GCP_VOICE_CLONING_FILE nor GCP_VOICE_CLONING_JSON_KEY is not provided`);
return t.end();
}
try {
const googleKey = process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
fs.readFileSync(process.env.GCP_CUSTOM_VOICE_FILE);
const voice_cloning_key = process.env.GCP_VOICE_CLONING_JSON_KEY ||
fs.readFileSync(process.env.GCP_VOICE_CLONING_FILE).toString();
const creds = JSON.parse(googleKey);
const opts = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
project_id: creds.project_id
},
},
language: 'en-US',
text: 'This is a test. This is only a test. This is a test. This is only a test. This is a test. This is only a test',
voice: {
voice_cloning_key
}
});
t.ok(!opts.servedFromCache, `successfully synthesized google voice cloning audio to ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('AWS speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -509,6 +554,7 @@ test('Custom Vendor speech synth tests', async(t) => {
text: 'This is a test. This is only a test',
});
t.ok(!opts.servedFromCache, `successfully synthesized custom vendor audio to ${opts.filePath}`);
t.ok(opts.filePath.endsWith('wav'), 'audio is cached as wav file');
let obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew`);
t.ok(obj.headers.Authorization == 'Bearer some_jwt_token', 'Custom Vendor Authentication Header is correct');
t.ok(obj.body.language == 'en-US', 'Custom Vendor Language is correct');
@@ -516,6 +562,21 @@ test('Custom Vendor speech synth tests', async(t) => {
t.ok(obj.body.type == 'text', 'Custom Vendor type is correct');
t.ok(obj.body.text == 'This is a test. This is only a test', 'Custom Vendor text is correct');
// Checking if cache is stored with wav format
opts = await synthAudio(stats, {
vendor: 'custom:somethingnew',
credentials: {
use_for_tts: 1,
custom_tts_url: "http://127.0.0.1:3100/somethingnew",
auth_token: 'some_jwt_token'
},
language: 'en-US',
voice: 'English-US.Female-1',
text: 'This is a test. This is only a test',
});
t.ok(opts.servedFromCache, `successfully get custom vendor cached audio to ${opts.filePath}`);
t.ok(opts.filePath.endsWith('wav'), 'audio is cached as wav file');
opts = await synthAudio(stats, {
vendor: 'custom:somethingnew2',
credentials: {
@@ -574,9 +635,9 @@ test('Elevenlabs speech synth tests', async(t) => {
t.end(err);
}
client.quit();
})
});
test('PlayHT speech synth tests', async(t) => {
const testPlayHT = async(t, voice_engine) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -584,26 +645,26 @@ test('PlayHT speech synth tests', async(t) => {
t.pass('skipping PlayHT speech synth tests since PLAYHT_API_KEY or PLAYHT_USER_ID is/are not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
const text = 'Hi there and welcome to jambones! ' + Date.now();
try {
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'playht',
credentials: {
api_key: process.env.PLAYHT_API_KEY,
user_id: process.env.PLAYHT_USER_ID,
voice_engine: 'PlayHT2.0-turbo',
voice_engine,
options: JSON.stringify({
quality: "medium",
quality: 'medium',
speed: 1,
seed: 1,
temperature: 1,
emotion: "female_happy",
emotion: 'female_happy',
voice_guidance: 3,
style_guidance: 20,
text_guidance: 1,
})
},
language: 'en-US',
language: 'english',
voice: 's3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json',
text,
renderForCaching: true
@@ -615,6 +676,14 @@ test('PlayHT speech synth tests', async(t) => {
t.end(err);
}
client.quit();
};
test('PlayHT speech synth tests', async(t) => {
await testPlayHT(t, 'PlayHT2.0-turbo');
});
test('PlayHT3.0 speech synth tests', async(t) => {
await testPlayHT(t, 'Play3.0');
});
test('rimelabs speech synth tests', async(t) => {