Compare commits

..
41 Commits
Author SHA1 Message Date
Dave Horton 7c91c537e4 0.1.18 2024-10-11 07:33:37 -04:00
Dave Horton 9747526664 Merge pull request #93 from jambonz/fix/playht_3.0
fixed playht3.0 cannot be played if credential is cached
2024-10-11 07:32:57 -04:00
Quan HL 31a0c7b02c fixed playht3.0 cannot be played if credential is cached 2024-10-11 12:05:38 +07:00
Dave Horton c18fbacd1b update playht3 2024-10-09 13:29:01 -04:00
Dave Horton b0fee6bbf1 Merge pull request #92 from jambonz/feat/playht30
support playht3.0
2024-10-09 13:26:53 -04:00
Quan HL f6cead6e92 add top_p and repetition_penalty to playht3.0 2024-10-03 19:24:23 +07:00
Quan HL 05fc96edc0 wip 2024-09-27 18:24:03 +07:00
Quan HL 6794a0b3be support playht3.0 2024-09-27 12:25:47 +07:00
Quan HL 1a04fd736c support playht3.0 2024-09-27 12:08:41 +07:00
Dave Horton 1846203807 0.1.16 2024-09-16 15:53:14 -04:00
Dave Horton 75be8658c1 Merge pull request #91 from jambonz/fix/diff_playht_voice_quality
fix playht has stream and cached audio differrent quality
2024-09-16 15:52:42 -04:00
Quan HL 91a5eebbaf fixed review comment 2024-09-16 18:46:10 +07:00
Quan HL c96f1e86ee fix review comment 2024-09-16 18:15:45 +07:00
Quan HL 8016c0886a fix playht has stream and cached audio differrent quality 2024-09-16 09:01:45 +07:00
Dave Horton 8f216e64d8 0.1.15 2024-08-12 09:30:09 -04:00
Dave Horton e1f4486e01 bump version 2024-08-12 09:27:12 -04:00
Dave Horton b0d6272974 Merge pull request #84 from jambonz/feat/precache_audio_with_tts_stream
support precache audio with tts stream enabled
2024-08-12 09:26:00 -04:00
Quan HL ef23b0807a add comment 2024-08-12 20:16:24 +07:00
Quan HL ab7e25243d improve on check precache 2024-08-12 20:10:43 +07:00
Quan HL bf0ea14423 install docker 2024-08-12 18:40:47 +07:00
Quan HL 305dabd84b wip 2024-08-12 18:35:48 +07:00
Quan HL b6a3fa5081 support precache audio with tts stream enabled 2024-08-12 18:29:01 +07:00
Dave Horton 73feadc4c4 0.1.13 2024-08-06 11:01:09 -04:00
Dave Horton aad0f4d62c Merge pull request #82 from jambonz/feat/deepgram_tts_endpoint
deepgram tts support endpoint for on-premise
2024-08-06 11:00:38 -04:00
Hoan Luu Huu 602b0cc60e Merge branch 'main' into feat/deepgram_tts_endpoint 2024-07-31 14:03:41 +07:00
Dave Horton 8511e762c8 version update 2024-07-30 07:29:25 -04:00
Dave Horton b75d3068c9 Merge pull request #81 from jambonz/feat/gh_fs_832
allow configure STS session expiry
2024-07-30 07:14:06 -04:00
Quan HL 461178e726 wip 2024-07-29 20:56:55 +07:00
Quan HL e0e4d47340 wip 2024-07-29 20:53:55 +07:00
Quan HL a595faa378 deepgram tts support endpoint for on-premise 2024-07-29 19:45:13 +07:00
Quan HL 7fd1e1a3c3 allow configure STS session expiry 2024-07-29 18:18:37 +07:00
Dave Horton 7f6a3d349c 0.1.11 2024-06-14 07:37:19 -04:00
Dave Horton 50429ff535 bump version 2024-06-14 07:36:53 -04:00
Dave Horton 3bf0ef8ea3 Merge pull request #80 from jambonz/fix/aws_arnrole
fix aws arnrole
2024-06-14 07:34:22 -04:00
Quan HL e9a5e83e36 wip 2024-06-14 15:04:19 +07:00
Quan HL 86a64ac091 wip 2024-06-14 15:00:56 +07:00
Quan HL 97e06b3ab3 wip 2024-06-14 10:23:22 +07:00
Quan HL 09e833d910 wip 2024-06-14 10:19:55 +07:00
Quan HL 8c4e12e54f wip 2024-06-14 10:18:41 +07:00
Quan HL 2642bd71a4 wip 2024-06-14 10:16:57 +07:00
Quan HL c4feac916f fix aws arnrole 2024-06-14 10:15:23 +07:00
11 changed files with 1729 additions and 1438 deletions
+5
View File
@@ -13,6 +13,11 @@ jobs:
with:
node-version: lts/*
- run: npm install
- name: Install Docker Compose
run: |
sudo curl -L "https://github.com/docker/compose/releases/download/1.29.2/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
sudo chmod +x /usr/local/bin/docker-compose
docker-compose --version
- run: npm run jslint
- run: sudo apt update && sudo apt install -y squid
- run: sudo cp test/squid.conf /etc/squid/squid.conf
+1 -1
View File
@@ -1,3 +1,3 @@
npm audit
#npm audit
npm run jslint:fix || true
npm test
+2
View File
@@ -1,6 +1,7 @@
const JAMBONES_TTS_TRIM_SILENCE = process.env.JAMBONES_TTS_TRIM_SILENCE;
const JAMBONES_DISABLE_TTS_STREAMING = process.env.JAMBONES_DISABLE_TTS_STREAMING;
const JAMBONES_DISABLE_AZURE_TTS_STREAMING = process.env.JAMBONES_DISABLE_AZURE_TTS_STREAMING;
const JAMBONES_EAGERLY_PRE_CACHE_AUDIO = process.env.JAMBONES_EAGERLY_PRE_CACHE_AUDIO;
const JAMBONES_HTTP_PROXY_IP = process.env.JAMBONES_HTTP_PROXY_IP;
const JAMBONES_HTTP_PROXY_PORT = process.env.JAMBONES_HTTP_PROXY_PORT;
@@ -18,6 +19,7 @@ module.exports = {
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
TMP_FOLDER,
HTTP_TIMEOUT
};
+10 -9
View File
@@ -1,20 +1,22 @@
const { STSClient, GetSessionTokenCommand, AssumeRoleCommand } = require('@aws-sdk/client-sts');
const {makeAwsKey, noopLogger} = require('./utils');
const debug = require('debug')('jambonz:speech-utils');
const EXPIRY = 3600;
const EXPIRY = process.env.AWS_STS_SESSION_DURATION || 3600;
// by default reset aws session before expiry time 10 mins
const CACHE_EXPIRY = process.env.AWS_STS_SESSION_RESET_EXPIRY || (EXPIRY - 600);
async function getAwsAuthToken(
logger, createHash, retrieveHash,
awsAccessKeyId, awsSecretAccessKey, awsRegion, roleArn = null) {
{accessKeyId, secretAccessKey, region, roleArn}) {
logger = logger || noopLogger;
try {
const key = makeAwsKey(roleArn || awsAccessKeyId);
const key = makeAwsKey(roleArn || accessKeyId);
const obj = await retrieveHash(key);
if (obj) return {...obj, servedFromCache: true};
let data;
if (roleArn) {
const stsClient = new STSClient({ region: awsRegion});
const stsClient = new STSClient({ region });
const roleToAssume = { RoleArn: roleArn, RoleSessionName: 'Jambonz_Speech', DurationSeconds: EXPIRY};
const command = new AssumeRoleCommand(roleToAssume);
@@ -22,10 +24,10 @@ async function getAwsAuthToken(
} else {
/* access token not found in cache, so generate it using STS */
const stsClient = new STSClient({
region: awsRegion,
region,
credentials: {
accessKeyId: awsAccessKeyId,
secretAccessKey: awsSecretAccessKey,
accessKeyId,
secretAccessKey,
}
});
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
@@ -39,8 +41,7 @@ async function getAwsAuthToken(
securityToken: data.Credentials.SessionToken
};
/* expire 10 minutes before the hour, so we don't lose the use of it during a call */
createHash(key, credentials, EXPIRY - 600)
createHash(key, credentials, CACHE_EXPIRY)
.catch((err) => logger.error(err, `Error saving hash for key ${key}`));
return {...credentials, servedFromCache: false};
+6 -1
View File
@@ -107,7 +107,12 @@ const getAwsVoices = async(_client, createHash, retrieveHash, logger, credential
} else if (roleArn) {
client = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
credentials: await getAwsAuthToken(
logger, createHash, retrieveHash,
{
region,
roleArn
}),
});
} else {
client = new PollyClient({region});
+86 -15
View File
@@ -20,7 +20,8 @@ const {
createKryptonClient,
createRivaClient,
noopLogger,
makeFilePath
makeFilePath,
makePlayhtKey
} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
@@ -44,6 +45,7 @@ const {
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
const OpenAI = require('openai');
@@ -86,7 +88,7 @@ const trimTrailingSilence = (buffer) => {
*/
async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, disableTtsStreaming, options
disableTtsCache, renderForCaching = false, disableTtsStreaming, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -151,21 +153,48 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
assert.ok(voice, 'synthAudio requires voice when verbio is used');
assert.ok(credentials.client_id, 'synthAudio requires client_id when verbio is used');
assert.ok(credentials.client_secret, 'synthAudio requires client_secret when verbio is used');
} else if ('deepgram' === vendor) {
if (!credentials.deepgram_tts_uri) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
}
}
const key = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text
text,
renderForCaching
});
let filePath;
filePath = makeFilePath(vendor, key, salt);
filePath = makeFilePath({vendor, key, salt, renderForCaching});
debug(`synth key is ${key}`);
let cached;
if (!disableTtsCache) {
cached = await client.get(key);
/**
* If we are using tts streaming and also precaching audio, audio could have been cached by streaming (r8)
* or here in speech-utils due to precaching (mp3), so we need to check for both keys.
*/
if (!cached && JAMBONES_EAGERLY_PRE_CACHE_AUDIO) {
const preCachekey = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text,
renderForCaching: true
});
cached = await client.get(preCachekey);
if (cached) {
// Precache audio is available update filpath with precache file extension.
filePath = makeFilePath({vendor, key, salt, renderForCaching: true});
}
}
}
if (cached) {
// found in cache - extend the expiry and use it
@@ -216,7 +245,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
});
break;
case 'playht':
audioBuffer = await synthPlayHT(logger, {
audioBuffer = await synthPlayHT(client, logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
break;
@@ -281,7 +310,12 @@ const synthPolly = async(createHash, retrieveHash, logger,
} else if (roleArn) {
polly = new PollyClient({
region,
credentials: await getAwsAuthToken(logger, createHash, retrieveHash, null, null, region, roleArn),
credentials: await getAwsAuthToken(
logger, createHash, retrieveHash,
{
region,
roleArn
}),
});
} else {
// AWS RoleArn assigned to Instance profile
@@ -722,12 +756,40 @@ const synthElevenlabs = async(logger, {
}
};
const synthPlayHT = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
const synthPlayHT = async(client, logger, {
credentials, options, stats, voice, language, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, user_id, voice_engine, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
let synthesizeUrl = 'https://api.play.ht/api/v2/tts/stream';
// If model is play3.0, the synthesizeUrl is got from authentication endpoint
if (voice_engine === 'Play3.0') {
try {
const post = bent('https://api.play.ht', 'POST', 'json', 201, {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
'Accept': 'application/json'
});
const key = makePlayhtKey(api_key);
const url = await client.get(key);
if (!url) {
const {inference_address, expires_at_ms} = await post('/api/v3/auth');
synthesizeUrl = inference_address;
const expiry = Math.floor((expires_at_ms - Date.now()) / 1000 - 30);
await client.set(key, inference_address, 'EX', expiry);
} else {
// Use cached URL
synthesizeUrl = url;
}
} catch (err) {
logger.info({err}, 'synth PlayHT returned error for authentication version 3.0');
stats.increment('tts.count', ['vendor:playht', 'accepted:no']);
throw err;
}
}
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
@@ -736,6 +798,7 @@ const synthPlayHT = async(logger, {
params += ',vendor=playht';
params += `,voice=${voice}`;
params += `,voice_engine=${voice_engine}`;
params += `,synthesize_url=${synthesizeUrl}`;
params += ',write_cache_file=1';
if (opts.quality) params += `,quality=${opts.quality}`;
if (opts.speed) params += `,speed=${opts.speed}`;
@@ -745,6 +808,8 @@ const synthPlayHT = async(logger, {
if (opts.voice_guidance) params += `,voice_guidance=${opts.voice_guidance}`;
if (opts.style_guidance) params += `,style_guidance=${opts.style_guidance}`;
if (opts.text_guidance) params += `,text_guidance=${opts.text_guidance}`;
if (opts.top_p) params += `,top_p=${opts.top_p}`;
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
params += '}';
return {
@@ -755,14 +820,18 @@ const synthPlayHT = async(logger, {
}
try {
const post = bent('https://api.play.ht', 'POST', 'buffer', {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
const post = bent('POST', 'buffer', {
...(voice_engine !== 'Play3.0' && {
'AUTHORIZATION': api_key,
'X-USER-ID': user_id,
}),
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post('/api/v2/tts/stream', {
const mp3 = await post(synthesizeUrl, {
text,
...(voice_engine === 'Play3.0' && { language }),
voice,
voice_engine,
output_format: 'mp3',
@@ -904,13 +973,14 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
};
const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
const {api_key} = credentials;
const {api_key, deepgram_tts_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=deepgram';
params += `,voice=${model}`;
params += ',write_cache_file=1';
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
params += '}';
return {
@@ -920,8 +990,9 @@ const synthDeepgram = async(logger, {credentials, stats, model, text, renderForC
};
}
try {
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
const post = bent(deepgram_tts_uri || 'https://api.deepgram.com', 'POST', 'buffer', {
// on-premise deepgram does not require to have api_key
...(api_key && {'Authorization': `Token ${api_key}`}),
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
+26 -9
View File
@@ -16,29 +16,31 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
*/
//const nuanceClientMap = new Map();
function makeSynthKey({account_sid = '', vendor, language, voice, engine = '', text}) {
function makeSynthKey({
account_sid = '', vendor, language, voice, engine = '', text,
renderForCaching = false}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension(vendor);
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension({vendor, renderForCaching});
const key = `tts${accountKey}:${namespace}:${hexHashKey}`;
return key;
}
function makeFilePath(vendor, key, salt = '') {
const extension = getFileExtension(vendor);
function makeFilePath({vendor, key, salt = '', renderForCaching = false}) {
const extension = getFileExtension({vendor, renderForCaching});
return `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt}`)}.${extension}`;
}
function getFileExtension(vendor) {
function getFileExtension({vendor, renderForCaching = false}) {
const mp3Extension = 'mp3';
const r8Extension = 'r8';
switch (vendor) {
case 'azure':
case 'microsoft':
if (!JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
if (!renderForCaching && !JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
return r8Extension;
} else {
return mp3Extension;
@@ -46,16 +48,25 @@ function getFileExtension(vendor) {
case 'deepgram':
case 'elevenlabs':
case 'rimlabs':
if (!JAMBONES_DISABLE_TTS_STREAMING) {
return r8Extension;
} else {
case 'playht':
if (renderForCaching || JAMBONES_DISABLE_TTS_STREAMING) {
return mp3Extension;
} else {
return r8Extension;
}
case 'nuance':
case 'nvidia':
case 'verbio':
return r8Extension;
default:
// If vendor is custom
if (vendor.startsWith('custom')) {
if (renderForCaching || JAMBONES_DISABLE_TTS_STREAMING) {
return mp3Extension;
} else {
return r8Extension;
}
}
return mp3Extension;
}
}
@@ -87,6 +98,11 @@ function makeAwsKey(awsAccessKeyId) {
return `aws:${hash.digest('hex')}`;
}
function makePlayhtKey(apiKey) {
const hash = crypto.createHash('sha1');
hash.update(apiKey);
return `playht:${hash.digest('hex')}`;
}
function makeVerbioKey(client_id) {
const hash = crypto.createHash('sha1');
hash.update(client_id);
@@ -160,6 +176,7 @@ module.exports = {
makeSynthKey,
makeNuanceKey,
makeIbmKey,
makePlayhtKey,
makeAwsKey,
makeVerbioKey,
getNuanceAccessToken,
+1565 -1391
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.1.9",
"version": "0.1.18",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -36,7 +36,7 @@
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.36.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
},
+10 -2
View File
@@ -19,12 +19,20 @@ test('AWS - create and cache auth token', async(t) => {
return;
}
try {
let obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
let obj = await getAwsAuthToken({
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
region: process.env.AWS_REGION
});
//console.log({obj}, 'received auth token from AWS');
t.ok(obj.securityToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
await sleep(250);
obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
obj = await getAwsAuthToken({
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
region: process.env.AWS_REGION
});
//console.log({obj}, 'received auth token from AWS - second request');
t.ok(obj.securityToken && obj.servedFromCache, 'successfully received access token from cache');
+16 -8
View File
@@ -574,9 +574,9 @@ test('Elevenlabs speech synth tests', async(t) => {
t.end(err);
}
client.quit();
})
});
test('PlayHT speech synth tests', async(t) => {
const testPlayHT = async(t, voice_engine) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -584,26 +584,26 @@ test('PlayHT speech synth tests', async(t) => {
t.pass('skipping PlayHT speech synth tests since PLAYHT_API_KEY or PLAYHT_USER_ID is/are not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
const text = 'Hi there and welcome to jambones! ' + Date.now();
try {
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'playht',
credentials: {
api_key: process.env.PLAYHT_API_KEY,
user_id: process.env.PLAYHT_USER_ID,
voice_engine: 'PlayHT2.0-turbo',
voice_engine,
options: JSON.stringify({
quality: "medium",
quality: 'medium',
speed: 1,
seed: 1,
temperature: 1,
emotion: "female_happy",
emotion: 'female_happy',
voice_guidance: 3,
style_guidance: 20,
text_guidance: 1,
})
},
language: 'en-US',
language: 'english',
voice: 's3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json',
text,
renderForCaching: true
@@ -615,6 +615,14 @@ test('PlayHT speech synth tests', async(t) => {
t.end(err);
}
client.quit();
};
test('PlayHT speech synth tests', async(t) => {
await testPlayHT(t, 'PlayHT2.0-turbo');
});
test('PlayHT3.0 speech synth tests', async(t) => {
await testPlayHT(t, 'Play3.0');
});
test('rimelabs speech synth tests', async(t) => {