Compare commits

...
Author SHA1 Message Date
snyk-bot 3d42bd3bde fix: upgrade @google-cloud/text-to-speech from 5.0.2 to 5.3.0
Snyk has created this PR to upgrade @google-cloud/text-to-speech from 5.0.2 to 5.3.0.

See this package in npm:
@google-cloud/text-to-speech

See this project in Snyk:
https://app.snyk.io/org/davehorton/project/6f41db32-9ff7-45b4-83ae-9512305f387b?utm_source=github&utm_medium=referral&page=upgrade-pr
2024-08-21 08:16:57 +00:00
Dave Horton 8f216e64d8 0.1.15 2024-08-12 09:30:09 -04:00
Dave Horton e1f4486e01 bump version 2024-08-12 09:27:12 -04:00
Dave Horton b0d6272974 Merge pull request #84 from jambonz/feat/precache_audio_with_tts_stream
support precache audio with tts stream enabled
2024-08-12 09:26:00 -04:00
Quan HL ef23b0807a add comment 2024-08-12 20:16:24 +07:00
Quan HL ab7e25243d improve on check precache 2024-08-12 20:10:43 +07:00
Quan HL bf0ea14423 install docker 2024-08-12 18:40:47 +07:00
Quan HL 305dabd84b wip 2024-08-12 18:35:48 +07:00
Quan HL b6a3fa5081 support precache audio with tts stream enabled 2024-08-12 18:29:01 +07:00
Dave Horton 73feadc4c4 0.1.13 2024-08-06 11:01:09 -04:00
Dave Horton aad0f4d62c Merge pull request #82 from jambonz/feat/deepgram_tts_endpoint
deepgram tts support endpoint for on-premise
2024-08-06 11:00:38 -04:00
Hoan Luu Huu 602b0cc60e Merge branch 'main' into feat/deepgram_tts_endpoint 2024-07-31 14:03:41 +07:00
Quan HL 461178e726 wip 2024-07-29 20:56:55 +07:00
Quan HL e0e4d47340 wip 2024-07-29 20:53:55 +07:00
Quan HL a595faa378 deepgram tts support endpoint for on-premise 2024-07-29 19:45:13 +07:00
6 changed files with 64 additions and 24 deletions
+5
View File
@@ -13,6 +13,11 @@ jobs:
with:
node-version: lts/*
- run: npm install
- name: Install Docker Compose
run: |
sudo curl -L "https://github.com/docker/compose/releases/download/1.29.2/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
sudo chmod +x /usr/local/bin/docker-compose
docker-compose --version
- run: npm run jslint
- run: sudo apt update && sudo apt install -y squid
- run: sudo cp test/squid.conf /etc/squid/squid.conf
+2
View File
@@ -1,6 +1,7 @@
const JAMBONES_TTS_TRIM_SILENCE = process.env.JAMBONES_TTS_TRIM_SILENCE;
const JAMBONES_DISABLE_TTS_STREAMING = process.env.JAMBONES_DISABLE_TTS_STREAMING;
const JAMBONES_DISABLE_AZURE_TTS_STREAMING = process.env.JAMBONES_DISABLE_AZURE_TTS_STREAMING;
const JAMBONES_EAGERLY_PRE_CACHE_AUDIO = process.env.JAMBONES_EAGERLY_PRE_CACHE_AUDIO;
const JAMBONES_HTTP_PROXY_IP = process.env.JAMBONES_HTTP_PROXY_IP;
const JAMBONES_HTTP_PROXY_PORT = process.env.JAMBONES_HTTP_PROXY_PORT;
@@ -18,6 +19,7 @@ module.exports = {
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
TMP_FOLDER,
HTTP_TIMEOUT
};
+36 -6
View File
@@ -44,6 +44,7 @@ const {
JAMBONES_HTTP_PROXY_IP,
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
const OpenAI = require('openai');
@@ -86,7 +87,7 @@ const trimTrailingSilence = (buffer) => {
*/
async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, disableTtsStreaming, options
disableTtsCache, renderForCaching = false, disableTtsStreaming, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -151,21 +152,48 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
assert.ok(voice, 'synthAudio requires voice when verbio is used');
assert.ok(credentials.client_id, 'synthAudio requires client_id when verbio is used');
assert.ok(credentials.client_secret, 'synthAudio requires client_secret when verbio is used');
} else if ('deepgram' === vendor) {
if (!credentials.deepgram_tts_uri) {
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
}
}
const key = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text
text,
renderForCaching
});
let filePath;
filePath = makeFilePath(vendor, key, salt);
filePath = makeFilePath({vendor, key, salt, renderForCaching});
debug(`synth key is ${key}`);
let cached;
if (!disableTtsCache) {
cached = await client.get(key);
/**
* If we are using tts streaming and also precaching audio, audio could have been cached by streaming (r8)
* or here in speech-utils due to precaching (mp3), so we need to check for both keys.
*/
if (!cached && JAMBONES_EAGERLY_PRE_CACHE_AUDIO) {
const preCachekey = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text,
renderForCaching: true
});
cached = await client.get(preCachekey);
if (cached) {
// Precache audio is available update filpath with precache file extension.
filePath = makeFilePath({vendor, key, salt, renderForCaching: true});
}
}
}
if (cached) {
// found in cache - extend the expiry and use it
@@ -909,13 +937,14 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
};
const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
const {api_key} = credentials;
const {api_key, deepgram_tts_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=deepgram';
params += `,voice=${model}`;
params += ',write_cache_file=1';
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
params += '}';
return {
@@ -925,8 +954,9 @@ const synthDeepgram = async(logger, {credentials, stats, model, text, renderForC
};
}
try {
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
const post = bent(deepgram_tts_uri || 'https://api.deepgram.com', 'POST', 'buffer', {
// on-premise deepgram does not require to have api_key
...(api_key && {'Authorization': `Token ${api_key}`}),
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
+9 -7
View File
@@ -16,29 +16,31 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
*/
//const nuanceClientMap = new Map();
function makeSynthKey({account_sid = '', vendor, language, voice, engine = '', text}) {
function makeSynthKey({
account_sid = '', vendor, language, voice, engine = '', text,
renderForCaching = false}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension(vendor);
const namespace = vendor.startsWith('custom') ? vendor : getFileExtension({vendor, renderForCaching});
const key = `tts${accountKey}:${namespace}:${hexHashKey}`;
return key;
}
function makeFilePath(vendor, key, salt = '') {
const extension = getFileExtension(vendor);
function makeFilePath({vendor, key, salt = '', renderForCaching = false}) {
const extension = getFileExtension({vendor, renderForCaching});
return `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt}`)}.${extension}`;
}
function getFileExtension(vendor) {
function getFileExtension({vendor, renderForCaching = false}) {
const mp3Extension = 'mp3';
const r8Extension = 'r8';
switch (vendor) {
case 'azure':
case 'microsoft':
if (!JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
if (!renderForCaching && !JAMBONES_DISABLE_TTS_STREAMING || JAMBONES_TTS_TRIM_SILENCE) {
return r8Extension;
} else {
return mp3Extension;
@@ -46,7 +48,7 @@ function getFileExtension(vendor) {
case 'deepgram':
case 'elevenlabs':
case 'rimlabs':
if (!JAMBONES_DISABLE_TTS_STREAMING) {
if (!renderForCaching && !JAMBONES_DISABLE_TTS_STREAMING) {
return r8Extension;
} else {
return mp3Extension;
+10 -9
View File
@@ -1,17 +1,17 @@
{
"name": "@jambonz/speech-utils",
"version": "0.1.12",
"version": "0.1.15",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.1.12",
"version": "0.1.15",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^5.0.2",
"@google-cloud/text-to-speech": "^5.3.0",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
@@ -1230,9 +1230,10 @@
}
},
"node_modules/@google-cloud/text-to-speech": {
"version": "5.0.2",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz",
"integrity": "sha512-Q11Ddh9eHKSDA3E/KSqMITgVprXb0XgIKuJP9F5ScJ1T9h+DNrbgIU7shd0QOlPqb8ruQRiTOqL08+Mq5R89Ow==",
"version": "5.3.0",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.3.0.tgz",
"integrity": "sha512-jc0TEHkSGrQErlaFCJ59YF/NKZUMBfrdpEc5it+AXVWbLvnVTbgCrxZn4ny+PFs310kyguaqLQ2qwScTDAaFxA==",
"license": "Apache-2.0",
"dependencies": {
"google-gax": "^4.0.3"
},
@@ -8338,9 +8339,9 @@
"dev": true
},
"@google-cloud/text-to-speech": {
"version": "5.0.2",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.0.2.tgz",
"integrity": "sha512-Q11Ddh9eHKSDA3E/KSqMITgVprXb0XgIKuJP9F5ScJ1T9h+DNrbgIU7shd0QOlPqb8ruQRiTOqL08+Mq5R89Ow==",
"version": "5.3.0",
"resolved": "https://registry.npmjs.org/@google-cloud/text-to-speech/-/text-to-speech-5.3.0.tgz",
"integrity": "sha512-jc0TEHkSGrQErlaFCJ59YF/NKZUMBfrdpEc5it+AXVWbLvnVTbgCrxZn4ny+PFs310kyguaqLQ2qwScTDAaFxA==",
"requires": {
"google-gax": "^4.0.3"
}
+2 -2
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.1.12",
"version": "0.1.15",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -28,7 +28,7 @@
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^5.0.2",
"@google-cloud/text-to-speech": "^5.3.0",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",