Compare commits

..
20 Commits
Author SHA1 Message Date
Dave Horton 1b1a0f19d0 0.0.34 2024-01-22 15:18:21 -05:00
Dave Horton 119ac50f7f Merge pull request #53 from jambonz/elevenlabs-streaming
Elevenlabs streaming
2024-01-22 15:17:33 -05:00
Dave Horton cb479f04d5 add param to synthAuydio to indicate whether audio is being generated specifically for caching purposes 2024-01-22 08:10:00 -05:00
Dave Horton 7e21e0b666 changes to support elevenlabs tts streaming 2024-01-21 21:36:53 -05:00
Dave Horton dabdb5b584 if streaming env is set prepare to use streaming tts 2024-01-20 14:15:13 -05:00
Dave Horton f36ba027d0 0.0.33 2023-12-25 22:13:29 -05:00
Dave Horton 08aae32975 Merge pull request #51 from jambonz/feat/deepgram
support deepgram
2023-12-25 22:10:50 -05:00
Quan HL 4cfc730d92 support deepgram 2023-12-26 09:26:31 +07:00
Dave Horton 4ef8538bcd 0.0.32 2023-12-18 10:21:12 -05:00
Dave Horton 16a746398f add CI badge to README 2023-12-06 10:06:43 -05:00
Dave Horton 4a75f353ee Merge pull request #35 from jambonz/feature/azure-proxy-support
add support for SetProxy when using azure tts
2023-12-06 09:57:11 -05:00
Dave Horton bbff7963fd make test explicit 2023-12-06 09:54:19 -05:00
Dave Horton 7b1d226403 test proxy using azure 2023-12-06 09:54:16 -05:00
Dave Horton 95c29ce105 squid config file so we can test azure proxy setting 2023-12-06 09:53:35 -05:00
Dave Horton a46ae01d9c add support for JAMBONES_HTTP_PROXY_IP and JAMBONES_HTTP_PROXY_PORT for azure tts 2023-12-06 09:53:35 -05:00
Dave Horton 562dd0ac79 0.0.31 2023-12-05 20:36:43 -05:00
Dave Horton a0612bd1f9 Merge pull request #50 from jambonz/fix/reuse_redis
cannot assign redis-client as createHash and retrieveHash
2023-12-03 19:58:58 -05:00
Quan HL 2dcbf2b0f1 cannot assign redis-client as createHash and retrieveHash 2023-12-04 06:17:02 +07:00
Dave Horton 048b7e871e 0.0.30 2023-11-30 16:11:34 -05:00
Dave Horton 4cae96eacb rename aws token from sessionToken to securityToken for consistency with AWS docs 2023-11-30 16:11:23 -05:00
10 changed files with 1820 additions and 2149 deletions
+5
View File
@@ -14,6 +14,9 @@ jobs:
node-version: lts/*
- run: npm install
- run: npm run jslint
- run: sudo apt update && sudo apt install -y squid
- run: sudo cp test/squid.conf /etc/squid/squid.conf
- run: sudo systemctl start squid
- run: npm test
env:
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }}
@@ -29,3 +32,5 @@ jobs:
ELEVENLABS_API_KEY: ${{ secrets.ELEVENLABS_API_KEY }}
ELEVENLABS_VOICE_ID: ${{ secrets.ELEVENLABS_VOICE_ID }}
ELEVENLABS_MODEL_ID: ${{ secrets.ELEVENLABS_MODEL_ID }}
JAMBONES_HTTP_PROXY_IP: 127.0.0.1
JAMBONES_HTTP_PROXY_PORT: 3128
+1 -1
View File
@@ -1,3 +1,3 @@
# speech-utils
# speech-utils ![CI](https://github.com/jambonz/speech-utils/workflows/CI/badge.svg)
TTS-related speech utilities for jambonz.
+1 -3
View File
@@ -2,13 +2,11 @@ const {noopLogger} = require('./lib/utils');
module.exports = (opts, logger) => {
logger = logger || noopLogger;
let client = opts.redis_client;
const {
client: redisClient,
client,
createHash,
retrieveHash
} = require('@jambonz/realtimedb-helpers')(opts, logger);
client = opts.redis_client || redisClient;
return {
client,
+1 -1
View File
@@ -27,7 +27,7 @@ async function getAwsAuthToken(
const credentials = {
accessKeyId: data.Credentials.AccessKeyId,
secretAccessKey: data.Credentials.SecretAccessKey,
sessionToken: data.Credentials.SessionToken
securityToken: data.Credentials.SessionToken
};
/* expire 10 minutes before the hour, so we don't lose the use of it during a call */
+63 -6
View File
@@ -76,7 +76,8 @@ const trimTrailingSilence = (buffer) => {
* the synthesized audio, and a variable indicating whether it was served from cache
*/
async function synthAudio(client, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache, options
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -84,7 +85,7 @@ async function synthAudio(client, logger, stats, { account_sid,
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper'].includes(vendor) ||
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs', 'whisper', 'deepgram'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
if ('google' === vendor) {
@@ -194,11 +195,22 @@ async function synthAudio(client, logger, stats, { account_sid,
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
audioBuffer = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, filePath
});
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
return audioBuffer;
}
else {
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
}
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
break;
case 'deepgram':
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
@@ -400,6 +412,12 @@ const synthMicrosoft = async(logger, {
speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
logger.debug(
`synthMicrosoft: using proxy ${process.env.JAMBONES_HTTP_PROXY_IP}:${process.env.JAMBONES_HTTP_PROXY_PORT}`);
speechConfig.setProxy(process.env.JAMBONES_HTTP_PROXY_IP, process.env.JAMBONES_HTTP_PROXY_PORT);
}
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
@@ -585,9 +603,29 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text}) => {
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
const optimize_streaming_latency = opts.optimize_streaming_latency ?
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
try {
@@ -634,8 +672,27 @@ const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
stats.increment('tts.count', ['vendor:openai', 'accepted:no']);
throw err;
}
}
;
};
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
const {api_key} = credentials;
try {
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(`/v1/speak?model=${model}`, {
text
});
return mp3;
} catch (err) {
logger.info({err}, 'synth Deepgram returned error');
stats.increment('tts.count', ['vendor:deepgram', 'accepted:no']);
throw err;
}
};
const getFileExtFromMime = (mime) => {
switch (mime) {
case 'audio/wav':
+1619 -2122
View File
File diff suppressed because it is too large Load Diff
+13 -13
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.29",
"version": "0.0.34",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -24,26 +24,26 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.359.0",
"@aws-sdk/client-sts": "^3.458.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^5.0.2",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
"debug": "^4.3.4",
"form-urlencoded": "^6.1.0",
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.32.0",
"openai": "^4.16.2",
"undici": "^5.21.0"
"microsoft-cognitiveservices-speech-sdk": "1.34.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
},
"devDependencies": {
"config": "^3.3.9",
"eslint": "^8.33.0",
"config": "^3.3.10",
"eslint": "^8.56.0",
"eslint-plugin-promise": "^6.1.1",
"nyc": "^15.1.0",
"pino": "^7.2.0",
"tape": "^5.1.1"
"pino": "^8.17.0",
"tape": "^5.7.3"
}
}
+2 -2
View File
@@ -21,12 +21,12 @@ test('AWS - create and cache auth token', async(t) => {
try {
let obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
//console.log({obj}, 'received auth token from AWS');
t.ok(obj.sessionToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
t.ok(obj.securityToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
await sleep(250);
obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
//console.log({obj}, 'received auth token from AWS - second request');
t.ok(obj.sessionToken && obj.servedFromCache, 'successfully received access token from cache');
t.ok(obj.securityToken && obj.servedFromCache, 'successfully received access token from cache');
await client.flushall();
t.end();
+85
View File
@@ -0,0 +1,85 @@
#
# Recommended minimum configuration:
#
# Example rule allowing access from your local networks.
# Adapt to list your (internal) IP networks from where browsing
# should be allowed
acl localnet src 0.0.0.1-0.255.255.255 # RFC 1122 "this" network (LAN)
acl localnet src 10.0.0.0/8 # RFC 1918 local private network (LAN)
acl localnet src 100.64.0.0/10 # RFC 6598 shared address space (CGN)
acl localnet src 169.254.0.0/16 # RFC 3927 link-local (directly plugged) machines
acl localnet src 172.16.0.0/12 # RFC 1918 local private network (LAN)
acl localnet src 192.168.0.0/16 # RFC 1918 local private network (LAN)
acl localnet src fc00::/7 # RFC 4193 local private network range
acl localnet src fe80::/10 # RFC 4291 link-local (directly plugged) machines
acl SSL_ports port 443
acl Safe_ports port 80 # http
acl Safe_ports port 21 # ftp
acl Safe_ports port 443 # https
acl Safe_ports port 70 # gopher
acl Safe_ports port 210 # wais
acl Safe_ports port 1025-65535 # unregistered ports
acl Safe_ports port 280 # http-mgmt
acl Safe_ports port 488 # gss-http
acl Safe_ports port 591 # filemaker
acl Safe_ports port 777 # multiling http
#
# Recommended minimum Access Permission configuration:
#
# Deny requests to certain unsafe ports
http_access deny !Safe_ports
# Deny CONNECT to other than secure SSL ports
http_access allow CONNECT !SSL_ports
# Only allow cachemgr access from localhost
http_access allow localhost manager
http_access deny manager
# This default configuration only allows localhost requests because a more
# permissive Squid installation could introduce new attack vectors into the
# network by proxying external TCP connections to unprotected services.
http_access allow localhost
# The two deny rules below are unnecessary in this default configuration
# because they are followed by a "deny all" rule. However, they may become
# critically important when you start allowing external requests below them.
# Protect web applications running on the same server as Squid. They often
# assume that only local users can access them at "localhost" ports.
http_access deny to_localhost
# Protect cloud servers that provide local users with sensitive info about
# their server via certain well-known link-local (a.k.a. APIPA) addresses.
http_access deny to_linklocal
#
# INSERT YOUR OWN RULE(S) HERE TO ALLOW ACCESS FROM YOUR CLIENTS
#
# For example, to allow access from your local networks, you may uncomment the
# following rule (and/or add rules that match your definition of "local"):
# http_access allow localnet
# And finally deny all other access to this proxy
http_access deny all
# Squid normally listens to port 3128
http_port 3128
# Uncomment and adjust the following to add a disk cache directory.
#cache_dir ufs /usr/local/var/cache/squid 100 16 256
# Leave coredumps in the first cache dir
coredump_dir /usr/local/var/cache/squid
#
# Add any of your own refresh_pattern entries above these.
#
refresh_pattern ^ftp: 1440 20% 10080
refresh_pattern ^gopher: 1440 0% 1440
refresh_pattern -i (/cgi-bin/|\?) 0 0% 0
refresh_pattern . 0 20% 4320
+30 -1
View File
@@ -182,7 +182,9 @@ test('Azure speech synth tests', async(t) => {
text: longText,
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
t.pass('successfully used proxy to reach microsoft tts service');
}
opts = await synthAudio(stats, {
vendor: 'microsoft',
@@ -512,6 +514,33 @@ test('whisper speech synth tests', async(t) => {
client.quit();
})
test('Deepgram speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.DEEPGRAM_API_KEY) {
t.pass('skipping Deepgram speech synth tests since DEEPGRAM_API_KEY');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'deepgram',
credentials: {
api_key: process.env.DEEPGRAM_API_KEY
},
model: 'alpha-asteria-en-v2',
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('TTS Cache tests', async(t) => {
const fn = require('..');
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);