mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1e57d00b55 | ||
|
|
4ead5ee417 | ||
|
|
a517f37473 | ||
|
|
3ac5a98e3a | ||
|
|
689fab6857 | ||
|
|
be3c484527 | ||
|
|
9827f7405d | ||
|
|
906a6b52b8 | ||
|
|
1f0b9ff539 | ||
|
|
b6f8368357 | ||
|
|
6bc487b3d1 | ||
|
|
9699abe0a3 | ||
|
|
5e1e7b17c6 | ||
|
|
1f52cd4f08 |
@@ -1,2 +1,3 @@
|
||||
# speech-utils
|
||||
TTS-related speech utilities for jambonz
|
||||
TTS-related speech utilities for jambonz.
|
||||
|
||||
|
||||
@@ -3,10 +3,12 @@ const {noopLogger} = require('./lib/utils');
|
||||
module.exports = (opts, logger) => {
|
||||
logger = logger || noopLogger;
|
||||
let client = opts.redis_client;
|
||||
if (!client) {
|
||||
const {client: redisClient} = require('@jambonz/realtimedb-helpers')(opts, logger);
|
||||
client = redisClient;
|
||||
}
|
||||
const {
|
||||
client: redisClient,
|
||||
createHash,
|
||||
retrieveHash
|
||||
} = require('@jambonz/realtimedb-helpers')(opts, logger);
|
||||
client = opts.redis_client || redisClient;
|
||||
|
||||
return {
|
||||
client,
|
||||
@@ -15,6 +17,7 @@ module.exports = (opts, logger) => {
|
||||
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
|
||||
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
|
||||
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
|
||||
getAwsAuthToken: require('./lib/get-aws-sts-token').bind(null, logger, createHash, retrieveHash),
|
||||
getTtsVoices: require('./lib/get-tts-voices').bind(null, client, logger),
|
||||
};
|
||||
};
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
const { STSClient, GetSessionTokenCommand } = require('@aws-sdk/client-sts');
|
||||
const {makeAwsKey, noopLogger} = require('./utils');
|
||||
const debug = require('debug')('jambonz:speech-utils');
|
||||
const EXPIRY = 3600;
|
||||
|
||||
async function getAwsAuthToken(
|
||||
logger,
|
||||
createHash, retrieveHash,
|
||||
awsAccessKeyId, awsSecretAccessKey, awsRegion) {
|
||||
logger = logger || noopLogger;
|
||||
try {
|
||||
const key = makeAwsKey(awsAccessKeyId);
|
||||
const obj = await retrieveHash(key);
|
||||
if (obj) return {...obj, servedFromCache: true};
|
||||
|
||||
/* access token not found in cache, so generate it using STS */
|
||||
const stsClient = new STSClient({
|
||||
region: awsRegion,
|
||||
credentials: {
|
||||
accessKeyId: awsAccessKeyId,
|
||||
secretAccessKey: awsSecretAccessKey,
|
||||
}
|
||||
});
|
||||
const command = new GetSessionTokenCommand({DurationSeconds: EXPIRY});
|
||||
const data = await stsClient.send(command);
|
||||
|
||||
const credentials = {
|
||||
accessKeyId: data.Credentials.AccessKeyId,
|
||||
secretAccessKey: data.Credentials.SecretAccessKey,
|
||||
sessionToken: data.Credentials.SessionToken
|
||||
};
|
||||
|
||||
/* expire 10 minutes before the hour, so we don't lose the use of it during a call */
|
||||
createHash(key, credentials, EXPIRY - 600)
|
||||
.catch((err) => logger.error(err, `Error saving hash for key ${key}`));
|
||||
|
||||
return {...credentials, servedFromCache: false};
|
||||
} catch (err) {
|
||||
debug(err, 'getAwsAuthToken: Error retrieving AWS auth token');
|
||||
logger.error(err, 'getAwsAuthToken: Error retrieving AWS auth token');
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = getAwsAuthToken;
|
||||
+10
-6
@@ -76,7 +76,7 @@ const trimTrailingSilence = (buffer) => {
|
||||
* the synthesized audio, and a variable indicating whether it was served from cache
|
||||
*/
|
||||
async function synthAudio(client, logger, stats, { account_sid,
|
||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache
|
||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache, options
|
||||
}) {
|
||||
let audioBuffer;
|
||||
let servedFromCache = false;
|
||||
@@ -194,7 +194,7 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
||||
break;
|
||||
case 'elevenlabs':
|
||||
audioBuffer = await synthElevenlabs(logger, {credentials, stats, language, voice, text, filePath});
|
||||
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
|
||||
break;
|
||||
case 'whisper':
|
||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
|
||||
@@ -585,21 +585,25 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
|
||||
}
|
||||
};
|
||||
|
||||
const synthElevenlabs = async(logger, {credentials, stats, language, voice, text}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text}) => {
|
||||
const {api_key, model_id, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
const optimize_streaming_latency = opts.optimize_streaming_latency ?
|
||||
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
|
||||
try {
|
||||
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
|
||||
'xi-api-key': api_key,
|
||||
'Accept': 'audio/mpeg',
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const mp3 = await post(`/v1/text-to-speech/${voice}`, {
|
||||
const mp3 = await post(`/v1/text-to-speech/${voice}${optimize_streaming_latency}`, {
|
||||
text,
|
||||
model_id,
|
||||
voice_settings: {
|
||||
stability: 0.5,
|
||||
similarity_boost: 0.5
|
||||
}
|
||||
},
|
||||
...opts
|
||||
});
|
||||
return mp3;
|
||||
} catch (err) {
|
||||
|
||||
@@ -43,6 +43,12 @@ function makeIbmKey(apiKey) {
|
||||
return `ibm:${hash.digest('hex')}`;
|
||||
}
|
||||
|
||||
function makeAwsKey(awsAccessKeyId) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(awsAccessKeyId);
|
||||
return `aws:${hash.digest('hex')}`;
|
||||
}
|
||||
|
||||
function makeNuanceKey(clientId, secret, scope) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(`${clientId}:${secret}:${scope}`);
|
||||
@@ -110,6 +116,7 @@ module.exports = {
|
||||
makeSynthKey,
|
||||
makeNuanceKey,
|
||||
makeIbmKey,
|
||||
makeAwsKey,
|
||||
getNuanceAccessToken,
|
||||
createNuanceClient,
|
||||
createKryptonClient,
|
||||
|
||||
Generated
+3054
-2381
File diff suppressed because it is too large
Load Diff
+3
-2
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.26",
|
||||
"version": "0.0.29",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -25,6 +25,7 @@
|
||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.359.0",
|
||||
"@aws-sdk/client-sts": "^3.458.0",
|
||||
"@google-cloud/text-to-speech": "^4.2.1",
|
||||
"@grpc/grpc-js": "^1.8.13",
|
||||
"@jambonz/realtimedb-helpers": "^0.8.7",
|
||||
@@ -33,7 +34,7 @@
|
||||
"form-urlencoded": "^6.1.0",
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^8.0.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "1.32.0",
|
||||
"openai": "^4.16.2",
|
||||
"undici": "^5.21.0"
|
||||
},
|
||||
|
||||
+40
@@ -0,0 +1,40 @@
|
||||
const test = require('tape').test ;
|
||||
const config = require('config');
|
||||
const opts = config.get('redis');
|
||||
const logger = require('pino')({level: 'error'});
|
||||
process.on('unhandledRejection', (reason, p) => {
|
||||
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
|
||||
});
|
||||
|
||||
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
test('AWS - create and cache auth token', async(t) => {
|
||||
const fn = require('..');
|
||||
const {client, getAwsAuthToken} = fn(opts, logger);
|
||||
|
||||
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
|
||||
t.pass('skipping AWS auth token tests since no AWS credentials provided');
|
||||
t.end();
|
||||
client.quit();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
let obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
|
||||
//console.log({obj}, 'received auth token from AWS');
|
||||
t.ok(obj.sessionToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
|
||||
|
||||
await sleep(250);
|
||||
obj = await getAwsAuthToken(process.env.AWS_ACCESS_KEY_ID, process.env.AWS_SECRET_ACCESS_KEY, process.env.AWS_REGION);
|
||||
//console.log({obj}, 'received auth token from AWS - second request');
|
||||
t.ok(obj.sessionToken && obj.servedFromCache, 'successfully received access token from cache');
|
||||
|
||||
await client.flushall();
|
||||
t.end();
|
||||
}
|
||||
catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
require('./docker_start');
|
||||
require('./synth');
|
||||
require('./list-voices');
|
||||
require('./aws');
|
||||
require('./ibm');
|
||||
require('./nuance');
|
||||
require('./docker_stop');
|
||||
|
||||
+10
-1
@@ -459,7 +459,16 @@ test('Elevenlabs speech synth tests', async(t) => {
|
||||
vendor: 'elevenlabs',
|
||||
credentials: {
|
||||
api_key: process.env.ELEVENLABS_API_KEY,
|
||||
model_id: process.env.ELEVENLABS_MODEL_ID
|
||||
model_id: process.env.ELEVENLABS_MODEL_ID,
|
||||
options: JSON.stringify({
|
||||
optimize_streaming_latency: 1,
|
||||
voice_settings: {
|
||||
similarity_boost: 1,
|
||||
stability: 0.8,
|
||||
style: 1,
|
||||
use_speaker_boost: true
|
||||
}
|
||||
})
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: process.env.ELEVENLABS_VOICE_ID,
|
||||
|
||||
Reference in New Issue
Block a user