Compare commits

..
34 Commits
Author SHA1 Message Date
Dave Horton eb2c39072b 0.0.22 2023-10-14 13:18:08 +02:00
Dave Horton e5932ffc18 Merge pull request #34 from jambonz/feat/elevenlabs
add elevenlabs
2023-10-14 07:17:14 -04:00
Quan HL 0a98c6a376 fix review comment 2023-10-14 18:13:04 +07:00
Quan HL ea153e9833 add elevenlabs 2023-10-12 14:22:15 +07:00
Dave Horton b5daeff047 0.0.21 2023-09-08 07:58:11 -04:00
Dave Horton da02926c9a Merge pull request #31 from jambonz/fix/onprem-azure
fix raw audio downloaded from onprem azure
2023-09-08 07:57:36 -04:00
Quan HL da3cdbb7aa fix raw audio downloaded from onprem azure 2023-09-08 15:42:57 +07:00
Dave Horton 625f147137 0.0.20 2023-08-30 21:08:52 -04:00
Dave Horton 2e5687978e 0.0.19 2023-08-30 21:08:41 -04:00
Dave Horton 897481d34c Merge pull request #29 from jambonz/feat/azure_fromhost
support self hosted microsoft
2023-08-30 21:07:44 -04:00
Quan HL bd5282e681 fix 2023-08-28 20:05:55 +07:00
Quan HL 95a1384f02 wip 2023-08-25 15:57:43 +07:00
Quan HL 35deeecf70 wip 2023-08-25 15:57:23 +07:00
Quan HL 9d2ac3273f wip 2023-08-25 13:53:09 +07:00
Quan HL b1049aad7f wip 2023-08-25 13:49:04 +07:00
Quan HL 40f51e7509 wip 2023-08-11 18:25:57 +07:00
Quan HL a0e2fe167c wip 2023-08-11 16:04:11 +07:00
Quan HL 1fa853faa3 wip 2023-08-11 13:30:11 +07:00
Quan HL 95e8d942b8 fix jslint 2023-08-09 18:00:17 +07:00
Quan HL 7a91876cd7 support self hosted microsoft 2023-08-09 17:56:38 +07:00
Dave Horton d07344ba3b Merge pull request #28 from jambonz/revert/ssml-silence-trim
revert change to _not_ trim silence when azure ssml is used
2023-07-25 12:34:22 -04:00
Dave Horton 44d8af2a96 revert change to _not_ trim silence when azure ssml is used 2023-07-25 11:08:06 -04:00
Dave Horton b530db9a62 0.0.18 2023-07-25 07:40:33 -04:00
Dave Horton 4c166c8eb4 synth_audio: dont trim silence for Azure when using SSML 2023-07-25 07:40:28 -04:00
Dave Horton 8246dbea21 0.0.17 2023-07-19 10:10:39 -04:00
Dave Horton 0084f6a468 update deps 2023-07-19 10:09:31 -04:00
Dave Horton 98f679f43a 0.0.16 2023-07-19 10:04:17 -04:00
Dave Horton 7e7841b5ff Merge pull request #23 from jambonz/feature/trim-silence
trim trailing silence from azure tts when JAMBONES_TTS_TRIM_SILENCE i…
2023-07-19 10:03:42 -04:00
Dave Horton 75ce537db1 linting 2023-07-19 10:02:10 -04:00
Dave Horton 830be783b8 trim trailing silence from azure tts when JAMBONES_TTS_TRIM_SILENCE is set 2023-07-19 10:00:31 -04:00
Dave Horton 38c3219425 Merge pull request #21 from jambonz/snyk-fix-255777535fcb48b70c964213f7dd9fe8
[Snyk] Security upgrade @aws-sdk/client-polly from 3.303.0 to 3.347.1
2023-06-07 13:09:27 -04:00
snyk-bot e37b96a9c2 fix: package.json & package-lock.json to reduce vulnerabilities
The following vulnerabilities are fixed with an upgrade:
- https://snyk.io/vuln/SNYK-JS-FASTXMLPARSER-5668858
2023-06-07 15:35:04 +00:00
Dave Horton c184fbae26 0.0.15 2023-06-03 09:15:48 -04:00
Dave Horton 66d33ebd60 change default tts cache duration to 4 hours 2023-06-03 09:15:45 -04:00
5 changed files with 1640 additions and 1699 deletions
+123 -9
View File
@@ -14,7 +14,13 @@ const {
CancellationDetails,
SpeechSynthesisOutputFormat
} = sdk;
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
const {
makeSynthKey,
createNuanceClient,
createKryptonClient,
createRivaClient,
noopLogger
} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const {
SynthesisRequest,
@@ -30,9 +36,27 @@ const {
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
const debug = require('debug')('jambonz:realtimedb-helpers');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 24 * 60) * 60; // cache tts for 24 hours
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
const TMP_FOLDER = '/tmp';
const trimTrailingSilence = (buffer) => {
assert.ok(buffer instanceof Buffer, 'trimTrailingSilence - argument is not a Buffer');
let offset = buffer.length;
while (offset > 0) {
// Get 16-bit value from the buffer (read in reverse)
const value = buffer.readUInt16BE(offset - 2);
if (value !== 0) {
break;
}
offset -= 2;
}
// Trim the silence from the end
return offset === buffer.length ? buffer : buffer.subarray(0, offset);
};
/**
* Synthesize speech to an mp3 file, and also cache the generated speech
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
@@ -58,7 +82,8 @@ async function synthAudio(client, logger, stats, { account_sid,
let rtt;
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm'].includes(vendor) ||
assert.ok(['google', 'aws', 'polly', 'microsoft',
'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid, not ${vendor}`);
if ('google' === vendor) {
@@ -104,7 +129,12 @@ async function synthAudio(client, logger, stats, { account_sid,
text
});
let filePath;
if (['nuance', 'nvidia'].includes(vendor)) {
if (['nuance', 'nvidia'].includes(vendor) ||
(
process.env.JAMBONES_TTS_TRIM_SILENCE &&
['microsoft', 'azure'].includes(vendor)
)
) {
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
}
else filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.mp3`;
@@ -154,6 +184,9 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'wellsaid':
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, stats, language, voice, text, filePath});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
({ audioBuffer, filePath } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
@@ -274,6 +307,48 @@ const synthIbm = async(logger, {credentials, stats, voice, text}) => {
}
};
async function _synthOnPremMicrosoft(logger, {
credentials,
stats,
language,
voice,
text,
filePath
}) {
const {use_custom_tts, custom_tts_endpoint_url} = credentials;
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
try {
const trimSilence = filePath.endsWith('.r8');
const post = bent('POST', 'buffer', {
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
'Content-Type': 'application/ssml+xml',
'User-Agent': 'Jambonz'
});
const mp3 = await post(custom_tts_endpoint_url, content);
return mp3;
} catch (err) {
logger.info({err}, '_synthMicrosoftByHttp returned error');
throw err;
}
}
const synthMicrosoft = async(logger, {
credentials,
stats,
@@ -283,21 +358,35 @@ const synthMicrosoft = async(logger, {
filePath
}) => {
try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
if (use_custom_tts && custom_tts_endpoint_url) {
return await _synthOnPremMicrosoft(logger, {
credentials,
stats,
language,
voice,
text,
filePath
});
}
const trimSilence = filePath.endsWith('.r8');
let content = text;
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
speechConfig.speechSynthesisLanguage = language;
speechConfig.speechSynthesisVoiceName = voice;
if (use_custom_tts && custom_tts_endpoint) {
speechConfig.endpointId = custom_tts_endpoint;
}
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
@@ -323,7 +412,9 @@ const synthMicrosoft = async(logger, {
reject(cancellation.errorDetails);
break;
case ResultReason.SynthesizingAudioCompleted:
resolve(Buffer.from(result.audioData));
let buffer = Buffer.from(result.audioData);
if (trimSilence) buffer = trimTrailingSilence(buffer);
resolve(buffer);
synthesizer.close();
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
break;
@@ -481,6 +572,29 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, stats, language, voice, text}) => {
const {api_key, model_id} = credentials;
try {
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
'xi-api-key': api_key,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(`/v1/text-to-speech/${voice}`, {
text,
model_id,
voice_settings: {
stability: 0.5,
similarity_boost: 0.5
}
});
return mp3;
} catch (err) {
logger.info({err}, 'synthEvenlabs returned error');
throw err;
}
};
const getFileExtFromMime = (mime) => {
switch (mime) {
case 'audio/wav':
-1
View File
@@ -106,7 +106,6 @@ const createRivaClient = async(rivaUri) => {
return client;
};
module.exports = {
makeSynthKey,
makeNuanceKey,
+1485 -1686
View File
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.14",
"version": "0.0.22",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -24,7 +24,7 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.303.0",
"@aws-sdk/client-polly": "^3.359.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"bent": "^7.3.12",
@@ -32,7 +32,7 @@
"form-urlencoded": "^6.1.0",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
"ioredis": "^5.3.2",
"undici": "^5.21.0"
},
+29
View File
@@ -411,6 +411,35 @@ test('Custom Vendor speech synth tests', async(t) => {
client.quit();
});
test('Elevenlabs speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.ELEVENLABS_API_KEY || !process.env.ELEVENLABS_VOICE_ID || !process.env.ELEVENLABS_MODEL_ID) {
t.pass('skipping IBM Watson speech synth tests since IBM_TTS_API_KEY or IBM_TTS_API_KEY not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
vendor: 'elevenlabs',
credentials: {
api_key: process.env.ELEVENLABS_API_KEY,
model_id: process.env.ELEVENLABS_MODEL_ID
},
language: 'en-US',
voice: process.env.ELEVENLABS_VOICE_ID,
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
})
test('TTS Cache tests', async(t) => {
const fn = require('..');
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);