Compare commits

...
6 Commits
Author SHA1 Message Date
Dave Horton 1e57d00b55 0.0.29 2023-11-30 08:56:56 -05:00
Dave Horton 4ead5ee417 Merge pull request #47 from jambonz/feat/update_elevenlabs
support elevenlabs options
2023-11-30 08:53:19 -05:00
Quan HL a517f37473 wip 2023-11-30 16:37:44 +07:00
Quan HL 3ac5a98e3a get elevenlabs options from syntheizier.options 2023-11-30 15:56:22 +07:00
Quan HL 689fab6857 get elevenlabs options from syntheizier.options 2023-11-30 15:48:26 +07:00
Quan HL be3c484527 support elevenlabs options 2023-11-30 13:29:51 +07:00
4 changed files with 23 additions and 10 deletions
+10 -6
View File
@@ -76,7 +76,7 @@ const trimTrailingSilence = (buffer) => {
* the synthesized audio, and a variable indicating whether it was served from cache
*/
async function synthAudio(client, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -194,7 +194,7 @@ async function synthAudio(client, logger, stats, { account_sid,
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, stats, language, voice, text, filePath});
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
@@ -585,21 +585,25 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, stats, language, voice, text}) => {
const {api_key, model_id} = credentials;
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
const optimize_streaming_latency = opts.optimize_streaming_latency ?
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
try {
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
'xi-api-key': api_key,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(`/v1/text-to-speech/${voice}`, {
const mp3 = await post(`/v1/text-to-speech/${voice}${optimize_streaming_latency}`, {
text,
model_id,
voice_settings: {
stability: 0.5,
similarity_boost: 0.5
}
},
...opts
});
return mp3;
} catch (err) {
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.28",
"version": "0.0.29",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.0.28",
"version": "0.0.29",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.359.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.28",
"version": "0.0.29",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
+10 -1
View File
@@ -459,7 +459,16 @@ test('Elevenlabs speech synth tests', async(t) => {
vendor: 'elevenlabs',
credentials: {
api_key: process.env.ELEVENLABS_API_KEY,
model_id: process.env.ELEVENLABS_MODEL_ID
model_id: process.env.ELEVENLABS_MODEL_ID,
options: JSON.stringify({
optimize_streaming_latency: 1,
voice_settings: {
similarity_boost: 1,
stability: 0.8,
style: 1,
use_speaker_boost: true
}
})
},
language: 'en-US',
voice: process.env.ELEVENLABS_VOICE_ID,