Compare commits

..
5 Commits
3 changed files with 1664 additions and 2138 deletions
+32 -3
View File
@@ -76,7 +76,8 @@ const trimTrailingSilence = (buffer) => {
* the synthesized audio, and a variable indicating whether it was served from cache * the synthesized audio, and a variable indicating whether it was served from cache
*/ */
async function synthAudio(client, logger, stats, { account_sid, async function synthAudio(client, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId, disableTtsCache, options vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, options
}) { }) {
let audioBuffer; let audioBuffer;
let servedFromCache = false; let servedFromCache = false;
@@ -194,7 +195,15 @@ async function synthAudio(client, logger, stats, { account_sid,
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath}); audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break; break;
case 'elevenlabs': case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath}); audioBuffer = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, filePath
});
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
return audioBuffer;
}
else {
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
}
break; break;
case 'whisper': case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text}); audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
@@ -594,9 +603,29 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
} }
}; };
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text}) => { const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
const {api_key, model_id, options: credOpts} = credentials; const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}'); const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
const optimize_streaming_latency = opts.optimize_streaming_latency ? const optimize_streaming_latency = opts.optimize_streaming_latency ?
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : ''; `?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
try { try {
+1619 -2122
View File
File diff suppressed because it is too large Load Diff
+13 -13
View File
@@ -1,6 +1,6 @@
{ {
"name": "@jambonz/speech-utils", "name": "@jambonz/speech-utils",
"version": "0.0.33", "version": "0.0.34",
"description": "TTS-related speech utilities for jambonz", "description": "TTS-related speech utilities for jambonz",
"main": "index.js", "main": "index.js",
"author": "Dave Horton", "author": "Dave Horton",
@@ -24,26 +24,26 @@
}, },
"homepage": "https://github.com/jambonz/speech-utils#readme", "homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": { "dependencies": {
"@aws-sdk/client-polly": "^3.359.0", "@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.458.0", "@aws-sdk/client-sts": "^3.496.0",
"@google-cloud/text-to-speech": "^4.2.1", "@google-cloud/text-to-speech": "^5.0.2",
"@grpc/grpc-js": "^1.8.13", "@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7", "@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12", "bent": "^7.3.12",
"debug": "^4.3.4", "debug": "^4.3.4",
"form-urlencoded": "^6.1.0", "form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2", "google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0", "ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.32.0", "microsoft-cognitiveservices-speech-sdk": "1.34.0",
"openai": "^4.16.2", "openai": "^4.25.0",
"undici": "^5.21.0" "undici": "^6.4.0"
}, },
"devDependencies": { "devDependencies": {
"config": "^3.3.9", "config": "^3.3.10",
"eslint": "^8.33.0", "eslint": "^8.56.0",
"eslint-plugin-promise": "^6.1.1", "eslint-plugin-promise": "^6.1.1",
"nyc": "^15.1.0", "nyc": "^15.1.0",
"pino": "^7.2.0", "pino": "^8.17.0",
"tape": "^5.1.1" "tape": "^5.7.3"
} }
} }