mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-09 21:14:25 +00:00
remove seemingly redundant code, reintroduce param to force bypass of tts streaming
This commit is contained in:
+12
-17
@@ -77,7 +77,7 @@ const trimTrailingSilence = (buffer) => {
|
|||||||
*/
|
*/
|
||||||
async function synthAudio(client, logger, stats, { account_sid,
|
async function synthAudio(client, logger, stats, { account_sid,
|
||||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
||||||
disableTtsCache, renderForCaching, options
|
disableTtsCache, renderForCaching, disableTtsStreaming, options
|
||||||
}) {
|
}) {
|
||||||
let audioBuffer;
|
let audioBuffer;
|
||||||
let servedFromCache = false;
|
let servedFromCache = false;
|
||||||
@@ -200,21 +200,14 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
break;
|
break;
|
||||||
case 'elevenlabs':
|
case 'elevenlabs':
|
||||||
audioBuffer = await synthElevenlabs(logger, {
|
audioBuffer = await synthElevenlabs(logger, {
|
||||||
credentials, options, stats, language, voice, text, renderForCaching, filePath
|
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
|
||||||
});
|
});
|
||||||
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
return audioBuffer;
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
|
|
||||||
}
|
|
||||||
break;
|
break;
|
||||||
case 'whisper':
|
case 'whisper':
|
||||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text, renderForCaching});
|
audioBuffer = await synthWhisper(logger, {
|
||||||
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
|
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
|
||||||
return audioBuffer;
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
}
|
|
||||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
|
|
||||||
break;
|
break;
|
||||||
case 'deepgram':
|
case 'deepgram':
|
||||||
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
|
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
|
||||||
@@ -611,12 +604,14 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
|
const synthElevenlabs = async(logger, {
|
||||||
|
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
|
||||||
|
}) => {
|
||||||
const {api_key, model_id, options: credOpts} = credentials;
|
const {api_key, model_id, options: credOpts} = credentials;
|
||||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||||
|
|
||||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
let params = '';
|
let params = '';
|
||||||
params += `{api_key=${api_key}`;
|
params += `{api_key=${api_key}`;
|
||||||
params += `,model_id=${model_id}`;
|
params += `,model_id=${model_id}`;
|
||||||
@@ -660,10 +655,10 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching}) => {
|
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
|
||||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
let params = '';
|
let params = '';
|
||||||
params += `{api_key=${api_key}`;
|
params += `{api_key=${api_key}`;
|
||||||
params += `,model_id=${model_id}`;
|
params += `,model_id=${model_id}`;
|
||||||
|
|||||||
+1
-1
@@ -537,7 +537,7 @@ test('Deepgram speech synth tests', async(t) => {
|
|||||||
credentials: {
|
credentials: {
|
||||||
api_key: process.env.DEEPGRAM_API_KEY
|
api_key: process.env.DEEPGRAM_API_KEY
|
||||||
},
|
},
|
||||||
model: 'alpha-asteria-en-v2',
|
model: 'aura-asteria-en',
|
||||||
text,
|
text,
|
||||||
});
|
});
|
||||||
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
||||||
|
|||||||
Reference in New Issue
Block a user