Compare commits

...
10 Commits
Author SHA1 Message Date
Dave Horton 4eabfbe4b7 0.0.43 2024-03-11 09:25:14 -04:00
Dave Horton dbfabeaddf Merge pull request #61 from jambonz/fix/duplicate-calls
remove seemingly redundant code, reintroduce param to force bypass of…
2024-03-09 18:25:15 -05:00
Dave Horton d0dfd07204 remove seemingly redundant code, reintroduce param to force bypass of tts streaming 2024-03-07 13:44:40 -05:00
Dave Horton 04a2466f54 Merge pull request #60 from jambonz/fix/deepgram_tts
update deepgram tts endpoint
2024-03-05 09:12:58 -05:00
Quan HL 0f9a9edc4d update deepgram tts endpoint 2024-03-05 20:55:19 +07:00
Dave Horton ced1a0ef0d 0.0.42 2024-02-20 20:34:51 -05:00
Dave Horton 1609d0b205 Merge pull request #57 from jambonz/feat/whisper_tts_stream
support whisper streaming
2024-02-20 20:33:21 -05:00
Quan HL ef8ada2793 wip 2024-02-20 20:52:14 +07:00
Quan HL 444ad2522f rebase 2024-02-19 15:32:50 +07:00
Quan HL 3cf9894b44 support whisper streaming 2024-02-05 11:49:38 +07:00
4 changed files with 33 additions and 18 deletions
+29 -14
View File
@@ -77,7 +77,7 @@ const trimTrailingSilence = (buffer) => {
*/
async function synthAudio(client, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching, options
disableTtsCache, renderForCaching, disableTtsStreaming, options
}) {
let audioBuffer;
let servedFromCache = false;
@@ -200,17 +200,14 @@ async function synthAudio(client, logger, stats, { account_sid,
break;
case 'elevenlabs':
audioBuffer = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, filePath
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
});
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
return audioBuffer;
}
else {
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
}
if (audioBuffer?.filePath) return audioBuffer;
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
audioBuffer = await synthWhisper(logger, {
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
if (audioBuffer?.filePath) return audioBuffer;
break;
case 'deepgram':
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
@@ -607,12 +604,14 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
}
};
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
const synthElevenlabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
@@ -656,8 +655,24 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
const {api_key, model_id, baseURL, timeout} = credentials;
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (speed) params += `,speed=${speed}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const openai = new OpenAI.OpenAI({
apiKey: api_key,
@@ -682,7 +697,7 @@ const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
const {api_key} = credentials;
try {
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', {
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
'Authorization': `Token ${api_key}`,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.41",
"version": "0.0.43",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.0.41",
"version": "0.0.43",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.41",
"version": "0.0.43",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
+1 -1
View File
@@ -537,7 +537,7 @@ test('Deepgram speech synth tests', async(t) => {
credentials: {
api_key: process.env.DEEPGRAM_API_KEY
},
model: 'alpha-asteria-en-v2',
model: 'aura-asteria-en',
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);