Compare commits

..
19 Commits
Author SHA1 Message Date
Dave Horton ced1a0ef0d 0.0.42 2024-02-20 20:34:51 -05:00
Dave Horton 1609d0b205 Merge pull request #57 from jambonz/feat/whisper_tts_stream
support whisper streaming
2024-02-20 20:33:21 -05:00
Quan HL ef8ada2793 wip 2024-02-20 20:52:14 +07:00
Quan HL 444ad2522f rebase 2024-02-19 15:32:50 +07:00
Dave Horton 1caea60803 0.0.41 2024-02-12 21:06:48 -05:00
Dave Horton 97c3588cfd bug: JAMBONES_DISABLE_TTS_STREAMING is now the env 2024-02-12 21:06:38 -05:00
Dave Horton da3aa5aadb 0.0.40 2024-02-12 12:43:11 -05:00
Dave Horton 2fe89f132c change elevenlabs default to streaming, can be disabled by env 2024-02-12 12:41:21 -05:00
Dave Horton 4bca840ba2 0.0.39 2024-02-08 14:55:50 -05:00
Dave Horton f858ccb781 for tts streaming we need to replace CR and LF with spaces, as we can not send text with those characters to freeswitch currently 2024-02-08 14:55:39 -05:00
Quan HL 3cf9894b44 support whisper streaming 2024-02-05 11:49:38 +07:00
Dave Horton 436b15d648 0.0.38 2024-01-26 11:09:24 -05:00
Dave Horton c3b7ea4cd1 fix prev commit 2024-01-26 11:08:06 -05:00
Dave Horton 4b5430d61d 0.0.37 2024-01-26 09:54:51 -05:00
Dave Horton 9fc8fe8341 Merge pull request #56 from jambonz/feat/cache-streaming
add function to add an audio file generated externally to cache
2024-01-26 09:54:24 -05:00
Dave Horton b31e40b8a5 add function to add an audio file generated externally to cache 2024-01-26 09:52:32 -05:00
Dave Horton 62e1c69f69 0.0.36 2024-01-25 13:14:31 -05:00
Dave Horton bc68b672ac Merge pull request #55 from jambonz/tts-streaming-cache-attribute
add tts param to indicate caching
2024-01-25 13:14:00 -05:00
Dave Horton da1e279128 add tts param to indicate caching 2024-01-25 13:07:49 -05:00
6 changed files with 73 additions and 9 deletions
+1
View File
@@ -12,6 +12,7 @@ module.exports = (opts, logger) => {
client,
getTtsSize: require('./lib/get-tts-size').bind(null, client, logger),
purgeTtsCache: require('./lib/purge-tts-cache').bind(null, client, logger),
addFileToCache: require('./lib/add-file-to-cache').bind(null, client, logger),
synthAudio: require('./lib/synth-audio').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
+30
View File
@@ -0,0 +1,30 @@
const fs = require('fs/promises');
const {noopLogger, makeSynthKey} = require('./utils');
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
async function addFileToCache(client, logger, path,
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
let key;
logger = logger || noopLogger;
try {
key = makeSynthKey({
account_sid,
vendor,
language: language || '',
voice: voice || deploymentId,
engine,
text,
});
const audioBuffer = await fs.readFile(path);
await client.setex(key, EXPIRES, audioBuffer.toString('base64'));
} catch (err) {
logger.error(err, 'addFileToCache: Error');
return;
}
logger.debug(`addFileToCache: added ${path} to cache with key ${key}`);
return key;
}
module.exports = addFileToCache;
+30 -5
View File
@@ -143,6 +143,10 @@ async function synthAudio(client, logger, stats, { account_sid,
(
process.env.JAMBONES_TTS_TRIM_SILENCE &&
['microsoft', 'azure'].includes(vendor)
) ||
(
!process.env.JAMBONES_DISABLE_TTS_STREAMING &&
vendor === 'elevenlabs'
)
) {
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
@@ -206,6 +210,10 @@ async function synthAudio(client, logger, stats, { account_sid,
}
break;
case 'whisper':
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text, renderForCaching});
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
return audioBuffer;
}
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
break;
case 'deepgram':
@@ -607,12 +615,13 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
params += ',write_cache_file=1';
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
@@ -620,7 +629,7 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
@@ -651,8 +660,24 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
const {api_key, model_id, baseURL, timeout} = credentials;
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (speed) params += `,speed=${speed}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const openai = new OpenAI.OpenAI({
apiKey: api_key,
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.35",
"version": "0.0.42",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.0.35",
"version": "0.0.42",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.35",
"version": "0.0.42",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
+9 -1
View File
@@ -20,7 +20,7 @@ const stats = {
test('Google speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
const {synthAudio, addFileToCache, client} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping google speech synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
@@ -58,6 +58,14 @@ test('Google speech synth tests', async(t) => {
});
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
const success = await addFileToCache(opts.filePath, {
vendor: 'google',
language: 'en-GB',
gender: 'FEMALE',
text: 'This is a test. This is only a test'
});
t.ok(success, `successfully added ${opts.filePath} to cache`);
opts = await synthAudio(stats, {
vendor: 'google',
credentials: {