mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ced1a0ef0d | ||
|
|
1609d0b205 | ||
|
|
ef8ada2793 | ||
|
|
444ad2522f | ||
|
|
1caea60803 | ||
|
|
97c3588cfd | ||
|
|
da3aa5aadb | ||
|
|
2fe89f132c | ||
|
|
4bca840ba2 | ||
|
|
f858ccb781 | ||
|
|
3cf9894b44 | ||
|
|
436b15d648 | ||
|
|
c3b7ea4cd1 |
@@ -4,10 +4,11 @@ const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; /
|
||||
|
||||
async function addFileToCache(client, logger, path,
|
||||
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
|
||||
let key;
|
||||
logger = logger || noopLogger;
|
||||
|
||||
try {
|
||||
const key = makeSynthKey({
|
||||
key = makeSynthKey({
|
||||
account_sid,
|
||||
vendor,
|
||||
language: language || '',
|
||||
@@ -15,15 +16,15 @@ async function addFileToCache(client, logger, path,
|
||||
engine,
|
||||
text,
|
||||
});
|
||||
const audioBuffer = fs.readFile(path);
|
||||
const audioBuffer = await fs.readFile(path);
|
||||
await client.setex(key, EXPIRES, audioBuffer.toString('base64'));
|
||||
} catch (err) {
|
||||
logger.error(err, 'addFileToCache: Error');
|
||||
return false;
|
||||
return;
|
||||
}
|
||||
|
||||
logger.debug(`addFileToCache: added ${path} to cache`);
|
||||
return true;
|
||||
logger.debug(`addFileToCache: added ${path} to cache with key ${key}`);
|
||||
return key;
|
||||
}
|
||||
|
||||
module.exports = addFileToCache;
|
||||
|
||||
+29
-5
@@ -143,6 +143,10 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
(
|
||||
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
||||
['microsoft', 'azure'].includes(vendor)
|
||||
) ||
|
||||
(
|
||||
!process.env.JAMBONES_DISABLE_TTS_STREAMING &&
|
||||
vendor === 'elevenlabs'
|
||||
)
|
||||
) {
|
||||
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
||||
@@ -206,6 +210,10 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
}
|
||||
break;
|
||||
case 'whisper':
|
||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text, renderForCaching});
|
||||
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
|
||||
return audioBuffer;
|
||||
}
|
||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
|
||||
break;
|
||||
case 'deepgram':
|
||||
@@ -607,8 +615,8 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
||||
const {api_key, model_id, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
|
||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
|
||||
let params = '';
|
||||
params += `{api_key=${api_key}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
@@ -621,7 +629,7 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
@@ -652,8 +660,24 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
||||
}
|
||||
};
|
||||
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
||||
const {api_key, model_id, baseURL, timeout} = credentials;
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching}) => {
|
||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching) {
|
||||
let params = '';
|
||||
params += `{api_key=${api_key}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
params += `,voice=${voice}`;
|
||||
params += ',write_cache_file=1';
|
||||
if (speed) params += `,speed=${speed}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
try {
|
||||
const openai = new OpenAI.OpenAI({
|
||||
apiKey: api_key,
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.37",
|
||||
"version": "0.0.42",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.37",
|
||||
"version": "0.0.42",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.496.0",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.37",
|
||||
"version": "0.0.42",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
|
||||
Reference in New Issue
Block a user