mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0d98f73c43 | ||
|
|
36670e0080 | ||
|
|
61672f9868 | ||
|
|
ddeca8eb99 | ||
|
|
6e8271b2f6 | ||
|
|
74a4938eb6 | ||
|
|
cb50f603cd | ||
|
|
9a524f00bc | ||
|
|
545df0b770 | ||
|
|
9409405769 | ||
|
|
7dc3bbdb01 | ||
|
|
f8f9de2645 | ||
|
|
467d7ede26 | ||
|
|
fc211ab2e7 | ||
|
|
536f8aab31 | ||
|
|
69b3fdffbe | ||
|
|
53fe72d89e | ||
|
|
e79c15c5da | ||
|
|
a4a427e174 | ||
|
|
b697bc5268 | ||
|
|
d35d7f0aec | ||
|
|
ed1c564fa2 | ||
|
|
7189d471c1 | ||
|
|
04080cc5ec | ||
|
|
3e5ab4af27 | ||
|
|
f06dddd2f7 | ||
|
|
7b4a71f55d | ||
|
|
d552b65618 | ||
|
|
f701b50244 | ||
|
|
199e502fbe | ||
|
|
6769779189 | ||
|
|
4b43c3986c | ||
|
|
e1292772e6 |
@@ -25,7 +25,7 @@ function getExtensionAndSampleRate(path) {
|
||||
}
|
||||
|
||||
async function addFileToCache(client, logger, path,
|
||||
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
|
||||
{account_sid, vendor, language, voice, deploymentId, engine, model, text, instructions}) {
|
||||
let key;
|
||||
logger = logger || noopLogger;
|
||||
|
||||
@@ -36,7 +36,9 @@ async function addFileToCache(client, logger, path,
|
||||
language: language || '',
|
||||
voice: voice || deploymentId,
|
||||
engine,
|
||||
model,
|
||||
text,
|
||||
instructions
|
||||
});
|
||||
const [extension, sampleRate] = getExtensionAndSampleRate(path);
|
||||
const audioBuffer = await fs.readFile(path);
|
||||
|
||||
@@ -12,7 +12,7 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
* @returns {object} result - {error, purgedCount}
|
||||
*/
|
||||
async function purgeTtsCache(client, logger, {all, account_sid, vendor,
|
||||
language, voice, deploymentId, engine, text} = {all: true}) {
|
||||
language, voice, deploymentId, engine, model, text, instructions} = {all: true}) {
|
||||
logger = logger || noopLogger;
|
||||
|
||||
let purgedCount = 0, error;
|
||||
@@ -33,7 +33,9 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
|
||||
language: language || '',
|
||||
voice: voice || deploymentId,
|
||||
engine,
|
||||
model,
|
||||
text,
|
||||
instructions
|
||||
});
|
||||
purgedCount = await client.del(key);
|
||||
if (purgedCount === 0) error = 'Specified item not found';
|
||||
|
||||
+44
-10
@@ -89,7 +89,7 @@ const trimTrailingSilence = (buffer) => {
|
||||
*/
|
||||
async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
|
||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
||||
disableTtsCache, renderForCaching = false, disableTtsStreaming, options
|
||||
disableTtsCache, renderForCaching = false, disableTtsStreaming, options, instructions
|
||||
}) {
|
||||
let audioData;
|
||||
let servedFromCache = false;
|
||||
@@ -169,7 +169,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
language: language || '',
|
||||
voice: voice || deploymentId,
|
||||
engine,
|
||||
text
|
||||
// model or model_id is used to identify the tts cache.
|
||||
model: model || credentials.model_id,
|
||||
text,
|
||||
instructions
|
||||
});
|
||||
|
||||
debug(`synth key is ${key}`);
|
||||
@@ -213,7 +216,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, text});
|
||||
break;
|
||||
case 'nvidia':
|
||||
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text});
|
||||
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text,
|
||||
renderForCaching, disableTtsStreaming});
|
||||
break;
|
||||
case 'ibm':
|
||||
audioData = await synthIbm(logger, {credentials, stats, voice, text});
|
||||
@@ -239,7 +243,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
break;
|
||||
case 'whisper':
|
||||
audioData = await synthWhisper(logger, {
|
||||
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
|
||||
credentials, stats, voice, text, instructions, renderForCaching, disableTtsStreaming});
|
||||
break;
|
||||
case 'verbio':
|
||||
audioData = await synthVerbio(client, logger, {
|
||||
@@ -706,8 +710,24 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
|
||||
});
|
||||
};
|
||||
|
||||
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
|
||||
const synthNvidia = async(client, logger, {
|
||||
credentials, stats, language, voice, model, text, renderForCaching, disableTtsStreaming
|
||||
}) => {
|
||||
const {riva_server_uri} = credentials;
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '';
|
||||
params += `{riva_server_uri=${riva_server_uri}`;
|
||||
params += `,voice=${voice}`;
|
||||
params += `,language=${language}`;
|
||||
params += ',write_cache_file=1';
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
let rivaClient, request;
|
||||
const sampleRate = 8000;
|
||||
try {
|
||||
@@ -787,9 +807,13 @@ const synthElevenlabs = async(logger, {
|
||||
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
|
||||
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
|
||||
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
|
||||
if (opts.voice_settings?.speed !== null && opts.voice_settings?.speed !== undefined)
|
||||
params += `,speed=${opts.voice_settings.speed}`;
|
||||
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
|
||||
if (opts.previous_text) params += `,previous_text=${opts.previous_text}`;
|
||||
if (opts.next_text) params += `,next_text=${opts.next_text}`;
|
||||
if (opts.pronunciation_dictionary_locators && Array.isArray(opts.pronunciation_dictionary_locators))
|
||||
params += `,pronunciation_dictionary_locators=${JSON.stringify(opts.pronunciation_dictionary_locators)}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -831,11 +855,10 @@ const synthElevenlabs = async(logger, {
|
||||
const synthPlayHT = async(client, logger, {
|
||||
credentials, options, stats, voice, language, text, renderForCaching, disableTtsStreaming
|
||||
}) => {
|
||||
const {api_key, user_id, voice_engine, options: credOpts} = credentials;
|
||||
const {api_key, user_id, voice_engine, playht_tts_uri, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
|
||||
let synthesizeUrl = 'https://api.play.ht/api/v2/tts/stream';
|
||||
|
||||
let synthesizeUrl = playht_tts_uri ? `${playht_tts_uri}/api/v2/tts/stream` : 'https://api.play.ht/api/v2/tts/stream';
|
||||
// If model is play3.0, the synthesizeUrl is got from authentication endpoint
|
||||
if (voice_engine === 'Play3.0') {
|
||||
try {
|
||||
@@ -924,7 +947,7 @@ const synthPlayHT = async(client, logger, {
|
||||
};
|
||||
|
||||
const synthRimelabs = async(logger, {
|
||||
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
|
||||
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming
|
||||
}) => {
|
||||
const {api_key, model_id, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
@@ -935,10 +958,16 @@ const synthRimelabs = async(logger, {
|
||||
params += `{api_key=${api_key}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
params += ',vendor=rimelabs';
|
||||
params += `,language=${language}`;
|
||||
params += `,voice=${voice}`;
|
||||
params += ',write_cache_file=1';
|
||||
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
|
||||
if (opts.reduceLatency) params += `,reduce_latency=${opts.reduceLatency}`;
|
||||
// Arcana model parameters
|
||||
if (opts.temperature) params += `,temperature=${opts.temperature}`;
|
||||
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
|
||||
if (opts.top_p) params += `,top_p=${opts.top_p}`;
|
||||
if (opts.max_tokens) params += `,max_tokens=${opts.max_tokens}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -960,6 +989,7 @@ const synthRimelabs = async(logger, {
|
||||
text,
|
||||
modelId: model_id,
|
||||
samplingRate: sampleRate,
|
||||
lang: language,
|
||||
...opts
|
||||
});
|
||||
return {
|
||||
@@ -1018,7 +1048,8 @@ const synthVerbio = async(client, logger, {credentials, stats, voice, text, rend
|
||||
}
|
||||
};
|
||||
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, text, instructions,
|
||||
renderForCaching, disableTtsStreaming}) => {
|
||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
@@ -1029,6 +1060,8 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
|
||||
params += `,voice=${voice}`;
|
||||
params += ',write_cache_file=1';
|
||||
if (speed) params += `,speed=${speed}`;
|
||||
// comma is used to separated parameters in freeswitch tts module
|
||||
if (instructions) params += `,instructions=${instructions.replace(/\n/g, ' ').replace(/,/g, ';')}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -1048,6 +1081,7 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
|
||||
model: model_id,
|
||||
voice,
|
||||
input: text,
|
||||
...(instructions && {instructions}),
|
||||
response_format: 'mp3'
|
||||
});
|
||||
return {
|
||||
|
||||
+10
-2
@@ -17,9 +17,17 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
//const nuanceClientMap = new Map();
|
||||
|
||||
function makeSynthKey({
|
||||
account_sid = '', vendor, language, voice, engine = '', text}) {
|
||||
account_sid = '',
|
||||
vendor,
|
||||
language,
|
||||
voice,
|
||||
engine = '',
|
||||
model = '',
|
||||
text,
|
||||
instructions = '',
|
||||
}) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
|
||||
hash.update(`${language}:${vendor}:${voice}:${engine}:${model}:${text}:${instructions}`);
|
||||
const hexHashKey = hash.digest('hex');
|
||||
const accountKey = account_sid ? `:${account_sid}` : '';
|
||||
const key = `tts${accountKey}:${hexHashKey}`;
|
||||
|
||||
Generated
+389
-712
File diff suppressed because it is too large
Load Diff
+4
-3
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.2.1",
|
||||
"version": "0.2.10",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -26,6 +26,7 @@
|
||||
},
|
||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||
"dependencies": {
|
||||
"23": "^0.0.0",
|
||||
"@aws-sdk/client-polly": "^3.496.0",
|
||||
"@aws-sdk/client-sts": "^3.496.0",
|
||||
"@cartesia/cartesia-js": "^2.1.0",
|
||||
@@ -38,8 +39,8 @@
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^8.0.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
|
||||
"openai": "^4.25.0",
|
||||
"undici": "^6.4.0"
|
||||
"openai": "^4.98.0",
|
||||
"undici": "^7.5.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"config": "^3.3.11",
|
||||
|
||||
+44
-9
@@ -5,7 +5,7 @@ const fs = require('fs');
|
||||
const {makeSynthKey} = require('../lib/utils');
|
||||
const logger = require('pino')();
|
||||
const bent = require('bent');
|
||||
const getJSON = bent('json')
|
||||
const getJSON = bent('json');
|
||||
|
||||
process.on('unhandledRejection', (reason, p) => {
|
||||
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
|
||||
@@ -720,7 +720,7 @@ test('Cartesia speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('rimelabs speech synth tests', async(t) => {
|
||||
test('rimelabs speech synth tests mist', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
@@ -730,7 +730,7 @@ test('rimelabs speech synth tests', async(t) => {
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones!';
|
||||
try {
|
||||
let opts = await synthAudio(stats, {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'rimelabs',
|
||||
credentials: {
|
||||
api_key: process.env.RIMELABS_API_KEY,
|
||||
@@ -740,7 +740,7 @@ test('rimelabs speech synth tests', async(t) => {
|
||||
reduceLatency: false
|
||||
})
|
||||
},
|
||||
language: 'en-US',
|
||||
language: 'eng',
|
||||
voice: 'amber',
|
||||
text,
|
||||
renderForCaching: true
|
||||
@@ -754,6 +754,40 @@ test('rimelabs speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('rimelabs speech synth tests mistv2', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.RIMELABS_API_KEY) {
|
||||
t.pass('skipping rimelabs speech synth tests since RIMELABS_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones!';
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'rimelabs',
|
||||
credentials: {
|
||||
api_key: process.env.RIMELABS_API_KEY,
|
||||
model_id: 'mistv2',
|
||||
options: JSON.stringify({
|
||||
speedAlpha: 1.0,
|
||||
reduceLatency: false
|
||||
})
|
||||
},
|
||||
language: 'spa',
|
||||
voice: 'pablo',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized rimelabs mistv2 audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('whisper speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
@@ -853,11 +887,12 @@ test('TTS Cache tests', async(t) => {
|
||||
// save some random tts keys to cache
|
||||
const minRecords = 8;
|
||||
for (const i in Array(minRecords).fill(0)) {
|
||||
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, model: i, text: i,
|
||||
instructions: i}), i);
|
||||
}
|
||||
const count = await getTtsSize();
|
||||
t.ok(count >= minRecords, 'getTtsSize worked.');
|
||||
|
||||
|
||||
const {purgedCount} = await purgeTtsCache();
|
||||
t.ok(purgedCount >= minRecords, `successfully purged at least ${minRecords} tts records from cache`);
|
||||
|
||||
@@ -872,7 +907,7 @@ test('TTS Cache tests', async(t) => {
|
||||
try {
|
||||
// save some random tts keys to cache
|
||||
for (const i in Array(10).fill(0)) {
|
||||
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
|
||||
}
|
||||
// save a specific key to tts cache
|
||||
const opts = {vendor: 'aws', language: 'en-US', voice: 'MALE', engine: 'Engine', text: 'Hello World!'};
|
||||
@@ -910,10 +945,10 @@ test('TTS Cache tests', async(t) => {
|
||||
const account_sid = "12412512_cabc_5aff"
|
||||
const account_sid2 = "22412512_cabc_5aff"
|
||||
for (const i in Array(minRecords).fill(0)) {
|
||||
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
|
||||
}
|
||||
for (const i in Array(minRecords).fill(0)) {
|
||||
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
|
||||
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
|
||||
}
|
||||
const {purgedCount} = await purgeTtsCache({account_sid});
|
||||
t.equal(purgedCount, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid}`);
|
||||
|
||||
Reference in New Issue
Block a user