Compare commits

...
Author SHA1 Message Date
Dave Horton 7dc3bbdb01 0.2.8 2025-05-08 09:31:40 -04:00
Dave Horton f8f9de2645 Merge pull request #110 from jambonz/feat/rimlabs_arcana
support rimelabs arcana
2025-05-06 09:22:35 -04:00
Quan HL 467d7ede26 support rimelabs arcana 2025-05-06 16:16:18 +07:00
Dave Horton fc211ab2e7 0.2.7 2025-04-28 19:31:59 -04:00
Dave Horton 536f8aab31 Merge pull request #109 from jambonz/feat/riva_tts
support riva tts stream
2025-04-28 19:31:30 -04:00
Hoan Luu Huu 69b3fdffbe Merge branch 'main' into feat/riva_tts 2025-04-28 08:47:44 +07:00
Dave Horton 53fe72d89e 0.2.6 2025-04-23 07:12:54 -04:00
Dave Horton e79c15c5da update version 2025-04-23 07:12:25 -04:00
Hoan Luu Huu a4a427e174 Merge branch 'main' into feat/riva_tts 2025-04-23 18:11:19 +07:00
Dave Horton b697bc5268 Merge pull request #108 from jambonz/feat/ell_tts_new_params
elevenlabs tts speed and pronunciation_dictionary_locators
2025-04-23 07:08:49 -04:00
Quan HL d35d7f0aec support riva tts stream 2025-04-23 17:54:21 +07:00
Quan HL ed1c564fa2 wip 2025-04-04 16:15:45 +07:00
Quan HL 7189d471c1 elevenlabs tts speed and pronunciation_dictionary_locators 2025-04-04 15:44:29 +07:00
Dave Horton 04080cc5ec 0.2.4 2025-03-19 21:48:31 -04:00
Dave Horton 3e5ab4af27 Merge pull request #107 from jambonz/update-deps
update undici
2025-03-19 21:48:02 -04:00
Dave Horton f06dddd2f7 update undici 2025-03-19 21:46:21 -04:00
Dave Horton 7b4a71f55d 0.2.3 2025-02-07 07:20:12 -05:00
Dave Horton d552b65618 Merge pull request #106 from jambonz/feat/rimelabs_voices
rimelabs support multiple model and languages
2025-02-07 07:19:47 -05:00
Quan HL f701b50244 rimelabs support multiple model and languages 2025-02-07 14:20:52 +07:00
Dave Horton 199e502fbe 0.2.2 2025-02-03 08:14:15 -05:00
Dave Horton 6769779189 Merge pull request #105 from jambonz/fix/gh_1059
tts key should include model
2025-02-03 08:13:54 -05:00
Quan HL 4b43c3986c wip 2025-02-02 19:03:09 +07:00
Quan HL e1292772e6 tts key should include model 2025-02-02 18:54:16 +07:00
7 changed files with 408 additions and 582 deletions
+2 -1
View File
@@ -25,7 +25,7 @@ function getExtensionAndSampleRate(path) {
}
async function addFileToCache(client, logger, path,
{account_sid, vendor, language, voice, deploymentId, engine, text}) {
{account_sid, vendor, language, voice, deploymentId, engine, model, text}) {
let key;
logger = logger || noopLogger;
@@ -36,6 +36,7 @@ async function addFileToCache(client, logger, path,
language: language || '',
voice: voice || deploymentId,
engine,
model,
text,
});
const [extension, sampleRate] = getExtensionAndSampleRate(path);
+2 -1
View File
@@ -12,7 +12,7 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
* @returns {object} result - {error, purgedCount}
*/
async function purgeTtsCache(client, logger, {all, account_sid, vendor,
language, voice, deploymentId, engine, text} = {all: true}) {
language, voice, deploymentId, engine, model, text} = {all: true}) {
logger = logger || noopLogger;
let purgedCount = 0, error;
@@ -33,6 +33,7 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
language: language || '',
voice: voice || deploymentId,
engine,
model,
text,
});
purgedCount = await client.del(key);
+33 -3
View File
@@ -169,6 +169,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
language: language || '',
voice: voice || deploymentId,
engine,
// model or model_id is used to identify the tts cache.
model: model || credentials.model_id,
text
});
@@ -213,7 +215,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, text});
break;
case 'nvidia':
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text});
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text,
renderForCaching, disableTtsStreaming});
break;
case 'ibm':
audioData = await synthIbm(logger, {credentials, stats, voice, text});
@@ -706,8 +709,24 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
});
};
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
const synthNvidia = async(client, logger, {
credentials, stats, language, voice, model, text, renderForCaching, disableTtsStreaming
}) => {
const {riva_server_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{riva_server_uri=${riva_server_uri}`;
params += `,voice=${voice}`;
params += `,language=${language}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
let rivaClient, request;
const sampleRate = 8000;
try {
@@ -787,9 +806,13 @@ const synthElevenlabs = async(logger, {
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
if (opts.voice_settings?.speed !== null && opts.voice_settings?.speed !== undefined)
params += `,speed=${opts.voice_settings.speed}`;
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
if (opts.previous_text) params += `,previous_text=${opts.previous_text}`;
if (opts.next_text) params += `,next_text=${opts.next_text}`;
if (opts.pronunciation_dictionary_locators && Array.isArray(opts.pronunciation_dictionary_locators))
params += `,pronunciation_dictionary_locators=${JSON.stringify(opts.pronunciation_dictionary_locators)}`;
params += '}';
return {
@@ -924,7 +947,7 @@ const synthPlayHT = async(client, logger, {
};
const synthRimelabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
@@ -935,10 +958,16 @@ const synthRimelabs = async(logger, {
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += ',vendor=rimelabs';
params += `,language=${language}`;
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
if (opts.reduceLatency) params += `,reduce_latency=${opts.reduceLatency}`;
// Arcana model parameters
if (opts.temperature) params += `,temperature=${opts.temperature}`;
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
if (opts.top_p) params += `,top_p=${opts.top_p}`;
if (opts.max_tokens) params += `,max_tokens=${opts.max_tokens}`;
params += '}';
return {
@@ -960,6 +989,7 @@ const synthRimelabs = async(logger, {
text,
modelId: model_id,
samplingRate: sampleRate,
lang: language,
...opts
});
return {
+9 -2
View File
@@ -17,9 +17,16 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
//const nuanceClientMap = new Map();
function makeSynthKey({
account_sid = '', vendor, language, voice, engine = '', text}) {
account_sid = '',
vendor,
language,
voice,
engine = '',
model = '',
text
}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${text}`);
hash.update(`${language}:${vendor}:${voice}:${engine}:${model}:${text}`);
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const key = `tts${accountKey}:${hexHashKey}`;
+320 -567
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.2.1",
"version": "0.2.8",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -39,7 +39,7 @@
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
"undici": "^7.5.0"
},
"devDependencies": {
"config": "^3.3.11",
+40 -6
View File
@@ -5,7 +5,7 @@ const fs = require('fs');
const {makeSynthKey} = require('../lib/utils');
const logger = require('pino')();
const bent = require('bent');
const getJSON = bent('json')
const getJSON = bent('json');
process.on('unhandledRejection', (reason, p) => {
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
@@ -720,7 +720,7 @@ test('Cartesia speech synth tests', async(t) => {
client.quit();
});
test('rimelabs speech synth tests', async(t) => {
test('rimelabs speech synth tests mist', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -730,7 +730,7 @@ test('rimelabs speech synth tests', async(t) => {
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
@@ -740,7 +740,7 @@ test('rimelabs speech synth tests', async(t) => {
reduceLatency: false
})
},
language: 'en-US',
language: 'eng',
voice: 'amber',
text,
renderForCaching: true
@@ -754,6 +754,40 @@ test('rimelabs speech synth tests', async(t) => {
client.quit();
});
test('rimelabs speech synth tests mistv2', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.RIMELABS_API_KEY) {
t.pass('skipping rimelabs speech synth tests since RIMELABS_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
const opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
model_id: 'mistv2',
options: JSON.stringify({
speedAlpha: 1.0,
reduceLatency: false
})
},
language: 'spa',
voice: 'pablo',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized rimelabs mistv2 audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('whisper speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -853,11 +887,11 @@ test('TTS Cache tests', async(t) => {
// save some random tts keys to cache
const minRecords = 8;
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, model: i, text: i}), i);
}
const count = await getTtsSize();
t.ok(count >= minRecords, 'getTtsSize worked.');
const {purgedCount} = await purgeTtsCache();
t.ok(purgedCount >= minRecords, `successfully purged at least ${minRecords} tts records from cache`);