Compare commits

...
14 Commits
Author SHA1 Message Date
Hoan Luu Huu 69b3fdffbe Merge branch 'main' into feat/riva_tts 2025-04-28 08:47:44 +07:00
Dave Horton 53fe72d89e 0.2.6 2025-04-23 07:12:54 -04:00
Dave Horton e79c15c5da update version 2025-04-23 07:12:25 -04:00
Hoan Luu Huu a4a427e174 Merge branch 'main' into feat/riva_tts 2025-04-23 18:11:19 +07:00
Dave Horton b697bc5268 Merge pull request #108 from jambonz/feat/ell_tts_new_params
elevenlabs tts speed and pronunciation_dictionary_locators
2025-04-23 07:08:49 -04:00
Quan HL d35d7f0aec support riva tts stream 2025-04-23 17:54:21 +07:00
Quan HL ed1c564fa2 wip 2025-04-04 16:15:45 +07:00
Quan HL 7189d471c1 elevenlabs tts speed and pronunciation_dictionary_locators 2025-04-04 15:44:29 +07:00
Dave Horton 04080cc5ec 0.2.4 2025-03-19 21:48:31 -04:00
Dave Horton 3e5ab4af27 Merge pull request #107 from jambonz/update-deps
update undici
2025-03-19 21:48:02 -04:00
Dave Horton f06dddd2f7 update undici 2025-03-19 21:46:21 -04:00
Dave Horton 7b4a71f55d 0.2.3 2025-02-07 07:20:12 -05:00
Dave Horton d552b65618 Merge pull request #106 from jambonz/feat/rimelabs_voices
rimelabs support multiple model and languages
2025-02-07 07:19:47 -05:00
Quan HL f701b50244 rimelabs support multiple model and languages 2025-02-07 14:20:52 +07:00
4 changed files with 385 additions and 575 deletions
+26 -3
View File
@@ -215,7 +215,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, text});
break;
case 'nvidia':
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text});
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text,
renderForCaching, disableTtsStreaming});
break;
case 'ibm':
audioData = await synthIbm(logger, {credentials, stats, voice, text});
@@ -708,8 +709,24 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
});
};
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
const synthNvidia = async(client, logger, {
credentials, stats, language, voice, model, text, renderForCaching, disableTtsStreaming
}) => {
const {riva_server_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{riva_server_uri=${riva_server_uri}`;
params += `,voice=${voice}`;
params += `,language=${language}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
let rivaClient, request;
const sampleRate = 8000;
try {
@@ -789,9 +806,13 @@ const synthElevenlabs = async(logger, {
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
if (opts.voice_settings?.speed !== null && opts.voice_settings?.speed !== undefined)
params += `,speed=${opts.voice_settings.speed}`;
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
if (opts.previous_text) params += `,previous_text=${opts.previous_text}`;
if (opts.next_text) params += `,next_text=${opts.next_text}`;
if (opts.pronunciation_dictionary_locators && Array.isArray(opts.pronunciation_dictionary_locators))
params += `,pronunciation_dictionary_locators=${JSON.stringify(opts.pronunciation_dictionary_locators)}`;
params += '}';
return {
@@ -926,7 +947,7 @@ const synthPlayHT = async(client, logger, {
};
const synthRimelabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
@@ -937,6 +958,7 @@ const synthRimelabs = async(logger, {
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += ',vendor=rimelabs';
params += `,language=${language}`;
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
@@ -962,6 +984,7 @@ const synthRimelabs = async(logger, {
text,
modelId: model_id,
samplingRate: sampleRate,
lang: language,
...opts
});
return {
+320 -567
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.2.2",
"version": "0.2.6",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -39,7 +39,7 @@
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
"undici": "^7.5.0"
},
"devDependencies": {
"config": "^3.3.11",
+37 -3
View File
@@ -720,7 +720,7 @@ test('Cartesia speech synth tests', async(t) => {
client.quit();
});
test('rimelabs speech synth tests', async(t) => {
test('rimelabs speech synth tests mist', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -730,7 +730,7 @@ test('rimelabs speech synth tests', async(t) => {
}
const text = 'Hi there and welcome to jambones!';
try {
let opts = await synthAudio(stats, {
const opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
@@ -740,7 +740,7 @@ test('rimelabs speech synth tests', async(t) => {
reduceLatency: false
})
},
language: 'en-US',
language: 'eng',
voice: 'amber',
text,
renderForCaching: true
@@ -754,6 +754,40 @@ test('rimelabs speech synth tests', async(t) => {
client.quit();
});
test('rimelabs speech synth tests mistv2', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.RIMELABS_API_KEY) {
t.pass('skipping rimelabs speech synth tests since RIMELABS_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambones!';
try {
const opts = await synthAudio(stats, {
vendor: 'rimelabs',
credentials: {
api_key: process.env.RIMELABS_API_KEY,
model_id: 'mistv2',
options: JSON.stringify({
speedAlpha: 1.0,
reduceLatency: false
})
},
language: 'spa',
voice: 'pablo',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized rimelabs mistv2 audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('whisper speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);