Compare commits

..
Author SHA1 Message Date
Dave Horton db135ee5ad 0.2.12 2025-06-11 11:08:34 +02:00
Dave Horton 5eac4c2ad8 Merge pull request #114 from vasudevanubrolu/feat/893-azure-ssml
Feat/893 azure ssml
2025-06-11 11:05:48 +02:00
vasudevanubrolu e23a1a6d09 feat/893 azure ssml add namespace check 2025-06-04 12:53:37 +05:30
vasudevanubrolu 3a78300a08 feat/893 azure ssml only change free text 2025-06-04 11:18:11 +05:30
vasudevanubrolu 1d7390e7ae feat/893 azure ssml fix only on condition 2025-05-30 14:55:22 +05:30
vasudevanubrolu 3c0940f657 feat/893 azure ssml lang syntax fix 2025-05-30 14:55:22 +05:30
Dave Horton 6fb6195b16 0.2.11 2025-05-27 10:09:01 -04:00
Dave Horton 2ff8587601 Merge pull request #113 from vasudevanubrolu/feat/893-azure-ssml
Feat/893 azure ssml
2025-05-27 10:08:48 -04:00
vasudevanubrolu 925bd26a70 feat/893 add add lang tag for accent to be picked 2025-05-27 13:41:07 +05:30
vasudevanubrolu ddea485f5f feat/893 azure ssml lang for on prem 2025-05-26 15:13:45 +05:30
vasudevanubrolu 49de25feb8 feat/893 ssml config 2025-05-26 12:10:22 +05:30
vasudevanubrolu 08b55b8d79 feat/893 support default azure ssml config 2025-05-26 12:10:22 +05:30
vasudevanubrolu 59bca302b9 feat/893 azure ssml based on env config 2025-05-26 12:10:22 +05:30
Dave Horton 0d98f73c43 0.2.10 2025-05-13 09:57:01 -04:00
Dave Horton 36670e0080 Merge pull request #111 from vasudevanubrolu/feat/864-playht-onprem
feat/864 playht on prem
2025-05-13 09:50:46 -04:00
vasudevanubrolu 61672f9868 feat/864 playht on prem pr changes 2025-05-13 18:47:26 +05:30
vasudevan-Kore ddeca8eb99 feat/864 playht on prem 2025-05-13 18:47:26 +05:30
Dave Horton 6e8271b2f6 0.2.9 2025-05-13 07:47:03 -04:00
Dave Horton 74a4938eb6 Merge pull request #112 from jambonz/feat/whisper_instructions
support openai whisper instructions
2025-05-13 07:46:34 -04:00
Quan HL cb50f603cd wip 2025-05-13 18:02:11 +07:00
Quan HL 9a524f00bc wip 2025-05-13 17:58:57 +07:00
Quan HL 545df0b770 wip 2025-05-13 17:55:41 +07:00
Quan HL 9409405769 support openai whisper instructions 2025-05-13 15:23:59 +07:00
Dave Horton 7dc3bbdb01 0.2.8 2025-05-08 09:31:40 -04:00
Dave Horton f8f9de2645 Merge pull request #110 from jambonz/feat/rimlabs_arcana
support rimelabs arcana
2025-05-06 09:22:35 -04:00
Quan HL 467d7ede26 support rimelabs arcana 2025-05-06 16:16:18 +07:00
Dave Horton fc211ab2e7 0.2.7 2025-04-28 19:31:59 -04:00
Dave Horton 536f8aab31 Merge pull request #109 from jambonz/feat/riva_tts
support riva tts stream
2025-04-28 19:31:30 -04:00
Hoan Luu Huu 69b3fdffbe Merge branch 'main' into feat/riva_tts 2025-04-28 08:47:44 +07:00
Dave Horton 53fe72d89e 0.2.6 2025-04-23 07:12:54 -04:00
Dave Horton e79c15c5da update version 2025-04-23 07:12:25 -04:00
Hoan Luu Huu a4a427e174 Merge branch 'main' into feat/riva_tts 2025-04-23 18:11:19 +07:00
Dave Horton b697bc5268 Merge pull request #108 from jambonz/feat/ell_tts_new_params
elevenlabs tts speed and pronunciation_dictionary_locators
2025-04-23 07:08:49 -04:00
Quan HL d35d7f0aec support riva tts stream 2025-04-23 17:54:21 +07:00
Quan HL ed1c564fa2 wip 2025-04-04 16:15:45 +07:00
Quan HL 7189d471c1 elevenlabs tts speed and pronunciation_dictionary_locators 2025-04-04 15:44:29 +07:00
Dave Horton 04080cc5ec 0.2.4 2025-03-19 21:48:31 -04:00
Dave Horton 3e5ab4af27 Merge pull request #107 from jambonz/update-deps
update undici
2025-03-19 21:48:02 -04:00
Dave Horton f06dddd2f7 update undici 2025-03-19 21:46:21 -04:00
Dave Horton 7b4a71f55d 0.2.3 2025-02-07 07:20:12 -05:00
Dave Horton d552b65618 Merge pull request #106 from jambonz/feat/rimelabs_voices
rimelabs support multiple model and languages
2025-02-07 07:19:47 -05:00
8 changed files with 460 additions and 738 deletions
+2 -1
View File
@@ -25,7 +25,7 @@ function getExtensionAndSampleRate(path) {
}
async function addFileToCache(client, logger, path,
{account_sid, vendor, language, voice, deploymentId, engine, model, text}) {
{account_sid, vendor, language, voice, deploymentId, engine, model, text, instructions}) {
let key;
logger = logger || noopLogger;
@@ -38,6 +38,7 @@ async function addFileToCache(client, logger, path,
engine,
model,
text,
instructions
});
const [extension, sampleRate] = getExtensionAndSampleRate(path);
const audioBuffer = await fs.readFile(path);
+3 -1
View File
@@ -2,6 +2,7 @@ const JAMBONES_TTS_TRIM_SILENCE = process.env.JAMBONES_TTS_TRIM_SILENCE;
const JAMBONES_DISABLE_TTS_STREAMING = process.env.JAMBONES_DISABLE_TTS_STREAMING;
const JAMBONES_DISABLE_AZURE_TTS_STREAMING = process.env.JAMBONES_DISABLE_AZURE_TTS_STREAMING;
const JAMBONES_EAGERLY_PRE_CACHE_AUDIO = process.env.JAMBONES_EAGERLY_PRE_CACHE_AUDIO;
const JAMBONES_AZURE_ENABLE_SSML = process.env.JAMBONES_AZURE_ENABLE_SSML;
const JAMBONES_HTTP_PROXY_IP = process.env.JAMBONES_HTTP_PROXY_IP;
const JAMBONES_HTTP_PROXY_PORT = process.env.JAMBONES_HTTP_PROXY_PORT;
@@ -21,5 +22,6 @@ module.exports = {
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_EAGERLY_PRE_CACHE_AUDIO,
TMP_FOLDER,
HTTP_TIMEOUT
HTTP_TIMEOUT,
JAMBONES_AZURE_ENABLE_SSML
};
+2 -1
View File
@@ -12,7 +12,7 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
* @returns {object} result - {error, purgedCount}
*/
async function purgeTtsCache(client, logger, {all, account_sid, vendor,
language, voice, deploymentId, engine, model, text} = {all: true}) {
language, voice, deploymentId, engine, model, text, instructions} = {all: true}) {
logger = logger || noopLogger;
let purgedCount = 0, error;
@@ -35,6 +35,7 @@ async function purgeTtsCache(client, logger, {all, account_sid, vendor,
engine,
model,
text,
instructions
});
purgedCount = await client.del(key);
if (purgedCount === 0) error = 'Specified item not found';
+52 -14
View File
@@ -47,6 +47,7 @@ const {
JAMBONES_HTTP_PROXY_PORT,
JAMBONES_TTS_CACHE_DURATION_MINS,
JAMBONES_TTS_TRIM_SILENCE,
JAMBONES_AZURE_ENABLE_SSML
} = require('./config');
const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
const OpenAI = require('openai');
@@ -89,7 +90,7 @@ const trimTrailingSilence = (buffer) => {
*/
async function synthAudio(client, createHash, retrieveHash, logger, stats, { account_sid,
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
disableTtsCache, renderForCaching = false, disableTtsStreaming, options
disableTtsCache, renderForCaching = false, disableTtsStreaming, options, instructions
}) {
let audioData;
let servedFromCache = false;
@@ -171,7 +172,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
engine,
// model or model_id is used to identify the tts cache.
model: model || credentials.model_id,
text
text,
instructions
});
debug(`synth key is ${key}`);
@@ -215,7 +217,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, text});
break;
case 'nvidia':
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text});
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text,
renderForCaching, disableTtsStreaming});
break;
case 'ibm':
audioData = await synthIbm(logger, {credentials, stats, voice, text});
@@ -241,7 +244,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
break;
case 'whisper':
audioData = await synthWhisper(logger, {
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
credentials, stats, voice, text, instructions, renderForCaching, disableTtsStreaming});
break;
case 'verbio':
audioData = await synthVerbio(client, logger, {
@@ -464,7 +467,6 @@ async function _synthOnPremMicrosoft(logger, {
}) {
const {use_custom_tts, custom_tts_endpoint_url, api_key} = credentials;
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
@@ -480,6 +482,10 @@ async function _synthOnPremMicrosoft(logger, {
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
else if (JAMBONES_AZURE_ENABLE_SSML && !content.startsWith('<speak')) {
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}"><lang xml:lang="${language}">${text}</lang></voice></speak>`;
}
try {
const trimSilence = JAMBONES_TTS_TRIM_SILENCE;
@@ -516,19 +522,23 @@ const synthMicrosoft = async(logger, {
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
else if (JAMBONES_AZURE_ENABLE_SSML && !content.startsWith('<speak')) {
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}"><lang xml:lang="${language}">${text}</lang></voice></speak>`;
}
if (!JAMBONES_DISABLE_TTS_STREAMING && !JAMBONES_DISABLE_AZURE_TTS_STREAMING &&
!renderForCaching && !disableTtsStreaming) {
let params = '';
@@ -708,8 +718,24 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
});
};
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
const synthNvidia = async(client, logger, {
credentials, stats, language, voice, model, text, renderForCaching, disableTtsStreaming
}) => {
const {riva_server_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{riva_server_uri=${riva_server_uri}`;
params += `,voice=${voice}`;
params += `,language=${language}`;
params += ',write_cache_file=1';
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
let rivaClient, request;
const sampleRate = 8000;
try {
@@ -789,9 +815,13 @@ const synthElevenlabs = async(logger, {
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
if (opts.voice_settings?.speed !== null && opts.voice_settings?.speed !== undefined)
params += `,speed=${opts.voice_settings.speed}`;
if (opts.voice_settings?.use_speaker_boost === false) params += ',use_speaker_boost=false';
if (opts.previous_text) params += `,previous_text=${opts.previous_text}`;
if (opts.next_text) params += `,next_text=${opts.next_text}`;
if (opts.pronunciation_dictionary_locators && Array.isArray(opts.pronunciation_dictionary_locators))
params += `,pronunciation_dictionary_locators=${JSON.stringify(opts.pronunciation_dictionary_locators)}`;
params += '}';
return {
@@ -833,11 +863,10 @@ const synthElevenlabs = async(logger, {
const synthPlayHT = async(client, logger, {
credentials, options, stats, voice, language, text, renderForCaching, disableTtsStreaming
}) => {
const {api_key, user_id, voice_engine, options: credOpts} = credentials;
const {api_key, user_id, voice_engine, playht_tts_uri, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
let synthesizeUrl = 'https://api.play.ht/api/v2/tts/stream';
let synthesizeUrl = playht_tts_uri ? `${playht_tts_uri}/api/v2/tts/stream` : 'https://api.play.ht/api/v2/tts/stream';
// If model is play3.0, the synthesizeUrl is got from authentication endpoint
if (voice_engine === 'Play3.0') {
try {
@@ -942,6 +971,11 @@ const synthRimelabs = async(logger, {
params += ',write_cache_file=1';
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
if (opts.reduceLatency) params += `,reduce_latency=${opts.reduceLatency}`;
// Arcana model parameters
if (opts.temperature) params += `,temperature=${opts.temperature}`;
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
if (opts.top_p) params += `,top_p=${opts.top_p}`;
if (opts.max_tokens) params += `,max_tokens=${opts.max_tokens}`;
params += '}';
return {
@@ -1022,7 +1056,8 @@ const synthVerbio = async(client, logger, {credentials, stats, voice, text, rend
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
const synthWhisper = async(logger, {credentials, stats, voice, text, instructions,
renderForCaching, disableTtsStreaming}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
@@ -1033,6 +1068,8 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (speed) params += `,speed=${speed}`;
// comma is used to separated parameters in freeswitch tts module
if (instructions) params += `,instructions=${instructions.replace(/\n/g, ' ').replace(/,/g, ';')}`;
params += '}';
return {
@@ -1052,6 +1089,7 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
model: model_id,
voice,
input: text,
...(instructions && {instructions}),
response_format: 'mp3'
});
return {
+3 -2
View File
@@ -23,10 +23,11 @@ function makeSynthKey({
voice,
engine = '',
model = '',
text
text,
instructions = '',
}) {
const hash = crypto.createHash('sha1');
hash.update(`${language}:${vendor}:${voice}:${engine}:${model}:${text}`);
hash.update(`${language}:${vendor}:${voice}:${engine}:${model}:${text}:${instructions}`);
const hexHashKey = hash.digest('hex');
const accountKey = account_sid ? `:${account_sid}` : '';
const key = `tts${accountKey}:${hexHashKey}`;
+389 -712
View File
File diff suppressed because it is too large Load Diff
+4 -3
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.2.2",
"version": "0.2.12",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -26,6 +26,7 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"23": "^0.0.0",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@cartesia/cartesia-js": "^2.1.0",
@@ -38,8 +39,8 @@
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.25.0",
"undici": "^6.4.0"
"openai": "^4.98.0",
"undici": "^7.5.0"
},
"devDependencies": {
"config": "^3.3.11",
+5 -4
View File
@@ -887,7 +887,8 @@ test('TTS Cache tests', async(t) => {
// save some random tts keys to cache
const minRecords = 8;
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, model: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, model: i, text: i,
instructions: i}), i);
}
const count = await getTtsSize();
t.ok(count >= minRecords, 'getTtsSize worked.');
@@ -906,7 +907,7 @@ test('TTS Cache tests', async(t) => {
try {
// save some random tts keys to cache
for (const i in Array(10).fill(0)) {
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
}
// save a specific key to tts cache
const opts = {vendor: 'aws', language: 'en-US', voice: 'MALE', engine: 'Engine', text: 'Hello World!'};
@@ -944,10 +945,10 @@ test('TTS Cache tests', async(t) => {
const account_sid = "12412512_cabc_5aff"
const account_sid2 = "22412512_cabc_5aff"
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({account_sid, vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
}
for (const i in Array(minRecords).fill(0)) {
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i}), i);
await client.set(makeSynthKey({account_sid: account_sid2, vendor: i, language: i, voice: i, engine: i, text: i, instructions: i}), i);
}
const {purgedCount} = await purgeTtsCache({account_sid});
t.equal(purgedCount, minRecords, `successfully purged at least ${minRecords} tts records from cache for account_sid:${account_sid}`);