Compare commits

...
70 Commits
Author SHA1 Message Date
Dave Horton 18c658d20c 1.0.3 2026-06-03 07:27:44 -04:00
Dave Horton 062608cf13 update lock file 2026-06-03 07:27:28 -04:00
Hoan Luu Huu a2ea94c14a support rimelabs coda (#142) 2026-06-03 07:24:13 -04:00
Dave Horton 04bb85ef81 1.0.1 2026-03-25 10:08:57 -04:00
Dave Horton c123f19898 remove ibm speech since it is not used (to my knowledge) and has dependencies with vulnerabilities (#141) 2026-03-25 10:08:30 -04:00
Dave Horton 305695d068 0.2.30 2026-01-22 08:03:44 -05:00
Dave Horton 1477752d40 update dep 2026-01-22 08:03:24 -05:00
Dave Horton fbedfe947f vuln 2026-01-22 07:58:06 -05:00
Dave Horton ab88facd52 Merge pull request #139 from jambonz/feat/google_gemini_tts
google tts support api_mode
2026-01-22 07:52:04 -05:00
Hoan HL a2be64da89 google tts support api_mode 2026-01-22 16:36:57 +07:00
Dave Horton c5e0f256e6 0.2.28 2026-01-17 21:39:59 -05:00
Dave Horton d0378751ba Merge pull request #135 from jambonz/feat/gemini_tts
support gemini tts
2026-01-17 21:39:28 -05:00
Hoan HL a7e391fcb2 wip 2026-01-18 08:45:14 +07:00
Hoan HL 10095de25d wip 2026-01-17 16:16:11 +07:00
Hoan HL 89007ba7cc wip 2026-01-17 15:13:47 +07:00
Hoan HL 29edccd5bf wip 2026-01-17 15:10:17 +07:00
Hoan HL ca9538030f wip 2026-01-14 17:37:38 +07:00
Hoan HL 417d58080e wip 2026-01-14 17:34:07 +07:00
Hoan HL 62fec6f5e4 wip 2026-01-14 16:00:43 +07:00
Hoan HL 490d23a703 wip 2026-01-14 15:46:29 +07:00
Hoan HL b91e6cc145 wip 2026-01-14 13:58:38 +07:00
Hoan HL 0e33358254 wip 2026-01-14 13:51:32 +07:00
Hoan HL 6b2b35acfb wip 2026-01-12 18:26:47 +07:00
Hoan HL ded60cb7aa wip 2026-01-12 17:18:11 +07:00
Hoan HL 460ca70ea7 wip 2026-01-12 12:55:43 +07:00
Hoan HL 0fbbbb8053 add testcases 2026-01-12 09:21:41 +07:00
Hoan HL 0ea7082da2 support gemini tts 2026-01-11 07:30:18 +07:00
Dave Horton 5f7e7458bb 0.2.27 2025-11-17 07:26:45 -05:00
Dave Horton f6714fb9e1 Merge pull request #134 from jambonz/fix/1628
fixed cartesia collect audio from stream
2025-11-13 07:15:51 -05:00
Hoan HL c04ef29f7c fixed cartesia collect audio from stream 2025-11-13 14:10:00 +07:00
Dave Horton 8154944252 0.2.26 2025-10-30 07:08:47 -04:00
Dave Horton 16fe8dce01 0.2.25 2025-10-30 07:08:10 -04:00
Dave Horton 11a955500d Merge pull request #132 from jambonz/feat/sonic_3
cartesia support volume for sonic3
2025-10-30 07:07:15 -04:00
Hoan HL c122129b55 wip 2025-10-30 12:29:09 +07:00
Hoan HL 41b26b966b wip 2025-10-30 12:08:38 +07:00
Hoan HL aa09c15b20 wip 2025-10-30 06:46:57 +07:00
Hoan HL 32d5f12638 cartesia support volume for sonic3 2025-10-30 05:57:38 +07:00
Dave Horton 4e336822a0 Merge pull request #131 from jambonz/feat/gh_fs_1384
support elevenlabs api_uri
2025-10-08 13:45:02 -04:00
Hoan HL b898d794b0 wip 2025-10-08 15:05:04 +07:00
Hoan HL fb754ca101 support elevenlabs api_uri 2025-10-08 10:26:17 +07:00
Dave Horton 8d8195be9a 0.2.24 2025-10-03 08:54:14 -04:00
Dave Horton eb4e1a773f Merge pull request #130 from jambonz/feat/disableTtsCache
set write_cache_file = 0 when disableTtsCache
2025-10-03 02:21:57 -04:00
Hoan HL 6fecb8755d set write_cache_file = 0 when disableTtsCache 2025-10-03 11:09:30 +07:00
Dave Horton fdb56cbc77 Merge pull request #128 from jambonz/feat/custom_tts_stream
support custom tts Stream
2025-09-11 09:24:18 -04:00
Quan HL 5328a60de8 support custom tts Stream 2025-09-11 09:18:42 -04:00
Dave Horton ea1ab301d5 0.2.23 2025-09-10 22:53:45 -04:00
Dave Horton 0a4994c5a3 Merge pull request #129 from jambonz/fix/fd_1372
remove optimize_streaming_latency as default option for elevenlabs
2025-09-10 22:53:27 -04:00
Quan HL cb71460189 remove optimize_streaming_latency as default option for elevenlabs 2025-09-11 09:44:53 +07:00
Dave Horton f08efa49b1 0.2.22 2025-08-20 18:03:08 -04:00
Dave Horton 80eda37ef6 Merge pull request #126 from jambonz/feat/revamp-playback-id
use synth key as playback id
2025-08-20 18:02:35 -04:00
Dave Horton 7768e58b19 use synth key as playback id 2025-08-20 17:11:45 -04:00
Dave Horton 6efc99ad83 0.2.21 2025-08-20 10:59:40 -04:00
Dave Horton 127d01a8c1 add playback_id setting to additional tts vendors 2025-08-20 10:58:28 -04:00
Dave Horton 1900b26d8b 0.2.20 2025-08-20 10:01:01 -04:00
Dave Horton aa25a6f6d9 Merge pull request #125 from jambonz/feat/add-playback-id
add playback_id to say metadata
2025-08-20 10:00:37 -04:00
Dave Horton 8dfe17b600 add playback_id to say metadata 2025-08-20 08:58:46 -04:00
Dave Horton 5463e9f56e 0.2.19 2025-08-17 09:35:07 -04:00
Dave Horton 2eb36af650 Merge pull request #124 from jambonz/fix/aws_tts
support mod_aws_tts with engine parameter
2025-08-17 09:23:58 -04:00
Quan HL b186cbc4f2 support mod_aws_tts with engine parameter 2025-08-17 06:44:40 +07:00
Dave Horton b1b9c182a9 0.2.18 2025-08-14 08:17:02 -04:00
Dave Horton 15c77626fc Merge pull request #123 from jambonz/feat/mod_aws_tts
support mod_aws_tts
2025-08-14 08:16:32 -04:00
Quan HL 76126ec0b3 wip 2025-08-14 18:49:29 +07:00
Quan HL 55431ee511 wip 2025-08-14 17:19:46 +07:00
Quan HL 1c76f74b4a wip 2025-08-14 17:13:05 +07:00
Quan HL 62ad8abb8e wip 2025-08-14 16:49:47 +07:00
Quan HL 92734aaedb wip 2025-08-14 16:38:36 +07:00
Quan HL fe4ccfe7d7 support mod_aws_tts 2025-08-14 16:34:40 +07:00
Dave Horton ab05976032 0.2.17 2025-08-13 07:41:24 -04:00
Dave Horton babd8abc51 Merge pull request #122 from jambonz/feat/resemble_tts_01
support resemble tts
2025-08-13 07:33:12 -04:00
Quan HL 91f85e16b2 support resemble tts 2025-08-13 14:00:24 +07:00
14 changed files with 3541 additions and 3218 deletions
+2 -9
View File
@@ -11,13 +11,8 @@ jobs:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: lts/*
node-version: '20'
- run: npm install
- name: Install Docker Compose
run: |
sudo curl -L "https://github.com/docker/compose/releases/download/1.29.2/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
sudo chmod +x /usr/local/bin/docker-compose
docker-compose --version
- run: npm run jslint
- run: sudo apt update && sudo apt install -y squid
- run: sudo cp test/squid.conf /etc/squid/squid.conf
@@ -28,9 +23,7 @@ jobs:
AWS_REGION: ${{ secrets.AWS_REGION }}
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
GCP_JSON_KEY: ${{ secrets.GCP_JSON_KEY }}
IBM_API_KEY: ${{ secrets.IBM_API_KEY }}
IBM_TTS_API_KEY: ${{ secrets.IBM_TTS_API_KEY }}
IBM_TTS_REGION: ${{ secrets.IBM_TTS_REGION }}
MICROSOFT_API_KEY: ${{ secrets.MICROSOFT_API_KEY }}
MICROSOFT_REGION: ${{ secrets.MICROSOFT_REGION }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
+1 -1
View File
@@ -16,7 +16,7 @@ module.exports = (opts, logger) => {
synthAudio: require('./lib/synth-audio').bind(null, client, createHash, retrieveHash, logger),
getVerbioAccessToken: require('./lib/get-verbio-token').bind(null, client, logger),
getNuanceAccessToken: require('./lib/get-nuance-access-token').bind(null, client, logger),
getIbmAccessToken: require('./lib/get-ibm-access-token').bind(null, client, logger),
getAwsAuthToken: require('./lib/get-aws-sts-token').bind(null, logger, createHash, retrieveHash),
getTtsVoices: require('./lib/get-tts-voices').bind(null, client, createHash, retrieveHash, logger),
};
-48
View File
@@ -1,48 +0,0 @@
const formurlencoded = require('form-urlencoded');
const {Pool} = require('undici');
const pool = new Pool('https://iam.cloud.ibm.com');
const {makeIbmKey, noopLogger} = require('./utils');
const { HTTP_TIMEOUT } = require('./config');
const debug = require('debug')('jambonz:realtimedb-helpers');
async function getIbmAccessToken(client, logger, apiKey) {
logger = logger || noopLogger;
try {
const key = makeIbmKey(apiKey);
const access_token = await client.get(key);
if (access_token) return {access_token, servedFromCache: true};
/* access token not found in cache, so fetch it from Ibm */
const payload = {
grant_type: 'urn:ibm:params:oauth:grant-type:apikey',
apikey: apiKey
};
const {statusCode, headers, body} = await pool.request({
path: '/identity/token',
method: 'POST',
headers: {
'Content-Type': 'application/x-www-form-urlencoded'
},
body: formurlencoded(payload),
timeout: HTTP_TIMEOUT,
followRedirects: false
});
if (200 !== statusCode) {
const json = await body.json();
logger.debug({statusCode, headers, body: json}, 'error fetching access token from Ibm');
const err = new Error();
err.statusCode = statusCode;
throw err;
}
const json = await body.json();
await client.set(key, json.access_token, 'EX', json.expires_in - 30);
return {...json, servedFromCache: false};
} catch (err) {
debug(err, 'getIbmAccessToken: Error retrieving Ibm access token');
logger.error(err, 'getIbmAccessToken: Error retrieving Ibm access token for client_id ${clientId}');
throw err;
}
}
module.exports = getIbmAccessToken;
+1 -20
View File
@@ -3,8 +3,6 @@ const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils')
const getNuanceAccessToken = require('./get-nuance-access-token');
const getVerbioAccessToken = require('./get-verbio-token');
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const ttsGoogle = require('@google-cloud/text-to-speech');
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
const getAwsAuthToken = require('./get-aws-sts-token');
@@ -12,21 +10,6 @@ const {Pool} = require('undici');
const { HTTP_TIMEOUT } = require('./config');
const verbioVoicePool = new Pool('https://us.rest.speechcenter.verbio.com');
const getIbmVoices = async(client, logger, credentials) => {
const {tts_region, tts_api_key} = credentials;
console.log(`region: ${tts_region}, api_key: ${tts_api_key}`);
const textToSpeech = new TextToSpeechV1({
authenticator: new IamAuthenticator({
apikey: tts_api_key,
}),
serviceUrl: `https://api.${tts_region}.text-to-speech.watson.cloud.ibm.com`
});
const voices = await textToSpeech.listVoices();
return voices;
};
const getNuanceVoices = async(client, logger, credentials) => {
const {client_id: clientId, secret: secret, nuance_tts_uri} = credentials;
@@ -165,14 +148,12 @@ const getVerbioVoices = async(client, logger, credentials) => {
async function getTtsVoices(client, createHash, retrieveHash, logger, {vendor, credentials}) {
logger = logger || noopLogger;
assert.ok(['nuance', 'ibm', 'google', 'aws', 'polly', 'verbio'].includes(vendor),
assert.ok(['nuance', 'google', 'aws', 'polly', 'verbio'].includes(vendor),
`getTtsVoices not supported for vendor ${vendor}`);
switch (vendor) {
case 'nuance':
return getNuanceVoices(client, logger, credentials);
case 'ibm':
return getIbmVoices(client, logger, credentials);
case 'google':
return getGoogleVoices(client, logger, credentials);
case 'aws':
+280 -172
View File
@@ -6,8 +6,7 @@ const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly'
const { CartesiaClient } = require('@cartesia/cartesia-js');
const sdk = require('microsoft-cognitiveservices-speech-sdk');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const {
ResultReason,
SpeechConfig,
@@ -53,7 +52,6 @@ const EXPIRES = JAMBONES_TTS_CACHE_DURATION_MINS;
const OpenAI = require('openai');
const getAwsAuthToken = require('./get-aws-sts-token');
const trimTrailingSilence = (buffer) => {
assert.ok(buffer instanceof Buffer, 'trimTrailingSilence - argument is not a Buffer');
@@ -97,7 +95,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
let rtt;
logger = logger || noopLogger;
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'ibm', 'elevenlabs',
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'elevenlabs',
'whisper', 'deepgram', 'playht', 'rimelabs', 'verbio', 'cartesia', 'inworld', 'resemble'].includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid ..etc, not ${vendor}`);
@@ -123,11 +121,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
assert.ok(language, 'synthAudio requires language when nvidia is used');
assert.ok(credentials.riva_server_uri, 'synthAudio requires riva_server_uri in credentials when nvidia is used');
}
else if ('ibm' === vendor) {
assert.ok(voice, 'synthAudio requires voice when ibm is used');
assert.ok(credentials.tts_region, 'synthAudio requires tts_region in credentials when ibm watson is used');
assert.ok(credentials.tts_api_key, 'synthAudio requires tts_api_key in credentials when nuance is used');
}
else if ('wellsaid' === vendor) {
language = 'en-US'; // WellSaid only supports English atm
assert.ok(voice, 'synthAudio requires voice when wellsaid is used');
@@ -205,74 +198,81 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
const startAt = process.hrtime();
switch (vendor) {
case 'google':
audioData = await synthGoogle(logger, {credentials, stats, language, voice, gender, text});
audioData = await synthGoogle(logger, {
credentials, stats, language, voice, gender, key, text, model, options, instructions,
renderForCaching, disableTtsStreaming, disableTtsCache
});
break;
case 'aws':
case 'polly':
vendorLabel = 'aws';
audioData = await synthPolly(createHash, retrieveHash, logger,
{credentials, stats, language, voice, text, engine});
{credentials, stats, language, voice, key, text, engine, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'azure':
case 'microsoft':
vendorLabel = 'microsoft';
audioData = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId,
renderForCaching, disableTtsStreaming});
audioData = await synthMicrosoft(logger, {credentials, stats, language, voice, key, text, deploymentId,
renderForCaching, disableTtsStreaming, disableTtsCache});
break;
case 'nuance':
model = model || 'enhanced';
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, text});
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, key, text});
break;
case 'nvidia':
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, text,
renderForCaching, disableTtsStreaming});
break;
case 'ibm':
audioData = await synthIbm(logger, {credentials, stats, voice, text});
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, key, text,
renderForCaching, disableTtsStreaming, disableTtsCache});
break;
case 'wellsaid':
audioData = await synthWellSaid(logger, {credentials, stats, language, voice, text});
audioData = await synthWellSaid(logger, {credentials, stats, language, voice, key, text});
break;
case 'elevenlabs':
audioData = await synthElevenlabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming});
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'playht':
audioData = await synthPlayHT(client, logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming});
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'cartesia':
audioData = await synthCartesia(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming});
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'inworld':
audioData = await synthInworld(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming});
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'rimelabs':
audioData = await synthRimelabs(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming});
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'whisper':
audioData = await synthWhisper(logger, {
credentials, stats, voice, text, instructions, renderForCaching, disableTtsStreaming});
credentials, stats, voice, key, text, instructions, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'verbio':
audioData = await synthVerbio(client, logger, {
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
credentials, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache});
if (audioData?.filePath) return audioData;
break;
case 'deepgram':
audioData = await synthDeepgram(logger, {credentials, stats, model, text,
renderForCaching, disableTtsStreaming});
audioData = await synthDeepgram(logger, {credentials, stats, model, key, text,
renderForCaching, disableTtsStreaming, disableTtsCache});
break;
case 'resemble':
audioData = await synthResemble(logger, {
credentials, stats, voice, text, options, renderForCaching, disableTtsStreaming});
credentials, stats, voice, key, text, options, renderForCaching, disableTtsStreaming, disableTtsCache});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
audioData = await synthCustomVendor(logger,
{credentials, stats, language, voice, text});
{credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache});
break;
default:
assert(`synthAudio: unsupported speech vendor ${vendor}`);
@@ -307,9 +307,43 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
}
const synthPolly = async(createHash, retrieveHash, logger,
{credentials, stats, language, voice, engine, text}) => {
{credentials, stats, language, voice, engine, key, text, renderForCaching, disableTtsStreaming, disableTtsCache}) => {
const {region, accessKeyId, secretAccessKey, roleArn} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `language=${language}`;
params += `,playback_id=${key}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += ',vendor=aws';
if (accessKeyId && secretAccessKey) {
if (accessKeyId) params += `,accessKeyId=${accessKeyId}`;
if (secretAccessKey) params += `,secretAccessKey=${secretAccessKey}`;
} else if (roleArn) {
const cred = await getAwsAuthToken(
logger, createHash, retrieveHash,
{
region,
roleArn
});
if (cred) {
params += `,accessKeyId=${cred.accessKeyId}`;
params += `,secretAccessKey=${cred.secretAccessKey}`;
params += `,sessionToken=${cred.sessionToken}`;
}
}
if (region) params += `,region=${region}`;
if (engine) params += `,engine=${engine}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const {region, accessKeyId, secretAccessKey, roleArn} = credentials;
let polly;
if (accessKeyId && secretAccessKey) {
polly = new PollyClient({
@@ -369,111 +403,129 @@ const synthPolly = async(createHash, retrieveHash, logger,
}
};
const synthGoogle = async(logger, {credentials, stats, language, voice, gender, text}) => {
const client = new ttsGoogle.TextToSpeechClient(credentials);
// If google custom voice cloning is used.
// At this time 31 Oct 2024, google node sdk has not support voice cloning yet.
if (typeof voice === 'object' && voice.voice_cloning_key) {
try {
const accessToken = await client.auth.getAccessToken();
const projectId = await client.getProjectId();
const post = bent('https://texttospeech.googleapis.com', 'POST', 'json', {
'Authorization': `Bearer ${accessToken}`,
'x-goog-user-project': projectId,
'Content-Type': 'application/json; charset=utf-8'
});
const synthGoogle = async(logger, {
credentials, stats, language, voice, gender, key, text, model, options, instructions,
renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const isGemini = !!model;
const isVoiceCloning = typeof voice === 'object' && voice.voice_cloning_key;
// HD voices have pattern like en-US-Chirp3-HD-Charon
const isHDVoice = typeof voice === 'string' && voice.includes('-HD-');
const payload = {
input: {
text
},
voice: {
language_code: language,
voice_clone: {
voice_cloning_key: voice.voice_cloning_key
}
},
audioConfig: {
// Cloning voice at this time still in v1 beta version, and it support LINEAR16 in Wav format, 24.000Hz
audioEncoding: 'LINEAR16',
sample_rate_hertz: 24000
}
};
const wav = await post('/v1beta1/text:synthesize', payload);
return {
audioContent: Buffer.from(wav.audioContent, 'base64'),
extension: 'wav',
sampleRate: 24000
};
} catch (err) {
logger.info({err: await err.text()}, 'synthGoogle returned error');
throw err;
// Streaming support for Google TTS (Gemini, HD voices, and standard voices)
// Voice cloning does not support streaming
if (!isVoiceCloning && !JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
// Strip SSML tags for Gemini TTS (it doesn't support SSML)
let inputText = text;
if (isGemini && text.startsWith('<speak>')) {
inputText = text.replace(/<[^>]*>/g, '').trim();
logger.info('synthGoogle: Gemini TTS does not support SSML, stripped tags from input');
}
let params = '{';
params += `credentials=${Buffer.from(JSON.stringify(credentials.credentials)).toString('base64')}`;
params += `,playback_id=${key}`;
params += ',vendor=google';
params += `,voice=${voice}`;
params += `,language_code=${language || 'en-US'}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
// api_mode: tts (standard), live (HD voices), gemini (Gemini TTS)
const apiMode = options?.apiMode || (isGemini ? 'gemini' : (isHDVoice ? 'live' : 'tts'));
params += `,api_mode=${apiMode}`;
if (model) params += `,model_name=${model}`;
if (gender) params += `,gender=${gender}`;
// comma is used to separate parameters in freeswitch tts module
const prompt = options?.prompt || instructions;
if (prompt) params += `,prompt=${prompt.replace(/\n/g, ' ').replace(/,/g, ';')}`;
params += '}';
return {
filePath: `say:${params}${(isGemini ? inputText : text).replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
const opts = {
voice: {
...(typeof voice === 'string' && {name: voice}),
...(typeof voice === 'object' && {customVoice: voice}),
const client = new ttsGoogle.TextToSpeechClient(credentials);
// Build input based on voice type
let input;
if (isGemini) {
// Gemini TTS does not support SSML - strip tags if present
let inputText = text;
if (text.startsWith('<speak>')) {
inputText = text.replace(/<[^>]*>/g, '').trim();
logger.info('synthGoogle: Gemini TTS does not support SSML, stripped tags from input');
}
// Use instructions as prompt for Gemini TTS style control, options.prompt can override
const prompt = options?.prompt || instructions;
input = {
text: inputText,
...(prompt && { prompt })
};
} else {
input = text.startsWith('<speak>') ? { ssml: text } : { text };
}
// Build voice selection params based on voice type
let voiceParams;
if (isGemini) {
voiceParams = {
languageCode: language || 'en-US',
name: voice,
modelName: model
};
} else if (isVoiceCloning) {
voiceParams = {
languageCode: language,
voiceClone: {
voiceCloningKey: voice.voice_cloning_key
}
};
} else {
voiceParams = {
...(typeof voice === 'string' && { name: voice }),
...(typeof voice === 'object' && { customVoice: voice }),
languageCode: language,
ssmlGender: gender || 'SSML_VOICE_GENDER_UNSPECIFIED'
},
audioConfig: {audioEncoding: 'MP3'}
};
Object.assign(opts, {input: text.startsWith('<speak>') ? {ssml: text} : {text}});
};
}
// Build audio config based on voice type
let audioConfig;
let extension;
let sampleRate;
if (isVoiceCloning) {
audioConfig = { audioEncoding: 'LINEAR16', sampleRateHertz: 24000 };
extension = 'wav';
sampleRate = 24000;
} else {
audioConfig = { audioEncoding: 'MP3' };
extension = 'mp3';
sampleRate = 8000;
}
const opts = { input, voice: voiceParams, audioConfig };
try {
const responses = await client.synthesizeSpeech(opts);
logger.debug({ opts }, 'synthGoogle: request');
const [response] = await client.synthesizeSpeech(opts);
stats.increment('tts.count', ['vendor:google', 'accepted:yes']);
client.close();
return {
audioContent: responses[0].audioContent,
extension: 'mp3',
sampleRate: 8000
audioContent: response.audioContent,
extension,
sampleRate
};
} catch (err) {
console.error(err);
logger.info({err, opts}, 'synthAudio: Error synthesizing speech using google');
logger.info({ err, opts }, 'synthAudio: Error synthesizing speech using google');
stats.increment('tts.count', ['vendor:google', 'accepted:no']);
client && client.close();
throw err;
}
};
const synthIbm = async(logger, {credentials, stats, voice, text}) => {
const {tts_api_key, tts_region} = credentials;
const params = {
text,
voice,
accept: 'audio/mp3'
};
try {
const textToSpeech = new TextToSpeechV1({
authenticator: new IamAuthenticator({
apikey: tts_api_key,
}),
serviceUrl: `https://api.${tts_region}.text-to-speech.watson.cloud.ibm.com`
});
const r = await textToSpeech.synthesize(params);
const chunks = [];
for await (const chunk of r.result) {
chunks.push(chunk);
}
return {
audioContent: Buffer.concat(chunks),
extension: 'mp3',
sampleRate: 8000
};
} catch (err) {
logger.info({err, params}, 'synthAudio: Error synthesizing speech using ibm');
stats.increment('tts.count', ['vendor:ibm', 'accepted:no']);
throw new Error(err.statusText || err.message);
}
};
async function _synthOnPremMicrosoft(logger, {
credentials,
language,
@@ -527,9 +579,11 @@ const synthMicrosoft = async(logger, {
stats,
language,
voice,
key,
text,
renderForCaching,
disableTtsStreaming
disableTtsStreaming,
disableTtsCache
}) => {
try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
@@ -556,12 +610,13 @@ const synthMicrosoft = async(logger, {
}
if (!JAMBONES_DISABLE_TTS_STREAMING && !JAMBONES_DISABLE_AZURE_TTS_STREAMING &&
!renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${apiKey}`;
let params = '{';
params += `api_key=${apiKey}`;
params += `,playback_id=${key}`;
params += `,language=${language}`;
params += ',vendor=microsoft';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (region) params += `,region=${region}`;
if (custom_tts_endpoint) params += `,endpointId=${custom_tts_endpoint}`;
if (custom_tts_endpoint_url) params += `,endpoint=${custom_tts_endpoint_url}`;
@@ -734,15 +789,16 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
};
const synthNvidia = async(client, logger, {
credentials, stats, language, voice, model, text, renderForCaching, disableTtsStreaming
credentials, stats, language, voice, model, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {riva_server_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{riva_server_uri=${riva_server_uri}`;
params += `,playback_id=${key}`;
params += `,voice=${voice}`;
params += `,language=${language}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += '}';
return {
@@ -782,9 +838,27 @@ const synthNvidia = async(client, logger, {
};
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text, filePath}) => {
const synthCustomVendor = async(logger, {credentials, stats, language, voice,
text, filePath, renderForCaching, disableTtsStreaming, key, disableTtsCache}) => {
const {vendor, auth_token, custom_tts_url} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `auth_token=${auth_token}`;
params += `,playback_id=${key}`;
params += `,custom_tts_url=${custom_tts_url}`;
params += ',vendor=custom';
params += `,voice=${voice}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += '}';
return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
try {
const post = bent('POST', {
'Authorization': `Bearer ${auth_token}`,
@@ -813,20 +887,24 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
};
const synthElevenlabs = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const {api_key, model_id, api_uri, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=elevenlabs';
params += `,voice=${voice}`;
params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (api_uri) params += `,api_uri=${api_uri}`;
if (opts.optimize_streaming_latency !== null && opts.optimize_streaming_latency !== undefined) {
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency}`;
}
if (opts.voice_settings?.similarity_boost) params += `,similarity_boost=${opts.voice_settings.similarity_boost}`;
if (opts.voice_settings?.stability) params += `,stability=${opts.voice_settings.stability}`;
if (opts.voice_settings?.style) params += `,style=${opts.voice_settings.style}`;
@@ -849,7 +927,7 @@ const synthElevenlabs = async(logger, {
const optimize_streaming_latency = opts.optimize_streaming_latency ?
`?optimize_streaming_latency=${opts.optimize_streaming_latency}` : '';
try {
const post = bent('https://api.elevenlabs.io', 'POST', 'buffer', {
const post = bent(`https://${api_uri || 'api.elevenlabs.io'}`, 'POST', 'buffer', {
'xi-api-key': api_key,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
@@ -876,7 +954,7 @@ const synthElevenlabs = async(logger, {
};
const synthPlayHT = async(client, logger, {
credentials, options, stats, voice, language, text, renderForCaching, disableTtsStreaming
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, user_id, voice_engine, playht_tts_uri, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
@@ -910,14 +988,15 @@ const synthPlayHT = async(client, logger, {
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += `,user_id=${user_id}`;
params += ',vendor=playht';
params += `,voice=${voice}`;
params += `,voice_engine=${voice_engine}`;
params += `,synthesize_url=${synthesizeUrl}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += `,language=${language}`;
if (opts.quality) params += `,quality=${opts.quality}`;
if (opts.speed) params += `,speed=${opts.speed}`;
@@ -970,19 +1049,20 @@ const synthPlayHT = async(client, logger, {
};
const synthInworld = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += `,model_id=${model_id}`;
params += ',vendor=inworld';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (opts.temperature) params += `,temperature=${opts.temperature}`;
if (opts.audioConfig?.pitch) params += `,pitch=${opts.pitch}`;
if (opts.audioConfig?.speakingRate) params += `,speakingRate=${opts.speakingRate}`;
@@ -1034,20 +1114,21 @@ const synthInworld = async(logger, {
};
const synthRimelabs = async(logger, {
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, model_id, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += `,model_id=${model_id}`;
params += ',vendor=rimelabs';
params += `,language=${language}`;
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (opts.speedAlpha) params += `,speed_alpha=${opts.speedAlpha}`;
if (opts.reduceLatency) params += `,reduce_latency=${opts.reduceLatency}`;
// Arcana model parameters
@@ -1055,6 +1136,8 @@ const synthRimelabs = async(logger, {
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
if (opts.top_p) params += `,top_p=${opts.top_p}`;
if (opts.max_tokens) params += `,max_tokens=${opts.max_tokens}`;
// Coda model parameters
if (opts.timeScaleFactor) params += `,time_scale_factor=${opts.timeScaleFactor}`;
params += '}';
return {
@@ -1090,18 +1173,21 @@ const synthRimelabs = async(logger, {
throw err;
}
};
const synthVerbio = async(client, logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
const synthVerbio = async(client, logger, {
credentials, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
//https://doc.speechcenter.verbio.com/#tag/Text-To-Speech-REST-API
if (text.length > 2000) {
throw new Error('Verbio cannot synthesize for the text length larger than 2000 characters');
}
const token = await getVerbioAccessToken(client, logger, credentials);
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{access_token=${token.access_token}`;
let params = '{';
params += `access_token=${token.access_token}`;
params += `,playback_id=${key}`;
params += ',vendor=verbio';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += '}';
return {
@@ -1135,17 +1221,18 @@ const synthVerbio = async(client, logger, {credentials, stats, voice, text, rend
}
};
const synthWhisper = async(logger, {credentials, stats, voice, text, instructions,
renderForCaching, disableTtsStreaming}) => {
const synthWhisper = async(logger, {credentials, stats, voice, key, text, instructions,
renderForCaching, disableTtsStreaming, disableTtsCache}) => {
const {api_key, model_id, baseURL, timeout, speed} = credentials;
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += `,model_id=${model_id}`;
params += ',vendor=whisper';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (speed) params += `,speed=${speed}`;
// comma is used to separated parameters in freeswitch tts module
if (instructions) params += `,instructions=${instructions.replace(/\n/g, ' ').replace(/,/g, ';')}`;
@@ -1183,14 +1270,16 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, instruction
}
};
const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
const synthDeepgram = async(logger, {credentials, stats, model, key, text, renderForCaching,
disableTtsStreaming, disableTtsCache}) => {
const {api_key, deepgram_tts_uri} = credentials;
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=deepgram';
params += `,voice=${model}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
params += '}';
@@ -1223,23 +1312,25 @@ const synthDeepgram = async(logger, {credentials, stats, model, text, renderForC
};
const synthCartesia = async(logger, {
credentials, options, stats, voice, language, text, renderForCaching, disableTtsStreaming
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, model_id, embedding, options: credOpts} = credentials;
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += `,model_id=${model_id}`;
params += ',vendor=cartesia';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += `,language=${language}`;
params += `,voice_mode=${embedding ? 'embedding' : 'id'}`;
if (embedding) params += `,embedding=${embedding}`;
if (opts.speed) params += `,speed=${opts.speed}`;
if (opts.emotion) params += `,emotion=${opts.emotion}`;
if (opts.volume) params += `,volume=${opts.volume}`;
params += '}';
return {
@@ -1252,7 +1343,7 @@ const synthCartesia = async(logger, {
try {
const client = new CartesiaClient({ apiKey: api_key });
const sampleRate = 48000;
const mp3 = await client.tts.bytes({
const mp3Stream = await client.tts.bytes({
modelId: model_id,
transcript: text,
voice: {
@@ -1265,13 +1356,20 @@ const synthCartesia = async(logger, {
id: voice
}
),
...(opts.speed || opts.emotion && {
...(model_id === 'sonic-2' && (opts.speed || opts.emotion) && {
experimentalControls: {
...(opts.speed !== null && opts.speed !== undefined && {speed: opts.speed}),
...(opts.emotion && {emotion: opts.emotion}),
...(opts.emotion && {emotion: [opts.emotion]}),
}
})
}),
},
...(model_id === 'sonic-3' && (opts.speed || opts.emotion || opts.volume) && {
generationConfig: {
...(opts.volume !== null && opts.volume !== undefined && {volume: opts.volume}),
...(opts.speed !== null && opts.speed !== undefined && {speed: opts.speed}),
...(opts.emotion !== null && opts.emotion !== undefined && {emotion: opts.emotion}),
}
}),
language: language,
outputFormat: {
container: 'mp3',
@@ -1279,8 +1377,16 @@ const synthCartesia = async(logger, {
sampleRate
},
});
// bytes() returns a ReadableStream - collect all chunks
const chunks = [];
for await (const chunk of mp3Stream) {
chunks.push(chunk);
}
const audioBuffer = Buffer.concat(chunks);
return {
audioContent: Buffer.from(mp3),
audioContent: audioBuffer,
extension: 'mp3',
sampleRate
};
@@ -1293,21 +1399,23 @@ const synthCartesia = async(logger, {
};
const synthResemble = async(logger, {
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
}) => {
const {api_key, resemble_tts_uri} = credentials;
const {api_key, resemble_tts_uri, resemble_tts_use_tls} = credentials;
const {project_uuid, use_hd} = options || {};
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=resemble';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (project_uuid) params += `,project_uuid=${project_uuid}`;
if (use_hd) params += `,use_hd=${use_hd}`;
if (resemble_tts_uri) params += `,endpoint=${resemble_tts_uri}`;
if (resemble_tts_use_tls) params += `,use_tls=${resemble_tts_use_tls}`;
params += '}';
+1 -7
View File
@@ -54,12 +54,6 @@ function makeBasicAuthHeader(username, password) {
return {Authorization: header};
}
function makeIbmKey(apiKey) {
const hash = crypto.createHash('sha1');
hash.update(apiKey);
return `ibm:${hash.digest('hex')}`;
}
function makeAwsKey(awsAccessKeyId) {
const hash = crypto.createHash('sha1');
hash.update(awsAccessKeyId);
@@ -143,7 +137,7 @@ const createRivaClient = async(rivaUri) => {
module.exports = {
makeSynthKey,
makeNuanceKey,
makeIbmKey,
makePlayhtKey,
makeAwsKey,
makeVerbioKey,
+2867 -2766
View File
File diff suppressed because it is too large Load Diff
+4 -5
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.2.16",
"version": "1.0.3",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -29,21 +29,20 @@
"23": "^0.0.0",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@cartesia/cartesia-js": "^2.1.0",
"@google-cloud/text-to-speech": "^5.5.0",
"@cartesia/cartesia-js": "^2.2.7",
"@google-cloud/text-to-speech": "^6.4.0",
"@grpc/grpc-js": "^1.9.14",
"@jambonz/realtimedb-helpers": "^0.8.7",
"bent": "^7.3.12",
"debug": "^4.3.4",
"form-urlencoded": "^6.1.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^11.0.0",
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
"openai": "^4.98.0",
"undici": "^7.5.0"
},
"devDependencies": {
"config": "^3.3.11",
"config": "^4.2.0",
"eslint": "^9.3.0",
"eslint-plugin-promise": "^6.2.0",
"husky": "^9.0.11",
+1 -1
View File
@@ -2,7 +2,7 @@ const test = require('tape').test ;
const exec = require('child_process').exec ;
test('starting docker network..', (t) => {
exec(`docker-compose -f ${__dirname}/docker-compose-testbed.yaml up -d`, (err, stdout, stderr) => {
exec(`docker compose -f ${__dirname}/docker-compose-testbed.yaml up -d`, (err, stdout, stderr) => {
setTimeout(() => {
t.end(err);
}, 2000);
+1 -1
View File
@@ -3,7 +3,7 @@ const exec = require('child_process').exec ;
test('stopping docker network..', (t) => {
t.timeoutAfter(10000);
exec(`docker-compose -f ${__dirname}/docker-compose-testbed.yaml down`, (err, stdout, stderr) => {
exec(`docker compose -f ${__dirname}/docker-compose-testbed.yaml down`, (err, stdout, stderr) => {
//console.log(`stderr: ${stderr}`);
process.exit(0);
});
-78
View File
@@ -1,78 +0,0 @@
const test = require('tape').test ;
const config = require('config');
const opts = config.get('redis');
const fs = require('fs');
const logger = require('pino')({level: 'error'});
process.on('unhandledRejection', (reason, p) => {
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
});
const stats = {
increment: () => {},
histogram: () => {}
};
test('IBM - create access key', async(t) => {
const fn = require('..');
const {client, getIbmAccessToken} = fn(opts, logger);
if (!process.env.IBM_API_KEY ) {
t.pass('skipping IBM test since no IBM api_key provided');
t.end();
client.quit();
return;
}
try {
let obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM');
t.ok(obj.access_token && !obj.servedFromCache, 'successfull received access token from IBM');
obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM - second request');
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
await client.flushall();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('IBM - retrieve tts voices test', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.IBM_TTS_API_KEY || !process.env.IBM_TTS_REGION) {
t.pass('skipping IBM test since no IBM api_key and/or region provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'ibm',
credentials: {
tts_api_key: process.env.IBM_TTS_API_KEY,
tts_region: process.env.IBM_TTS_REGION
}
};
const obj = await getTtsVoices(opts);
const {voices} = obj.result;
//console.log(JSON.stringify(voices));
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
await client.flushall();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
+1 -1
View File
@@ -2,6 +2,6 @@ require('./docker_start');
require('./synth');
require('./list-voices');
require('./aws');
require('./ibm');
require('./nuance');
require('./docker_stop');
-65
View File
@@ -38,71 +38,6 @@ test('Verbio - get Access key and voices', async(t) => {
client.quit();
});
test('IBM - create access key', async(t) => {
const fn = require('..');
const {client, getIbmAccessToken} = fn(opts, logger);
if (!process.env.IBM_API_KEY ) {
t.pass('skipping IBM test since no IBM api_key provided');
t.end();
client.quit();
return;
}
try {
let obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM');
t.ok(obj.access_token && !obj.servedFromCache, 'successfull received access token from IBM');
obj = await getIbmAccessToken(process.env.IBM_API_KEY);
//console.log({obj}, 'received access token from IBM - second request');
t.ok(obj.access_token && obj.servedFromCache, 'successfully received access token from cache');
await client.flushall();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('IBM - retrieve tts voices test', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
if (!process.env.IBM_TTS_API_KEY || !process.env.IBM_TTS_REGION) {
t.pass('skipping IBM test since no IBM api_key and/or region provided');
t.end();
client.quit();
return;
}
try {
const opts = {
vendor: 'ibm',
credentials: {
tts_api_key: process.env.IBM_TTS_API_KEY,
tts_region: process.env.IBM_TTS_REGION
}
};
const obj = await getTtsVoices(opts);
const {voices} = obj.result;
//console.log(JSON.stringify(voices));
t.ok(voices.length > 0 && voices[0].language,
`GetVoices: successfully retrieved ${voices.length} voices from IBM`);
await client.flushall();
t.end();
}
catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Nuance hosted tests', async(t) => {
const fn = require('..');
const {client, getTtsVoices} = fn(opts, logger);
+382 -44
View File
@@ -41,6 +41,7 @@ test('Google speech synth tests', async(t) => {
gender: 'FEMALE',
text: 'This is a test. This is only a test',
salt: 'foo.bar',
renderForCaching: true,
});
t.ok(!opts.servedFromCache, `successfully synthesized google audio to ${opts.filePath}`);
@@ -55,6 +56,7 @@ test('Google speech synth tests', async(t) => {
language: 'en-GB',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
renderForCaching: true,
});
t.ok(opts.servedFromCache, `successfully retrieved cached google audio from ${opts.filePath}`);
@@ -78,6 +80,7 @@ test('Google speech synth tests', async(t) => {
language: 'en-GB',
gender: 'FEMALE',
text: 'This is a test. This is only a test',
renderForCaching: true,
});
t.ok(!opts.servedFromCache, `successfully synthesized google audio regardless of current cache to ${opts.filePath}`);
} catch (err) {
@@ -114,7 +117,8 @@ GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided, GCP_CUSTOM_VOICE_M
voice: {
reportedUsage: 'REALTIME',
model: process.env.GCP_CUSTOM_VOICE_MODEL
}
},
renderForCaching: true,
});
t.ok(!opts.servedFromCache, `successfully synthesized google custom voice audio to ${opts.filePath}`);
} catch (err) {
@@ -132,7 +136,7 @@ test('Google speech voice cloning synth tests', async(t) => {
!process.env.GCP_CUSTOM_VOICE_JSON_KEY ||
!process.env.GCP_VOICE_CLONING_FILE &&
!process.env.GCP_VOICE_CLONING_JSON_KEY) {
t.pass(`skipping google speech synth tests since neither
t.pass(`skipping google speech synth tests since neither
GCP_CUSTOM_VOICE_FILE nor GCP_CUSTOM_VOICE_JSON_KEY provided,
GCP_VOICE_CLONING_FILE nor GCP_VOICE_CLONING_JSON_KEY is not provided`);
return t.end();
@@ -156,7 +160,8 @@ GCP_VOICE_CLONING_FILE nor GCP_VOICE_CLONING_JSON_KEY is not provided`);
text: 'This is a test. This is only a test. This is a test. This is only a test. This is a test. This is only a test',
voice: {
voice_cloning_key
}
},
renderForCaching: true,
});
t.ok(!opts.servedFromCache, `successfully synthesized google voice cloning audio to ${opts.filePath}`);
} catch (err) {
@@ -166,6 +171,374 @@ GCP_VOICE_CLONING_FILE nor GCP_VOICE_CLONING_JSON_KEY is not provided`);
client.quit();
});
test('Google Gemini TTS synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping Google Gemini TTS synth tests since neither GCP_FILE nor GCP_JSON_KEY provided');
return t.end();
}
try {
const str = process.env.GCP_JSON_KEY || fs.readFileSync(process.env.GCP_FILE);
const creds = JSON.parse(str);
const geminiModel = process.env.GCP_GEMINI_TTS_MODEL || 'gemini-2.5-flash-tts';
// Test Gemini TTS with model and instructions (both required for Gemini)
let result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Kore',
model: geminiModel,
text: 'Hello, this is a test of Google Gemini text to speech.',
instructions: 'Speak clearly and naturally.',
renderForCaching: true,
});
t.ok(!result.servedFromCache, `successfully synthesized Google Gemini TTS audio to ${result.filePath}`);
t.ok(result.filePath.endsWith('.mp3'), 'Gemini TTS audio file has correct extension');
// Test Gemini TTS with different voice and instructions
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Charon',
model: geminiModel,
text: 'Welcome to our service. How can I help you today?',
instructions: 'Speak in a warm, friendly and professional tone.',
renderForCaching: true,
});
t.ok(!result.servedFromCache, `successfully synthesized Gemini TTS with instructions to ${result.filePath}`);
// Test cache retrieval
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Kore',
model: geminiModel,
text: 'Hello, this is a test of Google Gemini text to speech.',
instructions: 'Speak clearly and naturally.',
renderForCaching: true,
});
t.ok(result.servedFromCache, `successfully retrieved Gemini TTS audio from cache ${result.filePath}`);
// Test SSML stripping (Gemini doesn't support SSML)
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Leda',
model: geminiModel,
text: '<speak>This SSML should be stripped for Gemini TTS.</speak>',
instructions: 'Speak naturally.',
disableTtsCache: true,
renderForCaching: true,
});
t.ok(!result.servedFromCache, `successfully synthesized Gemini TTS with SSML stripped to ${result.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Google TTS streaming tests (!JAMBONES_DISABLE_TTS_STREAMING)', async(t) => {
// Ensure streaming is enabled (default behavior)
delete process.env.JAMBONES_DISABLE_TTS_STREAMING;
// Clear require cache to reload config with new env var
delete require.cache[require.resolve('../lib/config')];
delete require.cache[require.resolve('../lib/synth-audio')];
delete require.cache[require.resolve('..')];
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping Google TTS streaming tests since neither GCP_FILE nor GCP_JSON_KEY provided');
return t.end();
}
try {
const str = process.env.GCP_JSON_KEY || fs.readFileSync(process.env.GCP_FILE);
const creds = JSON.parse(str);
const geminiModel = process.env.GCP_GEMINI_TTS_MODEL || 'gemini-2.5-flash-tts';
// Test 1: Standard voice streaming (use_live_api=0)
let result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'en-US-Wavenet-D',
gender: 'MALE',
text: 'This is a test of standard voice streaming.',
disableTtsCache: true
});
t.ok(result.filePath.startsWith('say:'), 'Standard voice returns streaming say: path');
t.ok(result.filePath.includes('vendor=google'), 'Standard voice streaming path contains vendor=google');
t.ok(result.filePath.includes('api_mode=tts'), 'Standard voice uses api_mode=tts');
t.ok(result.filePath.includes('voice=en-US-Wavenet-D'), 'Standard voice streaming path contains voice');
// Verify credentials are base64 encoded (no raw JSON braces that would break FreeSWitch parsing)
t.ok(result.filePath.includes('credentials='), 'Standard voice streaming path contains credentials');
t.ok(!result.filePath.includes('credentials={'), 'Credentials are not raw JSON (base64 encoded)');
// Test 2: HD voice streaming (api_mode=live)
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'en-US-Chirp3-HD-Charon',
text: 'This is a test of HD voice streaming.',
disableTtsCache: true
});
t.ok(result.filePath.startsWith('say:'), 'HD voice returns streaming say: path');
t.ok(result.filePath.includes('vendor=google'), 'HD voice streaming path contains vendor=google');
t.ok(result.filePath.includes('api_mode=live'), 'HD voice uses api_mode=live');
t.ok(result.filePath.includes('voice=en-US-Chirp3-HD-Charon'), 'HD voice streaming path contains voice');
// Test 3: Gemini TTS streaming (api_mode=gemini)
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Kore',
model: geminiModel,
text: 'This is a test of Gemini TTS streaming.',
instructions: 'Speak naturally.',
disableTtsCache: true
});
t.ok(result.filePath.startsWith('say:'), 'Gemini TTS returns streaming say: path');
t.ok(result.filePath.includes('vendor=google'), 'Gemini TTS streaming path contains vendor=google');
t.ok(result.filePath.includes('api_mode=gemini'), 'Gemini TTS uses api_mode=gemini');
t.ok(result.filePath.includes(`model_name=${geminiModel}`), 'Gemini TTS streaming path contains model_name');
t.ok(result.filePath.includes('prompt=Speak naturally.'), 'Gemini TTS streaming path contains prompt');
// Test 4: Gemini TTS with SSML stripping in streaming mode
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Leda',
model: geminiModel,
text: '<speak>This SSML should be stripped.</speak>',
instructions: 'Speak naturally.',
disableTtsCache: true
});
t.ok(result.filePath.startsWith('say:'), 'Gemini TTS with SSML returns streaming say: path');
t.ok(!result.filePath.includes('<speak>'), 'SSML tags are stripped from streaming path');
t.ok(result.filePath.includes('This SSML should be stripped.'), 'Text content is preserved after SSML stripping');
// Test 5: Gemini TTS with prompt containing special characters
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Kore',
model: geminiModel,
text: 'Testing special characters in prompt.',
options: { prompt: 'Speak in a warm, friendly tone' },
disableTtsCache: true
});
t.ok(result.filePath.startsWith('say:'), 'Gemini TTS with special chars returns streaming say: path');
// Commas in prompt should be replaced with semicolons
t.ok(result.filePath.includes('prompt=Speak in a warm; friendly tone'), 'Commas in prompt are escaped to semicolons');
// Test 6: options.apiMode override (force live on standard voice)
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'en-US-Wavenet-D',
text: 'Testing apiMode option override to live.',
options: { apiMode: 'live' },
disableTtsCache: true
});
t.ok(result.filePath.includes('api_mode=live'), 'options.apiMode=live overrides default for standard voice');
// Test 7: options.apiMode override (force gemini without model)
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Kore',
text: 'Testing apiMode option override to gemini.',
options: { apiMode: 'gemini' },
disableTtsCache: true
});
t.ok(result.filePath.includes('api_mode=gemini'), 'options.apiMode=gemini overrides default');
// Test 8: options.apiMode override (force tts on HD voice)
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'en-US-Chirp3-HD-Charon',
text: 'Testing apiMode option override to tts on HD voice.',
options: { apiMode: 'tts' },
disableTtsCache: true
});
t.ok(result.filePath.includes('api_mode=tts'), 'options.apiMode=tts overrides HD voice default');
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Google TTS non-streaming tests (JAMBONES_DISABLE_TTS_STREAMING=true)', async(t) => {
// Enable streaming disable flag
process.env.JAMBONES_DISABLE_TTS_STREAMING = 'true';
// Clear require cache to reload config with new env var
delete require.cache[require.resolve('../lib/config')];
delete require.cache[require.resolve('../lib/synth-audio')];
delete require.cache[require.resolve('..')];
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.GCP_FILE && !process.env.GCP_JSON_KEY) {
t.pass('skipping Google TTS non-streaming tests since neither GCP_FILE nor GCP_JSON_KEY provided');
delete process.env.JAMBONES_DISABLE_TTS_STREAMING;
return t.end();
}
try {
const str = process.env.GCP_JSON_KEY || fs.readFileSync(process.env.GCP_FILE);
const creds = JSON.parse(str);
const geminiModel = process.env.GCP_GEMINI_TTS_MODEL || 'gemini-2.5-flash-tts';
// Test 1: Standard voice falls back to non-streaming API
let result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'en-US-Wavenet-D',
gender: 'MALE',
text: 'This is a test with streaming disabled.',
disableTtsCache: true
});
t.ok(!result.filePath.startsWith('say:'), 'Standard voice does NOT return streaming say: path when disabled');
t.ok(result.filePath.endsWith('.mp3'), 'Standard voice returns mp3 file path');
// Test 2: HD voice falls back to non-streaming API
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'en-US-Chirp3-HD-Charon',
text: 'This is a test of HD voice with streaming disabled.',
disableTtsCache: true
});
t.ok(!result.filePath.startsWith('say:'), 'HD voice does NOT return streaming say: path when disabled');
t.ok(result.filePath.endsWith('.mp3'), 'HD voice returns mp3 file path');
// Test 3: Gemini TTS falls back to non-streaming API
result = await synthAudio(stats, {
vendor: 'google',
credentials: {
credentials: {
client_email: creds.client_email,
private_key: creds.private_key,
},
},
language: 'en-US',
voice: 'Kore',
model: geminiModel,
text: 'This is a test of Gemini TTS with streaming disabled.',
instructions: 'Speak naturally.',
disableTtsCache: true
});
t.ok(!result.filePath.startsWith('say:'), 'Gemini TTS does NOT return streaming say: path when disabled');
t.ok(result.filePath.endsWith('.mp3'), 'Gemini TTS returns mp3 file path');
} catch (err) {
console.error(err);
t.end(err);
} finally {
// Clean up: restore default behavior
delete process.env.JAMBONES_DISABLE_TTS_STREAMING;
delete require.cache[require.resolve('../lib/config')];
delete require.cache[require.resolve('../lib/synth-audio')];
delete require.cache[require.resolve('..')];
}
client.quit();
});
test('AWS speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -185,6 +558,7 @@ test('AWS speech synth tests', async(t) => {
language: 'en-US',
voice: 'Joey',
text: 'This is a test. This is only a test',
renderForCaching: true,
});
t.ok(!opts.servedFromCache, `successfully synthesized aws audio to ${opts.filePath}`);
@@ -198,6 +572,7 @@ test('AWS speech synth tests', async(t) => {
language: 'en-US',
voice: 'Joey',
text: 'This is a test. This is only a test',
renderForCaching: true,
});
t.ok(opts.servedFromCache, `successfully retrieved aws audio from cache ${opts.filePath}`);
} catch (err) {
@@ -497,46 +872,6 @@ test('Nvidia speech synth tests', async(t) => {
client.quit();
});
test('IBM watson speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.IBM_TTS_API_KEY || !process.env.IBM_TTS_REGION) {
t.pass('skipping IBM Watson speech synth tests since IBM_TTS_API_KEY or IBM_TTS_API_KEY not provided');
return t.end();
}
const text = `<speak> Hi there and welcome to jambones! jambones is the <sub alias="seapass">CPaaS</sub> designed with the needs of communication service providers in mind. This is an example of simple text-to-speech, but there is so much more you can do. Try us out!</speak>`;
try {
let opts = await synthAudio(stats, {
vendor: 'ibm',
credentials: {
tts_api_key: process.env.IBM_TTS_API_KEY,
tts_region: process.env.IBM_TTS_REGION,
},
language: 'en-US',
voice: 'en-US_AllisonV2Voice',
text,
});
t.ok(!opts.servedFromCache, `successfully synthesized ibm audio to ${opts.filePath}`);
opts = await synthAudio(stats, {
vendor: 'ibm',
credentials: {
tts_api_key: process.env.IBM_TTS_API_KEY,
tts_region: process.env.IBM_TTS_REGION,
},
language: 'en-US',
voice: 'en-US_AllisonV2Voice',
text,
});
t.ok(opts.servedFromCache, `successfully retrieved ibm audio from cache ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('Custom Vendor speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -552,6 +887,7 @@ test('Custom Vendor speech synth tests', async(t) => {
language: 'en-US',
voice: 'English-US.Female-1',
text: 'This is a test. This is only a test',
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized custom vendor audio to ${opts.filePath}`);
t.ok(opts.filePath.endsWith('wav'), 'audio is cached as wav file');
@@ -573,6 +909,7 @@ test('Custom Vendor speech synth tests', async(t) => {
language: 'en-US',
voice: 'English-US.Female-1',
text: 'This is a test. This is only a test',
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully get custom vendor cached audio to ${opts.filePath}`);
t.ok(opts.filePath.endsWith('wav'), 'audio is cached as wav file');
@@ -587,6 +924,7 @@ test('Custom Vendor speech synth tests', async(t) => {
language: 'en-US',
voice: 'English-US.Female-1',
text: '<speak>This is a test. This is only a test</speak>',
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized Custom Vendor audio to ${opts.filePath}`);
obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew2`);
@@ -711,7 +1049,7 @@ test('Cartesia speech synth tests', async(t) => {
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully playht eleven audio to ${opts.filePath}`);
t.ok(!opts.servedFromCache, `successfully cartesia eleven audio to ${opts.filePath}`);
} catch (err) {
console.error(JSON.stringify(err));