mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3ba1a14358 |
@@ -1,84 +0,0 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## Overview
|
||||
|
||||
`@jambonz/speech-utils` is a Node.js library providing TTS (Text-to-Speech) utilities for the jambonz CPaaS platform. It handles speech synthesis with caching through Redis and supports multiple TTS vendors.
|
||||
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
# Run tests (requires Docker for Redis)
|
||||
npm test
|
||||
|
||||
# Run linter
|
||||
npm run jslint
|
||||
|
||||
# Auto-fix lint issues
|
||||
npm run jslint:fix
|
||||
|
||||
# Generate coverage report
|
||||
npm run coverage
|
||||
```
|
||||
|
||||
## Testing
|
||||
|
||||
Tests use `tape` and require Redis. The test harness automatically starts/stops Redis via Docker Compose (`test/docker-compose-testbed.yaml`).
|
||||
|
||||
Most tests are conditional based on environment variables for vendor credentials:
|
||||
- `GCP_FILE` or `GCP_JSON_KEY` - Google TTS
|
||||
- `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_REGION` - AWS Polly
|
||||
- `MICROSOFT_API_KEY`, `MICROSOFT_REGION` - Azure TTS
|
||||
- `ELEVENLABS_API_KEY`, `ELEVENLABS_VOICE_ID`, `ELEVENLABS_MODEL_ID` - ElevenLabs
|
||||
- `OPENAI_API_KEY` - OpenAI Whisper TTS
|
||||
- And others per vendor
|
||||
|
||||
Redis config is in `config/test.json` (port 3379).
|
||||
|
||||
## Architecture
|
||||
|
||||
### Entry Point
|
||||
|
||||
`index.js` exports a factory function that takes Redis options and a logger, returning an object with these methods:
|
||||
- `synthAudio` - Main synthesis function
|
||||
- `getTtsVoices` - List available voices for a vendor
|
||||
- `purgeTtsCache` / `getTtsSize` / `addFileToCache` - Cache management
|
||||
- `getAwsAuthToken` - Token management
|
||||
|
||||
### Core Module: `lib/synth-audio.js`
|
||||
|
||||
The `synthAudio` function handles synthesis for all vendors. Key behaviors:
|
||||
1. **Cache check**: Generates SHA1 hash key from (vendor, language, voice, engine, model, text, instructions)
|
||||
2. **Streaming vs non-streaming**: When `JAMBONES_DISABLE_TTS_STREAMING` is not set and `renderForCaching=false`, returns `say:{params}text` format for FreeSWITCH streaming playback instead of generating files
|
||||
3. **Vendor dispatch**: Switch statement routes to vendor-specific synth functions (`synthGoogle`, `synthPolly`, `synthMicrosoft`, etc.)
|
||||
4. **Caching**: Stores audio as base64 JSON in Redis with configurable TTL (default 4 hours)
|
||||
|
||||
### Supported Vendors
|
||||
|
||||
google, aws/polly, microsoft/azure, nvidia (Riva), wellsaid, elevenlabs, cartesia, inworld, rimelabs, whisper (OpenAI), deepgram, resemble, custom:*
|
||||
|
||||
### gRPC Stubs
|
||||
|
||||
`stubs/riva/` contains generated protobuf/gRPC code for NVIDIA Riva.
|
||||
|
||||
## Environment Variables
|
||||
|
||||
Key configuration via env vars (see `lib/config.js`):
|
||||
- `JAMBONES_DISABLE_TTS_STREAMING` - Force non-streaming mode
|
||||
- `JAMBONES_DISABLE_AZURE_TTS_STREAMING` - Azure-specific streaming disable
|
||||
- `JAMBONES_TTS_CACHE_DURATION_MINS` - Cache TTL in minutes (default: 240)
|
||||
- `JAMBONES_TTS_TRIM_SILENCE` - Trim trailing silence from audio
|
||||
- `JAMBONES_TMP_FOLDER` - Temp folder for audio files (default: /tmp)
|
||||
- `JAMBONES_HTTP_PROXY_IP`, `JAMBONES_HTTP_PROXY_PORT` - HTTP proxy for Azure
|
||||
- `JAMBONES_AZURE_ENABLE_SSML` - Force SSML wrapper for Azure plain text
|
||||
|
||||
## Key Dependencies
|
||||
|
||||
- `@jambonz/realtimedb-helpers` - Redis client and hash utilities
|
||||
- `@google-cloud/text-to-speech` - Google TTS
|
||||
- `@aws-sdk/client-polly` - AWS Polly
|
||||
- `microsoft-cognitiveservices-speech-sdk` - Azure TTS
|
||||
- `@grpc/grpc-js` - gRPC for Riva
|
||||
- `openai` - OpenAI Whisper TTS
|
||||
- `bent` - HTTP client for REST-based vendors
|
||||
+13
-375
@@ -80,9 +80,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
logger = logger || noopLogger;
|
||||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
|
||||
'murf', 'xai']
|
||||
.includes(vendor) ||
|
||||
'whisper', 'deepgram', 'rimelabs', 'cartesia', 'inworld', 'resemble'].includes(vendor) ||
|
||||
vendor.startsWith('custom'),
|
||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||
if ('google' === vendor) {
|
||||
@@ -98,7 +96,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
else if ('nvidia' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
|
||||
assert.ok(language, 'synthAudio requires language when nvidia is used');
|
||||
assert.ok(credentials.riva_server_uri, 'synthAudio requires riva_server_uri in credentials when nvidia is used');
|
||||
assert.ok(credentials.riva_server_uri || credentials.api_key,
|
||||
'synthAudio requires riva_server_uri (self-hosted) or api_key (NVCF cloud) in credentials when nvidia is used');
|
||||
}
|
||||
else if ('wellsaid' === vendor) {
|
||||
language = 'en-US'; // WellSaid only supports English atm
|
||||
@@ -126,26 +125,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
if (!credentials.deepgram_tts_uri) {
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
|
||||
}
|
||||
} else if ('deepgramflux' === vendor) {
|
||||
// Deepgram Flux TTS (/v2/speak); the flux model rides on `voice`/`model`
|
||||
if (!credentials.deepgram_tts_uri) {
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgramflux is used');
|
||||
}
|
||||
} else if ('xai' === vendor) {
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when xai is used');
|
||||
} else if ('cartesia' === vendor) {
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when cartesia is used');
|
||||
assert.ok(credentials.model_id, 'synthAudio requires model_id when cartesia is used');
|
||||
} else if ('gradium' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when gradium is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
|
||||
} else if ('nineninesix' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nineninesix is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when nineninesix is used');
|
||||
assert.ok(credentials.model_id, 'synthAudio requires model_id when nineninesix is used');
|
||||
} else if ('murf' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when murf is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when murf is used');
|
||||
} else if (vendor === 'resemble') {
|
||||
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
||||
@@ -220,16 +202,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'gradium':
|
||||
audioData = await synthGradium(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'nineninesix':
|
||||
audioData = await synthNineninesix(logger, {
|
||||
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'inworld':
|
||||
audioData = await synthInworld(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
@@ -240,11 +212,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'murf':
|
||||
audioData = await synthMurf(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'whisper':
|
||||
audioData = await synthWhisper(logger, {
|
||||
credentials, stats, voice, key, text, instructions, renderForCaching, disableTtsStreaming,
|
||||
@@ -254,15 +221,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
audioData = await synthDeepgram(logger, {credentials, stats, model, key, text,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
break;
|
||||
case 'deepgramflux':
|
||||
audioData = await synthDeepgramFlux(logger, {credentials, stats, model: model || voice, key, text,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
break;
|
||||
case 'xai':
|
||||
audioData = await synthXai(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'resemble':
|
||||
audioData = await synthResemble(logger, {
|
||||
credentials, stats, voice, key, text, options, renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
@@ -401,32 +359,6 @@ const synthPolly = async(createHash, retrieveHash, logger,
|
||||
}
|
||||
};
|
||||
|
||||
/* google AudioConfig settings we support, as [google camelCase name, freeswitch param name] */
|
||||
const GOOGLE_AUDIO_SETTINGS = [
|
||||
['speakingRate', 'speaking_rate'],
|
||||
['pitch', 'pitch'],
|
||||
['volumeGainDb', 'volume_gain_db']
|
||||
];
|
||||
|
||||
/**
|
||||
* Extract google AudioConfig settings from the synthesizer options. They may be supplied
|
||||
* nested under an audioConfig property (mirroring google's AudioConfig object) or flat at
|
||||
* the top level, and either google's camelCase or snake_case names are accepted.
|
||||
* @see https://cloud.google.com/text-to-speech/docs/reference/rest/v1/text/synthesize#AudioConfig
|
||||
* @returns object keyed by google's camelCase names, holding only valid numeric settings
|
||||
*/
|
||||
const googleAudioConfig = (options) => {
|
||||
const provided = {...options, ...(options?.audioConfig || {})};
|
||||
const audioConfig = {};
|
||||
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
|
||||
const value = provided[name] ?? provided[snakeName];
|
||||
/* note: 0 is meaningful for pitch and volumeGainDb, so check for absence explicitly */
|
||||
if (value === undefined || value === null || value === '') continue;
|
||||
const num = Number(value);
|
||||
if (Number.isFinite(num)) audioConfig[name] = num;
|
||||
}
|
||||
return audioConfig;
|
||||
};
|
||||
|
||||
const synthGoogle = async(logger, {
|
||||
credentials, stats, language, voice, gender, key, text, model, options, instructions,
|
||||
@@ -462,15 +394,6 @@ const synthGoogle = async(logger, {
|
||||
// comma is used to separate parameters in freeswitch tts module
|
||||
const prompt = options?.prompt || instructions;
|
||||
if (prompt) params += `,prompt=${prompt.replace(/\n/g, ' ').replace(/,/g, ';')}`;
|
||||
/**
|
||||
* AudioConfig settings. Note google only honors these in some api modes:
|
||||
* tts applies all of them, live (HD voices) applies speakingRate only,
|
||||
* and gemini ignores them entirely (use prompt instead for style control).
|
||||
*/
|
||||
const audioSettings = googleAudioConfig(options);
|
||||
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
|
||||
if (name in audioSettings) params += `,${snakeName}=${audioSettings[name]}`;
|
||||
}
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -539,9 +462,6 @@ const synthGoogle = async(logger, {
|
||||
sampleRate = 8000;
|
||||
}
|
||||
|
||||
/* gemini voices do not support the AudioConfig settings; they use prompt for style control */
|
||||
if (!isGemini) Object.assign(audioConfig, googleAudioConfig(options));
|
||||
|
||||
const opts = { input, voice: voiceParams, audioConfig };
|
||||
|
||||
try {
|
||||
@@ -763,10 +683,16 @@ const synthWellSaid = async(logger, {credentials, stats, language, voice, gender
|
||||
const synthNvidia = async(client, logger, {
|
||||
credentials, stats, language, voice, model, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {riva_server_uri} = credentials;
|
||||
const {riva_server_uri, api_key, function_id} = credentials;
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '';
|
||||
params += `{riva_server_uri=${riva_server_uri}`;
|
||||
let params = '{';
|
||||
if (api_key) {
|
||||
/* NVCF cloud: mediajam connects to grpc.nvcf.nvidia.com using these */
|
||||
params += `NVIDIA_API_KEY=${api_key}`;
|
||||
if (function_id) params += `,NVIDIA_FUNCTION_ID=${function_id}`;
|
||||
} else {
|
||||
params += `riva_server_uri=${riva_server_uri}`;
|
||||
}
|
||||
params += `,playback_id=${key}`;
|
||||
params += `,voice=${voice}`;
|
||||
params += `,language=${language}`;
|
||||
@@ -782,7 +708,7 @@ const synthNvidia = async(client, logger, {
|
||||
let rivaClient, request;
|
||||
const sampleRate = 8000;
|
||||
try {
|
||||
rivaClient = await createRivaClient(riva_server_uri);
|
||||
rivaClient = await createRivaClient(riva_server_uri, {apiKey: api_key, functionId: function_id});
|
||||
request = new SynthesizeSpeechRequest();
|
||||
request.setVoiceName(voice);
|
||||
request.setLanguageCode(language);
|
||||
@@ -1050,72 +976,6 @@ const synthRimelabs = async(logger, {
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthMurf = async(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id, api_uri, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
/* param keys here must match mod_murf_tts's text_param handler */
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=murf';
|
||||
params += `,voice=${voice}`;
|
||||
if (model_id) params += `,model_id=${model_id}`;
|
||||
if (language) params += `,language=${language}`;
|
||||
if (api_uri) params += `,api_uri=${api_uri}`;
|
||||
if (opts.style) params += `,style=${opts.style}`;
|
||||
if (opts.rate !== undefined && opts.rate !== null) params += `,rate=${opts.rate}`;
|
||||
if (opts.pitch !== undefined && opts.pitch !== null) params += `,pitch=${opts.pitch}`;
|
||||
if (opts.variation !== undefined && opts.variation !== null) params += `,variation=${opts.variation}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
/* no Accept header: murf returns 406 if it doesn't match; the response
|
||||
container is selected by the `format` field in the body instead */
|
||||
const post = bent(api_uri || 'https://global.api.murf.ai', 'POST', 'buffer', {
|
||||
'api-key': api_key,
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
/* murf REST schema is documented loosely; field names follow the SDK params
|
||||
(voice_id/model/format/sample_rate) plus the websocket voice fields. */
|
||||
const audioContent = await post('/v1/speech/stream', {
|
||||
text,
|
||||
voice_id: voice,
|
||||
...(model_id && {model: model_id}),
|
||||
...(language && {locale: language}),
|
||||
...(opts.style && {style: opts.style}),
|
||||
...(opts.rate !== undefined && opts.rate !== null && {rate: opts.rate}),
|
||||
...(opts.pitch !== undefined && opts.pitch !== null && {pitch: opts.pitch}),
|
||||
...(opts.variation !== undefined && opts.variation !== null && {variation: opts.variation}),
|
||||
format: 'WAV',
|
||||
sample_rate: sampleRate,
|
||||
channel_type: 'MONO'
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'wav',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth murf returned error');
|
||||
stats.increment('tts.count', ['vendor:murf', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, key, text, instructions,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache}) => {
|
||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||
@@ -1206,116 +1066,6 @@ const synthDeepgram = async(logger, {credentials, stats, model, key, text, rende
|
||||
}
|
||||
};
|
||||
|
||||
// Deepgram Flux TTS — the conversation-native model served from /v2/speak.
|
||||
// Streaming rides the mediajam deepgramflux dialect via a say: filePath; the
|
||||
// batch/cache path POSTs to /v2/speak (mp3 is batch-only for Flux).
|
||||
const synthDeepgramFlux = async(logger, {credentials, stats, model, key, text, renderForCaching,
|
||||
disableTtsStreaming, disableTtsCache}) => {
|
||||
const {api_key, deepgram_tts_uri} = credentials;
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=deepgramflux';
|
||||
params += `,voice=${model}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (deepgram_tts_uri) params += `,endpoint=${deepgram_tts_uri}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
try {
|
||||
const post = bent(deepgram_tts_uri || 'https://api.deepgram.com', 'POST', 'buffer', {
|
||||
// on-premise deepgram does not require to have api_key
|
||||
...(api_key && {'Authorization': `Token ${api_key}`}),
|
||||
'Accept': 'audio/mpeg',
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post(`/v2/speak?model=${model}`, {
|
||||
text
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'mp3',
|
||||
sampleRate: 8000
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth Deepgram Flux returned error');
|
||||
stats.increment('tts.count', ['vendor:deepgramflux', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthXai = async(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, api_uri, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
const speed = opts.speed;
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=xai';
|
||||
if (voice) params += `,voice=${voice}`;
|
||||
if (language) params += `,language=${language}`;
|
||||
if (speed !== null && speed !== undefined) params += `,speed=${speed}`;
|
||||
if (opts.optimize_streaming_latency != null) {
|
||||
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency}`;
|
||||
}
|
||||
if (opts.text_normalization != null) params += `,text_normalization=${opts.text_normalization}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (api_uri) params += `,endpoint=${api_uri}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
try {
|
||||
const post = bent(`https://${api_uri || 'api.x.ai'}`, 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${api_key}`,
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post('/v1/tts', {
|
||||
text,
|
||||
language: language || 'auto',
|
||||
...(voice && {voice_id: voice}),
|
||||
...(speed !== null && speed !== undefined && {speed}),
|
||||
...(opts.optimize_streaming_latency != null && {optimize_streaming_latency: opts.optimize_streaming_latency}),
|
||||
...(opts.text_normalization != null && {text_normalization: opts.text_normalization}),
|
||||
output_format: {
|
||||
codec: 'wav',
|
||||
sample_rate: 8000
|
||||
}
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'wav',
|
||||
sampleRate: 8000
|
||||
};
|
||||
} catch (err) {
|
||||
// xAI errors are JSON {code, error} - read the body so the surfaced message isn't 'undefined'
|
||||
if (err.name === 'StatusError' && typeof err.text === 'function') {
|
||||
try {
|
||||
const body = await err.text();
|
||||
if (body) err.message = body;
|
||||
} catch (readErr) {
|
||||
logger.info({readErr}, 'synth xai: failed to read error response body');
|
||||
}
|
||||
}
|
||||
logger.info({err}, 'synth xai returned error');
|
||||
stats.increment('tts.count', ['vendor:xai', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthCartesia = async(logger, {
|
||||
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
@@ -1441,118 +1191,6 @@ const synthCartesia = async(logger, {
|
||||
|
||||
};
|
||||
|
||||
/* gradium.ai — json websocket for streaming, and a POST endpoint for the cache
|
||||
render. only_audio:true + pcm_8000 returns bare little-endian 16-bit samples,
|
||||
which is exactly the r8 container, so we avoid gradium's streaming wav header
|
||||
(it carries 0xffffffff as the RIFF size, since length is unknown up front).
|
||||
*/
|
||||
const synthGradium = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
const {json_config, pronunciation_id} = options || {};
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=gradium';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (model_id) params += `,model_id=${model_id}`;
|
||||
if (pronunciation_id) params += `,pronunciation_id=${pronunciation_id}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const post = bent('https://api.gradium.ai', 'POST', 'buffer', {
|
||||
'x-api-key': api_key,
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post('/api/post/speech/tts', {
|
||||
text,
|
||||
voice_id: voice,
|
||||
model_name: model_id || 'default',
|
||||
output_format: `pcm_${sampleRate}`,
|
||||
only_audio: true,
|
||||
...(json_config && {json_config}),
|
||||
...(pronunciation_id && {pronunciation_id})
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'r8',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth gradium returned error');
|
||||
stats.increment('tts.count', ['vendor:gradium', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
||||
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
||||
const synthNineninesix = async(logger, {
|
||||
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
params += ',vendor=nineninesix';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (language) params += `,language=${language}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const post = bent('https://api.nineninesix.ai', 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${api_key}`,
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post('/tts/bytes', {
|
||||
model_id,
|
||||
transcript: text,
|
||||
voice: {mode: 'id', id: voice},
|
||||
...(language && {language}),
|
||||
output_format: {
|
||||
container: 'wav',
|
||||
encoding: 'pcm_s16le',
|
||||
sample_rate: sampleRate
|
||||
}
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'wav',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth nineninesix returned error');
|
||||
stats.increment('tts.count', ['vendor:nineninesix', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthResemble = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
|
||||
+18
-3
@@ -38,9 +38,24 @@ function makeAwsKey(awsAccessKeyId) {
|
||||
return `aws:${hash.digest('hex')}`;
|
||||
}
|
||||
|
||||
const createRivaClient = async(rivaUri) => {
|
||||
const client = new RivaSpeechSynthesisClient(rivaUri, grpc.credentials.createInsecure());
|
||||
return client;
|
||||
// NVCF cloud TTS function-id default: ai-magpie-tts-multilingual (public)
|
||||
const NVIDIA_TTS_FUNCTION_ID = '877104f7-e885-42b9-8de8-f6e4c6303969';
|
||||
|
||||
const createRivaClient = async(rivaUri, {apiKey, functionId} = {}) => {
|
||||
if (apiKey) {
|
||||
/* NVCF cloud: TLS to grpc.nvcf.nvidia.com:443 with per-RPC metadata
|
||||
(function-id + Bearer api key) baked into the channel credentials */
|
||||
const callCreds = grpc.credentials.createFromMetadataGenerator((_params, cb) => {
|
||||
const md = new grpc.Metadata();
|
||||
md.add('function-id', functionId || NVIDIA_TTS_FUNCTION_ID);
|
||||
md.add('authorization', `Bearer ${apiKey}`);
|
||||
cb(null, md);
|
||||
});
|
||||
const creds = grpc.credentials.combineChannelCredentials(
|
||||
grpc.credentials.createSsl(), callCreds);
|
||||
return new RivaSpeechSynthesisClient('grpc.nvcf.nvidia.com:443', creds);
|
||||
}
|
||||
return new RivaSpeechSynthesisClient(rivaUri, grpc.credentials.createInsecure());
|
||||
};
|
||||
|
||||
module.exports = {
|
||||
|
||||
Generated
+63
-116
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.14",
|
||||
"version": "1.0.6",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.14",
|
||||
"version": "1.0.6",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"23": "^0.0.0",
|
||||
@@ -19,8 +19,9 @@
|
||||
"bent": "^7.3.12",
|
||||
"debug": "^4.3.4",
|
||||
"google-protobuf": "^3.21.2",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.51.0",
|
||||
"openai": "^4.98.0"
|
||||
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
|
||||
"openai": "^4.98.0",
|
||||
"undici": "^7.5.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"config": "^4.2.0",
|
||||
@@ -759,46 +760,6 @@
|
||||
"node": ">=18.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@azure/abort-controller": {
|
||||
"version": "2.2.0",
|
||||
"resolved": "https://registry.npmjs.org/@azure/abort-controller/-/abort-controller-2.2.0.tgz",
|
||||
"integrity": "sha512-fNAjWnA/nZ2jz31kxR/AqRaUT8ewHBw/WuBIosK0moMy1C9e5ValbDfFdIxJzVOOYaYkV/b2F1S4H/aHiqfVQg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"tslib": "^2.6.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@azure/core-auth": {
|
||||
"version": "1.11.0",
|
||||
"resolved": "https://registry.npmjs.org/@azure/core-auth/-/core-auth-1.11.0.tgz",
|
||||
"integrity": "sha512-IUZydyTUkDnYdstOW9pFOOUQlBjAepK5teihDE3x6yxsPJs/hsAaaYpeGxdxrgtOiJbBKSjKW7MDk7AEhb4LRg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@azure/abort-controller": "^2.1.2",
|
||||
"@azure/core-util": "^1.13.0",
|
||||
"tslib": "^2.6.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@azure/core-util": {
|
||||
"version": "1.14.0",
|
||||
"resolved": "https://registry.npmjs.org/@azure/core-util/-/core-util-1.14.0.tgz",
|
||||
"integrity": "sha512-9n2pWK61veAuN0V20t9lOuoV4CFMdyAZ1ygZzvBGk/pBBJRib/PjL9PLXa/aI2CcPpyHfqVsxxqLCYl6uZlfDw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@azure/abort-controller": "^2.1.2",
|
||||
"@typespec/ts-http-runtime": "^0.3.0",
|
||||
"tslib": "^2.6.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/code-frame": {
|
||||
"version": "7.26.2",
|
||||
"resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.26.2.tgz",
|
||||
@@ -2430,20 +2391,6 @@
|
||||
"resolved": "https://registry.npmjs.org/@types/webrtc/-/webrtc-0.0.37.tgz",
|
||||
"integrity": "sha512-JGAJC/ZZDhcrrmepU4sPLQLIOIAgs5oIK+Ieq90K8fdaNMhfdfqmYatJdgif1NDQtvrSlTOGJDUYHIDunuufOg=="
|
||||
},
|
||||
"node_modules/@typespec/ts-http-runtime": {
|
||||
"version": "0.3.8",
|
||||
"resolved": "https://registry.npmjs.org/@typespec/ts-http-runtime/-/ts-http-runtime-0.3.8.tgz",
|
||||
"integrity": "sha512-bLMpVcWZNzq6lYOybwFwOAR1IXKcHnhUNqYeHjl1bET/qE3jFPFH+p8Wrh3rU4xwdnifPxmKNESBYnvnmc75aA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"http-proxy-agent": "^7.0.0",
|
||||
"https-proxy-agent": "^7.0.0",
|
||||
"tslib": "^2.6.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/23": {
|
||||
"version": "0.0.0",
|
||||
"resolved": "https://registry.npmjs.org/23/-/23-0.0.0.tgz",
|
||||
@@ -5331,18 +5278,17 @@
|
||||
}
|
||||
},
|
||||
"node_modules/microsoft-cognitiveservices-speech-sdk": {
|
||||
"version": "1.51.0",
|
||||
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.51.0.tgz",
|
||||
"integrity": "sha512-BLLovv5PegOr5Lp52h4CSgL2c/ViFAh9LoU49ILNPyb/Z0giGI7nPV3xNLcmTtHLT+kfYNRGUIA62vORLzP8TA==",
|
||||
"version": "1.38.0",
|
||||
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.38.0.tgz",
|
||||
"integrity": "sha512-NA6J4eIDkeR9iN83rcn77Kn5AWQcizDEn1tLMjzRvSovUNB1FrZe0mWYO0fsGltUwMl3Ns5OZ3lGw42PU4fEYA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@azure/core-auth": "^1.9.0",
|
||||
"@types/webrtc": "^0.0.37",
|
||||
"agent-base": "^6.0.1",
|
||||
"bent": "^7.3.12",
|
||||
"https-proxy-agent": "^4.0.0",
|
||||
"uuid": "^11.1.1",
|
||||
"ws": "^8.21.0"
|
||||
"uuid": "^9.0.0",
|
||||
"ws": "^7.5.6"
|
||||
}
|
||||
},
|
||||
"node_modules/microsoft-cognitiveservices-speech-sdk/node_modules/https-proxy-agent": {
|
||||
@@ -5366,16 +5312,36 @@
|
||||
}
|
||||
},
|
||||
"node_modules/microsoft-cognitiveservices-speech-sdk/node_modules/uuid": {
|
||||
"version": "11.1.1",
|
||||
"resolved": "https://registry.npmjs.org/uuid/-/uuid-11.1.1.tgz",
|
||||
"integrity": "sha512-vIYxrBCC/N/K+Js3qSN88go7kIfNPssr/hHCesKCQNAjmgvYS2oqr69kIufEG+O4+PfezOH4EbIeHCfFov8ZgQ==",
|
||||
"version": "9.0.1",
|
||||
"resolved": "https://registry.npmjs.org/uuid/-/uuid-9.0.1.tgz",
|
||||
"integrity": "sha512-b+1eJOlsR9K8HJpow9Ok3fiWOWSIcIzXodvv0rQjVoOVNpWMpxf1wZNpt4y9h10odCNrqnYp1OBzRktckBe3sA==",
|
||||
"funding": [
|
||||
"https://github.com/sponsors/broofa",
|
||||
"https://github.com/sponsors/ctavan"
|
||||
],
|
||||
"license": "MIT",
|
||||
"bin": {
|
||||
"uuid": "dist/esm/bin/uuid"
|
||||
"uuid": "dist/bin/uuid"
|
||||
}
|
||||
},
|
||||
"node_modules/microsoft-cognitiveservices-speech-sdk/node_modules/ws": {
|
||||
"version": "7.5.10",
|
||||
"resolved": "https://registry.npmjs.org/ws/-/ws-7.5.10.tgz",
|
||||
"integrity": "sha512-+dbF1tHwZpXcbOJdVOkzLDxZP1ailvSxM6ZweXTegylPny803bFhA+vqBYw4s31NSAk4S2Qz+AKXK9a4wkdjcQ==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=8.3.0"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"bufferutil": "^4.0.1",
|
||||
"utf-8-validate": "^5.0.2"
|
||||
},
|
||||
"peerDependenciesMeta": {
|
||||
"bufferutil": {
|
||||
"optional": true
|
||||
},
|
||||
"utf-8-validate": {
|
||||
"optional": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/mime-db": {
|
||||
@@ -7065,6 +7031,15 @@
|
||||
"url": "https://github.com/sponsors/ljharb"
|
||||
}
|
||||
},
|
||||
"node_modules/undici": {
|
||||
"version": "7.24.5",
|
||||
"resolved": "https://registry.npmjs.org/undici/-/undici-7.24.5.tgz",
|
||||
"integrity": "sha512-3IWdCpjgxp15CbJnsi/Y9TCDE7HWVN19j1hmzVhoAkY/+CJx449tVxT5wZc1Gwg8J+P0LWvzlBzxYRnHJ+1i7Q==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=20.18.1"
|
||||
}
|
||||
},
|
||||
"node_modules/undici-types": {
|
||||
"version": "5.26.5",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-5.26.5.tgz",
|
||||
@@ -7949,34 +7924,6 @@
|
||||
"resolved": "https://registry.npmjs.org/@aws/lambda-invoke-store/-/lambda-invoke-store-0.2.4.tgz",
|
||||
"integrity": "sha512-iY8yvjE0y651BixKNPgmv1WrQc+GZ142sb0z4gYnChDDY2YqI4P/jsSopBWrKfAt7LOJAkOXt7rC/hms+WclQQ=="
|
||||
},
|
||||
"@azure/abort-controller": {
|
||||
"version": "2.2.0",
|
||||
"resolved": "https://registry.npmjs.org/@azure/abort-controller/-/abort-controller-2.2.0.tgz",
|
||||
"integrity": "sha512-fNAjWnA/nZ2jz31kxR/AqRaUT8ewHBw/WuBIosK0moMy1C9e5ValbDfFdIxJzVOOYaYkV/b2F1S4H/aHiqfVQg==",
|
||||
"requires": {
|
||||
"tslib": "^2.6.2"
|
||||
}
|
||||
},
|
||||
"@azure/core-auth": {
|
||||
"version": "1.11.0",
|
||||
"resolved": "https://registry.npmjs.org/@azure/core-auth/-/core-auth-1.11.0.tgz",
|
||||
"integrity": "sha512-IUZydyTUkDnYdstOW9pFOOUQlBjAepK5teihDE3x6yxsPJs/hsAaaYpeGxdxrgtOiJbBKSjKW7MDk7AEhb4LRg==",
|
||||
"requires": {
|
||||
"@azure/abort-controller": "^2.1.2",
|
||||
"@azure/core-util": "^1.13.0",
|
||||
"tslib": "^2.6.2"
|
||||
}
|
||||
},
|
||||
"@azure/core-util": {
|
||||
"version": "1.14.0",
|
||||
"resolved": "https://registry.npmjs.org/@azure/core-util/-/core-util-1.14.0.tgz",
|
||||
"integrity": "sha512-9n2pWK61veAuN0V20t9lOuoV4CFMdyAZ1ygZzvBGk/pBBJRib/PjL9PLXa/aI2CcPpyHfqVsxxqLCYl6uZlfDw==",
|
||||
"requires": {
|
||||
"@azure/abort-controller": "^2.1.2",
|
||||
"@typespec/ts-http-runtime": "^0.3.0",
|
||||
"tslib": "^2.6.2"
|
||||
}
|
||||
},
|
||||
"@babel/code-frame": {
|
||||
"version": "7.26.2",
|
||||
"resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.26.2.tgz",
|
||||
@@ -9171,16 +9118,6 @@
|
||||
"resolved": "https://registry.npmjs.org/@types/webrtc/-/webrtc-0.0.37.tgz",
|
||||
"integrity": "sha512-JGAJC/ZZDhcrrmepU4sPLQLIOIAgs5oIK+Ieq90K8fdaNMhfdfqmYatJdgif1NDQtvrSlTOGJDUYHIDunuufOg=="
|
||||
},
|
||||
"@typespec/ts-http-runtime": {
|
||||
"version": "0.3.8",
|
||||
"resolved": "https://registry.npmjs.org/@typespec/ts-http-runtime/-/ts-http-runtime-0.3.8.tgz",
|
||||
"integrity": "sha512-bLMpVcWZNzq6lYOybwFwOAR1IXKcHnhUNqYeHjl1bET/qE3jFPFH+p8Wrh3rU4xwdnifPxmKNESBYnvnmc75aA==",
|
||||
"requires": {
|
||||
"http-proxy-agent": "^7.0.0",
|
||||
"https-proxy-agent": "^7.0.0",
|
||||
"tslib": "^2.6.2"
|
||||
}
|
||||
},
|
||||
"abort-controller": {
|
||||
"version": "3.0.0",
|
||||
"resolved": "https://registry.npmjs.org/abort-controller/-/abort-controller-3.0.0.tgz",
|
||||
@@ -11169,17 +11106,16 @@
|
||||
"integrity": "sha512-/IXtbwEk5HTPyEwyKX6hGkYXxM9nbj64B+ilVJnC/R6B0pH5G4V3b0pVbL7DBj4tkhBAppbQUlf6F6Xl9LHu1g=="
|
||||
},
|
||||
"microsoft-cognitiveservices-speech-sdk": {
|
||||
"version": "1.51.0",
|
||||
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.51.0.tgz",
|
||||
"integrity": "sha512-BLLovv5PegOr5Lp52h4CSgL2c/ViFAh9LoU49ILNPyb/Z0giGI7nPV3xNLcmTtHLT+kfYNRGUIA62vORLzP8TA==",
|
||||
"version": "1.38.0",
|
||||
"resolved": "https://registry.npmjs.org/microsoft-cognitiveservices-speech-sdk/-/microsoft-cognitiveservices-speech-sdk-1.38.0.tgz",
|
||||
"integrity": "sha512-NA6J4eIDkeR9iN83rcn77Kn5AWQcizDEn1tLMjzRvSovUNB1FrZe0mWYO0fsGltUwMl3Ns5OZ3lGw42PU4fEYA==",
|
||||
"requires": {
|
||||
"@azure/core-auth": "^1.9.0",
|
||||
"@types/webrtc": "^0.0.37",
|
||||
"agent-base": "^6.0.1",
|
||||
"bent": "^7.3.12",
|
||||
"https-proxy-agent": "^4.0.0",
|
||||
"uuid": "^11.1.1",
|
||||
"ws": "^8.21.0"
|
||||
"uuid": "^9.0.0",
|
||||
"ws": "^7.5.6"
|
||||
},
|
||||
"dependencies": {
|
||||
"https-proxy-agent": {
|
||||
@@ -11199,9 +11135,15 @@
|
||||
}
|
||||
},
|
||||
"uuid": {
|
||||
"version": "11.1.1",
|
||||
"resolved": "https://registry.npmjs.org/uuid/-/uuid-11.1.1.tgz",
|
||||
"integrity": "sha512-vIYxrBCC/N/K+Js3qSN88go7kIfNPssr/hHCesKCQNAjmgvYS2oqr69kIufEG+O4+PfezOH4EbIeHCfFov8ZgQ=="
|
||||
"version": "9.0.1",
|
||||
"resolved": "https://registry.npmjs.org/uuid/-/uuid-9.0.1.tgz",
|
||||
"integrity": "sha512-b+1eJOlsR9K8HJpow9Ok3fiWOWSIcIzXodvv0rQjVoOVNpWMpxf1wZNpt4y9h10odCNrqnYp1OBzRktckBe3sA=="
|
||||
},
|
||||
"ws": {
|
||||
"version": "7.5.10",
|
||||
"resolved": "https://registry.npmjs.org/ws/-/ws-7.5.10.tgz",
|
||||
"integrity": "sha512-+dbF1tHwZpXcbOJdVOkzLDxZP1ailvSxM6ZweXTegylPny803bFhA+vqBYw4s31NSAk4S2Qz+AKXK9a4wkdjcQ==",
|
||||
"requires": {}
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -12405,6 +12347,11 @@
|
||||
"which-boxed-primitive": "^1.0.2"
|
||||
}
|
||||
},
|
||||
"undici": {
|
||||
"version": "7.24.5",
|
||||
"resolved": "https://registry.npmjs.org/undici/-/undici-7.24.5.tgz",
|
||||
"integrity": "sha512-3IWdCpjgxp15CbJnsi/Y9TCDE7HWVN19j1hmzVhoAkY/+CJx449tVxT5wZc1Gwg8J+P0LWvzlBzxYRnHJ+1i7Q=="
|
||||
},
|
||||
"undici-types": {
|
||||
"version": "5.26.5",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-5.26.5.tgz",
|
||||
|
||||
+4
-3
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.14",
|
||||
"version": "1.0.6",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -35,8 +35,9 @@
|
||||
"bent": "^7.3.12",
|
||||
"debug": "^4.3.4",
|
||||
"google-protobuf": "^3.21.2",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.51.0",
|
||||
"openai": "^4.98.0"
|
||||
"microsoft-cognitiveservices-speech-sdk": "1.38.0",
|
||||
"openai": "^4.98.0",
|
||||
"undici": "^7.5.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"config": "^4.2.0",
|
||||
|
||||
-247
@@ -442,95 +442,6 @@ test('Google TTS streaming tests (!JAMBONES_DISABLE_TTS_STREAMING)', async(t) =>
|
||||
});
|
||||
t.ok(result.filePath.includes('api_mode=tts'), 'options.apiMode=tts overrides HD voice default');
|
||||
|
||||
/* AudioConfig settings (speakingRate, pitch, volumeGainDb) */
|
||||
const googleCreds = {
|
||||
credentials: {
|
||||
client_email: creds.client_email,
|
||||
private_key: creds.private_key,
|
||||
},
|
||||
};
|
||||
|
||||
// Test 9: AudioConfig nested under options.audioConfig, as in google's API docs
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Wavenet-D',
|
||||
text: 'Testing nested audioConfig settings.',
|
||||
options: { audioConfig: { speakingRate: 1.4, pitch: -2.5, volumeGainDb: 6 } },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.includes(',speaking_rate=1.4'), 'nested audioConfig sets speaking_rate');
|
||||
t.ok(result.filePath.includes(',pitch=-2.5'), 'nested audioConfig sets pitch');
|
||||
t.ok(result.filePath.includes(',volume_gain_db=6'), 'nested audioConfig sets volume_gain_db');
|
||||
|
||||
// Test 10: AudioConfig flat at the top level of options, in camelCase or snake_case
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Wavenet-D',
|
||||
text: 'Testing flat audioConfig settings.',
|
||||
options: { speaking_rate: '0.8', volumeGainDb: 3 },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.includes(',speaking_rate=0.8'), 'flat snake_case speaking_rate is honored');
|
||||
t.ok(result.filePath.includes(',volume_gain_db=3'), 'flat camelCase volumeGainDb is honored');
|
||||
t.ok(!result.filePath.includes(',pitch='), 'unspecified audioConfig setting is omitted');
|
||||
|
||||
// Test 11: zero is a meaningful value for pitch and volumeGainDb, not an absent one
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Wavenet-D',
|
||||
text: 'Testing zero audioConfig settings.',
|
||||
options: { audioConfig: { pitch: 0, volumeGainDb: 0 } },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.includes(',pitch=0'), 'pitch=0 is passed through rather than dropped');
|
||||
t.ok(result.filePath.includes(',volume_gain_db=0'), 'volume_gain_db=0 is passed through rather than dropped');
|
||||
|
||||
// Test 12: non-numeric values are ignored, so they cannot corrupt the freeswitch param string
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Wavenet-D',
|
||||
text: 'Testing invalid audioConfig settings.',
|
||||
options: { audioConfig: { speakingRate: 'fast,evil=1', pitch: null, volumeGainDb: '' } },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(!result.filePath.includes('speaking_rate'), 'non-numeric speakingRate is ignored');
|
||||
t.ok(!result.filePath.includes('evil=1'), 'non-numeric value cannot inject extra params');
|
||||
t.ok(!result.filePath.includes('pitch='), 'null pitch is ignored');
|
||||
t.ok(!result.filePath.includes('volume_gain_db'), 'empty volumeGainDb is ignored');
|
||||
|
||||
// Test 13: no audioConfig supplied leaves the param string untouched
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Wavenet-D',
|
||||
text: 'Testing absent audioConfig settings.',
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(!/speaking_rate|pitch=|volume_gain_db/.test(result.filePath),
|
||||
'no audioConfig params are added when none are supplied');
|
||||
|
||||
// Test 14: HD voice (api_mode=live) also carries speaking_rate, the one setting google streams support
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Chirp3-HD-Charon',
|
||||
text: 'Testing audioConfig on an HD voice.',
|
||||
options: { audioConfig: { speakingRate: 1.25 } },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.includes('api_mode=live'), 'HD voice with audioConfig still uses api_mode=live');
|
||||
t.ok(result.filePath.includes(',speaking_rate=1.25'), 'HD voice streaming path carries speaking_rate');
|
||||
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
@@ -615,42 +526,6 @@ test('Google TTS non-streaming tests (JAMBONES_DISABLE_TTS_STREAMING=true)', asy
|
||||
t.ok(!result.filePath.startsWith('say:'), 'Gemini TTS does NOT return streaming say: path when disabled');
|
||||
t.ok(result.filePath.endsWith('.mp3'), 'Gemini TTS returns mp3 file path');
|
||||
|
||||
const googleCreds = {
|
||||
credentials: {
|
||||
client_email: creds.client_email,
|
||||
private_key: creds.private_key,
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
* Test 4: AudioConfig settings are accepted by the synthesize API.
|
||||
* Google rejects out-of-range values with a 400, so a successful render also confirms
|
||||
* the settings reached the request rather than being silently dropped.
|
||||
*/
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'en-US-Wavenet-D',
|
||||
text: 'This is a test of audioConfig with streaming disabled.',
|
||||
options: { audioConfig: { speakingRate: 1.4, pitch: -2.5, volumeGainDb: 6 } },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.endsWith('.mp3'), 'standard voice renders mp3 with audioConfig settings applied');
|
||||
|
||||
/* Test 5: gemini voices ignore the AudioConfig settings rather than failing on them */
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'google',
|
||||
credentials: googleCreds,
|
||||
language: 'en-US',
|
||||
voice: 'Kore',
|
||||
model: geminiModel,
|
||||
text: 'This is a test of audioConfig on Gemini TTS.',
|
||||
options: { audioConfig: { speakingRate: 1.4, pitch: -2.5, volumeGainDb: 6 } },
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.endsWith('.mp3'), 'gemini voice renders mp3 with audioConfig settings skipped');
|
||||
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
@@ -1058,65 +933,6 @@ test('Cartesia speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('gradium speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.GRADIUM_API_KEY) {
|
||||
t.pass('skipping gradium speech synth tests since GRADIUM_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'gradium',
|
||||
credentials: {
|
||||
api_key: process.env.GRADIUM_API_KEY,
|
||||
model_id: 'default'
|
||||
},
|
||||
voice: 'YTpq7expH9539ERJ',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthed gradium audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('nineninesix speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.NINENINESIX_API_KEY) {
|
||||
t.pass('skipping nineninesix speech synth tests since NINENINESIX_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'nineninesix',
|
||||
credentials: {
|
||||
api_key: process.env.NINENINESIX_API_KEY,
|
||||
model_id: 'gepard-1.0'
|
||||
},
|
||||
language: 'en',
|
||||
voice: '3ad7a827-7fd1-4954-bf35-47d4cc33d9ed',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthed nineninesix audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('inworld speech synth', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
@@ -1302,69 +1118,6 @@ test('Deepgram speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
})
|
||||
|
||||
test('Deepgram Flux speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.DEEPGRAM_API_KEY) {
|
||||
t.pass('skipping Deepgram Flux speech synth tests since DEEPGRAM_API_KEY');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones!';
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'deepgramflux',
|
||||
credentials: {
|
||||
api_key: process.env.DEEPGRAM_API_KEY
|
||||
},
|
||||
model: process.env.DEEPGRAM_FLUX_MODEL || 'flux-alexis-en',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized deepgramflux audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('xai speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.XAI_API_KEY) {
|
||||
t.pass('skipping xai speech synth tests - no XAI_API_KEY');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones!';
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'xai',
|
||||
credentials: {
|
||||
api_key: process.env.XAI_API_KEY,
|
||||
options: JSON.stringify({
|
||||
voice: process.env.XAI_VOICE || 'eve',
|
||||
speed: 1.0,
|
||||
optimize_streaming_latency: 1,
|
||||
text_normalization: true
|
||||
})
|
||||
},
|
||||
language: 'en',
|
||||
voice: process.env.XAI_VOICE || 'eve',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache && opts.filePath, `successfully synthesized xai audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('TTS Cache tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {purgeTtsCache, getTtsSize, client} = fn(opts, logger);
|
||||
|
||||
Reference in New Issue
Block a user