mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ab2b3f7fb2 | ||
|
|
e848d0276e | ||
|
|
d2447d477d | ||
|
|
ec8bbacc2c |
@@ -0,0 +1,84 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## Overview
|
||||
|
||||
`@jambonz/speech-utils` is a Node.js library providing TTS (Text-to-Speech) utilities for the jambonz CPaaS platform. It handles speech synthesis with caching through Redis and supports multiple TTS vendors.
|
||||
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
# Run tests (requires Docker for Redis)
|
||||
npm test
|
||||
|
||||
# Run linter
|
||||
npm run jslint
|
||||
|
||||
# Auto-fix lint issues
|
||||
npm run jslint:fix
|
||||
|
||||
# Generate coverage report
|
||||
npm run coverage
|
||||
```
|
||||
|
||||
## Testing
|
||||
|
||||
Tests use `tape` and require Redis. The test harness automatically starts/stops Redis via Docker Compose (`test/docker-compose-testbed.yaml`).
|
||||
|
||||
Most tests are conditional based on environment variables for vendor credentials:
|
||||
- `GCP_FILE` or `GCP_JSON_KEY` - Google TTS
|
||||
- `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_REGION` - AWS Polly
|
||||
- `MICROSOFT_API_KEY`, `MICROSOFT_REGION` - Azure TTS
|
||||
- `ELEVENLABS_API_KEY`, `ELEVENLABS_VOICE_ID`, `ELEVENLABS_MODEL_ID` - ElevenLabs
|
||||
- `OPENAI_API_KEY` - OpenAI Whisper TTS
|
||||
- And others per vendor
|
||||
|
||||
Redis config is in `config/test.json` (port 3379).
|
||||
|
||||
## Architecture
|
||||
|
||||
### Entry Point
|
||||
|
||||
`index.js` exports a factory function that takes Redis options and a logger, returning an object with these methods:
|
||||
- `synthAudio` - Main synthesis function
|
||||
- `getTtsVoices` - List available voices for a vendor
|
||||
- `purgeTtsCache` / `getTtsSize` / `addFileToCache` - Cache management
|
||||
- `getAwsAuthToken` - Token management
|
||||
|
||||
### Core Module: `lib/synth-audio.js`
|
||||
|
||||
The `synthAudio` function handles synthesis for all vendors. Key behaviors:
|
||||
1. **Cache check**: Generates SHA1 hash key from (vendor, language, voice, engine, model, text, instructions)
|
||||
2. **Streaming vs non-streaming**: When `JAMBONES_DISABLE_TTS_STREAMING` is not set and `renderForCaching=false`, returns `say:{params}text` format for FreeSWITCH streaming playback instead of generating files
|
||||
3. **Vendor dispatch**: Switch statement routes to vendor-specific synth functions (`synthGoogle`, `synthPolly`, `synthMicrosoft`, etc.)
|
||||
4. **Caching**: Stores audio as base64 JSON in Redis with configurable TTL (default 4 hours)
|
||||
|
||||
### Supported Vendors
|
||||
|
||||
google, aws/polly, microsoft/azure, nvidia (Riva), wellsaid, elevenlabs, cartesia, inworld, rimelabs, whisper (OpenAI), deepgram, resemble, custom:*
|
||||
|
||||
### gRPC Stubs
|
||||
|
||||
`stubs/riva/` contains generated protobuf/gRPC code for NVIDIA Riva.
|
||||
|
||||
## Environment Variables
|
||||
|
||||
Key configuration via env vars (see `lib/config.js`):
|
||||
- `JAMBONES_DISABLE_TTS_STREAMING` - Force non-streaming mode
|
||||
- `JAMBONES_DISABLE_AZURE_TTS_STREAMING` - Azure-specific streaming disable
|
||||
- `JAMBONES_TTS_CACHE_DURATION_MINS` - Cache TTL in minutes (default: 240)
|
||||
- `JAMBONES_TTS_TRIM_SILENCE` - Trim trailing silence from audio
|
||||
- `JAMBONES_TMP_FOLDER` - Temp folder for audio files (default: /tmp)
|
||||
- `JAMBONES_HTTP_PROXY_IP`, `JAMBONES_HTTP_PROXY_PORT` - HTTP proxy for Azure
|
||||
- `JAMBONES_AZURE_ENABLE_SSML` - Force SSML wrapper for Azure plain text
|
||||
|
||||
## Key Dependencies
|
||||
|
||||
- `@jambonz/realtimedb-helpers` - Redis client and hash utilities
|
||||
- `@google-cloud/text-to-speech` - Google TTS
|
||||
- `@aws-sdk/client-polly` - AWS Polly
|
||||
- `microsoft-cognitiveservices-speech-sdk` - Azure TTS
|
||||
- `@grpc/grpc-js` - gRPC for Riva
|
||||
- `openai` - OpenAI Whisper TTS
|
||||
- `bent` - HTTP client for REST-based vendors
|
||||
+131
-1
@@ -80,7 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
logger = logger || noopLogger;
|
||||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'inworld', 'resemble', 'murf', 'xai']
|
||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
|
||||
'murf', 'xai']
|
||||
.includes(vendor) ||
|
||||
vendor.startsWith('custom'),
|
||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||
@@ -135,6 +136,13 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
} else if ('cartesia' === vendor) {
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when cartesia is used');
|
||||
assert.ok(credentials.model_id, 'synthAudio requires model_id when cartesia is used');
|
||||
} else if ('gradium' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when gradium is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
|
||||
} else if ('nineninesix' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nineninesix is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when nineninesix is used');
|
||||
assert.ok(credentials.model_id, 'synthAudio requires model_id when nineninesix is used');
|
||||
} else if ('murf' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when murf is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when murf is used');
|
||||
@@ -212,6 +220,16 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'gradium':
|
||||
audioData = await synthGradium(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'nineninesix':
|
||||
audioData = await synthNineninesix(logger, {
|
||||
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'inworld':
|
||||
audioData = await synthInworld(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
@@ -1423,6 +1441,118 @@ const synthCartesia = async(logger, {
|
||||
|
||||
};
|
||||
|
||||
/* gradium.ai — json websocket for streaming, and a POST endpoint for the cache
|
||||
render. only_audio:true + pcm_8000 returns bare little-endian 16-bit samples,
|
||||
which is exactly the r8 container, so we avoid gradium's streaming wav header
|
||||
(it carries 0xffffffff as the RIFF size, since length is unknown up front).
|
||||
*/
|
||||
const synthGradium = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
const {json_config, pronunciation_id} = options || {};
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=gradium';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (model_id) params += `,model_id=${model_id}`;
|
||||
if (pronunciation_id) params += `,pronunciation_id=${pronunciation_id}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const post = bent('https://api.gradium.ai', 'POST', 'buffer', {
|
||||
'x-api-key': api_key,
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post('/api/post/speech/tts', {
|
||||
text,
|
||||
voice_id: voice,
|
||||
model_name: model_id || 'default',
|
||||
output_format: `pcm_${sampleRate}`,
|
||||
only_audio: true,
|
||||
...(json_config && {json_config}),
|
||||
...(pronunciation_id && {pronunciation_id})
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'r8',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth gradium returned error');
|
||||
stats.increment('tts.count', ['vendor:gradium', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
||||
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
||||
const synthNineninesix = async(logger, {
|
||||
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += `,model_id=${model_id}`;
|
||||
params += ',vendor=nineninesix';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (language) params += `,language=${language}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const post = bent('https://api.nineninesix.ai', 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${api_key}`,
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post('/tts/bytes', {
|
||||
model_id,
|
||||
transcript: text,
|
||||
voice: {mode: 'id', id: voice},
|
||||
...(language && {language}),
|
||||
output_format: {
|
||||
container: 'wav',
|
||||
encoding: 'pcm_s16le',
|
||||
sample_rate: sampleRate
|
||||
}
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'wav',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth nineninesix returned error');
|
||||
stats.increment('tts.count', ['vendor:nineninesix', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthResemble = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.12",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.12",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"23": "^0.0.0",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.10",
|
||||
"version": "1.0.12",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
|
||||
@@ -1058,6 +1058,65 @@ test('Cartesia speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('gradium speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.GRADIUM_API_KEY) {
|
||||
t.pass('skipping gradium speech synth tests since GRADIUM_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'gradium',
|
||||
credentials: {
|
||||
api_key: process.env.GRADIUM_API_KEY,
|
||||
model_id: 'default'
|
||||
},
|
||||
voice: 'YTpq7expH9539ERJ',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthed gradium audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('nineninesix speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.NINENINESIX_API_KEY) {
|
||||
t.pass('skipping nineninesix speech synth tests since NINENINESIX_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'nineninesix',
|
||||
credentials: {
|
||||
api_key: process.env.NINENINESIX_API_KEY,
|
||||
model_id: 'gepard-1.0'
|
||||
},
|
||||
language: 'en',
|
||||
voice: '3ad7a827-7fd1-4954-bf35-47d4cc33d9ed',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthed nineninesix audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('inworld speech synth', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
Reference in New Issue
Block a user