mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-04 07:43:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f4c3c7dc8b | ||
|
|
a32f71ac5a | ||
|
|
c8850998b5 | ||
|
|
e8b2009d29 | ||
|
|
27e07bb551 | ||
|
|
33456b93d9 | ||
|
|
70e91b5057 | ||
|
|
c066840f85 | ||
|
|
d89457f67d | ||
|
|
3c15976669 |
+92
-3
@@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
|
||||
'murf', 'xai']
|
||||
'murf', 'xai', 'fishaudio']
|
||||
.includes(vendor) ||
|
||||
vendor.startsWith('custom'),
|
||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||
@@ -149,6 +149,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
} else if (vendor === 'resemble') {
|
||||
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
||||
} else if ('fishaudio' === vendor) {
|
||||
/* no voice assert: fish synthesizes with its own default voice when
|
||||
reference_id is omitted, which is what the 'default' selection means */
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when fishaudio is used');
|
||||
}
|
||||
|
||||
const key = makeSynthKey({
|
||||
@@ -220,6 +224,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'fishaudio':
|
||||
audioData = await synthFishaudio(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'gradium':
|
||||
audioData = await synthGradium(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
@@ -941,8 +950,9 @@ const synthInworld = async(logger, {
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (opts.temperature) params += `,temperature=${opts.temperature}`;
|
||||
if (opts.audioConfig?.pitch) params += `,pitch=${opts.pitch}`;
|
||||
if (opts.audioConfig?.speakingRate) params += `,speakingRate=${opts.speakingRate}`;
|
||||
/* pitch and speakingRate are nested under audioConfig, matching Inworld's API */
|
||||
if (opts.audioConfig?.pitch) params += `,pitch=${opts.audioConfig.pitch}`;
|
||||
if (opts.audioConfig?.speakingRate) params += `,speakingRate=${opts.audioConfig.speakingRate}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -1462,6 +1472,10 @@ const synthGradium = async(logger, {
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (model_id) params += `,model_id=${model_id}`;
|
||||
if (pronunciation_id) params += `,pronunciation_id=${pronunciation_id}`;
|
||||
/* the say: param parser is brace-aware, so a nested json object survives intact */
|
||||
if (json_config) {
|
||||
params += `,json_config=${typeof json_config === 'string' ? json_config : JSON.stringify(json_config)}`;
|
||||
}
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -1500,6 +1514,81 @@ const synthGradium = async(logger, {
|
||||
|
||||
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
||||
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
||||
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
|
||||
render. format:pcm + sample_rate:8000 returns bare little-endian 16-bit samples,
|
||||
which is exactly the r8 container. we avoid fish's wav output because its RIFF
|
||||
header carries a placeholder size (0xffffff24) — length is unknown up front, as
|
||||
with gradium.
|
||||
|
||||
fish is a voice-cloning vendor: the "voice" is a reference_id returned by
|
||||
POST /model, and omitting it entirely synthesizes with fish's default voice.
|
||||
the sentinel value 'default' (the bundled fallback entry in the portal) means
|
||||
exactly that — send no reference_id.
|
||||
*/
|
||||
const synthFishaudio = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id, fishaudio_tts_uri} = credentials;
|
||||
const {reference_id, latency, chunk_length, speed, volume} = options || {};
|
||||
|
||||
/* free-text reference_id in the vendor options wins over the voice selector */
|
||||
const refId = reference_id || (voice && voice !== 'default' ? voice : null);
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=fishaudio';
|
||||
params += `,voice=${refId || 'default'}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (model_id) params += `,model_id=${model_id}`;
|
||||
if (latency) params += `,latency=${latency}`;
|
||||
if (chunk_length) params += `,chunk_length=${chunk_length}`;
|
||||
if (speed) params += `,speed=${speed}`;
|
||||
if (volume) params += `,volume=${volume}`;
|
||||
if (fishaudio_tts_uri) params += `,endpoint=${fishaudio_tts_uri}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const post = bent(fishaudio_tts_uri || 'https://api.fish.audio', 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${api_key}`,
|
||||
'Content-Type': 'application/json',
|
||||
/* the model is selected by header, not in the body */
|
||||
'model': model_id || 's2.1-pro'
|
||||
});
|
||||
const audioContent = await post('/v1/tts', {
|
||||
text,
|
||||
format: 'pcm',
|
||||
sample_rate: sampleRate,
|
||||
...(refId && {reference_id: refId}),
|
||||
...(latency && {latency}),
|
||||
...(chunk_length && {chunk_length: parseInt(chunk_length, 10)}),
|
||||
...((speed || volume) && {prosody: {
|
||||
...(speed && {speed: parseFloat(speed)}),
|
||||
...(volume && {volume: parseFloat(volume)})
|
||||
}})
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'r8',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth fishaudio returned error');
|
||||
stats.increment('tts.count', ['vendor:fishaudio', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthNineninesix = async(logger, {
|
||||
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
|
||||
Generated
+2
-14
@@ -1,15 +1,14 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.14",
|
||||
"version": "1.0.18",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.14",
|
||||
"version": "1.0.18",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"23": "^0.0.0",
|
||||
"@aws-sdk/client-polly": "^3.496.0",
|
||||
"@aws-sdk/client-sts": "^3.496.0",
|
||||
"@cartesia/cartesia-js": "^2.2.7",
|
||||
@@ -2444,12 +2443,6 @@
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/23": {
|
||||
"version": "0.0.0",
|
||||
"resolved": "https://registry.npmjs.org/23/-/23-0.0.0.tgz",
|
||||
"integrity": "sha512-uAETf9Okr72trtp1pNXYKhFCTTI1EKGcYMA8gw3jLGhlbaDX+grrNToEWrpt8luxRAvrZWBTvyB5wk3PpNjGQQ==",
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/abort-controller": {
|
||||
"version": "3.0.0",
|
||||
"resolved": "https://registry.npmjs.org/abort-controller/-/abort-controller-3.0.0.tgz",
|
||||
@@ -7367,11 +7360,6 @@
|
||||
}
|
||||
},
|
||||
"dependencies": {
|
||||
"23": {
|
||||
"version": "0.0.0",
|
||||
"resolved": "https://registry.npmjs.org/23/-/23-0.0.0.tgz",
|
||||
"integrity": "sha512-uAETf9Okr72trtp1pNXYKhFCTTI1EKGcYMA8gw3jLGhlbaDX+grrNToEWrpt8luxRAvrZWBTvyB5wk3PpNjGQQ=="
|
||||
},
|
||||
"@aashutoshrathi/word-wrap": {
|
||||
"version": "1.2.6",
|
||||
"resolved": "https://registry.npmjs.org/@aashutoshrathi/word-wrap/-/word-wrap-1.2.6.tgz",
|
||||
|
||||
+1
-2
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.14",
|
||||
"version": "1.0.18",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -25,7 +25,6 @@
|
||||
},
|
||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||
"dependencies": {
|
||||
"23": "^0.0.0",
|
||||
"@aws-sdk/client-polly": "^3.496.0",
|
||||
"@aws-sdk/client-sts": "^3.496.0",
|
||||
"@cartesia/cartesia-js": "^2.2.7",
|
||||
|
||||
+105
-1
@@ -1087,6 +1087,47 @@ test('gradium speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('fishaudio speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.FISHAUDIO_API_KEY) {
|
||||
t.pass('skipping fishaudio speech synth tests since FISHAUDIO_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||
try {
|
||||
/* voice 'default' means "send no reference_id" — fish's own default voice */
|
||||
const o = await synthAudio(stats, {
|
||||
vendor: 'fishaudio',
|
||||
credentials: {
|
||||
api_key: process.env.FISHAUDIO_API_KEY,
|
||||
model_id: 's2.1-pro'
|
||||
},
|
||||
voice: 'default',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!o.servedFromCache, `successfully synthed fishaudio audio to ${o.filePath}`);
|
||||
|
||||
/* the cache render must be raw 8k pcm (r8): fish's wav header carries a
|
||||
placeholder RIFF size, so we never ask for wav */
|
||||
const o2 = await synthAudio(stats, {
|
||||
vendor: 'fishaudio',
|
||||
credentials: {api_key: process.env.FISHAUDIO_API_KEY},
|
||||
voice: 'default',
|
||||
text: text + ' two',
|
||||
renderForCaching: true,
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(!o2.servedFromCache, 'fishaudio synthed a second uncached render');
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('nineninesix speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
@@ -1104,7 +1145,7 @@ test('nineninesix speech synth tests', async(t) => {
|
||||
model_id: 'gepard-1.0'
|
||||
},
|
||||
language: 'en',
|
||||
voice: '3ad7a827-7fd1-4954-bf35-47d4cc33d9ed',
|
||||
voice: '775e4dfd-819c-4325-92ec-250f487ef7e3',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
@@ -1147,6 +1188,69 @@ test('inworld speech synth', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('inworld streaming say: params', async(t) => {
|
||||
/* This test asserts the streaming say: path, so it must run with streaming
|
||||
enabled. The Google non-streaming test above sets
|
||||
JAMBONES_DISABLE_TTS_STREAMING and, on its no-credentials skip path,
|
||||
deletes the env var WITHOUT clearing the require cache — so lib/config can
|
||||
still be holding 'true' by the time we get here. Re-require to be
|
||||
independent of what ran before us.
|
||||
*/
|
||||
delete process.env.JAMBONES_DISABLE_TTS_STREAMING;
|
||||
delete require.cache[require.resolve('../lib/config')];
|
||||
delete require.cache[require.resolve('../lib/synth-audio')];
|
||||
delete require.cache[require.resolve('..')];
|
||||
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.INWORLD_API_KEY) {
|
||||
t.pass('skipping inworld streaming say: param tests since INWORLD_API_KEY is not provided');
|
||||
client.quit();
|
||||
return t.end();
|
||||
}
|
||||
|
||||
try {
|
||||
let result = await synthAudio(stats, {
|
||||
vendor: 'inworld',
|
||||
credentials: {api_key: process.env.INWORLD_API_KEY, model_id: 'inworld-tts-1.5-mini'},
|
||||
language: 'en',
|
||||
voice: 'Ashley',
|
||||
text: 'This is a test of inworld streaming.',
|
||||
options: {temperature: 0.9, audioConfig: {pitch: 2.5, speakingRate: 1.2}},
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(result.filePath.startsWith('say:'), 'inworld returns streaming say: path');
|
||||
t.ok(result.filePath.includes('vendor=inworld'), 'streaming path contains vendor=inworld');
|
||||
t.ok(result.filePath.includes('voice=Ashley'), 'streaming path contains voice');
|
||||
t.ok(result.filePath.includes('model_id=inworld-tts-1.5-mini'), 'streaming path contains model_id');
|
||||
t.ok(result.filePath.includes('temperature=0.9'), 'streaming path contains temperature');
|
||||
/* pitch and speakingRate are nested under audioConfig; they used to be read
|
||||
from the top level and emitted as "undefined"
|
||||
*/
|
||||
t.ok(result.filePath.includes('speakingRate=1.2'), 'audioConfig.speakingRate reaches the say: params');
|
||||
t.ok(result.filePath.includes('pitch=2.5'), 'audioConfig.pitch reaches the say: params');
|
||||
t.ok(!result.filePath.includes('undefined'), 'no undefined values in the say: params');
|
||||
|
||||
/* options omitted entirely: no stray keys */
|
||||
result = await synthAudio(stats, {
|
||||
vendor: 'inworld',
|
||||
credentials: {api_key: process.env.INWORLD_API_KEY, model_id: 'inworld-tts-1.5-mini'},
|
||||
language: 'en',
|
||||
voice: 'Ashley',
|
||||
text: 'This is a test of inworld streaming.',
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(!result.filePath.includes('speakingRate='), 'speakingRate omitted when unset');
|
||||
t.ok(!result.filePath.includes('pitch='), 'pitch omitted when unset');
|
||||
t.ok(!result.filePath.includes('undefined'), 'no undefined values when options are omitted');
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('resemble speech synth', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
Reference in New Issue
Block a user