mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5dc26c916d | ||
|
|
12e3364461 | ||
|
|
6c67f6faa0 |
+87
-1
@@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
|
|
||||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld',
|
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld',
|
||||||
'resemble', 'murf', 'xai', 'fishaudio']
|
'resemble', 'murf', 'xai', 'fishaudio', 'speechify']
|
||||||
.includes(vendor) ||
|
.includes(vendor) ||
|
||||||
vendor.startsWith('custom'),
|
vendor.startsWith('custom'),
|
||||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||||
@@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
} else if ('gradium' === vendor) {
|
} else if ('gradium' === vendor) {
|
||||||
assert.ok(voice, 'synthAudio requires voice when gradium is used');
|
assert.ok(voice, 'synthAudio requires voice when gradium is used');
|
||||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
|
||||||
|
} else if ('speechify' === vendor) {
|
||||||
|
assert.ok(voice, 'synthAudio requires voice when speechify is used');
|
||||||
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when speechify is used');
|
||||||
} else if ('kugelaudio' === vendor) {
|
} else if ('kugelaudio' === vendor) {
|
||||||
assert.ok(voice, 'synthAudio requires voice when kugelaudio is used');
|
assert.ok(voice, 'synthAudio requires voice when kugelaudio is used');
|
||||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used');
|
assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used');
|
||||||
@@ -242,6 +245,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
|||||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
disableTtsCache});
|
disableTtsCache});
|
||||||
break;
|
break;
|
||||||
|
case 'speechify':
|
||||||
|
audioData = await synthSpeechify(logger, {
|
||||||
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
|
disableTtsCache});
|
||||||
|
break;
|
||||||
case 'nineninesix':
|
case 'nineninesix':
|
||||||
audioData = await synthNineninesix(logger, {
|
audioData = await synthNineninesix(logger, {
|
||||||
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
@@ -1602,6 +1610,84 @@ const synthKugelaudio = async(logger, {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* simba-3.2 is English only; with no model set, other languages get simba-3.0 */
|
||||||
|
const speechifyModel = (model_id, language) => {
|
||||||
|
if (model_id) return model_id;
|
||||||
|
return !language || /^en/i.test(language) ? 'simba-3.2' : 'simba-3.0';
|
||||||
|
};
|
||||||
|
const synthSpeechify = async(logger, {
|
||||||
|
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||||
|
disableTtsCache
|
||||||
|
}) => {
|
||||||
|
const {api_key, model_id} = credentials;
|
||||||
|
let credOptions = credentials.options || {};
|
||||||
|
if (typeof credOptions === 'string') {
|
||||||
|
try {
|
||||||
|
credOptions = JSON.parse(credOptions);
|
||||||
|
} catch {
|
||||||
|
credOptions = {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const {api_uri, loudness_normalization, text_normalization} = {...credOptions, ...options};
|
||||||
|
const isSet = (v) => v !== null && v !== undefined;
|
||||||
|
const model = speechifyModel(options?.model_id || model_id, language);
|
||||||
|
|
||||||
|
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||||
|
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '{';
|
||||||
|
params += `api_key=${api_key}`;
|
||||||
|
params += `,playback_id=${key}`;
|
||||||
|
params += ',vendor=speechify';
|
||||||
|
params += `,voice=${voice}`;
|
||||||
|
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||||
|
params += `,model_id=${model}`;
|
||||||
|
if (language) params += `,language=${language}`;
|
||||||
|
if (api_uri) params += `,api_uri=${api_uri}`;
|
||||||
|
if (isSet(loudness_normalization)) params += `,loudness_normalization=${loudness_normalization}`;
|
||||||
|
if (isSet(text_normalization)) params += `,text_normalization=${text_normalization}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
/* other pcm rates are mislabelled 24 kHz on workspaces pinned before 2026-09-30 */
|
||||||
|
const sampleRate = 24000;
|
||||||
|
const host = (api_uri || 'api.speechify.ai').replace(/^[a-z]+:\/\//, '').replace(/\/$/, '');
|
||||||
|
const post = bent(`https://${host}`, 'POST', 'buffer', {
|
||||||
|
'Authorization': `Bearer ${api_key}`,
|
||||||
|
'Content-Type': 'application/json',
|
||||||
|
'Speechify-Caller': 'jambonz'
|
||||||
|
});
|
||||||
|
const toBool = (v) => v === true || v === 'true';
|
||||||
|
const speechOptions = {
|
||||||
|
...(isSet(loudness_normalization) && {loudness_normalization: toBool(loudness_normalization)}),
|
||||||
|
...(isSet(text_normalization) && {text_normalization: toBool(text_normalization)})
|
||||||
|
};
|
||||||
|
const audioContent = await post('/v1/audio/stream', {
|
||||||
|
input: text,
|
||||||
|
voice_id: voice,
|
||||||
|
model,
|
||||||
|
output_format: `pcm_${sampleRate}`,
|
||||||
|
...(language && {language}),
|
||||||
|
...(Object.keys(speechOptions).length && {options: speechOptions})
|
||||||
|
});
|
||||||
|
return {
|
||||||
|
audioContent,
|
||||||
|
extension: 'r24',
|
||||||
|
sampleRate
|
||||||
|
};
|
||||||
|
} catch (err) {
|
||||||
|
logger.info({err}, 'synth speechify returned error');
|
||||||
|
stats.increment('tts.count', ['vendor:speechify', 'accepted:no']);
|
||||||
|
throw err;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
||||||
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
||||||
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
|
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
|
||||||
|
|||||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "1.0.19",
|
"version": "1.0.21",
|
||||||
"lockfileVersion": 2,
|
"lockfileVersion": 2,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "1.0.19",
|
"version": "1.0.21",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@aws-sdk/client-polly": "^3.496.0",
|
"@aws-sdk/client-polly": "^3.496.0",
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "1.0.19",
|
"version": "1.0.21",
|
||||||
"description": "TTS-related speech utilities for jambonz",
|
"description": "TTS-related speech utilities for jambonz",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"author": "Dave Horton",
|
"author": "Dave Horton",
|
||||||
|
|||||||
@@ -1120,6 +1120,35 @@ test('kugelaudio speech synth tests', async(t) => {
|
|||||||
client.quit();
|
client.quit();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('speechify speech synth tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.SPEECHIFY_API_KEY) {
|
||||||
|
t.pass('skipping speechify speech synth tests since SPEECHIFY_API_KEY is not provided');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
const text = 'Hi there and welcome to jambonz! ' + Date.now();
|
||||||
|
try {
|
||||||
|
const opts = await synthAudio(stats, {
|
||||||
|
vendor: 'speechify',
|
||||||
|
credentials: {
|
||||||
|
api_key: process.env.SPEECHIFY_API_KEY
|
||||||
|
},
|
||||||
|
language: 'en-US',
|
||||||
|
voice: 'geffen_32',
|
||||||
|
text,
|
||||||
|
renderForCaching: true
|
||||||
|
});
|
||||||
|
t.ok(!opts.servedFromCache, `successfully synthed speechify audio to ${opts.filePath}`);
|
||||||
|
|
||||||
|
} catch (err) {
|
||||||
|
console.error(JSON.stringify(err));
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
test('fishaudio speech synth tests', async(t) => {
|
test('fishaudio speech synth tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {synthAudio, client} = fn(opts, logger);
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|||||||
Reference in New Issue
Block a user