mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-04 07:43:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
18917f49ff | ||
|
|
fa705694e1 | ||
|
|
89fe0b87e9 | ||
|
|
b928820906 | ||
|
|
f4c3c7dc8b | ||
|
|
a32f71ac5a | ||
|
|
c8850998b5 |
@@ -7,11 +7,18 @@ on:
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
id-token: write # required to request the GitHub OIDC token for AWS
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '20'
|
||||
- uses: aws-actions/configure-aws-credentials@v4
|
||||
with:
|
||||
role-to-assume: ${{ secrets.AWS_ROLE_ARN }}
|
||||
aws-region: us-east-1
|
||||
- run: npm install
|
||||
- run: npm run jslint
|
||||
- run: sudo apt update && sudo apt install -y squid
|
||||
@@ -19,10 +26,12 @@ jobs:
|
||||
- run: sudo systemctl start squid
|
||||
- run: npm test
|
||||
env:
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }}
|
||||
AWS_REGION: ${{ secrets.AWS_REGION }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
|
||||
# AWS_* are exported by configure-aws-credentials above as short-lived
|
||||
# OIDC credentials; no AWS keys are stored as repository secrets.
|
||||
GCP_JSON_KEY: ${{ secrets.GCP_JSON_KEY }}
|
||||
# Enables the "AWS speech synth tests by RoleArn" test, which exercises the
|
||||
# AssumeRole credential path used in production.
|
||||
AWS_ROLE_ARN: ${{ secrets.AWS_ROLE_ARN }}
|
||||
|
||||
MICROSOFT_API_KEY: ${{ secrets.MICROSOFT_API_KEY }}
|
||||
MICROSOFT_REGION: ${{ secrets.MICROSOFT_REGION }}
|
||||
|
||||
+176
-2
@@ -80,8 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
logger = logger || noopLogger;
|
||||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
|
||||
'murf', 'xai']
|
||||
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld',
|
||||
'resemble', 'murf', 'xai', 'fishaudio']
|
||||
.includes(vendor) ||
|
||||
vendor.startsWith('custom'),
|
||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||
@@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
} else if ('gradium' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when gradium is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
|
||||
} else if ('kugelaudio' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when kugelaudio is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used');
|
||||
} else if ('nineninesix' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nineninesix is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when nineninesix is used');
|
||||
@@ -149,6 +152,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
} else if (vendor === 'resemble') {
|
||||
assert.ok(voice, 'synthAudio requires voice when resemble is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when resemble is used');
|
||||
} else if ('fishaudio' === vendor) {
|
||||
/* no voice assert: fish synthesizes with its own default voice when
|
||||
reference_id is omitted, which is what the 'default' selection means */
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when fishaudio is used');
|
||||
}
|
||||
|
||||
const key = makeSynthKey({
|
||||
@@ -220,11 +227,21 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'fishaudio':
|
||||
audioData = await synthFishaudio(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'gradium':
|
||||
audioData = await synthGradium(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'kugelaudio':
|
||||
audioData = await synthKugelaudio(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'nineninesix':
|
||||
audioData = await synthNineninesix(logger, {
|
||||
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
@@ -1503,8 +1520,165 @@ const synthGradium = async(logger, {
|
||||
}
|
||||
};
|
||||
|
||||
/* kugelaudio — json websocket (/ws/tts/stream) for streaming, and POST /v1/tts/generate
|
||||
for the cache render. the POST streams back bare little-endian 16-bit samples at the
|
||||
requested sample_rate, which is exactly the r8 container at 8000.
|
||||
|
||||
voices are numeric ids (or public handles). language is an ISO 639-1 code that drives
|
||||
text normalization; jambonz carries BCP-47, so only the primary subtag is sent. the
|
||||
api rejects codes outside its list, so an unsupported or unset language is omitted
|
||||
and the voice's own language applies. options.api_uri pins a region
|
||||
(e.g. api.eu.kugelaudio.com).
|
||||
*/
|
||||
const KUGELAUDIO_LANGUAGES = ['ar', 'bg', 'bn', 'cs', 'da', 'de', 'el', 'en', 'es', 'fa', 'fi', 'fr', 'he', 'hi',
|
||||
'hr', 'hu', 'id', 'it', 'ja', 'ko', 'ms', 'nl', 'no', 'pl', 'pt', 'ro', 'ru', 'sk', 'sl', 'sr', 'sv', 'ta', 'th',
|
||||
'tr', 'uk', 'ur', 'vi', 'yue', 'zh'];
|
||||
const synthKugelaudio = async(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id} = credentials;
|
||||
const {api_uri, speed, cfg_scale, temperature, normalize, project_id, dictionary_ids} = options || {};
|
||||
const isSet = (v) => v !== null && v !== undefined;
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=kugelaudio';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
params += `,model_id=${model_id || 'kugel-3'}`;
|
||||
if (language) params += `,language=${language}`;
|
||||
if (api_uri) params += `,api_uri=${api_uri}`;
|
||||
if (isSet(speed)) params += `,speed=${speed}`;
|
||||
if (isSet(cfg_scale)) params += `,cfg_scale=${cfg_scale}`;
|
||||
if (isSet(temperature)) params += `,temperature=${temperature}`;
|
||||
if (isSet(normalize)) params += `,normalize=${normalize}`;
|
||||
if (isSet(project_id)) params += `,project_id=${project_id}`;
|
||||
/* the say: param parser is bracket-aware, so the json array survives intact */
|
||||
if (Array.isArray(dictionary_ids)) params += `,dictionary_ids=${JSON.stringify(dictionary_ids)}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const host = (api_uri || 'api.kugelaudio.com').replace(/^[a-z]+:\/\//, '').replace(/\/$/, '');
|
||||
const post = bent(`https://${host}`, 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${api_key}`,
|
||||
'Content-Type': 'application/json; charset=utf-8'
|
||||
});
|
||||
const voiceId = /^\d+$/.test(`${voice}`) ? Number(voice) : voice;
|
||||
const lang = language && language.split('-')[0].toLowerCase();
|
||||
const audioContent = await post('/v1/tts/generate', {
|
||||
text,
|
||||
voice_id: voiceId,
|
||||
model_id: model_id || 'kugel-3',
|
||||
sample_rate: sampleRate,
|
||||
...(KUGELAUDIO_LANGUAGES.includes(lang) && {language: lang}),
|
||||
...(isSet(speed) && {speed: Number(speed)}),
|
||||
...(isSet(cfg_scale) && {cfg_scale: Number(cfg_scale)}),
|
||||
...(isSet(temperature) && {temperature: Number(temperature)}),
|
||||
...(isSet(normalize) && {normalize: normalize === true || normalize === 'true'}),
|
||||
...(isSet(project_id) && {project_id: Number(project_id)}),
|
||||
...(Array.isArray(dictionary_ids) && {dictionary_ids})
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'r8',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth kugelaudio returned error');
|
||||
stats.increment('tts.count', ['vendor:kugelaudio', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
|
||||
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
|
||||
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
|
||||
render. format:pcm + sample_rate:8000 returns bare little-endian 16-bit samples,
|
||||
which is exactly the r8 container. we avoid fish's wav output because its RIFF
|
||||
header carries a placeholder size (0xffffff24) — length is unknown up front, as
|
||||
with gradium.
|
||||
|
||||
fish is a voice-cloning vendor: the "voice" is a reference_id returned by
|
||||
POST /model, and omitting it entirely synthesizes with fish's default voice.
|
||||
the sentinel value 'default' (the bundled fallback entry in the portal) means
|
||||
exactly that — send no reference_id.
|
||||
*/
|
||||
const synthFishaudio = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, model_id, fishaudio_tts_uri} = credentials;
|
||||
const {reference_id, latency, chunk_length, speed, volume} = options || {};
|
||||
|
||||
/* free-text reference_id in the vendor options wins over the voice selector */
|
||||
const refId = reference_id || (voice && voice !== 'default' ? voice : null);
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=fishaudio';
|
||||
params += `,voice=${refId || 'default'}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
if (model_id) params += `,model_id=${model_id}`;
|
||||
if (latency) params += `,latency=${latency}`;
|
||||
if (chunk_length) params += `,chunk_length=${chunk_length}`;
|
||||
if (speed) params += `,speed=${speed}`;
|
||||
if (volume) params += `,volume=${volume}`;
|
||||
if (fishaudio_tts_uri) params += `,endpoint=${fishaudio_tts_uri}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const sampleRate = 8000;
|
||||
const post = bent(fishaudio_tts_uri || 'https://api.fish.audio', 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${api_key}`,
|
||||
'Content-Type': 'application/json',
|
||||
/* the model is selected by header, not in the body */
|
||||
'model': model_id || 's2.1-pro'
|
||||
});
|
||||
const audioContent = await post('/v1/tts', {
|
||||
text,
|
||||
format: 'pcm',
|
||||
sample_rate: sampleRate,
|
||||
...(refId && {reference_id: refId}),
|
||||
...(latency && {latency}),
|
||||
...(chunk_length && {chunk_length: parseInt(chunk_length, 10)}),
|
||||
...((speed || volume) && {prosody: {
|
||||
...(speed && {speed: parseFloat(speed)}),
|
||||
...(volume && {volume: parseFloat(volume)})
|
||||
}})
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'r8',
|
||||
sampleRate
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth fishaudio returned error');
|
||||
stats.increment('tts.count', ['vendor:fishaudio', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthNineninesix = async(logger, {
|
||||
credentials, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.17",
|
||||
"version": "1.0.19",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.17",
|
||||
"version": "1.0.19",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.496.0",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "1.0.17",
|
||||
"version": "1.0.19",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
/**
|
||||
* Resolve AWS credentials for the test suite.
|
||||
*
|
||||
* Returns null when AWS is not configured, so the caller skips.
|
||||
*
|
||||
* When AWS_SESSION_TOKEN is set the credentials are temporary -- GitHub OIDC in CI, or
|
||||
* `aws sso login` locally -- and the key pair must not be passed through. Doing so sends
|
||||
* lib/get-aws-sts-token.js down its accessKeyId branch, which calls GetSessionToken, and
|
||||
* AWS rejects GetSessionToken when it is called with session credentials. Returning the
|
||||
* region alone routes to the SDK's default credential provider chain instead, which
|
||||
* handles temporary credentials correctly.
|
||||
*/
|
||||
module.exports = () => {
|
||||
const region = process.env.AWS_REGION;
|
||||
if (!region) return null;
|
||||
|
||||
const accessKeyId = process.env.AWS_ACCESS_KEY_ID;
|
||||
const secretAccessKey = process.env.AWS_SECRET_ACCESS_KEY;
|
||||
|
||||
if (process.env.AWS_SESSION_TOKEN) return {region};
|
||||
if (accessKeyId && secretAccessKey) return {accessKeyId, secretAccessKey, region};
|
||||
return {region};
|
||||
};
|
||||
+11
-11
@@ -6,33 +6,33 @@ process.on('unhandledRejection', (reason, p) => {
|
||||
console.log('Unhandled Rejection at: Promise', p, 'reason:', reason);
|
||||
});
|
||||
|
||||
const awsCredentials = require('./aws-credentials');
|
||||
|
||||
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
test('AWS - create and cache auth token', async(t) => {
|
||||
const fn = require('..');
|
||||
const {client, getAwsAuthToken} = fn(opts, logger);
|
||||
|
||||
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
|
||||
const credentials = awsCredentials();
|
||||
if (!credentials) {
|
||||
t.pass('skipping AWS auth token tests since no AWS credentials provided');
|
||||
t.end();
|
||||
client.quit();
|
||||
return;
|
||||
}
|
||||
// getAwsAuthToken derives its cache key from roleArn || accessKeyId || speech_credential_sid.
|
||||
// With temporary credentials none of the first two are passed, so supply a stable sid --
|
||||
// which is what production does for instance-profile credentials.
|
||||
const args = {...credentials, speech_credential_sid: 'test-aws-speech-credential'};
|
||||
|
||||
try {
|
||||
let obj = await getAwsAuthToken({
|
||||
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
|
||||
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
|
||||
region: process.env.AWS_REGION
|
||||
});
|
||||
let obj = await getAwsAuthToken(args);
|
||||
//console.log({obj}, 'received auth token from AWS');
|
||||
t.ok(obj.securityToken && !obj.servedFromCache, 'successfullY generated auth token from AWS');
|
||||
|
||||
await sleep(250);
|
||||
obj = await getAwsAuthToken({
|
||||
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
|
||||
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
|
||||
region: process.env.AWS_REGION
|
||||
});
|
||||
obj = await getAwsAuthToken(args);
|
||||
//console.log({obj}, 'received auth token from AWS - second request');
|
||||
t.ok(obj.securityToken && obj.servedFromCache, 'successfully received access token from cache');
|
||||
|
||||
|
||||
+5
-7
@@ -1,4 +1,5 @@
|
||||
const test = require('tape').test ;
|
||||
const awsCredentials = require('./aws-credentials');
|
||||
const config = require('config');
|
||||
const opts = config.get('redis');
|
||||
const fs = require('fs');
|
||||
@@ -45,18 +46,15 @@ test('AWS tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {client, getTtsVoices} = fn(opts, logger);
|
||||
|
||||
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
|
||||
t.pass('skipping AWS speech synth tests since AWS_ACCESS_KEY_ID, AWS_SECRET_ACCESS_KEY, or AWS_REGION not provided');
|
||||
const credentials = awsCredentials();
|
||||
if (!credentials) {
|
||||
t.pass('skipping AWS speech synth tests since AWS_REGION not provided');
|
||||
return t.end();
|
||||
}
|
||||
try {
|
||||
const opts = {
|
||||
vendor: 'aws',
|
||||
credentials: {
|
||||
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
|
||||
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
|
||||
region: process.env.AWS_REGION,
|
||||
}
|
||||
credentials
|
||||
};
|
||||
let result = await getTtsVoices(opts);
|
||||
t.ok(result?.Voices?.length > 0, `GetVoices: successfully retrieved ${result.Voices.length} voices from AWS`);
|
||||
|
||||
+89
-15
@@ -1,4 +1,5 @@
|
||||
const test = require('tape').test;
|
||||
const awsCredentials = require('./aws-credentials');
|
||||
const config = require('config');
|
||||
const opts = config.get('redis');
|
||||
const fs = require('fs');
|
||||
@@ -668,18 +669,15 @@ test('AWS speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.AWS_ACCESS_KEY_ID || !process.env.AWS_SECRET_ACCESS_KEY || !process.env.AWS_REGION) {
|
||||
t.pass('skipping AWS speech synth tests since AWS_ACCESS_KEY_ID, AWS_SECRET_ACCESS_KEY, or AWS_REGION not provided');
|
||||
const credentials = awsCredentials();
|
||||
if (!credentials) {
|
||||
t.pass('skipping AWS speech synth tests since AWS_REGION not provided');
|
||||
return t.end();
|
||||
}
|
||||
try {
|
||||
let opts = await synthAudio(stats, {
|
||||
vendor: 'aws',
|
||||
credentials: {
|
||||
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
|
||||
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
|
||||
region: process.env.AWS_REGION,
|
||||
},
|
||||
credentials,
|
||||
language: 'en-US',
|
||||
voice: 'Joey',
|
||||
text: 'This is a test. This is only a test',
|
||||
@@ -689,11 +687,7 @@ test('AWS speech synth tests', async(t) => {
|
||||
|
||||
opts = await synthAudio(stats, {
|
||||
vendor: 'aws',
|
||||
credentials: {
|
||||
accessKeyId: process.env.AWS_ACCESS_KEY_ID,
|
||||
secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY,
|
||||
region: process.env.AWS_REGION,
|
||||
},
|
||||
credentials,
|
||||
language: 'en-US',
|
||||
voice: 'Joey',
|
||||
text: 'This is a test. This is only a test',
|
||||
@@ -724,9 +718,18 @@ test('AWS speech synth tests by RoleArn', async(t) => {
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'Joey',
|
||||
text: 'This is a test. This is only a test',
|
||||
// Distinct text on purpose: the 'AWS speech synth tests' above cache audio for
|
||||
// the same vendor/voice/language, and identical text would hit that cache entry,
|
||||
// making servedFromCache true and this assertion fail.
|
||||
text: 'This is a roleArn test. This is only a roleArn test',
|
||||
// Without this, synthPolly returns the mediajam streaming params string, which
|
||||
// embeds accessKeyId/secretAccessKey/sessionToken (see lib/synth-audio.js).
|
||||
// Every other synth test in this file renders for caching; this one did not,
|
||||
// so it printed live credentials into a public CI log.
|
||||
renderForCaching: true,
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized aws by roleArn audio to ${opts.filePath}`);
|
||||
// Deliberately does not interpolate opts.filePath -- it can carry credentials.
|
||||
t.ok(!opts.servedFromCache, 'successfully synthesized aws by roleArn audio');
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
@@ -1087,6 +1090,77 @@ test('gradium speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('kugelaudio speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.KUGELAUDIO_API_KEY) {
|
||||
t.pass('skipping kugelaudio speech synth tests since KUGELAUDIO_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Guten Tag und willkommen bei jambonz! Ihre Bestellung kostet 12,99 Euro. ' + Date.now();
|
||||
try {
|
||||
const opts = await synthAudio(stats, {
|
||||
vendor: 'kugelaudio',
|
||||
credentials: {
|
||||
api_key: process.env.KUGELAUDIO_API_KEY,
|
||||
model_id: 'kugel-3'
|
||||
},
|
||||
language: 'de-DE',
|
||||
voice: '1930',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthed kugelaudio audio to ${opts.filePath}`);
|
||||
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('fishaudio speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.FISHAUDIO_API_KEY) {
|
||||
t.pass('skipping fishaudio speech synth tests since FISHAUDIO_API_KEY is not provided');
|
||||
return t.end();
|
||||
}
|
||||
const text = 'Hi there and welcome to jambones! ' + Date.now();
|
||||
try {
|
||||
/* voice 'default' means "send no reference_id" — fish's own default voice */
|
||||
const o = await synthAudio(stats, {
|
||||
vendor: 'fishaudio',
|
||||
credentials: {
|
||||
api_key: process.env.FISHAUDIO_API_KEY,
|
||||
model_id: 's2.1-pro'
|
||||
},
|
||||
voice: 'default',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
t.ok(!o.servedFromCache, `successfully synthed fishaudio audio to ${o.filePath}`);
|
||||
|
||||
/* the cache render must be raw 8k pcm (r8): fish's wav header carries a
|
||||
placeholder RIFF size, so we never ask for wav */
|
||||
const o2 = await synthAudio(stats, {
|
||||
vendor: 'fishaudio',
|
||||
credentials: {api_key: process.env.FISHAUDIO_API_KEY},
|
||||
voice: 'default',
|
||||
text: text + ' two',
|
||||
renderForCaching: true,
|
||||
disableTtsCache: true
|
||||
});
|
||||
t.ok(!o2.servedFromCache, 'fishaudio synthed a second uncached render');
|
||||
} catch (err) {
|
||||
console.error(JSON.stringify(err));
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('nineninesix speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
@@ -1104,7 +1178,7 @@ test('nineninesix speech synth tests', async(t) => {
|
||||
model_id: 'gepard-1.0'
|
||||
},
|
||||
language: 'en',
|
||||
voice: '3ad7a827-7fd1-4954-bf35-47d4cc33d9ed',
|
||||
voice: '775e4dfd-819c-4325-92ec-250f487ef7e3',
|
||||
text,
|
||||
renderForCaching: true
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user