mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b172caee49 | ||
|
|
58374ca0fa | ||
|
|
5a7e0d37f4 | ||
|
|
6fa68bc712 | ||
|
|
d914b26cac | ||
|
|
7c7be6bbb1 | ||
|
|
46127bb763 | ||
|
|
17305325ff | ||
|
|
0e883cf82a | ||
|
|
2f3a766713 | ||
|
|
b429645ec2 | ||
|
|
0becf33bab | ||
|
|
3d3875741c |
+10
-5
@@ -1,5 +1,5 @@
|
||||
const assert = require('assert');
|
||||
const {noopLogger, createNuanceClient} = require('./utils');
|
||||
const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils');
|
||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
|
||||
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
||||
@@ -21,15 +21,20 @@ const getIbmVoices = async(client, logger, credentials) => {
|
||||
};
|
||||
|
||||
const getNuanceVoices = async(client, logger, credentials) => {
|
||||
const {client_id: clientId, secret: secret} = credentials;
|
||||
const {client_id: clientId, secret: secret, nuance_tts_uri} = credentials;
|
||||
|
||||
return new Promise(async(resolve, reject) => {
|
||||
/* get a nuance access token */
|
||||
let token, nuanceClient;
|
||||
try {
|
||||
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts');
|
||||
token = access_token.access_token;
|
||||
nuanceClient = await createNuanceClient(token);
|
||||
if (nuance_tts_uri) {
|
||||
nuanceClient = await createKryptonClient(nuance_tts_uri);
|
||||
}
|
||||
else {
|
||||
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts');
|
||||
token = access_token.access_token;
|
||||
nuanceClient = await createNuanceClient(token);
|
||||
}
|
||||
} catch (err) {
|
||||
logger.error({err}, 'getTtsVoices: error retrieving access token');
|
||||
return reject(err);
|
||||
|
||||
+59
-17
@@ -16,7 +16,7 @@ const {
|
||||
CancellationDetails,
|
||||
SpeechSynthesisOutputFormat
|
||||
} = sdk;
|
||||
const {makeSynthKey, createNuanceClient, noopLogger, createRivaClient} = require('./utils');
|
||||
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
|
||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||
const {
|
||||
SynthesisRequest,
|
||||
@@ -32,7 +32,7 @@ const {
|
||||
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
|
||||
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
|
||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
const EXPIRES = process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 3600 * 24; // cache tts for 24 hours
|
||||
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 24 * 60) * 60; // cache tts for 24 hours
|
||||
const TMP_FOLDER = '/tmp';
|
||||
|
||||
/**
|
||||
@@ -75,8 +75,10 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
}
|
||||
else if ('nuance' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nuance is used');
|
||||
assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used');
|
||||
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used');
|
||||
if (!credentials.nuance_tts_uri) {
|
||||
assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used');
|
||||
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used');
|
||||
}
|
||||
}
|
||||
else if ('nvidia' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
|
||||
@@ -155,7 +157,8 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
|
||||
break;
|
||||
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
|
||||
audioBuffer = await synthCustomVendor(logger, {credentials, stats, language, voice, text});
|
||||
({ audioBuffer, filePath } = await synthCustomVendor(logger,
|
||||
{credentials, stats, language, voice, text, filePath}));
|
||||
break;
|
||||
default:
|
||||
assert(`synthAudio: unsupported speech vendor ${vendor}`);
|
||||
@@ -183,7 +186,14 @@ async function synthAudio(client, logger, stats, { account_sid,
|
||||
|
||||
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => {
|
||||
try {
|
||||
const polly = new PollyClient(credentials);
|
||||
const {region, accessKeyId, secretAccessKey} = credentials;
|
||||
const polly = new PollyClient({
|
||||
region,
|
||||
credentials: {
|
||||
accessKeyId,
|
||||
secretAccessKey
|
||||
}
|
||||
});
|
||||
const opts = {
|
||||
Engine: engine,
|
||||
OutputFormat: 'mp3',
|
||||
@@ -363,10 +373,16 @@ const synthWellSaid = async(logger, {credentials, stats, language, voice, gender
|
||||
};
|
||||
|
||||
const synthNuance = async(client, logger, {credentials, stats, voice, model, text}) => {
|
||||
/* get a nuance access token */
|
||||
const {client_id, secret} = credentials;
|
||||
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts');
|
||||
const nuanceClient = await createNuanceClient(access_token);
|
||||
let nuanceClient;
|
||||
const {client_id, secret, nuance_tts_uri} = credentials;
|
||||
if (nuance_tts_uri) {
|
||||
nuanceClient = await createKryptonClient(nuance_tts_uri);
|
||||
}
|
||||
else {
|
||||
/* get a nuance access token */
|
||||
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts');
|
||||
nuanceClient = await createNuanceClient(access_token);
|
||||
}
|
||||
|
||||
const v = new Voice();
|
||||
const p = new AudioParameters();
|
||||
@@ -440,30 +456,56 @@ const synthNvidia = async(client, logger, {credentials, stats, language, voice,
|
||||
};
|
||||
|
||||
|
||||
// CustomVendor accept only mp3
|
||||
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text}) => {
|
||||
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text, filePath}) => {
|
||||
const {vendor, auth_token, custom_tts_url} = credentials;
|
||||
|
||||
try {
|
||||
const post = bent('POST', 'buffer', {
|
||||
const post = bent('POST', {
|
||||
'Authorization': `Bearer ${auth_token}`,
|
||||
'Accept': 'audio/mpeg',
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
|
||||
const mp3 = await post(custom_tts_url, {
|
||||
const response = await post(custom_tts_url, {
|
||||
language,
|
||||
format: 'audio/mpeg',
|
||||
voice,
|
||||
type: text.startsWith('<speak>') ? 'ssml' : 'text',
|
||||
text
|
||||
});
|
||||
|
||||
return mp3;
|
||||
const regex = /\.[^\.]*$/g;
|
||||
const mime = response.headers['content-type'];
|
||||
const buffer = await response.arrayBuffer();
|
||||
return {
|
||||
audioBuffer: buffer,
|
||||
filePath: filePath.replace(regex, getFileExtFromMime(mime))
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, `Vendor ${vendor} returned error`);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const getFileExtFromMime = (mime) => {
|
||||
switch (mime) {
|
||||
case 'audio/wav':
|
||||
case 'audio/x-wav':
|
||||
return '.wav';
|
||||
case /audio\/l16.*rate=8000/.test(mime) ? mime : 'cant match value':
|
||||
return '.r8';
|
||||
case /audio\/l16.*rate=16000/.test(mime) ? mime : 'cant match value':
|
||||
return '.r16';
|
||||
case /audio\/l16.*rate=24000/.test(mime) ? mime : 'cant match value':
|
||||
return '.r24';
|
||||
case /audio\/l16.*rate=32000/.test(mime) ? mime : 'cant match value':
|
||||
return '.r32';
|
||||
case /audio\/l16.*rate=48000/.test(mime) ? mime : 'cant match value':
|
||||
return '.r48';
|
||||
case 'audio/mpeg':
|
||||
case 'audio/mp3':
|
||||
return '.mp3';
|
||||
default:
|
||||
return '.wav';
|
||||
}
|
||||
};
|
||||
|
||||
module.exports = synthAudio;
|
||||
|
||||
@@ -77,6 +77,11 @@ const getNuanceAccessToken = async(clientId, secret, scope = 'asr tts') => {
|
||||
return json.access_token;
|
||||
};
|
||||
|
||||
const createKryptonClient = async(uri) => {
|
||||
const client = new SynthesizerClient(uri, grpc.credentials.createInsecure());
|
||||
return client;
|
||||
};
|
||||
|
||||
const createNuanceClient = async(access_token) => {
|
||||
|
||||
//if (nuanceClientMap.has(access_token)) return nuanceClientMap.get(access_token);
|
||||
@@ -108,6 +113,7 @@ module.exports = {
|
||||
makeIbmKey,
|
||||
getNuanceAccessToken,
|
||||
createNuanceClient,
|
||||
createKryptonClient,
|
||||
createRivaClient,
|
||||
makeBasicAuthHeader,
|
||||
NUANCE_AUTH_ENDPOINT,
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.2",
|
||||
"version": "0.0.7",
|
||||
"lockfileVersion": 2,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.2",
|
||||
"version": "0.0.7",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.276.0",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.2",
|
||||
"version": "0.0.7",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
|
||||
+35
-4
@@ -12,12 +12,12 @@ const stats = {
|
||||
histogram: () => {}
|
||||
};
|
||||
|
||||
test('Nuance tests', async(t) => {
|
||||
test('Nuance hosted tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {client, getTtsVoices} = fn(opts, logger);
|
||||
|
||||
if (!process.env.NUANCE_CLIENT_ID || !process.env.NUANCE_SECRET ) {
|
||||
t.pass('skipping Nuance test since no Nuance client_id and secret provided');
|
||||
t.pass('skipping Nuance hosted test since no Nuance client_id and secret provided');
|
||||
t.end();
|
||||
client.quit();
|
||||
return;
|
||||
@@ -31,8 +31,39 @@ test('Nuance tests', async(t) => {
|
||||
}
|
||||
};
|
||||
let voices = await getTtsVoices(opts);
|
||||
//console.log(`received ${voices.length} voices from Nuance`);
|
||||
//console.log(JSON.stringify(voices));
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||
|
||||
await client.flushallAsync();
|
||||
|
||||
t.end();
|
||||
|
||||
}
|
||||
catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('Nuance on-prem tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {client, getTtsVoices} = fn(opts, logger);
|
||||
|
||||
if (!process.env.NUANCE_TTS_URI ) {
|
||||
t.pass('skipping Nuance on-prem test since no Nuance uri provided');
|
||||
t.end();
|
||||
client.quit();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const opts = {
|
||||
vendor: 'nuance',
|
||||
credentials: {
|
||||
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||
}
|
||||
};
|
||||
let voices = await getTtsVoices(opts);
|
||||
t.ok(voices.length > 0 && voices[0].language,
|
||||
`GetVoices: successfully retrieved ${voices.length} voices from Nuance`);
|
||||
|
||||
|
||||
+38
-2
@@ -212,7 +212,7 @@ test('Azure custom voice speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('Nuance speech synth tests', async(t) => {
|
||||
test('Nuance hosted speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
@@ -251,6 +251,43 @@ test('Nuance speech synth tests', async(t) => {
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('Nuance on-prem speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
|
||||
if (!process.env.NUANCE_TTS_URI) {
|
||||
t.pass('skipping Nuance on prem speech synth tests since NUANCE_TTS_URI not provided');
|
||||
return t.end();
|
||||
}
|
||||
try {
|
||||
let opts = await synthAudio(stats, {
|
||||
vendor: 'nuance',
|
||||
credentials: {
|
||||
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'Evan',
|
||||
text: 'This is a test of on-prem. This is only a test',
|
||||
});
|
||||
t.ok(!opts.servedFromCache, `successfully synthesized nuance audio to ${opts.filePath}`);
|
||||
|
||||
opts = await synthAudio(stats, {
|
||||
vendor: 'nuance',
|
||||
credentials: {
|
||||
nuance_tts_uri: process.env.NUANCE_TTS_URI
|
||||
},
|
||||
language: 'en-US',
|
||||
voice: 'Evan',
|
||||
text: 'This is a test of on-prem. This is only a test',
|
||||
});
|
||||
t.ok(opts.servedFromCache, `successfully retrieved nuance audio from cache ${opts.filePath}`);
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
t.end(err);
|
||||
}
|
||||
client.quit();
|
||||
});
|
||||
|
||||
test('Nvidia speech synth tests', async(t) => {
|
||||
const fn = require('..');
|
||||
const {synthAudio, client} = fn(opts, logger);
|
||||
@@ -348,7 +385,6 @@ test('Custom Vendor speech synth tests', async(t) => {
|
||||
let obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew`);
|
||||
t.ok(obj.headers.Authorization == 'Bearer some_jwt_token', 'Custom Vendor Authentication Header is correct');
|
||||
t.ok(obj.body.language == 'en-US', 'Custom Vendor Language is correct');
|
||||
t.ok(obj.body.format == 'audio/mpeg', 'Custom Vendor format is correct');
|
||||
t.ok(obj.body.voice == 'English-US.Female-1', 'Custom Vendor voice is correct');
|
||||
t.ok(obj.body.type == 'text', 'Custom Vendor type is correct');
|
||||
t.ok(obj.body.text == 'This is a test. This is only a test', 'Custom Vendor text is correct');
|
||||
|
||||
Reference in New Issue
Block a user