mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
chore: deprecate + remove verbio, nuance, playht speech vendor support (#144)
* chore: deprecate and remove verbio, nuance speech vendor support Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * chore: also deprecate and remove PlayHT speech vendor PlayHT was acquired and no longer provides the service. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
142323a151
commit
7d076bb8b4
@@ -1,49 +0,0 @@
|
||||
const formurlencoded = require('form-urlencoded');
|
||||
const {Pool} = require('undici');
|
||||
const pool = new Pool('https://auth.crt.nuance.com');
|
||||
const {makeNuanceKey, makeBasicAuthHeader, noopLogger} = require('./utils');
|
||||
const { HTTP_TIMEOUT } = require('./config');
|
||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
|
||||
async function getNuanceAccessToken(client, logger, clientId, secret, scope) {
|
||||
logger = logger || noopLogger;
|
||||
try {
|
||||
const key = makeNuanceKey(clientId, secret, scope);
|
||||
const access_token = await client.get(key);
|
||||
if (access_token) return {access_token, servedFromCache: true};
|
||||
|
||||
/* access token not found in cache, so fetch it from Nuance */
|
||||
const payload = {
|
||||
grant_type: 'client_credentials',
|
||||
scope
|
||||
};
|
||||
const auth = makeBasicAuthHeader(clientId, secret);
|
||||
const {statusCode, headers, body} = await pool.request({
|
||||
path: '/oauth2/token',
|
||||
method: 'POST',
|
||||
headers: {
|
||||
...auth,
|
||||
'Content-Type': 'application/x-www-form-urlencoded'
|
||||
},
|
||||
body: formurlencoded(payload),
|
||||
timeout: HTTP_TIMEOUT,
|
||||
followRedirects: false
|
||||
});
|
||||
|
||||
if (200 !== statusCode) {
|
||||
logger.debug({statusCode, headers, body: body.text()}, 'error fetching access token from Nuance');
|
||||
const err = new Error();
|
||||
err.statusCode = statusCode;
|
||||
throw err;
|
||||
}
|
||||
const json = await body.json();
|
||||
await client.set(key, json.access_token, 'EX', json.expires_in - 30);
|
||||
return {...json, servedFromCache: false};
|
||||
} catch (err) {
|
||||
debug(err, `getNuanceAccessToken: Error retrieving Nuance access token for client_id ${clientId}`);
|
||||
logger.error(err, `getNuanceAccessToken: Error retrieving Nuance access token for client_id ${clientId}`);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = getNuanceAccessToken;
|
||||
+2
-92
@@ -1,74 +1,8 @@
|
||||
const assert = require('assert');
|
||||
const {noopLogger, createNuanceClient, createKryptonClient} = require('./utils');
|
||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||
const getVerbioAccessToken = require('./get-verbio-token');
|
||||
const {GetVoicesRequest, Voice} = require('../stubs/nuance/synthesizer_pb');
|
||||
const {noopLogger} = require('./utils');
|
||||
const ttsGoogle = require('@google-cloud/text-to-speech');
|
||||
const { PollyClient, DescribeVoicesCommand } = require('@aws-sdk/client-polly');
|
||||
const getAwsAuthToken = require('./get-aws-sts-token');
|
||||
const {Pool} = require('undici');
|
||||
const { HTTP_TIMEOUT } = require('./config');
|
||||
const verbioVoicePool = new Pool('https://us.rest.speechcenter.verbio.com');
|
||||
|
||||
const getNuanceVoices = async(client, logger, credentials) => {
|
||||
const {client_id: clientId, secret: secret, nuance_tts_uri} = credentials;
|
||||
|
||||
return new Promise(async(resolve, reject) => {
|
||||
/* get a nuance access token */
|
||||
let token, nuanceClient;
|
||||
try {
|
||||
if (nuance_tts_uri) {
|
||||
nuanceClient = await createKryptonClient(nuance_tts_uri);
|
||||
}
|
||||
else {
|
||||
const access_token = await getNuanceAccessToken(client, logger, clientId, secret, 'tts');
|
||||
token = access_token.access_token;
|
||||
nuanceClient = await createNuanceClient(token);
|
||||
}
|
||||
} catch (err) {
|
||||
logger.error({err}, 'getTtsVoices: error retrieving access token');
|
||||
return reject(err);
|
||||
}
|
||||
/* retrieve all voices */
|
||||
const v = new Voice();
|
||||
const request = new GetVoicesRequest();
|
||||
request.setVoice(v);
|
||||
|
||||
nuanceClient.getVoices(request, (err, response) => {
|
||||
if (err) {
|
||||
logger.error({err, clientId, secret, token}, 'getTtsVoices: error retrieving voices');
|
||||
return reject(err);
|
||||
}
|
||||
|
||||
/* return all the voices that are not restricted and eliminate duplicates */
|
||||
const voices = response.getVoicesList()
|
||||
.map((v) => {
|
||||
return {
|
||||
language: v.getLanguage(),
|
||||
name: v.getName(),
|
||||
model: v.getModel(),
|
||||
gender: v.getGender() === 1 ? 'male' : 'female',
|
||||
restricted: v.getRestricted()
|
||||
};
|
||||
});
|
||||
const v = voices
|
||||
.filter((v) => v.restricted === false)
|
||||
.map((v) => {
|
||||
delete v.restricted;
|
||||
return v;
|
||||
})
|
||||
.sort((a, b) => {
|
||||
if (a.language < b.language) return -1;
|
||||
if (a.language > b.language) return 1;
|
||||
if (a.name < b.name) return -1;
|
||||
return 1;
|
||||
});
|
||||
const arr = [...new Set(v.map((v) => JSON.stringify(v)))]
|
||||
.map((v) => JSON.parse(v));
|
||||
resolve(arr);
|
||||
});
|
||||
});
|
||||
};
|
||||
|
||||
const getGoogleVoices = async(_client, logger, credentials) => {
|
||||
const client = new ttsGoogle.TextToSpeechClient({credentials});
|
||||
@@ -109,26 +43,6 @@ const getAwsVoices = async(_client, createHash, retrieveHash, logger, credential
|
||||
}
|
||||
};
|
||||
|
||||
const getVerbioVoices = async(client, logger, credentials) => {
|
||||
try {
|
||||
const access_token = await getVerbioAccessToken(client, logger, credentials);
|
||||
const { body} = await verbioVoicePool.request({
|
||||
path: '/api/v1/voices',
|
||||
method: 'GET',
|
||||
headers: {
|
||||
'Authorization': `Bearer ${access_token.access_token}`,
|
||||
'User-Agent': 'jambonz'
|
||||
},
|
||||
timeout: HTTP_TIMEOUT,
|
||||
followRedirects: false
|
||||
});
|
||||
return await body.json();
|
||||
} catch (err) {
|
||||
logger.info({err}, 'getVerbioVoices - failed to list voices for Verbio');
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Synthesize speech to an mp3 file, and also cache the generated speech
|
||||
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
|
||||
@@ -148,19 +62,15 @@ const getVerbioVoices = async(client, logger, credentials) => {
|
||||
async function getTtsVoices(client, createHash, retrieveHash, logger, {vendor, credentials}) {
|
||||
logger = logger || noopLogger;
|
||||
|
||||
assert.ok(['nuance', 'google', 'aws', 'polly', 'verbio'].includes(vendor),
|
||||
assert.ok(['google', 'aws', 'polly'].includes(vendor),
|
||||
`getTtsVoices not supported for vendor ${vendor}`);
|
||||
|
||||
switch (vendor) {
|
||||
case 'nuance':
|
||||
return getNuanceVoices(client, logger, credentials);
|
||||
case 'google':
|
||||
return getGoogleVoices(client, logger, credentials);
|
||||
case 'aws':
|
||||
case 'polly':
|
||||
return getAwsVoices(client, createHash, retrieveHash, logger, credentials);
|
||||
case 'verbio':
|
||||
return getVerbioVoices(client, logger, credentials);
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -1,51 +0,0 @@
|
||||
const {Pool} = require('undici');
|
||||
const { noopLogger, makeVerbioKey } = require('./utils');
|
||||
const { HTTP_TIMEOUT } = require('./config');
|
||||
const pool = new Pool('https://auth.speechcenter.verbio.com:444');
|
||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
|
||||
async function getVerbioAccessToken(client, logger, credentials) {
|
||||
logger = logger || noopLogger;
|
||||
const { client_id, client_secret } = credentials;
|
||||
try {
|
||||
const key = makeVerbioKey(client_id);
|
||||
const access_token = await client.get(key);
|
||||
if (access_token) {
|
||||
return {access_token, servedFromCache: true};
|
||||
}
|
||||
|
||||
const payload = {
|
||||
client_id,
|
||||
client_secret
|
||||
};
|
||||
|
||||
const {statusCode, headers, body} = await pool.request({
|
||||
path: '/api/v1/token',
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'User-Agent': 'jambonz'
|
||||
},
|
||||
body: JSON.stringify(payload),
|
||||
timeout: HTTP_TIMEOUT,
|
||||
followRedirects: false
|
||||
});
|
||||
|
||||
if (200 !== statusCode) {
|
||||
logger.debug({statusCode, headers, body: await body.text()}, 'error fetching access token from Verbio');
|
||||
const err = new Error();
|
||||
err.statusCode = statusCode;
|
||||
throw err;
|
||||
}
|
||||
const json = await body.json();
|
||||
const expiry = Math.floor(json.expiration_time - Date.now() / 1000 - 30);
|
||||
await client.set(key, json.access_token, 'EX', expiry);
|
||||
return {...json, servedFromCache: false};
|
||||
} catch (err) {
|
||||
debug(err, `getVerbioAccessToken: Error retrieving Verbio access token for client_id ${client_id}`);
|
||||
logger.error(err, `getVerbioAccessToken: Error retrieving Verbio access token for client_id ${client_id}`);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = getVerbioAccessToken;
|
||||
+4
-257
@@ -16,26 +16,10 @@ const {
|
||||
} = sdk;
|
||||
const {
|
||||
makeSynthKey,
|
||||
createNuanceClient,
|
||||
createKryptonClient,
|
||||
createRivaClient,
|
||||
noopLogger,
|
||||
makeFilePath,
|
||||
makePlayhtKey
|
||||
makeFilePath
|
||||
} = require('./utils');
|
||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||
const getVerbioAccessToken = require('./get-verbio-token');
|
||||
const {
|
||||
SynthesisRequest,
|
||||
Voice,
|
||||
AudioFormat,
|
||||
AudioParameters,
|
||||
PCM,
|
||||
Input,
|
||||
Text,
|
||||
SSML,
|
||||
EventParameters
|
||||
} = require('../stubs/nuance/synthesizer_pb');
|
||||
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
|
||||
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
|
||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
@@ -95,10 +79,10 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
let rtt;
|
||||
logger = logger || noopLogger;
|
||||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nuance', 'nvidia', 'elevenlabs',
|
||||
'whisper', 'deepgram', 'playht', 'rimelabs', 'verbio', 'cartesia', 'inworld', 'resemble'].includes(vendor) ||
|
||||
assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
|
||||
'whisper', 'deepgram', 'rimelabs', 'cartesia', 'inworld', 'resemble'].includes(vendor) ||
|
||||
vendor.startsWith('custom'),
|
||||
`synthAudio supported vendors are google, aws, microsoft, nuance, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
|
||||
if ('google' === vendor) {
|
||||
assert.ok(language, 'synthAudio requires language when google is used');
|
||||
}
|
||||
@@ -109,13 +93,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
assert.ok(language || deploymentId, 'synthAudio requires language when microsoft is used');
|
||||
assert.ok(voice || deploymentId, 'synthAudio requires voice when microsoft is used');
|
||||
}
|
||||
else if ('nuance' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nuance is used');
|
||||
if (!credentials.nuance_tts_uri) {
|
||||
assert.ok(credentials.client_id, 'synthAudio requires client_id in credentials when nuance is used');
|
||||
assert.ok(credentials.secret, 'synthAudio requires client_id in credentials when nuance is used');
|
||||
}
|
||||
}
|
||||
else if ('nvidia' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
|
||||
assert.ok(language, 'synthAudio requires language when nvidia is used');
|
||||
@@ -129,11 +106,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
assert.ok(voice, 'synthAudio requires voice when elevenlabs is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when elevenlabs is used');
|
||||
assert.ok(credentials.model_id, 'synthAudio requires model_id when elevenlabs is used');
|
||||
} else if ('playht' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when playht is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when playht is used');
|
||||
assert.ok(credentials.user_id, 'synthAudio requires user_id when playht is used');
|
||||
assert.ok(credentials.voice_engine, 'synthAudio requires voice_engine when playht is used');
|
||||
} else if ('inworld' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when inworld is used');
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when inworld is used');
|
||||
@@ -148,10 +120,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when whisper is used');
|
||||
} else if (vendor.startsWith('custom')) {
|
||||
assert.ok(credentials.custom_tts_url, `synthAudio requires custom_tts_url in credentials when ${vendor} is used`);
|
||||
} else if ('verbio' === vendor) {
|
||||
assert.ok(voice, 'synthAudio requires voice when verbio is used');
|
||||
assert.ok(credentials.client_id, 'synthAudio requires client_id when verbio is used');
|
||||
assert.ok(credentials.client_secret, 'synthAudio requires client_secret when verbio is used');
|
||||
} else if ('deepgram' === vendor) {
|
||||
if (!credentials.deepgram_tts_uri) {
|
||||
assert.ok(credentials.api_key, 'synthAudio requires api_key when deepgram is used');
|
||||
@@ -216,10 +184,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
audioData = await synthMicrosoft(logger, {credentials, stats, language, voice, key, text, deploymentId,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
break;
|
||||
case 'nuance':
|
||||
model = model || 'enhanced';
|
||||
audioData = await synthNuance(client, logger, {credentials, stats, voice, model, key, text});
|
||||
break;
|
||||
case 'nvidia':
|
||||
audioData = await synthNvidia(client, logger, {credentials, stats, language, voice, model, key, text,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
@@ -232,11 +196,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'playht':
|
||||
audioData = await synthPlayHT(client, logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'cartesia':
|
||||
audioData = await synthCartesia(logger, {
|
||||
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
|
||||
@@ -257,11 +216,6 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
|
||||
credentials, stats, voice, key, text, instructions, renderForCaching, disableTtsStreaming,
|
||||
disableTtsCache});
|
||||
break;
|
||||
case 'verbio':
|
||||
audioData = await synthVerbio(client, logger, {
|
||||
credentials, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
if (audioData?.filePath) return audioData;
|
||||
break;
|
||||
case 'deepgram':
|
||||
audioData = await synthDeepgram(logger, {credentials, stats, model, key, text,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache});
|
||||
@@ -725,70 +679,6 @@ const synthWellSaid = async(logger, {credentials, stats, language, voice, gender
|
||||
}
|
||||
};
|
||||
|
||||
const synthNuance = async(client, logger, {credentials, stats, voice, model, text}) => {
|
||||
let nuanceClient;
|
||||
const {client_id, secret, nuance_tts_uri} = credentials;
|
||||
if (nuance_tts_uri) {
|
||||
nuanceClient = await createKryptonClient(nuance_tts_uri);
|
||||
}
|
||||
else {
|
||||
/* get a nuance access token */
|
||||
const {access_token} = await getNuanceAccessToken(client, logger, client_id, secret, 'tts');
|
||||
nuanceClient = await createNuanceClient(access_token);
|
||||
}
|
||||
|
||||
const v = new Voice();
|
||||
const p = new AudioParameters();
|
||||
const f = new AudioFormat();
|
||||
const pcm = new PCM();
|
||||
const params = new EventParameters();
|
||||
const request = new SynthesisRequest();
|
||||
const input = new Input();
|
||||
|
||||
if (text.startsWith('<speak')) {
|
||||
const ssml = new SSML();
|
||||
ssml.setText(text);
|
||||
input.setSsml(ssml);
|
||||
}
|
||||
else {
|
||||
const t = new Text();
|
||||
t.setText(text);
|
||||
input.setText(t);
|
||||
}
|
||||
const sampleRate = 8000;
|
||||
pcm.setSampleRateHz(sampleRate);
|
||||
f.setPcm(pcm);
|
||||
p.setAudioFormat(f);
|
||||
v.setName(voice);
|
||||
v.setModel(model);
|
||||
request.setVoice(v);
|
||||
request.setAudioParams(p);
|
||||
request.setInput(input);
|
||||
request.setEventParams(params);
|
||||
request.setUserId('jambonz');
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
nuanceClient.unarySynthesize(request, (err, response) => {
|
||||
if (err) {
|
||||
console.error(err);
|
||||
return reject(err);
|
||||
}
|
||||
const status = response.getStatus();
|
||||
const code = status.getCode();
|
||||
if (code !== 200) {
|
||||
const message = status.getMessage();
|
||||
const details = status.getDetails();
|
||||
return reject({code, message, details});
|
||||
}
|
||||
resolve({
|
||||
audioContent: Buffer.from(response.getAudio()),
|
||||
extension: 'r8',
|
||||
sampleRate
|
||||
});
|
||||
});
|
||||
});
|
||||
};
|
||||
|
||||
const synthNvidia = async(client, logger, {
|
||||
credentials, stats, language, voice, model, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
@@ -954,101 +844,6 @@ const synthElevenlabs = async(logger, {
|
||||
}
|
||||
};
|
||||
|
||||
const synthPlayHT = async(client, logger, {
|
||||
credentials, options, stats, voice, language, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
const {api_key, user_id, voice_engine, playht_tts_uri, options: credOpts} = credentials;
|
||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||
|
||||
let synthesizeUrl = playht_tts_uri ? `${playht_tts_uri}/api/v2/tts/stream` : 'https://api.play.ht/api/v2/tts/stream';
|
||||
// If model is play3.0, the synthesizeUrl is got from authentication endpoint
|
||||
if (voice_engine === 'Play3.0') {
|
||||
try {
|
||||
const post = bent('https://api.play.ht', 'POST', 'json', 201, {
|
||||
'AUTHORIZATION': api_key,
|
||||
'X-USER-ID': user_id,
|
||||
'Accept': 'application/json'
|
||||
});
|
||||
const key = makePlayhtKey(api_key);
|
||||
const url = await client.get(key);
|
||||
if (!url) {
|
||||
const {inference_address, expires_at_ms} = await post('/api/v3/auth');
|
||||
synthesizeUrl = inference_address;
|
||||
const expiry = Math.floor((expires_at_ms - Date.now()) / 1000 - 30);
|
||||
await client.set(key, inference_address, 'EX', expiry);
|
||||
} else {
|
||||
// Use cached URL
|
||||
synthesizeUrl = url;
|
||||
}
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth PlayHT returned error for authentication version 3.0');
|
||||
stats.increment('tts.count', ['vendor:playht', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `api_key=${api_key}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += `,user_id=${user_id}`;
|
||||
params += ',vendor=playht';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,voice_engine=${voice_engine}`;
|
||||
params += `,synthesize_url=${synthesizeUrl}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
params += `,language=${language}`;
|
||||
if (opts.quality) params += `,quality=${opts.quality}`;
|
||||
if (opts.speed) params += `,speed=${opts.speed}`;
|
||||
if (opts.seed) params += `,style=${opts.seed}`;
|
||||
if (opts.temperature) params += `,temperature=${opts.temperature}`;
|
||||
if (opts.emotion) params += `,emotion=${opts.emotion}`;
|
||||
if (opts.voice_guidance) params += `,voice_guidance=${opts.voice_guidance}`;
|
||||
if (opts.style_guidance) params += `,style_guidance=${opts.style_guidance}`;
|
||||
if (opts.text_guidance) params += `,text_guidance=${opts.text_guidance}`;
|
||||
if (opts.top_p) params += `,top_p=${opts.top_p}`;
|
||||
if (opts.repetition_penalty) params += `,repetition_penalty=${opts.repetition_penalty}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const post = bent('POST', 'buffer', {
|
||||
...(voice_engine !== 'Play3.0' && {
|
||||
'AUTHORIZATION': api_key,
|
||||
'X-USER-ID': user_id,
|
||||
}),
|
||||
'Accept': 'audio/mpeg',
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
|
||||
const audioContent = await post(synthesizeUrl, {
|
||||
text,
|
||||
...(voice_engine === 'Play3.0' && { language }),
|
||||
voice,
|
||||
voice_engine,
|
||||
output_format: 'mp3',
|
||||
sample_rate: 8000,
|
||||
...opts
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'mp3',
|
||||
sampleRate: 8000
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth PlayHT returned error');
|
||||
stats.increment('tts.count', ['vendor:playht', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthInworld = async(logger, {
|
||||
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
@@ -1174,54 +969,6 @@ const synthRimelabs = async(logger, {
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
const synthVerbio = async(client, logger, {
|
||||
credentials, stats, voice, key, text, renderForCaching, disableTtsStreaming, disableTtsCache
|
||||
}) => {
|
||||
//https://doc.speechcenter.verbio.com/#tag/Text-To-Speech-REST-API
|
||||
if (text.length > 2000) {
|
||||
throw new Error('Verbio cannot synthesize for the text length larger than 2000 characters');
|
||||
}
|
||||
const token = await getVerbioAccessToken(client, logger, credentials);
|
||||
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||
let params = '{';
|
||||
params += `access_token=${token.access_token}`;
|
||||
params += `,playback_id=${key}`;
|
||||
params += ',vendor=verbio';
|
||||
params += `,voice=${voice}`;
|
||||
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||
servedFromCache: false,
|
||||
rtt: 0
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
const post = bent('https://us.rest.speechcenter.verbio.com', 'POST', 'buffer', {
|
||||
'Authorization': `Bearer ${token.access_token}`,
|
||||
'User-Agent': 'jambonz',
|
||||
'Content-Type': 'application/json'
|
||||
});
|
||||
const audioContent = await post('/api/v1/synthesize', {
|
||||
voice_id: voice,
|
||||
output_sample_rate: '8k',
|
||||
output_encoding: 'pcm16',
|
||||
text
|
||||
});
|
||||
return {
|
||||
audioContent,
|
||||
extension: 'r8',
|
||||
sampleRate: 8000
|
||||
};
|
||||
} catch (err) {
|
||||
logger.info({err}, 'synth Verbio returned error');
|
||||
stats.increment('tts.count', ['vendor:verbio', 'accepted:no']);
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
const synthWhisper = async(logger, {credentials, stats, voice, key, text, instructions,
|
||||
renderForCaching, disableTtsStreaming, disableTtsCache}) => {
|
||||
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||
|
||||
+1
-101
@@ -1,20 +1,7 @@
|
||||
const crypto = require('crypto');
|
||||
const {SynthesizerClient} = require('../stubs/nuance/synthesizer_grpc_pb');
|
||||
const {RivaSpeechSynthesisClient} = require('../stubs/riva/proto/riva_tts_grpc_pb');
|
||||
const {Pool} = require('undici');
|
||||
const pool = new Pool('https://auth.crt.nuance.com');
|
||||
const NUANCE_AUTH_ENDPOINT = 'tts.api.nuance.com:443';
|
||||
const grpc = require('@grpc/grpc-js');
|
||||
const formurlencoded = require('form-urlencoded');
|
||||
const { TMP_FOLDER, HTTP_TIMEOUT } = require('./config');
|
||||
|
||||
const debug = require('debug')('jambonz:realtimedb-helpers');
|
||||
/**
|
||||
* Future TODO: cache recently used connections to providers
|
||||
* to avoid connection overhead during a call.
|
||||
* Will need to periodically age them out to avoid memory leaks.
|
||||
*/
|
||||
//const nuanceClientMap = new Map();
|
||||
const { TMP_FOLDER } = require('./config');
|
||||
|
||||
function makeSynthKey({
|
||||
account_sid = '',
|
||||
@@ -45,90 +32,12 @@ const noopLogger = {
|
||||
error: () => {}
|
||||
};
|
||||
|
||||
const toBase64 = (str) => Buffer.from(str || '', 'utf8').toString('base64');
|
||||
|
||||
function makeBasicAuthHeader(username, password) {
|
||||
if (!username || !password) return {};
|
||||
const creds = `${encodeURIComponent(username)}:${password || ''}`;
|
||||
const header = `Basic ${toBase64(creds)}`;
|
||||
return {Authorization: header};
|
||||
}
|
||||
|
||||
function makeAwsKey(awsAccessKeyId) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(awsAccessKeyId);
|
||||
return `aws:${hash.digest('hex')}`;
|
||||
}
|
||||
|
||||
function makePlayhtKey(apiKey) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(apiKey);
|
||||
return `playht:${hash.digest('hex')}`;
|
||||
}
|
||||
function makeVerbioKey(client_id) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(client_id);
|
||||
return `verbio:${hash.digest('hex')}`;
|
||||
}
|
||||
|
||||
function makeNuanceKey(clientId, secret, scope) {
|
||||
const hash = crypto.createHash('sha1');
|
||||
hash.update(`${clientId}:${secret}:${scope}`);
|
||||
return `nuance:${hash.digest('hex')}`;
|
||||
}
|
||||
|
||||
const getNuanceAccessToken = async(clientId, secret, scope = 'asr tts') => {
|
||||
const payload = {
|
||||
grant_type: 'client_credentials',
|
||||
scope
|
||||
};
|
||||
const auth = makeBasicAuthHeader(clientId, secret);
|
||||
const {statusCode, headers, body} = await pool.request({
|
||||
path: '/oauth2/token',
|
||||
method: 'POST',
|
||||
headers: {
|
||||
...auth,
|
||||
'Content-Type': 'application/x-www-form-urlencoded'
|
||||
},
|
||||
body: formurlencoded(payload),
|
||||
timeout: HTTP_TIMEOUT,
|
||||
followRedirects: false
|
||||
});
|
||||
|
||||
if (200 !== statusCode) {
|
||||
debug({statusCode, headers, body: body.text()}, 'error fetching access token from Nuance');
|
||||
const err = new Error();
|
||||
err.statusCode = statusCode;
|
||||
throw err;
|
||||
}
|
||||
const json = await body.json();
|
||||
return json.access_token;
|
||||
};
|
||||
|
||||
const createKryptonClient = async(uri) => {
|
||||
const client = new SynthesizerClient(uri, grpc.credentials.createInsecure());
|
||||
return client;
|
||||
};
|
||||
|
||||
const createNuanceClient = async(access_token) => {
|
||||
|
||||
//if (nuanceClientMap.has(access_token)) return nuanceClientMap.get(access_token);
|
||||
|
||||
const generateMetadata = (params, callback) => {
|
||||
var metadata = new grpc.Metadata();
|
||||
metadata.add('authorization', `Bearer ${access_token}`);
|
||||
callback(null, metadata);
|
||||
};
|
||||
|
||||
const sslCreds = grpc.credentials.createSsl();
|
||||
const authCreds = grpc.credentials.createFromMetadataGenerator(generateMetadata);
|
||||
const combined_creds = grpc.credentials.combineChannelCredentials(sslCreds, authCreds);
|
||||
const client = new SynthesizerClient(NUANCE_AUTH_ENDPOINT, combined_creds);
|
||||
|
||||
//if (process.env.NUANCE_CACHE_TTS_CONNECTIONS) nuanceClientMap.set(access_token, client);
|
||||
return client;
|
||||
};
|
||||
|
||||
const createRivaClient = async(rivaUri) => {
|
||||
const client = new RivaSpeechSynthesisClient(rivaUri, grpc.credentials.createInsecure());
|
||||
return client;
|
||||
@@ -136,17 +45,8 @@ const createRivaClient = async(rivaUri) => {
|
||||
|
||||
module.exports = {
|
||||
makeSynthKey,
|
||||
makeNuanceKey,
|
||||
|
||||
makePlayhtKey,
|
||||
makeAwsKey,
|
||||
makeVerbioKey,
|
||||
getNuanceAccessToken,
|
||||
createNuanceClient,
|
||||
createKryptonClient,
|
||||
createRivaClient,
|
||||
makeBasicAuthHeader,
|
||||
NUANCE_AUTH_ENDPOINT,
|
||||
noopLogger,
|
||||
makeFilePath
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user