Compare commits

...
15 Commits
Author SHA1 Message Date
Dave Horton 6c5c8e734f 0.0.13 2023-05-10 07:39:21 -04:00
Dave Horton 4349cd7e40 Merge pull request #19 from jambonz/fix/nvidia
fixes for riva tts
2023-05-10 07:38:58 -04:00
Dave Horton b6058ca242 fix nvidia test 2023-05-10 07:36:25 -04:00
Dave Horton 68cbd63bbd minor logging 2023-05-09 13:59:38 -04:00
Dave Horton 521560e276 fixes for riva tts 2023-05-09 13:57:41 -04:00
Dave Horton d606141f57 fix issue in prev commit for microsoft 2023-04-01 13:19:27 -04:00
Dave Horton 0d58954537 bump version 2023-04-01 10:48:05 -04:00
Dave Horton 7c0eafded3 Merge pull request #18 from jambonz/fix/microsoft-buffer
fix: microsft retrun arrayBuffer not buffer, convert it now to buffer
2023-04-01 10:46:40 -04:00
Quan HL ab7d145288 fix: microsft retrun arrayBuffer not buffer, convert it now to buffer 2023-04-01 13:39:33 +07:00
Dave Horton 11746d3f22 bump version and minor changes 2023-03-31 20:05:07 -04:00
Dave Horton 3fcfbd10a1 Merge pull request #17 from jambonz/fix/imterim_audio_cut
fix: use synthesized audio data directly from microsoft sdk
2023-03-31 20:03:38 -04:00
Quan HL df8acbed0e fix: audioData is getter 2023-04-01 06:58:55 +07:00
Quan HL 4296ed7256 fix: user synthesized audio data directly from microsoft sdk 2023-04-01 06:37:33 +07:00
Quan HL f4b271c7b3 fix: user synthesized audio data directly from microsoft sdk 2023-04-01 06:36:15 +07:00
Dave Horton d5c71de27d update to latest speech packages 2023-03-31 15:56:18 -04:00
4 changed files with 1126 additions and 1423 deletions
+19 -23
View File
@@ -2,14 +2,12 @@ const assert = require('assert');
const fs = require('fs');
const bent = require('bent');
const ttsGoogle = require('@google-cloud/text-to-speech');
//const Polly = require('aws-sdk/clients/polly');
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
const sdk = require('microsoft-cognitiveservices-speech-sdk');
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
const { IamAuthenticator } = require('ibm-watson/auth');
const {
AudioConfig,
ResultReason,
SpeechConfig,
SpeechSynthesizer,
@@ -83,7 +81,7 @@ async function synthAudio(client, logger, stats, { account_sid,
else if ('nvidia' === vendor) {
assert.ok(voice, 'synthAudio requires voice when nvidia is used');
assert.ok(language, 'synthAudio requires language when nvidia is used');
assert.ok(credentials.riva_uri, 'synthAudio requires riva_uri in credentials when nuance is used');
assert.ok(credentials.riva_server_uri, 'synthAudio requires riva_server_uri in credentials when nvidia is used');
}
else if ('ibm' === vendor) {
assert.ok(voice, 'synthAudio requires voice when ibm is used');
@@ -172,8 +170,6 @@ async function synthAudio(client, logger, stats, { account_sid,
client.setexAsync(key, EXPIRES, audioBuffer.toString('base64'))
.catch((err) => logger.error(err, `error calling setex on key ${key}`));
if (['microsoft'].includes(vendor)) return {filePath, servedFromCache, rtt};
}
return new Promise((resolve, reject) => {
@@ -302,8 +298,7 @@ const synthMicrosoft = async(logger, {
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
const config = AudioConfig.fromAudioFileOutput(filePath);
const synthesizer = new SpeechSynthesizer(speechConfig, config);
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
@@ -328,12 +323,9 @@ const synthMicrosoft = async(logger, {
reject(cancellation.errorDetails);
break;
case ResultReason.SynthesizingAudioCompleted:
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
resolve(Buffer.from(result.audioData));
synthesizer.close();
fs.readFile(filePath, (err, data) => {
if (err) return reject(err);
resolve(data);
});
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
break;
default:
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
@@ -433,21 +425,25 @@ const synthNuance = async(client, logger, {credentials, stats, voice, model, tex
};
const synthNvidia = async(client, logger, {credentials, stats, language, voice, model, text}) => {
const {riva_uri} = credentials;
const rivaClient = await createRivaClient(riva_uri);
const request = new SynthesizeSpeechRequest();
request.setVoiceName(voice);
request.setLanguageCode(language);
request.setSampleRateHz(8000);
request.setEncoding(AudioEncoding.LINEAR_PCM);
request.setText(text);
const {riva_server_uri} = credentials;
let rivaClient, request;
try {
rivaClient = await createRivaClient(riva_server_uri);
request = new SynthesizeSpeechRequest();
request.setVoiceName(voice);
request.setLanguageCode(language);
request.setSampleRateHz(8000);
request.setEncoding(AudioEncoding.LINEAR_PCM);
request.setText(text);
} catch (err) {
logger.info({err}, 'error creating riva client');
return Promise.reject(err);
}
return new Promise((resolve, reject) => {
console.log(`language ${language} voice ${voice} model ${model} text ${text}`);
rivaClient.synthesize(request, (err, response) => {
if (err) {
console.error(err);
logger.info({err, voice, language}, 'error synthesizing speech using Nvidia');
return reject(err);
}
resolve(Buffer.from(response.getAudio()));
+1095 -1388
View File
File diff suppressed because it is too large Load Diff
+10 -10
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.8",
"version": "0.0.13",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -24,18 +24,18 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.276.0",
"@google-cloud/text-to-speech": "^4.2.0",
"@grpc/grpc-js": "^1.8.7",
"@jambonz/realtimedb-helpers": "^0.6.3",
"aws-sdk": "^2.1310.0",
"@aws-sdk/client-polly": "^3.303.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"@jambonz/promisify-redis": "^0.0.6",
"bent": "^7.3.12",
"debug": "^4.3.4",
"google-protobuf": "^3.21.2",
"ibm-watson": "^7.1.2",
"form-urlencoded": "^6.1.0",
"microsoft-cognitiveservices-speech-sdk": "^1.25.0",
"undici": "^5.19.1"
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
"redis": "^3.1.2",
"undici": "^5.21.0"
},
"devDependencies": {
"config": "^3.3.9",
+2 -2
View File
@@ -300,7 +300,7 @@ test('Nvidia speech synth tests', async(t) => {
let opts = await synthAudio(stats, {
vendor: 'nvidia',
credentials: {
riva_uri: process.env.RIVA_URI,
riva_server_uri: process.env.RIVA_URI,
},
language: 'en-US',
voice: 'English-US.Female-1',
@@ -311,7 +311,7 @@ test('Nvidia speech synth tests', async(t) => {
opts = await synthAudio(stats, {
vendor: 'nvidia',
credentials: {
riva_uri: process.env.RIVA_URI,
riva_server_uri: process.env.RIVA_URI,
},
language: 'en-US',
voice: 'English-US.Female-1',