mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
11746d3f22 | ||
|
|
3fcfbd10a1 | ||
|
|
df8acbed0e | ||
|
|
4296ed7256 | ||
|
|
f4b271c7b3 | ||
|
|
d5c71de27d |
+2
-9
@@ -2,14 +2,12 @@ const assert = require('assert');
|
||||
const fs = require('fs');
|
||||
const bent = require('bent');
|
||||
const ttsGoogle = require('@google-cloud/text-to-speech');
|
||||
//const Polly = require('aws-sdk/clients/polly');
|
||||
const { PollyClient, SynthesizeSpeechCommand } = require('@aws-sdk/client-polly');
|
||||
|
||||
const sdk = require('microsoft-cognitiveservices-speech-sdk');
|
||||
const TextToSpeechV1 = require('ibm-watson/text-to-speech/v1');
|
||||
const { IamAuthenticator } = require('ibm-watson/auth');
|
||||
const {
|
||||
AudioConfig,
|
||||
ResultReason,
|
||||
SpeechConfig,
|
||||
SpeechSynthesizer,
|
||||
@@ -302,8 +300,7 @@ const synthMicrosoft = async(logger, {
|
||||
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
||||
}
|
||||
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||
const config = AudioConfig.fromAudioFileOutput(filePath);
|
||||
const synthesizer = new SpeechSynthesizer(speechConfig, config);
|
||||
const synthesizer = new SpeechSynthesizer(speechConfig);
|
||||
|
||||
if (content.startsWith('<speak>')) {
|
||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||
@@ -329,11 +326,8 @@ const synthMicrosoft = async(logger, {
|
||||
break;
|
||||
case ResultReason.SynthesizingAudioCompleted:
|
||||
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
||||
resolve(result.audioData);
|
||||
synthesizer.close();
|
||||
fs.readFile(filePath, (err, data) => {
|
||||
if (err) return reject(err);
|
||||
resolve(data);
|
||||
});
|
||||
break;
|
||||
default:
|
||||
logger.info({result}, 'synthAudio: (Microsoft) unexpected result');
|
||||
@@ -444,7 +438,6 @@ const synthNvidia = async(client, logger, {credentials, stats, language, voice,
|
||||
request.setText(text);
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
console.log(`language ${language} voice ${voice} model ${model} text ${text}`);
|
||||
rivaClient.synthesize(request, (err, response) => {
|
||||
if (err) {
|
||||
console.error(err);
|
||||
|
||||
Generated
+1095
-1388
File diff suppressed because it is too large
Load Diff
+10
-10
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.8",
|
||||
"version": "0.0.10",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -24,18 +24,18 @@
|
||||
},
|
||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.276.0",
|
||||
"@google-cloud/text-to-speech": "^4.2.0",
|
||||
"@grpc/grpc-js": "^1.8.7",
|
||||
"@jambonz/realtimedb-helpers": "^0.6.3",
|
||||
"aws-sdk": "^2.1310.0",
|
||||
"@aws-sdk/client-polly": "^3.303.0",
|
||||
"@google-cloud/text-to-speech": "^4.2.1",
|
||||
"@grpc/grpc-js": "^1.8.13",
|
||||
"@jambonz/promisify-redis": "^0.0.6",
|
||||
"bent": "^7.3.12",
|
||||
"debug": "^4.3.4",
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^7.1.2",
|
||||
"form-urlencoded": "^6.1.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.25.0",
|
||||
"undici": "^5.19.1"
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^8.0.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
|
||||
"redis": "^3.1.2",
|
||||
"undici": "^5.21.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"config": "^3.3.9",
|
||||
|
||||
Reference in New Issue
Block a user