mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b5daeff047 | ||
|
|
da02926c9a | ||
|
|
da3cdbb7aa | ||
|
|
625f147137 | ||
|
|
2e5687978e | ||
|
|
897481d34c | ||
|
|
bd5282e681 | ||
|
|
95a1384f02 | ||
|
|
35deeecf70 | ||
|
|
9d2ac3273f | ||
|
|
b1049aad7f | ||
|
|
40f51e7509 | ||
|
|
a0e2fe167c | ||
|
|
1fa853faa3 | ||
|
|
95e8d942b8 | ||
|
|
7a91876cd7 | ||
|
|
d07344ba3b | ||
|
|
44d8af2a96 | ||
|
|
b530db9a62 | ||
|
|
4c166c8eb4 | ||
|
|
8246dbea21 | ||
|
|
0084f6a468 |
+63
-4
@@ -14,7 +14,13 @@ const {
|
||||
CancellationDetails,
|
||||
SpeechSynthesisOutputFormat
|
||||
} = sdk;
|
||||
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
|
||||
const {
|
||||
makeSynthKey,
|
||||
createNuanceClient,
|
||||
createKryptonClient,
|
||||
createRivaClient,
|
||||
noopLogger
|
||||
} = require('./utils');
|
||||
const getNuanceAccessToken = require('./get-nuance-access-token');
|
||||
const {
|
||||
SynthesisRequest,
|
||||
@@ -297,6 +303,48 @@ const synthIbm = async(logger, {credentials, stats, voice, text}) => {
|
||||
}
|
||||
};
|
||||
|
||||
async function _synthOnPremMicrosoft(logger, {
|
||||
credentials,
|
||||
stats,
|
||||
language,
|
||||
voice,
|
||||
text,
|
||||
filePath
|
||||
}) {
|
||||
const {use_custom_tts, custom_tts_endpoint_url} = credentials;
|
||||
let content = text;
|
||||
|
||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||
/**
|
||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||
* Otherwise sending plain text we get "Voice does not match"
|
||||
*/
|
||||
content = `<speak>${text}</speak>`;
|
||||
}
|
||||
|
||||
if (content.startsWith('<speak>')) {
|
||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
|
||||
// eslint-disable-next-line max-len
|
||||
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
|
||||
logger.info({content}, 'synthMicrosoft');
|
||||
}
|
||||
|
||||
try {
|
||||
const trimSilence = filePath.endsWith('.r8');
|
||||
const post = bent('POST', 'buffer', {
|
||||
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
|
||||
'Content-Type': 'application/ssml+xml',
|
||||
'User-Agent': 'Jambonz'
|
||||
});
|
||||
const mp3 = await post(custom_tts_endpoint_url, content);
|
||||
return mp3;
|
||||
} catch (err) {
|
||||
logger.info({err}, '_synthMicrosoftByHttp returned error');
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
const synthMicrosoft = async(logger, {
|
||||
credentials,
|
||||
stats,
|
||||
@@ -306,7 +354,17 @@ const synthMicrosoft = async(logger, {
|
||||
filePath
|
||||
}) => {
|
||||
try {
|
||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
|
||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
|
||||
if (use_custom_tts && custom_tts_endpoint_url) {
|
||||
return await _synthOnPremMicrosoft(logger, {
|
||||
credentials,
|
||||
stats,
|
||||
language,
|
||||
voice,
|
||||
text,
|
||||
filePath
|
||||
});
|
||||
}
|
||||
const trimSilence = filePath.endsWith('.r8');
|
||||
let content = text;
|
||||
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
||||
@@ -314,12 +372,13 @@ const synthMicrosoft = async(logger, {
|
||||
speechConfig.speechSynthesisVoiceName = voice;
|
||||
if (use_custom_tts && custom_tts_endpoint) {
|
||||
speechConfig.endpointId = custom_tts_endpoint;
|
||||
|
||||
}
|
||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||
/**
|
||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||
* Otherwise sending plain text we get "Voice does not match"
|
||||
*/
|
||||
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
||||
content = `<speak>${text}</speak>`;
|
||||
}
|
||||
speechConfig.speechSynthesisOutputFormat = trimSilence ?
|
||||
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
|
||||
|
||||
@@ -106,7 +106,6 @@ const createRivaClient = async(rivaUri) => {
|
||||
return client;
|
||||
};
|
||||
|
||||
|
||||
module.exports = {
|
||||
makeSynthKey,
|
||||
makeNuanceKey,
|
||||
|
||||
Generated
+1413
-1724
File diff suppressed because it is too large
Load Diff
+3
-3
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@jambonz/speech-utils",
|
||||
"version": "0.0.16",
|
||||
"version": "0.0.21",
|
||||
"description": "TTS-related speech utilities for jambonz",
|
||||
"main": "index.js",
|
||||
"author": "Dave Horton",
|
||||
@@ -24,7 +24,7 @@
|
||||
},
|
||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-polly": "^3.347.1",
|
||||
"@aws-sdk/client-polly": "^3.359.0",
|
||||
"@google-cloud/text-to-speech": "^4.2.1",
|
||||
"@grpc/grpc-js": "^1.8.13",
|
||||
"bent": "^7.3.12",
|
||||
@@ -32,7 +32,7 @@
|
||||
"form-urlencoded": "^6.1.0",
|
||||
"google-protobuf": "^3.21.2",
|
||||
"ibm-watson": "^8.0.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
|
||||
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
|
||||
"ioredis": "^5.3.2",
|
||||
"undici": "^5.21.0"
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user