Compare commits

...
22 Commits
Author SHA1 Message Date
Dave Horton b5daeff047 0.0.21 2023-09-08 07:58:11 -04:00
Dave Horton da02926c9a Merge pull request #31 from jambonz/fix/onprem-azure
fix raw audio downloaded from onprem azure
2023-09-08 07:57:36 -04:00
Quan HL da3cdbb7aa fix raw audio downloaded from onprem azure 2023-09-08 15:42:57 +07:00
Dave Horton 625f147137 0.0.20 2023-08-30 21:08:52 -04:00
Dave Horton 2e5687978e 0.0.19 2023-08-30 21:08:41 -04:00
Dave Horton 897481d34c Merge pull request #29 from jambonz/feat/azure_fromhost
support self hosted microsoft
2023-08-30 21:07:44 -04:00
Quan HL bd5282e681 fix 2023-08-28 20:05:55 +07:00
Quan HL 95a1384f02 wip 2023-08-25 15:57:43 +07:00
Quan HL 35deeecf70 wip 2023-08-25 15:57:23 +07:00
Quan HL 9d2ac3273f wip 2023-08-25 13:53:09 +07:00
Quan HL b1049aad7f wip 2023-08-25 13:49:04 +07:00
Quan HL 40f51e7509 wip 2023-08-11 18:25:57 +07:00
Quan HL a0e2fe167c wip 2023-08-11 16:04:11 +07:00
Quan HL 1fa853faa3 wip 2023-08-11 13:30:11 +07:00
Quan HL 95e8d942b8 fix jslint 2023-08-09 18:00:17 +07:00
Quan HL 7a91876cd7 support self hosted microsoft 2023-08-09 17:56:38 +07:00
Dave Horton d07344ba3b Merge pull request #28 from jambonz/revert/ssml-silence-trim
revert change to _not_ trim silence when azure ssml is used
2023-07-25 12:34:22 -04:00
Dave Horton 44d8af2a96 revert change to _not_ trim silence when azure ssml is used 2023-07-25 11:08:06 -04:00
Dave Horton b530db9a62 0.0.18 2023-07-25 07:40:33 -04:00
Dave Horton 4c166c8eb4 synth_audio: dont trim silence for Azure when using SSML 2023-07-25 07:40:28 -04:00
Dave Horton 8246dbea21 0.0.17 2023-07-19 10:10:39 -04:00
Dave Horton 0084f6a468 update deps 2023-07-19 10:09:31 -04:00
4 changed files with 1479 additions and 1732 deletions
+63 -4
View File
@@ -14,7 +14,13 @@ const {
CancellationDetails,
SpeechSynthesisOutputFormat
} = sdk;
const {makeSynthKey, createNuanceClient, createKryptonClient, createRivaClient, noopLogger} = require('./utils');
const {
makeSynthKey,
createNuanceClient,
createKryptonClient,
createRivaClient,
noopLogger
} = require('./utils');
const getNuanceAccessToken = require('./get-nuance-access-token');
const {
SynthesisRequest,
@@ -297,6 +303,48 @@ const synthIbm = async(logger, {credentials, stats, voice, text}) => {
}
};
async function _synthOnPremMicrosoft(logger, {
credentials,
stats,
language,
voice,
text,
filePath
}) {
const {use_custom_tts, custom_tts_endpoint_url} = credentials;
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
try {
const trimSilence = filePath.endsWith('.r8');
const post = bent('POST', 'buffer', {
'X-Microsoft-OutputFormat': trimSilence ? 'raw-8khz-16bit-mono-pcm' : 'audio-16khz-32kbitrate-mono-mp3',
'Content-Type': 'application/ssml+xml',
'User-Agent': 'Jambonz'
});
const mp3 = await post(custom_tts_endpoint_url, content);
return mp3;
} catch (err) {
logger.info({err}, '_synthMicrosoftByHttp returned error');
throw err;
}
}
const synthMicrosoft = async(logger, {
credentials,
stats,
@@ -306,7 +354,17 @@ const synthMicrosoft = async(logger, {
filePath
}) => {
try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
if (use_custom_tts && custom_tts_endpoint_url) {
return await _synthOnPremMicrosoft(logger, {
credentials,
stats,
language,
voice,
text,
filePath
});
}
const trimSilence = filePath.endsWith('.r8');
let content = text;
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
@@ -314,12 +372,13 @@ const synthMicrosoft = async(logger, {
speechConfig.speechSynthesisVoiceName = voice;
if (use_custom_tts && custom_tts_endpoint) {
speechConfig.endpointId = custom_tts_endpoint;
}
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
-1
View File
@@ -106,7 +106,6 @@ const createRivaClient = async(rivaUri) => {
return client;
};
module.exports = {
makeSynthKey,
makeNuanceKey,
+1413 -1724
View File
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.16",
"version": "0.0.21",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -24,7 +24,7 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"@aws-sdk/client-polly": "^3.347.1",
"@aws-sdk/client-polly": "^3.359.0",
"@google-cloud/text-to-speech": "^4.2.1",
"@grpc/grpc-js": "^1.8.13",
"bent": "^7.3.12",
@@ -32,7 +32,7 @@
"form-urlencoded": "^6.1.0",
"google-protobuf": "^3.21.2",
"ibm-watson": "^8.0.0",
"microsoft-cognitiveservices-speech-sdk": "^1.26.0",
"microsoft-cognitiveservices-speech-sdk": "^1.31.0",
"ioredis": "^5.3.2",
"undici": "^5.21.0"
},