Compare commits

...
22 Commits
Author SHA1 Message Date
Dave Horton bf229d0ab0 0.0.46 2024-04-03 13:22:20 -04:00
Dave Horton f08fedb8ca enable caching from azure tts streaming 2024-04-03 13:17:36 -04:00
Dave Horton fbed59e5de 0.0.45 2024-04-02 15:10:54 -04:00
Dave Horton 4f1685a365 Merge pull request #59 from jambonz/feat/azure_tts
support azure streaming
2024-03-30 09:21:07 -04:00
Quan HL 2701af102a wip 2024-03-30 17:49:14 +07:00
Quan HL 7f939b96d2 wip 2024-03-30 17:38:00 +07:00
Quan HL 4d58ca6daf wip 2024-03-30 17:34:49 +07:00
Hoan Luu Huu 16dd7a2805 Merge branch 'main' into feat/azure_tts 2024-03-30 17:04:01 +07:00
Dave Horton 8f3e930004 0.0.44 2024-03-20 19:43:26 -04:00
Dave Horton 3f4c444d82 add azure SSML tests 2024-03-20 19:43:18 -04:00
Hoan Luu Huu fd7d8b8bcd Merge pull request #62 from jambonz/feat/mod_dub
say command for freeswitch module to include vendor and voice
2024-03-20 13:40:51 +07:00
Hoan Luu Huu 46f833c7fa Merge branch 'main' into feat/mod_dub 2024-03-12 18:10:21 +07:00
Quan HL f3ab2baa6a wip 2024-03-10 07:34:55 +07:00
Hoan Luu Huu fb412e2ddf Merge branch 'main' into feat/mod_dub 2024-03-10 06:43:09 +07:00
Quan HL f06f96a6f0 wip 2024-03-10 06:41:46 +07:00
Hoan Luu Huu 2988e800b1 Merge branch 'main' into feat/azure_tts 2024-03-10 06:39:00 +07:00
Quan HL c3188e40bb support mod_dub 2024-03-09 16:59:00 +07:00
Quan HL 31a54f595b wip 2024-02-26 15:42:09 +07:00
Quan HL 3560a6d4d9 wip 2024-02-26 14:01:53 +07:00
Quan HL be8053db4f wip 2024-02-26 13:54:34 +07:00
Quan HL 4ffae38a3f wip 2024-02-26 13:49:37 +07:00
Quan HL 9e74760c39 support azure streaming 2024-02-26 13:33:22 +07:00
4 changed files with 103 additions and 21 deletions
+44 -18
View File
@@ -183,7 +183,9 @@ async function synthAudio(client, logger, stats, { account_sid,
case 'azure':
case 'microsoft':
vendorLabel = 'microsoft';
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId, filePath});
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId,
filePath, renderForCaching, disableTtsStreaming});
if (audioBuffer?.filePath) return audioBuffer;
break;
case 'nuance':
model = model || 'enhanced';
@@ -381,10 +383,47 @@ const synthMicrosoft = async(logger, {
language,
voice,
text,
filePath
filePath,
renderForCaching,
disableTtsStreaming
}) => {
try {
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
// let clean up the text
let content = text;
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${apiKey}`;
params += `,language=${language}`;
params += ',vendor=microsoft';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (region) params += `,region=${region}`;
if (custom_tts_endpoint) params += `,endpointId=${custom_tts_endpoint}`;
if (process.env.JAMBONES_HTTP_PROXY_IP) params += `,http_proxy_ip=${process.env.JAMBONES_HTTP_PROXY_IP}`;
if (process.env.JAMBONES_HTTP_PROXY_PORT) params += `,http_proxy_port=${process.env.JAMBONES_HTTP_PROXY_PORT}`;
params += '}';
return {
filePath: `say:${params}${content.replace(/\n/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}
if (use_custom_tts && custom_tts_endpoint_url) {
return await _synthOnPremMicrosoft(logger, {
credentials,
@@ -396,20 +435,12 @@ const synthMicrosoft = async(logger, {
});
}
const trimSilence = filePath.endsWith('.r8');
let content = text;
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
speechConfig.speechSynthesisLanguage = language;
speechConfig.speechSynthesisVoiceName = voice;
if (use_custom_tts && custom_tts_endpoint) {
speechConfig.endpointId = custom_tts_endpoint;
}
if (use_custom_tts && !content.startsWith('<speak')) {
/**
* Note: it seems that to use custom voice ssml is required with the voice attribute
* Otherwise sending plain text we get "Voice does not match"
*/
content = `<speak>${text}</speak>`;
}
speechConfig.speechSynthesisOutputFormat = trimSilence ?
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
@@ -421,14 +452,6 @@ const synthMicrosoft = async(logger, {
}
const synthesizer = new SpeechSynthesizer(speechConfig);
if (content.startsWith('<speak>')) {
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
// eslint-disable-next-line max-len
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
logger.info({content}, 'synthMicrosoft');
}
return new Promise((resolve, reject) => {
const speakAsync = content.startsWith('<speak') ?
synthesizer.speakSsmlAsync.bind(synthesizer) :
@@ -614,6 +637,8 @@ const synthElevenlabs = async(logger, {
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '';
params += `{api_key=${api_key}`;
params += ',vendor=elevenlabs';
params += `,voice=${voice}`;
params += `,model_id=${model_id}`;
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
params += ',write_cache_file=1';
@@ -662,6 +687,7 @@ const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCa
let params = '';
params += `{api_key=${api_key}`;
params += `,model_id=${model_id}`;
params += ',vendor=whisper';
params += `,voice=${voice}`;
params += ',write_cache_file=1';
if (speed) params += `,speed=${speed}`;
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.43",
"version": "0.0.46",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.0.43",
"version": "0.0.46",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.496.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.43",
"version": "0.0.46",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
+56
View File
@@ -188,6 +188,7 @@ test('Azure speech synth tests', async(t) => {
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text: longText,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
@@ -203,6 +204,7 @@ test('Azure speech synth tests', async(t) => {
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text: longText,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
} catch (err) {
@@ -212,6 +214,58 @@ test('Azure speech synth tests', async(t) => {
client.quit();
});
test('Azure SSML tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.MICROSOFT_API_KEY || !process.env.MICROSOFT_REGION) {
t.pass('skipping Microsoft speech synth tests since MICROSOFT_API_KEY or MICROSOFT_REGION not provided');
return t.end();
}
try {
const text = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="en-US">
<voice name="en-US-JennyMultilingualNeural">
<mstts:express-as style="cheerful" styledegree="2">That'd be just amazing!
</mstts:express-as>
</voice>
</speak>`;
let opts = await synthAudio(stats, {
vendor: 'microsoft',
credentials: {
api_key: process.env.MICROSOFT_API_KEY,
region: process.env.MICROSOFT_REGION,
},
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
t.pass('successfully used proxy to reach microsoft tts service');
}
opts = await synthAudio(stats, {
vendor: 'microsoft',
credentials: {
api_key: process.env.MICROSOFT_API_KEY,
region: process.env.MICROSOFT_REGION,
},
language: 'en-US',
voice: 'en-US-ChristopherNeural',
text,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
} catch (err) {
console.error(err);
t.end(err);
}
client.quit();
});
test('Azure custom voice speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
@@ -233,6 +287,7 @@ test('Azure custom voice speech synth tests', async(t) => {
language: 'en-US',
voice: process.env.MICROSOFT_CUSTOM_VOICE,
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
@@ -247,6 +302,7 @@ test('Azure custom voice speech synth tests', async(t) => {
language: 'en-US',
voice: process.env.MICROSOFT_CUSTOM_VOICE,
text,
renderForCaching: true
});
t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`);
} catch (err) {