mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-05 01:22:11 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e54e913fdd | ||
|
|
b088c0d7d9 | ||
|
|
f154a40692 | ||
|
|
0471b94ebe | ||
|
|
8279891dff | ||
|
|
f546ca998d | ||
|
|
bf229d0ab0 | ||
|
|
f08fedb8ca | ||
|
|
f3cc38089c | ||
|
|
fbed59e5de | ||
|
|
4f1685a365 | ||
|
|
2701af102a | ||
|
|
7f939b96d2 | ||
|
|
4d58ca6daf | ||
|
|
16dd7a2805 | ||
|
|
8f3e930004 | ||
|
|
3f4c444d82 | ||
|
|
fd7d8b8bcd | ||
|
|
46f833c7fa | ||
|
|
4eabfbe4b7 | ||
|
|
f3ab2baa6a | ||
|
|
fb412e2ddf | ||
|
|
f06f96a6f0 | ||
|
|
2988e800b1 | ||
|
|
dbfabeaddf | ||
|
|
c3188e40bb | ||
|
|
d0dfd07204 | ||
|
|
04a2466f54 | ||
|
|
0f9a9edc4d | ||
|
|
31a54f595b | ||
|
|
3560a6d4d9 | ||
|
|
be8053db4f | ||
|
|
4ffae38a3f | ||
|
|
9e74760c39 | ||
|
|
ced1a0ef0d | ||
|
|
1609d0b205 | ||
|
|
ef8ada2793 | ||
|
|
444ad2522f | ||
|
|
1caea60803 | ||
|
|
97c3588cfd | ||
|
|
da3aa5aadb | ||
|
|
2fe89f132c | ||
|
|
3cf9894b44 |
+95
-38
@@ -77,7 +77,7 @@ const trimTrailingSilence = (buffer) => {
|
|||||||
*/
|
*/
|
||||||
async function synthAudio(client, logger, stats, { account_sid,
|
async function synthAudio(client, logger, stats, { account_sid,
|
||||||
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
vendor, language, voice, gender, text, engine, salt, model, credentials, deploymentId,
|
||||||
disableTtsCache, renderForCaching, options
|
disableTtsCache, renderForCaching, disableTtsStreaming, options
|
||||||
}) {
|
}) {
|
||||||
let audioBuffer;
|
let audioBuffer;
|
||||||
let servedFromCache = false;
|
let servedFromCache = false;
|
||||||
@@ -141,12 +141,12 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
let filePath;
|
let filePath;
|
||||||
if (['nuance', 'nvidia'].includes(vendor) ||
|
if (['nuance', 'nvidia'].includes(vendor) ||
|
||||||
(
|
(
|
||||||
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
(process.env.JAMBONES_TTS_TRIM_SILENCE || !process.env.JAMBONES_DISABLE_TTS_STREAMING) &&
|
||||||
['microsoft', 'azure'].includes(vendor)
|
['microsoft', 'azure'].includes(vendor)
|
||||||
) ||
|
) ||
|
||||||
(
|
(
|
||||||
process.env.JAMBONES_ELEVENLABS_STREAMING &&
|
!process.env.JAMBONES_DISABLE_TTS_STREAMING &&
|
||||||
vendor === 'elevenlabs'
|
['elevenlabs', 'deepgram'].includes(vendor)
|
||||||
)
|
)
|
||||||
) {
|
) {
|
||||||
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
||||||
@@ -183,7 +183,9 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
case 'azure':
|
case 'azure':
|
||||||
case 'microsoft':
|
case 'microsoft':
|
||||||
vendorLabel = 'microsoft';
|
vendorLabel = 'microsoft';
|
||||||
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId, filePath});
|
audioBuffer = await synthMicrosoft(logger, {credentials, stats, language, voice, text, deploymentId,
|
||||||
|
filePath, renderForCaching, disableTtsStreaming});
|
||||||
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
break;
|
break;
|
||||||
case 'nuance':
|
case 'nuance':
|
||||||
model = model || 'enhanced';
|
model = model || 'enhanced';
|
||||||
@@ -200,20 +202,19 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
break;
|
break;
|
||||||
case 'elevenlabs':
|
case 'elevenlabs':
|
||||||
audioBuffer = await synthElevenlabs(logger, {
|
audioBuffer = await synthElevenlabs(logger, {
|
||||||
credentials, options, stats, language, voice, text, renderForCaching, filePath
|
credentials, options, stats, language, voice, text, renderForCaching, disableTtsStreaming, filePath
|
||||||
});
|
});
|
||||||
if (typeof audioBuffer === 'object' && audioBuffer.filePath) {
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
return audioBuffer;
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
audioBuffer = await synthElevenlabs(logger, {credentials, options, stats, language, voice, text, filePath});
|
|
||||||
}
|
|
||||||
break;
|
break;
|
||||||
case 'whisper':
|
case 'whisper':
|
||||||
audioBuffer = await synthWhisper(logger, {credentials, stats, voice, text});
|
audioBuffer = await synthWhisper(logger, {
|
||||||
|
credentials, stats, voice, text, renderForCaching, disableTtsStreaming});
|
||||||
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
break;
|
break;
|
||||||
case 'deepgram':
|
case 'deepgram':
|
||||||
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text});
|
audioBuffer = await synthDeepgram(logger, {credentials, stats, model, text,
|
||||||
|
renderForCaching, disableTtsStreaming});
|
||||||
|
if (audioBuffer?.filePath) return audioBuffer;
|
||||||
break;
|
break;
|
||||||
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
|
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
|
||||||
({ audioBuffer, filePath } = await synthCustomVendor(logger,
|
({ audioBuffer, filePath } = await synthCustomVendor(logger,
|
||||||
@@ -384,10 +385,47 @@ const synthMicrosoft = async(logger, {
|
|||||||
language,
|
language,
|
||||||
voice,
|
voice,
|
||||||
text,
|
text,
|
||||||
filePath
|
filePath,
|
||||||
|
renderForCaching,
|
||||||
|
disableTtsStreaming
|
||||||
}) => {
|
}) => {
|
||||||
try {
|
try {
|
||||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
|
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint, custom_tts_endpoint_url} = credentials;
|
||||||
|
// let clean up the text
|
||||||
|
let content = text;
|
||||||
|
if (use_custom_tts && !content.startsWith('<speak')) {
|
||||||
|
/**
|
||||||
|
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
||||||
|
* Otherwise sending plain text we get "Voice does not match"
|
||||||
|
*/
|
||||||
|
content = `<speak>${text}</speak>`;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (content.startsWith('<speak>')) {
|
||||||
|
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
||||||
|
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
|
||||||
|
// eslint-disable-next-line max-len
|
||||||
|
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
|
||||||
|
logger.info({content}, 'synthMicrosoft');
|
||||||
|
}
|
||||||
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '';
|
||||||
|
params += `{api_key=${apiKey}`;
|
||||||
|
params += `,language=${language}`;
|
||||||
|
params += ',vendor=microsoft';
|
||||||
|
params += `,voice=${voice}`;
|
||||||
|
params += ',write_cache_file=1';
|
||||||
|
if (region) params += `,region=${region}`;
|
||||||
|
if (custom_tts_endpoint) params += `,endpointId=${custom_tts_endpoint}`;
|
||||||
|
if (process.env.JAMBONES_HTTP_PROXY_IP) params += `,http_proxy_ip=${process.env.JAMBONES_HTTP_PROXY_IP}`;
|
||||||
|
if (process.env.JAMBONES_HTTP_PROXY_PORT) params += `,http_proxy_port=${process.env.JAMBONES_HTTP_PROXY_PORT}`;
|
||||||
|
params += '}';
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${content.replace(/\n/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
if (use_custom_tts && custom_tts_endpoint_url) {
|
if (use_custom_tts && custom_tts_endpoint_url) {
|
||||||
return await _synthOnPremMicrosoft(logger, {
|
return await _synthOnPremMicrosoft(logger, {
|
||||||
credentials,
|
credentials,
|
||||||
@@ -399,20 +437,12 @@ const synthMicrosoft = async(logger, {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
const trimSilence = filePath.endsWith('.r8');
|
const trimSilence = filePath.endsWith('.r8');
|
||||||
let content = text;
|
|
||||||
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
||||||
speechConfig.speechSynthesisLanguage = language;
|
speechConfig.speechSynthesisLanguage = language;
|
||||||
speechConfig.speechSynthesisVoiceName = voice;
|
speechConfig.speechSynthesisVoiceName = voice;
|
||||||
if (use_custom_tts && custom_tts_endpoint) {
|
if (use_custom_tts && custom_tts_endpoint) {
|
||||||
speechConfig.endpointId = custom_tts_endpoint;
|
speechConfig.endpointId = custom_tts_endpoint;
|
||||||
}
|
}
|
||||||
if (use_custom_tts && !content.startsWith('<speak')) {
|
|
||||||
/**
|
|
||||||
* Note: it seems that to use custom voice ssml is required with the voice attribute
|
|
||||||
* Otherwise sending plain text we get "Voice does not match"
|
|
||||||
*/
|
|
||||||
content = `<speak>${text}</speak>`;
|
|
||||||
}
|
|
||||||
speechConfig.speechSynthesisOutputFormat = trimSilence ?
|
speechConfig.speechSynthesisOutputFormat = trimSilence ?
|
||||||
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
|
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
|
||||||
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||||
@@ -424,14 +454,6 @@ const synthMicrosoft = async(logger, {
|
|||||||
}
|
}
|
||||||
const synthesizer = new SpeechSynthesizer(speechConfig);
|
const synthesizer = new SpeechSynthesizer(speechConfig);
|
||||||
|
|
||||||
if (content.startsWith('<speak>')) {
|
|
||||||
/* microsoft enforces some properties and uses voice xml element so if the user did not supply do it for them */
|
|
||||||
const words = content.slice(7, -8).trim().replace(/(\r\n|\n|\r)/gm, ' ');
|
|
||||||
// eslint-disable-next-line max-len
|
|
||||||
content = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="${language}"><voice name="${voice}">${words}</voice></speak>`;
|
|
||||||
logger.info({content}, 'synthMicrosoft');
|
|
||||||
}
|
|
||||||
|
|
||||||
return new Promise((resolve, reject) => {
|
return new Promise((resolve, reject) => {
|
||||||
const speakAsync = content.startsWith('<speak') ?
|
const speakAsync = content.startsWith('<speak') ?
|
||||||
synthesizer.speakSsmlAsync.bind(synthesizer) :
|
synthesizer.speakSsmlAsync.bind(synthesizer) :
|
||||||
@@ -607,14 +629,18 @@ const synthCustomVendor = async(logger, {credentials, stats, language, voice, te
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthElevenlabs = async(logger, {credentials, options, stats, language, voice, text, renderForCaching}) => {
|
const synthElevenlabs = async(logger, {
|
||||||
|
credentials, options, stats, voice, text, renderForCaching, disableTtsStreaming
|
||||||
|
}) => {
|
||||||
const {api_key, model_id, options: credOpts} = credentials;
|
const {api_key, model_id, options: credOpts} = credentials;
|
||||||
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
const opts = !!options && Object.keys(options).length !== 0 ? options : JSON.parse(credOpts || '{}');
|
||||||
|
|
||||||
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
|
||||||
if (process.env.JAMBONES_ELEVENLABS_STREAMING && !renderForCaching) {
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
let params = '';
|
let params = '';
|
||||||
params += `{api_key=${api_key}`;
|
params += `{api_key=${api_key}`;
|
||||||
|
params += ',vendor=elevenlabs';
|
||||||
|
params += `,voice=${voice}`;
|
||||||
params += `,model_id=${model_id}`;
|
params += `,model_id=${model_id}`;
|
||||||
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
|
params += `,optimize_streaming_latency=${opts.optimize_streaming_latency || 2}`;
|
||||||
params += ',write_cache_file=1';
|
params += ',write_cache_file=1';
|
||||||
@@ -656,8 +682,25 @@ const synthElevenlabs = async(logger, {credentials, options, stats, language, vo
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
const synthWhisper = async(logger, {credentials, stats, voice, text, renderForCaching, disableTtsStreaming}) => {
|
||||||
const {api_key, model_id, baseURL, timeout} = credentials;
|
const {api_key, model_id, baseURL, timeout, speed} = credentials;
|
||||||
|
/* if the env is set to stream then bag out, unless we are specifically rendering to generate a cache file */
|
||||||
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '';
|
||||||
|
params += `{api_key=${api_key}`;
|
||||||
|
params += `,model_id=${model_id}`;
|
||||||
|
params += ',vendor=whisper';
|
||||||
|
params += `,voice=${voice}`;
|
||||||
|
params += ',write_cache_file=1';
|
||||||
|
if (speed) params += `,speed=${speed}`;
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
const openai = new OpenAI.OpenAI({
|
const openai = new OpenAI.OpenAI({
|
||||||
apiKey: api_key,
|
apiKey: api_key,
|
||||||
@@ -679,10 +722,24 @@ const synthWhisper = async(logger, {credentials, stats, voice, text}) => {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
const synthDeepgram = async(logger, {credentials, stats, model, text}) => {
|
const synthDeepgram = async(logger, {credentials, stats, model, text, renderForCaching, disableTtsStreaming}) => {
|
||||||
const {api_key} = credentials;
|
const {api_key} = credentials;
|
||||||
|
if (!process.env.JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
|
||||||
|
let params = '';
|
||||||
|
params += `{api_key=${api_key}`;
|
||||||
|
params += ',vendor=deepgram';
|
||||||
|
params += `,voice=${model}`;
|
||||||
|
params += ',write_cache_file=1';
|
||||||
|
params += '}';
|
||||||
|
|
||||||
|
return {
|
||||||
|
filePath: `say:${params}${text.replace(/\n/g, ' ')}`,
|
||||||
|
servedFromCache: false,
|
||||||
|
rtt: 0
|
||||||
|
};
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
const post = bent('https://api.beta.deepgram.com', 'POST', 'buffer', {
|
const post = bent('https://api.deepgram.com', 'POST', 'buffer', {
|
||||||
'Authorization': `Token ${api_key}`,
|
'Authorization': `Token ${api_key}`,
|
||||||
'Accept': 'audio/mpeg',
|
'Accept': 'audio/mpeg',
|
||||||
'Content-Type': 'application/json'
|
'Content-Type': 'application/json'
|
||||||
|
|||||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "0.0.39",
|
"version": "0.0.48",
|
||||||
"lockfileVersion": 2,
|
"lockfileVersion": 2,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "0.0.39",
|
"version": "0.0.48",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@aws-sdk/client-polly": "^3.496.0",
|
"@aws-sdk/client-polly": "^3.496.0",
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "0.0.39",
|
"version": "0.0.48",
|
||||||
"description": "TTS-related speech utilities for jambonz",
|
"description": "TTS-related speech utilities for jambonz",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"author": "Dave Horton",
|
"author": "Dave Horton",
|
||||||
|
|||||||
+58
-1
@@ -188,6 +188,7 @@ test('Azure speech synth tests', async(t) => {
|
|||||||
language: 'en-US',
|
language: 'en-US',
|
||||||
voice: 'en-US-ChristopherNeural',
|
voice: 'en-US-ChristopherNeural',
|
||||||
text: longText,
|
text: longText,
|
||||||
|
renderForCaching: true
|
||||||
});
|
});
|
||||||
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
||||||
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
|
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
|
||||||
@@ -203,6 +204,7 @@ test('Azure speech synth tests', async(t) => {
|
|||||||
language: 'en-US',
|
language: 'en-US',
|
||||||
voice: 'en-US-ChristopherNeural',
|
voice: 'en-US-ChristopherNeural',
|
||||||
text: longText,
|
text: longText,
|
||||||
|
renderForCaching: true
|
||||||
});
|
});
|
||||||
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
|
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
@@ -212,6 +214,58 @@ test('Azure speech synth tests', async(t) => {
|
|||||||
client.quit();
|
client.quit();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('Azure SSML tests', async(t) => {
|
||||||
|
const fn = require('..');
|
||||||
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
|
|
||||||
|
if (!process.env.MICROSOFT_API_KEY || !process.env.MICROSOFT_REGION) {
|
||||||
|
t.pass('skipping Microsoft speech synth tests since MICROSOFT_API_KEY or MICROSOFT_REGION not provided');
|
||||||
|
return t.end();
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
const text = `<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="en-US">
|
||||||
|
<voice name="en-US-JennyMultilingualNeural">
|
||||||
|
<mstts:express-as style="cheerful" styledegree="2">That'd be just amazing!
|
||||||
|
</mstts:express-as>
|
||||||
|
</voice>
|
||||||
|
</speak>`;
|
||||||
|
|
||||||
|
let opts = await synthAudio(stats, {
|
||||||
|
vendor: 'microsoft',
|
||||||
|
credentials: {
|
||||||
|
api_key: process.env.MICROSOFT_API_KEY,
|
||||||
|
region: process.env.MICROSOFT_REGION,
|
||||||
|
},
|
||||||
|
language: 'en-US',
|
||||||
|
voice: 'en-US-ChristopherNeural',
|
||||||
|
text,
|
||||||
|
renderForCaching: true
|
||||||
|
});
|
||||||
|
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
||||||
|
if (process.env.JAMBONES_HTTP_PROXY_IP && process.env.JAMBONES_HTTP_PROXY_PORT) {
|
||||||
|
t.pass('successfully used proxy to reach microsoft tts service');
|
||||||
|
}
|
||||||
|
|
||||||
|
opts = await synthAudio(stats, {
|
||||||
|
vendor: 'microsoft',
|
||||||
|
credentials: {
|
||||||
|
api_key: process.env.MICROSOFT_API_KEY,
|
||||||
|
region: process.env.MICROSOFT_REGION,
|
||||||
|
},
|
||||||
|
language: 'en-US',
|
||||||
|
voice: 'en-US-ChristopherNeural',
|
||||||
|
text,
|
||||||
|
renderForCaching: true
|
||||||
|
});
|
||||||
|
t.ok(opts.servedFromCache, `successfully retrieved microsoft audio from cache ${opts.filePath}`);
|
||||||
|
} catch (err) {
|
||||||
|
console.error(err);
|
||||||
|
t.end(err);
|
||||||
|
}
|
||||||
|
client.quit();
|
||||||
|
});
|
||||||
|
|
||||||
|
|
||||||
test('Azure custom voice speech synth tests', async(t) => {
|
test('Azure custom voice speech synth tests', async(t) => {
|
||||||
const fn = require('..');
|
const fn = require('..');
|
||||||
const {synthAudio, client} = fn(opts, logger);
|
const {synthAudio, client} = fn(opts, logger);
|
||||||
@@ -233,6 +287,7 @@ test('Azure custom voice speech synth tests', async(t) => {
|
|||||||
language: 'en-US',
|
language: 'en-US',
|
||||||
voice: process.env.MICROSOFT_CUSTOM_VOICE,
|
voice: process.env.MICROSOFT_CUSTOM_VOICE,
|
||||||
text,
|
text,
|
||||||
|
renderForCaching: true
|
||||||
});
|
});
|
||||||
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
t.ok(!opts.servedFromCache, `successfully synthesized microsoft audio to ${opts.filePath}`);
|
||||||
|
|
||||||
@@ -247,6 +302,7 @@ test('Azure custom voice speech synth tests', async(t) => {
|
|||||||
language: 'en-US',
|
language: 'en-US',
|
||||||
voice: process.env.MICROSOFT_CUSTOM_VOICE,
|
voice: process.env.MICROSOFT_CUSTOM_VOICE,
|
||||||
text,
|
text,
|
||||||
|
renderForCaching: true
|
||||||
});
|
});
|
||||||
t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`);
|
t.ok(opts.servedFromCache, `successfully retrieved microsoft custom voice audio from cache ${opts.filePath}`);
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
@@ -537,8 +593,9 @@ test('Deepgram speech synth tests', async(t) => {
|
|||||||
credentials: {
|
credentials: {
|
||||||
api_key: process.env.DEEPGRAM_API_KEY
|
api_key: process.env.DEEPGRAM_API_KEY
|
||||||
},
|
},
|
||||||
model: 'alpha-asteria-en-v2',
|
model: 'aura-asteria-en',
|
||||||
text,
|
text,
|
||||||
|
renderForCaching: true
|
||||||
});
|
});
|
||||||
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
t.ok(!opts.servedFromCache, `successfully synthesized deepgram audio to ${opts.filePath}`);
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user