Compare commits

..
10 Commits
Author SHA1 Message Date
Dave Horton 6fa68bc712 more changes trying to get AWS V3 sdk properly configured 2023-03-21 08:20:29 -04:00
Dave Horton d914b26cac change to PollyClient 2023-03-21 07:52:06 -04:00
Dave Horton 7c7be6bbb1 bugfix: AWS polly 2023-03-20 15:33:00 -04:00
Dave Horton 46127bb763 bump version 2023-03-15 08:52:57 -04:00
Dave Horton 17305325ff Merge pull request #13 from jambonz/feat/aws-v3
fix: custom tts support multiple audio types
2023-03-15 08:51:25 -04:00
Quan HL 0e883cf82a fix: custom tts support multiple audio types 2023-03-15 13:43:35 +07:00
Dave Horton 2f3a766713 Merge pull request #12 from jambonz/#10
fix: JAMBONES_TTS_CACHE_DURATION_MINS in minute unit
2023-03-09 19:45:28 -05:00
Quan HL b429645ec2 fix: JAMBONES_TTS_CACHE_DURATION_MINS in minute unit 2023-03-10 06:41:22 +07:00
Quan HL 0becf33bab fix: JAMBONES_TTS_CACHE_DURATION_MINS in minute unit 2023-03-10 06:41:07 +07:00
Quan HL 3d3875741c fix: JAMBONES_TTS_CACHE_DURATION_MINS in minute unit 2023-03-10 06:31:17 +07:00
4 changed files with 47 additions and 14 deletions
+44 -10
View File
@@ -32,7 +32,7 @@ const {
const {SynthesizeSpeechRequest} = require('../stubs/riva/proto/riva_tts_pb');
const {AudioEncoding} = require('../stubs/riva/proto/riva_audio_pb');
const debug = require('debug')('jambonz:realtimedb-helpers');
const EXPIRES = process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 3600 * 24; // cache tts for 24 hours
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 24 * 60) * 60; // cache tts for 24 hours
const TMP_FOLDER = '/tmp';
/**
@@ -155,7 +155,8 @@ async function synthAudio(client, logger, stats, { account_sid,
audioBuffer = await synthWellSaid(logger, {credentials, stats, language, voice, text, filePath});
break;
case vendor.startsWith('custom') ? vendor : 'cant_match_value':
audioBuffer = await synthCustomVendor(logger, {credentials, stats, language, voice, text});
({ audioBuffer, filePath } = await synthCustomVendor(logger,
{credentials, stats, language, voice, text, filePath}));
break;
default:
assert(`synthAudio: unsupported speech vendor ${vendor}`);
@@ -183,7 +184,14 @@ async function synthAudio(client, logger, stats, { account_sid,
const synthPolly = async(logger, {credentials, stats, language, voice, engine, text}) => {
try {
const polly = new PollyClient(credentials);
const {region, accessKeyId, secretAccessKey} = credentials;
const polly = new PollyClient({
region,
credentials: {
accessKeyId,
secretAccessKey
}
});
const opts = {
Engine: engine,
OutputFormat: 'mp3',
@@ -440,30 +448,56 @@ const synthNvidia = async(client, logger, {credentials, stats, language, voice,
};
// CustomVendor accept only mp3
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text}) => {
const synthCustomVendor = async(logger, {credentials, stats, language, voice, text, filePath}) => {
const {vendor, auth_token, custom_tts_url} = credentials;
try {
const post = bent('POST', 'buffer', {
const post = bent('POST', {
'Authorization': `Bearer ${auth_token}`,
'Accept': 'audio/mpeg',
'Content-Type': 'application/json'
});
const mp3 = await post(custom_tts_url, {
const response = await post(custom_tts_url, {
language,
format: 'audio/mpeg',
voice,
type: text.startsWith('<speak>') ? 'ssml' : 'text',
text
});
return mp3;
const regex = /\.[^\.]*$/g;
const mime = response.headers['content-type'];
const buffer = await response.arrayBuffer();
return {
audioBuffer: buffer,
filePath: filePath.replace(regex, getFileExtFromMime(mime))
};
} catch (err) {
logger.info({err}, `Vendor ${vendor} returned error`);
throw err;
}
};
const getFileExtFromMime = (mime) => {
switch (mime) {
case 'audio/wav':
case 'audio/x-wav':
return '.wav';
case /audio\/l16.*rate=8000/.test(mime) ? mime : 'cant match value':
return '.r8';
case /audio\/l16.*rate=16000/.test(mime) ? mime : 'cant match value':
return '.r16';
case /audio\/l16.*rate=24000/.test(mime) ? mime : 'cant match value':
return '.r24';
case /audio\/l16.*rate=32000/.test(mime) ? mime : 'cant match value':
return '.r32';
case /audio\/l16.*rate=48000/.test(mime) ? mime : 'cant match value':
return '.r48';
case 'audio/mpeg':
case 'audio/mp3':
return '.mp3';
default:
return '.wav';
}
};
module.exports = synthAudio;
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.2",
"version": "0.0.5",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "0.0.2",
"version": "0.0.5",
"license": "MIT",
"dependencies": {
"@aws-sdk/client-polly": "^3.276.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "0.0.2",
"version": "0.0.6",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
-1
View File
@@ -348,7 +348,6 @@ test('Custom Vendor speech synth tests', async(t) => {
let obj = await getJSON(`http://127.0.0.1:3100/lastRequest/somethingnew`);
t.ok(obj.headers.Authorization == 'Bearer some_jwt_token', 'Custom Vendor Authentication Header is correct');
t.ok(obj.body.language == 'en-US', 'Custom Vendor Language is correct');
t.ok(obj.body.format == 'audio/mpeg', 'Custom Vendor format is correct');
t.ok(obj.body.voice == 'English-US.Female-1', 'Custom Vendor voice is correct');
t.ok(obj.body.type == 'text', 'Custom Vendor type is correct');
t.ok(obj.body.text == 'This is a test. This is only a test', 'Custom Vendor text is correct');