mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-03 23:33:59 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2e33e76cd5 | ||
|
|
44d8af2a96 | ||
|
|
b530db9a62 | ||
|
|
4c166c8eb4 | ||
|
|
8246dbea21 | ||
|
|
0084f6a468 | ||
|
|
98f679f43a | ||
|
|
7e7841b5ff | ||
|
|
75ce537db1 | ||
|
|
830be783b8 | ||
|
|
38c3219425 | ||
|
|
e37b96a9c2 |
+31
-3
@@ -33,6 +33,24 @@ const debug = require('debug')('jambonz:realtimedb-helpers');
|
|||||||
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
|
const EXPIRES = (process.env.JAMBONES_TTS_CACHE_DURATION_MINS || 4 * 60) * 60; // cache tts for 4 hours
|
||||||
const TMP_FOLDER = '/tmp';
|
const TMP_FOLDER = '/tmp';
|
||||||
|
|
||||||
|
|
||||||
|
const trimTrailingSilence = (buffer) => {
|
||||||
|
assert.ok(buffer instanceof Buffer, 'trimTrailingSilence - argument is not a Buffer');
|
||||||
|
|
||||||
|
let offset = buffer.length;
|
||||||
|
while (offset > 0) {
|
||||||
|
// Get 16-bit value from the buffer (read in reverse)
|
||||||
|
const value = buffer.readUInt16BE(offset - 2);
|
||||||
|
if (value !== 0) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
offset -= 2;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Trim the silence from the end
|
||||||
|
return offset === buffer.length ? buffer : buffer.subarray(0, offset);
|
||||||
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Synthesize speech to an mp3 file, and also cache the generated speech
|
* Synthesize speech to an mp3 file, and also cache the generated speech
|
||||||
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
|
* in redis (base64 format) for 24 hours so as to avoid unnecessarily paying
|
||||||
@@ -104,7 +122,12 @@ async function synthAudio(client, logger, stats, { account_sid,
|
|||||||
text
|
text
|
||||||
});
|
});
|
||||||
let filePath;
|
let filePath;
|
||||||
if (['nuance', 'nvidia'].includes(vendor)) {
|
if (['nuance', 'nvidia'].includes(vendor) ||
|
||||||
|
(
|
||||||
|
process.env.JAMBONES_TTS_TRIM_SILENCE &&
|
||||||
|
['microsoft', 'azure'].includes(vendor)
|
||||||
|
)
|
||||||
|
) {
|
||||||
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.r8`;
|
||||||
}
|
}
|
||||||
else filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.mp3`;
|
else filePath = `${TMP_FOLDER}/${key.replace('tts:', `tts-${salt || ''}`)}.mp3`;
|
||||||
@@ -284,6 +307,7 @@ const synthMicrosoft = async(logger, {
|
|||||||
}) => {
|
}) => {
|
||||||
try {
|
try {
|
||||||
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
|
const {api_key: apiKey, region, use_custom_tts, custom_tts_endpoint} = credentials;
|
||||||
|
const trimSilence = filePath.endsWith('.r8');
|
||||||
let content = text;
|
let content = text;
|
||||||
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
const speechConfig = SpeechConfig.fromSubscription(apiKey, region);
|
||||||
speechConfig.speechSynthesisLanguage = language;
|
speechConfig.speechSynthesisLanguage = language;
|
||||||
@@ -297,7 +321,9 @@ const synthMicrosoft = async(logger, {
|
|||||||
*/
|
*/
|
||||||
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
if (!content.startsWith('<speak')) content = `<speak>${text}</speak>`;
|
||||||
}
|
}
|
||||||
speechConfig.speechSynthesisOutputFormat = SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
speechConfig.speechSynthesisOutputFormat = trimSilence ?
|
||||||
|
SpeechSynthesisOutputFormat.Raw8Khz16BitMonoPcm :
|
||||||
|
SpeechSynthesisOutputFormat.Audio16Khz32KBitRateMonoMp3;
|
||||||
const synthesizer = new SpeechSynthesizer(speechConfig);
|
const synthesizer = new SpeechSynthesizer(speechConfig);
|
||||||
|
|
||||||
if (content.startsWith('<speak>')) {
|
if (content.startsWith('<speak>')) {
|
||||||
@@ -323,7 +349,9 @@ const synthMicrosoft = async(logger, {
|
|||||||
reject(cancellation.errorDetails);
|
reject(cancellation.errorDetails);
|
||||||
break;
|
break;
|
||||||
case ResultReason.SynthesizingAudioCompleted:
|
case ResultReason.SynthesizingAudioCompleted:
|
||||||
resolve(Buffer.from(result.audioData));
|
let buffer = Buffer.from(result.audioData);
|
||||||
|
if (trimSilence) buffer = trimTrailingSilence(buffer);
|
||||||
|
resolve(buffer);
|
||||||
synthesizer.close();
|
synthesizer.close();
|
||||||
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
stats.increment('tts.count', ['vendor:microsoft', 'accepted:yes']);
|
||||||
break;
|
break;
|
||||||
|
|||||||
Generated
+1478
-1375
File diff suppressed because it is too large
Load Diff
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@jambonz/speech-utils",
|
"name": "@jambonz/speech-utils",
|
||||||
"version": "0.0.15",
|
"version": "0.0.19",
|
||||||
"description": "TTS-related speech utilities for jambonz",
|
"description": "TTS-related speech utilities for jambonz",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"author": "Dave Horton",
|
"author": "Dave Horton",
|
||||||
@@ -24,7 +24,7 @@
|
|||||||
},
|
},
|
||||||
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
"homepage": "https://github.com/jambonz/speech-utils#readme",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@aws-sdk/client-polly": "^3.303.0",
|
"@aws-sdk/client-polly": "^3.359.0",
|
||||||
"@google-cloud/text-to-speech": "^4.2.1",
|
"@google-cloud/text-to-speech": "^4.2.1",
|
||||||
"@grpc/grpc-js": "^1.8.13",
|
"@grpc/grpc-js": "^1.8.13",
|
||||||
"bent": "^7.3.12",
|
"bent": "^7.3.12",
|
||||||
|
|||||||
Reference in New Issue
Block a user