mirror of
https://github.com/jambonz/speech-utils.git
synced 2026-10-09 21:14:25 +00:00
support google tts configuraiton (#149)
This commit is contained in:
@@ -383,6 +383,32 @@ const synthPolly = async(createHash, retrieveHash, logger,
|
||||
}
|
||||
};
|
||||
|
||||
/* google AudioConfig settings we support, as [google camelCase name, freeswitch param name] */
|
||||
const GOOGLE_AUDIO_SETTINGS = [
|
||||
['speakingRate', 'speaking_rate'],
|
||||
['pitch', 'pitch'],
|
||||
['volumeGainDb', 'volume_gain_db']
|
||||
];
|
||||
|
||||
/**
|
||||
* Extract google AudioConfig settings from the synthesizer options. They may be supplied
|
||||
* nested under an audioConfig property (mirroring google's AudioConfig object) or flat at
|
||||
* the top level, and either google's camelCase or snake_case names are accepted.
|
||||
* @see https://cloud.google.com/text-to-speech/docs/reference/rest/v1/text/synthesize#AudioConfig
|
||||
* @returns object keyed by google's camelCase names, holding only valid numeric settings
|
||||
*/
|
||||
const googleAudioConfig = (options) => {
|
||||
const provided = {...options, ...(options?.audioConfig || {})};
|
||||
const audioConfig = {};
|
||||
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
|
||||
const value = provided[name] ?? provided[snakeName];
|
||||
/* note: 0 is meaningful for pitch and volumeGainDb, so check for absence explicitly */
|
||||
if (value === undefined || value === null || value === '') continue;
|
||||
const num = Number(value);
|
||||
if (Number.isFinite(num)) audioConfig[name] = num;
|
||||
}
|
||||
return audioConfig;
|
||||
};
|
||||
|
||||
const synthGoogle = async(logger, {
|
||||
credentials, stats, language, voice, gender, key, text, model, options, instructions,
|
||||
@@ -418,6 +444,15 @@ const synthGoogle = async(logger, {
|
||||
// comma is used to separate parameters in freeswitch tts module
|
||||
const prompt = options?.prompt || instructions;
|
||||
if (prompt) params += `,prompt=${prompt.replace(/\n/g, ' ').replace(/,/g, ';')}`;
|
||||
/**
|
||||
* AudioConfig settings. Note google only honors these in some api modes:
|
||||
* tts applies all of them, live (HD voices) applies speakingRate only,
|
||||
* and gemini ignores them entirely (use prompt instead for style control).
|
||||
*/
|
||||
const audioSettings = googleAudioConfig(options);
|
||||
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
|
||||
if (name in audioSettings) params += `,${snakeName}=${audioSettings[name]}`;
|
||||
}
|
||||
params += '}';
|
||||
|
||||
return {
|
||||
@@ -486,6 +521,9 @@ const synthGoogle = async(logger, {
|
||||
sampleRate = 8000;
|
||||
}
|
||||
|
||||
/* gemini voices do not support the AudioConfig settings; they use prompt for style control */
|
||||
if (!isGemini) Object.assign(audioConfig, googleAudioConfig(options));
|
||||
|
||||
const opts = { input, voice: voiceParams, audioConfig };
|
||||
|
||||
try {
|
||||
|
||||
Reference in New Issue
Block a user