support google tts configuraiton (#149)

This commit is contained in:
Hoan Luu Huu
2026-07-28 20:58:26 -04:00
committed by GitHub
parent 04d1fb548b
commit 6a6de9b7a1
2 changed files with 163 additions and 0 deletions
+38
View File
@@ -383,6 +383,32 @@ const synthPolly = async(createHash, retrieveHash, logger,
}
};
/* google AudioConfig settings we support, as [google camelCase name, freeswitch param name] */
const GOOGLE_AUDIO_SETTINGS = [
['speakingRate', 'speaking_rate'],
['pitch', 'pitch'],
['volumeGainDb', 'volume_gain_db']
];
/**
* Extract google AudioConfig settings from the synthesizer options. They may be supplied
* nested under an audioConfig property (mirroring google's AudioConfig object) or flat at
* the top level, and either google's camelCase or snake_case names are accepted.
* @see https://cloud.google.com/text-to-speech/docs/reference/rest/v1/text/synthesize#AudioConfig
* @returns object keyed by google's camelCase names, holding only valid numeric settings
*/
const googleAudioConfig = (options) => {
const provided = {...options, ...(options?.audioConfig || {})};
const audioConfig = {};
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
const value = provided[name] ?? provided[snakeName];
/* note: 0 is meaningful for pitch and volumeGainDb, so check for absence explicitly */
if (value === undefined || value === null || value === '') continue;
const num = Number(value);
if (Number.isFinite(num)) audioConfig[name] = num;
}
return audioConfig;
};
const synthGoogle = async(logger, {
credentials, stats, language, voice, gender, key, text, model, options, instructions,
@@ -418,6 +444,15 @@ const synthGoogle = async(logger, {
// comma is used to separate parameters in freeswitch tts module
const prompt = options?.prompt || instructions;
if (prompt) params += `,prompt=${prompt.replace(/\n/g, ' ').replace(/,/g, ';')}`;
/**
* AudioConfig settings. Note google only honors these in some api modes:
* tts applies all of them, live (HD voices) applies speakingRate only,
* and gemini ignores them entirely (use prompt instead for style control).
*/
const audioSettings = googleAudioConfig(options);
for (const [name, snakeName] of GOOGLE_AUDIO_SETTINGS) {
if (name in audioSettings) params += `,${snakeName}=${audioSettings[name]}`;
}
params += '}';
return {
@@ -486,6 +521,9 @@ const synthGoogle = async(logger, {
sampleRate = 8000;
}
/* gemini voices do not support the AudioConfig settings; they use prompt for style control */
if (!isGemini) Object.assign(audioConfig, googleAudioConfig(options));
const opts = { input, voice: voiceParams, audioConfig };
try {