Compare commits

...

6 Commits

Author SHA1 Message Date
Hoan Luu Huu 30f2189986 support playht3.0 (#119)
* support playht3.0

* update language

* wip

* update top_p and repetition_penalty

---------

Co-authored-by: root <root@af6633a5cfe8>
2024-10-09 12:12:59 -04:00
Hoan Luu Huu ffde303446 fixed azure transcribe crashed if both endpointId and alternativeLang… (#124)
* fixed azure transcribe crashed if both endpointId and alternativeLangs are used

* fixed review comments

Signed-off-by: root <root@af6633a5cfe8>

---------

Signed-off-by: root <root@af6633a5cfe8>
Co-authored-by: root <root@af6633a5cfe8>
2024-10-07 09:52:58 -04:00
Hoan Luu Huu 7b4520c070 Playht delete circular buffer with mutex check (#123)
Signed-off-by: root <root@af6633a5cfe8>
Co-authored-by: root <root@af6633a5cfe8>
2024-10-05 10:02:07 -04:00
Lyle Pratt 9f7a06ce56 Update README.md (#9)
Updated ElevenLabs module to include an example of how to use the module as well as links to params in ElevenLabs docs.
2024-10-03 08:06:32 -04:00
Hoan Luu Huu f7f8f52283 fixed google asr max duration exceeded or no audio raised jambonz_transcribe::error (#120)
Co-authored-by: root <root@af6633a5cfe8>
2024-10-03 08:04:40 -04:00
Dave Horton 3f06a24b5d not clearing mark memory properly (#118)
* not clearing mark memory properly

* race condition where mark followed by audio and buffers for mark not completely allocated
2024-09-30 14:44:41 -04:00
7 changed files with 108 additions and 36 deletions
+3 -3
View File
@@ -557,7 +557,7 @@ namespace {
tech_pvt->streamingPreBuffer = nullptr;
}
if (nullptr == tech_pvt->pVecMarksInInventory) {
if (tech_pvt->pVecMarksInInventory) {
delete static_cast<std::deque<std::string>*>(tech_pvt->pVecMarksInInventory);
tech_pvt->pVecMarksInInventory = nullptr;
delete static_cast<std::deque<std::string>*>(tech_pvt->pVecMarksInUse);
@@ -979,7 +979,7 @@ extern "C" {
std::deque<std::string>* pVecInventory = nullptr;
std::deque<std::string>* pVecInUse = nullptr;
std::deque<std::string>* pVecCleared = nullptr;
if (nullptr != tech_pvt->pVecMarksInUse) {
if (tech_pvt->pVecMarksInInventory && tech_pvt->pVecMarksInUse && tech_pvt->pVecMarksCleared) {
pVecInventory = static_cast<std::deque<std::string>*>(tech_pvt->pVecMarksInInventory);
pVecInUse = static_cast<std::deque<std::string>*>(tech_pvt->pVecMarksInUse);
pVecCleared = static_cast<std::deque<std::string>*>(tech_pvt->pVecMarksCleared);
@@ -1032,7 +1032,7 @@ extern "C" {
if (samplesToCopy > 0) {
vector_add(fp, data, rframe->samples);
} else if (pVecInventory != nullptr && pVecInventory->size()) {
} else if (pVecInventory && pVecInventory->size()) {
// no bidirectional audio to dub but still have some mark in inventory, send them now
auto name = pVecInventory->front();
pVecInventory->pop_front();
+42 -18
View File
@@ -72,10 +72,7 @@ public:
if (switch_true(switch_channel_get_variable(channel, "AZURE_USE_OUTPUT_FORMAT_DETAILED"))) {
speechConfig->SetOutputFormat(OutputFormat::Detailed);
}
if (nullptr != endpointId) {
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(psession), SWITCH_LOG_DEBUG, "setting endpoint id: %s\n", endpointId);
speechConfig->SetEndpointId(endpointId);
}
if (!sdkInitialized && sdkLog) {
sdkInitialized = true;
speechConfig->SetProperty(PropertyId::Speech_LogFilename, sdkLog);
@@ -104,11 +101,31 @@ public:
languages.push_back( alt_langs[i]);
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(psession), SWITCH_LOG_DEBUG, "added alternative lang %s\n", alt_langs[i]);
}
auto autoDetectSourceLanguageConfig = AutoDetectSourceLanguageConfig::FromLanguages(languages);
std::vector<std::shared_ptr<SourceLanguageConfig>> sourceLanguageConfigs;
for (const auto& language : languages) {
std::shared_ptr<SourceLanguageConfig> sourceLanguageConfig;
if (endpointId != nullptr) {
sourceLanguageConfig = SourceLanguageConfig::FromLanguage(language, endpointId);
} else {
sourceLanguageConfig = SourceLanguageConfig::FromLanguage(language);
}
sourceLanguageConfigs.push_back(sourceLanguageConfig);
}
// Create AutoDetectSourceLanguageConfig from SourceLanguageConfigs
auto autoDetectSourceLanguageConfig = AutoDetectSourceLanguageConfig::FromSourceLanguageConfigs(sourceLanguageConfigs);
m_recognizer = SpeechRecognizer::FromConfig(speechConfig, autoDetectSourceLanguageConfig, audioConfig);
}
else {
auto sourceLanguageConfig = SourceLanguageConfig::FromLanguage(lang);
std::shared_ptr<SourceLanguageConfig> sourceLanguageConfig;
if (endpointId != nullptr) {
sourceLanguageConfig = SourceLanguageConfig::FromLanguage(lang, endpointId);
} else {
sourceLanguageConfig = SourceLanguageConfig::FromLanguage(lang);
}
m_recognizer = SpeechRecognizer::FromConfig(speechConfig, sourceLanguageConfig, audioConfig);
}
@@ -487,20 +504,27 @@ extern "C" {
auto read_codec = switch_core_session_get_read_codec(session);
uint32_t sampleRate = read_codec->implementation->actual_samples_per_second;
if (bug) {
struct cap_cb* existing_cb = (struct cap_cb*) switch_core_media_bug_get_user_data(bug);
GStreamer* existing_streamer = (GStreamer*) existing_cb->streamer;
existing_cb->is_keep_alive = 0;
if (!existing_streamer->hasConfigurationChanged(channels, lang, interim, sampleRate, region, subscriptionKey)) {
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "Reuse active azure connection.\n");
try {
struct cap_cb* existing_cb = (struct cap_cb*) switch_core_media_bug_get_user_data(bug);
GStreamer* existing_streamer = (GStreamer*) existing_cb->streamer;
existing_cb->is_keep_alive = 0;
if (!existing_streamer->hasConfigurationChanged(channels, lang, interim, sampleRate, region, subscriptionKey)) {
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "Reuse active azure connection.\n");
return SWITCH_STATUS_SUCCESS;
}
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "Azure configuration is changed, destroy old and create new azure connection\n");
reaper(existing_cb);
streamer = new GStreamer(sessionId, bugname, channels, lang, interim, sampleRate, region, subscriptionKey, responseHandler);
if (!existing_cb->vad) streamer->connect();
existing_cb->streamer = streamer;
*ppUserData = existing_cb;
return SWITCH_STATUS_SUCCESS;
} catch (std::exception& e) {
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_ERROR, "%s: Error initializing gstreamer: %s.\n",
switch_channel_get_name(channel), e.what());
return SWITCH_STATUS_FALSE;
}
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "Azure configuration is changed, destroy old and create new azure connection\n");
reaper(existing_cb);
streamer = new GStreamer(sessionId, bugname, channels, lang, interim, sampleRate, region, subscriptionKey, responseHandler);
if (!existing_cb->vad) streamer->connect();
existing_cb->streamer = streamer;
*ppUserData = existing_cb;
return SWITCH_STATUS_SUCCESS;
}
int err;
switch_threadattr_t *thd_attr = NULL;
+17 -9
View File
@@ -1,11 +1,11 @@
# mod_google_tts
A Freeswitch module that allows Google Text-to-Speech API to be used as a tts provider.
A Freeswitch module that allows Eleven Labs' Text-to-Speech API to be used as a tts provider.
## API
### Commands
This freeswitch module does not add any new commands, per se. Rather, it integrates into the Freeswitch TTS interface such that it is invoked when an application uses the mod_dptools `speak` command with a tts engine of `google_tts` and a voice equal to the language code associated to one of the [supported Wavenet voices](https://cloud.google.com/text-to-speech/docs/voices)
This freeswitch module does not add any new commands, per se. Rather, it integrates into the Freeswitch TTS interface such that it is invoked when an application uses the mod_dptools `speak` command with a tts engine of `elevenlabs` and a voice equal to the language code associated to one of the [supported Eleven Labs voices](https://elevenlabs.io/docs/api-reference/query-library)
### Events
None.
@@ -13,11 +13,19 @@ None.
## Usage
When using [drachtio-fsrmf](https://www.npmjs.com/package/drachtio-fsmrf), you can access this functionality via the speak method on the 'endpoint' object.
```js
ep.speak({
ttsEngine: 'google_tts',
voice: 'en-GB-Wavenet-A',
text: 'This aggression will not stand'
});
var text = "Hello World";
await endpoint.speak({
"ttsEngine": 'elevenlabs',
"voice": "W9OIfHh5DtdYiZUcFiql",
"text": `{use_speaker_boost=1,optimize_streaming_latency=4,style=0.5,stability=0.5,similarity_boost=0.75,api_key=XXYYZZ,model_id=eleven_turbo_v2}${text}`,
});
```
## Examples
[google_tts.js](../../examples/google_tts.js)
## Options
Documentation on these options can be found in Eleven Labs API docs: [Voice Settings](https://elevenlabs.io/docs/speech-synthesis/voice-settings)
- use_speaker_boost
- optimize_streaming_latency
- style
- stability
- similarity_boost
+4
View File
@@ -233,6 +233,10 @@ static void *SWITCH_THREAD_FUNC grpc_read_thread(switch_thread_t *thread, void *
auto speech_event_type = response.speech_event_type();
if (response.has_error()) {
Status status = response.error();
//error 11 is handled in finished session, avoid sending jambonz_transcribe::error event for this here.
if (11 == status.code()) {
continue;
}
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_ERROR, "grpc_read_thread: error %s (%d)\n", status.message().c_str(), status.code()) ;
cJSON* json = cJSON_CreateObject();
cJSON_AddStringToObject(json, "type", "error");
+20
View File
@@ -14,6 +14,10 @@ static void clearPlayht(playht_t* p, int freeAll) {
if (p->seed) free(p->seed);
if (p->temperature) free(p->temperature);
if (p->voice_engine) free(p->voice_engine);
if (p->synthesize_url) free(p->synthesize_url);
if (p->language) free(p->language);
if (p->top_p) free(p->top_p);
if (p->repetition_penalty) free(p->repetition_penalty);
if (p->emotion) free(p->emotion);
if (p->voice_guidance) free(p->voice_guidance);
if (p->style_guidance) free(p->style_guidance);
@@ -36,6 +40,10 @@ static void clearPlayht(playht_t* p, int freeAll) {
p->seed = NULL;
p->temperature = NULL;
p->voice_engine = NULL;
p->synthesize_url = NULL;
p->language = NULL;
p->top_p = NULL;
p->repetition_penalty = NULL;
p->emotion = NULL;
p->voice_guidance = NULL;
p->style_guidance = NULL;
@@ -154,6 +162,18 @@ static void p_text_param_tts(switch_speech_handle_t *sh, char *param, const char
} else if (0 == strcmp(param, "voice_engine")) {
if (p->voice_engine) free(p->voice_engine);
p->voice_engine = strdup(val);
} else if (0 == strcmp(param, "synthesize_url")) {
if (p->synthesize_url) free(p->synthesize_url);
p->synthesize_url = strdup(val);
} else if (0 == strcmp(param, "language")) {
if (p->language) free(p->language);
p->language = strdup(val);
} else if (0 == strcmp(param, "top_p")) {
if (p->top_p) free(p->top_p);
p->top_p = strdup(val);
} else if (0 == strcmp(param, "repetition_penalty")) {
if (p->repetition_penalty) free(p->repetition_penalty);
p->repetition_penalty = strdup(val);
} else if (0 == strcmp(param, "emotion")) {
if (p->emotion) free(p->emotion);
p->emotion = strdup(val);
+4
View File
@@ -11,10 +11,14 @@ typedef struct playht_data {
char *seed;
char *temperature;
char *voice_engine;
char *synthesize_url;
char *language;
char *emotion;
char *voice_guidance;
char *style_guidance;
char *text_guidance;
char *top_p;
char *repetition_penalty;
/* result data */
long response_code;
+18 -6
View File
@@ -780,7 +780,7 @@ extern "C" {
}
/* format url*/
std::string url = "https://api.play.ht/api/v2/tts/stream";
std::string url = p->synthesize_url;
/* create the JSON body */
cJSON * jResult = cJSON_CreateObject();
@@ -800,9 +800,6 @@ extern "C" {
cJSON_AddNumberToObject(jResult, "speed", val);
}
}
if (p->seed) {
cJSON_AddNumberToObject(jResult, "seed", atoi(p->seed));
}
if (p->temperature) {
cJSON_AddNumberToObject(jResult, "temperature", std::strtof(p->temperature, nullptr));
}
@@ -818,6 +815,15 @@ extern "C" {
if (p->text_guidance) {
cJSON_AddNumberToObject(jResult, "text_guidance", atoi(p->text_guidance));
}
if (strcmp(p->voice_engine, "Play3.0") == 0 && p->language) {
cJSON_AddStringToObject(jResult, "language", p->language);
}
if (p->repetition_penalty) {
cJSON_AddNumberToObject(jResult, "repetition_penalty", std::strtof(p->repetition_penalty, nullptr));
}
if (p->top_p) {
cJSON_AddNumberToObject(jResult, "top_p", std::strtof(p->top_p, nullptr));
}
char *json = cJSON_PrintUnformatted(jResult);
cJSON_Delete(jResult);
@@ -889,9 +895,12 @@ extern "C" {
/* call this function to close a socket */
curl_easy_setopt(easy, CURLOPT_CLOSESOCKETFUNCTION, close_socket);
// Play3.0 voice engine doesn't need authorization
if (strcmp(p->voice_engine, "Play3.0") != 0) {
conn->hdr_list = curl_slist_append(conn->hdr_list, api_key_stream.str().c_str());
conn->hdr_list = curl_slist_append(conn->hdr_list, user_id_stream.str().c_str());
}
conn->hdr_list = curl_slist_append(conn->hdr_list, api_key_stream.str().c_str());
conn->hdr_list = curl_slist_append(conn->hdr_list, user_id_stream.str().c_str());
conn->hdr_list = curl_slist_append(conn->hdr_list, "Accept: audio/mpeg");
conn->hdr_list = curl_slist_append(conn->hdr_list, "Content-Type: application/json");
curl_easy_setopt(easy, CURLOPT_HTTPHEADER, conn->hdr_list);
@@ -959,8 +968,11 @@ extern "C" {
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_DEBUG, "playht_speech_flush_tts, download complete? %s\n", download_complete ? "yes" : "no") ;
ConnInfo_t *conn = (ConnInfo_t *) p->conn;
CircularBuffer_t *cBuffer = (CircularBuffer_t *) p->circularBuffer;
// In multi threads, only delete the circular buffer when write and read buffer action finished using it.
switch_mutex_lock(p->mutex);
delete cBuffer;
p->circularBuffer = nullptr ;
switch_mutex_unlock(p->mutex);
if (conn) {
conn->flushed = true;