mirror of
https://github.com/jambonz/freeswitch-modules.git
synced 2026-01-25 02:08:27 +00:00
Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b003ab0875 | |||
| 81ceddf3d2 | |||
| 110a12d5a5 | |||
| fe1e4dcf11 | |||
| f828171b3b | |||
| e717ca7dd3 |
@@ -43,16 +43,16 @@ public:
|
||||
const char *sessionId,
|
||||
const char *bugname,
|
||||
u_int16_t channels,
|
||||
char *lang,
|
||||
char *lang,
|
||||
int interim,
|
||||
uint32_t samples_per_second,
|
||||
const char* region,
|
||||
const char* awsAccessKeyId,
|
||||
const char* region,
|
||||
const char* awsAccessKeyId,
|
||||
const char* awsSecretAccessKey,
|
||||
const char* awsSessionToken,
|
||||
responseHandler_t responseHandler
|
||||
) : m_sessionId(sessionId), m_bugname(bugname), m_finished(false), m_interim(interim), m_finishing(false), m_connected(false), m_connecting(false),
|
||||
m_packets(0), m_responseHandler(responseHandler), m_pStream(nullptr),
|
||||
m_packets(0), m_responseHandler(responseHandler), m_pStream(nullptr),
|
||||
m_audioBuffer(320 * (samples_per_second == 8000 ? 1 : 2), 15) {
|
||||
Aws::Client::ClientConfiguration config;
|
||||
if (region != nullptr && strlen(region) > 0) config.region = region;
|
||||
@@ -71,7 +71,7 @@ public:
|
||||
else {
|
||||
m_client = Aws::MakeUnique<TranscribeStreamingServiceClient>(ALLOC_TAG, config);
|
||||
}
|
||||
|
||||
|
||||
m_handler.SetTranscriptEventCallback([this](const TranscriptEvent& ev)
|
||||
{
|
||||
switch_core_session_t* psession = switch_core_session_locate(m_sessionId.c_str());
|
||||
@@ -132,7 +132,7 @@ public:
|
||||
|
||||
// send any buffered audio
|
||||
int nFrames = m_audioBuffer.getNumItems();
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_DEBUG, "GStreamer %p got stream ready, %d buffered frames\n", this, nFrames);
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_DEBUG, "GStreamer %p got stream ready, %d buffered frames\n", this, nFrames);
|
||||
if (nFrames) {
|
||||
char *p;
|
||||
do {
|
||||
@@ -142,19 +142,19 @@ public:
|
||||
}
|
||||
} while (p);
|
||||
}
|
||||
|
||||
|
||||
switch_core_session_rwunlock(psession);
|
||||
}
|
||||
};
|
||||
auto OnResponseCallback = [this](const TranscribeStreamingServiceClient* pClient,
|
||||
const Model::StartStreamTranscriptionRequest& request,
|
||||
const Model::StartStreamTranscriptionOutcome& outcome,
|
||||
auto OnResponseCallback = [this](const TranscribeStreamingServiceClient* pClient,
|
||||
const Model::StartStreamTranscriptionRequest& request,
|
||||
const Model::StartStreamTranscriptionOutcome& outcome,
|
||||
const std::shared_ptr<const Aws::Client::AsyncCallerContext>& context)
|
||||
{
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_DEBUG, "GStreamer %p stream got final response\n", this);
|
||||
switch_core_session_t* psession = switch_core_session_locate(m_sessionId.c_str());
|
||||
if (psession) {
|
||||
if (!outcome.IsSuccess()) {
|
||||
if (!outcome.IsSuccess()) {
|
||||
const TranscribeStreamingServiceError& err = outcome.GetError();
|
||||
auto message = err.GetMessage();
|
||||
auto exception = err.GetExceptionName();
|
||||
@@ -186,7 +186,7 @@ public:
|
||||
|
||||
|
||||
~GStreamer() {
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_DEBUG, "GStreamer::~GStreamer wrote %u packets %p\n", m_packets, this);
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_DEBUG, "GStreamer::~GStreamer wrote %u packets %p\n", m_packets, this);
|
||||
}
|
||||
|
||||
bool write(void* data, uint32_t datalen) {
|
||||
@@ -228,7 +228,7 @@ public:
|
||||
bool shutdownInitiated = false;
|
||||
while (true) {
|
||||
std::unique_lock<std::mutex> lk(m_mutex);
|
||||
m_cond.wait(lk, [&, this] {
|
||||
m_cond.wait(lk, [&, this] {
|
||||
return (!m_deqAudio.empty() && !m_finishing) || m_transcript.TranscriptHasBeenSet() || m_finished || (m_finishing && !shutdownInitiated);
|
||||
});
|
||||
|
||||
@@ -264,7 +264,7 @@ public:
|
||||
m_responseHandler(psession, s.str().c_str(), m_bugname.c_str());
|
||||
}
|
||||
TranscriptEvent empty;
|
||||
m_transcript = empty;
|
||||
m_transcript = empty;
|
||||
|
||||
switch_core_session_rwunlock(psession);
|
||||
}
|
||||
@@ -362,7 +362,7 @@ extern "C" {
|
||||
const char* secretAccessKey = std::getenv("AWS_SECRET_ACCESS_KEY");
|
||||
const char* region = std::getenv("AWS_REGION");
|
||||
if (NULL == accessKeyId && NULL == secretAccessKey) {
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_NOTICE,
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_NOTICE,
|
||||
"\"AWS_ACCESS_KEY_ID\" and/or \"AWS_SECRET_ACCESS_KEY\" env var not set; authentication will expect channel variables of same names to be set\n");
|
||||
}
|
||||
else {
|
||||
@@ -370,7 +370,7 @@ extern "C" {
|
||||
|
||||
}
|
||||
Aws::SDKOptions options;
|
||||
/*
|
||||
/*
|
||||
options.loggingOptions.logLevel = Aws::Utils::Logging::LogLevel::Trace;
|
||||
|
||||
Aws::Utils::Logging::InitializeAWSLogging(
|
||||
@@ -381,7 +381,7 @@ extern "C" {
|
||||
|
||||
return SWITCH_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
switch_status_t aws_transcribe_cleanup() {
|
||||
Aws::SDKOptions options;
|
||||
/*
|
||||
@@ -394,7 +394,7 @@ extern "C" {
|
||||
}
|
||||
|
||||
// start transcribe on a channel
|
||||
switch_status_t aws_transcribe_session_init(switch_core_session_t *session, responseHandler_t responseHandler,
|
||||
switch_status_t aws_transcribe_session_init(switch_core_session_t *session, responseHandler_t responseHandler,
|
||||
uint32_t samples_per_second, uint32_t channels, char* lang, int interim, char* bugname, void **ppUserData
|
||||
) {
|
||||
switch_status_t status = SWITCH_STATUS_SUCCESS;
|
||||
@@ -433,7 +433,7 @@ extern "C" {
|
||||
std::getenv("AWS_REGION")) {
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "Using env vars for aws authentication\n");
|
||||
strncpy(cb->awsAccessKeyId, std::getenv("AWS_ACCESS_KEY_ID"), 128);
|
||||
strncpy(cb->awsSecretAccessKey, std::getenv("AWS_SECRET_ACCESS_KEY"), 128);
|
||||
strncpy(cb->awsSecretAccessKey, std::getenv("AWS_SECRET_ACCESS_KEY"), 128);
|
||||
strncpy(cb->region, std::getenv("AWS_REGION"), MAX_REGION);
|
||||
}
|
||||
else {
|
||||
@@ -445,7 +445,7 @@ extern "C" {
|
||||
if (switch_mutex_init(&cb->mutex, SWITCH_MUTEX_NESTED, pool) != SWITCH_STATUS_SUCCESS) {
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_ERROR, "Error initializing mutex\n");
|
||||
status = SWITCH_STATUS_FALSE;
|
||||
goto done;
|
||||
goto done;
|
||||
}
|
||||
|
||||
cb->interim = interim;
|
||||
@@ -455,7 +455,7 @@ extern "C" {
|
||||
if (sampleRate != 8000) {
|
||||
cb->resampler = speex_resampler_init(1, sampleRate, 16000, SWITCH_RESAMPLE_QUALITY, &err);
|
||||
if (0 != err) {
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_ERROR, "%s: Error initializing resampler: %s.\n",
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_ERROR, "%s: Error initializing resampler: %s.\n",
|
||||
switch_channel_get_name(channel), speex_resampler_strerror(err));
|
||||
status = SWITCH_STATUS_FALSE;
|
||||
goto done;
|
||||
@@ -488,7 +488,7 @@ extern "C" {
|
||||
switch_vad_set_param(cb->vad, "silence_ms", silence_ms);
|
||||
switch_vad_set_param(cb->vad, "voice_ms", voice_ms);
|
||||
switch_vad_set_param(cb->vad, "debug", debug);
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "%s: delaying connection until vad, voice_ms %d, mode %d\n",
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG, "%s: delaying connection until vad, voice_ms %d, mode %d\n",
|
||||
switch_channel_get_name(channel), voice_ms, mode);
|
||||
}
|
||||
}
|
||||
@@ -499,7 +499,7 @@ extern "C" {
|
||||
switch_thread_create(&cb->thread, thd_attr, aws_transcribe_thread, cb, pool);
|
||||
|
||||
*ppUserData = cb;
|
||||
|
||||
|
||||
done:
|
||||
return status;
|
||||
}
|
||||
@@ -521,7 +521,7 @@ extern "C" {
|
||||
do {
|
||||
streamer = (GStreamer *) cb->streamer;
|
||||
if (streamer) break;
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG,
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG,
|
||||
"aws_transcribe_session_stop: waiting for streamer to come online..%s\n", bugname);
|
||||
switch_yield(10000); // wait 10ms
|
||||
} while (i++ < 3);
|
||||
@@ -557,7 +557,7 @@ extern "C" {
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_INFO, "%s Bug is not attached.\n", switch_channel_get_name(channel));
|
||||
return SWITCH_STATUS_FALSE;
|
||||
}
|
||||
|
||||
|
||||
switch_bool_t aws_transcribe_frame(switch_media_bug_t *bug, void* user_data) {
|
||||
switch_core_session_t *session = switch_core_media_bug_get_session(bug);
|
||||
uint8_t data[SWITCH_RECOMMENDED_BUFFER_SIZE];
|
||||
@@ -587,7 +587,7 @@ extern "C" {
|
||||
}
|
||||
|
||||
if (cb->resampler) {
|
||||
speex_resampler_process_interleaved_int(cb->resampler, (const spx_int16_t *) frame.data, (spx_uint32_t *) &in_len, &out[0], &out_len);
|
||||
speex_resampler_process_interleaved_int(cb->resampler, (const spx_int16_t *) frame.data, (spx_uint32_t *) &in_len, &out[0], &out_len);
|
||||
streamer->write( &out[0], sizeof(spx_int16_t) * out_len);
|
||||
}
|
||||
else {
|
||||
@@ -597,11 +597,11 @@ extern "C" {
|
||||
}
|
||||
}
|
||||
else {
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG,
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_DEBUG,
|
||||
"aws_transcribe_frame: not sending audio because aws channel has been closed\n");
|
||||
}
|
||||
switch_mutex_unlock(cb->mutex);
|
||||
}
|
||||
return SWITCH_TRUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -172,7 +172,7 @@ namespace {
|
||||
else if (customModel) oss << "&model=" << customModel;
|
||||
|
||||
if (var = switch_channel_get_variable(channel, "DEEPGRAM_SPEECH_MODEL_VERSION")) {
|
||||
oss << "&version";
|
||||
oss << "&version=";
|
||||
oss << var;
|
||||
}
|
||||
oss << "&language=";
|
||||
|
||||
@@ -742,7 +742,7 @@ extern "C" {
|
||||
std::string url;
|
||||
std::ostringstream url_stream;
|
||||
// always use sample_rate=8000 for support jambonz caching system.
|
||||
url_stream << "https://api.deepgram.com/v1/speak?model=" << d->voice_name << "&encoding=linear16&sample_rate=8000";
|
||||
url_stream << (d->endpoint != nullptr ? d->endpoint : "https://api.deepgram.com") << "/v1/speak?model=" << d->voice_name << "&encoding=linear16&sample_rate=8000";
|
||||
url = url_stream.str();
|
||||
|
||||
/* create the JSON body */
|
||||
|
||||
@@ -19,10 +19,12 @@ static void cleardeepgram(deepgram_t* d, int freeAll) {
|
||||
if (d->connect_time_ms) free(d->connect_time_ms);
|
||||
if (d->final_response_time_ms) free(d->final_response_time_ms);
|
||||
if (d->cache_filename) free(d->cache_filename);
|
||||
if (d->endpoint) free(d->endpoint);
|
||||
|
||||
|
||||
d->api_key = NULL;
|
||||
d->request_id = NULL;
|
||||
d->endpoint = NULL;
|
||||
|
||||
d->reported_model_name = NULL;
|
||||
d->reported_model_uuid = NULL;
|
||||
@@ -121,6 +123,9 @@ static void d_text_param_tts(switch_speech_handle_t *sh, char *param, const char
|
||||
if (0 == strcmp(param, "api_key")) {
|
||||
if (d->api_key) free(d->api_key);
|
||||
d->api_key = strdup(val);
|
||||
} else if (0 == strcmp(param, "endpoint")) {
|
||||
if (d->endpoint) free(d->endpoint);
|
||||
d->endpoint = strdup(val);
|
||||
} else if (0 == strcmp(param, "voice")) {
|
||||
if (d->voice_name) free(d->voice_name);
|
||||
d->voice_name = strdup(val);
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
typedef struct deepgram_data {
|
||||
char *voice_name;
|
||||
char *api_key;
|
||||
char *endpoint;
|
||||
|
||||
/* result data */
|
||||
long response_code;
|
||||
|
||||
+1
-1
@@ -271,7 +271,7 @@ SWITCH_STANDARD_API(dub_function)
|
||||
}
|
||||
else {
|
||||
char* text = argv[3];
|
||||
int gain = argc > 4 ? atoi(argv[4]) : 0;
|
||||
int gain = argc > 5 ? atoi(argv[5]) : 0;
|
||||
switch_log_printf(SWITCH_CHANNEL_SESSION_LOG(session), SWITCH_LOG_INFO, "sayOnTrack %s gain %d\n", text, gain);
|
||||
status = dub_say_on_track(session, track, text, gain);
|
||||
}
|
||||
|
||||
@@ -145,7 +145,7 @@ namespace {
|
||||
cJSON_AddStringToObject(json, "format", "raw");
|
||||
cJSON_AddStringToObject(json, "encoding", "LINEAR16");
|
||||
cJSON_AddBoolToObject(json, "interimResults", tech_pvt->interim);
|
||||
cJSON_AddNumberToObject(json, "sampleRateHz", 8000);
|
||||
cJSON_AddNumberToObject(json, "sampleRateHz", tech_pvt->sampling);
|
||||
if (var = switch_channel_get_variable(channel, "JAMBONZ_STT_OPTIONS")) {
|
||||
cJSON* jOptions = cJSON_Parse(var);
|
||||
if (jOptions) {
|
||||
@@ -353,7 +353,7 @@ extern "C" {
|
||||
}
|
||||
|
||||
switch_status_t jb_transcribe_session_init(switch_core_session_t *session,
|
||||
responseHandler_t responseHandler, uint32_t samples_per_second, uint32_t channels,
|
||||
responseHandler_t responseHandler, uint32_t samples_per_second, int desiredSampling, uint32_t channels,
|
||||
char* lang, int interim, char* bugname, void **ppUserData)
|
||||
{
|
||||
int err;
|
||||
@@ -365,7 +365,7 @@ extern "C" {
|
||||
return SWITCH_STATUS_FALSE;
|
||||
}
|
||||
|
||||
if (SWITCH_STATUS_SUCCESS != fork_data_init(tech_pvt, session, samples_per_second, 8000, channels, lang, interim, bugname, responseHandler)) {
|
||||
if (SWITCH_STATUS_SUCCESS != fork_data_init(tech_pvt, session, samples_per_second, desiredSampling, channels, lang, interim, bugname, responseHandler)) {
|
||||
destroy_tech_pvt(tech_pvt);
|
||||
return SWITCH_STATUS_FALSE;
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ int parse_ws_uri(switch_channel_t *channel, const char* szServerUri, char* host,
|
||||
switch_status_t jb_transcribe_init();
|
||||
switch_status_t jb_transcribe_cleanup();
|
||||
switch_status_t jb_transcribe_session_init(switch_core_session_t *session, responseHandler_t responseHandler,
|
||||
uint32_t samples_per_second, uint32_t channels, char* lang, int interim, char* bugname, void **ppUserData);
|
||||
uint32_t samples_per_second, int desiredSampling, uint32_t channels, char* lang, int interim, char* bugname, void **ppUserData);
|
||||
switch_status_t jb_transcribe_session_stop(switch_core_session_t *session, int channelIsClosing, char* bugname);
|
||||
switch_bool_t jb_transcribe_frame(switch_core_session_t *session, switch_media_bug_t *bug);
|
||||
|
||||
|
||||
@@ -73,6 +73,8 @@ static switch_status_t start_capture(switch_core_session_t *session, switch_medi
|
||||
switch_codec_implementation_t read_impl = { 0 };
|
||||
void *pUserData;
|
||||
uint32_t samples_per_second;
|
||||
uint32_t desiredSampling = 8000;
|
||||
const char* var;
|
||||
|
||||
if (!switch_channel_get_variable(channel, "JAMBONZ_STT_URL")) {
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_ERROR, "Missing JAMBONZ_STT_URL channel var\n");
|
||||
@@ -91,8 +93,11 @@ static switch_status_t start_capture(switch_core_session_t *session, switch_medi
|
||||
}
|
||||
|
||||
samples_per_second = !strcasecmp(read_impl.iananame, "g722") ? read_impl.actual_samples_per_second : read_impl.samples_per_second;
|
||||
|
||||
if (SWITCH_STATUS_FALSE == jb_transcribe_session_init(session, responseHandler, samples_per_second, flags & SMBF_STEREO ? 2 : 1, lang, interim, bugname, &pUserData)) {
|
||||
var = switch_channel_get_variable(channel, "JAMBONZ_STT_SAMPLING");
|
||||
if (var != NULL) {
|
||||
desiredSampling = atoi(var);
|
||||
}
|
||||
if (SWITCH_STATUS_FALSE == jb_transcribe_session_init(session, responseHandler, samples_per_second, desiredSampling, flags & SMBF_STEREO ? 2 : 1, lang, interim, bugname, &pUserData)) {
|
||||
switch_log_printf(SWITCH_CHANNEL_LOG, SWITCH_LOG_ERROR, "Error initializing jb speech session.\n");
|
||||
return SWITCH_STATUS_FALSE;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user