mirror of
https://github.com/jambonz/sbc-sip-sidecar.git
synced 2026-10-04 02:04:20 +00:00
restart regbots on drachtio reconnect (#151)
* restart regbots on drachtio reconnect * PR feedback * add check for multiple regbot instances
This commit is contained in:
@@ -44,6 +44,8 @@ const REGISTER_RESPONSE_REMOVE = process.env.REGISTER_RESPONSE_REMOVE?.split(','
|
||||
const JAMBONES_REGBOT_USER_AGENT = process.env.JAMBONES_REGBOT_USER_AGENT ;
|
||||
const JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL = process.env.JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL;
|
||||
const JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD = process.env.JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD;
|
||||
// how long (ms) to wait for a SIP response to a REGISTER before assuming it was lost and retrying
|
||||
const JAMBONES_REGBOT_RESPONSE_TIMEOUT = process.env.JAMBONES_REGBOT_RESPONSE_TIMEOUT;
|
||||
|
||||
/* Server control - external topology discovery and other server-control features (disabled unless truthy) */
|
||||
const JAMBONES_SERVER_CONTROL = process.env.JAMBONES_SERVER_CONTROL;
|
||||
@@ -85,5 +87,6 @@ module.exports = {
|
||||
JAMBONES_REGBOT_USER_AGENT,
|
||||
JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL,
|
||||
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD,
|
||||
JAMBONES_REGBOT_RESPONSE_TIMEOUT,
|
||||
JAMBONES_SERVER_CONTROL
|
||||
};
|
||||
|
||||
+52
-1
@@ -6,7 +6,8 @@ const {
|
||||
REGISTER_RESPONSE_REMOVE,
|
||||
JAMBONES_REGBOT_USER_AGENT,
|
||||
JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL,
|
||||
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD
|
||||
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD,
|
||||
JAMBONES_REGBOT_RESPONSE_TIMEOUT
|
||||
} = require('./config');
|
||||
const {isValidDomainOrIP, isValidIPv4} = require('./utils');
|
||||
const parseUri = require('drachtio-srf').parseUri;
|
||||
@@ -14,6 +15,10 @@ const DEFAULT_EXPIRES = (parseInt(JAMBONES_REGBOT_DEFAULT_EXPIRES_INTERVAL) || 3
|
||||
const MIN_EXPIRES = (parseInt(JAMBONES_REGBOT_MIN_EXPIRES_INTERVAL) || 90);
|
||||
const FAILURE_RETRY_INTERVAL = (parseInt(JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL) || 300);
|
||||
const REGISTER_FAILURE_THRESHOLD = (parseInt(JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD) || 3);
|
||||
// Backstop for a REGISTER whose response never arrives at all (dropped socket). Kept above
|
||||
// drachtio's own non-INVITE Timer F (~32s) so its 408 normally reaches us first and drives the
|
||||
// usual fail/backoff; this only fires when even that never comes.
|
||||
const RESPONSE_TIMEOUT = (parseInt(JAMBONES_REGBOT_RESPONSE_TIMEOUT) || 45000);
|
||||
const assert = require('assert');
|
||||
const version = require('../package.json').version;
|
||||
const useragent = JAMBONES_REGBOT_USER_AGENT || `Jambonz ${version}`;
|
||||
@@ -80,6 +85,8 @@ class Regbot {
|
||||
this.aor = `${this.fromUser}@${this.sip_realm}`;
|
||||
this.status = 'none';
|
||||
this.consecutiveRemoveFailures = 0;
|
||||
// monotonic per-attempt counter; only the latest attempt's response/watchdog is acted upon
|
||||
this.epoch = 0;
|
||||
}
|
||||
|
||||
async start(srf) {
|
||||
@@ -89,11 +96,25 @@ class Regbot {
|
||||
this.register(srf);
|
||||
}
|
||||
|
||||
/**
|
||||
* Force an immediate re-registration, e.g. after the drachtio server restarts.
|
||||
* Clears any pending (possibly stale) re-register timer and sends a fresh REGISTER now;
|
||||
* register() re-arms the timer chain from the response.
|
||||
*/
|
||||
reregister(srf) {
|
||||
if (this.retired) return;
|
||||
clearTimeout(this.timer);
|
||||
this.timer = null;
|
||||
this.register(srf);
|
||||
}
|
||||
|
||||
stop(srf) {
|
||||
this.retired = true;
|
||||
const { deleteEphemeralGateway } = srf.locals.realtimeDbHelpers;
|
||||
clearTimeout(this.timer);
|
||||
this.timer = null;
|
||||
clearTimeout(this.watchdog);
|
||||
this.watchdog = null;
|
||||
// remove any ephemeral gateways created for this regbot
|
||||
if (this.addresses && this.addresses.length) {
|
||||
this.addresses.forEach((ip) => {
|
||||
@@ -109,6 +130,8 @@ class Regbot {
|
||||
this.retired = true;
|
||||
clearTimeout(this.timer);
|
||||
this.timer = null;
|
||||
clearTimeout(this.watchdog);
|
||||
this.watchdog = null;
|
||||
}
|
||||
|
||||
configKey() {
|
||||
@@ -154,6 +177,8 @@ class Regbot {
|
||||
const { createEphemeralGateway } = srf.locals.realtimeDbHelpers;
|
||||
const { updateVoipCarriersRegisterStatus } = srf.locals.dbHelpers;
|
||||
const { writeAlerts, localSIPDomain } = srf.locals;
|
||||
// stamp this attempt so a late/stale response (e.g. after a watchdog-driven retry) is ignored
|
||||
const epoch = ++this.epoch;
|
||||
try {
|
||||
// transport
|
||||
const transport = (this.protocol.includes('/') ? this.protocol.substring(0, this.protocol.indexOf('/')) :
|
||||
@@ -195,6 +220,24 @@ class Regbot {
|
||||
proxy = `sip:${this.ipv4}${isIPv4 ? `:${this.port}` : ''};transport=${transport}`;
|
||||
this.logger.debug(`sending to registrar ${proxy}`);
|
||||
}
|
||||
/* Safety net for a REGISTER whose SIP response never arrives at all (srf.request has no
|
||||
timeout) -- e.g. drachtio dropped the socket mid-transaction so we never even get its
|
||||
408. Treat it exactly like the failure path (mark fail, record status, back off on
|
||||
FAILURE_RETRY_INTERVAL) rather than re-sending inline, so an unreachable registrar does
|
||||
not loop every RESPONSE_TIMEOUT. Bump the epoch so a late response is ignored. */
|
||||
clearTimeout(this.watchdog);
|
||||
this.watchdog = setTimeout(() => {
|
||||
if (this.retired || epoch !== this.epoch) return;
|
||||
this.watchdog = null;
|
||||
this.epoch++;
|
||||
this.status = 'fail';
|
||||
this.logger.info(`${this.aor}: no response to REGISTER within ${RESPONSE_TIMEOUT}ms, backing off`);
|
||||
updateVoipCarriersRegisterStatus(this.voip_carrier_sid, JSON.stringify({
|
||||
status: 'fail',
|
||||
reason: `no response within ${RESPONSE_TIMEOUT}ms`
|
||||
}));
|
||||
this.timer = setTimeout(this.register.bind(this, srf), FAILURE_RETRY_INTERVAL * 1000);
|
||||
}, RESPONSE_TIMEOUT);
|
||||
const req = await srf.request(`${scheme}:${this.sip_realm}`, {
|
||||
method: 'REGISTER',
|
||||
proxy,
|
||||
@@ -212,6 +255,12 @@ class Regbot {
|
||||
}
|
||||
});
|
||||
req.on('response', async(res) => {
|
||||
if (epoch !== this.epoch) {
|
||||
this.logger.info(`${this.aor}: ignoring stale REGISTER response (superseded by a newer attempt)`);
|
||||
return;
|
||||
}
|
||||
clearTimeout(this.watchdog);
|
||||
this.watchdog = null;
|
||||
if (this.retired) {
|
||||
this.logger.info(`${this.aor}: ignoring response, regbot has been retired`);
|
||||
return;
|
||||
@@ -347,6 +396,8 @@ class Regbot {
|
||||
}
|
||||
});
|
||||
} catch (err) {
|
||||
clearTimeout(this.watchdog);
|
||||
this.watchdog = null;
|
||||
this.logger.error({ err }, `${this.aor}: Error registering to ${this.ipv4}:${this.port}`);
|
||||
this.timer = setTimeout(this.register.bind(this, srf), FAILURE_RETRY_INTERVAL * 1000);
|
||||
updateVoipCarriersRegisterStatus(this.voip_carrier_sid, JSON.stringify({
|
||||
|
||||
@@ -168,13 +168,22 @@ const checkStatus = async(logger, srf) => {
|
||||
}
|
||||
else if (token && token !== myToken) {
|
||||
logger.info('Someone else grabbed the role! I need to stand down');
|
||||
srf.locals.regbot.active = false;
|
||||
regbots.forEach((rb) => rb.stop(srf));
|
||||
regbots.length = 0;
|
||||
/* reset hashes so the next updateCarrierRegbots rebuilds from scratch;
|
||||
otherwise the array stays empty until a carrier config change */
|
||||
carriersHash = '';
|
||||
gatewaysHash = '';
|
||||
}
|
||||
else {
|
||||
grabForTheWheel = true;
|
||||
regbots.forEach((rb) => rb.stop(srf));
|
||||
regbots.length = 0;
|
||||
/* reset hashes so the next updateCarrierRegbots rebuilds from scratch;
|
||||
otherwise the array stays empty until a carrier config change */
|
||||
carriersHash = '';
|
||||
gatewaysHash = '';
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -361,3 +370,55 @@ const updateCarrierRegbots = async(logger, srf) => {
|
||||
rebuildInProgress = false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Called when drachtio reconnects after a server restart.
|
||||
* The regbots survive in memory but their SIP state on the (restarted) drachtio server is gone,
|
||||
* so force each one to re-register immediately. If the array was emptied (e.g. by checkStatus)
|
||||
* but this instance is still the active holder, force a full rebuild instead.
|
||||
*/
|
||||
const resync = async(logger, srf) => {
|
||||
if (!srf.locals.regbot) return;
|
||||
if (!srf.locals.regbot.active) {
|
||||
logger.info('drachtio reconnect: not the active regbot holder, nothing to re-register');
|
||||
return;
|
||||
}
|
||||
if (regbots.length === 0) {
|
||||
/* nothing primed (e.g. cleared by checkStatus) -- force a full rebuild */
|
||||
carriersHash = '';
|
||||
gatewaysHash = '';
|
||||
return updateCarrierRegbots(logger, srf);
|
||||
}
|
||||
/* A registered bot with a live refresh timer keeps its binding at the carrier across a
|
||||
drachtio restart -- REGISTER is transaction-stateless on drachtio -- so leave it alone.
|
||||
Only bots with a REGISTER in flight (watchdog set) or not currently registered need to be
|
||||
re-driven; batch them so a large fleet does not REGISTER-storm all at once. */
|
||||
const broken = regbots.filter((rb) => rb.status !== 'registered' || rb.watchdog);
|
||||
if (broken.length === 0) {
|
||||
logger.info(`drachtio reconnect: all ${regbots.length} regbots healthy, nothing to re-register`);
|
||||
return;
|
||||
}
|
||||
logger.info(`drachtio reconnect: re-registering ${broken.length} of ${regbots.length} regbots`);
|
||||
let batch_count = 0;
|
||||
for (const rb of broken) {
|
||||
rb.reregister(srf);
|
||||
if (++batch_count >= JAMBONES_REGBOT_BATCH_SIZE) {
|
||||
batch_count = 0;
|
||||
await sleepFor(JAMBONES_REGBOT_BATCH_SLEEP_MS);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
module.exports.resync = resync;
|
||||
|
||||
// exposed for unit testing: stop all regbots and reset module state
|
||||
module.exports._resetForTest = () => {
|
||||
regbots.forEach((rb) => {
|
||||
rb.retired = true;
|
||||
clearTimeout(rb.timer);
|
||||
clearTimeout(rb.watchdog);
|
||||
});
|
||||
regbots.length = 0;
|
||||
carriersHash = '';
|
||||
gatewaysHash = '';
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user