restart regbots on drachtio reconnect (#151)

* restart regbots on drachtio reconnect

* PR feedback

* add check for multiple regbot instances
This commit is contained in:
Sam Machin
2026-09-09 15:02:23 +01:00
committed by GitHub
parent fe3f11a063
commit ce05d91e6d
7 changed files with 410 additions and 1 deletions
+3
View File
@@ -44,6 +44,8 @@ const REGISTER_RESPONSE_REMOVE = process.env.REGISTER_RESPONSE_REMOVE?.split(','
const JAMBONES_REGBOT_USER_AGENT = process.env.JAMBONES_REGBOT_USER_AGENT ;
const JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL = process.env.JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL;
const JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD = process.env.JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD;
// how long (ms) to wait for a SIP response to a REGISTER before assuming it was lost and retrying
const JAMBONES_REGBOT_RESPONSE_TIMEOUT = process.env.JAMBONES_REGBOT_RESPONSE_TIMEOUT;
/* Server control - external topology discovery and other server-control features (disabled unless truthy) */
const JAMBONES_SERVER_CONTROL = process.env.JAMBONES_SERVER_CONTROL;
@@ -85,5 +87,6 @@ module.exports = {
JAMBONES_REGBOT_USER_AGENT,
JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL,
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD,
JAMBONES_REGBOT_RESPONSE_TIMEOUT,
JAMBONES_SERVER_CONTROL
};
+52 -1
View File
@@ -6,7 +6,8 @@ const {
REGISTER_RESPONSE_REMOVE,
JAMBONES_REGBOT_USER_AGENT,
JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL,
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD,
JAMBONES_REGBOT_RESPONSE_TIMEOUT
} = require('./config');
const {isValidDomainOrIP, isValidIPv4} = require('./utils');
const parseUri = require('drachtio-srf').parseUri;
@@ -14,6 +15,10 @@ const DEFAULT_EXPIRES = (parseInt(JAMBONES_REGBOT_DEFAULT_EXPIRES_INTERVAL) || 3
const MIN_EXPIRES = (parseInt(JAMBONES_REGBOT_MIN_EXPIRES_INTERVAL) || 90);
const FAILURE_RETRY_INTERVAL = (parseInt(JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL) || 300);
const REGISTER_FAILURE_THRESHOLD = (parseInt(JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD) || 3);
// Backstop for a REGISTER whose response never arrives at all (dropped socket). Kept above
// drachtio's own non-INVITE Timer F (~32s) so its 408 normally reaches us first and drives the
// usual fail/backoff; this only fires when even that never comes.
const RESPONSE_TIMEOUT = (parseInt(JAMBONES_REGBOT_RESPONSE_TIMEOUT) || 45000);
const assert = require('assert');
const version = require('../package.json').version;
const useragent = JAMBONES_REGBOT_USER_AGENT || `Jambonz ${version}`;
@@ -80,6 +85,8 @@ class Regbot {
this.aor = `${this.fromUser}@${this.sip_realm}`;
this.status = 'none';
this.consecutiveRemoveFailures = 0;
// monotonic per-attempt counter; only the latest attempt's response/watchdog is acted upon
this.epoch = 0;
}
async start(srf) {
@@ -89,11 +96,25 @@ class Regbot {
this.register(srf);
}
/**
* Force an immediate re-registration, e.g. after the drachtio server restarts.
* Clears any pending (possibly stale) re-register timer and sends a fresh REGISTER now;
* register() re-arms the timer chain from the response.
*/
reregister(srf) {
if (this.retired) return;
clearTimeout(this.timer);
this.timer = null;
this.register(srf);
}
stop(srf) {
this.retired = true;
const { deleteEphemeralGateway } = srf.locals.realtimeDbHelpers;
clearTimeout(this.timer);
this.timer = null;
clearTimeout(this.watchdog);
this.watchdog = null;
// remove any ephemeral gateways created for this regbot
if (this.addresses && this.addresses.length) {
this.addresses.forEach((ip) => {
@@ -109,6 +130,8 @@ class Regbot {
this.retired = true;
clearTimeout(this.timer);
this.timer = null;
clearTimeout(this.watchdog);
this.watchdog = null;
}
configKey() {
@@ -154,6 +177,8 @@ class Regbot {
const { createEphemeralGateway } = srf.locals.realtimeDbHelpers;
const { updateVoipCarriersRegisterStatus } = srf.locals.dbHelpers;
const { writeAlerts, localSIPDomain } = srf.locals;
// stamp this attempt so a late/stale response (e.g. after a watchdog-driven retry) is ignored
const epoch = ++this.epoch;
try {
// transport
const transport = (this.protocol.includes('/') ? this.protocol.substring(0, this.protocol.indexOf('/')) :
@@ -195,6 +220,24 @@ class Regbot {
proxy = `sip:${this.ipv4}${isIPv4 ? `:${this.port}` : ''};transport=${transport}`;
this.logger.debug(`sending to registrar ${proxy}`);
}
/* Safety net for a REGISTER whose SIP response never arrives at all (srf.request has no
timeout) -- e.g. drachtio dropped the socket mid-transaction so we never even get its
408. Treat it exactly like the failure path (mark fail, record status, back off on
FAILURE_RETRY_INTERVAL) rather than re-sending inline, so an unreachable registrar does
not loop every RESPONSE_TIMEOUT. Bump the epoch so a late response is ignored. */
clearTimeout(this.watchdog);
this.watchdog = setTimeout(() => {
if (this.retired || epoch !== this.epoch) return;
this.watchdog = null;
this.epoch++;
this.status = 'fail';
this.logger.info(`${this.aor}: no response to REGISTER within ${RESPONSE_TIMEOUT}ms, backing off`);
updateVoipCarriersRegisterStatus(this.voip_carrier_sid, JSON.stringify({
status: 'fail',
reason: `no response within ${RESPONSE_TIMEOUT}ms`
}));
this.timer = setTimeout(this.register.bind(this, srf), FAILURE_RETRY_INTERVAL * 1000);
}, RESPONSE_TIMEOUT);
const req = await srf.request(`${scheme}:${this.sip_realm}`, {
method: 'REGISTER',
proxy,
@@ -212,6 +255,12 @@ class Regbot {
}
});
req.on('response', async(res) => {
if (epoch !== this.epoch) {
this.logger.info(`${this.aor}: ignoring stale REGISTER response (superseded by a newer attempt)`);
return;
}
clearTimeout(this.watchdog);
this.watchdog = null;
if (this.retired) {
this.logger.info(`${this.aor}: ignoring response, regbot has been retired`);
return;
@@ -347,6 +396,8 @@ class Regbot {
}
});
} catch (err) {
clearTimeout(this.watchdog);
this.watchdog = null;
this.logger.error({ err }, `${this.aor}: Error registering to ${this.ipv4}:${this.port}`);
this.timer = setTimeout(this.register.bind(this, srf), FAILURE_RETRY_INTERVAL * 1000);
updateVoipCarriersRegisterStatus(this.voip_carrier_sid, JSON.stringify({
+61
View File
@@ -168,13 +168,22 @@ const checkStatus = async(logger, srf) => {
}
else if (token && token !== myToken) {
logger.info('Someone else grabbed the role! I need to stand down');
srf.locals.regbot.active = false;
regbots.forEach((rb) => rb.stop(srf));
regbots.length = 0;
/* reset hashes so the next updateCarrierRegbots rebuilds from scratch;
otherwise the array stays empty until a carrier config change */
carriersHash = '';
gatewaysHash = '';
}
else {
grabForTheWheel = true;
regbots.forEach((rb) => rb.stop(srf));
regbots.length = 0;
/* reset hashes so the next updateCarrierRegbots rebuilds from scratch;
otherwise the array stays empty until a carrier config change */
carriersHash = '';
gatewaysHash = '';
}
}
else {
@@ -361,3 +370,55 @@ const updateCarrierRegbots = async(logger, srf) => {
rebuildInProgress = false;
}
};
/**
* Called when drachtio reconnects after a server restart.
* The regbots survive in memory but their SIP state on the (restarted) drachtio server is gone,
* so force each one to re-register immediately. If the array was emptied (e.g. by checkStatus)
* but this instance is still the active holder, force a full rebuild instead.
*/
const resync = async(logger, srf) => {
if (!srf.locals.regbot) return;
if (!srf.locals.regbot.active) {
logger.info('drachtio reconnect: not the active regbot holder, nothing to re-register');
return;
}
if (regbots.length === 0) {
/* nothing primed (e.g. cleared by checkStatus) -- force a full rebuild */
carriersHash = '';
gatewaysHash = '';
return updateCarrierRegbots(logger, srf);
}
/* A registered bot with a live refresh timer keeps its binding at the carrier across a
drachtio restart -- REGISTER is transaction-stateless on drachtio -- so leave it alone.
Only bots with a REGISTER in flight (watchdog set) or not currently registered need to be
re-driven; batch them so a large fleet does not REGISTER-storm all at once. */
const broken = regbots.filter((rb) => rb.status !== 'registered' || rb.watchdog);
if (broken.length === 0) {
logger.info(`drachtio reconnect: all ${regbots.length} regbots healthy, nothing to re-register`);
return;
}
logger.info(`drachtio reconnect: re-registering ${broken.length} of ${regbots.length} regbots`);
let batch_count = 0;
for (const rb of broken) {
rb.reregister(srf);
if (++batch_count >= JAMBONES_REGBOT_BATCH_SIZE) {
batch_count = 0;
await sleepFor(JAMBONES_REGBOT_BATCH_SLEEP_MS);
}
}
};
module.exports.resync = resync;
// exposed for unit testing: stop all regbots and reset module state
module.exports._resetForTest = () => {
regbots.forEach((rb) => {
rb.retired = true;
clearTimeout(rb.timer);
clearTimeout(rb.watchdog);
});
regbots.length = 0;
carriersHash = '';
gatewaysHash = '';
};