feat: per-SBC registration Call-ID, and hand off the regbot role on AWS scale-in (#156)

The Call-ID of outbound registrations was the sip_gateway_sid, the same on
every SBC. When the regbot role moved to the other SBC, the registrar saw a
refresh of an existing binding from a different source address and Contact.
Some registrars 200 such a refresh without updating their routing, so inbound
calls to the registered trunk fail with 404 until the binding is recreated.
The Call-ID is now sip_gateway_sid@<sending SBC public IP>: stable across
refreshes and restarts of one SBC, new when the role moves, so a move looks
like a new registration. register_status also records the sending SBC as
sbcAddress.

With AWS_LIFECYCLE_DRAIN enabled the sidecar polls IMDS
autoscaling/target-lifecycle-state (the signal inbound drains on; detection
only, inbound completes the lifecycle hook). When the instance is being scaled
in, the regbot holder releases the lease while still running instead of after
the instance is gone; until now the draining SBC kept the registrations, so
carriers kept sending registration-trunk calls to an SBC that answers new
INVITEs with 503. It never claims the role back. Once another SBC has claimed
it, the draining SBC un-REGISTERs (Expires: 0) the bindings whose Contact
carries its own IP. Bindings with an AoR or realm Contact are left alone: the
new SBC sends the same Contact, and its REGISTER has already replaced ours.

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Dave Horton
2026-10-01 13:33:36 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent bdd8fcda44
commit f5c675952a
10 changed files with 399 additions and 44 deletions
+44
View File
@@ -0,0 +1,44 @@
/* Detect that this instance is being scaled in by polling IMDS autoscaling/target-lifecycle-state,
* the same signal sbc-inbound drains on: it reads 'Terminated' once the Auto Scaling group moves
* the instance to Terminating:Wait. Detection only -- sbc-inbound owns the lifecycle hook and
* completes it once calls have drained, so this needs no Auto Scaling API access.
*
* IMDSv2 only: fetch a session token with PUT, then present it on the GET. */
const IMDS = 'http://169.254.169.254/latest';
const IMDS_TIMEOUT_MS = 2000;
const POLL_INTERVAL_MS = 20000;
const imds = async(path) => {
const tokenRes = await fetch(`${IMDS}/api/token`, {
method: 'PUT',
headers: {'X-aws-ec2-metadata-token-ttl-seconds': '60'},
signal: AbortSignal.timeout(IMDS_TIMEOUT_MS)
});
if (!tokenRes.ok) throw new Error(`IMDS token request failed: ${tokenRes.status}`);
const token = await tokenRes.text();
const res = await fetch(`${IMDS}/meta-data/${path}`, {
headers: {'X-aws-ec2-metadata-token': token},
signal: AbortSignal.timeout(IMDS_TIMEOUT_MS)
});
if (!res.ok) throw new Error(`IMDS ${path} request failed: ${res.status}`);
return res.text();
};
module.exports = (logger, onScaleIn, {interval = POLL_INTERVAL_MS} = {}) => {
let fired = false;
const timer = setInterval(async() => {
try {
const state = await imds('autoscaling/target-lifecycle-state');
if (state !== 'Terminated' || fired) return;
fired = true;
clearInterval(timer);
logger.info('AWS scale-in detected (target-lifecycle-state is Terminated)');
onScaleIn();
} catch (err) {
logger.warn({err}, 'Error polling IMDS autoscaling/target-lifecycle-state');
}
}, interval);
timer.unref();
logger.info('AWS lifecycle drain enabled: polling IMDS autoscaling/target-lifecycle-state');
return timer;
};
+6 -1
View File
@@ -47,6 +47,10 @@ const JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD = process.env.JAMBONES_REGBOT_R
// how long (ms) to wait for a SIP response to a REGISTER before assuming it was lost and retrying
const JAMBONES_REGBOT_RESPONSE_TIMEOUT = process.env.JAMBONES_REGBOT_RESPONSE_TIMEOUT;
/* AWS Auto Scaling: hand off the regbot role when this instance is being scaled in */
const AWS_LIFECYCLE_DRAIN = ['1', 'true', 'yes']
.includes((process.env.AWS_LIFECYCLE_DRAIN || '').trim().toLowerCase());
/* Server control - external topology discovery and other server-control features (disabled unless truthy) */
const JAMBONES_SERVER_CONTROL = process.env.JAMBONES_SERVER_CONTROL;
@@ -88,5 +92,6 @@ module.exports = {
JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL,
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD,
JAMBONES_REGBOT_RESPONSE_TIMEOUT,
JAMBONES_SERVER_CONTROL
JAMBONES_SERVER_CONTROL,
AWS_LIFECYCLE_DRAIN
};
+85 -42
View File
@@ -173,53 +173,95 @@ class Regbot {
};
}
/**
* Addressing for a REGISTER from this SBC.
* The Call-ID is per gateway *and* per sending SBC (sip_gateway_sid@public-ip): stable across
* refreshes and restarts of one SBC, but new when the regbot role moves to another SBC. A move
* then looks to the registrar like a new registration rather than a refresh of an existing
* binding from a different source address, which some registrars mishandle (they 200 the
* refresh but never update their routing).
*/
_requestParams(srf) {
const transport = (this.protocol.includes('/') ? this.protocol.substring(0, this.protocol.indexOf('/')) :
this.protocol).toLowerCase();
let scheme = 'sip';
if (transport === 'tls' && this.use_sips_scheme) scheme = 'sips';
let publicAddress = srf.locals.sbcPublicIpAddress.udp;
if (transport !== 'udp' && srf.locals.sbcPublicIpAddress[transport]) {
publicAddress = srf.locals.sbcPublicIpAddress[transport];
}
let contactAddress = this.aor;
if (this.use_public_ip_in_contact) {
contactAddress = `${this.fromUser}@${publicAddress}`;
}
else if (this.account_sip_realm) {
contactAddress = `${this.fromUser}@${this.account_sip_realm}`;
}
else if (srf.locals.localSIPDomain) {
contactAddress = `${this.fromUser}@${srf.locals.localSIPDomain}`;
}
let proxy;
if (this.outbound_sip_proxy) {
proxy = `sip:${this.outbound_sip_proxy};transport=${transport}`;
} else {
const isIPv4 = isValidIPv4(this.ipv4);
proxy = `sip:${this.ipv4}${isIPv4 ? `:${this.port}` : ''};transport=${transport}`;
}
const callId = `${this.sip_gateway_sid}@${publicAddress.split(':')[0]}`;
return {transport, scheme, publicAddress, contactAddress, proxy, callId};
}
/**
* Remove our binding at the registrar (Expires: 0), used when this SBC hands the regbot role
* to another SBC before going away. Only bindings whose Contact carries this SBC's own address
* are removed: when the Contact is the AoR or a sip realm, the other SBC sends the same Contact
* and its REGISTER (with a different Call-ID) has already replaced ours, so removing it would
* remove the other SBC's registration.
*/
async unregister(srf) {
if (!this.use_public_ip_in_contact) return false;
const {transport, scheme, contactAddress, proxy, callId} = this._requestParams(srf);
try {
const req = await srf.request(`${scheme}:${this.sip_realm}`, {
method: 'REGISTER',
proxy,
headers: {
'Call-ID': callId,
'From': this.from,
'To': this.from,
'Contact': `<${scheme}:${contactAddress};transport=${transport}>;expires=0`,
'Expires': 0,
'User-Agent': useragent
},
auth: {
username: this.username,
password: this.password
}
});
req.on('response', (res) => {
this.logger.info(`${this.aor}: got ${res.status} to un-REGISTER of ${contactAddress}`);
});
return true;
} catch (err) {
this.logger.info({err}, `${this.aor}: error sending un-REGISTER`);
return false;
}
}
async register(srf) {
const { createEphemeralGateway } = srf.locals.realtimeDbHelpers;
const { updateVoipCarriersRegisterStatus } = srf.locals.dbHelpers;
const { writeAlerts, AlertType, localSIPDomain } = srf.locals;
const { writeAlerts, AlertType } = srf.locals;
// stamp this attempt so a late/stale response (e.g. after a watchdog-driven retry) is ignored
const epoch = ++this.epoch;
try {
// transport
const transport = (this.protocol.includes('/') ? this.protocol.substring(0, this.protocol.indexOf('/')) :
this.protocol).toLowerCase();
// scheme
let scheme = 'sip';
if (transport === 'tls' && this.use_sips_scheme) scheme = 'sips';
let publicAddress = srf.locals.sbcPublicIpAddress.udp;
if (transport !== 'udp') {
if (srf.locals.sbcPublicIpAddress[transport]) {
publicAddress = srf.locals.sbcPublicIpAddress[transport];
}
else if (transport === 'tls') {
publicAddress = srf.locals.sbcPublicIpAddress.udp;
}
}
let contactAddress = this.aor;
if (this.use_public_ip_in_contact) {
contactAddress = `${this.fromUser}@${publicAddress}`;
}
else if (this.account_sip_realm) {
contactAddress = `${this.fromUser}@${this.account_sip_realm}`;
}
else if (localSIPDomain) {
contactAddress = `${this.fromUser}@${localSIPDomain}`;
}
this.logger.debug(`sending REGISTER for ${this.aor}`);
let proxy;
if (this.outbound_sip_proxy) {
proxy = `sip:${this.outbound_sip_proxy};transport=${transport}`;
this.logger.debug(`sending via proxy ${proxy}`);
} else {
const isIPv4 = isValidIPv4(this.ipv4);
proxy = `sip:${this.ipv4}${isIPv4 ? `:${this.port}` : ''};transport=${transport}`;
this.logger.debug(`sending to registrar ${proxy}`);
}
const {transport, scheme, publicAddress, contactAddress, proxy, callId} = this._requestParams(srf);
this.logger.debug(`sending REGISTER for ${this.aor} to ${proxy}`);
/* Safety net for a REGISTER whose SIP response never arrives at all (srf.request has no
timeout) -- e.g. drachtio dropped the socket mid-transaction so we never even get its
408. Treat it exactly like the failure path (mark fail, record status, back off on
@@ -242,7 +284,7 @@ class Regbot {
method: 'REGISTER',
proxy,
headers: {
'Call-ID': this.sip_gateway_sid,
'Call-ID': callId,
'From': this.from,
'To': this.from,
'Contact': `<${scheme}:${contactAddress};transport=${transport}>;expires=${DEFAULT_EXPIRES}`,
@@ -353,6 +395,7 @@ class Regbot {
reason: `${res.status} ${res.reason}`,
cseq: req.get('Cseq'),
callId: req.get('Call-Id'),
sbcAddress: publicAddress,
timestamp: timestamp,
expires: expires
}));
+87 -1
View File
@@ -4,6 +4,7 @@ const {
JAMBONES_CLUSTER_ID,
JAMBONES_REGBOT_BATCH_SLEEP_MS,
JAMBONES_REGBOT_BATCH_SIZE,
AWS_LIFECYCLE_DRAIN,
} = require('./config');
const short = require('short-uuid');
const Regbot = require('./regbot');
@@ -11,6 +12,11 @@ const { sleepFor } = require('./utils');
const MAX_INITIAL_DELAY = 15;
const REGBOT_STATUS_CHECK_INTERVAL = 60;
/* scale-in handoff: how often to look for a successor, how long to wait for one (a successor
claims within one status check interval), and how far to trail its registrations */
const HANDOFF_POLL_MS = 5000;
const HANDOFF_TIMEOUT_MS = 2 * REGBOT_STATUS_CHECK_INTERVAL * 1000;
const HANDOFF_GRACE_MS = 10000;
const regbotKey = `${(JAMBONES_CLUSTER_ID || 'default')}:regbot-token`;
const waitFor = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
let initialized = false;
@@ -137,6 +143,13 @@ module.exports = async(logger, srf) => {
/* check every so often if I need to go from inactive->active (or vice versa) */
setInterval(checkStatus.bind(null, logger, srf), REGBOT_STATUS_CHECK_INTERVAL * 1000);
/* on an AWS scale-in, hand the regbot role to another SBC while we are still up */
if (AWS_LIFECYCLE_DRAIN) {
require('./aws-lifecycle')(logger, () => {
handoff(logger, srf).catch((err) => logger.error({err}, 'regbot handoff failed'));
});
}
/* if I am the regbot holder, then kick it off */
if (srf.locals.regbot.active) {
updateCarrierRegbots(logger, srf)
@@ -150,7 +163,10 @@ module.exports = async(logger, srf) => {
const checkStatus = async(logger, srf) => {
const { addKeyNx, addKey, retrieveKey } = srf.locals.realtimeDbHelpers;
const { myToken, active } = srf.locals.regbot;
const { myToken, active, draining } = srf.locals.regbot;
/* a draining SBC has handed off the role and must never take it back */
if (draining) return;
logger.info({ active, myToken }, 'checking in on regbot status');
try {
@@ -409,9 +425,79 @@ const resync = async(logger, srf) => {
}
};
/**
* Called when this SBC starts draining for scale-in. If we hold the regbot role, give it up while
* we are still running rather than letting the lease expire after we are gone: the draining SBC
* answers new INVITEs with 503, and the carriers would otherwise keep sending calls for
* registration trunks here until the other SBC noticed the lapsed lease.
*
* Once another SBC has claimed the role (and so is registering with its own Call-ID), remove the
* bindings whose Contact points at this SBC, so registrars stop routing to an address that is
* going away.
*/
const handoff = async(logger, srf, opts = {}) => {
const {
pollMs = HANDOFF_POLL_MS,
timeoutMs = HANDOFF_TIMEOUT_MS,
graceMs = HANDOFF_GRACE_MS
} = opts;
const { retrieveKey, deleteKey } = srf.locals.realtimeDbHelpers;
const { myToken } = srf.locals.regbot;
srf.locals.regbot.draining = true;
if (!srf.locals.regbot.active) {
logger.info('scale-in: not the regbot holder, nothing to hand off');
return;
}
/* let an in-flight rebuild finish so it cannot start regbots after we stop them */
while (rebuildInProgress) await waitFor(250);
srf.locals.regbot.active = false;
const handedOff = regbots.splice(0, regbots.length);
/* stop refreshing but keep the ephemeral gateways: the new holder overwrites them */
handedOff.forEach((rb) => rb.stopTimer());
carriersHash = '';
gatewaysHash = '';
if (await retrieveKey(regbotKey) === myToken) await deleteKey(regbotKey);
logger.info(`scale-in: released regbot role (${handedOff.length} regbots), waiting for another SBC to claim it`);
let successor;
for (const deadline = Date.now() + timeoutMs; Date.now() < deadline;) {
await waitFor(pollMs);
const token = await retrieveKey(regbotKey);
if (token && token !== myToken) {
successor = token;
break;
}
}
if (!successor) {
logger.info('scale-in: no other SBC claimed the regbot role; leaving our bindings to expire');
return;
}
/* the successor registers in batches from the moment it claims the role; trail it by graceMs
at the same pace so each of its bindings is in place before we remove the matching old one */
await waitFor(graceMs);
const ours = handedOff.filter((rb) => rb.use_public_ip_in_contact);
logger.info(`scale-in: SBC ${successor} now holds the regbot role; ` +
`un-registering ${ours.length} bindings that point at this SBC`);
let batch_count = 0;
for (const rb of ours) {
await rb.unregister(srf);
if (++batch_count >= JAMBONES_REGBOT_BATCH_SIZE) {
batch_count = 0;
await sleepFor(JAMBONES_REGBOT_BATCH_SLEEP_MS);
}
}
};
module.exports.resync = resync;
module.exports.handoff = handoff;
// exposed for unit testing: stop all regbots and reset module state
module.exports._addForTest = (rb) => regbots.push(rb);
module.exports._resetForTest = () => {
regbots.forEach((rb) => {
rb.retired = true;