mirror of
https://github.com/jambonz/sbc-sip-sidecar.git
synced 2026-10-04 02:04:20 +00:00
feat: per-SBC registration Call-ID, and hand off the regbot role on AWS scale-in (#156)
The Call-ID of outbound registrations was the sip_gateway_sid, the same on every SBC. When the regbot role moved to the other SBC, the registrar saw a refresh of an existing binding from a different source address and Contact. Some registrars 200 such a refresh without updating their routing, so inbound calls to the registered trunk fail with 404 until the binding is recreated. The Call-ID is now sip_gateway_sid@<sending SBC public IP>: stable across refreshes and restarts of one SBC, new when the role moves, so a move looks like a new registration. register_status also records the sending SBC as sbcAddress. With AWS_LIFECYCLE_DRAIN enabled the sidecar polls IMDS autoscaling/target-lifecycle-state (the signal inbound drains on; detection only, inbound completes the lifecycle hook). When the instance is being scaled in, the regbot holder releases the lease while still running instead of after the instance is gone; until now the draining SBC kept the registrations, so carriers kept sending registration-trunk calls to an SBC that answers new INVITEs with 503. It never claims the role back. Once another SBC has claimed it, the draining SBC un-REGISTERs (Expires: 0) the bindings whose Contact carries its own IP. Bindings with an AoR or realm Contact are left alone: the new SBC sends the same Contact, and its REGISTER has already replaced ours. Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
bdd8fcda44
commit
f5c675952a
@@ -0,0 +1,44 @@
|
||||
/* Detect that this instance is being scaled in by polling IMDS autoscaling/target-lifecycle-state,
|
||||
* the same signal sbc-inbound drains on: it reads 'Terminated' once the Auto Scaling group moves
|
||||
* the instance to Terminating:Wait. Detection only -- sbc-inbound owns the lifecycle hook and
|
||||
* completes it once calls have drained, so this needs no Auto Scaling API access.
|
||||
*
|
||||
* IMDSv2 only: fetch a session token with PUT, then present it on the GET. */
|
||||
const IMDS = 'http://169.254.169.254/latest';
|
||||
const IMDS_TIMEOUT_MS = 2000;
|
||||
const POLL_INTERVAL_MS = 20000;
|
||||
|
||||
const imds = async(path) => {
|
||||
const tokenRes = await fetch(`${IMDS}/api/token`, {
|
||||
method: 'PUT',
|
||||
headers: {'X-aws-ec2-metadata-token-ttl-seconds': '60'},
|
||||
signal: AbortSignal.timeout(IMDS_TIMEOUT_MS)
|
||||
});
|
||||
if (!tokenRes.ok) throw new Error(`IMDS token request failed: ${tokenRes.status}`);
|
||||
const token = await tokenRes.text();
|
||||
const res = await fetch(`${IMDS}/meta-data/${path}`, {
|
||||
headers: {'X-aws-ec2-metadata-token': token},
|
||||
signal: AbortSignal.timeout(IMDS_TIMEOUT_MS)
|
||||
});
|
||||
if (!res.ok) throw new Error(`IMDS ${path} request failed: ${res.status}`);
|
||||
return res.text();
|
||||
};
|
||||
|
||||
module.exports = (logger, onScaleIn, {interval = POLL_INTERVAL_MS} = {}) => {
|
||||
let fired = false;
|
||||
const timer = setInterval(async() => {
|
||||
try {
|
||||
const state = await imds('autoscaling/target-lifecycle-state');
|
||||
if (state !== 'Terminated' || fired) return;
|
||||
fired = true;
|
||||
clearInterval(timer);
|
||||
logger.info('AWS scale-in detected (target-lifecycle-state is Terminated)');
|
||||
onScaleIn();
|
||||
} catch (err) {
|
||||
logger.warn({err}, 'Error polling IMDS autoscaling/target-lifecycle-state');
|
||||
}
|
||||
}, interval);
|
||||
timer.unref();
|
||||
logger.info('AWS lifecycle drain enabled: polling IMDS autoscaling/target-lifecycle-state');
|
||||
return timer;
|
||||
};
|
||||
+6
-1
@@ -47,6 +47,10 @@ const JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD = process.env.JAMBONES_REGBOT_R
|
||||
// how long (ms) to wait for a SIP response to a REGISTER before assuming it was lost and retrying
|
||||
const JAMBONES_REGBOT_RESPONSE_TIMEOUT = process.env.JAMBONES_REGBOT_RESPONSE_TIMEOUT;
|
||||
|
||||
/* AWS Auto Scaling: hand off the regbot role when this instance is being scaled in */
|
||||
const AWS_LIFECYCLE_DRAIN = ['1', 'true', 'yes']
|
||||
.includes((process.env.AWS_LIFECYCLE_DRAIN || '').trim().toLowerCase());
|
||||
|
||||
/* Server control - external topology discovery and other server-control features (disabled unless truthy) */
|
||||
const JAMBONES_SERVER_CONTROL = process.env.JAMBONES_SERVER_CONTROL;
|
||||
|
||||
@@ -88,5 +92,6 @@ module.exports = {
|
||||
JAMBONES_REGBOT_FAILURE_RETRY_INTERVAL,
|
||||
JAMBONES_REGBOT_REGISTER_FAILURE_THRESHOLD,
|
||||
JAMBONES_REGBOT_RESPONSE_TIMEOUT,
|
||||
JAMBONES_SERVER_CONTROL
|
||||
JAMBONES_SERVER_CONTROL,
|
||||
AWS_LIFECYCLE_DRAIN
|
||||
};
|
||||
|
||||
+85
-42
@@ -173,53 +173,95 @@ class Regbot {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Addressing for a REGISTER from this SBC.
|
||||
* The Call-ID is per gateway *and* per sending SBC (sip_gateway_sid@public-ip): stable across
|
||||
* refreshes and restarts of one SBC, but new when the regbot role moves to another SBC. A move
|
||||
* then looks to the registrar like a new registration rather than a refresh of an existing
|
||||
* binding from a different source address, which some registrars mishandle (they 200 the
|
||||
* refresh but never update their routing).
|
||||
*/
|
||||
_requestParams(srf) {
|
||||
const transport = (this.protocol.includes('/') ? this.protocol.substring(0, this.protocol.indexOf('/')) :
|
||||
this.protocol).toLowerCase();
|
||||
|
||||
let scheme = 'sip';
|
||||
if (transport === 'tls' && this.use_sips_scheme) scheme = 'sips';
|
||||
|
||||
let publicAddress = srf.locals.sbcPublicIpAddress.udp;
|
||||
if (transport !== 'udp' && srf.locals.sbcPublicIpAddress[transport]) {
|
||||
publicAddress = srf.locals.sbcPublicIpAddress[transport];
|
||||
}
|
||||
|
||||
let contactAddress = this.aor;
|
||||
if (this.use_public_ip_in_contact) {
|
||||
contactAddress = `${this.fromUser}@${publicAddress}`;
|
||||
}
|
||||
else if (this.account_sip_realm) {
|
||||
contactAddress = `${this.fromUser}@${this.account_sip_realm}`;
|
||||
}
|
||||
else if (srf.locals.localSIPDomain) {
|
||||
contactAddress = `${this.fromUser}@${srf.locals.localSIPDomain}`;
|
||||
}
|
||||
|
||||
let proxy;
|
||||
if (this.outbound_sip_proxy) {
|
||||
proxy = `sip:${this.outbound_sip_proxy};transport=${transport}`;
|
||||
} else {
|
||||
const isIPv4 = isValidIPv4(this.ipv4);
|
||||
proxy = `sip:${this.ipv4}${isIPv4 ? `:${this.port}` : ''};transport=${transport}`;
|
||||
}
|
||||
|
||||
const callId = `${this.sip_gateway_sid}@${publicAddress.split(':')[0]}`;
|
||||
return {transport, scheme, publicAddress, contactAddress, proxy, callId};
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove our binding at the registrar (Expires: 0), used when this SBC hands the regbot role
|
||||
* to another SBC before going away. Only bindings whose Contact carries this SBC's own address
|
||||
* are removed: when the Contact is the AoR or a sip realm, the other SBC sends the same Contact
|
||||
* and its REGISTER (with a different Call-ID) has already replaced ours, so removing it would
|
||||
* remove the other SBC's registration.
|
||||
*/
|
||||
async unregister(srf) {
|
||||
if (!this.use_public_ip_in_contact) return false;
|
||||
const {transport, scheme, contactAddress, proxy, callId} = this._requestParams(srf);
|
||||
try {
|
||||
const req = await srf.request(`${scheme}:${this.sip_realm}`, {
|
||||
method: 'REGISTER',
|
||||
proxy,
|
||||
headers: {
|
||||
'Call-ID': callId,
|
||||
'From': this.from,
|
||||
'To': this.from,
|
||||
'Contact': `<${scheme}:${contactAddress};transport=${transport}>;expires=0`,
|
||||
'Expires': 0,
|
||||
'User-Agent': useragent
|
||||
},
|
||||
auth: {
|
||||
username: this.username,
|
||||
password: this.password
|
||||
}
|
||||
});
|
||||
req.on('response', (res) => {
|
||||
this.logger.info(`${this.aor}: got ${res.status} to un-REGISTER of ${contactAddress}`);
|
||||
});
|
||||
return true;
|
||||
} catch (err) {
|
||||
this.logger.info({err}, `${this.aor}: error sending un-REGISTER`);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
async register(srf) {
|
||||
const { createEphemeralGateway } = srf.locals.realtimeDbHelpers;
|
||||
const { updateVoipCarriersRegisterStatus } = srf.locals.dbHelpers;
|
||||
const { writeAlerts, AlertType, localSIPDomain } = srf.locals;
|
||||
const { writeAlerts, AlertType } = srf.locals;
|
||||
// stamp this attempt so a late/stale response (e.g. after a watchdog-driven retry) is ignored
|
||||
const epoch = ++this.epoch;
|
||||
try {
|
||||
// transport
|
||||
const transport = (this.protocol.includes('/') ? this.protocol.substring(0, this.protocol.indexOf('/')) :
|
||||
this.protocol).toLowerCase();
|
||||
|
||||
// scheme
|
||||
let scheme = 'sip';
|
||||
if (transport === 'tls' && this.use_sips_scheme) scheme = 'sips';
|
||||
|
||||
let publicAddress = srf.locals.sbcPublicIpAddress.udp;
|
||||
if (transport !== 'udp') {
|
||||
if (srf.locals.sbcPublicIpAddress[transport]) {
|
||||
publicAddress = srf.locals.sbcPublicIpAddress[transport];
|
||||
}
|
||||
else if (transport === 'tls') {
|
||||
publicAddress = srf.locals.sbcPublicIpAddress.udp;
|
||||
}
|
||||
}
|
||||
|
||||
let contactAddress = this.aor;
|
||||
if (this.use_public_ip_in_contact) {
|
||||
contactAddress = `${this.fromUser}@${publicAddress}`;
|
||||
}
|
||||
else if (this.account_sip_realm) {
|
||||
contactAddress = `${this.fromUser}@${this.account_sip_realm}`;
|
||||
}
|
||||
else if (localSIPDomain) {
|
||||
contactAddress = `${this.fromUser}@${localSIPDomain}`;
|
||||
}
|
||||
|
||||
this.logger.debug(`sending REGISTER for ${this.aor}`);
|
||||
|
||||
let proxy;
|
||||
if (this.outbound_sip_proxy) {
|
||||
proxy = `sip:${this.outbound_sip_proxy};transport=${transport}`;
|
||||
this.logger.debug(`sending via proxy ${proxy}`);
|
||||
} else {
|
||||
const isIPv4 = isValidIPv4(this.ipv4);
|
||||
proxy = `sip:${this.ipv4}${isIPv4 ? `:${this.port}` : ''};transport=${transport}`;
|
||||
this.logger.debug(`sending to registrar ${proxy}`);
|
||||
}
|
||||
const {transport, scheme, publicAddress, contactAddress, proxy, callId} = this._requestParams(srf);
|
||||
this.logger.debug(`sending REGISTER for ${this.aor} to ${proxy}`);
|
||||
/* Safety net for a REGISTER whose SIP response never arrives at all (srf.request has no
|
||||
timeout) -- e.g. drachtio dropped the socket mid-transaction so we never even get its
|
||||
408. Treat it exactly like the failure path (mark fail, record status, back off on
|
||||
@@ -242,7 +284,7 @@ class Regbot {
|
||||
method: 'REGISTER',
|
||||
proxy,
|
||||
headers: {
|
||||
'Call-ID': this.sip_gateway_sid,
|
||||
'Call-ID': callId,
|
||||
'From': this.from,
|
||||
'To': this.from,
|
||||
'Contact': `<${scheme}:${contactAddress};transport=${transport}>;expires=${DEFAULT_EXPIRES}`,
|
||||
@@ -353,6 +395,7 @@ class Regbot {
|
||||
reason: `${res.status} ${res.reason}`,
|
||||
cseq: req.get('Cseq'),
|
||||
callId: req.get('Call-Id'),
|
||||
sbcAddress: publicAddress,
|
||||
timestamp: timestamp,
|
||||
expires: expires
|
||||
}));
|
||||
|
||||
@@ -4,6 +4,7 @@ const {
|
||||
JAMBONES_CLUSTER_ID,
|
||||
JAMBONES_REGBOT_BATCH_SLEEP_MS,
|
||||
JAMBONES_REGBOT_BATCH_SIZE,
|
||||
AWS_LIFECYCLE_DRAIN,
|
||||
} = require('./config');
|
||||
const short = require('short-uuid');
|
||||
const Regbot = require('./regbot');
|
||||
@@ -11,6 +12,11 @@ const { sleepFor } = require('./utils');
|
||||
|
||||
const MAX_INITIAL_DELAY = 15;
|
||||
const REGBOT_STATUS_CHECK_INTERVAL = 60;
|
||||
/* scale-in handoff: how often to look for a successor, how long to wait for one (a successor
|
||||
claims within one status check interval), and how far to trail its registrations */
|
||||
const HANDOFF_POLL_MS = 5000;
|
||||
const HANDOFF_TIMEOUT_MS = 2 * REGBOT_STATUS_CHECK_INTERVAL * 1000;
|
||||
const HANDOFF_GRACE_MS = 10000;
|
||||
const regbotKey = `${(JAMBONES_CLUSTER_ID || 'default')}:regbot-token`;
|
||||
const waitFor = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
||||
let initialized = false;
|
||||
@@ -137,6 +143,13 @@ module.exports = async(logger, srf) => {
|
||||
/* check every so often if I need to go from inactive->active (or vice versa) */
|
||||
setInterval(checkStatus.bind(null, logger, srf), REGBOT_STATUS_CHECK_INTERVAL * 1000);
|
||||
|
||||
/* on an AWS scale-in, hand the regbot role to another SBC while we are still up */
|
||||
if (AWS_LIFECYCLE_DRAIN) {
|
||||
require('./aws-lifecycle')(logger, () => {
|
||||
handoff(logger, srf).catch((err) => logger.error({err}, 'regbot handoff failed'));
|
||||
});
|
||||
}
|
||||
|
||||
/* if I am the regbot holder, then kick it off */
|
||||
if (srf.locals.regbot.active) {
|
||||
updateCarrierRegbots(logger, srf)
|
||||
@@ -150,7 +163,10 @@ module.exports = async(logger, srf) => {
|
||||
|
||||
const checkStatus = async(logger, srf) => {
|
||||
const { addKeyNx, addKey, retrieveKey } = srf.locals.realtimeDbHelpers;
|
||||
const { myToken, active } = srf.locals.regbot;
|
||||
const { myToken, active, draining } = srf.locals.regbot;
|
||||
|
||||
/* a draining SBC has handed off the role and must never take it back */
|
||||
if (draining) return;
|
||||
|
||||
logger.info({ active, myToken }, 'checking in on regbot status');
|
||||
try {
|
||||
@@ -409,9 +425,79 @@ const resync = async(logger, srf) => {
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Called when this SBC starts draining for scale-in. If we hold the regbot role, give it up while
|
||||
* we are still running rather than letting the lease expire after we are gone: the draining SBC
|
||||
* answers new INVITEs with 503, and the carriers would otherwise keep sending calls for
|
||||
* registration trunks here until the other SBC noticed the lapsed lease.
|
||||
*
|
||||
* Once another SBC has claimed the role (and so is registering with its own Call-ID), remove the
|
||||
* bindings whose Contact points at this SBC, so registrars stop routing to an address that is
|
||||
* going away.
|
||||
*/
|
||||
const handoff = async(logger, srf, opts = {}) => {
|
||||
const {
|
||||
pollMs = HANDOFF_POLL_MS,
|
||||
timeoutMs = HANDOFF_TIMEOUT_MS,
|
||||
graceMs = HANDOFF_GRACE_MS
|
||||
} = opts;
|
||||
const { retrieveKey, deleteKey } = srf.locals.realtimeDbHelpers;
|
||||
const { myToken } = srf.locals.regbot;
|
||||
|
||||
srf.locals.regbot.draining = true;
|
||||
if (!srf.locals.regbot.active) {
|
||||
logger.info('scale-in: not the regbot holder, nothing to hand off');
|
||||
return;
|
||||
}
|
||||
|
||||
/* let an in-flight rebuild finish so it cannot start regbots after we stop them */
|
||||
while (rebuildInProgress) await waitFor(250);
|
||||
|
||||
srf.locals.regbot.active = false;
|
||||
const handedOff = regbots.splice(0, regbots.length);
|
||||
/* stop refreshing but keep the ephemeral gateways: the new holder overwrites them */
|
||||
handedOff.forEach((rb) => rb.stopTimer());
|
||||
carriersHash = '';
|
||||
gatewaysHash = '';
|
||||
|
||||
if (await retrieveKey(regbotKey) === myToken) await deleteKey(regbotKey);
|
||||
logger.info(`scale-in: released regbot role (${handedOff.length} regbots), waiting for another SBC to claim it`);
|
||||
|
||||
let successor;
|
||||
for (const deadline = Date.now() + timeoutMs; Date.now() < deadline;) {
|
||||
await waitFor(pollMs);
|
||||
const token = await retrieveKey(regbotKey);
|
||||
if (token && token !== myToken) {
|
||||
successor = token;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!successor) {
|
||||
logger.info('scale-in: no other SBC claimed the regbot role; leaving our bindings to expire');
|
||||
return;
|
||||
}
|
||||
|
||||
/* the successor registers in batches from the moment it claims the role; trail it by graceMs
|
||||
at the same pace so each of its bindings is in place before we remove the matching old one */
|
||||
await waitFor(graceMs);
|
||||
const ours = handedOff.filter((rb) => rb.use_public_ip_in_contact);
|
||||
logger.info(`scale-in: SBC ${successor} now holds the regbot role; ` +
|
||||
`un-registering ${ours.length} bindings that point at this SBC`);
|
||||
let batch_count = 0;
|
||||
for (const rb of ours) {
|
||||
await rb.unregister(srf);
|
||||
if (++batch_count >= JAMBONES_REGBOT_BATCH_SIZE) {
|
||||
batch_count = 0;
|
||||
await sleepFor(JAMBONES_REGBOT_BATCH_SLEEP_MS);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
module.exports.resync = resync;
|
||||
module.exports.handoff = handoff;
|
||||
|
||||
// exposed for unit testing: stop all regbots and reset module state
|
||||
module.exports._addForTest = (rb) => regbots.push(rb);
|
||||
module.exports._resetForTest = () => {
|
||||
regbots.forEach((rb) => {
|
||||
rb.retired = true;
|
||||
|
||||
Reference in New Issue
Block a user