fix: complete autoscale scale-in reliably and reject new INVITEs while draining (#248)

Scale-in completion never happened: app.js holds the placeholder
Emitter that autoscale-manager returns synchronously (the real
SnsNotifier replaces it later inside an async IIFE), so the completion
poller never saw operationalState change; it also called the
nonexistent scaleIn() rather than completeScaleIn().  Instances in
Terminating:Wait therefore always burned the full lifecycle hook
heartbeat timeout.

In addition, nothing consumed dryUpCalls: a draining SBC kept accepting
new INVITEs sent directly to its public address right up until
termination.

Changes:
- complete the scale-in from within the ScaleIn handler in
  autoscale-manager, where the real notifier is in scope
- while draining, reject new INVITEs with 503 so senders fail over to
  another SBC (INVITE with Replaces is allowed through since it targets
  a call already in progress here)
- a server may run several sbc-inbound and sbc-outbound processes, and
  completing the hook when only this process is idle would terminate
  the instance while sibling processes still have calls; each process
  now reports its call count to redis (lib/call-count-reporter.js, with
  a companion change in sbc-outbound) and the draining process
  completes only when the server-wide count is zero on two consecutive
  checks, falling back to its own count if no reports are present

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Dave Horton
2026-08-31 08:54:19 -04:00
committed by GitHub
co-authored by Claude Fable 5
parent ccbebc015f
commit b7b707cc2e
4 changed files with 93 additions and 9 deletions
+11
View File
@@ -155,6 +155,17 @@ srf.locals = {
};
const activeCallIds = srf.locals.activeCallIds;
/* report our call count to redis so a draining process can count calls across
all sbc-inbound and sbc-outbound processes on this server */
if (!process.env.K8S && 'test' !== process.env.NODE_ENV) {
srf.locals.callCountReporter = require('./lib/call-count-reporter')({
logger,
addKey,
addToSet,
getCount: () => activeCallIds.size
});
}
const {
initLocals,
handleSipRec,
+42 -9
View File
@@ -23,23 +23,55 @@ module.exports = (logger) => {
const {srf} = require('..');
const {activeCallIds, removeFromRedis} = srf.locals;
/* reject new INVITEs with 503 so senders fail over to another SBC */
srf.locals.dryUpCalls = true;
/* remove our private IP from the set of active SBCs so rtp and fs know we are gone */
removeFromRedis();
/* if we have zero calls, we can complete the scale-in right now */
const calls = activeCallIds.size;
if (0 === calls) {
logger.info('scale-in can complete immediately as we have no calls in progress');
lifecycleEmitter.completeScaleIn();
}
else {
logger.info(`${calls} calls in progress; scale-in will complete when they are done`);
}
/* count calls in progress across all sbc-inbound and sbc-outbound
processes on this server, if they are reporting; otherwise
fall back to counting only our own */
const countServerCalls = async() => {
const reporter = srf.locals.callCountReporter;
if (!reporter) return activeCallIds.size;
const {retrieveSet, retrieveKey} = srf.locals.realtimeDbHelpers;
const keys = await retrieveSet(reporter.setName);
let count = 0;
for (const key of keys) {
count += parseInt(await retrieveKey(key), 10) || 0;
}
return Math.max(count, activeCallIds.size);
};
/* poll until calls have dried up, then complete the scale-in;
require two consecutive zero readings since reported counts
may be up to 15s stale */
let consecutiveZeroCounts = 0;
const timer = setInterval(async() => {
try {
const calls = await countServerCalls();
if (0 === calls) {
if (++consecutiveZeroCounts >= 2) {
clearInterval(timer);
logger.info('scale-in complete now that calls have dried up');
lifecycleEmitter.completeScaleIn();
}
}
else {
consecutiveZeroCounts = 0;
logger.info(`${calls} calls in progress on this server; scale-in will complete when they are done`);
}
} catch (err) {
logger.error({err}, 'Error counting calls in progress during scale-in');
}
}, 20000);
})
.on(LifeCycleEvents.StandbyEnter, () => {
lifecycleEmitter.dryUpCalls = true;
const {srf} = require('..');
const {removeFromRedis} = srf.locals;
srf.locals.dryUpCalls = true;
removeFromRedis();
logger.info('AWS enter pending state notification: begin drying up calls');
@@ -48,6 +80,7 @@ module.exports = (logger) => {
lifecycleEmitter.dryUpCalls = false;
const {srf} = require('..');
const {addToRedis} = srf.locals;
srf.locals.dryUpCalls = false;
addToRedis();
logger.info('AWS exit pending state notification: re-enable calls');
+31
View File
@@ -0,0 +1,31 @@
const os = require('os');
/**
* Periodically report this process's count of calls in progress to redis.
* A server may host several sbc-inbound and sbc-outbound processes; when one
* of them handles an autoscale drain it needs to know when the entire server
* has no calls in progress, not just its own process. Each process writes
* its own count under a per-pid key (with a short expiry, so keys from dead
* processes evaporate) and registers that key in a per-host set that the
* draining process can enumerate.
*/
const REPORT_INTERVAL = 15000;
const KEY_EXPIRY_SECS = 120;
module.exports = ({logger, addKey, addToSet, getCount}) => {
const prefix = process.env.JAMBONES_CLUSTER_ID || 'default';
const setName = `${prefix}:call-count-keys:${os.hostname()}`;
const key = `${prefix}:call-count:${os.hostname()}:${process.pid}`;
const report = () => {
addKey(key, `${getCount()}`, KEY_EXPIRY_SECS)
.catch((err) => logger.error({err}, 'call-count-reporter: error writing call count'));
};
addToSet(setName, key)
.catch((err) => logger.error({err}, `call-count-reporter: error adding ${key} to ${setName}`));
setInterval(report, REPORT_INTERVAL);
report();
return {key, setName};
};
+9
View File
@@ -34,6 +34,15 @@ module.exports = function(srf, logger) {
const initLocals = (req, res, next) => {
const callId = req.get('Call-ID');
/* if we are drying up calls prior to scale-in, reject new INVITEs so the
sender fails over to another SBC; allow INVITE with Replaces through
since it targets a call already in progress on this server */
if (srf.locals.dryUpCalls && !req.has('Replaces')) {
logger.info({callId}, 'rejecting INVITE with 503 as we are drying up calls before scale-in');
return res.send(503);
}
req.locals = req.locals || {callId};
req.locals.nudge = 0;