From b7b707cc2e2a1025623076f16446ea61bae429e0 Mon Sep 17 00:00:00 2001 From: Dave Horton Date: Mon, 31 Aug 2026 08:54:19 -0400 Subject: [PATCH] fix: complete autoscale scale-in reliably and reject new INVITEs while draining (#248) Scale-in completion never happened: app.js holds the placeholder Emitter that autoscale-manager returns synchronously (the real SnsNotifier replaces it later inside an async IIFE), so the completion poller never saw operationalState change; it also called the nonexistent scaleIn() rather than completeScaleIn(). Instances in Terminating:Wait therefore always burned the full lifecycle hook heartbeat timeout. In addition, nothing consumed dryUpCalls: a draining SBC kept accepting new INVITEs sent directly to its public address right up until termination. Changes: - complete the scale-in from within the ScaleIn handler in autoscale-manager, where the real notifier is in scope - while draining, reject new INVITEs with 503 so senders fail over to another SBC (INVITE with Replaces is allowed through since it targets a call already in progress here) - a server may run several sbc-inbound and sbc-outbound processes, and completing the hook when only this process is idle would terminate the instance while sibling processes still have calls; each process now reports its call count to redis (lib/call-count-reporter.js, with a companion change in sbc-outbound) and the draining process completes only when the server-wide count is zero on two consecutive checks, falling back to its own count if no reports are present Co-authored-by: Claude Fable 5 --- app.js | 11 ++++++++ lib/autoscale-manager.js | 51 +++++++++++++++++++++++++++++++------- lib/call-count-reporter.js | 31 +++++++++++++++++++++++ lib/middleware.js | 9 +++++++ 4 files changed, 93 insertions(+), 9 deletions(-) create mode 100644 lib/call-count-reporter.js diff --git a/app.js b/app.js index 6f64b76..83a29f3 100644 --- a/app.js +++ b/app.js @@ -155,6 +155,17 @@ srf.locals = { }; const activeCallIds = srf.locals.activeCallIds; +/* report our call count to redis so a draining process can count calls across + all sbc-inbound and sbc-outbound processes on this server */ +if (!process.env.K8S && 'test' !== process.env.NODE_ENV) { + srf.locals.callCountReporter = require('./lib/call-count-reporter')({ + logger, + addKey, + addToSet, + getCount: () => activeCallIds.size + }); +} + const { initLocals, handleSipRec, diff --git a/lib/autoscale-manager.js b/lib/autoscale-manager.js index 8cc5551..916c4e7 100644 --- a/lib/autoscale-manager.js +++ b/lib/autoscale-manager.js @@ -23,23 +23,55 @@ module.exports = (logger) => { const {srf} = require('..'); const {activeCallIds, removeFromRedis} = srf.locals; + /* reject new INVITEs with 503 so senders fail over to another SBC */ + srf.locals.dryUpCalls = true; + /* remove our private IP from the set of active SBCs so rtp and fs know we are gone */ removeFromRedis(); - /* if we have zero calls, we can complete the scale-in right now */ - const calls = activeCallIds.size; - if (0 === calls) { - logger.info('scale-in can complete immediately as we have no calls in progress'); - lifecycleEmitter.completeScaleIn(); - } - else { - logger.info(`${calls} calls in progress; scale-in will complete when they are done`); - } + /* count calls in progress across all sbc-inbound and sbc-outbound + processes on this server, if they are reporting; otherwise + fall back to counting only our own */ + const countServerCalls = async() => { + const reporter = srf.locals.callCountReporter; + if (!reporter) return activeCallIds.size; + const {retrieveSet, retrieveKey} = srf.locals.realtimeDbHelpers; + const keys = await retrieveSet(reporter.setName); + let count = 0; + for (const key of keys) { + count += parseInt(await retrieveKey(key), 10) || 0; + } + return Math.max(count, activeCallIds.size); + }; + + /* poll until calls have dried up, then complete the scale-in; + require two consecutive zero readings since reported counts + may be up to 15s stale */ + let consecutiveZeroCounts = 0; + const timer = setInterval(async() => { + try { + const calls = await countServerCalls(); + if (0 === calls) { + if (++consecutiveZeroCounts >= 2) { + clearInterval(timer); + logger.info('scale-in complete now that calls have dried up'); + lifecycleEmitter.completeScaleIn(); + } + } + else { + consecutiveZeroCounts = 0; + logger.info(`${calls} calls in progress on this server; scale-in will complete when they are done`); + } + } catch (err) { + logger.error({err}, 'Error counting calls in progress during scale-in'); + } + }, 20000); }) .on(LifeCycleEvents.StandbyEnter, () => { lifecycleEmitter.dryUpCalls = true; const {srf} = require('..'); const {removeFromRedis} = srf.locals; + srf.locals.dryUpCalls = true; removeFromRedis(); logger.info('AWS enter pending state notification: begin drying up calls'); @@ -48,6 +80,7 @@ module.exports = (logger) => { lifecycleEmitter.dryUpCalls = false; const {srf} = require('..'); const {addToRedis} = srf.locals; + srf.locals.dryUpCalls = false; addToRedis(); logger.info('AWS exit pending state notification: re-enable calls'); diff --git a/lib/call-count-reporter.js b/lib/call-count-reporter.js new file mode 100644 index 0000000..0e39ad6 --- /dev/null +++ b/lib/call-count-reporter.js @@ -0,0 +1,31 @@ +const os = require('os'); + +/** + * Periodically report this process's count of calls in progress to redis. + * A server may host several sbc-inbound and sbc-outbound processes; when one + * of them handles an autoscale drain it needs to know when the entire server + * has no calls in progress, not just its own process. Each process writes + * its own count under a per-pid key (with a short expiry, so keys from dead + * processes evaporate) and registers that key in a per-host set that the + * draining process can enumerate. + */ +const REPORT_INTERVAL = 15000; +const KEY_EXPIRY_SECS = 120; + +module.exports = ({logger, addKey, addToSet, getCount}) => { + const prefix = process.env.JAMBONES_CLUSTER_ID || 'default'; + const setName = `${prefix}:call-count-keys:${os.hostname()}`; + const key = `${prefix}:call-count:${os.hostname()}:${process.pid}`; + + const report = () => { + addKey(key, `${getCount()}`, KEY_EXPIRY_SECS) + .catch((err) => logger.error({err}, 'call-count-reporter: error writing call count')); + }; + + addToSet(setName, key) + .catch((err) => logger.error({err}, `call-count-reporter: error adding ${key} to ${setName}`)); + setInterval(report, REPORT_INTERVAL); + report(); + + return {key, setName}; +}; diff --git a/lib/middleware.js b/lib/middleware.js index b6f84f8..c9209f4 100644 --- a/lib/middleware.js +++ b/lib/middleware.js @@ -34,6 +34,15 @@ module.exports = function(srf, logger) { const initLocals = (req, res, next) => { const callId = req.get('Call-ID'); + + /* if we are drying up calls prior to scale-in, reject new INVITEs so the + sender fails over to another SBC; allow INVITE with Replaces through + since it targets a call already in progress on this server */ + if (srf.locals.dryUpCalls && !req.has('Replaces')) { + logger.info({callId}, 'rejecting INVITE with 503 as we are drying up calls before scale-in'); + return res.send(503); + } + req.locals = req.locals || {callId}; req.locals.nudge = 0;