mirror of
https://github.com/jambonz/sbc-inbound.git
synced 2026-10-04 02:04:22 +00:00
fix: complete autoscale scale-in reliably and reject new INVITEs while draining (#248)
Scale-in completion never happened: app.js holds the placeholder Emitter that autoscale-manager returns synchronously (the real SnsNotifier replaces it later inside an async IIFE), so the completion poller never saw operationalState change; it also called the nonexistent scaleIn() rather than completeScaleIn(). Instances in Terminating:Wait therefore always burned the full lifecycle hook heartbeat timeout. In addition, nothing consumed dryUpCalls: a draining SBC kept accepting new INVITEs sent directly to its public address right up until termination. Changes: - complete the scale-in from within the ScaleIn handler in autoscale-manager, where the real notifier is in scope - while draining, reject new INVITEs with 503 so senders fail over to another SBC (INVITE with Replaces is allowed through since it targets a call already in progress here) - a server may run several sbc-inbound and sbc-outbound processes, and completing the hook when only this process is idle would terminate the instance while sibling processes still have calls; each process now reports its call count to redis (lib/call-count-reporter.js, with a companion change in sbc-outbound) and the draining process completes only when the server-wide count is zero on two consecutive checks, falling back to its own count if no reports are present Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
ccbebc015f
commit
b7b707cc2e
@@ -155,6 +155,17 @@ srf.locals = {
|
|||||||
};
|
};
|
||||||
const activeCallIds = srf.locals.activeCallIds;
|
const activeCallIds = srf.locals.activeCallIds;
|
||||||
|
|
||||||
|
/* report our call count to redis so a draining process can count calls across
|
||||||
|
all sbc-inbound and sbc-outbound processes on this server */
|
||||||
|
if (!process.env.K8S && 'test' !== process.env.NODE_ENV) {
|
||||||
|
srf.locals.callCountReporter = require('./lib/call-count-reporter')({
|
||||||
|
logger,
|
||||||
|
addKey,
|
||||||
|
addToSet,
|
||||||
|
getCount: () => activeCallIds.size
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
const {
|
const {
|
||||||
initLocals,
|
initLocals,
|
||||||
handleSipRec,
|
handleSipRec,
|
||||||
|
|||||||
@@ -23,23 +23,55 @@ module.exports = (logger) => {
|
|||||||
const {srf} = require('..');
|
const {srf} = require('..');
|
||||||
const {activeCallIds, removeFromRedis} = srf.locals;
|
const {activeCallIds, removeFromRedis} = srf.locals;
|
||||||
|
|
||||||
|
/* reject new INVITEs with 503 so senders fail over to another SBC */
|
||||||
|
srf.locals.dryUpCalls = true;
|
||||||
|
|
||||||
/* remove our private IP from the set of active SBCs so rtp and fs know we are gone */
|
/* remove our private IP from the set of active SBCs so rtp and fs know we are gone */
|
||||||
removeFromRedis();
|
removeFromRedis();
|
||||||
|
|
||||||
/* if we have zero calls, we can complete the scale-in right now */
|
/* count calls in progress across all sbc-inbound and sbc-outbound
|
||||||
const calls = activeCallIds.size;
|
processes on this server, if they are reporting; otherwise
|
||||||
if (0 === calls) {
|
fall back to counting only our own */
|
||||||
logger.info('scale-in can complete immediately as we have no calls in progress');
|
const countServerCalls = async() => {
|
||||||
lifecycleEmitter.completeScaleIn();
|
const reporter = srf.locals.callCountReporter;
|
||||||
}
|
if (!reporter) return activeCallIds.size;
|
||||||
else {
|
const {retrieveSet, retrieveKey} = srf.locals.realtimeDbHelpers;
|
||||||
logger.info(`${calls} calls in progress; scale-in will complete when they are done`);
|
const keys = await retrieveSet(reporter.setName);
|
||||||
}
|
let count = 0;
|
||||||
|
for (const key of keys) {
|
||||||
|
count += parseInt(await retrieveKey(key), 10) || 0;
|
||||||
|
}
|
||||||
|
return Math.max(count, activeCallIds.size);
|
||||||
|
};
|
||||||
|
|
||||||
|
/* poll until calls have dried up, then complete the scale-in;
|
||||||
|
require two consecutive zero readings since reported counts
|
||||||
|
may be up to 15s stale */
|
||||||
|
let consecutiveZeroCounts = 0;
|
||||||
|
const timer = setInterval(async() => {
|
||||||
|
try {
|
||||||
|
const calls = await countServerCalls();
|
||||||
|
if (0 === calls) {
|
||||||
|
if (++consecutiveZeroCounts >= 2) {
|
||||||
|
clearInterval(timer);
|
||||||
|
logger.info('scale-in complete now that calls have dried up');
|
||||||
|
lifecycleEmitter.completeScaleIn();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
consecutiveZeroCounts = 0;
|
||||||
|
logger.info(`${calls} calls in progress on this server; scale-in will complete when they are done`);
|
||||||
|
}
|
||||||
|
} catch (err) {
|
||||||
|
logger.error({err}, 'Error counting calls in progress during scale-in');
|
||||||
|
}
|
||||||
|
}, 20000);
|
||||||
})
|
})
|
||||||
.on(LifeCycleEvents.StandbyEnter, () => {
|
.on(LifeCycleEvents.StandbyEnter, () => {
|
||||||
lifecycleEmitter.dryUpCalls = true;
|
lifecycleEmitter.dryUpCalls = true;
|
||||||
const {srf} = require('..');
|
const {srf} = require('..');
|
||||||
const {removeFromRedis} = srf.locals;
|
const {removeFromRedis} = srf.locals;
|
||||||
|
srf.locals.dryUpCalls = true;
|
||||||
removeFromRedis();
|
removeFromRedis();
|
||||||
|
|
||||||
logger.info('AWS enter pending state notification: begin drying up calls');
|
logger.info('AWS enter pending state notification: begin drying up calls');
|
||||||
@@ -48,6 +80,7 @@ module.exports = (logger) => {
|
|||||||
lifecycleEmitter.dryUpCalls = false;
|
lifecycleEmitter.dryUpCalls = false;
|
||||||
const {srf} = require('..');
|
const {srf} = require('..');
|
||||||
const {addToRedis} = srf.locals;
|
const {addToRedis} = srf.locals;
|
||||||
|
srf.locals.dryUpCalls = false;
|
||||||
addToRedis();
|
addToRedis();
|
||||||
|
|
||||||
logger.info('AWS exit pending state notification: re-enable calls');
|
logger.info('AWS exit pending state notification: re-enable calls');
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
const os = require('os');
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Periodically report this process's count of calls in progress to redis.
|
||||||
|
* A server may host several sbc-inbound and sbc-outbound processes; when one
|
||||||
|
* of them handles an autoscale drain it needs to know when the entire server
|
||||||
|
* has no calls in progress, not just its own process. Each process writes
|
||||||
|
* its own count under a per-pid key (with a short expiry, so keys from dead
|
||||||
|
* processes evaporate) and registers that key in a per-host set that the
|
||||||
|
* draining process can enumerate.
|
||||||
|
*/
|
||||||
|
const REPORT_INTERVAL = 15000;
|
||||||
|
const KEY_EXPIRY_SECS = 120;
|
||||||
|
|
||||||
|
module.exports = ({logger, addKey, addToSet, getCount}) => {
|
||||||
|
const prefix = process.env.JAMBONES_CLUSTER_ID || 'default';
|
||||||
|
const setName = `${prefix}:call-count-keys:${os.hostname()}`;
|
||||||
|
const key = `${prefix}:call-count:${os.hostname()}:${process.pid}`;
|
||||||
|
|
||||||
|
const report = () => {
|
||||||
|
addKey(key, `${getCount()}`, KEY_EXPIRY_SECS)
|
||||||
|
.catch((err) => logger.error({err}, 'call-count-reporter: error writing call count'));
|
||||||
|
};
|
||||||
|
|
||||||
|
addToSet(setName, key)
|
||||||
|
.catch((err) => logger.error({err}, `call-count-reporter: error adding ${key} to ${setName}`));
|
||||||
|
setInterval(report, REPORT_INTERVAL);
|
||||||
|
report();
|
||||||
|
|
||||||
|
return {key, setName};
|
||||||
|
};
|
||||||
@@ -34,6 +34,15 @@ module.exports = function(srf, logger) {
|
|||||||
|
|
||||||
const initLocals = (req, res, next) => {
|
const initLocals = (req, res, next) => {
|
||||||
const callId = req.get('Call-ID');
|
const callId = req.get('Call-ID');
|
||||||
|
|
||||||
|
/* if we are drying up calls prior to scale-in, reject new INVITEs so the
|
||||||
|
sender fails over to another SBC; allow INVITE with Replaces through
|
||||||
|
since it targets a call already in progress on this server */
|
||||||
|
if (srf.locals.dryUpCalls && !req.has('Replaces')) {
|
||||||
|
logger.info({callId}, 'rejecting INVITE with 503 as we are drying up calls before scale-in');
|
||||||
|
return res.send(503);
|
||||||
|
}
|
||||||
|
|
||||||
req.locals = req.locals || {callId};
|
req.locals = req.locals || {callId};
|
||||||
req.locals.nudge = 0;
|
req.locals.nudge = 0;
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user