mirror of
https://github.com/jambonz/sbc-inbound.git
synced 2026-10-04 02:04:22 +00:00
Scale-in completion never happened: app.js holds the placeholder Emitter that autoscale-manager returns synchronously (the real SnsNotifier replaces it later inside an async IIFE), so the completion poller never saw operationalState change; it also called the nonexistent scaleIn() rather than completeScaleIn(). Instances in Terminating:Wait therefore always burned the full lifecycle hook heartbeat timeout. In addition, nothing consumed dryUpCalls: a draining SBC kept accepting new INVITEs sent directly to its public address right up until termination. Changes: - complete the scale-in from within the ScaleIn handler in autoscale-manager, where the real notifier is in scope - while draining, reject new INVITEs with 503 so senders fail over to another SBC (INVITE with Replaces is allowed through since it targets a call already in progress here) - a server may run several sbc-inbound and sbc-outbound processes, and completing the hook when only this process is idle would terminate the instance while sibling processes still have calls; each process now reports its call count to redis (lib/call-count-reporter.js, with a companion change in sbc-outbound) and the draining process completes only when the server-wide count is zero on two consecutive checks, falling back to its own count if no reports are present Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
100 lines
3.8 KiB
JavaScript
100 lines
3.8 KiB
JavaScript
const noopLogger = {info: () => {}, error: () => {}};
|
|
const {LifeCycleEvents} = require('./constants');
|
|
const Emitter = require('events');
|
|
|
|
module.exports = (logger) => {
|
|
logger = logger || noopLogger;
|
|
|
|
// listen for SNS lifecycle changes
|
|
let lifecycleEmitter = new Emitter();
|
|
lifecycleEmitter.dryUpCalls = false;
|
|
if (process.env.AWS_SNS_TOPIC_ARN) {
|
|
|
|
(async function() {
|
|
try {
|
|
lifecycleEmitter = await require('./aws-sns-lifecycle')(logger);
|
|
|
|
lifecycleEmitter
|
|
.on(LifeCycleEvents.ScaleIn, async() => {
|
|
logger.info('AWS scale-in notification: begin drying up calls');
|
|
lifecycleEmitter.dryUpCalls = true;
|
|
lifecycleEmitter.operationalState = LifeCycleEvents.ScaleIn;
|
|
|
|
const {srf} = require('..');
|
|
const {activeCallIds, removeFromRedis} = srf.locals;
|
|
|
|
/* reject new INVITEs with 503 so senders fail over to another SBC */
|
|
srf.locals.dryUpCalls = true;
|
|
|
|
/* remove our private IP from the set of active SBCs so rtp and fs know we are gone */
|
|
removeFromRedis();
|
|
|
|
/* count calls in progress across all sbc-inbound and sbc-outbound
|
|
processes on this server, if they are reporting; otherwise
|
|
fall back to counting only our own */
|
|
const countServerCalls = async() => {
|
|
const reporter = srf.locals.callCountReporter;
|
|
if (!reporter) return activeCallIds.size;
|
|
const {retrieveSet, retrieveKey} = srf.locals.realtimeDbHelpers;
|
|
const keys = await retrieveSet(reporter.setName);
|
|
let count = 0;
|
|
for (const key of keys) {
|
|
count += parseInt(await retrieveKey(key), 10) || 0;
|
|
}
|
|
return Math.max(count, activeCallIds.size);
|
|
};
|
|
|
|
/* poll until calls have dried up, then complete the scale-in;
|
|
require two consecutive zero readings since reported counts
|
|
may be up to 15s stale */
|
|
let consecutiveZeroCounts = 0;
|
|
const timer = setInterval(async() => {
|
|
try {
|
|
const calls = await countServerCalls();
|
|
if (0 === calls) {
|
|
if (++consecutiveZeroCounts >= 2) {
|
|
clearInterval(timer);
|
|
logger.info('scale-in complete now that calls have dried up');
|
|
lifecycleEmitter.completeScaleIn();
|
|
}
|
|
}
|
|
else {
|
|
consecutiveZeroCounts = 0;
|
|
logger.info(`${calls} calls in progress on this server; scale-in will complete when they are done`);
|
|
}
|
|
} catch (err) {
|
|
logger.error({err}, 'Error counting calls in progress during scale-in');
|
|
}
|
|
}, 20000);
|
|
})
|
|
.on(LifeCycleEvents.StandbyEnter, () => {
|
|
lifecycleEmitter.dryUpCalls = true;
|
|
const {srf} = require('..');
|
|
const {removeFromRedis} = srf.locals;
|
|
srf.locals.dryUpCalls = true;
|
|
removeFromRedis();
|
|
|
|
logger.info('AWS enter pending state notification: begin drying up calls');
|
|
})
|
|
.on(LifeCycleEvents.StandbyExit, () => {
|
|
lifecycleEmitter.dryUpCalls = false;
|
|
const {srf} = require('..');
|
|
const {addToRedis} = srf.locals;
|
|
srf.locals.dryUpCalls = false;
|
|
addToRedis();
|
|
|
|
logger.info('AWS exit pending state notification: re-enable calls');
|
|
});
|
|
} catch (err) {
|
|
logger.error({err}, 'Failure creating SNS notifier, lifecycle events will be disabled');
|
|
}
|
|
})();
|
|
}
|
|
else if (process.env.K8S) {
|
|
lifecycleEmitter.scaleIn = () => process.exit(0);
|
|
}
|
|
|
|
return {lifecycleEmitter};
|
|
};
|
|
|