[glm-grade=A] fix(monitoring): DC-090 outage incidents follow displayed hysteresis status
checkForIncidents compared raw probe transitions while the dashboard badge (DC-086) follows post-hysteresis displayed status. A single raw down blip between two ups opened AND resolved a critical outage incident; a suppressed up blip during a real outage resolved it early. Incidents now open/resolve on displayed-vs-displayed transitions; previousDisplayed=null keeps legacy raw semantics for direct callers. 6 new parity tests + legacy checkService test moved to a 4-probe chain. Suite 2616/2616 (110). Verdict: urn:ump:yc5rdlnmnmhch5audc5fifgbt6d7moi2stqfs5vidsbh6x6zkvgq
This commit is contained in:
@@ -0,0 +1,186 @@
|
|||||||
|
/**
|
||||||
|
* Tests for DC-090: outage incidents follow the DISPLAYED (post-hysteresis)
|
||||||
|
* status — the same signal that flips the dashboard badge.
|
||||||
|
*
|
||||||
|
* - A single raw "down" blip that hysteresis suppresses opens NO outage
|
||||||
|
* incident (the DC-089-noted raw-transition bug).
|
||||||
|
* - A suppressed blip does not resolve a real open outage (UP_THRESHOLD=2).
|
||||||
|
* - DOWN_THRESHOLD consecutive downs open exactly ONE outage incident.
|
||||||
|
* - The incident payload carries the displayed snapshot, not the raw probe.
|
||||||
|
* - Direct callers without hysteresis state keep legacy raw semantics.
|
||||||
|
*
|
||||||
|
* The probe() helper replicates checkService's exact call order: capture the
|
||||||
|
* pre-probe raw + displayed state, recordStatus (updates both maps), then
|
||||||
|
* checkForIncidents with both previous states.
|
||||||
|
*/
|
||||||
|
|
||||||
|
'use strict';
|
||||||
|
|
||||||
|
const path = require('path');
|
||||||
|
const fs = require('fs');
|
||||||
|
const os = require('os');
|
||||||
|
|
||||||
|
// Use an isolated data dir so test history doesn't pollute the real one.
|
||||||
|
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'dashcaddy-incpar-'));
|
||||||
|
process.env.HEALTH_DATA_DIR = tmpDir;
|
||||||
|
process.env.HEALTH_CONFIG_FILE = path.join(tmpDir, 'health-config.json');
|
||||||
|
process.env.HEALTH_HISTORY_FILE = path.join(tmpDir, 'health-history.json');
|
||||||
|
|
||||||
|
// Module exports a singleton instance, not a class. Reset per-test state by
|
||||||
|
// replacing the relevant maps on the singleton in beforeEach.
|
||||||
|
const healthCheckerSingleton = require('../src/monitoring/health-checker');
|
||||||
|
const originalDownThreshold = process.env.HEALTH_DOWN_THRESHOLD;
|
||||||
|
const originalUpThreshold = process.env.HEALTH_UP_THRESHOLD;
|
||||||
|
|
||||||
|
function restoreEnv(name, value) {
|
||||||
|
if (value === undefined) delete process.env[name];
|
||||||
|
else process.env[name] = value;
|
||||||
|
}
|
||||||
|
|
||||||
|
function makeUp(serviceId = 'svc1') {
|
||||||
|
return {
|
||||||
|
serviceId,
|
||||||
|
timestamp: new Date().toISOString(),
|
||||||
|
status: 'up',
|
||||||
|
responseTime: 50,
|
||||||
|
statusCode: 200,
|
||||||
|
message: 'Service is healthy',
|
||||||
|
details: { headers: {}, bodyLength: 12 }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function makeDown(serviceId = 'svc1') {
|
||||||
|
return {
|
||||||
|
serviceId,
|
||||||
|
timestamp: new Date().toISOString(),
|
||||||
|
status: 'down',
|
||||||
|
responseTime: 50,
|
||||||
|
statusCode: 500,
|
||||||
|
message: 'fail',
|
||||||
|
details: { headers: {}, bodyLength: 0 }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
describe('DC-090: outage incidents follow the displayed (hysteresis) status', () => {
|
||||||
|
let hc;
|
||||||
|
let incidentCreatedSpy;
|
||||||
|
let incidentResolvedSpy;
|
||||||
|
|
||||||
|
beforeEach(() => {
|
||||||
|
healthCheckerSingleton.displayedStatus = new Map();
|
||||||
|
healthCheckerSingleton.consecutiveSinceChange = new Map();
|
||||||
|
healthCheckerSingleton.currentStatus = new Map();
|
||||||
|
healthCheckerSingleton.history = {};
|
||||||
|
healthCheckerSingleton.incidents = [];
|
||||||
|
healthCheckerSingleton.removeAllListeners('incident-created');
|
||||||
|
healthCheckerSingleton.removeAllListeners('incident-resolved');
|
||||||
|
incidentCreatedSpy = jest.fn();
|
||||||
|
incidentResolvedSpy = jest.fn();
|
||||||
|
healthCheckerSingleton.on('incident-created', incidentCreatedSpy);
|
||||||
|
healthCheckerSingleton.on('incident-resolved', incidentResolvedSpy);
|
||||||
|
hc = healthCheckerSingleton;
|
||||||
|
});
|
||||||
|
|
||||||
|
afterEach(() => {
|
||||||
|
restoreEnv('HEALTH_DOWN_THRESHOLD', originalDownThreshold);
|
||||||
|
restoreEnv('HEALTH_UP_THRESHOLD', originalUpThreshold);
|
||||||
|
});
|
||||||
|
|
||||||
|
afterAll(() => {
|
||||||
|
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||||
|
});
|
||||||
|
|
||||||
|
// Replicates checkService's record+incident sequence for one raw probe.
|
||||||
|
function probe(status, config = {}) {
|
||||||
|
const previousStatus = hc.currentStatus.get(status.serviceId);
|
||||||
|
const previousDisplayed = hc.displayedStatus.get(status.serviceId) || null;
|
||||||
|
hc.recordStatus(status.serviceId, status);
|
||||||
|
hc.checkForIncidents(status.serviceId, status, config, previousStatus, previousDisplayed);
|
||||||
|
}
|
||||||
|
|
||||||
|
test('a single down blip between two ups opens NO outage incident', () => {
|
||||||
|
probe(makeUp()); // baseline: displayed up
|
||||||
|
probe(makeDown()); // blip — hysteresis keeps displayed up
|
||||||
|
probe(makeUp()); // recovered
|
||||||
|
expect(hc.incidents).toHaveLength(0);
|
||||||
|
expect(incidentCreatedSpy).not.toHaveBeenCalled();
|
||||||
|
});
|
||||||
|
|
||||||
|
test('DOWN_THRESHOLD consecutive downs open exactly one outage incident (critical)', () => {
|
||||||
|
probe(makeUp());
|
||||||
|
probe(makeDown()); // counter=1, displayed still up
|
||||||
|
probe(makeDown()); // counter=2 → displayed flips down → incident
|
||||||
|
expect(hc.incidents).toHaveLength(1);
|
||||||
|
const incident = hc.incidents[0];
|
||||||
|
expect(incident.type).toBe('outage');
|
||||||
|
expect(incident.severity).toBe('critical');
|
||||||
|
expect(incident.status).toBe('open');
|
||||||
|
expect(incidentCreatedSpy).toHaveBeenCalledTimes(1);
|
||||||
|
|
||||||
|
probe(makeDown()); // still down — no new transition, no second incident
|
||||||
|
expect(hc.incidents).toHaveLength(1);
|
||||||
|
expect(incident.occurrences).toBe(1); // occurrences count displayed flips, not raw probes
|
||||||
|
expect(incidentCreatedSpy).toHaveBeenCalledTimes(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('the outage incident payload carries the displayed snapshot, not the raw blip', () => {
|
||||||
|
probe(makeUp());
|
||||||
|
const blip = makeDown();
|
||||||
|
blip.statusCode = 599;
|
||||||
|
probe(blip); // suppressed blip — must not appear in any incident
|
||||||
|
probe(makeDown()); // flip
|
||||||
|
expect(hc.incidents).toHaveLength(1);
|
||||||
|
// The incident's details snapshot is the probe that FLIPPED the displayed
|
||||||
|
// state (the second down), not the earlier suppressed blip.
|
||||||
|
expect(hc.incidents[0].details.statusCode).not.toBe(599);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('a suppressed up blip does not resolve a real open outage (UP_THRESHOLD=2)', () => {
|
||||||
|
process.env.HEALTH_UP_THRESHOLD = '2';
|
||||||
|
jest.resetModules();
|
||||||
|
const hc2 = require('../src/monitoring/health-checker');
|
||||||
|
hc2.displayedStatus = new Map();
|
||||||
|
hc2.consecutiveSinceChange = new Map();
|
||||||
|
hc2.currentStatus = new Map();
|
||||||
|
hc2.history = {};
|
||||||
|
hc2.incidents = [];
|
||||||
|
hc2.removeAllListeners('incident-created');
|
||||||
|
hc2.removeAllListeners('incident-resolved');
|
||||||
|
|
||||||
|
const p2 = (status) => {
|
||||||
|
const prevRaw = hc2.currentStatus.get(status.serviceId);
|
||||||
|
const prevDisp = hc2.displayedStatus.get(status.serviceId) || null;
|
||||||
|
hc2.recordStatus(status.serviceId, status);
|
||||||
|
hc2.checkForIncidents(status.serviceId, status, {}, prevRaw, prevDisp);
|
||||||
|
};
|
||||||
|
|
||||||
|
p2(makeUp());
|
||||||
|
p2(makeDown());
|
||||||
|
p2(makeDown()); // displayed down → outage opens
|
||||||
|
expect(hc2.incidents).toHaveLength(1);
|
||||||
|
expect(hc2.incidents[0].status).toBe('open');
|
||||||
|
|
||||||
|
p2(makeUp()); // counter=1 < UP_THRESHOLD=2 → displayed still down
|
||||||
|
expect(hc2.displayedStatus.get('svc1').status).toBe('down');
|
||||||
|
expect(hc2.incidents[0].status).toBe('open'); // NOT resolved by the blip
|
||||||
|
|
||||||
|
p2(makeUp()); // counter=2 → displayed up → incident resolves
|
||||||
|
expect(hc2.displayedStatus.get('svc1').status).toBe('up');
|
||||||
|
expect(hc2.incidents[0].status).toBe('resolved');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('legacy direct callers (no displayed state) keep raw transition semantics', () => {
|
||||||
|
hc.currentStatus.set('svc1', { status: 'up' });
|
||||||
|
const status = { status: 'down', timestamp: new Date().toISOString(), responseTime: 100 };
|
||||||
|
hc.checkForIncidents('svc1', status, {}); // 4-arg call, no previousDisplayed
|
||||||
|
expect(hc.incidents).toHaveLength(1);
|
||||||
|
expect(hc.incidents[0].type).toBe('outage');
|
||||||
|
});
|
||||||
|
|
||||||
|
test('slow-response detection still fires per-probe regardless of hysteresis', () => {
|
||||||
|
const slowUp = makeUp();
|
||||||
|
slowUp.responseTime = 6000;
|
||||||
|
probe(slowUp, { slowResponseThreshold: 5000 });
|
||||||
|
expect(hc.incidents.some(i => i.type === 'slow-response')).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -204,15 +204,22 @@ describe('HealthChecker', () => {
|
|||||||
});
|
});
|
||||||
|
|
||||||
it('opens and resolves an outage incident across real checkService transitions', async () => {
|
it('opens and resolves an outage incident across real checkService transitions', async () => {
|
||||||
|
// DC-090: incidents follow the DISPLAYED (post-hysteresis) status.
|
||||||
|
// DOWN_THRESHOLD defaults to 2, so it takes two consecutive failed
|
||||||
|
// probes to flip displayed down and open the outage; one up probe
|
||||||
|
// (UP_THRESHOLD=1) resolves it.
|
||||||
healthChecker._doRequest = jest.fn()
|
healthChecker._doRequest = jest.fn()
|
||||||
.mockResolvedValueOnce({ healthy: true, statusCode: 200, message: 'ok', details: {} })
|
.mockResolvedValueOnce({ healthy: true, statusCode: 200, message: 'ok', details: {} })
|
||||||
.mockResolvedValueOnce({ healthy: false, statusCode: 500, message: 'down', details: {} })
|
.mockResolvedValueOnce({ healthy: false, statusCode: 500, message: 'down', details: {} })
|
||||||
|
.mockResolvedValueOnce({ healthy: false, statusCode: 500, message: 'down', details: {} })
|
||||||
.mockResolvedValueOnce({ healthy: true, statusCode: 200, message: 'ok', details: {} });
|
.mockResolvedValueOnce({ healthy: true, statusCode: 200, message: 'ok', details: {} });
|
||||||
|
|
||||||
const config = { url: 'http://test.local' };
|
const config = { url: 'http://test.local' };
|
||||||
await healthChecker.checkService('svc1', config);
|
await healthChecker.checkService('svc1', config);
|
||||||
await healthChecker.checkService('svc1', config);
|
await healthChecker.checkService('svc1', config);
|
||||||
|
expect(healthChecker.incidents).toHaveLength(0); // one down alone: suppressed blip
|
||||||
|
|
||||||
|
await healthChecker.checkService('svc1', config); // second down flips displayed → open
|
||||||
expect(healthChecker.incidents).toHaveLength(1);
|
expect(healthChecker.incidents).toHaveLength(1);
|
||||||
expect(healthChecker.incidents[0]).toMatchObject({
|
expect(healthChecker.incidents[0]).toMatchObject({
|
||||||
serviceId: 'svc1',
|
serviceId: 'svc1',
|
||||||
@@ -220,7 +227,7 @@ describe('HealthChecker', () => {
|
|||||||
status: 'open'
|
status: 'open'
|
||||||
});
|
});
|
||||||
|
|
||||||
await healthChecker.checkService('svc1', config);
|
await healthChecker.checkService('svc1', config); // up resolves
|
||||||
expect(healthChecker.incidents[0].status).toBe('resolved');
|
expect(healthChecker.incidents[0].status).toBe('resolved');
|
||||||
expect(healthChecker.incidents[0].resolvedAt).toBeDefined();
|
expect(healthChecker.incidents[0].resolvedAt).toBeDefined();
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -199,8 +199,9 @@ class HealthChecker extends EventEmitter {
|
|||||||
}
|
}
|
||||||
|
|
||||||
const previousStatus = this.currentStatus.get(serviceId);
|
const previousStatus = this.currentStatus.get(serviceId);
|
||||||
|
const previousDisplayed = this.displayedStatus.get(serviceId);
|
||||||
this.recordStatus(serviceId, status);
|
this.recordStatus(serviceId, status);
|
||||||
this.checkForIncidents(serviceId, status, config, previousStatus);
|
this.checkForIncidents(serviceId, status, config, previousStatus, previousDisplayed);
|
||||||
|
|
||||||
return status;
|
return status;
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
@@ -223,8 +224,9 @@ class HealthChecker extends EventEmitter {
|
|||||||
this.consecutiveFailures.set(serviceId, (this.consecutiveFailures.get(serviceId) || 0) + 1);
|
this.consecutiveFailures.set(serviceId, (this.consecutiveFailures.get(serviceId) || 0) + 1);
|
||||||
|
|
||||||
const previousStatus = this.currentStatus.get(serviceId);
|
const previousStatus = this.currentStatus.get(serviceId);
|
||||||
|
const previousDisplayed = this.displayedStatus.get(serviceId);
|
||||||
this.recordStatus(serviceId, status);
|
this.recordStatus(serviceId, status);
|
||||||
this.checkForIncidents(serviceId, status, config, previousStatus);
|
this.checkForIncidents(serviceId, status, config, previousStatus, previousDisplayed);
|
||||||
|
|
||||||
return status;
|
return status;
|
||||||
}
|
}
|
||||||
@@ -453,10 +455,27 @@ class HealthChecker extends EventEmitter {
|
|||||||
/**
|
/**
|
||||||
* Check for incidents (downtime, slow response, etc.)
|
* Check for incidents (downtime, slow response, etc.)
|
||||||
*/
|
*/
|
||||||
checkForIncidents(serviceId, status, config, previous = this.currentStatus.get(serviceId)) {
|
checkForIncidents(serviceId, status, config, previous = this.currentStatus.get(serviceId), previousDisplayed = null) {
|
||||||
|
|
||||||
// Check for status change (up -> down or down -> up)
|
// DC-090: outage incidents follow the DISPLAYED (post-hysteresis) status —
|
||||||
if (previous && previous.status !== status.status) {
|
// the same signal that flips the dashboard badge. A single raw "down"
|
||||||
|
// blip that hysteresis suppresses must not open a critical outage
|
||||||
|
// incident (and a suppressed blip must not resolve a real one). When the
|
||||||
|
// caller supplies the pre-probe displayed state (checkService always
|
||||||
|
// does), transitions are evaluated displayed-vs-displayed using the
|
||||||
|
// post-recordStatus state in this.displayedStatus. Direct callers with
|
||||||
|
// no hysteresis state (previousDisplayed === null) keep the legacy
|
||||||
|
// raw-probe transition semantics.
|
||||||
|
if (previousDisplayed) {
|
||||||
|
const displayed = this.displayedStatus.get(serviceId);
|
||||||
|
if (displayed && displayed.status !== previousDisplayed.status) {
|
||||||
|
if (displayed.status === 'down') {
|
||||||
|
this.createIncident(serviceId, 'outage', 'Service is down', displayed);
|
||||||
|
} else if (displayed.status === 'up') {
|
||||||
|
this.resolveIncident(serviceId, 'outage', displayed);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else if (previous && previous.status !== status.status) {
|
||||||
if (status.status === 'down') {
|
if (status.status === 'down') {
|
||||||
this.createIncident(serviceId, 'outage', 'Service is down', status);
|
this.createIncident(serviceId, 'outage', 'Service is down', status);
|
||||||
} else if (status.status === 'up') {
|
} else if (status.status === 'up') {
|
||||||
|
|||||||
Reference in New Issue
Block a user