"""Tests for flap detection (hbd.server.flap) and its hook in notify.send_notification.""" import asyncio import json import pytest from hbd.server import flap, hbdclass, notify, users as users_mod @pytest.fixture(autouse=True) def clean_flap_state(): """Every test starts with an empty state and a 3-in-10-minutes config.""" flap._state.clear() flap.setup({"flap_count": 3, "flap_interval": 10}) yield flap._state.clear() flap.setup({}) def alerts(n, host="h1", service="cpu", level="CRITICAL"): return [flap.observe(host, service, level) for _ in range(n)] def advance(seconds, host="h1", service="cpu"): """Backdate a key's recorded timestamps to simulate the clock moving on.""" st = flap._state[(host, service)] st["events"] = [t - seconds for t in st["events"]] if st["ok_since"] is not None: st["ok_since"] -= seconds # --- tripping --------------------------------------------------------------- def test_alerts_up_to_flap_count_pass(): assert alerts(3) == [flap.PASS] * 3 assert flap.flapping_services("h1") == [] def test_exceeding_flap_count_trips_then_suppresses(): assert alerts(5) == [flap.PASS, flap.PASS, flap.PASS, flap.TRIP, flap.SUPPRESS] assert flap.flapping_services("h1") == ["cpu"] def test_warning_and_critical_both_count(): flap.observe("h1", "cpu", "WARNING") flap.observe("h1", "cpu", "CRITICAL") flap.observe("h1", "cpu", "WARNING") assert flap.observe("h1", "cpu", "CRITICAL") == flap.TRIP def test_alerts_outside_the_window_do_not_count(): alerts(3) advance(11 * 60) # older than flap_interval assert alerts(3) == [flap.PASS] * 3 assert flap.flapping_services("h1") == [] def test_recover_alone_never_trips(): assert [flap.observe("h1", "cpu", "RECOVER") for _ in range(5)] == [flap.PASS] * 5 # --- suppression while flapping -------------------------------------------- def test_recover_and_info_are_suppressed_while_flapping(): alerts(4) assert flap.observe("h1", "cpu", "RECOVER") == flap.SUPPRESS assert flap.observe("h1", "cpu", "INFO") == flap.SUPPRESS def test_info_passes_when_not_flapping(): assert flap.observe("h1", "cpu", "INFO") == flap.PASS # --- clearing --------------------------------------------------------------- def test_clears_silently_one_interval_after_recover(): alerts(4) flap.observe("h1", "cpu", "RECOVER") advance(10 * 60) assert flap.flapping_services("h1") == [] assert flap.observe("h1", "cpu", "CRITICAL") == flap.PASS def test_still_flapping_before_the_interval_elapses(): alerts(4) flap.observe("h1", "cpu", "RECOVER") advance(9 * 60) assert flap.flapping_services("h1") == ["cpu"] assert flap.observe("h1", "cpu", "CRITICAL") == flap.SUPPRESS def test_alert_during_the_quiet_window_keeps_it_flapping(): alerts(4) flap.observe("h1", "cpu", "RECOVER") advance(9 * 60) flap.observe("h1", "cpu", "CRITICAL") # resets the quiet window advance(2 * 60) assert flap.flapping_services("h1") == ["cpu"] # --- keying ----------------------------------------------------------------- def test_services_and_hosts_are_tracked_independently(): alerts(4, service="cpu") assert flap.observe("h1", "disk", "CRITICAL") == flap.PASS assert flap.observe("h2", "cpu", "CRITICAL") == flap.PASS assert flap.flapping_services("h1") == ["cpu"] assert flap.flapping_services("h2") == [] def test_host_level_events_use_the_empty_service_key(): alerts(4, service="") assert flap.flapping_services("h1") == [""] # --- disabled --------------------------------------------------------------- @pytest.mark.parametrize("cfg", [ {"flap_count": 0, "flap_interval": 10}, {"flap_count": 3, "flap_interval": 0}, {}, ]) def test_disabled_never_suppresses(cfg): flap.setup(cfg) assert alerts(20) == [flap.PASS] * 20 assert flap.flapping_services("h1") == [] # --- integration through the real dispatch path ----------------------------- NOTIFY_CFG = { "flap_count": 3, "flap_interval": 10, "notification_channels": {"ch1": {"type": "pushover", "token": "t", "user": "u"}}, "users": {"alice": {"notification_channels": ["ch1"]}}, } @pytest.fixture def delivered(monkeypatch): """Wire up a host, a user and a stub channel driver; collect what gets delivered.""" sent = [] monkeypatch.setattr(notify, "_config", dict(notify._config)) # restored on teardown notify.setup(NOTIFY_CFG) users_mod.load_users(NOTIFY_CFG) monkeypatch.setitem(notify._DRIVERS, "pushover", lambda cfg, n: sent.append(n.body) or True) host = hbdclass.Host("flaphost") host.watched = True host.owner = "alice" yield sent, host hbdclass.Host.hosts.pop("flaphost", None) users_mod.load_users({}) def _notify(level, service, body): n = notify.Notification(title=f"[{level}] flaphost", body=body, level=level, service=service) asyncio.run(notify.send_notification("flaphost", n)) def test_marker_on_the_tripping_notification_then_nothing(delivered): sent, host = delivered for i in range(5): _notify("CRITICAL", "cpu", f"cpu = 9{i}") _notify("RECOVER", "cpu", "cpu = 12") assert sent == ["cpu = 90", "cpu = 91", "cpu = 92", f"cpu = 93 {flap.FLAP_MARKER}"] assert host.stateinfo()["flapping"] == ["cpu"] def test_a_flapping_service_does_not_silence_another(delivered): sent, host = delivered for i in range(5): _notify("CRITICAL", "cpu", "cpu = 99") sent.clear() _notify("CRITICAL", "disk", "disk = 91") assert sent == ["disk = 91"] def test_notifications_resume_after_the_state_clears(delivered): sent, host = delivered for i in range(5): _notify("CRITICAL", "cpu", "cpu = 99") _notify("RECOVER", "cpu", "cpu = 12") advance(10 * 60, host="flaphost") assert host.stateinfo()["flapping"] == [] sent.clear() _notify("CRITICAL", "cpu", "cpu = 99") assert sent == ["cpu = 99"] def test_stateinfo_flapping_survives_json_encoding(delivered): sent, host = delivered for i in range(5): _notify("CRITICAL", "", "IPv4 overdue") # host-level, empty service key info = host.stateinfo() assert info["flapping"] == [""] assert json.loads(json.dumps(info))["flapping"] == [""]