diff --git a/README.md b/README.md
index 89163bc..cf93797 100644
--- a/README.md
+++ b/README.md
@@ -116,6 +116,11 @@ dyndomains:
# Threshold alert re-notification interval (seconds)
threshold_renotify_interval: 3600
+# Flap detection — silence a service/host after flap_count warning or critical
+# notifications within flap_interval minutes (flap_count: 0 disables)
+flap_count: 5
+flap_interval: 10
+
# Notification channels
notification_channels:
pushover_ops:
@@ -400,6 +405,8 @@ hosts:
Notifications are sent on state transitions (OK → WARNING, WARNING → CRITICAL, CRITICAL → OK). De-escalations (CRITICAL → WARNING) do not trigger a notification. Ongoing alerts generate a re-notification every `threshold_renotify_interval` seconds (default: 3600). Alerts can be acknowledged via the web UI or API to suppress re-notifications.
+A service or host that exceeds `flap_count` warning/critical notifications within `flap_interval` minutes is marked **flapping**: the tripping notification carries `Now flapping!! No more messages!` and further notifications are suppressed until it stays OK for `flap_interval` minutes. Flapping hosts are badged on the live dashboard; the event log keeps recording throughout. See [docs/NOTIFICATIONS.md](docs/NOTIFICATIONS.md#flap-detection).
+
### RTT thresholds
The server measures heartbeat round-trip time and supports RTT thresholds using the same format:
diff --git a/docs/NOTIFICATIONS.md b/docs/NOTIFICATIONS.md
index e35f9b4..effb259 100644
--- a/docs/NOTIFICATIONS.md
+++ b/docs/NOTIFICATIONS.md
@@ -9,6 +9,7 @@ Notifications are dispatched to the **owner and managers** of a host, each via t
```
Alert event (udp.py / threshold.py)
└─ notify.send_notification(host_name, Notification)
+ ├─ flap.observe(host, service, level) → pass | trip | suppress
├─ look up host.owner + host.managers
├─ for each user → user.notification_channels
└─ for each channel → _dispatch_to_channel (filtered by min_level)
@@ -19,6 +20,7 @@ Every notification carries:
- **body** — detail message (metric value, threshold, duration)
- **url** — link to the plugin metrics page (`{base_url}/plugins#{hostname}`)
- **level** — `RECOVER | WARNING | CRITICAL | INFO`
+- **service** — flap-detection key within the host (empty = the host itself)
## Configuration
@@ -268,6 +270,43 @@ min_level: WARNING
Reminder notifications (re-notify) are sent only for CRITICAL level alerts.
+## Flap detection
+
+A check that toggles between OK and alerting produces a notification per swing. Flap
+detection silences it after the first few.
+
+A **`(host, service)`** pair is flapping once it exceeds `flap_count` WARNING/CRITICAL
+notifications within `flap_interval` minutes. Threshold alerts key on their metric path,
+so a flapping disk check does not silence an unrelated CPU alert; connectivity, boot and
+shutdown events key on the host itself.
+
+```yaml
+flap_count: 5 # notifications within the window that trip flapping (0 disables)
+flap_interval: 10 # minutes — both the counting window and the quiet window
+```
+
+Lifecycle:
+
+| Event | Effect |
+|---|---|
+| Alerts 1..`flap_count` within the window | Delivered normally |
+| Alert `flap_count + 1` | Delivered with ` Now flapping!! No more messages!` appended to the body |
+| Every notification after that | Dropped — including RECOVER and INFO |
+| RECOVER while flapping | Dropped, and starts the `flap_interval` quiet window |
+| WARNING/CRITICAL during the quiet window | Restarts the quiet window; still flapping |
+| Quiet window elapses | Flapping ends **silently** — no notification |
+
+Only outbound notifications are suppressed. `notify.eventlog` keeps recording every event,
+so the journal and the `/log` page retain the full history of the flap.
+
+Flapping pairs appear in each host's `stateinfo()` under `flapping` (a list of service
+keys; `""` means the host itself) and render as an amber **flapping** badge next to the
+host name on the live dashboard, with the affected services in its tooltip.
+
+State lives at module level in `hbd/server/flap.py` and is never pickled — a server restart
+starts every check with a clean slate. Hosts with `watch: false` never reach the
+notification path, so they never flap.
+
## API reference
### `send_notification(host_name, notif) -> dict`
diff --git a/hbd/server/config.py b/hbd/server/config.py
index 335ad2f..22ee227 100644
--- a/hbd/server/config.py
+++ b/hbd/server/config.py
@@ -30,6 +30,10 @@ SERVER_DEFAULTS = {
"grace": 2, # Grace period (extra seconds before notifying after a missed heartbeat)
"threshold_renotify_interval": 3600, # Seconds between threshold re-notifications
+ # Flap detection (0 in either key disables it)
+ "flap_count": 5, # Warning/critical notifications within flap_interval that trip flapping
+ "flap_interval": 10, # Minutes: the counting window, and the quiet window after an OK
+
# User management
"users": {}, # username -> {full_name, avatar, password, admin, notification_channels}
"default_owner": None, # Username that owns hosts with no explicit owner
diff --git a/hbd/server/configio.py b/hbd/server/configio.py
index 6924c84..1793b4f 100644
--- a/hbd/server/configio.py
+++ b/hbd/server/configio.py
@@ -19,6 +19,7 @@ def _make_yaml() -> YAML:
_SERVER_KEYS = [
"hbd_port", "hbd_host", "ws_port", "wss_port", "hb_port",
"interval", "grace", "base_url", "threshold_renotify_interval",
+ "flap_count", "flap_interval",
"logfile", "pidfile", "pickfile", "journal_enabled", "journal_dir",
"journal_max_size", "journal_max_backups", "default_owner",
"default_threshold_config",
diff --git a/hbd/server/flap.py b/hbd/server/flap.py
new file mode 100644
index 0000000..e1eaa34
--- /dev/null
+++ b/hbd/server/flap.py
@@ -0,0 +1,119 @@
+"""Flap detection: silence checks that toggle faster than they are useful.
+
+A ``(host, service)`` pair is *flapping* once it exceeds ``flap_count``
+WARNING/CRITICAL notifications within ``flap_interval`` minutes. The
+notification that trips the threshold is delivered with ``FLAP_MARKER``
+appended; every notification for that pair afterwards is dropped. The state
+clears silently ``flap_interval`` minutes after a RECOVER, provided no further
+WARNING/CRITICAL arrived in the meantime.
+
+Only outbound notifications are affected — ``notify.eventlog`` keeps recording
+every event, so the journal and ``/log`` retain the full history of the flap.
+
+State lives here at module level rather than on ``Host`` so it is never
+pickled: a restart starts every check with a clean slate.
+"""
+
+import logging
+import time
+
+logger = logging.getLogger(__name__)
+
+FLAP_MARKER = "Now flapping!! No more messages!"
+
+# Actions returned by observe()
+PASS = "pass" # deliver unchanged
+TRIP = "trip" # deliver with FLAP_MARKER appended
+SUPPRESS = "suppress" # drop
+
+_count = 0 # flap_count: alerts within the window needed to trip
+_window = 0.0 # flap_interval in seconds
+
+# {(host, service): {"events": [ts, ...], "flapping": bool, "ok_since": float|None}}
+_state: dict = {}
+
+
+def setup(cfg) -> None:
+ """Read flap_count / flap_interval from *cfg* (also called on reload)."""
+ global _count, _window
+ _count = int(cfg.get("flap_count", 0) or 0)
+ _window = float(cfg.get("flap_interval", 0) or 0) * 60.0
+
+
+def _enabled() -> bool:
+ return _count > 0 and _window > 0
+
+
+def _label(key) -> str:
+ host, service = key
+ return f"{host}/{service}" if service else host
+
+
+def _sweep(st: dict, now: float) -> bool:
+ """Clear the flapping flag once the post-RECOVER quiet window has elapsed."""
+ if st["flapping"] and st["ok_since"] is not None and now - st["ok_since"] >= _window:
+ st["flapping"] = False
+ st["ok_since"] = None
+ st["events"].clear()
+ return True
+ return False
+
+
+def observe(host: str, service: str, level: str) -> str:
+ """Record a notification for *host*/*service* and return what to do with it.
+
+ Returns PASS, TRIP or SUPPRESS.
+ """
+ if not _enabled():
+ return PASS
+
+ now = time.time()
+ key = (host or "", service or "")
+ level = (level or "").upper()
+ st = _state.get(key)
+ if st is not None and _sweep(st, now):
+ logger.info("flapping cleared for %s", _label(key))
+
+ if level in ("WARNING", "CRITICAL"):
+ if st is None:
+ st = _state[key] = {"events": [], "flapping": False, "ok_since": None}
+ # An alert inside the quiet window means it never settled.
+ st["ok_since"] = None
+ st["events"] = [t for t in st["events"] if now - t < _window]
+ st["events"].append(now)
+ if st["flapping"]:
+ return SUPPRESS
+ if len(st["events"]) > _count:
+ st["flapping"] = True
+ logger.info(
+ "flapping detected for %s (%d alerts in %.0f min)",
+ _label(key), len(st["events"]), _window / 60,
+ )
+ return TRIP
+ return PASS
+
+ if st is None or not st["flapping"]:
+ return PASS
+ if level == "RECOVER":
+ # Starts the quiet window; the state ends silently when it elapses.
+ st["ok_since"] = now
+ return SUPPRESS
+
+
+def flapping_services(host: str) -> list:
+ """Return the services of *host* that are currently flapping.
+
+ An empty string in the result means the host itself (connectivity, boot,
+ shutdown) rather than a named service.
+ """
+ if not _enabled():
+ return []
+ now = time.time()
+ flapping = []
+ for (h, service), st in _state.items():
+ if h != host:
+ continue
+ _sweep(st, now)
+ if st["flapping"]:
+ flapping.append(service)
+ return sorted(flapping)
diff --git a/hbd/server/hbdclass.py b/hbd/server/hbdclass.py
index b5e190b..433ff1d 100644
--- a/hbd/server/hbdclass.py
+++ b/hbd/server/hbdclass.py
@@ -428,6 +428,10 @@ class Host:
ddict["alert_critical_unacked"] = critical_unacked
ddict["alert_critical_acked"] = critical_acked
+ # Flap detection state (module-level in flap.py, never pickled)
+ from . import flap
+ ddict["flapping"] = flap.flapping_services(self.name)
+
# User access
ddict["owner"] = getattr(self, "owner", None)
ddict["managers"] = list(getattr(self, "managers", []))
diff --git a/hbd/server/notify.py b/hbd/server/notify.py
index c5b9f27..b643fc6 100644
--- a/hbd/server/notify.py
+++ b/hbd/server/notify.py
@@ -25,6 +25,7 @@ from dataclasses import dataclass, field
from typing import Optional
from . import data
+from . import flap as flap_mod
from . import ws as ws_mod
logger = logging.getLogger(__name__)
@@ -63,6 +64,7 @@ class Notification:
body: str # detail message
level: str # RECOVER | WARNING | CRITICAL | INFO
url: str = "" # link to plugin metrics page
+ service: str = "" # flap-detection key within the host ("" = the host itself)
# ---------------------------------------------------------------------------
@@ -73,6 +75,7 @@ def setup(cfg: dict, loop: Optional[asyncio.AbstractEventLoop] = None):
"""Initialize notifier from configuration dict."""
global _config, _loop
_config = dict(cfg)
+ flap_mod.setup(_config)
if loop is not None:
_loop = loop
@@ -81,6 +84,7 @@ def reload_config(cfg: dict):
"""Reload notification configuration on SIGHUP."""
global _config
_config = dict(cfg)
+ flap_mod.setup(_config)
logger.info("Notification configuration reloaded")
@@ -439,11 +443,22 @@ async def send_notification(host_name: str, notif: Notification) -> dict:
notification_channels, and dispatches. Silently does nothing if
no users are configured.
+ Flap detection runs first: once *host_name*/*notif.service* is flapping the
+ notification is dropped, and the one that trips the state carries
+ ``flap.FLAP_MARKER``.
+
Returns a dict of {channel_name: bool} results.
"""
from . import users as users_mod
from . import hbdclass
+ action = flap_mod.observe(host_name, notif.service, notif.level)
+ if action == flap_mod.SUPPRESS:
+ logger.debug("flapping: suppressed %s notification for %s", notif.level, host_name)
+ return {}
+ if action == flap_mod.TRIP:
+ notif.body = f"{notif.body} {flap_mod.FLAP_MARKER}"
+
if not users_mod.users_enabled():
return {}
diff --git a/hbd/server/settings.py b/hbd/server/settings.py
index 30b5df3..b969201 100644
--- a/hbd/server/settings.py
+++ b/hbd/server/settings.py
@@ -386,6 +386,12 @@ def get_settings_sections(config: dict, threshold_checker=None, user=None) -> li
"Extra seconds to wait after a missed heartbeat before sending notifications.", editable=True),
field("threshold_renotify_interval", "Re-notify interval", "duration",
"How often to re-send notifications for ongoing threshold alerts.", editable=True),
+ field("flap_count", "Flap count", "number",
+ "Warning/critical notifications within the flap interval that mark a "
+ "service as flapping and silence it. 0 disables flap detection.", editable=True),
+ field("flap_interval", "Flap interval", "number",
+ "Minutes: the window flap count is measured over, and how long a service "
+ "must stay OK before flapping ends.", editable=True),
field("autosave_interval", "Autosave interval", "duration",
"How often the server saves its state to disk."),
field("base_url", "Base URL", "text",
diff --git a/hbd/server/templates/live.html b/hbd/server/templates/live.html
index 5e11c0c..8d5474d 100644
--- a/hbd/server/templates/live.html
+++ b/hbd/server/templates/live.html
@@ -45,6 +45,7 @@
.dot.down { background: var(--st-crit); box-shadow: 0 0 0 3px var(--st-crit-soft); }
.dot.overdue { background: transparent; border: 2px solid var(--st-warn); }
.stale { color: var(--st-faint); font-size: 10px; border: 1px solid var(--st-line); border-radius: 4px; padding: 1px 4px; font-family: var(--st-mono); white-space: nowrap; }
+ .flap { color: var(--st-warn); font-size: 10px; border: 1px solid var(--st-warn); border-radius: 4px; padding: 1px 4px; font-family: var(--st-mono); white-space: nowrap; }
.chip.outline { background: transparent; border: 1px solid var(--st-line); }
.lat { font-family: var(--st-mono); font-size: 12px; color: var(--st-muted); font-variant-numeric: tabular-nums; }
@@ -247,11 +248,17 @@
staleTag = 'v' + escHtml(String(data.hbc_version)) + '';
}
var dotCls = w ? (w.state === 'up' ? sev : (w.state === 'down' ? 'down' : 'overdue')) : 'off';
+ var flapTag = '';
+ var flapping = data.flapping || [];
+ if (flapping.length) {
+ var svcs = flapping.map(function (s) { return s || 'host'; }).join(', ');
+ flapTag = 'flapping';
+ }
row.innerHTML =
''
+ ''
- + '' + escHtml(name) + '' + staleTag + ''
+ + '' + escHtml(name) + '' + staleTag + flapTag + ''
+ '' + statusChip(data) + tipCard(data) + ''
+ '' + lat + ''
+ '' + alertChips(data) + ''
diff --git a/hbd/server/threshold.py b/hbd/server/threshold.py
index 07813d7..bc8ca64 100644
--- a/hbd/server/threshold.py
+++ b/hbd/server/threshold.py
@@ -1251,6 +1251,7 @@ class ThresholdChecker:
title=title,
body=body,
level=lvl,
+ service=short_path,
),
))
@@ -1534,6 +1535,7 @@ class ThresholdChecker:
title=f"[REMINDER/{alert_state.level.name}] {host_name} {short_path}",
body=body,
level=alert_state.level.name,
+ service=short_path,
),
))
logger.info("Re-notification sent: %s", message)
diff --git a/tests/test_flap.py b/tests/test_flap.py
new file mode 100644
index 0000000..72e6eca
--- /dev/null
+++ b/tests/test_flap.py
@@ -0,0 +1,197 @@
+"""Tests for flap detection (hbd.server.flap) and its hook in notify.send_notification."""
+import asyncio
+import json
+
+import pytest
+
+from hbd.server import flap, hbdclass, notify, users as users_mod
+
+
+@pytest.fixture(autouse=True)
+def clean_flap_state():
+ """Every test starts with an empty state and a 3-in-10-minutes config."""
+ flap._state.clear()
+ flap.setup({"flap_count": 3, "flap_interval": 10})
+ yield
+ flap._state.clear()
+ flap.setup({})
+
+
+def alerts(n, host="h1", service="cpu", level="CRITICAL"):
+ return [flap.observe(host, service, level) for _ in range(n)]
+
+
+def advance(seconds, host="h1", service="cpu"):
+ """Backdate a key's recorded timestamps to simulate the clock moving on."""
+ st = flap._state[(host, service)]
+ st["events"] = [t - seconds for t in st["events"]]
+ if st["ok_since"] is not None:
+ st["ok_since"] -= seconds
+
+
+# --- tripping ---------------------------------------------------------------
+
+def test_alerts_up_to_flap_count_pass():
+ assert alerts(3) == [flap.PASS] * 3
+ assert flap.flapping_services("h1") == []
+
+
+def test_exceeding_flap_count_trips_then_suppresses():
+ assert alerts(5) == [flap.PASS, flap.PASS, flap.PASS, flap.TRIP, flap.SUPPRESS]
+ assert flap.flapping_services("h1") == ["cpu"]
+
+
+def test_warning_and_critical_both_count():
+ flap.observe("h1", "cpu", "WARNING")
+ flap.observe("h1", "cpu", "CRITICAL")
+ flap.observe("h1", "cpu", "WARNING")
+ assert flap.observe("h1", "cpu", "CRITICAL") == flap.TRIP
+
+
+def test_alerts_outside_the_window_do_not_count():
+ alerts(3)
+ advance(11 * 60) # older than flap_interval
+ assert alerts(3) == [flap.PASS] * 3
+ assert flap.flapping_services("h1") == []
+
+
+def test_recover_alone_never_trips():
+ assert [flap.observe("h1", "cpu", "RECOVER") for _ in range(5)] == [flap.PASS] * 5
+
+
+# --- suppression while flapping --------------------------------------------
+
+def test_recover_and_info_are_suppressed_while_flapping():
+ alerts(4)
+ assert flap.observe("h1", "cpu", "RECOVER") == flap.SUPPRESS
+ assert flap.observe("h1", "cpu", "INFO") == flap.SUPPRESS
+
+
+def test_info_passes_when_not_flapping():
+ assert flap.observe("h1", "cpu", "INFO") == flap.PASS
+
+
+# --- clearing ---------------------------------------------------------------
+
+def test_clears_silently_one_interval_after_recover():
+ alerts(4)
+ flap.observe("h1", "cpu", "RECOVER")
+ advance(10 * 60)
+ assert flap.flapping_services("h1") == []
+ assert flap.observe("h1", "cpu", "CRITICAL") == flap.PASS
+
+
+def test_still_flapping_before_the_interval_elapses():
+ alerts(4)
+ flap.observe("h1", "cpu", "RECOVER")
+ advance(9 * 60)
+ assert flap.flapping_services("h1") == ["cpu"]
+ assert flap.observe("h1", "cpu", "CRITICAL") == flap.SUPPRESS
+
+
+def test_alert_during_the_quiet_window_keeps_it_flapping():
+ alerts(4)
+ flap.observe("h1", "cpu", "RECOVER")
+ advance(9 * 60)
+ flap.observe("h1", "cpu", "CRITICAL") # resets the quiet window
+ advance(2 * 60)
+ assert flap.flapping_services("h1") == ["cpu"]
+
+
+# --- keying -----------------------------------------------------------------
+
+def test_services_and_hosts_are_tracked_independently():
+ alerts(4, service="cpu")
+ assert flap.observe("h1", "disk", "CRITICAL") == flap.PASS
+ assert flap.observe("h2", "cpu", "CRITICAL") == flap.PASS
+ assert flap.flapping_services("h1") == ["cpu"]
+ assert flap.flapping_services("h2") == []
+
+
+def test_host_level_events_use_the_empty_service_key():
+ alerts(4, service="")
+ assert flap.flapping_services("h1") == [""]
+
+
+# --- disabled ---------------------------------------------------------------
+
+@pytest.mark.parametrize("cfg", [
+ {"flap_count": 0, "flap_interval": 10},
+ {"flap_count": 3, "flap_interval": 0},
+ {},
+])
+def test_disabled_never_suppresses(cfg):
+ flap.setup(cfg)
+ assert alerts(20) == [flap.PASS] * 20
+ assert flap.flapping_services("h1") == []
+
+
+# --- integration through the real dispatch path -----------------------------
+
+NOTIFY_CFG = {
+ "flap_count": 3,
+ "flap_interval": 10,
+ "notification_channels": {"ch1": {"type": "pushover", "token": "t", "user": "u"}},
+ "users": {"alice": {"notification_channels": ["ch1"]}},
+}
+
+
+@pytest.fixture
+def delivered(monkeypatch):
+ """Wire up a host, a user and a stub channel driver; collect what gets delivered."""
+ sent = []
+ monkeypatch.setattr(notify, "_config", dict(notify._config)) # restored on teardown
+ notify.setup(NOTIFY_CFG)
+ users_mod.load_users(NOTIFY_CFG)
+ monkeypatch.setitem(notify._DRIVERS, "pushover", lambda cfg, n: sent.append(n.body) or True)
+ host = hbdclass.Host("flaphost")
+ host.watched = True
+ host.owner = "alice"
+ yield sent, host
+ hbdclass.Host.hosts.pop("flaphost", None)
+ users_mod.load_users({})
+
+
+def _notify(level, service, body):
+ n = notify.Notification(title=f"[{level}] flaphost", body=body, level=level, service=service)
+ asyncio.run(notify.send_notification("flaphost", n))
+
+
+def test_marker_on_the_tripping_notification_then_nothing(delivered):
+ sent, host = delivered
+ for i in range(5):
+ _notify("CRITICAL", "cpu", f"cpu = 9{i}")
+ _notify("RECOVER", "cpu", "cpu = 12")
+
+ assert sent == ["cpu = 90", "cpu = 91", "cpu = 92", f"cpu = 93 {flap.FLAP_MARKER}"]
+ assert host.stateinfo()["flapping"] == ["cpu"]
+
+
+def test_a_flapping_service_does_not_silence_another(delivered):
+ sent, host = delivered
+ for i in range(5):
+ _notify("CRITICAL", "cpu", "cpu = 99")
+ sent.clear()
+ _notify("CRITICAL", "disk", "disk = 91")
+ assert sent == ["disk = 91"]
+
+
+def test_notifications_resume_after_the_state_clears(delivered):
+ sent, host = delivered
+ for i in range(5):
+ _notify("CRITICAL", "cpu", "cpu = 99")
+ _notify("RECOVER", "cpu", "cpu = 12")
+ advance(10 * 60, host="flaphost")
+ assert host.stateinfo()["flapping"] == []
+ sent.clear()
+ _notify("CRITICAL", "cpu", "cpu = 99")
+ assert sent == ["cpu = 99"]
+
+
+def test_stateinfo_flapping_survives_json_encoding(delivered):
+ sent, host = delivered
+ for i in range(5):
+ _notify("CRITICAL", "", "IPv4 overdue") # host-level, empty service key
+ info = host.stateinfo()
+ assert info["flapping"] == [""]
+ assert json.loads(json.dumps(info))["flapping"] == [""]