#!/usr/bin/env python3
"""van-ap-watchdog — keep the VanLink AP beaconing.

The AP wlan (rtw89, USB) shares a USB hub with the Starlink ethernet (r8152, USB).
When Starlink's link flaps the hub can re-enumerate and take the Wi-Fi radio down
with it. The hostapd.service drop-in (restart.conf) handles the common case — hostapd
exits and systemd restarts it (Restart=always, no start limit) until the interface
returns. This daemon is the backstop for the case systemd *can't* see: hostapd stays
running but the radio has wedged and stopped serving (dmesg "timed out to flush queues").

Every `interval` seconds it checks two things: hostapd's self-reported state via the
control socket (hostapd_cli status -> state=ENABLED), AND the kernel's ground truth for
the netdev (operstate up + still a port of the bridge). Both matter because they fail
independently: hostapd_cli keeps answering state=ENABLED off stale in-memory state after
the USB radio is torn down and re-enumerated underneath a still-running hostapd — the
netdev is recreated DOWN and dropped from the bridge, but hostapd never noticed and never
exited, so Restart=always never fired. The link check catches exactly that. If the AP is
unhealthy for `fail_threshold` checks in a row, it clears any failed state and restarts
hostapd. If the interface is simply gone (mid re-enumeration) it waits — there is
nothing to restart onto, and Restart=always reclaims it when it reappears. Stdlib only.
"""

import re
import subprocess
import sys
import time
from pathlib import Path

HOSTAPD_CONF = "/etc/hostapd/hostapd.conf"
CTRL_DIR = "/var/run/hostapd"
INTERVAL = 15          # seconds between health checks (backstop cadence, not first response)
FAIL_THRESHOLD = 2     # consecutive bad checks before restarting (debounces re-enum blips)


def log(msg, level="info"):
    # systemd journal severity prefixes (sd-daemon), same convention as van-thermal.
    pri = {"info": "<6>", "warn": "<4>", "crit": "<2>"}.get(level, "<6>")
    print(pri + msg, flush=True)


def _conf_value(key):
    """Read a `key=value` from hostapd.conf so there's one source of truth."""
    try:
        for line in Path(HOSTAPD_CONF).read_text().splitlines():
            m = re.match(rf"\s*{key}=(\S+)", line)
            if m:
                return m.group(1)
    except OSError as e:
        log(f"cannot read {HOSTAPD_CONF} ({e})", "crit")
    return None


def ap_ifname():
    """The AP interface name."""
    return _conf_value("interface")


def ap_bridge():
    """The bridge the AP netdev is enslaved to, or None if hostapd isn't bridging."""
    return _conf_value("bridge")


def iface_present(ifname):
    return Path(f"/sys/class/net/{ifname}").exists()


def link_healthy(ifname, bridge):
    """True when the netdev is actually carrying traffic: operationally up and, if
    hostapd bridges the AP, still a port of that bridge. This is the ground truth
    hostapd_cli can't see — after a USB re-enumeration the radio comes back as a fresh
    DOWN netdev outside the bridge while a stale hostapd still reports state=ENABLED."""
    try:
        operstate = Path(f"/sys/class/net/{ifname}/operstate").read_text().strip()
    except OSError:
        return False
    if operstate != "up":
        return False
    if bridge and not Path(f"/sys/class/net/{bridge}/brif/{ifname}").exists():
        return False
    return True


def ap_enabled(ifname):
    """True if hostapd reports the AP as beaconing (state=ENABLED). False if it's
    running but not enabled; None if the control socket is unreachable (hostapd down)."""
    try:
        out = subprocess.run(
            ["hostapd_cli", "-p", CTRL_DIR, "-i", ifname, "status"],
            capture_output=True, text=True, timeout=10,
        ).stdout
    except (OSError, subprocess.SubprocessError) as e:
        log(f"hostapd_cli failed: {e}", "warn")
        return None
    for line in out.splitlines():
        if line.startswith("state="):
            return line.strip() == "state=ENABLED"
    return None   # no state line -> couldn't talk to hostapd


def recover(ifname):
    log(f"AP {ifname} not beaconing — clearing failed state and restarting hostapd", "warn")
    # reset-failed first in case some other path parked the unit in a failed state;
    # harmless when it isn't.
    subprocess.run(["systemctl", "reset-failed", "hostapd"],
                   capture_output=True, text=True)
    r = subprocess.run(["systemctl", "restart", "hostapd"],
                       capture_output=True, text=True)
    if r.returncode != 0:
        log(f"hostapd restart returned {r.returncode}: {r.stderr.strip()}", "warn")


def main():
    ifname = ap_ifname()
    if not ifname:
        log("no interface= in hostapd.conf; nothing to watch", "crit")
        sys.exit(1)
    bridge = ap_bridge()
    log(f"van-ap-watchdog up: watching {ifname}"
        f"{f' on {bridge}' if bridge else ''} every {INTERVAL}s "
        f"(restart after {FAIL_THRESHOLD} bad checks)")

    bad = 0
    waiting = False   # latch so "interface absent" logs once, not every tick
    while True:
        if not iface_present(ifname):
            if not waiting:
                log(f"{ifname} absent — USB re-enumeration in progress; "
                    f"waiting for it to return (hostapd Restart=always will reclaim it)", "warn")
                waiting = True
            bad = 0
        else:
            waiting = False
            if ap_enabled(ifname) and link_healthy(ifname, bridge):
                if bad:
                    log(f"AP {ifname} beaconing again")
                bad = 0
            else:
                bad += 1
                if bad >= FAIL_THRESHOLD:
                    recover(ifname)
                    bad = 0
        time.sleep(INTERVAL)


if __name__ == "__main__":
    try:
        main()
    except KeyboardInterrupt:
        sys.exit(0)
