enodia-sentinal/enodia_sentinel/daemon.py
Luna 34bc09041a Add tamper-evidence: package-DB integrity, self-integrity, dead-man's switch
Answers two hard questions: "what guards the hashes FIM trusts?" and "how do we
make the sensor itself hard to silently disable?" — within the honest limit that
a root attacker who shares your privileges can't be fully stopped on-box, only
made loud.

- pkgdb.py: guards the package DB that `pacman -Qkk` trusts. Anchors a
  fingerprint of /var/lib/pacman/local (refreshed ONLY by the pacman hook) and
  cross-checks pacman.log; a DB change with no logged transaction is flagged
  pkgdb_tamper (CRITICAL, sid 100021) — catches an attacker rewriting a stored
  checksum to mask a modified binary. Verified end-to-end against a simulated
  hash-overwrite.
- selfprotect.py: Sentinel's own binaries/config/units/hook are always in the
  FIM watch set (self-integrity); a heartbeat is written each loop; an external
  `watchdog` command polls a remote dashboard and pushes if the sensor goes
  silent or unreachable — silence becomes the alarm.
- daemon: heartbeat + slow-cadence pkgdb check; DB anchor re-anchored in
  build_fim_baseline so the pacman hook refreshes it after each transaction
- web: /api/status now reports heartbeat_age/stale; dashboard shows it
- cli: pkgdb-check, watchdog (--url/--token/--max-age, bypasses min-severity)
- config: pkgdb_verify/_interval, heartbeat_max_age
- README: threat model (trust-anchor problem) + hardening layers (immutability,
  signed-package + external anchor); reframes "anti-rootkit" as tamper-evidence
- tests: +8 (DB fingerprint, anchor/transaction logic, pacman.log parse,
  heartbeat + watchdog verdicts). 80/80 pass

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-05-31 22:27:50 -07:00

263 lines
11 KiB
Python

# SPDX-License-Identifier: GPL-3.0-or-later
"""The detection daemon: sweep loop, cooldown dedup, baseline management.
Unlike the bash prototype, all loop state (cooldowns, last-scan timestamps,
baselines) lives in this object — no subshell-state surprises — and the
expensive filesystem-wide SUID scan is gated to its own slow cadence.
"""
from __future__ import annotations
import json
import os
import threading
import time
from pathlib import Path
from . import detectors, snapshot
from .alert import Alert
from .config import Config
from .system import SystemState, scan_suid_binaries
class Sentinel:
def __init__(self, cfg: Config) -> None:
self.cfg = cfg
self.start_time = time.time()
self.cooldowns: dict[str, float] = {}
# Cooldowns are touched by both the sweep loop and the eBPF event
# thread, so guard them.
self._cooldown_lock = threading.Lock()
self._exec_monitor = None
self.last_persist_scan = self.start_time
self.listener_baseline: set[str] = set()
self.suid_baseline: set[str] = set()
# SUID scan runs off the loop thread; the loop reads the latest result.
self._suid_current: list[str] | None = None
self._suid_thread: threading.Thread | None = None
self._last_suid_scan = 0.0
self._last_suid_baseline_refresh = self.start_time
# FIM: baseline + background scan state
self.fim_baseline: dict = {}
self._fim_thread: threading.Thread | None = None
self._last_fim_scan = 0.0
self._fim_pkg_thread: threading.Thread | None = None
self._last_fim_pkg = 0.0
self._last_pkgdb = 0.0
self._stop = threading.Event()
# -- baselines ---------------------------------------------------------
def build_baselines(self) -> None:
self.listener_baseline = SystemState().listener_keys()
self.suid_baseline = set(scan_suid_binaries(
extra_dirs=self.cfg.suid_scan_extra_dirs))
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
self._save(self.cfg.listener_baseline, sorted(self.listener_baseline))
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
def load_baselines(self) -> None:
self.listener_baseline = set(self._load(self.cfg.listener_baseline))
self.suid_baseline = set(self._load(self.cfg.suid_baseline))
@staticmethod
def _save(path: Path, data: list[str]) -> None:
path.write_text(json.dumps(data))
@staticmethod
def _load(path: Path) -> list[str]:
try:
return json.loads(path.read_text())
except (OSError, ValueError):
return []
# -- SUID scan (off the loop thread) -----------------------------------
def _maybe_scan_suid(self, now: float) -> None:
if self._suid_thread and self._suid_thread.is_alive():
return
if (now - self._last_suid_scan) < self.cfg.suid_scan_interval:
return
self._last_suid_scan = now
self._suid_thread = threading.Thread(target=self._scan_suid, daemon=True)
self._suid_thread.start()
def _scan_suid(self) -> None:
result = scan_suid_binaries(extra_dirs=self.cfg.suid_scan_extra_dirs)
self._suid_current = result
# Periodically fold the current state into the baseline so legitimately
# installed SUID binaries stop alerting after a while.
if (time.time() - self._last_suid_baseline_refresh) >= self.cfg.suid_refresh:
self.suid_baseline = set(result)
self._last_suid_baseline_refresh = time.time()
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
# -- FIM (off the loop thread) -----------------------------------------
def build_fim_baseline(self) -> int:
from . import pkgdb
from .fim import scan_paths
self.fim_baseline = scan_paths(self.cfg.fim_path_list())
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
self.cfg.fim_baseline.write_text(json.dumps(self.fim_baseline))
# Re-anchor the package DB here too, so the pacman hook (which calls
# fim-update) refreshes both after every legitimate transaction.
pkgdb.save_anchor(self.cfg)
return len(self.fim_baseline)
def _maybe_check_pkgdb(self, now: float) -> None:
if not self.cfg.pkgdb_verify:
return
if (now - self._last_pkgdb) < self.cfg.pkgdb_interval:
return
self._last_pkgdb = now
threading.Thread(target=self._check_pkgdb, daemon=True).start()
def _check_pkgdb(self) -> None:
from . import pkgdb
alert = pkgdb.check(self.cfg)
if alert:
self._on_exec_alert(alert)
def load_fim_baseline(self) -> None:
try:
self.fim_baseline = json.loads(self.cfg.fim_baseline.read_text())
except (OSError, ValueError):
self.fim_baseline = {}
if not self.fim_baseline:
self.build_fim_baseline()
def _maybe_scan_fim(self, now: float) -> None:
if not self.cfg.fim_enabled:
return
if self._fim_thread and self._fim_thread.is_alive():
return
if (now - self._last_fim_scan) < self.cfg.fim_scan_interval:
return
self._last_fim_scan = now
self._fim_thread = threading.Thread(target=self._scan_fim, daemon=True)
self._fim_thread.start()
def _scan_fim(self) -> None:
from .fim import diff, diff_alerts, scan_paths
current = scan_paths(self.cfg.fim_path_list())
# NB: the baseline is NOT updated here — only `fim-update` / the pacman
# hook refreshes it, so a change stays flagged until acknowledged.
for alert in diff_alerts(diff(self.fim_baseline, current)):
self._on_exec_alert(alert)
def _maybe_pkg_verify(self, now: float) -> None:
if not self.cfg.fim_pkg_verify:
return
if self._fim_pkg_thread and self._fim_pkg_thread.is_alive():
return
if (now - self._last_fim_pkg) < self.cfg.fim_pkg_verify_interval:
return
self._last_fim_pkg = now
self._fim_pkg_thread = threading.Thread(target=self._pkg_verify, daemon=True)
self._fim_pkg_thread.start()
def _pkg_verify(self) -> None:
from .fim import pacman_verify_alerts
for alert in pacman_verify_alerts():
self._on_exec_alert(alert)
# -- one sweep ---------------------------------------------------------
def sweep(self, *, force_suid: bool = False) -> list[Alert]:
now = time.time()
armed = (now - self.start_time) >= self.cfg.baseline_grace
if force_suid:
suid_binaries = scan_suid_binaries(
extra_dirs=self.cfg.suid_scan_extra_dirs)
else:
suid_binaries = self._suid_current # latest async result (may be None)
state = SystemState(
listener_baseline=self.listener_baseline if armed or force_suid else None,
suid_baseline=self.suid_baseline,
suid_binaries=suid_binaries if armed or force_suid else None,
persist_since=self.last_persist_scan if armed or force_suid else None,
)
alerts = list(detectors.run_all(state, self.cfg))
self.last_persist_scan = now
return alerts
def fresh_alerts(self, alerts: list[Alert], now: float) -> list[Alert]:
"""Drop alerts whose dedup key is still within cooldown (thread-safe)."""
out = []
with self._cooldown_lock:
for a in alerts:
prev = self.cooldowns.get(a.key, 0.0)
if (now - prev) >= self.cfg.cooldown:
self.cooldowns[a.key] = now
out.append(a)
return out
def _on_exec_alert(self, alert: Alert) -> None:
"""Callback for the eBPF exec monitor — same dedup + capture path."""
fresh = self.fresh_alerts([alert], time.time())
if fresh:
threading.Thread(
target=self._capture, args=(fresh,), daemon=True
).start()
# -- main loop ---------------------------------------------------------
def run(self) -> None:
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
try:
(self.cfg.log_dir / "sentinel.pid").write_text(str(os.getpid()))
except OSError:
pass
with open(self.cfg.events_log, "a") as fh:
fh.write(f"{time.strftime('%FT%T%z')} enodia-sentinel started\n")
self.build_baselines()
self.load_fim_baseline()
snapshot.prune(self.cfg)
self._start_exec_monitor()
sweeps = 0
while not self._stop.is_set():
now = time.time()
from .selfprotect import write_heartbeat
write_heartbeat(self.cfg) # dead-man's switch
if (now - self.start_time) >= self.cfg.baseline_grace:
self._maybe_scan_suid(now)
self._maybe_scan_fim(now)
self._maybe_pkg_verify(now)
self._maybe_check_pkgdb(now)
alerts = self.sweep()
fresh = self.fresh_alerts(alerts, now)
if fresh:
# capture off the loop thread so a slow snapshot never stalls
# detection; the SystemState used for forensics is rebuilt fresh
# inside the thread for accuracy.
threading.Thread(
target=self._capture, args=(fresh,), daemon=True
).start()
sweeps += 1
if sweeps % 20 == 0:
snapshot.prune(self.cfg)
self._stop.wait(self.cfg.sample_interval)
def _start_exec_monitor(self) -> None:
if not self.cfg.ebpf_exec_monitor:
with open(self.cfg.events_log, "a") as fh:
fh.write(f"{time.strftime('%FT%T%z')} "
"eBPF exec monitor: off (disabled in config)\n")
return
from .events.monitor import ExecMonitor
self._exec_monitor = ExecMonitor(self.cfg, self._on_exec_alert)
ok, reason = self._exec_monitor.start()
with open(self.cfg.events_log, "a") as fh:
status = "enabled" if ok else f"disabled ({reason})"
fh.write(f"{time.strftime('%FT%T%z')} eBPF exec monitor: {status}\n")
if not ok:
self._exec_monitor = None
def _capture(self, alerts: list[Alert]) -> None:
try:
snapshot.capture(alerts, SystemState(), self.cfg)
except Exception as exc: # never let a capture crash the daemon
with open(self.cfg.events_log, "a") as fh:
fh.write(f"{time.strftime('%FT%T%z')} capture error: {exc!r}\n")
def stop(self, *_a) -> None:
self._stop.set()
if self._exec_monitor is not None:
self._exec_monitor.stop()