Both zero-dependency (stdlib http.server + urllib), consistent with the project's no-dependency-tree stance for a security daemon. Web dashboard (enodia_sentinel/web.py + static/dashboard.html): - read-only JSON API over the log dir: /api/status, /api/alerts, /api/alerts/<id>, /api/events - bearer-token auth (constant-time; header or ?token=), required on non- loopback binds, auto-generated + persisted (0600) when unset - binds the host's Tailscale IP by default (auto-detected), reachable from the tailnet but not the LAN/internet - self-contained dark SPA: severity cards, live alert list, full snapshot viewer; 10s auto-refresh - path-traversal-safe alert lookup; `enodia-sentinel web` subcommand; daemon now writes a pidfile so the dashboard can show live status - hardened enodia-sentinel-web.service (read-only, no caps) Phone push (enodia_sentinel/notify/): - pluggable backends — ntfy, Pushover, generic webhook — each separating a pure build() (unit-tested, no network) from send() - a backend turns on when its config keys are set; pushes gated by notify_min_severity; severity → per-service priority/tags - fired from snapshot.capture on worker threads, errors swallowed - desktop notify-send retained Tests: +16 (9 web incl. a real-server 401/200 auth test, 7 notify request-build cases). 55/55 pass. Live end-to-end verified: daemon → alert → API → page. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
181 lines
7.4 KiB
Python
181 lines
7.4 KiB
Python
# SPDX-License-Identifier: GPL-3.0-or-later
|
|
"""The detection daemon: sweep loop, cooldown dedup, baseline management.
|
|
|
|
Unlike the bash prototype, all loop state (cooldowns, last-scan timestamps,
|
|
baselines) lives in this object — no subshell-state surprises — and the
|
|
expensive filesystem-wide SUID scan is gated to its own slow cadence.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import threading
|
|
import time
|
|
from pathlib import Path
|
|
|
|
from . import detectors, snapshot
|
|
from .alert import Alert
|
|
from .config import Config
|
|
from .system import SystemState, scan_suid_binaries
|
|
|
|
|
|
class Sentinel:
|
|
def __init__(self, cfg: Config) -> None:
|
|
self.cfg = cfg
|
|
self.start_time = time.time()
|
|
self.cooldowns: dict[str, float] = {}
|
|
# Cooldowns are touched by both the sweep loop and the eBPF event
|
|
# thread, so guard them.
|
|
self._cooldown_lock = threading.Lock()
|
|
self._exec_monitor = None
|
|
self.last_persist_scan = self.start_time
|
|
self.listener_baseline: set[str] = set()
|
|
self.suid_baseline: set[str] = set()
|
|
# SUID scan runs off the loop thread; the loop reads the latest result.
|
|
self._suid_current: list[str] | None = None
|
|
self._suid_thread: threading.Thread | None = None
|
|
self._last_suid_scan = 0.0
|
|
self._last_suid_baseline_refresh = self.start_time
|
|
self._stop = threading.Event()
|
|
|
|
# -- baselines ---------------------------------------------------------
|
|
def build_baselines(self) -> None:
|
|
self.listener_baseline = SystemState().listener_keys()
|
|
self.suid_baseline = set(scan_suid_binaries(
|
|
extra_dirs=self.cfg.suid_scan_extra_dirs))
|
|
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
|
|
self._save(self.cfg.listener_baseline, sorted(self.listener_baseline))
|
|
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
|
|
|
|
def load_baselines(self) -> None:
|
|
self.listener_baseline = set(self._load(self.cfg.listener_baseline))
|
|
self.suid_baseline = set(self._load(self.cfg.suid_baseline))
|
|
|
|
@staticmethod
|
|
def _save(path: Path, data: list[str]) -> None:
|
|
path.write_text(json.dumps(data))
|
|
|
|
@staticmethod
|
|
def _load(path: Path) -> list[str]:
|
|
try:
|
|
return json.loads(path.read_text())
|
|
except (OSError, ValueError):
|
|
return []
|
|
|
|
# -- SUID scan (off the loop thread) -----------------------------------
|
|
def _maybe_scan_suid(self, now: float) -> None:
|
|
if self._suid_thread and self._suid_thread.is_alive():
|
|
return
|
|
if (now - self._last_suid_scan) < self.cfg.suid_scan_interval:
|
|
return
|
|
self._last_suid_scan = now
|
|
self._suid_thread = threading.Thread(target=self._scan_suid, daemon=True)
|
|
self._suid_thread.start()
|
|
|
|
def _scan_suid(self) -> None:
|
|
result = scan_suid_binaries(extra_dirs=self.cfg.suid_scan_extra_dirs)
|
|
self._suid_current = result
|
|
# Periodically fold the current state into the baseline so legitimately
|
|
# installed SUID binaries stop alerting after a while.
|
|
if (time.time() - self._last_suid_baseline_refresh) >= self.cfg.suid_refresh:
|
|
self.suid_baseline = set(result)
|
|
self._last_suid_baseline_refresh = time.time()
|
|
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
|
|
|
|
# -- one sweep ---------------------------------------------------------
|
|
def sweep(self, *, force_suid: bool = False) -> list[Alert]:
|
|
now = time.time()
|
|
armed = (now - self.start_time) >= self.cfg.baseline_grace
|
|
|
|
if force_suid:
|
|
suid_binaries = scan_suid_binaries(
|
|
extra_dirs=self.cfg.suid_scan_extra_dirs)
|
|
else:
|
|
suid_binaries = self._suid_current # latest async result (may be None)
|
|
|
|
state = SystemState(
|
|
listener_baseline=self.listener_baseline if armed or force_suid else None,
|
|
suid_baseline=self.suid_baseline,
|
|
suid_binaries=suid_binaries if armed or force_suid else None,
|
|
persist_since=self.last_persist_scan if armed or force_suid else None,
|
|
)
|
|
alerts = list(detectors.run_all(state, self.cfg))
|
|
self.last_persist_scan = now
|
|
return alerts
|
|
|
|
def fresh_alerts(self, alerts: list[Alert], now: float) -> list[Alert]:
|
|
"""Drop alerts whose dedup key is still within cooldown (thread-safe)."""
|
|
out = []
|
|
with self._cooldown_lock:
|
|
for a in alerts:
|
|
prev = self.cooldowns.get(a.key, 0.0)
|
|
if (now - prev) >= self.cfg.cooldown:
|
|
self.cooldowns[a.key] = now
|
|
out.append(a)
|
|
return out
|
|
|
|
def _on_exec_alert(self, alert: Alert) -> None:
|
|
"""Callback for the eBPF exec monitor — same dedup + capture path."""
|
|
fresh = self.fresh_alerts([alert], time.time())
|
|
if fresh:
|
|
threading.Thread(
|
|
target=self._capture, args=(fresh,), daemon=True
|
|
).start()
|
|
|
|
# -- main loop ---------------------------------------------------------
|
|
def run(self) -> None:
|
|
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
(self.cfg.log_dir / "sentinel.pid").write_text(str(os.getpid()))
|
|
except OSError:
|
|
pass
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
fh.write(f"{time.strftime('%FT%T%z')} enodia-sentinel started\n")
|
|
self.build_baselines()
|
|
snapshot.prune(self.cfg)
|
|
self._start_exec_monitor()
|
|
sweeps = 0
|
|
while not self._stop.is_set():
|
|
now = time.time()
|
|
if (now - self.start_time) >= self.cfg.baseline_grace:
|
|
self._maybe_scan_suid(now)
|
|
alerts = self.sweep()
|
|
fresh = self.fresh_alerts(alerts, now)
|
|
if fresh:
|
|
# capture off the loop thread so a slow snapshot never stalls
|
|
# detection; the SystemState used for forensics is rebuilt fresh
|
|
# inside the thread for accuracy.
|
|
threading.Thread(
|
|
target=self._capture, args=(fresh,), daemon=True
|
|
).start()
|
|
sweeps += 1
|
|
if sweeps % 20 == 0:
|
|
snapshot.prune(self.cfg)
|
|
self._stop.wait(self.cfg.sample_interval)
|
|
|
|
def _start_exec_monitor(self) -> None:
|
|
if not self.cfg.ebpf_exec_monitor:
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
fh.write(f"{time.strftime('%FT%T%z')} "
|
|
"eBPF exec monitor: off (disabled in config)\n")
|
|
return
|
|
from .events.monitor import ExecMonitor
|
|
self._exec_monitor = ExecMonitor(self.cfg, self._on_exec_alert)
|
|
ok, reason = self._exec_monitor.start()
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
status = "enabled" if ok else f"disabled ({reason})"
|
|
fh.write(f"{time.strftime('%FT%T%z')} eBPF exec monitor: {status}\n")
|
|
if not ok:
|
|
self._exec_monitor = None
|
|
|
|
def _capture(self, alerts: list[Alert]) -> None:
|
|
try:
|
|
snapshot.capture(alerts, SystemState(), self.cfg)
|
|
except Exception as exc: # never let a capture crash the daemon
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
fh.write(f"{time.strftime('%FT%T%z')} capture error: {exc!r}\n")
|
|
|
|
def stop(self, *_a) -> None:
|
|
self._stop.set()
|
|
if self._exec_monitor is not None:
|
|
self._exec_monitor.stop()
|