enodia-sentinal/enodia_sentinel/daemon.py
Luna 0eb5077551 Add event-driven eBPF execve layer with a Snort-style rule engine
Closes polling's blind spot (processes that exit between sweeps) with a real
eBPF probe and a declarative, data-driven detection engine — inspired by Snort
(rule language, signature IDs) and OSSEC (host-IDS framing).

New events/ subpackage:
- bcc_source.py : eBPF C tracing execve (filename, argv[1..2], ppid, uid,
                  parent comm) over a perf buffer, loaded via bcc; lazy import
                  + available()/try-except so it fails closed to poll-only when
                  bcc/root/BTF are absent — a broken probe never downs the daemon
- exec_event.py : the ExecEvent type
- rules.py      : ExecRule (sid/msg/severity/classtype + path/exec/parent/argv
                  conditions) and ExecRuleEngine; 4 shipped rules (fileless exec
                  100001, reverse-shell argv 100002, web/DB→shell RCE 100003,
                  curl|sh 100004); operators add more via exec_rules_file TOML
- monitor.py    : runs the source on a thread, routes events through the engine

Integration:
- daemon starts the monitor, shares a lock-guarded cooldown with the sweep
  loop, and feeds event alerts into the same snapshot pipeline
- Alert gains Snort-style sid + classtype; retrofitted onto all 7 poll
  detectors; snapshots and JSON now carry them
- config: ebpf_exec_monitor (default on, degrades), exec_rules_file
- systemd: opt-in ebpf.conf drop-in (relaxes MemoryDenyWriteExecute + widens
  caps for bcc's JIT) so the base unit stays hardened for poll-only
- sentinel-redteam: ebpf_exec drill (short-lived /tmp exec + /dev/tcp argv the
  poller can't see); footer now uses the Python CLI

Tests: +14 cases for the rule engine (each default rule match/non-match, rule
validation, parent-exclude). 39/39 pass. Graceful non-root degradation verified.

NOTE: the eBPF C follows bcc's execsnoop pattern but could not be run here
(BPF needs root); it wants a root smoke-test on a real host.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-05-31 07:16:53 -07:00

173 lines
7.1 KiB
Python

# SPDX-License-Identifier: GPL-3.0-or-later
"""The detection daemon: sweep loop, cooldown dedup, baseline management.
Unlike the bash prototype, all loop state (cooldowns, last-scan timestamps,
baselines) lives in this object — no subshell-state surprises — and the
expensive filesystem-wide SUID scan is gated to its own slow cadence.
"""
from __future__ import annotations
import json
import threading
import time
from pathlib import Path
from . import detectors, snapshot
from .alert import Alert
from .config import Config
from .system import SystemState, scan_suid_binaries
class Sentinel:
def __init__(self, cfg: Config) -> None:
self.cfg = cfg
self.start_time = time.time()
self.cooldowns: dict[str, float] = {}
# Cooldowns are touched by both the sweep loop and the eBPF event
# thread, so guard them.
self._cooldown_lock = threading.Lock()
self._exec_monitor = None
self.last_persist_scan = self.start_time
self.listener_baseline: set[str] = set()
self.suid_baseline: set[str] = set()
# SUID scan runs off the loop thread; the loop reads the latest result.
self._suid_current: list[str] | None = None
self._suid_thread: threading.Thread | None = None
self._last_suid_scan = 0.0
self._last_suid_baseline_refresh = self.start_time
self._stop = threading.Event()
# -- baselines ---------------------------------------------------------
def build_baselines(self) -> None:
self.listener_baseline = SystemState().listener_keys()
self.suid_baseline = set(scan_suid_binaries(
extra_dirs=self.cfg.suid_scan_extra_dirs))
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
self._save(self.cfg.listener_baseline, sorted(self.listener_baseline))
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
def load_baselines(self) -> None:
self.listener_baseline = set(self._load(self.cfg.listener_baseline))
self.suid_baseline = set(self._load(self.cfg.suid_baseline))
@staticmethod
def _save(path: Path, data: list[str]) -> None:
path.write_text(json.dumps(data))
@staticmethod
def _load(path: Path) -> list[str]:
try:
return json.loads(path.read_text())
except (OSError, ValueError):
return []
# -- SUID scan (off the loop thread) -----------------------------------
def _maybe_scan_suid(self, now: float) -> None:
if self._suid_thread and self._suid_thread.is_alive():
return
if (now - self._last_suid_scan) < self.cfg.suid_scan_interval:
return
self._last_suid_scan = now
self._suid_thread = threading.Thread(target=self._scan_suid, daemon=True)
self._suid_thread.start()
def _scan_suid(self) -> None:
result = scan_suid_binaries(extra_dirs=self.cfg.suid_scan_extra_dirs)
self._suid_current = result
# Periodically fold the current state into the baseline so legitimately
# installed SUID binaries stop alerting after a while.
if (time.time() - self._last_suid_baseline_refresh) >= self.cfg.suid_refresh:
self.suid_baseline = set(result)
self._last_suid_baseline_refresh = time.time()
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
# -- one sweep ---------------------------------------------------------
def sweep(self, *, force_suid: bool = False) -> list[Alert]:
now = time.time()
armed = (now - self.start_time) >= self.cfg.baseline_grace
if force_suid:
suid_binaries = scan_suid_binaries(
extra_dirs=self.cfg.suid_scan_extra_dirs)
else:
suid_binaries = self._suid_current # latest async result (may be None)
state = SystemState(
listener_baseline=self.listener_baseline if armed or force_suid else None,
suid_baseline=self.suid_baseline,
suid_binaries=suid_binaries if armed or force_suid else None,
persist_since=self.last_persist_scan if armed or force_suid else None,
)
alerts = list(detectors.run_all(state, self.cfg))
self.last_persist_scan = now
return alerts
def fresh_alerts(self, alerts: list[Alert], now: float) -> list[Alert]:
"""Drop alerts whose dedup key is still within cooldown (thread-safe)."""
out = []
with self._cooldown_lock:
for a in alerts:
prev = self.cooldowns.get(a.key, 0.0)
if (now - prev) >= self.cfg.cooldown:
self.cooldowns[a.key] = now
out.append(a)
return out
def _on_exec_alert(self, alert: Alert) -> None:
"""Callback for the eBPF exec monitor — same dedup + capture path."""
fresh = self.fresh_alerts([alert], time.time())
if fresh:
threading.Thread(
target=self._capture, args=(fresh,), daemon=True
).start()
# -- main loop ---------------------------------------------------------
def run(self) -> None:
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
with open(self.cfg.events_log, "a") as fh:
fh.write(f"{time.strftime('%FT%T%z')} enodia-sentinel started\n")
self.build_baselines()
snapshot.prune(self.cfg)
self._start_exec_monitor()
sweeps = 0
while not self._stop.is_set():
now = time.time()
if (now - self.start_time) >= self.cfg.baseline_grace:
self._maybe_scan_suid(now)
alerts = self.sweep()
fresh = self.fresh_alerts(alerts, now)
if fresh:
# capture off the loop thread so a slow snapshot never stalls
# detection; the SystemState used for forensics is rebuilt fresh
# inside the thread for accuracy.
threading.Thread(
target=self._capture, args=(fresh,), daemon=True
).start()
sweeps += 1
if sweeps % 20 == 0:
snapshot.prune(self.cfg)
self._stop.wait(self.cfg.sample_interval)
def _start_exec_monitor(self) -> None:
if not self.cfg.ebpf_exec_monitor:
return
from .events.monitor import ExecMonitor
self._exec_monitor = ExecMonitor(self.cfg, self._on_exec_alert)
ok, reason = self._exec_monitor.start()
with open(self.cfg.events_log, "a") as fh:
status = "enabled" if ok else f"disabled ({reason})"
fh.write(f"{time.strftime('%FT%T%z')} eBPF exec monitor: {status}\n")
if not ok:
self._exec_monitor = None
def _capture(self, alerts: list[Alert]) -> None:
try:
snapshot.capture(alerts, SystemState(), self.cfg)
except Exception as exc: # never let a capture crash the daemon
with open(self.cfg.events_log, "a") as fh:
fh.write(f"{time.strftime('%FT%T%z')} capture error: {exc!r}\n")
def stop(self, *_a) -> None:
self._stop.set()
if self._exec_monitor is not None:
self._exec_monitor.stop()