Closes polling's blind spot (processes that exit between sweeps) with a real
eBPF probe and a declarative, data-driven detection engine — inspired by Snort
(rule language, signature IDs) and OSSEC (host-IDS framing).
New events/ subpackage:
- bcc_source.py : eBPF C tracing execve (filename, argv[1..2], ppid, uid,
parent comm) over a perf buffer, loaded via bcc; lazy import
+ available()/try-except so it fails closed to poll-only when
bcc/root/BTF are absent — a broken probe never downs the daemon
- exec_event.py : the ExecEvent type
- rules.py : ExecRule (sid/msg/severity/classtype + path/exec/parent/argv
conditions) and ExecRuleEngine; 4 shipped rules (fileless exec
100001, reverse-shell argv 100002, web/DB→shell RCE 100003,
curl|sh 100004); operators add more via exec_rules_file TOML
- monitor.py : runs the source on a thread, routes events through the engine
Integration:
- daemon starts the monitor, shares a lock-guarded cooldown with the sweep
loop, and feeds event alerts into the same snapshot pipeline
- Alert gains Snort-style sid + classtype; retrofitted onto all 7 poll
detectors; snapshots and JSON now carry them
- config: ebpf_exec_monitor (default on, degrades), exec_rules_file
- systemd: opt-in ebpf.conf drop-in (relaxes MemoryDenyWriteExecute + widens
caps for bcc's JIT) so the base unit stays hardened for poll-only
- sentinel-redteam: ebpf_exec drill (short-lived /tmp exec + /dev/tcp argv the
poller can't see); footer now uses the Python CLI
Tests: +14 cases for the rule engine (each default rule match/non-match, rule
validation, parent-exclude). 39/39 pass. Graceful non-root degradation verified.
NOTE: the eBPF C follows bcc's execsnoop pattern but could not be run here
(BPF needs root); it wants a root smoke-test on a real host.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
173 lines
7.1 KiB
Python
173 lines
7.1 KiB
Python
# SPDX-License-Identifier: GPL-3.0-or-later
|
|
"""The detection daemon: sweep loop, cooldown dedup, baseline management.
|
|
|
|
Unlike the bash prototype, all loop state (cooldowns, last-scan timestamps,
|
|
baselines) lives in this object — no subshell-state surprises — and the
|
|
expensive filesystem-wide SUID scan is gated to its own slow cadence.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import threading
|
|
import time
|
|
from pathlib import Path
|
|
|
|
from . import detectors, snapshot
|
|
from .alert import Alert
|
|
from .config import Config
|
|
from .system import SystemState, scan_suid_binaries
|
|
|
|
|
|
class Sentinel:
|
|
def __init__(self, cfg: Config) -> None:
|
|
self.cfg = cfg
|
|
self.start_time = time.time()
|
|
self.cooldowns: dict[str, float] = {}
|
|
# Cooldowns are touched by both the sweep loop and the eBPF event
|
|
# thread, so guard them.
|
|
self._cooldown_lock = threading.Lock()
|
|
self._exec_monitor = None
|
|
self.last_persist_scan = self.start_time
|
|
self.listener_baseline: set[str] = set()
|
|
self.suid_baseline: set[str] = set()
|
|
# SUID scan runs off the loop thread; the loop reads the latest result.
|
|
self._suid_current: list[str] | None = None
|
|
self._suid_thread: threading.Thread | None = None
|
|
self._last_suid_scan = 0.0
|
|
self._last_suid_baseline_refresh = self.start_time
|
|
self._stop = threading.Event()
|
|
|
|
# -- baselines ---------------------------------------------------------
|
|
def build_baselines(self) -> None:
|
|
self.listener_baseline = SystemState().listener_keys()
|
|
self.suid_baseline = set(scan_suid_binaries(
|
|
extra_dirs=self.cfg.suid_scan_extra_dirs))
|
|
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
|
|
self._save(self.cfg.listener_baseline, sorted(self.listener_baseline))
|
|
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
|
|
|
|
def load_baselines(self) -> None:
|
|
self.listener_baseline = set(self._load(self.cfg.listener_baseline))
|
|
self.suid_baseline = set(self._load(self.cfg.suid_baseline))
|
|
|
|
@staticmethod
|
|
def _save(path: Path, data: list[str]) -> None:
|
|
path.write_text(json.dumps(data))
|
|
|
|
@staticmethod
|
|
def _load(path: Path) -> list[str]:
|
|
try:
|
|
return json.loads(path.read_text())
|
|
except (OSError, ValueError):
|
|
return []
|
|
|
|
# -- SUID scan (off the loop thread) -----------------------------------
|
|
def _maybe_scan_suid(self, now: float) -> None:
|
|
if self._suid_thread and self._suid_thread.is_alive():
|
|
return
|
|
if (now - self._last_suid_scan) < self.cfg.suid_scan_interval:
|
|
return
|
|
self._last_suid_scan = now
|
|
self._suid_thread = threading.Thread(target=self._scan_suid, daemon=True)
|
|
self._suid_thread.start()
|
|
|
|
def _scan_suid(self) -> None:
|
|
result = scan_suid_binaries(extra_dirs=self.cfg.suid_scan_extra_dirs)
|
|
self._suid_current = result
|
|
# Periodically fold the current state into the baseline so legitimately
|
|
# installed SUID binaries stop alerting after a while.
|
|
if (time.time() - self._last_suid_baseline_refresh) >= self.cfg.suid_refresh:
|
|
self.suid_baseline = set(result)
|
|
self._last_suid_baseline_refresh = time.time()
|
|
self._save(self.cfg.suid_baseline, sorted(self.suid_baseline))
|
|
|
|
# -- one sweep ---------------------------------------------------------
|
|
def sweep(self, *, force_suid: bool = False) -> list[Alert]:
|
|
now = time.time()
|
|
armed = (now - self.start_time) >= self.cfg.baseline_grace
|
|
|
|
if force_suid:
|
|
suid_binaries = scan_suid_binaries(
|
|
extra_dirs=self.cfg.suid_scan_extra_dirs)
|
|
else:
|
|
suid_binaries = self._suid_current # latest async result (may be None)
|
|
|
|
state = SystemState(
|
|
listener_baseline=self.listener_baseline if armed or force_suid else None,
|
|
suid_baseline=self.suid_baseline,
|
|
suid_binaries=suid_binaries if armed or force_suid else None,
|
|
persist_since=self.last_persist_scan if armed or force_suid else None,
|
|
)
|
|
alerts = list(detectors.run_all(state, self.cfg))
|
|
self.last_persist_scan = now
|
|
return alerts
|
|
|
|
def fresh_alerts(self, alerts: list[Alert], now: float) -> list[Alert]:
|
|
"""Drop alerts whose dedup key is still within cooldown (thread-safe)."""
|
|
out = []
|
|
with self._cooldown_lock:
|
|
for a in alerts:
|
|
prev = self.cooldowns.get(a.key, 0.0)
|
|
if (now - prev) >= self.cfg.cooldown:
|
|
self.cooldowns[a.key] = now
|
|
out.append(a)
|
|
return out
|
|
|
|
def _on_exec_alert(self, alert: Alert) -> None:
|
|
"""Callback for the eBPF exec monitor — same dedup + capture path."""
|
|
fresh = self.fresh_alerts([alert], time.time())
|
|
if fresh:
|
|
threading.Thread(
|
|
target=self._capture, args=(fresh,), daemon=True
|
|
).start()
|
|
|
|
# -- main loop ---------------------------------------------------------
|
|
def run(self) -> None:
|
|
self.cfg.log_dir.mkdir(parents=True, exist_ok=True)
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
fh.write(f"{time.strftime('%FT%T%z')} enodia-sentinel started\n")
|
|
self.build_baselines()
|
|
snapshot.prune(self.cfg)
|
|
self._start_exec_monitor()
|
|
sweeps = 0
|
|
while not self._stop.is_set():
|
|
now = time.time()
|
|
if (now - self.start_time) >= self.cfg.baseline_grace:
|
|
self._maybe_scan_suid(now)
|
|
alerts = self.sweep()
|
|
fresh = self.fresh_alerts(alerts, now)
|
|
if fresh:
|
|
# capture off the loop thread so a slow snapshot never stalls
|
|
# detection; the SystemState used for forensics is rebuilt fresh
|
|
# inside the thread for accuracy.
|
|
threading.Thread(
|
|
target=self._capture, args=(fresh,), daemon=True
|
|
).start()
|
|
sweeps += 1
|
|
if sweeps % 20 == 0:
|
|
snapshot.prune(self.cfg)
|
|
self._stop.wait(self.cfg.sample_interval)
|
|
|
|
def _start_exec_monitor(self) -> None:
|
|
if not self.cfg.ebpf_exec_monitor:
|
|
return
|
|
from .events.monitor import ExecMonitor
|
|
self._exec_monitor = ExecMonitor(self.cfg, self._on_exec_alert)
|
|
ok, reason = self._exec_monitor.start()
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
status = "enabled" if ok else f"disabled ({reason})"
|
|
fh.write(f"{time.strftime('%FT%T%z')} eBPF exec monitor: {status}\n")
|
|
if not ok:
|
|
self._exec_monitor = None
|
|
|
|
def _capture(self, alerts: list[Alert]) -> None:
|
|
try:
|
|
snapshot.capture(alerts, SystemState(), self.cfg)
|
|
except Exception as exc: # never let a capture crash the daemon
|
|
with open(self.cfg.events_log, "a") as fh:
|
|
fh.write(f"{time.strftime('%FT%T%z')} capture error: {exc!r}\n")
|
|
|
|
def stop(self, *_a) -> None:
|
|
self._stop.set()
|
|
if self._exec_monitor is not None:
|
|
self._exec_monitor.stop()
|