Files
dipl-edr/agent/model.py
T

225 lines
8.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Anomaly detection — voting ensemble of three unsupervised models.
Pipeline:
IsolationForest + Local Outlier Factor + One-Class SVM
A process is flagged anomalous when at least 2 of 3 models agree.
Training data (in order of preference):
1. UNSW-NB15 dataset CSV (data/UNSW_NB15_training-set.csv)
2. Synthetic normal-behaviour data (fallback, 2000 rows)
"""
import logging
import os
from pathlib import Path
import numpy as np
import joblib
logger = logging.getLogger(__name__)
MODEL_PATH = Path(os.getenv("MODEL_PATH", "data/model.pkl"))
DATASET_PATH = Path(os.getenv("DATASET_PATH", "data/UNSW_NB15_training-set.csv"))
# Trusted desktop applications — ML classification skipped.
# Threshold alerts (CPU > 80 %, RAM > 3 GB) still fire from the backend.
TRUSTED_PROCESSES: frozenset[str] = frozenset({
"firefox", "firefox-bin", "Web Content", "Isolated Web Co",
"chrome", "chromium", "chromium-browser",
"brave", "brave-browser", "opera",
"electron", "Electron",
"telegram", "Telegram", "telegram-desktop",
"slack", "discord", "discord-bin",
"code", "code-oss", "vscodium", "cursor",
"zoom", "zoom-bin", "zoomus",
"obsidian",
"spotify", "vlc", "mpv",
"libreoffice", "soffice",
"thunderbird",
"gnome-shell", "plasmashell", "kwin_wayland", "kwin_x11",
"Xorg", "Xwayland",
})
# Feature vector order: [cpu_percent, memory_mb, open_files, connections]
FEATURE_COLS = ["cpu_percent", "memory_mb", "open_files", "connections"]
class EnsembleDetector:
"""Majority-vote ensemble: IsolationForest + LOF + One-Class SVM.
Each model casts a vote (+1 normal, -1 anomaly).
A process is anomalous when the vote sum is ≤ -1 (i.e. 2 or 3 models agree).
"""
def __init__(self) -> None:
self.iforest = None
self.lof = None
self.ocsvm = None
# ── training ─────────────────────────────────────────────────────────────
def train(self, X: np.ndarray) -> None:
"""Fit all three models on X and persist the ensemble to disk."""
from sklearn.ensemble import IsolationForest
from sklearn.neighbors import LocalOutlierFactor
from sklearn.svm import OneClassSVM
from sklearn.preprocessing import RobustScaler
logger.info("Training ensemble on %d samples …", X.shape[0])
# RobustScaler is median/IQR-based — less sensitive to extreme outliers
# than StandardScaler, which matters when training data may contain
# occasional noisy spikes.
scaler = RobustScaler()
X_scaled = scaler.fit_transform(X)
self.iforest = IsolationForest(
n_estimators=100,
contamination=0.01,
random_state=42,
n_jobs=-1,
)
self.iforest.fit(X_scaled)
logger.info(" IsolationForest trained")
# LOF in novelty=True mode so it can predict on new samples (not just
# the training set). n_neighbors=20 is a robust default for datasets
# of ~10005000 rows.
self.lof = LocalOutlierFactor(
n_neighbors=20,
contamination=0.01,
novelty=True,
n_jobs=-1,
)
self.lof.fit(X_scaled)
logger.info(" LocalOutlierFactor trained")
# One-Class SVM with RBF kernel. nu=0.01 ≈ expected outlier fraction.
# gamma="scale" adapts to feature variance automatically.
self.ocsvm = OneClassSVM(kernel="rbf", nu=0.01, gamma="scale")
self.ocsvm.fit(X_scaled)
logger.info(" OneClassSVM trained")
MODEL_PATH.parent.mkdir(parents=True, exist_ok=True)
joblib.dump({"iforest": self.iforest, "lof": self.lof,
"ocsvm": self.ocsvm, "scaler": scaler}, MODEL_PATH)
logger.info("Ensemble saved → %s", MODEL_PATH)
# Keep the scaler attached for inference
self._scaler = scaler
def load_or_train(self) -> None:
"""Load persisted ensemble or train a new one."""
if MODEL_PATH.exists():
bundle = joblib.load(MODEL_PATH)
# Support legacy single-model pkl (no "scaler" key)
if not isinstance(bundle, dict):
logger.info("Legacy single-model pkl detected — retraining ensemble")
MODEL_PATH.unlink()
self.load_or_train()
return
self.iforest = bundle["iforest"]
self.lof = bundle["lof"]
self.ocsvm = bundle["ocsvm"]
self._scaler = bundle["scaler"]
logger.info("Ensemble loaded from %s", MODEL_PATH)
return
logger.info("No saved model found — training now")
X = self._load_dataset()
self.train(X)
# ── data helpers ──────────────────────────────────────────────────────────
def _load_dataset(self) -> np.ndarray:
if DATASET_PATH.exists():
return self._load_unsw()
logger.warning("Dataset not found — using synthetic training data")
return self._synthetic_data()
def _load_unsw(self) -> np.ndarray:
import pandas as pd
df = pd.read_csv(DATASET_PATH)
numeric = df.select_dtypes(include=[np.number]).dropna()
X = numeric.iloc[:, :4].values.astype(np.float32)
logger.info("Loaded UNSW-NB15: %d rows", X.shape[0])
return X
def _synthetic_data(self) -> np.ndarray:
"""2000 rows of realistic normal Linux process behaviour.
Distribution:
75 % idle/background (CPU ≈ 0 %, low RAM)
20 % light workers (moderate CPU, moderate RAM)
5 % heavier services (higher RAM, some connections)
"""
rng = np.random.default_rng(42)
n = 2000
cpu_idle = rng.exponential(0.3, int(n * 0.75))
cpu_light = rng.uniform(0.5, 15, int(n * 0.20))
cpu_heavy = rng.uniform(15, 60, int(n * 0.05))
cpu = np.concatenate([cpu_idle, cpu_light, cpu_heavy])
mem_small = rng.uniform(0.5, 50, int(n * 0.50))
mem_medium = rng.uniform(50, 300, int(n * 0.30))
mem_large = rng.uniform(300, 800, int(n * 0.12))
mem_desktop = rng.uniform(800, 2000, int(n * 0.08))
mem = np.concatenate([mem_small, mem_medium, mem_large, mem_desktop])
files = np.concatenate([
rng.integers(0, 20, int(n * 0.70)).astype(float),
rng.integers(20, 80, int(n * 0.30)).astype(float),
])
conns = np.concatenate([
rng.integers(0, 2, int(n * 0.70)).astype(float),
rng.integers(2, 20, int(n * 0.30)).astype(float),
])
idx = rng.permutation(n)
return np.column_stack([cpu[idx], mem[idx], files[idx], conns[idx]])
# ── inference ─────────────────────────────────────────────────────────────
def predict(self, processes: list[dict]) -> list[dict]:
"""Annotate each process dict with is_anomaly=True/False.
Trusted processes skip ML — they can only be flagged by the
backend's threshold checks (CPU > 80 %, RAM > 3 GB).
"""
if not processes or self.iforest is None:
for p in processes:
p["is_anomaly"] = False
return processes
trusted_idx = {i for i, p in enumerate(processes)
if p["name"] in TRUSTED_PROCESSES}
candidates = [p for i, p in enumerate(processes) if i not in trusted_idx]
if candidates:
X_raw = np.array(
[[p["cpu_percent"], p["memory_mb"], p["open_files"], p["connections"]]
for p in candidates],
dtype=np.float32,
)
X = self._scaler.transform(X_raw)
# Each model returns +1 (normal) or -1 (anomaly)
v_if = self.iforest.predict(X) # IsolationForest
v_lof = self.lof.predict(X) # Local Outlier Factor
v_ocsvm = self.ocsvm.predict(X) # One-Class SVM
# Majority vote: flag anomaly when vote_sum ≤ -1 (≥ 2 of 3 agree)
vote_sum = v_if + v_lof + v_ocsvm
for proc, vs in zip(candidates, vote_sum):
proc["is_anomaly"] = bool(vs <= -1)
for i in trusted_idx:
processes[i]["is_anomaly"] = False
return processes
# Module-level alias so existing import `from model import AnomalyDetector`
# continues to work without changes in main.py.
AnomalyDetector = EnsembleDetector