Spaces:
Running on Zero
Running on Zero
| """ | |
| anomaly_model.py | |
| ----------------- | |
| Isolation Forest based anomaly detector for conveyor / crane motor sensor | |
| streams (motor temperature, vibration, current draw, belt speed). This | |
| powers the "Predictive Maintenance" tab -- flags abnormal equipment | |
| behaviour before it causes an unplanned stoppage, which is exactly the kind | |
| of workload modern intralogistics platforms (e.g. AS/RS, sorters, AGVs) | |
| generate continuously in production. | |
| Threshold calibration | |
| ---------------------- | |
| The model itself is fit unsupervised (IsolationForest never sees labels). | |
| However, converting its continuous anomaly score into a binary decision | |
| requires a threshold, and scikit-learn's default choices for this (a fixed | |
| "contamination" rate) are either a blind guess or -- if set to the true | |
| label rate -- a form of leakage (see build_artifacts.py for the full | |
| writeup of that bug and fix). The approach used here instead is standard | |
| practice for production anomaly detection: fit the model unsupervised on | |
| unlabelled data, then calibrate the decision threshold using a small | |
| *labelled calibration set* (e.g. a handful of confirmed historical | |
| incidents), completely separate from both model fitting and the final | |
| held-out test set used to report metrics. `calibrated_threshold` below is | |
| that calibrated cutoff, chosen to maximise F1 on the calibration set only. | |
| """ | |
| from dataclasses import dataclass | |
| from typing import Optional | |
| import joblib | |
| import numpy as np | |
| from sklearn.ensemble import IsolationForest | |
| from sklearn.preprocessing import StandardScaler | |
| FEATURES = ["motor_temp_c", "vibration_mm_s", "current_amps", "belt_speed_mps"] | |
| class AnomalyResult: | |
| is_anomaly: bool | |
| anomaly_score: float # higher = more anomalous, roughly in [0, 1] | |
| raw_score: float | |
| def build_model(contamination="auto", seed: int = 42) -> IsolationForest: | |
| return IsolationForest( | |
| n_estimators=200, | |
| contamination=contamination, | |
| random_state=seed, | |
| ) | |
| def score_reading( | |
| model: IsolationForest, | |
| scaler: StandardScaler, | |
| reading: dict, | |
| threshold: Optional[float] = None, | |
| ) -> AnomalyResult: | |
| """ | |
| Score a single sensor reading. | |
| If `threshold` is given, it's compared against the 0-1 anomaly_score | |
| (this is the calibrated-threshold path used by the deployed app). If | |
| omitted, falls back to the IsolationForest's own .predict() (its | |
| internal contamination-derived offset). | |
| """ | |
| x = np.array([[reading[f] for f in FEATURES]]) | |
| x_scaled = scaler.transform(x) | |
| raw = model.decision_function(x_scaled)[0] # higher = more normal | |
| # squash raw decision_function (~[-0.5, 0.5]) into a 0-1 "anomaly score" | |
| anomaly_score = float(np.clip(0.5 - raw, 0, 1)) | |
| if threshold is not None: | |
| is_anomaly = anomaly_score >= threshold | |
| else: | |
| pred = model.predict(x_scaled)[0] # 1 = normal, -1 = anomaly | |
| is_anomaly = (pred == -1) | |
| return AnomalyResult(is_anomaly=bool(is_anomaly), anomaly_score=anomaly_score, raw_score=float(raw)) | |
| def save_artifacts(model, scaler, model_path: str, scaler_path: str, threshold: Optional[float] = None, threshold_path: Optional[str] = None): | |
| joblib.dump(model, model_path) | |
| joblib.dump(scaler, scaler_path) | |
| if threshold is not None and threshold_path is not None: | |
| joblib.dump(float(threshold), threshold_path) | |
| def load_artifacts(model_path: str, scaler_path: str, threshold_path: Optional[str] = None): | |
| model = joblib.load(model_path) | |
| scaler = joblib.load(scaler_path) | |
| if threshold_path is not None: | |
| try: | |
| threshold = joblib.load(threshold_path) | |
| return model, scaler, threshold | |
| except FileNotFoundError: | |
| return model, scaler, None | |
| return model, scaler | |