"""Elevator Predictive Maintenance — inference code of the Hugging Face model repo shalev396/elevator-maintenance. Single source of truth for the input schema, the window feature engineering and the deployed RandomForest: the training code (../training) and the Space both import this file, so evaluation and serving run exactly the same code path. !! The model is trained on SYNTHETIC data (a seeded generator of one month of minutely readings from !! 11 elevator sensors). It is a demo of the windowing / feature-engineering workflow, not a model of !! any real elevator. Files next to this module (written by training/src/export.py): config.json sensors, window length, label lag, feature names, tuned decision threshold, versions rf.joblib fitted RandomForestClassifier (300 trees, class_weight="balanced") on 63 window features """ from __future__ import annotations import io import json import os from pathlib import Path import joblib import numpy as np import pandas as pd from sklearn.ensemble import RandomForestClassifier REPO_ID = "shalev396/elevator-maintenance" FRAMEWORK = "sklearn" SENSORS = ["temperature", "humidity", "vibration_rms", "vibration_peak", "motor_current", "motor_rpm", "door_cycles", "load_kg", "acoustic_db", "oil_level", "power_kw"] WINDOW = 60 # minutes of readings per input window (one row per minute) LABEL_LAG = 10 # the label is the machine state this many minutes AFTER the window ends FFT_TOP_K = 5 # top FFT magnitudes of vibration_rms used as features STATS = ("mean", "std", "min", "max", "slope") TIME_COLUMNS = ("timestamp", "date", "time", "datetime") DEFAULT_TIME = pd.Timestamp("2024-01-03 12:00") # weekday noon, used when the CSV has no timestamps CLASSES = ("failure", "healthy") def cuda_available() -> bool: """scikit-learn serves on CPU; this model has no CUDA code path.""" return False # --------------------------------------------------------------------------- features def feature_names(sensors: list[str] = SENSORS, fft_top_k: int = FFT_TOP_K) -> list[str]: """Names of the 63 engineered features, in model input order.""" names = [f"{s}_{stat}" for stat in STATS for s in sensors] names += [f"vib_fft_top{i + 1}" for i in range(fft_top_k)] return names + ["vib_fft_dominant_bin", "hour", "day_of_week"] def window_features(windows: np.ndarray, end_times, sensors: list[str] = SENSORS, fft_top_k: int = FFT_TOP_K) -> np.ndarray: """(n, window, channels) raw readings + n window-end timestamps -> (n, 63) float32 features. Per sensor: mean / std / min / max / least-squares slope over the window (5 x 11 = 55). vibration_rms spectrum: top-k FFT magnitudes (DC dropped) + index of the dominant bin (6). Time of the window end: hour + day of week (2). Fully vectorised. """ x = np.asarray(windows, dtype=np.float64) if x.ndim != 3: raise ValueError(f"expected (n, window, channels) windows, got shape {x.shape}") t = np.arange(x.shape[1]) - (x.shape[1] - 1) / 2.0 # zero-mean time axis slope = (x * t[None, :, None]).sum(axis=1) / float((t ** 2).sum()) stats = [x.mean(axis=1), x.std(axis=1), x.min(axis=1), x.max(axis=1), slope] vib = x[:, :, sensors.index("vibration_rms")] mags = np.abs(np.fft.rfft(vib, axis=1))[:, 1:] # drop the DC component k = min(fft_top_k, mags.shape[1]) top = np.sort(mags, axis=1)[:, ::-1][:, :k] dominant = np.argmax(mags, axis=1).astype(np.float64)[:, None] ends = pd.DatetimeIndex(end_times) hours = np.asarray(ends.hour, dtype=np.float64)[:, None] dows = np.asarray(ends.dayofweek, dtype=np.float64)[:, None] return np.hstack([*stats, top, dominant, hours, dows]).astype(np.float32) def build_forest(n_estimators: int = 300, seed: int = 42, n_jobs: int = -1) -> RandomForestClassifier: """The deployed classifier: balanced class weights for the ~9 % failure windows.""" return RandomForestClassifier(n_estimators=n_estimators, class_weight="balanced", n_jobs=n_jobs, random_state=seed) def tree_nodes(forest: RandomForestClassifier) -> int: """Size of a fitted forest as the total number of tree nodes (reported as its `params`).""" return int(sum(est.tree_.node_count for est in forest.estimators_)) # --------------------------------------------------------------------------- input parsing def read_window(data, sensors: list[str] = SENSORS, window: int = WINDOW) -> tuple[pd.DataFrame, pd.Timestamp | None]: """CSV path / CSV text / DataFrame / records -> (last `window` rows of the sensor columns, end time). One row per minute with the 11 sensor columns (any order, extra columns ignored). An optional first column named timestamp/date/time/datetime supplies the window-end time used for the hour and day-of-week features; without it a weekday noon is assumed. """ if isinstance(data, pd.DataFrame): frame = data.copy() elif isinstance(data, (list, dict)): frame = pd.DataFrame(data) elif isinstance(data, str) and "\n" in data.strip(): # CSV text (header + rows) frame = pd.read_csv(io.StringIO(data)) elif isinstance(data, (str, os.PathLike)): # CSV file path path = Path(data) if not path.is_file(): raise FileNotFoundError(f"CSV file not found: {data}") frame = pd.read_csv(path) elif isinstance(data, bytes): frame = pd.read_csv(io.BytesIO(data)) else: raise ValueError(f"unsupported input type {type(data).__name__}: pass a CSV path, CSV text or DataFrame") end_time = None lookup = {str(c).lower().strip(): c for c in frame.columns} time_col = next((lookup[c] for c in TIME_COLUMNS if c in lookup), None) if time_col is not None: parsed = pd.to_datetime(frame[time_col], errors="coerce") if parsed.notna().all() and len(parsed): end_time = pd.Timestamp(parsed.iloc[-1]) missing = [s for s in sensors if s not in lookup] if missing: raise ValueError(f"CSV is missing sensor column(s): {', '.join(missing)}") values = frame[[lookup[s] for s in sensors]].apply(pd.to_numeric, errors="coerce") values.columns = list(sensors) if values.isna().any().any(): bad = np.flatnonzero(values.isna().any(axis=1).to_numpy()) raise ValueError(f"empty or non-numeric sensor values in row(s) {bad[:10].tolist()} (0-based)") if len(values) < window: raise ValueError(f"need at least {window} rows (one per minute), got {len(values)}") return values.tail(window).reset_index(drop=True), end_time # --------------------------------------------------------------------------- inference class Predictor: """P(failure state LABEL_LAG minutes after the window) from the last WINDOW minutes of readings.""" def __init__(self, model_dir: str | Path, device: str = "cpu"): self.model_dir = Path(model_dir) self.device = "cpu" # CPU-only model (see cuda_available) self.config = json.loads(self._require("config.json").read_text(encoding="utf-8")) self.sensors = list(self.config["sensors"]) self.window = int(self.config["window"]) self.fft_top_k = int(self.config["fft_top_k"]) self.threshold = float(self.config["threshold"]) self.forest = joblib.load(self._require(self.config["weights"])) if self.forest.n_features_in_ != len(self.config["feature_names"]): raise ValueError("rf.joblib does not match config.json feature_names") def _require(self, filename: str) -> Path: path = self.model_dir / filename if not path.is_file(): raise FileNotFoundError(f"{path} not found: train and export the model first " "(training/notebook.ipynb) or download shalev396/elevator-maintenance") return path def features(self, data) -> np.ndarray: frame, end_time = read_window(data, self.sensors, self.window) return window_features(frame.to_numpy()[None], [end_time or DEFAULT_TIME], self.sensors, self.fft_top_k) def probability(self, data) -> float: """P(failure) for one window.""" return float(self.forest.predict_proba(self.features(data))[0, 1]) def predict(self, data) -> dict: """CSV path / CSV text / DataFrame of >= 60 minutely rows -> {"failure": p, "healthy": 1 - p}. The decision threshold tuned on validation (max F1) is `self.threshold` (config.json): flag the elevator for maintenance when failure >= threshold. """ p = self.probability(data) return {"failure": p, "healthy": 1.0 - p} def load(model_dir: str | Path, device: str = "cpu") -> Predictor: return Predictor(model_dir, device)