Detect data drift, concept drift, and performance degradation before they impact your business.
Traditional monitoring tracks if the system is up. ML monitoring tracks if the system is correct. A model can run perfectly while giving wrong predictions.
Your fraud model runs 24/7 with perfect uptime. But fraudsters changed tactics. The model misses 40% of fraud. Traditional monitoring shows green. You lose $2M. ML monitoring prevents this.
| Pillar | What to Monitor | Alert Threshold | Tools |
|---|---|---|---|
| Data Quality | Nulls, schema, volume | Any schema change, >5% nulls | Great Expectations |
| Data Drift | Feature distributions | PSI > 0.15 | Evidently, Whylogs |
| Model Performance | Accuracy, F1, latency | Accuracy drops > 5% | Prometheus + Grafana |
| Business KPIs | Revenue impact | KPI misses > 10% | Custom dashboards |
from evidently.report import Report
from evidently.metric_preset import DataDriftPreset, DataQualityPreset
class MLMonitor:
def __init__(self, reference_data):
self.reference = reference_data
def check_drift(self, current_data):
report = Report(metrics=[DataDriftPreset()])
report.run(reference_data=self.reference, current_data=current_data)
result = report.as_dict()
drift = result["metrics"][0]["result"]["dataset_drift"]
if drift:
print("DRIFT DETECTED! Triggering retraining...")
trigger_retraining()
else:
print("No drift. Model is healthy.")
report.save_html("drift_report.html")
return reportimport numpy as np
from scipy import stats
def calculate_psi(expected, actual, buckets=10):
breakpoints = np.percentile(expected, np.linspace(0, 100, buckets+1))
exp_counts = np.histogram(expected, breakpoints)[0] / len(expected)
act_counts = np.histogram(actual, breakpoints)[0] / len(actual)
exp_counts = np.clip(exp_counts, 0.001, None)
act_counts = np.clip(act_counts, 0.001, None)
return np.sum((act_counts - exp_counts) * np.log(act_counts / exp_counts))
# PSI < 0.1 = No drift
# PSI 0.1-0.2 = Moderate (investigate)
# PSI > 0.2 = Significant (retrain!)from prometheus_client import Counter, Histogram, Gauge, start_http_server
PREDICTIONS = Counter("model_predictions_total", "Total predictions",
["model", "result"])
LATENCY = Histogram("model_latency_seconds", "Inference latency", ["model"])
ACCURACY = Gauge("model_accuracy", "Rolling accuracy", ["model"])
DRIFT = Gauge("feature_drift_psi", "PSI score", ["model", "feature"])
def predict_with_monitoring(features):
import time
start = time.time()
prediction = model.predict(features)
PREDICTIONS.labels(model="churn_v3", result=str(prediction[0])).inc()
LATENCY.labels(model="churn_v3").observe(time.time() - start)
return prediction
start_http_server(8001) # Prometheus scrapes thisIf you use Prometheus + Grafana or Datadog already, ML monitoring just adds model-specific metrics to your existing dashboards.