Deploy ML models to production via REST APIs, batch jobs, and streaming with Docker and Kubernetes.
Latency: Hours
Example: Nightly churn scoring
Tools: Spark, Airflow
Latency: Milliseconds
Example: Fraud per transaction
Tools: BentoML, FastAPI
Latency: Near real-time
Example: IoT anomaly detection
Tools: Kafka + Flink
# service.py
import bentoml
import numpy as np
from bentoml.io import JSON, NumpyNdarray
churn_runner = bentoml.sklearn.get("churn_model:latest").to_runner()
svc = bentoml.Service("churn_prediction", runners=[churn_runner])
@svc.api(input=JSON(), output=JSON())
async def predict(input_data: dict) -> dict:
features = np.array([[
input_data["total_orders"],
input_data["avg_order_value"],
input_data["days_since_last_order"],
input_data["satisfaction_score"],
]])
prediction = await churn_runner.predict.async_run(features)
proba = await churn_runner.predict_proba.async_run(features)
return {
"will_churn": bool(prediction[0]),
"churn_probability": float(proba[0][1]),
"risk_level": "HIGH" if proba[0][1] > 0.7 else "LOW",
}
# Run: bentoml serve service:svc
# Build: bentoml build
# Docker: bentoml containerize churn_prediction:latestfrom airflow import DAG
from airflow.operators.python import PythonOperator
import mlflow, pandas as pd
def score_all_customers():
model = mlflow.pyfunc.load_model("models:/churn_predictor/Production")
customers = pd.read_sql("SELECT * FROM customer_features", engine)
customers["churn_prob"] = model.predict_proba(customers[features])[:, 1]
customers.to_sql("churn_scores", engine, if_exists="replace")
dag = DAG("churn_scoring", schedule_interval="0 2 * * *")
PythonOperator(task_id="score", python_callable=score_all_customers, dag=dag)# Dockerfile
FROM python:3.10-slim
WORKDIR /app
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY model/ ./model/
COPY service.py .
HEALTHCHECK CMD curl -f http://localhost:8000/health || exit 1
EXPOSE 8000
CMD ["bentoml", "serve", "service:svc", "--port", "8000"]apiVersion: apps/v1
kind: Deployment
metadata:
name: churn-model
spec:
replicas: 3
template:
spec:
containers:
- name: model-server
image: registry/churn-model:v3.0
ports: [{containerPort: 8000}]
resources:
requests: {memory: "512Mi", cpu: "500m"}
limits: {memory: "1Gi", cpu: "1000m"}Model deployment is 90% infrastructure. The only new part is that the app inside the container is a model instead of a data pipeline.