Serving with FastAPI
FastAPI wraps ML models in typed HTTP endpoints with automatic OpenAPI docs. Load the model once at startup, validate inputs with Pydantic, and expose health checks for orchestration.
Search across all documentation pages
FastAPI wraps ML models in typed HTTP endpoints with automatic OpenAPI docs. Load the model once at startup, validate inputs with Pydantic, and expose health checks for orchestration.
from fastapi import FastAPI
from pydantic import BaseModel
import joblib
app = FastAPI()
model = joblib.load("model.joblib")
class PredictRequest(BaseModel):
features: list[float]
@app.post("/predict")
def predict(req: PredictRequest) -> dict:
pred = model.predict([req.features])
return {"prediction": pred.tolist()}"""serving_fastapi.py - production-ready model API."""
from __future__ import annotations
from contextlib import asynccontextmanager
from typing import Any
import joblib
import numpy as np
from fastapi import FastAPI, HTTPException
from pydantic import BaseModel, Field
MODEL_PATH = "models/churn_pipeline.joblib"
MODEL_VERSION = "1.2.0"
model: Any = None
@asynccontextmanager
async def lifespan(app: FastAPI):
global model
model = joblib.load(MODEL_PATH)
yield
model = None
app = FastAPI(title="Churn Predictor", version=MODEL_VERSION, lifespan=lifespan)
class PredictRequest(BaseModel):
tenure: float = Field(ge=0)
monthly_charges: float = Field(ge=0)
total_charges: float = Field(ge=0)
class PredictResponse(BaseModel):
churn_probability: float
churn_prediction: bool
model_version: str
@app.get("/health")
def health() -> dict:
return {"status": "ok", "model_loaded": model is not None}
@app.post("/predict", response_model=PredictResponse)
def predict(req: PredictRequest) -> PredictResponse:
if model is None:
raise HTTPException(503, "Model not loaded")
features = np.array([[req.tenure, req.monthly_charges, req.total_charges]])
proba = model.predict_proba(features)[0][1]
return PredictResponse(
churn_probability=round(float(proba), 4),
churn_prediction=proba >= 0.5,
model_version=MODEL_VERSION,
)Run: uvicorn serving_fastapi:app --host 0.0.0.0 --port 8000
/health endpoint.| Alternative | Use When | Don't Use When |
|---|---|---|
| FastAPI | Python models, custom logic | Max throughput (use Triton) |
| BentoML | Packaged model serving | Simple single-endpoint API |
| Flask | Legacy apps | New projects (use FastAPI) |
| AWS Lambda + container | Low traffic, serverless | GPU inference |
Stack versions: This page was written for Python 3.14.0 (stable 3.14, maintenance 3.13), FastAPI 0.115+, Django 5.2, Flask 3.1, Pydantic 2, PyTorch 2.6+, pandas 2.2+, Polars 1.x, ruff 0.9+, and uv 0.6+.
Reviewed by Chris St. John·Last updated Jul 19, 2026