Shadow Deployment for ML Models: Testing New Models in Production Without Risk
You've trained a new model. Offline metrics look better than the current production model. But will it actually perform better with real users and real traffic?
Shadow deployment answers that question without risk. You run the new model in parallel with production, compare their predictions, and collect real-world performance data — all before any user sees a single output from the new model.
What Shadow Deployment Means for ML
In shadow mode, every incoming request is routed to both the champion (current production model) and the challenger (new model). The champion's prediction is returned to the user. The challenger's prediction is logged and discarded. No user is affected.
User Request
│
├──→ Champion Model → Response to User
│
└──→ Shadow Model → Prediction Logged (silently)This lets you answer the hardest question in ML deployment: "Will this model behave differently in production than it did on the test set?"
Implementing Shadow Mode
Basic Shadow Proxy
import asyncio
import logging
from typing import Any
from dataclasses import dataclass
logger = logging.getLogger(__name__)
@dataclass
class ShadowResult:
champion_prediction: Any
shadow_prediction: Any
request_id: str
agreed: bool
class ShadowDeployment:
def __init__(self, champion, shadow, metrics_client):
self.champion = champion
self.shadow = shadow
self.metrics = metrics_client
async def predict(self, features: dict, request_id: str) -> Any:
"""Return champion prediction; run shadow prediction in background."""
# Champion prediction is synchronous and blocking
champion_pred = self.champion.predict(features)
# Shadow prediction runs async in background — never blocks user
asyncio.create_task(
self._run_shadow(features, champion_pred, request_id)
)
return champion_pred
async def _run_shadow(self, features, champion_pred, request_id):
"""Run shadow model and log comparison — never raises exceptions."""
try:
shadow_pred = await asyncio.to_thread(
self.shadow.predict, features
)
agreed = self._predictions_agree(champion_pred, shadow_pred)
self.metrics.record({
"event": "shadow_prediction",
"request_id": request_id,
"champion": champion_pred,
"shadow": shadow_pred,
"agreed": agreed,
})
if not agreed:
logger.debug(
f"Shadow disagreement [{request_id}]: "
f"champion={champion_pred}, shadow={shadow_pred}"
)
except Exception as e:
# Shadow failures must NEVER affect the user experience
logger.error(f"Shadow prediction failed [{request_id}]: {e}")
self.metrics.record({"event": "shadow_error", "error": str(e)})
def _predictions_agree(self, champion, shadow) -> bool:
"""Check if predictions are meaningfully equivalent."""
if isinstance(champion, float):
return abs(champion - shadow) < 0.1 # Within 10% for regression
return champion == shadow # Exact match for classificationFastAPI Integration
from fastapi import FastAPI, Request
import uuid
app = FastAPI()
shadow_deployment = ShadowDeployment(champion_model, shadow_model, metrics)
@app.post("/predict")
async def predict(request: Request):
body = await request.json()
request_id = str(uuid.uuid4())
prediction = await shadow_deployment.predict(
features=body["features"],
request_id=request_id,
)
return {
"prediction": prediction,
"request_id": request_id,
}Analyzing Shadow Results
After collecting shadow data, analyze agreement rates and divergence patterns:
import pandas as pd
import numpy as np
from scipy import stats
def analyze_shadow_results(shadow_logs_df: pd.DataFrame) -> dict:
"""Analyze champion vs shadow prediction comparison."""
n_total = len(shadow_logs_df)
n_agreed = shadow_logs_df["agreed"].sum()
agreement_rate = n_agreed / n_total
# For regression models: distribution comparison
champion_scores = shadow_logs_df["champion"].values
shadow_scores = shadow_logs_df["shadow"].values
ks_stat, ks_p = stats.ks_2samp(champion_scores, shadow_scores)
# Disagreement analysis
disagreements = shadow_logs_df[~shadow_logs_df["agreed"]]
return {
"n_requests": n_total,
"agreement_rate": agreement_rate,
"champion_mean": champion_scores.mean(),
"shadow_mean": shadow_scores.mean(),
"ks_statistic": ks_stat,
"ks_p_value": ks_p,
"distribution_shift_detected": ks_p < 0.05,
"n_disagreements": len(disagreements),
"shadow_higher_rate": (shadow_scores > champion_scores).mean(),
}
def generate_shadow_report(shadow_logs_df: pd.DataFrame) -> str:
results = analyze_shadow_results(shadow_logs_df)
return f"""
# Shadow Deployment Report
## Overview
- Requests analyzed: {results['n_requests']:,}
- Agreement rate: {results['agreement_rate']:.1%}
## Prediction Distribution
- Champion mean score: {results['champion_mean']:.4f}
- Shadow mean score: {results['shadow_mean']:.4f}
- KS test p-value: {results['ks_p_value']:.4f}
- Distribution shift detected: {'YES ⚠️' if results['distribution_shift_detected'] else 'NO ✓'}
## Divergence Analysis
- Disagreements: {results['n_disagreements']:,}
- Shadow scored higher: {results['shadow_higher_rate']:.1%} of cases
## Recommendation
{'⚠️ Distribution shift detected — investigate before promoting' if results['distribution_shift_detected']
else '✓ Distributions similar — shadow model is a candidate for promotion'}
"""Champion-Challenger Testing
Champion-challenger goes one step further than shadow mode. You actually route a small percentage of real traffic to the challenger model and measure business outcomes — not just predictions.
import random
from enum import Enum
class ModelVariant(Enum):
CHAMPION = "champion"
CHALLENGER = "challenger"
class ChampionChallengerRouter:
def __init__(self, champion, challenger, challenger_traffic_pct=0.05):
self.champion = champion
self.challenger = challenger
self.challenger_pct = challenger_traffic_pct
self.metrics = {}
def predict(self, features: dict, user_id: str) -> tuple[Any, ModelVariant]:
"""Route to champion or challenger; return prediction and variant."""
# Deterministic routing: same user always gets same model
# (important for consistent UX and fair comparison)
use_challenger = self._should_use_challenger(user_id)
if use_challenger:
prediction = self.challenger.predict(features)
variant = ModelVariant.CHALLENGER
else:
prediction = self.champion.predict(features)
variant = ModelVariant.CHAMPION
return prediction, variant
def _should_use_challenger(self, user_id: str) -> bool:
"""Deterministic assignment based on user_id hash."""
import hashlib
hash_val = int(hashlib.md5(user_id.encode()).hexdigest(), 16)
return (hash_val % 100) < (self.challenger_pct * 100)
def record_outcome(self, user_id: str, variant: ModelVariant, converted: bool):
"""Record business outcome (conversion, click, etc.)."""
key = variant.value
if key not in self.metrics:
self.metrics[key] = {"conversions": 0, "total": 0}
self.metrics[key]["total"] += 1
if converted:
self.metrics[key]["conversions"] += 1
def get_conversion_rates(self) -> dict:
"""Compare conversion rates between champion and challenger."""
result = {}
for variant, data in self.metrics.items():
result[variant] = {
"conversion_rate": data["conversions"] / data["total"],
"n": data["total"],
}
return resultStatistical Significance Testing
Don't promote based on raw numbers — test for statistical significance:
from scipy import stats
def test_challenger_improvement(champion_metrics: dict, challenger_metrics: dict) -> dict:
"""Test if challenger is statistically significantly better than champion."""
# Two-proportion z-test
champion_conversions = champion_metrics["conversions"]
champion_total = champion_metrics["total"]
challenger_conversions = challenger_metrics["conversions"]
challenger_total = challenger_metrics["total"]
champion_rate = champion_conversions / champion_total
challenger_rate = challenger_conversions / challenger_total
# Pooled proportion
pooled = (champion_conversions + challenger_conversions) / (champion_total + challenger_total)
z_score = (challenger_rate - champion_rate) / (
(pooled * (1 - pooled) * (1/champion_total + 1/challenger_total)) ** 0.5
)
p_value = 1 - stats.norm.cdf(abs(z_score)) # One-tailed
relative_lift = (challenger_rate - champion_rate) / champion_rate
return {
"champion_rate": champion_rate,
"challenger_rate": challenger_rate,
"relative_lift": relative_lift,
"z_score": z_score,
"p_value": p_value,
"significant": p_value < 0.05,
"recommendation": "PROMOTE" if (p_value < 0.05 and relative_lift > 0) else "KEEP CHAMPION",
}Automated Shadow Testing in CI
Before even deploying to shadow, run automated comparison tests:
# tests/shadow/test_model_compatibility.py
import pytest
import numpy as np
from your_ml_project.models import ChampionModel, ChallengerModel
@pytest.fixture
def production_sample(test_data):
"""Sample of recent production traffic."""
return test_data.sample(n=1000, random_state=42)
def test_output_format_compatible(champion_model, challenger_model, production_sample):
"""Challenger must produce same output format as champion."""
features = production_sample.drop("label", axis=1)
champion_preds = champion_model.predict(features)
challenger_preds = challenger_model.predict(features)
assert champion_preds.shape == challenger_preds.shape
assert champion_preds.dtype == challenger_preds.dtype
def test_prediction_range_compatible(champion_model, challenger_model, production_sample):
"""Challenger predictions must be in same valid range."""
features = production_sample.drop("label", axis=1)
challenger_preds = challenger_model.predict_proba(features)[:, 1]
assert (challenger_preds >= 0).all(), "Challenger produced negative probabilities"
assert (challenger_preds <= 1).all(), "Challenger produced probabilities > 1"
assert not np.any(np.isnan(challenger_preds)), "Challenger produced NaN predictions"
def test_agreement_rate_acceptable(champion_model, challenger_model, production_sample):
"""Champion and challenger should broadly agree on most cases."""
features = production_sample.drop("label", axis=1)
champion_preds = champion_model.predict(features)
challenger_preds = challenger_model.predict(features)
agreement_rate = (champion_preds == challenger_preds).mean()
# 70%+ agreement expected — large divergence is a red flag
assert agreement_rate >= 0.70, \
f"Low agreement rate: {agreement_rate:.1%}. " \
f"Challenger may have a significant behavioral shift."
def test_challenger_faster_or_equivalent(champion_model, challenger_model, production_sample):
"""Challenger must not significantly increase latency."""
import time
features = production_sample.head(100).drop("label", axis=1)
start = time.perf_counter()
for _ in range(10):
champion_model.predict(features)
champion_ms = (time.perf_counter() - start) / 10 * 1000
start = time.perf_counter()
for _ in range(10):
challenger_model.predict(features)
challenger_ms = (time.perf_counter() - start) / 10 * 1000
# Allow up to 50% latency increase
assert challenger_ms <= champion_ms * 1.5, \
f"Challenger is {challenger_ms/champion_ms:.1f}x slower than champion"Promotion Decision Framework
Use this decision tree before promoting a challenger:
def should_promote_challenger(shadow_results, ab_results, safety_checks) -> tuple[bool, str]:
"""Systematic promotion decision."""
# Safety checks (hard gates — any failure blocks promotion)
if not safety_checks["no_errors_in_shadow"]:
return False, "Shadow mode produced errors — investigate before promoting"
if not safety_checks["latency_acceptable"]:
return False, f"Challenger latency {safety_checks['challenger_p99_ms']}ms exceeds budget"
if shadow_results["distribution_shift_detected"]:
return False, "Significant distribution shift detected in shadow — needs investigation"
# Quality checks (soft gates — require evidence of improvement)
if ab_results is None:
return False, "No A/B test results yet — run champion-challenger first"
if not ab_results["significant"]:
return False, f"Not statistically significant (p={ab_results['p_value']:.3f})"
if ab_results["relative_lift"] <= 0:
return False, f"Challenger not better: lift={ab_results['relative_lift']:.1%}"
return True, f"Promote: {ab_results['relative_lift']:.1%} lift (p={ab_results['p_value']:.4f})"Rollback Plan
Always have a rollback before you promote:
class ModelDeploymentManager:
def promote_challenger(self, challenger_version: str):
"""Promote challenger to production with automatic rollback capability."""
previous_version = self.get_production_version()
try:
self._deploy_to_production(challenger_version)
self._run_smoke_tests()
self._monitor_error_rate(duration_seconds=300) # 5-minute soak
logger.info(f"Promoted {challenger_version} to production")
self._archive_version(previous_version)
except Exception as e:
logger.error(f"Promotion failed, rolling back: {e}")
self._deploy_to_production(previous_version)
raise
def _monitor_error_rate(self, duration_seconds: int):
"""Monitor error rate post-promotion and raise if too high."""
import time
end_time = time.time() + duration_seconds
while time.time() < end_time:
error_rate = self.metrics.get_error_rate(last_seconds=60)
if error_rate > 0.01: # >1% error rate triggers rollback
raise RuntimeError(f"Error rate spiked to {error_rate:.1%} post-promotion")
time.sleep(30)Summary
Shadow deployment and champion-challenger testing give you production-quality validation before any user is affected:
- Shadow mode — run both models in parallel, log predictions, compare offline
- Champion-challenger — route small % of real traffic to challenger, measure business outcomes
- Statistical tests — don't promote based on gut feeling; require significance
- Automated pre-shadow tests — catch format and latency issues before shadow deployment
- Rollback readiness — every promotion needs a tested rollback plan
The confidence you gain from shadow data is worth the infrastructure complexity. A model that performs well on a held-out test set can still fail in production — shadow deployment catches those failures before they matter.