Testing SLOs and Error Budgets: Burn Rate Alerts and Compliance Validation
Your SLO says 99.9% availability. Your alert rule says it should fire when the error budget is burning 14x faster than sustainable. But has anyone tested whether that alert rule is correct? Whether the error budget calculation accounts for all failure modes? Whether the burn rate math is right?
SLOs are only as reliable as the tests behind them.
The SLO Testing Problem
SLOs translate business reliability commitments into engineering metrics. They're typically defined as:
SLO: 99.9% of requests succeed over a 30-day rolling window
Error budget: 0.1% × 30 days × 24 hours × 60 min = 43.2 minutes of downtimeThe components that need testing:
- SLI calculation — are you measuring the right thing? Is your success rate formula correct?
- Error budget math — does your error budget calculation match your SLO commitment?
- Burn rate alerts — do alerts fire at the right thresholds? Do they fire too early? Too late?
- SLO compliance reporting — does your compliance report accurately reflect the actual error rate?
- Multi-window alerts — do your Alertmanager rules implement multi-window burn rate correctly?
Testing SLI Formulas
The Service Level Indicator (SLI) is the measurement that backs your SLO. It must correctly represent what "good" means for your users.
import pytest
from your_service.slo import calculate_availability_sli
def test_sli_counts_all_error_codes():
"""SLI should count 5xx as failures, not 4xx"""
mock_metrics = {
"total_requests": 10000,
"5xx_requests": 50, # Server errors = failures
"4xx_requests": 100, # Client errors = NOT failures (user error, not our outage)
"timeout_requests": 25, # Timeouts = failures
}
sli = calculate_availability_sli(mock_metrics)
# SLI = (total - failures) / total = (10000 - 75) / 10000 = 0.9925
expected_sli = (10000 - 75) / 10000 # 50 5xx + 25 timeouts, NOT 4xx
assert abs(sli - expected_sli) < 0.0001
def test_sli_handles_zero_requests():
"""SLI with no requests should not cause division by zero"""
mock_metrics = {"total_requests": 0, "5xx_requests": 0}
# Should return 1.0 (no failures) or handle gracefully
sli = calculate_availability_sli(mock_metrics)
assert sli == 1.0 or sli is None
def test_sli_excludes_planned_maintenance():
"""Requests during planned maintenance windows should not count against SLI"""
requests_with_maintenance = [
{"timestamp": "2026-05-01T02:00:00Z", "status": 503, "maintenance": True},
{"timestamp": "2026-05-01T10:00:00Z", "status": 500, "maintenance": False},
]
sli = calculate_availability_sli_from_events(requests_with_maintenance)
# Only the non-maintenance 500 should count
# Total non-maintenance requests = 1, failures = 1, SLI = 0.0
assert sli == 0.0Testing Error Budget Calculations
class ErrorBudgetCalculator:
def __init__(self, slo_target: float, window_days: int):
self.slo_target = slo_target
self.window_days = window_days
self.total_minutes = window_days * 24 * 60
def error_budget_minutes(self) -> float:
"""Total error budget in minutes"""
return (1 - self.slo_target) * self.total_minutes
def remaining_budget_minutes(self, current_sli: float) -> float:
"""Remaining error budget given current SLI"""
actual_error_rate = 1 - current_sli
target_error_rate = 1 - self.slo_target
excess_error_rate = actual_error_rate - target_error_rate
if excess_error_rate <= 0:
return self.error_budget_minutes()
budget_consumed_minutes = excess_error_rate * self.total_minutes
return self.error_budget_minutes() - budget_consumed_minutes
def budget_consumed_percent(self, current_sli: float) -> float:
budget = self.error_budget_minutes()
remaining = self.remaining_budget_minutes(current_sli)
return (1 - remaining / budget) * 100
def test_error_budget_calculation():
calc = ErrorBudgetCalculator(slo_target=0.999, window_days=30)
# 99.9% SLO over 30 days = 43.2 minutes budget
assert abs(calc.error_budget_minutes() - 43.2) < 0.01
# Perfect reliability = full budget remaining
assert calc.remaining_budget_minutes(1.0) == calc.error_budget_minutes()
# Exactly at SLO = full budget remaining
assert calc.remaining_budget_minutes(0.999) == calc.error_budget_minutes()
# 0.1% worse than SLO over 30 days = 43.2 minutes consumed
# Total error rate = 0.2%, SLO allows 0.1%, excess = 0.1%
# 0.001 × 43200 min = 43.2 min consumed
remaining = calc.remaining_budget_minutes(0.998)
assert abs(remaining) < 1.0 # Approximately 0 remaining
def test_budget_consumed_percentage():
calc = ErrorBudgetCalculator(slo_target=0.999, window_days=30)
# Perfect reliability
assert calc.budget_consumed_percent(1.0) == 0.0
# Exactly at SLO
assert abs(calc.budget_consumed_percent(0.999)) < 0.01
# 50% budget consumed: error rate = 0.1% + 0.05% = 0.15%
# SLI = 0.9985
assert abs(calc.budget_consumed_percent(0.9985) - 50.0) < 1.0Testing Burn Rate Alerts
The Google SRE Workbook defines multi-window burn rate alerts. Test that your Prometheus rules implement them correctly.
Burn Rate Math Tests
def calculate_burn_rate(current_error_rate: float, slo_target: float) -> float:
"""
Burn rate = current error rate / allowed error rate
A burn rate of 1 = consuming budget at exactly the sustainable rate
A burn rate of 14 = consuming budget 14x faster than sustainable
"""
allowed_error_rate = 1 - slo_target
if allowed_error_rate == 0:
return float('inf')
return current_error_rate / allowed_error_rate
def test_burn_rate_at_slo_boundary():
"""Burn rate of 1 means consuming budget at exactly the sustainable rate"""
# At exactly the SLO target, burn rate = 1
burn_rate = calculate_burn_rate(0.001, 0.999)
assert abs(burn_rate - 1.0) < 0.001
def test_high_burn_rate():
"""14x burn rate consumes monthly budget in ~2 days"""
# Error rate 14x above allowed
burn_rate = calculate_burn_rate(0.014, 0.999) # 1.4% error rate
assert abs(burn_rate - 14.0) < 0.01
# At 14x burn rate, 30-day budget consumed in: 30 days / 14 = ~2.1 days
time_to_exhaustion_days = 30 / burn_rate
assert abs(time_to_exhaustion_days - 2.14) < 0.1
def test_burn_rate_alert_thresholds():
"""Verify alert thresholds match Google SRE Workbook recommendations"""
slo_target = 0.999
# Page immediately: 14.4x burn rate (consumes monthly budget in 2 days)
page_threshold_burn_rate = 14.4
page_threshold_error_rate = page_threshold_burn_rate * (1 - slo_target)
assert abs(page_threshold_error_rate - 0.01440) < 0.0001
# Ticket: 6x burn rate (consumes monthly budget in 5 days)
ticket_threshold_burn_rate = 6.0
ticket_threshold_error_rate = ticket_threshold_burn_rate * (1 - slo_target)
assert abs(ticket_threshold_error_rate - 0.006) < 0.0001Prometheus Alert Rule Testing
Test your Prometheus alerting rules with promtool:
# alert-rules.yaml
groups:
- name: slo
rules:
- alert: ErrorBudgetBurnRateHigh
expr: |
(
sum(rate(http_requests_total{status=~"5.."}[1h])) /
sum(rate(http_requests_total[1h]))
) > (14.4 * 0.001)
and
(
sum(rate(http_requests_total{status=~"5.."}[5m])) /
sum(rate(http_requests_total[5m]))
) > (14.4 * 0.001)
for: 2m
labels:
severity: critical
slo: api_availability
annotations:
summary: "Error budget burning at >14x rate"
burn_rate: "{{ printf \"%.1f\" $value }}"# Run alert rule unit tests
promtool test rules slo-alert-tests.yaml# slo-alert-tests.yaml
rule_files:
- alert-rules.yaml
tests:
- interval: 1m
input_series:
- series: 'http_requests_total{status="200"}'
values: "1000+1000x60" # 1000 req/min, growing
- series: 'http_requests_total{status="500"}'
values: "0+15x60" # 15 errors/min = 1.5% error rate = 15x burn rate
alert_rule_test:
- eval_time: 5m
alertname: ErrorBudgetBurnRateHigh
exp_alerts:
- exp_labels:
severity: critical
slo: api_availability
exp_annotations:
summary: "Error budget burning at >14x rate"
- interval: 1m
input_series:
- series: 'http_requests_total{status="200"}'
values: "1000+1000x60"
- series: 'http_requests_total{status="500"}'
values: "0+5x60" # 5 errors/min = 0.5% error rate = 5x burn rate (below threshold)
alert_rule_test:
- eval_time: 10m
alertname: ErrorBudgetBurnRateHigh
exp_alerts: [] # Should NOT fire at 5x burn rateSLO Compliance Reporting Tests
Your monthly SLO compliance report needs to be accurate.
from datetime import datetime, timedelta
import pytest
def test_slo_compliance_report_accuracy():
"""Verify compliance report matches manually calculated SLO"""
# Create a controlled set of events
window_start = datetime(2026, 5, 1)
window_end = datetime(2026, 5, 31, 23, 59, 59)
# Insert known good and bad events
test_events = generate_test_events(
total=1_000_000,
error_count=500, # 0.05% error rate = above 99.9% SLO
window=(window_start, window_end)
)
report = generate_slo_report(
service="test-service",
window_start=window_start,
window_end=window_end,
slo_target=0.999
)
expected_sli = (1_000_000 - 500) / 1_000_000 # = 0.9995
assert abs(report.sli - expected_sli) < 0.00001
assert report.is_compliant == True # 0.9995 > 0.999
# Verify error budget consumption
expected_budget_consumed_pct = (0.0005 / 0.001) * 100 # 50%
assert abs(report.budget_consumed_percent - expected_budget_consumed_pct) < 1.0
def test_slo_report_accounts_for_excluded_events():
"""Maintenance windows should be excluded from SLO calculation"""
maintenance_window = {
"start": datetime(2026, 5, 15, 2, 0),
"end": datetime(2026, 5, 15, 4, 0),
"reason": "Planned database migration"
}
# 10,000 errors during maintenance — should not count
events_during_maintenance = generate_test_events(
total=50000, error_count=10000,
window=(maintenance_window["start"], maintenance_window["end"])
)
# 0 errors outside maintenance
events_outside_maintenance = generate_test_events(
total=1_000_000, error_count=0,
window=(datetime(2026, 5, 1), datetime(2026, 5, 31))
)
report = generate_slo_report(
service="test-service",
window_start=datetime(2026, 5, 1),
window_end=datetime(2026, 5, 31),
slo_target=0.999,
maintenance_windows=[maintenance_window]
)
# Excluding maintenance errors, SLI should be 1.0
assert report.sli == 1.0
assert report.is_compliant == TrueAutomated SLO Dashboard Testing
Your Grafana dashboards display SLO metrics. Test that the queries are correct:
import requests
GRAFANA_URL = "http://localhost:3000"
def test_slo_dashboard_queries():
"""Verify SLO dashboard queries return valid data"""
# Get dashboard JSON
response = requests.get(
f"{GRAFANA_URL}/api/dashboards/uid/slo-dashboard",
auth=("admin", "admin")
)
dashboard = response.json()["dashboard"]
# Extract and validate all Prometheus queries
for panel in dashboard["panels"]:
for target in panel.get("targets", []):
if target.get("datasource", {}).get("type") == "prometheus":
query = target.get("expr", "")
# Test query against Prometheus
test_response = requests.get(
"http://localhost:9090/api/v1/query",
params={"query": query}
)
data = test_response.json()
assert data["status"] == "success", \
f"Dashboard query failed: {query}\nError: {data}"HelpMeTest provides continuous monitoring that feeds your SLO calculations — persistent health checks that measure availability, track response times, and give you the raw data you need to compute accurate SLIs and error budgets. Start free.