Testing SLOs and Error Budgets: Burn Rate Alerts and Compliance Validation

Testing SLOs and Error Budgets: Burn Rate Alerts and Compliance Validation

Your SLO says 99.9% availability. Your alert rule says it should fire when the error budget is burning 14x faster than sustainable. But has anyone tested whether that alert rule is correct? Whether the error budget calculation accounts for all failure modes? Whether the burn rate math is right?

SLOs are only as reliable as the tests behind them.

The SLO Testing Problem

SLOs translate business reliability commitments into engineering metrics. They're typically defined as:

SLO: 99.9% of requests succeed over a 30-day rolling window
Error budget: 0.1% × 30 days × 24 hours × 60 min = 43.2 minutes of downtime

The components that need testing:

  1. SLI calculation — are you measuring the right thing? Is your success rate formula correct?
  2. Error budget math — does your error budget calculation match your SLO commitment?
  3. Burn rate alerts — do alerts fire at the right thresholds? Do they fire too early? Too late?
  4. SLO compliance reporting — does your compliance report accurately reflect the actual error rate?
  5. Multi-window alerts — do your Alertmanager rules implement multi-window burn rate correctly?

Testing SLI Formulas

The Service Level Indicator (SLI) is the measurement that backs your SLO. It must correctly represent what "good" means for your users.

import pytest
from your_service.slo import calculate_availability_sli

def test_sli_counts_all_error_codes():
    """SLI should count 5xx as failures, not 4xx"""
    mock_metrics = {
        "total_requests": 10000,
        "5xx_requests": 50,     # Server errors = failures
        "4xx_requests": 100,    # Client errors = NOT failures (user error, not our outage)
        "timeout_requests": 25, # Timeouts = failures
    }
    
    sli = calculate_availability_sli(mock_metrics)
    
    # SLI = (total - failures) / total = (10000 - 75) / 10000 = 0.9925
    expected_sli = (10000 - 75) / 10000  # 50 5xx + 25 timeouts, NOT 4xx
    assert abs(sli - expected_sli) < 0.0001

def test_sli_handles_zero_requests():
    """SLI with no requests should not cause division by zero"""
    mock_metrics = {"total_requests": 0, "5xx_requests": 0}
    
    # Should return 1.0 (no failures) or handle gracefully
    sli = calculate_availability_sli(mock_metrics)
    assert sli == 1.0 or sli is None

def test_sli_excludes_planned_maintenance():
    """Requests during planned maintenance windows should not count against SLI"""
    requests_with_maintenance = [
        {"timestamp": "2026-05-01T02:00:00Z", "status": 503, "maintenance": True},
        {"timestamp": "2026-05-01T10:00:00Z", "status": 500, "maintenance": False},
    ]
    
    sli = calculate_availability_sli_from_events(requests_with_maintenance)
    
    # Only the non-maintenance 500 should count
    # Total non-maintenance requests = 1, failures = 1, SLI = 0.0
    assert sli == 0.0

Testing Error Budget Calculations

class ErrorBudgetCalculator:
    def __init__(self, slo_target: float, window_days: int):
        self.slo_target = slo_target
        self.window_days = window_days
        self.total_minutes = window_days * 24 * 60
    
    def error_budget_minutes(self) -> float:
        """Total error budget in minutes"""
        return (1 - self.slo_target) * self.total_minutes
    
    def remaining_budget_minutes(self, current_sli: float) -> float:
        """Remaining error budget given current SLI"""
        actual_error_rate = 1 - current_sli
        target_error_rate = 1 - self.slo_target
        excess_error_rate = actual_error_rate - target_error_rate
        
        if excess_error_rate <= 0:
            return self.error_budget_minutes()
        
        budget_consumed_minutes = excess_error_rate * self.total_minutes
        return self.error_budget_minutes() - budget_consumed_minutes
    
    def budget_consumed_percent(self, current_sli: float) -> float:
        budget = self.error_budget_minutes()
        remaining = self.remaining_budget_minutes(current_sli)
        return (1 - remaining / budget) * 100

def test_error_budget_calculation():
    calc = ErrorBudgetCalculator(slo_target=0.999, window_days=30)
    
    # 99.9% SLO over 30 days = 43.2 minutes budget
    assert abs(calc.error_budget_minutes() - 43.2) < 0.01
    
    # Perfect reliability = full budget remaining
    assert calc.remaining_budget_minutes(1.0) == calc.error_budget_minutes()
    
    # Exactly at SLO = full budget remaining
    assert calc.remaining_budget_minutes(0.999) == calc.error_budget_minutes()
    
    # 0.1% worse than SLO over 30 days = 43.2 minutes consumed
    # Total error rate = 0.2%, SLO allows 0.1%, excess = 0.1%
    # 0.001 × 43200 min = 43.2 min consumed
    remaining = calc.remaining_budget_minutes(0.998)
    assert abs(remaining) < 1.0  # Approximately 0 remaining

def test_budget_consumed_percentage():
    calc = ErrorBudgetCalculator(slo_target=0.999, window_days=30)
    
    # Perfect reliability
    assert calc.budget_consumed_percent(1.0) == 0.0
    
    # Exactly at SLO
    assert abs(calc.budget_consumed_percent(0.999)) < 0.01
    
    # 50% budget consumed: error rate = 0.1% + 0.05% = 0.15%
    # SLI = 0.9985
    assert abs(calc.budget_consumed_percent(0.9985) - 50.0) < 1.0

Testing Burn Rate Alerts

The Google SRE Workbook defines multi-window burn rate alerts. Test that your Prometheus rules implement them correctly.

Burn Rate Math Tests

def calculate_burn_rate(current_error_rate: float, slo_target: float) -> float:
    """
    Burn rate = current error rate / allowed error rate
    A burn rate of 1 = consuming budget at exactly the sustainable rate
    A burn rate of 14 = consuming budget 14x faster than sustainable
    """
    allowed_error_rate = 1 - slo_target
    if allowed_error_rate == 0:
        return float('inf')
    return current_error_rate / allowed_error_rate

def test_burn_rate_at_slo_boundary():
    """Burn rate of 1 means consuming budget at exactly the sustainable rate"""
    # At exactly the SLO target, burn rate = 1
    burn_rate = calculate_burn_rate(0.001, 0.999)
    assert abs(burn_rate - 1.0) < 0.001

def test_high_burn_rate():
    """14x burn rate consumes monthly budget in ~2 days"""
    # Error rate 14x above allowed
    burn_rate = calculate_burn_rate(0.014, 0.999)  # 1.4% error rate
    assert abs(burn_rate - 14.0) < 0.01
    
    # At 14x burn rate, 30-day budget consumed in: 30 days / 14 = ~2.1 days
    time_to_exhaustion_days = 30 / burn_rate
    assert abs(time_to_exhaustion_days - 2.14) < 0.1

def test_burn_rate_alert_thresholds():
    """Verify alert thresholds match Google SRE Workbook recommendations"""
    slo_target = 0.999
    
    # Page immediately: 14.4x burn rate (consumes monthly budget in 2 days)
    page_threshold_burn_rate = 14.4
    page_threshold_error_rate = page_threshold_burn_rate * (1 - slo_target)
    
    assert abs(page_threshold_error_rate - 0.01440) < 0.0001
    
    # Ticket: 6x burn rate (consumes monthly budget in 5 days)
    ticket_threshold_burn_rate = 6.0
    ticket_threshold_error_rate = ticket_threshold_burn_rate * (1 - slo_target)
    
    assert abs(ticket_threshold_error_rate - 0.006) < 0.0001

Prometheus Alert Rule Testing

Test your Prometheus alerting rules with promtool:

# alert-rules.yaml
groups:
  - name: slo
    rules:
      - alert: ErrorBudgetBurnRateHigh
        expr: |
          (
            sum(rate(http_requests_total{status=~"5.."}[1h])) /
            sum(rate(http_requests_total[1h]))
          ) > (14.4 * 0.001)
          and
          (
            sum(rate(http_requests_total{status=~"5.."}[5m])) /
            sum(rate(http_requests_total[5m]))
          ) > (14.4 * 0.001)
        for: 2m
        labels:
          severity: critical
          slo: api_availability
        annotations:
          summary: "Error budget burning at >14x rate"
          burn_rate: "{{ printf \"%.1f\" $value }}"
# Run alert rule unit tests
promtool test rules slo-alert-tests.yaml
# slo-alert-tests.yaml
rule_files:
  - alert-rules.yaml

tests:
  - interval: 1m
    input_series:
      - series: 'http_requests_total{status="200"}'
        values: "1000+1000x60"  # 1000 req/min, growing
      - series: 'http_requests_total{status="500"}'
        values: "0+15x60"       # 15 errors/min = 1.5% error rate = 15x burn rate
    
    alert_rule_test:
      - eval_time: 5m
        alertname: ErrorBudgetBurnRateHigh
        exp_alerts:
          - exp_labels:
              severity: critical
              slo: api_availability
            exp_annotations:
              summary: "Error budget burning at >14x rate"
  
  - interval: 1m
    input_series:
      - series: 'http_requests_total{status="200"}'
        values: "1000+1000x60"
      - series: 'http_requests_total{status="500"}'
        values: "0+5x60"  # 5 errors/min = 0.5% error rate = 5x burn rate (below threshold)
    
    alert_rule_test:
      - eval_time: 10m
        alertname: ErrorBudgetBurnRateHigh
        exp_alerts: []  # Should NOT fire at 5x burn rate

SLO Compliance Reporting Tests

Your monthly SLO compliance report needs to be accurate.

from datetime import datetime, timedelta
import pytest

def test_slo_compliance_report_accuracy():
    """Verify compliance report matches manually calculated SLO"""
    
    # Create a controlled set of events
    window_start = datetime(2026, 5, 1)
    window_end = datetime(2026, 5, 31, 23, 59, 59)
    
    # Insert known good and bad events
    test_events = generate_test_events(
        total=1_000_000,
        error_count=500,  # 0.05% error rate = above 99.9% SLO
        window=(window_start, window_end)
    )
    
    report = generate_slo_report(
        service="test-service",
        window_start=window_start,
        window_end=window_end,
        slo_target=0.999
    )
    
    expected_sli = (1_000_000 - 500) / 1_000_000  # = 0.9995
    
    assert abs(report.sli - expected_sli) < 0.00001
    assert report.is_compliant == True  # 0.9995 > 0.999
    
    # Verify error budget consumption
    expected_budget_consumed_pct = (0.0005 / 0.001) * 100  # 50%
    assert abs(report.budget_consumed_percent - expected_budget_consumed_pct) < 1.0

def test_slo_report_accounts_for_excluded_events():
    """Maintenance windows should be excluded from SLO calculation"""
    maintenance_window = {
        "start": datetime(2026, 5, 15, 2, 0),
        "end": datetime(2026, 5, 15, 4, 0),
        "reason": "Planned database migration"
    }
    
    # 10,000 errors during maintenance — should not count
    events_during_maintenance = generate_test_events(
        total=50000, error_count=10000,
        window=(maintenance_window["start"], maintenance_window["end"])
    )
    
    # 0 errors outside maintenance
    events_outside_maintenance = generate_test_events(
        total=1_000_000, error_count=0,
        window=(datetime(2026, 5, 1), datetime(2026, 5, 31))
    )
    
    report = generate_slo_report(
        service="test-service",
        window_start=datetime(2026, 5, 1),
        window_end=datetime(2026, 5, 31),
        slo_target=0.999,
        maintenance_windows=[maintenance_window]
    )
    
    # Excluding maintenance errors, SLI should be 1.0
    assert report.sli == 1.0
    assert report.is_compliant == True

Automated SLO Dashboard Testing

Your Grafana dashboards display SLO metrics. Test that the queries are correct:

import requests

GRAFANA_URL = "http://localhost:3000"

def test_slo_dashboard_queries():
    """Verify SLO dashboard queries return valid data"""
    
    # Get dashboard JSON
    response = requests.get(
        f"{GRAFANA_URL}/api/dashboards/uid/slo-dashboard",
        auth=("admin", "admin")
    )
    dashboard = response.json()["dashboard"]
    
    # Extract and validate all Prometheus queries
    for panel in dashboard["panels"]:
        for target in panel.get("targets", []):
            if target.get("datasource", {}).get("type") == "prometheus":
                query = target.get("expr", "")
                
                # Test query against Prometheus
                test_response = requests.get(
                    "http://localhost:9090/api/v1/query",
                    params={"query": query}
                )
                
                data = test_response.json()
                assert data["status"] == "success", \
                    f"Dashboard query failed: {query}\nError: {data}"

HelpMeTest provides continuous monitoring that feeds your SLO calculations — persistent health checks that measure availability, track response times, and give you the raw data you need to compute accurate SLIs and error budgets. Start free.

Read more

Start now free