slo-guard 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,26 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+ branches: [main]
8
+
9
+ jobs:
10
+ test:
11
+ runs-on: ubuntu-latest
12
+ strategy:
13
+ matrix:
14
+ python-version: ["3.9", "3.10", "3.11", "3.12"]
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - name: Set up Python ${{ matrix.python-version }}
18
+ uses: actions/setup-python@v5
19
+ with:
20
+ python-version: ${{ matrix.python-version }}
21
+ - name: Install dependencies
22
+ run: pip install -e ".[dev]"
23
+ - name: Lint
24
+ run: ruff check src tests
25
+ - name: Test
26
+ run: pytest --cov=slo_guard --cov-report=term-missing
@@ -0,0 +1,24 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ jobs:
8
+ publish:
9
+ runs-on: ubuntu-latest
10
+ environment: pypi
11
+ permissions:
12
+ id-token: write # required for PyPI trusted publishing
13
+ steps:
14
+ - uses: actions/checkout@v4
15
+ - name: Set up Python
16
+ uses: actions/setup-python@v5
17
+ with:
18
+ python-version: "3.12"
19
+ - name: Install build tool
20
+ run: pip install build
21
+ - name: Build sdist and wheel
22
+ run: python -m build
23
+ - name: Publish to PyPI
24
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,13 @@
1
+ __pycache__/
2
+ *.pyc
3
+ *.egg-info/
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ build/
7
+ dist/
8
+ .coverage
9
+ htmlcov/
10
+ .venv/
11
+ venv/
12
+ *.egg
13
+ .DS_Store
@@ -0,0 +1,14 @@
1
+ # Changelog
2
+
3
+ ## [0.1.0] - 2026-08-29
4
+
5
+ ### Added
6
+ - Core `SLO` model with error-budget and budget-remaining calculations.
7
+ - Multi-window, multi-burn-rate policy engine (`evaluate_policy`), generalized
8
+ so it reproduces the standard Google SRE Workbook constants (14.4 / 6 / 1)
9
+ for a 99.9%/30-day SLO and scales correctly for other targets/periods.
10
+ - Prometheus alerting rule generator (`build_alert_group`, `render_yaml`).
11
+ - YAML-based SLO config loader.
12
+ - CLI (`slo-guard rules`, `slo-guard budget`).
13
+ - Full test suite (pytest) covering core math, burn-rate policy, rule
14
+ generation, and config loading.
@@ -0,0 +1,42 @@
1
+ # Contributing to slo-guard
2
+
3
+ Thanks for considering a contribution! This project is small and focused,
4
+ so contributions of any size are welcome — bug reports, docs fixes,
5
+ new burn-rate policy presets, additional rule-generator backends
6
+ (e.g. Datadog, Grafana Alerting), etc.
7
+
8
+ ## Setup
9
+
10
+ ```bash
11
+ git clone https://github.com/itsmejoshi/slo-guard
12
+ cd slo-guard
13
+ pip install -e ".[dev]"
14
+ ```
15
+
16
+ ## Running tests
17
+
18
+ ```bash
19
+ pytest --cov=slo_guard --cov-report=term-missing
20
+ ```
21
+
22
+ ## Linting
23
+
24
+ ```bash
25
+ ruff check src tests
26
+ ```
27
+
28
+ ## Submitting changes
29
+
30
+ 1. Fork the repo and create a branch off `main`.
31
+ 2. Add tests for any new behavior — PRs without tests for new logic will be
32
+ asked to add them.
33
+ 3. Make sure `pytest` and `ruff check` pass locally.
34
+ 4. Open a PR with a clear description of the change and why it's needed.
35
+
36
+ ## Design principles
37
+
38
+ - Keep the core math (`core.py`, `burnrate.py`) dependency-free and provable
39
+ from first principles — no hard-coded magic numbers.
40
+ - Keep the library usable both as a CLI and as an importable Python API.
41
+ - Prefer clarity over cleverness; this is infrastructure tooling that people
42
+ will read during an incident.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sai Joshitha Kathari
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,123 @@
1
+ Metadata-Version: 2.5
2
+ Name: slo-guard
3
+ Version: 0.1.0
4
+ Summary: SLO error-budget tracking and multi-window burn-rate alerting for SRE/DevOps teams
5
+ Project-URL: Homepage, https://github.com/itsmejoshi/slo-guard
6
+ Project-URL: Repository, https://github.com/itsmejoshi/slo-guard
7
+ Project-URL: Issues, https://github.com/itsmejoshi/slo-guard/issues
8
+ Author-email: Sai Joshitha Kathari <kathari.saijoshitha@gmail.com>
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: devops,error-budget,monitoring,observability,prometheus,slo,sre
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: System Administrators
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.9
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: System :: Monitoring
22
+ Classifier: Topic :: System :: Systems Administration
23
+ Requires-Python: >=3.9
24
+ Requires-Dist: pyyaml>=6.0
25
+ Provides-Extra: dev
26
+ Requires-Dist: pytest-cov>=4.0; extra == 'dev'
27
+ Requires-Dist: pytest>=7.0; extra == 'dev'
28
+ Requires-Dist: ruff>=0.4; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ # slo-guard
32
+
33
+ Error-budget tracking and **multi-window, multi-burn-rate alerting** for SRE/DevOps teams, built around the technique described in Google's *SRE Workbook* ("Alerting on SLOs").
34
+
35
+ Instead of alerting the moment error rate ticks up, `slo-guard` computes how fast a service is burning its error budget relative to a uniform, budget-neutral pace — and generates ready-to-use Prometheus alerting rules from a single SLO definition.
36
+
37
+ ## Why burn-rate alerting?
38
+
39
+ A raw "error rate > X%" alert either fires too late (averaged over a long window) or too often (noisy on a short one). Burn-rate alerting fixes this by pairing a **long window** (to size the alert against your actual error budget) with a **short window** (to confirm the problem hasn't already resolved), at multiple severities:
40
+
41
+ | Window pair | Budget consumed | Meaning | Typical severity |
42
+ |--------------------|-----------------|-----------------------------------|-------------------|
43
+ | 1h / 5m | 2% in 1 hour | Fast, severe burn | Page |
44
+ | 6h / 30m | 5% in 6 hours | Sustained, moderate burn | Page |
45
+ | 3d / 6h | 10% in 3 days | Slow leak worth investigating | Ticket |
46
+
47
+ The exact multipliers are derived from your SLO's target and period, not hard-coded — a 99.9%/30-day SLO reproduces the well-known 14.4 / 6 / 1 burn-rate constants automatically; other targets and periods scale correctly.
48
+
49
+ ## Install
50
+
51
+ ```bash
52
+ pip install slo-guard
53
+ ```
54
+
55
+ ## Quick start
56
+
57
+ **1. Define your SLO** (`slo.yaml`):
58
+
59
+ ```yaml
60
+ name: checkout-availability
61
+ target: 0.999
62
+ period_days: 30
63
+ error_selector: 'http_requests_total{job="checkout",code=~"5.."}'
64
+ total_selector: 'http_requests_total{job="checkout"}'
65
+ ```
66
+
67
+ **2. Generate Prometheus alerting rules:**
68
+
69
+ ```bash
70
+ slo-guard rules --config slo.yaml --out checkout-rules.yml
71
+ ```
72
+
73
+ **3. Check current error-budget status:**
74
+
75
+ ```bash
76
+ slo-guard budget --config slo.yaml --bad-ratio 0.0015
77
+ ```
78
+
79
+ ```
80
+ SLO: checkout-availability (target=99.9000%, period=30d)
81
+ Observed bad-event ratio: 0.1500%
82
+ Error budget consumed: 150.00%
83
+ Error budget remaining: -50.00%
84
+ Remaining budget (time): -21600.0 minutes
85
+ STATUS: SLO VIOLATED -- error budget exhausted.
86
+ ```
87
+
88
+ **4. Or use it as a library:**
89
+
90
+ ```python
91
+ from slo_guard import SLO, evaluate_policy
92
+
93
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
94
+ print(slo.budget_consumed(bad_event_ratio=0.0005)) # 0.5 (50% of budget used)
95
+
96
+ for window in evaluate_policy(slo):
97
+ print(window["name"], window["burn_rate_threshold"])
98
+ ```
99
+
100
+ ## Development
101
+
102
+ ```bash
103
+ git clone https://github.com/itsmejoshi/slo-guard
104
+ cd slo-guard
105
+ pip install -e ".[dev]"
106
+ pytest
107
+ ```
108
+
109
+ ## Publishing (maintainer notes)
110
+
111
+ ```bash
112
+ python -m build
113
+ twine upload dist/*
114
+ ```
115
+
116
+ ## License
117
+
118
+ MIT — see [LICENSE](LICENSE).
119
+
120
+ ## Related reading
121
+
122
+ - Google SRE Workbook, ["Alerting on SLOs"](https://sre.google/workbook/alerting-on-slo/)
123
+ - Similar tools in this space: [Pyrra](https://github.com/pyrra-dev/pyrra), [Sloth](https://github.com/slok/sloth) — `slo-guard` focuses on being a minimal, dependency-light library you can embed directly in Python tooling or CI, rather than a standalone operator.
@@ -0,0 +1,93 @@
1
+ # slo-guard
2
+
3
+ Error-budget tracking and **multi-window, multi-burn-rate alerting** for SRE/DevOps teams, built around the technique described in Google's *SRE Workbook* ("Alerting on SLOs").
4
+
5
+ Instead of alerting the moment error rate ticks up, `slo-guard` computes how fast a service is burning its error budget relative to a uniform, budget-neutral pace — and generates ready-to-use Prometheus alerting rules from a single SLO definition.
6
+
7
+ ## Why burn-rate alerting?
8
+
9
+ A raw "error rate > X%" alert either fires too late (averaged over a long window) or too often (noisy on a short one). Burn-rate alerting fixes this by pairing a **long window** (to size the alert against your actual error budget) with a **short window** (to confirm the problem hasn't already resolved), at multiple severities:
10
+
11
+ | Window pair | Budget consumed | Meaning | Typical severity |
12
+ |--------------------|-----------------|-----------------------------------|-------------------|
13
+ | 1h / 5m | 2% in 1 hour | Fast, severe burn | Page |
14
+ | 6h / 30m | 5% in 6 hours | Sustained, moderate burn | Page |
15
+ | 3d / 6h | 10% in 3 days | Slow leak worth investigating | Ticket |
16
+
17
+ The exact multipliers are derived from your SLO's target and period, not hard-coded — a 99.9%/30-day SLO reproduces the well-known 14.4 / 6 / 1 burn-rate constants automatically; other targets and periods scale correctly.
18
+
19
+ ## Install
20
+
21
+ ```bash
22
+ pip install slo-guard
23
+ ```
24
+
25
+ ## Quick start
26
+
27
+ **1. Define your SLO** (`slo.yaml`):
28
+
29
+ ```yaml
30
+ name: checkout-availability
31
+ target: 0.999
32
+ period_days: 30
33
+ error_selector: 'http_requests_total{job="checkout",code=~"5.."}'
34
+ total_selector: 'http_requests_total{job="checkout"}'
35
+ ```
36
+
37
+ **2. Generate Prometheus alerting rules:**
38
+
39
+ ```bash
40
+ slo-guard rules --config slo.yaml --out checkout-rules.yml
41
+ ```
42
+
43
+ **3. Check current error-budget status:**
44
+
45
+ ```bash
46
+ slo-guard budget --config slo.yaml --bad-ratio 0.0015
47
+ ```
48
+
49
+ ```
50
+ SLO: checkout-availability (target=99.9000%, period=30d)
51
+ Observed bad-event ratio: 0.1500%
52
+ Error budget consumed: 150.00%
53
+ Error budget remaining: -50.00%
54
+ Remaining budget (time): -21600.0 minutes
55
+ STATUS: SLO VIOLATED -- error budget exhausted.
56
+ ```
57
+
58
+ **4. Or use it as a library:**
59
+
60
+ ```python
61
+ from slo_guard import SLO, evaluate_policy
62
+
63
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
64
+ print(slo.budget_consumed(bad_event_ratio=0.0005)) # 0.5 (50% of budget used)
65
+
66
+ for window in evaluate_policy(slo):
67
+ print(window["name"], window["burn_rate_threshold"])
68
+ ```
69
+
70
+ ## Development
71
+
72
+ ```bash
73
+ git clone https://github.com/itsmejoshi/slo-guard
74
+ cd slo-guard
75
+ pip install -e ".[dev]"
76
+ pytest
77
+ ```
78
+
79
+ ## Publishing (maintainer notes)
80
+
81
+ ```bash
82
+ python -m build
83
+ twine upload dist/*
84
+ ```
85
+
86
+ ## License
87
+
88
+ MIT — see [LICENSE](LICENSE).
89
+
90
+ ## Related reading
91
+
92
+ - Google SRE Workbook, ["Alerting on SLOs"](https://sre.google/workbook/alerting-on-slo/)
93
+ - Similar tools in this space: [Pyrra](https://github.com/pyrra-dev/pyrra), [Sloth](https://github.com/slok/sloth) — `slo-guard` focuses on being a minimal, dependency-light library you can embed directly in Python tooling or CI, rather than a standalone operator.
@@ -0,0 +1,10 @@
1
+ # Example SLO definition for slo-guard.
2
+ # Copy this file, rename it, and adjust the selectors to match your metrics.
3
+
4
+ name: checkout-availability
5
+ target: 0.999 # 99.9% of requests must succeed
6
+ period_days: 30 # rolling 30-day compliance window
7
+
8
+ # PromQL selectors used to compute the bad-event ratio.
9
+ error_selector: 'http_requests_total{job="checkout",code=~"5.."}'
10
+ total_selector: 'http_requests_total{job="checkout"}'
@@ -0,0 +1,55 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "slo-guard"
7
+ version = "0.1.0"
8
+ description = "SLO error-budget tracking and multi-window burn-rate alerting for SRE/DevOps teams"
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ requires-python = ">=3.9"
12
+ authors = [
13
+ { name = "Sai Joshitha Kathari", email = "kathari.saijoshitha@gmail.com" }
14
+ ]
15
+ keywords = ["sre", "slo", "observability", "prometheus", "monitoring", "devops", "error-budget"]
16
+ classifiers = [
17
+ "Development Status :: 4 - Beta",
18
+ "Intended Audience :: Developers",
19
+ "Intended Audience :: System Administrators",
20
+ "License :: OSI Approved :: MIT License",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3.9",
23
+ "Programming Language :: Python :: 3.10",
24
+ "Programming Language :: Python :: 3.11",
25
+ "Programming Language :: Python :: 3.12",
26
+ "Topic :: System :: Monitoring",
27
+ "Topic :: System :: Systems Administration",
28
+ ]
29
+ dependencies = [
30
+ "PyYAML>=6.0",
31
+ ]
32
+
33
+ [project.optional-dependencies]
34
+ dev = [
35
+ "pytest>=7.0",
36
+ "pytest-cov>=4.0",
37
+ "ruff>=0.4",
38
+ ]
39
+
40
+ [project.urls]
41
+ Homepage = "https://github.com/itsmejoshi/slo-guard"
42
+ Repository = "https://github.com/itsmejoshi/slo-guard"
43
+ Issues = "https://github.com/itsmejoshi/slo-guard/issues"
44
+
45
+ [project.scripts]
46
+ slo-guard = "slo_guard.cli:main"
47
+
48
+ [tool.hatch.build.targets.wheel]
49
+ packages = ["src/slo_guard"]
50
+
51
+ [tool.pytest.ini_options]
52
+ testpaths = ["tests"]
53
+
54
+ [tool.ruff]
55
+ line-length = 100
@@ -0,0 +1,28 @@
1
+ """slo-guard: SLO error-budget tracking and multi-window burn-rate alerting.
2
+
3
+ Quick start:
4
+
5
+ from slo_guard import SLO, evaluate_policy
6
+
7
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
8
+ print(slo.budget_consumed(bad_event_ratio=0.0015))
9
+ print(evaluate_policy(slo))
10
+ """
11
+ from .burnrate import DEFAULT_POLICY, BurnRateWindow, evaluate_policy
12
+ from .config import SLOConfig, load_config
13
+ from .core import SLO
14
+ from .rules import build_alert_group, render_yaml
15
+
16
+ __version__ = "0.1.0"
17
+
18
+ __all__ = [
19
+ "DEFAULT_POLICY",
20
+ "SLO",
21
+ "BurnRateWindow",
22
+ "SLOConfig",
23
+ "__version__",
24
+ "build_alert_group",
25
+ "evaluate_policy",
26
+ "load_config",
27
+ "render_yaml",
28
+ ]
@@ -0,0 +1,111 @@
1
+ """Multi-window, multi-burn-rate alert threshold calculations.
2
+
3
+ Based on the technique described in the Google SRE Workbook chapter
4
+ "Alerting on SLOs": rather than alerting on raw error ratio, alert on the
5
+ *rate* at which the error budget is being burned relative to a constant,
6
+ budget-neutral pace over the SLO period. A short window catches fast burns
7
+ quickly; a long window (paired with the same threshold) filters out noise
8
+ from brief blips, so both must fire together for a page-worthy alert.
9
+
10
+ This module computes the burn-rate threshold for an arbitrary window size
11
+ and SLO period from first principles, rather than hard-coding the constants
12
+ (14.4, 6, 1, ...) that appear in the workbook for the 99.9%/30-day case --
13
+ those fall out of this formula automatically.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+
19
+ from .core import SLO
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class BurnRateWindow:
24
+ """One entry in a multi-window alerting policy.
25
+
26
+ Attributes:
27
+ name: Short label, e.g. "fast", "medium", "slow".
28
+ long_window_hours: The long lookback window, in hours.
29
+ short_window_hours: The short lookback window, in hours (used to
30
+ confirm the alert has not already recovered).
31
+ budget_fraction: Fraction of the *total* error budget this window is
32
+ allowed to consume before paging, e.g. 0.02 for "2% of budget".
33
+ for_duration_minutes: How long the condition must hold before firing,
34
+ to suppress flapping.
35
+ severity: Free-text severity label, e.g. "page" or "ticket".
36
+ """
37
+
38
+ name: str
39
+ long_window_hours: float
40
+ short_window_hours: float
41
+ budget_fraction: float
42
+ for_duration_minutes: float
43
+ severity: str = "page"
44
+
45
+ def burn_rate_threshold(self, slo: SLO) -> float:
46
+ """The burn-rate multiplier that exhausts `budget_fraction` of the
47
+ error budget in exactly `long_window_hours`.
48
+
49
+ burn_rate = budget_fraction / (long_window / period)
50
+
51
+ A burn rate of 1.0 means "consuming the budget at exactly the
52
+ uniform pace needed to exhaust it right at the end of the period".
53
+ A burn rate of 14.4 means consuming it 14.4x faster than that.
54
+ """
55
+ period_hours = slo.period_days * 24
56
+ window_fraction_of_period = self.long_window_hours / period_hours
57
+ return self.budget_fraction / window_fraction_of_period
58
+
59
+
60
+ # The canonical 3-window policy from the SRE Workbook, generalized so it
61
+ # still produces the well-known 14.4 / 6 / 1 multipliers for a 99.9%/30-day
62
+ # SLO, but adapts correctly to any other target or period.
63
+ DEFAULT_POLICY: tuple[BurnRateWindow, ...] = (
64
+ BurnRateWindow(
65
+ name="fast",
66
+ long_window_hours=1,
67
+ short_window_hours=5 / 60,
68
+ budget_fraction=0.02,
69
+ for_duration_minutes=2,
70
+ severity="page",
71
+ ),
72
+ BurnRateWindow(
73
+ name="medium",
74
+ long_window_hours=6,
75
+ short_window_hours=0.5,
76
+ budget_fraction=0.05,
77
+ for_duration_minutes=15,
78
+ severity="page",
79
+ ),
80
+ BurnRateWindow(
81
+ name="slow",
82
+ long_window_hours=72,
83
+ short_window_hours=6,
84
+ budget_fraction=0.10,
85
+ for_duration_minutes=60,
86
+ severity="ticket",
87
+ ),
88
+ )
89
+
90
+
91
+ def evaluate_policy(
92
+ slo: SLO, policy: tuple[BurnRateWindow, ...] = DEFAULT_POLICY
93
+ ) -> list[dict]:
94
+ """Resolve a policy into concrete thresholds for a given SLO.
95
+
96
+ Returns a list of dicts (one per window) ready to feed into a rules
97
+ generator or to print as a report.
98
+ """
99
+ results = []
100
+ for window in policy:
101
+ results.append(
102
+ {
103
+ "name": window.name,
104
+ "long_window_hours": window.long_window_hours,
105
+ "short_window_hours": window.short_window_hours,
106
+ "burn_rate_threshold": window.burn_rate_threshold(slo),
107
+ "for_duration_minutes": window.for_duration_minutes,
108
+ "severity": window.severity,
109
+ }
110
+ )
111
+ return results
@@ -0,0 +1,67 @@
1
+ """Command-line interface for slo-guard.
2
+
3
+ Usage:
4
+ slo-guard rules --config examples/slo.yaml --out rules.yml
5
+ slo-guard budget --config examples/slo.yaml --bad-ratio 0.0015
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import sys
11
+
12
+ from .config import load_config
13
+ from .rules import render_yaml
14
+
15
+
16
+ def _cmd_rules(args: argparse.Namespace) -> int:
17
+ cfg = load_config(args.config)
18
+ text = render_yaml(cfg.slo, cfg.error_selector, cfg.total_selector)
19
+ if args.out:
20
+ with open(args.out, "w", encoding="utf-8") as fh:
21
+ fh.write(text)
22
+ print(f"Wrote Prometheus rules to {args.out}")
23
+ else:
24
+ print(text)
25
+ return 0
26
+
27
+
28
+ def _cmd_budget(args: argparse.Namespace) -> int:
29
+ cfg = load_config(args.config)
30
+ slo = cfg.slo
31
+ consumed = slo.budget_consumed(args.bad_ratio)
32
+ remaining = slo.budget_remaining(args.bad_ratio)
33
+ remaining_minutes = slo.budget_remaining_minutes(args.bad_ratio)
34
+
35
+ print(f"SLO: {slo.name} (target={slo.target:.4%}, period={slo.period_days}d)")
36
+ print(f"Observed bad-event ratio: {args.bad_ratio:.4%}")
37
+ print(f"Error budget consumed: {consumed:.2%}")
38
+ print(f"Error budget remaining: {remaining:.2%}")
39
+ print(f"Remaining budget (time): {remaining_minutes:.1f} minutes")
40
+ if remaining < 0:
41
+ print("STATUS: SLO VIOLATED -- error budget exhausted.", file=sys.stderr)
42
+ return 1
43
+ return 0
44
+
45
+
46
+ def main(argv: list[str] | None = None) -> int:
47
+ parser = argparse.ArgumentParser(prog="slo-guard")
48
+ sub = parser.add_subparsers(dest="command", required=True)
49
+
50
+ p_rules = sub.add_parser("rules", help="Generate Prometheus burn-rate alert rules")
51
+ p_rules.add_argument("--config", required=True, help="Path to SLO YAML config")
52
+ p_rules.add_argument("--out", help="Output file (default: stdout)")
53
+ p_rules.set_defaults(func=_cmd_rules)
54
+
55
+ p_budget = sub.add_parser("budget", help="Report current error-budget status")
56
+ p_budget.add_argument("--config", required=True, help="Path to SLO YAML config")
57
+ p_budget.add_argument(
58
+ "--bad-ratio", required=True, type=float, help="Observed bad-event ratio, e.g. 0.0015"
59
+ )
60
+ p_budget.set_defaults(func=_cmd_budget)
61
+
62
+ args = parser.parse_args(argv)
63
+ return args.func(args)
64
+
65
+
66
+ if __name__ == "__main__":
67
+ raise SystemExit(main())
@@ -0,0 +1,46 @@
1
+ """Load SLO definitions from a YAML config file.
2
+
3
+ Example config file (see examples/slo.yaml):
4
+
5
+ name: checkout-availability
6
+ target: 0.999
7
+ period_days: 30
8
+ error_selector: 'http_requests_total{job="checkout",code=~"5.."}'
9
+ total_selector: 'http_requests_total{job="checkout"}'
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass
14
+ from pathlib import Path
15
+
16
+ import yaml
17
+
18
+ from .core import SLO
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class SLOConfig:
23
+ slo: SLO
24
+ error_selector: str
25
+ total_selector: str
26
+
27
+
28
+ def load_config(path: str | Path) -> SLOConfig:
29
+ with open(path, "r", encoding="utf-8") as fh:
30
+ data = yaml.safe_load(fh)
31
+
32
+ required = {"name", "target", "error_selector", "total_selector"}
33
+ missing = required - data.keys()
34
+ if missing:
35
+ raise ValueError(f"Missing required config keys: {sorted(missing)}")
36
+
37
+ slo = SLO(
38
+ name=data["name"],
39
+ target=float(data["target"]),
40
+ period_days=int(data.get("period_days", 30)),
41
+ )
42
+ return SLOConfig(
43
+ slo=slo,
44
+ error_selector=data["error_selector"],
45
+ total_selector=data["total_selector"],
46
+ )
@@ -0,0 +1,62 @@
1
+ """Core SLO and error-budget math.
2
+
3
+ Implements the error-budget model described in the Google SRE Workbook
4
+ ("Implementing SLOs" and "Alerting on SLOs" chapters): an SLO defines an
5
+ acceptable ratio of bad events over a rolling period; the error budget is
6
+ the complementary "allowance" of bad events; and burn rate measures how
7
+ quickly that allowance is being consumed relative to a uniform, budget-neutral
8
+ pace.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from dataclasses import dataclass
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class SLO:
17
+ """A single Service Level Objective.
18
+
19
+ Attributes:
20
+ name: Human-readable identifier, e.g. "checkout-availability".
21
+ target: The objective as a fraction, e.g. 0.999 for "99.9%".
22
+ period_days: Rolling compliance window in days (commonly 28 or 30).
23
+ """
24
+
25
+ name: str
26
+ target: float
27
+ period_days: int = 30
28
+
29
+ def __post_init__(self) -> None:
30
+ if not (0 < self.target < 1):
31
+ raise ValueError(f"target must be between 0 and 1, got {self.target}")
32
+ if self.period_days <= 0:
33
+ raise ValueError(f"period_days must be positive, got {self.period_days}")
34
+
35
+ @property
36
+ def error_budget(self) -> float:
37
+ """Fraction of events allowed to be 'bad' over the period."""
38
+ return 1.0 - self.target
39
+
40
+ def budget_consumed(self, bad_event_ratio: float) -> float:
41
+ """Fraction of the total error budget consumed so far.
42
+
43
+ Args:
44
+ bad_event_ratio: Observed ratio of bad events over the period
45
+ (e.g. 5xx responses / total responses), between 0 and 1.
46
+
47
+ Returns:
48
+ 0.0 means no budget spent; 1.0 means the budget is exhausted;
49
+ values above 1.0 mean the SLO has already been violated.
50
+ """
51
+ if not (0 <= bad_event_ratio <= 1):
52
+ raise ValueError("bad_event_ratio must be between 0 and 1")
53
+ return bad_event_ratio / self.error_budget
54
+
55
+ def budget_remaining(self, bad_event_ratio: float) -> float:
56
+ """Fraction of error budget remaining (can go negative if exhausted)."""
57
+ return 1.0 - self.budget_consumed(bad_event_ratio)
58
+
59
+ def budget_remaining_minutes(self, bad_event_ratio: float) -> float:
60
+ """Convenience: remaining budget expressed in minutes of the period."""
61
+ total_minutes = self.period_days * 24 * 60
62
+ return self.budget_remaining(bad_event_ratio) * total_minutes
@@ -0,0 +1,89 @@
1
+ """Render burn-rate policies as Prometheus alerting rule YAML."""
2
+ from __future__ import annotations
3
+
4
+ from typing import Any
5
+
6
+ import yaml
7
+
8
+ from .burnrate import DEFAULT_POLICY, BurnRateWindow, evaluate_policy
9
+ from .core import SLO
10
+
11
+
12
+ def _hours_to_promql_duration(hours: float) -> str:
13
+ """Format an hour count as a Prometheus duration string."""
14
+ if hours < 1:
15
+ minutes = round(hours * 60)
16
+ return f"{minutes}m"
17
+ if hours == int(hours):
18
+ return f"{int(hours)}h"
19
+ return f"{hours}h"
20
+
21
+
22
+ def _ratio_expr(window_str: str, error_selector: str, total_selector: str) -> str:
23
+ """Build a PromQL expression for the bad-event ratio over a window."""
24
+ return (
25
+ f"(sum(rate({error_selector}[{window_str}])) / "
26
+ f"sum(rate({total_selector}[{window_str}])))"
27
+ )
28
+
29
+
30
+ def build_alert_group(
31
+ slo: SLO,
32
+ error_selector: str,
33
+ total_selector: str,
34
+ policy: tuple[BurnRateWindow, ...] = DEFAULT_POLICY,
35
+ extra_labels: dict[str, str] | None = None,
36
+ ) -> dict[str, Any]:
37
+ """Build a single Prometheus rule group for an SLO's burn-rate alerts.
38
+
39
+ Args:
40
+ slo: The SLO the alerts protect.
41
+ error_selector: PromQL selector for "bad" events, e.g.
42
+ 'http_requests_total{job="checkout",code=~"5.."}'.
43
+ total_selector: PromQL selector for all events, e.g.
44
+ 'http_requests_total{job="checkout"}'.
45
+ policy: The burn-rate windows to alert on.
46
+ extra_labels: Additional labels merged into every alert rule.
47
+ """
48
+ thresholds = evaluate_policy(slo, policy)
49
+ rules = []
50
+ for window, threshold in zip(policy, thresholds):
51
+ long_w = _hours_to_promql_duration(window.long_window_hours)
52
+ short_w = _hours_to_promql_duration(window.short_window_hours)
53
+ long_expr = _ratio_expr(long_w, error_selector, total_selector)
54
+ short_expr = _ratio_expr(short_w, error_selector, total_selector)
55
+ burn_threshold = threshold["burn_rate_threshold"] * slo.error_budget
56
+
57
+ expr = f"{long_expr} > {burn_threshold:.6g} and {short_expr} > {burn_threshold:.6g}"
58
+
59
+ labels = {"severity": window.severity, "slo": slo.name}
60
+ if extra_labels:
61
+ labels.update(extra_labels)
62
+
63
+ rules.append(
64
+ {
65
+ "alert": f"{slo.name}_burn_rate_{window.name}",
66
+ "expr": expr,
67
+ "for": f"{int(window.for_duration_minutes)}m",
68
+ "labels": labels,
69
+ "annotations": {
70
+ "summary": (
71
+ f"{slo.name}: burning error budget >{threshold['burn_rate_threshold']:.1f}x "
72
+ f"over {long_w} (confirmed over {short_w})"
73
+ ),
74
+ "description": (
75
+ f"Error ratio over {long_w} and {short_w} both exceed the "
76
+ f"budget-neutral threshold for the {slo.target:.4%} / "
77
+ f"{slo.period_days}d SLO. Severity: {window.severity}."
78
+ ),
79
+ },
80
+ }
81
+ )
82
+
83
+ return {"groups": [{"name": f"{slo.name}-burn-rate", "rules": rules}]}
84
+
85
+
86
+ def render_yaml(slo: SLO, error_selector: str, total_selector: str, **kwargs) -> str:
87
+ """Convenience wrapper: build the alert group and dump it as YAML text."""
88
+ group = build_alert_group(slo, error_selector, total_selector, **kwargs)
89
+ return yaml.dump(group, sort_keys=False, default_flow_style=False)
@@ -0,0 +1,43 @@
1
+ # Incident Response Runbook
2
+
3
+ ## Severity levels
4
+
5
+ | Level | Definition | Examples | Response |
6
+ |-------|------------|----------|----------|
7
+ | SEV1 | Full outage or data loss affecting all/most users | Site down, payments failing globally, data corruption | Page on-call immediately, open incident channel, notify leadership within 15 min |
8
+ | SEV2 | Significant degradation affecting a subset of users or a core feature | Elevated error rate on checkout, regional outage, major feature broken | Page on-call, open incident channel within 15 min |
9
+ | SEV3 | Minor issue, workaround available, limited user impact | Non-critical feature degraded, isolated errors | Ticket + fix during business hours, no page |
10
+ | SEV4 | Cosmetic or negligible impact | UI glitch, log noise | Backlog |
11
+
12
+ ## Roles during an incident
13
+
14
+ - **Incident Commander (IC):** Owns the response. Coordinates, makes calls, does not personally debug.
15
+ - **Communications Lead:** Owns status page updates and stakeholder comms so the IC and responders can focus on mitigation.
16
+ - **Subject Matter Expert(s) (SMEs):** Debug and mitigate. Report status to the IC, don't self-direct comms.
17
+
18
+ Small teams can combine IC + Comms; never combine IC + primary debugger on a SEV1/SEV2 — the person mitigating shouldn't also be fielding Slack questions.
19
+
20
+ ## Response sequence
21
+
22
+ 1. **Acknowledge.** Ack the page within the on-call SLA (commonly 5 minutes). Silence is the worst signal you can send.
23
+ 2. **Assess and declare.** Confirm real impact (check dashboards/SLO burn-rate alerts, not just the page text), assign a severity, and declare the incident in the incident channel/tool.
24
+ 3. **Stabilize before you diagnose.** If a recent deploy or config change correlates with the start of the incident, roll it back first and investigate root cause after service is restored. Mitigation beats explanation during an active incident.
25
+ 4. **Communicate on a cadence.** Post an update at a fixed interval (e.g. every 30 minutes for SEV1) even if the update is "still investigating, no new info" — silence reads as "IC lost control."
26
+ 5. **Confirm recovery.** Watch the relevant SLO/error-budget burn-rate alert clear (not just "error rate looks lower") before declaring resolved; burn-rate alerts are designed to require confirmation over a second window precisely to avoid premature all-clears.
27
+ 6. **Resolve and schedule the postmortem.** Resolving the page is not resolving the incident — schedule the postmortem within 1-2 business days while memory is fresh.
28
+
29
+ ## Using burn-rate alerts during an incident
30
+
31
+ Multi-window burn-rate alerts (see `slo-guard`) are designed to answer two questions responders always ask:
32
+
33
+ - **"How bad is this, really?"** — the burn-rate multiplier tells you how many times faster than "acceptable" the service is failing, independent of raw error-rate noise.
34
+ - **"Is it actually over?"** — because the alert requires both a long window and a short window to clear, a flapping recovery won't falsely signal "all clear."
35
+
36
+ If only the fast-burn (short-window) alert is firing and the slow-burn one is not, the issue is likely new and acute. If the slow-burn alert is firing but fast-burn is not, you likely have a longer-running, lower-grade leak worth a ticket rather than an all-hands page.
37
+
38
+ ## Anti-patterns to avoid
39
+
40
+ - Debugging in the incident channel with no IC — it becomes noise no one can follow.
41
+ - Making the on-call engineer also write the customer-facing status update during a SEV1.
42
+ - Declaring resolution the instant the error graph dips, without waiting for the burn-rate confirmation window.
43
+ - Skipping the postmortem because "we know what happened" — the point is the process, not just the notes.
@@ -0,0 +1,31 @@
1
+ # On-Call & Escalation Guide
2
+
3
+ ## On-call expectations
4
+
5
+ - **Ack SLA:** acknowledge a page within 5 minutes during on-call hours.
6
+ - **Handoff:** primary on-call reviews open incidents, muted alerts, and any in-flight action items with the incoming on-call at shift change. A handoff with no verbal/written summary is not a handoff.
7
+ - **Alert fatigue is a bug, not a fact of life.** If an alert fires and the response is consistently "ignore it," fix or delete the alert — don't normalize ignoring pages. Multi-window burn-rate alerting (long + short window agreement) exists specifically to cut this kind of noise; a page-worthy alert should page rarely and mean something when it does.
8
+
9
+ ## Escalation path
10
+
11
+ 1. **Primary on-call** — first responder, owns triage and initial mitigation.
12
+ 2. **Secondary on-call** — paged automatically if primary doesn't ack within the SLA, or pulled in by primary for a second pair of hands on a SEV1/SEV2.
13
+ 3. **Team lead / EM** — looped in for SEV1, for incidents needing cross-team coordination, or for customer-impacting incidents needing executive comms.
14
+ 4. **Incident Commander pool** (for orgs with a dedicated IC rotation) — takes the IC role for SEV1 so the primary responder can stay heads-down on mitigation.
15
+
16
+ Escalate early. Pulling in a second person 10 minutes into a SEV1 that resolves in 5 more minutes costs little; waiting 45 minutes to escalate a SEV1 that needed a second team from the start costs a lot.
17
+
18
+ ## Alert -> action mapping
19
+
20
+ | Alert type | What it means | First action |
21
+ |------------|----------------|--------------|
22
+ | Fast-burn (short+long window, high multiplier) | Severe, acute error-budget burn | Page immediately, treat as candidate SEV1/SEV2, check recent deploys first |
23
+ | Medium-burn | Sustained, moderate burn | Page, investigate within the hour |
24
+ | Slow-burn | Slow leak over days | Ticket, no page; review during business hours |
25
+ | Single-window-only fast alert (long window not yet confirmed) | Possible early signal or noise | Watch, don't page yet — this is why multi-window exists |
26
+
27
+ ## Runbook hygiene
28
+
29
+ - Every alert should link to a runbook (or explicitly note "no runbook needed, self-explanatory").
30
+ - Runbooks should be reviewed whenever the underlying service's architecture changes materially — a stale runbook during an incident is worse than no runbook, because it costs time to discover it's wrong.
31
+ - Keep runbooks in version control alongside the alerting rules that reference them, so they change together.
@@ -0,0 +1,64 @@
1
+ # Postmortem: [Incident Title]
2
+
3
+ **Date of incident:** YYYY-MM-DD
4
+ **Severity:** SEV1 / SEV2 / SEV3
5
+ **Duration:** HH:MM (start) — HH:MM (resolved)
6
+ **Authors:** [names]
7
+ **Status:** Draft / In review / Final
8
+
9
+ > This is a **blameless** postmortem. The goal is to understand systemic
10
+ > conditions that allowed the incident to happen, not to assign fault to
11
+ > an individual. Assume everyone involved made reasonable decisions given
12
+ > what they knew at the time.
13
+
14
+ ## Summary
15
+
16
+ One paragraph: what broke, who/what was affected, how it was resolved.
17
+
18
+ ## Impact
19
+
20
+ - Users/requests affected (rough %, or absolute count if known)
21
+ - Error budget consumed (reference the relevant SLO and burn-rate alert that fired)
22
+ - Revenue/SLA impact if applicable
23
+ - Duration of user-visible impact vs. total incident duration
24
+
25
+ ## Timeline
26
+
27
+ All times in UTC.
28
+
29
+ | Time | Event |
30
+ |------|-------|
31
+ | 14:02 | Deploy of service X v1.4.2 begins |
32
+ | 14:05 | Fast-burn alert fires for `checkout-availability` SLO |
33
+ | 14:07 | On-call acknowledges page |
34
+ | 14:12 | IC declared, incident channel opened |
35
+ | 14:20 | Root cause identified as v1.4.2 config regression |
36
+ | 14:23 | Rollback initiated |
37
+ | 14:31 | Burn-rate alert clears (both windows) |
38
+ | 14:35 | Incident resolved |
39
+
40
+ ## Root cause
41
+
42
+ What was the actual mechanism of failure? Go one or two levels past the immediate trigger — "a bad deploy" is a trigger, not a root cause; "the deploy pipeline had no automated check for X" is closer to root cause.
43
+
44
+ ## What went well
45
+
46
+ - Concrete, specific things — not "good communication" but "IC posted updates every 15 minutes, which kept stakeholders from pinging individually."
47
+
48
+ ## What went poorly
49
+
50
+ - Concrete, specific things — e.g. "the burn-rate alert fired but the runbook link in the alert annotation was stale."
51
+
52
+ ## Where we got lucky
53
+
54
+ Things that could have made this worse but didn't, by chance rather than by design. Worth turning into real safeguards.
55
+
56
+ ## Action items
57
+
58
+ | Action | Owner | Priority | Ticket |
59
+ |--------|-------|----------|--------|
60
+ | Add automated config validation to deploy pipeline | @owner | P1 | LINK |
61
+ | Update runbook link in alert annotation | @owner | P2 | LINK |
62
+ | Add canary stage before full rollout | @owner | P1 | LINK |
63
+
64
+ Every action item needs an owner and a ticket — "we should be more careful" is not an action item.
@@ -0,0 +1,63 @@
1
+ import pytest
2
+
3
+ from slo_guard.burnrate import DEFAULT_POLICY, evaluate_policy
4
+ from slo_guard.core import SLO
5
+
6
+
7
+ def test_default_policy_matches_known_constants_for_999_30d():
8
+ """For a 99.9% / 30-day SLO, the default policy should reproduce the
9
+ well-known 14.4 / 6 / 1 burn-rate multipliers from the Google SRE
10
+ Workbook's 'Alerting on SLOs' chapter."""
11
+ slo = SLO(name="checkout", target=0.999, period_days=30)
12
+ results = {r["name"]: r["burn_rate_threshold"] for r in evaluate_policy(slo)}
13
+
14
+ assert results["fast"] == pytest.approx(14.4, rel=1e-9)
15
+ assert results["medium"] == pytest.approx(6.0, rel=1e-9)
16
+ assert results["slow"] == pytest.approx(1.0, rel=1e-9)
17
+
18
+
19
+ def test_burn_rate_scales_with_different_period():
20
+ """A 90-day period should scale burn-rate thresholds proportionally
21
+ (same budget fraction consumed in the same window is a faster burn
22
+ relative to a longer total period)."""
23
+ slo_30 = SLO(name="a", target=0.999, period_days=30)
24
+ slo_90 = SLO(name="b", target=0.999, period_days=90)
25
+
26
+ fast_30 = evaluate_policy(slo_30)[0]["burn_rate_threshold"]
27
+ fast_90 = evaluate_policy(slo_90)[0]["burn_rate_threshold"]
28
+
29
+ assert fast_90 == pytest.approx(fast_30 * 3, rel=1e-9)
30
+
31
+
32
+ def test_burn_rate_scales_with_different_target():
33
+ """A looser target (larger error budget) means the same window burns a
34
+ smaller fraction of a bigger budget for the same multiplier -- so at a
35
+ fixed budget_fraction, the multiplier itself is independent of target;
36
+ it only depends on window/period. This test locks in that invariant."""
37
+ slo_tight = SLO(name="tight", target=0.999, period_days=30)
38
+ slo_loose = SLO(name="loose", target=0.99, period_days=30)
39
+
40
+ thresholds_tight = [r["burn_rate_threshold"] for r in evaluate_policy(slo_tight)]
41
+ thresholds_loose = [r["burn_rate_threshold"] for r in evaluate_policy(slo_loose)]
42
+
43
+ assert thresholds_tight == pytest.approx(thresholds_loose)
44
+
45
+
46
+ def test_policy_windows_ordered_fast_to_slow():
47
+ windows = [w.long_window_hours for w in DEFAULT_POLICY]
48
+ assert windows == sorted(windows)
49
+
50
+
51
+ def test_evaluate_policy_returns_all_fields():
52
+ slo = SLO(name="x", target=0.995, period_days=28)
53
+ results = evaluate_policy(slo)
54
+ assert len(results) == len(DEFAULT_POLICY)
55
+ for r in results:
56
+ assert set(r.keys()) == {
57
+ "name",
58
+ "long_window_hours",
59
+ "short_window_hours",
60
+ "burn_rate_threshold",
61
+ "for_duration_minutes",
62
+ "severity",
63
+ }
@@ -0,0 +1,48 @@
1
+ import textwrap
2
+
3
+ import pytest
4
+
5
+ from slo_guard.config import load_config
6
+
7
+
8
+ def _write_config(tmp_path, content):
9
+ path = tmp_path / "slo.yaml"
10
+ path.write_text(textwrap.dedent(content))
11
+ return path
12
+
13
+
14
+ def test_load_valid_config(tmp_path):
15
+ path = _write_config(
16
+ tmp_path,
17
+ """
18
+ name: checkout-availability
19
+ target: 0.999
20
+ period_days: 30
21
+ error_selector: 'http_requests_total{job="checkout",code=~"5.."}'
22
+ total_selector: 'http_requests_total{job="checkout"}'
23
+ """,
24
+ )
25
+ cfg = load_config(path)
26
+ assert cfg.slo.name == "checkout-availability"
27
+ assert cfg.slo.target == pytest.approx(0.999)
28
+ assert cfg.slo.period_days == 30
29
+
30
+
31
+ def test_load_config_defaults_period_days(tmp_path):
32
+ path = _write_config(
33
+ tmp_path,
34
+ """
35
+ name: x
36
+ target: 0.99
37
+ error_selector: 'a'
38
+ total_selector: 'b'
39
+ """,
40
+ )
41
+ cfg = load_config(path)
42
+ assert cfg.slo.period_days == 30
43
+
44
+
45
+ def test_load_config_missing_keys_raises(tmp_path):
46
+ path = _write_config(tmp_path, "name: x\ntarget: 0.99\n")
47
+ with pytest.raises(ValueError):
48
+ load_config(path)
@@ -0,0 +1,49 @@
1
+ import pytest
2
+
3
+ from slo_guard.core import SLO
4
+
5
+
6
+ def test_error_budget():
7
+ slo = SLO(name="test", target=0.999, period_days=30)
8
+ assert slo.error_budget == pytest.approx(0.001)
9
+
10
+
11
+ def test_budget_consumed_half():
12
+ slo = SLO(name="test", target=0.999, period_days=30)
13
+ # Half the budget: bad ratio = error_budget / 2
14
+ assert slo.budget_consumed(0.0005) == pytest.approx(0.5)
15
+
16
+
17
+ def test_budget_remaining():
18
+ slo = SLO(name="test", target=0.99, period_days=30)
19
+ assert slo.budget_remaining(0.0) == pytest.approx(1.0)
20
+ assert slo.budget_remaining(0.01) == pytest.approx(0.0)
21
+
22
+
23
+ def test_budget_exhausted_goes_negative():
24
+ slo = SLO(name="test", target=0.99, period_days=30)
25
+ assert slo.budget_remaining(0.02) < 0
26
+
27
+
28
+ def test_budget_remaining_minutes():
29
+ slo = SLO(name="test", target=0.999, period_days=30)
30
+ total_minutes = 30 * 24 * 60
31
+ assert slo.budget_remaining_minutes(0.0) == pytest.approx(total_minutes)
32
+
33
+
34
+ @pytest.mark.parametrize("bad_ratio", [-0.1, 1.1])
35
+ def test_invalid_bad_ratio_raises(bad_ratio):
36
+ slo = SLO(name="test", target=0.999)
37
+ with pytest.raises(ValueError):
38
+ slo.budget_consumed(bad_ratio)
39
+
40
+
41
+ @pytest.mark.parametrize("target", [0.0, 1.0, -0.5, 1.5])
42
+ def test_invalid_target_raises(target):
43
+ with pytest.raises(ValueError):
44
+ SLO(name="test", target=target)
45
+
46
+
47
+ def test_invalid_period_raises():
48
+ with pytest.raises(ValueError):
49
+ SLO(name="test", target=0.99, period_days=0)
@@ -0,0 +1,56 @@
1
+ import yaml
2
+
3
+ from slo_guard.core import SLO
4
+ from slo_guard.rules import build_alert_group, render_yaml
5
+
6
+ ERROR_SEL = 'http_requests_total{job="checkout",code=~"5.."}'
7
+ TOTAL_SEL = 'http_requests_total{job="checkout"}'
8
+
9
+
10
+ def test_build_alert_group_structure():
11
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
12
+ group = build_alert_group(slo, ERROR_SEL, TOTAL_SEL)
13
+
14
+ assert "groups" in group
15
+ assert len(group["groups"]) == 1
16
+ rules = group["groups"][0]["rules"]
17
+ assert len(rules) == 3
18
+ names = {r["alert"] for r in rules}
19
+ assert names == {
20
+ "checkout-availability_burn_rate_fast",
21
+ "checkout-availability_burn_rate_medium",
22
+ "checkout-availability_burn_rate_slow",
23
+ }
24
+
25
+
26
+ def test_alert_expr_contains_selectors_and_and_clause():
27
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
28
+ group = build_alert_group(slo, ERROR_SEL, TOTAL_SEL)
29
+ rule = group["groups"][0]["rules"][0]
30
+ assert ERROR_SEL in rule["expr"]
31
+ assert TOTAL_SEL in rule["expr"]
32
+ assert " and " in rule["expr"]
33
+
34
+
35
+ def test_alert_labels_include_severity_and_slo():
36
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
37
+ group = build_alert_group(slo, ERROR_SEL, TOTAL_SEL)
38
+ for rule in group["groups"][0]["rules"]:
39
+ assert rule["labels"]["slo"] == "checkout-availability"
40
+ assert rule["labels"]["severity"] in {"page", "ticket"}
41
+
42
+
43
+ def test_render_yaml_is_valid_yaml_and_roundtrips():
44
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
45
+ text = render_yaml(slo, ERROR_SEL, TOTAL_SEL)
46
+ parsed = yaml.safe_load(text)
47
+ assert parsed["groups"][0]["name"] == "checkout-availability-burn-rate"
48
+
49
+
50
+ def test_extra_labels_are_merged():
51
+ slo = SLO(name="checkout-availability", target=0.999, period_days=30)
52
+ group = build_alert_group(
53
+ slo, ERROR_SEL, TOTAL_SEL, extra_labels={"team": "payments"}
54
+ )
55
+ for rule in group["groups"][0]["rules"]:
56
+ assert rule["labels"]["team"] == "payments"