unflake 0.6.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
unflake/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Unflake: detect, score, and quarantine flaky tests. Stdlib only."""
2
+
3
+ __version__ = "0.6.1"
unflake/__main__.py ADDED
@@ -0,0 +1,14 @@
1
+ """Entry point. Works as `python -m unflake` AND as `python path/to/__main__.py`
2
+ (the latter is how the npx wrapper invokes the vendored core)."""
3
+
4
+ try:
5
+ from .cli import main
6
+ except ImportError: # pragma: no cover - path-invoked fallback
7
+ import os
8
+ import sys
9
+
10
+ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
11
+ from unflake.cli import main
12
+
13
+ if __name__ == "__main__":
14
+ raise SystemExit(main())
unflake/cli.py ADDED
@@ -0,0 +1,214 @@
1
+ """`unflake` CLI. Stdlib only. Exit codes: 0 clean, 1 findings, 2 usage."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+
9
+ from . import __version__
10
+ from .ingest import load_runs
11
+ from .patterns import SEVERITIES, Finding, scan_path
12
+ from .quarantine import FRAMEWORKS, emit
13
+ from .report import (
14
+ format_analyze_text,
15
+ format_sarif,
16
+ format_scan_json,
17
+ format_scan_text,
18
+ )
19
+ from .runner import collect
20
+ from .scaffold import FRAMEWORKS as SCAFFOLD_FRAMEWORKS
21
+ from .scaffold import init_framework
22
+ from .score import score_runs
23
+
24
+ ORDER = {"error": 0, "warning": 1, "info": 2}
25
+
26
+
27
+ def _meets_threshold(f: Finding, fail_on: str) -> bool:
28
+ return ORDER[f.severity] <= ORDER[fail_on]
29
+
30
+
31
+ def cmd_scan(args: argparse.Namespace) -> int:
32
+ findings: list[Finding] = []
33
+ for target in args.targets:
34
+ try:
35
+ findings.extend(scan_path(target, exclude=args.exclude))
36
+ except ValueError as exc:
37
+ print(f"unflake: error: {exc}", file=sys.stderr)
38
+ return 2
39
+ if args.format == "json":
40
+ print(format_scan_json(findings))
41
+ elif args.format == "sarif":
42
+ print(format_sarif(findings))
43
+ else:
44
+ print(format_scan_text(findings))
45
+ bad = [f for f in findings if _meets_threshold(f, args.fail_on)]
46
+ return 1 if bad else 0
47
+
48
+
49
+ def _print_suite(suite, fmt: str) -> None:
50
+ if fmt == "json":
51
+ print(json.dumps(suite.to_dict(), indent=2))
52
+ elif fmt == "sarif":
53
+ print(format_sarif([], suite))
54
+ else:
55
+ print(format_analyze_text(suite))
56
+
57
+
58
+ def _analyze_gate(suite, fail_on: str) -> int:
59
+ if fail_on == "never":
60
+ return 0
61
+ if fail_on == "flaky":
62
+ return 1 if (suite.flaky_count or suite.regression_count) else 0
63
+ return 0
64
+
65
+
66
+ def _quarantine_note(suite, framework: str | None) -> None:
67
+ if not framework:
68
+ return
69
+ flaky_ids = [t.id for t in suite.tests if t.verdict == "flaky"]
70
+ if not flaky_ids:
71
+ return
72
+ print()
73
+ print(emit(framework, flaky_ids))
74
+
75
+
76
+ def cmd_analyze(args: argparse.Namespace) -> int:
77
+ try:
78
+ runs = load_runs(args.results)
79
+ except ValueError as exc:
80
+ print(f"unflake: error: {exc}", file=sys.stderr)
81
+ return 2
82
+ if len(runs) < 2:
83
+ print("unflake: warning: only 1 run given — flakiness needs >= 2 runs "
84
+ "of the same suite to compare.", file=sys.stderr)
85
+ suite = score_runs(runs)
86
+ _print_suite(suite, args.format)
87
+ return _analyze_gate(suite, args.fail_on)
88
+
89
+
90
+ def cmd_quarantine(args: argparse.Namespace) -> int:
91
+ try:
92
+ runs = load_runs(args.results)
93
+ except ValueError as exc:
94
+ print(f"unflake: error: {exc}", file=sys.stderr)
95
+ return 2
96
+ suite = score_runs(runs)
97
+ regressions = [t.id for t in suite.tests if t.verdict == "regression"]
98
+ if regressions:
99
+ print("# unflake quarantine: NOT quarantining regressions "
100
+ "(real failures — fix them):", file=sys.stderr)
101
+ for rid in regressions:
102
+ print(f"# - {rid}", file=sys.stderr)
103
+ flaky_ids = [t.id for t in suite.tests if t.verdict == "flaky"]
104
+ if not flaky_ids:
105
+ print("# unflake quarantine: no flaky tests — nothing to quarantine.")
106
+ return 0
107
+ try:
108
+ print(emit(args.framework, flaky_ids))
109
+ except ValueError as exc:
110
+ print(f"unflake: error: {exc}", file=sys.stderr)
111
+ return 2
112
+ return 0
113
+
114
+
115
+ def cmd_run(args: argparse.Namespace) -> int:
116
+ if args.runs < 2:
117
+ print("unflake: error: --runs must be >= 2 (one run proves nothing)",
118
+ file=sys.stderr)
119
+ return 2
120
+ cmd = list(args.test_command or [])
121
+ if cmd[:1] == ["--"]:
122
+ cmd = cmd[1:] # argparse.REMAINDER keeps the separator
123
+ if not cmd:
124
+ print("unflake: error: no test command given after `--`", file=sys.stderr)
125
+ return 2
126
+ try:
127
+ files = collect(
128
+ cmd, runs=args.runs, out_dir=args.out_dir,
129
+ junit_flag=args.junit_flag, runner=args.runner, timeout=args.timeout,
130
+ progress=lambda m: print(f"unflake: {m}", file=sys.stderr),
131
+ )
132
+ except ValueError as exc:
133
+ print(f"unflake: error: {exc}", file=sys.stderr)
134
+ return 2
135
+ print(f"unflake: collected {len(files)} run(s) in {args.out_dir}", file=sys.stderr)
136
+ suite = score_runs(load_runs(files))
137
+ _print_suite(suite, args.format)
138
+ _quarantine_note(suite, args.quarantine)
139
+ return _analyze_gate(suite, args.fail_on)
140
+
141
+
142
+ def cmd_init(args: argparse.Namespace) -> int:
143
+ try:
144
+ print(init_framework(args.framework, root=args.root, write=args.write))
145
+ except ValueError as exc:
146
+ print(f"unflake: error: {exc}", file=sys.stderr)
147
+ return 2
148
+ return 0
149
+
150
+
151
+ def build_parser() -> argparse.ArgumentParser:
152
+ p = argparse.ArgumentParser(
153
+ prog="unflake",
154
+ description="Green CI without the rerun ritual — detect, score, quarantine flaky tests.",
155
+ )
156
+ p.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
157
+ sub = p.add_subparsers(dest="command", required=True)
158
+
159
+ s = sub.add_parser("scan", help="Static scan for flake anti-patterns (no test execution).")
160
+ s.add_argument("targets", nargs="+", help="File(s) or directorie(s) to scan.")
161
+ s.add_argument("--exclude", action="append", default=[],
162
+ help="Extra path-part glob to skip (repeatable; "
163
+ "defaults already skip .git, node_modules, venvs, build dirs).")
164
+ s.add_argument("--format", choices=["text", "json", "sarif"], default="text")
165
+ s.add_argument("--fail-on", choices=list(SEVERITIES), default="warning",
166
+ help="Exit 1 if any finding at/above this severity (default: warning).")
167
+ s.set_defaults(func=cmd_scan)
168
+
169
+ a = sub.add_parser("analyze", help="Score flakiness across >=2 result files (JUnit XML or JSON).")
170
+ a.add_argument("results", nargs="+", help="Result files from repeated runs, oldest first.")
171
+ a.add_argument("--format", choices=["text", "json", "sarif"], default="text")
172
+ a.add_argument("--fail-on", choices=["flaky", "never"], default="never",
173
+ help="'flaky' exits 1 on flaky tests OR regressions.")
174
+ a.set_defaults(func=cmd_analyze)
175
+
176
+ q = sub.add_parser("quarantine", help="Emit quarantine config for flaky tests (never deletes).")
177
+ q.add_argument("results", nargs="+", help="Result files from repeated runs, oldest first.")
178
+ q.add_argument("--framework", choices=list(FRAMEWORKS), required=True)
179
+ q.set_defaults(func=cmd_quarantine)
180
+
181
+ r = sub.add_parser("run", help="Run your test command N times, then score (collects JUnit when possible).")
182
+ r.add_argument("--runs", type=int, default=3, help="Repeat count, >= 2 (default: 3).")
183
+ r.add_argument("--out-dir", default=".unflake/runs", help="Where per-run files land.")
184
+ r.add_argument("--junit-flag", default=None,
185
+ help="Flag your runner uses for JUnit output, e.g. '--junitxml=' "
186
+ "(auto-detected for pytest; omit for exit-code-only mode).")
187
+ r.add_argument("--runner", default=None,
188
+ help="Verified runner preset (pytest, vitest, playwright, jest). "
189
+ "Others: use --junit-flag.")
190
+ r.add_argument("--timeout", type=float, default=None, help="Per-run timeout in seconds.")
191
+ r.add_argument("--format", choices=["text", "json", "sarif"], default="text")
192
+ r.add_argument("--fail-on", choices=["flaky", "never"], default="flaky")
193
+ r.add_argument("--quarantine", choices=list(FRAMEWORKS), default=None,
194
+ help="Also emit a quarantine snippet for this framework.")
195
+ r.add_argument("test_command", nargs=argparse.REMAINDER,
196
+ help="Test command after `--`, e.g. `-- pytest tests/ -q`.")
197
+ r.set_defaults(func=cmd_run)
198
+
199
+ i = sub.add_parser("init", help="Scaffold seeded-RNG / frozen-time helpers for a framework.")
200
+ i.add_argument("--framework", choices=list(SCAFFOLD_FRAMEWORKS), required=True)
201
+ i.add_argument("--root", default=".", help="Project root to write into.")
202
+ i.add_argument("--write", action="store_true",
203
+ help="Write files (idempotent). Without it, print to stdout.")
204
+ i.set_defaults(func=cmd_init)
205
+ return p
206
+
207
+
208
+ def main(argv: list[str] | None = None) -> int:
209
+ args = build_parser().parse_args(argv)
210
+ return int(args.func(args))
211
+
212
+
213
+ if __name__ == "__main__":
214
+ raise SystemExit(main())
unflake/ingest.py ADDED
@@ -0,0 +1,95 @@
1
+ """Ingest test results from JUnit XML and generic JSON result files.
2
+
3
+ Universal input: anything that can emit JUnit XML works
4
+ (pytest --junitxml, jest-junit, vitest --reporter=junit, Playwright,
5
+ JUnit/Gradle, go-junit-report, ...). No framework SDK required.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import xml.etree.ElementTree as ET
12
+ from pathlib import Path
13
+
14
+ PASS = "passed"
15
+ FAIL = "failed"
16
+ SKIP = "skipped"
17
+
18
+
19
+ def _test_id(classname: str, name: str, filepath: str = "") -> str:
20
+ if classname:
21
+ return f"{classname}::{name}"
22
+ if filepath:
23
+ return f"{filepath}::{name}"
24
+ return name
25
+
26
+
27
+ def parse_junit_xml(path: str | Path) -> dict[str, str]:
28
+ """Parse one JUnit XML file -> {test_id: status}."""
29
+ path = Path(path)
30
+ try:
31
+ tree = ET.parse(path)
32
+ except ET.ParseError as exc:
33
+ raise ValueError(f"{path}: not valid XML ({exc})") from exc
34
+ root = tree.getroot()
35
+ results: dict[str, str] = {}
36
+ suites = [root] if root.tag == "testsuite" else root.findall("testsuite")
37
+ if not suites and root.tag != "testsuite":
38
+ raise ValueError(f"{path}: no <testsuite> found, is this JUnit XML?")
39
+ for suite in suites:
40
+ for case in suite.iter("testcase"):
41
+ name = case.get("name", "unknown")
42
+ classname = case.get("classname", "") or ""
43
+ filepath = case.get("file", "") or ""
44
+ tid = _test_id(classname, name, filepath)
45
+ if case.find("failure") is not None or case.find("error") is not None:
46
+ status = FAIL
47
+ elif case.find("skipped") is not None:
48
+ status = SKIP
49
+ else:
50
+ status = PASS
51
+ results[tid] = status
52
+ return results
53
+
54
+
55
+ def parse_generic_json(path: str | Path) -> dict[str, str]:
56
+ """Parse {"tests": [{"id": ..., "status": "passed|failed|skipped"}]}."""
57
+ path = Path(path)
58
+ try:
59
+ data = json.loads(path.read_text(encoding="utf-8"))
60
+ except json.JSONDecodeError as exc:
61
+ raise ValueError(f"{path}: not valid JSON ({exc})") from exc
62
+ items = data.get("tests", data if isinstance(data, list) else [])
63
+ results: dict[str, str] = {}
64
+ for item in items:
65
+ tid = str(item.get("id", item.get("name", "unknown")))
66
+ status = str(item.get("status", item.get("outcome", PASS))).lower()
67
+ if status in {"failure", "fail", "failed", "error"}:
68
+ status = FAIL
69
+ elif status in {"skip", "skipped", "xskip"}:
70
+ status = SKIP
71
+ else:
72
+ status = PASS
73
+ results[tid] = status
74
+ return results
75
+
76
+
77
+ def parse_file(path: str | Path) -> dict[str, str]:
78
+ """Auto-detect format by extension/content -> {test_id: status}."""
79
+ path = Path(path)
80
+ if not path.exists():
81
+ raise ValueError(f"{path}: file not found")
82
+ if path.suffix.lower() == ".json":
83
+ return parse_generic_json(path)
84
+ head = path.read_bytes()[:2000].lstrip()
85
+ if head.startswith(b"{") or head.startswith(b"["):
86
+ return parse_generic_json(path)
87
+ return parse_junit_xml(path)
88
+
89
+
90
+ def load_runs(paths: list[str | Path]) -> list[dict[str, str]]:
91
+ """Load N result files (e.g. N repeated runs of the same suite)."""
92
+ runs = []
93
+ for p in paths:
94
+ runs.append(parse_file(p))
95
+ return runs
unflake/patterns.py ADDED
@@ -0,0 +1,243 @@
1
+ """Static scan for flaky-test anti-patterns. No execution, no dependencies.
2
+
3
+ Each rule is (id, severity, languages, check) so contributors can add
4
+ a rule with a single function + one test. See CONTRIBUTING.md.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import fnmatch
10
+ import io
11
+ import re
12
+ import tokenize
13
+ from dataclasses import dataclass
14
+ from pathlib import Path
15
+
16
+ SEVERITIES = ("error", "warning", "info")
17
+
18
+ TEST_FILE_HINT = re.compile(r"(test_|_test\.|\.spec\.|\.test\.)", re.IGNORECASE)
19
+ TEST_DIR_HINT = re.compile(r"(^|/)(test|tests|__tests__|spec|e2e)($|/)")
20
+ CODE_FENCE = "```"
21
+
22
+ SCAN_EXTENSIONS = {".py", ".js", ".ts", ".tsx", ".jsx", ".mts", ".cts", ".go", ".rb"}
23
+
24
+ PY_EXTS = (".py",)
25
+ JS_EXTS = (".js", ".ts", ".tsx", ".jsx", ".mts", ".cts")
26
+ ALL_EXTS = tuple(sorted(SCAN_EXTENSIONS))
27
+
28
+
29
+ @dataclass
30
+ class Finding:
31
+ rule: str
32
+ severity: str
33
+ file: str
34
+ line: int
35
+ message: str
36
+ fix: str = ""
37
+
38
+ def to_dict(self) -> dict:
39
+ return {
40
+ "rule": self.rule,
41
+ "severity": self.severity,
42
+ "file": self.file,
43
+ "line": self.line,
44
+ "message": self.message,
45
+ "fix": self.fix,
46
+ }
47
+
48
+
49
+ @dataclass
50
+ class Rule:
51
+ id: str
52
+ severity: str
53
+ pattern: re.Pattern
54
+ message: str
55
+ fix: str
56
+ extensions: tuple = ALL_EXTS
57
+ test_files_only: bool = True
58
+ # Precision valve: when downgrade_if matches the same line, the finding
59
+ # drops to downgrade_to with a note instead of firing at full severity.
60
+ downgrade_if: re.Pattern | None = None
61
+ downgrade_to: str = "info"
62
+ downgrade_note: str = ""
63
+
64
+
65
+ RULES: list[Rule] = [
66
+ Rule(
67
+ "FLK001", "warning",
68
+ re.compile(r"\btime\.sleep\s*\("),
69
+ "time.sleep() in tests makes timing-dependent flakes",
70
+ "Replace with polling (wait_for / waitFor / eventually) with a timeout, or freeze time.",
71
+ ),
72
+ Rule(
73
+ "FLK002", "warning",
74
+ re.compile(r"\b(datetime\.now|datetime\.utcnow|time\.time|Date\.now|new Date\(\))\s*\("),
75
+ "Unfrozen wall-clock reads make tests date/time dependent",
76
+ "Freeze time (freezegun, jest.useFakeTimers, MockDate) or inject a clock.",
77
+ ),
78
+ Rule(
79
+ "FLK003", "warning",
80
+ re.compile(r"\b(random\.(random|randint|choice|shuffle|uniform)|Math\.random\s*\()"),
81
+ "Unseeded randomness makes failures unreproducible",
82
+ "Seed the RNG (random.seed / faker.seed) and log the seed on failure.",
83
+ ),
84
+ Rule(
85
+ "FLK004", "warning",
86
+ re.compile(r"\b(requests\.(get|post|put|delete)|urllib\.request\.urlopen|fetch\s*\(|axios\.(get|post)|http\.(get|request))\s*\("),
87
+ "Real network calls in tests fail without mocks and leak outside hermetic CI",
88
+ "Mock the boundary (responses, nock, MSW, httpretty) or mark the test integration-only.",
89
+ ALL_EXTS, True,
90
+ re.compile(r"(localhost|127\.0\.0\.1|::1)"),
91
+ "info",
92
+ " (localhost server — hermetic but still a startup-race/port source)",
93
+ ),
94
+ Rule(
95
+ "FLK005", "info",
96
+ re.compile(r"for\s+\w+\s+in\s+set\s*\("),
97
+ "Iterating a set() has nondeterministic order across runs",
98
+ "Iterate over sorted(...) if order matters, or compare as sets.",
99
+ ),
100
+ Rule(
101
+ "FLK006", "info",
102
+ re.compile(r"(localhost|127\.0\.0\.1)\s*:\s*\d{2,5}|port\s*=\s*\d{2,5}"),
103
+ "Hardcoded ports collide on shared/parallel CI runners",
104
+ "Bind port 0 / let the OS pick, then read back the assigned port.",
105
+ ),
106
+ Rule(
107
+ "FLK007", "warning",
108
+ re.compile(r"(\.only\s*\(|describe\.only|it\.only|test\.only|@pytest\.mark\.skip\s*\(\s*\)|@unittest\.skip\s*\(\s*\))"),
109
+ ".only / bare skip markers suggest a suite that is already being triaged around flakes",
110
+ "Quarantine the flaky test explicitly (unflake quarantine) instead of .only/skip.",
111
+ ),
112
+ Rule(
113
+ "FLK008", "info",
114
+ re.compile(r"(concurrent|Parallel|Promise\.all|asyncio\.gather|ThreadPool|go\s+func)"),
115
+ "Concurrency without deterministic synchronization is a classic flake source",
116
+ "Join/await explicitly, avoid asserting on racy intermediate state.",
117
+ ALL_EXTS,
118
+ ),
119
+ Rule(
120
+ "FLK009", "warning",
121
+ re.compile(r"cy\.wait\(\s*\d+"),
122
+ "cy.wait(ms) fixed waits pass locally and flake on loaded CI runners",
123
+ "Wait on state instead: cy.intercept() + cy.wait('@alias'), or .should() retry-ability.",
124
+ JS_EXTS,
125
+ ),
126
+ Rule(
127
+ "FLK010", "warning",
128
+ re.compile(r"(waitForTimeout|wait_for_timeout)\s*\("),
129
+ "Fixed-time waits (waitForTimeout / wait_for_timeout) are Playwright's documented flake factory",
130
+ "Use web-first assertions: await expect(locator).toBeVisible() auto-retries.",
131
+ JS_EXTS + PY_EXTS,
132
+ ),
133
+ ]
134
+
135
+
136
+ def _is_test_file(path: Path) -> bool:
137
+ s = str(path)
138
+ return bool(TEST_FILE_HINT.search(path.name) or TEST_DIR_HINT.search(s))
139
+
140
+
141
+ DEFAULT_EXCLUDES = (".git", "node_modules", "__pycache__", ".venv", "venv",
142
+ "dist", "build", ".tox", ".pytest_cache", ".eggs", "*.egg-info")
143
+
144
+
145
+ def _string_spans(text: str) -> dict[int, list[tuple[int, int]]]:
146
+ """Map 1-based lineno -> string-literal column spans, via stdlib tokenize.
147
+
148
+ Matches fully inside a literal are fixture text, not executed code.
149
+ Python only; unparseable files yield {} (no skipping — never hide signal).
150
+ """
151
+ spans: dict[int, list[tuple[int, int]]] = {}
152
+ try:
153
+ toks = tokenize.generate_tokens(io.StringIO(text).readline)
154
+ for tok in toks:
155
+ if tok.type not in (tokenize.STRING, getattr(tokenize, "FSTRING_MIDDLE", -1)):
156
+ continue
157
+ (srow, scol), (erow, ecol) = tok.start, tok.end
158
+ for r in range(srow, erow + 1):
159
+ spans.setdefault(r, []).append(
160
+ (scol if r == srow else 0, ecol if r == erow else 10 ** 9))
161
+ except (tokenize.TokenError, SyntaxError, IndentationError):
162
+ return {}
163
+ return spans
164
+
165
+
166
+ def _in_literal(spans: dict[int, list[tuple[int, int]]],
167
+ lineno: int, start: int, end: int) -> bool:
168
+ return any(a <= start and end <= b for a, b in spans.get(lineno, []))
169
+
170
+
171
+ def _strip_code_fences(lines: list[str]) -> list[bool]:
172
+ """Return mask: True if the line is live code (not inside a markdown fence)."""
173
+ live = []
174
+ in_fence = False
175
+ for line in lines:
176
+ if line.strip().startswith(CODE_FENCE):
177
+ in_fence = not in_fence
178
+ live.append(False)
179
+ else:
180
+ live.append(not in_fence)
181
+ return live
182
+
183
+
184
+ def scan_file(path: str | Path) -> list[Finding]:
185
+ path = Path(path)
186
+ if path.suffix not in SCAN_EXTENSIONS:
187
+ return []
188
+ try:
189
+ text = path.read_text(encoding="utf-8", errors="replace")
190
+ except OSError:
191
+ return []
192
+ lines = text.splitlines()
193
+ live = _strip_code_fences(lines) if path.suffix == ".md" else [True] * len(lines)
194
+ is_test = _is_test_file(path)
195
+ literals = _string_spans(text) if path.suffix == ".py" else {}
196
+ findings: list[Finding] = []
197
+ for rule in RULES:
198
+ if rule.test_files_only and not is_test:
199
+ continue
200
+ if path.suffix not in rule.extensions:
201
+ continue
202
+ for i, line in enumerate(lines):
203
+ if not live[i]:
204
+ continue
205
+ m = rule.pattern.search(line)
206
+ if not m:
207
+ continue
208
+ if literals and _in_literal(literals, i + 1, m.start(), m.end()):
209
+ continue # fixture text in a literal, not executed code
210
+ severity, message = rule.severity, rule.message
211
+ if rule.downgrade_if and rule.downgrade_if.search(line):
212
+ severity, message = rule.downgrade_to, message + rule.downgrade_note
213
+ findings.append(Finding(
214
+ rule=rule.id, severity=severity,
215
+ file=str(path), line=i + 1,
216
+ message=message, fix=rule.fix,
217
+ ))
218
+ return sorted(findings, key=lambda f: (f.file, f.line))
219
+
220
+
221
+ def _excluded(path: Path, patterns: tuple[str, ...]) -> bool:
222
+ return any(fnmatch.fnmatchcase(part, pat)
223
+ for part in path.parts for pat in patterns)
224
+
225
+
226
+ def scan_path(target: str | Path,
227
+ exclude: list[str] | tuple[str, ...] | None = None) -> list[Finding]:
228
+ """Scan a file or directory recursively, skipping DEFAULT_EXCLUDES plus extras."""
229
+ patterns = tuple(DEFAULT_EXCLUDES) + tuple(exclude or ())
230
+ target = Path(target)
231
+ files: list[Path] = []
232
+ if target.is_file():
233
+ files = [target]
234
+ elif target.is_dir():
235
+ for ext in SCAN_EXTENSIONS:
236
+ files.extend(target.rglob(f"*{ext}"))
237
+ files = [f for f in files if not _excluded(f, patterns)]
238
+ else:
239
+ raise ValueError(f"{target}: no such file or directory")
240
+ findings: list[Finding] = []
241
+ for f in sorted(files):
242
+ findings.extend(scan_file(f))
243
+ return findings
unflake/quarantine.py ADDED
@@ -0,0 +1,78 @@
1
+ """Quarantine emitters: keep the build green WITHOUT deleting tests.
2
+
3
+ Philosophy (borrowed from the best CI teams): a quarantined test still
4
+ runs and still reports — it just can't fail the build. Deletion hides
5
+ signal; quarantine preserves it while unblocking everyone else.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+
12
+ _SAFE_K = re.compile(r"[A-Za-z0-9_:.\-\[\]]+\Z")
13
+
14
+
15
+ def _k_name(tid: str) -> tuple[str, bool]:
16
+ """Short test name for -k expressions. Returns (name, is_safe).
17
+
18
+ Unsafe names (spaces/quotes from parametrization) can't go in -k
19
+ verbatim — callers must verify and prefer --deselect with node ids.
20
+ We never silently mangle: mangling hides signal.
21
+ """
22
+ name = tid.split("::")[-1].split("[")[0]
23
+ if _SAFE_K.fullmatch(name):
24
+ return name, True
25
+ return name, False
26
+
27
+
28
+ def _pytest_k_expression(ids: list[str]) -> tuple[str, list[str]]:
29
+ shorts, risky = [], []
30
+ for tid in ids:
31
+ name, safe = _k_name(tid)
32
+ shorts.append(name)
33
+ if not safe:
34
+ risky.append(tid)
35
+ return "not " + " and not ".join(shorts), risky
36
+
37
+
38
+ def emit(framework: str, flaky_ids: list[str]) -> str:
39
+ fw = framework.lower()
40
+ if fw == "pytest":
41
+ k, risky = _pytest_k_expression(flaky_ids)
42
+ listed = "\n".join(f"# - {tid}" for tid in flaky_ids)
43
+ out = (
44
+ "# Generated by `unflake quarantine --framework pytest` — review, don't blindly commit.\n"
45
+ f"# Quarantined {len(flaky_ids)} flaky test(s):\n{listed}\n\n"
46
+ "# Option A: exclude them from the gating run, keep them in a nightly non-gating run:\n"
47
+ f"pytest -k \"{k}\"\n\n"
48
+ "# Option B: run ONLY the quarantined set (nightly, non-gating, still reporting):\n"
49
+ f"pytest -k \"{' or '.join(t.split('::')[-1].split('[')[0] for t in flaky_ids)}\" --junitxml=quarantine.xml\n"
50
+ )
51
+ if risky:
52
+ out += ("\n# WARNING: these ids contain spaces/quotes and can't go in -k verbatim —\n"
53
+ "# verify the expression matches, or use `pytest --deselect <nodeid>`:\n"
54
+ + "".join(f"# RISKY: {tid}\n" for tid in risky))
55
+ return out
56
+ if fw in {"jest", "vitest"}:
57
+ arr = ",\n".join(f' "{tid}"' for tid in flaky_ids)
58
+ flag = "--testPathIgnorePatterns" if fw == "jest" else "--exclude"
59
+ return (
60
+ f"// Generated by `unflake quarantine --framework {fw}`\n"
61
+ "// Quarantine list — wire into CI as a non-gating job, keep reporting.\n"
62
+ f"export const QUARANTINED = [\n{arr}\n];\n\n"
63
+ f"// CI tip ({fw}): run the gating suite excluding quarantine files,\n"
64
+ f"// then run quarantine separately with `{flag}` inverted so it still reports.\n"
65
+ )
66
+ if fw == "playwright":
67
+ joined = "|".join(t.split("::")[-1].split("[")[0] for t in flaky_ids)
68
+ return (
69
+ "# Generated by `unflake quarantine --framework playwright`\n"
70
+ f"# Gating run excludes the {len(flaky_ids)} quarantined test(s):\n"
71
+ f"npx playwright test --grep-invert \"{joined}\"\n\n"
72
+ "# Nightly non-gating run (still reporting):\n"
73
+ f"npx playwright test --grep \"{joined}\" --reporter=junit,html\n"
74
+ )
75
+ raise ValueError(f"unknown framework {framework!r} (try: pytest, jest, vitest, playwright)")
76
+
77
+
78
+ FRAMEWORKS = ("pytest", "jest", "vitest", "playwright")
unflake/report.py ADDED
@@ -0,0 +1,123 @@
1
+ """Report formatters: text (human), json (machine), sarif (code scanning)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+
7
+ from .patterns import RULES, Finding
8
+ from .score import SuiteScore
9
+
10
+ ORDER = {"error": 0, "warning": 1, "info": 2}
11
+
12
+
13
+ def _sarif_rules() -> list[dict]:
14
+ rules = [{
15
+ "id": f"unflake/{r.id}",
16
+ "shortDescription": {"text": r.message},
17
+ "help": {"text": r.fix},
18
+ "defaultConfiguration": {"level": {"error": "error", "warning": "warning"}.get(r.severity, "note")},
19
+ } for r in RULES]
20
+ rules += [
21
+ {"id": "unflake/FLAKY",
22
+ "shortDescription": {"text": "Mixed outcomes across repeated runs — quarantine candidate."},
23
+ "defaultConfiguration": {"level": "warning"}},
24
+ {"id": "unflake/REGRESSION",
25
+ "shortDescription": {"text": "Green-then-red-forever — real failure, never quarantine."},
26
+ "defaultConfiguration": {"level": "error"}},
27
+ ]
28
+ return rules
29
+
30
+
31
+ def format_scan_text(findings: list[Finding]) -> str:
32
+ if not findings:
33
+ return "unflake scan: clean — no flake anti-patterns found."
34
+ lines = [f"unflake scan: {len(findings)} finding(s)"]
35
+ for f in sorted(findings, key=lambda x: (ORDER[x.severity], x.file, x.line)):
36
+ lines.append(f" {f.severity.upper():7} [{f.rule}] {f.file}:{f.line}")
37
+ lines.append(f" {f.message}")
38
+ if f.fix:
39
+ lines.append(f" fix: {f.fix}")
40
+ return "\n".join(lines)
41
+
42
+
43
+ def format_scan_json(findings: list[Finding]) -> str:
44
+ return json.dumps({"findings": [f.to_dict() for f in findings]}, indent=2)
45
+
46
+
47
+ def format_sarif(findings: list[Finding], suite: SuiteScore | None = None) -> str:
48
+ results = []
49
+ for f in findings:
50
+ level = {"error": "error", "warning": "warning", "info": "note"}[f.severity]
51
+ results.append({
52
+ "ruleId": f"unflake/{f.rule}",
53
+ "level": level,
54
+ "message": {"text": f"{f.message} Fix: {f.fix}" if f.fix else f.message},
55
+ "locations": [{
56
+ "physicalLocation": {
57
+ "artifactLocation": {"uri": f.file},
58
+ "region": {"startLine": f.line},
59
+ }
60
+ }],
61
+ })
62
+ if suite:
63
+ for t in suite.tests:
64
+ if t.verdict == "flaky":
65
+ results.append({
66
+ "ruleId": "unflake/FLAKY",
67
+ "level": "warning",
68
+ "message": {"text": f"Flaky across {t.runs} runs: "
69
+ f"{t.passes} passed, {t.fails} failed."},
70
+ "locations": [{
71
+ "physicalLocation": {
72
+ "artifactLocation": {"uri": t.id},
73
+ "region": {"startLine": 1},
74
+ }
75
+ }],
76
+ })
77
+ elif t.verdict == "regression":
78
+ results.append({
79
+ "ruleId": "unflake/REGRESSION",
80
+ "level": "error",
81
+ "message": {"text": f"Regression: green until run {t.change_point}, "
82
+ f"red ever since. Fix — do not quarantine."},
83
+ "locations": [{
84
+ "physicalLocation": {
85
+ "artifactLocation": {"uri": t.id},
86
+ "region": {"startLine": 1},
87
+ }
88
+ }],
89
+ })
90
+ return json.dumps({
91
+ "$schema": "https://json.schemastore.org/sarif-2.1.0.json",
92
+ "version": "2.1.0",
93
+ "runs": [{
94
+ "tool": {"driver": {
95
+ "name": "unflake",
96
+ "informationUri": "https://github.com/YOUR-USER/unflake",
97
+ "rules": _sarif_rules(),
98
+ }},
99
+ "results": results,
100
+ }],
101
+ }, indent=2)
102
+
103
+
104
+ def format_analyze_text(suite: SuiteScore) -> str:
105
+ lines = [
106
+ f"FlakeScore: {suite.flake_score}/100 "
107
+ f"({suite.flaky_count} flaky of {suite.total} tests)"
108
+ ]
109
+ for t in suite.tests:
110
+ if t.verdict == "flaky":
111
+ lines.append(f" FLAKY {t.id} ({t.passes}P/{t.fails}F over {t.runs} runs)")
112
+ for t in suite.tests:
113
+ if t.verdict == "regression":
114
+ lines.append(f" REGRESSION {t.id} (green until run {t.change_point}, "
115
+ f"red ever since — fix this, do NOT quarantine it)")
116
+ for t in suite.tests:
117
+ if t.verdict == "new":
118
+ lines.append(f" NEW {t.id} (only in latest run — needs more runs before judging)")
119
+ stables = [t for t in suite.tests
120
+ if t.verdict in ("stable-pass", "stable-fail", "skipped")]
121
+ if stables:
122
+ lines.append(f" ... +{len(stables)} stable/skipped test(s) (see --format json for full list)")
123
+ return "\n".join(lines)
unflake/runner.py ADDED
@@ -0,0 +1,151 @@
1
+ """`unflake run`: collect the >=2 runs that flake-hunting requires.
2
+
3
+ Runs your test command N times, collects a JUnit XML per run, then scores.
4
+ With a JUnit flag the per-test analysis works; without one, each run's exit
5
+ code still gives an honest suite-level verdict (stable vs flaky suite).
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import os
11
+ import subprocess
12
+ from pathlib import Path
13
+
14
+ from .ingest import FAIL, PASS
15
+
16
+
17
+ def _detect_junit_flag(cmd: list[str]) -> str | None:
18
+ joined = " ".join(cmd)
19
+ if "pytest" in cmd[0] or "pytest" in joined.split()[0]:
20
+ return "--junitxml="
21
+ return None
22
+
23
+
24
+ # Verified presets: only runners proven end-to-end (see tests/test_e2e.py).
25
+ # Anything else uses --junit-flag explicitly — we'd rather error than guess.
26
+ # Each preset is (argv_fn, env_fn|None): some runners take the JUnit path
27
+ # as a flag (pytest, vitest), others as an env var (playwright).
28
+ def _pytest_args(dest: str) -> list[str]:
29
+ return [f"--junitxml={dest}"]
30
+
31
+
32
+ def _vitest_args(dest: str) -> list[str]:
33
+ return ["--reporter=junit", f"--outputFile={dest}"]
34
+
35
+
36
+ def _playwright_args(dest: str) -> list[str]:
37
+ return ["--reporter=junit"]
38
+
39
+
40
+ def _playwright_env(dest: str) -> dict[str, str]:
41
+ return {"PLAYWRIGHT_JUNIT_OUTPUT_NAME": dest}
42
+
43
+
44
+ def _jest_args(dest: str) -> list[str]:
45
+ # NOTE: space form — `--reporters=jest-junit` breaks reporter parsing.
46
+ return ["--reporters", "jest-junit"]
47
+
48
+
49
+ def _jest_env(dest: str) -> dict[str, str]:
50
+ return {"JEST_JUNIT_OUTPUT_FILE": dest}
51
+
52
+
53
+ RUNNERS = {
54
+ "pytest": (_pytest_args, None),
55
+ "vitest": (_vitest_args, None),
56
+ "playwright": (_playwright_args, _playwright_env),
57
+ "jest": (_jest_args, _jest_env),
58
+ }
59
+
60
+
61
+ def resolve_junit_flag(runner: str | None, junit_flag: str | None,
62
+ cmd: list[str]) -> str | None:
63
+ """Explicit --junit-flag wins; --runner pytest/vitest are verified; anything
64
+ else falls back to pytest auto-detect, else exit-code-only mode."""
65
+ if junit_flag:
66
+ return junit_flag
67
+ if runner:
68
+ if runner not in RUNNERS:
69
+ raise ValueError(f"unknown --runner {runner!r} (try: {', '.join(sorted(RUNNERS))})")
70
+ return None # preset handled via runner_args(), not a single flag
71
+ return _detect_junit_flag(cmd)
72
+
73
+
74
+ def runner_args(runner: str | None, dest: str) -> list[str] | None:
75
+ """Extra argv for a verified preset, or None."""
76
+ if runner and runner in RUNNERS:
77
+ return RUNNERS[runner][0](dest)
78
+ return None
79
+
80
+
81
+ def runner_env(runner: str | None, dest: str) -> dict[str, str]:
82
+ """Extra env for a verified preset (empty when the path goes via argv)."""
83
+ if runner and runner in RUNNERS:
84
+ fn = RUNNERS[runner][1]
85
+ return dict(fn(dest)) if fn else {}
86
+ return {}
87
+
88
+
89
+ def collect(cmd: list[str], runs: int, out_dir: str | Path,
90
+ junit_flag: str | None = None, runner: str | None = None,
91
+ timeout: float | None = None,
92
+ progress=None) -> list[Path]:
93
+ """Run cmd N times. Return per-run result files (JUnit XML if possible,
94
+ else generic JSON with one suite-level entry)."""
95
+ out = Path(out_dir)
96
+ out.mkdir(parents=True, exist_ok=True)
97
+ if runner and runner not in RUNNERS:
98
+ raise ValueError(f"unknown --runner {runner!r} (try: {', '.join(sorted(RUNNERS))})")
99
+ single = resolve_junit_flag(runner, junit_flag, cmd)
100
+ flag = single # legacy single-flag path (explicit --junit-flag or pytest auto-detect)
101
+ files: list[Path] = []
102
+ for i in range(1, runs + 1):
103
+ if progress:
104
+ progress(f"run {i}/{runs}: {' '.join(cmd)}")
105
+ log = open(out / f"run{i}.log", "w", encoding="utf-8") # closed below
106
+ try:
107
+ dest = out / f"run{i}.xml"
108
+ preset = runner_args(runner, str(dest))
109
+ if preset is not None:
110
+ full = [*cmd, *preset]
111
+ elif flag:
112
+ full = [*cmd, f"{flag}{dest}"]
113
+ else:
114
+ full = None
115
+ if full is not None:
116
+ run_env = dict(os.environ)
117
+ run_env.update(runner_env(runner, str(dest)))
118
+ try:
119
+ proc = subprocess.run(full, timeout=timeout, env=run_env,
120
+ stdout=log, stderr=subprocess.STDOUT)
121
+ ok = proc.returncode == 0
122
+ except subprocess.TimeoutExpired:
123
+ ok = False
124
+ if not dest.exists():
125
+ # Command swallowed the flag (or never wrote XML):
126
+ # fall back to an exit-code entry so the run still counts.
127
+ dest = _write_json_fallback(out, i, cmd, ok, timed_out=True)
128
+ files.append(dest)
129
+ else:
130
+ try:
131
+ proc = subprocess.run(cmd, timeout=timeout,
132
+ stdout=log, stderr=subprocess.STDOUT)
133
+ ok, timed_out = proc.returncode == 0, False
134
+ except subprocess.TimeoutExpired:
135
+ ok, timed_out = False, True
136
+ files.append(_write_json_fallback(out, i, cmd, ok, timed_out))
137
+ finally:
138
+ log.close()
139
+ return files
140
+
141
+
142
+ def _write_json_fallback(out: Path, i: int, cmd: list[str],
143
+ ok: bool, timed_out: bool = False) -> Path:
144
+ import json
145
+ dest = out / f"run{i}.json"
146
+ status = PASS if ok else FAIL
147
+ note = "suite exit 0" if ok else ("suite timed out" if timed_out else "suite exit != 0")
148
+ dest.write_text(json.dumps({"tests": [
149
+ {"id": f"suite::{' '.join(cmd)}", "status": status, "note": note},
150
+ ]}), encoding="utf-8")
151
+ return dest
unflake/scaffold.py ADDED
@@ -0,0 +1,89 @@
1
+ """`unflake init`: scaffold the two cheapest flake preventions.
2
+
3
+ 1. Seeded RNG + logged seed (unseeded randomness is unreproducible by definition).
4
+ 2. A frozen-time pointer (wall clocks are the #2 classic).
5
+
6
+ `--write` creates files idempotently (marker-guarded, never duplicates).
7
+ Without `--write`, snippets print to stdout with wiring instructions.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from pathlib import Path
13
+
14
+ MARKER = "# unflake:seed-logging"
15
+
16
+ PYTEST_SNIPPET = '''%(marker)s (added by `unflake init --framework pytest --write`)
17
+ # Deterministic RNG per test + logged seed: rerun any failure exactly with
18
+ # UNFLAKE_SEED=<seed> pytest ...
19
+ import os
20
+ import random
21
+
22
+ import pytest
23
+
24
+
25
+ @pytest.fixture(autouse=True)
26
+ def _unflake_seed(request):
27
+ seed = os.environ.get("UNFLAKE_SEED", "0")
28
+ random.seed(f"{seed}-{request.node.nodeid}")
29
+ print(f"\\n[unflake] seed={seed} test={request.node.nodeid}")
30
+ yield
31
+ '''
32
+
33
+ JEST_SNIPPET = '''// unflake:seed-logging (added by `unflake init --framework jest`)
34
+ // Deterministic RNG per test file. Wire into jest config:
35
+ // { setupFiles: ["<rootDir>/unflake.setup.cjs"] }
36
+ const seed = process.env.UNFLAKE_SEED || "0";
37
+ let s = [...seed].reduce((a, c) => (a * 33 + c.charCodeAt(0)) >>> 0, 7) || 7;
38
+ const file = expect.getState().testPath || "unknown";
39
+ console.log(`[unflake] seed=${seed} file=${file}`);
40
+ Math.random = () => (s = (s * 1103515245 + 12345) & 0x7fffffff) / 0x7fffffff;
41
+ '''
42
+
43
+ VITEST_SNIPPET = '''// unflake:seed-logging (added by `unflake init --framework vitest`)
44
+ // Wire into vitest config: export default { test: { setupFiles: ["./unflake.setup.js"] } }
45
+ const seed = process.env.UNFLAKE_SEED || "0";
46
+ console.log(`[unflake] seed=${seed}`);
47
+ '''
48
+
49
+ PLAYWRIGHT_SNIPPET = '''// unflake:seed-logging (added by `unflake init --framework playwright`)
50
+ // Playwright has built-in retries — prefer them over fixed waits:
51
+ // { retries: 2, use: { trace: "retain-on-failure" } }
52
+ // And NEVER page.waitForTimeout(): use web-first assertions:
53
+ // await expect(page.getByRole("button")).toBeVisible();
54
+ '''
55
+
56
+ SNIPPETS = {
57
+ "pytest": ("tests/conftest.py", PYTEST_SNIPPET),
58
+ "jest": ("unflake.setup.cjs", JEST_SNIPPET),
59
+ "vitest": ("unflake.setup.js", VITEST_SNIPPET),
60
+ "playwright": ("unflake.playwright-notes.js", PLAYWRIGHT_SNIPPET),
61
+ }
62
+
63
+ FRAMEWORKS = tuple(SNIPPETS)
64
+
65
+
66
+ def snippet(framework: str) -> tuple[str, str]:
67
+ """Return (relative_path, content) for a framework."""
68
+ if framework not in SNIPPETS:
69
+ raise ValueError(f"unknown framework {framework!r} (try: {', '.join(FRAMEWORKS)})")
70
+ rel, template = SNIPPETS[framework]
71
+ return rel, template % {"marker": MARKER}
72
+
73
+
74
+ def init_framework(framework: str, root: str | Path = ".", write: bool = False) -> str:
75
+ """Print or write the scaffold. Returns a human summary."""
76
+ rel, content = snippet(framework)
77
+ if not write:
78
+ return f"# {rel} — create this file with the content below:\n\n{content}"
79
+ dest = Path(root) / rel
80
+ dest.parent.mkdir(parents=True, exist_ok=True)
81
+ if dest.exists():
82
+ existing = dest.read_text(encoding="utf-8", errors="replace")
83
+ if MARKER in existing or "unflake:seed-logging" in existing:
84
+ return f"{dest}: already scaffolded — left untouched."
85
+ with dest.open("a", encoding="utf-8") as fh:
86
+ fh.write("\n" + content)
87
+ return f"{dest}: snippet appended (existing file preserved)."
88
+ dest.write_text(content, encoding="utf-8")
89
+ return f"{dest}: created. Rerun any failure exactly with UNFLAKE_SEED=<seed>."
unflake/score.py ADDED
@@ -0,0 +1,122 @@
1
+ """Flakiness scoring across repeated runs.
2
+
3
+ Pass result files oldest -> newest. Verdicts per test:
4
+
5
+ stable-pass always green
6
+ stable-fail always red — a real failure, NOT a flake, never quarantine
7
+ flaky mixed outcomes with no clean break — quarantine candidate
8
+ regression green, then red from some run on and never green again
9
+ (needs >= 3 runs; with 2 runs a pass->fail looks identical
10
+ to a flake, so it stays flaky) — fix, never quarantine
11
+ new only appeared in the latest run — needs more runs
12
+ skipped never produced a pass/fail
13
+
14
+ Suite FlakeScore: 0-100, share of non-flaky tests (higher = healthier).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from dataclasses import dataclass
20
+
21
+ from .ingest import FAIL, PASS, SKIP
22
+
23
+
24
+ @dataclass
25
+ class TestScore:
26
+ id: str
27
+ passes: int
28
+ fails: int
29
+ skips: int
30
+ runs: int
31
+ verdict: str
32
+ flakiness: float # 0.0 stable .. 1.0 maximally flaky (50/50 split)
33
+ change_point: int | None = None # 1-based run where a regression starts
34
+
35
+ def to_dict(self) -> dict:
36
+ return {
37
+ "id": self.id, "passes": self.passes, "fails": self.fails,
38
+ "skips": self.skips, "runs": self.runs,
39
+ "verdict": self.verdict, "flakiness": round(self.flakiness, 3),
40
+ "change_point": self.change_point,
41
+ }
42
+
43
+
44
+ @dataclass
45
+ class SuiteScore:
46
+ tests: list[TestScore]
47
+ flake_score: float
48
+ flaky_count: int
49
+ regression_count: int
50
+ total: int
51
+
52
+ def to_dict(self) -> dict:
53
+ return {
54
+ "flake_score": round(self.flake_score, 1),
55
+ "flaky_count": self.flaky_count,
56
+ "regression_count": self.regression_count,
57
+ "total": self.total,
58
+ "tests": [t.to_dict() for t in self.tests],
59
+ }
60
+
61
+ def issues(self) -> list[TestScore]:
62
+ """Actionable tests: flaky first, then regressions."""
63
+ return [t for t in self.tests if t.verdict in ("flaky", "regression")]
64
+
65
+
66
+ def _is_clean_break(statuses: list[str]) -> int | None:
67
+ """If statuses are pass* then fail* (skips ignored), return the 1-based
68
+ index of the first fail. Otherwise None."""
69
+ pf = [s for s in statuses if s in (PASS, FAIL)]
70
+ if len(pf) < 3 or PASS not in pf or FAIL not in pf:
71
+ return None
72
+ first_fail = pf.index(FAIL)
73
+ if first_fail == 0:
74
+ return None # failed from the start — stable-fail territory
75
+ if all(s == FAIL for s in pf[first_fail:]):
76
+ return statuses.index(FAIL) + 1 # 1-based, counting all runs
77
+ return None
78
+
79
+
80
+ def score_runs(runs: list[dict[str, str]]) -> SuiteScore:
81
+ if not runs:
82
+ raise ValueError("need at least one run to score")
83
+ ids: set[str] = set()
84
+ for run in runs:
85
+ ids.update(run.keys())
86
+ tests: list[TestScore] = []
87
+ for tid in sorted(ids):
88
+ statuses = [run.get(tid, "missing") for run in runs]
89
+ present = [s for s in statuses if s != "missing"]
90
+ passes = sum(1 for s in present if s == PASS)
91
+ fails = sum(1 for s in present if s == FAIL)
92
+ skips = sum(1 for s in present if s == SKIP)
93
+ change_point: int | None = None
94
+ if "missing" in statuses:
95
+ if len(present) == 1 and statuses[-1] != "missing":
96
+ verdict, flakiness = "new", 0.0
97
+ else:
98
+ verdict, flakiness = "flaky", 0.5 # appears/disappears
99
+ elif passes and fails:
100
+ cp = _is_clean_break(statuses)
101
+ if cp is not None:
102
+ verdict, flakiness, change_point = "regression", 0.0, cp
103
+ else:
104
+ rate = passes / (passes + fails)
105
+ verdict, flakiness = "flaky", round(1.0 - abs(rate - 0.5) * 2, 3)
106
+ elif fails and not passes:
107
+ verdict, flakiness = ("stable-fail", 0.0) if not skips else ("flaky", 0.3)
108
+ elif passes and not fails:
109
+ verdict, flakiness = ("stable-pass", 0.0) if not skips else ("flaky", 0.2)
110
+ else:
111
+ verdict, flakiness = "skipped", 0.0
112
+ tests.append(TestScore(tid, passes, fails, skips, len(runs),
113
+ verdict, flakiness, change_point))
114
+ flaky = [t for t in tests if t.verdict == "flaky"]
115
+ regressions = [t for t in tests if t.verdict == "regression"]
116
+ total = len(tests) or 1
117
+ flake_score = 100.0 * (total - len(flaky)) / total
118
+ order = {"flaky": 0, "regression": 1, "stable-fail": 2,
119
+ "new": 3, "stable-pass": 4, "skipped": 5}
120
+ tests.sort(key=lambda t: (order.get(t.verdict, 9), -t.flakiness, t.id))
121
+ return SuiteScore(tests, round(flake_score, 1), len(flaky),
122
+ len(regressions), len(tests))
@@ -0,0 +1,143 @@
1
+ Metadata-Version: 2.4
2
+ Name: unflake
3
+ Version: 0.6.1
4
+ Summary: Green CI without the rerun ritual — detect, score, and quarantine flaky tests
5
+ License: MIT
6
+ Project-URL: Homepage, https://github.com/YOUR-USER/unflake
7
+ Project-URL: Issues, https://github.com/YOUR-USER/unflake/issues
8
+ Keywords: testing,flaky-tests,ci,qa,agent-skills
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: Software Development :: Testing
14
+ Requires-Python: >=3.9
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ Dynamic: license-file
18
+
19
+ # Unflake 🟢
20
+
21
+ [![CI](https://github.com/YOUR-USER/unflake/actions/workflows/ci.yml/badge.svg)](https://github.com/YOUR-USER/unflake/actions/workflows/ci.yml)
22
+ [![OpenSSF Scorecard](https://api.scorecard.dev/projects/github.com/YOUR-USER/unflake/badge)](https://scorecard.dev/viewer/?uri=github.com/YOUR-USER/unflake)
23
+ [![PyPI](https://img.shields.io/pypi/v/unflake.svg)](https://pypi.org/project/unflake/)
24
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
25
+
26
+ **Green CI without the "just rerun it" ritual.**
27
+
28
+ ![Unflake 60-second demo](assets/terminal-demo.svg)
29
+
30
+ CI is red. You rerun. It's green. You merge, trusting nothing. Unflake hunts the flake instead: static scan in milliseconds, FlakeScore across repeated runs, regression-vs-flake verdicts, and quarantine configs that keep the build green *without deleting a single test*.
31
+
32
+ ![scan → run → calm](assets/pipeline.svg)
33
+
34
+ ```bash
35
+ pip install unflake
36
+ unflake scan tests/ # 1. smells, instantly
37
+ unflake run --runs 3 --runner pytest -- pytest tests/ -q # 2. collect + score
38
+ unflake quarantine run1.xml run2.xml --framework pytest # 3. quarantine, don't delete
39
+ unflake init --framework pytest --write # 4. seeded RNG scaffold (prevention)
40
+ ```
41
+
42
+ JS-first repo? Same tool, no Python packaging — just Python itself (3.9+):
43
+
44
+ ```bash
45
+ npx unflake-ci scan tests/
46
+ npx unflake-ci run --runs 3 --runner vitest -- npx vitest run tests/
47
+ ```
48
+
49
+ Proven on real suites: 3,676 tests × 5 runs across attrs/click/tqdm → FlakeScore 100,
50
+ 0 false quarantines ([study](benchmarks/BENCH.md)).
51
+
52
+ No install handy? `bash demo.sh` runs the whole loop on fixtures in 60 seconds.
53
+
54
+ > ⭐ If flaky tests have ever paged you, star this — it helps other devs find it.
55
+
56
+ ## Before / after
57
+
58
+ ```python
59
+ # before: the 3am pager
60
+ def test_checkout():
61
+ time.sleep(5) # pray the server is up
62
+ assert requests.get(API).ok # real network in a unit test
63
+ ```
64
+
65
+ ```bash
66
+ $ unflake scan tests/
67
+ WARNING [FLK001] tests/test_shop.py:4 — time.sleep() in tests makes timing-dependent flakes
68
+ fix: Replace with polling (wait_for / waitFor / eventually) with a timeout, or freeze time.
69
+ WARNING [FLK004] tests/test_shop.py:5 — Real network calls in tests fail without mocks ...
70
+ ```
71
+
72
+ ```bash
73
+ $ unflake run --runs 3 -- pytest tests/ -q
74
+ FlakeScore: 50.0/100 (2 flaky of 4 tests)
75
+ FLAKY test_cart::test_checkout (2P/1F over 3 runs)
76
+ FLAKY test_cart::test_discount (2P/1F over 3 runs)
77
+ ```
78
+
79
+ ```bash
80
+ $ unflake quarantine run1.xml run2.xml --framework pytest
81
+ pytest -k "not test_checkout and not test_discount" # gating run stays green
82
+ pytest -k "test_checkout or test_discount" # nightly run still reports
83
+ ```
84
+
85
+ Quarantine preserves signal; deletion hides it. Every quarantined test still runs — it just can't fail the build. **Regressions are never quarantined**: green-then-red-forever is a real failure — fix it.
86
+
87
+ ## Flaky or regression? Unflake knows the difference
88
+
89
+ Pass result files oldest → newest and Unflake separates three fates:
90
+
91
+ | Verdict | Meaning | Action |
92
+ |---|---|---|
93
+ | `FLAKY` | mixed outcomes, no clean break | quarantine + fix root cause |
94
+ | `REGRESSION` | green until run N, red ever since | fix now — never quarantine |
95
+ | `NEW` / stable | seen once / always green / always red | more runs / ship it / real bug |
96
+
97
+ Two runs can only suggest a flake; calling a regression needs ≥3. One run proves nothing — `unflake run` exists so there's no excuse.
98
+
99
+ ## Why not the others?
100
+
101
+ | | Unflake | Heavy flake platforms |
102
+ |---|---|---|
103
+ | Install | zero-dep, stdlib only, `pip install unflake` | OTel collectors, dashboards, SaaS |
104
+ | Input | JUnit XML from **any** framework (pytest ✅, Vitest ✅, Playwright ✅, Jest ✅ e2e — plus generic `--junit-flag`, Go, JUnit…) | per-framework reporters/plugins |
105
+ | First value | `scan` in milliseconds, no execution | needs history ingestion |
106
+ | Verdicts | flaky vs regression vs new, change-point included | usually just a score |
107
+ | Philosophy | quarantine, never delete | often auto-skip-and-forget |
108
+ | Agent-native | `SKILL.md` + harness adapters day one | docs page, if you're lucky |
109
+
110
+ ## Works with your harness
111
+
112
+ Claude Code · Codex · Copilot · Cursor · Gemini · Pi · OpenCode · Windsurf · Cline · Qoder — plus a GitHub Action (`action.yml`, SARIF → code scanning) and a pre-commit hook. Details: [`docs/ADAPTERS.md`](docs/ADAPTERS.md). The skill lives at [`skills/unflake/SKILL.md`](skills/unflake/SKILL.md).
113
+
114
+ 10 rules today, incl. the JS classics: `cy.wait(ms)` (FLK009) and `waitForTimeout` (FLK010).
115
+ Precision features: localhost-aware severities, string-literal awareness (Python),
116
+ `--exclude` globs, and SARIF rule metadata for code scanning.
117
+
118
+ ## Benchmarks (honest, reproducible)
119
+
120
+ Fixtures + real pytest e2e + a real-repo scan (`psf/requests`: 120 findings in ~0.1s — with the noise documented, not hidden). Method, numbers, and known limitations: [`benchmarks/BENCH.md`](benchmarks/BENCH.md). If a number doesn't reproduce, that's a bug — file it.
121
+
122
+ ## Contributing — built for drive-by PRs
123
+
124
+ - 🧪 New FLK rule = one regex + one test (`good first issue`)
125
+ - 🔌 New harness adapter or quarantine emitter = one file + one test
126
+ - 🌱 `unflake init` for your framework = one snippet
127
+ - 🌍 Translations welcome (`README.<lang>.md`)
128
+ - Full loop: [`CONTRIBUTING.md`](CONTRIBUTING.md)
129
+
130
+ ## Roadmap
131
+
132
+ - [x] v0.1 — scan / analyze / quarantine, SARIF, 4 frameworks, skill + adapters
133
+ - [x] v0.2 — `run` collector, regression-vs-flake verdicts, `init` scaffolder, FLK009/FLK010, pre-commit + GitHub Action, `pip install unflake`
134
+ - [x] v0.3 — `--runner` presets, per-run logs, real-pytest e2e in CI, launch kit (`demo.sh`, templates, `LAUNCH.md`), real-repo validation
135
+ - [x] v0.4 — precision (localhost-aware FLK004, literal-awareness), `--exclude`, SARIF rules, safe `-k` quoting, CI matrix 3.10–3.14
136
+ - [x] v0.5 — `unflake-ci` npx wrapper (verified pack→install→run), `--runner vitest` verified e2e
137
+ - [x] v0.6 — `--runner` playwright + jest verified e2e, real-repo study (3,676 tests, 0 false quarantines)
138
+ - [ ] Recall study v2 on suites with known flakes (specificity proven; recall needs wild flakes)
139
+ - [ ] JS/TS string-literal awareness + variable-host local-server detection (see BENCH.md)
140
+
141
+ ## License
142
+
143
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,16 @@
1
+ unflake/__init__.py,sha256=lns3NRcFIEftdwyjj6dzFeW1GyuI542ru-p2POnskTs,94
2
+ unflake/__main__.py,sha256=HlOZgR1S8N_FcV4UyArFybJgDBsd3tCfdOvbRzc5cww,447
3
+ unflake/cli.py,sha256=p-d_LPLABdhTQTaTajZHidSq5TEM_o1XCWhb3nPuWyo,8442
4
+ unflake/ingest.py,sha256=KkjDiJWvvGshliFGS5gPcp2PR4RYyBudKFDz_jZbvC8,3289
5
+ unflake/patterns.py,sha256=ZFomxU28yT1G1qKE-Ky73QLJjSjM5rWOkvme1xS2Fng,8942
6
+ unflake/quarantine.py,sha256=PoPpNHb5csPrcvdR78-7CgFPskZBH5Rm8WmwODNJ4lg,3431
7
+ unflake/report.py,sha256=ciIhbZMwRJVUqtrnm-UWEjCQj4yi7l8nHOy1g4hI97Y,4837
8
+ unflake/runner.py,sha256=tliEh2vf-FDMpXvTjpi1U6SsqJT_Fk96LEKzxj7GiUM,5659
9
+ unflake/scaffold.py,sha256=OiuHoaWMfB7oICiRbyJdqxlZQW-D8ydtBLsecfFjSek,3537
10
+ unflake/score.py,sha256=sLACgHvJpw8oW8-UTJp38QO-8PITqiUZJjVjF82xTPs,4672
11
+ unflake-0.6.1.dist-info/licenses/LICENSE,sha256=61Ot6dNjcmx1kKVNfniNm1PTbNF4N7zmNjxJ_H7VCao,1077
12
+ unflake-0.6.1.dist-info/METADATA,sha256=H9CXyanmYd5DhhX_l96Tts6RG_SemSHioRMT-2EliQg,7063
13
+ unflake-0.6.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
14
+ unflake-0.6.1.dist-info/entry_points.txt,sha256=BRz8fLVpu72NT5t1fgFH-tqpb_8qqW6e8tcup6wo8HY,45
15
+ unflake-0.6.1.dist-info/top_level.txt,sha256=m9wa0x1Y0vt-lPoM1gt7ihlXwbdqARUn2oRUdv7hZEQ,8
16
+ unflake-0.6.1.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ unflake = unflake.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Unflake Contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ unflake