unflake 0.6.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unflake/__init__.py +3 -0
- unflake/__main__.py +14 -0
- unflake/cli.py +214 -0
- unflake/ingest.py +95 -0
- unflake/patterns.py +243 -0
- unflake/quarantine.py +78 -0
- unflake/report.py +123 -0
- unflake/runner.py +151 -0
- unflake/scaffold.py +89 -0
- unflake/score.py +122 -0
- unflake-0.6.1.dist-info/METADATA +143 -0
- unflake-0.6.1.dist-info/RECORD +16 -0
- unflake-0.6.1.dist-info/WHEEL +5 -0
- unflake-0.6.1.dist-info/entry_points.txt +2 -0
- unflake-0.6.1.dist-info/licenses/LICENSE +21 -0
- unflake-0.6.1.dist-info/top_level.txt +1 -0
unflake/__init__.py
ADDED
unflake/__main__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Entry point. Works as `python -m unflake` AND as `python path/to/__main__.py`
|
|
2
|
+
(the latter is how the npx wrapper invokes the vendored core)."""
|
|
3
|
+
|
|
4
|
+
try:
|
|
5
|
+
from .cli import main
|
|
6
|
+
except ImportError: # pragma: no cover - path-invoked fallback
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
|
|
10
|
+
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
11
|
+
from unflake.cli import main
|
|
12
|
+
|
|
13
|
+
if __name__ == "__main__":
|
|
14
|
+
raise SystemExit(main())
|
unflake/cli.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""`unflake` CLI. Stdlib only. Exit codes: 0 clean, 1 findings, 2 usage."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .ingest import load_runs
|
|
11
|
+
from .patterns import SEVERITIES, Finding, scan_path
|
|
12
|
+
from .quarantine import FRAMEWORKS, emit
|
|
13
|
+
from .report import (
|
|
14
|
+
format_analyze_text,
|
|
15
|
+
format_sarif,
|
|
16
|
+
format_scan_json,
|
|
17
|
+
format_scan_text,
|
|
18
|
+
)
|
|
19
|
+
from .runner import collect
|
|
20
|
+
from .scaffold import FRAMEWORKS as SCAFFOLD_FRAMEWORKS
|
|
21
|
+
from .scaffold import init_framework
|
|
22
|
+
from .score import score_runs
|
|
23
|
+
|
|
24
|
+
ORDER = {"error": 0, "warning": 1, "info": 2}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _meets_threshold(f: Finding, fail_on: str) -> bool:
|
|
28
|
+
return ORDER[f.severity] <= ORDER[fail_on]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def cmd_scan(args: argparse.Namespace) -> int:
|
|
32
|
+
findings: list[Finding] = []
|
|
33
|
+
for target in args.targets:
|
|
34
|
+
try:
|
|
35
|
+
findings.extend(scan_path(target, exclude=args.exclude))
|
|
36
|
+
except ValueError as exc:
|
|
37
|
+
print(f"unflake: error: {exc}", file=sys.stderr)
|
|
38
|
+
return 2
|
|
39
|
+
if args.format == "json":
|
|
40
|
+
print(format_scan_json(findings))
|
|
41
|
+
elif args.format == "sarif":
|
|
42
|
+
print(format_sarif(findings))
|
|
43
|
+
else:
|
|
44
|
+
print(format_scan_text(findings))
|
|
45
|
+
bad = [f for f in findings if _meets_threshold(f, args.fail_on)]
|
|
46
|
+
return 1 if bad else 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _print_suite(suite, fmt: str) -> None:
|
|
50
|
+
if fmt == "json":
|
|
51
|
+
print(json.dumps(suite.to_dict(), indent=2))
|
|
52
|
+
elif fmt == "sarif":
|
|
53
|
+
print(format_sarif([], suite))
|
|
54
|
+
else:
|
|
55
|
+
print(format_analyze_text(suite))
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _analyze_gate(suite, fail_on: str) -> int:
|
|
59
|
+
if fail_on == "never":
|
|
60
|
+
return 0
|
|
61
|
+
if fail_on == "flaky":
|
|
62
|
+
return 1 if (suite.flaky_count or suite.regression_count) else 0
|
|
63
|
+
return 0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _quarantine_note(suite, framework: str | None) -> None:
|
|
67
|
+
if not framework:
|
|
68
|
+
return
|
|
69
|
+
flaky_ids = [t.id for t in suite.tests if t.verdict == "flaky"]
|
|
70
|
+
if not flaky_ids:
|
|
71
|
+
return
|
|
72
|
+
print()
|
|
73
|
+
print(emit(framework, flaky_ids))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def cmd_analyze(args: argparse.Namespace) -> int:
|
|
77
|
+
try:
|
|
78
|
+
runs = load_runs(args.results)
|
|
79
|
+
except ValueError as exc:
|
|
80
|
+
print(f"unflake: error: {exc}", file=sys.stderr)
|
|
81
|
+
return 2
|
|
82
|
+
if len(runs) < 2:
|
|
83
|
+
print("unflake: warning: only 1 run given — flakiness needs >= 2 runs "
|
|
84
|
+
"of the same suite to compare.", file=sys.stderr)
|
|
85
|
+
suite = score_runs(runs)
|
|
86
|
+
_print_suite(suite, args.format)
|
|
87
|
+
return _analyze_gate(suite, args.fail_on)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def cmd_quarantine(args: argparse.Namespace) -> int:
|
|
91
|
+
try:
|
|
92
|
+
runs = load_runs(args.results)
|
|
93
|
+
except ValueError as exc:
|
|
94
|
+
print(f"unflake: error: {exc}", file=sys.stderr)
|
|
95
|
+
return 2
|
|
96
|
+
suite = score_runs(runs)
|
|
97
|
+
regressions = [t.id for t in suite.tests if t.verdict == "regression"]
|
|
98
|
+
if regressions:
|
|
99
|
+
print("# unflake quarantine: NOT quarantining regressions "
|
|
100
|
+
"(real failures — fix them):", file=sys.stderr)
|
|
101
|
+
for rid in regressions:
|
|
102
|
+
print(f"# - {rid}", file=sys.stderr)
|
|
103
|
+
flaky_ids = [t.id for t in suite.tests if t.verdict == "flaky"]
|
|
104
|
+
if not flaky_ids:
|
|
105
|
+
print("# unflake quarantine: no flaky tests — nothing to quarantine.")
|
|
106
|
+
return 0
|
|
107
|
+
try:
|
|
108
|
+
print(emit(args.framework, flaky_ids))
|
|
109
|
+
except ValueError as exc:
|
|
110
|
+
print(f"unflake: error: {exc}", file=sys.stderr)
|
|
111
|
+
return 2
|
|
112
|
+
return 0
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def cmd_run(args: argparse.Namespace) -> int:
|
|
116
|
+
if args.runs < 2:
|
|
117
|
+
print("unflake: error: --runs must be >= 2 (one run proves nothing)",
|
|
118
|
+
file=sys.stderr)
|
|
119
|
+
return 2
|
|
120
|
+
cmd = list(args.test_command or [])
|
|
121
|
+
if cmd[:1] == ["--"]:
|
|
122
|
+
cmd = cmd[1:] # argparse.REMAINDER keeps the separator
|
|
123
|
+
if not cmd:
|
|
124
|
+
print("unflake: error: no test command given after `--`", file=sys.stderr)
|
|
125
|
+
return 2
|
|
126
|
+
try:
|
|
127
|
+
files = collect(
|
|
128
|
+
cmd, runs=args.runs, out_dir=args.out_dir,
|
|
129
|
+
junit_flag=args.junit_flag, runner=args.runner, timeout=args.timeout,
|
|
130
|
+
progress=lambda m: print(f"unflake: {m}", file=sys.stderr),
|
|
131
|
+
)
|
|
132
|
+
except ValueError as exc:
|
|
133
|
+
print(f"unflake: error: {exc}", file=sys.stderr)
|
|
134
|
+
return 2
|
|
135
|
+
print(f"unflake: collected {len(files)} run(s) in {args.out_dir}", file=sys.stderr)
|
|
136
|
+
suite = score_runs(load_runs(files))
|
|
137
|
+
_print_suite(suite, args.format)
|
|
138
|
+
_quarantine_note(suite, args.quarantine)
|
|
139
|
+
return _analyze_gate(suite, args.fail_on)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def cmd_init(args: argparse.Namespace) -> int:
|
|
143
|
+
try:
|
|
144
|
+
print(init_framework(args.framework, root=args.root, write=args.write))
|
|
145
|
+
except ValueError as exc:
|
|
146
|
+
print(f"unflake: error: {exc}", file=sys.stderr)
|
|
147
|
+
return 2
|
|
148
|
+
return 0
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
152
|
+
p = argparse.ArgumentParser(
|
|
153
|
+
prog="unflake",
|
|
154
|
+
description="Green CI without the rerun ritual — detect, score, quarantine flaky tests.",
|
|
155
|
+
)
|
|
156
|
+
p.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
157
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
158
|
+
|
|
159
|
+
s = sub.add_parser("scan", help="Static scan for flake anti-patterns (no test execution).")
|
|
160
|
+
s.add_argument("targets", nargs="+", help="File(s) or directorie(s) to scan.")
|
|
161
|
+
s.add_argument("--exclude", action="append", default=[],
|
|
162
|
+
help="Extra path-part glob to skip (repeatable; "
|
|
163
|
+
"defaults already skip .git, node_modules, venvs, build dirs).")
|
|
164
|
+
s.add_argument("--format", choices=["text", "json", "sarif"], default="text")
|
|
165
|
+
s.add_argument("--fail-on", choices=list(SEVERITIES), default="warning",
|
|
166
|
+
help="Exit 1 if any finding at/above this severity (default: warning).")
|
|
167
|
+
s.set_defaults(func=cmd_scan)
|
|
168
|
+
|
|
169
|
+
a = sub.add_parser("analyze", help="Score flakiness across >=2 result files (JUnit XML or JSON).")
|
|
170
|
+
a.add_argument("results", nargs="+", help="Result files from repeated runs, oldest first.")
|
|
171
|
+
a.add_argument("--format", choices=["text", "json", "sarif"], default="text")
|
|
172
|
+
a.add_argument("--fail-on", choices=["flaky", "never"], default="never",
|
|
173
|
+
help="'flaky' exits 1 on flaky tests OR regressions.")
|
|
174
|
+
a.set_defaults(func=cmd_analyze)
|
|
175
|
+
|
|
176
|
+
q = sub.add_parser("quarantine", help="Emit quarantine config for flaky tests (never deletes).")
|
|
177
|
+
q.add_argument("results", nargs="+", help="Result files from repeated runs, oldest first.")
|
|
178
|
+
q.add_argument("--framework", choices=list(FRAMEWORKS), required=True)
|
|
179
|
+
q.set_defaults(func=cmd_quarantine)
|
|
180
|
+
|
|
181
|
+
r = sub.add_parser("run", help="Run your test command N times, then score (collects JUnit when possible).")
|
|
182
|
+
r.add_argument("--runs", type=int, default=3, help="Repeat count, >= 2 (default: 3).")
|
|
183
|
+
r.add_argument("--out-dir", default=".unflake/runs", help="Where per-run files land.")
|
|
184
|
+
r.add_argument("--junit-flag", default=None,
|
|
185
|
+
help="Flag your runner uses for JUnit output, e.g. '--junitxml=' "
|
|
186
|
+
"(auto-detected for pytest; omit for exit-code-only mode).")
|
|
187
|
+
r.add_argument("--runner", default=None,
|
|
188
|
+
help="Verified runner preset (pytest, vitest, playwright, jest). "
|
|
189
|
+
"Others: use --junit-flag.")
|
|
190
|
+
r.add_argument("--timeout", type=float, default=None, help="Per-run timeout in seconds.")
|
|
191
|
+
r.add_argument("--format", choices=["text", "json", "sarif"], default="text")
|
|
192
|
+
r.add_argument("--fail-on", choices=["flaky", "never"], default="flaky")
|
|
193
|
+
r.add_argument("--quarantine", choices=list(FRAMEWORKS), default=None,
|
|
194
|
+
help="Also emit a quarantine snippet for this framework.")
|
|
195
|
+
r.add_argument("test_command", nargs=argparse.REMAINDER,
|
|
196
|
+
help="Test command after `--`, e.g. `-- pytest tests/ -q`.")
|
|
197
|
+
r.set_defaults(func=cmd_run)
|
|
198
|
+
|
|
199
|
+
i = sub.add_parser("init", help="Scaffold seeded-RNG / frozen-time helpers for a framework.")
|
|
200
|
+
i.add_argument("--framework", choices=list(SCAFFOLD_FRAMEWORKS), required=True)
|
|
201
|
+
i.add_argument("--root", default=".", help="Project root to write into.")
|
|
202
|
+
i.add_argument("--write", action="store_true",
|
|
203
|
+
help="Write files (idempotent). Without it, print to stdout.")
|
|
204
|
+
i.set_defaults(func=cmd_init)
|
|
205
|
+
return p
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def main(argv: list[str] | None = None) -> int:
|
|
209
|
+
args = build_parser().parse_args(argv)
|
|
210
|
+
return int(args.func(args))
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
if __name__ == "__main__":
|
|
214
|
+
raise SystemExit(main())
|
unflake/ingest.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Ingest test results from JUnit XML and generic JSON result files.
|
|
2
|
+
|
|
3
|
+
Universal input: anything that can emit JUnit XML works
|
|
4
|
+
(pytest --junitxml, jest-junit, vitest --reporter=junit, Playwright,
|
|
5
|
+
JUnit/Gradle, go-junit-report, ...). No framework SDK required.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import xml.etree.ElementTree as ET
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
PASS = "passed"
|
|
15
|
+
FAIL = "failed"
|
|
16
|
+
SKIP = "skipped"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _test_id(classname: str, name: str, filepath: str = "") -> str:
|
|
20
|
+
if classname:
|
|
21
|
+
return f"{classname}::{name}"
|
|
22
|
+
if filepath:
|
|
23
|
+
return f"{filepath}::{name}"
|
|
24
|
+
return name
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def parse_junit_xml(path: str | Path) -> dict[str, str]:
|
|
28
|
+
"""Parse one JUnit XML file -> {test_id: status}."""
|
|
29
|
+
path = Path(path)
|
|
30
|
+
try:
|
|
31
|
+
tree = ET.parse(path)
|
|
32
|
+
except ET.ParseError as exc:
|
|
33
|
+
raise ValueError(f"{path}: not valid XML ({exc})") from exc
|
|
34
|
+
root = tree.getroot()
|
|
35
|
+
results: dict[str, str] = {}
|
|
36
|
+
suites = [root] if root.tag == "testsuite" else root.findall("testsuite")
|
|
37
|
+
if not suites and root.tag != "testsuite":
|
|
38
|
+
raise ValueError(f"{path}: no <testsuite> found, is this JUnit XML?")
|
|
39
|
+
for suite in suites:
|
|
40
|
+
for case in suite.iter("testcase"):
|
|
41
|
+
name = case.get("name", "unknown")
|
|
42
|
+
classname = case.get("classname", "") or ""
|
|
43
|
+
filepath = case.get("file", "") or ""
|
|
44
|
+
tid = _test_id(classname, name, filepath)
|
|
45
|
+
if case.find("failure") is not None or case.find("error") is not None:
|
|
46
|
+
status = FAIL
|
|
47
|
+
elif case.find("skipped") is not None:
|
|
48
|
+
status = SKIP
|
|
49
|
+
else:
|
|
50
|
+
status = PASS
|
|
51
|
+
results[tid] = status
|
|
52
|
+
return results
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def parse_generic_json(path: str | Path) -> dict[str, str]:
|
|
56
|
+
"""Parse {"tests": [{"id": ..., "status": "passed|failed|skipped"}]}."""
|
|
57
|
+
path = Path(path)
|
|
58
|
+
try:
|
|
59
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
60
|
+
except json.JSONDecodeError as exc:
|
|
61
|
+
raise ValueError(f"{path}: not valid JSON ({exc})") from exc
|
|
62
|
+
items = data.get("tests", data if isinstance(data, list) else [])
|
|
63
|
+
results: dict[str, str] = {}
|
|
64
|
+
for item in items:
|
|
65
|
+
tid = str(item.get("id", item.get("name", "unknown")))
|
|
66
|
+
status = str(item.get("status", item.get("outcome", PASS))).lower()
|
|
67
|
+
if status in {"failure", "fail", "failed", "error"}:
|
|
68
|
+
status = FAIL
|
|
69
|
+
elif status in {"skip", "skipped", "xskip"}:
|
|
70
|
+
status = SKIP
|
|
71
|
+
else:
|
|
72
|
+
status = PASS
|
|
73
|
+
results[tid] = status
|
|
74
|
+
return results
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def parse_file(path: str | Path) -> dict[str, str]:
|
|
78
|
+
"""Auto-detect format by extension/content -> {test_id: status}."""
|
|
79
|
+
path = Path(path)
|
|
80
|
+
if not path.exists():
|
|
81
|
+
raise ValueError(f"{path}: file not found")
|
|
82
|
+
if path.suffix.lower() == ".json":
|
|
83
|
+
return parse_generic_json(path)
|
|
84
|
+
head = path.read_bytes()[:2000].lstrip()
|
|
85
|
+
if head.startswith(b"{") or head.startswith(b"["):
|
|
86
|
+
return parse_generic_json(path)
|
|
87
|
+
return parse_junit_xml(path)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def load_runs(paths: list[str | Path]) -> list[dict[str, str]]:
|
|
91
|
+
"""Load N result files (e.g. N repeated runs of the same suite)."""
|
|
92
|
+
runs = []
|
|
93
|
+
for p in paths:
|
|
94
|
+
runs.append(parse_file(p))
|
|
95
|
+
return runs
|
unflake/patterns.py
ADDED
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""Static scan for flaky-test anti-patterns. No execution, no dependencies.
|
|
2
|
+
|
|
3
|
+
Each rule is (id, severity, languages, check) so contributors can add
|
|
4
|
+
a rule with a single function + one test. See CONTRIBUTING.md.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import fnmatch
|
|
10
|
+
import io
|
|
11
|
+
import re
|
|
12
|
+
import tokenize
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
SEVERITIES = ("error", "warning", "info")
|
|
17
|
+
|
|
18
|
+
TEST_FILE_HINT = re.compile(r"(test_|_test\.|\.spec\.|\.test\.)", re.IGNORECASE)
|
|
19
|
+
TEST_DIR_HINT = re.compile(r"(^|/)(test|tests|__tests__|spec|e2e)($|/)")
|
|
20
|
+
CODE_FENCE = "```"
|
|
21
|
+
|
|
22
|
+
SCAN_EXTENSIONS = {".py", ".js", ".ts", ".tsx", ".jsx", ".mts", ".cts", ".go", ".rb"}
|
|
23
|
+
|
|
24
|
+
PY_EXTS = (".py",)
|
|
25
|
+
JS_EXTS = (".js", ".ts", ".tsx", ".jsx", ".mts", ".cts")
|
|
26
|
+
ALL_EXTS = tuple(sorted(SCAN_EXTENSIONS))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Finding:
|
|
31
|
+
rule: str
|
|
32
|
+
severity: str
|
|
33
|
+
file: str
|
|
34
|
+
line: int
|
|
35
|
+
message: str
|
|
36
|
+
fix: str = ""
|
|
37
|
+
|
|
38
|
+
def to_dict(self) -> dict:
|
|
39
|
+
return {
|
|
40
|
+
"rule": self.rule,
|
|
41
|
+
"severity": self.severity,
|
|
42
|
+
"file": self.file,
|
|
43
|
+
"line": self.line,
|
|
44
|
+
"message": self.message,
|
|
45
|
+
"fix": self.fix,
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class Rule:
|
|
51
|
+
id: str
|
|
52
|
+
severity: str
|
|
53
|
+
pattern: re.Pattern
|
|
54
|
+
message: str
|
|
55
|
+
fix: str
|
|
56
|
+
extensions: tuple = ALL_EXTS
|
|
57
|
+
test_files_only: bool = True
|
|
58
|
+
# Precision valve: when downgrade_if matches the same line, the finding
|
|
59
|
+
# drops to downgrade_to with a note instead of firing at full severity.
|
|
60
|
+
downgrade_if: re.Pattern | None = None
|
|
61
|
+
downgrade_to: str = "info"
|
|
62
|
+
downgrade_note: str = ""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
RULES: list[Rule] = [
|
|
66
|
+
Rule(
|
|
67
|
+
"FLK001", "warning",
|
|
68
|
+
re.compile(r"\btime\.sleep\s*\("),
|
|
69
|
+
"time.sleep() in tests makes timing-dependent flakes",
|
|
70
|
+
"Replace with polling (wait_for / waitFor / eventually) with a timeout, or freeze time.",
|
|
71
|
+
),
|
|
72
|
+
Rule(
|
|
73
|
+
"FLK002", "warning",
|
|
74
|
+
re.compile(r"\b(datetime\.now|datetime\.utcnow|time\.time|Date\.now|new Date\(\))\s*\("),
|
|
75
|
+
"Unfrozen wall-clock reads make tests date/time dependent",
|
|
76
|
+
"Freeze time (freezegun, jest.useFakeTimers, MockDate) or inject a clock.",
|
|
77
|
+
),
|
|
78
|
+
Rule(
|
|
79
|
+
"FLK003", "warning",
|
|
80
|
+
re.compile(r"\b(random\.(random|randint|choice|shuffle|uniform)|Math\.random\s*\()"),
|
|
81
|
+
"Unseeded randomness makes failures unreproducible",
|
|
82
|
+
"Seed the RNG (random.seed / faker.seed) and log the seed on failure.",
|
|
83
|
+
),
|
|
84
|
+
Rule(
|
|
85
|
+
"FLK004", "warning",
|
|
86
|
+
re.compile(r"\b(requests\.(get|post|put|delete)|urllib\.request\.urlopen|fetch\s*\(|axios\.(get|post)|http\.(get|request))\s*\("),
|
|
87
|
+
"Real network calls in tests fail without mocks and leak outside hermetic CI",
|
|
88
|
+
"Mock the boundary (responses, nock, MSW, httpretty) or mark the test integration-only.",
|
|
89
|
+
ALL_EXTS, True,
|
|
90
|
+
re.compile(r"(localhost|127\.0\.0\.1|::1)"),
|
|
91
|
+
"info",
|
|
92
|
+
" (localhost server — hermetic but still a startup-race/port source)",
|
|
93
|
+
),
|
|
94
|
+
Rule(
|
|
95
|
+
"FLK005", "info",
|
|
96
|
+
re.compile(r"for\s+\w+\s+in\s+set\s*\("),
|
|
97
|
+
"Iterating a set() has nondeterministic order across runs",
|
|
98
|
+
"Iterate over sorted(...) if order matters, or compare as sets.",
|
|
99
|
+
),
|
|
100
|
+
Rule(
|
|
101
|
+
"FLK006", "info",
|
|
102
|
+
re.compile(r"(localhost|127\.0\.0\.1)\s*:\s*\d{2,5}|port\s*=\s*\d{2,5}"),
|
|
103
|
+
"Hardcoded ports collide on shared/parallel CI runners",
|
|
104
|
+
"Bind port 0 / let the OS pick, then read back the assigned port.",
|
|
105
|
+
),
|
|
106
|
+
Rule(
|
|
107
|
+
"FLK007", "warning",
|
|
108
|
+
re.compile(r"(\.only\s*\(|describe\.only|it\.only|test\.only|@pytest\.mark\.skip\s*\(\s*\)|@unittest\.skip\s*\(\s*\))"),
|
|
109
|
+
".only / bare skip markers suggest a suite that is already being triaged around flakes",
|
|
110
|
+
"Quarantine the flaky test explicitly (unflake quarantine) instead of .only/skip.",
|
|
111
|
+
),
|
|
112
|
+
Rule(
|
|
113
|
+
"FLK008", "info",
|
|
114
|
+
re.compile(r"(concurrent|Parallel|Promise\.all|asyncio\.gather|ThreadPool|go\s+func)"),
|
|
115
|
+
"Concurrency without deterministic synchronization is a classic flake source",
|
|
116
|
+
"Join/await explicitly, avoid asserting on racy intermediate state.",
|
|
117
|
+
ALL_EXTS,
|
|
118
|
+
),
|
|
119
|
+
Rule(
|
|
120
|
+
"FLK009", "warning",
|
|
121
|
+
re.compile(r"cy\.wait\(\s*\d+"),
|
|
122
|
+
"cy.wait(ms) fixed waits pass locally and flake on loaded CI runners",
|
|
123
|
+
"Wait on state instead: cy.intercept() + cy.wait('@alias'), or .should() retry-ability.",
|
|
124
|
+
JS_EXTS,
|
|
125
|
+
),
|
|
126
|
+
Rule(
|
|
127
|
+
"FLK010", "warning",
|
|
128
|
+
re.compile(r"(waitForTimeout|wait_for_timeout)\s*\("),
|
|
129
|
+
"Fixed-time waits (waitForTimeout / wait_for_timeout) are Playwright's documented flake factory",
|
|
130
|
+
"Use web-first assertions: await expect(locator).toBeVisible() auto-retries.",
|
|
131
|
+
JS_EXTS + PY_EXTS,
|
|
132
|
+
),
|
|
133
|
+
]
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _is_test_file(path: Path) -> bool:
|
|
137
|
+
s = str(path)
|
|
138
|
+
return bool(TEST_FILE_HINT.search(path.name) or TEST_DIR_HINT.search(s))
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
DEFAULT_EXCLUDES = (".git", "node_modules", "__pycache__", ".venv", "venv",
|
|
142
|
+
"dist", "build", ".tox", ".pytest_cache", ".eggs", "*.egg-info")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _string_spans(text: str) -> dict[int, list[tuple[int, int]]]:
|
|
146
|
+
"""Map 1-based lineno -> string-literal column spans, via stdlib tokenize.
|
|
147
|
+
|
|
148
|
+
Matches fully inside a literal are fixture text, not executed code.
|
|
149
|
+
Python only; unparseable files yield {} (no skipping — never hide signal).
|
|
150
|
+
"""
|
|
151
|
+
spans: dict[int, list[tuple[int, int]]] = {}
|
|
152
|
+
try:
|
|
153
|
+
toks = tokenize.generate_tokens(io.StringIO(text).readline)
|
|
154
|
+
for tok in toks:
|
|
155
|
+
if tok.type not in (tokenize.STRING, getattr(tokenize, "FSTRING_MIDDLE", -1)):
|
|
156
|
+
continue
|
|
157
|
+
(srow, scol), (erow, ecol) = tok.start, tok.end
|
|
158
|
+
for r in range(srow, erow + 1):
|
|
159
|
+
spans.setdefault(r, []).append(
|
|
160
|
+
(scol if r == srow else 0, ecol if r == erow else 10 ** 9))
|
|
161
|
+
except (tokenize.TokenError, SyntaxError, IndentationError):
|
|
162
|
+
return {}
|
|
163
|
+
return spans
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _in_literal(spans: dict[int, list[tuple[int, int]]],
|
|
167
|
+
lineno: int, start: int, end: int) -> bool:
|
|
168
|
+
return any(a <= start and end <= b for a, b in spans.get(lineno, []))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _strip_code_fences(lines: list[str]) -> list[bool]:
|
|
172
|
+
"""Return mask: True if the line is live code (not inside a markdown fence)."""
|
|
173
|
+
live = []
|
|
174
|
+
in_fence = False
|
|
175
|
+
for line in lines:
|
|
176
|
+
if line.strip().startswith(CODE_FENCE):
|
|
177
|
+
in_fence = not in_fence
|
|
178
|
+
live.append(False)
|
|
179
|
+
else:
|
|
180
|
+
live.append(not in_fence)
|
|
181
|
+
return live
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def scan_file(path: str | Path) -> list[Finding]:
|
|
185
|
+
path = Path(path)
|
|
186
|
+
if path.suffix not in SCAN_EXTENSIONS:
|
|
187
|
+
return []
|
|
188
|
+
try:
|
|
189
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
190
|
+
except OSError:
|
|
191
|
+
return []
|
|
192
|
+
lines = text.splitlines()
|
|
193
|
+
live = _strip_code_fences(lines) if path.suffix == ".md" else [True] * len(lines)
|
|
194
|
+
is_test = _is_test_file(path)
|
|
195
|
+
literals = _string_spans(text) if path.suffix == ".py" else {}
|
|
196
|
+
findings: list[Finding] = []
|
|
197
|
+
for rule in RULES:
|
|
198
|
+
if rule.test_files_only and not is_test:
|
|
199
|
+
continue
|
|
200
|
+
if path.suffix not in rule.extensions:
|
|
201
|
+
continue
|
|
202
|
+
for i, line in enumerate(lines):
|
|
203
|
+
if not live[i]:
|
|
204
|
+
continue
|
|
205
|
+
m = rule.pattern.search(line)
|
|
206
|
+
if not m:
|
|
207
|
+
continue
|
|
208
|
+
if literals and _in_literal(literals, i + 1, m.start(), m.end()):
|
|
209
|
+
continue # fixture text in a literal, not executed code
|
|
210
|
+
severity, message = rule.severity, rule.message
|
|
211
|
+
if rule.downgrade_if and rule.downgrade_if.search(line):
|
|
212
|
+
severity, message = rule.downgrade_to, message + rule.downgrade_note
|
|
213
|
+
findings.append(Finding(
|
|
214
|
+
rule=rule.id, severity=severity,
|
|
215
|
+
file=str(path), line=i + 1,
|
|
216
|
+
message=message, fix=rule.fix,
|
|
217
|
+
))
|
|
218
|
+
return sorted(findings, key=lambda f: (f.file, f.line))
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _excluded(path: Path, patterns: tuple[str, ...]) -> bool:
|
|
222
|
+
return any(fnmatch.fnmatchcase(part, pat)
|
|
223
|
+
for part in path.parts for pat in patterns)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def scan_path(target: str | Path,
|
|
227
|
+
exclude: list[str] | tuple[str, ...] | None = None) -> list[Finding]:
|
|
228
|
+
"""Scan a file or directory recursively, skipping DEFAULT_EXCLUDES plus extras."""
|
|
229
|
+
patterns = tuple(DEFAULT_EXCLUDES) + tuple(exclude or ())
|
|
230
|
+
target = Path(target)
|
|
231
|
+
files: list[Path] = []
|
|
232
|
+
if target.is_file():
|
|
233
|
+
files = [target]
|
|
234
|
+
elif target.is_dir():
|
|
235
|
+
for ext in SCAN_EXTENSIONS:
|
|
236
|
+
files.extend(target.rglob(f"*{ext}"))
|
|
237
|
+
files = [f for f in files if not _excluded(f, patterns)]
|
|
238
|
+
else:
|
|
239
|
+
raise ValueError(f"{target}: no such file or directory")
|
|
240
|
+
findings: list[Finding] = []
|
|
241
|
+
for f in sorted(files):
|
|
242
|
+
findings.extend(scan_file(f))
|
|
243
|
+
return findings
|
unflake/quarantine.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Quarantine emitters: keep the build green WITHOUT deleting tests.
|
|
2
|
+
|
|
3
|
+
Philosophy (borrowed from the best CI teams): a quarantined test still
|
|
4
|
+
runs and still reports — it just can't fail the build. Deletion hides
|
|
5
|
+
signal; quarantine preserves it while unblocking everyone else.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
_SAFE_K = re.compile(r"[A-Za-z0-9_:.\-\[\]]+\Z")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _k_name(tid: str) -> tuple[str, bool]:
|
|
16
|
+
"""Short test name for -k expressions. Returns (name, is_safe).
|
|
17
|
+
|
|
18
|
+
Unsafe names (spaces/quotes from parametrization) can't go in -k
|
|
19
|
+
verbatim — callers must verify and prefer --deselect with node ids.
|
|
20
|
+
We never silently mangle: mangling hides signal.
|
|
21
|
+
"""
|
|
22
|
+
name = tid.split("::")[-1].split("[")[0]
|
|
23
|
+
if _SAFE_K.fullmatch(name):
|
|
24
|
+
return name, True
|
|
25
|
+
return name, False
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _pytest_k_expression(ids: list[str]) -> tuple[str, list[str]]:
|
|
29
|
+
shorts, risky = [], []
|
|
30
|
+
for tid in ids:
|
|
31
|
+
name, safe = _k_name(tid)
|
|
32
|
+
shorts.append(name)
|
|
33
|
+
if not safe:
|
|
34
|
+
risky.append(tid)
|
|
35
|
+
return "not " + " and not ".join(shorts), risky
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def emit(framework: str, flaky_ids: list[str]) -> str:
|
|
39
|
+
fw = framework.lower()
|
|
40
|
+
if fw == "pytest":
|
|
41
|
+
k, risky = _pytest_k_expression(flaky_ids)
|
|
42
|
+
listed = "\n".join(f"# - {tid}" for tid in flaky_ids)
|
|
43
|
+
out = (
|
|
44
|
+
"# Generated by `unflake quarantine --framework pytest` — review, don't blindly commit.\n"
|
|
45
|
+
f"# Quarantined {len(flaky_ids)} flaky test(s):\n{listed}\n\n"
|
|
46
|
+
"# Option A: exclude them from the gating run, keep them in a nightly non-gating run:\n"
|
|
47
|
+
f"pytest -k \"{k}\"\n\n"
|
|
48
|
+
"# Option B: run ONLY the quarantined set (nightly, non-gating, still reporting):\n"
|
|
49
|
+
f"pytest -k \"{' or '.join(t.split('::')[-1].split('[')[0] for t in flaky_ids)}\" --junitxml=quarantine.xml\n"
|
|
50
|
+
)
|
|
51
|
+
if risky:
|
|
52
|
+
out += ("\n# WARNING: these ids contain spaces/quotes and can't go in -k verbatim —\n"
|
|
53
|
+
"# verify the expression matches, or use `pytest --deselect <nodeid>`:\n"
|
|
54
|
+
+ "".join(f"# RISKY: {tid}\n" for tid in risky))
|
|
55
|
+
return out
|
|
56
|
+
if fw in {"jest", "vitest"}:
|
|
57
|
+
arr = ",\n".join(f' "{tid}"' for tid in flaky_ids)
|
|
58
|
+
flag = "--testPathIgnorePatterns" if fw == "jest" else "--exclude"
|
|
59
|
+
return (
|
|
60
|
+
f"// Generated by `unflake quarantine --framework {fw}`\n"
|
|
61
|
+
"// Quarantine list — wire into CI as a non-gating job, keep reporting.\n"
|
|
62
|
+
f"export const QUARANTINED = [\n{arr}\n];\n\n"
|
|
63
|
+
f"// CI tip ({fw}): run the gating suite excluding quarantine files,\n"
|
|
64
|
+
f"// then run quarantine separately with `{flag}` inverted so it still reports.\n"
|
|
65
|
+
)
|
|
66
|
+
if fw == "playwright":
|
|
67
|
+
joined = "|".join(t.split("::")[-1].split("[")[0] for t in flaky_ids)
|
|
68
|
+
return (
|
|
69
|
+
"# Generated by `unflake quarantine --framework playwright`\n"
|
|
70
|
+
f"# Gating run excludes the {len(flaky_ids)} quarantined test(s):\n"
|
|
71
|
+
f"npx playwright test --grep-invert \"{joined}\"\n\n"
|
|
72
|
+
"# Nightly non-gating run (still reporting):\n"
|
|
73
|
+
f"npx playwright test --grep \"{joined}\" --reporter=junit,html\n"
|
|
74
|
+
)
|
|
75
|
+
raise ValueError(f"unknown framework {framework!r} (try: pytest, jest, vitest, playwright)")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
FRAMEWORKS = ("pytest", "jest", "vitest", "playwright")
|
unflake/report.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Report formatters: text (human), json (machine), sarif (code scanning)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
from .patterns import RULES, Finding
|
|
8
|
+
from .score import SuiteScore
|
|
9
|
+
|
|
10
|
+
ORDER = {"error": 0, "warning": 1, "info": 2}
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _sarif_rules() -> list[dict]:
|
|
14
|
+
rules = [{
|
|
15
|
+
"id": f"unflake/{r.id}",
|
|
16
|
+
"shortDescription": {"text": r.message},
|
|
17
|
+
"help": {"text": r.fix},
|
|
18
|
+
"defaultConfiguration": {"level": {"error": "error", "warning": "warning"}.get(r.severity, "note")},
|
|
19
|
+
} for r in RULES]
|
|
20
|
+
rules += [
|
|
21
|
+
{"id": "unflake/FLAKY",
|
|
22
|
+
"shortDescription": {"text": "Mixed outcomes across repeated runs — quarantine candidate."},
|
|
23
|
+
"defaultConfiguration": {"level": "warning"}},
|
|
24
|
+
{"id": "unflake/REGRESSION",
|
|
25
|
+
"shortDescription": {"text": "Green-then-red-forever — real failure, never quarantine."},
|
|
26
|
+
"defaultConfiguration": {"level": "error"}},
|
|
27
|
+
]
|
|
28
|
+
return rules
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def format_scan_text(findings: list[Finding]) -> str:
|
|
32
|
+
if not findings:
|
|
33
|
+
return "unflake scan: clean — no flake anti-patterns found."
|
|
34
|
+
lines = [f"unflake scan: {len(findings)} finding(s)"]
|
|
35
|
+
for f in sorted(findings, key=lambda x: (ORDER[x.severity], x.file, x.line)):
|
|
36
|
+
lines.append(f" {f.severity.upper():7} [{f.rule}] {f.file}:{f.line}")
|
|
37
|
+
lines.append(f" {f.message}")
|
|
38
|
+
if f.fix:
|
|
39
|
+
lines.append(f" fix: {f.fix}")
|
|
40
|
+
return "\n".join(lines)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def format_scan_json(findings: list[Finding]) -> str:
|
|
44
|
+
return json.dumps({"findings": [f.to_dict() for f in findings]}, indent=2)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def format_sarif(findings: list[Finding], suite: SuiteScore | None = None) -> str:
|
|
48
|
+
results = []
|
|
49
|
+
for f in findings:
|
|
50
|
+
level = {"error": "error", "warning": "warning", "info": "note"}[f.severity]
|
|
51
|
+
results.append({
|
|
52
|
+
"ruleId": f"unflake/{f.rule}",
|
|
53
|
+
"level": level,
|
|
54
|
+
"message": {"text": f"{f.message} Fix: {f.fix}" if f.fix else f.message},
|
|
55
|
+
"locations": [{
|
|
56
|
+
"physicalLocation": {
|
|
57
|
+
"artifactLocation": {"uri": f.file},
|
|
58
|
+
"region": {"startLine": f.line},
|
|
59
|
+
}
|
|
60
|
+
}],
|
|
61
|
+
})
|
|
62
|
+
if suite:
|
|
63
|
+
for t in suite.tests:
|
|
64
|
+
if t.verdict == "flaky":
|
|
65
|
+
results.append({
|
|
66
|
+
"ruleId": "unflake/FLAKY",
|
|
67
|
+
"level": "warning",
|
|
68
|
+
"message": {"text": f"Flaky across {t.runs} runs: "
|
|
69
|
+
f"{t.passes} passed, {t.fails} failed."},
|
|
70
|
+
"locations": [{
|
|
71
|
+
"physicalLocation": {
|
|
72
|
+
"artifactLocation": {"uri": t.id},
|
|
73
|
+
"region": {"startLine": 1},
|
|
74
|
+
}
|
|
75
|
+
}],
|
|
76
|
+
})
|
|
77
|
+
elif t.verdict == "regression":
|
|
78
|
+
results.append({
|
|
79
|
+
"ruleId": "unflake/REGRESSION",
|
|
80
|
+
"level": "error",
|
|
81
|
+
"message": {"text": f"Regression: green until run {t.change_point}, "
|
|
82
|
+
f"red ever since. Fix — do not quarantine."},
|
|
83
|
+
"locations": [{
|
|
84
|
+
"physicalLocation": {
|
|
85
|
+
"artifactLocation": {"uri": t.id},
|
|
86
|
+
"region": {"startLine": 1},
|
|
87
|
+
}
|
|
88
|
+
}],
|
|
89
|
+
})
|
|
90
|
+
return json.dumps({
|
|
91
|
+
"$schema": "https://json.schemastore.org/sarif-2.1.0.json",
|
|
92
|
+
"version": "2.1.0",
|
|
93
|
+
"runs": [{
|
|
94
|
+
"tool": {"driver": {
|
|
95
|
+
"name": "unflake",
|
|
96
|
+
"informationUri": "https://github.com/YOUR-USER/unflake",
|
|
97
|
+
"rules": _sarif_rules(),
|
|
98
|
+
}},
|
|
99
|
+
"results": results,
|
|
100
|
+
}],
|
|
101
|
+
}, indent=2)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def format_analyze_text(suite: SuiteScore) -> str:
|
|
105
|
+
lines = [
|
|
106
|
+
f"FlakeScore: {suite.flake_score}/100 "
|
|
107
|
+
f"({suite.flaky_count} flaky of {suite.total} tests)"
|
|
108
|
+
]
|
|
109
|
+
for t in suite.tests:
|
|
110
|
+
if t.verdict == "flaky":
|
|
111
|
+
lines.append(f" FLAKY {t.id} ({t.passes}P/{t.fails}F over {t.runs} runs)")
|
|
112
|
+
for t in suite.tests:
|
|
113
|
+
if t.verdict == "regression":
|
|
114
|
+
lines.append(f" REGRESSION {t.id} (green until run {t.change_point}, "
|
|
115
|
+
f"red ever since — fix this, do NOT quarantine it)")
|
|
116
|
+
for t in suite.tests:
|
|
117
|
+
if t.verdict == "new":
|
|
118
|
+
lines.append(f" NEW {t.id} (only in latest run — needs more runs before judging)")
|
|
119
|
+
stables = [t for t in suite.tests
|
|
120
|
+
if t.verdict in ("stable-pass", "stable-fail", "skipped")]
|
|
121
|
+
if stables:
|
|
122
|
+
lines.append(f" ... +{len(stables)} stable/skipped test(s) (see --format json for full list)")
|
|
123
|
+
return "\n".join(lines)
|
unflake/runner.py
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""`unflake run`: collect the >=2 runs that flake-hunting requires.
|
|
2
|
+
|
|
3
|
+
Runs your test command N times, collects a JUnit XML per run, then scores.
|
|
4
|
+
With a JUnit flag the per-test analysis works; without one, each run's exit
|
|
5
|
+
code still gives an honest suite-level verdict (stable vs flaky suite).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
import subprocess
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from .ingest import FAIL, PASS
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _detect_junit_flag(cmd: list[str]) -> str | None:
|
|
18
|
+
joined = " ".join(cmd)
|
|
19
|
+
if "pytest" in cmd[0] or "pytest" in joined.split()[0]:
|
|
20
|
+
return "--junitxml="
|
|
21
|
+
return None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# Verified presets: only runners proven end-to-end (see tests/test_e2e.py).
|
|
25
|
+
# Anything else uses --junit-flag explicitly — we'd rather error than guess.
|
|
26
|
+
# Each preset is (argv_fn, env_fn|None): some runners take the JUnit path
|
|
27
|
+
# as a flag (pytest, vitest), others as an env var (playwright).
|
|
28
|
+
def _pytest_args(dest: str) -> list[str]:
|
|
29
|
+
return [f"--junitxml={dest}"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _vitest_args(dest: str) -> list[str]:
|
|
33
|
+
return ["--reporter=junit", f"--outputFile={dest}"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _playwright_args(dest: str) -> list[str]:
|
|
37
|
+
return ["--reporter=junit"]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _playwright_env(dest: str) -> dict[str, str]:
|
|
41
|
+
return {"PLAYWRIGHT_JUNIT_OUTPUT_NAME": dest}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _jest_args(dest: str) -> list[str]:
|
|
45
|
+
# NOTE: space form — `--reporters=jest-junit` breaks reporter parsing.
|
|
46
|
+
return ["--reporters", "jest-junit"]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _jest_env(dest: str) -> dict[str, str]:
|
|
50
|
+
return {"JEST_JUNIT_OUTPUT_FILE": dest}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
RUNNERS = {
|
|
54
|
+
"pytest": (_pytest_args, None),
|
|
55
|
+
"vitest": (_vitest_args, None),
|
|
56
|
+
"playwright": (_playwright_args, _playwright_env),
|
|
57
|
+
"jest": (_jest_args, _jest_env),
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def resolve_junit_flag(runner: str | None, junit_flag: str | None,
|
|
62
|
+
cmd: list[str]) -> str | None:
|
|
63
|
+
"""Explicit --junit-flag wins; --runner pytest/vitest are verified; anything
|
|
64
|
+
else falls back to pytest auto-detect, else exit-code-only mode."""
|
|
65
|
+
if junit_flag:
|
|
66
|
+
return junit_flag
|
|
67
|
+
if runner:
|
|
68
|
+
if runner not in RUNNERS:
|
|
69
|
+
raise ValueError(f"unknown --runner {runner!r} (try: {', '.join(sorted(RUNNERS))})")
|
|
70
|
+
return None # preset handled via runner_args(), not a single flag
|
|
71
|
+
return _detect_junit_flag(cmd)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def runner_args(runner: str | None, dest: str) -> list[str] | None:
|
|
75
|
+
"""Extra argv for a verified preset, or None."""
|
|
76
|
+
if runner and runner in RUNNERS:
|
|
77
|
+
return RUNNERS[runner][0](dest)
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def runner_env(runner: str | None, dest: str) -> dict[str, str]:
|
|
82
|
+
"""Extra env for a verified preset (empty when the path goes via argv)."""
|
|
83
|
+
if runner and runner in RUNNERS:
|
|
84
|
+
fn = RUNNERS[runner][1]
|
|
85
|
+
return dict(fn(dest)) if fn else {}
|
|
86
|
+
return {}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def collect(cmd: list[str], runs: int, out_dir: str | Path,
|
|
90
|
+
junit_flag: str | None = None, runner: str | None = None,
|
|
91
|
+
timeout: float | None = None,
|
|
92
|
+
progress=None) -> list[Path]:
|
|
93
|
+
"""Run cmd N times. Return per-run result files (JUnit XML if possible,
|
|
94
|
+
else generic JSON with one suite-level entry)."""
|
|
95
|
+
out = Path(out_dir)
|
|
96
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
if runner and runner not in RUNNERS:
|
|
98
|
+
raise ValueError(f"unknown --runner {runner!r} (try: {', '.join(sorted(RUNNERS))})")
|
|
99
|
+
single = resolve_junit_flag(runner, junit_flag, cmd)
|
|
100
|
+
flag = single # legacy single-flag path (explicit --junit-flag or pytest auto-detect)
|
|
101
|
+
files: list[Path] = []
|
|
102
|
+
for i in range(1, runs + 1):
|
|
103
|
+
if progress:
|
|
104
|
+
progress(f"run {i}/{runs}: {' '.join(cmd)}")
|
|
105
|
+
log = open(out / f"run{i}.log", "w", encoding="utf-8") # closed below
|
|
106
|
+
try:
|
|
107
|
+
dest = out / f"run{i}.xml"
|
|
108
|
+
preset = runner_args(runner, str(dest))
|
|
109
|
+
if preset is not None:
|
|
110
|
+
full = [*cmd, *preset]
|
|
111
|
+
elif flag:
|
|
112
|
+
full = [*cmd, f"{flag}{dest}"]
|
|
113
|
+
else:
|
|
114
|
+
full = None
|
|
115
|
+
if full is not None:
|
|
116
|
+
run_env = dict(os.environ)
|
|
117
|
+
run_env.update(runner_env(runner, str(dest)))
|
|
118
|
+
try:
|
|
119
|
+
proc = subprocess.run(full, timeout=timeout, env=run_env,
|
|
120
|
+
stdout=log, stderr=subprocess.STDOUT)
|
|
121
|
+
ok = proc.returncode == 0
|
|
122
|
+
except subprocess.TimeoutExpired:
|
|
123
|
+
ok = False
|
|
124
|
+
if not dest.exists():
|
|
125
|
+
# Command swallowed the flag (or never wrote XML):
|
|
126
|
+
# fall back to an exit-code entry so the run still counts.
|
|
127
|
+
dest = _write_json_fallback(out, i, cmd, ok, timed_out=True)
|
|
128
|
+
files.append(dest)
|
|
129
|
+
else:
|
|
130
|
+
try:
|
|
131
|
+
proc = subprocess.run(cmd, timeout=timeout,
|
|
132
|
+
stdout=log, stderr=subprocess.STDOUT)
|
|
133
|
+
ok, timed_out = proc.returncode == 0, False
|
|
134
|
+
except subprocess.TimeoutExpired:
|
|
135
|
+
ok, timed_out = False, True
|
|
136
|
+
files.append(_write_json_fallback(out, i, cmd, ok, timed_out))
|
|
137
|
+
finally:
|
|
138
|
+
log.close()
|
|
139
|
+
return files
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _write_json_fallback(out: Path, i: int, cmd: list[str],
|
|
143
|
+
ok: bool, timed_out: bool = False) -> Path:
|
|
144
|
+
import json
|
|
145
|
+
dest = out / f"run{i}.json"
|
|
146
|
+
status = PASS if ok else FAIL
|
|
147
|
+
note = "suite exit 0" if ok else ("suite timed out" if timed_out else "suite exit != 0")
|
|
148
|
+
dest.write_text(json.dumps({"tests": [
|
|
149
|
+
{"id": f"suite::{' '.join(cmd)}", "status": status, "note": note},
|
|
150
|
+
]}), encoding="utf-8")
|
|
151
|
+
return dest
|
unflake/scaffold.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""`unflake init`: scaffold the two cheapest flake preventions.
|
|
2
|
+
|
|
3
|
+
1. Seeded RNG + logged seed (unseeded randomness is unreproducible by definition).
|
|
4
|
+
2. A frozen-time pointer (wall clocks are the #2 classic).
|
|
5
|
+
|
|
6
|
+
`--write` creates files idempotently (marker-guarded, never duplicates).
|
|
7
|
+
Without `--write`, snippets print to stdout with wiring instructions.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
MARKER = "# unflake:seed-logging"
|
|
15
|
+
|
|
16
|
+
PYTEST_SNIPPET = '''%(marker)s (added by `unflake init --framework pytest --write`)
|
|
17
|
+
# Deterministic RNG per test + logged seed: rerun any failure exactly with
|
|
18
|
+
# UNFLAKE_SEED=<seed> pytest ...
|
|
19
|
+
import os
|
|
20
|
+
import random
|
|
21
|
+
|
|
22
|
+
import pytest
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@pytest.fixture(autouse=True)
|
|
26
|
+
def _unflake_seed(request):
|
|
27
|
+
seed = os.environ.get("UNFLAKE_SEED", "0")
|
|
28
|
+
random.seed(f"{seed}-{request.node.nodeid}")
|
|
29
|
+
print(f"\\n[unflake] seed={seed} test={request.node.nodeid}")
|
|
30
|
+
yield
|
|
31
|
+
'''
|
|
32
|
+
|
|
33
|
+
JEST_SNIPPET = '''// unflake:seed-logging (added by `unflake init --framework jest`)
|
|
34
|
+
// Deterministic RNG per test file. Wire into jest config:
|
|
35
|
+
// { setupFiles: ["<rootDir>/unflake.setup.cjs"] }
|
|
36
|
+
const seed = process.env.UNFLAKE_SEED || "0";
|
|
37
|
+
let s = [...seed].reduce((a, c) => (a * 33 + c.charCodeAt(0)) >>> 0, 7) || 7;
|
|
38
|
+
const file = expect.getState().testPath || "unknown";
|
|
39
|
+
console.log(`[unflake] seed=${seed} file=${file}`);
|
|
40
|
+
Math.random = () => (s = (s * 1103515245 + 12345) & 0x7fffffff) / 0x7fffffff;
|
|
41
|
+
'''
|
|
42
|
+
|
|
43
|
+
VITEST_SNIPPET = '''// unflake:seed-logging (added by `unflake init --framework vitest`)
|
|
44
|
+
// Wire into vitest config: export default { test: { setupFiles: ["./unflake.setup.js"] } }
|
|
45
|
+
const seed = process.env.UNFLAKE_SEED || "0";
|
|
46
|
+
console.log(`[unflake] seed=${seed}`);
|
|
47
|
+
'''
|
|
48
|
+
|
|
49
|
+
PLAYWRIGHT_SNIPPET = '''// unflake:seed-logging (added by `unflake init --framework playwright`)
|
|
50
|
+
// Playwright has built-in retries — prefer them over fixed waits:
|
|
51
|
+
// { retries: 2, use: { trace: "retain-on-failure" } }
|
|
52
|
+
// And NEVER page.waitForTimeout(): use web-first assertions:
|
|
53
|
+
// await expect(page.getByRole("button")).toBeVisible();
|
|
54
|
+
'''
|
|
55
|
+
|
|
56
|
+
SNIPPETS = {
|
|
57
|
+
"pytest": ("tests/conftest.py", PYTEST_SNIPPET),
|
|
58
|
+
"jest": ("unflake.setup.cjs", JEST_SNIPPET),
|
|
59
|
+
"vitest": ("unflake.setup.js", VITEST_SNIPPET),
|
|
60
|
+
"playwright": ("unflake.playwright-notes.js", PLAYWRIGHT_SNIPPET),
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
FRAMEWORKS = tuple(SNIPPETS)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def snippet(framework: str) -> tuple[str, str]:
|
|
67
|
+
"""Return (relative_path, content) for a framework."""
|
|
68
|
+
if framework not in SNIPPETS:
|
|
69
|
+
raise ValueError(f"unknown framework {framework!r} (try: {', '.join(FRAMEWORKS)})")
|
|
70
|
+
rel, template = SNIPPETS[framework]
|
|
71
|
+
return rel, template % {"marker": MARKER}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def init_framework(framework: str, root: str | Path = ".", write: bool = False) -> str:
|
|
75
|
+
"""Print or write the scaffold. Returns a human summary."""
|
|
76
|
+
rel, content = snippet(framework)
|
|
77
|
+
if not write:
|
|
78
|
+
return f"# {rel} — create this file with the content below:\n\n{content}"
|
|
79
|
+
dest = Path(root) / rel
|
|
80
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
if dest.exists():
|
|
82
|
+
existing = dest.read_text(encoding="utf-8", errors="replace")
|
|
83
|
+
if MARKER in existing or "unflake:seed-logging" in existing:
|
|
84
|
+
return f"{dest}: already scaffolded — left untouched."
|
|
85
|
+
with dest.open("a", encoding="utf-8") as fh:
|
|
86
|
+
fh.write("\n" + content)
|
|
87
|
+
return f"{dest}: snippet appended (existing file preserved)."
|
|
88
|
+
dest.write_text(content, encoding="utf-8")
|
|
89
|
+
return f"{dest}: created. Rerun any failure exactly with UNFLAKE_SEED=<seed>."
|
unflake/score.py
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Flakiness scoring across repeated runs.
|
|
2
|
+
|
|
3
|
+
Pass result files oldest -> newest. Verdicts per test:
|
|
4
|
+
|
|
5
|
+
stable-pass always green
|
|
6
|
+
stable-fail always red — a real failure, NOT a flake, never quarantine
|
|
7
|
+
flaky mixed outcomes with no clean break — quarantine candidate
|
|
8
|
+
regression green, then red from some run on and never green again
|
|
9
|
+
(needs >= 3 runs; with 2 runs a pass->fail looks identical
|
|
10
|
+
to a flake, so it stays flaky) — fix, never quarantine
|
|
11
|
+
new only appeared in the latest run — needs more runs
|
|
12
|
+
skipped never produced a pass/fail
|
|
13
|
+
|
|
14
|
+
Suite FlakeScore: 0-100, share of non-flaky tests (higher = healthier).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
|
|
21
|
+
from .ingest import FAIL, PASS, SKIP
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class TestScore:
|
|
26
|
+
id: str
|
|
27
|
+
passes: int
|
|
28
|
+
fails: int
|
|
29
|
+
skips: int
|
|
30
|
+
runs: int
|
|
31
|
+
verdict: str
|
|
32
|
+
flakiness: float # 0.0 stable .. 1.0 maximally flaky (50/50 split)
|
|
33
|
+
change_point: int | None = None # 1-based run where a regression starts
|
|
34
|
+
|
|
35
|
+
def to_dict(self) -> dict:
|
|
36
|
+
return {
|
|
37
|
+
"id": self.id, "passes": self.passes, "fails": self.fails,
|
|
38
|
+
"skips": self.skips, "runs": self.runs,
|
|
39
|
+
"verdict": self.verdict, "flakiness": round(self.flakiness, 3),
|
|
40
|
+
"change_point": self.change_point,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass
|
|
45
|
+
class SuiteScore:
|
|
46
|
+
tests: list[TestScore]
|
|
47
|
+
flake_score: float
|
|
48
|
+
flaky_count: int
|
|
49
|
+
regression_count: int
|
|
50
|
+
total: int
|
|
51
|
+
|
|
52
|
+
def to_dict(self) -> dict:
|
|
53
|
+
return {
|
|
54
|
+
"flake_score": round(self.flake_score, 1),
|
|
55
|
+
"flaky_count": self.flaky_count,
|
|
56
|
+
"regression_count": self.regression_count,
|
|
57
|
+
"total": self.total,
|
|
58
|
+
"tests": [t.to_dict() for t in self.tests],
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
def issues(self) -> list[TestScore]:
|
|
62
|
+
"""Actionable tests: flaky first, then regressions."""
|
|
63
|
+
return [t for t in self.tests if t.verdict in ("flaky", "regression")]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _is_clean_break(statuses: list[str]) -> int | None:
|
|
67
|
+
"""If statuses are pass* then fail* (skips ignored), return the 1-based
|
|
68
|
+
index of the first fail. Otherwise None."""
|
|
69
|
+
pf = [s for s in statuses if s in (PASS, FAIL)]
|
|
70
|
+
if len(pf) < 3 or PASS not in pf or FAIL not in pf:
|
|
71
|
+
return None
|
|
72
|
+
first_fail = pf.index(FAIL)
|
|
73
|
+
if first_fail == 0:
|
|
74
|
+
return None # failed from the start — stable-fail territory
|
|
75
|
+
if all(s == FAIL for s in pf[first_fail:]):
|
|
76
|
+
return statuses.index(FAIL) + 1 # 1-based, counting all runs
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def score_runs(runs: list[dict[str, str]]) -> SuiteScore:
|
|
81
|
+
if not runs:
|
|
82
|
+
raise ValueError("need at least one run to score")
|
|
83
|
+
ids: set[str] = set()
|
|
84
|
+
for run in runs:
|
|
85
|
+
ids.update(run.keys())
|
|
86
|
+
tests: list[TestScore] = []
|
|
87
|
+
for tid in sorted(ids):
|
|
88
|
+
statuses = [run.get(tid, "missing") for run in runs]
|
|
89
|
+
present = [s for s in statuses if s != "missing"]
|
|
90
|
+
passes = sum(1 for s in present if s == PASS)
|
|
91
|
+
fails = sum(1 for s in present if s == FAIL)
|
|
92
|
+
skips = sum(1 for s in present if s == SKIP)
|
|
93
|
+
change_point: int | None = None
|
|
94
|
+
if "missing" in statuses:
|
|
95
|
+
if len(present) == 1 and statuses[-1] != "missing":
|
|
96
|
+
verdict, flakiness = "new", 0.0
|
|
97
|
+
else:
|
|
98
|
+
verdict, flakiness = "flaky", 0.5 # appears/disappears
|
|
99
|
+
elif passes and fails:
|
|
100
|
+
cp = _is_clean_break(statuses)
|
|
101
|
+
if cp is not None:
|
|
102
|
+
verdict, flakiness, change_point = "regression", 0.0, cp
|
|
103
|
+
else:
|
|
104
|
+
rate = passes / (passes + fails)
|
|
105
|
+
verdict, flakiness = "flaky", round(1.0 - abs(rate - 0.5) * 2, 3)
|
|
106
|
+
elif fails and not passes:
|
|
107
|
+
verdict, flakiness = ("stable-fail", 0.0) if not skips else ("flaky", 0.3)
|
|
108
|
+
elif passes and not fails:
|
|
109
|
+
verdict, flakiness = ("stable-pass", 0.0) if not skips else ("flaky", 0.2)
|
|
110
|
+
else:
|
|
111
|
+
verdict, flakiness = "skipped", 0.0
|
|
112
|
+
tests.append(TestScore(tid, passes, fails, skips, len(runs),
|
|
113
|
+
verdict, flakiness, change_point))
|
|
114
|
+
flaky = [t for t in tests if t.verdict == "flaky"]
|
|
115
|
+
regressions = [t for t in tests if t.verdict == "regression"]
|
|
116
|
+
total = len(tests) or 1
|
|
117
|
+
flake_score = 100.0 * (total - len(flaky)) / total
|
|
118
|
+
order = {"flaky": 0, "regression": 1, "stable-fail": 2,
|
|
119
|
+
"new": 3, "stable-pass": 4, "skipped": 5}
|
|
120
|
+
tests.sort(key=lambda t: (order.get(t.verdict, 9), -t.flakiness, t.id))
|
|
121
|
+
return SuiteScore(tests, round(flake_score, 1), len(flaky),
|
|
122
|
+
len(regressions), len(tests))
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: unflake
|
|
3
|
+
Version: 0.6.1
|
|
4
|
+
Summary: Green CI without the rerun ritual — detect, score, and quarantine flaky tests
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/YOUR-USER/unflake
|
|
7
|
+
Project-URL: Issues, https://github.com/YOUR-USER/unflake/issues
|
|
8
|
+
Keywords: testing,flaky-tests,ci,qa,agent-skills
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Software Development :: Testing
|
|
14
|
+
Requires-Python: >=3.9
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Dynamic: license-file
|
|
18
|
+
|
|
19
|
+
# Unflake 🟢
|
|
20
|
+
|
|
21
|
+
[](https://github.com/YOUR-USER/unflake/actions/workflows/ci.yml)
|
|
22
|
+
[](https://scorecard.dev/viewer/?uri=github.com/YOUR-USER/unflake)
|
|
23
|
+
[](https://pypi.org/project/unflake/)
|
|
24
|
+
[](LICENSE)
|
|
25
|
+
|
|
26
|
+
**Green CI without the "just rerun it" ritual.**
|
|
27
|
+
|
|
28
|
+

|
|
29
|
+
|
|
30
|
+
CI is red. You rerun. It's green. You merge, trusting nothing. Unflake hunts the flake instead: static scan in milliseconds, FlakeScore across repeated runs, regression-vs-flake verdicts, and quarantine configs that keep the build green *without deleting a single test*.
|
|
31
|
+
|
|
32
|
+

|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install unflake
|
|
36
|
+
unflake scan tests/ # 1. smells, instantly
|
|
37
|
+
unflake run --runs 3 --runner pytest -- pytest tests/ -q # 2. collect + score
|
|
38
|
+
unflake quarantine run1.xml run2.xml --framework pytest # 3. quarantine, don't delete
|
|
39
|
+
unflake init --framework pytest --write # 4. seeded RNG scaffold (prevention)
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
JS-first repo? Same tool, no Python packaging — just Python itself (3.9+):
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
npx unflake-ci scan tests/
|
|
46
|
+
npx unflake-ci run --runs 3 --runner vitest -- npx vitest run tests/
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Proven on real suites: 3,676 tests × 5 runs across attrs/click/tqdm → FlakeScore 100,
|
|
50
|
+
0 false quarantines ([study](benchmarks/BENCH.md)).
|
|
51
|
+
|
|
52
|
+
No install handy? `bash demo.sh` runs the whole loop on fixtures in 60 seconds.
|
|
53
|
+
|
|
54
|
+
> ⭐ If flaky tests have ever paged you, star this — it helps other devs find it.
|
|
55
|
+
|
|
56
|
+
## Before / after
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
# before: the 3am pager
|
|
60
|
+
def test_checkout():
|
|
61
|
+
time.sleep(5) # pray the server is up
|
|
62
|
+
assert requests.get(API).ok # real network in a unit test
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
$ unflake scan tests/
|
|
67
|
+
WARNING [FLK001] tests/test_shop.py:4 — time.sleep() in tests makes timing-dependent flakes
|
|
68
|
+
fix: Replace with polling (wait_for / waitFor / eventually) with a timeout, or freeze time.
|
|
69
|
+
WARNING [FLK004] tests/test_shop.py:5 — Real network calls in tests fail without mocks ...
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
$ unflake run --runs 3 -- pytest tests/ -q
|
|
74
|
+
FlakeScore: 50.0/100 (2 flaky of 4 tests)
|
|
75
|
+
FLAKY test_cart::test_checkout (2P/1F over 3 runs)
|
|
76
|
+
FLAKY test_cart::test_discount (2P/1F over 3 runs)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
$ unflake quarantine run1.xml run2.xml --framework pytest
|
|
81
|
+
pytest -k "not test_checkout and not test_discount" # gating run stays green
|
|
82
|
+
pytest -k "test_checkout or test_discount" # nightly run still reports
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Quarantine preserves signal; deletion hides it. Every quarantined test still runs — it just can't fail the build. **Regressions are never quarantined**: green-then-red-forever is a real failure — fix it.
|
|
86
|
+
|
|
87
|
+
## Flaky or regression? Unflake knows the difference
|
|
88
|
+
|
|
89
|
+
Pass result files oldest → newest and Unflake separates three fates:
|
|
90
|
+
|
|
91
|
+
| Verdict | Meaning | Action |
|
|
92
|
+
|---|---|---|
|
|
93
|
+
| `FLAKY` | mixed outcomes, no clean break | quarantine + fix root cause |
|
|
94
|
+
| `REGRESSION` | green until run N, red ever since | fix now — never quarantine |
|
|
95
|
+
| `NEW` / stable | seen once / always green / always red | more runs / ship it / real bug |
|
|
96
|
+
|
|
97
|
+
Two runs can only suggest a flake; calling a regression needs ≥3. One run proves nothing — `unflake run` exists so there's no excuse.
|
|
98
|
+
|
|
99
|
+
## Why not the others?
|
|
100
|
+
|
|
101
|
+
| | Unflake | Heavy flake platforms |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| Install | zero-dep, stdlib only, `pip install unflake` | OTel collectors, dashboards, SaaS |
|
|
104
|
+
| Input | JUnit XML from **any** framework (pytest ✅, Vitest ✅, Playwright ✅, Jest ✅ e2e — plus generic `--junit-flag`, Go, JUnit…) | per-framework reporters/plugins |
|
|
105
|
+
| First value | `scan` in milliseconds, no execution | needs history ingestion |
|
|
106
|
+
| Verdicts | flaky vs regression vs new, change-point included | usually just a score |
|
|
107
|
+
| Philosophy | quarantine, never delete | often auto-skip-and-forget |
|
|
108
|
+
| Agent-native | `SKILL.md` + harness adapters day one | docs page, if you're lucky |
|
|
109
|
+
|
|
110
|
+
## Works with your harness
|
|
111
|
+
|
|
112
|
+
Claude Code · Codex · Copilot · Cursor · Gemini · Pi · OpenCode · Windsurf · Cline · Qoder — plus a GitHub Action (`action.yml`, SARIF → code scanning) and a pre-commit hook. Details: [`docs/ADAPTERS.md`](docs/ADAPTERS.md). The skill lives at [`skills/unflake/SKILL.md`](skills/unflake/SKILL.md).
|
|
113
|
+
|
|
114
|
+
10 rules today, incl. the JS classics: `cy.wait(ms)` (FLK009) and `waitForTimeout` (FLK010).
|
|
115
|
+
Precision features: localhost-aware severities, string-literal awareness (Python),
|
|
116
|
+
`--exclude` globs, and SARIF rule metadata for code scanning.
|
|
117
|
+
|
|
118
|
+
## Benchmarks (honest, reproducible)
|
|
119
|
+
|
|
120
|
+
Fixtures + real pytest e2e + a real-repo scan (`psf/requests`: 120 findings in ~0.1s — with the noise documented, not hidden). Method, numbers, and known limitations: [`benchmarks/BENCH.md`](benchmarks/BENCH.md). If a number doesn't reproduce, that's a bug — file it.
|
|
121
|
+
|
|
122
|
+
## Contributing — built for drive-by PRs
|
|
123
|
+
|
|
124
|
+
- 🧪 New FLK rule = one regex + one test (`good first issue`)
|
|
125
|
+
- 🔌 New harness adapter or quarantine emitter = one file + one test
|
|
126
|
+
- 🌱 `unflake init` for your framework = one snippet
|
|
127
|
+
- 🌍 Translations welcome (`README.<lang>.md`)
|
|
128
|
+
- Full loop: [`CONTRIBUTING.md`](CONTRIBUTING.md)
|
|
129
|
+
|
|
130
|
+
## Roadmap
|
|
131
|
+
|
|
132
|
+
- [x] v0.1 — scan / analyze / quarantine, SARIF, 4 frameworks, skill + adapters
|
|
133
|
+
- [x] v0.2 — `run` collector, regression-vs-flake verdicts, `init` scaffolder, FLK009/FLK010, pre-commit + GitHub Action, `pip install unflake`
|
|
134
|
+
- [x] v0.3 — `--runner` presets, per-run logs, real-pytest e2e in CI, launch kit (`demo.sh`, templates, `LAUNCH.md`), real-repo validation
|
|
135
|
+
- [x] v0.4 — precision (localhost-aware FLK004, literal-awareness), `--exclude`, SARIF rules, safe `-k` quoting, CI matrix 3.10–3.14
|
|
136
|
+
- [x] v0.5 — `unflake-ci` npx wrapper (verified pack→install→run), `--runner vitest` verified e2e
|
|
137
|
+
- [x] v0.6 — `--runner` playwright + jest verified e2e, real-repo study (3,676 tests, 0 false quarantines)
|
|
138
|
+
- [ ] Recall study v2 on suites with known flakes (specificity proven; recall needs wild flakes)
|
|
139
|
+
- [ ] JS/TS string-literal awareness + variable-host local-server detection (see BENCH.md)
|
|
140
|
+
|
|
141
|
+
## License
|
|
142
|
+
|
|
143
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
unflake/__init__.py,sha256=lns3NRcFIEftdwyjj6dzFeW1GyuI542ru-p2POnskTs,94
|
|
2
|
+
unflake/__main__.py,sha256=HlOZgR1S8N_FcV4UyArFybJgDBsd3tCfdOvbRzc5cww,447
|
|
3
|
+
unflake/cli.py,sha256=p-d_LPLABdhTQTaTajZHidSq5TEM_o1XCWhb3nPuWyo,8442
|
|
4
|
+
unflake/ingest.py,sha256=KkjDiJWvvGshliFGS5gPcp2PR4RYyBudKFDz_jZbvC8,3289
|
|
5
|
+
unflake/patterns.py,sha256=ZFomxU28yT1G1qKE-Ky73QLJjSjM5rWOkvme1xS2Fng,8942
|
|
6
|
+
unflake/quarantine.py,sha256=PoPpNHb5csPrcvdR78-7CgFPskZBH5Rm8WmwODNJ4lg,3431
|
|
7
|
+
unflake/report.py,sha256=ciIhbZMwRJVUqtrnm-UWEjCQj4yi7l8nHOy1g4hI97Y,4837
|
|
8
|
+
unflake/runner.py,sha256=tliEh2vf-FDMpXvTjpi1U6SsqJT_Fk96LEKzxj7GiUM,5659
|
|
9
|
+
unflake/scaffold.py,sha256=OiuHoaWMfB7oICiRbyJdqxlZQW-D8ydtBLsecfFjSek,3537
|
|
10
|
+
unflake/score.py,sha256=sLACgHvJpw8oW8-UTJp38QO-8PITqiUZJjVjF82xTPs,4672
|
|
11
|
+
unflake-0.6.1.dist-info/licenses/LICENSE,sha256=61Ot6dNjcmx1kKVNfniNm1PTbNF4N7zmNjxJ_H7VCao,1077
|
|
12
|
+
unflake-0.6.1.dist-info/METADATA,sha256=H9CXyanmYd5DhhX_l96Tts6RG_SemSHioRMT-2EliQg,7063
|
|
13
|
+
unflake-0.6.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
14
|
+
unflake-0.6.1.dist-info/entry_points.txt,sha256=BRz8fLVpu72NT5t1fgFH-tqpb_8qqW6e8tcup6wo8HY,45
|
|
15
|
+
unflake-0.6.1.dist-info/top_level.txt,sha256=m9wa0x1Y0vt-lPoM1gt7ihlXwbdqARUn2oRUdv7hZEQ,8
|
|
16
|
+
unflake-0.6.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Unflake Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
unflake
|