impact-gate 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- impact_gate/__init__.py +8 -0
- impact_gate/baseline.py +208 -0
- impact_gate/cli.py +201 -0
- impact_gate/config.py +123 -0
- impact_gate/core/__init__.py +10 -0
- impact_gate/core/config.py +102 -0
- impact_gate/core/gitplumb.py +153 -0
- impact_gate/core/impact.py +164 -0
- impact_gate/core/lizard_plugin.py +47 -0
- impact_gate/core/units.py +42 -0
- impact_gate/data/__init__.py +108 -0
- impact_gate/data/defaults.json +31 -0
- impact_gate/data/seed_percentiles.json +120 -0
- impact_gate/engine.py +122 -0
- impact_gate/ghapi.py +94 -0
- impact_gate/gitio.py +189 -0
- impact_gate/providers.py +163 -0
- impact_gate/report.py +173 -0
- impact_gate-0.1.0.dist-info/METADATA +194 -0
- impact_gate-0.1.0.dist-info/RECORD +24 -0
- impact_gate-0.1.0.dist-info/WHEEL +5 -0
- impact_gate-0.1.0.dist-info/entry_points.txt +2 -0
- impact_gate-0.1.0.dist-info/licenses/LICENSE +201 -0
- impact_gate-0.1.0.dist-info/top_level.txt +1 -0
impact_gate/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""impact-gate — report the change-impact of a change and gate on it.
|
|
2
|
+
|
|
3
|
+
A lightweight front-end over the change-impact measure (shared, single-sourced with
|
|
4
|
+
Surveyor / the PetClinic-Evolve harness). It scores a change against a base — a
|
|
5
|
+
committed range vs `main`, staged changes, or the working tree — reports the number,
|
|
6
|
+
and warns or blocks when the impact is too high.
|
|
7
|
+
"""
|
|
8
|
+
__version__ = "0.1.0"
|
impact_gate/baseline.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""The project baseline: the repo's own per-commit impact distribution, and the
|
|
2
|
+
empirical-Bayes blend of it with the shipped seed prior.
|
|
3
|
+
|
|
4
|
+
Scope. The walk starts from the merged base branch (default `main`) and follows its
|
|
5
|
+
first-parent mainline, so only landed work is measured; an unmerged in-flight branch is
|
|
6
|
+
never reached, because the walk only ever follows real merge commits.
|
|
7
|
+
|
|
8
|
+
One observation = one atomic landed change:
|
|
9
|
+
* a main non-merge commit -> its own diff vs its parent;
|
|
10
|
+
* a leaf MR (a merged branch with no MRs inside it) -> the net change from the branch
|
|
11
|
+
start (merge-base of the merge's parents) to the branch tip;
|
|
12
|
+
* a parent MR (a merged branch that contains MRs) -> its direct commits netted per run
|
|
13
|
+
between the merges on its spine, each run based at the preceding synced base; every
|
|
14
|
+
child MR recurses, and a merge that only syncs the parent/main branch in is skipped.
|
|
15
|
+
A merge is never scored as its own diff, so a long-running roll-up cannot inflate things.
|
|
16
|
+
|
|
17
|
+
Each observation is one composite impact = (Σ mutation + Σ godclass) * files. The
|
|
18
|
+
distribution of that over all observations is the baseline. The gate always grades the
|
|
19
|
+
composite; mutation cost stays a per-file ranking signal (report side), never a rival
|
|
20
|
+
distribution to gate against.
|
|
21
|
+
|
|
22
|
+
Grading. A change's grade blends its percentile rank in this project distribution (n
|
|
23
|
+
observations) with its rank against the per-language seed table, weighting the project by
|
|
24
|
+
w = n / (n + K)
|
|
25
|
+
so a shallow history leans on the seed and a deep one trusts itself. K is curve_prior_weight.
|
|
26
|
+
"""
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import bisect
|
|
30
|
+
import json
|
|
31
|
+
import re
|
|
32
|
+
from dataclasses import dataclass, field
|
|
33
|
+
|
|
34
|
+
from . import gitio
|
|
35
|
+
from .core.config import MeasureConfig
|
|
36
|
+
from .core.gitplumb import GitRepo
|
|
37
|
+
from .data import load_defaults, rank_in_table, seed_table
|
|
38
|
+
from .engine import score_change
|
|
39
|
+
|
|
40
|
+
_BD = load_defaults().get("baseline", {})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# --------------------------------------------------------------------------- model
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class Baseline:
|
|
47
|
+
"""The project's composite per-observation distribution (ascending)."""
|
|
48
|
+
n: int
|
|
49
|
+
dist: list[int] = field(default_factory=list)
|
|
50
|
+
head: str | None = None
|
|
51
|
+
base_ref: str | None = None
|
|
52
|
+
|
|
53
|
+
def rank(self, value: float) -> float:
|
|
54
|
+
return _percentile_rank(self.dist, value)
|
|
55
|
+
|
|
56
|
+
def to_dict(self) -> dict:
|
|
57
|
+
return {"_meta": {"tool": "impact-gate", "n": self.n, "head": self.head,
|
|
58
|
+
"base_ref": self.base_ref},
|
|
59
|
+
"distribution": self.dist}
|
|
60
|
+
|
|
61
|
+
@classmethod
|
|
62
|
+
def from_dict(cls, d: dict) -> "Baseline":
|
|
63
|
+
meta = d.get("_meta", {})
|
|
64
|
+
dist = [int(v) for v in (d.get("distribution") or [])]
|
|
65
|
+
n = int(meta.get("n", len(dist)))
|
|
66
|
+
return cls(n=n, dist=dist, head=meta.get("head"),
|
|
67
|
+
base_ref=meta.get("base_ref"))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _percentile_rank(sorted_vals: list[int], value: float) -> float:
|
|
71
|
+
"""Percentile rank of `value` in an ascending list: fraction below plus half the
|
|
72
|
+
ties, in [0, 100]. Above every observation -> 100; at or below the smallest -> ~0."""
|
|
73
|
+
n = len(sorted_vals)
|
|
74
|
+
if n == 0:
|
|
75
|
+
return 0.0
|
|
76
|
+
lo = bisect.bisect_left(sorted_vals, value)
|
|
77
|
+
hi = bisect.bisect_right(sorted_vals, value)
|
|
78
|
+
return round((lo + 0.5 * (hi - lo)) / n * 100, 2)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ----------------------------------------------------------------------- the walk
|
|
82
|
+
|
|
83
|
+
def build_baseline(repo_path: str, mcfg: MeasureConfig | None = None, *,
|
|
84
|
+
base_ref: str | None = None, max_commits: int | None = None,
|
|
85
|
+
exclude_subject_pattern: str | None = None) -> Baseline:
|
|
86
|
+
"""Walk the merged history of `base_ref` into the project's impact distribution."""
|
|
87
|
+
mcfg = mcfg or MeasureConfig()
|
|
88
|
+
base_ref = base_ref or _BD.get("base_ref") or "HEAD"
|
|
89
|
+
if max_commits is None:
|
|
90
|
+
max_commits = _BD.get("max_commits")
|
|
91
|
+
if exclude_subject_pattern is None:
|
|
92
|
+
exclude_subject_pattern = _BD.get("exclude_subject_pattern")
|
|
93
|
+
pat = re.compile(exclude_subject_pattern) if exclude_subject_pattern else None
|
|
94
|
+
|
|
95
|
+
parents = gitio.rev_parents(repo_path, base_ref)
|
|
96
|
+
mainline = gitio.mainline_commits(repo_path, base_ref, max_commits)
|
|
97
|
+
repo = GitRepo(repo_path)
|
|
98
|
+
comp: list[int] = []
|
|
99
|
+
seen: set[str] = set()
|
|
100
|
+
|
|
101
|
+
def emit(old_rev: str, new_rev: str) -> None:
|
|
102
|
+
score = score_change(gitio.diff_between(repo, old_rev, new_rev), mcfg)
|
|
103
|
+
if score.empty:
|
|
104
|
+
return
|
|
105
|
+
comp.append(score.impact)
|
|
106
|
+
|
|
107
|
+
def walk_mr(merge_sha: str, p1: str, tip: str) -> None:
|
|
108
|
+
if merge_sha in seen:
|
|
109
|
+
return
|
|
110
|
+
seen.add(merge_sha)
|
|
111
|
+
if pat and pat.search(gitio.commit_subject(repo_path, merge_sha)):
|
|
112
|
+
return # naming-convention exclusion: skip this MR entirely
|
|
113
|
+
start = gitio.merge_base(repo_path, p1, tip) or p1
|
|
114
|
+
spine = gitio.first_parent_spine(repo_path, start, tip)
|
|
115
|
+
base = start
|
|
116
|
+
for c in reversed(spine): # oldest first, along the branch
|
|
117
|
+
cps = parents.get(c) or gitio.rev_parents(repo_path, c).get(c, [])
|
|
118
|
+
if len(cps) < 2:
|
|
119
|
+
continue # direct commit: extends the run
|
|
120
|
+
prev = cps[0] # branch state just before this merge
|
|
121
|
+
if prev != base:
|
|
122
|
+
emit(base, prev) # net of the run of direct commits
|
|
123
|
+
for sec in cps[1:]:
|
|
124
|
+
if gitio.is_ancestor(repo_path, sec, p1):
|
|
125
|
+
continue # a sync of parent/main: already counted
|
|
126
|
+
walk_mr(c, cps[0], sec) # a child MR: recurse
|
|
127
|
+
base = c # advance past the merge
|
|
128
|
+
if base != tip:
|
|
129
|
+
emit(base, tip) # final run up to the branch tip
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
for c in mainline:
|
|
133
|
+
ps = parents.get(c, [])
|
|
134
|
+
if len(ps) >= 2: # an MR landed on the mainline
|
|
135
|
+
for sec in ps[1:]:
|
|
136
|
+
walk_mr(c, ps[0], sec)
|
|
137
|
+
else: # a direct commit on the mainline
|
|
138
|
+
emit(ps[0] if ps else gitio.EMPTY_TREE, c)
|
|
139
|
+
finally:
|
|
140
|
+
repo.close()
|
|
141
|
+
|
|
142
|
+
comp.sort()
|
|
143
|
+
return Baseline(n=len(comp), dist=comp,
|
|
144
|
+
head=gitio.head_sha(repo_path, base_ref), base_ref=base_ref)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
# ------------------------------------------------------------------- persistence
|
|
148
|
+
|
|
149
|
+
def save_baseline(baseline: Baseline, path: str) -> None:
|
|
150
|
+
with open(path, "w", encoding="utf-8") as fh:
|
|
151
|
+
json.dump(baseline.to_dict(), fh, indent=1)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def load_baseline(path: str) -> Baseline | None:
|
|
155
|
+
"""The cached baseline at `path`, or None if it is missing or unreadable."""
|
|
156
|
+
try:
|
|
157
|
+
with open(path, encoding="utf-8") as fh:
|
|
158
|
+
return Baseline.from_dict(json.load(fh))
|
|
159
|
+
except (OSError, ValueError):
|
|
160
|
+
return None
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# ------------------------------------------------------------------------ grading
|
|
164
|
+
|
|
165
|
+
@dataclass
|
|
166
|
+
class Grade:
|
|
167
|
+
percentile: float # the blended grade, 0..100
|
|
168
|
+
value: int # the composite impact that was graded
|
|
169
|
+
seed_percentile: float # rank against the shipped per-language seed
|
|
170
|
+
project_percentile: float | None # rank against project history (None at cold start)
|
|
171
|
+
weight: float # w = n / (n + K): the project's share of the blend
|
|
172
|
+
n: int # project observations behind the grade
|
|
173
|
+
language: str | None
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def dominant_language(score) -> str | None:
|
|
177
|
+
"""The language driving the change (highest-cost file), else None -> pooled seed."""
|
|
178
|
+
best, best_cost = None, -1
|
|
179
|
+
for f in score.files:
|
|
180
|
+
if f.lang and f.cost > best_cost:
|
|
181
|
+
best, best_cost = f.lang, f.cost
|
|
182
|
+
return best
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def grade_value(value: int, *, language: str | None,
|
|
186
|
+
baseline: Baseline | None, prior_weight_K: float) -> Grade:
|
|
187
|
+
seed_pct, seed_vals = seed_table(language)
|
|
188
|
+
seed_rank = rank_in_table(value, seed_pct, seed_vals)
|
|
189
|
+
n = baseline.n if baseline else 0
|
|
190
|
+
if n <= 0:
|
|
191
|
+
return Grade(seed_rank, value, seed_rank, None, 0.0, 0, language)
|
|
192
|
+
project_rank = baseline.rank(value)
|
|
193
|
+
denom = n + prior_weight_K
|
|
194
|
+
w = n / denom if denom > 0 else 1.0
|
|
195
|
+
blended = round(w * project_rank + (1 - w) * seed_rank, 2)
|
|
196
|
+
return Grade(blended, value, round(seed_rank, 2), round(project_rank, 2),
|
|
197
|
+
round(w, 4), n, language)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def grade_change(score, *, baseline: Baseline | None = None,
|
|
201
|
+
prior_weight_K: float = 200) -> Grade:
|
|
202
|
+
"""Grade a scored change by blending its project and seed percentile ranks.
|
|
203
|
+
|
|
204
|
+
The graded value is always the composite impact; the seed table is picked by the
|
|
205
|
+
change's dominant language.
|
|
206
|
+
"""
|
|
207
|
+
return grade_value(score.impact, language=dominant_language(score),
|
|
208
|
+
baseline=baseline, prior_weight_K=prior_weight_K)
|
impact_gate/cli.py
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""impact-gate command line.
|
|
2
|
+
|
|
3
|
+
impact-gate score [--mode staged|worktree|range] [--base main] ...
|
|
4
|
+
|
|
5
|
+
Exit codes: 0 = ok or warn (change allowed), 2 = blocked (impact too high, enforcement
|
|
6
|
+
'block'), 1 = usage/environment error. CI wrappers (GitHub/GitLab/Jenkins) call this
|
|
7
|
+
same command and translate the exit code + JSON into a check result / comment.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import os
|
|
13
|
+
import subprocess
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
from .core.config import MeasureConfig
|
|
17
|
+
|
|
18
|
+
from . import __version__, baseline, gitio, providers, report
|
|
19
|
+
from .config import ENFORCEMENTS, GateConfig
|
|
20
|
+
from .engine import score_change
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _add_score_args(p: argparse.ArgumentParser) -> None:
|
|
24
|
+
p.add_argument("--mode", choices=gitio.MODES, default="staged",
|
|
25
|
+
help="what to score: 'staged' (the commit you're about to make, "
|
|
26
|
+
"default), 'worktree' (uncommitted edits), or 'range' "
|
|
27
|
+
"(committed branch vs --base, for CI/PR).")
|
|
28
|
+
p.add_argument("--base", default="main",
|
|
29
|
+
help="base ref for --mode range (default: main). Use e.g. "
|
|
30
|
+
"origin/main in CI.")
|
|
31
|
+
p.add_argument("--repo", default=".", help="path to the git repo (default: .)")
|
|
32
|
+
p.add_argument("--format", choices=("text", "json", "markdown"), default="text")
|
|
33
|
+
p.add_argument("--config", help="path to an .impact-gate.yml (else auto-discovered in --repo)")
|
|
34
|
+
# threshold / enforcement overrides (win over the config file when given)
|
|
35
|
+
p.add_argument("--warn-at", type=int)
|
|
36
|
+
p.add_argument("--block-at", type=int)
|
|
37
|
+
p.add_argument("--enforcement", choices=ENFORCEMENTS)
|
|
38
|
+
p.add_argument("--tolerance", type=float)
|
|
39
|
+
p.add_argument("--measure-config", help="Surveyor-style YAML for ignore globs etc.")
|
|
40
|
+
# grading curve (percentile gate) overrides
|
|
41
|
+
p.add_argument("--curve", dest="curve_enabled", action="store_const", const=True,
|
|
42
|
+
default=None, help="gate on the change's percentile grade against the "
|
|
43
|
+
"baseline distribution instead of absolute thresholds")
|
|
44
|
+
p.add_argument("--baseline-file", dest="baseline_file",
|
|
45
|
+
help="project baseline cache to grade against (default: "
|
|
46
|
+
".impact-gate-baseline.json)")
|
|
47
|
+
p.add_argument("--warn-percentile", dest="warn_percentile", type=float)
|
|
48
|
+
p.add_argument("--block-percentile", dest="block_percentile", type=float)
|
|
49
|
+
p.add_argument("--curve-prior-weight", dest="curve_prior_weight", type=float,
|
|
50
|
+
help="K in the empirical-Bayes blend w = n/(n+K) between the project "
|
|
51
|
+
"baseline and the shipped seed. 0 grades PURELY against the "
|
|
52
|
+
"--baseline-file distribution (ignore the seed); large K leans on "
|
|
53
|
+
"the seed. Default from config (200).")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# Gate knobs an argparse flag may override on top of the config file, when given.
|
|
57
|
+
_OVERRIDE_ATTRS = ("warn_at", "block_at", "enforcement", "tolerance", "measure_config",
|
|
58
|
+
"curve_enabled", "baseline_file", "warn_percentile", "block_percentile",
|
|
59
|
+
"curve_prior_weight")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _resolve_config(args) -> GateConfig:
|
|
63
|
+
# Base policy from the policy port (local .impact-gate.yml today); CLI flags are the
|
|
64
|
+
# outermost layer and win over whatever the provider supplied.
|
|
65
|
+
cfg = providers.select_policy_provider(args.config, args.repo).policy(args.repo)
|
|
66
|
+
for attr in _OVERRIDE_ATTRS:
|
|
67
|
+
val = getattr(args, attr, None)
|
|
68
|
+
if val is not None:
|
|
69
|
+
setattr(cfg, attr, val)
|
|
70
|
+
cfg.validate()
|
|
71
|
+
return cfg
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _cmd_score(args) -> int:
|
|
75
|
+
cfg = _resolve_config(args)
|
|
76
|
+
mcfg = MeasureConfig.load(cfg.measure_config)
|
|
77
|
+
try:
|
|
78
|
+
changed = gitio.changed_files(args.repo, args.mode, args.base)
|
|
79
|
+
except gitio.DiffError as e:
|
|
80
|
+
print(f"impact-gate: {e}", file=sys.stderr)
|
|
81
|
+
return 1
|
|
82
|
+
|
|
83
|
+
score = score_change(changed, mcfg)
|
|
84
|
+
|
|
85
|
+
grade = None
|
|
86
|
+
if cfg.curve_enabled:
|
|
87
|
+
grades = providers.select_grade_provider(args.repo, cfg.baseline_file,
|
|
88
|
+
cfg.curve_prior_weight)
|
|
89
|
+
grade = grades.grade(providers.ChangeSummary.of(score), args.repo)
|
|
90
|
+
if grade is None: # backend unreachable -> shipped-seed only
|
|
91
|
+
grade = providers.seed_grade(score, cfg.curve_prior_weight)
|
|
92
|
+
level = cfg.level_for_grade(grade.percentile)
|
|
93
|
+
blocked = cfg.blocks_grade(grade.percentile)
|
|
94
|
+
else:
|
|
95
|
+
level = cfg.level(score.impact)
|
|
96
|
+
blocked = cfg.blocks(score.impact)
|
|
97
|
+
|
|
98
|
+
if args.format == "json":
|
|
99
|
+
print(report.render_json(score, cfg, level, args.mode, args.base, blocked, grade))
|
|
100
|
+
elif args.format == "markdown":
|
|
101
|
+
print(report.render_markdown(score, cfg, level, args.mode, args.base, blocked, grade))
|
|
102
|
+
else:
|
|
103
|
+
print(report.render_text(score, cfg, level, args.mode, args.base, blocked, grade))
|
|
104
|
+
if blocked:
|
|
105
|
+
print("\nimpact-gate: change BLOCKED. Impact exceeds the block threshold. "
|
|
106
|
+
"Simplify the change or refactor the code it touches, then retry.",
|
|
107
|
+
file=sys.stderr)
|
|
108
|
+
return 2 if blocked else 0
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _cmd_baseline(args) -> int:
|
|
112
|
+
"""Walk the merged mainline into the project distribution and cache it to disk."""
|
|
113
|
+
cfg = _resolve_config(args)
|
|
114
|
+
mcfg = MeasureConfig.load(cfg.measure_config)
|
|
115
|
+
out_path = os.path.join(args.repo, cfg.baseline_file)
|
|
116
|
+
try:
|
|
117
|
+
bl = baseline.build_baseline(
|
|
118
|
+
args.repo, mcfg,
|
|
119
|
+
base_ref=args.base_ref,
|
|
120
|
+
max_commits=args.max_commits,
|
|
121
|
+
exclude_subject_pattern=args.exclude_subject_pattern,
|
|
122
|
+
)
|
|
123
|
+
except (gitio.DiffError, OSError, ValueError,
|
|
124
|
+
subprocess.CalledProcessError) as e:
|
|
125
|
+
print(f"impact-gate: could not build baseline (is '{args.base_ref or 'main'}' "
|
|
126
|
+
f"a branch with history?): {e}", file=sys.stderr)
|
|
127
|
+
return 1
|
|
128
|
+
if bl.n == 0:
|
|
129
|
+
print(f"impact-gate: no landed changes found on '{bl.base_ref}'; nothing to "
|
|
130
|
+
"baseline. Check --base-ref points at a branch with history.",
|
|
131
|
+
file=sys.stderr)
|
|
132
|
+
return 1
|
|
133
|
+
providers.select_grade_provider(args.repo, cfg.baseline_file,
|
|
134
|
+
cfg.curve_prior_weight).publish(args.repo, bl)
|
|
135
|
+
print(f"impact-gate: baseline written to {out_path} "
|
|
136
|
+
f"({bl.n} observations from '{bl.base_ref}').")
|
|
137
|
+
return 0
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _cmd_comment(args) -> int:
|
|
141
|
+
from . import ghapi
|
|
142
|
+
token = args.token or os.environ.get("GITHUB_TOKEN")
|
|
143
|
+
repo = args.repo_slug or os.environ.get("GITHUB_REPOSITORY")
|
|
144
|
+
pr = args.pr or ghapi.detect_pr_number(os.environ.get("GITHUB_EVENT_PATH"))
|
|
145
|
+
if not token or not repo or not pr:
|
|
146
|
+
print("impact-gate: need a token, repo (owner/name), and PR number to comment "
|
|
147
|
+
"(GITHUB_TOKEN, GITHUB_REPOSITORY, GITHUB_EVENT_PATH are set in Actions).",
|
|
148
|
+
file=sys.stderr)
|
|
149
|
+
return 1
|
|
150
|
+
body = (open(args.body_file, encoding="utf-8").read()
|
|
151
|
+
if args.body_file else sys.stdin.read())
|
|
152
|
+
try:
|
|
153
|
+
result = ghapi.upsert_pr_comment(ghapi.GitHubAPI(token), repo, int(pr), body)
|
|
154
|
+
except Exception as e:
|
|
155
|
+
print(f"impact-gate: could not post PR comment: {e}", file=sys.stderr)
|
|
156
|
+
return 1
|
|
157
|
+
print(f"impact-gate: PR comment {result}")
|
|
158
|
+
return 0
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def main(argv: list[str] | None = None) -> int:
|
|
162
|
+
ap = argparse.ArgumentParser(prog="impact-gate",
|
|
163
|
+
description="Report and gate on the change-impact of a change.")
|
|
164
|
+
ap.add_argument("--version", action="version", version=f"impact-gate {__version__}")
|
|
165
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
166
|
+
s = sub.add_parser("score", help="score the current change and gate on it")
|
|
167
|
+
_add_score_args(s)
|
|
168
|
+
s.set_defaults(func=_cmd_score)
|
|
169
|
+
|
|
170
|
+
b = sub.add_parser("baseline",
|
|
171
|
+
help="build and cache the project baseline distribution")
|
|
172
|
+
b.add_argument("--repo", default=".", help="path to the git repo (default: .)")
|
|
173
|
+
b.add_argument("--config", help="path to an .impact-gate.yml (else auto-discovered)")
|
|
174
|
+
b.add_argument("--base-ref", dest="base_ref",
|
|
175
|
+
help="mainline branch to walk (default: main, or config's base_ref)")
|
|
176
|
+
b.add_argument("--max-commits", dest="max_commits", type=int,
|
|
177
|
+
help="cap how many recent mainline commits are walked")
|
|
178
|
+
b.add_argument("--exclude-subject-pattern", dest="exclude_subject_pattern",
|
|
179
|
+
help="regex on a merge subject to skip that MR entirely")
|
|
180
|
+
b.add_argument("--baseline-file", dest="baseline_file",
|
|
181
|
+
help="where to write the cache (default: .impact-gate-baseline.json)")
|
|
182
|
+
b.add_argument("--measure-config", help="Surveyor-style YAML for ignore globs etc.")
|
|
183
|
+
b.set_defaults(func=_cmd_baseline)
|
|
184
|
+
|
|
185
|
+
c = sub.add_parser("comment", help="upsert a sticky PR comment with a report (CI)")
|
|
186
|
+
c.add_argument("--body-file", help="markdown file to post (default: read stdin)")
|
|
187
|
+
c.add_argument("--repo-slug", help="owner/name (default: $GITHUB_REPOSITORY)")
|
|
188
|
+
c.add_argument("--pr", type=int, help="PR number (default: from $GITHUB_EVENT_PATH)")
|
|
189
|
+
c.add_argument("--token", help="GitHub token (default: $GITHUB_TOKEN)")
|
|
190
|
+
c.set_defaults(func=_cmd_comment)
|
|
191
|
+
|
|
192
|
+
args = ap.parse_args(argv)
|
|
193
|
+
try:
|
|
194
|
+
return args.func(args)
|
|
195
|
+
except ValueError as e: # config validation, bad args
|
|
196
|
+
print(f"impact-gate: {e}", file=sys.stderr)
|
|
197
|
+
return 1
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
if __name__ == "__main__":
|
|
201
|
+
sys.exit(main())
|
impact_gate/config.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Gate configuration: thresholds, enforcement mode, CI-adjustable tolerance.
|
|
2
|
+
|
|
3
|
+
Resolved in layers (later wins): built-in defaults -> `.impact-gate.yml` in the repo
|
|
4
|
+
-> CLI flags / CI inputs. The built-in defaults are not literals here; they are read
|
|
5
|
+
from `impact_gate/data/defaults.json`, so tuning them is a data edit, not a code change.
|
|
6
|
+
|
|
7
|
+
Two gating modes coexist. Absolute: `warn_at` / `block_at` are raw composite-impact
|
|
8
|
+
numbers. Curve (`curve_enabled`): a change is graded by its percentile against the
|
|
9
|
+
blended seed + project distribution, and `warn_percentile` / `block_percentile` gate on
|
|
10
|
+
that. The curve fields are wired here; `baseline.py` and the percentile gate consume
|
|
11
|
+
them (a later phase). Absolute stays the fallback when the curve is off or no baseline
|
|
12
|
+
exists yet.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
|
|
19
|
+
from .data import load_defaults
|
|
20
|
+
|
|
21
|
+
CONFIG_NAMES = (".impact-gate.yml", ".impact-gate.yaml")
|
|
22
|
+
ENFORCEMENTS = ("off", "warn", "block")
|
|
23
|
+
|
|
24
|
+
# Built-in defaults come from the shipped JSON, never from literals in this file.
|
|
25
|
+
_D = load_defaults()
|
|
26
|
+
_ABS = _D["absolute"]
|
|
27
|
+
_CURVE = _D["curve"]
|
|
28
|
+
|
|
29
|
+
# The knobs an .impact-gate.yml / CLI flag may override, and their loaders. Absolute and
|
|
30
|
+
# curve knobs share one path so a repo can set either mode's numbers in the same file.
|
|
31
|
+
_SCALAR_KEYS = ("warn_at", "block_at", "enforcement", "tolerance", "measure_config",
|
|
32
|
+
"curve_enabled", "warn_percentile", "block_percentile",
|
|
33
|
+
"curve_prior_weight", "baseline_file")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class GateConfig:
|
|
38
|
+
warn_at: int | None = _ABS["warn_at"] # impact above which to warn (None = never)
|
|
39
|
+
block_at: int | None = _ABS["block_at"] # impact above which to block (None = never)
|
|
40
|
+
enforcement: str = _D["enforcement"] # off | warn | block (block = too-high fails)
|
|
41
|
+
tolerance: float = _D["tolerance"] # CI multiplier on both thresholds (>1 = looser)
|
|
42
|
+
measure_config: str | None = None # optional Surveyor-style YAML (ignore globs)
|
|
43
|
+
# Grading curve (percentile-based). Consumed once baseline.py + the percentile gate land.
|
|
44
|
+
# The gate always scores the composite (change-level) impact; the mutation cost is a
|
|
45
|
+
# per-file signal used to rank which files to consider, not a rival gating metric.
|
|
46
|
+
curve_enabled: bool = _CURVE["enabled"] # gate on percentile vs absolute
|
|
47
|
+
warn_percentile: float = _CURVE["warn_percentile"]
|
|
48
|
+
block_percentile: float = _CURVE["block_percentile"]
|
|
49
|
+
curve_prior_weight: float = _CURVE["prior_weight_K"] # K in w = n / (n + K)
|
|
50
|
+
baseline_file: str = _CURVE["baseline_file"] # project distribution cache
|
|
51
|
+
|
|
52
|
+
def effective_warn(self) -> float | None:
|
|
53
|
+
return None if self.warn_at is None else self.warn_at * self.tolerance
|
|
54
|
+
|
|
55
|
+
def effective_block(self) -> float | None:
|
|
56
|
+
return None if self.block_at is None else self.block_at * self.tolerance
|
|
57
|
+
|
|
58
|
+
def level(self, impact: float) -> str:
|
|
59
|
+
"""'block' / 'warn' / 'ok' by threshold alone (independent of enforcement)."""
|
|
60
|
+
b, w = self.effective_block(), self.effective_warn()
|
|
61
|
+
if b is not None and impact > b:
|
|
62
|
+
return "block"
|
|
63
|
+
if w is not None and impact > w:
|
|
64
|
+
return "warn"
|
|
65
|
+
return "ok"
|
|
66
|
+
|
|
67
|
+
def blocks(self, impact: float) -> bool:
|
|
68
|
+
"""True only when enforcement is 'block' AND the impact clears block_at — the
|
|
69
|
+
one case that fails the gate. In 'warn' mode a too-high change still passes."""
|
|
70
|
+
return self.enforcement == "block" and self.level(impact) == "block"
|
|
71
|
+
|
|
72
|
+
def level_for_grade(self, percentile: float) -> str:
|
|
73
|
+
"""Curve mode: 'block' / 'warn' / 'ok' from a change's grade percentile. A change
|
|
74
|
+
at or above `block_percentile` blocks; at or above `warn_percentile` warns."""
|
|
75
|
+
if percentile >= self.block_percentile:
|
|
76
|
+
return "block"
|
|
77
|
+
if percentile >= self.warn_percentile:
|
|
78
|
+
return "warn"
|
|
79
|
+
return "ok"
|
|
80
|
+
|
|
81
|
+
def blocks_grade(self, percentile: float) -> bool:
|
|
82
|
+
"""Curve analogue of `blocks`: fails the gate only under 'block' enforcement."""
|
|
83
|
+
return (self.enforcement == "block"
|
|
84
|
+
and self.level_for_grade(percentile) == "block")
|
|
85
|
+
|
|
86
|
+
@classmethod
|
|
87
|
+
def load(cls, path: str | None = None, repo_path: str = ".") -> "GateConfig":
|
|
88
|
+
"""Load from an explicit path, else the first `.impact-gate.y*ml` in repo_path."""
|
|
89
|
+
cfg = cls()
|
|
90
|
+
found = path or _discover(repo_path)
|
|
91
|
+
if not found:
|
|
92
|
+
return cfg
|
|
93
|
+
import yaml # optional; only needed when a config file exists
|
|
94
|
+
with open(found) as fh:
|
|
95
|
+
data = yaml.safe_load(fh) or {}
|
|
96
|
+
for key in _SCALAR_KEYS:
|
|
97
|
+
if key in data and data[key] is not None:
|
|
98
|
+
setattr(cfg, key, data[key])
|
|
99
|
+
cfg.validate()
|
|
100
|
+
return cfg
|
|
101
|
+
|
|
102
|
+
def validate(self) -> None:
|
|
103
|
+
if self.enforcement not in ENFORCEMENTS:
|
|
104
|
+
raise ValueError(f"enforcement must be one of {ENFORCEMENTS}, got "
|
|
105
|
+
f"{self.enforcement!r}")
|
|
106
|
+
if self.tolerance <= 0:
|
|
107
|
+
raise ValueError("tolerance must be > 0")
|
|
108
|
+
for name in ("warn_percentile", "block_percentile"):
|
|
109
|
+
p = getattr(self, name)
|
|
110
|
+
if not 0 < p < 100:
|
|
111
|
+
raise ValueError(f"{name} must be within (0, 100), got {p}")
|
|
112
|
+
if self.warn_percentile > self.block_percentile:
|
|
113
|
+
raise ValueError("warn_percentile must be <= block_percentile")
|
|
114
|
+
if self.curve_prior_weight < 0:
|
|
115
|
+
raise ValueError("curve_prior_weight (K) must be >= 0")
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _discover(repo_path: str) -> str | None:
|
|
119
|
+
for name in CONFIG_NAMES:
|
|
120
|
+
p = os.path.join(repo_path, name)
|
|
121
|
+
if os.path.isfile(p):
|
|
122
|
+
return p
|
|
123
|
+
return None
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Self-contained structural change-impact core.
|
|
2
|
+
|
|
3
|
+
Vendored so the tool is standalone — no dependency on the (concluded) Surveyor
|
|
4
|
+
bug-finding experiment. Importing this package registers the default lizard plugin.
|
|
5
|
+
"""
|
|
6
|
+
from .units import Unit, get_plugin, register # noqa: F401
|
|
7
|
+
from . import lizard_plugin # noqa: F401 (registers default plugin)
|
|
8
|
+
from .impact import FileImpact, UnitImpact, compute_file_impact # noqa: F401
|
|
9
|
+
from .config import MeasureConfig # noqa: F401
|
|
10
|
+
from .gitplumb import GitRepo, parse_diff # noqa: F401
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Measure configuration: which files are source, how to group, ignore globs.
|
|
2
|
+
|
|
3
|
+
Structural-decay focused: languages, ignore globs, test-path heuristics, and the
|
|
4
|
+
rename similarity threshold. (No bug-keyword mining — that belonged to the earlier
|
|
5
|
+
defect-prediction experiment, not to a structural-decay gate.)
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import fnmatch
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
|
|
12
|
+
# Extension -> language label. lizard picks its reader by filename, so this table only
|
|
13
|
+
# decides "is this a source file worth parsing", plus the label used in reports.
|
|
14
|
+
LANG_BY_EXT: dict[str, str] = {
|
|
15
|
+
".java": "java",
|
|
16
|
+
".cs": "csharp",
|
|
17
|
+
".c": "c", ".h": "c", ".cc": "cpp", ".cpp": "cpp", ".cxx": "cpp",
|
|
18
|
+
".hpp": "cpp", ".hh": "cpp", ".hxx": "cpp",
|
|
19
|
+
".js": "javascript", ".jsx": "javascript", ".mjs": "javascript", ".cjs": "javascript",
|
|
20
|
+
".ts": "typescript", ".tsx": "typescript",
|
|
21
|
+
".py": "python",
|
|
22
|
+
".go": "go",
|
|
23
|
+
".kt": "kotlin", ".kts": "kotlin",
|
|
24
|
+
".swift": "swift",
|
|
25
|
+
".rb": "ruby",
|
|
26
|
+
".php": "php",
|
|
27
|
+
".rs": "rust",
|
|
28
|
+
".scala": "scala",
|
|
29
|
+
".m": "objectivec", ".mm": "objectivec",
|
|
30
|
+
".lua": "lua",
|
|
31
|
+
".ttcn": "ttcn",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
DEFAULT_IGNORE = [
|
|
35
|
+
"**/node_modules/**", "**/dist/**", "**/build/**", "**/target/**",
|
|
36
|
+
"**/out/**", "**/bin/**", "**/obj/**", "**/third_party/**",
|
|
37
|
+
"**/vendor/**", "**/vendors/**", "**/bower_components/**", "**/webjars/**",
|
|
38
|
+
"**/.venv/**", "**/venv/**", "**/__pycache__/**", "**/.git/**",
|
|
39
|
+
"**/*.min.js", "**/*.min.css", "**/*.bundle.js", "**/*.generated.*",
|
|
40
|
+
"**/generated/**", "**/gen/**",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
DEFAULT_TEST_PATTERNS = [
|
|
44
|
+
"**/test/**", "**/tests/**", "**/__tests__/**", "**/spec/**",
|
|
45
|
+
"**/*Test.*", "**/*Tests.*", "**/*_test.*", "**/test_*.*",
|
|
46
|
+
"**/*.test.*", "**/*.spec.*",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class MeasureConfig:
|
|
52
|
+
ignore: list[str] = field(default_factory=lambda: list(DEFAULT_IGNORE))
|
|
53
|
+
test_patterns: list[str] = field(default_factory=lambda: list(DEFAULT_TEST_PATTERNS))
|
|
54
|
+
lang_by_ext: dict[str, str] = field(default_factory=lambda: dict(LANG_BY_EXT))
|
|
55
|
+
rename_jaccard: float = 0.6 # body token-set similarity to call a rename
|
|
56
|
+
max_diff_lines: int = 200_000 # skip pathological mega-diffs (generated dumps)
|
|
57
|
+
|
|
58
|
+
def ext(self, path: str) -> str:
|
|
59
|
+
i = path.rfind(".")
|
|
60
|
+
return path[i:].lower() if i >= 0 else ""
|
|
61
|
+
|
|
62
|
+
def is_source(self, path: str) -> bool:
|
|
63
|
+
return not self.is_ignored(path) and self.ext(path) in self.lang_by_ext
|
|
64
|
+
|
|
65
|
+
def language(self, path: str) -> str | None:
|
|
66
|
+
return self.lang_by_ext.get(self.ext(path))
|
|
67
|
+
|
|
68
|
+
def is_ignored(self, path: str) -> bool:
|
|
69
|
+
# Match each glob against the path, and also with a leading "**/" stripped, so a
|
|
70
|
+
# pattern like "**/vendor/**" ignores a TOP-LEVEL vendor/ too. fnmatch's "*" spans
|
|
71
|
+
# "/", but "**/vendor/**" still requires a parent segment before "vendor", which
|
|
72
|
+
# let a repo-root vendor/ or node_modules/ slip through.
|
|
73
|
+
for pat in self.ignore:
|
|
74
|
+
if fnmatch.fnmatch(path, pat):
|
|
75
|
+
return True
|
|
76
|
+
if pat.startswith("**/") and fnmatch.fnmatch(path, pat[3:]):
|
|
77
|
+
return True
|
|
78
|
+
return False
|
|
79
|
+
|
|
80
|
+
def is_test(self, path: str) -> bool:
|
|
81
|
+
return any(fnmatch.fnmatch(path, pat) for pat in self.test_patterns)
|
|
82
|
+
|
|
83
|
+
@classmethod
|
|
84
|
+
def load(cls, path: str | None) -> "MeasureConfig":
|
|
85
|
+
cfg = cls()
|
|
86
|
+
if not path:
|
|
87
|
+
return cfg
|
|
88
|
+
import yaml # optional; only needed when a config file is passed
|
|
89
|
+
with open(path) as fh:
|
|
90
|
+
data = yaml.safe_load(fh) or {}
|
|
91
|
+
for key in ("rename_jaccard", "max_diff_lines"):
|
|
92
|
+
if key in data:
|
|
93
|
+
setattr(cfg, key, data[key])
|
|
94
|
+
# ignore / test_patterns APPEND to the defaults (a config adds, never silently
|
|
95
|
+
# drops). To start from scratch, set the matching `*_replace` key instead.
|
|
96
|
+
cfg.ignore = list(data["ignore_replace"]) if "ignore_replace" in data \
|
|
97
|
+
else cfg.ignore + list(data.get("ignore", []))
|
|
98
|
+
cfg.test_patterns = list(data["test_patterns_replace"]) if "test_patterns_replace" in data \
|
|
99
|
+
else cfg.test_patterns + list(data.get("test_patterns", []))
|
|
100
|
+
if "lang_by_ext" in data:
|
|
101
|
+
cfg.lang_by_ext.update(data["lang_by_ext"])
|
|
102
|
+
return cfg
|