impact-gate 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,8 @@
1
+ """impact-gate — report the change-impact of a change and gate on it.
2
+
3
+ A lightweight front-end over the change-impact measure (shared, single-sourced with
4
+ Surveyor / the PetClinic-Evolve harness). It scores a change against a base — a
5
+ committed range vs `main`, staged changes, or the working tree — reports the number,
6
+ and warns or blocks when the impact is too high.
7
+ """
8
+ __version__ = "0.1.0"
@@ -0,0 +1,208 @@
1
+ """The project baseline: the repo's own per-commit impact distribution, and the
2
+ empirical-Bayes blend of it with the shipped seed prior.
3
+
4
+ Scope. The walk starts from the merged base branch (default `main`) and follows its
5
+ first-parent mainline, so only landed work is measured; an unmerged in-flight branch is
6
+ never reached, because the walk only ever follows real merge commits.
7
+
8
+ One observation = one atomic landed change:
9
+ * a main non-merge commit -> its own diff vs its parent;
10
+ * a leaf MR (a merged branch with no MRs inside it) -> the net change from the branch
11
+ start (merge-base of the merge's parents) to the branch tip;
12
+ * a parent MR (a merged branch that contains MRs) -> its direct commits netted per run
13
+ between the merges on its spine, each run based at the preceding synced base; every
14
+ child MR recurses, and a merge that only syncs the parent/main branch in is skipped.
15
+ A merge is never scored as its own diff, so a long-running roll-up cannot inflate things.
16
+
17
+ Each observation is one composite impact = (Σ mutation + Σ godclass) * files. The
18
+ distribution of that over all observations is the baseline. The gate always grades the
19
+ composite; mutation cost stays a per-file ranking signal (report side), never a rival
20
+ distribution to gate against.
21
+
22
+ Grading. A change's grade blends its percentile rank in this project distribution (n
23
+ observations) with its rank against the per-language seed table, weighting the project by
24
+ w = n / (n + K)
25
+ so a shallow history leans on the seed and a deep one trusts itself. K is curve_prior_weight.
26
+ """
27
+ from __future__ import annotations
28
+
29
+ import bisect
30
+ import json
31
+ import re
32
+ from dataclasses import dataclass, field
33
+
34
+ from . import gitio
35
+ from .core.config import MeasureConfig
36
+ from .core.gitplumb import GitRepo
37
+ from .data import load_defaults, rank_in_table, seed_table
38
+ from .engine import score_change
39
+
40
+ _BD = load_defaults().get("baseline", {})
41
+
42
+
43
+ # --------------------------------------------------------------------------- model
44
+
45
+ @dataclass
46
+ class Baseline:
47
+ """The project's composite per-observation distribution (ascending)."""
48
+ n: int
49
+ dist: list[int] = field(default_factory=list)
50
+ head: str | None = None
51
+ base_ref: str | None = None
52
+
53
+ def rank(self, value: float) -> float:
54
+ return _percentile_rank(self.dist, value)
55
+
56
+ def to_dict(self) -> dict:
57
+ return {"_meta": {"tool": "impact-gate", "n": self.n, "head": self.head,
58
+ "base_ref": self.base_ref},
59
+ "distribution": self.dist}
60
+
61
+ @classmethod
62
+ def from_dict(cls, d: dict) -> "Baseline":
63
+ meta = d.get("_meta", {})
64
+ dist = [int(v) for v in (d.get("distribution") or [])]
65
+ n = int(meta.get("n", len(dist)))
66
+ return cls(n=n, dist=dist, head=meta.get("head"),
67
+ base_ref=meta.get("base_ref"))
68
+
69
+
70
+ def _percentile_rank(sorted_vals: list[int], value: float) -> float:
71
+ """Percentile rank of `value` in an ascending list: fraction below plus half the
72
+ ties, in [0, 100]. Above every observation -> 100; at or below the smallest -> ~0."""
73
+ n = len(sorted_vals)
74
+ if n == 0:
75
+ return 0.0
76
+ lo = bisect.bisect_left(sorted_vals, value)
77
+ hi = bisect.bisect_right(sorted_vals, value)
78
+ return round((lo + 0.5 * (hi - lo)) / n * 100, 2)
79
+
80
+
81
+ # ----------------------------------------------------------------------- the walk
82
+
83
+ def build_baseline(repo_path: str, mcfg: MeasureConfig | None = None, *,
84
+ base_ref: str | None = None, max_commits: int | None = None,
85
+ exclude_subject_pattern: str | None = None) -> Baseline:
86
+ """Walk the merged history of `base_ref` into the project's impact distribution."""
87
+ mcfg = mcfg or MeasureConfig()
88
+ base_ref = base_ref or _BD.get("base_ref") or "HEAD"
89
+ if max_commits is None:
90
+ max_commits = _BD.get("max_commits")
91
+ if exclude_subject_pattern is None:
92
+ exclude_subject_pattern = _BD.get("exclude_subject_pattern")
93
+ pat = re.compile(exclude_subject_pattern) if exclude_subject_pattern else None
94
+
95
+ parents = gitio.rev_parents(repo_path, base_ref)
96
+ mainline = gitio.mainline_commits(repo_path, base_ref, max_commits)
97
+ repo = GitRepo(repo_path)
98
+ comp: list[int] = []
99
+ seen: set[str] = set()
100
+
101
+ def emit(old_rev: str, new_rev: str) -> None:
102
+ score = score_change(gitio.diff_between(repo, old_rev, new_rev), mcfg)
103
+ if score.empty:
104
+ return
105
+ comp.append(score.impact)
106
+
107
+ def walk_mr(merge_sha: str, p1: str, tip: str) -> None:
108
+ if merge_sha in seen:
109
+ return
110
+ seen.add(merge_sha)
111
+ if pat and pat.search(gitio.commit_subject(repo_path, merge_sha)):
112
+ return # naming-convention exclusion: skip this MR entirely
113
+ start = gitio.merge_base(repo_path, p1, tip) or p1
114
+ spine = gitio.first_parent_spine(repo_path, start, tip)
115
+ base = start
116
+ for c in reversed(spine): # oldest first, along the branch
117
+ cps = parents.get(c) or gitio.rev_parents(repo_path, c).get(c, [])
118
+ if len(cps) < 2:
119
+ continue # direct commit: extends the run
120
+ prev = cps[0] # branch state just before this merge
121
+ if prev != base:
122
+ emit(base, prev) # net of the run of direct commits
123
+ for sec in cps[1:]:
124
+ if gitio.is_ancestor(repo_path, sec, p1):
125
+ continue # a sync of parent/main: already counted
126
+ walk_mr(c, cps[0], sec) # a child MR: recurse
127
+ base = c # advance past the merge
128
+ if base != tip:
129
+ emit(base, tip) # final run up to the branch tip
130
+
131
+ try:
132
+ for c in mainline:
133
+ ps = parents.get(c, [])
134
+ if len(ps) >= 2: # an MR landed on the mainline
135
+ for sec in ps[1:]:
136
+ walk_mr(c, ps[0], sec)
137
+ else: # a direct commit on the mainline
138
+ emit(ps[0] if ps else gitio.EMPTY_TREE, c)
139
+ finally:
140
+ repo.close()
141
+
142
+ comp.sort()
143
+ return Baseline(n=len(comp), dist=comp,
144
+ head=gitio.head_sha(repo_path, base_ref), base_ref=base_ref)
145
+
146
+
147
+ # ------------------------------------------------------------------- persistence
148
+
149
+ def save_baseline(baseline: Baseline, path: str) -> None:
150
+ with open(path, "w", encoding="utf-8") as fh:
151
+ json.dump(baseline.to_dict(), fh, indent=1)
152
+
153
+
154
+ def load_baseline(path: str) -> Baseline | None:
155
+ """The cached baseline at `path`, or None if it is missing or unreadable."""
156
+ try:
157
+ with open(path, encoding="utf-8") as fh:
158
+ return Baseline.from_dict(json.load(fh))
159
+ except (OSError, ValueError):
160
+ return None
161
+
162
+
163
+ # ------------------------------------------------------------------------ grading
164
+
165
+ @dataclass
166
+ class Grade:
167
+ percentile: float # the blended grade, 0..100
168
+ value: int # the composite impact that was graded
169
+ seed_percentile: float # rank against the shipped per-language seed
170
+ project_percentile: float | None # rank against project history (None at cold start)
171
+ weight: float # w = n / (n + K): the project's share of the blend
172
+ n: int # project observations behind the grade
173
+ language: str | None
174
+
175
+
176
+ def dominant_language(score) -> str | None:
177
+ """The language driving the change (highest-cost file), else None -> pooled seed."""
178
+ best, best_cost = None, -1
179
+ for f in score.files:
180
+ if f.lang and f.cost > best_cost:
181
+ best, best_cost = f.lang, f.cost
182
+ return best
183
+
184
+
185
+ def grade_value(value: int, *, language: str | None,
186
+ baseline: Baseline | None, prior_weight_K: float) -> Grade:
187
+ seed_pct, seed_vals = seed_table(language)
188
+ seed_rank = rank_in_table(value, seed_pct, seed_vals)
189
+ n = baseline.n if baseline else 0
190
+ if n <= 0:
191
+ return Grade(seed_rank, value, seed_rank, None, 0.0, 0, language)
192
+ project_rank = baseline.rank(value)
193
+ denom = n + prior_weight_K
194
+ w = n / denom if denom > 0 else 1.0
195
+ blended = round(w * project_rank + (1 - w) * seed_rank, 2)
196
+ return Grade(blended, value, round(seed_rank, 2), round(project_rank, 2),
197
+ round(w, 4), n, language)
198
+
199
+
200
+ def grade_change(score, *, baseline: Baseline | None = None,
201
+ prior_weight_K: float = 200) -> Grade:
202
+ """Grade a scored change by blending its project and seed percentile ranks.
203
+
204
+ The graded value is always the composite impact; the seed table is picked by the
205
+ change's dominant language.
206
+ """
207
+ return grade_value(score.impact, language=dominant_language(score),
208
+ baseline=baseline, prior_weight_K=prior_weight_K)
impact_gate/cli.py ADDED
@@ -0,0 +1,201 @@
1
+ """impact-gate command line.
2
+
3
+ impact-gate score [--mode staged|worktree|range] [--base main] ...
4
+
5
+ Exit codes: 0 = ok or warn (change allowed), 2 = blocked (impact too high, enforcement
6
+ 'block'), 1 = usage/environment error. CI wrappers (GitHub/GitLab/Jenkins) call this
7
+ same command and translate the exit code + JSON into a check result / comment.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import os
13
+ import subprocess
14
+ import sys
15
+
16
+ from .core.config import MeasureConfig
17
+
18
+ from . import __version__, baseline, gitio, providers, report
19
+ from .config import ENFORCEMENTS, GateConfig
20
+ from .engine import score_change
21
+
22
+
23
+ def _add_score_args(p: argparse.ArgumentParser) -> None:
24
+ p.add_argument("--mode", choices=gitio.MODES, default="staged",
25
+ help="what to score: 'staged' (the commit you're about to make, "
26
+ "default), 'worktree' (uncommitted edits), or 'range' "
27
+ "(committed branch vs --base, for CI/PR).")
28
+ p.add_argument("--base", default="main",
29
+ help="base ref for --mode range (default: main). Use e.g. "
30
+ "origin/main in CI.")
31
+ p.add_argument("--repo", default=".", help="path to the git repo (default: .)")
32
+ p.add_argument("--format", choices=("text", "json", "markdown"), default="text")
33
+ p.add_argument("--config", help="path to an .impact-gate.yml (else auto-discovered in --repo)")
34
+ # threshold / enforcement overrides (win over the config file when given)
35
+ p.add_argument("--warn-at", type=int)
36
+ p.add_argument("--block-at", type=int)
37
+ p.add_argument("--enforcement", choices=ENFORCEMENTS)
38
+ p.add_argument("--tolerance", type=float)
39
+ p.add_argument("--measure-config", help="Surveyor-style YAML for ignore globs etc.")
40
+ # grading curve (percentile gate) overrides
41
+ p.add_argument("--curve", dest="curve_enabled", action="store_const", const=True,
42
+ default=None, help="gate on the change's percentile grade against the "
43
+ "baseline distribution instead of absolute thresholds")
44
+ p.add_argument("--baseline-file", dest="baseline_file",
45
+ help="project baseline cache to grade against (default: "
46
+ ".impact-gate-baseline.json)")
47
+ p.add_argument("--warn-percentile", dest="warn_percentile", type=float)
48
+ p.add_argument("--block-percentile", dest="block_percentile", type=float)
49
+ p.add_argument("--curve-prior-weight", dest="curve_prior_weight", type=float,
50
+ help="K in the empirical-Bayes blend w = n/(n+K) between the project "
51
+ "baseline and the shipped seed. 0 grades PURELY against the "
52
+ "--baseline-file distribution (ignore the seed); large K leans on "
53
+ "the seed. Default from config (200).")
54
+
55
+
56
+ # Gate knobs an argparse flag may override on top of the config file, when given.
57
+ _OVERRIDE_ATTRS = ("warn_at", "block_at", "enforcement", "tolerance", "measure_config",
58
+ "curve_enabled", "baseline_file", "warn_percentile", "block_percentile",
59
+ "curve_prior_weight")
60
+
61
+
62
+ def _resolve_config(args) -> GateConfig:
63
+ # Base policy from the policy port (local .impact-gate.yml today); CLI flags are the
64
+ # outermost layer and win over whatever the provider supplied.
65
+ cfg = providers.select_policy_provider(args.config, args.repo).policy(args.repo)
66
+ for attr in _OVERRIDE_ATTRS:
67
+ val = getattr(args, attr, None)
68
+ if val is not None:
69
+ setattr(cfg, attr, val)
70
+ cfg.validate()
71
+ return cfg
72
+
73
+
74
+ def _cmd_score(args) -> int:
75
+ cfg = _resolve_config(args)
76
+ mcfg = MeasureConfig.load(cfg.measure_config)
77
+ try:
78
+ changed = gitio.changed_files(args.repo, args.mode, args.base)
79
+ except gitio.DiffError as e:
80
+ print(f"impact-gate: {e}", file=sys.stderr)
81
+ return 1
82
+
83
+ score = score_change(changed, mcfg)
84
+
85
+ grade = None
86
+ if cfg.curve_enabled:
87
+ grades = providers.select_grade_provider(args.repo, cfg.baseline_file,
88
+ cfg.curve_prior_weight)
89
+ grade = grades.grade(providers.ChangeSummary.of(score), args.repo)
90
+ if grade is None: # backend unreachable -> shipped-seed only
91
+ grade = providers.seed_grade(score, cfg.curve_prior_weight)
92
+ level = cfg.level_for_grade(grade.percentile)
93
+ blocked = cfg.blocks_grade(grade.percentile)
94
+ else:
95
+ level = cfg.level(score.impact)
96
+ blocked = cfg.blocks(score.impact)
97
+
98
+ if args.format == "json":
99
+ print(report.render_json(score, cfg, level, args.mode, args.base, blocked, grade))
100
+ elif args.format == "markdown":
101
+ print(report.render_markdown(score, cfg, level, args.mode, args.base, blocked, grade))
102
+ else:
103
+ print(report.render_text(score, cfg, level, args.mode, args.base, blocked, grade))
104
+ if blocked:
105
+ print("\nimpact-gate: change BLOCKED. Impact exceeds the block threshold. "
106
+ "Simplify the change or refactor the code it touches, then retry.",
107
+ file=sys.stderr)
108
+ return 2 if blocked else 0
109
+
110
+
111
+ def _cmd_baseline(args) -> int:
112
+ """Walk the merged mainline into the project distribution and cache it to disk."""
113
+ cfg = _resolve_config(args)
114
+ mcfg = MeasureConfig.load(cfg.measure_config)
115
+ out_path = os.path.join(args.repo, cfg.baseline_file)
116
+ try:
117
+ bl = baseline.build_baseline(
118
+ args.repo, mcfg,
119
+ base_ref=args.base_ref,
120
+ max_commits=args.max_commits,
121
+ exclude_subject_pattern=args.exclude_subject_pattern,
122
+ )
123
+ except (gitio.DiffError, OSError, ValueError,
124
+ subprocess.CalledProcessError) as e:
125
+ print(f"impact-gate: could not build baseline (is '{args.base_ref or 'main'}' "
126
+ f"a branch with history?): {e}", file=sys.stderr)
127
+ return 1
128
+ if bl.n == 0:
129
+ print(f"impact-gate: no landed changes found on '{bl.base_ref}'; nothing to "
130
+ "baseline. Check --base-ref points at a branch with history.",
131
+ file=sys.stderr)
132
+ return 1
133
+ providers.select_grade_provider(args.repo, cfg.baseline_file,
134
+ cfg.curve_prior_weight).publish(args.repo, bl)
135
+ print(f"impact-gate: baseline written to {out_path} "
136
+ f"({bl.n} observations from '{bl.base_ref}').")
137
+ return 0
138
+
139
+
140
+ def _cmd_comment(args) -> int:
141
+ from . import ghapi
142
+ token = args.token or os.environ.get("GITHUB_TOKEN")
143
+ repo = args.repo_slug or os.environ.get("GITHUB_REPOSITORY")
144
+ pr = args.pr or ghapi.detect_pr_number(os.environ.get("GITHUB_EVENT_PATH"))
145
+ if not token or not repo or not pr:
146
+ print("impact-gate: need a token, repo (owner/name), and PR number to comment "
147
+ "(GITHUB_TOKEN, GITHUB_REPOSITORY, GITHUB_EVENT_PATH are set in Actions).",
148
+ file=sys.stderr)
149
+ return 1
150
+ body = (open(args.body_file, encoding="utf-8").read()
151
+ if args.body_file else sys.stdin.read())
152
+ try:
153
+ result = ghapi.upsert_pr_comment(ghapi.GitHubAPI(token), repo, int(pr), body)
154
+ except Exception as e:
155
+ print(f"impact-gate: could not post PR comment: {e}", file=sys.stderr)
156
+ return 1
157
+ print(f"impact-gate: PR comment {result}")
158
+ return 0
159
+
160
+
161
+ def main(argv: list[str] | None = None) -> int:
162
+ ap = argparse.ArgumentParser(prog="impact-gate",
163
+ description="Report and gate on the change-impact of a change.")
164
+ ap.add_argument("--version", action="version", version=f"impact-gate {__version__}")
165
+ sub = ap.add_subparsers(dest="cmd", required=True)
166
+ s = sub.add_parser("score", help="score the current change and gate on it")
167
+ _add_score_args(s)
168
+ s.set_defaults(func=_cmd_score)
169
+
170
+ b = sub.add_parser("baseline",
171
+ help="build and cache the project baseline distribution")
172
+ b.add_argument("--repo", default=".", help="path to the git repo (default: .)")
173
+ b.add_argument("--config", help="path to an .impact-gate.yml (else auto-discovered)")
174
+ b.add_argument("--base-ref", dest="base_ref",
175
+ help="mainline branch to walk (default: main, or config's base_ref)")
176
+ b.add_argument("--max-commits", dest="max_commits", type=int,
177
+ help="cap how many recent mainline commits are walked")
178
+ b.add_argument("--exclude-subject-pattern", dest="exclude_subject_pattern",
179
+ help="regex on a merge subject to skip that MR entirely")
180
+ b.add_argument("--baseline-file", dest="baseline_file",
181
+ help="where to write the cache (default: .impact-gate-baseline.json)")
182
+ b.add_argument("--measure-config", help="Surveyor-style YAML for ignore globs etc.")
183
+ b.set_defaults(func=_cmd_baseline)
184
+
185
+ c = sub.add_parser("comment", help="upsert a sticky PR comment with a report (CI)")
186
+ c.add_argument("--body-file", help="markdown file to post (default: read stdin)")
187
+ c.add_argument("--repo-slug", help="owner/name (default: $GITHUB_REPOSITORY)")
188
+ c.add_argument("--pr", type=int, help="PR number (default: from $GITHUB_EVENT_PATH)")
189
+ c.add_argument("--token", help="GitHub token (default: $GITHUB_TOKEN)")
190
+ c.set_defaults(func=_cmd_comment)
191
+
192
+ args = ap.parse_args(argv)
193
+ try:
194
+ return args.func(args)
195
+ except ValueError as e: # config validation, bad args
196
+ print(f"impact-gate: {e}", file=sys.stderr)
197
+ return 1
198
+
199
+
200
+ if __name__ == "__main__":
201
+ sys.exit(main())
impact_gate/config.py ADDED
@@ -0,0 +1,123 @@
1
+ """Gate configuration: thresholds, enforcement mode, CI-adjustable tolerance.
2
+
3
+ Resolved in layers (later wins): built-in defaults -> `.impact-gate.yml` in the repo
4
+ -> CLI flags / CI inputs. The built-in defaults are not literals here; they are read
5
+ from `impact_gate/data/defaults.json`, so tuning them is a data edit, not a code change.
6
+
7
+ Two gating modes coexist. Absolute: `warn_at` / `block_at` are raw composite-impact
8
+ numbers. Curve (`curve_enabled`): a change is graded by its percentile against the
9
+ blended seed + project distribution, and `warn_percentile` / `block_percentile` gate on
10
+ that. The curve fields are wired here; `baseline.py` and the percentile gate consume
11
+ them (a later phase). Absolute stays the fallback when the curve is off or no baseline
12
+ exists yet.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ from dataclasses import dataclass
18
+
19
+ from .data import load_defaults
20
+
21
+ CONFIG_NAMES = (".impact-gate.yml", ".impact-gate.yaml")
22
+ ENFORCEMENTS = ("off", "warn", "block")
23
+
24
+ # Built-in defaults come from the shipped JSON, never from literals in this file.
25
+ _D = load_defaults()
26
+ _ABS = _D["absolute"]
27
+ _CURVE = _D["curve"]
28
+
29
+ # The knobs an .impact-gate.yml / CLI flag may override, and their loaders. Absolute and
30
+ # curve knobs share one path so a repo can set either mode's numbers in the same file.
31
+ _SCALAR_KEYS = ("warn_at", "block_at", "enforcement", "tolerance", "measure_config",
32
+ "curve_enabled", "warn_percentile", "block_percentile",
33
+ "curve_prior_weight", "baseline_file")
34
+
35
+
36
+ @dataclass
37
+ class GateConfig:
38
+ warn_at: int | None = _ABS["warn_at"] # impact above which to warn (None = never)
39
+ block_at: int | None = _ABS["block_at"] # impact above which to block (None = never)
40
+ enforcement: str = _D["enforcement"] # off | warn | block (block = too-high fails)
41
+ tolerance: float = _D["tolerance"] # CI multiplier on both thresholds (>1 = looser)
42
+ measure_config: str | None = None # optional Surveyor-style YAML (ignore globs)
43
+ # Grading curve (percentile-based). Consumed once baseline.py + the percentile gate land.
44
+ # The gate always scores the composite (change-level) impact; the mutation cost is a
45
+ # per-file signal used to rank which files to consider, not a rival gating metric.
46
+ curve_enabled: bool = _CURVE["enabled"] # gate on percentile vs absolute
47
+ warn_percentile: float = _CURVE["warn_percentile"]
48
+ block_percentile: float = _CURVE["block_percentile"]
49
+ curve_prior_weight: float = _CURVE["prior_weight_K"] # K in w = n / (n + K)
50
+ baseline_file: str = _CURVE["baseline_file"] # project distribution cache
51
+
52
+ def effective_warn(self) -> float | None:
53
+ return None if self.warn_at is None else self.warn_at * self.tolerance
54
+
55
+ def effective_block(self) -> float | None:
56
+ return None if self.block_at is None else self.block_at * self.tolerance
57
+
58
+ def level(self, impact: float) -> str:
59
+ """'block' / 'warn' / 'ok' by threshold alone (independent of enforcement)."""
60
+ b, w = self.effective_block(), self.effective_warn()
61
+ if b is not None and impact > b:
62
+ return "block"
63
+ if w is not None and impact > w:
64
+ return "warn"
65
+ return "ok"
66
+
67
+ def blocks(self, impact: float) -> bool:
68
+ """True only when enforcement is 'block' AND the impact clears block_at — the
69
+ one case that fails the gate. In 'warn' mode a too-high change still passes."""
70
+ return self.enforcement == "block" and self.level(impact) == "block"
71
+
72
+ def level_for_grade(self, percentile: float) -> str:
73
+ """Curve mode: 'block' / 'warn' / 'ok' from a change's grade percentile. A change
74
+ at or above `block_percentile` blocks; at or above `warn_percentile` warns."""
75
+ if percentile >= self.block_percentile:
76
+ return "block"
77
+ if percentile >= self.warn_percentile:
78
+ return "warn"
79
+ return "ok"
80
+
81
+ def blocks_grade(self, percentile: float) -> bool:
82
+ """Curve analogue of `blocks`: fails the gate only under 'block' enforcement."""
83
+ return (self.enforcement == "block"
84
+ and self.level_for_grade(percentile) == "block")
85
+
86
+ @classmethod
87
+ def load(cls, path: str | None = None, repo_path: str = ".") -> "GateConfig":
88
+ """Load from an explicit path, else the first `.impact-gate.y*ml` in repo_path."""
89
+ cfg = cls()
90
+ found = path or _discover(repo_path)
91
+ if not found:
92
+ return cfg
93
+ import yaml # optional; only needed when a config file exists
94
+ with open(found) as fh:
95
+ data = yaml.safe_load(fh) or {}
96
+ for key in _SCALAR_KEYS:
97
+ if key in data and data[key] is not None:
98
+ setattr(cfg, key, data[key])
99
+ cfg.validate()
100
+ return cfg
101
+
102
+ def validate(self) -> None:
103
+ if self.enforcement not in ENFORCEMENTS:
104
+ raise ValueError(f"enforcement must be one of {ENFORCEMENTS}, got "
105
+ f"{self.enforcement!r}")
106
+ if self.tolerance <= 0:
107
+ raise ValueError("tolerance must be > 0")
108
+ for name in ("warn_percentile", "block_percentile"):
109
+ p = getattr(self, name)
110
+ if not 0 < p < 100:
111
+ raise ValueError(f"{name} must be within (0, 100), got {p}")
112
+ if self.warn_percentile > self.block_percentile:
113
+ raise ValueError("warn_percentile must be <= block_percentile")
114
+ if self.curve_prior_weight < 0:
115
+ raise ValueError("curve_prior_weight (K) must be >= 0")
116
+
117
+
118
+ def _discover(repo_path: str) -> str | None:
119
+ for name in CONFIG_NAMES:
120
+ p = os.path.join(repo_path, name)
121
+ if os.path.isfile(p):
122
+ return p
123
+ return None
@@ -0,0 +1,10 @@
1
+ """Self-contained structural change-impact core.
2
+
3
+ Vendored so the tool is standalone — no dependency on the (concluded) Surveyor
4
+ bug-finding experiment. Importing this package registers the default lizard plugin.
5
+ """
6
+ from .units import Unit, get_plugin, register # noqa: F401
7
+ from . import lizard_plugin # noqa: F401 (registers default plugin)
8
+ from .impact import FileImpact, UnitImpact, compute_file_impact # noqa: F401
9
+ from .config import MeasureConfig # noqa: F401
10
+ from .gitplumb import GitRepo, parse_diff # noqa: F401
@@ -0,0 +1,102 @@
1
+ """Measure configuration: which files are source, how to group, ignore globs.
2
+
3
+ Structural-decay focused: languages, ignore globs, test-path heuristics, and the
4
+ rename similarity threshold. (No bug-keyword mining — that belonged to the earlier
5
+ defect-prediction experiment, not to a structural-decay gate.)
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import fnmatch
10
+ from dataclasses import dataclass, field
11
+
12
+ # Extension -> language label. lizard picks its reader by filename, so this table only
13
+ # decides "is this a source file worth parsing", plus the label used in reports.
14
+ LANG_BY_EXT: dict[str, str] = {
15
+ ".java": "java",
16
+ ".cs": "csharp",
17
+ ".c": "c", ".h": "c", ".cc": "cpp", ".cpp": "cpp", ".cxx": "cpp",
18
+ ".hpp": "cpp", ".hh": "cpp", ".hxx": "cpp",
19
+ ".js": "javascript", ".jsx": "javascript", ".mjs": "javascript", ".cjs": "javascript",
20
+ ".ts": "typescript", ".tsx": "typescript",
21
+ ".py": "python",
22
+ ".go": "go",
23
+ ".kt": "kotlin", ".kts": "kotlin",
24
+ ".swift": "swift",
25
+ ".rb": "ruby",
26
+ ".php": "php",
27
+ ".rs": "rust",
28
+ ".scala": "scala",
29
+ ".m": "objectivec", ".mm": "objectivec",
30
+ ".lua": "lua",
31
+ ".ttcn": "ttcn",
32
+ }
33
+
34
+ DEFAULT_IGNORE = [
35
+ "**/node_modules/**", "**/dist/**", "**/build/**", "**/target/**",
36
+ "**/out/**", "**/bin/**", "**/obj/**", "**/third_party/**",
37
+ "**/vendor/**", "**/vendors/**", "**/bower_components/**", "**/webjars/**",
38
+ "**/.venv/**", "**/venv/**", "**/__pycache__/**", "**/.git/**",
39
+ "**/*.min.js", "**/*.min.css", "**/*.bundle.js", "**/*.generated.*",
40
+ "**/generated/**", "**/gen/**",
41
+ ]
42
+
43
+ DEFAULT_TEST_PATTERNS = [
44
+ "**/test/**", "**/tests/**", "**/__tests__/**", "**/spec/**",
45
+ "**/*Test.*", "**/*Tests.*", "**/*_test.*", "**/test_*.*",
46
+ "**/*.test.*", "**/*.spec.*",
47
+ ]
48
+
49
+
50
+ @dataclass
51
+ class MeasureConfig:
52
+ ignore: list[str] = field(default_factory=lambda: list(DEFAULT_IGNORE))
53
+ test_patterns: list[str] = field(default_factory=lambda: list(DEFAULT_TEST_PATTERNS))
54
+ lang_by_ext: dict[str, str] = field(default_factory=lambda: dict(LANG_BY_EXT))
55
+ rename_jaccard: float = 0.6 # body token-set similarity to call a rename
56
+ max_diff_lines: int = 200_000 # skip pathological mega-diffs (generated dumps)
57
+
58
+ def ext(self, path: str) -> str:
59
+ i = path.rfind(".")
60
+ return path[i:].lower() if i >= 0 else ""
61
+
62
+ def is_source(self, path: str) -> bool:
63
+ return not self.is_ignored(path) and self.ext(path) in self.lang_by_ext
64
+
65
+ def language(self, path: str) -> str | None:
66
+ return self.lang_by_ext.get(self.ext(path))
67
+
68
+ def is_ignored(self, path: str) -> bool:
69
+ # Match each glob against the path, and also with a leading "**/" stripped, so a
70
+ # pattern like "**/vendor/**" ignores a TOP-LEVEL vendor/ too. fnmatch's "*" spans
71
+ # "/", but "**/vendor/**" still requires a parent segment before "vendor", which
72
+ # let a repo-root vendor/ or node_modules/ slip through.
73
+ for pat in self.ignore:
74
+ if fnmatch.fnmatch(path, pat):
75
+ return True
76
+ if pat.startswith("**/") and fnmatch.fnmatch(path, pat[3:]):
77
+ return True
78
+ return False
79
+
80
+ def is_test(self, path: str) -> bool:
81
+ return any(fnmatch.fnmatch(path, pat) for pat in self.test_patterns)
82
+
83
+ @classmethod
84
+ def load(cls, path: str | None) -> "MeasureConfig":
85
+ cfg = cls()
86
+ if not path:
87
+ return cfg
88
+ import yaml # optional; only needed when a config file is passed
89
+ with open(path) as fh:
90
+ data = yaml.safe_load(fh) or {}
91
+ for key in ("rename_jaccard", "max_diff_lines"):
92
+ if key in data:
93
+ setattr(cfg, key, data[key])
94
+ # ignore / test_patterns APPEND to the defaults (a config adds, never silently
95
+ # drops). To start from scratch, set the matching `*_replace` key instead.
96
+ cfg.ignore = list(data["ignore_replace"]) if "ignore_replace" in data \
97
+ else cfg.ignore + list(data.get("ignore", []))
98
+ cfg.test_patterns = list(data["test_patterns_replace"]) if "test_patterns_replace" in data \
99
+ else cfg.test_patterns + list(data.get("test_patterns", []))
100
+ if "lang_by_ext" in data:
101
+ cfg.lang_by_ext.update(data["lang_by_ext"])
102
+ return cfg