argocd-source-lint 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. argocd_source_lint/__init__.py +10 -0
  2. argocd_source_lint/app_projects.py +53 -0
  3. argocd_source_lint/applicationset.py +451 -0
  4. argocd_source_lint/baseline.py +94 -0
  5. argocd_source_lint/cli.py +221 -0
  6. argocd_source_lint/coverage.py +284 -0
  7. argocd_source_lint/fsutil.py +199 -0
  8. argocd_source_lint/git_context.py +301 -0
  9. argocd_source_lint/globs.py +70 -0
  10. argocd_source_lint/loader.py +153 -0
  11. argocd_source_lint/models.py +125 -0
  12. argocd_source_lint/policy.py +60 -0
  13. argocd_source_lint/reporters/__init__.py +0 -0
  14. argocd_source_lint/reporters/fingerprint.py +19 -0
  15. argocd_source_lint/reporters/gitlab_codequality.py +36 -0
  16. argocd_source_lint/reporters/json_report.py +18 -0
  17. argocd_source_lint/reporters/junit.py +67 -0
  18. argocd_source_lint/reporters/sarif.py +125 -0
  19. argocd_source_lint/reporters/table.py +53 -0
  20. argocd_source_lint/rules/__init__.py +0 -0
  21. argocd_source_lint/rules/base.py +37 -0
  22. argocd_source_lint/rules/broken_values_ref.py +199 -0
  23. argocd_source_lint/rules/double_coverage.py +64 -0
  24. argocd_source_lint/rules/duplicate_application_name.py +67 -0
  25. argocd_source_lint/rules/hpa_selfheal_conflict.py +153 -0
  26. argocd_source_lint/rules/known-operators.yaml +89 -0
  27. argocd_source_lint/rules/malformed_ignore_diff_jq_expression.py +62 -0
  28. argocd_source_lint/rules/malformed_ignore_diff_pointer.py +55 -0
  29. argocd_source_lint/rules/malformed_sync_wave.py +79 -0
  30. argocd_source_lint/rules/missing_ignore_diff.py +181 -0
  31. argocd_source_lint/rules/orphan_source.py +105 -0
  32. argocd_source_lint/rules/phantom_target.py +73 -0
  33. argocd_source_lint/rules/project_scope.py +204 -0
  34. argocd_source_lint/rules/revision_mismatch.py +61 -0
  35. argocd_source_lint/rules/sync_validation_disabled.py +53 -0
  36. argocd_source_lint/rules/unknown_resource_hook.py +116 -0
  37. argocd_source_lint/rules/unknown_sync_option.py +151 -0
  38. argocd_source_lint-1.0.0.dist-info/METADATA +306 -0
  39. argocd_source_lint-1.0.0.dist-info/RECORD +42 -0
  40. argocd_source_lint-1.0.0.dist-info/WHEEL +4 -0
  41. argocd_source_lint-1.0.0.dist-info/entry_points.txt +2 -0
  42. argocd_source_lint-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,10 @@
1
+ from __future__ import annotations
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+
6
+ def tool_version() -> str:
7
+ try:
8
+ return version("argocd-source-lint")
9
+ except PackageNotFoundError:
10
+ return "0.0.0-dev"
@@ -0,0 +1,53 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import Any
5
+
6
+ from argocd_source_lint.fsutil import discover_documents
7
+ from argocd_source_lint.models import AppProject, AppProjectDestination
8
+
9
+
10
+ def discover_app_projects(repo_root: Path) -> list[AppProject]:
11
+ """Walks the repo for `kind: AppProject` manifests — used only by
12
+ `project-scope-violation`. A project referenced by an Application but
13
+ not found here is either genuinely undeclared (ArgoCD's own
14
+ permissive auto-created "default") or managed out-of-band in another
15
+ repo — the rule itself decides which, this just reports what's here."""
16
+ projects: list[AppProject] = []
17
+ for manifest_path, doc in discover_documents(repo_root):
18
+ if _is_app_project(doc):
19
+ projects.append(_build_app_project(doc, manifest_path, repo_root))
20
+ return projects
21
+
22
+
23
+ def _is_app_project(doc: dict[str, Any]) -> bool:
24
+ return doc.get("kind") == "AppProject" and str(doc.get("apiVersion", "")).startswith(
25
+ "argoproj.io/"
26
+ )
27
+
28
+
29
+ def _build_app_project(doc: dict[str, Any], manifest_path: Path, repo_root: Path) -> AppProject:
30
+ metadata = doc.get("metadata") or {}
31
+ spec = doc.get("spec") or {}
32
+
33
+ destinations = [
34
+ AppProjectDestination(
35
+ server=entry.get("server"),
36
+ name=entry.get("name"),
37
+ namespace=entry.get("namespace"),
38
+ )
39
+ for entry in spec.get("destinations") or []
40
+ if isinstance(entry, dict)
41
+ ]
42
+
43
+ try:
44
+ source_file = manifest_path.relative_to(repo_root)
45
+ except ValueError:
46
+ source_file = manifest_path
47
+
48
+ return AppProject(
49
+ name=metadata.get("name", ""),
50
+ source_repos=[str(entry) for entry in spec.get("sourceRepos") or []],
51
+ destinations=destinations,
52
+ source_file=source_file,
53
+ )
@@ -0,0 +1,451 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from dataclasses import dataclass
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from ruamel.yaml import YAML
9
+ from ruamel.yaml.error import YAMLError
10
+
11
+ from argocd_source_lint.fsutil import discover_documents, is_within_budget, walk_tree
12
+ from argocd_source_lint.git_context import (
13
+ is_local_repo_url,
14
+ materialize_revision,
15
+ revision_matches_checkout,
16
+ )
17
+ from argocd_source_lint.globs import match_glob
18
+ from argocd_source_lint.loader import build_application
19
+ from argocd_source_lint.models import Application, Finding, Severity
20
+
21
+ RULE_ID = "unresolvable-generator"
22
+
23
+
24
+ @dataclass(frozen=True, slots=True)
25
+ class GeneratorContext:
26
+ """Everything a generator resolver needs besides its own generator
27
+ dict, threaded unchanged through every level of `_resolve_generator`'s
28
+ recursion (`matrix`/`merge` calling back into it for each child) --
29
+ bundled here instead of six positional parameters repeated across
30
+ every one of those functions."""
31
+
32
+ repo_root: Path
33
+ local_origin: str | None
34
+ appset_name: str
35
+ source_file: Path
36
+ severity: Severity
37
+
38
+
39
+ # Classic ApplicationSet templating (`{{key}}`, valyala/fasttemplate) —
40
+ # `spec.goTemplate: true` switches to Go template syntax instead, which is
41
+ # a different rendering engine entirely and out of scope v1 (see below).
42
+ _PLACEHOLDER_RE = re.compile(r"\{\{\s*([\w.\-]+)\s*\}\}")
43
+
44
+ _yaml_safe = YAML(typ="safe")
45
+
46
+
47
+ def discover(
48
+ repo_root: Path, local_origin: str | None, severity: Severity
49
+ ) -> tuple[list[Application], list[Finding]]:
50
+ """Expands every `ApplicationSet` in the repo into the `Application`s
51
+ its generators would produce, so the existing rules apply to them
52
+ unchanged. A generator this tool can't resolve locally (requires a
53
+ live cluster/API, or Go-template rendering) produces one `info`
54
+ finding instead of guessing — same principle as an external
55
+ `Application` source (see DESIGN.md)."""
56
+ applications: list[Application] = []
57
+ findings: list[Finding] = []
58
+
59
+ for manifest_path, doc in discover_documents(repo_root):
60
+ if not _is_application_set(doc):
61
+ continue
62
+ expanded_apps, doc_findings = _expand(doc, manifest_path, repo_root, local_origin, severity)
63
+ applications.extend(expanded_apps)
64
+ findings.extend(doc_findings)
65
+
66
+ return applications, findings
67
+
68
+
69
+ def _is_application_set(doc: dict[str, Any]) -> bool:
70
+ return doc.get("kind") == "ApplicationSet" and str(doc.get("apiVersion", "")).startswith(
71
+ "argoproj.io/"
72
+ )
73
+
74
+
75
+ def _expand(
76
+ doc: dict[str, Any],
77
+ manifest_path: Path,
78
+ repo_root: Path,
79
+ local_origin: str | None,
80
+ severity: Severity,
81
+ ) -> tuple[list[Application], list[Finding]]:
82
+ metadata = doc.get("metadata", {}) or {}
83
+ spec = doc.get("spec", {}) or {}
84
+ try:
85
+ source_file = manifest_path.relative_to(repo_root)
86
+ except ValueError:
87
+ source_file = manifest_path
88
+ ctx = GeneratorContext(
89
+ repo_root=repo_root,
90
+ local_origin=local_origin,
91
+ appset_name=metadata.get("name", ""),
92
+ source_file=source_file,
93
+ severity=severity,
94
+ )
95
+
96
+ if spec.get("goTemplate"):
97
+ return [], [
98
+ _finding(
99
+ ctx,
100
+ "`goTemplate: true` (Go template rendering) is out of scope v1 — "
101
+ "only the classic `{{key}}` substitution is supported.",
102
+ )
103
+ ]
104
+
105
+ template = spec.get("template") or {}
106
+ applications: list[Application] = []
107
+ findings: list[Finding] = []
108
+
109
+ for generator in spec.get("generators") or []:
110
+ if not isinstance(generator, dict):
111
+ continue
112
+ param_sets, generator_findings = _resolve_generator(generator, ctx)
113
+ findings.extend(generator_findings)
114
+ for params in param_sets:
115
+ applications.append(
116
+ _build_generated_application(template, params, manifest_path, repo_root)
117
+ )
118
+
119
+ return applications, findings
120
+
121
+
122
+ def _build_generated_application(
123
+ template: dict[str, Any], params: dict[str, str], manifest_path: Path, repo_root: Path
124
+ ) -> Application:
125
+ rendered = _substitute(template, params)
126
+ doc = {
127
+ "apiVersion": "argoproj.io/v1alpha1",
128
+ "kind": "Application",
129
+ "metadata": rendered.get("metadata") or {},
130
+ "spec": rendered.get("spec") or {},
131
+ }
132
+ return build_application(doc, manifest_path, repo_root)
133
+
134
+
135
+ def _resolve_generator(
136
+ generator: dict[str, Any], ctx: GeneratorContext
137
+ ) -> tuple[list[dict[str, str]], list[Finding]]:
138
+ if generator.get("selector"):
139
+ return [], [
140
+ _finding(
141
+ ctx,
142
+ "generator has a `selector` (label filter on the generated params) — "
143
+ "this tool doesn't evaluate label selectors, out of scope v1; every "
144
+ "combination is left unexpanded rather than guessed at.",
145
+ )
146
+ ]
147
+
148
+ if "list" in generator:
149
+ return _resolve_list(generator.get("list") or {}), []
150
+ if "git" in generator:
151
+ return _resolve_git(generator.get("git") or {}, ctx)
152
+ if "matrix" in generator:
153
+ return _resolve_matrix(generator.get("matrix") or {}, ctx)
154
+ if "merge" in generator:
155
+ return _resolve_merge(generator.get("merge") or {}, ctx)
156
+
157
+ kind = next(iter(generator), "unknown")
158
+ return [], [
159
+ _finding(
160
+ ctx,
161
+ f"`{kind}` generator requires live cluster/API access — out of scope v1, "
162
+ "this tool only reads the local Git checkout.",
163
+ )
164
+ ]
165
+
166
+
167
+ def _resolve_list(list_generator: dict[str, Any]) -> list[dict[str, str]]:
168
+ elements = list_generator.get("elements") or []
169
+ return [_flatten_params(element) for element in elements if isinstance(element, dict)]
170
+
171
+
172
+ def _resolve_git(
173
+ git_generator: dict[str, Any], ctx: GeneratorContext
174
+ ) -> tuple[list[dict[str, str]], list[Finding]]:
175
+ repo_url = git_generator.get("repoURL", "")
176
+ if not is_local_repo_url(repo_url, ctx.local_origin):
177
+ return [], [
178
+ _finding(
179
+ ctx,
180
+ "git generator targets a different repo — out of scope v1, this tool "
181
+ "only verifies sources in the repo it runs in, see DESIGN.md.",
182
+ )
183
+ ]
184
+
185
+ # The generator's own `revision` is independent of any generated
186
+ # Application's `targetRevision` (confirmed against the upstream Git
187
+ # generator docs) -- discovering `directories`/`files` from the
188
+ # checked-out working tree regardless was a real, silent-wrong-result
189
+ # gap: a `revision` pinned away from HEAD would enumerate today's
190
+ # directory structure, not the pinned one. Same snapshot mechanism as
191
+ # `coverage._resolved_source_root` (DESIGN.md "targetRevision drift"),
192
+ # reused here rather than a second implementation.
193
+ revision = git_generator.get("revision") or "HEAD"
194
+ matches_checkout = revision_matches_checkout(ctx.repo_root, revision)
195
+ if matches_checkout is None:
196
+ # `is None`, not `is False` -- a revision that doesn't resolve at
197
+ # all must never fall through as "matches HEAD" by default.
198
+ return [], [
199
+ _finding(
200
+ ctx,
201
+ f"git generator's revision `{revision}` missing from the local "
202
+ "checkout — unable to tell which directories/files it would "
203
+ "discover. Add `fetch-depth: 0` or fetch the branch in question "
204
+ "in CI.",
205
+ severity=Severity.UNVERIFIABLE,
206
+ )
207
+ ]
208
+
209
+ base_root = ctx.repo_root
210
+ if matches_checkout is False:
211
+ snapshot_root = materialize_revision(ctx.repo_root, revision)
212
+ if snapshot_root is None:
213
+ return [], [
214
+ _finding(
215
+ ctx,
216
+ f"git generator's revision `{revision}` could not be extracted "
217
+ "from the local checkout.",
218
+ severity=Severity.UNVERIFIABLE,
219
+ )
220
+ ]
221
+ base_root = snapshot_root
222
+
223
+ param_sets: list[dict[str, str]] = []
224
+ param_sets.extend(_resolve_git_directories(base_root, git_generator.get("directories") or []))
225
+ param_sets.extend(_resolve_git_files(base_root, git_generator.get("files") or []))
226
+ return param_sets, []
227
+
228
+
229
+ def _resolve_git_directories(repo_root: Path, entries: list[Any]) -> list[dict[str, str]]:
230
+ all_dirs = _list_local_directories(repo_root)
231
+ included: set[str] = set()
232
+ excluded: set[str] = set()
233
+ for entry in entries:
234
+ if not isinstance(entry, dict):
235
+ continue
236
+ pattern = str(entry.get("path", ""))
237
+ matches = {d for d in all_dirs if match_glob(pattern, d)}
238
+ if entry.get("exclude"):
239
+ excluded |= matches
240
+ else:
241
+ included |= matches
242
+
243
+ return [_directory_params(d) for d in sorted(included - excluded)]
244
+
245
+
246
+ def _resolve_git_files(repo_root: Path, entries: list[Any]) -> list[dict[str, str]]:
247
+ all_files = _list_local_files(repo_root)
248
+ matched: set[str] = set()
249
+ for entry in entries:
250
+ if not isinstance(entry, dict):
251
+ continue
252
+ pattern = str(entry.get("path", ""))
253
+ matched |= {f for f in all_files if match_glob(pattern, f)}
254
+
255
+ param_sets: list[dict[str, str]] = []
256
+ for rel_path in sorted(matched):
257
+ content = _load_params_file(repo_root / rel_path)
258
+ if content is None:
259
+ continue
260
+ params = _flatten_params(content)
261
+ params.update(_directory_params(rel_path))
262
+ param_sets.append(params)
263
+ return param_sets
264
+
265
+
266
+ # Real-world matrix uses (environments x regions, clusters x apps) rarely
267
+ # reach even the low hundreds -- generous headroom, chosen the same way as
268
+ # fsutil's `_MAX_EXPANDED_NODES`: far below where the cost actually starts
269
+ # to matter. Confirmed for real, not theoretical: two `list` generators of
270
+ # 5,000 small elements each (comfortably under the alias-bomb node budget
271
+ # on their own -- that budget catches a densely *aliased* document, not a
272
+ # large but flat one) produced 25,000,000 combinations in ~7s for the
273
+ # combine step alone, before a single generated Application is even built
274
+ # or run through a rule.
275
+ _MAX_MATRIX_COMBINATIONS = 10_000
276
+
277
+
278
+ def _resolve_matrix(
279
+ matrix_generator: dict[str, Any], ctx: GeneratorContext
280
+ ) -> tuple[list[dict[str, str]], list[Finding]]:
281
+ children = matrix_generator.get("generators") or []
282
+
283
+ if len(children) > 2:
284
+ # ArgoCD's own matrix generator only supports combining exactly
285
+ # two child generators -- the controller reports an error on more
286
+ # (see DESIGN.md), it doesn't just behave unpredictably. Guessing
287
+ # at a 3+-way cartesian product here would report Applications
288
+ # ArgoCD itself would never actually generate.
289
+ return [], [
290
+ _finding(
291
+ ctx,
292
+ "matrix generator has more than 2 child generators — ArgoCD only "
293
+ "supports combining exactly two and errors out on more, so this tool "
294
+ "doesn't guess at what it would generate either.",
295
+ )
296
+ ]
297
+
298
+ findings: list[Finding] = []
299
+ param_lists: list[list[dict[str, str]]] = []
300
+
301
+ for child in children:
302
+ if not isinstance(child, dict):
303
+ continue
304
+ params, child_findings = _resolve_generator(child, ctx)
305
+ findings.extend(child_findings)
306
+ param_lists.append(params)
307
+
308
+ if len(param_lists) < 2 or any(not params for params in param_lists):
309
+ return [], findings
310
+
311
+ total_combinations = 1
312
+ for params in param_lists:
313
+ total_combinations *= len(params)
314
+ if total_combinations > _MAX_MATRIX_COMBINATIONS:
315
+ sizes = " x ".join(str(len(params)) for params in param_lists)
316
+ return [], [
317
+ *findings,
318
+ _finding(
319
+ ctx,
320
+ f"matrix generator would produce {total_combinations} combinations "
321
+ f"({sizes}) — over this tool's {_MAX_MATRIX_COMBINATIONS} safety limit, "
322
+ "not computed rather than risking an expensive or unbounded cartesian "
323
+ "product.",
324
+ ),
325
+ ]
326
+
327
+ combined: list[dict[str, str]] = [{}]
328
+ for params in param_lists:
329
+ combined = [{**base, **entry} for base in combined for entry in params]
330
+ return combined, findings
331
+
332
+
333
+ def _resolve_merge(
334
+ merge_generator: dict[str, Any], ctx: GeneratorContext
335
+ ) -> tuple[list[dict[str, str]], list[Finding]]:
336
+ merge_keys = [str(key) for key in (merge_generator.get("mergeKeys") or [])]
337
+ if not merge_keys:
338
+ return [], [
339
+ _finding(
340
+ ctx,
341
+ "merge generator has no `mergeKeys` — matching semantics are "
342
+ "unspecified upstream, this tool doesn't guess at them.",
343
+ )
344
+ ]
345
+
346
+ children = merge_generator.get("generators") or []
347
+ findings: list[Finding] = []
348
+ child_results: list[tuple[list[dict[str, str]], bool]] = []
349
+
350
+ for child in children:
351
+ if not isinstance(child, dict):
352
+ continue
353
+ params, child_findings = _resolve_generator(child, ctx)
354
+ findings.extend(child_findings)
355
+ child_results.append((params, bool(child_findings)))
356
+
357
+ if not child_results:
358
+ return [], findings
359
+
360
+ base_params, base_unresolvable = child_results[0]
361
+ if base_unresolvable:
362
+ # No base entries to match against at all — same reasoning as
363
+ # matrix's cross product being empty when a factor is empty.
364
+ return [], findings
365
+
366
+ # Base entries are kept even without a match in a later generator
367
+ # (see DESIGN.md); a later generator only overrides fields on an
368
+ # entry whose merge keys already match one from the base, and its
369
+ # own non-matching entries are discarded rather than added as new
370
+ # ones. An unresolvable later generator just contributes no override
371
+ # — its own finding above already flags the gap, so this doesn't
372
+ # silently drop the (fully known) base entries over it.
373
+ merged = [dict(entry) for entry in base_params]
374
+ by_key = {tuple(entry.get(key, "") for key in merge_keys): entry for entry in merged}
375
+
376
+ for params, was_unresolvable in child_results[1:]:
377
+ if was_unresolvable:
378
+ continue
379
+ for entry in params:
380
+ target = by_key.get(tuple(entry.get(key, "") for key in merge_keys))
381
+ if target is not None:
382
+ target.update(entry)
383
+
384
+ return merged, findings
385
+
386
+
387
+ def _list_local_directories(repo_root: Path) -> list[str]:
388
+ return [
389
+ path.relative_to(repo_root).as_posix() for path in walk_tree(repo_root) if path.is_dir()
390
+ ]
391
+
392
+
393
+ def _list_local_files(repo_root: Path) -> list[str]:
394
+ return [
395
+ path.relative_to(repo_root).as_posix() for path in walk_tree(repo_root) if path.is_file()
396
+ ]
397
+
398
+
399
+ def _directory_params(rel_path: str) -> dict[str, str]:
400
+ basename = rel_path.rsplit("/", 1)[-1]
401
+ normalized = re.sub(r"[^A-Za-z0-9-]", "-", basename).strip("-").lower() or "x"
402
+ return {"path": rel_path, "path.basename": basename, "path.basenameNormalized": normalized}
403
+
404
+
405
+ def _load_params_file(path: Path) -> Any:
406
+ if not path.is_file():
407
+ return None
408
+ try:
409
+ with path.open("r", encoding="utf-8") as f:
410
+ content = _yaml_safe.load(f) # valid JSON is also valid YAML
411
+ except (YAMLError, UnicodeDecodeError, OSError):
412
+ return None
413
+ # This bypasses fsutil.load_yaml_documents' own budget check (a
414
+ # `files:` generator target isn't a `kind: Application`-shaped
415
+ # document, so it never goes through that path) -- checked directly
416
+ # instead of letting a YAML alias bomb reach _flatten_params' str().
417
+ return content if is_within_budget(content) else None
418
+
419
+
420
+ def _flatten_params(obj: Any, prefix: str = "") -> dict[str, str]:
421
+ flat: dict[str, str] = {}
422
+ if isinstance(obj, dict):
423
+ for key, value in obj.items():
424
+ dotted = f"{prefix}.{key}" if prefix else str(key)
425
+ flat.update(_flatten_params(value, dotted))
426
+ elif isinstance(obj, list):
427
+ if prefix:
428
+ flat[prefix] = str(obj)
429
+ elif prefix:
430
+ flat[prefix] = "" if obj is None else str(obj)
431
+ return flat
432
+
433
+
434
+ def _substitute(value: Any, params: dict[str, str]) -> Any:
435
+ if isinstance(value, str):
436
+ return _PLACEHOLDER_RE.sub(lambda m: params.get(m.group(1), m.group(0)), value)
437
+ if isinstance(value, dict):
438
+ return {key: _substitute(v, params) for key, v in value.items()}
439
+ if isinstance(value, list):
440
+ return [_substitute(v, params) for v in value]
441
+ return value
442
+
443
+
444
+ def _finding(ctx: GeneratorContext, message: str, *, severity: Severity | None = None) -> Finding:
445
+ return Finding(
446
+ rule_id=RULE_ID,
447
+ severity=ctx.severity if severity is None else severity,
448
+ application=ctx.appset_name,
449
+ message=message,
450
+ file=ctx.source_file,
451
+ )
@@ -0,0 +1,94 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import Any
5
+
6
+ from ruamel.yaml import YAML
7
+
8
+ from argocd_source_lint.models import Finding
9
+
10
+ DEFAULT_BASELINE_FILENAME = ".argocd-lint-baseline.yaml"
11
+
12
+ _HEADER = (
13
+ "# Findings accepted at a point in time — see DESIGN.md.\n"
14
+ "# New findings still block CI; regenerate with `--write-baseline`.\n"
15
+ )
16
+
17
+ FindingKey = tuple[str, str, str, str]
18
+
19
+
20
+ def _key(finding: Finding) -> FindingKey:
21
+ return (finding.rule_id, finding.file.as_posix(), finding.application, finding.message)
22
+
23
+
24
+ def _key_from_entry(entry: dict[str, Any]) -> FindingKey:
25
+ return (
26
+ entry.get("rule_id", ""),
27
+ entry.get("file", ""),
28
+ entry.get("application", ""),
29
+ entry.get("message", ""),
30
+ )
31
+
32
+
33
+ def load_baseline(repo_root: Path, filename: str = DEFAULT_BASELINE_FILENAME) -> set[FindingKey]:
34
+ """Loads the baseline file from the repo root. Missing file means an
35
+ empty baseline (nothing accepted yet), same convention as
36
+ `policy.load_policy`."""
37
+ baseline_file = repo_root / filename
38
+ if not baseline_file.exists():
39
+ return set()
40
+
41
+ yaml = YAML(typ="safe")
42
+ with baseline_file.open("r", encoding="utf-8") as f:
43
+ raw_entries = yaml.load(f) or []
44
+ return {_key_from_entry(entry) for entry in raw_entries}
45
+
46
+
47
+ def write_baseline(
48
+ repo_root: Path, findings: list[Finding], filename: str = DEFAULT_BASELINE_FILENAME
49
+ ) -> Path:
50
+ """Writes every current finding to the baseline file, accepting them
51
+ all at once — the onboarding path for an existing repo. Overwrites any
52
+ previous baseline outright (it is meant to be regenerated, not
53
+ hand-merged)."""
54
+ entries = [
55
+ {
56
+ "rule_id": finding.rule_id,
57
+ "file": finding.file.as_posix(),
58
+ "application": finding.application,
59
+ "message": finding.message,
60
+ }
61
+ for finding in findings
62
+ ]
63
+
64
+ baseline_file = repo_root / filename
65
+ yaml = YAML()
66
+ yaml.default_flow_style = False
67
+ with baseline_file.open("w", encoding="utf-8") as f:
68
+ f.write(_HEADER)
69
+ yaml.dump(entries, f)
70
+ return baseline_file
71
+
72
+
73
+ def split_by_baseline(
74
+ findings: list[Finding], baseline: set[FindingKey]
75
+ ) -> tuple[list[Finding], list[Finding]]:
76
+ """Splits `findings` into (new, known) — `known` are already accepted
77
+ in the baseline and should be suppressed from the report and the exit
78
+ code, `new` are everything else."""
79
+ new: list[Finding] = []
80
+ known: list[Finding] = []
81
+ for finding in findings:
82
+ (known if _key(finding) in baseline else new).append(finding)
83
+ return new, known
84
+
85
+
86
+ def stale_baseline_entries(findings: list[Finding], baseline: set[FindingKey]) -> set[FindingKey]:
87
+ """Baseline entries matching none of the current `findings` — the
88
+ underlying issue was fixed, renamed, or the file/Application removed,
89
+ so the entry no longer suppresses anything. Purely informational
90
+ (never affects the exit code): a stale entry is dead weight, not a
91
+ new risk, but left to grow forever it stops being something a
92
+ reviewer can actually read (see DESIGN.md)."""
93
+ current = {_key(finding) for finding in findings}
94
+ return baseline - current