crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/ratchet.py ADDED
@@ -0,0 +1,236 @@
1
+ """The committed ratchet: per-function high-water CRAP marks for functions above target.
2
+
3
+ Pure text <-> entries. Marks only ever tighten: an improvement lowers or drops
4
+ an entry on update; a regression NEVER raises one (it surfaces as a verdict
5
+ failure instead); brand-new debt enters only through the audited override.
6
+ Identity is (path, long_name): spans drift with every edit, names survive.
7
+
8
+ The file opens with a metric stamp comment naming the analysis version and the
9
+ lizard that produced the numbers. Marks measured under different rules are not
10
+ comparable, and without the stamp that reads as a clean run.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ from typing import NamedTuple
15
+
16
+ from .score import ScoredRow
17
+
18
+ _HEADER = "path\tlong_name\tcrap"
19
+
20
+
21
+ class RatchetEntry(NamedTuple):
22
+ path: str
23
+ long_name: str
24
+ crap: float
25
+
26
+
27
+ def stamp_text(analysis_version: int, lizard_version: str) -> str:
28
+ """The metric identity a set of marks was measured under, as one line."""
29
+ return f"crapkit-analysis={analysis_version} lizard={lizard_version}"
30
+
31
+
32
+ def metric_version() -> str:
33
+ # Imported here, not at module scope: the git merge driver runs this module
34
+ # in a temp dir with no analysis stack, and it passes its own stamp through.
35
+ import lizard
36
+
37
+ from .analyze import ANALYSIS_VERSION
38
+ return stamp_text(ANALYSIS_VERSION, lizard.version)
39
+
40
+
41
+ def read_stamp(text: str) -> str:
42
+ """The metric a marks file was written under; "" for one written before stamping."""
43
+ for line in text.splitlines():
44
+ if line.startswith("#"):
45
+ return line[1:].strip()
46
+ if line.strip():
47
+ return ""
48
+ return ""
49
+
50
+
51
+ def stamp_conflict(recorded: str, current: str) -> str | None:
52
+ """The refusal when marks and the running metric disagree; None when they compare.
53
+
54
+ An unstamped file has nothing to disagree with — the caller warns instead.
55
+ """
56
+ if not recorded or recorded == current:
57
+ return None
58
+ return (f"ratchet marks were recorded under [{recorded}] but this run measures "
59
+ f"[{current}] — CRAP scores are not comparable across metric versions; "
60
+ "re-baseline with `crapkit ratchet seed`")
61
+
62
+
63
+ def _is_skippable(line: str) -> bool:
64
+ """Blank lines, the header and comment lines (the metric stamp) carry no mark."""
65
+ return not line.strip() or line == _HEADER or line.startswith("#")
66
+
67
+
68
+ def load_ratchet(text: str) -> list[RatchetEntry]:
69
+ entries = []
70
+ for i, line in enumerate(text.splitlines()):
71
+ if _is_skippable(line):
72
+ continue
73
+ parts = line.split("\t")
74
+ if len(parts) != 3:
75
+ raise ValueError(f"ratchet line {i + 1} has {len(parts)} fields, expected 3: {line!r}")
76
+ entries.append(RatchetEntry(parts[0], parts[1], float(parts[2])))
77
+ return entries
78
+
79
+
80
+ def mark_for(entries: list[RatchetEntry], path: str, long_name: str) -> float | None:
81
+ """One function's recorded high-water mark, or None when it carries no mark."""
82
+ for e in entries:
83
+ if e.path == path and e.long_name == long_name:
84
+ return e.crap
85
+ return None
86
+
87
+
88
+ def dump_ratchet(entries: list[RatchetEntry], *, stamp: str | None = None) -> str:
89
+ """`stamp` None takes the running metric; a version string is written verbatim
90
+ and "" writes none, which is how the merge driver keeps two legacy sides legacy."""
91
+ version = metric_version() if stamp is None else stamp
92
+ lines = [f"# {version}"] if version else []
93
+ lines.append(_HEADER)
94
+ for e in sorted(entries, key=lambda e: (e.path, e.long_name)):
95
+ lines.append(f"{e.path}\t{e.long_name}\t{e.crap:.4f}")
96
+ return "\n".join(lines) + "\n"
97
+
98
+
99
+ def _merge_key(b: float | None, o: float | None, t: float | None) -> float | None:
100
+ """git 3-way semantics per key: the changed side wins over the unchanged one.
101
+ Both changed: keep the mark (it can only fall; prune is re-runnable) at min."""
102
+ if o == t:
103
+ return o
104
+ if o == b:
105
+ return t
106
+ if t == b:
107
+ return o
108
+ # both changed differently; both-None is impossible past the o == t check
109
+ return min(x for x in (o, t) if x is not None)
110
+
111
+
112
+ def merge_ratchets(base: list[RatchetEntry], ours: list[RatchetEntry],
113
+ theirs: list[RatchetEntry]) -> list[RatchetEntry]:
114
+ b = {(e.path, e.long_name): e.crap for e in base}
115
+ o = {(e.path, e.long_name): e.crap for e in ours}
116
+ t = {(e.path, e.long_name): e.crap for e in theirs}
117
+ merged = []
118
+ for key in sorted(set(b) | set(o) | set(t)):
119
+ crap = _merge_key(b.get(key), o.get(key), t.get(key))
120
+ if crap is not None:
121
+ merged.append(RatchetEntry(key[0], key[1], crap))
122
+ return merged
123
+
124
+
125
+ def seed_ratchet(prior: list[RatchetEntry], fresh: list[ScoredRow], *, target: int,
126
+ scope_targets: dict[str, int] | None = None) -> tuple[list[RatchetEntry], int, int]:
127
+ """First-class mark entry: record every over-ceiling function at its current
128
+ CRAP. A mark never rises, seeding included — an existing lower mark stays."""
129
+ from .verify import worst_twins
130
+
131
+ ceilings = scope_targets or {}
132
+ marks = {(e.path, e.long_name): e for e in prior}
133
+ added = tightened = 0
134
+ for key, row in worst_twins(fresh).items():
135
+ if row.crap <= ceilings.get(row.scope, target):
136
+ continue
137
+ a, t = _seed_mark(marks, key, row.crap)
138
+ added += a
139
+ tightened += t
140
+ return sorted(marks.values(), key=lambda e: (e.path, e.long_name)), added, tightened
141
+
142
+
143
+ def _seed_mark(marks: dict, key: tuple, crap: float) -> tuple[int, int]:
144
+ """Place one mark; returns (added, tightened) deltas. A mark never rises."""
145
+ old = marks.get(key)
146
+ mark = round(crap, 4)
147
+ if old is not None and mark >= old.crap:
148
+ return 0, 0
149
+ marks[key] = RatchetEntry(key[0], key[1], mark)
150
+ return (1, 0) if old is None else (0, 1)
151
+
152
+
153
+ def _repath(entries: list[RatchetEntry], dest_of) -> tuple[list[RatchetEntry], int]:
154
+ """Rewrite paths by a per-mark chooser (None leaves a mark alone); values never move."""
155
+ out = []
156
+ moved = 0
157
+ for e in entries:
158
+ dest = dest_of(e)
159
+ if dest is None:
160
+ out.append(e)
161
+ continue
162
+ out.append(e._replace(path=dest))
163
+ moved += 1
164
+ return sorted(out, key=lambda e: (e.path, e.long_name)), moved
165
+
166
+
167
+ def _moved_path(path: str, old: str, new: str) -> str | None:
168
+ """One mark's destination under an explicit move, or None when `old` misses it."""
169
+ if old.endswith("/"):
170
+ return f"{new.rstrip('/')}/{path[len(old):]}" if path.startswith(old) else None
171
+ return new if path == old else None
172
+
173
+
174
+ def move_marks(entries: list[RatchetEntry], old: str,
175
+ new: str) -> tuple[list[RatchetEntry], int]:
176
+ """Re-path marks at their recorded values. A trailing "/" on `old` moves a whole
177
+ directory; anything else matches one exact path."""
178
+ return _repath(entries, lambda e: _moved_path(e.path, old, new))
179
+
180
+
181
+ def _rename_target(entry: RatchetEntry, present: set, renames: dict[str, str]) -> str | None:
182
+ """Where a mark should follow a renamed file, or None to leave it alone.
183
+
184
+ Three conditions, all required: the function is gone from its recorded path,
185
+ git calls that path renamed, and the SAME long_name exists at the new one. A
186
+ copy fails the first (the source survives), so its mark never travels.
187
+ """
188
+ if (entry.path, entry.long_name) in present:
189
+ return None
190
+ dest = renames.get(entry.path)
191
+ if dest is None or (dest, entry.long_name) not in present:
192
+ return None
193
+ return dest
194
+
195
+
196
+ def follow_renames(prior: list[RatchetEntry], fresh: list[ScoredRow],
197
+ renames: dict[str, str]) -> tuple[list[RatchetEntry], int]:
198
+ """Marks for renamed files, re-pathed. Runs BEFORE prune so a rename reads as a
199
+ move, not as code that left the repo and forfeits its high-water mark."""
200
+ from .verify import worst_twins
201
+
202
+ present = set(worst_twins(fresh))
203
+ return _repath(prior, lambda e: _rename_target(e, present, renames))
204
+
205
+
206
+ def prune_ratchet(prior: list[RatchetEntry],
207
+ fresh: list[ScoredRow]) -> tuple[list[RatchetEntry], int]:
208
+ """Deliberate mark exit: drop entries whose function is absent from the run.
209
+ The automatic update keeps them (an exclude glob or a lane outage also removes
210
+ rows); prune is the human confirming the code is really gone."""
211
+ from .verify import worst_twins
212
+
213
+ present = set(worst_twins(fresh))
214
+ kept = [e for e in prior if (e.path, e.long_name) in present]
215
+ return kept, len(prior) - len(kept)
216
+
217
+
218
+ def update_ratchet(prior: list[RatchetEntry], fresh: list[ScoredRow], *, target: int,
219
+ scope_targets: dict[str, int] | None = None) -> list[RatchetEntry]:
220
+ from .verify import worst_twins
221
+
222
+ fresh_by_key = worst_twins(fresh)
223
+ updated = []
224
+ for entry in prior:
225
+ row = fresh_by_key.get((entry.path, entry.long_name))
226
+ if row is None:
227
+ # Absent from the scored rows is NOT proof the code is gone — an
228
+ # exclude glob or a lane outage also removes it, and dropping the
229
+ # entry would erase an audited override's only diff-visible record.
230
+ # Stale entries are inert (verify checks only present functions).
231
+ updated.append(entry)
232
+ continue
233
+ if row.crap <= (scope_targets or {}).get(row.scope, target):
234
+ continue # fixed for real: below the scope's ceiling needs no mark
235
+ updated.append(RatchetEntry(entry.path, entry.long_name, min(entry.crap, round(row.crap, 4))))
236
+ return sorted(updated, key=lambda e: (e.path, e.long_name))
@@ -0,0 +1,135 @@
1
+ """Burn-down analytics from the ratchet file's own git history. Pure.
2
+
3
+ No timestamps live in the TSV; the commits that changed it carry them, so a
4
+ fixed history reports deterministically. Ages and velocity anchor on the
5
+ NEWEST commit in the history, never the wall clock.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ DAY = 86400
10
+
11
+
12
+ def _mark_line(line: str) -> tuple | None:
13
+ """((path, long_name), crap) when a +/- patch line is a mark row; else None."""
14
+ body = line[1:]
15
+ if body.startswith(("++ ", "-- ", "path\t")):
16
+ return None
17
+ parts = body.split("\t")
18
+ if len(parts) != 3:
19
+ return None
20
+ try:
21
+ return (parts[0], parts[1]), float(parts[2])
22
+ except ValueError:
23
+ return None
24
+
25
+
26
+ def _commit_delta(patch: str) -> tuple[dict, dict]:
27
+ """The marks one commit's patch added and removed, keyed."""
28
+ added: dict = {}
29
+ removed: dict = {}
30
+ for line in patch.splitlines():
31
+ if not line or line[0] not in "+-":
32
+ continue
33
+ mark = _mark_line(line)
34
+ if mark is None:
35
+ continue
36
+ (added if line[0] == "+" else removed)[mark[0]] = mark[1]
37
+ return added, removed
38
+
39
+
40
+ def mark_events(patches: list[tuple[int, str]]) -> list[tuple]:
41
+ """(ts, key, 'added'|'dropped', crap) per commit. A key removed and re-added
42
+ in the same commit is a tightening and emits nothing — only entry and
43
+ repayment count."""
44
+ events = []
45
+ for ts, patch in patches:
46
+ added, removed = _commit_delta(patch)
47
+ for key in sorted(set(added) - set(removed)):
48
+ events.append((ts, key, "added", added[key]))
49
+ for key in sorted(set(removed) - set(added)):
50
+ events.append((ts, key, "dropped", removed[key]))
51
+ return events
52
+
53
+
54
+ def _drop_velocity(dropped: list[int], anchor: int) -> dict:
55
+ return {"dropped_last_30d": sum(1 for ts in dropped if anchor - ts <= 30 * DAY),
56
+ "dropped_last_90d": sum(1 for ts in dropped if anchor - ts <= 90 * DAY)}
57
+
58
+
59
+ def _expired_marks(report: dict, max_age_months: int | None) -> list[str]:
60
+ if max_age_months is None:
61
+ return []
62
+ limit = max_age_months * 30
63
+ # "oldest" is age-sorted, so any expired mark appears in it; a backlog of
64
+ # more than its cap still reports the worst offenders
65
+ return [f"mark {e['long_name']} in {e['path']} is {e['age_days']}d old (limit {limit}d)"
66
+ for e in report["oldest"] if e["age_days"] > limit]
67
+
68
+
69
+ def _stalled_repayment(report: dict, min_repaid_30d: int | None) -> list[str]:
70
+ if min_repaid_30d is None or not report["open"]:
71
+ return []
72
+ if report["dropped_last_30d"] >= min_repaid_30d:
73
+ return []
74
+ return [f"repayment stalled: {report['dropped_last_30d']} mark(s) repaid in 30d "
75
+ f"(policy wants {min_repaid_30d})"]
76
+
77
+
78
+ def policy_violations(report: dict, max_age_months: int | None,
79
+ min_repaid_30d: int | None) -> list[str]:
80
+ """Findings against the debt policy knobs; no knobs, no findings."""
81
+ return _expired_marks(report, max_age_months) + _stalled_repayment(report, min_repaid_30d)
82
+
83
+
84
+ def _replay(events: list[tuple]) -> tuple[dict, dict, list[int]]:
85
+ """The committed state: when each surviving mark entered, what it is worth,
86
+ and the timestamp of every repayment."""
87
+ entered: dict = {}
88
+ crap: dict = {}
89
+ dropped: list[int] = []
90
+ for ts, key, kind, value in events:
91
+ if kind == "added":
92
+ entered[key] = ts
93
+ crap[key] = value
94
+ else:
95
+ entered.pop(key, None)
96
+ crap.pop(key, None)
97
+ dropped.append(ts)
98
+ return entered, crap, dropped
99
+
100
+
101
+ def _open_marks(entered: dict, working: dict | None, anchor: int) -> dict:
102
+ """Open marks keyed to when they entered. The working tree says WHICH marks
103
+ are open, history says how old each one is; a mark with no commit behind it
104
+ entered at the anchor and reports 0d."""
105
+ if working is None:
106
+ return entered
107
+ return {key: entered.get(key, anchor) for key in working}
108
+
109
+
110
+ def _uncommitted(crap: dict, working: dict | None) -> int:
111
+ """Marks the working tree and the newest committed version disagree on:
112
+ added, repaid or tightened on disk and not committed yet."""
113
+ if working is None:
114
+ return 0
115
+ return sum(1 for key in set(crap) | set(working) if crap.get(key) != working.get(key))
116
+
117
+
118
+ def _age_rows(opened: dict, anchor: int) -> list[dict]:
119
+ return sorted(({"path": k[0], "long_name": k[1], "age_days": (anchor - ts) // DAY}
120
+ for k, ts in opened.items()),
121
+ key=lambda e: (-e["age_days"], e["path"], e["long_name"]))
122
+
123
+
124
+ def report_from_events(events: list[tuple], working: dict | None = None) -> dict:
125
+ """`working` is the marks on disk, keyed (path, long_name) -> crap. It decides
126
+ which marks are open, because a seed that has not been committed is still debt
127
+ somebody owes. Repayment counts and velocity stay on committed history, which
128
+ is the only place a timestamp exists. None reports the committed state alone.
129
+ """
130
+ entered, crap, dropped = _replay(events)
131
+ anchor = max((ts for ts, *_ in events), default=0)
132
+ opened = _open_marks(entered, working, anchor)
133
+ return {"open": len(opened), "dropped_total": len(dropped), "anchor_ts": anchor,
134
+ "uncommitted": _uncommitted(crap, working),
135
+ "oldest": _age_rows(opened, anchor)[:20], **_drop_velocity(dropped, anchor)}
crapkit/sarif.py ADDED
@@ -0,0 +1,82 @@
1
+ """SARIF 2.1.0 and GitHub workflow-command emission. Pure builders.
2
+
3
+ Code-scanning UIs and PR annotation bots consume this; ruleIds, levels, and
4
+ locations are contract. Every uri is repo-relative with forward slashes.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from . import __version__
9
+
10
+ _RULES = (
11
+ {"id": "crapkit/over-target",
12
+ "shortDescription": {"text": "CRAP score above the scope ceiling"}},
13
+ {"id": "crapkit/gate",
14
+ "shortDescription": {"text": "touched function over the complexity gate"}},
15
+ {"id": "crapkit/ratchet-regression",
16
+ "shortDescription": {"text": "a recorded CRAP mark got worse"}},
17
+ )
18
+
19
+
20
+ def _result(rule_id: str, level: str, path: str, line: int, text: str) -> dict:
21
+ return {
22
+ "ruleId": rule_id, "level": level,
23
+ "message": {"text": text},
24
+ "locations": [{"physicalLocation": {
25
+ "artifactLocation": {"uri": path.replace("\\", "/")},
26
+ "region": {"startLine": line},
27
+ }}],
28
+ }
29
+
30
+
31
+ def over_target_results(scored, scope_targets: dict, target: int) -> list[dict]:
32
+ out = []
33
+ for r in scored:
34
+ ceiling = scope_targets.get(r.scope, target)
35
+ if r.crap <= ceiling:
36
+ continue
37
+ out.append(_result(
38
+ "crapkit/over-target", "warning", r.path, r.start,
39
+ f"{r.long_name}: CRAP {r.crap:.1f} over ceiling {ceiling} "
40
+ f"(ccn {r.ccn}, cov {r.cov:.0%}) -> {r.remedy}"))
41
+ return out
42
+
43
+
44
+ def gate_results(violations) -> list[dict]:
45
+ return [_result("crapkit/gate", "error", v.path, v.start,
46
+ f"{v.long_name}: CRAP {v.crap:.1f} (ccn {v.ccn}, cov {v.cov:.0%}) -> {v.remedy}")
47
+ for v in violations]
48
+
49
+
50
+ def regression_results(regressions) -> list[dict]:
51
+ # the ratchet stores no line numbers; line 1 anchors the file-level finding
52
+ return [_result("crapkit/ratchet-regression", "error", r.path, 1,
53
+ f"{r.long_name}: recorded {r.recorded} -> fresh {r.fresh_crap}")
54
+ for r in regressions]
55
+
56
+
57
+ def sarif_document(results: list[dict]) -> dict:
58
+ return {
59
+ "$schema": "https://raw.githubusercontent.com/oasis-tcs/sarif-spec/"
60
+ "master/Schemata/sarif-schema-2.1.0.json",
61
+ "version": "2.1.0",
62
+ "runs": [{
63
+ "tool": {"driver": {
64
+ "name": "crapkit",
65
+ "version": __version__,
66
+ "informationUri": "https://github.com/JeanFrancoisGagne/crapkit",
67
+ "rules": [dict(r) for r in _RULES],
68
+ }},
69
+ "results": results,
70
+ }],
71
+ }
72
+
73
+
74
+ def _esc(text: str) -> str:
75
+ return text.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
76
+
77
+
78
+ def github_annotation(result: dict) -> str:
79
+ loc = result["locations"][0]["physicalLocation"]
80
+ return (f"::{result['level']} file={loc['artifactLocation']['uri']},"
81
+ f"line={loc['region']['startLine']},title={result['ruleId']}"
82
+ f"::{_esc(result['message']['text'])}")
crapkit/sarifio.py ADDED
@@ -0,0 +1,49 @@
1
+ """Writing a SARIF document to disk, one finding at a time.
2
+
3
+ json.dump reaches CPython's pure-Python encoder whenever indent is set, so a
4
+ 36,767-finding report was serialized character by character in Python: 718 ms,
5
+ against 171 ms for the same findings handed to the C encoder. Nothing in the
6
+ SARIF spec asks for indentation and no consumer of this file is a human
7
+ scrolling 17 MB, so the document is written compact and streamed.
8
+
9
+ The skeleton comes from the builder itself, encoded with an empty results
10
+ array and split at that array. Findings are then C-encoded one at a time into
11
+ the gap, which lands byte for byte where json.dumps of the whole document
12
+ would have put them — and never builds that document as a string.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ from pathlib import Path
18
+
19
+ from .sarif import sarif_document
20
+
21
+ _EMPTY_RESULTS = '"results": []'
22
+
23
+
24
+ def _frame(document: dict) -> tuple[str, str]:
25
+ """The document either side of its results array."""
26
+ text = json.dumps(document, sort_keys=True)
27
+ head, found, tail = text.partition(_EMPTY_RESULTS)
28
+ if not found:
29
+ raise ValueError("sarif document has no results array to stream into")
30
+ return head + '"results": [', "]" + tail
31
+
32
+
33
+ def _write_results(handle, results) -> None:
34
+ # ", " is json.dumps' own item separator, which is what keeps the spliced
35
+ # bytes identical to a whole-document encode.
36
+ separator = ""
37
+ for result in results:
38
+ handle.write(separator + json.dumps(result, sort_keys=True))
39
+ separator = ", "
40
+
41
+
42
+ def write_sarif(path: Path | str, results: list[dict]) -> None:
43
+ """Write the SARIF report for these findings. newline="\\n" is the
44
+ determinism contract: the bytes must not pick up the host's separator."""
45
+ head, tail = _frame(sarif_document([]))
46
+ with open(path, "w", encoding="utf-8", newline="\n") as handle:
47
+ handle.write(head)
48
+ _write_results(handle, results)
49
+ handle.write(tail)