okstra 0.143.0 → 0.145.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -1
- package/docs/architecture.md +18 -2
- package/docs/cli.md +39 -2
- package/docs/project-structure-overview.md +19 -6
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/coding-preflight/overview.md +1 -1
- package/runtime/prompts/lead/convergence.md +11 -3
- package/runtime/prompts/lead/okstra-lead-contract.md +7 -1
- package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
- package/runtime/prompts/profiles/_common-contract.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +48 -2
- package/runtime/prompts/profiles/change-impact-analysis.md +24 -0
- package/runtime/prompts/profiles/feature-analysis.md +24 -0
- package/runtime/prompts/profiles/forbidden-actions.json +18 -0
- package/runtime/prompts/profiles/project-analysis.md +24 -0
- package/runtime/prompts/wizard/prompts.ko.json +44 -1
- package/runtime/python/okstra_ctl/analysis_inputs.py +369 -0
- package/runtime/python/okstra_ctl/clarification_items.py +74 -1
- package/runtime/python/okstra_ctl/mutation_probe.py +1263 -0
- package/runtime/python/okstra_ctl/render.py +77 -4
- package/runtime/python/okstra_ctl/render_final_report.py +13 -4
- package/runtime/python/okstra_ctl/report_views.py +134 -3
- package/runtime/python/okstra_ctl/run.py +118 -0
- package/runtime/python/okstra_ctl/run_context.py +34 -2
- package/runtime/python/okstra_ctl/schema_excerpt.py +12 -4
- package/runtime/python/okstra_ctl/self_mock_signals.py +183 -0
- package/runtime/python/okstra_ctl/user_response.py +309 -3
- package/runtime/python/okstra_ctl/wizard.py +545 -32
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +3 -0
- package/runtime/python/okstra_ctl/workflow.py +22 -0
- package/runtime/schemas/final-report-v1.0.schema.json +849 -3
- package/runtime/skills/okstra-run/SKILL.md +13 -1
- package/runtime/templates/reports/change-impact-analysis-input.template.md +58 -0
- package/runtime/templates/reports/feature-analysis-input.template.md +59 -0
- package/runtime/templates/reports/final-report.template.md +220 -0
- package/runtime/templates/reports/i18n/en.json +8 -0
- package/runtime/templates/reports/i18n/ko.json +8 -0
- package/runtime/templates/reports/project-analysis-input.template.md +58 -0
- package/runtime/templates/reports/report.js +84 -5
- package/runtime/templates/reports/user-response.template.md +19 -1
- package/runtime/validators/detect_self_mock.py +220 -0
- package/runtime/validators/validate-report-views.py +61 -7
- package/runtime/validators/validate-run.py +518 -0
- package/runtime/validators/validate_analysis_report.py +864 -0
- package/src/commands/execute/render-bundle.mjs +3 -0
|
@@ -0,0 +1,1263 @@
|
|
|
1
|
+
"""Tool-agnostic front door for gate B of the self-mock gate — mutation probing.
|
|
2
|
+
|
|
3
|
+
Gate A (``validators/detect_self_mock.py``) catches a test that stubs its own
|
|
4
|
+
subject by syntax. Gate B catches the ones syntax cannot see: if a mutant of a
|
|
5
|
+
changed line goes UNDETECTED by the stage's own suite, no test in it constrains
|
|
6
|
+
that line. Undetected covers two distinct failures, and both count — the test
|
|
7
|
+
ran the code and asserted nothing about it, or the test never reached the code
|
|
8
|
+
at all. The second is what a self-mocked test actually looks like from the
|
|
9
|
+
outside: stubbing the subject's own method means the real production line never
|
|
10
|
+
executes, so it is the signal this gate most needs to keep.
|
|
11
|
+
|
|
12
|
+
Every external mutation tool (Stryker, cargo-mutants, PIT) answers a different
|
|
13
|
+
CLI and a different report format, so each is wrapped in an `Adapter` and this
|
|
14
|
+
module owns everything that must NOT differ between them: which language has an
|
|
15
|
+
adapter at all, whether that adapter's tool is actually available, which mutants
|
|
16
|
+
are in scope, and how survivors become a verdict. Adapters parse; they do not
|
|
17
|
+
judge.
|
|
18
|
+
|
|
19
|
+
Two rules are load-bearing and both are about refusing a quiet pass:
|
|
20
|
+
|
|
21
|
+
- Nothing degrades silently. A language with no adapter, an adapter whose tool
|
|
22
|
+
is not declared, and a missing diff each answer `unsupported(<reason>)` naming
|
|
23
|
+
the cause. `validate-run.py` folds `unsupported(...)` down to gate A only, so
|
|
24
|
+
a silent "PASS" here would be indistinguishable from a real one.
|
|
25
|
+
- Only mutants covering a line the diff added or modified count. A survivor
|
|
26
|
+
elsewhere is pre-existing debt that this stage neither introduced nor is
|
|
27
|
+
blocked by.
|
|
28
|
+
- Having nothing to mutate is never a PASS. Production-source selection and the
|
|
29
|
+
empty-set refusal live in `run_probe`, ahead of every adapter, so "the tool
|
|
30
|
+
never ran" cannot be reported in the same words as "the tool ran and found
|
|
31
|
+
nothing undetected". Each adapter would otherwise have to re-earn that
|
|
32
|
+
distinction, and each one is a chance to lose it.
|
|
33
|
+
- A sibling language's PASS may stand over a CAPABILITY GAP, never over an
|
|
34
|
+
inspection failure. `--diff` is one shared file, so "incomplete for Rust" also
|
|
35
|
+
indicts the input that scoped the passing TypeScript verdict. The three reason
|
|
36
|
+
classes are defined once here (`classify_reason`) and read by both the merge
|
|
37
|
+
and `validate-run.py`'s gate.
|
|
38
|
+
- The two inputs must AGREE, completely. `changed_files` and the diff arrive
|
|
39
|
+
from separate git commands, so `run_probe` refuses to run unless the diff
|
|
40
|
+
names EVERY changed source — one uncovered target is enough to hide a real
|
|
41
|
+
survivor, since its lines never enter the intersection. A diff that names them
|
|
42
|
+
all but adds no line anywhere is refused too: there is nothing to verify, which
|
|
43
|
+
is not the same as verifying and finding nothing.
|
|
44
|
+
- An undetected mutant must be PLACEABLE, in BOTH halves of its `(file, line)`
|
|
45
|
+
key. Each adapter screens its own undetected rows — the line through
|
|
46
|
+
`_coerce_line`, the same test `evaluate` uses, and the file by requiring a
|
|
47
|
+
non-empty string — so a survivor that cannot be located fails its report
|
|
48
|
+
instead of being dropped on the way to the intersection. `evaluate`'s own check then stands as defense
|
|
49
|
+
for a future adapter that forgets, not as the live path for a real survivor.
|
|
50
|
+
- Only a CONCLUSIVE trial is evidence. A mutant that failed to compile, was
|
|
51
|
+
skipped, or never finished says nothing about the tests, so those outcomes are
|
|
52
|
+
recognised but not counted; a report made entirely of them is
|
|
53
|
+
`unsupported(no-conclusive-mutants)`. Every outcome vocabulary is matched as an
|
|
54
|
+
ALLOWLIST — an unrecognised word fails the report rather than being read as
|
|
55
|
+
"a test caught it".
|
|
56
|
+
|
|
57
|
+
"Adapters parse, they never judge" is enforced, not merely stated: `run_probe`
|
|
58
|
+
refuses any result whose `status` is outside the `PASS`/`FAIL`/`unsupported(...)`
|
|
59
|
+
vocabulary, because a fourth value matches neither of `validate-run.py`'s
|
|
60
|
+
branches and would pass by falling between them.
|
|
61
|
+
|
|
62
|
+
`ADAPTERS` is keyed by the `self_mock_signals.EXT_TO_LANG` vocabulary — the same
|
|
63
|
+
lang names gate A resolves a changed file to — so both gates answer to one set
|
|
64
|
+
of language keys: `ts_js` (Stryker), `rust` (cargo-mutants), and `java`/`kotlin`
|
|
65
|
+
(PIT, whose diff scoping is not wired up yet — see `PitAdapter`).
|
|
66
|
+
"""
|
|
67
|
+
from __future__ import annotations
|
|
68
|
+
|
|
69
|
+
import json
|
|
70
|
+
import re
|
|
71
|
+
import shutil
|
|
72
|
+
import subprocess
|
|
73
|
+
import sys
|
|
74
|
+
from pathlib import Path
|
|
75
|
+
from typing import NamedTuple, Protocol
|
|
76
|
+
|
|
77
|
+
from .self_mock_signals import (
|
|
78
|
+
EXT_TO_LANG,
|
|
79
|
+
partition_waived_entries,
|
|
80
|
+
selfmock_path_key,
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
# `survived` is read by a human in the sidecar, so it is trimmed. The verdict and
|
|
84
|
+
# the logged total are taken before the trim — the cap shortens the report, never
|
|
85
|
+
# the finding.
|
|
86
|
+
SURVIVOR_CAP = 50
|
|
87
|
+
|
|
88
|
+
ProbeResult = dict[str, object]
|
|
89
|
+
|
|
90
|
+
_HUNK = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
|
|
91
|
+
_DIFF_GIT_RE = re.compile(r"^diff --git a/(.+) b/(.+)$")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class Adapter(Protocol):
|
|
95
|
+
"""One external mutation tool, reduced to what the probe needs from it.
|
|
96
|
+
|
|
97
|
+
An adapter parses; it does not judge and it does not select. `run_probe`
|
|
98
|
+
picks the production sources and refuses an empty set before any adapter is
|
|
99
|
+
reached, so `run` is only ever called with at least one real target. That
|
|
100
|
+
ordering is the contract: "selected nothing, therefore PASS" is the vacuous
|
|
101
|
+
pass this gate exists to prevent, and every adapter would otherwise have to
|
|
102
|
+
remember not to reinvent it.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
name: str
|
|
106
|
+
|
|
107
|
+
def is_declared(self, worktree: Path | None) -> bool:
|
|
108
|
+
"""True when this tool is installed/configured in `worktree`.
|
|
109
|
+
|
|
110
|
+
Must answer from files alone — no subprocess, no network. `run_probe`
|
|
111
|
+
consults it before `run`, so shelling out here would be the very failure
|
|
112
|
+
it guards against.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
def run(
|
|
116
|
+
self,
|
|
117
|
+
targets: list[Path],
|
|
118
|
+
diff_path: Path | None,
|
|
119
|
+
worktree: Path | None,
|
|
120
|
+
) -> ProbeResult:
|
|
121
|
+
"""Mutate `targets` (already selected, never empty) and report via `evaluate`.
|
|
122
|
+
|
|
123
|
+
Anything that stops the tool producing a usable report — it was never
|
|
124
|
+
installed, it died, the report is unreadable, its scope cannot be
|
|
125
|
+
narrowed to the diff — is `unsupported(<reason>)`, never `PASS`.
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
STRYKER_REPORT_PATH = Path("reports/mutation/mutation.json")
|
|
130
|
+
|
|
131
|
+
# Stryker's `--mutate` takes PRODUCTION sources. Handing it a spec file mutates
|
|
132
|
+
# the test instead of the code, and handing it nothing at all makes it fall back
|
|
133
|
+
# to mutating the whole project. These mirror Stryker's own default test
|
|
134
|
+
# excludes; matching is case-insensitive so `Foo.Spec.ts` is caught too.
|
|
135
|
+
_TEST_NAME_MARKERS = (".spec.", ".test.")
|
|
136
|
+
_TEST_DIR_SEGMENTS = ("test", "tests", "spec", "__tests__")
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _is_test_source(path: Path) -> bool:
|
|
140
|
+
"""True for a file a mutation tool must not mutate.
|
|
141
|
+
|
|
142
|
+
Mutating test code produces mutants that sit on the stage's changed lines
|
|
143
|
+
and fail it for nothing. A gate that raises false alarms is one people learn
|
|
144
|
+
to route around, which costs more than the mutants it would have caught.
|
|
145
|
+
"""
|
|
146
|
+
name = path.name.lower()
|
|
147
|
+
if any(marker in name for marker in _TEST_NAME_MARKERS):
|
|
148
|
+
return True
|
|
149
|
+
# Only directory components — a production file may legitimately be named
|
|
150
|
+
# `spec.ts`, and `parts[:-1]` keeps the filename out of the comparison.
|
|
151
|
+
return any(part.lower() in _TEST_DIR_SEGMENTS for part in path.parts[:-1])
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def production_sources(changed_files: list[Path], lang: str) -> list[Path]:
|
|
155
|
+
"""The changed files of `lang` that a mutation tool may target."""
|
|
156
|
+
return [
|
|
157
|
+
p
|
|
158
|
+
for p in changed_files
|
|
159
|
+
if EXT_TO_LANG.get(p.suffix) == lang and not _is_test_source(p)
|
|
160
|
+
]
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _run_stryker_cli(targets: list[Path], worktree: Path | None) -> None:
|
|
164
|
+
"""Invoke the real Stryker CLI, scoped to `targets`, writing a json report.
|
|
165
|
+
|
|
166
|
+
`--no-install` keeps the promise `is_declared` makes: plain `npx stryker`
|
|
167
|
+
fetches the package from the registry when it is not installed locally, so
|
|
168
|
+
an undeclared worktree would quietly go to the network instead of reporting
|
|
169
|
+
`unsupported(...)`. Without an install this now fails fast and leaves no
|
|
170
|
+
report, which surfaces as `unsupported(report-unavailable)`.
|
|
171
|
+
|
|
172
|
+
A surviving mutant makes Stryker exit non-zero, which is a normal outcome
|
|
173
|
+
here rather than an error — the report is what carries the verdict, so the
|
|
174
|
+
exit status is deliberately not checked.
|
|
175
|
+
"""
|
|
176
|
+
subprocess.run(
|
|
177
|
+
[
|
|
178
|
+
"npx",
|
|
179
|
+
"--no-install",
|
|
180
|
+
"stryker",
|
|
181
|
+
"run",
|
|
182
|
+
"--reporters",
|
|
183
|
+
"json",
|
|
184
|
+
"--mutate",
|
|
185
|
+
",".join(str(t) for t in targets),
|
|
186
|
+
],
|
|
187
|
+
cwd=str(worktree) if worktree is not None else None,
|
|
188
|
+
check=False,
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# Stryker's own model is `Undetected = Survived + NoCoverage`, and both accuse
|
|
193
|
+
# the tests: `Survived` means the code ran and nothing asserted on it,
|
|
194
|
+
# `NoCoverage` means no test reached the code at all. NoCoverage is in fact the
|
|
195
|
+
# purest self-mock fingerprint — stubbing the subject's own method stops the
|
|
196
|
+
# real production line from ever executing. Every other status (`Killed`,
|
|
197
|
+
# `Timeout`, `RuntimeError`, `CompileError`, `Ignored`) either means a test
|
|
198
|
+
# caught the mutant or that no usable trial happened, so none of them counts.
|
|
199
|
+
#
|
|
200
|
+
# Deliberately local to this adapter: cargo-mutants and PIT report their
|
|
201
|
+
# outcomes in different vocabularies, so this must not be hoisted into the
|
|
202
|
+
# shared layer.
|
|
203
|
+
_STRYKER_UNDETECTED = ("Survived", "NoCoverage")
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# The FULL vocabulary Stryker emits. Matching against an allowlist rather than
|
|
207
|
+
# "anything that is not undetected" is what stops an unrecognised word — a
|
|
208
|
+
# renamed status, a newer Stryker, a truncated report — from being read as "a
|
|
209
|
+
# test caught it" and quietly clearing the run.
|
|
210
|
+
_STRYKER_STATUSES = (
|
|
211
|
+
"Killed",
|
|
212
|
+
"Survived",
|
|
213
|
+
"NoCoverage",
|
|
214
|
+
"Timeout",
|
|
215
|
+
"RuntimeError",
|
|
216
|
+
"CompileError",
|
|
217
|
+
"Ignored",
|
|
218
|
+
"Pending",
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
# The subset that represents a trial that actually finished and therefore says
|
|
222
|
+
# something about the tests. `CompileError` and `RuntimeError` mean the mutant
|
|
223
|
+
# never produced a usable trial, `Ignored` means it was skipped, and `Pending`
|
|
224
|
+
# means the run was cut short before it got there.
|
|
225
|
+
_STRYKER_CONCLUSIVE = ("Killed", "Survived", "NoCoverage", "Timeout")
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _stryker_line(mutant: dict) -> int | None:
|
|
229
|
+
"""The new-side line of a Stryker mutant, or `None` if it cannot be read.
|
|
230
|
+
|
|
231
|
+
Guards each hop rather than trusting the shape: a `location` or `start` that
|
|
232
|
+
is not an object would otherwise raise `AttributeError` out of `run_probe`.
|
|
233
|
+
That crash is worse than it looks — the probe runs before the sidecar is
|
|
234
|
+
written, so it would also destroy gate A's static result and leave the run
|
|
235
|
+
with no sidecar at all. Every malformed shape converges on `report-unparsed`
|
|
236
|
+
instead.
|
|
237
|
+
"""
|
|
238
|
+
location = mutant.get("location")
|
|
239
|
+
if not isinstance(location, dict):
|
|
240
|
+
return None
|
|
241
|
+
start = location.get("start")
|
|
242
|
+
if not isinstance(start, dict):
|
|
243
|
+
return None
|
|
244
|
+
return _coerce_line(start.get("line"))
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _survivors_from_report(
|
|
248
|
+
report: dict, worktree: Path | None
|
|
249
|
+
) -> ParsedReport | None:
|
|
250
|
+
"""Flatten a Stryker json report into rows plus the two trial counts.
|
|
251
|
+
|
|
252
|
+
`None` means the report could not be read at all, which is not the same
|
|
253
|
+
answer as "no undetected mutants" and must never become one.
|
|
254
|
+
|
|
255
|
+
`status` is carried through because the two undetected outcomes need
|
|
256
|
+
different fixes — write a real assertion (`Survived`) versus cover the code
|
|
257
|
+
at all (`NoCoverage`) — and the sidecar reader cannot tell them apart once
|
|
258
|
+
they are merged. `evaluate` treats the rows opaquely, so the extra key costs
|
|
259
|
+
the shared layer nothing.
|
|
260
|
+
"""
|
|
261
|
+
files = report.get("files")
|
|
262
|
+
if files is None:
|
|
263
|
+
files = {}
|
|
264
|
+
if not isinstance(files, dict):
|
|
265
|
+
return None
|
|
266
|
+
rows: list[dict] = []
|
|
267
|
+
conclusive = 0
|
|
268
|
+
observed = 0
|
|
269
|
+
for name, entry in files.items():
|
|
270
|
+
if not isinstance(entry, dict):
|
|
271
|
+
return None
|
|
272
|
+
mutants = entry.get("mutants")
|
|
273
|
+
if mutants is None:
|
|
274
|
+
mutants = []
|
|
275
|
+
if not isinstance(mutants, list):
|
|
276
|
+
return None
|
|
277
|
+
for mutant in mutants:
|
|
278
|
+
if not isinstance(mutant, dict):
|
|
279
|
+
return None
|
|
280
|
+
status = mutant.get("status")
|
|
281
|
+
if status not in _STRYKER_STATUSES:
|
|
282
|
+
return None
|
|
283
|
+
observed += 1
|
|
284
|
+
if status in _STRYKER_CONCLUSIVE:
|
|
285
|
+
conclusive += 1
|
|
286
|
+
if status not in _STRYKER_UNDETECTED:
|
|
287
|
+
# A DETECTED mutant's position is never matched against the diff,
|
|
288
|
+
# so a malformed location cannot change the verdict and is not
|
|
289
|
+
# worth failing a run over. Only undetected ones are read below.
|
|
290
|
+
continue
|
|
291
|
+
line = _stryker_line(mutant)
|
|
292
|
+
if line is None:
|
|
293
|
+
# An undetected mutant we cannot PLACE is unverifiable: `evaluate`
|
|
294
|
+
# could only drop it, turning the most serious finding a report
|
|
295
|
+
# carries into a PASS with a note on stderr.
|
|
296
|
+
return None
|
|
297
|
+
rows.append(
|
|
298
|
+
{
|
|
299
|
+
"file": _worktree_relative(name, worktree),
|
|
300
|
+
"line": line,
|
|
301
|
+
"mutant": mutant.get("mutatorName"),
|
|
302
|
+
"status": status,
|
|
303
|
+
}
|
|
304
|
+
)
|
|
305
|
+
return ParsedReport(rows, conclusive, observed)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
class ParsedReport(NamedTuple):
|
|
309
|
+
"""One tool's report, reduced to what the shared guards need from it.
|
|
310
|
+
|
|
311
|
+
`conclusive` is deliberately not `len(survivors)` nor `observed`: a mutant
|
|
312
|
+
that failed to compile, was skipped, or never finished says nothing about
|
|
313
|
+
the tests. Counting those as trials is how a run that proved nothing ends up
|
|
314
|
+
reported as clean.
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
survivors: list[dict]
|
|
318
|
+
conclusive: int
|
|
319
|
+
observed: int
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _verdict_from_parsed(
|
|
323
|
+
parsed: ParsedReport | None,
|
|
324
|
+
diff_path: Path | None,
|
|
325
|
+
worktree: Path | None,
|
|
326
|
+
tool: str,
|
|
327
|
+
) -> ProbeResult:
|
|
328
|
+
"""The guards every adapter needs once its own report shape is parsed.
|
|
329
|
+
|
|
330
|
+
Report-shape parsing stays local to each adapter — the formats have nothing
|
|
331
|
+
in common — but what an unreadable report, a mutant-free report and a report
|
|
332
|
+
of nothing but inconclusive trials MEAN is identical across tools, and all
|
|
333
|
+
three are answers only `unsupported` can carry. Keeping the decision here
|
|
334
|
+
means a new adapter inherits it instead of re-deriving it.
|
|
335
|
+
"""
|
|
336
|
+
if parsed is None:
|
|
337
|
+
return unsupported("report-unparsed", tool=tool)
|
|
338
|
+
if parsed.observed == 0:
|
|
339
|
+
# The tool ran but produced nothing to detect.
|
|
340
|
+
return unsupported("no-mutants-generated", tool=tool)
|
|
341
|
+
if parsed.conclusive == 0:
|
|
342
|
+
# Mutants existed but not one of them completed a real trial — an
|
|
343
|
+
# all-unviable build, or a truncated run. Never evidence of a good suite.
|
|
344
|
+
return unsupported("no-conclusive-mutants", tool=tool)
|
|
345
|
+
return evaluate(parsed.survivors, diff_path, worktree, tool=tool)
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
class StrykerAdapter:
|
|
349
|
+
"""Gate B for TypeScript / JavaScript, wrapping the Stryker CLI.
|
|
350
|
+
|
|
351
|
+
The CLI call is injected (`runner`) rather than hard-coded, so a test can
|
|
352
|
+
substitute the external tool — the genuine collaborator — while the
|
|
353
|
+
adapter's own selection, parsing and path handling still run for real.
|
|
354
|
+
"""
|
|
355
|
+
|
|
356
|
+
name = "stryker"
|
|
357
|
+
|
|
358
|
+
def __init__(self, runner=_run_stryker_cli, report_path=STRYKER_REPORT_PATH):
|
|
359
|
+
self._runner = runner
|
|
360
|
+
self._report_path = Path(report_path)
|
|
361
|
+
|
|
362
|
+
def is_declared(self, worktree: Path | None) -> bool:
|
|
363
|
+
"""True when Stryker is actually INSTALLED here. No subprocess, no network.
|
|
364
|
+
|
|
365
|
+
A `package.json` entry is not enough. `node_modules` is gitignored, so a
|
|
366
|
+
fresh stage worktree routinely declares `@stryker-mutator/core` with no
|
|
367
|
+
binary present; answering True there sends the run into
|
|
368
|
+
`npx --no-install`, which fails, writes no report, and BLOCKS the stage on
|
|
369
|
+
`report-unavailable` — failing a run over a tool nobody installed.
|
|
370
|
+
Requiring the binary makes that case `tool-not-declared`, a capability
|
|
371
|
+
gap, which is non-blocking. Symmetric with `CargoMutantsAdapter`, which
|
|
372
|
+
already requires the executable rather than the manifest entry.
|
|
373
|
+
|
|
374
|
+
`run_probe` calls this before `run`, so it must answer from files alone —
|
|
375
|
+
shelling out to check would be the very failure it is guarding against.
|
|
376
|
+
"""
|
|
377
|
+
if worktree is None:
|
|
378
|
+
return False
|
|
379
|
+
return (Path(worktree) / "node_modules" / ".bin" / "stryker").exists()
|
|
380
|
+
|
|
381
|
+
def run(
|
|
382
|
+
self,
|
|
383
|
+
targets: list[Path],
|
|
384
|
+
diff_path: Path | None,
|
|
385
|
+
worktree: Path | None,
|
|
386
|
+
) -> ProbeResult:
|
|
387
|
+
# Drop any earlier report FIRST. Single-stage final-verification reuses
|
|
388
|
+
# the implementation stage worktree, so a previous run's mutation.json is
|
|
389
|
+
# genuinely on disk; if Stryker then dies before writing, reading it back
|
|
390
|
+
# would report that older run's verdict for code it never saw.
|
|
391
|
+
self._report_file(worktree).unlink(missing_ok=True)
|
|
392
|
+
self._runner(targets, worktree)
|
|
393
|
+
report = self._read_report(worktree)
|
|
394
|
+
if report is None:
|
|
395
|
+
return unsupported("report-unavailable", tool=self.name)
|
|
396
|
+
return _verdict_from_parsed(
|
|
397
|
+
_survivors_from_report(report, worktree), diff_path, worktree, self.name
|
|
398
|
+
)
|
|
399
|
+
|
|
400
|
+
def _report_file(self, worktree: Path | None) -> Path:
|
|
401
|
+
if worktree is None:
|
|
402
|
+
return self._report_path
|
|
403
|
+
return Path(worktree) / self._report_path
|
|
404
|
+
|
|
405
|
+
def _read_report(self, worktree: Path | None) -> dict | None:
|
|
406
|
+
"""The parsed json report, or `None` when the run left none behind."""
|
|
407
|
+
path = self._report_file(worktree)
|
|
408
|
+
if not path.is_file():
|
|
409
|
+
return None
|
|
410
|
+
try:
|
|
411
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
412
|
+
except (OSError, json.JSONDecodeError):
|
|
413
|
+
return None
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
CARGO_OUTCOMES_PATH = Path("mutants.out/outcomes.json")
|
|
417
|
+
|
|
418
|
+
# cargo-mutants' own vocabulary: `caught` (a test failed, good), `missed` (no
|
|
419
|
+
# test failed), `unviable` (the mutant did not compile) and `timeout`. Only
|
|
420
|
+
# `missed` accuses the tests. Unlike Stryker and PIT there is no separate
|
|
421
|
+
# "not covered" word — cargo-mutants folds uncovered code into `missed`, so this
|
|
422
|
+
# tuple is one entry rather than two.
|
|
423
|
+
#
|
|
424
|
+
# Adapter-local on purpose: Stryker says `Survived`/`NoCoverage`, PIT says
|
|
425
|
+
# `SURVIVED`/`NO_COVERAGE`. Hoisting any of them into the shared layer would
|
|
426
|
+
# make one tool's vocabulary silently govern another's report.
|
|
427
|
+
_CARGO_UNDETECTED = ("missed",)
|
|
428
|
+
|
|
429
|
+
# The FULL vocabulary, matched as an allowlist. An unrecognised word is exactly
|
|
430
|
+
# the unverified case this parser must fail closed on: treating it as "not
|
|
431
|
+
# undetected" would let a renamed or suffixed outcome clear the whole run.
|
|
432
|
+
#
|
|
433
|
+
# UNVERIFIED, and the first thing a real cargo-mutants run must settle: if
|
|
434
|
+
# `outcomes.json` also carries a baseline scenario (a `summary` such as
|
|
435
|
+
# "Success"), every run here becomes `unsupported(report-unparsed)` and this
|
|
436
|
+
# adapter is permanently inert — and `len(entries)` would count that baseline as
|
|
437
|
+
# a mutant besides. Confirm the vocabulary AND whether a baseline row exists,
|
|
438
|
+
# then fix the allowlist and the `observed` count together. Do not add the word
|
|
439
|
+
# on speculation; a wrong guess here is exactly what the allowlist exists to
|
|
440
|
+
# catch.
|
|
441
|
+
_CARGO_OUTCOMES = ("caught", "missed", "unviable", "timeout")
|
|
442
|
+
|
|
443
|
+
# `unviable` means the mutant did not compile, so the tests never ran against
|
|
444
|
+
# it. Only the rest represent a finished trial.
|
|
445
|
+
_CARGO_CONCLUSIVE = ("caught", "missed", "timeout")
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def _run_cargo_mutants_cli(diff_path: Path, worktree: Path | None) -> None:
|
|
449
|
+
"""Invoke cargo-mutants scoped to `diff_path`, writing `mutants.out/`.
|
|
450
|
+
|
|
451
|
+
`--in-diff` restricts testing to mutants overlapping the diff's changed
|
|
452
|
+
regions; the file is expected to carry `b/`-prefixed names, which is exactly
|
|
453
|
+
what the `git diff` output the verifier writes contains. `--no-shuffle`
|
|
454
|
+
keeps the report order stable between runs.
|
|
455
|
+
|
|
456
|
+
A missed mutant makes cargo-mutants exit non-zero, which is the normal
|
|
457
|
+
outcome here rather than an error — the report carries the verdict.
|
|
458
|
+
"""
|
|
459
|
+
subprocess.run(
|
|
460
|
+
["cargo", "mutants", "--no-shuffle", "--in-diff", str(diff_path)],
|
|
461
|
+
cwd=str(worktree) if worktree is not None else None,
|
|
462
|
+
check=False,
|
|
463
|
+
)
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _cargo_mutant_location(entry: dict) -> tuple[object, object, object]:
|
|
467
|
+
"""`(file, line, description)` for one outcomes.json entry.
|
|
468
|
+
|
|
469
|
+
UNVERIFIED SHAPE. The reachable cargo-mutants documentation (mutants.rs)
|
|
470
|
+
pins the outcome words and that `mutants.out/outcomes.json` carries the
|
|
471
|
+
results, but not the key names inside it. This reads the nesting the tool is
|
|
472
|
+
believed to use, and every caller treats an unreadable entry as a reason to
|
|
473
|
+
fail the whole report rather than to skip a row — so if this guess is wrong
|
|
474
|
+
the run reports `unsupported(report-unparsed)` instead of a PASS bought with
|
|
475
|
+
our own parsing error. Confirm against a real run in Task 11.
|
|
476
|
+
"""
|
|
477
|
+
scenario = entry.get("scenario")
|
|
478
|
+
mutant = scenario.get("Mutant") if isinstance(scenario, dict) else None
|
|
479
|
+
if not isinstance(mutant, dict):
|
|
480
|
+
return None, None, None
|
|
481
|
+
return (
|
|
482
|
+
mutant.get("file"),
|
|
483
|
+
mutant.get("line"),
|
|
484
|
+
mutant.get("description") or mutant.get("function"),
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _survivors_from_cargo_outcomes(
|
|
489
|
+
report: dict, worktree: Path | None
|
|
490
|
+
) -> ParsedReport | None:
|
|
491
|
+
"""Undetected rows + trial counts, or `None` if the report cannot be read.
|
|
492
|
+
|
|
493
|
+
`None` is deliberately distinct from an empty result: "no missed mutants" is
|
|
494
|
+
evidence, "this report is not the shape we can read" is not, and only the
|
|
495
|
+
first may become a PASS.
|
|
496
|
+
"""
|
|
497
|
+
entries = report.get("outcomes")
|
|
498
|
+
if not isinstance(entries, list):
|
|
499
|
+
return None
|
|
500
|
+
rows: list[dict] = []
|
|
501
|
+
conclusive = 0
|
|
502
|
+
for entry in entries:
|
|
503
|
+
if not isinstance(entry, dict):
|
|
504
|
+
return None
|
|
505
|
+
summary = entry.get("summary")
|
|
506
|
+
if not isinstance(summary, str):
|
|
507
|
+
return None
|
|
508
|
+
word = summary.lower()
|
|
509
|
+
if word not in _CARGO_OUTCOMES:
|
|
510
|
+
return None
|
|
511
|
+
if word in _CARGO_CONCLUSIVE:
|
|
512
|
+
conclusive += 1
|
|
513
|
+
if word not in _CARGO_UNDETECTED:
|
|
514
|
+
continue
|
|
515
|
+
file, raw_line, describe = _cargo_mutant_location(entry)
|
|
516
|
+
line = _coerce_line(raw_line)
|
|
517
|
+
# Both halves of the placement key must actually place. A non-string
|
|
518
|
+
# `file` would survive `str()` into a key no diff can ever match — and
|
|
519
|
+
# unlike an unplaceable line, that one is dropped without even a note on
|
|
520
|
+
# stderr. The report shape is unverified, so an object here is plausible.
|
|
521
|
+
if not isinstance(file, str) or not file or line is None:
|
|
522
|
+
return None
|
|
523
|
+
rows.append(
|
|
524
|
+
{
|
|
525
|
+
"file": _worktree_relative(file, worktree),
|
|
526
|
+
"line": line,
|
|
527
|
+
"mutant": describe,
|
|
528
|
+
"status": summary,
|
|
529
|
+
}
|
|
530
|
+
)
|
|
531
|
+
return ParsedReport(rows, conclusive, len(entries))
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
class CargoMutantsAdapter:
|
|
535
|
+
"""Gate B for Rust, wrapping the cargo-mutants CLI."""
|
|
536
|
+
|
|
537
|
+
name = "cargo-mutants"
|
|
538
|
+
|
|
539
|
+
def __init__(self, runner=_run_cargo_mutants_cli, report_path=CARGO_OUTCOMES_PATH):
|
|
540
|
+
self._runner = runner
|
|
541
|
+
self._report_path = Path(report_path)
|
|
542
|
+
|
|
543
|
+
def is_declared(self, worktree: Path | None) -> bool:
|
|
544
|
+
"""A crate to mutate AND the subcommand installed. No subprocess.
|
|
545
|
+
|
|
546
|
+
`shutil.which` only stats candidate paths, so asking whether the binary
|
|
547
|
+
exists costs nothing and starts nothing.
|
|
548
|
+
"""
|
|
549
|
+
if worktree is None:
|
|
550
|
+
return False
|
|
551
|
+
if not (Path(worktree) / "Cargo.toml").is_file():
|
|
552
|
+
return False
|
|
553
|
+
return shutil.which("cargo-mutants") is not None
|
|
554
|
+
|
|
555
|
+
def run(
|
|
556
|
+
self,
|
|
557
|
+
targets: list[Path],
|
|
558
|
+
diff_path: Path | None,
|
|
559
|
+
worktree: Path | None,
|
|
560
|
+
) -> ProbeResult:
|
|
561
|
+
# `targets` is deliberately unused: the shared selection in `run_probe`
|
|
562
|
+
# decides WHETHER there is Rust production code worth running on, while
|
|
563
|
+
# `--in-diff` decides WHAT gets mutated. Passing a file list as well
|
|
564
|
+
# would give cargo-mutants a second, redundant scope to disagree with.
|
|
565
|
+
#
|
|
566
|
+
# Checked before running: `--in-diff` is the only thing keeping this to
|
|
567
|
+
# the stage's own changes, and without it cargo-mutants would mutate the
|
|
568
|
+
# whole crate — slow, and full of findings this stage never caused.
|
|
569
|
+
if diff_path is None or not Path(diff_path).is_file():
|
|
570
|
+
return unsupported("diff-unavailable", tool=self.name)
|
|
571
|
+
self._report_file(worktree).unlink(missing_ok=True)
|
|
572
|
+
self._runner(diff_path, worktree)
|
|
573
|
+
report = self._read_report(worktree)
|
|
574
|
+
if report is None:
|
|
575
|
+
return unsupported("report-unavailable", tool=self.name)
|
|
576
|
+
return _verdict_from_parsed(
|
|
577
|
+
_survivors_from_cargo_outcomes(report, worktree),
|
|
578
|
+
diff_path,
|
|
579
|
+
worktree,
|
|
580
|
+
self.name,
|
|
581
|
+
)
|
|
582
|
+
|
|
583
|
+
def _report_file(self, worktree: Path | None) -> Path:
|
|
584
|
+
if worktree is None:
|
|
585
|
+
return self._report_path
|
|
586
|
+
return Path(worktree) / self._report_path
|
|
587
|
+
|
|
588
|
+
def _read_report(self, worktree: Path | None) -> dict | None:
|
|
589
|
+
path = self._report_file(worktree)
|
|
590
|
+
if not path.is_file():
|
|
591
|
+
return None
|
|
592
|
+
try:
|
|
593
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
594
|
+
except (OSError, json.JSONDecodeError):
|
|
595
|
+
return None
|
|
596
|
+
return data if isinstance(data, dict) else None
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
_PIT_BUILD_FILES = ("pom.xml", "build.gradle", "build.gradle.kts")
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
class PitAdapter:
|
|
603
|
+
"""Gate B for Java / Kotlin — currently honest about not being able to run.
|
|
604
|
+
|
|
605
|
+
PIT's own statuses are SURVIVED, KILLED, NO_COVERAGE, TIMED_OUT, NON_VIABLE,
|
|
606
|
+
MEMORY_ERROR and RUN_ERROR, with undetected = SURVIVED + NO_COVERAGE, the
|
|
607
|
+
same two-failure shape Stryker has. Parsing that is the easy part.
|
|
608
|
+
|
|
609
|
+
Scoping is what stops it. PIT is not incapable of SCM scoping — it can
|
|
610
|
+
restrict analysis to files changed in source control — but two concrete
|
|
611
|
+
things block wiring it up here:
|
|
612
|
+
|
|
613
|
+
1. That scoping is the `scmMutationCoverage` **Maven goal**. There is no
|
|
614
|
+
equivalent among the build files `is_declared` accepts: a Gradle project
|
|
615
|
+
(`build.gradle` / `build.gradle.kts`) has no such goal to invoke, so the
|
|
616
|
+
adapter cannot offer one story for the languages it is registered under.
|
|
617
|
+
2. Joining PIT's report back to the diff needs a mapping from its
|
|
618
|
+
class-oriented output to repo paths — an unverified report-path↔FQCN
|
|
619
|
+
derivation that depends on each project's source-root layout. Guessing it
|
|
620
|
+
is how a survivor silently misses the changed-line intersection.
|
|
621
|
+
|
|
622
|
+
Running unscoped `mutationCoverage` instead would mutate the whole module:
|
|
623
|
+
slow enough that a stage would time out, and it would report mutants on
|
|
624
|
+
lines this stage never touched. So this adapter reports what is true — gate
|
|
625
|
+
B is not wired for the JVM yet — and lets gate A carry the run. `is_declared`
|
|
626
|
+
still distinguishes "PIT is not configured" from "PIT is configured but we
|
|
627
|
+
cannot scope it", which is the difference an operator acts on.
|
|
628
|
+
|
|
629
|
+
Wiring this up later means: a Maven-only path invoking `scmMutationCoverage`,
|
|
630
|
+
the XML `outputFormats` report, and a verified path mapping — none of which
|
|
631
|
+
should be guessed.
|
|
632
|
+
"""
|
|
633
|
+
|
|
634
|
+
name = "pit"
|
|
635
|
+
|
|
636
|
+
def is_declared(self, worktree: Path | None) -> bool:
|
|
637
|
+
"""True when a build file in this worktree configures PIT."""
|
|
638
|
+
if worktree is None:
|
|
639
|
+
return False
|
|
640
|
+
root = Path(worktree)
|
|
641
|
+
for build_file in _PIT_BUILD_FILES:
|
|
642
|
+
path = root / build_file
|
|
643
|
+
if not path.is_file():
|
|
644
|
+
continue
|
|
645
|
+
try:
|
|
646
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
647
|
+
except OSError:
|
|
648
|
+
continue
|
|
649
|
+
if "pitest" in text.lower():
|
|
650
|
+
return True
|
|
651
|
+
return False
|
|
652
|
+
|
|
653
|
+
def run(
|
|
654
|
+
self,
|
|
655
|
+
targets: list[Path],
|
|
656
|
+
diff_path: Path | None,
|
|
657
|
+
worktree: Path | None,
|
|
658
|
+
) -> ProbeResult:
|
|
659
|
+
return unsupported("diff-scope-unavailable", tool=self.name)
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
# Keyed by the `self_mock_signals.EXT_TO_LANG` vocabulary, which folds `.ts`,
|
|
663
|
+
# `.tsx`, `.js`, `.jsx` and `.mjs` into the single lang `ts_js`. A language
|
|
664
|
+
# absent here answers `unsupported(no-adapter:<lang>)`, which is the honest report.
|
|
665
|
+
_PIT = PitAdapter()
|
|
666
|
+
ADAPTERS: dict[str, Adapter] = {
|
|
667
|
+
"ts_js": StrykerAdapter(),
|
|
668
|
+
"rust": CargoMutantsAdapter(),
|
|
669
|
+
"java": _PIT,
|
|
670
|
+
"kotlin": _PIT,
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def unsupported(reason: str, tool: str | None = None) -> ProbeResult:
|
|
675
|
+
"""The one way to say "gate B did not run", always naming why."""
|
|
676
|
+
return {
|
|
677
|
+
"status": f"unsupported({reason})",
|
|
678
|
+
"tool": tool,
|
|
679
|
+
"survived": [],
|
|
680
|
+
"waived": [],
|
|
681
|
+
}
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
class DiffScope(NamedTuple):
|
|
685
|
+
"""What a unified diff says about a stage, in the two forms the gate needs.
|
|
686
|
+
|
|
687
|
+
`touched` — file → the new-side lines it added or modified. This is what
|
|
688
|
+
scopes a surviving mutant to work the stage actually did.
|
|
689
|
+
|
|
690
|
+
`mentioned` — every file the diff NAMES in a header, whether or not it added
|
|
691
|
+
a line. A renamed or mode-changed file is mentioned but not touched, and
|
|
692
|
+
that distinction is what lets the coverage check be total: every changed
|
|
693
|
+
source must be mentioned, while only the ones with added lines can carry
|
|
694
|
+
findings. Checking `touched` instead would fail a stage for renaming a file.
|
|
695
|
+
"""
|
|
696
|
+
|
|
697
|
+
touched: dict[str, set[int]]
|
|
698
|
+
mentioned: set[str]
|
|
699
|
+
|
|
700
|
+
|
|
701
|
+
def read_diff(diff_path: Path | None) -> DiffScope | None:
|
|
702
|
+
"""Parse a unified diff, or `None` when it cannot be read at all.
|
|
703
|
+
|
|
704
|
+
A modification shows up as a `-`/`+` pair, so tracking the `+` side alone
|
|
705
|
+
covers both added and modified lines.
|
|
706
|
+
|
|
707
|
+
`None` and an empty scope are different answers and callers must keep them
|
|
708
|
+
apart: `None` means the diff was unreadable (`unsupported(diff-unavailable)`),
|
|
709
|
+
while an empty `touched` means it was read and added nothing anywhere.
|
|
710
|
+
"""
|
|
711
|
+
if diff_path is None or not Path(diff_path).is_file():
|
|
712
|
+
return None
|
|
713
|
+
try:
|
|
714
|
+
text = Path(diff_path).read_text(encoding="utf-8", errors="replace")
|
|
715
|
+
except OSError:
|
|
716
|
+
return None
|
|
717
|
+
touched: dict[str, set[int]] = {}
|
|
718
|
+
mentioned: set[str] = set()
|
|
719
|
+
current: str | None = None
|
|
720
|
+
line_no = 0
|
|
721
|
+
for raw in text.splitlines():
|
|
722
|
+
if raw.startswith("diff --git "):
|
|
723
|
+
m = _DIFF_GIT_RE.match(raw)
|
|
724
|
+
if m:
|
|
725
|
+
mentioned.update(_strip_diff_prefix(g) for g in m.groups())
|
|
726
|
+
# A pure rename or mode change emits no `---`/`+++` pair and no
|
|
727
|
+
# hunk, so this header is the ONLY place those files are named.
|
|
728
|
+
# Both sides are recorded: a rename names the old and the new path.
|
|
729
|
+
continue
|
|
730
|
+
if raw.startswith("--- ") or raw.startswith("+++ "):
|
|
731
|
+
target = raw[4:].strip()
|
|
732
|
+
named = None if target == "/dev/null" else _strip_diff_prefix(target)
|
|
733
|
+
if named is not None:
|
|
734
|
+
mentioned.add(named)
|
|
735
|
+
if raw.startswith("+++ "):
|
|
736
|
+
# `/dev/null` on the new side is a deletion: nothing to mutate.
|
|
737
|
+
current = named
|
|
738
|
+
continue
|
|
739
|
+
if raw.startswith("@@"):
|
|
740
|
+
m = _HUNK.match(raw)
|
|
741
|
+
if m is None:
|
|
742
|
+
# A header we cannot read — a combined/merge diff (`@@@ -a -b +c @@@`)
|
|
743
|
+
# or a truncated file. Skipping it would drop every line in the
|
|
744
|
+
# hunk, and in a MIXED diff the coverage check would still be
|
|
745
|
+
# satisfied by the ordinary hunks while these changes went
|
|
746
|
+
# unchecked. The whole diff is unreadable instead.
|
|
747
|
+
return None
|
|
748
|
+
line_no = int(m.group(1))
|
|
749
|
+
continue
|
|
750
|
+
if current is None or not line_no:
|
|
751
|
+
continue
|
|
752
|
+
if raw.startswith("+"):
|
|
753
|
+
touched.setdefault(current, set()).add(line_no)
|
|
754
|
+
line_no += 1
|
|
755
|
+
elif raw.startswith(" ") or raw == "":
|
|
756
|
+
# Allowlist, not "anything but `-`": ``
|
|
757
|
+
# occupies no new-side line, and counting it would slide every later
|
|
758
|
+
# `+` down by one — the intersection would then miss the changed line
|
|
759
|
+
# and report PASS. Only a context line advances the counter.
|
|
760
|
+
line_no += 1
|
|
761
|
+
return DiffScope(touched, mentioned)
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
def _strip_diff_prefix(target: str) -> str:
|
|
765
|
+
"""Drop git's `b/` prefix and any trailing tab-separated timestamp."""
|
|
766
|
+
path = target.split("\t", 1)[0]
|
|
767
|
+
if path.startswith("a/") or path.startswith("b/"):
|
|
768
|
+
path = path[2:]
|
|
769
|
+
return selfmock_path_key(path)
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def _coerce_line(value: object) -> int | None:
|
|
773
|
+
"""A reported line as the `int` the diff index is keyed on, or `None`.
|
|
774
|
+
|
|
775
|
+
A tool that reports `"11"` names the same line as `11`; treating them as
|
|
776
|
+
different would drop the survivor and read as PASS. Anything that will not
|
|
777
|
+
coerce — a word, a `10-12` range, an object — places nowhere.
|
|
778
|
+
|
|
779
|
+
The single definition of "placeable": the adapters screen undetected mutants
|
|
780
|
+
with it so an unplaceable one fails its report, and `evaluate` applies it
|
|
781
|
+
again as the shared defense. Two spellings of this test could disagree, and
|
|
782
|
+
the row would fall between them into a PASS.
|
|
783
|
+
"""
|
|
784
|
+
try:
|
|
785
|
+
return int(value)
|
|
786
|
+
except (TypeError, ValueError):
|
|
787
|
+
return None
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
def _new_side_line(survivor: dict) -> int | None:
|
|
791
|
+
"""The survivor's placeable line, or `None`.
|
|
792
|
+
|
|
793
|
+
Reaching `None` here means an adapter let an unplaceable row through; the
|
|
794
|
+
row is skipped and logged rather than raised on, so one bad row cannot abort
|
|
795
|
+
a whole verdict.
|
|
796
|
+
"""
|
|
797
|
+
return _coerce_line(survivor.get("line"))
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _worktree_relative(path: str, worktree: Path | None) -> str:
|
|
801
|
+
"""Re-spell an absolute tool path relative to the worktree root.
|
|
802
|
+
|
|
803
|
+
Mutation tools report absolute file names; a diff names them relative to the
|
|
804
|
+
worktree. `selfmock_path_key` folds cosmetic drift but not this, so without
|
|
805
|
+
the conversion every survivor misses the intersection and the stage reads
|
|
806
|
+
PASS. A path already relative, or outside the worktree, is left alone.
|
|
807
|
+
"""
|
|
808
|
+
if worktree is None:
|
|
809
|
+
return path
|
|
810
|
+
try:
|
|
811
|
+
return str(Path(path).relative_to(Path(worktree)))
|
|
812
|
+
except ValueError:
|
|
813
|
+
return path
|
|
814
|
+
|
|
815
|
+
|
|
816
|
+
def _diff_key(survivor: dict, worktree: Path | None) -> str:
|
|
817
|
+
"""The spelling of a survivor's file that `read_diff` keys `touched` on."""
|
|
818
|
+
return selfmock_path_key(
|
|
819
|
+
_worktree_relative(str(survivor.get("file", "")), worktree)
|
|
820
|
+
)
|
|
821
|
+
|
|
822
|
+
|
|
823
|
+
def evaluate(
|
|
824
|
+
survivors: list[dict], diff_path: Path | None, worktree: Path | None, tool: str
|
|
825
|
+
) -> ProbeResult:
|
|
826
|
+
"""Turn an adapter's raw survivors into a verdict. Adapters must route here.
|
|
827
|
+
|
|
828
|
+
`survivors` are `{file, line, mutant}` rows. Two spellings are reconciled
|
|
829
|
+
here so no adapter has to get them right on its own — either mistake would
|
|
830
|
+
drop the survivor silently into a PASS:
|
|
831
|
+
|
|
832
|
+
- `file` is folded to worktree-relative and then through `selfmock_path_key`,
|
|
833
|
+
which also absorbs `./` prefixes, duplicated slashes and Windows
|
|
834
|
+
separators. Adapters should still emit worktree-relative paths, because
|
|
835
|
+
that is what lands in the sidecar for a human to read.
|
|
836
|
+
- `line` must be the NEW-side (post-change) line number. A string is coerced,
|
|
837
|
+
but an old-side number simply points somewhere else and cannot be rescued.
|
|
838
|
+
"""
|
|
839
|
+
scope = read_diff(diff_path)
|
|
840
|
+
if scope is None:
|
|
841
|
+
return unsupported("diff-unavailable", tool=tool)
|
|
842
|
+
touched = scope.touched
|
|
843
|
+
unplaceable = [s for s in survivors if _new_side_line(s) is None]
|
|
844
|
+
if unplaceable:
|
|
845
|
+
print(
|
|
846
|
+
f"mutation-probe: {tool} reported {len(unplaceable)} mutant(s) with no "
|
|
847
|
+
"usable line number; they cannot be matched against the diff and are "
|
|
848
|
+
"not counted",
|
|
849
|
+
file=sys.stderr,
|
|
850
|
+
)
|
|
851
|
+
covered = [
|
|
852
|
+
s
|
|
853
|
+
for s in survivors
|
|
854
|
+
if _new_side_line(s) in touched.get(_diff_key(s, worktree), ())
|
|
855
|
+
]
|
|
856
|
+
if len(covered) > SURVIVOR_CAP:
|
|
857
|
+
print(
|
|
858
|
+
f"mutation-probe: {tool} left {len(covered)} surviving mutants on changed "
|
|
859
|
+
f"lines; reporting the first {SURVIVOR_CAP} "
|
|
860
|
+
f"({len(covered) - SURVIVOR_CAP} more not listed)",
|
|
861
|
+
file=sys.stderr,
|
|
862
|
+
)
|
|
863
|
+
return {
|
|
864
|
+
"status": "FAIL" if covered else "PASS",
|
|
865
|
+
"tool": tool,
|
|
866
|
+
"survived": covered[:SURVIVOR_CAP],
|
|
867
|
+
# How many were actually found, before the cap shortened the list. The
|
|
868
|
+
# verdict is decided from this, never from the trimmed list: a reader can
|
|
869
|
+
# only waive what the report shows, so recomputing from `survived` alone
|
|
870
|
+
# would let the visible ones be cleared while the overflow stayed unseen.
|
|
871
|
+
"survivedTotal": len(covered),
|
|
872
|
+
"waived": [],
|
|
873
|
+
}
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
def _targets_absent_from_diff(
|
|
877
|
+
targets: list[Path], mentioned: set[str], worktree: Path | None
|
|
878
|
+
) -> list[str]:
|
|
879
|
+
"""The changed sources the diff never names. Empty means the two inputs agree.
|
|
880
|
+
|
|
881
|
+
TOTAL coverage, not overlap. Overlap left a real hole: with targets [A, B]
|
|
882
|
+
and a diff covering only A, the check passed on A while a survivor in B fell
|
|
883
|
+
outside the intersection and vanished into a PASS. Asking `mentioned` rather
|
|
884
|
+
than `touched` is what makes totality affordable — a renamed or mode-changed
|
|
885
|
+
target is named by the diff without adding a line, so it satisfies this
|
|
886
|
+
without being expected to carry findings.
|
|
887
|
+
"""
|
|
888
|
+
return [
|
|
889
|
+
key
|
|
890
|
+
for target in targets
|
|
891
|
+
if (key := selfmock_path_key(_worktree_relative(str(target), worktree)))
|
|
892
|
+
not in mentioned
|
|
893
|
+
]
|
|
894
|
+
|
|
895
|
+
|
|
896
|
+
def mutation_waivers(waivers) -> list[dict]:
|
|
897
|
+
"""The entries in the shared waiver file that belong to gate B.
|
|
898
|
+
|
|
899
|
+
ONE file (`<task_root>/qa/self-mock-waivers.json`) serves both gates so the
|
|
900
|
+
user manages a single place and `waiverSource` stays singular. A static
|
|
901
|
+
waiver names a `signal`; a mutation waiver names a `mutant` — the field is
|
|
902
|
+
the discriminator, so there is no `kind` for anyone to get wrong.
|
|
903
|
+
|
|
904
|
+
The filter is explicit rather than relying on the keys failing to line up: a
|
|
905
|
+
survivor whose `mutant` the tool left null would otherwise share the `None`
|
|
906
|
+
third element with a static entry and be cleared by it.
|
|
907
|
+
"""
|
|
908
|
+
return [
|
|
909
|
+
w
|
|
910
|
+
for w in (waivers or [])
|
|
911
|
+
if isinstance(w, dict)
|
|
912
|
+
and isinstance(w.get("mutant"), str)
|
|
913
|
+
and w["mutant"].strip()
|
|
914
|
+
]
|
|
915
|
+
|
|
916
|
+
|
|
917
|
+
def _apply_waivers(result: ProbeResult, waivers) -> ProbeResult:
|
|
918
|
+
"""Move user-acknowledged survivors out of `survived` and re-decide.
|
|
919
|
+
|
|
920
|
+
Mirrors gate A exactly, including what it refuses to do: this only MATCHES.
|
|
921
|
+
An entry missing `reason` or `acknowledgedBy` is carried into `waived` so
|
|
922
|
+
`validate-run.py` can block on it — rejecting it here would let the run that
|
|
923
|
+
produced the finding excuse itself by writing the file.
|
|
924
|
+
|
|
925
|
+
Only a real verdict can be waived. There is no finding to excuse when the
|
|
926
|
+
status is `unsupported(...)`, and letting a waiver rewrite that would turn
|
|
927
|
+
"gate B never ran" into a pass.
|
|
928
|
+
"""
|
|
929
|
+
if result["status"] not in ("PASS", "FAIL"):
|
|
930
|
+
return result
|
|
931
|
+
entries = mutation_waivers(waivers)
|
|
932
|
+
if not entries:
|
|
933
|
+
return result
|
|
934
|
+
listed = list(result["survived"])
|
|
935
|
+
remaining, waived = partition_waived_entries(listed, entries, "mutant")
|
|
936
|
+
return {
|
|
937
|
+
**result,
|
|
938
|
+
"status": _waived_verdict(result, listed, remaining),
|
|
939
|
+
"survived": remaining,
|
|
940
|
+
"waived": list(result["waived"]) + waived,
|
|
941
|
+
}
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
def _waived_verdict(
|
|
945
|
+
result: ProbeResult, listed: list[dict], remaining: list[dict]
|
|
946
|
+
) -> str:
|
|
947
|
+
"""`PASS` only when EVERY survivor was waived — including any the cap hid.
|
|
948
|
+
|
|
949
|
+
A user can only acknowledge what they were shown. Waiving every listed
|
|
950
|
+
survivor of a truncated report must not clear the ones that were trimmed:
|
|
951
|
+
they were never reviewed, and no acknowledgement for them can exist.
|
|
952
|
+
|
|
953
|
+
`survivedTotal` is what `evaluate` found before the cap. Each way it can be
|
|
954
|
+
untrustworthy fails closed:
|
|
955
|
+
|
|
956
|
+
- smaller than the list it accompanies — the result contradicts itself, so
|
|
957
|
+
nothing in it is reliable enough to clear a finding;
|
|
958
|
+
- absent — a list sitting exactly at the cap may be truncated, so it is
|
|
959
|
+
treated as if it is. Every real adapter routes through `evaluate` and does
|
|
960
|
+
carry the total, so this only catches a hand-built result.
|
|
961
|
+
"""
|
|
962
|
+
if remaining:
|
|
963
|
+
return "FAIL"
|
|
964
|
+
total = result.get("survivedTotal")
|
|
965
|
+
if isinstance(total, int) and total >= len(listed):
|
|
966
|
+
return "FAIL" if total > len(listed) else "PASS"
|
|
967
|
+
if total is not None:
|
|
968
|
+
return "FAIL"
|
|
969
|
+
return "FAIL" if len(listed) >= SURVIVOR_CAP else "PASS"
|
|
970
|
+
|
|
971
|
+
|
|
972
|
+
def run_probe(
|
|
973
|
+
lang: str,
|
|
974
|
+
changed_files: list[Path],
|
|
975
|
+
diff_path: Path | None,
|
|
976
|
+
worktree: Path | None,
|
|
977
|
+
waivers=(),
|
|
978
|
+
) -> ProbeResult:
|
|
979
|
+
"""Probe `lang`'s changed files, or say precisely why it could not be done.
|
|
980
|
+
|
|
981
|
+
`changed_files` must be EVERY file the stage changed, not a pre-filtered
|
|
982
|
+
subset — the adapter selects its own production sources from it. Forwarding
|
|
983
|
+
only the changed TEST files (gate A's `scannedFiles`, which is the tempting
|
|
984
|
+
thing to reuse) leaves nothing to mutate, and the run reports a vacuous PASS
|
|
985
|
+
with gate B silently dead behind it.
|
|
986
|
+
"""
|
|
987
|
+
adapter = ADAPTERS.get(lang)
|
|
988
|
+
if adapter is None:
|
|
989
|
+
return unsupported(f"no-adapter:{lang}")
|
|
990
|
+
# Checked before `run` so a missing binary surfaces as its own reason rather
|
|
991
|
+
# than as an adapter-specific crash or, worse, an empty survivor list.
|
|
992
|
+
if not adapter.is_declared(worktree):
|
|
993
|
+
return unsupported("tool-not-declared", tool=adapter.name)
|
|
994
|
+
# Selected HERE rather than inside each adapter: three adapters each doing
|
|
995
|
+
# their own selection is three chances to answer PASS for a run that never
|
|
996
|
+
# mutated anything. An adapter is never handed an empty set.
|
|
997
|
+
targets = production_sources(changed_files, lang)
|
|
998
|
+
if not targets:
|
|
999
|
+
return unsupported("no-production-sources", tool=adapter.name)
|
|
1000
|
+
# `--changed-file` and `--diff` are built by two separate git commands, so
|
|
1001
|
+
# nothing guarantees they describe the same work. If the diff covers none of
|
|
1002
|
+
# the files about to be mutated, every survivor falls outside the
|
|
1003
|
+
# intersection and the stage is stamped PASS having verified nothing.
|
|
1004
|
+
scope = read_diff(diff_path)
|
|
1005
|
+
if scope is None:
|
|
1006
|
+
return unsupported("diff-unavailable", tool=adapter.name)
|
|
1007
|
+
missing = _targets_absent_from_diff(targets, scope.mentioned, worktree)
|
|
1008
|
+
if missing:
|
|
1009
|
+
print(
|
|
1010
|
+
f"mutation-probe: {adapter.name} was handed {len(missing)} changed "
|
|
1011
|
+
f"source(s) the diff never names: {missing}",
|
|
1012
|
+
file=sys.stderr,
|
|
1013
|
+
)
|
|
1014
|
+
return unsupported("diff-incomplete", tool=adapter.name)
|
|
1015
|
+
if not scope.touched:
|
|
1016
|
+
# The diff names every target but adds no line anywhere — a deletion-only
|
|
1017
|
+
# or header-only diff. There is nothing for gate B to verify, which is
|
|
1018
|
+
# not the same as verifying it and finding nothing.
|
|
1019
|
+
return unsupported("diff-adds-no-line", tool=adapter.name)
|
|
1020
|
+
result = adapter.run(targets, diff_path, worktree)
|
|
1021
|
+
if not _states_a_known_verdict(result):
|
|
1022
|
+
return unsupported("adapter-malformed-status", tool=adapter.name)
|
|
1023
|
+
return _apply_waivers(result, waivers)
|
|
1024
|
+
|
|
1025
|
+
|
|
1026
|
+
def _states_a_known_verdict(result: object) -> bool:
|
|
1027
|
+
"""True when `result` speaks the vocabulary the gate downstream understands.
|
|
1028
|
+
|
|
1029
|
+
"Adapters parse, they never judge" is only a rule if something checks it.
|
|
1030
|
+
`validate-run.py` blocks on `FAIL` and folds `unsupported(...)` down to gate
|
|
1031
|
+
A; a third value matches neither and would sail through both. Enforcing it
|
|
1032
|
+
on the interface means each new adapter inherits the check instead of
|
|
1033
|
+
re-earning trust.
|
|
1034
|
+
"""
|
|
1035
|
+
if not isinstance(result, dict):
|
|
1036
|
+
return False
|
|
1037
|
+
status = result.get("status")
|
|
1038
|
+
if not isinstance(status, str):
|
|
1039
|
+
return False
|
|
1040
|
+
if status in ("PASS", "FAIL"):
|
|
1041
|
+
return True
|
|
1042
|
+
if not (status.startswith("unsupported(") and status.endswith(")")):
|
|
1043
|
+
return False
|
|
1044
|
+
# An empty reason is the one thing `unsupported` must never be: "gate B did
|
|
1045
|
+
# not run" with no way to find out why is indistinguishable from a silent skip.
|
|
1046
|
+
return bool(_unsupported_reason(status))
|
|
1047
|
+
|
|
1048
|
+
|
|
1049
|
+
def _unsupported_reason(status: str) -> str:
|
|
1050
|
+
"""The text inside `unsupported(...)`, or `""` when there is none."""
|
|
1051
|
+
return status[len("unsupported(") : -1].strip()
|
|
1052
|
+
|
|
1053
|
+
|
|
1054
|
+
def _probe_langs(
|
|
1055
|
+
changed_files: list[Path], worktree: Path | None
|
|
1056
|
+
) -> dict[str, list[Path]]:
|
|
1057
|
+
"""Group the stage's changed files by the language key each adapter answers to.
|
|
1058
|
+
|
|
1059
|
+
Paths are folded to worktree-relative FIRST. An adapter decides what to
|
|
1060
|
+
exclude from its own path shape — Stryker drops anything under a `test`,
|
|
1061
|
+
`tests` or `spec` directory — so an absolute path drags the checkout's own
|
|
1062
|
+
ancestors into that decision and can exclude real production code whenever
|
|
1063
|
+
the worktree happens to live under such a directory. That failure is silent:
|
|
1064
|
+
the gate simply finds nothing to mutate and reports a vacuous PASS.
|
|
1065
|
+
"""
|
|
1066
|
+
grouped: dict[str, list[Path]] = {}
|
|
1067
|
+
for path in changed_files:
|
|
1068
|
+
relative = Path(_worktree_relative(str(path), worktree))
|
|
1069
|
+
lang = EXT_TO_LANG.get(relative.suffix)
|
|
1070
|
+
if lang is None:
|
|
1071
|
+
continue
|
|
1072
|
+
grouped.setdefault(lang, []).append(relative)
|
|
1073
|
+
return grouped
|
|
1074
|
+
|
|
1075
|
+
|
|
1076
|
+
# --- reason classification SSOT ---------------------------------------------
|
|
1077
|
+
#
|
|
1078
|
+
# Every `unsupported(<reason>)` this module can emit falls into exactly one of
|
|
1079
|
+
# three classes. BOTH consumers read this one definition — `_merge_probes` below
|
|
1080
|
+
# decides whether a sibling language's PASS may stand, and
|
|
1081
|
+
# `validate-run.py::_validate_selfmock` decides whether the run blocks. A second,
|
|
1082
|
+
# parallel table is how those two drift apart.
|
|
1083
|
+
CAPABILITY_GAP = "capability-gap"
|
|
1084
|
+
NOTHING_TO_VERIFY = "nothing-to-verify"
|
|
1085
|
+
INTEGRITY_INSPECTION = "integrity-inspection"
|
|
1086
|
+
|
|
1087
|
+
# Gate B declared this boundary in advance: it never claimed to cover that
|
|
1088
|
+
# language here. Non-blocking, and a sibling's PASS may stand over it — without
|
|
1089
|
+
# that, gate B would be unusable in any polyglot repo and would wedge every repo
|
|
1090
|
+
# with no mutation tooling installed.
|
|
1091
|
+
CAPABILITY_GAP_REASONS = frozenset(
|
|
1092
|
+
{
|
|
1093
|
+
"tool-not-declared",
|
|
1094
|
+
"diff-scope-unavailable",
|
|
1095
|
+
"no-production-sources",
|
|
1096
|
+
# No changed file is in a language any adapter covers — a Go, Ruby or C#
|
|
1097
|
+
# stage. Gate A's trigger is extension-agnostic, so those repos reach
|
|
1098
|
+
# gate B on every run; blocking here would wedge all of them.
|
|
1099
|
+
"no-changed-sources",
|
|
1100
|
+
}
|
|
1101
|
+
)
|
|
1102
|
+
|
|
1103
|
+
# The tool ran to completion and the changed code genuinely offered nothing to
|
|
1104
|
+
# check. That is a property of the code, not a failure of the run, so it is also
|
|
1105
|
+
# non-blocking. `diff-adds-no-line` lives here because deletion-only and
|
|
1106
|
+
# rename-only stages are legitimate and must not wedge.
|
|
1107
|
+
NOTHING_TO_VERIFY_REASONS = frozenset(
|
|
1108
|
+
{
|
|
1109
|
+
"no-mutants-generated",
|
|
1110
|
+
"diff-adds-no-line",
|
|
1111
|
+
}
|
|
1112
|
+
)
|
|
1113
|
+
|
|
1114
|
+
# We tried to inspect and cannot trust the answer. Never overridden by a sibling
|
|
1115
|
+
# PASS, and BLOCKING at the gate: each of these is a fixable fault in the run's
|
|
1116
|
+
# own inputs or output, not a boundary anyone declared.
|
|
1117
|
+
#
|
|
1118
|
+
# `no-conclusive-mutants` belongs here rather than with "nothing to verify":
|
|
1119
|
+
# mutants existed and not one completed a trial, so the same mutant reports
|
|
1120
|
+
# `Survived` when the run finishes and `Pending` when it is cut short. Classing
|
|
1121
|
+
# it as a gap would let the verdict turn on whether the run was interrupted.
|
|
1122
|
+
INTEGRITY_INSPECTION_REASONS = frozenset(
|
|
1123
|
+
{
|
|
1124
|
+
"diff-incomplete",
|
|
1125
|
+
"diff-unavailable",
|
|
1126
|
+
"report-unavailable",
|
|
1127
|
+
"report-unparsed",
|
|
1128
|
+
"adapter-malformed-status",
|
|
1129
|
+
"no-conclusive-mutants",
|
|
1130
|
+
}
|
|
1131
|
+
)
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def _strip_tool_qualifier(reason: str) -> str:
|
|
1135
|
+
"""Drop the `<tool>:` prefix `_merged_reasons` adds, keeping `no-adapter:<lang>`.
|
|
1136
|
+
|
|
1137
|
+
The sidecar records the MERGED reason, so what reaches the gate looks like
|
|
1138
|
+
`stryker:diff-incomplete`. `no-adapter:<lang>` carries its colon as part of
|
|
1139
|
+
the reason itself and is never tool-qualified (its `tool` is `None`), so it
|
|
1140
|
+
is recognised before the split rather than being truncated to a bare lang.
|
|
1141
|
+
"""
|
|
1142
|
+
if reason.startswith("no-adapter:"):
|
|
1143
|
+
return reason
|
|
1144
|
+
_, sep, rest = reason.partition(":")
|
|
1145
|
+
return rest if sep else reason
|
|
1146
|
+
|
|
1147
|
+
|
|
1148
|
+
def _classify_one(reason: str) -> str:
|
|
1149
|
+
bare = _strip_tool_qualifier(reason.strip())
|
|
1150
|
+
if bare.startswith("no-adapter:") or bare in CAPABILITY_GAP_REASONS:
|
|
1151
|
+
return CAPABILITY_GAP
|
|
1152
|
+
if bare in NOTHING_TO_VERIFY_REASONS:
|
|
1153
|
+
return NOTHING_TO_VERIFY
|
|
1154
|
+
# Fail closed: a reason nobody classified is treated as an inspection
|
|
1155
|
+
# failure, so forgetting to classify one blocks rather than passes.
|
|
1156
|
+
return INTEGRITY_INSPECTION
|
|
1157
|
+
|
|
1158
|
+
|
|
1159
|
+
def classify_reason(status: str) -> str:
|
|
1160
|
+
"""Classify an `unsupported(<reason>)` status into one of the three classes.
|
|
1161
|
+
|
|
1162
|
+
A merged status can carry several `; `-joined reasons; the most severe class
|
|
1163
|
+
wins, so one inspection failure among capability gaps still governs.
|
|
1164
|
+
"""
|
|
1165
|
+
reasons = [r for r in _unsupported_reason(status).split(";") if r.strip()]
|
|
1166
|
+
classes = {_classify_one(r) for r in reasons}
|
|
1167
|
+
if not classes or INTEGRITY_INSPECTION in classes:
|
|
1168
|
+
return INTEGRITY_INSPECTION
|
|
1169
|
+
if NOTHING_TO_VERIFY in classes:
|
|
1170
|
+
return NOTHING_TO_VERIFY
|
|
1171
|
+
return CAPABILITY_GAP
|
|
1172
|
+
|
|
1173
|
+
|
|
1174
|
+
def _merge_probes(results: list[ProbeResult]) -> ProbeResult:
|
|
1175
|
+
"""Fold one verdict per language into the single one the sidecar records.
|
|
1176
|
+
|
|
1177
|
+
Any `FAIL` fails the stage — that precedence is absolute.
|
|
1178
|
+
|
|
1179
|
+
Otherwise a single real `PASS` carries the stage, but ONLY over capability
|
|
1180
|
+
gaps. A probed language's clean result is evidence and an uncoverable
|
|
1181
|
+
language is merely silence; a language whose INSPECTION failed is neither,
|
|
1182
|
+
and letting a sibling speak for it is how total coverage inside one language
|
|
1183
|
+
becomes partial coverage across two.
|
|
1184
|
+
|
|
1185
|
+
When nothing was probed at all, every reason is kept: which language went
|
|
1186
|
+
unchecked is the operator's next action.
|
|
1187
|
+
"""
|
|
1188
|
+
statuses = [r["status"] for r in results]
|
|
1189
|
+
survived = [row for r in results for row in r["survived"]]
|
|
1190
|
+
waived = [row for r in results for row in r["waived"]]
|
|
1191
|
+
ran = [r["tool"] for r in results if r["status"] in ("PASS", "FAIL") and r["tool"]]
|
|
1192
|
+
unsupported_results = [
|
|
1193
|
+
r for r in results if str(r["status"]).startswith("unsupported(")
|
|
1194
|
+
]
|
|
1195
|
+
# One classification, shared with the gate: a sibling PASS may stand over a
|
|
1196
|
+
# capability gap or a language with nothing to verify, never over an
|
|
1197
|
+
# inspection failure.
|
|
1198
|
+
blocking = [
|
|
1199
|
+
r
|
|
1200
|
+
for r in unsupported_results
|
|
1201
|
+
if classify_reason(str(r["status"])) == INTEGRITY_INSPECTION
|
|
1202
|
+
]
|
|
1203
|
+
if "FAIL" in statuses:
|
|
1204
|
+
status = "FAIL"
|
|
1205
|
+
elif blocking:
|
|
1206
|
+
# Reported from the blocking reasons alone: those are what the operator
|
|
1207
|
+
# has to fix, and folding in the capability gaps would bury them.
|
|
1208
|
+
status = f"unsupported({'; '.join(_merged_reasons(blocking))})"
|
|
1209
|
+
elif "PASS" in statuses:
|
|
1210
|
+
status = "PASS"
|
|
1211
|
+
else:
|
|
1212
|
+
status = f"unsupported({'; '.join(_merged_reasons(unsupported_results))})"
|
|
1213
|
+
return {
|
|
1214
|
+
"status": status,
|
|
1215
|
+
"tool": ",".join(sorted(set(ran))) or None,
|
|
1216
|
+
"survived": survived,
|
|
1217
|
+
# Summed so the sidecar still shows how many were found even when a
|
|
1218
|
+
# language's list was capped; `survived` alone would understate it.
|
|
1219
|
+
"survivedTotal": sum(
|
|
1220
|
+
r["survivedTotal"]
|
|
1221
|
+
if isinstance(r.get("survivedTotal"), int)
|
|
1222
|
+
else len(r["survived"])
|
|
1223
|
+
for r in results
|
|
1224
|
+
),
|
|
1225
|
+
"waived": waived,
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def _merged_reasons(results: list[ProbeResult]) -> list[str]:
|
|
1230
|
+
"""Each language's reason, tool-qualified where the reason alone is ambiguous."""
|
|
1231
|
+
reasons: list[str] = []
|
|
1232
|
+
for r in results:
|
|
1233
|
+
reason = _unsupported_reason(str(r["status"]))
|
|
1234
|
+
# `no-adapter:<lang>` already names its language; `tool-not-declared` and
|
|
1235
|
+
# friends do not, so the tool that produced them is prepended.
|
|
1236
|
+
if r["tool"]:
|
|
1237
|
+
reason = f"{r['tool']}:{reason}"
|
|
1238
|
+
if reason not in reasons:
|
|
1239
|
+
reasons.append(reason)
|
|
1240
|
+
return reasons
|
|
1241
|
+
|
|
1242
|
+
|
|
1243
|
+
def probe_changed_files(
|
|
1244
|
+
changed_files: list[Path],
|
|
1245
|
+
diff_path: Path | None,
|
|
1246
|
+
worktree: Path | None,
|
|
1247
|
+
waivers=(),
|
|
1248
|
+
) -> ProbeResult:
|
|
1249
|
+
"""Probe every language the stage touched and merge the verdicts into one.
|
|
1250
|
+
|
|
1251
|
+
This is the entry point the detector writes its sidecar from. `changed_files`
|
|
1252
|
+
is the stage's WHOLE changed set — each adapter selects its own production
|
|
1253
|
+
sources out of it (see `run_probe`).
|
|
1254
|
+
"""
|
|
1255
|
+
grouped = _probe_langs(changed_files, worktree)
|
|
1256
|
+
if not grouped:
|
|
1257
|
+
return unsupported("no-changed-sources")
|
|
1258
|
+
return _merge_probes(
|
|
1259
|
+
[
|
|
1260
|
+
run_probe(lang, files, diff_path, worktree, waivers)
|
|
1261
|
+
for lang, files in sorted(grouped.items())
|
|
1262
|
+
]
|
|
1263
|
+
)
|