okstra 0.143.0 → 0.145.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +4 -1
  2. package/docs/architecture.md +18 -2
  3. package/docs/cli.md +39 -2
  4. package/docs/project-structure-overview.md +19 -6
  5. package/package.json +1 -1
  6. package/runtime/BUILD.json +2 -2
  7. package/runtime/prompts/coding-preflight/overview.md +1 -1
  8. package/runtime/prompts/lead/convergence.md +11 -3
  9. package/runtime/prompts/lead/okstra-lead-contract.md +7 -1
  10. package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
  11. package/runtime/prompts/profiles/_common-contract.md +1 -1
  12. package/runtime/prompts/profiles/_implementation-verifier.md +48 -2
  13. package/runtime/prompts/profiles/change-impact-analysis.md +24 -0
  14. package/runtime/prompts/profiles/feature-analysis.md +24 -0
  15. package/runtime/prompts/profiles/forbidden-actions.json +18 -0
  16. package/runtime/prompts/profiles/project-analysis.md +24 -0
  17. package/runtime/prompts/wizard/prompts.ko.json +44 -1
  18. package/runtime/python/okstra_ctl/analysis_inputs.py +369 -0
  19. package/runtime/python/okstra_ctl/clarification_items.py +74 -1
  20. package/runtime/python/okstra_ctl/mutation_probe.py +1263 -0
  21. package/runtime/python/okstra_ctl/render.py +77 -4
  22. package/runtime/python/okstra_ctl/render_final_report.py +13 -4
  23. package/runtime/python/okstra_ctl/report_views.py +134 -3
  24. package/runtime/python/okstra_ctl/run.py +118 -0
  25. package/runtime/python/okstra_ctl/run_context.py +34 -2
  26. package/runtime/python/okstra_ctl/schema_excerpt.py +12 -4
  27. package/runtime/python/okstra_ctl/self_mock_signals.py +183 -0
  28. package/runtime/python/okstra_ctl/user_response.py +309 -3
  29. package/runtime/python/okstra_ctl/wizard.py +545 -32
  30. package/runtime/python/okstra_ctl/worker_prompt_policy.py +3 -0
  31. package/runtime/python/okstra_ctl/workflow.py +22 -0
  32. package/runtime/schemas/final-report-v1.0.schema.json +849 -3
  33. package/runtime/skills/okstra-run/SKILL.md +13 -1
  34. package/runtime/templates/reports/change-impact-analysis-input.template.md +58 -0
  35. package/runtime/templates/reports/feature-analysis-input.template.md +59 -0
  36. package/runtime/templates/reports/final-report.template.md +220 -0
  37. package/runtime/templates/reports/i18n/en.json +8 -0
  38. package/runtime/templates/reports/i18n/ko.json +8 -0
  39. package/runtime/templates/reports/project-analysis-input.template.md +58 -0
  40. package/runtime/templates/reports/report.js +84 -5
  41. package/runtime/templates/reports/user-response.template.md +19 -1
  42. package/runtime/validators/detect_self_mock.py +220 -0
  43. package/runtime/validators/validate-report-views.py +61 -7
  44. package/runtime/validators/validate-run.py +518 -0
  45. package/runtime/validators/validate_analysis_report.py +864 -0
  46. package/src/commands/execute/render-bundle.mjs +3 -0
@@ -0,0 +1,1263 @@
1
+ """Tool-agnostic front door for gate B of the self-mock gate — mutation probing.
2
+
3
+ Gate A (``validators/detect_self_mock.py``) catches a test that stubs its own
4
+ subject by syntax. Gate B catches the ones syntax cannot see: if a mutant of a
5
+ changed line goes UNDETECTED by the stage's own suite, no test in it constrains
6
+ that line. Undetected covers two distinct failures, and both count — the test
7
+ ran the code and asserted nothing about it, or the test never reached the code
8
+ at all. The second is what a self-mocked test actually looks like from the
9
+ outside: stubbing the subject's own method means the real production line never
10
+ executes, so it is the signal this gate most needs to keep.
11
+
12
+ Every external mutation tool (Stryker, cargo-mutants, PIT) answers a different
13
+ CLI and a different report format, so each is wrapped in an `Adapter` and this
14
+ module owns everything that must NOT differ between them: which language has an
15
+ adapter at all, whether that adapter's tool is actually available, which mutants
16
+ are in scope, and how survivors become a verdict. Adapters parse; they do not
17
+ judge.
18
+
19
+ Two rules are load-bearing and both are about refusing a quiet pass:
20
+
21
+ - Nothing degrades silently. A language with no adapter, an adapter whose tool
22
+ is not declared, and a missing diff each answer `unsupported(<reason>)` naming
23
+ the cause. `validate-run.py` folds `unsupported(...)` down to gate A only, so
24
+ a silent "PASS" here would be indistinguishable from a real one.
25
+ - Only mutants covering a line the diff added or modified count. A survivor
26
+ elsewhere is pre-existing debt that this stage neither introduced nor is
27
+ blocked by.
28
+ - Having nothing to mutate is never a PASS. Production-source selection and the
29
+ empty-set refusal live in `run_probe`, ahead of every adapter, so "the tool
30
+ never ran" cannot be reported in the same words as "the tool ran and found
31
+ nothing undetected". Each adapter would otherwise have to re-earn that
32
+ distinction, and each one is a chance to lose it.
33
+ - A sibling language's PASS may stand over a CAPABILITY GAP, never over an
34
+ inspection failure. `--diff` is one shared file, so "incomplete for Rust" also
35
+ indicts the input that scoped the passing TypeScript verdict. The three reason
36
+ classes are defined once here (`classify_reason`) and read by both the merge
37
+ and `validate-run.py`'s gate.
38
+ - The two inputs must AGREE, completely. `changed_files` and the diff arrive
39
+ from separate git commands, so `run_probe` refuses to run unless the diff
40
+ names EVERY changed source — one uncovered target is enough to hide a real
41
+ survivor, since its lines never enter the intersection. A diff that names them
42
+ all but adds no line anywhere is refused too: there is nothing to verify, which
43
+ is not the same as verifying and finding nothing.
44
+ - An undetected mutant must be PLACEABLE, in BOTH halves of its `(file, line)`
45
+ key. Each adapter screens its own undetected rows — the line through
46
+ `_coerce_line`, the same test `evaluate` uses, and the file by requiring a
47
+ non-empty string — so a survivor that cannot be located fails its report
48
+ instead of being dropped on the way to the intersection. `evaluate`'s own check then stands as defense
49
+ for a future adapter that forgets, not as the live path for a real survivor.
50
+ - Only a CONCLUSIVE trial is evidence. A mutant that failed to compile, was
51
+ skipped, or never finished says nothing about the tests, so those outcomes are
52
+ recognised but not counted; a report made entirely of them is
53
+ `unsupported(no-conclusive-mutants)`. Every outcome vocabulary is matched as an
54
+ ALLOWLIST — an unrecognised word fails the report rather than being read as
55
+ "a test caught it".
56
+
57
+ "Adapters parse, they never judge" is enforced, not merely stated: `run_probe`
58
+ refuses any result whose `status` is outside the `PASS`/`FAIL`/`unsupported(...)`
59
+ vocabulary, because a fourth value matches neither of `validate-run.py`'s
60
+ branches and would pass by falling between them.
61
+
62
+ `ADAPTERS` is keyed by the `self_mock_signals.EXT_TO_LANG` vocabulary — the same
63
+ lang names gate A resolves a changed file to — so both gates answer to one set
64
+ of language keys: `ts_js` (Stryker), `rust` (cargo-mutants), and `java`/`kotlin`
65
+ (PIT, whose diff scoping is not wired up yet — see `PitAdapter`).
66
+ """
67
+ from __future__ import annotations
68
+
69
+ import json
70
+ import re
71
+ import shutil
72
+ import subprocess
73
+ import sys
74
+ from pathlib import Path
75
+ from typing import NamedTuple, Protocol
76
+
77
+ from .self_mock_signals import (
78
+ EXT_TO_LANG,
79
+ partition_waived_entries,
80
+ selfmock_path_key,
81
+ )
82
+
83
+ # `survived` is read by a human in the sidecar, so it is trimmed. The verdict and
84
+ # the logged total are taken before the trim — the cap shortens the report, never
85
+ # the finding.
86
+ SURVIVOR_CAP = 50
87
+
88
+ ProbeResult = dict[str, object]
89
+
90
+ _HUNK = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@")
91
+ _DIFF_GIT_RE = re.compile(r"^diff --git a/(.+) b/(.+)$")
92
+
93
+
94
+ class Adapter(Protocol):
95
+ """One external mutation tool, reduced to what the probe needs from it.
96
+
97
+ An adapter parses; it does not judge and it does not select. `run_probe`
98
+ picks the production sources and refuses an empty set before any adapter is
99
+ reached, so `run` is only ever called with at least one real target. That
100
+ ordering is the contract: "selected nothing, therefore PASS" is the vacuous
101
+ pass this gate exists to prevent, and every adapter would otherwise have to
102
+ remember not to reinvent it.
103
+ """
104
+
105
+ name: str
106
+
107
+ def is_declared(self, worktree: Path | None) -> bool:
108
+ """True when this tool is installed/configured in `worktree`.
109
+
110
+ Must answer from files alone — no subprocess, no network. `run_probe`
111
+ consults it before `run`, so shelling out here would be the very failure
112
+ it guards against.
113
+ """
114
+
115
+ def run(
116
+ self,
117
+ targets: list[Path],
118
+ diff_path: Path | None,
119
+ worktree: Path | None,
120
+ ) -> ProbeResult:
121
+ """Mutate `targets` (already selected, never empty) and report via `evaluate`.
122
+
123
+ Anything that stops the tool producing a usable report — it was never
124
+ installed, it died, the report is unreadable, its scope cannot be
125
+ narrowed to the diff — is `unsupported(<reason>)`, never `PASS`.
126
+ """
127
+
128
+
129
+ STRYKER_REPORT_PATH = Path("reports/mutation/mutation.json")
130
+
131
+ # Stryker's `--mutate` takes PRODUCTION sources. Handing it a spec file mutates
132
+ # the test instead of the code, and handing it nothing at all makes it fall back
133
+ # to mutating the whole project. These mirror Stryker's own default test
134
+ # excludes; matching is case-insensitive so `Foo.Spec.ts` is caught too.
135
+ _TEST_NAME_MARKERS = (".spec.", ".test.")
136
+ _TEST_DIR_SEGMENTS = ("test", "tests", "spec", "__tests__")
137
+
138
+
139
+ def _is_test_source(path: Path) -> bool:
140
+ """True for a file a mutation tool must not mutate.
141
+
142
+ Mutating test code produces mutants that sit on the stage's changed lines
143
+ and fail it for nothing. A gate that raises false alarms is one people learn
144
+ to route around, which costs more than the mutants it would have caught.
145
+ """
146
+ name = path.name.lower()
147
+ if any(marker in name for marker in _TEST_NAME_MARKERS):
148
+ return True
149
+ # Only directory components — a production file may legitimately be named
150
+ # `spec.ts`, and `parts[:-1]` keeps the filename out of the comparison.
151
+ return any(part.lower() in _TEST_DIR_SEGMENTS for part in path.parts[:-1])
152
+
153
+
154
+ def production_sources(changed_files: list[Path], lang: str) -> list[Path]:
155
+ """The changed files of `lang` that a mutation tool may target."""
156
+ return [
157
+ p
158
+ for p in changed_files
159
+ if EXT_TO_LANG.get(p.suffix) == lang and not _is_test_source(p)
160
+ ]
161
+
162
+
163
+ def _run_stryker_cli(targets: list[Path], worktree: Path | None) -> None:
164
+ """Invoke the real Stryker CLI, scoped to `targets`, writing a json report.
165
+
166
+ `--no-install` keeps the promise `is_declared` makes: plain `npx stryker`
167
+ fetches the package from the registry when it is not installed locally, so
168
+ an undeclared worktree would quietly go to the network instead of reporting
169
+ `unsupported(...)`. Without an install this now fails fast and leaves no
170
+ report, which surfaces as `unsupported(report-unavailable)`.
171
+
172
+ A surviving mutant makes Stryker exit non-zero, which is a normal outcome
173
+ here rather than an error — the report is what carries the verdict, so the
174
+ exit status is deliberately not checked.
175
+ """
176
+ subprocess.run(
177
+ [
178
+ "npx",
179
+ "--no-install",
180
+ "stryker",
181
+ "run",
182
+ "--reporters",
183
+ "json",
184
+ "--mutate",
185
+ ",".join(str(t) for t in targets),
186
+ ],
187
+ cwd=str(worktree) if worktree is not None else None,
188
+ check=False,
189
+ )
190
+
191
+
192
+ # Stryker's own model is `Undetected = Survived + NoCoverage`, and both accuse
193
+ # the tests: `Survived` means the code ran and nothing asserted on it,
194
+ # `NoCoverage` means no test reached the code at all. NoCoverage is in fact the
195
+ # purest self-mock fingerprint — stubbing the subject's own method stops the
196
+ # real production line from ever executing. Every other status (`Killed`,
197
+ # `Timeout`, `RuntimeError`, `CompileError`, `Ignored`) either means a test
198
+ # caught the mutant or that no usable trial happened, so none of them counts.
199
+ #
200
+ # Deliberately local to this adapter: cargo-mutants and PIT report their
201
+ # outcomes in different vocabularies, so this must not be hoisted into the
202
+ # shared layer.
203
+ _STRYKER_UNDETECTED = ("Survived", "NoCoverage")
204
+
205
+
206
+ # The FULL vocabulary Stryker emits. Matching against an allowlist rather than
207
+ # "anything that is not undetected" is what stops an unrecognised word — a
208
+ # renamed status, a newer Stryker, a truncated report — from being read as "a
209
+ # test caught it" and quietly clearing the run.
210
+ _STRYKER_STATUSES = (
211
+ "Killed",
212
+ "Survived",
213
+ "NoCoverage",
214
+ "Timeout",
215
+ "RuntimeError",
216
+ "CompileError",
217
+ "Ignored",
218
+ "Pending",
219
+ )
220
+
221
+ # The subset that represents a trial that actually finished and therefore says
222
+ # something about the tests. `CompileError` and `RuntimeError` mean the mutant
223
+ # never produced a usable trial, `Ignored` means it was skipped, and `Pending`
224
+ # means the run was cut short before it got there.
225
+ _STRYKER_CONCLUSIVE = ("Killed", "Survived", "NoCoverage", "Timeout")
226
+
227
+
228
+ def _stryker_line(mutant: dict) -> int | None:
229
+ """The new-side line of a Stryker mutant, or `None` if it cannot be read.
230
+
231
+ Guards each hop rather than trusting the shape: a `location` or `start` that
232
+ is not an object would otherwise raise `AttributeError` out of `run_probe`.
233
+ That crash is worse than it looks — the probe runs before the sidecar is
234
+ written, so it would also destroy gate A's static result and leave the run
235
+ with no sidecar at all. Every malformed shape converges on `report-unparsed`
236
+ instead.
237
+ """
238
+ location = mutant.get("location")
239
+ if not isinstance(location, dict):
240
+ return None
241
+ start = location.get("start")
242
+ if not isinstance(start, dict):
243
+ return None
244
+ return _coerce_line(start.get("line"))
245
+
246
+
247
+ def _survivors_from_report(
248
+ report: dict, worktree: Path | None
249
+ ) -> ParsedReport | None:
250
+ """Flatten a Stryker json report into rows plus the two trial counts.
251
+
252
+ `None` means the report could not be read at all, which is not the same
253
+ answer as "no undetected mutants" and must never become one.
254
+
255
+ `status` is carried through because the two undetected outcomes need
256
+ different fixes — write a real assertion (`Survived`) versus cover the code
257
+ at all (`NoCoverage`) — and the sidecar reader cannot tell them apart once
258
+ they are merged. `evaluate` treats the rows opaquely, so the extra key costs
259
+ the shared layer nothing.
260
+ """
261
+ files = report.get("files")
262
+ if files is None:
263
+ files = {}
264
+ if not isinstance(files, dict):
265
+ return None
266
+ rows: list[dict] = []
267
+ conclusive = 0
268
+ observed = 0
269
+ for name, entry in files.items():
270
+ if not isinstance(entry, dict):
271
+ return None
272
+ mutants = entry.get("mutants")
273
+ if mutants is None:
274
+ mutants = []
275
+ if not isinstance(mutants, list):
276
+ return None
277
+ for mutant in mutants:
278
+ if not isinstance(mutant, dict):
279
+ return None
280
+ status = mutant.get("status")
281
+ if status not in _STRYKER_STATUSES:
282
+ return None
283
+ observed += 1
284
+ if status in _STRYKER_CONCLUSIVE:
285
+ conclusive += 1
286
+ if status not in _STRYKER_UNDETECTED:
287
+ # A DETECTED mutant's position is never matched against the diff,
288
+ # so a malformed location cannot change the verdict and is not
289
+ # worth failing a run over. Only undetected ones are read below.
290
+ continue
291
+ line = _stryker_line(mutant)
292
+ if line is None:
293
+ # An undetected mutant we cannot PLACE is unverifiable: `evaluate`
294
+ # could only drop it, turning the most serious finding a report
295
+ # carries into a PASS with a note on stderr.
296
+ return None
297
+ rows.append(
298
+ {
299
+ "file": _worktree_relative(name, worktree),
300
+ "line": line,
301
+ "mutant": mutant.get("mutatorName"),
302
+ "status": status,
303
+ }
304
+ )
305
+ return ParsedReport(rows, conclusive, observed)
306
+
307
+
308
+ class ParsedReport(NamedTuple):
309
+ """One tool's report, reduced to what the shared guards need from it.
310
+
311
+ `conclusive` is deliberately not `len(survivors)` nor `observed`: a mutant
312
+ that failed to compile, was skipped, or never finished says nothing about
313
+ the tests. Counting those as trials is how a run that proved nothing ends up
314
+ reported as clean.
315
+ """
316
+
317
+ survivors: list[dict]
318
+ conclusive: int
319
+ observed: int
320
+
321
+
322
+ def _verdict_from_parsed(
323
+ parsed: ParsedReport | None,
324
+ diff_path: Path | None,
325
+ worktree: Path | None,
326
+ tool: str,
327
+ ) -> ProbeResult:
328
+ """The guards every adapter needs once its own report shape is parsed.
329
+
330
+ Report-shape parsing stays local to each adapter — the formats have nothing
331
+ in common — but what an unreadable report, a mutant-free report and a report
332
+ of nothing but inconclusive trials MEAN is identical across tools, and all
333
+ three are answers only `unsupported` can carry. Keeping the decision here
334
+ means a new adapter inherits it instead of re-deriving it.
335
+ """
336
+ if parsed is None:
337
+ return unsupported("report-unparsed", tool=tool)
338
+ if parsed.observed == 0:
339
+ # The tool ran but produced nothing to detect.
340
+ return unsupported("no-mutants-generated", tool=tool)
341
+ if parsed.conclusive == 0:
342
+ # Mutants existed but not one of them completed a real trial — an
343
+ # all-unviable build, or a truncated run. Never evidence of a good suite.
344
+ return unsupported("no-conclusive-mutants", tool=tool)
345
+ return evaluate(parsed.survivors, diff_path, worktree, tool=tool)
346
+
347
+
348
+ class StrykerAdapter:
349
+ """Gate B for TypeScript / JavaScript, wrapping the Stryker CLI.
350
+
351
+ The CLI call is injected (`runner`) rather than hard-coded, so a test can
352
+ substitute the external tool — the genuine collaborator — while the
353
+ adapter's own selection, parsing and path handling still run for real.
354
+ """
355
+
356
+ name = "stryker"
357
+
358
+ def __init__(self, runner=_run_stryker_cli, report_path=STRYKER_REPORT_PATH):
359
+ self._runner = runner
360
+ self._report_path = Path(report_path)
361
+
362
+ def is_declared(self, worktree: Path | None) -> bool:
363
+ """True when Stryker is actually INSTALLED here. No subprocess, no network.
364
+
365
+ A `package.json` entry is not enough. `node_modules` is gitignored, so a
366
+ fresh stage worktree routinely declares `@stryker-mutator/core` with no
367
+ binary present; answering True there sends the run into
368
+ `npx --no-install`, which fails, writes no report, and BLOCKS the stage on
369
+ `report-unavailable` — failing a run over a tool nobody installed.
370
+ Requiring the binary makes that case `tool-not-declared`, a capability
371
+ gap, which is non-blocking. Symmetric with `CargoMutantsAdapter`, which
372
+ already requires the executable rather than the manifest entry.
373
+
374
+ `run_probe` calls this before `run`, so it must answer from files alone —
375
+ shelling out to check would be the very failure it is guarding against.
376
+ """
377
+ if worktree is None:
378
+ return False
379
+ return (Path(worktree) / "node_modules" / ".bin" / "stryker").exists()
380
+
381
+ def run(
382
+ self,
383
+ targets: list[Path],
384
+ diff_path: Path | None,
385
+ worktree: Path | None,
386
+ ) -> ProbeResult:
387
+ # Drop any earlier report FIRST. Single-stage final-verification reuses
388
+ # the implementation stage worktree, so a previous run's mutation.json is
389
+ # genuinely on disk; if Stryker then dies before writing, reading it back
390
+ # would report that older run's verdict for code it never saw.
391
+ self._report_file(worktree).unlink(missing_ok=True)
392
+ self._runner(targets, worktree)
393
+ report = self._read_report(worktree)
394
+ if report is None:
395
+ return unsupported("report-unavailable", tool=self.name)
396
+ return _verdict_from_parsed(
397
+ _survivors_from_report(report, worktree), diff_path, worktree, self.name
398
+ )
399
+
400
+ def _report_file(self, worktree: Path | None) -> Path:
401
+ if worktree is None:
402
+ return self._report_path
403
+ return Path(worktree) / self._report_path
404
+
405
+ def _read_report(self, worktree: Path | None) -> dict | None:
406
+ """The parsed json report, or `None` when the run left none behind."""
407
+ path = self._report_file(worktree)
408
+ if not path.is_file():
409
+ return None
410
+ try:
411
+ return json.loads(path.read_text(encoding="utf-8"))
412
+ except (OSError, json.JSONDecodeError):
413
+ return None
414
+
415
+
416
+ CARGO_OUTCOMES_PATH = Path("mutants.out/outcomes.json")
417
+
418
+ # cargo-mutants' own vocabulary: `caught` (a test failed, good), `missed` (no
419
+ # test failed), `unviable` (the mutant did not compile) and `timeout`. Only
420
+ # `missed` accuses the tests. Unlike Stryker and PIT there is no separate
421
+ # "not covered" word — cargo-mutants folds uncovered code into `missed`, so this
422
+ # tuple is one entry rather than two.
423
+ #
424
+ # Adapter-local on purpose: Stryker says `Survived`/`NoCoverage`, PIT says
425
+ # `SURVIVED`/`NO_COVERAGE`. Hoisting any of them into the shared layer would
426
+ # make one tool's vocabulary silently govern another's report.
427
+ _CARGO_UNDETECTED = ("missed",)
428
+
429
+ # The FULL vocabulary, matched as an allowlist. An unrecognised word is exactly
430
+ # the unverified case this parser must fail closed on: treating it as "not
431
+ # undetected" would let a renamed or suffixed outcome clear the whole run.
432
+ #
433
+ # UNVERIFIED, and the first thing a real cargo-mutants run must settle: if
434
+ # `outcomes.json` also carries a baseline scenario (a `summary` such as
435
+ # "Success"), every run here becomes `unsupported(report-unparsed)` and this
436
+ # adapter is permanently inert — and `len(entries)` would count that baseline as
437
+ # a mutant besides. Confirm the vocabulary AND whether a baseline row exists,
438
+ # then fix the allowlist and the `observed` count together. Do not add the word
439
+ # on speculation; a wrong guess here is exactly what the allowlist exists to
440
+ # catch.
441
+ _CARGO_OUTCOMES = ("caught", "missed", "unviable", "timeout")
442
+
443
+ # `unviable` means the mutant did not compile, so the tests never ran against
444
+ # it. Only the rest represent a finished trial.
445
+ _CARGO_CONCLUSIVE = ("caught", "missed", "timeout")
446
+
447
+
448
+ def _run_cargo_mutants_cli(diff_path: Path, worktree: Path | None) -> None:
449
+ """Invoke cargo-mutants scoped to `diff_path`, writing `mutants.out/`.
450
+
451
+ `--in-diff` restricts testing to mutants overlapping the diff's changed
452
+ regions; the file is expected to carry `b/`-prefixed names, which is exactly
453
+ what the `git diff` output the verifier writes contains. `--no-shuffle`
454
+ keeps the report order stable between runs.
455
+
456
+ A missed mutant makes cargo-mutants exit non-zero, which is the normal
457
+ outcome here rather than an error — the report carries the verdict.
458
+ """
459
+ subprocess.run(
460
+ ["cargo", "mutants", "--no-shuffle", "--in-diff", str(diff_path)],
461
+ cwd=str(worktree) if worktree is not None else None,
462
+ check=False,
463
+ )
464
+
465
+
466
+ def _cargo_mutant_location(entry: dict) -> tuple[object, object, object]:
467
+ """`(file, line, description)` for one outcomes.json entry.
468
+
469
+ UNVERIFIED SHAPE. The reachable cargo-mutants documentation (mutants.rs)
470
+ pins the outcome words and that `mutants.out/outcomes.json` carries the
471
+ results, but not the key names inside it. This reads the nesting the tool is
472
+ believed to use, and every caller treats an unreadable entry as a reason to
473
+ fail the whole report rather than to skip a row — so if this guess is wrong
474
+ the run reports `unsupported(report-unparsed)` instead of a PASS bought with
475
+ our own parsing error. Confirm against a real run in Task 11.
476
+ """
477
+ scenario = entry.get("scenario")
478
+ mutant = scenario.get("Mutant") if isinstance(scenario, dict) else None
479
+ if not isinstance(mutant, dict):
480
+ return None, None, None
481
+ return (
482
+ mutant.get("file"),
483
+ mutant.get("line"),
484
+ mutant.get("description") or mutant.get("function"),
485
+ )
486
+
487
+
488
+ def _survivors_from_cargo_outcomes(
489
+ report: dict, worktree: Path | None
490
+ ) -> ParsedReport | None:
491
+ """Undetected rows + trial counts, or `None` if the report cannot be read.
492
+
493
+ `None` is deliberately distinct from an empty result: "no missed mutants" is
494
+ evidence, "this report is not the shape we can read" is not, and only the
495
+ first may become a PASS.
496
+ """
497
+ entries = report.get("outcomes")
498
+ if not isinstance(entries, list):
499
+ return None
500
+ rows: list[dict] = []
501
+ conclusive = 0
502
+ for entry in entries:
503
+ if not isinstance(entry, dict):
504
+ return None
505
+ summary = entry.get("summary")
506
+ if not isinstance(summary, str):
507
+ return None
508
+ word = summary.lower()
509
+ if word not in _CARGO_OUTCOMES:
510
+ return None
511
+ if word in _CARGO_CONCLUSIVE:
512
+ conclusive += 1
513
+ if word not in _CARGO_UNDETECTED:
514
+ continue
515
+ file, raw_line, describe = _cargo_mutant_location(entry)
516
+ line = _coerce_line(raw_line)
517
+ # Both halves of the placement key must actually place. A non-string
518
+ # `file` would survive `str()` into a key no diff can ever match — and
519
+ # unlike an unplaceable line, that one is dropped without even a note on
520
+ # stderr. The report shape is unverified, so an object here is plausible.
521
+ if not isinstance(file, str) or not file or line is None:
522
+ return None
523
+ rows.append(
524
+ {
525
+ "file": _worktree_relative(file, worktree),
526
+ "line": line,
527
+ "mutant": describe,
528
+ "status": summary,
529
+ }
530
+ )
531
+ return ParsedReport(rows, conclusive, len(entries))
532
+
533
+
534
+ class CargoMutantsAdapter:
535
+ """Gate B for Rust, wrapping the cargo-mutants CLI."""
536
+
537
+ name = "cargo-mutants"
538
+
539
+ def __init__(self, runner=_run_cargo_mutants_cli, report_path=CARGO_OUTCOMES_PATH):
540
+ self._runner = runner
541
+ self._report_path = Path(report_path)
542
+
543
+ def is_declared(self, worktree: Path | None) -> bool:
544
+ """A crate to mutate AND the subcommand installed. No subprocess.
545
+
546
+ `shutil.which` only stats candidate paths, so asking whether the binary
547
+ exists costs nothing and starts nothing.
548
+ """
549
+ if worktree is None:
550
+ return False
551
+ if not (Path(worktree) / "Cargo.toml").is_file():
552
+ return False
553
+ return shutil.which("cargo-mutants") is not None
554
+
555
+ def run(
556
+ self,
557
+ targets: list[Path],
558
+ diff_path: Path | None,
559
+ worktree: Path | None,
560
+ ) -> ProbeResult:
561
+ # `targets` is deliberately unused: the shared selection in `run_probe`
562
+ # decides WHETHER there is Rust production code worth running on, while
563
+ # `--in-diff` decides WHAT gets mutated. Passing a file list as well
564
+ # would give cargo-mutants a second, redundant scope to disagree with.
565
+ #
566
+ # Checked before running: `--in-diff` is the only thing keeping this to
567
+ # the stage's own changes, and without it cargo-mutants would mutate the
568
+ # whole crate — slow, and full of findings this stage never caused.
569
+ if diff_path is None or not Path(diff_path).is_file():
570
+ return unsupported("diff-unavailable", tool=self.name)
571
+ self._report_file(worktree).unlink(missing_ok=True)
572
+ self._runner(diff_path, worktree)
573
+ report = self._read_report(worktree)
574
+ if report is None:
575
+ return unsupported("report-unavailable", tool=self.name)
576
+ return _verdict_from_parsed(
577
+ _survivors_from_cargo_outcomes(report, worktree),
578
+ diff_path,
579
+ worktree,
580
+ self.name,
581
+ )
582
+
583
+ def _report_file(self, worktree: Path | None) -> Path:
584
+ if worktree is None:
585
+ return self._report_path
586
+ return Path(worktree) / self._report_path
587
+
588
+ def _read_report(self, worktree: Path | None) -> dict | None:
589
+ path = self._report_file(worktree)
590
+ if not path.is_file():
591
+ return None
592
+ try:
593
+ data = json.loads(path.read_text(encoding="utf-8"))
594
+ except (OSError, json.JSONDecodeError):
595
+ return None
596
+ return data if isinstance(data, dict) else None
597
+
598
+
599
+ _PIT_BUILD_FILES = ("pom.xml", "build.gradle", "build.gradle.kts")
600
+
601
+
602
+ class PitAdapter:
603
+ """Gate B for Java / Kotlin — currently honest about not being able to run.
604
+
605
+ PIT's own statuses are SURVIVED, KILLED, NO_COVERAGE, TIMED_OUT, NON_VIABLE,
606
+ MEMORY_ERROR and RUN_ERROR, with undetected = SURVIVED + NO_COVERAGE, the
607
+ same two-failure shape Stryker has. Parsing that is the easy part.
608
+
609
+ Scoping is what stops it. PIT is not incapable of SCM scoping — it can
610
+ restrict analysis to files changed in source control — but two concrete
611
+ things block wiring it up here:
612
+
613
+ 1. That scoping is the `scmMutationCoverage` **Maven goal**. There is no
614
+ equivalent among the build files `is_declared` accepts: a Gradle project
615
+ (`build.gradle` / `build.gradle.kts`) has no such goal to invoke, so the
616
+ adapter cannot offer one story for the languages it is registered under.
617
+ 2. Joining PIT's report back to the diff needs a mapping from its
618
+ class-oriented output to repo paths — an unverified report-path↔FQCN
619
+ derivation that depends on each project's source-root layout. Guessing it
620
+ is how a survivor silently misses the changed-line intersection.
621
+
622
+ Running unscoped `mutationCoverage` instead would mutate the whole module:
623
+ slow enough that a stage would time out, and it would report mutants on
624
+ lines this stage never touched. So this adapter reports what is true — gate
625
+ B is not wired for the JVM yet — and lets gate A carry the run. `is_declared`
626
+ still distinguishes "PIT is not configured" from "PIT is configured but we
627
+ cannot scope it", which is the difference an operator acts on.
628
+
629
+ Wiring this up later means: a Maven-only path invoking `scmMutationCoverage`,
630
+ the XML `outputFormats` report, and a verified path mapping — none of which
631
+ should be guessed.
632
+ """
633
+
634
+ name = "pit"
635
+
636
+ def is_declared(self, worktree: Path | None) -> bool:
637
+ """True when a build file in this worktree configures PIT."""
638
+ if worktree is None:
639
+ return False
640
+ root = Path(worktree)
641
+ for build_file in _PIT_BUILD_FILES:
642
+ path = root / build_file
643
+ if not path.is_file():
644
+ continue
645
+ try:
646
+ text = path.read_text(encoding="utf-8", errors="replace")
647
+ except OSError:
648
+ continue
649
+ if "pitest" in text.lower():
650
+ return True
651
+ return False
652
+
653
+ def run(
654
+ self,
655
+ targets: list[Path],
656
+ diff_path: Path | None,
657
+ worktree: Path | None,
658
+ ) -> ProbeResult:
659
+ return unsupported("diff-scope-unavailable", tool=self.name)
660
+
661
+
662
+ # Keyed by the `self_mock_signals.EXT_TO_LANG` vocabulary, which folds `.ts`,
663
+ # `.tsx`, `.js`, `.jsx` and `.mjs` into the single lang `ts_js`. A language
664
+ # absent here answers `unsupported(no-adapter:<lang>)`, which is the honest report.
665
+ _PIT = PitAdapter()
666
+ ADAPTERS: dict[str, Adapter] = {
667
+ "ts_js": StrykerAdapter(),
668
+ "rust": CargoMutantsAdapter(),
669
+ "java": _PIT,
670
+ "kotlin": _PIT,
671
+ }
672
+
673
+
674
+ def unsupported(reason: str, tool: str | None = None) -> ProbeResult:
675
+ """The one way to say "gate B did not run", always naming why."""
676
+ return {
677
+ "status": f"unsupported({reason})",
678
+ "tool": tool,
679
+ "survived": [],
680
+ "waived": [],
681
+ }
682
+
683
+
684
+ class DiffScope(NamedTuple):
685
+ """What a unified diff says about a stage, in the two forms the gate needs.
686
+
687
+ `touched` — file → the new-side lines it added or modified. This is what
688
+ scopes a surviving mutant to work the stage actually did.
689
+
690
+ `mentioned` — every file the diff NAMES in a header, whether or not it added
691
+ a line. A renamed or mode-changed file is mentioned but not touched, and
692
+ that distinction is what lets the coverage check be total: every changed
693
+ source must be mentioned, while only the ones with added lines can carry
694
+ findings. Checking `touched` instead would fail a stage for renaming a file.
695
+ """
696
+
697
+ touched: dict[str, set[int]]
698
+ mentioned: set[str]
699
+
700
+
701
+ def read_diff(diff_path: Path | None) -> DiffScope | None:
702
+ """Parse a unified diff, or `None` when it cannot be read at all.
703
+
704
+ A modification shows up as a `-`/`+` pair, so tracking the `+` side alone
705
+ covers both added and modified lines.
706
+
707
+ `None` and an empty scope are different answers and callers must keep them
708
+ apart: `None` means the diff was unreadable (`unsupported(diff-unavailable)`),
709
+ while an empty `touched` means it was read and added nothing anywhere.
710
+ """
711
+ if diff_path is None or not Path(diff_path).is_file():
712
+ return None
713
+ try:
714
+ text = Path(diff_path).read_text(encoding="utf-8", errors="replace")
715
+ except OSError:
716
+ return None
717
+ touched: dict[str, set[int]] = {}
718
+ mentioned: set[str] = set()
719
+ current: str | None = None
720
+ line_no = 0
721
+ for raw in text.splitlines():
722
+ if raw.startswith("diff --git "):
723
+ m = _DIFF_GIT_RE.match(raw)
724
+ if m:
725
+ mentioned.update(_strip_diff_prefix(g) for g in m.groups())
726
+ # A pure rename or mode change emits no `---`/`+++` pair and no
727
+ # hunk, so this header is the ONLY place those files are named.
728
+ # Both sides are recorded: a rename names the old and the new path.
729
+ continue
730
+ if raw.startswith("--- ") or raw.startswith("+++ "):
731
+ target = raw[4:].strip()
732
+ named = None if target == "/dev/null" else _strip_diff_prefix(target)
733
+ if named is not None:
734
+ mentioned.add(named)
735
+ if raw.startswith("+++ "):
736
+ # `/dev/null` on the new side is a deletion: nothing to mutate.
737
+ current = named
738
+ continue
739
+ if raw.startswith("@@"):
740
+ m = _HUNK.match(raw)
741
+ if m is None:
742
+ # A header we cannot read — a combined/merge diff (`@@@ -a -b +c @@@`)
743
+ # or a truncated file. Skipping it would drop every line in the
744
+ # hunk, and in a MIXED diff the coverage check would still be
745
+ # satisfied by the ordinary hunks while these changes went
746
+ # unchecked. The whole diff is unreadable instead.
747
+ return None
748
+ line_no = int(m.group(1))
749
+ continue
750
+ if current is None or not line_no:
751
+ continue
752
+ if raw.startswith("+"):
753
+ touched.setdefault(current, set()).add(line_no)
754
+ line_no += 1
755
+ elif raw.startswith(" ") or raw == "":
756
+ # Allowlist, not "anything but `-`": ``
757
+ # occupies no new-side line, and counting it would slide every later
758
+ # `+` down by one — the intersection would then miss the changed line
759
+ # and report PASS. Only a context line advances the counter.
760
+ line_no += 1
761
+ return DiffScope(touched, mentioned)
762
+
763
+
764
+ def _strip_diff_prefix(target: str) -> str:
765
+ """Drop git's `b/` prefix and any trailing tab-separated timestamp."""
766
+ path = target.split("\t", 1)[0]
767
+ if path.startswith("a/") or path.startswith("b/"):
768
+ path = path[2:]
769
+ return selfmock_path_key(path)
770
+
771
+
772
+ def _coerce_line(value: object) -> int | None:
773
+ """A reported line as the `int` the diff index is keyed on, or `None`.
774
+
775
+ A tool that reports `"11"` names the same line as `11`; treating them as
776
+ different would drop the survivor and read as PASS. Anything that will not
777
+ coerce — a word, a `10-12` range, an object — places nowhere.
778
+
779
+ The single definition of "placeable": the adapters screen undetected mutants
780
+ with it so an unplaceable one fails its report, and `evaluate` applies it
781
+ again as the shared defense. Two spellings of this test could disagree, and
782
+ the row would fall between them into a PASS.
783
+ """
784
+ try:
785
+ return int(value)
786
+ except (TypeError, ValueError):
787
+ return None
788
+
789
+
790
+ def _new_side_line(survivor: dict) -> int | None:
791
+ """The survivor's placeable line, or `None`.
792
+
793
+ Reaching `None` here means an adapter let an unplaceable row through; the
794
+ row is skipped and logged rather than raised on, so one bad row cannot abort
795
+ a whole verdict.
796
+ """
797
+ return _coerce_line(survivor.get("line"))
798
+
799
+
800
+ def _worktree_relative(path: str, worktree: Path | None) -> str:
801
+ """Re-spell an absolute tool path relative to the worktree root.
802
+
803
+ Mutation tools report absolute file names; a diff names them relative to the
804
+ worktree. `selfmock_path_key` folds cosmetic drift but not this, so without
805
+ the conversion every survivor misses the intersection and the stage reads
806
+ PASS. A path already relative, or outside the worktree, is left alone.
807
+ """
808
+ if worktree is None:
809
+ return path
810
+ try:
811
+ return str(Path(path).relative_to(Path(worktree)))
812
+ except ValueError:
813
+ return path
814
+
815
+
816
+ def _diff_key(survivor: dict, worktree: Path | None) -> str:
817
+ """The spelling of a survivor's file that `read_diff` keys `touched` on."""
818
+ return selfmock_path_key(
819
+ _worktree_relative(str(survivor.get("file", "")), worktree)
820
+ )
821
+
822
+
823
+ def evaluate(
824
+ survivors: list[dict], diff_path: Path | None, worktree: Path | None, tool: str
825
+ ) -> ProbeResult:
826
+ """Turn an adapter's raw survivors into a verdict. Adapters must route here.
827
+
828
+ `survivors` are `{file, line, mutant}` rows. Two spellings are reconciled
829
+ here so no adapter has to get them right on its own — either mistake would
830
+ drop the survivor silently into a PASS:
831
+
832
+ - `file` is folded to worktree-relative and then through `selfmock_path_key`,
833
+ which also absorbs `./` prefixes, duplicated slashes and Windows
834
+ separators. Adapters should still emit worktree-relative paths, because
835
+ that is what lands in the sidecar for a human to read.
836
+ - `line` must be the NEW-side (post-change) line number. A string is coerced,
837
+ but an old-side number simply points somewhere else and cannot be rescued.
838
+ """
839
+ scope = read_diff(diff_path)
840
+ if scope is None:
841
+ return unsupported("diff-unavailable", tool=tool)
842
+ touched = scope.touched
843
+ unplaceable = [s for s in survivors if _new_side_line(s) is None]
844
+ if unplaceable:
845
+ print(
846
+ f"mutation-probe: {tool} reported {len(unplaceable)} mutant(s) with no "
847
+ "usable line number; they cannot be matched against the diff and are "
848
+ "not counted",
849
+ file=sys.stderr,
850
+ )
851
+ covered = [
852
+ s
853
+ for s in survivors
854
+ if _new_side_line(s) in touched.get(_diff_key(s, worktree), ())
855
+ ]
856
+ if len(covered) > SURVIVOR_CAP:
857
+ print(
858
+ f"mutation-probe: {tool} left {len(covered)} surviving mutants on changed "
859
+ f"lines; reporting the first {SURVIVOR_CAP} "
860
+ f"({len(covered) - SURVIVOR_CAP} more not listed)",
861
+ file=sys.stderr,
862
+ )
863
+ return {
864
+ "status": "FAIL" if covered else "PASS",
865
+ "tool": tool,
866
+ "survived": covered[:SURVIVOR_CAP],
867
+ # How many were actually found, before the cap shortened the list. The
868
+ # verdict is decided from this, never from the trimmed list: a reader can
869
+ # only waive what the report shows, so recomputing from `survived` alone
870
+ # would let the visible ones be cleared while the overflow stayed unseen.
871
+ "survivedTotal": len(covered),
872
+ "waived": [],
873
+ }
874
+
875
+
876
+ def _targets_absent_from_diff(
877
+ targets: list[Path], mentioned: set[str], worktree: Path | None
878
+ ) -> list[str]:
879
+ """The changed sources the diff never names. Empty means the two inputs agree.
880
+
881
+ TOTAL coverage, not overlap. Overlap left a real hole: with targets [A, B]
882
+ and a diff covering only A, the check passed on A while a survivor in B fell
883
+ outside the intersection and vanished into a PASS. Asking `mentioned` rather
884
+ than `touched` is what makes totality affordable — a renamed or mode-changed
885
+ target is named by the diff without adding a line, so it satisfies this
886
+ without being expected to carry findings.
887
+ """
888
+ return [
889
+ key
890
+ for target in targets
891
+ if (key := selfmock_path_key(_worktree_relative(str(target), worktree)))
892
+ not in mentioned
893
+ ]
894
+
895
+
896
+ def mutation_waivers(waivers) -> list[dict]:
897
+ """The entries in the shared waiver file that belong to gate B.
898
+
899
+ ONE file (`<task_root>/qa/self-mock-waivers.json`) serves both gates so the
900
+ user manages a single place and `waiverSource` stays singular. A static
901
+ waiver names a `signal`; a mutation waiver names a `mutant` — the field is
902
+ the discriminator, so there is no `kind` for anyone to get wrong.
903
+
904
+ The filter is explicit rather than relying on the keys failing to line up: a
905
+ survivor whose `mutant` the tool left null would otherwise share the `None`
906
+ third element with a static entry and be cleared by it.
907
+ """
908
+ return [
909
+ w
910
+ for w in (waivers or [])
911
+ if isinstance(w, dict)
912
+ and isinstance(w.get("mutant"), str)
913
+ and w["mutant"].strip()
914
+ ]
915
+
916
+
917
+ def _apply_waivers(result: ProbeResult, waivers) -> ProbeResult:
918
+ """Move user-acknowledged survivors out of `survived` and re-decide.
919
+
920
+ Mirrors gate A exactly, including what it refuses to do: this only MATCHES.
921
+ An entry missing `reason` or `acknowledgedBy` is carried into `waived` so
922
+ `validate-run.py` can block on it — rejecting it here would let the run that
923
+ produced the finding excuse itself by writing the file.
924
+
925
+ Only a real verdict can be waived. There is no finding to excuse when the
926
+ status is `unsupported(...)`, and letting a waiver rewrite that would turn
927
+ "gate B never ran" into a pass.
928
+ """
929
+ if result["status"] not in ("PASS", "FAIL"):
930
+ return result
931
+ entries = mutation_waivers(waivers)
932
+ if not entries:
933
+ return result
934
+ listed = list(result["survived"])
935
+ remaining, waived = partition_waived_entries(listed, entries, "mutant")
936
+ return {
937
+ **result,
938
+ "status": _waived_verdict(result, listed, remaining),
939
+ "survived": remaining,
940
+ "waived": list(result["waived"]) + waived,
941
+ }
942
+
943
+
944
+ def _waived_verdict(
945
+ result: ProbeResult, listed: list[dict], remaining: list[dict]
946
+ ) -> str:
947
+ """`PASS` only when EVERY survivor was waived — including any the cap hid.
948
+
949
+ A user can only acknowledge what they were shown. Waiving every listed
950
+ survivor of a truncated report must not clear the ones that were trimmed:
951
+ they were never reviewed, and no acknowledgement for them can exist.
952
+
953
+ `survivedTotal` is what `evaluate` found before the cap. Each way it can be
954
+ untrustworthy fails closed:
955
+
956
+ - smaller than the list it accompanies — the result contradicts itself, so
957
+ nothing in it is reliable enough to clear a finding;
958
+ - absent — a list sitting exactly at the cap may be truncated, so it is
959
+ treated as if it is. Every real adapter routes through `evaluate` and does
960
+ carry the total, so this only catches a hand-built result.
961
+ """
962
+ if remaining:
963
+ return "FAIL"
964
+ total = result.get("survivedTotal")
965
+ if isinstance(total, int) and total >= len(listed):
966
+ return "FAIL" if total > len(listed) else "PASS"
967
+ if total is not None:
968
+ return "FAIL"
969
+ return "FAIL" if len(listed) >= SURVIVOR_CAP else "PASS"
970
+
971
+
972
+ def run_probe(
973
+ lang: str,
974
+ changed_files: list[Path],
975
+ diff_path: Path | None,
976
+ worktree: Path | None,
977
+ waivers=(),
978
+ ) -> ProbeResult:
979
+ """Probe `lang`'s changed files, or say precisely why it could not be done.
980
+
981
+ `changed_files` must be EVERY file the stage changed, not a pre-filtered
982
+ subset — the adapter selects its own production sources from it. Forwarding
983
+ only the changed TEST files (gate A's `scannedFiles`, which is the tempting
984
+ thing to reuse) leaves nothing to mutate, and the run reports a vacuous PASS
985
+ with gate B silently dead behind it.
986
+ """
987
+ adapter = ADAPTERS.get(lang)
988
+ if adapter is None:
989
+ return unsupported(f"no-adapter:{lang}")
990
+ # Checked before `run` so a missing binary surfaces as its own reason rather
991
+ # than as an adapter-specific crash or, worse, an empty survivor list.
992
+ if not adapter.is_declared(worktree):
993
+ return unsupported("tool-not-declared", tool=adapter.name)
994
+ # Selected HERE rather than inside each adapter: three adapters each doing
995
+ # their own selection is three chances to answer PASS for a run that never
996
+ # mutated anything. An adapter is never handed an empty set.
997
+ targets = production_sources(changed_files, lang)
998
+ if not targets:
999
+ return unsupported("no-production-sources", tool=adapter.name)
1000
+ # `--changed-file` and `--diff` are built by two separate git commands, so
1001
+ # nothing guarantees they describe the same work. If the diff covers none of
1002
+ # the files about to be mutated, every survivor falls outside the
1003
+ # intersection and the stage is stamped PASS having verified nothing.
1004
+ scope = read_diff(diff_path)
1005
+ if scope is None:
1006
+ return unsupported("diff-unavailable", tool=adapter.name)
1007
+ missing = _targets_absent_from_diff(targets, scope.mentioned, worktree)
1008
+ if missing:
1009
+ print(
1010
+ f"mutation-probe: {adapter.name} was handed {len(missing)} changed "
1011
+ f"source(s) the diff never names: {missing}",
1012
+ file=sys.stderr,
1013
+ )
1014
+ return unsupported("diff-incomplete", tool=adapter.name)
1015
+ if not scope.touched:
1016
+ # The diff names every target but adds no line anywhere — a deletion-only
1017
+ # or header-only diff. There is nothing for gate B to verify, which is
1018
+ # not the same as verifying it and finding nothing.
1019
+ return unsupported("diff-adds-no-line", tool=adapter.name)
1020
+ result = adapter.run(targets, diff_path, worktree)
1021
+ if not _states_a_known_verdict(result):
1022
+ return unsupported("adapter-malformed-status", tool=adapter.name)
1023
+ return _apply_waivers(result, waivers)
1024
+
1025
+
1026
+ def _states_a_known_verdict(result: object) -> bool:
1027
+ """True when `result` speaks the vocabulary the gate downstream understands.
1028
+
1029
+ "Adapters parse, they never judge" is only a rule if something checks it.
1030
+ `validate-run.py` blocks on `FAIL` and folds `unsupported(...)` down to gate
1031
+ A; a third value matches neither and would sail through both. Enforcing it
1032
+ on the interface means each new adapter inherits the check instead of
1033
+ re-earning trust.
1034
+ """
1035
+ if not isinstance(result, dict):
1036
+ return False
1037
+ status = result.get("status")
1038
+ if not isinstance(status, str):
1039
+ return False
1040
+ if status in ("PASS", "FAIL"):
1041
+ return True
1042
+ if not (status.startswith("unsupported(") and status.endswith(")")):
1043
+ return False
1044
+ # An empty reason is the one thing `unsupported` must never be: "gate B did
1045
+ # not run" with no way to find out why is indistinguishable from a silent skip.
1046
+ return bool(_unsupported_reason(status))
1047
+
1048
+
1049
+ def _unsupported_reason(status: str) -> str:
1050
+ """The text inside `unsupported(...)`, or `""` when there is none."""
1051
+ return status[len("unsupported(") : -1].strip()
1052
+
1053
+
1054
+ def _probe_langs(
1055
+ changed_files: list[Path], worktree: Path | None
1056
+ ) -> dict[str, list[Path]]:
1057
+ """Group the stage's changed files by the language key each adapter answers to.
1058
+
1059
+ Paths are folded to worktree-relative FIRST. An adapter decides what to
1060
+ exclude from its own path shape — Stryker drops anything under a `test`,
1061
+ `tests` or `spec` directory — so an absolute path drags the checkout's own
1062
+ ancestors into that decision and can exclude real production code whenever
1063
+ the worktree happens to live under such a directory. That failure is silent:
1064
+ the gate simply finds nothing to mutate and reports a vacuous PASS.
1065
+ """
1066
+ grouped: dict[str, list[Path]] = {}
1067
+ for path in changed_files:
1068
+ relative = Path(_worktree_relative(str(path), worktree))
1069
+ lang = EXT_TO_LANG.get(relative.suffix)
1070
+ if lang is None:
1071
+ continue
1072
+ grouped.setdefault(lang, []).append(relative)
1073
+ return grouped
1074
+
1075
+
1076
+ # --- reason classification SSOT ---------------------------------------------
1077
+ #
1078
+ # Every `unsupported(<reason>)` this module can emit falls into exactly one of
1079
+ # three classes. BOTH consumers read this one definition — `_merge_probes` below
1080
+ # decides whether a sibling language's PASS may stand, and
1081
+ # `validate-run.py::_validate_selfmock` decides whether the run blocks. A second,
1082
+ # parallel table is how those two drift apart.
1083
+ CAPABILITY_GAP = "capability-gap"
1084
+ NOTHING_TO_VERIFY = "nothing-to-verify"
1085
+ INTEGRITY_INSPECTION = "integrity-inspection"
1086
+
1087
+ # Gate B declared this boundary in advance: it never claimed to cover that
1088
+ # language here. Non-blocking, and a sibling's PASS may stand over it — without
1089
+ # that, gate B would be unusable in any polyglot repo and would wedge every repo
1090
+ # with no mutation tooling installed.
1091
+ CAPABILITY_GAP_REASONS = frozenset(
1092
+ {
1093
+ "tool-not-declared",
1094
+ "diff-scope-unavailable",
1095
+ "no-production-sources",
1096
+ # No changed file is in a language any adapter covers — a Go, Ruby or C#
1097
+ # stage. Gate A's trigger is extension-agnostic, so those repos reach
1098
+ # gate B on every run; blocking here would wedge all of them.
1099
+ "no-changed-sources",
1100
+ }
1101
+ )
1102
+
1103
+ # The tool ran to completion and the changed code genuinely offered nothing to
1104
+ # check. That is a property of the code, not a failure of the run, so it is also
1105
+ # non-blocking. `diff-adds-no-line` lives here because deletion-only and
1106
+ # rename-only stages are legitimate and must not wedge.
1107
+ NOTHING_TO_VERIFY_REASONS = frozenset(
1108
+ {
1109
+ "no-mutants-generated",
1110
+ "diff-adds-no-line",
1111
+ }
1112
+ )
1113
+
1114
+ # We tried to inspect and cannot trust the answer. Never overridden by a sibling
1115
+ # PASS, and BLOCKING at the gate: each of these is a fixable fault in the run's
1116
+ # own inputs or output, not a boundary anyone declared.
1117
+ #
1118
+ # `no-conclusive-mutants` belongs here rather than with "nothing to verify":
1119
+ # mutants existed and not one completed a trial, so the same mutant reports
1120
+ # `Survived` when the run finishes and `Pending` when it is cut short. Classing
1121
+ # it as a gap would let the verdict turn on whether the run was interrupted.
1122
+ INTEGRITY_INSPECTION_REASONS = frozenset(
1123
+ {
1124
+ "diff-incomplete",
1125
+ "diff-unavailable",
1126
+ "report-unavailable",
1127
+ "report-unparsed",
1128
+ "adapter-malformed-status",
1129
+ "no-conclusive-mutants",
1130
+ }
1131
+ )
1132
+
1133
+
1134
+ def _strip_tool_qualifier(reason: str) -> str:
1135
+ """Drop the `<tool>:` prefix `_merged_reasons` adds, keeping `no-adapter:<lang>`.
1136
+
1137
+ The sidecar records the MERGED reason, so what reaches the gate looks like
1138
+ `stryker:diff-incomplete`. `no-adapter:<lang>` carries its colon as part of
1139
+ the reason itself and is never tool-qualified (its `tool` is `None`), so it
1140
+ is recognised before the split rather than being truncated to a bare lang.
1141
+ """
1142
+ if reason.startswith("no-adapter:"):
1143
+ return reason
1144
+ _, sep, rest = reason.partition(":")
1145
+ return rest if sep else reason
1146
+
1147
+
1148
+ def _classify_one(reason: str) -> str:
1149
+ bare = _strip_tool_qualifier(reason.strip())
1150
+ if bare.startswith("no-adapter:") or bare in CAPABILITY_GAP_REASONS:
1151
+ return CAPABILITY_GAP
1152
+ if bare in NOTHING_TO_VERIFY_REASONS:
1153
+ return NOTHING_TO_VERIFY
1154
+ # Fail closed: a reason nobody classified is treated as an inspection
1155
+ # failure, so forgetting to classify one blocks rather than passes.
1156
+ return INTEGRITY_INSPECTION
1157
+
1158
+
1159
+ def classify_reason(status: str) -> str:
1160
+ """Classify an `unsupported(<reason>)` status into one of the three classes.
1161
+
1162
+ A merged status can carry several `; `-joined reasons; the most severe class
1163
+ wins, so one inspection failure among capability gaps still governs.
1164
+ """
1165
+ reasons = [r for r in _unsupported_reason(status).split(";") if r.strip()]
1166
+ classes = {_classify_one(r) for r in reasons}
1167
+ if not classes or INTEGRITY_INSPECTION in classes:
1168
+ return INTEGRITY_INSPECTION
1169
+ if NOTHING_TO_VERIFY in classes:
1170
+ return NOTHING_TO_VERIFY
1171
+ return CAPABILITY_GAP
1172
+
1173
+
1174
+ def _merge_probes(results: list[ProbeResult]) -> ProbeResult:
1175
+ """Fold one verdict per language into the single one the sidecar records.
1176
+
1177
+ Any `FAIL` fails the stage — that precedence is absolute.
1178
+
1179
+ Otherwise a single real `PASS` carries the stage, but ONLY over capability
1180
+ gaps. A probed language's clean result is evidence and an uncoverable
1181
+ language is merely silence; a language whose INSPECTION failed is neither,
1182
+ and letting a sibling speak for it is how total coverage inside one language
1183
+ becomes partial coverage across two.
1184
+
1185
+ When nothing was probed at all, every reason is kept: which language went
1186
+ unchecked is the operator's next action.
1187
+ """
1188
+ statuses = [r["status"] for r in results]
1189
+ survived = [row for r in results for row in r["survived"]]
1190
+ waived = [row for r in results for row in r["waived"]]
1191
+ ran = [r["tool"] for r in results if r["status"] in ("PASS", "FAIL") and r["tool"]]
1192
+ unsupported_results = [
1193
+ r for r in results if str(r["status"]).startswith("unsupported(")
1194
+ ]
1195
+ # One classification, shared with the gate: a sibling PASS may stand over a
1196
+ # capability gap or a language with nothing to verify, never over an
1197
+ # inspection failure.
1198
+ blocking = [
1199
+ r
1200
+ for r in unsupported_results
1201
+ if classify_reason(str(r["status"])) == INTEGRITY_INSPECTION
1202
+ ]
1203
+ if "FAIL" in statuses:
1204
+ status = "FAIL"
1205
+ elif blocking:
1206
+ # Reported from the blocking reasons alone: those are what the operator
1207
+ # has to fix, and folding in the capability gaps would bury them.
1208
+ status = f"unsupported({'; '.join(_merged_reasons(blocking))})"
1209
+ elif "PASS" in statuses:
1210
+ status = "PASS"
1211
+ else:
1212
+ status = f"unsupported({'; '.join(_merged_reasons(unsupported_results))})"
1213
+ return {
1214
+ "status": status,
1215
+ "tool": ",".join(sorted(set(ran))) or None,
1216
+ "survived": survived,
1217
+ # Summed so the sidecar still shows how many were found even when a
1218
+ # language's list was capped; `survived` alone would understate it.
1219
+ "survivedTotal": sum(
1220
+ r["survivedTotal"]
1221
+ if isinstance(r.get("survivedTotal"), int)
1222
+ else len(r["survived"])
1223
+ for r in results
1224
+ ),
1225
+ "waived": waived,
1226
+ }
1227
+
1228
+
1229
+ def _merged_reasons(results: list[ProbeResult]) -> list[str]:
1230
+ """Each language's reason, tool-qualified where the reason alone is ambiguous."""
1231
+ reasons: list[str] = []
1232
+ for r in results:
1233
+ reason = _unsupported_reason(str(r["status"]))
1234
+ # `no-adapter:<lang>` already names its language; `tool-not-declared` and
1235
+ # friends do not, so the tool that produced them is prepended.
1236
+ if r["tool"]:
1237
+ reason = f"{r['tool']}:{reason}"
1238
+ if reason not in reasons:
1239
+ reasons.append(reason)
1240
+ return reasons
1241
+
1242
+
1243
+ def probe_changed_files(
1244
+ changed_files: list[Path],
1245
+ diff_path: Path | None,
1246
+ worktree: Path | None,
1247
+ waivers=(),
1248
+ ) -> ProbeResult:
1249
+ """Probe every language the stage touched and merge the verdicts into one.
1250
+
1251
+ This is the entry point the detector writes its sidecar from. `changed_files`
1252
+ is the stage's WHOLE changed set — each adapter selects its own production
1253
+ sources out of it (see `run_probe`).
1254
+ """
1255
+ grouped = _probe_langs(changed_files, worktree)
1256
+ if not grouped:
1257
+ return unsupported("no-changed-sources")
1258
+ return _merge_probes(
1259
+ [
1260
+ run_probe(lang, files, diff_path, worktree, waivers)
1261
+ for lang, files in sorted(grouped.items())
1262
+ ]
1263
+ )