mjolnir-qa 0.5.30 → 0.5.32
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/README.md +484 -425
- package/dist/cli.d.mts +1 -1
- package/dist/cli.mjs +2 -2
- package/dist/mcp/stdio.mjs +2 -2
- package/package.json +5 -2
package/CHANGELOG.md
CHANGED
|
@@ -9,6 +9,19 @@ Rule behavior changes (new rules, FP-rate changes against the corpus,
|
|
|
9
9
|
severity changes) are first-class entries here — rule IDs are immutable
|
|
10
10
|
once shipped, so this file is the record of what changed between versions.
|
|
11
11
|
|
|
12
|
+
## [0.5.32] — 2026-09-08
|
|
13
|
+
|
|
14
|
+
### Changes since 0.5.31
|
|
15
|
+
|
|
16
|
+
- P0: repo state + truth drift — single measured count with drift lock (#62)
|
|
17
|
+
|
|
18
|
+
## [0.5.31] — 2026-09-08
|
|
19
|
+
|
|
20
|
+
### Changes since 0.5.30
|
|
21
|
+
|
|
22
|
+
- chore: resync managed surface stamp to v0.5.30 (docs narrative landed on the release state)
|
|
23
|
+
- docs: README narrative refresh + reproducible flow/architecture assets with contract locks (docs:flow, docs:architecture)
|
|
24
|
+
|
|
12
25
|
## [0.5.30] — 2026-09-08
|
|
13
26
|
|
|
14
27
|
### Changes since 0.5.29
|
package/README.md
CHANGED
|
@@ -2,10 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
<img src="assets/readme/logo.png" alt="Mjölnir — Verification Trust Engine" width="800" />
|
|
4
4
|
|
|
5
|
-
###
|
|
5
|
+
### Tests tell you what passed. Mjölnir tells you what you can trust.
|
|
6
6
|
|
|
7
|
-
**Verification Trust Engine
|
|
8
|
-
|
|
7
|
+
**Mjölnir is a Verification Trust Engine.** Test frameworks verify your
|
|
8
|
+
software. Mjölnir verifies the system that does the verifying — the test
|
|
9
|
+
suite, the run artifacts and the CI pipeline — and reports a worthiness
|
|
10
|
+
score with the evidence behind every deduction.
|
|
9
11
|
|
|
10
12
|
[](https://www.npmjs.com/package/mjolnir-qa)
|
|
11
13
|
[](https://www.npmjs.com/package/mjolnir-qa)
|
|
@@ -17,9 +19,7 @@ pipelines, reports a worthiness score, and shows exactly where trust breaks.
|
|
|
17
19
|
npx mjolnir-qa@latest
|
|
18
20
|
```
|
|
19
21
|
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
[See it work](#-see-it-work) · [Quickstart](#-quickstart) · [Who it's for](#-who-is-this-for) · [Why not a linter](#-mjölnir-is-not-another-linter) · [What it verifies](#-what-mjölnir-verifies) · [Scoring](#-how-the-score-works) · [Runtime evidence](#-runtime-evidence) · [CI](#-ci-integration) · [Agents & MCP](#-works-with-your-agent) · [Docs](#-documentation) · [Contributing](#-contributing)
|
|
22
|
+
[See it work](#see-it-work) · [Quickstart](#quickstart) · [What it finds](#what-mjölnir-finds) · [Score](#the-worthiness-score) · [Evidence model](#the-evidence-model) · [Forensics](#runtime-forensics) · [CI](#ci-integrity) · [Agents](#ai-agents) · [Security](#trust-and-security) · [Limits](#what-mjölnir-cannot-tell-you) · [Docs](#documentation)
|
|
23
23
|
|
|
24
24
|
<details>
|
|
25
25
|
<summary>Read this in another language — 22 translations</summary>
|
|
@@ -35,73 +35,105 @@ it; `npm run docs:translations` reports how far.
|
|
|
35
35
|
|
|
36
36
|
---
|
|
37
37
|
|
|
38
|
-
##
|
|
38
|
+
## The problem
|
|
39
|
+
|
|
40
|
+
A green pipeline is a claim, not a proof. The same checkmark is printed
|
|
41
|
+
whether a suite genuinely verified your product or merely failed to
|
|
42
|
+
contradict it. Every one of these ships green:
|
|
43
|
+
|
|
44
|
+
- a committed `.only` that ran 3 tests instead of 900
|
|
45
|
+
- a `continue-on-error: true` on the job that was supposed to gate
|
|
46
|
+
- a `|| true` after the test command
|
|
47
|
+
- a test that asserts nothing, or whose body is empty
|
|
48
|
+
- a retry wrapper that turns a real failure into a lucky pass
|
|
49
|
+
- a report the workflow uploads but never actually generated
|
|
50
|
+
- a hard sleep holding a race condition together until the day it doesn't
|
|
51
|
+
|
|
52
|
+
None of these are exotic, and none of them turn the pipeline red. They look
|
|
53
|
+
intentional to a reviewer — which is exactly why they survive.
|
|
54
|
+
|
|
55
|
+
## The Mjölnir Principle
|
|
56
|
+
|
|
57
|
+
> ### No evidence. No proof.
|
|
58
|
+
|
|
59
|
+
Mjölnir would rather say _unknown_ than manufacture confidence. Where a
|
|
60
|
+
conventional tool rounds silence up to "fine", it stops and names the gap:
|
|
61
|
+
|
|
62
|
+
| Situation | What Mjölnir reports |
|
|
63
|
+
| ------------------------------------ | ------------------------------------------------------- |
|
|
64
|
+
| No test declarations found | Score `null` — **UNKNOWN**, never a fabricated 100 |
|
|
65
|
+
| No baseline / no comparable revision | **UNKNOWN**, with the reason named — never an assumed 0 |
|
|
66
|
+
| Scan truncated (budget, unreadable) | **PARTIAL**, exit `2` — never presented as clean |
|
|
67
|
+
|
|
68
|
+
Unknown is a valid answer, and this is the reason: a tool that says
|
|
69
|
+
"verified" when it does not know is the same failure mode as a CI gate
|
|
70
|
+
that says green when it never ran.
|
|
71
|
+
|
|
72
|
+
## How it works
|
|
73
|
+
|
|
74
|
+
Mjölnir sits between your test system and your release decision. It reads
|
|
75
|
+
the suite, the CI workflows and — when you point it at one — the artifacts
|
|
76
|
+
of a real run.
|
|
39
77
|
|
|
40
|
-
<!-- Plays inline on github.com only: <video> is rendered for GitHub's own
|
|
41
|
-
user-content CDN, never for a repo-relative path. The <a> below is the
|
|
42
|
-
fallback for every other renderer (npm, mirrors, offline clones). -->
|
|
43
78
|
<p align="center">
|
|
44
|
-
<
|
|
45
|
-
src="https://github.com/user-attachments/assets/0e1af1e4-1e27-4c1c-9ec4-2717d194df05"
|
|
46
|
-
poster="https://raw.githubusercontent.com/Sergey-Bar/Mjolnir/main/assets/video/mjolnir-demo-poster.png"
|
|
47
|
-
controls
|
|
48
|
-
muted
|
|
49
|
-
playsinline
|
|
50
|
-
width="900"></video>
|
|
79
|
+
<img src="assets/readme/architecture.svg" alt="Mjölnir reads the test suite and the CI pipeline statically, and reads the Playwright JSON and JUnit XML artifacts of a real run. It discovers, analyzes, correlates and measures across four evidence streams — test quality, CI integrity, runtime forensics and selector health — stamping each finding E0 observation, E1 pattern evidence or E2 deterministic proof, weighted none, half and full. A trust ladder L0 to L5 shows the top three rungs require a real run. Out come findings, a worthiness score of 75 out of 100 labelled NEEDS WORK, and a CI gate on the frozen exit codes 0, 1, 2, 10 and 20. An agent loop runs scan, evidence, handoff, AI agent, re-scan, proof." width="1600" />
|
|
51
80
|
</p>
|
|
52
81
|
|
|
82
|
+
It does not run your tests, install your dependencies, or execute the code
|
|
83
|
+
it scans. Static analysis reads source text; forensics reads report files
|
|
84
|
+
that already exist on disk.
|
|
85
|
+
|
|
86
|
+
<sub>Generated by `npm run docs:architecture` and drift-locked in CI; the
|
|
87
|
+
score, counts and rule ID are read from
|
|
88
|
+
[`script.demo.json`](assets/video/script.demo.json), not written by hand.
|
|
89
|
+
Open [`architecture.svg`](assets/readme/architecture.svg) on its own for
|
|
90
|
+
the full-resolution version.</sub>
|
|
91
|
+
|
|
92
|
+
---
|
|
93
|
+
|
|
94
|
+
## See it work
|
|
95
|
+
|
|
96
|
+
One false-green CI gate — caught, fixed with the tool's own printed fix,
|
|
97
|
+
and re-proved by a second scan.
|
|
98
|
+
|
|
53
99
|
<p align="center">
|
|
54
|
-
<
|
|
55
|
-
re-proved, then handed to an agent.
|
|
56
|
-
<a href="assets/video/mjolnir-demo.mp4">Download the 1440p MP4</a> if the
|
|
57
|
-
player above doesn't load.
|
|
100
|
+
<img src="assets/readme/flow.svg" alt="npx mjolnir-qa@latest. A large grey 75 labelled NEEDS WORK, an arrow marked ONE FIX above and RE-SCANNED below, then a large lit 90 labelled WORTHY. Beneath: set -o pipefail, && not a semicolon, no continue-on-error — the fix the report printed, closing QA-CI-009 and QA-CI-001. Then 27 findings to 23, 4 errors to 1. Finally: 90, not 100 — the suite's other problems are still real." width="900" />
|
|
58
101
|
</p>
|
|
59
102
|
|
|
60
|
-
<sub>Every frame is real CLI output — the 75 → 90 score change is a real
|
|
61
|
-
re-scan after applying the fix the tool itself printed, never a mockup.
|
|
62
|
-
Rendered by `npm run docs:video` from
|
|
63
|
-
[`assets/video/script.demo.json`](assets/video/script.demo.json);
|
|
64
|
-
[`tests/contract/video-script.spec.ts`](tests/contract/video-script.spec.ts)
|
|
65
|
-
fails CI if that script stops matching what the CLI prints, or if the
|
|
66
|
-
findings the video shows as fixed turn out to still be there.</sub>
|
|
67
|
-
|
|
68
103
|
<details>
|
|
69
|
-
<summary><strong>Prefer it
|
|
104
|
+
<summary><strong>Prefer to watch it?</strong> The same run, as a 42-second recording</summary>
|
|
70
105
|
|
|
106
|
+
<!-- Plays inline on github.com only: <video> is rendered for GitHub's own
|
|
107
|
+
user-content CDN, never for a repo-relative path. The link below is
|
|
108
|
+
the fallback for every other renderer (npm, mirrors, offline clones). -->
|
|
71
109
|
<p align="center">
|
|
72
|
-
<
|
|
110
|
+
<video
|
|
111
|
+
src="https://github.com/user-attachments/assets/0e1af1e4-1e27-4c1c-9ec4-2717d194df05"
|
|
112
|
+
poster="https://raw.githubusercontent.com/Sergey-Bar/Mjolnir/main/assets/video/mjolnir-demo-poster.png"
|
|
113
|
+
controls
|
|
114
|
+
muted
|
|
115
|
+
playsinline
|
|
116
|
+
width="900"></video>
|
|
73
117
|
</p>
|
|
74
118
|
|
|
75
|
-
<sub>
|
|
76
|
-
|
|
77
|
-
`
|
|
78
|
-
|
|
79
|
-
|
|
119
|
+
<sub>Found, fixed, re-proved, then handed to an agent. If the player above
|
|
120
|
+
doesn't load, the file is
|
|
121
|
+
[`assets/video/mjolnir-demo.mp4`](assets/video/mjolnir-demo.mp4). Rendered
|
|
122
|
+
by `npm run docs:video`. The full `--verbose` report of the same scan is
|
|
123
|
+
[`demo.svg`](assets/readme/demo.svg) (`npm run docs:demo`).</sub>
|
|
80
124
|
|
|
81
125
|
</details>
|
|
82
126
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
masking a job, a `|| true` swallowing an exit code, hard sleeps, a
|
|
89
|
-
brittle selector, hardcoded staging URLs, a `networkidle` wait.
|
|
90
|
-
3. It turned each into a concrete finding with a rule ID, a location and a
|
|
91
|
-
fix — and a single score you can gate a PR on.
|
|
92
|
-
4. `mjolnir handoff` turned the findings into a remediation plan — evidence,
|
|
93
|
-
constraints and a copy-pastable prompt per finding — that Claude Code
|
|
94
|
-
(or any other agent) can work through, with the tool's own verification
|
|
95
|
-
discipline built in.
|
|
96
|
-
|
|
97
|
-
An 89-second tour of `explain` and `forensics` is built from
|
|
98
|
-
[`script.tour.json`](assets/video/script.tour.json) by `npm run docs:video`
|
|
99
|
-
— not committed as an MP4 (~16MB), so check
|
|
100
|
-
[Releases](../../releases) or render it yourself.
|
|
127
|
+
<sub>Every number above is read from
|
|
128
|
+
[`script.demo.json`](assets/video/script.demo.json) — the same values
|
|
129
|
+
[`video-script.spec.ts`](tests/contract/video-script.spec.ts) checks
|
|
130
|
+
against real CLI output — and the diff quotes the two committed workflows
|
|
131
|
+
verbatim. Regenerate with `npm run docs:flow`; drift-locked in CI.</sub>
|
|
101
132
|
|
|
102
133
|
### One finding, up close
|
|
103
134
|
|
|
104
|
-
|
|
135
|
+
`mjolnir explain QA-CI-001` prints a rule's whole trust record — including
|
|
136
|
+
its measured false-positive rate and the tier that rate earned it:
|
|
105
137
|
|
|
106
138
|
```text
|
|
107
139
|
▚ QA-CI-001 — continue-on-error masks a failing verification gate
|
|
@@ -135,28 +167,29 @@ HOW TO VERIFY THE FIX
|
|
|
135
167
|
Docs: mjolnir rules --md (full catalog, this rule included)
|
|
136
168
|
```
|
|
137
169
|
|
|
138
|
-
That is the unit of value: not a style nit, but a place where
|
|
139
|
-
|
|
170
|
+
That is the unit of value: not a style nit, but a place where CI is
|
|
171
|
+
reporting a pass it did not earn.
|
|
140
172
|
|
|
141
173
|
---
|
|
142
174
|
|
|
143
|
-
##
|
|
144
|
-
|
|
145
|
-
Run it against a repo for a full report and a worthiness score:
|
|
175
|
+
## Quickstart
|
|
146
176
|
|
|
147
177
|
```bash
|
|
148
178
|
npx mjolnir-qa@latest
|
|
149
179
|
```
|
|
150
180
|
|
|
151
|
-
|
|
152
|
-
|
|
181
|
+
That is the whole product: it scans the current directory, prints a
|
|
182
|
+
worthiness score and the findings behind it, and exits `0` if nothing at or
|
|
183
|
+
above the gate was found. **In CI, use the changed-scope form** — it
|
|
184
|
+
attributes findings to what your branch introduced, so a legacy suite does
|
|
185
|
+
not drown a first PR:
|
|
153
186
|
|
|
154
187
|
```bash
|
|
155
188
|
npx mjolnir-qa@latest --scope changed
|
|
156
189
|
```
|
|
157
190
|
|
|
158
|
-
|
|
159
|
-
|
|
191
|
+
`mjolnir ci install` writes that as a GitHub Actions workflow — advisory by
|
|
192
|
+
default, never blocking until you say so.
|
|
160
193
|
|
|
161
194
|
| Command | What it does |
|
|
162
195
|
| ----------------------------------- | ------------------------------------------------ |
|
|
@@ -165,235 +198,145 @@ and you're done. Everything else is optional.
|
|
|
165
198
|
| `mjolnir ci install` | Generate the advisory PR workflow |
|
|
166
199
|
| `mjolnir explain QA-CI-001` | What / why / fix + measured FP rate for one rule |
|
|
167
200
|
| `mjolnir why src/a.spec.ts:42` | Why this exact line was flagged — never a gate |
|
|
168
|
-
| `mjolnir
|
|
201
|
+
| `mjolnir forensics ./test-results/` | Runtime evidence from a real run |
|
|
202
|
+
| `mjolnir handoff` | Remediation plan for a coding agent |
|
|
169
203
|
| `mjolnir --json` / `--format sarif` | Machine-readable / GitHub Code Scanning |
|
|
170
204
|
| `mjolnir --strict` | Also run quarantine-tier rules (higher FP risk) |
|
|
171
|
-
| `mjolnir --cache` | Incremental re-scans via a local verdict cache |
|
|
172
|
-
|
|
173
|
-
<details>
|
|
174
|
-
<summary><strong>When something's flaky</strong></summary>
|
|
175
|
-
|
|
176
|
-
| Command | What it does |
|
|
177
|
-
| ----------------------------------- | --------------------------------------------------- |
|
|
178
|
-
| `mjolnir forensics ./test-results/` | Real run data → `TRUE-FLAKE` verdicts, `FLAKY.md` |
|
|
179
|
-
| `mjolnir triage ./test-results/` | Quarantine proposal from execution history |
|
|
180
|
-
| `mjolnir pw-report ./test-results/` | Playwright run summary — retries / flakes / slowest |
|
|
181
|
-
| `mjolnir doctor:playwright` | Playwright-only deep scan + Selector Health Score |
|
|
182
|
-
|
|
183
|
-
</details>
|
|
184
205
|
|
|
185
206
|
<details>
|
|
186
|
-
<summary><strong>
|
|
187
|
-
|
|
188
|
-
| Command
|
|
189
|
-
|
|
|
190
|
-
| `mjolnir
|
|
191
|
-
| `mjolnir
|
|
192
|
-
| `mjolnir
|
|
193
|
-
| `mjolnir
|
|
194
|
-
| `mjolnir
|
|
195
|
-
| `mjolnir
|
|
196
|
-
| `mjolnir
|
|
197
|
-
| `mjolnir
|
|
198
|
-
| `mjolnir
|
|
199
|
-
| `mjolnir
|
|
200
|
-
| `mjolnir
|
|
201
|
-
| `mjolnir
|
|
202
|
-
| `mjolnir
|
|
203
|
-
| `mjolnir
|
|
204
|
-
| `mjolnir
|
|
205
|
-
| `mjolnir
|
|
207
|
+
<summary><strong>Everything else</strong> — flake triage, reporting, governance</summary>
|
|
208
|
+
|
|
209
|
+
| Command | What it does |
|
|
210
|
+
| ----------------------------------- | ------------------------------------------------------ |
|
|
211
|
+
| `mjolnir triage ./test-results/` | Quarantine proposal from execution history |
|
|
212
|
+
| `mjolnir pw-report ./test-results/` | Playwright run summary — retries / flakes / slowest |
|
|
213
|
+
| `mjolnir doctor:playwright` | Playwright-only deep scan + Selector Health Score |
|
|
214
|
+
| `mjolnir fix --dry-run` / `fix` | Safe auto-fixes, each re-scanned to prove it landed |
|
|
215
|
+
| `mjolnir baseline` / `diff` | Snapshot findings, then report only new/worsened |
|
|
216
|
+
| `mjolnir impact --since <ref>` | What a commit introduced vs resolved |
|
|
217
|
+
| `mjolnir summary` | CI annotations + step summary from a saved report |
|
|
218
|
+
| `mjolnir pr-comment` | A scoped PR comment, as Markdown |
|
|
219
|
+
| `mjolnir debt` | Test-debt register with a cost model |
|
|
220
|
+
| `mjolnir handover` | New-QA onboarding map of the suite |
|
|
221
|
+
| `mjolnir init` | Detect frameworks + setup checklist (never overwrites) |
|
|
222
|
+
| `mjolnir suppressions` | List suppressed findings — governance transparency |
|
|
223
|
+
| `mjolnir rules --unmeasured` | The rules running on assumption, not measurement |
|
|
224
|
+
| `mjolnir rules --md` | Full rule catalog (JSON or Markdown) |
|
|
225
|
+
| `mjolnir doctor` | Self-audit of Mjölnir's own rule base |
|
|
226
|
+
| `mjolnir create-rule <ID>` | Scaffold a new rule + fixtures |
|
|
227
|
+
| `mjolnir stats` | Local all-time counters of fixes seen |
|
|
228
|
+
| `mjolnir badge` | shields.io endpoint JSON + snippet |
|
|
229
|
+
| `mjolnir --cache` | Incremental re-scans via a local verdict cache |
|
|
230
|
+
| `mjolnir --format mermaid` | Test-architecture diagram for a PR comment |
|
|
231
|
+
|
|
232
|
+
`mjolnir help <command>` prints usage, examples and the next step for any
|
|
233
|
+
of them.
|
|
206
234
|
|
|
207
235
|
</details>
|
|
208
236
|
|
|
209
|
-
|
|
210
|
-
|
|
237
|
+
Requires **Node.js ≥ 22.18**. Runs on Windows, macOS and Linux. Install
|
|
238
|
+
globally with `npm i -g mjolnir-qa` if you prefer it over `npx`.
|
|
211
239
|
|
|
212
240
|
---
|
|
213
241
|
|
|
214
|
-
##
|
|
215
|
-
|
|
216
|
-
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
|
233
|
-
|
|
|
234
|
-
|
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
| Reads real run data for `TRUE-FLAKE` verdicts | ❌ | ❌ | ❌ | ✅ |
|
|
240
|
-
| Deterministic (same input → same output) | ✅ | ✅ | ❌ | ✅ |
|
|
241
|
-
| Cost per scan | free | free | tokens | **zero** (local) |
|
|
242
|
-
|
|
243
|
-
\*`eslint-plugin-jest` (`expect-expect`) and `eslint-plugin-playwright`
|
|
244
|
-
(`expect-expect`, `no-wait-for-timeout`) cover these for their respective
|
|
245
|
-
frameworks.
|
|
246
|
-
|
|
247
|
-
**Use AI review too.** It catches nuance, intent, and design flaws no regex
|
|
248
|
-
can find. Mjölnir catches the structural patterns AI overlooks because they
|
|
249
|
-
look "intentional" — a committed `.only`, a swallowed exit code, a
|
|
250
|
-
`continue-on-error` on a test job (that one under `--strict`). Those aren't
|
|
251
|
-
bugs that need reasoning; they're facts that need scanning.
|
|
252
|
-
|
|
253
|
-
---
|
|
254
|
-
|
|
255
|
-
## 🔨 What Mjölnir verifies
|
|
256
|
-
|
|
257
|
-
| | |
|
|
258
|
-
| --- | ----------------------------------------------------------------------------------------------------------------- |
|
|
259
|
-
| ⚖️ | **Worthiness Score** — one number, transparent deduction table, no black box |
|
|
260
|
-
| 🎭 | **Selector Health Score** — grades your Playwright locators, not just your pass rate |
|
|
261
|
-
| 🔬 | **Runtime forensics** — reads real Playwright/JUnit run data to catch `TRUE-FLAKE`, not just static guesses |
|
|
262
|
-
| 🚨 | **CI-integrity rules** — catches `\|\| true` by default; `continue-on-error` detection runs under `--strict` |
|
|
263
|
-
| 🐍 | **All four Playwright bindings** — TypeScript, Python, Java, C#/.NET — plus pytest, JUnit/TestNG and CI workflows |
|
|
264
|
-
| 🔒 | **Local-first** — zero network calls while scanning, zero telemetry, runs in seconds |
|
|
265
|
-
|
|
266
|
-
Every rule ships with must-fire **and** must-not-fire fixtures. A rule that
|
|
267
|
-
fires on its own negative fixture cannot ship — that's the false-positive
|
|
268
|
-
firewall.
|
|
269
|
-
|
|
270
|
-
**The rule catalog.** Every family is collapsed below; the generated
|
|
271
|
-
full catalog lives in [`docs/rules/`](docs/rules/),
|
|
272
|
-
[what it checks](https://sergey-bar.github.io/Mjolnir/guide/what-it-checks),
|
|
273
|
-
or `mjolnir rules --md`.
|
|
274
|
-
|
|
275
|
-
> **Tier** — `quarantine` rules run only under `--strict` and never gate
|
|
276
|
-
> (capped to info); the severity shown is the authored severity.
|
|
277
|
-
|
|
278
|
-
<details>
|
|
279
|
-
<summary><strong>Test Hygiene</strong></summary>
|
|
280
|
-
|
|
281
|
-
| ID | Rule | Severity | Tier |
|
|
282
|
-
| ----------- | --------------------------------------------------- | -------- | ---------- |
|
|
283
|
-
| QA-TEST-001 | Focused test committed (`.only`, `fit`) | error | quarantine |
|
|
284
|
-
| QA-TEST-002 | Skipped test without justification | error | quarantine |
|
|
285
|
-
| QA-TEST-002 | Skipped test with tracked justification | warning | quarantine |
|
|
286
|
-
| QA-TEST-003 | Test with no assertions | error | quarantine |
|
|
287
|
-
| QA-TEST-004 | Hard sleep (`waitForTimeout`, `sleep()`, `delay()`) | warning | extended |
|
|
288
|
-
| QA-TEST-006 | Retry abuse hiding flakiness | warning | quarantine |
|
|
289
|
-
| QA-TEST-010 | Empty test body | error | quarantine |
|
|
290
|
-
|
|
291
|
-
</details>
|
|
292
|
-
|
|
293
|
-
<details>
|
|
294
|
-
<summary><strong>Test Quality</strong></summary>
|
|
295
|
-
|
|
296
|
-
| ID | Rule | Severity | Tier |
|
|
297
|
-
| ------------ | --------------------------- | -------- | ---------- |
|
|
298
|
-
| QA-TQUAL-001 | Mock-only verification | info | quarantine |
|
|
299
|
-
| QA-TQUAL-002 | Tautological assertion | error | quarantine |
|
|
300
|
-
| QA-TQUAL-009 | Unawaited promise assertion | error | quarantine |
|
|
301
|
-
| QA-TQUAL-011 | Commented-out tests | warning | extended |
|
|
302
|
-
|
|
303
|
-
</details>
|
|
304
|
-
|
|
305
|
-
<details>
|
|
306
|
-
<summary><strong>Playwright 🎭</strong></summary>
|
|
307
|
-
|
|
308
|
-
| ID | Rule | Severity | Tier |
|
|
309
|
-
| --------- | ---------------------------------------- | -------- | ---------- |
|
|
310
|
-
| QA-PW-002 | Unawaited locator assertion | error | core |
|
|
311
|
-
| QA-PW-003 | `page.pause()` / `test.only()` committed | error | core |
|
|
312
|
-
| QA-PW-004 | Brittle CSS/XPath selectors | warning | quarantine |
|
|
313
|
-
| QA-PW-005 | Business logic inside `page.evaluate()` | info | quarantine |
|
|
314
|
-
| QA-PW-114 | Legacy element handles (`page.$`) | info | quarantine |
|
|
315
|
-
| QA-PW-118 | `networkidle` waits (flaky by design) | info | quarantine |
|
|
316
|
-
| QA-PW-123 | Hardcoded environment URLs | warning | quarantine |
|
|
317
|
-
|
|
318
|
-
</details>
|
|
319
|
-
|
|
320
|
-
<details>
|
|
321
|
-
<summary><strong>CI Integrity</strong></summary>
|
|
322
|
-
|
|
323
|
-
| ID | Rule | Severity | Tier |
|
|
324
|
-
| --------- | ----------------------------------------------------------------- | -------- | ---------- |
|
|
325
|
-
| QA-CI-001 | `continue-on-error` masks failures | error | quarantine |
|
|
326
|
-
| QA-CI-002 | `\|\| true` swallows exit codes | error | extended |
|
|
327
|
-
| QA-CI-005 | Report consumed but never generated | error | quarantine |
|
|
328
|
-
| QA-CI-007 | Retry wrappers around tests | warning | extended |
|
|
329
|
-
| QA-CI-008 | Always-success step masks failures | error | quarantine |
|
|
330
|
-
| QA-CI-009 | Test exit code not propagated (`\|` without pipefail, `;` chains) | error | extended |
|
|
331
|
-
| QA-CI-010 | Tests skipped where they must block (skip-on-PR guards) | error | quarantine |
|
|
332
|
-
|
|
333
|
-
</details>
|
|
334
|
-
|
|
335
|
-
<details>
|
|
336
|
-
<summary><strong>Python / pytest 🐍</strong></summary>
|
|
337
|
-
|
|
338
|
-
| ID | Rule | Severity | Tier |
|
|
339
|
-
| --------- | ----------------------------------------- | -------- | ---------- |
|
|
340
|
-
| QA-PY-002 | Skipped test (`skip`, non-strict `xfail`) | warning | core |
|
|
341
|
-
| QA-PY-003 | Test function with no assertions | error | quarantine |
|
|
342
|
-
| QA-PY-005 | `time.sleep()` in tests | warning | extended |
|
|
343
|
-
| QA-PY-006 | Empty test body (`pass`) | info | quarantine |
|
|
344
|
-
| QA-PY-010 | Random/time dependence without freeze | info | quarantine |
|
|
345
|
-
| QA-PY-012 | Tautological assertion | error | quarantine |
|
|
346
|
-
|
|
347
|
-
20 Python rules total (QA-PY-001…012 pytest hygiene + QA-PY-101…108 Playwright-Python).
|
|
348
|
-
|
|
349
|
-
</details>
|
|
350
|
-
|
|
351
|
-
<details>
|
|
352
|
-
<summary><strong>Java / JUnit · TestNG ☕</strong></summary>
|
|
353
|
-
|
|
354
|
-
| ID | Rule | Severity | Tier |
|
|
355
|
-
| --------- | ---------------------------------------- | -------- | ---------- |
|
|
356
|
-
| QA-JV-101 | Disabled test (`@Disabled`) | warning | core |
|
|
357
|
-
| QA-JV-102 | Hard sleep (`Thread.sleep()`) | warning | extended |
|
|
358
|
-
| QA-JV-103 | Test method with no assertions | error | extended |
|
|
359
|
-
| QA-JV-105 | Playwright `waitForTimeout()` hard sleep | warning | core |
|
|
360
|
-
| QA-JV-106 | Brittle selector instead of role locator | warning | quarantine |
|
|
361
|
-
| QA-JV-108 | Hardcoded environment URL in test | info | quarantine |
|
|
362
|
-
| QA-JV-111 | Blanket `page.route("**")` mock | info | quarantine |
|
|
363
|
-
|
|
364
|
-
</details>
|
|
242
|
+
## What Mjölnir finds
|
|
243
|
+
|
|
244
|
+
**<!-- census:total-rules -->99 rules<!-- /census:total-rules -->** in four families — **test hygiene**, **test quality**,
|
|
245
|
+
**Playwright**, **CI integrity** — over TypeScript/JavaScript, Python,
|
|
246
|
+
Java, C# and GitHub Actions YAML, covering Playwright in all four bindings
|
|
247
|
+
plus pytest, JUnit, TestNG, NUnit, xUnit, MSTest, Jest, Vitest and Mocha,
|
|
248
|
+
with starter coverage for Cypress and Selenium. Ten of them, so the shape
|
|
249
|
+
is clear:
|
|
250
|
+
|
|
251
|
+
| ID | Rule | Severity | Tier |
|
|
252
|
+
| ------------ | ----------------------------------------------------------------- | -------- | ---------- |
|
|
253
|
+
| QA-CI-001 | `continue-on-error` masks a failing verification gate | error | quarantine |
|
|
254
|
+
| QA-CI-009 | Test exit code not propagated (`\|` without pipefail, `;` chains) | error | extended |
|
|
255
|
+
| QA-TEST-001 | Focused test committed (`.only`, `fit`) | error | quarantine |
|
|
256
|
+
| QA-TEST-003 | Test with no assertions | error | quarantine |
|
|
257
|
+
| QA-TQUAL-009 | Unawaited promise assertion | error | quarantine |
|
|
258
|
+
| QA-PW-002 | Unawaited locator assertion | error | core |
|
|
259
|
+
| QA-PW-004 | Brittle CSS/XPath selectors | warning | quarantine |
|
|
260
|
+
| QA-PW-118 | `networkidle` waits (flaky by design) | info | quarantine |
|
|
261
|
+
| QA-PY-002 | Skipped test (`skip`, non-strict `xfail`) | warning | core |
|
|
262
|
+
| QA-CS-103 | Test method with no assertions | error | core |
|
|
263
|
+
|
|
264
|
+
The full catalog is generated from the registry, never hand-maintained:
|
|
265
|
+
`mjolnir rules --md`, [`docs/rules/`](docs/rules/), or the
|
|
266
|
+
[what-it-checks guide](https://sergey-bar.github.io/Mjolnir/guide/what-it-checks).
|
|
365
267
|
|
|
366
268
|
<details>
|
|
367
|
-
<summary><strong>
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
|
373
|
-
|
|
|
374
|
-
| QA-
|
|
375
|
-
| QA-
|
|
376
|
-
| QA-
|
|
377
|
-
| QA-
|
|
269
|
+
<summary><strong>Every rule named in this README, in one table</strong> — the other 53 live in <code>mjolnir rules --md</code></summary>
|
|
270
|
+
|
|
271
|
+
> `quarantine` rules run only under `--strict` and never gate (capped to
|
|
272
|
+
> info); the severity shown is the authored severity.
|
|
273
|
+
|
|
274
|
+
| ID | Family | Rule | Severity | Tier |
|
|
275
|
+
| ------------ | ---------- | ------------------------------------------------------------------- | ----------------------------- | ------------------------- |
|
|
276
|
+
| QA-TEST-001 | Hygiene | Focused test committed (`.only`, `fit`) | error | quarantine |
|
|
277
|
+
| QA-TEST-002 | Hygiene | Skipped test — escalates to `error` without a tracked justification | warning | quarantine |
|
|
278
|
+
| QA-TEST-003 | Hygiene | Test with no assertions | error | quarantine |
|
|
279
|
+
| QA-TEST-004 | Hygiene | Hard sleep (`waitForTimeout`, `sleep()`, `delay()`) | warning | extended |
|
|
280
|
+
| QA-TEST-006 | Hygiene | Retry abuse hiding flakiness | warning | quarantine |
|
|
281
|
+
| QA-TEST-010 | Hygiene | Empty test body | error | quarantine |
|
|
282
|
+
| QA-TQUAL-001 | Quality | Mock-only verification | info | quarantine |
|
|
283
|
+
| QA-TQUAL-002 | Quality | Tautological assertion | error | quarantine |
|
|
284
|
+
| QA-TQUAL-009 | Quality | Unawaited promise assertion | error | quarantine |
|
|
285
|
+
| QA-TQUAL-011 | Quality | Commented-out tests | warning | extended |
|
|
286
|
+
| QA-PW-002 | Playwright | Unawaited locator assertion | error | core |
|
|
287
|
+
| QA-PW-003 | Playwright | `page.pause()` / `test.only()` committed | error | core |
|
|
288
|
+
| QA-PW-004 | Playwright | Brittle CSS/XPath selectors | warning | quarantine |
|
|
289
|
+
| QA-PW-005 | Playwright | Business logic inside `page.evaluate()` | info | quarantine |
|
|
290
|
+
| QA-PW-107 | Playwright | `toBeVisible` where `toBeInViewport` fits better | info | quarantine |
|
|
291
|
+
| QA-PW-114 | Playwright | Legacy element handles (`page.$`) | info | quarantine |
|
|
292
|
+
| QA-PW-118 | Playwright | `networkidle` waits (flaky by design) | info | quarantine |
|
|
293
|
+
| QA-PW-123 | Playwright | Hardcoded environment URLs | warning | quarantine |
|
|
294
|
+
| QA-PW-140 | Playwright | Screenshot without `maxDiffPixelRatio` | warning | core |
|
|
295
|
+
| QA-CI-001 | CI | `continue-on-error` masks a failing gate | error | quarantine |
|
|
296
|
+
| QA-CI-002 | CI | ` | | true` swallows exit codes | error | extended |
|
|
297
|
+
| QA-CI-005 | CI | Report consumed but never generated | error | quarantine |
|
|
298
|
+
| QA-CI-007 | CI | Retry wrappers around tests | warning | extended |
|
|
299
|
+
| QA-CI-008 | CI | Always-success step masks failures | error | quarantine |
|
|
300
|
+
| QA-CI-009 | CI | Exit code not propagated (` | `without pipefail,`;` chains) | error | extended |
|
|
301
|
+
| QA-CI-010 | CI | Tests skipped where they must block | error | quarantine |
|
|
302
|
+
| QA-PY-002 | Python | Skipped test (`skip`, non-strict `xfail`) | warning | core |
|
|
303
|
+
| QA-PY-003 | Python | Test function with no assertions | error | quarantine |
|
|
304
|
+
| QA-PY-005 | Python | `time.sleep()` in tests | warning | extended |
|
|
305
|
+
| QA-PY-006 | Python | Empty test body (`pass`) | info | quarantine |
|
|
306
|
+
| QA-PY-010 | Python | Random/time dependence without freeze | info | quarantine |
|
|
307
|
+
| QA-PY-012 | Python | Tautological assertion | error | quarantine |
|
|
308
|
+
| QA-JV-101 | Java | Disabled test (`@Disabled`) | warning | core |
|
|
309
|
+
| QA-JV-102 | Java | Hard sleep (`Thread.sleep()`) | warning | extended |
|
|
310
|
+
| QA-JV-103 | Java | Test method with no assertions | error | extended |
|
|
311
|
+
| QA-JV-105 | Java | Playwright `waitForTimeout()` hard sleep | warning | core |
|
|
312
|
+
| QA-JV-106 | Java | Brittle selector instead of role locator | warning | quarantine |
|
|
313
|
+
| QA-JV-108 | Java | Hardcoded environment URL in test | info | quarantine |
|
|
314
|
+
| QA-JV-111 | Java | Blanket `page.route("**")` mock | info | quarantine |
|
|
315
|
+
| QA-CS-101 | C# | Skipped test (`[Ignore]`, `[Fact(Skip=)]`) | warning | core |
|
|
316
|
+
| QA-CS-102 | C# | Hard sleep (`Thread.Sleep` / `Task.Delay`) | warning | core |
|
|
317
|
+
| QA-CS-103 | C# | Test method with no assertions | error | core |
|
|
318
|
+
| QA-CS-105 | C# | `WaitForTimeoutAsync()` hard sleep | warning | extended |
|
|
319
|
+
| QA-CS-106 | C# | Brittle selector instead of role locator | warning | quarantine |
|
|
320
|
+
| QA-CS-108 | C# | Hardcoded environment URL in test | info | quarantine |
|
|
321
|
+
| QA-CS-111 | C# | Blanket `page.RouteAsync("**")` mock | info | quarantine |
|
|
322
|
+
|
|
323
|
+
Python also ships QA-PY-001…012 (pytest hygiene) and QA-PY-101…108
|
|
324
|
+
(Playwright-Python); Cypress and Selenium have starter sets of three
|
|
325
|
+
rules each.
|
|
378
326
|
|
|
379
327
|
</details>
|
|
380
328
|
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
> mjolnir rules --md
|
|
386
|
-
> ```
|
|
387
|
-
>
|
|
388
|
-
> Per-rule pages live under [`docs/rules/`](docs/rules/).
|
|
329
|
+
Every rule ships with a must-fire **and** a must-not-fire fixture; a rule
|
|
330
|
+
that fires on its own negative fixture cannot ship. That is the
|
|
331
|
+
false-positive firewall, and `mjolnir doctor` enforces it in this
|
|
332
|
+
repository's own CI.
|
|
389
333
|
|
|
390
334
|
### Selector Health Score
|
|
391
335
|
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
a structural accident (CSS chains, XPath) — and scores the file 0–100:
|
|
336
|
+
`mjolnir doctor:playwright` grades every locator by how it finds an
|
|
337
|
+
element — the way a user identifies it (role, label, text), an explicit
|
|
338
|
+
contract (`data-testid`), or a structural accident (CSS chains, XPath) —
|
|
339
|
+
and scores the file 0–100:
|
|
397
340
|
|
|
398
341
|
```text
|
|
399
342
|
▚ SELECTOR HEALTH
|
|
@@ -407,108 +350,156 @@ e2e/checkout.spec.ts
|
|
|
407
350
|
role/text: 3 · testid: 1 · css-chains: 1 ⚠ · xpath: 0
|
|
408
351
|
```
|
|
409
352
|
|
|
410
|
-
|
|
411
|
-
|
|
353
|
+
This is **resilience, not correctness**.
|
|
354
|
+
`.btn.btn-primary > div:nth-child(2)` passes today and keeps passing until
|
|
355
|
+
someone touches the markup. A low score never claims the test is broken —
|
|
356
|
+
only that its future depends on markup nobody promised to keep.
|
|
412
357
|
|
|
413
|
-
|
|
358
|
+
---
|
|
414
359
|
|
|
415
|
-
|
|
416
|
-
[docs/FP-AUDIT.md](docs/FP-AUDIT.md)). The other 21 ship on the author's
|
|
417
|
-
estimate. Every scan footer tells you how many of the rules that _fired_
|
|
418
|
-
are measured; `mjolnir rules --unmeasured` lists the ones that aren't;
|
|
419
|
-
every rule's `mjolnir explain` page states its status. We publish the rate
|
|
420
|
-
even when it's ugly — QA-PW-107 audits at 95% and is quarantined for it.
|
|
421
|
-
Growing that number is the project's continuing work.
|
|
360
|
+
## The Worthiness Score
|
|
422
361
|
|
|
423
|
-
|
|
362
|
+
<table>
|
|
363
|
+
<tr>
|
|
364
|
+
<td width="50%" align="center" valign="bottom">
|
|
365
|
+
<img src="assets/readme/score-gauge.svg" alt="The hammer sweeping every score from 0 to 100 — cracked below 50 (UNWORTHY), strained 50-79 (NEEDS WORK), charged 80-99 (WORTHY), forged at 100 (FORGED) — then holding on FORGED before it loops" width="355" height="430" />
|
|
366
|
+
</td>
|
|
367
|
+
<td width="50%" align="center" valign="bottom">
|
|
368
|
+
<img src="assets/readme/terminal-hero.svg" alt="Mjölnir's deduction breakdown — WORTHINESS 75/100 NEEDS WORK, a diagnostics-by-category bar chart, the per-severity deduction box, and a FIX THIS FIRST list" width="337" height="430" />
|
|
369
|
+
</td>
|
|
370
|
+
</tr>
|
|
371
|
+
<tr>
|
|
372
|
+
<td align="center"><strong>What the score means</strong></td>
|
|
373
|
+
<td align="center"><strong>Where the points went</strong></td>
|
|
374
|
+
</tr>
|
|
375
|
+
</table>
|
|
424
376
|
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
[
|
|
377
|
+
<sub>Left: every score 0–100 through the real `deriveScoreState`. Right: a
|
|
378
|
+
real strict scan of `examples/demo-repo`. Both generated
|
|
379
|
+
(`npm run docs:gauge` · `npm run docs:hero`) and drift-locked
|
|
380
|
+
([gauge](tests/contract/score-gauge-asset-reproducibility.spec.ts) ·
|
|
381
|
+
[breakdown](tests/contract/hero-asset-reproducibility.spec.ts)).</sub>
|
|
429
382
|
|
|
430
|
-
|
|
383
|
+
| Score | Verdict |
|
|
384
|
+
| --------- | ---------------------------------------- |
|
|
385
|
+
| `0 – 49` | **UNWORTHY** |
|
|
386
|
+
| `50 – 79` | **NEEDS WORK** |
|
|
387
|
+
| `80 – 99` | **WORTHY** |
|
|
388
|
+
| `100` | **FORGED** |
|
|
389
|
+
| `null` | **UNKNOWN** — no test declarations found |
|
|
390
|
+
|
|
391
|
+
**How it is computed.** Severity sets a base deduction — `error −8`,
|
|
392
|
+
`warning −3`, `info −1` — which the evidence level then discounts: E2 pays
|
|
393
|
+
full, E1 half (rounded down), E0 nothing. The total is normalized by suite
|
|
394
|
+
exposure (deductions per test declaration, not per file), and the terminal
|
|
395
|
+
prints the same discounted numbers the score used. No hidden second model:
|
|
396
|
+
[docs/SCORING.md](docs/SCORING.md) ·
|
|
397
|
+
[scoring guide](https://sergey-bar.github.io/Mjolnir/guide/scoring).
|
|
431
398
|
|
|
432
|
-
|
|
433
|
-
|
|
399
|
+
**What 100 does not mean.** Not that the software is correct, the suite
|
|
400
|
+
adequate, or the product free of defects. Exactly one thing: **none of
|
|
401
|
+
Mjölnir's evaluated rules produced a deduction under this scan and this
|
|
402
|
+
evidence model.**
|
|
434
403
|
|
|
435
|
-
|
|
436
|
-
it at one, a run report. A clean scan is not a passing suite.
|
|
437
|
-
- **It cannot tell you an assertion is _wrong_.** `expect(total).toBe(41)`
|
|
438
|
-
is a perfectly healthy-looking test. Mjölnir finds tests that can't fail
|
|
439
|
-
and pipelines that can't go red — not tests that check the wrong thing.
|
|
440
|
-
- **A 100 is not proof of a good suite.** It means none of these 99 rules
|
|
441
|
-
fired. Coverage of your actual risk is a different question, and this
|
|
442
|
-
tool does not pretend to answer it.
|
|
443
|
-
- **21 of 99 rules ship on an estimate**, not a measured rate — and they
|
|
444
|
-
say so, per rule, in `mjolnir explain`.
|
|
445
|
-
- **E1 findings are heuristics.** They are positioned to be worth reading,
|
|
446
|
-
not to be applied blindly; the evidence level is attached to every
|
|
447
|
-
finding precisely so you can tell the difference.
|
|
448
|
-
- **An empty repo scores `null`, never 100.** "Unknown" is a verdict here.
|
|
404
|
+
---
|
|
449
405
|
|
|
450
|
-
|
|
406
|
+
## The evidence model
|
|
451
407
|
|
|
452
|
-
Every
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
reports nothing is exactly the false green this project exists to catch.
|
|
456
|
-
The result is uploaded as a build artifact on every run.
|
|
408
|
+
Every finding carries the strength of the evidence behind it. This is the
|
|
409
|
+
difference between a tool that reports patterns and a tool you can gate a
|
|
410
|
+
release on.
|
|
457
411
|
|
|
458
|
-
|
|
412
|
+
```text
|
|
413
|
+
STATIC SIGNAL → EVIDENCE LEVEL → RUNTIME CORROBORATION → TRUST DECISION
|
|
414
|
+
```
|
|
459
415
|
|
|
460
|
-
|
|
416
|
+
| Level | Name | Means | Deduction |
|
|
417
|
+
| ------ | ------------------- | --------------------------------------------------------- | --------- |
|
|
418
|
+
| **E2** | Deterministic proof | The defect is structurally present in the code as written | Full |
|
|
419
|
+
| **E1** | Pattern evidence | A pattern strongly associated with the defect was matched | Half |
|
|
420
|
+
| **E0** | Observation | Worth knowing; not a claim that anything is wrong | Zero |
|
|
461
421
|
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
422
|
+
Confidence in a detection is not strength of proof: a rule can be certain
|
|
423
|
+
it matched what it looked for and still be looking at a heuristic. So E1
|
|
424
|
+
findings are positioned to be read and judged, never applied blindly — and
|
|
425
|
+
that boundary is stamped on the finding in the terminal, the JSON and the
|
|
426
|
+
agent handoff.
|
|
465
427
|
|
|
466
|
-
|
|
428
|
+
Runtime evidence raises the ceiling. Given a real run report, a finding
|
|
429
|
+
climbs a six-rung **trust ladder** from `L0` (observation) to `L5` (the run
|
|
430
|
+
verdict corroborates the defect class); the top three rungs structurally
|
|
431
|
+
require runtime evidence, so a static-only finding can never claim them.
|
|
432
|
+
Rung by rung: [docs/TERMINOLOGY.md](docs/TERMINOLOGY.md).
|
|
467
433
|
|
|
468
|
-
|
|
434
|
+
### How much of this is measured
|
|
469
435
|
|
|
470
|
-
|
|
471
|
-
|
|
436
|
+
**<!-- census:measured-of-total -->78 of 99<!-- /census:measured-of-total --> rules carry a false-positive rate measured against real OSS code**
|
|
437
|
+
(≥ 10 hand-classified findings each — [docs/FP-AUDIT.md](docs/FP-AUDIT.md)).
|
|
438
|
+
The other <!-- census:unmeasured -->21<!-- /census:unmeasured --> ship on the author's estimate and say so, per rule, in
|
|
439
|
+
`mjolnir explain`; `mjolnir rules --unmeasured` lists them, and every scan
|
|
440
|
+
footer reports how many of the rules that actually _fired_ are measured.
|
|
472
441
|
|
|
473
|
-
|
|
442
|
+
The rate is published even when unflattering: QA-PW-107 audits at 95% and
|
|
443
|
+
is quarantined for it. **Mjölnir measures its own uncertainty** — that is
|
|
444
|
+
the product, not a caveat.
|
|
474
445
|
|
|
475
|
-
|
|
446
|
+
### Trust tiers
|
|
476
447
|
|
|
477
|
-
|
|
478
|
-
</tr>
|
|
479
|
-
</table>
|
|
448
|
+
Tiers follow measured false-positive behavior, not opinion:
|
|
480
449
|
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
[breakdown](tests/contract/hero-asset-reproducibility.spec.ts)).</sub>
|
|
450
|
+
| Tier | Measured FP | Behavior |
|
|
451
|
+
| -------------- | ----------- | --------------------------------------------- |
|
|
452
|
+
| **core** | ≤ 10% | Default report, gates |
|
|
453
|
+
| **extended** | ≤ 30% | Default report, lower confidence |
|
|
454
|
+
| **quarantine** | > 30% | `--strict` only, capped to info — never gates |
|
|
455
|
+
| _unmeasured_ | n < 10 | Cannot be promoted to core until measured |
|
|
488
456
|
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
deductions mean weak signals cost less. The terminal shows the same
|
|
492
|
-
discounted numbers the score uses — no black box.
|
|
457
|
+
Promotion and demotion rules, plus per-language maturity:
|
|
458
|
+
[rule lifecycle](https://sergey-bar.github.io/Mjolnir/reference/rule-lifecycle).
|
|
493
459
|
|
|
494
|
-
|
|
495
|
-
(deterministic defect, full deduction), **≥ 80/WORTHY** and **50–79/NEEDS
|
|
496
|
-
WORK** findings are mostly **E1** (heuristic pattern, half deduction), and
|
|
497
|
-
**E0** (observation) findings cost nothing — info only. The tagline "we
|
|
498
|
-
prove it" refers to this system: E2 findings are structural proof; E1
|
|
499
|
-
findings are correctly-positioned warnings, not formal proofs.
|
|
460
|
+
---
|
|
500
461
|
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
462
|
+
## Why this is not a linter
|
|
463
|
+
|
|
464
|
+
Linters tell you whether code follows rules. Mjölnir tells you whether your
|
|
465
|
+
verification can be trusted.
|
|
466
|
+
|
|
467
|
+
| | Linters (ESLint, SonarQube) | Coverage tools | AI code review | **Mjölnir** |
|
|
468
|
+
| -------------------------------------------------------- | :-------------------------: | :------------: | :------------: | :--------------: |
|
|
469
|
+
| Scores the **verification system**, not the product code | ❌ | ❌ | ❌ | ✅ |
|
|
470
|
+
| CI workflow integrity (`continue-on-error`, `\|\| true`) | ❌ | ❌ | only the diff | ✅ |
|
|
471
|
+
| Grades Playwright locator resilience (Selector Health) | ❌ | ❌ | ❌ | ✅ |
|
|
472
|
+
| Reads real run data for `TRUE-FLAKE` verdicts | ❌ | ❌ | ❌ | ✅ |
|
|
473
|
+
| Publishes a measured false-positive rate per rule | ❌ | ❌ | ❌ | ✅ |
|
|
474
|
+
| Flags tests with no assertions | ✅\* | ❌ | sometimes | ✅ |
|
|
475
|
+
| Catches hard sleeps (`waitForTimeout`, `time.sleep`) | ✅\* | ❌ | sometimes | ✅ |
|
|
476
|
+
| Deterministic (same input → same output) | ✅ | ✅ | ❌ | ✅ |
|
|
477
|
+
| Cost per scan | free | free | tokens | **zero** (local) |
|
|
478
|
+
|
|
479
|
+
<sub>\*Covered by `eslint-plugin-jest` / `eslint-plugin-playwright`
|
|
480
|
+
(`expect-expect`, `no-wait-for-timeout`) and by SonarQube's own assertion
|
|
481
|
+
rules. Columns describe default behavior aimed at test-suite verification;
|
|
482
|
+
plugins, paid tiers and custom rules change some answers. A positioning
|
|
483
|
+
summary, not a benchmark.</sub>
|
|
484
|
+
|
|
485
|
+
**Use AI review too.** It catches nuance, intent and design flaws no regex
|
|
486
|
+
can find. Mjölnir catches what AI overlooks because it looks intentional —
|
|
487
|
+
a committed `.only`, a swallowed exit code, a `continue-on-error` on a test
|
|
488
|
+
job. Those are not defects that need reasoning; they are facts that need
|
|
489
|
+
scanning.
|
|
505
490
|
|
|
506
491
|
---
|
|
507
492
|
|
|
508
|
-
##
|
|
493
|
+
## Runtime forensics
|
|
509
494
|
|
|
510
|
-
Static
|
|
511
|
-
|
|
495
|
+
Static analysis reasons about code that was never run. Forensics reads what
|
|
496
|
+
actually happened — Playwright JSON reports and JUnit XML from any runner:
|
|
497
|
+
|
|
498
|
+
```text
|
|
499
|
+
Static analysis → what the code appears to do
|
|
500
|
+
Runtime evidence → what the run actually did
|
|
501
|
+
both → a finding that can climb the trust ladder
|
|
502
|
+
```
|
|
512
503
|
|
|
513
504
|
```bash
|
|
514
505
|
mjolnir forensics ./test-results/
|
|
@@ -525,14 +516,23 @@ FAILING declines an expired card (e2e/checkout.spec.ts)
|
|
|
525
516
|
████░░░░░░░░░░░░░░░░ 1.1s · 1 attempt
|
|
526
517
|
```
|
|
527
518
|
|
|
528
|
-
|
|
529
|
-
|
|
519
|
+
`TRUE-FLAKE` is not "this test retried". It is precise: the test **failed
|
|
520
|
+
at least one attempt and then finished green** — a lucky pass, flagged
|
|
521
|
+
regardless of the final checkmark. `mjolnir triage` turns that history into
|
|
522
|
+
a quarantine proposal; `mjolnir pw-report` summarizes a run.
|
|
530
523
|
|
|
531
524
|
---
|
|
532
525
|
|
|
533
|
-
##
|
|
526
|
+
## CI integrity
|
|
527
|
+
|
|
528
|
+
A test can pass while the pipeline around it is incapable of failing.
|
|
529
|
+
Mjölnir reads the workflows too — `continue-on-error`, `|| true`,
|
|
530
|
+
unpropagated exit codes, always-success steps, reports consumed but never
|
|
531
|
+
generated, and gates skipped on the very events that should block. Each
|
|
532
|
+
finding names the job, the step and the line, and carries its own evidence
|
|
533
|
+
level; none of them is a claim about CI in general.
|
|
534
534
|
|
|
535
|
-
One command generates
|
|
535
|
+
One command generates the PR workflow — advisory by default:
|
|
536
536
|
|
|
537
537
|
```bash
|
|
538
538
|
mjolnir ci install
|
|
@@ -547,33 +547,47 @@ Or wire it into GitHub Code Scanning natively via SARIF:
|
|
|
547
547
|
sarif_file: mjolnir.sarif
|
|
548
548
|
```
|
|
549
549
|
|
|
550
|
-
Editor and pipeline setup
|
|
550
|
+
Editor and pipeline setup: [docs/SARIF-INTEGRATION.md](docs/SARIF-INTEGRATION.md).
|
|
551
|
+
|
|
552
|
+
### Changed-scope attribution
|
|
553
|
+
|
|
554
|
+
```bash
|
|
555
|
+
npx mjolnir-qa@latest --scope changed
|
|
556
|
+
```
|
|
551
557
|
|
|
552
|
-
|
|
558
|
+
Findings are attributed to the lines your branch added, against the
|
|
559
|
+
**merge-base**. The scope is the same file set a full scan discovers —
|
|
560
|
+
TS/JS specs and adapter configs, `test_*.py`, `*Test.java`, `*Tests.cs`,
|
|
561
|
+
`.github/workflows/*.yml` — plus uncommitted and untracked working-tree
|
|
562
|
+
changes, so it works before you commit. The base resolves
|
|
563
|
+
`main → master → origin/main → origin/master → origin/HEAD`; override with
|
|
564
|
+
`--base <ref>`.
|
|
553
565
|
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
target, different default branch — it degrades honestly: findings fall
|
|
559
|
-
back to full-file attribution and the report says so. Override the base
|
|
560
|
-
ref with `--base <ref>`.
|
|
566
|
+
When the merge-base cannot be resolved — shallow clone, detached HEAD,
|
|
567
|
+
non-git target — findings fall back to full-file attribution **and the
|
|
568
|
+
report says that it did.** A silent fallback would be the same class of
|
|
569
|
+
defect this tool exists to catch.
|
|
561
570
|
|
|
562
571
|
---
|
|
563
572
|
|
|
564
|
-
##
|
|
573
|
+
## AI agents
|
|
574
|
+
|
|
575
|
+
Findings are only worth something if something acts on them.
|
|
576
|
+
|
|
577
|
+
```text
|
|
578
|
+
SCAN → EVIDENCE → HANDOFF → AGENT → RE-SCAN → PROOF
|
|
579
|
+
```
|
|
565
580
|
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
terminal output and hope".
|
|
581
|
+
**AI writes the fix. Mjölnir verifies the fix.** Proof comes from the
|
|
582
|
+
re-scan, never from the agent's own report of success.
|
|
569
583
|
|
|
570
|
-
| Command | What the agent gets
|
|
571
|
-
| ----------------- |
|
|
572
|
-
| `mjolnir mcp` |
|
|
573
|
-
| `mjolnir handoff` | A saved `--json` report becomes a deterministic Markdown
|
|
574
|
-
| `mjolnir install` | Writes the agent
|
|
584
|
+
| Command | What the agent gets |
|
|
585
|
+
| ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
586
|
+
| `mjolnir mcp` | An [MCP](https://modelcontextprotocol.io) server over stdio — `scan`, `explain` and `diff` become callable tools. |
|
|
587
|
+
| `mjolnir handoff` | A saved `--json` report becomes a deterministic Markdown plan: what was detected, the evidence boundary per finding, what must **not** change, how to verify. |
|
|
588
|
+
| `mjolnir install` | Writes into the agent surfaces your repo already has — `.claude/`, `.cursor/`, `.kilo/`, `AGENTS.md` — so it re-scans before claiming it is done. |
|
|
575
589
|
|
|
576
|
-
Add
|
|
590
|
+
Add it to a client that ships its own CLI:
|
|
577
591
|
|
|
578
592
|
```bash
|
|
579
593
|
claude mcp add mjolnir -- npx -y mjolnir-qa@latest mcp
|
|
@@ -589,24 +603,72 @@ Or to any client that takes an `mcpServers` block:
|
|
|
589
603
|
}
|
|
590
604
|
```
|
|
591
605
|
|
|
592
|
-
Everything a machine consumes — MCP tool results, `--json`, SARIF — comes
|
|
593
|
-
off one canonical result under a versioned, additive-only schema, so a
|
|
594
|
-
consumer never has to reconstruct semantics for itself:
|
|
595
|
-
[the machine contract](docs/machine-contract.md) (`contractVersion: 1`).
|
|
596
|
-
|
|
597
606
|
**The guardrail matters more than the convenience.** Every finding in a
|
|
598
|
-
handoff carries its
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
607
|
+
handoff carries its boundary: **E2** — _deterministic, check the location
|
|
608
|
+
and apply the fix_; **E1** — _REQUIRES CONFIRMATION, the observation alone
|
|
609
|
+
does not prove the defect_. An agent that fixes E1 blindly, suppresses a
|
|
610
|
+
rule, or edits a rule to raise the score is doing the exact thing this tool
|
|
611
|
+
exists to catch — so the artifact says so, in the prompt, next to the
|
|
612
|
+
finding.
|
|
613
|
+
|
|
614
|
+
---
|
|
615
|
+
|
|
616
|
+
## Trust and security
|
|
617
|
+
|
|
618
|
+
**Local-first, zero telemetry.** No network-capable API — `fetch`, `http`,
|
|
619
|
+
`https`, `net`, `dns`, `dgram`, WebSocket — exists anywhere in `src/`, and
|
|
620
|
+
[`privacy-network-isolation.spec.ts`](tests/contract/privacy-network-isolation.spec.ts)
|
|
621
|
+
fails the build if one appears (it also bars `eval` and `new Function`).
|
|
622
|
+
Scanning untrusted code never executes it: static analysis reads source
|
|
623
|
+
text, and forensics parses report files that already exist on disk.
|
|
624
|
+
|
|
625
|
+
Two caveats worth stating: `npx` itself fetches the package before
|
|
626
|
+
anything runs, and the guarantee covers `src/` — not third-party plugins.
|
|
627
|
+
|
|
628
|
+
**Plugins are not sandboxed, and this will not be dressed up.** JS plugins
|
|
629
|
+
(`mjolnir-rules/*.mjs`, or npm packages under `"plugins"`) run with full
|
|
630
|
+
Node privileges — the same trust model as ESLint or Vitest plugins. So
|
|
631
|
+
loading them is opt-in **per scan**: without `--enable-plugins` (or
|
|
632
|
+
`MJOLNIR_ENABLE_PLUGINS=1`) the sources are never loaded, and a stderr
|
|
633
|
+
notice lists what was skipped. JSON rule manifests execute no code by
|
|
634
|
+
design, and core rule-ID prefixes are reserved so a plugin cannot
|
|
635
|
+
impersonate one. Vulnerabilities: [SECURITY.md](SECURITY.md).
|
|
636
|
+
|
|
637
|
+
### We run it on ourselves
|
|
638
|
+
|
|
639
|
+
A verification trust engine has no standing unless it is itself verifiable.
|
|
640
|
+
Every CI run scans this repository **with the build that same run
|
|
641
|
+
produced**, and the gate fails on any error-severity finding — but also on
|
|
642
|
+
a **partial** scan or a **crashed rule**, because a truncated self-scan
|
|
643
|
+
that reports nothing is precisely the false green this project exists to
|
|
644
|
+
catch. `mjolnir doctor` re-audits the rule base in the same run (fixture
|
|
645
|
+
firewall, tier honesty, the core-tier cap), where an INCONCLUSIVE check
|
|
646
|
+
fails exactly like a failing one. Both reports are uploaded as build
|
|
647
|
+
artifacts.
|
|
648
|
+
|
|
649
|
+
---
|
|
650
|
+
|
|
651
|
+
## What Mjölnir cannot tell you
|
|
652
|
+
|
|
653
|
+
- **It does not run your tests.** A clean scan is not a passing suite.
|
|
654
|
+
- **It cannot tell you an assertion is _wrong_.** `expect(total).toBe(41)`
|
|
655
|
+
looks perfectly healthy. Mjölnir finds tests that _cannot fail_ and
|
|
656
|
+
pipelines that _cannot go red_ — not tests that check the wrong thing.
|
|
657
|
+
- **It does not prove business correctness.** Nothing here says your
|
|
658
|
+
product does what the requirement asked for.
|
|
659
|
+
- **A 100 is not proof of a good suite.** Whether your suite covers your
|
|
660
|
+
actual risk is a different question, and this tool does not answer it.
|
|
661
|
+
- **<!-- census:unmeasured-of-total -->21 of 99<!-- /census:unmeasured-of-total --> rules ship on an estimate**, not a measured rate — disclosed
|
|
662
|
+
per rule, not buried here.
|
|
663
|
+
- **E1 is not E2.** Heuristic findings are worth reading, not worth
|
|
664
|
+
applying blindly.
|
|
665
|
+
- **An empty repo scores `null`, never 100.**
|
|
604
666
|
|
|
605
667
|
---
|
|
606
668
|
|
|
607
|
-
##
|
|
669
|
+
## Exit codes and the machine contract
|
|
608
670
|
|
|
609
|
-
Frozen — safe to build CI logic on:
|
|
671
|
+
Frozen surfaces — safe to build CI logic on:
|
|
610
672
|
|
|
611
673
|
| Exit code | Meaning |
|
|
612
674
|
| --------- | --------------------------------------------------------------- |
|
|
@@ -616,78 +678,75 @@ Frozen — safe to build CI logic on:
|
|
|
616
678
|
| `10` | Usage error (bad flag, missing target) |
|
|
617
679
|
| `20` | Internal error |
|
|
618
680
|
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
---
|
|
623
|
-
|
|
624
|
-
## 🔒 Trust model
|
|
625
|
-
|
|
626
|
-
**Local-first, zero telemetry, no false proof.** Scanning untrusted code
|
|
627
|
-
never executes it.
|
|
681
|
+
`2` is deliberately distinct from `0`: a scan that did not finish has not
|
|
682
|
+
found nothing — it has not finished looking.
|
|
628
683
|
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
`
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
loaded, and a stderr notice lists exactly what was skipped. JSON rule
|
|
635
|
-
manifests declare regex patterns and execute no code by design. Core
|
|
636
|
-
rule-ID prefixes are reserved, so a plugin cannot impersonate one.
|
|
637
|
-
|
|
638
|
-
Scoring math, rule lifecycle, architecture and the tree-sitter roadmap:
|
|
639
|
-
[docs/SCORING.md](docs/SCORING.md) ·
|
|
640
|
-
[docs/RULE-LIFECYCLE.md](docs/RULE-LIFECYCLE.md) ·
|
|
641
|
-
[CONTRIBUTING.md](CONTRIBUTING.md) ·
|
|
642
|
-
[docs site](https://sergey-bar.github.io/Mjolnir/).
|
|
684
|
+
Everything a machine consumes — MCP tool results, `--json`, SARIF 2.1 —
|
|
685
|
+
comes off one canonical result under a versioned, **additive-only** schema
|
|
686
|
+
(`schemaVersion: 1`, `contractVersion: 1`), so no consumer reconstructs
|
|
687
|
+
semantics from rendered text: [the machine contract](docs/machine-contract.md).
|
|
688
|
+
Rule IDs (`QA-<FAMILY>-NNN`) are immutable once shipped and never reused.
|
|
643
689
|
|
|
644
690
|
---
|
|
645
691
|
|
|
646
|
-
##
|
|
692
|
+
## Documentation
|
|
647
693
|
|
|
648
694
|
| Document | What's in it |
|
|
649
695
|
| ------------------------------------------------------ | ------------------------------------------------- |
|
|
650
696
|
| [docs/SCORING.md](docs/SCORING.md) | Score normalization + evidence weighting |
|
|
651
697
|
| [docs/TERMINOLOGY.md](docs/TERMINOLOGY.md) | Canonical vocabulary — one word per concept |
|
|
652
698
|
| [docs/FP-AUDIT.md](docs/FP-AUDIT.md) | Measured false-positive rates + method |
|
|
653
|
-
| [docs/RULE-LIFECYCLE.md](docs/RULE-LIFECYCLE.md) | Rule states, suppression, deprecation
|
|
699
|
+
| [docs/RULE-LIFECYCLE.md](docs/RULE-LIFECYCLE.md) | Rule states, tiers, suppression, deprecation |
|
|
654
700
|
| [docs/VERSIONING.md](docs/VERSIONING.md) | Semver policy, frozen surfaces, deprecation cycle |
|
|
701
|
+
| [docs/machine-contract.md](docs/machine-contract.md) | The canonical machine-readable result |
|
|
655
702
|
| [docs/SARIF-INTEGRATION.md](docs/SARIF-INTEGRATION.md) | SARIF output + editor/CI setup |
|
|
656
703
|
| [docs/rules/](docs/rules/) | Generated per-rule catalog |
|
|
657
704
|
| [CONTRIBUTING.md](CONTRIBUTING.md) | Dev setup + contribution workflow |
|
|
658
705
|
| [SUPPORT.md](SUPPORT.md) | Where to ask, report and get help |
|
|
659
|
-
| [CHANGELOG.md](CHANGELOG.md) | Release history |
|
|
660
706
|
| [SECURITY.md](SECURITY.md) | Vulnerability reporting |
|
|
707
|
+
| [CHANGELOG.md](CHANGELOG.md) | Release history |
|
|
708
|
+
|
|
709
|
+
Full docs site: <https://sergey-bar.github.io/Mjolnir/>.
|
|
661
710
|
|
|
662
711
|
---
|
|
663
712
|
|
|
664
|
-
##
|
|
713
|
+
## Status
|
|
665
714
|
|
|
666
|
-
**v0.5.x · open beta.** The JSON schema and exit codes are frozen
|
|
667
|
-
TypeScript and Python have the broadest measured coverage; Java
|
|
668
|
-
newer — read them through the
|
|
669
|
-
|
|
715
|
+
**v0.5.x · open beta.** The JSON schema and the exit codes are frozen
|
|
716
|
+
contracts. TypeScript and Python have the broadest measured coverage; Java
|
|
717
|
+
and C# are newer — read them through the
|
|
718
|
+
[maturity table](https://sergey-bar.github.io/Mjolnir/reference/rule-lifecycle).
|
|
719
|
+
Honest scope, no invented dates:
|
|
720
|
+
[the public roadmap](https://sergey-bar.github.io/Mjolnir/reference/roadmap).
|
|
670
721
|
|
|
671
722
|
---
|
|
672
723
|
|
|
673
|
-
##
|
|
724
|
+
## Contributing
|
|
674
725
|
|
|
675
|
-
New rules are the easiest first contribution
|
|
676
|
-
rule plus its must-fire **and** must-not-fire fixtures
|
|
677
|
-
intentionally fails its fixtures until
|
|
678
|
-
|
|
726
|
+
New rules are the easiest first contribution. One command scaffolds the
|
|
727
|
+
rule plus its must-fire **and** must-not-fire fixtures — and the generated
|
|
728
|
+
rule intentionally fails its own fixtures until real detection is
|
|
729
|
+
implemented, because a stub that ships is a rule nobody measured:
|
|
679
730
|
|
|
680
731
|
```bash
|
|
681
732
|
mjolnir create-rule QA-PW-140 --title "Screenshot without diff bound"
|
|
682
733
|
```
|
|
683
734
|
|
|
684
|
-
|
|
685
|
-
firewall laws are in [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
735
|
+
Dev setup, the standing-gate commands, and the anti-creep and
|
|
736
|
+
fixture-firewall laws are in [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
686
737
|
|
|
687
738
|
---
|
|
688
739
|
|
|
740
|
+
## The Mjölnir Standard
|
|
741
|
+
|
|
742
|
+
Don't ask whether the tests passed.
|
|
743
|
+
|
|
744
|
+
Ask whether the evidence proves they deserve to be trusted.
|
|
745
|
+
|
|
689
746
|
<div align="center">
|
|
690
747
|
|
|
748
|
+
---
|
|
749
|
+
|
|
691
750
|
**Stop shipping tests you can't trust.**
|
|
692
751
|
|
|
693
752
|
```bash
|
package/dist/cli.d.mts
CHANGED
|
@@ -728,7 +728,7 @@ declare const runScan: typeof runScan$1, buildUniversalRules: typeof buildUniver
|
|
|
728
728
|
* `scripts/sync-sarif-version.cjs` on release and guarded by
|
|
729
729
|
* `tests/version-consistency.spec.ts` locally.
|
|
730
730
|
*/
|
|
731
|
-
declare const CLI_VERSION = "0.5.
|
|
731
|
+
declare const CLI_VERSION = "0.5.32";
|
|
732
732
|
/** A usage-error detail: the offending token, when one exists. */
|
|
733
733
|
interface UsageErrorDetail {
|
|
734
734
|
/** The unknown flag or rejected value (e.g. `--nope`, `loud`). */
|
package/dist/cli.mjs
CHANGED
|
@@ -13290,7 +13290,7 @@ function renderSarif(result, repoRootUri) {
|
|
|
13290
13290
|
tool: { driver: {
|
|
13291
13291
|
name: "Mjölnir",
|
|
13292
13292
|
informationUri: "https://github.com/Sergey-Bar/Mjolnir",
|
|
13293
|
-
version: "0.5.
|
|
13293
|
+
version: "0.5.32",
|
|
13294
13294
|
rules: [...rules.values()].map((r) => {
|
|
13295
13295
|
const meta = RULES.find((x) => x.id === r.id);
|
|
13296
13296
|
return {
|
|
@@ -18662,7 +18662,7 @@ const { runScan, buildUniversalRules, fallbackWorkspace, pathMatchesGlob, isVali
|
|
|
18662
18662
|
* `scripts/sync-sarif-version.cjs` on release and guarded by
|
|
18663
18663
|
* `tests/version-consistency.spec.ts` locally.
|
|
18664
18664
|
*/
|
|
18665
|
-
const CLI_VERSION = "0.5.
|
|
18665
|
+
const CLI_VERSION = "0.5.32";
|
|
18666
18666
|
function parseArgs(argv, onError) {
|
|
18667
18667
|
const args = {
|
|
18668
18668
|
target: ".",
|
package/dist/mcp/stdio.mjs
CHANGED
|
@@ -13289,7 +13289,7 @@ function renderSarif(result, repoRootUri) {
|
|
|
13289
13289
|
tool: { driver: {
|
|
13290
13290
|
name: "Mjölnir",
|
|
13291
13291
|
informationUri: "https://github.com/Sergey-Bar/Mjolnir",
|
|
13292
|
-
version: "0.5.
|
|
13292
|
+
version: "0.5.32",
|
|
13293
13293
|
rules: [...rules.values()].map((r) => {
|
|
13294
13294
|
const meta = RULES.find((x) => x.id === r.id);
|
|
13295
13295
|
return {
|
|
@@ -18298,7 +18298,7 @@ const { runScan, buildUniversalRules, fallbackWorkspace, pathMatchesGlob, isVali
|
|
|
18298
18298
|
* `scripts/sync-sarif-version.cjs` on release and guarded by
|
|
18299
18299
|
* `tests/version-consistency.spec.ts` locally.
|
|
18300
18300
|
*/
|
|
18301
|
-
const CLI_VERSION = "0.5.
|
|
18301
|
+
const CLI_VERSION = "0.5.32";
|
|
18302
18302
|
function parseArgs(argv, onError) {
|
|
18303
18303
|
const args = {
|
|
18304
18304
|
target: ".",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mjolnir-qa",
|
|
3
|
-
"version": "0.5.
|
|
3
|
+
"version": "0.5.32",
|
|
4
4
|
"description": "Mjölnir — the Verification Trust Engine for QA. Audits test suites and CI pipelines, reports a worthiness score and prioritized findings.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"engines": {
|
|
@@ -36,6 +36,7 @@
|
|
|
36
36
|
"corpus:sample": "tsx scripts/corpus-sample.ts",
|
|
37
37
|
"fp-audit:generate": "tsx scripts/generate-fp-audit-table.ts",
|
|
38
38
|
"docs:rules": "tsx scripts/generate-rule-docs.ts",
|
|
39
|
+
"docs:counts": "tsx scripts/generate-counts.ts",
|
|
39
40
|
"docs:capability": "tsx scripts/generate-capability-matrix.ts",
|
|
40
41
|
"docs:machine-contract": "tsx scripts/generate-machine-contract-doc.ts",
|
|
41
42
|
"docs:hero": "tsx scripts/generate-readme-hero.ts",
|
|
@@ -53,7 +54,9 @@
|
|
|
53
54
|
"docs:translations": "node scripts/check-readme-translations.mjs",
|
|
54
55
|
"self-scan": "node dist/cli.mjs .",
|
|
55
56
|
"prepare": "husky",
|
|
56
|
-
"prepublishOnly": "npm run build"
|
|
57
|
+
"prepublishOnly": "npm run build",
|
|
58
|
+
"docs:flow": "tsx scripts/generate-readme-flow.ts",
|
|
59
|
+
"docs:architecture": "tsx scripts/generate-readme-architecture.ts"
|
|
57
60
|
},
|
|
58
61
|
"keywords": [
|
|
59
62
|
"qa",
|