canary-test-cli 7.1.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +327 -0
- package/agents/skills/canary:generate.md +49 -0
- package/agents/skills/canary:init.md +37 -0
- package/agents/skills/canary:migrate.md +66 -0
- package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
- package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
- package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
- package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
- package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
- package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
- package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
- package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
- package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
- package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
- package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
- package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
- package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
- package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
- package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
- package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
- package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
- package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
- package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
- package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
- package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
- package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
- package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
- package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
- package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
- package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
- package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
- package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
- package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
- package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
- package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
- package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
- package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
- package/agents/skills/lib/parse-args.mjs +275 -0
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +49 -72
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +27 -19
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/skill-dispatch.js +115 -0
- package/dist/engine/core/skill-examples.js +103 -3
- package/dist/engine/core/skill-registry.js +59 -4
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/test-files.js +77 -0
- package/dist/engine/core/vacuity-scanner.js +330 -15
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/analysis-emit.js +7 -2
- package/dist/engine/guardian/cli.js +277 -249
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +354 -223
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +171 -51
- package/dist/engine/workflow-cli.js +85 -65
- package/dist/reporters/testtracker.d.ts +1 -1
- package/dist/reporters/testtracker.js +1 -1
- package/package.json +3 -2
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-savant
|
|
3
|
+
description:
|
|
4
|
+
Order-dependence and isolation detector for test suites. A deterministic
|
|
5
|
+
static scan flags the shared-state smells which predict order-dependent tests
|
|
6
|
+
- a module-level mutable a test writes to, a setup with no matching teardown,
|
|
7
|
+
a mutated process singleton, an order-coupled name - with no test execution,
|
|
8
|
+
so it runs anywhere node does and on every PR. An opt-in confirming pass
|
|
9
|
+
(--confirm) shuffles the suite under a pinned seed and bisects the prefix to
|
|
10
|
+
name the polluting test. Advisory by default; pytest and vitest idioms.
|
|
11
|
+
cli: scripts/cli.mjs
|
|
12
|
+
requires: [node>=20]
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
# Canary Savant
|
|
16
|
+
|
|
17
|
+
A test that only passes because of the tests that ran before it is a lie that
|
|
18
|
+
passes CI. Savant finds shared-state leakage and names the culprit. The **static
|
|
19
|
+
pass** is the cheap half that runs on every PR and points at the _suspects_. The
|
|
20
|
+
**confirming pass** (`--confirm`, opt-in) proves a leak by shuffling the suite
|
|
21
|
+
under a pinned seed and bisecting the prefix to name the polluter — not just the
|
|
22
|
+
victim.
|
|
23
|
+
|
|
24
|
+
Tier-0 by the definition in
|
|
25
|
+
[ADR 0015](../../../../docs/knowledge/decisions/0015-skill-capability-vocabulary.md):
|
|
26
|
+
deterministic, no network, no agent or LLM. Also no secrets and no dependency on
|
|
27
|
+
any other skill. Both passes qualify — `--confirm` runs the suite locally, which
|
|
28
|
+
needs no network and no model.
|
|
29
|
+
|
|
30
|
+
## Rules (static pass — suspects)
|
|
31
|
+
|
|
32
|
+
| Rule | Severity | Fires on |
|
|
33
|
+
| --------------------------------- | -------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
34
|
+
| `SV001-module-mutable-global` | medium | A module-scope mutable (`= {}`, `= []`, `set()`, `dict()`, `list()`, or a top-level JS `let`/`var`/`const` object/array) that some line later mutates in place (`.append`/`.add`/`[...] =`/`+=`/`.attr =`). Fires on the **declaration**, the leak's source. |
|
|
35
|
+
| `SV002-missing-teardown` | medium | A **class/all-scoped** setup whose matching teardown is absent: pytest `setup_class`/`setUpClass`, or vitest/jest `beforeAll`. Per-test setup (`setUp`/`setup_method`/`beforeEach`) is excluded - it rebuilds state each test, so it does not leak. |
|
|
36
|
+
| `SV003-shared-singleton-mutation` | low | A process-global singleton assigned without restore: `os.environ[...] =`, `sys.modules[...] =`, `process.env.X =`. Reads and `==` comparisons never fire, and neither does a file that demonstrably restores the global — a same-key (or computed-loop) restore in `afterEach`/`afterAll`/teardown/post-`yield` fixture code or an in-test `try`/`finally`, or a write-back from a snapshot saved from that global. |
|
|
37
|
+
| `SV004-order-coupled-name` | low | A test name or comment that encodes ordering: `test_1_…`, a **terminal** ordinal (`test_first()`, `test_last()` - not `test_first_match_wins`), `must run before …`, `it('… run first')`. |
|
|
38
|
+
|
|
39
|
+
A finding is a **suspect, not a verdict.** A module dict that is only ever read
|
|
40
|
+
is a legitimate constant and does not fire; only a _mutated_ one does.
|
|
41
|
+
|
|
42
|
+
## Framework conditioning
|
|
43
|
+
|
|
44
|
+
The setup/teardown idioms (`SV002`) differ by ecosystem, so the rule is
|
|
45
|
+
conditioned on the file: Python files are read with pytest/unittest markers, JS
|
|
46
|
+
and TS with vitest/jest markers. The idioms do not collide across languages, so
|
|
47
|
+
each file is judged by its own ecosystem's conventions.
|
|
48
|
+
|
|
49
|
+
## Fidelity limits (AST-lite, on purpose)
|
|
50
|
+
|
|
51
|
+
The static pass is a scanner with no parser dependency, so it ships anywhere
|
|
52
|
+
`node` does. The cost, stated plainly:
|
|
53
|
+
|
|
54
|
+
- **`SV001` mutation is file-scoped, not flow-scoped.** Any in-place mutation of
|
|
55
|
+
a module-level name anywhere in the file indicts the declaration, even if the
|
|
56
|
+
mutation sits in a helper rather than a test body. A shared-state leak is a
|
|
57
|
+
shared-state leak regardless of which function does the writing.
|
|
58
|
+
- **`SV002` is presence-based, not pairing-based.** A file with one `setUpClass`
|
|
59
|
+
and one `tearDownClass` is considered balanced even if a _second_ class lacks
|
|
60
|
+
teardown. It also only judges class/all-scoped setup, so a genuinely leaky
|
|
61
|
+
per-test setup (rare) is missed - the deliberate false-positive/false-negative
|
|
62
|
+
trade from dogfooding.
|
|
63
|
+
- **Comment-blind for code rules.** `SV003` skips commented-out lines, and
|
|
64
|
+
`SV002` judges both halves of its pair against a code-only projection of the
|
|
65
|
+
file (comments dropped, string contents blanked) - so neither a prose mention
|
|
66
|
+
of `afterAll` nor a fixture string containing one can stand in for the real
|
|
67
|
+
teardown (#732). `SV004` deliberately is _not_ comment-blind, because an
|
|
68
|
+
ordering note in a comment is exactly the self-reported dependence it looks
|
|
69
|
+
for.
|
|
70
|
+
- **String-aware for code anchors.** An `SV003` (or `SV004` name-pattern) match
|
|
71
|
+
that _starts_ inside a string literal on its own line is rejected as fixture
|
|
72
|
+
data. `SV004`'s directive-text alternatives stay unfiltered on purpose: their
|
|
73
|
+
signal (test titles, docstrings) legitimately lives inside strings.
|
|
74
|
+
- **Restoration check is file-level.** `SV003` suppression cannot verify that a
|
|
75
|
+
fixture actually applies to the mutating test, and a computed-key loop restore
|
|
76
|
+
is assumed to cover the whole family — the same file-wide trade blackhawk
|
|
77
|
+
makes for frozen clocks. `vi.unstubAllEnvs`/`monkeypatch` never suppress a
|
|
78
|
+
direct assignment: they only undo their own mutations.
|
|
79
|
+
- **Line-scoped.** A declaration or call split across lines can be missed.
|
|
80
|
+
- **A missed suspect costs less than a false one** — the same bias as
|
|
81
|
+
canary-blackhawk.
|
|
82
|
+
|
|
83
|
+
## Inline suppression (per-line escape hatch)
|
|
84
|
+
|
|
85
|
+
A suite that legitimately contains a suspect line — most commonly a fixture
|
|
86
|
+
string feeding a tool that _tests_ these detections — can suppress a single
|
|
87
|
+
finding with an inline pragma, same dialect as `blackhawk-ignore`:
|
|
88
|
+
|
|
89
|
+
```py
|
|
90
|
+
# savant-ignore SV004 -- fixture: directive-text input the rule under test must detect
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
- **Reason required** (the `-- reason` tail) — keeps suppressions honest and
|
|
94
|
+
greppable, like `eslint-disable-next-line` / `# noqa`. A bare
|
|
95
|
+
`savant-ignore SV004` suppresses nothing.
|
|
96
|
+
- **Rule-scoped** (`SV004`, or the full `SV004-order-coupled-name`) so it never
|
|
97
|
+
blanket-silences the line; another rule firing on the same line still fires.
|
|
98
|
+
- **Placement:** the pragma may trail the offending line or sit on the line
|
|
99
|
+
directly above it. Comma-separate ids to suppress several (`SV003,SV004`).
|
|
100
|
+
- **Counted separately:** suppressed findings are reported as
|
|
101
|
+
`N suppressed (inline savant-ignore)` and in the JSON `summary.suppressed`,
|
|
102
|
+
out of the actionable total — so a genuinely clean suite can read zero
|
|
103
|
+
findings while the known-OK fixtures stay visible.
|
|
104
|
+
- **String-guarded:** pragma text that itself sits _inside_ a string literal is
|
|
105
|
+
fixture data, not a directive — it neither suppresses its own line nor the
|
|
106
|
+
next one.
|
|
107
|
+
|
|
108
|
+
## Which files get scanned
|
|
109
|
+
|
|
110
|
+
A directory walk only visits **test** files — `*.test.*`, `*.spec.*`,
|
|
111
|
+
`test_*.py`, `*_test.py`, or any supported source under `tests/`, `test/`,
|
|
112
|
+
`__tests__/`, `e2e/`, `spec/`. A file named explicitly on the command line is
|
|
113
|
+
always scanned. Supported suffixes: `.py`, `.js`, `.jsx`, `.ts`, `.tsx`, `.mjs`,
|
|
114
|
+
`.cjs`. Dependency directories (`node_modules`, `.venv`, …) are never walked.
|
|
115
|
+
|
|
116
|
+
## Invocation
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
# Scan the repo's test files (advisory - always exits 0):
|
|
120
|
+
canary skills run canary-savant
|
|
121
|
+
|
|
122
|
+
# Scan a specific suite:
|
|
123
|
+
canary skills run canary-savant -- tests/unit
|
|
124
|
+
|
|
125
|
+
# Machine-readable findings:
|
|
126
|
+
canary skills run canary-savant -- tests --json
|
|
127
|
+
|
|
128
|
+
# Fail the step on any suspect:
|
|
129
|
+
canary skills run canary-savant -- tests --strict
|
|
130
|
+
|
|
131
|
+
# Usage, options, and the full rule list (exits 0):
|
|
132
|
+
canary skills run canary-savant -- --help
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
An unknown flag is rejected with `unrecognized arguments: <flag>` and exit 2.
|
|
136
|
+
`--seed` is validated as a determinism flag rather than being allowed to decay
|
|
137
|
+
into a random seed. All three failures exit 2:
|
|
138
|
+
|
|
139
|
+
| Input | Error |
|
|
140
|
+
| ---------------------------------- | ------------------------------------------------------------ |
|
|
141
|
+
| missing value, or a following flag | `argument --seed: expected one argument` |
|
|
142
|
+
| `abc`, `3.7`, `1e999`, `0x10`, `` | `argument --seed: invalid int value: '<value>'` |
|
|
143
|
+
| beyond ±(2^53 − 1) | `argument --seed: seed out of safe integer range: '<value>'` |
|
|
144
|
+
|
|
145
|
+
The last one matters because `Number()` rounds silently above 2^53 − 1, so the
|
|
146
|
+
seed actually used would differ from the seed asked for — the same lie the whole
|
|
147
|
+
flag exists to prevent.
|
|
148
|
+
|
|
149
|
+
`--seed 42` and `--seed=42` are equivalent, and both accept negative seeds
|
|
150
|
+
(`--seed -5`, `--seed=-5`). Use `--` to end option parsing when a path itself
|
|
151
|
+
starts with a dash.
|
|
152
|
+
|
|
153
|
+
### The confirming pass — dynamic confirmation (`--confirm`, opt-in)
|
|
154
|
+
|
|
155
|
+
`--confirm` runs the suite in declared order, re-runs it shuffled under a pinned
|
|
156
|
+
seed, and for each order-dependent victim runs it alone and **bisects the prefix
|
|
157
|
+
to name the polluter** — the earlier test whose state leaked. Opt-in because a
|
|
158
|
+
shuffled re-run at least doubles wall-clock; always prints the seed and a
|
|
159
|
+
copy-pasteable reproduce command.
|
|
160
|
+
|
|
161
|
+
```bash
|
|
162
|
+
# Confirm order-dependence, pinning the seed for reproducibility:
|
|
163
|
+
canary skills run canary-savant -- tests --confirm --seed 424242
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
**pytest and vitest** are both supported; savant auto-detects from the target
|
|
167
|
+
(extensions, then a directory scan, then config files). pytest needs a shuffle
|
|
168
|
+
plugin (`pytest-randomly` or `pytest-random-order`) and declines loudly if none
|
|
169
|
+
is installed; vitest's shuffle is built in (`--sequence.shuffle`), so no plugin
|
|
170
|
+
is required. Node drives the project's _own_ runner — savant orchestrates it, it
|
|
171
|
+
does not run in the target's language.
|
|
172
|
+
|
|
173
|
+
**Polluter bisect is pytest-only.** vitest has no CLI-driven ordered per-test
|
|
174
|
+
execution, so a vitest target gets victim _detection_ (which tests break under
|
|
175
|
+
shuffle) but not culprit _naming_. For pytest, node ids are captured via
|
|
176
|
+
`--collect-only`, so class-based layouts (`file.py::Class::test`) re-run
|
|
177
|
+
correctly.
|
|
178
|
+
|
|
179
|
+
`--json` shape:
|
|
180
|
+
|
|
181
|
+
```json
|
|
182
|
+
{
|
|
183
|
+
"schema_version": 1,
|
|
184
|
+
"findings": [
|
|
185
|
+
{
|
|
186
|
+
"file": "tests/test_cache.py",
|
|
187
|
+
"line": 3,
|
|
188
|
+
"rule_id": "SV001-module-mutable-global",
|
|
189
|
+
"severity": "medium",
|
|
190
|
+
"snippet": "_CACHE = {}",
|
|
191
|
+
"why": "a module-level mutable is written by a test, so state leaks into whatever test runs next"
|
|
192
|
+
}
|
|
193
|
+
],
|
|
194
|
+
"summary": {
|
|
195
|
+
"files_scanned": 8,
|
|
196
|
+
"findings": 1,
|
|
197
|
+
"by_severity": { "medium": 1 },
|
|
198
|
+
"suppressed": 0
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
## Fixing what it finds
|
|
204
|
+
|
|
205
|
+
| Finding | Fix |
|
|
206
|
+
| ------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
207
|
+
| `SV001` | Move the mutable into a fixture that rebuilds it per test, or reset it in teardown. Module-level mutable state shared across tests is the leak. |
|
|
208
|
+
| `SV002` | Add the matching teardown (`teardown_method`, `afterEach`, a `yield` fixture) so acquired state is released. |
|
|
209
|
+
| `SV003` | Use a restoring helper — pytest `monkeypatch.setenv`, or save/restore around the test — instead of assigning the global directly. |
|
|
210
|
+
| `SV004` | Make the test self-contained so order stops mattering, then drop the ordering hint from the name/comment. |
|
|
211
|
+
|
|
212
|
+
## Dogfooding and the `--strict` promotion path
|
|
213
|
+
|
|
214
|
+
canary runs savant's static pass over its **own** test suite on every PR
|
|
215
|
+
(`.github/workflows/harness-quality.yml`, the `Skills (JS)` job), **advisory**:
|
|
216
|
+
it prints suspects to the log and always exits 0. Tuning the rules against that
|
|
217
|
+
real suite dropped the backlog from 37 findings to a handful of genuine
|
|
218
|
+
suspects; the structural residue (savant's own suite testing SV004's
|
|
219
|
+
directive-text detection, #496) is suppressed with per-site `savant-ignore`
|
|
220
|
+
reasons and stays visible in the suppressed count. Promote to blocking by
|
|
221
|
+
appending `--strict` to that step once the remaining suspects are triaged (fixed
|
|
222
|
+
or confirmed benign) - the same advisory-first path every canary gate takes.
|
|
223
|
+
|
|
224
|
+
## Roadmap
|
|
225
|
+
|
|
226
|
+
- **Shipped:** the static pass; the dynamic confirming pass (baseline → shuffle
|
|
227
|
+
→ classify) with isolation + polluter bisect (pytest); vitest as a
|
|
228
|
+
confirming-pass classify target; pytest node-id capture for class-based
|
|
229
|
+
layouts; advisory CI gate dogfooded on canary's own suite (rules tuned to kill
|
|
230
|
+
the dominant false positives).
|
|
231
|
+
- **Remaining:** flip the advisory gate to `--strict` once the suspect backlog
|
|
232
|
+
is triaged; vitest polluter naming is out of scope until vitest gains ordered
|
|
233
|
+
per-test execution. See `docs/changes/canary-savant/proposal.md`.
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// canary-savant -- order-dependence and isolation detector (Tier-1 static scan).
|
|
3
|
+
//
|
|
4
|
+
// Phase 1 ships the always-on static "suspect" tier: an AST-lite scan that
|
|
5
|
+
// flags the shared-state smells that predict order-dependent tests -- module-
|
|
6
|
+
// level mutables written by tests, setup without teardown, mutated process
|
|
7
|
+
// singletons, order-coupled names -- with no test execution. The opt-in
|
|
8
|
+
// dynamic confirmer (--confirm) lands in a later phase.
|
|
9
|
+
//
|
|
10
|
+
// <paths> files or directories to scan (default: the current directory).
|
|
11
|
+
// --json emit machine-readable findings instead of human text.
|
|
12
|
+
// --strict exit 1 when there are findings (default is advisory: exit 0).
|
|
13
|
+
//
|
|
14
|
+
// Tier-0 in the real sense -- no LLM, no network, no secrets, no dependency on
|
|
15
|
+
// any other skill.
|
|
16
|
+
//
|
|
17
|
+
// Invoked via `canary skills run canary-savant -- [paths] [--json] [--strict]`.
|
|
18
|
+
|
|
19
|
+
import fs from 'node:fs';
|
|
20
|
+
import { scanPaths, toJson } from './scanner.mjs';
|
|
21
|
+
import { confirm, locatePolluters, realPolluterSeams } from './runner.mjs';
|
|
22
|
+
import { RULES } from './rules.mjs';
|
|
23
|
+
import {
|
|
24
|
+
createParser,
|
|
25
|
+
formatUsageError,
|
|
26
|
+
EXIT_USAGE,
|
|
27
|
+
} from '../../../lib/parse-args.mjs';
|
|
28
|
+
|
|
29
|
+
export const SCHEMA_VERSION = 1;
|
|
30
|
+
|
|
31
|
+
// --- no-silent-abstention (#508 D2, skill-CLI convention half) ---------------
|
|
32
|
+
//
|
|
33
|
+
// Skill CLIs are deliberately self-contained -- no engine import, no shared
|
|
34
|
+
// module -- so they cannot call `gateOutcome`. They honour the doctrine by
|
|
35
|
+
// CONVENTION instead, emitting the same greppable line the engine helper does.
|
|
36
|
+
// The skill-layer conformance registry (agents/skills/test/gate-conformance.
|
|
37
|
+
// test.ts) is what holds them to it: a row whose fixture collapses the
|
|
38
|
+
// denominator and asserts the loud outcome.
|
|
39
|
+
//
|
|
40
|
+
// U+26A0 / U+2014 are written as escapes so this source stays ASCII, matching
|
|
41
|
+
// ts/src/core/gate-result.ts.
|
|
42
|
+
const ABSTAINED_LINE =
|
|
43
|
+
'\u{26A0} Abstained \u{2014} verified zero items; this is not a pass.';
|
|
44
|
+
|
|
45
|
+
const PREFIX = 'canary-savant:';
|
|
46
|
+
|
|
47
|
+
// The rules block is GENERATED from RULES, never hand-typed, so a new rule
|
|
48
|
+
// appears in --help the moment it is registered.
|
|
49
|
+
const USAGE =
|
|
50
|
+
'usage: canary-savant [-h] [--json] [--strict] [--confirm] [--seed N] [--]\n' +
|
|
51
|
+
' [path ...]\n' +
|
|
52
|
+
'\n' +
|
|
53
|
+
'Order-dependence and isolation detector: flags the shared-state smells that\n' +
|
|
54
|
+
'predict order-dependent tests, and optionally confirms them dynamically.\n' +
|
|
55
|
+
'\n' +
|
|
56
|
+
'positional arguments:\n' +
|
|
57
|
+
' path files or directories to scan (default: the current directory)\n' +
|
|
58
|
+
'\n' +
|
|
59
|
+
'options:\n' +
|
|
60
|
+
' -h, --help show this help message and exit\n' +
|
|
61
|
+
' --json emit machine-readable findings instead of human text\n' +
|
|
62
|
+
' --strict exit 1 when there are findings (default is advisory: exit 0)\n' +
|
|
63
|
+
' --confirm run the opt-in Tier-2 dynamic confirmer (executes the suite)\n' +
|
|
64
|
+
' --seed N shuffle seed for --confirm (default: random)\n' +
|
|
65
|
+
'\n' +
|
|
66
|
+
'rules:\n' +
|
|
67
|
+
RULES.map((r) => ` ${r.ruleId} (${r.severity})`).join('\n');
|
|
68
|
+
|
|
69
|
+
function summary(result) {
|
|
70
|
+
const bySeverity = {};
|
|
71
|
+
for (const f of result.findings) {
|
|
72
|
+
bySeverity[f.severity] = (bySeverity[f.severity] || 0) + 1;
|
|
73
|
+
}
|
|
74
|
+
return {
|
|
75
|
+
files_scanned: result.filesScanned,
|
|
76
|
+
findings: result.findings.length,
|
|
77
|
+
by_severity: bySeverity,
|
|
78
|
+
suppressed: result.suppressed ?? 0,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// A trailing "N suppressed" note keeps inline-ignored lines visible but out of
|
|
83
|
+
// the actionable total - the pattern canary-blackhawk (#393) and the
|
|
84
|
+
// PR-guardian sticky comment use.
|
|
85
|
+
function suppressedNote(result) {
|
|
86
|
+
const n = result.suppressed ?? 0;
|
|
87
|
+
return n ? `\n${n} suppressed (inline savant-ignore).` : '';
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function renderText(result) {
|
|
91
|
+
const count = result.findings.length;
|
|
92
|
+
const files = result.filesScanned;
|
|
93
|
+
const fp = files === 1 ? '' : 's';
|
|
94
|
+
// #508: zero suspects over zero scanned files is an ABSENT result, not a
|
|
95
|
+
// clean one. Findings outrank abstention, so this is the no-findings path.
|
|
96
|
+
if (!count && !files) {
|
|
97
|
+
return (
|
|
98
|
+
`${ABSTAINED_LINE} No file matched the given paths, so there is ` +
|
|
99
|
+
'nothing to report. Point at a directory that holds test files, or ' +
|
|
100
|
+
'pass a file directly.'
|
|
101
|
+
);
|
|
102
|
+
}
|
|
103
|
+
if (!count) {
|
|
104
|
+
return (
|
|
105
|
+
`No order-dependence suspects (${files} file${fp} scanned).` +
|
|
106
|
+
suppressedNote(result)
|
|
107
|
+
);
|
|
108
|
+
}
|
|
109
|
+
const sp = count === 1 ? '' : 's';
|
|
110
|
+
const lines = [
|
|
111
|
+
`${count} order-dependence suspect${sp} in ${files} file${fp}:`,
|
|
112
|
+
'',
|
|
113
|
+
];
|
|
114
|
+
for (const f of result.findings) {
|
|
115
|
+
lines.push(` ${f.file}:${f.line} [${f.severity}] ${f.ruleId}`);
|
|
116
|
+
lines.push(` ${f.snippet}`);
|
|
117
|
+
lines.push(` why: ${f.why}`);
|
|
118
|
+
}
|
|
119
|
+
lines.push('');
|
|
120
|
+
lines.push(
|
|
121
|
+
'Advisory by default. Re-run with --strict to fail the step on findings.',
|
|
122
|
+
);
|
|
123
|
+
return lines.join('\n') + suppressedNote(result);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* savant takes paths, so it gets the `--` end-of-options terminator and treats
|
|
128
|
+
* a lone `-` as a positional, as argparse does.
|
|
129
|
+
*
|
|
130
|
+
* `--seed` is the flag that made the shared int type worth having: a bad value
|
|
131
|
+
* used to decay to `Math.floor(Math.random() * 1e6)` at exit 0, silently
|
|
132
|
+
* randomizing the one flag whose entire purpose is reproducibility. The shared
|
|
133
|
+
* parser rejects a non-integer AND an integer past the safe range (#479).
|
|
134
|
+
*/
|
|
135
|
+
export const CLI_SPEC = {
|
|
136
|
+
prog: 'canary-savant',
|
|
137
|
+
booleans: {
|
|
138
|
+
'--json': 'json',
|
|
139
|
+
'--strict': 'strict',
|
|
140
|
+
'--confirm': 'confirm',
|
|
141
|
+
},
|
|
142
|
+
values: { '--seed': { key: 'seed', type: 'int' } },
|
|
143
|
+
positionals: { key: 'paths', defaults: ['.'] },
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
const parseArgs = createParser(CLI_SPEC);
|
|
147
|
+
|
|
148
|
+
export function renderConfirm(dyn) {
|
|
149
|
+
if (
|
|
150
|
+
dyn.status === 'no_plugin' ||
|
|
151
|
+
dyn.status === 'baseline_red' ||
|
|
152
|
+
dyn.status === 'unknown_framework'
|
|
153
|
+
) {
|
|
154
|
+
return `\nTier 2 (dynamic): skipped - ${dyn.message}`;
|
|
155
|
+
}
|
|
156
|
+
const lines = [`\nTier 2 (dynamic): seed ${dyn.seed}`];
|
|
157
|
+
if (!dyn.victims.length) {
|
|
158
|
+
lines.push(' No order-dependence confirmed under this seed.');
|
|
159
|
+
} else {
|
|
160
|
+
let namedAny = false;
|
|
161
|
+
for (const v of dyn.victims) {
|
|
162
|
+
lines.push(` order-dependent: ${v.victim}`);
|
|
163
|
+
if (v.polluter) {
|
|
164
|
+
lines.push(` polluted by: ${v.polluter}`);
|
|
165
|
+
if (v.reproduce) lines.push(` reproduce: ${v.reproduce}`);
|
|
166
|
+
namedAny = true;
|
|
167
|
+
} else if (v.note) {
|
|
168
|
+
lines.push(` ${v.note}`);
|
|
169
|
+
} else if (v.exhausted) {
|
|
170
|
+
lines.push(
|
|
171
|
+
` no single culprit isolated; smallest reproducing prefix: ` +
|
|
172
|
+
`${v.minimalPrefix?.length ?? '?'} test(s)`,
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
if (!namedAny && dyn.reproduce) {
|
|
177
|
+
lines.push(` reproduce: ${dyn.reproduce}`);
|
|
178
|
+
}
|
|
179
|
+
if (dyn.framework === 'vitest') {
|
|
180
|
+
lines.push(
|
|
181
|
+
' (vitest: victims detected; polluter bisect is pytest-only)',
|
|
182
|
+
);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
if (dyn.nondeterministic.length) {
|
|
186
|
+
lines.push(
|
|
187
|
+
` (${dyn.nondeterministic.length} nondeterministic flake(s) - not order, handed off)`,
|
|
188
|
+
);
|
|
189
|
+
}
|
|
190
|
+
return lines.join('\n');
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
export function main(argv = []) {
|
|
194
|
+
const { positionals: paths, opts, help, error } = parseArgs(argv);
|
|
195
|
+
|
|
196
|
+
// Usage and parse errors resolve before any filesystem work, so `--help`
|
|
197
|
+
// never reports a missing path and a typo never half-runs a scan.
|
|
198
|
+
if (help) {
|
|
199
|
+
console.log(USAGE);
|
|
200
|
+
return 0;
|
|
201
|
+
}
|
|
202
|
+
if (error) {
|
|
203
|
+
console.error(formatUsageError(CLI_SPEC.prog, error));
|
|
204
|
+
return EXIT_USAGE;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
for (const entry of paths) {
|
|
208
|
+
if (!fs.existsSync(entry)) {
|
|
209
|
+
console.error(`${PREFIX} path not found: ${entry}`);
|
|
210
|
+
return 1;
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const result = scanPaths(paths);
|
|
215
|
+
|
|
216
|
+
let dyn;
|
|
217
|
+
if (opts.confirm) {
|
|
218
|
+
const seed = Number.isFinite(opts.seed)
|
|
219
|
+
? opts.seed
|
|
220
|
+
: Math.floor(Math.random() * 1e6);
|
|
221
|
+
dyn = confirm(paths, { seed });
|
|
222
|
+
// Phase 3: name the polluter behind each confirmed victim. pytest only -
|
|
223
|
+
// vitest has no CLI-driven ordered per-test execution, so it gets victim
|
|
224
|
+
// detection (Phase 2) but not polluter bisection.
|
|
225
|
+
if (
|
|
226
|
+
dyn.status === 'ok' &&
|
|
227
|
+
dyn.victims.length &&
|
|
228
|
+
dyn.framework === 'pytest'
|
|
229
|
+
) {
|
|
230
|
+
dyn.victims = locatePolluters(
|
|
231
|
+
dyn.victims,
|
|
232
|
+
dyn.order,
|
|
233
|
+
realPolluterSeams(paths),
|
|
234
|
+
);
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
if (opts.json) {
|
|
239
|
+
const payload = {
|
|
240
|
+
schema_version: SCHEMA_VERSION,
|
|
241
|
+
findings: result.findings.map(toJson),
|
|
242
|
+
summary: summary(result),
|
|
243
|
+
};
|
|
244
|
+
if (dyn) {
|
|
245
|
+
payload.dynamic = {
|
|
246
|
+
status: dyn.status,
|
|
247
|
+
seed: dyn.seed,
|
|
248
|
+
victims: dyn.victims,
|
|
249
|
+
nondeterministic: dyn.nondeterministic,
|
|
250
|
+
reproduce: dyn.reproduce,
|
|
251
|
+
...(dyn.message ? { message: dyn.message } : {}),
|
|
252
|
+
};
|
|
253
|
+
}
|
|
254
|
+
console.log(JSON.stringify(payload, null, 2));
|
|
255
|
+
} else {
|
|
256
|
+
console.log(renderText(result) + (dyn ? renderConfirm(dyn) : ''));
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const hasViolation =
|
|
260
|
+
result.findings.length > 0 || (dyn?.victims.length ?? 0) > 0;
|
|
261
|
+
// Advisory by default (D3); --strict inherits EXIT_ABSTAINED (3) on a
|
|
262
|
+
// collapsed denominator, distinct from 1 ("found something real").
|
|
263
|
+
if (opts.strict && !result.filesScanned) return 3;
|
|
264
|
+
return opts.strict && hasViolation ? 1 : 0;
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
// Direct execution (the skill runner execs this file via its shebang).
|
|
268
|
+
//
|
|
269
|
+
// `process.exitCode`, not `process.exit()`: a large `--json` payload exceeds
|
|
270
|
+
// the pipe buffer, and `process.exit` tears the process down mid-write, leaving
|
|
271
|
+
// truncated JSON that still exits 0 (#791).
|
|
272
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
273
|
+
process.exitCode = main(process.argv.slice(2));
|
|
274
|
+
}
|