canary-test-cli 7.2.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +23 -4
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +23 -16
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +3 -1
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +20 -3
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +1 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +15 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/lib/parse-args.mjs +200 -139
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +46 -7
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +13 -18
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/vacuity-scanner.js +151 -6
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/cli.js +180 -264
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +262 -430
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +48 -32
- package/dist/engine/workflow-cli.js +85 -65
- package/package.json +1 -1
package/agents/skills/README.md
CHANGED
|
@@ -13,8 +13,9 @@ it), see [Guides](../../docs/guides/index.md).
|
|
|
13
13
|
|
|
14
14
|
```text
|
|
15
15
|
agents/skills/
|
|
16
|
-
├── claude-code/ # Claude Code skills (
|
|
16
|
+
├── claude-code/ # Claude Code skills (24)
|
|
17
17
|
│ ├── canary-add-framework/
|
|
18
|
+
│ ├── canary-batwoman/
|
|
18
19
|
│ ├── canary-blackhawk/
|
|
19
20
|
│ ├── canary-cassandra/
|
|
20
21
|
│ ├── canary-ci-ready/
|
|
@@ -30,9 +31,11 @@ agents/skills/
|
|
|
30
31
|
│ ├── canary-pr-guardian/
|
|
31
32
|
│ ├── canary-promote-test/
|
|
32
33
|
│ ├── canary-savant/
|
|
34
|
+
│ ├── canary-screech/
|
|
33
35
|
│ ├── canary-setup-harness/
|
|
34
36
|
│ ├── canary-shadow/
|
|
35
37
|
│ ├── canary-ship/
|
|
38
|
+
│ ├── canary-strix/
|
|
36
39
|
│ ├── canary-test-pipeline/
|
|
37
40
|
│ └── canary-test-reporter/
|
|
38
41
|
└── README.md # this file
|
|
@@ -90,6 +93,21 @@ slash-command entry points.
|
|
|
90
93
|
into a Markdown and/or JSON report with pass/fail/flaky/skipped counts.
|
|
91
94
|
Complements `canary-fail-fast` (which aborts early) by summarising the full
|
|
92
95
|
run at the end.
|
|
96
|
+
- [`canary-screech`](./claude-code/canary-screech/SKILL.md) — Bundled executable
|
|
97
|
+
skill (`scripts/cli.mjs`). Broken-main siren: reads the cross-run history
|
|
98
|
+
store, decides whether the default branch is red, and emits a one-page blast
|
|
99
|
+
(culprit commit range, failure cluster, owning area, quarantine-or-revert
|
|
100
|
+
recommendation, chat-ready block) as a markdown artifact plus a `::error`
|
|
101
|
+
annotation. The cross-run complement to the two above — neither of them can
|
|
102
|
+
tell that the branch itself went red.
|
|
103
|
+
|
|
104
|
+
### Closure auditing
|
|
105
|
+
|
|
106
|
+
- [`canary-batwoman`](./claude-code/canary-batwoman/SKILL.md) — Reports whether
|
|
107
|
+
the files a closed issue's fix changed have actually **executed** since that
|
|
108
|
+
fix merged. GitHub closes an issue on a keyword match, which checks neither
|
|
109
|
+
that the fix works nor that it ever ran. Advisory, and the only skill here
|
|
110
|
+
that requires the network (`gh`): deterministic, network, no agent.
|
|
93
111
|
|
|
94
112
|
### Test hygiene & reliability
|
|
95
113
|
|
|
@@ -204,9 +222,10 @@ Use the canary-generate-test skill to write a load test for /v1/search.
|
|
|
204
222
|
Most skills here are documentation, not executable artifacts — they describe
|
|
205
223
|
_how an agent should behave_, not a function to call. Several are bundled
|
|
206
224
|
executable skills with their own CLI entry point (`cli:` in frontmatter).
|
|
207
|
-
`canary-fail-fast`, `canary-katana`, and `canary-blackhawk`
|
|
208
|
-
(`scripts/cli.mjs`); `canary-instrument` and
|
|
209
|
-
Python entry (`scripts/cli.py`). Run those
|
|
225
|
+
`canary-fail-fast`, `canary-katana`, `canary-screech`, and `canary-blackhawk`
|
|
226
|
+
ship a Node entry (`scripts/cli.mjs`); `canary-instrument` and
|
|
227
|
+
`canary-test-reporter` ship a Python entry (`scripts/cli.py`). Run those
|
|
228
|
+
directly, e.g.:
|
|
210
229
|
|
|
211
230
|
```bash
|
|
212
231
|
node agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs --help
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-batwoman
|
|
3
|
+
description:
|
|
4
|
+
Closure auditing — reports whether the files a closed issue's fix changed have
|
|
5
|
+
actually EXECUTED since that fix merged. GitHub closes an issue on a keyword
|
|
6
|
+
match in a PR body, which checks neither that the fix works nor that it ever
|
|
7
|
+
ran; batwoman answers only the second question, per changed file, and names
|
|
8
|
+
the files it could not answer for. Use when the user asks "did that fix
|
|
9
|
+
actually run", "is this issue really done", "audit a closed issue", or after a
|
|
10
|
+
batch of merges. Advisory and read-only — it never asserts correctness, never
|
|
11
|
+
fails a job, and never reopens an issue. NOT a test runner, NOT a coverage
|
|
12
|
+
tool (a dormant workflow has whatever coverage it always had), and NOT a
|
|
13
|
+
correctness check.
|
|
14
|
+
cli: canary batwoman
|
|
15
|
+
requires: [node>=20, gh]
|
|
16
|
+
---
|
|
17
|
+
|
|
18
|
+
# Canary Batwoman
|
|
19
|
+
|
|
20
|
+
GitHub closes an issue when a merged PR body matches `Closes #N`. That is a
|
|
21
|
+
**string match with no denominator**: nothing checks that the fix works, and
|
|
22
|
+
nothing checks that the fix ever ran. The issue moves to `CLOSED` on the
|
|
23
|
+
strength of a keyword.
|
|
24
|
+
|
|
25
|
+
Batwoman audits the claim. For each file the closing PR changed, it reports
|
|
26
|
+
whether that artifact has executed since the merge — and, just as importantly,
|
|
27
|
+
which files it could not decide about.
|
|
28
|
+
|
|
29
|
+
## The founding case
|
|
30
|
+
|
|
31
|
+
canary#749 fixed a false-green in `.github/workflows/refresh-arch-baseline.yml`.
|
|
32
|
+
It merged as `1e0c05b` and auto-closed on the keyword. Verified afterwards:
|
|
33
|
+
|
|
34
|
+
- 11 script unit tests passed
|
|
35
|
+
- 97 static workflow assertions passed
|
|
36
|
+
- the workflow itself **had not run since twelve days before the fix merged** —
|
|
37
|
+
it is label-triggered, so it stayed dormant until someone applied the label
|
|
38
|
+
|
|
39
|
+
Every gate the repo owns read that as done. The fix was real and well-tested;
|
|
40
|
+
its end-to-end path was unproven. **Verified and exercised are different
|
|
41
|
+
properties**, and that gap is what batwoman makes visible.
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
canary batwoman --help
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
That one needs no credentials, no network and no fixtures. A real audit needs
|
|
50
|
+
the repository named and `gh` authenticated:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
export GITHUB_REPOSITORY=owner/name # never inferred from a git remote
|
|
54
|
+
canary batwoman --issue 749 # the report
|
|
55
|
+
canary batwoman --issue 749 --json # the machine shape, for CI
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
`GITHUB_REPOSITORY` is required rather than guessed. Auditing the wrong
|
|
59
|
+
repository's run history would produce a confident answer about the wrong thing,
|
|
60
|
+
which is worse than refusing.
|
|
61
|
+
|
|
62
|
+
## What the five statuses mean
|
|
63
|
+
|
|
64
|
+
| status | meaning |
|
|
65
|
+
| ---------------- | -------------------------------------------------------------------------- |
|
|
66
|
+
| `exercised` | a run started after the merge, and it names which |
|
|
67
|
+
| `not-exercised` | the artifact has not run since — with the `on:` trigger explaining why |
|
|
68
|
+
| `abstain` | a probe looked and could not tell (unreadable history, untraceable script) |
|
|
69
|
+
| `no-probe` | nothing looked: a registry gap, fixable by adding a probe |
|
|
70
|
+
| `not-applicable` | prose, config, or a file this change deleted |
|
|
71
|
+
|
|
72
|
+
`abstain` and `no-probe` are deliberately separate, and neither is folded into
|
|
73
|
+
anything resembling a pass. **The summary always prints all five counts**, and
|
|
74
|
+
they always sum to the changed-file total — a report over a subset presented as
|
|
75
|
+
a report over the whole is the exact defect batwoman exists to detect.
|
|
76
|
+
|
|
77
|
+
There is no `assessed` field, no score, and no success token anywhere in the
|
|
78
|
+
output. A convenient `passed: true` in the JSON is the field a CI wrapper would
|
|
79
|
+
grow later; it does not exist, and a test asserts it stays that way.
|
|
80
|
+
|
|
81
|
+
## What it will not tell you
|
|
82
|
+
|
|
83
|
+
- **Whether the fix is correct.** Batwoman proves execution, never behaviour. A
|
|
84
|
+
workflow that ran and did the wrong thing reports `exercised`.
|
|
85
|
+
- **Whether coverage is adequate.** A dormant workflow has whatever coverage it
|
|
86
|
+
always had; that is the point.
|
|
87
|
+
- **Anything about a file it has no probe for.** `ts/src/**` is an honest
|
|
88
|
+
`no-probe` row in v1, countable rather than silent.
|
|
89
|
+
|
|
90
|
+
## Probes shipped
|
|
91
|
+
|
|
92
|
+
| probe | matches | decides by |
|
|
93
|
+
| ----------------- | ------------------------- | ------------------------------------------------------------------------ |
|
|
94
|
+
| `workflow` | `.github/workflows/*.yml` | a run created after the merge; reads `on:` to explain a dormant one |
|
|
95
|
+
| `workflow-script` | `scripts/*.mjs` | the workflows that name it, inheriting `exercised` if **any** caller ran |
|
|
96
|
+
| `no-execution` | `*.md`, config manifests | returns `not-applicable`, reading nothing |
|
|
97
|
+
|
|
98
|
+
A script nothing references **abstains** rather than reporting as never-run: "no
|
|
99
|
+
workflow calls this" is a statement about the repo, not about the script, which
|
|
100
|
+
may still be run by a hand or a hook.
|
|
101
|
+
|
|
102
|
+
## Requires the network, and says so
|
|
103
|
+
|
|
104
|
+
Batwoman shells out to `gh` for run history and for the closing PR. That is a
|
|
105
|
+
property, not a tier — unlike the deterministic offline detectors alongside it
|
|
106
|
+
(`canary-savant`, `canary-blackhawk`, `canary-cassandra`, `canary-katana`),
|
|
107
|
+
which run anywhere node does.
|
|
108
|
+
|
|
109
|
+
An unauthenticated or missing `gh` does not degrade to a clean report. It fails
|
|
110
|
+
loudly, and every affected file becomes an `abstain` naming the failure. Per
|
|
111
|
+
this repo's standing rule, **cannot-verify is a finding, not a skip**.
|
|
112
|
+
|
|
113
|
+
## In CI
|
|
114
|
+
|
|
115
|
+
`.github/workflows/batwoman.yml` runs it on every push to `main`, deriving the
|
|
116
|
+
issue from the merge commit's closing keyword — the very keyword match batwoman
|
|
117
|
+
questions, which is the right input precisely because it is what GitHub acted
|
|
118
|
+
on. Advisory: every step is `continue-on-error`, and a merge that closes nothing
|
|
119
|
+
is reported and skipped rather than passing silently over an empty set.
|
|
@@ -3,14 +3,15 @@ name: canary-cassandra
|
|
|
3
3
|
description: >
|
|
4
4
|
Vacuous-test detection — finds tests that PASS WITHOUT PROVING ANYTHING: an
|
|
5
5
|
assertion that compares a value with itself, a test that never invokes the
|
|
6
|
-
target it claims to cover, and a test whose every assertion is an absence
|
|
7
|
-
observed on a bystander rather than on the
|
|
8
|
-
says "why did this pass against the bug",
|
|
9
|
-
anything", "audit my suite for vacuous
|
|
10
|
-
a bug shipped through a green suite.
|
|
11
|
-
execution. NOT for tests with zero
|
|
12
|
-
LINT-006), NOT for flaky tests
|
|
13
|
-
— a vacuous test has coverage,
|
|
6
|
+
target it claims to cover, and a test whose every assertion is an absence, or
|
|
7
|
+
a trivially true presence check, observed on a bystander rather than on the
|
|
8
|
+
code under test. Use when the user says "why did this pass against the bug",
|
|
9
|
+
"are these tests actually testing anything", "audit my suite for vacuous
|
|
10
|
+
tests", "green but worthless", or after a bug shipped through a green suite.
|
|
11
|
+
Advisory and deterministic — no LLM, no execution. NOT for tests with zero
|
|
12
|
+
assertions (that is `canary review-test`'s LINT-006), NOT for flaky tests
|
|
13
|
+
(canary-flake-hunter), and NOT a coverage tool — a vacuous test has coverage,
|
|
14
|
+
which is exactly why coverage never caught it.
|
|
14
15
|
cli: scripts/cli.mjs
|
|
15
16
|
requires: [node>=20]
|
|
16
17
|
---
|
|
@@ -29,15 +30,20 @@ Cassandra is Tier-0: deterministic, no LLM, no network, no execution.
|
|
|
29
30
|
|
|
30
31
|
## What it finds
|
|
31
32
|
|
|
32
|
-
| Rule | Severity | Fires on
|
|
33
|
-
| --------- | -------- |
|
|
34
|
-
| `VAC-001` | critical | An assertion whose expectation is identical to the value it checks — `expect(true).toBe(true)`, `assert x == x`
|
|
35
|
-
| `VAC-002` | warning | The test never references the target it claims to cover
|
|
36
|
-
| `VAC-003` | warning | Every assertion in the test asserts an _absence_, and none of them observes the target — so nothing proves it ran
|
|
33
|
+
| Rule | Severity | Fires on |
|
|
34
|
+
| --------- | -------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
35
|
+
| `VAC-001` | critical | An assertion whose expectation is identical to the value it checks — `expect(true).toBe(true)`, `assert x == x` |
|
|
36
|
+
| `VAC-002` | warning | The test never references the target it claims to cover |
|
|
37
|
+
| `VAC-003` | warning | Every assertion in the test asserts an _absence_, and none of them observes the target — so nothing proves it ran |
|
|
38
|
+
| `VAC-005` | warning | Every assertion is a trivially true _presence_ check (`toBeDefined`, `toBeTruthy`, `assert x is not None`) on a value the test built itself before the target ran |
|
|
37
39
|
|
|
38
40
|
`VAC-001` is deterministic, hence `critical`: no implementation can fail it.
|
|
39
|
-
`VAC-
|
|
40
|
-
|
|
41
|
+
`VAC-005` abstains (no finding) whenever it cannot prove the subject is a
|
|
42
|
+
bystander, such as a name bound in a hook or a multi-line initialiser. `VAC-004`
|
|
43
|
+
is reserved for the self-excusing-skip rule
|
|
44
|
+
(`docs/changes/vac-004-self-excusing-skip/`). `VAC-002` and `VAC-003` depend on
|
|
45
|
+
resolving a target, which is inference, so they are `warning` and carry a
|
|
46
|
+
fidelity tier.
|
|
41
47
|
|
|
42
48
|
## Run it
|
|
43
49
|
|
|
@@ -45,7 +51,8 @@ Two doors, one detector. Both run the same engine rules, so they cannot disagree
|
|
|
45
51
|
about a finding or about the denominator.
|
|
46
52
|
|
|
47
53
|
```bash
|
|
48
|
-
canary vacuity-check tests/ # human-readable
|
|
54
|
+
canary vacuity-check tests/ # human-readable, skips counted per reason
|
|
55
|
+
canary vacuity-check tests/ --verbose # ...plus every skipped test at file:line
|
|
49
56
|
canary vacuity-check tests/ --json # verdict + denominator + skips
|
|
50
57
|
canary vacuity-check tests/a.test.ts # one file
|
|
51
58
|
```
|
|
@@ -4,7 +4,8 @@
|
|
|
4
4
|
// Finds tests that PASS WITHOUT PROVING ANYTHING: an assertion that compares a
|
|
5
5
|
// value with itself (VAC-001), a test that never invokes the target it claims
|
|
6
6
|
// to cover (VAC-002), and a test whose every assertion is an absence observed
|
|
7
|
-
// on a bystander (VAC-003)
|
|
7
|
+
// on a bystander (VAC-003), and a test whose every assertion is a trivially
|
|
8
|
+
// true presence check on a value the test built itself (VAC-005).
|
|
8
9
|
//
|
|
9
10
|
// <paths> files or directories to scan (default: the current directory).
|
|
10
11
|
// --json emit machine-readable findings instead of human text.
|
|
@@ -53,6 +54,7 @@ const USAGE =
|
|
|
53
54
|
' VAC-001 (critical) assertion compares a value with itself\n' +
|
|
54
55
|
' VAC-002 (warning) the test never invokes the target it covers\n' +
|
|
55
56
|
' VAC-003 (warning) every assertion is an absence, on a bystander\n' +
|
|
57
|
+
' VAC-005 (warning) every assertion is trivial presence, on a bystander\n' +
|
|
56
58
|
'\n' +
|
|
57
59
|
'The denominator is TESTS read, not files. A zero denominator exits 3 under\n' +
|
|
58
60
|
'--strict; it is never reported as a clean scan.';
|
|
@@ -14,6 +14,14 @@ Analyses a test suite across five dimensions and produces a readiness score. Use
|
|
|
14
14
|
this before promoting a suite to CI, or as the convergence gate in
|
|
15
15
|
`/canary-test-pipeline`.
|
|
16
16
|
|
|
17
|
+
**Deterministic scorer:** run `canary ci-ready [--root <dir>] [--json]` first.
|
|
18
|
+
It scores every check that has a real input and reports `skip`, naming the
|
|
19
|
+
missing input, for every check that does not. A skip is never a pass. Today only
|
|
20
|
+
flakiness has a producer behind it, so expect the other four to skip until
|
|
21
|
+
their inputs exist. The verdict is `ready` (all five passed), `incomplete`
|
|
22
|
+
(nothing failed, something skipped), `not-ready` (exit 1) or `abstained`
|
|
23
|
+
(nothing scored, exit 3).
|
|
24
|
+
|
|
17
25
|
## When to Use
|
|
18
26
|
|
|
19
27
|
- Before wiring a new test suite into CI for the first time
|
|
@@ -30,8 +38,9 @@ Run all five checks and score each pass / warn / fail.
|
|
|
30
38
|
|
|
31
39
|
### 1. Coverage depth
|
|
32
40
|
|
|
33
|
-
Read `.canary/test-inventory.json` if present.
|
|
34
|
-
|
|
41
|
+
Read `.canary/test-inventory.json` if present. Nothing in canary produces this
|
|
42
|
+
file yet (there is no `canary coverage` command), so when it is absent this
|
|
43
|
+
check is a `skip`, not a pass or a fail.
|
|
35
44
|
|
|
36
45
|
Default threshold: depth ≥ 2 for all endpoints in critical areas. Override with
|
|
37
46
|
`--threshold <n>`.
|
|
@@ -47,6 +56,11 @@ Default threshold: depth ≥ 2 for all endpoints in critical areas. Override wit
|
|
|
47
56
|
Read `test-results/quarantine-ledger.json` (or the path in
|
|
48
57
|
`.canary/company.json` under `quarantine_ledger_path` if set).
|
|
49
58
|
|
|
59
|
+
No tool writes a quarantine ledger yet, so `canary ci-ready` scores flakiness
|
|
60
|
+
from the run-history store instead: the last 30 runs in
|
|
61
|
+
`test-results/reports/history-v2.jsonl`. Any test flaking in 10% or more of its
|
|
62
|
+
runs fails the check, any lower flake rate warns, and no flakes passes.
|
|
63
|
+
|
|
50
64
|
A quarantined test is acceptable only when it has a linked open issue (Jira or
|
|
51
65
|
GitHub). Check issue state:
|
|
52
66
|
|
|
@@ -83,7 +97,10 @@ Cross-reference the top 5 risk-scored areas from `critical-areas.json` against
|
|
|
83
97
|
|
|
84
98
|
### 5. Suite runtime
|
|
85
99
|
|
|
86
|
-
|
|
100
|
+
Run history lives in `test-results/reports/history-v2.jsonl`. The store does not
|
|
101
|
+
record run or test durations today, so there is no p95 to compute and
|
|
102
|
+
`canary ci-ready` reports this check as `skip`. The scoring below applies once
|
|
103
|
+
durations are recorded.
|
|
87
104
|
|
|
88
105
|
**With harness MCP available:** score the p95 against trend history rather than
|
|
89
106
|
an absolute clock. Call `get_perf_baselines` and compare this run's p95 to the
|
|
@@ -75,6 +75,7 @@ of the full digest — cheaper and more focused:
|
|
|
75
75
|
| --------------------------------------- | -------------------------------------------------------------------- |
|
|
76
76
|
| Flaky tests fleet-wide | `canary analyze flaky --window-runs 30 --min-rate-pct 10 --json` |
|
|
77
77
|
| Failure spikes | `canary analyze spikes --delta-pp 20 --json` |
|
|
78
|
+
| CI-run flakes (reruns to green) | `canary analyze gh-flaky --repo <owner/name> --json` |
|
|
78
79
|
| Cross-suite common failures | `canary analyze common-failures --min-suites 2 --json` |
|
|
79
80
|
| Newly broken tests after a green streak | `canary analyze regression-candidates --json` |
|
|
80
81
|
| "Area health" / degrading areas | See the caveat below — this dimension does not currently return data |
|
|
@@ -58,6 +58,21 @@ canary guardian pr-check --format json
|
|
|
58
58
|
|
|
59
59
|
Findings are `untested-new-code` gaps. If there are none, report clean and stop.
|
|
60
60
|
|
|
61
|
+
**Coverage regressions (#606).** `--coverage` alone answers only "is this
|
|
62
|
+
changed unit covered at all?". To also catch a unit whose coverage _fell_
|
|
63
|
+
against the base branch, pass the base ref's report as well:
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
canary guardian pr-check --coverage lcov.info --base-coverage base-lcov.info --format json
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
That adds `coverage-regression` findings, graded by how many percentage points
|
|
70
|
+
were lost. Most CI never uploads a base-branch artifact; without
|
|
71
|
+
`--base-coverage` the run degrades **loudly** to "delta unavailable — head-only"
|
|
72
|
+
and `coverage_delta.status` reports `unavailable`. Read that field before
|
|
73
|
+
treating an empty finding list as "no regressions" — a run that compared nothing
|
|
74
|
+
has abstained, not passed.
|
|
75
|
+
|
|
61
76
|
### Phase 1 — Quality audit (Tier ≥ 1, read-only)
|
|
62
77
|
|
|
63
78
|
Export the availability signal so the probe reports the ceiling, then audit the
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-screech
|
|
3
|
+
description:
|
|
4
|
+
Broken-main siren. Reads the cross-run history store, decides whether the
|
|
5
|
+
default branch is red, and emits a one-page blast — culprit commit range,
|
|
6
|
+
failure cluster, owning area, a quarantine-or-revert recommendation, and a
|
|
7
|
+
chat-ready block — as a standalone markdown artifact plus a GitHub Actions
|
|
8
|
+
`::error` annotation. Self-contained; emits only, never posts.
|
|
9
|
+
cli: scripts/cli.mjs
|
|
10
|
+
requires: [node>=20]
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
# Canary Screech
|
|
14
|
+
|
|
15
|
+
The **cross-run, branch-level** member of the failure-surfacing family:
|
|
16
|
+
|
|
17
|
+
| Skill | Scope | Knows the branch went red? |
|
|
18
|
+
| ---------------------- | ----------------------- | -------------------------- |
|
|
19
|
+
| `canary-fail-fast` | in-run; aborts early | no |
|
|
20
|
+
| `canary-test-reporter` | per-run summary | no |
|
|
21
|
+
| **`canary-screech`** | cross-run, branch-level | **yes** |
|
|
22
|
+
|
|
23
|
+
"The default branch is red" is not a fact any single run holds. It lives in the
|
|
24
|
+
sequence of runs, which is why this skill reads the run-history store rather
|
|
25
|
+
than a results file.
|
|
26
|
+
|
|
27
|
+
## Signal source
|
|
28
|
+
|
|
29
|
+
`test-results/reports/history-v2.jsonl` — the run-history store, one `RunRecord`
|
|
30
|
+
JSON object per line. No network, no credentials, no service, and no write
|
|
31
|
+
access to anything except the `--out` artifact path. A GH Actions webhook and a
|
|
32
|
+
polling loop were both considered and rejected: this family's contract is
|
|
33
|
+
deterministic and self-contained, and the store already carries `branch`,
|
|
34
|
+
`commit_sha`, `timestamp`, the pass/fail counts, and per-test `area` and
|
|
35
|
+
`failure_category`.
|
|
36
|
+
|
|
37
|
+
## What it emits
|
|
38
|
+
|
|
39
|
+
1. A standalone **markdown one-pager** (`--out`) with five sections: culprit
|
|
40
|
+
commit range, failure cluster, owning area, recommendation, chat-ready block.
|
|
41
|
+
2. A single GitHub Actions **`::error` annotation**, so the break shows in the
|
|
42
|
+
Checks UI rather than only in a file.
|
|
43
|
+
|
|
44
|
+
It emits the chat block. It does not post it — there is no Slack, Teams, or
|
|
45
|
+
webhook integration, by design.
|
|
46
|
+
|
|
47
|
+
## Recommendation
|
|
48
|
+
|
|
49
|
+
| Verdict | When |
|
|
50
|
+
| ------------- | --------------------------------------------------------------------- |
|
|
51
|
+
| `revert` | every failure sits in one owning area, attributable to one commit |
|
|
52
|
+
| `quarantine` | failures span several areas, or the culprit range is not attributable |
|
|
53
|
+
| `investigate` | the failures carry no `area` at all |
|
|
54
|
+
|
|
55
|
+
`investigate` is deliberate: a confident recommendation derived from absent data
|
|
56
|
+
is worse than no recommendation.
|
|
57
|
+
|
|
58
|
+
## Abstention
|
|
59
|
+
|
|
60
|
+
A store with no run for the requested branch is a **zero denominator**. The
|
|
61
|
+
skill prints a loud `ABSTAINED` line and never the green copy — "main looks
|
|
62
|
+
fine" derived from zero observations is the false-green this repo keeps
|
|
63
|
+
re-learning. Under `--strict` that abstention exits `3`.
|
|
64
|
+
|
|
65
|
+
## Invocation
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
# Advisory (exit 0 whatever it finds) — the default:
|
|
69
|
+
canary skills run canary-screech -- --history test-results/reports/history-v2.jsonl
|
|
70
|
+
|
|
71
|
+
# Watch a non-default branch and write the artifact:
|
|
72
|
+
canary skills run canary-screech -- \
|
|
73
|
+
--history test-results/reports/history-v2.jsonl \
|
|
74
|
+
--branch release/7.x \
|
|
75
|
+
--out test-results/reports/screech.md
|
|
76
|
+
|
|
77
|
+
# Gate a workflow on it:
|
|
78
|
+
canary skills run canary-screech -- \
|
|
79
|
+
--history test-results/reports/history-v2.jsonl --strict
|
|
80
|
+
|
|
81
|
+
# Usage and the full flag list (exits 0):
|
|
82
|
+
canary skills run canary-screech -- --help
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
`--history` is required. `--branch` defaults to `main`.
|
|
86
|
+
|
|
87
|
+
### Exit codes
|
|
88
|
+
|
|
89
|
+
| Code | Meaning |
|
|
90
|
+
| ---- | ------------------------------------------------- |
|
|
91
|
+
| `0` | advisory mode always; or `--strict` and green |
|
|
92
|
+
| `1` | `--strict` and the branch is red; or a read error |
|
|
93
|
+
| `2` | usage error |
|
|
94
|
+
| `3` | `--strict` and the skill abstained |
|
|
95
|
+
|
|
96
|
+
## CI wiring (GitHub Actions)
|
|
97
|
+
|
|
98
|
+
Run it on a schedule or after the default-branch test job, with `if: always()`
|
|
99
|
+
so a failed test step does not suppress the siren:
|
|
100
|
+
|
|
101
|
+
```yaml
|
|
102
|
+
- name: Broken-main siren
|
|
103
|
+
if: always()
|
|
104
|
+
run: |
|
|
105
|
+
canary skills run canary-screech -- \
|
|
106
|
+
--history test-results/reports/history-v2.jsonl \
|
|
107
|
+
--out test-results/reports/screech.md \
|
|
108
|
+
--strict
|
|
109
|
+
```
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
// blast -- render the one-page broken-main blast. Pure.
|
|
2
|
+
//
|
|
3
|
+
// Three outputs from one assessment, because the same fact has three audiences:
|
|
4
|
+
// markdown the standalone artifact a human opens
|
|
5
|
+
// annotations a single `::error` line so the GitHub Checks UI shows it
|
|
6
|
+
// chatBlock plain text the on-call pastes into whatever chat they use
|
|
7
|
+
//
|
|
8
|
+
// The chat block is emitted, never posted. The skill has no credentials, no
|
|
9
|
+
// webhook, and no write access to anything but the artifact path it was given.
|
|
10
|
+
|
|
11
|
+
const CROSS = '\u274c'; // cross mark
|
|
12
|
+
const CHECK = '\u2705'; // white heavy check mark
|
|
13
|
+
const DASH = '\u2014'; // em dash
|
|
14
|
+
|
|
15
|
+
/** The first meaningful line of an error blob, clipped. */
|
|
16
|
+
function firstLine(error, limit = 160) {
|
|
17
|
+
if (!error) return '(no error message)';
|
|
18
|
+
for (const raw of String(error).split(/\r\n|\r|\n/)) {
|
|
19
|
+
const line = raw.trim();
|
|
20
|
+
if (line) return line.slice(0, limit);
|
|
21
|
+
}
|
|
22
|
+
return '(no error message)';
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const NEXT_STEP = {
|
|
26
|
+
revert: 'Revert the culprit commit. One area, one attributable commit.',
|
|
27
|
+
quarantine:
|
|
28
|
+
'Quarantine the failing tests and open a tracking issue. The break is not cleanly attributable to one commit.',
|
|
29
|
+
investigate:
|
|
30
|
+
'Investigate before acting. The failing tests carry no owning area, so neither a revert nor a quarantine can be aimed.',
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Render the blast.
|
|
35
|
+
*
|
|
36
|
+
* @param {{branch: string, assessment: object, cluster: object|null}} input
|
|
37
|
+
* @returns {{markdown: string, annotations: string[], chatBlock: string}}
|
|
38
|
+
*/
|
|
39
|
+
export function renderBlast({ branch, assessment, cluster }) {
|
|
40
|
+
if (assessment.state === 'abstained') {
|
|
41
|
+
const text =
|
|
42
|
+
`${CROSS} canary-screech ABSTAINED ${DASH} the run-history store holds no run for \`${branch}\`.\n\n` +
|
|
43
|
+
'Zero observations is not a healthy branch. Point `--history` at a store ' +
|
|
44
|
+
'that this branch actually writes to, or wire the suite to record runs.';
|
|
45
|
+
return {
|
|
46
|
+
markdown: `# canary-screech ${DASH} ${branch}\n\n${text}\n`,
|
|
47
|
+
annotations: [],
|
|
48
|
+
chatBlock: text,
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
if (assessment.state === 'green') {
|
|
53
|
+
const text = `${CHECK} \`${branch}\` is green as of ${assessment.latest.commit_sha ?? '(unknown commit)'}.`;
|
|
54
|
+
return {
|
|
55
|
+
markdown: `# canary-screech ${DASH} ${branch}\n\n${text}\n`,
|
|
56
|
+
annotations: [],
|
|
57
|
+
chatBlock: text,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const { culpritRange, firstRed } = assessment;
|
|
62
|
+
const from =
|
|
63
|
+
culpritRange.from ?? '(unknown - no green run recorded before the break)';
|
|
64
|
+
const to = culpritRange.to ?? '(unknown)';
|
|
65
|
+
const failureCount = cluster.clusters.reduce((n, c) => n + c.tests.length, 0);
|
|
66
|
+
|
|
67
|
+
const chatLines = [
|
|
68
|
+
`${CROSS} ${branch} is RED ${DASH} ${failureCount} failing test${failureCount === 1 ? '' : 's'}`,
|
|
69
|
+
`culprit range: ${from}..${to}`,
|
|
70
|
+
`owning area: ${cluster.owningArea ?? '(none recorded)'}`,
|
|
71
|
+
`recommendation: ${cluster.recommendation.toUpperCase()} ${DASH} ${NEXT_STEP[cluster.recommendation]}`,
|
|
72
|
+
`first red run: ${firstRed.run_id} (${firstRed.suite}) at ${firstRed.timestamp}`,
|
|
73
|
+
];
|
|
74
|
+
const chatBlock = chatLines.join('\n');
|
|
75
|
+
|
|
76
|
+
const lines = [
|
|
77
|
+
`# ${CROSS} canary-screech ${DASH} \`${branch}\` is red`,
|
|
78
|
+
'',
|
|
79
|
+
`**First red run:** \`${firstRed.run_id}\` (${firstRed.suite}) at ${firstRed.timestamp}`,
|
|
80
|
+
'',
|
|
81
|
+
'## Culprit commit range',
|
|
82
|
+
'',
|
|
83
|
+
`\`${from}\` .. \`${to}\``,
|
|
84
|
+
'',
|
|
85
|
+
];
|
|
86
|
+
if (!culpritRange.bounded) {
|
|
87
|
+
lines.push(
|
|
88
|
+
'> The lower bound is unknown: the store holds no green run for this ' +
|
|
89
|
+
'branch before the break, so the range below is open-ended.',
|
|
90
|
+
'',
|
|
91
|
+
);
|
|
92
|
+
}
|
|
93
|
+
lines.push('## Failure cluster', '');
|
|
94
|
+
for (const c of cluster.clusters) {
|
|
95
|
+
lines.push(`### ${c.category} (${c.tests.length})`, '');
|
|
96
|
+
for (const t of c.tests) {
|
|
97
|
+
lines.push(`- \`${t.test_name}\` ${DASH} ${firstLine(t.error_text)}`);
|
|
98
|
+
}
|
|
99
|
+
lines.push('');
|
|
100
|
+
}
|
|
101
|
+
lines.push(
|
|
102
|
+
'## Owning area',
|
|
103
|
+
'',
|
|
104
|
+
cluster.areas.length
|
|
105
|
+
? cluster.areas.map((a) => `- \`${a.area}\` (${a.count})`).join('\n')
|
|
106
|
+
: '_No failing test carries an `area`, so ownership could not be derived._',
|
|
107
|
+
'',
|
|
108
|
+
'## Recommendation',
|
|
109
|
+
'',
|
|
110
|
+
`**${cluster.recommendation.toUpperCase()}** ${DASH} ${NEXT_STEP[cluster.recommendation]}`,
|
|
111
|
+
'',
|
|
112
|
+
'## Chat-ready block',
|
|
113
|
+
'',
|
|
114
|
+
'```text',
|
|
115
|
+
chatBlock,
|
|
116
|
+
'```',
|
|
117
|
+
'',
|
|
118
|
+
);
|
|
119
|
+
|
|
120
|
+
const annotations = [
|
|
121
|
+
`::error title=Broken branch::${branch} is red ${DASH} ${failureCount} failing test${failureCount === 1 ? '' : 's'} in ${cluster.owningArea ?? 'an unrecorded area'}; culprit ${from}..${to}; recommendation: ${cluster.recommendation}`,
|
|
122
|
+
];
|
|
123
|
+
|
|
124
|
+
return { markdown: lines.join('\n'), annotations, chatBlock };
|
|
125
|
+
}
|