canary-test-cli 7.2.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/agents/skills/README.md +23 -4
  2. package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
  3. package/agents/skills/claude-code/canary-cassandra/SKILL.md +23 -16
  4. package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +3 -1
  5. package/agents/skills/claude-code/canary-ci-ready/SKILL.md +20 -3
  6. package/agents/skills/claude-code/canary-fleet-health/SKILL.md +1 -0
  7. package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +15 -0
  8. package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
  9. package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
  10. package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
  11. package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
  12. package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
  13. package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
  14. package/agents/skills/lib/parse-args.mjs +200 -139
  15. package/dist/engine/analysis/batwoman/audit.js +39 -0
  16. package/dist/engine/analysis/batwoman/closure.js +159 -0
  17. package/dist/engine/analysis/batwoman/gh-history.js +119 -0
  18. package/dist/engine/analysis/batwoman/probes.js +195 -0
  19. package/dist/engine/analysis/batwoman/registry.js +142 -0
  20. package/dist/engine/analysis/batwoman/render.js +194 -0
  21. package/dist/engine/analysis/batwoman/run-window.js +122 -0
  22. package/dist/engine/analysis/batwoman/text.js +84 -0
  23. package/dist/engine/analysis/batwoman/triggers.js +122 -0
  24. package/dist/engine/analysis/batwoman/verdict.js +64 -0
  25. package/dist/engine/analysis/cli.js +47 -14
  26. package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
  27. package/dist/engine/batwoman-cli.js +119 -0
  28. package/dist/engine/ci-ready-cli.js +71 -0
  29. package/dist/engine/cli-commands.js +46 -7
  30. package/dist/engine/cli.core.js +16 -0
  31. package/dist/engine/company-knowledge-cli.js +10 -2
  32. package/dist/engine/core/ci-ready.js +112 -0
  33. package/dist/engine/core/company-knowledge.js +8 -0
  34. package/dist/engine/core/migrator.js +147 -20
  35. package/dist/engine/core/permission-matrix.js +219 -0
  36. package/dist/engine/core/quality-scorer.js +13 -18
  37. package/dist/engine/core/scaling-curve.js +143 -0
  38. package/dist/engine/core/string-literals.js +3 -1
  39. package/dist/engine/core/vacuity-scanner.js +151 -6
  40. package/dist/engine/core/workflow-discovery.js +41 -23
  41. package/dist/engine/guardian/adjudication-github.js +136 -0
  42. package/dist/engine/guardian/adjudication.js +119 -340
  43. package/dist/engine/guardian/cli.js +180 -264
  44. package/dist/engine/guardian/coverage.js +2 -1
  45. package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
  46. package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
  47. package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
  48. package/dist/engine/guardian/diff-coverage/paths.js +5 -9
  49. package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
  50. package/dist/engine/guardian/diff-extractor.js +31 -32
  51. package/dist/engine/guardian/pr-check.js +262 -430
  52. package/dist/engine/guardian/pr-comment.js +35 -58
  53. package/dist/engine/guardian/weak-test.js +236 -0
  54. package/dist/engine/mcp-server.js +67 -4
  55. package/dist/engine/permission-matrix-cli.js +51 -0
  56. package/dist/engine/scaling-curve-cli.js +147 -0
  57. package/dist/engine/skills-cli.js +48 -32
  58. package/dist/engine/workflow-cli.js +85 -65
  59. package/package.json +1 -1
@@ -13,8 +13,9 @@ it), see [Guides](../../docs/guides/index.md).
13
13
 
14
14
  ```text
15
15
  agents/skills/
16
- ├── claude-code/ # Claude Code skills (21)
16
+ ├── claude-code/ # Claude Code skills (24)
17
17
  │ ├── canary-add-framework/
18
+ │ ├── canary-batwoman/
18
19
  │ ├── canary-blackhawk/
19
20
  │ ├── canary-cassandra/
20
21
  │ ├── canary-ci-ready/
@@ -30,9 +31,11 @@ agents/skills/
30
31
  │ ├── canary-pr-guardian/
31
32
  │ ├── canary-promote-test/
32
33
  │ ├── canary-savant/
34
+ │ ├── canary-screech/
33
35
  │ ├── canary-setup-harness/
34
36
  │ ├── canary-shadow/
35
37
  │ ├── canary-ship/
38
+ │ ├── canary-strix/
36
39
  │ ├── canary-test-pipeline/
37
40
  │ └── canary-test-reporter/
38
41
  └── README.md # this file
@@ -90,6 +93,21 @@ slash-command entry points.
90
93
  into a Markdown and/or JSON report with pass/fail/flaky/skipped counts.
91
94
  Complements `canary-fail-fast` (which aborts early) by summarising the full
92
95
  run at the end.
96
+ - [`canary-screech`](./claude-code/canary-screech/SKILL.md) — Bundled executable
97
+ skill (`scripts/cli.mjs`). Broken-main siren: reads the cross-run history
98
+ store, decides whether the default branch is red, and emits a one-page blast
99
+ (culprit commit range, failure cluster, owning area, quarantine-or-revert
100
+ recommendation, chat-ready block) as a markdown artifact plus a `::error`
101
+ annotation. The cross-run complement to the two above — neither of them can
102
+ tell that the branch itself went red.
103
+
104
+ ### Closure auditing
105
+
106
+ - [`canary-batwoman`](./claude-code/canary-batwoman/SKILL.md) — Reports whether
107
+ the files a closed issue's fix changed have actually **executed** since that
108
+ fix merged. GitHub closes an issue on a keyword match, which checks neither
109
+ that the fix works nor that it ever ran. Advisory, and the only skill here
110
+ that requires the network (`gh`): deterministic, network, no agent.
93
111
 
94
112
  ### Test hygiene & reliability
95
113
 
@@ -204,9 +222,10 @@ Use the canary-generate-test skill to write a load test for /v1/search.
204
222
  Most skills here are documentation, not executable artifacts — they describe
205
223
  _how an agent should behave_, not a function to call. Several are bundled
206
224
  executable skills with their own CLI entry point (`cli:` in frontmatter).
207
- `canary-fail-fast`, `canary-katana`, and `canary-blackhawk` ship a Node entry
208
- (`scripts/cli.mjs`); `canary-instrument` and `canary-test-reporter` ship a
209
- Python entry (`scripts/cli.py`). Run those directly, e.g.:
225
+ `canary-fail-fast`, `canary-katana`, `canary-screech`, and `canary-blackhawk`
226
+ ship a Node entry (`scripts/cli.mjs`); `canary-instrument` and
227
+ `canary-test-reporter` ship a Python entry (`scripts/cli.py`). Run those
228
+ directly, e.g.:
210
229
 
211
230
  ```bash
212
231
  node agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs --help
@@ -0,0 +1,119 @@
1
+ ---
2
+ name: canary-batwoman
3
+ description:
4
+ Closure auditing — reports whether the files a closed issue's fix changed have
5
+ actually EXECUTED since that fix merged. GitHub closes an issue on a keyword
6
+ match in a PR body, which checks neither that the fix works nor that it ever
7
+ ran; batwoman answers only the second question, per changed file, and names
8
+ the files it could not answer for. Use when the user asks "did that fix
9
+ actually run", "is this issue really done", "audit a closed issue", or after a
10
+ batch of merges. Advisory and read-only — it never asserts correctness, never
11
+ fails a job, and never reopens an issue. NOT a test runner, NOT a coverage
12
+ tool (a dormant workflow has whatever coverage it always had), and NOT a
13
+ correctness check.
14
+ cli: canary batwoman
15
+ requires: [node>=20, gh]
16
+ ---
17
+
18
+ # Canary Batwoman
19
+
20
+ GitHub closes an issue when a merged PR body matches `Closes #N`. That is a
21
+ **string match with no denominator**: nothing checks that the fix works, and
22
+ nothing checks that the fix ever ran. The issue moves to `CLOSED` on the
23
+ strength of a keyword.
24
+
25
+ Batwoman audits the claim. For each file the closing PR changed, it reports
26
+ whether that artifact has executed since the merge — and, just as importantly,
27
+ which files it could not decide about.
28
+
29
+ ## The founding case
30
+
31
+ canary#749 fixed a false-green in `.github/workflows/refresh-arch-baseline.yml`.
32
+ It merged as `1e0c05b` and auto-closed on the keyword. Verified afterwards:
33
+
34
+ - 11 script unit tests passed
35
+ - 97 static workflow assertions passed
36
+ - the workflow itself **had not run since twelve days before the fix merged** —
37
+ it is label-triggered, so it stayed dormant until someone applied the label
38
+
39
+ Every gate the repo owns read that as done. The fix was real and well-tested;
40
+ its end-to-end path was unproven. **Verified and exercised are different
41
+ properties**, and that gap is what batwoman makes visible.
42
+
43
+ ## Usage
44
+
45
+ ```bash
46
+ canary batwoman --help
47
+ ```
48
+
49
+ That one needs no credentials, no network and no fixtures. A real audit needs
50
+ the repository named and `gh` authenticated:
51
+
52
+ ```bash
53
+ export GITHUB_REPOSITORY=owner/name # never inferred from a git remote
54
+ canary batwoman --issue 749 # the report
55
+ canary batwoman --issue 749 --json # the machine shape, for CI
56
+ ```
57
+
58
+ `GITHUB_REPOSITORY` is required rather than guessed. Auditing the wrong
59
+ repository's run history would produce a confident answer about the wrong thing,
60
+ which is worse than refusing.
61
+
62
+ ## What the five statuses mean
63
+
64
+ | status | meaning |
65
+ | ---------------- | -------------------------------------------------------------------------- |
66
+ | `exercised` | a run started after the merge, and it names which |
67
+ | `not-exercised` | the artifact has not run since — with the `on:` trigger explaining why |
68
+ | `abstain` | a probe looked and could not tell (unreadable history, untraceable script) |
69
+ | `no-probe` | nothing looked: a registry gap, fixable by adding a probe |
70
+ | `not-applicable` | prose, config, or a file this change deleted |
71
+
72
+ `abstain` and `no-probe` are deliberately separate, and neither is folded into
73
+ anything resembling a pass. **The summary always prints all five counts**, and
74
+ they always sum to the changed-file total — a report over a subset presented as
75
+ a report over the whole is the exact defect batwoman exists to detect.
76
+
77
+ There is no `assessed` field, no score, and no success token anywhere in the
78
+ output. A convenient `passed: true` in the JSON is the field a CI wrapper would
79
+ grow later; it does not exist, and a test asserts it stays that way.
80
+
81
+ ## What it will not tell you
82
+
83
+ - **Whether the fix is correct.** Batwoman proves execution, never behaviour. A
84
+ workflow that ran and did the wrong thing reports `exercised`.
85
+ - **Whether coverage is adequate.** A dormant workflow has whatever coverage it
86
+ always had; that is the point.
87
+ - **Anything about a file it has no probe for.** `ts/src/**` is an honest
88
+ `no-probe` row in v1, countable rather than silent.
89
+
90
+ ## Probes shipped
91
+
92
+ | probe | matches | decides by |
93
+ | ----------------- | ------------------------- | ------------------------------------------------------------------------ |
94
+ | `workflow` | `.github/workflows/*.yml` | a run created after the merge; reads `on:` to explain a dormant one |
95
+ | `workflow-script` | `scripts/*.mjs` | the workflows that name it, inheriting `exercised` if **any** caller ran |
96
+ | `no-execution` | `*.md`, config manifests | returns `not-applicable`, reading nothing |
97
+
98
+ A script nothing references **abstains** rather than reporting as never-run: "no
99
+ workflow calls this" is a statement about the repo, not about the script, which
100
+ may still be run by a hand or a hook.
101
+
102
+ ## Requires the network, and says so
103
+
104
+ Batwoman shells out to `gh` for run history and for the closing PR. That is a
105
+ property, not a tier — unlike the deterministic offline detectors alongside it
106
+ (`canary-savant`, `canary-blackhawk`, `canary-cassandra`, `canary-katana`),
107
+ which run anywhere node does.
108
+
109
+ An unauthenticated or missing `gh` does not degrade to a clean report. It fails
110
+ loudly, and every affected file becomes an `abstain` naming the failure. Per
111
+ this repo's standing rule, **cannot-verify is a finding, not a skip**.
112
+
113
+ ## In CI
114
+
115
+ `.github/workflows/batwoman.yml` runs it on every push to `main`, deriving the
116
+ issue from the merge commit's closing keyword — the very keyword match batwoman
117
+ questions, which is the right input precisely because it is what GitHub acted
118
+ on. Advisory: every step is `continue-on-error`, and a merge that closes nothing
119
+ is reported and skipped rather than passing silently over an empty set.
@@ -3,14 +3,15 @@ name: canary-cassandra
3
3
  description: >
4
4
  Vacuous-test detection — finds tests that PASS WITHOUT PROVING ANYTHING: an
5
5
  assertion that compares a value with itself, a test that never invokes the
6
- target it claims to cover, and a test whose every assertion is an absence
7
- observed on a bystander rather than on the code under test. Use when the user
8
- says "why did this pass against the bug", "are these tests actually testing
9
- anything", "audit my suite for vacuous tests", "green but worthless", or after
10
- a bug shipped through a green suite. Advisory and deterministic — no LLM, no
11
- execution. NOT for tests with zero assertions (that is `canary review-test`'s
12
- LINT-006), NOT for flaky tests (canary-flake-hunter), and NOT a coverage tool
13
- — a vacuous test has coverage, which is exactly why coverage never caught it.
6
+ target it claims to cover, and a test whose every assertion is an absence, or
7
+ a trivially true presence check, observed on a bystander rather than on the
8
+ code under test. Use when the user says "why did this pass against the bug",
9
+ "are these tests actually testing anything", "audit my suite for vacuous
10
+ tests", "green but worthless", or after a bug shipped through a green suite.
11
+ Advisory and deterministic — no LLM, no execution. NOT for tests with zero
12
+ assertions (that is `canary review-test`'s LINT-006), NOT for flaky tests
13
+ (canary-flake-hunter), and NOT a coverage tool — a vacuous test has coverage,
14
+ which is exactly why coverage never caught it.
14
15
  cli: scripts/cli.mjs
15
16
  requires: [node>=20]
16
17
  ---
@@ -29,15 +30,20 @@ Cassandra is Tier-0: deterministic, no LLM, no network, no execution.
29
30
 
30
31
  ## What it finds
31
32
 
32
- | Rule | Severity | Fires on |
33
- | --------- | -------- | ----------------------------------------------------------------------------------------------------------------- |
34
- | `VAC-001` | critical | An assertion whose expectation is identical to the value it checks — `expect(true).toBe(true)`, `assert x == x` |
35
- | `VAC-002` | warning | The test never references the target it claims to cover |
36
- | `VAC-003` | warning | Every assertion in the test asserts an _absence_, and none of them observes the target — so nothing proves it ran |
33
+ | Rule | Severity | Fires on |
34
+ | --------- | -------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- |
35
+ | `VAC-001` | critical | An assertion whose expectation is identical to the value it checks — `expect(true).toBe(true)`, `assert x == x` |
36
+ | `VAC-002` | warning | The test never references the target it claims to cover |
37
+ | `VAC-003` | warning | Every assertion in the test asserts an _absence_, and none of them observes the target — so nothing proves it ran |
38
+ | `VAC-005` | warning | Every assertion is a trivially true _presence_ check (`toBeDefined`, `toBeTruthy`, `assert x is not None`) on a value the test built itself before the target ran |
37
39
 
38
40
  `VAC-001` is deterministic, hence `critical`: no implementation can fail it.
39
- `VAC-002` and `VAC-003` depend on resolving a target, which is inference, so
40
- they are `warning` and carry a fidelity tier.
41
+ `VAC-005` abstains (no finding) whenever it cannot prove the subject is a
42
+ bystander, such as a name bound in a hook or a multi-line initialiser. `VAC-004`
43
+ is reserved for the self-excusing-skip rule
44
+ (`docs/changes/vac-004-self-excusing-skip/`). `VAC-002` and `VAC-003` depend on
45
+ resolving a target, which is inference, so they are `warning` and carry a
46
+ fidelity tier.
41
47
 
42
48
  ## Run it
43
49
 
@@ -45,7 +51,8 @@ Two doors, one detector. Both run the same engine rules, so they cannot disagree
45
51
  about a finding or about the denominator.
46
52
 
47
53
  ```bash
48
- canary vacuity-check tests/ # human-readable
54
+ canary vacuity-check tests/ # human-readable, skips counted per reason
55
+ canary vacuity-check tests/ --verbose # ...plus every skipped test at file:line
49
56
  canary vacuity-check tests/ --json # verdict + denominator + skips
50
57
  canary vacuity-check tests/a.test.ts # one file
51
58
  ```
@@ -4,7 +4,8 @@
4
4
  // Finds tests that PASS WITHOUT PROVING ANYTHING: an assertion that compares a
5
5
  // value with itself (VAC-001), a test that never invokes the target it claims
6
6
  // to cover (VAC-002), and a test whose every assertion is an absence observed
7
- // on a bystander (VAC-003).
7
+ // on a bystander (VAC-003), and a test whose every assertion is a trivially
8
+ // true presence check on a value the test built itself (VAC-005).
8
9
  //
9
10
  // <paths> files or directories to scan (default: the current directory).
10
11
  // --json emit machine-readable findings instead of human text.
@@ -53,6 +54,7 @@ const USAGE =
53
54
  ' VAC-001 (critical) assertion compares a value with itself\n' +
54
55
  ' VAC-002 (warning) the test never invokes the target it covers\n' +
55
56
  ' VAC-003 (warning) every assertion is an absence, on a bystander\n' +
57
+ ' VAC-005 (warning) every assertion is trivial presence, on a bystander\n' +
56
58
  '\n' +
57
59
  'The denominator is TESTS read, not files. A zero denominator exits 3 under\n' +
58
60
  '--strict; it is never reported as a clean scan.';
@@ -14,6 +14,14 @@ Analyses a test suite across five dimensions and produces a readiness score. Use
14
14
  this before promoting a suite to CI, or as the convergence gate in
15
15
  `/canary-test-pipeline`.
16
16
 
17
+ **Deterministic scorer:** run `canary ci-ready [--root <dir>] [--json]` first.
18
+ It scores every check that has a real input and reports `skip`, naming the
19
+ missing input, for every check that does not. A skip is never a pass. Today only
20
+ flakiness has a producer behind it, so expect the other four to skip until
21
+ their inputs exist. The verdict is `ready` (all five passed), `incomplete`
22
+ (nothing failed, something skipped), `not-ready` (exit 1) or `abstained`
23
+ (nothing scored, exit 3).
24
+
17
25
  ## When to Use
18
26
 
19
27
  - Before wiring a new test suite into CI for the first time
@@ -30,8 +38,9 @@ Run all five checks and score each pass / warn / fail.
30
38
 
31
39
  ### 1. Coverage depth
32
40
 
33
- Read `.canary/test-inventory.json` if present. If absent or older than 7 days,
34
- run `canary coverage` to generate fresh data.
41
+ Read `.canary/test-inventory.json` if present. Nothing in canary produces this
42
+ file yet (there is no `canary coverage` command), so when it is absent this
43
+ check is a `skip`, not a pass or a fail.
35
44
 
36
45
  Default threshold: depth ≥ 2 for all endpoints in critical areas. Override with
37
46
  `--threshold <n>`.
@@ -47,6 +56,11 @@ Default threshold: depth ≥ 2 for all endpoints in critical areas. Override wit
47
56
  Read `test-results/quarantine-ledger.json` (or the path in
48
57
  `.canary/company.json` under `quarantine_ledger_path` if set).
49
58
 
59
+ No tool writes a quarantine ledger yet, so `canary ci-ready` scores flakiness
60
+ from the run-history store instead: the last 30 runs in
61
+ `test-results/reports/history-v2.jsonl`. Any test flaking in 10% or more of its
62
+ runs fails the check, any lower flake rate warns, and no flakes passes.
63
+
50
64
  A quarantined test is acceptable only when it has a linked open issue (Jira or
51
65
  GitHub). Check issue state:
52
66
 
@@ -83,7 +97,10 @@ Cross-reference the top 5 risk-scored areas from `critical-areas.json` against
83
97
 
84
98
  ### 5. Suite runtime
85
99
 
86
- Read `test-results/run-history.ndjson`. Use the p95 of the last 10 runs.
100
+ Run history lives in `test-results/reports/history-v2.jsonl`. The store does not
101
+ record run or test durations today, so there is no p95 to compute and
102
+ `canary ci-ready` reports this check as `skip`. The scoring below applies once
103
+ durations are recorded.
87
104
 
88
105
  **With harness MCP available:** score the p95 against trend history rather than
89
106
  an absolute clock. Call `get_perf_baselines` and compare this run's p95 to the
@@ -75,6 +75,7 @@ of the full digest — cheaper and more focused:
75
75
  | --------------------------------------- | -------------------------------------------------------------------- |
76
76
  | Flaky tests fleet-wide | `canary analyze flaky --window-runs 30 --min-rate-pct 10 --json` |
77
77
  | Failure spikes | `canary analyze spikes --delta-pp 20 --json` |
78
+ | CI-run flakes (reruns to green) | `canary analyze gh-flaky --repo <owner/name> --json` |
78
79
  | Cross-suite common failures | `canary analyze common-failures --min-suites 2 --json` |
79
80
  | Newly broken tests after a green streak | `canary analyze regression-candidates --json` |
80
81
  | "Area health" / degrading areas | See the caveat below — this dimension does not currently return data |
@@ -58,6 +58,21 @@ canary guardian pr-check --format json
58
58
 
59
59
  Findings are `untested-new-code` gaps. If there are none, report clean and stop.
60
60
 
61
+ **Coverage regressions (#606).** `--coverage` alone answers only "is this
62
+ changed unit covered at all?". To also catch a unit whose coverage _fell_
63
+ against the base branch, pass the base ref's report as well:
64
+
65
+ ```bash
66
+ canary guardian pr-check --coverage lcov.info --base-coverage base-lcov.info --format json
67
+ ```
68
+
69
+ That adds `coverage-regression` findings, graded by how many percentage points
70
+ were lost. Most CI never uploads a base-branch artifact; without
71
+ `--base-coverage` the run degrades **loudly** to "delta unavailable — head-only"
72
+ and `coverage_delta.status` reports `unavailable`. Read that field before
73
+ treating an empty finding list as "no regressions" — a run that compared nothing
74
+ has abstained, not passed.
75
+
61
76
  ### Phase 1 — Quality audit (Tier ≥ 1, read-only)
62
77
 
63
78
  Export the availability signal so the probe reports the ceiling, then audit the
@@ -0,0 +1,109 @@
1
+ ---
2
+ name: canary-screech
3
+ description:
4
+ Broken-main siren. Reads the cross-run history store, decides whether the
5
+ default branch is red, and emits a one-page blast — culprit commit range,
6
+ failure cluster, owning area, a quarantine-or-revert recommendation, and a
7
+ chat-ready block — as a standalone markdown artifact plus a GitHub Actions
8
+ `::error` annotation. Self-contained; emits only, never posts.
9
+ cli: scripts/cli.mjs
10
+ requires: [node>=20]
11
+ ---
12
+
13
+ # Canary Screech
14
+
15
+ The **cross-run, branch-level** member of the failure-surfacing family:
16
+
17
+ | Skill | Scope | Knows the branch went red? |
18
+ | ---------------------- | ----------------------- | -------------------------- |
19
+ | `canary-fail-fast` | in-run; aborts early | no |
20
+ | `canary-test-reporter` | per-run summary | no |
21
+ | **`canary-screech`** | cross-run, branch-level | **yes** |
22
+
23
+ "The default branch is red" is not a fact any single run holds. It lives in the
24
+ sequence of runs, which is why this skill reads the run-history store rather
25
+ than a results file.
26
+
27
+ ## Signal source
28
+
29
+ `test-results/reports/history-v2.jsonl` — the run-history store, one `RunRecord`
30
+ JSON object per line. No network, no credentials, no service, and no write
31
+ access to anything except the `--out` artifact path. A GH Actions webhook and a
32
+ polling loop were both considered and rejected: this family's contract is
33
+ deterministic and self-contained, and the store already carries `branch`,
34
+ `commit_sha`, `timestamp`, the pass/fail counts, and per-test `area` and
35
+ `failure_category`.
36
+
37
+ ## What it emits
38
+
39
+ 1. A standalone **markdown one-pager** (`--out`) with five sections: culprit
40
+ commit range, failure cluster, owning area, recommendation, chat-ready block.
41
+ 2. A single GitHub Actions **`::error` annotation**, so the break shows in the
42
+ Checks UI rather than only in a file.
43
+
44
+ It emits the chat block. It does not post it — there is no Slack, Teams, or
45
+ webhook integration, by design.
46
+
47
+ ## Recommendation
48
+
49
+ | Verdict | When |
50
+ | ------------- | --------------------------------------------------------------------- |
51
+ | `revert` | every failure sits in one owning area, attributable to one commit |
52
+ | `quarantine` | failures span several areas, or the culprit range is not attributable |
53
+ | `investigate` | the failures carry no `area` at all |
54
+
55
+ `investigate` is deliberate: a confident recommendation derived from absent data
56
+ is worse than no recommendation.
57
+
58
+ ## Abstention
59
+
60
+ A store with no run for the requested branch is a **zero denominator**. The
61
+ skill prints a loud `ABSTAINED` line and never the green copy — "main looks
62
+ fine" derived from zero observations is the false-green this repo keeps
63
+ re-learning. Under `--strict` that abstention exits `3`.
64
+
65
+ ## Invocation
66
+
67
+ ```bash
68
+ # Advisory (exit 0 whatever it finds) — the default:
69
+ canary skills run canary-screech -- --history test-results/reports/history-v2.jsonl
70
+
71
+ # Watch a non-default branch and write the artifact:
72
+ canary skills run canary-screech -- \
73
+ --history test-results/reports/history-v2.jsonl \
74
+ --branch release/7.x \
75
+ --out test-results/reports/screech.md
76
+
77
+ # Gate a workflow on it:
78
+ canary skills run canary-screech -- \
79
+ --history test-results/reports/history-v2.jsonl --strict
80
+
81
+ # Usage and the full flag list (exits 0):
82
+ canary skills run canary-screech -- --help
83
+ ```
84
+
85
+ `--history` is required. `--branch` defaults to `main`.
86
+
87
+ ### Exit codes
88
+
89
+ | Code | Meaning |
90
+ | ---- | ------------------------------------------------- |
91
+ | `0` | advisory mode always; or `--strict` and green |
92
+ | `1` | `--strict` and the branch is red; or a read error |
93
+ | `2` | usage error |
94
+ | `3` | `--strict` and the skill abstained |
95
+
96
+ ## CI wiring (GitHub Actions)
97
+
98
+ Run it on a schedule or after the default-branch test job, with `if: always()`
99
+ so a failed test step does not suppress the siren:
100
+
101
+ ```yaml
102
+ - name: Broken-main siren
103
+ if: always()
104
+ run: |
105
+ canary skills run canary-screech -- \
106
+ --history test-results/reports/history-v2.jsonl \
107
+ --out test-results/reports/screech.md \
108
+ --strict
109
+ ```
@@ -0,0 +1,125 @@
1
+ // blast -- render the one-page broken-main blast. Pure.
2
+ //
3
+ // Three outputs from one assessment, because the same fact has three audiences:
4
+ // markdown the standalone artifact a human opens
5
+ // annotations a single `::error` line so the GitHub Checks UI shows it
6
+ // chatBlock plain text the on-call pastes into whatever chat they use
7
+ //
8
+ // The chat block is emitted, never posted. The skill has no credentials, no
9
+ // webhook, and no write access to anything but the artifact path it was given.
10
+
11
+ const CROSS = '\u274c'; // cross mark
12
+ const CHECK = '\u2705'; // white heavy check mark
13
+ const DASH = '\u2014'; // em dash
14
+
15
+ /** The first meaningful line of an error blob, clipped. */
16
+ function firstLine(error, limit = 160) {
17
+ if (!error) return '(no error message)';
18
+ for (const raw of String(error).split(/\r\n|\r|\n/)) {
19
+ const line = raw.trim();
20
+ if (line) return line.slice(0, limit);
21
+ }
22
+ return '(no error message)';
23
+ }
24
+
25
+ const NEXT_STEP = {
26
+ revert: 'Revert the culprit commit. One area, one attributable commit.',
27
+ quarantine:
28
+ 'Quarantine the failing tests and open a tracking issue. The break is not cleanly attributable to one commit.',
29
+ investigate:
30
+ 'Investigate before acting. The failing tests carry no owning area, so neither a revert nor a quarantine can be aimed.',
31
+ };
32
+
33
+ /**
34
+ * Render the blast.
35
+ *
36
+ * @param {{branch: string, assessment: object, cluster: object|null}} input
37
+ * @returns {{markdown: string, annotations: string[], chatBlock: string}}
38
+ */
39
+ export function renderBlast({ branch, assessment, cluster }) {
40
+ if (assessment.state === 'abstained') {
41
+ const text =
42
+ `${CROSS} canary-screech ABSTAINED ${DASH} the run-history store holds no run for \`${branch}\`.\n\n` +
43
+ 'Zero observations is not a healthy branch. Point `--history` at a store ' +
44
+ 'that this branch actually writes to, or wire the suite to record runs.';
45
+ return {
46
+ markdown: `# canary-screech ${DASH} ${branch}\n\n${text}\n`,
47
+ annotations: [],
48
+ chatBlock: text,
49
+ };
50
+ }
51
+
52
+ if (assessment.state === 'green') {
53
+ const text = `${CHECK} \`${branch}\` is green as of ${assessment.latest.commit_sha ?? '(unknown commit)'}.`;
54
+ return {
55
+ markdown: `# canary-screech ${DASH} ${branch}\n\n${text}\n`,
56
+ annotations: [],
57
+ chatBlock: text,
58
+ };
59
+ }
60
+
61
+ const { culpritRange, firstRed } = assessment;
62
+ const from =
63
+ culpritRange.from ?? '(unknown - no green run recorded before the break)';
64
+ const to = culpritRange.to ?? '(unknown)';
65
+ const failureCount = cluster.clusters.reduce((n, c) => n + c.tests.length, 0);
66
+
67
+ const chatLines = [
68
+ `${CROSS} ${branch} is RED ${DASH} ${failureCount} failing test${failureCount === 1 ? '' : 's'}`,
69
+ `culprit range: ${from}..${to}`,
70
+ `owning area: ${cluster.owningArea ?? '(none recorded)'}`,
71
+ `recommendation: ${cluster.recommendation.toUpperCase()} ${DASH} ${NEXT_STEP[cluster.recommendation]}`,
72
+ `first red run: ${firstRed.run_id} (${firstRed.suite}) at ${firstRed.timestamp}`,
73
+ ];
74
+ const chatBlock = chatLines.join('\n');
75
+
76
+ const lines = [
77
+ `# ${CROSS} canary-screech ${DASH} \`${branch}\` is red`,
78
+ '',
79
+ `**First red run:** \`${firstRed.run_id}\` (${firstRed.suite}) at ${firstRed.timestamp}`,
80
+ '',
81
+ '## Culprit commit range',
82
+ '',
83
+ `\`${from}\` .. \`${to}\``,
84
+ '',
85
+ ];
86
+ if (!culpritRange.bounded) {
87
+ lines.push(
88
+ '> The lower bound is unknown: the store holds no green run for this ' +
89
+ 'branch before the break, so the range below is open-ended.',
90
+ '',
91
+ );
92
+ }
93
+ lines.push('## Failure cluster', '');
94
+ for (const c of cluster.clusters) {
95
+ lines.push(`### ${c.category} (${c.tests.length})`, '');
96
+ for (const t of c.tests) {
97
+ lines.push(`- \`${t.test_name}\` ${DASH} ${firstLine(t.error_text)}`);
98
+ }
99
+ lines.push('');
100
+ }
101
+ lines.push(
102
+ '## Owning area',
103
+ '',
104
+ cluster.areas.length
105
+ ? cluster.areas.map((a) => `- \`${a.area}\` (${a.count})`).join('\n')
106
+ : '_No failing test carries an `area`, so ownership could not be derived._',
107
+ '',
108
+ '## Recommendation',
109
+ '',
110
+ `**${cluster.recommendation.toUpperCase()}** ${DASH} ${NEXT_STEP[cluster.recommendation]}`,
111
+ '',
112
+ '## Chat-ready block',
113
+ '',
114
+ '```text',
115
+ chatBlock,
116
+ '```',
117
+ '',
118
+ );
119
+
120
+ const annotations = [
121
+ `::error title=Broken branch::${branch} is red ${DASH} ${failureCount} failing test${failureCount === 1 ? '' : 's'} in ${cluster.owningArea ?? 'an unrecorded area'}; culprit ${from}..${to}; recommendation: ${cluster.recommendation}`,
122
+ ];
123
+
124
+ return { markdown: lines.join('\n'), annotations, chatBlock };
125
+ }