canary-test-cli 7.1.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/agents/skills/README.md +327 -0
  2. package/agents/skills/canary:generate.md +49 -0
  3. package/agents/skills/canary:init.md +37 -0
  4. package/agents/skills/canary:migrate.md +66 -0
  5. package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
  6. package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
  7. package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
  8. package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
  9. package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
  10. package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
  11. package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
  12. package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
  13. package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
  14. package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
  15. package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
  16. package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
  17. package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
  18. package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
  19. package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
  20. package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
  21. package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
  22. package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
  23. package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
  24. package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
  25. package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
  26. package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
  27. package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
  28. package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
  29. package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
  30. package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
  31. package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
  32. package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
  33. package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
  34. package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
  35. package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
  36. package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
  37. package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
  38. package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
  39. package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
  40. package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
  41. package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
  42. package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
  43. package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
  44. package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
  45. package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
  46. package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
  47. package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
  48. package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
  49. package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
  50. package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
  51. package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
  52. package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
  53. package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
  54. package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
  55. package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
  56. package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
  57. package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
  58. package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
  59. package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
  60. package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
  61. package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
  62. package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
  63. package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
  64. package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
  65. package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
  66. package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
  67. package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
  68. package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
  69. package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
  70. package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
  71. package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
  72. package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
  73. package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
  74. package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
  75. package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
  76. package/agents/skills/lib/parse-args.mjs +275 -0
  77. package/dist/engine/analysis/batwoman/audit.js +39 -0
  78. package/dist/engine/analysis/batwoman/closure.js +159 -0
  79. package/dist/engine/analysis/batwoman/gh-history.js +119 -0
  80. package/dist/engine/analysis/batwoman/probes.js +195 -0
  81. package/dist/engine/analysis/batwoman/registry.js +142 -0
  82. package/dist/engine/analysis/batwoman/render.js +194 -0
  83. package/dist/engine/analysis/batwoman/run-window.js +122 -0
  84. package/dist/engine/analysis/batwoman/text.js +84 -0
  85. package/dist/engine/analysis/batwoman/triggers.js +122 -0
  86. package/dist/engine/analysis/batwoman/verdict.js +64 -0
  87. package/dist/engine/analysis/cli.js +47 -14
  88. package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
  89. package/dist/engine/batwoman-cli.js +119 -0
  90. package/dist/engine/ci-ready-cli.js +71 -0
  91. package/dist/engine/cli-commands.js +49 -72
  92. package/dist/engine/cli.core.js +16 -0
  93. package/dist/engine/company-knowledge-cli.js +10 -2
  94. package/dist/engine/core/ci-ready.js +112 -0
  95. package/dist/engine/core/company-knowledge.js +8 -0
  96. package/dist/engine/core/migrator.js +147 -20
  97. package/dist/engine/core/permission-matrix.js +219 -0
  98. package/dist/engine/core/quality-scorer.js +27 -19
  99. package/dist/engine/core/scaling-curve.js +143 -0
  100. package/dist/engine/core/skill-dispatch.js +115 -0
  101. package/dist/engine/core/skill-examples.js +103 -3
  102. package/dist/engine/core/skill-registry.js +59 -4
  103. package/dist/engine/core/string-literals.js +3 -1
  104. package/dist/engine/core/test-files.js +77 -0
  105. package/dist/engine/core/vacuity-scanner.js +330 -15
  106. package/dist/engine/core/workflow-discovery.js +41 -23
  107. package/dist/engine/guardian/adjudication-github.js +136 -0
  108. package/dist/engine/guardian/adjudication.js +119 -340
  109. package/dist/engine/guardian/analysis-emit.js +7 -2
  110. package/dist/engine/guardian/cli.js +277 -249
  111. package/dist/engine/guardian/coverage.js +2 -1
  112. package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
  113. package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
  114. package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
  115. package/dist/engine/guardian/diff-coverage/paths.js +5 -9
  116. package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
  117. package/dist/engine/guardian/diff-extractor.js +31 -32
  118. package/dist/engine/guardian/pr-check.js +354 -223
  119. package/dist/engine/guardian/pr-comment.js +35 -58
  120. package/dist/engine/guardian/weak-test.js +236 -0
  121. package/dist/engine/mcp-server.js +67 -4
  122. package/dist/engine/permission-matrix-cli.js +51 -0
  123. package/dist/engine/scaling-curve-cli.js +147 -0
  124. package/dist/engine/skills-cli.js +171 -51
  125. package/dist/engine/workflow-cli.js +85 -65
  126. package/dist/reporters/testtracker.d.ts +1 -1
  127. package/dist/reporters/testtracker.js +1 -1
  128. package/package.json +3 -2
@@ -0,0 +1,248 @@
1
+ ---
2
+ name: canary-add-framework
3
+ description: >
4
+ Add a new testing framework to Canary's registry end-to-end — aligns the
5
+ classifier↔registry contract, authors the
6
+ `ts/src/data/frameworks/registry.json` entry, validates the execution command,
7
+ and ensures every classifier `test_type` still resolves to a non-null
8
+ framework. Use for "add support for a new framework", "add k6", "support
9
+ cypress", "we need Locust", or when a classifier `test_type` exists with no
10
+ framework backing it (registry gap). Not for choosing between existing
11
+ frameworks at runtime (that's the recommender's job) or adding a one-off
12
+ framework name with no real CLI.
13
+ ---
14
+
15
+ # Canary: Add Framework
16
+
17
+ > Add a new testing framework to Canary's registry end-to-end. Confirms the
18
+ > classifier↔registry contract, authors the registry entry, validates the
19
+ > execution command, and ensures every classifier `test_type` still resolves to
20
+ > a non-null framework.
21
+
22
+ ## When to Use
23
+
24
+ - When the user asks to add support for a new framework ("add k6", "support
25
+ cypress", "we need Locust")
26
+ - When a classifier `test_type` exists but has no framework backing it (registry
27
+ gap)
28
+ - When migrating from a deprecated framework — adding the new entry before
29
+ removing the old one
30
+ - NOT for choosing between existing frameworks at runtime — that's the
31
+ recommender's job
32
+ - NOT for adding a one-off LLM-generated framework name that doesn't exist as a
33
+ real tool — frameworks must be real, runnable CLIs
34
+
35
+ ## Process
36
+
37
+ ### Phase 1: SCOPE — Confirm the Framework Belongs in the Registry
38
+
39
+ 1. **Verify the framework is real and installable.** It needs a runnable CLI
40
+ invocation (the `execution_command` template). If you can't write a one-line
41
+ shell command that runs a single test file, it doesn't fit Canary's model.
42
+ 2. **Identify the category.** Match to an existing `category` value (`e2e_ui`,
43
+ `python_unit`, `api`, `performance`, `frontend_unit`, etc.). New categories
44
+ require a coordinated classifier update — see Phase 2.
45
+ 3. **Check for overlap.** Is there already a `preferred` framework in this
46
+ category? If yes, decide intentionally: is the new framework a replacement
47
+ (demote the old to `legacy`), a peer (add as `supported`), or the new
48
+ preference (demote the old, mark new as `preferred`)?
49
+ 4. **Confirm the user wants this maintained.** Each registry entry is a
50
+ maintenance commitment. A framework added casually but never validated rots
51
+ into a routing trap.
52
+
53
+ ### Phase 2: ALIGN — Classifier ↔ Registry Contract
54
+
55
+ 1. **Find the classifier rule that emits this `test_type`.** Open
56
+ `ts/src/core/classifier.ts` and locate the heuristic block that produces the
57
+ target category.
58
+ 2. **If no classifier rule exists for the category:** Add one _before_ the
59
+ registry entry. A framework with no routing path is a dead entry. Add a
60
+ heuristic that maps natural-language signals (keywords, phrases) to the new
61
+ `test_type`.
62
+ 3. **If the category exists but the classifier never emits it confidently:**
63
+ Strengthen the rule's heuristics. Confidence below 0.7 should trigger a
64
+ clarification, not a generation.
65
+ 4. **Run the contract tests:**
66
+
67
+ ```bash
68
+ pytest tests/unit/test_orchestrator.py
69
+ ```
70
+
71
+ Every `test_type` the classifier can emit must resolve to ≥1 framework via
72
+ `get_by_category` — the orchestrator tests assert non-null framework
73
+ resolution per category. This is the gate.
74
+
75
+ ### Phase 3: AUTHOR — Write the Registry Entry
76
+
77
+ 1. **Open `ts/src/data/frameworks/registry.json`.** Add a new object to
78
+ `frameworks[]`. Required fields:
79
+ - `name` (unique slug)
80
+ - `display_name`
81
+ - `category` (must match the classifier emission)
82
+ - `languages`
83
+ - `file_extensions`
84
+ - `execution_command` (with `{file}` placeholder)
85
+ - `status` (`preferred` / `supported` / `legacy`)
86
+ 2. **Add recommended metadata.** `maturity`, `community_size`,
87
+ `recommended_for`, `strengths`, `avoid_when`, `ecosystems`. These show up in
88
+ the recommender's reasoning string — sparse entries produce sparse
89
+ recommendations.
90
+ 3. **Validate the `execution_command`.** It must run a single test file when
91
+ `{file}` is substituted. Test manually:
92
+
93
+ ```bash
94
+ echo '<minimal test>' > /tmp/sample.<ext>
95
+ <execution_command with /tmp/sample.<ext> substituted>
96
+ ```
97
+
98
+ Confirm exit code 0 (or expected non-zero) and no missing-config errors.
99
+
100
+ 4. **Confirm the `{file}` placeholder is the only substitution.** The executor
101
+ tokenizes the template with `shlex.split` before substituting `{file}`, so
102
+ the file path stays a single argv element; paths with spaces must remain
103
+ intact.
104
+
105
+ ### Phase 4: VERIFY — Generate, Execute, Promote-Dry
106
+
107
+ 1. **Generate a test that should route to the new framework.** Run the
108
+ `/canary-write-test` slash command in Claude Code with a prompt that should
109
+ hit the new `test_type`.
110
+
111
+ 2. **Check the trace.** Classification matches the expected category,
112
+ recommendation picks the new framework (or the existing preferred — confirm
113
+ this matches your intent), output file has the right extension.
114
+ 3. **Execute with `--execute`.** Confirm the framework's CLI actually runs the
115
+ generated file. A passing execution validates the `execution_command`
116
+ template under the real Canary invocation path.
117
+ 4. **Don't promote.** This is a registry-validation run, not a real test
118
+ promotion. Leave the artifact in `tests/generated/`.
119
+
120
+ ### Phase 5: DOCUMENT — Update Guides and State
121
+
122
+ 1. **Update `docs/guides/framework-registry.md`** if you added a new category,
123
+ changed the contract surface, or introduced a new optional field schema.
124
+ 2. **Update the LLM-providers guide** if the new framework only works with
125
+ specific provider output styles (rare; flag this as a smell if it's true).
126
+ 3. **Log to `docs/CANARY_STATE.md`.** One line: framework name, category,
127
+ status, date added. This is the project ledger; future maintainers grep here
128
+ first.
129
+ 4. **Open a PR.** Title: `feat(registry): add <framework-name> support`. Body
130
+ should include the validation steps you ran and the generated/executed
131
+ sample.
132
+
133
+ ## Canary Integration
134
+
135
+ - **`ts/src/data/frameworks/registry.json`** — Source of truth for entries.
136
+ - **`ts/src/core/framework-registry.ts`** — Loader and lookup methods. Don't add
137
+ lookup helpers here unless the existing five (`get_all_frameworks`,
138
+ `get_by_category`, `get_preferred_by_category`, `find_by_name`,
139
+ `match_by_language`) genuinely don't fit.
140
+ - **`ts/src/core/classifier.ts`** — Routing rules. Coordinate changes with
141
+ registry entries.
142
+ - **`tests/unit/test_orchestrator.py`** — Contract enforcement. Non-null
143
+ resolution per `test_type` (api → pytest, e2e_ui → playwright, performance →
144
+ k6, etc.) is asserted here. `tests/unit/test_factory.py` separately enforces
145
+ the LLM provider matrix.
146
+
147
+ ## Success Criteria
148
+
149
+ - New entry validates against the contract test (no orphaned `test_type` in
150
+ classifier, no null framework resolution in registry)
151
+ - The `/canary-write-test` slash command with a routing prompt resolves to the
152
+ new framework
153
+ - `--execute` runs the generated file successfully via the new
154
+ `execution_command`
155
+ - `framework-registry.md` documents any new category or schema field
156
+ - `CANARY_STATE.md` ledger entry is present
157
+ - PR includes the validation transcript
158
+
159
+ ## Rationalizations to Reject
160
+
161
+ | Rationalization | Why It Is Wrong |
162
+ | ------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
163
+ | "I'll add the registry entry now and update the classifier later" | Orphaned registry entries are routing traps. Either both land together or neither does. |
164
+ | "Marking it `preferred` is fine, no need to demote the existing entry" | Multiple `preferred` entries in one category means `get_preferred_by_category` picks by registry order, not by merit. Demote the loser explicitly. |
165
+ | "I tested the command with my path — Canary's substitution will work too" | Test the substituted command with a path that contains spaces. The executor's `shlex.split`-then-substitute order exists specifically to handle this case; a broken template will silently corrupt argv. |
166
+ | "It only needs to work for the happy-path prompt I tried" | If the classifier emits this `test_type` with confidence ≥0.7 for _any_ phrasing, the framework must handle the resulting generated file. Test at least two distinct prompts that route to the new entry. |
167
+ | "Sparse metadata is fine, we can fill it in later" | The recommender's reasoning string is read by humans and agents. Sparse metadata produces unhelpful recommendations — fill it in now while you have the context. |
168
+
169
+ ## Examples
170
+
171
+ ### Example: Adding Cypress as a peer to Playwright
172
+
173
+ **Scope:** Cypress already has community demand and a CI image. Category
174
+ `e2e_ui` already has Playwright as `preferred`. Decision: add Cypress as
175
+ `supported`, keep Playwright as `preferred`.
176
+
177
+ **Classifier:** Already emits `test_type=e2e_ui` on UI-test prompts. No
178
+ classifier change needed.
179
+
180
+ **Entry (abridged):**
181
+
182
+ ```json
183
+ {
184
+ "name": "cypress",
185
+ "display_name": "Cypress",
186
+ "category": "e2e_ui",
187
+ "languages": ["javascript", "typescript"],
188
+ "file_extensions": ["cy.ts", "cy.js"],
189
+ "execution_command": "npx --yes cypress run --spec {file}",
190
+ "status": "supported",
191
+ "strengths": ["Time-travel debugging", "Strong DX for UI tests"],
192
+ "avoid_when": ["Cross-browser parity required (Playwright is stronger)"]
193
+ }
194
+ ```
195
+
196
+ **Validation:** Generated a UI test prompt, confirmed recommender still picked
197
+ Playwright (correct — `preferred` wins). Verified Cypress is accessible by
198
+ routing a prompt with explicit Cypress mention through a future
199
+ explicit-framework override.
200
+
201
+ ### Example: Adding k6 to a new performance category
202
+
203
+ **Scope:** Performance category doesn't exist in the registry yet. Classifier
204
+ emits `test_type=performance` on load/stress prompts, but the registry has no
205
+ matching entry — this is the orphan case.
206
+
207
+ **Phase 2 first:** Confirmed `ts/src/core/classifier.ts` already has the
208
+ performance rule (`if "performance" in p or "load test" in p ...`). No
209
+ classifier change needed.
210
+
211
+ **Phase 3:**
212
+
213
+ ```json
214
+ {
215
+ "name": "k6",
216
+ "display_name": "k6",
217
+ "category": "performance",
218
+ "languages": ["javascript"],
219
+ "file_extensions": ["js"],
220
+ "execution_command": "k6 run {file}",
221
+ "status": "preferred",
222
+ "recommended_for": ["HTTP load testing", "Stress and spike tests"],
223
+ "avoid_when": ["Browser-level performance (use Playwright tracing)"]
224
+ }
225
+ ```
226
+
227
+ **Validation:** Ran the `/canary-write-test` slash command with "Load test
228
+ /v1/search at 200 RPS for 5 minutes". Confirmed classifier emits
229
+ `test_type=performance` at 0.95 confidence, recommender picks k6, k6 CLI runs
230
+ the generated file without error.
231
+
232
+ ## Escalation
233
+
234
+ - **When the framework requires a non-trivial config file (e.g.,
235
+ `playwright.config.ts`):** Document the setup in `avoid_when` or in
236
+ `recommended_for`. If Canary is expected to generate the config too, that's a
237
+ scaffolder change, not just a registry change.
238
+ - **When two frameworks legitimately should both be `preferred` for different
239
+ sub-cases:** Split the category. A single category should have a single
240
+ `preferred`. If the split isn't clean, surface this to the user as a design
241
+ decision rather than fudging the registry.
242
+ - **When the classifier doesn't reliably emit the target `test_type`:** Don't
243
+ add the registry entry yet. Strengthen the classifier first — an unreachable
244
+ entry is dead weight.
245
+ - **When the framework's CLI doesn't accept a single-file argument:** Canary's
246
+ model is one-test-per-file generation. Frameworks that only run by directory
247
+ or by tag don't fit cleanly; raise this with the user before forcing a
248
+ workaround.
@@ -0,0 +1,119 @@
1
+ ---
2
+ name: canary-batwoman
3
+ description:
4
+ Closure auditing — reports whether the files a closed issue's fix changed have
5
+ actually EXECUTED since that fix merged. GitHub closes an issue on a keyword
6
+ match in a PR body, which checks neither that the fix works nor that it ever
7
+ ran; batwoman answers only the second question, per changed file, and names
8
+ the files it could not answer for. Use when the user asks "did that fix
9
+ actually run", "is this issue really done", "audit a closed issue", or after a
10
+ batch of merges. Advisory and read-only — it never asserts correctness, never
11
+ fails a job, and never reopens an issue. NOT a test runner, NOT a coverage
12
+ tool (a dormant workflow has whatever coverage it always had), and NOT a
13
+ correctness check.
14
+ cli: canary batwoman
15
+ requires: [node>=20, gh]
16
+ ---
17
+
18
+ # Canary Batwoman
19
+
20
+ GitHub closes an issue when a merged PR body matches `Closes #N`. That is a
21
+ **string match with no denominator**: nothing checks that the fix works, and
22
+ nothing checks that the fix ever ran. The issue moves to `CLOSED` on the
23
+ strength of a keyword.
24
+
25
+ Batwoman audits the claim. For each file the closing PR changed, it reports
26
+ whether that artifact has executed since the merge — and, just as importantly,
27
+ which files it could not decide about.
28
+
29
+ ## The founding case
30
+
31
+ canary#749 fixed a false-green in `.github/workflows/refresh-arch-baseline.yml`.
32
+ It merged as `1e0c05b` and auto-closed on the keyword. Verified afterwards:
33
+
34
+ - 11 script unit tests passed
35
+ - 97 static workflow assertions passed
36
+ - the workflow itself **had not run since twelve days before the fix merged** —
37
+ it is label-triggered, so it stayed dormant until someone applied the label
38
+
39
+ Every gate the repo owns read that as done. The fix was real and well-tested;
40
+ its end-to-end path was unproven. **Verified and exercised are different
41
+ properties**, and that gap is what batwoman makes visible.
42
+
43
+ ## Usage
44
+
45
+ ```bash
46
+ canary batwoman --help
47
+ ```
48
+
49
+ That one needs no credentials, no network and no fixtures. A real audit needs
50
+ the repository named and `gh` authenticated:
51
+
52
+ ```bash
53
+ export GITHUB_REPOSITORY=owner/name # never inferred from a git remote
54
+ canary batwoman --issue 749 # the report
55
+ canary batwoman --issue 749 --json # the machine shape, for CI
56
+ ```
57
+
58
+ `GITHUB_REPOSITORY` is required rather than guessed. Auditing the wrong
59
+ repository's run history would produce a confident answer about the wrong thing,
60
+ which is worse than refusing.
61
+
62
+ ## What the five statuses mean
63
+
64
+ | status | meaning |
65
+ | ---------------- | -------------------------------------------------------------------------- |
66
+ | `exercised` | a run started after the merge, and it names which |
67
+ | `not-exercised` | the artifact has not run since — with the `on:` trigger explaining why |
68
+ | `abstain` | a probe looked and could not tell (unreadable history, untraceable script) |
69
+ | `no-probe` | nothing looked: a registry gap, fixable by adding a probe |
70
+ | `not-applicable` | prose, config, or a file this change deleted |
71
+
72
+ `abstain` and `no-probe` are deliberately separate, and neither is folded into
73
+ anything resembling a pass. **The summary always prints all five counts**, and
74
+ they always sum to the changed-file total — a report over a subset presented as
75
+ a report over the whole is the exact defect batwoman exists to detect.
76
+
77
+ There is no `assessed` field, no score, and no success token anywhere in the
78
+ output. A convenient `passed: true` in the JSON is the field a CI wrapper would
79
+ grow later; it does not exist, and a test asserts it stays that way.
80
+
81
+ ## What it will not tell you
82
+
83
+ - **Whether the fix is correct.** Batwoman proves execution, never behaviour. A
84
+ workflow that ran and did the wrong thing reports `exercised`.
85
+ - **Whether coverage is adequate.** A dormant workflow has whatever coverage it
86
+ always had; that is the point.
87
+ - **Anything about a file it has no probe for.** `ts/src/**` is an honest
88
+ `no-probe` row in v1, countable rather than silent.
89
+
90
+ ## Probes shipped
91
+
92
+ | probe | matches | decides by |
93
+ | ----------------- | ------------------------- | ------------------------------------------------------------------------ |
94
+ | `workflow` | `.github/workflows/*.yml` | a run created after the merge; reads `on:` to explain a dormant one |
95
+ | `workflow-script` | `scripts/*.mjs` | the workflows that name it, inheriting `exercised` if **any** caller ran |
96
+ | `no-execution` | `*.md`, config manifests | returns `not-applicable`, reading nothing |
97
+
98
+ A script nothing references **abstains** rather than reporting as never-run: "no
99
+ workflow calls this" is a statement about the repo, not about the script, which
100
+ may still be run by a hand or a hook.
101
+
102
+ ## Requires the network, and says so
103
+
104
+ Batwoman shells out to `gh` for run history and for the closing PR. That is a
105
+ property, not a tier — unlike the deterministic offline detectors alongside it
106
+ (`canary-savant`, `canary-blackhawk`, `canary-cassandra`, `canary-katana`),
107
+ which run anywhere node does.
108
+
109
+ An unauthenticated or missing `gh` does not degrade to a clean report. It fails
110
+ loudly, and every affected file becomes an `abstain` naming the failure. Per
111
+ this repo's standing rule, **cannot-verify is a finding, not a skip**.
112
+
113
+ ## In CI
114
+
115
+ `.github/workflows/batwoman.yml` runs it on every push to `main`, deriving the
116
+ issue from the merge commit's closing keyword — the very keyword match batwoman
117
+ questions, which is the right input precisely because it is what GitHub acted
118
+ on. Advisory: every step is `continue-on-error`, and a merge that closes nothing
119
+ is reported and skipped rather than passing silently over an empty set.
@@ -0,0 +1,170 @@
1
+ ---
2
+ name: canary-blackhawk
3
+ description:
4
+ Temporal-dependency linter for test files — statically flags tests that lean
5
+ on wall-clock time, a real delay, or the local timezone, the ones that pass
6
+ all day and fail at midnight, across a DST boundary, or on Feb 29. Suppresses
7
+ itself when the file already installs a frozen clock (fake timers, freezegun,
8
+ time_machine, MockDate). Self-contained, deterministic, advisory by default.
9
+ cli: scripts/cli.mjs
10
+ requires: [node>=20]
11
+ ---
12
+
13
+ # Canary Blackhawk
14
+
15
+ A test that reads the wall clock is a test with a scheduled outage. It passes
16
+ every run you watch, then fails at 23:59:59, on the Sunday the clocks move, or
17
+ on Feb 29. Blackhawk finds those lines before the calendar does.
18
+
19
+ Tier-0 deterministic analysis: no LLM, no network, no secrets, no dependency on
20
+ any other skill.
21
+
22
+ ## Rules
23
+
24
+ | Rule | Severity | Fires on |
25
+ | ------------------------------ | -------- | --------------------------------------------------------------------------------------------------------------------------------------------- |
26
+ | `BH001-wall-clock` | high | `Date.now()`, bare `new Date()`, `moment()`, `datetime.now()` / `.today()` / `.utcnow()`, `date.today()`, `time.time()`, `pd.Timestamp.now()` |
27
+ | `BH002-real-delay` | medium | `time.sleep(n)` and `setTimeout(fn, n)` with a literal `n > 0` |
28
+ | `BH003-local-timezone` | medium | `toLocaleString` / `toLocaleDateString` / `toLocaleTimeString`, `strftime('…%Z…')` or `%z` |
29
+ | `BH004-naive-datetime-compare` | low | a comparison against `datetime(2024, …)` or `strptime(…)` with no `tzinfo` / `timezone.utc` / `ZoneInfo` / `pytz` on the line |
30
+
31
+ A pinned constructor is never flagged: `new Date('2024-01-01T00:00:00Z')`,
32
+ `moment('2024-01-01')`, and `datetime(2024, 1, 1)` on its own are all fine.
33
+
34
+ ## Framework-conditioned suppression (the important part)
35
+
36
+ The single biggest failure mode of a temporal linter is firing on tests that
37
+ **already handle time correctly**. So when a file installs a frozen clock, every
38
+ clock-dependent rule (`BH001`, `BH002`, `BH004`) goes quiet for that file.
39
+ Markers:
40
+
41
+ `vi.useFakeTimers` · `vi.setSystemTime` · `jest.useFakeTimers` ·
42
+ `jest.setSystemTime` · `sinon.useFakeTimers` · `MockDate` · `freeze_time` /
43
+ `freezegun` · `time_machine`
44
+
45
+ Two deliberate choices inside that:
46
+
47
+ - **Suppression is file-wide, not block-scoped.** `vi.useFakeTimers()` inside a
48
+ `beforeEach` governs tests declared above it, and answering "is the clock
49
+ frozen _here_" accurately needs a real parser. Blackhawk errs toward silence.
50
+ - **`BH003` is never suppressed.** Freezing the clock pins _when_ a test runs,
51
+ never _where_. A frozen clock does not stop `toLocaleString()` from returning
52
+ a different string on a developer laptop than on a UTC runner.
53
+
54
+ ## Inline suppression (per-line escape hatch)
55
+
56
+ File-wide frozen-clock suppression is too coarse when one intentional real wait
57
+ sits among otherwise-deterministic tests. An inline pragma silences a single
58
+ finding:
59
+
60
+ ```js
61
+ // blackhawk-ignore BH002 -- real MongoDB claim race; deterministic wait impossible
62
+ await new Promise((r) => setTimeout(r, 25));
63
+ ```
64
+
65
+ - **Reason required** (the `-- reason` tail) — keeps suppressions honest and
66
+ greppable, like `eslint-disable-next-line` / `# noqa`.
67
+ - **Rule-scoped** (`BH002`, or the full `BH002-real-delay`) so it never
68
+ blanket-silences the line; a `BH001` on the same line still fires.
69
+ - **Placement:** the pragma may trail the offending line or sit on the line
70
+ directly above it. Comma-separate ids to suppress several (`BH002,BH004`).
71
+ - **Counted separately:** suppressed findings are reported as `N suppressed`,
72
+ out of the actionable total — so a genuinely clean suite can read zero
73
+ findings while the known-OK waits stay visible.
74
+
75
+ ## Fidelity limits (regex/AST-lite, on purpose)
76
+
77
+ Blackhawk is a line scanner with no TypeScript parser dependency, so it ships
78
+ anywhere `node` does. The cost, stated plainly:
79
+
80
+ - **Line-scoped.** A call split across lines (`setTimeout(\n fn,\n 500\n)`) is
81
+ missed, as is a `setTimeout` whose callback body contains a comma.
82
+ - **Comment-blind, one level deep.** Lines starting with `#`, `//`, `*`, `/*`,
83
+ or a docstring quote are skipped; a multi-line block comment whose inner lines
84
+ do not start with `*` is still scanned.
85
+ - **String-aware, one line deep.** A match that _starts_ inside a string literal
86
+ on its own line is rejected as fixture data (`pyFile('time.sleep(1)')` does
87
+ not fire), while an anchor whose pattern merely reaches into quotes
88
+ (`strftime('..%Z')`) still does; template `${...}` interpolation is code. A
89
+ string spanning lines is only seen on its opening line, so a fixture blob's
90
+ continuation lines are still scanned like code.
91
+ - **No type awareness.** `.toLocaleString()` on a `Number` reads the same as on
92
+ a `Date`.
93
+ - **Suppression is a substring match.** A mention of `freezegun` in a comment
94
+ silences the file. That direction is intentional: a missed finding costs less
95
+ than a false one.
96
+
97
+ ## Which files get scanned
98
+
99
+ A directory walk only visits **test** files — `*.test.*`, `*.spec.*`,
100
+ `test_*.py`, `*_test.py`, or any supported source under `tests/`, `test/`,
101
+ `__tests__/`, `e2e/`, `spec/`. A file named explicitly on the command line is
102
+ always scanned, test-looking or not. Supported suffixes: `.py`, `.js`, `.jsx`,
103
+ `.ts`, `.tsx`, `.mjs`, `.cjs`.
104
+
105
+ ## Invocation
106
+
107
+ ```bash
108
+ # Scan the repo's test files (advisory — always exits 0):
109
+ canary skills run canary-blackhawk
110
+
111
+ # Scan a specific suite:
112
+ canary skills run canary-blackhawk -- tests/e2e
113
+
114
+ # Machine-readable findings:
115
+ canary skills run canary-blackhawk -- tests --json
116
+
117
+ # Fail the step on any finding:
118
+ canary skills run canary-blackhawk -- tests --strict
119
+
120
+ # Usage, options, and the full rule list (exits 0):
121
+ canary skills run canary-blackhawk -- --help
122
+ ```
123
+
124
+ An unknown flag is rejected with `unrecognized arguments: <flag>` and exit 2;
125
+ use `--` to end option parsing when a path itself starts with a dash.
126
+
127
+ `--json` shape:
128
+
129
+ ```json
130
+ {
131
+ "schema_version": 1,
132
+ "findings": [
133
+ {
134
+ "file": "tests/clock.spec.ts",
135
+ "line": 12,
136
+ "rule_id": "BH001-wall-clock",
137
+ "severity": "high",
138
+ "snippet": "const t = Date.now();",
139
+ "why": "reads the wall clock, so the assertion depends on when the suite runs..."
140
+ }
141
+ ],
142
+ "summary": {
143
+ "files_scanned": 8,
144
+ "findings": 1,
145
+ "by_severity": { "high": 1 },
146
+ "suppressed": 0
147
+ }
148
+ }
149
+ ```
150
+
151
+ ## CI wiring (GitHub Actions)
152
+
153
+ Advisory first, then promote to blocking once the backlog is drained and the
154
+ signal is trusted — the same path every canary gate takes.
155
+
156
+ ```yaml
157
+ - name: Temporal-dependency lint (advisory)
158
+ run: canary skills run canary-blackhawk -- tests
159
+ # Once clean, add --strict to make new offenders fail the PR:
160
+ # run: canary skills run canary-blackhawk -- tests --strict
161
+ ```
162
+
163
+ ## Fixing what it finds
164
+
165
+ | Finding | Fix |
166
+ | ------- | ------------------------------------------------------------------------------------------------------------------------------------------------ |
167
+ | `BH001` | Freeze the clock (`vi.useFakeTimers()` + `vi.setSystemTime(...)`, `@freeze_time("2024-01-01")`) or inject a clock the test controls. |
168
+ | `BH002` | Advance a fake timer (`vi.advanceTimersByTime(500)`) or await the real condition instead of a duration. |
169
+ | `BH003` | Assert on a UTC representation (`toISOString()`, `strftime('%Y-%m-%dT%H:%M:%SZ')` on a UTC datetime), or pin the locale and timezone explicitly. |
170
+ | `BH004` | Attach a timezone: `datetime(2024, 1, 1, tzinfo=timezone.utc)`. |