canary-test-cli 7.1.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +327 -0
- package/agents/skills/canary:generate.md +49 -0
- package/agents/skills/canary:init.md +37 -0
- package/agents/skills/canary:migrate.md +66 -0
- package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
- package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
- package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
- package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
- package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
- package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
- package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
- package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
- package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
- package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
- package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
- package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
- package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
- package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
- package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
- package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
- package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
- package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
- package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
- package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
- package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
- package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
- package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
- package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
- package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
- package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
- package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
- package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
- package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
- package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
- package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
- package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
- package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
- package/agents/skills/lib/parse-args.mjs +275 -0
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +49 -72
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +27 -19
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/skill-dispatch.js +115 -0
- package/dist/engine/core/skill-examples.js +103 -3
- package/dist/engine/core/skill-registry.js +59 -4
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/test-files.js +77 -0
- package/dist/engine/core/vacuity-scanner.js +330 -15
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/analysis-emit.js +7 -2
- package/dist/engine/guardian/cli.js +277 -249
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +354 -223
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +171 -51
- package/dist/engine/workflow-cli.js +85 -65
- package/dist/reporters/testtracker.d.ts +1 -1
- package/dist/reporters/testtracker.js +1 -1
- package/package.json +3 -2
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-edge-case-discovery
|
|
3
|
+
description: >
|
|
4
|
+
Given a feature description, function signature, or existing test suite,
|
|
5
|
+
surfaces edge cases worth testing across six categories. Explanation depth
|
|
6
|
+
scales to user skill level. Optionally focuses on critical areas when
|
|
7
|
+
critical-areas.json is present.
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# Canary: Edge Case Discovery
|
|
11
|
+
|
|
12
|
+
Surfaces the edge cases that tests typically miss: the inputs and conditions
|
|
13
|
+
that work in demos but break in production.
|
|
14
|
+
|
|
15
|
+
## When to Use
|
|
16
|
+
|
|
17
|
+
- After writing happy-path tests: "what else should I test?"
|
|
18
|
+
|
|
19
|
+
- When reviewing a feature for robustness
|
|
20
|
+
|
|
21
|
+
- As Phase 2 of `/canary-test-pipeline`
|
|
22
|
+
|
|
23
|
+
- When asked "what edge cases should I cover?"
|
|
24
|
+
|
|
25
|
+
## Input
|
|
26
|
+
|
|
27
|
+
Provide one of:
|
|
28
|
+
|
|
29
|
+
- A feature description: `"points accrual on tier upgrade"`
|
|
30
|
+
|
|
31
|
+
- A function signature:
|
|
32
|
+
`accruePoints(memberId: string, amount: number): Promise<Result>`
|
|
33
|
+
|
|
34
|
+
- A test file path: `tests/loyalty/points.spec.ts`
|
|
35
|
+
|
|
36
|
+
- Or nothing — Canary will infer from open files and recent context
|
|
37
|
+
|
|
38
|
+
If `.canary/critical-areas.json` is present, focus edge case discovery on the
|
|
39
|
+
highest-risk areas first (rank_score ≥ 0.6).
|
|
40
|
+
|
|
41
|
+
## The Six Categories
|
|
42
|
+
|
|
43
|
+
For each category, generate specific, actionable cases — not generic advice.
|
|
44
|
+
|
|
45
|
+
### 1. Boundary values
|
|
46
|
+
|
|
47
|
+
Zero, negative, max integer, empty string, null, undefined, one-off-by-one. For
|
|
48
|
+
amounts: 0, 1, MAX_SAFE_INTEGER, -1, 0.001 (floating point). For strings: empty
|
|
49
|
+
string, whitespace-only, max-length + 1 character.
|
|
50
|
+
|
|
51
|
+
### 2. Race conditions
|
|
52
|
+
|
|
53
|
+
Concurrent writes to the same resource. Double-submit (user clicks twice). Stale
|
|
54
|
+
reads after an update. Lock contention. Out-of-order async responses.
|
|
55
|
+
|
|
56
|
+
### 3. Locale and timezone
|
|
57
|
+
|
|
58
|
+
DST transition times. Dates at midnight UTC vs local time. Non-ASCII characters
|
|
59
|
+
in names and addresses. RTL text in string fields. Locale-specific number
|
|
60
|
+
formats (1.000,00 vs 1,000.00). Emoji in text fields.
|
|
61
|
+
|
|
62
|
+
### 4. Partial network
|
|
63
|
+
|
|
64
|
+
Request timeout mid-flight. Dropped connection after partial response. Retry
|
|
65
|
+
storms (client retries while server is still processing). Response truncation.
|
|
66
|
+
|
|
67
|
+
### 5. Unexpected input shapes
|
|
68
|
+
|
|
69
|
+
Extra fields the schema doesn't expect. Missing required fields. Wrong types
|
|
70
|
+
(string where number expected). SQL or script injection strings. Deeply nested
|
|
71
|
+
objects. Arrays where scalars expected.
|
|
72
|
+
|
|
73
|
+
### 6. Accessibility
|
|
74
|
+
|
|
75
|
+
Keyboard-only navigation paths. Missing ARIA labels. Focus trap conditions.
|
|
76
|
+
Screen reader text for dynamic content. Colour contrast for status indicators.
|
|
77
|
+
_(Only include if the input is a UI feature or test.)_
|
|
78
|
+
|
|
79
|
+
## Output Depth
|
|
80
|
+
|
|
81
|
+
**Consult the resolved persona; do not infer one here.** The engine owns that
|
|
82
|
+
decision (issue #462) so every skill adapts the same way and a downstream
|
|
83
|
+
overlay has one place to override.
|
|
84
|
+
|
|
85
|
+
Resolve it in this order:
|
|
86
|
+
|
|
87
|
+
1. `--level <id>` if the caller passed one. It is the explicit override and
|
|
88
|
+
always wins.
|
|
89
|
+
2. The `persona` block on a `canary__analyze_file` response. It carries `id`,
|
|
90
|
+
`depth`, `formats`, `reasoning`, plus `source`, `reason`, and `signals` — the
|
|
91
|
+
evidence behind the choice.
|
|
92
|
+
3. Neither available → use `junior`, the engine's fallback. Erring explanatory
|
|
93
|
+
is deliberate: over-explaining is a mild annoyance, under-explaining silently
|
|
94
|
+
fails a manual tester.
|
|
95
|
+
|
|
96
|
+
Render from the persona's `depth`, not from its `id` — an overlay may add
|
|
97
|
+
personas this table never listed:
|
|
98
|
+
|
|
99
|
+
| `depth` | Output style |
|
|
100
|
+
| -------- | ----------------------------------- |
|
|
101
|
+
| `terse` | Bullet list of cases only |
|
|
102
|
+
| `brief` | Cases + one-line _why this matters_ |
|
|
103
|
+
| `guided` | Cases + numbered reproduction steps |
|
|
104
|
+
|
|
105
|
+
When `reasoning` is true, say which persona shaped the output and why —
|
|
106
|
+
`persona.reason` is written to be quoted.
|
|
107
|
+
|
|
108
|
+
Read `source` before trusting the persona, and treat the three values
|
|
109
|
+
differently:
|
|
110
|
+
|
|
111
|
+
- `explicit` — the reader chose this. Do not second-guess it.
|
|
112
|
+
- `detected` — inferred from at least two independent signals about the project.
|
|
113
|
+
Reliable enough to act on, still a guess; mention it is overridable with
|
|
114
|
+
`CANARY_PERSONA` or `--level` if the depth seems wrong for them.
|
|
115
|
+
- `fallback` — the engine **declined** to guess, which is the common case. That
|
|
116
|
+
is not a failure and not a reason to go hunting for signals of your own: it
|
|
117
|
+
means the evidence did not clear the floor, and re-deriving an audience here
|
|
118
|
+
is exactly the hand-rolling this section replaced. Use the persona as given.
|
|
119
|
+
|
|
120
|
+
**Never infer `terse` on your own.** Stripping explanation from a reader who
|
|
121
|
+
turns out to be a manual tester is the expensive direction of this mistake;
|
|
122
|
+
`fallback` is deliberately explanatory for that reason. Note that `manual` is
|
|
123
|
+
not inferable from a file analysis at all — it is reached by an explicit choice
|
|
124
|
+
— so a `fallback` persona may well be a manual tester.
|
|
125
|
+
|
|
126
|
+
The shipped personas are `sdet` (terse), `junior` (brief), and `manual`
|
|
127
|
+
(guided); the definitions live in `ts/src/data/personas/registry.json`.
|
|
128
|
+
|
|
129
|
+
## Output Format (`sdet` example)
|
|
130
|
+
|
|
131
|
+
```text
|
|
132
|
+
Edge cases — points accrual on tier upgrade
|
|
133
|
+
|
|
134
|
+
Boundary values
|
|
135
|
+
· amount = 0
|
|
136
|
+
· amount = MAX_SAFE_INTEGER
|
|
137
|
+
· memberId empty string
|
|
138
|
+
· memberId with special characters
|
|
139
|
+
|
|
140
|
+
Race conditions
|
|
141
|
+
· concurrent accrual calls for the same memberId
|
|
142
|
+
· double-submit within 100ms
|
|
143
|
+
|
|
144
|
+
...
|
|
145
|
+
|
|
146
|
+
Suggested next: /canary-write-test "add edge case tests for accruePoints boundary values"
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Flags
|
|
150
|
+
|
|
151
|
+
- `--level sdet|junior|manual` — output depth. Overrides the resolved persona;
|
|
152
|
+
the default is the persona the engine resolved (see **Output Depth**).
|
|
153
|
+
|
|
154
|
+
## Related skills
|
|
155
|
+
|
|
156
|
+
- `/canary-critical-areas` — produces `critical-areas.json` used for focus
|
|
157
|
+
|
|
158
|
+
- `/canary-write-test` — generates tests for the surfaced cases
|
|
159
|
+
|
|
160
|
+
- `/canary-test-pipeline` — Phase 2
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
name: canary-edge-case-discovery
|
|
2
|
+
version: '1.0.0'
|
|
3
|
+
description:
|
|
4
|
+
Surface edge cases worth testing across six categories from a feature
|
|
5
|
+
description, function signature, or existing suite; explanation depth scales
|
|
6
|
+
to user skill level.
|
|
7
|
+
stability: static
|
|
8
|
+
triggers:
|
|
9
|
+
- manual
|
|
10
|
+
platforms:
|
|
11
|
+
- claude-code
|
|
12
|
+
type: rigid
|
|
13
|
+
tools: []
|
|
14
|
+
tier: 1
|
|
15
|
+
depends_on:
|
|
16
|
+
- canary-ci-ready
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-fail-fast
|
|
3
|
+
description:
|
|
4
|
+
Surface test failures fast and loud — audit a Playwright config for fail-fast
|
|
5
|
+
knobs (maxFailures/forbidOnly/retries) and print a loud, categorized failure
|
|
6
|
+
digest to the CI log + ::error annotations at run end, failing the step so the
|
|
7
|
+
signal can't be missed. Self-contained (bundles its own Playwright JSON parser
|
|
8
|
+
and failure categorizer).
|
|
9
|
+
cli: scripts/cli.mjs
|
|
10
|
+
requires: [node>=20]
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
# Canary Fail-Fast
|
|
14
|
+
|
|
15
|
+
Make test failures **fast** (abort a broken run early) and **loud** (surface in
|
|
16
|
+
the CI log + Checks, not a file). Two halves:
|
|
17
|
+
|
|
18
|
+
1. **Fail-fast config audit** — flags the absence of `maxFailures`,
|
|
19
|
+
`forbidOnly`, and `retries` in a `playwright.config.*`.
|
|
20
|
+
2. **Loud run-end digest** — a terse, categorized failure summary to stdout +
|
|
21
|
+
`::error` annotations, with a non-zero exit.
|
|
22
|
+
|
|
23
|
+
Self-contained: it bundles a minimal Playwright JSON parser and the failure
|
|
24
|
+
categorizer, so it has no dependency on any other skill.
|
|
25
|
+
|
|
26
|
+
## Fail-fast config (paste into `playwright.config.ts`)
|
|
27
|
+
|
|
28
|
+
```ts
|
|
29
|
+
export default defineConfig({
|
|
30
|
+
// Fail fast in CI: abort once enough has clearly broken, never on local runs.
|
|
31
|
+
forbidOnly: !!process.env.CI, // a stray test.only fails the build
|
|
32
|
+
maxFailures: process.env.CI ? 10 : 0, // stop the run after 10 failures in CI
|
|
33
|
+
retries: process.env.CI ? 2 : 0, // absorb flakes in CI; surface them locally
|
|
34
|
+
// ...your existing config
|
|
35
|
+
});
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## Invocation
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
# Loud failure digest from a Playwright JSON run (exits non-zero on failures):
|
|
42
|
+
canary skills run canary-fail-fast -- --results test-results/results.json
|
|
43
|
+
|
|
44
|
+
# Audit the fail-fast config:
|
|
45
|
+
canary skills run canary-fail-fast -- --config playwright.config.ts
|
|
46
|
+
|
|
47
|
+
# Both at once:
|
|
48
|
+
canary skills run canary-fail-fast -- \
|
|
49
|
+
--results test-results/results.json \
|
|
50
|
+
--config playwright.config.ts
|
|
51
|
+
|
|
52
|
+
# Usage and the full flag list (exits 0):
|
|
53
|
+
canary skills run canary-fail-fast -- --help
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
At least one of `--results` / `--config` is required. The digest exits `1` when
|
|
57
|
+
any test failed (so the CI step fails); the config audit alone never fails the
|
|
58
|
+
build.
|
|
59
|
+
|
|
60
|
+
## CI wiring (GitHub Actions)
|
|
61
|
+
|
|
62
|
+
Run after the Playwright step with `if: always()` so the digest surfaces even
|
|
63
|
+
when the test step already failed:
|
|
64
|
+
|
|
65
|
+
```yaml
|
|
66
|
+
- name: Run Playwright
|
|
67
|
+
run:
|
|
68
|
+
npx playwright test --reporter=json --output-file=test-results/results.json
|
|
69
|
+
|
|
70
|
+
- name: Fail-fast digest
|
|
71
|
+
if: always()
|
|
72
|
+
run: |
|
|
73
|
+
canary skills run canary-fail-fast -- \
|
|
74
|
+
--results test-results/results.json
|
|
75
|
+
```
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// canary-fail-fast -- surface test failures fast and loud. Ported behavior-for-
|
|
3
|
+
// behavior from the Python original.
|
|
4
|
+
//
|
|
5
|
+
// Two halves:
|
|
6
|
+
// --config <playwright.config.*> audit fail-fast knobs (maxFailures/forbidOnly/
|
|
7
|
+
// retries); print recommendations (read-only).
|
|
8
|
+
// --results <playwright.json> print a loud, categorized failure digest to
|
|
9
|
+
// the CI log + ::error annotations; exit non-
|
|
10
|
+
// zero on any failure so the step fails.
|
|
11
|
+
//
|
|
12
|
+
// At least one of --config / --results is required. Self-contained -- no
|
|
13
|
+
// external skill dependency.
|
|
14
|
+
//
|
|
15
|
+
// Invoked via `canary skills run canary-fail-fast -- --results <json> [--config <path>]`.
|
|
16
|
+
|
|
17
|
+
import fs from 'node:fs';
|
|
18
|
+
|
|
19
|
+
import {
|
|
20
|
+
createParser,
|
|
21
|
+
formatUsageError,
|
|
22
|
+
EXIT_USAGE,
|
|
23
|
+
} from '../../../lib/parse-args.mjs';
|
|
24
|
+
import { parseFailures } from './parse.mjs';
|
|
25
|
+
import { buildDigest } from './digest.mjs';
|
|
26
|
+
import { checkConfig } from './fastfail_check.mjs';
|
|
27
|
+
|
|
28
|
+
const PREFIX = 'canary-fail-fast:';
|
|
29
|
+
const DASH = '\u2014'; // em dash (see digest.mjs)
|
|
30
|
+
|
|
31
|
+
const USAGE =
|
|
32
|
+
'usage: canary-fail-fast [-h] [--results PATH] [--config PATH]\n' +
|
|
33
|
+
'\n' +
|
|
34
|
+
'Fail-fast config audit + loud run-end failure digest.';
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Both value flags are optional here -- main enforces "at least one of
|
|
38
|
+
* --results / --config" itself, because argparse has no vocabulary for that.
|
|
39
|
+
* The four shared invariants live in the shared parser (#479).
|
|
40
|
+
*/
|
|
41
|
+
export const CLI_SPEC = {
|
|
42
|
+
prog: 'canary-fail-fast',
|
|
43
|
+
values: {
|
|
44
|
+
'--results': { key: 'results' },
|
|
45
|
+
'--config': { key: 'config' },
|
|
46
|
+
},
|
|
47
|
+
};
|
|
48
|
+
|
|
49
|
+
const parseArgs = createParser(CLI_SPEC);
|
|
50
|
+
|
|
51
|
+
export function main(argv = []) {
|
|
52
|
+
const { opts: args, help, error } = parseArgs(argv);
|
|
53
|
+
|
|
54
|
+
if (help) {
|
|
55
|
+
console.log(USAGE);
|
|
56
|
+
return 0;
|
|
57
|
+
}
|
|
58
|
+
if (error) {
|
|
59
|
+
console.error(formatUsageError(CLI_SPEC.prog, error));
|
|
60
|
+
return EXIT_USAGE;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
if (!args.results && !args.config) {
|
|
64
|
+
console.error(
|
|
65
|
+
`${PREFIX} nothing to do ${DASH} pass --results and/or --config.`,
|
|
66
|
+
);
|
|
67
|
+
return 1;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// ---- config audit -----------------------------------------------------
|
|
71
|
+
if (args.config) {
|
|
72
|
+
let text;
|
|
73
|
+
try {
|
|
74
|
+
text = fs.readFileSync(args.config, 'utf8');
|
|
75
|
+
} catch (exc) {
|
|
76
|
+
console.error(`${PREFIX} cannot read config: ${exc.message}`);
|
|
77
|
+
return 1;
|
|
78
|
+
}
|
|
79
|
+
const recs = checkConfig(text);
|
|
80
|
+
if (recs.length) {
|
|
81
|
+
console.log('Fail-fast config recommendations:');
|
|
82
|
+
for (const r of recs) console.log(` - ${r}`);
|
|
83
|
+
} else {
|
|
84
|
+
console.log('Fail-fast config OK.');
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// ---- failure digest ---------------------------------------------------
|
|
89
|
+
let exitCode = 0;
|
|
90
|
+
if (args.results) {
|
|
91
|
+
if (!fs.existsSync(args.results)) {
|
|
92
|
+
console.error(`${PREFIX} results file not found: ${args.results}`);
|
|
93
|
+
return 1;
|
|
94
|
+
}
|
|
95
|
+
let failures;
|
|
96
|
+
try {
|
|
97
|
+
failures = parseFailures(args.results);
|
|
98
|
+
} catch (exc) {
|
|
99
|
+
console.error(`${PREFIX} ${exc.message}`);
|
|
100
|
+
return 1;
|
|
101
|
+
}
|
|
102
|
+
const d = buildDigest(failures);
|
|
103
|
+
console.log(d.text);
|
|
104
|
+
for (const ann of d.annotations) console.log(ann);
|
|
105
|
+
exitCode = d.exitCode;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
return exitCode;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// Direct execution (the skill runner execs this file via its shebang).
|
|
112
|
+
//
|
|
113
|
+
// `process.exitCode`, not `process.exit()`: a large `--json` payload exceeds
|
|
114
|
+
// the pipe buffer, and `process.exit` tears the process down mid-write, leaving
|
|
115
|
+
// truncated JSON that still exits 0 (#791).
|
|
116
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
117
|
+
process.exitCode = main(process.argv.slice(2));
|
|
118
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
// digest -- loud, categorized failure digest (pure). Ported behavior-for-
|
|
2
|
+
// behavior from the Python original.
|
|
3
|
+
//
|
|
4
|
+
// Turns a list of failures into a terse CI-log digest + `::error` workflow
|
|
5
|
+
// annotations + a non-zero exit code, so an engineer triages from the run log
|
|
6
|
+
// without opening the HTML report.
|
|
7
|
+
|
|
8
|
+
import { FAILURE_CATEGORIES, categorizeFailure } from './failures.mjs';
|
|
9
|
+
|
|
10
|
+
// Output glyphs kept as \u escapes so the source stays ASCII while the emitted
|
|
11
|
+
// text is byte-identical to the Python original.
|
|
12
|
+
const CHECK = '\u2705'; // white heavy check mark
|
|
13
|
+
const CROSS = '\u274c'; // cross mark
|
|
14
|
+
const DASH = '\u2014'; // em dash
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* @typedef {{text: string, annotations: string[], exitCode: number}} Digest
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
function firstLine(error, limit = 160) {
|
|
21
|
+
if (!error) return '(no error message)';
|
|
22
|
+
for (const raw of error.split(/\r\n|\r|\n/)) {
|
|
23
|
+
const line = raw.trim();
|
|
24
|
+
if (line) return line.slice(0, limit);
|
|
25
|
+
}
|
|
26
|
+
return '(no error message)';
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Build the digest text, `::error` annotations, and exit code from failures. */
|
|
30
|
+
export function buildDigest(failures) {
|
|
31
|
+
if (!failures.length) {
|
|
32
|
+
return { text: `${CHECK} 0 failing tests.`, annotations: [], exitCode: 0 };
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const n = failures.length;
|
|
36
|
+
const byCat = new Map();
|
|
37
|
+
for (const f of failures) {
|
|
38
|
+
const cat = categorizeFailure(f.error);
|
|
39
|
+
if (!byCat.has(cat)) byCat.set(cat, []);
|
|
40
|
+
byCat.get(cat).push(f);
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const lines = [
|
|
44
|
+
`${CROSS} ${n} failing test${n !== 1 ? 's' : ''} ${DASH} triage by category:`,
|
|
45
|
+
'',
|
|
46
|
+
];
|
|
47
|
+
for (const cat of FAILURE_CATEGORIES) {
|
|
48
|
+
const bucket = byCat.get(cat);
|
|
49
|
+
if (!bucket || !bucket.length) continue;
|
|
50
|
+
lines.push(` ${cat} (${bucket.length}):`);
|
|
51
|
+
for (const f of bucket) {
|
|
52
|
+
lines.push(` - ${f.title} ${DASH} ${firstLine(f.error)}`);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
const text = lines.join('\n');
|
|
56
|
+
|
|
57
|
+
const annotations = [];
|
|
58
|
+
for (const f of failures) {
|
|
59
|
+
const cat = categorizeFailure(f.error);
|
|
60
|
+
let loc = '';
|
|
61
|
+
if (f.file) loc += `file=${f.file},`;
|
|
62
|
+
if (f.line !== null && f.line !== undefined) loc += `line=${f.line},`;
|
|
63
|
+
annotations.push(
|
|
64
|
+
`::error ${loc}title=Test failure::${f.title} ${DASH} ${cat}: ${firstLine(f.error)}`,
|
|
65
|
+
);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
return { text, annotations, exitCode: 1 };
|
|
69
|
+
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
// failures -- heuristic failure categorization (self-contained, pure). Ported
|
|
2
|
+
// behavior-for-behavior from the Python original.
|
|
3
|
+
//
|
|
4
|
+
// Order matters -- the most distinctive signals are checked first so 4xx/5xx
|
|
5
|
+
// status-code patterns don't swallow schema errors that happen to mention a
|
|
6
|
+
// status code in a response preview.
|
|
7
|
+
|
|
8
|
+
export const FAILURE_CATEGORIES = [
|
|
9
|
+
'schema',
|
|
10
|
+
'auth',
|
|
11
|
+
'server',
|
|
12
|
+
'client',
|
|
13
|
+
'timeout',
|
|
14
|
+
'network',
|
|
15
|
+
'other',
|
|
16
|
+
];
|
|
17
|
+
|
|
18
|
+
// [category, pattern] pairs in match-priority order (distinct from the display
|
|
19
|
+
// order in FAILURE_CATEGORIES). Patterns are case-insensitive and stateless
|
|
20
|
+
// (no `g` flag), mirroring Python's re.search with re.IGNORECASE.
|
|
21
|
+
//
|
|
22
|
+
// Intentional deviation (exotic input only): the `\b` word boundaries use
|
|
23
|
+
// JavaScript's ASCII-only word-character definition, whereas Python's `re` is
|
|
24
|
+
// Unicode-aware by default. An accented/non-ASCII letter glued directly to a
|
|
25
|
+
// status code (a "u"-with-umlaut immediately before `401`, say) can therefore
|
|
26
|
+
// categorize differently than the Python original. This is negligible for real
|
|
27
|
+
// Playwright error messages, which are
|
|
28
|
+
// ASCII around status codes; documented so the behavior reads as deliberate.
|
|
29
|
+
const RULES = [
|
|
30
|
+
[
|
|
31
|
+
'schema',
|
|
32
|
+
/ZodError|invalid[_ ]type|unrecognized key|expected .+ received|at path "|\bzod\b/i,
|
|
33
|
+
],
|
|
34
|
+
[
|
|
35
|
+
'auth',
|
|
36
|
+
/\b401\b|unauthorized|\b403\b|forbidden|invalid(?: auth)? token|token expired/i,
|
|
37
|
+
],
|
|
38
|
+
['timeout', /timeout|timed out|etimedout|deadline exceeded/i],
|
|
39
|
+
[
|
|
40
|
+
'network',
|
|
41
|
+
/econnrefused|enotfound|econnreset|socket hang up|getaddrinfo|network request failed/i,
|
|
42
|
+
],
|
|
43
|
+
[
|
|
44
|
+
'server',
|
|
45
|
+
/\b5\d{2}\b|internal server error|bad gateway|service unavailable|gateway timeout/i,
|
|
46
|
+
],
|
|
47
|
+
[
|
|
48
|
+
'client',
|
|
49
|
+
/\b4(?:0[045-9]|1\d|2\d)\b|bad request|not found|unprocessable|conflict/i,
|
|
50
|
+
],
|
|
51
|
+
];
|
|
52
|
+
|
|
53
|
+
/** Return the category of a failure error message ('other' when unknown). */
|
|
54
|
+
export function categorizeFailure(error) {
|
|
55
|
+
if (!error) return 'other';
|
|
56
|
+
for (const [category, pattern] of RULES) {
|
|
57
|
+
if (pattern.test(error)) return category;
|
|
58
|
+
}
|
|
59
|
+
return 'other';
|
|
60
|
+
}
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
// fastfail_check -- fail-fast config audit for a Playwright config (pure, read-
|
|
2
|
+
// only). Ported behavior-for-behavior from the Python original.
|
|
3
|
+
//
|
|
4
|
+
// Scans a `playwright.config.*` for the fail-fast knobs a suite should set so a
|
|
5
|
+
// broken CI run aborts early instead of burning the whole matrix. Never edits
|
|
6
|
+
// the config -- it recommends, and CANONICAL is the block to paste in.
|
|
7
|
+
|
|
8
|
+
const DASH = '\u2014'; // em dash (see digest.mjs)
|
|
9
|
+
|
|
10
|
+
// The block to paste into playwright.config.ts. Exported for reference; the CLI
|
|
11
|
+
// does not emit it.
|
|
12
|
+
export const CANONICAL = `export default defineConfig({
|
|
13
|
+
// Fail fast in CI: abort once enough has clearly broken, never on local runs.
|
|
14
|
+
forbidOnly: !!process.env.CI, // a stray test.only fails the build
|
|
15
|
+
maxFailures: process.env.CI ? 10 : 0, // stop the run after 10 failures in CI
|
|
16
|
+
retries: process.env.CI ? 2 : 0, // absorb flakes in CI; surface them locally
|
|
17
|
+
// ...your existing config
|
|
18
|
+
});
|
|
19
|
+
`;
|
|
20
|
+
|
|
21
|
+
// knob name -> why it matters, in the Python dict's insertion order.
|
|
22
|
+
const KNOBS = [
|
|
23
|
+
['forbidOnly', 'a stray `test.only` silently skips the rest of the suite'],
|
|
24
|
+
[
|
|
25
|
+
'maxFailures',
|
|
26
|
+
'a broken run keeps burning the matrix instead of aborting early',
|
|
27
|
+
],
|
|
28
|
+
[
|
|
29
|
+
'retries',
|
|
30
|
+
'flakes either fail the build or hide locally without a CI retry policy',
|
|
31
|
+
],
|
|
32
|
+
];
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Return one recommendation per fail-fast knob missing from the config text.
|
|
36
|
+
* Empty array means all knobs are present. Substring scan -- good enough to flag
|
|
37
|
+
* absence; it does not validate the knob's value.
|
|
38
|
+
*/
|
|
39
|
+
export function checkConfig(text) {
|
|
40
|
+
return KNOBS.filter(([knob]) => !text.includes(knob)).map(
|
|
41
|
+
([knob, why]) => `Add \`${knob}\` ${DASH} without it, ${why}.`,
|
|
42
|
+
);
|
|
43
|
+
}
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
// parse -- minimal Playwright JSON parser (failing tests only, self-contained).
|
|
2
|
+
// Ported behavior-for-behavior from the Python original.
|
|
3
|
+
//
|
|
4
|
+
// Walks the Playwright JSON reporter's nested suites/specs/tests and returns the
|
|
5
|
+
// real failures. A failed/unexpected test with a passing retry is flaky and
|
|
6
|
+
// excluded; without one it is a failure. Leading non-JSON banners are stripped
|
|
7
|
+
// before parsing (matches the reporter's defensive indexOf('{')).
|
|
8
|
+
|
|
9
|
+
import fs from 'node:fs';
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* @typedef {{title: string, status: string, file: (string|null),
|
|
13
|
+
* line: (number|null), error: (string|null)}} Failure
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
/** Build a Failure with the Python dataclass defaults (absent -> null). */
|
|
17
|
+
export function Failure(fields) {
|
|
18
|
+
return {
|
|
19
|
+
title: fields.title,
|
|
20
|
+
status: fields.status,
|
|
21
|
+
file: fields.file ?? null,
|
|
22
|
+
line: fields.line ?? null,
|
|
23
|
+
error: fields.error ?? null,
|
|
24
|
+
};
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function isObject(x) {
|
|
28
|
+
return x !== null && typeof x === 'object' && !Array.isArray(x);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Mirror Python dict.get: `.get` on a non-dict (str/list/number/None) raises
|
|
32
|
+
// AttributeError, which parseFailures converts into a structured error. Here a
|
|
33
|
+
// non-object access throws TypeError, caught by the same conversion.
|
|
34
|
+
function dget(obj, key, dflt) {
|
|
35
|
+
if (!isObject(obj)) throw new TypeError('value is not an object');
|
|
36
|
+
return key in obj ? obj[key] : dflt;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Mirror Python's `container or []` for a for-loop target. Python truthiness
|
|
40
|
+
// differs from JS: an empty object `{}` and empty string `""` are FALSY in
|
|
41
|
+
// Python but TRUTHY in JS. A naive `x || []` would turn `{"suites": {}}` into a
|
|
42
|
+
// thrown "unexpected structure" (JS iterates the truthy `{}`) where Python
|
|
43
|
+
// quietly iterates nothing -- flipping a CI pass into a fail. So:
|
|
44
|
+
// - any Python-falsy value (null/undefined/false/0/""/[]/{}) -> [] (iterate nothing)
|
|
45
|
+
// - an Array -> the array
|
|
46
|
+
// - any other truthy value (non-empty object/string/number) -> returned as-is,
|
|
47
|
+
// so `for...of` reproduces Python's iterate-then-`.get`-fails path (a string
|
|
48
|
+
// iterates its chars; a non-empty object/number is non-iterable and throws),
|
|
49
|
+
// which surfaces as the same "unexpected structure" error.
|
|
50
|
+
function asItems(v) {
|
|
51
|
+
if (Array.isArray(v)) return v;
|
|
52
|
+
if (v && typeof v === 'object') return Object.keys(v).length ? v : [];
|
|
53
|
+
if (v && typeof v !== 'object') return v; // non-empty string / truthy number
|
|
54
|
+
return []; // null/undefined/false/0/""
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** Parse a Playwright JSON results file, returning only the real failures. */
|
|
58
|
+
export function parseFailures(resultsPath) {
|
|
59
|
+
if (!fs.existsSync(resultsPath)) return [];
|
|
60
|
+
|
|
61
|
+
let text = fs.readFileSync(resultsPath, 'utf8');
|
|
62
|
+
const braceAt = text.indexOf('{');
|
|
63
|
+
if (braceAt > 0) text = text.slice(braceAt);
|
|
64
|
+
|
|
65
|
+
let data;
|
|
66
|
+
try {
|
|
67
|
+
data = JSON.parse(text);
|
|
68
|
+
} catch (exc) {
|
|
69
|
+
throw new Error(`results file is not valid JSON: ${exc.message}`);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
if (!isObject(data)) {
|
|
73
|
+
throw new Error("results file's top-level value must be an object");
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const failures = [];
|
|
77
|
+
try {
|
|
78
|
+
for (const suite of asItems(data.suites)) {
|
|
79
|
+
processSuite(suite, failures, '', '');
|
|
80
|
+
}
|
|
81
|
+
} catch (exc) {
|
|
82
|
+
if (exc instanceof TypeError) {
|
|
83
|
+
throw new Error(
|
|
84
|
+
`results file has an unexpected structure: ${exc.message}`,
|
|
85
|
+
);
|
|
86
|
+
}
|
|
87
|
+
throw exc;
|
|
88
|
+
}
|
|
89
|
+
return failures;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function processSuite(suite, failures, parentPath, suiteFile) {
|
|
93
|
+
const suiteTitle = dget(suite, 'title', '');
|
|
94
|
+
const suitePath = parentPath ? `${parentPath} > ${suiteTitle}` : suiteTitle;
|
|
95
|
+
const currentFile = dget(suite, 'file', undefined) || suiteFile;
|
|
96
|
+
|
|
97
|
+
for (const child of asItems(dget(suite, 'suites', []))) {
|
|
98
|
+
processSuite(child, failures, suitePath, currentFile);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
for (const spec of asItems(dget(suite, 'specs', []))) {
|
|
102
|
+
const specPath = `${suitePath} > ${dget(spec, 'title', '')}`;
|
|
103
|
+
const specLocation = dget(spec, 'location', undefined) || {};
|
|
104
|
+
for (const test of asItems(dget(spec, 'tests', []))) {
|
|
105
|
+
const testTitle =
|
|
106
|
+
dget(test, 'title', undefined) || dget(spec, 'title', '');
|
|
107
|
+
const testLocation = dget(test, 'location', undefined) || {};
|
|
108
|
+
const results = asItems(dget(test, 'results', undefined));
|
|
109
|
+
|
|
110
|
+
const status = dget(test, 'status', 'unknown');
|
|
111
|
+
if (status !== 'unexpected' && status !== 'failed') continue;
|
|
112
|
+
|
|
113
|
+
const hasPassingRetry = results.some((r) => {
|
|
114
|
+
const s = dget(r, 'status', undefined);
|
|
115
|
+
return s === 'passed' || s === 'expected';
|
|
116
|
+
});
|
|
117
|
+
if (hasPassingRetry) continue; // flaky -- excluded from the failure count
|
|
118
|
+
|
|
119
|
+
let error = null;
|
|
120
|
+
if (results.length) {
|
|
121
|
+
const last = results[results.length - 1];
|
|
122
|
+
const err = dget(last, 'error', undefined) || {};
|
|
123
|
+
error = dget(err, 'message', undefined);
|
|
124
|
+
if (error === undefined || error === null) {
|
|
125
|
+
const errs = dget(last, 'errors', undefined) || [];
|
|
126
|
+
if (errs.length) {
|
|
127
|
+
error = dget(errs[0], 'message', undefined);
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
failures.push(
|
|
133
|
+
Failure({
|
|
134
|
+
title: `${specPath} > ${testTitle}`,
|
|
135
|
+
status,
|
|
136
|
+
file:
|
|
137
|
+
dget(testLocation, 'file', undefined) ||
|
|
138
|
+
dget(specLocation, 'file', undefined) ||
|
|
139
|
+
currentFile,
|
|
140
|
+
line:
|
|
141
|
+
dget(testLocation, 'line', undefined) ||
|
|
142
|
+
dget(specLocation, 'line', undefined) ||
|
|
143
|
+
null,
|
|
144
|
+
error,
|
|
145
|
+
}),
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
}
|