canary-test-cli 7.1.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +327 -0
- package/agents/skills/canary:generate.md +49 -0
- package/agents/skills/canary:init.md +37 -0
- package/agents/skills/canary:migrate.md +66 -0
- package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
- package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
- package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
- package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
- package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
- package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
- package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
- package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
- package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
- package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
- package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
- package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
- package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
- package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
- package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
- package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
- package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
- package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
- package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
- package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
- package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
- package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
- package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
- package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
- package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
- package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
- package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
- package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
- package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
- package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
- package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
- package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
- package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
- package/agents/skills/lib/parse-args.mjs +275 -0
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +49 -72
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +27 -19
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/skill-dispatch.js +115 -0
- package/dist/engine/core/skill-examples.js +103 -3
- package/dist/engine/core/skill-registry.js +59 -4
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/test-files.js +77 -0
- package/dist/engine/core/vacuity-scanner.js +330 -15
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/analysis-emit.js +7 -2
- package/dist/engine/guardian/cli.js +277 -249
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +354 -223
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +171 -51
- package/dist/engine/workflow-cli.js +85 -65
- package/dist/reporters/testtracker.d.ts +1 -1
- package/dist/reporters/testtracker.js +1 -1
- package/package.json +3 -2
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-ci-ready
|
|
3
|
+
description: >
|
|
4
|
+
Analyses a test suite for CI readiness: coverage depth, flakiness, assertion
|
|
5
|
+
quality, critical path coverage, and suite runtime. Accepts documented
|
|
6
|
+
failures (quarantined tests with linked open issues count as verified).
|
|
7
|
+
Investigates config/auth failures using the consuming repo's declared
|
|
8
|
+
user_catalog_skill.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Canary: CI Ready
|
|
12
|
+
|
|
13
|
+
Analyses a test suite across five dimensions and produces a readiness score. Use
|
|
14
|
+
this before promoting a suite to CI, or as the convergence gate in
|
|
15
|
+
`/canary-test-pipeline`.
|
|
16
|
+
|
|
17
|
+
**Deterministic scorer:** run `canary ci-ready [--root <dir>] [--json]` first.
|
|
18
|
+
It scores every check that has a real input and reports `skip`, naming the
|
|
19
|
+
missing input, for every check that does not. A skip is never a pass. Today only
|
|
20
|
+
flakiness has a producer behind it, so expect the other four to skip until
|
|
21
|
+
their inputs exist. The verdict is `ready` (all five passed), `incomplete`
|
|
22
|
+
(nothing failed, something skipped), `not-ready` (exit 1) or `abstained`
|
|
23
|
+
(nothing scored, exit 3).
|
|
24
|
+
|
|
25
|
+
## When to Use
|
|
26
|
+
|
|
27
|
+
- Before wiring a new test suite into CI for the first time
|
|
28
|
+
|
|
29
|
+
- When a CI run is failing and you need to understand why
|
|
30
|
+
|
|
31
|
+
- As part of `/canary-test-pipeline` (Phase 0 and convergence gate)
|
|
32
|
+
|
|
33
|
+
- When asked "is this suite ready for CI?"
|
|
34
|
+
|
|
35
|
+
## The Five Checks
|
|
36
|
+
|
|
37
|
+
Run all five checks and score each pass / warn / fail.
|
|
38
|
+
|
|
39
|
+
### 1. Coverage depth
|
|
40
|
+
|
|
41
|
+
Read `.canary/test-inventory.json` if present. Nothing in canary produces this
|
|
42
|
+
file yet (there is no `canary coverage` command), so when it is absent this
|
|
43
|
+
check is a `skip`, not a pass or a fail.
|
|
44
|
+
|
|
45
|
+
Default threshold: depth ≥ 2 for all endpoints in critical areas. Override with
|
|
46
|
+
`--threshold <n>`.
|
|
47
|
+
|
|
48
|
+
- **pass** — all critical-area endpoints at depth ≥ threshold
|
|
49
|
+
|
|
50
|
+
- **warn** — some endpoints at depth 1 (hit but unasserted)
|
|
51
|
+
|
|
52
|
+
- **fail** — any critical-area endpoint at depth 0
|
|
53
|
+
|
|
54
|
+
### 2. Flakiness
|
|
55
|
+
|
|
56
|
+
Read `test-results/quarantine-ledger.json` (or the path in
|
|
57
|
+
`.canary/company.json` under `quarantine_ledger_path` if set).
|
|
58
|
+
|
|
59
|
+
No tool writes a quarantine ledger yet, so `canary ci-ready` scores flakiness
|
|
60
|
+
from the run-history store instead: the last 30 runs in
|
|
61
|
+
`test-results/reports/history-v2.jsonl`. Any test flaking in 10% or more of its
|
|
62
|
+
runs fails the check, any lower flake rate warns, and no flakes passes.
|
|
63
|
+
|
|
64
|
+
A quarantined test is acceptable only when it has a linked open issue (Jira or
|
|
65
|
+
GitHub). Check issue state:
|
|
66
|
+
|
|
67
|
+
- Linked issue **open** → counts as verified (documented, tracked)
|
|
68
|
+
|
|
69
|
+
- Linked issue **closed** → flag: quarantine should be resolved
|
|
70
|
+
|
|
71
|
+
- **No linked issue** → fail: unlinked quarantine blocks CI-ready
|
|
72
|
+
|
|
73
|
+
### 3. Assertion quality
|
|
74
|
+
|
|
75
|
+
Read depth scores from the inventory. In critical-area endpoints:
|
|
76
|
+
|
|
77
|
+
- **pass** — all tests at depth ≥ 2 (shaped assertions: result.ok or equivalent)
|
|
78
|
+
|
|
79
|
+
- **warn** — some tests at depth 1 (status-only assertions)
|
|
80
|
+
|
|
81
|
+
- **fail** — majority of critical-path tests at depth 1
|
|
82
|
+
|
|
83
|
+
### 4. Critical path coverage
|
|
84
|
+
|
|
85
|
+
Only run this check if `.canary/critical-areas.json` is present.
|
|
86
|
+
|
|
87
|
+
Cross-reference the top 5 risk-scored areas from `critical-areas.json` against
|
|
88
|
+
`test-inventory.json`:
|
|
89
|
+
|
|
90
|
+
- **pass** — all top-5 areas have at least one test at depth ≥ 1
|
|
91
|
+
|
|
92
|
+
- **warn** — one area uncovered
|
|
93
|
+
|
|
94
|
+
- **fail** — two or more top areas uncovered
|
|
95
|
+
|
|
96
|
+
- **skip** — `critical-areas.json` absent (note this in output, not a failure)
|
|
97
|
+
|
|
98
|
+
### 5. Suite runtime
|
|
99
|
+
|
|
100
|
+
Run history lives in `test-results/reports/history-v2.jsonl`. The store does not
|
|
101
|
+
record run or test durations today, so there is no p95 to compute and
|
|
102
|
+
`canary ci-ready` reports this check as `skip`. The scoring below applies once
|
|
103
|
+
durations are recorded.
|
|
104
|
+
|
|
105
|
+
**With harness MCP available:** score the p95 against trend history rather than
|
|
106
|
+
an absolute clock. Call `get_perf_baselines` and compare this run's p95 to the
|
|
107
|
+
recorded baseline for the suite:
|
|
108
|
+
|
|
109
|
+
- **pass** — p95 within the baseline's tolerance, or an improvement
|
|
110
|
+
- **warn** — p95 regressed past tolerance but under 2× the baseline
|
|
111
|
+
- **fail** — p95 at or over 2× the baseline
|
|
112
|
+
- **skip** — no baseline recorded yet (this is a baseline-capture run, not a
|
|
113
|
+
failure — say so in the output)
|
|
114
|
+
|
|
115
|
+
After scoring, record the run back into the baseline with
|
|
116
|
+
`update_perf_baselines` so the trend keeps moving. Do **not** record a run that
|
|
117
|
+
failed for unrelated reasons — a broken run's runtime is not a data point.
|
|
118
|
+
|
|
119
|
+
Regression beats absolute here. A suite that has always taken 11 minutes is a
|
|
120
|
+
fact of life; a suite that went from 3 minutes to 11 this week is the actual
|
|
121
|
+
signal, and an absolute threshold cannot tell those apart — it fails the first
|
|
122
|
+
forever and stays silent on the second until it crosses the line.
|
|
123
|
+
|
|
124
|
+
**Fallback (no MCP):** judge against absolute thresholds.
|
|
125
|
+
|
|
126
|
+
- **pass** — p95 under the configured timeout (default: 5 minutes)
|
|
127
|
+
- **warn** — p95 between 5–10 minutes
|
|
128
|
+
- **fail** — p95 over 10 minutes, or no run history (cannot assess)
|
|
129
|
+
|
|
130
|
+
State which mode was used in the output — "runtime vs. baseline" or "runtime vs.
|
|
131
|
+
absolute threshold" — so a reader knows whether a pass means "no regression" or
|
|
132
|
+
merely "under the clock".
|
|
133
|
+
|
|
134
|
+
## User Catalog Investigation
|
|
135
|
+
|
|
136
|
+
When a test fails with an auth, permission, or configuration error:
|
|
137
|
+
|
|
138
|
+
1. Read `user_catalog_skill` from `.canary/company.json`
|
|
139
|
+
2. If present: invoke `canary skills run <user_catalog_skill>` with the required
|
|
140
|
+
attributes from the error context; surface any matching user as a suggestion
|
|
141
|
+
3. If absent, or no matching user found: present constructively —
|
|
142
|
+
|
|
143
|
+
> "This failure may be a test user or test data configuration issue. Check
|
|
144
|
+
> your user catalog if you have one, or set up the required test data before
|
|
145
|
+
> re-running."
|
|
146
|
+
|
|
147
|
+
Never reference a specific catalog skill by name in output.
|
|
148
|
+
|
|
149
|
+
## Output Format
|
|
150
|
+
|
|
151
|
+
```text
|
|
152
|
+
CI Readiness — <repo-name>
|
|
153
|
+
|
|
154
|
+
✓ / ⚠ / ✗ <check name> <brief finding>
|
|
155
|
+
...
|
|
156
|
+
|
|
157
|
+
Score: N/5 — CI-READY or NOT CI-READY
|
|
158
|
+
|
|
159
|
+
Runtime scored vs. baseline | vs. absolute threshold
|
|
160
|
+
|
|
161
|
+
<gap list with suggested next actions>
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Score of 5/5 = CI-READY. Any fail = NOT CI-READY. Warns do not block.
|
|
165
|
+
|
|
166
|
+
## Flags
|
|
167
|
+
|
|
168
|
+
- `--threshold <n>` — minimum depth for coverage check (default: 2)
|
|
169
|
+
|
|
170
|
+
## Related skills
|
|
171
|
+
|
|
172
|
+
- `/canary-test-pipeline` — orchestrates this skill as Phase 0 and convergence
|
|
173
|
+
gate
|
|
174
|
+
|
|
175
|
+
- `/canary-critical-areas` — produces `critical-areas.json` used by check 4
|
|
176
|
+
|
|
177
|
+
- `canary-unquarantine` (overlay) — resolves quarantined tests once bugs are
|
|
178
|
+
fixed
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
name: canary-ci-ready
|
|
2
|
+
version: '1.0.0'
|
|
3
|
+
description:
|
|
4
|
+
Analyse a test suite for CI readiness across coverage depth, flakiness,
|
|
5
|
+
assertion quality, critical-path coverage, and runtime; documented/quarantined
|
|
6
|
+
failures count as verified.
|
|
7
|
+
stability: static
|
|
8
|
+
triggers:
|
|
9
|
+
- manual
|
|
10
|
+
platforms:
|
|
11
|
+
- claude-code
|
|
12
|
+
type: rigid
|
|
13
|
+
tools: []
|
|
14
|
+
tier: 1
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-company-knowledge
|
|
3
|
+
description: >
|
|
4
|
+
Scaffold .canary/company.json — the org-specific pointer file (Confluence
|
|
5
|
+
spaces, Jira projects, internal docs/domains, MCP servers, dashboard, and the
|
|
6
|
+
user-catalog skill) that canary-ci-ready and canary-failure-impact silently
|
|
7
|
+
assume already exists. Use when the user says "set up company knowledge",
|
|
8
|
+
"bootstrap company.json", "no company.json", "canary company-knowledge init",
|
|
9
|
+
or when canary-ci-ready/canary-failure-impact report no user-catalog config
|
|
10
|
+
for auth/config failures. Scaffolds the file and prompts for the fields that
|
|
11
|
+
genuinely cannot be inferred — it does not claim full automation.
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
# Canary: Company Knowledge Init
|
|
15
|
+
|
|
16
|
+
Wraps `canary company-knowledge init` to scaffold `.canary/company.json` — the
|
|
17
|
+
_pointers-only_ file that lets Canary skills reach into your org's internal
|
|
18
|
+
tooling (Confluence, Jira, internal docs, MCP servers, a user catalog) without
|
|
19
|
+
ever committing proprietary content to this open-core repo.
|
|
20
|
+
|
|
21
|
+
This file is silently assumed to exist by
|
|
22
|
+
[`canary-ci-ready`](../canary-ci-ready/SKILL.md) (user-catalog investigation for
|
|
23
|
+
auth/config failures) and
|
|
24
|
+
[`canary-failure-impact`](../canary-failure-impact/SKILL.md) (same). If neither
|
|
25
|
+
has ever been run with a populated `company.json`, both degrade to a generic
|
|
26
|
+
"check your user catalog if you have one" message. This skill closes that gap.
|
|
27
|
+
|
|
28
|
+
## When to Use
|
|
29
|
+
|
|
30
|
+
- First-time setup for a new project or team — "we need to configure company
|
|
31
|
+
knowledge for canary"
|
|
32
|
+
- `canary-ci-ready` or `canary-failure-impact` fell back to the generic
|
|
33
|
+
user-catalog message and the user wants real investigation instead
|
|
34
|
+
- `canary company-knowledge show` reports "No company knowledge configured"
|
|
35
|
+
- Re-running setup to add a field that was skipped the first time (safe — `init`
|
|
36
|
+
merges onto existing values by default)
|
|
37
|
+
- NOT for storing secrets, API keys, or tokens — this file holds pointers only;
|
|
38
|
+
the CLI's secret heuristic (`_looks_like_secret`) rejects
|
|
39
|
+
`sk-`/`token`/`bearer`-shaped values and anything over 128 chars outside
|
|
40
|
+
`notes`
|
|
41
|
+
- NOT for company-specific proprietary content (client names, internal runbooks,
|
|
42
|
+
populated data) — per this repo's open-core boundary (`AGENTS.md`), that lives
|
|
43
|
+
only in a private overlay under `.canary/skills/` or the org's own tooling,
|
|
44
|
+
reached _via_ the pointers this file stores
|
|
45
|
+
|
|
46
|
+
## What This Skill Cannot Automate
|
|
47
|
+
|
|
48
|
+
Every field in `.canary/company.json` is org-specific and cannot be reliably
|
|
49
|
+
inferred from the codebase alone — this repo is public/open-core by design, so
|
|
50
|
+
nothing in it names a real company, Jira project, or internal host. Don't claim
|
|
51
|
+
otherwise. What this skill _can_ do:
|
|
52
|
+
|
|
53
|
+
- Scaffold the file and `.gitignore` entry so it exists and is never
|
|
54
|
+
accidentally committed
|
|
55
|
+
- Prompt for each field with the exact validation format the CLI expects, so the
|
|
56
|
+
user isn't guessing at schema
|
|
57
|
+
- Detect when a value looks like a secret and refuse it before it reaches disk
|
|
58
|
+
- Merge onto existing values, so re-running is always safe
|
|
59
|
+
|
|
60
|
+
What it cannot do: know your Confluence space key, your Jira project prefix, or
|
|
61
|
+
which MCP server your org runs. Those come from the user.
|
|
62
|
+
|
|
63
|
+
## Process
|
|
64
|
+
|
|
65
|
+
### Phase 1: CHECK — Does It Already Exist?
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
canary company-knowledge show
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
- **"No company knowledge configured"** → proceed to Phase 2 (fresh scaffold).
|
|
72
|
+
- **Populated output** → note which fields are already set (shown with
|
|
73
|
+
`sources:` — `~/.canary/company.json`, `.canary/company.json`, or an env
|
|
74
|
+
layer). Proceed to Phase 2 in merge mode (no `--force`) so existing values
|
|
75
|
+
survive as defaults.
|
|
76
|
+
- **`⚠` error line** (secret-like value or malformed JSON in an existing layer)
|
|
77
|
+
→ surface the exact warning to the user before continuing; that layer is being
|
|
78
|
+
skipped entirely until fixed.
|
|
79
|
+
|
|
80
|
+
### Phase 2: SCAFFOLD — Run the Interactive Init
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
canary company-knowledge init
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Walk the user through each prompt. For each field, explain what it's for and
|
|
87
|
+
give the expected shape before they answer — the CLI accepts blank input to skip
|
|
88
|
+
or keep the current value:
|
|
89
|
+
|
|
90
|
+
| Field | Format | Purpose |
|
|
91
|
+
| -------------------- | ---------------------------------------------------------- | ---------------------------------------------- |
|
|
92
|
+
| `confluence_spaces` | comma-separated, uppercase (`QA, ENG`) | spaces the LLM should consult for org docs |
|
|
93
|
+
| `jira_projects` | comma-separated, uppercase (`PROJ, OPS`) | projects for ticket cross-referencing |
|
|
94
|
+
| `internal_doc_urls` | one URL per line, `http(s)://` only | specific reference docs to fetch via MCP |
|
|
95
|
+
| `internal_domains` | comma-separated hostnames (`corp.example.com`) | flags internal-only URLs in generated content |
|
|
96
|
+
| `mcp_servers` | comma-separated identifiers (`plugin_atlassian_atlassian`) | which configured MCP server(s) back the above |
|
|
97
|
+
| `claude_code_skills` | comma-separated slugs (`team:skill-name`) | project-overlay skills to surface as available |
|
|
98
|
+
| `notes` | free text, ≤2048 chars, no secrets | anything else the LLM should know |
|
|
99
|
+
|
|
100
|
+
Fields not prompted by `init` (`dashboard_url`, `dashboard_token_env`,
|
|
101
|
+
`otel_exporter_endpoint`) can be added by hand-editing the JSON afterward —
|
|
102
|
+
mention this if the user needs dashboard/OTel wiring; don't skip it silently.
|
|
103
|
+
|
|
104
|
+
If the user offers a value that looks like a credential (starts with `sk-`,
|
|
105
|
+
`token`, `bearer`, or is unusually long), stop and remind them: secrets go in
|
|
106
|
+
environment variables, never in `company.json`. The CLI will reject the whole
|
|
107
|
+
layer if one slips through — better to catch it before submitting.
|
|
108
|
+
|
|
109
|
+
### Phase 3: WIRE THE USER-CATALOG SKILL (Known Schema Gap)
|
|
110
|
+
|
|
111
|
+
`canary-ci-ready` and `canary-failure-impact` both read a `user_catalog_skill`
|
|
112
|
+
key directly from `.canary/company.json` to investigate auth/config failures.
|
|
113
|
+
**This key is not part of the CLI's validated schema** — it isn't in
|
|
114
|
+
`CompanyKnowledge`'s known fields, `company-knowledge init` never prompts for
|
|
115
|
+
it, and `company-knowledge show` will list it under "ignored unknown field" if
|
|
116
|
+
you inspect warnings. It still works for the two consuming skills because they
|
|
117
|
+
read the raw JSON file directly rather than going through the Python loader —
|
|
118
|
+
but be transparent with the user that this is an informal extension, not a
|
|
119
|
+
first-class field, until the schema catches up.
|
|
120
|
+
|
|
121
|
+
To wire it:
|
|
122
|
+
|
|
123
|
+
1. Ask the user which project-overlay skill (if any) looks up test users — e.g.
|
|
124
|
+
`team:user-lookup`. If they don't have one, skip this phase; the consuming
|
|
125
|
+
skills degrade gracefully to a generic prompt.
|
|
126
|
+
2. If they gave a normal skill slug, prefer adding it to `claude_code_skills` (a
|
|
127
|
+
real, validated field) _and_ separately hand-edit `.canary/company.json` to
|
|
128
|
+
add the literal key:
|
|
129
|
+
|
|
130
|
+
```json
|
|
131
|
+
{
|
|
132
|
+
"claude_code_skills": ["team:user-lookup"],
|
|
133
|
+
"user_catalog_skill": "team:user-lookup"
|
|
134
|
+
}
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
3. Confirm the value matches what `canary skills run <user_catalog_skill>`
|
|
138
|
+
expects as an identifier (same slug format as `claude_code_skills`:
|
|
139
|
+
`^[a-z0-9][a-z0-9_-]*(:[a-z0-9][a-z0-9_-]*)?$`).
|
|
140
|
+
|
|
141
|
+
### Phase 4: VERIFY
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
canary company-knowledge show
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
- Confirm every field the user just set appears in the printed output.
|
|
148
|
+
- Confirm no `⚠` warnings (unknown fields other than the intentional
|
|
149
|
+
`user_catalog_skill` extension, dropped invalid entries, secret detections).
|
|
150
|
+
- Confirm `.gitignore` now contains a `.canary/` line — `init` adds it
|
|
151
|
+
automatically, but verify if the project already had a `.gitignore` with
|
|
152
|
+
unusual formatting.
|
|
153
|
+
- If JSON output is useful for the user's own tooling: `--json`.
|
|
154
|
+
|
|
155
|
+
## Error Handling
|
|
156
|
+
|
|
157
|
+
| Situation | What Happens | What To Do |
|
|
158
|
+
| --------------------------------------------------- | ---------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
|
|
159
|
+
| `.canary/company.json` already exists, no `--force` | `init` shows existing values as defaults and merges | Normal — just re-run `init` |
|
|
160
|
+
| Secret-like value entered | That whole layer is dropped; `show` prints a red `✗` | Remove the value, use an env var, re-run |
|
|
161
|
+
| Malformed JSON in an existing file | Layer skipped with a parse error | Fix the JSON by hand, then re-run `show` to confirm |
|
|
162
|
+
| Invalid entry format (e.g. lowercase Jira key) | Entry silently dropped, warning logged | Re-enter in the correct case/format |
|
|
163
|
+
| Unknown field in the file | Warned, not fatal, field ignored by the loader | Expected for `user_catalog_skill` today (Phase 3) — otherwise likely a typo |
|
|
164
|
+
| User has no Confluence/Jira/MCP setup at all | Every field is legitimately empty | Skip `init` entirely; `canary-ci-ready`/`canary-failure-impact` degrade to their generic prompts, which is correct behavior, not a bug |
|
|
165
|
+
|
|
166
|
+
## Examples
|
|
167
|
+
|
|
168
|
+
### Example: First-time team lead setup
|
|
169
|
+
|
|
170
|
+
**Prompt:** "Set up company knowledge for canary on this repo."
|
|
171
|
+
|
|
172
|
+
**Action:** Run `company-knowledge show` → "No company knowledge configured."
|
|
173
|
+
Run `company-knowledge init`. Walk through each field; the team lead has a
|
|
174
|
+
Confluence space (`QA`), a Jira project (`OPS`), and an Atlassian MCP server
|
|
175
|
+
(`plugin_atlassian_atlassian`) but no internal dashboard yet. Leave
|
|
176
|
+
`dashboard_url` unset. Ask about a user-catalog skill — they don't have one, so
|
|
177
|
+
Phase 3 is skipped. Verify with `show`.
|
|
178
|
+
|
|
179
|
+
### Example: Retrofitting `user_catalog_skill` onto an existing file
|
|
180
|
+
|
|
181
|
+
**Prompt:** "`canary-ci-ready` keeps telling me to check my user catalog
|
|
182
|
+
manually, but we have a skill for that."
|
|
183
|
+
|
|
184
|
+
**Action:** Run `show` — confirm `.canary/company.json` already has
|
|
185
|
+
`confluence_spaces`/`jira_projects` set from a prior run. Ask for the skill slug
|
|
186
|
+
(`team:test-user-lookup`). Add it to both `claude_code_skills` and the raw
|
|
187
|
+
`user_catalog_skill` key per Phase 3. Re-run `show` to confirm no new warnings,
|
|
188
|
+
then re-run `canary-ci-ready` on a known failing auth test to confirm the lookup
|
|
189
|
+
now fires.
|
|
190
|
+
|
|
191
|
+
## Related Skills
|
|
192
|
+
|
|
193
|
+
- [`canary-ci-ready`](../canary-ci-ready/SKILL.md) — consumes
|
|
194
|
+
`user_catalog_skill` for auth/config failure investigation
|
|
195
|
+
- [`canary-failure-impact`](../canary-failure-impact/SKILL.md) — same
|
|
196
|
+
investigation pattern, different trigger context
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-critical-areas
|
|
3
|
+
description: >
|
|
4
|
+
Risk-based test prioritisation. Given a codebase or diff, identifies which
|
|
5
|
+
areas carry the most risk using git churn, downstream dependents,
|
|
6
|
+
business-critical signals, and existing coverage depth. Produces a ranked list
|
|
7
|
+
with recommended test types per area.
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# Canary: Critical Areas
|
|
11
|
+
|
|
12
|
+
Identifies the highest-risk areas of a codebase so test effort goes where it
|
|
13
|
+
matters most. Uses multiple signals, degrades gracefully when advanced tooling
|
|
14
|
+
is unavailable.
|
|
15
|
+
|
|
16
|
+
## When to Use
|
|
17
|
+
|
|
18
|
+
- Before writing new tests: "where should I focus?"
|
|
19
|
+
|
|
20
|
+
- After a large diff lands: "what did this change put at risk?"
|
|
21
|
+
|
|
22
|
+
- As Phase 1 of `/canary-test-pipeline`
|
|
23
|
+
|
|
24
|
+
- When asked to prioritise test coverage
|
|
25
|
+
|
|
26
|
+
## Signals
|
|
27
|
+
|
|
28
|
+
Collect all available signals, score each area, and rank by composite risk
|
|
29
|
+
score.
|
|
30
|
+
|
|
31
|
+
### 1. Churn / hotspot (always available)
|
|
32
|
+
|
|
33
|
+
**With harness MCP available:** call `detect_anomalies` (metric `hotspotScore`,
|
|
34
|
+
plus its co-change / single-point-of-failure signals). Harness's hotspot score
|
|
35
|
+
already blends churn with structural risk, so use it directly as this signal and
|
|
36
|
+
skip the raw `git log` pass. Normalise the returned scores to 0–1.
|
|
37
|
+
|
|
38
|
+
**Fallback (no MCP):**
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
git log --stat --since="90 days ago" -- <path> | grep -c "^"
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Files changed most frequently in the last 90 days score higher. Normalise to 0–1
|
|
45
|
+
across all files in scope.
|
|
46
|
+
|
|
47
|
+
### 2. Downstream dependents
|
|
48
|
+
|
|
49
|
+
**With harness MCP available:** call `get_impact` for each candidate file
|
|
50
|
+
(`filePath`, `mode: "summary"`) and read the affected-node counts it returns
|
|
51
|
+
(tests / docs / code grouped by type). This is the purpose-built impact
|
|
52
|
+
primitive — do **not** hand-walk `get_relationships` edge-by-edge; `get_impact`
|
|
53
|
+
already computes the transitive downstream set. More affected nodes ⇒ higher
|
|
54
|
+
score.
|
|
55
|
+
|
|
56
|
+
**Fallback (no MCP):** scan for `import` statements referencing each file using
|
|
57
|
+
`grep -r`. Count unique files that import each candidate.
|
|
58
|
+
|
|
59
|
+
Files with more inbound dependents score higher — a change here breaks more.
|
|
60
|
+
|
|
61
|
+
### 3. Business-critical / critical-path flags
|
|
62
|
+
|
|
63
|
+
**With harness MCP available:** call `get_critical_paths` and add a fixed boost
|
|
64
|
+
(+0.3) to any area whose functions appear in the returned perf-critical set.
|
|
65
|
+
Also query `ask_graph` for `business_fact` nodes associated with each area; any
|
|
66
|
+
business-critical annotation adds the same +0.3 boost (apply the boost once,
|
|
67
|
+
whichever signal fires).
|
|
68
|
+
|
|
69
|
+
**Fallback:** skip this signal silently (do not penalise the score).
|
|
70
|
+
|
|
71
|
+
### 4. Coverage depth boost
|
|
72
|
+
|
|
73
|
+
If `.canary/test-inventory.json` is present: files whose endpoints are at depth
|
|
74
|
+
0 or 1 receive a boost (+0.15) — low depth in a high-churn file is especially
|
|
75
|
+
risky.
|
|
76
|
+
|
|
77
|
+
## Risk Score
|
|
78
|
+
|
|
79
|
+
```text
|
|
80
|
+
risk_score = (churn * 0.35) + (dependents * 0.35) + (business_critical * 0.30) + depth_boost
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Capped at 1.0. Round to 2 decimal places.
|
|
84
|
+
|
|
85
|
+
## Output
|
|
86
|
+
|
|
87
|
+
```text
|
|
88
|
+
Critical areas — <repo> (<N> files analysed)
|
|
89
|
+
|
|
90
|
+
1. src/loyalty/points.service.ts risk 0.92 ████████████
|
|
91
|
+
signals: high churn · 12 dependents · business_critical
|
|
92
|
+
recommended: api + integration tests
|
|
93
|
+
|
|
94
|
+
2. src/billing/charge.service.ts risk 0.78 ██████████
|
|
95
|
+
signals: high churn · billing domain
|
|
96
|
+
recommended: api tests · /canary-failure-impact suggested
|
|
97
|
+
|
|
98
|
+
...
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Show at most 10 areas. If more than 10 qualify, note the total and offer to show
|
|
102
|
+
all.
|
|
103
|
+
|
|
104
|
+
## Optional Artifact
|
|
105
|
+
|
|
106
|
+
When `--save` flag is passed (or when invoked by `/canary-test-pipeline`), write
|
|
107
|
+
`.canary/critical-areas.json`:
|
|
108
|
+
|
|
109
|
+
```json
|
|
110
|
+
{
|
|
111
|
+
"generated": "<ISO timestamp>",
|
|
112
|
+
"areas": [
|
|
113
|
+
{
|
|
114
|
+
"path": "src/loyalty/points.service.ts",
|
|
115
|
+
"risk_score": 0.92,
|
|
116
|
+
"signals": ["high_churn", "many_dependents", "business_critical"],
|
|
117
|
+
"recommended_test_types": ["api", "integration"],
|
|
118
|
+
"summary": "High-churn service with 12 downstream dependents"
|
|
119
|
+
}
|
|
120
|
+
]
|
|
121
|
+
}
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
This file is consumed as opt-in context by `/canary-edge-cases` and
|
|
125
|
+
`/canary-failure-impact`.
|
|
126
|
+
|
|
127
|
+
## Flags
|
|
128
|
+
|
|
129
|
+
- `--diff <git ref>` — scope analysis to files changed in a diff
|
|
130
|
+
(`git diff <ref>...HEAD`)
|
|
131
|
+
|
|
132
|
+
- `--save` — write `critical-areas.json`
|
|
133
|
+
|
|
134
|
+
## Related skills
|
|
135
|
+
|
|
136
|
+
- `/canary-ci-ready` — check 4 consumes `critical-areas.json`
|
|
137
|
+
|
|
138
|
+
- `/canary-edge-cases` — focuses edge cases on critical areas when JSON present
|
|
139
|
+
|
|
140
|
+
- `/canary-failure-impact` — focuses tracing on critical paths when JSON present
|
|
141
|
+
|
|
142
|
+
- `/canary-test-pipeline` — Phase 1
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
name: canary-critical-areas
|
|
2
|
+
version: '1.0.0'
|
|
3
|
+
description:
|
|
4
|
+
Risk-rank a codebase or diff by git churn, downstream dependents,
|
|
5
|
+
business-critical signals, and coverage depth; outputs a ranked list with
|
|
6
|
+
recommended test types per area.
|
|
7
|
+
stability: static
|
|
8
|
+
triggers:
|
|
9
|
+
- manual
|
|
10
|
+
platforms:
|
|
11
|
+
- claude-code
|
|
12
|
+
type: rigid
|
|
13
|
+
tools: []
|
|
14
|
+
tier: 1
|
|
15
|
+
depends_on:
|
|
16
|
+
- canary-ci-ready
|