canary-test-cli 7.1.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +327 -0
- package/agents/skills/canary:generate.md +49 -0
- package/agents/skills/canary:init.md +37 -0
- package/agents/skills/canary:migrate.md +66 -0
- package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
- package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
- package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
- package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
- package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
- package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
- package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
- package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
- package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
- package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
- package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
- package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
- package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
- package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
- package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
- package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
- package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
- package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
- package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
- package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
- package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
- package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
- package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
- package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
- package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
- package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
- package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
- package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
- package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
- package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
- package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
- package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
- package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
- package/agents/skills/lib/parse-args.mjs +275 -0
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +49 -72
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +27 -19
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/skill-dispatch.js +115 -0
- package/dist/engine/core/skill-examples.js +103 -3
- package/dist/engine/core/skill-registry.js +59 -4
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/test-files.js +77 -0
- package/dist/engine/core/vacuity-scanner.js +330 -15
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/analysis-emit.js +7 -2
- package/dist/engine/guardian/cli.js +277 -249
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +354 -223
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +171 -51
- package/dist/engine/workflow-cli.js +85 -65
- package/dist/reporters/testtracker.d.ts +1 -1
- package/dist/reporters/testtracker.js +1 -1
- package/package.json +3 -2
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
// span_reader -- merge/correlate OTel span JSONL files into a Trace (pure,
|
|
2
|
+
// read-only). Ported from Python span_reader.py, behavior-for-behavior.
|
|
3
|
+
//
|
|
4
|
+
// Reads one or more `otel-spans.<worker>.jsonl` files (one JSON span object
|
|
5
|
+
// per line, written by otel_bootstrap/instrument.mjs), groups spans by
|
|
6
|
+
// `traceId`, resolves each trace's root span (the one carrying a `test.id`
|
|
7
|
+
// attribute -- set by otel_bootstrap/playwright-fixture.ts's root-span
|
|
8
|
+
// fixture), and attaches that trace's HTTP child spans to the resolved test.
|
|
9
|
+
// Traces with no `test.id`-attributed root bucket their HTTP spans under the
|
|
10
|
+
// synthetic test id "__setup__" (traffic outside any test, e.g. global setup).
|
|
11
|
+
//
|
|
12
|
+
// Assumed span envelope: {traceId, spanId, parentSpanId, name, startTime,
|
|
13
|
+
// duration_ms, attributes{}} -- matches exactly what instrument.mjs's
|
|
14
|
+
// JsonlFileSpanExporter writes; no `endTime` key is emitted (duration_ms +
|
|
15
|
+
// startTime cover it). HTTP attributes keyed http.method/http.request.method,
|
|
16
|
+
// http.url, http.route, http.status_code.
|
|
17
|
+
|
|
18
|
+
import fs from 'node:fs';
|
|
19
|
+
import path from 'node:path';
|
|
20
|
+
|
|
21
|
+
import { RequestSpan, TestTrace, Trace } from './run_types.mjs';
|
|
22
|
+
|
|
23
|
+
const SETUP_TEST_ID = '__setup__';
|
|
24
|
+
|
|
25
|
+
// Mirrors Python's Path.glob("otel-spans.*.jsonl"): the `*` matches any run of
|
|
26
|
+
// characters (including none) within a single path segment.
|
|
27
|
+
const SPAN_FILE_RE = /^otel-spans\..*\.jsonl$/;
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Read every `otel-spans.*.jsonl` file under `spansDir`, correlate spans to
|
|
31
|
+
* their test roots, and return the Trace. A missing or non-directory path
|
|
32
|
+
* yields an empty trace (never throws) -- the same as the Python version.
|
|
33
|
+
* @param {string} spansDir
|
|
34
|
+
*/
|
|
35
|
+
export function readTraces(spansDir) {
|
|
36
|
+
const byTrace = new Map();
|
|
37
|
+
|
|
38
|
+
if (isDir(spansDir)) {
|
|
39
|
+
const files = fs
|
|
40
|
+
.readdirSync(spansDir)
|
|
41
|
+
.filter((name) => SPAN_FILE_RE.test(name))
|
|
42
|
+
.sort();
|
|
43
|
+
for (const name of files) {
|
|
44
|
+
for (const span of readJsonl(path.join(spansDir, name))) {
|
|
45
|
+
const traceId = span.traceId;
|
|
46
|
+
if (!traceId) continue;
|
|
47
|
+
if (!byTrace.has(traceId)) byTrace.set(traceId, []);
|
|
48
|
+
byTrace.get(traceId).push(span);
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
const byTest = [];
|
|
54
|
+
const setupRequests = [];
|
|
55
|
+
let spansTotal = 0;
|
|
56
|
+
|
|
57
|
+
for (const [traceId, spans] of byTrace) {
|
|
58
|
+
const root = spans.find(isTestRoot) ?? null;
|
|
59
|
+
const httpSpans = spans.filter((s) => s !== root && isHttpSpan(s));
|
|
60
|
+
const requests = httpSpans.map(toRequestSpan);
|
|
61
|
+
spansTotal += requests.length;
|
|
62
|
+
|
|
63
|
+
if (root === null) {
|
|
64
|
+
setupRequests.push(...requests);
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const attrs = root.attributes ?? {};
|
|
69
|
+
byTest.push(
|
|
70
|
+
TestTrace({
|
|
71
|
+
test_id: attrs['test.id'] ?? '',
|
|
72
|
+
test_title: attrs['test.title'] ?? '',
|
|
73
|
+
test_file: attrs['test.file'] ?? '',
|
|
74
|
+
trace_id: traceId,
|
|
75
|
+
outcome: attrs['test.outcome'] ?? '',
|
|
76
|
+
requests,
|
|
77
|
+
}),
|
|
78
|
+
);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
if (setupRequests.length) {
|
|
82
|
+
byTest.push(
|
|
83
|
+
TestTrace({
|
|
84
|
+
test_id: SETUP_TEST_ID,
|
|
85
|
+
test_title: '',
|
|
86
|
+
test_file: '',
|
|
87
|
+
trace_id: '',
|
|
88
|
+
outcome: '',
|
|
89
|
+
requests: setupRequests,
|
|
90
|
+
}),
|
|
91
|
+
);
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
return Trace({ spans_total: spansTotal, by_test: byTest });
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function isDir(p) {
|
|
98
|
+
try {
|
|
99
|
+
return fs.statSync(p).isDirectory();
|
|
100
|
+
} catch {
|
|
101
|
+
return false;
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function readJsonl(filePath) {
|
|
106
|
+
const spans = [];
|
|
107
|
+
// NDJSON records are \n- or \r\n-delimited. We intentionally do NOT split on
|
|
108
|
+
// the exotic Unicode line separators that Python's str.splitlines() also
|
|
109
|
+
// breaks on (U+2028, U+2029, U+0085, ...): instrument.mjs writes those RAW
|
|
110
|
+
// inside JSON string values, so splitting on them would fragment a valid
|
|
111
|
+
// record and misattribute its request to __setup__.
|
|
112
|
+
for (const raw of fs.readFileSync(filePath, 'utf8').split(/\r\n|\n/)) {
|
|
113
|
+
const line = raw.trim();
|
|
114
|
+
if (!line) continue;
|
|
115
|
+
try {
|
|
116
|
+
spans.push(JSON.parse(line));
|
|
117
|
+
} catch {
|
|
118
|
+
continue; // malformed/torn line (e.g. a crashed worker's last write)
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
return spans;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function isTestRoot(span) {
|
|
125
|
+
return 'test.id' in (span.attributes ?? {});
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function isHttpSpan(span) {
|
|
129
|
+
const attrs = span.attributes ?? {};
|
|
130
|
+
return 'http.method' in attrs || 'http.request.method' in attrs;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Reconstruct the request URL across both OTel HTTP semantic conventions.
|
|
135
|
+
*
|
|
136
|
+
* The old convention put the whole thing in one attribute, `http.url`. The
|
|
137
|
+
* current one (stable since semconv 1.23) splits it: `url.full` on some
|
|
138
|
+
* instrumentations, and otherwise the pieces -- `url.scheme`, `server.address`,
|
|
139
|
+
* `server.port`, `url.path`. `@opentelemetry/auto-instrumentations-node`, which
|
|
140
|
+
* `otel_bootstrap/instrument.mjs` tells consumers to install, emits the SPLIT
|
|
141
|
+
* form and no `http.url` at all.
|
|
142
|
+
*
|
|
143
|
+
* Reading only `http.url` therefore produced `url: ""` on every request while
|
|
144
|
+
* the reader still reported a full span count and exit 0 -- the artifact looked
|
|
145
|
+
* written and was empty of the one field a coverage consumer needs. Port is
|
|
146
|
+
* omitted when it is the scheme default, so the URL matches what a spec or a
|
|
147
|
+
* HAR would say.
|
|
148
|
+
*/
|
|
149
|
+
function requestUrl(attrs) {
|
|
150
|
+
const direct = attrs['url.full'] ?? attrs['http.url'];
|
|
151
|
+
if (direct) return direct;
|
|
152
|
+
|
|
153
|
+
const scheme = attrs['url.scheme'] ?? attrs['http.scheme'];
|
|
154
|
+
const host =
|
|
155
|
+
attrs['server.address'] ?? attrs['net.peer.name'] ?? attrs['http.host'];
|
|
156
|
+
const path = attrs['url.path'] ?? attrs['http.target'] ?? '';
|
|
157
|
+
if (!scheme || !host) return typeof path === 'string' ? path : '';
|
|
158
|
+
|
|
159
|
+
const port = attrs['server.port'] ?? attrs['net.peer.port'];
|
|
160
|
+
const isDefaultPort =
|
|
161
|
+
port == null ||
|
|
162
|
+
(scheme === 'http' && +port === 80) ||
|
|
163
|
+
(scheme === 'https' && +port === 443);
|
|
164
|
+
const authority = isDefaultPort ? host : `${host}:${port}`;
|
|
165
|
+
const query = attrs['url.query'] ? `?${attrs['url.query']}` : '';
|
|
166
|
+
return `${scheme}://${authority}${path}${query}`;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
function toRequestSpan(span) {
|
|
170
|
+
const attrs = span.attributes ?? {};
|
|
171
|
+
const method = attrs['http.method'] || attrs['http.request.method'] || '';
|
|
172
|
+
return RequestSpan({
|
|
173
|
+
method,
|
|
174
|
+
url: requestUrl(attrs),
|
|
175
|
+
// `http.route` is a SERVER-side attribute and is never present on the client
|
|
176
|
+
// spans this reader consumes. Kept for producers that do supply it (a future
|
|
177
|
+
// server-side producer under the same v1 contract), but its absence here is
|
|
178
|
+
// normal, not a gap.
|
|
179
|
+
route: attrs['http.route'] ?? null,
|
|
180
|
+
// Renamed in the current convention. Old name kept as a fallback.
|
|
181
|
+
status:
|
|
182
|
+
attrs['http.response.status_code'] ?? attrs['http.status_code'] ?? null,
|
|
183
|
+
duration_ms: span.duration_ms ?? 0,
|
|
184
|
+
span_id: span.spanId ?? '',
|
|
185
|
+
started_at: span.startTime ?? '',
|
|
186
|
+
});
|
|
187
|
+
}
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: canary-katana
|
|
3
|
+
description:
|
|
4
|
+
Quarantines deleted and newly-skipped tests instead of letting them vanish.
|
|
5
|
+
Captures every removed or skipped test with provenance (who, when, which
|
|
6
|
+
commit, why) into an append-only ledger, and alarms in exactly one case — the
|
|
7
|
+
deletion removed the last coverage of a symbol critical-areas.json marks
|
|
8
|
+
high-risk. Silent by default, degrades to recording-only when critical-area
|
|
9
|
+
data is missing. Self-contained, deterministic, advisory by default.
|
|
10
|
+
cli: scripts/cli.mjs
|
|
11
|
+
requires: [node>=20]
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
# Canary Katana
|
|
15
|
+
|
|
16
|
+
Named for Tatsu Yamashiro's Soultaker — the blade that captures the soul of
|
|
17
|
+
whatever it cuts. A deleted test is coverage that leaves without a trace: the
|
|
18
|
+
suite still goes green, the gap is invisible, and nobody notices until the bug
|
|
19
|
+
it caught ships. Katana catches every test as it is removed or muted, records
|
|
20
|
+
who took it and why, and raises its voice only when the cut was the last thing
|
|
21
|
+
guarding a critical path.
|
|
22
|
+
|
|
23
|
+
Tier-0 deterministic analysis: no LLM, no network, no secrets, no dependency on
|
|
24
|
+
any other skill at runtime.
|
|
25
|
+
|
|
26
|
+
## What it captures
|
|
27
|
+
|
|
28
|
+
| Event | Detected from a diff |
|
|
29
|
+
| --------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
30
|
+
| `removed` | a `def test_*` / `async def test_*` (Python) or `describe`/`it`/`test('…')` (JS/TS) that left on a `-` line **and did not come back on the `+` side** |
|
|
31
|
+
| `skipped` | a `+`-side skip/mute marker: `@pytest.mark.skip` / `skipif` / `xfail`, or `it.skip` / `test.skip` / `describe.skip`, `it.only` / `test.only`, `xit` / `xdescribe` / `fit` |
|
|
32
|
+
|
|
33
|
+
A test flipped in place from `it('x')` to `it.skip('x')` is **one** event, not
|
|
34
|
+
two: the skip supersedes the removal so the ledger never double-counts a
|
|
35
|
+
mute-in-place as both a deletion and a skip.
|
|
36
|
+
|
|
37
|
+
The general form of that rule is the emphasis above: a `(file, title)` that
|
|
38
|
+
reappears on the `+` side was **modified, not removed**. Without it, any rewrite
|
|
39
|
+
of a declaration line recorded a deletion of a test that is still in the tree —
|
|
40
|
+
a prettier reflow of a long signature, or a `.skip` lifted in place, was enough
|
|
41
|
+
(#783). The ledger is append-only, so such a row is permanent and cannot be
|
|
42
|
+
corrected without the hand-edit the ledger exists to prevent; and a consumer
|
|
43
|
+
that attributes on the newest matching row would hand every test under a phantom
|
|
44
|
+
removal of a `describe` the wrong ticket. A **rename** is still a removal — the
|
|
45
|
+
old title's coverage really is gone.
|
|
46
|
+
|
|
47
|
+
## The one thing it alarms on
|
|
48
|
+
|
|
49
|
+
Most test deletions are legitimate — dead feature removal, genuine dedup — so
|
|
50
|
+
alarming on every one is nag fatigue within a week, and a gate people mute is
|
|
51
|
+
worse than no gate. Katana is **silent by default** and alarms only when a
|
|
52
|
+
removed test was the **last coverage** of a symbol listed in
|
|
53
|
+
`critical-areas.json` (produced by `canary-critical-areas`).
|
|
54
|
+
|
|
55
|
+
- **name-matched** — the removed test's name matches an area symbol and no other
|
|
56
|
+
test still covers it. Severity `critical` when the area's `risk_score` is high
|
|
57
|
+
(≥ 0.7), otherwise `high`.
|
|
58
|
+
- **heuristic** — only the test's _directory_ maps to the area (no name match).
|
|
59
|
+
Always severity `medium`, and flagged as lower fidelity.
|
|
60
|
+
|
|
61
|
+
### Degradation is loud and safe
|
|
62
|
+
|
|
63
|
+
When `critical-areas.json` is missing or malformed, katana records everything
|
|
64
|
+
but alarms on nothing, printing:
|
|
65
|
+
|
|
66
|
+
```text
|
|
67
|
+
critical-area data unavailable, recording only, not alarming
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Degradation never manufactures a failure — even under `--strict`, a degraded run
|
|
71
|
+
exits `0`.
|
|
72
|
+
|
|
73
|
+
## The ledger
|
|
74
|
+
|
|
75
|
+
Append-only JSON at `.canary/quarantine.json` (override with `--ledger`). Each
|
|
76
|
+
row carries full provenance so a vanished test leaves a trail:
|
|
77
|
+
|
|
78
|
+
```json
|
|
79
|
+
{
|
|
80
|
+
"schema_version": 2,
|
|
81
|
+
"entries": [
|
|
82
|
+
{
|
|
83
|
+
"test": "test_points_service_earns",
|
|
84
|
+
"file": "tests/test_points.py",
|
|
85
|
+
"kind": "removed",
|
|
86
|
+
"marker": "",
|
|
87
|
+
"commit": "…40 hex…",
|
|
88
|
+
"author": "Ada Lovelace",
|
|
89
|
+
"date": "2026-07-20T10:00:00+00:00",
|
|
90
|
+
"reason": "chore: drop points coverage",
|
|
91
|
+
"cause": "",
|
|
92
|
+
"issue": "",
|
|
93
|
+
"expiry": ""
|
|
94
|
+
}
|
|
95
|
+
]
|
|
96
|
+
}
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Re-running on the same change adds nothing (entries de-duplicate); a corrupt
|
|
100
|
+
ledger is a hard error, never silently overwritten.
|
|
101
|
+
|
|
102
|
+
### Schema v2: why a row is out, not just how it left (#771)
|
|
103
|
+
|
|
104
|
+
`cause`, `issue` and `expiry` are written by the quarantine producer, not by
|
|
105
|
+
katana. Katana records what it can observe from a diff — a test was removed or
|
|
106
|
+
skipped — and leaves `cause` empty, because "someone deleted this in commit
|
|
107
|
+
abc123" is provenance, not a judgement about why the test is out of the suite.
|
|
108
|
+
|
|
109
|
+
`reason` and `cause` are deliberately separate. `reason` is **derived** (the
|
|
110
|
+
commit subject). `cause` is **asserted** — one of `flaky`, `product-defect`,
|
|
111
|
+
`blocked-data`, `obsolete`. Collapsing them would dress an auto-derived string
|
|
112
|
+
up as a claim someone stands behind.
|
|
113
|
+
|
|
114
|
+
**One row per `(test, file)` may state a cause, and a caused row wins.** A row
|
|
115
|
+
with a cause supersedes a causeless row for the same pair, and a causeless row
|
|
116
|
+
is dropped when a caused row already exists. This is the one place the ledger is
|
|
117
|
+
not purely append-only, and it exists because the alternative is worse: katana
|
|
118
|
+
recording `{kind: 'skipped', cause: ''}` and a quarantine producer recording
|
|
119
|
+
`{kind: 'skipped', cause: 'product-defect', issue: …}` differ in every-field
|
|
120
|
+
identity, so **both** would persist — and a consumer that fails on an unlinked
|
|
121
|
+
quarantine (`canary-ci-ready` does) would fail on the causeless row while the
|
|
122
|
+
linked row sat beside it. The ledger would be contradicting itself about one
|
|
123
|
+
test.
|
|
124
|
+
|
|
125
|
+
History survives that rule: only rows differing in cause-bearing state collapse.
|
|
126
|
+
Two caused rows, or two causeless rows, keep the full-field identity and both
|
|
127
|
+
remain.
|
|
128
|
+
|
|
129
|
+
### Where `issue` comes from, and why the trailer keeps its own name
|
|
130
|
+
|
|
131
|
+
`issue` is the bug the quarantine is waiting on. It has two sources, and both
|
|
132
|
+
land in the same field:
|
|
133
|
+
|
|
134
|
+
- **A `Ticket:` commit trailer**, read by katana at capture time (`Bug:` and
|
|
135
|
+
`Tracked:` are accepted spellings). This is the low-friction path: the person
|
|
136
|
+
switching the test off names the bug in the commit that does it.
|
|
137
|
+
- **A quarantine producer**, writing a caused row directly.
|
|
138
|
+
|
|
139
|
+
v1 called this field `ticket` (#781). It is folded into `issue` here rather than
|
|
140
|
+
kept alongside, because two fields answering "what is this waiting on" is how a
|
|
141
|
+
consumer ends up reading the empty one — and the consumer is specific:
|
|
142
|
+
`canary-ci-ready` fails a quarantine with no **linked issue**, in either Jira or
|
|
143
|
+
GitHub. The schema now uses the consumer's word. A v1 row's `ticket` migrates
|
|
144
|
+
onto `issue` on load, so no recorded link is lost.
|
|
145
|
+
|
|
146
|
+
`Ticket:` survives as the name of the **trailer**, which is a mechanism rather
|
|
147
|
+
than a schema: it is what you type in a commit message, and renaming it would
|
|
148
|
+
invalidate the trailers already written without teaching anyone anything.
|
|
149
|
+
|
|
150
|
+
Empty is a real and important state, not a gap to paper over. A test switched
|
|
151
|
+
off with nothing to chase is the worst thing this ledger can record, and it can
|
|
152
|
+
only be seen if it is recorded honestly.
|
|
153
|
+
|
|
154
|
+
A v1 file is normalized on load, so every row comes back carrying the v2 fields
|
|
155
|
+
(empty where unrecorded). That is what makes writing `schema_version: 2` honest
|
|
156
|
+
— the version claims these rows have these fields, and after load they do.
|
|
157
|
+
Stamping the version over un-migrated rows would make it a promise the file does
|
|
158
|
+
not keep.
|
|
159
|
+
|
|
160
|
+
## Invocation
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
# Diff the current branch against its merge-base, record, advise (exit 0):
|
|
164
|
+
canary skills run canary-katana
|
|
165
|
+
|
|
166
|
+
# Feed an explicit diff and a critical-areas map:
|
|
167
|
+
canary skills run canary-katana -- \
|
|
168
|
+
--diff-file changes.diff --critical-areas .canary/critical-areas.json
|
|
169
|
+
|
|
170
|
+
# Machine-readable:
|
|
171
|
+
canary skills run canary-katana -- --json
|
|
172
|
+
|
|
173
|
+
# Fail the step only when a critical path loses its last coverage:
|
|
174
|
+
canary skills run canary-katana -- --strict
|
|
175
|
+
|
|
176
|
+
# Usage and options (exits 0, and writes nothing to the ledger):
|
|
177
|
+
canary skills run canary-katana -- --help
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
Value flags (`--repo`, `--diff-file`, `--ledger`, `--critical-areas`) accept
|
|
181
|
+
both `--repo <path>` and `--repo=<path>`, matching `canary-instrument` and
|
|
182
|
+
`canary-fail-fast`.
|
|
183
|
+
|
|
184
|
+
An unknown flag is rejected with `unrecognized arguments: <flag>` and exit 2,
|
|
185
|
+
and a value flag left without a usable value is
|
|
186
|
+
`argument <flag>: expected one argument` (exit 2). That covers all three ways
|
|
187
|
+
the value can go missing: the flag is last, the next token is another flag, or
|
|
188
|
+
the value is empty — in either the `--repo=` spelling or, the one shells
|
|
189
|
+
actually produce, `--repo "$UNSET_VAR"`. Empty is rejected rather than accepted
|
|
190
|
+
because `--repo ''` would resolve the ledger to `path.join('', '.canary', ...)`
|
|
191
|
+
and write it into the process CWD instead of the target repo.
|
|
192
|
+
|
|
193
|
+
All of these are decided before any diff is read or ledger entry is appended, so
|
|
194
|
+
a usage request or a typo never mutates the working tree.
|
|
195
|
+
|
|
196
|
+
`--json` shape:
|
|
197
|
+
|
|
198
|
+
```json
|
|
199
|
+
{
|
|
200
|
+
"schema_version": 2,
|
|
201
|
+
"captured": [
|
|
202
|
+
{ "name": "…", "file": "…", "kind": "removed", "line": 3, "marker": "" }
|
|
203
|
+
],
|
|
204
|
+
"findings": [
|
|
205
|
+
{
|
|
206
|
+
"kind": "last-coverage-removed",
|
|
207
|
+
"test": "…",
|
|
208
|
+
"file": "…",
|
|
209
|
+
"area": "src/loyalty/points.service.ts",
|
|
210
|
+
"fidelity": "name-matched",
|
|
211
|
+
"severity": "critical",
|
|
212
|
+
"evidence": "…"
|
|
213
|
+
}
|
|
214
|
+
],
|
|
215
|
+
"ledger": ".canary/quarantine.json"
|
|
216
|
+
}
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
A degraded run adds a top-level `"degraded_notice"` and an empty `findings`.
|
|
220
|
+
|
|
221
|
+
## CI wiring (GitHub Actions)
|
|
222
|
+
|
|
223
|
+
Advisory first, then promote to blocking once the ledger is trusted — the same
|
|
224
|
+
path every canary gate takes.
|
|
225
|
+
|
|
226
|
+
```yaml
|
|
227
|
+
- name: Quarantine deleted tests (advisory)
|
|
228
|
+
run:
|
|
229
|
+
canary skills run canary-katana -- --critical-areas
|
|
230
|
+
.canary/critical-areas.json
|
|
231
|
+
# Once trusted, add --strict so a last-coverage loss fails the PR:
|
|
232
|
+
# run: canary skills run canary-katana -- --critical-areas .canary/critical-areas.json --strict
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
## Fidelity limits (regex/diff-lite, on purpose)
|
|
236
|
+
|
|
237
|
+
- **Line-scoped diff parsing.** A declaration split across lines can be missed;
|
|
238
|
+
katana errs toward recording the clear cases.
|
|
239
|
+
- **Name/dir coverage is heuristic.** "Last coverage" is inferred from test
|
|
240
|
+
names and directory layout, not a real coverage run — treat `heuristic`
|
|
241
|
+
findings as prompts to look, not verdicts.
|
|
242
|
+
- **Provenance needs git.** Fed a `--diff-file` outside a git repo, author and
|
|
243
|
+
commit are recorded as `unknown` / empty rather than guessed.
|