canary-test-cli 7.1.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +327 -0
- package/agents/skills/canary:generate.md +49 -0
- package/agents/skills/canary:init.md +37 -0
- package/agents/skills/canary:migrate.md +66 -0
- package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
- package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
- package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
- package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
- package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
- package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
- package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
- package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
- package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
- package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
- package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
- package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
- package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
- package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
- package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
- package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
- package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
- package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
- package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
- package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
- package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
- package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
- package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
- package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
- package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
- package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
- package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
- package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
- package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
- package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
- package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
- package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
- package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
- package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
- package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
- package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
- package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
- package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
- package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
- package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
- package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
- package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
- package/agents/skills/lib/parse-args.mjs +275 -0
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +49 -72
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +27 -19
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/skill-dispatch.js +115 -0
- package/dist/engine/core/skill-examples.js +103 -3
- package/dist/engine/core/skill-registry.js +59 -4
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/test-files.js +77 -0
- package/dist/engine/core/vacuity-scanner.js +330 -15
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/analysis-emit.js +7 -2
- package/dist/engine/guardian/cli.js +277 -249
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +354 -223
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +171 -51
- package/dist/engine/workflow-cli.js +85 -65
- package/dist/reporters/testtracker.d.ts +1 -1
- package/dist/reporters/testtracker.js +1 -1
- package/package.json +3 -2
|
@@ -1,368 +1,147 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Guardian
|
|
3
|
-
* (#490).
|
|
2
|
+
* Guardian adjudication without reactions (ADR 0025, #938).
|
|
4
3
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* gate. Reviewers already give the lowest-friction feedback available: a 👍
|
|
9
|
-
* (true positive) or 👎 (false positive) reaction on the guardian's sticky
|
|
10
|
-
* comment. This module reads those reactions back off the comment the guardian
|
|
11
|
-
* already upserts by marker, and persists a per-PR adjudication record to the
|
|
12
|
-
* existing `.harness/analyses/` channel (no new store — see
|
|
13
|
-
* {@link module:./analysis-emit}).
|
|
4
|
+
* The soft→hard promotion rests on `precision = TP / (TP + FP)`. Reaction
|
|
5
|
+
* collection (#490) read an input nobody produces (0 reactions on 290 stickies),
|
|
6
|
+
* so precision is now DERIVED on demand from signals the team already leaves:
|
|
14
7
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
8
|
+
* - **true positive**: a coverage-verified finding on the first sticky revision
|
|
9
|
+
* is gone from the last one while its file is still in the merged diff (a
|
|
10
|
+
* later commit covered it).
|
|
11
|
+
* - **intentional**: the merged diff adds `canary:allow-untested <reason>`.
|
|
12
|
+
* - **false positive**: the reason starts with `fp:`.
|
|
13
|
+
* - **ambiguous**: it disappeared with no coverage evidence (heuristic/graph
|
|
14
|
+
* tier, or the file left the diff).
|
|
15
|
+
* - **unresolved**: still active at merge. Not a false positive.
|
|
20
16
|
*
|
|
21
|
-
*
|
|
22
|
-
* findings
|
|
23
|
-
*
|
|
24
|
-
* reviewers react to neither — the sample is small and self-selected, and every
|
|
25
|
-
* rendered surface states the sample size rather than presenting the number as
|
|
26
|
-
* ground truth.
|
|
17
|
+
* Nothing is stored. Precision is `null` below {@link PRECISION_FLOOR}
|
|
18
|
+
* adjudicated findings (TP + FP), and every excluded PR is counted in a
|
|
19
|
+
* disclosed denominator rather than dropped (#508, ADR 0009).
|
|
27
20
|
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
* uses {@link FakeReactionsClient}.
|
|
21
|
+
* PURE: no network, no filesystem. GitHub access lives in
|
|
22
|
+
* `adjudication-github.ts` behind an injected seam.
|
|
31
23
|
*/
|
|
32
|
-
import {
|
|
33
|
-
import {
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
const
|
|
24
|
+
import { STICKY_MARKER } from './pr-comment.js';
|
|
25
|
+
import { suppressionReason } from './pr-check.js';
|
|
26
|
+
/** Minimum TP + FP before precision is reported as a number. */
|
|
27
|
+
export const PRECISION_FLOOR = 30;
|
|
28
|
+
// A finding row: `| <sev> | [`path`](url) or `path` | what | fidelity |`. The
|
|
29
|
+
// header and separator never open their second cell with a backtick.
|
|
30
|
+
const FINDING_ROW_RE = /^\|[^|]*\|\s*\[?`([^`]+)`.*\|\s*([a-z-]+)\s*\|\s*$/;
|
|
31
|
+
const TABLE_HEADER = '| Sev | File |';
|
|
32
|
+
const HEADING = 'Canary PR Guardian';
|
|
39
33
|
/**
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
* The filenames are not provably distinct, though: a branch named
|
|
45
|
-
* `adjudication/pr-42` sanitizes through `analysisFilename` to exactly
|
|
46
|
-
* `canary-pr-guardian-adjudication-pr-42.json`, colliding with this prefix.
|
|
47
|
-
* What actually keeps the precision summary honest is the `source` field —
|
|
48
|
-
* `loadAdjudicationRecords` requires `source === ADJUDICATION_SOURCE` plus
|
|
49
|
-
* numeric `tp`/`fp`, so a findings record landing on that name is skipped, not
|
|
50
|
-
* mis-tallied. Read the field, never the filename.
|
|
34
|
+
* `fp:` (any case, optional space before the colon) marks a false positive;
|
|
35
|
+
* any other reason is intentional. A bare `fp:` still counts: the prefix is
|
|
36
|
+
* the reviewer's verdict, the text after it is only the explanation.
|
|
51
37
|
*/
|
|
52
|
-
export
|
|
53
|
-
|
|
54
|
-
const THUMBS_UP = '+1';
|
|
55
|
-
const THUMBS_DOWN = '-1';
|
|
56
|
-
// Loud notices carry an em-dash as output data; escaped per the ASCII-source rule.
|
|
57
|
-
const EM_DASH = '\u{2014}';
|
|
58
|
-
/** In-memory {@link ReactionsClient} for unit tests — no network. */
|
|
59
|
-
export class FakeReactionsClient {
|
|
60
|
-
comments;
|
|
61
|
-
reactionsByComment;
|
|
62
|
-
constructor(init = {}) {
|
|
63
|
-
this.comments = init.comments ?? [];
|
|
64
|
-
this.reactionsByComment = new Map(Object.entries(init.reactions ?? {}).map(([id, rows]) => [
|
|
65
|
-
Number(id),
|
|
66
|
-
rows,
|
|
67
|
-
]));
|
|
68
|
-
}
|
|
69
|
-
async listComments() {
|
|
70
|
-
return this.comments;
|
|
71
|
-
}
|
|
72
|
-
async listReactions(commentId) {
|
|
73
|
-
return this.reactionsByComment.get(commentId) ?? [];
|
|
74
|
-
}
|
|
38
|
+
export function suppressionKind(reason) {
|
|
39
|
+
return /^fp\s*:/i.test(reason.trim()) ? 'false-positive' : 'intentional';
|
|
75
40
|
}
|
|
76
|
-
/**
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
prNumber;
|
|
86
|
-
static API = 'https://api.github.com';
|
|
87
|
-
read;
|
|
88
|
-
constructor(repo, prNumber, token, read) {
|
|
89
|
-
this.repo = repo;
|
|
90
|
-
this.prNumber = prNumber;
|
|
91
|
-
this.read =
|
|
92
|
-
read ??
|
|
93
|
-
restPageReader({
|
|
94
|
-
Authorization: `Bearer ${token}`,
|
|
95
|
-
Accept: 'application/vnd.github+json',
|
|
96
|
-
'X-GitHub-Api-Version': '2022-11-28',
|
|
97
|
-
'User-Agent': 'canary-pr-guardian',
|
|
98
|
-
}, (status, url) => new Error(`GitHub API ${status}: ${url}`));
|
|
99
|
-
}
|
|
100
|
-
async listComments() {
|
|
101
|
-
const url = `${RestReactionsClient.API}/repos/${this.repo}/issues/${this.prNumber}/comments`;
|
|
102
|
-
return (await readAllPages(url, this.read));
|
|
103
|
-
}
|
|
104
|
-
async listReactions(commentId) {
|
|
105
|
-
const url = `${RestReactionsClient.API}/repos/${this.repo}/issues/comments/${commentId}/reactions`;
|
|
106
|
-
const result = await readAllPages(url, this.read);
|
|
107
|
-
const rows = [];
|
|
108
|
-
for (const raw of result) {
|
|
109
|
-
if (typeof raw !== 'object' || raw === null)
|
|
41
|
+
/** Suppressions ADDED by the merged diff, per path (`fp:` wins on a tie). */
|
|
42
|
+
export function suppressionsByPath(files) {
|
|
43
|
+
const out = new Map();
|
|
44
|
+
for (const file of files) {
|
|
45
|
+
for (const line of (file.patch ?? '').split('\n')) {
|
|
46
|
+
if (!line.startsWith('+') || line.startsWith('+++'))
|
|
47
|
+
continue;
|
|
48
|
+
const reason = suppressionReason(line.slice(1));
|
|
49
|
+
if (reason === null || out.get(file.filename) === 'false-positive')
|
|
110
50
|
continue;
|
|
111
|
-
|
|
112
|
-
const content = typeof rec.content === 'string' ? rec.content : '';
|
|
113
|
-
const user = typeof rec.user?.login === 'string' ? rec.user.login : 'unknown';
|
|
114
|
-
if (content)
|
|
115
|
-
rows.push({ user, content });
|
|
51
|
+
out.set(file.filename, suppressionKind(reason));
|
|
116
52
|
}
|
|
117
|
-
return rows;
|
|
118
|
-
}
|
|
119
|
-
}
|
|
120
|
-
/**
|
|
121
|
-
* Tally verdict reactions: one vote per user, bots excluded (PURE).
|
|
122
|
-
*
|
|
123
|
-
* - Only `+1`/`-1` carry a verdict; every other content is ignored.
|
|
124
|
-
* - Logins ending in `[bot]` are excluded so the guardian's own automation (or
|
|
125
|
-
* any other bot) can never inflate its own precision.
|
|
126
|
-
* - A user who reacted both 👍 and 👎 is contradictory: counted as `ambiguous`
|
|
127
|
-
* and excluded from both TP and FP rather than guessed at.
|
|
128
|
-
*/
|
|
129
|
-
export function tallyAdjudications(reactions) {
|
|
130
|
-
const up = new Set();
|
|
131
|
-
const down = new Set();
|
|
132
|
-
for (const reaction of reactions) {
|
|
133
|
-
if (reaction.user.endsWith('[bot]'))
|
|
134
|
-
continue;
|
|
135
|
-
if (reaction.content === THUMBS_UP)
|
|
136
|
-
up.add(reaction.user);
|
|
137
|
-
else if (reaction.content === THUMBS_DOWN)
|
|
138
|
-
down.add(reaction.user);
|
|
139
|
-
}
|
|
140
|
-
let ambiguous = 0;
|
|
141
|
-
for (const user of up) {
|
|
142
|
-
if (down.has(user))
|
|
143
|
-
ambiguous += 1;
|
|
144
53
|
}
|
|
145
|
-
return
|
|
146
|
-
tp: up.size - ambiguous,
|
|
147
|
-
fp: down.size - ambiguous,
|
|
148
|
-
ambiguous,
|
|
149
|
-
};
|
|
54
|
+
return out;
|
|
150
55
|
}
|
|
151
|
-
// A findings-table row in the sticky comment: `| <icon> <sev> | `path`... |`.
|
|
152
|
-
// The header row's second cell is ` File ` and the separator's is ` --- `,
|
|
153
|
-
// neither of which starts with a backtick, so anchoring on the second cell's
|
|
154
|
-
// leading backtick selects exactly the finding rows. Paths never contain `|`
|
|
155
|
-
// or backticks (see `fileLabel` in pr-check.ts), so the naive anchor is safe.
|
|
156
|
-
//
|
|
157
|
-
// The optional `[` accommodates the permalinked cell — `fileLabel` wraps the
|
|
158
|
-
// path as `[`path`](<blob url>)` whenever a blob base is resolvable, which is
|
|
159
|
-
// the normal case in CI. Without it this regex matched nothing on every posted
|
|
160
|
-
// comment and `activeFindingPaths` returned `[]`, zeroing the precision
|
|
161
|
-
// denominator silently instead of failing (#490, #508).
|
|
162
|
-
const FINDING_ROW_RE = /^\|[^|]*\|\s*\[?`([^`]+)`/;
|
|
163
56
|
/**
|
|
164
|
-
*
|
|
165
|
-
*
|
|
166
|
-
*
|
|
167
|
-
* current finding set, so a reaction is attributed to what the reviewer saw.
|
|
168
|
-
* Returns `[]` for a no-gaps body (no table).
|
|
57
|
+
* Parse the finding rows of one sticky revision. Returns `null` for a body
|
|
58
|
+
* that is not a guardian sticky, or whose findings table yields no row: an
|
|
59
|
+
* unparseable body is unknown, never zero findings.
|
|
169
60
|
*/
|
|
170
|
-
export function
|
|
171
|
-
const
|
|
172
|
-
|
|
61
|
+
export function parseStickyFindings(body) {
|
|
62
|
+
const trimmed = body.trimStart();
|
|
63
|
+
if (!trimmed.startsWith(STICKY_MARKER) || !body.includes(HEADING))
|
|
64
|
+
return null;
|
|
65
|
+
const rows = [];
|
|
66
|
+
for (const line of body.split(/\r\n|\r|\n/)) {
|
|
173
67
|
const match = FINDING_ROW_RE.exec(line);
|
|
174
68
|
if (match)
|
|
175
|
-
|
|
69
|
+
rows.push({ path: match[1], fidelity: match[2] });
|
|
176
70
|
}
|
|
177
|
-
|
|
71
|
+
if (rows.length === 0 && body.includes(TABLE_HEADER))
|
|
72
|
+
return null;
|
|
73
|
+
return rows;
|
|
178
74
|
}
|
|
179
|
-
/**
|
|
180
|
-
function
|
|
181
|
-
|
|
75
|
+
/** Classify one first-revision finding against the merge-time evidence. */
|
|
76
|
+
export function classifyFinding(finding, ctx) {
|
|
77
|
+
const suppression = ctx.suppressions.get(finding.path);
|
|
78
|
+
if (suppression)
|
|
79
|
+
return suppression;
|
|
80
|
+
if (ctx.last.some((f) => f.path === finding.path))
|
|
81
|
+
return 'unresolved';
|
|
82
|
+
const covered = finding.fidelity === 'coverage-verified' &&
|
|
83
|
+
ctx.mergedPaths.has(finding.path);
|
|
84
|
+
return covered ? 'true-positive' : 'ambiguous';
|
|
182
85
|
}
|
|
183
|
-
|
|
184
|
-
export function buildAdjudicationRecord(init) {
|
|
185
|
-
const findingPaths = activeFindingPaths(init.commentBody);
|
|
186
|
-
const single = findingPaths.length === 1;
|
|
86
|
+
function emptyCounts() {
|
|
187
87
|
return {
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
granularity: single ? 'finding' : 'run',
|
|
194
|
-
attributedPath: single ? findingPaths[0] : null,
|
|
195
|
-
findingPaths,
|
|
196
|
-
tp: init.tally.tp,
|
|
197
|
-
fp: init.tally.fp,
|
|
198
|
-
ambiguous: init.tally.ambiguous,
|
|
199
|
-
collectedAt: init.collectedAt ?? isoUtcNow(),
|
|
88
|
+
'true-positive': 0,
|
|
89
|
+
'false-positive': 0,
|
|
90
|
+
intentional: 0,
|
|
91
|
+
ambiguous: 0,
|
|
92
|
+
unresolved: 0,
|
|
200
93
|
};
|
|
201
94
|
}
|
|
202
|
-
/** `
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
return
|
|
213
|
-
}
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
* for different PRs never collide — the store is append-only across PRs.
|
|
222
|
-
*
|
|
223
|
-
* Never throws for an expected shape: a missing comment, zero reactions, or an
|
|
224
|
-
* unavailable channel each return a distinct non-`collected` result so the
|
|
225
|
-
* caller can report honestly instead of crashing the gate.
|
|
226
|
-
*/
|
|
227
|
-
export async function collectAdjudications(client, args) {
|
|
228
|
-
const sticky = findSticky(await client.listComments(), args.marker ?? STICKY_MARKER);
|
|
229
|
-
if (sticky === null) {
|
|
230
|
-
return { action: 'no-comment', path: null, record: null, notice: null };
|
|
231
|
-
}
|
|
232
|
-
const tally = tallyAdjudications(await client.listReactions(sticky.id));
|
|
233
|
-
if (tally.tp + tally.fp + tally.ambiguous === 0) {
|
|
234
|
-
return { action: 'no-reactions', path: null, record: null, notice: null };
|
|
235
|
-
}
|
|
236
|
-
const record = buildAdjudicationRecord({
|
|
237
|
-
repo: args.repo,
|
|
238
|
-
prNumber: args.prNumber,
|
|
239
|
-
commentId: sticky.id,
|
|
240
|
-
commentBody: sticky.body,
|
|
241
|
-
tally,
|
|
242
|
-
collectedAt: args.collectedAt,
|
|
243
|
-
});
|
|
244
|
-
if (!isChannelAvailable(args.analysesDir)) {
|
|
245
|
-
return {
|
|
246
|
-
action: 'unavailable',
|
|
247
|
-
path: null,
|
|
248
|
-
record,
|
|
249
|
-
notice: 'guardian: harness analyses channel unavailable (.harness/ absent) ' +
|
|
250
|
-
`${EM_DASH} adjudication not persisted`,
|
|
251
|
-
};
|
|
252
|
-
}
|
|
253
|
-
const target = join(args.analysesDir, adjudicationFilename(args.prNumber));
|
|
254
|
-
try {
|
|
255
|
-
mkdirSync(args.analysesDir, { recursive: true });
|
|
256
|
-
// Atomic write (same-dir temp + rename), matching analysis-emit: a torn
|
|
257
|
-
// record would poison every later precision summary.
|
|
258
|
-
const tmp = join(args.analysesDir, `.tmp-${randomBytes(8).toString('hex')}.json`);
|
|
259
|
-
writeFileSync(tmp, JSON.stringify(record, null, 2), 'utf-8');
|
|
260
|
-
try {
|
|
261
|
-
renameSync(tmp, target);
|
|
262
|
-
}
|
|
263
|
-
catch (err) {
|
|
264
|
-
try {
|
|
265
|
-
unlinkSync(tmp);
|
|
266
|
-
}
|
|
267
|
-
catch {
|
|
268
|
-
// best-effort cleanup
|
|
269
|
-
}
|
|
270
|
-
throw err;
|
|
271
|
-
}
|
|
272
|
-
}
|
|
273
|
-
catch (exc) {
|
|
274
|
-
const message = exc instanceof Error ? exc.message : String(exc);
|
|
275
|
-
return {
|
|
276
|
-
action: 'unavailable',
|
|
277
|
-
path: null,
|
|
278
|
-
record,
|
|
279
|
-
notice: `guardian: adjudication write failed (${message}) ${EM_DASH} ` +
|
|
280
|
-
'adjudication not persisted',
|
|
281
|
-
};
|
|
282
|
-
}
|
|
283
|
-
return { action: 'collected', path: target, record, notice: null };
|
|
284
|
-
}
|
|
285
|
-
/**
|
|
286
|
-
* Load every adjudication record under `analysesDir` (best-effort).
|
|
287
|
-
*
|
|
288
|
-
* Reads only `canary-pr-guardian-adjudication-*.json`; pr-check findings
|
|
289
|
-
* records and harness's own records are never touched. A malformed or
|
|
290
|
-
* wrong-`source` file is skipped, never fatal — one corrupt record must not
|
|
291
|
-
* take down the precision report.
|
|
292
|
-
*/
|
|
293
|
-
export function loadAdjudicationRecords(analysesDir) {
|
|
294
|
-
let names;
|
|
295
|
-
try {
|
|
296
|
-
names = readdirSync(analysesDir);
|
|
297
|
-
}
|
|
298
|
-
catch {
|
|
299
|
-
return [];
|
|
300
|
-
}
|
|
301
|
-
const records = [];
|
|
302
|
-
for (const name of names.sort()) {
|
|
303
|
-
if (!name.startsWith(`${ADJUDICATION_SOURCE}-`) || !name.endsWith('.json'))
|
|
304
|
-
continue;
|
|
305
|
-
try {
|
|
306
|
-
const raw = JSON.parse(readFileSync(join(analysesDir, name), 'utf-8'));
|
|
307
|
-
if (raw !== null &&
|
|
308
|
-
typeof raw === 'object' &&
|
|
309
|
-
raw.source === ADJUDICATION_SOURCE &&
|
|
310
|
-
typeof raw.tp === 'number' &&
|
|
311
|
-
typeof raw.fp === 'number') {
|
|
312
|
-
records.push(raw);
|
|
313
|
-
}
|
|
314
|
-
}
|
|
315
|
-
catch {
|
|
316
|
-
// skip malformed record
|
|
317
|
-
}
|
|
318
|
-
}
|
|
319
|
-
return records;
|
|
95
|
+
/** Tally one PR into `report`, or count it as excluded. */
|
|
96
|
+
function tallyPr(pr, report) {
|
|
97
|
+
if (pr.revisions === null || pr.revisions.length === 0) {
|
|
98
|
+
report.prs.noHistory += 1;
|
|
99
|
+
return;
|
|
100
|
+
}
|
|
101
|
+
const first = parseStickyFindings(pr.revisions[0]);
|
|
102
|
+
const last = parseStickyFindings(pr.revisions.at(-1));
|
|
103
|
+
if (first === null || last === null) {
|
|
104
|
+
report.prs.unparseable += 1;
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
const ctx = {
|
|
108
|
+
last,
|
|
109
|
+
suppressions: suppressionsByPath(pr.files),
|
|
110
|
+
mergedPaths: new Set(pr.files.map((f) => f.filename)),
|
|
111
|
+
};
|
|
112
|
+
for (const finding of first)
|
|
113
|
+
report.counts[classifyFinding(finding, ctx)]++;
|
|
320
114
|
}
|
|
321
|
-
/**
|
|
322
|
-
export function
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
tp += record.tp;
|
|
329
|
-
fp += record.fp;
|
|
330
|
-
ambiguous += record.ambiguous ?? 0;
|
|
331
|
-
if (record.tp + record.fp > 0)
|
|
332
|
-
prCount += 1;
|
|
333
|
-
}
|
|
334
|
-
const adjudicated = tp + fp;
|
|
335
|
-
return {
|
|
336
|
-
adjudicated,
|
|
337
|
-
tp,
|
|
338
|
-
fp,
|
|
339
|
-
ambiguous,
|
|
340
|
-
prCount,
|
|
341
|
-
precision: adjudicated === 0 ? null : tp / adjudicated,
|
|
115
|
+
/** Derive the report over the PRs that carried a sticky (PURE). */
|
|
116
|
+
export function deriveReport(prs, scanned) {
|
|
117
|
+
const report = {
|
|
118
|
+
counts: emptyCounts(),
|
|
119
|
+
adjudicated: 0,
|
|
120
|
+
precision: null,
|
|
121
|
+
prs: { scanned, withSticky: prs.length, noHistory: 0, unparseable: 0 },
|
|
342
122
|
};
|
|
123
|
+
for (const pr of prs)
|
|
124
|
+
tallyPr(pr, report);
|
|
125
|
+
const { 'true-positive': tp, 'false-positive': fp } = report.counts;
|
|
126
|
+
report.adjudicated = tp + fp;
|
|
127
|
+
if (report.adjudicated >= PRECISION_FLOOR)
|
|
128
|
+
report.precision = tp / (tp + fp);
|
|
129
|
+
return report;
|
|
343
130
|
}
|
|
344
|
-
/**
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
*
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
const ambiguousNote = summary.ambiguous > 0
|
|
360
|
-
? ` ${summary.ambiguous} contradictory verdict(s) excluded.`
|
|
361
|
-
: '';
|
|
362
|
-
return (`guardian precision: ${pct}% (${summary.tp} true / ${summary.fp} false ` +
|
|
363
|
-
`positive${summary.adjudicated === 1 ? '' : 's'}, n=${summary.adjudicated} ` +
|
|
364
|
-
`across ${summary.prCount} PR(s)).${ambiguousNote} Sample is ` +
|
|
365
|
-
`self-selected (reviewers who chose to react) ${EM_DASH} a signal, not ` +
|
|
366
|
-
`ground truth.`);
|
|
131
|
+
/** Render the report; the number never appears without its sample size. */
|
|
132
|
+
export function renderReport(report) {
|
|
133
|
+
const c = report.counts;
|
|
134
|
+
const sample = `n=${report.adjudicated}: ${c['true-positive']} TP / ${c['false-positive']} FP`;
|
|
135
|
+
const value = report.precision === null
|
|
136
|
+
? `unknown (N < ${PRECISION_FLOOR}) (${sample})`
|
|
137
|
+
: `${(report.precision * 100).toFixed(1).replace(/\.0$/, '')}% (${sample})`;
|
|
138
|
+
const p = report.prs;
|
|
139
|
+
return [
|
|
140
|
+
`guardian precision: ${value}`,
|
|
141
|
+
`excluded from precision: ${c.intentional} intentional, ${c.ambiguous} ambiguous, ${c.unresolved} unresolved at merge`,
|
|
142
|
+
`merged PRs: ${p.scanned} scanned, ${p.withSticky} with a guardian sticky, ` +
|
|
143
|
+
`${p.noHistory} excluded: no edit history, ${p.unparseable} excluded: unparseable sticky`,
|
|
144
|
+
'`fp:` suppressions are a convention: under-use biases precision upward.',
|
|
145
|
+
].join('\n');
|
|
367
146
|
}
|
|
368
147
|
//# sourceMappingURL=adjudication.js.map
|
|
@@ -36,13 +36,17 @@ import { mkdirSync, renameSync, statSync, unlinkSync, writeFileSync, } from 'nod
|
|
|
36
36
|
import { dirname, join } from 'node:path';
|
|
37
37
|
import { ensureAscii } from '../util/ensure-ascii.js';
|
|
38
38
|
import { coverageDegradedNotice, coverageStatus, } from './coverage.js';
|
|
39
|
-
import { combineNotices, renderFindings } from './pr-check.js';
|
|
39
|
+
import { combineNotices, renderFindings, } from './pr-check.js';
|
|
40
40
|
// 1.1 adds the additive `coverage` block (#554); readers of 1.0 are unaffected.
|
|
41
41
|
// 1.2 adds the additive `skipped` list (#582). Additive again, and bumped again
|
|
42
42
|
// for the reason recorded in #572: a reader that pins a version must be able to
|
|
43
43
|
// tell which fields it can rely on being present, and silence about a new field
|
|
44
44
|
// is indistinguishable from the field being absent for a real reason.
|
|
45
|
-
|
|
45
|
+
// 1.3 adds the additive `provenance` block (#761). Same additive rule. This is
|
|
46
|
+
// the field an archived record needs most: when a run is questioned WEEKS later
|
|
47
|
+
// from its uploaded artifact, `checked: 43` is only interpretable next to the
|
|
48
|
+
// diff endpoints that produced it.
|
|
49
|
+
export const SCHEMA_VERSION = '1.3';
|
|
46
50
|
const ANALYSIS_SOURCE = 'canary-pr-guardian';
|
|
47
51
|
const REF_SAFE = /[^A-Za-z0-9._-]/g;
|
|
48
52
|
const REF_MAX = 100; // cap the sanitized ref so a long branch never hits ENAMETOOLONG
|
|
@@ -111,6 +115,7 @@ export function buildAnalysisRecord(findings, args) {
|
|
|
111
115
|
? null
|
|
112
116
|
: { status: coverageStatus(coverage), ...coverage },
|
|
113
117
|
skipped: args.skipped ?? [],
|
|
118
|
+
provenance: args.provenance ?? null,
|
|
114
119
|
summary: {
|
|
115
120
|
total: findings.length,
|
|
116
121
|
unaddressed: active.length,
|