canary-test-cli 7.2.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +23 -4
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +23 -16
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +3 -1
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +20 -3
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +1 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +15 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/lib/parse-args.mjs +200 -139
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +46 -7
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +13 -18
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/vacuity-scanner.js +151 -6
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/cli.js +180 -264
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +262 -430
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +48 -32
- package/dist/engine/workflow-cli.js +85 -65
- package/package.json +1 -1
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The three shipped probes (spec Phase 2).
|
|
3
|
+
*
|
|
4
|
+
* Each answers one question about one artifact type: has this thing executed
|
|
5
|
+
* since the fix merged? None of them asserts the fix is correct, and none of
|
|
6
|
+
* them reaches the network directly -- `ExerciseContext.runs` is the single
|
|
7
|
+
* seam, and the `gh` implementation behind it lives in `gh-history.ts`.
|
|
8
|
+
*
|
|
9
|
+
* **A probe that cannot decide throws or abstains; it never guesses.** The
|
|
10
|
+
* registry turns a throw into `abstain`, so a probe is free to let a broken
|
|
11
|
+
* `RunHistoryPort` surface rather than inventing a verdict over history nobody
|
|
12
|
+
* managed to read. Reporting "not exercised" on an unread history would be a
|
|
13
|
+
* claim with no evidence behind it -- the exact defect batwoman exists to find.
|
|
14
|
+
*/
|
|
15
|
+
import { readFileSync, readdirSync } from 'node:fs';
|
|
16
|
+
import { basename, join } from 'node:path';
|
|
17
|
+
import { day, judgeRuns } from './run-window.js';
|
|
18
|
+
import { explain, } from './verdict.js';
|
|
19
|
+
const WORKFLOW_DIR = '.github/workflows';
|
|
20
|
+
const WORKFLOW_RE = /^\.github\/workflows\/[^/]+\.ya?ml$/;
|
|
21
|
+
/** Top-level `scripts/*.mjs` only: nested helpers are not workflow entry points. */
|
|
22
|
+
const SCRIPT_RE = /^scripts\/[^/]+\.mjs$/;
|
|
23
|
+
/**
|
|
24
|
+
* `.github/workflows/*.yml` -- decided by run history, explained by `on:`.
|
|
25
|
+
*/
|
|
26
|
+
export function workflowProbe() {
|
|
27
|
+
return {
|
|
28
|
+
id: 'workflow',
|
|
29
|
+
artifact: 'workflow',
|
|
30
|
+
matches: (file) => WORKFLOW_RE.test(file),
|
|
31
|
+
async probe(file, ctx) {
|
|
32
|
+
return judgeRuns(file, await ctx.runs.runsForWorkflow(file), ctx);
|
|
33
|
+
},
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
/** Workflow files that mention `name`, as repo-relative paths. */
|
|
37
|
+
function workflowsReferencing(root, name) {
|
|
38
|
+
let entries;
|
|
39
|
+
try {
|
|
40
|
+
entries = readdirSync(join(root, WORKFLOW_DIR));
|
|
41
|
+
}
|
|
42
|
+
catch {
|
|
43
|
+
return [];
|
|
44
|
+
}
|
|
45
|
+
// Bounded by a non-word character so `scripts/lint.mjs` is not considered
|
|
46
|
+
// called by a workflow whose only mention is `scripts/lint-staged.mjs`.
|
|
47
|
+
const mention = new RegExp(`(^|[^\\w-])${name.replace(/\./g, '\\.')}(\\W|$)`);
|
|
48
|
+
const found = [];
|
|
49
|
+
for (const entry of entries.sort()) {
|
|
50
|
+
if (!/\.ya?ml$/.test(entry))
|
|
51
|
+
continue;
|
|
52
|
+
const path = `${WORKFLOW_DIR}/${entry}`;
|
|
53
|
+
try {
|
|
54
|
+
if (mention.test(readFileSync(join(root, path), 'utf8')))
|
|
55
|
+
found.push(path);
|
|
56
|
+
}
|
|
57
|
+
catch {
|
|
58
|
+
// An unreadable workflow cannot be shown to reference the script, so it
|
|
59
|
+
// is not counted as one. It is skipped rather than made fatal.
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
return found;
|
|
63
|
+
}
|
|
64
|
+
/** A caller's file name, which is what a reader recognises it by. */
|
|
65
|
+
function named(verdict) {
|
|
66
|
+
return basename(verdict.file);
|
|
67
|
+
}
|
|
68
|
+
/** Nothing under `.github/workflows` mentions the script. */
|
|
69
|
+
function unreferencedScript(file) {
|
|
70
|
+
return {
|
|
71
|
+
file,
|
|
72
|
+
status: 'abstain',
|
|
73
|
+
explanation: explain('No workflow under .github/workflows references this script, so its ' +
|
|
74
|
+
'execution cannot be traced. It may still be run by hand or from a ' +
|
|
75
|
+
'command this scan cannot see.'),
|
|
76
|
+
evidence: `searched ${WORKFLOW_DIR} for ${basename(file)}`,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
/** At least one calling workflow ran after the merge, so the script ran too. */
|
|
80
|
+
function scriptRanWith(file, ran, callers, ctx) {
|
|
81
|
+
return {
|
|
82
|
+
file,
|
|
83
|
+
status: 'exercised',
|
|
84
|
+
explanation: explain(`It is called by ${named(ran)}, which has run since this fix merged, ` +
|
|
85
|
+
'so the script ran with it.'),
|
|
86
|
+
evidence: `${callers.length} referencing workflow(s); ${named(ran)} ran after ${day(ctx.mergedAt)}`,
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
/** Every calling workflow has been dormant since the merge. */
|
|
90
|
+
function everyCallerDormant(file, judged, ctx) {
|
|
91
|
+
return {
|
|
92
|
+
file,
|
|
93
|
+
status: 'not-exercised',
|
|
94
|
+
explanation: explain(`Every workflow that calls it (${judged.map(named).join(', ')}) has ` +
|
|
95
|
+
'been dormant since this fix merged, so the script has not run.'),
|
|
96
|
+
evidence: `${judged.length} referencing workflow(s), none run after ${day(ctx.mergedAt)}`,
|
|
97
|
+
};
|
|
98
|
+
}
|
|
99
|
+
/** At least one calling workflow's history could not be read back far enough. */
|
|
100
|
+
function callerHistoryUnknown(file, blind, judged) {
|
|
101
|
+
return {
|
|
102
|
+
file,
|
|
103
|
+
status: 'abstain',
|
|
104
|
+
explanation: explain(`The run history of ${blind.map(named).join(', ')} was truncated before ` +
|
|
105
|
+
'reaching the merge, so whether this script has run since cannot be ' +
|
|
106
|
+
'told. This is not a claim that it has not run.'),
|
|
107
|
+
evidence: `${judged.length} referencing workflow(s); ${blind.length} with a truncated history`,
|
|
108
|
+
};
|
|
109
|
+
}
|
|
110
|
+
/**
|
|
111
|
+
* `scripts/*.mjs` -- resolved statically to the workflows that call it, then
|
|
112
|
+
* decided by their run history.
|
|
113
|
+
*
|
|
114
|
+
* **Abstains when nothing references it**, rather than reporting it never
|
|
115
|
+
* ran. Those are different facts: "no workflow calls this" is a statement
|
|
116
|
+
* about the repo, and it may well be run by a human, a hook, or a workflow
|
|
117
|
+
* that builds the command dynamically. Calling that `not-exercised` would
|
|
118
|
+
* assert something no file supports (spec D3).
|
|
119
|
+
*/
|
|
120
|
+
export function workflowScriptProbe() {
|
|
121
|
+
return {
|
|
122
|
+
id: 'workflow-script',
|
|
123
|
+
artifact: 'workflow script',
|
|
124
|
+
matches: (file) => SCRIPT_RE.test(file),
|
|
125
|
+
async probe(file, ctx) {
|
|
126
|
+
const callers = workflowsReferencing(ctx.root, basename(file));
|
|
127
|
+
if (callers.length === 0)
|
|
128
|
+
return unreferencedScript(file);
|
|
129
|
+
const judged = await Promise.all(callers.map(async (caller) => judgeRuns(caller, await ctx.runs.runsForWorkflow(caller), ctx)));
|
|
130
|
+
const ran = judged.find((v) => v.status === 'exercised');
|
|
131
|
+
if (ran !== undefined)
|
|
132
|
+
return scriptRanWith(file, ran, callers, ctx);
|
|
133
|
+
// A caller whose own history was truncated has not been shown to be
|
|
134
|
+
// dormant, so the script cannot be called dormant either. Folding an
|
|
135
|
+
// abstention into "every caller dormant" would launder the one status
|
|
136
|
+
// that admits ignorance into a claim (spec Phase 3).
|
|
137
|
+
const blind = judged.filter((v) => v.status === 'abstain');
|
|
138
|
+
if (blind.length > 0)
|
|
139
|
+
return callerHistoryUnknown(file, blind, judged);
|
|
140
|
+
return everyCallerDormant(file, judged, ctx);
|
|
141
|
+
},
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
/** Files that are read, not run. Extensions first, then exact names. */
|
|
145
|
+
const NO_EXECUTION_EXT = /\.(md|json|ya?ml|toml|ini|txt)$/;
|
|
146
|
+
const NO_EXECUTION_NAMES = new Set([
|
|
147
|
+
'.gitignore',
|
|
148
|
+
'.gitattributes',
|
|
149
|
+
'.npmrc',
|
|
150
|
+
'.nvmrc',
|
|
151
|
+
'.prettierignore',
|
|
152
|
+
'.editorconfig',
|
|
153
|
+
'LICENSE',
|
|
154
|
+
'CODEOWNERS',
|
|
155
|
+
]);
|
|
156
|
+
/**
|
|
157
|
+
* Prose and configuration -- `not-applicable` without reading anything.
|
|
158
|
+
*
|
|
159
|
+
* A workflow is YAML, so the extension alone would swallow every file in
|
|
160
|
+
* `.github/workflows/` and declare it unrunnable. Registration order already
|
|
161
|
+
* puts `workflowProbe` first, but relying on that alone would mean a reordered
|
|
162
|
+
* list silently reclassified every workflow as `not-applicable` -- a whole
|
|
163
|
+
* artifact type disappearing into a status that asks nothing of anyone, which
|
|
164
|
+
* is precisely the silence batwoman exists to remove. The exclusion is stated
|
|
165
|
+
* here so the probe is correct standing alone, and the order is asserted
|
|
166
|
+
* separately.
|
|
167
|
+
*/
|
|
168
|
+
export function noExecutionProbe() {
|
|
169
|
+
return {
|
|
170
|
+
id: 'no-execution',
|
|
171
|
+
artifact: 'documentation or configuration',
|
|
172
|
+
matches: (file) => !WORKFLOW_RE.test(file) &&
|
|
173
|
+
(NO_EXECUTION_EXT.test(file) || NO_EXECUTION_NAMES.has(basename(file))),
|
|
174
|
+
probe(file) {
|
|
175
|
+
return Promise.resolve({
|
|
176
|
+
file,
|
|
177
|
+
status: 'not-applicable',
|
|
178
|
+
explanation: explain('This file is read rather than run, so "has it executed" is not a ' +
|
|
179
|
+
'question it can answer.'),
|
|
180
|
+
});
|
|
181
|
+
},
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* The shipped probes, in match order.
|
|
186
|
+
*
|
|
187
|
+
* Order matters: a workflow file matches both `workflowProbe` and
|
|
188
|
+
* `noExecutionProbe`'s `.yml` extension, and `matchProbe` takes the first
|
|
189
|
+
* match. Everything unmatched becomes a `no-probe` row, which is the honest
|
|
190
|
+
* state for `ts/src/**` in v1 and is countable rather than silent.
|
|
191
|
+
*/
|
|
192
|
+
export function shippedProbes() {
|
|
193
|
+
return [workflowProbe(), workflowScriptProbe(), noExecutionProbe()];
|
|
194
|
+
}
|
|
195
|
+
//# sourceMappingURL=probes.js.map
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The probe registry: matching, dispatch, and the two named non-answers.
|
|
3
|
+
*
|
|
4
|
+
* Per spec D3 the gaps have to be countable, so an unmatched file produces a
|
|
5
|
+
* `no-probe` verdict that names what it could not classify rather than
|
|
6
|
+
* silence. The noun phrase comes from the path, not from a probe -- by
|
|
7
|
+
* construction no probe matched, so no probe can supply it.
|
|
8
|
+
* `ExerciseProbe.artifact` is the matched-probe counterpart, supplied by the
|
|
9
|
+
* probes themselves (see `probes.ts`).
|
|
10
|
+
*/
|
|
11
|
+
import { CLAIMING_STATUSES, EXERCISE_STATUSES, explain, isExplanation, } from './verdict.js';
|
|
12
|
+
/**
|
|
13
|
+
* Path patterns to the noun phrase a NO PROBE row prints.
|
|
14
|
+
*
|
|
15
|
+
* Order is load-bearing: `*.test.ts` must be read as a test file before the
|
|
16
|
+
* generic source-module rule claims it, and a skill's `SKILL.md` before the
|
|
17
|
+
* generic document rule. The final fallback is a phrase rather than an empty
|
|
18
|
+
* string, because a row that names nothing is the silence D3 exists to remove.
|
|
19
|
+
*/
|
|
20
|
+
const ARTIFACT_RULES = [
|
|
21
|
+
[/\.(test|spec)\.[cm]?[jt]sx?$/, 'test file'],
|
|
22
|
+
[/^agents\/skills\/.+\.md$/, 'skill document'],
|
|
23
|
+
[/\.(json|ya?ml|toml|ini|lock)$/, 'config'],
|
|
24
|
+
[/\.[cm]?[jt]sx?$/, 'source module'],
|
|
25
|
+
[/\.md$/, 'document'],
|
|
26
|
+
];
|
|
27
|
+
/** The noun phrase for a file, for use in a `no-probe` row. */
|
|
28
|
+
export function describeArtifact(file) {
|
|
29
|
+
for (const [pattern, noun] of ARTIFACT_RULES) {
|
|
30
|
+
if (pattern.test(file))
|
|
31
|
+
return noun;
|
|
32
|
+
}
|
|
33
|
+
return 'unrecognised artifact';
|
|
34
|
+
}
|
|
35
|
+
/** The first probe claiming this file, or null. */
|
|
36
|
+
export function matchProbe(probes, file) {
|
|
37
|
+
return probes.find((probe) => probe.matches(file)) ?? null;
|
|
38
|
+
}
|
|
39
|
+
/** An `abstain` naming the probe and what was wrong with its answer. */
|
|
40
|
+
function distrust(probeId, file, fault) {
|
|
41
|
+
return {
|
|
42
|
+
file,
|
|
43
|
+
status: 'abstain',
|
|
44
|
+
explanation: explain(`the ${probeId} probe answered about this file but batwoman could not ` +
|
|
45
|
+
`trust its verdict, because ${fault}; the file is therefore ` +
|
|
46
|
+
'unassessed rather than clean.'),
|
|
47
|
+
evidence: `${probeId} probe answer`,
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* What is wrong with a probe's answer, or `null` if nothing is.
|
|
52
|
+
*
|
|
53
|
+
* The type boundary in `verdict.ts` makes each of these unrepresentable in
|
|
54
|
+
* TypeScript, so this is the seam for answers arriving from outside it: a
|
|
55
|
+
* JavaScript probe, a plugin, a `JSON.parse`. It matters because an out-of-
|
|
56
|
+
* union status reaches `tallyVerdicts`, where `byStatus[status] += 1` yields
|
|
57
|
+
* `NaN` and the summary line stops summing to its own denominator -- a
|
|
58
|
+
* false-green shape one level up from the one batwoman detects.
|
|
59
|
+
*/
|
|
60
|
+
function faultIn(answer, file) {
|
|
61
|
+
if (typeof answer !== 'object' || answer === null) {
|
|
62
|
+
return (`it returned ${answer === null ? 'null' : typeof answer} rather ` +
|
|
63
|
+
'than a verdict');
|
|
64
|
+
}
|
|
65
|
+
const verdict = answer;
|
|
66
|
+
if (verdict.file !== file) {
|
|
67
|
+
return `it answered about ${String(verdict.file)} instead`;
|
|
68
|
+
}
|
|
69
|
+
if (!EXERCISE_STATUSES.includes(verdict.status)) {
|
|
70
|
+
return `'${String(verdict.status)}' is not one of the five statuses`;
|
|
71
|
+
}
|
|
72
|
+
if (!isExplanation(verdict.explanation)) {
|
|
73
|
+
return ('it supplied no explanation, and a row that names a file and says ' +
|
|
74
|
+
'nothing about it is not a verdict');
|
|
75
|
+
}
|
|
76
|
+
const claims = CLAIMING_STATUSES.includes(verdict.status);
|
|
77
|
+
if (claims && typeof verdict.evidence !== 'string') {
|
|
78
|
+
return (`a '${String(verdict.status)}' claim must name the evidence ` +
|
|
79
|
+
'behind it, and this one named none');
|
|
80
|
+
}
|
|
81
|
+
return null;
|
|
82
|
+
}
|
|
83
|
+
/** One file's verdict, via the probe that claims it. */
|
|
84
|
+
export async function probeFile(probes, file, ctx) {
|
|
85
|
+
// Answered ahead of dispatch, and ahead of the no-probe fallback. A file the
|
|
86
|
+
// closing PR deleted has nothing left to execute and no longer exists to
|
|
87
|
+
// classify, so neither "nothing looked" nor a port failure is the right
|
|
88
|
+
// report for it -- the answer does not depend on run history at all
|
|
89
|
+
// (spec Phase 3).
|
|
90
|
+
if (ctx.deleted.has(file)) {
|
|
91
|
+
return {
|
|
92
|
+
file,
|
|
93
|
+
status: 'not-applicable',
|
|
94
|
+
explanation: explain('It was deleted by this change, so there is nothing left to run.'),
|
|
95
|
+
evidence: 'deleted by this change',
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
const probe = matchProbe(probes, file);
|
|
99
|
+
if (probe === null) {
|
|
100
|
+
return {
|
|
101
|
+
file,
|
|
102
|
+
status: 'no-probe',
|
|
103
|
+
explanation: explain(`batwoman has no probe for this ${describeArtifact(file)}, so ` +
|
|
104
|
+
'nothing looked at whether it has run since the fix merged.'),
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
try {
|
|
108
|
+
const answer = await probe.probe(file, ctx);
|
|
109
|
+
const fault = faultIn(answer, file);
|
|
110
|
+
return fault === null
|
|
111
|
+
? answer
|
|
112
|
+
: distrust(probe.id, file, fault);
|
|
113
|
+
}
|
|
114
|
+
catch (err) {
|
|
115
|
+
// Cannot-verify is a finding, not a pass (spec criterion 4). A probe whose
|
|
116
|
+
// evidence source failed knows strictly less than one that never ran, so
|
|
117
|
+
// the only honest answer is abstain -- and the `await` inside the `try` is
|
|
118
|
+
// load-bearing: without it a rejected promise escapes the catch.
|
|
119
|
+
return {
|
|
120
|
+
file,
|
|
121
|
+
status: 'abstain',
|
|
122
|
+
explanation: explain(`the ${probe.id} probe looked at this file but could not decide ` +
|
|
123
|
+
'whether it ran, because reading its evidence failed: ' +
|
|
124
|
+
`${err instanceof Error ? err.message : String(err)}.`),
|
|
125
|
+
evidence: `${probe.id} probe error`,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* Probe every changed file, in input order, one at a time.
|
|
131
|
+
*
|
|
132
|
+
* Sequential rather than `Promise.all`: the real port shells out to `gh`, and a
|
|
133
|
+
* fan-out over a whole changed-file set would turn one audit into a burst of
|
|
134
|
+
* API calls. Order is preserved so the report is reproducible.
|
|
135
|
+
*/
|
|
136
|
+
export async function probeAll(probes, files, ctx) {
|
|
137
|
+
const verdicts = [];
|
|
138
|
+
for (const file of files)
|
|
139
|
+
verdicts.push(await probeFile(probes, file, ctx));
|
|
140
|
+
return verdicts;
|
|
141
|
+
}
|
|
142
|
+
//# sourceMappingURL=registry.js.map
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* batwoman's report, rendered through canary's persona registry (spec D7).
|
|
3
|
+
*
|
|
4
|
+
* Two rules bind harder than anything else here, and both are asserted in
|
|
5
|
+
* `ts/test/batwoman-render-invariants.test.ts` across all three registers
|
|
6
|
+
* rather than once against the default:
|
|
7
|
+
*
|
|
8
|
+
* **There is no success-only path.** Every run ends with one column per status,
|
|
9
|
+
* including the two that mean batwoman could not decide. A detector that
|
|
10
|
+
* covered two of seven changed files and printed a clean token would be a pass
|
|
11
|
+
* over a denominator of two presented as a pass over seven -- the exact defect
|
|
12
|
+
* batwoman exists to catch, committed by batwoman (spec D5).
|
|
13
|
+
*
|
|
14
|
+
* **Every row keeps its sentence.** `ExerciseVerdict.explanation` renders in
|
|
15
|
+
* every register, terse included. What the terse register drops is the material
|
|
16
|
+
* gated by the persona's `reasoning` flag -- the evidence line and the
|
|
17
|
+
* next-step guidance -- not the observation-and-cause sentence itself, which is
|
|
18
|
+
* the difference between a verdict and a status code.
|
|
19
|
+
*/
|
|
20
|
+
import { EXERCISE_STATUSES, tallyVerdicts, } from './verdict.js';
|
|
21
|
+
import { WIDTH, fitLine, stamp, wrap } from './text.js';
|
|
22
|
+
/** Column labels, in `EXERCISE_STATUSES` order. */
|
|
23
|
+
const SUMMARY_LABELS = {
|
|
24
|
+
exercised: 'exercised',
|
|
25
|
+
'not-exercised': 'not exercised',
|
|
26
|
+
abstain: 'abstained',
|
|
27
|
+
'no-probe': 'no probe',
|
|
28
|
+
'not-applicable': 'n/a',
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* The one line every run prints.
|
|
32
|
+
*
|
|
33
|
+
* The changed-file total leads so the denominator is read before any count;
|
|
34
|
+
* the render suite asserts the columns sum to it.
|
|
35
|
+
*/
|
|
36
|
+
export function summaryLine(verdicts) {
|
|
37
|
+
const { changed, byStatus } = tallyVerdicts(verdicts);
|
|
38
|
+
const columns = EXERCISE_STATUSES.map((status) => `${byStatus[status]} ${SUMMARY_LABELS[status]}`);
|
|
39
|
+
return [`${changed} changed`, ...columns].join(' · ');
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* One ` Label value` header row, wrapped with a hanging indent.
|
|
43
|
+
*
|
|
44
|
+
* The header wraps for the same reason the body does, and it is the amendment
|
|
45
|
+
* this module needed most: both values it carries are free text of unbounded
|
|
46
|
+
* length. A commit subject is whatever the author typed, and
|
|
47
|
+
* `ResolvedPersona.reason` is a composed sentence -- an unknown register
|
|
48
|
+
* produces one naming every candidate. Emitted unwrapped, those two lines were
|
|
49
|
+
* 83 and 199 columns wide against a 78-column report, and only short fixtures
|
|
50
|
+
* kept that from being visible.
|
|
51
|
+
*/
|
|
52
|
+
function labelled(label, value) {
|
|
53
|
+
const prefix = ` ${label.padEnd(9)} `;
|
|
54
|
+
const lines = wrap(value, WIDTH, ' '.repeat(prefix.length));
|
|
55
|
+
const [first, ...rest] = lines;
|
|
56
|
+
// An empty value used to emit ` Closed by` and nothing else -- a label
|
|
57
|
+
// dangling over blank space, which is the same silent-empty shape the verdict
|
|
58
|
+
// model now forbids outright. Naming the gap keeps it countable by eye.
|
|
59
|
+
if (first === undefined)
|
|
60
|
+
return [`${prefix}(none recorded)`];
|
|
61
|
+
// The first line swaps the blank indent for the label; the rest keep it, so
|
|
62
|
+
// continuations align under the value rather than under the label.
|
|
63
|
+
return [prefix + first.slice(prefix.length), ...rest];
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* A commit oid at the length a human reads and pastes.
|
|
67
|
+
*
|
|
68
|
+
* The closure adapter carries the full 40-character oid, which is right for
|
|
69
|
+
* data and wrong for a 78-column report: printed whole it pushed the subject
|
|
70
|
+
* onto a second line and split the hash across the wrap, which reads as
|
|
71
|
+
* corruption rather than as a hash. Seven is git's own short form, and what
|
|
72
|
+
* the spec's sample shows. A value that is not an oid is left alone -- an
|
|
73
|
+
* abbreviated non-hash would be a lie about what it is.
|
|
74
|
+
*/
|
|
75
|
+
function abbreviate(sha) {
|
|
76
|
+
return /^[0-9a-f]{40}$/i.test(sha) ? sha.slice(0, 7) : sha;
|
|
77
|
+
}
|
|
78
|
+
function headerLines(options) {
|
|
79
|
+
const { header, repo, persona } = options;
|
|
80
|
+
return [
|
|
81
|
+
`canary batwoman — ${repo}#${header.issue}`,
|
|
82
|
+
'',
|
|
83
|
+
...labelled('Closed by', `${abbreviate(header.mergeSha)} ${header.mergeSubject}`),
|
|
84
|
+
...labelled('Merged', stamp(header.mergedAt)),
|
|
85
|
+
// Printed so a reader who got terse output when they wanted guided output
|
|
86
|
+
// can tell a short report from a truncated one (spec criterion 9).
|
|
87
|
+
...labelled('Register', `${persona.persona.id} (${persona.persona.label}) — ` +
|
|
88
|
+
`${persona.source}: ${persona.reason}`),
|
|
89
|
+
'',
|
|
90
|
+
];
|
|
91
|
+
}
|
|
92
|
+
/** Section headings, in report order: the findings first. */
|
|
93
|
+
const SECTIONS = [
|
|
94
|
+
['not-exercised', 'NOT EXERCISED'],
|
|
95
|
+
['abstain', 'ABSTAINED'],
|
|
96
|
+
['no-probe', 'NO PROBE'],
|
|
97
|
+
['exercised', 'EXERCISED'],
|
|
98
|
+
['not-applicable', 'NOT APPLICABLE'],
|
|
99
|
+
];
|
|
100
|
+
/**
|
|
101
|
+
* A section heading.
|
|
102
|
+
*
|
|
103
|
+
* The first finding section states its denominator inline; the rest are counted
|
|
104
|
+
* against the same total one line below, in the summary that always prints.
|
|
105
|
+
*/
|
|
106
|
+
function heading(label, status, rows, total) {
|
|
107
|
+
const files = (count) => (count === 1 ? 'file' : 'files');
|
|
108
|
+
return status === 'not-exercised'
|
|
109
|
+
? ` ${label} — ${rows} of ${total} changed ${files(total)}`
|
|
110
|
+
: ` ${label} — ${rows} ${files(rows)}`;
|
|
111
|
+
}
|
|
112
|
+
function briefRows(verdicts, reasoning) {
|
|
113
|
+
const lines = [];
|
|
114
|
+
for (const verdict of verdicts) {
|
|
115
|
+
lines.push(...fitLine(` ${verdict.file}`, WIDTH, ' '));
|
|
116
|
+
lines.push(...wrap(verdict.explanation, WIDTH, ' '));
|
|
117
|
+
if (reasoning && verdict.evidence !== undefined) {
|
|
118
|
+
lines.push(...fitLine(` Read: ${verdict.evidence}`, WIDTH, ' '));
|
|
119
|
+
}
|
|
120
|
+
lines.push('');
|
|
121
|
+
}
|
|
122
|
+
return lines;
|
|
123
|
+
}
|
|
124
|
+
/**
|
|
125
|
+
* What a reader can actually do about each status.
|
|
126
|
+
*
|
|
127
|
+
* `exercised` and `not-applicable` are absent on purpose: there is nothing to
|
|
128
|
+
* ask for, and inventing a step for them would be the guided register's version
|
|
129
|
+
* of a success token.
|
|
130
|
+
*/
|
|
131
|
+
const NEXT_STEPS = {
|
|
132
|
+
'not-exercised': 'Run the path this file belongs to, then re-run canary batwoman to ' +
|
|
133
|
+
'confirm it moved.',
|
|
134
|
+
abstain: 'Check the evidence source named above, then re-run canary batwoman so ' +
|
|
135
|
+
'this file gets a verdict instead of a gap.',
|
|
136
|
+
'no-probe': 'Add an ExerciseProbe for this artifact type so the gap stops being ' +
|
|
137
|
+
'uncountable, or confirm by hand that the file ran.',
|
|
138
|
+
};
|
|
139
|
+
function guidedRows(verdicts, status) {
|
|
140
|
+
const lines = [];
|
|
141
|
+
verdicts.forEach((verdict, index) => {
|
|
142
|
+
lines.push(...fitLine(` ${index + 1}. ${verdict.file}`, WIDTH, ' '));
|
|
143
|
+
lines.push(...wrap(verdict.explanation, WIDTH, ' '));
|
|
144
|
+
if (verdict.evidence !== undefined) {
|
|
145
|
+
lines.push(...fitLine(` Read: ${verdict.evidence}`, WIDTH, ' '));
|
|
146
|
+
}
|
|
147
|
+
const step = NEXT_STEPS[status];
|
|
148
|
+
if (step !== undefined) {
|
|
149
|
+
lines.push(...wrap(`Next: ${step}`, WIDTH, ' '));
|
|
150
|
+
}
|
|
151
|
+
lines.push('');
|
|
152
|
+
});
|
|
153
|
+
return lines;
|
|
154
|
+
}
|
|
155
|
+
function terseRows(verdicts) {
|
|
156
|
+
return verdicts.flatMap((verdict) => [
|
|
157
|
+
...fitLine(` - ${verdict.file} ${verdict.status}`, WIDTH, ' '),
|
|
158
|
+
...wrap(verdict.explanation, WIDTH, ' '),
|
|
159
|
+
]);
|
|
160
|
+
}
|
|
161
|
+
function bodyLines(options) {
|
|
162
|
+
const { verdicts, persona } = options;
|
|
163
|
+
const lines = [];
|
|
164
|
+
for (const [status, label] of SECTIONS) {
|
|
165
|
+
const rows = verdicts.filter((verdict) => verdict.status === status);
|
|
166
|
+
// An empty section is omitted, not printed as a zero: the counts that
|
|
167
|
+
// matter are in the summary line, which prints all five unconditionally.
|
|
168
|
+
if (rows.length === 0)
|
|
169
|
+
continue;
|
|
170
|
+
if (persona.persona.depth === 'terse') {
|
|
171
|
+
lines.push(...terseRows(rows));
|
|
172
|
+
continue;
|
|
173
|
+
}
|
|
174
|
+
if (persona.persona.depth === 'guided') {
|
|
175
|
+
lines.push(heading(label, status, rows.length, verdicts.length));
|
|
176
|
+
lines.push(...guidedRows(rows, status));
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
lines.push(heading(label, status, rows.length, verdicts.length));
|
|
180
|
+
lines.push(...briefRows(rows, persona.persona.reasoning));
|
|
181
|
+
}
|
|
182
|
+
return lines;
|
|
183
|
+
}
|
|
184
|
+
/** The whole report, as one string whose last content line is the summary. */
|
|
185
|
+
export function renderReport(options) {
|
|
186
|
+
return [
|
|
187
|
+
...headerLines(options),
|
|
188
|
+
...bodyLines(options),
|
|
189
|
+
' ' + '─'.repeat(62),
|
|
190
|
+
' ' + summaryLine(options.verdicts),
|
|
191
|
+
'',
|
|
192
|
+
].join('\n');
|
|
193
|
+
}
|
|
194
|
+
//# sourceMappingURL=render.js.map
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Turning one fetched window of run history into a verdict.
|
|
3
|
+
*
|
|
4
|
+
* Separated from the probes because it is the part that reasons about *time
|
|
5
|
+
* and evidence* rather than about artifact types: which runs postdate the
|
|
6
|
+
* merge, whether the window saw enough to justify a negative answer, and how
|
|
7
|
+
* a dormant workflow explains itself. Both the workflow probe and the
|
|
8
|
+
* workflow-script probe decide through this one function, so the two cannot
|
|
9
|
+
* drift apart in what they will claim.
|
|
10
|
+
*/
|
|
11
|
+
import { readFileSync } from 'node:fs';
|
|
12
|
+
import { join } from 'node:path';
|
|
13
|
+
import { load } from 'js-yaml';
|
|
14
|
+
import { describeTriggers } from './triggers.js';
|
|
15
|
+
import { explain, } from './verdict.js';
|
|
16
|
+
/** `2026-08-10`, the grain a human reasons about a run in. */
|
|
17
|
+
export function day(when) {
|
|
18
|
+
return when.toISOString().slice(0, 10);
|
|
19
|
+
}
|
|
20
|
+
/** The most recent run, or undefined. Order from the port is not assumed. */
|
|
21
|
+
function latest(runs) {
|
|
22
|
+
return [...runs].sort((a, b) => b.createdAt.getTime() - a.createdAt.getTime())[0];
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* The workflow's trigger clause, or null when the file cannot be read.
|
|
26
|
+
*
|
|
27
|
+
* Deliberately swallows every read and parse error. The trigger explanation
|
|
28
|
+
* and the run-history verdict come from different places and fail
|
|
29
|
+
* independently: a workflow deleted by the very PR under audit still has a run
|
|
30
|
+
* history worth reporting, and losing that verdict because its file is gone
|
|
31
|
+
* would be a self-inflicted abstention.
|
|
32
|
+
*/
|
|
33
|
+
function triggerClause(root, file) {
|
|
34
|
+
try {
|
|
35
|
+
const doc = load(readFileSync(join(root, file), 'utf8'));
|
|
36
|
+
if (typeof doc !== 'object' || doc === null)
|
|
37
|
+
return null;
|
|
38
|
+
const record = doc;
|
|
39
|
+
// `on` is read under both spellings. Under a YAML 1.1 resolver an
|
|
40
|
+
// unquoted `on:` becomes boolean true and the block vanishes; js-yaml
|
|
41
|
+
// 5.4.1 keeps it a string, so this is defence against a parser change
|
|
42
|
+
// rather than a bug being worked around.
|
|
43
|
+
return describeTriggers(record['on'] ?? record['true']);
|
|
44
|
+
}
|
|
45
|
+
catch {
|
|
46
|
+
return null;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
/** The sentence a dormant workflow gets: the fact, then the cause. */
|
|
50
|
+
function dormantExplanation(last, clause) {
|
|
51
|
+
const fact = last === undefined
|
|
52
|
+
? 'It has no recorded runs at all.'
|
|
53
|
+
: `Last ran ${day(last.createdAt)}, before this fix merged.`;
|
|
54
|
+
// Without a clause the sentence stops at the fact. Saying less is the
|
|
55
|
+
// correct degradation; a guessed cause would be acted on.
|
|
56
|
+
return clause === null
|
|
57
|
+
? fact
|
|
58
|
+
: `${fact} It is ${clause}, so it has not run since.`;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* True when a *negative* answer from this window would be unfounded.
|
|
62
|
+
*
|
|
63
|
+
* **Narrower than the spec's wording, deliberately.** The spec asks the probe
|
|
64
|
+
* to abstain whenever the fetched page "does not reach back past `mergedAt`",
|
|
65
|
+
* reasoning that absence of a qualifying run would then be indistinguishable
|
|
66
|
+
* from truncation. That rationale does not survive contact with the API:
|
|
67
|
+
* `gh run list` returns runs newest-first (verified against this repo), so
|
|
68
|
+
* every run beyond the page is OLDER than every run in it. For a qualifying
|
|
69
|
+
* run to hide outside the window, the oldest fetched run would have to
|
|
70
|
+
* postdate the merge -- and that same condition puts a qualifying run *inside*
|
|
71
|
+
* the window, where the check above has already found it. A dropped-off page
|
|
72
|
+
* therefore cannot conceal a run after the merge.
|
|
73
|
+
*
|
|
74
|
+
* What remains genuinely blind is a window that is both empty and truncated:
|
|
75
|
+
* it reports nothing and admits there is more, which is an absence of evidence
|
|
76
|
+
* rather than evidence of absence. That case abstains.
|
|
77
|
+
*
|
|
78
|
+
* The `complete` flag is kept on {@link RunHistory} regardless. It costs one
|
|
79
|
+
* comparison, it is the honest description of what the port fetched, and it is
|
|
80
|
+
* what makes this argument checkable instead of assumed -- if the ordering
|
|
81
|
+
* guarantee ever changes, this is the one function that has to change with it.
|
|
82
|
+
*/
|
|
83
|
+
function windowIsBlind(history, _mergedAt) {
|
|
84
|
+
return !history.complete && history.runs.length === 0;
|
|
85
|
+
}
|
|
86
|
+
/** Decide one workflow from its run history. Shared with the script probe. */
|
|
87
|
+
export function judgeRuns(file, history, ctx) {
|
|
88
|
+
const runs = history.runs;
|
|
89
|
+
const after = runs.filter((r) => r.createdAt > ctx.mergedAt);
|
|
90
|
+
const mostRecent = latest(after);
|
|
91
|
+
// Checked before truncation on purpose: a run after the merge is a positive
|
|
92
|
+
// finding, and nothing hidden further back could overturn it.
|
|
93
|
+
if (mostRecent !== undefined) {
|
|
94
|
+
return {
|
|
95
|
+
file,
|
|
96
|
+
status: 'exercised',
|
|
97
|
+
explanation: explain(`It has run ${after.length === 1 ? 'once' : `${after.length} times`} ` +
|
|
98
|
+
`since this fix merged, most recently on ${day(mostRecent.createdAt)}.`),
|
|
99
|
+
evidence: `${runs.length} recorded run(s) for ${file}; ${after.length} after ${day(ctx.mergedAt)}`,
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
if (windowIsBlind(history, ctx.mergedAt)) {
|
|
103
|
+
return {
|
|
104
|
+
file,
|
|
105
|
+
status: 'abstain',
|
|
106
|
+
explanation: explain('Its run history came back empty but incomplete, so nothing was ' +
|
|
107
|
+
'learned about whether it has run. This is not a claim that it ' +
|
|
108
|
+
'has not run.'),
|
|
109
|
+
evidence: `an empty but truncated run history for ${file}; nothing reaching back to ${day(ctx.mergedAt)}`,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
const last = latest(runs);
|
|
113
|
+
return {
|
|
114
|
+
file,
|
|
115
|
+
status: 'not-exercised',
|
|
116
|
+
explanation: explain(dormantExplanation(last, triggerClause(ctx.root, file))),
|
|
117
|
+
evidence: last === undefined
|
|
118
|
+
? `no recorded runs for ${file}`
|
|
119
|
+
: `${runs.length} recorded run(s) for ${file}, none after ${day(ctx.mergedAt)}`,
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
//# sourceMappingURL=run-window.js.map
|