canary-test-cli 7.2.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +23 -4
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +23 -16
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +3 -1
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +20 -3
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +1 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +15 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/lib/parse-args.mjs +200 -139
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +46 -7
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +13 -18
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/vacuity-scanner.js +151 -6
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/cli.js +180 -264
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +262 -430
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +48 -32
- package/dist/engine/workflow-cli.js +85 -65
- package/package.json +1 -1
|
@@ -19,6 +19,10 @@
|
|
|
19
19
|
// `test/skill-cli-conformance.test.ts` discovers every SKILL.md declaring
|
|
20
20
|
// `cli:` and asserts the module exports the `CLI_SPEC` it passed here, so a
|
|
21
21
|
// sixth hand-rolled copy fails CI instead of quietly starting the cycle again.
|
|
22
|
+
//
|
|
23
|
+
// The parse is split into one small helper per decision (#906): the single
|
|
24
|
+
// 155-line closure it replaced measured cyclomatic complexity 42, and every
|
|
25
|
+
// branch of it is pinned end to end in `test/parse-args.test.ts`.
|
|
22
26
|
|
|
23
27
|
/** Exit code argparse reserves for usage errors; the whole family follows it. */
|
|
24
28
|
export const EXIT_USAGE = 2;
|
|
@@ -42,6 +46,191 @@ function nullProtoMap(entries) {
|
|
|
42
46
|
return Object.assign(Object.create(null), entries);
|
|
43
47
|
}
|
|
44
48
|
|
|
49
|
+
function expectedOneArgument(flag) {
|
|
50
|
+
return `argument ${flag}: expected one argument`;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Fail loudly at construction on a spec that cannot be satisfied -- a rename
|
|
55
|
+
* that leaves a stale default or a required flag behind is otherwise silent.
|
|
56
|
+
*/
|
|
57
|
+
function assertSpecSatisfiable(config) {
|
|
58
|
+
const { booleans, values, defaults, required, VALUES } = config;
|
|
59
|
+
const declaredKeys = new Set([
|
|
60
|
+
...Object.values(booleans),
|
|
61
|
+
...Object.values(values).map((v) => v.key),
|
|
62
|
+
]);
|
|
63
|
+
for (const key of Object.keys(defaults)) {
|
|
64
|
+
if (!declaredKeys.has(key)) {
|
|
65
|
+
throw new Error(`createParser: defaults names unknown key '${key}'`);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
for (const flag of required) {
|
|
69
|
+
if (VALUES[flag] === undefined) {
|
|
70
|
+
throw new Error(`createParser: required names undeclared flag '${flag}'`);
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Booleans start false, value flags start at their default or null. */
|
|
76
|
+
function initialOpts({ booleans, values, defaults }) {
|
|
77
|
+
const opts = Object.create(null);
|
|
78
|
+
for (const key of Object.values(booleans)) opts[key] = false;
|
|
79
|
+
for (const { key } of Object.values(values)) {
|
|
80
|
+
opts[key] = key in defaults ? defaults[key] : null;
|
|
81
|
+
}
|
|
82
|
+
Object.assign(opts, defaults);
|
|
83
|
+
return opts;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Split `--flag=value` once, up front, so both spellings share one path. */
|
|
87
|
+
function splitInlineValue(arg) {
|
|
88
|
+
const eq = arg.startsWith('--') ? arg.indexOf('=') : -1;
|
|
89
|
+
if (eq === -1) return { flag: arg, inline: null };
|
|
90
|
+
return { flag: arg.slice(0, eq), inline: arg.slice(eq + 1) };
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* A leading '-' normally means "the next flag, not my value" -- but a
|
|
95
|
+
* well-formed integer is a legitimate value for an int flag, so `--seed -5`
|
|
96
|
+
* and `--seed=-5` stay the same command.
|
|
97
|
+
*/
|
|
98
|
+
function cannotBeValue(def, next) {
|
|
99
|
+
if (next === undefined) return true;
|
|
100
|
+
return next.startsWith('-') && !(def.type === 'int' && INT_RE.test(next));
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** The raw value text and how many extra argv tokens it used, or an error. */
|
|
104
|
+
function readRawValue(flag, def, inline, next) {
|
|
105
|
+
if (inline !== null) return { raw: inline, consumed: 0 };
|
|
106
|
+
if (cannotBeValue(def, next)) return { error: expectedOneArgument(flag) };
|
|
107
|
+
return { raw: next, consumed: 1 };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/** Validate an int value's syntax AND its exactness. */
|
|
111
|
+
function parseIntValue(flag, raw) {
|
|
112
|
+
// Validate the VALUE, not just its presence: a flag whose purpose is
|
|
113
|
+
// determinism must not decay to a default when its value is junk.
|
|
114
|
+
if (!INT_RE.test(raw)) {
|
|
115
|
+
return { error: `argument ${flag}: invalid int value: '${raw}'` };
|
|
116
|
+
}
|
|
117
|
+
// Syntactically an integer is not enough: Number() silently rounds past
|
|
118
|
+
// 2^53-1, so `--seed 9007199254740993` would RUN with ...992 -- the value
|
|
119
|
+
// used differing from the value asked for, which is the exact class of lie
|
|
120
|
+
// a determinism flag must not tell.
|
|
121
|
+
const parsed = Number(raw);
|
|
122
|
+
if (!Number.isSafeInteger(parsed)) {
|
|
123
|
+
return { error: `argument ${flag}: integer out of safe range: '${raw}'` };
|
|
124
|
+
}
|
|
125
|
+
return { value: parsed };
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function coerceValue(flag, def, raw) {
|
|
129
|
+
// Empty is the missing-value case wearing a disguise. `--repo=` is typed by
|
|
130
|
+
// nobody, but `--repo "$UNSET_VAR"` expands to `--repo ''` in any shell, and
|
|
131
|
+
// an accepted empty path silently retargets writes at the process CWD.
|
|
132
|
+
if (raw === '') return { error: expectedOneArgument(flag) };
|
|
133
|
+
if (def.type === 'int') return parseIntValue(flag, raw);
|
|
134
|
+
return { value: raw };
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/** Store a value flag's value; the outcome says how many tokens it took. */
|
|
138
|
+
function takeValue(state, flag, inline, index) {
|
|
139
|
+
const def = state.VALUES[flag];
|
|
140
|
+
const read = readRawValue(flag, def, inline, state.argv[index + 1]);
|
|
141
|
+
if (read.error !== undefined) return read;
|
|
142
|
+
const coerced = coerceValue(flag, def, read.raw);
|
|
143
|
+
if (coerced.error !== undefined) return coerced;
|
|
144
|
+
state.opts[def.key] = coerced.value;
|
|
145
|
+
return { consumed: read.consumed };
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/** A lone `-` is a positional, as argparse treats it. */
|
|
149
|
+
function isPositional(state, arg) {
|
|
150
|
+
if (!state.positionals) return false;
|
|
151
|
+
return arg === '-' || !arg.startsWith('-');
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** A token that is neither help nor `--`: a flag, a positional, or junk. */
|
|
155
|
+
function readOptionToken(state, arg, index) {
|
|
156
|
+
const { flag, inline } = splitInlineValue(arg);
|
|
157
|
+
if (state.BOOLEANS[flag] !== undefined && inline === null) {
|
|
158
|
+
state.opts[state.BOOLEANS[flag]] = true;
|
|
159
|
+
return { consumed: 0 };
|
|
160
|
+
}
|
|
161
|
+
if (state.VALUES[flag] !== undefined) {
|
|
162
|
+
return takeValue(state, flag, inline, index);
|
|
163
|
+
}
|
|
164
|
+
if (isPositional(state, arg)) {
|
|
165
|
+
state.found.push(arg);
|
|
166
|
+
return { consumed: 0 };
|
|
167
|
+
}
|
|
168
|
+
return { error: `unrecognized arguments: ${arg}` };
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
function endOptions(state) {
|
|
172
|
+
if (!state.positionals) return { error: 'unrecognized arguments: --' };
|
|
173
|
+
state.endOfOptions = true;
|
|
174
|
+
return { consumed: 0 };
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* One token's outcome: `{consumed}` to keep scanning, or `{help}` / `{error}`
|
|
179
|
+
* to stop.
|
|
180
|
+
*/
|
|
181
|
+
function readToken(state, index) {
|
|
182
|
+
const arg = state.argv[index];
|
|
183
|
+
if (state.endOfOptions) {
|
|
184
|
+
state.found.push(arg);
|
|
185
|
+
return { consumed: 0 };
|
|
186
|
+
}
|
|
187
|
+
// Help short-circuits everything, including the required-flag check --
|
|
188
|
+
// otherwise `--help` reports the arguments it is being asked to explain as
|
|
189
|
+
// missing.
|
|
190
|
+
if (arg === '-h' || arg === '--help') return { help: true };
|
|
191
|
+
if (arg === '--') return endOptions(state);
|
|
192
|
+
return readOptionToken(state, arg, index);
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** Scan argv; returns the stopping outcome, or null when argv ran out. */
|
|
196
|
+
function scanTokens(state) {
|
|
197
|
+
for (let i = 0; i < state.argv.length; i += 1) {
|
|
198
|
+
const outcome = readToken(state, i);
|
|
199
|
+
if (outcome.consumed === undefined) return outcome;
|
|
200
|
+
i += outcome.consumed;
|
|
201
|
+
}
|
|
202
|
+
return null;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
function missingRequired({ required, VALUES }, opts) {
|
|
206
|
+
return required.filter((flag) => {
|
|
207
|
+
const value = opts[VALUES[flag].key];
|
|
208
|
+
return value === null || value === undefined;
|
|
209
|
+
});
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
function parseArgv(config, argv) {
|
|
213
|
+
const opts = initialOpts(config);
|
|
214
|
+
const found = [];
|
|
215
|
+
const result = { opts, positionals: found, help: false, error: null };
|
|
216
|
+
const state = { ...config, argv, opts, found, endOfOptions: false };
|
|
217
|
+
|
|
218
|
+
const stop = scanTokens(state);
|
|
219
|
+
if (stop !== null) return Object.assign(result, stop);
|
|
220
|
+
|
|
221
|
+
const missing = missingRequired(config, opts);
|
|
222
|
+
if (missing.length) {
|
|
223
|
+
result.error = `the following arguments are required: ${missing.join(', ')}`;
|
|
224
|
+
return result;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
const { positionals } = config;
|
|
228
|
+
if (positionals && !found.length && positionals.defaults) {
|
|
229
|
+
found.push(...positionals.defaults);
|
|
230
|
+
}
|
|
231
|
+
return result;
|
|
232
|
+
}
|
|
233
|
+
|
|
45
234
|
/**
|
|
46
235
|
* Build a parser from a declarative spec.
|
|
47
236
|
*
|
|
@@ -69,146 +258,18 @@ export function createParser(spec) {
|
|
|
69
258
|
|
|
70
259
|
if (!prog) throw new Error('createParser: spec.prog is required');
|
|
71
260
|
|
|
72
|
-
const
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
if (!declaredKeys.has(key)) {
|
|
83
|
-
throw new Error(`createParser: defaults names unknown key '${key}'`);
|
|
84
|
-
}
|
|
85
|
-
}
|
|
86
|
-
for (const flag of required) {
|
|
87
|
-
if (VALUES[flag] === undefined) {
|
|
88
|
-
throw new Error(`createParser: required names undeclared flag '${flag}'`);
|
|
89
|
-
}
|
|
90
|
-
}
|
|
261
|
+
const config = {
|
|
262
|
+
booleans,
|
|
263
|
+
values,
|
|
264
|
+
defaults,
|
|
265
|
+
required,
|
|
266
|
+
positionals,
|
|
267
|
+
BOOLEANS: nullProtoMap(booleans),
|
|
268
|
+
VALUES: nullProtoMap(values),
|
|
269
|
+
};
|
|
270
|
+
assertSpecSatisfiable(config);
|
|
91
271
|
|
|
92
272
|
return function parse(argv = []) {
|
|
93
|
-
|
|
94
|
-
for (const key of Object.values(booleans)) opts[key] = false;
|
|
95
|
-
for (const { key } of Object.values(values)) {
|
|
96
|
-
opts[key] = key in defaults ? defaults[key] : null;
|
|
97
|
-
}
|
|
98
|
-
Object.assign(opts, defaults);
|
|
99
|
-
|
|
100
|
-
const found = [];
|
|
101
|
-
const result = { opts, positionals: found, help: false, error: null };
|
|
102
|
-
const fail = (message) => {
|
|
103
|
-
result.error = message;
|
|
104
|
-
return result;
|
|
105
|
-
};
|
|
106
|
-
|
|
107
|
-
let endOfOptions = false;
|
|
108
|
-
|
|
109
|
-
for (let i = 0; i < argv.length; i += 1) {
|
|
110
|
-
const arg = argv[i];
|
|
111
|
-
|
|
112
|
-
if (endOfOptions) {
|
|
113
|
-
found.push(arg);
|
|
114
|
-
continue;
|
|
115
|
-
}
|
|
116
|
-
|
|
117
|
-
// Help short-circuits everything, including the required-flag check
|
|
118
|
-
// below -- otherwise `--help` reports the arguments it is being asked to
|
|
119
|
-
// explain as missing.
|
|
120
|
-
if (arg === '-h' || arg === '--help') {
|
|
121
|
-
result.help = true;
|
|
122
|
-
return result;
|
|
123
|
-
}
|
|
124
|
-
|
|
125
|
-
if (arg === '--') {
|
|
126
|
-
if (!positionals) return fail('unrecognized arguments: --');
|
|
127
|
-
endOfOptions = true;
|
|
128
|
-
continue;
|
|
129
|
-
}
|
|
130
|
-
|
|
131
|
-
// Split `--flag=value` once, up front, so both spellings share one path.
|
|
132
|
-
const eq = arg.startsWith('--') ? arg.indexOf('=') : -1;
|
|
133
|
-
const flag = eq === -1 ? arg : arg.slice(0, eq);
|
|
134
|
-
const inline = eq === -1 ? null : arg.slice(eq + 1);
|
|
135
|
-
|
|
136
|
-
if (BOOLEANS[flag] !== undefined && inline === null) {
|
|
137
|
-
opts[BOOLEANS[flag]] = true;
|
|
138
|
-
continue;
|
|
139
|
-
}
|
|
140
|
-
|
|
141
|
-
const def = VALUES[flag];
|
|
142
|
-
if (def !== undefined) {
|
|
143
|
-
let raw;
|
|
144
|
-
if (inline !== null) {
|
|
145
|
-
raw = inline;
|
|
146
|
-
} else {
|
|
147
|
-
const next = argv[i + 1];
|
|
148
|
-
// A leading '-' normally means "the next flag, not my value" -- but a
|
|
149
|
-
// well-formed integer is a legitimate value for an int flag, so
|
|
150
|
-
// `--seed -5` and `--seed=-5` stay the same command.
|
|
151
|
-
const looksLikeFlag =
|
|
152
|
-
next !== undefined &&
|
|
153
|
-
next.startsWith('-') &&
|
|
154
|
-
!(def.type === 'int' && INT_RE.test(next));
|
|
155
|
-
if (next === undefined || looksLikeFlag) {
|
|
156
|
-
return fail(`argument ${flag}: expected one argument`);
|
|
157
|
-
}
|
|
158
|
-
raw = next;
|
|
159
|
-
i += 1;
|
|
160
|
-
}
|
|
161
|
-
// Empty is the missing-value case wearing a disguise. `--repo=` is
|
|
162
|
-
// typed by nobody, but `--repo "$UNSET_VAR"` expands to `--repo ''` in
|
|
163
|
-
// any shell, and an accepted empty path silently retargets writes at
|
|
164
|
-
// the process CWD.
|
|
165
|
-
if (raw === '') return fail(`argument ${flag}: expected one argument`);
|
|
166
|
-
if (def.type === 'int') {
|
|
167
|
-
// Validate the VALUE, not just its presence: a flag whose purpose is
|
|
168
|
-
// determinism must not decay to a default when its value is junk.
|
|
169
|
-
if (!INT_RE.test(raw)) {
|
|
170
|
-
return fail(`argument ${flag}: invalid int value: '${raw}'`);
|
|
171
|
-
}
|
|
172
|
-
// Syntactically an integer is not enough: Number() silently rounds
|
|
173
|
-
// past 2^53-1, so `--seed 9007199254740993` would RUN with ...992 --
|
|
174
|
-
// the value used differing from the value asked for, which is the
|
|
175
|
-
// exact class of lie a determinism flag must not tell.
|
|
176
|
-
const parsed = Number(raw);
|
|
177
|
-
if (!Number.isSafeInteger(parsed)) {
|
|
178
|
-
return fail(
|
|
179
|
-
`argument ${flag}: integer out of safe range: '${raw}'`,
|
|
180
|
-
);
|
|
181
|
-
}
|
|
182
|
-
opts[def.key] = parsed;
|
|
183
|
-
} else {
|
|
184
|
-
opts[def.key] = raw;
|
|
185
|
-
}
|
|
186
|
-
continue;
|
|
187
|
-
}
|
|
188
|
-
|
|
189
|
-
// A lone `-` is a positional, as argparse treats it.
|
|
190
|
-
if (positionals && (arg === '-' || !arg.startsWith('-'))) {
|
|
191
|
-
found.push(arg);
|
|
192
|
-
continue;
|
|
193
|
-
}
|
|
194
|
-
|
|
195
|
-
return fail(`unrecognized arguments: ${arg}`);
|
|
196
|
-
}
|
|
197
|
-
|
|
198
|
-
const missing = required.filter((flag) => {
|
|
199
|
-
const value = opts[VALUES[flag].key];
|
|
200
|
-
return value === null || value === undefined;
|
|
201
|
-
});
|
|
202
|
-
if (missing.length) {
|
|
203
|
-
return fail(
|
|
204
|
-
`the following arguments are required: ${missing.join(', ')}`,
|
|
205
|
-
);
|
|
206
|
-
}
|
|
207
|
-
|
|
208
|
-
if (positionals && !found.length && positionals.defaults) {
|
|
209
|
-
found.push(...positionals.defaults);
|
|
210
|
-
}
|
|
211
|
-
|
|
212
|
-
return result;
|
|
273
|
+
return parseArgv(config, argv);
|
|
213
274
|
};
|
|
214
275
|
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One issue, audited end to end (spec Phase 4).
|
|
3
|
+
*
|
|
4
|
+
* The composition every surface shares: resolve the change that closed the
|
|
5
|
+
* issue, probe each file it touched, and hand back the verdicts with the
|
|
6
|
+
* header they belong to. It lives here rather than in the CLI so that the
|
|
7
|
+
* order of operations -- and the rule that the changed-file set is the
|
|
8
|
+
* denominator -- is stated once, in the analysis layer, where the Phase 5 CI
|
|
9
|
+
* wrapper can reuse it without going through argv.
|
|
10
|
+
*
|
|
11
|
+
* **The two failure classes are deliberately different.** A closure that will
|
|
12
|
+
* not resolve throws: without it there is no changed-file set, so there is no
|
|
13
|
+
* denominator and nothing honest to report. A run history that will not read
|
|
14
|
+
* does not throw: the denominator is known, and "could not tell" is a verdict
|
|
15
|
+
* worth printing per file. Collapsing the second into the first would trade a
|
|
16
|
+
* report full of abstentions for no report at all.
|
|
17
|
+
*/
|
|
18
|
+
import { resolveClosure as defaultResolveClosure } from './closure.js';
|
|
19
|
+
import { ghRunHistory } from './gh-history.js';
|
|
20
|
+
import { shippedProbes } from './probes.js';
|
|
21
|
+
import { probeAll } from './registry.js';
|
|
22
|
+
/**
|
|
23
|
+
* Audit `issue` in `repo`, reading the working tree at `root` for static
|
|
24
|
+
* resolution (which workflow calls a script, what an `on:` block says).
|
|
25
|
+
*/
|
|
26
|
+
export async function auditIssue(issue, repo, root, deps = {}) {
|
|
27
|
+
const resolve = deps.resolveClosure ?? defaultResolveClosure;
|
|
28
|
+
const history = deps.runHistory ?? ((r) => ghRunHistory(r));
|
|
29
|
+
const closure = await resolve(issue, repo);
|
|
30
|
+
const verdicts = await probeAll(shippedProbes(), closure.files, {
|
|
31
|
+
mergedAt: closure.header.mergedAt,
|
|
32
|
+
repo,
|
|
33
|
+
runs: history(repo),
|
|
34
|
+
root,
|
|
35
|
+
deleted: closure.deleted,
|
|
36
|
+
});
|
|
37
|
+
return { closure, verdicts };
|
|
38
|
+
}
|
|
39
|
+
//# sourceMappingURL=audit.js.map
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolving an issue number to the change that closed it (spec Phase 4).
|
|
3
|
+
*
|
|
4
|
+
* GitHub closes an issue on a keyword match in a PR body. That is the fact
|
|
5
|
+
* batwoman exists to question, and this module is where the questioning starts:
|
|
6
|
+
* it finds the PR the keyword pointed at, when it merged, and every file it
|
|
7
|
+
* touched -- the denominator the whole report is measured against.
|
|
8
|
+
*
|
|
9
|
+
* **Nothing here ever degrades to an empty answer.** An issue with no merged
|
|
10
|
+
* closing PR, an unreachable `gh`, unparseable output, and a merged PR that
|
|
11
|
+
* reports no files at all are all thrown. An empty changed-file set would
|
|
12
|
+
* render as a clean report over a denominator of zero, which is a pass over
|
|
13
|
+
* nothing presented as a pass -- the exact defect batwoman detects, committed
|
|
14
|
+
* by batwoman.
|
|
15
|
+
*/
|
|
16
|
+
import { defaultSubprocess, } from '../../core/workflow-discovery.js';
|
|
17
|
+
/** How long to wait on one `gh` call, in seconds. */
|
|
18
|
+
const GH_TIMEOUT_SECONDS = 30;
|
|
19
|
+
/** Run `gh`, or throw with the reason. Never returns a degraded result. */
|
|
20
|
+
function gh(subprocess, args, what) {
|
|
21
|
+
// A throwing runner (a missing `gh`) propagates untouched: its message names
|
|
22
|
+
// the real cause, and the caller reports that rather than a guess.
|
|
23
|
+
const result = subprocess([...args], { timeout: GH_TIMEOUT_SECONDS });
|
|
24
|
+
if (result.returncode !== 0) {
|
|
25
|
+
throw new Error(`${what} failed (exit ${result.returncode}): ` +
|
|
26
|
+
`${result.stderr.trim() || 'no stderr'}`);
|
|
27
|
+
}
|
|
28
|
+
try {
|
|
29
|
+
return JSON.parse(result.stdout);
|
|
30
|
+
}
|
|
31
|
+
catch {
|
|
32
|
+
// Not treated as "nothing found". Unreadable output means the question
|
|
33
|
+
// went unanswered, and an unanswered question is not an empty answer.
|
|
34
|
+
throw new Error(`${what} returned unparseable JSON`);
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
/** The PR numbers GitHub records as having closed the issue. */
|
|
38
|
+
function closingPrNumbers(issue, doc) {
|
|
39
|
+
const refs = doc
|
|
40
|
+
?.closedByPullRequestsReferences;
|
|
41
|
+
const numbers = Array.isArray(refs)
|
|
42
|
+
? refs
|
|
43
|
+
.map((r) => r?.number)
|
|
44
|
+
.filter((n) => typeof n === 'number')
|
|
45
|
+
: [];
|
|
46
|
+
if (numbers.length === 0) {
|
|
47
|
+
throw new Error(`issue ${issue} has no closing pull request, so there is no change to ` +
|
|
48
|
+
'audit. It may have been closed by hand.');
|
|
49
|
+
}
|
|
50
|
+
return numbers;
|
|
51
|
+
}
|
|
52
|
+
/** One PR's merge facts, or null when it never merged. */
|
|
53
|
+
function readPr(subprocess, repo, number) {
|
|
54
|
+
const doc = gh(subprocess, [
|
|
55
|
+
'gh',
|
|
56
|
+
'pr',
|
|
57
|
+
'view',
|
|
58
|
+
'--repo',
|
|
59
|
+
repo,
|
|
60
|
+
String(number),
|
|
61
|
+
'--json',
|
|
62
|
+
'number,mergedAt,mergeCommit,title',
|
|
63
|
+
], `gh pr view ${number}`);
|
|
64
|
+
if (typeof doc.mergedAt !== 'string')
|
|
65
|
+
return null;
|
|
66
|
+
const mergedAt = new Date(doc.mergedAt);
|
|
67
|
+
if (Number.isNaN(mergedAt.getTime()))
|
|
68
|
+
return null;
|
|
69
|
+
const oid = doc.mergeCommit?.oid;
|
|
70
|
+
return {
|
|
71
|
+
number,
|
|
72
|
+
mergedAt,
|
|
73
|
+
mergeSha: typeof oid === 'string' ? oid : '(unknown)',
|
|
74
|
+
subject: typeof doc.title === 'string' ? doc.title : '(no subject)',
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
/** The changed paths and the deleted subset, from the files endpoint. */
|
|
78
|
+
function readFiles(subprocess, repo, number) {
|
|
79
|
+
const payload = gh(subprocess, [
|
|
80
|
+
'gh',
|
|
81
|
+
'api',
|
|
82
|
+
`repos/${repo}/pulls/${number}/files`,
|
|
83
|
+
'--paginate',
|
|
84
|
+
'--slurp',
|
|
85
|
+
], `gh api pulls/${number}/files`);
|
|
86
|
+
if (!Array.isArray(payload)) {
|
|
87
|
+
throw new Error(`gh api pulls/${number}/files did not return a list`);
|
|
88
|
+
}
|
|
89
|
+
// `--paginate --slurp` yields an array of PAGES, each an array of rows --
|
|
90
|
+
// verified against the live endpoint, where a single page still arrives
|
|
91
|
+
// wrapped. `--jq` cannot do the flattening here: gh rejects `--slurp`
|
|
92
|
+
// together with `--jq`, which a fixture-only test cannot discover because
|
|
93
|
+
// the fixture never runs gh. One level is flattened, and an already-flat
|
|
94
|
+
// list is accepted so the shape is not load-bearing.
|
|
95
|
+
const rows = payload.every((page) => Array.isArray(page))
|
|
96
|
+
? payload.flat()
|
|
97
|
+
: payload;
|
|
98
|
+
const files = [];
|
|
99
|
+
const deleted = new Set();
|
|
100
|
+
for (const row of rows) {
|
|
101
|
+
if (typeof row?.filename !== 'string')
|
|
102
|
+
continue;
|
|
103
|
+
files.push(row.filename);
|
|
104
|
+
// `status` is the only trustworthy deletion signal. An additions count of
|
|
105
|
+
// zero cannot mean deleted: a file gutted to an empty stub looks identical
|
|
106
|
+
// and is very much still there to run.
|
|
107
|
+
if (row.status === 'removed')
|
|
108
|
+
deleted.add(row.filename);
|
|
109
|
+
}
|
|
110
|
+
if (files.length === 0) {
|
|
111
|
+
throw new Error(`pull request ${number} reports no changed files, which cannot be ` +
|
|
112
|
+
'right for a merged change. Refusing to report over an empty ' +
|
|
113
|
+
'denominator.');
|
|
114
|
+
}
|
|
115
|
+
return { files, deleted };
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Resolve `issue` to the merged change that closed it.
|
|
119
|
+
*
|
|
120
|
+
* When more than one PR is recorded as closing the issue -- GitHub permits it
|
|
121
|
+
* -- the most recently merged one wins. Picking by array order would make the
|
|
122
|
+
* report depend on the order GitHub happened to return, and the latest merge
|
|
123
|
+
* is the one whose code is actually standing in the tree.
|
|
124
|
+
*/
|
|
125
|
+
export async function resolveClosure(issue, repo, subprocess = defaultSubprocess) {
|
|
126
|
+
const doc = gh(subprocess, [
|
|
127
|
+
'gh',
|
|
128
|
+
'issue',
|
|
129
|
+
'view',
|
|
130
|
+
'--repo',
|
|
131
|
+
repo,
|
|
132
|
+
String(issue),
|
|
133
|
+
'--json',
|
|
134
|
+
'number,closedByPullRequestsReferences',
|
|
135
|
+
], `gh issue view ${issue}`);
|
|
136
|
+
const candidates = closingPrNumbers(issue, doc);
|
|
137
|
+
const merged = candidates
|
|
138
|
+
.map((n) => readPr(subprocess, repo, n))
|
|
139
|
+
.filter((pr) => pr !== null)
|
|
140
|
+
.sort((a, b) => b.mergedAt.getTime() - a.mergedAt.getTime());
|
|
141
|
+
const chosen = merged[0];
|
|
142
|
+
if (chosen === undefined) {
|
|
143
|
+
throw new Error(`issue ${issue} is closed, but its closing pull request ` +
|
|
144
|
+
`(${candidates.join(', ')}) was not merged, so nothing shipped to audit.`);
|
|
145
|
+
}
|
|
146
|
+
const { files, deleted } = readFiles(subprocess, repo, chosen.number);
|
|
147
|
+
return {
|
|
148
|
+
header: {
|
|
149
|
+
issue,
|
|
150
|
+
mergeSha: chosen.mergeSha,
|
|
151
|
+
mergeSubject: chosen.subject,
|
|
152
|
+
mergedAt: chosen.mergedAt,
|
|
153
|
+
},
|
|
154
|
+
pullRequest: chosen.number,
|
|
155
|
+
files,
|
|
156
|
+
deleted,
|
|
157
|
+
};
|
|
158
|
+
}
|
|
159
|
+
//# sourceMappingURL=closure.js.map
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `RunHistoryPort` over the `gh` CLI (spec Phase 3, criterion 4).
|
|
3
|
+
*
|
|
4
|
+
* This is batwoman's only outward call, and therefore the only place where
|
|
5
|
+
* "I could not find out" can be mistaken for "it did not happen". Every
|
|
6
|
+
* failure path here is loud: a missing `gh`, a non-zero exit, and output that
|
|
7
|
+
* will not parse all throw, because the registry turns a throw into `abstain`
|
|
8
|
+
* and an abstention is the honest report. The one thing this module must never
|
|
9
|
+
* do is return an empty history for a question it failed to ask -- that would
|
|
10
|
+
* read downstream as "this workflow has never run", which is the false green
|
|
11
|
+
* batwoman was built to catch.
|
|
12
|
+
*
|
|
13
|
+
* The subprocess is injected through the same {@link SubprocessRun} seam
|
|
14
|
+
* `ticket-updater.ts` uses, so tests never shell out.
|
|
15
|
+
*/
|
|
16
|
+
import { defaultSubprocess, } from '../../core/workflow-discovery.js';
|
|
17
|
+
/**
|
|
18
|
+
* How many runs to request per workflow.
|
|
19
|
+
*
|
|
20
|
+
* Explicit on purpose. `gh run list` defaults to 20 (the spec says 30; the CLI
|
|
21
|
+
* says otherwise) and reports nothing about the window it used, so a repo with
|
|
22
|
+
* a busy workflow would answer from a window of days and say so nowhere.
|
|
23
|
+
* Naming the number here makes the window a stated fact, and a page that fills
|
|
24
|
+
* to exactly this many rows is what {@link RunHistory.complete} reports as
|
|
25
|
+
* truncated.
|
|
26
|
+
*/
|
|
27
|
+
export const GH_RUN_LIMIT = 100;
|
|
28
|
+
/** How long to wait on one `gh` call, in seconds. */
|
|
29
|
+
const GH_TIMEOUT_SECONDS = 30;
|
|
30
|
+
/**
|
|
31
|
+
* One row to a {@link WorkflowRun}, or null when it cannot be placed in time.
|
|
32
|
+
*
|
|
33
|
+
* A row with no usable `createdAt` is dropped rather than defaulted: a run
|
|
34
|
+
* with a guessed timestamp could land on either side of the merge and decide
|
|
35
|
+
* the verdict. Dropping the row loses one piece of evidence; keeping it with
|
|
36
|
+
* an invented date would manufacture one.
|
|
37
|
+
*/
|
|
38
|
+
function toRun(row) {
|
|
39
|
+
if (typeof row.createdAt !== 'string')
|
|
40
|
+
return null;
|
|
41
|
+
const createdAt = new Date(row.createdAt);
|
|
42
|
+
if (Number.isNaN(createdAt.getTime()))
|
|
43
|
+
return null;
|
|
44
|
+
return {
|
|
45
|
+
createdAt,
|
|
46
|
+
conclusion: typeof row.conclusion === 'string' ? row.conclusion : null,
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
/** The argv for one workflow's history. Kept in one place so tests can pin it. */
|
|
50
|
+
function ghArgs(repo, workflowPath) {
|
|
51
|
+
return [
|
|
52
|
+
'gh',
|
|
53
|
+
'run',
|
|
54
|
+
'list',
|
|
55
|
+
'--repo',
|
|
56
|
+
repo,
|
|
57
|
+
'--workflow',
|
|
58
|
+
workflowPath,
|
|
59
|
+
'--limit',
|
|
60
|
+
String(GH_RUN_LIMIT),
|
|
61
|
+
'--json',
|
|
62
|
+
'createdAt,conclusion',
|
|
63
|
+
];
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* A `RunHistoryPort` that reads real run history through `gh`.
|
|
67
|
+
*
|
|
68
|
+
* @param repo `owner/name`, passed explicitly so the port does not depend on
|
|
69
|
+
* the process's working directory.
|
|
70
|
+
* @param subprocess injected for tests; defaults to the shared `spawnSync`
|
|
71
|
+
* runner.
|
|
72
|
+
*/
|
|
73
|
+
export function ghRunHistory(repo, subprocess = defaultSubprocess) {
|
|
74
|
+
return {
|
|
75
|
+
// `async` so that a synchronous failure below -- a missing `gh` throws
|
|
76
|
+
// straight out of spawnSync -- reaches the caller as a REJECTED promise
|
|
77
|
+
// rather than as a throw during port construction. The registry catches
|
|
78
|
+
// rejections to record an abstention; a synchronous throw would escape it.
|
|
79
|
+
async runsForWorkflow(workflowPath) {
|
|
80
|
+
return fetchHistory(repo, workflowPath, subprocess);
|
|
81
|
+
},
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
function fetchHistory(repo, workflowPath, subprocess) {
|
|
85
|
+
// A throwing runner (a missing `gh` is the common one) is left to propagate
|
|
86
|
+
// untouched: its message names the real cause, and the registry will record
|
|
87
|
+
// the abstention with that message attached.
|
|
88
|
+
const result = subprocess(ghArgs(repo, workflowPath), {
|
|
89
|
+
timeout: GH_TIMEOUT_SECONDS,
|
|
90
|
+
});
|
|
91
|
+
if (result.returncode !== 0) {
|
|
92
|
+
throw new Error(`gh run list failed for ${workflowPath} (exit ${result.returncode}): ` +
|
|
93
|
+
`${result.stderr.trim() || 'no stderr'}`);
|
|
94
|
+
}
|
|
95
|
+
let parsed;
|
|
96
|
+
try {
|
|
97
|
+
parsed = JSON.parse(result.stdout);
|
|
98
|
+
}
|
|
99
|
+
catch {
|
|
100
|
+
// Deliberately not treated as "no runs". Unreadable output means the
|
|
101
|
+
// question went unanswered, and an unanswered question is an abstention.
|
|
102
|
+
throw new Error(`gh run list returned unparseable JSON for ${workflowPath}`);
|
|
103
|
+
}
|
|
104
|
+
if (!Array.isArray(parsed)) {
|
|
105
|
+
throw new Error(`gh run list returned ${typeof parsed}, not a list, for ${workflowPath}`);
|
|
106
|
+
}
|
|
107
|
+
const runs = parsed
|
|
108
|
+
.map((row) => toRun((row ?? {})))
|
|
109
|
+
.filter((run) => run !== null);
|
|
110
|
+
return {
|
|
111
|
+
runs,
|
|
112
|
+
// Measured against the ROWS gh returned, not against the runs that
|
|
113
|
+
// survived parsing. A dropped malformed row still proves the page was
|
|
114
|
+
// full, and counting the survivors instead would report a truncated
|
|
115
|
+
// window as complete -- turning a dropped row into a false verdict.
|
|
116
|
+
complete: parsed.length < GH_RUN_LIMIT,
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
//# sourceMappingURL=gh-history.js.map
|