canary-test-cli 7.2.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/skills/README.md +23 -4
- package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
- package/agents/skills/claude-code/canary-cassandra/SKILL.md +23 -16
- package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +3 -1
- package/agents/skills/claude-code/canary-ci-ready/SKILL.md +20 -3
- package/agents/skills/claude-code/canary-fleet-health/SKILL.md +1 -0
- package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +15 -0
- package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
- package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
- package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
- package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
- package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
- package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
- package/agents/skills/lib/parse-args.mjs +200 -139
- package/dist/engine/analysis/batwoman/audit.js +39 -0
- package/dist/engine/analysis/batwoman/closure.js +159 -0
- package/dist/engine/analysis/batwoman/gh-history.js +119 -0
- package/dist/engine/analysis/batwoman/probes.js +195 -0
- package/dist/engine/analysis/batwoman/registry.js +142 -0
- package/dist/engine/analysis/batwoman/render.js +194 -0
- package/dist/engine/analysis/batwoman/run-window.js +122 -0
- package/dist/engine/analysis/batwoman/text.js +84 -0
- package/dist/engine/analysis/batwoman/triggers.js +122 -0
- package/dist/engine/analysis/batwoman/verdict.js +64 -0
- package/dist/engine/analysis/cli.js +47 -14
- package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
- package/dist/engine/batwoman-cli.js +119 -0
- package/dist/engine/ci-ready-cli.js +71 -0
- package/dist/engine/cli-commands.js +46 -7
- package/dist/engine/cli.core.js +16 -0
- package/dist/engine/company-knowledge-cli.js +10 -2
- package/dist/engine/core/ci-ready.js +112 -0
- package/dist/engine/core/company-knowledge.js +8 -0
- package/dist/engine/core/migrator.js +147 -20
- package/dist/engine/core/permission-matrix.js +219 -0
- package/dist/engine/core/quality-scorer.js +13 -18
- package/dist/engine/core/scaling-curve.js +143 -0
- package/dist/engine/core/string-literals.js +3 -1
- package/dist/engine/core/vacuity-scanner.js +151 -6
- package/dist/engine/core/workflow-discovery.js +41 -23
- package/dist/engine/guardian/adjudication-github.js +136 -0
- package/dist/engine/guardian/adjudication.js +119 -340
- package/dist/engine/guardian/cli.js +180 -264
- package/dist/engine/guardian/coverage.js +2 -1
- package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
- package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
- package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
- package/dist/engine/guardian/diff-coverage/paths.js +5 -9
- package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
- package/dist/engine/guardian/diff-extractor.js +31 -32
- package/dist/engine/guardian/pr-check.js +262 -430
- package/dist/engine/guardian/pr-comment.js +35 -58
- package/dist/engine/guardian/weak-test.js +236 -0
- package/dist/engine/mcp-server.js +67 -4
- package/dist/engine/permission-matrix-cli.js +51 -0
- package/dist/engine/scaling-curve-cli.js +147 -0
- package/dist/engine/skills-cli.js +48 -32
- package/dist/engine/workflow-cli.js +85 -65
- package/package.json +1 -1
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text layout for the batwoman report.
|
|
3
|
+
*
|
|
4
|
+
* These are the report's typographic primitives -- how prose is broken to a
|
|
5
|
+
* width and how an assembled row is fitted to one. They are separated from
|
|
6
|
+
* `render.ts` because they know nothing about verdicts, registers or sections:
|
|
7
|
+
* they take strings and widths and return strings. Composition lives next
|
|
8
|
+
* door; measuring and breaking lines lives here.
|
|
9
|
+
*/
|
|
10
|
+
/** `2026-08-22 17:34 UTC`, stable across the runner's timezone. */
|
|
11
|
+
export function stamp(when) {
|
|
12
|
+
return `${when.toISOString().slice(0, 16).replace('T', ' ')} UTC`;
|
|
13
|
+
}
|
|
14
|
+
/** The report's column limit. Header and body wrap to the same width. */
|
|
15
|
+
export const WIDTH = 78;
|
|
16
|
+
/**
|
|
17
|
+
* Greedy word wrap with a fixed indent. A too-long word gets its own line.
|
|
18
|
+
*
|
|
19
|
+
* **Explanations are single-paragraph prose.** `split(/\s+/)` normalises all
|
|
20
|
+
* internal whitespace, so a newline becomes a space and a run of spaces becomes
|
|
21
|
+
* one. That is a decision, not an accident, and it is recorded here because
|
|
22
|
+
* the probes compose explanations out of evidence, and the temptation
|
|
23
|
+
* to embed a command, a YAML fragment or an indented log line is real. A verdict
|
|
24
|
+
* sentence is a sentence: the thing a probe wants to quote belongs in
|
|
25
|
+
* `evidence`, which {@link fitLine} wraps without reflowing it, or in a future
|
|
26
|
+
* field that declares itself preformatted. Reflowing a YAML fragment silently
|
|
27
|
+
* would be worse than refusing it, so if that need arrives, add the field --
|
|
28
|
+
* do not relax this.
|
|
29
|
+
*/
|
|
30
|
+
export function wrap(text, width, indent) {
|
|
31
|
+
const lines = [];
|
|
32
|
+
let current = '';
|
|
33
|
+
for (const word of text.split(/\s+/).filter((w) => w !== '')) {
|
|
34
|
+
const candidate = current === '' ? word : `${current} ${word}`;
|
|
35
|
+
if (indent.length + candidate.length <= width || current === '') {
|
|
36
|
+
current = candidate;
|
|
37
|
+
}
|
|
38
|
+
else {
|
|
39
|
+
lines.push(indent + current);
|
|
40
|
+
current = word;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
if (current !== '')
|
|
44
|
+
lines.push(indent + current);
|
|
45
|
+
return lines;
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* Fit an already-assembled line to the width, breaking it only if it overflows.
|
|
49
|
+
*
|
|
50
|
+
* `wrap` cannot do this job: a file path is a single whitespace-free token, so
|
|
51
|
+
* `wrap` would faithfully put all 86 columns of it on one line. This breaks at a
|
|
52
|
+
* space when there is one inside the remaining room and chops the token when
|
|
53
|
+
* there is not, which is the only way a path fits at all.
|
|
54
|
+
*
|
|
55
|
+
* A line that already fits is returned untouched, so every short row keeps its
|
|
56
|
+
* exact spacing -- including the two spaces the terse register puts between a
|
|
57
|
+
* path and its status.
|
|
58
|
+
*/
|
|
59
|
+
export function fitLine(line, width, continuation) {
|
|
60
|
+
if (line.length <= width)
|
|
61
|
+
return [line];
|
|
62
|
+
const lines = [];
|
|
63
|
+
let rest = line;
|
|
64
|
+
let indent = '';
|
|
65
|
+
while (indent.length + rest.length > width && rest !== '') {
|
|
66
|
+
const room = width - indent.length;
|
|
67
|
+
const space = rest.lastIndexOf(' ', room);
|
|
68
|
+
// A space break is only taken when it fills at least half the line. Every
|
|
69
|
+
// row here starts with a marker -- two spaces of indent, ` - `, ` 1. `
|
|
70
|
+
// -- and the last space *within the room* of a long path is the one right
|
|
71
|
+
// after that marker. Breaking there emitted ` -` and ` 1.` alone on a
|
|
72
|
+
// line, and for a plain indented path an empty one. A path with no spaces
|
|
73
|
+
// in it has to be chopped somewhere; chopping it at the width reads better
|
|
74
|
+
// than orphaning its bullet.
|
|
75
|
+
const cut = space * 2 >= room ? space : room;
|
|
76
|
+
lines.push((indent + rest.slice(0, cut)).trimEnd());
|
|
77
|
+
rest = rest.slice(cut).trimStart();
|
|
78
|
+
indent = continuation;
|
|
79
|
+
}
|
|
80
|
+
if (rest !== '')
|
|
81
|
+
lines.push(indent + rest);
|
|
82
|
+
return lines;
|
|
83
|
+
}
|
|
84
|
+
//# sourceMappingURL=text.js.map
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Turning a workflow's `on:` block into the clause that explains its dormancy.
|
|
3
|
+
*
|
|
4
|
+
* This exists because of the founding case. #749 fixed
|
|
5
|
+
* `refresh-arch-baseline.yml`, and that workflow had not run in twelve days --
|
|
6
|
+
* not because anything was broken, but because it is gated on a label nobody
|
|
7
|
+
* had applied. A report that says only "has not run" sends its reader to look
|
|
8
|
+
* for a break that is not there, and the time they lose is time batwoman
|
|
9
|
+
* caused. The trigger is the difference between a fact and an explanation.
|
|
10
|
+
*
|
|
11
|
+
* **Unrecognised shapes return `null`, never a guess.** The probe degrades to
|
|
12
|
+
* the bare run-history sentence when that happens. Inventing a plausible
|
|
13
|
+
* trigger would put a fabricated cause into a human's report, which is a worse
|
|
14
|
+
* failure than saying less: a wrong cause is acted on, silence is asked about.
|
|
15
|
+
*/
|
|
16
|
+
/** Trigger names worth explaining, mapped to the clause each contributes. */
|
|
17
|
+
const SIMPLE_CLAUSES = {
|
|
18
|
+
push: 'on push',
|
|
19
|
+
pull_request: 'on a pull request',
|
|
20
|
+
pull_request_target: 'on a pull request',
|
|
21
|
+
workflow_dispatch: 'manually',
|
|
22
|
+
workflow_call: 'when another workflow calls it',
|
|
23
|
+
schedule: 'on a schedule',
|
|
24
|
+
release: 'on a release',
|
|
25
|
+
issues: 'on issue activity',
|
|
26
|
+
issue_comment: 'on an issue comment',
|
|
27
|
+
};
|
|
28
|
+
/**
|
|
29
|
+
* Triggers that mean "this sits still until a human does something specific".
|
|
30
|
+
*
|
|
31
|
+
* Only these earn the word "only", and only when they stand alone: a workflow
|
|
32
|
+
* that also runs on push is not dormant, and calling it dormant would be the
|
|
33
|
+
* same fabrication this module refuses elsewhere.
|
|
34
|
+
*/
|
|
35
|
+
const DORMANT_ALONE = new Set(['workflow_dispatch', 'pull_request_labeled']);
|
|
36
|
+
/** The `on:` block's keys, from either the mapping or the bare-list form. */
|
|
37
|
+
function triggerNames(on) {
|
|
38
|
+
if (Array.isArray(on))
|
|
39
|
+
return on.filter((n) => typeof n === 'string');
|
|
40
|
+
if (typeof on === 'object' && on !== null)
|
|
41
|
+
return Object.keys(on);
|
|
42
|
+
return [];
|
|
43
|
+
}
|
|
44
|
+
/** The config a trigger carries, or undefined for the bare-list form. */
|
|
45
|
+
function configFor(on, name) {
|
|
46
|
+
if (typeof on === 'object' && on !== null && !Array.isArray(on)) {
|
|
47
|
+
return on[name];
|
|
48
|
+
}
|
|
49
|
+
return undefined;
|
|
50
|
+
}
|
|
51
|
+
/** `{ types: ['labeled'] }` -> true. The shape #749's workflow uses. */
|
|
52
|
+
function isLabelGated(config) {
|
|
53
|
+
if (typeof config !== 'object' || config === null)
|
|
54
|
+
return false;
|
|
55
|
+
const types = config.types;
|
|
56
|
+
return Array.isArray(types) && types.length === 1 && types[0] === 'labeled';
|
|
57
|
+
}
|
|
58
|
+
/** `{ branches: ['main'] }` -> 'main'. Multiple branches are not enumerated. */
|
|
59
|
+
function branchClause(config) {
|
|
60
|
+
if (typeof config !== 'object' || config === null)
|
|
61
|
+
return '';
|
|
62
|
+
const branches = config.branches;
|
|
63
|
+
if (!Array.isArray(branches) || branches.length !== 1)
|
|
64
|
+
return '';
|
|
65
|
+
const only = branches[0];
|
|
66
|
+
return typeof only === 'string' ? ` to ${only}` : '';
|
|
67
|
+
}
|
|
68
|
+
/** One trigger's clause, or null when it is not one we explain. */
|
|
69
|
+
function clauseFor(name, config) {
|
|
70
|
+
if (name === 'pull_request' && isLabelGated(config)) {
|
|
71
|
+
return 'when a label is added to a pull request';
|
|
72
|
+
}
|
|
73
|
+
const base = SIMPLE_CLAUSES[name];
|
|
74
|
+
if (base === undefined)
|
|
75
|
+
return null;
|
|
76
|
+
return name === 'push' ? `${base}${branchClause(config)}` : base;
|
|
77
|
+
}
|
|
78
|
+
/** The key a clause was derived from, for the dormancy test. */
|
|
79
|
+
function dormancyKey(name, config) {
|
|
80
|
+
return name === 'pull_request' && isLabelGated(config)
|
|
81
|
+
? 'pull_request_labeled'
|
|
82
|
+
: name;
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* A human clause naming why a workflow runs -- `triggered on push to main`,
|
|
86
|
+
* `triggered only when a label is added to a pull request`.
|
|
87
|
+
*
|
|
88
|
+
* Returns `null` when the block is absent, malformed, or names only events
|
|
89
|
+
* this module has no wording for. The caller must handle null by saying less,
|
|
90
|
+
* not by filling in.
|
|
91
|
+
*/
|
|
92
|
+
export function describeTriggers(on) {
|
|
93
|
+
const names = triggerNames(on);
|
|
94
|
+
if (names.length === 0)
|
|
95
|
+
return null;
|
|
96
|
+
const clauses = [];
|
|
97
|
+
const keys = [];
|
|
98
|
+
// Manual dispatch reads last regardless of where it sat in the YAML. Key
|
|
99
|
+
// order is an authoring accident -- the same workflow written two ways would
|
|
100
|
+
// otherwise produce two different sentences -- and "or manually" is the
|
|
101
|
+
// afterthought clause in English anyway.
|
|
102
|
+
const ordered = [...names].sort((a, b) => Number(a === 'workflow_dispatch') - Number(b === 'workflow_dispatch'));
|
|
103
|
+
for (const name of ordered) {
|
|
104
|
+
const config = configFor(on, name);
|
|
105
|
+
const clause = clauseFor(name, config);
|
|
106
|
+
// An unrecognised event is skipped rather than echoed: printing
|
|
107
|
+
// `triggered on some_future_event` would read as an explanation while
|
|
108
|
+
// telling the reader nothing they did not already see in the file.
|
|
109
|
+
if (clause === null)
|
|
110
|
+
continue;
|
|
111
|
+
clauses.push(clause);
|
|
112
|
+
keys.push(dormancyKey(name, config));
|
|
113
|
+
}
|
|
114
|
+
if (clauses.length === 0)
|
|
115
|
+
return null;
|
|
116
|
+
const only = clauses.length === 1 && keys[0] !== undefined && DORMANT_ALONE.has(keys[0]);
|
|
117
|
+
const joined = clauses.length === 1
|
|
118
|
+
? clauses[0]
|
|
119
|
+
: `${clauses.slice(0, -1).join(', ')}, or ${clauses[clauses.length - 1]}`;
|
|
120
|
+
return `triggered ${only ? 'only ' : ''}${joined}`;
|
|
121
|
+
}
|
|
122
|
+
//# sourceMappingURL=triggers.js.map
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* canary-batwoman's verdict model (spec `docs/changes/canary-batwoman/`).
|
|
3
|
+
*
|
|
4
|
+
* Batwoman answers one question per changed file -- has this artifact executed
|
|
5
|
+
* since the fix merged -- and answers with one of five values, never with a
|
|
6
|
+
* confidence score. A score attached to a guess reads as evidence; "I have no
|
|
7
|
+
* probe for this" does not (spec D4).
|
|
8
|
+
*
|
|
9
|
+
* `abstain` and `no-probe` are deliberately separate. The first means a probe
|
|
10
|
+
* looked and could not tell: report it, investigate it. The second means
|
|
11
|
+
* nothing looked: a registry gap, fixable by adding a probe. Collapsing them
|
|
12
|
+
* would hide which of the two a given file suffers from (spec D3).
|
|
13
|
+
*/
|
|
14
|
+
/** The five statuses, in report order. */
|
|
15
|
+
export const EXERCISE_STATUSES = [
|
|
16
|
+
'exercised',
|
|
17
|
+
'not-exercised',
|
|
18
|
+
'abstain',
|
|
19
|
+
'no-probe',
|
|
20
|
+
'not-applicable',
|
|
21
|
+
];
|
|
22
|
+
/** The two statuses that make a positive claim about whether the file ran. */
|
|
23
|
+
export const CLAIMING_STATUSES = ['exercised', 'not-exercised'];
|
|
24
|
+
/** Thrown by {@link explain}. Named so a probe's failure reads as its own. */
|
|
25
|
+
export class EmptyExplanationError extends Error {
|
|
26
|
+
constructor() {
|
|
27
|
+
super('an ExerciseVerdict explanation must be a full sentence, never empty: ' +
|
|
28
|
+
'a row that names a file and says nothing about it is the silence ' +
|
|
29
|
+
'batwoman exists to remove');
|
|
30
|
+
this.name = 'EmptyExplanationError';
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
/** True when `text` could be an {@link Explanation}. Whitespace is not a sentence. */
|
|
34
|
+
export function isExplanation(text) {
|
|
35
|
+
return typeof text === 'string' && text.trim() !== '';
|
|
36
|
+
}
|
|
37
|
+
/** The only constructor for an {@link Explanation}. Rejects the empty string. */
|
|
38
|
+
export function explain(text) {
|
|
39
|
+
if (!isExplanation(text))
|
|
40
|
+
throw new EmptyExplanationError();
|
|
41
|
+
return text;
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Count verdicts by status.
|
|
45
|
+
*
|
|
46
|
+
* There is deliberately no `assessed` field. A derived "files batwoman could
|
|
47
|
+
* decide about" figure is exactly the shape spec criterion 3 forbids: it makes
|
|
48
|
+
* `abstain` and `no-probe` disappear into a denominator that looks like
|
|
49
|
+
* coverage. A caller wanting a subtotal has to write the addition itself, in
|
|
50
|
+
* the open, where a reviewer can see which statuses it folded.
|
|
51
|
+
*/
|
|
52
|
+
export function tallyVerdicts(verdicts) {
|
|
53
|
+
const byStatus = {
|
|
54
|
+
exercised: 0,
|
|
55
|
+
'not-exercised': 0,
|
|
56
|
+
abstain: 0,
|
|
57
|
+
'no-probe': 0,
|
|
58
|
+
'not-applicable': 0,
|
|
59
|
+
};
|
|
60
|
+
for (const verdict of verdicts)
|
|
61
|
+
byStatus[verdict.status] += 1;
|
|
62
|
+
return { changed: verdicts.length, byStatus };
|
|
63
|
+
}
|
|
64
|
+
//# sourceMappingURL=verdict.js.map
|
|
@@ -6,8 +6,10 @@
|
|
|
6
6
|
*
|
|
7
7
|
* Follows the guardian CLI conventions (see `../cli-common.ts`): a
|
|
8
8
|
* {@link createAnalyzeCommand} factory wired to an injectable {@link AnalyzeDeps},
|
|
9
|
-
* and `normalizeUsageExit` on every command so usage errors exit 2.
|
|
10
|
-
*
|
|
9
|
+
* and `normalizeUsageExit` on every command so usage errors exit 2. The
|
|
10
|
+
* history-backed subcommands raise no business exit and return 0 (matching
|
|
11
|
+
* Python). `gh-flaky` (#884) has no Python original and follows the CLI-wide
|
|
12
|
+
* gate contract: 1 for candidates, 3 when abstained or unverifiable.
|
|
11
13
|
*
|
|
12
14
|
* Python->TS fidelity notes:
|
|
13
15
|
* - `json.dumps(x, indent=2)` -> {@link jsonIndent2} (byte-exact + ensure_ascii).
|
|
@@ -26,9 +28,11 @@
|
|
|
26
28
|
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
27
29
|
import { join } from 'node:path';
|
|
28
30
|
import { Command, InvalidArgumentError, Option } from 'commander';
|
|
29
|
-
import { jsonIndent2, normalizeUsageExit } from '../cli-common.js';
|
|
31
|
+
import { CliExitError, jsonIndent2, normalizeUsageExit, } from '../cli-common.js';
|
|
30
32
|
import { gateOutcome } from '../core/gate-result.js';
|
|
33
|
+
import { defaultSubprocess, } from '../core/workflow-discovery.js';
|
|
31
34
|
import { AnalysisEngine } from './engine.js';
|
|
35
|
+
import { GH_FLAKY_RUN_LIMIT, ghFlakyExitCode, renderGhFlaky, scanGhFlaky, } from './gh-flaky/gh-run-attempts.js';
|
|
32
36
|
import { buildCommonFailuresReport, buildFlakyTestsReport, buildRegressionCandidatesReport, buildFailureSpikesReport, } from './reports.js';
|
|
33
37
|
import { makeStore } from '../history/store.js';
|
|
34
38
|
const DEFAULT_HISTORY_PATH = 'test-results/reports/history-v2.jsonl';
|
|
@@ -45,6 +49,7 @@ export function defaultAnalyzeDeps() {
|
|
|
45
49
|
// Where a remote backend cannot answer a given section, the command says so
|
|
46
50
|
// by name (see `cannotVerifyRawRecords`) rather than rendering an empty one.
|
|
47
51
|
makeStore: (dbUrl) => makeStore(dbUrl, DEFAULT_HISTORY_PATH),
|
|
52
|
+
runGh: defaultSubprocess,
|
|
48
53
|
};
|
|
49
54
|
}
|
|
50
55
|
/**
|
|
@@ -296,17 +301,7 @@ async function commonFailuresCmd(opts, deps) {
|
|
|
296
301
|
for (const record of await store.readAll()) {
|
|
297
302
|
if (opts.since && (record.timestamp ?? '') < opts.since)
|
|
298
303
|
continue;
|
|
299
|
-
|
|
300
|
-
if ((t.status === 'failed' || t.status === 'flaky') && t.error_text) {
|
|
301
|
-
rows.push({
|
|
302
|
-
test_name: t.test_name,
|
|
303
|
-
suite: record.suite ?? '',
|
|
304
|
-
failure_category: t.failure_category ?? 'other',
|
|
305
|
-
error_text: t.error_text ?? '',
|
|
306
|
-
run_count: 1,
|
|
307
|
-
});
|
|
308
|
-
}
|
|
309
|
-
}
|
|
304
|
+
rows.push(...failureRowsOf(record));
|
|
310
305
|
}
|
|
311
306
|
if (opts.json) {
|
|
312
307
|
deps.out(jsonIndent2(rows));
|
|
@@ -315,6 +310,32 @@ async function commonFailuresCmd(opts, deps) {
|
|
|
315
310
|
deps.out(buildCommonFailuresReport(rows, opts.minSuites));
|
|
316
311
|
}
|
|
317
312
|
}
|
|
313
|
+
/** One run record's failed/flaky tests as rows (split out for perf, #884). */
|
|
314
|
+
function failureRowsOf(record) {
|
|
315
|
+
const suite = record.suite ?? '';
|
|
316
|
+
return (record.tests ?? [])
|
|
317
|
+
.filter((t) => (t.status === 'failed' || t.status === 'flaky') && t.error_text)
|
|
318
|
+
.map((t) => ({
|
|
319
|
+
test_name: t.test_name,
|
|
320
|
+
suite,
|
|
321
|
+
failure_category: t.failure_category ?? 'other',
|
|
322
|
+
error_text: t.error_text ?? '',
|
|
323
|
+
run_count: 1,
|
|
324
|
+
}));
|
|
325
|
+
}
|
|
326
|
+
/** Gate-shaped: exit 1 on candidates, 3 when abstained or unverifiable. */
|
|
327
|
+
function ghFlakyCmd(opts, deps) {
|
|
328
|
+
const report = scanGhFlaky(opts.repo, opts.limitRuns, deps.runGh);
|
|
329
|
+
if (opts.json) {
|
|
330
|
+
deps.out(jsonIndent2(report));
|
|
331
|
+
}
|
|
332
|
+
else {
|
|
333
|
+
deps.out(renderGhFlaky(report).join('\n'));
|
|
334
|
+
}
|
|
335
|
+
const code = ghFlakyExitCode(report);
|
|
336
|
+
if (code !== 0)
|
|
337
|
+
throw new CliExitError(code);
|
|
338
|
+
}
|
|
318
339
|
async function regressionCandidatesCmd(opts, deps) {
|
|
319
340
|
const store = deps.makeStore(opts.dbUrl);
|
|
320
341
|
if (await abstainOnEmptyHistory(store, deps, opts.json === true, 'regression candidates')) {
|
|
@@ -433,6 +454,18 @@ export function createAnalyzeCommand(depsInit = {}) {
|
|
|
433
454
|
.action(async (opts, cmd) => {
|
|
434
455
|
await spikesCmd(resolveUnitFlags(opts, cmd, deps, [DELTA_ALIAS]), deps);
|
|
435
456
|
});
|
|
457
|
+
program
|
|
458
|
+
.command('gh-flaky')
|
|
459
|
+
.description('Flake signals from GitHub Actions runs: same-SHA outcome flips and ' +
|
|
460
|
+
'reruns to green. Exits 3 when a signature could not be verified.')
|
|
461
|
+
.requiredOption('--repo <owner/name>', 'GitHub repository to scan.')
|
|
462
|
+
.addOption(new Option('--limit-runs <runs>', 'How many recent RUNS to scan.')
|
|
463
|
+
.default(GH_FLAKY_RUN_LIMIT)
|
|
464
|
+
.argParser(parseRuns('--limit-runs')))
|
|
465
|
+
.option('--json', 'Emit the full report as JSON.')
|
|
466
|
+
.action((opts) => {
|
|
467
|
+
ghFlakyCmd(opts, deps);
|
|
468
|
+
});
|
|
436
469
|
program
|
|
437
470
|
.command('area-health')
|
|
438
471
|
.description('Area degradation trends over time.')
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
import { EXIT_ABSTAINED, gateOutcome } from '../../core/gate-result.js';
|
|
2
|
+
/** Default window, stated explicitly: gh's own default is silent about it. */
|
|
3
|
+
export const GH_FLAKY_RUN_LIMIT = 100;
|
|
4
|
+
const GH_TIMEOUT_SECONDS = 30;
|
|
5
|
+
/** Conclusions that are not a completed outcome ("" is still in progress). */
|
|
6
|
+
const NON_OUTCOME_NAMES = 'action_required neutral skipped stale';
|
|
7
|
+
const NON_OUTCOMES = new Set(['', ...NON_OUTCOME_NAMES.split(' ')]);
|
|
8
|
+
const isOutcome = (c) => c !== null && !NON_OUTCOMES.has(c);
|
|
9
|
+
const SIGNATURES = ['same-sha-flip', 'rerun-attempt'];
|
|
10
|
+
const label = (s) => (s === 'same-sha-flip' ? 'same-SHA flip' : s);
|
|
11
|
+
function ghJson(run, cmd) {
|
|
12
|
+
const result = run(cmd, { timeout: GH_TIMEOUT_SECONDS });
|
|
13
|
+
if (result.returncode !== 0) {
|
|
14
|
+
throw new Error(`exit ${result.returncode}: ${result.stderr.trim()}`);
|
|
15
|
+
}
|
|
16
|
+
return JSON.parse(result.stdout);
|
|
17
|
+
}
|
|
18
|
+
function toRunRow(raw) {
|
|
19
|
+
const r = (raw ?? {});
|
|
20
|
+
if (typeof r.databaseId !== 'number' || typeof r.headSha !== 'string') {
|
|
21
|
+
return null;
|
|
22
|
+
}
|
|
23
|
+
return {
|
|
24
|
+
id: r.databaseId,
|
|
25
|
+
headSha: r.headSha,
|
|
26
|
+
workflow: typeof r.workflowName === 'string' ? r.workflowName : '',
|
|
27
|
+
conclusion: typeof r.conclusion === 'string' ? r.conclusion : null,
|
|
28
|
+
// Never defaulted to 1: that would verify a rerun signature never read.
|
|
29
|
+
attempt: typeof r.attempt === 'number' ? r.attempt : null,
|
|
30
|
+
};
|
|
31
|
+
}
|
|
32
|
+
function listRuns(repo, limit, run) {
|
|
33
|
+
const cmd = ['gh', 'run', 'list', '--repo', repo, '--limit', String(limit)];
|
|
34
|
+
cmd.push('--json', 'databaseId,headSha,workflowName,conclusion,attempt');
|
|
35
|
+
const parsed = ghJson(run, cmd);
|
|
36
|
+
if (!Array.isArray(parsed))
|
|
37
|
+
throw new Error('output was not a list');
|
|
38
|
+
const rows = parsed.map(toRunRow).filter((r) => r !== null);
|
|
39
|
+
return { rows, rawCount: parsed.length };
|
|
40
|
+
}
|
|
41
|
+
function sameShaFlips(rows) {
|
|
42
|
+
const groups = new Map();
|
|
43
|
+
for (const { workflow, headSha, conclusion } of rows) {
|
|
44
|
+
if (!isOutcome(conclusion))
|
|
45
|
+
continue;
|
|
46
|
+
const key = JSON.stringify([workflow, headSha]);
|
|
47
|
+
groups.set(key, [...(groups.get(key) ?? []), conclusion]);
|
|
48
|
+
}
|
|
49
|
+
const flips = [];
|
|
50
|
+
for (const [key, all] of groups) {
|
|
51
|
+
const conclusions = [...new Set(all)];
|
|
52
|
+
if (!conclusions.includes('success') || conclusions.length < 2)
|
|
53
|
+
continue;
|
|
54
|
+
const [workflow, headSha] = JSON.parse(key);
|
|
55
|
+
flips.push({ signature: 'same-sha-flip', headSha, workflow, conclusions });
|
|
56
|
+
}
|
|
57
|
+
return flips;
|
|
58
|
+
}
|
|
59
|
+
/** Read every earlier attempt of every green rerun. */
|
|
60
|
+
function rerunSignals(repo, rows, run) {
|
|
61
|
+
const read = { candidates: [], unverifiable: [], nonOutcome: 0 };
|
|
62
|
+
for (const row of rows) {
|
|
63
|
+
if (row.conclusion !== 'success')
|
|
64
|
+
continue;
|
|
65
|
+
for (let n = 1; n < (row.attempt ?? 1); n++) {
|
|
66
|
+
readAttempt(repo, row, n, run, read);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
return read;
|
|
70
|
+
}
|
|
71
|
+
function readAttempt(repo, row, attempt, run, read) {
|
|
72
|
+
const source = `/repos/${repo}/actions/runs/${row.id}/attempts/${attempt}`;
|
|
73
|
+
try {
|
|
74
|
+
const body = ghJson(run, ['gh', 'api', source.slice(1)]);
|
|
75
|
+
const c = typeof body?.conclusion === 'string' ? body.conclusion : null;
|
|
76
|
+
if (!isOutcome(c))
|
|
77
|
+
read.nonOutcome++;
|
|
78
|
+
if (!isOutcome(c) || c === 'success')
|
|
79
|
+
return;
|
|
80
|
+
const { headSha, workflow, id: runId } = row;
|
|
81
|
+
const hit = { headSha, workflow, runId, earlierConclusion: c };
|
|
82
|
+
read.candidates.push({ signature: 'rerun-attempt', ...hit });
|
|
83
|
+
}
|
|
84
|
+
catch (e) {
|
|
85
|
+
const reason = e instanceof Error ? e.message : String(e);
|
|
86
|
+
read.unverifiable.push({ runId: row.id, attempt, source, reason });
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
function abstained(repo, reason) {
|
|
90
|
+
return {
|
|
91
|
+
repo,
|
|
92
|
+
verdict: 'abstained',
|
|
93
|
+
runsChecked: 0,
|
|
94
|
+
verifiedAgainst: [],
|
|
95
|
+
candidates: [],
|
|
96
|
+
unverifiable: [],
|
|
97
|
+
complete: false,
|
|
98
|
+
window: 'no runs read',
|
|
99
|
+
skippedRows: 0,
|
|
100
|
+
missingAttemptRows: 0,
|
|
101
|
+
nonOutcomeRows: 0,
|
|
102
|
+
notes: [],
|
|
103
|
+
reason,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
function verdictOf(candidates, unverified) {
|
|
107
|
+
if (candidates > 0)
|
|
108
|
+
return 'candidates';
|
|
109
|
+
return unverified > 0 ? 'flake-signal-unverifiable' : 'verified-zero';
|
|
110
|
+
}
|
|
111
|
+
/** Every disclosure, in words, for both the text and JSON output. */
|
|
112
|
+
function notesFor(r) {
|
|
113
|
+
const kinds = `in progress, ${NON_OUTCOME_NAMES.split(' ').join(', ')}`;
|
|
114
|
+
const notes = [
|
|
115
|
+
[r.complete ? 0 : 1, r.window],
|
|
116
|
+
[
|
|
117
|
+
r.skippedRows,
|
|
118
|
+
`${r.skippedRows} malformed run row(s) from gh skipped; not checked`,
|
|
119
|
+
],
|
|
120
|
+
[
|
|
121
|
+
r.missingAttemptRows,
|
|
122
|
+
`${r.missingAttemptRows} run row(s) carried no attempt number; rerun-attempt not verified`,
|
|
123
|
+
],
|
|
124
|
+
[
|
|
125
|
+
r.nonOutcomeRows,
|
|
126
|
+
`${r.nonOutcomeRows} run row(s) had no completed outcome (${kinds}); never a flip or rerun candidate`,
|
|
127
|
+
],
|
|
128
|
+
];
|
|
129
|
+
return notes.filter(([n]) => n > 0).map(([, text]) => text);
|
|
130
|
+
}
|
|
131
|
+
/** Scan the last `limit` runs of `repo` for both flake signatures. */
|
|
132
|
+
export function scanGhFlaky(repo, limit, run) {
|
|
133
|
+
let rows;
|
|
134
|
+
let rawCount;
|
|
135
|
+
try {
|
|
136
|
+
({ rows, rawCount } = listRuns(repo, limit, run));
|
|
137
|
+
}
|
|
138
|
+
catch (e) {
|
|
139
|
+
const why = e instanceof Error ? e.message : String(e);
|
|
140
|
+
return abstained(repo, `gh run list could not be read (${why}).`);
|
|
141
|
+
}
|
|
142
|
+
if (rows.length === 0) {
|
|
143
|
+
return abstained(repo, 'gh run list returned zero runs.');
|
|
144
|
+
}
|
|
145
|
+
const reruns = rerunSignals(repo, rows, run);
|
|
146
|
+
const candidates = [...sameShaFlips(rows), ...reruns.candidates];
|
|
147
|
+
const missingAttemptRows = rows.filter((r) => r.attempt === null).length;
|
|
148
|
+
const unverified = reruns.unverifiable.length + missingAttemptRows;
|
|
149
|
+
const nonOutcomeRows = rows.filter((r) => !isOutcome(r.conclusion)).length + reruns.nonOutcome;
|
|
150
|
+
const complete = rawCount < limit; // disclosed; the verdict covers the window
|
|
151
|
+
const report = {
|
|
152
|
+
repo,
|
|
153
|
+
verdict: verdictOf(candidates.length, unverified),
|
|
154
|
+
runsChecked: rows.length,
|
|
155
|
+
verifiedAgainst: SIGNATURES.slice(0, unverified > 0 ? 1 : 2),
|
|
156
|
+
candidates,
|
|
157
|
+
unverifiable: reruns.unverifiable,
|
|
158
|
+
complete,
|
|
159
|
+
window: complete
|
|
160
|
+
? `window complete: all ${rows.length} runs checked`
|
|
161
|
+
: `window truncated at ${limit} runs; older runs unchecked`,
|
|
162
|
+
skippedRows: rawCount - rows.length,
|
|
163
|
+
missingAttemptRows,
|
|
164
|
+
nonOutcomeRows,
|
|
165
|
+
notes: [],
|
|
166
|
+
};
|
|
167
|
+
report.notes = notesFor(report);
|
|
168
|
+
return report;
|
|
169
|
+
}
|
|
170
|
+
function describeCandidate(c) {
|
|
171
|
+
if (c.signature === 'rerun-attempt') {
|
|
172
|
+
return (` rerun-attempt run ${c.runId} (${c.workflow} @ ${c.headSha}): ` +
|
|
173
|
+
`an earlier attempt concluded ${c.earlierConclusion}, re-run to success`);
|
|
174
|
+
}
|
|
175
|
+
return (` same-SHA flip ${c.workflow} @ ${c.headSha}: ` +
|
|
176
|
+
`${(c.conclusions ?? []).join(' / ')}`);
|
|
177
|
+
}
|
|
178
|
+
/** Human-readable report. Never prints a zero without its signatures. */
|
|
179
|
+
export function renderGhFlaky(report) {
|
|
180
|
+
if (report.verdict === 'abstained') {
|
|
181
|
+
const { summaryLine } = gateOutcome({ checked: 0, findings: [] }, 'gate');
|
|
182
|
+
return [`${summaryLine} gh-flaky for ${report.repo}: ${report.reason}`];
|
|
183
|
+
}
|
|
184
|
+
const lines = [`gh-flaky ${report.repo}: ${report.verdict}`];
|
|
185
|
+
lines.push(...report.candidates.map(describeCandidate));
|
|
186
|
+
for (const u of report.unverifiable) {
|
|
187
|
+
lines.push(` unverifiable run ${u.runId}: ${u.source} (${u.reason})`);
|
|
188
|
+
}
|
|
189
|
+
if (report.verdict === 'flake-signal-unverifiable') {
|
|
190
|
+
lines.push('flake-signal-unverifiable: rerun-attempt is NOT verified.');
|
|
191
|
+
}
|
|
192
|
+
else if (report.verdict === 'verified-zero') {
|
|
193
|
+
lines.push('0 candidates.');
|
|
194
|
+
}
|
|
195
|
+
const names = report.verifiedAgainst.map(label).join(', ');
|
|
196
|
+
lines.push(`Verified against: ${names} (${report.runsChecked} runs).`);
|
|
197
|
+
lines.push(...report.notes.map((n) => `Note: ${n}.`));
|
|
198
|
+
return lines;
|
|
199
|
+
}
|
|
200
|
+
/** Gate contract: 0 verified zero, 1 candidates, 3 abstained or unverifiable. */
|
|
201
|
+
export function ghFlakyExitCode(report) {
|
|
202
|
+
if (report.verdict === 'verified-zero')
|
|
203
|
+
return 0;
|
|
204
|
+
return report.verdict === 'candidates' ? 1 : EXIT_ABSTAINED;
|
|
205
|
+
}
|
|
206
|
+
//# sourceMappingURL=gh-run-attempts.js.map
|