canary-test-cli 7.1.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/agents/skills/README.md +327 -0
  2. package/agents/skills/canary:generate.md +49 -0
  3. package/agents/skills/canary:init.md +37 -0
  4. package/agents/skills/canary:migrate.md +66 -0
  5. package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
  6. package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
  7. package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
  8. package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
  9. package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
  10. package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
  11. package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
  12. package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
  13. package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
  14. package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
  15. package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
  16. package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
  17. package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
  18. package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
  19. package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
  20. package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
  21. package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
  22. package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
  23. package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
  24. package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
  25. package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
  26. package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
  27. package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
  28. package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
  29. package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
  30. package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
  31. package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
  32. package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
  33. package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
  34. package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
  35. package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
  36. package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
  37. package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
  38. package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
  39. package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
  40. package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
  41. package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
  42. package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
  43. package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
  44. package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
  45. package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
  46. package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
  47. package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
  48. package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
  49. package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
  50. package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
  51. package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
  52. package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
  53. package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
  54. package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
  55. package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
  56. package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
  57. package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
  58. package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
  59. package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
  60. package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
  61. package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
  62. package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
  63. package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
  64. package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
  65. package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
  66. package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
  67. package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
  68. package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
  69. package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
  70. package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
  71. package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
  72. package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
  73. package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
  74. package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
  75. package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
  76. package/agents/skills/lib/parse-args.mjs +275 -0
  77. package/dist/engine/analysis/batwoman/audit.js +39 -0
  78. package/dist/engine/analysis/batwoman/closure.js +159 -0
  79. package/dist/engine/analysis/batwoman/gh-history.js +119 -0
  80. package/dist/engine/analysis/batwoman/probes.js +195 -0
  81. package/dist/engine/analysis/batwoman/registry.js +142 -0
  82. package/dist/engine/analysis/batwoman/render.js +194 -0
  83. package/dist/engine/analysis/batwoman/run-window.js +122 -0
  84. package/dist/engine/analysis/batwoman/text.js +84 -0
  85. package/dist/engine/analysis/batwoman/triggers.js +122 -0
  86. package/dist/engine/analysis/batwoman/verdict.js +64 -0
  87. package/dist/engine/analysis/cli.js +47 -14
  88. package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
  89. package/dist/engine/batwoman-cli.js +119 -0
  90. package/dist/engine/ci-ready-cli.js +71 -0
  91. package/dist/engine/cli-commands.js +49 -72
  92. package/dist/engine/cli.core.js +16 -0
  93. package/dist/engine/company-knowledge-cli.js +10 -2
  94. package/dist/engine/core/ci-ready.js +112 -0
  95. package/dist/engine/core/company-knowledge.js +8 -0
  96. package/dist/engine/core/migrator.js +147 -20
  97. package/dist/engine/core/permission-matrix.js +219 -0
  98. package/dist/engine/core/quality-scorer.js +27 -19
  99. package/dist/engine/core/scaling-curve.js +143 -0
  100. package/dist/engine/core/skill-dispatch.js +115 -0
  101. package/dist/engine/core/skill-examples.js +103 -3
  102. package/dist/engine/core/skill-registry.js +59 -4
  103. package/dist/engine/core/string-literals.js +3 -1
  104. package/dist/engine/core/test-files.js +77 -0
  105. package/dist/engine/core/vacuity-scanner.js +330 -15
  106. package/dist/engine/core/workflow-discovery.js +41 -23
  107. package/dist/engine/guardian/adjudication-github.js +136 -0
  108. package/dist/engine/guardian/adjudication.js +119 -340
  109. package/dist/engine/guardian/analysis-emit.js +7 -2
  110. package/dist/engine/guardian/cli.js +277 -249
  111. package/dist/engine/guardian/coverage.js +2 -1
  112. package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
  113. package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
  114. package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
  115. package/dist/engine/guardian/diff-coverage/paths.js +5 -9
  116. package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
  117. package/dist/engine/guardian/diff-extractor.js +31 -32
  118. package/dist/engine/guardian/pr-check.js +354 -223
  119. package/dist/engine/guardian/pr-comment.js +35 -58
  120. package/dist/engine/guardian/weak-test.js +236 -0
  121. package/dist/engine/mcp-server.js +67 -4
  122. package/dist/engine/permission-matrix-cli.js +51 -0
  123. package/dist/engine/scaling-curve-cli.js +147 -0
  124. package/dist/engine/skills-cli.js +171 -51
  125. package/dist/engine/workflow-cli.js +85 -65
  126. package/dist/reporters/testtracker.d.ts +1 -1
  127. package/dist/reporters/testtracker.js +1 -1
  128. package/package.json +3 -2
@@ -0,0 +1,125 @@
1
+ // blast -- render the one-page broken-main blast. Pure.
2
+ //
3
+ // Three outputs from one assessment, because the same fact has three audiences:
4
+ // markdown the standalone artifact a human opens
5
+ // annotations a single `::error` line so the GitHub Checks UI shows it
6
+ // chatBlock plain text the on-call pastes into whatever chat they use
7
+ //
8
+ // The chat block is emitted, never posted. The skill has no credentials, no
9
+ // webhook, and no write access to anything but the artifact path it was given.
10
+
11
+ const CROSS = '\u274c'; // cross mark
12
+ const CHECK = '\u2705'; // white heavy check mark
13
+ const DASH = '\u2014'; // em dash
14
+
15
+ /** The first meaningful line of an error blob, clipped. */
16
+ function firstLine(error, limit = 160) {
17
+ if (!error) return '(no error message)';
18
+ for (const raw of String(error).split(/\r\n|\r|\n/)) {
19
+ const line = raw.trim();
20
+ if (line) return line.slice(0, limit);
21
+ }
22
+ return '(no error message)';
23
+ }
24
+
25
+ const NEXT_STEP = {
26
+ revert: 'Revert the culprit commit. One area, one attributable commit.',
27
+ quarantine:
28
+ 'Quarantine the failing tests and open a tracking issue. The break is not cleanly attributable to one commit.',
29
+ investigate:
30
+ 'Investigate before acting. The failing tests carry no owning area, so neither a revert nor a quarantine can be aimed.',
31
+ };
32
+
33
+ /**
34
+ * Render the blast.
35
+ *
36
+ * @param {{branch: string, assessment: object, cluster: object|null}} input
37
+ * @returns {{markdown: string, annotations: string[], chatBlock: string}}
38
+ */
39
+ export function renderBlast({ branch, assessment, cluster }) {
40
+ if (assessment.state === 'abstained') {
41
+ const text =
42
+ `${CROSS} canary-screech ABSTAINED ${DASH} the run-history store holds no run for \`${branch}\`.\n\n` +
43
+ 'Zero observations is not a healthy branch. Point `--history` at a store ' +
44
+ 'that this branch actually writes to, or wire the suite to record runs.';
45
+ return {
46
+ markdown: `# canary-screech ${DASH} ${branch}\n\n${text}\n`,
47
+ annotations: [],
48
+ chatBlock: text,
49
+ };
50
+ }
51
+
52
+ if (assessment.state === 'green') {
53
+ const text = `${CHECK} \`${branch}\` is green as of ${assessment.latest.commit_sha ?? '(unknown commit)'}.`;
54
+ return {
55
+ markdown: `# canary-screech ${DASH} ${branch}\n\n${text}\n`,
56
+ annotations: [],
57
+ chatBlock: text,
58
+ };
59
+ }
60
+
61
+ const { culpritRange, firstRed } = assessment;
62
+ const from =
63
+ culpritRange.from ?? '(unknown - no green run recorded before the break)';
64
+ const to = culpritRange.to ?? '(unknown)';
65
+ const failureCount = cluster.clusters.reduce((n, c) => n + c.tests.length, 0);
66
+
67
+ const chatLines = [
68
+ `${CROSS} ${branch} is RED ${DASH} ${failureCount} failing test${failureCount === 1 ? '' : 's'}`,
69
+ `culprit range: ${from}..${to}`,
70
+ `owning area: ${cluster.owningArea ?? '(none recorded)'}`,
71
+ `recommendation: ${cluster.recommendation.toUpperCase()} ${DASH} ${NEXT_STEP[cluster.recommendation]}`,
72
+ `first red run: ${firstRed.run_id} (${firstRed.suite}) at ${firstRed.timestamp}`,
73
+ ];
74
+ const chatBlock = chatLines.join('\n');
75
+
76
+ const lines = [
77
+ `# ${CROSS} canary-screech ${DASH} \`${branch}\` is red`,
78
+ '',
79
+ `**First red run:** \`${firstRed.run_id}\` (${firstRed.suite}) at ${firstRed.timestamp}`,
80
+ '',
81
+ '## Culprit commit range',
82
+ '',
83
+ `\`${from}\` .. \`${to}\``,
84
+ '',
85
+ ];
86
+ if (!culpritRange.bounded) {
87
+ lines.push(
88
+ '> The lower bound is unknown: the store holds no green run for this ' +
89
+ 'branch before the break, so the range below is open-ended.',
90
+ '',
91
+ );
92
+ }
93
+ lines.push('## Failure cluster', '');
94
+ for (const c of cluster.clusters) {
95
+ lines.push(`### ${c.category} (${c.tests.length})`, '');
96
+ for (const t of c.tests) {
97
+ lines.push(`- \`${t.test_name}\` ${DASH} ${firstLine(t.error_text)}`);
98
+ }
99
+ lines.push('');
100
+ }
101
+ lines.push(
102
+ '## Owning area',
103
+ '',
104
+ cluster.areas.length
105
+ ? cluster.areas.map((a) => `- \`${a.area}\` (${a.count})`).join('\n')
106
+ : '_No failing test carries an `area`, so ownership could not be derived._',
107
+ '',
108
+ '## Recommendation',
109
+ '',
110
+ `**${cluster.recommendation.toUpperCase()}** ${DASH} ${NEXT_STEP[cluster.recommendation]}`,
111
+ '',
112
+ '## Chat-ready block',
113
+ '',
114
+ '```text',
115
+ chatBlock,
116
+ '```',
117
+ '',
118
+ );
119
+
120
+ const annotations = [
121
+ `::error title=Broken branch::${branch} is red ${DASH} ${failureCount} failing test${failureCount === 1 ? '' : 's'} in ${cluster.owningArea ?? 'an unrecorded area'}; culprit ${from}..${to}; recommendation: ${cluster.recommendation}`,
122
+ ];
123
+
124
+ return { markdown: lines.join('\n'), annotations, chatBlock };
125
+ }
@@ -0,0 +1,128 @@
1
+ #!/usr/bin/env node
2
+ // canary-screech -- broken-main siren (#591).
3
+ //
4
+ // The cross-run member of the failure-surfacing family. canary-fail-fast aborts
5
+ // inside one run; canary-test-reporter summarises one run; neither can say the
6
+ // default branch went red, because that fact only exists in the SEQUENCE of
7
+ // runs. This skill reads that sequence out of the run-history store and emits a
8
+ // one-page blast: culprit commit range, failure cluster, owning area, a
9
+ // quarantine-or-revert recommendation, and a chat-ready block.
10
+ //
11
+ // It emits. It does not post: no Slack, no Teams, no webhook, no `gh`. Output
12
+ // is a markdown artifact (--out) plus a `::error` GitHub Actions annotation,
13
+ // the same channel canary-fail-fast uses.
14
+ //
15
+ // Advisory by default (#508 D3): exit 0 whatever it finds, unless --strict.
16
+ // Under --strict: 1 = red, 3 = abstained (zero runs for the branch), 0 = green.
17
+ //
18
+ // Invoked via `canary skills run canary-screech -- --history <jsonl> [...]`.
19
+
20
+ import fs from 'node:fs';
21
+ import path from 'node:path';
22
+
23
+ import {
24
+ createParser,
25
+ formatUsageError,
26
+ EXIT_USAGE,
27
+ } from '../../../lib/parse-args.mjs';
28
+ import { loadRuns, runsForBranch } from './history.mjs';
29
+ import { assessBranch } from './redness.mjs';
30
+ import { clusterFailures } from './cluster.mjs';
31
+ import { renderBlast } from './blast.mjs';
32
+
33
+ const PREFIX = 'canary-screech:';
34
+
35
+ /** Exit code reserved family-wide for "abstained" (#508 D4). */
36
+ const EXIT_ABSTAINED = 3;
37
+
38
+ const USAGE =
39
+ 'usage: canary-screech [-h] --history PATH [--branch NAME] [--out PATH] [--strict]\n' +
40
+ '\n' +
41
+ 'Broken-main siren: one-page blast when the default branch goes red.';
42
+
43
+ export const CLI_SPEC = {
44
+ prog: 'canary-screech',
45
+ booleans: { '--strict': 'strict' },
46
+ values: {
47
+ '--history': { key: 'history' },
48
+ '--branch': { key: 'branch' },
49
+ '--out': { key: 'out' },
50
+ },
51
+ defaults: { branch: 'main' },
52
+ required: ['--history'],
53
+ };
54
+
55
+ const parseArgs = createParser(CLI_SPEC);
56
+
57
+ /**
58
+ * Write the artifact, or return the message explaining why it could not be.
59
+ * Extracted so `main` stays under the complexity the perf gate allows.
60
+ *
61
+ * @returns {string|null} null on success
62
+ */
63
+ function writeArtifact(out, markdown) {
64
+ try {
65
+ fs.mkdirSync(path.dirname(path.resolve(out)), { recursive: true });
66
+ fs.writeFileSync(out, markdown, 'utf8');
67
+ return null;
68
+ } catch (exc) {
69
+ return `cannot write artifact: ${exc.message}`;
70
+ }
71
+ }
72
+
73
+ /** The `--strict` exit contract. Advisory callers never reach this. */
74
+ function strictExitFor(state) {
75
+ if (state === 'abstained') return EXIT_ABSTAINED;
76
+ return state === 'red' ? 1 : 0;
77
+ }
78
+
79
+ export function main(argv = []) {
80
+ const { opts: args, help, error } = parseArgs(argv);
81
+
82
+ if (help) {
83
+ console.log(USAGE);
84
+ return 0;
85
+ }
86
+ if (error) {
87
+ console.error(formatUsageError(CLI_SPEC.prog, error));
88
+ return EXIT_USAGE;
89
+ }
90
+
91
+ let runs;
92
+ try {
93
+ runs = runsForBranch(loadRuns(args.history), args.branch);
94
+ } catch (exc) {
95
+ // A bad --history path is an error, never "the branch looks fine".
96
+ console.error(`${PREFIX} ${exc.message}`);
97
+ return 1;
98
+ }
99
+
100
+ const assessment = assessBranch(runs);
101
+ const cluster =
102
+ assessment.state === 'red'
103
+ ? clusterFailures(assessment.firstRed, assessment.culpritRange)
104
+ : null;
105
+ const blast = renderBlast({ branch: args.branch, assessment, cluster });
106
+
107
+ console.log(blast.markdown);
108
+ for (const annotation of blast.annotations) console.log(annotation);
109
+
110
+ if (args.out) {
111
+ const failure = writeArtifact(args.out, blast.markdown);
112
+ if (failure) {
113
+ console.error(`${PREFIX} ${failure}`);
114
+ return 1;
115
+ }
116
+ }
117
+
118
+ return args.strict ? strictExitFor(assessment.state) : 0;
119
+ }
120
+
121
+ // Direct execution (the skill runner execs this file via its shebang).
122
+ //
123
+ // `process.exitCode`, not `process.exit()`: a large payload exceeds the pipe
124
+ // buffer, and `process.exit` tears the process down mid-write, leaving
125
+ // truncated output that still exits 0 (#791).
126
+ if (import.meta.url === `file://${process.argv[1]}`) {
127
+ process.exitCode = main(process.argv.slice(2));
128
+ }
@@ -0,0 +1,97 @@
1
+ // cluster -- what broke together, who owns it, and what to do about it. Pure.
2
+ //
3
+ // The one-pager's value is not the list of failing tests (the run log already
4
+ // has that). It is the SHAPE of the break: whether one thing broke in one
5
+ // place, or many things broke everywhere. Those two shapes want opposite
6
+ // responses, which is why the recommendation is derived from the shape rather
7
+ // than from the failure count.
8
+
9
+ const UNCATEGORIZED = 'uncategorized';
10
+
11
+ /** A test row counts as a failure when it did not pass. */
12
+ function isFailure(test) {
13
+ const status = String(test?.status ?? '').toLowerCase();
14
+ return (
15
+ status === 'failed' || status === 'unexpected' || status === 'timedout'
16
+ );
17
+ }
18
+
19
+ /**
20
+ * @typedef {object} Cluster
21
+ * @property {string} category
22
+ * @property {object[]} tests
23
+ */
24
+
25
+ /** Failures grouped by category, largest group first. */
26
+ function byCategory(failures) {
27
+ const groups = new Map();
28
+ for (const test of failures) {
29
+ const category = test.failure_category || UNCATEGORIZED;
30
+ if (!groups.has(category)) groups.set(category, []);
31
+ groups.get(category).push(test);
32
+ }
33
+ // Ties broken by name so the one-pager is stable across runs -- a report that
34
+ // reshuffles itself is a report nobody diffs.
35
+ return [...groups.entries()]
36
+ .map(([category, tests]) => ({ category, tests }))
37
+ .sort(
38
+ (a, b) =>
39
+ b.tests.length - a.tests.length || a.category.localeCompare(b.category),
40
+ );
41
+ }
42
+
43
+ /** Failure counts per owning area, busiest first. Areas are often absent. */
44
+ function byArea(failures) {
45
+ const counts = new Map();
46
+ for (const test of failures) {
47
+ if (test.area) counts.set(test.area, (counts.get(test.area) ?? 0) + 1);
48
+ }
49
+ return [...counts.entries()]
50
+ .map(([area, count]) => ({ area, count }))
51
+ .sort((a, b) => b.count - a.count || a.area.localeCompare(b.area));
52
+ }
53
+
54
+ /**
55
+ * The response the break's shape argues for.
56
+ *
57
+ * @param {{area: string, count: number}[]} areas
58
+ * @param {{commits: string[], bounded: boolean}} culpritRange
59
+ */
60
+ function recommendFor(areas, culpritRange) {
61
+ // No area data at all. A recommendation here would be derived from nothing,
62
+ // which is exactly the confident-on-absent-data shape this repo keeps getting
63
+ // burned by -- so it declines to make one.
64
+ if (!areas.length) return 'investigate';
65
+
66
+ // One place, one attributable commit: the cheap, complete fix is a revert.
67
+ const commits = culpritRange?.commits ?? [];
68
+ const attributable = culpritRange?.bounded === true && commits.length === 1;
69
+ if (areas.length === 1 && attributable) return 'revert';
70
+
71
+ // Several areas, or an un-attributable range: nothing is cleanly revertable,
72
+ // so isolate the failures and keep the branch moving.
73
+ return 'quarantine';
74
+ }
75
+
76
+ /**
77
+ * Cluster a red run's failures and derive the response.
78
+ *
79
+ * @param {object} run the first red run
80
+ * @param {{commits: string[], bounded: boolean}} culpritRange
81
+ * @returns {{clusters: Cluster[], areas: {area: string, count: number}[],
82
+ * owningArea: string|null,
83
+ * recommendation: 'revert'|'quarantine'|'investigate'}}
84
+ */
85
+ export function clusterFailures(run, culpritRange) {
86
+ const failures = (run?.tests ?? []).filter(isFailure);
87
+ const areas = byArea(failures);
88
+ return {
89
+ clusters: byCategory(failures),
90
+ areas,
91
+ // The busiest area, or null when nothing recorded one. With several areas
92
+ // in play this is a pointer, not an owner -- which is why the
93
+ // recommendation refuses to treat it as one.
94
+ owningArea: areas[0]?.area ?? null,
95
+ recommendation: recommendFor(areas, culpritRange),
96
+ };
97
+ }
@@ -0,0 +1,73 @@
1
+ // history -- read the run-history store canary-screech watches.
2
+ //
3
+ // The store is `test-results/reports/history-v2.jsonl`: one RunRecord JSON
4
+ // object per line (the shape in `ts/src/history/record.ts`). This module is the
5
+ // only part of the skill that touches the filesystem; everything downstream is
6
+ // pure.
7
+ //
8
+ // Deliberately NOT tolerant of a missing file. Returning `[]` there would be
9
+ // byte-identical to a genuinely empty store, and the caller would print an
10
+ // abstention -- a plausible-looking, wrong answer -- when the truth is that
11
+ // `--history` points at nothing. A typo'd path must look like a typo'd path.
12
+
13
+ import fs from 'node:fs';
14
+
15
+ /** The fields this skill reads. Everything else in the record is ignored. */
16
+
17
+ /**
18
+ * Parse the JSONL store into records.
19
+ *
20
+ * @param {string} file path to history-v2.jsonl
21
+ * @returns {object[]} one record per non-blank line, in file order
22
+ */
23
+ export function loadRuns(file) {
24
+ if (!fs.existsSync(file)) {
25
+ throw new Error(`history store not found: ${file}`);
26
+ }
27
+ const lines = fs.readFileSync(file, 'utf8').split(/\r\n|\r|\n/);
28
+ const runs = [];
29
+ for (let i = 0; i < lines.length; i += 1) {
30
+ const line = lines[i].trim();
31
+ if (!line) continue;
32
+ let record;
33
+ try {
34
+ record = JSON.parse(line);
35
+ } catch (exc) {
36
+ throw new Error(
37
+ `malformed history record at line ${i + 1}: ${exc.message}`,
38
+ );
39
+ }
40
+ if (
41
+ record === null ||
42
+ typeof record !== 'object' ||
43
+ Array.isArray(record)
44
+ ) {
45
+ throw new Error(
46
+ `malformed history record at line ${i + 1}: not an object`,
47
+ );
48
+ }
49
+ runs.push(record);
50
+ }
51
+ return runs;
52
+ }
53
+
54
+ /**
55
+ * The runs for one branch, oldest first.
56
+ *
57
+ * Ordered by `timestamp`, not by file order: the store is append-only per
58
+ * writer, and several suites append to it concurrently, so file order is not
59
+ * chronological order. Getting this backwards would name the wrong run as the
60
+ * first red one, which is the whole culprit range.
61
+ *
62
+ * @param {object[]} runs
63
+ * @param {string} branch
64
+ * @returns {object[]}
65
+ */
66
+ export function runsForBranch(runs, branch) {
67
+ return runs
68
+ .filter((r) => r.branch === branch)
69
+ .slice()
70
+ .sort((a, b) =>
71
+ String(a.timestamp ?? '').localeCompare(String(b.timestamp ?? '')),
72
+ );
73
+ }
@@ -0,0 +1,94 @@
1
+ // redness -- is this branch red, and which commits are implicated. Pure.
2
+ //
3
+ // "Red" is a property of the LATEST run on the branch, but the culprit range is
4
+ // a property of the transition into red, so the walk goes backwards from the
5
+ // latest run to the FIRST consecutive red one. Naming the latest red run as the
6
+ // culprit is the obvious mistake: by the time a human looks, main has usually
7
+ // taken several more commits while staying broken, and reverting the newest of
8
+ // them fixes nothing.
9
+ //
10
+ // Split into four small functions rather than one readable-looking block: the
11
+ // perf gate scores the block at cyclomatic 12 / 53 lines, and the shape it is
12
+ // objecting to is real -- three distinct verdicts sharing one return statement.
13
+
14
+ /** A run counts as failing when it recorded at least one failed test. */
15
+ function isRed(run) {
16
+ return Number(run.failed ?? 0) > 0;
17
+ }
18
+
19
+ /**
20
+ * @typedef {object} CulpritRange
21
+ * @property {string|null} from last known-green commit, or null if unknown
22
+ * @property {string|null} to commit of the first red run
23
+ * @property {string[]} commits distinct commits inside the range
24
+ * @property {boolean} bounded false when no green run precedes the break
25
+ */
26
+
27
+ /** Index of the first run in the unbroken red tail ending at the latest run. */
28
+ function firstRedIndex(runs) {
29
+ let i = runs.length - 1;
30
+ while (i > 0 && isRed(runs[i - 1])) i -= 1;
31
+ return i;
32
+ }
33
+
34
+ /** The red verdict, with the range the break is attributable to. */
35
+ function redVerdict(runs, index) {
36
+ const firstRed = runs[index];
37
+ const lastGreen = index > 0 ? runs[index - 1] : null;
38
+
39
+ // The commits this store can actually SEE inside the range. The interval is
40
+ // `(lastGreen, firstRed]`, and its interior may hold commits that no run ever
41
+ // observed -- enumerating those needs git, which this skill deliberately does
42
+ // not shell out to. So the list is the observed suspect only, and
43
+ // `bounded: false` is how "the lower bound is unknown" is said out loud
44
+ // rather than implied away. `clusterFailures` accepts a longer list so a
45
+ // git-aware caller can widen the range without changing this contract.
46
+ const commits = firstRed.commit_sha ? [firstRed.commit_sha] : [];
47
+
48
+ return {
49
+ state: 'red',
50
+ latest: runs[runs.length - 1],
51
+ firstRed,
52
+ lastGreen,
53
+ culpritRange: {
54
+ from: lastGreen?.commit_sha ?? null,
55
+ to: firstRed.commit_sha ?? null,
56
+ commits,
57
+ bounded: lastGreen !== null,
58
+ },
59
+ };
60
+ }
61
+
62
+ /**
63
+ * Assess a branch from its runs, oldest first.
64
+ *
65
+ * @param {object[]} runs oldest-first runs for ONE branch
66
+ * @returns {{state: 'red'|'green'|'abstained', latest: object|null,
67
+ * firstRed: object|null, lastGreen: object|null,
68
+ * culpritRange: CulpritRange|null}}
69
+ */
70
+ export function assessBranch(runs) {
71
+ // Zero denominator. Not green -- nothing was observed, so nothing is known.
72
+ if (!runs.length) {
73
+ return {
74
+ state: 'abstained',
75
+ latest: null,
76
+ firstRed: null,
77
+ lastGreen: null,
78
+ culpritRange: null,
79
+ };
80
+ }
81
+
82
+ const latest = runs[runs.length - 1];
83
+ if (!isRed(latest)) {
84
+ return {
85
+ state: 'green',
86
+ latest,
87
+ firstRed: null,
88
+ lastGreen: latest,
89
+ culpritRange: null,
90
+ };
91
+ }
92
+
93
+ return redVerdict(runs, firstRedIndex(runs));
94
+ }