canary-test-cli 7.1.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/agents/skills/README.md +327 -0
  2. package/agents/skills/canary:generate.md +49 -0
  3. package/agents/skills/canary:init.md +37 -0
  4. package/agents/skills/canary:migrate.md +66 -0
  5. package/agents/skills/claude-code/canary-add-framework/SKILL.md +248 -0
  6. package/agents/skills/claude-code/canary-batwoman/SKILL.md +119 -0
  7. package/agents/skills/claude-code/canary-blackhawk/SKILL.md +170 -0
  8. package/agents/skills/claude-code/canary-blackhawk/scripts/cli.mjs +188 -0
  9. package/agents/skills/claude-code/canary-blackhawk/scripts/rules.mjs +120 -0
  10. package/agents/skills/claude-code/canary-blackhawk/scripts/scanner.mjs +244 -0
  11. package/agents/skills/claude-code/canary-blackhawk/scripts/string-literals.mjs +116 -0
  12. package/agents/skills/claude-code/canary-cassandra/SKILL.md +187 -0
  13. package/agents/skills/claude-code/canary-cassandra/scripts/cli.mjs +270 -0
  14. package/agents/skills/claude-code/canary-cassandra/scripts/engine.mjs +95 -0
  15. package/agents/skills/claude-code/canary-ci-ready/SKILL.md +178 -0
  16. package/agents/skills/claude-code/canary-ci-ready/skill.yaml +14 -0
  17. package/agents/skills/claude-code/canary-company-knowledge/SKILL.md +196 -0
  18. package/agents/skills/claude-code/canary-critical-areas/SKILL.md +142 -0
  19. package/agents/skills/claude-code/canary-critical-areas/skill.yaml +16 -0
  20. package/agents/skills/claude-code/canary-edge-case-discovery/SKILL.md +160 -0
  21. package/agents/skills/claude-code/canary-edge-case-discovery/skill.yaml +16 -0
  22. package/agents/skills/claude-code/canary-fail-fast/SKILL.md +75 -0
  23. package/agents/skills/claude-code/canary-fail-fast/scripts/cli.mjs +118 -0
  24. package/agents/skills/claude-code/canary-fail-fast/scripts/digest.mjs +69 -0
  25. package/agents/skills/claude-code/canary-fail-fast/scripts/failures.mjs +60 -0
  26. package/agents/skills/claude-code/canary-fail-fast/scripts/fastfail_check.mjs +43 -0
  27. package/agents/skills/claude-code/canary-fail-fast/scripts/parse.mjs +149 -0
  28. package/agents/skills/claude-code/canary-failure-impact/SKILL.md +153 -0
  29. package/agents/skills/claude-code/canary-failure-impact/skill.yaml +15 -0
  30. package/agents/skills/claude-code/canary-fleet-health/SKILL.md +197 -0
  31. package/agents/skills/claude-code/canary-generate-test/SKILL.md +185 -0
  32. package/agents/skills/claude-code/canary-instrument/SKILL.md +157 -0
  33. package/agents/skills/claude-code/canary-instrument/scripts/cli.mjs +178 -0
  34. package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/instrument.mjs +96 -0
  35. package/agents/skills/claude-code/canary-instrument/scripts/otel_bootstrap/playwright-fixture.ts +44 -0
  36. package/agents/skills/claude-code/canary-instrument/scripts/run_types.mjs +81 -0
  37. package/agents/skills/claude-code/canary-instrument/scripts/span_reader.mjs +187 -0
  38. package/agents/skills/claude-code/canary-katana/SKILL.md +243 -0
  39. package/agents/skills/claude-code/canary-katana/scripts/alarm.mjs +296 -0
  40. package/agents/skills/claude-code/canary-katana/scripts/cli.mjs +247 -0
  41. package/agents/skills/claude-code/canary-katana/scripts/diffscan.mjs +0 -0
  42. package/agents/skills/claude-code/canary-katana/scripts/ledger.mjs +183 -0
  43. package/agents/skills/claude-code/canary-pr-guardian/SKILL.md +144 -0
  44. package/agents/skills/claude-code/canary-pr-guardian/skill.yaml +17 -0
  45. package/agents/skills/claude-code/canary-promote-test/SKILL.md +228 -0
  46. package/agents/skills/claude-code/canary-savant/SKILL.md +233 -0
  47. package/agents/skills/claude-code/canary-savant/scripts/cli.mjs +274 -0
  48. package/agents/skills/claude-code/canary-savant/scripts/restoration.mjs +274 -0
  49. package/agents/skills/claude-code/canary-savant/scripts/rules.mjs +168 -0
  50. package/agents/skills/claude-code/canary-savant/scripts/runner.mjs +572 -0
  51. package/agents/skills/claude-code/canary-savant/scripts/scanner.mjs +374 -0
  52. package/agents/skills/claude-code/canary-savant/scripts/string-literals.mjs +116 -0
  53. package/agents/skills/claude-code/canary-screech/SKILL.md +109 -0
  54. package/agents/skills/claude-code/canary-screech/scripts/blast.mjs +125 -0
  55. package/agents/skills/claude-code/canary-screech/scripts/cli.mjs +128 -0
  56. package/agents/skills/claude-code/canary-screech/scripts/cluster.mjs +97 -0
  57. package/agents/skills/claude-code/canary-screech/scripts/history.mjs +73 -0
  58. package/agents/skills/claude-code/canary-screech/scripts/redness.mjs +94 -0
  59. package/agents/skills/claude-code/canary-setup-harness/SKILL.md +263 -0
  60. package/agents/skills/claude-code/canary-shadow/SKILL.md +131 -0
  61. package/agents/skills/claude-code/canary-shadow/scripts/cases.example.json +32 -0
  62. package/agents/skills/claude-code/canary-shadow/scripts/cli.mjs +195 -0
  63. package/agents/skills/claude-code/canary-ship/SKILL.md +177 -0
  64. package/agents/skills/claude-code/canary-ship/skill.yaml +16 -0
  65. package/agents/skills/claude-code/canary-strix/SKILL.md +130 -0
  66. package/agents/skills/claude-code/canary-strix/scripts/cli.mjs +255 -0
  67. package/agents/skills/claude-code/canary-strix/scripts/scanner.mjs +252 -0
  68. package/agents/skills/claude-code/canary-strix/scripts/terms.mjs +132 -0
  69. package/agents/skills/claude-code/canary-test-pipeline/SKILL.md +159 -0
  70. package/agents/skills/claude-code/canary-test-pipeline/skill.yaml +19 -0
  71. package/agents/skills/claude-code/canary-test-reporter/SKILL.md +138 -0
  72. package/agents/skills/claude-code/canary-test-reporter/scripts/cli.mjs +98 -0
  73. package/agents/skills/claude-code/canary-test-reporter/scripts/json_report.mjs +58 -0
  74. package/agents/skills/claude-code/canary-test-reporter/scripts/parse.mjs +216 -0
  75. package/agents/skills/claude-code/canary-test-reporter/scripts/render.mjs +114 -0
  76. package/agents/skills/lib/parse-args.mjs +275 -0
  77. package/dist/engine/analysis/batwoman/audit.js +39 -0
  78. package/dist/engine/analysis/batwoman/closure.js +159 -0
  79. package/dist/engine/analysis/batwoman/gh-history.js +119 -0
  80. package/dist/engine/analysis/batwoman/probes.js +195 -0
  81. package/dist/engine/analysis/batwoman/registry.js +142 -0
  82. package/dist/engine/analysis/batwoman/render.js +194 -0
  83. package/dist/engine/analysis/batwoman/run-window.js +122 -0
  84. package/dist/engine/analysis/batwoman/text.js +84 -0
  85. package/dist/engine/analysis/batwoman/triggers.js +122 -0
  86. package/dist/engine/analysis/batwoman/verdict.js +64 -0
  87. package/dist/engine/analysis/cli.js +47 -14
  88. package/dist/engine/analysis/gh-flaky/gh-run-attempts.js +206 -0
  89. package/dist/engine/batwoman-cli.js +119 -0
  90. package/dist/engine/ci-ready-cli.js +71 -0
  91. package/dist/engine/cli-commands.js +49 -72
  92. package/dist/engine/cli.core.js +16 -0
  93. package/dist/engine/company-knowledge-cli.js +10 -2
  94. package/dist/engine/core/ci-ready.js +112 -0
  95. package/dist/engine/core/company-knowledge.js +8 -0
  96. package/dist/engine/core/migrator.js +147 -20
  97. package/dist/engine/core/permission-matrix.js +219 -0
  98. package/dist/engine/core/quality-scorer.js +27 -19
  99. package/dist/engine/core/scaling-curve.js +143 -0
  100. package/dist/engine/core/skill-dispatch.js +115 -0
  101. package/dist/engine/core/skill-examples.js +103 -3
  102. package/dist/engine/core/skill-registry.js +59 -4
  103. package/dist/engine/core/string-literals.js +3 -1
  104. package/dist/engine/core/test-files.js +77 -0
  105. package/dist/engine/core/vacuity-scanner.js +330 -15
  106. package/dist/engine/core/workflow-discovery.js +41 -23
  107. package/dist/engine/guardian/adjudication-github.js +136 -0
  108. package/dist/engine/guardian/adjudication.js +119 -340
  109. package/dist/engine/guardian/analysis-emit.js +7 -2
  110. package/dist/engine/guardian/cli.js +277 -249
  111. package/dist/engine/guardian/coverage.js +2 -1
  112. package/dist/engine/guardian/diff-coverage/coverage-delta.js +162 -0
  113. package/dist/engine/guardian/diff-coverage/formats/cobertura.js +45 -1
  114. package/dist/engine/guardian/diff-coverage/orchestrator.js +25 -21
  115. package/dist/engine/guardian/diff-coverage/paths.js +5 -9
  116. package/dist/engine/guardian/diff-coverage/report-tier.js +88 -12
  117. package/dist/engine/guardian/diff-extractor.js +31 -32
  118. package/dist/engine/guardian/pr-check.js +354 -223
  119. package/dist/engine/guardian/pr-comment.js +35 -58
  120. package/dist/engine/guardian/weak-test.js +236 -0
  121. package/dist/engine/mcp-server.js +67 -4
  122. package/dist/engine/permission-matrix-cli.js +51 -0
  123. package/dist/engine/scaling-curve-cli.js +147 -0
  124. package/dist/engine/skills-cli.js +171 -51
  125. package/dist/engine/workflow-cli.js +85 -65
  126. package/dist/reporters/testtracker.d.ts +1 -1
  127. package/dist/reporters/testtracker.js +1 -1
  128. package/package.json +3 -2
@@ -25,10 +25,9 @@
25
25
  * to the later CLI wave. This module is the pure library logic.
26
26
  */
27
27
  import { readFileSync } from 'node:fs';
28
- import { extname, join } from 'node:path';
28
+ import { join } from 'node:path';
29
29
  import { readJsonWithWarning } from '../core/config-validation.js';
30
- import { isAssertionFreeTest } from '../core/quality-scorer.js';
31
- import { Fidelity, coverageDegradedNotice, coverageStatus, isSourcePath, isTestPath, isTestSupportPath, isTypeOnlyModule, } from './coverage.js';
30
+ import { Fidelity, coverageDegradedNotice, coverageDeltaNotice, coverageDeltaStatus, coverageCauses, coverageStatus, isSourcePath, isTestPath, isTestSupportPath, isTypeOnlyModule, } from './coverage.js';
32
31
  import { Severity, severitySortKey } from './impact-mapper.js';
33
32
  import { ensureAscii } from '../util/ensure-ascii.js';
34
33
  const HUNK_RE = /^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@/;
@@ -69,72 +68,143 @@ function mergeLines(lines) {
69
68
  */
70
69
  export function scopeDiff(diffText) {
71
70
  const addedByPath = new Map();
72
- let currentPath = null;
73
- let newLineno = 0;
74
- let skipCurrent = false;
71
+ walkDiff(diffText, (line) => {
72
+ if (!line.added)
73
+ return;
74
+ const lines = addedByPath.get(line.path);
75
+ if (lines)
76
+ lines.push(line.lineno);
77
+ else
78
+ addedByPath.set(line.path, [line.lineno]);
79
+ });
80
+ const units = [];
81
+ for (const [path, lines] of addedByPath) {
82
+ units.push({ path, added_ranges: mergeLines(lines) });
83
+ }
84
+ return units;
85
+ }
86
+ // Single-character C escapes git's `quote_c_style` emits, per `sq_lookup`.
87
+ const C_ESCAPES = {
88
+ a: 0x07,
89
+ b: 0x08,
90
+ f: 0x0c,
91
+ n: 0x0a,
92
+ r: 0x0d,
93
+ t: 0x09,
94
+ v: 0x0b,
95
+ '"': 0x22,
96
+ '\\': 0x5c,
97
+ };
98
+ const UTF8_DECODER = new TextDecoder('utf-8');
99
+ /**
100
+ * Undo git's C-style path quoting (`core.quotePath`, on by default).
101
+ *
102
+ * A path with a non-ASCII (or control) byte is emitted by `git diff` wrapped in
103
+ * double quotes with its bytes octal-escaped:
104
+ * `+++ "b/caf\303\251.ts"`. Left as-is, the quotes ride along in the unit's
105
+ * path, so the file resolves to nothing: `extname` reads `.ts"`,
106
+ * {@link isSourcePath} says "not program source", and
107
+ * {@link filterHeuristicNoise} drops the finding — a silent false negative on
108
+ * every non-ASCII-named file. A value that is not quoted is returned verbatim.
109
+ */
110
+ function unquoteCStyle(value) {
111
+ if (!isCQuoted(value))
112
+ return value;
113
+ const body = value.slice(1, -1);
114
+ const bytes = [];
115
+ for (let i = 0; i < body.length; i++) {
116
+ const ch = body[i];
117
+ if (ch !== '\\') {
118
+ // Any non-escaped character is already a decoded code point; re-encode it
119
+ // so the whole path decodes as one UTF-8 byte stream.
120
+ for (const byte of new TextEncoder().encode(ch))
121
+ bytes.push(byte);
122
+ continue;
123
+ }
124
+ const next = body[i + 1];
125
+ if (next === undefined)
126
+ return value; // trailing backslash → not quoted
127
+ if (next >= '0' && next <= '7') {
128
+ const octal = /^[0-7]{1,3}/.exec(body.slice(i + 1))[0];
129
+ bytes.push(Number.parseInt(octal, 8) & 0xff);
130
+ i += octal.length;
131
+ continue;
132
+ }
133
+ const mapped = C_ESCAPES[next];
134
+ if (mapped === undefined)
135
+ return value; // unknown escape → leave alone
136
+ bytes.push(mapped);
137
+ i += 1;
138
+ }
139
+ return UTF8_DECODER.decode(new Uint8Array(bytes));
140
+ }
141
+ /**
142
+ * True when `value` is wrapped in the double quotes git uses for C-style path
143
+ * quoting. Split out of {@link unquoteCStyle} to keep its decode loop under
144
+ * the cyclomatic-complexity threshold.
145
+ */
146
+ function isCQuoted(value) {
147
+ return value.length >= 2 && value.startsWith('"') && value.endsWith('"');
148
+ }
149
+ /** The new-side path a `+++ ` header names, or `null` for a deleted file. */
150
+ function headerPath(line) {
151
+ // A quoted path is unquoted BEFORE the `b/` strip: the quotes wrap the
152
+ // prefix too (`"b/caf\303\251.ts"`), so stripping first would never match.
153
+ const target = unquoteCStyle(line.slice(4).trim());
154
+ if (target === '/dev/null')
155
+ return null;
156
+ // Strip the conventional "b/" prefix.
157
+ return target.startsWith('b/') ? target.slice(2) : target;
158
+ }
159
+ /**
160
+ * Walk a unified diff, calling `emit` once per NEW-SIDE line, in file order.
161
+ *
162
+ * The single parser behind {@link scopeDiff}, {@link addedContentByPath} and
163
+ * `visibleLinesByPath`. Those three read different things off the same walk —
164
+ * added line numbers, added text, and added-plus-context text — and each used
165
+ * to carry its own copy of the header/hunk bookkeeping, which is three places
166
+ * for a diff-format edge case to be fixed in two of.
167
+ *
168
+ * Removed (`-`) lines and the `` marker are not
169
+ * emitted and do not advance the new-side counter, since neither exists on that
170
+ * side. Deleted files (`+++ /dev/null`) emit nothing at all.
171
+ *
172
+ * FIX 7: `--- `/`+++ ` are file headers ONLY before the first hunk of a file.
173
+ * Inside a hunk body a `+++ ...` line is ADDED CONTENT whose real text is
174
+ * `++ ...`, and mistaking it for a header loses the rest of the file.
175
+ */
176
+ export function walkDiff(diffText, emit) {
177
+ let path = null;
178
+ let lineno = 0;
75
179
  let inHunk = false;
76
180
  for (const line of splitLines(diffText)) {
77
181
  if (line.startsWith('diff --git')) {
78
- // New file block begins → leave any prior hunk body; path is set by the
79
- // upcoming `+++ ` header.
182
+ // A new file block begins → leave any prior hunk body; the path is set
183
+ // by the upcoming `+++ ` header.
80
184
  inHunk = false;
81
- currentPath = null;
82
- skipCurrent = false;
185
+ path = null;
83
186
  continue;
84
187
  }
85
- // `--- `/`+++ ` are file headers ONLY before the first hunk of a file. Once
86
- // inside a hunk body a `+++ ...` line is an ADDED content line whose real
87
- // text is `++ ...` and must not be mistaken for a header (FIX 7).
88
188
  if (!inHunk && line.startsWith('+++ ')) {
89
- const target = line.slice(4).trim();
90
- if (target === '/dev/null') {
91
- skipCurrent = true;
92
- currentPath = null;
93
- continue;
94
- }
95
- skipCurrent = false;
96
- // Strip the conventional "b/" prefix.
97
- currentPath = target.startsWith('b/') ? target.slice(2) : target;
98
- if (!addedByPath.has(currentPath))
99
- addedByPath.set(currentPath, []);
189
+ path = headerPath(line);
100
190
  continue;
101
191
  }
102
- if (!inHunk && line.startsWith('--- ')) {
103
- // Old-file header; ignored (path comes from +++).
192
+ // Old-file header; ignored (the path comes from `+++`).
193
+ if (!inHunk && line.startsWith('--- '))
104
194
  continue;
105
- }
106
195
  const hunk = HUNK_RE.exec(line);
107
196
  if (hunk) {
108
- newLineno = Number.parseInt(hunk[1], 10);
197
+ lineno = Number.parseInt(hunk[1], 10);
109
198
  inHunk = true;
110
199
  continue;
111
200
  }
112
- if (skipCurrent || currentPath === null)
113
- continue;
114
- if (line.startsWith('+')) {
115
- addedByPath.get(currentPath).push(newLineno);
116
- newLineno += 1;
117
- }
118
- else if (line.startsWith('-')) {
119
- // Removed line: does not advance the new-file counter.
201
+ if (path === null)
120
202
  continue;
121
- }
122
- else if (line.startsWith('\\')) {
123
- // "" — metadata, ignore.
203
+ if (line.startsWith('-') || line.startsWith('\\'))
124
204
  continue;
125
- }
126
- else {
127
- // Context line (leading space) or blank — advances new-file counter.
128
- newLineno += 1;
129
- }
205
+ emit({ path, lineno, text: line.slice(1), added: line.startsWith('+') });
206
+ lineno += 1;
130
207
  }
131
- const units = [];
132
- for (const [path, lines] of addedByPath) {
133
- if (lines.length === 0)
134
- continue;
135
- units.push({ path, added_ranges: mergeLines(lines) });
136
- }
137
- return units;
138
208
  }
139
209
  // FIX 2 (signal-quality): a changed file whose ADDED lines are ONLY imports /
140
210
  // re-exports (no real declarations or logic) is a barrel/index file
@@ -173,48 +243,20 @@ function isNeutralLine(stripped) {
173
243
  /**
174
244
  * Map each changed file to the CONTENT of its added (`+`) lines.
175
245
  *
176
- * Mirrors {@link scopeDiff}'s parser but captures the added-line *text* (the
177
- * `+` stripped) rather than line numbers. Deleted files (`+++ /dev/null`) are
178
- * excluded; a `+++ ` line inside a hunk body is added content, not a header
179
- * (same FIX 7 guard as `scopeDiff`).
246
+ * The same walk {@link scopeDiff} takes, reading the added-line *text* rather
247
+ * than its line number.
180
248
  */
181
- function addedContentByPath(diffText) {
249
+ export function addedContentByPath(diffText) {
182
250
  const added = new Map();
183
- let currentPath = null;
184
- let skipCurrent = false;
185
- let inHunk = false;
186
- for (const line of splitLines(diffText)) {
187
- if (line.startsWith('diff --git')) {
188
- inHunk = false;
189
- currentPath = null;
190
- skipCurrent = false;
191
- continue;
192
- }
193
- if (!inHunk && line.startsWith('+++ ')) {
194
- const target = line.slice(4).trim();
195
- if (target === '/dev/null') {
196
- skipCurrent = true;
197
- currentPath = null;
198
- continue;
199
- }
200
- skipCurrent = false;
201
- currentPath = target.startsWith('b/') ? target.slice(2) : target;
202
- if (!added.has(currentPath))
203
- added.set(currentPath, []);
204
- continue;
205
- }
206
- if (!inHunk && line.startsWith('--- '))
207
- continue;
208
- if (HUNK_RE.test(line)) {
209
- inHunk = true;
210
- continue;
211
- }
212
- if (skipCurrent || currentPath === null)
213
- continue;
214
- if (line.startsWith('+')) {
215
- added.get(currentPath).push(line.slice(1));
216
- }
217
- }
251
+ walkDiff(diffText, (line) => {
252
+ if (!line.added)
253
+ return;
254
+ const texts = added.get(line.path);
255
+ if (texts)
256
+ texts.push(line.text);
257
+ else
258
+ added.set(line.path, [line.text]);
259
+ });
218
260
  return added;
219
261
  }
220
262
  /**
@@ -656,91 +698,63 @@ export function buildFindings(results) {
656
698
  }
657
699
  return [...findings].sort((a, b) => severitySortKey(a.severity) - severitySortKey(b.severity));
658
700
  }
659
- // Map a test file's extension to the framework whose assertion/test patterns
660
- // the quality scorer should use. Unknown → pytest (the scorer's own fallback).
661
- const TEST_FRAMEWORK_BY_EXT = {
662
- '.py': 'pytest',
663
- '.ts': 'vitest',
664
- '.tsx': 'vitest',
665
- '.js': 'vitest',
666
- '.jsx': 'vitest',
667
- '.mjs': 'vitest',
668
- '.cjs': 'vitest',
669
- };
670
- function frameworkForTestPath(path) {
671
- return TEST_FRAMEWORK_BY_EXT[extname(path).toLowerCase()] ?? 'pytest';
672
- }
673
- // A test-function signature / decorator / block-close / comment — lines that
674
- // are not a test *body*. If a diff's added lines are ONLY these (e.g. a rename
675
- // that adds just `def test_new():` while the asserting body stays as context),
676
- // there is no added body to judge and we must not flag it.
677
- const TEST_SIGNATURE_RE = /^\s*(?:async\s+)?def\s+test\w*\s*\(|^\s*(?:it|test|describe)\s*\(/;
678
- const BLOCK_DELIMITERS = new Set(['})', '});', '}', ')', '{']);
679
701
  /**
680
- * True iff the added lines contain a real body line — not just a test
681
- * signature, decorator, comment, or a bare block delimiter.
702
+ * Percentage-point bands for a coverage **regression** (#606).
703
+ *
704
+ * A drop is graded by how far it fell and stops at `HIGH` — it never reaches
705
+ * `CRITICAL`. An uncovered new block is a fact about one artifact; a drop is a
706
+ * *relative* measurement across two, and guardian cannot verify that the base
707
+ * artifact it was handed is genuinely the base of this PR. The top of the scale
708
+ * is reserved for what guardian can prove on its own.
682
709
  */
683
- function hasAddedTestBody(added) {
684
- for (const line of added) {
685
- const stripped = line.trim();
686
- if (!stripped)
687
- continue;
688
- if (stripped.startsWith('#') ||
689
- stripped.startsWith('//') ||
690
- stripped.startsWith('@') ||
691
- stripped.startsWith('*') ||
692
- stripped.startsWith('/*')) {
693
- continue;
694
- }
695
- if (BLOCK_DELIMITERS.has(stripped))
696
- continue;
697
- if (TEST_SIGNATURE_RE.test(line))
698
- continue;
699
- return true;
700
- }
701
- return false;
710
+ const REGRESSION_HIGH_POINTS = 20;
711
+ const REGRESSION_MEDIUM_POINTS = 5;
712
+ /** `92.0% (23/25)` — a ratio a reviewer can check without doing the division. */
713
+ function ratioLabel(ratio) {
714
+ const pct = ((ratio.covered / ratio.coverable) * 100).toFixed(1);
715
+ return `${pct}% (${ratio.covered}/${ratio.coverable})`;
702
716
  }
703
717
  /**
704
- * Advisory `weak-test` findings for ADDED tests that assert nothing.
718
+ * Turn regressed base-vs-head deltas into `coverage-regression` findings (#606).
705
719
  *
706
- * Consumes the test-path units {@link filterTestUnits} sets aside (a test file
707
- * needs no test of its own, but an added test that asserts nothing is itself a
708
- * gap). Scores only the diff's *added* lines per test file via
709
- * {@link isAssertionFreeTest} — a high-precision signal (a real test function
710
- * with zero assertions), so snapshot/table-driven tests are not flagged. These
711
- * findings are `LOW`/`weak-test` and are **never** gated (see
712
- * {@link computeExitCode}): they surface, never block.
720
+ * Only `regressed` deltas become findings — an improvement and a flat result
721
+ * are not news. Fidelity is `COVERAGE_VERIFIED` because both sides of the
722
+ * comparison were measured by a real coverage run; there is no graph or
723
+ * heuristic path to a delta, and a tier that cannot measure must not guess one.
724
+ *
725
+ * The evidence carries both ratios AND both raw counts on purpose: base and
726
+ * head can disagree about how many lines are coverable at all (a diff adds
727
+ * lines; two producers may instrument differently), and a bare pair of
728
+ * percentages would hide that.
713
729
  */
714
- export function buildWeakTestFindings(testUnits, diffText) {
715
- const addedByPath = addedContentByPath(diffText);
730
+ export function buildRegressionFindings(deltas) {
716
731
  const findings = [];
717
- for (const unit of testUnits) {
718
- const added = addedByPath.get(unit.path);
719
- if (!added || added.length === 0)
720
- continue;
721
- // A rename adds only the signature line (body is unchanged context) —
722
- // nothing new to judge, so don't flag it (FP guard).
723
- if (!hasAddedTestBody(added))
732
+ for (const delta of deltas) {
733
+ if (!delta.regressed)
724
734
  continue;
725
- const code = added.join('\n');
726
- const framework = frameworkForTestPath(unit.path);
727
- if (isAssertionFreeTest(code, framework)) {
728
- findings.push(new GuardianFinding({
729
- path: unit.path,
730
- unit: unit.path,
731
- kind: 'weak-test',
732
- fidelity: Fidelity.Heuristic,
733
- severity: Severity.LOW,
734
- evidence: 'added test asserts nothing (advisory — never blocks the gate)',
735
- suggestion: 'add at least one assertion, or delete the test if it is a placeholder.',
736
- added_ranges: [...unit.added_ranges],
737
- }));
738
- }
735
+ const points = delta.dropPoints;
736
+ const severity = points >= REGRESSION_HIGH_POINTS
737
+ ? Severity.HIGH
738
+ : points >= REGRESSION_MEDIUM_POINTS
739
+ ? Severity.MEDIUM
740
+ : Severity.LOW;
741
+ findings.push(new GuardianFinding({
742
+ path: delta.path,
743
+ unit: delta.path,
744
+ kind: 'coverage-regression',
745
+ fidelity: Fidelity.CoverageVerified,
746
+ severity,
747
+ evidence: `coverage fell ${points.toFixed(1)} points on a file this change ` +
748
+ `touches: base ${ratioLabel(delta.base)} ${ARROW} head ` +
749
+ `${ratioLabel(delta.head)}`,
750
+ suggestion: `Restore the lost coverage in \`${delta.path}\` — a test that ` +
751
+ 'exercised this file on the base ref no longer reaches part of it.',
752
+ }));
739
753
  }
740
- return findings;
754
+ return [...findings].sort((a, b) => severitySortKey(a.severity) - severitySortKey(b.severity));
741
755
  }
742
756
  /** Flatten inclusive `[start, end]` ranges into a sorted list of line numbers. */
743
- function linesInRanges(ranges) {
757
+ export function linesInRanges(ranges) {
744
758
  const lines = new Set();
745
759
  for (const [start, end] of ranges) {
746
760
  for (let ln = start; ln <= end; ln++)
@@ -754,7 +768,7 @@ function linesInRanges(ranges) {
754
768
  * Requires a `//`/`#` comment leader (FIX 1), strips a trailing inline-comment
755
769
  * close (e.g. `*​/`), and trims surrounding whitespace.
756
770
  */
757
- function suppressionReason(line) {
771
+ export function suppressionReason(line) {
758
772
  const match = SUPPRESS_RE.exec(line);
759
773
  if (match === null)
760
774
  return null;
@@ -846,26 +860,11 @@ export function computeExitCode(findings, gate) {
846
860
  return 0;
847
861
  }
848
862
  const STICKY_MARKER = '<!-- canary-pr-guardian -->';
849
- // Severity → status icon for the sticky comment (encodes severity in form, not
850
- // just text, so the most urgent findings read at a glance).
851
- //
852
- // Written as `\u{...}` escapes, not literal glyphs: this file is `.ts`, and the
853
- // house rule keeps emitted non-ASCII out of non-Markdown source (see the
854
- // "Output data glyphs" block in `cli.ts`). They are emitted verbatim.
855
863
  /**
856
- * Character budget for a rendered sticky comment (#457).
857
- *
858
- * GitHub rejects an issue/PR comment body over **65,536** characters. The post
859
- * path reports that as "could not post", so an over-long body means the gate
860
- * silently produces nothing on exactly the large PRs that need it most -- the
861
- * same silent-green failure #369 was filed for.
862
- *
863
- * 60,000 leaves ~5.5k of headroom for anything appended outside
864
- * `renderFindings` (degradation annotations, upsert wrappers) without inviting
865
- * a body that only *just* fits and then breaks when a filename grows.
866
- *
867
- * The cap applies ONLY to the comment. The `--emit-analysis` JSON record is the
868
- * authoritative complete set and is never truncated.
864
+ * Character budget for a rendered sticky comment (#457). GitHub rejects a body
865
+ * over 65,536 characters, which would silently post nothing on the largest PRs
866
+ * (#369); 60,000 leaves headroom for appended annotations. The comment only:
867
+ * the `--emit-analysis` record is complete and never truncated.
869
868
  */
870
869
  export const COMMENT_CHAR_BUDGET = 60_000;
871
870
  /** The line that accounts for findings the budget could not fit (#457). */
@@ -909,6 +908,26 @@ function findingDict(finding) {
909
908
  uncovered_lines: finding.uncovered_lines,
910
909
  };
911
910
  }
911
+ /** Short display form for a rev: 10 chars of a sha, a ref name verbatim. */
912
+ function shortRev(rev) {
913
+ if (!rev)
914
+ return '?';
915
+ return /^[0-9a-f]{40}$/i.test(rev) ? rev.slice(0, 10) : rev;
916
+ }
917
+ /**
918
+ * The one-line diff provenance shown on every surface (#761).
919
+ *
920
+ * Deliberately terse and always present — a line that appears only when
921
+ * something is wrong teaches readers to ignore it when it does appear.
922
+ */
923
+ export const MERGE_REF_WARNING = 'HEAD is a pull_request MERGE REF, not the PR head, so this diff spans ' +
924
+ 'commits merged into the base branch and is WIDER than the PR';
925
+ export function provenanceLine(p) {
926
+ const noun = p.fileCount === 1 ? 'file' : 'files';
927
+ const range = `${shortRev(p.base)}...${shortRev(p.head)}`;
928
+ const warn = p.mergeRef ? ` ${EM_DASH} ${MERGE_REF_WARNING}` : '';
929
+ return `Diff: \`${range}\` (${p.fileCount} ${noun}, via ${p.origin})${warn}`;
930
+ }
912
931
  /**
913
932
  * Join every degradation notice this run produced into one line, dropping the
914
933
  * empty ones. Notices are independent (the agent tier and the coverage input
@@ -923,19 +942,90 @@ function coverageBlock(state) {
923
942
  return { status: coverageStatus(state), ...state };
924
943
  }
925
944
  /**
926
- * The comment body for a run with zero active findings.
927
- *
928
- * #554: the ✅ all-clear headline is reserved for a run whose coverage report
929
- * spoke to every changed file. Anything less says so in the body — a `<sub>`
930
- * footer under a green headline is read as boilerplate, and this is the exact
931
- * shape that let 43 coverage-blind PRs read as covered.
945
+ * True when this run VERIFIED NO COVERAGE and every finding is a naming guess
946
+ * (#761): a confident count over a zero coverage denominator is an abstention.
947
+ * Not this case: `unitsTotal === 0` (no claim either way), any coverage- or
948
+ * graph-verified finding (real evidence), or no findings at all (the per-cause
949
+ * headline already says what was not verified, #928).
950
+ */
951
+ export function isCoverageAbstention(coverage, findings) {
952
+ if (!coverage || coverage.unitsTotal === 0)
953
+ return false;
954
+ if (coverageStatus(coverage) !== 'unavailable')
955
+ return false;
956
+ if (findings.length === 0)
957
+ return false;
958
+ return findings.every((f) => f.fidelity === Fidelity.Heuristic);
959
+ }
960
+ /**
961
+ * The abstention headline (#761) — states what was NOT verified, and never a
962
+ * count of findings, which is what reads as a measured result.
963
+ */
964
+ function abstentionHeadline(checked) {
965
+ const noun = checked === 1 ? 'file' : 'files';
966
+ return (`${WARNING} abstained: no coverage data ` +
967
+ `(${checked} ${noun} judged heuristically)`);
968
+ }
969
+ /** The body paragraph that stops the heuristic findings reading as a verdict. */
970
+ const ABSTENTION_BODY = 'No coverage report reached this run, so nothing below is a coverage ' +
971
+ 'verdict — every finding is a filename-level guess. A gate that verified ' +
972
+ 'zero items has abstained; this is not a pass.';
973
+ const files = (n) => `${n} file${n === 1 ? '' : 's'}`;
974
+ /** The unverified-unit causes present on a run, worst first (#928). */
975
+ function causeEntries(state) {
976
+ const { stale, scopeGap, nonCoverable } = coverageCauses(state);
977
+ const trees = (state.instrumentedTrees ?? []).map((t) => t || '.');
978
+ const entries = [
979
+ [
980
+ stale,
981
+ `${WARNING} coverage report stale: ${stale} changed ${files(stale).split(' ')[1]} missing from it`,
982
+ 'coverage report stale (changed lines missing from it)',
983
+ ],
984
+ [
985
+ scopeGap,
986
+ `${WARNING} not coverage-checked: ${files(scopeGap)} outside instrumented trees (${trees.join(', ')})`,
987
+ 'not coverage-checked (outside instrumented trees)',
988
+ ],
989
+ [
990
+ nonCoverable,
991
+ `${WHITE_CHECK} nothing to test: only non-executable lines changed (${files(nonCoverable)})`,
992
+ 'nothing to test (only non-executable lines changed)',
993
+ ],
994
+ ];
995
+ return entries.filter(([n]) => n > 0);
996
+ }
997
+ /**
998
+ * One headline per cause (#928) plus the lesser causes as count lines. ✅ only
999
+ * when the report spoke to every changed unit (#554): any B or C unit is ⚠️.
1000
+ */
1001
+ function coverageHeadline(state, notice) {
1002
+ const clean = `${WHITE_CHECK} no test-coverage gaps`;
1003
+ if (state?.unitsTotal === 0) {
1004
+ return [`${WHITE_CHECK} nothing to test: no source files changed`, []];
1005
+ }
1006
+ if (!state?.parsed) {
1007
+ const blind = `${WARNING} no gaps found, but coverage was ${state ? coverageStatus(state) : ''}`;
1008
+ return [notice ? blind : clean, []];
1009
+ }
1010
+ const [worst, ...rest] = causeEntries(state);
1011
+ // Matched units plus non-executable ones: nothing is missing, so plain ✅.
1012
+ if (!worst || (worst[1].startsWith(WHITE_CHECK) && state.unitsMatched > 0)) {
1013
+ return [clean, []];
1014
+ }
1015
+ return [worst[1], rest.map(([n, , label]) => `- ${files(n)}: ${label}`)];
1016
+ }
1017
+ /**
1018
+ * The comment body for a run with zero active findings. The notice goes in the
1019
+ * body, not only a `<sub>` footer, which is read as boilerplate (#554).
932
1020
  */
933
- function noGapsLines(coverageState, suppressedCount) {
1021
+ function noGapsLines(coverageState, suppressedCount, abstained = false, checked = 0) {
934
1022
  const notice = coverageState ? coverageDegradedNotice(coverageState) : null;
935
- const headline = notice
936
- ? `${WARNING} no gaps found, but coverage was ${coverageStatus(coverageState)}`
937
- : `${WHITE_CHECK} no test-coverage gaps`;
938
- const lines = [`## ${BABY_CHICK} Canary PR Guardian ${EM_DASH} ${headline}`];
1023
+ const [cause, counts] = coverageHeadline(coverageState, notice);
1024
+ const headline = abstained ? abstentionHeadline(checked) : cause;
1025
+ const lines = [
1026
+ `## ${BABY_CHICK} Canary PR Guardian ${EM_DASH} ${headline}`,
1027
+ ...counts,
1028
+ ];
939
1029
  if (notice) {
940
1030
  lines.push(`> **${notice}**`, '', 'Zero files matched is an abstention, not a pass — nothing here is ' +
941
1031
  'evidence that the changed lines are covered.');
@@ -947,12 +1037,18 @@ function noGapsLines(coverageState, suppressedCount) {
947
1037
  }
948
1038
  export function renderFindings(findings, fmt, tier = 0, degradedNotice = null, gateMeta = null, blobBase = null) {
949
1039
  const ordered = [...findings].sort((a, b) => severitySortKey(a.severity) - severitySortKey(b.severity));
1040
+ // #761: an abstained run never headlines a count, on any surface.
1041
+ const abstained = gateMeta?.abstained === true;
950
1042
  // #554: the coverage ladder's own degradation, stated alongside the tier's.
951
1043
  const coverageState = gateMeta?.coverage ?? null;
952
1044
  const coverageNotice = coverageState
953
1045
  ? coverageDegradedNotice(coverageState)
954
1046
  : null;
955
- const notice = combineNotices(degradedNotice, coverageNotice);
1047
+ // #606: the delta's own degradation is a THIRD independent reason a run can
1048
+ // be blind, so it joins the same combined notice rather than replacing one.
1049
+ const deltaState = gateMeta?.coverageDelta ?? null;
1050
+ const deltaNotice = deltaState ? coverageDeltaNotice(deltaState) : null;
1051
+ const notice = combineNotices(degradedNotice, coverageNotice, deltaNotice);
956
1052
  if (fmt === 'json') {
957
1053
  const payload = {
958
1054
  findings: ordered.map(findingDict),
@@ -970,6 +1066,18 @@ export function renderFindings(findings, fmt, tier = 0, degradedNotice = null, g
970
1066
  payload['skipped'] = gateMeta.skipped ?? [];
971
1067
  if (coverageState)
972
1068
  payload['coverage'] = coverageBlock(coverageState);
1069
+ // #606: the delta's denominator, so a machine consumer can tell "no
1070
+ // regressions" from "never compared" without re-deriving it.
1071
+ if (deltaState) {
1072
+ payload['coverage_delta'] = {
1073
+ status: coverageDeltaStatus(deltaState),
1074
+ ...deltaState,
1075
+ };
1076
+ }
1077
+ // #761: machine consumers need the diff's endpoints for the same reason
1078
+ // humans do — every count in this payload is scoped by them.
1079
+ if (gateMeta.provenance)
1080
+ payload['provenance'] = { ...gateMeta.provenance };
973
1081
  }
974
1082
  return ensureAscii(JSON.stringify(payload, null, 2));
975
1083
  }
@@ -1018,24 +1126,38 @@ export function renderFindings(findings, fmt, tier = 0, degradedNotice = null, g
1018
1126
  const CONFIDENCE_NOTE = 'Confidence — **coverage-verified**: measured from a real coverage run · ' +
1019
1127
  '**graph-verified**: inferred from the call graph · **heuristic**: filename ' +
1020
1128
  `guess (lowest). tier ${tier}: deterministic check, no LLM.`;
1021
- const footerLine = `<sub>${CONFIDENCE_NOTE}${notice ? ` ${EM_DASH} ${notice}` : ''}</sub>`;
1022
- // #554: a coverage-blind run must not present as a run that checked and found
1023
- // nothing. The notice goes in the BODY, not only the footer — a `<sub>` line
1024
- // under a green headline is read as boilerplate.
1129
+ // #928: the coverage notice is rendered in the body, so the footer omits it.
1130
+ const footerNotice = combineNotices(degradedNotice, deltaNotice);
1131
+ const footerLine = `<sub>${CONFIDENCE_NOTE}${footerNotice ? ` ${EM_DASH} ${footerNotice}` : ''}</sub>`;
1132
+ // #554: a coverage-blind run must not present as one that checked and passed.
1025
1133
  const coverageLine = coverageNotice ? `> **${coverageNotice}**` : null;
1134
+ // #761: shown on EVERY comment, clean or not. The run that motivated this was
1135
+ // a findings run whose findings were all phantom, so gating the line on a
1136
+ // problem guardian had not detected would have hidden it exactly when needed.
1137
+ const provLine = gateMeta?.provenance
1138
+ ? `<sub>${provenanceLine(gateMeta.provenance)}</sub>`
1139
+ : null;
1026
1140
  if (fmt === 'comment') {
1027
1141
  const fileCount = new Set(active.map((f) => f.path)).size;
1028
1142
  const lines = [STICKY_MARKER];
1029
1143
  if (active.length === 0) {
1030
- lines.push(...noGapsLines(coverageState, suppressed.length));
1144
+ lines.push(...noGapsLines(coverageState, suppressed.length, abstained, gateMeta?.checked ?? 0));
1031
1145
  }
1032
1146
  else {
1033
1147
  const noun = fileCount === 1 ? 'file needs' : 'files need';
1148
+ // #761: on an abstained run the headline states the abstention instead of
1149
+ // a count. The findings stay in the table below — they are useful, they
1150
+ // are just not a coverage verdict, and a count headline is exactly what
1151
+ // makes a reader take them for one.
1034
1152
  lines.push(`## ${BABY_CHICK} Canary PR Guardian ${EM_DASH} ` +
1035
- `${fileCount} ${noun} test coverage`);
1036
- lines.push('These lines were changed by this PR but no test exercises them. Add or ' +
1037
- 'extend a test that covers them, or mark the line ' +
1038
- '`// canary:allow-untested <reason>` if it is intentionally untested.');
1153
+ (abstained
1154
+ ? abstentionHeadline(gateMeta?.checked ?? 0)
1155
+ : `${fileCount} ${noun} test coverage`));
1156
+ lines.push(abstained
1157
+ ? ABSTENTION_BODY
1158
+ : 'These lines were changed by this PR but no test exercises them. Add or ' +
1159
+ 'extend a test that covers them, or mark the line ' +
1160
+ '`// canary:allow-untested <reason>` if it is intentionally untested.');
1039
1161
  if (coverageLine)
1040
1162
  lines.push('', coverageLine);
1041
1163
  lines.push('', '| Sev | File | What is uncovered, and what to do | Confidence |', '| --- | --- | --- | --- |');
@@ -1069,24 +1191,41 @@ export function renderFindings(findings, fmt, tier = 0, degradedNotice = null, g
1069
1191
  lines.push('', `<sub>${suppressed.length} finding(s) suppressed as intentional and not counted above.</sub>`);
1070
1192
  }
1071
1193
  }
1194
+ // Directly above the confidence footer: provenance and confidence are the
1195
+ // two "how much should I trust this" facts, so they read as one block.
1196
+ if (provLine)
1197
+ lines.push('', provLine);
1072
1198
  lines.push('', footerLine);
1073
1199
  return lines.join('\n');
1074
1200
  }
1075
1201
  // fmt == "text" (default fallback): plain, no markdown/HTML.
1076
- const cleanHeadline = coverageNotice
1077
- ? // #554: same rule as the comment surface — a blind run never claims clean.
1078
- `Canary PR Guardian — no gaps found, but coverage was ${coverageStatus(coverageState)}`
1079
- : 'Canary PR Guardian — no test-coverage gaps';
1202
+ // #554/#928: the comment surface's per-cause headline, without its glyph.
1203
+ const [cause, causeCounts] = coverageHeadline(coverageState, coverageNotice);
1204
+ const cleanHeadline = `Canary PR Guardian — ${cause.replace(/^\S+ /, '')}`;
1205
+ // #761: the same rule on the surface an engineer reads at their desk. The
1206
+ // headline is stripped of the comment surface's markdown-era glyph so the
1207
+ // terminal line stays plain text.
1208
+ const textAbstention = `Canary PR Guardian — ` +
1209
+ abstentionHeadline(gateMeta?.checked ?? 0).replace(`${WARNING} `, '');
1080
1210
  const lines = [
1081
- active.length === 0
1082
- ? cleanHeadline
1083
- : `Canary PR Guardian — ${new Set(active.map((f) => f.path)).size} file(s) need test coverage`,
1211
+ abstained
1212
+ ? textAbstention
1213
+ : active.length === 0
1214
+ ? cleanHeadline
1215
+ : `Canary PR Guardian — ${new Set(active.map((f) => f.path)).size} file(s) need test coverage`,
1216
+ ...(abstained || active.length > 0 ? [] : causeCounts),
1084
1217
  ];
1085
1218
  for (const finding of ordered) {
1086
1219
  const unit = finding.unit && finding.unit !== finding.path ? ` → ${finding.unit}` : '';
1087
1220
  const mark = finding.suppressed ? ' (suppressed)' : '';
1088
1221
  lines.push(`[${finding.severity}] ${finding.path}${unit} — ${finding.evidence} (${finding.fidelity})${mark}`);
1089
1222
  }
1223
+ // #761: the terminal surface gets the same provenance the comment does —
1224
+ // this is the one an engineer reads at their desk, where a wrong `--diff` is
1225
+ // likeliest.
1226
+ if (gateMeta?.provenance) {
1227
+ lines.push(provenanceLine(gateMeta.provenance).replace(/`/g, ''));
1228
+ }
1090
1229
  let footer = `tier ${tier}: deterministic check, no LLM`;
1091
1230
  if (notice)
1092
1231
  footer += ` - ${notice}`;
@@ -1142,16 +1281,8 @@ export const DEFAULT_SKIP_GLOBS = [
1142
1281
  '**/__generated__/**',
1143
1282
  ];
1144
1283
  /**
1145
- * Parsed `canary.guardian` config block.
1146
- *
1147
- * Phase 1 stores every field but only `pr_*` gate/tier drive behavior.
1148
- * `skip_globs` and the `precommit_*`/`coverage_paths` fields are read into the
1149
- * object (scaffold) for later phases (SC-2 skip, SC-5 tier).
1150
- *
1151
- * `skip_globs` defaults to docs/markdown PLUS generated/dependency artifacts
1152
- * (lockfiles, `dist`/`build` outputs, minified JS, snapshots — see
1153
- * {@link DEFAULT_SKIP_GLOBS}) so noise-only paths skip out of the box; an
1154
- * explicit `skipGlobs` in config (even `[]`) overrides it.
1284
+ * Parsed `canary.guardian` config block. `skip_globs` defaults to
1285
+ * {@link DEFAULT_SKIP_GLOBS}; an explicit `skipGlobs` (even `[]`) overrides it.
1155
1286
  */
1156
1287
  export class GuardianConfig {
1157
1288
  pr_enabled;