mandrel 2.64.0 → 2.66.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.agents/agents/acceptance-critic.md +8 -7
  2. package/.agents/agents/auditor.md +20 -20
  3. package/.agents/agents/plan-critic.md +8 -7
  4. package/.agents/agents/story-worker.md +7 -7
  5. package/.agents/audit-checklists/quality.md +3 -0
  6. package/.agents/docs/agentrc-reference.json +1 -9
  7. package/.agents/docs/configuration.md +8 -7
  8. package/.agents/docs/execution-reference.md +27 -5
  9. package/.agents/instructions.md +10 -12
  10. package/.agents/rules/ci-remediation.md +3 -3
  11. package/.agents/rules/gherkin-standards.md +3 -2
  12. package/.agents/rules/git-conventions-reference.md +12 -3
  13. package/.agents/rules/git-conventions.md +9 -7
  14. package/.agents/rules/testing-standards.md +8 -7
  15. package/.agents/runtime-deps.json +1 -1
  16. package/.agents/schemas/agentrc.schema.json +6 -13
  17. package/.agents/schemas/audit-rules.schema.json +1 -1
  18. package/.agents/schemas/story-deliver-terminal.schema.json +5 -0
  19. package/.agents/scripts/bootstrap.js +102 -91
  20. package/.agents/scripts/check-context-budget.js +1 -1
  21. package/.agents/scripts/lib/ITicketingProvider.js +1 -3
  22. package/.agents/scripts/lib/audit-suite/findings.js +1 -17
  23. package/.agents/scripts/lib/audit-suite/frontmatter.js +0 -28
  24. package/.agents/scripts/lib/audit-suite/index.js +0 -6
  25. package/.agents/scripts/lib/audit-suite/selector.js +0 -31
  26. package/.agents/scripts/lib/baselines/duplication-scanner.js +17 -7
  27. package/.agents/scripts/lib/bootstrap/agents-md-fold.js +156 -0
  28. package/.agents/scripts/lib/bootstrap/commit-push.js +1 -1
  29. package/.agents/scripts/lib/bootstrap/manifest.js +2 -2
  30. package/.agents/scripts/lib/bootstrap/project-bootstrap.js +91 -107
  31. package/.agents/scripts/lib/cli/standard-args.js +60 -76
  32. package/.agents/scripts/lib/cli-args.js +26 -0
  33. package/.agents/scripts/lib/config/gates/shared.js +3 -3
  34. package/.agents/scripts/lib/config/review-chain-default.js +13 -0
  35. package/.agents/scripts/lib/config-settings-schema-delivery.js +2 -2
  36. package/.agents/scripts/lib/config-settings-schema-quality.js +11 -13
  37. package/.agents/scripts/lib/doc-tiers.js +25 -6
  38. package/.agents/scripts/lib/feedback-loop/graduate-steps.js +205 -0
  39. package/.agents/scripts/lib/feedback-loop/graduator-core.js +47 -782
  40. package/.agents/scripts/lib/feedback-loop/graduator-gh.js +449 -0
  41. package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
  42. package/.agents/scripts/lib/observability/close-telemetry.js +330 -0
  43. package/.agents/scripts/lib/observability/metrics-ledger.js +0 -72
  44. package/.agents/scripts/lib/observability/runtime-friction.js +2 -0
  45. package/.agents/scripts/lib/observability/signal-validator.js +17 -5
  46. package/.agents/scripts/lib/orchestration/code-review.js +33 -6
  47. package/.agents/scripts/lib/orchestration/epic-rollup.js +29 -12
  48. package/.agents/scripts/lib/orchestration/merge-block-class.js +20 -4
  49. package/.agents/scripts/lib/orchestration/merge-poll.js +41 -22
  50. package/.agents/scripts/lib/orchestration/plan-metrics.js +76 -63
  51. package/.agents/scripts/lib/orchestration/required-checks.js +147 -0
  52. package/.agents/scripts/lib/orchestration/review-providers/code-review.js +203 -0
  53. package/.agents/scripts/lib/orchestration/review-providers/review-provider-factory.js +29 -4
  54. package/.agents/scripts/lib/orchestration/review-providers/security-review.js +3 -2
  55. package/.agents/scripts/lib/orchestration/run-epilogue.js +6 -0
  56. package/.agents/scripts/lib/orchestration/single-story-close/failed-terminal.js +1 -0
  57. package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +2 -12
  58. package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +370 -268
  59. package/.agents/scripts/lib/orchestration/single-story-close/phases/options.js +21 -7
  60. package/.agents/scripts/lib/orchestration/single-story-close/phases/post-land.js +112 -82
  61. package/.agents/scripts/lib/orchestration/single-story-close/phases/review-override.js +4 -0
  62. package/.agents/scripts/lib/orchestration/single-story-close/runner.js +393 -313
  63. package/.agents/scripts/lib/orchestration/story-close/phases/review-core.js +12 -87
  64. package/.agents/scripts/lib/orchestration/story-deliver-terminal.js +3 -0
  65. package/.agents/scripts/lib/orchestration/ticket-validator.js +19 -36
  66. package/.agents/scripts/lib/signals/detectors/common.js +63 -51
  67. package/.agents/scripts/lib/templates/decomposer-prompts.js +5 -24
  68. package/.agents/scripts/lib/transpile.js +28 -3
  69. package/.agents/scripts/providers/github/issues.js +14 -23
  70. package/.agents/scripts/single-story-close.js +10 -2
  71. package/.agents/scripts/single-story-confirm-merge.js +267 -238
  72. package/.agents/scripts/sync-claude-agents.js +1 -1
  73. package/.agents/skills/core/idea-refinement/SKILL.md +6 -6
  74. package/.agents/skills/stack/qa/qa-harness/SKILL.md +1 -2
  75. package/.agents/workflows/audit-architecture.md +5 -4
  76. package/.agents/workflows/audit-documentation.md +5 -5
  77. package/.agents/workflows/audit-performance.md +10 -10
  78. package/.agents/workflows/audit-quality.md +42 -7
  79. package/.agents/workflows/helpers/acceptance-self-eval.md +9 -9
  80. package/.agents/workflows/helpers/audit-lens-core.md +30 -57
  81. package/.agents/workflows/helpers/code-review.md +15 -38
  82. package/.agents/workflows/helpers/deliver-digest.md +2 -2
  83. package/.agents/workflows/helpers/deliver-reference.md +7 -3
  84. package/.agents/workflows/helpers/deliver-story.md +9 -1
  85. package/.agents/workflows/helpers/parallel-tooling.md +16 -18
  86. package/.agents/workflows/helpers/plan-reference.md +9 -8
  87. package/.agents/workflows/mandrel-deliver.md +3 -2
  88. package/.agents/workflows/mandrel-plan.md +11 -7
  89. package/.agents/workflows/mandrel-update.md +5 -3
  90. package/docs/CHANGELOG.md +57 -0
  91. package/lib/cli/claude-code-version.js +73 -0
  92. package/lib/cli/doctor.js +2 -2
  93. package/lib/cli/guarded-sync.js +87 -0
  94. package/lib/cli/registry.js +9 -0
  95. package/lib/cli/sync-agents.js +9 -92
  96. package/lib/cli/sync-commands.js +9 -101
  97. package/lib/cli/uninstall.js +37 -9
  98. package/lib/migrations/index.js +2 -0
  99. package/lib/migrations/steps/2.65.0-fold-claude-md-into-agents-md.js +38 -0
  100. package/package.json +3 -2
  101. package/.agents/scripts/lib/audit-suite/lens-diff-floor.js +0 -99
  102. package/.agents/scripts/lib/audit-suite/runner.js +0 -205
  103. package/.agents/scripts/lib/audit-suite/substitutions.js +0 -96
  104. package/.agents/scripts/lib/audit-suite/workflow-loader.js +0 -37
  105. package/.agents/scripts/lib/orchestration/story-close/phases/local-lens-review.js +0 -234
@@ -8,13 +8,29 @@ import { checkVerdict, classifyRollupEntry } from './check-state.js';
8
8
 
9
9
  /** Fixed poll interval; default for `delivery.mergeWatch.maxBudgetSeconds`. */
10
10
  export const DEFAULT_INTERVAL_SECONDS = 30;
11
+
12
+ /** Checks green, PR unmerged: the merge is imminent, so observe it sooner. */
13
+ const GREEN_INTERVAL_SECONDS = 10;
14
+
15
+ /**
16
+ * @param {string|undefined} checksStatus
17
+ * @param {number} intervalSeconds The in-flight cadence.
18
+ * @returns {number} Milliseconds until the next poll.
19
+ */
20
+ export function pollIntervalMs(checksStatus, intervalSeconds) {
21
+ const seconds =
22
+ checksStatus === 'success'
23
+ ? Math.min(GREEN_INTERVAL_SECONDS, intervalSeconds)
24
+ : intervalSeconds;
25
+ return seconds * 1000;
26
+ }
11
27
  export const DEFAULT_MAX_BUDGET_SECONDS = 3600;
12
28
 
13
29
  /** Bounds every `gh` spawn so a hang degrades to the probe-error path. */
14
30
  export const MERGE_WAIT_GH_TIMEOUT_MS = 60_000;
15
31
 
16
32
  /**
17
- * Aggregate over EVERY check (the rollup has no `isRequired`): `failure`
33
+ * Aggregate over EVERY check (the view rollup has no `isRequired`): `failure`
18
34
  * means "something is red", not "blocked" — see {@link failingChecksBlockMerge}.
19
35
  */
20
36
  export function deriveChecksStatus(statusCheckRollup) {
@@ -47,7 +63,7 @@ export function isPrMerged(pr) {
47
63
  * @param {{ conclusion?: string, state?: string }} [check]
48
64
  * @returns {string|null}
49
65
  */
50
- function redConclusionOf(check) {
66
+ export function redConclusionOf(check) {
51
67
  const conclusion = String(check?.conclusion ?? '').toUpperCase();
52
68
  if (conclusion === 'FAILURE' || conclusion === 'ERROR') return conclusion;
53
69
  const state = String(check?.state ?? '').toUpperCase();
@@ -61,17 +77,29 @@ function redConclusionOf(check) {
61
77
  * @param {{ name?: string, context?: string }} [check]
62
78
  * @returns {string|null}
63
79
  */
64
- function readRunName(check) {
80
+ export function readRunName(check) {
65
81
  for (const value of [check?.name, check?.context]) {
66
82
  if (typeof value === 'string' && value) return value;
67
83
  }
68
84
  return null;
69
85
  }
70
86
 
87
+ /**
88
+ * @param {{ status?: string, state?: string }} [check]
89
+ * @returns {boolean}
90
+ */
91
+ export function isRunInFlight(check) {
92
+ const status = String(check?.status ?? '').toUpperCase();
93
+ // `status` is empty on a StatusContext, so it falls to the `state` branch.
94
+ if (status) return status !== 'COMPLETED';
95
+ const state = String(check?.state ?? '').toUpperCase();
96
+ return state === 'PENDING' || state === 'EXPECTED';
97
+ }
98
+
71
99
  /**
72
100
  * Head-anchored evidence; `null` on an empty rollup (the caller falls back
73
- * to consecutive probes). Reads EVERY run despite the names — required-ness
74
- * is attributed via `BLOCKED` in {@link requiredCheckFailedBlocksMerge}.
101
+ * to consecutive probes). Reads EVERY run — the unscoped rule used when
102
+ * GitHub's required attribution is unavailable (see `required-checks.js`).
75
103
  *
76
104
  * @param {Array<{status?: string, conclusion?: string, state?: string}>} statusCheckRollup
77
105
  * @returns {{ requiredRunFailed: boolean, requiredRunInFlight: boolean } | null}
@@ -80,22 +108,12 @@ export function deriveRequiredRunEvidence(statusCheckRollup) {
80
108
  if (!Array.isArray(statusCheckRollup) || statusCheckRollup.length === 0) {
81
109
  return null;
82
110
  }
83
- let requiredRunFailed = false;
84
- let requiredRunInFlight = false;
85
- for (const check of statusCheckRollup) {
86
- const status = String(check?.status ?? '').toUpperCase();
87
- const state = String(check?.state ?? '').toUpperCase();
88
- // `status` is empty on a StatusContext, so it falls to the `state` branch.
89
- if (status && status !== 'COMPLETED') {
90
- requiredRunInFlight = true;
91
- } else if (state === 'PENDING' || state === 'EXPECTED') {
92
- requiredRunInFlight = true;
93
- }
94
- if (redConclusionOf(check)) {
95
- requiredRunFailed = true;
96
- }
97
- }
98
- return { requiredRunFailed, requiredRunInFlight };
111
+ return {
112
+ requiredRunFailed: statusCheckRollup.some(
113
+ (c) => redConclusionOf(c) !== null,
114
+ ),
115
+ requiredRunInFlight: statusCheckRollup.some(isRunInFlight),
116
+ };
99
117
  }
100
118
 
101
119
  /** The `mergeStateStatus` meaning GitHub itself gates the merge. */
@@ -139,7 +157,8 @@ export function formatChecksFailedReason(prProbe, evidencePath) {
139
157
 
140
158
  /**
141
159
  * A genuinely red REQUIRED check: gated, no review owns `BLOCKED`, a run is
142
- * red and none in flight. No evidence → false (consecutive-probe path).
160
+ * red and none in flight — with GitHub attribution, a required run is red and
161
+ * no re-run of it is in flight. No evidence → false (consecutive-probe path).
143
162
  *
144
163
  * @param {{ checksStatus?: string, mergeStateStatus?: string,
145
164
  * reviewDecision?: string,
@@ -231,75 +231,88 @@ function recordTimestamp(entry) {
231
231
  * }|null}
232
232
  */
233
233
  export function summarizePlanMetrics(ledger, opts = {}) {
234
- const all = ledger?.entries ?? [];
235
- const since = typeof opts.since === 'string' ? opts.since : null;
236
- // ISO-8601 UTC strings compare correctly as strings.
237
- const entries =
238
- since === null
239
- ? all
240
- : all.filter((e) => {
241
- const stamp = recordTimestamp(e);
242
- return stamp !== null && stamp >= since;
243
- });
234
+ const entries = scopeEntries(ledger?.entries ?? [], opts.since);
244
235
  if (entries.length === 0) return null;
245
- const byCli = {};
246
- const byMode = {};
247
- const criticSkipsByCritic = {};
248
- let criticSkips = 0;
249
- let failures = 0;
250
- let totalDurationMs = 0;
251
- let firstStartedAt = null;
252
- let lastEndedAt = null;
253
- const invocationEntries = [];
254
- for (const e of entries) {
255
- if (e.kind === PLAN_METRICS_KIND_CRITIC_SKIP) {
256
- criticSkips += 1;
257
- if (typeof e.critic === 'string') {
258
- criticSkipsByCritic[e.critic] =
259
- (criticSkipsByCritic[e.critic] ?? 0) + 1;
260
- }
261
- continue;
262
- }
263
- if (typeof e.kind === 'string') {
264
- // Other kinded records (e.g. `findings-yield`) are not invocations.
265
- continue;
266
- }
267
- invocationEntries.push(e);
268
- byCli[e.cli] = (byCli[e.cli] ?? 0) + 1;
269
- if (typeof e.mode === 'string') byMode[e.mode] = (byMode[e.mode] ?? 0) + 1;
270
- if (e.ok !== true) failures += 1;
271
- if (typeof e.durationMs === 'number') totalDurationMs += e.durationMs;
272
- if (typeof e.startedAt === 'string') {
273
- if (firstStartedAt === null || e.startedAt < firstStartedAt) {
274
- firstStartedAt = e.startedAt;
275
- }
276
- }
277
- if (typeof e.endedAt === 'string') {
278
- if (lastEndedAt === null || e.endedAt > lastEndedAt) {
279
- lastEndedAt = e.endedAt;
280
- }
281
- }
282
- }
283
- let spanMs = null;
284
- if (firstStartedAt !== null && lastEndedAt !== null) {
285
- const span = Date.parse(lastEndedAt) - Date.parse(firstStartedAt);
286
- if (Number.isFinite(span)) spanMs = Math.max(0, span);
287
- }
236
+ const skips = tallyCriticSkips(entries);
237
+ const invocations = tallyInvocations(
238
+ entries.filter((e) => typeof e.kind !== 'string'),
239
+ );
288
240
  return {
289
- invocations: invocationEntries.length,
290
- failures,
291
- byCli,
292
- byMode,
293
- criticSkips,
294
- criticSkipsByCritic,
295
- firstStartedAt,
296
- lastEndedAt,
297
- spanMs,
298
- totalDurationMs,
241
+ invocations: invocations.count,
242
+ failures: invocations.failures,
243
+ byCli: invocations.byCli,
244
+ byMode: invocations.byMode,
245
+ criticSkips: skips.count,
246
+ criticSkipsByCritic: skips.byCritic,
247
+ firstStartedAt: invocations.firstStartedAt,
248
+ lastEndedAt: invocations.lastEndedAt,
249
+ spanMs: spanBetween(invocations.firstStartedAt, invocations.lastEndedAt),
250
+ totalDurationMs: invocations.totalDurationMs,
299
251
  malformedLines: ledger?.malformedLines ?? 0,
300
252
  };
301
253
  }
302
254
 
255
+ function scopeEntries(all, since) {
256
+ if (typeof since !== 'string') return all;
257
+ // ISO-8601 UTC strings compare correctly as strings.
258
+ return all.filter((e) => {
259
+ const stamp = recordTimestamp(e);
260
+ return stamp !== null && stamp >= since;
261
+ });
262
+ }
263
+
264
+ function increment(counts, key) {
265
+ counts[key] = (counts[key] ?? 0) + 1;
266
+ }
267
+
268
+ function tallyCriticSkips(entries) {
269
+ const byCritic = {};
270
+ let count = 0;
271
+ for (const e of entries) {
272
+ if (e.kind !== PLAN_METRICS_KIND_CRITIC_SKIP) continue;
273
+ count += 1;
274
+ if (typeof e.critic === 'string') increment(byCritic, e.critic);
275
+ }
276
+ return { count, byCritic };
277
+ }
278
+
279
+ function earlierStamp(current, candidate) {
280
+ if (typeof candidate !== 'string') return current;
281
+ return current === null || candidate < current ? candidate : current;
282
+ }
283
+
284
+ function laterStamp(current, candidate) {
285
+ if (typeof candidate !== 'string') return current;
286
+ return current === null || candidate > current ? candidate : current;
287
+ }
288
+
289
+ function tallyInvocations(entries) {
290
+ const tally = {
291
+ count: entries.length,
292
+ failures: 0,
293
+ byCli: {},
294
+ byMode: {},
295
+ totalDurationMs: 0,
296
+ firstStartedAt: null,
297
+ lastEndedAt: null,
298
+ };
299
+ for (const e of entries) {
300
+ increment(tally.byCli, e.cli);
301
+ if (typeof e.mode === 'string') increment(tally.byMode, e.mode);
302
+ if (e.ok !== true) tally.failures += 1;
303
+ if (typeof e.durationMs === 'number') tally.totalDurationMs += e.durationMs;
304
+ tally.firstStartedAt = earlierStamp(tally.firstStartedAt, e.startedAt);
305
+ tally.lastEndedAt = laterStamp(tally.lastEndedAt, e.endedAt);
306
+ }
307
+ return tally;
308
+ }
309
+
310
+ function spanBetween(first, last) {
311
+ if (first === null || last === null) return null;
312
+ const span = Date.parse(last) - Date.parse(first);
313
+ return Number.isFinite(span) ? Math.max(0, span) : null;
314
+ }
315
+
303
316
  /**
304
317
  * @param {ReturnType<typeof summarizePlanMetrics>} summary
305
318
  * @returns {string}
@@ -0,0 +1,147 @@
1
+ /**
2
+ * GitHub's per-PR required-check attribution (GraphQL `isRequired`), not
3
+ * `.agentrc` requiredChecks, which are local command names.
4
+ */
5
+
6
+ import {
7
+ deriveRequiredRunEvidence,
8
+ failingChecksBlockMerge,
9
+ isRunInFlight,
10
+ MERGE_WAIT_GH_TIMEOUT_MS,
11
+ readRunName,
12
+ redConclusionOf,
13
+ } from './merge-poll.js';
14
+
15
+ /**
16
+ * A required run is red and no re-run of that same check is in flight.
17
+ *
18
+ * @param {Array<object>} statusCheckRollup non-empty
19
+ * @param {Set<string>} requiredNames
20
+ */
21
+ function deriveAttributedEvidence(statusCheckRollup, requiredNames) {
22
+ const failedRequired = new Set();
23
+ for (const check of statusCheckRollup) {
24
+ const name = readRunName(check);
25
+ if (name && requiredNames.has(name) && redConclusionOf(check)) {
26
+ failedRequired.add(name);
27
+ }
28
+ }
29
+ let requiredRunInFlight = false;
30
+ let runInFlight = false;
31
+ for (const check of statusCheckRollup) {
32
+ if (!isRunInFlight(check)) continue;
33
+ runInFlight = true;
34
+ if (failedRequired.has(readRunName(check))) requiredRunInFlight = true;
35
+ }
36
+ return {
37
+ requiredRunFailed: failedRequired.size > 0,
38
+ requiredRunInFlight,
39
+ runInFlight,
40
+ attribution: 'github',
41
+ };
42
+ }
43
+
44
+ const REQUIRED_CHECKS_QUERY =
45
+ 'query($id: ID!, $n: Int!) { node(id: $id) { ... on PullRequest { ' +
46
+ 'commits(last: 1) { nodes { commit { statusCheckRollup { ' +
47
+ 'contexts(first: 100) { nodes { __typename ' +
48
+ '... on CheckRun { name isRequired(pullRequestNumber: $n) } ' +
49
+ '... on StatusContext { context isRequired(pullRequestNumber: $n) } ' +
50
+ '} } } } } } } } }';
51
+
52
+ /**
53
+ * @param {object|string} result `gh api graphql` output
54
+ * @returns {Set<string>}
55
+ */
56
+ function parseRequiredNames(result) {
57
+ const text = typeof result === 'string' ? result : result?.stdout;
58
+ const parsed = JSON.parse(String(text ?? ''));
59
+ if (Array.isArray(parsed?.errors) && parsed.errors.length > 0) {
60
+ throw new Error('graphql errors reading required checks');
61
+ }
62
+ const nodes =
63
+ parsed?.data?.node?.commits?.nodes?.[0]?.commit?.statusCheckRollup?.contexts
64
+ ?.nodes;
65
+ if (!Array.isArray(nodes)) throw new Error('required-check contexts absent');
66
+ const names = new Set();
67
+ for (const node of nodes) {
68
+ const name = readRunName(node);
69
+ if (node?.isRequired === true && name) names.add(name);
70
+ }
71
+ return names;
72
+ }
73
+
74
+ /** Per-head cache: required attribution is read at most once per PR head. */
75
+ const requiredNamesCache = new Map();
76
+
77
+ /**
78
+ * Required check names, cached per PR head. `null` on any failure.
79
+ *
80
+ * @param {{ prNodeId?: string, prNumber: number|string, headSha?: string,
81
+ * gh: { api: Function }, timeoutMs?: number }} args
82
+ * @returns {Promise<Set<string>|null>}
83
+ */
84
+ async function readRequiredCheckNames({
85
+ prNodeId,
86
+ prNumber,
87
+ headSha,
88
+ gh,
89
+ timeoutMs = MERGE_WAIT_GH_TIMEOUT_MS,
90
+ }) {
91
+ if (!prNodeId || !headSha) return null;
92
+ const key = `${prNodeId}@${headSha}`;
93
+ if (requiredNamesCache.has(key)) return requiredNamesCache.get(key);
94
+ try {
95
+ const names = parseRequiredNames(
96
+ await gh.api({
97
+ method: 'POST',
98
+ endpoint: 'graphql',
99
+ body: {
100
+ query: REQUIRED_CHECKS_QUERY,
101
+ variables: { id: prNodeId, n: Number(prNumber) },
102
+ },
103
+ execOpts: { timeoutMs },
104
+ }),
105
+ );
106
+ requiredNamesCache.set(key, names);
107
+ return names;
108
+ } catch {
109
+ return null;
110
+ }
111
+ }
112
+
113
+ /**
114
+ * Scoped evidence when a red gates the merge and attribution reads; else
115
+ * the unscoped rule.
116
+ *
117
+ * @param {{ view?: object, checksStatus?: string, prNumber: number|string,
118
+ * gh: object, ghTimeoutMs?: number, readFn?: Function }} args
119
+ * @returns {Promise<object|null>}
120
+ */
121
+ export async function readProbeRunEvidence({
122
+ view,
123
+ checksStatus,
124
+ prNumber,
125
+ gh,
126
+ ghTimeoutMs,
127
+ readFn = readRequiredCheckNames,
128
+ }) {
129
+ const rollup = view?.statusCheckRollup;
130
+ const gated = failingChecksBlockMerge({
131
+ checksStatus,
132
+ mergeStateStatus: view?.mergeStateStatus,
133
+ });
134
+ const names = gated
135
+ ? await readFn({
136
+ prNodeId: view?.id,
137
+ prNumber,
138
+ headSha: view?.headRefOid,
139
+ gh,
140
+ timeoutMs: ghTimeoutMs,
141
+ })
142
+ : null;
143
+ if (names instanceof Set && Array.isArray(rollup) && rollup.length > 0) {
144
+ return deriveAttributedEvidence(rollup, names);
145
+ }
146
+ return deriveRequiredRunEvidence(rollup);
147
+ }
@@ -0,0 +1,203 @@
1
+ /**
2
+ * review-providers/code-review.js — a low-effort model bug review of the
3
+ * Story diff through `claude --print --effort low`. It asks only for
4
+ * merge-blocking problems, so every parsed finding is `critical` and halts
5
+ * close before auto-merge. The reviewer is handed the diff range, the Story
6
+ * id and the diff text — never `acceptance[]` or the self-eval verdict: this
7
+ * is a bug review, not a second acceptance scoring.
8
+ *
9
+ * A missing CLI throws at construction (the default chain entry is
10
+ * `optional: true`, so hosts without it skip). A review that cannot run or
11
+ * whose output does not parse degrades to one non-halting `suggestion`.
12
+ *
13
+ * @typedef {import('./types.js').Finding} Finding
14
+ * @typedef {import('./types.js').ReviewInput} ReviewInput
15
+ * @typedef {import('./types.js').ReviewProvider} ReviewProvider
16
+ */
17
+
18
+ import { spawnCapture } from '../../child-exec.js';
19
+ import { gitSpawn } from '../../git-utils.js';
20
+ import { PROJECT_ROOT } from '../../project-root.js';
21
+ import { parseProviderFindings } from './parse-findings.js';
22
+ import { renderDepthDirective } from './review-depth.js';
23
+ import { probeClaudeCli } from './security-review.js';
24
+
25
+ const CLAUDE_ARGS = Object.freeze(['--print', '--effort', 'low']);
26
+ const INVOKE_TIMEOUT_MS = 10 * 60 * 1000;
27
+ /** Larger diffs are truncated; the reviewer is told so. */
28
+ const MAX_DIFF_CHARS = 200_000;
29
+
30
+ const PROMPT_HEAD =
31
+ 'You are reviewing the diff `{baseRef}...{headRef}` for Story #{ticketId} ' +
32
+ 'before it merges. {depthDirective}\n\n' +
33
+ 'Report ONLY problems you would block this merge for: a bug that makes ' +
34
+ 'the change behave incorrectly, crash, lose data, or break an existing ' +
35
+ 'caller. Do not report style, naming, refactoring ideas, missing tests or ' +
36
+ 'anything you would merge anyway. For each problem give the file, the ' +
37
+ 'line in the new version, why it is wrong, and how to show it fails (an ' +
38
+ 'input, command or test that exposes it).\n\n' +
39
+ 'The diff below is authoritative — files on disk may not reflect it. ' +
40
+ 'Emit ONLY a JSON array on stdout with this exact shape, no prose around ' +
41
+ 'it:\n\n' +
42
+ '[{"title":"...","body":"Why it is wrong: ... How to show it fails: ...",' +
43
+ '"file":"...","line":1,"category":"bug"}]\n\n' +
44
+ 'Emit [] if there is nothing you would block the merge for.\n\n';
45
+
46
+ /**
47
+ * @param {ReviewInput} input
48
+ * @param {string} diff
49
+ * @returns {string}
50
+ */
51
+ function buildPrompt(input, diff) {
52
+ const truncated = diff.length > MAX_DIFF_CHARS;
53
+ const body = truncated ? diff.slice(0, MAX_DIFF_CHARS) : diff;
54
+ const note = truncated
55
+ ? `\n[diff truncated at ${MAX_DIFF_CHARS} characters]\n`
56
+ : '';
57
+ const head = PROMPT_HEAD.replace('{baseRef}', input.baseRef)
58
+ .replace('{headRef}', input.headRef)
59
+ .replace('{ticketId}', String(input.ticketId))
60
+ .replace('{depthDirective}', renderDepthDirective(input.depth));
61
+ return `${head}<diff>\n${body}${note}\n</diff>\n`;
62
+ }
63
+
64
+ /**
65
+ * The prompt rides stdin so no shell ever quotes it.
66
+ *
67
+ * @param {string} prompt
68
+ * @param {Function} [run] - `spawnSync`-shaped seam for tests.
69
+ * @returns {{ status: number, stdout: string, stderr: string }}
70
+ */
71
+ function invokeClaude(prompt, run) {
72
+ return spawnCapture('claude', [...CLAUDE_ARGS], {
73
+ cwd: PROJECT_ROOT,
74
+ input: prompt,
75
+ shell: process.platform === 'win32',
76
+ timeout: INVOKE_TIMEOUT_MS,
77
+ ...(run ? { run } : {}),
78
+ });
79
+ }
80
+
81
+ /** A model often fences its JSON; strip one surrounding fence. */
82
+ function stripFence(text) {
83
+ const match = /^\s*```[a-z]*\s*\n([\s\S]*?)\n\s*```\s*$/i.exec(text ?? '');
84
+ return match ? match[1] : (text ?? '');
85
+ }
86
+
87
+ /**
88
+ * @param {string} title
89
+ * @param {string} detail
90
+ * @returns {Finding}
91
+ */
92
+ function advisory(title, detail) {
93
+ return {
94
+ severity: 'suggestion',
95
+ title,
96
+ body:
97
+ `${detail} The model bug review did not produce a verdict — inspect ` +
98
+ 'the diff manually before merging. Advisory only; the chain did not halt.',
99
+ category: 'bug',
100
+ };
101
+ }
102
+
103
+ /**
104
+ * @param {string} stdout
105
+ * @returns {Finding[]}
106
+ */
107
+ function parseFindings(stdout) {
108
+ try {
109
+ return parseProviderFindings(stripFence(stdout), {
110
+ errorPrefix: '[code-review] Failed to parse reviewer stdout as JSON',
111
+ mapSeverity: () => 'critical',
112
+ defaultCategory: 'bug',
113
+ });
114
+ } catch (err) {
115
+ return [
116
+ advisory(
117
+ 'Code review output not parseable as JSON',
118
+ `The reviewer returned text that did not parse as a JSON findings array (${err.message}).`,
119
+ ),
120
+ ];
121
+ }
122
+ }
123
+
124
+ /**
125
+ * @param {ReviewInput} input
126
+ */
127
+ function assertInput(input) {
128
+ const { baseRef, headRef, ticketId } = input ?? {};
129
+ if (!baseRef || !headRef) {
130
+ throw new TypeError(
131
+ '[code-review] runReview requires baseRef and headRef.',
132
+ );
133
+ }
134
+ if (!Number.isInteger(ticketId) || ticketId <= 0) {
135
+ throw new TypeError(
136
+ '[code-review] runReview requires a positive integer ticketId.',
137
+ );
138
+ }
139
+ }
140
+
141
+ /**
142
+ * @param {{
143
+ * probeFn?: () => boolean,
144
+ * gitSpawnFn?: typeof gitSpawn,
145
+ * spawnFn?: Function,
146
+ * logger?: { info?: Function, warn?: Function },
147
+ * }} [deps] - `spawnFn` replaces `spawnSync` for the `claude` call.
148
+ * @returns {ReviewProvider}
149
+ */
150
+ export function createCodeReviewProviderForRegistry(deps = {}) {
151
+ const probeFn = deps.probeFn ?? probeClaudeCli;
152
+ if (!probeFn()) {
153
+ throw new Error(
154
+ '[ReviewProviderFactory] codeReview provider "code-review" requires ' +
155
+ 'the `claude` CLI on PATH but it was not detected. Install the ' +
156
+ 'Claude Code CLI, or keep the entry `optional: true` so hosts ' +
157
+ 'without it skip the model bug review.',
158
+ );
159
+ }
160
+ const gitSpawnFn = deps.gitSpawnFn ?? gitSpawn;
161
+ const { spawnFn, logger } = deps;
162
+
163
+ return {
164
+ async runReview(input) {
165
+ assertInput(input);
166
+ const { baseRef, headRef, ticketId } = input;
167
+ const diff = gitSpawnFn(
168
+ PROJECT_ROOT,
169
+ 'diff',
170
+ '--no-color',
171
+ `${baseRef}...${headRef}`,
172
+ );
173
+ if (diff.status !== 0) {
174
+ return [
175
+ advisory(
176
+ 'Code review could not read the diff',
177
+ `\`git diff ${baseRef}...${headRef}\` failed: ${diff.stderr || '<no output>'}.`,
178
+ ),
179
+ ];
180
+ }
181
+ if (diff.stdout.trim().length === 0) return [];
182
+
183
+ logger?.info?.(
184
+ `[code-review] Invoking claude --print --effort low for Story #${ticketId} (${baseRef}...${headRef})...`,
185
+ );
186
+ const result = invokeClaude(buildPrompt(input, diff.stdout), spawnFn);
187
+ if (result.status !== 0) {
188
+ logger?.warn?.(
189
+ `[code-review] claude exited ${result.status}; emitting advisory.`,
190
+ );
191
+ return [
192
+ advisory(
193
+ 'Code review did not complete',
194
+ `\`claude --print\` exited with status ${result.status}: ${
195
+ result.stderr || result.stdout || '<no output>'
196
+ }.`,
197
+ ),
198
+ ];
199
+ }
200
+ return parseFindings(result.stdout);
201
+ },
202
+ };
203
+ }