mandrel 2.64.0 → 2.66.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/acceptance-critic.md +8 -7
- package/.agents/agents/auditor.md +20 -20
- package/.agents/agents/plan-critic.md +8 -7
- package/.agents/agents/story-worker.md +7 -7
- package/.agents/audit-checklists/quality.md +3 -0
- package/.agents/docs/agentrc-reference.json +1 -9
- package/.agents/docs/configuration.md +8 -7
- package/.agents/docs/execution-reference.md +27 -5
- package/.agents/instructions.md +10 -12
- package/.agents/rules/ci-remediation.md +3 -3
- package/.agents/rules/gherkin-standards.md +3 -2
- package/.agents/rules/git-conventions-reference.md +12 -3
- package/.agents/rules/git-conventions.md +9 -7
- package/.agents/rules/testing-standards.md +8 -7
- package/.agents/runtime-deps.json +1 -1
- package/.agents/schemas/agentrc.schema.json +6 -13
- package/.agents/schemas/audit-rules.schema.json +1 -1
- package/.agents/schemas/story-deliver-terminal.schema.json +5 -0
- package/.agents/scripts/bootstrap.js +102 -91
- package/.agents/scripts/check-context-budget.js +1 -1
- package/.agents/scripts/lib/ITicketingProvider.js +1 -3
- package/.agents/scripts/lib/audit-suite/findings.js +1 -17
- package/.agents/scripts/lib/audit-suite/frontmatter.js +0 -28
- package/.agents/scripts/lib/audit-suite/index.js +0 -6
- package/.agents/scripts/lib/audit-suite/selector.js +0 -31
- package/.agents/scripts/lib/baselines/duplication-scanner.js +17 -7
- package/.agents/scripts/lib/bootstrap/agents-md-fold.js +156 -0
- package/.agents/scripts/lib/bootstrap/commit-push.js +1 -1
- package/.agents/scripts/lib/bootstrap/manifest.js +2 -2
- package/.agents/scripts/lib/bootstrap/project-bootstrap.js +91 -107
- package/.agents/scripts/lib/cli/standard-args.js +60 -76
- package/.agents/scripts/lib/cli-args.js +26 -0
- package/.agents/scripts/lib/config/gates/shared.js +3 -3
- package/.agents/scripts/lib/config/review-chain-default.js +13 -0
- package/.agents/scripts/lib/config-settings-schema-delivery.js +2 -2
- package/.agents/scripts/lib/config-settings-schema-quality.js +11 -13
- package/.agents/scripts/lib/doc-tiers.js +25 -6
- package/.agents/scripts/lib/feedback-loop/graduate-steps.js +205 -0
- package/.agents/scripts/lib/feedback-loop/graduator-core.js +47 -782
- package/.agents/scripts/lib/feedback-loop/graduator-gh.js +449 -0
- package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
- package/.agents/scripts/lib/observability/close-telemetry.js +330 -0
- package/.agents/scripts/lib/observability/metrics-ledger.js +0 -72
- package/.agents/scripts/lib/observability/runtime-friction.js +2 -0
- package/.agents/scripts/lib/observability/signal-validator.js +17 -5
- package/.agents/scripts/lib/orchestration/code-review.js +33 -6
- package/.agents/scripts/lib/orchestration/epic-rollup.js +29 -12
- package/.agents/scripts/lib/orchestration/merge-block-class.js +20 -4
- package/.agents/scripts/lib/orchestration/merge-poll.js +41 -22
- package/.agents/scripts/lib/orchestration/plan-metrics.js +76 -63
- package/.agents/scripts/lib/orchestration/required-checks.js +147 -0
- package/.agents/scripts/lib/orchestration/review-providers/code-review.js +203 -0
- package/.agents/scripts/lib/orchestration/review-providers/review-provider-factory.js +29 -4
- package/.agents/scripts/lib/orchestration/review-providers/security-review.js +3 -2
- package/.agents/scripts/lib/orchestration/run-epilogue.js +6 -0
- package/.agents/scripts/lib/orchestration/single-story-close/failed-terminal.js +1 -0
- package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +2 -12
- package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +370 -268
- package/.agents/scripts/lib/orchestration/single-story-close/phases/options.js +21 -7
- package/.agents/scripts/lib/orchestration/single-story-close/phases/post-land.js +112 -82
- package/.agents/scripts/lib/orchestration/single-story-close/phases/review-override.js +4 -0
- package/.agents/scripts/lib/orchestration/single-story-close/runner.js +393 -313
- package/.agents/scripts/lib/orchestration/story-close/phases/review-core.js +12 -87
- package/.agents/scripts/lib/orchestration/story-deliver-terminal.js +3 -0
- package/.agents/scripts/lib/orchestration/ticket-validator.js +19 -36
- package/.agents/scripts/lib/signals/detectors/common.js +63 -51
- package/.agents/scripts/lib/templates/decomposer-prompts.js +5 -24
- package/.agents/scripts/lib/transpile.js +28 -3
- package/.agents/scripts/providers/github/issues.js +14 -23
- package/.agents/scripts/single-story-close.js +10 -2
- package/.agents/scripts/single-story-confirm-merge.js +267 -238
- package/.agents/scripts/sync-claude-agents.js +1 -1
- package/.agents/skills/core/idea-refinement/SKILL.md +6 -6
- package/.agents/skills/stack/qa/qa-harness/SKILL.md +1 -2
- package/.agents/workflows/audit-architecture.md +5 -4
- package/.agents/workflows/audit-documentation.md +5 -5
- package/.agents/workflows/audit-performance.md +10 -10
- package/.agents/workflows/audit-quality.md +42 -7
- package/.agents/workflows/helpers/acceptance-self-eval.md +9 -9
- package/.agents/workflows/helpers/audit-lens-core.md +30 -57
- package/.agents/workflows/helpers/code-review.md +15 -38
- package/.agents/workflows/helpers/deliver-digest.md +2 -2
- package/.agents/workflows/helpers/deliver-reference.md +7 -3
- package/.agents/workflows/helpers/deliver-story.md +9 -1
- package/.agents/workflows/helpers/parallel-tooling.md +16 -18
- package/.agents/workflows/helpers/plan-reference.md +9 -8
- package/.agents/workflows/mandrel-deliver.md +3 -2
- package/.agents/workflows/mandrel-plan.md +11 -7
- package/.agents/workflows/mandrel-update.md +5 -3
- package/docs/CHANGELOG.md +57 -0
- package/lib/cli/claude-code-version.js +73 -0
- package/lib/cli/doctor.js +2 -2
- package/lib/cli/guarded-sync.js +87 -0
- package/lib/cli/registry.js +9 -0
- package/lib/cli/sync-agents.js +9 -92
- package/lib/cli/sync-commands.js +9 -101
- package/lib/cli/uninstall.js +37 -9
- package/lib/migrations/index.js +2 -0
- package/lib/migrations/steps/2.65.0-fold-claude-md-into-agents-md.js +38 -0
- package/package.json +3 -2
- package/.agents/scripts/lib/audit-suite/lens-diff-floor.js +0 -99
- package/.agents/scripts/lib/audit-suite/runner.js +0 -205
- package/.agents/scripts/lib/audit-suite/substitutions.js +0 -96
- package/.agents/scripts/lib/audit-suite/workflow-loader.js +0 -37
- package/.agents/scripts/lib/orchestration/story-close/phases/local-lens-review.js +0 -234
|
@@ -8,13 +8,29 @@ import { checkVerdict, classifyRollupEntry } from './check-state.js';
|
|
|
8
8
|
|
|
9
9
|
/** Fixed poll interval; default for `delivery.mergeWatch.maxBudgetSeconds`. */
|
|
10
10
|
export const DEFAULT_INTERVAL_SECONDS = 30;
|
|
11
|
+
|
|
12
|
+
/** Checks green, PR unmerged: the merge is imminent, so observe it sooner. */
|
|
13
|
+
const GREEN_INTERVAL_SECONDS = 10;
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {string|undefined} checksStatus
|
|
17
|
+
* @param {number} intervalSeconds The in-flight cadence.
|
|
18
|
+
* @returns {number} Milliseconds until the next poll.
|
|
19
|
+
*/
|
|
20
|
+
export function pollIntervalMs(checksStatus, intervalSeconds) {
|
|
21
|
+
const seconds =
|
|
22
|
+
checksStatus === 'success'
|
|
23
|
+
? Math.min(GREEN_INTERVAL_SECONDS, intervalSeconds)
|
|
24
|
+
: intervalSeconds;
|
|
25
|
+
return seconds * 1000;
|
|
26
|
+
}
|
|
11
27
|
export const DEFAULT_MAX_BUDGET_SECONDS = 3600;
|
|
12
28
|
|
|
13
29
|
/** Bounds every `gh` spawn so a hang degrades to the probe-error path. */
|
|
14
30
|
export const MERGE_WAIT_GH_TIMEOUT_MS = 60_000;
|
|
15
31
|
|
|
16
32
|
/**
|
|
17
|
-
* Aggregate over EVERY check (the rollup has no `isRequired`): `failure`
|
|
33
|
+
* Aggregate over EVERY check (the view rollup has no `isRequired`): `failure`
|
|
18
34
|
* means "something is red", not "blocked" — see {@link failingChecksBlockMerge}.
|
|
19
35
|
*/
|
|
20
36
|
export function deriveChecksStatus(statusCheckRollup) {
|
|
@@ -47,7 +63,7 @@ export function isPrMerged(pr) {
|
|
|
47
63
|
* @param {{ conclusion?: string, state?: string }} [check]
|
|
48
64
|
* @returns {string|null}
|
|
49
65
|
*/
|
|
50
|
-
function redConclusionOf(check) {
|
|
66
|
+
export function redConclusionOf(check) {
|
|
51
67
|
const conclusion = String(check?.conclusion ?? '').toUpperCase();
|
|
52
68
|
if (conclusion === 'FAILURE' || conclusion === 'ERROR') return conclusion;
|
|
53
69
|
const state = String(check?.state ?? '').toUpperCase();
|
|
@@ -61,17 +77,29 @@ function redConclusionOf(check) {
|
|
|
61
77
|
* @param {{ name?: string, context?: string }} [check]
|
|
62
78
|
* @returns {string|null}
|
|
63
79
|
*/
|
|
64
|
-
function readRunName(check) {
|
|
80
|
+
export function readRunName(check) {
|
|
65
81
|
for (const value of [check?.name, check?.context]) {
|
|
66
82
|
if (typeof value === 'string' && value) return value;
|
|
67
83
|
}
|
|
68
84
|
return null;
|
|
69
85
|
}
|
|
70
86
|
|
|
87
|
+
/**
|
|
88
|
+
* @param {{ status?: string, state?: string }} [check]
|
|
89
|
+
* @returns {boolean}
|
|
90
|
+
*/
|
|
91
|
+
export function isRunInFlight(check) {
|
|
92
|
+
const status = String(check?.status ?? '').toUpperCase();
|
|
93
|
+
// `status` is empty on a StatusContext, so it falls to the `state` branch.
|
|
94
|
+
if (status) return status !== 'COMPLETED';
|
|
95
|
+
const state = String(check?.state ?? '').toUpperCase();
|
|
96
|
+
return state === 'PENDING' || state === 'EXPECTED';
|
|
97
|
+
}
|
|
98
|
+
|
|
71
99
|
/**
|
|
72
100
|
* Head-anchored evidence; `null` on an empty rollup (the caller falls back
|
|
73
|
-
* to consecutive probes). Reads EVERY run
|
|
74
|
-
*
|
|
101
|
+
* to consecutive probes). Reads EVERY run — the unscoped rule used when
|
|
102
|
+
* GitHub's required attribution is unavailable (see `required-checks.js`).
|
|
75
103
|
*
|
|
76
104
|
* @param {Array<{status?: string, conclusion?: string, state?: string}>} statusCheckRollup
|
|
77
105
|
* @returns {{ requiredRunFailed: boolean, requiredRunInFlight: boolean } | null}
|
|
@@ -80,22 +108,12 @@ export function deriveRequiredRunEvidence(statusCheckRollup) {
|
|
|
80
108
|
if (!Array.isArray(statusCheckRollup) || statusCheckRollup.length === 0) {
|
|
81
109
|
return null;
|
|
82
110
|
}
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
if (status && status !== 'COMPLETED') {
|
|
90
|
-
requiredRunInFlight = true;
|
|
91
|
-
} else if (state === 'PENDING' || state === 'EXPECTED') {
|
|
92
|
-
requiredRunInFlight = true;
|
|
93
|
-
}
|
|
94
|
-
if (redConclusionOf(check)) {
|
|
95
|
-
requiredRunFailed = true;
|
|
96
|
-
}
|
|
97
|
-
}
|
|
98
|
-
return { requiredRunFailed, requiredRunInFlight };
|
|
111
|
+
return {
|
|
112
|
+
requiredRunFailed: statusCheckRollup.some(
|
|
113
|
+
(c) => redConclusionOf(c) !== null,
|
|
114
|
+
),
|
|
115
|
+
requiredRunInFlight: statusCheckRollup.some(isRunInFlight),
|
|
116
|
+
};
|
|
99
117
|
}
|
|
100
118
|
|
|
101
119
|
/** The `mergeStateStatus` meaning GitHub itself gates the merge. */
|
|
@@ -139,7 +157,8 @@ export function formatChecksFailedReason(prProbe, evidencePath) {
|
|
|
139
157
|
|
|
140
158
|
/**
|
|
141
159
|
* A genuinely red REQUIRED check: gated, no review owns `BLOCKED`, a run is
|
|
142
|
-
* red and none in flight
|
|
160
|
+
* red and none in flight — with GitHub attribution, a required run is red and
|
|
161
|
+
* no re-run of it is in flight. No evidence → false (consecutive-probe path).
|
|
143
162
|
*
|
|
144
163
|
* @param {{ checksStatus?: string, mergeStateStatus?: string,
|
|
145
164
|
* reviewDecision?: string,
|
|
@@ -231,75 +231,88 @@ function recordTimestamp(entry) {
|
|
|
231
231
|
* }|null}
|
|
232
232
|
*/
|
|
233
233
|
export function summarizePlanMetrics(ledger, opts = {}) {
|
|
234
|
-
const
|
|
235
|
-
const since = typeof opts.since === 'string' ? opts.since : null;
|
|
236
|
-
// ISO-8601 UTC strings compare correctly as strings.
|
|
237
|
-
const entries =
|
|
238
|
-
since === null
|
|
239
|
-
? all
|
|
240
|
-
: all.filter((e) => {
|
|
241
|
-
const stamp = recordTimestamp(e);
|
|
242
|
-
return stamp !== null && stamp >= since;
|
|
243
|
-
});
|
|
234
|
+
const entries = scopeEntries(ledger?.entries ?? [], opts.since);
|
|
244
235
|
if (entries.length === 0) return null;
|
|
245
|
-
const
|
|
246
|
-
const
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
let failures = 0;
|
|
250
|
-
let totalDurationMs = 0;
|
|
251
|
-
let firstStartedAt = null;
|
|
252
|
-
let lastEndedAt = null;
|
|
253
|
-
const invocationEntries = [];
|
|
254
|
-
for (const e of entries) {
|
|
255
|
-
if (e.kind === PLAN_METRICS_KIND_CRITIC_SKIP) {
|
|
256
|
-
criticSkips += 1;
|
|
257
|
-
if (typeof e.critic === 'string') {
|
|
258
|
-
criticSkipsByCritic[e.critic] =
|
|
259
|
-
(criticSkipsByCritic[e.critic] ?? 0) + 1;
|
|
260
|
-
}
|
|
261
|
-
continue;
|
|
262
|
-
}
|
|
263
|
-
if (typeof e.kind === 'string') {
|
|
264
|
-
// Other kinded records (e.g. `findings-yield`) are not invocations.
|
|
265
|
-
continue;
|
|
266
|
-
}
|
|
267
|
-
invocationEntries.push(e);
|
|
268
|
-
byCli[e.cli] = (byCli[e.cli] ?? 0) + 1;
|
|
269
|
-
if (typeof e.mode === 'string') byMode[e.mode] = (byMode[e.mode] ?? 0) + 1;
|
|
270
|
-
if (e.ok !== true) failures += 1;
|
|
271
|
-
if (typeof e.durationMs === 'number') totalDurationMs += e.durationMs;
|
|
272
|
-
if (typeof e.startedAt === 'string') {
|
|
273
|
-
if (firstStartedAt === null || e.startedAt < firstStartedAt) {
|
|
274
|
-
firstStartedAt = e.startedAt;
|
|
275
|
-
}
|
|
276
|
-
}
|
|
277
|
-
if (typeof e.endedAt === 'string') {
|
|
278
|
-
if (lastEndedAt === null || e.endedAt > lastEndedAt) {
|
|
279
|
-
lastEndedAt = e.endedAt;
|
|
280
|
-
}
|
|
281
|
-
}
|
|
282
|
-
}
|
|
283
|
-
let spanMs = null;
|
|
284
|
-
if (firstStartedAt !== null && lastEndedAt !== null) {
|
|
285
|
-
const span = Date.parse(lastEndedAt) - Date.parse(firstStartedAt);
|
|
286
|
-
if (Number.isFinite(span)) spanMs = Math.max(0, span);
|
|
287
|
-
}
|
|
236
|
+
const skips = tallyCriticSkips(entries);
|
|
237
|
+
const invocations = tallyInvocations(
|
|
238
|
+
entries.filter((e) => typeof e.kind !== 'string'),
|
|
239
|
+
);
|
|
288
240
|
return {
|
|
289
|
-
invocations:
|
|
290
|
-
failures,
|
|
291
|
-
byCli,
|
|
292
|
-
byMode,
|
|
293
|
-
criticSkips,
|
|
294
|
-
criticSkipsByCritic,
|
|
295
|
-
firstStartedAt,
|
|
296
|
-
lastEndedAt,
|
|
297
|
-
spanMs,
|
|
298
|
-
totalDurationMs,
|
|
241
|
+
invocations: invocations.count,
|
|
242
|
+
failures: invocations.failures,
|
|
243
|
+
byCli: invocations.byCli,
|
|
244
|
+
byMode: invocations.byMode,
|
|
245
|
+
criticSkips: skips.count,
|
|
246
|
+
criticSkipsByCritic: skips.byCritic,
|
|
247
|
+
firstStartedAt: invocations.firstStartedAt,
|
|
248
|
+
lastEndedAt: invocations.lastEndedAt,
|
|
249
|
+
spanMs: spanBetween(invocations.firstStartedAt, invocations.lastEndedAt),
|
|
250
|
+
totalDurationMs: invocations.totalDurationMs,
|
|
299
251
|
malformedLines: ledger?.malformedLines ?? 0,
|
|
300
252
|
};
|
|
301
253
|
}
|
|
302
254
|
|
|
255
|
+
function scopeEntries(all, since) {
|
|
256
|
+
if (typeof since !== 'string') return all;
|
|
257
|
+
// ISO-8601 UTC strings compare correctly as strings.
|
|
258
|
+
return all.filter((e) => {
|
|
259
|
+
const stamp = recordTimestamp(e);
|
|
260
|
+
return stamp !== null && stamp >= since;
|
|
261
|
+
});
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
function increment(counts, key) {
|
|
265
|
+
counts[key] = (counts[key] ?? 0) + 1;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
function tallyCriticSkips(entries) {
|
|
269
|
+
const byCritic = {};
|
|
270
|
+
let count = 0;
|
|
271
|
+
for (const e of entries) {
|
|
272
|
+
if (e.kind !== PLAN_METRICS_KIND_CRITIC_SKIP) continue;
|
|
273
|
+
count += 1;
|
|
274
|
+
if (typeof e.critic === 'string') increment(byCritic, e.critic);
|
|
275
|
+
}
|
|
276
|
+
return { count, byCritic };
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
function earlierStamp(current, candidate) {
|
|
280
|
+
if (typeof candidate !== 'string') return current;
|
|
281
|
+
return current === null || candidate < current ? candidate : current;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
function laterStamp(current, candidate) {
|
|
285
|
+
if (typeof candidate !== 'string') return current;
|
|
286
|
+
return current === null || candidate > current ? candidate : current;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
function tallyInvocations(entries) {
|
|
290
|
+
const tally = {
|
|
291
|
+
count: entries.length,
|
|
292
|
+
failures: 0,
|
|
293
|
+
byCli: {},
|
|
294
|
+
byMode: {},
|
|
295
|
+
totalDurationMs: 0,
|
|
296
|
+
firstStartedAt: null,
|
|
297
|
+
lastEndedAt: null,
|
|
298
|
+
};
|
|
299
|
+
for (const e of entries) {
|
|
300
|
+
increment(tally.byCli, e.cli);
|
|
301
|
+
if (typeof e.mode === 'string') increment(tally.byMode, e.mode);
|
|
302
|
+
if (e.ok !== true) tally.failures += 1;
|
|
303
|
+
if (typeof e.durationMs === 'number') tally.totalDurationMs += e.durationMs;
|
|
304
|
+
tally.firstStartedAt = earlierStamp(tally.firstStartedAt, e.startedAt);
|
|
305
|
+
tally.lastEndedAt = laterStamp(tally.lastEndedAt, e.endedAt);
|
|
306
|
+
}
|
|
307
|
+
return tally;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
function spanBetween(first, last) {
|
|
311
|
+
if (first === null || last === null) return null;
|
|
312
|
+
const span = Date.parse(last) - Date.parse(first);
|
|
313
|
+
return Number.isFinite(span) ? Math.max(0, span) : null;
|
|
314
|
+
}
|
|
315
|
+
|
|
303
316
|
/**
|
|
304
317
|
* @param {ReturnType<typeof summarizePlanMetrics>} summary
|
|
305
318
|
* @returns {string}
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* GitHub's per-PR required-check attribution (GraphQL `isRequired`), not
|
|
3
|
+
* `.agentrc` requiredChecks, which are local command names.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import {
|
|
7
|
+
deriveRequiredRunEvidence,
|
|
8
|
+
failingChecksBlockMerge,
|
|
9
|
+
isRunInFlight,
|
|
10
|
+
MERGE_WAIT_GH_TIMEOUT_MS,
|
|
11
|
+
readRunName,
|
|
12
|
+
redConclusionOf,
|
|
13
|
+
} from './merge-poll.js';
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* A required run is red and no re-run of that same check is in flight.
|
|
17
|
+
*
|
|
18
|
+
* @param {Array<object>} statusCheckRollup non-empty
|
|
19
|
+
* @param {Set<string>} requiredNames
|
|
20
|
+
*/
|
|
21
|
+
function deriveAttributedEvidence(statusCheckRollup, requiredNames) {
|
|
22
|
+
const failedRequired = new Set();
|
|
23
|
+
for (const check of statusCheckRollup) {
|
|
24
|
+
const name = readRunName(check);
|
|
25
|
+
if (name && requiredNames.has(name) && redConclusionOf(check)) {
|
|
26
|
+
failedRequired.add(name);
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
let requiredRunInFlight = false;
|
|
30
|
+
let runInFlight = false;
|
|
31
|
+
for (const check of statusCheckRollup) {
|
|
32
|
+
if (!isRunInFlight(check)) continue;
|
|
33
|
+
runInFlight = true;
|
|
34
|
+
if (failedRequired.has(readRunName(check))) requiredRunInFlight = true;
|
|
35
|
+
}
|
|
36
|
+
return {
|
|
37
|
+
requiredRunFailed: failedRequired.size > 0,
|
|
38
|
+
requiredRunInFlight,
|
|
39
|
+
runInFlight,
|
|
40
|
+
attribution: 'github',
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const REQUIRED_CHECKS_QUERY =
|
|
45
|
+
'query($id: ID!, $n: Int!) { node(id: $id) { ... on PullRequest { ' +
|
|
46
|
+
'commits(last: 1) { nodes { commit { statusCheckRollup { ' +
|
|
47
|
+
'contexts(first: 100) { nodes { __typename ' +
|
|
48
|
+
'... on CheckRun { name isRequired(pullRequestNumber: $n) } ' +
|
|
49
|
+
'... on StatusContext { context isRequired(pullRequestNumber: $n) } ' +
|
|
50
|
+
'} } } } } } } } }';
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* @param {object|string} result `gh api graphql` output
|
|
54
|
+
* @returns {Set<string>}
|
|
55
|
+
*/
|
|
56
|
+
function parseRequiredNames(result) {
|
|
57
|
+
const text = typeof result === 'string' ? result : result?.stdout;
|
|
58
|
+
const parsed = JSON.parse(String(text ?? ''));
|
|
59
|
+
if (Array.isArray(parsed?.errors) && parsed.errors.length > 0) {
|
|
60
|
+
throw new Error('graphql errors reading required checks');
|
|
61
|
+
}
|
|
62
|
+
const nodes =
|
|
63
|
+
parsed?.data?.node?.commits?.nodes?.[0]?.commit?.statusCheckRollup?.contexts
|
|
64
|
+
?.nodes;
|
|
65
|
+
if (!Array.isArray(nodes)) throw new Error('required-check contexts absent');
|
|
66
|
+
const names = new Set();
|
|
67
|
+
for (const node of nodes) {
|
|
68
|
+
const name = readRunName(node);
|
|
69
|
+
if (node?.isRequired === true && name) names.add(name);
|
|
70
|
+
}
|
|
71
|
+
return names;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Per-head cache: required attribution is read at most once per PR head. */
|
|
75
|
+
const requiredNamesCache = new Map();
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Required check names, cached per PR head. `null` on any failure.
|
|
79
|
+
*
|
|
80
|
+
* @param {{ prNodeId?: string, prNumber: number|string, headSha?: string,
|
|
81
|
+
* gh: { api: Function }, timeoutMs?: number }} args
|
|
82
|
+
* @returns {Promise<Set<string>|null>}
|
|
83
|
+
*/
|
|
84
|
+
async function readRequiredCheckNames({
|
|
85
|
+
prNodeId,
|
|
86
|
+
prNumber,
|
|
87
|
+
headSha,
|
|
88
|
+
gh,
|
|
89
|
+
timeoutMs = MERGE_WAIT_GH_TIMEOUT_MS,
|
|
90
|
+
}) {
|
|
91
|
+
if (!prNodeId || !headSha) return null;
|
|
92
|
+
const key = `${prNodeId}@${headSha}`;
|
|
93
|
+
if (requiredNamesCache.has(key)) return requiredNamesCache.get(key);
|
|
94
|
+
try {
|
|
95
|
+
const names = parseRequiredNames(
|
|
96
|
+
await gh.api({
|
|
97
|
+
method: 'POST',
|
|
98
|
+
endpoint: 'graphql',
|
|
99
|
+
body: {
|
|
100
|
+
query: REQUIRED_CHECKS_QUERY,
|
|
101
|
+
variables: { id: prNodeId, n: Number(prNumber) },
|
|
102
|
+
},
|
|
103
|
+
execOpts: { timeoutMs },
|
|
104
|
+
}),
|
|
105
|
+
);
|
|
106
|
+
requiredNamesCache.set(key, names);
|
|
107
|
+
return names;
|
|
108
|
+
} catch {
|
|
109
|
+
return null;
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Scoped evidence when a red gates the merge and attribution reads; else
|
|
115
|
+
* the unscoped rule.
|
|
116
|
+
*
|
|
117
|
+
* @param {{ view?: object, checksStatus?: string, prNumber: number|string,
|
|
118
|
+
* gh: object, ghTimeoutMs?: number, readFn?: Function }} args
|
|
119
|
+
* @returns {Promise<object|null>}
|
|
120
|
+
*/
|
|
121
|
+
export async function readProbeRunEvidence({
|
|
122
|
+
view,
|
|
123
|
+
checksStatus,
|
|
124
|
+
prNumber,
|
|
125
|
+
gh,
|
|
126
|
+
ghTimeoutMs,
|
|
127
|
+
readFn = readRequiredCheckNames,
|
|
128
|
+
}) {
|
|
129
|
+
const rollup = view?.statusCheckRollup;
|
|
130
|
+
const gated = failingChecksBlockMerge({
|
|
131
|
+
checksStatus,
|
|
132
|
+
mergeStateStatus: view?.mergeStateStatus,
|
|
133
|
+
});
|
|
134
|
+
const names = gated
|
|
135
|
+
? await readFn({
|
|
136
|
+
prNodeId: view?.id,
|
|
137
|
+
prNumber,
|
|
138
|
+
headSha: view?.headRefOid,
|
|
139
|
+
gh,
|
|
140
|
+
timeoutMs: ghTimeoutMs,
|
|
141
|
+
})
|
|
142
|
+
: null;
|
|
143
|
+
if (names instanceof Set && Array.isArray(rollup) && rollup.length > 0) {
|
|
144
|
+
return deriveAttributedEvidence(rollup, names);
|
|
145
|
+
}
|
|
146
|
+
return deriveRequiredRunEvidence(rollup);
|
|
147
|
+
}
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* review-providers/code-review.js — a low-effort model bug review of the
|
|
3
|
+
* Story diff through `claude --print --effort low`. It asks only for
|
|
4
|
+
* merge-blocking problems, so every parsed finding is `critical` and halts
|
|
5
|
+
* close before auto-merge. The reviewer is handed the diff range, the Story
|
|
6
|
+
* id and the diff text — never `acceptance[]` or the self-eval verdict: this
|
|
7
|
+
* is a bug review, not a second acceptance scoring.
|
|
8
|
+
*
|
|
9
|
+
* A missing CLI throws at construction (the default chain entry is
|
|
10
|
+
* `optional: true`, so hosts without it skip). A review that cannot run or
|
|
11
|
+
* whose output does not parse degrades to one non-halting `suggestion`.
|
|
12
|
+
*
|
|
13
|
+
* @typedef {import('./types.js').Finding} Finding
|
|
14
|
+
* @typedef {import('./types.js').ReviewInput} ReviewInput
|
|
15
|
+
* @typedef {import('./types.js').ReviewProvider} ReviewProvider
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { spawnCapture } from '../../child-exec.js';
|
|
19
|
+
import { gitSpawn } from '../../git-utils.js';
|
|
20
|
+
import { PROJECT_ROOT } from '../../project-root.js';
|
|
21
|
+
import { parseProviderFindings } from './parse-findings.js';
|
|
22
|
+
import { renderDepthDirective } from './review-depth.js';
|
|
23
|
+
import { probeClaudeCli } from './security-review.js';
|
|
24
|
+
|
|
25
|
+
const CLAUDE_ARGS = Object.freeze(['--print', '--effort', 'low']);
|
|
26
|
+
const INVOKE_TIMEOUT_MS = 10 * 60 * 1000;
|
|
27
|
+
/** Larger diffs are truncated; the reviewer is told so. */
|
|
28
|
+
const MAX_DIFF_CHARS = 200_000;
|
|
29
|
+
|
|
30
|
+
const PROMPT_HEAD =
|
|
31
|
+
'You are reviewing the diff `{baseRef}...{headRef}` for Story #{ticketId} ' +
|
|
32
|
+
'before it merges. {depthDirective}\n\n' +
|
|
33
|
+
'Report ONLY problems you would block this merge for: a bug that makes ' +
|
|
34
|
+
'the change behave incorrectly, crash, lose data, or break an existing ' +
|
|
35
|
+
'caller. Do not report style, naming, refactoring ideas, missing tests or ' +
|
|
36
|
+
'anything you would merge anyway. For each problem give the file, the ' +
|
|
37
|
+
'line in the new version, why it is wrong, and how to show it fails (an ' +
|
|
38
|
+
'input, command or test that exposes it).\n\n' +
|
|
39
|
+
'The diff below is authoritative — files on disk may not reflect it. ' +
|
|
40
|
+
'Emit ONLY a JSON array on stdout with this exact shape, no prose around ' +
|
|
41
|
+
'it:\n\n' +
|
|
42
|
+
'[{"title":"...","body":"Why it is wrong: ... How to show it fails: ...",' +
|
|
43
|
+
'"file":"...","line":1,"category":"bug"}]\n\n' +
|
|
44
|
+
'Emit [] if there is nothing you would block the merge for.\n\n';
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* @param {ReviewInput} input
|
|
48
|
+
* @param {string} diff
|
|
49
|
+
* @returns {string}
|
|
50
|
+
*/
|
|
51
|
+
function buildPrompt(input, diff) {
|
|
52
|
+
const truncated = diff.length > MAX_DIFF_CHARS;
|
|
53
|
+
const body = truncated ? diff.slice(0, MAX_DIFF_CHARS) : diff;
|
|
54
|
+
const note = truncated
|
|
55
|
+
? `\n[diff truncated at ${MAX_DIFF_CHARS} characters]\n`
|
|
56
|
+
: '';
|
|
57
|
+
const head = PROMPT_HEAD.replace('{baseRef}', input.baseRef)
|
|
58
|
+
.replace('{headRef}', input.headRef)
|
|
59
|
+
.replace('{ticketId}', String(input.ticketId))
|
|
60
|
+
.replace('{depthDirective}', renderDepthDirective(input.depth));
|
|
61
|
+
return `${head}<diff>\n${body}${note}\n</diff>\n`;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* The prompt rides stdin so no shell ever quotes it.
|
|
66
|
+
*
|
|
67
|
+
* @param {string} prompt
|
|
68
|
+
* @param {Function} [run] - `spawnSync`-shaped seam for tests.
|
|
69
|
+
* @returns {{ status: number, stdout: string, stderr: string }}
|
|
70
|
+
*/
|
|
71
|
+
function invokeClaude(prompt, run) {
|
|
72
|
+
return spawnCapture('claude', [...CLAUDE_ARGS], {
|
|
73
|
+
cwd: PROJECT_ROOT,
|
|
74
|
+
input: prompt,
|
|
75
|
+
shell: process.platform === 'win32',
|
|
76
|
+
timeout: INVOKE_TIMEOUT_MS,
|
|
77
|
+
...(run ? { run } : {}),
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** A model often fences its JSON; strip one surrounding fence. */
|
|
82
|
+
function stripFence(text) {
|
|
83
|
+
const match = /^\s*```[a-z]*\s*\n([\s\S]*?)\n\s*```\s*$/i.exec(text ?? '');
|
|
84
|
+
return match ? match[1] : (text ?? '');
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* @param {string} title
|
|
89
|
+
* @param {string} detail
|
|
90
|
+
* @returns {Finding}
|
|
91
|
+
*/
|
|
92
|
+
function advisory(title, detail) {
|
|
93
|
+
return {
|
|
94
|
+
severity: 'suggestion',
|
|
95
|
+
title,
|
|
96
|
+
body:
|
|
97
|
+
`${detail} The model bug review did not produce a verdict — inspect ` +
|
|
98
|
+
'the diff manually before merging. Advisory only; the chain did not halt.',
|
|
99
|
+
category: 'bug',
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* @param {string} stdout
|
|
105
|
+
* @returns {Finding[]}
|
|
106
|
+
*/
|
|
107
|
+
function parseFindings(stdout) {
|
|
108
|
+
try {
|
|
109
|
+
return parseProviderFindings(stripFence(stdout), {
|
|
110
|
+
errorPrefix: '[code-review] Failed to parse reviewer stdout as JSON',
|
|
111
|
+
mapSeverity: () => 'critical',
|
|
112
|
+
defaultCategory: 'bug',
|
|
113
|
+
});
|
|
114
|
+
} catch (err) {
|
|
115
|
+
return [
|
|
116
|
+
advisory(
|
|
117
|
+
'Code review output not parseable as JSON',
|
|
118
|
+
`The reviewer returned text that did not parse as a JSON findings array (${err.message}).`,
|
|
119
|
+
),
|
|
120
|
+
];
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* @param {ReviewInput} input
|
|
126
|
+
*/
|
|
127
|
+
function assertInput(input) {
|
|
128
|
+
const { baseRef, headRef, ticketId } = input ?? {};
|
|
129
|
+
if (!baseRef || !headRef) {
|
|
130
|
+
throw new TypeError(
|
|
131
|
+
'[code-review] runReview requires baseRef and headRef.',
|
|
132
|
+
);
|
|
133
|
+
}
|
|
134
|
+
if (!Number.isInteger(ticketId) || ticketId <= 0) {
|
|
135
|
+
throw new TypeError(
|
|
136
|
+
'[code-review] runReview requires a positive integer ticketId.',
|
|
137
|
+
);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* @param {{
|
|
143
|
+
* probeFn?: () => boolean,
|
|
144
|
+
* gitSpawnFn?: typeof gitSpawn,
|
|
145
|
+
* spawnFn?: Function,
|
|
146
|
+
* logger?: { info?: Function, warn?: Function },
|
|
147
|
+
* }} [deps] - `spawnFn` replaces `spawnSync` for the `claude` call.
|
|
148
|
+
* @returns {ReviewProvider}
|
|
149
|
+
*/
|
|
150
|
+
export function createCodeReviewProviderForRegistry(deps = {}) {
|
|
151
|
+
const probeFn = deps.probeFn ?? probeClaudeCli;
|
|
152
|
+
if (!probeFn()) {
|
|
153
|
+
throw new Error(
|
|
154
|
+
'[ReviewProviderFactory] codeReview provider "code-review" requires ' +
|
|
155
|
+
'the `claude` CLI on PATH but it was not detected. Install the ' +
|
|
156
|
+
'Claude Code CLI, or keep the entry `optional: true` so hosts ' +
|
|
157
|
+
'without it skip the model bug review.',
|
|
158
|
+
);
|
|
159
|
+
}
|
|
160
|
+
const gitSpawnFn = deps.gitSpawnFn ?? gitSpawn;
|
|
161
|
+
const { spawnFn, logger } = deps;
|
|
162
|
+
|
|
163
|
+
return {
|
|
164
|
+
async runReview(input) {
|
|
165
|
+
assertInput(input);
|
|
166
|
+
const { baseRef, headRef, ticketId } = input;
|
|
167
|
+
const diff = gitSpawnFn(
|
|
168
|
+
PROJECT_ROOT,
|
|
169
|
+
'diff',
|
|
170
|
+
'--no-color',
|
|
171
|
+
`${baseRef}...${headRef}`,
|
|
172
|
+
);
|
|
173
|
+
if (diff.status !== 0) {
|
|
174
|
+
return [
|
|
175
|
+
advisory(
|
|
176
|
+
'Code review could not read the diff',
|
|
177
|
+
`\`git diff ${baseRef}...${headRef}\` failed: ${diff.stderr || '<no output>'}.`,
|
|
178
|
+
),
|
|
179
|
+
];
|
|
180
|
+
}
|
|
181
|
+
if (diff.stdout.trim().length === 0) return [];
|
|
182
|
+
|
|
183
|
+
logger?.info?.(
|
|
184
|
+
`[code-review] Invoking claude --print --effort low for Story #${ticketId} (${baseRef}...${headRef})...`,
|
|
185
|
+
);
|
|
186
|
+
const result = invokeClaude(buildPrompt(input, diff.stdout), spawnFn);
|
|
187
|
+
if (result.status !== 0) {
|
|
188
|
+
logger?.warn?.(
|
|
189
|
+
`[code-review] claude exited ${result.status}; emitting advisory.`,
|
|
190
|
+
);
|
|
191
|
+
return [
|
|
192
|
+
advisory(
|
|
193
|
+
'Code review did not complete',
|
|
194
|
+
`\`claude --print\` exited with status ${result.status}: ${
|
|
195
|
+
result.stderr || result.stdout || '<no output>'
|
|
196
|
+
}.`,
|
|
197
|
+
),
|
|
198
|
+
];
|
|
199
|
+
}
|
|
200
|
+
return parseFindings(result.stdout);
|
|
201
|
+
},
|
|
202
|
+
};
|
|
203
|
+
}
|