argus-reviewer-e2e 0.3.1 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +84 -71
- package/action/action.yml +129 -11
- package/action/approval-review.mjs +13 -3
- package/action/bootstrap.mjs +2 -0
- package/action/emit-review.mjs +16 -0
- package/action/runtime.mjs +20 -0
- package/action/sticky-comment.cjs +1260 -479
- package/dist/cli.d.ts +103 -7
- package/dist/cli.js +1202 -186
- package/dist/config.d.ts +96 -11
- package/dist/config.js +102 -4
- package/dist/detect.d.ts +29 -2
- package/dist/detect.js +98 -7
- package/dist/driver/browser.d.ts +32 -0
- package/dist/driver/browser.js +56 -1
- package/dist/driver/target.d.ts +4 -1
- package/dist/driver/target.js +27 -6
- package/dist/engine/actions.d.ts +5 -0
- package/dist/engine/actions.js +8 -0
- package/dist/engine/explore.d.ts +78 -0
- package/dist/engine/explore.js +373 -0
- package/dist/engine/loop.d.ts +2 -2
- package/dist/engine/loop.js +8 -8
- package/dist/engine/prompts.d.ts +28 -1
- package/dist/engine/prompts.js +88 -0
- package/dist/evidence/ci.d.ts +13 -1
- package/dist/evidence/ci.js +38 -3
- package/dist/evidence/gate.d.ts +8 -0
- package/dist/evidence/gate.js +1 -1
- package/dist/evidence/link.js +1 -1
- package/dist/executor/a0.d.ts +114 -1
- package/dist/executor/a0.js +216 -4
- package/dist/fsutil.d.ts +3 -2
- package/dist/fsutil.js +7 -4
- package/dist/journal/schema.d.ts +1 -1
- package/dist/log.d.ts +2 -1
- package/dist/log.js +10 -2
- package/dist/mention.d.ts +45 -0
- package/dist/mention.js +107 -0
- package/dist/pipeline/app.d.ts +126 -0
- package/dist/pipeline/app.js +250 -0
- package/dist/pipeline/budget.d.ts +1 -0
- package/dist/pipeline/budget.js +1 -1
- package/dist/pipeline/verify.d.ts +20 -3
- package/dist/pipeline/verify.js +189 -35
- package/dist/probe/persist.d.ts +68 -0
- package/dist/probe/persist.js +184 -0
- package/dist/probe/queue.d.ts +12 -0
- package/dist/probe/queue.js +10 -2
- package/dist/report/brand-assets.generated.d.ts +9 -0
- package/dist/report/brand-assets.generated.js +8 -0
- package/dist/report/comment.d.ts +99 -6
- package/dist/report/comment.js +292 -103
- package/dist/report/html.d.ts +50 -0
- package/dist/report/html.js +879 -0
- package/dist/report/manifest.d.ts +29 -0
- package/dist/report/manifest.js +37 -0
- package/dist/report/run.d.ts +54 -1
- package/dist/report/run.js +34 -9
- package/dist/report/viewmodel.d.ts +91 -0
- package/dist/report/viewmodel.js +241 -0
- package/dist/review/adjudicate.d.ts +6 -6
- package/dist/review/adjudicate.js +2 -2
- package/dist/review/inline.d.ts +44 -0
- package/dist/review/inline.js +95 -0
- package/dist/review/packs.d.ts +21 -0
- package/dist/review/packs.js +47 -0
- package/dist/review/scope.d.ts +16 -0
- package/dist/review/scope.js +74 -0
- package/dist/review/secrets.d.ts +10 -10
- package/dist/review/secrets.js +7 -7
- package/dist/review/testfiles.d.ts +18 -0
- package/dist/review/testfiles.js +26 -0
- package/dist/review/triage.d.ts +1 -1
- package/dist/review/triage.js +10 -10
- package/dist/review/validate.d.ts +41 -0
- package/dist/review/validate.js +76 -0
- package/dist/ui/errors.d.ts +54 -0
- package/dist/ui/errors.js +236 -0
- package/dist/ui/style.d.ts +34 -0
- package/dist/ui/style.js +48 -0
- package/dist/ui/summary.d.ts +38 -0
- package/dist/ui/summary.js +101 -0
- package/dist/vision/cost.d.ts +1 -1
- package/dist/vision/decisions.d.ts +9 -3
- package/dist/vision/decisions.js +31 -21
- package/dist/vision/openrouter.d.ts +4 -0
- package/dist/vision/openrouter.js +30 -4
- package/package.json +11 -2
|
@@ -16,471 +16,1078 @@ function formatUsd(n) {
|
|
|
16
16
|
return `$${(n || 0).toFixed(6)}`
|
|
17
17
|
}
|
|
18
18
|
|
|
19
|
-
/** Escape a report string for one markdown table cell
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
19
|
+
/** Escape a report string for one markdown table cell — and mask
|
|
20
|
+
* secret-shaped tokens so a leaked credential never reaches a PR comment. */
|
|
21
|
+
const SECRET_PATTERNS = [
|
|
22
|
+
/sk-or-[A-Za-z0-9_-]{4,}/g,
|
|
23
|
+
/sk-[A-Za-z0-9_-]{8,}/g,
|
|
24
|
+
/gh[pousr]_[A-Za-z0-9_]{8,}/g,
|
|
25
|
+
/github_pat_[A-Za-z0-9_]{8,}/g,
|
|
26
|
+
/xox[baprs]-[A-Za-z0-9-]{8,}/g,
|
|
27
|
+
/AKIA[A-Z0-9]{16}/g,
|
|
28
|
+
/npm_[A-Za-z0-9]{8,}/g,
|
|
29
|
+
/Bearer\s+[A-Za-z0-9._~+/=-]{10,}/gi,
|
|
30
|
+
/eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{5,}/g,
|
|
31
|
+
/:\/\/[^/\s:@]{1,64}:[^/\s:@]{6,}@/g,
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
function maskSecrets(s) {
|
|
35
|
+
let out = s
|
|
36
|
+
for (const re of SECRET_PATTERNS) out = out.replace(re, '•••')
|
|
37
|
+
return out
|
|
25
38
|
}
|
|
26
39
|
|
|
27
|
-
function
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
lines.push(
|
|
34
|
-
'`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.',
|
|
35
|
-
)
|
|
36
|
-
lines.push('')
|
|
37
|
-
lines.push('This status is intentionally neutral, not a failure.')
|
|
38
|
-
lines.push('')
|
|
39
|
-
return lines.join('\n')
|
|
40
|
+
function cell(s, max = 200) {
|
|
41
|
+
return maskSecrets(
|
|
42
|
+
String(s ?? '')
|
|
43
|
+
.replace(/\|/g, '\\|')
|
|
44
|
+
.replace(/[\r\n]+/g, ' '),
|
|
45
|
+
).slice(0, max)
|
|
40
46
|
}
|
|
41
47
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
return lines.join('\n')
|
|
48
|
+
/** Inline code span around a cell-safe string. The fence is one backtick
|
|
49
|
+
* longer than any run inside, so report text cannot close it early. */
|
|
50
|
+
function code(s) {
|
|
51
|
+
const t = cell(s)
|
|
52
|
+
const longest = Math.max(0, ...(t.match(/`+/g) ?? []).map((r) => r.length))
|
|
53
|
+
const fence = '`'.repeat(longest + 1)
|
|
54
|
+
const pad = t.startsWith('`') || t.endsWith('`') ? ' ' : ''
|
|
55
|
+
return `${fence}${pad}${t}${pad}${fence}`
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function plural(n, one, many = `${one}s`) {
|
|
59
|
+
return `${n} ${n === 1 ? one : many}`
|
|
55
60
|
}
|
|
56
61
|
|
|
57
|
-
function
|
|
58
|
-
if (!
|
|
62
|
+
function formatDuration(ms) {
|
|
63
|
+
if (typeof ms !== 'number' || !Number.isFinite(ms) || ms < 0) return undefined
|
|
64
|
+
return ms < 1000 ? `${Math.round(ms)}ms` : `${(ms / 1000).toFixed(1)}s`
|
|
65
|
+
}
|
|
59
66
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
67
|
+
// --- Ocellus vocabulary (DESIGN.md 6.7) --------------------------------------
|
|
68
|
+
// This file's own copy of the src/report/viewmodel.ts vocabulary (plan KTD2):
|
|
69
|
+
// the action loads it with require() and no build step. The comment-golden
|
|
70
|
+
// parity test pins both copies equal.
|
|
71
|
+
const STATUS_GLYPH = {
|
|
72
|
+
passed: '●',
|
|
73
|
+
failed: '⊘',
|
|
74
|
+
skipped: '–',
|
|
75
|
+
blocked: '⊖',
|
|
76
|
+
unavailable: '◌',
|
|
77
|
+
inconclusive: '◐',
|
|
78
|
+
}
|
|
79
|
+
const PROOF_LEVELS = ['suspected', 'corroborated', 'exercised', 'reproduced']
|
|
80
|
+
const SEVERITY_GLYPH = { bug: '◆', risk: '◈', nit: '○', q: '□' }
|
|
81
|
+
const SEVERITY_LABEL = { bug: 'bug', risk: 'risk', nit: 'nit', q: 'question' }
|
|
82
|
+
const VERDICT_STATUS = { approve: 'passed', needs_changes: 'failed', pass: 'passed' }
|
|
83
|
+
const VERDICT_LABEL = { approve: 'approve', needs_changes: 'needs changes', pass: 'clean' }
|
|
69
84
|
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
pushReviewTop(lines, codeReview)
|
|
75
|
-
lines.push(
|
|
76
|
-
`**Summary:** ${report.totals.passed}/${report.totals.tests} passed · ` +
|
|
77
|
-
`${report.totals.visionCalls} vision calls · ` +
|
|
78
|
-
`${formatUsd(report.totals.visionCostUsd)} spend · ` +
|
|
79
|
-
`${report.totals.sandboxSeconds.toFixed(1)}s sandbox`,
|
|
80
|
-
)
|
|
81
|
-
lines.push(
|
|
82
|
-
`**Fingerprint cache:** ${report.totals.cacheHits ?? 0} hit(s) · ` +
|
|
83
|
-
`${report.totals.cacheMisses ?? 0} miss(es) · ${report.totals.cacheHeals ?? 0} heal(s)`,
|
|
84
|
-
)
|
|
85
|
-
lines.push('')
|
|
85
|
+
function proofMeter(level) {
|
|
86
|
+
const filled = PROOF_LEVELS.indexOf(level ?? '') + 1
|
|
87
|
+
return '▰'.repeat(filled) + '▱'.repeat(PROOF_LEVELS.length - filled)
|
|
88
|
+
}
|
|
86
89
|
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
90
|
+
/** Status is always glyph plus lowercase word (R2). */
|
|
91
|
+
function statusText(status) {
|
|
92
|
+
return `${STATUS_GLYPH[status]} ${status}`
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Proof cell: blank for a lane that did not run, the empty meter plus
|
|
96
|
+
* "none" for one that ran and proved nothing, otherwise meter plus word. */
|
|
97
|
+
function proofText(level) {
|
|
98
|
+
if (level === null) return ''
|
|
99
|
+
if (level === 'none') return `${proofMeter(undefined)} none`
|
|
100
|
+
return `${proofMeter(level)} ${level}`
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function ladderLevel(status) {
|
|
104
|
+
return PROOF_LEVELS.includes(status) ? status : 'suspected'
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function findingsOf(codeReview) {
|
|
108
|
+
return codeReview && Array.isArray(codeReview.findings) ? codeReview.findings : []
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/** Strongest proof any finding reached; a review with no evidence is a suspicion. */
|
|
112
|
+
function bestFindingProof(codeReview) {
|
|
113
|
+
let best = 0
|
|
114
|
+
for (const f of findingsOf(codeReview)) {
|
|
115
|
+
best = Math.max(best, PROOF_LEVELS.indexOf(f.evidence?.status))
|
|
93
116
|
}
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
117
|
+
return PROOF_LEVELS[best]
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** How far a lane's result is proven (A4). a0 is self-reported, the browser
|
|
121
|
+
* lanes exercise the app, the review is as strong as its best evidence. */
|
|
122
|
+
function laneProof(lane, status, codeReview) {
|
|
123
|
+
if (status === 'skipped') return null
|
|
124
|
+
if (status === 'blocked' || status === 'unavailable') return 'none'
|
|
125
|
+
if (status === 'inconclusive' || lane === 'a0') return 'suspected'
|
|
126
|
+
if (lane === 'review') return bestFindingProof(codeReview)
|
|
127
|
+
return 'exercised'
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// --- run-manifest lanes -------------------------------------------------------
|
|
131
|
+
// The manifest is the shared evidence contract (R15): these labels/statuses/
|
|
132
|
+
// costs/head fields must stay identical to the TUI and dashboard rendering —
|
|
133
|
+
// the parity contract test enforces it.
|
|
134
|
+
const MANIFEST_LANE_ORDER = ['review', 'flow', 'app', 'a0']
|
|
135
|
+
|
|
136
|
+
// The commit-status surface validates at least as strictly as the display
|
|
137
|
+
// surfaces (viewmodel.isRunManifest / collect.validManifest) — this file is
|
|
138
|
+
// what can publish a green check, so it never trusts a shallow sniff.
|
|
139
|
+
function validManifest(m) {
|
|
140
|
+
const num = (n) => typeof n === 'number' && Number.isFinite(n)
|
|
141
|
+
if (m === null || typeof m !== 'object' || Array.isArray(m)) return false
|
|
142
|
+
if (m.schemaVersion !== 1 || typeof m.runId !== 'string') return false
|
|
143
|
+
if (typeof m.startedAt !== 'string') return false
|
|
144
|
+
if (m.identity === null || typeof m.identity !== 'object' || Array.isArray(m.identity)) {
|
|
145
|
+
return false
|
|
146
|
+
}
|
|
147
|
+
for (const v of [
|
|
148
|
+
m.identity.repo,
|
|
149
|
+
m.identity.pr,
|
|
150
|
+
m.identity.intendedHeadSha,
|
|
151
|
+
m.identity.checkoutSha,
|
|
152
|
+
m.identity.baseSha,
|
|
153
|
+
m.identity.runNonce,
|
|
154
|
+
]) {
|
|
155
|
+
if (v !== undefined && typeof v !== 'string') return false
|
|
156
|
+
}
|
|
157
|
+
if (m.aggregate === null || typeof m.aggregate !== 'object') return false
|
|
158
|
+
// aggregate.status must be a real lane status and ok a real boolean —
|
|
159
|
+
// a type-confused aggregate must fail the gate, not reach the renderer.
|
|
160
|
+
if (!Object.hasOwn(STATUS_GLYPH, m.aggregate.status)) return false
|
|
161
|
+
if (m.aggregate.ok !== true && m.aggregate.ok !== false) return false
|
|
162
|
+
if (!num(m.aggregate.calls) || !num(m.aggregate.tokens) || !num(m.aggregate.costUsd)) {
|
|
163
|
+
return false
|
|
164
|
+
}
|
|
165
|
+
if (m.lanes === null || typeof m.lanes !== 'object') return false
|
|
166
|
+
return MANIFEST_LANE_ORDER.every((id) => {
|
|
167
|
+
const lane = m.lanes[id]
|
|
168
|
+
return (
|
|
169
|
+
lane !== null &&
|
|
170
|
+
typeof lane === 'object' &&
|
|
171
|
+
lane.lane === id &&
|
|
172
|
+
typeof lane.selected === 'boolean' &&
|
|
173
|
+
Object.hasOwn(STATUS_GLYPH, lane.status) &&
|
|
174
|
+
lane.usage !== null &&
|
|
175
|
+
typeof lane.usage === 'object' &&
|
|
176
|
+
num(lane.usage.calls) &&
|
|
177
|
+
num(lane.usage.tokens) &&
|
|
178
|
+
num(lane.usage.costUsd) &&
|
|
179
|
+
lane.budget !== null &&
|
|
180
|
+
typeof lane.budget === 'object'
|
|
181
|
+
)
|
|
182
|
+
})
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
function manifestLanes(manifest) {
|
|
186
|
+
const lanes = manifest && manifest.lanes ? manifest.lanes : {}
|
|
187
|
+
return MANIFEST_LANE_ORDER.map((id) => lanes[id]).filter(
|
|
188
|
+
(l) => l !== null && typeof l === 'object' && Object.hasOwn(STATUS_GLYPH, l.status),
|
|
97
189
|
)
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/** One lane-table row per manifest lane, canonical order. */
|
|
193
|
+
function manifestLaneRows(manifest, codeReview) {
|
|
194
|
+
return manifestLanes(manifest).map((lane) => {
|
|
195
|
+
if (lane.selected !== true) {
|
|
196
|
+
return { lane: lane.lane, status: 'skipped', result: 'not selected', proof: null, spend: '' }
|
|
103
197
|
}
|
|
104
|
-
|
|
198
|
+
const usage = lane.usage || {}
|
|
199
|
+
return {
|
|
200
|
+
lane: lane.lane,
|
|
201
|
+
status: lane.status,
|
|
202
|
+
result: lane.reason ?? lane.summary ?? '',
|
|
203
|
+
proof: laneProof(lane.lane, lane.status, codeReview),
|
|
204
|
+
spend: usage.metered === true ? formatUsd(usage.costUsd) : 'unmetered',
|
|
205
|
+
}
|
|
206
|
+
})
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
function reproducedCount(codeReview) {
|
|
210
|
+
return typeof codeReview.provenBlockers === 'number'
|
|
211
|
+
? codeReview.provenBlockers
|
|
212
|
+
: findingsOf(codeReview).filter((f) => f.evidence?.status === 'reproduced').length
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/** Review-lane row synthesized from code-review.json when no manifest exists.
|
|
216
|
+
* A missing report means the review step crashed: failed, never skipped. */
|
|
217
|
+
function reviewLaneRow(codeReview) {
|
|
218
|
+
if (!codeReview) {
|
|
219
|
+
return { lane: 'review', status: 'failed', result: 'no code-review.json', proof: 'none', spend: '' }
|
|
105
220
|
}
|
|
106
|
-
|
|
107
|
-
|
|
221
|
+
if (codeReview.skipped) {
|
|
222
|
+
return { lane: 'review', status: 'skipped', result: codeReview.summary ?? 'skipped', proof: null, spend: '' }
|
|
223
|
+
}
|
|
224
|
+
const status = codeReview.ok === true ? 'passed' : 'failed'
|
|
225
|
+
return {
|
|
226
|
+
lane: 'review',
|
|
227
|
+
status,
|
|
228
|
+
result: `${plural(findingsOf(codeReview).length, 'finding')}, ${reproducedCount(codeReview)} reproduced`,
|
|
229
|
+
proof: laneProof('review', status, codeReview),
|
|
230
|
+
spend: formatUsd(codeReview.visionCostUsd),
|
|
231
|
+
}
|
|
232
|
+
}
|
|
108
233
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`,
|
|
118
|
-
)
|
|
234
|
+
function healsOf(report) {
|
|
235
|
+
return (report.tests ?? []).flatMap((t) => t.healEvents ?? [])
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/** Flow-lane row synthesized from run.json when no manifest exists. */
|
|
239
|
+
function flowLaneRow(report) {
|
|
240
|
+
if (!report) {
|
|
241
|
+
return { lane: 'flow', status: 'skipped', result: 'run lane disabled', proof: null, spend: '' }
|
|
119
242
|
}
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
243
|
+
const totals = report.totals ?? {}
|
|
244
|
+
const heals = healsOf(report).length
|
|
245
|
+
const status = report.ok === true ? 'passed' : 'failed'
|
|
246
|
+
return {
|
|
247
|
+
lane: 'flow',
|
|
248
|
+
status,
|
|
249
|
+
result: `${totals.passed ?? 0} of ${totals.tests ?? 0} tests passed${heals > 0 ? `, ${heals} healed` : ''}`,
|
|
250
|
+
proof: laneProof('flow', status, undefined),
|
|
251
|
+
spend: formatUsd(totals.visionCostUsd),
|
|
252
|
+
}
|
|
253
|
+
}
|
|
123
254
|
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
const
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
: '$0.00'
|
|
134
|
-
lines.push(`| Per-call cost (avg) | ${perCall} |`)
|
|
135
|
-
for (const model of Object.keys(report.totals.callsByModel ?? {}).sort()) {
|
|
136
|
-
lines.push(`| Calls (${model}) | ${report.totals.callsByModel[model]} |`)
|
|
137
|
-
lines.push(`| Spend (${model}) | ${formatUsd(report.totals.costByModel?.[model] ?? 0)} |`)
|
|
138
|
-
}
|
|
139
|
-
lines.push(`| Total vision spend | ${formatUsd(report.totals.visionCostUsd)} |`)
|
|
140
|
-
lines.push(`| Sandbox seconds | ${report.totals.sandboxSeconds.toFixed(1)}s |`)
|
|
141
|
-
if (budgetCap > 0) {
|
|
142
|
-
lines.push(`| Budget cap | ${formatUsd(budgetCap)} |`)
|
|
143
|
-
lines.push(`| Budget exceeded | ${report.totals.budgetExceeded ? '⚠️ yes' : '✅ no'} |`)
|
|
255
|
+
// --- the one layout (R6) ---------------------------------------------------------
|
|
256
|
+
// header, verdict line, lane table, findings summary, folds, footer. Every
|
|
257
|
+
// body below is this layout with different parts; the first screen stays
|
|
258
|
+
// within 12 lines before the first fold.
|
|
259
|
+
|
|
260
|
+
function laneTable(rows) {
|
|
261
|
+
const lines = ['| Status | Lane | Result | Proof | Spend |', '|---|---|---|---|--:|']
|
|
262
|
+
for (const r of rows) {
|
|
263
|
+
lines.push(`| ${statusText(r.status)} | ${cell(r.lane)} | ${cell(r.result)} | ${proofText(r.proof)} | ${r.spend} |`)
|
|
144
264
|
}
|
|
145
|
-
lines
|
|
146
|
-
|
|
147
|
-
lines.push('')
|
|
265
|
+
return lines
|
|
266
|
+
}
|
|
148
267
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
268
|
+
/** Header word: a failed run is never shown with a positive verdict. */
|
|
269
|
+
function headline(ok, aggregateStatus, codeReview) {
|
|
270
|
+
const verdict =
|
|
271
|
+
codeReview && !codeReview.skipped && Object.hasOwn(VERDICT_LABEL, codeReview.verdict)
|
|
272
|
+
? codeReview.verdict
|
|
273
|
+
: undefined
|
|
274
|
+
if (ok !== true) {
|
|
275
|
+
if (verdict === 'needs_changes') return `${STATUS_GLYPH.failed} ${VERDICT_LABEL.needs_changes}`
|
|
276
|
+
const s = aggregateStatus !== undefined && !['passed', 'skipped'].includes(aggregateStatus)
|
|
277
|
+
? aggregateStatus
|
|
278
|
+
: 'failed'
|
|
279
|
+
return statusText(s)
|
|
280
|
+
}
|
|
281
|
+
if (verdict !== undefined) return `${STATUS_GLYPH[VERDICT_STATUS[verdict]]} ${VERDICT_LABEL[verdict]}`
|
|
282
|
+
return statusText(aggregateStatus ?? 'passed')
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/** Bold lead of the verdict line: the one fact a reader needs first. */
|
|
286
|
+
function verdictLead(rows, codeReview) {
|
|
287
|
+
const findings = findingsOf(codeReview)
|
|
288
|
+
const reviewed = codeReview && !codeReview.skipped
|
|
289
|
+
if (reviewed) {
|
|
290
|
+
const reproduced = reproducedCount(codeReview)
|
|
291
|
+
if (reproduced > 0) {
|
|
292
|
+
const files = new Set(
|
|
293
|
+
findings.filter((f) => f.evidence?.status === 'reproduced').map((f) => f.file),
|
|
294
|
+
)
|
|
295
|
+
const where = files.size === 1 ? ` in ${code([...files][0])}` : ''
|
|
296
|
+
return `**${plural(reproduced, 'finding')} reproduced**${where}`
|
|
158
297
|
}
|
|
159
298
|
}
|
|
299
|
+
const failing = rows.filter((r) => r.lane !== 'review' && !['passed', 'skipped'].includes(r.status))
|
|
300
|
+
if (failing.length > 0) {
|
|
301
|
+
return `**${failing.map((r) => `${cell(r.lane)} ${r.status}`).join(', ')}**`
|
|
302
|
+
}
|
|
303
|
+
if (reviewed && findings.length > 0) return `**${plural(findings.length, 'finding')}, none reproduced**`
|
|
304
|
+
const review = rows.find((r) => r.lane === 'review')
|
|
305
|
+
if (review !== undefined && !['passed', 'skipped'].includes(review.status)) {
|
|
306
|
+
return `**review ${review.status}**`
|
|
307
|
+
}
|
|
308
|
+
if (reviewed) return '**No findings**'
|
|
309
|
+
return rows.some((r) => r.status !== 'skipped') ? '**All selected lanes passed**' : '**No lane ran**'
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
function verdictLine({ rows, codeReview, headSha, binding, costUsd, durationMs }) {
|
|
313
|
+
const bits = [verdictLead(rows, codeReview)]
|
|
314
|
+
if (typeof headSha === 'string' && headSha !== '') bits.push(`head ${code(headSha.slice(0, 7))}`)
|
|
315
|
+
if (binding && binding.status === 'mismatch') bits.push('head binding mismatch')
|
|
316
|
+
bits.push(formatUsd(costUsd))
|
|
317
|
+
const duration = formatDuration(durationMs)
|
|
318
|
+
if (duration !== undefined) bits.push(duration)
|
|
319
|
+
return bits.join(' · ')
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
const NO_REVIEW_REPORT = 'No code review report was found; check the action logs before merging.'
|
|
323
|
+
const NO_REVIEW_ATTACHED = 'No code review report is attached to this run.'
|
|
324
|
+
|
|
325
|
+
function findingsLine(codeReview, missing = NO_REVIEW_REPORT) {
|
|
326
|
+
if (!codeReview) return missing
|
|
327
|
+
if (codeReview.skipped) return `Code review skipped: ${cell(codeReview.summary, 300)}`
|
|
328
|
+
const findings = findingsOf(codeReview)
|
|
329
|
+
const sev = { bug: 0, risk: 0, nit: 0, q: 0 }
|
|
330
|
+
for (const f of findings) if (Object.hasOwn(sev, f.severity)) sev[f.severity] += 1
|
|
331
|
+
const parts = [
|
|
332
|
+
`${SEVERITY_GLYPH.bug} ${plural(sev.bug, 'bug')}`,
|
|
333
|
+
`${SEVERITY_GLYPH.risk} ${plural(sev.risk, 'risk')}`,
|
|
334
|
+
`${SEVERITY_GLYPH.nit} ${plural(sev.nit, 'nit')}`,
|
|
335
|
+
]
|
|
336
|
+
if (sev.q > 0) parts.push(`${SEVERITY_GLYPH.q} ${plural(sev.q, 'question')}`)
|
|
337
|
+
// p alone is never proof: high-confidence is counted apart from reproduced
|
|
338
|
+
// (KTD2). Serialized counts win; the recount serves older reports and
|
|
339
|
+
// P_FALLBACK_GATE must match P_TRUE_POSITIVE_THRESHOLD in src/cli.ts.
|
|
340
|
+
const confident =
|
|
341
|
+
typeof codeReview.highConfidenceBlockers === 'number'
|
|
342
|
+
? codeReview.highConfidenceBlockers
|
|
343
|
+
: findings.filter((f) => typeof f.p === 'number' && f.p >= P_FALLBACK_GATE).length
|
|
344
|
+
if (confident > 0) parts.push(`${confident} high-confidence`)
|
|
345
|
+
// Counts serialized comments carrying a committable block (what lands on
|
|
346
|
+
// the PR), not every finding the model offered a patch for.
|
|
347
|
+
const suggestions = Array.isArray(codeReview.reviewComments)
|
|
348
|
+
? codeReview.reviewComments.filter((c) => extractSuggestion(c.body ?? '') !== '').length
|
|
349
|
+
: findings.filter((f) => typeof f.suggestion === 'string' && f.suggestion !== '').length
|
|
350
|
+
if (suggestions > 0) parts.push(`${plural(suggestions, 'suggestion')} ready to commit`)
|
|
351
|
+
return parts.join(' · ')
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
const P_FALLBACK_GATE = 0.7
|
|
355
|
+
|
|
356
|
+
function fold(lines, title, body) {
|
|
357
|
+
if (body.length === 0) return
|
|
358
|
+
lines.push('<details>')
|
|
359
|
+
lines.push(`<summary>${title}</summary>`)
|
|
160
360
|
lines.push('')
|
|
361
|
+
lines.push(...body)
|
|
362
|
+
if (body[body.length - 1] !== '') lines.push('')
|
|
161
363
|
lines.push('</details>')
|
|
162
364
|
lines.push('')
|
|
365
|
+
}
|
|
163
366
|
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
367
|
+
function footer(meta, runUrl) {
|
|
368
|
+
const bits = [`Argus ${meta.version ?? packageVersion()}`]
|
|
369
|
+
const url = runUrl ?? meta.runUrl
|
|
370
|
+
if (url) bits.push(`[workflow run and evidence](${url})`)
|
|
371
|
+
// U14 (KTD12, Q13): name where report.html sits in the consumer's upload.
|
|
372
|
+
if (meta.reportHtml) bits.push(`report ${code(meta.reportHtml)} in the run artifacts`)
|
|
373
|
+
bits.push('self-hosted, BYOK')
|
|
374
|
+
return `<sub>${bits.join(' · ')}</sub>`
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
function packageVersion() {
|
|
378
|
+
return require('../package.json').version
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
/** KTD12: the comment budget (DESIGN.md 10), well under GitHub's 65,536-char cap. */
|
|
382
|
+
const COMMENT_BUDGET_BYTES = 20 * 1024
|
|
383
|
+
|
|
384
|
+
/** Fold keys in the order they collapse when a body is over budget (KTD12),
|
|
385
|
+
* then the remaining folds so the post never fails on length (R10). */
|
|
386
|
+
const COLLAPSE_ORDER = ['diagnostics', 'spend', 'heals', 'findings', 'explore', 'tests']
|
|
387
|
+
|
|
388
|
+
/** One-line pointer that replaces a collapsed fold. Lines in `keep` (the
|
|
389
|
+
* hidden `@argus persist` payload) survive the collapse. */
|
|
390
|
+
function collapsedFold(title, keep) {
|
|
391
|
+
return [
|
|
392
|
+
`<details><summary>${title}: omitted to keep this comment under 20 KB</summary>` +
|
|
393
|
+
'The full detail is in the report files and the workflow run linked below.</details>',
|
|
394
|
+
'',
|
|
395
|
+
...keep.flatMap((k) => [k, '']),
|
|
396
|
+
]
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
/** The shared layout. `folds` is a list of [title, bodyLines, {key, keep}].
|
|
400
|
+
* Notices (one quoted line each) sit between the findings summary and the
|
|
401
|
+
* folds. Length is measured after the full render; while over budget, folds
|
|
402
|
+
* collapse to a pointer in COLLAPSE_ORDER. */
|
|
403
|
+
function layout({ status, verdict, rows, summary, folds, meta, runUrl }) {
|
|
404
|
+
const budget = typeof meta.budgetBytes === 'number' ? meta.budgetBytes : COMMENT_BUDGET_BYTES
|
|
405
|
+
const notices = Array.isArray(meta.notices) ? meta.notices : []
|
|
406
|
+
const collapsed = new Set()
|
|
407
|
+
const render = () => {
|
|
408
|
+
const lines = [SENTINEL, `### Argus: ${status}`, '', verdict, '', ...laneTable(rows), '', summary, '']
|
|
409
|
+
for (const n of notices) lines.push(`> ${n}`, '')
|
|
410
|
+
for (const [title, body, opts = {}] of folds) {
|
|
411
|
+
if (body.length === 0) continue
|
|
412
|
+
if (collapsed.has(opts.key)) lines.push(...collapsedFold(title, opts.keep ?? []))
|
|
413
|
+
else fold(lines, title, body)
|
|
175
414
|
}
|
|
415
|
+
lines.push(footer(meta, runUrl))
|
|
176
416
|
lines.push('')
|
|
417
|
+
return lines.join('\n')
|
|
177
418
|
}
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
419
|
+
let body = render()
|
|
420
|
+
for (const key of COLLAPSE_ORDER) {
|
|
421
|
+
if (Buffer.byteLength(body) <= budget) break
|
|
422
|
+
const present = folds.some(([, b, opts = {}]) => opts.key === key && b.length > 0)
|
|
423
|
+
if (!present) continue
|
|
424
|
+
collapsed.add(key)
|
|
425
|
+
body = render()
|
|
181
426
|
}
|
|
182
|
-
|
|
183
|
-
|
|
427
|
+
return body
|
|
428
|
+
}
|
|
184
429
|
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
430
|
+
// --- folds ---------------------------------------------------------------------
|
|
431
|
+
|
|
432
|
+
const MAX_FINDING_ROWS = 25
|
|
433
|
+
|
|
434
|
+
function severityText(s) {
|
|
435
|
+
return Object.hasOwn(SEVERITY_GLYPH, s) ? `${SEVERITY_GLYPH[s]} ${SEVERITY_LABEL[s]}` : cell(s ?? 'unknown')
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
function findingsFold(codeReview, inlinePlan) {
|
|
439
|
+
if (!codeReview || codeReview.skipped) return []
|
|
440
|
+
const body = []
|
|
441
|
+
const findings = findingsOf(codeReview)
|
|
442
|
+
if (codeReview.summary) {
|
|
443
|
+
body.push(cell(codeReview.summary, 1000))
|
|
444
|
+
body.push('')
|
|
190
445
|
}
|
|
191
|
-
if (
|
|
192
|
-
|
|
193
|
-
|
|
446
|
+
if (findings.length > 0) {
|
|
447
|
+
body.push('| Severity | Proof | p | Category | Location | Finding |')
|
|
448
|
+
body.push('|---|---|---|---|---|---|')
|
|
449
|
+
// Findings/evidence strings are model- and probe-emitted: sanitized per
|
|
450
|
+
// cell, and the section is bounded so an oversized report can't push the
|
|
451
|
+
// body past GitHub's 65536-char comment limit.
|
|
452
|
+
for (const f of findings.slice(0, MAX_FINDING_ROWS)) {
|
|
453
|
+
const level = ladderLevel(f.evidence?.status)
|
|
454
|
+
const p = typeof f.p === 'number' ? f.p.toFixed(2) : ''
|
|
455
|
+
const where = typeof f.line === 'number' ? `${f.file}:${f.line}` : f.file
|
|
456
|
+
// An inconclusive link says the same thing on every row; Diagnostics states it once.
|
|
457
|
+
const evidence =
|
|
458
|
+
f.evidence?.detail && f.evidence.status !== 'inconclusive' ? `<br>evidence: ${cell(f.evidence.detail)}` : ''
|
|
459
|
+
body.push(
|
|
460
|
+
`| ${severityText(f.severity)} | ${proofText(level)} | ${p} | ${cell(f.category ?? '')} | ` +
|
|
461
|
+
`${code(where)} | ${cell(f.message)}${evidence} |`,
|
|
462
|
+
)
|
|
463
|
+
}
|
|
464
|
+
if (findings.length > MAX_FINDING_ROWS) {
|
|
465
|
+
body.push(`| | | | | | ${findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`)
|
|
466
|
+
}
|
|
467
|
+
body.push('')
|
|
194
468
|
}
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
)
|
|
213
|
-
lines.push(
|
|
214
|
-
`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`,
|
|
469
|
+
// Serialized comments that didn't post: the maxComments cap plus post-time
|
|
470
|
+
// drops (off-diff anchors, retry-ladder discards).
|
|
471
|
+
if (inlinePlan !== undefined && inlinePlan.dropped > 0) {
|
|
472
|
+
const overflow = inlinePlan.overflow ?? 0
|
|
473
|
+
const reasons = []
|
|
474
|
+
if (overflow > 0) reasons.push(`\`review.maxComments\` cap ${inlinePlan.cap}`)
|
|
475
|
+
if (inlinePlan.dropped - overflow > 0) {
|
|
476
|
+
reasons.push(`${inlinePlan.dropped - overflow} outside the PR diff`)
|
|
477
|
+
}
|
|
478
|
+
body.push(`*${plural(inlinePlan.dropped, 'inline comment')} not posted: ${reasons.join(', ')}.*`)
|
|
479
|
+
body.push('')
|
|
480
|
+
}
|
|
481
|
+
// E1.U3: reproduced probes render copy-pasteable source plus the payload
|
|
482
|
+
// `@argus persist` parses back. Bounded (2 probes, 8KB each) and fenced;
|
|
483
|
+
// ``` runs inside the source are flattened so it can't break the fence.
|
|
484
|
+
const persistable = (codeReview.probes ?? []).filter(
|
|
485
|
+
(p) => p.outcome === 'reproduced' && typeof p.path === 'string' && typeof p.content === 'string',
|
|
215
486
|
)
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
lines.push(
|
|
220
|
-
`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
|
|
487
|
+
if (persistable.length > 0) {
|
|
488
|
+
body.push(
|
|
489
|
+
'Reproduced probes: comment `@argus persist` to open a regression-test PR, or copy them into your suite.',
|
|
221
490
|
)
|
|
222
|
-
|
|
223
|
-
|
|
491
|
+
body.push('')
|
|
492
|
+
for (const p of persistable.slice(0, 2)) {
|
|
493
|
+
const rendered = p.content.replace(/`{3,}/g, '``').slice(0, 8 * 1024)
|
|
494
|
+
body.push('<details>')
|
|
495
|
+
body.push(`<summary>Probe ${code(p.path)}</summary>`)
|
|
496
|
+
body.push('')
|
|
497
|
+
body.push('```ts')
|
|
498
|
+
body.push(rendered)
|
|
499
|
+
body.push('```')
|
|
500
|
+
body.push('</details>')
|
|
501
|
+
body.push('')
|
|
502
|
+
}
|
|
503
|
+
if (persistable.length > 2) {
|
|
504
|
+
body.push(`*${persistable.length - 2} more in \`code-review.json\`.*`)
|
|
505
|
+
body.push('')
|
|
506
|
+
}
|
|
224
507
|
}
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
508
|
+
const payload = persistPayloadOf(codeReview)
|
|
509
|
+
if (payload !== undefined) {
|
|
510
|
+
body.push(payload)
|
|
511
|
+
body.push('')
|
|
512
|
+
}
|
|
513
|
+
return body
|
|
514
|
+
}
|
|
228
515
|
|
|
229
|
-
|
|
516
|
+
/** The hidden `@argus persist` payload, which must reach the posted body even
|
|
517
|
+
* when the Findings fold collapses: the persist command parses it back. */
|
|
518
|
+
function persistPayloadOf(codeReview) {
|
|
519
|
+
return codeReview && typeof codeReview.persistPayload === 'string' && codeReview.persistPayload.length < 32768
|
|
520
|
+
? codeReview.persistPayload
|
|
521
|
+
: undefined
|
|
522
|
+
}
|
|
230
523
|
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
lines.push('')
|
|
240
|
-
lines.push('---')
|
|
241
|
-
lines.push('')
|
|
242
|
-
lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
|
|
243
|
-
lines.push('')
|
|
244
|
-
return lines.join('\n')
|
|
524
|
+
/** Findings fold entry for layout(): collapsible, keeping the persist payload. */
|
|
525
|
+
function findingsEntry(codeReview, inlinePlan) {
|
|
526
|
+
const payload = persistPayloadOf(codeReview)
|
|
527
|
+
return [
|
|
528
|
+
`Findings (${findingsOf(codeReview).length})`,
|
|
529
|
+
findingsFold(codeReview, inlinePlan),
|
|
530
|
+
{ key: 'findings', keep: payload !== undefined ? [payload] : [] },
|
|
531
|
+
]
|
|
245
532
|
}
|
|
246
533
|
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
// kinds, what needs me" before the <details> fold. ⛔ counts
|
|
251
|
-
// probe-reproduced findings only — a high p alone is never "proven" —
|
|
252
|
-
// while ◎ carries the Jev-confidence count; the two overlap when a
|
|
253
|
-
// finding is both (KTD2). The serialized blocker counts win when present
|
|
254
|
-
// (same numbers reviewBody prints); the recount is the fallback for
|
|
255
|
-
// reports predating them — P_FALLBACK_GATE must match
|
|
256
|
-
// P_TRUE_POSITIVE_THRESHOLD in src/cli.ts.
|
|
257
|
-
|
|
258
|
-
const REVIEW_VERDICT_ICON = { pass: '✅', approve: '👍', needs_changes: '🔴' }
|
|
259
|
-
const P_FALLBACK_GATE = 0.7
|
|
534
|
+
function assertionStatus(verdict) {
|
|
535
|
+
return verdict === 'pass' ? 'passed' : verdict === 'fail' ? 'failed' : 'skipped'
|
|
536
|
+
}
|
|
260
537
|
|
|
261
|
-
function
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
const
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
lines.push('')
|
|
271
|
-
return
|
|
538
|
+
function testsFold(report) {
|
|
539
|
+
const tests = report.tests ?? []
|
|
540
|
+
if (tests.length === 0) return []
|
|
541
|
+
const body = ['| Test | Result | Calls | Spend | Heals | Asserts |', '|---|---|--:|--:|--:|--:|']
|
|
542
|
+
for (const t of tests) {
|
|
543
|
+
body.push(
|
|
544
|
+
`| ${cell(t.name)} | ${statusText(t.ok ? 'passed' : 'failed')} | ${t.visionCalls ?? 0} | ` +
|
|
545
|
+
`${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`,
|
|
546
|
+
)
|
|
272
547
|
}
|
|
273
|
-
|
|
274
|
-
for (const
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
const
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
: findings.filter((f) => f.evidence?.status === 'reproduced').length
|
|
287
|
-
if (reproduced > 0) tail.push(`⛔ ${reproduced} reproduced`)
|
|
288
|
-
const confident =
|
|
289
|
-
typeof codeReview.highConfidenceBlockers === 'number'
|
|
290
|
-
? codeReview.highConfidenceBlockers
|
|
291
|
-
: findings.filter((f) => typeof f.p === 'number' && f.p >= P_FALLBACK_GATE).length
|
|
292
|
-
if (confident > 0) tail.push(`◎ ${confident} high-confidence`)
|
|
293
|
-
lines.push(
|
|
294
|
-
`🐛 ${sev.bug} · ⚠️ ${sev.risk} · 💡 ${sev.nit} · ❓ ${sev.q}` +
|
|
295
|
-
(tail.length > 0 ? ` — ${tail.join(' · ')}` : ''),
|
|
296
|
-
)
|
|
297
|
-
lines.push('')
|
|
548
|
+
body.push('')
|
|
549
|
+
for (const t of tests) {
|
|
550
|
+
if (!t.asserts || t.asserts.length === 0) continue
|
|
551
|
+
body.push(`**${cell(t.name)}**`)
|
|
552
|
+
for (const a of t.asserts) {
|
|
553
|
+
body.push(`- ${statusText(assertionStatus(a.verdict))} · *${cell(a.question)}*: ${cell(a.reasoning, 500)}`)
|
|
554
|
+
}
|
|
555
|
+
body.push('')
|
|
556
|
+
}
|
|
557
|
+
const videos = report.artifacts?.videos ?? []
|
|
558
|
+
for (const v of videos) body.push(`- video: ${code(v)}`)
|
|
559
|
+
if (videos.length > 0) body.push('')
|
|
560
|
+
return body
|
|
298
561
|
}
|
|
299
562
|
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
563
|
+
function healsFold(heals) {
|
|
564
|
+
return heals.map((h) => `- ${code(h.instruction)} healed with ${cell(h.model || 'unknown model')}`)
|
|
565
|
+
}
|
|
566
|
+
|
|
567
|
+
const CAPTURE_LABEL = {
|
|
568
|
+
'console-error': 'console error',
|
|
569
|
+
pageerror: 'page error',
|
|
570
|
+
'request-failed': 'failed request',
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
// U4a exploratory lane: observed runtime anomalies from the browser session.
|
|
574
|
+
// Evidence only; captures never change the verdict.
|
|
575
|
+
function exploreFold(report) {
|
|
576
|
+
const explore = report.explore
|
|
577
|
+
if (!explore || explore.enabled !== true) return []
|
|
578
|
+
const body = []
|
|
579
|
+
if (typeof explore.skipped === 'string') {
|
|
580
|
+
body.push(`Explore skipped: ${cell(explore.skipped)}`)
|
|
581
|
+
return body
|
|
582
|
+
}
|
|
583
|
+
if (typeof explore.steps === 'number') {
|
|
584
|
+
const pages = typeof explore.visited === 'number' ? explore.visited : 0
|
|
585
|
+
const spend = typeof explore.visionCostUsd === 'number' ? ` · ${formatUsd(explore.visionCostUsd)}` : ''
|
|
586
|
+
body.push(
|
|
587
|
+
`explored **${explore.steps}** step(s) across **${pages}** page(s), ` +
|
|
588
|
+
`stopped: ${cell(explore.stopReason ?? 'unknown')}${spend}`,
|
|
313
589
|
)
|
|
590
|
+
body.push('')
|
|
314
591
|
}
|
|
315
|
-
//
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
592
|
+
// The same signature can appear on several test reports from one file's
|
|
593
|
+
// shared browser session; collapse before rendering.
|
|
594
|
+
const seen = new Map()
|
|
595
|
+
for (const c of [...(explore.captures ?? []), ...(report.tests ?? []).flatMap((t) => t.captures ?? [])]) {
|
|
596
|
+
const key = `${c.kind}|${c.text}|${c.url ?? ''}`
|
|
597
|
+
const existing = seen.get(key)
|
|
598
|
+
if (existing) existing.count += c.count
|
|
599
|
+
else seen.set(key, { ...c })
|
|
600
|
+
}
|
|
601
|
+
const caps = [...seen.values()]
|
|
602
|
+
if (caps.length === 0) {
|
|
603
|
+
body.push('No page errors, console errors, or failed same-origin requests captured.')
|
|
604
|
+
return body
|
|
605
|
+
}
|
|
606
|
+
for (const c of caps.slice(0, 10)) {
|
|
607
|
+
const times = c.count > 1 ? ` ×${c.count}` : ''
|
|
608
|
+
const target = c.url !== undefined ? ` at ${code(c.url)}` : ''
|
|
609
|
+
body.push(`- observed · ${CAPTURE_LABEL[c.kind] ?? cell(c.kind)}${times}: ${code(c.text)}${target}`)
|
|
610
|
+
}
|
|
611
|
+
if (caps.length > 10) body.push(`- ${caps.length - 10} more distinct captures`)
|
|
612
|
+
body.push('')
|
|
613
|
+
body.push('*Observed findings are evidence only; they do not change the verdict.*')
|
|
614
|
+
return body
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
function spendFold(manifest, report, codeReview) {
|
|
618
|
+
const body = []
|
|
619
|
+
if (manifest !== undefined) {
|
|
620
|
+
body.push('| Lane | Model | Calls | Tokens | Spend |', '|---|---|--:|--:|--:|')
|
|
621
|
+
for (const lane of manifestLanes(manifest)) {
|
|
622
|
+
if (lane.selected !== true) continue
|
|
623
|
+
const usage = lane.usage || {}
|
|
624
|
+
const spend = usage.metered === true ? formatUsd(usage.costUsd) : 'unmetered'
|
|
625
|
+
body.push(
|
|
626
|
+
`| ${cell(lane.lane)} | ${cell(lane.model ?? usage.model ?? '')} | ${usage.calls ?? 0} | ` +
|
|
627
|
+
`${usage.tokens ?? 0} | ${spend} |`,
|
|
326
628
|
)
|
|
327
629
|
}
|
|
630
|
+
const agg = manifest.aggregate || {}
|
|
631
|
+
body.push(`| Total | | ${agg.calls ?? 0} | ${agg.tokens ?? 0} | ${formatUsd(agg.costUsd)} |`)
|
|
632
|
+
body.push('')
|
|
633
|
+
const over = manifestLanes(manifest).filter((l) => l.selected === true && l.budget?.exceeded === true)
|
|
634
|
+
if (over.length > 0) {
|
|
635
|
+
body.push(`**Budget exceeded:** ${over.map((l) => cell(l.lane)).join(', ')}`)
|
|
636
|
+
body.push('')
|
|
637
|
+
}
|
|
638
|
+
}
|
|
639
|
+
if (report !== undefined) {
|
|
640
|
+
const t = report.totals ?? {}
|
|
641
|
+
const budgetCap = report.config?.budgetUsd ?? 0
|
|
642
|
+
body.push('| Line item | Value |', '|---|--:|')
|
|
643
|
+
body.push(`| Vision calls | ${t.visionCalls ?? 0} |`)
|
|
644
|
+
const perCall = t.visionCalls > 0 ? formatUsd(t.visionCostUsd / t.visionCalls) : formatUsd(0)
|
|
645
|
+
body.push(`| Per-call cost (avg) | ${perCall} |`)
|
|
646
|
+
for (const model of Object.keys(t.callsByModel ?? {}).sort()) {
|
|
647
|
+
body.push(`| Calls (${cell(model)}) | ${t.callsByModel[model]} |`)
|
|
648
|
+
body.push(`| Spend (${cell(model)}) | ${formatUsd(t.costByModel?.[model] ?? 0)} |`)
|
|
649
|
+
}
|
|
650
|
+
body.push(`| Total vision spend | ${formatUsd(t.visionCostUsd)} |`)
|
|
651
|
+
body.push(`| Sandbox seconds | ${(t.sandboxSeconds ?? 0).toFixed(1)}s |`)
|
|
652
|
+
if (budgetCap > 0) {
|
|
653
|
+
body.push(`| Budget cap | ${formatUsd(budgetCap)} |`)
|
|
654
|
+
body.push(`| Budget exceeded | ${t.budgetExceeded ? 'yes' : 'no'} |`)
|
|
655
|
+
}
|
|
656
|
+
body.push('')
|
|
328
657
|
}
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
658
|
+
// Cache economics: the run report's totals when present, else the manifest's flow lane.
|
|
659
|
+
const cache = report !== undefined
|
|
660
|
+
? { hits: report.totals?.cacheHits, misses: report.totals?.cacheMisses, heals: report.totals?.cacheHeals }
|
|
661
|
+
: manifest?.lanes?.flow?.cache
|
|
662
|
+
if (cache && typeof cache === 'object') {
|
|
663
|
+
body.push(
|
|
664
|
+
`**Fingerprint cache:** ${cache.hits ?? 0} hit(s) · ${cache.misses ?? 0} miss(es) · ${cache.heals ?? 0} heal(s)`,
|
|
333
665
|
)
|
|
666
|
+
body.push('')
|
|
334
667
|
}
|
|
335
|
-
if (
|
|
336
|
-
|
|
668
|
+
if (codeReview && !codeReview.skipped) {
|
|
669
|
+
body.push(
|
|
670
|
+
`**Code review:** ${cell(codeReview.model ?? 'unknown model')} · ${codeReview.tokens ?? 0} tokens · ` +
|
|
671
|
+
`${formatUsd(codeReview.visionCostUsd)}`,
|
|
672
|
+
)
|
|
673
|
+
body.push('')
|
|
337
674
|
}
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
// for the markdown table and bound the section so an oversized report
|
|
353
|
-
// can't push the body past GitHub's 65536-char comment limit.
|
|
354
|
-
const MAX_FINDING_ROWS = 25
|
|
355
|
-
for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
|
|
356
|
-
const ev = f.evidence
|
|
357
|
-
? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
|
|
358
|
-
: '—'
|
|
359
|
-
// U8 — Jev P(true positive); unadjudicated findings render '—'.
|
|
360
|
-
const p = typeof f.p === 'number' ? f.p.toFixed(2) : '—'
|
|
361
|
-
lines.push(
|
|
362
|
-
`| \`${cell(f.file)}\` | ${cell(f.severity)} | ${p} | ${cell(f.category ?? '—')} | ${ev} | ${cell(f.message)} |`,
|
|
675
|
+
return body
|
|
676
|
+
}
|
|
677
|
+
|
|
678
|
+
function diagnosticsFold(manifest, report, codeReview) {
|
|
679
|
+
const items = []
|
|
680
|
+
const binding = manifest?.lanes?.review?.headBinding ?? (codeReview && !codeReview.skipped ? codeReview.headBinding : undefined)
|
|
681
|
+
if (binding) items.push(`Head binding: ${cell(binding.status)}, ${cell(binding.detail)}`)
|
|
682
|
+
if (codeReview && !codeReview.skipped) {
|
|
683
|
+
const sc = codeReview.scope
|
|
684
|
+
if (sc && sc.excludedFiles > 0) {
|
|
685
|
+
const sample = Array.isArray(sc.excludedSample) ? sc.excludedSample.slice(0, 5).map((p) => code(p)).join(', ') : ''
|
|
686
|
+
items.push(
|
|
687
|
+
`Review scope: ${sc.reviewedFiles} of ${sc.totalFiles} changed files reviewed; ` +
|
|
688
|
+
`${sc.excludedFiles} excluded by \`review.exclude\`${sample ? ` (${sample})` : ''}`,
|
|
363
689
|
)
|
|
364
690
|
}
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
691
|
+
const val = codeReview.validation
|
|
692
|
+
if (val && val.dropped > 0) {
|
|
693
|
+
const LABEL = {
|
|
694
|
+
file_not_in_diff: 'file not in the diff',
|
|
695
|
+
file_excluded: 'file excluded from review',
|
|
696
|
+
file_deleted: 'file deleted at head',
|
|
697
|
+
line_beyond_file: 'line past end of file',
|
|
698
|
+
line_outside_diff: 'line outside changed hunks',
|
|
699
|
+
}
|
|
700
|
+
const parts = Object.entries(val.byReason ?? {}).map(([k, n]) => `${n} ${LABEL[k] ?? cell(k)}`)
|
|
701
|
+
items.push(`Findings dropped by validation: ${val.dropped} (${parts.join(', ')})`)
|
|
369
702
|
}
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
703
|
+
if (typeof codeReview.testFileCapped === 'number' && codeReview.testFileCapped > 0) {
|
|
704
|
+
items.push(`Test-file findings capped at nit: ${codeReview.testFileCapped}`)
|
|
705
|
+
}
|
|
706
|
+
const t = codeReview.triage
|
|
707
|
+
if (t) {
|
|
708
|
+
if (t.unadjudicated === true) {
|
|
709
|
+
items.push('Risk triage unavailable: the confidence model did not respond.')
|
|
710
|
+
} else {
|
|
711
|
+
items.push(
|
|
712
|
+
`Risk triage: risk ${cell(t.risk ?? '?')}/5` +
|
|
713
|
+
`${typeof t.needsDeepReview === 'number' ? ` · deep-review ${t.needsDeepReview.toFixed(2)}` : ''}` +
|
|
714
|
+
`${t.topRiskArea !== undefined ? ` · top area ${code(t.topRiskArea)}` : ''}` +
|
|
715
|
+
` (${cell(t.mode)})`,
|
|
716
|
+
)
|
|
717
|
+
}
|
|
718
|
+
}
|
|
719
|
+
// Evidence links that could not conclude (no repo index, CI unreachable):
|
|
720
|
+
// one line per distinct reason instead of one per finding (U6).
|
|
721
|
+
const inconclusive = new Map()
|
|
722
|
+
for (const f of findingsOf(codeReview)) {
|
|
723
|
+
if (f.evidence?.status !== 'inconclusive' || !f.evidence.detail) continue
|
|
724
|
+
inconclusive.set(f.evidence.detail, (inconclusive.get(f.evidence.detail) ?? 0) + 1)
|
|
725
|
+
}
|
|
726
|
+
for (const [detail, n] of inconclusive) {
|
|
727
|
+
items.push(`CI evidence inconclusive for ${plural(n, 'finding')}: ${cell(detail)}`)
|
|
728
|
+
}
|
|
729
|
+
if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
|
|
730
|
+
const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
|
|
731
|
+
items.push(`Probes: ${codeReview.probes.length} run, ${reproduced} reproduced`)
|
|
732
|
+
}
|
|
733
|
+
if (typeof codeReview.probeLaneSkipped === 'string') {
|
|
734
|
+
items.push(`Probe lane skipped: ${cell(codeReview.probeLaneSkipped)}`)
|
|
735
|
+
}
|
|
736
|
+
// Secrets-lane audit: adjudicated/suppressed counts, never literals.
|
|
737
|
+
const scan = codeReview.secretsScan
|
|
738
|
+
if (scan) {
|
|
739
|
+
if (typeof scan.skipped === 'string') {
|
|
740
|
+
items.push(`Secrets scan skipped: ${cell(scan.skipped)}`)
|
|
741
|
+
} else if (Array.isArray(scan.records)) {
|
|
742
|
+
const suppressed = scan.records.filter((r) => r.suppressed).length
|
|
743
|
+
const unadj = scan.records.filter((r) => !r.adjudicated).length
|
|
744
|
+
items.push(
|
|
745
|
+
`Secrets scan: ${scan.records.length} candidate(s)` +
|
|
380
746
|
`${suppressed > 0 ? `, ${suppressed} adjudicated-suppressed` : ''}` +
|
|
381
747
|
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
382
|
-
`${
|
|
748
|
+
`${scan.overflow > 0 ? `, ${scan.overflow} over cap` : ''}`,
|
|
749
|
+
)
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
// U8 adjudication audit: shows what the confidence model removed.
|
|
753
|
+
const fa = codeReview.findingAdjudication
|
|
754
|
+
if (fa && Array.isArray(fa.records)) {
|
|
755
|
+
if (fa.unadjudicated === true) {
|
|
756
|
+
items.push('Finding adjudication unavailable: the confidence model did not respond, nothing suppressed.')
|
|
757
|
+
} else {
|
|
758
|
+
const suppressed = fa.records.filter((r) => r.suppressed).length
|
|
759
|
+
const unadj = fa.records.filter((r) => !r.adjudicated).length
|
|
760
|
+
items.push(
|
|
761
|
+
`Adjudication: ${fa.records.length} finding(s) scored` +
|
|
762
|
+
`${suppressed > 0 ? `, ${suppressed} suppressed (nit/q)` : ''}` +
|
|
763
|
+
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
764
|
+
`${fa.overflow > 0 ? `, ${fa.overflow} over cap` : ''}`,
|
|
383
765
|
)
|
|
384
766
|
}
|
|
385
|
-
lines.push('')
|
|
386
767
|
}
|
|
387
768
|
}
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
769
|
+
for (const [k, v] of Object.entries(report?.trace ?? {})) items.push(`${cell(k)}: ${code(v)}`)
|
|
770
|
+
return items.map((i) => `- ${i}`)
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
// --- bodies ----------------------------------------------------------------------
|
|
774
|
+
|
|
775
|
+
function renderMissingKeyBody(meta = {}) {
|
|
776
|
+
return layout({
|
|
777
|
+
status: statusText('skipped'),
|
|
778
|
+
verdict:
|
|
779
|
+
'**Not run:** `OPENROUTER_API_KEY` is not configured, so no lane ran. This status is neutral, not a failure.',
|
|
780
|
+
rows: MANIFEST_LANE_ORDER.map((lane) => ({ lane, status: 'skipped', result: 'no API key', proof: null, spend: '' })),
|
|
781
|
+
summary:
|
|
782
|
+
'Fix: add the key as a repository secret, then re-run the workflow: `gh secret set OPENROUTER_API_KEY`',
|
|
783
|
+
folds: [],
|
|
784
|
+
meta,
|
|
785
|
+
})
|
|
786
|
+
}
|
|
787
|
+
|
|
788
|
+
function renderNoReportBody(reportDir, runUrl, meta = {}) {
|
|
789
|
+
return layout({
|
|
790
|
+
status: statusText('failed'),
|
|
791
|
+
verdict:
|
|
792
|
+
`**No report:** the run step produced no \`run.json\` under ${code(reportDir)}. ` +
|
|
793
|
+
'The commit status fails closed.',
|
|
794
|
+
rows: [{ lane: 'flow', status: 'failed', result: 'no run.json', proof: 'none', spend: '' }],
|
|
795
|
+
summary: 'Check the action logs before merging.',
|
|
796
|
+
folds: [],
|
|
797
|
+
meta,
|
|
798
|
+
runUrl,
|
|
799
|
+
})
|
|
800
|
+
}
|
|
801
|
+
|
|
802
|
+
/** Lane table rows (string lines) for a parsed run-manifest.json. */
|
|
803
|
+
function renderManifestLanes(manifest, codeReview) {
|
|
804
|
+
return laneTable(manifestLaneRows(manifest, codeReview))
|
|
805
|
+
}
|
|
806
|
+
|
|
807
|
+
function manifestDuration(manifest) {
|
|
808
|
+
const ms = Date.parse(manifest.finishedAt) - Date.parse(manifest.startedAt)
|
|
809
|
+
return Number.isFinite(ms) ? ms : undefined
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
/**
|
|
813
|
+
* Manifest-only body: a verify run whose lanes produced no run.json
|
|
814
|
+
* (review-only, or a flow lane that never reached the browser). The
|
|
815
|
+
* aggregate status is the verdict; lane detail lives in the manifest.
|
|
816
|
+
*/
|
|
817
|
+
function renderManifestBody(manifest, codeReview, runUrl, meta = {}) {
|
|
818
|
+
const aggregate = (manifest && manifest.aggregate) || {}
|
|
819
|
+
const rows = manifestLaneRows(manifest, codeReview)
|
|
820
|
+
return layout({
|
|
821
|
+
status: headline(aggregate.ok === true, aggregate.status, codeReview),
|
|
822
|
+
verdict: verdictLine({
|
|
823
|
+
rows,
|
|
824
|
+
codeReview,
|
|
825
|
+
headSha: manifest.identity?.intendedHeadSha,
|
|
826
|
+
binding: manifest.lanes?.review?.headBinding,
|
|
827
|
+
costUsd: aggregate.costUsd,
|
|
828
|
+
durationMs: manifestDuration(manifest),
|
|
829
|
+
}),
|
|
830
|
+
rows,
|
|
831
|
+
summary: findingsLine(codeReview, NO_REVIEW_ATTACHED),
|
|
832
|
+
folds: [
|
|
833
|
+
findingsEntry(codeReview, undefined),
|
|
834
|
+
['Spend ledger', spendFold(manifest, undefined, codeReview), { key: 'spend' }],
|
|
835
|
+
['Diagnostics', diagnosticsFold(manifest, undefined, codeReview), { key: 'diagnostics' }],
|
|
836
|
+
],
|
|
837
|
+
meta,
|
|
838
|
+
runUrl,
|
|
839
|
+
})
|
|
840
|
+
}
|
|
841
|
+
|
|
842
|
+
function renderBody(report, codeReview, runUrl, ok, inlinePlan, manifest, meta = {}) {
|
|
843
|
+
if (!report) return renderMissingKeyBody(meta)
|
|
844
|
+
// The verify manifest is the lane contract; without one, the review and
|
|
845
|
+
// flow rows are synthesized from their own reports.
|
|
846
|
+
const rows = manifest !== undefined
|
|
847
|
+
? manifestLaneRows(manifest, codeReview)
|
|
848
|
+
: [reviewLaneRow(codeReview), flowLaneRow(report)]
|
|
849
|
+
const reviewSpend = codeReview && !codeReview.skipped ? codeReview.visionCostUsd ?? 0 : 0
|
|
850
|
+
const heals = healsOf(report)
|
|
851
|
+
return layout({
|
|
852
|
+
status: headline(ok, manifest?.aggregate?.status, codeReview),
|
|
853
|
+
verdict: verdictLine({
|
|
854
|
+
rows,
|
|
855
|
+
codeReview,
|
|
856
|
+
headSha: manifest?.identity?.intendedHeadSha ?? codeReview?.headBinding?.intendedSha,
|
|
857
|
+
binding: manifest?.lanes?.review?.headBinding ?? codeReview?.headBinding,
|
|
858
|
+
costUsd: manifest !== undefined ? manifest.aggregate.costUsd : (report.totals?.visionCostUsd ?? 0) + reviewSpend,
|
|
859
|
+
durationMs: manifest !== undefined ? manifestDuration(manifest) : report.durationMs,
|
|
860
|
+
}),
|
|
861
|
+
rows,
|
|
862
|
+
summary: findingsLine(codeReview),
|
|
863
|
+
folds: [
|
|
864
|
+
findingsEntry(codeReview, inlinePlan),
|
|
865
|
+
[`Tests (${report.tests?.length ?? 0})`, testsFold(report), { key: 'tests' }],
|
|
866
|
+
[`Heals (${heals.length}): review before merging`, healsFold(heals), { key: 'heals' }],
|
|
867
|
+
['Exploratory', exploreFold(report), { key: 'explore' }],
|
|
868
|
+
['Spend ledger', spendFold(manifest, report, codeReview), { key: 'spend' }],
|
|
869
|
+
['Diagnostics', diagnosticsFold(manifest, report, codeReview), { key: 'diagnostics' }],
|
|
870
|
+
],
|
|
871
|
+
meta,
|
|
872
|
+
runUrl,
|
|
873
|
+
})
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
// Sticky body for `run: 'false'` consumers: no run.json exists by design, so
|
|
877
|
+
// the body and conclusion reflect code review alone.
|
|
878
|
+
function renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan, manifest, meta = {}) {
|
|
879
|
+
const rows = manifest !== undefined
|
|
880
|
+
? manifestLaneRows(manifest, codeReview)
|
|
881
|
+
: [reviewLaneRow(codeReview), flowLaneRow(undefined)]
|
|
882
|
+
const reviewed = codeReview && !codeReview.skipped
|
|
883
|
+
return layout({
|
|
884
|
+
status: headline(ok, manifest?.aggregate?.status, codeReview),
|
|
885
|
+
verdict: verdictLine({
|
|
886
|
+
rows,
|
|
887
|
+
codeReview,
|
|
888
|
+
headSha: manifest?.identity?.intendedHeadSha ?? (reviewed ? codeReview.headBinding?.intendedSha : undefined),
|
|
889
|
+
binding: manifest?.lanes?.review?.headBinding ?? (reviewed ? codeReview.headBinding : undefined),
|
|
890
|
+
costUsd: manifest !== undefined ? manifest.aggregate.costUsd : reviewed ? codeReview.visionCostUsd : 0,
|
|
891
|
+
durationMs: manifest !== undefined ? manifestDuration(manifest) : undefined,
|
|
892
|
+
}),
|
|
893
|
+
rows,
|
|
894
|
+
summary: findingsLine(codeReview),
|
|
895
|
+
folds: [
|
|
896
|
+
findingsEntry(codeReview, inlinePlan),
|
|
897
|
+
['Spend ledger', spendFold(manifest, undefined, codeReview), { key: 'spend' }],
|
|
898
|
+
['Diagnostics', diagnosticsFold(manifest, undefined, codeReview), { key: 'diagnostics' }],
|
|
899
|
+
],
|
|
900
|
+
meta,
|
|
901
|
+
runUrl,
|
|
902
|
+
})
|
|
903
|
+
}
|
|
904
|
+
|
|
905
|
+
// --- manifest states (R11) -------------------------------------------------------
|
|
906
|
+
// A manifest is ok, missing, unreadable (did not parse or failed validation)
|
|
907
|
+
// or stale (bound to another head or another run). Each degraded state is
|
|
908
|
+
// named in the comment with a fix; a stale one shows both SHAs.
|
|
909
|
+
|
|
910
|
+
/** A git SHA shortened for display, or undefined when the field is not 7-40
|
|
911
|
+
* hex characters: a tampered manifest cannot plant text in the banner. */
|
|
912
|
+
function shortSha(sha) {
|
|
913
|
+
return typeof sha === 'string' && /^[0-9a-f]{7,40}$/i.test(sha) ? sha.slice(0, 7) : undefined
|
|
914
|
+
}
|
|
915
|
+
|
|
916
|
+
/**
|
|
917
|
+
* Classify run-manifest.json. `raw` is the file text, or undefined when the
|
|
918
|
+
* file is absent. The manifest decides the status only when it validates like
|
|
919
|
+
* a real manifest AND binds this run's head AND this run's nonce; residue and
|
|
920
|
+
* plants fall through to whatever serialized evidence survived the same gate.
|
|
921
|
+
*/
|
|
922
|
+
function resolveManifest(raw, { headSha, nonce }) {
|
|
923
|
+
if (raw === undefined) return { state: 'missing' }
|
|
924
|
+
let parsed
|
|
925
|
+
try {
|
|
926
|
+
parsed = JSON.parse(raw)
|
|
927
|
+
} catch {
|
|
928
|
+
return { state: 'unreadable', reason: 'parse' }
|
|
929
|
+
}
|
|
930
|
+
if (!validManifest(parsed)) return { state: 'unreadable', reason: 'invalid' }
|
|
931
|
+
const manifestSha = parsed.identity?.intendedHeadSha
|
|
932
|
+
const otherHead = manifestSha !== headSha
|
|
933
|
+
const otherRun = nonce !== undefined && nonce !== '' && parsed.identity?.runNonce !== nonce
|
|
934
|
+
if (otherHead || otherRun) {
|
|
935
|
+
return { state: 'stale', manifestSha: shortSha(manifestSha), headSha: shortSha(headSha), sameHead: !otherHead }
|
|
936
|
+
}
|
|
937
|
+
return { state: 'ok', manifest: parsed }
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
function shaText(short) {
|
|
941
|
+
return short === undefined ? 'unknown' : code(short)
|
|
942
|
+
}
|
|
943
|
+
|
|
944
|
+
const VERIFY_STEP = '"Run selected Argus lanes"'
|
|
945
|
+
|
|
946
|
+
/** Lead, consequence and fix for a degraded manifest state. */
|
|
947
|
+
function manifestStateCopy(ms) {
|
|
948
|
+
switch (ms.state) {
|
|
949
|
+
case 'missing':
|
|
950
|
+
return {
|
|
951
|
+
lead: '**No manifest:** the verify step wrote no `run-manifest.json`',
|
|
952
|
+
fix: `Fix: open the ${VERIFY_STEP} step log to see where it stopped, then re-run the workflow.`,
|
|
953
|
+
}
|
|
954
|
+
case 'unreadable':
|
|
955
|
+
return {
|
|
956
|
+
lead:
|
|
957
|
+
ms.reason === 'parse'
|
|
958
|
+
? '**Manifest unreadable:** `run-manifest.json` did not parse'
|
|
959
|
+
: '**Manifest unreadable:** `run-manifest.json` failed validation',
|
|
960
|
+
fix: `Fix: re-run the workflow. If it happens again, check the ${VERIFY_STEP} step log.`,
|
|
961
|
+
}
|
|
962
|
+
case 'stale':
|
|
963
|
+
return {
|
|
964
|
+
lead: ms.sameHead
|
|
965
|
+
? `**Manifest stale:** the manifest for head ${shaText(ms.headSha)} came from another workflow run`
|
|
966
|
+
: `**Manifest stale:** manifest ${shaText(ms.manifestSha)} ≠ head ${shaText(ms.headSha)}`,
|
|
967
|
+
fix: 'Fix: re-run the workflow on the current head.',
|
|
968
|
+
}
|
|
969
|
+
default:
|
|
970
|
+
return undefined
|
|
971
|
+
}
|
|
972
|
+
}
|
|
973
|
+
|
|
974
|
+
/** Whether a missing manifest is worth naming: only runs that had a verify
|
|
975
|
+
* step write one. `@argus` mention runs (issue_comment) never do. */
|
|
976
|
+
function manifestExpected(ms, eventName) {
|
|
977
|
+
return ms.state !== 'missing' || eventName !== 'issue_comment'
|
|
978
|
+
}
|
|
979
|
+
|
|
980
|
+
/** Body for a run whose manifest is degraded and whose run.json is absent:
|
|
981
|
+
* nothing trustworthy says how the lanes went, so the status fails closed. */
|
|
982
|
+
function renderManifestStateBody(ms, codeReview, runUrl, meta = {}) {
|
|
983
|
+
const copy = manifestStateCopy(ms)
|
|
984
|
+
return layout({
|
|
985
|
+
status: statusText('failed'),
|
|
986
|
+
verdict: `${copy.lead}${ms.state === 'missing' ? '' : ', so it was ignored'}. The commit status fails closed.`,
|
|
987
|
+
rows: [reviewLaneRow(codeReview)],
|
|
988
|
+
summary: copy.fix,
|
|
989
|
+
folds: [findingsEntry(codeReview, undefined)],
|
|
990
|
+
meta,
|
|
991
|
+
runUrl,
|
|
992
|
+
})
|
|
993
|
+
}
|
|
994
|
+
|
|
995
|
+
/** Ignored-evidence notice: a report whose head/run binding does not match. */
|
|
996
|
+
function staleEvidenceNotice(file) {
|
|
997
|
+
return `A \`${file}\` was found but its head/run binding does not match this run, so it was ignored.`
|
|
998
|
+
}
|
|
999
|
+
|
|
1000
|
+
/**
|
|
1001
|
+
* Pick and render the sticky body for one run. `ev` carries what main() read:
|
|
1002
|
+
* hasKey, runDisabled, eventName, report, codeReview, manifestState, inlinePlan,
|
|
1003
|
+
* ok, reportDir, staleEvidence, runUrl, reportHtml.
|
|
1004
|
+
*/
|
|
1005
|
+
function renderSticky(ev, baseMeta = {}) {
|
|
1006
|
+
const runUrl = ev.runUrl
|
|
1007
|
+
if (!ev.hasKey) return renderMissingKeyBody({ ...baseMeta, runUrl })
|
|
1008
|
+
const ms = ev.manifestState ?? { state: 'missing' }
|
|
1009
|
+
const manifest = ms.state === 'ok' ? ms.manifest : undefined
|
|
1010
|
+
// report.html is written beside the manifest by the same verify run. Only
|
|
1011
|
+
// a fresh, valid manifest vouches for it; otherwise it may be residue.
|
|
1012
|
+
const meta =
|
|
1013
|
+
manifest !== undefined && typeof ev.reportHtml === 'string' && ev.reportHtml !== ''
|
|
1014
|
+
? { ...baseMeta, reportHtml: ev.reportHtml }
|
|
1015
|
+
: baseMeta
|
|
1016
|
+
const named = ms.state !== 'ok' && manifestExpected(ms, ev.eventName)
|
|
1017
|
+
const notices = []
|
|
1018
|
+
if (named) {
|
|
1019
|
+
const copy = manifestStateCopy(ms)
|
|
1020
|
+
notices.push(`${copy.lead}, so lanes come from the individual reports. ${copy.fix}`)
|
|
1021
|
+
}
|
|
1022
|
+
for (const file of ev.staleEvidence ?? []) notices.push(staleEvidenceNotice(file))
|
|
1023
|
+
const withNotices = { ...meta, notices }
|
|
1024
|
+
if (ev.runDisabled) {
|
|
1025
|
+
return renderReviewOnlyBody(ev.codeReview, runUrl, ev.ok, ev.inlinePlan, manifest, withNotices)
|
|
1026
|
+
}
|
|
1027
|
+
if (ev.report === undefined) {
|
|
1028
|
+
if (manifest !== undefined) return renderManifestBody(manifest, ev.codeReview, runUrl, withNotices)
|
|
1029
|
+
const rest = { ...meta, notices: notices.slice(named ? 1 : 0) }
|
|
1030
|
+
if (named) return renderManifestStateBody(ms, ev.codeReview, runUrl, rest)
|
|
1031
|
+
return renderNoReportBody(ev.reportDir, runUrl, rest)
|
|
1032
|
+
}
|
|
1033
|
+
return renderBody(ev.report, ev.codeReview, runUrl, ev.ok, ev.inlinePlan, manifest, withNotices)
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
// --- GitHub write failures (R9) ---------------------------------------------------
|
|
1037
|
+
// A failed write never fails silently: the reason goes to the job summary
|
|
1038
|
+
// and a warning, and the commit status is still attempted.
|
|
1039
|
+
|
|
1040
|
+
function headerOf(e, name) {
|
|
1041
|
+
const h = e?.response?.headers
|
|
1042
|
+
return h && typeof h === 'object' ? h[name] : undefined
|
|
1043
|
+
}
|
|
1044
|
+
|
|
1045
|
+
/** Plain-words reason and fix for a failed GitHub API call. HTTP codes stay
|
|
1046
|
+
* in the Diagnostics part (R5). */
|
|
1047
|
+
function describeApiError(e, scope) {
|
|
1048
|
+
const status = e?.status
|
|
1049
|
+
const remaining = headerOf(e, 'x-ratelimit-remaining')
|
|
1050
|
+
const message = String(e?.message ?? 'unknown error')
|
|
1051
|
+
if (status === 429 || String(remaining) === '0' || /rate limit/i.test(message)) {
|
|
1052
|
+
const reset = Number(headerOf(e, 'x-ratelimit-reset'))
|
|
1053
|
+
const at = Number.isFinite(reset) && reset > 0
|
|
1054
|
+
? new Date(reset * 1000).toISOString().replace(/\.\d{3}Z$/, 'Z')
|
|
1055
|
+
: undefined
|
|
1056
|
+
return {
|
|
1057
|
+
reason: 'GitHub API rate limit reached',
|
|
1058
|
+
fix: at !== undefined ? `The limit resets at ${at}; re-run the workflow after that.` : 'Re-run the workflow later.',
|
|
398
1059
|
}
|
|
399
|
-
lines.push(`*+${inlinePlan.dropped} inline comment(s) not posted — ${reasons.join(', ')}.*`)
|
|
400
|
-
lines.push('')
|
|
401
1060
|
}
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
const fa = codeReview.findingAdjudication
|
|
407
|
-
if (fa.unadjudicated === true) {
|
|
408
|
-
lines.push('*🧮 adjudication: unadjudicated — Jev unavailable, nothing suppressed.*')
|
|
409
|
-
} else {
|
|
410
|
-
const suppressed = fa.records.filter((r) => r.suppressed).length
|
|
411
|
-
const unadj = fa.records.filter((r) => !r.adjudicated).length
|
|
412
|
-
lines.push(
|
|
413
|
-
`*🧮 adjudication: ${fa.records.length} finding(s) scored` +
|
|
414
|
-
`${suppressed > 0 ? `, ${suppressed} suppressed (nit/q)` : ''}` +
|
|
415
|
-
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
416
|
-
`${fa.overflow > 0 ? `, +${fa.overflow} over cap` : ''}.*`,
|
|
417
|
-
)
|
|
1061
|
+
if (status === 401 || status === 403) {
|
|
1062
|
+
return {
|
|
1063
|
+
reason: 'permission denied',
|
|
1064
|
+
fix: `Give the workflow \`${scope}\` permission. Pull requests from forks get a read-only token.`,
|
|
418
1065
|
}
|
|
419
|
-
lines.push('')
|
|
420
1066
|
}
|
|
421
|
-
|
|
422
|
-
|
|
1067
|
+
if (status === 404) {
|
|
1068
|
+
return { reason: 'not found', fix: 'The token cannot see this pull request or comment; check the workflow token.' }
|
|
1069
|
+
}
|
|
1070
|
+
if (status === 422) {
|
|
1071
|
+
return { reason: 'GitHub rejected the request as invalid', fix: 'Re-run the workflow; if it repeats, open an issue with the job log.' }
|
|
1072
|
+
}
|
|
1073
|
+
return { reason: 'GitHub API error', fix: 'Re-run the workflow; if it repeats, check the GitHub status page.' }
|
|
423
1074
|
}
|
|
424
1075
|
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
const
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
lines.push(
|
|
440
|
-
`**Summary:** code-review only (run lane disabled) — review skipped: ${cell(codeReview.summary)}`,
|
|
441
|
-
)
|
|
442
|
-
} else {
|
|
443
|
-
lines.push(
|
|
444
|
-
`**Summary:** code review only (run lane disabled) · verdict **${codeReview.verdict}** · ` +
|
|
445
|
-
`${codeReview.findings.length} finding(s) · ${codeReview.model} · ` +
|
|
446
|
-
`${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
|
|
447
|
-
)
|
|
448
|
-
}
|
|
449
|
-
lines.push('')
|
|
450
|
-
lines.push('<details>')
|
|
451
|
-
lines.push('<summary>🚥 Pre-merge checks</summary>')
|
|
452
|
-
lines.push('')
|
|
453
|
-
lines.push('| Check | Status | Explanation |')
|
|
454
|
-
lines.push('| --- | --- | --- |')
|
|
455
|
-
lines.push('| OpenRouter key | ✅ Passed | `OPENROUTER_API_KEY` configured |')
|
|
456
|
-
if (codeReview && !codeReview.skipped) {
|
|
457
|
-
const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
|
|
458
|
-
lines.push(
|
|
459
|
-
`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
|
|
460
|
-
)
|
|
461
|
-
} else {
|
|
462
|
-
lines.push(
|
|
463
|
-
`| Code review | ${codeReview ? '⚪ Skipped' : '❌ Failed'} | ${cell(codeReview?.summary ?? 'no report')} |`,
|
|
464
|
-
)
|
|
1076
|
+
/** Report one failed write to the job summary and as a warning. Never throws. */
|
|
1077
|
+
async function reportWriteFailure(what, e, scope) {
|
|
1078
|
+
const { reason, fix } = describeApiError(e, scope)
|
|
1079
|
+
const diagnostic = maskSecrets(`GitHub API ${e?.status ?? 'error'}: ${String(e?.message ?? 'unknown error')}`)
|
|
1080
|
+
core.warning(`Argus could not ${what}: ${reason}. ${fix}`)
|
|
1081
|
+
try {
|
|
1082
|
+
await core.summary
|
|
1083
|
+
.addRaw(
|
|
1084
|
+
`### Argus could not ${what}\n\n${reason[0].toUpperCase()}${reason.slice(1)}. ${fix}\n\n` +
|
|
1085
|
+
`<details><summary>Diagnostics</summary>\n\n${diagnostic}\n\n</details>\n\n`,
|
|
1086
|
+
)
|
|
1087
|
+
.write()
|
|
1088
|
+
} catch (summaryErr) {
|
|
1089
|
+
core.warning(`job summary write failed: ${summaryErr?.message ?? 'unknown error'}`)
|
|
465
1090
|
}
|
|
466
|
-
lines.push('')
|
|
467
|
-
lines.push('</details>')
|
|
468
|
-
lines.push('')
|
|
469
|
-
pushCodeReviewDetails(lines, codeReview, inlinePlan)
|
|
470
|
-
if (runUrl) lines.push(`[View run](${runUrl})`)
|
|
471
|
-
lines.push('')
|
|
472
|
-
lines.push('<details>')
|
|
473
|
-
lines.push('<summary>✨ Actions</summary>')
|
|
474
|
-
lines.push('')
|
|
475
|
-
lines.push('- [ ] Re-run argus-reviewer')
|
|
476
|
-
lines.push('')
|
|
477
|
-
lines.push('</details>')
|
|
478
|
-
lines.push('')
|
|
479
|
-
lines.push('---')
|
|
480
|
-
lines.push('')
|
|
481
|
-
lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
|
|
482
|
-
lines.push('')
|
|
483
|
-
return lines.join('\n')
|
|
484
1091
|
}
|
|
485
1092
|
|
|
486
1093
|
// ---------------------------------------------------------------------------
|
|
@@ -492,8 +1099,8 @@ function renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan) {
|
|
|
492
1099
|
// (KTD5), and POSTs one batched review with the serialized event plus a
|
|
493
1100
|
// bounded retry ladder (R4/KTD4).
|
|
494
1101
|
|
|
495
|
-
/** djb2 → 8 hex chars. Must match shortHash() in src/
|
|
496
|
-
* dedupKey suffix is this hash over the raw suggestion text. */
|
|
1102
|
+
/** djb2 → 8 hex chars. Must match shortHash() in src/review/inline.ts — the
|
|
1103
|
+
* CLI's dedupKey suffix is this hash over the raw suggestion text. */
|
|
497
1104
|
function shortHash(s) {
|
|
498
1105
|
let h = 5381
|
|
499
1106
|
for (let i = 0; i < s.length; i++) h = ((h << 5) + h + s.charCodeAt(i)) | 0
|
|
@@ -509,12 +1116,64 @@ function extractSuggestion(body) {
|
|
|
509
1116
|
return m === null ? '' : m[2]
|
|
510
1117
|
}
|
|
511
1118
|
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
1119
|
+
// --- inline comment identity (plan KTD4) --------------------------------------
|
|
1120
|
+
// This file's copy of src/review/inline.ts (KTD2); the action-contract parity
|
|
1121
|
+
// test pins both equal. A legacy body (`**argus-reviewer <sev>:** <msg>`) and
|
|
1122
|
+
// an Ocellus body (sentinel, `<glyph> **<sev>** · <proof>`, message) key to
|
|
1123
|
+
// the same value for the same finding, so an upgrade never re-posts.
|
|
1124
|
+
const INLINE_SENTINEL = '<!-- argus-reviewer:inline -->'
|
|
1125
|
+
const LEGACY_PREFIX = '**argus-reviewer'
|
|
1126
|
+
const LEGACY_LINE = /^\*\*argus-reviewer ([^:*]+):\*\* ?(.*)$/
|
|
1127
|
+
const SEVERITY_LINE = /^(?:\S+ )?\*\*([^*]+)\*\* · /
|
|
1128
|
+
const CATEGORY_SUFFIX = /\s*`(?:correctness|security|performance|usability|convention|other)`$/
|
|
1129
|
+
const MESSAGE_PREFIX = /^(?:L\d+(?:-\d+)?:|\p{Extended_Pictographic}\u{FE0F}?|(?:bug|risk|nit|q|question):)\s*/iu
|
|
1130
|
+
const LABEL_TO_SEVERITY = new Map(Object.entries(SEVERITY_LABEL).map(([s, label]) => [label, s]))
|
|
1131
|
+
|
|
1132
|
+
/** Strip the model's `L<n>: <emoji> <sev>:` prefix so the sentence leads. */
|
|
1133
|
+
function normalizeFindingMessage(message) {
|
|
1134
|
+
let out = message.trim()
|
|
1135
|
+
for (let i = 0; i < 4; i++) {
|
|
1136
|
+
const next = out.replace(MESSAGE_PREFIX, '')
|
|
1137
|
+
if (next === out) break
|
|
1138
|
+
out = next
|
|
1139
|
+
}
|
|
1140
|
+
return out
|
|
1141
|
+
}
|
|
1142
|
+
|
|
1143
|
+
function keyMessage(message) {
|
|
1144
|
+
return normalizeFindingMessage(message.replace(CATEGORY_SUFFIX, '')).replace(/\s+/g, ' ').trim()
|
|
1145
|
+
}
|
|
1146
|
+
|
|
1147
|
+
/** Severity and message of an Argus inline comment in either format. */
|
|
1148
|
+
function parseInlineBody(body) {
|
|
1149
|
+
const lines = body.split(/\r?\n/)
|
|
1150
|
+
if (body.startsWith(LEGACY_PREFIX)) {
|
|
1151
|
+
const m = LEGACY_LINE.exec(lines[0] ?? '')
|
|
1152
|
+
if (m === null) return undefined
|
|
1153
|
+
return { severity: m[1].trim(), message: keyMessage(m[2]) }
|
|
1154
|
+
}
|
|
1155
|
+
if (lines[0] === INLINE_SENTINEL) {
|
|
1156
|
+
const m = SEVERITY_LINE.exec(lines[1] ?? '')
|
|
1157
|
+
if (m === null) return undefined
|
|
1158
|
+
const word = m[1].trim()
|
|
1159
|
+
return { severity: LABEL_TO_SEVERITY.get(word) ?? word, message: keyMessage(lines[2] ?? '') }
|
|
1160
|
+
}
|
|
1161
|
+
return undefined
|
|
1162
|
+
}
|
|
1163
|
+
|
|
1164
|
+
/** Argus's own inline comment: the legacy prefix or the sentinel, at the very start. */
|
|
1165
|
+
function isArgusInlineBody(body) {
|
|
1166
|
+
return body.startsWith(LEGACY_PREFIX) || body.startsWith(INLINE_SENTINEL)
|
|
1167
|
+
}
|
|
1168
|
+
|
|
1169
|
+
/** KTD4 key `path:line:severity:normalizedMessage:hash8(suggestion)`, rebuilt
|
|
1170
|
+
* from a posted body; identical to the key the CLI serialized. */
|
|
515
1171
|
function postedDedupKey(c) {
|
|
516
1172
|
const body = c.body ?? ''
|
|
517
|
-
|
|
1173
|
+
const parsed = parseInlineBody(body)
|
|
1174
|
+
const hash = shortHash(extractSuggestion(body))
|
|
1175
|
+
if (parsed === undefined) return `${c.path}:${c.line}:${body.split('\n')[0]}:${hash}`
|
|
1176
|
+
return `${c.path}:${c.line}:${parsed.severity}:${parsed.message}:${hash}`
|
|
518
1177
|
}
|
|
519
1178
|
|
|
520
1179
|
/** Fetch every page of a list endpoint (100/page, octokit shape). */
|
|
@@ -576,12 +1235,17 @@ async function planInlineComments(pr, codeReview) {
|
|
|
576
1235
|
|
|
577
1236
|
// R9/KTD6 freshness — a planted or stale report must never produce
|
|
578
1237
|
// committable suggestions or a blocking review. The sticky still posts;
|
|
579
|
-
// it renders status text, not code.
|
|
580
|
-
|
|
1238
|
+
// it renders status text, not code. Head sha is forgeable (public), so
|
|
1239
|
+
// when a run id exists the report must also carry it.
|
|
1240
|
+
const expectedNonce = (process.env.GITHUB_RUN_ID ?? '').trim()
|
|
1241
|
+
if (
|
|
1242
|
+
codeReview.headBinding?.intendedSha !== pr.head.sha ||
|
|
1243
|
+
(expectedNonce !== '' && codeReview.runNonce !== expectedNonce)
|
|
1244
|
+
) {
|
|
581
1245
|
core.warning(
|
|
582
1246
|
`code-review.json head binding ` +
|
|
583
1247
|
`(${codeReview.headBinding?.intendedSha ?? 'missing'}) does not match ` +
|
|
584
|
-
`PR head ${pr.head.sha}
|
|
1248
|
+
`PR head ${pr.head.sha}; skipping inline review`,
|
|
585
1249
|
)
|
|
586
1250
|
return undefined
|
|
587
1251
|
}
|
|
@@ -599,17 +1263,13 @@ async function planInlineComments(pr, codeReview) {
|
|
|
599
1263
|
try {
|
|
600
1264
|
// R10 dedup — paginate fully and scope to the current head so comments on
|
|
601
1265
|
// older commits can't suppress still-valid findings. Keys are
|
|
602
|
-
// reconstructed from the posted body
|
|
603
|
-
// suggestion), so a re-run with
|
|
604
|
-
// instead of colliding.
|
|
1266
|
+
// reconstructed from the posted body in either format (KTD4: severity,
|
|
1267
|
+
// normalized message, hash of the embedded suggestion), so a re-run with
|
|
1268
|
+
// a corrected suggestion posts the fix instead of colliding.
|
|
605
1269
|
const posted = new Set()
|
|
606
1270
|
const existing = await listAll((p) => github.rest.pulls.listReviewComments(p), prRef)
|
|
607
1271
|
for (const c of existing) {
|
|
608
|
-
if (
|
|
609
|
-
c.commit_id === pr.head.sha &&
|
|
610
|
-
typeof c.body === 'string' &&
|
|
611
|
-
c.body.startsWith('**argus-reviewer')
|
|
612
|
-
) {
|
|
1272
|
+
if (c.commit_id === pr.head.sha && typeof c.body === 'string' && isArgusInlineBody(c.body)) {
|
|
613
1273
|
posted.add(postedDedupKey(c))
|
|
614
1274
|
}
|
|
615
1275
|
}
|
|
@@ -638,24 +1298,30 @@ async function planInlineComments(pr, codeReview) {
|
|
|
638
1298
|
highConfidenceBlockers: codeReview.highConfidenceBlockers ?? 0,
|
|
639
1299
|
}
|
|
640
1300
|
} catch (e) {
|
|
641
|
-
core.warning(`review planning failed: ${e.message}
|
|
1301
|
+
core.warning(`review planning failed: ${e.message}; the sticky comment still posts`)
|
|
642
1302
|
return undefined
|
|
643
1303
|
}
|
|
644
1304
|
}
|
|
645
1305
|
|
|
646
|
-
/** Review body: verdict
|
|
647
|
-
* p-gated are never lumped). Always present
|
|
648
|
-
* body. Carries the sentinel so KTD5 dismissal can
|
|
1306
|
+
/** Review body: the verdict word with its glyph, then honest blocker counts
|
|
1307
|
+
* (R6: reproduced and p-gated are never lumped). Always present, since
|
|
1308
|
+
* REQUEST_CHANGES requires a body. Carries the sentinel so KTD5 dismissal can
|
|
1309
|
+
* self-identify. */
|
|
649
1310
|
function reviewBody(plan, note) {
|
|
650
|
-
const
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
}
|
|
654
|
-
if (plan.
|
|
655
|
-
|
|
1311
|
+
const verdict = Object.hasOwn(VERDICT_LABEL, plan.verdict ?? '')
|
|
1312
|
+
? `${STATUS_GLYPH[VERDICT_STATUS[plan.verdict]]} ${VERDICT_LABEL[plan.verdict]}`
|
|
1313
|
+
: `${STATUS_GLYPH.inconclusive} verdict unknown`
|
|
1314
|
+
const parts = [`**Argus: ${verdict}**`]
|
|
1315
|
+
if (plan.provenBlockers > 0) parts.push(plural(plan.provenBlockers, 'reproduced blocker'))
|
|
1316
|
+
if (plan.highConfidenceBlockers > 0) parts.push(plural(plan.highConfidenceBlockers, 'high-confidence blocker'))
|
|
1317
|
+
let body = `${SENTINEL}\n${parts.join(' · ')}`
|
|
1318
|
+
if (note !== undefined) {
|
|
1319
|
+
body += `\n\n*${note.text}*`
|
|
1320
|
+
// Raw API status/message stays available but out of the reading path.
|
|
1321
|
+
if (note.diagnostic) {
|
|
1322
|
+
body += `\n\n<details><summary>Diagnostics</summary>\n\n${note.diagnostic}\n\n</details>`
|
|
1323
|
+
}
|
|
656
1324
|
}
|
|
657
|
-
let body = `${SENTINEL}\n**argus-reviewer** — ${parts.join(' · ')}`
|
|
658
|
-
if (note !== undefined) body += `\n\n*${note}*`
|
|
659
1325
|
return body
|
|
660
1326
|
}
|
|
661
1327
|
|
|
@@ -733,7 +1399,12 @@ async function postInlineComments(pr, plan) {
|
|
|
733
1399
|
lastErr = e
|
|
734
1400
|
if (attempt === 0 && event === 'REQUEST_CHANGES' && (e.status === 403 || e.status === 422)) {
|
|
735
1401
|
event = 'COMMENT'
|
|
736
|
-
note =
|
|
1402
|
+
note = {
|
|
1403
|
+
text:
|
|
1404
|
+
'Posted as a comment instead of requesting changes: GitHub did not allow a ' +
|
|
1405
|
+
'change request here (for example on your own PR, or without write permission).',
|
|
1406
|
+
diagnostic: `GitHub API ${e.status}: ${e.message}`,
|
|
1407
|
+
}
|
|
737
1408
|
continue
|
|
738
1409
|
}
|
|
739
1410
|
if (e.status === 422 && comments.length > 0) {
|
|
@@ -747,7 +1418,70 @@ async function postInlineComments(pr, plan) {
|
|
|
747
1418
|
break
|
|
748
1419
|
}
|
|
749
1420
|
}
|
|
750
|
-
core.warning(`review post failed: ${lastErr?.message ?? 'unknown error'}
|
|
1421
|
+
core.warning(`review post failed: ${lastErr?.message ?? 'unknown error'}; the sticky comment still posts`)
|
|
1422
|
+
}
|
|
1423
|
+
|
|
1424
|
+
/** Create or update the one sticky comment (R9). The lookup reads every page,
|
|
1425
|
+
* so a PR with more than 100 comments still has one sticky. A failed lookup
|
|
1426
|
+
* skips the post rather than risk a duplicate sticky. */
|
|
1427
|
+
async function postSticky(owner, repo, pr, body) {
|
|
1428
|
+
let existing
|
|
1429
|
+
try {
|
|
1430
|
+
const comments = await listAll((p) => github.rest.issues.listComments(p), {
|
|
1431
|
+
owner,
|
|
1432
|
+
repo,
|
|
1433
|
+
issue_number: pr.number,
|
|
1434
|
+
})
|
|
1435
|
+
existing = comments.find((c) => c.body && c.body.includes(SENTINEL))
|
|
1436
|
+
} catch (e) {
|
|
1437
|
+
await reportWriteFailure('find the existing PR comment, so it did not post one', e, 'pull-requests: read')
|
|
1438
|
+
return
|
|
1439
|
+
}
|
|
1440
|
+
try {
|
|
1441
|
+
if (existing) {
|
|
1442
|
+
await github.rest.issues.updateComment({ owner, repo, comment_id: existing.id, body })
|
|
1443
|
+
} else {
|
|
1444
|
+
await github.rest.issues.createComment({ owner, repo, issue_number: pr.number, body })
|
|
1445
|
+
}
|
|
1446
|
+
} catch (e) {
|
|
1447
|
+
await reportWriteFailure(existing ? 'update the PR comment' : 'post the PR comment', e, 'pull-requests: write')
|
|
1448
|
+
}
|
|
1449
|
+
}
|
|
1450
|
+
|
|
1451
|
+
// --- commit status (R8) ----------------------------------------------------------
|
|
1452
|
+
|
|
1453
|
+
/** GitHub rejects a commit status description over 140 characters. */
|
|
1454
|
+
const STATUS_DESCRIPTION_MAX = 140
|
|
1455
|
+
|
|
1456
|
+
/** Join segments with ` · `, dropping whole trailing segments until the text
|
|
1457
|
+
* fits; a lone first segment that still overflows is cut on a code point
|
|
1458
|
+
* boundary and ends with an ellipsis, so no glyph is ever split. */
|
|
1459
|
+
function fitStatusDescription(parts) {
|
|
1460
|
+
const kept = [...parts]
|
|
1461
|
+
let text = kept.join(' · ')
|
|
1462
|
+
while (text.length > STATUS_DESCRIPTION_MAX && kept.length > 1) {
|
|
1463
|
+
kept.pop()
|
|
1464
|
+
text = kept.join(' · ')
|
|
1465
|
+
}
|
|
1466
|
+
if (text.length <= STATUS_DESCRIPTION_MAX) return text
|
|
1467
|
+
const points = [...text]
|
|
1468
|
+
while (points.length > 0 && points.join('').length > STATUS_DESCRIPTION_MAX - 1) points.pop()
|
|
1469
|
+
return `${points.join('').trimEnd()}…`
|
|
1470
|
+
}
|
|
1471
|
+
|
|
1472
|
+
/** The comment verdict line in status form: `<glyph> <verdict> · <n> findings · $<total>`.
|
|
1473
|
+
* Cost and verdict come from the same sources the sticky header uses. */
|
|
1474
|
+
function statusDescription({ conclusion, hasKey, ok, report, codeReview, manifest }) {
|
|
1475
|
+
if (conclusion === 'neutral') {
|
|
1476
|
+
return fitStatusDescription([statusText('skipped'), hasKey ? 'no lanes ran' : 'no OPENROUTER_API_KEY'])
|
|
1477
|
+
}
|
|
1478
|
+
const parts = [headline(ok, manifest?.aggregate?.status, codeReview)]
|
|
1479
|
+
const reviewed = codeReview && !codeReview.skipped
|
|
1480
|
+
if (reviewed) parts.push(plural(findingsOf(codeReview).length, 'finding'))
|
|
1481
|
+
const reviewSpend = reviewed ? codeReview.visionCostUsd ?? 0 : 0
|
|
1482
|
+
const cost = manifest !== undefined ? manifest.aggregate.costUsd : (report?.totals?.visionCostUsd ?? 0) + reviewSpend
|
|
1483
|
+
parts.push(formatUsd(cost))
|
|
1484
|
+
return fitStatusDescription(parts)
|
|
751
1485
|
}
|
|
752
1486
|
|
|
753
1487
|
async function main() {
|
|
@@ -765,20 +1499,52 @@ async function main() {
|
|
|
765
1499
|
|
|
766
1500
|
let report
|
|
767
1501
|
let codeReview
|
|
1502
|
+
let manifest
|
|
1503
|
+
let manifestState = { state: 'missing' }
|
|
1504
|
+
// Every evidence file is run-scoped. GITHUB_RUN_ID is not knowable when a
|
|
1505
|
+
// commit or a planted file is authored (freshness, not secrecy — it is
|
|
1506
|
+
// public once the run exists), so a file that cannot present this run's
|
|
1507
|
+
// id is residue or plant and is ignored. No env → local/dogfood path,
|
|
1508
|
+
// where the gate is off by design.
|
|
1509
|
+
const expectedNonce = (process.env.GITHUB_RUN_ID ?? '').trim()
|
|
1510
|
+
const staleEvidence = []
|
|
768
1511
|
if (hasKey) {
|
|
769
1512
|
try {
|
|
770
1513
|
const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
|
|
771
|
-
|
|
1514
|
+
const parsed = JSON.parse(raw)
|
|
1515
|
+
if (expectedNonce !== '' && parsed?.runNonce !== expectedNonce) {
|
|
1516
|
+
staleEvidence.push('run.json')
|
|
1517
|
+
} else {
|
|
1518
|
+
report = parsed
|
|
1519
|
+
}
|
|
772
1520
|
} catch {
|
|
773
1521
|
report = undefined
|
|
774
1522
|
}
|
|
775
1523
|
try {
|
|
776
1524
|
const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
|
|
777
|
-
|
|
1525
|
+
const parsed = JSON.parse(raw)
|
|
1526
|
+
if (expectedNonce !== '' && parsed?.runNonce !== expectedNonce) {
|
|
1527
|
+
staleEvidence.push('code-review.json')
|
|
1528
|
+
} else {
|
|
1529
|
+
codeReview = parsed
|
|
1530
|
+
}
|
|
778
1531
|
} catch {
|
|
779
1532
|
codeReview = undefined
|
|
780
1533
|
}
|
|
1534
|
+
let raw
|
|
1535
|
+
try {
|
|
1536
|
+
raw = fs.readFileSync(path.join(reportDir, 'run-manifest.json'), 'utf8')
|
|
1537
|
+
} catch (e) {
|
|
1538
|
+
// Absent is "missing"; any other read error is "unreadable".
|
|
1539
|
+
raw = e && e.code === 'ENOENT' ? undefined : ''
|
|
1540
|
+
}
|
|
1541
|
+
manifestState = resolveManifest(raw, { headSha: pr ? pr.head.sha : context.sha, nonce: expectedNonce })
|
|
1542
|
+
manifest = manifestState.state === 'ok' ? manifestState.manifest : undefined
|
|
781
1543
|
}
|
|
1544
|
+
const reportHtmlPath = path.join(reportDir, 'report.html')
|
|
1545
|
+
const reportHtml = fs.existsSync(reportHtmlPath)
|
|
1546
|
+
? path.relative(process.env.GITHUB_WORKSPACE || process.cwd(), reportHtmlPath).split(path.sep).join('/')
|
|
1547
|
+
: undefined
|
|
782
1548
|
|
|
783
1549
|
// Missing code-review.json after a continue-on-error step means the review
|
|
784
1550
|
// crashed, not that it skipped — an intentional skip writes ok+skipped.
|
|
@@ -787,8 +1553,17 @@ async function main() {
|
|
|
787
1553
|
// run: 'false' consumers have no run.json by design — the conclusion then
|
|
788
1554
|
// reflects the code-review verdict alone.
|
|
789
1555
|
const runDisabled = process.env.ARGUS_RUN_DISABLED === '1'
|
|
790
|
-
|
|
791
|
-
|
|
1556
|
+
// A verify manifest is the authoritative lane verdict when it exists —
|
|
1557
|
+
// its aggregate already fails closed on missing lane reports. An
|
|
1558
|
+
// all-skipped aggregate (e.g. a push event where the review lane has no
|
|
1559
|
+
// PR to inspect) is a legitimate non-verdict — neutral, not failure,
|
|
1560
|
+
// matching the pre-manifest codeReviewOk contract.
|
|
1561
|
+
const aggregateSkipped =
|
|
1562
|
+
manifest !== undefined && manifest.aggregate.status === 'skipped'
|
|
1563
|
+
const ok = manifest !== undefined
|
|
1564
|
+
? manifest.aggregate.ok === true
|
|
1565
|
+
: (runDisabled || report?.ok === true) && codeReviewOk
|
|
1566
|
+
const conclusion = !hasKey || aggregateSkipped ? 'neutral' : ok ? 'success' : 'failure'
|
|
792
1567
|
// Freshness + dedup + diff validation for the serialized review surface,
|
|
793
1568
|
// computed before the sticky body renders so the "+N not posted" note is
|
|
794
1569
|
// truthful. The review posts BEFORE the sticky so retry-ladder drops land
|
|
@@ -796,56 +1571,41 @@ async function main() {
|
|
|
796
1571
|
// postInlineComments a no-op with no API calls.
|
|
797
1572
|
const inlinePlan = hasKey ? await planInlineComments(pr, codeReview) : undefined
|
|
798
1573
|
if (pr) await postInlineComments(pr, inlinePlan)
|
|
799
|
-
const body =
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
if (existing) {
|
|
816
|
-
await github.rest.issues.updateComment({
|
|
817
|
-
owner,
|
|
818
|
-
repo,
|
|
819
|
-
comment_id: existing.id,
|
|
820
|
-
body,
|
|
821
|
-
})
|
|
822
|
-
} else {
|
|
823
|
-
await github.rest.issues.createComment({
|
|
824
|
-
owner,
|
|
825
|
-
repo,
|
|
826
|
-
issue_number: pr.number,
|
|
827
|
-
body,
|
|
828
|
-
})
|
|
829
|
-
}
|
|
830
|
-
}
|
|
1574
|
+
const body = renderSticky({
|
|
1575
|
+
hasKey,
|
|
1576
|
+
runDisabled,
|
|
1577
|
+
eventName: context.eventName,
|
|
1578
|
+
report,
|
|
1579
|
+
codeReview,
|
|
1580
|
+
manifestState,
|
|
1581
|
+
inlinePlan,
|
|
1582
|
+
ok,
|
|
1583
|
+
reportDir,
|
|
1584
|
+
staleEvidence,
|
|
1585
|
+
runUrl,
|
|
1586
|
+
reportHtml,
|
|
1587
|
+
})
|
|
1588
|
+
|
|
1589
|
+
if (pr) await postSticky(owner, repo, pr, body)
|
|
831
1590
|
|
|
832
1591
|
const sha = pr ? pr.head.sha : context.sha
|
|
833
1592
|
// Commit statuses have no 'neutral'; a 'pending' skip would wedge a
|
|
834
1593
|
// required check forever, so skip maps to success with a clear label.
|
|
835
1594
|
const state = conclusion === 'failure' ? 'failure' : 'success'
|
|
836
|
-
const description =
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
1595
|
+
const description = statusDescription({ conclusion, hasKey, ok, report, codeReview, manifest })
|
|
1596
|
+
try {
|
|
1597
|
+
await github.rest.repos.createCommitStatus({
|
|
1598
|
+
owner,
|
|
1599
|
+
repo,
|
|
1600
|
+
sha,
|
|
1601
|
+
state,
|
|
1602
|
+
description,
|
|
1603
|
+
context: 'argus-reviewer',
|
|
1604
|
+
target_url: runUrl,
|
|
1605
|
+
})
|
|
1606
|
+
} catch (e) {
|
|
1607
|
+
await reportWriteFailure('set the commit status', e, 'statuses: write')
|
|
1608
|
+
}
|
|
849
1609
|
|
|
850
1610
|
core.setOutput('conclusion', conclusion)
|
|
851
1611
|
}
|
|
@@ -859,12 +1619,33 @@ async function run(runtime) {
|
|
|
859
1619
|
|
|
860
1620
|
module.exports = {
|
|
861
1621
|
run,
|
|
1622
|
+
STATUS_GLYPH,
|
|
1623
|
+
PROOF_LEVELS,
|
|
1624
|
+
SEVERITY_GLYPH,
|
|
1625
|
+
SEVERITY_LABEL,
|
|
1626
|
+
VERDICT_STATUS,
|
|
1627
|
+
VERDICT_LABEL,
|
|
1628
|
+
proofMeter,
|
|
1629
|
+
validManifest,
|
|
862
1630
|
renderBody,
|
|
863
1631
|
renderReviewOnlyBody,
|
|
1632
|
+
renderManifestBody,
|
|
1633
|
+
renderManifestLanes,
|
|
864
1634
|
renderMissingKeyBody,
|
|
865
1635
|
renderNoReportBody,
|
|
1636
|
+
renderManifestStateBody,
|
|
1637
|
+
renderSticky,
|
|
1638
|
+
resolveManifest,
|
|
1639
|
+
COMMENT_BUDGET_BYTES,
|
|
866
1640
|
planInlineComments,
|
|
867
1641
|
postInlineComments,
|
|
868
1642
|
shortHash,
|
|
869
1643
|
postedDedupKey,
|
|
1644
|
+
INLINE_SENTINEL,
|
|
1645
|
+
normalizeFindingMessage,
|
|
1646
|
+
parseInlineBody,
|
|
1647
|
+
isArgusInlineBody,
|
|
1648
|
+
fitStatusDescription,
|
|
1649
|
+
statusDescription,
|
|
1650
|
+
reviewBody,
|
|
870
1651
|
}
|