@plainconceptsplatform/workflows 0.6.1 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +105 -88
- package/dist/catalog-installation.d.ts +39 -2
- package/dist/catalog-installation.js +172 -109
- package/dist/index.js +128 -73
- package/dist/package-baseline.d.ts +25 -0
- package/dist/package-baseline.js +138 -0
- package/dist/route-processing.d.ts +0 -2
- package/dist/route-processing.js +22 -91
- package/dist/stack-defaults.js +18 -12
- package/dist/tui.js +27 -43
- package/dist/worker-env.d.ts +46 -0
- package/dist/worker-env.js +179 -0
- package/dist/workflow-catalog.d.ts +4 -2
- package/dist/workflow-catalog.js +5 -3
- package/loops/actions/add-issue-labels/action.yml +20 -0
- package/loops/actions/audit-close/action.yml +180 -128
- package/loops/actions/classify-route/classify-route.sh +8 -2
- package/loops/actions/housekeeping/action.yml +251 -0
- package/loops/actions/merge-agent-pr/action.yml +13 -0
- package/loops/actions/report-workflow-errors/action.yml +385 -0
- package/loops/actions/validate-merge-gate-output/validate-merge-gate-output.sh +13 -1
- package/loops/actions/validate-triage-output/action.yml +1 -1
- package/loops/actions/validate-triage-output/validate-triage-output.sh +9 -5
- package/loops/actions/verify-composite-actions/verify-composite-actions.sh +53 -0
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +847 -170
- package/loops/templates/agentics/agentics-error-report.yml +97 -0
- package/loops/templates/opencode/opencode.ci.json +1 -1
- package/loops/workflows/agent-apply-review.md +452 -469
- package/loops/workflows/agent-audit.md +201 -213
- package/loops/workflows/agent-implement.md +616 -640
- package/loops/workflows/agent-merge-gate.md +830 -844
- package/loops/workflows/agent-refine.md +599 -633
- package/loops/workflows/agent-release.md +244 -258
- package/loops/workflows/agent-triage.md +476 -447
- package/loops/workflows/authorize-bot-work.yml +26 -6
- package/loops/workflows/work-router.yml +1185 -1038
- package/package.json +9 -8
- package/dist/action-validation.test.d.ts +0 -1
- package/dist/action-validation.test.js +0 -87
- package/dist/catalog-installation.test.d.ts +0 -1
- package/dist/catalog-installation.test.js +0 -485
- package/dist/catalog-listing.test.d.ts +0 -1
- package/dist/catalog-listing.test.js +0 -150
- package/dist/index.test.d.ts +0 -1
- package/dist/index.test.js +0 -273
- package/dist/repository-inspection.test.d.ts +0 -1
- package/dist/repository-inspection.test.js +0 -77
- package/dist/route-processing.test.d.ts +0 -1
- package/dist/route-processing.test.js +0 -283
- package/dist/stack-defaults.test.d.ts +0 -1
- package/dist/stack-defaults.test.js +0 -266
- package/dist/tui.test.d.ts +0 -1
- package/dist/tui.test.js +0 -249
- package/dist/workflow-catalog.test.d.ts +0 -1
- package/dist/workflow-catalog.test.js +0 -29
- package/loops/actions/stale-recovery/action.yml +0 -288
- package/loops/actions/update-changelog/action.yml +0 -113
|
@@ -0,0 +1,385 @@
|
|
|
1
|
+
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/report-workflow-errors/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
+
name: Report workflow errors upstream
|
|
3
|
+
description: >-
|
|
4
|
+
Summarise the last day of failures in the workflows this package ships and file the summary
|
|
5
|
+
upstream so the package can be improved. Deterministic: no model runs, and nothing that could
|
|
6
|
+
identify this repository or its contents leaves it.
|
|
7
|
+
|
|
8
|
+
inputs:
|
|
9
|
+
token:
|
|
10
|
+
description: Token for THIS repository. Read-only; used to list runs and jobs.
|
|
11
|
+
required: true
|
|
12
|
+
upstream-token:
|
|
13
|
+
description: >-
|
|
14
|
+
Token that can open an issue in the upstream repository. Empty means compute the report,
|
|
15
|
+
attach it to the run, and file nothing.
|
|
16
|
+
required: false
|
|
17
|
+
default: ''
|
|
18
|
+
upstream-repo:
|
|
19
|
+
description: The repository the report is filed in, as owner/name.
|
|
20
|
+
required: false
|
|
21
|
+
default: PlainConceptsPlatform/Agentic-Workflows
|
|
22
|
+
lookback-hours:
|
|
23
|
+
description: How far back to look for failures.
|
|
24
|
+
required: false
|
|
25
|
+
default: '24'
|
|
26
|
+
dry-run:
|
|
27
|
+
description: Compute and print the report, file nothing.
|
|
28
|
+
required: false
|
|
29
|
+
default: 'false'
|
|
30
|
+
|
|
31
|
+
outputs:
|
|
32
|
+
filed:
|
|
33
|
+
description: The upstream issue number, or empty if nothing was filed.
|
|
34
|
+
value: ${{ steps.report.outputs.filed }}
|
|
35
|
+
|
|
36
|
+
runs:
|
|
37
|
+
using: composite
|
|
38
|
+
steps:
|
|
39
|
+
- name: Collect and file
|
|
40
|
+
id: report
|
|
41
|
+
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
|
|
42
|
+
env:
|
|
43
|
+
UPSTREAM_TOKEN: ${{ inputs.upstream-token }}
|
|
44
|
+
UPSTREAM_REPO: ${{ inputs.upstream-repo }}
|
|
45
|
+
LOOKBACK_HOURS: ${{ inputs.lookback-hours }}
|
|
46
|
+
DRY_RUN: ${{ inputs.dry-run }}
|
|
47
|
+
with:
|
|
48
|
+
github-token: ${{ inputs.token }}
|
|
49
|
+
script: |
|
|
50
|
+
// Every consumer of this package is a private repository, and this is the one job that
|
|
51
|
+
// sends anything out of one. The rule it is built around: no free text ever leaves
|
|
52
|
+
// here. Not a log line, not a branch name, not an issue title, not a file path. What
|
|
53
|
+
// leaves is a fixed set of enumerable facts about workflows this package itself
|
|
54
|
+
// ships -- their names, their job and step names, conclusions, timings and counts --
|
|
55
|
+
// plus the NAME of a matched error pattern from the catalogue below. The final body
|
|
56
|
+
// is checked against a leak scanner and the job fails rather than files if the
|
|
57
|
+
// scanner trips.
|
|
58
|
+
|
|
59
|
+
const dryRun = process.env.DRY_RUN === 'true';
|
|
60
|
+
const upstreamToken = process.env.UPSTREAM_TOKEN;
|
|
61
|
+
const [upstreamOwner, upstreamName] = process.env.UPSTREAM_REPO.split('/');
|
|
62
|
+
const since = new Date(Date.now() - Number(process.env.LOOKBACK_HOURS) * 3600 * 1000);
|
|
63
|
+
|
|
64
|
+
// Only workflows this package ships. A consumer's own workflow name is theirs and can
|
|
65
|
+
// describe a product, a customer or an environment ("Deploy: <customer> staging"), so
|
|
66
|
+
// an allowlist by filename is the boundary. Anything not matching is counted and
|
|
67
|
+
// otherwise ignored.
|
|
68
|
+
const OWNED = /^(work-router|agent-(refine|implement|triage|apply-review|merge-gate|audit|release)|authorize-bot-work|agentics-(checks|maintenance))\.(yml|lock\.yml)$/;
|
|
69
|
+
const ownedName = (path) => (path ?? '').split('/').pop() ?? '';
|
|
70
|
+
|
|
71
|
+
// The catalogue. Each entry is a failure shape worth fixing in the package. Only the
|
|
72
|
+
// `id` is ever reported -- never the text that matched it, because the surrounding log
|
|
73
|
+
// line carries repository paths, branch names and issue titles.
|
|
74
|
+
const CATALOGUE = [
|
|
75
|
+
{ id: 'runner-never-assigned', test: /waiting for a runner|no runner matching|requested labels/i, note: 'A job waited for a self-hosted runner that never arrived.' },
|
|
76
|
+
{ id: 'runner-lost-mid-job', test: /lost communication with the server|runner has received a shutdown signal|The self-hosted runner:.*lost/i, note: 'The runner disappeared while a job was running.' },
|
|
77
|
+
{ id: 'job-cancelled-by-timeout', test: /The job running on runner .* has exceeded the maximum execution time|timeout-minutes/i, note: 'A job hit its timeout instead of finishing.' },
|
|
78
|
+
{ id: 'out-of-disk', test: /no space left on device|You are running out of disk space/i, note: 'The runner ran out of disk.' },
|
|
79
|
+
{ id: 'out-of-memory', test: /JavaScript heap out of memory|Killed\s+process|OOMKilled/i, note: 'A step was killed for memory.' },
|
|
80
|
+
{ id: 'jq-error', test: /^jq: error|jq: error \(at/im, note: 'A jq filter in the router or an action died on its input.' },
|
|
81
|
+
{ id: 'gh-api-403', test: /HTTP 403|Resource not accessible by integration/i, note: 'A step called an endpoint its token has no scope for.' },
|
|
82
|
+
{ id: 'gh-api-404', test: /HTTP 404/i, note: 'A step called an endpoint or resource that does not exist.' },
|
|
83
|
+
{ id: 'gh-api-rate-limit', test: /API rate limit exceeded|secondary rate limit/i, note: 'A step was rate limited.' },
|
|
84
|
+
{ id: 'gh-api-409-conflict', test: /HTTP 409|Update is not a fast forward/i, note: 'A push or merge raced another writer.' },
|
|
85
|
+
{ id: 'model-quota', test: /quota|insufficient_quota|429 Too Many Requests/i, note: 'The model gateway refused the request for quota reasons.' },
|
|
86
|
+
{ id: 'model-gateway-unreachable', test: /ECONNREFUSED|ETIMEDOUT|getaddrinfo|502 Bad Gateway|503 Service Unavailable/i, note: 'The model gateway or another network dependency was unreachable.' },
|
|
87
|
+
{ id: 'agent-produced-no-output', test: /no agent output|output file (is )?(missing|empty)|safe.outputs.*empty/i, note: 'The agent finished without emitting any safe output.' },
|
|
88
|
+
{ id: 'lock-out-of-date', test: /lock file is out of date|recompile|\.lock\.yml.*differs/i, note: 'A compiled lock file did not match its source.' },
|
|
89
|
+
{ id: 'action-not-found', test: /Can't find '.*' in|Unable to resolve action/i, note: 'A workflow referenced an action or path that is not present.' },
|
|
90
|
+
{ id: 'yaml-or-expression-invalid', test: /Invalid workflow file|Unrecognized named-value|Unexpected symbol/i, note: 'A workflow failed to parse or an expression did not evaluate.' },
|
|
91
|
+
];
|
|
92
|
+
|
|
93
|
+
// The leak scanner. Runs on the finished body, and its failure is fatal: a report that
|
|
94
|
+
// might carry private information is not filed at all. Deliberately paranoid, because
|
|
95
|
+
// the cost of a false positive is one missing daily report and the cost of a false
|
|
96
|
+
// negative is a private repository described in a public-facing issue.
|
|
97
|
+
const owner = context.repo.owner;
|
|
98
|
+
const repo = context.repo.repo;
|
|
99
|
+
const leakChecks = [
|
|
100
|
+
{ what: 'the repository name', test: new RegExp(`\\b${repo.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i') },
|
|
101
|
+
{ what: 'the owner name', test: new RegExp(`\\b${owner.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'i') },
|
|
102
|
+
{ what: 'a github.com URL', test: /https?:\/\/[^\s)]*github\.com\/[A-Za-z0-9._-]+\/[A-Za-z0-9._-]+/ },
|
|
103
|
+
{ what: 'an email address', test: /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/ },
|
|
104
|
+
{ what: 'an absolute path', test: /(?:^|\s)(?:\/(?:home|Users|github|runner|mnt)\/|[A-Za-z]:\\)/m },
|
|
105
|
+
{ what: 'a long hex or base64 blob that could be a token or a commit', test: /\b(?:gh[pousr]_[A-Za-z0-9]{16,}|[A-Za-z0-9+/]{40,}={0,2}|[0-9a-f]{32,})\b/ },
|
|
106
|
+
{ what: 'an issue or pull request reference', test: /(?:^|[^A-Za-z0-9_/])#\d+/ },
|
|
107
|
+
{ what: 'a branch or ref path', test: /\brefs\/heads\/|\b(?:feature|fix|chore|bot)\/[A-Za-z0-9._-]+/ },
|
|
108
|
+
];
|
|
109
|
+
|
|
110
|
+
// Which version of the package is reporting. Without this a bug fixed weeks ago keeps
|
|
111
|
+
// arriving upstream with no way to tell it is stale: the first report filed by this
|
|
112
|
+
// job was 19 occurrences of a jq bug that a release had already fixed, and nothing in
|
|
113
|
+
// it said so. The version is the package's own name for itself, read from the
|
|
114
|
+
// ownership header the installer stamps, so it carries nothing of the repository.
|
|
115
|
+
let packageVersion = 'unknown';
|
|
116
|
+
try {
|
|
117
|
+
const header = require('node:fs')
|
|
118
|
+
.readFileSync('.github/workflows/work-router.yml', 'utf8')
|
|
119
|
+
.split('\n')[0];
|
|
120
|
+
packageVersion = /@plainconceptsplatform\/workflows@([0-9][0-9A-Za-z.\-+]*)/.exec(header)?.[1] ?? 'unknown';
|
|
121
|
+
} catch {
|
|
122
|
+
core.info('could not read the installed package version from the router header');
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const runs = await github.paginate(github.rest.actions.listWorkflowRunsForRepo, {
|
|
126
|
+
...context.repo, created: `>=${since.toISOString()}`, per_page: 100,
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
// signature -> what we know about it. The signature is the dedup key both here and
|
|
130
|
+
// upstream, and every part of it is package-owned or an enumerated value.
|
|
131
|
+
//
|
|
132
|
+
// THE FINDING SHAPE IS THE PRIVACY BOUNDARY. bodyFor() renders a finding, so a field
|
|
133
|
+
// added here is a field sent to another repository. These eight are all
|
|
134
|
+
// package-owned names or enumerated values; nothing derived from a log, a branch, a
|
|
135
|
+
// path or an issue belongs on this object. The route matrix asserts this exact list,
|
|
136
|
+
// so adding a ninth field fails the build rather than leaking on the next cron.
|
|
137
|
+
const FINDING_FIELDS = ['workflow', 'job', 'step', 'conclusion', 'patternId', 'count', 'runnerLabels', 'attempts'];
|
|
138
|
+
const findings = new Map();
|
|
139
|
+
let ownedRuns = 0;
|
|
140
|
+
let ownedFailures = 0;
|
|
141
|
+
let skippedForeign = 0;
|
|
142
|
+
let queueWaitTotal = 0;
|
|
143
|
+
let queueWaitCount = 0;
|
|
144
|
+
let queueWaitWorst = 0;
|
|
145
|
+
|
|
146
|
+
// A run that never gets a runner never fails: it sits queued until GitHub expires it
|
|
147
|
+
// a day later. On a two-VM fleet that is the most common way work disappears, and it
|
|
148
|
+
// is invisible in every failure count, so it is detected by state and age instead.
|
|
149
|
+
const STUCK_AFTER_MS = 45 * 60 * 1000;
|
|
150
|
+
let stuckInQueue = 0;
|
|
151
|
+
|
|
152
|
+
for (const run of runs) {
|
|
153
|
+
const file = ownedName(run.path);
|
|
154
|
+
if (!OWNED.test(file)) { skippedForeign += 1; continue; }
|
|
155
|
+
ownedRuns += 1;
|
|
156
|
+
|
|
157
|
+
if (['queued', 'waiting', 'pending'].includes(run.status) &&
|
|
158
|
+
Date.now() - new Date(run.created_at).getTime() > STUCK_AFTER_MS) {
|
|
159
|
+
stuckInQueue += 1;
|
|
160
|
+
const signature = `${file} · (whole run) · (never started) · queued · runner-never-assigned`;
|
|
161
|
+
const seen = findings.get(signature) ?? {
|
|
162
|
+
workflow: file, job: '(whole run)', step: '(never started)',
|
|
163
|
+
conclusion: 'queued', patternId: 'runner-never-assigned', count: 0,
|
|
164
|
+
runnerLabels: new Set(), attempts: new Set(),
|
|
165
|
+
};
|
|
166
|
+
seen.count += 1;
|
|
167
|
+
seen.attempts.add(run.run_attempt ?? 1);
|
|
168
|
+
findings.set(signature, seen);
|
|
169
|
+
continue;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
if (!['failure', 'cancelled', 'timed_out', 'startup_failure'].includes(run.conclusion)) continue;
|
|
173
|
+
ownedFailures += 1;
|
|
174
|
+
|
|
175
|
+
let jobs = [];
|
|
176
|
+
try {
|
|
177
|
+
jobs = await github.paginate(github.rest.actions.listJobsForWorkflowRunAttempt, {
|
|
178
|
+
...context.repo, run_id: run.id, attempt_number: run.run_attempt ?? 1, per_page: 100,
|
|
179
|
+
});
|
|
180
|
+
} catch { continue; }
|
|
181
|
+
|
|
182
|
+
for (const job of jobs) {
|
|
183
|
+
if (job.started_at && job.created_at) {
|
|
184
|
+
const wait = (new Date(job.started_at) - new Date(job.created_at)) / 1000;
|
|
185
|
+
if (wait >= 0) {
|
|
186
|
+
queueWaitTotal += wait; queueWaitCount += 1;
|
|
187
|
+
queueWaitWorst = Math.max(queueWaitWorst, Math.round(wait));
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
if (!['failure', 'cancelled', 'timed_out'].includes(job.conclusion)) continue;
|
|
191
|
+
|
|
192
|
+
const step = (job.steps ?? []).find((s) => ['failure', 'timed_out'].includes(s.conclusion));
|
|
193
|
+
// Job and step names come from files this package ships, so they are the
|
|
194
|
+
// package's own words. A `run-name` is not: it interpolates issue titles.
|
|
195
|
+
const stepName = step?.name ?? '(no failing step recorded)';
|
|
196
|
+
|
|
197
|
+
let patternId = 'unrecognised';
|
|
198
|
+
try {
|
|
199
|
+
const { data: log } = await github.rest.actions.downloadJobLogsForWorkflowRun({
|
|
200
|
+
...context.repo, job_id: job.id,
|
|
201
|
+
});
|
|
202
|
+
const text = typeof log === 'string' ? log : Buffer.from(log).toString('utf8');
|
|
203
|
+
// The tail is where the failure is, and reading less of the log is less to get
|
|
204
|
+
// wrong. Only the matched entry's id crosses into the report.
|
|
205
|
+
const tail = text.slice(-20000);
|
|
206
|
+
patternId = CATALOGUE.find((entry) => entry.test.test(tail))?.id ?? 'unrecognised';
|
|
207
|
+
} catch {
|
|
208
|
+
patternId = 'log-unavailable';
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
const signature = `${file} · ${job.name} · ${stepName} · ${job.conclusion} · ${patternId}`;
|
|
212
|
+
const seen = findings.get(signature) ?? {
|
|
213
|
+
workflow: file, job: job.name, step: stepName,
|
|
214
|
+
conclusion: job.conclusion, patternId, count: 0,
|
|
215
|
+
runnerLabels: new Set(), attempts: new Set(),
|
|
216
|
+
};
|
|
217
|
+
seen.count += 1;
|
|
218
|
+
seen.attempts.add(run.run_attempt ?? 1);
|
|
219
|
+
for (const label of job.labels ?? []) seen.runnerLabels.add(label);
|
|
220
|
+
findings.set(signature, seen);
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
const catalogueNote = (id) => CATALOGUE.find((entry) => entry.id === id)?.note
|
|
225
|
+
?? (id === 'log-unavailable' ? 'The job log could not be read, so the failure could not be classified.'
|
|
226
|
+
: 'The log matched no known failure shape. Worth adding one to the catalogue.');
|
|
227
|
+
|
|
228
|
+
const ranked = [...findings.values()].sort((a, b) => b.count - a.count);
|
|
229
|
+
const hours = process.env.LOOKBACK_HOURS;
|
|
230
|
+
|
|
231
|
+
// Enforce the boundary at run time as well as in the build. A field that reached a
|
|
232
|
+
// finding by some path the static check did not see stops the job here rather than
|
|
233
|
+
// being rendered into an issue in another repository.
|
|
234
|
+
for (const finding of ranked) {
|
|
235
|
+
const unexpected = Object.keys(finding).filter((key) => !FINDING_FIELDS.includes(key));
|
|
236
|
+
if (unexpected.length > 0) {
|
|
237
|
+
core.setFailed(`a finding carries ${unexpected.join(', ')}, which is not in the reportable field list. Nothing was filed.`);
|
|
238
|
+
return;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
const bodyFor = (finding) => [
|
|
243
|
+
`<!-- pcp-error-report: ${finding.workflow}|${finding.job}|${finding.step}|${finding.patternId} -->`,
|
|
244
|
+
'',
|
|
245
|
+
'Filed automatically by the `report-workflow-errors` job in a repository that installs',
|
|
246
|
+
'this package. It carries no information about that repository: only the names this',
|
|
247
|
+
'package gives its own workflows, jobs and steps, and a classification of the failure.',
|
|
248
|
+
'',
|
|
249
|
+
'| | |',
|
|
250
|
+
'|---|---|',
|
|
251
|
+
`| Workflow | \`${finding.workflow}\` |`,
|
|
252
|
+
`| Job | \`${finding.job}\` |`,
|
|
253
|
+
`| Failing step | \`${finding.step}\` |`,
|
|
254
|
+
`| Conclusion | \`${finding.conclusion}\` |`,
|
|
255
|
+
`| Classification | \`${finding.patternId}\` |`,
|
|
256
|
+
`| Runner labels | ${[...finding.runnerLabels].map((l) => `\`${l}\``).join(', ') || 'not reported'} |`,
|
|
257
|
+
`| Occurrences in the last ${hours}h | ${finding.count} |`,
|
|
258
|
+
`| Re-run attempts involved | ${[...finding.attempts].sort().join(', ')} |`,
|
|
259
|
+
`| Package version reporting | \`${packageVersion}\` |`,
|
|
260
|
+
'',
|
|
261
|
+
`**What this classification means.** ${catalogueNote(finding.patternId)}`,
|
|
262
|
+
'',
|
|
263
|
+
'### Occurrences',
|
|
264
|
+
'',
|
|
265
|
+
`- ${new Date().toISOString().slice(0, 10)}: ${finding.count} in ${hours}h, on \`${packageVersion}\``,
|
|
266
|
+
].join('\n');
|
|
267
|
+
|
|
268
|
+
const titleFor = (finding) =>
|
|
269
|
+
`${finding.workflow}: ${finding.patternId} in "${finding.step}"`;
|
|
270
|
+
|
|
271
|
+
// The scanner. A body that trips it is never filed, and the job goes red so the
|
|
272
|
+
// pattern that produced it gets fixed rather than quietly dropped every night.
|
|
273
|
+
const scan = (text, label) => {
|
|
274
|
+
const tripped = leakChecks.filter((check) => check.test.test(text));
|
|
275
|
+
if (tripped.length === 0) return true;
|
|
276
|
+
core.error(`${label} was withheld: the leak scanner matched ${tripped.map((t) => t.what).join('; ')}`);
|
|
277
|
+
return false;
|
|
278
|
+
};
|
|
279
|
+
|
|
280
|
+
const summaryRows = ranked.map((f) => [
|
|
281
|
+
`\`${f.workflow}\``, `\`${f.job}\``, `\`${f.patternId}\``, String(f.count),
|
|
282
|
+
]);
|
|
283
|
+
await core.summary
|
|
284
|
+
.addHeading('Workflow error report', 2)
|
|
285
|
+
.addList([
|
|
286
|
+
`Runs from workflows this package ships: ${ownedRuns}`,
|
|
287
|
+
`Of those, failed: ${ownedFailures}`,
|
|
288
|
+
`Runs from the repository's own workflows, not inspected: ${skippedForeign}`,
|
|
289
|
+
`Runs still queued after 45 minutes, so no runner ever took them: ${stuckInQueue}`,
|
|
290
|
+
`Distinct failure signatures: ${ranked.length}`,
|
|
291
|
+
queueWaitCount > 0
|
|
292
|
+
? `Runner queue wait: ${Math.round(queueWaitTotal / queueWaitCount)}s mean, ${queueWaitWorst}s worst`
|
|
293
|
+
: 'Runner queue wait: no jobs started in the window',
|
|
294
|
+
])
|
|
295
|
+
.addTable([
|
|
296
|
+
[{ data: 'Workflow', header: true }, { data: 'Job', header: true }, { data: 'Classification', header: true }, { data: 'Count', header: true }],
|
|
297
|
+
...summaryRows,
|
|
298
|
+
])
|
|
299
|
+
.write();
|
|
300
|
+
|
|
301
|
+
if (ranked.length === 0) {
|
|
302
|
+
core.info('No failures in the window for any workflow this package ships.');
|
|
303
|
+
core.setOutput('filed', '');
|
|
304
|
+
return;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
// Build every report first and scan them all, so a leak is reported once with the
|
|
308
|
+
// full picture rather than one night at a time.
|
|
309
|
+
const reports = [];
|
|
310
|
+
let leaked = 0;
|
|
311
|
+
for (const finding of ranked) {
|
|
312
|
+
const title = titleFor(finding);
|
|
313
|
+
const body = bodyFor(finding);
|
|
314
|
+
if (!scan(`${title}\n${body}`, `The report for ${finding.patternId}`)) { leaked += 1; continue; }
|
|
315
|
+
reports.push({ finding, title, body });
|
|
316
|
+
core.info(`report ready: ${title} (${finding.count})`);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
if (!upstreamToken) {
|
|
320
|
+
// Not silent and not red: the report exists in the job summary, and the run says
|
|
321
|
+
// plainly why it went no further.
|
|
322
|
+
core.warning(`${reports.length} report(s) computed but not filed: no upstream token is configured. The full report is in this job summary.`);
|
|
323
|
+
core.setOutput('filed', '');
|
|
324
|
+
if (leaked > 0) core.setFailed(`${leaked} report(s) were withheld by the leak scanner; fix the field that carried private text.`);
|
|
325
|
+
return;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
const upstream = getOctokit(upstreamToken);
|
|
329
|
+
const filed = [];
|
|
330
|
+
|
|
331
|
+
for (const { finding, title, body } of reports) {
|
|
332
|
+
const marker = body.split('\n')[0];
|
|
333
|
+
// One issue per signature, updated rather than repeated. Searching by the marker
|
|
334
|
+
// keeps that true across every consumer: two repositories hitting the same package
|
|
335
|
+
// bug land on the same issue and its count is the number that matters.
|
|
336
|
+
let existing;
|
|
337
|
+
try {
|
|
338
|
+
const { data } = await upstream.rest.search.issuesAndPullRequests({
|
|
339
|
+
q: `repo:${upstreamOwner}/${upstreamName} is:issue is:open in:body "${marker}"`,
|
|
340
|
+
per_page: 5,
|
|
341
|
+
});
|
|
342
|
+
existing = data.items.find((item) => (item.body ?? '').includes(marker));
|
|
343
|
+
} catch (error) {
|
|
344
|
+
core.warning(`could not search upstream for an existing report: HTTP ${error.status ?? '?'}`);
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
if (dryRun) {
|
|
348
|
+
core.info(`[dry-run] would ${existing ? `append to #${existing.number}` : 'open an issue'}: ${title}`);
|
|
349
|
+
continue;
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
try {
|
|
353
|
+
if (existing) {
|
|
354
|
+
// Append one dated line rather than rewriting the body, so the history of how
|
|
355
|
+
// often this fires stays readable.
|
|
356
|
+
const line = `- ${new Date().toISOString().slice(0, 10)}: ${finding.count} in ${hours}h, on \`${packageVersion}\``;
|
|
357
|
+
const updated = (existing.body ?? '').includes(line)
|
|
358
|
+
? existing.body
|
|
359
|
+
: `${existing.body}\n${line}`;
|
|
360
|
+
if (!scan(updated, `The appended report for ${finding.patternId}`)) { leaked += 1; continue; }
|
|
361
|
+
await upstream.rest.issues.update({
|
|
362
|
+
owner: upstreamOwner, repo: upstreamName,
|
|
363
|
+
issue_number: existing.number, body: updated,
|
|
364
|
+
});
|
|
365
|
+
filed.push(existing.number);
|
|
366
|
+
core.info(`appended to ${upstreamOwner}/${upstreamName}#${existing.number}`);
|
|
367
|
+
} else {
|
|
368
|
+
const { data: created } = await upstream.rest.issues.create({
|
|
369
|
+
owner: upstreamOwner, repo: upstreamName, title, body,
|
|
370
|
+
labels: ['fleet-error-report'],
|
|
371
|
+
});
|
|
372
|
+
filed.push(created.number);
|
|
373
|
+
core.info(`opened ${upstreamOwner}/${upstreamName}#${created.number}`);
|
|
374
|
+
}
|
|
375
|
+
} catch (error) {
|
|
376
|
+
// A missing label, a token without issues:write, a repository that moved. Say so;
|
|
377
|
+
// do not fail the whole report for one signature.
|
|
378
|
+
core.warning(`could not file the report for ${finding.patternId}: HTTP ${error.status ?? '?'} ${error.message ?? ''}`);
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
core.setOutput('filed', filed.join(','));
|
|
383
|
+
if (leaked > 0) {
|
|
384
|
+
core.setFailed(`${leaked} report(s) were withheld by the leak scanner. Nothing was filed for them; fix the field that carried private text.`);
|
|
385
|
+
}
|
|
@@ -26,8 +26,20 @@ jq -r --arg issue "$issue_number" --arg conclusion "$ci_conclusion" '
|
|
|
26
26
|
)
|
|
27
27
|
| .[0] // "invalid") as $outcome
|
|
28
28
|
| ([$items[] | select(.type == "push_to_pull_request_branch")] | length) as $pushes
|
|
29
|
+
# remediated used to require conclusion == "failure", which threw correct work away. The prompt
|
|
30
|
+
# tells the agent to merge main in, verify and push when CI is green but the pull request
|
|
31
|
+
# conflicts. That is a real and common state: a conflicting pull request has no merge ref, so
|
|
32
|
+
# GitHub can never run CI on that head, and the belt falls back to the last verdict on the
|
|
33
|
+
# branch, which is usually success. The agent did the job, the validator called it invalid,
|
|
34
|
+
# conclude was skipped, and because this worker stages its outputs the resolved merge commit
|
|
35
|
+
# was discarded. The belt then dispatched again on the same verdict, up to six times, each a
|
|
36
|
+
# full run on the single-slot merge belt. A push carrying a remediated verdict is remediation
|
|
37
|
+
# whatever CI last said; what still matters is that exactly one push comes with it.
|
|
38
|
+
#
|
|
39
|
+
# No apostrophes in here. This block sits inside the single-quoted jq program, and one
|
|
40
|
+
# apostrophe closes that string and breaks the script.
|
|
29
41
|
| if $outcome == "merge" and $conclusion == "success" and $pushes == 0 then "merge"
|
|
30
|
-
elif $outcome == "remediated" and $
|
|
42
|
+
elif $outcome == "remediated" and $pushes == 1 then "remediated"
|
|
31
43
|
elif $outcome == "review" and $pushes == 0 then "review"
|
|
32
44
|
else "invalid"
|
|
33
45
|
end
|
|
@@ -13,7 +13,7 @@ outputs:
|
|
|
13
13
|
description: Whether the output contains a usable triage comment with a verdict.
|
|
14
14
|
value: ${{ steps.validate.outputs.valid }}
|
|
15
15
|
outcome:
|
|
16
|
-
description: "Deterministic triage outcome: pass, needs-info, block, or invalid."
|
|
16
|
+
description: "Deterministic triage outcome: pass, needs-info, needs-maintainer, block, or invalid."
|
|
17
17
|
value: ${{ steps.validate.outputs.outcome }}
|
|
18
18
|
runs:
|
|
19
19
|
using: composite
|
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
2
|
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/validate-triage-output/validate-triage-output.sh. Update with `workflows update --force`; consumer edits may be overwritten.
|
|
3
|
-
# Print the deterministic triage outcome: pass, needs-info, block, or invalid.
|
|
3
|
+
# Print the deterministic triage outcome: pass, needs-info, needs-maintainer, block, or invalid.
|
|
4
|
+
#
|
|
5
|
+
# needs-maintainer and block are both refusals of product-owner intake, and they part company on what
|
|
6
|
+
# happens next: needs-maintainer keeps the issue open with the review label so a maintainer can take it
|
|
7
|
+
# on, block closes it. Neither value is a substring of another, so the alternation order below is free.
|
|
4
8
|
|
|
5
9
|
set -euo pipefail
|
|
6
10
|
|
|
@@ -13,21 +17,21 @@ if [ ! -f "$output_file" ] || ! jq -e '.items | arrays' "$output_file" >/dev/nul
|
|
|
13
17
|
fi
|
|
14
18
|
|
|
15
19
|
# The agent emits exactly one add_comment on the source issue. The comment body
|
|
16
|
-
# must contain a verdict line matching **Verdict:** pass|needs-info|block.
|
|
20
|
+
# must contain a verdict line matching **Verdict:** pass|needs-info|needs-maintainer|block.
|
|
17
21
|
# capture() returns an object — use a named group (?<v>...) to extract the value.
|
|
18
22
|
# test() before capture() avoids jq errors on non-matching bodies.
|
|
19
23
|
jq -r --arg issue "$issue_number" '
|
|
20
24
|
def extract_verdict:
|
|
21
25
|
[.items[] | select(.type == "add_comment" and (.item_number | tostring) == $issue and (.body | type == "string"))]
|
|
22
26
|
| map(.body |
|
|
23
|
-
if test("\\*\\*Verdict:\\*\\*\\s*(pass|needs-info|block)"; "i") then
|
|
24
|
-
capture("\\*\\*Verdict:\\*\\*\\s*(?<v>pass|needs-info|block)"; "i").v | ascii_downcase
|
|
27
|
+
if test("\\*\\*Verdict:\\*\\*\\s*(pass|needs-info|needs-maintainer|block)"; "i") then
|
|
28
|
+
capture("\\*\\*Verdict:\\*\\*\\s*(?<v>pass|needs-info|needs-maintainer|block)"; "i").v | ascii_downcase
|
|
25
29
|
else empty end
|
|
26
30
|
)
|
|
27
31
|
| .[0] // "none";
|
|
28
32
|
|
|
29
33
|
extract_verdict as $verdict |
|
|
30
|
-
if $verdict == "pass" or $verdict == "needs-info" or $verdict == "block" then
|
|
34
|
+
if $verdict == "pass" or $verdict == "needs-info" or $verdict == "needs-maintainer" or $verdict == "block" then
|
|
31
35
|
$verdict
|
|
32
36
|
else
|
|
33
37
|
"invalid"
|
|
@@ -38,6 +38,59 @@ while IFS= read -r manifest; do
|
|
|
38
38
|
continue
|
|
39
39
|
fi
|
|
40
40
|
|
|
41
|
+
# `actions/github-script` steps carry real JavaScript inside a YAML string, and nothing on
|
|
42
|
+
# the way to production parses it: not the YAML check above, not actionlint, not gh aw
|
|
43
|
+
# compile. A typo there is a green push and a red run at whatever hour the schedule fires.
|
|
44
|
+
# The same shape of bug shipped in a jq filter in the router and broke an hourly job in a
|
|
45
|
+
# consumer for a day, so every inline script gets syntax-checked here.
|
|
46
|
+
#
|
|
47
|
+
# `${{ }}` is substituted before checking: it is a runner expression, not JavaScript, and
|
|
48
|
+
# standing in a string literal keeps the surrounding syntax intact.
|
|
49
|
+
scripts="$(python3 - "$manifest" <<'PY' 2>/dev/null || true
|
|
50
|
+
import re, sys, yaml, pathlib, tempfile, os
|
|
51
|
+
|
|
52
|
+
manifest = pathlib.Path(sys.argv[1])
|
|
53
|
+
doc = yaml.safe_load(manifest.read_text(encoding='utf-8')) or {}
|
|
54
|
+
out = []
|
|
55
|
+
for index, step in enumerate(((doc.get('runs') or {}).get('steps') or [])):
|
|
56
|
+
if not isinstance(step, dict):
|
|
57
|
+
continue
|
|
58
|
+
if 'github-script' not in str(step.get('uses', '')):
|
|
59
|
+
continue
|
|
60
|
+
body = ((step.get('with') or {}).get('script'))
|
|
61
|
+
if not isinstance(body, str):
|
|
62
|
+
continue
|
|
63
|
+
body = re.sub(r'\$\{\{[^}]*\}\}', '"__expr__"', body)
|
|
64
|
+
# github-script runs the body as the content of an async function, so a top-level `return`
|
|
65
|
+
# and a top-level `await` are both legal there. Wrap it the same way or every early return
|
|
66
|
+
# reads as a syntax error.
|
|
67
|
+
handle, path = tempfile.mkstemp(suffix='.mjs')
|
|
68
|
+
with os.fdopen(handle, 'w', encoding='utf-8') as sink:
|
|
69
|
+
sink.write('async function __ghScript(github, context, core, exec, io, glob, require, getOctokit) {\n')
|
|
70
|
+
sink.write(body)
|
|
71
|
+
sink.write('\n}\n')
|
|
72
|
+
out.append('%s\t%s' % (step.get('name', 'step %d' % index), path))
|
|
73
|
+
print('\n'.join(out))
|
|
74
|
+
PY
|
|
75
|
+
)"
|
|
76
|
+
|
|
77
|
+
script_ok=1
|
|
78
|
+
while IFS=$'\t' read -r step_name script_file; do
|
|
79
|
+
[ -n "${script_file:-}" ] || continue
|
|
80
|
+
if ! node --check "$script_file" 2>/tmp/node-check.err; then
|
|
81
|
+
script_ok=0
|
|
82
|
+
echo "FAIL: ${rel} step '${step_name}' has a JavaScript syntax error:" >&2
|
|
83
|
+
sed -n '1,6p' /tmp/node-check.err >&2
|
|
84
|
+
fi
|
|
85
|
+
rm -f "$script_file"
|
|
86
|
+
done <<<"$scripts"
|
|
87
|
+
rm -f /tmp/node-check.err
|
|
88
|
+
|
|
89
|
+
if [ "$script_ok" -eq 0 ]; then
|
|
90
|
+
FAIL=$((FAIL + 1))
|
|
91
|
+
continue
|
|
92
|
+
fi
|
|
93
|
+
|
|
41
94
|
PASS=$((PASS + 1))
|
|
42
95
|
done < <(find "$ACTIONS_DIR" -name 'action.yml' | sort)
|
|
43
96
|
|