@plainconceptsplatform/workflows 0.21.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +0 -0
- package/loops/actions/attach-screenshots/action.yml +100 -0
- package/loops/actions/classify-route/action.yml +100 -100
- package/loops/actions/classify-route/classify-route.sh +10 -9
- package/loops/actions/decide-agent-retry/action.yml +72 -0
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +116 -7
- package/loops/templates/ci/app-ci-dotnet-next.yml +9 -9
- package/loops/templates/ci/app-ci-node-monorepo.yml +11 -11
- package/loops/templates/release/github-release.yml +1 -1
- package/loops/workflows/agent-apply-review.md +65 -0
- package/loops/workflows/agent-merge-gate.md +53 -5
- package/loops/workflows/agent-refine.md +66 -0
- package/loops/workflows/agent-triage.md +66 -0
- package/loops/workflows/agent-visual-verify.md +248 -0
- package/loops/workflows/work-router.yml +5 -2
- package/package.json +8 -9
|
@@ -62,7 +62,7 @@ jobs:
|
|
|
62
62
|
api:
|
|
63
63
|
name: API (.NET)
|
|
64
64
|
if: github.event_name != 'schedule'
|
|
65
|
-
runs-on:
|
|
65
|
+
runs-on: RunnerLandingZone
|
|
66
66
|
timeout-minutes: 30
|
|
67
67
|
services:
|
|
68
68
|
sqlserver:
|
|
@@ -123,7 +123,7 @@ jobs:
|
|
|
123
123
|
mutation-api:
|
|
124
124
|
name: Mutation (.NET, diff)
|
|
125
125
|
if: github.event_name == 'pull_request'
|
|
126
|
-
runs-on:
|
|
126
|
+
runs-on: RunnerLandingZone
|
|
127
127
|
timeout-minutes: 25
|
|
128
128
|
defaults:
|
|
129
129
|
run:
|
|
@@ -169,7 +169,7 @@ jobs:
|
|
|
169
169
|
web:
|
|
170
170
|
name: Web (Next.js)
|
|
171
171
|
if: github.event_name != 'schedule'
|
|
172
|
-
runs-on:
|
|
172
|
+
runs-on: RunnerLandingZone
|
|
173
173
|
timeout-minutes: 15
|
|
174
174
|
steps:
|
|
175
175
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -200,7 +200,7 @@ jobs:
|
|
|
200
200
|
mutation-web:
|
|
201
201
|
name: Mutation (web, diff)
|
|
202
202
|
if: github.event_name == 'pull_request'
|
|
203
|
-
runs-on:
|
|
203
|
+
runs-on: RunnerLandingZone
|
|
204
204
|
timeout-minutes: 25
|
|
205
205
|
steps:
|
|
206
206
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -241,14 +241,14 @@ jobs:
|
|
|
241
241
|
|
|
242
242
|
secret-scan:
|
|
243
243
|
name: Secret scan (TruffleHog)
|
|
244
|
-
runs-on:
|
|
244
|
+
runs-on: RunnerLandingZone
|
|
245
245
|
steps:
|
|
246
246
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
247
247
|
- run: docker run --rm -v "${{ github.workspace }}:/repo" "$TRUFFLEHOG_IMAGE" filesystem /repo --only-verified --fail --no-update
|
|
248
248
|
|
|
249
249
|
deps-and-iac-scan:
|
|
250
250
|
name: Dependencies and IaC (Trivy)
|
|
251
|
-
runs-on:
|
|
251
|
+
runs-on: RunnerLandingZone
|
|
252
252
|
steps:
|
|
253
253
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
254
254
|
- uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0
|
|
@@ -262,14 +262,14 @@ jobs:
|
|
|
262
262
|
|
|
263
263
|
sast-scan:
|
|
264
264
|
name: SAST (Semgrep)
|
|
265
|
-
runs-on:
|
|
265
|
+
runs-on: RunnerLandingZone
|
|
266
266
|
steps:
|
|
267
267
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
268
268
|
- run: docker run --rm -v "${{ github.workspace }}:/src" -w /src "$SEMGREP_IMAGE" semgrep scan --error --metrics=off --oss-only --disable-version-check --config p/default --config p/security-audit --config p/secrets --config p/csharp --config p/typescript --config p/react --exclude-rule yaml.github-actions.security.github-actions-mutable-action-tag.github-actions-mutable-action-tag
|
|
269
269
|
|
|
270
270
|
sbom:
|
|
271
271
|
name: SBOM (CycloneDX)
|
|
272
|
-
runs-on:
|
|
272
|
+
runs-on: RunnerLandingZone
|
|
273
273
|
steps:
|
|
274
274
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
275
275
|
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0
|
|
@@ -296,7 +296,7 @@ jobs:
|
|
|
296
296
|
github.event.pull_request.head.repo.full_name == github.repository &&
|
|
297
297
|
github.event.pull_request.draft == false &&
|
|
298
298
|
github.event.pull_request.user.type == 'Bot'
|
|
299
|
-
runs-on:
|
|
299
|
+
runs-on: RunnerLandingZone
|
|
300
300
|
timeout-minutes: 5
|
|
301
301
|
permissions:
|
|
302
302
|
contents: read
|
|
@@ -38,7 +38,7 @@ jobs:
|
|
|
38
38
|
lint:
|
|
39
39
|
name: Lint (Biome)
|
|
40
40
|
if: github.event_name != 'schedule'
|
|
41
|
-
runs-on:
|
|
41
|
+
runs-on: RunnerLandingZone
|
|
42
42
|
steps:
|
|
43
43
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
44
44
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -51,7 +51,7 @@ jobs:
|
|
|
51
51
|
web:
|
|
52
52
|
name: Web (Next.js static export)
|
|
53
53
|
if: github.event_name != 'schedule'
|
|
54
|
-
runs-on:
|
|
54
|
+
runs-on: RunnerLandingZone
|
|
55
55
|
steps:
|
|
56
56
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
57
57
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -73,7 +73,7 @@ jobs:
|
|
|
73
73
|
mutation-web:
|
|
74
74
|
name: Mutation (web, diff)
|
|
75
75
|
if: github.event_name == 'pull_request'
|
|
76
|
-
runs-on:
|
|
76
|
+
runs-on: RunnerLandingZone
|
|
77
77
|
timeout-minutes: 25
|
|
78
78
|
steps:
|
|
79
79
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -103,7 +103,7 @@ jobs:
|
|
|
103
103
|
name: E2E (Playwright)
|
|
104
104
|
if: github.event_name != 'schedule'
|
|
105
105
|
needs: web
|
|
106
|
-
runs-on:
|
|
106
|
+
runs-on: RunnerLandingZone
|
|
107
107
|
steps:
|
|
108
108
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
109
109
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -128,7 +128,7 @@ jobs:
|
|
|
128
128
|
name: Desktop (Electron)
|
|
129
129
|
if: github.event_name != 'schedule'
|
|
130
130
|
needs: web
|
|
131
|
-
runs-on:
|
|
131
|
+
runs-on: RunnerLandingZone
|
|
132
132
|
steps:
|
|
133
133
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
134
134
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -152,7 +152,7 @@ jobs:
|
|
|
152
152
|
mobile:
|
|
153
153
|
name: Mobile (Capacitor)
|
|
154
154
|
if: github.event_name != 'schedule'
|
|
155
|
-
runs-on:
|
|
155
|
+
runs-on: RunnerLandingZone
|
|
156
156
|
steps:
|
|
157
157
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
158
158
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -171,14 +171,14 @@ jobs:
|
|
|
171
171
|
|
|
172
172
|
secret-scan:
|
|
173
173
|
name: Secret scan (TruffleHog)
|
|
174
|
-
runs-on:
|
|
174
|
+
runs-on: RunnerLandingZone
|
|
175
175
|
steps:
|
|
176
176
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
177
177
|
- run: docker run --rm -v "${{ github.workspace }}:/repo" "$TRUFFLEHOG_IMAGE" filesystem /repo --only-verified --fail --no-update
|
|
178
178
|
|
|
179
179
|
deps-and-iac-scan:
|
|
180
180
|
name: Dependencies and IaC (Trivy)
|
|
181
|
-
runs-on:
|
|
181
|
+
runs-on: RunnerLandingZone
|
|
182
182
|
steps:
|
|
183
183
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
184
184
|
- uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0
|
|
@@ -192,14 +192,14 @@ jobs:
|
|
|
192
192
|
|
|
193
193
|
sast-scan:
|
|
194
194
|
name: SAST (Semgrep)
|
|
195
|
-
runs-on:
|
|
195
|
+
runs-on: RunnerLandingZone
|
|
196
196
|
steps:
|
|
197
197
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
198
198
|
- run: docker run --rm -v "${{ github.workspace }}:/src" -w /src "$SEMGREP_IMAGE" semgrep scan --error --metrics=off --oss-only --disable-version-check --config p/default --config p/security-audit --config p/secrets --config p/typescript --config p/react --exclude-rule yaml.github-actions.security.github-actions-mutable-action-tag.github-actions-mutable-action-tag
|
|
199
199
|
|
|
200
200
|
sbom:
|
|
201
201
|
name: SBOM (CycloneDX)
|
|
202
|
-
runs-on:
|
|
202
|
+
runs-on: RunnerLandingZone
|
|
203
203
|
steps:
|
|
204
204
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
205
205
|
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0
|
|
@@ -226,7 +226,7 @@ jobs:
|
|
|
226
226
|
github.event.pull_request.head.repo.full_name == github.repository &&
|
|
227
227
|
github.event.pull_request.draft == false &&
|
|
228
228
|
github.event.pull_request.user.type == 'Bot'
|
|
229
|
-
runs-on:
|
|
229
|
+
runs-on: RunnerLandingZone
|
|
230
230
|
timeout-minutes: 5
|
|
231
231
|
permissions:
|
|
232
232
|
contents: read
|
|
@@ -12,6 +12,18 @@ env:
|
|
|
12
12
|
STALLED_LABEL: stalled
|
|
13
13
|
PR_PENDING_LABEL: pr-pending
|
|
14
14
|
REVIEW_MARKER: "<!-- agent-apply-review -->"
|
|
15
|
+
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
16
|
+
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
17
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-3' not
|
|
18
|
+
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
19
|
+
# because only the implement worker could do this.
|
|
20
|
+
APPLY_REVIEW_ATTEMPT_MARKER: "<!-- agent-apply-review-attempt -->"
|
|
21
|
+
MAX_ATTEMPTS: "5"
|
|
22
|
+
PARK_AT_ATTEMPT: "4"
|
|
23
|
+
# Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
|
|
24
|
+
# worse day would fall outside a six-minute window and park for a fault that clears by itself.
|
|
25
|
+
RETRY_UNDER_MINUTES: "10"
|
|
26
|
+
RETRY_COMMENT: "That is what a provider outage looks like -- the run ended before it could produce an answer -- so this is being tried again from the start. It is a fresh run rather than a continuation: nothing is carried over from the attempt that failed."
|
|
15
27
|
INCOMPLETE_COMMENT: "Applying the review feedback ended without an outcome. This worker has no retry of its own: it runs again when somebody reviews or comments on the pull request, and the issue is flagged so it is not lost until then."
|
|
16
28
|
ISSUE_CONTEXT_PATH: /tmp/gh-aw/agent/issue-context.json
|
|
17
29
|
GH_AW_ALLOWED_BOTS: "platform-devbox[bot],github-actions[bot]"
|
|
@@ -38,6 +50,11 @@ imports:
|
|
|
38
50
|
on:
|
|
39
51
|
workflow_call:
|
|
40
52
|
inputs:
|
|
53
|
+
attempts_so_far:
|
|
54
|
+
description: Runs already made for this work that died before producing an answer. Filled by the worker when it re-dispatches itself, not by people.
|
|
55
|
+
required: false
|
|
56
|
+
type: string
|
|
57
|
+
default: '0'
|
|
41
58
|
pr-number:
|
|
42
59
|
description: Pull request number to apply review feedback on.
|
|
43
60
|
required: true
|
|
@@ -285,6 +302,8 @@ jobs:
|
|
|
285
302
|
permissions:
|
|
286
303
|
contents: read
|
|
287
304
|
issues: write
|
|
305
|
+
# the retry re-enters through the router, which is a workflow_dispatch
|
|
306
|
+
actions: write
|
|
288
307
|
steps:
|
|
289
308
|
- name: Checkout workflow actions
|
|
290
309
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -296,13 +315,58 @@ jobs:
|
|
|
296
315
|
with:
|
|
297
316
|
client-id: ${{ secrets.BOT_APP_ID }}
|
|
298
317
|
private-key: ${{ secrets.BOT_PRIVATE_KEY }}
|
|
318
|
+
- name: Decide whether this failure is worth repeating
|
|
319
|
+
id: decide
|
|
320
|
+
uses: ./.github/actions/decide-agent-retry
|
|
321
|
+
with:
|
|
322
|
+
token: ${{ github.token }}
|
|
323
|
+
attempts-so-far: ${{ inputs.attempts_so_far }}
|
|
324
|
+
park-at: ${{ env.PARK_AT_ATTEMPT }}
|
|
325
|
+
under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
|
|
326
|
+
# Recorded before any label moves, so a failure in the steps below leaves a run that can be
|
|
327
|
+
# counted rather than work released with nothing to show for it.
|
|
328
|
+
- name: Report the failed attempt
|
|
329
|
+
if: steps.decide.outputs.retry == 'true'
|
|
330
|
+
uses: ./.github/actions/create-issue-comment
|
|
331
|
+
with:
|
|
332
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
333
|
+
issue-number: ${{ needs.subject.outputs.issue }}
|
|
334
|
+
body: |
|
|
335
|
+
${{ env.APPLY_REVIEW_ATTEMPT_MARKER }}
|
|
336
|
+
Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
|
|
337
|
+
${{ env.RETRY_COMMENT }}
|
|
338
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
339
|
+
- name: Release the reservation for the retry
|
|
340
|
+
if: steps.decide.outputs.retry == 'true'
|
|
341
|
+
uses: ./.github/actions/remove-issue-labels
|
|
342
|
+
with:
|
|
343
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
344
|
+
issue-number: ${{ needs.subject.outputs.issue }}
|
|
345
|
+
labels: ${{ env.WORKING_LABEL }}
|
|
346
|
+
- name: Send the work back through the router
|
|
347
|
+
if: steps.decide.outputs.retry == 'true'
|
|
348
|
+
env:
|
|
349
|
+
GH_TOKEN: ${{ github.token }}
|
|
350
|
+
REPO: ${{ github.repository }}
|
|
351
|
+
REF: ${{ github.event.repository.default_branch }}
|
|
352
|
+
SUBJECT: ${{ inputs.pr-number }}
|
|
353
|
+
NEXT: ${{ steps.decide.outputs.next }}
|
|
354
|
+
run: |
|
|
355
|
+
set -euo pipefail
|
|
356
|
+
# The provider recovers in seconds, so pause before re-entering rather than dispatching
|
|
357
|
+
# back into the same outage. The router's own classify and authorize jobs add more.
|
|
358
|
+
sleep 30
|
|
359
|
+
gh workflow run work-router.yml --repo "$REPO" --ref "$REF" -f operation=apply-review -f pr-number="$SUBJECT" -f attempts_so_far="$NEXT"
|
|
360
|
+
echo "Re-dispatched apply-review for $SUBJECT as attempt $NEXT."
|
|
299
361
|
- name: Release the issue
|
|
362
|
+
if: steps.decide.outputs.retry != 'true'
|
|
300
363
|
uses: ./.github/actions/remove-issue-labels
|
|
301
364
|
with:
|
|
302
365
|
token: ${{ steps.app-token.outputs.token }}
|
|
303
366
|
issue-number: ${{ needs.subject.outputs.issue }}
|
|
304
367
|
labels: ${{ env.WORKING_LABEL }}
|
|
305
368
|
- name: Flag for human review
|
|
369
|
+
if: steps.decide.outputs.retry != 'true'
|
|
306
370
|
uses: ./.github/actions/add-issue-labels
|
|
307
371
|
with:
|
|
308
372
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -311,6 +375,7 @@ jobs:
|
|
|
311
375
|
${{ env.REVIEW_LABEL }}
|
|
312
376
|
${{ env.STALLED_LABEL }}
|
|
313
377
|
- name: Report missing review feedback outcome
|
|
378
|
+
if: steps.decide.outputs.retry != 'true'
|
|
314
379
|
uses: ./.github/actions/create-issue-comment
|
|
315
380
|
with:
|
|
316
381
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -72,6 +72,14 @@ env:
|
|
|
72
72
|
# one unusable report can keep every other bot pull request in the repository waiting.
|
|
73
73
|
PARK_AT_UNUSABLE_OUTPUT: "2"
|
|
74
74
|
ISSUE_CONTEXT_PATH: /tmp/gh-aw/agent/issue-context.json
|
|
75
|
+
# Visual verification runs after auto-merge is decided and before the merge itself. The worker
|
|
76
|
+
# runs /repo-verify, drives a browser via agent-browser, and attaches screenshots to the linked
|
|
77
|
+
# issue. Failures never block the merge. Consumers opt out by setting this to "false".
|
|
78
|
+
VISUAL_VERIFY_ENABLED: "true"
|
|
79
|
+
VISUAL_VERIFY_BUILD_COMMAND: ""
|
|
80
|
+
VISUAL_VERIFY_START_COMMAND: ""
|
|
81
|
+
VISUAL_VERIFY_PORT: "3000"
|
|
82
|
+
VISUAL_VERIFY_WAIT_SECONDS: "30"
|
|
75
83
|
GH_AW_ALLOWED_BOTS: "platform-devbox[bot],github-actions[bot]"
|
|
76
84
|
GIT_AUTHOR_NAME: "github-actions[bot]"
|
|
77
85
|
GIT_AUTHOR_EMAIL: "github-actions[bot]@users.noreply.github.com"
|
|
@@ -449,6 +457,13 @@ jobs:
|
|
|
449
457
|
```
|
|
450
458
|
|
|
451
459
|
Findings and verification on the linked issue: #${{ needs.subject.outputs.issue }}. [View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
460
|
+
- name: Run visual verification
|
|
461
|
+
if: needs.validate_output.outputs.outcome == 'auto-merge' && env.VISUAL_VERIFY_ENABLED == 'true'
|
|
462
|
+
uses: ./.github/workflows/agent-visual-verify.lock.yml
|
|
463
|
+
with:
|
|
464
|
+
pr-number: ${{ needs.subject.outputs.pr }}
|
|
465
|
+
linked-issue: ${{ needs.subject.outputs.issue }}
|
|
466
|
+
continue-on-error: true
|
|
452
467
|
- name: Merge approved pull request
|
|
453
468
|
if: needs.validate_output.outputs.outcome == 'auto-merge'
|
|
454
469
|
env:
|
|
@@ -898,9 +913,35 @@ timeout-minutes: 120
|
|
|
898
913
|
reassurance about their absence.
|
|
899
914
|
|
|
900
915
|
**5c. Check the acceptance criteria.** The issue context at `${{ env.ISSUE_CONTEXT_PATH }}`
|
|
901
|
-
says what this change was supposed to do. Confirm the diff does it.
|
|
902
|
-
|
|
903
|
-
|
|
916
|
+
says what this change was supposed to do. Confirm the diff does it. The double-check the
|
|
917
|
+
rest of this step asks you to do is the gate's own: a defect an agent did not notice and CI
|
|
918
|
+
did not catch is a more expensive rollback than a wrong auto-merge.
|
|
919
|
+
|
|
920
|
+
Set `acceptanceCriteriaMet` to `false` when you can name a criterion the diff does not
|
|
921
|
+
satisfy. Name every criterion that is missing or wrong, in `reason` and in `findings` if
|
|
922
|
+
the gap is a defect (it often is). An empty `findings` array with
|
|
923
|
+
`acceptanceCriteriaMet: false` is accepted but carries little evidence: prefer one entry
|
|
924
|
+
per gap, with `category: correctness`, the file and line, and a `suggestedFix`.
|
|
925
|
+
|
|
926
|
+
**Correctness remediation.** When `acceptanceCriteriaMet` is `false`, you may fix the
|
|
927
|
+
gap and push the fix the same way you fix a failed CI run — this is the same
|
|
928
|
+
`remediated` verdict the merge-conflict and CI-failure paths use, and it re-runs CI and
|
|
929
|
+
the gate on the new head. To do it:
|
|
930
|
+
|
|
931
|
+
- Produce the patch that closes each unmet criterion. Every criterion you named must be
|
|
932
|
+
addressed by the patch, or the next gate cycle reproduces this verdict.
|
|
933
|
+
- Run the verification commands below (scoped to the files you changed) before the push.
|
|
934
|
+
- Push the fix using `push_to_pull_request_branch` (pr_number: ${{ needs.subject.outputs.pr }},
|
|
935
|
+
branch: the current PR branch), then emit the `add_comment` with
|
|
936
|
+
**Verdict:** remediated. Set `acceptanceCriteriaMet` to `false` in the report: the
|
|
937
|
+
criteria were unmet when you reviewed, and the fix is what addresses them. The next
|
|
938
|
+
cycle validates the fix.
|
|
939
|
+
- Do not rebase, reset, amend or otherwise rewrite history: the push is fast-forward only.
|
|
940
|
+
|
|
941
|
+
If you cannot fix a criterion in one pass — it needs a decision, a question, or a code path
|
|
942
|
+
you cannot trace — do not push. Report `assessed` with `acceptanceCriteriaMet: false` and no
|
|
943
|
+
push. The workflow sends the pull request to a human, which is the correct action when the
|
|
944
|
+
gap is beyond a focused repair.
|
|
904
945
|
|
|
905
946
|
**5d. Answer the recoverability checklist.** How easy would this be to undo if it were
|
|
906
947
|
wrong? Cite the diff for each answer, and record the ones that fired in
|
|
@@ -991,8 +1032,9 @@ timeout-minutes: 120
|
|
|
991
1032
|
|
|
992
1033
|
7. Say which of two things you did, and nothing more.
|
|
993
1034
|
|
|
994
|
-
- **`remediated`** — CI failed
|
|
995
|
-
|
|
1035
|
+
- **`remediated`** — something was wrong (CI failed, the branch conflicted, or the diff did
|
|
1036
|
+
not satisfy the acceptance criteria), you fixed it, you verified the fix, and you are
|
|
1037
|
+
pushing it. Exactly one `push_to_pull_request_branch` goes with this word.
|
|
996
1038
|
- **`assessed`** — you reviewed the change and are reporting what you found. No push.
|
|
997
1039
|
|
|
998
1040
|
These are the only two words the workflow accepts. You do not write `merge`, `review`,
|
|
@@ -1041,6 +1083,12 @@ timeout-minutes: 120
|
|
|
1041
1083
|
|
|
1042
1084
|
`"findings": []` on a clean change is the expected output, not a failure to do the job.
|
|
1043
1085
|
|
|
1086
|
+
When `acceptanceCriteriaMet` is `false` and you are pushing a fix, the verdict must be
|
|
1087
|
+
`remediated`, not `assessed`: the validator refuses an `assessed` verdict that carries a
|
|
1088
|
+
push. Set `acceptanceCriteriaMet` to `false` in the report you push with the fix — it
|
|
1089
|
+
describes the code you reviewed, not the fix you just produced. The next gate cycle
|
|
1090
|
+
re-evaluates the updated diff and sets it to `true` (or finds another gap).
|
|
1091
|
+
|
|
1044
1092
|
The workflow applies comments, labels, merges, and closures with the App token. Reading the
|
|
1045
1093
|
repository, running verification commands and delegating a finding to be checked are all part
|
|
1046
1094
|
of the job. What is restricted is what leaves this run: the only safe outputs you may call are
|
|
@@ -24,6 +24,18 @@ env:
|
|
|
24
24
|
# re-running a decision produces the same decision. Created idempotently where it is applied.
|
|
25
25
|
STALLED_LABEL: stalled
|
|
26
26
|
REFINE_MARKER: "<!-- agent-refine -->"
|
|
27
|
+
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
28
|
+
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
29
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-3' not
|
|
30
|
+
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
31
|
+
# because only the implement worker could do this.
|
|
32
|
+
REFINE_ATTEMPT_MARKER: "<!-- agent-refine-attempt -->"
|
|
33
|
+
MAX_ATTEMPTS: "5"
|
|
34
|
+
PARK_AT_ATTEMPT: "4"
|
|
35
|
+
# Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
|
|
36
|
+
# worse day would fall outside a six-minute window and park for a fault that clears by itself.
|
|
37
|
+
RETRY_UNDER_MINUTES: "10"
|
|
38
|
+
RETRY_COMMENT: "That is what a provider outage looks like -- the run ended before it could produce an answer -- so this is being tried again from the start. It is a fresh run rather than a continuation: nothing is carried over from the attempt that failed."
|
|
27
39
|
DRAFT_MARKER: "<!-- agent-refine-draft -->"
|
|
28
40
|
INITIAL_MODE: first
|
|
29
41
|
RESPONSE_MODE: rerefine
|
|
@@ -72,6 +84,11 @@ imports:
|
|
|
72
84
|
on:
|
|
73
85
|
workflow_call:
|
|
74
86
|
inputs:
|
|
87
|
+
attempts_so_far:
|
|
88
|
+
description: Runs already made for this work that died before producing an answer. Filled by the worker when it re-dispatches itself, not by people.
|
|
89
|
+
required: false
|
|
90
|
+
type: string
|
|
91
|
+
default: '0'
|
|
75
92
|
issue-number:
|
|
76
93
|
description: Issue number to refine.
|
|
77
94
|
required: true
|
|
@@ -369,6 +386,8 @@ jobs:
|
|
|
369
386
|
permissions:
|
|
370
387
|
contents: read
|
|
371
388
|
issues: write
|
|
389
|
+
# the retry re-enters through the router, which is a workflow_dispatch
|
|
390
|
+
actions: write
|
|
372
391
|
steps:
|
|
373
392
|
- name: Checkout workflow actions
|
|
374
393
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -378,13 +397,59 @@ jobs:
|
|
|
378
397
|
with:
|
|
379
398
|
client-id: ${{ secrets.BOT_APP_ID }}
|
|
380
399
|
private-key: ${{ secrets.BOT_PRIVATE_KEY }}
|
|
400
|
+
- name: Decide whether this failure is worth repeating
|
|
401
|
+
id: decide
|
|
402
|
+
uses: ./.github/actions/decide-agent-retry
|
|
403
|
+
with:
|
|
404
|
+
token: ${{ github.token }}
|
|
405
|
+
attempts-so-far: ${{ inputs.attempts_so_far }}
|
|
406
|
+
park-at: ${{ env.PARK_AT_ATTEMPT }}
|
|
407
|
+
under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
|
|
408
|
+
# Recorded before any label moves, so a failure in the steps below leaves a run that can be
|
|
409
|
+
# counted rather than work released with nothing to show for it.
|
|
410
|
+
- name: Report the failed attempt
|
|
411
|
+
if: steps.decide.outputs.retry == 'true'
|
|
412
|
+
uses: ./.github/actions/create-issue-comment
|
|
413
|
+
with:
|
|
414
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
415
|
+
issue-number: ${{ inputs.issue-number }}
|
|
416
|
+
body: |
|
|
417
|
+
${{ env.REFINE_ATTEMPT_MARKER }}
|
|
418
|
+
Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
|
|
419
|
+
${{ env.RETRY_COMMENT }}
|
|
420
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
421
|
+
- name: Release the reservation for the retry
|
|
422
|
+
if: steps.decide.outputs.retry == 'true'
|
|
423
|
+
uses: ./.github/actions/remove-issue-labels
|
|
424
|
+
with:
|
|
425
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
426
|
+
issue-number: ${{ inputs.issue-number }}
|
|
427
|
+
labels: ${{ env.WORKING_LABEL }}
|
|
428
|
+
- name: Send the work back through the router
|
|
429
|
+
if: steps.decide.outputs.retry == 'true'
|
|
430
|
+
env:
|
|
431
|
+
GH_TOKEN: ${{ github.token }}
|
|
432
|
+
REPO: ${{ github.repository }}
|
|
433
|
+
REF: ${{ github.event.repository.default_branch }}
|
|
434
|
+
SUBJECT: ${{ inputs.issue-number }}
|
|
435
|
+
NEXT: ${{ steps.decide.outputs.next }}
|
|
436
|
+
MODE: ${{ inputs.mode }}
|
|
437
|
+
run: |
|
|
438
|
+
set -euo pipefail
|
|
439
|
+
# The provider recovers in seconds, so pause before re-entering rather than dispatching
|
|
440
|
+
# back into the same outage. The router's own classify and authorize jobs add more.
|
|
441
|
+
sleep 30
|
|
442
|
+
gh workflow run work-router.yml --repo "$REPO" --ref "$REF" -f operation=refine -f issue-number="$SUBJECT" -f mode="$MODE" -f attempts_so_far="$NEXT"
|
|
443
|
+
echo "Re-dispatched refine for $SUBJECT as attempt $NEXT."
|
|
381
444
|
- name: Release the issue
|
|
445
|
+
if: steps.decide.outputs.retry != 'true'
|
|
382
446
|
uses: ./.github/actions/remove-issue-labels
|
|
383
447
|
with:
|
|
384
448
|
token: ${{ steps.app-token.outputs.token }}
|
|
385
449
|
issue-number: ${{ inputs.issue-number }}
|
|
386
450
|
labels: ${{ env.WORKING_LABEL }}
|
|
387
451
|
- name: Flag for human review
|
|
452
|
+
if: steps.decide.outputs.retry != 'true'
|
|
388
453
|
uses: ./.github/actions/add-issue-labels
|
|
389
454
|
with:
|
|
390
455
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -393,6 +458,7 @@ jobs:
|
|
|
393
458
|
${{ env.REVIEW_LABEL }}
|
|
394
459
|
${{ env.STALLED_LABEL }}
|
|
395
460
|
- name: Report missing refinement outcome
|
|
461
|
+
if: steps.decide.outputs.retry != 'true'
|
|
396
462
|
uses: ./.github/actions/create-issue-comment
|
|
397
463
|
with:
|
|
398
464
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -17,6 +17,18 @@ env:
|
|
|
17
17
|
STALLED_LABEL: stalled
|
|
18
18
|
REFINE_LABEL: refine
|
|
19
19
|
TRIAGE_MARKER: "<!-- agent-triage -->"
|
|
20
|
+
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
21
|
+
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
22
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-3' not
|
|
23
|
+
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
24
|
+
# because only the implement worker could do this.
|
|
25
|
+
TRIAGE_ATTEMPT_MARKER: "<!-- agent-triage-attempt -->"
|
|
26
|
+
MAX_ATTEMPTS: "5"
|
|
27
|
+
PARK_AT_ATTEMPT: "4"
|
|
28
|
+
# Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
|
|
29
|
+
# worse day would fall outside a six-minute window and park for a fault that clears by itself.
|
|
30
|
+
RETRY_UNDER_MINUTES: "10"
|
|
31
|
+
RETRY_COMMENT: "That is what a provider outage looks like -- the run ended before it could produce an answer -- so this is being tried again from the start. It is a fresh run rather than a continuation: nothing is carried over from the attempt that failed."
|
|
20
32
|
MAX_TRIAGE_ROUNDS: "3"
|
|
21
33
|
INCOMPLETE_COMMENT: "Automated triage ended without an outcome. The triage label remains for a retry."
|
|
22
34
|
SAFE_OUTPUT_COMMENT_PREFIX: "Triage assessment"
|
|
@@ -61,6 +73,11 @@ imports:
|
|
|
61
73
|
on:
|
|
62
74
|
workflow_call:
|
|
63
75
|
inputs:
|
|
76
|
+
attempts_so_far:
|
|
77
|
+
description: Runs already made for this work that died before producing an answer. Filled by the worker when it re-dispatches itself, not by people.
|
|
78
|
+
required: false
|
|
79
|
+
type: string
|
|
80
|
+
default: '0'
|
|
64
81
|
issue-number:
|
|
65
82
|
description: Issue number to triage.
|
|
66
83
|
required: true
|
|
@@ -295,6 +312,8 @@ jobs:
|
|
|
295
312
|
permissions:
|
|
296
313
|
contents: read
|
|
297
314
|
issues: write
|
|
315
|
+
# the retry re-enters through the router, which is a workflow_dispatch
|
|
316
|
+
actions: write
|
|
298
317
|
steps:
|
|
299
318
|
- name: Checkout workflow actions
|
|
300
319
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -304,13 +323,59 @@ jobs:
|
|
|
304
323
|
with:
|
|
305
324
|
client-id: ${{ secrets.BOT_APP_ID }}
|
|
306
325
|
private-key: ${{ secrets.BOT_PRIVATE_KEY }}
|
|
326
|
+
- name: Decide whether this failure is worth repeating
|
|
327
|
+
id: decide
|
|
328
|
+
uses: ./.github/actions/decide-agent-retry
|
|
329
|
+
with:
|
|
330
|
+
token: ${{ github.token }}
|
|
331
|
+
attempts-so-far: ${{ inputs.attempts_so_far }}
|
|
332
|
+
park-at: ${{ env.PARK_AT_ATTEMPT }}
|
|
333
|
+
under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
|
|
334
|
+
# Recorded before any label moves, so a failure in the steps below leaves a run that can be
|
|
335
|
+
# counted rather than work released with nothing to show for it.
|
|
336
|
+
- name: Report the failed attempt
|
|
337
|
+
if: steps.decide.outputs.retry == 'true'
|
|
338
|
+
uses: ./.github/actions/create-issue-comment
|
|
339
|
+
with:
|
|
340
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
341
|
+
issue-number: ${{ inputs.issue-number }}
|
|
342
|
+
body: |
|
|
343
|
+
${{ env.TRIAGE_ATTEMPT_MARKER }}
|
|
344
|
+
Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
|
|
345
|
+
${{ env.RETRY_COMMENT }}
|
|
346
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
347
|
+
- name: Release the reservation for the retry
|
|
348
|
+
if: steps.decide.outputs.retry == 'true'
|
|
349
|
+
uses: ./.github/actions/remove-issue-labels
|
|
350
|
+
with:
|
|
351
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
352
|
+
issue-number: ${{ inputs.issue-number }}
|
|
353
|
+
labels: ${{ env.WORKING_LABEL }}
|
|
354
|
+
- name: Send the work back through the router
|
|
355
|
+
if: steps.decide.outputs.retry == 'true'
|
|
356
|
+
env:
|
|
357
|
+
GH_TOKEN: ${{ github.token }}
|
|
358
|
+
REPO: ${{ github.repository }}
|
|
359
|
+
REF: ${{ github.event.repository.default_branch }}
|
|
360
|
+
SUBJECT: ${{ inputs.issue-number }}
|
|
361
|
+
NEXT: ${{ steps.decide.outputs.next }}
|
|
362
|
+
MODE: ${{ inputs.mode }}
|
|
363
|
+
run: |
|
|
364
|
+
set -euo pipefail
|
|
365
|
+
# The provider recovers in seconds, so pause before re-entering rather than dispatching
|
|
366
|
+
# back into the same outage. The router's own classify and authorize jobs add more.
|
|
367
|
+
sleep 30
|
|
368
|
+
gh workflow run work-router.yml --repo "$REPO" --ref "$REF" -f operation=triage -f issue-number="$SUBJECT" -f mode="$MODE" -f attempts_so_far="$NEXT"
|
|
369
|
+
echo "Re-dispatched triage for $SUBJECT as attempt $NEXT."
|
|
307
370
|
- name: Release the issue
|
|
371
|
+
if: steps.decide.outputs.retry != 'true'
|
|
308
372
|
uses: ./.github/actions/remove-issue-labels
|
|
309
373
|
with:
|
|
310
374
|
token: ${{ steps.app-token.outputs.token }}
|
|
311
375
|
issue-number: ${{ inputs.issue-number }}
|
|
312
376
|
labels: ${{ env.WORKING_LABEL }}
|
|
313
377
|
- name: Flag for human review
|
|
378
|
+
if: steps.decide.outputs.retry != 'true'
|
|
314
379
|
uses: ./.github/actions/add-issue-labels
|
|
315
380
|
with:
|
|
316
381
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -319,6 +384,7 @@ jobs:
|
|
|
319
384
|
${{ env.REVIEW_LABEL }}
|
|
320
385
|
${{ env.STALLED_LABEL }}
|
|
321
386
|
- name: Report missing triage outcome
|
|
387
|
+
if: steps.decide.outputs.retry != 'true'
|
|
322
388
|
uses: ./.github/actions/create-issue-comment
|
|
323
389
|
with:
|
|
324
390
|
token: ${{ steps.app-token.outputs.token }}
|