@plainconceptsplatform/workflows 0.20.3 → 0.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +0 -0
- package/loops/actions/classify-route/action.yml +100 -100
- package/loops/actions/classify-route/classify-route.sh +10 -9
- package/loops/actions/decide-agent-retry/action.yml +72 -0
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +223 -19
- package/loops/templates/ci/app-ci-dotnet-next.yml +9 -9
- package/loops/templates/ci/app-ci-node-monorepo.yml +11 -11
- package/loops/templates/release/github-release.yml +1 -1
- package/loops/workflows/agent-apply-review.md +65 -0
- package/loops/workflows/agent-implement.md +99 -6
- package/loops/workflows/agent-merge-gate.md +42 -7
- package/loops/workflows/agent-refine.md +66 -0
- package/loops/workflows/agent-triage.md +66 -0
- package/loops/workflows/shared/opencode-ci.md +206 -206
- package/loops/workflows/work-router.yml +7 -4
- package/package.json +8 -9
|
@@ -62,7 +62,7 @@ jobs:
|
|
|
62
62
|
api:
|
|
63
63
|
name: API (.NET)
|
|
64
64
|
if: github.event_name != 'schedule'
|
|
65
|
-
runs-on:
|
|
65
|
+
runs-on: RunnerLandingZone
|
|
66
66
|
timeout-minutes: 30
|
|
67
67
|
services:
|
|
68
68
|
sqlserver:
|
|
@@ -123,7 +123,7 @@ jobs:
|
|
|
123
123
|
mutation-api:
|
|
124
124
|
name: Mutation (.NET, diff)
|
|
125
125
|
if: github.event_name == 'pull_request'
|
|
126
|
-
runs-on:
|
|
126
|
+
runs-on: RunnerLandingZone
|
|
127
127
|
timeout-minutes: 25
|
|
128
128
|
defaults:
|
|
129
129
|
run:
|
|
@@ -169,7 +169,7 @@ jobs:
|
|
|
169
169
|
web:
|
|
170
170
|
name: Web (Next.js)
|
|
171
171
|
if: github.event_name != 'schedule'
|
|
172
|
-
runs-on:
|
|
172
|
+
runs-on: RunnerLandingZone
|
|
173
173
|
timeout-minutes: 15
|
|
174
174
|
steps:
|
|
175
175
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -200,7 +200,7 @@ jobs:
|
|
|
200
200
|
mutation-web:
|
|
201
201
|
name: Mutation (web, diff)
|
|
202
202
|
if: github.event_name == 'pull_request'
|
|
203
|
-
runs-on:
|
|
203
|
+
runs-on: RunnerLandingZone
|
|
204
204
|
timeout-minutes: 25
|
|
205
205
|
steps:
|
|
206
206
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -241,14 +241,14 @@ jobs:
|
|
|
241
241
|
|
|
242
242
|
secret-scan:
|
|
243
243
|
name: Secret scan (TruffleHog)
|
|
244
|
-
runs-on:
|
|
244
|
+
runs-on: RunnerLandingZone
|
|
245
245
|
steps:
|
|
246
246
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
247
247
|
- run: docker run --rm -v "${{ github.workspace }}:/repo" "$TRUFFLEHOG_IMAGE" filesystem /repo --only-verified --fail --no-update
|
|
248
248
|
|
|
249
249
|
deps-and-iac-scan:
|
|
250
250
|
name: Dependencies and IaC (Trivy)
|
|
251
|
-
runs-on:
|
|
251
|
+
runs-on: RunnerLandingZone
|
|
252
252
|
steps:
|
|
253
253
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
254
254
|
- uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0
|
|
@@ -262,14 +262,14 @@ jobs:
|
|
|
262
262
|
|
|
263
263
|
sast-scan:
|
|
264
264
|
name: SAST (Semgrep)
|
|
265
|
-
runs-on:
|
|
265
|
+
runs-on: RunnerLandingZone
|
|
266
266
|
steps:
|
|
267
267
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
268
268
|
- run: docker run --rm -v "${{ github.workspace }}:/src" -w /src "$SEMGREP_IMAGE" semgrep scan --error --metrics=off --oss-only --disable-version-check --config p/default --config p/security-audit --config p/secrets --config p/csharp --config p/typescript --config p/react --exclude-rule yaml.github-actions.security.github-actions-mutable-action-tag.github-actions-mutable-action-tag
|
|
269
269
|
|
|
270
270
|
sbom:
|
|
271
271
|
name: SBOM (CycloneDX)
|
|
272
|
-
runs-on:
|
|
272
|
+
runs-on: RunnerLandingZone
|
|
273
273
|
steps:
|
|
274
274
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
275
275
|
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0
|
|
@@ -296,7 +296,7 @@ jobs:
|
|
|
296
296
|
github.event.pull_request.head.repo.full_name == github.repository &&
|
|
297
297
|
github.event.pull_request.draft == false &&
|
|
298
298
|
github.event.pull_request.user.type == 'Bot'
|
|
299
|
-
runs-on:
|
|
299
|
+
runs-on: RunnerLandingZone
|
|
300
300
|
timeout-minutes: 5
|
|
301
301
|
permissions:
|
|
302
302
|
contents: read
|
|
@@ -38,7 +38,7 @@ jobs:
|
|
|
38
38
|
lint:
|
|
39
39
|
name: Lint (Biome)
|
|
40
40
|
if: github.event_name != 'schedule'
|
|
41
|
-
runs-on:
|
|
41
|
+
runs-on: RunnerLandingZone
|
|
42
42
|
steps:
|
|
43
43
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
44
44
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -51,7 +51,7 @@ jobs:
|
|
|
51
51
|
web:
|
|
52
52
|
name: Web (Next.js static export)
|
|
53
53
|
if: github.event_name != 'schedule'
|
|
54
|
-
runs-on:
|
|
54
|
+
runs-on: RunnerLandingZone
|
|
55
55
|
steps:
|
|
56
56
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
57
57
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -73,7 +73,7 @@ jobs:
|
|
|
73
73
|
mutation-web:
|
|
74
74
|
name: Mutation (web, diff)
|
|
75
75
|
if: github.event_name == 'pull_request'
|
|
76
|
-
runs-on:
|
|
76
|
+
runs-on: RunnerLandingZone
|
|
77
77
|
timeout-minutes: 25
|
|
78
78
|
steps:
|
|
79
79
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -103,7 +103,7 @@ jobs:
|
|
|
103
103
|
name: E2E (Playwright)
|
|
104
104
|
if: github.event_name != 'schedule'
|
|
105
105
|
needs: web
|
|
106
|
-
runs-on:
|
|
106
|
+
runs-on: RunnerLandingZone
|
|
107
107
|
steps:
|
|
108
108
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
109
109
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -128,7 +128,7 @@ jobs:
|
|
|
128
128
|
name: Desktop (Electron)
|
|
129
129
|
if: github.event_name != 'schedule'
|
|
130
130
|
needs: web
|
|
131
|
-
runs-on:
|
|
131
|
+
runs-on: RunnerLandingZone
|
|
132
132
|
steps:
|
|
133
133
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
134
134
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -152,7 +152,7 @@ jobs:
|
|
|
152
152
|
mobile:
|
|
153
153
|
name: Mobile (Capacitor)
|
|
154
154
|
if: github.event_name != 'schedule'
|
|
155
|
-
runs-on:
|
|
155
|
+
runs-on: RunnerLandingZone
|
|
156
156
|
steps:
|
|
157
157
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
158
158
|
- uses: pnpm/action-setup@f40ffcd9367d9f12939873eb1018b921a783ffaa # v4
|
|
@@ -171,14 +171,14 @@ jobs:
|
|
|
171
171
|
|
|
172
172
|
secret-scan:
|
|
173
173
|
name: Secret scan (TruffleHog)
|
|
174
|
-
runs-on:
|
|
174
|
+
runs-on: RunnerLandingZone
|
|
175
175
|
steps:
|
|
176
176
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
177
177
|
- run: docker run --rm -v "${{ github.workspace }}:/repo" "$TRUFFLEHOG_IMAGE" filesystem /repo --only-verified --fail --no-update
|
|
178
178
|
|
|
179
179
|
deps-and-iac-scan:
|
|
180
180
|
name: Dependencies and IaC (Trivy)
|
|
181
|
-
runs-on:
|
|
181
|
+
runs-on: RunnerLandingZone
|
|
182
182
|
steps:
|
|
183
183
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
184
184
|
- uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0
|
|
@@ -192,14 +192,14 @@ jobs:
|
|
|
192
192
|
|
|
193
193
|
sast-scan:
|
|
194
194
|
name: SAST (Semgrep)
|
|
195
|
-
runs-on:
|
|
195
|
+
runs-on: RunnerLandingZone
|
|
196
196
|
steps:
|
|
197
197
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
198
198
|
- run: docker run --rm -v "${{ github.workspace }}:/src" -w /src "$SEMGREP_IMAGE" semgrep scan --error --metrics=off --oss-only --disable-version-check --config p/default --config p/security-audit --config p/secrets --config p/typescript --config p/react --exclude-rule yaml.github-actions.security.github-actions-mutable-action-tag.github-actions-mutable-action-tag
|
|
199
199
|
|
|
200
200
|
sbom:
|
|
201
201
|
name: SBOM (CycloneDX)
|
|
202
|
-
runs-on:
|
|
202
|
+
runs-on: RunnerLandingZone
|
|
203
203
|
steps:
|
|
204
204
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
205
205
|
- uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0
|
|
@@ -226,7 +226,7 @@ jobs:
|
|
|
226
226
|
github.event.pull_request.head.repo.full_name == github.repository &&
|
|
227
227
|
github.event.pull_request.draft == false &&
|
|
228
228
|
github.event.pull_request.user.type == 'Bot'
|
|
229
|
-
runs-on:
|
|
229
|
+
runs-on: RunnerLandingZone
|
|
230
230
|
timeout-minutes: 5
|
|
231
231
|
permissions:
|
|
232
232
|
contents: read
|
|
@@ -12,6 +12,18 @@ env:
|
|
|
12
12
|
STALLED_LABEL: stalled
|
|
13
13
|
PR_PENDING_LABEL: pr-pending
|
|
14
14
|
REVIEW_MARKER: "<!-- agent-apply-review -->"
|
|
15
|
+
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
16
|
+
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
17
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-3' not
|
|
18
|
+
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
19
|
+
# because only the implement worker could do this.
|
|
20
|
+
APPLY_REVIEW_ATTEMPT_MARKER: "<!-- agent-apply-review-attempt -->"
|
|
21
|
+
MAX_ATTEMPTS: "5"
|
|
22
|
+
PARK_AT_ATTEMPT: "4"
|
|
23
|
+
# Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
|
|
24
|
+
# worse day would fall outside a six-minute window and park for a fault that clears by itself.
|
|
25
|
+
RETRY_UNDER_MINUTES: "10"
|
|
26
|
+
RETRY_COMMENT: "That is what a provider outage looks like -- the run ended before it could produce an answer -- so this is being tried again from the start. It is a fresh run rather than a continuation: nothing is carried over from the attempt that failed."
|
|
15
27
|
INCOMPLETE_COMMENT: "Applying the review feedback ended without an outcome. This worker has no retry of its own: it runs again when somebody reviews or comments on the pull request, and the issue is flagged so it is not lost until then."
|
|
16
28
|
ISSUE_CONTEXT_PATH: /tmp/gh-aw/agent/issue-context.json
|
|
17
29
|
GH_AW_ALLOWED_BOTS: "platform-devbox[bot],github-actions[bot]"
|
|
@@ -38,6 +50,11 @@ imports:
|
|
|
38
50
|
on:
|
|
39
51
|
workflow_call:
|
|
40
52
|
inputs:
|
|
53
|
+
attempts_so_far:
|
|
54
|
+
description: Runs already made for this work that died before producing an answer. Filled by the worker when it re-dispatches itself, not by people.
|
|
55
|
+
required: false
|
|
56
|
+
type: string
|
|
57
|
+
default: '0'
|
|
41
58
|
pr-number:
|
|
42
59
|
description: Pull request number to apply review feedback on.
|
|
43
60
|
required: true
|
|
@@ -285,6 +302,8 @@ jobs:
|
|
|
285
302
|
permissions:
|
|
286
303
|
contents: read
|
|
287
304
|
issues: write
|
|
305
|
+
# the retry re-enters through the router, which is a workflow_dispatch
|
|
306
|
+
actions: write
|
|
288
307
|
steps:
|
|
289
308
|
- name: Checkout workflow actions
|
|
290
309
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -296,13 +315,58 @@ jobs:
|
|
|
296
315
|
with:
|
|
297
316
|
client-id: ${{ secrets.BOT_APP_ID }}
|
|
298
317
|
private-key: ${{ secrets.BOT_PRIVATE_KEY }}
|
|
318
|
+
- name: Decide whether this failure is worth repeating
|
|
319
|
+
id: decide
|
|
320
|
+
uses: ./.github/actions/decide-agent-retry
|
|
321
|
+
with:
|
|
322
|
+
token: ${{ github.token }}
|
|
323
|
+
attempts-so-far: ${{ inputs.attempts_so_far }}
|
|
324
|
+
park-at: ${{ env.PARK_AT_ATTEMPT }}
|
|
325
|
+
under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
|
|
326
|
+
# Recorded before any label moves, so a failure in the steps below leaves a run that can be
|
|
327
|
+
# counted rather than work released with nothing to show for it.
|
|
328
|
+
- name: Report the failed attempt
|
|
329
|
+
if: steps.decide.outputs.retry == 'true'
|
|
330
|
+
uses: ./.github/actions/create-issue-comment
|
|
331
|
+
with:
|
|
332
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
333
|
+
issue-number: ${{ needs.subject.outputs.issue }}
|
|
334
|
+
body: |
|
|
335
|
+
${{ env.APPLY_REVIEW_ATTEMPT_MARKER }}
|
|
336
|
+
Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
|
|
337
|
+
${{ env.RETRY_COMMENT }}
|
|
338
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
339
|
+
- name: Release the reservation for the retry
|
|
340
|
+
if: steps.decide.outputs.retry == 'true'
|
|
341
|
+
uses: ./.github/actions/remove-issue-labels
|
|
342
|
+
with:
|
|
343
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
344
|
+
issue-number: ${{ needs.subject.outputs.issue }}
|
|
345
|
+
labels: ${{ env.WORKING_LABEL }}
|
|
346
|
+
- name: Send the work back through the router
|
|
347
|
+
if: steps.decide.outputs.retry == 'true'
|
|
348
|
+
env:
|
|
349
|
+
GH_TOKEN: ${{ github.token }}
|
|
350
|
+
REPO: ${{ github.repository }}
|
|
351
|
+
REF: ${{ github.event.repository.default_branch }}
|
|
352
|
+
SUBJECT: ${{ inputs.pr-number }}
|
|
353
|
+
NEXT: ${{ steps.decide.outputs.next }}
|
|
354
|
+
run: |
|
|
355
|
+
set -euo pipefail
|
|
356
|
+
# The provider recovers in seconds, so pause before re-entering rather than dispatching
|
|
357
|
+
# back into the same outage. The router's own classify and authorize jobs add more.
|
|
358
|
+
sleep 30
|
|
359
|
+
gh workflow run work-router.yml --repo "$REPO" --ref "$REF" -f operation=apply-review -f pr-number="$SUBJECT" -f attempts_so_far="$NEXT"
|
|
360
|
+
echo "Re-dispatched apply-review for $SUBJECT as attempt $NEXT."
|
|
299
361
|
- name: Release the issue
|
|
362
|
+
if: steps.decide.outputs.retry != 'true'
|
|
300
363
|
uses: ./.github/actions/remove-issue-labels
|
|
301
364
|
with:
|
|
302
365
|
token: ${{ steps.app-token.outputs.token }}
|
|
303
366
|
issue-number: ${{ needs.subject.outputs.issue }}
|
|
304
367
|
labels: ${{ env.WORKING_LABEL }}
|
|
305
368
|
- name: Flag for human review
|
|
369
|
+
if: steps.decide.outputs.retry != 'true'
|
|
306
370
|
uses: ./.github/actions/add-issue-labels
|
|
307
371
|
with:
|
|
308
372
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -311,6 +375,7 @@ jobs:
|
|
|
311
375
|
${{ env.REVIEW_LABEL }}
|
|
312
376
|
${{ env.STALLED_LABEL }}
|
|
313
377
|
- name: Report missing review feedback outcome
|
|
378
|
+
if: steps.decide.outputs.retry != 'true'
|
|
314
379
|
uses: ./.github/actions/create-issue-comment
|
|
315
380
|
with:
|
|
316
381
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -16,7 +16,25 @@ env:
|
|
|
16
16
|
# re-running a decision produces the same decision. Created idempotently where it is applied.
|
|
17
17
|
STALLED_LABEL: stalled
|
|
18
18
|
PR_PENDING_LABEL: pr-pending
|
|
19
|
-
|
|
19
|
+
# Five ways a run ends with no pull request, and they are not the same thing. One sentence
|
|
20
|
+
# covered all of them, which is why three issues in two days said "nothing landed" and none of
|
|
21
|
+
# them could be acted on: one agent gave up in a minute, another did more work than the run that
|
|
22
|
+
# succeeded and never asked for a pull request, and a third was told to stop. The retry that
|
|
23
|
+
# follows is right for some and wasted on the rest. Each is one line: the compiler flattens a
|
|
24
|
+
# multi-line env value.
|
|
25
|
+
#
|
|
26
|
+
# The agent produced nothing at all. Measured on Pliny-Bot #191: the model was called, replied
|
|
27
|
+
# with almost nothing each time, and emitted no safe output.
|
|
28
|
+
NO_OUTPUT_COMMENT: "The implementation run produced nothing at all: no patch, and no request to open a pull request. That is a failed run rather than a decision, so the issue keeps `implement` and is flagged for a retry."
|
|
29
|
+
# The agent wrote the code and never asked for it to be published. Measured on Pliny-Bot #175,
|
|
30
|
+
# whose run produced more output than the run that succeeded the same hour.
|
|
31
|
+
PATCH_WITHOUT_REQUEST_COMMENT: "The implementation wrote code but never asked for a pull request, so nothing was published and the patch is only in the run. The work is not lost -- it is in the run's artifacts -- but it has to be asked for again. The issue keeps `implement` and is flagged for a retry."
|
|
32
|
+
# The agent asked, and the handler did not apply it. Different from a push that conflicted:
|
|
33
|
+
# there, gh-aw kept the patch as an issue and said so.
|
|
34
|
+
OUTPUT_NOT_APPLIED_COMMENT: "The implementation asked for a pull request and it was not applied, so the request exists in the run and nothing reached the repository. That is a failure on our side of the handoff rather than the agent's, and it is flagged for a retry."
|
|
35
|
+
# The agent was asked to do the work and reported there was none. A decision, not a failure:
|
|
36
|
+
# retrying it produces the same answer, so it is not flagged stalled and the janitor leaves it.
|
|
37
|
+
NOOP_COMMENT: "The implementation ran and reported there was nothing to do. That is an answer rather than a failure, so the issue is not queued for another attempt. If it is wrong, say what is missing and label it `implement` again."
|
|
20
38
|
# Said when the agent DID write the code and the push failed. gh-aw pushes through the GraphQL
|
|
21
39
|
# signed-commits API, which rebases onto the current parent, so a `main` that moved under a long
|
|
22
40
|
# run conflicts; gh-aw keeps the work by filing the patch as an issue rather than dropping it,
|
|
@@ -299,8 +317,83 @@ jobs:
|
|
|
299
317
|
${{ env.IMPLEMENT_MARKER }}
|
|
300
318
|
${{ env.PUSH_CONFLICT_COMMENT }}
|
|
301
319
|
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
302
|
-
|
|
303
|
-
|
|
320
|
+
# Which of the remaining four happened, from what the agent itself reported.
|
|
321
|
+
#
|
|
322
|
+
# `output_types` and `has_patch` are the agent job's own outputs, and gh-aw gates on the
|
|
323
|
+
# first of them itself, so they are read here rather than invented. They are also the only
|
|
324
|
+
# way to tell these apart from `conclude`: the usage artifact that carries the token counts
|
|
325
|
+
# is uploaded by `conclusion`, and `conclusion` already depends on this job, so naming it is
|
|
326
|
+
# a dependency cycle that does not compile.
|
|
327
|
+
#
|
|
328
|
+
# Ordered most specific first, and each condition excludes the ones above it, so exactly one
|
|
329
|
+
# of the four fires. Pliny-Bot #191 is the last of them and #175 is the one before it: both
|
|
330
|
+
# said "nothing landed" and neither could be acted on, while #175's run had produced more
|
|
331
|
+
# output than the run that succeeded in the same hour.
|
|
332
|
+
- name: Flag a request that was not applied
|
|
333
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && contains(needs.agent.outputs.output_types, 'create_pull_request')
|
|
334
|
+
uses: ./.github/actions/add-issue-labels
|
|
335
|
+
with:
|
|
336
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
337
|
+
issue-number: ${{ inputs.issue-number }}
|
|
338
|
+
labels: |-
|
|
339
|
+
${{ env.REVIEW_LABEL }}
|
|
340
|
+
${{ env.STALLED_LABEL }}
|
|
341
|
+
- name: Say the request was not applied
|
|
342
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && contains(needs.agent.outputs.output_types, 'create_pull_request')
|
|
343
|
+
uses: ./.github/actions/create-issue-comment
|
|
344
|
+
with:
|
|
345
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
346
|
+
issue-number: ${{ inputs.issue-number }}
|
|
347
|
+
body: |
|
|
348
|
+
${{ env.IMPLEMENT_MARKER }}
|
|
349
|
+
${{ env.OUTPUT_NOT_APPLIED_COMMENT }}
|
|
350
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
351
|
+
- name: Release an issue the agent found nothing to do on
|
|
352
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && contains(needs.agent.outputs.output_types, 'noop')
|
|
353
|
+
uses: ./.github/actions/add-issue-labels
|
|
354
|
+
with:
|
|
355
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
356
|
+
issue-number: ${{ inputs.issue-number }}
|
|
357
|
+
labels: |-
|
|
358
|
+
${{ env.REVIEW_LABEL }}
|
|
359
|
+
- name: Take the implement label off a decided issue
|
|
360
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && contains(needs.agent.outputs.output_types, 'noop')
|
|
361
|
+
uses: ./.github/actions/remove-issue-labels
|
|
362
|
+
with:
|
|
363
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
364
|
+
issue-number: ${{ inputs.issue-number }}
|
|
365
|
+
labels: ${{ env.IMPLEMENT_LABEL }}
|
|
366
|
+
- name: Say there was nothing to do
|
|
367
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && contains(needs.agent.outputs.output_types, 'noop')
|
|
368
|
+
uses: ./.github/actions/create-issue-comment
|
|
369
|
+
with:
|
|
370
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
371
|
+
issue-number: ${{ inputs.issue-number }}
|
|
372
|
+
body: |
|
|
373
|
+
${{ env.IMPLEMENT_MARKER }}
|
|
374
|
+
${{ env.NOOP_COMMENT }}
|
|
375
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
376
|
+
- name: Flag a patch nobody asked to publish
|
|
377
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && !contains(needs.agent.outputs.output_types, 'noop') && needs.agent.outputs.has_patch == 'true'
|
|
378
|
+
uses: ./.github/actions/add-issue-labels
|
|
379
|
+
with:
|
|
380
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
381
|
+
issue-number: ${{ inputs.issue-number }}
|
|
382
|
+
labels: |-
|
|
383
|
+
${{ env.REVIEW_LABEL }}
|
|
384
|
+
${{ env.STALLED_LABEL }}
|
|
385
|
+
- name: Say the patch was never asked for
|
|
386
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && !contains(needs.agent.outputs.output_types, 'noop') && needs.agent.outputs.has_patch == 'true'
|
|
387
|
+
uses: ./.github/actions/create-issue-comment
|
|
388
|
+
with:
|
|
389
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
390
|
+
issue-number: ${{ inputs.issue-number }}
|
|
391
|
+
body: |
|
|
392
|
+
${{ env.IMPLEMENT_MARKER }}
|
|
393
|
+
${{ env.PATCH_WITHOUT_REQUEST_COMMENT }}
|
|
394
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
395
|
+
- name: Flag a run that produced nothing
|
|
396
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && !contains(needs.agent.outputs.output_types, 'noop') && needs.agent.outputs.has_patch != 'true'
|
|
304
397
|
uses: ./.github/actions/add-issue-labels
|
|
305
398
|
with:
|
|
306
399
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -308,15 +401,15 @@ jobs:
|
|
|
308
401
|
labels: |-
|
|
309
402
|
${{ env.REVIEW_LABEL }}
|
|
310
403
|
${{ env.STALLED_LABEL }}
|
|
311
|
-
- name: Say
|
|
312
|
-
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '')
|
|
404
|
+
- name: Say the run produced nothing
|
|
405
|
+
if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '') && !contains(needs.agent.outputs.output_types, 'create_pull_request') && !contains(needs.agent.outputs.output_types, 'noop') && needs.agent.outputs.has_patch != 'true'
|
|
313
406
|
uses: ./.github/actions/create-issue-comment
|
|
314
407
|
with:
|
|
315
408
|
token: ${{ steps.app-token.outputs.token }}
|
|
316
409
|
issue-number: ${{ inputs.issue-number }}
|
|
317
410
|
body: |
|
|
318
411
|
${{ env.IMPLEMENT_MARKER }}
|
|
319
|
-
${{ env.
|
|
412
|
+
${{ env.NO_OUTPUT_COMMENT }}
|
|
320
413
|
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
321
414
|
incomplete:
|
|
322
415
|
needs: [agent, safe_outputs, eligibility]
|
|
@@ -285,7 +285,7 @@ jobs:
|
|
|
285
285
|
Protected files:
|
|
286
286
|
${{ needs.protected_changes.outputs.files }}
|
|
287
287
|
|
|
288
|
-
Required owners:
|
|
288
|
+
Required owners: `${{ needs.protected_changes.outputs.required_owners || 'none configured' }}`
|
|
289
289
|
|
|
290
290
|
**Verdict:** owner-review
|
|
291
291
|
|
|
@@ -439,7 +439,9 @@ jobs:
|
|
|
439
439
|
| Files / lines | ${{ needs.protected_changes.outputs.files_changed }} / ${{ needs.protected_changes.outputs.lines_changed }} |
|
|
440
440
|
| Protected paths | ${{ needs.protected_changes.outputs.requires_review == 'true' && 'yes' || 'no' }} |
|
|
441
441
|
| Owner paths | ${{ needs.protected_changes.outputs.owner_hit == 'true' && 'yes' || 'no' }} |
|
|
442
|
-
|
|
442
|
+
${{ needs.validate_output.outputs.outcome == 'owner-review'
|
|
443
|
+
&& format('| Required owners | `{0}` |', needs.protected_changes.outputs.required_owners != '' && needs.protected_changes.outputs.required_owners || 'none configured, see the matched paths above')
|
|
444
|
+
|| '' }}
|
|
443
445
|
|
|
444
446
|
Why this blast radius:
|
|
445
447
|
```
|
|
@@ -896,9 +898,35 @@ timeout-minutes: 120
|
|
|
896
898
|
reassurance about their absence.
|
|
897
899
|
|
|
898
900
|
**5c. Check the acceptance criteria.** The issue context at `${{ env.ISSUE_CONTEXT_PATH }}`
|
|
899
|
-
says what this change was supposed to do. Confirm the diff does it.
|
|
900
|
-
|
|
901
|
-
|
|
901
|
+
says what this change was supposed to do. Confirm the diff does it. The double-check the
|
|
902
|
+
rest of this step asks you to do is the gate's own: a defect an agent did not notice and CI
|
|
903
|
+
did not catch is a more expensive rollback than a wrong auto-merge.
|
|
904
|
+
|
|
905
|
+
Set `acceptanceCriteriaMet` to `false` when you can name a criterion the diff does not
|
|
906
|
+
satisfy. Name every criterion that is missing or wrong, in `reason` and in `findings` if
|
|
907
|
+
the gap is a defect (it often is). An empty `findings` array with
|
|
908
|
+
`acceptanceCriteriaMet: false` is accepted but carries little evidence: prefer one entry
|
|
909
|
+
per gap, with `category: correctness`, the file and line, and a `suggestedFix`.
|
|
910
|
+
|
|
911
|
+
**Correctness remediation.** When `acceptanceCriteriaMet` is `false`, you may fix the
|
|
912
|
+
gap and push the fix the same way you fix a failed CI run — this is the same
|
|
913
|
+
`remediated` verdict the merge-conflict and CI-failure paths use, and it re-runs CI and
|
|
914
|
+
the gate on the new head. To do it:
|
|
915
|
+
|
|
916
|
+
- Produce the patch that closes each unmet criterion. Every criterion you named must be
|
|
917
|
+
addressed by the patch, or the next gate cycle reproduces this verdict.
|
|
918
|
+
- Run the verification commands below (scoped to the files you changed) before the push.
|
|
919
|
+
- Push the fix using `push_to_pull_request_branch` (pr_number: ${{ needs.subject.outputs.pr }},
|
|
920
|
+
branch: the current PR branch), then emit the `add_comment` with
|
|
921
|
+
**Verdict:** remediated. Set `acceptanceCriteriaMet` to `false` in the report: the
|
|
922
|
+
criteria were unmet when you reviewed, and the fix is what addresses them. The next
|
|
923
|
+
cycle validates the fix.
|
|
924
|
+
- Do not rebase, reset, amend or otherwise rewrite history: the push is fast-forward only.
|
|
925
|
+
|
|
926
|
+
If you cannot fix a criterion in one pass — it needs a decision, a question, or a code path
|
|
927
|
+
you cannot trace — do not push. Report `assessed` with `acceptanceCriteriaMet: false` and no
|
|
928
|
+
push. The workflow sends the pull request to a human, which is the correct action when the
|
|
929
|
+
gap is beyond a focused repair.
|
|
902
930
|
|
|
903
931
|
**5d. Answer the recoverability checklist.** How easy would this be to undo if it were
|
|
904
932
|
wrong? Cite the diff for each answer, and record the ones that fired in
|
|
@@ -989,8 +1017,9 @@ timeout-minutes: 120
|
|
|
989
1017
|
|
|
990
1018
|
7. Say which of two things you did, and nothing more.
|
|
991
1019
|
|
|
992
|
-
- **`remediated`** — CI failed
|
|
993
|
-
|
|
1020
|
+
- **`remediated`** — something was wrong (CI failed, the branch conflicted, or the diff did
|
|
1021
|
+
not satisfy the acceptance criteria), you fixed it, you verified the fix, and you are
|
|
1022
|
+
pushing it. Exactly one `push_to_pull_request_branch` goes with this word.
|
|
994
1023
|
- **`assessed`** — you reviewed the change and are reporting what you found. No push.
|
|
995
1024
|
|
|
996
1025
|
These are the only two words the workflow accepts. You do not write `merge`, `review`,
|
|
@@ -1039,6 +1068,12 @@ timeout-minutes: 120
|
|
|
1039
1068
|
|
|
1040
1069
|
`"findings": []` on a clean change is the expected output, not a failure to do the job.
|
|
1041
1070
|
|
|
1071
|
+
When `acceptanceCriteriaMet` is `false` and you are pushing a fix, the verdict must be
|
|
1072
|
+
`remediated`, not `assessed`: the validator refuses an `assessed` verdict that carries a
|
|
1073
|
+
push. Set `acceptanceCriteriaMet` to `false` in the report you push with the fix — it
|
|
1074
|
+
describes the code you reviewed, not the fix you just produced. The next gate cycle
|
|
1075
|
+
re-evaluates the updated diff and sets it to `true` (or finds another gap).
|
|
1076
|
+
|
|
1042
1077
|
The workflow applies comments, labels, merges, and closures with the App token. Reading the
|
|
1043
1078
|
repository, running verification commands and delegating a finding to be checked are all part
|
|
1044
1079
|
of the job. What is restricted is what leaves this run: the only safe outputs you may call are
|
|
@@ -24,6 +24,18 @@ env:
|
|
|
24
24
|
# re-running a decision produces the same decision. Created idempotently where it is applied.
|
|
25
25
|
STALLED_LABEL: stalled
|
|
26
26
|
REFINE_MARKER: "<!-- agent-refine -->"
|
|
27
|
+
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
28
|
+
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
29
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-3' not
|
|
30
|
+
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
31
|
+
# because only the implement worker could do this.
|
|
32
|
+
REFINE_ATTEMPT_MARKER: "<!-- agent-refine-attempt -->"
|
|
33
|
+
MAX_ATTEMPTS: "5"
|
|
34
|
+
PARK_AT_ATTEMPT: "4"
|
|
35
|
+
# Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
|
|
36
|
+
# worse day would fall outside a six-minute window and park for a fault that clears by itself.
|
|
37
|
+
RETRY_UNDER_MINUTES: "10"
|
|
38
|
+
RETRY_COMMENT: "That is what a provider outage looks like -- the run ended before it could produce an answer -- so this is being tried again from the start. It is a fresh run rather than a continuation: nothing is carried over from the attempt that failed."
|
|
27
39
|
DRAFT_MARKER: "<!-- agent-refine-draft -->"
|
|
28
40
|
INITIAL_MODE: first
|
|
29
41
|
RESPONSE_MODE: rerefine
|
|
@@ -72,6 +84,11 @@ imports:
|
|
|
72
84
|
on:
|
|
73
85
|
workflow_call:
|
|
74
86
|
inputs:
|
|
87
|
+
attempts_so_far:
|
|
88
|
+
description: Runs already made for this work that died before producing an answer. Filled by the worker when it re-dispatches itself, not by people.
|
|
89
|
+
required: false
|
|
90
|
+
type: string
|
|
91
|
+
default: '0'
|
|
75
92
|
issue-number:
|
|
76
93
|
description: Issue number to refine.
|
|
77
94
|
required: true
|
|
@@ -369,6 +386,8 @@ jobs:
|
|
|
369
386
|
permissions:
|
|
370
387
|
contents: read
|
|
371
388
|
issues: write
|
|
389
|
+
# the retry re-enters through the router, which is a workflow_dispatch
|
|
390
|
+
actions: write
|
|
372
391
|
steps:
|
|
373
392
|
- name: Checkout workflow actions
|
|
374
393
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
@@ -378,13 +397,59 @@ jobs:
|
|
|
378
397
|
with:
|
|
379
398
|
client-id: ${{ secrets.BOT_APP_ID }}
|
|
380
399
|
private-key: ${{ secrets.BOT_PRIVATE_KEY }}
|
|
400
|
+
- name: Decide whether this failure is worth repeating
|
|
401
|
+
id: decide
|
|
402
|
+
uses: ./.github/actions/decide-agent-retry
|
|
403
|
+
with:
|
|
404
|
+
token: ${{ github.token }}
|
|
405
|
+
attempts-so-far: ${{ inputs.attempts_so_far }}
|
|
406
|
+
park-at: ${{ env.PARK_AT_ATTEMPT }}
|
|
407
|
+
under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
|
|
408
|
+
# Recorded before any label moves, so a failure in the steps below leaves a run that can be
|
|
409
|
+
# counted rather than work released with nothing to show for it.
|
|
410
|
+
- name: Report the failed attempt
|
|
411
|
+
if: steps.decide.outputs.retry == 'true'
|
|
412
|
+
uses: ./.github/actions/create-issue-comment
|
|
413
|
+
with:
|
|
414
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
415
|
+
issue-number: ${{ inputs.issue-number }}
|
|
416
|
+
body: |
|
|
417
|
+
${{ env.REFINE_ATTEMPT_MARKER }}
|
|
418
|
+
Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
|
|
419
|
+
${{ env.RETRY_COMMENT }}
|
|
420
|
+
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
421
|
+
- name: Release the reservation for the retry
|
|
422
|
+
if: steps.decide.outputs.retry == 'true'
|
|
423
|
+
uses: ./.github/actions/remove-issue-labels
|
|
424
|
+
with:
|
|
425
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
426
|
+
issue-number: ${{ inputs.issue-number }}
|
|
427
|
+
labels: ${{ env.WORKING_LABEL }}
|
|
428
|
+
- name: Send the work back through the router
|
|
429
|
+
if: steps.decide.outputs.retry == 'true'
|
|
430
|
+
env:
|
|
431
|
+
GH_TOKEN: ${{ github.token }}
|
|
432
|
+
REPO: ${{ github.repository }}
|
|
433
|
+
REF: ${{ github.event.repository.default_branch }}
|
|
434
|
+
SUBJECT: ${{ inputs.issue-number }}
|
|
435
|
+
NEXT: ${{ steps.decide.outputs.next }}
|
|
436
|
+
MODE: ${{ inputs.mode }}
|
|
437
|
+
run: |
|
|
438
|
+
set -euo pipefail
|
|
439
|
+
# The provider recovers in seconds, so pause before re-entering rather than dispatching
|
|
440
|
+
# back into the same outage. The router's own classify and authorize jobs add more.
|
|
441
|
+
sleep 30
|
|
442
|
+
gh workflow run work-router.yml --repo "$REPO" --ref "$REF" -f operation=refine -f issue-number="$SUBJECT" -f mode="$MODE" -f attempts_so_far="$NEXT"
|
|
443
|
+
echo "Re-dispatched refine for $SUBJECT as attempt $NEXT."
|
|
381
444
|
- name: Release the issue
|
|
445
|
+
if: steps.decide.outputs.retry != 'true'
|
|
382
446
|
uses: ./.github/actions/remove-issue-labels
|
|
383
447
|
with:
|
|
384
448
|
token: ${{ steps.app-token.outputs.token }}
|
|
385
449
|
issue-number: ${{ inputs.issue-number }}
|
|
386
450
|
labels: ${{ env.WORKING_LABEL }}
|
|
387
451
|
- name: Flag for human review
|
|
452
|
+
if: steps.decide.outputs.retry != 'true'
|
|
388
453
|
uses: ./.github/actions/add-issue-labels
|
|
389
454
|
with:
|
|
390
455
|
token: ${{ steps.app-token.outputs.token }}
|
|
@@ -393,6 +458,7 @@ jobs:
|
|
|
393
458
|
${{ env.REVIEW_LABEL }}
|
|
394
459
|
${{ env.STALLED_LABEL }}
|
|
395
460
|
- name: Report missing refinement outcome
|
|
461
|
+
if: steps.decide.outputs.retry != 'true'
|
|
396
462
|
uses: ./.github/actions/create-issue-comment
|
|
397
463
|
with:
|
|
398
464
|
token: ${{ steps.app-token.outputs.token }}
|