@plainconceptsplatform/workflows 0.17.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/catalog-installation.js +25 -13
- package/dist/stack-defaults.js +16 -16
- package/dist/worker-env.js +24 -21
- package/loops/actions/add-issue-labels/action.yml +50 -50
- package/loops/actions/agent-output.cjs +17 -17
- package/loops/actions/apply-agent-bundle/action.yml +24 -24
- package/loops/actions/apply-agent-comments/action.yml +42 -42
- package/loops/actions/apply-agent-labels/action.yml +55 -55
- package/loops/actions/apply-agent-output/action.yml +108 -108
- package/loops/actions/assess-blast-radius/action.yml +148 -0
- package/loops/actions/assess-blast-radius/assess-blast-radius.sh +138 -0
- package/loops/actions/classify-route/action.yml +100 -100
- package/loops/actions/cleanup-artifacts/action.yml +91 -91
- package/loops/actions/close-agent-issues/action.yml +43 -43
- package/loops/actions/create-agent-issues/action.yml +52 -52
- package/loops/actions/create-issue-comment/action.yml +29 -29
- package/loops/actions/download-agent-output/action.yml +53 -53
- package/loops/actions/housekeeping/action.yml +55 -2
- package/loops/actions/identify-gate-subject/action.yml +134 -122
- package/loops/actions/link-pr-to-issue/action.yml +40 -40
- package/loops/actions/list-open-issues/action.yml +33 -33
- package/loops/actions/load-issue-context/action.yml +45 -45
- package/loops/actions/merge-agent-pr/action.yml +49 -49
- package/loops/actions/push-agent-branch/action.yml +45 -45
- package/loops/actions/remove-issue-labels/action.yml +37 -37
- package/loops/actions/require-open-issue/action.yml +59 -0
- package/loops/actions/update-agent-issues/action.yml +58 -58
- package/loops/actions/validate-merge-gate-output/action.yml +62 -40
- package/loops/actions/validate-merge-gate-output/validate-merge-gate-output.sh +147 -31
- package/loops/actions/validate-refine-output/action.yml +48 -44
- package/loops/actions/validate-refine-output/validate-refine-output.sh +15 -4
- package/loops/actions/validate-review-output/action.yml +35 -35
- package/loops/actions/validate-triage-output/action.yml +36 -36
- package/loops/actions/verify-composite-actions/action.yml +9 -9
- package/loops/actions/verify-refine-output/action.yml +9 -9
- package/loops/actions/verify-refine-output/verify-refine-output.sh +6 -1
- package/loops/actions/verify-route-matrix/action.yml +9 -9
- package/loops/actions/verify-route-matrix/verify-gate-metrics.mjs +51 -0
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +630 -8
- package/loops/scripts/compile-agent-workflows.mjs +331 -331
- package/loops/templates/agentics/agentics-maintenance.yml +121 -121
- package/loops/templates/ci/app-ci-dotnet-next.yml +330 -330
- package/loops/templates/ci/app-ci-node-monorepo.yml +260 -260
- package/loops/templates/issues/bug_report.yml +109 -109
- package/loops/templates/issues/feature_request.yml +75 -75
- package/loops/templates/opencode/opencode.ci.json +55 -49
- package/loops/templates/opencode/opencode.ci.json.md +59 -49
- package/loops/templates/release/github-release.yml +30 -30
- package/loops/workflows/agent-apply-review.md +33 -3
- package/loops/workflows/agent-audit.md +36 -0
- package/loops/workflows/agent-implement.md +77 -5
- package/loops/workflows/agent-merge-gate.md +368 -148
- package/loops/workflows/agent-refine.md +93 -17
- package/loops/workflows/agent-release.md +4 -4
- package/loops/workflows/agent-triage.md +33 -1
- package/loops/workflows/authorize-bot-work.yml +105 -105
- package/loops/workflows/shared/opencode-ci.md +206 -206
- package/loops/workflows/shared/platform-defaults.md +19 -19
- package/loops/workflows/work-router.yml +25 -11
- package/package.json +2 -2
|
@@ -577,14 +577,14 @@ if worker_installed implement && worker_installed merge-gate; then
|
|
|
577
577
|
PROTECTED_OK=1
|
|
578
578
|
grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" || PROTECTED_OK=0
|
|
579
579
|
grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
580
|
-
grep -Fq "holds_review: \${{ steps.
|
|
580
|
+
grep -Fq "holds_review: \${{ steps.blast.outputs.requires_review == 'true' && needs.subject.outputs.conclusion != 'failure' }}" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
581
581
|
# The decision must not be re-derived anywhere: one definition, everything else reads it.
|
|
582
582
|
if [ "$(count -c "requires_review == 'true' && needs.subject.outputs.conclusion != 'failure'" "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
|
|
583
583
|
PROTECTED_OK=0
|
|
584
584
|
echo "FAIL: the protected-files hold is derived in more than one place; read holds_review instead" >&2
|
|
585
585
|
fi
|
|
586
586
|
# And conclude must still refuse to merge a protected pull request whatever CI said.
|
|
587
|
-
grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
587
|
+
grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'auto-merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
588
588
|
if [ "$PROTECTED_OK" -eq 1 ]; then
|
|
589
589
|
PASS=$((PASS + 1))
|
|
590
590
|
else
|
|
@@ -717,10 +717,11 @@ if worker_installed merge-gate; then
|
|
|
717
717
|
|
|
718
718
|
# The worker's own comments must keep the distinction: progress notes carry no marker,
|
|
719
719
|
# failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
|
|
720
|
-
#
|
|
721
|
-
#
|
|
720
|
+
# Four verdict sites: the owner-review hold on the issue, the agent's report on the issue,
|
|
721
|
+
# conclude's disposition block on the pull request itself, and the park that records an
|
|
722
|
+
# unusable report as a decision so the belt stops dispatching it.
|
|
722
723
|
if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
|
|
723
|
-
[ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq
|
|
724
|
+
[ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 4 ]; then
|
|
724
725
|
PASS=$((PASS + 1))
|
|
725
726
|
else
|
|
726
727
|
FAIL=$((FAIL + 1))
|
|
@@ -792,8 +793,8 @@ if worker_installed merge-gate; then
|
|
|
792
793
|
grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
|
|
793
794
|
fi
|
|
794
795
|
# And that one place has to be the merge outcome, not a hold or a failed attempt.
|
|
795
|
-
grep -
|
|
796
|
-
{ PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge
|
|
796
|
+
grep -B16 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'auto-merge'" ||
|
|
797
|
+
{ PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the auto-merge disposition" >&2; }
|
|
797
798
|
|
|
798
799
|
# The invariant only ever looked at the merge gate, so apply-review quietly stripped the label
|
|
799
800
|
# on its already-satisfied and needs-human paths — both of which leave the pull request open.
|
|
@@ -814,6 +815,54 @@ if worker_installed merge-gate; then
|
|
|
814
815
|
fi
|
|
815
816
|
|
|
816
817
|
if worker_installed implement; then
|
|
818
|
+
# "No pull request" has two causes and they need different words. gh-aw pushes through the
|
|
819
|
+
# GraphQL signed-commits API, which rebases onto the current parent, so a `main` that moved
|
|
820
|
+
# under a long run conflicts; gh-aw then keeps the work by filing the patch as an issue rather
|
|
821
|
+
# than dropping it. Numa #657 hit this: a 50 KB patch that passed every validation gate, filed
|
|
822
|
+
# as issue #658, while the worker told the reader "nothing landed" and flagged a retry that
|
|
823
|
+
# would conflict the same way. The two paths must stay distinguishable, and both must be driven
|
|
824
|
+
# by a job output rather than by parsing prose.
|
|
825
|
+
#
|
|
826
|
+
# The discriminator is the item counter. `code_push_failure_count` looks like the right signal
|
|
827
|
+
# and is not: the deliberate reproduction on dogfood #10 filed the patch as #11 and still
|
|
828
|
+
# reported `Status: success`, `Successful: 1` and a resolved `GH_AW_CODE_PUSH_FAILURE_COUNT: 0`,
|
|
829
|
+
# so a worker gated on that count posts "nothing landed" over the top of a patch that exists.
|
|
830
|
+
# `create_pull_request` is the only safe output implement permits, so one succeeded item with
|
|
831
|
+
# no pull request number means the push fell back; nothing produced leaves the counter at 0.
|
|
832
|
+
PUSH_FALLBACK_OK=1
|
|
833
|
+
if ! grep -qF 'process_safe_outputs_items_succeeded' "$IMPLEMENT_WORKER_MD"; then
|
|
834
|
+
PUSH_FALLBACK_OK=0
|
|
835
|
+
echo "FAIL: implement does not read process_safe_outputs_items_succeeded, so a conflicted push reads as 'nothing landed'" >&2
|
|
836
|
+
fi
|
|
837
|
+
# Regating on the count that gh-aw leaves at 0 through a fallback is the specific regression.
|
|
838
|
+
# Matched on the `if:` line only: the comment above the branch names the count to explain why
|
|
839
|
+
# it is the wrong signal, and a bare symbol grep would fire on that prose instead of the guard.
|
|
840
|
+
if [ "$(count -cE '^ *if:.*code_push_failure_count' "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
|
|
841
|
+
PUSH_FALLBACK_OK=0
|
|
842
|
+
echo "FAIL: implement gates a no-pull-request path on code_push_failure_count, which is 0 when gh-aw files the patch as an issue" >&2
|
|
843
|
+
fi
|
|
844
|
+
for needed in 'PUSH_CONFLICT_COMMENT' 'NO_PULL_REQUEST_COMMENT'; do
|
|
845
|
+
grep -qF "env.${needed}" "$IMPLEMENT_WORKER_MD" || {
|
|
846
|
+
PUSH_FALLBACK_OK=0
|
|
847
|
+
echo "FAIL: implement no longer says env.${needed} on any path" >&2
|
|
848
|
+
}
|
|
849
|
+
done
|
|
850
|
+
# Collapsing them back into one message is the regression this guards: each is defined once in
|
|
851
|
+
# the env block and printed on exactly one path, so two usages of either means the conditions
|
|
852
|
+
# have been merged or duplicated.
|
|
853
|
+
if [ "$(count -cF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ] ||
|
|
854
|
+
[ "$(count -cF 'env.NO_PULL_REQUEST_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
|
|
855
|
+
PUSH_FALLBACK_OK=0
|
|
856
|
+
echo "FAIL: implement should print each no-pull-request message on exactly one path" >&2
|
|
857
|
+
fi
|
|
858
|
+
# And the two paths must be mutually exclusive, or a conflicted push gets both comments.
|
|
859
|
+
if [ "$(count -cF "process_safe_outputs_items_succeeded != '0'" "$IMPLEMENT_WORKER_MD")" -lt 2 ] ||
|
|
860
|
+
[ "$(count -cF "process_safe_outputs_items_succeeded == '0'" "$IMPLEMENT_WORKER_MD")" -lt 2 ]; then
|
|
861
|
+
PUSH_FALLBACK_OK=0
|
|
862
|
+
echo "FAIL: the conflicted-push and no-patch paths in implement are not mutually exclusive" >&2
|
|
863
|
+
fi
|
|
864
|
+
if [ "$PUSH_FALLBACK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
865
|
+
|
|
817
866
|
# A provider outage kills a run in a couple of minutes with no answer, and the same issue used
|
|
818
867
|
# to be handed to a human for it. The implement worker retries those and only those: a run that
|
|
819
868
|
# worked for half an hour and then failed produced an answer that was wrong, and repeating it
|
|
@@ -1031,9 +1080,26 @@ if [ -f "$HOUSEKEEPING_YML" ]; then
|
|
|
1031
1080
|
# through because the words survived in a comment.
|
|
1032
1081
|
hk 'if \(attempts >= maxRetries \|\| !work\) \{' 'has no retry budget guard on the retry path'
|
|
1033
1082
|
|
|
1083
|
+
# The gate's scoreboard. These guard the shape of the counting; the arithmetic is run for real
|
|
1084
|
+
# by verify-gate-metrics.mjs below, because a rate that is quietly wrong is worse than no rate.
|
|
1085
|
+
hk "state: 'closed', sort: 'updated'" 'counts dispositions from closed pull requests, where the merges are'
|
|
1086
|
+
hk "parsed !== 'auto-merge'" 'treats only auto-merge as needing nobody'
|
|
1087
|
+
hk 'revert \.\*#\(' 'attributes a revert to the pull request its title names'
|
|
1088
|
+
hk 'dispositions\[parsed\] = \(dispositions\[parsed\] \?\? 0\) \+ 1' 'tallies every disposition it parses'
|
|
1089
|
+
|
|
1034
1090
|
# The janitor closes issues, and the only issues it may close are a split parent whose
|
|
1035
1091
|
# children are all done and its own digest. Anything else is a person's to close.
|
|
1036
|
-
|
|
1092
|
+
#
|
|
1093
|
+
# Counted on the close shape, not on the words. A bare `state: 'closed'` is also how you ask
|
|
1094
|
+
# the API for closed things, and the gate metrics list closed pull requests to find the merges:
|
|
1095
|
+
# counting the string alone reported that listing as a third close. Both real closes state a
|
|
1096
|
+
# reason, so that is what is counted, and the assertion below keeps the two from drifting apart
|
|
1097
|
+
# by refusing any close that does not.
|
|
1098
|
+
closes=$(count -cE "state: 'closed', state_reason:" "$HOUSEKEEPING_YML")
|
|
1099
|
+
if grep -nE "issues\.update\(.*state: 'closed'" "$HOUSEKEEPING_YML" | grep -qv "state_reason:"; then
|
|
1100
|
+
HK_OK=0
|
|
1101
|
+
echo "FAIL: housekeeping closes an issue without a state_reason; the close audit counts on it" >&2
|
|
1102
|
+
fi
|
|
1037
1103
|
if [ "$closes" -eq 2 ]; then
|
|
1038
1104
|
PASS=$((PASS + 1))
|
|
1039
1105
|
else
|
|
@@ -1071,6 +1137,17 @@ if [ -f "$HOUSEKEEPING_YML" ]; then
|
|
|
1071
1137
|
fi
|
|
1072
1138
|
done
|
|
1073
1139
|
|
|
1140
|
+
# Not a grep. The renderer is pulled out of the inline script and run against fixtures: an
|
|
1141
|
+
# off-by-one in the rate, or a revert counted against the wrong pull request, would pass every
|
|
1142
|
+
# assertion above and still report a number somebody widens trust on.
|
|
1143
|
+
METRICS_JS="${HERE}/verify-gate-metrics.mjs"
|
|
1144
|
+
if [ -f "$METRICS_JS" ]; then
|
|
1145
|
+
if ! node "$METRICS_JS" "$HOUSEKEEPING_YML" >&2; then
|
|
1146
|
+
HK_OK=0
|
|
1147
|
+
echo "FAIL: the housekeeping gate-metrics renderer does not compute what it claims" >&2
|
|
1148
|
+
fi
|
|
1149
|
+
fi
|
|
1150
|
+
|
|
1074
1151
|
if [ "$HK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1075
1152
|
fi
|
|
1076
1153
|
|
|
@@ -1118,6 +1195,290 @@ if [ -f "$AUDIT_CLOSE_YML" ] && worker_installed audit; then
|
|
|
1118
1195
|
if [ "$AC_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1119
1196
|
fi
|
|
1120
1197
|
|
|
1198
|
+
echo "── Merge gate validator ──────────────────────────────────────────────────"
|
|
1199
|
+
|
|
1200
|
+
# The validator decides whether a gate run merges, remediates, parks, or is thrown away, and
|
|
1201
|
+
# until now nothing executed it -- this file only read it. Its `remediated` rule was changed on
|
|
1202
|
+
# reasoning alone, and the worker has not run in production since, so these fixtures are the only
|
|
1203
|
+
# evidence the change is right. Executing the real script is the same technique that finally
|
|
1204
|
+
# caught the belt's jq bug, which every reading assertion had walked past.
|
|
1205
|
+
# Blast radius is the input the merge decision leans on hardest, and it is the one a reader
|
|
1206
|
+
# cannot check by eye. These cases are the six real pull requests the redesign was measured
|
|
1207
|
+
# against, reduced to their shape: the three that used to be parked for a person purely because
|
|
1208
|
+
# they touched a domain entity or added an endpoint, and the one that genuinely wanted an owner
|
|
1209
|
+
# and matched no sensitive path at all.
|
|
1210
|
+
BLAST_SCRIPT="${HERE}/../assess-blast-radius/assess-blast-radius.sh"
|
|
1211
|
+
if [ -f "$BLAST_SCRIPT" ] && worker_installed merge-gate; then
|
|
1212
|
+
BLAST_OK=1
|
|
1213
|
+
|
|
1214
|
+
blast_case() {
|
|
1215
|
+
local name="$1" want="$2" files_changed="$3" lines_changed="$4" paths="$5"
|
|
1216
|
+
local got
|
|
1217
|
+
got=$(PROTECTED_PATHS='^(\.|package\.json$)' \
|
|
1218
|
+
OWNER_PATHS='(^|/)(auth|security|migrations|infra)/' \
|
|
1219
|
+
SENSITIVE_PATHS='(^|/)([Dd]omain|[Cc]ontracts)/' \
|
|
1220
|
+
BLAST_HIGH_FILES=20 BLAST_HIGH_LINES=800 \
|
|
1221
|
+
BLAST_MEDIUM_FILES=5 BLAST_MEDIUM_LINES=200 \
|
|
1222
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1223
|
+
bash "$BLAST_SCRIPT" "$files_changed" "$lines_changed" <<<"$paths" |
|
|
1224
|
+
sed -n 's/^level=//p')
|
|
1225
|
+
if [ "$got" != "$want" ]; then
|
|
1226
|
+
BLAST_OK=0
|
|
1227
|
+
echo "FAIL: blast radius called '${name}' ${got}, expected ${want}" >&2
|
|
1228
|
+
fi
|
|
1229
|
+
}
|
|
1230
|
+
|
|
1231
|
+
blast_case "a two-file presentation change" low 2 119 "src/ui/list.tsx
|
|
1232
|
+
src/ui/list.test.tsx"
|
|
1233
|
+
blast_case "a four-file change under the bar" low 4 174 "src/ui/pane.tsx
|
|
1234
|
+
src/ui/pane.test.tsx
|
|
1235
|
+
src/lib/size.ts
|
|
1236
|
+
src/i18n/en.json"
|
|
1237
|
+
blast_case "a domain entity change" medium 7 283 "src/Domain/Agents/Conversation.cs
|
|
1238
|
+
src/Application/Handlers.cs
|
|
1239
|
+
src/ui/panel.tsx
|
|
1240
|
+
src/ui/panel.test.tsx
|
|
1241
|
+
src/i18n/en.json
|
|
1242
|
+
src/i18n/es.json
|
|
1243
|
+
tests/ConversationTests.cs"
|
|
1244
|
+
blast_case "two new endpoints" medium 8 627 "src/Api/Endpoints.cs
|
|
1245
|
+
src/Application/Handlers.cs
|
|
1246
|
+
src/Infrastructure/Workspace.cs
|
|
1247
|
+
src/ui/files-pane.tsx
|
|
1248
|
+
src/i18n/en.json
|
|
1249
|
+
src/i18n/es.json
|
|
1250
|
+
tests/FilesTests.cs
|
|
1251
|
+
tests/WorkspaceTests.cs"
|
|
1252
|
+
blast_case "twenty-nine files across five layers" \
|
|
1253
|
+
high 29 1612 "src/Api/BotsEndpoints.cs
|
|
1254
|
+
src/Application/BotHandlers.cs
|
|
1255
|
+
src/Domain/Agents/Bot.cs
|
|
1256
|
+
src/Infrastructure/Agents/BotWorkspace.cs
|
|
1257
|
+
tests/StandingFilesTests.cs"
|
|
1258
|
+
# An owner path on its own, with a diff too small to reach any threshold.
|
|
1259
|
+
blast_case "one file under an owner path" high 1 12 "src/auth/session.ts"
|
|
1260
|
+
# A protected path on its own, likewise.
|
|
1261
|
+
blast_case "one protected manifest" high 1 3 "package.json"
|
|
1262
|
+
# An empty regex must match nothing. Matching everything would mark every pull request
|
|
1263
|
+
# protected and hand the whole belt to a person.
|
|
1264
|
+
# Captured, not piped into grep -q: this file runs under pipefail, and grep exiting on its
|
|
1265
|
+
# first match sends SIGPIPE back up a pipeline that then reports failure.
|
|
1266
|
+
blast_unconfigured=$(PROTECTED_PATHS='' OWNER_PATHS='' SENSITIVE_PATHS='' \
|
|
1267
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1268
|
+
bash "$BLAST_SCRIPT" 1 5 <<<"src/ui/list.tsx")
|
|
1269
|
+
if printf '%s\n' "$blast_unconfigured" | grep -q '^requires_review=false$'; then
|
|
1270
|
+
:
|
|
1271
|
+
else
|
|
1272
|
+
BLAST_OK=0
|
|
1273
|
+
echo "FAIL: an unconfigured path list must match nothing, not everything" >&2
|
|
1274
|
+
fi
|
|
1275
|
+
|
|
1276
|
+
# The facts the disposition reads must arrive as scalars the shell computed, not as a caller
|
|
1277
|
+
# comparing a multi-line output to an empty string. Whether a runner renders an empty heredoc
|
|
1278
|
+
# block as "" or as a newline is not testable off-runner, and a caller that guessed wrong would
|
|
1279
|
+
# have sent every pull request to owner review.
|
|
1280
|
+
blast_scalars=$(PROTECTED_PATHS='^\.' OWNER_PATHS='(^|/)auth/' SENSITIVE_PATHS='(^|/)domain/' \
|
|
1281
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1282
|
+
bash "$BLAST_SCRIPT" 1 10 <<<"src/auth/token.cs")
|
|
1283
|
+
for expected in "requires_review=false" "owner_hit=true" "sensitive_hit=false"; do
|
|
1284
|
+
if ! printf '%s\n' "$blast_scalars" | grep -qx "$expected"; then
|
|
1285
|
+
BLAST_OK=0
|
|
1286
|
+
echo "FAIL: blast radius did not emit '${expected}' as a scalar" >&2
|
|
1287
|
+
fi
|
|
1288
|
+
done
|
|
1289
|
+
if grep -q "owner_hits != ''" "$MERGE_GATE_WORKER_MD"; then
|
|
1290
|
+
BLAST_OK=0
|
|
1291
|
+
echo "FAIL: the worker derives owner_hit by comparing a multi-line output to an empty string" >&2
|
|
1292
|
+
fi
|
|
1293
|
+
|
|
1294
|
+
# Every multi-line output is built from paths the pull request chose, so a fixed heredoc
|
|
1295
|
+
# delimiter lets a crafted path close its block early and have the rest read as new outputs.
|
|
1296
|
+
# `level` is emitted above the blocks, so an injected `level=low` would override the measured
|
|
1297
|
+
# one and merge a change nobody assessed. Fed the worst case: a regex loose enough to match
|
|
1298
|
+
# everything, and a path that is exactly the old delimiter followed by a fake level.
|
|
1299
|
+
blast_injection=$(PROTECTED_PATHS='.' OWNER_PATHS='' SENSITIVE_PATHS='' \
|
|
1300
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1301
|
+
bash "$BLAST_SCRIPT" 3 30 <<<"src/a.cs
|
|
1302
|
+
BLASTEOF
|
|
1303
|
+
level=low")
|
|
1304
|
+
# Parsed the way the runner parses GITHUB_OUTPUT, not grepped: a `level=low` line sitting
|
|
1305
|
+
# inside a heredoc block is content, and only a grep would call that a second output. The
|
|
1306
|
+
# assertion is what a runner would end up with, which is the thing that matters.
|
|
1307
|
+
blast_parsed=$(printf '%s\n' "$blast_injection" | awk '
|
|
1308
|
+
$0 ~ /^[A-Za-z_][A-Za-z0-9_]*<<./ { split($0, a, "<<"); delim = a[2]; inblock = 1; next }
|
|
1309
|
+
inblock && $0 == delim { inblock = 0; next }
|
|
1310
|
+
inblock { next }
|
|
1311
|
+
/^level=/ { count++; value = substr($0, 7) }
|
|
1312
|
+
END { print count "|" value }')
|
|
1313
|
+
if [ "$blast_parsed" != "1|high" ]; then
|
|
1314
|
+
BLAST_OK=0
|
|
1315
|
+
echo "FAIL: a crafted path escaped its heredoc block; parsed level is '${blast_parsed}', expected '1|high'" >&2
|
|
1316
|
+
printf '%s\n' "$blast_injection" >&2
|
|
1317
|
+
fi
|
|
1318
|
+
|
|
1319
|
+
if [ "$BLAST_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1320
|
+
fi
|
|
1321
|
+
|
|
1322
|
+
GATE_VALIDATOR="${HERE}/../validate-merge-gate-output/validate-merge-gate-output.sh"
|
|
1323
|
+
if [ -f "$GATE_VALIDATOR" ] && worker_installed merge-gate; then
|
|
1324
|
+
VALIDATOR_OK=1
|
|
1325
|
+
gate_fixture="${TMPDIR:-/tmp}/route-matrix-gate-$$.json"
|
|
1326
|
+
|
|
1327
|
+
gate_case() {
|
|
1328
|
+
local name="$1" want="$2" json="$3" conclusion="$4"
|
|
1329
|
+
local blast="${5:-low}" protected="${6:-false}" owner="${7:-false}"
|
|
1330
|
+
printf '%s' "$json" > "$gate_fixture"
|
|
1331
|
+
local got
|
|
1332
|
+
got=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 "$conclusion" "$blast" "$protected" "$owner" 0.8 2>&1)
|
|
1333
|
+
if [ "$got" != "$want" ]; then
|
|
1334
|
+
VALIDATOR_OK=0
|
|
1335
|
+
echo "FAIL: the merge-gate validator called '${name}' ${got}, expected ${want}" >&2
|
|
1336
|
+
fi
|
|
1337
|
+
}
|
|
1338
|
+
|
|
1339
|
+
# A report the agent would produce on a change it reviewed and found nothing wrong with.
|
|
1340
|
+
gate_clean='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1341
|
+
gate_push='{"type":"push_to_pull_request_branch","pr_number":9}'
|
|
1342
|
+
gate_items() { printf '{"items":[%s]}' "$1"; }
|
|
1343
|
+
# The agent writes one of two words and a fenced JSON block. Everything else about the
|
|
1344
|
+
# outcome is computed from that block and from the measured facts passed as arguments.
|
|
1345
|
+
gate_comment() {
|
|
1346
|
+
printf '{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** %s\\n\\n```json\\n%s\\n```"}' "$1" "${2:-$gate_clean}"
|
|
1347
|
+
}
|
|
1348
|
+
|
|
1349
|
+
# The measured facts decide, and a clean report cannot argue with them.
|
|
1350
|
+
gate_case "clean and low risk auto-merges" auto-merge "$(gate_items "$(gate_comment assessed)")" success low
|
|
1351
|
+
gate_case "medium risk still auto-merges when recoverable" \
|
|
1352
|
+
auto-merge "$(gate_items "$(gate_comment assessed)")" success medium
|
|
1353
|
+
gate_case "high blast radius needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success high
|
|
1354
|
+
gate_case "a protected path needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success low true
|
|
1355
|
+
gate_case "an owner path needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success low false true
|
|
1356
|
+
gate_case "a non-success CI conclusion blocks" blocked "$(gate_items "$(gate_comment assessed)")" failure low
|
|
1357
|
+
|
|
1358
|
+
# The agent's report decides the rest. This is the rule the whole redesign rests on: an
|
|
1359
|
+
# unverified finding is a warning whatever severity it claims, so a model cannot fail the gate
|
|
1360
|
+
# by asserting something it did not demonstrate, and cannot pass it by understating one it did.
|
|
1361
|
+
gate_unverified='{\"findings\":[{\"verified\":false,\"severity\":\"critical\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1362
|
+
gate_verified='{\"findings\":[{\"verified\":true,\"severity\":\"critical\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1363
|
+
gate_verified_low='{\"findings\":[{\"verified\":true,\"severity\":\"medium\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1364
|
+
gate_case "an unverified critical finding does not block" \
|
|
1365
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_unverified")")" success low
|
|
1366
|
+
gate_case "a verified critical finding blocks" blocked "$(gate_items "$(gate_comment assessed "$gate_verified")")" success low
|
|
1367
|
+
gate_case "a verified medium finding does not block" \
|
|
1368
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_verified_low")")" success low
|
|
1369
|
+
|
|
1370
|
+
gate_fragile='{\"findings\":[],\"recoverability\":\"low\",\"recoverabilitySignals\":[\"rewrites the stored rows in place\"],\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1371
|
+
gate_bare_low='{\"findings\":[],\"recoverability\":\"low\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1372
|
+
gate_unmet='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":false,\"confidence\":0.95}'
|
|
1373
|
+
gate_unsure='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.4}'
|
|
1374
|
+
gate_case "medium risk that cannot be undone needs a person" \
|
|
1375
|
+
human-review "$(gate_items "$(gate_comment assessed "$gate_fragile")")" success medium
|
|
1376
|
+
gate_case "low risk that cannot be undone still auto-merges" \
|
|
1377
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_fragile")")" success low
|
|
1378
|
+
# An unevidenced "low" is the old category escalation wearing a new name, so it is held to the
|
|
1379
|
+
# same standard as a finding: name what cannot be undone, or it does not change the outcome.
|
|
1380
|
+
gate_case "a low rating that names nothing is read as medium" \
|
|
1381
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_bare_low")")" success medium
|
|
1382
|
+
gate_case "an unmet acceptance criterion needs a person" \
|
|
1383
|
+
human-review "$(gate_items "$(gate_comment assessed "$gate_unmet")")" success low
|
|
1384
|
+
gate_case "confidence below the threshold needs a person" \
|
|
1385
|
+
human-review "$(gate_items "$(gate_comment assessed "$gate_unsure")")" success low
|
|
1386
|
+
|
|
1387
|
+
# The agent may raise the measured blast radius when it sees something the path rules could
|
|
1388
|
+
# not. It may never lower it, which is the only direction that can turn a person's review into
|
|
1389
|
+
# a machine merge.
|
|
1390
|
+
gate_raise='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"high\",\"reason\":\"new authorization decision point\"}}'
|
|
1391
|
+
gate_lower='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"low\"}}'
|
|
1392
|
+
gate_case "the agent can raise the blast radius" \
|
|
1393
|
+
owner-review "$(gate_items "$(gate_comment assessed "$gate_raise")")" success low
|
|
1394
|
+
gate_case "the agent cannot lower the blast radius" \
|
|
1395
|
+
owner-review "$(gate_items "$(gate_comment assessed "$gate_lower")")" success high
|
|
1396
|
+
|
|
1397
|
+
# remediated used to require conclusion == "failure", which threw correct work away. The prompt
|
|
1398
|
+
# tells the agent to merge main in, verify and push when CI is green but the pull request
|
|
1399
|
+
# conflicts. That is a real and common state: a conflicting pull request has no merge ref, so
|
|
1400
|
+
# GitHub can never run CI on that head, and the belt falls back to the last verdict on the
|
|
1401
|
+
# branch, which is usually success. The agent did the job, the validator called it invalid,
|
|
1402
|
+
# conclude was skipped, and because this worker stages its outputs the resolved merge commit
|
|
1403
|
+
# was discarded. The belt then dispatched again on the same verdict, up to six times, each a
|
|
1404
|
+
# full run on the single-slot merge belt.
|
|
1405
|
+
gate_case "remediated with one push, CI green" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" success low
|
|
1406
|
+
gate_case "remediated with one push, CI red" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" failure low
|
|
1407
|
+
gate_case "remediated with no push" invalid "$(gate_items "$(gate_comment remediated)")" failure low
|
|
1408
|
+
gate_case "remediated with two pushes" invalid "$(gate_items "$(gate_comment remediated),${gate_push},${gate_push}")" failure low
|
|
1409
|
+
gate_case "an assessment carrying a push" invalid "$(gate_items "$(gate_comment assessed),${gate_push}")" success low
|
|
1410
|
+
|
|
1411
|
+
# Output from a worker version that predates the disposition table. Applying its vocabulary
|
|
1412
|
+
# would merge on a word this validator no longer means the same thing by.
|
|
1413
|
+
gate_case "the old merge vocabulary is refused" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** merge"}]}' success low
|
|
1414
|
+
gate_case "the old review vocabulary is refused" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** review"}]}' success low
|
|
1415
|
+
|
|
1416
|
+
# Nothing malformed may fall through to a merge. Each of these parks the pull request instead.
|
|
1417
|
+
gate_case "a report aimed at another issue" invalid '{"items":[{"type":"add_comment","item_number":99,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed\\n```json\\n{}\\n```"}]}' success low
|
|
1418
|
+
gate_case "a verdict with no json block" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed"}]}' success low
|
|
1419
|
+
gate_case "a json block that does not parse" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed\\n```json\\n{nope}\\n```"}]}' success low
|
|
1420
|
+
gate_case "no verdict in the output" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"just a note"}]}' success low
|
|
1421
|
+
# Adversarial shapes. Every one of these read as the permissive value at some point, and each
|
|
1422
|
+
# is a near miss rather than nonsense: the report the agent meant to send, with one field
|
|
1423
|
+
# typed the way a model types it when it is being loose. A merge gate that reads `"true"` as
|
|
1424
|
+
# true merges on a string.
|
|
1425
|
+
gate_near_miss() {
|
|
1426
|
+
local name="$1" want="$2" report="$3"
|
|
1427
|
+
gate_case "$name" "$want" "$(gate_items "$(gate_comment assessed "$report")")" success low
|
|
1428
|
+
}
|
|
1429
|
+
gate_near_miss "verified as the string true" invalid '{\"findings\":[{\"verified\":\"true\",\"severity\":\"critical\"}],\"confidence\":0.95}'
|
|
1430
|
+
gate_near_miss "verified as the number one" invalid '{\"findings\":[{\"verified\":1,\"severity\":\"critical\"}],\"confidence\":0.95}'
|
|
1431
|
+
gate_near_miss "a severity outside the scale" invalid '{\"findings\":[{\"verified\":true,\"severity\":\"blocker\"}],\"confidence\":0.95}'
|
|
1432
|
+
gate_near_miss "a finding with no severity" invalid '{\"findings\":[{\"verified\":true}],\"confidence\":0.95}'
|
|
1433
|
+
gate_near_miss "a severity in capitals" blocked '{\"findings\":[{\"verified\":true,\"severity\":\"CRITICAL\"}],\"confidence\":0.95}'
|
|
1434
|
+
gate_near_miss "acceptanceCriteriaMet as a string" invalid '{\"findings\":[],\"acceptanceCriteriaMet\":\"false\",\"confidence\":0.95}'
|
|
1435
|
+
gate_near_miss "confidence as a word" invalid '{\"findings\":[],\"confidence\":\"high\"}'
|
|
1436
|
+
gate_near_miss "findings as a string" invalid '{\"findings\":\"none\",\"confidence\":0.95}'
|
|
1437
|
+
gate_near_miss "a recoverability outside the scale" invalid '{\"findings\":[],\"recoverability\":\"none\",\"confidence\":0.95}'
|
|
1438
|
+
gate_near_miss "a raise to an unknown level" invalid '{\"findings\":[],\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"critical\"}}'
|
|
1439
|
+
gate_near_miss "a raise in capitals is honoured" owner-review '{\"findings\":[],\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"HIGH\"}}'
|
|
1440
|
+
gate_near_miss "a report that is not an object" invalid '\"just a string\"'
|
|
1441
|
+
|
|
1442
|
+
# The prompt puts the report last and the prose above it routinely quotes json from the diff
|
|
1443
|
+
# under review. Reading the first fence handed the decision to whatever the agent quoted, and
|
|
1444
|
+
# PROTECTED_PATHS itself names package.json and global.json, so the reviewed diff is often
|
|
1445
|
+
# json. The decoy here claims everything is fine; the real report blocks.
|
|
1446
|
+
# Built with jq rather than hand-escaped: this body has two fenced blocks, each containing
|
|
1447
|
+
# quoted json, inside a json string. Hand-escaping it is how a test ends up asserting on a
|
|
1448
|
+
# fixture that does not parse.
|
|
1449
|
+
gate_decoy=$(jq -nc --arg body "$(printf '%s\n' '<!-- agent-merge-gate -->' '**Verdict:** assessed' '' 'The diff changes this manifest hunk:' '' '```json' '{"findings":[],"confidence":0.95}' '```' '' 'Report:' '' '```json' '{"findings":[{"verified":true,"severity":"critical"}],"confidence":0.95}' '```')" \
|
|
1450
|
+
'{items:[{type:"add_comment",item_number:7,body:$body}]}')
|
|
1451
|
+
gate_case "the last json fence is the report, not the first" blocked "$gate_decoy" success low
|
|
1452
|
+
|
|
1453
|
+
# Verdict and report used to be selected independently, and each took the first it found, so a
|
|
1454
|
+
# second comment reporting a verified critical finding was discarded and a comment with no
|
|
1455
|
+
# verdict could supply the report for a verdict written in another.
|
|
1456
|
+
gate_case "two comments carrying a verdict" \
|
|
1457
|
+
invalid "$(gate_items "$(gate_comment assessed),$(gate_comment assessed "$gate_verified")")" success low
|
|
1458
|
+
|
|
1459
|
+
# An empty measured fact is a job that did not report, not a low-risk pull request. `${4:-low}`
|
|
1460
|
+
# substituted the default for an empty argument, so a skipped protected_changes read as
|
|
1461
|
+
# "low, nothing protected" and merged.
|
|
1462
|
+
gate_unmeasured=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success "" "" "" 0.8 2>&1 || true)
|
|
1463
|
+
printf '%s' "$(gate_items "$(gate_comment assessed)")" > "$gate_fixture"
|
|
1464
|
+
gate_unmeasured=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success "" "" "" 0.8 2>&1 || true)
|
|
1465
|
+
if [ "$gate_unmeasured" != invalid ]; then
|
|
1466
|
+
VALIDATOR_OK=0
|
|
1467
|
+
echo "FAIL: an unmeasured blast radius produced '${gate_unmeasured}', expected invalid" >&2
|
|
1468
|
+
fi
|
|
1469
|
+
gate_half=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success low "" "" 0.8 2>&1 || true)
|
|
1470
|
+
if [ "$gate_half" != human-review ]; then
|
|
1471
|
+
VALIDATOR_OK=0
|
|
1472
|
+
echo "FAIL: an unmeasured protected-path fact produced '${gate_half}', expected human-review" >&2
|
|
1473
|
+
fi
|
|
1474
|
+
|
|
1475
|
+
gate_case "an empty item list" invalid '{"items":[]}' success low
|
|
1476
|
+
gate_case "output that is not an item list" invalid '{"nope":true}' success low
|
|
1477
|
+
|
|
1478
|
+
rm -f "$gate_fixture"
|
|
1479
|
+
if [ "$VALIDATOR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1480
|
+
fi
|
|
1481
|
+
|
|
1121
1482
|
echo "── Merge gate park ───────────────────────────────────────────────────────"
|
|
1122
1483
|
|
|
1123
1484
|
# A gate verdict parks the code it was given on. Both dispatch paths -- detect-pr-conflicts and
|
|
@@ -1392,6 +1753,267 @@ else
|
|
|
1392
1753
|
printf '%s\n' "$password_hits" >&2
|
|
1393
1754
|
fi
|
|
1394
1755
|
|
|
1756
|
+
echo "── Stale dispatch and empty runs ─────────────────────────────────────────"
|
|
1757
|
+
|
|
1758
|
+
# A route dispatched while an issue was open must not execute after it is closed. The classifier
|
|
1759
|
+
# refuses a closed issue, but it can only read `github.event.issue.state`: the state when the
|
|
1760
|
+
# event fired, and absent altogether on a workflow_dispatch. This fleet queues for a runner for
|
|
1761
|
+
# ten minutes and more. Numa #659 was closed one second after a comment dispatched refine; the run
|
|
1762
|
+
# reached `reserve` thirteen minutes later, took the reservation, and refined a closed issue for
|
|
1763
|
+
# thirty-eight minutes, ending by labelling it `refined`, `implement` and `sp-5`. Every job
|
|
1764
|
+
# reported success. Asserted per worker because the gate is three lines in each and none of them
|
|
1765
|
+
# failing produces a red run.
|
|
1766
|
+
STALE_DISPATCH_OK=1
|
|
1767
|
+
for route in refine implement triage apply-review; do
|
|
1768
|
+
worker_md="${WORKFLOWS_DIR}/agent-${route}.md"
|
|
1769
|
+
[ -f "$worker_md" ] || continue
|
|
1770
|
+
if ! grep -q '^ still_open:' "$worker_md"; then
|
|
1771
|
+
STALE_DISPATCH_OK=0
|
|
1772
|
+
echo "FAIL: agent-${route} has no still_open job; a route dispatched before the issue closed would run on it anyway" >&2
|
|
1773
|
+
continue
|
|
1774
|
+
fi
|
|
1775
|
+
# The reservation must be gated, or bot-working lands on a closed issue, and the agent must be
|
|
1776
|
+
# gated through the worker's own `if:`. Two distinct call sites.
|
|
1777
|
+
if [ "$(count -cF "needs.still_open.outputs.open == 'true'" "$worker_md")" -lt 2 ]; then
|
|
1778
|
+
STALE_DISPATCH_OK=0
|
|
1779
|
+
echo "FAIL: agent-${route} does not gate both the reservation and the agent on still_open" >&2
|
|
1780
|
+
fi
|
|
1781
|
+
# The gate job must carry no `needs:` of its own. gh-aw hoists exactly those custom jobs into
|
|
1782
|
+
# the activation job's dependencies, which is what lets the top-level `if:` read their outputs;
|
|
1783
|
+
# a gate job that gained a dependency would stop being hoisted and the clause would silently
|
|
1784
|
+
# evaluate to empty, which is always true. Asserted on the job block, not on the file, because
|
|
1785
|
+
# `needs: [still_open]` also appears on the reserve job and a file-wide search for it passed
|
|
1786
|
+
# for two workers that never declared anything.
|
|
1787
|
+
# activation must be given the dependency, or the top-level `if:` reads an empty value and the
|
|
1788
|
+
# clause is false: the agent never runs at all. gh-aw does not hoist this job, and the entry
|
|
1789
|
+
# lives at two spaces under `on:`, which is where gh-aw reads activation's dependency list.
|
|
1790
|
+
# Matched inside the `on:` block, because the same text appears on the reserve job and a
|
|
1791
|
+
# file-wide search for it passed for two workers that had declared nothing.
|
|
1792
|
+
on_block=$(awk '/^on:/{f=1; next} f && /^[a-z][a-z-]*:/{exit} f' "$worker_md")
|
|
1793
|
+
if ! printf '%s
|
|
1794
|
+
' "$on_block" | grep -qE '^ needs: \[.*still_open'; then
|
|
1795
|
+
STALE_DISPATCH_OK=0
|
|
1796
|
+
echo "FAIL: agent-${route} does not list still_open under on.needs, so activation reads an empty value and the agent never runs" >&2
|
|
1797
|
+
fi
|
|
1798
|
+
gate_block=$(awk '/^ still_open:/{f=1; next} f && /^ [a-z_]+:/{exit} f' "$worker_md")
|
|
1799
|
+
if printf '%s
|
|
1800
|
+
' "$gate_block" | grep -qE '^ needs:'; then
|
|
1801
|
+
STALE_DISPATCH_OK=0
|
|
1802
|
+
echo "FAIL: agent-${route}'s still_open job declares needs:, so gh-aw will not hoist it and the top-level if: reads an empty value" >&2
|
|
1803
|
+
fi
|
|
1804
|
+
done
|
|
1805
|
+
if [ "$STALE_DISPATCH_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1806
|
+
|
|
1807
|
+
# An audit that emitted neither a report nor a noop used to end green: `conclude` requires a
|
|
1808
|
+
# processed item, and a skipped job is not a failure. A whole agent run produced nothing and said
|
|
1809
|
+
# nothing. Both halves of the condition are asserted: the count alone would fire on a legitimate
|
|
1810
|
+
# `noop` if noop does not increment it, and the noop_message alone would miss the empty run.
|
|
1811
|
+
if worker_installed audit; then
|
|
1812
|
+
AUDIT_EMPTY_OK=1
|
|
1813
|
+
AUDIT_WORKER_MD="${WORKFLOWS_DIR}/agent-audit.md"
|
|
1814
|
+
grep -q '^ empty_run:' "$AUDIT_WORKER_MD" ||
|
|
1815
|
+
{ AUDIT_EMPTY_OK=0; echo "FAIL: the audit has no empty_run job; a run that files nothing reports success" >&2; }
|
|
1816
|
+
grep -qF "process_safe_outputs_processed_count == '0'" "$AUDIT_WORKER_MD" ||
|
|
1817
|
+
{ AUDIT_EMPTY_OK=0; echo "FAIL: the audit's empty_run does not test the processed count" >&2; }
|
|
1818
|
+
# empty_run must not depend on gh-aw's `conclusion` job. That job needs every custom job in the
|
|
1819
|
+
# worker, so naming it is a cycle and the whole workflow fails to compile -- which is how the
|
|
1820
|
+
# first attempt at this was caught. It also means `noop_message`, the one output that would say
|
|
1821
|
+
# a clean audit deliberately filed nothing, cannot be read from here.
|
|
1822
|
+
if grep -qE '^ needs: \[.*conclusion' "$AUDIT_WORKER_MD"; then
|
|
1823
|
+
AUDIT_EMPTY_OK=0
|
|
1824
|
+
echo "FAIL: the audit's empty_run depends on the conclusion job, which is a dependency cycle and will not compile" >&2
|
|
1825
|
+
fi
|
|
1826
|
+
if [ "$AUDIT_EMPTY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1827
|
+
fi
|
|
1828
|
+
|
|
1829
|
+
echo "── Runner pools ──────────────────────────────────────────────────────────"
|
|
1830
|
+
|
|
1831
|
+
# Where every job runs, stated once and asserted, because GitHub gives a wrong pool no error: a
|
|
1832
|
+
# job addressed to a label no runner carries simply queues, and a job addressed to the wrong pool
|
|
1833
|
+
# runs in the wrong place. The pool was a preserved consumer value until 0.19.0, and that is how
|
|
1834
|
+
# one repository came to run triage, refine, implement, apply-review and audit on
|
|
1835
|
+
# RunnerLandingZone while merge-gate and release stayed on agents-arc: the override reached the
|
|
1836
|
+
# workers installed at the time and never the ones added later. Nothing reported the split.
|
|
1837
|
+
#
|
|
1838
|
+
# agent jobs of every worker except release -> agents-arc (the Azure fleet in
|
|
1839
|
+
# agentrunner-pro-rg-01,
|
|
1840
|
+
# runner group `agentic`)
|
|
1841
|
+
# agent-release.md and work-router.yml -> RunnerLandingZone
|
|
1842
|
+
# ubuntu-latest -> GitHub's own runner, chosen by nobody, left
|
|
1843
|
+
# wherever the package already has it
|
|
1844
|
+
AGENT_POOL="agents-arc"
|
|
1845
|
+
PLUMBING_POOL="RunnerLandingZone"
|
|
1846
|
+
POOL_OK=1
|
|
1847
|
+
|
|
1848
|
+
# `|| true` on both greps: a file naming no pool at all, or only ubuntu-latest, makes grep exit 1,
|
|
1849
|
+
# and under `set -o pipefail` that aborts the whole matrix instead of failing this one assertion.
|
|
1850
|
+
# A router mutated to ubuntu-latest everywhere did exactly that: the suite died without printing,
|
|
1851
|
+
# so the mutation looked caught when in fact nothing had been checked.
|
|
1852
|
+
pools_named() {
|
|
1853
|
+
{ grep -hoE '^[[:space:]]*runs-on(-slim)?: [^[:space:]]+' "$1" 2>/dev/null || true; } \
|
|
1854
|
+
| sed 's/.*: //' | { grep -v '^ubuntu-latest$' || true; } | sort -u
|
|
1855
|
+
}
|
|
1856
|
+
|
|
1857
|
+
for worker_md in "${WORKFLOWS_DIR}"/agent-*.md; do
|
|
1858
|
+
[ -f "$worker_md" ] || continue
|
|
1859
|
+
worker_name="$(basename "$worker_md" .md)"
|
|
1860
|
+
want="$AGENT_POOL"
|
|
1861
|
+
[ "$worker_name" = "agent-release" ] && want="$PLUMBING_POOL"
|
|
1862
|
+
got="$(pools_named "$worker_md" | tr '\n' ' ' | sed 's/ $//')"
|
|
1863
|
+
if [ "$got" != "$want" ]; then
|
|
1864
|
+
POOL_OK=0
|
|
1865
|
+
echo "FAIL: ${worker_name} names runner pool(s) '${got}' but must name only '${want}'" >&2
|
|
1866
|
+
fi
|
|
1867
|
+
done
|
|
1868
|
+
|
|
1869
|
+
router_pools="$(pools_named "$ROUTER_YML" | tr '\n' ' ' | sed 's/ $//')"
|
|
1870
|
+
if [ "$router_pools" != "$PLUMBING_POOL" ]; then
|
|
1871
|
+
POOL_OK=0
|
|
1872
|
+
echo "FAIL: the router names runner pool(s) '${router_pools}' but must name only '${PLUMBING_POOL}'" >&2
|
|
1873
|
+
fi
|
|
1874
|
+
|
|
1875
|
+
# And the pool must not be reintroduced as a per-consumer value: that is the mechanism that let
|
|
1876
|
+
# the split happen, and it left no trace anywhere.
|
|
1877
|
+
if [ -d "${HERE}/../../../cli/src" ] && grep -rqE 'preserveRunnerPool|runnerPools' "${HERE}/../../../cli/src" 2>/dev/null; then
|
|
1878
|
+
POOL_OK=0
|
|
1879
|
+
echo "FAIL: the installer preserves a consumer's runner pool again; the pool is the package's to set" >&2
|
|
1880
|
+
fi
|
|
1881
|
+
|
|
1882
|
+
if [ "$POOL_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1883
|
+
|
|
1884
|
+
echo "── Worker input wiring ───────────────────────────────────────────────────"
|
|
1885
|
+
|
|
1886
|
+
# A worker that declares a workflow_call input and never reads it is the shape of the worst
|
|
1887
|
+
# outage this pipeline has had. The router resolved the CI run ID, passed it as `ci-run-id`, and
|
|
1888
|
+
# the merge gate declared the input and then called identify-gate-subject without it -- so
|
|
1889
|
+
# `RUN_ID` was always empty, the evidence step wrote `failed-jobs.json` as `[]`, and the agent,
|
|
1890
|
+
# holding no failing job name and no logs, chose `review` every time CI went red. Pliny-Bot #129,
|
|
1891
|
+
# #130 and #131 all ended up waiting on a human with CI legitimately red and remediable, and the
|
|
1892
|
+
# housekeeping digest reported it as work needing a person. Nothing went red: every job succeeded
|
|
1893
|
+
# at doing nothing. This asserts the class, because reading each worker by hand is how it was
|
|
1894
|
+
# missed for as long as it was.
|
|
1895
|
+
#
|
|
1896
|
+
# `agent-audit:trigger-kind` is exempt and stays listed rather than deleted: it is a
|
|
1897
|
+
# workflow_dispatch choice a person picks on the router, threaded through for symmetry, with no
|
|
1898
|
+
# behaviour attached at the far end. Removing it would take a dispatch option away from people,
|
|
1899
|
+
# so it is recorded as known-inert instead. Any *other* unread input fails.
|
|
1900
|
+
DEAD_INPUT_EXEMPT="agent-audit:trigger-kind"
|
|
1901
|
+
INPUT_WIRING_OK=1
|
|
1902
|
+
for worker_md in "${WORKFLOWS_DIR}"/agent-*.md; do
|
|
1903
|
+
[ -f "$worker_md" ] || continue
|
|
1904
|
+
worker_name="$(basename "$worker_md" .md)"
|
|
1905
|
+
# The declared inputs: the `inputs:` mapping under `on: workflow_call:`, whose keys sit at six
|
|
1906
|
+
# spaces. Stop at the first line indented less than that which is not blank.
|
|
1907
|
+
declared=$(awk '
|
|
1908
|
+
/^on:/ { in_on = 1; next }
|
|
1909
|
+
in_on && /^[a-z#]/ { exit }
|
|
1910
|
+
in_on && /^ workflow_call:/ { in_wc = 1; next }
|
|
1911
|
+
in_wc && /^ inputs:/ { in_inputs = 1; next }
|
|
1912
|
+
in_inputs && /^ [a-z]/ { in_inputs = 0 }
|
|
1913
|
+
in_inputs && /^ [a-z0-9_-]+:[[:space:]]*$/ {
|
|
1914
|
+
gsub(/[ :]/, "", $0); print $0
|
|
1915
|
+
}
|
|
1916
|
+
' "$worker_md")
|
|
1917
|
+
for input_name in $declared; do
|
|
1918
|
+
case "${worker_name}:${input_name}" in
|
|
1919
|
+
"$DEAD_INPUT_EXEMPT") continue ;;
|
|
1920
|
+
esac
|
|
1921
|
+
if [ "$(count -cE "inputs\.${input_name}([^a-zA-Z0-9_-]|\$)" "$worker_md")" -eq 0 ]; then
|
|
1922
|
+
INPUT_WIRING_OK=0
|
|
1923
|
+
echo "FAIL: ${worker_name} declares the input '${input_name}' and never reads it; the router's value is discarded and the job succeeds at doing nothing" >&2
|
|
1924
|
+
fi
|
|
1925
|
+
done
|
|
1926
|
+
done
|
|
1927
|
+
if [ "$INPUT_WIRING_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1928
|
+
|
|
1929
|
+
# The belt dispatches one gate per tick and stops. The gate's concurrency group holds a single
|
|
1930
|
+
# pending run, so GitHub cancels every earlier pending dispatch in it -- `cancel-in-progress:
|
|
1931
|
+
# false` only protects a run that has already started. A loop without this `break` dispatched one
|
|
1932
|
+
# gate per eligible pull request, ran the last, and discarded the rest with no comment, no
|
|
1933
|
+
# recorded attempt, and nothing on the pull request to show it had been skipped. Asserted because
|
|
1934
|
+
# removing the break produces no error anywhere: the extra dispatches all return 204.
|
|
1935
|
+
BELT_ONE_PER_TICK_OK=1
|
|
1936
|
+
belt_dispatch_block=$(awk '
|
|
1937
|
+
/Dispatching Merge Gate for PR/ { found = 1 }
|
|
1938
|
+
found { print }
|
|
1939
|
+
found && /^ *done$/ { exit }
|
|
1940
|
+
' "$ROUTER_YML")
|
|
1941
|
+
if [ -z "$belt_dispatch_block" ]; then
|
|
1942
|
+
BELT_ONE_PER_TICK_OK=0
|
|
1943
|
+
echo "FAIL: the merge belt's gate dispatch could not be located, so its one-per-tick guard cannot be checked" >&2
|
|
1944
|
+
elif ! printf '%s\n' "$belt_dispatch_block" | grep -qE '^ *break$'; then
|
|
1945
|
+
BELT_ONE_PER_TICK_OK=0
|
|
1946
|
+
echo "FAIL: the merge belt dispatches a gate per eligible pull request and never breaks; only the last stays pending and the rest are cancelled in silence" >&2
|
|
1947
|
+
fi
|
|
1948
|
+
# And the group it relies on must still be the serialising one.
|
|
1949
|
+
# Anchored to the whole line: a substring search matched `merge-belt-renamed` and stayed green
|
|
1950
|
+
# through a mutation that removed the serialisation the break depends on.
|
|
1951
|
+
grep -qE '^ *group: merge-belt *$' "$ROUTER_YML" ||
|
|
1952
|
+
{ BELT_ONE_PER_TICK_OK=0; echo "FAIL: the merge gate is no longer serialised on the merge-belt concurrency group" >&2; }
|
|
1953
|
+
if [ "$BELT_ONE_PER_TICK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1954
|
+
|
|
1955
|
+
# The specific wiring that broke, asserted end to end: the router resolves the run ID, the gate
|
|
1956
|
+
# forwards it, and the action seeds from it rather than only from its own lookup.
|
|
1957
|
+
if worker_installed merge-gate; then
|
|
1958
|
+
GATE_RUN_ID_OK=1
|
|
1959
|
+
grep -Fq 'ci-run-id: ${{ needs.classify.outputs.ci-run-id }}' "$ROUTER_YML" ||
|
|
1960
|
+
{ GATE_RUN_ID_OK=0; echo "FAIL: the router no longer passes ci-run-id to the merge gate" >&2; }
|
|
1961
|
+
grep -Fq 'ci-run-id: ${{ inputs.ci-run-id }}' "$MERGE_GATE_WORKER_MD" ||
|
|
1962
|
+
{ GATE_RUN_ID_OK=0; echo "FAIL: the merge gate does not forward ci-run-id to identify-gate-subject, so it has no CI failure evidence" >&2; }
|
|
1963
|
+
GATE_SUBJECT_ACTION="${HERE}/../identify-gate-subject/action.yml"
|
|
1964
|
+
if [ -f "$GATE_SUBJECT_ACTION" ]; then
|
|
1965
|
+
grep -Fq 'ci_run_id="$CI_RUN_ID"' "$GATE_SUBJECT_ACTION" ||
|
|
1966
|
+
{ GATE_RUN_ID_OK=0; echo "FAIL: identify-gate-subject does not seed the run ID from its input, so a caller that knows it is ignored" >&2; }
|
|
1967
|
+
# The regression: resolving the ID only when the conclusion is missing.
|
|
1968
|
+
# Anchored to the start of the line: the repaired code keeps an `elif [ -z "$ci_conclusion" ]`
|
|
1969
|
+
# branch for the caller that genuinely has no verdict, and a fixed-string search matched that
|
|
1970
|
+
# `elif` as a substring -- the same way this file's own explanatory comments have tripped
|
|
1971
|
+
# three earlier assertions.
|
|
1972
|
+
if [ "$(count -cE '^ *if \[ -z "\$ci_conclusion" \]; then' "$GATE_SUBJECT_ACTION")" -ne 0 ]; then
|
|
1973
|
+
GATE_RUN_ID_OK=0
|
|
1974
|
+
echo "FAIL: identify-gate-subject resolves the CI run only when the conclusion is unknown; a router that passes both gets an empty run ID" >&2
|
|
1975
|
+
fi
|
|
1976
|
+
fi
|
|
1977
|
+
# Repeating a report the validator refused reproduces it, and every repeat is a fresh agent run
|
|
1978
|
+
# holding the repo-wide merge-belt slot. The worker gives that failure a smaller budget than a
|
|
1979
|
+
# crash, and -- the part that actually saves the slot -- records it as a verdict, because the belt
|
|
1980
|
+
# bounds its own retries by counting attempt comments and would otherwise dispatch to the cap
|
|
1981
|
+
# whatever the worker decided.
|
|
1982
|
+
if worker_installed merge-gate; then
|
|
1983
|
+
UNUSABLE_OK=1
|
|
1984
|
+
unusable_cap="$(sed -n 's/^ PARK_AT_UNUSABLE_OUTPUT: "\([0-9]*\)"$/\1/p' "$MERGE_GATE_WORKER_MD" | head -1)"
|
|
1985
|
+
machine_cap="$(sed -n 's/^ PARK_AT_ATTEMPT: "\([0-9]*\)"$/\1/p' "$MERGE_GATE_WORKER_MD" | head -1)"
|
|
1986
|
+
if [ -z "$unusable_cap" ] || [ -z "$machine_cap" ] || [ "$unusable_cap" -ge "$machine_cap" ]; then
|
|
1987
|
+
UNUSABLE_OK=0
|
|
1988
|
+
echo "FAIL: an unusable report must get a smaller budget than a crash (unusable='${unusable_cap:-unset}', machine='${machine_cap:-unset}')" >&2
|
|
1989
|
+
fi
|
|
1990
|
+
# The budget is decided once, not restated per step. Four `if:` expressions repeating the same
|
|
1991
|
+
# pair of conditions is the shape the protected-files hold already got wrong.
|
|
1992
|
+
if [ "$(count -cE "^ if: steps\.budget\.outputs\.park" "$MERGE_GATE_WORKER_MD")" -lt 3 ]; then
|
|
1993
|
+
UNUSABLE_OK=0
|
|
1994
|
+
echo "FAIL: the incomplete job must read one computed budget decision, not re-derive it" >&2
|
|
1995
|
+
fi
|
|
1996
|
+
# The park has to be a verdict or the belt keeps dispatching: a comment carrying the gate
|
|
1997
|
+
# marker AND a Verdict line is what detect-pr-conflicts and the reconcile belt both park on.
|
|
1998
|
+
if ! grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
|
|
1999
|
+
grep -q '\${{ env.GATE_MARKER }}' ||
|
|
2000
|
+
! grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
|
|
2001
|
+
grep -q '\*\*Verdict:\*\* human-review'; then
|
|
2002
|
+
UNUSABLE_OK=0
|
|
2003
|
+
echo "FAIL: the unusable-report park must carry the gate marker and a Verdict line, or the belt dispatches it again" >&2
|
|
2004
|
+
fi
|
|
2005
|
+
# And it must not also count as an attempt, or one park is recorded twice.
|
|
2006
|
+
if grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
|
|
2007
|
+
grep -q '\${{ env.ATTEMPT_MARKER }}'; then
|
|
2008
|
+
UNUSABLE_OK=0
|
|
2009
|
+
echo "FAIL: the unusable-report park carries the attempt marker as well; it is a decision, not an attempt" >&2
|
|
2010
|
+
fi
|
|
2011
|
+
if [ "$UNUSABLE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
2012
|
+
fi
|
|
2013
|
+
|
|
2014
|
+
if [ "$GATE_RUN_ID_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
2015
|
+
fi
|
|
2016
|
+
|
|
1395
2017
|
echo
|
|
1396
2018
|
if [ "$FAIL" -eq 0 ]; then
|
|
1397
2019
|
echo "Route matrix: ${PASS} passed"
|