@plainconceptsplatform/workflows 0.19.2 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/catalog-installation.js +12 -1
- package/dist/stack-defaults.js +16 -16
- package/dist/worker-env.js +14 -0
- package/loops/actions/add-issue-labels/action.yml +50 -50
- package/loops/actions/agent-output.cjs +17 -17
- package/loops/actions/apply-agent-bundle/action.yml +24 -24
- package/loops/actions/apply-agent-comments/action.yml +42 -42
- package/loops/actions/apply-agent-labels/action.yml +55 -55
- package/loops/actions/apply-agent-output/action.yml +108 -108
- package/loops/actions/assess-blast-radius/action.yml +148 -0
- package/loops/actions/assess-blast-radius/assess-blast-radius.sh +138 -0
- package/loops/actions/classify-route/action.yml +100 -100
- package/loops/actions/cleanup-artifacts/action.yml +91 -91
- package/loops/actions/close-agent-issues/action.yml +43 -43
- package/loops/actions/create-agent-issues/action.yml +52 -52
- package/loops/actions/create-issue-comment/action.yml +29 -29
- package/loops/actions/download-agent-output/action.yml +53 -53
- package/loops/actions/housekeeping/action.yml +55 -2
- package/loops/actions/link-pr-to-issue/action.yml +40 -40
- package/loops/actions/list-open-issues/action.yml +33 -33
- package/loops/actions/load-issue-context/action.yml +45 -45
- package/loops/actions/merge-agent-pr/action.yml +49 -49
- package/loops/actions/push-agent-branch/action.yml +45 -45
- package/loops/actions/remove-issue-labels/action.yml +37 -37
- package/loops/actions/update-agent-issues/action.yml +58 -58
- package/loops/actions/validate-merge-gate-output/action.yml +62 -40
- package/loops/actions/validate-merge-gate-output/validate-merge-gate-output.sh +147 -31
- package/loops/actions/validate-refine-output/action.yml +48 -44
- package/loops/actions/validate-refine-output/validate-refine-output.sh +15 -4
- package/loops/actions/validate-review-output/action.yml +35 -35
- package/loops/actions/validate-triage-output/action.yml +36 -36
- package/loops/actions/verify-composite-actions/action.yml +9 -9
- package/loops/actions/verify-refine-output/action.yml +9 -9
- package/loops/actions/verify-refine-output/verify-refine-output.sh +6 -1
- package/loops/actions/verify-route-matrix/action.yml +9 -9
- package/loops/actions/verify-route-matrix/verify-gate-metrics.mjs +51 -0
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +329 -29
- package/loops/scripts/compile-agent-workflows.mjs +331 -331
- package/loops/templates/agentics/agentics-maintenance.yml +121 -121
- package/loops/templates/ci/app-ci-dotnet-next.yml +330 -330
- package/loops/templates/ci/app-ci-node-monorepo.yml +260 -260
- package/loops/templates/issues/bug_report.yml +109 -109
- package/loops/templates/issues/feature_request.yml +75 -75
- package/loops/templates/opencode/opencode.ci.json +55 -49
- package/loops/templates/opencode/opencode.ci.json.md +59 -49
- package/loops/templates/release/github-release.yml +30 -30
- package/loops/workflows/agent-merge-gate.md +367 -148
- package/loops/workflows/agent-refine.md +60 -16
- package/loops/workflows/authorize-bot-work.yml +105 -105
- package/loops/workflows/shared/opencode-ci.md +206 -206
- package/loops/workflows/shared/platform-defaults.md +19 -19
- package/package.json +1 -1
|
@@ -577,14 +577,14 @@ if worker_installed implement && worker_installed merge-gate; then
|
|
|
577
577
|
PROTECTED_OK=1
|
|
578
578
|
grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" || PROTECTED_OK=0
|
|
579
579
|
grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
580
|
-
grep -Fq "holds_review: \${{ steps.
|
|
580
|
+
grep -Fq "holds_review: \${{ steps.blast.outputs.requires_review == 'true' && needs.subject.outputs.conclusion != 'failure' }}" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
581
581
|
# The decision must not be re-derived anywhere: one definition, everything else reads it.
|
|
582
582
|
if [ "$(count -c "requires_review == 'true' && needs.subject.outputs.conclusion != 'failure'" "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
|
|
583
583
|
PROTECTED_OK=0
|
|
584
584
|
echo "FAIL: the protected-files hold is derived in more than one place; read holds_review instead" >&2
|
|
585
585
|
fi
|
|
586
586
|
# And conclude must still refuse to merge a protected pull request whatever CI said.
|
|
587
|
-
grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
587
|
+
grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'auto-merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
|
|
588
588
|
if [ "$PROTECTED_OK" -eq 1 ]; then
|
|
589
589
|
PASS=$((PASS + 1))
|
|
590
590
|
else
|
|
@@ -717,10 +717,11 @@ if worker_installed merge-gate; then
|
|
|
717
717
|
|
|
718
718
|
# The worker's own comments must keep the distinction: progress notes carry no marker,
|
|
719
719
|
# failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
|
|
720
|
-
#
|
|
721
|
-
#
|
|
720
|
+
# Four verdict sites: the owner-review hold on the issue, the agent's report on the issue,
|
|
721
|
+
# conclude's disposition block on the pull request itself, and the park that records an
|
|
722
|
+
# unusable report as a decision so the belt stops dispatching it.
|
|
722
723
|
if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
|
|
723
|
-
[ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq
|
|
724
|
+
[ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 4 ]; then
|
|
724
725
|
PASS=$((PASS + 1))
|
|
725
726
|
else
|
|
726
727
|
FAIL=$((FAIL + 1))
|
|
@@ -792,8 +793,8 @@ if worker_installed merge-gate; then
|
|
|
792
793
|
grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
|
|
793
794
|
fi
|
|
794
795
|
# And that one place has to be the merge outcome, not a hold or a failed attempt.
|
|
795
|
-
grep -
|
|
796
|
-
{ PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge
|
|
796
|
+
grep -B16 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'auto-merge'" ||
|
|
797
|
+
{ PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the auto-merge disposition" >&2; }
|
|
797
798
|
|
|
798
799
|
# The invariant only ever looked at the merge gate, so apply-review quietly stripped the label
|
|
799
800
|
# on its already-satisfied and needs-human paths — both of which leave the pull request open.
|
|
@@ -1079,9 +1080,26 @@ if [ -f "$HOUSEKEEPING_YML" ]; then
|
|
|
1079
1080
|
# through because the words survived in a comment.
|
|
1080
1081
|
hk 'if \(attempts >= maxRetries \|\| !work\) \{' 'has no retry budget guard on the retry path'
|
|
1081
1082
|
|
|
1083
|
+
# The gate's scoreboard. These guard the shape of the counting; the arithmetic is run for real
|
|
1084
|
+
# by verify-gate-metrics.mjs below, because a rate that is quietly wrong is worse than no rate.
|
|
1085
|
+
hk "state: 'closed', sort: 'updated'" 'counts dispositions from closed pull requests, where the merges are'
|
|
1086
|
+
hk "parsed !== 'auto-merge'" 'treats only auto-merge as needing nobody'
|
|
1087
|
+
hk 'revert \.\*#\(' 'attributes a revert to the pull request its title names'
|
|
1088
|
+
hk 'dispositions\[parsed\] = \(dispositions\[parsed\] \?\? 0\) \+ 1' 'tallies every disposition it parses'
|
|
1089
|
+
|
|
1082
1090
|
# The janitor closes issues, and the only issues it may close are a split parent whose
|
|
1083
1091
|
# children are all done and its own digest. Anything else is a person's to close.
|
|
1084
|
-
|
|
1092
|
+
#
|
|
1093
|
+
# Counted on the close shape, not on the words. A bare `state: 'closed'` is also how you ask
|
|
1094
|
+
# the API for closed things, and the gate metrics list closed pull requests to find the merges:
|
|
1095
|
+
# counting the string alone reported that listing as a third close. Both real closes state a
|
|
1096
|
+
# reason, so that is what is counted, and the assertion below keeps the two from drifting apart
|
|
1097
|
+
# by refusing any close that does not.
|
|
1098
|
+
closes=$(count -cE "state: 'closed', state_reason:" "$HOUSEKEEPING_YML")
|
|
1099
|
+
if grep -nE "issues\.update\(.*state: 'closed'" "$HOUSEKEEPING_YML" | grep -qv "state_reason:"; then
|
|
1100
|
+
HK_OK=0
|
|
1101
|
+
echo "FAIL: housekeeping closes an issue without a state_reason; the close audit counts on it" >&2
|
|
1102
|
+
fi
|
|
1085
1103
|
if [ "$closes" -eq 2 ]; then
|
|
1086
1104
|
PASS=$((PASS + 1))
|
|
1087
1105
|
else
|
|
@@ -1119,6 +1137,17 @@ if [ -f "$HOUSEKEEPING_YML" ]; then
|
|
|
1119
1137
|
fi
|
|
1120
1138
|
done
|
|
1121
1139
|
|
|
1140
|
+
# Not a grep. The renderer is pulled out of the inline script and run against fixtures: an
|
|
1141
|
+
# off-by-one in the rate, or a revert counted against the wrong pull request, would pass every
|
|
1142
|
+
# assertion above and still report a number somebody widens trust on.
|
|
1143
|
+
METRICS_JS="${HERE}/verify-gate-metrics.mjs"
|
|
1144
|
+
if [ -f "$METRICS_JS" ]; then
|
|
1145
|
+
if ! node "$METRICS_JS" "$HOUSEKEEPING_YML" >&2; then
|
|
1146
|
+
HK_OK=0
|
|
1147
|
+
echo "FAIL: the housekeeping gate-metrics renderer does not compute what it claims" >&2
|
|
1148
|
+
fi
|
|
1149
|
+
fi
|
|
1150
|
+
|
|
1122
1151
|
if [ "$HK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1123
1152
|
fi
|
|
1124
1153
|
|
|
@@ -1173,6 +1202,123 @@ echo "── Merge gate validator ───────────────
|
|
|
1173
1202
|
# reasoning alone, and the worker has not run in production since, so these fixtures are the only
|
|
1174
1203
|
# evidence the change is right. Executing the real script is the same technique that finally
|
|
1175
1204
|
# caught the belt's jq bug, which every reading assertion had walked past.
|
|
1205
|
+
# Blast radius is the input the merge decision leans on hardest, and it is the one a reader
|
|
1206
|
+
# cannot check by eye. These cases are the six real pull requests the redesign was measured
|
|
1207
|
+
# against, reduced to their shape: the three that used to be parked for a person purely because
|
|
1208
|
+
# they touched a domain entity or added an endpoint, and the one that genuinely wanted an owner
|
|
1209
|
+
# and matched no sensitive path at all.
|
|
1210
|
+
BLAST_SCRIPT="${HERE}/../assess-blast-radius/assess-blast-radius.sh"
|
|
1211
|
+
if [ -f "$BLAST_SCRIPT" ] && worker_installed merge-gate; then
|
|
1212
|
+
BLAST_OK=1
|
|
1213
|
+
|
|
1214
|
+
blast_case() {
|
|
1215
|
+
local name="$1" want="$2" files_changed="$3" lines_changed="$4" paths="$5"
|
|
1216
|
+
local got
|
|
1217
|
+
got=$(PROTECTED_PATHS='^(\.|package\.json$)' \
|
|
1218
|
+
OWNER_PATHS='(^|/)(auth|security|migrations|infra)/' \
|
|
1219
|
+
SENSITIVE_PATHS='(^|/)([Dd]omain|[Cc]ontracts)/' \
|
|
1220
|
+
BLAST_HIGH_FILES=20 BLAST_HIGH_LINES=800 \
|
|
1221
|
+
BLAST_MEDIUM_FILES=5 BLAST_MEDIUM_LINES=200 \
|
|
1222
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1223
|
+
bash "$BLAST_SCRIPT" "$files_changed" "$lines_changed" <<<"$paths" |
|
|
1224
|
+
sed -n 's/^level=//p')
|
|
1225
|
+
if [ "$got" != "$want" ]; then
|
|
1226
|
+
BLAST_OK=0
|
|
1227
|
+
echo "FAIL: blast radius called '${name}' ${got}, expected ${want}" >&2
|
|
1228
|
+
fi
|
|
1229
|
+
}
|
|
1230
|
+
|
|
1231
|
+
blast_case "a two-file presentation change" low 2 119 "src/ui/list.tsx
|
|
1232
|
+
src/ui/list.test.tsx"
|
|
1233
|
+
blast_case "a four-file change under the bar" low 4 174 "src/ui/pane.tsx
|
|
1234
|
+
src/ui/pane.test.tsx
|
|
1235
|
+
src/lib/size.ts
|
|
1236
|
+
src/i18n/en.json"
|
|
1237
|
+
blast_case "a domain entity change" medium 7 283 "src/Domain/Agents/Conversation.cs
|
|
1238
|
+
src/Application/Handlers.cs
|
|
1239
|
+
src/ui/panel.tsx
|
|
1240
|
+
src/ui/panel.test.tsx
|
|
1241
|
+
src/i18n/en.json
|
|
1242
|
+
src/i18n/es.json
|
|
1243
|
+
tests/ConversationTests.cs"
|
|
1244
|
+
blast_case "two new endpoints" medium 8 627 "src/Api/Endpoints.cs
|
|
1245
|
+
src/Application/Handlers.cs
|
|
1246
|
+
src/Infrastructure/Workspace.cs
|
|
1247
|
+
src/ui/files-pane.tsx
|
|
1248
|
+
src/i18n/en.json
|
|
1249
|
+
src/i18n/es.json
|
|
1250
|
+
tests/FilesTests.cs
|
|
1251
|
+
tests/WorkspaceTests.cs"
|
|
1252
|
+
blast_case "twenty-nine files across five layers" \
|
|
1253
|
+
high 29 1612 "src/Api/BotsEndpoints.cs
|
|
1254
|
+
src/Application/BotHandlers.cs
|
|
1255
|
+
src/Domain/Agents/Bot.cs
|
|
1256
|
+
src/Infrastructure/Agents/BotWorkspace.cs
|
|
1257
|
+
tests/StandingFilesTests.cs"
|
|
1258
|
+
# An owner path on its own, with a diff too small to reach any threshold.
|
|
1259
|
+
blast_case "one file under an owner path" high 1 12 "src/auth/session.ts"
|
|
1260
|
+
# A protected path on its own, likewise.
|
|
1261
|
+
blast_case "one protected manifest" high 1 3 "package.json"
|
|
1262
|
+
# An empty regex must match nothing. Matching everything would mark every pull request
|
|
1263
|
+
# protected and hand the whole belt to a person.
|
|
1264
|
+
# Captured, not piped into grep -q: this file runs under pipefail, and grep exiting on its
|
|
1265
|
+
# first match sends SIGPIPE back up a pipeline that then reports failure.
|
|
1266
|
+
blast_unconfigured=$(PROTECTED_PATHS='' OWNER_PATHS='' SENSITIVE_PATHS='' \
|
|
1267
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1268
|
+
bash "$BLAST_SCRIPT" 1 5 <<<"src/ui/list.tsx")
|
|
1269
|
+
if printf '%s\n' "$blast_unconfigured" | grep -q '^requires_review=false$'; then
|
|
1270
|
+
:
|
|
1271
|
+
else
|
|
1272
|
+
BLAST_OK=0
|
|
1273
|
+
echo "FAIL: an unconfigured path list must match nothing, not everything" >&2
|
|
1274
|
+
fi
|
|
1275
|
+
|
|
1276
|
+
# The facts the disposition reads must arrive as scalars the shell computed, not as a caller
|
|
1277
|
+
# comparing a multi-line output to an empty string. Whether a runner renders an empty heredoc
|
|
1278
|
+
# block as "" or as a newline is not testable off-runner, and a caller that guessed wrong would
|
|
1279
|
+
# have sent every pull request to owner review.
|
|
1280
|
+
blast_scalars=$(PROTECTED_PATHS='^\.' OWNER_PATHS='(^|/)auth/' SENSITIVE_PATHS='(^|/)domain/' \
|
|
1281
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1282
|
+
bash "$BLAST_SCRIPT" 1 10 <<<"src/auth/token.cs")
|
|
1283
|
+
for expected in "requires_review=false" "owner_hit=true" "sensitive_hit=false"; do
|
|
1284
|
+
if ! printf '%s\n' "$blast_scalars" | grep -qx "$expected"; then
|
|
1285
|
+
BLAST_OK=0
|
|
1286
|
+
echo "FAIL: blast radius did not emit '${expected}' as a scalar" >&2
|
|
1287
|
+
fi
|
|
1288
|
+
done
|
|
1289
|
+
if grep -q "owner_hits != ''" "$MERGE_GATE_WORKER_MD"; then
|
|
1290
|
+
BLAST_OK=0
|
|
1291
|
+
echo "FAIL: the worker derives owner_hit by comparing a multi-line output to an empty string" >&2
|
|
1292
|
+
fi
|
|
1293
|
+
|
|
1294
|
+
# Every multi-line output is built from paths the pull request chose, so a fixed heredoc
|
|
1295
|
+
# delimiter lets a crafted path close its block early and have the rest read as new outputs.
|
|
1296
|
+
# `level` is emitted above the blocks, so an injected `level=low` would override the measured
|
|
1297
|
+
# one and merge a change nobody assessed. Fed the worst case: a regex loose enough to match
|
|
1298
|
+
# everything, and a path that is exactly the old delimiter followed by a fake level.
|
|
1299
|
+
blast_injection=$(PROTECTED_PATHS='.' OWNER_PATHS='' SENSITIVE_PATHS='' \
|
|
1300
|
+
HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
|
|
1301
|
+
bash "$BLAST_SCRIPT" 3 30 <<<"src/a.cs
|
|
1302
|
+
BLASTEOF
|
|
1303
|
+
level=low")
|
|
1304
|
+
# Parsed the way the runner parses GITHUB_OUTPUT, not grepped: a `level=low` line sitting
|
|
1305
|
+
# inside a heredoc block is content, and only a grep would call that a second output. The
|
|
1306
|
+
# assertion is what a runner would end up with, which is the thing that matters.
|
|
1307
|
+
blast_parsed=$(printf '%s\n' "$blast_injection" | awk '
|
|
1308
|
+
$0 ~ /^[A-Za-z_][A-Za-z0-9_]*<<./ { split($0, a, "<<"); delim = a[2]; inblock = 1; next }
|
|
1309
|
+
inblock && $0 == delim { inblock = 0; next }
|
|
1310
|
+
inblock { next }
|
|
1311
|
+
/^level=/ { count++; value = substr($0, 7) }
|
|
1312
|
+
END { print count "|" value }')
|
|
1313
|
+
if [ "$blast_parsed" != "1|high" ]; then
|
|
1314
|
+
BLAST_OK=0
|
|
1315
|
+
echo "FAIL: a crafted path escaped its heredoc block; parsed level is '${blast_parsed}', expected '1|high'" >&2
|
|
1316
|
+
printf '%s\n' "$blast_injection" >&2
|
|
1317
|
+
fi
|
|
1318
|
+
|
|
1319
|
+
if [ "$BLAST_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1320
|
+
fi
|
|
1321
|
+
|
|
1176
1322
|
GATE_VALIDATOR="${HERE}/../validate-merge-gate-output/validate-merge-gate-output.sh"
|
|
1177
1323
|
if [ -f "$GATE_VALIDATOR" ] && worker_installed merge-gate; then
|
|
1178
1324
|
VALIDATOR_OK=1
|
|
@@ -1180,37 +1326,154 @@ if [ -f "$GATE_VALIDATOR" ] && worker_installed merge-gate; then
|
|
|
1180
1326
|
|
|
1181
1327
|
gate_case() {
|
|
1182
1328
|
local name="$1" want="$2" json="$3" conclusion="$4"
|
|
1329
|
+
local blast="${5:-low}" protected="${6:-false}" owner="${7:-false}"
|
|
1183
1330
|
printf '%s' "$json" > "$gate_fixture"
|
|
1184
1331
|
local got
|
|
1185
|
-
got=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 "$conclusion" 2>&1)
|
|
1332
|
+
got=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 "$conclusion" "$blast" "$protected" "$owner" 0.8 2>&1)
|
|
1186
1333
|
if [ "$got" != "$want" ]; then
|
|
1187
1334
|
VALIDATOR_OK=0
|
|
1188
1335
|
echo "FAIL: the merge-gate validator called '${name}' ${got}, expected ${want}" >&2
|
|
1189
1336
|
fi
|
|
1190
1337
|
}
|
|
1191
1338
|
|
|
1192
|
-
|
|
1339
|
+
# A report the agent would produce on a change it reviewed and found nothing wrong with.
|
|
1340
|
+
gate_clean='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1193
1341
|
gate_push='{"type":"push_to_pull_request_branch","pr_number":9}'
|
|
1194
1342
|
gate_items() { printf '{"items":[%s]}' "$1"; }
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
#
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
gate_case "
|
|
1206
|
-
gate_case "
|
|
1207
|
-
gate_case "
|
|
1208
|
-
gate_case "
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1343
|
+
# The agent writes one of two words and a fenced JSON block. Everything else about the
|
|
1344
|
+
# outcome is computed from that block and from the measured facts passed as arguments.
|
|
1345
|
+
gate_comment() {
|
|
1346
|
+
printf '{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** %s\\n\\n```json\\n%s\\n```"}' "$1" "${2:-$gate_clean}"
|
|
1347
|
+
}
|
|
1348
|
+
|
|
1349
|
+
# The measured facts decide, and a clean report cannot argue with them.
|
|
1350
|
+
gate_case "clean and low risk auto-merges" auto-merge "$(gate_items "$(gate_comment assessed)")" success low
|
|
1351
|
+
gate_case "medium risk still auto-merges when recoverable" \
|
|
1352
|
+
auto-merge "$(gate_items "$(gate_comment assessed)")" success medium
|
|
1353
|
+
gate_case "high blast radius needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success high
|
|
1354
|
+
gate_case "a protected path needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success low true
|
|
1355
|
+
gate_case "an owner path needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success low false true
|
|
1356
|
+
gate_case "a non-success CI conclusion blocks" blocked "$(gate_items "$(gate_comment assessed)")" failure low
|
|
1357
|
+
|
|
1358
|
+
# The agent's report decides the rest. This is the rule the whole redesign rests on: an
|
|
1359
|
+
# unverified finding is a warning whatever severity it claims, so a model cannot fail the gate
|
|
1360
|
+
# by asserting something it did not demonstrate, and cannot pass it by understating one it did.
|
|
1361
|
+
gate_unverified='{\"findings\":[{\"verified\":false,\"severity\":\"critical\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1362
|
+
gate_verified='{\"findings\":[{\"verified\":true,\"severity\":\"critical\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1363
|
+
gate_verified_low='{\"findings\":[{\"verified\":true,\"severity\":\"medium\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1364
|
+
gate_case "an unverified critical finding does not block" \
|
|
1365
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_unverified")")" success low
|
|
1366
|
+
gate_case "a verified critical finding blocks" blocked "$(gate_items "$(gate_comment assessed "$gate_verified")")" success low
|
|
1367
|
+
gate_case "a verified medium finding does not block" \
|
|
1368
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_verified_low")")" success low
|
|
1369
|
+
|
|
1370
|
+
gate_fragile='{\"findings\":[],\"recoverability\":\"low\",\"recoverabilitySignals\":[\"rewrites the stored rows in place\"],\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1371
|
+
gate_bare_low='{\"findings\":[],\"recoverability\":\"low\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
|
|
1372
|
+
gate_unmet='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":false,\"confidence\":0.95}'
|
|
1373
|
+
gate_unsure='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.4}'
|
|
1374
|
+
gate_case "medium risk that cannot be undone needs a person" \
|
|
1375
|
+
human-review "$(gate_items "$(gate_comment assessed "$gate_fragile")")" success medium
|
|
1376
|
+
gate_case "low risk that cannot be undone still auto-merges" \
|
|
1377
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_fragile")")" success low
|
|
1378
|
+
# An unevidenced "low" is the old category escalation wearing a new name, so it is held to the
|
|
1379
|
+
# same standard as a finding: name what cannot be undone, or it does not change the outcome.
|
|
1380
|
+
gate_case "a low rating that names nothing is read as medium" \
|
|
1381
|
+
auto-merge "$(gate_items "$(gate_comment assessed "$gate_bare_low")")" success medium
|
|
1382
|
+
gate_case "an unmet acceptance criterion needs a person" \
|
|
1383
|
+
human-review "$(gate_items "$(gate_comment assessed "$gate_unmet")")" success low
|
|
1384
|
+
gate_case "confidence below the threshold needs a person" \
|
|
1385
|
+
human-review "$(gate_items "$(gate_comment assessed "$gate_unsure")")" success low
|
|
1386
|
+
|
|
1387
|
+
# The agent may raise the measured blast radius when it sees something the path rules could
|
|
1388
|
+
# not. It may never lower it, which is the only direction that can turn a person's review into
|
|
1389
|
+
# a machine merge.
|
|
1390
|
+
gate_raise='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"high\",\"reason\":\"new authorization decision point\"}}'
|
|
1391
|
+
gate_lower='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"low\"}}'
|
|
1392
|
+
gate_case "the agent can raise the blast radius" \
|
|
1393
|
+
owner-review "$(gate_items "$(gate_comment assessed "$gate_raise")")" success low
|
|
1394
|
+
gate_case "the agent cannot lower the blast radius" \
|
|
1395
|
+
owner-review "$(gate_items "$(gate_comment assessed "$gate_lower")")" success high
|
|
1396
|
+
|
|
1397
|
+
# remediated used to require conclusion == "failure", which threw correct work away. The prompt
|
|
1398
|
+
# tells the agent to merge main in, verify and push when CI is green but the pull request
|
|
1399
|
+
# conflicts. That is a real and common state: a conflicting pull request has no merge ref, so
|
|
1400
|
+
# GitHub can never run CI on that head, and the belt falls back to the last verdict on the
|
|
1401
|
+
# branch, which is usually success. The agent did the job, the validator called it invalid,
|
|
1402
|
+
# conclude was skipped, and because this worker stages its outputs the resolved merge commit
|
|
1403
|
+
# was discarded. The belt then dispatched again on the same verdict, up to six times, each a
|
|
1404
|
+
# full run on the single-slot merge belt.
|
|
1405
|
+
gate_case "remediated with one push, CI green" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" success low
|
|
1406
|
+
gate_case "remediated with one push, CI red" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" failure low
|
|
1407
|
+
gate_case "remediated with no push" invalid "$(gate_items "$(gate_comment remediated)")" failure low
|
|
1408
|
+
gate_case "remediated with two pushes" invalid "$(gate_items "$(gate_comment remediated),${gate_push},${gate_push}")" failure low
|
|
1409
|
+
gate_case "an assessment carrying a push" invalid "$(gate_items "$(gate_comment assessed),${gate_push}")" success low
|
|
1410
|
+
|
|
1411
|
+
# Output from a worker version that predates the disposition table. Applying its vocabulary
|
|
1412
|
+
# would merge on a word this validator no longer means the same thing by.
|
|
1413
|
+
gate_case "the old merge vocabulary is refused" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** merge"}]}' success low
|
|
1414
|
+
gate_case "the old review vocabulary is refused" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** review"}]}' success low
|
|
1415
|
+
|
|
1416
|
+
# Nothing malformed may fall through to a merge. Each of these parks the pull request instead.
|
|
1417
|
+
gate_case "a report aimed at another issue" invalid '{"items":[{"type":"add_comment","item_number":99,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed\\n```json\\n{}\\n```"}]}' success low
|
|
1418
|
+
gate_case "a verdict with no json block" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed"}]}' success low
|
|
1419
|
+
gate_case "a json block that does not parse" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed\\n```json\\n{nope}\\n```"}]}' success low
|
|
1420
|
+
gate_case "no verdict in the output" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"just a note"}]}' success low
|
|
1421
|
+
# Adversarial shapes. Every one of these read as the permissive value at some point, and each
|
|
1422
|
+
# is a near miss rather than nonsense: the report the agent meant to send, with one field
|
|
1423
|
+
# typed the way a model types it when it is being loose. A merge gate that reads `"true"` as
|
|
1424
|
+
# true merges on a string.
|
|
1425
|
+
gate_near_miss() {
|
|
1426
|
+
local name="$1" want="$2" report="$3"
|
|
1427
|
+
gate_case "$name" "$want" "$(gate_items "$(gate_comment assessed "$report")")" success low
|
|
1428
|
+
}
|
|
1429
|
+
gate_near_miss "verified as the string true" invalid '{\"findings\":[{\"verified\":\"true\",\"severity\":\"critical\"}],\"confidence\":0.95}'
|
|
1430
|
+
gate_near_miss "verified as the number one" invalid '{\"findings\":[{\"verified\":1,\"severity\":\"critical\"}],\"confidence\":0.95}'
|
|
1431
|
+
gate_near_miss "a severity outside the scale" invalid '{\"findings\":[{\"verified\":true,\"severity\":\"blocker\"}],\"confidence\":0.95}'
|
|
1432
|
+
gate_near_miss "a finding with no severity" invalid '{\"findings\":[{\"verified\":true}],\"confidence\":0.95}'
|
|
1433
|
+
gate_near_miss "a severity in capitals" blocked '{\"findings\":[{\"verified\":true,\"severity\":\"CRITICAL\"}],\"confidence\":0.95}'
|
|
1434
|
+
gate_near_miss "acceptanceCriteriaMet as a string" invalid '{\"findings\":[],\"acceptanceCriteriaMet\":\"false\",\"confidence\":0.95}'
|
|
1435
|
+
gate_near_miss "confidence as a word" invalid '{\"findings\":[],\"confidence\":\"high\"}'
|
|
1436
|
+
gate_near_miss "findings as a string" invalid '{\"findings\":\"none\",\"confidence\":0.95}'
|
|
1437
|
+
gate_near_miss "a recoverability outside the scale" invalid '{\"findings\":[],\"recoverability\":\"none\",\"confidence\":0.95}'
|
|
1438
|
+
gate_near_miss "a raise to an unknown level" invalid '{\"findings\":[],\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"critical\"}}'
|
|
1439
|
+
gate_near_miss "a raise in capitals is honoured" owner-review '{\"findings\":[],\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"HIGH\"}}'
|
|
1440
|
+
gate_near_miss "a report that is not an object" invalid '\"just a string\"'
|
|
1441
|
+
|
|
1442
|
+
# The prompt puts the report last and the prose above it routinely quotes json from the diff
|
|
1443
|
+
# under review. Reading the first fence handed the decision to whatever the agent quoted, and
|
|
1444
|
+
# PROTECTED_PATHS itself names package.json and global.json, so the reviewed diff is often
|
|
1445
|
+
# json. The decoy here claims everything is fine; the real report blocks.
|
|
1446
|
+
# Built with jq rather than hand-escaped: this body has two fenced blocks, each containing
|
|
1447
|
+
# quoted json, inside a json string. Hand-escaping it is how a test ends up asserting on a
|
|
1448
|
+
# fixture that does not parse.
|
|
1449
|
+
gate_decoy=$(jq -nc --arg body "$(printf '%s\n' '<!-- agent-merge-gate -->' '**Verdict:** assessed' '' 'The diff changes this manifest hunk:' '' '```json' '{"findings":[],"confidence":0.95}' '```' '' 'Report:' '' '```json' '{"findings":[{"verified":true,"severity":"critical"}],"confidence":0.95}' '```')" \
|
|
1450
|
+
'{items:[{type:"add_comment",item_number:7,body:$body}]}')
|
|
1451
|
+
gate_case "the last json fence is the report, not the first" blocked "$gate_decoy" success low
|
|
1452
|
+
|
|
1453
|
+
# Verdict and report used to be selected independently, and each took the first it found, so a
|
|
1454
|
+
# second comment reporting a verified critical finding was discarded and a comment with no
|
|
1455
|
+
# verdict could supply the report for a verdict written in another.
|
|
1456
|
+
gate_case "two comments carrying a verdict" \
|
|
1457
|
+
invalid "$(gate_items "$(gate_comment assessed),$(gate_comment assessed "$gate_verified")")" success low
|
|
1458
|
+
|
|
1459
|
+
# An empty measured fact is a job that did not report, not a low-risk pull request. `${4:-low}`
|
|
1460
|
+
# substituted the default for an empty argument, so a skipped protected_changes read as
|
|
1461
|
+
# "low, nothing protected" and merged.
|
|
1462
|
+
gate_unmeasured=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success "" "" "" 0.8 2>&1 || true)
|
|
1463
|
+
printf '%s' "$(gate_items "$(gate_comment assessed)")" > "$gate_fixture"
|
|
1464
|
+
gate_unmeasured=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success "" "" "" 0.8 2>&1 || true)
|
|
1465
|
+
if [ "$gate_unmeasured" != invalid ]; then
|
|
1466
|
+
VALIDATOR_OK=0
|
|
1467
|
+
echo "FAIL: an unmeasured blast radius produced '${gate_unmeasured}', expected invalid" >&2
|
|
1468
|
+
fi
|
|
1469
|
+
gate_half=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success low "" "" 0.8 2>&1 || true)
|
|
1470
|
+
if [ "$gate_half" != human-review ]; then
|
|
1471
|
+
VALIDATOR_OK=0
|
|
1472
|
+
echo "FAIL: an unmeasured protected-path fact produced '${gate_half}', expected human-review" >&2
|
|
1473
|
+
fi
|
|
1474
|
+
|
|
1475
|
+
gate_case "an empty item list" invalid '{"items":[]}' success low
|
|
1476
|
+
gate_case "output that is not an item list" invalid '{"nope":true}' success low
|
|
1214
1477
|
|
|
1215
1478
|
rm -f "$gate_fixture"
|
|
1216
1479
|
if [ "$VALIDATOR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
@@ -1711,6 +1974,43 @@ if worker_installed merge-gate; then
|
|
|
1711
1974
|
echo "FAIL: identify-gate-subject resolves the CI run only when the conclusion is unknown; a router that passes both gets an empty run ID" >&2
|
|
1712
1975
|
fi
|
|
1713
1976
|
fi
|
|
1977
|
+
# Repeating a report the validator refused reproduces it, and every repeat is a fresh agent run
|
|
1978
|
+
# holding the repo-wide merge-belt slot. The worker gives that failure a smaller budget than a
|
|
1979
|
+
# crash, and -- the part that actually saves the slot -- records it as a verdict, because the belt
|
|
1980
|
+
# bounds its own retries by counting attempt comments and would otherwise dispatch to the cap
|
|
1981
|
+
# whatever the worker decided.
|
|
1982
|
+
if worker_installed merge-gate; then
|
|
1983
|
+
UNUSABLE_OK=1
|
|
1984
|
+
unusable_cap="$(sed -n 's/^ PARK_AT_UNUSABLE_OUTPUT: "\([0-9]*\)"$/\1/p' "$MERGE_GATE_WORKER_MD" | head -1)"
|
|
1985
|
+
machine_cap="$(sed -n 's/^ PARK_AT_ATTEMPT: "\([0-9]*\)"$/\1/p' "$MERGE_GATE_WORKER_MD" | head -1)"
|
|
1986
|
+
if [ -z "$unusable_cap" ] || [ -z "$machine_cap" ] || [ "$unusable_cap" -ge "$machine_cap" ]; then
|
|
1987
|
+
UNUSABLE_OK=0
|
|
1988
|
+
echo "FAIL: an unusable report must get a smaller budget than a crash (unusable='${unusable_cap:-unset}', machine='${machine_cap:-unset}')" >&2
|
|
1989
|
+
fi
|
|
1990
|
+
# The budget is decided once, not restated per step. Four `if:` expressions repeating the same
|
|
1991
|
+
# pair of conditions is the shape the protected-files hold already got wrong.
|
|
1992
|
+
if [ "$(count -cE "^ if: steps\.budget\.outputs\.park" "$MERGE_GATE_WORKER_MD")" -lt 3 ]; then
|
|
1993
|
+
UNUSABLE_OK=0
|
|
1994
|
+
echo "FAIL: the incomplete job must read one computed budget decision, not re-derive it" >&2
|
|
1995
|
+
fi
|
|
1996
|
+
# The park has to be a verdict or the belt keeps dispatching: a comment carrying the gate
|
|
1997
|
+
# marker AND a Verdict line is what detect-pr-conflicts and the reconcile belt both park on.
|
|
1998
|
+
if ! grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
|
|
1999
|
+
grep -q '\${{ env.GATE_MARKER }}' ||
|
|
2000
|
+
! grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
|
|
2001
|
+
grep -q '\*\*Verdict:\*\* human-review'; then
|
|
2002
|
+
UNUSABLE_OK=0
|
|
2003
|
+
echo "FAIL: the unusable-report park must carry the gate marker and a Verdict line, or the belt dispatches it again" >&2
|
|
2004
|
+
fi
|
|
2005
|
+
# And it must not also count as an attempt, or one park is recorded twice.
|
|
2006
|
+
if grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
|
|
2007
|
+
grep -q '\${{ env.ATTEMPT_MARKER }}'; then
|
|
2008
|
+
UNUSABLE_OK=0
|
|
2009
|
+
echo "FAIL: the unusable-report park carries the attempt marker as well; it is a decision, not an attempt" >&2
|
|
2010
|
+
fi
|
|
2011
|
+
if [ "$UNUSABLE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
2012
|
+
fi
|
|
2013
|
+
|
|
1714
2014
|
if [ "$GATE_RUN_ID_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1715
2015
|
fi
|
|
1716
2016
|
|