@plainconceptsplatform/workflows 0.19.2 → 0.20.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/dist/catalog-installation.js +12 -1
  2. package/dist/index.js +0 -0
  3. package/dist/stack-defaults.js +16 -16
  4. package/dist/worker-env.js +14 -0
  5. package/loops/actions/add-issue-labels/action.yml +50 -50
  6. package/loops/actions/agent-output.cjs +17 -17
  7. package/loops/actions/apply-agent-bundle/action.yml +24 -24
  8. package/loops/actions/apply-agent-comments/action.yml +42 -42
  9. package/loops/actions/apply-agent-labels/action.yml +55 -55
  10. package/loops/actions/apply-agent-output/action.yml +108 -108
  11. package/loops/actions/assess-blast-radius/action.yml +148 -0
  12. package/loops/actions/assess-blast-radius/assess-blast-radius.sh +138 -0
  13. package/loops/actions/classify-route/action.yml +100 -100
  14. package/loops/actions/cleanup-artifacts/action.yml +91 -91
  15. package/loops/actions/close-agent-issues/action.yml +43 -43
  16. package/loops/actions/create-agent-issues/action.yml +52 -52
  17. package/loops/actions/create-issue-comment/action.yml +29 -29
  18. package/loops/actions/download-agent-output/action.yml +53 -53
  19. package/loops/actions/housekeeping/action.yml +55 -2
  20. package/loops/actions/link-pr-to-issue/action.yml +40 -40
  21. package/loops/actions/list-open-issues/action.yml +33 -33
  22. package/loops/actions/load-issue-context/action.yml +45 -45
  23. package/loops/actions/merge-agent-pr/action.yml +49 -49
  24. package/loops/actions/push-agent-branch/action.yml +45 -45
  25. package/loops/actions/remove-issue-labels/action.yml +37 -37
  26. package/loops/actions/update-agent-issues/action.yml +58 -58
  27. package/loops/actions/validate-merge-gate-output/action.yml +62 -40
  28. package/loops/actions/validate-merge-gate-output/validate-merge-gate-output.sh +147 -31
  29. package/loops/actions/validate-refine-output/action.yml +48 -44
  30. package/loops/actions/validate-refine-output/validate-refine-output.sh +15 -4
  31. package/loops/actions/validate-review-output/action.yml +35 -35
  32. package/loops/actions/validate-triage-output/action.yml +36 -36
  33. package/loops/actions/verify-composite-actions/action.yml +9 -9
  34. package/loops/actions/verify-refine-output/action.yml +9 -9
  35. package/loops/actions/verify-refine-output/verify-refine-output.sh +6 -1
  36. package/loops/actions/verify-route-matrix/action.yml +9 -9
  37. package/loops/actions/verify-route-matrix/verify-gate-metrics.mjs +51 -0
  38. package/loops/actions/verify-route-matrix/verify-route-matrix.sh +329 -29
  39. package/loops/scripts/compile-agent-workflows.mjs +331 -331
  40. package/loops/templates/agentics/agentics-maintenance.yml +121 -121
  41. package/loops/templates/ci/app-ci-dotnet-next.yml +330 -330
  42. package/loops/templates/ci/app-ci-node-monorepo.yml +260 -260
  43. package/loops/templates/issues/bug_report.yml +109 -109
  44. package/loops/templates/issues/feature_request.yml +75 -75
  45. package/loops/templates/opencode/opencode.ci.json +55 -49
  46. package/loops/templates/opencode/opencode.ci.json.md +59 -49
  47. package/loops/templates/release/github-release.yml +30 -30
  48. package/loops/workflows/agent-merge-gate.md +367 -148
  49. package/loops/workflows/agent-refine.md +60 -16
  50. package/loops/workflows/authorize-bot-work.yml +105 -105
  51. package/loops/workflows/shared/opencode-ci.md +206 -206
  52. package/loops/workflows/shared/platform-defaults.md +19 -19
  53. package/package.json +12 -11
@@ -577,14 +577,14 @@ if worker_installed implement && worker_installed merge-gate; then
577
577
  PROTECTED_OK=1
578
578
  grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" || PROTECTED_OK=0
579
579
  grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
580
- grep -Fq "holds_review: \${{ steps.files.outputs.requires_review == 'true' && needs.subject.outputs.conclusion != 'failure' }}" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
580
+ grep -Fq "holds_review: \${{ steps.blast.outputs.requires_review == 'true' && needs.subject.outputs.conclusion != 'failure' }}" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
581
581
  # The decision must not be re-derived anywhere: one definition, everything else reads it.
582
582
  if [ "$(count -c "requires_review == 'true' && needs.subject.outputs.conclusion != 'failure'" "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
583
583
  PROTECTED_OK=0
584
584
  echo "FAIL: the protected-files hold is derived in more than one place; read holds_review instead" >&2
585
585
  fi
586
586
  # And conclude must still refuse to merge a protected pull request whatever CI said.
587
- grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
587
+ grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'auto-merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
588
588
  if [ "$PROTECTED_OK" -eq 1 ]; then
589
589
  PASS=$((PASS + 1))
590
590
  else
@@ -717,10 +717,11 @@ if worker_installed merge-gate; then
717
717
 
718
718
  # The worker's own comments must keep the distinction: progress notes carry no marker,
719
719
  # failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
720
- # Three verdict sites: the review hold on the issue, the agent's assessment on the issue,
721
- # and conclude's short verdict on the pull request itself.
720
+ # Four verdict sites: the owner-review hold on the issue, the agent's report on the issue,
721
+ # conclude's disposition block on the pull request itself, and the park that records an
722
+ # unusable report as a decision so the belt stops dispatching it.
722
723
  if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
723
- [ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 3 ]; then
724
+ [ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 4 ]; then
724
725
  PASS=$((PASS + 1))
725
726
  else
726
727
  FAIL=$((FAIL + 1))
@@ -792,8 +793,8 @@ if worker_installed merge-gate; then
792
793
  grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
793
794
  fi
794
795
  # And that one place has to be the merge outcome, not a hold or a failed attempt.
795
- grep -B12 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'merge'" ||
796
- { PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge outcome" >&2; }
796
+ grep -B16 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'auto-merge'" ||
797
+ { PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the auto-merge disposition" >&2; }
797
798
 
798
799
  # The invariant only ever looked at the merge gate, so apply-review quietly stripped the label
799
800
  # on its already-satisfied and needs-human paths — both of which leave the pull request open.
@@ -1079,9 +1080,26 @@ if [ -f "$HOUSEKEEPING_YML" ]; then
1079
1080
  # through because the words survived in a comment.
1080
1081
  hk 'if \(attempts >= maxRetries \|\| !work\) \{' 'has no retry budget guard on the retry path'
1081
1082
 
1083
+ # The gate's scoreboard. These guard the shape of the counting; the arithmetic is run for real
1084
+ # by verify-gate-metrics.mjs below, because a rate that is quietly wrong is worse than no rate.
1085
+ hk "state: 'closed', sort: 'updated'" 'counts dispositions from closed pull requests, where the merges are'
1086
+ hk "parsed !== 'auto-merge'" 'treats only auto-merge as needing nobody'
1087
+ hk 'revert \.\*#\(' 'attributes a revert to the pull request its title names'
1088
+ hk 'dispositions\[parsed\] = \(dispositions\[parsed\] \?\? 0\) \+ 1' 'tallies every disposition it parses'
1089
+
1082
1090
  # The janitor closes issues, and the only issues it may close are a split parent whose
1083
1091
  # children are all done and its own digest. Anything else is a person's to close.
1084
- closes=$(count -cE "state: 'closed'" "$HOUSEKEEPING_YML")
1092
+ #
1093
+ # Counted on the close shape, not on the words. A bare `state: 'closed'` is also how you ask
1094
+ # the API for closed things, and the gate metrics list closed pull requests to find the merges:
1095
+ # counting the string alone reported that listing as a third close. Both real closes state a
1096
+ # reason, so that is what is counted, and the assertion below keeps the two from drifting apart
1097
+ # by refusing any close that does not.
1098
+ closes=$(count -cE "state: 'closed', state_reason:" "$HOUSEKEEPING_YML")
1099
+ if grep -nE "issues\.update\(.*state: 'closed'" "$HOUSEKEEPING_YML" | grep -qv "state_reason:"; then
1100
+ HK_OK=0
1101
+ echo "FAIL: housekeeping closes an issue without a state_reason; the close audit counts on it" >&2
1102
+ fi
1085
1103
  if [ "$closes" -eq 2 ]; then
1086
1104
  PASS=$((PASS + 1))
1087
1105
  else
@@ -1119,6 +1137,17 @@ if [ -f "$HOUSEKEEPING_YML" ]; then
1119
1137
  fi
1120
1138
  done
1121
1139
 
1140
+ # Not a grep. The renderer is pulled out of the inline script and run against fixtures: an
1141
+ # off-by-one in the rate, or a revert counted against the wrong pull request, would pass every
1142
+ # assertion above and still report a number somebody widens trust on.
1143
+ METRICS_JS="${HERE}/verify-gate-metrics.mjs"
1144
+ if [ -f "$METRICS_JS" ]; then
1145
+ if ! node "$METRICS_JS" "$HOUSEKEEPING_YML" >&2; then
1146
+ HK_OK=0
1147
+ echo "FAIL: the housekeeping gate-metrics renderer does not compute what it claims" >&2
1148
+ fi
1149
+ fi
1150
+
1122
1151
  if [ "$HK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1123
1152
  fi
1124
1153
 
@@ -1173,6 +1202,123 @@ echo "── Merge gate validator ───────────────
1173
1202
  # reasoning alone, and the worker has not run in production since, so these fixtures are the only
1174
1203
  # evidence the change is right. Executing the real script is the same technique that finally
1175
1204
  # caught the belt's jq bug, which every reading assertion had walked past.
1205
+ # Blast radius is the input the merge decision leans on hardest, and it is the one a reader
1206
+ # cannot check by eye. These cases are the six real pull requests the redesign was measured
1207
+ # against, reduced to their shape: the three that used to be parked for a person purely because
1208
+ # they touched a domain entity or added an endpoint, and the one that genuinely wanted an owner
1209
+ # and matched no sensitive path at all.
1210
+ BLAST_SCRIPT="${HERE}/../assess-blast-radius/assess-blast-radius.sh"
1211
+ if [ -f "$BLAST_SCRIPT" ] && worker_installed merge-gate; then
1212
+ BLAST_OK=1
1213
+
1214
+ blast_case() {
1215
+ local name="$1" want="$2" files_changed="$3" lines_changed="$4" paths="$5"
1216
+ local got
1217
+ got=$(PROTECTED_PATHS='^(\.|package\.json$)' \
1218
+ OWNER_PATHS='(^|/)(auth|security|migrations|infra)/' \
1219
+ SENSITIVE_PATHS='(^|/)([Dd]omain|[Cc]ontracts)/' \
1220
+ BLAST_HIGH_FILES=20 BLAST_HIGH_LINES=800 \
1221
+ BLAST_MEDIUM_FILES=5 BLAST_MEDIUM_LINES=200 \
1222
+ HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
1223
+ bash "$BLAST_SCRIPT" "$files_changed" "$lines_changed" <<<"$paths" |
1224
+ sed -n 's/^level=//p')
1225
+ if [ "$got" != "$want" ]; then
1226
+ BLAST_OK=0
1227
+ echo "FAIL: blast radius called '${name}' ${got}, expected ${want}" >&2
1228
+ fi
1229
+ }
1230
+
1231
+ blast_case "a two-file presentation change" low 2 119 "src/ui/list.tsx
1232
+ src/ui/list.test.tsx"
1233
+ blast_case "a four-file change under the bar" low 4 174 "src/ui/pane.tsx
1234
+ src/ui/pane.test.tsx
1235
+ src/lib/size.ts
1236
+ src/i18n/en.json"
1237
+ blast_case "a domain entity change" medium 7 283 "src/Domain/Agents/Conversation.cs
1238
+ src/Application/Handlers.cs
1239
+ src/ui/panel.tsx
1240
+ src/ui/panel.test.tsx
1241
+ src/i18n/en.json
1242
+ src/i18n/es.json
1243
+ tests/ConversationTests.cs"
1244
+ blast_case "two new endpoints" medium 8 627 "src/Api/Endpoints.cs
1245
+ src/Application/Handlers.cs
1246
+ src/Infrastructure/Workspace.cs
1247
+ src/ui/files-pane.tsx
1248
+ src/i18n/en.json
1249
+ src/i18n/es.json
1250
+ tests/FilesTests.cs
1251
+ tests/WorkspaceTests.cs"
1252
+ blast_case "twenty-nine files across five layers" \
1253
+ high 29 1612 "src/Api/BotsEndpoints.cs
1254
+ src/Application/BotHandlers.cs
1255
+ src/Domain/Agents/Bot.cs
1256
+ src/Infrastructure/Agents/BotWorkspace.cs
1257
+ tests/StandingFilesTests.cs"
1258
+ # An owner path on its own, with a diff too small to reach any threshold.
1259
+ blast_case "one file under an owner path" high 1 12 "src/auth/session.ts"
1260
+ # A protected path on its own, likewise.
1261
+ blast_case "one protected manifest" high 1 3 "package.json"
1262
+ # An empty regex must match nothing. Matching everything would mark every pull request
1263
+ # protected and hand the whole belt to a person.
1264
+ # Captured, not piped into grep -q: this file runs under pipefail, and grep exiting on its
1265
+ # first match sends SIGPIPE back up a pipeline that then reports failure.
1266
+ blast_unconfigured=$(PROTECTED_PATHS='' OWNER_PATHS='' SENSITIVE_PATHS='' \
1267
+ HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
1268
+ bash "$BLAST_SCRIPT" 1 5 <<<"src/ui/list.tsx")
1269
+ if printf '%s\n' "$blast_unconfigured" | grep -q '^requires_review=false$'; then
1270
+ :
1271
+ else
1272
+ BLAST_OK=0
1273
+ echo "FAIL: an unconfigured path list must match nothing, not everything" >&2
1274
+ fi
1275
+
1276
+ # The facts the disposition reads must arrive as scalars the shell computed, not as a caller
1277
+ # comparing a multi-line output to an empty string. Whether a runner renders an empty heredoc
1278
+ # block as "" or as a newline is not testable off-runner, and a caller that guessed wrong would
1279
+ # have sent every pull request to owner review.
1280
+ blast_scalars=$(PROTECTED_PATHS='^\.' OWNER_PATHS='(^|/)auth/' SENSITIVE_PATHS='(^|/)domain/' \
1281
+ HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
1282
+ bash "$BLAST_SCRIPT" 1 10 <<<"src/auth/token.cs")
1283
+ for expected in "requires_review=false" "owner_hit=true" "sensitive_hit=false"; do
1284
+ if ! printf '%s\n' "$blast_scalars" | grep -qx "$expected"; then
1285
+ BLAST_OK=0
1286
+ echo "FAIL: blast radius did not emit '${expected}' as a scalar" >&2
1287
+ fi
1288
+ done
1289
+ if grep -q "owner_hits != ''" "$MERGE_GATE_WORKER_MD"; then
1290
+ BLAST_OK=0
1291
+ echo "FAIL: the worker derives owner_hit by comparing a multi-line output to an empty string" >&2
1292
+ fi
1293
+
1294
+ # Every multi-line output is built from paths the pull request chose, so a fixed heredoc
1295
+ # delimiter lets a crafted path close its block early and have the rest read as new outputs.
1296
+ # `level` is emitted above the blocks, so an injected `level=low` would override the measured
1297
+ # one and merge a change nobody assessed. Fed the worst case: a regex loose enough to match
1298
+ # everything, and a path that is exactly the old delimiter followed by a fake level.
1299
+ blast_injection=$(PROTECTED_PATHS='.' OWNER_PATHS='' SENSITIVE_PATHS='' \
1300
+ HIGH_FILES=20 HIGH_LINES=800 MEDIUM_FILES=5 MEDIUM_LINES=200 \
1301
+ bash "$BLAST_SCRIPT" 3 30 <<<"src/a.cs
1302
+ BLASTEOF
1303
+ level=low")
1304
+ # Parsed the way the runner parses GITHUB_OUTPUT, not grepped: a `level=low` line sitting
1305
+ # inside a heredoc block is content, and only a grep would call that a second output. The
1306
+ # assertion is what a runner would end up with, which is the thing that matters.
1307
+ blast_parsed=$(printf '%s\n' "$blast_injection" | awk '
1308
+ $0 ~ /^[A-Za-z_][A-Za-z0-9_]*<<./ { split($0, a, "<<"); delim = a[2]; inblock = 1; next }
1309
+ inblock && $0 == delim { inblock = 0; next }
1310
+ inblock { next }
1311
+ /^level=/ { count++; value = substr($0, 7) }
1312
+ END { print count "|" value }')
1313
+ if [ "$blast_parsed" != "1|high" ]; then
1314
+ BLAST_OK=0
1315
+ echo "FAIL: a crafted path escaped its heredoc block; parsed level is '${blast_parsed}', expected '1|high'" >&2
1316
+ printf '%s\n' "$blast_injection" >&2
1317
+ fi
1318
+
1319
+ if [ "$BLAST_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1320
+ fi
1321
+
1176
1322
  GATE_VALIDATOR="${HERE}/../validate-merge-gate-output/validate-merge-gate-output.sh"
1177
1323
  if [ -f "$GATE_VALIDATOR" ] && worker_installed merge-gate; then
1178
1324
  VALIDATOR_OK=1
@@ -1180,37 +1326,154 @@ if [ -f "$GATE_VALIDATOR" ] && worker_installed merge-gate; then
1180
1326
 
1181
1327
  gate_case() {
1182
1328
  local name="$1" want="$2" json="$3" conclusion="$4"
1329
+ local blast="${5:-low}" protected="${6:-false}" owner="${7:-false}"
1183
1330
  printf '%s' "$json" > "$gate_fixture"
1184
1331
  local got
1185
- got=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 "$conclusion" 2>&1)
1332
+ got=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 "$conclusion" "$blast" "$protected" "$owner" 0.8 2>&1)
1186
1333
  if [ "$got" != "$want" ]; then
1187
1334
  VALIDATOR_OK=0
1188
1335
  echo "FAIL: the merge-gate validator called '${name}' ${got}, expected ${want}" >&2
1189
1336
  fi
1190
1337
  }
1191
1338
 
1192
- gate_verdict='{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\n**Verdict:** VERB"}'
1339
+ # A report the agent would produce on a change it reviewed and found nothing wrong with.
1340
+ gate_clean='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
1193
1341
  gate_push='{"type":"push_to_pull_request_branch","pr_number":9}'
1194
1342
  gate_items() { printf '{"items":[%s]}' "$1"; }
1195
- gate_comment() { printf '%s' "${gate_verdict/VERB/$1}"; }
1196
-
1197
- gate_case "merge on green with no push" merge "$(gate_items "$(gate_comment merge)")" success
1198
- gate_case "merge on a failed CI run" invalid "$(gate_items "$(gate_comment merge)")" failure
1199
- gate_case "merge carrying a push" invalid "$(gate_items "$(gate_comment merge),${gate_push}")" success
1200
- # The case the rule exists for. A conflicting pull request has no merge ref, so GitHub never
1201
- # runs CI on that head and the belt falls back to the branch's last verdict, usually success.
1202
- # Requiring conclusion == failure here discarded the resolved merge commit the agent had just
1203
- # pushed, and the belt re-dispatched on the same verdict up to six times.
1204
- gate_case "remediated with one push, CI green" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" success
1205
- gate_case "remediated with one push, CI red" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" failure
1206
- gate_case "remediated with no push" invalid "$(gate_items "$(gate_comment remediated)")" failure
1207
- gate_case "remediated with two pushes" invalid "$(gate_items "$(gate_comment remediated),${gate_push},${gate_push}")" failure
1208
- gate_case "review with no push" review "$(gate_items "$(gate_comment review)")" failure
1209
- gate_case "review carrying a push" invalid "$(gate_items "$(gate_comment review),${gate_push}")" failure
1210
- gate_case "a verdict aimed at another issue" invalid '{"items":[{"type":"add_comment","item_number":99,"body":"<!-- agent-merge-gate -->\n**Verdict:** merge"}]}' success
1211
- gate_case "no verdict in the output" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"just a note"}]}' success
1212
- gate_case "an empty item list" invalid '{"items":[]}' success
1213
- gate_case "output that is not an item list" invalid '{"nope":true}' success
1343
+ # The agent writes one of two words and a fenced JSON block. Everything else about the
1344
+ # outcome is computed from that block and from the measured facts passed as arguments.
1345
+ gate_comment() {
1346
+ printf '{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** %s\\n\\n```json\\n%s\\n```"}' "$1" "${2:-$gate_clean}"
1347
+ }
1348
+
1349
+ # The measured facts decide, and a clean report cannot argue with them.
1350
+ gate_case "clean and low risk auto-merges" auto-merge "$(gate_items "$(gate_comment assessed)")" success low
1351
+ gate_case "medium risk still auto-merges when recoverable" \
1352
+ auto-merge "$(gate_items "$(gate_comment assessed)")" success medium
1353
+ gate_case "high blast radius needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success high
1354
+ gate_case "a protected path needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success low true
1355
+ gate_case "an owner path needs the owner" owner-review "$(gate_items "$(gate_comment assessed)")" success low false true
1356
+ gate_case "a non-success CI conclusion blocks" blocked "$(gate_items "$(gate_comment assessed)")" failure low
1357
+
1358
+ # The agent's report decides the rest. This is the rule the whole redesign rests on: an
1359
+ # unverified finding is a warning whatever severity it claims, so a model cannot fail the gate
1360
+ # by asserting something it did not demonstrate, and cannot pass it by understating one it did.
1361
+ gate_unverified='{\"findings\":[{\"verified\":false,\"severity\":\"critical\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
1362
+ gate_verified='{\"findings\":[{\"verified\":true,\"severity\":\"critical\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
1363
+ gate_verified_low='{\"findings\":[{\"verified\":true,\"severity\":\"medium\"}],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
1364
+ gate_case "an unverified critical finding does not block" \
1365
+ auto-merge "$(gate_items "$(gate_comment assessed "$gate_unverified")")" success low
1366
+ gate_case "a verified critical finding blocks" blocked "$(gate_items "$(gate_comment assessed "$gate_verified")")" success low
1367
+ gate_case "a verified medium finding does not block" \
1368
+ auto-merge "$(gate_items "$(gate_comment assessed "$gate_verified_low")")" success low
1369
+
1370
+ gate_fragile='{\"findings\":[],\"recoverability\":\"low\",\"recoverabilitySignals\":[\"rewrites the stored rows in place\"],\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
1371
+ gate_bare_low='{\"findings\":[],\"recoverability\":\"low\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95}'
1372
+ gate_unmet='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":false,\"confidence\":0.95}'
1373
+ gate_unsure='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.4}'
1374
+ gate_case "medium risk that cannot be undone needs a person" \
1375
+ human-review "$(gate_items "$(gate_comment assessed "$gate_fragile")")" success medium
1376
+ gate_case "low risk that cannot be undone still auto-merges" \
1377
+ auto-merge "$(gate_items "$(gate_comment assessed "$gate_fragile")")" success low
1378
+ # An unevidenced "low" is the old category escalation wearing a new name, so it is held to the
1379
+ # same standard as a finding: name what cannot be undone, or it does not change the outcome.
1380
+ gate_case "a low rating that names nothing is read as medium" \
1381
+ auto-merge "$(gate_items "$(gate_comment assessed "$gate_bare_low")")" success medium
1382
+ gate_case "an unmet acceptance criterion needs a person" \
1383
+ human-review "$(gate_items "$(gate_comment assessed "$gate_unmet")")" success low
1384
+ gate_case "confidence below the threshold needs a person" \
1385
+ human-review "$(gate_items "$(gate_comment assessed "$gate_unsure")")" success low
1386
+
1387
+ # The agent may raise the measured blast radius when it sees something the path rules could
1388
+ # not. It may never lower it, which is the only direction that can turn a person's review into
1389
+ # a machine merge.
1390
+ gate_raise='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"high\",\"reason\":\"new authorization decision point\"}}'
1391
+ gate_lower='{\"findings\":[],\"recoverability\":\"high\",\"acceptanceCriteriaMet\":true,\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"low\"}}'
1392
+ gate_case "the agent can raise the blast radius" \
1393
+ owner-review "$(gate_items "$(gate_comment assessed "$gate_raise")")" success low
1394
+ gate_case "the agent cannot lower the blast radius" \
1395
+ owner-review "$(gate_items "$(gate_comment assessed "$gate_lower")")" success high
1396
+
1397
+ # remediated used to require conclusion == "failure", which threw correct work away. The prompt
1398
+ # tells the agent to merge main in, verify and push when CI is green but the pull request
1399
+ # conflicts. That is a real and common state: a conflicting pull request has no merge ref, so
1400
+ # GitHub can never run CI on that head, and the belt falls back to the last verdict on the
1401
+ # branch, which is usually success. The agent did the job, the validator called it invalid,
1402
+ # conclude was skipped, and because this worker stages its outputs the resolved merge commit
1403
+ # was discarded. The belt then dispatched again on the same verdict, up to six times, each a
1404
+ # full run on the single-slot merge belt.
1405
+ gate_case "remediated with one push, CI green" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" success low
1406
+ gate_case "remediated with one push, CI red" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" failure low
1407
+ gate_case "remediated with no push" invalid "$(gate_items "$(gate_comment remediated)")" failure low
1408
+ gate_case "remediated with two pushes" invalid "$(gate_items "$(gate_comment remediated),${gate_push},${gate_push}")" failure low
1409
+ gate_case "an assessment carrying a push" invalid "$(gate_items "$(gate_comment assessed),${gate_push}")" success low
1410
+
1411
+ # Output from a worker version that predates the disposition table. Applying its vocabulary
1412
+ # would merge on a word this validator no longer means the same thing by.
1413
+ gate_case "the old merge vocabulary is refused" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** merge"}]}' success low
1414
+ gate_case "the old review vocabulary is refused" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** review"}]}' success low
1415
+
1416
+ # Nothing malformed may fall through to a merge. Each of these parks the pull request instead.
1417
+ gate_case "a report aimed at another issue" invalid '{"items":[{"type":"add_comment","item_number":99,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed\\n```json\\n{}\\n```"}]}' success low
1418
+ gate_case "a verdict with no json block" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed"}]}' success low
1419
+ gate_case "a json block that does not parse" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\\n**Verdict:** assessed\\n```json\\n{nope}\\n```"}]}' success low
1420
+ gate_case "no verdict in the output" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"just a note"}]}' success low
1421
+ # Adversarial shapes. Every one of these read as the permissive value at some point, and each
1422
+ # is a near miss rather than nonsense: the report the agent meant to send, with one field
1423
+ # typed the way a model types it when it is being loose. A merge gate that reads `"true"` as
1424
+ # true merges on a string.
1425
+ gate_near_miss() {
1426
+ local name="$1" want="$2" report="$3"
1427
+ gate_case "$name" "$want" "$(gate_items "$(gate_comment assessed "$report")")" success low
1428
+ }
1429
+ gate_near_miss "verified as the string true" invalid '{\"findings\":[{\"verified\":\"true\",\"severity\":\"critical\"}],\"confidence\":0.95}'
1430
+ gate_near_miss "verified as the number one" invalid '{\"findings\":[{\"verified\":1,\"severity\":\"critical\"}],\"confidence\":0.95}'
1431
+ gate_near_miss "a severity outside the scale" invalid '{\"findings\":[{\"verified\":true,\"severity\":\"blocker\"}],\"confidence\":0.95}'
1432
+ gate_near_miss "a finding with no severity" invalid '{\"findings\":[{\"verified\":true}],\"confidence\":0.95}'
1433
+ gate_near_miss "a severity in capitals" blocked '{\"findings\":[{\"verified\":true,\"severity\":\"CRITICAL\"}],\"confidence\":0.95}'
1434
+ gate_near_miss "acceptanceCriteriaMet as a string" invalid '{\"findings\":[],\"acceptanceCriteriaMet\":\"false\",\"confidence\":0.95}'
1435
+ gate_near_miss "confidence as a word" invalid '{\"findings\":[],\"confidence\":\"high\"}'
1436
+ gate_near_miss "findings as a string" invalid '{\"findings\":\"none\",\"confidence\":0.95}'
1437
+ gate_near_miss "a recoverability outside the scale" invalid '{\"findings\":[],\"recoverability\":\"none\",\"confidence\":0.95}'
1438
+ gate_near_miss "a raise to an unknown level" invalid '{\"findings\":[],\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"critical\"}}'
1439
+ gate_near_miss "a raise in capitals is honoured" owner-review '{\"findings\":[],\"confidence\":0.95,\"blastRadiusRaise\":{\"to\":\"HIGH\"}}'
1440
+ gate_near_miss "a report that is not an object" invalid '\"just a string\"'
1441
+
1442
+ # The prompt puts the report last and the prose above it routinely quotes json from the diff
1443
+ # under review. Reading the first fence handed the decision to whatever the agent quoted, and
1444
+ # PROTECTED_PATHS itself names package.json and global.json, so the reviewed diff is often
1445
+ # json. The decoy here claims everything is fine; the real report blocks.
1446
+ # Built with jq rather than hand-escaped: this body has two fenced blocks, each containing
1447
+ # quoted json, inside a json string. Hand-escaping it is how a test ends up asserting on a
1448
+ # fixture that does not parse.
1449
+ gate_decoy=$(jq -nc --arg body "$(printf '%s\n' '<!-- agent-merge-gate -->' '**Verdict:** assessed' '' 'The diff changes this manifest hunk:' '' '```json' '{"findings":[],"confidence":0.95}' '```' '' 'Report:' '' '```json' '{"findings":[{"verified":true,"severity":"critical"}],"confidence":0.95}' '```')" \
1450
+ '{items:[{type:"add_comment",item_number:7,body:$body}]}')
1451
+ gate_case "the last json fence is the report, not the first" blocked "$gate_decoy" success low
1452
+
1453
+ # Verdict and report used to be selected independently, and each took the first it found, so a
1454
+ # second comment reporting a verified critical finding was discarded and a comment with no
1455
+ # verdict could supply the report for a verdict written in another.
1456
+ gate_case "two comments carrying a verdict" \
1457
+ invalid "$(gate_items "$(gate_comment assessed),$(gate_comment assessed "$gate_verified")")" success low
1458
+
1459
+ # An empty measured fact is a job that did not report, not a low-risk pull request. `${4:-low}`
1460
+ # substituted the default for an empty argument, so a skipped protected_changes read as
1461
+ # "low, nothing protected" and merged.
1462
+ gate_unmeasured=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success "" "" "" 0.8 2>&1 || true)
1463
+ printf '%s' "$(gate_items "$(gate_comment assessed)")" > "$gate_fixture"
1464
+ gate_unmeasured=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success "" "" "" 0.8 2>&1 || true)
1465
+ if [ "$gate_unmeasured" != invalid ]; then
1466
+ VALIDATOR_OK=0
1467
+ echo "FAIL: an unmeasured blast radius produced '${gate_unmeasured}', expected invalid" >&2
1468
+ fi
1469
+ gate_half=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 success low "" "" 0.8 2>&1 || true)
1470
+ if [ "$gate_half" != human-review ]; then
1471
+ VALIDATOR_OK=0
1472
+ echo "FAIL: an unmeasured protected-path fact produced '${gate_half}', expected human-review" >&2
1473
+ fi
1474
+
1475
+ gate_case "an empty item list" invalid '{"items":[]}' success low
1476
+ gate_case "output that is not an item list" invalid '{"nope":true}' success low
1214
1477
 
1215
1478
  rm -f "$gate_fixture"
1216
1479
  if [ "$VALIDATOR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
@@ -1711,6 +1974,43 @@ if worker_installed merge-gate; then
1711
1974
  echo "FAIL: identify-gate-subject resolves the CI run only when the conclusion is unknown; a router that passes both gets an empty run ID" >&2
1712
1975
  fi
1713
1976
  fi
1977
+ # Repeating a report the validator refused reproduces it, and every repeat is a fresh agent run
1978
+ # holding the repo-wide merge-belt slot. The worker gives that failure a smaller budget than a
1979
+ # crash, and -- the part that actually saves the slot -- records it as a verdict, because the belt
1980
+ # bounds its own retries by counting attempt comments and would otherwise dispatch to the cap
1981
+ # whatever the worker decided.
1982
+ if worker_installed merge-gate; then
1983
+ UNUSABLE_OK=1
1984
+ unusable_cap="$(sed -n 's/^ PARK_AT_UNUSABLE_OUTPUT: "\([0-9]*\)"$/\1/p' "$MERGE_GATE_WORKER_MD" | head -1)"
1985
+ machine_cap="$(sed -n 's/^ PARK_AT_ATTEMPT: "\([0-9]*\)"$/\1/p' "$MERGE_GATE_WORKER_MD" | head -1)"
1986
+ if [ -z "$unusable_cap" ] || [ -z "$machine_cap" ] || [ "$unusable_cap" -ge "$machine_cap" ]; then
1987
+ UNUSABLE_OK=0
1988
+ echo "FAIL: an unusable report must get a smaller budget than a crash (unusable='${unusable_cap:-unset}', machine='${machine_cap:-unset}')" >&2
1989
+ fi
1990
+ # The budget is decided once, not restated per step. Four `if:` expressions repeating the same
1991
+ # pair of conditions is the shape the protected-files hold already got wrong.
1992
+ if [ "$(count -cE "^ if: steps\.budget\.outputs\.park" "$MERGE_GATE_WORKER_MD")" -lt 3 ]; then
1993
+ UNUSABLE_OK=0
1994
+ echo "FAIL: the incomplete job must read one computed budget decision, not re-derive it" >&2
1995
+ fi
1996
+ # The park has to be a verdict or the belt keeps dispatching: a comment carrying the gate
1997
+ # marker AND a Verdict line is what detect-pr-conflicts and the reconcile belt both park on.
1998
+ if ! grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
1999
+ grep -q '\${{ env.GATE_MARKER }}' ||
2000
+ ! grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
2001
+ grep -q '\*\*Verdict:\*\* human-review'; then
2002
+ UNUSABLE_OK=0
2003
+ echo "FAIL: the unusable-report park must carry the gate marker and a Verdict line, or the belt dispatches it again" >&2
2004
+ fi
2005
+ # And it must not also count as an attempt, or one park is recorded twice.
2006
+ if grep -A 16 "Record an unusable report as a decision" "$MERGE_GATE_WORKER_MD" |
2007
+ grep -q '\${{ env.ATTEMPT_MARKER }}'; then
2008
+ UNUSABLE_OK=0
2009
+ echo "FAIL: the unusable-report park carries the attempt marker as well; it is a decision, not an attempt" >&2
2010
+ fi
2011
+ if [ "$UNUSABLE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
2012
+ fi
2013
+
1714
2014
  if [ "$GATE_RUN_ID_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1715
2015
  fi
1716
2016