@plainconceptsplatform/workflows 0.6.1 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/README.md +105 -88
  2. package/dist/catalog-installation.d.ts +39 -2
  3. package/dist/catalog-installation.js +172 -109
  4. package/dist/index.js +128 -73
  5. package/dist/package-baseline.d.ts +25 -0
  6. package/dist/package-baseline.js +138 -0
  7. package/dist/route-processing.d.ts +0 -2
  8. package/dist/route-processing.js +22 -91
  9. package/dist/stack-defaults.js +18 -12
  10. package/dist/tui.js +27 -43
  11. package/dist/worker-env.d.ts +46 -0
  12. package/dist/worker-env.js +179 -0
  13. package/dist/workflow-catalog.d.ts +4 -2
  14. package/dist/workflow-catalog.js +5 -3
  15. package/loops/actions/add-issue-labels/action.yml +20 -0
  16. package/loops/actions/audit-close/action.yml +180 -128
  17. package/loops/actions/classify-route/classify-route.sh +8 -2
  18. package/loops/actions/housekeeping/action.yml +251 -0
  19. package/loops/actions/merge-agent-pr/action.yml +13 -0
  20. package/loops/actions/report-workflow-errors/action.yml +385 -0
  21. package/loops/actions/validate-merge-gate-output/validate-merge-gate-output.sh +13 -1
  22. package/loops/actions/validate-triage-output/action.yml +1 -1
  23. package/loops/actions/validate-triage-output/validate-triage-output.sh +9 -5
  24. package/loops/actions/verify-composite-actions/verify-composite-actions.sh +53 -0
  25. package/loops/actions/verify-route-matrix/verify-route-matrix.sh +847 -170
  26. package/loops/templates/agentics/agentics-error-report.yml +97 -0
  27. package/loops/templates/opencode/opencode.ci.json +1 -1
  28. package/loops/workflows/agent-apply-review.md +452 -469
  29. package/loops/workflows/agent-audit.md +201 -213
  30. package/loops/workflows/agent-implement.md +616 -640
  31. package/loops/workflows/agent-merge-gate.md +830 -844
  32. package/loops/workflows/agent-refine.md +599 -633
  33. package/loops/workflows/agent-release.md +244 -258
  34. package/loops/workflows/agent-triage.md +476 -447
  35. package/loops/workflows/authorize-bot-work.yml +26 -6
  36. package/loops/workflows/work-router.yml +1185 -1038
  37. package/package.json +9 -8
  38. package/dist/action-validation.test.d.ts +0 -1
  39. package/dist/action-validation.test.js +0 -87
  40. package/dist/catalog-installation.test.d.ts +0 -1
  41. package/dist/catalog-installation.test.js +0 -485
  42. package/dist/catalog-listing.test.d.ts +0 -1
  43. package/dist/catalog-listing.test.js +0 -150
  44. package/dist/index.test.d.ts +0 -1
  45. package/dist/index.test.js +0 -273
  46. package/dist/repository-inspection.test.d.ts +0 -1
  47. package/dist/repository-inspection.test.js +0 -77
  48. package/dist/route-processing.test.d.ts +0 -1
  49. package/dist/route-processing.test.js +0 -283
  50. package/dist/stack-defaults.test.d.ts +0 -1
  51. package/dist/stack-defaults.test.js +0 -266
  52. package/dist/tui.test.d.ts +0 -1
  53. package/dist/tui.test.js +0 -249
  54. package/dist/workflow-catalog.test.d.ts +0 -1
  55. package/dist/workflow-catalog.test.js +0 -29
  56. package/loops/actions/stale-recovery/action.yml +0 -288
  57. package/loops/actions/update-changelog/action.yml +0 -113
@@ -3,23 +3,69 @@
3
3
  # Exercise the router's real classifier. This sources classify-route.sh rather than
4
4
  # restating it, so a change to the route table cannot pass here by being copied twice.
5
5
  #
6
+ # The same file runs in every consumer, whatever subset of workers it installed: the router is
7
+ # regenerated for that subset, so every assertion about a worker or about its router job is
8
+ # conditional on the worker file being present. The classifier is the complete route table in
9
+ # every repository (a route with no job is a no-op run), so its assertions are unconditional.
10
+ #
6
11
  # This file greps workflow sources for literal `${{ ... }}` expressions on purpose.
7
12
  # shellcheck disable=SC2016
8
13
 
9
14
  set -euo pipefail
10
15
 
11
16
  HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
12
- ROUTER_YML="${HERE}/../../workflows/work-router.yml"
13
- IMPLEMENT_WORKER_MD="${HERE}/../../workflows/agent-implement.md"
14
- MERGE_GATE_WORKER_MD="${HERE}/../../workflows/agent-merge-gate.md"
17
+ WORKFLOWS_DIR="${HERE}/../../workflows"
18
+ ROUTER_YML="${WORKFLOWS_DIR}/work-router.yml"
19
+ IMPLEMENT_WORKER_MD="${WORKFLOWS_DIR}/agent-implement.md"
20
+ MERGE_GATE_WORKER_MD="${WORKFLOWS_DIR}/agent-merge-gate.md"
21
+
22
+ # The audit slot is per repository and lives in the router's own env: block, which a real run
23
+ # exports into the classify step. Export it here too, or this file would test the classifier's
24
+ # fallback rather than the cron the router actually fires on.
25
+ router_env() {
26
+ sed -n "s/^ $1: *//p" "$ROUTER_YML" | head -1 | sed -e 's/^"\(.*\)"$/\1/' -e "s/^'\(.*\)'$/\1/"
27
+ }
28
+ AUDIT_CRON="$(router_env AUDIT_CRON)"
29
+ export AUDIT_CRON
15
30
 
16
31
  # shellcheck source-path=SCRIPTDIR
17
32
  # shellcheck source=../classify-route/classify-route.sh
18
33
  source "${HERE}/../classify-route/classify-route.sh"
19
34
 
35
+ # Every worker route the package knows. The ones with a worker file here are installed; the
36
+ # others must have no job in this router. Plumbing routes are in every router.
37
+ ALL_WORKER_ROUTES=(refine implement triage apply-review merge-gate audit release)
38
+ PLUMBING_ROUTES=(bot-approve audit-close cleanup-artifacts reconcile-bot-pr-runs housekeeping validate)
39
+
40
+ worker_installed() {
41
+ [ -f "${WORKFLOWS_DIR}/agent-$1.md" ]
42
+ }
43
+
44
+ INSTALLED_ROUTES=()
45
+ EXCLUDED_ROUTES=()
46
+ for route in "${ALL_WORKER_ROUTES[@]}"; do
47
+ if worker_installed "$route"; then
48
+ INSTALLED_ROUTES+=("$route")
49
+ else
50
+ EXCLUDED_ROUTES+=("$route")
51
+ fi
52
+ done
53
+ echo "Installed workers: ${INSTALLED_ROUTES[*]:-(none)}"
54
+ [ "${#EXCLUDED_ROUTES[@]}" -eq 0 ] || echo "Not installed: ${EXCLUDED_ROUTES[*]}"
55
+
20
56
  PASS=0
21
57
  FAIL=0
22
58
 
59
+ # `grep -c` exits 1 when it counts zero, and this file runs under `set -e`, so writing
60
+ # `n=$(grep -c ...)` against a pattern that is absent ended the whole suite at whatever section
61
+ # it had reached, with no error printed and no FAIL counted. That made an assertion of the form
62
+ # "this pattern must be GONE" impossible to write here: the moment it held, the suite died. Every
63
+ # count goes through this instead, where zero is an answer rather than a failure. Callers pass
64
+ # their own grep flags.
65
+ count() {
66
+ grep "$@" 2>/dev/null || true
67
+ }
68
+
23
69
  # Classify one event and read a single field out of the result.
24
70
  route_field() {
25
71
  local field="$1"
@@ -106,9 +152,6 @@ assert_route "the bot's own comment never re-enters refine" none \
106
152
  assert_route "a comment on an issue without refine routes nowhere" none \
107
153
  EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
108
154
  'ISSUE_LABELS=["bug"]' EVENT_ISSUE_NUMBER=42
109
- assert_route "the bot's own comment never re-enters direct" none \
110
- EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=Bot \
111
- 'ISSUE_LABELS=["direct"]' EVENT_ISSUE_NUMBER=42
112
155
  assert_route "a comment on a triage issue re-triages" triage \
113
156
  EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
114
157
  'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
@@ -120,12 +163,22 @@ assert_route "the bot's own comment never re-enters triage" none \
120
163
  'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
121
164
 
122
165
  echo "── Closed issues ─────────────────────────────────────────────────────────"
123
- assert_route "a closing comment on a refine issue does not re-refine" none EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User ISSUE_STATE=closed 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
124
- assert_route "a comment on a closed issue never re-triages" none EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User ISSUE_STATE=closed 'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
125
- assert_route "a work label added to a closed issue routes nowhere" none EVENT=issues ACTION=labeled LABEL=bot-working ISSUE_STATE=closed 'ISSUE_LABELS=["implement"]' EVENT_ISSUE_NUMBER=42
126
- assert_route "a closed issue reopened as opened still routes nowhere while closed" none EVENT=issues ACTION=opened ISSUE_STATE=closed 'ISSUE_LABELS=[]' EVENT_ISSUE_NUMBER=42
127
- assert_route "a comment on a closed pull request still routes to apply-review" apply-review EVENT=issue_comment COMMENT_ON_PR=true ISSUE_STATE=closed EVENT_ISSUE_NUMBER=7
128
- assert_route "an open refine issue is unaffected by the closed guard" refine EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User ISSUE_STATE=open 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
166
+ assert_route "a closing comment on a refine issue does not re-refine" none \
167
+ EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
168
+ ISSUE_STATE=closed 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
169
+ assert_route "a comment on a closed issue never re-triages" none \
170
+ EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
171
+ ISSUE_STATE=closed 'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
172
+ assert_route "a work label added to a closed issue routes nowhere" none \
173
+ EVENT=issues ACTION=labeled LABEL=bot-working ISSUE_STATE=closed \
174
+ 'ISSUE_LABELS=["implement"]' EVENT_ISSUE_NUMBER=42
175
+ assert_route "a closed issue reopened as opened still routes nowhere while closed" none \
176
+ EVENT=issues ACTION=opened ISSUE_STATE=closed 'ISSUE_LABELS=[]' EVENT_ISSUE_NUMBER=42
177
+ assert_route "a comment on a closed pull request still routes to apply-review" apply-review \
178
+ EVENT=issue_comment COMMENT_ON_PR=true ISSUE_STATE=closed EVENT_ISSUE_NUMBER=7
179
+ assert_route "an open refine issue is unaffected by the closed guard" refine \
180
+ EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
181
+ ISSUE_STATE=open 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
129
182
 
130
183
  echo "── Review events ─────────────────────────────────────────────────────────"
131
184
  assert_route "a review comment routes to apply-review" apply-review \
@@ -162,7 +215,7 @@ while read -r cron; do
162
215
  PASS=$((PASS + 1))
163
216
  echo " ${cron} -> ${selected}"
164
217
  fi
165
- done < <(sed -n 's/^ *- cron: "\(.*\)"$/\1/p' "$ROUTER_YML")
218
+ done < <(sed -n 's/^ *- cron: "\([^"]*\)".*/\1/p' "$ROUTER_YML")
166
219
 
167
220
  assert_route "an unknown cron routes nowhere" none EVENT=schedule "SCHEDULE=0 0 30 2 *"
168
221
 
@@ -173,16 +226,12 @@ assert_route "refine dispatch rejects a non-numeric issue" none \
173
226
  EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=abc
174
227
  assert_route "refine dispatch accepts a positive issue" refine \
175
228
  EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=42
176
- assert_route "direct dispatch needs an issue number" none \
177
- EVENT=workflow_dispatch OPERATION=direct INPUT_ISSUE_NUMBER=
178
229
  assert_route "triage dispatch accepts a positive issue" triage \
179
230
  EVENT=workflow_dispatch OPERATION=triage INPUT_ISSUE_NUMBER=42
180
231
  assert_route "triage dispatch needs an issue number" none \
181
232
  EVENT=workflow_dispatch OPERATION=triage INPUT_ISSUE_NUMBER=
182
233
  assert "triage dispatch defaults to first pass" first \
183
234
  "$(route_field triage-mode EVENT=workflow_dispatch OPERATION=triage INPUT_ISSUE_NUMBER=42)"
184
- assert_route "batch dispatch needs an issue number" none \
185
- EVENT=workflow_dispatch OPERATION=batch INPUT_ISSUE_NUMBER=
186
235
  assert_route "merge-gate dispatch needs a pull request number" none \
187
236
  EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=0
188
237
  assert_route "merge-gate dispatch accepts a positive pull request" merge-gate \
@@ -199,6 +248,8 @@ assert "implement dispatch forwards the attempt count" 2 \
199
248
  "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=implement INPUT_ISSUE_NUMBER=42 INPUT_ATTEMPTS_SO_FAR=2)"
200
249
  assert "a refine dispatch carries no implement attempts" 0 \
201
250
  "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=42 INPUT_ATTEMPTS_SO_FAR=2)"
251
+ assert_route "release dispatch needs no numbers" release \
252
+ EVENT=workflow_dispatch OPERATION=release
202
253
  assert_route "reconcile-bot-pr-runs dispatch needs no numbers" reconcile-bot-pr-runs \
203
254
  EVENT=workflow_dispatch OPERATION=reconcile-bot-pr-runs
204
255
  assert_route "an unknown operation routes nowhere" none \
@@ -209,13 +260,199 @@ assert "a scheduled audit reports its trigger kind" scheduled \
209
260
  assert "a dispatched audit reports its trigger kind" manual \
210
261
  "$(route_field trigger-kind EVENT=workflow_dispatch OPERATION=audit INPUT_TRIGGER_KIND=manual)"
211
262
 
263
+ echo "── Prompt hygiene ────────────────────────────────────────────────────────"
264
+
265
+ # Everything below a worker's frontmatter is the prompt. Three things must not be in one.
266
+ #
267
+ # A `gh` call, because the shared CI agent config says the GitHub CLI is intentionally
268
+ # unauthenticated and the agent must never use it for GitHub reads or writes. A prompt that
269
+ # orders one burns turns and fails; implement carried a `gh pr list` for weeks, asking the agent
270
+ # to redo a check the router had already done before dispatching it.
271
+ #
272
+ # A duplicate step number, because these are ordered instruction lists and a step that says
273
+ # "go to step 6" cannot resolve when there are two. implement had two 6s with contradictory
274
+ # rules ("exactly one" and "at least one" safe output), refine had two 5s, apply-review two 9s.
275
+ #
276
+ # A Mermaid diagram, because it is documentation that the model is charged for on every run and
277
+ # then told to ignore. They live in docs/diagrams.md.
278
+ PROMPT_OK=1
279
+ for worker in "${WORKFLOWS_DIR}"/agent-*.md; do
280
+ [ -f "$worker" ] || continue
281
+ name=$(basename "$worker")
282
+ # The prompt starts after the closing --- of the frontmatter.
283
+ fm_end=$(awk 'NR>1 && /^---[[:space:]]*$/{print NR; exit}' "$worker")
284
+ [ -n "$fm_end" ] || { PROMPT_OK=0; echo "FAIL: ${name} has no frontmatter terminator" >&2; continue; }
285
+ prompt=$(tail -n "+$((fm_end + 1))" "$worker")
286
+
287
+ if grep -qE '(^|[^[:alnum:]_-])gh (pr|issue|api|run|release|workflow|auth) ' <<<"$prompt"; then
288
+ PROMPT_OK=0
289
+ echo "FAIL: ${name}'s prompt tells the agent to run gh; the CLI is unauthenticated in CI" >&2
290
+ grep -nE '(^|[^[:alnum:]_-])gh (pr|issue|api|run|release|workflow|auth) ' <<<"$prompt" >&2
291
+ fi
292
+
293
+ if grep -q '```mermaid' <<<"$prompt"; then
294
+ PROMPT_OK=0
295
+ echo "FAIL: ${name}'s prompt contains a Mermaid diagram; diagrams belong in docs/diagrams.md" >&2
296
+ fi
297
+
298
+ # Top-level steps only: an indented "1." is a sub-list and numbers restart legitimately.
299
+ dupes=$(grep -oE '^[0-9]+\. ' <<<"$prompt" | tr -d '. ' | sort -n | uniq -d | tr '\n' ' ')
300
+ if [ -n "${dupes// /}" ]; then
301
+ PROMPT_OK=0
302
+ echo "FAIL: ${name}'s prompt repeats step number(s): ${dupes}" >&2
303
+ fi
304
+
305
+ # A multi-line env value does not survive compilation: gh-aw joins it onto one line in the
306
+ # lock, so a Markdown table written across five lines in the source reaches the agent as a
307
+ # single unreadable row. Verified against a compiled lock. Block the YAML block scalars that
308
+ # produce one, so the flattening is a failed check rather than a silently useless value.
309
+ block_scalars=$(awk '
310
+ /^env:$/ { inenv = 1; next }
311
+ inenv && /^[^ ]/ { inenv = 0 }
312
+ inenv && /^ [A-Za-z_][A-Za-z0-9_]*: *[|>]-?[0-9]* *$/ { print $1 }
313
+ ' "$worker" | tr -d ':' | tr '\n' ' ')
314
+ if [ -n "${block_scalars// /}" ]; then
315
+ PROMPT_OK=0
316
+ echo "FAIL: ${name} declares env value(s) as a multi-line block, which the compiler flattens: ${block_scalars}" >&2
317
+ fi
318
+ done
319
+ if [ "$PROMPT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
320
+
212
321
  echo "── Router wiring ─────────────────────────────────────────────────────────"
213
322
 
323
+ # Two router values are needed where GitHub evaluates no expression, so the installer mirrors
324
+ # them out of env: into a literal. A copy that drifts fails in the direction that hurts: the
325
+ # gate silently reads a CI workflow nobody runs, or the audit fires on a cron the classifier
326
+ # maps to no route. Neither produces a red run, so assert the copies here.
327
+ MIRROR_OK=1
328
+ ci_name="$(router_env CI_WORKFLOW_NAME)"
329
+ trigger_name="$(sed -n 's/^ *workflows: \["\(.*\)"\] *$/\1/p' "$ROUTER_YML" | head -1)"
330
+ if [ -z "$ci_name" ]; then
331
+ MIRROR_OK=0; echo "FAIL: work-router.yml defines no CI_WORKFLOW_NAME in its env: block" >&2
332
+ elif [ "$ci_name" != "$trigger_name" ]; then
333
+ MIRROR_OK=0
334
+ echo "FAIL: the workflow_run trigger names '${trigger_name}' but env.CI_WORKFLOW_NAME is '${ci_name}'" >&2
335
+ fi
336
+ # The audit cron only exists in a router that installed the audit worker.
337
+ if worker_installed audit; then
338
+ cron_line="$(sed -n 's/^ *- cron: "\([^"]*\)" # audit slot.*/\1/p' "$ROUTER_YML" | head -1)"
339
+ if [ -z "$AUDIT_CRON" ]; then
340
+ MIRROR_OK=0; echo "FAIL: work-router.yml defines no AUDIT_CRON in its env: block" >&2
341
+ elif [ "$AUDIT_CRON" != "$cron_line" ]; then
342
+ MIRROR_OK=0
343
+ echo "FAIL: the audit slot cron is '${cron_line}' but env.AUDIT_CRON is '${AUDIT_CRON}'" >&2
344
+ fi
345
+ fi
346
+ # And nothing may go back to naming the CI workflow directly: a second literal is a second
347
+ # thing to keep in step, and the one that gets forgotten is the one inside a jq filter.
348
+ if [ "$(count -c '"App: CI"' "$ROUTER_YML")" -gt 2 ]; then
349
+ MIRROR_OK=0
350
+ echo "FAIL: work-router.yml hardcodes the CI workflow name outside env: and the mirrored trigger" >&2
351
+ grep -n '"App: CI"' "$ROUTER_YML" >&2
352
+ fi
353
+
354
+ # Bot logins are the same shape of problem. Most sites read env.TRUSTED_BOTS, but a job-level
355
+ # `if:` cannot: GitHub does not expose the env context there, so bot-approve keeps literals and
356
+ # they have to agree. Assert every bot login written anywhere in the router is in the list.
357
+ trusted="$(router_env TRUSTED_BOTS)"
358
+ if [ -z "$trusted" ]; then
359
+ MIRROR_OK=0; echo "FAIL: work-router.yml defines no TRUSTED_BOTS in its env: block" >&2
360
+ else
361
+ while IFS= read -r login; do
362
+ [ -n "$login" ] || continue
363
+ case " $trusted " in
364
+ *" $login "*) ;;
365
+ *)
366
+ MIRROR_OK=0
367
+ echo "FAIL: work-router.yml names bot '${login}' but env.TRUSTED_BOTS does not list it" >&2
368
+ ;;
369
+ esac
370
+ done < <(grep -oE "'(app/[a-z-]+|[a-z-]+\[bot\])'|\"(app/[a-z-]+|[a-z-]+\[bot\])\"" "$ROUTER_YML" |
371
+ tr -d "'\"" | sort -u)
372
+ fi
373
+ if [ "$MIRROR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
374
+
375
+ # Run the belt's own jq, rather than reading it. The filter that picks which open pull requests
376
+ # the hourly reconcile job acts on was written as
377
+ # ((env.TRUSTED_BOTS | split(" ")) | index(.user.login) != null)
378
+ # which dies at runtime with `Cannot index array with string "user"`, because inside index() the
379
+ # input is the array, not the pull request. Every assertion here passed: one checked the bot
380
+ # logins were listed in TRUSTED_BOTS, another that the router named no bot outside that list.
381
+ # Nothing executed the program. It failed hourly in production for a day, on the one job whose
382
+ # purpose is to keep stuck pull requests moving. Extract it and give it inputs.
383
+ BELT_OK=1
384
+ bot_pr_filter=$(awk '/jq -r --arg repo "\$REPO"/{found=1;next} found && /^ *'"'"' \|$/{exit} found' "$ROUTER_YML")
385
+ if [ -z "$bot_pr_filter" ]; then
386
+ BELT_OK=0
387
+ echo "FAIL: could not extract the open-pull-request filter from work-router.yml" >&2
388
+ else
389
+ # One of each: a trusted App under both spellings, a human, a draft, and a fork.
390
+ belt_fixture='[
391
+ {"number":11,"draft":false,"user":{"login":"app/github-actions"},"head":{"ref":"a","sha":"s1","repo":{"full_name":"o/r"}}},
392
+ {"number":12,"draft":false,"user":{"login":"platform-devbox[bot]"},"head":{"ref":"b","sha":"s2","repo":{"full_name":"o/r"}}},
393
+ {"number":13,"draft":false,"user":{"login":"a-person"},"head":{"ref":"c","sha":"s3","repo":{"full_name":"o/r"}}},
394
+ {"number":14,"draft":true,"user":{"login":"app/github-actions"},"head":{"ref":"d","sha":"s4","repo":{"full_name":"o/r"}}},
395
+ {"number":15,"draft":false,"user":{"login":"app/github-actions"},"head":{"ref":"e","sha":"s5","repo":{"full_name":"fork/r"}}}
396
+ ]'
397
+ if ! selected=$(printf '%s' "$belt_fixture" |
398
+ TRUSTED_BOTS="$trusted" jq -r --arg repo "o/r" "$bot_pr_filter" 2>&1 | cut -f1 | tr '\n' ' '); then
399
+ BELT_OK=0
400
+ echo "FAIL: the open-pull-request filter does not run: ${selected}" >&2
401
+ elif [ "$(echo "$selected" | tr -s ' ')" != "11 12 " ]; then
402
+ BELT_OK=0
403
+ echo "FAIL: the belt selected pull requests [${selected}]; expected the two bot-authored ones (11 12)" >&2
404
+ fi
405
+ fi
406
+ if [ "$BELT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
407
+
408
+ # `review` must not be a one-way door. It used to be: authorize-bot-work refused to fire on an
409
+ # issue carrying it, and the classifier refuses to route while it is set, so a person adding
410
+ # `refine` to a parked issue got nothing at all — no run, no comment, no error. Triage's own
411
+ # needs-maintainer verdict tells the maintainer to do exactly that, so the bot was giving an
412
+ # instruction the machine ignored. authorize-bot-work now clears `review` before handing over,
413
+ # which is what makes the human's decision stick.
414
+ AUTHORIZE_YML="${WORKFLOWS_DIR}/authorize-bot-work.yml"
415
+ if [ -f "$AUTHORIZE_YML" ]; then
416
+ DOOR_OK=1
417
+ authorize_if=$(sed -n '/^ if: >/,/^ runs-on:/p' "$AUTHORIZE_YML")
418
+ if grep -q "labels\.\*\.name, 'review'" <<<"$authorize_if"; then
419
+ DOOR_OK=0
420
+ echo "FAIL: authorize-bot-work refuses issues carrying review; a human could not un-park one" >&2
421
+ fi
422
+ # It must still refuse the bot, and an issue another run already owns.
423
+ grep -q "endsWith(github.actor, '\[bot\]')" <<<"$authorize_if" || {
424
+ DOOR_OK=0
425
+ echo "FAIL: authorize-bot-work no longer excludes bot actors; it would re-trigger itself" >&2
426
+ }
427
+ grep -q "labels\.\*\.name, 'bot-working'" <<<"$authorize_if" || {
428
+ DOOR_OK=0
429
+ echo "FAIL: authorize-bot-work no longer excludes an issue a run already owns" >&2
430
+ }
431
+ # The hand-off has to clear review BEFORE adding bot-working, because bot-working is the event
432
+ # the classifier reads: the other order raises an event whose payload still carries review.
433
+ remove_line=$(grep -n -- '--remove-label "review"' "$AUTHORIZE_YML" | head -1 | cut -d: -f1)
434
+ add_line=$(grep -n -- '--add-label "bot-working"' "$AUTHORIZE_YML" | head -1 | cut -d: -f1)
435
+ if [ -z "$remove_line" ] || [ -z "$add_line" ] || [ "$remove_line" -ge "$add_line" ]; then
436
+ DOOR_OK=0
437
+ echo "FAIL: authorize-bot-work must remove review before adding bot-working (review=${remove_line:-none} bot-working=${add_line:-none})" >&2
438
+ fi
439
+ if [ "$DOOR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
440
+ fi
441
+
442
+ # The classifier's own guard stays: it stops the bot re-triggering itself while a human is
443
+ # needed. Both halves matter, so assert the pair rather than either alone.
444
+ assert_route "a bot-working event on a review-labelled issue still routes nowhere" none \
445
+ EVENT=issues ACTION=labeled LABEL=bot-working ACTOR=platform-devbox[bot] \
446
+ 'ISSUE_LABELS=["implement","review"]' EVENT_ISSUE_NUMBER=42
447
+ assert_route "and routes normally once review has been cleared" implement \
448
+ EVENT=issues ACTION=labeled LABEL=bot-working ACTOR=platform-devbox[bot] \
449
+ 'ISSUE_LABELS=["implement"]' EVENT_ISSUE_NUMBER=42
450
+
214
451
  # GitHub evaluates every Actions expression in a workflow file, including ones written inside
215
452
  # shell comments. An empty pair is not a valid expression and fails the whole file to parse,
216
453
  # with an error that points at a line number rather than saying what is wrong. Prose about
217
454
  # expressions must not contain one.
218
- empty_expr=$(grep -rl -e '${{[[:space:]]*}}' "${HERE}/../../workflows"/*.yml "${HERE}/../../workflows"/*.md 2>/dev/null || true)
455
+ empty_expr=$(grep -rl -e '${{[[:space:]]*}}' "${WORKFLOWS_DIR}"/*.yml "${WORKFLOWS_DIR}"/*.md 2>/dev/null || true)
219
456
  if [ -z "$empty_expr" ]; then
220
457
  PASS=$((PASS + 1))
221
458
  else
@@ -224,15 +461,16 @@ else
224
461
  while IFS= read -r offending; do echo " $offending" >&2; done <<<"$empty_expr"
225
462
  fi
226
463
 
227
-
228
464
  # A hyphen inside a ${{ }} property path is parsed as subtraction, so the reference silently
229
465
  # resolves to nothing and the rendered prompt keeps the raw expression. Underscores only.
230
- if ! grep -qE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD"; then
231
- PASS=$((PASS + 1))
232
- else
233
- FAIL=$((FAIL + 1))
234
- echo "FAIL: implement worker reads a hyphenated job output inside an expression" >&2
235
- grep -nE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD" >&2
466
+ if worker_installed implement; then
467
+ if ! grep -qE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD"; then
468
+ PASS=$((PASS + 1))
469
+ else
470
+ FAIL=$((FAIL + 1))
471
+ echo "FAIL: implement worker reads a hyphenated job output inside an expression" >&2
472
+ grep -nE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD" >&2
473
+ fi
236
474
  fi
237
475
 
238
476
  # A worker that prints `${{ env.NAME }}` without defining NAME in its own env: block renders
@@ -241,7 +479,8 @@ fi
241
479
  # worker prints must be defined in that worker. The values are consumer-owned (a consumer may
242
480
  # split VERIFY_COMMANDS per area, or keep one); only the wiring is asserted here.
243
481
  VERIFY_OK=1
244
- for worker in "${HERE}/../../workflows"/agent-*.md; do
482
+ for worker in "${WORKFLOWS_DIR}"/agent-*.md; do
483
+ [ -f "$worker" ] || continue
245
484
  while read -r name; do
246
485
  [ -n "$name" ] || continue
247
486
  if ! grep -q "^ ${name}:" "$worker"; then
@@ -252,13 +491,29 @@ for worker in "${HERE}/../../workflows"/agent-*.md; do
252
491
  done
253
492
  if [ "$VERIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
254
493
 
255
- if grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" &&
256
- grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" &&
257
- grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.subject.outputs.conclusion == 'failure'" "$MERGE_GATE_WORKER_MD"; then
258
- PASS=$((PASS + 1))
259
- else
260
- FAIL=$((FAIL + 1))
261
- echo "FAIL: protected changes must allow failed-CI repair while remaining held from merge" >&2
494
+ # A protected path holds the merge for a human but must never stop the agent repairing failed
495
+ # CI on those same files, or the pull request strands with nobody able to fix it. That pair of
496
+ # conditions is decided once, in protected_changes.outputs.holds_review, and read everywhere
497
+ # else; it used to be restated at eight call sites. Auto-merge stays blocked separately, by
498
+ # conclude's own guard on requires_review, which holds even when CI failed.
499
+ if worker_installed implement && worker_installed merge-gate; then
500
+ PROTECTED_OK=1
501
+ grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" || PROTECTED_OK=0
502
+ grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
503
+ grep -Fq "holds_review: \${{ steps.files.outputs.requires_review == 'true' && needs.subject.outputs.conclusion != 'failure' }}" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
504
+ # The decision must not be re-derived anywhere: one definition, everything else reads it.
505
+ if [ "$(count -c "requires_review == 'true' && needs.subject.outputs.conclusion != 'failure'" "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
506
+ PROTECTED_OK=0
507
+ echo "FAIL: the protected-files hold is derived in more than one place; read holds_review instead" >&2
508
+ fi
509
+ # And conclude must still refuse to merge a protected pull request whatever CI said.
510
+ grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
511
+ if [ "$PROTECTED_OK" -eq 1 ]; then
512
+ PASS=$((PASS + 1))
513
+ else
514
+ FAIL=$((FAIL + 1))
515
+ echo "FAIL: protected changes must allow failed-CI repair while remaining held from merge" >&2
516
+ fi
262
517
  fi
263
518
 
264
519
  # gh-aw folds the worker's top-level `if:` into the generated activation job but computes
@@ -266,56 +521,59 @@ fi
266
521
  # `needs:` are hoisted. A guard with its own `needs:` (protected_changes needs subject) is read
267
522
  # before it has run, resolves to '' and gates nothing, unless it is listed in `on.needs`, the
268
523
  # documented way to add jobs to pre_activation and activation. Inline list form is expected.
269
- TOP_IF="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n 's/^if: //p')"
270
- ON_NEEDS="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n '/^on:$/,/^[a-z]/p' |
271
- sed -n 's/^ needs: *\[\(.*\)\].*/\1/p' | tr -d ' ' | tr ',' '\n')"
272
- ACTIVATION_OK=1
273
- [ -n "$TOP_IF" ] || { ACTIVATION_OK=0; echo "FAIL: could not read the merge-gate worker's top-level if" >&2; }
274
- while read -r job; do
275
- [ -n "$job" ] || continue
276
- if tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n "/^ ${job}:$/,/^ [a-z_]*:$/p" | grep -q '^ needs:' &&
277
- ! grep -qx "$job" <<<"$ON_NEEDS"; then
278
- ACTIVATION_OK=0
279
- echo "FAIL: merge-gate top-level if reads needs.${job}, which has its own needs and is not in on.needs; activation would read it before it runs" >&2
280
- fi
281
- done < <(grep -oE 'needs\.[a-z_]+\.' <<<"$TOP_IF" | sed 's/^needs\.//; s/\.$//' | sort -u)
282
- if [ "$ACTIVATION_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
283
-
284
- # The merge belt is serial for the whole repository: several overnight pull requests
285
- # mean every merge moves the default branch under the rest, and gates running at once
286
- # rebase onto bases other gates are about to invalidate. A per-issue group here would
287
- # reintroduce that race, so assert the repo-wide lock is the one in use.
288
- if grep -A7 'call-merge-gate:' "$ROUTER_YML" | grep -q 'group: merge-belt'; then
289
- PASS=$((PASS + 1))
290
- else
291
- FAIL=$((FAIL + 1))
292
- echo "FAIL: call-merge-gate must hold the repo-wide merge-belt lock" >&2
524
+ if worker_installed merge-gate; then
525
+ TOP_IF="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n 's/^if: //p')"
526
+ ON_NEEDS="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n '/^on:$/,/^[a-z]/p' |
527
+ sed -n 's/^ needs: *\[\(.*\)\].*/\1/p' | tr -d ' ' | tr ',' '\n')"
528
+ ACTIVATION_OK=1
529
+ [ -n "$TOP_IF" ] || { ACTIVATION_OK=0; echo "FAIL: could not read the merge-gate worker's top-level if" >&2; }
530
+ while read -r job; do
531
+ [ -n "$job" ] || continue
532
+ if tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n "/^ ${job}:$/,/^ [a-z_]*:$/p" | grep -q '^ needs:' &&
533
+ ! grep -qx "$job" <<<"$ON_NEEDS"; then
534
+ ACTIVATION_OK=0
535
+ echo "FAIL: merge-gate top-level if reads needs.${job}, which has its own needs and is not in on.needs; activation would read it before it runs" >&2
536
+ fi
537
+ done < <(grep -oE 'needs\.[a-z_]+\.' <<<"$TOP_IF" | sed 's/^needs\.//; s/\.$//' | sort -u)
538
+ if [ "$ACTIVATION_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
539
+
540
+ # The merge belt is serial for the whole repository: several overnight pull requests
541
+ # mean every merge moves the default branch under the rest, and gates running at once
542
+ # rebase onto bases other gates are about to invalidate. A per-issue group here would
543
+ # reintroduce that race, so assert the repo-wide lock is the one in use.
544
+ if grep -A7 'call-merge-gate:' "$ROUTER_YML" | grep -q 'group: merge-belt'; then
545
+ PASS=$((PASS + 1))
546
+ else
547
+ FAIL=$((FAIL + 1))
548
+ echo "FAIL: call-merge-gate must hold the repo-wide merge-belt lock" >&2
549
+ fi
293
550
  fi
294
551
 
295
552
  # A verdict is the gate marker AND a `**Verdict:**` line together. Comments carrying the
296
553
  # marker alone were progress notes and failed attempts, and the reconcile belt read every
297
554
  # one of them as final: a crashed or OOM-killed gate parked its pull request for the rest
298
555
  # of the night. Attempts are counted separately, capped, and reset by any new CI run.
556
+ # The belt lives in the router's plumbing jobs, so this holds in every repository.
299
557
  BELT_OK=1
300
558
  if ! grep -q 'agent-merge-gate-attempt' "$ROUTER_YML"; then
301
559
  BELT_OK=0; echo "FAIL: router never counts gate attempts" >&2
302
560
  fi
303
- if [ "$(grep -cF 'contains("<!-- agent-merge-gate -->")) and (.body | contains("**Verdict:**"))' "$ROUTER_YML")" -lt 4 ]; then
561
+ if [ "$(count -cF 'contains("<!-- agent-merge-gate -->")) and (.body | contains("**Verdict:**"))' "$ROUTER_YML")" -lt 4 ]; then
304
562
  BELT_OK=0; echo "FAIL: verdict detection must pair the gate marker with a Verdict line in both dispatch paths" >&2
305
563
  fi
306
- if [ "$(grep -c 'attempts_so_far' "$ROUTER_YML")" -lt 2 ]; then
564
+ if [ "$(count -c 'attempts_so_far' "$ROUTER_YML")" -lt 2 ]; then
307
565
  BELT_OK=0; echo "FAIL: dispatch sites must forward attempts_so_far" >&2
308
566
  fi
309
567
  # A second gate for a pull request whose gate is already queued or running reads the same CI
310
568
  # verdict and is cancelled by the single-slot merge-belt queue (two cancellations on 2026-09-06).
311
- if [ "$(grep -c 'a merge-gate run is already live' "$ROUTER_YML")" -lt 2 ]; then
569
+ if [ "$(count -c 'a merge-gate run is already live' "$ROUTER_YML")" -lt 2 ]; then
312
570
  BELT_OK=0; echo "FAIL: both dispatch paths must skip a pull request whose gate is already live" >&2
313
571
  fi
314
572
  # A conflicting pull request has no refs/pull/N/merge for GitHub to build, so a `pull_request`
315
573
  # CI workflow can never run on that head. Requiring a fresh verdict before dispatching deadlocks
316
574
  # the belt: only the gate resolves the conflict, and the gate never runs. Both paths fall back to
317
575
  # the branch's last verdict when, and only when, the pull request is conflicting.
318
- if [ "$(grep -c 'conflicts, so CI cannot run on' "$ROUTER_YML")" -lt 2 ]; then
576
+ if [ "$(count -c 'conflicts, so CI cannot run on' "$ROUTER_YML")" -lt 2 ]; then
319
577
  BELT_OK=0
320
578
  echo "FAIL: both dispatch paths must gate a conflicting pull request that can never get fresh CI" >&2
321
579
  fi
@@ -327,8 +585,8 @@ fi
327
585
  # reports "not conflicting" for exactly the stale pull requests the fallback is for. Observed
328
586
  # twice in production: the fallback logged "no completed CI run" for a pull request that
329
587
  # `gh pr view` reported as CONFLICTING from a warm cache seconds later.
330
- if [ "$(grep -c 'mergeable_state()' "$ROUTER_YML")" -lt 2 ] ||
331
- [ "$(grep -c 'mergeable_now=$(mergeable_state' "$ROUTER_YML")" -lt 2 ]; then
588
+ if [ "$(count -c 'mergeable_state()' "$ROUTER_YML")" -lt 2 ] ||
589
+ [ "$(count -c 'mergeable_now=$(mergeable_state' "$ROUTER_YML")" -lt 2 ]; then
332
590
  BELT_OK=0
333
591
  echo "FAIL: both dispatch paths must poll the mergeable state; a single read answers UNKNOWN" >&2
334
592
  fi
@@ -338,7 +596,7 @@ if [ "$BELT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
338
596
  # CI never reaches the router's CI-completion route. The package ships a dispatch-merge-gate job
339
597
  # in templates/ci that hands the verdict over from inside CI; a consumer CI workflow, where one
340
598
  # exists beside the router, must carry it or bot pull requests wait for the hourly belt.
341
- for ci in "${HERE}/../../workflows/ci.yml" "${HERE}/../../workflows/app-ci.yml"; do
599
+ for ci in "${WORKFLOWS_DIR}/ci.yml" "${WORKFLOWS_DIR}/app-ci.yml"; do
342
600
  [ -f "$ci" ] || continue
343
601
  if grep -q 'operation=merge-gate' "$ci"; then
344
602
  PASS=$((PASS + 1))
@@ -368,27 +626,29 @@ while read -r name; do
368
626
  done < <(grep -oE 'needs\.classify\.outputs\.[a-zA-Z0-9_-]+' "$ROUTER_YML" | sed 's/.*\.//' | sort -u)
369
627
  if [ "$CLASSIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
370
628
 
371
- # fromJson('') is a hard failure ("Error reading JToken"), and a workflow_call input arrives as
372
- # '' whenever the caller passes an empty expression, declared default or not. The gate must never
373
- # hand a raw input to fromJson; `inputs.x || '0'` reads the empty case as zero.
374
- if ! grep -qE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD"; then
375
- PASS=$((PASS + 1))
376
- else
377
- FAIL=$((FAIL + 1))
378
- echo "FAIL: merge-gate worker calls fromJson on a raw input; an empty caller value kills the job" >&2
379
- grep -nE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD" >&2
380
- fi
629
+ if worker_installed merge-gate; then
630
+ # fromJson('') is a hard failure ("Error reading JToken"), and a workflow_call input arrives as
631
+ # '' whenever the caller passes an empty expression, declared default or not. The gate must never
632
+ # hand a raw input to fromJson; `inputs.x || '0'` reads the empty case as zero.
633
+ if ! grep -qE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD"; then
634
+ PASS=$((PASS + 1))
635
+ else
636
+ FAIL=$((FAIL + 1))
637
+ echo "FAIL: merge-gate worker calls fromJson on a raw input; an empty caller value kills the job" >&2
638
+ grep -nE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD" >&2
639
+ fi
381
640
 
382
- # The worker's own comments must keep the distinction: progress notes carry no marker,
383
- # failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
384
- # Three verdict sites: the review hold on the issue, the agent's assessment on the issue,
385
- # and conclude's short verdict on the pull request itself.
386
- if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
387
- [ "$(grep -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 3 ]; then
388
- PASS=$((PASS + 1))
389
- else
390
- FAIL=$((FAIL + 1))
391
- echo "FAIL: merge-gate worker must keep verdict and attempt markers distinct" >&2
641
+ # The worker's own comments must keep the distinction: progress notes carry no marker,
642
+ # failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
643
+ # Three verdict sites: the review hold on the issue, the agent's assessment on the issue,
644
+ # and conclude's short verdict on the pull request itself.
645
+ if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
646
+ [ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 3 ]; then
647
+ PASS=$((PASS + 1))
648
+ else
649
+ FAIL=$((FAIL + 1))
650
+ echo "FAIL: merge-gate worker must keep verdict and attempt markers distinct" >&2
651
+ fi
392
652
  fi
393
653
 
394
654
  # add-issue-labels and remove-issue-labels split `labels` on newlines. A caller that joined two
@@ -397,7 +657,7 @@ fi
397
657
  # and review together for a day. Callers use block scalars, one label per line; the actions also
398
658
  # accept commas so a consumer copy of an old caller keeps working.
399
659
  LABELS_OK=1
400
- if grep -nE '^[[:space:]]+labels: [^|>].*,' "${HERE}/../../workflows"/agent-*.md >&2; then
660
+ if grep -nE '^[[:space:]]+labels: [^|>].*,' "${WORKFLOWS_DIR}"/agent-*.md >&2 2>/dev/null; then
401
661
  LABELS_OK=0
402
662
  echo "FAIL: a worker passes comma-joined labels to a label action; use a block scalar, one label per line" >&2
403
663
  fi
@@ -409,83 +669,105 @@ for action in add-issue-labels remove-issue-labels; do
409
669
  done
410
670
  if [ "$LABELS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
411
671
 
412
- # The agent's fix reaches the branch as a bundle applied fast-forward only (apply-agent-output).
413
- # gh-aw's push tool description tells the model to rebase, and a rebased branch cannot
414
- # fast-forward: the push is refused and the verdict is lost (Pliny-Bot run 33952565835). The
415
- # worker must start on the pull request branch and must never say `git rebase`. Its progress
416
- # comment is posted on the first attempt only; retries are recorded by the attempt comment.
417
- BRANCH_OK=1
418
- # Path B: staged safe outputs, applied by conclude with the App token. Without `staged: true`
419
- # gh-aw's safe_outputs job writes too, and it runs first: it pushed a flattened single-parent
420
- # commit with GITHUB_TOKEN, which lost the agent's merge, left the pull request conflicting,
421
- # and started no CI, because GITHUB_TOKEN writes raise no events.
422
- if ! grep -qE '^ staged: true' "$MERGE_GATE_WORKER_MD"; then
423
- BRANCH_OK=0; echo "FAIL: merge-gate safe-outputs must be staged; conclude owns the write path" >&2
424
- fi
425
- if grep -q 'git rebase' "$MERGE_GATE_WORKER_MD"; then
426
- BRANCH_OK=0; echo "FAIL: merge-gate worker tells the agent to rebase; the push is fast-forward only" >&2
427
- fi
428
- if ! grep -q 'name: Check out the pull request branch' "$MERGE_GATE_WORKER_MD"; then
429
- BRANCH_OK=0; echo "FAIL: merge-gate worker must check out the pull request branch before the agent starts" >&2
430
- fi
431
- if ! grep -qF "conclusion == 'failure' && (inputs.attempts_so_far || '0') == '0'" "$MERGE_GATE_WORKER_MD"; then
432
- BRANCH_OK=0; echo "FAIL: the reserve job's progress comment must be posted on the first attempt only" >&2
433
- fi
434
- # A conflicting pull request has no CI run to read logs from, so the gate is handed empty
435
- # failure artifacts. Read on its own that looks like "no evidence", and the agent asked for a
436
- # human instead of resolving the conflict that caused it.
437
- if ! grep -qF 'Empty failure evidence is not a reason to ask for review' "$MERGE_GATE_WORKER_MD"; then
438
- BRANCH_OK=0
439
- echo "FAIL: the gate must treat empty failure evidence on a conflicting PR as the conflict to fix" >&2
672
+ if worker_installed merge-gate; then
673
+ # The agent's fix reaches the branch as a bundle applied fast-forward only (apply-agent-output).
674
+ # gh-aw's push tool description tells the model to rebase, and a rebased branch cannot
675
+ # fast-forward: the push is refused and the verdict is lost (Pliny-Bot run 33952565835). The
676
+ # worker must start on the pull request branch and must never say `git rebase`. Its progress
677
+ # comment is posted on the first attempt only; retries are recorded by the attempt comment.
678
+ BRANCH_OK=1
679
+ # Path B: staged safe outputs, applied by conclude with the App token. Without `staged: true`
680
+ # gh-aw's safe_outputs job writes too, and it runs first: it pushed a flattened single-parent
681
+ # commit with GITHUB_TOKEN, which lost the agent's merge, left the pull request conflicting,
682
+ # and started no CI, because GITHUB_TOKEN writes raise no events.
683
+ if ! grep -qE '^ staged: true' "$MERGE_GATE_WORKER_MD"; then
684
+ BRANCH_OK=0; echo "FAIL: merge-gate safe-outputs must be staged; conclude owns the write path" >&2
685
+ fi
686
+ if grep -q 'git rebase' "$MERGE_GATE_WORKER_MD"; then
687
+ BRANCH_OK=0; echo "FAIL: merge-gate worker tells the agent to rebase; the push is fast-forward only" >&2
688
+ fi
689
+ if ! grep -q 'name: Check out the pull request branch' "$MERGE_GATE_WORKER_MD"; then
690
+ BRANCH_OK=0; echo "FAIL: merge-gate worker must check out the pull request branch before the agent starts" >&2
691
+ fi
692
+ if ! grep -qF "conclusion == 'failure' && (inputs.attempts_so_far || '0') == '0'" "$MERGE_GATE_WORKER_MD"; then
693
+ BRANCH_OK=0; echo "FAIL: the reserve job's progress comment must be posted on the first attempt only" >&2
694
+ fi
695
+ # A conflicting pull request has no CI run to read logs from, so the gate is handed empty
696
+ # failure artifacts. Read on its own that looks like "no evidence", and the agent asked for a
697
+ # human instead of resolving the conflict that caused it.
698
+ if ! grep -qF 'Empty failure evidence is not a reason to ask for review' "$MERGE_GATE_WORKER_MD"; then
699
+ BRANCH_OK=0
700
+ echo "FAIL: the gate must treat empty failure evidence on a conflicting PR as the conflict to fix" >&2
701
+ fi
702
+ if [ "$BRANCH_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
703
+
704
+ # pr-pending means a pull request for this issue is open and waiting. Only merging retires it.
705
+ # Every other path (the protected-files hold, a review verdict, a failed attempt) leaves the
706
+ # pull request open, and stripping the label there produced a board where issues with open
707
+ # pull requests looked like they had none. It went unnoticed while the label actions silently
708
+ # removed nothing, so the two bugs hid each other.
709
+ PENDING_OK=1
710
+ grep -q '^ PR_PENDING_LABEL:' "$MERGE_GATE_WORKER_MD" ||
711
+ { PENDING_OK=0; echo "FAIL: merge gate lost its PR_PENDING_LABEL definition" >&2; }
712
+ if [ "$(count -c '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
713
+ PENDING_OK=0
714
+ echo "FAIL: pr-pending must be removed in exactly one place, the merge path" >&2
715
+ grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
716
+ fi
717
+ # And that one place has to be the merge outcome, not a hold or a failed attempt.
718
+ grep -B12 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'merge'" ||
719
+ { PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge outcome" >&2; }
720
+
721
+ # The invariant only ever looked at the merge gate, so apply-review quietly stripped the label
722
+ # on its already-satisfied and needs-human paths — both of which leave the pull request open.
723
+ # The one file the check ignored was the one breaking it. Look at every worker: implement adds
724
+ # the label, merge-gate removes it on merge, nobody else may touch it.
725
+ for worker in "${WORKFLOWS_DIR}"/agent-*.md; do
726
+ [ -f "$worker" ] || continue
727
+ case "$(basename "$worker")" in
728
+ agent-merge-gate.md | agent-implement.md) continue ;;
729
+ esac
730
+ if grep -q 'remove-issue-labels' "$worker" &&
731
+ grep -A8 'remove-issue-labels' "$worker" | grep -q 'env.PR_PENDING_LABEL'; then
732
+ PENDING_OK=0
733
+ echo "FAIL: $(basename "$worker") removes pr-pending; only the merge gate's merge path may" >&2
734
+ fi
735
+ done
736
+ if [ "$PENDING_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
440
737
  fi
441
- if [ "$BRANCH_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
442
-
443
- # pr-pending means a pull request for this issue is open and waiting. Only merging retires it.
444
- # Every other path (the protected-files hold, a review verdict, a failed attempt) leaves the
445
- # pull request open, and stripping the label there produced a board where issues with open
446
- # pull requests looked like they had none. It went unnoticed while the label actions silently
447
- # removed nothing, so the two bugs hid each other.
448
- PENDING_OK=1
449
- grep -q '^ PR_PENDING_LABEL:' "$MERGE_GATE_WORKER_MD" ||
450
- { PENDING_OK=0; echo "FAIL: merge gate lost its PR_PENDING_LABEL definition" >&2; }
451
- if [ "$(grep -c '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
452
- PENDING_OK=0
453
- echo "FAIL: pr-pending must be removed in exactly one place, the merge path" >&2
454
- grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
738
+
739
+ if worker_installed implement; then
740
+ # A provider outage kills a run in a couple of minutes with no answer, and the same issue used
741
+ # to be handed to a human for it. The implement worker retries those and only those: a run that
742
+ # worked for half an hour and then failed produced an answer that was wrong, and repeating it
743
+ # costs the fleet the same half hour to be wrong again.
744
+ IMPLEMENT_RETRY_OK=1
745
+ for needle in 'RETRY_UNDER_MINUTES' 'ATTEMPT_MARKER' 'attempts_so_far' 'operation=implement'; do
746
+ grep -qF "$needle" "$IMPLEMENT_WORKER_MD" || {
747
+ IMPLEMENT_RETRY_OK=0
748
+ echo "FAIL: implement worker lost its retry belt: no '$needle'" >&2
749
+ }
750
+ done
751
+ # Park and retry are mutually exclusive: the retry path must never add the review label, and
752
+ # the park path must never re-dispatch.
753
+ grep -A3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" ||
754
+ grep -B3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" || {
755
+ IMPLEMENT_RETRY_OK=0
756
+ echo "FAIL: the implement worker must not flag review on a run it is about to retry" >&2
757
+ }
758
+ if [ "$IMPLEMENT_RETRY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
455
759
  fi
456
- # And that one place has to be the merge outcome, not a hold or a failed attempt.
457
- grep -B12 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'merge'" ||
458
- { PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge outcome" >&2; }
459
- if [ "$PENDING_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
460
-
461
- # A provider outage kills a run in a couple of minutes with no answer, and the same issue used
462
- # to be handed to a human for it. The implement worker retries those and only those: a run that
463
- # worked for half an hour and then failed produced an answer that was wrong, and repeating it
464
- # costs the fleet the same half hour to be wrong again.
465
- IMPLEMENT_RETRY_OK=1
466
- for needle in 'RETRY_UNDER_MINUTES' 'ATTEMPT_MARKER' 'attempts_so_far' 'operation=implement'; do
467
- grep -qF "$needle" "$IMPLEMENT_WORKER_MD" || {
468
- IMPLEMENT_RETRY_OK=0
469
- echo "FAIL: implement worker lost its retry belt: no '$needle'" >&2
470
- }
471
- done
472
- # Park and retry are mutually exclusive: the retry path must never add the review label, and
473
- # the park path must never re-dispatch.
474
- grep -A3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" ||
475
- grep -B3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" || {
476
- IMPLEMENT_RETRY_OK=0
477
- echo "FAIL: the implement worker must not flag review on a run it is about to retry" >&2
478
- }
479
- if [ "$IMPLEMENT_RETRY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
480
760
 
481
- # A failed attempt must not strip `implement`: identify-gate-subject refuses an issue
482
- # without it, so the first crash would starve every retry at the subject check.
483
- if grep -A6 'Park the issue' "$MERGE_GATE_WORKER_MD" | grep -q 'REVIEW_LABEL' &&
484
- ! grep -qF 'labels: ${{ env.WORKING_LABEL }},${{ env.IMPLEMENT_LABEL }}' "$MERGE_GATE_WORKER_MD"; then
485
- PASS=$((PASS + 1))
486
- else
487
- FAIL=$((FAIL + 1))
488
- echo "FAIL: the incomplete job must keep implement and only park on an exhausted budget" >&2
761
+ if worker_installed merge-gate; then
762
+ # A failed attempt must not strip `implement`: identify-gate-subject refuses an issue
763
+ # without it, so the first crash would starve every retry at the subject check.
764
+ if grep -A9 'Park the issue' "$MERGE_GATE_WORKER_MD" | grep -q 'REVIEW_LABEL' &&
765
+ ! grep -qF 'labels: ${{ env.WORKING_LABEL }},${{ env.IMPLEMENT_LABEL }}' "$MERGE_GATE_WORKER_MD"; then
766
+ PASS=$((PASS + 1))
767
+ else
768
+ FAIL=$((FAIL + 1))
769
+ echo "FAIL: the incomplete job must keep implement and only park on an exhausted budget" >&2
770
+ fi
489
771
  fi
490
772
 
491
773
  # This repository is public. Every route a human can start from a comment, a review or a
@@ -493,6 +775,7 @@ fi
493
775
  # writes code. Asserted here because removing the gate would otherwise be a silent, one-line
494
776
  # change that nothing fails on.
495
777
  for route in refine implement apply-review; do
778
+ worker_installed "$route" || continue
496
779
  if grep -qE "route == '${route}'.*needs\.authorize\.outputs\.trusted == 'true'" "$ROUTER_YML"; then
497
780
  PASS=$((PASS + 1))
498
781
  else
@@ -503,17 +786,83 @@ done
503
786
 
504
787
  # Triage runs under a trusted App identity. Outside collaborators are admitted only to
505
788
  # the deterministic dispatcher; the worker call itself requires a trusted actor.
506
- if grep -qE "dispatch-triage:.*" "$ROUTER_YML" && \
507
- grep -qE "route == 'triage'.*is_outside_collaborator == 'true'" "$ROUTER_YML" && \
508
- grep -qE "route == 'triage'.*trusted == 'true'" "$ROUTER_YML"; then
509
- PASS=$((PASS + 1))
510
- else
511
- FAIL=$((FAIL + 1))
512
- echo "FAIL: route 'triage' does not dispatch outside collaborators and require a trusted worker actor" >&2
789
+ if worker_installed triage; then
790
+ if grep -qE "dispatch-triage:.*" "$ROUTER_YML" && \
791
+ grep -qE "route == 'triage'.*is_outside_collaborator == 'true'" "$ROUTER_YML" && \
792
+ grep -qE "route == 'triage'.*trusted == 'true'" "$ROUTER_YML"; then
793
+ PASS=$((PASS + 1))
794
+ else
795
+ FAIL=$((FAIL + 1))
796
+ echo "FAIL: route 'triage' does not dispatch outside collaborators and require a trusted worker actor" >&2
797
+ fi
513
798
  fi
514
799
 
515
- for route in refine implement triage apply-review merge-gate audit bot-approve \
516
- audit-close cleanup-artifacts reconcile-bot-pr-runs validate release; do
800
+ # Out of scope is the wrong door, not a rejection, and it must never close an issue. Numa#654
801
+ # was a reproducible authorization defect that passed nine of ten checks and was closed as
802
+ # not_planned with every label stripped, so nobody would ever have found it. The verdict for
803
+ # that case is needs-maintainer: open, review label, a maintainer adds refine to take it on.
804
+ # Only block closes, and only for work that cannot be done or is unsafe.
805
+ if worker_installed triage; then
806
+ TRIAGE_WORKER_MD="${WORKFLOWS_DIR}/agent-triage.md"
807
+ VALIDATE_TRIAGE_SH="${HERE}/../validate-triage-output/validate-triage-output.sh"
808
+ TRIAGE_OK=1
809
+
810
+ # The validator is what turns the agent's prose into the outcome the jobs branch on. A
811
+ # verdict it does not know becomes "invalid", which skips conclude entirely and reports the
812
+ # run incomplete, so the prompt and this script have to agree on all four names.
813
+ # Comment lines stripped first: the file explains the verdicts in prose above the program,
814
+ # and a plain search finds the name there even after it has been dropped from the jq
815
+ # alternation, which is exactly the regression this is meant to catch.
816
+ validate_program=$(grep -v '^[[:space:]]*#' "$VALIDATE_TRIAGE_SH")
817
+ for verdict in pass needs-info needs-maintainer block; do
818
+ if [ "$(count -c -- "$verdict" <<<"$validate_program")" -lt 3 ]; then
819
+ TRIAGE_OK=0
820
+ echo "FAIL: validate-triage-output.sh does not accept the '${verdict}' verdict in test(), capture() and the guard" >&2
821
+ fi
822
+ if ! grep -qF "\`**Verdict:** ${verdict}\`" "$TRIAGE_WORKER_MD"; then
823
+ TRIAGE_OK=0
824
+ echo "FAIL: the triage prompt does not offer '**Verdict:** ${verdict}'" >&2
825
+ fi
826
+ done
827
+
828
+ # The assertion this whole route turns on: exactly one step closes an issue, and it is
829
+ # reached only by a block verdict.
830
+ closes=$(count -c "state: 'closed'" "$TRIAGE_WORKER_MD")
831
+ if [ "$closes" -ne 1 ]; then
832
+ TRIAGE_OK=0
833
+ echo "FAIL: agent-triage.md closes an issue in ${closes} places; expected exactly one" >&2
834
+ elif ! grep -B12 "state: 'closed'" "$TRIAGE_WORKER_MD" | grep -q "outcome == 'block'"; then
835
+ TRIAGE_OK=0
836
+ echo "FAIL: the triage close step is not guarded on a block verdict alone" >&2
837
+ fi
838
+ if grep -q "outcome == 'needs-maintainer'" "$TRIAGE_WORKER_MD"; then
839
+ if grep -A6 "outcome == 'needs-maintainer'" "$TRIAGE_WORKER_MD" | grep -q "state: 'closed'"; then
840
+ TRIAGE_OK=0
841
+ echo "FAIL: a needs-maintainer verdict closes the issue; it must stay open" >&2
842
+ fi
843
+ else
844
+ TRIAGE_OK=0
845
+ echo "FAIL: agent-triage.md has no needs-maintainer branch in conclude" >&2
846
+ fi
847
+
848
+ # Parked, not looping: review goes on so a human sees it, triage comes off so a later
849
+ # comment does not re-enter triage and put it out of scope again for ever.
850
+ maintainer_block=$(sed -n "/outcome == 'needs-maintainer'/,/outcome == 'block'/p" "$TRIAGE_WORKER_MD")
851
+ grep -q 'env.REVIEW_LABEL' <<<"$maintainer_block" || {
852
+ TRIAGE_OK=0
853
+ echo "FAIL: the needs-maintainer branch does not add the review label" >&2
854
+ }
855
+ grep -q 'env.TRIAGE_LABEL' <<<"$maintainer_block" || {
856
+ TRIAGE_OK=0
857
+ echo "FAIL: the needs-maintainer branch does not remove the triage label, so comments would re-trigger triage" >&2
858
+ }
859
+
860
+ if [ "$TRIAGE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
861
+ fi
862
+
863
+ # Every installed worker and every plumbing route has a job; a worker that is not installed
864
+ # has none, or the router would call a lock file that does not exist.
865
+ for route in "${INSTALLED_ROUTES[@]}" "${PLUMBING_ROUTES[@]}"; do
517
866
  if grep -q "route == '${route}'" "$ROUTER_YML"; then
518
867
  PASS=$((PASS + 1))
519
868
  else
@@ -521,6 +870,14 @@ for route in refine implement triage apply-review merge-gate audit bot-approve \
521
870
  echo "FAIL: work-router.yml has no job for route '${route}'" >&2
522
871
  fi
523
872
  done
873
+ for route in "${EXCLUDED_ROUTES[@]}"; do
874
+ if grep -q "route == '${route}'" "$ROUTER_YML"; then
875
+ FAIL=$((FAIL + 1))
876
+ echo "FAIL: work-router.yml has a job for route '${route}' but agent-${route}.md is not installed" >&2
877
+ else
878
+ PASS=$((PASS + 1))
879
+ fi
880
+ done
524
881
 
525
882
  while read -r operation; do
526
883
  if grep -q "route == '${operation}'" "$ROUTER_YML"; then
@@ -536,7 +893,7 @@ done < <(sed -n '/^ operation:/,/^ issue-number:/p' "$ROUTER_YML" |
536
893
  # contents and issues and failed nightly on a 403 that named the endpoint and nothing else.
537
894
  # Paired job-to-scope rather than parsed out of each action: the jobs that touch pull
538
895
  # requests are few and known, and naming them here is what makes the omission visible.
539
- for pr_job in audit-close reconcile-bot-pr-runs detect-pr-conflicts; do
896
+ for pr_job in audit-close reconcile-bot-pr-runs detect-pr-conflicts housekeeping; do
540
897
  if ! grep -q "^ ${pr_job}:$" "$ROUTER_YML"; then
541
898
  continue
542
899
  fi
@@ -552,10 +909,330 @@ done
552
909
  # No written-down passwords in anything this repository ships. A throwaway credential for
553
910
  # a test container is still a policy finding, and one sat in every consumer's CI for weeks
554
911
  # until a scan found it rather than us. Two shapes: a password-ish name assigned a quoted
912
+ echo "── Housekeeping ──────────────────────────────────────────────────────────"
913
+
914
+ # The janitor is the only thing in the fleet that deletes a branch and closes an issue nobody
915
+ # asked it to close, and it runs unattended every six hours. Its guardrails are one-line
916
+ # conditions that would be easy to lose in an edit and impossible to notice afterwards, so
917
+ # they are asserted rather than trusted.
918
+ HOUSEKEEPING_YML="${HERE}/../housekeeping/action.yml"
919
+ if [ -f "$HOUSEKEEPING_YML" ]; then
920
+ HK_OK=1
921
+ hk() {
922
+ grep -qE "$1" "$HOUSEKEEPING_YML" || { HK_OK=0; echo "FAIL: housekeeping ${2}" >&2; }
923
+ }
924
+
925
+ # Every write goes through act(), which is the only place dry-run is honoured. A second
926
+ # write path would make --dry-run a lie exactly once, on the run that deletes something.
927
+ hk 'const act = async' 'has no act\(\) wrapper, so dry-run cannot be enforced in one place'
928
+ writes=$(count -cE 'github\.rest\.(issues\.(create|update|createComment|removeLabel|addLabels)|git\.deleteRef|actions\.createWorkflowDispatch)\(' "$HOUSEKEEPING_YML")
929
+ outside=$(awk '
930
+ /await act\(/ { inact = 1 }
931
+ inact && /github\.rest\.(issues\.(create|update|createComment|removeLabel|addLabels)|git\.deleteRef|actions\.createWorkflowDispatch)\(/ { seen++ }
932
+ inact && /^ \}\);$/ { inact = 0 }
933
+ END { print seen + 0 }
934
+ ' "$HOUSEKEEPING_YML")
935
+ if [ "$writes" -ne "$outside" ]; then
936
+ HK_OK=0
937
+ echo "FAIL: housekeeping performs ${writes} write(s) but only ${outside} are inside act(); dry-run would not cover the rest" >&2
938
+ fi
939
+
940
+ # A branch is someone's work until its pull request is finished. All three guards have to
941
+ # hold: never the default branch, never one with an open pull request, and never one whose
942
+ # pull requests were not all opened by a bot.
943
+ hk "branch\.name === defaultBranch. continue" 'can delete the default branch'
944
+ hk "p\.state === 'open'\)\) continue" 'can delete a branch whose pull request is still open'
945
+ hk 'isBot\(p\.user' 'can delete a branch from a human pull request'
946
+ hk 'forBranch\.length === 0. continue' 'can delete a branch that never had a pull request'
947
+
948
+ # Retrying a decision reproduces it. Only a park the machine caused carries `stalled`, and
949
+ # only those may be re-dispatched; everything else is reported.
950
+ hk "labels\.includes\('stalled'\)" 'retries parks that were decisions, not machine failures'
951
+ # Match the guard, not the phrase. `attempts >= maxRetries` also appears in the line that
952
+ # labels the digest entry, so grepping for the words alone still passed with the guard
953
+ # deleted from the `if` -- the same weak-assertion shape that let a deleted triage verdict
954
+ # through because the words survived in a comment.
955
+ hk 'if \(attempts >= maxRetries \|\| !work\) \{' 'has no retry budget guard on the retry path'
956
+
957
+ # The janitor closes issues, and the only issues it may close are a split parent whose
958
+ # children are all done and its own digest. Anything else is a person's to close.
959
+ closes=$(count -cE "state: 'closed'" "$HOUSEKEEPING_YML")
960
+ if [ "$closes" -eq 2 ]; then
961
+ PASS=$((PASS + 1))
962
+ else
963
+ HK_OK=0
964
+ echo "FAIL: housekeeping closes issues in ${closes} place(s); only the split parent and its own digest are allowed" >&2
965
+ fi
966
+
967
+ # A retry is a workflow_dispatch, and GitHub starts no workflow run from an event raised
968
+ # with GITHUB_TOKEN. Wiring the default token here would make every retry a silent no-op:
969
+ # green run, comment posted, labels removed, and nothing ever picks the issue up again.
970
+ hk_job=$(sed -n '/^ housekeeping:$/,/^ [a-z0-9_-]*:$/p' "$ROUTER_YML")
971
+ if printf '%s' "$hk_job" | grep -q 'app-token.outputs.token'; then
972
+ PASS=$((PASS + 1))
973
+ else
974
+ HK_OK=0
975
+ echo "FAIL: the housekeeping job passes a token that cannot start a workflow run; retries would silently do nothing" >&2
976
+ fi
977
+ # Deleting a ref needs contents: write. Without it every delete answers 403 and the sweep
978
+ # reports success having removed nothing.
979
+ if printf '%s' "$hk_job" | grep -qE '^ contents: write$'; then
980
+ PASS=$((PASS + 1))
981
+ else
982
+ HK_OK=0
983
+ echo "FAIL: the housekeeping job deletes branches but grants no contents: write scope" >&2
984
+ fi
985
+
986
+ # Every knob the action takes is a repository's to change, so each has to come from the
987
+ # router's env: block, which is the one part of the file `workflows update` preserves.
988
+ for knob in HOUSEKEEPING_RETRY_AFTER_HOURS HOUSEKEEPING_MAX_RETRIES HOUSEKEEPING_STALE_PR_DAYS HOUSEKEEPING_DIGEST_TITLE; do
989
+ if [ -n "$(router_env "$knob")" ] && printf '%s' "$hk_job" | grep -q "env.${knob}"; then
990
+ PASS=$((PASS + 1))
991
+ else
992
+ HK_OK=0
993
+ echo "FAIL: ${knob} is not both declared in the router env: block and read by the housekeeping job" >&2
994
+ fi
995
+ done
996
+
997
+ if [ "$HK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
998
+ fi
999
+
1000
+ # The audit chain closes its own reports, and every way it can be wrong is silent: a report
1001
+ # closed as completed with nothing implemented, or a report pinned open forever so the next
1002
+ # audit never runs. Both happened. Assert the three conditions that decide it.
1003
+ AUDIT_CLOSE_YML="${HERE}/../audit-close/action.yml"
1004
+ if [ -f "$AUDIT_CLOSE_YML" ] && worker_installed audit; then
1005
+ AC_OK=1
1006
+
1007
+ # A report referencing no issues had no work done on it. Closing that as `completed` is how
1008
+ # an audit used to end with every finding closed and nothing implemented.
1009
+ if grep -qE 'resolved === 0\) \{' "$AUDIT_CLOSE_YML" &&
1010
+ ! sed -n '/resolved === 0) {/,/^ }$/p' "$AUDIT_CLOSE_YML" | grep -q "state: 'closed'"; then
1011
+ PASS=$((PASS + 1))
1012
+ else
1013
+ AC_OK=0
1014
+ echo "FAIL: audit-close closes a report that references no issues; nothing was implemented from it" >&2
1015
+ fi
1016
+
1017
+ # Only a real closing keyword may pin a report open. One pattern here required the literal
1018
+ # `#closes #12` and matched nothing; the other matched a bare `#12` anywhere in any open
1019
+ # pull request and pinned the report open for as long as that pull request lived.
1020
+ if grep -q 'clos(?:e|es|ed)' "$AUDIT_CLOSE_YML" && ! grep -qF '#(?:closes?' "$AUDIT_CLOSE_YML"; then
1021
+ PASS=$((PASS + 1))
1022
+ else
1023
+ AC_OK=0
1024
+ echo "FAIL: audit-close still carries the dead '#closes #N' pattern or lost its closing-keyword match" >&2
1025
+ fi
1026
+
1027
+ # The backpressure query has to exclude what the chain marks stale, or three abandoned
1028
+ # reports disable the weekly audit permanently and the run skips green every week.
1029
+ # Read the query line itself, not the file. The prose above it explains what
1030
+ # `-label:stale-audit` is for, so a grep of the whole file passed with the exclusion deleted
1031
+ # from the query -- matching the comment that describes it.
1032
+ AUDIT_WORKER_MD="${WORKFLOWS_DIR}/agent-audit.md"
1033
+ audit_query="$(sed -n 's/^ *query: *"\(.*\)" *$/\1/p' "$AUDIT_WORKER_MD" | head -1)"
1034
+ if [[ "$audit_query" == *-label:stale-audit* ]] && grep -q "STALE_LABEL = 'stale-audit'" "$AUDIT_CLOSE_YML"; then
1035
+ PASS=$((PASS + 1))
1036
+ else
1037
+ AC_OK=0
1038
+ echo "FAIL: the audit backpressure query and audit-close disagree about stale-audit; abandoned reports would block every future audit" >&2
1039
+ fi
1040
+
1041
+ if [ "$AC_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1042
+ fi
1043
+
1044
+ echo "── Merge gate park ───────────────────────────────────────────────────────"
1045
+
1046
+ # A gate verdict parks the code it was given on. Both dispatch paths -- detect-pr-conflicts and
1047
+ # the reconcile belt -- used to compare the standing verdict against the CI finish time, which
1048
+ # made the park worthless: any later run on the same commits was newer than the verdict, so the
1049
+ # belt re-dispatched a pull request a human already owned and reset its attempt budget at the
1050
+ # same time. Lyceum PR #13 sat parked for six days while that happened. Asserted because both
1051
+ # comparisons are one line and neither failing produces a red run.
1052
+ if worker_installed merge-gate; then
1053
+ GATE_OK=1
1054
+
1055
+ # Neither path may key the park to CI timing again.
1056
+ stale_horizon=$(count -cE 'verdict" \\> "\$ci_finished"|latest_verdict" \\> "\$ci_finished"' "$ROUTER_YML")
1057
+ if [ "$stale_horizon" -eq 0 ]; then
1058
+ PASS=$((PASS + 1))
1059
+ else
1060
+ GATE_OK=0
1061
+ echo "FAIL: the merge-gate park is keyed to the CI finish time in ${stale_horizon} place(s); a CI re-run would reopen a park a person owns" >&2
1062
+ fi
1063
+
1064
+ # Both must fall back to the CI time only when the head commit cannot be read.
1065
+ horizons=$(count -cE '\$\{head_committed:-\$ci_finished\}' "$ROUTER_YML")
1066
+ if [ "$horizons" -eq 2 ]; then
1067
+ PASS=$((PASS + 1))
1068
+ else
1069
+ GATE_OK=0
1070
+ echo "FAIL: ${horizons} of the 2 gate dispatch paths key their park to the head commit" >&2
1071
+ fi
1072
+
1073
+ # One cap, not four literals, and it has to match what the worker tells the reader.
1074
+ router_cap="$(router_env MAX_GATE_ATTEMPTS)"
1075
+ worker_cap="$(sed -n 's/^ MAX_ATTEMPTS: "\([0-9]*\)"$/\1/p' "${WORKFLOWS_DIR}/agent-merge-gate.md" | head -1)"
1076
+ if [ -n "$router_cap" ] && [ "$router_cap" = "$worker_cap" ]; then
1077
+ PASS=$((PASS + 1))
1078
+ else
1079
+ GATE_OK=0
1080
+ echo "FAIL: the belt gives up after '${router_cap:-unset}' attempts but agent-merge-gate.md tells the reader '${worker_cap:-unset}'" >&2
1081
+ fi
1082
+ # And no path may go back to a literal. Counting the word `6` would match a hundred things,
1083
+ # so this looks only at the attempt comparison and the message beside it.
1084
+ if ! grep -qE '"\$attempts" -ge 6|attempts \+ 1\)\) of 6' "$ROUTER_YML"; then
1085
+ PASS=$((PASS + 1))
1086
+ else
1087
+ GATE_OK=0
1088
+ echo "FAIL: the gate attempt cap is hardcoded in work-router.yml instead of read from env.MAX_GATE_ATTEMPTS" >&2
1089
+ grep -nE '"\$attempts" -ge 6|attempts \+ 1\)\) of 6' "$ROUTER_YML" >&2
1090
+ fi
1091
+
1092
+ if [ "$GATE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1093
+ fi
1094
+
1095
+ echo "── Expression functions ──────────────────────────────────────────────────"
1096
+
1097
+ # GitHub's expression language has eleven functions and no more. There is no `split()`, no
1098
+ # `length()`, no `replace()`, and calling one is not a warning: the workflow fails to load with
1099
+ # "Unrecognized function", which shows up as a run that never starts. `split(env.REPO, '/')[0]`
1100
+ # was written into a template here and only caught by hand. actionlint would find it, but it
1101
+ # does not read composite manifests and is not installed in every consumer, so the same rule
1102
+ # lives here where the rest of the invariants are.
1103
+ readonly GH_EXPRESSION_FUNCTIONS='contains|startsWith|endsWith|format|join|toJSON|toJson|fromJSON|fromJson|hashFiles|success|always|cancelled|failure'
1104
+ EXPR_OK=1
1105
+ while IFS= read -r workflow; do
1106
+ # Only inside an expression. The same word in a `run:` block is shell or JavaScript.
1107
+ offenders="$(grep -oE '\$\{\{[^}]*\}\}' "$workflow" |
1108
+ grep -oE '[a-zA-Z_][a-zA-Z0-9_]*\(' |
1109
+ tr -d '(' |
1110
+ grep -vE "^(${GH_EXPRESSION_FUNCTIONS})$" |
1111
+ sort -u || true)"
1112
+ if [ -n "$offenders" ]; then
1113
+ EXPR_OK=0
1114
+ echo "FAIL: $(basename "$workflow") calls $(echo "$offenders" | tr '\n' ' ')which GitHub expressions do not have; the workflow will not load" >&2
1115
+ fi
1116
+ done < <(find "$WORKFLOWS_DIR" "${HERE}/../.." -maxdepth 3 -name '*.yml' -not -name '*.lock.yml' 2>/dev/null | sort -u)
1117
+ if [ "$EXPR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1118
+
1119
+ echo "── Error report privacy ──────────────────────────────────────────────────"
1120
+
1121
+ # Every repository that installs this package is private, and the error report is the only job
1122
+ # that sends anything out of one. Its whole safety argument is four properties, each of which
1123
+ # is one line that an edit could remove without any run going red, so all four are asserted.
1124
+ ERROR_REPORT_YML="${HERE}/../report-workflow-errors/action.yml"
1125
+ if [ -f "$ERROR_REPORT_YML" ]; then
1126
+ ER_OK=1
1127
+ er() {
1128
+ grep -qE "$1" "$ERROR_REPORT_YML" || { ER_OK=0; echo "FAIL: report-workflow-errors ${2}" >&2; }
1129
+ }
1130
+
1131
+ # 1. The scanner, and its teeth. A body that trips it must not be filed, and the run must go
1132
+ # red so the field that carried private text gets fixed instead of leaking again tomorrow.
1133
+ #
1134
+ # Match the declaration and count the sites, never the bare symbol. A grep for `leakChecks`
1135
+ # passed with the declaration renamed, because the name survives where scan() uses it, and a
1136
+ # grep for `core.setFailed` passed with one of the two calls turned into core.info. Both of
1137
+ # those were mutation-tested and both let a broken privacy guard through.
1138
+ er 'const leakChecks = \[' 'declares no leak scanner'
1139
+ teeth=$(count -cE 'core\.setFailed.*withheld by the leak scanner' "$ERROR_REPORT_YML")
1140
+ if [ "$teeth" -ge 2 ]; then
1141
+ PASS=$((PASS + 1))
1142
+ else
1143
+ ER_OK=0
1144
+ echo "FAIL: report-workflow-errors fails the run on a leak in only ${teeth} of its 2 exit paths" >&2
1145
+ fi
1146
+ # Every write upstream has to be behind a scan. Counting is enough here because both are few
1147
+ # and named, and a new write added without a guard moves the counts apart.
1148
+ scans=$(count -cE 'if \(!scan\(' "$ERROR_REPORT_YML")
1149
+ upstream_writes=$(count -cE 'upstream\.rest\.issues\.(create|update)\(' "$ERROR_REPORT_YML")
1150
+ if [ "$scans" -ge "$upstream_writes" ] && [ "$upstream_writes" -gt 0 ]; then
1151
+ PASS=$((PASS + 1))
1152
+ else
1153
+ ER_OK=0
1154
+ echo "FAIL: report-workflow-errors makes ${upstream_writes} upstream write(s) behind only ${scans} leak scan(s)" >&2
1155
+ fi
1156
+ for guard in 'the repository name' 'the owner name' 'a github.com URL' 'an email address' 'an absolute path'; do
1157
+ er "\{ what: '${guard}', test:" "no longer scans for ${guard}"
1158
+ done
1159
+
1160
+ # 2. The allowlist. A consumer's own workflow name can describe a product, a customer or an
1161
+ # environment; only the names this package gives its own files may be reported.
1162
+ er 'const OWNED = /\^\(work-router' 'has no workflow allowlist, so a repository-specific workflow name could be reported'
1163
+ er 'skippedForeign' 'does not account for the workflows it declined to inspect'
1164
+
1165
+ # 3. No raw log text. The catalogue matches the log tail and only the matched entry's id is
1166
+ # kept; a change that put the matched text in the report would be the leak.
1167
+ #
1168
+ # The finding object is that boundary, because bodyFor() renders a finding, so the check is
1169
+ # on its shape: the fields the object literals actually set must be exactly the declared
1170
+ # reportable list. An earlier version of this tried to spot log text in the body with a
1171
+ # regex over the whole file, and a mutation that added `${finding.logText.slice(0, 400)}`
1172
+ # walked straight past it -- the pattern was case-sensitive and the inserted label said
1173
+ # "Summary". Comparing two sets has no such gap.
1174
+ er 'patternId = CATALOGUE\.find' 'no longer classifies the log through the catalogue'
1175
+ declared=$(sed -n "s/^ *const FINDING_FIELDS = \[\(.*\)\];$/\1/p" "$ERROR_REPORT_YML" |
1176
+ tr -d " '" | tr ',' '\n' | sort -u | tr '\n' ' ')
1177
+ # Every key set in a finding object literal, plus every key assigned onto one afterwards.
1178
+ assigned=$( { sed -n '/const seen = findings\.get/,/^ };$/p' "$ERROR_REPORT_YML" |
1179
+ grep -oE '[a-zA-Z_][a-zA-Z0-9_]*:' | tr -d ':'
1180
+ grep -oE 'seen\.[a-zA-Z_][a-zA-Z0-9_]*' "$ERROR_REPORT_YML" | cut -d. -f2
1181
+ } | sort -u | tr '\n' ' ')
1182
+ if [ -n "${declared// /}" ] && [ "$declared" = "$assigned" ]; then
1183
+ PASS=$((PASS + 1))
1184
+ else
1185
+ ER_OK=0
1186
+ echo "FAIL: report-workflow-errors builds findings with fields that are not the declared reportable set" >&2
1187
+ echo " declared: ${declared:-(none)}" >&2
1188
+ echo " assigned: ${assigned:-(none)}" >&2
1189
+ fi
1190
+ # And the run-time half of the same boundary, so a field that arrives by a path the check
1191
+ # above cannot see stops the job instead of being rendered upstream.
1192
+ er 'not in the reportable field list' 'does not check the finding shape at run time'
1193
+
1194
+ # 4. No model. A model asked to summarise a failure paraphrases whatever the log held, which
1195
+ # is the one thing that must not cross the boundary. This job stays deterministic.
1196
+ if grep -qiE '(engine:|opencode|safe-outputs|OPENAI_API_KEY)' "$ERROR_REPORT_YML"; then
1197
+ ER_OK=0
1198
+ echo "FAIL: report-workflow-errors reaches for a model; the report must stay deterministic" >&2
1199
+ else
1200
+ PASS=$((PASS + 1))
1201
+ fi
1202
+
1203
+ # The workflow that drives it is an optional template: installed under .github/workflows in a
1204
+ # consumer, and still in templates/ upstream. Check whichever is present, so the assertions
1205
+ # run in the package's own CI rather than only after somebody installs it.
1206
+ ERROR_REPORT_WORKFLOW="${WORKFLOWS_DIR}/agentics-error-report.yml"
1207
+ [ -f "$ERROR_REPORT_WORKFLOW" ] ||
1208
+ ERROR_REPORT_WORKFLOW="${HERE}/../../templates/agentics/agentics-error-report.yml"
1209
+ if [ -f "$ERROR_REPORT_WORKFLOW" ]; then
1210
+ # It reads this repository and writes nothing to it. A write scope here would mean the job
1211
+ # that talks to another repository can also change this one.
1212
+ if grep -qE '^ (contents|actions): read$' "$ERROR_REPORT_WORKFLOW" &&
1213
+ ! grep -qE '^ [a-z-]+: write$' "$ERROR_REPORT_WORKFLOW"; then
1214
+ PASS=$((PASS + 1))
1215
+ else
1216
+ ER_OK=0
1217
+ echo "FAIL: agentics-error-report.yml grants a write scope; it must be read-only in the repository it reports on" >&2
1218
+ fi
1219
+ # The upstream token is scoped to the upstream repository alone, never the default token.
1220
+ if grep -q 'upstream-token: ${{ steps.upstream-token.outputs.token }}' "$ERROR_REPORT_WORKFLOW" &&
1221
+ grep -qE '^ repositories: \$\{\{ env\.UPSTREAM_NAME \}\}$' "$ERROR_REPORT_WORKFLOW"; then
1222
+ PASS=$((PASS + 1))
1223
+ else
1224
+ ER_OK=0
1225
+ echo "FAIL: agentics-error-report.yml does not scope its upstream token to the upstream repository" >&2
1226
+ fi
1227
+ fi
1228
+
1229
+ if [ "$ER_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1230
+ fi
1231
+
555
1232
  # value, and a command-line flag given one; a line containing a dollar sign is taken to be
556
1233
  # an expression or a shell variable and allowed. Paths resolve relative to this script, so
557
1234
  # upstream this reads the templates and in a consumer it reads the real workflows.
558
- PASSWORD_SCAN_DIRS=("${HERE}/../../workflows")
1235
+ PASSWORD_SCAN_DIRS=("${WORKFLOWS_DIR}")
559
1236
  [ -d "${HERE}/../../templates/ci" ] && PASSWORD_SCAN_DIRS+=("${HERE}/../../templates/ci")
560
1237
  [ -d "${HERE}/../../templates/agentics" ] && PASSWORD_SCAN_DIRS+=("${HERE}/../../templates/agentics")
561
1238
  password_hits=$(