@plainconceptsplatform/workflows 0.5.1 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/README.md +105 -88
  2. package/dist/catalog-installation.d.ts +67 -0
  3. package/dist/catalog-installation.js +485 -0
  4. package/dist/catalog-listing.d.ts +13 -0
  5. package/dist/catalog-listing.js +70 -0
  6. package/dist/index.d.ts +2 -0
  7. package/dist/index.js +273 -0
  8. package/dist/package-baseline.d.ts +25 -0
  9. package/dist/package-baseline.js +138 -0
  10. package/dist/repository-inspection.d.ts +23 -0
  11. package/dist/repository-inspection.js +113 -0
  12. package/dist/route-processing.d.ts +4 -0
  13. package/dist/route-processing.js +80 -0
  14. package/dist/stack-defaults.d.ts +11 -0
  15. package/dist/stack-defaults.js +110 -0
  16. package/dist/tui.d.ts +25 -0
  17. package/dist/tui.js +271 -0
  18. package/dist/worker-env.d.ts +46 -0
  19. package/dist/worker-env.js +179 -0
  20. package/dist/workflow-catalog.d.ts +27 -0
  21. package/dist/workflow-catalog.js +57 -0
  22. package/loops/actions/add-issue-labels/action.yml +22 -2
  23. package/loops/actions/apply-agent-bundle/apply-bundle.sh +14 -1
  24. package/loops/actions/audit-close/action.yml +180 -128
  25. package/loops/actions/classify-route/action.yml +7 -0
  26. package/loops/actions/classify-route/classify-route.sh +28 -7
  27. package/loops/actions/cleanup-artifacts/action.yml +38 -12
  28. package/loops/actions/housekeeping/action.yml +251 -0
  29. package/loops/actions/identify-gate-subject/action.yml +3 -1
  30. package/loops/actions/merge-agent-pr/action.yml +13 -0
  31. package/loops/actions/remove-issue-labels/action.yml +4 -2
  32. package/loops/actions/report-workflow-errors/action.yml +385 -0
  33. package/loops/actions/validate-merge-gate-output/validate-merge-gate-output.sh +13 -1
  34. package/loops/actions/validate-refine-output/validate-refine-output.sh +18 -2
  35. package/loops/actions/validate-triage-output/action.yml +1 -1
  36. package/loops/actions/validate-triage-output/validate-triage-output.sh +9 -5
  37. package/loops/actions/verify-composite-actions/verify-composite-actions.sh +53 -0
  38. package/loops/actions/verify-refine-output/verify-refine-output.sh +20 -0
  39. package/loops/actions/verify-route-matrix/verify-route-matrix.sh +1017 -56
  40. package/loops/scripts/compile-agent-workflows.mjs +96 -1
  41. package/loops/scripts/merge-changelog.mjs +76 -0
  42. package/loops/templates/agentics/actionlint.yaml +13 -0
  43. package/loops/templates/agentics/agentics-checks.yml +203 -86
  44. package/loops/templates/agentics/agentics-error-report.yml +97 -0
  45. package/loops/templates/ci/app-ci-dotnet-next.yml +64 -2
  46. package/loops/templates/ci/app-ci-node-monorepo.yml +52 -0
  47. package/loops/templates/opencode/opencode.ci.json +3 -3
  48. package/loops/templates/opencode/opencode.ci.json.md +4 -1
  49. package/loops/workflows/agent-apply-review.md +452 -465
  50. package/loops/workflows/agent-audit.md +201 -213
  51. package/loops/workflows/agent-implement.md +616 -538
  52. package/loops/workflows/agent-merge-gate.md +830 -730
  53. package/loops/workflows/agent-refine.md +599 -609
  54. package/loops/workflows/agent-release.md +244 -258
  55. package/loops/workflows/agent-triage.md +476 -447
  56. package/loops/workflows/authorize-bot-work.yml +26 -6
  57. package/loops/workflows/work-router.yml +1185 -862
  58. package/package.json +9 -8
  59. package/loops/actions/stale-recovery/action.yml +0 -288
  60. package/loops/actions/update-changelog/action.yml +0 -113
@@ -2,22 +2,70 @@
2
2
  # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-route-matrix/verify-route-matrix.sh. Update with `workflows update --force`; consumer edits may be overwritten.
3
3
  # Exercise the router's real classifier. This sources classify-route.sh rather than
4
4
  # restating it, so a change to the route table cannot pass here by being copied twice.
5
+ #
6
+ # The same file runs in every consumer, whatever subset of workers it installed: the router is
7
+ # regenerated for that subset, so every assertion about a worker or about its router job is
8
+ # conditional on the worker file being present. The classifier is the complete route table in
9
+ # every repository (a route with no job is a no-op run), so its assertions are unconditional.
10
+ #
11
+ # This file greps workflow sources for literal `${{ ... }}` expressions on purpose.
12
+ # shellcheck disable=SC2016
5
13
 
6
14
  set -euo pipefail
7
15
 
8
16
  HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
9
- ROUTER_YML="${HERE}/../../workflows/work-router.yml"
10
- AUTHORIZE_YML="${HERE}/../../workflows/authorize-bot-work.yml"
11
- IMPLEMENT_WORKER_MD="${HERE}/../../workflows/agent-implement.md"
12
- MERGE_GATE_WORKER_MD="${HERE}/../../workflows/agent-merge-gate.md"
17
+ WORKFLOWS_DIR="${HERE}/../../workflows"
18
+ ROUTER_YML="${WORKFLOWS_DIR}/work-router.yml"
19
+ IMPLEMENT_WORKER_MD="${WORKFLOWS_DIR}/agent-implement.md"
20
+ MERGE_GATE_WORKER_MD="${WORKFLOWS_DIR}/agent-merge-gate.md"
21
+
22
+ # The audit slot is per repository and lives in the router's own env: block, which a real run
23
+ # exports into the classify step. Export it here too, or this file would test the classifier's
24
+ # fallback rather than the cron the router actually fires on.
25
+ router_env() {
26
+ sed -n "s/^ $1: *//p" "$ROUTER_YML" | head -1 | sed -e 's/^"\(.*\)"$/\1/' -e "s/^'\(.*\)'$/\1/"
27
+ }
28
+ AUDIT_CRON="$(router_env AUDIT_CRON)"
29
+ export AUDIT_CRON
13
30
 
14
31
  # shellcheck source-path=SCRIPTDIR
15
32
  # shellcheck source=../classify-route/classify-route.sh
16
33
  source "${HERE}/../classify-route/classify-route.sh"
17
34
 
35
+ # Every worker route the package knows. The ones with a worker file here are installed; the
36
+ # others must have no job in this router. Plumbing routes are in every router.
37
+ ALL_WORKER_ROUTES=(refine implement triage apply-review merge-gate audit release)
38
+ PLUMBING_ROUTES=(bot-approve audit-close cleanup-artifacts reconcile-bot-pr-runs housekeeping validate)
39
+
40
+ worker_installed() {
41
+ [ -f "${WORKFLOWS_DIR}/agent-$1.md" ]
42
+ }
43
+
44
+ INSTALLED_ROUTES=()
45
+ EXCLUDED_ROUTES=()
46
+ for route in "${ALL_WORKER_ROUTES[@]}"; do
47
+ if worker_installed "$route"; then
48
+ INSTALLED_ROUTES+=("$route")
49
+ else
50
+ EXCLUDED_ROUTES+=("$route")
51
+ fi
52
+ done
53
+ echo "Installed workers: ${INSTALLED_ROUTES[*]:-(none)}"
54
+ [ "${#EXCLUDED_ROUTES[@]}" -eq 0 ] || echo "Not installed: ${EXCLUDED_ROUTES[*]}"
55
+
18
56
  PASS=0
19
57
  FAIL=0
20
58
 
59
+ # `grep -c` exits 1 when it counts zero, and this file runs under `set -e`, so writing
60
+ # `n=$(grep -c ...)` against a pattern that is absent ended the whole suite at whatever section
61
+ # it had reached, with no error printed and no FAIL counted. That made an assertion of the form
62
+ # "this pattern must be GONE" impossible to write here: the moment it held, the suite died. Every
63
+ # count goes through this instead, where zero is an answer rather than a failure. Callers pass
64
+ # their own grep flags.
65
+ count() {
66
+ grep "$@" 2>/dev/null || true
67
+ }
68
+
21
69
  # Classify one event and read a single field out of the result.
22
70
  route_field() {
23
71
  local field="$1"
@@ -90,6 +138,8 @@ assert "refine label starts a first pass" first \
90
138
  echo "── Comment events ────────────────────────────────────────────────────────"
91
139
  assert_route "a comment on a pull request routes to apply-review" apply-review \
92
140
  EVENT=issue_comment COMMENT_ON_PR=true EVENT_ISSUE_NUMBER=7
141
+ assert_route "the bot's own comment on a pull request never re-enters apply-review" none \
142
+ EVENT=issue_comment COMMENT_ON_PR=true COMMENT_SENDER_TYPE=Bot EVENT_ISSUE_NUMBER=7
93
143
  assert_route "an author reply on a refine issue re-refines" refine \
94
144
  EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
95
145
  'ISSUE_LABELS=["refine","review"]' EVENT_ISSUE_NUMBER=42
@@ -102,9 +152,6 @@ assert_route "the bot's own comment never re-enters refine" none \
102
152
  assert_route "a comment on an issue without refine routes nowhere" none \
103
153
  EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
104
154
  'ISSUE_LABELS=["bug"]' EVENT_ISSUE_NUMBER=42
105
- assert_route "the bot's own comment never re-enters direct" none \
106
- EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=Bot \
107
- 'ISSUE_LABELS=["direct"]' EVENT_ISSUE_NUMBER=42
108
155
  assert_route "a comment on a triage issue re-triages" triage \
109
156
  EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
110
157
  'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
@@ -116,12 +163,22 @@ assert_route "the bot's own comment never re-enters triage" none \
116
163
  'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
117
164
 
118
165
  echo "── Closed issues ─────────────────────────────────────────────────────────"
119
- assert_route "a closing comment on a refine issue does not re-refine" none EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User ISSUE_STATE=closed 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
120
- assert_route "a comment on a closed issue never re-triages" none EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User ISSUE_STATE=closed 'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
121
- assert_route "a work label added to a closed issue routes nowhere" none EVENT=issues ACTION=labeled LABEL=bot-working ISSUE_STATE=closed 'ISSUE_LABELS=["implement"]' EVENT_ISSUE_NUMBER=42
122
- assert_route "a closed issue reopened as opened still routes nowhere while closed" none EVENT=issues ACTION=opened ISSUE_STATE=closed 'ISSUE_LABELS=[]' EVENT_ISSUE_NUMBER=42
123
- assert_route "a comment on a closed pull request still routes to apply-review" apply-review EVENT=issue_comment COMMENT_ON_PR=true ISSUE_STATE=closed EVENT_ISSUE_NUMBER=7
124
- assert_route "an open refine issue is unaffected by the closed guard" refine EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User ISSUE_STATE=open 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
166
+ assert_route "a closing comment on a refine issue does not re-refine" none \
167
+ EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
168
+ ISSUE_STATE=closed 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
169
+ assert_route "a comment on a closed issue never re-triages" none \
170
+ EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
171
+ ISSUE_STATE=closed 'ISSUE_LABELS=["triage"]' EVENT_ISSUE_NUMBER=42
172
+ assert_route "a work label added to a closed issue routes nowhere" none \
173
+ EVENT=issues ACTION=labeled LABEL=bot-working ISSUE_STATE=closed \
174
+ 'ISSUE_LABELS=["implement"]' EVENT_ISSUE_NUMBER=42
175
+ assert_route "a closed issue reopened as opened still routes nowhere while closed" none \
176
+ EVENT=issues ACTION=opened ISSUE_STATE=closed 'ISSUE_LABELS=[]' EVENT_ISSUE_NUMBER=42
177
+ assert_route "a comment on a closed pull request still routes to apply-review" apply-review \
178
+ EVENT=issue_comment COMMENT_ON_PR=true ISSUE_STATE=closed EVENT_ISSUE_NUMBER=7
179
+ assert_route "an open refine issue is unaffected by the closed guard" refine \
180
+ EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
181
+ ISSUE_STATE=open 'ISSUE_LABELS=["refine"]' EVENT_ISSUE_NUMBER=42
125
182
 
126
183
  echo "── Review events ─────────────────────────────────────────────────────────"
127
184
  assert_route "a review comment routes to apply-review" apply-review \
@@ -158,7 +215,7 @@ while read -r cron; do
158
215
  PASS=$((PASS + 1))
159
216
  echo " ${cron} -> ${selected}"
160
217
  fi
161
- done < <(sed -n 's/^ *- cron: "\(.*\)"$/\1/p' "$ROUTER_YML")
218
+ done < <(sed -n 's/^ *- cron: "\([^"]*\)".*/\1/p' "$ROUTER_YML")
162
219
 
163
220
  assert_route "an unknown cron routes nowhere" none EVENT=schedule "SCHEDULE=0 0 30 2 *"
164
221
 
@@ -169,20 +226,30 @@ assert_route "refine dispatch rejects a non-numeric issue" none \
169
226
  EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=abc
170
227
  assert_route "refine dispatch accepts a positive issue" refine \
171
228
  EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=42
172
- assert_route "direct dispatch needs an issue number" none \
173
- EVENT=workflow_dispatch OPERATION=direct INPUT_ISSUE_NUMBER=
174
229
  assert_route "triage dispatch accepts a positive issue" triage \
175
230
  EVENT=workflow_dispatch OPERATION=triage INPUT_ISSUE_NUMBER=42
176
231
  assert_route "triage dispatch needs an issue number" none \
177
232
  EVENT=workflow_dispatch OPERATION=triage INPUT_ISSUE_NUMBER=
178
233
  assert "triage dispatch defaults to first pass" first \
179
234
  "$(route_field triage-mode EVENT=workflow_dispatch OPERATION=triage INPUT_ISSUE_NUMBER=42)"
180
- assert_route "batch dispatch needs an issue number" none \
181
- EVENT=workflow_dispatch OPERATION=batch INPUT_ISSUE_NUMBER=
182
235
  assert_route "merge-gate dispatch needs a pull request number" none \
183
236
  EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=0
184
237
  assert_route "merge-gate dispatch accepts a positive pull request" merge-gate \
185
238
  EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=7
239
+ assert "merge-gate dispatch defaults its attempt count to zero" 0 \
240
+ "$(route_field merge-gate-attempts EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=7)"
241
+ assert "merge-gate dispatch forwards the attempt count" 3 \
242
+ "$(route_field merge-gate-attempts EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=7 INPUT_ATTEMPTS_SO_FAR=3)"
243
+ # The implement worker re-dispatches itself when a run dies before producing an answer, so the
244
+ # count has to survive the round trip or the budget never advances and the retry never stops.
245
+ assert "implement dispatch defaults its attempt count to zero" 0 \
246
+ "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=implement INPUT_ISSUE_NUMBER=42)"
247
+ assert "implement dispatch forwards the attempt count" 2 \
248
+ "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=implement INPUT_ISSUE_NUMBER=42 INPUT_ATTEMPTS_SO_FAR=2)"
249
+ assert "a refine dispatch carries no implement attempts" 0 \
250
+ "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=42 INPUT_ATTEMPTS_SO_FAR=2)"
251
+ assert_route "release dispatch needs no numbers" release \
252
+ EVENT=workflow_dispatch OPERATION=release
186
253
  assert_route "reconcile-bot-pr-runs dispatch needs no numbers" reconcile-bot-pr-runs \
187
254
  EVENT=workflow_dispatch OPERATION=reconcile-bot-pr-runs
188
255
  assert_route "an unknown operation routes nowhere" none \
@@ -193,57 +260,514 @@ assert "a scheduled audit reports its trigger kind" scheduled \
193
260
  assert "a dispatched audit reports its trigger kind" manual \
194
261
  "$(route_field trigger-kind EVENT=workflow_dispatch OPERATION=audit INPUT_TRIGGER_KIND=manual)"
195
262
 
263
+ echo "── Prompt hygiene ────────────────────────────────────────────────────────"
264
+
265
+ # Everything below a worker's frontmatter is the prompt. Three things must not be in one.
266
+ #
267
+ # A `gh` call, because the shared CI agent config says the GitHub CLI is intentionally
268
+ # unauthenticated and the agent must never use it for GitHub reads or writes. A prompt that
269
+ # orders one burns turns and fails; implement carried a `gh pr list` for weeks, asking the agent
270
+ # to redo a check the router had already done before dispatching it.
271
+ #
272
+ # A duplicate step number, because these are ordered instruction lists and a step that says
273
+ # "go to step 6" cannot resolve when there are two. implement had two 6s with contradictory
274
+ # rules ("exactly one" and "at least one" safe output), refine had two 5s, apply-review two 9s.
275
+ #
276
+ # A Mermaid diagram, because it is documentation that the model is charged for on every run and
277
+ # then told to ignore. They live in docs/diagrams.md.
278
+ PROMPT_OK=1
279
+ for worker in "${WORKFLOWS_DIR}"/agent-*.md; do
280
+ [ -f "$worker" ] || continue
281
+ name=$(basename "$worker")
282
+ # The prompt starts after the closing --- of the frontmatter.
283
+ fm_end=$(awk 'NR>1 && /^---[[:space:]]*$/{print NR; exit}' "$worker")
284
+ [ -n "$fm_end" ] || { PROMPT_OK=0; echo "FAIL: ${name} has no frontmatter terminator" >&2; continue; }
285
+ prompt=$(tail -n "+$((fm_end + 1))" "$worker")
286
+
287
+ if grep -qE '(^|[^[:alnum:]_-])gh (pr|issue|api|run|release|workflow|auth) ' <<<"$prompt"; then
288
+ PROMPT_OK=0
289
+ echo "FAIL: ${name}'s prompt tells the agent to run gh; the CLI is unauthenticated in CI" >&2
290
+ grep -nE '(^|[^[:alnum:]_-])gh (pr|issue|api|run|release|workflow|auth) ' <<<"$prompt" >&2
291
+ fi
292
+
293
+ if grep -q '```mermaid' <<<"$prompt"; then
294
+ PROMPT_OK=0
295
+ echo "FAIL: ${name}'s prompt contains a Mermaid diagram; diagrams belong in docs/diagrams.md" >&2
296
+ fi
297
+
298
+ # Top-level steps only: an indented "1." is a sub-list and numbers restart legitimately.
299
+ dupes=$(grep -oE '^[0-9]+\. ' <<<"$prompt" | tr -d '. ' | sort -n | uniq -d | tr '\n' ' ')
300
+ if [ -n "${dupes// /}" ]; then
301
+ PROMPT_OK=0
302
+ echo "FAIL: ${name}'s prompt repeats step number(s): ${dupes}" >&2
303
+ fi
304
+
305
+ # A multi-line env value does not survive compilation: gh-aw joins it onto one line in the
306
+ # lock, so a Markdown table written across five lines in the source reaches the agent as a
307
+ # single unreadable row. Verified against a compiled lock. Block the YAML block scalars that
308
+ # produce one, so the flattening is a failed check rather than a silently useless value.
309
+ block_scalars=$(awk '
310
+ /^env:$/ { inenv = 1; next }
311
+ inenv && /^[^ ]/ { inenv = 0 }
312
+ inenv && /^ [A-Za-z_][A-Za-z0-9_]*: *[|>]-?[0-9]* *$/ { print $1 }
313
+ ' "$worker" | tr -d ':' | tr '\n' ' ')
314
+ if [ -n "${block_scalars// /}" ]; then
315
+ PROMPT_OK=0
316
+ echo "FAIL: ${name} declares env value(s) as a multi-line block, which the compiler flattens: ${block_scalars}" >&2
317
+ fi
318
+ done
319
+ if [ "$PROMPT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
320
+
196
321
  echo "── Router wiring ─────────────────────────────────────────────────────────"
197
322
 
323
+ # Two router values are needed where GitHub evaluates no expression, so the installer mirrors
324
+ # them out of env: into a literal. A copy that drifts fails in the direction that hurts: the
325
+ # gate silently reads a CI workflow nobody runs, or the audit fires on a cron the classifier
326
+ # maps to no route. Neither produces a red run, so assert the copies here.
327
+ MIRROR_OK=1
328
+ ci_name="$(router_env CI_WORKFLOW_NAME)"
329
+ trigger_name="$(sed -n 's/^ *workflows: \["\(.*\)"\] *$/\1/p' "$ROUTER_YML" | head -1)"
330
+ if [ -z "$ci_name" ]; then
331
+ MIRROR_OK=0; echo "FAIL: work-router.yml defines no CI_WORKFLOW_NAME in its env: block" >&2
332
+ elif [ "$ci_name" != "$trigger_name" ]; then
333
+ MIRROR_OK=0
334
+ echo "FAIL: the workflow_run trigger names '${trigger_name}' but env.CI_WORKFLOW_NAME is '${ci_name}'" >&2
335
+ fi
336
+ # The audit cron only exists in a router that installed the audit worker.
337
+ if worker_installed audit; then
338
+ cron_line="$(sed -n 's/^ *- cron: "\([^"]*\)" # audit slot.*/\1/p' "$ROUTER_YML" | head -1)"
339
+ if [ -z "$AUDIT_CRON" ]; then
340
+ MIRROR_OK=0; echo "FAIL: work-router.yml defines no AUDIT_CRON in its env: block" >&2
341
+ elif [ "$AUDIT_CRON" != "$cron_line" ]; then
342
+ MIRROR_OK=0
343
+ echo "FAIL: the audit slot cron is '${cron_line}' but env.AUDIT_CRON is '${AUDIT_CRON}'" >&2
344
+ fi
345
+ fi
346
+ # And nothing may go back to naming the CI workflow directly: a second literal is a second
347
+ # thing to keep in step, and the one that gets forgotten is the one inside a jq filter.
348
+ if [ "$(count -c '"App: CI"' "$ROUTER_YML")" -gt 2 ]; then
349
+ MIRROR_OK=0
350
+ echo "FAIL: work-router.yml hardcodes the CI workflow name outside env: and the mirrored trigger" >&2
351
+ grep -n '"App: CI"' "$ROUTER_YML" >&2
352
+ fi
353
+
354
+ # Bot logins are the same shape of problem. Most sites read env.TRUSTED_BOTS, but a job-level
355
+ # `if:` cannot: GitHub does not expose the env context there, so bot-approve keeps literals and
356
+ # they have to agree. Assert every bot login written anywhere in the router is in the list.
357
+ trusted="$(router_env TRUSTED_BOTS)"
358
+ if [ -z "$trusted" ]; then
359
+ MIRROR_OK=0; echo "FAIL: work-router.yml defines no TRUSTED_BOTS in its env: block" >&2
360
+ else
361
+ while IFS= read -r login; do
362
+ [ -n "$login" ] || continue
363
+ case " $trusted " in
364
+ *" $login "*) ;;
365
+ *)
366
+ MIRROR_OK=0
367
+ echo "FAIL: work-router.yml names bot '${login}' but env.TRUSTED_BOTS does not list it" >&2
368
+ ;;
369
+ esac
370
+ done < <(grep -oE "'(app/[a-z-]+|[a-z-]+\[bot\])'|\"(app/[a-z-]+|[a-z-]+\[bot\])\"" "$ROUTER_YML" |
371
+ tr -d "'\"" | sort -u)
372
+ fi
373
+ if [ "$MIRROR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
374
+
375
+ # Run the belt's own jq, rather than reading it. The filter that picks which open pull requests
376
+ # the hourly reconcile job acts on was written as
377
+ # ((env.TRUSTED_BOTS | split(" ")) | index(.user.login) != null)
378
+ # which dies at runtime with `Cannot index array with string "user"`, because inside index() the
379
+ # input is the array, not the pull request. Every assertion here passed: one checked the bot
380
+ # logins were listed in TRUSTED_BOTS, another that the router named no bot outside that list.
381
+ # Nothing executed the program. It failed hourly in production for a day, on the one job whose
382
+ # purpose is to keep stuck pull requests moving. Extract it and give it inputs.
383
+ BELT_OK=1
384
+ bot_pr_filter=$(awk '/jq -r --arg repo "\$REPO"/{found=1;next} found && /^ *'"'"' \|$/{exit} found' "$ROUTER_YML")
385
+ if [ -z "$bot_pr_filter" ]; then
386
+ BELT_OK=0
387
+ echo "FAIL: could not extract the open-pull-request filter from work-router.yml" >&2
388
+ else
389
+ # One of each: a trusted App under both spellings, a human, a draft, and a fork.
390
+ belt_fixture='[
391
+ {"number":11,"draft":false,"user":{"login":"app/github-actions"},"head":{"ref":"a","sha":"s1","repo":{"full_name":"o/r"}}},
392
+ {"number":12,"draft":false,"user":{"login":"platform-devbox[bot]"},"head":{"ref":"b","sha":"s2","repo":{"full_name":"o/r"}}},
393
+ {"number":13,"draft":false,"user":{"login":"a-person"},"head":{"ref":"c","sha":"s3","repo":{"full_name":"o/r"}}},
394
+ {"number":14,"draft":true,"user":{"login":"app/github-actions"},"head":{"ref":"d","sha":"s4","repo":{"full_name":"o/r"}}},
395
+ {"number":15,"draft":false,"user":{"login":"app/github-actions"},"head":{"ref":"e","sha":"s5","repo":{"full_name":"fork/r"}}}
396
+ ]'
397
+ if ! selected=$(printf '%s' "$belt_fixture" |
398
+ TRUSTED_BOTS="$trusted" jq -r --arg repo "o/r" "$bot_pr_filter" 2>&1 | cut -f1 | tr '\n' ' '); then
399
+ BELT_OK=0
400
+ echo "FAIL: the open-pull-request filter does not run: ${selected}" >&2
401
+ elif [ "$(echo "$selected" | tr -s ' ')" != "11 12 " ]; then
402
+ BELT_OK=0
403
+ echo "FAIL: the belt selected pull requests [${selected}]; expected the two bot-authored ones (11 12)" >&2
404
+ fi
405
+ fi
406
+ if [ "$BELT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
407
+
408
+ # `review` must not be a one-way door. It used to be: authorize-bot-work refused to fire on an
409
+ # issue carrying it, and the classifier refuses to route while it is set, so a person adding
410
+ # `refine` to a parked issue got nothing at all — no run, no comment, no error. Triage's own
411
+ # needs-maintainer verdict tells the maintainer to do exactly that, so the bot was giving an
412
+ # instruction the machine ignored. authorize-bot-work now clears `review` before handing over,
413
+ # which is what makes the human's decision stick.
414
+ AUTHORIZE_YML="${WORKFLOWS_DIR}/authorize-bot-work.yml"
415
+ if [ -f "$AUTHORIZE_YML" ]; then
416
+ DOOR_OK=1
417
+ authorize_if=$(sed -n '/^ if: >/,/^ runs-on:/p' "$AUTHORIZE_YML")
418
+ if grep -q "labels\.\*\.name, 'review'" <<<"$authorize_if"; then
419
+ DOOR_OK=0
420
+ echo "FAIL: authorize-bot-work refuses issues carrying review; a human could not un-park one" >&2
421
+ fi
422
+ # It must still refuse the bot, and an issue another run already owns.
423
+ grep -q "endsWith(github.actor, '\[bot\]')" <<<"$authorize_if" || {
424
+ DOOR_OK=0
425
+ echo "FAIL: authorize-bot-work no longer excludes bot actors; it would re-trigger itself" >&2
426
+ }
427
+ grep -q "labels\.\*\.name, 'bot-working'" <<<"$authorize_if" || {
428
+ DOOR_OK=0
429
+ echo "FAIL: authorize-bot-work no longer excludes an issue a run already owns" >&2
430
+ }
431
+ # The hand-off has to clear review BEFORE adding bot-working, because bot-working is the event
432
+ # the classifier reads: the other order raises an event whose payload still carries review.
433
+ remove_line=$(grep -n -- '--remove-label "review"' "$AUTHORIZE_YML" | head -1 | cut -d: -f1)
434
+ add_line=$(grep -n -- '--add-label "bot-working"' "$AUTHORIZE_YML" | head -1 | cut -d: -f1)
435
+ if [ -z "$remove_line" ] || [ -z "$add_line" ] || [ "$remove_line" -ge "$add_line" ]; then
436
+ DOOR_OK=0
437
+ echo "FAIL: authorize-bot-work must remove review before adding bot-working (review=${remove_line:-none} bot-working=${add_line:-none})" >&2
438
+ fi
439
+ if [ "$DOOR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
440
+ fi
441
+
442
+ # The classifier's own guard stays: it stops the bot re-triggering itself while a human is
443
+ # needed. Both halves matter, so assert the pair rather than either alone.
444
+ assert_route "a bot-working event on a review-labelled issue still routes nowhere" none \
445
+ EVENT=issues ACTION=labeled LABEL=bot-working ACTOR=platform-devbox[bot] \
446
+ 'ISSUE_LABELS=["implement","review"]' EVENT_ISSUE_NUMBER=42
447
+ assert_route "and routes normally once review has been cleared" implement \
448
+ EVENT=issues ACTION=labeled LABEL=bot-working ACTOR=platform-devbox[bot] \
449
+ 'ISSUE_LABELS=["implement"]' EVENT_ISSUE_NUMBER=42
450
+
198
451
  # GitHub evaluates every Actions expression in a workflow file, including ones written inside
199
452
  # shell comments. An empty pair is not a valid expression and fails the whole file to parse,
200
453
  # with an error that points at a line number rather than saying what is wrong. Prose about
201
454
  # expressions must not contain one.
202
- empty_expr=$(grep -rl -e '${{[[:space:]]*}}' "${HERE}/../../workflows"/*.yml "${HERE}/../../workflows"/*.md 2>/dev/null || true)
455
+ empty_expr=$(grep -rl -e '${{[[:space:]]*}}' "${WORKFLOWS_DIR}"/*.yml "${WORKFLOWS_DIR}"/*.md 2>/dev/null || true)
203
456
  if [ -z "$empty_expr" ]; then
204
457
  PASS=$((PASS + 1))
205
458
  else
206
459
  FAIL=$((FAIL + 1))
207
460
  echo "FAIL: workflow files contain an empty Actions expression:" >&2
208
- printf ' %s
209
- ' $empty_expr >&2
461
+ while IFS= read -r offending; do echo " $offending" >&2; done <<<"$empty_expr"
210
462
  fi
211
463
 
212
-
213
464
  # A hyphen inside a ${{ }} property path is parsed as subtraction, so the reference silently
214
465
  # resolves to nothing and the rendered prompt keeps the raw expression. Underscores only.
215
- if ! grep -qE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD"; then
216
- PASS=$((PASS + 1))
217
- else
218
- FAIL=$((FAIL + 1))
219
- echo "FAIL: implement worker reads a hyphenated job output inside an expression" >&2
220
- grep -nE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD" >&2
466
+ if worker_installed implement; then
467
+ if ! grep -qE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD"; then
468
+ PASS=$((PASS + 1))
469
+ else
470
+ FAIL=$((FAIL + 1))
471
+ echo "FAIL: implement worker reads a hyphenated job output inside an expression" >&2
472
+ grep -nE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD" >&2
473
+ fi
221
474
  fi
222
475
 
223
- # A worker that prints a ${VERIFY_COMMANDS_*} block without setting it renders an empty command
224
- # block, and the model invents its own build line. That is how a child shipped `dotnet build
225
- # --no-restore` against an unrestored workspace. The commands are split per area so a change
226
- # that only touches apps/web does not pay for a cold Release build of the API.
476
+ # A worker that prints `${{ env.NAME }}` without defining NAME in its own env: block renders
477
+ # an empty value, and the model fills the gap itself. That is how a child shipped `dotnet build
478
+ # --no-restore` against an unrestored workspace: the verification block was empty. Every name a
479
+ # worker prints must be defined in that worker. The values are consumer-owned (a consumer may
480
+ # split VERIFY_COMMANDS per area, or keep one); only the wiring is asserted here.
227
481
  VERIFY_OK=1
228
- for VAR in VERIFY_COMMANDS_API VERIFY_COMMANDS_WEB; do
229
- if grep -q "env\.$VAR" "$IMPLEMENT_WORKER_MD" && ! grep -q "^ $VAR:" "$IMPLEMENT_WORKER_MD"; then
230
- VERIFY_OK=0
231
- echo "FAIL: implement worker prints $VAR without defining it" >&2
482
+ for worker in "${WORKFLOWS_DIR}"/agent-*.md; do
483
+ [ -f "$worker" ] || continue
484
+ while read -r name; do
485
+ [ -n "$name" ] || continue
486
+ if ! grep -q "^ ${name}:" "$worker"; then
487
+ VERIFY_OK=0
488
+ echo "FAIL: $(basename "$worker") prints env.${name} without defining it" >&2
489
+ fi
490
+ done < <(grep -oE '\$\{\{ *env\.[A-Za-z_][A-Za-z0-9_]* *\}\}' "$worker" | sed -E 's/.*env\.([A-Za-z_][A-Za-z0-9_]*).*/\1/' | sort -u)
491
+ done
492
+ if [ "$VERIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
493
+
494
+ # A protected path holds the merge for a human but must never stop the agent repairing failed
495
+ # CI on those same files, or the pull request strands with nobody able to fix it. That pair of
496
+ # conditions is decided once, in protected_changes.outputs.holds_review, and read everywhere
497
+ # else; it used to be restated at eight call sites. Auto-merge stays blocked separately, by
498
+ # conclude's own guard on requires_review, which holds even when CI failed.
499
+ if worker_installed implement && worker_installed merge-gate; then
500
+ PROTECTED_OK=1
501
+ grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" || PROTECTED_OK=0
502
+ grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
503
+ grep -Fq "holds_review: \${{ steps.files.outputs.requires_review == 'true' && needs.subject.outputs.conclusion != 'failure' }}" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
504
+ # The decision must not be re-derived anywhere: one definition, everything else reads it.
505
+ if [ "$(count -c "requires_review == 'true' && needs.subject.outputs.conclusion != 'failure'" "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
506
+ PROTECTED_OK=0
507
+ echo "FAIL: the protected-files hold is derived in more than one place; read holds_review instead" >&2
508
+ fi
509
+ # And conclude must still refuse to merge a protected pull request whatever CI said.
510
+ grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.validate_output.outputs.outcome != 'merge'" "$MERGE_GATE_WORKER_MD" || PROTECTED_OK=0
511
+ if [ "$PROTECTED_OK" -eq 1 ]; then
512
+ PASS=$((PASS + 1))
513
+ else
514
+ FAIL=$((FAIL + 1))
515
+ echo "FAIL: protected changes must allow failed-CI repair while remaining held from merge" >&2
516
+ fi
517
+ fi
518
+
519
+ # gh-aw folds the worker's top-level `if:` into the generated activation job but computes
520
+ # activation's `needs` on its own: only custom jobs the prompt references AND that declare no
521
+ # `needs:` are hoisted. A guard with its own `needs:` (protected_changes needs subject) is read
522
+ # before it has run, resolves to '' and gates nothing, unless it is listed in `on.needs`, the
523
+ # documented way to add jobs to pre_activation and activation. Inline list form is expected.
524
+ if worker_installed merge-gate; then
525
+ TOP_IF="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n 's/^if: //p')"
526
+ ON_NEEDS="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n '/^on:$/,/^[a-z]/p' |
527
+ sed -n 's/^ needs: *\[\(.*\)\].*/\1/p' | tr -d ' ' | tr ',' '\n')"
528
+ ACTIVATION_OK=1
529
+ [ -n "$TOP_IF" ] || { ACTIVATION_OK=0; echo "FAIL: could not read the merge-gate worker's top-level if" >&2; }
530
+ while read -r job; do
531
+ [ -n "$job" ] || continue
532
+ if tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n "/^ ${job}:$/,/^ [a-z_]*:$/p" | grep -q '^ needs:' &&
533
+ ! grep -qx "$job" <<<"$ON_NEEDS"; then
534
+ ACTIVATION_OK=0
535
+ echo "FAIL: merge-gate top-level if reads needs.${job}, which has its own needs and is not in on.needs; activation would read it before it runs" >&2
536
+ fi
537
+ done < <(grep -oE 'needs\.[a-z_]+\.' <<<"$TOP_IF" | sed 's/^needs\.//; s/\.$//' | sort -u)
538
+ if [ "$ACTIVATION_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
539
+
540
+ # The merge belt is serial for the whole repository: several overnight pull requests
541
+ # mean every merge moves the default branch under the rest, and gates running at once
542
+ # rebase onto bases other gates are about to invalidate. A per-issue group here would
543
+ # reintroduce that race, so assert the repo-wide lock is the one in use.
544
+ if grep -A7 'call-merge-gate:' "$ROUTER_YML" | grep -q 'group: merge-belt'; then
545
+ PASS=$((PASS + 1))
546
+ else
547
+ FAIL=$((FAIL + 1))
548
+ echo "FAIL: call-merge-gate must hold the repo-wide merge-belt lock" >&2
549
+ fi
550
+ fi
551
+
552
+ # A verdict is the gate marker AND a `**Verdict:**` line together. Comments carrying the
553
+ # marker alone were progress notes and failed attempts, and the reconcile belt read every
554
+ # one of them as final: a crashed or OOM-killed gate parked its pull request for the rest
555
+ # of the night. Attempts are counted separately, capped, and reset by any new CI run.
556
+ # The belt lives in the router's plumbing jobs, so this holds in every repository.
557
+ BELT_OK=1
558
+ if ! grep -q 'agent-merge-gate-attempt' "$ROUTER_YML"; then
559
+ BELT_OK=0; echo "FAIL: router never counts gate attempts" >&2
560
+ fi
561
+ if [ "$(count -cF 'contains("<!-- agent-merge-gate -->")) and (.body | contains("**Verdict:**"))' "$ROUTER_YML")" -lt 4 ]; then
562
+ BELT_OK=0; echo "FAIL: verdict detection must pair the gate marker with a Verdict line in both dispatch paths" >&2
563
+ fi
564
+ if [ "$(count -c 'attempts_so_far' "$ROUTER_YML")" -lt 2 ]; then
565
+ BELT_OK=0; echo "FAIL: dispatch sites must forward attempts_so_far" >&2
566
+ fi
567
+ # A second gate for a pull request whose gate is already queued or running reads the same CI
568
+ # verdict and is cancelled by the single-slot merge-belt queue (two cancellations on 2026-09-06).
569
+ if [ "$(count -c 'a merge-gate run is already live' "$ROUTER_YML")" -lt 2 ]; then
570
+ BELT_OK=0; echo "FAIL: both dispatch paths must skip a pull request whose gate is already live" >&2
571
+ fi
572
+ # A conflicting pull request has no refs/pull/N/merge for GitHub to build, so a `pull_request`
573
+ # CI workflow can never run on that head. Requiring a fresh verdict before dispatching deadlocks
574
+ # the belt: only the gate resolves the conflict, and the gate never runs. Both paths fall back to
575
+ # the branch's last verdict when, and only when, the pull request is conflicting.
576
+ if [ "$(count -c 'conflicts, so CI cannot run on' "$ROUTER_YML")" -lt 2 ]; then
577
+ BELT_OK=0
578
+ echo "FAIL: both dispatch paths must gate a conflicting pull request that can never get fresh CI" >&2
579
+ fi
580
+ # That fallback has to read the computed mergeable state. The REST boolean is null until GitHub
581
+ # recomputes it, and stays null for a pull request nobody has opened recently, which is exactly
582
+ # the stale conflicting pull request the fallback exists for: it never fired once in production.
583
+ # The state has to be polled, not read once. GitHub computes mergeability on demand and the
584
+ # first read answers UNKNOWN (or null through REST) while it works it out, so a single read
585
+ # reports "not conflicting" for exactly the stale pull requests the fallback is for. Observed
586
+ # twice in production: the fallback logged "no completed CI run" for a pull request that
587
+ # `gh pr view` reported as CONFLICTING from a warm cache seconds later.
588
+ if [ "$(count -c 'mergeable_state()' "$ROUTER_YML")" -lt 2 ] ||
589
+ [ "$(count -c 'mergeable_now=$(mergeable_state' "$ROUTER_YML")" -lt 2 ]; then
590
+ BELT_OK=0
591
+ echo "FAIL: both dispatch paths must poll the mergeable state; a single read answers UNKNOWN" >&2
592
+ fi
593
+ if [ "$BELT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
594
+
595
+ # GitHub delivers workflow_run only for CI runs whose actor is a human, so a bot pull request's
596
+ # CI never reaches the router's CI-completion route. The package ships a dispatch-merge-gate job
597
+ # in templates/ci that hands the verdict over from inside CI; a consumer CI workflow, where one
598
+ # exists beside the router, must carry it or bot pull requests wait for the hourly belt.
599
+ for ci in "${WORKFLOWS_DIR}/ci.yml" "${WORKFLOWS_DIR}/app-ci.yml"; do
600
+ [ -f "$ci" ] || continue
601
+ if grep -q 'operation=merge-gate' "$ci"; then
602
+ PASS=$((PASS + 1))
603
+ else
604
+ FAIL=$((FAIL + 1))
605
+ echo "FAIL: $(basename "$ci") has no dispatch-merge-gate job; bot pull requests would wait for the hourly belt" >&2
232
606
  fi
233
607
  done
234
- if grep -qE 'env\.VERIFY_COMMANDS[^_]' "$IMPLEMENT_WORKER_MD"; then
235
- VERIFY_OK=0
236
- echo "FAIL: implement worker still references the unscoped VERIFY_COMMANDS" >&2
608
+
609
+ # The router forwards a fact to a worker by reading `needs.classify.outputs.<x>`; a name the
610
+ # classify job does not export resolves to '' with no error. That is how the gate received
611
+ # attempts_so_far='' (the classifier emitted merge-gate-attempts, the job never exported it),
612
+ # fromJson('') killed the incomplete job before its attempt comment, and the belt re-dispatched
613
+ # the same crash every hour. Every name the router reads must be exported by the classify job.
614
+ CLASSIFY_EXPORTS="$(tr -d '\r' <"$ROUTER_YML" |
615
+ sed -n '/^ classify:$/,/^ [a-z-]*:$/p' |
616
+ sed -n '/^ outputs:$/,/^ [a-z]*:$/p' |
617
+ sed -n 's/^ \([a-zA-Z0-9_-]*\):.*/\1/p')"
618
+ CLASSIFY_OK=1
619
+ [ -n "$CLASSIFY_EXPORTS" ] || { CLASSIFY_OK=0; echo "FAIL: could not read the classify job's outputs from work-router.yml" >&2; }
620
+ while read -r name; do
621
+ [ -n "$name" ] || continue
622
+ if ! grep -qx "$name" <<<"$CLASSIFY_EXPORTS"; then
623
+ CLASSIFY_OK=0
624
+ echo "FAIL: work-router.yml reads needs.classify.outputs.${name} but the classify job does not export it" >&2
625
+ fi
626
+ done < <(grep -oE 'needs\.classify\.outputs\.[a-zA-Z0-9_-]+' "$ROUTER_YML" | sed 's/.*\.//' | sort -u)
627
+ if [ "$CLASSIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
628
+
629
+ if worker_installed merge-gate; then
630
+ # fromJson('') is a hard failure ("Error reading JToken"), and a workflow_call input arrives as
631
+ # '' whenever the caller passes an empty expression, declared default or not. The gate must never
632
+ # hand a raw input to fromJson; `inputs.x || '0'` reads the empty case as zero.
633
+ if ! grep -qE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD"; then
634
+ PASS=$((PASS + 1))
635
+ else
636
+ FAIL=$((FAIL + 1))
637
+ echo "FAIL: merge-gate worker calls fromJson on a raw input; an empty caller value kills the job" >&2
638
+ grep -nE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD" >&2
639
+ fi
640
+
641
+ # The worker's own comments must keep the distinction: progress notes carry no marker,
642
+ # failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
643
+ # Three verdict sites: the review hold on the issue, the agent's assessment on the issue,
644
+ # and conclude's short verdict on the pull request itself.
645
+ if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
646
+ [ "$(count -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 3 ]; then
647
+ PASS=$((PASS + 1))
648
+ else
649
+ FAIL=$((FAIL + 1))
650
+ echo "FAIL: merge-gate worker must keep verdict and attempt markers distinct" >&2
651
+ fi
237
652
  fi
238
- if [ "$VERIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
239
653
 
240
- if grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" &&
241
- grep -Fq 'protected-files: allowed' "$MERGE_GATE_WORKER_MD" &&
242
- grep -Fq "needs.protected_changes.outputs.requires_review != 'true' || needs.subject.outputs.conclusion == 'failure'" "$MERGE_GATE_WORKER_MD"; then
243
- PASS=$((PASS + 1))
244
- else
245
- FAIL=$((FAIL + 1))
246
- echo "FAIL: protected changes must allow failed-CI repair while remaining held from merge" >&2
654
+ # add-issue-labels and remove-issue-labels split `labels` on newlines. A caller that joined two
655
+ # names with a comma removed one label called "bot-working,pr-pending": a 404 the action swallows
656
+ # on purpose, so the release never happened and Pliny-Bot #49/#54 carried implement, pr-pending
657
+ # and review together for a day. Callers use block scalars, one label per line; the actions also
658
+ # accept commas so a consumer copy of an old caller keeps working.
659
+ LABELS_OK=1
660
+ if grep -nE '^[[:space:]]+labels: [^|>].*,' "${WORKFLOWS_DIR}"/agent-*.md >&2 2>/dev/null; then
661
+ LABELS_OK=0
662
+ echo "FAIL: a worker passes comma-joined labels to a label action; use a block scalar, one label per line" >&2
663
+ fi
664
+ for action in add-issue-labels remove-issue-labels; do
665
+ if ! grep -qF 'split(/\r?\n|,/)' "${HERE}/../${action}/action.yml"; then
666
+ LABELS_OK=0
667
+ echo "FAIL: ${action} must accept comma-separated labels as well as one per line" >&2
668
+ fi
669
+ done
670
+ if [ "$LABELS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
671
+
672
+ if worker_installed merge-gate; then
673
+ # The agent's fix reaches the branch as a bundle applied fast-forward only (apply-agent-output).
674
+ # gh-aw's push tool description tells the model to rebase, and a rebased branch cannot
675
+ # fast-forward: the push is refused and the verdict is lost (Pliny-Bot run 33952565835). The
676
+ # worker must start on the pull request branch and must never say `git rebase`. Its progress
677
+ # comment is posted on the first attempt only; retries are recorded by the attempt comment.
678
+ BRANCH_OK=1
679
+ # Path B: staged safe outputs, applied by conclude with the App token. Without `staged: true`
680
+ # gh-aw's safe_outputs job writes too, and it runs first: it pushed a flattened single-parent
681
+ # commit with GITHUB_TOKEN, which lost the agent's merge, left the pull request conflicting,
682
+ # and started no CI, because GITHUB_TOKEN writes raise no events.
683
+ if ! grep -qE '^ staged: true' "$MERGE_GATE_WORKER_MD"; then
684
+ BRANCH_OK=0; echo "FAIL: merge-gate safe-outputs must be staged; conclude owns the write path" >&2
685
+ fi
686
+ if grep -q 'git rebase' "$MERGE_GATE_WORKER_MD"; then
687
+ BRANCH_OK=0; echo "FAIL: merge-gate worker tells the agent to rebase; the push is fast-forward only" >&2
688
+ fi
689
+ if ! grep -q 'name: Check out the pull request branch' "$MERGE_GATE_WORKER_MD"; then
690
+ BRANCH_OK=0; echo "FAIL: merge-gate worker must check out the pull request branch before the agent starts" >&2
691
+ fi
692
+ if ! grep -qF "conclusion == 'failure' && (inputs.attempts_so_far || '0') == '0'" "$MERGE_GATE_WORKER_MD"; then
693
+ BRANCH_OK=0; echo "FAIL: the reserve job's progress comment must be posted on the first attempt only" >&2
694
+ fi
695
+ # A conflicting pull request has no CI run to read logs from, so the gate is handed empty
696
+ # failure artifacts. Read on its own that looks like "no evidence", and the agent asked for a
697
+ # human instead of resolving the conflict that caused it.
698
+ if ! grep -qF 'Empty failure evidence is not a reason to ask for review' "$MERGE_GATE_WORKER_MD"; then
699
+ BRANCH_OK=0
700
+ echo "FAIL: the gate must treat empty failure evidence on a conflicting PR as the conflict to fix" >&2
701
+ fi
702
+ if [ "$BRANCH_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
703
+
704
+ # pr-pending means a pull request for this issue is open and waiting. Only merging retires it.
705
+ # Every other path (the protected-files hold, a review verdict, a failed attempt) leaves the
706
+ # pull request open, and stripping the label there produced a board where issues with open
707
+ # pull requests looked like they had none. It went unnoticed while the label actions silently
708
+ # removed nothing, so the two bugs hid each other.
709
+ PENDING_OK=1
710
+ grep -q '^ PR_PENDING_LABEL:' "$MERGE_GATE_WORKER_MD" ||
711
+ { PENDING_OK=0; echo "FAIL: merge gate lost its PR_PENDING_LABEL definition" >&2; }
712
+ if [ "$(count -c '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
713
+ PENDING_OK=0
714
+ echo "FAIL: pr-pending must be removed in exactly one place, the merge path" >&2
715
+ grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
716
+ fi
717
+ # And that one place has to be the merge outcome, not a hold or a failed attempt.
718
+ grep -B12 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'merge'" ||
719
+ { PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge outcome" >&2; }
720
+
721
+ # The invariant only ever looked at the merge gate, so apply-review quietly stripped the label
722
+ # on its already-satisfied and needs-human paths — both of which leave the pull request open.
723
+ # The one file the check ignored was the one breaking it. Look at every worker: implement adds
724
+ # the label, merge-gate removes it on merge, nobody else may touch it.
725
+ for worker in "${WORKFLOWS_DIR}"/agent-*.md; do
726
+ [ -f "$worker" ] || continue
727
+ case "$(basename "$worker")" in
728
+ agent-merge-gate.md | agent-implement.md) continue ;;
729
+ esac
730
+ if grep -q 'remove-issue-labels' "$worker" &&
731
+ grep -A8 'remove-issue-labels' "$worker" | grep -q 'env.PR_PENDING_LABEL'; then
732
+ PENDING_OK=0
733
+ echo "FAIL: $(basename "$worker") removes pr-pending; only the merge gate's merge path may" >&2
734
+ fi
735
+ done
736
+ if [ "$PENDING_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
737
+ fi
738
+
739
+ if worker_installed implement; then
740
+ # A provider outage kills a run in a couple of minutes with no answer, and the same issue used
741
+ # to be handed to a human for it. The implement worker retries those and only those: a run that
742
+ # worked for half an hour and then failed produced an answer that was wrong, and repeating it
743
+ # costs the fleet the same half hour to be wrong again.
744
+ IMPLEMENT_RETRY_OK=1
745
+ for needle in 'RETRY_UNDER_MINUTES' 'ATTEMPT_MARKER' 'attempts_so_far' 'operation=implement'; do
746
+ grep -qF "$needle" "$IMPLEMENT_WORKER_MD" || {
747
+ IMPLEMENT_RETRY_OK=0
748
+ echo "FAIL: implement worker lost its retry belt: no '$needle'" >&2
749
+ }
750
+ done
751
+ # Park and retry are mutually exclusive: the retry path must never add the review label, and
752
+ # the park path must never re-dispatch.
753
+ grep -A3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" ||
754
+ grep -B3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" || {
755
+ IMPLEMENT_RETRY_OK=0
756
+ echo "FAIL: the implement worker must not flag review on a run it is about to retry" >&2
757
+ }
758
+ if [ "$IMPLEMENT_RETRY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
759
+ fi
760
+
761
+ if worker_installed merge-gate; then
762
+ # A failed attempt must not strip `implement`: identify-gate-subject refuses an issue
763
+ # without it, so the first crash would starve every retry at the subject check.
764
+ if grep -A9 'Park the issue' "$MERGE_GATE_WORKER_MD" | grep -q 'REVIEW_LABEL' &&
765
+ ! grep -qF 'labels: ${{ env.WORKING_LABEL }},${{ env.IMPLEMENT_LABEL }}' "$MERGE_GATE_WORKER_MD"; then
766
+ PASS=$((PASS + 1))
767
+ else
768
+ FAIL=$((FAIL + 1))
769
+ echo "FAIL: the incomplete job must keep implement and only park on an exhausted budget" >&2
770
+ fi
247
771
  fi
248
772
 
249
773
  # This repository is public. Every route a human can start from a comment, a review or a
@@ -251,6 +775,7 @@ fi
251
775
  # writes code. Asserted here because removing the gate would otherwise be a silent, one-line
252
776
  # change that nothing fails on.
253
777
  for route in refine implement apply-review; do
778
+ worker_installed "$route" || continue
254
779
  if grep -qE "route == '${route}'.*needs\.authorize\.outputs\.trusted == 'true'" "$ROUTER_YML"; then
255
780
  PASS=$((PASS + 1))
256
781
  else
@@ -261,17 +786,83 @@ done
261
786
 
262
787
  # Triage runs under a trusted App identity. Outside collaborators are admitted only to
263
788
  # the deterministic dispatcher; the worker call itself requires a trusted actor.
264
- if grep -qE "dispatch-triage:.*" "$ROUTER_YML" && \
265
- grep -qE "route == 'triage'.*is_outside_collaborator == 'true'" "$ROUTER_YML" && \
266
- grep -qE "route == 'triage'.*trusted == 'true'" "$ROUTER_YML"; then
267
- PASS=$((PASS + 1))
268
- else
269
- FAIL=$((FAIL + 1))
270
- echo "FAIL: route 'triage' does not dispatch outside collaborators and require a trusted worker actor" >&2
789
+ if worker_installed triage; then
790
+ if grep -qE "dispatch-triage:.*" "$ROUTER_YML" && \
791
+ grep -qE "route == 'triage'.*is_outside_collaborator == 'true'" "$ROUTER_YML" && \
792
+ grep -qE "route == 'triage'.*trusted == 'true'" "$ROUTER_YML"; then
793
+ PASS=$((PASS + 1))
794
+ else
795
+ FAIL=$((FAIL + 1))
796
+ echo "FAIL: route 'triage' does not dispatch outside collaborators and require a trusted worker actor" >&2
797
+ fi
271
798
  fi
272
799
 
273
- for route in refine implement triage apply-review merge-gate audit bot-approve \
274
- audit-close cleanup-artifacts reconcile-bot-pr-runs validate release; do
800
+ # Out of scope is the wrong door, not a rejection, and it must never close an issue. Numa#654
801
+ # was a reproducible authorization defect that passed nine of ten checks and was closed as
802
+ # not_planned with every label stripped, so nobody would ever have found it. The verdict for
803
+ # that case is needs-maintainer: open, review label, a maintainer adds refine to take it on.
804
+ # Only block closes, and only for work that cannot be done or is unsafe.
805
+ if worker_installed triage; then
806
+ TRIAGE_WORKER_MD="${WORKFLOWS_DIR}/agent-triage.md"
807
+ VALIDATE_TRIAGE_SH="${HERE}/../validate-triage-output/validate-triage-output.sh"
808
+ TRIAGE_OK=1
809
+
810
+ # The validator is what turns the agent's prose into the outcome the jobs branch on. A
811
+ # verdict it does not know becomes "invalid", which skips conclude entirely and reports the
812
+ # run incomplete, so the prompt and this script have to agree on all four names.
813
+ # Comment lines stripped first: the file explains the verdicts in prose above the program,
814
+ # and a plain search finds the name there even after it has been dropped from the jq
815
+ # alternation, which is exactly the regression this is meant to catch.
816
+ validate_program=$(grep -v '^[[:space:]]*#' "$VALIDATE_TRIAGE_SH")
817
+ for verdict in pass needs-info needs-maintainer block; do
818
+ if [ "$(count -c -- "$verdict" <<<"$validate_program")" -lt 3 ]; then
819
+ TRIAGE_OK=0
820
+ echo "FAIL: validate-triage-output.sh does not accept the '${verdict}' verdict in test(), capture() and the guard" >&2
821
+ fi
822
+ if ! grep -qF "\`**Verdict:** ${verdict}\`" "$TRIAGE_WORKER_MD"; then
823
+ TRIAGE_OK=0
824
+ echo "FAIL: the triage prompt does not offer '**Verdict:** ${verdict}'" >&2
825
+ fi
826
+ done
827
+
828
+ # The assertion this whole route turns on: exactly one step closes an issue, and it is
829
+ # reached only by a block verdict.
830
+ closes=$(count -c "state: 'closed'" "$TRIAGE_WORKER_MD")
831
+ if [ "$closes" -ne 1 ]; then
832
+ TRIAGE_OK=0
833
+ echo "FAIL: agent-triage.md closes an issue in ${closes} places; expected exactly one" >&2
834
+ elif ! grep -B12 "state: 'closed'" "$TRIAGE_WORKER_MD" | grep -q "outcome == 'block'"; then
835
+ TRIAGE_OK=0
836
+ echo "FAIL: the triage close step is not guarded on a block verdict alone" >&2
837
+ fi
838
+ if grep -q "outcome == 'needs-maintainer'" "$TRIAGE_WORKER_MD"; then
839
+ if grep -A6 "outcome == 'needs-maintainer'" "$TRIAGE_WORKER_MD" | grep -q "state: 'closed'"; then
840
+ TRIAGE_OK=0
841
+ echo "FAIL: a needs-maintainer verdict closes the issue; it must stay open" >&2
842
+ fi
843
+ else
844
+ TRIAGE_OK=0
845
+ echo "FAIL: agent-triage.md has no needs-maintainer branch in conclude" >&2
846
+ fi
847
+
848
+ # Parked, not looping: review goes on so a human sees it, triage comes off so a later
849
+ # comment does not re-enter triage and put it out of scope again for ever.
850
+ maintainer_block=$(sed -n "/outcome == 'needs-maintainer'/,/outcome == 'block'/p" "$TRIAGE_WORKER_MD")
851
+ grep -q 'env.REVIEW_LABEL' <<<"$maintainer_block" || {
852
+ TRIAGE_OK=0
853
+ echo "FAIL: the needs-maintainer branch does not add the review label" >&2
854
+ }
855
+ grep -q 'env.TRIAGE_LABEL' <<<"$maintainer_block" || {
856
+ TRIAGE_OK=0
857
+ echo "FAIL: the needs-maintainer branch does not remove the triage label, so comments would re-trigger triage" >&2
858
+ }
859
+
860
+ if [ "$TRIAGE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
861
+ fi
862
+
863
+ # Every installed worker and every plumbing route has a job; a worker that is not installed
864
+ # has none, or the router would call a lock file that does not exist.
865
+ for route in "${INSTALLED_ROUTES[@]}" "${PLUMBING_ROUTES[@]}"; do
275
866
  if grep -q "route == '${route}'" "$ROUTER_YML"; then
276
867
  PASS=$((PASS + 1))
277
868
  else
@@ -279,6 +870,14 @@ for route in refine implement triage apply-review merge-gate audit bot-approve \
279
870
  echo "FAIL: work-router.yml has no job for route '${route}'" >&2
280
871
  fi
281
872
  done
873
+ for route in "${EXCLUDED_ROUTES[@]}"; do
874
+ if grep -q "route == '${route}'" "$ROUTER_YML"; then
875
+ FAIL=$((FAIL + 1))
876
+ echo "FAIL: work-router.yml has a job for route '${route}' but agent-${route}.md is not installed" >&2
877
+ else
878
+ PASS=$((PASS + 1))
879
+ fi
880
+ done
282
881
 
283
882
  while read -r operation; do
284
883
  if grep -q "route == '${operation}'" "$ROUTER_YML"; then
@@ -290,6 +889,368 @@ while read -r operation; do
290
889
  done < <(sed -n '/^ operation:/,/^ issue-number:/p' "$ROUTER_YML" |
291
890
  sed -n 's/^ - //p')
292
891
 
892
+ # A router job that reads pull requests has to say so. audit-close listed them with only
893
+ # contents and issues and failed nightly on a 403 that named the endpoint and nothing else.
894
+ # Paired job-to-scope rather than parsed out of each action: the jobs that touch pull
895
+ # requests are few and known, and naming them here is what makes the omission visible.
896
+ for pr_job in audit-close reconcile-bot-pr-runs detect-pr-conflicts housekeeping; do
897
+ if ! grep -q "^ ${pr_job}:$" "$ROUTER_YML"; then
898
+ continue
899
+ fi
900
+ pr_scopes=$(sed -n "/^ ${pr_job}:$/,/^ steps:$/p" "$ROUTER_YML")
901
+ if printf '%s' "$pr_scopes" | grep -qE '^ pull-requests: (read|write)$'; then
902
+ PASS=$((PASS + 1))
903
+ else
904
+ FAIL=$((FAIL + 1))
905
+ echo "FAIL: router job '${pr_job}' reads pull requests but grants no pull-requests scope" >&2
906
+ fi
907
+ done
908
+
909
+ # No written-down passwords in anything this repository ships. A throwaway credential for
910
+ # a test container is still a policy finding, and one sat in every consumer's CI for weeks
911
+ # until a scan found it rather than us. Two shapes: a password-ish name assigned a quoted
912
+ echo "── Housekeeping ──────────────────────────────────────────────────────────"
913
+
914
+ # The janitor is the only thing in the fleet that deletes a branch and closes an issue nobody
915
+ # asked it to close, and it runs unattended every six hours. Its guardrails are one-line
916
+ # conditions that would be easy to lose in an edit and impossible to notice afterwards, so
917
+ # they are asserted rather than trusted.
918
+ HOUSEKEEPING_YML="${HERE}/../housekeeping/action.yml"
919
+ if [ -f "$HOUSEKEEPING_YML" ]; then
920
+ HK_OK=1
921
+ hk() {
922
+ grep -qE "$1" "$HOUSEKEEPING_YML" || { HK_OK=0; echo "FAIL: housekeeping ${2}" >&2; }
923
+ }
924
+
925
+ # Every write goes through act(), which is the only place dry-run is honoured. A second
926
+ # write path would make --dry-run a lie exactly once, on the run that deletes something.
927
+ hk 'const act = async' 'has no act\(\) wrapper, so dry-run cannot be enforced in one place'
928
+ writes=$(count -cE 'github\.rest\.(issues\.(create|update|createComment|removeLabel|addLabels)|git\.deleteRef|actions\.createWorkflowDispatch)\(' "$HOUSEKEEPING_YML")
929
+ outside=$(awk '
930
+ /await act\(/ { inact = 1 }
931
+ inact && /github\.rest\.(issues\.(create|update|createComment|removeLabel|addLabels)|git\.deleteRef|actions\.createWorkflowDispatch)\(/ { seen++ }
932
+ inact && /^ \}\);$/ { inact = 0 }
933
+ END { print seen + 0 }
934
+ ' "$HOUSEKEEPING_YML")
935
+ if [ "$writes" -ne "$outside" ]; then
936
+ HK_OK=0
937
+ echo "FAIL: housekeeping performs ${writes} write(s) but only ${outside} are inside act(); dry-run would not cover the rest" >&2
938
+ fi
939
+
940
+ # A branch is someone's work until its pull request is finished. All three guards have to
941
+ # hold: never the default branch, never one with an open pull request, and never one whose
942
+ # pull requests were not all opened by a bot.
943
+ hk "branch\.name === defaultBranch. continue" 'can delete the default branch'
944
+ hk "p\.state === 'open'\)\) continue" 'can delete a branch whose pull request is still open'
945
+ hk 'isBot\(p\.user' 'can delete a branch from a human pull request'
946
+ hk 'forBranch\.length === 0. continue' 'can delete a branch that never had a pull request'
947
+
948
+ # Retrying a decision reproduces it. Only a park the machine caused carries `stalled`, and
949
+ # only those may be re-dispatched; everything else is reported.
950
+ hk "labels\.includes\('stalled'\)" 'retries parks that were decisions, not machine failures'
951
+ # Match the guard, not the phrase. `attempts >= maxRetries` also appears in the line that
952
+ # labels the digest entry, so grepping for the words alone still passed with the guard
953
+ # deleted from the `if` -- the same weak-assertion shape that let a deleted triage verdict
954
+ # through because the words survived in a comment.
955
+ hk 'if \(attempts >= maxRetries \|\| !work\) \{' 'has no retry budget guard on the retry path'
956
+
957
+ # The janitor closes issues, and the only issues it may close are a split parent whose
958
+ # children are all done and its own digest. Anything else is a person's to close.
959
+ closes=$(count -cE "state: 'closed'" "$HOUSEKEEPING_YML")
960
+ if [ "$closes" -eq 2 ]; then
961
+ PASS=$((PASS + 1))
962
+ else
963
+ HK_OK=0
964
+ echo "FAIL: housekeeping closes issues in ${closes} place(s); only the split parent and its own digest are allowed" >&2
965
+ fi
966
+
967
+ # A retry is a workflow_dispatch, and GitHub starts no workflow run from an event raised
968
+ # with GITHUB_TOKEN. Wiring the default token here would make every retry a silent no-op:
969
+ # green run, comment posted, labels removed, and nothing ever picks the issue up again.
970
+ hk_job=$(sed -n '/^ housekeeping:$/,/^ [a-z0-9_-]*:$/p' "$ROUTER_YML")
971
+ if printf '%s' "$hk_job" | grep -q 'app-token.outputs.token'; then
972
+ PASS=$((PASS + 1))
973
+ else
974
+ HK_OK=0
975
+ echo "FAIL: the housekeeping job passes a token that cannot start a workflow run; retries would silently do nothing" >&2
976
+ fi
977
+ # Deleting a ref needs contents: write. Without it every delete answers 403 and the sweep
978
+ # reports success having removed nothing.
979
+ if printf '%s' "$hk_job" | grep -qE '^ contents: write$'; then
980
+ PASS=$((PASS + 1))
981
+ else
982
+ HK_OK=0
983
+ echo "FAIL: the housekeeping job deletes branches but grants no contents: write scope" >&2
984
+ fi
985
+
986
+ # Every knob the action takes is a repository's to change, so each has to come from the
987
+ # router's env: block, which is the one part of the file `workflows update` preserves.
988
+ for knob in HOUSEKEEPING_RETRY_AFTER_HOURS HOUSEKEEPING_MAX_RETRIES HOUSEKEEPING_STALE_PR_DAYS HOUSEKEEPING_DIGEST_TITLE; do
989
+ if [ -n "$(router_env "$knob")" ] && printf '%s' "$hk_job" | grep -q "env.${knob}"; then
990
+ PASS=$((PASS + 1))
991
+ else
992
+ HK_OK=0
993
+ echo "FAIL: ${knob} is not both declared in the router env: block and read by the housekeeping job" >&2
994
+ fi
995
+ done
996
+
997
+ if [ "$HK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
998
+ fi
999
+
1000
+ # The audit chain closes its own reports, and every way it can be wrong is silent: a report
1001
+ # closed as completed with nothing implemented, or a report pinned open forever so the next
1002
+ # audit never runs. Both happened. Assert the three conditions that decide it.
1003
+ AUDIT_CLOSE_YML="${HERE}/../audit-close/action.yml"
1004
+ if [ -f "$AUDIT_CLOSE_YML" ] && worker_installed audit; then
1005
+ AC_OK=1
1006
+
1007
+ # A report referencing no issues had no work done on it. Closing that as `completed` is how
1008
+ # an audit used to end with every finding closed and nothing implemented.
1009
+ if grep -qE 'resolved === 0\) \{' "$AUDIT_CLOSE_YML" &&
1010
+ ! sed -n '/resolved === 0) {/,/^ }$/p' "$AUDIT_CLOSE_YML" | grep -q "state: 'closed'"; then
1011
+ PASS=$((PASS + 1))
1012
+ else
1013
+ AC_OK=0
1014
+ echo "FAIL: audit-close closes a report that references no issues; nothing was implemented from it" >&2
1015
+ fi
1016
+
1017
+ # Only a real closing keyword may pin a report open. One pattern here required the literal
1018
+ # `#closes #12` and matched nothing; the other matched a bare `#12` anywhere in any open
1019
+ # pull request and pinned the report open for as long as that pull request lived.
1020
+ if grep -q 'clos(?:e|es|ed)' "$AUDIT_CLOSE_YML" && ! grep -qF '#(?:closes?' "$AUDIT_CLOSE_YML"; then
1021
+ PASS=$((PASS + 1))
1022
+ else
1023
+ AC_OK=0
1024
+ echo "FAIL: audit-close still carries the dead '#closes #N' pattern or lost its closing-keyword match" >&2
1025
+ fi
1026
+
1027
+ # The backpressure query has to exclude what the chain marks stale, or three abandoned
1028
+ # reports disable the weekly audit permanently and the run skips green every week.
1029
+ # Read the query line itself, not the file. The prose above it explains what
1030
+ # `-label:stale-audit` is for, so a grep of the whole file passed with the exclusion deleted
1031
+ # from the query -- matching the comment that describes it.
1032
+ AUDIT_WORKER_MD="${WORKFLOWS_DIR}/agent-audit.md"
1033
+ audit_query="$(sed -n 's/^ *query: *"\(.*\)" *$/\1/p' "$AUDIT_WORKER_MD" | head -1)"
1034
+ if [[ "$audit_query" == *-label:stale-audit* ]] && grep -q "STALE_LABEL = 'stale-audit'" "$AUDIT_CLOSE_YML"; then
1035
+ PASS=$((PASS + 1))
1036
+ else
1037
+ AC_OK=0
1038
+ echo "FAIL: the audit backpressure query and audit-close disagree about stale-audit; abandoned reports would block every future audit" >&2
1039
+ fi
1040
+
1041
+ if [ "$AC_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1042
+ fi
1043
+
1044
+ echo "── Merge gate park ───────────────────────────────────────────────────────"
1045
+
1046
+ # A gate verdict parks the code it was given on. Both dispatch paths -- detect-pr-conflicts and
1047
+ # the reconcile belt -- used to compare the standing verdict against the CI finish time, which
1048
+ # made the park worthless: any later run on the same commits was newer than the verdict, so the
1049
+ # belt re-dispatched a pull request a human already owned and reset its attempt budget at the
1050
+ # same time. Lyceum PR #13 sat parked for six days while that happened. Asserted because both
1051
+ # comparisons are one line and neither failing produces a red run.
1052
+ if worker_installed merge-gate; then
1053
+ GATE_OK=1
1054
+
1055
+ # Neither path may key the park to CI timing again.
1056
+ stale_horizon=$(count -cE 'verdict" \\> "\$ci_finished"|latest_verdict" \\> "\$ci_finished"' "$ROUTER_YML")
1057
+ if [ "$stale_horizon" -eq 0 ]; then
1058
+ PASS=$((PASS + 1))
1059
+ else
1060
+ GATE_OK=0
1061
+ echo "FAIL: the merge-gate park is keyed to the CI finish time in ${stale_horizon} place(s); a CI re-run would reopen a park a person owns" >&2
1062
+ fi
1063
+
1064
+ # Both must fall back to the CI time only when the head commit cannot be read.
1065
+ horizons=$(count -cE '\$\{head_committed:-\$ci_finished\}' "$ROUTER_YML")
1066
+ if [ "$horizons" -eq 2 ]; then
1067
+ PASS=$((PASS + 1))
1068
+ else
1069
+ GATE_OK=0
1070
+ echo "FAIL: ${horizons} of the 2 gate dispatch paths key their park to the head commit" >&2
1071
+ fi
1072
+
1073
+ # One cap, not four literals, and it has to match what the worker tells the reader.
1074
+ router_cap="$(router_env MAX_GATE_ATTEMPTS)"
1075
+ worker_cap="$(sed -n 's/^ MAX_ATTEMPTS: "\([0-9]*\)"$/\1/p' "${WORKFLOWS_DIR}/agent-merge-gate.md" | head -1)"
1076
+ if [ -n "$router_cap" ] && [ "$router_cap" = "$worker_cap" ]; then
1077
+ PASS=$((PASS + 1))
1078
+ else
1079
+ GATE_OK=0
1080
+ echo "FAIL: the belt gives up after '${router_cap:-unset}' attempts but agent-merge-gate.md tells the reader '${worker_cap:-unset}'" >&2
1081
+ fi
1082
+ # And no path may go back to a literal. Counting the word `6` would match a hundred things,
1083
+ # so this looks only at the attempt comparison and the message beside it.
1084
+ if ! grep -qE '"\$attempts" -ge 6|attempts \+ 1\)\) of 6' "$ROUTER_YML"; then
1085
+ PASS=$((PASS + 1))
1086
+ else
1087
+ GATE_OK=0
1088
+ echo "FAIL: the gate attempt cap is hardcoded in work-router.yml instead of read from env.MAX_GATE_ATTEMPTS" >&2
1089
+ grep -nE '"\$attempts" -ge 6|attempts \+ 1\)\) of 6' "$ROUTER_YML" >&2
1090
+ fi
1091
+
1092
+ if [ "$GATE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1093
+ fi
1094
+
1095
+ echo "── Expression functions ──────────────────────────────────────────────────"
1096
+
1097
+ # GitHub's expression language has eleven functions and no more. There is no `split()`, no
1098
+ # `length()`, no `replace()`, and calling one is not a warning: the workflow fails to load with
1099
+ # "Unrecognized function", which shows up as a run that never starts. `split(env.REPO, '/')[0]`
1100
+ # was written into a template here and only caught by hand. actionlint would find it, but it
1101
+ # does not read composite manifests and is not installed in every consumer, so the same rule
1102
+ # lives here where the rest of the invariants are.
1103
+ readonly GH_EXPRESSION_FUNCTIONS='contains|startsWith|endsWith|format|join|toJSON|toJson|fromJSON|fromJson|hashFiles|success|always|cancelled|failure'
1104
+ EXPR_OK=1
1105
+ while IFS= read -r workflow; do
1106
+ # Only inside an expression. The same word in a `run:` block is shell or JavaScript.
1107
+ offenders="$(grep -oE '\$\{\{[^}]*\}\}' "$workflow" |
1108
+ grep -oE '[a-zA-Z_][a-zA-Z0-9_]*\(' |
1109
+ tr -d '(' |
1110
+ grep -vE "^(${GH_EXPRESSION_FUNCTIONS})$" |
1111
+ sort -u || true)"
1112
+ if [ -n "$offenders" ]; then
1113
+ EXPR_OK=0
1114
+ echo "FAIL: $(basename "$workflow") calls $(echo "$offenders" | tr '\n' ' ')which GitHub expressions do not have; the workflow will not load" >&2
1115
+ fi
1116
+ done < <(find "$WORKFLOWS_DIR" "${HERE}/../.." -maxdepth 3 -name '*.yml' -not -name '*.lock.yml' 2>/dev/null | sort -u)
1117
+ if [ "$EXPR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1118
+
1119
+ echo "── Error report privacy ──────────────────────────────────────────────────"
1120
+
1121
+ # Every repository that installs this package is private, and the error report is the only job
1122
+ # that sends anything out of one. Its whole safety argument is four properties, each of which
1123
+ # is one line that an edit could remove without any run going red, so all four are asserted.
1124
+ ERROR_REPORT_YML="${HERE}/../report-workflow-errors/action.yml"
1125
+ if [ -f "$ERROR_REPORT_YML" ]; then
1126
+ ER_OK=1
1127
+ er() {
1128
+ grep -qE "$1" "$ERROR_REPORT_YML" || { ER_OK=0; echo "FAIL: report-workflow-errors ${2}" >&2; }
1129
+ }
1130
+
1131
+ # 1. The scanner, and its teeth. A body that trips it must not be filed, and the run must go
1132
+ # red so the field that carried private text gets fixed instead of leaking again tomorrow.
1133
+ #
1134
+ # Match the declaration and count the sites, never the bare symbol. A grep for `leakChecks`
1135
+ # passed with the declaration renamed, because the name survives where scan() uses it, and a
1136
+ # grep for `core.setFailed` passed with one of the two calls turned into core.info. Both of
1137
+ # those were mutation-tested and both let a broken privacy guard through.
1138
+ er 'const leakChecks = \[' 'declares no leak scanner'
1139
+ teeth=$(count -cE 'core\.setFailed.*withheld by the leak scanner' "$ERROR_REPORT_YML")
1140
+ if [ "$teeth" -ge 2 ]; then
1141
+ PASS=$((PASS + 1))
1142
+ else
1143
+ ER_OK=0
1144
+ echo "FAIL: report-workflow-errors fails the run on a leak in only ${teeth} of its 2 exit paths" >&2
1145
+ fi
1146
+ # Every write upstream has to be behind a scan. Counting is enough here because both are few
1147
+ # and named, and a new write added without a guard moves the counts apart.
1148
+ scans=$(count -cE 'if \(!scan\(' "$ERROR_REPORT_YML")
1149
+ upstream_writes=$(count -cE 'upstream\.rest\.issues\.(create|update)\(' "$ERROR_REPORT_YML")
1150
+ if [ "$scans" -ge "$upstream_writes" ] && [ "$upstream_writes" -gt 0 ]; then
1151
+ PASS=$((PASS + 1))
1152
+ else
1153
+ ER_OK=0
1154
+ echo "FAIL: report-workflow-errors makes ${upstream_writes} upstream write(s) behind only ${scans} leak scan(s)" >&2
1155
+ fi
1156
+ for guard in 'the repository name' 'the owner name' 'a github.com URL' 'an email address' 'an absolute path'; do
1157
+ er "\{ what: '${guard}', test:" "no longer scans for ${guard}"
1158
+ done
1159
+
1160
+ # 2. The allowlist. A consumer's own workflow name can describe a product, a customer or an
1161
+ # environment; only the names this package gives its own files may be reported.
1162
+ er 'const OWNED = /\^\(work-router' 'has no workflow allowlist, so a repository-specific workflow name could be reported'
1163
+ er 'skippedForeign' 'does not account for the workflows it declined to inspect'
1164
+
1165
+ # 3. No raw log text. The catalogue matches the log tail and only the matched entry's id is
1166
+ # kept; a change that put the matched text in the report would be the leak.
1167
+ #
1168
+ # The finding object is that boundary, because bodyFor() renders a finding, so the check is
1169
+ # on its shape: the fields the object literals actually set must be exactly the declared
1170
+ # reportable list. An earlier version of this tried to spot log text in the body with a
1171
+ # regex over the whole file, and a mutation that added `${finding.logText.slice(0, 400)}`
1172
+ # walked straight past it -- the pattern was case-sensitive and the inserted label said
1173
+ # "Summary". Comparing two sets has no such gap.
1174
+ er 'patternId = CATALOGUE\.find' 'no longer classifies the log through the catalogue'
1175
+ declared=$(sed -n "s/^ *const FINDING_FIELDS = \[\(.*\)\];$/\1/p" "$ERROR_REPORT_YML" |
1176
+ tr -d " '" | tr ',' '\n' | sort -u | tr '\n' ' ')
1177
+ # Every key set in a finding object literal, plus every key assigned onto one afterwards.
1178
+ assigned=$( { sed -n '/const seen = findings\.get/,/^ };$/p' "$ERROR_REPORT_YML" |
1179
+ grep -oE '[a-zA-Z_][a-zA-Z0-9_]*:' | tr -d ':'
1180
+ grep -oE 'seen\.[a-zA-Z_][a-zA-Z0-9_]*' "$ERROR_REPORT_YML" | cut -d. -f2
1181
+ } | sort -u | tr '\n' ' ')
1182
+ if [ -n "${declared// /}" ] && [ "$declared" = "$assigned" ]; then
1183
+ PASS=$((PASS + 1))
1184
+ else
1185
+ ER_OK=0
1186
+ echo "FAIL: report-workflow-errors builds findings with fields that are not the declared reportable set" >&2
1187
+ echo " declared: ${declared:-(none)}" >&2
1188
+ echo " assigned: ${assigned:-(none)}" >&2
1189
+ fi
1190
+ # And the run-time half of the same boundary, so a field that arrives by a path the check
1191
+ # above cannot see stops the job instead of being rendered upstream.
1192
+ er 'not in the reportable field list' 'does not check the finding shape at run time'
1193
+
1194
+ # 4. No model. A model asked to summarise a failure paraphrases whatever the log held, which
1195
+ # is the one thing that must not cross the boundary. This job stays deterministic.
1196
+ if grep -qiE '(engine:|opencode|safe-outputs|OPENAI_API_KEY)' "$ERROR_REPORT_YML"; then
1197
+ ER_OK=0
1198
+ echo "FAIL: report-workflow-errors reaches for a model; the report must stay deterministic" >&2
1199
+ else
1200
+ PASS=$((PASS + 1))
1201
+ fi
1202
+
1203
+ # The workflow that drives it is an optional template: installed under .github/workflows in a
1204
+ # consumer, and still in templates/ upstream. Check whichever is present, so the assertions
1205
+ # run in the package's own CI rather than only after somebody installs it.
1206
+ ERROR_REPORT_WORKFLOW="${WORKFLOWS_DIR}/agentics-error-report.yml"
1207
+ [ -f "$ERROR_REPORT_WORKFLOW" ] ||
1208
+ ERROR_REPORT_WORKFLOW="${HERE}/../../templates/agentics/agentics-error-report.yml"
1209
+ if [ -f "$ERROR_REPORT_WORKFLOW" ]; then
1210
+ # It reads this repository and writes nothing to it. A write scope here would mean the job
1211
+ # that talks to another repository can also change this one.
1212
+ if grep -qE '^ (contents|actions): read$' "$ERROR_REPORT_WORKFLOW" &&
1213
+ ! grep -qE '^ [a-z-]+: write$' "$ERROR_REPORT_WORKFLOW"; then
1214
+ PASS=$((PASS + 1))
1215
+ else
1216
+ ER_OK=0
1217
+ echo "FAIL: agentics-error-report.yml grants a write scope; it must be read-only in the repository it reports on" >&2
1218
+ fi
1219
+ # The upstream token is scoped to the upstream repository alone, never the default token.
1220
+ if grep -q 'upstream-token: ${{ steps.upstream-token.outputs.token }}' "$ERROR_REPORT_WORKFLOW" &&
1221
+ grep -qE '^ repositories: \$\{\{ env\.UPSTREAM_NAME \}\}$' "$ERROR_REPORT_WORKFLOW"; then
1222
+ PASS=$((PASS + 1))
1223
+ else
1224
+ ER_OK=0
1225
+ echo "FAIL: agentics-error-report.yml does not scope its upstream token to the upstream repository" >&2
1226
+ fi
1227
+ fi
1228
+
1229
+ if [ "$ER_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1230
+ fi
1231
+
1232
+ # value, and a command-line flag given one; a line containing a dollar sign is taken to be
1233
+ # an expression or a shell variable and allowed. Paths resolve relative to this script, so
1234
+ # upstream this reads the templates and in a consumer it reads the real workflows.
1235
+ PASSWORD_SCAN_DIRS=("${WORKFLOWS_DIR}")
1236
+ [ -d "${HERE}/../../templates/ci" ] && PASSWORD_SCAN_DIRS+=("${HERE}/../../templates/ci")
1237
+ [ -d "${HERE}/../../templates/agentics" ] && PASSWORD_SCAN_DIRS+=("${HERE}/../../templates/agentics")
1238
+ password_hits=$(
1239
+ find "${PASSWORD_SCAN_DIRS[@]}" -type f \( -name '*.yml' -o -name '*.yaml' \) \
1240
+ -not -name '*.lock.yml' -print0 |
1241
+ xargs -0 -r grep -nEi \
1242
+ -e "(password|passwd|pwd)[\"']?[[:space:]]*[:=][[:space:]]*[\"'][^\"']" \
1243
+ -e "(^|[[:space:]])(-P|--password)[[:space:]=]*[\"'][^\"']" |
1244
+ grep -v '[$]' || true
1245
+ )
1246
+ if [ -z "$password_hits" ]; then
1247
+ PASS=$((PASS + 1))
1248
+ else
1249
+ FAIL=$((FAIL + 1))
1250
+ echo "FAIL: a password is written down in a shipped workflow; use a secret or derive it per run" >&2
1251
+ printf '%s\n' "$password_hits" >&2
1252
+ fi
1253
+
293
1254
  echo
294
1255
  if [ "$FAIL" -eq 0 ]; then
295
1256
  echo "Route matrix: ${PASS} passed"