@plainconceptsplatform/workflows 0.5.1 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/dist/action-validation.test.d.ts +1 -0
  2. package/dist/action-validation.test.js +87 -0
  3. package/dist/catalog-installation.d.ts +30 -0
  4. package/dist/catalog-installation.js +422 -0
  5. package/dist/catalog-installation.test.d.ts +1 -0
  6. package/dist/catalog-installation.test.js +485 -0
  7. package/dist/catalog-listing.d.ts +13 -0
  8. package/dist/catalog-listing.js +70 -0
  9. package/dist/catalog-listing.test.d.ts +1 -0
  10. package/dist/catalog-listing.test.js +150 -0
  11. package/dist/index.d.ts +2 -0
  12. package/dist/index.js +218 -0
  13. package/dist/index.test.d.ts +1 -0
  14. package/dist/index.test.js +273 -0
  15. package/dist/repository-inspection.d.ts +23 -0
  16. package/dist/repository-inspection.js +113 -0
  17. package/dist/repository-inspection.test.d.ts +1 -0
  18. package/dist/repository-inspection.test.js +77 -0
  19. package/dist/route-processing.d.ts +6 -0
  20. package/dist/route-processing.js +149 -0
  21. package/dist/route-processing.test.d.ts +1 -0
  22. package/dist/route-processing.test.js +283 -0
  23. package/dist/stack-defaults.d.ts +11 -0
  24. package/dist/stack-defaults.js +104 -0
  25. package/dist/stack-defaults.test.d.ts +1 -0
  26. package/dist/stack-defaults.test.js +266 -0
  27. package/dist/tui.d.ts +25 -0
  28. package/dist/tui.js +287 -0
  29. package/dist/tui.test.d.ts +1 -0
  30. package/dist/tui.test.js +249 -0
  31. package/dist/workflow-catalog.d.ts +25 -0
  32. package/dist/workflow-catalog.js +55 -0
  33. package/dist/workflow-catalog.test.d.ts +1 -0
  34. package/dist/workflow-catalog.test.js +29 -0
  35. package/loops/actions/add-issue-labels/action.yml +2 -2
  36. package/loops/actions/apply-agent-bundle/apply-bundle.sh +14 -1
  37. package/loops/actions/classify-route/action.yml +7 -0
  38. package/loops/actions/classify-route/classify-route.sh +20 -5
  39. package/loops/actions/cleanup-artifacts/action.yml +38 -12
  40. package/loops/actions/identify-gate-subject/action.yml +3 -1
  41. package/loops/actions/remove-issue-labels/action.yml +4 -2
  42. package/loops/actions/validate-refine-output/validate-refine-output.sh +18 -2
  43. package/loops/actions/verify-refine-output/verify-refine-output.sh +20 -0
  44. package/loops/actions/verify-route-matrix/verify-route-matrix.sh +300 -16
  45. package/loops/scripts/compile-agent-workflows.mjs +96 -1
  46. package/loops/scripts/merge-changelog.mjs +76 -0
  47. package/loops/templates/agentics/actionlint.yaml +13 -0
  48. package/loops/templates/agentics/agentics-checks.yml +203 -86
  49. package/loops/templates/ci/app-ci-dotnet-next.yml +64 -2
  50. package/loops/templates/ci/app-ci-node-monorepo.yml +52 -0
  51. package/loops/templates/opencode/opencode.ci.json +2 -2
  52. package/loops/templates/opencode/opencode.ci.json.md +4 -1
  53. package/loops/workflows/agent-apply-review.md +5 -1
  54. package/loops/workflows/agent-implement.md +105 -3
  55. package/loops/workflows/agent-merge-gate.md +139 -25
  56. package/loops/workflows/agent-refine.md +26 -2
  57. package/loops/workflows/work-router.yml +227 -51
  58. package/package.json +1 -1
@@ -56,6 +56,26 @@ assert_output 'wrong issue output is invalid' invalid \
56
56
  assert_output 'complete output cannot update another issue' invalid \
57
57
  '{"items":[{"type":"update_issue","item_number":42,"body":"# User story"},{"type":"update_issue","item_number":7,"body":"# Other story"}]}'
58
58
 
59
+ # From a real run. The agent's first update_issue went out malformed, the bridge counted it
60
+ # as spent, the retry carrying the body was refused, and the "Refinement complete" comment
61
+ # went out regardless. Invalid is the only safe reading: the body was never replaced. Note
62
+ # what it must not be read as — a question. Nobody asked one, so "questions" would leave the
63
+ # issue waiting on an answer that is never coming.
64
+ assert_output 'a completion claim with no body is invalid' invalid \
65
+ '{"items":[{"type":"add_comment","item_number":42,"body":"<!-- agent-refine -->\nRefinement update\nRefinement complete. The implement label has been added and the implement workflow will start shortly."},{"type":"report_incomplete","reason":"update_issue limit reached"}]}'
66
+
67
+ # The other side of it: a run that did replace the body and also reported a difficulty has
68
+ # done the work, and its work is not thrown away for having said so.
69
+ assert_output 'a refined body survives a reported difficulty' complete \
70
+ '{"items":[{"type":"update_issue","item_number":42,"body":"# User story"},{"type":"report_incomplete","reason":"a tool was slow"}]}'
71
+ assert_output 'a split survives a note about a missing tool' split \
72
+ '{"items":[{"type":"update_issue","item_number":42,"body":"# Epic"},{"type":"create_issue","title":"One","body":"First"},{"type":"create_issue","title":"Two","body":"Second"},{"type":"missing_tool","reason":"no browser"}]}'
73
+
74
+ # And a run that left nothing but a signal has done nothing.
75
+ assert_output 'only a run signal is invalid' invalid \
76
+ '{"items":[{"type":"report_incomplete","reason":"gave up"}]}'
77
+
78
+
59
79
  echo
60
80
  if [ "$FAIL" -eq 0 ]; then
61
81
  echo "Refine output validation: ${PASS} passed"
@@ -2,12 +2,14 @@
2
2
  # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-route-matrix/verify-route-matrix.sh. Update with `workflows update --force`; consumer edits may be overwritten.
3
3
  # Exercise the router's real classifier. This sources classify-route.sh rather than
4
4
  # restating it, so a change to the route table cannot pass here by being copied twice.
5
+ #
6
+ # This file greps workflow sources for literal `${{ ... }}` expressions on purpose.
7
+ # shellcheck disable=SC2016
5
8
 
6
9
  set -euo pipefail
7
10
 
8
11
  HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
9
12
  ROUTER_YML="${HERE}/../../workflows/work-router.yml"
10
- AUTHORIZE_YML="${HERE}/../../workflows/authorize-bot-work.yml"
11
13
  IMPLEMENT_WORKER_MD="${HERE}/../../workflows/agent-implement.md"
12
14
  MERGE_GATE_WORKER_MD="${HERE}/../../workflows/agent-merge-gate.md"
13
15
 
@@ -90,6 +92,8 @@ assert "refine label starts a first pass" first \
90
92
  echo "── Comment events ────────────────────────────────────────────────────────"
91
93
  assert_route "a comment on a pull request routes to apply-review" apply-review \
92
94
  EVENT=issue_comment COMMENT_ON_PR=true EVENT_ISSUE_NUMBER=7
95
+ assert_route "the bot's own comment on a pull request never re-enters apply-review" none \
96
+ EVENT=issue_comment COMMENT_ON_PR=true COMMENT_SENDER_TYPE=Bot EVENT_ISSUE_NUMBER=7
93
97
  assert_route "an author reply on a refine issue re-refines" refine \
94
98
  EVENT=issue_comment COMMENT_ON_PR=false COMMENT_SENDER_TYPE=User \
95
99
  'ISSUE_LABELS=["refine","review"]' EVENT_ISSUE_NUMBER=42
@@ -183,6 +187,18 @@ assert_route "merge-gate dispatch needs a pull request number" none \
183
187
  EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=0
184
188
  assert_route "merge-gate dispatch accepts a positive pull request" merge-gate \
185
189
  EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=7
190
+ assert "merge-gate dispatch defaults its attempt count to zero" 0 \
191
+ "$(route_field merge-gate-attempts EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=7)"
192
+ assert "merge-gate dispatch forwards the attempt count" 3 \
193
+ "$(route_field merge-gate-attempts EVENT=workflow_dispatch OPERATION=merge-gate INPUT_PR_NUMBER=7 INPUT_ATTEMPTS_SO_FAR=3)"
194
+ # The implement worker re-dispatches itself when a run dies before producing an answer, so the
195
+ # count has to survive the round trip or the budget never advances and the retry never stops.
196
+ assert "implement dispatch defaults its attempt count to zero" 0 \
197
+ "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=implement INPUT_ISSUE_NUMBER=42)"
198
+ assert "implement dispatch forwards the attempt count" 2 \
199
+ "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=implement INPUT_ISSUE_NUMBER=42 INPUT_ATTEMPTS_SO_FAR=2)"
200
+ assert "a refine dispatch carries no implement attempts" 0 \
201
+ "$(route_field implement-attempts EVENT=workflow_dispatch OPERATION=refine INPUT_ISSUE_NUMBER=42 INPUT_ATTEMPTS_SO_FAR=2)"
186
202
  assert_route "reconcile-bot-pr-runs dispatch needs no numbers" reconcile-bot-pr-runs \
187
203
  EVENT=workflow_dispatch OPERATION=reconcile-bot-pr-runs
188
204
  assert_route "an unknown operation routes nowhere" none \
@@ -205,8 +221,7 @@ if [ -z "$empty_expr" ]; then
205
221
  else
206
222
  FAIL=$((FAIL + 1))
207
223
  echo "FAIL: workflow files contain an empty Actions expression:" >&2
208
- printf ' %s
209
- ' $empty_expr >&2
224
+ while IFS= read -r offending; do echo " $offending" >&2; done <<<"$empty_expr"
210
225
  fi
211
226
 
212
227
 
@@ -220,21 +235,21 @@ else
220
235
  grep -nE 'needs\.[a-z_]+\.outputs\.[a-zA-Z0-9_]*-' "$IMPLEMENT_WORKER_MD" >&2
221
236
  fi
222
237
 
223
- # A worker that prints a ${VERIFY_COMMANDS_*} block without setting it renders an empty command
224
- # block, and the model invents its own build line. That is how a child shipped `dotnet build
225
- # --no-restore` against an unrestored workspace. The commands are split per area so a change
226
- # that only touches apps/web does not pay for a cold Release build of the API.
238
+ # A worker that prints `${{ env.NAME }}` without defining NAME in its own env: block renders
239
+ # an empty value, and the model fills the gap itself. That is how a child shipped `dotnet build
240
+ # --no-restore` against an unrestored workspace: the verification block was empty. Every name a
241
+ # worker prints must be defined in that worker. The values are consumer-owned (a consumer may
242
+ # split VERIFY_COMMANDS per area, or keep one); only the wiring is asserted here.
227
243
  VERIFY_OK=1
228
- for VAR in VERIFY_COMMANDS_API VERIFY_COMMANDS_WEB; do
229
- if grep -q "env\.$VAR" "$IMPLEMENT_WORKER_MD" && ! grep -q "^ $VAR:" "$IMPLEMENT_WORKER_MD"; then
230
- VERIFY_OK=0
231
- echo "FAIL: implement worker prints $VAR without defining it" >&2
232
- fi
244
+ for worker in "${HERE}/../../workflows"/agent-*.md; do
245
+ while read -r name; do
246
+ [ -n "$name" ] || continue
247
+ if ! grep -q "^ ${name}:" "$worker"; then
248
+ VERIFY_OK=0
249
+ echo "FAIL: $(basename "$worker") prints env.${name} without defining it" >&2
250
+ fi
251
+ done < <(grep -oE '\$\{\{ *env\.[A-Za-z_][A-Za-z0-9_]* *\}\}' "$worker" | sed -E 's/.*env\.([A-Za-z_][A-Za-z0-9_]*).*/\1/' | sort -u)
233
252
  done
234
- if grep -qE 'env\.VERIFY_COMMANDS[^_]' "$IMPLEMENT_WORKER_MD"; then
235
- VERIFY_OK=0
236
- echo "FAIL: implement worker still references the unscoped VERIFY_COMMANDS" >&2
237
- fi
238
253
  if [ "$VERIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
239
254
 
240
255
  if grep -Fq 'protected-files: allowed' "$IMPLEMENT_WORKER_MD" &&
@@ -246,6 +261,233 @@ else
246
261
  echo "FAIL: protected changes must allow failed-CI repair while remaining held from merge" >&2
247
262
  fi
248
263
 
264
+ # gh-aw folds the worker's top-level `if:` into the generated activation job but computes
265
+ # activation's `needs` on its own: only custom jobs the prompt references AND that declare no
266
+ # `needs:` are hoisted. A guard with its own `needs:` (protected_changes needs subject) is read
267
+ # before it has run, resolves to '' and gates nothing, unless it is listed in `on.needs`, the
268
+ # documented way to add jobs to pre_activation and activation. Inline list form is expected.
269
+ TOP_IF="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n 's/^if: //p')"
270
+ ON_NEEDS="$(tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n '/^on:$/,/^[a-z]/p' |
271
+ sed -n 's/^ needs: *\[\(.*\)\].*/\1/p' | tr -d ' ' | tr ',' '\n')"
272
+ ACTIVATION_OK=1
273
+ [ -n "$TOP_IF" ] || { ACTIVATION_OK=0; echo "FAIL: could not read the merge-gate worker's top-level if" >&2; }
274
+ while read -r job; do
275
+ [ -n "$job" ] || continue
276
+ if tr -d '\r' <"$MERGE_GATE_WORKER_MD" | sed -n "/^ ${job}:$/,/^ [a-z_]*:$/p" | grep -q '^ needs:' &&
277
+ ! grep -qx "$job" <<<"$ON_NEEDS"; then
278
+ ACTIVATION_OK=0
279
+ echo "FAIL: merge-gate top-level if reads needs.${job}, which has its own needs and is not in on.needs; activation would read it before it runs" >&2
280
+ fi
281
+ done < <(grep -oE 'needs\.[a-z_]+\.' <<<"$TOP_IF" | sed 's/^needs\.//; s/\.$//' | sort -u)
282
+ if [ "$ACTIVATION_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
283
+
284
+ # The merge belt is serial for the whole repository: several overnight pull requests
285
+ # mean every merge moves the default branch under the rest, and gates running at once
286
+ # rebase onto bases other gates are about to invalidate. A per-issue group here would
287
+ # reintroduce that race, so assert the repo-wide lock is the one in use.
288
+ if grep -A7 'call-merge-gate:' "$ROUTER_YML" | grep -q 'group: merge-belt'; then
289
+ PASS=$((PASS + 1))
290
+ else
291
+ FAIL=$((FAIL + 1))
292
+ echo "FAIL: call-merge-gate must hold the repo-wide merge-belt lock" >&2
293
+ fi
294
+
295
+ # A verdict is the gate marker AND a `**Verdict:**` line together. Comments carrying the
296
+ # marker alone were progress notes and failed attempts, and the reconcile belt read every
297
+ # one of them as final: a crashed or OOM-killed gate parked its pull request for the rest
298
+ # of the night. Attempts are counted separately, capped, and reset by any new CI run.
299
+ BELT_OK=1
300
+ if ! grep -q 'agent-merge-gate-attempt' "$ROUTER_YML"; then
301
+ BELT_OK=0; echo "FAIL: router never counts gate attempts" >&2
302
+ fi
303
+ if [ "$(grep -cF 'contains("<!-- agent-merge-gate -->")) and (.body | contains("**Verdict:**"))' "$ROUTER_YML")" -lt 4 ]; then
304
+ BELT_OK=0; echo "FAIL: verdict detection must pair the gate marker with a Verdict line in both dispatch paths" >&2
305
+ fi
306
+ if [ "$(grep -c 'attempts_so_far' "$ROUTER_YML")" -lt 2 ]; then
307
+ BELT_OK=0; echo "FAIL: dispatch sites must forward attempts_so_far" >&2
308
+ fi
309
+ # A second gate for a pull request whose gate is already queued or running reads the same CI
310
+ # verdict and is cancelled by the single-slot merge-belt queue (two cancellations on 2026-09-06).
311
+ if [ "$(grep -c 'a merge-gate run is already live' "$ROUTER_YML")" -lt 2 ]; then
312
+ BELT_OK=0; echo "FAIL: both dispatch paths must skip a pull request whose gate is already live" >&2
313
+ fi
314
+ # A conflicting pull request has no refs/pull/N/merge for GitHub to build, so a `pull_request`
315
+ # CI workflow can never run on that head. Requiring a fresh verdict before dispatching deadlocks
316
+ # the belt: only the gate resolves the conflict, and the gate never runs. Both paths fall back to
317
+ # the branch's last verdict when, and only when, the pull request is conflicting.
318
+ if [ "$(grep -c 'conflicts, so CI cannot run on' "$ROUTER_YML")" -lt 2 ]; then
319
+ BELT_OK=0
320
+ echo "FAIL: both dispatch paths must gate a conflicting pull request that can never get fresh CI" >&2
321
+ fi
322
+ # That fallback has to read the computed mergeable state. The REST boolean is null until GitHub
323
+ # recomputes it, and stays null for a pull request nobody has opened recently, which is exactly
324
+ # the stale conflicting pull request the fallback exists for: it never fired once in production.
325
+ # The state has to be polled, not read once. GitHub computes mergeability on demand and the
326
+ # first read answers UNKNOWN (or null through REST) while it works it out, so a single read
327
+ # reports "not conflicting" for exactly the stale pull requests the fallback is for. Observed
328
+ # twice in production: the fallback logged "no completed CI run" for a pull request that
329
+ # `gh pr view` reported as CONFLICTING from a warm cache seconds later.
330
+ if [ "$(grep -c 'mergeable_state()' "$ROUTER_YML")" -lt 2 ] ||
331
+ [ "$(grep -c 'mergeable_now=$(mergeable_state' "$ROUTER_YML")" -lt 2 ]; then
332
+ BELT_OK=0
333
+ echo "FAIL: both dispatch paths must poll the mergeable state; a single read answers UNKNOWN" >&2
334
+ fi
335
+ if [ "$BELT_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
336
+
337
+ # GitHub delivers workflow_run only for CI runs whose actor is a human, so a bot pull request's
338
+ # CI never reaches the router's CI-completion route. The package ships a dispatch-merge-gate job
339
+ # in templates/ci that hands the verdict over from inside CI; a consumer CI workflow, where one
340
+ # exists beside the router, must carry it or bot pull requests wait for the hourly belt.
341
+ for ci in "${HERE}/../../workflows/ci.yml" "${HERE}/../../workflows/app-ci.yml"; do
342
+ [ -f "$ci" ] || continue
343
+ if grep -q 'operation=merge-gate' "$ci"; then
344
+ PASS=$((PASS + 1))
345
+ else
346
+ FAIL=$((FAIL + 1))
347
+ echo "FAIL: $(basename "$ci") has no dispatch-merge-gate job; bot pull requests would wait for the hourly belt" >&2
348
+ fi
349
+ done
350
+
351
+ # The router forwards a fact to a worker by reading `needs.classify.outputs.<x>`; a name the
352
+ # classify job does not export resolves to '' with no error. That is how the gate received
353
+ # attempts_so_far='' (the classifier emitted merge-gate-attempts, the job never exported it),
354
+ # fromJson('') killed the incomplete job before its attempt comment, and the belt re-dispatched
355
+ # the same crash every hour. Every name the router reads must be exported by the classify job.
356
+ CLASSIFY_EXPORTS="$(tr -d '\r' <"$ROUTER_YML" |
357
+ sed -n '/^ classify:$/,/^ [a-z-]*:$/p' |
358
+ sed -n '/^ outputs:$/,/^ [a-z]*:$/p' |
359
+ sed -n 's/^ \([a-zA-Z0-9_-]*\):.*/\1/p')"
360
+ CLASSIFY_OK=1
361
+ [ -n "$CLASSIFY_EXPORTS" ] || { CLASSIFY_OK=0; echo "FAIL: could not read the classify job's outputs from work-router.yml" >&2; }
362
+ while read -r name; do
363
+ [ -n "$name" ] || continue
364
+ if ! grep -qx "$name" <<<"$CLASSIFY_EXPORTS"; then
365
+ CLASSIFY_OK=0
366
+ echo "FAIL: work-router.yml reads needs.classify.outputs.${name} but the classify job does not export it" >&2
367
+ fi
368
+ done < <(grep -oE 'needs\.classify\.outputs\.[a-zA-Z0-9_-]+' "$ROUTER_YML" | sed 's/.*\.//' | sort -u)
369
+ if [ "$CLASSIFY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
370
+
371
+ # fromJson('') is a hard failure ("Error reading JToken"), and a workflow_call input arrives as
372
+ # '' whenever the caller passes an empty expression, declared default or not. The gate must never
373
+ # hand a raw input to fromJson; `inputs.x || '0'` reads the empty case as zero.
374
+ if ! grep -qE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD"; then
375
+ PASS=$((PASS + 1))
376
+ else
377
+ FAIL=$((FAIL + 1))
378
+ echo "FAIL: merge-gate worker calls fromJson on a raw input; an empty caller value kills the job" >&2
379
+ grep -nE "fromJson\(inputs\.[a-zA-Z0-9_]+\)" "$MERGE_GATE_WORKER_MD" >&2
380
+ fi
381
+
382
+ # The worker's own comments must keep the distinction: progress notes carry no marker,
383
+ # failed attempts carry the attempt marker, verdicts carry the marker AND the Verdict line.
384
+ # Three verdict sites: the review hold on the issue, the agent's assessment on the issue,
385
+ # and conclude's short verdict on the pull request itself.
386
+ if grep -q 'ATTEMPT_MARKER: "<!-- agent-merge-gate-attempt -->"' "$MERGE_GATE_WORKER_MD" &&
387
+ [ "$(grep -c '\${{ env.GATE_MARKER }}' "$MERGE_GATE_WORKER_MD")" -eq 3 ]; then
388
+ PASS=$((PASS + 1))
389
+ else
390
+ FAIL=$((FAIL + 1))
391
+ echo "FAIL: merge-gate worker must keep verdict and attempt markers distinct" >&2
392
+ fi
393
+
394
+ # add-issue-labels and remove-issue-labels split `labels` on newlines. A caller that joined two
395
+ # names with a comma removed one label called "bot-working,pr-pending": a 404 the action swallows
396
+ # on purpose, so the release never happened and Pliny-Bot #49/#54 carried implement, pr-pending
397
+ # and review together for a day. Callers use block scalars, one label per line; the actions also
398
+ # accept commas so a consumer copy of an old caller keeps working.
399
+ LABELS_OK=1
400
+ if grep -nE '^[[:space:]]+labels: [^|>].*,' "${HERE}/../../workflows"/agent-*.md >&2; then
401
+ LABELS_OK=0
402
+ echo "FAIL: a worker passes comma-joined labels to a label action; use a block scalar, one label per line" >&2
403
+ fi
404
+ for action in add-issue-labels remove-issue-labels; do
405
+ if ! grep -qF 'split(/\r?\n|,/)' "${HERE}/../${action}/action.yml"; then
406
+ LABELS_OK=0
407
+ echo "FAIL: ${action} must accept comma-separated labels as well as one per line" >&2
408
+ fi
409
+ done
410
+ if [ "$LABELS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
411
+
412
+ # The agent's fix reaches the branch as a bundle applied fast-forward only (apply-agent-output).
413
+ # gh-aw's push tool description tells the model to rebase, and a rebased branch cannot
414
+ # fast-forward: the push is refused and the verdict is lost (Pliny-Bot run 33952565835). The
415
+ # worker must start on the pull request branch and must never say `git rebase`. Its progress
416
+ # comment is posted on the first attempt only; retries are recorded by the attempt comment.
417
+ BRANCH_OK=1
418
+ # Path B: staged safe outputs, applied by conclude with the App token. Without `staged: true`
419
+ # gh-aw's safe_outputs job writes too, and it runs first: it pushed a flattened single-parent
420
+ # commit with GITHUB_TOKEN, which lost the agent's merge, left the pull request conflicting,
421
+ # and started no CI, because GITHUB_TOKEN writes raise no events.
422
+ if ! grep -qE '^ staged: true' "$MERGE_GATE_WORKER_MD"; then
423
+ BRANCH_OK=0; echo "FAIL: merge-gate safe-outputs must be staged; conclude owns the write path" >&2
424
+ fi
425
+ if grep -q 'git rebase' "$MERGE_GATE_WORKER_MD"; then
426
+ BRANCH_OK=0; echo "FAIL: merge-gate worker tells the agent to rebase; the push is fast-forward only" >&2
427
+ fi
428
+ if ! grep -q 'name: Check out the pull request branch' "$MERGE_GATE_WORKER_MD"; then
429
+ BRANCH_OK=0; echo "FAIL: merge-gate worker must check out the pull request branch before the agent starts" >&2
430
+ fi
431
+ if ! grep -qF "conclusion == 'failure' && (inputs.attempts_so_far || '0') == '0'" "$MERGE_GATE_WORKER_MD"; then
432
+ BRANCH_OK=0; echo "FAIL: the reserve job's progress comment must be posted on the first attempt only" >&2
433
+ fi
434
+ # A conflicting pull request has no CI run to read logs from, so the gate is handed empty
435
+ # failure artifacts. Read on its own that looks like "no evidence", and the agent asked for a
436
+ # human instead of resolving the conflict that caused it.
437
+ if ! grep -qF 'Empty failure evidence is not a reason to ask for review' "$MERGE_GATE_WORKER_MD"; then
438
+ BRANCH_OK=0
439
+ echo "FAIL: the gate must treat empty failure evidence on a conflicting PR as the conflict to fix" >&2
440
+ fi
441
+ if [ "$BRANCH_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
442
+
443
+ # pr-pending means a pull request for this issue is open and waiting. Only merging retires it.
444
+ # Every other path (the protected-files hold, a review verdict, a failed attempt) leaves the
445
+ # pull request open, and stripping the label there produced a board where issues with open
446
+ # pull requests looked like they had none. It went unnoticed while the label actions silently
447
+ # removed nothing, so the two bugs hid each other.
448
+ PENDING_OK=1
449
+ grep -q '^ PR_PENDING_LABEL:' "$MERGE_GATE_WORKER_MD" ||
450
+ { PENDING_OK=0; echo "FAIL: merge gate lost its PR_PENDING_LABEL definition" >&2; }
451
+ if [ "$(grep -c '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD")" -ne 1 ]; then
452
+ PENDING_OK=0
453
+ echo "FAIL: pr-pending must be removed in exactly one place, the merge path" >&2
454
+ grep -n '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" >&2
455
+ fi
456
+ # And that one place has to be the merge outcome, not a hold or a failed attempt.
457
+ grep -B12 '\${{ env.PR_PENDING_LABEL }}' "$MERGE_GATE_WORKER_MD" | grep -q "outcome == 'merge'" ||
458
+ { PENDING_OK=0; echo "FAIL: the only pr-pending removal must sit under the merge outcome" >&2; }
459
+ if [ "$PENDING_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
460
+
461
+ # A provider outage kills a run in a couple of minutes with no answer, and the same issue used
462
+ # to be handed to a human for it. The implement worker retries those and only those: a run that
463
+ # worked for half an hour and then failed produced an answer that was wrong, and repeating it
464
+ # costs the fleet the same half hour to be wrong again.
465
+ IMPLEMENT_RETRY_OK=1
466
+ for needle in 'RETRY_UNDER_MINUTES' 'ATTEMPT_MARKER' 'attempts_so_far' 'operation=implement'; do
467
+ grep -qF "$needle" "$IMPLEMENT_WORKER_MD" || {
468
+ IMPLEMENT_RETRY_OK=0
469
+ echo "FAIL: implement worker lost its retry belt: no '$needle'" >&2
470
+ }
471
+ done
472
+ # Park and retry are mutually exclusive: the retry path must never add the review label, and
473
+ # the park path must never re-dispatch.
474
+ grep -A3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" ||
475
+ grep -B3 'Flag for human review' "$IMPLEMENT_WORKER_MD" | grep -q "retry != 'true'" || {
476
+ IMPLEMENT_RETRY_OK=0
477
+ echo "FAIL: the implement worker must not flag review on a run it is about to retry" >&2
478
+ }
479
+ if [ "$IMPLEMENT_RETRY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
480
+
481
+ # A failed attempt must not strip `implement`: identify-gate-subject refuses an issue
482
+ # without it, so the first crash would starve every retry at the subject check.
483
+ if grep -A6 'Park the issue' "$MERGE_GATE_WORKER_MD" | grep -q 'REVIEW_LABEL' &&
484
+ ! grep -qF 'labels: ${{ env.WORKING_LABEL }},${{ env.IMPLEMENT_LABEL }}' "$MERGE_GATE_WORKER_MD"; then
485
+ PASS=$((PASS + 1))
486
+ else
487
+ FAIL=$((FAIL + 1))
488
+ echo "FAIL: the incomplete job must keep implement and only park on an exhausted budget" >&2
489
+ fi
490
+
249
491
  # This repository is public. Every route a human can start from a comment, a review or a
250
492
  # label must pass the authorize gate, or anyone able to comment can start a model run that
251
493
  # writes code. Asserted here because removing the gate would otherwise be a silent, one-line
@@ -290,6 +532,48 @@ while read -r operation; do
290
532
  done < <(sed -n '/^ operation:/,/^ issue-number:/p' "$ROUTER_YML" |
291
533
  sed -n 's/^ - //p')
292
534
 
535
+ # A router job that reads pull requests has to say so. audit-close listed them with only
536
+ # contents and issues and failed nightly on a 403 that named the endpoint and nothing else.
537
+ # Paired job-to-scope rather than parsed out of each action: the jobs that touch pull
538
+ # requests are few and known, and naming them here is what makes the omission visible.
539
+ for pr_job in audit-close reconcile-bot-pr-runs detect-pr-conflicts; do
540
+ if ! grep -q "^ ${pr_job}:$" "$ROUTER_YML"; then
541
+ continue
542
+ fi
543
+ pr_scopes=$(sed -n "/^ ${pr_job}:$/,/^ steps:$/p" "$ROUTER_YML")
544
+ if printf '%s' "$pr_scopes" | grep -qE '^ pull-requests: (read|write)$'; then
545
+ PASS=$((PASS + 1))
546
+ else
547
+ FAIL=$((FAIL + 1))
548
+ echo "FAIL: router job '${pr_job}' reads pull requests but grants no pull-requests scope" >&2
549
+ fi
550
+ done
551
+
552
+ # No written-down passwords in anything this repository ships. A throwaway credential for
553
+ # a test container is still a policy finding, and one sat in every consumer's CI for weeks
554
+ # until a scan found it rather than us. Two shapes: a password-ish name assigned a quoted
555
+ # value, and a command-line flag given one; a line containing a dollar sign is taken to be
556
+ # an expression or a shell variable and allowed. Paths resolve relative to this script, so
557
+ # upstream this reads the templates and in a consumer it reads the real workflows.
558
+ PASSWORD_SCAN_DIRS=("${HERE}/../../workflows")
559
+ [ -d "${HERE}/../../templates/ci" ] && PASSWORD_SCAN_DIRS+=("${HERE}/../../templates/ci")
560
+ [ -d "${HERE}/../../templates/agentics" ] && PASSWORD_SCAN_DIRS+=("${HERE}/../../templates/agentics")
561
+ password_hits=$(
562
+ find "${PASSWORD_SCAN_DIRS[@]}" -type f \( -name '*.yml' -o -name '*.yaml' \) \
563
+ -not -name '*.lock.yml' -print0 |
564
+ xargs -0 -r grep -nEi \
565
+ -e "(password|passwd|pwd)[\"']?[[:space:]]*[:=][[:space:]]*[\"'][^\"']" \
566
+ -e "(^|[[:space:]])(-P|--password)[[:space:]=]*[\"'][^\"']" |
567
+ grep -v '[$]' || true
568
+ )
569
+ if [ -z "$password_hits" ]; then
570
+ PASS=$((PASS + 1))
571
+ else
572
+ FAIL=$((FAIL + 1))
573
+ echo "FAIL: a password is written down in a shipped workflow; use a secret or derive it per run" >&2
574
+ printf '%s\n' "$password_hits" >&2
575
+ fi
576
+
293
577
  echo
294
578
  if [ "$FAIL" -eq 0 ]; then
295
579
  echo "Route matrix: ${PASS} passed"
@@ -19,6 +19,50 @@ function resolveGhPath() {
19
19
  return "gh";
20
20
  }
21
21
 
22
+ // gh-aw folds a workflow's top-level `if:` into the activation job but computes activation.needs
23
+ // on its own, so a compiled `if:` can read `needs.<job>` for a job activation never waits for.
24
+ // GitHub does not reject that: the reference resolves to '' at runtime and the clause is silently
25
+ // true (agent-merge-gate's protected_changes guard, Pliny-Bot run 34042143350). actionlint would
26
+ // flag it, but it is deliberately not run on generated files (it does not model concurrency.queue
27
+ // or job.workflow_*). This is the one check that reads the locks, and it needs no dependency:
28
+ // gh-aw emits jobs at two spaces, job keys at four, list items at six.
29
+ function undeclaredNeeds(lock) {
30
+ const violations = [];
31
+ let inJobs = false;
32
+ let job = null;
33
+ let inList = false;
34
+ let jobs = 0;
35
+ let needs = new Set();
36
+ let refs = new Set();
37
+ const flush = () => {
38
+ if (!job) return;
39
+ for (const ref of refs) if (!needs.has(ref)) violations.push({ job, ref, needs: [...needs].sort() });
40
+ };
41
+ for (const line of lock.split("\n")) {
42
+ if (/^jobs:\s*$/.test(line)) { inJobs = true; continue; }
43
+ if (inJobs && /^[^\s#]/.test(line)) inJobs = false;
44
+ if (!inJobs) continue;
45
+ const start = line.match(/^ ([a-z_][a-zA-Z0-9_-]*):\s*$/);
46
+ if (start) { flush(); job = start[1]; jobs += 1; needs = new Set(); refs = new Set(); inList = false; continue; }
47
+ if (!job) continue;
48
+ if (/^ needs:\s*$/.test(line)) { inList = true; continue; }
49
+ const inline = line.match(/^ needs:\s*(.+?)\s*$/);
50
+ if (inline) {
51
+ for (const n of inline[1].replace(/[\[\]]/g, "").split(",")) if (n.trim()) needs.add(n.trim());
52
+ inList = false;
53
+ continue;
54
+ }
55
+ if (inList) {
56
+ const item = line.match(/^ - (\S+)\s*$/);
57
+ if (item) { needs.add(item[1]); continue; }
58
+ inList = false;
59
+ }
60
+ for (const m of line.matchAll(/needs\.([a-z_][a-zA-Z0-9_-]*)\./g)) refs.add(m[1]);
61
+ }
62
+ flush();
63
+ return { jobs, violations };
64
+ }
65
+
22
66
  const compile = spawnSync(resolveGhPath(), ["aw", "compile", "--strict", "--dir", workflowDirectory], {
23
67
  stdio: "inherit",
24
68
  shell: false,
@@ -31,12 +75,43 @@ if (compile.error?.code === "ENOENT" || compile.status === null) {
31
75
 
32
76
  if (compile.status !== 0) process.exit(compile.status ?? 1);
33
77
 
78
+ function lintLock(file, text) {
79
+ const lint = undeclaredNeeds(text);
80
+ if (lint.jobs === 0) {
81
+ process.stderr.write(
82
+ `${file}: the needs lint found no jobs; the gh-aw lock layout changed. ` +
83
+ `Re-anchor undeclaredNeeds() in loops/scripts/compile-agent-workflows.mjs.\n`,
84
+ );
85
+ process.exitCode = 1;
86
+ }
87
+ for (const v of lint.violations) {
88
+ process.stderr.write(
89
+ `${file}: job "${v.job}" reads needs.${v.ref} but its needs are [${v.needs.join(", ")}]; ` +
90
+ `the reference resolves to '' at runtime. For activation, list the job under on.needs in the ` +
91
+ `source frontmatter; for a custom job, add it to that job's needs.\n`,
92
+ );
93
+ process.exitCode = 1;
94
+ }
95
+ }
96
+
97
+ // gh aw compile leaves a lock alone when its frontmatter and body hashes still match the source.
98
+ // This script then ran over a lock it had already patched: AGENTMEMORY_URL was inserted a second
99
+ // time, the opencode install line was wrapped in itself, and the consumer freshness check could
100
+ // never be clean (Pliny-Bot agentics-checks run 33963600563). The sentinel marks a processed
101
+ // lock; a lock that carries it is linted and otherwise left as it is.
102
+ const SENTINEL = "# post-processed by scripts/compile-agent-workflows.mjs; run it again and nothing changes";
103
+
34
104
  for (const file of readdirSync(workflowDirectory)) {
35
105
  if (!file.endsWith(".lock.yml")) continue;
36
106
 
37
107
  const path = join(workflowDirectory, file);
38
108
  const content = readFileSync(path, "utf8");
109
+ if (content.includes(SENTINEL)) {
110
+ lintLock(file, content);
111
+ continue;
112
+ }
39
113
  const patched = content
114
+ .replace(/^(# This file was automatically generated by gh-aw[^\n]*\n)/m, `$1${SENTINEL}\n`)
40
115
  .replaceAll("opencode run --print-logs --log-level DEBUG", "opencode run --log-level ERROR")
41
116
  .replaceAll("opencode run --print-logs --log-level ERROR", "opencode run --log-level ERROR")
42
117
  .replaceAll("opencode run --log-level ERROR", "opencode run --log-level ERROR")
@@ -138,6 +213,16 @@ for (const file of readdirSync(workflowDirectory)) {
138
213
  ' /tmp/gh-aw/aw-*.patch\n' +
139
214
  ' /tmp/gh-aw/aw-*.bundle\n' +
140
215
  ' /tmp/gh-aw/agent_output.json\n')
216
+ // "A failed run is already a red run": report-failure-as-issue: false and
217
+ // GH_AW_REPORT_FAILED_JOBS: "false" were supposed to silence issue creation, but
218
+ // 0.87.5's engine-failure reporter files an "[aw] ... failed" issue anyway (seen as
219
+ // Pliny-Bot #57: an OOM-killed gate produced a fresh issue nobody needs, while the
220
+ // retry belt is the actual remediation). `if: false` keeps the step compiled but
221
+ // never runs it, so the failure stays visible in the run and in the belt's attempt
222
+ // comment, and out of the issue tracker.
223
+ .replace(
224
+ /^([ \t]+)- name: Report failed jobs\n([ \t]*)id: report_failed_jobs\n[ \t]*if: always\(\)\n/m,
225
+ '$1- name: Report failed jobs\n$2id: report_failed_jobs\n$2if: false # never: failures surface in the run and the retry belt\n')
141
226
 
142
227
  // v0.87.5's arc-dind mode stages the engine CLI to a daemon-visible path but assumes the
143
228
  // Copilot engine: command -v copilot is empty under engine: opencode and cp "" fails the
@@ -222,6 +307,7 @@ for (const file of readdirSync(workflowDirectory)) {
222
307
  for (const check of [
223
308
  { name: "bundle upload glob", ok: /\n( +)- name: Upload agent artifacts\n\1 if: always\(\)\n\1 continue-on-error: true\n\1 uses: actions\/upload-artifact@[^\n]*\n\1 with:\n\1 name: \$\{\{ needs\.activation\.outputs\.artifact_prefix \}\}agent\n\1 path: \|\n(?:\1 [^\n]*\n)*\1 \/tmp\/gh-aw\/aw-\*\.bundle\n/ },
224
309
  { name: "placeholder step intact", ok: /\n( +)- name: Write agent output placeholder if missing\n\1 if: always\(\)\n\1 run: \|\n\1 if \[ ! -f \/tmp\/gh-aw\/agent_output\.json \]; then\n\1 echo '\{"items":\[\]\}' > \/tmp\/gh-aw\/agent_output\.json\n\1 fi\n/ },
310
+ { name: "failed-jobs reporter disabled", ok: /- name: Report failed jobs\n[ \t]*id: report_failed_jobs\n[ \t]*if: false[^\n]*\n/ },
225
311
  ]) {
226
312
  if (!check.ok.test(rewritten)) {
227
313
  process.stderr.write(
@@ -230,7 +316,16 @@ for (const file of readdirSync(workflowDirectory)) {
230
316
  );
231
317
  process.exitCode = 1;
232
318
  }
233
- }
319
+ }
320
+
321
+ if (!rewritten.includes(SENTINEL)) {
322
+ process.stderr.write(
323
+ `${file}: the generated-file header changed and the post-processing sentinel was not placed; ` +
324
+ `re-anchor the SENTINEL insert in loops/scripts/compile-agent-workflows.mjs.\n`,
325
+ );
326
+ process.exitCode = 1;
327
+ }
328
+ lintLock(file, rewritten);
234
329
 
235
330
  if (rewritten !== content) writeFileSync(path, rewritten);
236
331
  }
@@ -0,0 +1,76 @@
1
+ // Managed by @plainconceptsplatform/workflows. Source: loops/scripts/merge-changelog.mjs. Update with `workflows update --force`; consumer edits may be overwritten.
2
+ //
3
+ // A git merge driver for a newest-first changelog array.
4
+ //
5
+ // Every merge into the default branch invalidates every other open pull request that added a
6
+ // changelog entry, because they all insert at the top of the same list. The conflict is real
7
+ // but it is never interesting: both entries belong, newest first. Left to the default driver
8
+ // it costs a full model run per sibling pull request, or a person, and it recurs on every
9
+ // merge for as long as there is more than one pull request in flight.
10
+ //
11
+ // Registered from .gitattributes as `merge=changelog`, so git calls this instead of producing
12
+ // conflict markers. Falls back to exit 1 (a normal conflict) whenever it cannot be sure, so a
13
+ // genuine edit to the same entry is still escalated rather than silently mangled.
14
+ //
15
+ // Usage (git's merge driver contract): merge-changelog.mjs %O %A %B
16
+ // %O ancestor, %A ours (written back with the result), %B theirs.
17
+ import { readFileSync, writeFileSync } from "node:fs";
18
+
19
+ const [ancestorPath, oursPath, theirsPath] = process.argv.slice(2);
20
+
21
+ /** Parse, or return undefined so the caller can bail out to a normal conflict. */
22
+ function read(path) {
23
+ try {
24
+ const value = JSON.parse(readFileSync(path, "utf8"));
25
+ return Array.isArray(value?.changes) ? value : undefined;
26
+ } catch {
27
+ return undefined;
28
+ }
29
+ }
30
+
31
+ const ours = read(oursPath);
32
+ const theirs = read(theirsPath);
33
+ const ancestor = read(ancestorPath) ?? { changes: [] };
34
+
35
+ if (!ours || !theirs) {
36
+ process.stderr.write("merge-changelog: not a changelog document on both sides; leaving the conflict\n");
37
+ process.exit(1);
38
+ }
39
+
40
+ // An entry is identified by its commit when it has one, and by the whole record otherwise.
41
+ const identify = (entry) => entry?.commit ?? JSON.stringify(entry);
42
+
43
+ // Anything either side changed relative to the ancestor, plus everything the ancestor had.
44
+ // Union by identity: an append on both sides is the case this exists for.
45
+ const merged = new Map();
46
+ for (const entry of [...ancestor.changes, ...theirs.changes, ...ours.changes]) {
47
+ merged.set(identify(entry), entry);
48
+ }
49
+
50
+ // An entry the ancestor had and both sides removed should stay removed.
51
+ const kept = [...merged.values()].filter((entry) => {
52
+ const id = identify(entry);
53
+ const inAncestor = ancestor.changes.some((candidate) => identify(candidate) === id);
54
+ if (!inAncestor) return true;
55
+ return ours.changes.some((candidate) => identify(candidate) === id) ||
56
+ theirs.changes.some((candidate) => identify(candidate) === id);
57
+ });
58
+
59
+ // If the same identity carries different content on the two sides, somebody edited an entry
60
+ // rather than adding one. That is a real conflict and a person should look at it.
61
+ for (const entry of kept) {
62
+ const id = identify(entry);
63
+ const mine = ours.changes.find((candidate) => identify(candidate) === id);
64
+ const yours = theirs.changes.find((candidate) => identify(candidate) === id);
65
+ if (mine && yours && JSON.stringify(mine) !== JSON.stringify(yours)) {
66
+ process.stderr.write(`merge-changelog: entry ${id} differs on both sides; leaving the conflict\n`);
67
+ process.exit(1);
68
+ }
69
+ }
70
+
71
+ // Newest first, which is the order the file is read in and the order the page renders.
72
+ kept.sort((a, b) => String(b.timestamp ?? "").localeCompare(String(a.timestamp ?? "")));
73
+
74
+ const result = { ...theirs, ...ours, changes: kept };
75
+ writeFileSync(oursPath, `${JSON.stringify(result, null, 2)}\n`);
76
+ process.stderr.write(`merge-changelog: combined ${ours.changes.length} + ${theirs.changes.length} entries into ${kept.length}\n`);
@@ -0,0 +1,13 @@
1
+ # Managed by @plainconceptsplatform/workflows. Source: loops/templates/agentics/actionlint.yaml. Update with `workflows update --force`; consumer edits may be overwritten.
2
+ # Runners this repository actually has.
3
+ #
4
+ # actionlint knows GitHub's own labels and rejects everything else, so a workflow that
5
+ # names one of our runner groups was reported as an error on every pull request. Both
6
+ # of these are real: agents-arc runs the agentic workflows, RunnerLandingZone runs the
7
+ # work router. Listing them here is how actionlint is told a label is self-hosted
8
+ # rather than a typo, and it keeps that knowledge in a file the workflow generator
9
+ # does not overwrite.
10
+ self-hosted-runner:
11
+ labels:
12
+ - agents-arc
13
+ - RunnerLandingZone