@plainconceptsplatform/workflows 0.16.4 → 0.19.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -814,6 +814,54 @@ if worker_installed merge-gate; then
814
814
  fi
815
815
 
816
816
  if worker_installed implement; then
817
+ # "No pull request" has two causes and they need different words. gh-aw pushes through the
818
+ # GraphQL signed-commits API, which rebases onto the current parent, so a `main` that moved
819
+ # under a long run conflicts; gh-aw then keeps the work by filing the patch as an issue rather
820
+ # than dropping it. Numa #657 hit this: a 50 KB patch that passed every validation gate, filed
821
+ # as issue #658, while the worker told the reader "nothing landed" and flagged a retry that
822
+ # would conflict the same way. The two paths must stay distinguishable, and both must be driven
823
+ # by a job output rather than by parsing prose.
824
+ #
825
+ # The discriminator is the item counter. `code_push_failure_count` looks like the right signal
826
+ # and is not: the deliberate reproduction on dogfood #10 filed the patch as #11 and still
827
+ # reported `Status: success`, `Successful: 1` and a resolved `GH_AW_CODE_PUSH_FAILURE_COUNT: 0`,
828
+ # so a worker gated on that count posts "nothing landed" over the top of a patch that exists.
829
+ # `create_pull_request` is the only safe output implement permits, so one succeeded item with
830
+ # no pull request number means the push fell back; nothing produced leaves the counter at 0.
831
+ PUSH_FALLBACK_OK=1
832
+ if ! grep -qF 'process_safe_outputs_items_succeeded' "$IMPLEMENT_WORKER_MD"; then
833
+ PUSH_FALLBACK_OK=0
834
+ echo "FAIL: implement does not read process_safe_outputs_items_succeeded, so a conflicted push reads as 'nothing landed'" >&2
835
+ fi
836
+ # Regating on the count that gh-aw leaves at 0 through a fallback is the specific regression.
837
+ # Matched on the `if:` line only: the comment above the branch names the count to explain why
838
+ # it is the wrong signal, and a bare symbol grep would fire on that prose instead of the guard.
839
+ if [ "$(count -cE '^ *if:.*code_push_failure_count' "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
840
+ PUSH_FALLBACK_OK=0
841
+ echo "FAIL: implement gates a no-pull-request path on code_push_failure_count, which is 0 when gh-aw files the patch as an issue" >&2
842
+ fi
843
+ for needed in 'PUSH_CONFLICT_COMMENT' 'NO_PULL_REQUEST_COMMENT'; do
844
+ grep -qF "env.${needed}" "$IMPLEMENT_WORKER_MD" || {
845
+ PUSH_FALLBACK_OK=0
846
+ echo "FAIL: implement no longer says env.${needed} on any path" >&2
847
+ }
848
+ done
849
+ # Collapsing them back into one message is the regression this guards: each is defined once in
850
+ # the env block and printed on exactly one path, so two usages of either means the conditions
851
+ # have been merged or duplicated.
852
+ if [ "$(count -cF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ] ||
853
+ [ "$(count -cF 'env.NO_PULL_REQUEST_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
854
+ PUSH_FALLBACK_OK=0
855
+ echo "FAIL: implement should print each no-pull-request message on exactly one path" >&2
856
+ fi
857
+ # And the two paths must be mutually exclusive, or a conflicted push gets both comments.
858
+ if [ "$(count -cF "process_safe_outputs_items_succeeded != '0'" "$IMPLEMENT_WORKER_MD")" -lt 2 ] ||
859
+ [ "$(count -cF "process_safe_outputs_items_succeeded == '0'" "$IMPLEMENT_WORKER_MD")" -lt 2 ]; then
860
+ PUSH_FALLBACK_OK=0
861
+ echo "FAIL: the conflicted-push and no-patch paths in implement are not mutually exclusive" >&2
862
+ fi
863
+ if [ "$PUSH_FALLBACK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
864
+
817
865
  # A provider outage kills a run in a couple of minutes with no answer, and the same issue used
818
866
  # to be handed to a human for it. The implement worker retries those and only those: a run that
819
867
  # worked for half an hour and then failed produced an answer that was wrong, and repeating it
@@ -1118,6 +1166,56 @@ if [ -f "$AUDIT_CLOSE_YML" ] && worker_installed audit; then
1118
1166
  if [ "$AC_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1119
1167
  fi
1120
1168
 
1169
+ echo "── Merge gate validator ──────────────────────────────────────────────────"
1170
+
1171
+ # The validator decides whether a gate run merges, remediates, parks, or is thrown away, and
1172
+ # until now nothing executed it -- this file only read it. Its `remediated` rule was changed on
1173
+ # reasoning alone, and the worker has not run in production since, so these fixtures are the only
1174
+ # evidence the change is right. Executing the real script is the same technique that finally
1175
+ # caught the belt's jq bug, which every reading assertion had walked past.
1176
+ GATE_VALIDATOR="${HERE}/../validate-merge-gate-output/validate-merge-gate-output.sh"
1177
+ if [ -f "$GATE_VALIDATOR" ] && worker_installed merge-gate; then
1178
+ VALIDATOR_OK=1
1179
+ gate_fixture="${TMPDIR:-/tmp}/route-matrix-gate-$$.json"
1180
+
1181
+ gate_case() {
1182
+ local name="$1" want="$2" json="$3" conclusion="$4"
1183
+ printf '%s' "$json" > "$gate_fixture"
1184
+ local got
1185
+ got=$(bash "$GATE_VALIDATOR" "$gate_fixture" 7 "$conclusion" 2>&1)
1186
+ if [ "$got" != "$want" ]; then
1187
+ VALIDATOR_OK=0
1188
+ echo "FAIL: the merge-gate validator called '${name}' ${got}, expected ${want}" >&2
1189
+ fi
1190
+ }
1191
+
1192
+ gate_verdict='{"type":"add_comment","item_number":7,"body":"<!-- agent-merge-gate -->\n**Verdict:** VERB"}'
1193
+ gate_push='{"type":"push_to_pull_request_branch","pr_number":9}'
1194
+ gate_items() { printf '{"items":[%s]}' "$1"; }
1195
+ gate_comment() { printf '%s' "${gate_verdict/VERB/$1}"; }
1196
+
1197
+ gate_case "merge on green with no push" merge "$(gate_items "$(gate_comment merge)")" success
1198
+ gate_case "merge on a failed CI run" invalid "$(gate_items "$(gate_comment merge)")" failure
1199
+ gate_case "merge carrying a push" invalid "$(gate_items "$(gate_comment merge),${gate_push}")" success
1200
+ # The case the rule exists for. A conflicting pull request has no merge ref, so GitHub never
1201
+ # runs CI on that head and the belt falls back to the branch's last verdict, usually success.
1202
+ # Requiring conclusion == failure here discarded the resolved merge commit the agent had just
1203
+ # pushed, and the belt re-dispatched on the same verdict up to six times.
1204
+ gate_case "remediated with one push, CI green" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" success
1205
+ gate_case "remediated with one push, CI red" remediated "$(gate_items "$(gate_comment remediated),${gate_push}")" failure
1206
+ gate_case "remediated with no push" invalid "$(gate_items "$(gate_comment remediated)")" failure
1207
+ gate_case "remediated with two pushes" invalid "$(gate_items "$(gate_comment remediated),${gate_push},${gate_push}")" failure
1208
+ gate_case "review with no push" review "$(gate_items "$(gate_comment review)")" failure
1209
+ gate_case "review carrying a push" invalid "$(gate_items "$(gate_comment review),${gate_push}")" failure
1210
+ gate_case "a verdict aimed at another issue" invalid '{"items":[{"type":"add_comment","item_number":99,"body":"<!-- agent-merge-gate -->\n**Verdict:** merge"}]}' success
1211
+ gate_case "no verdict in the output" invalid '{"items":[{"type":"add_comment","item_number":7,"body":"just a note"}]}' success
1212
+ gate_case "an empty item list" invalid '{"items":[]}' success
1213
+ gate_case "output that is not an item list" invalid '{"nope":true}' success
1214
+
1215
+ rm -f "$gate_fixture"
1216
+ if [ "$VALIDATOR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1217
+ fi
1218
+
1121
1219
  echo "── Merge gate park ───────────────────────────────────────────────────────"
1122
1220
 
1123
1221
  # A gate verdict parks the code it was given on. Both dispatch paths -- detect-pr-conflicts and
@@ -1392,6 +1490,230 @@ else
1392
1490
  printf '%s\n' "$password_hits" >&2
1393
1491
  fi
1394
1492
 
1493
+ echo "── Stale dispatch and empty runs ─────────────────────────────────────────"
1494
+
1495
+ # A route dispatched while an issue was open must not execute after it is closed. The classifier
1496
+ # refuses a closed issue, but it can only read `github.event.issue.state`: the state when the
1497
+ # event fired, and absent altogether on a workflow_dispatch. This fleet queues for a runner for
1498
+ # ten minutes and more. Numa #659 was closed one second after a comment dispatched refine; the run
1499
+ # reached `reserve` thirteen minutes later, took the reservation, and refined a closed issue for
1500
+ # thirty-eight minutes, ending by labelling it `refined`, `implement` and `sp-5`. Every job
1501
+ # reported success. Asserted per worker because the gate is three lines in each and none of them
1502
+ # failing produces a red run.
1503
+ STALE_DISPATCH_OK=1
1504
+ for route in refine implement triage apply-review; do
1505
+ worker_md="${WORKFLOWS_DIR}/agent-${route}.md"
1506
+ [ -f "$worker_md" ] || continue
1507
+ if ! grep -q '^ still_open:' "$worker_md"; then
1508
+ STALE_DISPATCH_OK=0
1509
+ echo "FAIL: agent-${route} has no still_open job; a route dispatched before the issue closed would run on it anyway" >&2
1510
+ continue
1511
+ fi
1512
+ # The reservation must be gated, or bot-working lands on a closed issue, and the agent must be
1513
+ # gated through the worker's own `if:`. Two distinct call sites.
1514
+ if [ "$(count -cF "needs.still_open.outputs.open == 'true'" "$worker_md")" -lt 2 ]; then
1515
+ STALE_DISPATCH_OK=0
1516
+ echo "FAIL: agent-${route} does not gate both the reservation and the agent on still_open" >&2
1517
+ fi
1518
+ # The gate job must carry no `needs:` of its own. gh-aw hoists exactly those custom jobs into
1519
+ # the activation job's dependencies, which is what lets the top-level `if:` read their outputs;
1520
+ # a gate job that gained a dependency would stop being hoisted and the clause would silently
1521
+ # evaluate to empty, which is always true. Asserted on the job block, not on the file, because
1522
+ # `needs: [still_open]` also appears on the reserve job and a file-wide search for it passed
1523
+ # for two workers that never declared anything.
1524
+ # activation must be given the dependency, or the top-level `if:` reads an empty value and the
1525
+ # clause is false: the agent never runs at all. gh-aw does not hoist this job, and the entry
1526
+ # lives at two spaces under `on:`, which is where gh-aw reads activation's dependency list.
1527
+ # Matched inside the `on:` block, because the same text appears on the reserve job and a
1528
+ # file-wide search for it passed for two workers that had declared nothing.
1529
+ on_block=$(awk '/^on:/{f=1; next} f && /^[a-z][a-z-]*:/{exit} f' "$worker_md")
1530
+ if ! printf '%s
1531
+ ' "$on_block" | grep -qE '^ needs: \[.*still_open'; then
1532
+ STALE_DISPATCH_OK=0
1533
+ echo "FAIL: agent-${route} does not list still_open under on.needs, so activation reads an empty value and the agent never runs" >&2
1534
+ fi
1535
+ gate_block=$(awk '/^ still_open:/{f=1; next} f && /^ [a-z_]+:/{exit} f' "$worker_md")
1536
+ if printf '%s
1537
+ ' "$gate_block" | grep -qE '^ needs:'; then
1538
+ STALE_DISPATCH_OK=0
1539
+ echo "FAIL: agent-${route}'s still_open job declares needs:, so gh-aw will not hoist it and the top-level if: reads an empty value" >&2
1540
+ fi
1541
+ done
1542
+ if [ "$STALE_DISPATCH_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1543
+
1544
+ # An audit that emitted neither a report nor a noop used to end green: `conclude` requires a
1545
+ # processed item, and a skipped job is not a failure. A whole agent run produced nothing and said
1546
+ # nothing. Both halves of the condition are asserted: the count alone would fire on a legitimate
1547
+ # `noop` if noop does not increment it, and the noop_message alone would miss the empty run.
1548
+ if worker_installed audit; then
1549
+ AUDIT_EMPTY_OK=1
1550
+ AUDIT_WORKER_MD="${WORKFLOWS_DIR}/agent-audit.md"
1551
+ grep -q '^ empty_run:' "$AUDIT_WORKER_MD" ||
1552
+ { AUDIT_EMPTY_OK=0; echo "FAIL: the audit has no empty_run job; a run that files nothing reports success" >&2; }
1553
+ grep -qF "process_safe_outputs_processed_count == '0'" "$AUDIT_WORKER_MD" ||
1554
+ { AUDIT_EMPTY_OK=0; echo "FAIL: the audit's empty_run does not test the processed count" >&2; }
1555
+ # empty_run must not depend on gh-aw's `conclusion` job. That job needs every custom job in the
1556
+ # worker, so naming it is a cycle and the whole workflow fails to compile -- which is how the
1557
+ # first attempt at this was caught. It also means `noop_message`, the one output that would say
1558
+ # a clean audit deliberately filed nothing, cannot be read from here.
1559
+ if grep -qE '^ needs: \[.*conclusion' "$AUDIT_WORKER_MD"; then
1560
+ AUDIT_EMPTY_OK=0
1561
+ echo "FAIL: the audit's empty_run depends on the conclusion job, which is a dependency cycle and will not compile" >&2
1562
+ fi
1563
+ if [ "$AUDIT_EMPTY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1564
+ fi
1565
+
1566
+ echo "── Runner pools ──────────────────────────────────────────────────────────"
1567
+
1568
+ # Where every job runs, stated once and asserted, because GitHub gives a wrong pool no error: a
1569
+ # job addressed to a label no runner carries simply queues, and a job addressed to the wrong pool
1570
+ # runs in the wrong place. The pool was a preserved consumer value until 0.19.0, and that is how
1571
+ # one repository came to run triage, refine, implement, apply-review and audit on
1572
+ # RunnerLandingZone while merge-gate and release stayed on agents-arc: the override reached the
1573
+ # workers installed at the time and never the ones added later. Nothing reported the split.
1574
+ #
1575
+ # agent jobs of every worker except release -> agents-arc (the Azure fleet in
1576
+ # agentrunner-pro-rg-01,
1577
+ # runner group `agentic`)
1578
+ # agent-release.md and work-router.yml -> RunnerLandingZone
1579
+ # ubuntu-latest -> GitHub's own runner, chosen by nobody, left
1580
+ # wherever the package already has it
1581
+ AGENT_POOL="agents-arc"
1582
+ PLUMBING_POOL="RunnerLandingZone"
1583
+ POOL_OK=1
1584
+
1585
+ # `|| true` on both greps: a file naming no pool at all, or only ubuntu-latest, makes grep exit 1,
1586
+ # and under `set -o pipefail` that aborts the whole matrix instead of failing this one assertion.
1587
+ # A router mutated to ubuntu-latest everywhere did exactly that: the suite died without printing,
1588
+ # so the mutation looked caught when in fact nothing had been checked.
1589
+ pools_named() {
1590
+ { grep -hoE '^[[:space:]]*runs-on(-slim)?: [^[:space:]]+' "$1" 2>/dev/null || true; } \
1591
+ | sed 's/.*: //' | { grep -v '^ubuntu-latest$' || true; } | sort -u
1592
+ }
1593
+
1594
+ for worker_md in "${WORKFLOWS_DIR}"/agent-*.md; do
1595
+ [ -f "$worker_md" ] || continue
1596
+ worker_name="$(basename "$worker_md" .md)"
1597
+ want="$AGENT_POOL"
1598
+ [ "$worker_name" = "agent-release" ] && want="$PLUMBING_POOL"
1599
+ got="$(pools_named "$worker_md" | tr '\n' ' ' | sed 's/ $//')"
1600
+ if [ "$got" != "$want" ]; then
1601
+ POOL_OK=0
1602
+ echo "FAIL: ${worker_name} names runner pool(s) '${got}' but must name only '${want}'" >&2
1603
+ fi
1604
+ done
1605
+
1606
+ router_pools="$(pools_named "$ROUTER_YML" | tr '\n' ' ' | sed 's/ $//')"
1607
+ if [ "$router_pools" != "$PLUMBING_POOL" ]; then
1608
+ POOL_OK=0
1609
+ echo "FAIL: the router names runner pool(s) '${router_pools}' but must name only '${PLUMBING_POOL}'" >&2
1610
+ fi
1611
+
1612
+ # And the pool must not be reintroduced as a per-consumer value: that is the mechanism that let
1613
+ # the split happen, and it left no trace anywhere.
1614
+ if [ -d "${HERE}/../../../cli/src" ] && grep -rqE 'preserveRunnerPool|runnerPools' "${HERE}/../../../cli/src" 2>/dev/null; then
1615
+ POOL_OK=0
1616
+ echo "FAIL: the installer preserves a consumer's runner pool again; the pool is the package's to set" >&2
1617
+ fi
1618
+
1619
+ if [ "$POOL_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1620
+
1621
+ echo "── Worker input wiring ───────────────────────────────────────────────────"
1622
+
1623
+ # A worker that declares a workflow_call input and never reads it is the shape of the worst
1624
+ # outage this pipeline has had. The router resolved the CI run ID, passed it as `ci-run-id`, and
1625
+ # the merge gate declared the input and then called identify-gate-subject without it -- so
1626
+ # `RUN_ID` was always empty, the evidence step wrote `failed-jobs.json` as `[]`, and the agent,
1627
+ # holding no failing job name and no logs, chose `review` every time CI went red. Pliny-Bot #129,
1628
+ # #130 and #131 all ended up waiting on a human with CI legitimately red and remediable, and the
1629
+ # housekeeping digest reported it as work needing a person. Nothing went red: every job succeeded
1630
+ # at doing nothing. This asserts the class, because reading each worker by hand is how it was
1631
+ # missed for as long as it was.
1632
+ #
1633
+ # `agent-audit:trigger-kind` is exempt and stays listed rather than deleted: it is a
1634
+ # workflow_dispatch choice a person picks on the router, threaded through for symmetry, with no
1635
+ # behaviour attached at the far end. Removing it would take a dispatch option away from people,
1636
+ # so it is recorded as known-inert instead. Any *other* unread input fails.
1637
+ DEAD_INPUT_EXEMPT="agent-audit:trigger-kind"
1638
+ INPUT_WIRING_OK=1
1639
+ for worker_md in "${WORKFLOWS_DIR}"/agent-*.md; do
1640
+ [ -f "$worker_md" ] || continue
1641
+ worker_name="$(basename "$worker_md" .md)"
1642
+ # The declared inputs: the `inputs:` mapping under `on: workflow_call:`, whose keys sit at six
1643
+ # spaces. Stop at the first line indented less than that which is not blank.
1644
+ declared=$(awk '
1645
+ /^on:/ { in_on = 1; next }
1646
+ in_on && /^[a-z#]/ { exit }
1647
+ in_on && /^ workflow_call:/ { in_wc = 1; next }
1648
+ in_wc && /^ inputs:/ { in_inputs = 1; next }
1649
+ in_inputs && /^ [a-z]/ { in_inputs = 0 }
1650
+ in_inputs && /^ [a-z0-9_-]+:[[:space:]]*$/ {
1651
+ gsub(/[ :]/, "", $0); print $0
1652
+ }
1653
+ ' "$worker_md")
1654
+ for input_name in $declared; do
1655
+ case "${worker_name}:${input_name}" in
1656
+ "$DEAD_INPUT_EXEMPT") continue ;;
1657
+ esac
1658
+ if [ "$(count -cE "inputs\.${input_name}([^a-zA-Z0-9_-]|\$)" "$worker_md")" -eq 0 ]; then
1659
+ INPUT_WIRING_OK=0
1660
+ echo "FAIL: ${worker_name} declares the input '${input_name}' and never reads it; the router's value is discarded and the job succeeds at doing nothing" >&2
1661
+ fi
1662
+ done
1663
+ done
1664
+ if [ "$INPUT_WIRING_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1665
+
1666
+ # The belt dispatches one gate per tick and stops. The gate's concurrency group holds a single
1667
+ # pending run, so GitHub cancels every earlier pending dispatch in it -- `cancel-in-progress:
1668
+ # false` only protects a run that has already started. A loop without this `break` dispatched one
1669
+ # gate per eligible pull request, ran the last, and discarded the rest with no comment, no
1670
+ # recorded attempt, and nothing on the pull request to show it had been skipped. Asserted because
1671
+ # removing the break produces no error anywhere: the extra dispatches all return 204.
1672
+ BELT_ONE_PER_TICK_OK=1
1673
+ belt_dispatch_block=$(awk '
1674
+ /Dispatching Merge Gate for PR/ { found = 1 }
1675
+ found { print }
1676
+ found && /^ *done$/ { exit }
1677
+ ' "$ROUTER_YML")
1678
+ if [ -z "$belt_dispatch_block" ]; then
1679
+ BELT_ONE_PER_TICK_OK=0
1680
+ echo "FAIL: the merge belt's gate dispatch could not be located, so its one-per-tick guard cannot be checked" >&2
1681
+ elif ! printf '%s\n' "$belt_dispatch_block" | grep -qE '^ *break$'; then
1682
+ BELT_ONE_PER_TICK_OK=0
1683
+ echo "FAIL: the merge belt dispatches a gate per eligible pull request and never breaks; only the last stays pending and the rest are cancelled in silence" >&2
1684
+ fi
1685
+ # And the group it relies on must still be the serialising one.
1686
+ # Anchored to the whole line: a substring search matched `merge-belt-renamed` and stayed green
1687
+ # through a mutation that removed the serialisation the break depends on.
1688
+ grep -qE '^ *group: merge-belt *$' "$ROUTER_YML" ||
1689
+ { BELT_ONE_PER_TICK_OK=0; echo "FAIL: the merge gate is no longer serialised on the merge-belt concurrency group" >&2; }
1690
+ if [ "$BELT_ONE_PER_TICK_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1691
+
1692
+ # The specific wiring that broke, asserted end to end: the router resolves the run ID, the gate
1693
+ # forwards it, and the action seeds from it rather than only from its own lookup.
1694
+ if worker_installed merge-gate; then
1695
+ GATE_RUN_ID_OK=1
1696
+ grep -Fq 'ci-run-id: ${{ needs.classify.outputs.ci-run-id }}' "$ROUTER_YML" ||
1697
+ { GATE_RUN_ID_OK=0; echo "FAIL: the router no longer passes ci-run-id to the merge gate" >&2; }
1698
+ grep -Fq 'ci-run-id: ${{ inputs.ci-run-id }}' "$MERGE_GATE_WORKER_MD" ||
1699
+ { GATE_RUN_ID_OK=0; echo "FAIL: the merge gate does not forward ci-run-id to identify-gate-subject, so it has no CI failure evidence" >&2; }
1700
+ GATE_SUBJECT_ACTION="${HERE}/../identify-gate-subject/action.yml"
1701
+ if [ -f "$GATE_SUBJECT_ACTION" ]; then
1702
+ grep -Fq 'ci_run_id="$CI_RUN_ID"' "$GATE_SUBJECT_ACTION" ||
1703
+ { GATE_RUN_ID_OK=0; echo "FAIL: identify-gate-subject does not seed the run ID from its input, so a caller that knows it is ignored" >&2; }
1704
+ # The regression: resolving the ID only when the conclusion is missing.
1705
+ # Anchored to the start of the line: the repaired code keeps an `elif [ -z "$ci_conclusion" ]`
1706
+ # branch for the caller that genuinely has no verdict, and a fixed-string search matched that
1707
+ # `elif` as a substring -- the same way this file's own explanatory comments have tripped
1708
+ # three earlier assertions.
1709
+ if [ "$(count -cE '^ *if \[ -z "\$ci_conclusion" \]; then' "$GATE_SUBJECT_ACTION")" -ne 0 ]; then
1710
+ GATE_RUN_ID_OK=0
1711
+ echo "FAIL: identify-gate-subject resolves the CI run only when the conclusion is unknown; a router that passes both gets an empty run ID" >&2
1712
+ fi
1713
+ fi
1714
+ if [ "$GATE_RUN_ID_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1715
+ fi
1716
+
1395
1717
  echo
1396
1718
  if [ "$FAIL" -eq 0 ]; then
1397
1719
  echo "Route matrix: ${PASS} passed"
@@ -49,7 +49,37 @@ on:
49
49
 
50
50
  # Rung 4. Router has classified the event; this job validates PR ownership and checks for
51
51
  # substantive feedback. A custom job, not `on.steps`, because the prompt needs these values.
52
+ # The gate job that the top-level `if:` reads. gh-aw folds that `if:` into the generated
53
+ # activation job but gives activation no dependency on the job, so the reference resolves
54
+ # to '' and the clause is false -- the agent would never run. The package's own validator
55
+ # catches it after compilation; this is the line it asks for, the same one the merge gate
56
+ # uses for protected_changes.
57
+ needs: [still_open]
52
58
  jobs:
59
+ # A route dispatched while the issue was open must not execute after it has been closed. The
60
+ # classifier can only see `github.event.issue.state`, which is the state when the event fired
61
+ # and is absent on a workflow_dispatch, and this fleet queues for a runner for ten minutes and
62
+ # more. Numa #659 was closed one second after a comment dispatched refine; the run reached
63
+ # `reserve` thirteen minutes later and refined a closed issue to completion. Read now, once,
64
+ # and gate both the reservation and the agent on it.
65
+ still_open:
66
+ runs-on: agents-arc
67
+ permissions:
68
+ contents: read
69
+ issues: read
70
+ outputs:
71
+ open: ${{ steps.state.outputs.open }}
72
+ steps:
73
+ - name: Checkout workflow actions
74
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
75
+ with:
76
+ persist-credentials: false
77
+ - name: Read the issue state
78
+ id: state
79
+ uses: ./.github/actions/require-open-issue
80
+ with:
81
+ token: ${{ github.token }}
82
+ issue-number: ${{ inputs.issue-number }}
53
83
  subject:
54
84
  runs-on: agents-arc
55
85
  permissions:
@@ -118,8 +148,8 @@ jobs:
118
148
  echo "PR #$PR has $unresolved unresolved thread(s)"
119
149
 
120
150
  reserve:
121
- needs: subject
122
- if: needs.subject.outputs.found == 'true' && needs.subject.outputs.issue != ''
151
+ needs: [subject, still_open]
152
+ if: needs.subject.outputs.found == 'true' && needs.subject.outputs.issue != '' && needs.still_open.outputs.open == 'true'
123
153
  runs-on: agents-arc
124
154
  permissions:
125
155
  contents: read
@@ -290,7 +320,7 @@ jobs:
290
320
  ${{ env.INCOMPLETE_COMMENT }}
291
321
  [View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
292
322
 
293
- if: needs.subject.outputs.found == 'true'
323
+ if: needs.subject.outputs.found == 'true' && needs.still_open.outputs.open == 'true'
294
324
 
295
325
  runs-on: agents-arc
296
326
  runs-on-slim: agents-arc
@@ -119,6 +119,42 @@ jobs:
119
119
  bug
120
120
  refine
121
121
 
122
+ # An audit that produced nothing reported success. The whole run -- a full agent, its tokens,
123
+ # its half hour -- ended with `conclude` skipped, because that job requires a processed item,
124
+ # and a skipped job leaves the run green. Numa's audit on 2026-09-10 did exactly that:
125
+ # `agent_output.json` 24 bytes, `safe-output-items.jsonl` empty, every job success, no report
126
+ # filed and nothing anywhere saying so. The next scheduled audit would have looked identical.
127
+ #
128
+ # This fires when the agent succeeded and the safe-outputs handler processed nothing at all.
129
+ #
130
+ # Step 6 of the prompt gives a clean codebase its own outcome -- call `noop` and stop -- and
131
+ # that outcome must not be reported as a failure. `noop` is a registered safe output here
132
+ # (`noop: max 1`), so the handler receives it as a message and the processed count is not zero,
133
+ # which keeps a deliberately empty audit out of this branch. That is read from the handler
134
+ # configuration rather than observed in a run: no audit in these repositories has yet emitted a
135
+ # noop. The direct signal, `noop_message`, belongs to gh-aw's `conclusion` job, and that job
136
+ # depends on every custom job here, so naming it is a dependency cycle and the worker will not
137
+ # compile -- the first version of this job was rejected for exactly that.
138
+ #
139
+ # If a genuinely clean audit ever fails here, that is the assumption breaking, and the fix is to
140
+ # carry the noop through a job this one can depend on rather than to widen the condition.
141
+ empty_run:
142
+ needs: [agent, safe_outputs]
143
+ if: >
144
+ needs.agent.result == 'success' &&
145
+ needs.safe_outputs.outputs.process_safe_outputs_processed_count == '0'
146
+ runs-on: agents-arc
147
+ permissions:
148
+ contents: read
149
+ steps:
150
+ - name: Report an audit that produced no outcome
151
+ env:
152
+ RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
153
+ run: |
154
+ set -euo pipefail
155
+ echo "::error::The audit agent finished without emitting a report or a noop. The prompt requires one of the two: create_issue with the findings, or noop when the codebase is clean. Nothing was filed and no reason was given, so this run is a failure rather than a clean audit. The next scheduled audit will try again. ${RUN_URL}"
156
+ exit 1
157
+
122
158
  safe-outputs:
123
159
  # A failed run is already a red run. An issue per failure buries the real backlog
124
160
  # under noise nobody closes.
@@ -17,6 +17,13 @@ env:
17
17
  STALLED_LABEL: stalled
18
18
  PR_PENDING_LABEL: pr-pending
19
19
  NO_PULL_REQUEST_COMMENT: "The implementation run finished without producing a pull request. Nothing was lost, but nothing landed either: the issue keeps `implement` and is flagged for a retry."
20
+ # Said when the agent DID write the code and the push failed. gh-aw pushes through the GraphQL
21
+ # signed-commits API, which rebases onto the current parent, so a `main` that moved under a long
22
+ # run conflicts; gh-aw keeps the work by filing the patch as an issue rather than dropping it,
23
+ # and comments the link on this issue itself. Telling someone "nothing landed" over the top of
24
+ # that sends them to reimplement work that already exists. One line: the compiler flattens a
25
+ # multi-line env value.
26
+ PUSH_CONFLICT_COMMENT: "The implementation produced a patch, but pushing it failed: it no longer applies to `main`, which moved while this ran. gh-aw filed the patch as a separate issue rather than losing it, and linked it in its own comment above. The work is there and needs rebasing onto current `main`, not writing again."
20
27
  GIT_AUTHOR_NAME: "github-actions[bot]"
21
28
  GIT_AUTHOR_EMAIL: "github-actions[bot]@users.noreply.github.com"
22
29
  GIT_COMMITTER_NAME: "github-actions[bot]"
@@ -70,7 +77,37 @@ on:
70
77
  required: false
71
78
  type: string
72
79
  default: '0'
80
+ # The gate job that the top-level `if:` reads. gh-aw folds that `if:` into the generated
81
+ # activation job but gives activation no dependency on the job, so the reference resolves
82
+ # to '' and the clause is false -- the agent would never run. The package's own validator
83
+ # catches it after compilation; this is the line it asks for, the same one the merge gate
84
+ # uses for protected_changes.
85
+ needs: [still_open]
73
86
  jobs:
87
+ # A route dispatched while the issue was open must not execute after it has been closed. The
88
+ # classifier can only see `github.event.issue.state`, which is the state when the event fired
89
+ # and is absent on a workflow_dispatch, and this fleet queues for a runner for ten minutes and
90
+ # more. Numa #659 was closed one second after a comment dispatched refine; the run reached
91
+ # `reserve` thirteen minutes later and refined a closed issue to completion. Read now, once,
92
+ # and gate both the reservation and the agent on it.
93
+ still_open:
94
+ runs-on: agents-arc
95
+ permissions:
96
+ contents: read
97
+ issues: read
98
+ outputs:
99
+ open: ${{ steps.state.outputs.open }}
100
+ steps:
101
+ - name: Checkout workflow actions
102
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
103
+ with:
104
+ persist-credentials: false
105
+ - name: Read the issue state
106
+ id: state
107
+ uses: ./.github/actions/require-open-issue
108
+ with:
109
+ token: ${{ github.token }}
110
+ issue-number: ${{ inputs.issue-number }}
74
111
  eligibility:
75
112
  runs-on: agents-arc
76
113
  permissions:
@@ -107,8 +144,8 @@ jobs:
107
144
  echo "eligible=true" >> "$GITHUB_OUTPUT"
108
145
 
109
146
  reserve:
110
- needs: [eligibility]
111
- if: needs.eligibility.outputs.eligible == 'true'
147
+ needs: [eligibility, still_open]
148
+ if: needs.eligibility.outputs.eligible == 'true' && needs.still_open.outputs.open == 'true'
112
149
  runs-on: agents-arc
113
150
  permissions:
114
151
  contents: read
@@ -227,8 +264,43 @@ jobs:
227
264
  # comment, and no bot-working — which also hid it from the hourly stale-reservation sweep.
228
265
  # Comments do not re-trigger implement, so nothing on any path would ever look at it again.
229
266
  # It was the only failure in the fleet that signalled nobody at all.
267
+ #
268
+ # There are two ways to reach "no pull request", and telling a person they are the same
269
+ # thing wastes their time. gh-aw pushes through the GraphQL signed-commits API, which
270
+ # rebases the commit range onto the current parent; when `main` has moved under a long run
271
+ # the rebase conflicts, and gh-aw keeps the work by filing the patch as an issue instead of
272
+ # dropping it, commenting the link on this issue itself. Saying "nothing landed" over the
273
+ # top of that is false: the patch exists and needs rebasing, not reimplementing. Seen on
274
+ # Numa #657, where the same change had landed on main by hand while the agent was writing
275
+ # it, and reproduced deliberately on dogfood #10 -> #11.
276
+ #
277
+ # The signal is the item counter, not `code_push_failure_count`. gh-aw treats the fallback
278
+ # as a *successful* outcome for the item -- the dogfood run logged `Status: success`,
279
+ # `Successful: 1` and a resolved `GH_AW_CODE_PUSH_FAILURE_COUNT: 0` while filing #11 -- so
280
+ # gating on that count posted the wrong message. `create_pull_request` is the only safe
281
+ # output this worker permits, so one succeeded item with no pull request number can only
282
+ # mean the push fell back to an issue. Nothing produced at all leaves the counter at 0.
283
+ - name: Flag a patch that could not be pushed
284
+ if: needs.safe_outputs.outputs.created_pr_number == '' && needs.safe_outputs.outputs.process_safe_outputs_items_succeeded != '0' && needs.safe_outputs.outputs.process_safe_outputs_items_succeeded != ''
285
+ uses: ./.github/actions/add-issue-labels
286
+ with:
287
+ token: ${{ steps.app-token.outputs.token }}
288
+ issue-number: ${{ inputs.issue-number }}
289
+ labels: |-
290
+ ${{ env.REVIEW_LABEL }}
291
+ ${{ env.STALLED_LABEL }}
292
+ - name: Say where the patch went
293
+ if: needs.safe_outputs.outputs.created_pr_number == '' && needs.safe_outputs.outputs.process_safe_outputs_items_succeeded != '0' && needs.safe_outputs.outputs.process_safe_outputs_items_succeeded != ''
294
+ uses: ./.github/actions/create-issue-comment
295
+ with:
296
+ token: ${{ steps.app-token.outputs.token }}
297
+ issue-number: ${{ inputs.issue-number }}
298
+ body: |
299
+ ${{ env.IMPLEMENT_MARKER }}
300
+ ${{ env.PUSH_CONFLICT_COMMENT }}
301
+ [View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
230
302
  - name: Flag a run that produced no pull request
231
- if: needs.safe_outputs.outputs.created_pr_number == ''
303
+ if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '')
232
304
  uses: ./.github/actions/add-issue-labels
233
305
  with:
234
306
  token: ${{ steps.app-token.outputs.token }}
@@ -237,7 +309,7 @@ jobs:
237
309
  ${{ env.REVIEW_LABEL }}
238
310
  ${{ env.STALLED_LABEL }}
239
311
  - name: Say so on the issue
240
- if: needs.safe_outputs.outputs.created_pr_number == ''
312
+ if: needs.safe_outputs.outputs.created_pr_number == '' && (needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '0' || needs.safe_outputs.outputs.process_safe_outputs_items_succeeded == '')
241
313
  uses: ./.github/actions/create-issue-comment
242
314
  with:
243
315
  token: ${{ steps.app-token.outputs.token }}
@@ -436,7 +508,7 @@ jobs:
436
508
  needs: [eligibility]
437
509
  if: needs.eligibility.outputs.eligible == 'true'
438
510
 
439
- if: inputs.issue-number != ''
511
+ if: inputs.issue-number != '' && needs.still_open.outputs.open == 'true'
440
512
 
441
513
  runs-on: agents-arc
442
514
  runs-on-slim: agents-arc
@@ -114,6 +114,7 @@ jobs:
114
114
  token: ${{ github.token }}
115
115
  pr-number: ${{ inputs.pr-number }}
116
116
  ci-conclusion: ${{ inputs.ci-conclusion }}
117
+ ci-run-id: ${{ inputs.ci-run-id }}
117
118
  linked-issue: ${{ inputs.linked-issue }}
118
119
  require-label: ${{ env.IMPLEMENT_LABEL }}
119
120
  - name: Block a pull request with requested changes
@@ -76,9 +76,41 @@ on:
76
76
  required: false
77
77
  type: string
78
78
  default: first
79
+ # The gate job that the top-level `if:` reads. gh-aw folds that `if:` into the generated
80
+ # activation job but gives activation no dependency on the job, so the reference resolves
81
+ # to '' and the clause is false -- the agent would never run. The package's own validator
82
+ # catches it after compilation; this is the line it asks for, the same one the merge gate
83
+ # uses for protected_changes.
84
+ needs: [still_open]
79
85
 
80
86
  jobs:
87
+ # A route dispatched while the issue was open must not execute after it has been closed. The
88
+ # classifier can only see `github.event.issue.state`, which is the state when the event fired
89
+ # and is absent on a workflow_dispatch, and this fleet queues for a runner for ten minutes and
90
+ # more. Numa #659 was closed one second after a comment dispatched refine; the run reached
91
+ # `reserve` thirteen minutes later and refined a closed issue to completion. Read now, once,
92
+ # and gate both the reservation and the agent on it.
93
+ still_open:
94
+ runs-on: agents-arc
95
+ permissions:
96
+ contents: read
97
+ issues: read
98
+ outputs:
99
+ open: ${{ steps.state.outputs.open }}
100
+ steps:
101
+ - name: Checkout workflow actions
102
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
103
+ with:
104
+ persist-credentials: false
105
+ - name: Read the issue state
106
+ id: state
107
+ uses: ./.github/actions/require-open-issue
108
+ with:
109
+ token: ${{ github.token }}
110
+ issue-number: ${{ inputs.issue-number }}
81
111
  reserve:
112
+ needs: [still_open]
113
+ if: needs.still_open.outputs.open == 'true'
82
114
  runs-on: agents-arc
83
115
  permissions:
84
116
  contents: read
@@ -364,7 +396,7 @@ jobs:
364
396
  ${{ env.INCOMPLETE_COMMENT }}
365
397
  [View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
366
398
 
367
- if: inputs.issue-number != ''
399
+ if: inputs.issue-number != '' && needs.still_open.outputs.open == 'true'
368
400
 
369
401
  runs-on: agents-arc
370
402
  runs-on-slim: agents-arc