@plainconceptsplatform/workflows 0.20.2 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/index.js +0 -0
  2. package/dist/stack-defaults.js +16 -16
  3. package/dist/workflow-catalog.d.ts +1 -1
  4. package/dist/workflow-catalog.js +2 -0
  5. package/loops/actions/add-issue-labels/action.yml +50 -50
  6. package/loops/actions/agent-output.cjs +17 -17
  7. package/loops/actions/apply-agent-bundle/action.yml +24 -24
  8. package/loops/actions/apply-agent-comments/action.yml +42 -42
  9. package/loops/actions/apply-agent-labels/action.yml +55 -55
  10. package/loops/actions/apply-agent-output/action.yml +108 -108
  11. package/loops/actions/classify-route/action.yml +100 -100
  12. package/loops/actions/cleanup-artifacts/action.yml +91 -91
  13. package/loops/actions/close-agent-issues/action.yml +43 -43
  14. package/loops/actions/collect-app-errors/action.yml +297 -0
  15. package/loops/actions/collect-app-errors/group-and-redact.mjs +252 -0
  16. package/loops/actions/collect-app-errors/query-app-errors.sh +104 -0
  17. package/loops/actions/create-agent-issues/action.yml +52 -52
  18. package/loops/actions/create-issue-comment/action.yml +29 -29
  19. package/loops/actions/download-agent-output/action.yml +53 -53
  20. package/loops/actions/link-pr-to-issue/action.yml +40 -40
  21. package/loops/actions/list-open-issues/action.yml +33 -33
  22. package/loops/actions/load-issue-context/action.yml +45 -45
  23. package/loops/actions/merge-agent-pr/action.yml +49 -49
  24. package/loops/actions/push-agent-branch/action.yml +45 -45
  25. package/loops/actions/remove-issue-labels/action.yml +37 -37
  26. package/loops/actions/update-agent-issues/action.yml +58 -58
  27. package/loops/actions/validate-review-output/action.yml +35 -35
  28. package/loops/actions/validate-triage-output/action.yml +36 -36
  29. package/loops/actions/verify-app-errors/action.yml +11 -0
  30. package/loops/actions/verify-app-errors/verify-app-errors.mjs +216 -0
  31. package/loops/actions/verify-composite-actions/action.yml +9 -9
  32. package/loops/actions/verify-refine-output/action.yml +9 -9
  33. package/loops/actions/verify-route-matrix/action.yml +9 -9
  34. package/loops/actions/verify-route-matrix/verify-route-matrix.sh +233 -12
  35. package/loops/scripts/compile-agent-workflows.mjs +331 -331
  36. package/loops/templates/agentics/agentics-app-errors.yml +168 -0
  37. package/loops/templates/agentics/agentics-maintenance.yml +121 -121
  38. package/loops/templates/ci/app-ci-dotnet-next.yml +330 -330
  39. package/loops/templates/ci/app-ci-node-monorepo.yml +260 -260
  40. package/loops/templates/issues/bug_report.yml +109 -109
  41. package/loops/templates/issues/feature_request.yml +75 -75
  42. package/loops/templates/release/github-release.yml +30 -30
  43. package/loops/workflows/agent-implement.md +99 -6
  44. package/loops/workflows/agent-merge-gate.md +4 -2
  45. package/loops/workflows/authorize-bot-work.yml +105 -105
  46. package/loops/workflows/shared/opencode-ci.md +2 -2
  47. package/loops/workflows/shared/platform-defaults.md +19 -19
  48. package/loops/workflows/work-router.yml +2 -2
  49. package/package.json +9 -8
@@ -0,0 +1,216 @@
1
+ // Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-app-errors/verify-app-errors.mjs. Update with workflows update --force; consumer edits may be overwritten.
2
+ //
3
+ // Executes the real grouping and redaction against rows shaped the way Application Insights
4
+ // actually returns them.
5
+ //
6
+ // It runs the shipped module rather than grepping it. The workflow-error report's logic lives
7
+ // inside a `script:` string in its action.yml, so nothing can execute it and its tests are
8
+ // greps; that was the reason to put this one in a module, and this is the payoff.
9
+
10
+ import {
11
+ FINDING_FIELDS,
12
+ MAX_FRAMES,
13
+ clip,
14
+ fingerprint,
15
+ leaks,
16
+ ownFrames,
17
+ prepare,
18
+ render,
19
+ scrub,
20
+ toFindings,
21
+ unexpectedFields,
22
+ } from "../collect-app-errors/group-and-redact.mjs";
23
+
24
+ let pass = 0;
25
+ let fail = 0;
26
+
27
+ function check(label, condition, detail = "") {
28
+ if (condition) {
29
+ pass += 1;
30
+ return;
31
+ }
32
+
33
+ fail += 1;
34
+ console.error(`FAIL: ${label}${detail ? `\n ${detail}` : ""}`);
35
+ }
36
+
37
+ function section(title) {
38
+ console.log(`\n── ${title} ──`);
39
+ }
40
+
41
+ /** A row in the shape `az monitor log-analytics query` returns, from a real exception. */
42
+ function row(overrides = {}) {
43
+ return {
44
+ ProblemId: "System.InvalidOperationException at Pliny.Application.Agents.RunExecutor.ExecuteAsync",
45
+ ExceptionType: "System.InvalidOperationException",
46
+ OperationName: "POST Runs/Start",
47
+ AppRoleName: "plinybot-pre-app-01",
48
+ Occurrences: 42,
49
+ Operations: 3,
50
+ FirstSeen: "2026-09-13T01:10:00Z",
51
+ LastSeen: "2026-09-13T06:40:00Z",
52
+ AnyOperationId: "6f1c1b0e9a2f4d55",
53
+ AnyDetails: [
54
+ {
55
+ parsedStack: [
56
+ { method: "Pliny.Application.Agents.RunExecutor.ExecuteAsync", fileName: "/home/vsts/work/1/s/src/RunExecutor.cs", line: 512 },
57
+ { method: "Microsoft.AspNetCore.Routing.EndpointMiddleware.Invoke", fileName: "/_/src/Http/Routing.cs", line: 90 },
58
+ ],
59
+ },
60
+ ],
61
+ ...overrides,
62
+ };
63
+ }
64
+
65
+ const options = { environment: "pre", lookbackHours: "24", ownCodePrefix: "Pliny" };
66
+
67
+ section("Scrubbing");
68
+
69
+ check(
70
+ "a GUID is masked",
71
+ scrub("run 6f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e failed") === "run {guid} failed",
72
+ scrub("run 6f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e failed"));
73
+
74
+ check(
75
+ "every GUID is masked, not just the first",
76
+ !/[0-9a-f]{8}-[0-9a-f]{4}/i.test(scrub("a 6f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e b 7f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e")));
77
+
78
+ check("an email address is masked", scrub("from someone@example.com") === "from {email}");
79
+
80
+ check(
81
+ "a build-machine path is masked",
82
+ clip("at /home/vsts/work/1/s/src/RunExecutor.cs line 5") === "at {path} line 5",
83
+ clip("at /home/vsts/work/1/s/src/RunExecutor.cs line 5"));
84
+
85
+ check(
86
+ "a query string is masked but the address survives",
87
+ clip("GET https://api.example.com/v1/items?token=abc123") === "GET https://api.example.com/v1/items?{query}",
88
+ clip("GET https://api.example.com/v1/items?token=abc123"));
89
+
90
+ check(
91
+ "a pattern split across a newline is still caught",
92
+ !/example\.com/.test(scrub("mail\nto")) && clip("someone@example.com\nnext") === "{email} next",
93
+ clip("someone@example.com\nnext"));
94
+
95
+ section("Frames");
96
+
97
+ check("only our own frames survive", ownFrames(row().AnyDetails, "Pliny").length === 1);
98
+
99
+ check(
100
+ "a frame carries the method and never the path",
101
+ ownFrames(row().AnyDetails, "Pliny")[0] === "Pliny.Application.Agents.RunExecutor.ExecuteAsync");
102
+
103
+ check(
104
+ "an empty prefix publishes no frames at all",
105
+ ownFrames(row().AnyDetails, "").length === 0);
106
+
107
+ check(
108
+ "frames are capped",
109
+ ownFrames(
110
+ [{ parsedStack: Array.from({ length: 40 }, (_, i) => ({ method: `Pliny.Frame${i}` })) }],
111
+ "Pliny",
112
+ ).length === MAX_FRAMES);
113
+
114
+ check("a row with no details does not throw", ownFrames(undefined, "Pliny").length === 0);
115
+
116
+ section("Fingerprints");
117
+
118
+ check(
119
+ "the same problem in the same place is the same fingerprint",
120
+ fingerprint("a", "b", "pre") === fingerprint("a", "b", "pre"));
121
+
122
+ check(
123
+ "the same problem in another environment is a different one",
124
+ fingerprint("a", "b", "pre") !== fingerprint("a", "b", "pro"));
125
+
126
+ check("a fingerprint is short enough to read", fingerprint("a", "b", "pre").length === 12);
127
+
128
+ section("Thresholds");
129
+
130
+ check(
131
+ "a finding below the floor is dropped",
132
+ toFindings([row({ Occurrences: 2 })], { ...options, minOccurrences: 5 }).length === 0);
133
+
134
+ check(
135
+ "a finding on the floor is kept",
136
+ toFindings([row({ Occurrences: 5 })], { ...options, minOccurrences: 5 }).length === 1);
137
+
138
+ check(
139
+ "the loudest comes first",
140
+ toFindings([row({ Occurrences: 5, ProblemId: "quiet" }), row({ Occurrences: 99, ProblemId: "loud" })], options)[0]
141
+ .problemId === "loud");
142
+
143
+ section("The field allowlist");
144
+
145
+ check(
146
+ "a finding carries exactly the agreed fields",
147
+ unexpectedFields(toFindings([row()], options)[0]).length === 0);
148
+
149
+ check(
150
+ "a field nobody agreed to publish is named",
151
+ unexpectedFields({ ...toFindings([row()], options)[0], runId: "6f1c1b0e" }).join() === "runId");
152
+
153
+ check(
154
+ "an ordinary batch is not refused",
155
+ prepare([row()], options).refused === "" && prepare([row()], options).prepared.length === 1);
156
+
157
+ check(
158
+ "run_id is not a reportable field",
159
+ !FINDING_FIELDS.includes("runId") && !FINDING_FIELDS.includes("run_id"));
160
+
161
+ section("Rendering and the scanner");
162
+
163
+ const rendered = render(toFindings([row()], options)[0], options);
164
+
165
+ check("the marker is the first line", rendered.body.startsWith("<!-- pcp-app-error: "));
166
+
167
+ check("the marker carries the fingerprint", rendered.body.includes(toFindings([row()], options)[0].fingerprint));
168
+
169
+ check("the count is said plainly", rendered.body.includes("**42 occurrence(s)**"));
170
+
171
+ check("a clean report trips nothing", leaks(`${rendered.title}\n${rendered.body}`).length === 0,
172
+ leaks(`${rendered.title}\n${rendered.body}`).join("; "));
173
+
174
+ check(
175
+ "no build-machine path reaches the body",
176
+ !rendered.body.includes("/home/vsts"));
177
+
178
+ // Two ways to have no frames, and they ask the reader for different things. Pliny-Bot #191 was
179
+ // filed with OWN_CODE_PREFIX set to "Pliny" and still said the repository had not said which
180
+ // assemblies were its own -- it sent a reader to configure a knob that had been set the day
181
+ // before. Only the unconfigured case was covered here, which is why the wording survived.
182
+ check(
183
+ "no prefix configured says which knob to set",
184
+ render(toFindings([row()], { ...options, ownCodePrefix: "" })[0], { ...options, ownCodePrefix: "" })
185
+ .body.includes("has not said which assemblies are its own"));
186
+
187
+ check(
188
+ "a prefix that matched nothing does not claim the prefix is missing",
189
+ !render(toFindings([row()], { ...options, ownCodePrefix: "NoSuchAssembly" })[0], { ...options, ownCodePrefix: "NoSuchAssembly" })
190
+ .body.includes("has not said which assemblies are its own"));
191
+
192
+ check(
193
+ "a prefix that matched nothing names the prefix it tried",
194
+ render(toFindings([row()], { ...options, ownCodePrefix: "NoSuchAssembly" })[0], { ...options, ownCodePrefix: "NoSuchAssembly" })
195
+ .body.includes("No frame in this exception belongs to `NoSuchAssembly`"));
196
+
197
+ section("Fail-closed, but not fail-dead");
198
+
199
+ const mixed = prepare(
200
+ [row(), row({ ProblemId: "Secret", ExceptionType: "AccountKey=abc123def456;Other" })],
201
+ options);
202
+
203
+ check("the clean report is still prepared", mixed.prepared.length === 1, JSON.stringify(mixed.withheld));
204
+
205
+ check("the tripped report is withheld", mixed.withheld.length === 1);
206
+
207
+ check("the caller is told what matched", (mixed.withheld[0]?.matched ?? []).length > 0);
208
+
209
+ check("a withheld report does not refuse the run", mixed.refused === "");
210
+
211
+ check("nothing is prepared from no rows", prepare([], options).prepared.length === 0);
212
+
213
+ check("a non-array does not throw", prepare(null, options).prepared.length === 0);
214
+
215
+ console.log(`\n${pass} passed, ${fail} failed`);
216
+ process.exit(fail > 0 ? 1 : 0);
@@ -1,9 +1,9 @@
1
- # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-composite-actions/action.yml. Update with workflows update --force; consumer edits may be overwritten.
2
- name: Verify composite actions
3
- description: Check every local composite action manifest parses and uses only contexts a composite action actually has.
4
- runs:
5
- using: composite
6
- steps:
7
- - name: Validate composite action manifests
8
- shell: bash
9
- run: bash "${GITHUB_ACTION_PATH}/verify-composite-actions.sh"
1
+ # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-composite-actions/action.yml. Update with workflows update --force; consumer edits may be overwritten.
2
+ name: Verify composite actions
3
+ description: Check every local composite action manifest parses and uses only contexts a composite action actually has.
4
+ runs:
5
+ using: composite
6
+ steps:
7
+ - name: Validate composite action manifests
8
+ shell: bash
9
+ run: bash "${GITHUB_ACTION_PATH}/verify-composite-actions.sh"
@@ -1,9 +1,9 @@
1
- # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-refine-output/action.yml. Update with workflows update --force; consumer edits may be overwritten.
2
- name: Verify refine output validation
3
- description: Exercise deterministic Refine output validation regressions.
4
- runs:
5
- using: composite
6
- steps:
7
- - name: Verify refinement outcomes
8
- shell: bash
9
- run: bash "${{ github.action_path }}/verify-refine-output.sh"
1
+ # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-refine-output/action.yml. Update with workflows update --force; consumer edits may be overwritten.
2
+ name: Verify refine output validation
3
+ description: Exercise deterministic Refine output validation regressions.
4
+ runs:
5
+ using: composite
6
+ steps:
7
+ - name: Verify refinement outcomes
8
+ shell: bash
9
+ run: bash "${{ github.action_path }}/verify-refine-output.sh"
@@ -1,9 +1,9 @@
1
- # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-route-matrix/action.yml. Update with workflows update --force; consumer edits may be overwritten.
2
- name: Verify route matrix
3
- description: Exercise the router's classifier against every supported event, and check that each route and dispatch operation has a job in work-router.yml.
4
- runs:
5
- using: composite
6
- steps:
7
- - name: Run route matrix verification
8
- shell: bash
9
- run: bash "${GITHUB_ACTION_PATH}/verify-route-matrix.sh"
1
+ # Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-route-matrix/action.yml. Update with workflows update --force; consumer edits may be overwritten.
2
+ name: Verify route matrix
3
+ description: Exercise the router's classifier against every supported event, and check that each route and dispatch operation has a job in work-router.yml.
4
+ runs:
5
+ using: composite
6
+ steps:
7
+ - name: Run route matrix verification
8
+ shell: bash
9
+ run: bash "${GITHUB_ACTION_PATH}/verify-route-matrix.sh"
@@ -372,6 +372,52 @@ else
372
372
  fi
373
373
  if [ "$MIRROR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
374
374
 
375
+ # Third-party action pins are the same shape of problem. Every `uses:` in a package-managed
376
+ # file is pinned to a full commit SHA, and one action must resolve to exactly one SHA across
377
+ # the whole package: the router, the workers, the shared mechanics and the composite actions
378
+ # are installed together, so two pins for one action mean two versions of it run in the same
379
+ # repository, and the older one is invisible until it breaks. That is how the merge-belt
380
+ # dispatchers ran create-github-app-token v2.0.6 for a release while every worker ran v3.2.0:
381
+ # nothing failed, nothing flagged it, and the action that receives the App's private key was
382
+ # the stale one. Templates are excluded: they are consumer-owned from installation and
383
+ # deliberately carry their own versions (the CI templates pin download-artifact v6 while the
384
+ # loops pin v8). A floating tag (`@v4`) is a missing pin and fails the same check.
385
+ PIN_OK=1
386
+ declare -A PIN_SEEN
387
+ PKG_ROOT="$(cd "${HERE}/../../.." && pwd)"
388
+ pin_scan() {
389
+ local file="$1" line action ref
390
+ # Resolve to an absolute path so the package-relative name in the failure reads like the
391
+ # source tree, not like this script's location.
392
+ file="$(cd "$(dirname "$file")" && pwd)/$(basename "$file")"
393
+ local shown="${file#"${PKG_ROOT}/"}"
394
+ # Anchored on the line start so a comment quoting a `uses:` cannot match, and tolerant of
395
+ # both spellings (`uses:` as a mapping key and `- uses:` as a list item). Local composite
396
+ # actions (`./.github/actions/...`) and docker refs carry no `@` and never match.
397
+ local pin_re='^[[:space:]]*-?[[:space:]]*uses:[[:space:]]+([A-Za-z0-9_.-]+/[A-Za-z0-9_./-]+)@([A-Za-z0-9._-]+)'
398
+ while IFS= read -r line; do
399
+ [[ "$line" =~ $pin_re ]] || continue
400
+ action="${BASH_REMATCH[1]}"
401
+ ref="${BASH_REMATCH[2]}"
402
+ if [ "${PIN_SEEN[$action]+set}" ] && [ "${PIN_SEEN[$action]}" != "$ref" ]; then
403
+ PIN_OK=0
404
+ echo "FAIL: ${shown} pins ${action}@${ref} but ${PIN_SEEN[$action]} is pinned elsewhere; one action, one pin" >&2
405
+ else
406
+ PIN_SEEN[$action]="$ref"
407
+ fi
408
+ if [ "${#ref}" -ne 40 ]; then
409
+ PIN_OK=0
410
+ echo "FAIL: ${shown} uses ${action}@${ref}; a tag is mutable, pin the full commit SHA" >&2
411
+ fi
412
+ done < "$file"
413
+ }
414
+ for pin_file in "$ROUTER_YML" "${WORKFLOWS_DIR}"/agent-*.md "${WORKFLOWS_DIR}"/shared/*.md \
415
+ "${WORKFLOWS_DIR}"/authorize-bot-work.yml "${HERE}/.."/*/action.yml; do
416
+ [ -f "$pin_file" ] || continue
417
+ pin_scan "$pin_file"
418
+ done
419
+ if [ "$PIN_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
420
+
375
421
  # Run the belt's own jq, rather than reading it. The filter that picks which open pull requests
376
422
  # the hourly reconcile job acts on was written as
377
423
  # ((env.TRUSTED_BOTS | split(" ")) | index(.user.login) != null)
@@ -841,19 +887,16 @@ if worker_installed implement; then
841
887
  PUSH_FALLBACK_OK=0
842
888
  echo "FAIL: implement gates a no-pull-request path on code_push_failure_count, which is 0 when gh-aw files the patch as an issue" >&2
843
889
  fi
844
- for needed in 'PUSH_CONFLICT_COMMENT' 'NO_PULL_REQUEST_COMMENT'; do
845
- grep -qF "env.${needed}" "$IMPLEMENT_WORKER_MD" || {
846
- PUSH_FALLBACK_OK=0
847
- echo "FAIL: implement no longer says env.${needed} on any path" >&2
848
- }
849
- done
850
- # Collapsing them back into one message is the regression this guards: each is defined once in
851
- # the env block and printed on exactly one path, so two usages of either means the conditions
852
- # have been merged or duplicated.
853
- if [ "$(count -cF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ] ||
854
- [ "$(count -cF 'env.NO_PULL_REQUEST_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
890
+ grep -qF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD" || {
891
+ PUSH_FALLBACK_OK=0
892
+ echo "FAIL: implement no longer says env.PUSH_CONFLICT_COMMENT on any path" >&2
893
+ }
894
+ # Collapsing this back into whatever covers the other endings is the regression this guards: it
895
+ # is defined once in the env block and printed on exactly one path. The four messages it used to
896
+ # be paired with are asserted together under "Ways a run ends with nothing".
897
+ if [ "$(count -cF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
855
898
  PUSH_FALLBACK_OK=0
856
- echo "FAIL: implement should print each no-pull-request message on exactly one path" >&2
899
+ echo "FAIL: implement should print the conflicted-push message on exactly one path" >&2
857
900
  fi
858
901
  # And the two paths must be mutually exclusive, or a conflicted push gets both comments.
859
902
  if [ "$(count -cF "process_safe_outputs_items_succeeded != '0'" "$IMPLEMENT_WORKER_MD")" -lt 2 ] ||
@@ -1202,6 +1245,31 @@ echo "── Merge gate validator ───────────────
1202
1245
  # reasoning alone, and the worker has not run in production since, so these fixtures are the only
1203
1246
  # evidence the change is right. Executing the real script is the same technique that finally
1204
1247
  # caught the belt's jq bug, which every reading assertion had walked past.
1248
+ # The gate comments on every bot pull request, so anything it writes there is written dozens of
1249
+ # times a week. A CODEOWNERS of `* @someone` matches every pull request, so the owners row was
1250
+ # naming four people on routine auto-merges and @-mentioning each of them, which notifies them.
1251
+ # The row belongs to the disposition that actually wants an owner, and the names go in a code
1252
+ # span so the comment reports who owns the area without pinging them: on owner-review the gate
1253
+ # requests the review properly, and that is the notification.
1254
+ if worker_installed merge-gate; then
1255
+ OWNERS_OK=1
1256
+ if ! grep -q "outcome == 'owner-review'" "$MERGE_GATE_WORKER_MD"; then
1257
+ OWNERS_OK=0
1258
+ echo "FAIL: the gate comment names required owners unconditionally; gate the row on owner-review" >&2
1259
+ fi
1260
+ # Every rendering of required_owners has to sit inside a code span.
1261
+ while IFS= read -r line; do
1262
+ case "$line" in
1263
+ *'`{0}`'*|*'`${{ needs.protected_changes.outputs.required_owners'*) continue ;;
1264
+ *)
1265
+ OWNERS_OK=0
1266
+ echo "FAIL: required_owners is rendered without a code span, so the gate @-mentions people: ${line# }" >&2
1267
+ ;;
1268
+ esac
1269
+ done < <(grep -F 'required_owners' "$MERGE_GATE_WORKER_MD" | grep -vE '^\s*(#|[a-z_]+:)' | grep -F 'Required owners')
1270
+ if [ "$OWNERS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1271
+ fi
1272
+
1205
1273
  # Blast radius is the input the merge decision leans on hardest, and it is the one a reader
1206
1274
  # cannot check by eye. These cases are the six real pull requests the redesign was measured
1207
1275
  # against, reduced to their shape: the three that used to be parked for a person purely because
@@ -1872,6 +1940,71 @@ if worker_installed audit; then
1872
1940
  if [ "$AUDIT_EMPTY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
1873
1941
  fi
1874
1942
 
1943
+ echo "── Ways a run ends with nothing ──────────────────────────────────────────"
1944
+
1945
+ # A run can end without a pull request in five ways and they need different answers. One sentence
1946
+ # covered all of them, and three issues in two days said "nothing landed": one agent gave up in a
1947
+ # minute, one did more work than the run that succeeded the same hour and never asked for a pull
1948
+ # request, one was told there was nothing to do. Nobody could act on any of them, and the retry
1949
+ # that followed was right for some and wasted on the rest.
1950
+ #
1951
+ # Asserted because collapsing two of these back together is a one-line edit that turns every run
1952
+ # green again while saying less.
1953
+ if worker_installed implement; then
1954
+ ENDINGS_OK=1
1955
+ ENDING_MESSAGES="PUSH_CONFLICT_COMMENT OUTPUT_NOT_APPLIED_COMMENT NOOP_COMMENT PATCH_WITHOUT_REQUEST_COMMENT NO_OUTPUT_COMMENT"
1956
+
1957
+ for message in $ENDING_MESSAGES; do
1958
+ # Defined once in the env block and printed on exactly one path. Two usages means two
1959
+ # branches now say the same thing, which is the collapse this guards.
1960
+ if [ "$(count -cF "env.${message}" "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
1961
+ ENDINGS_OK=0
1962
+ echo "FAIL: implement should print ${message} on exactly one path" >&2
1963
+ fi
1964
+ done
1965
+
1966
+ # The sentence that used to cover all five. Its text survives as NO_OUTPUT_COMMENT; the name
1967
+ # coming back means the branches were merged again.
1968
+ if [ "$(count -cF 'NO_PULL_REQUEST_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
1969
+ ENDINGS_OK=0
1970
+ echo "FAIL: implement still carries NO_PULL_REQUEST_COMMENT, which said the same thing about five different endings" >&2
1971
+ fi
1972
+
1973
+ # What tells them apart. Both are the agent job's own outputs, and gh-aw gates on the first
1974
+ # itself, so losing them means guessing again.
1975
+ for signal in 'needs.agent.outputs.output_types' 'needs.agent.outputs.has_patch'; do
1976
+ if [ "$(count -cE "^ *if:.*${signal}" "$IMPLEMENT_WORKER_MD")" -eq 0 ]; then
1977
+ ENDINGS_OK=0
1978
+ echo "FAIL: implement no longer reads ${signal}, so it cannot tell its no-pull-request endings apart" >&2
1979
+ fi
1980
+ done
1981
+
1982
+ # `!x == 'true'` parses as `(!x) == 'true'` in a GitHub expression, which compares a boolean to
1983
+ # a string and is always false. Written wrong here first, and it would have made the
1984
+ # produced-nothing branch unreachable while every other branch still looked right.
1985
+ if [ "$(count -cF "!needs.agent.outputs.has_patch ==" "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
1986
+ ENDINGS_OK=0
1987
+ echo "FAIL: implement negates has_patch as \`!x == 'true'\`, which binds as \`(!x) == 'true'\` and never matches; use != 'true'" >&2
1988
+ fi
1989
+
1990
+ # A bot that reported nothing to do made a decision. The janitor retries a stalled issue, and
1991
+ # re-running it produces the same answer, so this one path must not be flagged.
1992
+ noop_block=$(awk "/name: Say there was nothing to do/{found=1} found{print} found && /View this workflow run/{exit}" "$IMPLEMENT_WORKER_MD")
1993
+ if [ -n "$noop_block" ] && printf '%s' "$noop_block" | grep -qF 'STALLED_LABEL'; then
1994
+ ENDINGS_OK=0
1995
+ echo "FAIL: implement flags a deliberate noop as stalled, so the janitor will retry a decision until it runs out of attempts" >&2
1996
+ fi
1997
+
1998
+ # The usage artifact carrying the token counts is uploaded by gh-aw's conclusion job, and that
1999
+ # job already depends on conclude. Naming it is a cycle and the worker stops compiling.
2000
+ if [ "$(count -cE "^ *needs:.*conclusion" "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
2001
+ ENDINGS_OK=0
2002
+ echo "FAIL: implement depends on the conclusion job, which depends on conclude; that cycle does not compile" >&2
2003
+ fi
2004
+
2005
+ if [ "$ENDINGS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
2006
+ fi
2007
+
1875
2008
  echo "── Runner pools ──────────────────────────────────────────────────────────"
1876
2009
 
1877
2010
  # Where every job runs, stated once and asserted, because GitHub gives a wrong pool no error: a
@@ -2060,6 +2193,94 @@ fi
2060
2193
  if [ "$GATE_RUN_ID_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
2061
2194
  fi
2062
2195
 
2196
+ echo "── App error privacy ─────────────────────────────────────────────────────"
2197
+
2198
+ # The application error report files into the private repository rather than out of it, so
2199
+ # its danger is the opposite one: not that a name escapes, but that an identifier joining a
2200
+ # stack trace back to one person's conversation gets written down beside it.
2201
+ #
2202
+ # Four properties hold that line and each is a line an edit could drop with nothing going red.
2203
+ # Asserted the way the error report's are: call sites and set equality, never a bare symbol,
2204
+ # because a grep for a name passes against a guard that has been renamed rather than removed.
2205
+ APP_ERRORS_MJS="${HERE}/../collect-app-errors/group-and-redact.mjs"
2206
+ APP_ERRORS_SH="${HERE}/../collect-app-errors/query-app-errors.sh"
2207
+ APP_ERRORS_YML="${HERE}/../collect-app-errors/action.yml"
2208
+
2209
+ if [ -f "$APP_ERRORS_MJS" ]; then
2210
+ AE_OK=1
2211
+ ae() {
2212
+ grep -qE "$1" "$2" || { AE_OK=0; echo "FAIL: collect-app-errors ${3}" >&2; }
2213
+ }
2214
+
2215
+ # 1. The query reads the two tables the application's own telemetry goes to, and no others.
2216
+ # AppServiceConsoleLogs and ContainerAppConsoleLogs live in the same workspace and carry raw
2217
+ # engine stdout, which the no-content rule does not govern. A union across them is an
2218
+ # incident rather than a wider query.
2219
+ #
2220
+ # AppTraces is deliberate and was added after the first dry run against a real workspace
2221
+ # returned nought exceptions on a service that had been failing all morning: nothing in
2222
+ # these applications calls RecordException, so every failure they care about is caught in
2223
+ # code, logged through ILogger, and lands in AppTraces alone.
2224
+ ae '^ AppExceptions$' "$APP_ERRORS_SH" 'does not query AppExceptions by name'
2225
+ ae '^ AppTraces$' "$APP_ERRORS_SH" 'does not query AppTraces, where caught failures land'
2226
+ # The heredoc alone. The comment above it names the console tables in order to say they are
2227
+ # never read, and a grep over the whole file cannot tell the warning from the offence.
2228
+ query_body="$(sed -n '/^read -r -d .. QUERY <<KQL/,/^KQL$/p' "$APP_ERRORS_SH")"
2229
+ if printf '%s' "$query_body" | grep -qE 'ConsoleLogs|AppRequests|AppDependencies|union \*|search '; then
2230
+ AE_OK=0
2231
+ echo "FAIL: collect-app-errors reads a table it has no business reading" >&2
2232
+ fi
2233
+
2234
+ # 2. Counts are scaled for sampling. A pre environment samples at 0.3, so count() reads a
2235
+ # problem that happened two hundred times as sixty, and it falls under a floor set to catch
2236
+ # it. This is a correctness guard that looks like a style one.
2237
+ ae 'sum\(ItemCount\)' "$APP_ERRORS_SH" 'counts rows instead of summing ItemCount'
2238
+
2239
+ # 3. The field allowlist is what actually holds the line, because a redaction rule filters
2240
+ # values and cannot see a field somebody adds next year. Compared as a set, so reordering is
2241
+ # fine and an addition is not. run_id is the one that matters: it is unhashed on this
2242
+ # telemetry and joins to a conversation and to a person.
2243
+ expected_fields="exceptionType frames fingerprint firstSeen lastSeen occurrences operationName operations problemId roleName"
2244
+ actual_fields="$(sed -n '/export const FINDING_FIELDS = \[/,/\];/p' "$APP_ERRORS_MJS" |
2245
+ grep -oE '"[a-zA-Z]+"' | tr -d '"' | sort | tr '\n' ' ' | sed 's/ $//')"
2246
+ if [ "$actual_fields" = "$(echo "$expected_fields" | tr ' ' '\n' | sort | tr '\n' ' ' | sed 's/ $//')" ]; then
2247
+ PASS=$((PASS + 1))
2248
+ else
2249
+ FAIL=$((FAIL + 1))
2250
+ echo "FAIL: collect-app-errors reportable fields changed: ${actual_fields}" >&2
2251
+ fi
2252
+ # Plain grep, not the count() helper: that is `grep || true`, so an `if count -q` is true
2253
+ # whatever it finds and the guard never fires. Three of these were written that way first
2254
+ # and all three failed open, which is the same shape of bug this file exists to catch.
2255
+ if sed -n '/export const FINDING_FIELDS = \[/,/\];/p' "$APP_ERRORS_MJS" | grep -qiE 'run_?id'; then
2256
+ AE_OK=0
2257
+ echo "FAIL: collect-app-errors lists a run id as reportable; it joins telemetry to a person" >&2
2258
+ fi
2259
+
2260
+ # 4. Scrub, then scan, then refuse. Scrubbing masks what is common and safely replaceable so
2261
+ # one unlucky operation name does not leave the job red and filing nothing forever; the
2262
+ # scanner is the backstop for whatever that missed; and a withheld report still turns the
2263
+ # run red so the field that carried it gets fixed.
2264
+ ae 'export const SCRUBS = \[' "$APP_ERRORS_MJS" 'declares no scrubbing'
2265
+ ae 'export const LEAK_CHECKS = \[' "$APP_ERRORS_MJS" 'declares no leak scanner'
2266
+ ae 'core\.setFailed.*withheld by the leak scanner' "$APP_ERRORS_YML" 'does not go red on a withheld report'
2267
+ ae 'not in the reportable field list' "$APP_ERRORS_MJS" 'does not refuse an unlisted field'
2268
+ ae 'core\.setFailed\(`Nothing was filed' "$APP_ERRORS_YML" 'does not stop when the module refuses'
2269
+
2270
+ # A GUID must be masked on the way in, not merely detected on the way out. Both, in fact:
2271
+ # the scrub is what keeps the job alive and the check is what keeps it honest.
2272
+ ae '\{guid\}' "$APP_ERRORS_MJS" 'does not mask GUIDs'
2273
+
2274
+ # 5. No model. This is the same assertion the error report carries, for the same reason: a
2275
+ # model summarising a stack trace paraphrases it, and a paraphrase is a wrong issue.
2276
+ if grep -qiE '(engine:|opencode|safe-outputs|OPENAI_API_KEY)' "$APP_ERRORS_YML"; then
2277
+ AE_OK=0
2278
+ echo "FAIL: collect-app-errors has grown a model" >&2
2279
+ fi
2280
+
2281
+ if [ "$AE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
2282
+ fi
2283
+
2063
2284
  echo
2064
2285
  if [ "$FAIL" -eq 0 ]; then
2065
2286
  echo "Route matrix: ${PASS} passed"