@plainconceptsplatform/workflows 0.20.2 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +0 -0
- package/dist/stack-defaults.js +16 -16
- package/dist/workflow-catalog.d.ts +1 -1
- package/dist/workflow-catalog.js +2 -0
- package/loops/actions/add-issue-labels/action.yml +50 -50
- package/loops/actions/agent-output.cjs +17 -17
- package/loops/actions/apply-agent-bundle/action.yml +24 -24
- package/loops/actions/apply-agent-comments/action.yml +42 -42
- package/loops/actions/apply-agent-labels/action.yml +55 -55
- package/loops/actions/apply-agent-output/action.yml +108 -108
- package/loops/actions/classify-route/action.yml +100 -100
- package/loops/actions/cleanup-artifacts/action.yml +91 -91
- package/loops/actions/close-agent-issues/action.yml +43 -43
- package/loops/actions/collect-app-errors/action.yml +297 -0
- package/loops/actions/collect-app-errors/group-and-redact.mjs +252 -0
- package/loops/actions/collect-app-errors/query-app-errors.sh +104 -0
- package/loops/actions/create-agent-issues/action.yml +52 -52
- package/loops/actions/create-issue-comment/action.yml +29 -29
- package/loops/actions/download-agent-output/action.yml +53 -53
- package/loops/actions/link-pr-to-issue/action.yml +40 -40
- package/loops/actions/list-open-issues/action.yml +33 -33
- package/loops/actions/load-issue-context/action.yml +45 -45
- package/loops/actions/merge-agent-pr/action.yml +49 -49
- package/loops/actions/push-agent-branch/action.yml +45 -45
- package/loops/actions/remove-issue-labels/action.yml +37 -37
- package/loops/actions/update-agent-issues/action.yml +58 -58
- package/loops/actions/validate-review-output/action.yml +35 -35
- package/loops/actions/validate-triage-output/action.yml +36 -36
- package/loops/actions/verify-app-errors/action.yml +11 -0
- package/loops/actions/verify-app-errors/verify-app-errors.mjs +216 -0
- package/loops/actions/verify-composite-actions/action.yml +9 -9
- package/loops/actions/verify-refine-output/action.yml +9 -9
- package/loops/actions/verify-route-matrix/action.yml +9 -9
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +233 -12
- package/loops/scripts/compile-agent-workflows.mjs +331 -331
- package/loops/templates/agentics/agentics-app-errors.yml +168 -0
- package/loops/templates/agentics/agentics-maintenance.yml +121 -121
- package/loops/templates/ci/app-ci-dotnet-next.yml +330 -330
- package/loops/templates/ci/app-ci-node-monorepo.yml +260 -260
- package/loops/templates/issues/bug_report.yml +109 -109
- package/loops/templates/issues/feature_request.yml +75 -75
- package/loops/templates/release/github-release.yml +30 -30
- package/loops/workflows/agent-implement.md +99 -6
- package/loops/workflows/agent-merge-gate.md +4 -2
- package/loops/workflows/authorize-bot-work.yml +105 -105
- package/loops/workflows/shared/opencode-ci.md +2 -2
- package/loops/workflows/shared/platform-defaults.md +19 -19
- package/loops/workflows/work-router.yml +2 -2
- package/package.json +9 -8
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
// Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-app-errors/verify-app-errors.mjs. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
+
//
|
|
3
|
+
// Executes the real grouping and redaction against rows shaped the way Application Insights
|
|
4
|
+
// actually returns them.
|
|
5
|
+
//
|
|
6
|
+
// It runs the shipped module rather than grepping it. The workflow-error report's logic lives
|
|
7
|
+
// inside a `script:` string in its action.yml, so nothing can execute it and its tests are
|
|
8
|
+
// greps; that was the reason to put this one in a module, and this is the payoff.
|
|
9
|
+
|
|
10
|
+
import {
|
|
11
|
+
FINDING_FIELDS,
|
|
12
|
+
MAX_FRAMES,
|
|
13
|
+
clip,
|
|
14
|
+
fingerprint,
|
|
15
|
+
leaks,
|
|
16
|
+
ownFrames,
|
|
17
|
+
prepare,
|
|
18
|
+
render,
|
|
19
|
+
scrub,
|
|
20
|
+
toFindings,
|
|
21
|
+
unexpectedFields,
|
|
22
|
+
} from "../collect-app-errors/group-and-redact.mjs";
|
|
23
|
+
|
|
24
|
+
let pass = 0;
|
|
25
|
+
let fail = 0;
|
|
26
|
+
|
|
27
|
+
function check(label, condition, detail = "") {
|
|
28
|
+
if (condition) {
|
|
29
|
+
pass += 1;
|
|
30
|
+
return;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
fail += 1;
|
|
34
|
+
console.error(`FAIL: ${label}${detail ? `\n ${detail}` : ""}`);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function section(title) {
|
|
38
|
+
console.log(`\n── ${title} ──`);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** A row in the shape `az monitor log-analytics query` returns, from a real exception. */
|
|
42
|
+
function row(overrides = {}) {
|
|
43
|
+
return {
|
|
44
|
+
ProblemId: "System.InvalidOperationException at Pliny.Application.Agents.RunExecutor.ExecuteAsync",
|
|
45
|
+
ExceptionType: "System.InvalidOperationException",
|
|
46
|
+
OperationName: "POST Runs/Start",
|
|
47
|
+
AppRoleName: "plinybot-pre-app-01",
|
|
48
|
+
Occurrences: 42,
|
|
49
|
+
Operations: 3,
|
|
50
|
+
FirstSeen: "2026-09-13T01:10:00Z",
|
|
51
|
+
LastSeen: "2026-09-13T06:40:00Z",
|
|
52
|
+
AnyOperationId: "6f1c1b0e9a2f4d55",
|
|
53
|
+
AnyDetails: [
|
|
54
|
+
{
|
|
55
|
+
parsedStack: [
|
|
56
|
+
{ method: "Pliny.Application.Agents.RunExecutor.ExecuteAsync", fileName: "/home/vsts/work/1/s/src/RunExecutor.cs", line: 512 },
|
|
57
|
+
{ method: "Microsoft.AspNetCore.Routing.EndpointMiddleware.Invoke", fileName: "/_/src/Http/Routing.cs", line: 90 },
|
|
58
|
+
],
|
|
59
|
+
},
|
|
60
|
+
],
|
|
61
|
+
...overrides,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
const options = { environment: "pre", lookbackHours: "24", ownCodePrefix: "Pliny" };
|
|
66
|
+
|
|
67
|
+
section("Scrubbing");
|
|
68
|
+
|
|
69
|
+
check(
|
|
70
|
+
"a GUID is masked",
|
|
71
|
+
scrub("run 6f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e failed") === "run {guid} failed",
|
|
72
|
+
scrub("run 6f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e failed"));
|
|
73
|
+
|
|
74
|
+
check(
|
|
75
|
+
"every GUID is masked, not just the first",
|
|
76
|
+
!/[0-9a-f]{8}-[0-9a-f]{4}/i.test(scrub("a 6f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e b 7f1c1b0e-9a2f-4d55-8b3e-1f2a3b4c5d6e")));
|
|
77
|
+
|
|
78
|
+
check("an email address is masked", scrub("from someone@example.com") === "from {email}");
|
|
79
|
+
|
|
80
|
+
check(
|
|
81
|
+
"a build-machine path is masked",
|
|
82
|
+
clip("at /home/vsts/work/1/s/src/RunExecutor.cs line 5") === "at {path} line 5",
|
|
83
|
+
clip("at /home/vsts/work/1/s/src/RunExecutor.cs line 5"));
|
|
84
|
+
|
|
85
|
+
check(
|
|
86
|
+
"a query string is masked but the address survives",
|
|
87
|
+
clip("GET https://api.example.com/v1/items?token=abc123") === "GET https://api.example.com/v1/items?{query}",
|
|
88
|
+
clip("GET https://api.example.com/v1/items?token=abc123"));
|
|
89
|
+
|
|
90
|
+
check(
|
|
91
|
+
"a pattern split across a newline is still caught",
|
|
92
|
+
!/example\.com/.test(scrub("mail\nto")) && clip("someone@example.com\nnext") === "{email} next",
|
|
93
|
+
clip("someone@example.com\nnext"));
|
|
94
|
+
|
|
95
|
+
section("Frames");
|
|
96
|
+
|
|
97
|
+
check("only our own frames survive", ownFrames(row().AnyDetails, "Pliny").length === 1);
|
|
98
|
+
|
|
99
|
+
check(
|
|
100
|
+
"a frame carries the method and never the path",
|
|
101
|
+
ownFrames(row().AnyDetails, "Pliny")[0] === "Pliny.Application.Agents.RunExecutor.ExecuteAsync");
|
|
102
|
+
|
|
103
|
+
check(
|
|
104
|
+
"an empty prefix publishes no frames at all",
|
|
105
|
+
ownFrames(row().AnyDetails, "").length === 0);
|
|
106
|
+
|
|
107
|
+
check(
|
|
108
|
+
"frames are capped",
|
|
109
|
+
ownFrames(
|
|
110
|
+
[{ parsedStack: Array.from({ length: 40 }, (_, i) => ({ method: `Pliny.Frame${i}` })) }],
|
|
111
|
+
"Pliny",
|
|
112
|
+
).length === MAX_FRAMES);
|
|
113
|
+
|
|
114
|
+
check("a row with no details does not throw", ownFrames(undefined, "Pliny").length === 0);
|
|
115
|
+
|
|
116
|
+
section("Fingerprints");
|
|
117
|
+
|
|
118
|
+
check(
|
|
119
|
+
"the same problem in the same place is the same fingerprint",
|
|
120
|
+
fingerprint("a", "b", "pre") === fingerprint("a", "b", "pre"));
|
|
121
|
+
|
|
122
|
+
check(
|
|
123
|
+
"the same problem in another environment is a different one",
|
|
124
|
+
fingerprint("a", "b", "pre") !== fingerprint("a", "b", "pro"));
|
|
125
|
+
|
|
126
|
+
check("a fingerprint is short enough to read", fingerprint("a", "b", "pre").length === 12);
|
|
127
|
+
|
|
128
|
+
section("Thresholds");
|
|
129
|
+
|
|
130
|
+
check(
|
|
131
|
+
"a finding below the floor is dropped",
|
|
132
|
+
toFindings([row({ Occurrences: 2 })], { ...options, minOccurrences: 5 }).length === 0);
|
|
133
|
+
|
|
134
|
+
check(
|
|
135
|
+
"a finding on the floor is kept",
|
|
136
|
+
toFindings([row({ Occurrences: 5 })], { ...options, minOccurrences: 5 }).length === 1);
|
|
137
|
+
|
|
138
|
+
check(
|
|
139
|
+
"the loudest comes first",
|
|
140
|
+
toFindings([row({ Occurrences: 5, ProblemId: "quiet" }), row({ Occurrences: 99, ProblemId: "loud" })], options)[0]
|
|
141
|
+
.problemId === "loud");
|
|
142
|
+
|
|
143
|
+
section("The field allowlist");
|
|
144
|
+
|
|
145
|
+
check(
|
|
146
|
+
"a finding carries exactly the agreed fields",
|
|
147
|
+
unexpectedFields(toFindings([row()], options)[0]).length === 0);
|
|
148
|
+
|
|
149
|
+
check(
|
|
150
|
+
"a field nobody agreed to publish is named",
|
|
151
|
+
unexpectedFields({ ...toFindings([row()], options)[0], runId: "6f1c1b0e" }).join() === "runId");
|
|
152
|
+
|
|
153
|
+
check(
|
|
154
|
+
"an ordinary batch is not refused",
|
|
155
|
+
prepare([row()], options).refused === "" && prepare([row()], options).prepared.length === 1);
|
|
156
|
+
|
|
157
|
+
check(
|
|
158
|
+
"run_id is not a reportable field",
|
|
159
|
+
!FINDING_FIELDS.includes("runId") && !FINDING_FIELDS.includes("run_id"));
|
|
160
|
+
|
|
161
|
+
section("Rendering and the scanner");
|
|
162
|
+
|
|
163
|
+
const rendered = render(toFindings([row()], options)[0], options);
|
|
164
|
+
|
|
165
|
+
check("the marker is the first line", rendered.body.startsWith("<!-- pcp-app-error: "));
|
|
166
|
+
|
|
167
|
+
check("the marker carries the fingerprint", rendered.body.includes(toFindings([row()], options)[0].fingerprint));
|
|
168
|
+
|
|
169
|
+
check("the count is said plainly", rendered.body.includes("**42 occurrence(s)**"));
|
|
170
|
+
|
|
171
|
+
check("a clean report trips nothing", leaks(`${rendered.title}\n${rendered.body}`).length === 0,
|
|
172
|
+
leaks(`${rendered.title}\n${rendered.body}`).join("; "));
|
|
173
|
+
|
|
174
|
+
check(
|
|
175
|
+
"no build-machine path reaches the body",
|
|
176
|
+
!rendered.body.includes("/home/vsts"));
|
|
177
|
+
|
|
178
|
+
// Two ways to have no frames, and they ask the reader for different things. Pliny-Bot #191 was
|
|
179
|
+
// filed with OWN_CODE_PREFIX set to "Pliny" and still said the repository had not said which
|
|
180
|
+
// assemblies were its own -- it sent a reader to configure a knob that had been set the day
|
|
181
|
+
// before. Only the unconfigured case was covered here, which is why the wording survived.
|
|
182
|
+
check(
|
|
183
|
+
"no prefix configured says which knob to set",
|
|
184
|
+
render(toFindings([row()], { ...options, ownCodePrefix: "" })[0], { ...options, ownCodePrefix: "" })
|
|
185
|
+
.body.includes("has not said which assemblies are its own"));
|
|
186
|
+
|
|
187
|
+
check(
|
|
188
|
+
"a prefix that matched nothing does not claim the prefix is missing",
|
|
189
|
+
!render(toFindings([row()], { ...options, ownCodePrefix: "NoSuchAssembly" })[0], { ...options, ownCodePrefix: "NoSuchAssembly" })
|
|
190
|
+
.body.includes("has not said which assemblies are its own"));
|
|
191
|
+
|
|
192
|
+
check(
|
|
193
|
+
"a prefix that matched nothing names the prefix it tried",
|
|
194
|
+
render(toFindings([row()], { ...options, ownCodePrefix: "NoSuchAssembly" })[0], { ...options, ownCodePrefix: "NoSuchAssembly" })
|
|
195
|
+
.body.includes("No frame in this exception belongs to `NoSuchAssembly`"));
|
|
196
|
+
|
|
197
|
+
section("Fail-closed, but not fail-dead");
|
|
198
|
+
|
|
199
|
+
const mixed = prepare(
|
|
200
|
+
[row(), row({ ProblemId: "Secret", ExceptionType: "AccountKey=abc123def456;Other" })],
|
|
201
|
+
options);
|
|
202
|
+
|
|
203
|
+
check("the clean report is still prepared", mixed.prepared.length === 1, JSON.stringify(mixed.withheld));
|
|
204
|
+
|
|
205
|
+
check("the tripped report is withheld", mixed.withheld.length === 1);
|
|
206
|
+
|
|
207
|
+
check("the caller is told what matched", (mixed.withheld[0]?.matched ?? []).length > 0);
|
|
208
|
+
|
|
209
|
+
check("a withheld report does not refuse the run", mixed.refused === "");
|
|
210
|
+
|
|
211
|
+
check("nothing is prepared from no rows", prepare([], options).prepared.length === 0);
|
|
212
|
+
|
|
213
|
+
check("a non-array does not throw", prepare(null, options).prepared.length === 0);
|
|
214
|
+
|
|
215
|
+
console.log(`\n${pass} passed, ${fail} failed`);
|
|
216
|
+
process.exit(fail > 0 ? 1 : 0);
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-composite-actions/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
-
name: Verify composite actions
|
|
3
|
-
description: Check every local composite action manifest parses and uses only contexts a composite action actually has.
|
|
4
|
-
runs:
|
|
5
|
-
using: composite
|
|
6
|
-
steps:
|
|
7
|
-
- name: Validate composite action manifests
|
|
8
|
-
shell: bash
|
|
9
|
-
run: bash "${GITHUB_ACTION_PATH}/verify-composite-actions.sh"
|
|
1
|
+
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-composite-actions/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
+
name: Verify composite actions
|
|
3
|
+
description: Check every local composite action manifest parses and uses only contexts a composite action actually has.
|
|
4
|
+
runs:
|
|
5
|
+
using: composite
|
|
6
|
+
steps:
|
|
7
|
+
- name: Validate composite action manifests
|
|
8
|
+
shell: bash
|
|
9
|
+
run: bash "${GITHUB_ACTION_PATH}/verify-composite-actions.sh"
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-refine-output/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
-
name: Verify refine output validation
|
|
3
|
-
description: Exercise deterministic Refine output validation regressions.
|
|
4
|
-
runs:
|
|
5
|
-
using: composite
|
|
6
|
-
steps:
|
|
7
|
-
- name: Verify refinement outcomes
|
|
8
|
-
shell: bash
|
|
9
|
-
run: bash "${{ github.action_path }}/verify-refine-output.sh"
|
|
1
|
+
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-refine-output/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
+
name: Verify refine output validation
|
|
3
|
+
description: Exercise deterministic Refine output validation regressions.
|
|
4
|
+
runs:
|
|
5
|
+
using: composite
|
|
6
|
+
steps:
|
|
7
|
+
- name: Verify refinement outcomes
|
|
8
|
+
shell: bash
|
|
9
|
+
run: bash "${{ github.action_path }}/verify-refine-output.sh"
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-route-matrix/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
-
name: Verify route matrix
|
|
3
|
-
description: Exercise the router's classifier against every supported event, and check that each route and dispatch operation has a job in work-router.yml.
|
|
4
|
-
runs:
|
|
5
|
-
using: composite
|
|
6
|
-
steps:
|
|
7
|
-
- name: Run route matrix verification
|
|
8
|
-
shell: bash
|
|
9
|
-
run: bash "${GITHUB_ACTION_PATH}/verify-route-matrix.sh"
|
|
1
|
+
# Managed by @plainconceptsplatform/workflows. Source: loops/actions/verify-route-matrix/action.yml. Update with workflows update --force; consumer edits may be overwritten.
|
|
2
|
+
name: Verify route matrix
|
|
3
|
+
description: Exercise the router's classifier against every supported event, and check that each route and dispatch operation has a job in work-router.yml.
|
|
4
|
+
runs:
|
|
5
|
+
using: composite
|
|
6
|
+
steps:
|
|
7
|
+
- name: Run route matrix verification
|
|
8
|
+
shell: bash
|
|
9
|
+
run: bash "${GITHUB_ACTION_PATH}/verify-route-matrix.sh"
|
|
@@ -372,6 +372,52 @@ else
|
|
|
372
372
|
fi
|
|
373
373
|
if [ "$MIRROR_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
374
374
|
|
|
375
|
+
# Third-party action pins are the same shape of problem. Every `uses:` in a package-managed
|
|
376
|
+
# file is pinned to a full commit SHA, and one action must resolve to exactly one SHA across
|
|
377
|
+
# the whole package: the router, the workers, the shared mechanics and the composite actions
|
|
378
|
+
# are installed together, so two pins for one action mean two versions of it run in the same
|
|
379
|
+
# repository, and the older one is invisible until it breaks. That is how the merge-belt
|
|
380
|
+
# dispatchers ran create-github-app-token v2.0.6 for a release while every worker ran v3.2.0:
|
|
381
|
+
# nothing failed, nothing flagged it, and the action that receives the App's private key was
|
|
382
|
+
# the stale one. Templates are excluded: they are consumer-owned from installation and
|
|
383
|
+
# deliberately carry their own versions (the CI templates pin download-artifact v6 while the
|
|
384
|
+
# loops pin v8). A floating tag (`@v4`) is a missing pin and fails the same check.
|
|
385
|
+
PIN_OK=1
|
|
386
|
+
declare -A PIN_SEEN
|
|
387
|
+
PKG_ROOT="$(cd "${HERE}/../../.." && pwd)"
|
|
388
|
+
pin_scan() {
|
|
389
|
+
local file="$1" line action ref
|
|
390
|
+
# Resolve to an absolute path so the package-relative name in the failure reads like the
|
|
391
|
+
# source tree, not like this script's location.
|
|
392
|
+
file="$(cd "$(dirname "$file")" && pwd)/$(basename "$file")"
|
|
393
|
+
local shown="${file#"${PKG_ROOT}/"}"
|
|
394
|
+
# Anchored on the line start so a comment quoting a `uses:` cannot match, and tolerant of
|
|
395
|
+
# both spellings (`uses:` as a mapping key and `- uses:` as a list item). Local composite
|
|
396
|
+
# actions (`./.github/actions/...`) and docker refs carry no `@` and never match.
|
|
397
|
+
local pin_re='^[[:space:]]*-?[[:space:]]*uses:[[:space:]]+([A-Za-z0-9_.-]+/[A-Za-z0-9_./-]+)@([A-Za-z0-9._-]+)'
|
|
398
|
+
while IFS= read -r line; do
|
|
399
|
+
[[ "$line" =~ $pin_re ]] || continue
|
|
400
|
+
action="${BASH_REMATCH[1]}"
|
|
401
|
+
ref="${BASH_REMATCH[2]}"
|
|
402
|
+
if [ "${PIN_SEEN[$action]+set}" ] && [ "${PIN_SEEN[$action]}" != "$ref" ]; then
|
|
403
|
+
PIN_OK=0
|
|
404
|
+
echo "FAIL: ${shown} pins ${action}@${ref} but ${PIN_SEEN[$action]} is pinned elsewhere; one action, one pin" >&2
|
|
405
|
+
else
|
|
406
|
+
PIN_SEEN[$action]="$ref"
|
|
407
|
+
fi
|
|
408
|
+
if [ "${#ref}" -ne 40 ]; then
|
|
409
|
+
PIN_OK=0
|
|
410
|
+
echo "FAIL: ${shown} uses ${action}@${ref}; a tag is mutable, pin the full commit SHA" >&2
|
|
411
|
+
fi
|
|
412
|
+
done < "$file"
|
|
413
|
+
}
|
|
414
|
+
for pin_file in "$ROUTER_YML" "${WORKFLOWS_DIR}"/agent-*.md "${WORKFLOWS_DIR}"/shared/*.md \
|
|
415
|
+
"${WORKFLOWS_DIR}"/authorize-bot-work.yml "${HERE}/.."/*/action.yml; do
|
|
416
|
+
[ -f "$pin_file" ] || continue
|
|
417
|
+
pin_scan "$pin_file"
|
|
418
|
+
done
|
|
419
|
+
if [ "$PIN_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
420
|
+
|
|
375
421
|
# Run the belt's own jq, rather than reading it. The filter that picks which open pull requests
|
|
376
422
|
# the hourly reconcile job acts on was written as
|
|
377
423
|
# ((env.TRUSTED_BOTS | split(" ")) | index(.user.login) != null)
|
|
@@ -841,19 +887,16 @@ if worker_installed implement; then
|
|
|
841
887
|
PUSH_FALLBACK_OK=0
|
|
842
888
|
echo "FAIL: implement gates a no-pull-request path on code_push_failure_count, which is 0 when gh-aw files the patch as an issue" >&2
|
|
843
889
|
fi
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
#
|
|
851
|
-
|
|
852
|
-
# have been merged or duplicated.
|
|
853
|
-
if [ "$(count -cF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ] ||
|
|
854
|
-
[ "$(count -cF 'env.NO_PULL_REQUEST_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
|
|
890
|
+
grep -qF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD" || {
|
|
891
|
+
PUSH_FALLBACK_OK=0
|
|
892
|
+
echo "FAIL: implement no longer says env.PUSH_CONFLICT_COMMENT on any path" >&2
|
|
893
|
+
}
|
|
894
|
+
# Collapsing this back into whatever covers the other endings is the regression this guards: it
|
|
895
|
+
# is defined once in the env block and printed on exactly one path. The four messages it used to
|
|
896
|
+
# be paired with are asserted together under "Ways a run ends with nothing".
|
|
897
|
+
if [ "$(count -cF 'env.PUSH_CONFLICT_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
|
|
855
898
|
PUSH_FALLBACK_OK=0
|
|
856
|
-
echo "FAIL: implement should print
|
|
899
|
+
echo "FAIL: implement should print the conflicted-push message on exactly one path" >&2
|
|
857
900
|
fi
|
|
858
901
|
# And the two paths must be mutually exclusive, or a conflicted push gets both comments.
|
|
859
902
|
if [ "$(count -cF "process_safe_outputs_items_succeeded != '0'" "$IMPLEMENT_WORKER_MD")" -lt 2 ] ||
|
|
@@ -1202,6 +1245,31 @@ echo "── Merge gate validator ───────────────
|
|
|
1202
1245
|
# reasoning alone, and the worker has not run in production since, so these fixtures are the only
|
|
1203
1246
|
# evidence the change is right. Executing the real script is the same technique that finally
|
|
1204
1247
|
# caught the belt's jq bug, which every reading assertion had walked past.
|
|
1248
|
+
# The gate comments on every bot pull request, so anything it writes there is written dozens of
|
|
1249
|
+
# times a week. A CODEOWNERS of `* @someone` matches every pull request, so the owners row was
|
|
1250
|
+
# naming four people on routine auto-merges and @-mentioning each of them, which notifies them.
|
|
1251
|
+
# The row belongs to the disposition that actually wants an owner, and the names go in a code
|
|
1252
|
+
# span so the comment reports who owns the area without pinging them: on owner-review the gate
|
|
1253
|
+
# requests the review properly, and that is the notification.
|
|
1254
|
+
if worker_installed merge-gate; then
|
|
1255
|
+
OWNERS_OK=1
|
|
1256
|
+
if ! grep -q "outcome == 'owner-review'" "$MERGE_GATE_WORKER_MD"; then
|
|
1257
|
+
OWNERS_OK=0
|
|
1258
|
+
echo "FAIL: the gate comment names required owners unconditionally; gate the row on owner-review" >&2
|
|
1259
|
+
fi
|
|
1260
|
+
# Every rendering of required_owners has to sit inside a code span.
|
|
1261
|
+
while IFS= read -r line; do
|
|
1262
|
+
case "$line" in
|
|
1263
|
+
*'`{0}`'*|*'`${{ needs.protected_changes.outputs.required_owners'*) continue ;;
|
|
1264
|
+
*)
|
|
1265
|
+
OWNERS_OK=0
|
|
1266
|
+
echo "FAIL: required_owners is rendered without a code span, so the gate @-mentions people: ${line# }" >&2
|
|
1267
|
+
;;
|
|
1268
|
+
esac
|
|
1269
|
+
done < <(grep -F 'required_owners' "$MERGE_GATE_WORKER_MD" | grep -vE '^\s*(#|[a-z_]+:)' | grep -F 'Required owners')
|
|
1270
|
+
if [ "$OWNERS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1271
|
+
fi
|
|
1272
|
+
|
|
1205
1273
|
# Blast radius is the input the merge decision leans on hardest, and it is the one a reader
|
|
1206
1274
|
# cannot check by eye. These cases are the six real pull requests the redesign was measured
|
|
1207
1275
|
# against, reduced to their shape: the three that used to be parked for a person purely because
|
|
@@ -1872,6 +1940,71 @@ if worker_installed audit; then
|
|
|
1872
1940
|
if [ "$AUDIT_EMPTY_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
1873
1941
|
fi
|
|
1874
1942
|
|
|
1943
|
+
echo "── Ways a run ends with nothing ──────────────────────────────────────────"
|
|
1944
|
+
|
|
1945
|
+
# A run can end without a pull request in five ways and they need different answers. One sentence
|
|
1946
|
+
# covered all of them, and three issues in two days said "nothing landed": one agent gave up in a
|
|
1947
|
+
# minute, one did more work than the run that succeeded the same hour and never asked for a pull
|
|
1948
|
+
# request, one was told there was nothing to do. Nobody could act on any of them, and the retry
|
|
1949
|
+
# that followed was right for some and wasted on the rest.
|
|
1950
|
+
#
|
|
1951
|
+
# Asserted because collapsing two of these back together is a one-line edit that turns every run
|
|
1952
|
+
# green again while saying less.
|
|
1953
|
+
if worker_installed implement; then
|
|
1954
|
+
ENDINGS_OK=1
|
|
1955
|
+
ENDING_MESSAGES="PUSH_CONFLICT_COMMENT OUTPUT_NOT_APPLIED_COMMENT NOOP_COMMENT PATCH_WITHOUT_REQUEST_COMMENT NO_OUTPUT_COMMENT"
|
|
1956
|
+
|
|
1957
|
+
for message in $ENDING_MESSAGES; do
|
|
1958
|
+
# Defined once in the env block and printed on exactly one path. Two usages means two
|
|
1959
|
+
# branches now say the same thing, which is the collapse this guards.
|
|
1960
|
+
if [ "$(count -cF "env.${message}" "$IMPLEMENT_WORKER_MD")" -ne 1 ]; then
|
|
1961
|
+
ENDINGS_OK=0
|
|
1962
|
+
echo "FAIL: implement should print ${message} on exactly one path" >&2
|
|
1963
|
+
fi
|
|
1964
|
+
done
|
|
1965
|
+
|
|
1966
|
+
# The sentence that used to cover all five. Its text survives as NO_OUTPUT_COMMENT; the name
|
|
1967
|
+
# coming back means the branches were merged again.
|
|
1968
|
+
if [ "$(count -cF 'NO_PULL_REQUEST_COMMENT' "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
|
|
1969
|
+
ENDINGS_OK=0
|
|
1970
|
+
echo "FAIL: implement still carries NO_PULL_REQUEST_COMMENT, which said the same thing about five different endings" >&2
|
|
1971
|
+
fi
|
|
1972
|
+
|
|
1973
|
+
# What tells them apart. Both are the agent job's own outputs, and gh-aw gates on the first
|
|
1974
|
+
# itself, so losing them means guessing again.
|
|
1975
|
+
for signal in 'needs.agent.outputs.output_types' 'needs.agent.outputs.has_patch'; do
|
|
1976
|
+
if [ "$(count -cE "^ *if:.*${signal}" "$IMPLEMENT_WORKER_MD")" -eq 0 ]; then
|
|
1977
|
+
ENDINGS_OK=0
|
|
1978
|
+
echo "FAIL: implement no longer reads ${signal}, so it cannot tell its no-pull-request endings apart" >&2
|
|
1979
|
+
fi
|
|
1980
|
+
done
|
|
1981
|
+
|
|
1982
|
+
# `!x == 'true'` parses as `(!x) == 'true'` in a GitHub expression, which compares a boolean to
|
|
1983
|
+
# a string and is always false. Written wrong here first, and it would have made the
|
|
1984
|
+
# produced-nothing branch unreachable while every other branch still looked right.
|
|
1985
|
+
if [ "$(count -cF "!needs.agent.outputs.has_patch ==" "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
|
|
1986
|
+
ENDINGS_OK=0
|
|
1987
|
+
echo "FAIL: implement negates has_patch as \`!x == 'true'\`, which binds as \`(!x) == 'true'\` and never matches; use != 'true'" >&2
|
|
1988
|
+
fi
|
|
1989
|
+
|
|
1990
|
+
# A bot that reported nothing to do made a decision. The janitor retries a stalled issue, and
|
|
1991
|
+
# re-running it produces the same answer, so this one path must not be flagged.
|
|
1992
|
+
noop_block=$(awk "/name: Say there was nothing to do/{found=1} found{print} found && /View this workflow run/{exit}" "$IMPLEMENT_WORKER_MD")
|
|
1993
|
+
if [ -n "$noop_block" ] && printf '%s' "$noop_block" | grep -qF 'STALLED_LABEL'; then
|
|
1994
|
+
ENDINGS_OK=0
|
|
1995
|
+
echo "FAIL: implement flags a deliberate noop as stalled, so the janitor will retry a decision until it runs out of attempts" >&2
|
|
1996
|
+
fi
|
|
1997
|
+
|
|
1998
|
+
# The usage artifact carrying the token counts is uploaded by gh-aw's conclusion job, and that
|
|
1999
|
+
# job already depends on conclude. Naming it is a cycle and the worker stops compiling.
|
|
2000
|
+
if [ "$(count -cE "^ *needs:.*conclusion" "$IMPLEMENT_WORKER_MD")" -ne 0 ]; then
|
|
2001
|
+
ENDINGS_OK=0
|
|
2002
|
+
echo "FAIL: implement depends on the conclusion job, which depends on conclude; that cycle does not compile" >&2
|
|
2003
|
+
fi
|
|
2004
|
+
|
|
2005
|
+
if [ "$ENDINGS_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
2006
|
+
fi
|
|
2007
|
+
|
|
1875
2008
|
echo "── Runner pools ──────────────────────────────────────────────────────────"
|
|
1876
2009
|
|
|
1877
2010
|
# Where every job runs, stated once and asserted, because GitHub gives a wrong pool no error: a
|
|
@@ -2060,6 +2193,94 @@ fi
|
|
|
2060
2193
|
if [ "$GATE_RUN_ID_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
2061
2194
|
fi
|
|
2062
2195
|
|
|
2196
|
+
echo "── App error privacy ─────────────────────────────────────────────────────"
|
|
2197
|
+
|
|
2198
|
+
# The application error report files into the private repository rather than out of it, so
|
|
2199
|
+
# its danger is the opposite one: not that a name escapes, but that an identifier joining a
|
|
2200
|
+
# stack trace back to one person's conversation gets written down beside it.
|
|
2201
|
+
#
|
|
2202
|
+
# Four properties hold that line and each is a line an edit could drop with nothing going red.
|
|
2203
|
+
# Asserted the way the error report's are: call sites and set equality, never a bare symbol,
|
|
2204
|
+
# because a grep for a name passes against a guard that has been renamed rather than removed.
|
|
2205
|
+
APP_ERRORS_MJS="${HERE}/../collect-app-errors/group-and-redact.mjs"
|
|
2206
|
+
APP_ERRORS_SH="${HERE}/../collect-app-errors/query-app-errors.sh"
|
|
2207
|
+
APP_ERRORS_YML="${HERE}/../collect-app-errors/action.yml"
|
|
2208
|
+
|
|
2209
|
+
if [ -f "$APP_ERRORS_MJS" ]; then
|
|
2210
|
+
AE_OK=1
|
|
2211
|
+
ae() {
|
|
2212
|
+
grep -qE "$1" "$2" || { AE_OK=0; echo "FAIL: collect-app-errors ${3}" >&2; }
|
|
2213
|
+
}
|
|
2214
|
+
|
|
2215
|
+
# 1. The query reads the two tables the application's own telemetry goes to, and no others.
|
|
2216
|
+
# AppServiceConsoleLogs and ContainerAppConsoleLogs live in the same workspace and carry raw
|
|
2217
|
+
# engine stdout, which the no-content rule does not govern. A union across them is an
|
|
2218
|
+
# incident rather than a wider query.
|
|
2219
|
+
#
|
|
2220
|
+
# AppTraces is deliberate and was added after the first dry run against a real workspace
|
|
2221
|
+
# returned nought exceptions on a service that had been failing all morning: nothing in
|
|
2222
|
+
# these applications calls RecordException, so every failure they care about is caught in
|
|
2223
|
+
# code, logged through ILogger, and lands in AppTraces alone.
|
|
2224
|
+
ae '^ AppExceptions$' "$APP_ERRORS_SH" 'does not query AppExceptions by name'
|
|
2225
|
+
ae '^ AppTraces$' "$APP_ERRORS_SH" 'does not query AppTraces, where caught failures land'
|
|
2226
|
+
# The heredoc alone. The comment above it names the console tables in order to say they are
|
|
2227
|
+
# never read, and a grep over the whole file cannot tell the warning from the offence.
|
|
2228
|
+
query_body="$(sed -n '/^read -r -d .. QUERY <<KQL/,/^KQL$/p' "$APP_ERRORS_SH")"
|
|
2229
|
+
if printf '%s' "$query_body" | grep -qE 'ConsoleLogs|AppRequests|AppDependencies|union \*|search '; then
|
|
2230
|
+
AE_OK=0
|
|
2231
|
+
echo "FAIL: collect-app-errors reads a table it has no business reading" >&2
|
|
2232
|
+
fi
|
|
2233
|
+
|
|
2234
|
+
# 2. Counts are scaled for sampling. A pre environment samples at 0.3, so count() reads a
|
|
2235
|
+
# problem that happened two hundred times as sixty, and it falls under a floor set to catch
|
|
2236
|
+
# it. This is a correctness guard that looks like a style one.
|
|
2237
|
+
ae 'sum\(ItemCount\)' "$APP_ERRORS_SH" 'counts rows instead of summing ItemCount'
|
|
2238
|
+
|
|
2239
|
+
# 3. The field allowlist is what actually holds the line, because a redaction rule filters
|
|
2240
|
+
# values and cannot see a field somebody adds next year. Compared as a set, so reordering is
|
|
2241
|
+
# fine and an addition is not. run_id is the one that matters: it is unhashed on this
|
|
2242
|
+
# telemetry and joins to a conversation and to a person.
|
|
2243
|
+
expected_fields="exceptionType frames fingerprint firstSeen lastSeen occurrences operationName operations problemId roleName"
|
|
2244
|
+
actual_fields="$(sed -n '/export const FINDING_FIELDS = \[/,/\];/p' "$APP_ERRORS_MJS" |
|
|
2245
|
+
grep -oE '"[a-zA-Z]+"' | tr -d '"' | sort | tr '\n' ' ' | sed 's/ $//')"
|
|
2246
|
+
if [ "$actual_fields" = "$(echo "$expected_fields" | tr ' ' '\n' | sort | tr '\n' ' ' | sed 's/ $//')" ]; then
|
|
2247
|
+
PASS=$((PASS + 1))
|
|
2248
|
+
else
|
|
2249
|
+
FAIL=$((FAIL + 1))
|
|
2250
|
+
echo "FAIL: collect-app-errors reportable fields changed: ${actual_fields}" >&2
|
|
2251
|
+
fi
|
|
2252
|
+
# Plain grep, not the count() helper: that is `grep || true`, so an `if count -q` is true
|
|
2253
|
+
# whatever it finds and the guard never fires. Three of these were written that way first
|
|
2254
|
+
# and all three failed open, which is the same shape of bug this file exists to catch.
|
|
2255
|
+
if sed -n '/export const FINDING_FIELDS = \[/,/\];/p' "$APP_ERRORS_MJS" | grep -qiE 'run_?id'; then
|
|
2256
|
+
AE_OK=0
|
|
2257
|
+
echo "FAIL: collect-app-errors lists a run id as reportable; it joins telemetry to a person" >&2
|
|
2258
|
+
fi
|
|
2259
|
+
|
|
2260
|
+
# 4. Scrub, then scan, then refuse. Scrubbing masks what is common and safely replaceable so
|
|
2261
|
+
# one unlucky operation name does not leave the job red and filing nothing forever; the
|
|
2262
|
+
# scanner is the backstop for whatever that missed; and a withheld report still turns the
|
|
2263
|
+
# run red so the field that carried it gets fixed.
|
|
2264
|
+
ae 'export const SCRUBS = \[' "$APP_ERRORS_MJS" 'declares no scrubbing'
|
|
2265
|
+
ae 'export const LEAK_CHECKS = \[' "$APP_ERRORS_MJS" 'declares no leak scanner'
|
|
2266
|
+
ae 'core\.setFailed.*withheld by the leak scanner' "$APP_ERRORS_YML" 'does not go red on a withheld report'
|
|
2267
|
+
ae 'not in the reportable field list' "$APP_ERRORS_MJS" 'does not refuse an unlisted field'
|
|
2268
|
+
ae 'core\.setFailed\(`Nothing was filed' "$APP_ERRORS_YML" 'does not stop when the module refuses'
|
|
2269
|
+
|
|
2270
|
+
# A GUID must be masked on the way in, not merely detected on the way out. Both, in fact:
|
|
2271
|
+
# the scrub is what keeps the job alive and the check is what keeps it honest.
|
|
2272
|
+
ae '\{guid\}' "$APP_ERRORS_MJS" 'does not mask GUIDs'
|
|
2273
|
+
|
|
2274
|
+
# 5. No model. This is the same assertion the error report carries, for the same reason: a
|
|
2275
|
+
# model summarising a stack trace paraphrases it, and a paraphrase is a wrong issue.
|
|
2276
|
+
if grep -qiE '(engine:|opencode|safe-outputs|OPENAI_API_KEY)' "$APP_ERRORS_YML"; then
|
|
2277
|
+
AE_OK=0
|
|
2278
|
+
echo "FAIL: collect-app-errors has grown a model" >&2
|
|
2279
|
+
fi
|
|
2280
|
+
|
|
2281
|
+
if [ "$AE_OK" -eq 1 ]; then PASS=$((PASS + 1)); else FAIL=$((FAIL + 1)); fi
|
|
2282
|
+
fi
|
|
2283
|
+
|
|
2063
2284
|
echo
|
|
2064
2285
|
if [ "$FAIL" -eq 0 ]; then
|
|
2065
2286
|
echo "Route matrix: ${PASS} passed"
|