@mmerterden/multi-agent-pipeline 16.29.0 → 16.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +82 -0
  2. package/docs/features.md +14 -0
  3. package/install/_common.mjs +45 -4
  4. package/package.json +1 -1
  5. package/pipeline/commands/multi-agent/design-check/SKILL.md +6 -5
  6. package/pipeline/commands/multi-agent/help/SKILL.md +13 -12
  7. package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
  8. package/pipeline/commands/multi-agent/sync/SKILL.md +3 -4
  9. package/pipeline/lib/credential-inventory.sh +1 -0
  10. package/pipeline/lib/repo-hygiene.sh +164 -0
  11. package/pipeline/lib/vercel-deploy.sh +41 -22
  12. package/pipeline/multi-agent-refs/channels/pr.md +37 -1
  13. package/pipeline/multi-agent-refs/features/doctor.md +12 -0
  14. package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
  15. package/pipeline/multi-agent-refs/features/visual-evidence.md +103 -20
  16. package/pipeline/multi-agent-refs/keychain.md +1 -0
  17. package/pipeline/multi-agent-refs/knowledge.md +1 -1
  18. package/pipeline/multi-agent-refs/phases/phase-0-init.md +36 -7
  19. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  20. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +14 -2
  21. package/pipeline/multi-agent-refs/phases/phase-5-test.md +12 -2
  22. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +23 -0
  23. package/pipeline/preferences-template.json +2 -1
  24. package/pipeline/schemas/agent-state.schema.json +79 -1
  25. package/pipeline/schemas/prefs.schema.json +24 -1
  26. package/pipeline/schemas/token-budget.json +10 -10
  27. package/pipeline/scripts/bulk-read.sh +13 -5
  28. package/pipeline/scripts/capture-evidence.sh +170 -5
  29. package/pipeline/scripts/doctor.mjs +56 -1
  30. package/pipeline/scripts/evidence-gate.mjs +31 -2
  31. package/pipeline/scripts/gc-tmp.sh +30 -0
  32. package/pipeline/scripts/gc-worktrees.sh +32 -10
  33. package/pipeline/scripts/offload-ref.sh +13 -5
  34. package/pipeline/scripts/probe-evidence-capability.sh +250 -0
  35. package/pipeline/scripts/purge.sh +11 -0
  36. package/pipeline/scripts/run-ui-tests.sh +380 -0
  37. package/pipeline/scripts/worktree-finalize.sh +9 -0
  38. package/pipeline/skills/.skill-manifest.json +19 -11
  39. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +10 -1
  40. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +2 -1
@@ -0,0 +1,250 @@
1
+ #!/usr/bin/env bash
2
+ #
3
+ # probe-evidence-capability.sh - measure what visual evidence this machine and
4
+ # this repo can actually produce, BEFORE the user is asked to choose a test depth.
5
+ #
6
+ # Contract: multi-agent-refs/features/visual-evidence.md section 4.
7
+ #
8
+ # The order is probe, then question, then run. Offering "unit + UI test with a
9
+ # screen recording" and only then discovering there is no UI test target, or no
10
+ # booted simulator, spends the user's answer on something that cannot happen. So
11
+ # the options are built from what this prints.
12
+ #
13
+ # Three rules this file exists to keep:
14
+ #
15
+ # 1. Absence carries a reason. Every empty value is paired with a *_REASON, so
16
+ # a closed option can say why it is closed instead of vanishing from the menu.
17
+ # 2. Unmeasurable is `unknown`, never `false`. A probe that could not look and a
18
+ # probe that looked and found nothing are different facts, and collapsing
19
+ # them lets a missing tool read as a clean negative.
20
+ # 3. Detection is not reimplemented here. The UI test answer comes from
21
+ # run-ui-tests.sh detect, which is also what actually runs the tests; a
22
+ # second copy is a second place for the answer to drift.
23
+ #
24
+ # Usage:
25
+ # probe-evidence-capability.sh --platform <ios|android> [--repo <path>]
26
+ # [--changed <f>[,<f>...]] [--json]
27
+ # [--json-out <path>] [--only all|device]
28
+ #
29
+ # Output: KEY='VALUE' lines (eval-able; every value is shell-quoted), or a JSON object with --json (shaped for
30
+ # state.evidenceCapability). --json-out writes that JSON to a file while stdout
31
+ # stays KEY=VALUE, so one run serves both the shell that builds the menu and the
32
+ # state that records the measurement - two runs would mean two repo scans and
33
+ # two chances to disagree.
34
+ #
35
+ # --only device skips the UI-test detection. Phase 3 re-checks the device right
36
+ # before recording, and nothing else it would re-scan can have changed.
37
+ #
38
+ # Exit: 0 probed (whatever the verdicts), 2 usage.
39
+ # Never non-zero for an absent capability: absence is the finding.
40
+ set -uo pipefail
41
+
42
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
43
+ PLATFORM=""; REPO="$PWD"; CHANGED=""; AS_JSON=0; JSON_OUT=""; ONLY="all"
44
+
45
+ while [ "$#" -gt 0 ]; do
46
+ case "$1" in
47
+ --platform) PLATFORM="${2:-}"; shift 2 ;;
48
+ --repo) REPO="${2:-}"; shift 2 ;;
49
+ --changed) CHANGED="${CHANGED:+$CHANGED,}${2:-}"; shift 2 ;;
50
+ --json) AS_JSON=1; shift ;;
51
+ --json-out) JSON_OUT="${2:-}"; shift 2 ;;
52
+ --only) ONLY="${2:-all}"; shift 2 ;;
53
+ *) echo "probe-evidence-capability: unknown option $1" >&2; exit 2 ;;
54
+ esac
55
+ done
56
+ case "$ONLY" in all | device) ;; *)
57
+ echo "probe-evidence-capability: --only takes 'all' or 'device'" >&2; exit 2 ;;
58
+ esac
59
+
60
+ case "$PLATFORM" in ios | android) ;; *)
61
+ echo "usage: probe-evidence-capability.sh --platform <ios|android> [--repo <path>] [--changed <f>] [--json]" >&2
62
+ exit 2 ;;
63
+ esac
64
+ [ -d "$REPO" ] || { echo "probe-evidence-capability: no such repo directory: $REPO" >&2; exit 2; }
65
+
66
+ UI_TEST_TARGET=""; UI_TEST_TARGETS=""; UI_TEST_TARGET_REASON=""
67
+ UI_TEST_MATCHES=""; UI_TEST_MATCH_REASON=""
68
+ DEVICE=""; DEVICE_REASON=""
69
+ RECORDER="unknown"; RECORDER_REASON=""
70
+ MCP="unknown"; MCP_REASON=""
71
+
72
+ # ---- UI test target, delegated ------------------------------------------------
73
+ RUNNER="$HERE/run-ui-tests.sh"
74
+ if [ "$ONLY" = "device" ]; then
75
+ # Phase 3 re-checks the device right before recording, and only the device.
76
+ # Delegating detection there would re-walk the whole repo for a row that
77
+ # cannot have changed, which cost 5 seconds of a phase that is already holding
78
+ # a green build.
79
+ UI_TEST_TARGET_REASON="not probed (--only device)"
80
+ elif [ ! -x "$RUNNER" ] && [ ! -f "$RUNNER" ]; then
81
+ UI_TEST_TARGET_REASON="run-ui-tests.sh not found beside this script"
82
+ else
83
+ DETECT_ARGS=(detect --platform "$PLATFORM" --repo "$REPO")
84
+ [ -n "$CHANGED" ] && DETECT_ARGS+=(--changed "$CHANGED")
85
+ while IFS='=' read -r k v; do
86
+ case "$k" in
87
+ UI_TEST_TARGET) UI_TEST_TARGET="$v" ;;
88
+ UI_TEST_TARGETS) UI_TEST_TARGETS="$v" ;;
89
+ UI_TEST_TARGET_REASON) UI_TEST_TARGET_REASON="$v" ;;
90
+ UI_TEST_MATCHES) UI_TEST_MATCHES="$v" ;;
91
+ UI_TEST_MATCH_REASON) UI_TEST_MATCH_REASON="$v" ;;
92
+ esac
93
+ done < <(bash "$RUNNER" "${DETECT_ARGS[@]}" 2>/dev/null)
94
+ fi
95
+
96
+ # ---- Device -------------------------------------------------------------------
97
+ case "$PLATFORM" in
98
+ ios)
99
+ if ! command -v xcrun >/dev/null 2>&1; then
100
+ DEVICE_REASON="xcrun unavailable; cannot tell whether a simulator exists"
101
+ else
102
+ DEVICE=$(xcrun simctl list devices booted 2>/dev/null | grep -oE '[0-9A-F-]{36}' | head -1)
103
+ if [ -z "$DEVICE" ]; then
104
+ # Bootable is not booted, and the difference is a question the user can
105
+ # act on: one is "start your simulator", the other is "this machine has
106
+ # no iOS runtime installed".
107
+ if xcrun simctl list devices available 2>/dev/null | grep -q "([0-9A-F-]\{36\})"; then
108
+ DEVICE_REASON="no booted simulator, but one is available to boot"
109
+ else
110
+ DEVICE_REASON="no iOS simulator available on this machine"
111
+ fi
112
+ fi
113
+ fi
114
+ ;;
115
+ android)
116
+ if ! command -v adb >/dev/null 2>&1; then
117
+ DEVICE_REASON="adb unavailable; cannot tell whether a device is attached"
118
+ else
119
+ DEVICE=$(adb devices 2>/dev/null | awk 'NR>1 && $2=="device" {print $1; exit}')
120
+ [ -n "$DEVICE" ] || DEVICE_REASON="no attached device or running emulator"
121
+ fi
122
+ ;;
123
+ esac
124
+
125
+ # ---- Recorder -----------------------------------------------------------------
126
+ # The recorder is the same CLI the capture needs, so this answers "would
127
+ # capture-evidence.sh video start work" rather than "is some recorder installed".
128
+ case "$PLATFORM" in
129
+ ios)
130
+ if command -v xcrun >/dev/null 2>&1; then RECORDER="true"; else
131
+ RECORDER="false"; RECORDER_REASON="xcrun unavailable"
132
+ fi
133
+ ;;
134
+ android)
135
+ if command -v adb >/dev/null 2>&1; then RECORDER="true"; else
136
+ RECORDER="false"; RECORDER_REASON="adb unavailable"
137
+ fi
138
+ ;;
139
+ esac
140
+ if [ "$RECORDER" = "true" ] && ! command -v ffprobe >/dev/null 2>&1; then
141
+ # Not fatal. ffprobe only verifies the result, so its absence downgrades what
142
+ # can be CHECKED about a recording, never whether one can be made.
143
+ RECORDER_REASON="ffprobe unavailable; a recording cannot be verified after capture"
144
+ fi
145
+
146
+ # ---- Toolkit MCP --------------------------------------------------------------
147
+ # Registration only, never a handshake: a probe that spawns the server costs
148
+ # seconds at intake, and the question this answers is whether tier 2 may be
149
+ # offered at all.
150
+ CLAUDE_JSON="$HOME/.claude.json"
151
+ SETTINGS_JSON="$HOME/.claude/settings.json"
152
+ if command -v node >/dev/null 2>&1; then
153
+ MCP=$(node -e '
154
+ const fs = require("fs");
155
+ const hit = (p) => {
156
+ try {
157
+ const j = JSON.parse(fs.readFileSync(p, "utf8"));
158
+ return Object.keys(j.mcpServers || {}).some((k) => /toolkit/i.test(k));
159
+ } catch { return false; }
160
+ };
161
+ process.stdout.write(process.argv.slice(1).some(hit) ? "true" : "false");
162
+ ' "$CLAUDE_JSON" "$SETTINGS_JSON" 2>/dev/null)
163
+ [ -n "$MCP" ] || MCP="unknown"
164
+ [ "$MCP" = "false" ] && MCP_REASON="no mcpServers entry matching /toolkit/i"
165
+ [ "$MCP" = "unknown" ] && MCP_REASON="could not read the MCP registration files"
166
+ else
167
+ MCP="unknown"; MCP_REASON="node unavailable; cannot read the MCP registration"
168
+ fi
169
+
170
+ # ---- Tier verdicts ------------------------------------------------------------
171
+ # Computed before the report so both output forms carry them: a caller that
172
+ # persists the JSON and a caller that evals the KEY=VALUE lines must not have to
173
+ # re-derive the same booleans and risk deriving them differently.
174
+ # TIER1 needs a target AND a device AND a recorder; TIER2 drops the target and
175
+ # adds MCP; tier 3 is always reachable because "record nothing and say why" is.
176
+ T1=closed; T2=closed
177
+ [ -n "$UI_TEST_TARGET$UI_TEST_TARGETS" ] && [ -n "$DEVICE" ] && [ "$RECORDER" = "true" ] && T1=open
178
+ [ -n "$DEVICE" ] && [ "$RECORDER" = "true" ] && [ "$MCP" = "true" ] && T2=open
179
+
180
+ # --only device did not look for a target, so tier 1 is UNKNOWN here, not closed.
181
+ # Reporting it closed would be rule 2 broken by the file that states it: a caller
182
+ # re-checking the device before recording would read "tier 1 unavailable" from a
183
+ # measurement that never ran and downgrade a recording it could have made. Tier 2
184
+ # does not depend on the target, so it stays a real verdict.
185
+ [ "$ONLY" = "device" ] && T1=unknown
186
+
187
+ emit_json() {
188
+ UI_TEST_TARGET="$UI_TEST_TARGET" UI_TEST_TARGETS="$UI_TEST_TARGETS" \
189
+ UI_TEST_TARGET_REASON="$UI_TEST_TARGET_REASON" UI_TEST_MATCHES="$UI_TEST_MATCHES" \
190
+ UI_TEST_MATCH_REASON="$UI_TEST_MATCH_REASON" DEVICE="$DEVICE" DEVICE_REASON="$DEVICE_REASON" \
191
+ RECORDER="$RECORDER" RECORDER_REASON="$RECORDER_REASON" MCP="$MCP" MCP_REASON="$MCP_REASON" \
192
+ PLATFORM="$PLATFORM" TIER1="$T1" TIER2="$T2" \
193
+ node -e '
194
+ const e = process.env;
195
+ const list = (s) => (s ? s.split(",").filter(Boolean) : []);
196
+ const tri = (s) => (s === "true" ? true : s === "false" ? false : null);
197
+ process.stdout.write(JSON.stringify({
198
+ platform: e.PLATFORM,
199
+ uiTestTarget: e.UI_TEST_TARGET || null,
200
+ uiTestTargets: list(e.UI_TEST_TARGETS),
201
+ uiTestTargetReason: e.UI_TEST_TARGET_REASON || null,
202
+ matchingTests: list(e.UI_TEST_MATCHES),
203
+ matchingTestsReason: e.UI_TEST_MATCH_REASON || null,
204
+ device: e.DEVICE || null,
205
+ deviceReason: e.DEVICE_REASON || null,
206
+ recorder: tri(e.RECORDER),
207
+ recorderReason: e.RECORDER_REASON || null,
208
+ mcp: tri(e.MCP),
209
+ mcpReason: e.MCP_REASON || null,
210
+ tier1: e.TIER1,
211
+ tier2: e.TIER2,
212
+ }, null, 2) + "\n");
213
+ '
214
+ }
215
+
216
+ if [ -n "$JSON_OUT" ]; then
217
+ mkdir -p "$(dirname "$JSON_OUT")" 2>/dev/null || true
218
+ emit_json > "$JSON_OUT" || {
219
+ echo "probe-evidence-capability: could not write $JSON_OUT" >&2
220
+ exit 2
221
+ }
222
+ fi
223
+
224
+ if [ "$AS_JSON" -eq 1 ]; then
225
+ emit_json
226
+ exit 0
227
+ fi
228
+
229
+ # Every value is single-quoted, because the caller EVALS this. The reasons are
230
+ # prose - "no booted simulator, but one is available to boot" - and an unquoted
231
+ # assignment makes eval run `booted` as a command and assign the first word. The
232
+ # bug is invisible on a machine where the reasons happen to be empty, which is
233
+ # exactly the machine a developer tests on.
234
+ q() { printf "%s='%s'" "$1" "$(printf '%s' "$2" | sed "s/'/'\\\\''/g")"; printf '\n'; }
235
+
236
+ q EVIDENCE_PLATFORM "$PLATFORM"
237
+ q EVIDENCE_UI_TEST_TARGET "$UI_TEST_TARGET"
238
+ q EVIDENCE_UI_TEST_TARGETS "$UI_TEST_TARGETS"
239
+ q EVIDENCE_UI_TEST_TARGET_REASON "$UI_TEST_TARGET_REASON"
240
+ q EVIDENCE_MATCHING_TESTS "$UI_TEST_MATCHES"
241
+ q EVIDENCE_MATCHING_TESTS_REASON "$UI_TEST_MATCH_REASON"
242
+ q EVIDENCE_DEVICE "$DEVICE"
243
+ q EVIDENCE_DEVICE_REASON "$DEVICE_REASON"
244
+ q EVIDENCE_RECORDER "$RECORDER"
245
+ q EVIDENCE_RECORDER_REASON "$RECORDER_REASON"
246
+ q EVIDENCE_MCP "$MCP"
247
+ q EVIDENCE_MCP_REASON "$MCP_REASON"
248
+
249
+ q EVIDENCE_TIER1 "$T1"
250
+ q EVIDENCE_TIER2 "$T2"
@@ -231,5 +231,16 @@ fi
231
231
  # Drop the now-empty .worktrees/ shell (only when truly empty).
232
232
  [ -d "$WT_ROOT" ] && rmdir "$WT_ROOT" 2>/dev/null
233
233
 
234
+ # purge is the whole-repo teardown, so it is the one place the managed exclude
235
+ # block should come back out - a per-task finalize must not, because other tasks
236
+ # in the same repo still need the guard. Empty artefact parents go with it.
237
+ # Only the marker-delimited block is removed; lines a human added stay.
238
+ _MA_HYG="$(dirname "${BASH_SOURCE[0]}")/../lib/repo-hygiene.sh"
239
+ if [ -f "$_MA_HYG" ]; then
240
+ . "$_MA_HYG"
241
+ ma_hygiene_prune_empty "$MAIN_WT"
242
+ ma_hygiene_release_exclusions "$MAIN_WT"
243
+ fi
244
+
234
245
  echo "══ purge: removed $removed_wt worktree(s) and $removed_br local branch(es) in $MAIN_WT ══"
235
246
  exit 0
@@ -0,0 +1,380 @@
1
+ #!/usr/bin/env bash
2
+ #
3
+ # run-ui-tests.sh - find the repo's own UI test target, pick the tests that
4
+ # cover the files this task changed, and run them.
5
+ #
6
+ # Contract: multi-agent-refs/features/visual-evidence.md section 4 (video tier 1).
7
+ #
8
+ # Tier 1 of the flow video is "the repo's own UI test drove the screen". That is
9
+ # worth more than a generated flow for one reason: the test already runs in CI,
10
+ # so the recording shows a path somebody committed to keeping green. This script
11
+ # is the half that finds and runs it; capture-evidence.sh records around it.
12
+ #
13
+ # Two sub-commands, deliberately in one file so detection has exactly one
14
+ # implementation. The capability probe needs the same answer BEFORE the user is
15
+ # asked which test depth to run, and a second copy of "does this repo have a UI
16
+ # test target" is a second place for the answer to drift.
17
+ #
18
+ # run-ui-tests.sh detect --platform <ios|android> [--repo <path>] [--changed <f>[,<f>...]]
19
+ # Print KEY=VALUE lines and exit. Never builds, never runs a test.
20
+ # UI_TEST_TARGET=<name>| (empty until one candidate is chosen)
21
+ # UI_TEST_TARGETS=<name[,name...]> (every candidate found)
22
+ # UI_TEST_TARGET_REASON=<why it is empty>
23
+ # UI_TEST_MATCHES=<Target/Class[,Target/Class...]>|
24
+ # UI_TEST_MATCH_REASON=<why it is empty>
25
+ # UI_TEST_CONTAINER=<-project X|-workspace X>|
26
+ # UI_TEST_SCHEME=<name>|
27
+ #
28
+ # run-ui-tests.sh run --platform <ios|android> [--repo <path>] [--changed <f>...]
29
+ # [--device <udid|serial>] [--log <path>] [--all]
30
+ # Run the matching tests (or the whole UI suite with --all).
31
+ #
32
+ # Env:
33
+ # UI_TEST_LOG default "$PWD/.pipeline/ui-test.log"
34
+ #
35
+ # Exit:
36
+ # 0 ran, and the run passed
37
+ # 1 ran, and the run failed
38
+ # 2 usage / environment
39
+ # 3 a UI test target exists but nothing matches the changed files
40
+ # 4 no UI test target in this project
41
+ #
42
+ # 3 and 4 are reasons to fall to video tier 2, not failures. Only 1 is a red test.
43
+ set -uo pipefail
44
+
45
+ MODE="${1:-}"
46
+ shift 2>/dev/null || true
47
+
48
+ PLATFORM=""; REPO="$PWD"; CHANGED=""; DEVICE=""; LOG=""; RUN_ALL=0
49
+ while [ "$#" -gt 0 ]; do
50
+ case "$1" in
51
+ --platform) PLATFORM="${2:-}"; shift 2 ;;
52
+ --repo) REPO="${2:-}"; shift 2 ;;
53
+ --changed) CHANGED="${CHANGED:+$CHANGED,}${2:-}"; shift 2 ;;
54
+ --device) DEVICE="${2:-}"; shift 2 ;;
55
+ --log) LOG="${2:-}"; shift 2 ;;
56
+ --all) RUN_ALL=1; shift ;;
57
+ *) echo "run-ui-tests: unknown option $1" >&2; exit 2 ;;
58
+ esac
59
+ done
60
+
61
+ case "$MODE" in detect | run) ;; *)
62
+ echo "usage: run-ui-tests.sh detect|run --platform <ios|android> [--repo <path>] [--changed <f>]" >&2
63
+ exit 2 ;;
64
+ esac
65
+ case "$PLATFORM" in ios | android) ;; *)
66
+ echo "run-ui-tests: unsupported platform '$PLATFORM'" >&2; exit 2 ;;
67
+ esac
68
+ [ -d "$REPO" ] || { echo "run-ui-tests: no such repo directory: $REPO" >&2; exit 2; }
69
+
70
+ LOG="${LOG:-${UI_TEST_LOG:-$PWD/.pipeline/ui-test.log}}"
71
+
72
+ TARGET=""; TARGET_REASON=""; TARGETS=""; SOURCES=""
73
+ CONTAINER_ARGV=()
74
+ MATCHES=""; MATCH_REASON=""
75
+ CONTAINER=""; SCHEME=""
76
+
77
+ # The names a changed UI file can be known by inside a UI test: the type it
78
+ # declares, and the file's own basename. A UI test refers to a screen by its
79
+ # accessibility identifier far more often than by its type name, and identifiers
80
+ # are generated from the same names, so both land on the same string often enough
81
+ # to be worth trying. This is a heuristic and the caller is told so: an empty
82
+ # match set falls to tier 2 rather than claiming the screen is untested.
83
+ changed_names() {
84
+ # The trailing newline matters: `while read` returns non-zero on an
85
+ # unterminated final line and drops it, so a one-element list read this way
86
+ # yields nothing at all and every change looks like it has no testable name.
87
+ printf '%s\n' "$CHANGED" | tr ',' '\n' | while IFS= read -r f; do
88
+ [ -n "$f" ] || continue
89
+ b="${f##*/}"
90
+ printf '%s\n' "${b%.*}"
91
+ done | sort -u
92
+ }
93
+
94
+ # Detection reads the filesystem, never `xcodebuild -list`. On a real app
95
+ # workspace that call resolves the SPM graph first and took 72 seconds here,
96
+ # and detection runs at intake while the user is waiting for a question. The
97
+ # scheme it would have told us is only needed to RUN, so it is resolved there,
98
+ # behind a timeout.
99
+ resolve_ios_container() {
100
+ local ws proj
101
+ # CONTAINER is reported as text for the detect contract, but the RUN path uses
102
+ # CONTAINER_ARGV: a checkout under "~/My Projects/" splits an unquoted
103
+ # -project /path/with space into two arguments and xcodebuild is handed a
104
+ # project that does not exist.
105
+ # A standalone .xcworkspace wins, but *.xcodeproj/project.xcworkspace is the
106
+ # implicit one Xcode keeps inside every project. Passing that to -workspace
107
+ # builds a different, schemeless container, so it must never be treated as a
108
+ # workspace the repo chose.
109
+ ws=$(find "$REPO" -maxdepth 2 -name "*.xcworkspace" -not -path "*/.*" \
110
+ -not -path "*.xcodeproj/*" 2>/dev/null | head -1)
111
+ proj=$(find "$REPO" -maxdepth 2 -name "*.xcodeproj" -not -path "*/.*" 2>/dev/null | head -1)
112
+ if [ -n "$ws" ]; then
113
+ CONTAINER="-workspace $ws"
114
+ CONTAINER_ARGV=(-workspace "$ws")
115
+ elif [ -n "$proj" ]; then
116
+ CONTAINER="-project $proj"
117
+ CONTAINER_ARGV=(-project "$proj")
118
+ fi
119
+ }
120
+
121
+ # Prune every dotted directory, not a hand-listed few. `.worktrees` is where the
122
+ # pipeline puts other tasks' checkouts: scanning it makes this run match a UI test
123
+ # belonging to somebody else's branch, and the name `worktrees` in a prune list
124
+ # does not match `.worktrees`.
125
+ swift_sources() {
126
+ find "$REPO" \
127
+ -name ".*" -type d -prune -o \
128
+ -type d \( -name .build -o -name Pods -o -name build -o -name DerivedData \
129
+ -o -name node_modules \) -prune -o \
130
+ -type f -name "*.swift" -print0 2>/dev/null
131
+ }
132
+
133
+ # The signal is XCUIApplication, not a directory called *UITests. In the
134
+ # reference app 477 files sit under a *UITests path and exactly 2 drive the UI;
135
+ # the other 475 are snapshot tests, which render a view and compare pixels
136
+ # without ever launching the app. Recording video around one of those produces a
137
+ # still frame and calls it a flow. XCUIApplication is the only API that drives
138
+ # another process's UI, which is precisely the precondition a flow recording has.
139
+ # Computed once into SOURCES by detect_ios, never memoised inside a function: the
140
+ # consumers read it through $(...) and a variable a subshell assigns is gone by
141
+ # the time the parent looks. It is a repo-wide grep, and running it twice put the
142
+ # intake probe at 13 seconds with the user waiting on a question.
143
+ ui_test_sources() {
144
+ printf '%s\n' "$SOURCES" | grep -v '^$'
145
+ }
146
+
147
+ # The target is the nearest ancestor directory whose name ends in Tests - the
148
+ # test bundle, by Apple's own template convention. Derived from the path because
149
+ # `xcodebuild -list` costs over a minute on a real workspace and detection runs
150
+ # while the user waits for a question.
151
+ target_of() {
152
+ printf '%s\n' "$1" | awk -F/ '{
153
+ for (i = NF - 1; i > 0; i--)
154
+ if (tolower($i) ~ /tests$/) { print $i; exit }
155
+ }'
156
+ }
157
+
158
+ ios_targets() {
159
+ ui_test_sources | while IFS= read -r f; do target_of "$f"; done | sort -u
160
+ }
161
+
162
+ detect_ios() {
163
+ resolve_ios_container
164
+ [ -n "$CONTAINER" ] || { TARGET_REASON="no xcodeproj or xcworkspace under $REPO"; return; }
165
+
166
+ SOURCES=$(swift_sources | xargs -0 grep -l "XCUIApplication" 2>/dev/null)
167
+ TARGETS=$(ios_targets | paste -sd, -)
168
+ [ -n "$TARGETS" ] || { TARGET_REASON="no swift test source drives XCUIApplication under $REPO"; return; }
169
+
170
+ # One candidate is an answer on its own; several are not, until a match names
171
+ # one. Reporting a single target when there are several would be the same guess
172
+ # with a more confident face on it.
173
+ case "$TARGETS" in
174
+ *,*) TARGET="" ;;
175
+ *) TARGET="$TARGETS" ;;
176
+ esac
177
+
178
+ [ -n "$CHANGED" ] || { MATCH_REASON="no changed-file list supplied"; return; }
179
+
180
+ local names hits=""
181
+ names=$(changed_names)
182
+ [ -n "$names" ] || { MATCH_REASON="changed-file list held no usable names"; return; }
183
+
184
+ while IFS= read -r src; do
185
+ [ -n "$src" ] || continue
186
+ local tgt cls
187
+ tgt=$(target_of "$src")
188
+ [ -n "$tgt" ] || continue
189
+ cls=$(grep -oE 'class[[:space:]]+[A-Za-z0-9_]+' "$src" 2>/dev/null | head -1 | awk '{print $2}')
190
+ [ -n "$cls" ] || continue
191
+ while IFS= read -r n; do
192
+ [ -n "$n" ] || continue
193
+ if grep -qF "$n" "$src" 2>/dev/null; then
194
+ case ",$hits," in *",$tgt/$cls,"*) ;; *) hits="${hits:+$hits,}$tgt/$cls" ;; esac
195
+ break
196
+ fi
197
+ done <<EOF
198
+ $names
199
+ EOF
200
+ done <<EOF
201
+ $(ui_test_sources)
202
+ EOF
203
+
204
+ MATCHES="$hits"
205
+ if [ -n "$MATCHES" ]; then
206
+ [ -n "$TARGET" ] || TARGET="${MATCHES%%/*}"
207
+ else
208
+ MATCH_REASON="no UI test class mentions any changed file name"
209
+ [ -n "$TARGET" ] || TARGET_REASON="$(printf '%s\n' "$TARGETS" | tr ',' '\n' | wc -l | tr -d ' ') candidates and no match to choose between them"
210
+ fi
211
+ }
212
+
213
+ android_test_dirs() {
214
+ find "$REPO" \
215
+ -name ".*" -type d -prune -o \
216
+ -type d -name build -prune -o \
217
+ -type d -name "androidTest" -print 2>/dev/null
218
+ }
219
+
220
+ # module path -> Gradle task path, e.g. feature/auth/impl -> :feature:auth:impl
221
+ android_module_of() {
222
+ printf '%s\n' "${1#"$REPO"/}" | sed 's#/src/androidTest.*##; s#^#:#; s#/#:#g'
223
+ }
224
+
225
+ detect_android() {
226
+ local dirs
227
+ dirs=$(android_test_dirs)
228
+ [ -n "$dirs" ] || { TARGET_REASON="no src/androidTest source set under $REPO"; return; }
229
+
230
+ CONTAINER="$REPO"
231
+ TARGETS=$(printf '%s\n' "$dirs" | while IFS= read -r d; do
232
+ [ -n "$d" ] || continue
233
+ printf '%s:connectedAndroidTest\n' "$(android_module_of "$d")"
234
+ done | sort -u | paste -sd, -)
235
+
236
+ # Same rule as iOS: several candidates is not an answer until a match picks
237
+ # one. The reference app has eight instrumentation source sets.
238
+ case "$TARGETS" in
239
+ *,*) TARGET="" ;;
240
+ *) TARGET="$TARGETS"; SCHEME="${TARGET%:connectedAndroidTest}" ;;
241
+ esac
242
+
243
+ [ -n "$CHANGED" ] || { MATCH_REASON="no changed-file list supplied"; return; }
244
+
245
+ local names hits="" hit_target=""
246
+ names=$(changed_names)
247
+ [ -n "$names" ] || { MATCH_REASON="changed-file list held no usable names"; return; }
248
+
249
+ while IFS= read -r src; do
250
+ [ -n "$src" ] || continue
251
+ local cls pkg
252
+ cls=$(grep -oE 'class[[:space:]]+[A-Za-z0-9_]+' "$src" 2>/dev/null | head -1 | awk '{print $2}')
253
+ pkg=$(grep -oE '^package[[:space:]]+[A-Za-z0-9_.]+' "$src" 2>/dev/null | head -1 | awk '{print $2}')
254
+ [ -n "$cls" ] || continue
255
+ while IFS= read -r n; do
256
+ [ -n "$n" ] || continue
257
+ if grep -qF "$n" "$src" 2>/dev/null; then
258
+ local fq="${pkg:+$pkg.}$cls"
259
+ case ",$hits," in *",$fq,"*) ;; *) hits="${hits:+$hits,}$fq" ;; esac
260
+ # The module that owns the matched test is the one to run.
261
+ [ -n "$hit_target" ] || hit_target="$(android_module_of "$src"):connectedAndroidTest"
262
+ break
263
+ fi
264
+ done <<EOF
265
+ $names
266
+ EOF
267
+ done <<EOF
268
+ $(android_test_dirs | while IFS= read -r d; do find "$d" -type f \( -name "*.kt" -o -name "*.java" \) 2>/dev/null; done)
269
+ EOF
270
+
271
+ MATCHES="$hits"
272
+ if [ -n "$MATCHES" ]; then
273
+ [ -n "$TARGET" ] || { TARGET="$hit_target"; SCHEME="${TARGET%:connectedAndroidTest}"; }
274
+ else
275
+ MATCH_REASON="no instrumentation test class mentions any changed file name"
276
+ [ -n "$TARGET" ] || TARGET_REASON="$(printf '%s\n' "$TARGETS" | tr ',' '\n' | wc -l | tr -d ' ') candidates and no match to choose between them"
277
+ fi
278
+ }
279
+
280
+ case "$PLATFORM" in
281
+ ios) detect_ios ;;
282
+ android) detect_android ;;
283
+ esac
284
+
285
+ if [ "$MODE" = "detect" ]; then
286
+ printf 'UI_TEST_TARGET=%s\n' "$TARGET"
287
+ printf 'UI_TEST_TARGETS=%s\n' "$TARGETS"
288
+ printf 'UI_TEST_TARGET_REASON=%s\n' "$TARGET_REASON"
289
+ printf 'UI_TEST_MATCHES=%s\n' "$MATCHES"
290
+ printf 'UI_TEST_MATCH_REASON=%s\n' "$MATCH_REASON"
291
+ printf 'UI_TEST_CONTAINER=%s\n' "$CONTAINER"
292
+ printf 'UI_TEST_SCHEME=%s\n' "$SCHEME"
293
+ exit 0
294
+ fi
295
+
296
+ # run
297
+ #
298
+ # 4 and 3 are different facts and the caller acts on them differently: 4 means
299
+ # this repo has nothing to record a flow from, 3 means it does but nothing covers
300
+ # what changed. Keying 4 off TARGET rather than TARGETS conflated them, because
301
+ # TARGET is deliberately empty while several candidates exist and no match has
302
+ # chosen between them.
303
+ [ -n "$TARGETS$TARGET" ] || { echo "run-ui-tests: ${TARGET_REASON:-no ui test target}" >&2; exit 4; }
304
+ if [ -z "$MATCHES" ] && [ "$RUN_ALL" -eq 0 ]; then
305
+ echo "run-ui-tests: ${MATCH_REASON:-no matching test}" >&2
306
+ exit 3
307
+ fi
308
+ [ -n "$TARGET" ] || {
309
+ echo "run-ui-tests: ${TARGET_REASON:-several candidates and no match to choose between them}" >&2
310
+ exit 3
311
+ }
312
+
313
+ mkdir -p "$(dirname "$LOG")"
314
+
315
+ case "$PLATFORM" in
316
+ ios)
317
+ ONLY_ARGV=()
318
+ if [ "$RUN_ALL" -eq 0 ]; then
319
+ OLDIFS="$IFS"; IFS=','
320
+ for m in $MATCHES; do ONLY_ARGV+=("-only-testing:$m"); done
321
+ IFS="$OLDIFS"
322
+ else
323
+ ONLY_ARGV=("-only-testing:$TARGET")
324
+ fi
325
+ # The scheme is resolved here and not in detect: `xcodebuild -list` walks the
326
+ # SPM graph and can take over a minute on a real app, which detect cannot
327
+ # afford. Behind a timeout, because a resolver that hangs would otherwise hang
328
+ # the phase; on timeout fall back to the target name, which is the scheme name
329
+ # under Apple's own template.
330
+ if [ -z "$SCHEME" ]; then
331
+ TO=""
332
+ command -v timeout >/dev/null 2>&1 && TO="timeout 180"
333
+ # shellcheck disable=SC2086
334
+ LIST=$(cd "$REPO" && $TO xcodebuild -list -json "${CONTAINER_ARGV[@]}" 2>/dev/null)
335
+ SCHEME=$(printf '%s' "$LIST" | node -e '
336
+ let s=""; process.stdin.on("data",d=>s+=d).on("end",()=>{
337
+ let j={}; try{j=JSON.parse(s)}catch{ process.exit(0) }
338
+ const c=j.project||j.workspace||{};
339
+ const sch=c.schemes||[];
340
+ process.stdout.write(sch.find(n=>/uitests?$/i.test(n))||sch[0]||"");
341
+ });' 2>/dev/null)
342
+ [ -n "$SCHEME" ] || SCHEME="$TARGET"
343
+ fi
344
+
345
+ # `id=booted` is not a destination xcodebuild accepts, so resolve the udid
346
+ # rather than assigning a placeholder and hoping it is overwritten.
347
+ if [ -z "$DEVICE" ]; then
348
+ DEVICE=$(xcrun simctl list devices booted 2>/dev/null | grep -oE '[0-9A-F-]{36}' | head -1)
349
+ [ -n "$DEVICE" ] || { echo "run-ui-tests: no booted simulator and no --device" >&2; exit 2; }
350
+ fi
351
+ DEST="platform=iOS Simulator,id=$DEVICE"
352
+ (cd "$REPO" && xcodebuild test "${CONTAINER_ARGV[@]}" -scheme "$SCHEME" \
353
+ -destination "$DEST" "${ONLY_ARGV[@]}") >"$LOG" 2>&1
354
+ RC=$?
355
+ ;;
356
+ android)
357
+ command -v adb >/dev/null 2>&1 || { echo "run-ui-tests: adb unavailable" >&2; exit 2; }
358
+ adb shell true >/dev/null 2>&1 || { echo "run-ui-tests: no attached device" >&2; exit 2; }
359
+ GRADLE="$REPO/gradlew"
360
+ [ -x "$GRADLE" ] || { echo "run-ui-tests: no executable gradlew at $GRADLE" >&2; exit 2; }
361
+ ARGS=""
362
+ if [ "$RUN_ALL" -eq 0 ]; then
363
+ ARGS="-Pandroid.testInstrumentationRunnerArguments.class=$MATCHES"
364
+ fi
365
+ # shellcheck disable=SC2086
366
+ (cd "$REPO" && "$GRADLE" "$TARGET" $ARGS) >"$LOG" 2>&1
367
+ RC=$?
368
+ ;;
369
+ esac
370
+
371
+ printf 'UI_TEST_LOG=%s\n' "$LOG"
372
+ printf 'UI_TEST_SELECTED=%s\n' "${MATCHES:-$TARGET}"
373
+
374
+ # The exit code alone is not the verdict here for the same reason it is not one
375
+ # for the build: a runner that died before it reached the tests also exits
376
+ # non-zero, and a caller that reads only the code cannot tell "the UI test found
377
+ # a bug" from "the simulator never booted". The log is the evidence; the caller
378
+ # runs evidence-gate.mjs over it.
379
+ [ "$RC" -eq 0 ] && exit 0
380
+ exit 1
@@ -318,6 +318,15 @@ if git -C "$rp_pr" worktree remove "$rp_wt" 2>/dev/null; then
318
318
  ' 2>/dev/null || true
319
319
  fi
320
320
 
321
+ # The worktree is gone, but its parent `.worktrees/` stays behind as an empty
322
+ # directory and reads as leftover to anyone looking at the checkout. rmdir,
323
+ # never rm -rf: it fails harmlessly while any other task still has one.
324
+ _MA_HYG="$(dirname "${BASH_SOURCE[0]}")/../lib/repo-hygiene.sh"
325
+ if [ -f "$_MA_HYG" ]; then
326
+ . "$_MA_HYG"
327
+ ma_hygiene_prune_empty "$rp_pr"
328
+ fi
329
+
321
330
  emit
322
331
  exit 0
323
332
  fi