@mmerterden/multi-agent-pipeline 16.28.0 → 16.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -2
- package/README.md +4 -4
- package/README.tr.md +3 -3
- package/docs/architecture.md +3 -3
- package/docs/ecosystem.md +5 -5
- package/docs/features.md +14 -0
- package/install/claude.mjs +17 -0
- package/package.json +1 -1
- package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +93 -0
- package/pipeline/commands/multi-agent/design-check/SKILL.md +6 -5
- package/pipeline/commands/multi-agent/doctor/SKILL.md +78 -0
- package/pipeline/commands/multi-agent/help/SKILL.md +15 -12
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/setup/SKILL.md +14 -1
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -9
- package/pipeline/commands/multi-agent/update/SKILL.md +12 -0
- package/pipeline/lib/_jira-auth.sh +99 -0
- package/pipeline/lib/analysis-jira-write.sh +203 -0
- package/pipeline/lib/issue-fetcher.sh +4 -4
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/channels/pr.md +37 -1
- package/pipeline/multi-agent-refs/cross-cli-contract.md +3 -3
- package/pipeline/multi-agent-refs/features/analysis-jira.md +128 -0
- package/pipeline/multi-agent-refs/features/doctor.md +197 -0
- package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
- package/pipeline/multi-agent-refs/features/visual-evidence.md +103 -20
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +38 -7
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +13 -1
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +11 -1
- package/pipeline/multi-agent-refs/phases/phase-6-commit.md +23 -0
- package/pipeline/multi-agent-refs/picker-contract.md +35 -0
- package/pipeline/multi-agent-refs/tracker-contract.md +5 -1
- package/pipeline/preferences-template.json +1 -1
- package/pipeline/schemas/agent-state.schema.json +84 -1
- package/pipeline/schemas/analysis-spec.schema.json +336 -95
- package/pipeline/schemas/prefs.schema.json +80 -3
- package/pipeline/schemas/token-budget.json +10 -10
- package/pipeline/scripts/analysis-story-tree.mjs +441 -0
- package/pipeline/scripts/capture-evidence.sh +170 -5
- package/pipeline/scripts/doctor.mjs +758 -0
- package/pipeline/scripts/evidence-gate.mjs +31 -2
- package/pipeline/scripts/phase-tracker.sh +97 -17
- package/pipeline/scripts/probe-evidence-capability.sh +250 -0
- package/pipeline/scripts/run-ui-tests.sh +380 -0
- package/pipeline/scripts/scan-agent-config.sh +48 -10
- package/pipeline/scripts/skill-siblings.mjs +1 -1
- package/pipeline/skills/shared/core/multi-agent-analysis-jira/SKILL.md +94 -0
- package/pipeline/skills/shared/core/multi-agent-doctor/SKILL.md +79 -0
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +10 -1
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +13 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -6
- package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +18 -0
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
#
|
|
3
|
+
# run-ui-tests.sh - find the repo's own UI test target, pick the tests that
|
|
4
|
+
# cover the files this task changed, and run them.
|
|
5
|
+
#
|
|
6
|
+
# Contract: multi-agent-refs/features/visual-evidence.md section 4 (video tier 1).
|
|
7
|
+
#
|
|
8
|
+
# Tier 1 of the flow video is "the repo's own UI test drove the screen". That is
|
|
9
|
+
# worth more than a generated flow for one reason: the test already runs in CI,
|
|
10
|
+
# so the recording shows a path somebody committed to keeping green. This script
|
|
11
|
+
# is the half that finds and runs it; capture-evidence.sh records around it.
|
|
12
|
+
#
|
|
13
|
+
# Two sub-commands, deliberately in one file so detection has exactly one
|
|
14
|
+
# implementation. The capability probe needs the same answer BEFORE the user is
|
|
15
|
+
# asked which test depth to run, and a second copy of "does this repo have a UI
|
|
16
|
+
# test target" is a second place for the answer to drift.
|
|
17
|
+
#
|
|
18
|
+
# run-ui-tests.sh detect --platform <ios|android> [--repo <path>] [--changed <f>[,<f>...]]
|
|
19
|
+
# Print KEY=VALUE lines and exit. Never builds, never runs a test.
|
|
20
|
+
# UI_TEST_TARGET=<name>| (empty until one candidate is chosen)
|
|
21
|
+
# UI_TEST_TARGETS=<name[,name...]> (every candidate found)
|
|
22
|
+
# UI_TEST_TARGET_REASON=<why it is empty>
|
|
23
|
+
# UI_TEST_MATCHES=<Target/Class[,Target/Class...]>|
|
|
24
|
+
# UI_TEST_MATCH_REASON=<why it is empty>
|
|
25
|
+
# UI_TEST_CONTAINER=<-project X|-workspace X>|
|
|
26
|
+
# UI_TEST_SCHEME=<name>|
|
|
27
|
+
#
|
|
28
|
+
# run-ui-tests.sh run --platform <ios|android> [--repo <path>] [--changed <f>...]
|
|
29
|
+
# [--device <udid|serial>] [--log <path>] [--all]
|
|
30
|
+
# Run the matching tests (or the whole UI suite with --all).
|
|
31
|
+
#
|
|
32
|
+
# Env:
|
|
33
|
+
# UI_TEST_LOG default "$PWD/.pipeline/ui-test.log"
|
|
34
|
+
#
|
|
35
|
+
# Exit:
|
|
36
|
+
# 0 ran, and the run passed
|
|
37
|
+
# 1 ran, and the run failed
|
|
38
|
+
# 2 usage / environment
|
|
39
|
+
# 3 a UI test target exists but nothing matches the changed files
|
|
40
|
+
# 4 no UI test target in this project
|
|
41
|
+
#
|
|
42
|
+
# 3 and 4 are reasons to fall to video tier 2, not failures. Only 1 is a red test.
|
|
43
|
+
set -uo pipefail
|
|
44
|
+
|
|
45
|
+
MODE="${1:-}"
|
|
46
|
+
shift 2>/dev/null || true
|
|
47
|
+
|
|
48
|
+
PLATFORM=""; REPO="$PWD"; CHANGED=""; DEVICE=""; LOG=""; RUN_ALL=0
|
|
49
|
+
while [ "$#" -gt 0 ]; do
|
|
50
|
+
case "$1" in
|
|
51
|
+
--platform) PLATFORM="${2:-}"; shift 2 ;;
|
|
52
|
+
--repo) REPO="${2:-}"; shift 2 ;;
|
|
53
|
+
--changed) CHANGED="${CHANGED:+$CHANGED,}${2:-}"; shift 2 ;;
|
|
54
|
+
--device) DEVICE="${2:-}"; shift 2 ;;
|
|
55
|
+
--log) LOG="${2:-}"; shift 2 ;;
|
|
56
|
+
--all) RUN_ALL=1; shift ;;
|
|
57
|
+
*) echo "run-ui-tests: unknown option $1" >&2; exit 2 ;;
|
|
58
|
+
esac
|
|
59
|
+
done
|
|
60
|
+
|
|
61
|
+
case "$MODE" in detect | run) ;; *)
|
|
62
|
+
echo "usage: run-ui-tests.sh detect|run --platform <ios|android> [--repo <path>] [--changed <f>]" >&2
|
|
63
|
+
exit 2 ;;
|
|
64
|
+
esac
|
|
65
|
+
case "$PLATFORM" in ios | android) ;; *)
|
|
66
|
+
echo "run-ui-tests: unsupported platform '$PLATFORM'" >&2; exit 2 ;;
|
|
67
|
+
esac
|
|
68
|
+
[ -d "$REPO" ] || { echo "run-ui-tests: no such repo directory: $REPO" >&2; exit 2; }
|
|
69
|
+
|
|
70
|
+
LOG="${LOG:-${UI_TEST_LOG:-$PWD/.pipeline/ui-test.log}}"
|
|
71
|
+
|
|
72
|
+
TARGET=""; TARGET_REASON=""; TARGETS=""; SOURCES=""
|
|
73
|
+
CONTAINER_ARGV=()
|
|
74
|
+
MATCHES=""; MATCH_REASON=""
|
|
75
|
+
CONTAINER=""; SCHEME=""
|
|
76
|
+
|
|
77
|
+
# The names a changed UI file can be known by inside a UI test: the type it
|
|
78
|
+
# declares, and the file's own basename. A UI test refers to a screen by its
|
|
79
|
+
# accessibility identifier far more often than by its type name, and identifiers
|
|
80
|
+
# are generated from the same names, so both land on the same string often enough
|
|
81
|
+
# to be worth trying. This is a heuristic and the caller is told so: an empty
|
|
82
|
+
# match set falls to tier 2 rather than claiming the screen is untested.
|
|
83
|
+
changed_names() {
|
|
84
|
+
# The trailing newline matters: `while read` returns non-zero on an
|
|
85
|
+
# unterminated final line and drops it, so a one-element list read this way
|
|
86
|
+
# yields nothing at all and every change looks like it has no testable name.
|
|
87
|
+
printf '%s\n' "$CHANGED" | tr ',' '\n' | while IFS= read -r f; do
|
|
88
|
+
[ -n "$f" ] || continue
|
|
89
|
+
b="${f##*/}"
|
|
90
|
+
printf '%s\n' "${b%.*}"
|
|
91
|
+
done | sort -u
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
# Detection reads the filesystem, never `xcodebuild -list`. On a real app
|
|
95
|
+
# workspace that call resolves the SPM graph first and took 72 seconds here,
|
|
96
|
+
# and detection runs at intake while the user is waiting for a question. The
|
|
97
|
+
# scheme it would have told us is only needed to RUN, so it is resolved there,
|
|
98
|
+
# behind a timeout.
|
|
99
|
+
resolve_ios_container() {
|
|
100
|
+
local ws proj
|
|
101
|
+
# CONTAINER is reported as text for the detect contract, but the RUN path uses
|
|
102
|
+
# CONTAINER_ARGV: a checkout under "~/My Projects/" splits an unquoted
|
|
103
|
+
# -project /path/with space into two arguments and xcodebuild is handed a
|
|
104
|
+
# project that does not exist.
|
|
105
|
+
# A standalone .xcworkspace wins, but *.xcodeproj/project.xcworkspace is the
|
|
106
|
+
# implicit one Xcode keeps inside every project. Passing that to -workspace
|
|
107
|
+
# builds a different, schemeless container, so it must never be treated as a
|
|
108
|
+
# workspace the repo chose.
|
|
109
|
+
ws=$(find "$REPO" -maxdepth 2 -name "*.xcworkspace" -not -path "*/.*" \
|
|
110
|
+
-not -path "*.xcodeproj/*" 2>/dev/null | head -1)
|
|
111
|
+
proj=$(find "$REPO" -maxdepth 2 -name "*.xcodeproj" -not -path "*/.*" 2>/dev/null | head -1)
|
|
112
|
+
if [ -n "$ws" ]; then
|
|
113
|
+
CONTAINER="-workspace $ws"
|
|
114
|
+
CONTAINER_ARGV=(-workspace "$ws")
|
|
115
|
+
elif [ -n "$proj" ]; then
|
|
116
|
+
CONTAINER="-project $proj"
|
|
117
|
+
CONTAINER_ARGV=(-project "$proj")
|
|
118
|
+
fi
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
# Prune every dotted directory, not a hand-listed few. `.worktrees` is where the
|
|
122
|
+
# pipeline puts other tasks' checkouts: scanning it makes this run match a UI test
|
|
123
|
+
# belonging to somebody else's branch, and the name `worktrees` in a prune list
|
|
124
|
+
# does not match `.worktrees`.
|
|
125
|
+
swift_sources() {
|
|
126
|
+
find "$REPO" \
|
|
127
|
+
-name ".*" -type d -prune -o \
|
|
128
|
+
-type d \( -name .build -o -name Pods -o -name build -o -name DerivedData \
|
|
129
|
+
-o -name node_modules \) -prune -o \
|
|
130
|
+
-type f -name "*.swift" -print0 2>/dev/null
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
# The signal is XCUIApplication, not a directory called *UITests. In the
|
|
134
|
+
# reference app 477 files sit under a *UITests path and exactly 2 drive the UI;
|
|
135
|
+
# the other 475 are snapshot tests, which render a view and compare pixels
|
|
136
|
+
# without ever launching the app. Recording video around one of those produces a
|
|
137
|
+
# still frame and calls it a flow. XCUIApplication is the only API that drives
|
|
138
|
+
# another process's UI, which is precisely the precondition a flow recording has.
|
|
139
|
+
# Computed once into SOURCES by detect_ios, never memoised inside a function: the
|
|
140
|
+
# consumers read it through $(...) and a variable a subshell assigns is gone by
|
|
141
|
+
# the time the parent looks. It is a repo-wide grep, and running it twice put the
|
|
142
|
+
# intake probe at 13 seconds with the user waiting on a question.
|
|
143
|
+
ui_test_sources() {
|
|
144
|
+
printf '%s\n' "$SOURCES" | grep -v '^$'
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
# The target is the nearest ancestor directory whose name ends in Tests - the
|
|
148
|
+
# test bundle, by Apple's own template convention. Derived from the path because
|
|
149
|
+
# `xcodebuild -list` costs over a minute on a real workspace and detection runs
|
|
150
|
+
# while the user waits for a question.
|
|
151
|
+
target_of() {
|
|
152
|
+
printf '%s\n' "$1" | awk -F/ '{
|
|
153
|
+
for (i = NF - 1; i > 0; i--)
|
|
154
|
+
if (tolower($i) ~ /tests$/) { print $i; exit }
|
|
155
|
+
}'
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
ios_targets() {
|
|
159
|
+
ui_test_sources | while IFS= read -r f; do target_of "$f"; done | sort -u
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
detect_ios() {
|
|
163
|
+
resolve_ios_container
|
|
164
|
+
[ -n "$CONTAINER" ] || { TARGET_REASON="no xcodeproj or xcworkspace under $REPO"; return; }
|
|
165
|
+
|
|
166
|
+
SOURCES=$(swift_sources | xargs -0 grep -l "XCUIApplication" 2>/dev/null)
|
|
167
|
+
TARGETS=$(ios_targets | paste -sd, -)
|
|
168
|
+
[ -n "$TARGETS" ] || { TARGET_REASON="no swift test source drives XCUIApplication under $REPO"; return; }
|
|
169
|
+
|
|
170
|
+
# One candidate is an answer on its own; several are not, until a match names
|
|
171
|
+
# one. Reporting a single target when there are several would be the same guess
|
|
172
|
+
# with a more confident face on it.
|
|
173
|
+
case "$TARGETS" in
|
|
174
|
+
*,*) TARGET="" ;;
|
|
175
|
+
*) TARGET="$TARGETS" ;;
|
|
176
|
+
esac
|
|
177
|
+
|
|
178
|
+
[ -n "$CHANGED" ] || { MATCH_REASON="no changed-file list supplied"; return; }
|
|
179
|
+
|
|
180
|
+
local names hits=""
|
|
181
|
+
names=$(changed_names)
|
|
182
|
+
[ -n "$names" ] || { MATCH_REASON="changed-file list held no usable names"; return; }
|
|
183
|
+
|
|
184
|
+
while IFS= read -r src; do
|
|
185
|
+
[ -n "$src" ] || continue
|
|
186
|
+
local tgt cls
|
|
187
|
+
tgt=$(target_of "$src")
|
|
188
|
+
[ -n "$tgt" ] || continue
|
|
189
|
+
cls=$(grep -oE 'class[[:space:]]+[A-Za-z0-9_]+' "$src" 2>/dev/null | head -1 | awk '{print $2}')
|
|
190
|
+
[ -n "$cls" ] || continue
|
|
191
|
+
while IFS= read -r n; do
|
|
192
|
+
[ -n "$n" ] || continue
|
|
193
|
+
if grep -qF "$n" "$src" 2>/dev/null; then
|
|
194
|
+
case ",$hits," in *",$tgt/$cls,"*) ;; *) hits="${hits:+$hits,}$tgt/$cls" ;; esac
|
|
195
|
+
break
|
|
196
|
+
fi
|
|
197
|
+
done <<EOF
|
|
198
|
+
$names
|
|
199
|
+
EOF
|
|
200
|
+
done <<EOF
|
|
201
|
+
$(ui_test_sources)
|
|
202
|
+
EOF
|
|
203
|
+
|
|
204
|
+
MATCHES="$hits"
|
|
205
|
+
if [ -n "$MATCHES" ]; then
|
|
206
|
+
[ -n "$TARGET" ] || TARGET="${MATCHES%%/*}"
|
|
207
|
+
else
|
|
208
|
+
MATCH_REASON="no UI test class mentions any changed file name"
|
|
209
|
+
[ -n "$TARGET" ] || TARGET_REASON="$(printf '%s\n' "$TARGETS" | tr ',' '\n' | wc -l | tr -d ' ') candidates and no match to choose between them"
|
|
210
|
+
fi
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
android_test_dirs() {
|
|
214
|
+
find "$REPO" \
|
|
215
|
+
-name ".*" -type d -prune -o \
|
|
216
|
+
-type d -name build -prune -o \
|
|
217
|
+
-type d -name "androidTest" -print 2>/dev/null
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
# module path -> Gradle task path, e.g. feature/auth/impl -> :feature:auth:impl
|
|
221
|
+
android_module_of() {
|
|
222
|
+
printf '%s\n' "${1#"$REPO"/}" | sed 's#/src/androidTest.*##; s#^#:#; s#/#:#g'
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
detect_android() {
|
|
226
|
+
local dirs
|
|
227
|
+
dirs=$(android_test_dirs)
|
|
228
|
+
[ -n "$dirs" ] || { TARGET_REASON="no src/androidTest source set under $REPO"; return; }
|
|
229
|
+
|
|
230
|
+
CONTAINER="$REPO"
|
|
231
|
+
TARGETS=$(printf '%s\n' "$dirs" | while IFS= read -r d; do
|
|
232
|
+
[ -n "$d" ] || continue
|
|
233
|
+
printf '%s:connectedAndroidTest\n' "$(android_module_of "$d")"
|
|
234
|
+
done | sort -u | paste -sd, -)
|
|
235
|
+
|
|
236
|
+
# Same rule as iOS: several candidates is not an answer until a match picks
|
|
237
|
+
# one. The reference app has eight instrumentation source sets.
|
|
238
|
+
case "$TARGETS" in
|
|
239
|
+
*,*) TARGET="" ;;
|
|
240
|
+
*) TARGET="$TARGETS"; SCHEME="${TARGET%:connectedAndroidTest}" ;;
|
|
241
|
+
esac
|
|
242
|
+
|
|
243
|
+
[ -n "$CHANGED" ] || { MATCH_REASON="no changed-file list supplied"; return; }
|
|
244
|
+
|
|
245
|
+
local names hits="" hit_target=""
|
|
246
|
+
names=$(changed_names)
|
|
247
|
+
[ -n "$names" ] || { MATCH_REASON="changed-file list held no usable names"; return; }
|
|
248
|
+
|
|
249
|
+
while IFS= read -r src; do
|
|
250
|
+
[ -n "$src" ] || continue
|
|
251
|
+
local cls pkg
|
|
252
|
+
cls=$(grep -oE 'class[[:space:]]+[A-Za-z0-9_]+' "$src" 2>/dev/null | head -1 | awk '{print $2}')
|
|
253
|
+
pkg=$(grep -oE '^package[[:space:]]+[A-Za-z0-9_.]+' "$src" 2>/dev/null | head -1 | awk '{print $2}')
|
|
254
|
+
[ -n "$cls" ] || continue
|
|
255
|
+
while IFS= read -r n; do
|
|
256
|
+
[ -n "$n" ] || continue
|
|
257
|
+
if grep -qF "$n" "$src" 2>/dev/null; then
|
|
258
|
+
local fq="${pkg:+$pkg.}$cls"
|
|
259
|
+
case ",$hits," in *",$fq,"*) ;; *) hits="${hits:+$hits,}$fq" ;; esac
|
|
260
|
+
# The module that owns the matched test is the one to run.
|
|
261
|
+
[ -n "$hit_target" ] || hit_target="$(android_module_of "$src"):connectedAndroidTest"
|
|
262
|
+
break
|
|
263
|
+
fi
|
|
264
|
+
done <<EOF
|
|
265
|
+
$names
|
|
266
|
+
EOF
|
|
267
|
+
done <<EOF
|
|
268
|
+
$(android_test_dirs | while IFS= read -r d; do find "$d" -type f \( -name "*.kt" -o -name "*.java" \) 2>/dev/null; done)
|
|
269
|
+
EOF
|
|
270
|
+
|
|
271
|
+
MATCHES="$hits"
|
|
272
|
+
if [ -n "$MATCHES" ]; then
|
|
273
|
+
[ -n "$TARGET" ] || { TARGET="$hit_target"; SCHEME="${TARGET%:connectedAndroidTest}"; }
|
|
274
|
+
else
|
|
275
|
+
MATCH_REASON="no instrumentation test class mentions any changed file name"
|
|
276
|
+
[ -n "$TARGET" ] || TARGET_REASON="$(printf '%s\n' "$TARGETS" | tr ',' '\n' | wc -l | tr -d ' ') candidates and no match to choose between them"
|
|
277
|
+
fi
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
case "$PLATFORM" in
|
|
281
|
+
ios) detect_ios ;;
|
|
282
|
+
android) detect_android ;;
|
|
283
|
+
esac
|
|
284
|
+
|
|
285
|
+
if [ "$MODE" = "detect" ]; then
|
|
286
|
+
printf 'UI_TEST_TARGET=%s\n' "$TARGET"
|
|
287
|
+
printf 'UI_TEST_TARGETS=%s\n' "$TARGETS"
|
|
288
|
+
printf 'UI_TEST_TARGET_REASON=%s\n' "$TARGET_REASON"
|
|
289
|
+
printf 'UI_TEST_MATCHES=%s\n' "$MATCHES"
|
|
290
|
+
printf 'UI_TEST_MATCH_REASON=%s\n' "$MATCH_REASON"
|
|
291
|
+
printf 'UI_TEST_CONTAINER=%s\n' "$CONTAINER"
|
|
292
|
+
printf 'UI_TEST_SCHEME=%s\n' "$SCHEME"
|
|
293
|
+
exit 0
|
|
294
|
+
fi
|
|
295
|
+
|
|
296
|
+
# run
|
|
297
|
+
#
|
|
298
|
+
# 4 and 3 are different facts and the caller acts on them differently: 4 means
|
|
299
|
+
# this repo has nothing to record a flow from, 3 means it does but nothing covers
|
|
300
|
+
# what changed. Keying 4 off TARGET rather than TARGETS conflated them, because
|
|
301
|
+
# TARGET is deliberately empty while several candidates exist and no match has
|
|
302
|
+
# chosen between them.
|
|
303
|
+
[ -n "$TARGETS$TARGET" ] || { echo "run-ui-tests: ${TARGET_REASON:-no ui test target}" >&2; exit 4; }
|
|
304
|
+
if [ -z "$MATCHES" ] && [ "$RUN_ALL" -eq 0 ]; then
|
|
305
|
+
echo "run-ui-tests: ${MATCH_REASON:-no matching test}" >&2
|
|
306
|
+
exit 3
|
|
307
|
+
fi
|
|
308
|
+
[ -n "$TARGET" ] || {
|
|
309
|
+
echo "run-ui-tests: ${TARGET_REASON:-several candidates and no match to choose between them}" >&2
|
|
310
|
+
exit 3
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
mkdir -p "$(dirname "$LOG")"
|
|
314
|
+
|
|
315
|
+
case "$PLATFORM" in
|
|
316
|
+
ios)
|
|
317
|
+
ONLY_ARGV=()
|
|
318
|
+
if [ "$RUN_ALL" -eq 0 ]; then
|
|
319
|
+
OLDIFS="$IFS"; IFS=','
|
|
320
|
+
for m in $MATCHES; do ONLY_ARGV+=("-only-testing:$m"); done
|
|
321
|
+
IFS="$OLDIFS"
|
|
322
|
+
else
|
|
323
|
+
ONLY_ARGV=("-only-testing:$TARGET")
|
|
324
|
+
fi
|
|
325
|
+
# The scheme is resolved here and not in detect: `xcodebuild -list` walks the
|
|
326
|
+
# SPM graph and can take over a minute on a real app, which detect cannot
|
|
327
|
+
# afford. Behind a timeout, because a resolver that hangs would otherwise hang
|
|
328
|
+
# the phase; on timeout fall back to the target name, which is the scheme name
|
|
329
|
+
# under Apple's own template.
|
|
330
|
+
if [ -z "$SCHEME" ]; then
|
|
331
|
+
TO=""
|
|
332
|
+
command -v timeout >/dev/null 2>&1 && TO="timeout 180"
|
|
333
|
+
# shellcheck disable=SC2086
|
|
334
|
+
LIST=$(cd "$REPO" && $TO xcodebuild -list -json "${CONTAINER_ARGV[@]}" 2>/dev/null)
|
|
335
|
+
SCHEME=$(printf '%s' "$LIST" | node -e '
|
|
336
|
+
let s=""; process.stdin.on("data",d=>s+=d).on("end",()=>{
|
|
337
|
+
let j={}; try{j=JSON.parse(s)}catch{ process.exit(0) }
|
|
338
|
+
const c=j.project||j.workspace||{};
|
|
339
|
+
const sch=c.schemes||[];
|
|
340
|
+
process.stdout.write(sch.find(n=>/uitests?$/i.test(n))||sch[0]||"");
|
|
341
|
+
});' 2>/dev/null)
|
|
342
|
+
[ -n "$SCHEME" ] || SCHEME="$TARGET"
|
|
343
|
+
fi
|
|
344
|
+
|
|
345
|
+
# `id=booted` is not a destination xcodebuild accepts, so resolve the udid
|
|
346
|
+
# rather than assigning a placeholder and hoping it is overwritten.
|
|
347
|
+
if [ -z "$DEVICE" ]; then
|
|
348
|
+
DEVICE=$(xcrun simctl list devices booted 2>/dev/null | grep -oE '[0-9A-F-]{36}' | head -1)
|
|
349
|
+
[ -n "$DEVICE" ] || { echo "run-ui-tests: no booted simulator and no --device" >&2; exit 2; }
|
|
350
|
+
fi
|
|
351
|
+
DEST="platform=iOS Simulator,id=$DEVICE"
|
|
352
|
+
(cd "$REPO" && xcodebuild test "${CONTAINER_ARGV[@]}" -scheme "$SCHEME" \
|
|
353
|
+
-destination "$DEST" "${ONLY_ARGV[@]}") >"$LOG" 2>&1
|
|
354
|
+
RC=$?
|
|
355
|
+
;;
|
|
356
|
+
android)
|
|
357
|
+
command -v adb >/dev/null 2>&1 || { echo "run-ui-tests: adb unavailable" >&2; exit 2; }
|
|
358
|
+
adb shell true >/dev/null 2>&1 || { echo "run-ui-tests: no attached device" >&2; exit 2; }
|
|
359
|
+
GRADLE="$REPO/gradlew"
|
|
360
|
+
[ -x "$GRADLE" ] || { echo "run-ui-tests: no executable gradlew at $GRADLE" >&2; exit 2; }
|
|
361
|
+
ARGS=""
|
|
362
|
+
if [ "$RUN_ALL" -eq 0 ]; then
|
|
363
|
+
ARGS="-Pandroid.testInstrumentationRunnerArguments.class=$MATCHES"
|
|
364
|
+
fi
|
|
365
|
+
# shellcheck disable=SC2086
|
|
366
|
+
(cd "$REPO" && "$GRADLE" "$TARGET" $ARGS) >"$LOG" 2>&1
|
|
367
|
+
RC=$?
|
|
368
|
+
;;
|
|
369
|
+
esac
|
|
370
|
+
|
|
371
|
+
printf 'UI_TEST_LOG=%s\n' "$LOG"
|
|
372
|
+
printf 'UI_TEST_SELECTED=%s\n' "${MATCHES:-$TARGET}"
|
|
373
|
+
|
|
374
|
+
# The exit code alone is not the verdict here for the same reason it is not one
|
|
375
|
+
# for the build: a runner that died before it reached the tests also exits
|
|
376
|
+
# non-zero, and a caller that reads only the code cannot tell "the UI test found
|
|
377
|
+
# a bug" from "the simulator never booted". The log is the evidence; the caller
|
|
378
|
+
# runs evidence-gate.mjs over it.
|
|
379
|
+
[ "$RC" -eq 0 ] && exit 0
|
|
380
|
+
exit 1
|
|
@@ -18,7 +18,24 @@ set -uo pipefail
|
|
|
18
18
|
|
|
19
19
|
# ROOT defaults to the repo root; SCAN_ROOT overrides it (used by the smoke test
|
|
20
20
|
# to point at a fixture tree of planted-bad configs).
|
|
21
|
-
|
|
21
|
+
#
|
|
22
|
+
# Two layouts, and this script ships into the second one. From the checkout, $0
|
|
23
|
+
# is <repo>/pipeline/scripts/... and `../..` is the repo. From an install it is
|
|
24
|
+
# ~/.claude/scripts/... and `../..` is the HOME directory, where none of the
|
|
25
|
+
# shipped config paths exist - the scan then found zero files and called itself
|
|
26
|
+
# clean. Same defect class as the one skill-siblings.mjs carried, so it gets the
|
|
27
|
+
# same two-candidate resolution: try the repo layout, fall back to the install.
|
|
28
|
+
if [ -n "${SCAN_ROOT:-}" ]; then
|
|
29
|
+
ROOT="$SCAN_ROOT"
|
|
30
|
+
else
|
|
31
|
+
_here="$(cd "$(dirname "$0")" && pwd)"
|
|
32
|
+
_repo="$(cd "$_here/../.." && pwd)"
|
|
33
|
+
if [ -d "$_repo/pipeline/agents" ]; then
|
|
34
|
+
ROOT="$_repo"
|
|
35
|
+
else
|
|
36
|
+
ROOT="$HOME/.claude"
|
|
37
|
+
fi
|
|
38
|
+
fi
|
|
22
39
|
|
|
23
40
|
HIGH=0; MED=0
|
|
24
41
|
high() { HIGH=$((HIGH+1)); echo " [HIGH] $1"; }
|
|
@@ -27,18 +44,39 @@ med() { MED=$((MED+1)); echo " [MEDIUM] $1"; }
|
|
|
27
44
|
# Shipped config surface (globs expanded safely; missing paths are skipped).
|
|
28
45
|
TARGETS=()
|
|
29
46
|
add() { [ -e "$1" ] && TARGETS+=("$1"); }
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
47
|
+
# The same surface under either layout: `pipeline/`-prefixed in the checkout,
|
|
48
|
+
# flat under ~/.claude once installed.
|
|
49
|
+
if [ -d "$ROOT/pipeline/agents" ]; then
|
|
50
|
+
add "$ROOT/install/templates/claude-hooks.json"
|
|
51
|
+
add "$ROOT/install/templates/copilot-instructions.md"
|
|
52
|
+
add "$ROOT/pipeline/preferences-template.json"
|
|
53
|
+
for f in "$ROOT"/pipeline/agents/*.md; do add "$f"; done
|
|
54
|
+
# (figma component skills + their scripts now live in the ai-*-toolkit
|
|
55
|
+
# marketplace plugin, not in this repo, so there is nothing figma to scan here.)
|
|
56
|
+
add "$ROOT/pipeline/scripts/agent-guard.sh"
|
|
57
|
+
add "$ROOT/pipeline/scripts/pre-commit-check.sh"
|
|
58
|
+
else
|
|
59
|
+
# settings.json is deliberately NOT in this list. It is the USER's file, not
|
|
60
|
+
# something the pipeline ships, and their own permission choices are theirs to
|
|
61
|
+
# make - scanning it turned a personal setting into a release blocker.
|
|
62
|
+
for f in "$ROOT"/templates/*; do add "$f"; done
|
|
63
|
+
for f in "$ROOT"/agents/*.md; do add "$f"; done
|
|
64
|
+
add "$ROOT/scripts/agent-guard.sh"
|
|
65
|
+
add "$ROOT/scripts/pre-commit-check.sh"
|
|
66
|
+
fi
|
|
38
67
|
|
|
39
68
|
rel() { echo "${1#$ROOT/}"; }
|
|
40
69
|
|
|
41
|
-
echo "→ scanning ${#TARGETS[@]} shipped config files"
|
|
70
|
+
echo "→ scanning ${#TARGETS[@]} shipped config files (root: $ROOT)"
|
|
71
|
+
|
|
72
|
+
# Zero targets is a resolution failure, not a clean bill: `"${TARGETS[@]}"` on an
|
|
73
|
+
# empty array is also an unbound-variable error under bash 3.2 with `set -u`, so
|
|
74
|
+
# the script died here instead of saying what went wrong.
|
|
75
|
+
if [ "${#TARGETS[@]}" -eq 0 ]; then
|
|
76
|
+
echo " [HIGH] no shipped config file found under $ROOT - nothing was scanned"
|
|
77
|
+
echo "→ config hygiene: 1 HIGH, 0 MEDIUM"
|
|
78
|
+
exit 1
|
|
79
|
+
fi
|
|
42
80
|
|
|
43
81
|
for f in "${TARGETS[@]}"; do
|
|
44
82
|
[ -f "$f" ] || continue
|
|
@@ -60,7 +60,7 @@ const CMD_DIR = IN_REPO ? REPO_CMD_DIR : join(HOME, ".claude", "commands", "mult
|
|
|
60
60
|
const CORE_DIR = join(REPO_ROOT, "pipeline", "skills", "shared", "core");
|
|
61
61
|
const COPILOT_DIR = join(HOME, ".copilot", "skills");
|
|
62
62
|
const CODEX_DIR = join(HOME, ".codex", "multi-agent-refs", "commands");
|
|
63
|
-
// The dispatcher is the one skill Codex installs as a SKILL: the
|
|
63
|
+
// The dispatcher is the one skill Codex installs as a SKILL: the sub-commands
|
|
64
64
|
// become refs there, because Codex truncates a large skills block. Resolving it
|
|
65
65
|
// under the refs tree like the others reported a false "absent" - and a sibling
|
|
66
66
|
// check that invents a missing copy is worse than none, since it sends the next
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: multi-agent-analysis-jira
|
|
3
|
+
language: en
|
|
4
|
+
description: "Turn a rendered analysis document into a Jira story tree: derive stories from the document's own rule ids, check coverage both ways, preview every byte, then create only what does not already exist. Use when an analysis is final and the work needs tickets."
|
|
5
|
+
user-invocable: true
|
|
6
|
+
argument-hint: "[analysis.md] [--project KEY] - optional; with no argument, pick from the documents this run emitted"
|
|
7
|
+
not-for: create-jira, jira
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# multi-agent analysis-jira
|
|
11
|
+
|
|
12
|
+
**Input**: $ARGUMENTS
|
|
13
|
+
|
|
14
|
+
Reads an analysis document as a work breakdown and creates the tree. It never
|
|
15
|
+
invents a story: every node comes from an id the document defines.
|
|
16
|
+
|
|
17
|
+
Contract, severities and reasoning:
|
|
18
|
+
`$HOME/.claude/multi-agent-refs/features/analysis-jira.md`.
|
|
19
|
+
|
|
20
|
+
## Phase 1 - Plan, offline
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
node "$HOME/.claude/scripts/analysis-story-tree.mjs" "<analysis.md>" --json > /tmp/ma-plan.json
|
|
24
|
+
node "$HOME/.claude/scripts/analysis-story-tree.mjs" "<analysis.md>"
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Exit 4 means the document still carries an open placeholder. Stop and report it:
|
|
28
|
+
the step is `/multi-agent:analysis-resolve`, and a tree built from an open
|
|
29
|
+
question publishes the gap as work somebody is now assigned.
|
|
30
|
+
|
|
31
|
+
Exit 2 means nothing could be derived. Say so; do not improvise a tree.
|
|
32
|
+
|
|
33
|
+
## Phase 2 - Preview, in full
|
|
34
|
+
|
|
35
|
+
Show the human-readable output verbatim, in `outputLanguage`. It already carries
|
|
36
|
+
what makes a preview meaningful:
|
|
37
|
+
|
|
38
|
+
- every node, its source ids and its identity label
|
|
39
|
+
- the coverage verdict **on its own line**
|
|
40
|
+
- every field beside the pref key it came from, so a wrong setting shows here
|
|
41
|
+
rather than in Jira afterwards
|
|
42
|
+
- the write count
|
|
43
|
+
|
|
44
|
+
Then the dry run, which is what proves the writer agrees with the plan:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
bash "$HOME/.claude/lib/analysis-jira-write.sh" \
|
|
48
|
+
--plan /tmp/ma-plan.json --project "<KEY>" --dry-run
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Phase 3 - Approve
|
|
52
|
+
|
|
53
|
+
One `AskUserQuestion`. When the verdict is `ok` or `incomplete`:
|
|
54
|
+
|
|
55
|
+
- **Create the tree** - proceed
|
|
56
|
+
- **Show a node in full** - print one node's body, ask again
|
|
57
|
+
- **Cancel** - stop, write nothing
|
|
58
|
+
|
|
59
|
+
When the verdict is `unverifiable`, the approve option is **replaced**, never
|
|
60
|
+
reworded:
|
|
61
|
+
|
|
62
|
+
- **Create it, unverified** - the tree will be written and recorded as unchecked
|
|
63
|
+
- **Show a node in full**
|
|
64
|
+
- **Cancel**
|
|
65
|
+
|
|
66
|
+
A run that could not be checked is allowed. One that looks checked when it was
|
|
67
|
+
not is the defect, so the approval has to name it and a plain "Approve" must not
|
|
68
|
+
be reachable.
|
|
69
|
+
|
|
70
|
+
`incomplete` prints its uncovered and invented ids before the question. An
|
|
71
|
+
invented id means the plan cites something the document does not define - treat
|
|
72
|
+
that as a defect in the plan, not a warning to click past.
|
|
73
|
+
|
|
74
|
+
## Phase 4 - Write
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
bash "$HOME/.claude/lib/analysis-jira-write.sh" \
|
|
78
|
+
--plan /tmp/ma-plan.json --project "<KEY>"
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
It searches by label first and skips what exists. Re-running is safe and is the
|
|
82
|
+
intended recovery from any failure: the ledger makes a half-written tree
|
|
83
|
+
findable rather than duplicable.
|
|
84
|
+
|
|
85
|
+
**Never pass `--force`-like flags, and never update an existing node.** Jira has
|
|
86
|
+
no backup path for fields other than description.
|
|
87
|
+
|
|
88
|
+
## Phase 5 - Report
|
|
89
|
+
|
|
90
|
+
Keys created, keys skipped, the coverage verdict, and the ledger path. If the
|
|
91
|
+
verdict was `unverifiable`, say that in the report too - not only at approval
|
|
92
|
+
time, because the report is what gets pasted elsewhere.
|
|
93
|
+
|
|
94
|
+
**Stop. No worktree, no branch, no dev chain.**
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: multi-agent-doctor
|
|
3
|
+
language: en
|
|
4
|
+
description: "Health check for the installed pipeline: layout, preferences, credentials, hooks and host capabilities, each with one actionable step. Exit code is the verdict. Use when a run failed for an environmental reason, before a sync, or after an update."
|
|
5
|
+
user-invocable: true
|
|
6
|
+
argument-hint: "[--probe] [--explain] [--json] [--list-checks] - optional; --probe also makes one network request per configured service"
|
|
7
|
+
not-for: scan, setup
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
# multi-agent doctor
|
|
11
|
+
|
|
12
|
+
**Input**: $ARGUMENTS
|
|
13
|
+
|
|
14
|
+
Answers one question about this machine: **would a run work here, and if not,
|
|
15
|
+
what is the single next step?**
|
|
16
|
+
|
|
17
|
+
Every failure it reports has already reached a user, and each one arrived the
|
|
18
|
+
same way - late, mid-run, after the pickers had been answered. A missing script
|
|
19
|
+
fails at the call. Malformed preferences fail after Phase 0 has asked five
|
|
20
|
+
questions. A token in a remote URL does not fail at all; it leaks.
|
|
21
|
+
|
|
22
|
+
## Run it
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
node "$HOME/.claude/scripts/doctor.mjs" $ARGUMENTS
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Then report the output in `outputLanguage`. Do not re-word the steps: each line
|
|
29
|
+
already carries one imperative step, and paraphrasing is how a step turns into
|
|
30
|
+
advice.
|
|
31
|
+
|
|
32
|
+
**Answer the one check a script cannot.** `task-tools` asks whether THIS session
|
|
33
|
+
carries `TaskCreate` / `TaskUpdate`, and only the agent can see its own tool
|
|
34
|
+
list. Pass what you know:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
node "$HOME/.claude/scripts/doctor.mjs" --task-tools=yes # they are in your tools
|
|
38
|
+
node "$HOME/.claude/scripts/doctor.mjs" --task-tools=no # they are not
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Without the flag that check prints `SKIP`, which is correct - reporting "absent"
|
|
42
|
+
from a script that never looked is the defect this check is about.
|
|
43
|
+
|
|
44
|
+
## The exit code is the product
|
|
45
|
+
|
|
46
|
+
| Code | Meaning |
|
|
47
|
+
|---|---|
|
|
48
|
+
| 0 | healthy |
|
|
49
|
+
| 1 | degraded - at least one WARN |
|
|
50
|
+
| 2 | blocked - a run will fail or leak |
|
|
51
|
+
| 3 | usage error |
|
|
52
|
+
| 4 | indeterminate - the layout did not resolve, so nothing was checked |
|
|
53
|
+
|
|
54
|
+
4 matters as much as 2. Without it, "I could not look" borrows the code for
|
|
55
|
+
"I looked and it is fine".
|
|
56
|
+
|
|
57
|
+
## What it does not do
|
|
58
|
+
|
|
59
|
+
It recommends; it never fixes. It does not rewrite a remote URL, edit
|
|
60
|
+
`settings.json`, or touch a credential. A remote carrying a token may be the only
|
|
61
|
+
credential that repo has, the remote may be a mirror a script depends on
|
|
62
|
+
verbatim, and this can run inside a checkout the user does not own.
|
|
63
|
+
|
|
64
|
+
For an embedded credential the honest step is **not** "hide it". A token that
|
|
65
|
+
reached `.git/config` is already burned - it is in the shell history and readable
|
|
66
|
+
by anything that can read the working tree. The step is to revoke it at its host.
|
|
67
|
+
|
|
68
|
+
## Flags
|
|
69
|
+
|
|
70
|
+
| Flag | Effect |
|
|
71
|
+
|---|---|
|
|
72
|
+
| `--probe` | also make one authenticated request per configured service, and start the MCP server to count its tools. Off by default: a network check is the user's decision to spend. |
|
|
73
|
+
| `--explain` | print the detail behind a finding (which scripts, which repos) |
|
|
74
|
+
| `--json` | the same verdict in machine form, same exit code |
|
|
75
|
+
| `--list-checks` | the check ids, one per line |
|
|
76
|
+
| `--task-tools=yes\|no` | answer the check only the caller can see |
|
|
77
|
+
|
|
78
|
+
Every check, its severity and its reasoning:
|
|
79
|
+
`$HOME/.claude/multi-agent-refs/features/doctor.md`.
|
|
@@ -43,5 +43,14 @@ Lets you switch to the task branch for manual testing in Xcode before the PR is
|
|
|
43
43
|
5. **Wait for the user's response**
|
|
44
44
|
|
|
45
45
|
6. **Based on the result**:
|
|
46
|
-
- **OK** →
|
|
46
|
+
- **OK** → an "ok" is a claim, and the pass has to be substantiated before it is recorded. First write `$WORKTREE/.pipeline/manual-test.json`, one entry per acceptance criterion taken from the analysis doc test plan, the plan tasks, or the user's own words:
|
|
47
|
+
```json
|
|
48
|
+
{"criteria":[{"spec":"<quote>","source":"analysis 15.2 | plan task 3 | user","observed":"<what was seen>","verdict":"pass|fail|not-tested","reason":"<required when not-tested>","screenshot":"<path or null>"}],"verdict":"passed|failed"}
|
|
49
|
+
```
|
|
50
|
+
then run the gate, adding `--require-screenshot` when `state.visualEvidence.required` is true:
|
|
51
|
+
```bash
|
|
52
|
+
node $HOME/.claude/scripts/evidence-gate.mjs --claim manual --status passed \
|
|
53
|
+
--evidence "$WORKTREE/.pipeline/manual-test.json"
|
|
54
|
+
```
|
|
55
|
+
Exit 1 means the "ok" is not accepted: name the criterion that lacks evidence and wait for the next reply. Exit 0 → recreate the worktree, proceed to Phase 6. Full contract: `$HOME/.claude/multi-agent-refs/phases/phase-5-test.md` step 5.
|
|
47
56
|
- **Fix needed** → recreate the worktree, apply the fix
|
|
@@ -5,6 +5,19 @@ description: "First-run setup wizard: keychain token discovery, Git Identity onb
|
|
|
5
5
|
user-invocable: true
|
|
6
6
|
---
|
|
7
7
|
|
|
8
|
+
## Step 0 - What is actually wrong
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
node "$HOME/.claude/scripts/doctor.mjs"
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Run it before the first question and again at the end. The first run turns setup
|
|
15
|
+
from a fixed script into a targeted one: there is no point asking for a token
|
|
16
|
+
that is already mapped and answering. The second run is the only honest way to
|
|
17
|
+
end - "setup complete" is a claim, and the doctor's exit code is the evidence for
|
|
18
|
+
or against it. Report both in `outputLanguage`, and if the second run still shows
|
|
19
|
+
a BLOCK, say so plainly rather than closing on the word "complete".
|
|
20
|
+
|
|
8
21
|
## Setup (Keychain Token + Git Identity Onboarding)
|
|
9
22
|
|
|
10
23
|
Self-contained setup - works with inline `security` commands, no external script **required**. If `$HOME/.claude/scripts/keychain-save.sh` exists, it can be used as an interactive alternative but is not mandatory. The `setup` command starts interactive onboarding.
|