@mmerterden/multi-agent-pipeline 16.28.0 → 16.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -2
- package/README.md +4 -4
- package/README.tr.md +3 -3
- package/docs/architecture.md +3 -3
- package/docs/ecosystem.md +5 -5
- package/docs/features.md +14 -0
- package/install/claude.mjs +17 -0
- package/package.json +1 -1
- package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +93 -0
- package/pipeline/commands/multi-agent/design-check/SKILL.md +6 -5
- package/pipeline/commands/multi-agent/doctor/SKILL.md +78 -0
- package/pipeline/commands/multi-agent/help/SKILL.md +15 -12
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/setup/SKILL.md +14 -1
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -9
- package/pipeline/commands/multi-agent/update/SKILL.md +12 -0
- package/pipeline/lib/_jira-auth.sh +99 -0
- package/pipeline/lib/analysis-jira-write.sh +203 -0
- package/pipeline/lib/issue-fetcher.sh +4 -4
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/channels/pr.md +37 -1
- package/pipeline/multi-agent-refs/cross-cli-contract.md +3 -3
- package/pipeline/multi-agent-refs/features/analysis-jira.md +128 -0
- package/pipeline/multi-agent-refs/features/doctor.md +197 -0
- package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
- package/pipeline/multi-agent-refs/features/visual-evidence.md +103 -20
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +38 -7
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +13 -1
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +11 -1
- package/pipeline/multi-agent-refs/phases/phase-6-commit.md +23 -0
- package/pipeline/multi-agent-refs/picker-contract.md +35 -0
- package/pipeline/multi-agent-refs/tracker-contract.md +5 -1
- package/pipeline/preferences-template.json +1 -1
- package/pipeline/schemas/agent-state.schema.json +84 -1
- package/pipeline/schemas/analysis-spec.schema.json +336 -95
- package/pipeline/schemas/prefs.schema.json +80 -3
- package/pipeline/schemas/token-budget.json +10 -10
- package/pipeline/scripts/analysis-story-tree.mjs +441 -0
- package/pipeline/scripts/capture-evidence.sh +170 -5
- package/pipeline/scripts/doctor.mjs +758 -0
- package/pipeline/scripts/evidence-gate.mjs +31 -2
- package/pipeline/scripts/phase-tracker.sh +97 -17
- package/pipeline/scripts/probe-evidence-capability.sh +250 -0
- package/pipeline/scripts/run-ui-tests.sh +380 -0
- package/pipeline/scripts/scan-agent-config.sh +48 -10
- package/pipeline/scripts/skill-siblings.mjs +1 -1
- package/pipeline/skills/shared/core/multi-agent-analysis-jira/SKILL.md +94 -0
- package/pipeline/skills/shared/core/multi-agent-doctor/SKILL.md +79 -0
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +10 -1
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +13 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -6
- package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +18 -0
|
@@ -6,12 +6,19 @@
|
|
|
6
6
|
#
|
|
7
7
|
# Contract: multi-agent-refs/features/visual-evidence.md
|
|
8
8
|
#
|
|
9
|
-
#
|
|
9
|
+
# Three jobs, deliberately in one script because they share the naming convention
|
|
10
10
|
# that the Jira comment references by filename:
|
|
11
11
|
#
|
|
12
12
|
# capture-evidence.sh after --task <id> --platform <ios|android> --label <slug>
|
|
13
13
|
# Capture the current screen of the running app into the evidence dir.
|
|
14
14
|
#
|
|
15
|
+
# capture-evidence.sh video start|stop --task <id> --platform <ios|android> [--label <slug>]
|
|
16
|
+
# Writes <task>[-<label>]-flow.mp4; the label is optional because one
|
|
17
|
+
# recording per task is the common case.
|
|
18
|
+
# Record the screen while something else drives the app (a UI test run, or
|
|
19
|
+
# an MCP-driven flow). start and stop are separate because the recording
|
|
20
|
+
# wraps a run whose duration is not known in advance.
|
|
21
|
+
#
|
|
15
22
|
# capture-evidence.sh fit --file <path> [--max-mb <n>]
|
|
16
23
|
# capture-evidence.sh limits
|
|
17
24
|
# Print the resolved visualEvidence settings as KEY=VALUE lines.
|
|
@@ -120,6 +127,163 @@ case "$MODE" in
|
|
|
120
127
|
printf '%s\n' "$OUT"
|
|
121
128
|
;;
|
|
122
129
|
|
|
130
|
+
video)
|
|
131
|
+
# The flow recording. Shell, not MCP, for the same three reasons the "after"
|
|
132
|
+
# capture is: the screenshot path already shells out, a run whose host has no
|
|
133
|
+
# toolkit MCP registered still produces evidence, and Phase 3 / Phase 5 are
|
|
134
|
+
# exactly where an MCP call is contested. start and stop are separate because
|
|
135
|
+
# the recording wraps a test run whose duration is not known in advance - a
|
|
136
|
+
# fixed --duration either truncates the test or trails dead screen after it.
|
|
137
|
+
ACTION="${1:-}"
|
|
138
|
+
shift 2>/dev/null || true
|
|
139
|
+
# LABEL defaults to EMPTY, not to "flow": the suffix is already "-flow", so a
|
|
140
|
+
# default of "flow" produced <task>-flow-flow.mp4 while every renderer cites
|
|
141
|
+
# <task>-flow.mp4. One recording per task is the common case and it should not
|
|
142
|
+
# have to name itself twice to get the filename the contract quotes.
|
|
143
|
+
TASK=""; PLATFORM=""; LABEL=""
|
|
144
|
+
while [ "$#" -gt 0 ]; do
|
|
145
|
+
case "$1" in
|
|
146
|
+
--task) TASK="${2:-}"; shift 2 ;;
|
|
147
|
+
--platform) PLATFORM="${2:-}"; shift 2 ;;
|
|
148
|
+
--label) LABEL="${2:-}"; shift 2 ;;
|
|
149
|
+
*) echo "capture-evidence: unknown option $1" >&2; exit 2 ;;
|
|
150
|
+
esac
|
|
151
|
+
done
|
|
152
|
+
case "$ACTION" in start | stop) ;; *)
|
|
153
|
+
echo "usage: capture-evidence.sh video start|stop --task <id> --platform <ios|android> [--label <slug>]" >&2
|
|
154
|
+
exit 2 ;;
|
|
155
|
+
esac
|
|
156
|
+
[ -n "$TASK" ] && [ -n "$PLATFORM" ] || {
|
|
157
|
+
echo "usage: capture-evidence.sh video start|stop --task <id> --platform <ios|android> [--label <slug>]" >&2
|
|
158
|
+
exit 2
|
|
159
|
+
}
|
|
160
|
+
case "$PLATFORM" in ios | android) ;; *)
|
|
161
|
+
echo "capture-evidence: unsupported platform '$PLATFORM'" >&2; exit 2 ;;
|
|
162
|
+
esac
|
|
163
|
+
|
|
164
|
+
mkdir -p "$EVIDENCE_DIR"
|
|
165
|
+
SLUG="${TASK}${LABEL:+-$LABEL}"
|
|
166
|
+
OUT="$EVIDENCE_DIR/${SLUG}-flow.mp4"
|
|
167
|
+
PIDFILE="$EVIDENCE_DIR/.${SLUG}.recpid"
|
|
168
|
+
WATCHFILE="$EVIDENCE_DIR/.${SLUG}.watchpid"
|
|
169
|
+
REMOTE="/sdcard/_ma_${SLUG}.mp4"
|
|
170
|
+
|
|
171
|
+
# screenrecord's own ceiling is 180s and it is not negotiable, so a
|
|
172
|
+
# maxVideoSeconds above it would silently become 180 on Android and stay
|
|
173
|
+
# honoured on iOS - two platforms disagreeing about one preference. Clamp in
|
|
174
|
+
# one place and say so.
|
|
175
|
+
CAP="$MAX_VIDEO_SECONDS"
|
|
176
|
+
if [ "$CAP" -gt 180 ] 2>/dev/null; then
|
|
177
|
+
echo "capture-evidence: maxVideoSeconds $CAP clamped to 180 (screenrecord ceiling)" >&2
|
|
178
|
+
CAP=180
|
|
179
|
+
fi
|
|
180
|
+
|
|
181
|
+
if [ "$ACTION" = "start" ]; then
|
|
182
|
+
[ -f "$PIDFILE" ] && { echo "capture-evidence: a recording for $SLUG is already running" >&2; exit 2; }
|
|
183
|
+
rm -f "$OUT"
|
|
184
|
+
case "$PLATFORM" in
|
|
185
|
+
ios)
|
|
186
|
+
command -v xcrun >/dev/null 2>&1 || { echo "capture-evidence: xcrun unavailable" >&2; exit 4; }
|
|
187
|
+
xcrun simctl list devices booted 2>/dev/null | grep -q "(Booted)" \
|
|
188
|
+
|| { echo "capture-evidence: no booted simulator" >&2; exit 4; }
|
|
189
|
+
# h264 rather than the hevc default: an hevc mp4 does not play in the
|
|
190
|
+
# Jira attachment preview or in several browsers, which turns the
|
|
191
|
+
# artefact into a download nobody opens.
|
|
192
|
+
xcrun simctl io booted recordVideo --codec h264 --force "$OUT" >/dev/null 2>&1 &
|
|
193
|
+
REC_PID=$!
|
|
194
|
+
;;
|
|
195
|
+
android)
|
|
196
|
+
command -v adb >/dev/null 2>&1 || { echo "capture-evidence: adb unavailable" >&2; exit 4; }
|
|
197
|
+
adb shell true >/dev/null 2>&1 || { echo "capture-evidence: no attached device" >&2; exit 4; }
|
|
198
|
+
adb shell rm -f "$REMOTE" >/dev/null 2>&1 || true
|
|
199
|
+
adb shell screenrecord --time-limit "$CAP" "$REMOTE" >/dev/null 2>&1 &
|
|
200
|
+
REC_PID=$!
|
|
201
|
+
;;
|
|
202
|
+
esac
|
|
203
|
+
|
|
204
|
+
# A recorder that dies on the first frame leaves a pid file and an empty
|
|
205
|
+
# path, and the caller then "stops" a recording that never ran. Give it a
|
|
206
|
+
# beat and check it actually started.
|
|
207
|
+
sleep 1
|
|
208
|
+
kill -0 "$REC_PID" 2>/dev/null || {
|
|
209
|
+
echo "capture-evidence: recorder exited immediately" >&2
|
|
210
|
+
exit 4
|
|
211
|
+
}
|
|
212
|
+
# On Android the local process is the adb CLIENT, which stays alive even
|
|
213
|
+
# when screenrecord failed on the device, so liveness there proves only
|
|
214
|
+
# that adb is running. The remote file existing is what proves a recording
|
|
215
|
+
# began.
|
|
216
|
+
if [ "$PLATFORM" = "android" ]; then
|
|
217
|
+
adb shell "[ -e '$REMOTE' ]" >/dev/null 2>&1 || {
|
|
218
|
+
kill "$REC_PID" 2>/dev/null || true
|
|
219
|
+
echo "capture-evidence: screenrecord did not start on the device" >&2
|
|
220
|
+
exit 4
|
|
221
|
+
}
|
|
222
|
+
fi
|
|
223
|
+
printf '%s\n' "$REC_PID" > "$PIDFILE"
|
|
224
|
+
|
|
225
|
+
# iOS has no --time-limit, so the cap is ours to enforce. Without this a
|
|
226
|
+
# hung UI test records until the disk complains.
|
|
227
|
+
if [ "$PLATFORM" = "ios" ]; then
|
|
228
|
+
(sleep "$CAP"; kill -INT "$REC_PID" 2>/dev/null || true) >/dev/null 2>&1 &
|
|
229
|
+
printf '%s\n' "$!" > "$WATCHFILE"
|
|
230
|
+
fi
|
|
231
|
+
printf '%s\n' "$OUT"
|
|
232
|
+
exit 0
|
|
233
|
+
fi
|
|
234
|
+
|
|
235
|
+
# stop
|
|
236
|
+
[ -f "$PIDFILE" ] || { echo "capture-evidence: no recording in progress for $SLUG" >&2; exit 2; }
|
|
237
|
+
REC_PID=$(cat "$PIDFILE" 2>/dev/null)
|
|
238
|
+
rm -f "$PIDFILE"
|
|
239
|
+
if [ -f "$WATCHFILE" ]; then
|
|
240
|
+
kill "$(cat "$WATCHFILE" 2>/dev/null)" 2>/dev/null || true
|
|
241
|
+
rm -f "$WATCHFILE"
|
|
242
|
+
fi
|
|
243
|
+
|
|
244
|
+
case "$PLATFORM" in
|
|
245
|
+
ios)
|
|
246
|
+
# SIGINT, not SIGTERM: simctl only writes the moov atom and closes the
|
|
247
|
+
# container on INT. A TERM leaves an mp4 that every player refuses.
|
|
248
|
+
kill -INT "$REC_PID" 2>/dev/null || true
|
|
249
|
+
i=0
|
|
250
|
+
while kill -0 "$REC_PID" 2>/dev/null && [ "$i" -lt 100 ]; do sleep 0.1; i=$((i + 1)); done
|
|
251
|
+
kill -0 "$REC_PID" 2>/dev/null && kill -TERM "$REC_PID" 2>/dev/null || true
|
|
252
|
+
;;
|
|
253
|
+
android)
|
|
254
|
+
# Match on the output path, not on the program name: a bare
|
|
255
|
+
# `pkill screenrecord` stops every recording on the device, including a
|
|
256
|
+
# second pipeline run's and anything the user started by hand. Fall back
|
|
257
|
+
# to the blunt form only when -f is unavailable.
|
|
258
|
+
adb shell pkill -2 -f "$REMOTE" >/dev/null 2>&1 \
|
|
259
|
+
|| adb shell pkill -2 screenrecord >/dev/null 2>&1 || true
|
|
260
|
+
# screenrecord finalises the container after the signal; pulling straight
|
|
261
|
+
# away yields a truncated file that looks like a successful capture.
|
|
262
|
+
sleep 2
|
|
263
|
+
adb pull "$REMOTE" "$OUT" >/dev/null 2>&1 || true
|
|
264
|
+
adb shell rm -f "$REMOTE" >/dev/null 2>&1 || true
|
|
265
|
+
;;
|
|
266
|
+
esac
|
|
267
|
+
|
|
268
|
+
[ -s "$OUT" ] || { echo "capture-evidence: recording produced no file" >&2; exit 4; }
|
|
269
|
+
|
|
270
|
+
# Both recorders encode on change, so a flow over a screen that never moved
|
|
271
|
+
# produces a valid two-frame mp4 a fraction of a second long. The file is not
|
|
272
|
+
# broken and must not be discarded, but it is not evidence of a flow either,
|
|
273
|
+
# and the duration is the only thing that can tell the two apart. Say so and
|
|
274
|
+
# let the caller record it; never assert this duration against wall clock,
|
|
275
|
+
# which is what makes a correct static capture look like a failure.
|
|
276
|
+
if command -v ffprobe >/dev/null 2>&1; then
|
|
277
|
+
SECS=$(ffprobe -v error -show_entries format=duration -of csv=p=0 "$OUT" 2>/dev/null)
|
|
278
|
+
case "$SECS" in
|
|
279
|
+
"" ) echo "capture-evidence: ffprobe could not read the recording; duration unknown" >&2 ;;
|
|
280
|
+
* ) awk -v d="$SECS" 'BEGIN { exit (d < 1.0) ? 0 : 1 }' \
|
|
281
|
+
&& echo "capture-evidence: recording is ${SECS}s - the screen did not change during it" >&2 ;;
|
|
282
|
+
esac
|
|
283
|
+
fi
|
|
284
|
+
printf '%s\n' "$OUT"
|
|
285
|
+
;;
|
|
286
|
+
|
|
123
287
|
fit)
|
|
124
288
|
FILE=""; MAX_MB="$MAX_ATTACH_MB"
|
|
125
289
|
while [ "$#" -gt 0 ]; do
|
|
@@ -174,14 +338,15 @@ case "$MODE" in
|
|
|
174
338
|
;;
|
|
175
339
|
|
|
176
340
|
limits)
|
|
177
|
-
# The
|
|
178
|
-
#
|
|
179
|
-
#
|
|
341
|
+
# The resolved settings as KEY=VALUE, so visualEvidence.maxVideoSeconds is one
|
|
342
|
+
# value read in one place rather than a number repeated across documents. The
|
|
343
|
+
# `video` mode above reads the same function, so the cap a caller prints and
|
|
344
|
+
# the cap the recorder enforces cannot drift apart.
|
|
180
345
|
prefs_visual
|
|
181
346
|
;;
|
|
182
347
|
|
|
183
348
|
*)
|
|
184
|
-
echo "usage: capture-evidence.sh after|fit|limits ..." >&2
|
|
349
|
+
echo "usage: capture-evidence.sh after|video|fit|limits ..." >&2
|
|
185
350
|
exit 2
|
|
186
351
|
;;
|
|
187
352
|
esac
|