@mmerterden/multi-agent-pipeline 16.28.0 → 16.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -2
- package/README.md +4 -4
- package/README.tr.md +3 -3
- package/docs/architecture.md +3 -3
- package/docs/ecosystem.md +5 -5
- package/docs/features.md +14 -0
- package/install/claude.mjs +17 -0
- package/package.json +1 -1
- package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +93 -0
- package/pipeline/commands/multi-agent/design-check/SKILL.md +6 -5
- package/pipeline/commands/multi-agent/doctor/SKILL.md +78 -0
- package/pipeline/commands/multi-agent/help/SKILL.md +15 -12
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/setup/SKILL.md +14 -1
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -9
- package/pipeline/commands/multi-agent/update/SKILL.md +12 -0
- package/pipeline/lib/_jira-auth.sh +99 -0
- package/pipeline/lib/analysis-jira-write.sh +203 -0
- package/pipeline/lib/issue-fetcher.sh +4 -4
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/channels/pr.md +37 -1
- package/pipeline/multi-agent-refs/cross-cli-contract.md +3 -3
- package/pipeline/multi-agent-refs/features/analysis-jira.md +128 -0
- package/pipeline/multi-agent-refs/features/doctor.md +197 -0
- package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
- package/pipeline/multi-agent-refs/features/visual-evidence.md +103 -20
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +38 -7
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +13 -1
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +11 -1
- package/pipeline/multi-agent-refs/phases/phase-6-commit.md +23 -0
- package/pipeline/multi-agent-refs/picker-contract.md +35 -0
- package/pipeline/multi-agent-refs/tracker-contract.md +5 -1
- package/pipeline/preferences-template.json +1 -1
- package/pipeline/schemas/agent-state.schema.json +84 -1
- package/pipeline/schemas/analysis-spec.schema.json +336 -95
- package/pipeline/schemas/prefs.schema.json +80 -3
- package/pipeline/schemas/token-budget.json +10 -10
- package/pipeline/scripts/analysis-story-tree.mjs +441 -0
- package/pipeline/scripts/capture-evidence.sh +170 -5
- package/pipeline/scripts/doctor.mjs +758 -0
- package/pipeline/scripts/evidence-gate.mjs +31 -2
- package/pipeline/scripts/phase-tracker.sh +97 -17
- package/pipeline/scripts/probe-evidence-capability.sh +250 -0
- package/pipeline/scripts/run-ui-tests.sh +380 -0
- package/pipeline/scripts/scan-agent-config.sh +48 -10
- package/pipeline/scripts/skill-siblings.mjs +1 -1
- package/pipeline/skills/shared/core/multi-agent-analysis-jira/SKILL.md +94 -0
- package/pipeline/skills/shared/core/multi-agent-doctor/SKILL.md +79 -0
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +10 -1
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +13 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -6
- package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +18 -0
|
@@ -19,6 +19,9 @@
|
|
|
19
19
|
// manual is JSON-based (Phase 5 manual-test.json, see below)
|
|
20
20
|
// --status the claimed status (only "passed" is gated; others pass through)
|
|
21
21
|
// --evidence <path> the log / artifact that must substantiate a "passed" claim
|
|
22
|
+
// --require-screenshot manual claim only: every "pass" criterion must name a
|
|
23
|
+
// screenshot that exists on disk. Set by Phase 5 when
|
|
24
|
+
// state.visualEvidence.required is true.
|
|
22
25
|
// --success-pattern override the default success regex for the claim type (marker claims only)
|
|
23
26
|
// --failure-pattern override the default failure regex for the claim type (marker claims only)
|
|
24
27
|
//
|
|
@@ -120,7 +123,7 @@ function nonEmptyString(v) {
|
|
|
120
123
|
// Manual-test evidence is a structured document, not a log: every acceptance
|
|
121
124
|
// criterion must say what was observed and how it came out. A "fail" anywhere
|
|
122
125
|
// is decisive; an untested criterion is only acceptable when it says why.
|
|
123
|
-
function gateManual(claim, evidence, body) {
|
|
126
|
+
function gateManual(claim, evidence, body, requireScreenshot) {
|
|
124
127
|
let doc;
|
|
125
128
|
try {
|
|
126
129
|
doc = JSON.parse(body);
|
|
@@ -154,6 +157,32 @@ function gateManual(claim, evidence, body) {
|
|
|
154
157
|
`${claim} evidence shows a failed criterion "${failed.spec}" - pass claim contradicted by ${evidence}`,
|
|
155
158
|
);
|
|
156
159
|
}
|
|
160
|
+
// A screenshot path is only evidence when a file is behind it. The field has
|
|
161
|
+
// been in the document shape since the gate shipped and was never read, so a
|
|
162
|
+
// criterion carrying "screenshot": null passed as a verified manual test on a
|
|
163
|
+
// UI change - which is the one case the picture was added for. Checked only
|
|
164
|
+
// when the caller says visual evidence is required for this run, because on a
|
|
165
|
+
// backend change there is nothing to photograph.
|
|
166
|
+
if (requireScreenshot) {
|
|
167
|
+
const shotless = criteria.find(
|
|
168
|
+
(item) => item.verdict === "pass" && !nonEmptyString(item.screenshot),
|
|
169
|
+
);
|
|
170
|
+
if (shotless) {
|
|
171
|
+
fail(
|
|
172
|
+
`${claim} evidence has passing criterion "${shotless.spec}" with no screenshot, and this run requires visual evidence: ${evidence} (default-FAIL)`,
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
const missing = criteria.find(
|
|
176
|
+
(item) =>
|
|
177
|
+
item.verdict === "pass" && nonEmptyString(item.screenshot) && !existsSync(item.screenshot),
|
|
178
|
+
);
|
|
179
|
+
if (missing) {
|
|
180
|
+
fail(
|
|
181
|
+
`${claim} evidence names a screenshot that is not on disk: ${missing.screenshot} (default-FAIL)`,
|
|
182
|
+
);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
157
186
|
const untested = criteria.find(
|
|
158
187
|
(item) => item.verdict === "not-tested" && !nonEmptyString(item.reason),
|
|
159
188
|
);
|
|
@@ -203,7 +232,7 @@ if (body.trim().length === 0) {
|
|
|
203
232
|
}
|
|
204
233
|
|
|
205
234
|
if (JSON_CLAIMS.has(claim)) {
|
|
206
|
-
gateManual(claim, evidence, body);
|
|
235
|
+
gateManual(claim, evidence, body, args["require-screenshot"] === true);
|
|
207
236
|
}
|
|
208
237
|
|
|
209
238
|
const successRe = userRegExp(args["success-pattern"], DEFAULT_MARKERS[claim].success, "success");
|
|
@@ -570,10 +570,11 @@ tracker_next_hint() {
|
|
|
570
570
|
[ "${TRACKER_QUIET:-0}" = "1" ] && return 0
|
|
571
571
|
name=$(load_state | jq -r --arg id "$pid" '(.phases[] | select(.id == $id) | .name) // ""')
|
|
572
572
|
case "$(host_kind)" in
|
|
573
|
-
|
|
574
|
-
#
|
|
575
|
-
#
|
|
576
|
-
|
|
573
|
+
claude) mirror="TaskUpdate(\"$(subjects "$pid")\", status=\"$status\") - the subject is re-read from \`phase-tracker.sh subjects $pid\` so the widget carries the model, the elapsed time and the tokens; if TaskUpdate is not one of your tools, paste the card above into your reply VERBATIM in a code block (every phase on its own line, keep the elapsed and token columns; do not redraw or compact it)" ;;
|
|
574
|
+
# Not every session carries the task tools: Claude Code provides them by
|
|
575
|
+
# default only up to Opus 4.7 / Sonnet 4.6, a default that landed in
|
|
576
|
+
# v2.1.268. Naming the fallback on the same line is what keeps a newer model
|
|
577
|
+
# from advancing eight phases in silence.
|
|
577
578
|
codex) mirror="update_plan: set step \"Phase $pid $name\" to $status (send the FULL step list, it is not a delta)" ;;
|
|
578
579
|
*) mirror="no task widget on this host - reprint the card above inside your reply text" ;;
|
|
579
580
|
esac
|
|
@@ -664,6 +665,61 @@ EOF
|
|
|
664
665
|
printf '\n'
|
|
665
666
|
}
|
|
666
667
|
|
|
668
|
+
# The native widget renders one row per task and gives us exactly one string to
|
|
669
|
+
# fill it: the subject. So the numbers a reader wants - which model is spending,
|
|
670
|
+
# how long the phase has run, what it cost - have to travel IN the subject or not
|
|
671
|
+
# at all. `render` already computes all of it for the fallback card; this prints
|
|
672
|
+
# the same values in the one shape the host will accept.
|
|
673
|
+
#
|
|
674
|
+
# Every segment is omitted while it is empty, so a pending phase is just its name
|
|
675
|
+
# and a finished one carries its whole bill. Separator is ` - `, not a middle dot:
|
|
676
|
+
# a subject is a task title and travels through surfaces that are not a terminal.
|
|
677
|
+
subjects() {
|
|
678
|
+
need_jq
|
|
679
|
+
local state want="${1:-}"
|
|
680
|
+
state=$(load_state)
|
|
681
|
+
local prices_json='{"prices":{}}'
|
|
682
|
+
if [ -f "$COST_TABLE" ]; then
|
|
683
|
+
prices_json=$(cat "$COST_TABLE" 2>/dev/null) || prices_json='{"prices":{}}'
|
|
684
|
+
echo "$prices_json" | jq empty 2>/dev/null || prices_json='{"prices":{}}'
|
|
685
|
+
fi
|
|
686
|
+
local now_s; now_s=$(now_epoch)
|
|
687
|
+
local rows
|
|
688
|
+
rows=$(echo "$state" | jq -r --argjson prices "$prices_json" "$COST_JQ_DEFS"'
|
|
689
|
+
def usd_of(p): cost_usd_of($prices.prices[p.model // ""] // null; (p.tokens_in // 0); (p.tokens_out // 0); (p.tokens_cached // 0));
|
|
690
|
+
.phases // []
|
|
691
|
+
| sort_by(.id | (tonumber? // 9999))
|
|
692
|
+
| .[] | [
|
|
693
|
+
(.id // ""), (.name // ""), (.status // ""),
|
|
694
|
+
(.started_at // ""), (.completed_at // ""), (.model // ""),
|
|
695
|
+
(((.tokens_in // 0) + (.tokens_out // 0)) | tostring),
|
|
696
|
+
(usd_of(.) | if . == null then "" else (((. * 100) | floor) / 100 | tostring) end)
|
|
697
|
+
] | join("\u001f")')
|
|
698
|
+
local pid pname pstatus p_start p_end pmodel ptok pusd
|
|
699
|
+
while IFS=$'\x1f' read -r pid pname pstatus p_start p_end pmodel ptok pusd; do
|
|
700
|
+
[ -n "$pid" ] || continue
|
|
701
|
+
[ -z "$want" ] || [ "$want" = "$pid" ] || continue
|
|
702
|
+
local line="Phase $pid $pname"
|
|
703
|
+
[ -n "$pmodel" ] && line="$line - $pmodel"
|
|
704
|
+
local s_ep e_ep el=""
|
|
705
|
+
s_ep=$(iso_to_epoch "$p_start")
|
|
706
|
+
e_ep=$(iso_to_epoch "$p_end")
|
|
707
|
+
if [ -n "$s_ep" ]; then
|
|
708
|
+
case "$pstatus" in
|
|
709
|
+
completed|failed|skipped) [ -n "$e_ep" ] && el=$((e_ep - s_ep)) ;;
|
|
710
|
+
in_progress) el=$((now_s - s_ep)) ;;
|
|
711
|
+
esac
|
|
712
|
+
fi
|
|
713
|
+
if [ -n "$el" ] && [ "$el" -lt 0 ] 2>/dev/null; then el=0; fi
|
|
714
|
+
[ -n "$el" ] && line="$line - $(format_elapsed "$el")"
|
|
715
|
+
if [ "${ptok:-0}" -gt 0 ] 2>/dev/null; then
|
|
716
|
+
line="$line - $(format_tokens "$ptok") tok"
|
|
717
|
+
[ -n "$pusd" ] && line="$line - $(printf '~$%.2f' "$pusd" 2>/dev/null || echo '')"
|
|
718
|
+
fi
|
|
719
|
+
printf '%s\n' "$line"
|
|
720
|
+
done <<< "$rows"
|
|
721
|
+
}
|
|
722
|
+
|
|
667
723
|
render() {
|
|
668
724
|
need_jq
|
|
669
725
|
local state
|
|
@@ -1159,6 +1215,10 @@ GATE
|
|
|
1159
1215
|
fi
|
|
1160
1216
|
;;
|
|
1161
1217
|
|
|
1218
|
+
subjects)
|
|
1219
|
+
subjects "${1:-}"
|
|
1220
|
+
;;
|
|
1221
|
+
|
|
1162
1222
|
tiles)
|
|
1163
1223
|
need_jq
|
|
1164
1224
|
tiles_state=$(load_state)
|
|
@@ -1169,22 +1229,42 @@ GATE
|
|
|
1169
1229
|
}
|
|
1170
1230
|
case "$(host_kind)" in
|
|
1171
1231
|
claude)
|
|
1172
|
-
# The
|
|
1173
|
-
#
|
|
1174
|
-
#
|
|
1175
|
-
#
|
|
1176
|
-
#
|
|
1177
|
-
#
|
|
1178
|
-
#
|
|
1232
|
+
# The task tools are not in every session, and the reason is the model,
|
|
1233
|
+
# not the CLI version. Claude Code provides TaskCreate / TaskUpdate by
|
|
1234
|
+
# default only on Claude 3.x, Opus 4 through 4.7, Sonnet 4 through 4.6
|
|
1235
|
+
# and Haiku 4.5; on any newer model it leaves them out unless the user
|
|
1236
|
+
# opts in, and that default landed in v2.1.268. This contract was written
|
|
1237
|
+
# when the tools were universal and stayed true for years, so nothing
|
|
1238
|
+
# noticed: the tracker state was written correctly, every phase advanced,
|
|
1239
|
+
# and the screen stayed empty for a whole run.
|
|
1240
|
+
#
|
|
1241
|
+
# A shell script cannot probe the model's tool list, so the branch is
|
|
1242
|
+
# taken where the information actually lives - the model knows which
|
|
1243
|
+
# tools it has. And the fallback is not enough on its own: a user looking
|
|
1244
|
+
# at a missing widget needs the one command that brings it back, which is
|
|
1245
|
+
# why the opt-in is printed next to it.
|
|
1179
1246
|
echo "REQUIRED - create one native tile per phase, in this exact order,"
|
|
1180
1247
|
echo "BEFORE any TaskUpdate. The widget renders by creation order, not by"
|
|
1181
1248
|
echo "phase number, so an out-of-order call scrambles the stack."
|
|
1182
|
-
|
|
1249
|
+
subjects | sed 's/^/ TaskCreate(subject: "/; s/$/")/'
|
|
1250
|
+
echo
|
|
1251
|
+
echo "At every phase boundary re-run \`phase-tracker.sh subjects <id>\` and"
|
|
1252
|
+
echo "pass that line as the subject of the TaskUpdate. The subject is the only"
|
|
1253
|
+
echo "string the native widget renders, so the model, the elapsed time and the"
|
|
1254
|
+
echo "token count reach the user through it or not at all."
|
|
1183
1255
|
echo
|
|
1184
|
-
echo "IF TaskCreate IS NOT ONE OF YOUR TOOLS this
|
|
1185
|
-
echo "
|
|
1186
|
-
echo "
|
|
1187
|
-
echo "
|
|
1256
|
+
echo "IF TaskCreate IS NOT ONE OF YOUR TOOLS this session's model does not"
|
|
1257
|
+
echo "carry them (default since v2.1.268 on models newer than Opus 4.7 /"
|
|
1258
|
+
echo "Sonnet 4.6). Then the card below IS the widget, and at every phase"
|
|
1259
|
+
echo "boundary you run \`phase-tracker.sh render\` and paste its output"
|
|
1260
|
+
echo "into your reply VERBATIM, inside a code block - every phase on"
|
|
1261
|
+
echo "its own line, with the elapsed time, the token count and the total"
|
|
1262
|
+
echo "row exactly as printed. Do NOT redraw it, do not compact phases"
|
|
1263
|
+
echo "onto one line, do not drop the columns: those numbers are the"
|
|
1264
|
+
echo "whole reason the card is worth showing. Tool output is collapsed,"
|
|
1265
|
+
echo "so a card left in stdout never reaches the user."
|
|
1266
|
+
echo "Say once, in outputLanguage, that the native widget returns with:"
|
|
1267
|
+
echo " CLAUDE_CODE_ENABLE_TODO_TOOLS=1 claude"
|
|
1188
1268
|
echo
|
|
1189
1269
|
render
|
|
1190
1270
|
;;
|
|
@@ -1212,7 +1292,7 @@ GATE
|
|
|
1212
1292
|
;;
|
|
1213
1293
|
|
|
1214
1294
|
*)
|
|
1215
|
-
echo "phase-tracker: unknown action '$ACTION' (use init|add|update|sub|tokens|model|meta|now|cost|render)" >&2
|
|
1295
|
+
echo "phase-tracker: unknown action '$ACTION' (use init|add|update|sub|tokens|model|meta|now|cost|render|subjects)" >&2
|
|
1216
1296
|
exit 64
|
|
1217
1297
|
;;
|
|
1218
1298
|
esac
|
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
#
|
|
3
|
+
# probe-evidence-capability.sh - measure what visual evidence this machine and
|
|
4
|
+
# this repo can actually produce, BEFORE the user is asked to choose a test depth.
|
|
5
|
+
#
|
|
6
|
+
# Contract: multi-agent-refs/features/visual-evidence.md section 4.
|
|
7
|
+
#
|
|
8
|
+
# The order is probe, then question, then run. Offering "unit + UI test with a
|
|
9
|
+
# screen recording" and only then discovering there is no UI test target, or no
|
|
10
|
+
# booted simulator, spends the user's answer on something that cannot happen. So
|
|
11
|
+
# the options are built from what this prints.
|
|
12
|
+
#
|
|
13
|
+
# Three rules this file exists to keep:
|
|
14
|
+
#
|
|
15
|
+
# 1. Absence carries a reason. Every empty value is paired with a *_REASON, so
|
|
16
|
+
# a closed option can say why it is closed instead of vanishing from the menu.
|
|
17
|
+
# 2. Unmeasurable is `unknown`, never `false`. A probe that could not look and a
|
|
18
|
+
# probe that looked and found nothing are different facts, and collapsing
|
|
19
|
+
# them lets a missing tool read as a clean negative.
|
|
20
|
+
# 3. Detection is not reimplemented here. The UI test answer comes from
|
|
21
|
+
# run-ui-tests.sh detect, which is also what actually runs the tests; a
|
|
22
|
+
# second copy is a second place for the answer to drift.
|
|
23
|
+
#
|
|
24
|
+
# Usage:
|
|
25
|
+
# probe-evidence-capability.sh --platform <ios|android> [--repo <path>]
|
|
26
|
+
# [--changed <f>[,<f>...]] [--json]
|
|
27
|
+
# [--json-out <path>] [--only all|device]
|
|
28
|
+
#
|
|
29
|
+
# Output: KEY='VALUE' lines (eval-able; every value is shell-quoted), or a JSON object with --json (shaped for
|
|
30
|
+
# state.evidenceCapability). --json-out writes that JSON to a file while stdout
|
|
31
|
+
# stays KEY=VALUE, so one run serves both the shell that builds the menu and the
|
|
32
|
+
# state that records the measurement - two runs would mean two repo scans and
|
|
33
|
+
# two chances to disagree.
|
|
34
|
+
#
|
|
35
|
+
# --only device skips the UI-test detection. Phase 3 re-checks the device right
|
|
36
|
+
# before recording, and nothing else it would re-scan can have changed.
|
|
37
|
+
#
|
|
38
|
+
# Exit: 0 probed (whatever the verdicts), 2 usage.
|
|
39
|
+
# Never non-zero for an absent capability: absence is the finding.
|
|
40
|
+
set -uo pipefail
|
|
41
|
+
|
|
42
|
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
43
|
+
PLATFORM=""; REPO="$PWD"; CHANGED=""; AS_JSON=0; JSON_OUT=""; ONLY="all"
|
|
44
|
+
|
|
45
|
+
while [ "$#" -gt 0 ]; do
|
|
46
|
+
case "$1" in
|
|
47
|
+
--platform) PLATFORM="${2:-}"; shift 2 ;;
|
|
48
|
+
--repo) REPO="${2:-}"; shift 2 ;;
|
|
49
|
+
--changed) CHANGED="${CHANGED:+$CHANGED,}${2:-}"; shift 2 ;;
|
|
50
|
+
--json) AS_JSON=1; shift ;;
|
|
51
|
+
--json-out) JSON_OUT="${2:-}"; shift 2 ;;
|
|
52
|
+
--only) ONLY="${2:-all}"; shift 2 ;;
|
|
53
|
+
*) echo "probe-evidence-capability: unknown option $1" >&2; exit 2 ;;
|
|
54
|
+
esac
|
|
55
|
+
done
|
|
56
|
+
case "$ONLY" in all | device) ;; *)
|
|
57
|
+
echo "probe-evidence-capability: --only takes 'all' or 'device'" >&2; exit 2 ;;
|
|
58
|
+
esac
|
|
59
|
+
|
|
60
|
+
case "$PLATFORM" in ios | android) ;; *)
|
|
61
|
+
echo "usage: probe-evidence-capability.sh --platform <ios|android> [--repo <path>] [--changed <f>] [--json]" >&2
|
|
62
|
+
exit 2 ;;
|
|
63
|
+
esac
|
|
64
|
+
[ -d "$REPO" ] || { echo "probe-evidence-capability: no such repo directory: $REPO" >&2; exit 2; }
|
|
65
|
+
|
|
66
|
+
UI_TEST_TARGET=""; UI_TEST_TARGETS=""; UI_TEST_TARGET_REASON=""
|
|
67
|
+
UI_TEST_MATCHES=""; UI_TEST_MATCH_REASON=""
|
|
68
|
+
DEVICE=""; DEVICE_REASON=""
|
|
69
|
+
RECORDER="unknown"; RECORDER_REASON=""
|
|
70
|
+
MCP="unknown"; MCP_REASON=""
|
|
71
|
+
|
|
72
|
+
# ---- UI test target, delegated ------------------------------------------------
|
|
73
|
+
RUNNER="$HERE/run-ui-tests.sh"
|
|
74
|
+
if [ "$ONLY" = "device" ]; then
|
|
75
|
+
# Phase 3 re-checks the device right before recording, and only the device.
|
|
76
|
+
# Delegating detection there would re-walk the whole repo for a row that
|
|
77
|
+
# cannot have changed, which cost 5 seconds of a phase that is already holding
|
|
78
|
+
# a green build.
|
|
79
|
+
UI_TEST_TARGET_REASON="not probed (--only device)"
|
|
80
|
+
elif [ ! -x "$RUNNER" ] && [ ! -f "$RUNNER" ]; then
|
|
81
|
+
UI_TEST_TARGET_REASON="run-ui-tests.sh not found beside this script"
|
|
82
|
+
else
|
|
83
|
+
DETECT_ARGS=(detect --platform "$PLATFORM" --repo "$REPO")
|
|
84
|
+
[ -n "$CHANGED" ] && DETECT_ARGS+=(--changed "$CHANGED")
|
|
85
|
+
while IFS='=' read -r k v; do
|
|
86
|
+
case "$k" in
|
|
87
|
+
UI_TEST_TARGET) UI_TEST_TARGET="$v" ;;
|
|
88
|
+
UI_TEST_TARGETS) UI_TEST_TARGETS="$v" ;;
|
|
89
|
+
UI_TEST_TARGET_REASON) UI_TEST_TARGET_REASON="$v" ;;
|
|
90
|
+
UI_TEST_MATCHES) UI_TEST_MATCHES="$v" ;;
|
|
91
|
+
UI_TEST_MATCH_REASON) UI_TEST_MATCH_REASON="$v" ;;
|
|
92
|
+
esac
|
|
93
|
+
done < <(bash "$RUNNER" "${DETECT_ARGS[@]}" 2>/dev/null)
|
|
94
|
+
fi
|
|
95
|
+
|
|
96
|
+
# ---- Device -------------------------------------------------------------------
|
|
97
|
+
case "$PLATFORM" in
|
|
98
|
+
ios)
|
|
99
|
+
if ! command -v xcrun >/dev/null 2>&1; then
|
|
100
|
+
DEVICE_REASON="xcrun unavailable; cannot tell whether a simulator exists"
|
|
101
|
+
else
|
|
102
|
+
DEVICE=$(xcrun simctl list devices booted 2>/dev/null | grep -oE '[0-9A-F-]{36}' | head -1)
|
|
103
|
+
if [ -z "$DEVICE" ]; then
|
|
104
|
+
# Bootable is not booted, and the difference is a question the user can
|
|
105
|
+
# act on: one is "start your simulator", the other is "this machine has
|
|
106
|
+
# no iOS runtime installed".
|
|
107
|
+
if xcrun simctl list devices available 2>/dev/null | grep -q "([0-9A-F-]\{36\})"; then
|
|
108
|
+
DEVICE_REASON="no booted simulator, but one is available to boot"
|
|
109
|
+
else
|
|
110
|
+
DEVICE_REASON="no iOS simulator available on this machine"
|
|
111
|
+
fi
|
|
112
|
+
fi
|
|
113
|
+
fi
|
|
114
|
+
;;
|
|
115
|
+
android)
|
|
116
|
+
if ! command -v adb >/dev/null 2>&1; then
|
|
117
|
+
DEVICE_REASON="adb unavailable; cannot tell whether a device is attached"
|
|
118
|
+
else
|
|
119
|
+
DEVICE=$(adb devices 2>/dev/null | awk 'NR>1 && $2=="device" {print $1; exit}')
|
|
120
|
+
[ -n "$DEVICE" ] || DEVICE_REASON="no attached device or running emulator"
|
|
121
|
+
fi
|
|
122
|
+
;;
|
|
123
|
+
esac
|
|
124
|
+
|
|
125
|
+
# ---- Recorder -----------------------------------------------------------------
|
|
126
|
+
# The recorder is the same CLI the capture needs, so this answers "would
|
|
127
|
+
# capture-evidence.sh video start work" rather than "is some recorder installed".
|
|
128
|
+
case "$PLATFORM" in
|
|
129
|
+
ios)
|
|
130
|
+
if command -v xcrun >/dev/null 2>&1; then RECORDER="true"; else
|
|
131
|
+
RECORDER="false"; RECORDER_REASON="xcrun unavailable"
|
|
132
|
+
fi
|
|
133
|
+
;;
|
|
134
|
+
android)
|
|
135
|
+
if command -v adb >/dev/null 2>&1; then RECORDER="true"; else
|
|
136
|
+
RECORDER="false"; RECORDER_REASON="adb unavailable"
|
|
137
|
+
fi
|
|
138
|
+
;;
|
|
139
|
+
esac
|
|
140
|
+
if [ "$RECORDER" = "true" ] && ! command -v ffprobe >/dev/null 2>&1; then
|
|
141
|
+
# Not fatal. ffprobe only verifies the result, so its absence downgrades what
|
|
142
|
+
# can be CHECKED about a recording, never whether one can be made.
|
|
143
|
+
RECORDER_REASON="ffprobe unavailable; a recording cannot be verified after capture"
|
|
144
|
+
fi
|
|
145
|
+
|
|
146
|
+
# ---- Toolkit MCP --------------------------------------------------------------
|
|
147
|
+
# Registration only, never a handshake: a probe that spawns the server costs
|
|
148
|
+
# seconds at intake, and the question this answers is whether tier 2 may be
|
|
149
|
+
# offered at all.
|
|
150
|
+
CLAUDE_JSON="$HOME/.claude.json"
|
|
151
|
+
SETTINGS_JSON="$HOME/.claude/settings.json"
|
|
152
|
+
if command -v node >/dev/null 2>&1; then
|
|
153
|
+
MCP=$(node -e '
|
|
154
|
+
const fs = require("fs");
|
|
155
|
+
const hit = (p) => {
|
|
156
|
+
try {
|
|
157
|
+
const j = JSON.parse(fs.readFileSync(p, "utf8"));
|
|
158
|
+
return Object.keys(j.mcpServers || {}).some((k) => /toolkit/i.test(k));
|
|
159
|
+
} catch { return false; }
|
|
160
|
+
};
|
|
161
|
+
process.stdout.write(process.argv.slice(1).some(hit) ? "true" : "false");
|
|
162
|
+
' "$CLAUDE_JSON" "$SETTINGS_JSON" 2>/dev/null)
|
|
163
|
+
[ -n "$MCP" ] || MCP="unknown"
|
|
164
|
+
[ "$MCP" = "false" ] && MCP_REASON="no mcpServers entry matching /toolkit/i"
|
|
165
|
+
[ "$MCP" = "unknown" ] && MCP_REASON="could not read the MCP registration files"
|
|
166
|
+
else
|
|
167
|
+
MCP="unknown"; MCP_REASON="node unavailable; cannot read the MCP registration"
|
|
168
|
+
fi
|
|
169
|
+
|
|
170
|
+
# ---- Tier verdicts ------------------------------------------------------------
|
|
171
|
+
# Computed before the report so both output forms carry them: a caller that
|
|
172
|
+
# persists the JSON and a caller that evals the KEY=VALUE lines must not have to
|
|
173
|
+
# re-derive the same booleans and risk deriving them differently.
|
|
174
|
+
# TIER1 needs a target AND a device AND a recorder; TIER2 drops the target and
|
|
175
|
+
# adds MCP; tier 3 is always reachable because "record nothing and say why" is.
|
|
176
|
+
T1=closed; T2=closed
|
|
177
|
+
[ -n "$UI_TEST_TARGET$UI_TEST_TARGETS" ] && [ -n "$DEVICE" ] && [ "$RECORDER" = "true" ] && T1=open
|
|
178
|
+
[ -n "$DEVICE" ] && [ "$RECORDER" = "true" ] && [ "$MCP" = "true" ] && T2=open
|
|
179
|
+
|
|
180
|
+
# --only device did not look for a target, so tier 1 is UNKNOWN here, not closed.
|
|
181
|
+
# Reporting it closed would be rule 2 broken by the file that states it: a caller
|
|
182
|
+
# re-checking the device before recording would read "tier 1 unavailable" from a
|
|
183
|
+
# measurement that never ran and downgrade a recording it could have made. Tier 2
|
|
184
|
+
# does not depend on the target, so it stays a real verdict.
|
|
185
|
+
[ "$ONLY" = "device" ] && T1=unknown
|
|
186
|
+
|
|
187
|
+
emit_json() {
|
|
188
|
+
UI_TEST_TARGET="$UI_TEST_TARGET" UI_TEST_TARGETS="$UI_TEST_TARGETS" \
|
|
189
|
+
UI_TEST_TARGET_REASON="$UI_TEST_TARGET_REASON" UI_TEST_MATCHES="$UI_TEST_MATCHES" \
|
|
190
|
+
UI_TEST_MATCH_REASON="$UI_TEST_MATCH_REASON" DEVICE="$DEVICE" DEVICE_REASON="$DEVICE_REASON" \
|
|
191
|
+
RECORDER="$RECORDER" RECORDER_REASON="$RECORDER_REASON" MCP="$MCP" MCP_REASON="$MCP_REASON" \
|
|
192
|
+
PLATFORM="$PLATFORM" TIER1="$T1" TIER2="$T2" \
|
|
193
|
+
node -e '
|
|
194
|
+
const e = process.env;
|
|
195
|
+
const list = (s) => (s ? s.split(",").filter(Boolean) : []);
|
|
196
|
+
const tri = (s) => (s === "true" ? true : s === "false" ? false : null);
|
|
197
|
+
process.stdout.write(JSON.stringify({
|
|
198
|
+
platform: e.PLATFORM,
|
|
199
|
+
uiTestTarget: e.UI_TEST_TARGET || null,
|
|
200
|
+
uiTestTargets: list(e.UI_TEST_TARGETS),
|
|
201
|
+
uiTestTargetReason: e.UI_TEST_TARGET_REASON || null,
|
|
202
|
+
matchingTests: list(e.UI_TEST_MATCHES),
|
|
203
|
+
matchingTestsReason: e.UI_TEST_MATCH_REASON || null,
|
|
204
|
+
device: e.DEVICE || null,
|
|
205
|
+
deviceReason: e.DEVICE_REASON || null,
|
|
206
|
+
recorder: tri(e.RECORDER),
|
|
207
|
+
recorderReason: e.RECORDER_REASON || null,
|
|
208
|
+
mcp: tri(e.MCP),
|
|
209
|
+
mcpReason: e.MCP_REASON || null,
|
|
210
|
+
tier1: e.TIER1,
|
|
211
|
+
tier2: e.TIER2,
|
|
212
|
+
}, null, 2) + "\n");
|
|
213
|
+
'
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
if [ -n "$JSON_OUT" ]; then
|
|
217
|
+
mkdir -p "$(dirname "$JSON_OUT")" 2>/dev/null || true
|
|
218
|
+
emit_json > "$JSON_OUT" || {
|
|
219
|
+
echo "probe-evidence-capability: could not write $JSON_OUT" >&2
|
|
220
|
+
exit 2
|
|
221
|
+
}
|
|
222
|
+
fi
|
|
223
|
+
|
|
224
|
+
if [ "$AS_JSON" -eq 1 ]; then
|
|
225
|
+
emit_json
|
|
226
|
+
exit 0
|
|
227
|
+
fi
|
|
228
|
+
|
|
229
|
+
# Every value is single-quoted, because the caller EVALS this. The reasons are
|
|
230
|
+
# prose - "no booted simulator, but one is available to boot" - and an unquoted
|
|
231
|
+
# assignment makes eval run `booted` as a command and assign the first word. The
|
|
232
|
+
# bug is invisible on a machine where the reasons happen to be empty, which is
|
|
233
|
+
# exactly the machine a developer tests on.
|
|
234
|
+
q() { printf "%s='%s'" "$1" "$(printf '%s' "$2" | sed "s/'/'\\\\''/g")"; printf '\n'; }
|
|
235
|
+
|
|
236
|
+
q EVIDENCE_PLATFORM "$PLATFORM"
|
|
237
|
+
q EVIDENCE_UI_TEST_TARGET "$UI_TEST_TARGET"
|
|
238
|
+
q EVIDENCE_UI_TEST_TARGETS "$UI_TEST_TARGETS"
|
|
239
|
+
q EVIDENCE_UI_TEST_TARGET_REASON "$UI_TEST_TARGET_REASON"
|
|
240
|
+
q EVIDENCE_MATCHING_TESTS "$UI_TEST_MATCHES"
|
|
241
|
+
q EVIDENCE_MATCHING_TESTS_REASON "$UI_TEST_MATCH_REASON"
|
|
242
|
+
q EVIDENCE_DEVICE "$DEVICE"
|
|
243
|
+
q EVIDENCE_DEVICE_REASON "$DEVICE_REASON"
|
|
244
|
+
q EVIDENCE_RECORDER "$RECORDER"
|
|
245
|
+
q EVIDENCE_RECORDER_REASON "$RECORDER_REASON"
|
|
246
|
+
q EVIDENCE_MCP "$MCP"
|
|
247
|
+
q EVIDENCE_MCP_REASON "$MCP_REASON"
|
|
248
|
+
|
|
249
|
+
q EVIDENCE_TIER1 "$T1"
|
|
250
|
+
q EVIDENCE_TIER2 "$T2"
|