@mmerterden/multi-agent-pipeline 16.28.0 → 16.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/CHANGELOG.md +119 -2
  2. package/README.md +4 -4
  3. package/README.tr.md +3 -3
  4. package/docs/architecture.md +3 -3
  5. package/docs/ecosystem.md +5 -5
  6. package/docs/features.md +14 -0
  7. package/install/claude.mjs +17 -0
  8. package/package.json +1 -1
  9. package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +93 -0
  10. package/pipeline/commands/multi-agent/design-check/SKILL.md +6 -5
  11. package/pipeline/commands/multi-agent/doctor/SKILL.md +78 -0
  12. package/pipeline/commands/multi-agent/help/SKILL.md +15 -12
  13. package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/setup/SKILL.md +14 -1
  15. package/pipeline/commands/multi-agent/sync/SKILL.md +12 -9
  16. package/pipeline/commands/multi-agent/update/SKILL.md +12 -0
  17. package/pipeline/lib/_jira-auth.sh +99 -0
  18. package/pipeline/lib/analysis-jira-write.sh +203 -0
  19. package/pipeline/lib/issue-fetcher.sh +4 -4
  20. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  21. package/pipeline/multi-agent-refs/channels/pr.md +37 -1
  22. package/pipeline/multi-agent-refs/cross-cli-contract.md +3 -3
  23. package/pipeline/multi-agent-refs/features/analysis-jira.md +128 -0
  24. package/pipeline/multi-agent-refs/features/doctor.md +197 -0
  25. package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
  26. package/pipeline/multi-agent-refs/features/visual-evidence.md +103 -20
  27. package/pipeline/multi-agent-refs/phases/phase-0-init.md +38 -7
  28. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +13 -1
  29. package/pipeline/multi-agent-refs/phases/phase-5-test.md +11 -1
  30. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +23 -0
  31. package/pipeline/multi-agent-refs/picker-contract.md +35 -0
  32. package/pipeline/multi-agent-refs/tracker-contract.md +5 -1
  33. package/pipeline/preferences-template.json +1 -1
  34. package/pipeline/schemas/agent-state.schema.json +84 -1
  35. package/pipeline/schemas/analysis-spec.schema.json +336 -95
  36. package/pipeline/schemas/prefs.schema.json +80 -3
  37. package/pipeline/schemas/token-budget.json +10 -10
  38. package/pipeline/scripts/analysis-story-tree.mjs +441 -0
  39. package/pipeline/scripts/capture-evidence.sh +170 -5
  40. package/pipeline/scripts/doctor.mjs +758 -0
  41. package/pipeline/scripts/evidence-gate.mjs +31 -2
  42. package/pipeline/scripts/phase-tracker.sh +97 -17
  43. package/pipeline/scripts/probe-evidence-capability.sh +250 -0
  44. package/pipeline/scripts/run-ui-tests.sh +380 -0
  45. package/pipeline/scripts/scan-agent-config.sh +48 -10
  46. package/pipeline/scripts/skill-siblings.mjs +1 -1
  47. package/pipeline/skills/shared/core/multi-agent-analysis-jira/SKILL.md +94 -0
  48. package/pipeline/skills/shared/core/multi-agent-doctor/SKILL.md +79 -0
  49. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +10 -1
  50. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +13 -0
  51. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -6
  52. package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +18 -0
@@ -19,6 +19,9 @@
19
19
  // manual is JSON-based (Phase 5 manual-test.json, see below)
20
20
  // --status the claimed status (only "passed" is gated; others pass through)
21
21
  // --evidence <path> the log / artifact that must substantiate a "passed" claim
22
+ // --require-screenshot manual claim only: every "pass" criterion must name a
23
+ // screenshot that exists on disk. Set by Phase 5 when
24
+ // state.visualEvidence.required is true.
22
25
  // --success-pattern override the default success regex for the claim type (marker claims only)
23
26
  // --failure-pattern override the default failure regex for the claim type (marker claims only)
24
27
  //
@@ -120,7 +123,7 @@ function nonEmptyString(v) {
120
123
  // Manual-test evidence is a structured document, not a log: every acceptance
121
124
  // criterion must say what was observed and how it came out. A "fail" anywhere
122
125
  // is decisive; an untested criterion is only acceptable when it says why.
123
- function gateManual(claim, evidence, body) {
126
+ function gateManual(claim, evidence, body, requireScreenshot) {
124
127
  let doc;
125
128
  try {
126
129
  doc = JSON.parse(body);
@@ -154,6 +157,32 @@ function gateManual(claim, evidence, body) {
154
157
  `${claim} evidence shows a failed criterion "${failed.spec}" - pass claim contradicted by ${evidence}`,
155
158
  );
156
159
  }
160
+ // A screenshot path is only evidence when a file is behind it. The field has
161
+ // been in the document shape since the gate shipped and was never read, so a
162
+ // criterion carrying "screenshot": null passed as a verified manual test on a
163
+ // UI change - which is the one case the picture was added for. Checked only
164
+ // when the caller says visual evidence is required for this run, because on a
165
+ // backend change there is nothing to photograph.
166
+ if (requireScreenshot) {
167
+ const shotless = criteria.find(
168
+ (item) => item.verdict === "pass" && !nonEmptyString(item.screenshot),
169
+ );
170
+ if (shotless) {
171
+ fail(
172
+ `${claim} evidence has passing criterion "${shotless.spec}" with no screenshot, and this run requires visual evidence: ${evidence} (default-FAIL)`,
173
+ );
174
+ }
175
+ const missing = criteria.find(
176
+ (item) =>
177
+ item.verdict === "pass" && nonEmptyString(item.screenshot) && !existsSync(item.screenshot),
178
+ );
179
+ if (missing) {
180
+ fail(
181
+ `${claim} evidence names a screenshot that is not on disk: ${missing.screenshot} (default-FAIL)`,
182
+ );
183
+ }
184
+ }
185
+
157
186
  const untested = criteria.find(
158
187
  (item) => item.verdict === "not-tested" && !nonEmptyString(item.reason),
159
188
  );
@@ -203,7 +232,7 @@ if (body.trim().length === 0) {
203
232
  }
204
233
 
205
234
  if (JSON_CLAIMS.has(claim)) {
206
- gateManual(claim, evidence, body);
235
+ gateManual(claim, evidence, body, args["require-screenshot"] === true);
207
236
  }
208
237
 
209
238
  const successRe = userRegExp(args["success-pattern"], DEFAULT_MARKERS[claim].success, "success");
@@ -570,10 +570,11 @@ tracker_next_hint() {
570
570
  [ "${TRACKER_QUIET:-0}" = "1" ] && return 0
571
571
  name=$(load_state | jq -r --arg id "$pid" '(.phases[] | select(.id == $id) | .name) // ""')
572
572
  case "$(host_kind)" in
573
- # Not every Claude Code build ships the tile API, and the shell cannot see
574
- # the model's tool list. Naming the fallback on the same line is what keeps a
575
- # build without TaskUpdate from advancing eight phases in silence.
576
- claude) mirror="TaskUpdate(\"Phase $pid $name\", status=\"$status\") - or, if TaskUpdate is not one of your tools, reprint the card above inside your reply text" ;;
573
+ claude) mirror="TaskUpdate(\"$(subjects "$pid")\", status=\"$status\") - the subject is re-read from \`phase-tracker.sh subjects $pid\` so the widget carries the model, the elapsed time and the tokens; if TaskUpdate is not one of your tools, paste the card above into your reply VERBATIM in a code block (every phase on its own line, keep the elapsed and token columns; do not redraw or compact it)" ;;
574
+ # Not every session carries the task tools: Claude Code provides them by
575
+ # default only up to Opus 4.7 / Sonnet 4.6, a default that landed in
576
+ # v2.1.268. Naming the fallback on the same line is what keeps a newer model
577
+ # from advancing eight phases in silence.
577
578
  codex) mirror="update_plan: set step \"Phase $pid $name\" to $status (send the FULL step list, it is not a delta)" ;;
578
579
  *) mirror="no task widget on this host - reprint the card above inside your reply text" ;;
579
580
  esac
@@ -664,6 +665,61 @@ EOF
664
665
  printf '\n'
665
666
  }
666
667
 
668
+ # The native widget renders one row per task and gives us exactly one string to
669
+ # fill it: the subject. So the numbers a reader wants - which model is spending,
670
+ # how long the phase has run, what it cost - have to travel IN the subject or not
671
+ # at all. `render` already computes all of it for the fallback card; this prints
672
+ # the same values in the one shape the host will accept.
673
+ #
674
+ # Every segment is omitted while it is empty, so a pending phase is just its name
675
+ # and a finished one carries its whole bill. Separator is ` - `, not a middle dot:
676
+ # a subject is a task title and travels through surfaces that are not a terminal.
677
+ subjects() {
678
+ need_jq
679
+ local state want="${1:-}"
680
+ state=$(load_state)
681
+ local prices_json='{"prices":{}}'
682
+ if [ -f "$COST_TABLE" ]; then
683
+ prices_json=$(cat "$COST_TABLE" 2>/dev/null) || prices_json='{"prices":{}}'
684
+ echo "$prices_json" | jq empty 2>/dev/null || prices_json='{"prices":{}}'
685
+ fi
686
+ local now_s; now_s=$(now_epoch)
687
+ local rows
688
+ rows=$(echo "$state" | jq -r --argjson prices "$prices_json" "$COST_JQ_DEFS"'
689
+ def usd_of(p): cost_usd_of($prices.prices[p.model // ""] // null; (p.tokens_in // 0); (p.tokens_out // 0); (p.tokens_cached // 0));
690
+ .phases // []
691
+ | sort_by(.id | (tonumber? // 9999))
692
+ | .[] | [
693
+ (.id // ""), (.name // ""), (.status // ""),
694
+ (.started_at // ""), (.completed_at // ""), (.model // ""),
695
+ (((.tokens_in // 0) + (.tokens_out // 0)) | tostring),
696
+ (usd_of(.) | if . == null then "" else (((. * 100) | floor) / 100 | tostring) end)
697
+ ] | join("\u001f")')
698
+ local pid pname pstatus p_start p_end pmodel ptok pusd
699
+ while IFS=$'\x1f' read -r pid pname pstatus p_start p_end pmodel ptok pusd; do
700
+ [ -n "$pid" ] || continue
701
+ [ -z "$want" ] || [ "$want" = "$pid" ] || continue
702
+ local line="Phase $pid $pname"
703
+ [ -n "$pmodel" ] && line="$line - $pmodel"
704
+ local s_ep e_ep el=""
705
+ s_ep=$(iso_to_epoch "$p_start")
706
+ e_ep=$(iso_to_epoch "$p_end")
707
+ if [ -n "$s_ep" ]; then
708
+ case "$pstatus" in
709
+ completed|failed|skipped) [ -n "$e_ep" ] && el=$((e_ep - s_ep)) ;;
710
+ in_progress) el=$((now_s - s_ep)) ;;
711
+ esac
712
+ fi
713
+ if [ -n "$el" ] && [ "$el" -lt 0 ] 2>/dev/null; then el=0; fi
714
+ [ -n "$el" ] && line="$line - $(format_elapsed "$el")"
715
+ if [ "${ptok:-0}" -gt 0 ] 2>/dev/null; then
716
+ line="$line - $(format_tokens "$ptok") tok"
717
+ [ -n "$pusd" ] && line="$line - $(printf '~$%.2f' "$pusd" 2>/dev/null || echo '')"
718
+ fi
719
+ printf '%s\n' "$line"
720
+ done <<< "$rows"
721
+ }
722
+
667
723
  render() {
668
724
  need_jq
669
725
  local state
@@ -1159,6 +1215,10 @@ GATE
1159
1215
  fi
1160
1216
  ;;
1161
1217
 
1218
+ subjects)
1219
+ subjects "${1:-}"
1220
+ ;;
1221
+
1162
1222
  tiles)
1163
1223
  need_jq
1164
1224
  tiles_state=$(load_state)
@@ -1169,22 +1229,42 @@ GATE
1169
1229
  }
1170
1230
  case "$(host_kind)" in
1171
1231
  claude)
1172
- # The native tile API is not in every Claude Code build. This script
1173
- # cannot probe the model's tool list, and a run that fires nothing and
1174
- # says nothing is the failure that reached a user: the tracker state was
1175
- # written correctly, every phase advanced, and the screen stayed empty
1176
- # for the whole run. So the fallback is named here rather than assumed,
1177
- # and the branch is taken where the information actually lives - the
1178
- # model knows which tools it has, the shell does not.
1232
+ # The task tools are not in every session, and the reason is the model,
1233
+ # not the CLI version. Claude Code provides TaskCreate / TaskUpdate by
1234
+ # default only on Claude 3.x, Opus 4 through 4.7, Sonnet 4 through 4.6
1235
+ # and Haiku 4.5; on any newer model it leaves them out unless the user
1236
+ # opts in, and that default landed in v2.1.268. This contract was written
1237
+ # when the tools were universal and stayed true for years, so nothing
1238
+ # noticed: the tracker state was written correctly, every phase advanced,
1239
+ # and the screen stayed empty for a whole run.
1240
+ #
1241
+ # A shell script cannot probe the model's tool list, so the branch is
1242
+ # taken where the information actually lives - the model knows which
1243
+ # tools it has. And the fallback is not enough on its own: a user looking
1244
+ # at a missing widget needs the one command that brings it back, which is
1245
+ # why the opt-in is printed next to it.
1179
1246
  echo "REQUIRED - create one native tile per phase, in this exact order,"
1180
1247
  echo "BEFORE any TaskUpdate. The widget renders by creation order, not by"
1181
1248
  echo "phase number, so an out-of-order call scrambles the stack."
1182
- echo "$tiles_state" | jq -r '.phases[] | " TaskCreate(subject: \"Phase \(.id) \(.name)\")"'
1249
+ subjects | sed 's/^/ TaskCreate(subject: "/; s/$/")/'
1250
+ echo
1251
+ echo "At every phase boundary re-run \`phase-tracker.sh subjects <id>\` and"
1252
+ echo "pass that line as the subject of the TaskUpdate. The subject is the only"
1253
+ echo "string the native widget renders, so the model, the elapsed time and the"
1254
+ echo "token count reach the user through it or not at all."
1183
1255
  echo
1184
- echo "IF TaskCreate IS NOT ONE OF YOUR TOOLS this build has no native tile"
1185
- echo "API. Do not skip the signal: the card below IS the widget then, and"
1186
- echo "it must be reprinted inside your reply at every phase boundary - "
1187
- echo "tool output is collapsed, so a card left in stdout never arrives."
1256
+ echo "IF TaskCreate IS NOT ONE OF YOUR TOOLS this session's model does not"
1257
+ echo "carry them (default since v2.1.268 on models newer than Opus 4.7 /"
1258
+ echo "Sonnet 4.6). Then the card below IS the widget, and at every phase"
1259
+ echo "boundary you run \`phase-tracker.sh render\` and paste its output"
1260
+ echo "into your reply VERBATIM, inside a code block - every phase on"
1261
+ echo "its own line, with the elapsed time, the token count and the total"
1262
+ echo "row exactly as printed. Do NOT redraw it, do not compact phases"
1263
+ echo "onto one line, do not drop the columns: those numbers are the"
1264
+ echo "whole reason the card is worth showing. Tool output is collapsed,"
1265
+ echo "so a card left in stdout never reaches the user."
1266
+ echo "Say once, in outputLanguage, that the native widget returns with:"
1267
+ echo " CLAUDE_CODE_ENABLE_TODO_TOOLS=1 claude"
1188
1268
  echo
1189
1269
  render
1190
1270
  ;;
@@ -1212,7 +1292,7 @@ GATE
1212
1292
  ;;
1213
1293
 
1214
1294
  *)
1215
- echo "phase-tracker: unknown action '$ACTION' (use init|add|update|sub|tokens|model|meta|now|cost|render)" >&2
1295
+ echo "phase-tracker: unknown action '$ACTION' (use init|add|update|sub|tokens|model|meta|now|cost|render|subjects)" >&2
1216
1296
  exit 64
1217
1297
  ;;
1218
1298
  esac
@@ -0,0 +1,250 @@
1
+ #!/usr/bin/env bash
2
+ #
3
+ # probe-evidence-capability.sh - measure what visual evidence this machine and
4
+ # this repo can actually produce, BEFORE the user is asked to choose a test depth.
5
+ #
6
+ # Contract: multi-agent-refs/features/visual-evidence.md section 4.
7
+ #
8
+ # The order is probe, then question, then run. Offering "unit + UI test with a
9
+ # screen recording" and only then discovering there is no UI test target, or no
10
+ # booted simulator, spends the user's answer on something that cannot happen. So
11
+ # the options are built from what this prints.
12
+ #
13
+ # Three rules this file exists to keep:
14
+ #
15
+ # 1. Absence carries a reason. Every empty value is paired with a *_REASON, so
16
+ # a closed option can say why it is closed instead of vanishing from the menu.
17
+ # 2. Unmeasurable is `unknown`, never `false`. A probe that could not look and a
18
+ # probe that looked and found nothing are different facts, and collapsing
19
+ # them lets a missing tool read as a clean negative.
20
+ # 3. Detection is not reimplemented here. The UI test answer comes from
21
+ # run-ui-tests.sh detect, which is also what actually runs the tests; a
22
+ # second copy is a second place for the answer to drift.
23
+ #
24
+ # Usage:
25
+ # probe-evidence-capability.sh --platform <ios|android> [--repo <path>]
26
+ # [--changed <f>[,<f>...]] [--json]
27
+ # [--json-out <path>] [--only all|device]
28
+ #
29
+ # Output: KEY='VALUE' lines (eval-able; every value is shell-quoted), or a JSON object with --json (shaped for
30
+ # state.evidenceCapability). --json-out writes that JSON to a file while stdout
31
+ # stays KEY=VALUE, so one run serves both the shell that builds the menu and the
32
+ # state that records the measurement - two runs would mean two repo scans and
33
+ # two chances to disagree.
34
+ #
35
+ # --only device skips the UI-test detection. Phase 3 re-checks the device right
36
+ # before recording, and nothing else it would re-scan can have changed.
37
+ #
38
+ # Exit: 0 probed (whatever the verdicts), 2 usage.
39
+ # Never non-zero for an absent capability: absence is the finding.
40
+ set -uo pipefail
41
+
42
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
43
+ PLATFORM=""; REPO="$PWD"; CHANGED=""; AS_JSON=0; JSON_OUT=""; ONLY="all"
44
+
45
+ while [ "$#" -gt 0 ]; do
46
+ case "$1" in
47
+ --platform) PLATFORM="${2:-}"; shift 2 ;;
48
+ --repo) REPO="${2:-}"; shift 2 ;;
49
+ --changed) CHANGED="${CHANGED:+$CHANGED,}${2:-}"; shift 2 ;;
50
+ --json) AS_JSON=1; shift ;;
51
+ --json-out) JSON_OUT="${2:-}"; shift 2 ;;
52
+ --only) ONLY="${2:-all}"; shift 2 ;;
53
+ *) echo "probe-evidence-capability: unknown option $1" >&2; exit 2 ;;
54
+ esac
55
+ done
56
+ case "$ONLY" in all | device) ;; *)
57
+ echo "probe-evidence-capability: --only takes 'all' or 'device'" >&2; exit 2 ;;
58
+ esac
59
+
60
+ case "$PLATFORM" in ios | android) ;; *)
61
+ echo "usage: probe-evidence-capability.sh --platform <ios|android> [--repo <path>] [--changed <f>] [--json]" >&2
62
+ exit 2 ;;
63
+ esac
64
+ [ -d "$REPO" ] || { echo "probe-evidence-capability: no such repo directory: $REPO" >&2; exit 2; }
65
+
66
+ UI_TEST_TARGET=""; UI_TEST_TARGETS=""; UI_TEST_TARGET_REASON=""
67
+ UI_TEST_MATCHES=""; UI_TEST_MATCH_REASON=""
68
+ DEVICE=""; DEVICE_REASON=""
69
+ RECORDER="unknown"; RECORDER_REASON=""
70
+ MCP="unknown"; MCP_REASON=""
71
+
72
+ # ---- UI test target, delegated ------------------------------------------------
73
+ RUNNER="$HERE/run-ui-tests.sh"
74
+ if [ "$ONLY" = "device" ]; then
75
+ # Phase 3 re-checks the device right before recording, and only the device.
76
+ # Delegating detection there would re-walk the whole repo for a row that
77
+ # cannot have changed, which cost 5 seconds of a phase that is already holding
78
+ # a green build.
79
+ UI_TEST_TARGET_REASON="not probed (--only device)"
80
+ elif [ ! -x "$RUNNER" ] && [ ! -f "$RUNNER" ]; then
81
+ UI_TEST_TARGET_REASON="run-ui-tests.sh not found beside this script"
82
+ else
83
+ DETECT_ARGS=(detect --platform "$PLATFORM" --repo "$REPO")
84
+ [ -n "$CHANGED" ] && DETECT_ARGS+=(--changed "$CHANGED")
85
+ while IFS='=' read -r k v; do
86
+ case "$k" in
87
+ UI_TEST_TARGET) UI_TEST_TARGET="$v" ;;
88
+ UI_TEST_TARGETS) UI_TEST_TARGETS="$v" ;;
89
+ UI_TEST_TARGET_REASON) UI_TEST_TARGET_REASON="$v" ;;
90
+ UI_TEST_MATCHES) UI_TEST_MATCHES="$v" ;;
91
+ UI_TEST_MATCH_REASON) UI_TEST_MATCH_REASON="$v" ;;
92
+ esac
93
+ done < <(bash "$RUNNER" "${DETECT_ARGS[@]}" 2>/dev/null)
94
+ fi
95
+
96
+ # ---- Device -------------------------------------------------------------------
97
+ case "$PLATFORM" in
98
+ ios)
99
+ if ! command -v xcrun >/dev/null 2>&1; then
100
+ DEVICE_REASON="xcrun unavailable; cannot tell whether a simulator exists"
101
+ else
102
+ DEVICE=$(xcrun simctl list devices booted 2>/dev/null | grep -oE '[0-9A-F-]{36}' | head -1)
103
+ if [ -z "$DEVICE" ]; then
104
+ # Bootable is not booted, and the difference is a question the user can
105
+ # act on: one is "start your simulator", the other is "this machine has
106
+ # no iOS runtime installed".
107
+ if xcrun simctl list devices available 2>/dev/null | grep -q "([0-9A-F-]\{36\})"; then
108
+ DEVICE_REASON="no booted simulator, but one is available to boot"
109
+ else
110
+ DEVICE_REASON="no iOS simulator available on this machine"
111
+ fi
112
+ fi
113
+ fi
114
+ ;;
115
+ android)
116
+ if ! command -v adb >/dev/null 2>&1; then
117
+ DEVICE_REASON="adb unavailable; cannot tell whether a device is attached"
118
+ else
119
+ DEVICE=$(adb devices 2>/dev/null | awk 'NR>1 && $2=="device" {print $1; exit}')
120
+ [ -n "$DEVICE" ] || DEVICE_REASON="no attached device or running emulator"
121
+ fi
122
+ ;;
123
+ esac
124
+
125
+ # ---- Recorder -----------------------------------------------------------------
126
+ # The recorder is the same CLI the capture needs, so this answers "would
127
+ # capture-evidence.sh video start work" rather than "is some recorder installed".
128
+ case "$PLATFORM" in
129
+ ios)
130
+ if command -v xcrun >/dev/null 2>&1; then RECORDER="true"; else
131
+ RECORDER="false"; RECORDER_REASON="xcrun unavailable"
132
+ fi
133
+ ;;
134
+ android)
135
+ if command -v adb >/dev/null 2>&1; then RECORDER="true"; else
136
+ RECORDER="false"; RECORDER_REASON="adb unavailable"
137
+ fi
138
+ ;;
139
+ esac
140
+ if [ "$RECORDER" = "true" ] && ! command -v ffprobe >/dev/null 2>&1; then
141
+ # Not fatal. ffprobe only verifies the result, so its absence downgrades what
142
+ # can be CHECKED about a recording, never whether one can be made.
143
+ RECORDER_REASON="ffprobe unavailable; a recording cannot be verified after capture"
144
+ fi
145
+
146
+ # ---- Toolkit MCP --------------------------------------------------------------
147
+ # Registration only, never a handshake: a probe that spawns the server costs
148
+ # seconds at intake, and the question this answers is whether tier 2 may be
149
+ # offered at all.
150
+ CLAUDE_JSON="$HOME/.claude.json"
151
+ SETTINGS_JSON="$HOME/.claude/settings.json"
152
+ if command -v node >/dev/null 2>&1; then
153
+ MCP=$(node -e '
154
+ const fs = require("fs");
155
+ const hit = (p) => {
156
+ try {
157
+ const j = JSON.parse(fs.readFileSync(p, "utf8"));
158
+ return Object.keys(j.mcpServers || {}).some((k) => /toolkit/i.test(k));
159
+ } catch { return false; }
160
+ };
161
+ process.stdout.write(process.argv.slice(1).some(hit) ? "true" : "false");
162
+ ' "$CLAUDE_JSON" "$SETTINGS_JSON" 2>/dev/null)
163
+ [ -n "$MCP" ] || MCP="unknown"
164
+ [ "$MCP" = "false" ] && MCP_REASON="no mcpServers entry matching /toolkit/i"
165
+ [ "$MCP" = "unknown" ] && MCP_REASON="could not read the MCP registration files"
166
+ else
167
+ MCP="unknown"; MCP_REASON="node unavailable; cannot read the MCP registration"
168
+ fi
169
+
170
+ # ---- Tier verdicts ------------------------------------------------------------
171
+ # Computed before the report so both output forms carry them: a caller that
172
+ # persists the JSON and a caller that evals the KEY=VALUE lines must not have to
173
+ # re-derive the same booleans and risk deriving them differently.
174
+ # TIER1 needs a target AND a device AND a recorder; TIER2 drops the target and
175
+ # adds MCP; tier 3 is always reachable because "record nothing and say why" is.
176
+ T1=closed; T2=closed
177
+ [ -n "$UI_TEST_TARGET$UI_TEST_TARGETS" ] && [ -n "$DEVICE" ] && [ "$RECORDER" = "true" ] && T1=open
178
+ [ -n "$DEVICE" ] && [ "$RECORDER" = "true" ] && [ "$MCP" = "true" ] && T2=open
179
+
180
+ # --only device did not look for a target, so tier 1 is UNKNOWN here, not closed.
181
+ # Reporting it closed would be rule 2 broken by the file that states it: a caller
182
+ # re-checking the device before recording would read "tier 1 unavailable" from a
183
+ # measurement that never ran and downgrade a recording it could have made. Tier 2
184
+ # does not depend on the target, so it stays a real verdict.
185
+ [ "$ONLY" = "device" ] && T1=unknown
186
+
187
+ emit_json() {
188
+ UI_TEST_TARGET="$UI_TEST_TARGET" UI_TEST_TARGETS="$UI_TEST_TARGETS" \
189
+ UI_TEST_TARGET_REASON="$UI_TEST_TARGET_REASON" UI_TEST_MATCHES="$UI_TEST_MATCHES" \
190
+ UI_TEST_MATCH_REASON="$UI_TEST_MATCH_REASON" DEVICE="$DEVICE" DEVICE_REASON="$DEVICE_REASON" \
191
+ RECORDER="$RECORDER" RECORDER_REASON="$RECORDER_REASON" MCP="$MCP" MCP_REASON="$MCP_REASON" \
192
+ PLATFORM="$PLATFORM" TIER1="$T1" TIER2="$T2" \
193
+ node -e '
194
+ const e = process.env;
195
+ const list = (s) => (s ? s.split(",").filter(Boolean) : []);
196
+ const tri = (s) => (s === "true" ? true : s === "false" ? false : null);
197
+ process.stdout.write(JSON.stringify({
198
+ platform: e.PLATFORM,
199
+ uiTestTarget: e.UI_TEST_TARGET || null,
200
+ uiTestTargets: list(e.UI_TEST_TARGETS),
201
+ uiTestTargetReason: e.UI_TEST_TARGET_REASON || null,
202
+ matchingTests: list(e.UI_TEST_MATCHES),
203
+ matchingTestsReason: e.UI_TEST_MATCH_REASON || null,
204
+ device: e.DEVICE || null,
205
+ deviceReason: e.DEVICE_REASON || null,
206
+ recorder: tri(e.RECORDER),
207
+ recorderReason: e.RECORDER_REASON || null,
208
+ mcp: tri(e.MCP),
209
+ mcpReason: e.MCP_REASON || null,
210
+ tier1: e.TIER1,
211
+ tier2: e.TIER2,
212
+ }, null, 2) + "\n");
213
+ '
214
+ }
215
+
216
+ if [ -n "$JSON_OUT" ]; then
217
+ mkdir -p "$(dirname "$JSON_OUT")" 2>/dev/null || true
218
+ emit_json > "$JSON_OUT" || {
219
+ echo "probe-evidence-capability: could not write $JSON_OUT" >&2
220
+ exit 2
221
+ }
222
+ fi
223
+
224
+ if [ "$AS_JSON" -eq 1 ]; then
225
+ emit_json
226
+ exit 0
227
+ fi
228
+
229
+ # Every value is single-quoted, because the caller EVALS this. The reasons are
230
+ # prose - "no booted simulator, but one is available to boot" - and an unquoted
231
+ # assignment makes eval run `booted` as a command and assign the first word. The
232
+ # bug is invisible on a machine where the reasons happen to be empty, which is
233
+ # exactly the machine a developer tests on.
234
+ q() { printf "%s='%s'" "$1" "$(printf '%s' "$2" | sed "s/'/'\\\\''/g")"; printf '\n'; }
235
+
236
+ q EVIDENCE_PLATFORM "$PLATFORM"
237
+ q EVIDENCE_UI_TEST_TARGET "$UI_TEST_TARGET"
238
+ q EVIDENCE_UI_TEST_TARGETS "$UI_TEST_TARGETS"
239
+ q EVIDENCE_UI_TEST_TARGET_REASON "$UI_TEST_TARGET_REASON"
240
+ q EVIDENCE_MATCHING_TESTS "$UI_TEST_MATCHES"
241
+ q EVIDENCE_MATCHING_TESTS_REASON "$UI_TEST_MATCH_REASON"
242
+ q EVIDENCE_DEVICE "$DEVICE"
243
+ q EVIDENCE_DEVICE_REASON "$DEVICE_REASON"
244
+ q EVIDENCE_RECORDER "$RECORDER"
245
+ q EVIDENCE_RECORDER_REASON "$RECORDER_REASON"
246
+ q EVIDENCE_MCP "$MCP"
247
+ q EVIDENCE_MCP_REASON "$MCP_REASON"
248
+
249
+ q EVIDENCE_TIER1 "$T1"
250
+ q EVIDENCE_TIER2 "$T2"