@mmerterden/multi-agent-pipeline 17.0.0 → 17.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README.md +49 -4
  3. package/README.tr.md +50 -4
  4. package/docs/architecture.md +3 -3
  5. package/docs/ecosystem.md +5 -5
  6. package/install/templates/multi-agent-autopilot.plist.template +79 -0
  7. package/package.json +1 -1
  8. package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +64 -0
  9. package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +173 -0
  10. package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +74 -0
  11. package/pipeline/commands/multi-agent/channels/SKILL.md +41 -12
  12. package/pipeline/commands/multi-agent/help/SKILL.md +41 -35
  13. package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/sync/SKILL.md +10 -9
  15. package/pipeline/commands/multi-agent/update/SKILL.md +1 -1
  16. package/pipeline/lib/autopilot-activation.sh +117 -0
  17. package/pipeline/lib/autopilot-state.sh +150 -0
  18. package/pipeline/lib/issue-fetcher.sh +18 -1
  19. package/pipeline/lib/plan-todos.sh +18 -0
  20. package/pipeline/multi-agent-refs/channels/jira.md +80 -20
  21. package/pipeline/multi-agent-refs/channels/pr.md +65 -19
  22. package/pipeline/multi-agent-refs/cross-cli-contract.md +6 -5
  23. package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -7
  24. package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -1
  25. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +17 -15
  26. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -1
  27. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +1 -1
  28. package/pipeline/multi-agent-refs/readiness-review.md +7 -1
  29. package/pipeline/multi-agent-refs/rules.md +3 -11
  30. package/pipeline/multi-agent-refs/tracker-contract.md +32 -0
  31. package/pipeline/schemas/autopilot-config.schema.json +149 -0
  32. package/pipeline/schemas/token-budget.json +2 -2
  33. package/pipeline/scripts/autopilot-arming.mjs +147 -0
  34. package/pipeline/scripts/autopilot-intake.mjs +383 -0
  35. package/pipeline/scripts/autopilot-menubar.swift +361 -0
  36. package/pipeline/scripts/autopilot-runner.mjs +349 -0
  37. package/pipeline/scripts/autopilot-status.sh +212 -0
  38. package/pipeline/scripts/jira-search.sh +70 -0
  39. package/pipeline/scripts/phase-tracker.sh +134 -12
  40. package/pipeline/scripts/probe-evidence-capability.sh +27 -3
  41. package/pipeline/scripts/run-ui-tests.sh +113 -4
  42. package/pipeline/skills/.skill-manifest.json +16 -4
  43. package/pipeline/skills/shared/core/multi-agent-autopilot-off/SKILL.md +67 -0
  44. package/pipeline/skills/shared/core/multi-agent-autopilot-on/SKILL.md +146 -0
  45. package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +64 -0
  46. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +62 -11
  47. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -8
@@ -575,7 +575,12 @@ tracker_next_hint() {
575
575
  # default only up to Opus 4.7 / Sonnet 4.6, a default that landed in
576
576
  # v2.1.268. Naming the fallback on the same line is what keeps a newer model
577
577
  # from advancing eight phases in silence.
578
- codex) mirror="update_plan: set step \"Phase $pid $name\" to $status (send the FULL step list, it is not a delta)" ;;
578
+ # `subjects` now carries Phase 2's plan steps as indented rows, and
579
+ # update_plan takes the whole list anyway, so re-reading it is what puts
580
+ # those steps on the Codex plan without a second mechanism. Codex has no
581
+ # dependency concept; the "(bekliyor: ...)" suffix inside the step text is
582
+ # where that information survives on this host.
583
+ codex) mirror="update_plan: set step \"Phase $pid $name\" to $status, re-reading the FULL list from \`phase-tracker.sh subjects\` (it is not a delta, and the indented rows are plan steps that belong in it)" ;;
579
584
  *) mirror="no task widget on this host - reprint the card above inside your reply text" ;;
580
585
  esac
581
586
  printf '\n-- NEXT (required) --\n %s\n' "$mirror"
@@ -717,9 +722,89 @@ subjects() {
717
722
  [ -n "$pusd" ] && line="$line - $(printf '~$%.2f' "$pusd" 2>/dev/null || echo '')"
718
723
  fi
719
724
  printf '%s\n' "$line"
725
+
726
+ # Sub-phases ride out on the SAME list, indented in the subject string.
727
+ #
728
+ # The card has drawn these since sub-phases existed; the widget never has,
729
+ # and the widget is the surface the user actually looks at. Phase 2's plan
730
+ # is the case that made the gap matter: the tasks, their order and their
731
+ # dependencies are computed, stored and used to drive Phase 3's picker, and
732
+ # none of it was visible anywhere the user was looking.
733
+ #
734
+ # Indentation is two spaces INSIDE the subject because the widget takes
735
+ # exactly one string per row and has no parent field - there is no nesting
736
+ # to ask for, so the only place a hierarchy can be expressed is the string.
737
+ if [ -z "$want" ] || [ "$want" = "$pid" ]; then
738
+ echo "$state" | jq -r --arg pid "$pid" '
739
+ ([.phases[] | select(.id == $pid)][0].subs) // []
740
+ | .[]
741
+ | " " + (.id // "") + " " + (.name // "")
742
+ + (if ((.deps // []) | length) > 0
743
+ then " (bekliyor: " + ((.deps // []) | join(", ")) + ")"
744
+ else "" end)'
745
+ fi
720
746
  done <<< "$rows"
721
747
  }
722
748
 
749
+ # The TaskCreate/TaskUpdate script for this host, in creation order.
750
+ #
751
+ # Creation order is the ONLY ordering the native widget has - it renders by the
752
+ # order tiles were made, not by any id inside them - so a plan that arrives at
753
+ # Phase 2 cannot simply be appended: its rows would land after Phase 7. The
754
+ # answer is a rebuild at that one boundary, which is the same thing `:resume`
755
+ # already does for a different reason.
756
+ tiles_script() {
757
+ need_jq
758
+ local state; state=$(load_state)
759
+ local has_subs
760
+ has_subs=$(echo "$state" | jq '[.phases[]?.subs[]?] | length')
761
+
762
+ if [ "${has_subs:-0}" -gt 0 ]; then
763
+ printf 'The tile list has sub-steps now, and the widget orders by CREATION.
764
+ '
765
+ printf 'Delete every existing tile (TaskUpdate status="deleted"), then create
766
+ '
767
+ printf 'these in this exact order. Rebuilding is cheaper than a list whose
768
+ '
769
+ printf 'steps sit after the phase they belong to.
770
+
771
+ '
772
+ else
773
+ printf 'REQUIRED - create one native tile per phase, in this exact order,
774
+ '
775
+ printf 'BEFORE any TaskUpdate. The widget renders by creation order, not by
776
+ '
777
+ printf 'phase number, so an out-of-order call scrambles the stack.
778
+
779
+ '
780
+ fi
781
+
782
+ subjects | sed 's/^/ TaskCreate(subject: "/; s/$/")/'
783
+
784
+ # Dependencies are a second pass on purpose: a tile cannot be blocked by one
785
+ # that does not exist yet, and TaskCreate has no dependency field.
786
+ #
787
+ # The lines below name tiles by their SUBJECT, not by the plan's own `t1`/`t2`
788
+ # ids. `addBlockedBy` takes the ids TaskCreate handed back, which no shell
789
+ # script can know - printing `addBlockedBy: ["t1"]` would be a line that looks
790
+ # executable and is not, which is worse than a line that admits what it needs.
791
+ local deps
792
+ deps=$(echo "$state" | jq -r '
793
+ [.phases[]? | .subs[]?] as $all
794
+ | $all[] | select(((.deps // []) | length) > 0)
795
+ | . as $s
796
+ | " TaskUpdate(<tile \" " + ($s.name // "") + "\">, addBlockedBy: [" +
797
+ (($s.deps // [])
798
+ | map(. as $d | (($all[] | select(.id == $d) | "<tile \" " + (.name // "") + "\">") // $d))
799
+ | join(", ")) + "])"')
800
+ if [ -n "$deps" ]; then
801
+ printf '\nThen the dependencies, so the widget says WHY a step has not started.\n'
802
+ printf 'addBlockedBy takes the ids TaskCreate returned; <tile "..."> below is\n'
803
+ printf 'the tile whose subject is that string.\n'
804
+ printf '%s\n' "$deps"
805
+ fi
806
+ }
807
+
723
808
  render() {
724
809
  need_jq
725
810
  local state
@@ -888,11 +973,14 @@ render() {
888
973
  if [ "$sub_count" != "0" ] && [ "$sub_count" -gt 0 ] 2>/dev/null; then
889
974
  local sub_data
890
975
  sub_data=$(echo "$state" | jq -r --arg pid "$pid" '
891
- ([.phases[] | select(.id == $pid)][0].subs) // [] | .[] | [(.id // ""), (.name // ""), (.status // "")] | join("")
976
+ ([.phases[] | select(.id == $pid)][0].subs) // [] | .[] | [(.id // ""), (.name // ""), (.status // ""), ((.deps // []) | join(", "))] | join("")
892
977
  ')
893
978
  local j=0
894
- local sid sname sstatus sg connector
895
- while IFS=$'\x1f' read -r sid sname sstatus; do
979
+ local sid sname sstatus sdeps sg connector
980
+ while IFS=$'\x1f' read -r sid sname sstatus sdeps; do
981
+ # The card is the only surface a Copilot session has, so a dependency
982
+ # that is invisible here is invisible to that user entirely.
983
+ [ -n "$sdeps" ] && sname="$sname (bekliyor: $sdeps)"
896
984
  sg=$(glyph_for "$sstatus")
897
985
  connector="├─"
898
986
  if [ "$j" = "$((sub_count - 1))" ]; then connector="└─"; fi
@@ -1052,10 +1140,45 @@ GATE
1052
1140
  usage_live_ping "$(basename "$TRACKER_DIR")" "$PID"
1053
1141
  ;;
1054
1142
 
1143
+ plan)
1144
+ # Phase 2's plan, turned into sub-phases of the phase that will execute it.
1145
+ #
1146
+ # The parsing lives here rather than in the phase document for one reason:
1147
+ # sub-phases are this file's structure, and a jq blob in a phase doc is a
1148
+ # second definition of that structure which nothing keeps in step. It also
1149
+ # keeps the doc to one line, which the aggregate phase-doc budget cares about.
1150
+ #
1151
+ # Reads planning-output.schema.json on stdin: tasks[] with id, title and an
1152
+ # optional dependsOn[]. Status is `pending` for all of them - Phase 3 moves
1153
+ # them with `sub`, and pre-marking work as started is the lie the tracker
1154
+ # exists to avoid.
1155
+ need_jq
1156
+ [ "$#" -ge 1 ] || { echo "plan needs <phase_id> (plan JSON on stdin)" >&2; exit 64; }
1157
+ PID="$1"
1158
+ PLAN_IN=$(cat)
1159
+ printf '%s' "$PLAN_IN" | jq empty 2>/dev/null || {
1160
+ echo "phase-tracker: plan input is not valid JSON" >&2
1161
+ exit 65
1162
+ }
1163
+ N=0
1164
+ while IFS=$'\x1f' read -r tid ttitle tdeps; do
1165
+ [ -n "$tid" ] || continue
1166
+ "$SELF_PATH" sub "$PID" "$tid" "$ttitle" pending "$tdeps" >/dev/null || true
1167
+ N=$((N + 1))
1168
+ done <<EOF
1169
+ $(printf '%s' "$PLAN_IN" | jq -r '
1170
+ (.tasks // [])[]
1171
+ | [(.id // ""), (.title // ""), ((.dependsOn // []) | join(","))]
1172
+ | join("\u001f")')
1173
+ EOF
1174
+ [ "$N" -gt 0 ] || { echo "phase-tracker: plan carried no tasks[]" >&2; exit 65; }
1175
+ printf 'phase-tracker: %s plan step(s) attached to phase %s\n' "$N" "$PID"
1176
+ tiles_script
1177
+ ;;
1055
1178
  sub)
1056
1179
  need_jq
1057
- [ "$#" -ge 3 ] || { echo "sub needs <phase_id> <sub_id> <name> [status]" >&2; exit 64; }
1058
- PID="$1"; SID="$2"; SNAME="$3"; SSTATUS="${4:-pending}"
1180
+ [ "$#" -ge 3 ] || { echo "sub needs <phase_id> <sub_id> <name> [status] [deps]" >&2; exit 64; }
1181
+ PID="$1"; SID="$2"; SNAME="$3"; SSTATUS="${4:-pending}"; SDEPS="${5:-}"
1059
1182
  case "$SSTATUS" in
1060
1183
  pending|in_progress|completed|failed|skipped) ;;
1061
1184
  *) echo "bad status: $SSTATUS" >&2; exit 64 ;;
@@ -1065,13 +1188,15 @@ GATE
1065
1188
  # If sub with this id exists under this phase, update; else append.
1066
1189
  new=$(echo "$state" | jq \
1067
1190
  --arg pid "$PID" --arg sid "$SID" --arg sname "$SNAME" --arg ss "$SSTATUS" \
1191
+ --argjson deps "$(printf '%s\n' "$SDEPS" | jq -Rc 'split(",") | map(select(length > 0))')" \
1068
1192
  '
1069
1193
  .phases |= map(
1070
1194
  if .id == $pid then
1071
1195
  if any(.subs[]?; .id == $sid) then
1072
- .subs |= map(if .id == $sid then .name = $sname | .status = $ss else . end)
1196
+ .subs |= map(if .id == $sid then .name = $sname | .status = $ss
1197
+ | (if ($deps | length) > 0 then .deps = $deps else . end) else . end)
1073
1198
  else
1074
- .subs += [{"id":$sid,"name":$sname,"status":$ss}]
1199
+ .subs += [{"id":$sid,"name":$sname,"status":$ss,"deps":$deps}]
1075
1200
  end
1076
1201
  else . end
1077
1202
  )')
@@ -1243,10 +1368,7 @@ GATE
1243
1368
  # tools it has. And the fallback is not enough on its own: a user looking
1244
1369
  # at a missing widget needs the one command that brings it back, which is
1245
1370
  # why the opt-in is printed next to it.
1246
- echo "REQUIRED - create one native tile per phase, in this exact order,"
1247
- echo "BEFORE any TaskUpdate. The widget renders by creation order, not by"
1248
- echo "phase number, so an out-of-order call scrambles the stack."
1249
- subjects | sed 's/^/ TaskCreate(subject: "/; s/$/")/'
1371
+ tiles_script
1250
1372
  echo
1251
1373
  echo "At every phase boundary re-run \`phase-tracker.sh subjects <id>\` and"
1252
1374
  echo "pass that line as the subject of the TaskUpdate. The subject is the only"
@@ -22,7 +22,7 @@
22
22
  # second copy is a second place for the answer to drift.
23
23
  #
24
24
  # Usage:
25
- # probe-evidence-capability.sh --platform <ios|android> [--repo <path>]
25
+ # probe-evidence-capability.sh --platform <ios|android|web> [--repo <path>]
26
26
  # [--changed <f>[,<f>...]] [--json]
27
27
  # [--json-out <path>] [--only all|device]
28
28
  #
@@ -57,8 +57,8 @@ case "$ONLY" in all | device) ;; *)
57
57
  echo "probe-evidence-capability: --only takes 'all' or 'device'" >&2; exit 2 ;;
58
58
  esac
59
59
 
60
- case "$PLATFORM" in ios | android) ;; *)
61
- echo "usage: probe-evidence-capability.sh --platform <ios|android> [--repo <path>] [--changed <f>] [--json]" >&2
60
+ case "$PLATFORM" in ios | android | web) ;; *)
61
+ echo "usage: probe-evidence-capability.sh --platform <ios|android|web> [--repo <path>] [--changed <f>] [--json]" >&2
62
62
  exit 2 ;;
63
63
  esac
64
64
  [ -d "$REPO" ] || { echo "probe-evidence-capability: no such repo directory: $REPO" >&2; exit 2; }
@@ -120,6 +120,21 @@ case "$PLATFORM" in
120
120
  [ -n "$DEVICE" ] || DEVICE_REASON="no attached device or running emulator"
121
121
  fi
122
122
  ;;
123
+ web)
124
+ # The "device" on web is the browser the runner drives, and it is installed
125
+ # per project rather than per machine. `npx --no-install` is the same guard
126
+ # run-ui-tests.sh uses, so this answers the question the run will actually
127
+ # face instead of a friendlier one.
128
+ if ! command -v npx >/dev/null 2>&1; then
129
+ DEVICE_REASON="npx unavailable; cannot tell whether a browser runner is installed"
130
+ elif (cd "$REPO" && npx --no-install playwright --version >/dev/null 2>&1); then
131
+ DEVICE="playwright"
132
+ elif (cd "$REPO" && npx --no-install cypress version >/dev/null 2>&1); then
133
+ DEVICE="cypress"
134
+ else
135
+ DEVICE_REASON="no browser runner installed in this project (npm install first)"
136
+ fi
137
+ ;;
123
138
  esac
124
139
 
125
140
  # ---- Recorder -----------------------------------------------------------------
@@ -136,6 +151,15 @@ case "$PLATFORM" in
136
151
  RECORDER="false"; RECORDER_REASON="adb unavailable"
137
152
  fi
138
153
  ;;
154
+ web)
155
+ # Nothing external records a browser here: Playwright and Cypress both write
156
+ # their own video, so the recorder IS the runner. Claiming a recorder when
157
+ # the runner is absent would open tier 1 on a machine that cannot produce a
158
+ # single frame.
159
+ if [ -n "$DEVICE" ]; then RECORDER="true"; else
160
+ RECORDER="false"; RECORDER_REASON="${DEVICE_REASON:-no browser runner}"
161
+ fi
162
+ ;;
139
163
  esac
140
164
  if [ "$RECORDER" = "true" ] && ! command -v ffprobe >/dev/null 2>&1; then
141
165
  # Not fatal. ffprobe only verifies the result, so its absence downgrades what
@@ -15,7 +15,7 @@
15
15
  # asked which test depth to run, and a second copy of "does this repo have a UI
16
16
  # test target" is a second place for the answer to drift.
17
17
  #
18
- # run-ui-tests.sh detect --platform <ios|android> [--repo <path>] [--changed <f>[,<f>...]]
18
+ # run-ui-tests.sh detect --platform <ios|android|web> [--repo <path>] [--changed <f>[,<f>...]]
19
19
  # Print KEY=VALUE lines and exit. Never builds, never runs a test.
20
20
  # UI_TEST_TARGET=<name>| (empty until one candidate is chosen)
21
21
  # UI_TEST_TARGETS=<name[,name...]> (every candidate found)
@@ -25,7 +25,7 @@
25
25
  # UI_TEST_CONTAINER=<-project X|-workspace X>|
26
26
  # UI_TEST_SCHEME=<name>|
27
27
  #
28
- # run-ui-tests.sh run --platform <ios|android> [--repo <path>] [--changed <f>...]
28
+ # run-ui-tests.sh run --platform <ios|android|web> [--repo <path>] [--changed <f>...]
29
29
  # [--device <udid|serial>] [--log <path>] [--all]
30
30
  # Run the matching tests (or the whole UI suite with --all).
31
31
  #
@@ -59,10 +59,10 @@ while [ "$#" -gt 0 ]; do
59
59
  done
60
60
 
61
61
  case "$MODE" in detect | run) ;; *)
62
- echo "usage: run-ui-tests.sh detect|run --platform <ios|android> [--repo <path>] [--changed <f>]" >&2
62
+ echo "usage: run-ui-tests.sh detect|run --platform <ios|android|web> [--repo <path>] [--changed <f>]" >&2
63
63
  exit 2 ;;
64
64
  esac
65
- case "$PLATFORM" in ios | android) ;; *)
65
+ case "$PLATFORM" in ios | android | web) ;; *)
66
66
  echo "run-ui-tests: unsupported platform '$PLATFORM'" >&2; exit 2 ;;
67
67
  esac
68
68
  [ -d "$REPO" ] || { echo "run-ui-tests: no such repo directory: $REPO" >&2; exit 2; }
@@ -277,9 +277,94 @@ EOF
277
277
  fi
278
278
  }
279
279
 
280
+ # Web had no arm here at all until v17.1.0, which meant a repo with a full
281
+ # Playwright suite reported "no UI test target" and every downstream consumer
282
+ # believed it: the evidence probe closed tier 1, the PR body said UI tests were
283
+ # not run, and the reason it gave was true of the runner rather than of the repo.
284
+ # That is the failure mode this file exists to avoid, and it was shipped for web.
285
+ web_spec_files() {
286
+ find "$REPO" \
287
+ -name ".*" -type d -prune -o \
288
+ -type d \( -name node_modules -o -name dist -o -name build -o -name out \
289
+ -o -name coverage -o -name .next -o -name test-results \
290
+ -o -name playwright-report \) -prune -o \
291
+ -type f \( -name "*.spec.ts" -o -name "*.spec.tsx" -o -name "*.spec.js" \
292
+ -o -name "*.spec.mjs" -o -name "*.e2e.ts" -o -name "*.e2e.js" \
293
+ -o -name "*.cy.ts" -o -name "*.cy.tsx" -o -name "*.cy.js" \) \
294
+ -print 2>/dev/null
295
+ }
296
+
297
+ # The signal is the import, not a directory called e2e. A `*.spec.ts` under
298
+ # `src/` is almost always a vitest or jest unit test rendering into jsdom: no
299
+ # browser, no navigation, nothing to record a flow from. `@playwright/test` and
300
+ # `cy.visit(` are the two APIs that actually drive a browser, which is the same
301
+ # precondition XCUIApplication carries on iOS - and the same reason the iOS arm
302
+ # refuses to treat 477 snapshot tests as UI tests.
303
+ detect_web() {
304
+ local f pw="" cy="" n
305
+ while IFS= read -r f; do
306
+ [ -n "$f" ] || continue
307
+ if grep -qF '@playwright/test' "$f" 2>/dev/null; then
308
+ pw="${pw}${f}
309
+ "
310
+ elif grep -qE 'cy\.(visit|get)\(' "$f" 2>/dev/null; then
311
+ cy="${cy}${f}
312
+ "
313
+ fi
314
+ done <<EOF
315
+ $(web_spec_files)
316
+ EOF
317
+
318
+ [ -n "$pw$cy" ] || { TARGET_REASON="no spec imports @playwright/test and none calls cy.visit() under $REPO"; return; }
319
+
320
+ CONTAINER="$REPO"
321
+ [ -n "$pw" ] && TARGETS="playwright"
322
+ [ -n "$cy" ] && TARGETS="${TARGETS:+$TARGETS,}cypress"
323
+ case "$TARGETS" in
324
+ *,*) TARGET="" ;;
325
+ *) TARGET="$TARGETS"; SCHEME="$TARGET" ;;
326
+ esac
327
+
328
+ [ -n "$CHANGED" ] || { MATCH_REASON="no changed-file list supplied"; return; }
329
+ local names hits="" hit_runner=""
330
+ names=$(changed_names)
331
+ [ -n "$names" ] || { MATCH_REASON="changed-file list held no usable names"; return; }
332
+
333
+ # Matches are spec FILES, not test titles: both runners take a path on the
334
+ # command line, and a title filter (`-g`) silently matches nothing when the
335
+ # title was reworded.
336
+ while IFS= read -r f; do
337
+ [ -n "$f" ] || continue
338
+ while IFS= read -r n; do
339
+ [ -n "$n" ] || continue
340
+ if grep -qF "$n" "$f" 2>/dev/null; then
341
+ rel="${f#"$REPO"/}"
342
+ case ",$hits," in *",$rel,"*) ;; *) hits="${hits:+$hits,}$rel" ;; esac
343
+ if [ -z "$hit_runner" ]; then
344
+ grep -qF '@playwright/test' "$f" 2>/dev/null && hit_runner="playwright" || hit_runner="cypress"
345
+ fi
346
+ break
347
+ fi
348
+ done <<EOF2
349
+ $names
350
+ EOF2
351
+ done <<EOF
352
+ $pw$cy
353
+ EOF
354
+
355
+ MATCHES="$hits"
356
+ if [ -n "$MATCHES" ]; then
357
+ [ -n "$TARGET" ] || { TARGET="$hit_runner"; SCHEME="$TARGET"; }
358
+ else
359
+ MATCH_REASON="no browser-driving spec mentions any changed file name"
360
+ [ -n "$TARGET" ] || TARGET_REASON="both playwright and cypress specs exist and no match chose between them"
361
+ fi
362
+ }
363
+
280
364
  case "$PLATFORM" in
281
365
  ios) detect_ios ;;
282
366
  android) detect_android ;;
367
+ web) detect_web ;;
283
368
  esac
284
369
 
285
370
  if [ "$MODE" = "detect" ]; then
@@ -366,6 +451,30 @@ case "$PLATFORM" in
366
451
  (cd "$REPO" && "$GRADLE" "$TARGET" $ARGS) >"$LOG" 2>&1
367
452
  RC=$?
368
453
  ;;
454
+ web)
455
+ command -v npx >/dev/null 2>&1 || { echo "run-ui-tests: npx unavailable" >&2; exit 2; }
456
+ WEB_ARGV=()
457
+ case "$TARGET" in
458
+ playwright)
459
+ # --no-install: npx would otherwise DOWNLOAD the runner on a machine that
460
+ # does not have it, which turns "this repo has no UI tests" into a silent
461
+ # network fetch mid-phase.
462
+ WEB_ARGV=(npx --no-install playwright test --reporter=line)
463
+ if [ "$RUN_ALL" -eq 0 ]; then
464
+ OLDIFS="$IFS"; IFS=','
465
+ for m in $MATCHES; do WEB_ARGV+=("$m"); done
466
+ IFS="$OLDIFS"
467
+ fi
468
+ ;;
469
+ cypress)
470
+ WEB_ARGV=(npx --no-install cypress run)
471
+ [ "$RUN_ALL" -eq 0 ] && WEB_ARGV+=(--spec "$MATCHES")
472
+ ;;
473
+ *) echo "run-ui-tests: unknown web runner '$TARGET'" >&2; exit 2 ;;
474
+ esac
475
+ (cd "$REPO" && "${WEB_ARGV[@]}") >"$LOG" 2>&1
476
+ RC=$?
477
+ ;;
369
478
  esac
370
479
 
371
480
  printf 'UI_TEST_LOG=%s\n' "$LOG"
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "schemaVersion": "1.0.0",
3
- "generatedAt": "2026-09-13T12:09:01Z",
4
- "skillCount": 209,
3
+ "generatedAt": "2026-09-14T11:56:29Z",
4
+ "skillCount": 212,
5
5
  "entries": [
6
6
  {
7
7
  "path": "shared/core/apple-archive-compliance/SKILL.md",
@@ -23,6 +23,18 @@
23
23
  "path": "shared/core/multi-agent-analysis/SKILL.md",
24
24
  "sha256": "21bb520e781123d998e4ef5f6ee54e7403774f64e094153645cbde4112a7a8a6"
25
25
  },
26
+ {
27
+ "path": "shared/core/multi-agent-autopilot-off/SKILL.md",
28
+ "sha256": "c141ad85bcba8320d5108708db2045cb3d3b1578b564eb1724f4c81202bdd133"
29
+ },
30
+ {
31
+ "path": "shared/core/multi-agent-autopilot-on/SKILL.md",
32
+ "sha256": "2d9c12084bb7c74f38a35c810a0a20648847173bf0b1a3bc4c8c4318a5dcb0d1"
33
+ },
34
+ {
35
+ "path": "shared/core/multi-agent-autopilot-status/SKILL.md",
36
+ "sha256": "065958fb3ff608d7d0ea36b3eef4614da442742dc69b00799cf1658848cee786"
37
+ },
26
38
  {
27
39
  "path": "shared/core/multi-agent-autopilot/SKILL.md",
28
40
  "sha256": "c02525ad003c34d53834ecd04eb150764b36779f4e08941c9016926418e74333"
@@ -33,7 +45,7 @@
33
45
  },
34
46
  {
35
47
  "path": "shared/core/multi-agent-channels/SKILL.md",
36
- "sha256": "e43f5cfa92d5016b296710863b4cab91eb7c0be0ce13018a1c34d15807604fa7"
48
+ "sha256": "fa128726aa0009699e7ac2de566812e7192681db98c90dfc5da501ad50a05583"
37
49
  },
38
50
  {
39
51
  "path": "shared/core/multi-agent-complaint-analysis/SKILL.md",
@@ -189,7 +201,7 @@
189
201
  },
190
202
  {
191
203
  "path": "shared/core/multi-agent-sync/SKILL.md",
192
- "sha256": "8f033b904e90926b15f6e5a8f8703b47a17373e5450921514633ea0a9edaf327"
204
+ "sha256": "51e9a8a9053ce9d0450bcaf66555075f83c340203bb753946938f4b29cf93b79"
193
205
  },
194
206
  {
195
207
  "path": "shared/core/multi-agent-test-accessibility/SKILL.md",
@@ -0,0 +1,67 @@
1
+ ---
2
+ name: multi-agent-autopilot-off
3
+ language: en
4
+ description: "Turn off continuous mode on this machine: the schedule is removed, work already running finishes, the repo selection is kept. Use when the machine should stop picking work up."
5
+ user-invocable: true
6
+ argument-hint: "[--now] - --now also stops the item currently running"
7
+ ---
8
+
9
+
10
+ # multi-agent autopilot-off - stop picking work up
11
+
12
+ Removes the schedule. **Work already in flight finishes** unless you pass
13
+ `--now`: an item stopped mid-development leaves a worktree and a half-written
14
+ branch, which is the exact state this release spent its time cleaning up.
15
+
16
+ ## Steps
17
+
18
+ ### 1. Say what is in flight before stopping anything
19
+
20
+ ```bash
21
+ bash "$HOME/.claude/scripts/autopilot-status.sh" --short
22
+ ```
23
+
24
+ When something is running, name it and how long it has been going. "Stopped"
25
+ means something different when an item is 2 minutes from a PR than when the queue
26
+ is idle, and the user is the one who knows which.
27
+
28
+ ### 2. Remove the schedule
29
+
30
+ ```bash
31
+ L="com.multi-agent.autopilot"
32
+ launchctl bootout "gui/$(id -u)/$L" 2>/dev/null || launchctl unload "$HOME/Library/LaunchAgents/$L.plist" 2>/dev/null
33
+ rm -f "$HOME/Library/LaunchAgents/$L.plist"
34
+ ```
35
+
36
+ The plist is removed rather than left disabled, because its presence is the
37
+ definition of on: a disabled-but-present job is a third state nobody can read
38
+ from the outside.
39
+
40
+ ### 3. Release the sleep assertion and the indicator
41
+
42
+ ```bash
43
+ pkill -f 'caffeinate -i -w' 2>/dev/null || true # only the one this mode holds
44
+ pkill -f "$HOME/.claude/autopilot/bin/menubar" 2>/dev/null || true
45
+ ```
46
+
47
+ ### 4. `--now` only: end the in-flight item deliberately
48
+
49
+ Without `--now` nothing here runs. With it, the child session is stopped and the
50
+ run is marked `abandoned`, its worktree is removed **unless it holds uncommitted
51
+ work**, in which case the work is stashed to `autopilot/abandoned/<task-id>` and
52
+ the worktree is kept. Same contract as `gc-abandoned.sh`; losing a day of edits
53
+ is worse than 750 MB.
54
+
55
+ ### 5. Keep the selection
56
+
57
+ `~/.claude/autopilot/config.json`, `queue.json` and `attempted.jsonl` all stay.
58
+ Turning the mode back on must not re-ask which repos, and the attempt history is
59
+ what stops an item that already failed twice from being retried forever.
60
+
61
+ To forget the selection as well: `rm -rf ~/.claude/autopilot`. Say that rather
62
+ than doing it - "off" and "forget everything" are different requests.
63
+
64
+ ### 6. Report
65
+
66
+ State that it is off, what was left running or finishing, and that the repo
67
+ selection is kept. If an item was stashed, name the branch.
@@ -0,0 +1,146 @@
1
+ ---
2
+ name: multi-agent-autopilot-on
3
+ language: en
4
+ description: "Turn on continuous mode on THIS machine: pick the repos it watches; labelled GitHub issues and assigned+labelled Jira items then run in a worktree and stop at an open PR. Use when the machine should pick work up unattended."
5
+ user-invocable: true
6
+ argument-hint: "(no arguments - opens the repo picker)"
7
+ ---
8
+
9
+
10
+ # multi-agent autopilot-on - continuous mode, on this machine
11
+
12
+ Turns on the mode that picks work up without being asked. **No repo is included
13
+ by default and none is ever added implicitly** - you choose, here, every time.
14
+
15
+ Why a picker rather than a label alone: measured on one machine, 66 repos grant
16
+ push and a good half belong to other people. A label is a filter; anyone who can
17
+ open an issue in a repo you happen to have push on could add one. The gate is
18
+ this selection, and it is per machine.
19
+
20
+ ## Steps
21
+
22
+ ### 1. Prerequisites, stated before anything is written
23
+
24
+ ```bash
25
+ gh auth status >/dev/null 2>&1 || echo "BLOCKED: gh is not authenticated - run 'gh auth login'"
26
+ command -v jq >/dev/null 2>&1 || echo "BLOCKED: jq is required"
27
+ node "$HOME/.claude/scripts/doctor.mjs" >/dev/null 2>&1; [ "$?" -ge 2 ] && echo "BLOCKED: doctor reports a blocking problem - fix it first"
28
+ ```
29
+
30
+ Any BLOCKED line stops here and is reported. Arming unattended work on an install
31
+ that `doctor` calls broken produces a queue of failures, one per item.
32
+
33
+ ### 2. The repo list, both halves of it
34
+
35
+ ```bash
36
+ bash "$HOME/.claude/lib/autopilot-activation.sh" list
37
+ ```
38
+
39
+ Two states come back and **both are shown**:
40
+
41
+ - `eligible` - push rights and a checkout on this machine
42
+ - `unavailable` - push rights, no local checkout, listed WITH that reason
43
+
44
+ Never hide the second list. Dropping 51 of 66 rows silently looks exactly like a
45
+ permissions problem, and the user goes hunting for a token fault that does not
46
+ exist.
47
+
48
+ ### 3. Pick, with the current selection already checked
49
+
50
+ Read `~/.claude/autopilot/config.json` if it exists, then one `AskUserQuestion`
51
+ (`multiSelect: true`) over the **eligible** rows:
52
+
53
+ - `question` (in `outputLanguage`): which repos this machine should pick work up from
54
+ - `header`: "Repos" (English, UI contract)
55
+ - `options`: one per eligible repo, currently-selected ones marked "(seçili)" / "(selected)"
56
+
57
+ **This is how repos are added and removed.** Checking a new one adds it;
58
+ unchecking an existing one removes it. There is no `add-repo` / `remove-repo`
59
+ command, because two entry points to one list is two lists.
60
+
61
+ Unavailable repos are printed above the question as a plain list with their
62
+ reason, not offered as options.
63
+
64
+ ### 4. Sources and label, per selected repo
65
+
66
+ One `AskUserQuestion` per newly added repo (already-configured ones keep their
67
+ answers - re-running must not re-ask what was already settled):
68
+
69
+ - Sources: `GitHub issues` / `Jira` / both
70
+ - Label: default **`agent-queue`** on both sides, editable
71
+
72
+ The same name works on both sides and both mechanisms already exist:
73
+
74
+ | Source | Query | Extra condition |
75
+ |---|---|---|
76
+ | GitHub | `gh issue list --repo <r> --label <label> --state open` | none - the label is the whole gate |
77
+ | Jira | `assignee = currentUser() AND labels = "<label>" AND resolution = EMPTY AND status not in (Done, Closed, Cancelled)` | **assigned to you**, so a label somebody else adds is not enough |
78
+
79
+ Two Jira caveats worth saying out loud when Jira is selected: labels are one
80
+ global namespace across the whole instance and anyone can write them, so on a
81
+ shared instance prefer a qualified name (`agent-queue-<team>`); and the Labels
82
+ field has to be on the edit screen for the issue types in play. An instance where
83
+ neither holds is a `jiraJql` config change - a saved filter, a component - not a
84
+ redesign.
85
+
86
+ ### 5. Write the config
87
+
88
+ Validate against `schemas/autopilot-config.schema.json`, then write
89
+ `~/.claude/autopilot/config.json` (0600, in a 0700 directory) via
90
+ `ma_ap_write` in `lib/autopilot-state.sh`. Defaults that are not asked:
91
+ `slots: 1`, `scanIntervalSeconds: 120`, `maxAttempts: 2`,
92
+ `askOnLowMaturity: true`, `costCeilingUsd: 25`, `depthRouter: "off"`.
93
+
94
+ ### 6. Build the menu bar indicator (best effort)
95
+
96
+ ```bash
97
+ SRC="$HOME/.claude/scripts/autopilot-menubar.swift"
98
+ OUT="$HOME/.claude/autopilot/bin/menubar"
99
+ if command -v swiftc >/dev/null 2>&1; then
100
+ swiftc -O "$SRC" -o "$OUT" 2>/dev/null && chmod 700 "$OUT"
101
+ fi
102
+ ```
103
+
104
+ Built from source on demand, never shipped as a binary: the package contains only
105
+ text today, and a compiled artifact in it would need signing and notarization to
106
+ be distributable. No `swiftc` means no indicator and nothing else changes - say
107
+ so in one line rather than failing. `autopilot-status` works either way.
108
+
109
+ ### 7. Confirm the schedule, naming the file
110
+
111
+ Render `templates/multi-agent-autopilot.plist.template` and show the resolved
112
+ **path** and interval, then `AskUserQuestion`:
113
+
114
+ - `question`: install the schedule at `~/Library/LaunchAgents/com.multi-agent.autopilot.plist` and start picking work up?
115
+ - options: `{ label: "Start" }` / `{ label: "Configure only", description: "Save the repo selection, do not schedule anything yet" }`
116
+
117
+ "Configure only" is a real answer and the safe default to offer first: the queue
118
+ can be watched with `autopilot-status` for a while before anything runs.
119
+
120
+ On **Start**:
121
+
122
+ ```bash
123
+ launchctl bootstrap "gui/$(id -u)" "$HOME/Library/LaunchAgents/com.multi-agent.autopilot.plist" 2>/dev/null \
124
+ || launchctl load "$HOME/Library/LaunchAgents/com.multi-agent.autopilot.plist"
125
+ ```
126
+
127
+ ### 8. Report what is now true
128
+
129
+ Say all five, because each one is something a user has asked about after the
130
+ fact: which repos and which labels; that it survives a restart with no command to
131
+ run (launchd loads at **login** - not boot, and that is correct, because the login
132
+ keychain is what unlocks the tokens, so a run before login could not reach Jira or
133
+ GitHub anyway); that the machine will be kept awake **only on AC**; the slot count
134
+ and the ceiling this machine would allow (`ma_ap_slot_ceiling`); and that
135
+ `/multi-agent:autopilot-off` is the only way to stop it.
136
+
137
+ ## What this does not do
138
+
139
+ - **No cap on PRs.** As many items as carry the label become as many PRs. The
140
+ bounds are `costCeilingUsd` over a rolling 24 hours and what the machine can
141
+ hold - never a daily count.
142
+ - **Does not merge.** The runner stops at an open PR; the merge decision stays
143
+ yours.
144
+ - **Does not touch an attended run.** Both are worktrees and per-repo concurrency
145
+ is 1, so the queue skips a repo you are working in rather than competing for
146
+ `.git/index.lock`.