@mmerterden/multi-agent-pipeline 18.0.0 → 19.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (234) hide show
  1. package/CHANGELOG.md +287 -0
  2. package/README.md +36 -20
  3. package/README.tr.md +14 -16
  4. package/docs/adr/0002-instruction-driven-flag.md +1 -0
  5. package/docs/adr/0005-lazy-phase-docs.md +11 -1
  6. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  7. package/docs/adr/0010-own-code-graph.md +1 -0
  8. package/docs/adr/0014-six-phase-consolidation.md +134 -0
  9. package/docs/adr/README.md +2 -1
  10. package/docs/architecture.md +37 -38
  11. package/docs/best-practices.md +1 -1
  12. package/docs/ecosystem.md +46 -27
  13. package/docs/engineering.md +1 -1
  14. package/docs/facts.json +61 -0
  15. package/docs/features.md +55 -54
  16. package/docs/performance.md +5 -5
  17. package/docs/recovery-guide.md +17 -17
  18. package/docs/token-budget-history.md +3 -1
  19. package/index.js +2 -2
  20. package/install/_codex-agents.mjs +1 -1
  21. package/install/templates/claude-hooks.json +1 -1
  22. package/install/templates/codex-instructions.md +1 -1
  23. package/install/templates/copilot-instructions.md +28 -28
  24. package/manifest.json +234 -216
  25. package/package.json +2 -2
  26. package/pipeline/agents/dev-critic.md +7 -7
  27. package/pipeline/commands/figma-to-swiftui.md +1 -1
  28. package/pipeline/commands/multi-agent/SKILL.md +9 -9
  29. package/pipeline/commands/multi-agent/analysis/SKILL.md +15 -15
  30. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -2
  31. package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
  32. package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
  33. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
  34. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  35. package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
  36. package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
  37. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  38. package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
  39. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
  40. package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
  41. package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
  42. package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
  43. package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
  44. package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
  45. package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
  46. package/pipeline/commands/multi-agent/review/SKILL.md +2 -2
  47. package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
  48. package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
  49. package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
  50. package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
  51. package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
  52. package/pipeline/commands/multi-agent/status/SKILL.md +5 -5
  53. package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
  54. package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
  55. package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
  56. package/pipeline/lib/credential-inventory.sh +1 -1
  57. package/pipeline/lib/fetch-fortify.sh +1 -1
  58. package/pipeline/lib/model-dispatch.sh +140 -0
  59. package/pipeline/lib/model-rung.sh +142 -0
  60. package/pipeline/lib/outbound-gate.mjs +14 -0
  61. package/pipeline/lib/phase-schema.mjs +88 -0
  62. package/pipeline/lib/plan-todos.sh +5 -5
  63. package/pipeline/lib/route-state.sh +161 -0
  64. package/pipeline/lib/run-paths.sh +2 -2
  65. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  66. package/pipeline/multi-agent-refs/_dev-context.md +6 -6
  67. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  68. package/pipeline/multi-agent-refs/analysis/evidence.md +2 -11
  69. package/pipeline/multi-agent-refs/analysis/intake.md +7 -7
  70. package/pipeline/multi-agent-refs/analysis/locked.md +48 -22
  71. package/pipeline/multi-agent-refs/analysis/redesign.md +1 -1
  72. package/pipeline/multi-agent-refs/analysis/render.md +10 -10
  73. package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
  74. package/pipeline/multi-agent-refs/analysis/review.md +2 -2
  75. package/pipeline/multi-agent-refs/analysis/synthesis.md +13 -7
  76. package/pipeline/multi-agent-refs/analysis-template-corporate.md +9 -9
  77. package/pipeline/multi-agent-refs/analysis-template.md +19 -19
  78. package/pipeline/multi-agent-refs/android-guide.md +1 -1
  79. package/pipeline/multi-agent-refs/audit-guide.md +13 -13
  80. package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
  81. package/pipeline/multi-agent-refs/channels/jira.md +3 -3
  82. package/pipeline/multi-agent-refs/channels/pr.md +4 -4
  83. package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
  84. package/pipeline/multi-agent-refs/component-dispatch.md +8 -8
  85. package/pipeline/multi-agent-refs/conventions-defaults.md +2 -2
  86. package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
  87. package/pipeline/multi-agent-refs/features/analysis-jira.md +1 -1
  88. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +4 -4
  89. package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
  90. package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
  91. package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
  92. package/pipeline/multi-agent-refs/features/doctor.md +3 -3
  93. package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
  94. package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
  95. package/pipeline/multi-agent-refs/features/model-fallback.md +41 -5
  96. package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
  97. package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
  98. package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
  99. package/pipeline/multi-agent-refs/features/review-multi-repo.md +2 -2
  100. package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
  101. package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
  102. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
  103. package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
  104. package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
  105. package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
  106. package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
  107. package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
  108. package/pipeline/multi-agent-refs/knowledge.md +11 -11
  109. package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
  110. package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
  111. package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
  112. package/pipeline/multi-agent-refs/phases/modes.md +30 -30
  113. package/pipeline/multi-agent-refs/phases/operations.md +8 -8
  114. package/pipeline/multi-agent-refs/phases/phase-0-init.md +24 -24
  115. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
  116. package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
  117. package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
  118. package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
  119. package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
  120. package/pipeline/multi-agent-refs/phases.md +44 -48
  121. package/pipeline/multi-agent-refs/picker-contract.md +1 -1
  122. package/pipeline/multi-agent-refs/progress-contract.md +6 -6
  123. package/pipeline/multi-agent-refs/readiness-review.md +1 -1
  124. package/pipeline/multi-agent-refs/rules.md +7 -7
  125. package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
  126. package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
  127. package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
  128. package/pipeline/preferences-template.json +9 -1
  129. package/pipeline/rules/figma-pipeline.md +8 -8
  130. package/pipeline/rules/outside-the-pipeline.md +1 -1
  131. package/pipeline/schemas/agent-state.schema.json +50 -50
  132. package/pipeline/schemas/analysis-output.schema.json +3 -3
  133. package/pipeline/schemas/analysis-spec.schema.json +2 -2
  134. package/pipeline/schemas/autopilot-config.schema.json +1 -1
  135. package/pipeline/schemas/code-graph.schema.json +1 -1
  136. package/pipeline/schemas/criteria-manifest.schema.json +1 -1
  137. package/pipeline/schemas/dev-critic-output.schema.json +1 -1
  138. package/pipeline/schemas/diff-risk.schema.json +1 -1
  139. package/pipeline/schemas/figma-project-config.schema.json +1 -1
  140. package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
  141. package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
  142. package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
  143. package/pipeline/schemas/phases.json +105 -0
  144. package/pipeline/schemas/plan-todos.schema.json +5 -5
  145. package/pipeline/schemas/planning-output.schema.json +1 -1
  146. package/pipeline/schemas/prefs.schema.json +102 -58
  147. package/pipeline/schemas/reviewer-output.schema.json +3 -3
  148. package/pipeline/schemas/route-config.schema.json +74 -0
  149. package/pipeline/schemas/scope-check.schema.json +1 -1
  150. package/pipeline/schemas/secret-patterns.json +124 -0
  151. package/pipeline/schemas/test-gap.schema.json +1 -1
  152. package/pipeline/schemas/token-budget.json +12 -18
  153. package/pipeline/schemas/triage-output.schema.json +6 -6
  154. package/pipeline/scripts/README.md +3 -3
  155. package/pipeline/scripts/_code-graph.mjs +2 -2
  156. package/pipeline/scripts/_run-paths.mjs +2 -2
  157. package/pipeline/scripts/_smoke-root.sh +1 -1
  158. package/pipeline/scripts/aggregate-metrics.mjs +1 -1
  159. package/pipeline/scripts/build-references.mjs +2 -2
  160. package/pipeline/scripts/bulk-read.sh +10 -1
  161. package/pipeline/scripts/capture-flush.sh +8 -8
  162. package/pipeline/scripts/capture-resume.sh +3 -3
  163. package/pipeline/scripts/classify-plan-safety.mjs +1 -1
  164. package/pipeline/scripts/cost-table.json +8 -1
  165. package/pipeline/scripts/diff-explain.mjs +1 -1
  166. package/pipeline/scripts/doctor.mjs +3 -3
  167. package/pipeline/scripts/gc-abandoned.sh +3 -3
  168. package/pipeline/scripts/gc-tmp.sh +1 -1
  169. package/pipeline/scripts/gc-worktrees.sh +1 -1
  170. package/pipeline/scripts/gen-facts.mjs +280 -0
  171. package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
  172. package/pipeline/scripts/gen-ref-toc.mjs +1 -1
  173. package/pipeline/scripts/graph-report.mjs +1 -1
  174. package/pipeline/scripts/jira-attach.sh +1 -1
  175. package/pipeline/scripts/learn-from-transcripts.mjs +1 -1
  176. package/pipeline/scripts/learning-curve.mjs +2 -2
  177. package/pipeline/scripts/log-metric.sh +17 -4
  178. package/pipeline/scripts/memory-save.sh +1 -1
  179. package/pipeline/scripts/migrate-prefs.mjs +22 -5
  180. package/pipeline/scripts/phase-banner.sh +20 -20
  181. package/pipeline/scripts/phase-tracker.sh +12 -12
  182. package/pipeline/scripts/plan-coverage-gate.mjs +2 -2
  183. package/pipeline/scripts/pre-commit-check.sh +30 -1
  184. package/pipeline/scripts/render-agent-log-cost.sh +1 -1
  185. package/pipeline/scripts/render-work-summary.sh +3 -3
  186. package/pipeline/scripts/review-file-filter.mjs +1 -1
  187. package/pipeline/scripts/run-aggregator.mjs +13 -6
  188. package/pipeline/scripts/run-metrics.mjs +1 -1
  189. package/pipeline/scripts/runs-index.mjs +11 -1
  190. package/pipeline/scripts/scan-skills.sh +26 -0
  191. package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
  192. package/pipeline/scripts/smoke-schema-validation.sh +26 -7
  193. package/pipeline/scripts/token-budget-report.mjs +13 -2
  194. package/pipeline/scripts/triage-memory.mjs +2 -2
  195. package/pipeline/scripts/validate-analysis-doc.mjs +274 -43
  196. package/pipeline/scripts/validate-planning.mjs +1 -1
  197. package/pipeline/scripts/validate-reviewer.mjs +1 -1
  198. package/pipeline/scripts/validate-state.mjs +45 -5
  199. package/pipeline/scripts/validate-triage.mjs +3 -3
  200. package/pipeline/scripts/verify-citations.mjs +1 -1
  201. package/pipeline/scripts/worktree-finalize.sh +5 -5
  202. package/pipeline/scripts/write-state.mjs +32 -0
  203. package/pipeline/skills/.skill-manifest.json +38 -22
  204. package/pipeline/skills/.skills-index.json +49 -5
  205. package/pipeline/skills/shared/README.md +10 -6
  206. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +8 -8
  207. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +8 -8
  208. package/pipeline/skills/shared/core/multi-agent/SKILL.md +81 -82
  209. package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
  210. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
  211. package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
  212. package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
  213. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
  214. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  215. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
  216. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
  217. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
  218. package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
  219. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
  220. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
  221. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
  222. package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
  223. package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
  224. package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
  225. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
  226. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +5 -5
  227. package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
  228. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
  229. package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +1 -1
  230. package/pipeline/skills/shared/external/signal-community/SKILL.md +8 -1
  231. package/pipeline/skills/skills-index.md +8 -4
  232. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
  233. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
  234. package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
@@ -43,7 +43,7 @@
43
43
  #
44
44
  # Examples:
45
45
  # phase-tracker.sh init "TASK-123"
46
- # for p in 0:Init 1:Analysis 2:Planning 3:Dev 4:Review 5:Test 6:Commit 7:Report; do
46
+ # for p in 0:Init 1:Plan 2:Dev 3:Review 4:Commit 5:Report; do
47
47
  # phase-tracker.sh add "${p%%:*}" "${p#*:}"
48
48
  # done
49
49
  # phase-tracker.sh update 0 in_progress
@@ -141,7 +141,7 @@ need_jq() {
141
141
  # Live usage ping (best-effort). Emits one per-phase update to the private
142
142
  # dashboard via usage-report.mjs, always status=running so per-phase updates
143
143
  # never fold the run into rollup counters - the terminal fold comes only from
144
- # the Phase 7 / halt emit that reads the real run status. The emitter no-ops
144
+ # the Phase 5 / halt emit that reads the real run status. The emitter no-ops
145
145
  # unless prefs.global.usageLog.enabled; detached so it never blocks a boundary.
146
146
  usage_live_ping() {
147
147
  # Smoke runs exercise this script's state handling, never the live dashboard -
@@ -461,7 +461,7 @@ save_state() {
461
461
 
462
462
  # --- read-modify-write lock ---------------------------------------------------
463
463
  # tmp+rename makes each save atomic, but two concurrent invocations (e.g. two
464
- # parallel Phase 4 reviewers reporting `tokens`) both load the same state and
464
+ # parallel Phase 3 reviewers reporting `tokens`) both load the same state and
465
465
  # the second save silently drops the first delta. Guard every load->save
466
466
  # sequence with a portable mkdir spinlock (macOS bash 3.2: no flock builtin).
467
467
  # The wait is bounded and the lock FAILS OPEN with a warning so a stuck lock
@@ -612,8 +612,8 @@ tracker_next_hint() {
612
612
  # Not every session carries the task tools: Claude Code provides them by
613
613
  # default only up to Opus 4.7 / Sonnet 4.6, a default that landed in
614
614
  # v2.1.268. Naming the fallback on the same line is what keeps a newer model
615
- # from advancing eight phases in silence.
616
- # `subjects` now carries Phase 2's plan steps as indented rows, and
615
+ # from advancing six phases in silence.
616
+ # `subjects` now carries Phase 1's plan steps as indented rows, and
617
617
  # update_plan takes the whole list anyway, so re-reading it is what puts
618
618
  # those steps on the Codex plan without a second mechanism. Codex has no
619
619
  # dependency concept; the "(bekliyor: ...)" suffix inside the step text is
@@ -643,7 +643,7 @@ format_span() {
643
643
  }
644
644
 
645
645
  # The end-of-run report: what the pipeline spent, phase by phase, and what it
646
- # has to say about phases it could not price. Printed by Phase 7 next to the
646
+ # has to say about phases it could not price. Printed by Phase 5 next to the
647
647
  # work summary, which covers what actually changed on disk.
648
648
  report() {
649
649
  need_jq
@@ -776,9 +776,9 @@ subjects() {
776
776
  # Sub-phases ride out on the SAME list, indented in the subject string.
777
777
  #
778
778
  # The card has drawn these since sub-phases existed; the widget never has,
779
- # and the widget is the surface the user actually looks at. Phase 2's plan
779
+ # and the widget is the surface the user actually looks at. Phase 1's plan
780
780
  # is the case that made the gap matter: the tasks, their order and their
781
- # dependencies are computed, stored and used to drive Phase 3's picker, and
781
+ # dependencies are computed, stored and used to drive Phase 2's picker, and
782
782
  # none of it was visible anywhere the user was looking.
783
783
  #
784
784
  # Indentation is two spaces INSIDE the subject because the widget takes
@@ -800,7 +800,7 @@ subjects() {
800
800
  #
801
801
  # Creation order is the ONLY ordering the native widget has - it renders by the
802
802
  # order tiles were made, not by any id inside them - so a plan that arrives at
803
- # Phase 2 cannot simply be appended: its rows would land after Phase 7. The
803
+ # Phase 2 cannot simply be appended: its rows would land after Phase 5. The
804
804
  # answer is a rebuild at that one boundary, which is the same thing `:resume`
805
805
  # already does for a different reason.
806
806
  tiles_script() {
@@ -811,7 +811,7 @@ tiles_script() {
811
811
  has_subs=$(echo "$state" | jq '[.phases[]?.subs[]?] | length')
812
812
 
813
813
  # A list carrying sub-steps has to be rebuilt whole, so the rebuild wins over a
814
- # narrowed batch: appending to it would land the new rows after Phase 7.
814
+ # narrowed batch: appending to it would land the new rows after Phase 5.
815
815
  [ "${has_subs:-0}" -gt 0 ] && new_only=""
816
816
 
817
817
  if [ -n "$new_only" ]; then
@@ -1214,7 +1214,7 @@ GATE
1214
1214
  ;;
1215
1215
 
1216
1216
  plan)
1217
- # Phase 2's plan, turned into sub-phases of the phase that will execute it.
1217
+ # Phase 1's plan, turned into sub-phases of the phase that will execute it.
1218
1218
  #
1219
1219
  # The parsing lives here rather than in the phase document for one reason:
1220
1220
  # sub-phases are this file's structure, and a jq blob in a phase doc is a
@@ -1222,7 +1222,7 @@ GATE
1222
1222
  # keeps the doc to one line, which the aggregate phase-doc budget cares about.
1223
1223
  #
1224
1224
  # Reads planning-output.schema.json on stdin: tasks[] with id, title and an
1225
- # optional dependsOn[]. Status is `pending` for all of them - Phase 3 moves
1225
+ # optional dependsOn[]. Status is `pending` for all of them - Phase 2 moves
1226
1226
  # them with `sub`, and pre-marking work as started is the lie the tracker
1227
1227
  # exists to avoid.
1228
1228
  need_jq
@@ -4,7 +4,7 @@
4
4
  // Phase 4 answers "is what changed correct" and Step 1.45 answers "did every
5
5
  // planned TEST land". Nothing answered "did every planned TASK land". The
6
6
  // criteria manifest's denominator is rule IDs, not plan steps, and the only
7
- // place a step's status surfaced was render-work-summary.sh at Phase 7 - a
7
+ // place a step's status surfaced was render-work-summary.sh at Phase 5 - a
8
8
  // report, printed after the commit. So a plan with seven steps could ship five
9
9
  // and read as done.
10
10
  //
@@ -116,7 +116,7 @@ if (isMain) {
116
116
  "",
117
117
  "Fails when a plan step never reached a terminal status, when a skip or",
118
118
  "failure carries no reason, or when an analysis Section 14 `Add new` file",
119
- "is missing from the tree. Run before Phase 6 commit.",
119
+ "is missing from the tree. Run before Phase 4 commit.",
120
120
  "",
121
121
  ].join("\n"),
122
122
  );
@@ -148,12 +148,41 @@ scan_file() {
148
148
  FOUND=1
149
149
  fi
150
150
 
151
- # High-signal provider token prefixes (low false-positive rate)
151
+ # High-signal provider token prefixes (low false-positive rate).
152
+ #
153
+ # This set and the one in lib/outbound-gate.mjs cover the same providers, and
154
+ # smoke-secret-parity.sh holds them to it by running a fake token of each
155
+ # shape through BOTH.
152
156
  if echo "$content" | grep -qE '(ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{36}|github_pat_[A-Za-z0-9_]{60,}|xox[baprs]-[A-Za-z0-9-]{12,}|sk_live_[A-Za-z0-9]{20,}|rk_live_[A-Za-z0-9]{20,}|AIza[0-9A-Za-z_-]{35}|npm_[A-Za-z0-9]{36}|glpat-[A-Za-z0-9_-]{20,}'; then
153
157
  echo "BLOCKED: Provider access token in $file" >&2
154
158
  FOUND=1
155
159
  fi
156
160
 
161
+ # Model-provider and ML-hub keys. `sk-ant-` and `sk-proj-` are checked before
162
+ # the bare `sk-` form so the message names the provider a reader can revoke.
163
+ if echo "$content" | grep -qE 'sk-ant-[A-Za-z0-9_-]{32,}|sk-proj-[A-Za-z0-9_-]{32,}|pplx-[A-Za-z0-9]{32,}|hf_[A-Za-z0-9]{30,}'; then
164
+ echo "BLOCKED: Model-provider API key in $file" >&2
165
+ FOUND=1
166
+ fi
167
+
168
+ # Figma personal access token (figd_) and the MCP OAuth token (figu_). Same
169
+ # shape, same blast radius: both read every file the account can reach.
170
+ if echo "$content" | grep -qE 'fig[a-z]_[A-Za-z0-9_-]{20,}'; then
171
+ echo "BLOCKED: Figma token in $file" >&2
172
+ FOUND=1
173
+ fi
174
+
175
+ # A URL carrying its own credentials, which is how a `git remote -v` paste
176
+ # leaks, and an Authorization header copied out of a curl trace.
177
+ if echo "$content" | grep -qE '[a-z][a-z0-9+.-]*://[^[:space:]/:@]+:[^[:space:]/@]+@'; then
178
+ echo "BLOCKED: URL with embedded credentials in $file" >&2
179
+ FOUND=1
180
+ fi
181
+ if echo "$content" | grep -qiE 'Authorization:[[:space:]]*(Bearer|Basic)[[:space:]]+[A-Za-z0-9._~+/=-]{16,}'; then
182
+ echo "BLOCKED: Authorization header with a token in $file" >&2
183
+ FOUND=1
184
+ fi
185
+
157
186
  # JWT (three base64url segments - header.payload.signature)
158
187
  if echo "$content" | grep -qE 'eyJ[A-Za-z0-9_-]{10,}\.eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}'; then
159
188
  echo "BLOCKED: JWT in $file" >&2
@@ -7,7 +7,7 @@
7
7
  #
8
8
  # This is the agent-log counterpart to render-cost-summary.sh (which
9
9
  # targets PR/Jira channel bodies). The agent-log version is rendered
10
- # unconditionally on every Phase 7 run; the channels version is opt-in
10
+ # unconditionally on every Phase 5 run; the channels version is opt-in
11
11
  # via prefs.global.reportContent.costSummary.
12
12
  #
13
13
  # Usage:
@@ -2,7 +2,7 @@
2
2
  # render-work-summary.sh - v7.1.0
3
3
  #
4
4
  # Reads agent-state.json + phase-tracker.json + git diff, emits a concise
5
- # "### Work Summary" markdown block for Phase 7 channels dispatch. Intended
5
+ # "### Work Summary" markdown block for Phase 5 channels dispatch. Intended
6
6
  # as an executive-summary companion to the existing Normal Analysis /
7
7
  # Technical Details / Test Scenarios sections - gives Jira/Confluence
8
8
  # readers and PR reviewers a single-screen answer to "what actually
@@ -85,7 +85,7 @@ if [ -n "$WORKTREE" ]; then
85
85
  [ -f "$WORKTREE/triage-output.json" ] && TRIAGE_FILE="$WORKTREE/triage-output.json"
86
86
  fi
87
87
 
88
- # Fall back to the salvaged artefacts when the worktree is gone. Phase 6 removes it
88
+ # Fall back to the salvaged artefacts when the worktree is gone. Phase 4 removes it
89
89
  # once the PR is open (worktree-finalize), and this script used to resolve state
90
90
  # ONLY from the worktree - so it exited 2 and the whole Work Summary silently
91
91
  # disappeared from the PR body and the Jira comment. render-agent-log-cost.sh has
@@ -203,7 +203,7 @@ total_add=0; total_del=0
203
203
  # repository check. Testing -d "$WORKTREE/.git" would be wrong regardless: in a
204
204
  # linked worktree .git is a file, not a directory.
205
205
  # Prefer the worktree; fall back to the project root with the branch by name. A
206
- # finalized task (Phase 6 removed the worktree after the PR) has no worktree but
206
+ # finalized task (Phase 4 removed the worktree after the PR) has no worktree but
207
207
  # the branch is still local, so without this the Changed-files section vanished
208
208
  # from the PR body and the Jira comment even though the data was right there.
209
209
  DIFF_IN=""; DIFF_TIP=""
@@ -4,7 +4,7 @@
4
4
  * @file review-file-filter.mjs - decide what the reviewers are asked to read.
5
5
  *
6
6
  * Phase 4 has a size cap and no exclusion list. When the diff exceeds the
7
- * budget the cap truncates the LARGEST files first (`phase-4-review.md`, Step
7
+ * budget the cap truncates the LARGEST files first (`phase-3-review.md`, Step
8
8
  * 1.9), so a regenerated lockfile or a snapshot dump does not merely waste
9
9
  * tokens - it is the thing that survives while real code is cut. The cheapest
10
10
  * fix is to decide what is worth reading before the cap decides what fits.
@@ -39,6 +39,7 @@ import { join, dirname } from "node:path";
39
39
  import { fileURLToPath } from "node:url";
40
40
  import { costUsd } from "./_cost.mjs";
41
41
  import { resolveRunFile, canonicalRunDir } from "./_run-paths.mjs";
42
+ import { toCurrentPhase, CURRENT_SCHEMA } from "../lib/phase-schema.mjs";
42
43
 
43
44
  const __dirname = dirname(fileURLToPath(import.meta.url));
44
45
 
@@ -157,16 +158,22 @@ function computeCost(modelKey, tokensIn, tokensOut, costTable) {
157
158
  }
158
159
 
159
160
  /**
160
- * Heuristic: phase 1/2 use Opus (per personas + dev-mode docs); phase 3 uses
161
- * Sonnet by default unless the run is a Short pipeline (Opus). Phase 4 review
162
- * is multi-model - span events carry per-reviewer model when OTel is on.
161
+ * Heuristic: Plan (1) uses Opus per personas + dev-mode docs; Dev (2) uses
162
+ * Sonnet by default unless the run is a Short pipeline (Opus); Review (3)
163
+ * triage is Opus and the reviewer breakdown comes from spans.
163
164
  *
164
165
  * This is a fallback when phase-tracker doesn't record the model used. Live
165
166
  * runs that emit OTel spans get per-call model attribution from the spans.
167
+ *
168
+ * The phase number is normalised first. The old body compared against the
169
+ * literals "1", "2" and "4", which under the six-phase contract are Plan, Dev
170
+ * and Commit - so a pre-v19 row and a post-v19 row would have been priced
171
+ * against different phases with no error anywhere.
166
172
  */
167
- function inferModelForPhase(phaseId) {
168
- if (phaseId === "1" || phaseId === "2") return "opus";
169
- if (phaseId === "4") return "opus"; // triage is opus; reviewer breakdown comes from spans
173
+ function inferModelForPhase(phaseId, schema = CURRENT_SCHEMA) {
174
+ const p = toCurrentPhase(phaseId, schema);
175
+ if (p === 1) return "opus";
176
+ if (p === 3) return "opus";
170
177
  return "sonnet";
171
178
  }
172
179
 
@@ -24,7 +24,7 @@
24
24
  * consensusVerdict - unanimous-* / split / unverified
25
25
  * buildPassed - per-repo build outcome
26
26
  * diff.filesTouched / locAdded / - size of the change, from state.diffRisk
27
- * locRemoved / maxScore (Phase 4 Step 1.75); null when the
27
+ * locRemoved / maxScore (Phase 3 Step 1.75); null when the
28
28
  * run predates the persisted totals
29
29
  * reviewDelta.stillPresentFinal / - cross-round classification of the
30
30
  * resolvedTotal / tripped last iteration (state.reviewIterations[].delta)
@@ -42,6 +42,16 @@ import { listRuns, logsRoot, resolveRunDir, taskIdVariants } from "./_run-paths.
42
42
  import { runMain } from "../lib/fatal.mjs";
43
43
  import { invokedDirectly } from "../lib/invoked-directly.mjs";
44
44
 
45
+ /**
46
+ * The phase from which a run counts as "waiting on you", read from the phase
47
+ * contract rather than written here. This threshold moved once already (it was
48
+ * 6 under the eight-phase contract, it is 4 under six) and nothing connected it
49
+ * to the renumbering, so it would have silently regrouped every run.
50
+ */
51
+ const PHASE_WAITING_FROM = JSON.parse(
52
+ readFileSync(new URL("../schemas/phases.json", import.meta.url), "utf8"),
53
+ ).thresholds.waitingFromPhase;
54
+
45
55
  const GROUPS = {
46
56
  waiting: "Waiting on you",
47
57
  stopped: "Stopped mid-development",
@@ -166,7 +176,7 @@ function groupOf(state) {
166
176
  if (status === "awaiting_input" || status === "awaiting-user-test-main-checkout")
167
177
  return "waiting";
168
178
  if (prUrl) return "waiting";
169
- if (Number.isFinite(phase) && phase >= 6) return "waiting";
179
+ if (Number.isFinite(phase) && phase >= PHASE_WAITING_FROM) return "waiting";
170
180
  if (Number.isFinite(phase) && phase === 0) return "question";
171
181
  return "stopped";
172
182
  }
@@ -123,6 +123,14 @@ tree_grep() {
123
123
  return 0
124
124
  }
125
125
 
126
+ # Case-insensitive variant. Prose written to steer a model is written by hand,
127
+ # so its capitalisation is whatever the author felt like, and a case-sensitive
128
+ # pattern would miss "Ignore all previous instructions" by one letter.
129
+ tree_grep_i() {
130
+ tr '\n' '\0' < "$SCAN_LIST" | xargs -0 grep -nHEi -- "$1" 2>/dev/null
131
+ return 0
132
+ }
133
+
126
134
  # stdin: "path:line:content" grep hits -> stdout: "fileIdx|path|line|content"
127
135
  index_hits() {
128
136
  awk -v listfile="$SCAN_LIST" '
@@ -260,6 +268,24 @@ if [ "$THRESHOLD_RANK" -ge 1 ]; then
260
268
  seq=$((seq+1))
261
269
  emit_raw "$HIT_IDX" 9 "$seq" high "$HIT_FILE" "$HIT_LINE" "chmod-then-exec" "script made executable and immediately invoked"
262
270
  done < <(tree_grep 'chmod[[:space:]]+\+x[[:space:]]+[^&;]+[[:space:]]*(&&|;)[[:space:]]*\./' | index_hits)
271
+
272
+ # Prompt injection (OWASP LLM01). The other families ask what a skill makes
273
+ # the MACHINE do; this one asks what it makes the MODEL do. A skill is
274
+ # instructions loaded straight into the context that decides everything after
275
+ # it, so text telling the model to drop its instructions, recite its prompt,
276
+ # or act behind the user's back is an attack delivered as prose - and no
277
+ # pattern above can see it, because nothing is executed.
278
+ #
279
+ # `high`, not `critical`: these phrasings can appear in a skill that DESCRIBES
280
+ # the attack, this scanner's own documentation being the obvious case, so a
281
+ # hit is a line a human reads rather than a verdict.
282
+ seq=0
283
+ while IFS= read -r hit; do
284
+ [ -z "$hit" ] && continue
285
+ parse_hit "$hit"
286
+ seq=$((seq+1))
287
+ emit_raw "$HIT_IDX" 13 "$seq" high "$HIT_FILE" "$HIT_LINE" "prompt-injection" "text instructing the model to override, disclose or hide - OWASP LLM01"
288
+ done < <(tree_grep_i 'ignore (all )?(the )?(previous|prior|above|earlier) (instructions|prompts?|rules?)|disregard (your|the|any) (system prompt|previous instructions|instructions)|(reveal|print|output|repeat|show) (your|the) (system prompt|full instructions)|(do not|don'"'"'t|never) (tell|inform) the (user|operator)|without (telling|informing|asking) the (user|operator)|(always|automatically) (approve|confirm) [^.]{0,30}(without|regardless)|you are now (a|an|the) ' | index_hits)
263
289
  fi
264
290
 
265
291
  # --- medium families (rank 2) ---------------------------------------------
@@ -204,21 +204,21 @@ echo "→ reviewer-count contract (Claude=3, Copilot=3, Codex=3)"
204
204
  # a broken installation. _smoke-root.sh handles both layouts.
205
205
  # shellcheck source=pipeline/scripts/_smoke-root.sh
206
206
  . "$(dirname "${BASH_SOURCE[0]}")/_smoke-root.sh"
207
- P4="${MA_REFS:+$MA_REFS/phases/phase-4-review.md}"
207
+ P4="${MA_REFS:+$MA_REFS/phases/phase-3-review.md}"
208
208
  REVSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/reviewer-output.schema.json}"
209
209
  TRSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/triage-output.schema.json}"
210
210
 
211
211
  if [ -z "$P4" ] || [ ! -f "$P4" ]; then
212
- echo " ↷ SKIP: phase-4-review.md not present in this $MA_LAYOUT layout"
212
+ echo " ↷ SKIP: phase-3-review.md not present in this $MA_LAYOUT layout"
213
213
  # Two independent statements, because they can drift apart: the count sentence is
214
214
  # the contract, and the matrix is what a reader dispatches from. Three regexes over
215
215
  # overlapping prose used to stand in for this and let the count sentence 300 lines
216
216
  # further down go stale for a whole release without failing.
217
217
  elif grep -qF "Claude Code 3, Copilot CLI 3, Codex CLI 3" "$P4" \
218
218
  && grep -qE '^\| Reviewer 3 .*\|.*\|.*\|.*\|' "$P4"; then
219
- pass "phase-4-review declares Claude=3 / Copilot=3 / Codex=3 reviewers, and the matrix has all three host columns"
219
+ pass "phase-3-review declares Claude=3 / Copilot=3 / Codex=3 reviewers, and the matrix has all three host columns"
220
220
  else
221
- fail "phase-4-review does not declare the CLI-aware reviewer count for all three hosts"
221
+ fail "phase-3-review does not declare the CLI-aware reviewer count for all three hosts"
222
222
  fi
223
223
 
224
224
  # The two Codex constraints are silent-failure shaped, so the contract has to name
@@ -226,9 +226,9 @@ fi
226
226
  # 4-slot ceiling (orchestrator included) is why the count is 3 and not more.
227
227
  if [ -n "$P4" ] && [ -f "$P4" ]; then
228
228
  if grep -q 'fork_turns' "$P4" && grep -qiE "concurrency|slots" "$P4"; then
229
- pass "phase-4-review documents the fork_turns override rule + the concurrency ceiling"
229
+ pass "phase-3-review documents the fork_turns override rule + the concurrency ceiling"
230
230
  else
231
- fail "phase-4-review must document fork_turns and the Codex concurrency ceiling"
231
+ fail "phase-3-review must document fork_turns and the Codex concurrency ceiling"
232
232
  fi
233
233
  fi
234
234
 
@@ -24,6 +24,26 @@ TEMPLATE="$SMOKE_DIR/../preferences-template.json"
24
24
  [ -f "$TEMPLATE" ] || TEMPLATE="$HOME/multi-agent-pipeline/pipeline/preferences-template.json"
25
25
  LIVE_PREFS="$HOME/.claude/multi-agent-preferences.json"
26
26
 
27
+ # The migration target, derived the way migrate-prefs.mjs derives it: the last
28
+ # entry of the schema's schemaVersion enum, read from the schema that sits
29
+ # beside the migrator being checked. It used to be grepped as a
30
+ # `TARGET_VERSION = "x.y.z"` literal out of the migrator, and when that literal
31
+ # was replaced by the schema read - precisely because a literal had drifted a
32
+ # minor behind - the grep started matching nothing and both checks that depend
33
+ # on it reported "no reference point". A gate that reads the source of truth
34
+ # cannot go stale against it; a gate that reads a transcription of it can.
35
+ migration_target() {
36
+ local schema="$1/../schemas/prefs.schema.json"
37
+ [ -f "$schema" ] || schema="$PREFS_SCHEMA"
38
+ [ -f "$schema" ] || return 1
39
+ node -e "
40
+ const sv = JSON.parse(require('fs').readFileSync('$schema','utf8'))?.properties?.schemaVersion;
41
+ const v = sv?.const ?? sv?.enum?.at(-1);
42
+ if (!v) process.exit(1);
43
+ process.stdout.write(v);
44
+ " 2>/dev/null
45
+ }
46
+
27
47
  # ──────────────────────────────────────────────────────────────────────────
28
48
  echo "→ 1. Schema files parse as JSON"
29
49
  for f in "$PREFS_SCHEMA" "$STATE_SCHEMA"; do
@@ -145,10 +165,9 @@ if [ -f "$TEMPLATE" ]; then
145
165
  # every old entry in the migrator's accepted set became load-bearing purely to
146
166
  # rescue the template this gate was holding back. A template behind the target
147
167
  # is a defect, not the expected shape.
148
- TEMPLATE_TARGET=$(grep -oE 'TARGET_VERSION = "[0-9.]+"' "$SMOKE_DIR/migrate-prefs.mjs" 2>/dev/null \
149
- | grep -oE '[0-9]+\.[0-9]+\.[0-9]+')
168
+ TEMPLATE_TARGET=$(migration_target "$SMOKE_DIR")
150
169
  if [ -z "$TEMPLATE_TARGET" ]; then
151
- fail "cannot read TARGET_VERSION from migrate-prefs.mjs - the check has no reference point"
170
+ fail "cannot derive the migration target from prefs.schema.json - the check has no reference point"
152
171
  elif [ "$TVER" = "$TEMPLATE_TARGET" ]; then
153
172
  pass "template schemaVersion: $TVER (at the migration target)"
154
173
  else
@@ -201,17 +220,17 @@ if [ -f "$LIVE_PREFS" ]; then
201
220
  # correctly migrated to 2.4.0 hit the else arm and FAILED as "unknown". The
202
221
  # gate was rejecting the only fully-migrated state it exists to encourage.
203
222
  #
204
- # Reading the target from the migrator, and the accepted set from the schema
205
- # enum, means this can never disagree with them again.
223
+ # Reading the target and the accepted set from the same schema enum the
224
+ # migrator reads means this can never disagree with it again.
206
225
  MIGRATOR="$SMOKE_DIR/migrate-prefs.mjs"
207
- TARGET=$(grep -oE 'TARGET_VERSION = "[0-9.]+"' "$MIGRATOR" 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+')
226
+ TARGET=$(migration_target "$(dirname "$MIGRATOR")")
208
227
  KNOWN=$(node -e "
209
228
  const s = require('$PREFS_SCHEMA');
210
229
  process.stdout.write((s.properties.schemaVersion.enum || []).join(' '));
211
230
  " 2>/dev/null)
212
231
 
213
232
  if [ -z "$TARGET" ]; then
214
- fail "cannot read TARGET_VERSION from migrate-prefs.mjs - the check has no reference point"
233
+ fail "cannot derive the migration target from prefs.schema.json - the check has no reference point"
215
234
  elif [ "$LVER" = "$TARGET" ]; then
216
235
  pass "live prefs at the migration target (v$TARGET)"
217
236
  elif [ "$LVER" = "none" ]; then
@@ -21,6 +21,7 @@
21
21
  import fs from "node:fs";
22
22
  import os from "node:os";
23
23
  import path from "node:path";
24
+ import { rowPhase } from "../lib/phase-schema.mjs";
24
25
 
25
26
  const args = parseArgs(process.argv.slice(2));
26
27
  const METRICS =
@@ -64,7 +65,12 @@ for (const e of events) {
64
65
  models: new Set(),
65
66
  });
66
67
  const t = byTask.get(e.task_id);
67
- t.phases.add(e.phase);
68
+ // Normalised to the current vocabulary before it goes in the set. This log
69
+ // is append-only across the v19.0.0 renumbering, so the raw field carries
70
+ // two different meanings and a set of raw values cannot be compared to
71
+ // anything.
72
+ const cp = rowPhase(e);
73
+ if (cp !== null) t.phases.add(cp);
68
74
  const d = e.details || {};
69
75
  if (typeof d.tokens_in === "number") t.tokens_in += d.tokens_in;
70
76
  if (typeof d.tokens_out === "number") t.tokens_out += d.tokens_out;
@@ -113,7 +119,12 @@ for (const [taskId, t] of byTask) {
113
119
  summary.totalTokensOut += t.tokens_out;
114
120
  summary.totalUsd += usd;
115
121
  const total = t.tokens_in + t.tokens_out;
116
- const isMulti = t.phases.has("4") && t.premium_calls > 5; // heuristic
122
+ // Heuristic: a task that reached Review with many premium calls is a
123
+ // multi-repo run. Keyed on Review (3), not on the literal "4" this line used
124
+ // to carry - under the six-phase contract 4 is Commit, and every run reaches
125
+ // Commit, so the old literal would have classified nearly everything as
126
+ // multi-repo and applied the wrong ceiling.
127
+ const isMulti = t.phases.has(3) && t.premium_calls > 5;
117
128
  const tokenLimit = isMulti ? limits.perTaskTokensMultiRepo : limits.perTaskTokens;
118
129
  const usdLimit = isMulti ? limits.perTaskUsdMultiRepo : limits.perTaskUsd;
119
130
  if (total > tokenLimit)
@@ -3,8 +3,8 @@
3
3
  /**
4
4
  * @file triage-memory.mjs - v8.3.0
5
5
  *
6
- * Lightweight, zero-dep persistence layer for past Phase 4 triage findings.
7
- * Used by Phase 7 to ingest results, by Phase 1/4 to look up prior art when
6
+ * Lightweight, zero-dep persistence layer for past Phase 3 triage findings.
7
+ * Used by Phase 5 to ingest results, by Phase 1/4 to look up prior art when
8
8
  * a new task overlaps with a previous one, and by /multi-agent:search.
9
9
  *
10
10
  * Storage: append-only JSONL, one file per repo: