@mmerterden/multi-agent-pipeline 19.1.3 → 20.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/CHANGELOG.md +150 -7
  2. package/README.md +60 -50
  3. package/README.tr.md +55 -46
  4. package/docs/adr/0002-instruction-driven-flag.md +6 -5
  5. package/docs/adr/0005-lazy-phase-docs.md +2 -2
  6. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  7. package/docs/adr/0009-claude-stack-skills-plugin-only.md +1 -1
  8. package/docs/adr/0010-own-code-graph.md +5 -4
  9. package/docs/adr/0012-macos-only.md +2 -2
  10. package/docs/adr/0013-lsp-code-intelligence.md +2 -2
  11. package/docs/adr/0014-six-phase-consolidation.md +9 -9
  12. package/docs/adr/0015-one-pipeline-no-depth-answer.md +83 -0
  13. package/docs/adr/0016-the-run-shape-is-asked-not-typed.md +69 -0
  14. package/docs/adr/README.md +18 -16
  15. package/docs/architecture.md +2 -2
  16. package/docs/ecosystem.md +8 -9
  17. package/docs/facts.json +5 -8
  18. package/docs/features.md +4 -5
  19. package/docs/token-budget-history.md +1 -1
  20. package/install/_common.mjs +14 -6
  21. package/install/_mcp-register.mjs +1 -1
  22. package/install/_plugin-skills.mjs +3 -4
  23. package/install/copilot.mjs +5 -5
  24. package/install/templates/copilot-instructions.md +7 -16
  25. package/manifest.json +135 -138
  26. package/package.json +1 -1
  27. package/pipeline/commands/multi-agent/SKILL.md +6 -8
  28. package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -0
  29. package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +2 -0
  30. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -0
  31. package/pipeline/commands/multi-agent/autopilot/SKILL.md +2 -0
  32. package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +2 -0
  33. package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +1 -1
  34. package/pipeline/commands/multi-agent/build-optimize/SKILL.md +2 -0
  35. package/pipeline/commands/multi-agent/channels/SKILL.md +1 -1
  36. package/pipeline/commands/multi-agent/create-jira/SKILL.md +2 -0
  37. package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
  38. package/pipeline/commands/multi-agent/forget/SKILL.md +2 -0
  39. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +4 -2
  40. package/pipeline/commands/multi-agent/help/SKILL.md +21 -27
  41. package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +5 -4
  42. package/pipeline/commands/multi-agent/issue/SKILL.md +2 -0
  43. package/pipeline/commands/multi-agent/jira/SKILL.md +2 -0
  44. package/pipeline/commands/multi-agent/language/SKILL.md +2 -0
  45. package/pipeline/commands/multi-agent/prune-logs/SKILL.md +2 -0
  46. package/pipeline/commands/multi-agent/purge/SKILL.md +2 -0
  47. package/pipeline/commands/multi-agent/resume/SKILL.md +177 -48
  48. package/pipeline/commands/multi-agent/save/SKILL.md +2 -0
  49. package/pipeline/commands/multi-agent/setup/SKILL.md +4 -4
  50. package/pipeline/commands/multi-agent/stack/SKILL.md +2 -0
  51. package/pipeline/commands/multi-agent/sync/SKILL.md +6 -7
  52. package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +2 -0
  53. package/pipeline/commands/multi-agent/uninstall/SKILL.md +2 -0
  54. package/pipeline/commands/sim-test.md +4 -4
  55. package/pipeline/lib/repo-hygiene.sh +1 -1
  56. package/pipeline/multi-agent-refs/analysis/locked.md +2 -2
  57. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  58. package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
  59. package/pipeline/multi-agent-refs/analysis/synthesis.md +1 -1
  60. package/pipeline/multi-agent-refs/analysis-template.md +1 -1
  61. package/pipeline/multi-agent-refs/channels/jira.md +8 -8
  62. package/pipeline/multi-agent-refs/component-dispatch.md +0 -8
  63. package/pipeline/multi-agent-refs/cross-cli-contract.md +10 -11
  64. package/pipeline/multi-agent-refs/features/base-branch-evidence.md +2 -2
  65. package/pipeline/multi-agent-refs/features/external-context-injection.md +2 -0
  66. package/pipeline/multi-agent-refs/features/review-delta.md +1 -1
  67. package/pipeline/multi-agent-refs/features/review-multi-repo.md +3 -3
  68. package/pipeline/multi-agent-refs/features/scope-check.md +1 -1
  69. package/pipeline/multi-agent-refs/features/skill-conformance.md +1 -1
  70. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
  71. package/pipeline/multi-agent-refs/features/visual-evidence.md +2 -1
  72. package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
  73. package/pipeline/multi-agent-refs/generate-issue.md +2 -0
  74. package/pipeline/multi-agent-refs/issue-jira-triad.md +2 -0
  75. package/pipeline/multi-agent-refs/keychain.md +2 -0
  76. package/pipeline/multi-agent-refs/knowledge.md +0 -7
  77. package/pipeline/multi-agent-refs/outside-the-pipeline.md +1 -1
  78. package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
  79. package/pipeline/multi-agent-refs/phases/modes.md +32 -108
  80. package/pipeline/multi-agent-refs/phases/operations.md +3 -1
  81. package/pipeline/multi-agent-refs/phases/phase-0-init.md +23 -42
  82. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +10 -21
  83. package/pipeline/multi-agent-refs/phases/phase-2-dev.md +13 -44
  84. package/pipeline/multi-agent-refs/phases/phase-3-review.md +19 -22
  85. package/pipeline/multi-agent-refs/phases/phase-4-commit.md +6 -6
  86. package/pipeline/multi-agent-refs/phases/phase-5-report.md +2 -2
  87. package/pipeline/multi-agent-refs/phases.md +9 -11
  88. package/pipeline/multi-agent-refs/progress-contract.md +1 -1
  89. package/pipeline/multi-agent-refs/readiness-review.md +2 -0
  90. package/pipeline/multi-agent-refs/rules.md +2 -2
  91. package/pipeline/multi-agent-refs/tracker-contract.md +9 -40
  92. package/pipeline/multi-agent-refs/wiki-capture.md +3 -2
  93. package/pipeline/preferences-template.json +2 -2
  94. package/pipeline/rules/figma-pipeline.md +1 -1
  95. package/pipeline/schemas/agent-state.schema.json +5 -10
  96. package/pipeline/schemas/migrations/prefs-2.7.0-to-2.8.0.mjs +33 -0
  97. package/pipeline/schemas/phases.json +3 -24
  98. package/pipeline/schemas/prefs.schema.json +5 -5
  99. package/pipeline/scripts/autopilot-runner.mjs +6 -7
  100. package/pipeline/scripts/build-references.mjs +3 -3
  101. package/pipeline/scripts/build-stack-plugins.mjs +1 -1
  102. package/pipeline/scripts/bulk-read.sh +6 -4
  103. package/pipeline/scripts/cost-table.json +1 -1
  104. package/pipeline/scripts/doctor.mjs +4 -4
  105. package/pipeline/scripts/gc-refs.sh +1 -1
  106. package/pipeline/scripts/gen-mode-dispatch.mjs +11 -41
  107. package/pipeline/scripts/learnings-ledger.mjs +1 -1
  108. package/pipeline/scripts/match-skills.mjs +4 -4
  109. package/pipeline/scripts/memory-load.sh +3 -3
  110. package/pipeline/scripts/migrate-prefs.mjs +18 -17
  111. package/pipeline/scripts/phase-tracker.sh +2 -2
  112. package/pipeline/scripts/phase0-exit-gate.mjs +1 -1
  113. package/pipeline/scripts/plan-coverage-gate.mjs +6 -6
  114. package/pipeline/scripts/run-aggregator.mjs +3 -3
  115. package/pipeline/scripts/runs-index.mjs +7 -7
  116. package/pipeline/scripts/scope-check-gate.mjs +1 -1
  117. package/pipeline/scripts/smoke-schema-validation.sh +9 -12
  118. package/pipeline/scripts/usage-report.mjs +5 -7
  119. package/pipeline/scripts/validate-analysis-doc.mjs +3 -3
  120. package/pipeline/scripts/worktree-finalize.sh +2 -2
  121. package/pipeline/scripts/write-state.mjs +22 -11
  122. package/pipeline/skills/.skill-manifest.json +9 -21
  123. package/pipeline/skills/.skills-index.json +6 -39
  124. package/pipeline/skills/shared/README.md +5 -8
  125. package/pipeline/skills/shared/core/multi-agent/SKILL.md +10 -13
  126. package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +1 -1
  127. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +4 -5
  128. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +13 -16
  129. package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +2 -3
  130. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +51 -15
  131. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -6
  132. package/pipeline/skills/skills-index.md +3 -6
  133. package/pipeline/commands/multi-agent/local/SKILL.md +0 -132
  134. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +0 -142
  135. package/pipeline/commands/multi-agent/resume-local/SKILL.md +0 -114
  136. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +0 -41
  137. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +0 -55
  138. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +0 -51
@@ -66,7 +66,7 @@
66
66
  },
67
67
  "worktreePath": {
68
68
  "type": ["string", "null"],
69
- "description": "Absolute path to the task worktree. Null in --local mode."
69
+ "description": "Absolute path to the task worktree. Null when the Step 5b picker answered local."
70
70
  },
71
71
  "branch": {
72
72
  "type": "string",
@@ -274,7 +274,7 @@
274
274
  "workspaceSource": {
275
275
  "type": "string",
276
276
  "enum": ["asked", "command", "autopilot"],
277
- "description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = :local / --local / :local-autopilot stated it up front, or a flow that only ever builds worktrees; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
277
+ "description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = a flow that only ever builds worktrees stated it up front; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
278
278
  },
279
279
  "remoteType": {
280
280
  "type": "string",
@@ -476,15 +476,10 @@
476
476
  "type": "boolean",
477
477
  "default": false
478
478
  },
479
- "onlyDevelop": {
480
- "type": "boolean",
481
- "default": false,
482
- "description": "Short pipeline - phases 1 and 2 are skipped. Set by the Phase 0 Step 7.5 depth picker as of v16.0.0; the key and every reader of it are unchanged. Phase 3 Review still runs (v14.0.0+). Always false in an autopilot run, which never asks the depth question."
483
- },
484
479
  "localMode": {
485
480
  "type": "boolean",
486
481
  "default": false,
487
- "description": "--local flag - no worktree, direct branch in projectRoot."
482
+ "description": "Step 5b answered local - no worktree, direct branch in projectRoot."
488
483
  },
489
484
  "instructionDriven": {
490
485
  "type": "boolean",
@@ -498,7 +493,7 @@
498
493
  "analysis": {
499
494
  "type": ["object", "null"],
500
495
  "additionalProperties": true,
501
- "description": "Phase 1 analysis-document outcome. Absent in Short runs, which produce no document by design.",
496
+ "description": "Phase 1 analysis-document outcome. Phase 1 runs in every mode, so a document is always produced; what varies is how many of its sections the evidence supported.",
502
497
  "properties": {
503
498
  "docStatus": {
504
499
  "type": "string",
@@ -1530,7 +1525,7 @@
1530
1525
  }
1531
1526
  }
1532
1527
  },
1533
- "description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot and --local entry."
1528
+ "description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot entry and by any run whose workspace is local."
1534
1529
  },
1535
1530
  "host": {
1536
1531
  "type": ["string", "null"],
@@ -0,0 +1,33 @@
1
+ /**
2
+ * prefs-2.7.0-to-2.8.0.mjs - preferences migration
3
+ *
4
+ * v20.0.0 folds `/multi-agent:resume-local` into `/multi-agent:resume`. One
5
+ * command now asks which unfinished work to pick up: a run that stopped
6
+ * mid-phase, or a branch carrying work no run ever produced. The preference
7
+ * that decides whether triage-accepted findings are auto-fixed applies to the
8
+ * second path, so it moves with it: `global.resumeLocal` -> `global.resume`.
9
+ *
10
+ * The value is carried, never reset. Someone who turned auto-fix on for the
11
+ * tail meant it for exactly the work this path still handles.
12
+ *
13
+ * @param {object} data - parsed multi-agent-preferences.json
14
+ * @returns {object} - migrated data with schemaVersion "2.8.0"
15
+ */
16
+ export default function migrate(data) {
17
+ const out = JSON.parse(JSON.stringify(data));
18
+
19
+ out.schemaVersion = "2.8.0";
20
+
21
+ if (!out.global || typeof out.global !== "object") return out;
22
+
23
+ const legacy = out.global.resumeLocal;
24
+ if (!out.global.resume || typeof out.global.resume !== "object") {
25
+ out.global.resume = {};
26
+ }
27
+ if (typeof out.global.resume.autoFix !== "boolean") {
28
+ out.global.resume.autoFix = typeof legacy?.autoFix === "boolean" ? legacy.autoFix : false;
29
+ }
30
+ delete out.global.resumeLocal;
31
+
32
+ return out;
33
+ }
@@ -6,7 +6,6 @@
6
6
  "phaseSchema": 2,
7
7
  "phaseSchemaNote": "The phase vocabulary generation. metrics.jsonl is append-only and v19.0.0 renumbered the phases, so every line written from v19.0.0 on carries this number and a line without the field is generation 1. Aggregators pick their name table from it; without it phase 3 means Dev in old rows and Review in new ones and no reader can tell them apart.",
8
8
  "legacyPhaseNames": {
9
- "note": "Generation 1 phase names, kept so an aggregator can label a historical metrics.jsonl row correctly instead of printing the current name for a number that meant something else. Read-only history: nothing emits these any more. The generation 1 -> 2 number map is not repeated here, it is the `was` array on each phase above.",
10
9
  "0": "Init",
11
10
  "1": "Analysis",
12
11
  "2": "Planning",
@@ -14,7 +13,8 @@
14
13
  "4": "Review",
15
14
  "5": "Test",
16
15
  "6": "Commit",
17
- "7": "Report"
16
+ "7": "Report",
17
+ "note": "Generation 1 phase names, kept so an aggregator can label a historical metrics.jsonl row correctly instead of printing the current name for a number that meant something else. Read-only history: nothing emits these any more. The generation 1 -> 2 number map is not repeated here, it is the `was` array on each phase above."
18
18
  },
19
19
  "phases": [
20
20
  {
@@ -64,35 +64,14 @@
64
64
  "modes": {
65
65
  "full": {
66
66
  "phases": [0, 1, 2, 3, 4, 5],
67
- "local": false,
68
- "autopilot": false,
69
- "depth": true
67
+ "autopilot": false
70
68
  },
71
69
  "autopilot": {
72
70
  "phases": [0, 1, 2, 3, 4, 5],
73
- "local": false,
74
- "autopilot": true
75
- },
76
- "local": {
77
- "phases": [0, 1, 2, 3, 4, 5],
78
- "local": true,
79
- "autopilot": false,
80
- "depth": true
81
- },
82
- "local-autopilot": {
83
- "phases": [0, 1, 2, 3, 4, 5],
84
- "local": true,
85
71
  "autopilot": true
86
72
  },
87
- "full-local": {
88
- "phases": [0, 1, 2, 3, 4, 5],
89
- "local": true,
90
- "autopilot": false,
91
- "depth": true
92
- },
93
73
  "analysis": {
94
74
  "phases": [0, 1, 3, 4, 5],
95
- "local": false,
96
75
  "autopilot": false
97
76
  }
98
77
  },
@@ -14,8 +14,8 @@
14
14
  "properties": {
15
15
  "schemaVersion": {
16
16
  "type": "string",
17
- "enum": ["2.0.0", "2.1.0", "2.2.0", "2.3.0", "2.4.0", "2.5.0", "2.6.0", "2.7.0"],
18
- "description": "v2.0.0: pre-v3.7. v2.1.0: v3.7+ adds identities[].servicePatMap, platformIdentityRouting, recentGroups, recentBranches, serviceStatus, settings, expanded keychainMapping. v2.2.0: v6.0.0 formalizes v5.7 / v5.8 additions (reportChannels, reportContent with technicalAnalysis, wikiScope, autopilotReportTimeoutSeconds) that had been running as 2.1.0 sub-migrations without a proper version bump. v2.5.0: v14.0.0 adds global.skillConformance (Phase 4 criteria resolution) and declares global.ship.autoFix, which the tail command's spec had referenced as global.finish.autoFix without ever declaring it. v2.6.0: v15.0.0 renames global.ship to global.resumeLocal (/multi-agent:ship -> :resume-local). v2.7.0: v19.0.0 drops the analysis Lite mode (global.analysisPhase.mode keeps a single value 'full') and adds global.modelRouting, which ships disabled."
17
+ "enum": ["2.0.0", "2.1.0", "2.2.0", "2.3.0", "2.4.0", "2.5.0", "2.6.0", "2.7.0", "2.8.0"],
18
+ "description": "v2.0.0: pre-v3.7. v2.1.0: v3.7+ adds identities[].servicePatMap, platformIdentityRouting, recentGroups, recentBranches, serviceStatus, settings, expanded keychainMapping. v2.2.0: v6.0.0 formalizes v5.7 / v5.8 additions (reportChannels, reportContent with technicalAnalysis, wikiScope, autopilotReportTimeoutSeconds) that had been running as 2.1.0 sub-migrations without a proper version bump. v2.5.0: v14.0.0 adds global.skillConformance (Phase 4 criteria resolution) and declares global.ship.autoFix, which the tail command's spec had referenced as global.finish.autoFix without ever declaring it. v2.6.0: v15.0.0 renames global.ship to global.resumeLocal (/multi-agent:ship -> :resume-local). v2.7.0: v19.0.0 drops the analysis Lite mode (global.analysisPhase.mode keeps a single value 'full') and adds global.modelRouting, which ships disabled. v2.8.0: v20.0.0 folds :resume-local into :resume and renames global.resumeLocal to global.resume."
19
19
  },
20
20
  "global": {
21
21
  "type": "object",
@@ -1923,15 +1923,15 @@
1923
1923
  }
1924
1924
  }
1925
1925
  },
1926
- "resumeLocal": {
1926
+ "resume": {
1927
1927
  "type": "object",
1928
1928
  "additionalProperties": false,
1929
- "description": "/multi-agent:resume-local (formerly :ship, :finish) - the tail that runs review + build/test + PR + report over work already on the branch.",
1929
+ "description": "/multi-agent:resume - picking up unfinished work, either a run that stopped mid-phase or a branch carrying work no run produced. These keys apply to the second path, which runs review + build/test + PR + report over the branch diff.",
1930
1930
  "properties": {
1931
1931
  "autoFix": {
1932
1932
  "type": "boolean",
1933
1933
  "default": false,
1934
- "description": "When true, resume-local auto-fixes triage-accepted blocking/important findings and re-reviews instead of asking. Equivalent to passing `autopilot` on every run. Default false: resume-local operates on work the user wrote by hand, so silently rewriting it is the surprising option."
1934
+ "description": "When true, the tail path auto-fixes triage-accepted blocking/important findings and re-reviews instead of asking. Equivalent to passing `autopilot` on every run. Default false: this path operates on work the user wrote by hand, so silently rewriting it is the surprising option."
1935
1935
  }
1936
1936
  }
1937
1937
  },
@@ -587,13 +587,12 @@ function main() {
587
587
  const attempts = readAttempted();
588
588
  const breaker = consecutiveFailures(attempts);
589
589
  if (BREAKER_LIMIT > 0 && breaker.count >= BREAKER_LIMIT) {
590
- // HALF-OPEN, not latched. The first version of this returned here on every
591
- // tick, and the only writer of attempted.jsonl is downstream of this
592
- // return - so once it opened, no new attempt could ever be recorded, the
593
- // consecutive count could never fall, and the runner was stopped for good.
594
- // The log line said "until an attempt succeeds" and the feature doc said
595
- // "one successful attempt clears it": both described a state the code made
596
- // unreachable.
590
+ // HALF-OPEN, not latched. The only writer of attempted.jsonl is downstream
591
+ // of this return, so returning on every tick would mean that once the
592
+ // breaker opened no new attempt could be recorded, the consecutive count
593
+ // could never fall, and the runner would be stopped for good - while the
594
+ // log line promises "until an attempt succeeds" and the feature doc
595
+ // promises "one successful attempt clears it".
597
596
  //
598
597
  // A breaker that cannot re-close is not a breaker, it is an off switch with
599
598
  // a misleading label. So after a cooldown one probe is let through: if the
@@ -2,9 +2,9 @@
2
2
  // build-references.mjs - deterministic Section 21 References table for an
3
3
  // /multi-agent:analysis document (Locked 33).
4
4
  //
5
- // The references table used to be prose the model filled in. That lists what an
6
- // author remembers consulting, which is a different set from what the run
7
- // actually fetched: sources that failed to load vanish silently, scope decisions
5
+ // The references table is built, not written as prose the model fills in. Prose
6
+ // lists what an author remembers consulting, which is a different set from what
7
+ // the run actually fetched: sources that failed to load vanish silently, scope decisions
8
8
  // made in conversation never appear, and a Figma link with no node id points at
9
9
  // whatever the file looks like today. This turns the table into a projection of
10
10
  // `state.analysisSpec.evidence.*` so the two cannot disagree.
@@ -72,7 +72,7 @@ const PLUGINS_REPO = getArg("--plugins-repo", join(HOME, "multi-agent-plugins"))
72
72
  // nowhere else: on a CI runner the checkout is under the workspace directory,
73
73
  // so --check-routing reported "source not found" and three tests failed on
74
74
  // every clean host while passing locally. That is the whole class of bug CI
75
- // exists to catch, and it survived because CI was asleep.
75
+ // exists to catch, which is why this path is not left to CI alone.
76
76
  const PIPE_ROOT = getArg("--pipeline", join(dirname(fileURLToPath(import.meta.url)), "..", ".."));
77
77
  const EXTERNAL = join(PIPE_ROOT, "pipeline/skills/shared/external");
78
78
  const DRY = args.includes("--dry-run");
@@ -91,10 +91,12 @@ if command -v jq >/dev/null 2>&1; then
91
91
  fi
92
92
  [ -n "$MODEL" ] || MODEL="${PREF_MODEL:-haiku}"
93
93
 
94
- # bulk-read is one of the two call sites the pipeline makes itself, so routing
95
- # may send it outside the Anthropic ladder - unlike a subagent, nothing here
96
- # belongs to the host. An explicit --model still wins: a caller that named a
97
- # rung asked for that rung.
94
+ # Routing picks the rung INSIDE the Anthropic ladder. The worker below is the
95
+ # `claude` CLI, which speaks that ladder and nothing else, so a rung on another
96
+ # provider is refused at dispatch rather than returned and failed. Sending this
97
+ # read to a third-party provider would put the file's full text outside the
98
+ # account that owns it, and that is a decision a user makes, not a default.
99
+ # An explicit --model still wins: a caller that named a rung asked for that rung.
98
100
  if [ -z "${MODEL_EXPLICIT:-}" ] && [ -x "$HERE/../lib/model-dispatch.sh" ]; then
99
101
  MODEL=$("$HERE/../lib/model-dispatch.sh" bulk-read \
100
102
  --phase "$PHASE" --default "$MODEL" 2>/dev/null) || MODEL="${PREF_MODEL:-haiku}"
@@ -16,7 +16,7 @@
16
16
  "cacheReadPerMtok": 0.5,
17
17
  "modelId": "claude-opus-5",
18
18
  "provider": "anthropic",
19
- "note": "Second tier - dev phase on a Short run, Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
19
+ "note": "Second tier - Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
20
20
  },
21
21
  "sonnet": {
22
22
  "inPerMtok": 3.0,
@@ -1051,10 +1051,10 @@ function main() {
1051
1051
  process.exitCode = code;
1052
1052
  }
1053
1053
 
1054
- // Exported so the gate can DRIVE the engine instead of reading it. The BLOCK
1055
- // downgrade used to be asserted with a regex over this file's own source, which
1056
- // a comment carrying the same two substrings satisfied just as well as the code
1057
- // did - the enforcement could be deleted outright and the gate stayed green.
1054
+ // Exported so the gate can DRIVE the engine instead of reading it. Asserting the
1055
+ // BLOCK downgrade with a regex over this file's own source proves nothing: a
1056
+ // comment carrying the same two substrings satisfies it as well as the code does,
1057
+ // so the enforcement could be deleted outright and the gate would stay green.
1058
1058
  export { report, results as __results, MAY_BLOCK as __mayBlock };
1059
1059
 
1060
1060
  // realpath on both sides: `import.meta.url` is already resolved, while argv[1]
@@ -4,7 +4,7 @@
4
4
  # `offload-ref.sh` parks build logs, diffs and test output under
5
5
  # <root>/.multi-agent/refs/<node_id>.md so a phase prompt can carry a pointer
6
6
  # instead of the whole log. In worktree modes that directory dies with the
7
- # worktree. In the --local modes there is no worktree: the refs land in the
7
+ # worktree. With a local workspace there is no worktree: the refs land in the
8
8
  # real checkout and nothing ever removes them. They are gitignored, so they are
9
9
  # invisible to `git status` and grow without bound.
10
10
  #
@@ -16,12 +16,9 @@ import { readFileSync } from "node:fs";
16
16
  * node pipeline/scripts/gen-mode-dispatch.mjs --mode=autopilot # 0..5
17
17
  * node pipeline/scripts/gen-mode-dispatch.mjs --mode=analysis # 0/1/3/4/5 (no Dev)
18
18
  *
19
- * v16.0.0 removed the four dev-* modes. Depth is no longer a command name: the
20
- * Phase 0 Step 7.5 picker asks Full or Short and sets `state.onlyDevelop`. That
21
- * answer arrives long after the tracker boots at Step -1, so v17.5.0 splits
22
- * registration for the two modes that ask it (`full`, `local`): Phase 0 at Step -1,
23
- * the rest once depth has named them. Everything else still registers its whole
24
- * set up front, and a generated phase set is per-COMMAND, never per-depth.
19
+ * There is one pipeline. A mode's phase set is a property of the COMMAND and is
20
+ * known before the tracker boots, so every mode registers its whole set at
21
+ * Step -1 and no registration is deferred.
25
22
  *
26
23
  * Companion smoke `smoke-mode-dispatch-drift.sh` regenerates the section for
27
24
  * each mode file, diffs against the on-disk content, and fails on drift.
@@ -63,9 +60,7 @@ const MODES = Object.fromEntries(
63
60
  name,
64
61
  {
65
62
  phases: modePhases(name),
66
- local: m.local,
67
63
  autopilot: m.autopilot,
68
- ...(m.depth ? { depth: true } : {}),
69
64
  },
70
65
  ]),
71
66
  );
@@ -86,9 +81,6 @@ const phaseSequence = activeIds.map((id) => `Phase ${id}`).join(" → ");
86
81
  const MODE_LABELS = {
87
82
  autopilot: "`autopilot`",
88
83
  full: "full-pipeline",
89
- local: "`--local`",
90
- "local-autopilot": "`--local autopilot`",
91
- "full-local": "full-pipeline + `--local`",
92
84
  };
93
85
  const modeLabel = MODE_LABELS[MODE] ?? `\`${MODE}\``;
94
86
 
@@ -98,46 +90,24 @@ const skipNote =
98
90
  ? `${modeLabel} mode does NOT TaskCreate phases ${skippedIds.join("/")} - those are not part of the ${modeLabel} phase set (\`${spec.phases.join(" ")}\`). Only register tiles for the active set.`
99
91
  : `${modeLabel} mode TaskCreates all ${PHASES.length} phases (no phase is skipped).`;
100
92
 
101
- const orderingNote = `**All TaskCreate calls in a batch fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${spec.depth ? `Phase 0 at Step -1, then the rest in ascending order at Step 7.5 (${phaseSequence} minus whatever the depth answer drops)` : phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
102
-
103
- const localCaveat = spec.local
104
- ? `\n> **Local mode:** no worktree is created, work happens on the current branch. Phase 0 Init still calls \`init\` - the \`--local\` flag is stored in tracker-state.json, and \`:resume\` restores the correct CWD.\n`
105
- : "";
93
+ const orderingNote = `**All TaskCreate calls in a batch fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
106
94
 
107
95
  const banner = spec.autopilot
108
96
  ? `\n> **Autopilot mode:** user confirmations are skipped. The tracker is still mandatory - autopilot agent calls cannot skip it; skipping breaks \`smoke-tracker-contract.sh\`.\n`
109
97
  : "";
110
98
 
111
- // Registration shape. A mode that asks the depth question does not know its phase
112
- // set at Step -1 (the answer needs taskType, which needs the fetched issue and the
113
- // branch), so it registers Phase 0 there and the rest at Step 7.5. Drawing eight
114
- // tiles beside the question that decides whether two of them run is the failure this
115
- // split exists to remove.
116
- const fullSet = spec.phases.filter((p) => p !== "0:Init");
117
- const shortSet = fullSet.filter((p) => Number(p.split(":")[0]) >= 2);
99
+ // Registration shape. Every mode knows its whole phase set at Step -1, so every
100
+ // tile is created there and nothing is appended later.
118
101
  const loopFor = (list) =>
119
102
  `for p in ${list.map((x) => `"${x}"`).join(" ")}; do\n bash $HOME/.claude/scripts/phase-tracker.sh add "\${p%%:*}" "\${p#*:}"\ndone`;
120
103
 
121
- const initLoop = spec.depth
122
- ? 'bash $HOME/.claude/scripts/phase-tracker.sh add 0 "Init"\nbash $HOME/.claude/scripts/phase-tracker.sh tiles'
123
- : `${loopFor(spec.phases)}`;
124
-
125
- const deferredBlock = spec.depth
126
- ? `
127
- # Phase 0 Step 7.5, immediately after the depth answer - the first moment this
128
- # mode knows its phase set. Full:
129
- ${loopFor(fullSet)}
130
- # Short (Analysis and Planning are not run, so they get no tile at all):
131
- ${loopFor(shortSet)}
132
- # Then the widget, narrowed to the phases that do not have a tile yet:
133
- bash $HOME/.claude/scripts/phase-tracker.sh tiles --new
134
- `
135
- : "";
104
+ const initLoop = loopFor(spec.phases);
105
+ const deferredBlock = "";
136
106
 
137
107
  const out = `## Required: Phase Tracker Contract
138
108
 
139
109
  **The phase tracker is mandatory** - the agent cannot skip it. Full spec: [\`$HOME/.claude/multi-agent-refs/tracker-contract.md\`]($HOME/.claude/multi-agent-refs/tracker-contract.md).
140
- ${banner}${localCaveat}
110
+ ${banner}
141
111
  Two channels run in parallel at every phase boundary:
142
112
 
143
113
  1. **State channel** (every CLI, identical): \`phase-tracker.sh\` writes to \`tracker-state.json\`. Drives \`:resume\`, \`:log\`, \`:status\`.
@@ -160,11 +130,11 @@ bash $HOME/.claude/scripts/phase-tracker.sh tokens <N> <in> <out> [cached]
160
130
 
161
131
  In Claude Code the agent MUST also drive the native TaskList widget so the user sees a sticky phase tile stack - this is the only progress signal Claude Code surfaces. Skipping these calls is the #1 source of "I don't see any phases" complaints.
162
132
 
163
- **TaskCreate ordering (strict)**: All TaskCreate calls in a registration batch fire in strict phase-number order BEFORE any TaskUpdate in that batch, and a later batch only ever appends phases numbered above everything already registered.${spec.depth ? " This mode registers in two batches (Step -1, then Step 7.5), so `tiles --new` narrows the second one and the Phase 0 tile is never created twice." : ""} The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. \`1 ✓ · 2 ✓ · 4 ✓ · 0 ▶ · 3 ☐\`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".
133
+ **TaskCreate ordering (strict)**: All TaskCreate calls in a registration batch fire in strict phase-number order BEFORE any TaskUpdate in that batch, and a later batch only ever appends phases numbered above everything already registered. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. \`1 ✓ · 2 ✓ · 4 ✓ · 0 ▶ · 3 ☐\`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".
164
134
 
165
135
  \`\`\`text
166
136
  # Register one tile per phase, capture the taskId, persist it:
167
- for each phase in ${spec.depth ? `0:Init at Step -1, then ${fullSet.join(", ")} (Full) or ${shortSet.join(", ")} (Short) at Step 7.5` : spec.phases.join(", ")}:
137
+ for each phase in ${spec.phases.join(", ")}:
168
138
  TaskCreate({ subject: "Phase <N>: <Name>", activeForm: "<doing-form>" })
169
139
  -> returns taskId
170
140
  bash $HOME/.claude/scripts/phase-tracker.sh meta <N> tasklist_id "<taskId>"
@@ -475,7 +475,7 @@ function cmdBrief() {
475
475
  );
476
476
  ordered = fused.slice(0, max).map((x) => pool[x.index]);
477
477
  // A task whose words match nothing in the ledger must not silently inject an
478
- // empty block where the newest entries used to be: fall back to recency.
478
+ // empty block in place of the newest entries: fall back to recency.
479
479
  if (ordered.length === 0) ordered = pool.slice().reverse().slice(0, max);
480
480
  }
481
481
 
@@ -40,10 +40,10 @@ const here = dirname(fileURLToPath(import.meta.url));
40
40
  *
41
41
  * This script lives in two shapes: `<repo>/pipeline/scripts/` in a checkout, and
42
42
  * `~/.claude/scripts/` (or `~/.copilot`, `~/.codex`) once installed. The default
43
- * used to be `resolve(here, "..", "..") + "/pipeline/skills/.skills-index.json"`,
44
- * which is right in a checkout and resolves to `$HOME/pipeline/skills/...` from an
45
- * installed tree - a path that has never existed. So dynamic skill loading exited 1
46
- * on every real install while passing its own smoke, which runs from the repo.
43
+ * is NOT `resolve(here, "..", "..") + "/pipeline/skills/.skills-index.json"`.
44
+ * That is right in a checkout and resolves to `$HOME/pipeline/skills/...` from an
45
+ * installed tree - a path that does not exist. Dynamic skill loading would exit 1
46
+ * on every real install while passing a smoke that runs from the repo.
47
47
  *
48
48
  * Installed layout is tried first: that is where a user's run happens.
49
49
  */
@@ -11,9 +11,9 @@
11
11
  # scripts/_retrieval.mjs) and the most relevant are printed. Without it the
12
12
  # index order is kept, which is what every existing caller gets.
13
13
  #
14
- # The cap used to be a bare `head -30`. That is a truncation, not a summary: the
15
- # thirty-first pointer was invisible however precisely it matched the task, and
16
- # the block got less useful the longer a repo was worked on - the opposite of
14
+ # The cap is not a bare `head -30`. That is a truncation, not a summary: the
15
+ # thirty-first pointer stays invisible however precisely it matches the task, and
16
+ # the block gets less useful the longer a repo is worked on - the opposite of
17
17
  # what accumulated memory is for.
18
18
  #
19
19
  # Honors `prefs.global.perRepoMemory`. When the pref is off OR the prefs file
@@ -357,31 +357,32 @@ function migrate(prefs) {
357
357
  out.global.skillConformance.blockOnCoverageGap = false;
358
358
  changes.push("added skillConformance.blockOnCoverageGap=false (report, do not halt)");
359
359
  }
360
- // Rename chain for the pipeline-tail command: /multi-agent:finish (pre-v2.5.0)
361
- // -> :ship (v2.5.0) -> :resume-local (current). The autoFix value survives every
362
- // hop - carry whichever legacy block still holds one rather than reset it.
363
- if (!out.global.resumeLocal || typeof out.global.resumeLocal !== "object") {
364
- out.global.resumeLocal = {};
365
- changes.push("added resumeLocal (/multi-agent:ship renamed to :resume-local)");
366
- }
367
- if (typeof out.global.resumeLocal.autoFix !== "boolean") {
368
- const legacyAutoFix =
369
- typeof out.global.ship?.autoFix === "boolean"
370
- ? out.global.ship.autoFix
371
- : out.global.finish?.autoFix;
372
- out.global.resumeLocal.autoFix = typeof legacyAutoFix === "boolean" ? legacyAutoFix : false;
360
+ // Rename chain for the tail that runs over work already on the branch:
361
+ // global.finish -> global.ship -> global.resumeLocal -> global.resume, the
362
+ // last hop following the command itself into /multi-agent:resume. The
363
+ // autoFix value survives every hop - carry whichever legacy block still
364
+ // holds one rather than reset it.
365
+ if (!out.global.resume || typeof out.global.resume !== "object") {
366
+ out.global.resume = {};
367
+ changes.push("added resume (:resume-local folded into :resume)");
368
+ }
369
+ if (typeof out.global.resume.autoFix !== "boolean") {
370
+ const legacyAutoFix = ["resumeLocal", "ship", "finish"]
371
+ .map((k) => out.global[k]?.autoFix)
372
+ .find((v) => typeof v === "boolean");
373
+ out.global.resume.autoFix = typeof legacyAutoFix === "boolean" ? legacyAutoFix : false;
373
374
  changes.push(
374
375
  typeof legacyAutoFix === "boolean"
375
- ? "moved ship/finish autoFix -> resumeLocal.autoFix (value preserved)"
376
- : "added resumeLocal.autoFix=false",
376
+ ? "moved tail autoFix -> resume.autoFix (value preserved)"
377
+ : "added resume.autoFix=false",
377
378
  );
378
379
  }
379
- for (const legacy of ["ship", "finish"]) {
380
+ for (const legacy of ["resumeLocal", "ship", "finish"]) {
380
381
  if (out.global[legacy] && typeof out.global[legacy] === "object") {
381
382
  delete out.global[legacy].autoFix;
382
383
  if (Object.keys(out.global[legacy]).length === 0) {
383
384
  delete out.global[legacy];
384
- changes.push(`removed obsolete ${legacy} block (renamed to resumeLocal)`);
385
+ changes.push(`removed obsolete ${legacy} block (renamed to resume)`);
385
386
  }
386
387
  }
387
388
  }
@@ -1132,7 +1132,7 @@ case "$ACTION" in
1132
1132
  state=$(load_state)
1133
1133
  # Idempotent: a phase id that already exists is a no-op (name, status, and
1134
1134
  # token history are preserved). This is what lets continuation commands
1135
- # (/multi-agent:resume-local, resume) re-declare their phase set against a
1135
+ # (/multi-agent:resume, /multi-agent:manual-test) re-declare their phase set against a
1136
1136
  # pre-existing tracker without duplicating or resetting tiles.
1137
1137
  new=$(echo "$state" | jq --arg id "$PID" --arg name "$PNAME" '
1138
1138
  if any(.phases[]?; .id == $id) then .
@@ -1167,7 +1167,7 @@ case "$ACTION" in
1167
1167
  # completion is recoverable (record, then re-run) and loses no work; a
1168
1168
  # phase that genuinely made no LLM call says so with --no-llm.
1169
1169
  if [ "$STATUS" = "completed" ] && [ "$NO_LLM" -eq 0 ]; then
1170
- case " ${TRACKER_LLM_PHASES:-1 2 3 4} " in
1170
+ case " ${TRACKER_LLM_PHASES:-1 2 3 4 5} " in
1171
1171
  *" $PID "*)
1172
1172
  RECORDED=$(load_state | jq -r --arg id "$PID" \
1173
1173
  '(.phases[] | select(.id == $id) | (.tokens_in // 0) + (.tokens_out // 0)) // 0')
@@ -270,7 +270,7 @@ export function evaluate(state, extraInput = "") {
270
270
  failures.push(
271
271
  `workspaceSource="autopilot" on an interactive run. Autopilot resolves the workspace ` +
272
272
  `to a worktree because an unattended commit in the user's own checkout is what ` +
273
- `worktrees prevent; an interactive run asks (Step 5b) or is told by :local / --local.`,
273
+ `worktrees prevent; an interactive run asks at Step 5b, and the multi-repo bridge records "command".`,
274
274
  );
275
275
  }
276
276
 
@@ -77,9 +77,9 @@ export function filesToAdd(markdown) {
77
77
  if (!path || /^-+$/.test(path)) continue;
78
78
  if (path.includes("<") || tag.includes("<")) continue;
79
79
  // No header-row filter: the tag column of the header reads "Etiket / Tag",
80
- // which the `Add new` test below already rejects. The filter that used to be
81
- // here matched on the PATH column instead and would have skipped a real row
82
- // whose path begins with a `File/` directory.
80
+ // which the `Add new` test below already rejects. A filter here would have to
81
+ // match on the PATH column instead, and would skip a real row whose path
82
+ // begins with a `File/` directory.
83
83
  if (!/^add new$/i.test(tag)) continue;
84
84
  rows.push(path);
85
85
  }
@@ -150,9 +150,9 @@ if (isMain) {
150
150
 
151
151
  // No plan at all is not a pass, and neither is an empty one: a Phase 2 that
152
152
  // produced no steps, or a state whose todos got cleared, would otherwise read
153
- // as "0/0 accounted for" and let the commit through. A Short run has no Phase
154
- // 2, and the caller is expected to skip this gate for those modes rather than
155
- // let it report clean.
153
+ // as "0/0 accounted for" and let the commit through. A mode with no Phase 2
154
+ // (analysis) is expected to skip this gate explicitly rather than let it
155
+ // report clean.
156
156
  if (!Array.isArray(todos) || todos.length === 0) {
157
157
  const msg = Array.isArray(todos)
158
158
  ? "the plan has zero steps - that is an unusable plan, not a clean one; skip this gate explicitly for modes with no Phase 2"
@@ -87,8 +87,8 @@ function die(msg) {
87
87
  // Path resolution lives in _run-paths.mjs: tracker-state.json is written to
88
88
  // the flat <root>/<task-id>/ while agent-state.json documents the nested
89
89
  // <root>/<project>/<task-id>/, and both layouts are populated in practice.
90
- // This used to be a hand-rolled candidate list here, a second one in
91
- // render-cost-summary.sh and a third in render-agent-log-cost.sh.
90
+ // Resolved once here rather than as a hand-rolled candidate list in each of
91
+ // render-cost-summary.sh and render-agent-log-cost.sh as well.
92
92
  function resolveTrackerCandidates() {
93
93
  const candidates = [];
94
94
  if (flags["task-dir"]) {
@@ -159,7 +159,7 @@ function computeCost(modelKey, tokensIn, tokensOut, costTable) {
159
159
 
160
160
  /**
161
161
  * Heuristic: Plan (1) uses Opus per personas + dev-mode docs; Dev (2) uses
162
- * Sonnet by default unless the run is a Short pipeline (Opus); Review (3)
162
+ * Sonnet by default unless routing names another rung; Review (3)
163
163
  * triage is Opus and the reviewer breakdown comes from spans.
164
164
  *
165
165
  * This is a fallback when phase-tracker doesn't record the model used. Live
@@ -44,9 +44,9 @@ import { invokedDirectly } from "../lib/invoked-directly.mjs";
44
44
 
45
45
  /**
46
46
  * The phase from which a run counts as "waiting on you", read from the phase
47
- * contract rather than written here. This threshold moved once already (it was
48
- * 6 under the eight-phase contract, it is 4 under six) and nothing connected it
49
- * to the renumbering, so it would have silently regrouped every run.
47
+ * contract rather than written here. The threshold follows the phase count (6
48
+ * under an eight-phase contract, 4 under six), so a literal would survive a
49
+ * renumbering unchanged and silently regroup every run.
50
50
  */
51
51
  const PHASE_WAITING_FROM = JSON.parse(
52
52
  readFileSync(new URL("../schemas/phases.json", import.meta.url), "utf8"),
@@ -203,10 +203,10 @@ export function buildIndex() {
203
203
  salvaged: Boolean(statePath && statePath.includes(`${run.dir}/artifacts/`)),
204
204
  // What KIND of record this is, before asking whether it is healthy.
205
205
  //
206
- // 74 of the 103 runs on this machine have a tracker file and no agent
207
- // state, and the first version of this index called every one of them
208
- // `stateReadable: false` - so a panel built on it announced "74 runs
209
- // could not be read" about records that are not damaged and never had
206
+ // A run with a tracker file and no agent state is the common case, not
207
+ // damage: most runs on a machine are shaped that way. Calling them
208
+ // `stateReadable: false` makes a panel announce "N runs could not be
209
+ // read" about records that are intact and never had
210
210
  // agent state to begin with. `/multi-agent:analysis` says so in its own
211
211
  // description ("no worktree, no commits, no dev chain"), and design-check
212
212
  // is the same shape.
@@ -4,7 +4,7 @@
4
4
  *
5
5
  * Phase 3 writes `.pipeline/scope-check.json` before the handoff: one reason
6
6
  * per touched file, the things it deliberately did not do, and the simplifier
7
- * rationales that used to be discarded. This gate compares that record with
7
+ * rationales that are otherwise discarded. This gate compares that record with
8
8
  * the real diff so an unjustified file cannot ride into review unnoticed.
9
9
  *
10
10
  * Usage: