@mmerterden/multi-agent-pipeline 17.4.0 → 17.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +136 -0
  2. package/README.md +23 -5
  3. package/README.tr.md +23 -5
  4. package/docs/adr/0013-lsp-code-intelligence.md +102 -0
  5. package/docs/adr/README.md +1 -0
  6. package/docs/token-budget-history.md +1 -1
  7. package/install/templates/copilot-instructions.md +9 -3
  8. package/package.json +1 -1
  9. package/pipeline/commands/multi-agent/analysis/SKILL.md +3 -3
  10. package/pipeline/commands/multi-agent/autopilot/SKILL.md +3 -3
  11. package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +5 -3
  12. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  13. package/pipeline/commands/multi-agent/local/SKILL.md +17 -6
  14. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +3 -3
  15. package/pipeline/lib/multi-repo-pipeline.sh +26 -0
  16. package/pipeline/multi-agent-refs/features/base-branch-evidence.md +222 -0
  17. package/pipeline/multi-agent-refs/features/code-intelligence.md +80 -0
  18. package/pipeline/multi-agent-refs/phases/modes.md +23 -3
  19. package/pipeline/multi-agent-refs/phases/phase-0-init.md +96 -71
  20. package/pipeline/multi-agent-refs/phases/phase-7-report.md +1 -1
  21. package/pipeline/multi-agent-refs/phases.md +7 -2
  22. package/pipeline/multi-agent-refs/picker-contract.md +37 -5
  23. package/pipeline/multi-agent-refs/tracker-contract.md +25 -14
  24. package/pipeline/schemas/agent-state.schema.json +88 -4
  25. package/pipeline/schemas/prefs.schema.json +22 -0
  26. package/pipeline/schemas/token-budget.json +2 -2
  27. package/pipeline/scripts/autopilot-runner.mjs +292 -45
  28. package/pipeline/scripts/base-branch-candidates.mjs +599 -0
  29. package/pipeline/scripts/gc-abandoned.sh +5 -3
  30. package/pipeline/scripts/gen-mode-dispatch.mjs +39 -16
  31. package/pipeline/scripts/phase-tracker.sh +39 -2
  32. package/pipeline/scripts/phase0-exit-gate.mjs +128 -0
  33. package/pipeline/scripts/verify-citations.mjs +84 -2
  34. package/pipeline/skills/.skill-manifest.json +2 -2
  35. package/pipeline/skills/shared/core/multi-agent/SKILL.md +1 -1
@@ -17,9 +17,10 @@
17
17
  *
18
18
  * v16.0.0 removed the four dev-* modes. Depth is no longer a command name: the
19
19
  * Phase 0 Step 7.5 picker asks Full or Short and sets `state.onlyDevelop`. That
20
- * answer arrives long after the tracker boots at Step -1, so a Short run
21
- * registers its full set and flips Phases 1 and 2 to `skipped` when it learns.
22
- * A generated phase set is therefore per-COMMAND, never per-depth.
20
+ * answer arrives long after the tracker boots at Step -1, so v17.5.0 splits
21
+ * registration for the two modes that ask it (`full`, `local`): Phase 0 at Step -1,
22
+ * the rest once depth has named them. Everything else still registers its whole
23
+ * set up front, and a generated phase set is per-COMMAND, never per-depth.
23
24
  *
24
25
  * Companion smoke `smoke-mode-dispatch-drift.sh` regenerates the section for
25
26
  * each mode file, diffs against the on-disk content, and fails on drift.
@@ -60,10 +61,10 @@ const PHASE_NAMES_NO_TEST = PHASE_NAMES.filter((p) => p !== "5:Test");
60
61
  */
61
62
  const MODES = {
62
63
  autopilot: { phases: PHASE_NAMES_NO_TEST, local: false, autopilot: true },
63
- full: { phases: PHASE_NAMES, local: false, autopilot: false },
64
- local: { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false },
64
+ full: { phases: PHASE_NAMES, local: false, autopilot: false, depth: true },
65
+ local: { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false, depth: true },
65
66
  "local-autopilot": { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: true },
66
- "full-local": { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false },
67
+ "full-local": { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false, depth: true },
67
68
  // Analysis produces a document, not code: no Dev, no Test, and Phase 6
68
69
  // publishes instead of committing. Same 8-phase contract, four of them
69
70
  // reinterpreted.
@@ -81,8 +82,6 @@ if (!spec) {
81
82
  process.exit(2);
82
83
  }
83
84
 
84
- const phaseLoop = spec.phases.map((p) => `"${p}"`).join(" ");
85
-
86
85
  const ALL_PHASE_IDS = PHASE_NAMES.map((p) => p.split(":")[0]);
87
86
  const activeIds = spec.phases.map((p) => p.split(":")[0]);
88
87
  const skippedIds = ALL_PHASE_IDS.filter((id) => !activeIds.includes(id));
@@ -104,7 +103,7 @@ const skipNote =
104
103
  ? `${modeLabel} mode does NOT TaskCreate phases ${skippedIds.join("/")} - those are not part of the ${modeLabel} phase set (\`${spec.phases.join(" ")}\`). Only register tiles for the active set.`
105
104
  : `${modeLabel} mode TaskCreates all 8 phases (no phase is skipped).`;
106
105
 
107
- const orderingNote = `**All TaskCreate calls fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
106
+ const orderingNote = `**All TaskCreate calls in a batch fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${spec.depth ? `Phase 0 at Step -1, then the rest in ascending order at Step 7.5 (${phaseSequence} minus whatever the depth answer drops)` : phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
108
107
 
109
108
  const localCaveat = spec.local
110
109
  ? `\n> **Local mode:** no worktree is created, work happens on the current branch. Phase 0 Init still calls \`init\` - the \`--local\` flag is stored in tracker-state.json, and \`:resume\` restores the correct CWD.\n`
@@ -114,6 +113,32 @@ const banner = spec.autopilot
114
113
  ? `\n> **Autopilot mode:** user confirmations are skipped. The tracker is still mandatory - autopilot agent calls cannot skip it; skipping breaks \`smoke-tracker-contract.sh\`.\n`
115
114
  : "";
116
115
 
116
+ // Registration shape. A mode that asks the depth question does not know its phase
117
+ // set at Step -1 (the answer needs taskType, which needs the fetched issue and the
118
+ // branch), so it registers Phase 0 there and the rest at Step 7.5. Drawing eight
119
+ // tiles beside the question that decides whether two of them run is the failure this
120
+ // split exists to remove.
121
+ const fullSet = spec.phases.filter((p) => p !== "0:Init");
122
+ const shortSet = fullSet.filter((p) => Number(p.split(":")[0]) >= 3);
123
+ const loopFor = (list) =>
124
+ `for p in ${list.map((x) => `"${x}"`).join(" ")}; do\n bash $HOME/.claude/scripts/phase-tracker.sh add "\${p%%:*}" "\${p#*:}"\ndone`;
125
+
126
+ const initLoop = spec.depth
127
+ ? 'bash $HOME/.claude/scripts/phase-tracker.sh add 0 "Init"\nbash $HOME/.claude/scripts/phase-tracker.sh tiles'
128
+ : `${loopFor(spec.phases)}`;
129
+
130
+ const deferredBlock = spec.depth
131
+ ? `
132
+ # Phase 0 Step 7.5, immediately after the depth answer - the first moment this
133
+ # mode knows its phase set. Full:
134
+ ${loopFor(fullSet)}
135
+ # Short (Analysis and Planning are not run, so they get no tile at all):
136
+ ${loopFor(shortSet)}
137
+ # Then the widget, narrowed to the phases that do not have a tile yet:
138
+ bash $HOME/.claude/scripts/phase-tracker.sh tiles --new
139
+ `
140
+ : "";
141
+
117
142
  const out = `## Required: Phase Tracker Contract
118
143
 
119
144
  **The phase tracker is mandatory** - the agent cannot skip it. Full spec: [\`$HOME/.claude/multi-agent-refs/tracker-contract.md\`]($HOME/.claude/multi-agent-refs/tracker-contract.md).
@@ -126,11 +151,9 @@ Two channels run in parallel at every phase boundary:
126
151
  \`\`\`bash
127
152
  # Phase 0, very first shell call (every CLI):
128
153
  bash $HOME/.claude/scripts/phase-tracker.sh init "$TASK_ID"
129
- for p in ${phaseLoop}; do
130
- bash $HOME/.claude/scripts/phase-tracker.sh add "\${p%%:*}" "\${p#*:}"
131
- done
154
+ ${initLoop}
132
155
  bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
133
-
156
+ ${deferredBlock}
134
157
  # Every phase boundary (every CLI):
135
158
  bash $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress|completed|failed|skipped
136
159
 
@@ -142,11 +165,11 @@ bash $HOME/.claude/scripts/phase-tracker.sh tokens <N> <in> <out> [cached]
142
165
 
143
166
  In Claude Code the agent MUST also drive the native TaskList widget so the user sees a sticky phase tile stack - this is the only progress signal Claude Code surfaces. Skipping these calls is the #1 source of "I don't see any phases" complaints.
144
167
 
145
- **TaskCreate ordering (strict)**: All TaskCreate calls fire in strict phase-number order BEFORE any TaskUpdate is applied. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. \`1 ✓ · 2 ✓ · 4 ✓ · 0 ▶ · 3 ☐\`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".
168
+ **TaskCreate ordering (strict)**: All TaskCreate calls in a registration batch fire in strict phase-number order BEFORE any TaskUpdate in that batch, and a later batch only ever appends phases numbered above everything already registered.${spec.depth ? " This mode registers in two batches (Step -1, then Step 7.5), so `tiles --new` narrows the second one and the Phase 0 tile is never created twice." : ""} The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. \`1 ✓ · 2 ✓ · 4 ✓ · 0 ▶ · 3 ☐\`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".
146
169
 
147
170
  \`\`\`text
148
- # Phase 0 startup - register one tile per phase (0..N), capture the taskId, persist it:
149
- for each phase in ${spec.phases.join(", ")}:
171
+ # Register one tile per phase, capture the taskId, persist it:
172
+ for each phase in ${spec.depth ? `0:Init at Step -1, then ${fullSet.join(", ")} (Full) or ${shortSet.join(", ")} (Short) at Step 7.5` : spec.phases.join(", ")}:
150
173
  TaskCreate({ subject: "Phase <N>: <Name>", activeForm: "<doing-form>" })
151
174
  -> returns taskId
152
175
  bash $HOME/.claude/scripts/phase-tracker.sh meta <N> tasklist_id "<taskId>"
@@ -681,8 +681,17 @@ EOF
681
681
  # a subject is a task title and travels through surfaces that are not a terminal.
682
682
  subjects() {
683
683
  need_jq
684
- local state want="${1:-}"
684
+ local state want="${1:-}" new_only=""
685
+ # `--new` narrows to phases that have no tile yet, which is how a deferred
686
+ # registration batch (Phase 0 at Step -1, the rest once depth is known) avoids
687
+ # asking the host to create a tile it already has. A phase whose tasklist_id was
688
+ # never recorded has, as far as anything can tell, no tile.
689
+ if [ "$want" = "--new" ]; then new_only=1; want=""; fi
685
690
  state=$(load_state)
691
+ local new_ids=""
692
+ if [ -n "$new_only" ]; then
693
+ new_ids=" $(echo "$state" | jq -r '[.phases[]? | select(((.meta.tasklist_id // "") | tostring) == "") | .id] | join(" ")') "
694
+ fi
686
695
  local prices_json='{"prices":{}}'
687
696
  if [ -f "$COST_TABLE" ]; then
688
697
  prices_json=$(cat "$COST_TABLE" 2>/dev/null) || prices_json='{"prices":{}}'
@@ -704,6 +713,9 @@ subjects() {
704
713
  while IFS=$'\x1f' read -r pid pname pstatus p_start p_end pmodel ptok pusd; do
705
714
  [ -n "$pid" ] || continue
706
715
  [ -z "$want" ] || [ "$want" = "$pid" ] || continue
716
+ if [ -n "$new_only" ]; then
717
+ case "$new_ids" in *" $pid "*) ;; *) continue ;; esac
718
+ fi
707
719
  local line="Phase $pid $pname"
708
720
  [ -n "$pmodel" ] && line="$line - $pmodel"
709
721
  local s_ep e_ep el=""
@@ -755,10 +767,33 @@ subjects() {
755
767
  # already does for a different reason.
756
768
  tiles_script() {
757
769
  need_jq
770
+ local new_only="${1:-}"
758
771
  local state; state=$(load_state)
759
772
  local has_subs
760
773
  has_subs=$(echo "$state" | jq '[.phases[]?.subs[]?] | length')
761
774
 
775
+ # A list carrying sub-steps has to be rebuilt whole, so the rebuild wins over a
776
+ # narrowed batch: appending to it would land the new rows after Phase 7.
777
+ [ "${has_subs:-0}" -gt 0 ] && new_only=""
778
+
779
+ if [ -n "$new_only" ]; then
780
+ local pending_new
781
+ pending_new=$(subjects --new)
782
+ if [ -z "$pending_new" ]; then
783
+ printf 'Every registered phase already has a tile - nothing to create.\n'
784
+ return 0
785
+ fi
786
+ printf 'REQUIRED - create one native tile per phase below, in this exact order.
787
+ '
788
+ printf 'Tiles already created keep their place; every phase here is numbered
789
+ '
790
+ printf 'above them, so the widget stays in phase order.
791
+
792
+ '
793
+ printf '%s\n' "$pending_new" | sed 's/^/ TaskCreate(subject: "/; s/$/")/'
794
+ return 0
795
+ fi
796
+
762
797
  if [ "${has_subs:-0}" -gt 0 ]; then
763
798
  printf 'The tile list has sub-steps now, and the widget orders by CREATION.
764
799
  '
@@ -1346,6 +1381,8 @@ EOF
1346
1381
 
1347
1382
  tiles)
1348
1383
  need_jq
1384
+ tiles_new=""
1385
+ [ "${1:-}" = "--new" ] && tiles_new=1
1349
1386
  tiles_state=$(load_state)
1350
1387
  tiles_count=$(echo "$tiles_state" | jq '[.phases[]?] | length')
1351
1388
  [ "${tiles_count:-0}" -gt 0 ] || {
@@ -1368,7 +1405,7 @@ EOF
1368
1405
  # tools it has. And the fallback is not enough on its own: a user looking
1369
1406
  # at a missing widget needs the one command that brings it back, which is
1370
1407
  # why the opt-in is printed next to it.
1371
- tiles_script
1408
+ tiles_script "$tiles_new"
1372
1409
  echo
1373
1410
  echo "At every phase boundary re-run \`phase-tracker.sh subjects <id>\` and"
1374
1411
  echo "pass that line as the subject of the TaskUpdate. The subject is the only"
@@ -20,6 +20,20 @@
20
20
  * `agent-state.json` is the only durable evidence they happened - so the gate asserts
21
21
  * their output too, not just `taskType`.
22
22
  *
23
+ * A third run got past that, because `baseBranch` and `baseFetchStatus` can be filled
24
+ * in by a branch that was never chosen. The remote had one PR-targetable branch, the
25
+ * one-option `AskUserQuestion` was refused by the host (its schema needs two), and the
26
+ * run announced "only candidate, continuing with it" and carried on. The branch was
27
+ * right; nothing was asked. `baseBranchSource` is the field that separates those two,
28
+ * and the dev-context picker never ran at all - no `siblings`, so Phase 4's parity
29
+ * cross-check had nothing to read and could not tell an empty answer from no answer.
30
+ *
31
+ * A fourth showed the same shape one layer down: `git fetch origin` failed on a
32
+ * restricted network, `git branch -r` printed the remote-tracking cache anyway, and a
33
+ * weeks-old local list was presented as the remote's answer. Degrading to local refs is
34
+ * correct; reporting them as remote is not, so `baseBranchEvidence.refProvenance` has to
35
+ * agree with `baseFetchStatus`.
36
+ *
23
37
  * A phase that reports success without its output is worse than one that fails:
24
38
  * every later phase then reasons from a field that is not there. So this is a
25
39
  * gate, not a lint - the spec already said what to write, and prose alone did
@@ -157,6 +171,120 @@ export function evaluate(state, extraInput = "") {
157
171
  );
158
172
  }
159
173
 
174
+ // Which rule decided the base branch. `asked` and `input` are the only two an
175
+ // interactive run can honestly record: `remembered`, `default` and `derived` are the
176
+ // autopilot resolutions, and an interactive run that reaches for them has skipped its
177
+ // picker. This is the assertion `baseBranch` alone cannot make - a branch announced in
178
+ // prose and a branch chosen by the user leave the same value behind. A branch the run
179
+ // derived from the issue and the user then confirmed is still `asked`; the derivation
180
+ // lives in `baseBranchEvidence`, which is a record, not a permission.
181
+ const BRANCH_SOURCES = ["asked", "input", "remembered", "default", "derived"];
182
+ const branchSource =
183
+ typeof state.baseBranchSource === "string" ? state.baseBranchSource.trim() : "";
184
+ if (baseBranch && !BRANCH_SOURCES.includes(branchSource)) {
185
+ failures.push(
186
+ `agent-state.json has baseBranchSource="${branchSource || "<unset>"}"; Step 3 must ` +
187
+ `record one of ${BRANCH_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
188
+ `branch the user chose from one the run picked and announced.`,
189
+ );
190
+ }
191
+ const isAutopilot = state.autopilot === true;
192
+ const AUTOPILOT_ONLY_SOURCES = ["remembered", "default", "derived"];
193
+ if (!isAutopilot && AUTOPILOT_ONLY_SOURCES.includes(branchSource)) {
194
+ failures.push(
195
+ `baseBranchSource="${branchSource}" on an interactive run. Those three are autopilot ` +
196
+ `resolutions; an interactive run asks (Step 3 is not skippable) or takes the base ` +
197
+ `from the task reference. A one-candidate filter is still asked, with a second ` +
198
+ `option - picker-contract.md, "Two options or it is not a question".`,
199
+ );
200
+ }
201
+
202
+ // What the base branch was chosen from, and what that list was worth.
203
+ //
204
+ // Two separate silent failures live here. The first: `git fetch origin` can fail on a
205
+ // restricted network while `git branch -r` still prints a full, confident list - the
206
+ // remote-tracking cache - so a weeks-old local guess gets presented as the remote's
207
+ // answer. Falling back to local refs is correct; not saying so is not. The second:
208
+ // `derived` claims the issue named the branch, and a claim with no evidence behind it
209
+ // is `default` wearing a hat.
210
+ const evidence =
211
+ state.baseBranchEvidence && typeof state.baseBranchEvidence === "object"
212
+ ? state.baseBranchEvidence
213
+ : null;
214
+ const DEGRADED_FETCH = ["cached-stale", "local-branch"];
215
+ if (DEGRADED_FETCH.includes(fetchStatus)) {
216
+ if (!evidence) {
217
+ failures.push(
218
+ `baseFetchStatus="${fetchStatus}" but state.baseBranchEvidence is absent. A run whose ` +
219
+ `fetch failed listed its branches from local refs; the record of that is what stops a ` +
220
+ `local-only guess being read afterwards as the remote's answer.`,
221
+ );
222
+ } else if (evidence.refProvenance !== "local") {
223
+ failures.push(
224
+ `baseFetchStatus="${fetchStatus}" but baseBranchEvidence.refProvenance=` +
225
+ `"${evidence.refProvenance || "<unset>"}". The fetch failed, so the candidate list came ` +
226
+ `from local refs and must say so - "remote" here is the silent degradation this field exists to catch.`,
227
+ );
228
+ }
229
+ }
230
+ if (branchSource === "derived") {
231
+ const ISSUE_DERIVED = new Set(["issue-version", "linked-release"]);
232
+ const cands = evidence && Array.isArray(evidence.candidates) ? evidence.candidates : [];
233
+ const chosen = cands.find((c) => c && c.branch === baseBranch);
234
+ const hasIssueEvidence =
235
+ chosen &&
236
+ Array.isArray(chosen.evidence) &&
237
+ chosen.evidence.some((e) => e && ISSUE_DERIVED.has(e.kind));
238
+ if (!hasIssueEvidence) {
239
+ failures.push(
240
+ `baseBranchSource="derived" but baseBranchEvidence carries no issue-version or ` +
241
+ `linked-release evidence for "${baseBranch}". "Derived" names a specific claim - the ` +
242
+ `issue's version field or a linked release issue pointed at this branch - and without ` +
243
+ `that record it is the sort-order default under a better name.`,
244
+ );
245
+ }
246
+ if (evidence && evidence.ambiguous === true) {
247
+ failures.push(
248
+ `baseBranchSource="derived" with baseBranchEvidence.ambiguous=true. Two or more ` +
249
+ `candidates tied at the top score; autopilot does not break a tie by picking one.`,
250
+ );
251
+ }
252
+ }
253
+
254
+ // Step 5b decides where the branch lives, and `localMode` alone cannot say
255
+ // whether anyone decided: `false` is both "the user chose a worktree" and
256
+ // "nothing asked and the default stood". Same shape as baseBranchSource, and
257
+ // the same reason - an autopilot run resolves it rather than asking, so
258
+ // `autopilot` is a legal source there and nowhere else.
259
+ const WORKSPACE_SOURCES = ["asked", "command", "autopilot"];
260
+ const workspaceSource =
261
+ typeof state.workspaceSource === "string" ? state.workspaceSource.trim() : "";
262
+ if (!WORKSPACE_SOURCES.includes(workspaceSource)) {
263
+ failures.push(
264
+ `agent-state.json has workspaceSource="${workspaceSource || "<unset>"}"; Step 5b must ` +
265
+ `record one of ${WORKSPACE_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
266
+ `worktree the user chose from one no question was asked about.`,
267
+ );
268
+ }
269
+ if (!isAutopilot && workspaceSource === "autopilot") {
270
+ failures.push(
271
+ `workspaceSource="autopilot" on an interactive run. Autopilot resolves the workspace ` +
272
+ `to a worktree because an unattended commit in the user's own checkout is what ` +
273
+ `worktrees prevent; an interactive run asks (Step 5b) or is told by :local / --local.`,
274
+ );
275
+ }
276
+
277
+ // Step 2b's dev-context picker writes siblings[], empty included. The empty array is
278
+ // the record that it ran; absent, a multi-repo task silently became a single-repo one
279
+ // and Phase 4's parity cross-check lost its fourth counterpart source.
280
+ if (!Array.isArray(state.siblings)) {
281
+ failures.push(
282
+ "agent-state.json has no siblings array. Step 2b runs the dev-context picker on " +
283
+ "every input type and persists the result, `[]` included; an absent field means " +
284
+ "the picker never ran, so extra repos and read-only counterparts were never offered.",
285
+ );
286
+ }
287
+
160
288
  // Worktree isolation is a standing rule: never develop in the primary checkout.
161
289
  const worktree = typeof state.worktreePath === "string" ? state.worktreePath.trim() : "";
162
290
  const projectRoot = typeof state.projectRoot === "string" ? state.projectRoot.trim() : "";
@@ -67,6 +67,79 @@ export function unsafePath(p) {
67
67
  return null;
68
68
  }
69
69
 
70
+ /**
71
+ * Extensions a markdown citation may carry.
72
+ *
73
+ * A list, not a pattern, and that direction is deliberate: the cost of a
74
+ * missing extension is one citation going unchecked, while the cost of an open
75
+ * pattern is the gate failing a correct document over a hostname. Add to it
76
+ * when a real repository file is missed - never widen it to a wildcard.
77
+ *
78
+ * `com`, `net`, `io`, `dev`, `tr` are absent on purpose.
79
+ */
80
+ export const SOURCE_EXTENSIONS = new Set([
81
+ "swift",
82
+ "kt",
83
+ "kts",
84
+ "java",
85
+ "m",
86
+ "mm",
87
+ "h",
88
+ "hpp",
89
+ "c",
90
+ "cc",
91
+ "cpp",
92
+ "js",
93
+ "mjs",
94
+ "cjs",
95
+ "jsx",
96
+ "ts",
97
+ "tsx",
98
+ "py",
99
+ "rb",
100
+ "go",
101
+ "rs",
102
+ "php",
103
+ "sh",
104
+ "bash",
105
+ "zsh",
106
+ "pl",
107
+ "sql",
108
+ "json",
109
+ "yml",
110
+ "yaml",
111
+ "toml",
112
+ "xml",
113
+ "plist",
114
+ "xcconfig",
115
+ "pbxproj",
116
+ "gradle",
117
+ "properties",
118
+ "cfg",
119
+ "ini",
120
+ "env",
121
+ "md",
122
+ "mdx",
123
+ "txt",
124
+ "csv",
125
+ "tsv",
126
+ "strings",
127
+ "stringsdict",
128
+ "html",
129
+ "css",
130
+ "scss",
131
+ "vue",
132
+ "svelte",
133
+ "graphql",
134
+ "proto",
135
+ "entitlements",
136
+ "storyboard",
137
+ "xib",
138
+ "lock",
139
+ "gemspec",
140
+ "podspec",
141
+ ]);
142
+
70
143
  /**
71
144
  * Pull `{file, line}` pairs out of whatever was handed over.
72
145
  *
@@ -124,6 +197,14 @@ export function extract(raw) {
124
197
  // Fenced blocks are skipped for the same class of reason: a code sample that
125
198
  // happens to contain `foo.js:12` is an illustration, not a claim about this
126
199
  // repository.
200
+ //
201
+ // The URL strip below only removes what carries a scheme. A host written
202
+ // bare - `wiki.example.com:8443/x`, and an analysis document is full of
203
+ // them - still looks exactly like a path with an extension and a line
204
+ // number, and this gate would fail the document over `.com`. So the
205
+ // extension has to be one a repository actually contains. This is the same
206
+ // defect that was already fixed once here for scheme-carrying URLs; the bare
207
+ // form survived it.
127
208
  let fenced = false;
128
209
  raw.split("\n").forEach((line, i) => {
129
210
  if (/^\s*```/.test(line)) {
@@ -132,8 +213,9 @@ export function extract(raw) {
132
213
  }
133
214
  if (fenced) return;
134
215
  const cleaned = line.replace(/\b[a-z][a-z0-9+.-]*:\/\/\S+/gi, " ");
135
- for (const m of cleaned.matchAll(/([A-Za-z0-9_./-]+\.[A-Za-z][A-Za-z0-9]*):(\d+)/g)) {
136
- push(m[1], Number(m[2]), `line ${i + 1}`);
216
+ for (const m of cleaned.matchAll(/([A-Za-z0-9_./-]+\.([A-Za-z][A-Za-z0-9]*)):(\d+)/g)) {
217
+ if (!SOURCE_EXTENSIONS.has(m[2].toLowerCase())) continue;
218
+ push(m[1], Number(m[3]), `line ${i + 1}`);
137
219
  }
138
220
  });
139
221
  return out;
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schemaVersion": "1.0.0",
3
- "generatedAt": "2026-09-14T11:56:29Z",
3
+ "generatedAt": "2026-09-15T12:49:28Z",
4
4
  "skillCount": 212,
5
5
  "entries": [
6
6
  {
@@ -237,7 +237,7 @@
237
237
  },
238
238
  {
239
239
  "path": "shared/core/multi-agent/SKILL.md",
240
- "sha256": "de56a86c64f1e6f56d1d4e5f0e129f04e44c36d930caa3857dd5c46689e8e501"
240
+ "sha256": "faac7ceb535d18778efe0ea91c19534c8e076bc709e7b1dbc567f5e71692bbbd"
241
241
  },
242
242
  {
243
243
  "path": "shared/external/accessibility-compliance-accessibility-audit/SKILL.md",
@@ -316,7 +316,7 @@ All TaskCreate calls fire in strict phase-number order BEFORE any TaskUpdate is
316
316
  - `multi-agent-local`, `multi-agent-autopilot`, `multi-agent-local-autopilot`: 0 → 1 → 2 → 3 → 4 → 6 → 7 (the interactive Phase 5 gate needs a worktree checkout and an attended run)
317
317
  - `multi-agent-analysis`: 0 → 1 → 2 → 4 → 6 → 7 (no code, so no Dev and no Test)
318
318
 
319
- Depth does not change the SET. A Short run (the Phase 0 Step 7.5 answer) registers its command's full set and flips Phases 1 and 2 to `skipped` when the answer lands at Step 7.5 - the tracker boots at Step -1, long before the question can be asked.
319
+ Depth decides WHICH of the command's set is registered. The tracker boots at Step -1, long before the depth question can be asked, so it registers Phase 0 alone and the remaining tiles are created at Step 7.5 once the answer is known - a Short run never draws an Analysis or Planning tile.
320
320
 
321
321
  Full contract: `refs/tracker-contract.md` section "TaskCreate ordering (strict)".
322
322