@mmerterden/multi-agent-pipeline 17.4.0 → 17.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +136 -0
- package/README.md +23 -5
- package/README.tr.md +23 -5
- package/docs/adr/0013-lsp-code-intelligence.md +102 -0
- package/docs/adr/README.md +1 -0
- package/docs/token-budget-history.md +1 -1
- package/install/templates/copilot-instructions.md +9 -3
- package/package.json +1 -1
- package/pipeline/commands/multi-agent/analysis/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +5 -3
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/local/SKILL.md +17 -6
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +3 -3
- package/pipeline/lib/multi-repo-pipeline.sh +26 -0
- package/pipeline/multi-agent-refs/features/base-branch-evidence.md +222 -0
- package/pipeline/multi-agent-refs/features/code-intelligence.md +80 -0
- package/pipeline/multi-agent-refs/phases/modes.md +23 -3
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +96 -71
- package/pipeline/multi-agent-refs/phases/phase-7-report.md +1 -1
- package/pipeline/multi-agent-refs/phases.md +7 -2
- package/pipeline/multi-agent-refs/picker-contract.md +37 -5
- package/pipeline/multi-agent-refs/tracker-contract.md +25 -14
- package/pipeline/schemas/agent-state.schema.json +88 -4
- package/pipeline/schemas/prefs.schema.json +22 -0
- package/pipeline/schemas/token-budget.json +2 -2
- package/pipeline/scripts/autopilot-runner.mjs +292 -45
- package/pipeline/scripts/base-branch-candidates.mjs +599 -0
- package/pipeline/scripts/gc-abandoned.sh +5 -3
- package/pipeline/scripts/gen-mode-dispatch.mjs +39 -16
- package/pipeline/scripts/phase-tracker.sh +39 -2
- package/pipeline/scripts/phase0-exit-gate.mjs +128 -0
- package/pipeline/scripts/verify-citations.mjs +84 -2
- package/pipeline/skills/.skill-manifest.json +2 -2
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +1 -1
|
@@ -17,9 +17,10 @@
|
|
|
17
17
|
*
|
|
18
18
|
* v16.0.0 removed the four dev-* modes. Depth is no longer a command name: the
|
|
19
19
|
* Phase 0 Step 7.5 picker asks Full or Short and sets `state.onlyDevelop`. That
|
|
20
|
-
* answer arrives long after the tracker boots at Step -1, so
|
|
21
|
-
*
|
|
22
|
-
*
|
|
20
|
+
* answer arrives long after the tracker boots at Step -1, so v17.5.0 splits
|
|
21
|
+
* registration for the two modes that ask it (`full`, `local`): Phase 0 at Step -1,
|
|
22
|
+
* the rest once depth has named them. Everything else still registers its whole
|
|
23
|
+
* set up front, and a generated phase set is per-COMMAND, never per-depth.
|
|
23
24
|
*
|
|
24
25
|
* Companion smoke `smoke-mode-dispatch-drift.sh` regenerates the section for
|
|
25
26
|
* each mode file, diffs against the on-disk content, and fails on drift.
|
|
@@ -60,10 +61,10 @@ const PHASE_NAMES_NO_TEST = PHASE_NAMES.filter((p) => p !== "5:Test");
|
|
|
60
61
|
*/
|
|
61
62
|
const MODES = {
|
|
62
63
|
autopilot: { phases: PHASE_NAMES_NO_TEST, local: false, autopilot: true },
|
|
63
|
-
full: { phases: PHASE_NAMES, local: false, autopilot: false },
|
|
64
|
-
local: { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false },
|
|
64
|
+
full: { phases: PHASE_NAMES, local: false, autopilot: false, depth: true },
|
|
65
|
+
local: { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false, depth: true },
|
|
65
66
|
"local-autopilot": { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: true },
|
|
66
|
-
"full-local": { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false },
|
|
67
|
+
"full-local": { phases: PHASE_NAMES_NO_TEST, local: true, autopilot: false, depth: true },
|
|
67
68
|
// Analysis produces a document, not code: no Dev, no Test, and Phase 6
|
|
68
69
|
// publishes instead of committing. Same 8-phase contract, four of them
|
|
69
70
|
// reinterpreted.
|
|
@@ -81,8 +82,6 @@ if (!spec) {
|
|
|
81
82
|
process.exit(2);
|
|
82
83
|
}
|
|
83
84
|
|
|
84
|
-
const phaseLoop = spec.phases.map((p) => `"${p}"`).join(" ");
|
|
85
|
-
|
|
86
85
|
const ALL_PHASE_IDS = PHASE_NAMES.map((p) => p.split(":")[0]);
|
|
87
86
|
const activeIds = spec.phases.map((p) => p.split(":")[0]);
|
|
88
87
|
const skippedIds = ALL_PHASE_IDS.filter((id) => !activeIds.includes(id));
|
|
@@ -104,7 +103,7 @@ const skipNote =
|
|
|
104
103
|
? `${modeLabel} mode does NOT TaskCreate phases ${skippedIds.join("/")} - those are not part of the ${modeLabel} phase set (\`${spec.phases.join(" ")}\`). Only register tiles for the active set.`
|
|
105
104
|
: `${modeLabel} mode TaskCreates all 8 phases (no phase is skipped).`;
|
|
106
105
|
|
|
107
|
-
const orderingNote = `**All TaskCreate calls fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
|
|
106
|
+
const orderingNote = `**All TaskCreate calls in a batch fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${spec.depth ? `Phase 0 at Step -1, then the rest in ascending order at Step 7.5 (${phaseSequence} minus whatever the depth answer drops)` : phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
|
|
108
107
|
|
|
109
108
|
const localCaveat = spec.local
|
|
110
109
|
? `\n> **Local mode:** no worktree is created, work happens on the current branch. Phase 0 Init still calls \`init\` - the \`--local\` flag is stored in tracker-state.json, and \`:resume\` restores the correct CWD.\n`
|
|
@@ -114,6 +113,32 @@ const banner = spec.autopilot
|
|
|
114
113
|
? `\n> **Autopilot mode:** user confirmations are skipped. The tracker is still mandatory - autopilot agent calls cannot skip it; skipping breaks \`smoke-tracker-contract.sh\`.\n`
|
|
115
114
|
: "";
|
|
116
115
|
|
|
116
|
+
// Registration shape. A mode that asks the depth question does not know its phase
|
|
117
|
+
// set at Step -1 (the answer needs taskType, which needs the fetched issue and the
|
|
118
|
+
// branch), so it registers Phase 0 there and the rest at Step 7.5. Drawing eight
|
|
119
|
+
// tiles beside the question that decides whether two of them run is the failure this
|
|
120
|
+
// split exists to remove.
|
|
121
|
+
const fullSet = spec.phases.filter((p) => p !== "0:Init");
|
|
122
|
+
const shortSet = fullSet.filter((p) => Number(p.split(":")[0]) >= 3);
|
|
123
|
+
const loopFor = (list) =>
|
|
124
|
+
`for p in ${list.map((x) => `"${x}"`).join(" ")}; do\n bash $HOME/.claude/scripts/phase-tracker.sh add "\${p%%:*}" "\${p#*:}"\ndone`;
|
|
125
|
+
|
|
126
|
+
const initLoop = spec.depth
|
|
127
|
+
? 'bash $HOME/.claude/scripts/phase-tracker.sh add 0 "Init"\nbash $HOME/.claude/scripts/phase-tracker.sh tiles'
|
|
128
|
+
: `${loopFor(spec.phases)}`;
|
|
129
|
+
|
|
130
|
+
const deferredBlock = spec.depth
|
|
131
|
+
? `
|
|
132
|
+
# Phase 0 Step 7.5, immediately after the depth answer - the first moment this
|
|
133
|
+
# mode knows its phase set. Full:
|
|
134
|
+
${loopFor(fullSet)}
|
|
135
|
+
# Short (Analysis and Planning are not run, so they get no tile at all):
|
|
136
|
+
${loopFor(shortSet)}
|
|
137
|
+
# Then the widget, narrowed to the phases that do not have a tile yet:
|
|
138
|
+
bash $HOME/.claude/scripts/phase-tracker.sh tiles --new
|
|
139
|
+
`
|
|
140
|
+
: "";
|
|
141
|
+
|
|
117
142
|
const out = `## Required: Phase Tracker Contract
|
|
118
143
|
|
|
119
144
|
**The phase tracker is mandatory** - the agent cannot skip it. Full spec: [\`$HOME/.claude/multi-agent-refs/tracker-contract.md\`]($HOME/.claude/multi-agent-refs/tracker-contract.md).
|
|
@@ -126,11 +151,9 @@ Two channels run in parallel at every phase boundary:
|
|
|
126
151
|
\`\`\`bash
|
|
127
152
|
# Phase 0, very first shell call (every CLI):
|
|
128
153
|
bash $HOME/.claude/scripts/phase-tracker.sh init "$TASK_ID"
|
|
129
|
-
|
|
130
|
-
bash $HOME/.claude/scripts/phase-tracker.sh add "\${p%%:*}" "\${p#*:}"
|
|
131
|
-
done
|
|
154
|
+
${initLoop}
|
|
132
155
|
bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
|
|
133
|
-
|
|
156
|
+
${deferredBlock}
|
|
134
157
|
# Every phase boundary (every CLI):
|
|
135
158
|
bash $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress|completed|failed|skipped
|
|
136
159
|
|
|
@@ -142,11 +165,11 @@ bash $HOME/.claude/scripts/phase-tracker.sh tokens <N> <in> <out> [cached]
|
|
|
142
165
|
|
|
143
166
|
In Claude Code the agent MUST also drive the native TaskList widget so the user sees a sticky phase tile stack - this is the only progress signal Claude Code surfaces. Skipping these calls is the #1 source of "I don't see any phases" complaints.
|
|
144
167
|
|
|
145
|
-
**TaskCreate ordering (strict)**: All TaskCreate calls fire in strict phase-number order BEFORE any TaskUpdate is
|
|
168
|
+
**TaskCreate ordering (strict)**: All TaskCreate calls in a registration batch fire in strict phase-number order BEFORE any TaskUpdate in that batch, and a later batch only ever appends phases numbered above everything already registered.${spec.depth ? " This mode registers in two batches (Step -1, then Step 7.5), so `tiles --new` narrows the second one and the Phase 0 tile is never created twice." : ""} The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. \`1 ✓ · 2 ✓ · 4 ✓ · 0 ▶ · 3 ☐\`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".
|
|
146
169
|
|
|
147
170
|
\`\`\`text
|
|
148
|
-
#
|
|
149
|
-
for each phase in ${spec.phases.join(", ")}:
|
|
171
|
+
# Register one tile per phase, capture the taskId, persist it:
|
|
172
|
+
for each phase in ${spec.depth ? `0:Init at Step -1, then ${fullSet.join(", ")} (Full) or ${shortSet.join(", ")} (Short) at Step 7.5` : spec.phases.join(", ")}:
|
|
150
173
|
TaskCreate({ subject: "Phase <N>: <Name>", activeForm: "<doing-form>" })
|
|
151
174
|
-> returns taskId
|
|
152
175
|
bash $HOME/.claude/scripts/phase-tracker.sh meta <N> tasklist_id "<taskId>"
|
|
@@ -681,8 +681,17 @@ EOF
|
|
|
681
681
|
# a subject is a task title and travels through surfaces that are not a terminal.
|
|
682
682
|
subjects() {
|
|
683
683
|
need_jq
|
|
684
|
-
local state want="${1:-}"
|
|
684
|
+
local state want="${1:-}" new_only=""
|
|
685
|
+
# `--new` narrows to phases that have no tile yet, which is how a deferred
|
|
686
|
+
# registration batch (Phase 0 at Step -1, the rest once depth is known) avoids
|
|
687
|
+
# asking the host to create a tile it already has. A phase whose tasklist_id was
|
|
688
|
+
# never recorded has, as far as anything can tell, no tile.
|
|
689
|
+
if [ "$want" = "--new" ]; then new_only=1; want=""; fi
|
|
685
690
|
state=$(load_state)
|
|
691
|
+
local new_ids=""
|
|
692
|
+
if [ -n "$new_only" ]; then
|
|
693
|
+
new_ids=" $(echo "$state" | jq -r '[.phases[]? | select(((.meta.tasklist_id // "") | tostring) == "") | .id] | join(" ")') "
|
|
694
|
+
fi
|
|
686
695
|
local prices_json='{"prices":{}}'
|
|
687
696
|
if [ -f "$COST_TABLE" ]; then
|
|
688
697
|
prices_json=$(cat "$COST_TABLE" 2>/dev/null) || prices_json='{"prices":{}}'
|
|
@@ -704,6 +713,9 @@ subjects() {
|
|
|
704
713
|
while IFS=$'\x1f' read -r pid pname pstatus p_start p_end pmodel ptok pusd; do
|
|
705
714
|
[ -n "$pid" ] || continue
|
|
706
715
|
[ -z "$want" ] || [ "$want" = "$pid" ] || continue
|
|
716
|
+
if [ -n "$new_only" ]; then
|
|
717
|
+
case "$new_ids" in *" $pid "*) ;; *) continue ;; esac
|
|
718
|
+
fi
|
|
707
719
|
local line="Phase $pid $pname"
|
|
708
720
|
[ -n "$pmodel" ] && line="$line - $pmodel"
|
|
709
721
|
local s_ep e_ep el=""
|
|
@@ -755,10 +767,33 @@ subjects() {
|
|
|
755
767
|
# already does for a different reason.
|
|
756
768
|
tiles_script() {
|
|
757
769
|
need_jq
|
|
770
|
+
local new_only="${1:-}"
|
|
758
771
|
local state; state=$(load_state)
|
|
759
772
|
local has_subs
|
|
760
773
|
has_subs=$(echo "$state" | jq '[.phases[]?.subs[]?] | length')
|
|
761
774
|
|
|
775
|
+
# A list carrying sub-steps has to be rebuilt whole, so the rebuild wins over a
|
|
776
|
+
# narrowed batch: appending to it would land the new rows after Phase 7.
|
|
777
|
+
[ "${has_subs:-0}" -gt 0 ] && new_only=""
|
|
778
|
+
|
|
779
|
+
if [ -n "$new_only" ]; then
|
|
780
|
+
local pending_new
|
|
781
|
+
pending_new=$(subjects --new)
|
|
782
|
+
if [ -z "$pending_new" ]; then
|
|
783
|
+
printf 'Every registered phase already has a tile - nothing to create.\n'
|
|
784
|
+
return 0
|
|
785
|
+
fi
|
|
786
|
+
printf 'REQUIRED - create one native tile per phase below, in this exact order.
|
|
787
|
+
'
|
|
788
|
+
printf 'Tiles already created keep their place; every phase here is numbered
|
|
789
|
+
'
|
|
790
|
+
printf 'above them, so the widget stays in phase order.
|
|
791
|
+
|
|
792
|
+
'
|
|
793
|
+
printf '%s\n' "$pending_new" | sed 's/^/ TaskCreate(subject: "/; s/$/")/'
|
|
794
|
+
return 0
|
|
795
|
+
fi
|
|
796
|
+
|
|
762
797
|
if [ "${has_subs:-0}" -gt 0 ]; then
|
|
763
798
|
printf 'The tile list has sub-steps now, and the widget orders by CREATION.
|
|
764
799
|
'
|
|
@@ -1346,6 +1381,8 @@ EOF
|
|
|
1346
1381
|
|
|
1347
1382
|
tiles)
|
|
1348
1383
|
need_jq
|
|
1384
|
+
tiles_new=""
|
|
1385
|
+
[ "${1:-}" = "--new" ] && tiles_new=1
|
|
1349
1386
|
tiles_state=$(load_state)
|
|
1350
1387
|
tiles_count=$(echo "$tiles_state" | jq '[.phases[]?] | length')
|
|
1351
1388
|
[ "${tiles_count:-0}" -gt 0 ] || {
|
|
@@ -1368,7 +1405,7 @@ EOF
|
|
|
1368
1405
|
# tools it has. And the fallback is not enough on its own: a user looking
|
|
1369
1406
|
# at a missing widget needs the one command that brings it back, which is
|
|
1370
1407
|
# why the opt-in is printed next to it.
|
|
1371
|
-
tiles_script
|
|
1408
|
+
tiles_script "$tiles_new"
|
|
1372
1409
|
echo
|
|
1373
1410
|
echo "At every phase boundary re-run \`phase-tracker.sh subjects <id>\` and"
|
|
1374
1411
|
echo "pass that line as the subject of the TaskUpdate. The subject is the only"
|
|
@@ -20,6 +20,20 @@
|
|
|
20
20
|
* `agent-state.json` is the only durable evidence they happened - so the gate asserts
|
|
21
21
|
* their output too, not just `taskType`.
|
|
22
22
|
*
|
|
23
|
+
* A third run got past that, because `baseBranch` and `baseFetchStatus` can be filled
|
|
24
|
+
* in by a branch that was never chosen. The remote had one PR-targetable branch, the
|
|
25
|
+
* one-option `AskUserQuestion` was refused by the host (its schema needs two), and the
|
|
26
|
+
* run announced "only candidate, continuing with it" and carried on. The branch was
|
|
27
|
+
* right; nothing was asked. `baseBranchSource` is the field that separates those two,
|
|
28
|
+
* and the dev-context picker never ran at all - no `siblings`, so Phase 4's parity
|
|
29
|
+
* cross-check had nothing to read and could not tell an empty answer from no answer.
|
|
30
|
+
*
|
|
31
|
+
* A fourth showed the same shape one layer down: `git fetch origin` failed on a
|
|
32
|
+
* restricted network, `git branch -r` printed the remote-tracking cache anyway, and a
|
|
33
|
+
* weeks-old local list was presented as the remote's answer. Degrading to local refs is
|
|
34
|
+
* correct; reporting them as remote is not, so `baseBranchEvidence.refProvenance` has to
|
|
35
|
+
* agree with `baseFetchStatus`.
|
|
36
|
+
*
|
|
23
37
|
* A phase that reports success without its output is worse than one that fails:
|
|
24
38
|
* every later phase then reasons from a field that is not there. So this is a
|
|
25
39
|
* gate, not a lint - the spec already said what to write, and prose alone did
|
|
@@ -157,6 +171,120 @@ export function evaluate(state, extraInput = "") {
|
|
|
157
171
|
);
|
|
158
172
|
}
|
|
159
173
|
|
|
174
|
+
// Which rule decided the base branch. `asked` and `input` are the only two an
|
|
175
|
+
// interactive run can honestly record: `remembered`, `default` and `derived` are the
|
|
176
|
+
// autopilot resolutions, and an interactive run that reaches for them has skipped its
|
|
177
|
+
// picker. This is the assertion `baseBranch` alone cannot make - a branch announced in
|
|
178
|
+
// prose and a branch chosen by the user leave the same value behind. A branch the run
|
|
179
|
+
// derived from the issue and the user then confirmed is still `asked`; the derivation
|
|
180
|
+
// lives in `baseBranchEvidence`, which is a record, not a permission.
|
|
181
|
+
const BRANCH_SOURCES = ["asked", "input", "remembered", "default", "derived"];
|
|
182
|
+
const branchSource =
|
|
183
|
+
typeof state.baseBranchSource === "string" ? state.baseBranchSource.trim() : "";
|
|
184
|
+
if (baseBranch && !BRANCH_SOURCES.includes(branchSource)) {
|
|
185
|
+
failures.push(
|
|
186
|
+
`agent-state.json has baseBranchSource="${branchSource || "<unset>"}"; Step 3 must ` +
|
|
187
|
+
`record one of ${BRANCH_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
|
|
188
|
+
`branch the user chose from one the run picked and announced.`,
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
const isAutopilot = state.autopilot === true;
|
|
192
|
+
const AUTOPILOT_ONLY_SOURCES = ["remembered", "default", "derived"];
|
|
193
|
+
if (!isAutopilot && AUTOPILOT_ONLY_SOURCES.includes(branchSource)) {
|
|
194
|
+
failures.push(
|
|
195
|
+
`baseBranchSource="${branchSource}" on an interactive run. Those three are autopilot ` +
|
|
196
|
+
`resolutions; an interactive run asks (Step 3 is not skippable) or takes the base ` +
|
|
197
|
+
`from the task reference. A one-candidate filter is still asked, with a second ` +
|
|
198
|
+
`option - picker-contract.md, "Two options or it is not a question".`,
|
|
199
|
+
);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
// What the base branch was chosen from, and what that list was worth.
|
|
203
|
+
//
|
|
204
|
+
// Two separate silent failures live here. The first: `git fetch origin` can fail on a
|
|
205
|
+
// restricted network while `git branch -r` still prints a full, confident list - the
|
|
206
|
+
// remote-tracking cache - so a weeks-old local guess gets presented as the remote's
|
|
207
|
+
// answer. Falling back to local refs is correct; not saying so is not. The second:
|
|
208
|
+
// `derived` claims the issue named the branch, and a claim with no evidence behind it
|
|
209
|
+
// is `default` wearing a hat.
|
|
210
|
+
const evidence =
|
|
211
|
+
state.baseBranchEvidence && typeof state.baseBranchEvidence === "object"
|
|
212
|
+
? state.baseBranchEvidence
|
|
213
|
+
: null;
|
|
214
|
+
const DEGRADED_FETCH = ["cached-stale", "local-branch"];
|
|
215
|
+
if (DEGRADED_FETCH.includes(fetchStatus)) {
|
|
216
|
+
if (!evidence) {
|
|
217
|
+
failures.push(
|
|
218
|
+
`baseFetchStatus="${fetchStatus}" but state.baseBranchEvidence is absent. A run whose ` +
|
|
219
|
+
`fetch failed listed its branches from local refs; the record of that is what stops a ` +
|
|
220
|
+
`local-only guess being read afterwards as the remote's answer.`,
|
|
221
|
+
);
|
|
222
|
+
} else if (evidence.refProvenance !== "local") {
|
|
223
|
+
failures.push(
|
|
224
|
+
`baseFetchStatus="${fetchStatus}" but baseBranchEvidence.refProvenance=` +
|
|
225
|
+
`"${evidence.refProvenance || "<unset>"}". The fetch failed, so the candidate list came ` +
|
|
226
|
+
`from local refs and must say so - "remote" here is the silent degradation this field exists to catch.`,
|
|
227
|
+
);
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
if (branchSource === "derived") {
|
|
231
|
+
const ISSUE_DERIVED = new Set(["issue-version", "linked-release"]);
|
|
232
|
+
const cands = evidence && Array.isArray(evidence.candidates) ? evidence.candidates : [];
|
|
233
|
+
const chosen = cands.find((c) => c && c.branch === baseBranch);
|
|
234
|
+
const hasIssueEvidence =
|
|
235
|
+
chosen &&
|
|
236
|
+
Array.isArray(chosen.evidence) &&
|
|
237
|
+
chosen.evidence.some((e) => e && ISSUE_DERIVED.has(e.kind));
|
|
238
|
+
if (!hasIssueEvidence) {
|
|
239
|
+
failures.push(
|
|
240
|
+
`baseBranchSource="derived" but baseBranchEvidence carries no issue-version or ` +
|
|
241
|
+
`linked-release evidence for "${baseBranch}". "Derived" names a specific claim - the ` +
|
|
242
|
+
`issue's version field or a linked release issue pointed at this branch - and without ` +
|
|
243
|
+
`that record it is the sort-order default under a better name.`,
|
|
244
|
+
);
|
|
245
|
+
}
|
|
246
|
+
if (evidence && evidence.ambiguous === true) {
|
|
247
|
+
failures.push(
|
|
248
|
+
`baseBranchSource="derived" with baseBranchEvidence.ambiguous=true. Two or more ` +
|
|
249
|
+
`candidates tied at the top score; autopilot does not break a tie by picking one.`,
|
|
250
|
+
);
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// Step 5b decides where the branch lives, and `localMode` alone cannot say
|
|
255
|
+
// whether anyone decided: `false` is both "the user chose a worktree" and
|
|
256
|
+
// "nothing asked and the default stood". Same shape as baseBranchSource, and
|
|
257
|
+
// the same reason - an autopilot run resolves it rather than asking, so
|
|
258
|
+
// `autopilot` is a legal source there and nowhere else.
|
|
259
|
+
const WORKSPACE_SOURCES = ["asked", "command", "autopilot"];
|
|
260
|
+
const workspaceSource =
|
|
261
|
+
typeof state.workspaceSource === "string" ? state.workspaceSource.trim() : "";
|
|
262
|
+
if (!WORKSPACE_SOURCES.includes(workspaceSource)) {
|
|
263
|
+
failures.push(
|
|
264
|
+
`agent-state.json has workspaceSource="${workspaceSource || "<unset>"}"; Step 5b must ` +
|
|
265
|
+
`record one of ${WORKSPACE_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
|
|
266
|
+
`worktree the user chose from one no question was asked about.`,
|
|
267
|
+
);
|
|
268
|
+
}
|
|
269
|
+
if (!isAutopilot && workspaceSource === "autopilot") {
|
|
270
|
+
failures.push(
|
|
271
|
+
`workspaceSource="autopilot" on an interactive run. Autopilot resolves the workspace ` +
|
|
272
|
+
`to a worktree because an unattended commit in the user's own checkout is what ` +
|
|
273
|
+
`worktrees prevent; an interactive run asks (Step 5b) or is told by :local / --local.`,
|
|
274
|
+
);
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
// Step 2b's dev-context picker writes siblings[], empty included. The empty array is
|
|
278
|
+
// the record that it ran; absent, a multi-repo task silently became a single-repo one
|
|
279
|
+
// and Phase 4's parity cross-check lost its fourth counterpart source.
|
|
280
|
+
if (!Array.isArray(state.siblings)) {
|
|
281
|
+
failures.push(
|
|
282
|
+
"agent-state.json has no siblings array. Step 2b runs the dev-context picker on " +
|
|
283
|
+
"every input type and persists the result, `[]` included; an absent field means " +
|
|
284
|
+
"the picker never ran, so extra repos and read-only counterparts were never offered.",
|
|
285
|
+
);
|
|
286
|
+
}
|
|
287
|
+
|
|
160
288
|
// Worktree isolation is a standing rule: never develop in the primary checkout.
|
|
161
289
|
const worktree = typeof state.worktreePath === "string" ? state.worktreePath.trim() : "";
|
|
162
290
|
const projectRoot = typeof state.projectRoot === "string" ? state.projectRoot.trim() : "";
|
|
@@ -67,6 +67,79 @@ export function unsafePath(p) {
|
|
|
67
67
|
return null;
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
/**
|
|
71
|
+
* Extensions a markdown citation may carry.
|
|
72
|
+
*
|
|
73
|
+
* A list, not a pattern, and that direction is deliberate: the cost of a
|
|
74
|
+
* missing extension is one citation going unchecked, while the cost of an open
|
|
75
|
+
* pattern is the gate failing a correct document over a hostname. Add to it
|
|
76
|
+
* when a real repository file is missed - never widen it to a wildcard.
|
|
77
|
+
*
|
|
78
|
+
* `com`, `net`, `io`, `dev`, `tr` are absent on purpose.
|
|
79
|
+
*/
|
|
80
|
+
export const SOURCE_EXTENSIONS = new Set([
|
|
81
|
+
"swift",
|
|
82
|
+
"kt",
|
|
83
|
+
"kts",
|
|
84
|
+
"java",
|
|
85
|
+
"m",
|
|
86
|
+
"mm",
|
|
87
|
+
"h",
|
|
88
|
+
"hpp",
|
|
89
|
+
"c",
|
|
90
|
+
"cc",
|
|
91
|
+
"cpp",
|
|
92
|
+
"js",
|
|
93
|
+
"mjs",
|
|
94
|
+
"cjs",
|
|
95
|
+
"jsx",
|
|
96
|
+
"ts",
|
|
97
|
+
"tsx",
|
|
98
|
+
"py",
|
|
99
|
+
"rb",
|
|
100
|
+
"go",
|
|
101
|
+
"rs",
|
|
102
|
+
"php",
|
|
103
|
+
"sh",
|
|
104
|
+
"bash",
|
|
105
|
+
"zsh",
|
|
106
|
+
"pl",
|
|
107
|
+
"sql",
|
|
108
|
+
"json",
|
|
109
|
+
"yml",
|
|
110
|
+
"yaml",
|
|
111
|
+
"toml",
|
|
112
|
+
"xml",
|
|
113
|
+
"plist",
|
|
114
|
+
"xcconfig",
|
|
115
|
+
"pbxproj",
|
|
116
|
+
"gradle",
|
|
117
|
+
"properties",
|
|
118
|
+
"cfg",
|
|
119
|
+
"ini",
|
|
120
|
+
"env",
|
|
121
|
+
"md",
|
|
122
|
+
"mdx",
|
|
123
|
+
"txt",
|
|
124
|
+
"csv",
|
|
125
|
+
"tsv",
|
|
126
|
+
"strings",
|
|
127
|
+
"stringsdict",
|
|
128
|
+
"html",
|
|
129
|
+
"css",
|
|
130
|
+
"scss",
|
|
131
|
+
"vue",
|
|
132
|
+
"svelte",
|
|
133
|
+
"graphql",
|
|
134
|
+
"proto",
|
|
135
|
+
"entitlements",
|
|
136
|
+
"storyboard",
|
|
137
|
+
"xib",
|
|
138
|
+
"lock",
|
|
139
|
+
"gemspec",
|
|
140
|
+
"podspec",
|
|
141
|
+
]);
|
|
142
|
+
|
|
70
143
|
/**
|
|
71
144
|
* Pull `{file, line}` pairs out of whatever was handed over.
|
|
72
145
|
*
|
|
@@ -124,6 +197,14 @@ export function extract(raw) {
|
|
|
124
197
|
// Fenced blocks are skipped for the same class of reason: a code sample that
|
|
125
198
|
// happens to contain `foo.js:12` is an illustration, not a claim about this
|
|
126
199
|
// repository.
|
|
200
|
+
//
|
|
201
|
+
// The URL strip below only removes what carries a scheme. A host written
|
|
202
|
+
// bare - `wiki.example.com:8443/x`, and an analysis document is full of
|
|
203
|
+
// them - still looks exactly like a path with an extension and a line
|
|
204
|
+
// number, and this gate would fail the document over `.com`. So the
|
|
205
|
+
// extension has to be one a repository actually contains. This is the same
|
|
206
|
+
// defect that was already fixed once here for scheme-carrying URLs; the bare
|
|
207
|
+
// form survived it.
|
|
127
208
|
let fenced = false;
|
|
128
209
|
raw.split("\n").forEach((line, i) => {
|
|
129
210
|
if (/^\s*```/.test(line)) {
|
|
@@ -132,8 +213,9 @@ export function extract(raw) {
|
|
|
132
213
|
}
|
|
133
214
|
if (fenced) return;
|
|
134
215
|
const cleaned = line.replace(/\b[a-z][a-z0-9+.-]*:\/\/\S+/gi, " ");
|
|
135
|
-
for (const m of cleaned.matchAll(/([A-Za-z0-9_./-]+\.[A-Za-z][A-Za-z0-9]*):(\d+)/g)) {
|
|
136
|
-
|
|
216
|
+
for (const m of cleaned.matchAll(/([A-Za-z0-9_./-]+\.([A-Za-z][A-Za-z0-9]*)):(\d+)/g)) {
|
|
217
|
+
if (!SOURCE_EXTENSIONS.has(m[2].toLowerCase())) continue;
|
|
218
|
+
push(m[1], Number(m[3]), `line ${i + 1}`);
|
|
137
219
|
}
|
|
138
220
|
});
|
|
139
221
|
return out;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schemaVersion": "1.0.0",
|
|
3
|
-
"generatedAt": "2026-09-
|
|
3
|
+
"generatedAt": "2026-09-15T12:49:28Z",
|
|
4
4
|
"skillCount": 212,
|
|
5
5
|
"entries": [
|
|
6
6
|
{
|
|
@@ -237,7 +237,7 @@
|
|
|
237
237
|
},
|
|
238
238
|
{
|
|
239
239
|
"path": "shared/core/multi-agent/SKILL.md",
|
|
240
|
-
"sha256": "
|
|
240
|
+
"sha256": "faac7ceb535d18778efe0ea91c19534c8e076bc709e7b1dbc567f5e71692bbbd"
|
|
241
241
|
},
|
|
242
242
|
{
|
|
243
243
|
"path": "shared/external/accessibility-compliance-accessibility-audit/SKILL.md",
|
|
@@ -316,7 +316,7 @@ All TaskCreate calls fire in strict phase-number order BEFORE any TaskUpdate is
|
|
|
316
316
|
- `multi-agent-local`, `multi-agent-autopilot`, `multi-agent-local-autopilot`: 0 → 1 → 2 → 3 → 4 → 6 → 7 (the interactive Phase 5 gate needs a worktree checkout and an attended run)
|
|
317
317
|
- `multi-agent-analysis`: 0 → 1 → 2 → 4 → 6 → 7 (no code, so no Dev and no Test)
|
|
318
318
|
|
|
319
|
-
Depth
|
|
319
|
+
Depth decides WHICH of the command's set is registered. The tracker boots at Step -1, long before the depth question can be asked, so it registers Phase 0 alone and the remaining tiles are created at Step 7.5 once the answer is known - a Short run never draws an Analysis or Planning tile.
|
|
320
320
|
|
|
321
321
|
Full contract: `refs/tracker-contract.md` section "TaskCreate ordering (strict)".
|
|
322
322
|
|