@mmerterden/multi-agent-pipeline 16.29.0 → 16.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +82 -0
  2. package/docs/features.md +14 -0
  3. package/install/_common.mjs +45 -4
  4. package/package.json +1 -1
  5. package/pipeline/commands/multi-agent/design-check/SKILL.md +6 -5
  6. package/pipeline/commands/multi-agent/help/SKILL.md +13 -12
  7. package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
  8. package/pipeline/commands/multi-agent/sync/SKILL.md +3 -4
  9. package/pipeline/lib/credential-inventory.sh +1 -0
  10. package/pipeline/lib/repo-hygiene.sh +164 -0
  11. package/pipeline/lib/vercel-deploy.sh +41 -22
  12. package/pipeline/multi-agent-refs/channels/pr.md +37 -1
  13. package/pipeline/multi-agent-refs/features/doctor.md +12 -0
  14. package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
  15. package/pipeline/multi-agent-refs/features/visual-evidence.md +103 -20
  16. package/pipeline/multi-agent-refs/keychain.md +1 -0
  17. package/pipeline/multi-agent-refs/knowledge.md +1 -1
  18. package/pipeline/multi-agent-refs/phases/phase-0-init.md +36 -7
  19. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  20. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +14 -2
  21. package/pipeline/multi-agent-refs/phases/phase-5-test.md +12 -2
  22. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +23 -0
  23. package/pipeline/preferences-template.json +2 -1
  24. package/pipeline/schemas/agent-state.schema.json +79 -1
  25. package/pipeline/schemas/prefs.schema.json +24 -1
  26. package/pipeline/schemas/token-budget.json +10 -10
  27. package/pipeline/scripts/bulk-read.sh +13 -5
  28. package/pipeline/scripts/capture-evidence.sh +170 -5
  29. package/pipeline/scripts/doctor.mjs +56 -1
  30. package/pipeline/scripts/evidence-gate.mjs +31 -2
  31. package/pipeline/scripts/gc-tmp.sh +30 -0
  32. package/pipeline/scripts/gc-worktrees.sh +32 -10
  33. package/pipeline/scripts/offload-ref.sh +13 -5
  34. package/pipeline/scripts/probe-evidence-capability.sh +250 -0
  35. package/pipeline/scripts/purge.sh +11 -0
  36. package/pipeline/scripts/run-ui-tests.sh +380 -0
  37. package/pipeline/scripts/worktree-finalize.sh +9 -0
  38. package/pipeline/skills/.skill-manifest.json +19 -11
  39. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +10 -1
  40. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +2 -1
@@ -1,11 +1,11 @@
1
1
  {
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/token-budget.json",
4
- "description": "Per-phase token budget for lazy-loaded pipeline docs. Enforced by smoke-token-budget.sh.",
4
+ "description": "Per-phase token budget for lazy-loaded pipeline docs. Enforced by smoke-token-budget.sh. Raised in 16.30.0 for the four phases that gained a visual-evidence step (probe + depth question in 0, recording in 3 and 5, host resolution in 6): the docs were compressed first and these numbers are the measured residual, not headroom.",
5
5
  "phases": {
6
6
  "phase-0-init": {
7
- "max_tokens": 12500,
8
- "warn_tokens": 11900
7
+ "max_tokens": 13000,
8
+ "warn_tokens": 12400
9
9
  },
10
10
  "phase-1-analysis": {
11
11
  "max_tokens": 4600,
@@ -16,26 +16,26 @@
16
16
  "warn_tokens": 5650
17
17
  },
18
18
  "phase-3-dev": {
19
- "max_tokens": 8950,
20
- "warn_tokens": 7900
19
+ "max_tokens": 9250,
20
+ "warn_tokens": 8600
21
21
  },
22
22
  "phase-4-review": {
23
23
  "max_tokens": 15150,
24
24
  "warn_tokens": 12950
25
25
  },
26
26
  "phase-5-test": {
27
- "max_tokens": 2900,
28
- "warn_tokens": 2550
27
+ "max_tokens": 3050,
28
+ "warn_tokens": 2850
29
29
  },
30
30
  "phase-6-commit": {
31
- "max_tokens": 6150,
32
- "warn_tokens": 5450
31
+ "max_tokens": 6550,
32
+ "warn_tokens": 6100
33
33
  },
34
34
  "phase-7-report": {
35
35
  "max_tokens": 6350,
36
36
  "warn_tokens": 5600
37
37
  }
38
38
  },
39
- "total_max_tokens": 60750,
39
+ "total_max_tokens": 62400,
40
40
  "note": "Token estimate = ceil(chars / 4). Per-phase budget rule: warn = current+10% (rounded to nearest 50), max = current+25%. Gives ~6 edit cycles of headroom before warn trips - intentionally quiet under normal maintenance, loud when a phase grows unusually. Only the active phase is loaded (lazy). Recalibrated at v10.0.0 after the validator/consistency/simplifier/lesson gate contracts landed in phases 1-4. Recalibrated again at v10.9.0 after the verify-by-test (Phase 4 Step 3.7), update-check (Phase 0 Step 0.6), immutable-test (Phase 3 GREEN) and redTests re-entry contracts landed - Step 3.7 prose was compressed to a pointer into refs/features/verify-by-test.md before the recalibration. Total bumped 50000 -> 51000 at v12.5.0 after the worktree residue/traversal-prune contract (Phase 0 + Phase 5 heal) and the Reflexion causal-diagnosis contract (Phase 4 lesson memory) landed; the prose was compressed first (161 tokens reclaimed) and every per-phase max still passes - only the aggregate needed room. Recalibrated again at v13.6.0 after the install-relative path correction: an instruction that names `pipeline/scripts/x` resolves only from a repo checkout, and a run happens in the user's worktree, so 157 references across these docs moved to `$HOME/.claude/...` at +5 bytes each - 196 tokens of pure correctness cost. Same discipline as before: prose was compressed FIRST (149 tokens reclaimed, by pointing Phase 1's Figma tier table at the Phase 0 probe that already resolved it and Phase 4's Codex constraints at the always-loaded AGENTS.md block), and only then were the budgets moved. Five warn lines had been permanently amber, which makes the amber tier useless as a signal, so every warn was reset to the documented current+10% and the four maxes that the new warn would have collided with were reset to current+25%. Aggregate 51000 -> 51500. Total bumped 51500 -> 52200 at v14.0.0 after Phase 4 Review entered the four --dev mode phase sets and the criteria-resolution contract (Step 1.78) landed. Same discipline as every prior bump: prose was compressed FIRST, 820 tokens reclaimed, before the number moved. Two of those compressions are structural rather than cosmetic - the hardcoded SwiftUI interaction list in Step 1.5 and the SwiftUI convention paragraph in Step 2.8 were transcriptions of rules that now live in a scoped registry, so keeping them here would have re-created the drift this release exists to remove, and the third moved the Step 1.78 full contract into refs/features/skill-conformance.md leaving a pointer. What remains is contract text that cannot be inferred: the manifest's four consumer-visible parts, the conformance checklist the reviewers must return, and the fail-closed semantics. Every per-phase max still passes (phase-4 12405/14750); only the aggregate needed room. Total bumped 52200 -> 52700 at v14.1.0 after two more contracts landed: stack skill routing (Phase 3 pre-flight step 9) and worktree finalize (Phase 6 step 9). Compression came first, as always, and twice: 224 tokens out of Phase 3 by pointing its criteria-ledger and routing steps at their feature files instead of restating them, and 190 out of Phase 6 by moving the finalize contract into refs/features/worktree-finalize.md and leaving the invocation plus the exit-3 semantics. Both new contracts follow the pattern the earlier ones set: the phase doc carries the call and the decision, the feature file carries the reasoning, and the feature files are outside this budget because it loops only the eight phase-N-* keys. Every per-phase max still passes (phase-3 7677/8950, phase-6 5223/6150 and both under warn); only the aggregate needed room. Total bumped 52700 -> 52750 for the Phase 0 Step 3 branch-persistence correction: the step wrote the legacy `projects[].branches` while the TTL filter two sections below read `global.recentBranches`, and both spots named a `{name, lastUsed}` shape the schema rejects (`branch` required, `additionalProperties: false`), so the recent-branch picker option could never populate and a literal implementation would have failed prefs validation. Naming the right target, the right key and the legacy field to avoid costs 41 tokens over the one line it replaces. Compression came first and was applied three times to the replacement text itself, from 120 tokens down to 66, by moving the rationale out of the phase doc entirely: the reasoning now lives where it is enforced, in the migrate-prefs carry-forward comment and the smoke-pref-migration f7 block, leaving the phase doc with only the instruction. 50 was the smallest step that clears it; phase-0-init sits at 10893/12400, far under its own max, so this is purely an aggregate ceiling. v15.0.0: total 52750 -> 53100, the stack-skill tables in phase-1/2/4 now carry plugin-namespaced names (ai-<stack>-toolkit:<skill>) - functional prefixes, ~170 tokens. v15.10.0: total 53350 -> 53950 for the memory-recall + context-offload contracts (Phase 1 two-block durable-knowledge injection and its telemetry, Phase 3 build-log offload pipe, Phase 4 ranked prior art, offload pipe and recall telemetry). Compression came first and twice, taking the new prose from 1168 tokens to 580: the reasoning behind the two blocks lives in multi-agent-refs/prompt-assembly.md and the reasoning behind the offload filter lives in the offload-ref.sh header, both outside this budget, so the phase docs carry only the call, the pref that gates it and the one fact an agent cannot infer - that the evidence gate still reads the whole build log, so offloading changes what is read, never what counts as a verified pass. Every per-phase max still passes (phase-3 7985/8950, phase-4 12997/14750); phase-3 and phase-4 crossed their warn lines and are left amber on purpose, because that is the signal that those two docs are the next ones needing structural compression rather than another bump. v15.13.0: total 53950 -> 54050 for the prefs-to-flag bridges. Five settings had shipped declared-but-inert: contextOffload.minLines and .tailLines (fixed in 15.11.0), learningsLedger.maxBriefEntries, and testGap.scanTree and .promoteSeverity - the last two declared in the schema AND implemented as flags in the scanner, with nothing in between reading the pref and passing the flag. Wiring three of them costs the phase docs 94 tokens, which is the wiring itself and not prose: two `--max` substitutions and a three-line GAP_FLAGS block. Compression came first and twice, as always: the rationale that would have sat in phase-5 now lives in the header of smoke-prefs-consumed.sh, the gate that makes this class fail a build instead of shipping, and a `--severity-promote` table row was dropped because the invocation above it now shows the flag and names the pref that triggers it, which the row did not. 100 was the smallest step that clears it. Every per-phase max still passes; phase-3 and phase-4 remain amber on purpose. v15.14.0: total 54050 -> 54400 for the supported-version gate. Phase 0 Step 0.6 stopped being purely advisory: a release can now publish an npm dist-tag `required` that names the oldest runnable version, and below it the run halts instead of nagging. What the phase doc has to carry is the part an agent cannot infer - the third stdout field, that the halt is identical in autopilot, and that the run must NOT continue on the freshly updated install because its docs were already loaded from the old version. Compression came first, as always, and took the new prose from 469 tokens to 337: the rationale for the floor, the exemption list, the fail-open rules and the `npm dist-tag add` recipe all moved to multi-agent-refs/rules.md \"Supported Version Gate\" (loaded by 25 commands, outside this budget) and to the header of require-supported-version.sh, leaving the phase doc with the call, the decision table and the halt. 350 was the smallest step that clears it. Every per-phase max still passes (phase-0-init 11230/12400); phase-3 and phase-4 remain amber on purpose. v15.17.0: total 54400 -> 54900 for the Phase 1 analysis-document step. Phase 2 and Phase 3 pre-flights had BLOCKED on `analysis/<feature>-<platform>.md` since v9.0.0 while nothing produced it, so a full run either aborted at Phase 2 or the model ignored its own BLOCKING contract; Step 4 is the producer. What the phase doc carries is only what cannot be inferred: the when-table (taskType x Figma reference), the four refs in load order, the two artefacts, and that the doc validator fails closed. Compression came first and took the step from 745 tokens to 497: the history of why the gap existed moved to the CHANGELOG, the per-ref one-line descriptions moved into the refs' own headers, and the autopilot carve-out collapsed to one clause. The 17.4k-token analysis engine itself is NOT in this budget - it moved out of commands/ into multi-agent-refs/analysis/{locked,evidence,synthesis,render}.md, loaded on demand, which also took analysis/SKILL.md from 18081 to 5974 tokens and retired its lint grace entry. 500 was the smallest step that clears it; phase-1-analysis sits at 4338/4600 and is amber on purpose, like phase-3 and phase-4. v15.18.0: total 54900 -> 55250 for analysis mode. Three phase docs gained a mode branch that cannot be inferred: Phase 4 reviews a document instead of a diff (validator, the one question reviewers answer, the open-question walk), and Phase 6 publishes instead of committing. Compression came first and was applied twice to the new prose and once to old: the Phase 4 branch went from 320 tokens to 180 and the Phase 6 branch from 190 to 120 by pointing at multi-agent-refs/analysis/{resolve,render}.md, which now hold the walks themselves, and the front-matter parse contract stopped being spelled out in both pre-flights. The analysis engine keeps leaving this budget rather than entering it: intake joined locked/evidence/synthesis/render/resolve in multi-agent-refs/analysis/, which is what let analysis/SKILL.md drop under the 6000 hard cap after its grace entry was retired. 350 was the smallest step that clears it; phase-4 and phase-6 are amber on purpose, as phase-1 and phase-3 already were. v15.20.0: total 55250 -> 55500 for the TDD bridge. Phase 3 pre-flight read the analysis doc's concept table and even said test method names come from it, while nothing read Section 15 - so the RED step invented tests and the analysis test matrix never reached development. Phase 3 step 5b now loads it into state.dev.testPlan[] and Phase 4 step 1.45 cross-checks every planned row against a real test, which is what turns \"analysis quality is output quality\" from a slogan into a finding. Compression came first on both blocks, 300 tokens down to 175, by dropping the enumerated failure modes to one line each and the rationale to one clause; the reasoning lives in the CHANGELOG. 250 was the smallest step that clears it. v15.21.0: total 55500 -> 55800 for the post-analysis confirmation. Phase 2 gained Step 0.9, the last human checkpoint before Phase 3: derived values are shown for confirmation and only Section 20 rows are asked, through the resolve engine that already exists in refs. It belongs here rather than Phase 4 because Phase 4 runs after development, where an answer arrives too late to change anything. Compression came first and twice, 430 tokens down to 250, by collapsing the derived-vs-asked explanation to one sentence each and moving the walk itself to multi-agent-refs/analysis/resolve.md, which Phase 4 and analysis-resolve already mount. 300 was the smallest step that clears it. v15.22.0: total 55800 -> 55900 for the analyst-toolkit hooks. Phase 1 Step 4 now names the two prefs that decide whether a document is produced at all and how deep it goes (forceFull, mode) - the first of those had shipped declared-but-inert and smoke-prefs-consumed caught it - and Phase 4 triage gained one clause: a finding that blames a third-party library asks evidence-github whether it is already open upstream, which turns it into a deferred item with a citation instead of Phase 3 rework on code that is not ours. Compression came first and three times, taking the new prose from 220 tokens to 110, and the Phase 1d evidence contract itself never entered this budget - it lives in multi-agent-refs/analysis/evidence.md beside the phases it belongs to. 100 was the smallest step that clears it, leaving 34 tokens of headroom. phase-4 stays amber and the debt named at v15.10.0 stands: it is the doc that needs structural compression rather than another bump, and the two candidates are the inline triage JSON shape and the 3.4 telemetry block, both of which restate something already authoritative elsewhere. v16.0.0: total 55900 -> 56350 for the depth picker. `--dev` and the four dev-* commands are gone; depth is Phase 0 Step 7.5, which costs phase-0-init a step it did not have. Compression came first and three times, taking the step from 530 tokens to 300: the question wording, the per-taskType recommendation and the mode tables all live in phases/modes.md (outside this budget), so the phase doc carries only what an agent cannot infer - that the step runs after Step 7 and why, who is exempt, that ASK_CHOICE_DEFAULT must be passed explicitly because ask-choice.sh takes the FIRST option on a non-TTY, and that Short flips the Phase 1/2 tiles late rather than pre-marking them. The phase-4 telemetry block named as compression debt at v15.22.0 was collapsed to an emit() helper (-27) and the four dev-* mode files left the tree entirely, but neither offsets a genuinely new phase step. 450 was the smallest step that clears it, leaving 119 tokens of headroom. phase-4 remains amber and its other named candidate, the inline triage JSON shape, was left alone on purpose: it is the prompt the triage agent is handed, not a restatement for readers. v16.2.0: total 56350 -> 56600 for the spec-freshness and reuse-tag contracts. Phase 3 step 3 had compared `state.run.lastAnalysisDigest` since it was written, against a key nothing ever set and that the state schema did not declare, so the staleness branch was unreachable and every run reported fresh by default. Phase 1 now persists the digest and a `base_commit` anchor, and step 3 gained the repo-drift half the digest cannot see: a reused document keeps a matching digest precisely because its evidence inputs did not change, while the code underneath it moved. The second contract is the Section 14 tag reaching development: Phase 2 carries it onto the todo as `sourceTag` and Phase 3 treats it as an instruction, which is what stops a Reuse row from being re-implemented. Compression came first and took the four additions from 380 tokens to 214, by moving every rationale clause out of the phase docs: why the commit anchor exists rather than a digest recomputation lives in this note and the CHANGELOG, and the schema descriptions carry the field semantics. The baseline had 9 tokens of headroom, so no addition of any size could have fit without a bump. 250 was the smallest step that clears it, leaving 45 tokens. phase-3 and phase-4 remain amber. v16.13.0: total 57600 -> 57700 for the code-graph injection and the fable-rung switch. Phase 1 gained Step 2.6 (query the graph, hand Explore a ranked starting set), Phase 7 gained the post-branch graph refresh, and Phase 0 Step 0 gained one line: a prefs switch that resolves every preferredModel: fable persona to opus for the run, which also collapses the Phase 4 Claude Code panel from three reviewers to two. Compression came first and mostly structurally: of roughly 1,630 tokens of new contract text, 1,310 never entered this budget at all - the whole code-graph contract lives in multi-agent-refs/features/code-graph.md (604) and the fable switch's scope table, per-host effects and cost-accounting consequence live in features/model-fallback.md (+707), leaving the phase docs with the call, the pref that gates it and the one fact an agent cannot infer. Phase 4 was compressed on top of that: its TLDR restated the reviewer matrix 270 lines below it, so 36 tokens came back and the doc nets +6 despite carrying two new clauses. One of those clauses is a correction rather than a feature - the consensus rule still said reviewerCount is 2 on Claude Code, which stopped being true when the third reviewer landed in 16.12.0, and the cross-CLI smoke never caught it because it reads the matrix line instead. 100 was the smallest step that clears it, leaving 54 tokens. phase-3 and phase-4 remain amber. v16.17.0: total 57700 -> 57850 for the platform-parity cross-check. Phase 4 gained Step 1.8: when dev-context carries a counterpart app repo, the review compares the change against the other platform on four axes. Compression came first and structurally, as always - of roughly 1,610 tokens of new contract text, 1,490 never entered this budget at all, because the four axes, the file cap, the graph-query recipe, the read-only prohibitions and the rule that an extractor miss may not be reported as an absence all live in multi-agent-refs/platform-parity.md. The step itself was then cut from ~200 tokens to 120 by deleting everything the ref already owns, leaving the trigger, the pointer and the two facts an agent must not infer: the counterpart repo is read-only, and parity findings are never blocking. The baseline had 13 tokens of headroom, so no addition of any size could have fit without a bump. 150 was the smallest step that clears it, leaving 35 tokens. phase-3 and phase-4 remain amber, and phase-4's structural-compression debt still stands. v16.20.0: phase-4-review max 14750 -> 15150 and total 58250 -> 60250 for the cross-round review delta, the scope self-check handoff and the circuit-breaker wiring. Compression came first and structurally: of roughly 3,900 tokens of new contract text, 2,700 never entered this budget at all - the previous-round-findings block, the scope-self-check block, the Step 3.8 state merge, telemetry and picker wording live in multi-agent-refs/features/review-delta.md, and the scope-check record rules and consumers in features/scope-check.md - so the phase docs carry the call, the pref that gates it and the exit table. The Phase 3 stability rule and the trigger-3 write were cut twice more before the bump; phase-3 stays under its max (8692/8950). Phase 4 is the first per-phase max raised since v10.9.0: the doc gained three steps that cannot be inferred (a per-round triage file, a prefix block that changes what reviewers report, and a halt condition), and its structural-compression debt (the inline triage JSON shape, named at v15.10.0) still stands and is the next candidate. 15150 and 60250 were the smallest steps that clear it, leaving 25 and 45 tokens. v16.23.0: phase-0-init max 12400 -> 12500 and total 60250 -> 60500 for the widget-registration call and the accounting gate. Phase 0 gained the `tiles` call and the exit-3 rule, Phase 7 gained the run report; together they are contract an agent cannot infer - which call registers this host's widget, and that a completion is refused without recorded spend. Compression came first and twice, taking the new prose from 472 tokens to 255: the per-host call list moved into tracker-contract.md \"The card is not the widget\" and the record-then-rerun recovery into \"Accounting is a gate\", both outside this budget, leaving the phase docs with the call and the one fact that cannot be looked up. 100 and 250 were the smallest steps that clear it, leaving 74 and 40 tokens. phase-3 and phase-4 remain amber. v16.24.0: total 60500 -> 60750 for visual evidence. Four phase docs gained one instruction each that cannot be inferred: Phase 0 keeps the issue's own images as the pre-fix evidence, Phase 3 captures the fixed state (there and not Phase 5, because every autopilot and --local entry drops Phase 5), Phase 5 hosts the flow recording when it runs, and Phase 6 blocks on a required artefact that is neither attached nor explained. Compression came first and twice, 42 tokens back, and the contract itself never entered this budget: the trigger matrix, the three video tiers, the size-degradation ladder and both render shapes live in multi-agent-refs/features/visual-evidence.md. phase-0-init cleared its own max without a bump. 250 was the smallest step that clears the aggregate. phase-3 and phase-4 remain amber."
41
41
  }
@@ -124,11 +124,19 @@ ROOT=$(git rev-parse --show-toplevel 2>/dev/null || true)
124
124
  REFS_DIR="$ROOT/.multi-agent/refs"
125
125
  NODE_ID=""
126
126
  if mkdir -p "$REFS_DIR" 2>/dev/null; then
127
- GITIGNORE="$ROOT/.multi-agent/.gitignore"
128
- if [ ! -f "$GITIGNORE" ]; then
129
- printf '# Local run artefacts - never commit.\nmemory/\nrefs/\n' > "$GITIGNORE"
130
- elif ! grep -q '^refs/$' "$GITIGNORE" 2>/dev/null; then
131
- printf 'refs/\n' >> "$GITIGNORE"
127
+ # Shared with offload-ref.sh via pipeline/lib/repo-hygiene.sh. `../lib` resolves in
128
+ # both layouts: pipeline/scripts -> pipeline/lib in the repo, ~/.claude/scripts ->
129
+ # ~/.claude/lib installed. Fall back to writing it inline if the lib is absent,
130
+ # so this script stays usable standalone.
131
+ _MA_HYG="$(dirname "${BASH_SOURCE[0]}")/../lib/repo-hygiene.sh"
132
+ if [ -f "$_MA_HYG" ]; then
133
+ . "$_MA_HYG"
134
+ ma_hygiene_local_gitignore "$ROOT"
135
+ else
136
+ GITIGNORE="$ROOT/.multi-agent/.gitignore"
137
+ if [ ! -f "$GITIGNORE" ] || ! grep -q '^\*$' "$GITIGNORE" 2>/dev/null; then
138
+ printf '# Local run artefacts - never commit. Ignores this file too.\n*\n' > "$GITIGNORE"
139
+ fi
132
140
  fi
133
141
  if command -v shasum >/dev/null 2>&1; then
134
142
  DIGEST=$(shasum -a 256 "$FILE" | awk '{print substr($1,1,8)}')
@@ -6,12 +6,19 @@
6
6
  #
7
7
  # Contract: multi-agent-refs/features/visual-evidence.md
8
8
  #
9
- # Two jobs, deliberately in one script because they share the naming convention
9
+ # Three jobs, deliberately in one script because they share the naming convention
10
10
  # that the Jira comment references by filename:
11
11
  #
12
12
  # capture-evidence.sh after --task <id> --platform <ios|android> --label <slug>
13
13
  # Capture the current screen of the running app into the evidence dir.
14
14
  #
15
+ # capture-evidence.sh video start|stop --task <id> --platform <ios|android> [--label <slug>]
16
+ # Writes <task>[-<label>]-flow.mp4; the label is optional because one
17
+ # recording per task is the common case.
18
+ # Record the screen while something else drives the app (a UI test run, or
19
+ # an MCP-driven flow). start and stop are separate because the recording
20
+ # wraps a run whose duration is not known in advance.
21
+ #
15
22
  # capture-evidence.sh fit --file <path> [--max-mb <n>]
16
23
  # capture-evidence.sh limits
17
24
  # Print the resolved visualEvidence settings as KEY=VALUE lines.
@@ -120,6 +127,163 @@ case "$MODE" in
120
127
  printf '%s\n' "$OUT"
121
128
  ;;
122
129
 
130
+ video)
131
+ # The flow recording. Shell, not MCP, for the same three reasons the "after"
132
+ # capture is: the screenshot path already shells out, a run whose host has no
133
+ # toolkit MCP registered still produces evidence, and Phase 3 / Phase 5 are
134
+ # exactly where an MCP call is contested. start and stop are separate because
135
+ # the recording wraps a test run whose duration is not known in advance - a
136
+ # fixed --duration either truncates the test or trails dead screen after it.
137
+ ACTION="${1:-}"
138
+ shift 2>/dev/null || true
139
+ # LABEL defaults to EMPTY, not to "flow": the suffix is already "-flow", so a
140
+ # default of "flow" produced <task>-flow-flow.mp4 while every renderer cites
141
+ # <task>-flow.mp4. One recording per task is the common case and it should not
142
+ # have to name itself twice to get the filename the contract quotes.
143
+ TASK=""; PLATFORM=""; LABEL=""
144
+ while [ "$#" -gt 0 ]; do
145
+ case "$1" in
146
+ --task) TASK="${2:-}"; shift 2 ;;
147
+ --platform) PLATFORM="${2:-}"; shift 2 ;;
148
+ --label) LABEL="${2:-}"; shift 2 ;;
149
+ *) echo "capture-evidence: unknown option $1" >&2; exit 2 ;;
150
+ esac
151
+ done
152
+ case "$ACTION" in start | stop) ;; *)
153
+ echo "usage: capture-evidence.sh video start|stop --task <id> --platform <ios|android> [--label <slug>]" >&2
154
+ exit 2 ;;
155
+ esac
156
+ [ -n "$TASK" ] && [ -n "$PLATFORM" ] || {
157
+ echo "usage: capture-evidence.sh video start|stop --task <id> --platform <ios|android> [--label <slug>]" >&2
158
+ exit 2
159
+ }
160
+ case "$PLATFORM" in ios | android) ;; *)
161
+ echo "capture-evidence: unsupported platform '$PLATFORM'" >&2; exit 2 ;;
162
+ esac
163
+
164
+ mkdir -p "$EVIDENCE_DIR"
165
+ SLUG="${TASK}${LABEL:+-$LABEL}"
166
+ OUT="$EVIDENCE_DIR/${SLUG}-flow.mp4"
167
+ PIDFILE="$EVIDENCE_DIR/.${SLUG}.recpid"
168
+ WATCHFILE="$EVIDENCE_DIR/.${SLUG}.watchpid"
169
+ REMOTE="/sdcard/_ma_${SLUG}.mp4"
170
+
171
+ # screenrecord's own ceiling is 180s and it is not negotiable, so a
172
+ # maxVideoSeconds above it would silently become 180 on Android and stay
173
+ # honoured on iOS - two platforms disagreeing about one preference. Clamp in
174
+ # one place and say so.
175
+ CAP="$MAX_VIDEO_SECONDS"
176
+ if [ "$CAP" -gt 180 ] 2>/dev/null; then
177
+ echo "capture-evidence: maxVideoSeconds $CAP clamped to 180 (screenrecord ceiling)" >&2
178
+ CAP=180
179
+ fi
180
+
181
+ if [ "$ACTION" = "start" ]; then
182
+ [ -f "$PIDFILE" ] && { echo "capture-evidence: a recording for $SLUG is already running" >&2; exit 2; }
183
+ rm -f "$OUT"
184
+ case "$PLATFORM" in
185
+ ios)
186
+ command -v xcrun >/dev/null 2>&1 || { echo "capture-evidence: xcrun unavailable" >&2; exit 4; }
187
+ xcrun simctl list devices booted 2>/dev/null | grep -q "(Booted)" \
188
+ || { echo "capture-evidence: no booted simulator" >&2; exit 4; }
189
+ # h264 rather than the hevc default: an hevc mp4 does not play in the
190
+ # Jira attachment preview or in several browsers, which turns the
191
+ # artefact into a download nobody opens.
192
+ xcrun simctl io booted recordVideo --codec h264 --force "$OUT" >/dev/null 2>&1 &
193
+ REC_PID=$!
194
+ ;;
195
+ android)
196
+ command -v adb >/dev/null 2>&1 || { echo "capture-evidence: adb unavailable" >&2; exit 4; }
197
+ adb shell true >/dev/null 2>&1 || { echo "capture-evidence: no attached device" >&2; exit 4; }
198
+ adb shell rm -f "$REMOTE" >/dev/null 2>&1 || true
199
+ adb shell screenrecord --time-limit "$CAP" "$REMOTE" >/dev/null 2>&1 &
200
+ REC_PID=$!
201
+ ;;
202
+ esac
203
+
204
+ # A recorder that dies on the first frame leaves a pid file and an empty
205
+ # path, and the caller then "stops" a recording that never ran. Give it a
206
+ # beat and check it actually started.
207
+ sleep 1
208
+ kill -0 "$REC_PID" 2>/dev/null || {
209
+ echo "capture-evidence: recorder exited immediately" >&2
210
+ exit 4
211
+ }
212
+ # On Android the local process is the adb CLIENT, which stays alive even
213
+ # when screenrecord failed on the device, so liveness there proves only
214
+ # that adb is running. The remote file existing is what proves a recording
215
+ # began.
216
+ if [ "$PLATFORM" = "android" ]; then
217
+ adb shell "[ -e '$REMOTE' ]" >/dev/null 2>&1 || {
218
+ kill "$REC_PID" 2>/dev/null || true
219
+ echo "capture-evidence: screenrecord did not start on the device" >&2
220
+ exit 4
221
+ }
222
+ fi
223
+ printf '%s\n' "$REC_PID" > "$PIDFILE"
224
+
225
+ # iOS has no --time-limit, so the cap is ours to enforce. Without this a
226
+ # hung UI test records until the disk complains.
227
+ if [ "$PLATFORM" = "ios" ]; then
228
+ (sleep "$CAP"; kill -INT "$REC_PID" 2>/dev/null || true) >/dev/null 2>&1 &
229
+ printf '%s\n' "$!" > "$WATCHFILE"
230
+ fi
231
+ printf '%s\n' "$OUT"
232
+ exit 0
233
+ fi
234
+
235
+ # stop
236
+ [ -f "$PIDFILE" ] || { echo "capture-evidence: no recording in progress for $SLUG" >&2; exit 2; }
237
+ REC_PID=$(cat "$PIDFILE" 2>/dev/null)
238
+ rm -f "$PIDFILE"
239
+ if [ -f "$WATCHFILE" ]; then
240
+ kill "$(cat "$WATCHFILE" 2>/dev/null)" 2>/dev/null || true
241
+ rm -f "$WATCHFILE"
242
+ fi
243
+
244
+ case "$PLATFORM" in
245
+ ios)
246
+ # SIGINT, not SIGTERM: simctl only writes the moov atom and closes the
247
+ # container on INT. A TERM leaves an mp4 that every player refuses.
248
+ kill -INT "$REC_PID" 2>/dev/null || true
249
+ i=0
250
+ while kill -0 "$REC_PID" 2>/dev/null && [ "$i" -lt 100 ]; do sleep 0.1; i=$((i + 1)); done
251
+ kill -0 "$REC_PID" 2>/dev/null && kill -TERM "$REC_PID" 2>/dev/null || true
252
+ ;;
253
+ android)
254
+ # Match on the output path, not on the program name: a bare
255
+ # `pkill screenrecord` stops every recording on the device, including a
256
+ # second pipeline run's and anything the user started by hand. Fall back
257
+ # to the blunt form only when -f is unavailable.
258
+ adb shell pkill -2 -f "$REMOTE" >/dev/null 2>&1 \
259
+ || adb shell pkill -2 screenrecord >/dev/null 2>&1 || true
260
+ # screenrecord finalises the container after the signal; pulling straight
261
+ # away yields a truncated file that looks like a successful capture.
262
+ sleep 2
263
+ adb pull "$REMOTE" "$OUT" >/dev/null 2>&1 || true
264
+ adb shell rm -f "$REMOTE" >/dev/null 2>&1 || true
265
+ ;;
266
+ esac
267
+
268
+ [ -s "$OUT" ] || { echo "capture-evidence: recording produced no file" >&2; exit 4; }
269
+
270
+ # Both recorders encode on change, so a flow over a screen that never moved
271
+ # produces a valid two-frame mp4 a fraction of a second long. The file is not
272
+ # broken and must not be discarded, but it is not evidence of a flow either,
273
+ # and the duration is the only thing that can tell the two apart. Say so and
274
+ # let the caller record it; never assert this duration against wall clock,
275
+ # which is what makes a correct static capture look like a failure.
276
+ if command -v ffprobe >/dev/null 2>&1; then
277
+ SECS=$(ffprobe -v error -show_entries format=duration -of csv=p=0 "$OUT" 2>/dev/null)
278
+ case "$SECS" in
279
+ "" ) echo "capture-evidence: ffprobe could not read the recording; duration unknown" >&2 ;;
280
+ * ) awk -v d="$SECS" 'BEGIN { exit (d < 1.0) ? 0 : 1 }' \
281
+ && echo "capture-evidence: recording is ${SECS}s - the screen did not change during it" >&2 ;;
282
+ esac
283
+ fi
284
+ printf '%s\n' "$OUT"
285
+ ;;
286
+
123
287
  fit)
124
288
  FILE=""; MAX_MB="$MAX_ATTACH_MB"
125
289
  while [ "$#" -gt 0 ]; do
@@ -174,14 +338,15 @@ case "$MODE" in
174
338
  ;;
175
339
 
176
340
  limits)
177
- # The flow recorder lives in the phase doc (it drives MCP tools, not a
178
- # shell); it reads its duration cap from here so visualEvidence.maxVideoSeconds
179
- # is one resolved value rather than a number repeated in two documents.
341
+ # The resolved settings as KEY=VALUE, so visualEvidence.maxVideoSeconds is one
342
+ # value read in one place rather than a number repeated across documents. The
343
+ # `video` mode above reads the same function, so the cap a caller prints and
344
+ # the cap the recorder enforces cannot drift apart.
180
345
  prefs_visual
181
346
  ;;
182
347
 
183
348
  *)
184
- echo "usage: capture-evidence.sh after|fit|limits ..." >&2
349
+ echo "usage: capture-evidence.sh after|video|fit|limits ..." >&2
185
350
  exit 2
186
351
  ;;
187
352
  esac
@@ -84,6 +84,7 @@ const CHECK_IDS = [
84
84
  "task-tools",
85
85
  "mcp-registration",
86
86
  "disk-space",
87
+ "worktree-residue",
87
88
  ];
88
89
 
89
90
  // Only these five may return BLOCK. Enforced below, not merely documented: a
@@ -641,6 +642,59 @@ function checkDiskSpace() {
641
642
  ok("disk-space");
642
643
  }
643
644
 
645
+ // A finished task removes its own worktree at PR time, but a run that dies
646
+ // before Phase 6 never reaches that step and nothing else collects it: the
647
+ // finalizer only runs on success and gc-worktrees only sweeps entries git no
648
+ // longer knows about. So worktrees accumulate silently, and each one is a full
649
+ // second checkout. Measured on one real iOS repo: 15 left behind, 11 GB.
650
+ //
651
+ // Only the repo the caller is standing in is examined. A project name in prefs
652
+ // is a name, not a path, and guessing checkout locations to report a number is
653
+ // how a diagnostic starts lying.
654
+ function checkWorktreeResidue() {
655
+ let repo;
656
+ try {
657
+ repo = execFileSync("git", ["rev-parse", "--show-toplevel"], {
658
+ encoding: "utf8",
659
+ stdio: ["ignore", "pipe", "ignore"],
660
+ }).trim();
661
+ } catch {
662
+ skip("worktree-residue", "not inside a git repository, so there is no checkout to measure");
663
+ return;
664
+ }
665
+ const wt = join(repo, ".worktrees");
666
+ let entries;
667
+ try {
668
+ entries = readdirSync(wt).filter((n) => n !== ".DS_Store" && n !== ".archive");
669
+ } catch {
670
+ ok("worktree-residue");
671
+ return;
672
+ }
673
+ if (entries.length === 0) {
674
+ ok("worktree-residue");
675
+ return;
676
+ }
677
+ let gb = null;
678
+ try {
679
+ const out = execFileSync("du", ["-sk", wt], { encoding: "utf8" });
680
+ const kb = Number(out.trim().split(/\s+/)[0]);
681
+ if (Number.isFinite(kb)) gb = kb / 1024 / 1024;
682
+ } catch {
683
+ /* du is not everywhere; the count alone is still worth reporting */
684
+ }
685
+ const size = gb === null ? "" : `, ${gb.toFixed(1)} GB`;
686
+ if (entries.length >= 5 || (gb !== null && gb >= 2)) {
687
+ report(
688
+ "worktree-residue",
689
+ "WARN",
690
+ `${entries.length} worktree(s) left under ${wt}${size}; runs that stopped before Phase 6 are never collected`,
691
+ "run /multi-agent:garbage-collect, or /multi-agent:kill for a task you know is dead",
692
+ );
693
+ return;
694
+ }
695
+ ok("worktree-residue");
696
+ }
697
+
644
698
  /* ------------------------------------------------------------------ main -- */
645
699
 
646
700
  function main() {
@@ -675,7 +729,7 @@ function main() {
675
729
  } else {
676
730
  process.stdout.write(`${line.severity} ${line.id} - ${line.problem} - ${line.step}\n`);
677
731
  process.stdout.write(
678
- "\n-> indeterminate: the layout did not resolve, so 13 checks did not run\n",
732
+ `\n-> indeterminate: the layout did not resolve, so ${CHECK_IDS.length - 1} checks did not run\n`,
679
733
  );
680
734
  }
681
735
  process.exitCode = 4;
@@ -694,6 +748,7 @@ function main() {
694
748
  checkTaskTools();
695
749
  checkMcpRegistration();
696
750
  checkDiskSpace();
751
+ checkWorktreeResidue();
697
752
 
698
753
  const blocked = results.filter((r) => r.severity === "BLOCK");
699
754
  const warned = results.filter((r) => r.severity === "WARN");
@@ -19,6 +19,9 @@
19
19
  // manual is JSON-based (Phase 5 manual-test.json, see below)
20
20
  // --status the claimed status (only "passed" is gated; others pass through)
21
21
  // --evidence <path> the log / artifact that must substantiate a "passed" claim
22
+ // --require-screenshot manual claim only: every "pass" criterion must name a
23
+ // screenshot that exists on disk. Set by Phase 5 when
24
+ // state.visualEvidence.required is true.
22
25
  // --success-pattern override the default success regex for the claim type (marker claims only)
23
26
  // --failure-pattern override the default failure regex for the claim type (marker claims only)
24
27
  //
@@ -120,7 +123,7 @@ function nonEmptyString(v) {
120
123
  // Manual-test evidence is a structured document, not a log: every acceptance
121
124
  // criterion must say what was observed and how it came out. A "fail" anywhere
122
125
  // is decisive; an untested criterion is only acceptable when it says why.
123
- function gateManual(claim, evidence, body) {
126
+ function gateManual(claim, evidence, body, requireScreenshot) {
124
127
  let doc;
125
128
  try {
126
129
  doc = JSON.parse(body);
@@ -154,6 +157,32 @@ function gateManual(claim, evidence, body) {
154
157
  `${claim} evidence shows a failed criterion "${failed.spec}" - pass claim contradicted by ${evidence}`,
155
158
  );
156
159
  }
160
+ // A screenshot path is only evidence when a file is behind it. The field has
161
+ // been in the document shape since the gate shipped and was never read, so a
162
+ // criterion carrying "screenshot": null passed as a verified manual test on a
163
+ // UI change - which is the one case the picture was added for. Checked only
164
+ // when the caller says visual evidence is required for this run, because on a
165
+ // backend change there is nothing to photograph.
166
+ if (requireScreenshot) {
167
+ const shotless = criteria.find(
168
+ (item) => item.verdict === "pass" && !nonEmptyString(item.screenshot),
169
+ );
170
+ if (shotless) {
171
+ fail(
172
+ `${claim} evidence has passing criterion "${shotless.spec}" with no screenshot, and this run requires visual evidence: ${evidence} (default-FAIL)`,
173
+ );
174
+ }
175
+ const missing = criteria.find(
176
+ (item) =>
177
+ item.verdict === "pass" && nonEmptyString(item.screenshot) && !existsSync(item.screenshot),
178
+ );
179
+ if (missing) {
180
+ fail(
181
+ `${claim} evidence names a screenshot that is not on disk: ${missing.screenshot} (default-FAIL)`,
182
+ );
183
+ }
184
+ }
185
+
157
186
  const untested = criteria.find(
158
187
  (item) => item.verdict === "not-tested" && !nonEmptyString(item.reason),
159
188
  );
@@ -203,7 +232,7 @@ if (body.trim().length === 0) {
203
232
  }
204
233
 
205
234
  if (JSON_CLAIMS.has(claim)) {
206
- gateManual(claim, evidence, body);
235
+ gateManual(claim, evidence, body, args["require-screenshot"] === true);
207
236
  }
208
237
 
209
238
  const successRe = userRegExp(args["success-pattern"], DEFAULT_MARKERS[claim].success, "success");
@@ -97,8 +97,24 @@ NAME_ARGS=(
97
97
  -o -name 'context-links-*'
98
98
  -o -name 'context-by-type-*'
99
99
  -o -name 'analysis-*'
100
+ -o -name 'complaint-analysis-*'
101
+ -o -name 'generate-issue-*'
102
+ -o -name 'cred-inventory-*'
103
+ -o -name 'pr-body-*'
104
+ -o \( -name '*-wiki' -type d \)
100
105
  )
101
106
 
107
+ # `complaint-analysis-*` needs its own entry: find matches the basename against
108
+ # the whole glob, so `analysis-*` never matched it. Same for the other three,
109
+ # which were writing into $ROOT with nothing sweeping them (phase-0-init.md,
110
+ # phase-6-commit.md, generate-issue.md).
111
+ #
112
+ # `*-wiki` is the one loose pattern here, and a suffix glob over $ROOT could
113
+ # name something a person created. It is narrowed twice: `-type d` above, and
114
+ # the `.git` check in the loop below - the pipeline's only `*-wiki` artefact is
115
+ # a shallow clone (analysis/evidence.md), so a directory without a `.git` is
116
+ # somebody else's and is left alone.
117
+
102
118
  MTIME_ARGS=()
103
119
  if [ "$OLDER_MIN" -gt 0 ]; then
104
120
  MTIME_ARGS=(-mmin "+$OLDER_MIN")
@@ -114,6 +130,20 @@ while IFS= read -r -d '' p; do
114
130
  skipped_active=$((skipped_active + 1))
115
131
  continue
116
132
  fi
133
+ # The `.git` requirement belongs to the `*-wiki` pattern alone. Checking the
134
+ # suffix first would re-narrow entries an unambiguous prefix already claimed:
135
+ # `multi-agent-foo-wiki` ends in `-wiki`, and demanding a `.git` there would
136
+ # spare scratch that `multi-agent-*` has always swept unconditionally.
137
+ case "${p##*/}" in
138
+ multi-agent-* | issue-progress-* | channels-* | context-links-* | context-by-type-* | \
139
+ analysis-* | complaint-analysis-* | generate-issue-* | cred-inventory-* | pr-body-*) ;;
140
+ *-wiki)
141
+ # See the NAME_ARGS note: only our shallow clones, never a user's folder.
142
+ if [ ! -e "$p/.git" ]; then
143
+ continue
144
+ fi
145
+ ;;
146
+ esac
117
147
  matches+=("$p")
118
148
  done < <(find "$ROOT" -mindepth 1 -maxdepth 1 \( "${NAME_ARGS[@]}" \) ${MTIME_ARGS[@]+"${MTIME_ARGS[@]}"} -print0 2>/dev/null)
119
149
 
@@ -12,8 +12,13 @@
12
12
  # "Subproject commit" entry by a blanket `git add -A` before the
13
13
  # residue guard existed. --yes runs `git rm --cached` (the commit
14
14
  # itself stays yours to make).
15
- # 4. missing residue guard -> ensures `.worktrees/` is in
16
- # .git/info/exclude so the gitlink class cannot re-occur.
15
+ # 4. missing residue guard -> writes the managed block in
16
+ # .git/info/exclude (`.worktrees/`, `.pipeline/`, `.multi-agent/` and
17
+ # the loose run logs) so none of them can reach the index. The block is
18
+ # rewritten on each call, so a repo guarded by the old one-line form is
19
+ # upgraded in place.
20
+ # 5. empty artefact parents -> rmdir only, so a `.worktrees/` or
21
+ # `.pipeline/` left behind by a finished run stops reading as residue.
17
22
  #
18
23
  # Registered, healthy worktrees are NEVER touched. Finishing a task removes its
19
24
  # own worktree in Phase 6 (worktree-finalize, v14.1.0+); killing one is
@@ -143,22 +148,39 @@ $gitlinks
143
148
  EOF
144
149
  fi
145
150
 
146
- # 4. Residue guard: keep .worktrees/ out of the index from now on.
151
+ # 4. Residue guard: keep every pipeline artefact out of the index from now on.
152
+ # `.worktrees/` was the only entry for a long time, which left `.pipeline/` and
153
+ # `.multi-agent/` writable into the branch; the managed block in repo-hygiene.sh
154
+ # covers all of them and is rewritten on each call, so an old checkout upgrades.
147
155
  ex="$(git -C "$REPO" rev-parse --path-format=absolute --git-common-dir 2>/dev/null)/info/exclude"
148
- if [ -n "${ex%/info/exclude}" ] && ! grep -qxF '.worktrees/' "$ex" 2>/dev/null; then
149
- if [ "$DELETE" -eq 1 ]; then
150
- mkdir -p "$(dirname "$ex")" \
151
- && printf '.worktrees/\n' >> "$ex" \
152
- && { echo "→ added .worktrees/ to .git/info/exclude (residue guard)"; changed=$((changed + 1)); }
156
+ _MA_HYG="$(dirname "${BASH_SOURCE[0]}")/../lib/repo-hygiene.sh"
157
+ [ -f "$_MA_HYG" ] && . "$_MA_HYG"
158
+ if [ -n "${ex%/info/exclude}" ] && ! grep -qF "${MA_HYGIENE_BEGIN:-# >>> multi-agent pipeline (managed) >>>}" "$ex" 2>/dev/null; then
159
+ if ! command -v ma_hygiene_ensure_exclusions >/dev/null 2>&1; then
160
+ # The guard is the whole point of step 4. Going quiet when the library is
161
+ # missing would leave the repo unguarded and say nothing, which is the one
162
+ # outcome worse than failing.
163
+ echo "gc-worktrees: WARN - $_MA_HYG not found; the residue guard was NOT written" >&2
164
+ elif [ "$DELETE" -eq 1 ]; then
165
+ ma_hygiene_ensure_exclusions "$REPO"
166
+ echo "→ wrote the managed residue block to .git/info/exclude"
167
+ changed=$((changed + 1))
153
168
  else
154
- echo "would add .worktrees/ to .git/info/exclude (residue guard)"
169
+ echo "would write the managed residue block to .git/info/exclude"
155
170
  fi
156
171
  fi
157
172
 
173
+ # 5. Artefact parents left empty by a finished run read as leftovers to anyone
174
+ # looking at the checkout. rmdir, never rm -rf: it fails harmlessly if anything
175
+ # is still there.
176
+ if [ "$DELETE" -eq 1 ] && command -v ma_hygiene_prune_empty >/dev/null 2>&1; then
177
+ ma_hygiene_prune_empty "$REPO"
178
+ fi
179
+
158
180
  total=$(( ${#orphans[@]} + $(printf '%s' "$gitlinks" | grep -c . || true) ))
159
181
  if [ "$DELETE" -eq 1 ]; then
160
182
  echo "══ gc-worktrees: applied $changed change(s) in $REPO ══"
161
- elif [ "$total" -eq 0 ] && [ "$pruned" -eq 0 ] && { [ -z "${ex%/info/exclude}" ] || grep -qxF '.worktrees/' "$ex" 2>/dev/null; }; then
183
+ elif [ "$total" -eq 0 ] && [ "$pruned" -eq 0 ] && { [ -z "${ex%/info/exclude}" ] || grep -qF "${MA_HYGIENE_BEGIN:-# >>> multi-agent pipeline (managed) >>>}" "$ex" 2>/dev/null; }; then
162
184
  echo "gc-worktrees: no worktree residue in $REPO - nothing to do"
163
185
  else
164
186
  echo "══ gc-worktrees: dry-run - re-run with --yes to apply ══"
@@ -158,11 +158,19 @@ mkdir -p "$REFS_DIR" 2>/dev/null || { cat "$BUF"; exit 0; }
158
158
 
159
159
  # Never commit an offloaded payload: it is a build log, and the worktree it
160
160
  # belongs to is deleted at the end of the task.
161
- GITIGNORE="$ROOT/.multi-agent/.gitignore"
162
- if [ ! -f "$GITIGNORE" ]; then
163
- printf '# Local run artefacts - never commit.\nmemory/\nrefs/\n' > "$GITIGNORE"
164
- elif ! grep -q '^refs/$' "$GITIGNORE"; then
165
- printf 'refs/\n' >> "$GITIGNORE"
161
+ # Shared with bulk-read.sh via pipeline/lib/repo-hygiene.sh. `../lib` resolves in
162
+ # both layouts: pipeline/scripts -> pipeline/lib in the repo, ~/.claude/scripts ->
163
+ # ~/.claude/lib installed. Fall back to writing it inline if the lib is absent,
164
+ # so this script stays usable standalone.
165
+ _MA_HYG="$(dirname "${BASH_SOURCE[0]}")/../lib/repo-hygiene.sh"
166
+ if [ -f "$_MA_HYG" ]; then
167
+ . "$_MA_HYG"
168
+ ma_hygiene_local_gitignore "$ROOT"
169
+ else
170
+ GITIGNORE="$ROOT/.multi-agent/.gitignore"
171
+ if [ ! -f "$GITIGNORE" ] || ! grep -q '^\*$' "$GITIGNORE" 2>/dev/null; then
172
+ printf '# Local run artefacts - never commit. Ignores this file too.\n*\n' > "$GITIGNORE"
173
+ fi
166
174
  fi
167
175
 
168
176
  SLUG=$(printf '%s' "$LABEL" | tr '[:upper:]' '[:lower:]' | tr -c 'a-z0-9' '-' | sed 's/-\{1,\}/-/g; s/^-//; s/-$//')