@mmerterden/multi-agent-pipeline 17.5.1 → 18.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +276 -0
- package/README.md +59 -1
- package/README.tr.md +57 -0
- package/docs/adr/0011-dormant-ci.md +25 -1
- package/docs/features.md +24 -0
- package/docs/server-readiness.md +188 -0
- package/docs/token-budget-history.md +1 -1
- package/index.js +16 -1
- package/install/_common.mjs +42 -17
- package/install/_dev-only-files.mjs +8 -0
- package/install/_unattended-profile.mjs +113 -0
- package/install/index.mjs +48 -0
- package/install/templates/claude-hooks.json +13 -1
- package/manifest.json +1049 -0
- package/package.json +5 -2
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/feedback/SKILL.md +7 -1
- package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/issue/SKILL.md +13 -1
- package/pipeline/commands/multi-agent/jira/SKILL.md +13 -1
- package/pipeline/commands/multi-agent/resume/SKILL.md +16 -1
- package/pipeline/commands/multi-agent/setup/SKILL.md +14 -16
- package/pipeline/commands/multi-agent/status/SKILL.md +52 -21
- package/pipeline/commands/multi-agent/update/SKILL.md +13 -56
- package/pipeline/lib/_jira-auth.sh +8 -0
- package/pipeline/lib/analysis-jira-write.sh +32 -0
- package/pipeline/lib/ask-choice.sh +13 -2
- package/pipeline/lib/autopilot-state.sh +8 -0
- package/pipeline/lib/fatal.mjs +129 -0
- package/pipeline/lib/figma-mcp-refresh.sh +18 -0
- package/pipeline/lib/figma-screenshot.sh +18 -0
- package/pipeline/lib/invoked-directly.mjs +43 -0
- package/pipeline/lib/jira-publish.sh +42 -0
- package/pipeline/lib/md2confluence-v3.py +47 -0
- package/pipeline/lib/outbound-gate.mjs +175 -0
- package/pipeline/lib/plan-todos.sh +27 -6
- package/pipeline/lib/post-pr-review.sh +77 -8
- package/pipeline/lib/repo-hygiene.sh +8 -3
- package/pipeline/lib/require-jq.sh +40 -0
- package/pipeline/lib/run-paths.sh +335 -0
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +70 -0
- package/pipeline/multi-agent-refs/features/code-graph.md +20 -0
- package/pipeline/multi-agent-refs/features/cost-analysis.md +93 -0
- package/pipeline/multi-agent-refs/features/doctor.md +68 -0
- package/pipeline/multi-agent-refs/features/maturity-followup.md +166 -0
- package/pipeline/multi-agent-refs/features/package-manager.md +80 -0
- package/pipeline/multi-agent-refs/features/usage-reporting.md +79 -0
- package/pipeline/multi-agent-refs/features/verify-by-test.md +1 -1
- package/pipeline/multi-agent-refs/features/verify.md +83 -0
- package/pipeline/multi-agent-refs/phases/operations.md +13 -2
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +6 -3
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +8 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +1 -1
- package/pipeline/multi-agent-refs/picker-contract.md +1 -1
- package/pipeline/multi-agent-refs/unattended-contract.md +129 -0
- package/pipeline/preferences-template.json +1 -1
- package/pipeline/schemas/agent-state.schema.json +122 -11
- package/pipeline/schemas/prefs.schema.json +35 -0
- package/pipeline/schemas/token-budget.json +2 -2
- package/pipeline/scripts/_run-paths.mjs +372 -0
- package/pipeline/scripts/aggregate-metrics.mjs +64 -64
- package/pipeline/scripts/autopilot-arming.mjs +2 -1
- package/pipeline/scripts/autopilot-intake.mjs +2 -1
- package/pipeline/scripts/autopilot-runner.mjs +206 -2
- package/pipeline/scripts/build-references.mjs +2 -1
- package/pipeline/scripts/build-stack-plugins.mjs +10 -2
- package/pipeline/scripts/capture-evidence.sh +7 -2
- package/pipeline/scripts/classify-plan-safety.mjs +2 -1
- package/pipeline/scripts/cost-analyze.mjs +600 -0
- package/pipeline/scripts/cost-budget-check.mjs +4 -12
- package/pipeline/scripts/council-view.mjs +2 -1
- package/pipeline/scripts/crush-json.mjs +2 -1
- package/pipeline/scripts/diff-explain.mjs +6 -9
- package/pipeline/scripts/diff-risk-score.mjs +2 -1
- package/pipeline/scripts/doctor.mjs +203 -4
- package/pipeline/scripts/evidence-gate.mjs +9 -3
- package/pipeline/scripts/feedback-send.mjs +13 -3
- package/pipeline/scripts/gc-abandoned.sh +29 -13
- package/pipeline/scripts/gc-worktrees.sh +11 -4
- package/pipeline/scripts/github-ssh-setup.sh +64 -7
- package/pipeline/scripts/graph-mermaid.mjs +4 -2
- package/pipeline/scripts/graph-report.mjs +155 -1
- package/pipeline/scripts/keychain-save.sh +101 -30
- package/pipeline/scripts/learn-from-transcripts.mjs +2 -1
- package/pipeline/scripts/learning-curve.mjs +34 -29
- package/pipeline/scripts/make-manifest.mjs +199 -0
- package/pipeline/scripts/maturity-followup.mjs +294 -0
- package/pipeline/scripts/migrate-prefs.mjs +2 -1
- package/pipeline/scripts/migrate-state.mjs +94 -4
- package/pipeline/scripts/package-manager.mjs +310 -0
- package/pipeline/scripts/phase-banner.sh +6 -2
- package/pipeline/scripts/phase-tracker.sh +41 -3
- package/pipeline/scripts/plan-coverage-gate.mjs +6 -2
- package/pipeline/scripts/pre-commit-check.sh +7 -0
- package/pipeline/scripts/pre-push-check.sh +7 -0
- package/pipeline/scripts/purge.sh +23 -6
- package/pipeline/scripts/render-agent-log-cost.sh +9 -2
- package/pipeline/scripts/render-cost-summary.sh +9 -2
- package/pipeline/scripts/render-work-summary.sh +11 -4
- package/pipeline/scripts/review-file-filter.mjs +4 -2
- package/pipeline/scripts/review-scope.mjs +2 -1
- package/pipeline/scripts/routine-registry.mjs +2 -1
- package/pipeline/scripts/run-aggregator.mjs +13 -14
- package/pipeline/scripts/run-metrics.mjs +3 -1
- package/pipeline/scripts/runs-index.mjs +343 -0
- package/pipeline/scripts/scorecard-snapshot.mjs +178 -0
- package/pipeline/scripts/search-logs.sh +18 -0
- package/pipeline/scripts/test-gap-scan.mjs +2 -1
- package/pipeline/scripts/test-integrity-gate.mjs +2 -1
- package/pipeline/scripts/update-issue-progress.sh +56 -7
- package/pipeline/scripts/usage-register.mjs +271 -0
- package/pipeline/scripts/usage-report.mjs +14 -3
- package/pipeline/scripts/validate-analysis-doc.mjs +2 -1
- package/pipeline/scripts/validate-code-graph.mjs +6 -3
- package/pipeline/scripts/validate-complaint-doc.mjs +2 -1
- package/pipeline/scripts/validate-diff-risk.mjs +6 -3
- package/pipeline/scripts/validate-test-gap.mjs +6 -3
- package/pipeline/scripts/validate-triage.mjs +3 -1
- package/pipeline/scripts/verify-citations.mjs +4 -2
- package/pipeline/scripts/verify.mjs +327 -0
- package/pipeline/scripts/worktree-finalize.sh +13 -4
- package/pipeline/scripts/write-state.mjs +154 -15
- package/pipeline/skills/.skill-manifest.json +6 -6
- package/pipeline/skills/.skills-index.json +56 -1
- package/pipeline/skills/shared/README.md +8 -3
- package/pipeline/skills/shared/core/multi-agent-issue/SKILL.md +14 -0
- package/pipeline/skills/shared/core/multi-agent-jira/SKILL.md +14 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +13 -0
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +33 -9
- package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +6 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +2 -1
- package/pipeline/skills/skills-index.md +6 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mmerterden/multi-agent-pipeline",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "18.0.0",
|
|
4
4
|
"description": "8-phase AI development pipeline with full orchestration on Claude Code, Copilot CLI and Codex CLI. Analysis, planning, TDD, CLI-aware parallel review with consensus surfacing + Fable triage, default-FAIL evidence gates, secret + intent guards, per-phase cost ledger, persistent learnings memory, wiki generation, commit automation. Token-preserving uninstall.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.js",
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
},
|
|
15
15
|
"scripts": {
|
|
16
16
|
"start": "node index.js",
|
|
17
|
-
"test": "npm run format:check && node --test test/*.test.mjs && node pipeline/scripts/run-smokes.mjs && node pipeline/scripts/lint-skills.mjs && node pipeline/scripts/lint-personas.mjs && node pipeline/scripts/lint-mcp-refs.mjs && node pipeline/scripts/eval-triage.mjs && node pipeline/scripts/eval-golden-tasks.mjs && node pipeline/scripts/eval-intent.mjs && node pipeline/scripts/eval-recall.mjs && node pipeline/scripts/validate-schemas.mjs && node pipeline/scripts/validate-prefs.mjs && node pipeline/scripts/scorecard.mjs",
|
|
17
|
+
"test": "npm run format:check && npm run lint && node --test test/*.test.mjs && node pipeline/scripts/run-smokes.mjs && node pipeline/scripts/lint-skills.mjs && node pipeline/scripts/lint-personas.mjs && node pipeline/scripts/lint-mcp-refs.mjs && node pipeline/scripts/eval-triage.mjs && node pipeline/scripts/eval-golden-tasks.mjs && node pipeline/scripts/eval-intent.mjs && node pipeline/scripts/eval-recall.mjs && node pipeline/scripts/validate-schemas.mjs && node pipeline/scripts/validate-prefs.mjs && node pipeline/scripts/scorecard.mjs",
|
|
18
18
|
"test:unit": "node --test test/*.test.mjs",
|
|
19
19
|
"test:smoke": "node pipeline/scripts/run-smokes.mjs",
|
|
20
20
|
"lint:skills": "node pipeline/scripts/lint-skills.mjs",
|
|
@@ -25,6 +25,7 @@
|
|
|
25
25
|
"format": "prettier --write \"**/*.{js,mjs,json,yml}\" --ignore-path .gitignore --ignore-path .prettierignore",
|
|
26
26
|
"format:check": "prettier --check \"**/*.{js,mjs,json,yml}\" --ignore-path .gitignore --ignore-path .prettierignore",
|
|
27
27
|
"scorecard": "node pipeline/scripts/scorecard.mjs",
|
|
28
|
+
"prepack": "node pipeline/scripts/make-manifest.mjs",
|
|
28
29
|
"gate": "bash pipeline/scripts/pre-push-check.sh --run"
|
|
29
30
|
},
|
|
30
31
|
"keywords": [
|
|
@@ -69,6 +70,8 @@
|
|
|
69
70
|
],
|
|
70
71
|
"files": [
|
|
71
72
|
"index.js",
|
|
73
|
+
"manifest.json",
|
|
74
|
+
"manifest.sig",
|
|
72
75
|
"install.js",
|
|
73
76
|
"install/**/*",
|
|
74
77
|
"pipeline/**/*",
|
|
@@ -48,7 +48,7 @@ Classification schema lives in `$HOME/.claude/multi-agent-refs/_input-parser.md`
|
|
|
48
48
|
| 7 | `issue` | full picker | account → repo (multi) → issue → maturity → dev-context |
|
|
49
49
|
| 8 | Free-text | `freetext` | account → repo (single) → dev-context (maturity skip) |
|
|
50
50
|
|
|
51
|
-
**Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** -
|
|
51
|
+
**Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - a blocker asks before it halts, and an autopilot run can be told to ask on the item itself (`$HOME/.claude/multi-agent-refs/features/maturity-followup.md`). Picker `header` renders in English (the 12-char chip); the `question`, each option's `label` and each option's `description` render in `outputLanguage`, per the canonical matrix in `multi-agent-refs/rules.md`.
|
|
52
52
|
|
|
53
53
|
Lib scripts (`~/.claude/lib/`):
|
|
54
54
|
- `account-resolver.sh` - keychain account inventory
|
|
@@ -44,7 +44,13 @@ One message, sent to the maintainer's admin panel. This exists because the alter
|
|
|
44
44
|
|
|
45
45
|
## Auth and reachability
|
|
46
46
|
|
|
47
|
-
Reuses the usage ingest token (`prefs.global.keychainMapping.usage_ingest`), so nothing extra has to be onboarded
|
|
47
|
+
Reuses the usage ingest token (`prefs.global.keychainMapping.usage_ingest`), so nothing extra has to be onboarded. When none resolves, register one first and then send:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
node "$HOME/.claude/scripts/usage-register.mjs" --feedback --quiet
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
`--feedback` is what makes this work for a machine that opted out of telemetry: it mints the token and leaves `usageLog.enabled` untouched. If registration is unavailable (offline, endpoint down), the command says so and names the command that fixes it rather than failing quietly.
|
|
48
54
|
|
|
49
55
|
`usageLog.optOut` does **not** silence this. Telemetry is passive collection and opting out of it is a real choice; feedback is a deliberate act by the person typing the command, and dropping a message somebody chose to send would be worse than not offering the command at all.
|
|
50
56
|
|
|
@@ -25,7 +25,7 @@ No worktree, no branch, no commit, no pipeline chaining.
|
|
|
25
25
|
| `refresh` | same as `build` | A full rebuild takes seconds, so there is no separate incremental path |
|
|
26
26
|
| `ask "<question>"` | `graph-query.mjs "<question>" --budget N` | Token-budgeted traversal; default budget 2000 |
|
|
27
27
|
| `affected "<symbol>"` | `graph-affected.mjs "<symbol>" --depth N` | Reverse traversal: the blast radius of a change |
|
|
28
|
-
| `report` | `graph-report.mjs` | Writes `GRAPH_REPORT.md` beside the graph |
|
|
28
|
+
| `report` | `graph-report.mjs` | Writes `GRAPH_REPORT.md` beside the graph: hubs, modules, external dependencies, unconnected files, and symbols no other file references (candidates only - the extractor is regex, not a parser, so nothing gates on that list) |
|
|
29
29
|
| `status` | `graph-report.mjs --status` | One line: stack, scale, build time and whether `baseCommit` still matches HEAD. Never read the graph file yourself - it is 22MB on a large repo |
|
|
30
30
|
|
|
31
31
|
With no argument, run `status`, then offer `build` when no graph exists and
|
|
@@ -56,10 +56,22 @@ After selection, inspect `maturity` from `~/.claude/lib/issue-fetcher.sh` (same
|
|
|
56
56
|
|
|
57
57
|
| Outcome | Behavior |
|
|
58
58
|
|---|---|
|
|
59
|
-
| `blockers` non-empty (e.g. `description_empty`, `status_closed`) | **
|
|
59
|
+
| `blockers` non-empty (e.g. `description_empty`, `status_closed`) | **Ask, then halt** - see "Blockers" below. Default is still the halt |
|
|
60
60
|
| only `warnings` | AskUserQuestion: show summary + "Continue?". Autopilot auto-continues; warnings logged to `agent-log.md` |
|
|
61
61
|
| both empty | Continue silently |
|
|
62
62
|
|
|
63
|
+
**Blockers: ask, do not just stop.** `$HOME/.claude/multi-agent-refs/features/maturity-followup.md`
|
|
64
|
+
owns this. In short: an interactive run asks at this step (open the item and fix it /
|
|
65
|
+
continue without it, recording which gap was waved through / abort) instead of ending
|
|
66
|
+
with a summary; an autopilot run with `prefs.global.maturityFollowup.autopilotCommentsOnIssue`
|
|
67
|
+
on posts ONE comment on the item asking for what is missing, then halts on the circuit
|
|
68
|
+
breaker with `state.waitingFor = "maturity"` so `resume` re-enters THIS step with the item
|
|
69
|
+
re-fetched. Both defaults keep today's behaviour: the comment is off, the halt is the halt.
|
|
70
|
+
An edit is a reason to re-check, never proof the gap closed - the check re-runs on the new
|
|
71
|
+
content, and only a DIFFERENT gap set ever earns a second comment. Read the item's existing
|
|
72
|
+
comments FIRST and derive the prior ask from the newest one of ours (`priorFromComments`):
|
|
73
|
+
a scan is a new run with a fresh state file, so state alone would make every scan a first ask.
|
|
74
|
+
|
|
63
75
|
`PROMPT_LANG=en` is passed to `issue-fetcher.sh` (`promptLanguage` is locked to `"en"`).
|
|
64
76
|
|
|
65
77
|
### [4/4] Dev context (`_dev-context`)
|
|
@@ -90,12 +90,24 @@ After issue selection, inspect `maturity` from `~/.claude/lib/issue-fetcher.sh`:
|
|
|
90
90
|
|
|
91
91
|
| Outcome | Behavior |
|
|
92
92
|
|---|---|
|
|
93
|
-
| `blockers` non-empty | **
|
|
93
|
+
| `blockers` non-empty | **Ask, then halt** - see "Blockers" below. Default is still the halt |
|
|
94
94
|
| `warnings` contains `description_empty_parent_available` | Tailored parent-description question above (not the generic one) |
|
|
95
95
|
| `warnings` contains `description_empty_sibling_available` | Same question, sibling key substituted for the parent's |
|
|
96
96
|
| other `warnings` only | AskUserQuestion: show summary + "Continue?". Autopilot auto-continues; warnings logged to `agent-log.md` |
|
|
97
97
|
| both empty (score ≥ 90) | Continue silently |
|
|
98
98
|
|
|
99
|
+
**Blockers: ask, do not just stop.** `$HOME/.claude/multi-agent-refs/features/maturity-followup.md`
|
|
100
|
+
owns this. In short: an interactive run asks at this step (open the item and fix it /
|
|
101
|
+
continue without it, recording which gap was waved through / abort) instead of ending
|
|
102
|
+
with a summary; an autopilot run with `prefs.global.maturityFollowup.autopilotCommentsOnIssue`
|
|
103
|
+
on posts ONE comment on the item asking for what is missing, then halts on the circuit
|
|
104
|
+
breaker with `state.waitingFor = "maturity"` so `resume` re-enters THIS step with the item
|
|
105
|
+
re-fetched. Both defaults keep today's behaviour: the comment is off, the halt is the halt.
|
|
106
|
+
An edit is a reason to re-check, never proof the gap closed - the check re-runs on the new
|
|
107
|
+
content, and only a DIFFERENT gap set ever earns a second comment. Read the item's existing
|
|
108
|
+
comments FIRST and derive the prior ask from the newest one of ours (`priorFromComments`):
|
|
109
|
+
a scan is a new run with a fresh state file, so state alone would make every scan a first ask.
|
|
110
|
+
|
|
99
111
|
> **Resolution check**: If Jira `resolution` is set (`Fixed`, `Done`, `Won't Do`, `Resolved`...), `already_resolved` is added - the issue was closed already. To reopen, clear `resolution` in Jira first.
|
|
100
112
|
|
|
101
113
|
> **fixVersions**: When `descriptor.fixVersions` is non-empty (e.g. `["v1.48.0"]`) and a matching `release/v1.48.0` branch exists in `prefs.projects[*].branches`, suggest that branch. Otherwise fall back to `develop`/cwd current.
|
|
@@ -38,6 +38,21 @@ Resume a paused or failed task from the last successful phase.
|
|
|
38
38
|
- Claude Code: `TaskCreate` every phase from the state file in phase order, then `TaskUpdate` each to its stored status, and replace every `tasklist_id` meta with the new IDs.
|
|
39
39
|
- Other CLIs: a single `bash $HOME/.claude/scripts/phase-tracker.sh render`.
|
|
40
40
|
|
|
41
|
-
5. **Continue the pipeline
|
|
41
|
+
5. **Continue the pipeline.** Read `state.waitingFor` FIRST: when it names a step, the
|
|
42
|
+
run re-enters THAT step rather than the next phase. `currentPhase + 1` is the fallback,
|
|
43
|
+
not the rule - a run that stopped mid-phase to ask a human has `currentPhase` pointing at
|
|
44
|
+
the phase it is still inside, so resuming past it skips the question permanently. That
|
|
45
|
+
was already true of Phase 7's channels pause, which documented itself as resumable
|
|
46
|
+
through this field while this file never mentioned it.
|
|
47
|
+
|
|
48
|
+
| `waitingFor` | Re-entry |
|
|
49
|
+
|---|---|
|
|
50
|
+
| `maturity` | Phase 0, the maturity step, with the item **re-fetched** and re-scored - an edit is a reason to look again, never proof the gap closed (`$HOME/.claude/multi-agent-refs/features/maturity-followup.md`) |
|
|
51
|
+
| `user-channels-choice` | Phase 7, the channels multi-select, with the stored `channelsInput` |
|
|
52
|
+
| absent | `currentPhase + 1`, as before (same pipeline as the main multi-agent command) |
|
|
53
|
+
|
|
54
|
+
Clear `waitingFor` in the same write that records the answer, the moment the step is
|
|
55
|
+
re-entered. A field that outlives the question it asked sends every later resume back
|
|
56
|
+
to the step the user already answered.
|
|
42
57
|
|
|
43
58
|
6. **Log**: `🔄 Resumed {JIRA-KEY}-{id} from Phase {N}`
|
|
@@ -209,9 +209,7 @@ Full key list and shape: `$HOME/.claude/multi-agent-refs/keychain.md`.
|
|
|
209
209
|
|
|
210
210
|
`null` = not mapped (missing or skipped). Pipeline phases read this mapping to retrieve tokens dynamically - never hardcoded key names.
|
|
211
211
|
|
|
212
|
-
### Step 2.7 - Operational reporting
|
|
213
|
-
|
|
214
|
-
Only relevant when the admin has issued this user a token. Since v15.8.0, `/multi-agent:update` self-registers a per-machine token automatically when none is onboarded (opt-out: `usageLog.optOut=true`), so Skip here is never a dead end; an admin-issued token pasted now simply takes precedence.
|
|
212
|
+
### Step 2.7 - Operational reporting
|
|
215
213
|
|
|
216
214
|
Ask (in `outputLanguage`), and proceed only on an explicit yes:
|
|
217
215
|
|
|
@@ -220,27 +218,27 @@ Do you have an operational-reporting token from your admin?
|
|
|
220
218
|
[ Paste token / Skip ]
|
|
221
219
|
```
|
|
222
220
|
|
|
223
|
-
On paste, store
|
|
221
|
+
On paste, store it in the credential store ONLY - never in a file, prefs value,
|
|
222
|
+
git or any synced tree - under the standard per-user name, so it is revocable on
|
|
223
|
+
its own:
|
|
224
224
|
|
|
225
225
|
```bash
|
|
226
226
|
~/.claude/lib/credential-store.sh set "${USER}_Usage_Ingest_Token" "<pasted-token>"
|
|
227
227
|
```
|
|
228
228
|
|
|
229
|
-
Then
|
|
229
|
+
Then, **whether they pasted or skipped**, run:
|
|
230
230
|
|
|
231
231
|
```bash
|
|
232
|
-
node -
|
|
233
|
-
const fs=require("fs"),os=require("os"),p=os.homedir()+"/.claude/multi-agent-preferences.json";
|
|
234
|
-
const j=JSON.parse(fs.readFileSync(p,"utf8"));
|
|
235
|
-
j.global=j.global||{}; j.global.keychainMapping=j.global.keychainMapping||{};
|
|
236
|
-
j.global.keychainMapping.usage_ingest=process.argv[1];
|
|
237
|
-
j.global.usageLog=Object.assign({enabled:true},j.global.usageLog||{},{enabled:true});
|
|
238
|
-
fs.writeFileSync(p,JSON.stringify(j,null,2)+"\n");
|
|
239
|
-
' "${USER}_Usage_Ingest_Token"
|
|
240
|
-
echo " -> operational reporting configured (token in credential store)"
|
|
232
|
+
node "$HOME/.claude/scripts/usage-register.mjs"
|
|
241
233
|
```
|
|
242
234
|
|
|
243
|
-
|
|
235
|
+
It maps whatever token exists and switches reporting on, and requests a
|
|
236
|
+
per-machine write-only one when none resolves. This call is why the step exists:
|
|
237
|
+
registration used to happen only inside `/multi-agent:update`, so a user who
|
|
238
|
+
installed, ran setup and never ran update never registered and never reported -
|
|
239
|
+
which reads in the panel exactly like nobody using the pipeline. What is sent,
|
|
240
|
+
what never is, and `usageLog.optOut`:
|
|
241
|
+
`$HOME/.claude/multi-agent-refs/features/usage-reporting.md`.
|
|
244
242
|
|
|
245
243
|
### Auto-learned fields (no setup step needed)
|
|
246
244
|
|
|
@@ -802,7 +800,7 @@ All tokens are optional in the sense that every service can be answered with Ski
|
|
|
802
800
|
|
|
803
801
|
### Step 8 - Enforcement hooks (optional, Claude Code)
|
|
804
802
|
|
|
805
|
-
Offer to merge `$HOME/.claude/templates/claude-hooks.json`: three `PreToolUse` gates that block on a non-zero exit (secret scan, agent-guard, read-size) plus
|
|
803
|
+
Offer to merge `$HOME/.claude/templates/claude-hooks.json`: three `PreToolUse` gates that block on a non-zero exit (secret scan, agent-guard, read-size) plus three capture hooks that block nothing (`PreCompact`, `SessionEnd`, `SessionStart`). What each does: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
|
|
806
804
|
|
|
807
805
|
- Ask (picker): "Install the pipeline's hooks into `~/.claude/settings.json`?" Default Yes.
|
|
808
806
|
- On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase 7 then loses its findings exactly as it did before they existed. Preserve existing hooks; never duplicate a matcher already calling the same script.
|
|
@@ -9,36 +9,49 @@ Show every active and completed task as a table.
|
|
|
9
9
|
|
|
10
10
|
## Steps
|
|
11
11
|
|
|
12
|
-
1. **
|
|
13
|
-
|
|
14
|
-
- `~/my-figma-app/.worktrees/`
|
|
15
|
-
- `~/my-ui-components/.worktrees/`
|
|
12
|
+
1. **Ask the producer, do not go looking.** One command answers the whole
|
|
13
|
+
question:
|
|
16
14
|
|
|
17
|
-
2. **Scan worktrees** - read `agent-state.json` in each worktree dir. **Also scan the log dir**, because a task whose PR is open has no worktree any more (Phase 6 removes it and salvages its state): `find $HOME/.claude/logs/multi-agent -maxdepth 4 -name agent-state.json -path '*/artifacts/*'`. (`-maxdepth 4`, not 3: Phase 6 always passes `--project`, so the salvaged copy lands at `<project>/<task-id>/artifacts/agent-state.json`, which a depth-3 scan can never reach.) Merge both sets by `taskId`, preferring the worktree copy when both exist, and render a finalized task with its `worktreeRemovedAt` rather than omitting it - a task that shipped should not vanish from status.
|
|
18
15
|
```bash
|
|
19
|
-
|
|
16
|
+
node "$HOME/.claude/scripts/runs-index.mjs" # grouped table
|
|
17
|
+
node "$HOME/.claude/scripts/runs-index.mjs" --json # the same records
|
|
20
18
|
```
|
|
21
19
|
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
20
|
+
`runs-index.mjs` resolves through `lib/run-paths.sh` / `scripts/_run-paths.mjs`,
|
|
21
|
+
so it sees both directory layouts (`<project>/<id>/` and the flat `<id>/`),
|
|
22
|
+
the salvaged `artifacts/` copy Phase 6 leaves behind, and every spelling of a
|
|
23
|
+
task id - and it counts a run that exists in both layouts once. Do NOT
|
|
24
|
+
re-scan the tree by hand: the earlier instruction here listed three
|
|
25
|
+
hard-coded `.worktrees/` paths and a single `find` depth, and on a real
|
|
26
|
+
install that combination missed a quarter of the runs and double-counted
|
|
27
|
+
others. It also scanned worktrees for `agent-state.json`, which Phase 0 has
|
|
28
|
+
never written there ("never inside the worktree", `phases/phase-0-init.md`).
|
|
29
|
+
|
|
30
|
+
The JSON and the table are rendered from the same records, so a dashboard and
|
|
31
|
+
this command cannot disagree.
|
|
32
|
+
|
|
33
|
+
2. **Fields per run** (already in the output): `taskId`, `project`, `branch`,
|
|
34
|
+
`currentPhase`, `status`, `startedAt`, `worktreePath`, `prUrl`, `autopilot`,
|
|
35
|
+
`phases[]`, `tokens`, `estUsd`, `group`, plus `duplicateOf` when the run also
|
|
36
|
+
exists in the other layout and `salvaged` when its state is the Phase 6 copy.
|
|
37
|
+
|
|
38
|
+
3. **Groups are computed, not judged.** `runs-index.mjs` assigns `group` by the
|
|
39
|
+
table below; report what it returns rather than re-deriving it. `in_progress`
|
|
40
|
+
alone cannot tell a run that is waiting for you from one that died, and on
|
|
41
|
+
this machine that difference covered 21 runs.
|
|
31
42
|
|
|
32
43
|
| Group | Test | Action offered |
|
|
33
44
|
|---|---|---|
|
|
34
|
-
| Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase 6
|
|
35
|
-
| Stopped mid-development | anything else past phase 0 | `resume #N` or `kill #N` |
|
|
36
|
-
| Left at a question | phase 0 | `garbage-collect --abandoned` - nothing was built |
|
|
45
|
+
| `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >= 6 | `resume #N` - the work landed, it needs your answer |
|
|
46
|
+
| `stopped` - Stopped mid-development | anything else past phase 0 | `resume #N` or `kill #N` |
|
|
47
|
+
| `question` - Left at a question | phase 0 | `garbage-collect --abandoned` - nothing was built |
|
|
48
|
+
| `unknown` - Status not recorded | no `status` field | say so; offer nothing |
|
|
37
49
|
|
|
38
|
-
A run with no `status` is **not** placed in
|
|
39
|
-
finding, and calling it dead is the same false claim in the other
|
|
50
|
+
A run with no `status` is **not** placed in an actionable group. Unknown is
|
|
51
|
+
not a finding, and calling it dead is the same false claim in the other
|
|
52
|
+
direction.
|
|
40
53
|
|
|
41
|
-
4. **Render as a table**, grouped per
|
|
54
|
+
4. **Render as a table**, grouped per step 3, with the group as a section heading:
|
|
42
55
|
```
|
|
43
56
|
🤖 Multi-Agent Tasks
|
|
44
57
|
|
|
@@ -59,6 +72,24 @@ Show every active and completed task as a table.
|
|
|
59
72
|
|
|
60
73
|
Report `N lesson(s) minable from transcripts - /multi-agent:refactor to review` and `N open pipeline observation(s)` when either count is above zero, and print nothing when both are zero. Neither runs a model and neither writes anything: the miner is dry-run by default. Zero candidates alongside a non-zero `toolResultsExamined` means there is nothing to find; zero of both means the read is broken, and that is worth saying rather than reporting a clean queue.
|
|
61
74
|
|
|
75
|
+
7. **What the spend is doing** (one line, and only when there is something to
|
|
76
|
+
say). `cost-budget-check.mjs` watches ONE run against ONE ceiling, which is
|
|
77
|
+
blind to the two ways a budget actually empties: a drift no single run trips,
|
|
78
|
+
and one pathological session that burns a week in an hour while every run
|
|
79
|
+
stays under its cap.
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
node "$HOME/.claude/scripts/cost-analyze.mjs" burn --json 2>/dev/null
|
|
83
|
+
node "$HOME/.claude/scripts/cost-analyze.mjs" anomaly --days 30 --json 2>/dev/null
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Report a line only when either exits 10 - an accelerating day, or a day out
|
|
87
|
+
of family - and say nothing otherwise. The figures are estimates priced from
|
|
88
|
+
`cost-table.json` at LIST price, so on a subscription they are the right
|
|
89
|
+
number for comparing days to each other and the wrong number to call a bill;
|
|
90
|
+
say which when quoting one. `UNMEASURED` means the transcripts could not be
|
|
91
|
+
read, and it is reported as that rather than as zero.
|
|
92
|
+
|
|
62
93
|
5. **Quick command hints** - based on state:
|
|
63
94
|
- Paused task → suggest `resume #N`
|
|
64
95
|
- Done task → suggest `log #N`
|
|
@@ -111,65 +111,22 @@ A git clone of the pipeline repo is a maintainer workspace, kept in sync by `/mu
|
|
|
111
111
|
fi
|
|
112
112
|
```
|
|
113
113
|
|
|
114
|
-
5b. **Auto-configure operational reporting.**
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
Resolution order: env `MULTI_AGENT_USAGE_TOKEN`, then `usageLog.token`, then
|
|
121
|
-
the Keychain item named by `keychainMapping.usage_ingest`, then
|
|
122
|
-
self-registration. Registration failing (offline, endpoint down, admin turned
|
|
123
|
-
ingest off) leaves reporting off with one status line - never an error.
|
|
124
|
-
**Opt-out is `usageLog.optOut: true`**: it blocks both the auto-enable and the
|
|
125
|
-
self-registration permanently; print the opt-out hint on first auto-enable.
|
|
114
|
+
5b. **Auto-configure operational reporting.** One call, and it is the same call
|
|
115
|
+
`/multi-agent:setup` and a first run make - the registration used to live here
|
|
116
|
+
as forty lines of shell, so a user who installed, ran setup and never ran
|
|
117
|
+
update was never registered and never reported, which reads in the panel
|
|
118
|
+
exactly like nobody using the pipeline.
|
|
119
|
+
|
|
126
120
|
```bash
|
|
127
|
-
|
|
128
|
-
# Reads go through node, not jq. node is a declared engine (>=20.11) so it
|
|
129
|
-
# is always there; jq is not, and gating this block on it meant a machine
|
|
130
|
-
# without jq silently never registered and never reported - which reads in
|
|
131
|
-
# the panel exactly like nobody using the pipeline.
|
|
132
|
-
if [ -f "$PREFS" ]; then
|
|
133
|
-
pref() { node -e 'const fs=require("fs");let v;try{v=process.argv[2].split(".").reduce((a,k)=>a?.[k],JSON.parse(fs.readFileSync(process.argv[1],"utf8")))}catch{};process.stdout.write(v==null?"":String(v))' "$PREFS" "$1" 2>/dev/null; }
|
|
134
|
-
ENABLED=$(pref global.usageLog.enabled)
|
|
135
|
-
OPTOUT=$(pref global.usageLog.optOut)
|
|
136
|
-
if [ "$ENABLED" != "true" ] && [ "$OPTOUT" != "true" ]; then
|
|
137
|
-
TOK="${MULTI_AGENT_USAGE_TOKEN:-}"
|
|
138
|
-
[ -z "$TOK" ] && TOK=$(pref global.usageLog.token)
|
|
139
|
-
if [ -z "$TOK" ]; then
|
|
140
|
-
KNAME=$(pref global.keychainMapping.usage_ingest)
|
|
141
|
-
[ -n "$KNAME" ] && TOK=$(bash "$HOME/.claude/lib/credential-store.sh" get "$KNAME" 2>/dev/null)
|
|
142
|
-
fi
|
|
143
|
-
if [ -z "$TOK" ]; then
|
|
144
|
-
EP=$(pref global.usageLog.endpoint); [ -z "$EP" ] && EP="https://mmerterden.vercel.app/api/usage/ingest"
|
|
145
|
-
REG_EP="${EP%/ingest}/register"
|
|
146
|
-
# Telemetry identity is the GitHub account name, never the git
|
|
147
|
-
# identity.name (which can carry a corporate title). username -> live
|
|
148
|
-
# gh login -> OS user.
|
|
149
|
-
RUSER=$(pref global.identities.0.username)
|
|
150
|
-
[ -z "$RUSER" ] && RUSER=$(gh api user --jq .login 2>/dev/null || echo "")
|
|
151
|
-
[ -z "$RUSER" ] && RUSER="$USER"
|
|
152
|
-
RESP=$(curl -sSL -m 10 -X POST -H "Content-Type: application/json" \
|
|
153
|
-
--data "{\"u\":\"$RUSER\",\"c\":\"$(hostname -s 2>/dev/null || echo unknown)\"}" \
|
|
154
|
-
"$REG_EP" 2>/dev/null)
|
|
155
|
-
TOK=$(printf '%s' "$RESP" | node -e 'let b="";process.stdin.on("data",d=>b+=d).on("end",()=>{try{process.stdout.write(String(JSON.parse(b).token||""))}catch{}})' 2>/dev/null)
|
|
156
|
-
if [ -n "$TOK" ]; then
|
|
157
|
-
printf '%s' "$TOK" | bash "$HOME/.claude/lib/credential-store.sh" set "${USER}_Usage_Ingest_Token" "$(cat)"
|
|
158
|
-
node -e 'const fs=require("fs"),p=process.argv[1];const j=JSON.parse(fs.readFileSync(p,"utf8"));j.global=j.global||{};j.global.keychainMapping=j.global.keychainMapping||{};j.global.keychainMapping.usage_ingest=process.argv[2];fs.writeFileSync(p,JSON.stringify(j,null,2)+"\n");' "$PREFS" "${USER}_Usage_Ingest_Token"
|
|
159
|
-
echo " -> operational reporting: registered this machine (write-only token in credential store)"
|
|
160
|
-
echo " opt out any time: set global.usageLog.optOut=true in multi-agent-preferences.json"
|
|
161
|
-
fi
|
|
162
|
-
fi
|
|
163
|
-
if [ -n "$TOK" ]; then
|
|
164
|
-
node -e 'const fs=require("fs"),p=process.argv[1];const j=JSON.parse(fs.readFileSync(p,"utf8"));j.global=j.global||{};j.global.usageLog=j.global.usageLog||{};j.global.usageLog.enabled=true;fs.writeFileSync(p,JSON.stringify(j,null,2)+"\n");' "$PREFS"
|
|
165
|
-
echo " -> operational reporting configured"
|
|
166
|
-
else
|
|
167
|
-
echo " -> operational reporting left off (no token; registration unreachable or disabled)"
|
|
168
|
-
fi
|
|
169
|
-
fi
|
|
170
|
-
fi
|
|
121
|
+
node "$HOME/.claude/scripts/usage-register.mjs"
|
|
171
122
|
```
|
|
172
123
|
|
|
124
|
+
It requests a per-machine **write-only** token when none resolves, stores it
|
|
125
|
+
only in the credential store, and flips `usageLog.enabled` on. `usageLog.optOut:
|
|
126
|
+
true` blocks it permanently. Offline, endpoint down or ingest disabled leaves
|
|
127
|
+
reporting off with one status line - never an error. Contract and the data it
|
|
128
|
+
sends: `$HOME/.claude/multi-agent-refs/features/usage-reporting.md`.
|
|
129
|
+
|
|
173
130
|
6. **Show the new version and its changes** (from the packaged CHANGELOG - there is no git history on this channel):
|
|
174
131
|
```bash
|
|
175
132
|
NEW=$(tr -d '[:space:]' < "$HOME/.claude/.pipeline-version" 2>/dev/null)
|
|
@@ -40,6 +40,14 @@ JIRA_AUTH_PREFS="${JIRA_AUTH_PREFS:-$HOME/.claude/multi-agent-preferences.json}"
|
|
|
40
40
|
|
|
41
41
|
_jira_auth_pref() { # _jira_auth_pref <jq path> -> value or empty
|
|
42
42
|
[ -f "$JIRA_AUTH_PREFS" ] || { printf ''; return 0; }
|
|
43
|
+
# Without jq this printed empty and returned 0, so "no jq" and "key not set"
|
|
44
|
+
# were the same answer and the caller went on to authenticate with nothing -
|
|
45
|
+
# surfacing much later as a 401 that blames the credential.
|
|
46
|
+
if ! command -v jq >/dev/null 2>&1; then
|
|
47
|
+
echo "jq not found - cannot read $JIRA_AUTH_PREFS (install: brew install jq)" >&2
|
|
48
|
+
printf ''
|
|
49
|
+
return 3
|
|
50
|
+
fi
|
|
43
51
|
jq -r "$1 // empty" "$JIRA_AUTH_PREFS" 2>/dev/null || printf ''
|
|
44
52
|
}
|
|
45
53
|
|
|
@@ -125,6 +125,30 @@ fi
|
|
|
125
125
|
# back in CREATED_KEY. Returning the key on stdout too meant a caller using
|
|
126
126
|
# command substitution swallowed the report - the first dry run printed a header
|
|
127
127
|
# and nothing else, and the tree looked empty.
|
|
128
|
+
# Same gate, same five candidate paths, same refusal as jira-publish.sh and
|
|
129
|
+
# post-pr-review.sh. This file is the only ISSUE-CREATION path in the tree, and
|
|
130
|
+
# a summary plus a description is outbound text like any other - it was the one
|
|
131
|
+
# writer the gate did not cover.
|
|
132
|
+
ma_outbound_gate_text() {
|
|
133
|
+
local text="$1" og="" tmp rc
|
|
134
|
+
for c in "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)/outbound-gate.mjs" \
|
|
135
|
+
"$HOME/.claude/lib/outbound-gate.mjs" \
|
|
136
|
+
"$HOME/.copilot/lib/outbound-gate.mjs" \
|
|
137
|
+
"$HOME/.codex/lib/outbound-gate.mjs"; do
|
|
138
|
+
[ -f "$c" ] && { og="$c"; break; }
|
|
139
|
+
done
|
|
140
|
+
if [ -z "$og" ]; then
|
|
141
|
+
echo "outbound-gate.mjs not found - refusing to create unchecked issues." >&2
|
|
142
|
+
return 7
|
|
143
|
+
fi
|
|
144
|
+
tmp="$(mktemp)"
|
|
145
|
+
printf '%s' "$text" > "$tmp"
|
|
146
|
+
node "$og" --file "$tmp"
|
|
147
|
+
rc=$?
|
|
148
|
+
rm -f "$tmp"
|
|
149
|
+
return $rc
|
|
150
|
+
}
|
|
151
|
+
|
|
128
152
|
CREATED_KEY=""
|
|
129
153
|
create_issue() { # create_issue <label> <summary> <issuetype> <parentKey|""> <description>
|
|
130
154
|
local label="$1" summary="$2" itype="$3" parent="$4" desc="$5" body key existing
|
|
@@ -146,6 +170,14 @@ create_issue() { # create_issue <label> <summary> <issuetype> <parentKey|""> <d
|
|
|
146
170
|
echo " create $itype $summary [$label]"
|
|
147
171
|
return 0
|
|
148
172
|
fi
|
|
173
|
+
# After the dry-run branch, because a dry run publishes nothing, and before
|
|
174
|
+
# the ledger line, because an intent recorded for a POST that never happens
|
|
175
|
+
# is a false entry in the only record of what was attempted.
|
|
176
|
+
if ! ma_outbound_gate_text "$summary
|
|
177
|
+
$desc"; then
|
|
178
|
+
echo "ERR: outbound gate refused '$summary'; nothing was created" >&2
|
|
179
|
+
return 1
|
|
180
|
+
fi
|
|
149
181
|
ledger intent "$label"
|
|
150
182
|
local resp
|
|
151
183
|
resp="$(printf '%s' "$body" | jira_api POST "/rest/api/2/issue" --data @- || echo "")"
|
|
@@ -14,6 +14,9 @@
|
|
|
14
14
|
#
|
|
15
15
|
# Non-interactive / autopilot / CI:
|
|
16
16
|
# - ASK_CHOICE_DEFAULT=<label-or-1based-index> picks without prompting.
|
|
17
|
+
# - MULTI_AGENT_UNATTENDED=1 means nobody is watching even if a terminal is
|
|
18
|
+
# attached (screen, tmux, a login shell on a server). Same resolution as
|
|
19
|
+
# the no-TTY case.
|
|
17
20
|
# - If stdin is not a TTY and no default is set, the FIRST option is chosen
|
|
18
21
|
# and a notice is written to stderr (never blocks an automated run).
|
|
19
22
|
#
|
|
@@ -64,8 +67,16 @@ if [ -n "${ASK_CHOICE_DEFAULT:-}" ]; then
|
|
|
64
67
|
fi
|
|
65
68
|
|
|
66
69
|
# Non-interactive with no usable default: pick the first option, don't block.
|
|
67
|
-
|
|
68
|
-
|
|
70
|
+
#
|
|
71
|
+
# "No TTY" is the usual shape of that, but it is not the only one. A server run
|
|
72
|
+
# under screen, tmux or a login shell HAS a terminal and still has nobody in
|
|
73
|
+
# front of it, and there the TTY test says "ask" and the process waits forever.
|
|
74
|
+
# MULTI_AGENT_UNATTENDED=1 is the operator saying so out loud; see
|
|
75
|
+
# refs/unattended-contract.md. Unset, nothing below changes.
|
|
76
|
+
if [ ! -t 0 ] || [ "${MULTI_AGENT_UNATTENDED:-}" = "1" ]; then
|
|
77
|
+
why="no TTY"
|
|
78
|
+
[ "${MULTI_AGENT_UNATTENDED:-}" = "1" ] && why="MULTI_AGENT_UNATTENDED=1"
|
|
79
|
+
echo "ask-choice: $why and no ASK_CHOICE_DEFAULT - selecting first option '${OPTIONS[0]}'" >&2
|
|
69
80
|
printf '%s\n' "${OPTIONS[0]}"
|
|
70
81
|
exit 0
|
|
71
82
|
fi
|
|
@@ -133,6 +133,14 @@ ma_ap_read() { # $1 = filename; empty and exit 1 when absent
|
|
|
133
133
|
# jq with a default, so a caller never has to distinguish "key absent" from
|
|
134
134
|
# "file absent" from "file unparseable" - all three mean "use the default".
|
|
135
135
|
ma_ap_cfg() { # $1 = jq path, $2 = default
|
|
136
|
+
# An unreadable queue must not read as an EMPTY queue. Without this, a
|
|
137
|
+
# machine with no jq made the runner conclude there was no work and go
|
|
138
|
+
# quiet - the worst failure this mode can have, because it looks like
|
|
139
|
+
# success.
|
|
140
|
+
if ! command -v jq >/dev/null 2>&1; then
|
|
141
|
+
echo "jq not found - cannot read the autopilot config (install: brew install jq)" >&2
|
|
142
|
+
return 3
|
|
143
|
+
fi
|
|
136
144
|
local v
|
|
137
145
|
v=$(ma_ap_read config.json 2>/dev/null | jq -r "$1 // empty" 2>/dev/null)
|
|
138
146
|
[ -n "$v" ] && printf '%s\n' "$v" || printf '%s\n' "$2"
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* fatal.mjs - one line instead of a stack dump, and whatever was held gets
|
|
3
|
+
* released.
|
|
4
|
+
*
|
|
5
|
+
* Node's default for an uncaught throw or a rejected promise nobody awaited is
|
|
6
|
+
* a stack trace on stderr and exit 1. That is the right default for a library
|
|
7
|
+
* and the wrong one here for two reasons.
|
|
8
|
+
*
|
|
9
|
+
* The first is legibility. Every script in this tree reports its own failures
|
|
10
|
+
* as `name: what went wrong`, one line, and a caller - a shell gate, a phase
|
|
11
|
+
* step, a person reading a run log - is written against that shape. A stack
|
|
12
|
+
* dump in the middle of it reads as a crash of the pipeline rather than as a
|
|
13
|
+
* failure of one step, and the actual message sits four frames down.
|
|
14
|
+
*
|
|
15
|
+
* The second is state, and it is the one that costs something. A script that
|
|
16
|
+
* dies between acquiring an advisory lock and releasing it leaves the lock on
|
|
17
|
+
* disk. The next writer then waits out the full acquire window before deciding
|
|
18
|
+
* the holder is stale - and if it judges wrong it takes a lock a LIVE writer
|
|
19
|
+
* holds. `cleanup` exists for exactly that: the handler releases before it
|
|
20
|
+
* exits, so the abnormal path leaves the same state behind as the normal one.
|
|
21
|
+
*
|
|
22
|
+
* What this does NOT do is swallow anything. Every path still exits non-zero,
|
|
23
|
+
* and the message still names the error; only the stack goes, and only when
|
|
24
|
+
* the error carries a message worth reading on its own.
|
|
25
|
+
*
|
|
26
|
+
* Usage:
|
|
27
|
+
* import { runMain } from "../lib/fatal.mjs";
|
|
28
|
+
* runMain("write-state", main, { cleanup: releaseLock });
|
|
29
|
+
*
|
|
30
|
+
* @module pipeline/lib/fatal
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
/** Guards against a cleanup that runs twice when a throw follows a rejection. */
|
|
34
|
+
let cleanedUp = false;
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Run `cleanup` at most once, and never let it become the failure it is
|
|
38
|
+
* cleaning up after: a cleanup that throws inside a fatal handler replaces the
|
|
39
|
+
* original error with its own, which is how a lock bug ends up reported as a
|
|
40
|
+
* permissions bug.
|
|
41
|
+
*
|
|
42
|
+
* @param {(() => void)|undefined} cleanup
|
|
43
|
+
*/
|
|
44
|
+
function runCleanup(cleanup) {
|
|
45
|
+
if (cleanedUp || typeof cleanup !== "function") return;
|
|
46
|
+
cleanedUp = true;
|
|
47
|
+
try {
|
|
48
|
+
cleanup();
|
|
49
|
+
} catch {
|
|
50
|
+
// Deliberately silent. The caller is already exiting with the real error.
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* The message a human should see. An Error with a message prints the message;
|
|
56
|
+
* anything else (a thrown string, a rejected non-Error) prints its own
|
|
57
|
+
* stringification, because dropping it would leave the line with nothing in it.
|
|
58
|
+
*
|
|
59
|
+
* @param {unknown} err
|
|
60
|
+
* @returns {string}
|
|
61
|
+
*/
|
|
62
|
+
function describe(err) {
|
|
63
|
+
if (err instanceof Error && err.message) return err.message;
|
|
64
|
+
if (typeof err === "string" && err) return err;
|
|
65
|
+
try {
|
|
66
|
+
return JSON.stringify(err);
|
|
67
|
+
} catch {
|
|
68
|
+
return String(err);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Catch what escapes: a rejected promise nobody awaited, and a throw from a
|
|
74
|
+
* callback that no try/catch surrounds. Both are invisible to a try/catch
|
|
75
|
+
* around main(), which is why wrapping main() alone is not enough.
|
|
76
|
+
*
|
|
77
|
+
* MULTI_AGENT_FATAL_STACK=1 restores the stack for debugging. It is off by
|
|
78
|
+
* default because the line is for the person running the pipeline, not for the
|
|
79
|
+
* person maintaining it.
|
|
80
|
+
*
|
|
81
|
+
* @param {{name:string, cleanup?:()=>void, code?:number}} opts
|
|
82
|
+
*/
|
|
83
|
+
export function installFatalHandlers({ name, cleanup, code = 1 }) {
|
|
84
|
+
const report = (kind, err) => {
|
|
85
|
+
runCleanup(cleanup);
|
|
86
|
+
// `runs-index.mjs --json | head` is not a failure: head closes the pipe and
|
|
87
|
+
// the next write raises EPIPE. Reporting it as a fatal would turn the most
|
|
88
|
+
// ordinary way of looking at a large output into a red line and a non-zero
|
|
89
|
+
// exit, and the reader that went away is not listening to the complaint
|
|
90
|
+
// anyway. Quiet, zero, the way every other CLI treats it.
|
|
91
|
+
if (err && err.code === "EPIPE") process.exit(0);
|
|
92
|
+
process.stderr.write(`${name}: ${kind} - ${describe(err)}\n`);
|
|
93
|
+
if (process.env.MULTI_AGENT_FATAL_STACK === "1" && err instanceof Error && err.stack) {
|
|
94
|
+
process.stderr.write(`${err.stack}\n`);
|
|
95
|
+
}
|
|
96
|
+
// exit(), not exitCode: the event loop may hold a listener that would keep
|
|
97
|
+
// the process alive after the failure, and a script that has already
|
|
98
|
+
// reported a fatal must not go on to do more work.
|
|
99
|
+
process.exit(code);
|
|
100
|
+
};
|
|
101
|
+
process.on("unhandledRejection", (err) => report("unhandled rejection", err));
|
|
102
|
+
process.on("uncaughtException", (err) => report("uncaught exception", err));
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Install the handlers and run main, sync or async, with the same contract for
|
|
107
|
+
* both. A successful run is untouched: no exit call, so `process.exitCode` set
|
|
108
|
+
* inside main survives, and stdout drains the way it would have.
|
|
109
|
+
*
|
|
110
|
+
* @param {string} name script name as it should appear in the message
|
|
111
|
+
* @param {() => unknown} main
|
|
112
|
+
* @param {{cleanup?:()=>void, code?:number}} [opts]
|
|
113
|
+
* @returns {Promise<void>}
|
|
114
|
+
*/
|
|
115
|
+
export async function runMain(name, main, opts = {}) {
|
|
116
|
+
const { cleanup, code = 1 } = opts;
|
|
117
|
+
installFatalHandlers({ name, cleanup, code });
|
|
118
|
+
try {
|
|
119
|
+
await main();
|
|
120
|
+
} catch (err) {
|
|
121
|
+
runCleanup(cleanup);
|
|
122
|
+
if (err && err.code === "EPIPE") process.exit(0);
|
|
123
|
+
process.stderr.write(`${name}: ${describe(err)}\n`);
|
|
124
|
+
if (process.env.MULTI_AGENT_FATAL_STACK === "1" && err instanceof Error && err.stack) {
|
|
125
|
+
process.stderr.write(`${err.stack}\n`);
|
|
126
|
+
}
|
|
127
|
+
process.exit(code);
|
|
128
|
+
}
|
|
129
|
+
}
|