@mmerterden/multi-agent-pipeline 17.6.0 → 18.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/README.md +43 -1
- package/README.tr.md +41 -0
- package/docs/adr/0011-dormant-ci.md +25 -1
- package/docs/server-readiness.md +188 -0
- package/index.js +16 -1
- package/install/_common.mjs +42 -17
- package/install/_dev-only-files.mjs +8 -0
- package/install/_unattended-profile.mjs +113 -0
- package/install/index.mjs +48 -0
- package/manifest.json +1049 -0
- package/package.json +5 -2
- package/pipeline/commands/multi-agent/status/SKILL.md +52 -21
- package/pipeline/lib/_jira-auth.sh +8 -0
- package/pipeline/lib/analysis-jira-write.sh +32 -0
- package/pipeline/lib/ask-choice.sh +13 -2
- package/pipeline/lib/autopilot-state.sh +8 -0
- package/pipeline/lib/fatal.mjs +129 -0
- package/pipeline/lib/figma-mcp-refresh.sh +18 -0
- package/pipeline/lib/figma-screenshot.sh +18 -0
- package/pipeline/lib/invoked-directly.mjs +43 -0
- package/pipeline/lib/jira-publish.sh +42 -0
- package/pipeline/lib/md2confluence-v3.py +47 -0
- package/pipeline/lib/outbound-gate.mjs +175 -0
- package/pipeline/lib/plan-todos.sh +27 -6
- package/pipeline/lib/post-pr-review.sh +77 -8
- package/pipeline/lib/repo-hygiene.sh +8 -3
- package/pipeline/lib/require-jq.sh +40 -0
- package/pipeline/lib/run-paths.sh +335 -0
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +70 -0
- package/pipeline/multi-agent-refs/features/cost-analysis.md +93 -0
- package/pipeline/multi-agent-refs/features/doctor.md +45 -0
- package/pipeline/multi-agent-refs/features/verify.md +83 -0
- package/pipeline/multi-agent-refs/phases/operations.md +13 -2
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -1
- package/pipeline/multi-agent-refs/unattended-contract.md +129 -0
- package/pipeline/scripts/_run-paths.mjs +372 -0
- package/pipeline/scripts/aggregate-metrics.mjs +64 -64
- package/pipeline/scripts/autopilot-arming.mjs +2 -1
- package/pipeline/scripts/autopilot-intake.mjs +2 -1
- package/pipeline/scripts/autopilot-runner.mjs +206 -2
- package/pipeline/scripts/build-references.mjs +2 -1
- package/pipeline/scripts/build-stack-plugins.mjs +10 -2
- package/pipeline/scripts/capture-evidence.sh +7 -2
- package/pipeline/scripts/classify-plan-safety.mjs +2 -1
- package/pipeline/scripts/cost-analyze.mjs +600 -0
- package/pipeline/scripts/cost-budget-check.mjs +4 -12
- package/pipeline/scripts/council-view.mjs +2 -1
- package/pipeline/scripts/crush-json.mjs +2 -1
- package/pipeline/scripts/diff-explain.mjs +6 -9
- package/pipeline/scripts/diff-risk-score.mjs +2 -1
- package/pipeline/scripts/doctor.mjs +138 -4
- package/pipeline/scripts/evidence-gate.mjs +9 -3
- package/pipeline/scripts/feedback-send.mjs +12 -2
- package/pipeline/scripts/gc-abandoned.sh +29 -13
- package/pipeline/scripts/gc-worktrees.sh +11 -4
- package/pipeline/scripts/github-ssh-setup.sh +64 -7
- package/pipeline/scripts/graph-mermaid.mjs +4 -2
- package/pipeline/scripts/keychain-save.sh +101 -30
- package/pipeline/scripts/learn-from-transcripts.mjs +2 -1
- package/pipeline/scripts/learning-curve.mjs +34 -29
- package/pipeline/scripts/make-manifest.mjs +199 -0
- package/pipeline/scripts/migrate-prefs.mjs +2 -1
- package/pipeline/scripts/migrate-state.mjs +94 -4
- package/pipeline/scripts/phase-banner.sh +6 -2
- package/pipeline/scripts/phase-tracker.sh +41 -3
- package/pipeline/scripts/plan-coverage-gate.mjs +6 -2
- package/pipeline/scripts/pre-commit-check.sh +7 -0
- package/pipeline/scripts/pre-push-check.sh +7 -0
- package/pipeline/scripts/purge.sh +23 -6
- package/pipeline/scripts/render-agent-log-cost.sh +9 -2
- package/pipeline/scripts/render-cost-summary.sh +9 -2
- package/pipeline/scripts/render-work-summary.sh +11 -4
- package/pipeline/scripts/review-file-filter.mjs +4 -2
- package/pipeline/scripts/review-scope.mjs +2 -1
- package/pipeline/scripts/routine-registry.mjs +2 -1
- package/pipeline/scripts/run-aggregator.mjs +13 -14
- package/pipeline/scripts/run-metrics.mjs +3 -1
- package/pipeline/scripts/runs-index.mjs +343 -0
- package/pipeline/scripts/scorecard-snapshot.mjs +178 -0
- package/pipeline/scripts/search-logs.sh +18 -0
- package/pipeline/scripts/test-gap-scan.mjs +2 -1
- package/pipeline/scripts/test-integrity-gate.mjs +2 -1
- package/pipeline/scripts/update-issue-progress.sh +56 -7
- package/pipeline/scripts/usage-report.mjs +12 -1
- package/pipeline/scripts/validate-analysis-doc.mjs +2 -1
- package/pipeline/scripts/validate-code-graph.mjs +6 -3
- package/pipeline/scripts/validate-complaint-doc.mjs +2 -1
- package/pipeline/scripts/validate-diff-risk.mjs +6 -3
- package/pipeline/scripts/validate-test-gap.mjs +6 -3
- package/pipeline/scripts/validate-triage.mjs +3 -1
- package/pipeline/scripts/verify-citations.mjs +4 -2
- package/pipeline/scripts/verify.mjs +327 -0
- package/pipeline/scripts/worktree-finalize.sh +13 -4
- package/pipeline/scripts/write-state.mjs +154 -15
- package/pipeline/skills/.skill-manifest.json +2 -2
- package/pipeline/skills/.skills-index.json +56 -1
- package/pipeline/skills/shared/README.md +8 -3
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +33 -9
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +2 -1
- package/pipeline/skills/skills-index.md +6 -1
|
@@ -514,7 +514,7 @@ done
|
|
|
514
514
|
|
|
515
515
|
State file in multi-repo mode:
|
|
516
516
|
- Single shared `agent-state.json` lives at `$HOME/.claude/logs/multi-agent/{first-project}/{task-id}/agent-state.json` (anchored on the first repo for back-compat with `multi-agent log`/`status` commands)
|
|
517
|
-
- Every
|
|
517
|
+
- Every write to it, creation included, goes through `node $HOME/.claude/scripts/write-state.mjs` - the required mechanism in `operations.md` "Writing `agent-state.json`", and the race a per-repo read-modify-write loses `projects[]` entries to.
|
|
518
518
|
- `state.projects[]` holds per-repo `{name, root, worktreePath, branch, baseBranch, identity, platform, baseFetchStatus, commit, pr, pushAttempts, buildStatus}` - see `agent-state.schema.json`
|
|
519
519
|
- Scalar fields (`project`, `projectRoot`, `worktreePath`, `branch`, `baseBranch`, `identity`) mirror `projects[0]` so legacy phases that read scalars keep working
|
|
520
520
|
- Atomicity: if any repo's worktree creation fails (collision aborted, fetch aborted, disk full), roll back already-created worktrees: `git -C $proj worktree remove --force $WT_PATH; git -C $proj branch -D $BRANCH`. Never leave a partial multi-repo state.
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# Unattended contract (`MULTI_AGENT_UNATTENDED=1`)
|
|
2
|
+
|
|
3
|
+
<!-- toc -->
|
|
4
|
+
- [What the variable means](#what-the-variable-means)
|
|
5
|
+
- [The three effects](#the-three-effects)
|
|
6
|
+
- [What it does NOT do](#what-it-does-not-do)
|
|
7
|
+
- [Who honours it](#who-honours-it)
|
|
8
|
+
- [The model side already has a contract](#the-model-side-already-has-a-contract)
|
|
9
|
+
- [Default behaviour is unchanged, and a gate says so](#default-behaviour-is-unchanged-and-a-gate-says-so)
|
|
10
|
+
- [The permission posture](#the-permission-posture)
|
|
11
|
+
<!-- /toc -->
|
|
12
|
+
|
|
13
|
+
## What the variable means
|
|
14
|
+
|
|
15
|
+
`MULTI_AGENT_UNATTENDED=1` is the operator stating that **nobody is watching
|
|
16
|
+
this process**, and it is the only way to state it reliably.
|
|
17
|
+
|
|
18
|
+
The usual test for that is `[ -t 0 ]`: no terminal, no human. It is wrong in
|
|
19
|
+
both directions on a server. A run under `screen`, `tmux`, `ssh -t` or a login
|
|
20
|
+
shell HAS a terminal and still has nobody in front of it, so the test says
|
|
21
|
+
"ask" and the process waits for an answer that will never come - printing
|
|
22
|
+
nothing, exiting never, and looking exactly like slow work. That is the failure
|
|
23
|
+
this contract exists to remove, and it is not hypothetical: two setup scripts
|
|
24
|
+
shipped that way.
|
|
25
|
+
|
|
26
|
+
The variable is opt-in and defaults to absent. Nothing in this file describes
|
|
27
|
+
behaviour that changes on a machine that does not set it.
|
|
28
|
+
|
|
29
|
+
## The three effects
|
|
30
|
+
|
|
31
|
+
**1. No prompt blocks.** Every shell entry point that asks a question either
|
|
32
|
+
resolves it from a documented default or refuses with a reason on stderr and a
|
|
33
|
+
non-zero exit. Waiting is never one of the outcomes. Refusing beats hanging:
|
|
34
|
+
one of them can be read in a log.
|
|
35
|
+
|
|
36
|
+
**2. Output is greppable.** Colour is off even when stdout is a terminal.
|
|
37
|
+
ANSI escapes in a log file make it unsearchable, and on a server the terminal
|
|
38
|
+
that is attached is not the one anyone reads.
|
|
39
|
+
|
|
40
|
+
**3. A secret is never a prompt.** Credentials arrive through stdin or a file
|
|
41
|
+
path, never through an interactive read and never through an environment
|
|
42
|
+
variable. An env value is inherited by every child process and is visible to
|
|
43
|
+
`ps e` on some systems; a pipe stays in the one process that needs it.
|
|
44
|
+
|
|
45
|
+
## What it does NOT do
|
|
46
|
+
|
|
47
|
+
- It does not grant permissions. An unattended run still needs a permission
|
|
48
|
+
posture, which is a separate opt-in (`install --unattended`) and a separate
|
|
49
|
+
doctor check.
|
|
50
|
+
- It does not suppress errors. A run that cannot proceed still fails; it just
|
|
51
|
+
fails visibly instead of hanging.
|
|
52
|
+
- It does not change any default. With the variable unset, every script below
|
|
53
|
+
behaves exactly as it did before this contract existed.
|
|
54
|
+
|
|
55
|
+
## Who honours it
|
|
56
|
+
|
|
57
|
+
Paths below are install-relative: `lib/` and `scripts/` under the host root,
|
|
58
|
+
which is `~/.claude`, `~/.copilot` or `~/.codex` depending on the install. A
|
|
59
|
+
ref that ships to users must not name a checkout path, because a run happens
|
|
60
|
+
in the user's worktree and the checkout is not there.
|
|
61
|
+
|
|
62
|
+
| Entry point | Without the variable | With `MULTI_AGENT_UNATTENDED=1` |
|
|
63
|
+
|---|---|---|
|
|
64
|
+
| `lib/ask-choice.sh` | TTY: renders the menu and reads. No TTY: first option, notice on stderr | First option (or `ASK_CHOICE_DEFAULT`), never prompts |
|
|
65
|
+
| `scripts/github-ssh-setup.sh` | TTY: three questions. No TTY: refuses, naming `SSH_SETUP_EMAIL` | Refuses the same way even with a terminal |
|
|
66
|
+
| `scripts/keychain-save.sh` | TTY: menu + secret prompt. No TTY: refuses, naming `--stdin` / `--json` | Refuses the same way even with a terminal |
|
|
67
|
+
| `scripts/phase-banner.sh` | Colour when stdout is a TTY and `TERM != dumb` | Plain text |
|
|
68
|
+
|
|
69
|
+
Anything not in this table does not read the variable. That is deliberate: a
|
|
70
|
+
list of four that is true beats a claim of coverage that is not.
|
|
71
|
+
|
|
72
|
+
## The model side already has a contract
|
|
73
|
+
|
|
74
|
+
The prompt-level question - what an agent does when it would call
|
|
75
|
+
`AskUserQuestion` and no one can answer - is `refs/picker-contract.md`, section
|
|
76
|
+
"Autopilot / non-interactive contract", and it predates this file. It resolves
|
|
77
|
+
from the remembered choice first, then the documented default, and records
|
|
78
|
+
which rule fired so the run stays readable afterwards.
|
|
79
|
+
|
|
80
|
+
The two are different layers and should not be merged. The picker contract
|
|
81
|
+
governs a model deciding; this file governs a process waiting. A run on a
|
|
82
|
+
server needs both, and only one of them can be enforced by a gate.
|
|
83
|
+
|
|
84
|
+
## Default behaviour is unchanged, and a gate says so
|
|
85
|
+
|
|
86
|
+
`smoke-unattended-profile.sh` (a maintainer gate, not shipped) asserts both directions. With the variable set,
|
|
87
|
+
each entry point above resolves or refuses under a hard timeout. With it unset,
|
|
88
|
+
each one produces byte-identical output to the behaviour it had before - the
|
|
89
|
+
assertion that matters most, because the whole point is that a local
|
|
90
|
+
interactive machine is not affected by any of this.
|
|
91
|
+
|
|
92
|
+
## The permission posture
|
|
93
|
+
|
|
94
|
+
Everything above concerns a process that would otherwise WAIT. There is a second
|
|
95
|
+
way an unattended run stops, and it does not wait at all.
|
|
96
|
+
|
|
97
|
+
autopilot spawns its child with `--permission-prompts none`. That stops Claude
|
|
98
|
+
Code from ASKING; it grants nothing. The tools the child then calls still have
|
|
99
|
+
to be allowed, and on a fresh machine they are not - so the run stops at the
|
|
100
|
+
first tool call, with no prompt anywhere for a person to answer. From the
|
|
101
|
+
outside it is indistinguishable from a queue with nothing to do.
|
|
102
|
+
|
|
103
|
+
`install --unattended` writes the profile that closes it:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
npx @mmerterden/multi-agent-pipeline install --unattended --dry-run # show it
|
|
107
|
+
npx @mmerterden/multi-agent-pipeline install --unattended # write it
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Three properties, each one a promise to somebody who did not ask for this:
|
|
111
|
+
|
|
112
|
+
1. A DEFAULT install writes no permissions at all. Widening a permission set is
|
|
113
|
+
not something an installer does to a person who wanted files copied.
|
|
114
|
+
2. The profile is printed in full, with a reason per line, BEFORE anything is
|
|
115
|
+
written. "The installer changed my permissions" must never be a thing
|
|
116
|
+
discovered afterwards.
|
|
117
|
+
3. It is additive and idempotent. An entry a person added by hand survives, a
|
|
118
|
+
narrower rule is kept alongside, unrelated settings are untouched, and a
|
|
119
|
+
second run changes nothing. A `settings.json` that does not parse is refused
|
|
120
|
+
rather than overwritten.
|
|
121
|
+
|
|
122
|
+
The entries are broad, and that breadth is what unattended operation costs
|
|
123
|
+
rather than an oversight: a run builds and tests whatever the target repo uses,
|
|
124
|
+
so an allowlist narrow enough to be interesting is one the first unfamiliar repo
|
|
125
|
+
stops at. `doctor --profile=server` reports `unattended-permissions` against the
|
|
126
|
+
same rule the writer applies, so the two cannot disagree.
|
|
127
|
+
|
|
128
|
+
`smoke-unattended-install-profile.sh` asserts all of it, including that a plain
|
|
129
|
+
install still writes nothing.
|
|
@@ -0,0 +1,372 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file _run-paths.mjs - the single resolver for pipeline run-state paths.
|
|
3
|
+
*
|
|
4
|
+
* A run's files live under the log root in ONE of two layouts:
|
|
5
|
+
*
|
|
6
|
+
* nested <root>/<project>/<taskId>/ documented canonical; what
|
|
7
|
+
* agent-state.json, purge.sh's
|
|
8
|
+
* per-project .counter and
|
|
9
|
+
* prune-logs.sh reason about
|
|
10
|
+
* flat <root>/<taskId>/ what phase-tracker.sh has always
|
|
11
|
+
* written tracker-state.json to
|
|
12
|
+
*
|
|
13
|
+
* Both are real and both are populated, so every reader has to accept both.
|
|
14
|
+
* Before this module three callers hand-rolled their own candidate list -
|
|
15
|
+
* run-aggregator.mjs, render-cost-summary.sh and render-agent-log-cost.sh -
|
|
16
|
+
* and the single-level `<root>/*\/` globs elsewhere saw only one layout, so a
|
|
17
|
+
* run could be counted twice or missed entirely depending on which glob ran.
|
|
18
|
+
*
|
|
19
|
+
* This module does NOT change where anything is written. Measured on a real
|
|
20
|
+
* install, 90 of 103 runs are flat and every tracker-state.json is, so moving
|
|
21
|
+
* the writers would relocate the majority layout to satisfy a document. READS
|
|
22
|
+
* accept both, newest wins, and a taskId present in both layouts is ONE run,
|
|
23
|
+
* not two. Relocation is opt-in and explicit: `migrate-state.mjs --relocate`.
|
|
24
|
+
*
|
|
25
|
+
* Zero runtime dependencies (ADR-0004). Shell twin: pipeline/lib/run-paths.sh -
|
|
26
|
+
* the two must agree, and smoke-run-path-canonical.sh asserts that they do.
|
|
27
|
+
*
|
|
28
|
+
* @module pipeline/scripts/_run-paths
|
|
29
|
+
*/
|
|
30
|
+
|
|
31
|
+
import { existsSync, lstatSync, readdirSync, readFileSync, realpathSync, statSync } from "node:fs";
|
|
32
|
+
import { join } from "node:path";
|
|
33
|
+
import { homedir } from "node:os";
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Files that mark a directory as a run directory rather than a project
|
|
37
|
+
* directory. A project directory holds run directories; a run directory holds
|
|
38
|
+
* at least one of these.
|
|
39
|
+
*/
|
|
40
|
+
export const RUN_MARKERS = ["agent-state.json", "tracker-state.json", "agent-log.md"];
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Depth-1 names under the log root that are namespaces, not runs and not
|
|
44
|
+
* projects. They are skipped by listRuns so a namespace never surfaces as a
|
|
45
|
+
* phantom run with no phase.
|
|
46
|
+
*/
|
|
47
|
+
export const RESERVED_DIRS = new Set([
|
|
48
|
+
"review-watch",
|
|
49
|
+
"jira-backups",
|
|
50
|
+
"shadow-git",
|
|
51
|
+
"_analysis-jira",
|
|
52
|
+
"prompts",
|
|
53
|
+
]);
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* The log root. LOGS_ROOT overrides it, which is what the smokes use so they
|
|
57
|
+
* never touch a real run.
|
|
58
|
+
*
|
|
59
|
+
* @returns {string}
|
|
60
|
+
*/
|
|
61
|
+
export function logsRoot() {
|
|
62
|
+
return process.env.LOGS_ROOT || join(homedir(), ".claude", "logs", "multi-agent");
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* The path a NEW run's files should be written to.
|
|
67
|
+
*
|
|
68
|
+
* @param {string} taskId
|
|
69
|
+
* @param {string|null|undefined} project omitted only when the project is
|
|
70
|
+
* genuinely unknown, which yields the flat path rather than inventing a slug
|
|
71
|
+
* @returns {string}
|
|
72
|
+
*/
|
|
73
|
+
export function canonicalRunDir(taskId, project) {
|
|
74
|
+
const root = logsRoot();
|
|
75
|
+
return project ? join(root, project, taskId) : join(root, taskId);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Phase 6 removes the worktree and salvages the run's files into an
|
|
80
|
+
* `artifacts/` subdirectory of the same run directory. A finished run therefore
|
|
81
|
+
* keeps its state one level deeper, and a reader that only looks at the top
|
|
82
|
+
* level reports a shipped task as having no state at all.
|
|
83
|
+
*/
|
|
84
|
+
export const ARTIFACTS_SUBDIR = "artifacts";
|
|
85
|
+
|
|
86
|
+
function isRunDir(dir) {
|
|
87
|
+
if (RUN_MARKERS.some((m) => existsSync(join(dir, m)))) return true;
|
|
88
|
+
return RUN_MARKERS.some((m) => existsSync(join(dir, ARTIFACTS_SUBDIR, m)));
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Newest marker mtime in WHOLE SECONDS.
|
|
93
|
+
*
|
|
94
|
+
* Seconds, not milliseconds, because the shell twin can only read seconds
|
|
95
|
+
* (`stat -c %Y` / `stat -f %m`). Comparing at different precisions made the two
|
|
96
|
+
* implementations pick different winners for a run written to both layouts
|
|
97
|
+
* within the same second - which is the common case, since both writes happen
|
|
98
|
+
* in one Phase 0.
|
|
99
|
+
*/
|
|
100
|
+
function newestMarkerMtime(dir) {
|
|
101
|
+
let newest = 0;
|
|
102
|
+
for (const base of [dir, join(dir, ARTIFACTS_SUBDIR)]) {
|
|
103
|
+
for (const m of RUN_MARKERS) {
|
|
104
|
+
try {
|
|
105
|
+
const t = Math.floor(statSync(join(base, m)).mtimeMs / 1000);
|
|
106
|
+
if (t > newest) newest = t;
|
|
107
|
+
} catch {
|
|
108
|
+
// Marker absent or unreadable - not every run has all three.
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return newest;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Total order over candidate directories for one task id. Negative when `a`
|
|
117
|
+
* should win.
|
|
118
|
+
*
|
|
119
|
+
* mtime alone is not an order: two directories written in the same second tie,
|
|
120
|
+
* and both implementations then fell back to their own traversal order and
|
|
121
|
+
* disagreed. The tail of this comparator exists so the answer is defined:
|
|
122
|
+
* richer record first (agent-state.json is what every reader actually wants),
|
|
123
|
+
* then the documented nested layout, then the path itself.
|
|
124
|
+
*/
|
|
125
|
+
function compareCandidates(a, b) {
|
|
126
|
+
// A bridge and its target hold the same bytes, so neither is "newer". Prefer
|
|
127
|
+
// the real directory: a reader that reports a path should report the one the
|
|
128
|
+
// file actually lives in.
|
|
129
|
+
const aBridge = isBridgeDir(a) ? 1 : 0;
|
|
130
|
+
const bBridge = isBridgeDir(b) ? 1 : 0;
|
|
131
|
+
if (aBridge !== bBridge) return aBridge - bBridge;
|
|
132
|
+
const byTime = newestMarkerMtime(b) - newestMarkerMtime(a);
|
|
133
|
+
if (byTime !== 0) return byTime;
|
|
134
|
+
const aState = existsSync(join(a, "agent-state.json")) ? 1 : 0;
|
|
135
|
+
const bState = existsSync(join(b, "agent-state.json")) ? 1 : 0;
|
|
136
|
+
if (aState !== bState) return bState - aState;
|
|
137
|
+
const aNested = a.split("/").length;
|
|
138
|
+
const bNested = b.split("/").length;
|
|
139
|
+
if (aNested !== bNested) return bNested - aNested;
|
|
140
|
+
return a < b ? -1 : a > b ? 1 : 0;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Every directory a run with this id could live in, most-specific first.
|
|
145
|
+
* Existence is NOT checked - this is the candidate list; resolveRunDir picks.
|
|
146
|
+
*
|
|
147
|
+
* @param {string} taskId
|
|
148
|
+
* @param {string|null|undefined} project
|
|
149
|
+
* @returns {string[]}
|
|
150
|
+
*/
|
|
151
|
+
export function runDirCandidates(taskId, project) {
|
|
152
|
+
const root = logsRoot();
|
|
153
|
+
const out = [];
|
|
154
|
+
if (project) out.push(join(root, project, taskId));
|
|
155
|
+
out.push(join(root, taskId));
|
|
156
|
+
if (!project) {
|
|
157
|
+
// No project given: the run may still be nested under one. Scan depth-1
|
|
158
|
+
// directories for a child with this id rather than failing the lookup.
|
|
159
|
+
let entries;
|
|
160
|
+
try {
|
|
161
|
+
entries = readdirSync(root, { withFileTypes: true });
|
|
162
|
+
} catch {
|
|
163
|
+
return out;
|
|
164
|
+
}
|
|
165
|
+
for (const e of entries) {
|
|
166
|
+
if (!e.isDirectory() || RESERVED_DIRS.has(e.name) || e.name === taskId) continue;
|
|
167
|
+
const cand = join(root, e.name, taskId);
|
|
168
|
+
if (isRunDir(cand)) out.push(cand);
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
return out;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* The directory holding this run's files, or null when the run is unknown.
|
|
176
|
+
* When a taskId exists in more than one layout the most recently written one
|
|
177
|
+
* wins, so a resumed run is not read from its stale twin.
|
|
178
|
+
*
|
|
179
|
+
* @param {string} taskId
|
|
180
|
+
* @param {string|null|undefined} [project]
|
|
181
|
+
* @returns {string|null}
|
|
182
|
+
*/
|
|
183
|
+
export function resolveRunDir(taskId, project) {
|
|
184
|
+
const present = runDirCandidates(taskId, project).filter(isRunDir);
|
|
185
|
+
if (!present.length) return null;
|
|
186
|
+
if (present.length === 1) return present[0];
|
|
187
|
+
return present.slice().sort(compareCandidates)[0];
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* The spellings one task id has been written under, in preference order.
|
|
192
|
+
*
|
|
193
|
+
* `#316` reaches the pipeline from a GitHub issue reference, `316` from the
|
|
194
|
+
* bare-number input class, and `task-316` from an older directory convention.
|
|
195
|
+
* Callers used to inline this list; centralising it is the point of this
|
|
196
|
+
* module, and dropping any spelling would silently stop resolving old runs.
|
|
197
|
+
*
|
|
198
|
+
* @param {string} id
|
|
199
|
+
* @returns {string[]} unique, order-preserving
|
|
200
|
+
*/
|
|
201
|
+
export function taskIdVariants(id) {
|
|
202
|
+
const raw = String(id);
|
|
203
|
+
const bare = raw.replace(/^#/, "");
|
|
204
|
+
return [...new Set([raw, bare, `task-${bare}`])];
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* resolveRunDir over every spelling of the id, first hit wins.
|
|
209
|
+
*
|
|
210
|
+
* @param {string} taskId
|
|
211
|
+
* @param {string|null|undefined} [project]
|
|
212
|
+
* @returns {string|null}
|
|
213
|
+
*/
|
|
214
|
+
export function resolveRunDirAny(taskId, project) {
|
|
215
|
+
for (const v of taskIdVariants(taskId)) {
|
|
216
|
+
const dir = resolveRunDir(v, project);
|
|
217
|
+
if (dir) return dir;
|
|
218
|
+
}
|
|
219
|
+
return null;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
/**
|
|
223
|
+
* The full path to one of a run's files, or null when the run is unknown.
|
|
224
|
+
* Every spelling of the id is tried.
|
|
225
|
+
*
|
|
226
|
+
* @param {string} taskId
|
|
227
|
+
* @param {string} filename e.g. "agent-state.json"
|
|
228
|
+
* @param {string|null|undefined} [project]
|
|
229
|
+
* @returns {string|null}
|
|
230
|
+
*/
|
|
231
|
+
export function resolveRunFile(taskId, filename, project) {
|
|
232
|
+
for (const v of taskIdVariants(taskId)) {
|
|
233
|
+
const dir = resolveRunDir(v, project);
|
|
234
|
+
if (!dir) continue;
|
|
235
|
+
// Top level first, then the salvaged copy Phase 6 leaves behind.
|
|
236
|
+
for (const base of [dir, join(dir, ARTIFACTS_SUBDIR)]) {
|
|
237
|
+
const p = join(base, filename);
|
|
238
|
+
if (existsSync(p)) return p;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
return null;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* True when two candidate directories are two views of ONE record rather than
|
|
246
|
+
* two copies of it.
|
|
247
|
+
*
|
|
248
|
+
* Some nested run directories are symlink bridges into the flat copy - a
|
|
249
|
+
* previous attempt to reconcile the two layouts. Counting those as duplicates
|
|
250
|
+
* overstates the drift (21 reported, 18 real on the install this was measured
|
|
251
|
+
* on) and makes "which one is newer" a question about a link's own mtime.
|
|
252
|
+
*
|
|
253
|
+
* @param {string} a
|
|
254
|
+
* @param {string} b
|
|
255
|
+
* @returns {boolean}
|
|
256
|
+
*/
|
|
257
|
+
function isSameRecord(a, b) {
|
|
258
|
+
for (const m of RUN_MARKERS) {
|
|
259
|
+
try {
|
|
260
|
+
if (realpathSync(join(a, m)) === realpathSync(join(b, m))) return true;
|
|
261
|
+
} catch {
|
|
262
|
+
// Marker missing on one side - try the next one.
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
return false;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
/**
|
|
269
|
+
* True when any marker in this directory is itself a symlink.
|
|
270
|
+
*
|
|
271
|
+
* lstat on the LEAF, not a realpath comparison: realpath also resolves parent
|
|
272
|
+
* directories, and on macOS a temp tree under /var is reached through a symlink
|
|
273
|
+
* to /private/var, so realpath(x) !== x is true for every path there. That made
|
|
274
|
+
* both sides of a pair look like bridges, the rule collapsed, and the two
|
|
275
|
+
* implementations picked different winners. The shell twin tests `[ -L ]`, and
|
|
276
|
+
* this has to mean the same thing.
|
|
277
|
+
*/
|
|
278
|
+
function isBridgeDir(dir) {
|
|
279
|
+
for (const m of RUN_MARKERS) {
|
|
280
|
+
try {
|
|
281
|
+
if (lstatSync(join(dir, m)).isSymbolicLink()) return true;
|
|
282
|
+
} catch {
|
|
283
|
+
// Missing marker is not a bridge.
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
return false;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
function readProjectHint(dir) {
|
|
290
|
+
// agent-state.json v2.1.0 mirrors projects[0] onto scalars; older revisions
|
|
291
|
+
// carry `project` as a string or an object with a name. Take whichever is a
|
|
292
|
+
// non-empty string and do not guess beyond that.
|
|
293
|
+
const statePath = existsSync(join(dir, "agent-state.json"))
|
|
294
|
+
? join(dir, "agent-state.json")
|
|
295
|
+
: join(dir, ARTIFACTS_SUBDIR, "agent-state.json");
|
|
296
|
+
try {
|
|
297
|
+
const s = JSON.parse(readFileSync(statePath, "utf-8"));
|
|
298
|
+
const cands = [s.projectSlug, s.project?.name, s.project, s.projects?.[0]?.name];
|
|
299
|
+
for (const c of cands) if (typeof c === "string" && c.trim()) return c.trim();
|
|
300
|
+
} catch {
|
|
301
|
+
// No state file, or malformed - the layout still tells us what it can.
|
|
302
|
+
}
|
|
303
|
+
return null;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
/**
|
|
307
|
+
* Every run under the log root, both layouts, deduplicated by task id.
|
|
308
|
+
*
|
|
309
|
+
* A task id present in both layouts collapses to one entry - chosen by
|
|
310
|
+
* compareCandidates - with `duplicateOf` naming the other so a caller can
|
|
311
|
+
* report the drift instead of silently hiding it.
|
|
312
|
+
*
|
|
313
|
+
* `project` is derived from the LAYOUT alone, so the shell twin can produce the
|
|
314
|
+
* same four columns without parsing JSON (it has no jq guarantee). Anything
|
|
315
|
+
* read out of the state file is `projectHint`, a JS-only enrichment that
|
|
316
|
+
* runs-index.mjs uses and the cross-implementation contract does not cover.
|
|
317
|
+
*
|
|
318
|
+
* Ordering is byte-wise, not locale-aware, for the same reason: `sort` in the
|
|
319
|
+
* twin is byte-wise, and a collation difference is a diff nobody can act on.
|
|
320
|
+
*
|
|
321
|
+
* @returns {{taskId:string,project:string|null,dir:string,layout:"nested"|"flat",duplicateOf:string|null,projectHint:string|null}[]}
|
|
322
|
+
*/
|
|
323
|
+
export function listRuns() {
|
|
324
|
+
const root = logsRoot();
|
|
325
|
+
let entries;
|
|
326
|
+
try {
|
|
327
|
+
entries = readdirSync(root, { withFileTypes: true });
|
|
328
|
+
} catch {
|
|
329
|
+
return [];
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
/** @type {Map<string, {taskId:string,project:string|null,dir:string,layout:"nested"|"flat",duplicateOf:string|null}>} */
|
|
333
|
+
const byId = new Map();
|
|
334
|
+
|
|
335
|
+
const offer = (taskId, project, dir, layout) => {
|
|
336
|
+
const prev = byId.get(taskId);
|
|
337
|
+
if (!prev) {
|
|
338
|
+
byId.set(taskId, { taskId, project, dir, layout, duplicateOf: null });
|
|
339
|
+
return;
|
|
340
|
+
}
|
|
341
|
+
const keepNew = compareCandidates(dir, prev.dir) < 0;
|
|
342
|
+
const winner = keepNew ? { taskId, project, dir, layout } : prev;
|
|
343
|
+
const loserDir = keepNew ? prev.dir : dir;
|
|
344
|
+
// A symlink bridge is one record seen twice, not drift worth reporting.
|
|
345
|
+
const dup = isSameRecord(dir, prev.dir) ? (prev.duplicateOf ?? null) : loserDir;
|
|
346
|
+
byId.set(taskId, { ...winner, duplicateOf: dup });
|
|
347
|
+
};
|
|
348
|
+
|
|
349
|
+
for (const e of entries) {
|
|
350
|
+
if (!e.isDirectory() || RESERVED_DIRS.has(e.name)) continue;
|
|
351
|
+
const dir = join(root, e.name);
|
|
352
|
+
if (isRunDir(dir)) {
|
|
353
|
+
offer(e.name, null, dir, "flat");
|
|
354
|
+
continue;
|
|
355
|
+
}
|
|
356
|
+
let kids;
|
|
357
|
+
try {
|
|
358
|
+
kids = readdirSync(dir, { withFileTypes: true });
|
|
359
|
+
} catch {
|
|
360
|
+
continue;
|
|
361
|
+
}
|
|
362
|
+
for (const k of kids) {
|
|
363
|
+
if (!k.isDirectory()) continue;
|
|
364
|
+
const kd = join(dir, k.name);
|
|
365
|
+
if (isRunDir(kd)) offer(k.name, e.name, kd, "nested");
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
return [...byId.values()]
|
|
370
|
+
.map((r) => ({ ...r, projectHint: r.project ?? readProjectHint(r.dir) }))
|
|
371
|
+
.sort((a, b) => (a.taskId < b.taskId ? -1 : a.taskId > b.taskId ? 1 : 0));
|
|
372
|
+
}
|
|
@@ -221,12 +221,13 @@ const summary = {
|
|
|
221
221
|
events_by_type: byEvent,
|
|
222
222
|
};
|
|
223
223
|
|
|
224
|
+
// An if/else chain rather than three blocks each ending in process.exit(0).
|
|
225
|
+
// stdout to a pipe is asynchronous and process.exit() discards what has not
|
|
226
|
+
// drained, so `aggregate-metrics.mjs --json | jq` read a payload cut at a
|
|
227
|
+
// buffer boundary. Choosing the renderer with `else` needs no exit at all.
|
|
224
228
|
if (opts.json) {
|
|
225
229
|
console.log(JSON.stringify(summary, null, 2));
|
|
226
|
-
|
|
227
|
-
}
|
|
228
|
-
|
|
229
|
-
if (opts.markdown) {
|
|
230
|
+
} else if (opts.markdown) {
|
|
230
231
|
const lines = [];
|
|
231
232
|
lines.push(`# Pipeline Metrics Summary`);
|
|
232
233
|
lines.push(``);
|
|
@@ -288,74 +289,73 @@ if (opts.markdown) {
|
|
|
288
289
|
);
|
|
289
290
|
}
|
|
290
291
|
console.log(lines.join("\n"));
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
console.log(`
|
|
297
|
-
console.log(`
|
|
298
|
-
console.log(
|
|
299
|
-
console.log(`
|
|
300
|
-
console.log(
|
|
301
|
-
console.log(`
|
|
302
|
-
console.log(`
|
|
303
|
-
console.log(` cycles per task avg: ${summary.review_cycles.avg}`);
|
|
304
|
-
console.log(` cycles per task p95: ${summary.review_cycles.p95}`);
|
|
305
|
-
console.log(``);
|
|
306
|
-
console.log(`Triage classification (sum across all reviews):`);
|
|
307
|
-
console.log(` raw findings: ${summary.triage_totals.raw}`);
|
|
308
|
-
console.log(
|
|
309
|
-
` accepted: ${summary.triage_totals.accepted} (rate ${fmt(summary.triage_totals.accept_rate)})`,
|
|
310
|
-
);
|
|
311
|
-
console.log(` deferred: ${summary.triage_totals.deferred}`);
|
|
312
|
-
console.log(
|
|
313
|
-
` rejected: ${summary.triage_totals.rejected} (rate ${fmt(summary.triage_totals.reject_rate)})`,
|
|
314
|
-
);
|
|
315
|
-
|
|
316
|
-
if (Object.keys(summary.edge_cases).length) {
|
|
292
|
+
} else {
|
|
293
|
+
// Plain-text rendering
|
|
294
|
+
const fmt = (n) => (n === null ? " - " : String(n));
|
|
295
|
+
console.log(`Multi-Agent Pipeline - metrics summary`);
|
|
296
|
+
console.log(`source: ${summary.source}`);
|
|
297
|
+
console.log(`events: ${summary.total_events} (${summary.parse_errors} parse errors)`);
|
|
298
|
+
console.log(`unique tasks: ${summary.unique_tasks}`);
|
|
299
|
+
console.log(``);
|
|
300
|
+
console.log(`Reviews:`);
|
|
301
|
+
console.log(` completed: ${summary.reviews_completed}`);
|
|
302
|
+
console.log(` cycles per task avg: ${summary.review_cycles.avg}`);
|
|
303
|
+
console.log(` cycles per task p95: ${summary.review_cycles.p95}`);
|
|
317
304
|
console.log(``);
|
|
318
|
-
console.log(`Triage
|
|
319
|
-
|
|
320
|
-
|
|
305
|
+
console.log(`Triage classification (sum across all reviews):`);
|
|
306
|
+
console.log(` raw findings: ${summary.triage_totals.raw}`);
|
|
307
|
+
console.log(
|
|
308
|
+
` accepted: ${summary.triage_totals.accepted} (rate ${fmt(summary.triage_totals.accept_rate)})`,
|
|
309
|
+
);
|
|
310
|
+
console.log(` deferred: ${summary.triage_totals.deferred}`);
|
|
311
|
+
console.log(
|
|
312
|
+
` rejected: ${summary.triage_totals.rejected} (rate ${fmt(summary.triage_totals.reject_rate)})`,
|
|
313
|
+
);
|
|
314
|
+
|
|
315
|
+
if (Object.keys(summary.edge_cases).length) {
|
|
316
|
+
console.log(``);
|
|
317
|
+
console.log(`Triage edge cases:`);
|
|
318
|
+
for (const [k, v] of Object.entries(summary.edge_cases).sort((a, b) => b[1] - a[1])) {
|
|
319
|
+
console.log(` ${k.padEnd(28)} ${v}`);
|
|
320
|
+
}
|
|
321
321
|
}
|
|
322
|
-
}
|
|
323
322
|
|
|
324
|
-
if (Object.keys(summary.rework_iterations).length) {
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
323
|
+
if (Object.keys(summary.rework_iterations).length) {
|
|
324
|
+
console.log(``);
|
|
325
|
+
console.log(`Phase 3 rework iterations:`);
|
|
326
|
+
for (const [it, n] of Object.entries(summary.rework_iterations).sort(
|
|
327
|
+
(a, b) => Number(a[0]) - Number(b[0]),
|
|
328
|
+
)) {
|
|
329
|
+
console.log(` iteration ${it.padEnd(5)} ${n}`);
|
|
330
|
+
}
|
|
331
331
|
}
|
|
332
|
-
}
|
|
333
332
|
|
|
334
|
-
if (Object.keys(summary.language_preference).length) {
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
333
|
+
if (Object.keys(summary.language_preference).length) {
|
|
334
|
+
console.log(``);
|
|
335
|
+
console.log(`Language preference (events tagged with lang=):`);
|
|
336
|
+
for (const [lang, n] of Object.entries(summary.language_preference)) {
|
|
337
|
+
console.log(` ${lang.padEnd(8)} ${n}`);
|
|
338
|
+
}
|
|
339
339
|
}
|
|
340
|
-
}
|
|
341
340
|
|
|
342
|
-
if (Object.keys(summary.cost_per_model).length) {
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
341
|
+
if (Object.keys(summary.cost_per_model).length) {
|
|
342
|
+
console.log(``);
|
|
343
|
+
console.log(`Cost / token telemetry (per model):`);
|
|
344
|
+
console.log(
|
|
345
|
+
` ${"model".padEnd(20)} ${"calls".padStart(6)} ${"duration_ms".padStart(13)} ${"tokens_in".padStart(11)} ${"tokens_out".padStart(11)} ${"cached".padStart(11)} ${"cache%".padStart(7)}`,
|
|
346
|
+
);
|
|
347
|
+
for (const [m, t] of Object.entries(summary.cost_per_model).sort(
|
|
348
|
+
(a, b) => b[1].calls - a[1].calls,
|
|
349
|
+
)) {
|
|
350
|
+
const pct = t.cache_ratio === null ? " - " : `${Math.round(t.cache_ratio * 100)}%`;
|
|
351
|
+
console.log(
|
|
352
|
+
` ${m.padEnd(20)} ${String(t.calls).padStart(6)} ${String(t.duration_ms).padStart(13)} ${String(t.tokens_in).padStart(11)} ${String(t.tokens_out).padStart(11)} ${String(t.tokens_cached).padStart(11)} ${pct.padStart(7)}`,
|
|
353
|
+
);
|
|
354
|
+
}
|
|
355
|
+
const oc = summary.cache_reuse;
|
|
356
|
+
const ocPct = oc.cache_ratio === null ? " - " : `${Math.round(oc.cache_ratio * 100)}%`;
|
|
352
357
|
console.log(
|
|
353
|
-
`
|
|
358
|
+
` overall prompt-cache reuse: ${oc.tokens_cached} / ${oc.tokens_in + oc.tokens_cached} input tokens = ${ocPct}`,
|
|
354
359
|
);
|
|
355
360
|
}
|
|
356
|
-
const oc = summary.cache_reuse;
|
|
357
|
-
const ocPct = oc.cache_ratio === null ? " - " : `${Math.round(oc.cache_ratio * 100)}%`;
|
|
358
|
-
console.log(
|
|
359
|
-
` overall prompt-cache reuse: ${oc.tokens_cached} / ${oc.tokens_in + oc.tokens_cached} input tokens = ${ocPct}`,
|
|
360
|
-
);
|
|
361
361
|
}
|
|
@@ -32,6 +32,7 @@
|
|
|
32
32
|
import { existsSync, readFileSync } from "node:fs";
|
|
33
33
|
import { join } from "node:path";
|
|
34
34
|
import { homedir } from "node:os";
|
|
35
|
+
import { invokedDirectly } from "../lib/invoked-directly.mjs";
|
|
35
36
|
|
|
36
37
|
const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopilot");
|
|
37
38
|
const PREFS =
|
|
@@ -142,6 +143,6 @@ function main(argv) {
|
|
|
142
143
|
return 0;
|
|
143
144
|
}
|
|
144
145
|
|
|
145
|
-
if (import.meta.url
|
|
146
|
+
if (invokedDirectly(import.meta.url)) {
|
|
146
147
|
process.exit(main(process.argv));
|
|
147
148
|
}
|