@mmerterden/multi-agent-pipeline 17.6.0 → 19.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +310 -0
- package/README.md +76 -18
- package/README.tr.md +55 -16
- package/docs/adr/0002-instruction-driven-flag.md +1 -0
- package/docs/adr/0005-lazy-phase-docs.md +11 -1
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0010-own-code-graph.md +1 -0
- package/docs/adr/0011-dormant-ci.md +25 -1
- package/docs/adr/0014-six-phase-consolidation.md +134 -0
- package/docs/adr/README.md +2 -1
- package/docs/architecture.md +37 -38
- package/docs/best-practices.md +1 -1
- package/docs/ecosystem.md +37 -26
- package/docs/engineering.md +1 -1
- package/docs/facts.json +45 -0
- package/docs/features.md +54 -53
- package/docs/performance.md +5 -5
- package/docs/recovery-guide.md +9 -9
- package/docs/server-readiness.md +188 -0
- package/docs/token-budget-history.md +3 -1
- package/index.js +18 -3
- package/install/_codex-agents.mjs +1 -1
- package/install/_common.mjs +42 -17
- package/install/_dev-only-files.mjs +8 -0
- package/install/_unattended-profile.mjs +113 -0
- package/install/index.mjs +48 -0
- package/install/templates/claude-hooks.json +1 -1
- package/install/templates/codex-instructions.md +1 -1
- package/install/templates/copilot-instructions.md +28 -28
- package/manifest.json +1065 -0
- package/package.json +6 -3
- package/pipeline/agents/dev-critic.md +3 -3
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +8 -8
- package/pipeline/commands/multi-agent/analysis/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
- package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
- package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
- package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
- package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
- package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
- package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/status/SKILL.md +54 -23
- package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
- package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
- package/pipeline/lib/_jira-auth.sh +8 -0
- package/pipeline/lib/analysis-jira-write.sh +32 -0
- package/pipeline/lib/ask-choice.sh +13 -2
- package/pipeline/lib/autopilot-state.sh +8 -0
- package/pipeline/lib/credential-inventory.sh +1 -1
- package/pipeline/lib/fatal.mjs +129 -0
- package/pipeline/lib/fetch-fortify.sh +1 -1
- package/pipeline/lib/figma-mcp-refresh.sh +18 -0
- package/pipeline/lib/figma-screenshot.sh +18 -0
- package/pipeline/lib/invoked-directly.mjs +43 -0
- package/pipeline/lib/jira-publish.sh +42 -0
- package/pipeline/lib/md2confluence-v3.py +47 -0
- package/pipeline/lib/model-rung.sh +142 -0
- package/pipeline/lib/outbound-gate.mjs +175 -0
- package/pipeline/lib/phase-schema.mjs +88 -0
- package/pipeline/lib/plan-todos.sh +32 -11
- package/pipeline/lib/post-pr-review.sh +77 -8
- package/pipeline/lib/repo-hygiene.sh +8 -3
- package/pipeline/lib/require-jq.sh +40 -0
- package/pipeline/lib/route-state.sh +161 -0
- package/pipeline/lib/run-paths.sh +335 -0
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +1 -1
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/analysis/evidence.md +0 -9
- package/pipeline/multi-agent-refs/analysis/intake.md +1 -1
- package/pipeline/multi-agent-refs/analysis/locked.md +21 -22
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/analysis/synthesis.md +12 -6
- package/pipeline/multi-agent-refs/android-guide.md +1 -1
- package/pipeline/multi-agent-refs/audit-guide.md +13 -13
- package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
- package/pipeline/multi-agent-refs/channels/jira.md +3 -3
- package/pipeline/multi-agent-refs/channels/pr.md +4 -4
- package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +3 -3
- package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +74 -4
- package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
- package/pipeline/multi-agent-refs/features/cost-analysis.md +93 -0
- package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
- package/pipeline/multi-agent-refs/features/doctor.md +47 -2
- package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
- package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
- package/pipeline/multi-agent-refs/features/model-fallback.md +5 -5
- package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
- package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
- package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
- package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
- package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
- package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
- package/pipeline/multi-agent-refs/features/verify.md +83 -0
- package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
- package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
- package/pipeline/multi-agent-refs/knowledge.md +11 -11
- package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
- package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
- package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
- package/pipeline/multi-agent-refs/phases/modes.md +30 -30
- package/pipeline/multi-agent-refs/phases/operations.md +21 -10
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +25 -25
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
- package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
- package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
- package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
- package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
- package/pipeline/multi-agent-refs/phases.md +44 -48
- package/pipeline/multi-agent-refs/picker-contract.md +1 -1
- package/pipeline/multi-agent-refs/progress-contract.md +6 -6
- package/pipeline/multi-agent-refs/readiness-review.md +1 -1
- package/pipeline/multi-agent-refs/rules.md +7 -7
- package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
- package/pipeline/multi-agent-refs/unattended-contract.md +129 -0
- package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
- package/pipeline/preferences-template.json +9 -1
- package/pipeline/rules/outside-the-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +50 -50
- package/pipeline/schemas/analysis-output.schema.json +2 -2
- package/pipeline/schemas/autopilot-config.schema.json +1 -1
- package/pipeline/schemas/code-graph.schema.json +1 -1
- package/pipeline/schemas/criteria-manifest.schema.json +1 -1
- package/pipeline/schemas/dev-critic-output.schema.json +1 -1
- package/pipeline/schemas/diff-risk.schema.json +1 -1
- package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
- package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
- package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
- package/pipeline/schemas/phases.json +105 -0
- package/pipeline/schemas/plan-todos.schema.json +5 -5
- package/pipeline/schemas/planning-output.schema.json +1 -1
- package/pipeline/schemas/prefs.schema.json +100 -56
- package/pipeline/schemas/reviewer-output.schema.json +3 -3
- package/pipeline/schemas/route-config.schema.json +74 -0
- package/pipeline/schemas/scope-check.schema.json +1 -1
- package/pipeline/schemas/test-gap.schema.json +1 -1
- package/pipeline/schemas/token-budget.json +12 -18
- package/pipeline/schemas/triage-output.schema.json +6 -6
- package/pipeline/scripts/README.md +3 -3
- package/pipeline/scripts/_code-graph.mjs +2 -2
- package/pipeline/scripts/_run-paths.mjs +372 -0
- package/pipeline/scripts/_smoke-root.sh +1 -1
- package/pipeline/scripts/aggregate-metrics.mjs +65 -65
- package/pipeline/scripts/autopilot-arming.mjs +2 -1
- package/pipeline/scripts/autopilot-intake.mjs +2 -1
- package/pipeline/scripts/autopilot-runner.mjs +206 -2
- package/pipeline/scripts/build-references.mjs +2 -1
- package/pipeline/scripts/build-stack-plugins.mjs +10 -2
- package/pipeline/scripts/capture-evidence.sh +7 -2
- package/pipeline/scripts/capture-flush.sh +8 -8
- package/pipeline/scripts/capture-resume.sh +3 -3
- package/pipeline/scripts/classify-plan-safety.mjs +3 -2
- package/pipeline/scripts/cost-analyze.mjs +600 -0
- package/pipeline/scripts/cost-budget-check.mjs +4 -12
- package/pipeline/scripts/council-view.mjs +2 -1
- package/pipeline/scripts/crush-json.mjs +2 -1
- package/pipeline/scripts/diff-explain.mjs +7 -10
- package/pipeline/scripts/diff-risk-score.mjs +2 -1
- package/pipeline/scripts/doctor.mjs +140 -6
- package/pipeline/scripts/evidence-gate.mjs +9 -3
- package/pipeline/scripts/feedback-send.mjs +12 -2
- package/pipeline/scripts/gc-abandoned.sh +32 -16
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +12 -5
- package/pipeline/scripts/gen-facts.mjs +175 -0
- package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
- package/pipeline/scripts/gen-ref-toc.mjs +1 -1
- package/pipeline/scripts/github-ssh-setup.sh +64 -7
- package/pipeline/scripts/graph-mermaid.mjs +4 -2
- package/pipeline/scripts/graph-report.mjs +1 -1
- package/pipeline/scripts/jira-attach.sh +1 -1
- package/pipeline/scripts/keychain-save.sh +101 -30
- package/pipeline/scripts/learn-from-transcripts.mjs +3 -2
- package/pipeline/scripts/learning-curve.mjs +36 -31
- package/pipeline/scripts/log-metric.sh +17 -4
- package/pipeline/scripts/make-manifest.mjs +199 -0
- package/pipeline/scripts/memory-save.sh +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +24 -6
- package/pipeline/scripts/migrate-state.mjs +94 -4
- package/pipeline/scripts/phase-banner.sh +26 -22
- package/pipeline/scripts/phase-tracker.sh +48 -10
- package/pipeline/scripts/plan-coverage-gate.mjs +8 -4
- package/pipeline/scripts/pre-commit-check.sh +7 -0
- package/pipeline/scripts/pre-push-check.sh +7 -0
- package/pipeline/scripts/purge.sh +23 -6
- package/pipeline/scripts/render-agent-log-cost.sh +10 -3
- package/pipeline/scripts/render-cost-summary.sh +9 -2
- package/pipeline/scripts/render-work-summary.sh +14 -7
- package/pipeline/scripts/review-file-filter.mjs +5 -3
- package/pipeline/scripts/review-scope.mjs +2 -1
- package/pipeline/scripts/routine-registry.mjs +2 -1
- package/pipeline/scripts/run-aggregator.mjs +26 -20
- package/pipeline/scripts/run-metrics.mjs +4 -2
- package/pipeline/scripts/runs-index.mjs +353 -0
- package/pipeline/scripts/scorecard-snapshot.mjs +178 -0
- package/pipeline/scripts/search-logs.sh +18 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
- package/pipeline/scripts/smoke-schema-validation.sh +26 -7
- package/pipeline/scripts/test-gap-scan.mjs +2 -1
- package/pipeline/scripts/test-integrity-gate.mjs +2 -1
- package/pipeline/scripts/token-budget-report.mjs +13 -2
- package/pipeline/scripts/triage-memory.mjs +2 -2
- package/pipeline/scripts/update-issue-progress.sh +56 -7
- package/pipeline/scripts/usage-report.mjs +12 -1
- package/pipeline/scripts/validate-analysis-doc.mjs +75 -18
- package/pipeline/scripts/validate-code-graph.mjs +6 -3
- package/pipeline/scripts/validate-complaint-doc.mjs +2 -1
- package/pipeline/scripts/validate-diff-risk.mjs +6 -3
- package/pipeline/scripts/validate-planning.mjs +1 -1
- package/pipeline/scripts/validate-reviewer.mjs +1 -1
- package/pipeline/scripts/validate-state.mjs +45 -5
- package/pipeline/scripts/validate-test-gap.mjs +6 -3
- package/pipeline/scripts/validate-triage.mjs +6 -4
- package/pipeline/scripts/verify-citations.mjs +4 -2
- package/pipeline/scripts/verify.mjs +327 -0
- package/pipeline/scripts/worktree-finalize.sh +18 -9
- package/pipeline/scripts/write-state.mjs +154 -15
- package/pipeline/skills/.skill-manifest.json +37 -21
- package/pipeline/skills/.skills-index.json +104 -5
- package/pipeline/skills/shared/README.md +15 -6
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +2 -2
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +69 -71
- package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
- package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
- package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
- package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
- package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
- package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +35 -11
- package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +2 -1
- package/pipeline/skills/skills-index.md +13 -4
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
|
@@ -0,0 +1,353 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* @file runs-index.mjs - one deterministic answer to "what runs exist and
|
|
4
|
+
* where is each one".
|
|
5
|
+
*
|
|
6
|
+
* `/multi-agent:status` used to answer this by telling the model to go and find
|
|
7
|
+
* the files itself: scan three hard-coded `.worktrees/` paths, then `find` the
|
|
8
|
+
* log tree at one depth, then merge. Three problems with that. It cannot be
|
|
9
|
+
* called by anything that is not a model, the depth was wrong for half the
|
|
10
|
+
* layouts, and two invocations could disagree because nothing pinned the
|
|
11
|
+
* traversal. A UI, a gate, or a second phase reading the same question got a
|
|
12
|
+
* different answer than the terminal did.
|
|
13
|
+
*
|
|
14
|
+
* This is the producer. `--json` and the human table are rendered from the SAME
|
|
15
|
+
* in-memory records, so a dashboard and a terminal cannot disagree - the rule
|
|
16
|
+
* autopilot-status.sh already follows for the autopilot half of the picture.
|
|
17
|
+
*
|
|
18
|
+
* Grouping follows the contract in commands/multi-agent/status/SKILL.md 3b,
|
|
19
|
+
* including its final clause: a run with no `status` is placed in no group at
|
|
20
|
+
* all. Unknown is not a finding, and calling it dead is the same false claim in
|
|
21
|
+
* the other direction.
|
|
22
|
+
*
|
|
23
|
+
* Read-only. Never writes, never migrates, never deletes.
|
|
24
|
+
*
|
|
25
|
+
* Usage:
|
|
26
|
+
* node runs-index.mjs # human table, grouped
|
|
27
|
+
* node runs-index.mjs --json # the same records as JSON
|
|
28
|
+
* node runs-index.mjs --group waiting # one group only
|
|
29
|
+
* node runs-index.mjs --task-id <id> # one run
|
|
30
|
+
*
|
|
31
|
+
* Exit codes:
|
|
32
|
+
* 0 - answered (including "no runs")
|
|
33
|
+
* 2 - usage error
|
|
34
|
+
*
|
|
35
|
+
* @module pipeline/scripts/runs-index
|
|
36
|
+
*/
|
|
37
|
+
|
|
38
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
39
|
+
import { join } from "node:path";
|
|
40
|
+
import { costUsd } from "./_cost.mjs";
|
|
41
|
+
import { listRuns, logsRoot, resolveRunDir, taskIdVariants } from "./_run-paths.mjs";
|
|
42
|
+
import { runMain } from "../lib/fatal.mjs";
|
|
43
|
+
import { invokedDirectly } from "../lib/invoked-directly.mjs";
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* The phase from which a run counts as "waiting on you", read from the phase
|
|
47
|
+
* contract rather than written here. This threshold moved once already (it was
|
|
48
|
+
* 6 under the eight-phase contract, it is 4 under six) and nothing connected it
|
|
49
|
+
* to the renumbering, so it would have silently regrouped every run.
|
|
50
|
+
*/
|
|
51
|
+
const PHASE_WAITING_FROM = JSON.parse(
|
|
52
|
+
readFileSync(new URL("../schemas/phases.json", import.meta.url), "utf8"),
|
|
53
|
+
).thresholds.waitingFromPhase;
|
|
54
|
+
|
|
55
|
+
const GROUPS = {
|
|
56
|
+
waiting: "Waiting on you",
|
|
57
|
+
stopped: "Stopped mid-development",
|
|
58
|
+
question: "Left at a question",
|
|
59
|
+
unknown: "Status not recorded",
|
|
60
|
+
};
|
|
61
|
+
|
|
62
|
+
function parseArgs(argv) {
|
|
63
|
+
const flags = { json: false, group: null, taskId: null };
|
|
64
|
+
for (let i = 0; i < argv.length; i++) {
|
|
65
|
+
const a = argv[i];
|
|
66
|
+
if (a === "--json") flags.json = true;
|
|
67
|
+
else if (a === "--group") flags.group = argv[++i];
|
|
68
|
+
else if (a.startsWith("--group=")) flags.group = a.slice(8);
|
|
69
|
+
else if (a === "--task-id") flags.taskId = argv[++i];
|
|
70
|
+
else if (a.startsWith("--task-id=")) flags.taskId = a.slice(10);
|
|
71
|
+
else if (a === "-h" || a === "--help") flags.help = true;
|
|
72
|
+
else {
|
|
73
|
+
process.stderr.write(`runs-index: unknown argument ${a}\n`);
|
|
74
|
+
process.exit(2);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
if (flags.group && !Object.hasOwn(GROUPS, flags.group)) {
|
|
78
|
+
process.stderr.write(
|
|
79
|
+
`runs-index: unknown group ${flags.group} (want ${Object.keys(GROUPS).join(", ")})\n`,
|
|
80
|
+
);
|
|
81
|
+
process.exit(2);
|
|
82
|
+
}
|
|
83
|
+
return flags;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function readJson(path) {
|
|
87
|
+
if (!path || !existsSync(path)) return null;
|
|
88
|
+
try {
|
|
89
|
+
return JSON.parse(readFileSync(path, "utf-8"));
|
|
90
|
+
} catch {
|
|
91
|
+
// A truncated state file is a fact about the run, not a reason to refuse
|
|
92
|
+
// the whole index. It surfaces as `stateReadable: false`.
|
|
93
|
+
return null;
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function firstFile(dir, name) {
|
|
98
|
+
for (const base of [dir, join(dir, "artifacts")]) {
|
|
99
|
+
const p = join(base, name);
|
|
100
|
+
if (existsSync(p)) return p;
|
|
101
|
+
}
|
|
102
|
+
return null;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
let COST_TABLE = null;
|
|
106
|
+
function rateFor(model) {
|
|
107
|
+
if (!COST_TABLE) {
|
|
108
|
+
COST_TABLE = readJson(new URL("./cost-table.json", import.meta.url).pathname) ?? { prices: {} };
|
|
109
|
+
}
|
|
110
|
+
if (!model) return null;
|
|
111
|
+
const prices = COST_TABLE.prices ?? {};
|
|
112
|
+
// phase-tracker.sh stores the short tier name ("opus", "fable"), which is the
|
|
113
|
+
// table's own key - the same lookup run-aggregator.mjs does. A caller that
|
|
114
|
+
// stored the full model id instead is matched on the table's `modelId`
|
|
115
|
+
// rather than guessed at by prefix: "gpt-5.6" is a prefix of "gpt-5.6-terra"
|
|
116
|
+
// and those are two different prices.
|
|
117
|
+
if (prices[model]) return prices[model];
|
|
118
|
+
for (const rate of Object.values(prices)) {
|
|
119
|
+
if (rate?.modelId === model) return rate;
|
|
120
|
+
}
|
|
121
|
+
return null;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Phase rows plus token/cost totals, from the tracker document.
|
|
126
|
+
*
|
|
127
|
+
* @param {object|null} tracker
|
|
128
|
+
*/
|
|
129
|
+
function summarisePhases(tracker) {
|
|
130
|
+
const phases = Array.isArray(tracker?.phases) ? tracker.phases : [];
|
|
131
|
+
let tokensIn = 0;
|
|
132
|
+
let tokensOut = 0;
|
|
133
|
+
let tokensCached = 0;
|
|
134
|
+
let usd = 0;
|
|
135
|
+
const rows = phases.map((p) => {
|
|
136
|
+
// phase-tracker.sh writes flat `tokens_in` / `tokens_out` / `tokens_cached`
|
|
137
|
+
// on each phase, not a nested `tokens` object.
|
|
138
|
+
const tin = Number(p.tokens_in ?? 0) || 0;
|
|
139
|
+
const tout = Number(p.tokens_out ?? 0) || 0;
|
|
140
|
+
const tcached = Number(p.tokens_cached ?? 0) || 0;
|
|
141
|
+
tokensIn += tin;
|
|
142
|
+
tokensOut += tout;
|
|
143
|
+
tokensCached += tcached;
|
|
144
|
+
const c = costUsd(rateFor(p.model), tin, tout, tcached);
|
|
145
|
+
if (typeof c === "number") usd += c;
|
|
146
|
+
return {
|
|
147
|
+
id: String(p.id ?? ""),
|
|
148
|
+
name: p.name ?? "",
|
|
149
|
+
status: p.status ?? "pending",
|
|
150
|
+
model: p.model ?? null,
|
|
151
|
+
startedAt: p.started_at ?? null,
|
|
152
|
+
completedAt: p.completed_at ?? null,
|
|
153
|
+
now: p.now ?? null,
|
|
154
|
+
subs: Array.isArray(p.subs) ? p.subs.length : 0,
|
|
155
|
+
};
|
|
156
|
+
});
|
|
157
|
+
return {
|
|
158
|
+
phases: rows,
|
|
159
|
+
startedAt: tracker?.started_at ?? null,
|
|
160
|
+
tokens: { in: tokensIn, out: tokensOut, cached: tokensCached },
|
|
161
|
+
estUsd: Number(usd.toFixed(4)),
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* The group a run belongs to, per status/SKILL.md 3b.
|
|
167
|
+
*
|
|
168
|
+
* @param {object|null} state
|
|
169
|
+
* @returns {"waiting"|"stopped"|"question"|"unknown"}
|
|
170
|
+
*/
|
|
171
|
+
function groupOf(state) {
|
|
172
|
+
const status = state?.status;
|
|
173
|
+
if (!status) return "unknown";
|
|
174
|
+
const phase = Number(state?.currentPhase);
|
|
175
|
+
const prUrl = state?.pr?.url ?? state?.prUrl ?? null;
|
|
176
|
+
if (status === "awaiting_input" || status === "awaiting-user-test-main-checkout")
|
|
177
|
+
return "waiting";
|
|
178
|
+
if (prUrl) return "waiting";
|
|
179
|
+
if (Number.isFinite(phase) && phase >= PHASE_WAITING_FROM) return "waiting";
|
|
180
|
+
if (Number.isFinite(phase) && phase === 0) return "question";
|
|
181
|
+
return "stopped";
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/**
|
|
185
|
+
* Every run, enriched, in one stable order.
|
|
186
|
+
*
|
|
187
|
+
* @returns {object[]}
|
|
188
|
+
*/
|
|
189
|
+
export function buildIndex() {
|
|
190
|
+
return listRuns().map((run) => {
|
|
191
|
+
const statePath = firstFile(run.dir, "agent-state.json");
|
|
192
|
+
const trackerPath = firstFile(run.dir, "tracker-state.json");
|
|
193
|
+
const state = readJson(statePath);
|
|
194
|
+
const tracker = readJson(trackerPath);
|
|
195
|
+
const phases = summarisePhases(tracker);
|
|
196
|
+
const prUrl = state?.pr?.url ?? state?.prUrl ?? null;
|
|
197
|
+
return {
|
|
198
|
+
taskId: run.taskId,
|
|
199
|
+
project: run.project ?? run.projectHint ?? null,
|
|
200
|
+
dir: run.dir,
|
|
201
|
+
layout: run.layout,
|
|
202
|
+
duplicateOf: run.duplicateOf,
|
|
203
|
+
salvaged: Boolean(statePath && statePath.includes(`${run.dir}/artifacts/`)),
|
|
204
|
+
// What KIND of record this is, before asking whether it is healthy.
|
|
205
|
+
//
|
|
206
|
+
// 74 of the 103 runs on this machine have a tracker file and no agent
|
|
207
|
+
// state, and the first version of this index called every one of them
|
|
208
|
+
// `stateReadable: false` - so a panel built on it announced "74 runs
|
|
209
|
+
// could not be read" about records that are not damaged and never had
|
|
210
|
+
// agent state to begin with. `/multi-agent:analysis` says so in its own
|
|
211
|
+
// description ("no worktree, no commits, no dev chain"), and design-check
|
|
212
|
+
// is the same shape.
|
|
213
|
+
//
|
|
214
|
+
// Derived from what is on disk, not from the task id: `ANALYSIS-` and
|
|
215
|
+
// `DC-` are naming conventions, and a convention is not a contract.
|
|
216
|
+
kind: statePath ? "pipeline" : trackerPath ? "tracker-only" : "empty",
|
|
217
|
+
// Now the health question, and it only applies where state was expected.
|
|
218
|
+
// A tracker-only record is `true` because there is nothing it failed to
|
|
219
|
+
// read; "could not read" and "there is none" are different answers and
|
|
220
|
+
// only the first one asks anyone to do something.
|
|
221
|
+
stateReadable: statePath ? Boolean(state) : true,
|
|
222
|
+
status: state?.status ?? null,
|
|
223
|
+
currentPhase: Number.isFinite(Number(state?.currentPhase))
|
|
224
|
+
? Number(state.currentPhase)
|
|
225
|
+
: null,
|
|
226
|
+
branch: state?.branch ?? state?.branchName ?? null,
|
|
227
|
+
baseBranch: state?.baseBranch ?? null,
|
|
228
|
+
startedAt: state?.startedAt ?? phases.startedAt,
|
|
229
|
+
worktreePath: state?.worktreePath ?? null,
|
|
230
|
+
prUrl,
|
|
231
|
+
autopilot: Boolean(state?.autopilot),
|
|
232
|
+
schemaVersion: state?.schemaVersion ?? null,
|
|
233
|
+
rev: Number.isInteger(state?.rev) ? state.rev : null,
|
|
234
|
+
group: groupOf(state),
|
|
235
|
+
phases: phases.phases,
|
|
236
|
+
tokens: phases.tokens,
|
|
237
|
+
estUsd: phases.estUsd,
|
|
238
|
+
// How many tracker writes went through without the lock. Non-zero means
|
|
239
|
+
// two writers were in the critical section and one of their token
|
|
240
|
+
// deltas was dropped, so the numbers on this run are LOW. Reporting the
|
|
241
|
+
// count rather than the loss, because the size of what was dropped is
|
|
242
|
+
// exactly what nobody measured.
|
|
243
|
+
unlockedWrites: Number.isInteger(tracker?.unlockedWrites) ? tracker.unlockedWrites : 0,
|
|
244
|
+
};
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
function fmtDuration(startedAt) {
|
|
249
|
+
if (!startedAt) return "-";
|
|
250
|
+
const t = Date.parse(startedAt);
|
|
251
|
+
if (!Number.isFinite(t)) return "-";
|
|
252
|
+
const mins = Math.max(0, Math.round((Date.now() - t) / 60000));
|
|
253
|
+
if (mins < 60) return `${mins}m`;
|
|
254
|
+
const h = Math.floor(mins / 60);
|
|
255
|
+
if (h < 48) return `${h}h`;
|
|
256
|
+
return `${Math.floor(h / 24)}d`;
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
function renderHuman(records) {
|
|
260
|
+
const out = [];
|
|
261
|
+
const total = records.length;
|
|
262
|
+
out.push(`multi-agent runs - ${total} under ${logsRoot()}`);
|
|
263
|
+
|
|
264
|
+
const dupes = records.filter((r) => r.duplicateOf);
|
|
265
|
+
if (dupes.length) {
|
|
266
|
+
out.push(
|
|
267
|
+
` ${dupes.length} run(s) also have a copy in the other directory layout; ` +
|
|
268
|
+
`the newer one is shown. migrate-state.mjs --all reports them.`,
|
|
269
|
+
);
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
for (const key of Object.keys(GROUPS)) {
|
|
273
|
+
const rows = records.filter((r) => r.group === key);
|
|
274
|
+
if (!rows.length) continue;
|
|
275
|
+
out.push("");
|
|
276
|
+
out.push(`${GROUPS[key]} (${rows.length})`);
|
|
277
|
+
out.push(
|
|
278
|
+
" ID Phase Status Branch Age ~USD",
|
|
279
|
+
);
|
|
280
|
+
for (const r of rows) {
|
|
281
|
+
const phase = r.currentPhase === null ? " -" : `${r.currentPhase}/7`;
|
|
282
|
+
out.push(
|
|
283
|
+
" " +
|
|
284
|
+
[
|
|
285
|
+
r.taskId.padEnd(30).slice(0, 30),
|
|
286
|
+
String(phase).padStart(5),
|
|
287
|
+
(r.status ?? "-").padEnd(17).slice(0, 17),
|
|
288
|
+
(r.branch ?? "-").padEnd(24).slice(0, 24),
|
|
289
|
+
fmtDuration(r.startedAt).padStart(5),
|
|
290
|
+
(r.estUsd ? r.estUsd.toFixed(2) : "-").padStart(6),
|
|
291
|
+
].join(" "),
|
|
292
|
+
);
|
|
293
|
+
}
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
const hints = {
|
|
297
|
+
waiting: "resume #N - the work landed, it needs your answer",
|
|
298
|
+
stopped: "resume #N or kill #N",
|
|
299
|
+
question: "garbage-collect --abandoned - nothing was built",
|
|
300
|
+
};
|
|
301
|
+
const present = Object.keys(hints).filter((k) => records.some((r) => r.group === k));
|
|
302
|
+
if (present.length) {
|
|
303
|
+
out.push("");
|
|
304
|
+
for (const k of present) out.push(` ${GROUPS[k]}: ${hints[k]}`);
|
|
305
|
+
}
|
|
306
|
+
return out.join("\n");
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
function main() {
|
|
310
|
+
const flags = parseArgs(process.argv.slice(2));
|
|
311
|
+
if (flags.help) {
|
|
312
|
+
process.stdout.write(
|
|
313
|
+
"Usage: runs-index.mjs [--json] [--group waiting|stopped|question|unknown] [--task-id <id>]\n",
|
|
314
|
+
);
|
|
315
|
+
return 0;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
let records = buildIndex();
|
|
319
|
+
|
|
320
|
+
if (flags.taskId) {
|
|
321
|
+
const wanted = new Set(taskIdVariants(flags.taskId));
|
|
322
|
+
records = records.filter((r) => wanted.has(r.taskId));
|
|
323
|
+
if (!records.length) {
|
|
324
|
+
const dir = resolveRunDir(flags.taskId);
|
|
325
|
+
process.stderr.write(
|
|
326
|
+
`runs-index: no run for ${flags.taskId}${dir ? ` (directory ${dir} has no markers)` : ""}\n`,
|
|
327
|
+
);
|
|
328
|
+
return 0;
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
if (flags.group) records = records.filter((r) => r.group === flags.group);
|
|
332
|
+
|
|
333
|
+
if (flags.json) {
|
|
334
|
+
process.stdout.write(
|
|
335
|
+
JSON.stringify({ logsRoot: logsRoot(), count: records.length, runs: records }, null, 2) +
|
|
336
|
+
"\n",
|
|
337
|
+
);
|
|
338
|
+
} else {
|
|
339
|
+
process.stdout.write(renderHuman(records) + "\n");
|
|
340
|
+
}
|
|
341
|
+
return 0;
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
if (invokedDirectly(import.meta.url)) {
|
|
345
|
+
// process.exitCode, not process.exit(): stdout to a PIPE is asynchronous and
|
|
346
|
+
// process.exit() throws away whatever has not drained. 103 runs render to
|
|
347
|
+
// ~200KB of JSON, so a caller doing `runs-index.mjs --json | jq` received the
|
|
348
|
+
// first 64KB and a parse error. A terminal hid it, because stdout to a TTY is
|
|
349
|
+
// synchronous - and a terminal is where this was tested.
|
|
350
|
+
runMain("runs-index", () => {
|
|
351
|
+
process.exitCode = main();
|
|
352
|
+
});
|
|
353
|
+
}
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scorecard-snapshot.mjs - keep what the scorecard said, and say what moved.
|
|
4
|
+
*
|
|
5
|
+
* The scorecard answers "does every mechanical claim hold right now". It
|
|
6
|
+
* cannot answer "is this better or worse than last week", and that second
|
|
7
|
+
* question is the one that catches slow rot: a metric that has been failing
|
|
8
|
+
* for a month reads identically to one that broke an hour ago.
|
|
9
|
+
*
|
|
10
|
+
* WHAT THIS DELIBERATELY DOES NOT DO: collapse the result into a 0-100 score.
|
|
11
|
+
* ruflo's scorecard does, and the discipline worth taking from it is
|
|
12
|
+
* "measurable and comparable over time", not the number. The scorecard already
|
|
13
|
+
* reports twelve measured metrics AND four it refuses to measure - a single
|
|
14
|
+
* figure would hide both halves, and an unmeasured category would silently
|
|
15
|
+
* count as zero or as full marks depending on an arithmetic choice nobody
|
|
16
|
+
* would ever read. A diff keeps every metric answering for itself.
|
|
17
|
+
*
|
|
18
|
+
* Snapshots live in ~/.claude/state/scorecard/<iso>.json and are never pruned
|
|
19
|
+
* by this script: they are small, and a history that deletes itself cannot
|
|
20
|
+
* answer the question it was kept for.
|
|
21
|
+
*
|
|
22
|
+
* Usage:
|
|
23
|
+
* scorecard-snapshot.mjs --save run the scorecard, store a snapshot
|
|
24
|
+
* scorecard-snapshot.mjs --diff latest vs the one before it
|
|
25
|
+
* scorecard-snapshot.mjs --diff a.json b.json two snapshots by path
|
|
26
|
+
* scorecard-snapshot.mjs --list what is stored
|
|
27
|
+
*
|
|
28
|
+
* Exit codes:
|
|
29
|
+
* 0 - nothing regressed (or a save succeeded)
|
|
30
|
+
* 1 - at least one metric that used to pass now fails
|
|
31
|
+
* 2 - usage error, or not enough snapshots to compare
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
import { execFileSync } from "node:child_process";
|
|
35
|
+
import { existsSync, mkdirSync, readdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
36
|
+
import { homedir } from "node:os";
|
|
37
|
+
import { dirname, join } from "node:path";
|
|
38
|
+
import { fileURLToPath } from "node:url";
|
|
39
|
+
import { runMain } from "../lib/fatal.mjs";
|
|
40
|
+
|
|
41
|
+
const ROOT = join(dirname(fileURLToPath(import.meta.url)), "..", "..");
|
|
42
|
+
const STORE = process.env.SCORECARD_STORE || join(homedir(), ".claude", "state", "scorecard");
|
|
43
|
+
|
|
44
|
+
/** Metric identity. Category plus metric name, because neither is unique alone. */
|
|
45
|
+
const keyOf = (r) => `${r.category} :: ${r.metric}`;
|
|
46
|
+
|
|
47
|
+
function runScorecard() {
|
|
48
|
+
// `--json` prints the report on stdout; a failing metric is a non-zero exit
|
|
49
|
+
// and is exactly the case worth snapshotting, so the status is captured
|
|
50
|
+
// rather than thrown.
|
|
51
|
+
try {
|
|
52
|
+
return JSON.parse(
|
|
53
|
+
execFileSync("node", ["pipeline/scripts/scorecard.mjs", "--json"], {
|
|
54
|
+
cwd: ROOT,
|
|
55
|
+
encoding: "utf-8",
|
|
56
|
+
maxBuffer: 32 * 1024 * 1024,
|
|
57
|
+
}),
|
|
58
|
+
);
|
|
59
|
+
} catch (err) {
|
|
60
|
+
const out = err?.stdout;
|
|
61
|
+
if (typeof out === "string" && out.trim().startsWith("{")) return JSON.parse(out);
|
|
62
|
+
throw new Error(`scorecard did not produce JSON: ${err?.message ?? err}`, { cause: err });
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function snapshots() {
|
|
67
|
+
if (!existsSync(STORE)) return [];
|
|
68
|
+
return (
|
|
69
|
+
readdirSync(STORE)
|
|
70
|
+
.filter((f) => f.endsWith(".json"))
|
|
71
|
+
// ISO-8601 sorts lexically in time order, which is the whole reason the
|
|
72
|
+
// file is named after the timestamp rather than a counter.
|
|
73
|
+
.sort()
|
|
74
|
+
.map((f) => join(STORE, f))
|
|
75
|
+
);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function save() {
|
|
79
|
+
const report = runScorecard();
|
|
80
|
+
mkdirSync(STORE, { recursive: true });
|
|
81
|
+
const at = new Date().toISOString().replace(/[:.]/g, "-");
|
|
82
|
+
const path = join(STORE, `${at}.json`);
|
|
83
|
+
writeFileSync(path, JSON.stringify({ at: new Date().toISOString(), ...report }, null, 2) + "\n");
|
|
84
|
+
process.stdout.write(`scorecard snapshot: ${path}\n`);
|
|
85
|
+
process.stdout.write(` ${report.passed} passed, ${report.failed} failed\n`);
|
|
86
|
+
return 0;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function diff(aPath, bPath) {
|
|
90
|
+
const a = JSON.parse(readFileSync(aPath, "utf-8"));
|
|
91
|
+
const b = JSON.parse(readFileSync(bPath, "utf-8"));
|
|
92
|
+
// UNMEASURED rows carry no `ok` at all - they are the categories the
|
|
93
|
+
// scorecard refuses to score. Comparing verdicts across them would invent
|
|
94
|
+
// one, which is the thing that section exists to avoid.
|
|
95
|
+
const measured = (rs) => (rs || []).filter((r) => r.kind !== "UNMEASURED");
|
|
96
|
+
const ma = new Map(measured(a.results).map((r) => [keyOf(r), r]));
|
|
97
|
+
const mb = new Map(measured(b.results).map((r) => [keyOf(r), r]));
|
|
98
|
+
|
|
99
|
+
const regressed = [];
|
|
100
|
+
const fixed = [];
|
|
101
|
+
const added = [];
|
|
102
|
+
const removed = [];
|
|
103
|
+
const changed = [];
|
|
104
|
+
|
|
105
|
+
for (const [k, rb] of mb) {
|
|
106
|
+
const ra = ma.get(k);
|
|
107
|
+
if (!ra) {
|
|
108
|
+
added.push(rb);
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
if (ra.ok && !rb.ok) regressed.push({ k, ra, rb });
|
|
112
|
+
else if (!ra.ok && rb.ok) fixed.push({ k, ra, rb });
|
|
113
|
+
// A metric whose verdict held but whose DETAIL moved is the early warning:
|
|
114
|
+
// coverage sliding from 72 to 69 passes the floor and is still the thing
|
|
115
|
+
// that will fail next month.
|
|
116
|
+
else if (ra.detail !== rb.detail) changed.push({ k, ra, rb });
|
|
117
|
+
}
|
|
118
|
+
for (const [k, ra] of ma) if (!mb.has(k)) removed.push({ k, ra });
|
|
119
|
+
|
|
120
|
+
const out = [];
|
|
121
|
+
out.push(`scorecard diff - ${a.at ?? aPath} → ${b.at ?? bPath}`);
|
|
122
|
+
out.push(` ${a.passed}/${a.passed + a.failed} → ${b.passed}/${b.passed + b.failed} passing`);
|
|
123
|
+
const section = (title, rows, fmt) => {
|
|
124
|
+
if (!rows.length) return;
|
|
125
|
+
out.push("");
|
|
126
|
+
out.push(`${title} (${rows.length})`);
|
|
127
|
+
for (const r of rows) out.push(` ${fmt(r)}`);
|
|
128
|
+
};
|
|
129
|
+
section("REGRESSED", regressed, ({ k, rb }) => `${k}\n now: ${rb.detail ?? "(no detail)"}`);
|
|
130
|
+
section("FIXED", fixed, ({ k, rb }) => `${k}\n now: ${rb.detail ?? "(no detail)"}`);
|
|
131
|
+
section("NEW METRIC", added, (r) => `${keyOf(r)} ${r.ok ? "passing" : "FAILING"}`);
|
|
132
|
+
section("METRIC GONE", removed, ({ k }) => `${k} - no longer measured`);
|
|
133
|
+
section(
|
|
134
|
+
"SAME VERDICT, DIFFERENT NUMBERS",
|
|
135
|
+
changed,
|
|
136
|
+
({ k, ra, rb }) =>
|
|
137
|
+
`${k}\n was: ${ra.detail ?? "(none)"}\n now: ${rb.detail ?? "(none)"}`,
|
|
138
|
+
);
|
|
139
|
+
|
|
140
|
+
if (!regressed.length && !fixed.length && !added.length && !removed.length && !changed.length) {
|
|
141
|
+
out.push("");
|
|
142
|
+
out.push(" nothing moved");
|
|
143
|
+
}
|
|
144
|
+
process.stdout.write(out.join("\n") + "\n");
|
|
145
|
+
return regressed.length ? 1 : 0;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
function main() {
|
|
149
|
+
const args = process.argv.slice(2);
|
|
150
|
+
if (args.includes("--save")) return save();
|
|
151
|
+
if (args.includes("--list")) {
|
|
152
|
+
const s = snapshots();
|
|
153
|
+
process.stdout.write(s.length ? s.join("\n") + "\n" : `no snapshots under ${STORE}\n`);
|
|
154
|
+
return 0;
|
|
155
|
+
}
|
|
156
|
+
if (args.includes("--diff")) {
|
|
157
|
+
const paths = args.filter((a) => !a.startsWith("--"));
|
|
158
|
+
if (paths.length === 2) return diff(paths[0], paths[1]);
|
|
159
|
+
const s = snapshots();
|
|
160
|
+
if (s.length < 2) {
|
|
161
|
+
process.stderr.write(
|
|
162
|
+
`need two snapshots to compare, found ${s.length} under ${STORE}\n` +
|
|
163
|
+
` take one with: node pipeline/scripts/scorecard-snapshot.mjs --save\n`,
|
|
164
|
+
);
|
|
165
|
+
return 2;
|
|
166
|
+
}
|
|
167
|
+
return diff(s[s.length - 2], s[s.length - 1]);
|
|
168
|
+
}
|
|
169
|
+
process.stderr.write("usage: scorecard-snapshot.mjs --save | --diff [a.json b.json] | --list\n");
|
|
170
|
+
return 2;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// `process.exitCode`, never `process.exit()`: stdout to a pipe is asynchronous
|
|
174
|
+
// and exit() discards whatever has not drained, which truncated two scripts in
|
|
175
|
+
// this repo at a buffer boundary and only ever when piped.
|
|
176
|
+
runMain("scorecard-snapshot", () => {
|
|
177
|
+
process.exitCode = main();
|
|
178
|
+
});
|
|
@@ -30,6 +30,24 @@
|
|
|
30
30
|
|
|
31
31
|
set -uo pipefail
|
|
32
32
|
|
|
33
|
+
# jq is not optional on this path. Without the guard below a missing binary
|
|
34
|
+
# renders as EMPTY DATA and the work continues on it; see lib/require-jq.sh.
|
|
35
|
+
for _rq in "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)/require-jq.sh" \
|
|
36
|
+
"$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")/../lib" 2>/dev/null && pwd)/require-jq.sh" \
|
|
37
|
+
"$HOME/.claude/lib/require-jq.sh" \
|
|
38
|
+
"$HOME/.copilot/lib/require-jq.sh" \
|
|
39
|
+
"$HOME/.codex/lib/require-jq.sh"; do
|
|
40
|
+
[ -f "$_rq" ] || continue
|
|
41
|
+
# shellcheck source=/dev/null
|
|
42
|
+
. "$_rq" && break
|
|
43
|
+
done
|
|
44
|
+
unset _rq
|
|
45
|
+
if ! command -v ma_require_jq >/dev/null 2>&1; then
|
|
46
|
+
# The helper itself is missing, which is an install problem, not a jq one.
|
|
47
|
+
ma_require_jq() { command -v jq >/dev/null 2>&1 || { echo "jq not found - cannot ${1:-continue}." >&2; return 1; }; }
|
|
48
|
+
fi
|
|
49
|
+
ma_require_jq "search the logs" || exit 3
|
|
50
|
+
|
|
33
51
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
34
52
|
|
|
35
53
|
ROOT="$HOME/.claude/logs/multi-agent"
|
|
@@ -204,21 +204,21 @@ echo "→ reviewer-count contract (Claude=3, Copilot=3, Codex=3)"
|
|
|
204
204
|
# a broken installation. _smoke-root.sh handles both layouts.
|
|
205
205
|
# shellcheck source=pipeline/scripts/_smoke-root.sh
|
|
206
206
|
. "$(dirname "${BASH_SOURCE[0]}")/_smoke-root.sh"
|
|
207
|
-
P4="${MA_REFS:+$MA_REFS/phases/phase-
|
|
207
|
+
P4="${MA_REFS:+$MA_REFS/phases/phase-3-review.md}"
|
|
208
208
|
REVSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/reviewer-output.schema.json}"
|
|
209
209
|
TRSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/triage-output.schema.json}"
|
|
210
210
|
|
|
211
211
|
if [ -z "$P4" ] || [ ! -f "$P4" ]; then
|
|
212
|
-
echo " ↷ SKIP: phase-
|
|
212
|
+
echo " ↷ SKIP: phase-3-review.md not present in this $MA_LAYOUT layout"
|
|
213
213
|
# Two independent statements, because they can drift apart: the count sentence is
|
|
214
214
|
# the contract, and the matrix is what a reader dispatches from. Three regexes over
|
|
215
215
|
# overlapping prose used to stand in for this and let the count sentence 300 lines
|
|
216
216
|
# further down go stale for a whole release without failing.
|
|
217
217
|
elif grep -qF "Claude Code 3, Copilot CLI 3, Codex CLI 3" "$P4" \
|
|
218
218
|
&& grep -qE '^\| Reviewer 3 .*\|.*\|.*\|.*\|' "$P4"; then
|
|
219
|
-
pass "phase-
|
|
219
|
+
pass "phase-3-review declares Claude=3 / Copilot=3 / Codex=3 reviewers, and the matrix has all three host columns"
|
|
220
220
|
else
|
|
221
|
-
fail "phase-
|
|
221
|
+
fail "phase-3-review does not declare the CLI-aware reviewer count for all three hosts"
|
|
222
222
|
fi
|
|
223
223
|
|
|
224
224
|
# The two Codex constraints are silent-failure shaped, so the contract has to name
|
|
@@ -226,9 +226,9 @@ fi
|
|
|
226
226
|
# 4-slot ceiling (orchestrator included) is why the count is 3 and not more.
|
|
227
227
|
if [ -n "$P4" ] && [ -f "$P4" ]; then
|
|
228
228
|
if grep -q 'fork_turns' "$P4" && grep -qiE "concurrency|slots" "$P4"; then
|
|
229
|
-
pass "phase-
|
|
229
|
+
pass "phase-3-review documents the fork_turns override rule + the concurrency ceiling"
|
|
230
230
|
else
|
|
231
|
-
fail "phase-
|
|
231
|
+
fail "phase-3-review must document fork_turns and the Codex concurrency ceiling"
|
|
232
232
|
fi
|
|
233
233
|
fi
|
|
234
234
|
|
|
@@ -24,6 +24,26 @@ TEMPLATE="$SMOKE_DIR/../preferences-template.json"
|
|
|
24
24
|
[ -f "$TEMPLATE" ] || TEMPLATE="$HOME/multi-agent-pipeline/pipeline/preferences-template.json"
|
|
25
25
|
LIVE_PREFS="$HOME/.claude/multi-agent-preferences.json"
|
|
26
26
|
|
|
27
|
+
# The migration target, derived the way migrate-prefs.mjs derives it: the last
|
|
28
|
+
# entry of the schema's schemaVersion enum, read from the schema that sits
|
|
29
|
+
# beside the migrator being checked. It used to be grepped as a
|
|
30
|
+
# `TARGET_VERSION = "x.y.z"` literal out of the migrator, and when that literal
|
|
31
|
+
# was replaced by the schema read - precisely because a literal had drifted a
|
|
32
|
+
# minor behind - the grep started matching nothing and both checks that depend
|
|
33
|
+
# on it reported "no reference point". A gate that reads the source of truth
|
|
34
|
+
# cannot go stale against it; a gate that reads a transcription of it can.
|
|
35
|
+
migration_target() {
|
|
36
|
+
local schema="$1/../schemas/prefs.schema.json"
|
|
37
|
+
[ -f "$schema" ] || schema="$PREFS_SCHEMA"
|
|
38
|
+
[ -f "$schema" ] || return 1
|
|
39
|
+
node -e "
|
|
40
|
+
const sv = JSON.parse(require('fs').readFileSync('$schema','utf8'))?.properties?.schemaVersion;
|
|
41
|
+
const v = sv?.const ?? sv?.enum?.at(-1);
|
|
42
|
+
if (!v) process.exit(1);
|
|
43
|
+
process.stdout.write(v);
|
|
44
|
+
" 2>/dev/null
|
|
45
|
+
}
|
|
46
|
+
|
|
27
47
|
# ──────────────────────────────────────────────────────────────────────────
|
|
28
48
|
echo "→ 1. Schema files parse as JSON"
|
|
29
49
|
for f in "$PREFS_SCHEMA" "$STATE_SCHEMA"; do
|
|
@@ -145,10 +165,9 @@ if [ -f "$TEMPLATE" ]; then
|
|
|
145
165
|
# every old entry in the migrator's accepted set became load-bearing purely to
|
|
146
166
|
# rescue the template this gate was holding back. A template behind the target
|
|
147
167
|
# is a defect, not the expected shape.
|
|
148
|
-
TEMPLATE_TARGET=$(
|
|
149
|
-
| grep -oE '[0-9]+\.[0-9]+\.[0-9]+')
|
|
168
|
+
TEMPLATE_TARGET=$(migration_target "$SMOKE_DIR")
|
|
150
169
|
if [ -z "$TEMPLATE_TARGET" ]; then
|
|
151
|
-
fail "cannot
|
|
170
|
+
fail "cannot derive the migration target from prefs.schema.json - the check has no reference point"
|
|
152
171
|
elif [ "$TVER" = "$TEMPLATE_TARGET" ]; then
|
|
153
172
|
pass "template schemaVersion: $TVER (at the migration target)"
|
|
154
173
|
else
|
|
@@ -201,17 +220,17 @@ if [ -f "$LIVE_PREFS" ]; then
|
|
|
201
220
|
# correctly migrated to 2.4.0 hit the else arm and FAILED as "unknown". The
|
|
202
221
|
# gate was rejecting the only fully-migrated state it exists to encourage.
|
|
203
222
|
#
|
|
204
|
-
# Reading the target
|
|
205
|
-
#
|
|
223
|
+
# Reading the target and the accepted set from the same schema enum the
|
|
224
|
+
# migrator reads means this can never disagree with it again.
|
|
206
225
|
MIGRATOR="$SMOKE_DIR/migrate-prefs.mjs"
|
|
207
|
-
TARGET=$(
|
|
226
|
+
TARGET=$(migration_target "$(dirname "$MIGRATOR")")
|
|
208
227
|
KNOWN=$(node -e "
|
|
209
228
|
const s = require('$PREFS_SCHEMA');
|
|
210
229
|
process.stdout.write((s.properties.schemaVersion.enum || []).join(' '));
|
|
211
230
|
" 2>/dev/null)
|
|
212
231
|
|
|
213
232
|
if [ -z "$TARGET" ]; then
|
|
214
|
-
fail "cannot
|
|
233
|
+
fail "cannot derive the migration target from prefs.schema.json - the check has no reference point"
|
|
215
234
|
elif [ "$LVER" = "$TARGET" ]; then
|
|
216
235
|
pass "live prefs at the migration target (v$TARGET)"
|
|
217
236
|
elif [ "$LVER" = "none" ]; then
|
|
@@ -392,8 +392,9 @@ function main() {
|
|
|
392
392
|
gaps,
|
|
393
393
|
};
|
|
394
394
|
|
|
395
|
+
// Returning, not exiting: `gaps` grows with the repo and process.exit() would
|
|
396
|
+
// cut the payload at the pipe buffer.
|
|
395
397
|
process.stdout.write(PRETTY ? JSON.stringify(out, null, 2) + "\n" : JSON.stringify(out) + "\n");
|
|
396
|
-
process.exit(0);
|
|
397
398
|
}
|
|
398
399
|
|
|
399
400
|
try {
|
|
@@ -29,6 +29,7 @@
|
|
|
29
29
|
* only on a setup error (unreadable/invalid input is treated as "no findings").
|
|
30
30
|
*/
|
|
31
31
|
import { readFileSync } from "node:fs";
|
|
32
|
+
import { runMain } from "../lib/fatal.mjs";
|
|
32
33
|
|
|
33
34
|
const args = process.argv.slice(2);
|
|
34
35
|
const fileFlag = args.indexOf("--file");
|
|
@@ -92,4 +93,4 @@ function main() {
|
|
|
92
93
|
);
|
|
93
94
|
}
|
|
94
95
|
|
|
95
|
-
main
|
|
96
|
+
runMain("test-integrity-gate", main);
|