@mmerterden/multi-agent-pipeline 18.0.0 → 19.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +183 -0
- package/README.md +34 -18
- package/README.tr.md +14 -16
- package/docs/adr/0002-instruction-driven-flag.md +1 -0
- package/docs/adr/0005-lazy-phase-docs.md +11 -1
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0010-own-code-graph.md +1 -0
- package/docs/adr/0014-six-phase-consolidation.md +134 -0
- package/docs/adr/README.md +2 -1
- package/docs/architecture.md +37 -38
- package/docs/best-practices.md +1 -1
- package/docs/ecosystem.md +37 -26
- package/docs/engineering.md +1 -1
- package/docs/facts.json +45 -0
- package/docs/features.md +54 -53
- package/docs/performance.md +5 -5
- package/docs/recovery-guide.md +9 -9
- package/docs/token-budget-history.md +3 -1
- package/index.js +2 -2
- package/install/_codex-agents.mjs +1 -1
- package/install/templates/claude-hooks.json +1 -1
- package/install/templates/codex-instructions.md +1 -1
- package/install/templates/copilot-instructions.md +28 -28
- package/manifest.json +209 -193
- package/package.json +2 -2
- package/pipeline/agents/dev-critic.md +3 -3
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +8 -8
- package/pipeline/commands/multi-agent/analysis/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
- package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
- package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
- package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
- package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
- package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
- package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/status/SKILL.md +5 -5
- package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
- package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
- package/pipeline/lib/credential-inventory.sh +1 -1
- package/pipeline/lib/fetch-fortify.sh +1 -1
- package/pipeline/lib/model-rung.sh +142 -0
- package/pipeline/lib/phase-schema.mjs +88 -0
- package/pipeline/lib/plan-todos.sh +5 -5
- package/pipeline/lib/route-state.sh +161 -0
- package/pipeline/lib/run-paths.sh +2 -2
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +1 -1
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/analysis/evidence.md +0 -9
- package/pipeline/multi-agent-refs/analysis/intake.md +1 -1
- package/pipeline/multi-agent-refs/analysis/locked.md +21 -22
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/analysis/synthesis.md +12 -6
- package/pipeline/multi-agent-refs/android-guide.md +1 -1
- package/pipeline/multi-agent-refs/audit-guide.md +13 -13
- package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
- package/pipeline/multi-agent-refs/channels/jira.md +3 -3
- package/pipeline/multi-agent-refs/channels/pr.md +4 -4
- package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +3 -3
- package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +4 -4
- package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
- package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
- package/pipeline/multi-agent-refs/features/doctor.md +2 -2
- package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
- package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
- package/pipeline/multi-agent-refs/features/model-fallback.md +5 -5
- package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
- package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
- package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
- package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
- package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
- package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
- package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
- package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
- package/pipeline/multi-agent-refs/knowledge.md +11 -11
- package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
- package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
- package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
- package/pipeline/multi-agent-refs/phases/modes.md +30 -30
- package/pipeline/multi-agent-refs/phases/operations.md +8 -8
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +24 -24
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
- package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
- package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
- package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
- package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
- package/pipeline/multi-agent-refs/phases.md +44 -48
- package/pipeline/multi-agent-refs/picker-contract.md +1 -1
- package/pipeline/multi-agent-refs/progress-contract.md +6 -6
- package/pipeline/multi-agent-refs/readiness-review.md +1 -1
- package/pipeline/multi-agent-refs/rules.md +7 -7
- package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
- package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
- package/pipeline/preferences-template.json +9 -1
- package/pipeline/rules/outside-the-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +50 -50
- package/pipeline/schemas/analysis-output.schema.json +2 -2
- package/pipeline/schemas/autopilot-config.schema.json +1 -1
- package/pipeline/schemas/code-graph.schema.json +1 -1
- package/pipeline/schemas/criteria-manifest.schema.json +1 -1
- package/pipeline/schemas/dev-critic-output.schema.json +1 -1
- package/pipeline/schemas/diff-risk.schema.json +1 -1
- package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
- package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
- package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
- package/pipeline/schemas/phases.json +105 -0
- package/pipeline/schemas/plan-todos.schema.json +5 -5
- package/pipeline/schemas/planning-output.schema.json +1 -1
- package/pipeline/schemas/prefs.schema.json +100 -56
- package/pipeline/schemas/reviewer-output.schema.json +3 -3
- package/pipeline/schemas/route-config.schema.json +74 -0
- package/pipeline/schemas/scope-check.schema.json +1 -1
- package/pipeline/schemas/test-gap.schema.json +1 -1
- package/pipeline/schemas/token-budget.json +12 -18
- package/pipeline/schemas/triage-output.schema.json +6 -6
- package/pipeline/scripts/README.md +3 -3
- package/pipeline/scripts/_code-graph.mjs +2 -2
- package/pipeline/scripts/_run-paths.mjs +2 -2
- package/pipeline/scripts/_smoke-root.sh +1 -1
- package/pipeline/scripts/aggregate-metrics.mjs +1 -1
- package/pipeline/scripts/capture-flush.sh +8 -8
- package/pipeline/scripts/capture-resume.sh +3 -3
- package/pipeline/scripts/classify-plan-safety.mjs +1 -1
- package/pipeline/scripts/diff-explain.mjs +1 -1
- package/pipeline/scripts/doctor.mjs +2 -2
- package/pipeline/scripts/gc-abandoned.sh +3 -3
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +1 -1
- package/pipeline/scripts/gen-facts.mjs +175 -0
- package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
- package/pipeline/scripts/gen-ref-toc.mjs +1 -1
- package/pipeline/scripts/graph-report.mjs +1 -1
- package/pipeline/scripts/jira-attach.sh +1 -1
- package/pipeline/scripts/learn-from-transcripts.mjs +1 -1
- package/pipeline/scripts/learning-curve.mjs +2 -2
- package/pipeline/scripts/log-metric.sh +17 -4
- package/pipeline/scripts/memory-save.sh +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +22 -5
- package/pipeline/scripts/phase-banner.sh +20 -20
- package/pipeline/scripts/phase-tracker.sh +7 -7
- package/pipeline/scripts/plan-coverage-gate.mjs +2 -2
- package/pipeline/scripts/render-agent-log-cost.sh +1 -1
- package/pipeline/scripts/render-work-summary.sh +3 -3
- package/pipeline/scripts/review-file-filter.mjs +1 -1
- package/pipeline/scripts/run-aggregator.mjs +13 -6
- package/pipeline/scripts/run-metrics.mjs +1 -1
- package/pipeline/scripts/runs-index.mjs +11 -1
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
- package/pipeline/scripts/smoke-schema-validation.sh +26 -7
- package/pipeline/scripts/token-budget-report.mjs +13 -2
- package/pipeline/scripts/triage-memory.mjs +2 -2
- package/pipeline/scripts/validate-analysis-doc.mjs +73 -17
- package/pipeline/scripts/validate-planning.mjs +1 -1
- package/pipeline/scripts/validate-reviewer.mjs +1 -1
- package/pipeline/scripts/validate-state.mjs +45 -5
- package/pipeline/scripts/validate-triage.mjs +3 -3
- package/pipeline/scripts/worktree-finalize.sh +5 -5
- package/pipeline/skills/.skill-manifest.json +37 -21
- package/pipeline/skills/.skills-index.json +49 -5
- package/pipeline/skills/shared/README.md +10 -6
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +2 -2
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +69 -71
- package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
- package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
- package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
- package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
- package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
- package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
- package/pipeline/skills/skills-index.md +8 -4
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/scope-check.schema.json",
|
|
4
4
|
"version": "1.0.0",
|
|
5
5
|
"title": "Multi-Agent Pipeline - Phase 3 scope self-check",
|
|
6
|
-
"description": "Written by Phase
|
|
6
|
+
"description": "Written by Phase 2 Step 3.7 to $WORKTREE/.pipeline/scope-check.json before the Phase 4 handoff: one stated reason per file in the diff, the changes deliberately not made, and the code-simplifier rationales. scope-check-gate.mjs compares files[] with the real diff; Phase 4 injects the record as <scope-self-check>; Phase 4 builds the PR Changes bullets and the follow-up list from it.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"required": ["version", "taskId", "files"],
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/test-gap.schema.json",
|
|
4
4
|
"version": "1.0.0",
|
|
5
|
-
"title": "Multi-Agent Pipeline - Phase
|
|
5
|
+
"title": "Multi-Agent Pipeline - Phase 3 test gap report",
|
|
6
6
|
"description": "Output of test-gap-scan.mjs. Lists symbols added/changed in the diff that have no paired test file or no matching test method. Advisory in default mode; opt-in blocking via prefs.testGap.blockingThreshold.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
@@ -1,32 +1,26 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/token-budget.json",
|
|
4
|
-
"description": "Per-phase token ceilings for the lazy-loaded pipeline docs, enforced by smoke-token-budget.sh. Only the ACTIVE phase is loaded at run time and nothing truncates a phase document, so these numbers govern what may be written, not what a run receives. Two rules, and the split is the point. `max_tokens` and `total_max_tokens` are committed constants a human owns: computing them would end the gate, because a ceiling that is always current x k can never fail. The warn tier is NOT stored - the gate derives it per phase from that phase's own git history (median + 3*MAD of its historical per-commit deltas, which is robust to the one 980-token release that makes sigma meaningless on phase-4), because a soft line maintained by hand rots, and this one rotted twice: it was reset at v13.6.0 when five lines had gone permanently amber, and six of eight were amber again by v17.1.0. A ceiling more than 25% above the measurement is stale and fails, which is the 'only ratchets down' rule the SKILL.md grace list always had and this budget never did. Change history: docs/token-budget-history.md.",
|
|
4
|
+
"description": "Per-phase token ceilings for the lazy-loaded pipeline docs, enforced by smoke-token-budget.sh. Only the ACTIVE phase is loaded at run time and nothing truncates a phase document, so these numbers govern what may be written, not what a run receives. Two rules, and the split is the point. `max_tokens` and `total_max_tokens` are committed constants a human owns: computing them would end the gate, because a ceiling that is always current x k can never fail. The warn tier is NOT stored - the gate derives it per phase from that phase's own git history (median + 3*MAD of its historical per-commit deltas, which is robust to the one 980-token release that makes sigma meaningless on phase-4), because a soft line maintained by hand rots, and this one rotted twice: it was reset at v13.6.0 when five lines had gone permanently amber, and six of eight were amber again by v17.1.0. A ceiling more than 25% above the measurement is stale and fails, which is the 'only ratchets down' rule the SKILL.md grace list always had and this budget never did. Ceilings are DERIVED from schemas/phases.json and re-measured after the v19.0.0 six-phase merge, not summed from the eight old ones: the two-sided rule fails a ceiling more than 25% above the measurement, so a sum would have shipped stale by construction. Change history: docs/token-budget-history.md.",
|
|
5
5
|
"phases": {
|
|
6
6
|
"phase-0-init": {
|
|
7
7
|
"max_tokens": 13400
|
|
8
8
|
},
|
|
9
|
-
"phase-1-
|
|
10
|
-
"max_tokens":
|
|
9
|
+
"phase-1-plan": {
|
|
10
|
+
"max_tokens": 10000
|
|
11
11
|
},
|
|
12
|
-
"phase-2-
|
|
13
|
-
"max_tokens":
|
|
14
|
-
},
|
|
15
|
-
"phase-3-dev": {
|
|
16
|
-
"max_tokens": 9450
|
|
12
|
+
"phase-2-dev": {
|
|
13
|
+
"max_tokens": 10650
|
|
17
14
|
},
|
|
18
|
-
"phase-
|
|
19
|
-
"max_tokens":
|
|
15
|
+
"phase-3-review": {
|
|
16
|
+
"max_tokens": 17200
|
|
20
17
|
},
|
|
21
|
-
"phase-
|
|
22
|
-
"max_tokens":
|
|
23
|
-
},
|
|
24
|
-
"phase-6-commit": {
|
|
25
|
-
"max_tokens": 6550
|
|
18
|
+
"phase-4-commit": {
|
|
19
|
+
"max_tokens": 6500
|
|
26
20
|
},
|
|
27
|
-
"phase-
|
|
28
|
-
"max_tokens":
|
|
21
|
+
"phase-5-report": {
|
|
22
|
+
"max_tokens": 5550
|
|
29
23
|
}
|
|
30
24
|
},
|
|
31
|
-
"total_max_tokens":
|
|
25
|
+
"total_max_tokens": 63300
|
|
32
26
|
}
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/triage-output.schema.json",
|
|
4
4
|
"version": "3.4.0",
|
|
5
|
-
"title": "Multi-Agent Pipeline - Phase
|
|
6
|
-
"description": "Contract for the Opus triage agent's JSON output in Phase 4 Step 3. Triage consumes merged reviewer findings and splits them into accepted/deferred/rejected. Only `accepted` blocking/important items trigger Phase 3 rework. v3.1.0 adds the optional `consensus` block so triage can surface reviewer-agreement risk (false consensus among same-base-model reviewers) instead of silently merging. v3.2.0 adds the optional per-finding `verification` block written by Phase
|
|
5
|
+
"title": "Multi-Agent Pipeline - Phase 3 triage output",
|
|
6
|
+
"description": "Contract for the Opus triage agent's JSON output in Phase 4 Step 3. Triage consumes merged reviewer findings and splits them into accepted/deferred/rejected. Only `accepted` blocking/important items trigger Phase 3 rework. v3.1.0 adds the optional `consensus` block so triage can surface reviewer-agreement risk (false consensus among same-base-model reviewers) instead of silently merging. v3.2.0 adds the optional per-finding `verification` block written by Phase 3 Step 3.7 (verify-by-test): the empirical repro-test outcome for accepted blocking findings. v3.3.0 carries ruleId + criteriaSource through triage so a finding that cites a stable rule ID keeps that citation into Phase 4 and Phase 5, and the lesson loop can key durable learnings by rule. v3.4.0 adds the optional per-finding fingerprint (finding-fingerprint.mjs) so review-delta.mjs can tell still-present, resolved and new findings apart between rounds.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"required": ["accepted", "deferred", "rejected", "approved"],
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
},
|
|
18
18
|
"deferred": {
|
|
19
19
|
"type": "array",
|
|
20
|
-
"description": "Findings that are real but out of current task scope. Surfaced in Phase
|
|
20
|
+
"description": "Findings that are real but out of current task scope. Surfaced in Phase 5 report; not actioned.",
|
|
21
21
|
"items": {
|
|
22
22
|
"type": "object",
|
|
23
23
|
"additionalProperties": false,
|
|
@@ -55,7 +55,7 @@
|
|
|
55
55
|
},
|
|
56
56
|
"approved": {
|
|
57
57
|
"type": "boolean",
|
|
58
|
-
"description": "True if no accepted BLOCKING items remain. The pipeline uses this to gate Phase 5 / Phase
|
|
58
|
+
"description": "True if no accepted BLOCKING items remain. The pipeline uses this to gate Phase 5 / Phase 4 entry."
|
|
59
59
|
},
|
|
60
60
|
"consensus": {
|
|
61
61
|
"$ref": "#/$defs/consensus"
|
|
@@ -111,7 +111,7 @@
|
|
|
111
111
|
},
|
|
112
112
|
"disagreements": {
|
|
113
113
|
"type": "array",
|
|
114
|
-
"description": "Findings where reviewers split (existence or severity). Surfaced verbatim in the Phase
|
|
114
|
+
"description": "Findings where reviewers split (existence or severity). Surfaced verbatim in the Phase 5 report and at the Step 4 user checkpoint.",
|
|
115
115
|
"items": {
|
|
116
116
|
"type": "object",
|
|
117
117
|
"additionalProperties": false,
|
|
@@ -142,7 +142,7 @@
|
|
|
142
142
|
"verification": {
|
|
143
143
|
"type": "object",
|
|
144
144
|
"additionalProperties": false,
|
|
145
|
-
"description": "v3.2.0 verify-by-test outcome (Phase
|
|
145
|
+
"description": "v3.2.0 verify-by-test outcome (Phase 3 Step 3.7, opt-in via prefs.global.verifyByTest). confirmed = repro test failed as the finding predicts (finding stands, test kept as the Phase 3 RED test); not-reproduced = repro test passed under evidence-gate (finding downgraded to deferred); inconclusive = compile error / timeout / not unit-testable (judgment verdict stands).",
|
|
146
146
|
"required": ["result"],
|
|
147
147
|
"properties": {
|
|
148
148
|
"result": {
|
|
@@ -19,10 +19,10 @@ Validate contracts. Each emits `══ <name> smoke: N passed, M failed ══`
|
|
|
19
19
|
|
|
20
20
|
### Phase contracts
|
|
21
21
|
- `smoke-phase-0-multi-repo.sh` - Phase 0 multi-repo mode fetch + worktree atomicity
|
|
22
|
-
- `smoke-phase-6-multi.sh` - Phase
|
|
22
|
+
- `smoke-phase-6-multi.sh` - Phase 4 multi-repo commit/PR cross-linking
|
|
23
23
|
- `smoke-phase-banner.sh` + `smoke-phase-tracker.sh` - Phase UI output contracts
|
|
24
|
-
- `smoke-phase4-triage.sh` - Phase
|
|
25
|
-
- `smoke-verify-by-test.sh` - Phase
|
|
24
|
+
- `smoke-phase4-triage.sh` - Phase 3 reviewer → triage flow
|
|
25
|
+
- `smoke-verify-by-test.sh` - Phase 3 Step 3.7 verify-by-test contract (v10.8.0)
|
|
26
26
|
- `smoke-handoff-contract.sh` - phase-boundary structured handoff + handoff-first resume (v10.8.0)
|
|
27
27
|
- `smoke-update-check.sh` - Phase 0 Step 0.6 update-check + required-floor contract (v10.9.0, floor v15.14.0)
|
|
28
28
|
- `smoke-context-links.sh` - context-link-extractor classification contract, all types (v15.14.0)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @file _code-graph.mjs - deterministic, LLM-free code graph for Phase 1 and Phase
|
|
2
|
+
* @file _code-graph.mjs - deterministic, LLM-free code graph for Phase 1 and Phase 5.
|
|
3
3
|
*
|
|
4
|
-
* Phase 1 narrows its Explore fan-out with this graph, and Phase
|
|
4
|
+
* Phase 1 narrows its Explore fan-out with this graph, and Phase 5 refreshes it
|
|
5
5
|
* instead of hand-writing architecture.md from a single task's window. The
|
|
6
6
|
* design follows graphify (Graphify-Labs/graphify): AST-quality extraction with
|
|
7
7
|
* zero LLM cost, one graph file, an incremental manifest gate, god-node hubs,
|
|
@@ -76,7 +76,7 @@ export function canonicalRunDir(taskId, project) {
|
|
|
76
76
|
}
|
|
77
77
|
|
|
78
78
|
/**
|
|
79
|
-
* Phase
|
|
79
|
+
* Phase 4 removes the worktree and salvages the run's files into an
|
|
80
80
|
* `artifacts/` subdirectory of the same run directory. A finished run therefore
|
|
81
81
|
* keeps its state one level deeper, and a reader that only looks at the top
|
|
82
82
|
* level reports a shipped task as having no state at all.
|
|
@@ -232,7 +232,7 @@ export function resolveRunFile(taskId, filename, project) {
|
|
|
232
232
|
for (const v of taskIdVariants(taskId)) {
|
|
233
233
|
const dir = resolveRunDir(v, project);
|
|
234
234
|
if (!dir) continue;
|
|
235
|
-
// Top level first, then the salvaged copy Phase
|
|
235
|
+
// Top level first, then the salvaged copy Phase 4 leaves behind.
|
|
236
236
|
for (const base of [dir, join(dir, ARTIFACTS_SUBDIR)]) {
|
|
237
237
|
const p = join(base, filename);
|
|
238
238
|
if (existsSync(p)) return p;
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
#
|
|
21
21
|
# Usage (from any script in pipeline/scripts/):
|
|
22
22
|
# . "$(dirname "${BASH_SOURCE[0]}")/_smoke-root.sh"
|
|
23
|
-
# grep -q needle "$MA_REFS/phases/phase-
|
|
23
|
+
# grep -q needle "$MA_REFS/phases/phase-3-review.md"
|
|
24
24
|
#
|
|
25
25
|
# Exports:
|
|
26
26
|
# MA_LAYOUT "repo" | "install"
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
// by --since=<ISO date> / --task-id=<id> / --phase=<N>.
|
|
7
7
|
//
|
|
8
8
|
// Output is plain-text by default; pass --json for machine-readable output
|
|
9
|
-
// (used by Phase
|
|
9
|
+
// (used by Phase 5 to embed a metrics block in the run report).
|
|
10
10
|
//
|
|
11
11
|
// Aggregations:
|
|
12
12
|
// - Tasks completed
|
|
@@ -5,14 +5,14 @@
|
|
|
5
5
|
#
|
|
6
6
|
# WHY THIS EXISTS
|
|
7
7
|
#
|
|
8
|
-
# Every persistent write used to live in Phase
|
|
9
|
-
# ledger distill, the knowledge-base append, the code-graph refresh. And Phase
|
|
8
|
+
# Every persistent write used to live in Phase 5: triage ingest, the learnings
|
|
9
|
+
# ledger distill, the knowledge-base append, the code-graph refresh. And Phase 5
|
|
10
10
|
# is, by the pipeline's own admission in features/code-graph.md, the phase a run
|
|
11
11
|
# is LEAST likely to reach. A run killed in Phase 3, a session that hits its
|
|
12
12
|
# context ceiling, a crash after review - each one threw away everything it had
|
|
13
13
|
# established, and the next run on the same repo rediscovered it from scratch.
|
|
14
14
|
#
|
|
15
|
-
# So the writes move here, and Phase
|
|
15
|
+
# So the writes move here, and Phase 5 becomes the LAST flush rather than the
|
|
16
16
|
# only one. Phase boundaries call this, and so does SessionEnd. Nothing about
|
|
17
17
|
# the trigger depends on a model noticing that a moment qualifies: it hangs on
|
|
18
18
|
# a phase transition and on process exit, both objectively visible without any
|
|
@@ -20,15 +20,15 @@
|
|
|
20
20
|
#
|
|
21
21
|
# What it does NOT do: call a model. Everything here is derived from artefacts
|
|
22
22
|
# already on disk (triage-output.json) plus agent-state.json. The parts of
|
|
23
|
-
# Phase
|
|
24
|
-
# per-repo memory synthesis - stay in Phase
|
|
23
|
+
# Phase 5 that genuinely need a model - the knowledge-base extraction, the
|
|
24
|
+
# per-repo memory synthesis - stay in Phase 5, because a hook cannot think.
|
|
25
25
|
#
|
|
26
26
|
# Usage:
|
|
27
27
|
# ./capture-flush.sh [--state <agent-state.json>] [--if-stale] [--json] [--quiet]
|
|
28
28
|
#
|
|
29
29
|
# --state the run to flush. Default: resolved from the newest task dir
|
|
30
30
|
# under $HOME/.claude/logs/multi-agent (see resolve_state).
|
|
31
|
-
# --if-stale flush only when the run did NOT complete Phase
|
|
31
|
+
# --if-stale flush only when the run did NOT complete Phase 5 - the
|
|
32
32
|
# SessionEnd case. A finished run has already flushed.
|
|
33
33
|
# --json machine-readable result for a caller that wants to count rows.
|
|
34
34
|
# --quiet no stdout. Exit status still distinguishes the outcomes.
|
|
@@ -101,13 +101,13 @@ WORKTREE=$(jq -r '.worktreePath // empty' "$STATE" 2>/dev/null)
|
|
|
101
101
|
PHASE7=$(jq -r '[.phases[]? | select((.id // "") == "7") | .status] | first // ""' "$STATE" 2>/dev/null)
|
|
102
102
|
|
|
103
103
|
if [ "$IF_STALE" -eq 1 ] && [ "$PHASE7" = "completed" ]; then
|
|
104
|
-
say "capture-flush: ${TASK_ID:-run} already completed Phase
|
|
104
|
+
say "capture-flush: ${TASK_ID:-run} already completed Phase 5 - nothing stale"
|
|
105
105
|
[ "$JSON" -eq 1 ] && printf '{"status":"noop","reason":"already-flushed","taskId":"%s"}\n' "$TASK_ID"
|
|
106
106
|
exit 0
|
|
107
107
|
fi
|
|
108
108
|
|
|
109
109
|
# The triage artefact is the only input either store needs, and it has two homes:
|
|
110
|
-
# Phase
|
|
110
|
+
# Phase 4 removes the worktree once the PR is open, so the salvaged copy under
|
|
111
111
|
# artifactsPath is tried FIRST. Reading the worktree path first would degrade
|
|
112
112
|
# silently for exactly the runs this script exists to rescue.
|
|
113
113
|
TRIAGE=""
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
# the user learns to skip, which costs more than it saves.
|
|
12
12
|
#
|
|
13
13
|
# It answers two questions:
|
|
14
|
-
# - Is there a pipeline run that stopped before Phase
|
|
14
|
+
# - Is there a pipeline run that stopped before Phase 5? Name it and the resume
|
|
15
15
|
# command, because that run's work is recoverable and its findings are not
|
|
16
16
|
# yet in the durable stores until it flushes.
|
|
17
17
|
# - Is the pipeline's own observation queue stale? (>= REVIEW_DAYS since the
|
|
@@ -32,7 +32,7 @@ REVIEW_DAYS=7
|
|
|
32
32
|
# Three buckets, not one line. Measured on the development machine: 22 runs read
|
|
33
33
|
# `in_progress` and every one was over a day old, but they are not one failure.
|
|
34
34
|
# 13 stopped at Phase 0, which is almost entirely questions - those runs never
|
|
35
|
-
# started. 3 had their PR already open and were waiting at Phase
|
|
35
|
+
# started. 3 had their PR already open and were waiting at Phase 4/7, where the
|
|
36
36
|
# pipeline pauses ON PURPOSE for channel selection. 4 died mid-development.
|
|
37
37
|
#
|
|
38
38
|
# Reporting the newest one of those as "stopped at Phase N" told the truth about
|
|
@@ -71,7 +71,7 @@ if [ -d "$LOGS" ] && command -v jq >/dev/null 2>&1; then
|
|
|
71
71
|
PR=$(jq -r 'if (.pr|type)=="string" then .pr elif (.pr|type)=="object" then (.pr.url // .pr.number // "") else "" end | tostring' "$f" 2>/dev/null)
|
|
72
72
|
|
|
73
73
|
# Waiting for you, not broken: an open PR means the work landed, and
|
|
74
|
-
# Phase
|
|
74
|
+
# Phase 5 pauses for channel selection by design (modes.md).
|
|
75
75
|
if [ "$STATUS" = "awaiting_input" ] || [ -n "$PR" ] ||
|
|
76
76
|
[ "$PHASE" = "6" ] || [ "$PHASE" = "7" ]; then
|
|
77
77
|
AWAITING_N=$((AWAITING_N + 1))
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
// classify-plan-safety.mjs - v7.0.G
|
|
3
3
|
//
|
|
4
|
-
// Heuristic safety classifier for a Phase
|
|
4
|
+
// Heuristic safety classifier for a Phase 1 plan. Autopilot's "zero user
|
|
5
5
|
// interaction" contract is fine for small, predictable tasks but dangerous
|
|
6
6
|
// when a plan touches the security path, deletes files without paired
|
|
7
7
|
// tests, or sprawls across many files. This script inspects the plan and
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
/**
|
|
4
4
|
* @file diff-explain.mjs - v7.8.0 Paket B
|
|
5
5
|
*
|
|
6
|
-
* Bridges Phase
|
|
6
|
+
* Bridges Phase 3 triage output to the actual git diff. For each accepted /
|
|
7
7
|
* deferred / rejected finding, locates the corresponding hunk in the branch's
|
|
8
8
|
* diff against base and renders an annotated markdown report.
|
|
9
9
|
*
|
|
@@ -744,7 +744,7 @@ function checkDiskSpace() {
|
|
|
744
744
|
}
|
|
745
745
|
|
|
746
746
|
// A finished task removes its own worktree at PR time, but a run that dies
|
|
747
|
-
// before Phase
|
|
747
|
+
// before Phase 4 never reaches that step and nothing else collects it: the
|
|
748
748
|
// finalizer only runs on success and gc-worktrees only sweeps entries git no
|
|
749
749
|
// longer knows about. So worktrees accumulate silently, and each one is a full
|
|
750
750
|
// second checkout. Measured on one real iOS repo: 15 left behind, 11 GB.
|
|
@@ -788,7 +788,7 @@ function checkWorktreeResidue() {
|
|
|
788
788
|
report(
|
|
789
789
|
"worktree-residue",
|
|
790
790
|
"WARN",
|
|
791
|
-
`${entries.length} worktree(s) left under ${wt}${size}; runs that stopped before Phase
|
|
791
|
+
`${entries.length} worktree(s) left under ${wt}${size}; runs that stopped before Phase 4 are never collected`,
|
|
792
792
|
"run /multi-agent:garbage-collect, or /multi-agent:kill for a task you know is dead",
|
|
793
793
|
);
|
|
794
794
|
return;
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
# Three things keep this from being a foot-gun, and each is a rule rather than a
|
|
12
12
|
# heuristic:
|
|
13
13
|
#
|
|
14
|
-
# 1. A run WAITING FOR YOU is never reaped. An open PR, Phase
|
|
14
|
+
# 1. A run WAITING FOR YOU is never reaped. An open PR, Phase 4 or 7, or
|
|
15
15
|
# `status: awaiting_input` means the work landed and the pipeline is
|
|
16
16
|
# holding for an answer by design. That is finished work, not residue.
|
|
17
17
|
# 2. A path that is not strictly inside `<repo>/.worktrees/` is never removed.
|
|
@@ -49,7 +49,7 @@
|
|
|
49
49
|
#
|
|
50
50
|
# State is moved to `<run-dir>/artifacts/` before the worktree goes, so the log,
|
|
51
51
|
# the findings and the phase history survive the sweep. The same recovery
|
|
52
|
-
# contract Phase
|
|
52
|
+
# contract Phase 4 uses when it removes a finished worktree.
|
|
53
53
|
#
|
|
54
54
|
# SAFE BY DEFAULT: dry-run. Removes nothing until you pass --yes.
|
|
55
55
|
#
|
|
@@ -271,7 +271,7 @@ if [ "$STATE_ONLY" -eq 0 ] && [ -d "$REPOS" ]; then
|
|
|
271
271
|
ph=$(printf '%s' "$row" | cut -f5)
|
|
272
272
|
|
|
273
273
|
# Rule 1 again: landed work is not residue, whichever pass finds it. Order
|
|
274
|
-
# matters here - the phase-
|
|
274
|
+
# matters here - the phase-4/5 test is a proxy for "still waiting", and a
|
|
275
275
|
# run whose status already says it FINISHED is not waiting for anyone. A
|
|
276
276
|
# complete run sitting at phase 7 was read as waiting and never reaped,
|
|
277
277
|
# which is how three finished worktrees held 5.1 GB indefinitely.
|
|
@@ -107,7 +107,7 @@ NAME_ARGS=(
|
|
|
107
107
|
# `complaint-analysis-*` needs its own entry: find matches the basename against
|
|
108
108
|
# the whole glob, so `analysis-*` never matched it. Same for the other three,
|
|
109
109
|
# which were writing into $ROOT with nothing sweeping them (phase-0-init.md,
|
|
110
|
-
# phase-
|
|
110
|
+
# phase-4-commit.md, generate-issue.md).
|
|
111
111
|
#
|
|
112
112
|
# `*-wiki` is the one loose pattern here, and a suffix glob over $ROOT could
|
|
113
113
|
# name something a person created. It is narrowed twice: `-type d` above, and
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
# `.pipeline/` left behind by a finished run stops reading as residue.
|
|
22
22
|
#
|
|
23
23
|
# Registered, healthy worktrees are NEVER touched. Finishing a task removes its
|
|
24
|
-
# own worktree in Phase
|
|
24
|
+
# own worktree in Phase 4 (worktree-finalize, v14.1.0+); killing one is
|
|
25
25
|
# /multi-agent:kill. This sweep only reaps orphans neither of those left behind.
|
|
26
26
|
#
|
|
27
27
|
# SAFE BY DEFAULT: dry-run. Deletes nothing until you pass --yes.
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* @file gen-facts.mjs - the numbers the website is allowed to state, derived.
|
|
4
|
+
*
|
|
5
|
+
* The site carried its own copies: `ArchitectureContent.tsx` said "6 faz + 51
|
|
6
|
+
* komut" while the repo had six phases and sixty commands. Nobody noticed,
|
|
7
|
+
* because a number written into a React component is guarded by nothing. This
|
|
8
|
+
* is the same defect `smoke-phase-contract.sh` fixes inside the pipeline, and
|
|
9
|
+
* the fix is the same shape: one producer, everything else reads it.
|
|
10
|
+
*
|
|
11
|
+
* Source of truth per field:
|
|
12
|
+
* phases pipeline/schemas/phases.json - the phase contract itself
|
|
13
|
+
* commandCount a count of pipeline/commands/multi-agent/<name>/SKILL.md
|
|
14
|
+
* skillCount a count of pipeline/skills/shared/external/<name>/SKILL.md
|
|
15
|
+
* toolCount the toolkit's own tools/list response, when reachable
|
|
16
|
+
* version package.json
|
|
17
|
+
*
|
|
18
|
+
* Counts come from the filesystem at the moment of writing, never from a README
|
|
19
|
+
* or a previous version of this file.
|
|
20
|
+
*
|
|
21
|
+
* Usage:
|
|
22
|
+
* node gen-facts.mjs write docs/facts.json
|
|
23
|
+
* node gen-facts.mjs --check exit 1 if the file on disk is stale
|
|
24
|
+
* node gen-facts.mjs --stdout print, write nothing
|
|
25
|
+
*
|
|
26
|
+
* Exit codes:
|
|
27
|
+
* 0 - written, or up to date
|
|
28
|
+
* 1 - --check found drift
|
|
29
|
+
* 2 - a source could not be read (nothing is written)
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import { readFileSync, writeFileSync, readdirSync, existsSync } from "node:fs";
|
|
33
|
+
import { join, dirname } from "node:path";
|
|
34
|
+
import { fileURLToPath } from "node:url";
|
|
35
|
+
import { spawnSync } from "node:child_process";
|
|
36
|
+
|
|
37
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
38
|
+
const ROOT = join(HERE, "..", "..");
|
|
39
|
+
const OUT = join(ROOT, "docs", "facts.json");
|
|
40
|
+
|
|
41
|
+
function die(msg) {
|
|
42
|
+
console.error(`gen-facts: ${msg}`);
|
|
43
|
+
process.exit(2);
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function countSkillDirs(rel) {
|
|
47
|
+
const dir = join(ROOT, rel);
|
|
48
|
+
if (!existsSync(dir)) die(`missing directory: ${rel}`);
|
|
49
|
+
return readdirSync(dir, { withFileTypes: true }).filter(
|
|
50
|
+
(e) => e.isDirectory() && existsSync(join(dir, e.name, "SKILL.md")),
|
|
51
|
+
).length;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
let contract;
|
|
55
|
+
try {
|
|
56
|
+
contract = JSON.parse(readFileSync(join(ROOT, "pipeline", "schemas", "phases.json"), "utf8"));
|
|
57
|
+
} catch (err) {
|
|
58
|
+
die(`cannot read the phase contract - ${err.message}`);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
let pkg;
|
|
62
|
+
try {
|
|
63
|
+
pkg = JSON.parse(readFileSync(join(ROOT, "package.json"), "utf8"));
|
|
64
|
+
} catch (err) {
|
|
65
|
+
die(`cannot read package.json - ${err.message}`);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
// The toolkit is a separate repo and may not be checked out beside this one.
|
|
69
|
+
// Absent is reported as null rather than as a guess: a site that prints a stale
|
|
70
|
+
// tool count is the problem this file exists to remove, and a made-up one is
|
|
71
|
+
// worse than none.
|
|
72
|
+
let toolCount = null;
|
|
73
|
+
let toolkitVersion = null;
|
|
74
|
+
const toolkitDir = join(ROOT, "..", "multi-agent-toolkit-mcp");
|
|
75
|
+
if (existsSync(join(toolkitDir, "package.json"))) {
|
|
76
|
+
try {
|
|
77
|
+
toolkitVersion =
|
|
78
|
+
JSON.parse(readFileSync(join(toolkitDir, "package.json"), "utf8")).version ?? null;
|
|
79
|
+
} catch {
|
|
80
|
+
toolkitVersion = null;
|
|
81
|
+
}
|
|
82
|
+
// Asked of the server, not counted out of its source. Three tool families
|
|
83
|
+
// are defined in modules the main file only spreads in, so a regex over
|
|
84
|
+
// index.js returns a number that looks plausible and is wrong - the first
|
|
85
|
+
// version of this script said 2 when the answer was 99. tools/list is what
|
|
86
|
+
// a client actually receives, which is the number the site should print.
|
|
87
|
+
try {
|
|
88
|
+
const req =
|
|
89
|
+
[
|
|
90
|
+
JSON.stringify({
|
|
91
|
+
jsonrpc: "2.0",
|
|
92
|
+
id: 1,
|
|
93
|
+
method: "initialize",
|
|
94
|
+
params: {
|
|
95
|
+
protocolVersion: "2024-11-05",
|
|
96
|
+
capabilities: {},
|
|
97
|
+
clientInfo: { name: "gen-facts", version: "1" },
|
|
98
|
+
},
|
|
99
|
+
}),
|
|
100
|
+
JSON.stringify({ jsonrpc: "2.0", method: "notifications/initialized" }),
|
|
101
|
+
JSON.stringify({ jsonrpc: "2.0", id: 2, method: "tools/list", params: {} }),
|
|
102
|
+
].join("\n") + "\n";
|
|
103
|
+
const res = spawnSync(process.execPath, [join(toolkitDir, "index.js")], {
|
|
104
|
+
input: req,
|
|
105
|
+
encoding: "utf8",
|
|
106
|
+
timeout: 30000,
|
|
107
|
+
});
|
|
108
|
+
for (const line of (res.stdout || "").split("\n")) {
|
|
109
|
+
if (!line.trim()) continue;
|
|
110
|
+
let msg;
|
|
111
|
+
try {
|
|
112
|
+
msg = JSON.parse(line);
|
|
113
|
+
} catch {
|
|
114
|
+
continue;
|
|
115
|
+
}
|
|
116
|
+
if (msg.id === 2 && Array.isArray(msg.result?.tools)) toolCount = msg.result.tools.length;
|
|
117
|
+
}
|
|
118
|
+
} catch {
|
|
119
|
+
toolCount = null;
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const facts = {
|
|
124
|
+
$comment:
|
|
125
|
+
"Generated by pipeline/scripts/gen-facts.mjs. Do not hand-edit: the site reads this, and a number edited here instead of at its source is the drift this file removes.",
|
|
126
|
+
generatedAt: new Date().toISOString().slice(0, 10),
|
|
127
|
+
version: pkg.version,
|
|
128
|
+
phaseSchema: contract.phaseSchema,
|
|
129
|
+
phases: contract.phases.map((p) => ({ id: p.id, name: p.name })),
|
|
130
|
+
phaseCount: contract.phases.length,
|
|
131
|
+
modes: Object.fromEntries(Object.entries(contract.modes).map(([k, v]) => [k, v.phases])),
|
|
132
|
+
commandCount: countSkillDirs("pipeline/commands/multi-agent"),
|
|
133
|
+
skillCount: countSkillDirs("pipeline/skills/shared/external"),
|
|
134
|
+
toolCount,
|
|
135
|
+
toolkitVersion,
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
const body = `${JSON.stringify(facts, null, 2)}\n`;
|
|
139
|
+
|
|
140
|
+
if (process.argv.includes("--stdout")) {
|
|
141
|
+
process.stdout.write(body);
|
|
142
|
+
process.exit(0);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
if (process.argv.includes("--check")) {
|
|
146
|
+
if (!existsSync(OUT)) {
|
|
147
|
+
console.error("gen-facts: docs/facts.json does not exist - run without --check");
|
|
148
|
+
process.exit(1);
|
|
149
|
+
}
|
|
150
|
+
const onDisk = JSON.parse(readFileSync(OUT, "utf8"));
|
|
151
|
+
const fresh = JSON.parse(body);
|
|
152
|
+
// generatedAt is a timestamp, not a fact: comparing it would fail every day.
|
|
153
|
+
delete onDisk.generatedAt;
|
|
154
|
+
delete fresh.generatedAt;
|
|
155
|
+
if (JSON.stringify(onDisk) !== JSON.stringify(fresh)) {
|
|
156
|
+
console.error(
|
|
157
|
+
"gen-facts: docs/facts.json is stale - run: node pipeline/scripts/gen-facts.mjs",
|
|
158
|
+
);
|
|
159
|
+
for (const k of Object.keys(fresh)) {
|
|
160
|
+
if (JSON.stringify(onDisk[k]) !== JSON.stringify(fresh[k])) {
|
|
161
|
+
console.error(
|
|
162
|
+
` ${k}: on disk ${JSON.stringify(onDisk[k])}, derived ${JSON.stringify(fresh[k])}`,
|
|
163
|
+
);
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
process.exit(1);
|
|
167
|
+
}
|
|
168
|
+
console.log("gen-facts: docs/facts.json is current");
|
|
169
|
+
process.exit(0);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
writeFileSync(OUT, body);
|
|
173
|
+
console.log(
|
|
174
|
+
`gen-facts: wrote docs/facts.json (${facts.phaseCount} phases, ${facts.commandCount} commands, ${facts.skillCount} skills, ${facts.toolCount ?? "?"} tools)`,
|
|
175
|
+
);
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
+
import { readFileSync } from "node:fs";
|
|
2
3
|
/**
|
|
3
4
|
* gen-mode-dispatch.mjs - emit the canonical Phase Tracker Contract block
|
|
4
5
|
* for a given mode dispatch file (full / local / autopilot / analysis).
|
|
@@ -10,10 +11,10 @@
|
|
|
10
11
|
* call, anti-pattern warning, and ref link are identical.
|
|
11
12
|
*
|
|
12
13
|
* Usage:
|
|
13
|
-
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=full # 0..
|
|
14
|
-
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=local # 0
|
|
15
|
-
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=autopilot # 0
|
|
16
|
-
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=analysis # 0/1/
|
|
14
|
+
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=full # 0..5
|
|
15
|
+
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=local # 0..5 + local-mode caveat
|
|
16
|
+
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=autopilot # 0..5
|
|
17
|
+
* node pipeline/scripts/gen-mode-dispatch.mjs --mode=analysis # 0/1/3/4/5 (no Dev)
|
|
17
18
|
*
|
|
18
19
|
* v16.0.0 removed the four dev-* modes. Depth is no longer a command name: the
|
|
19
20
|
* Phase 0 Step 7.5 picker asks Full or Short and sets `state.onlyDevelop`. That
|
|
@@ -35,22 +36,20 @@ const args = Object.fromEntries(
|
|
|
35
36
|
|
|
36
37
|
const MODE = args.mode;
|
|
37
38
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
"
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
// v16.0.0 leaves exactly one command: `/multi-agent` itself.
|
|
53
|
-
const PHASE_NAMES_NO_TEST = PHASE_NAMES.filter((p) => p !== "5:Test");
|
|
39
|
+
// The phase list has one owner: schemas/phases.json. This file used to restate
|
|
40
|
+
// it, which is how eight independent copies came to exist. smoke-phase-contract.sh
|
|
41
|
+
// holds every remaining copy to the same file.
|
|
42
|
+
const CONTRACT = JSON.parse(
|
|
43
|
+
readFileSync(new URL("../schemas/phases.json", import.meta.url), "utf8"),
|
|
44
|
+
);
|
|
45
|
+
const PHASES = CONTRACT.phases;
|
|
46
|
+
const modePhases = (name) =>
|
|
47
|
+
CONTRACT.modes[name].phases.map((id) => {
|
|
48
|
+
const p = PHASES.find((x) => x.id === id);
|
|
49
|
+
return `${p.id}:${p.name}`;
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
const PHASE_NAMES = PHASES.map((p) => `${p.id}:${p.name}`);
|
|
54
53
|
|
|
55
54
|
/**
|
|
56
55
|
* Mode → phases + flag descriptor. The descriptor controls the autopilot /
|
|
@@ -59,21 +58,17 @@ const PHASE_NAMES_NO_TEST = PHASE_NAMES.filter((p) => p !== "5:Test");
|
|
|
59
58
|
* `smoke-mode-dispatch-drift.sh` can byte-equal-diff each on-disk section
|
|
60
59
|
* against its generator output.
|
|
61
60
|
*/
|
|
62
|
-
const MODES =
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
local: false,
|
|
74
|
-
autopilot: false,
|
|
75
|
-
},
|
|
76
|
-
};
|
|
61
|
+
const MODES = Object.fromEntries(
|
|
62
|
+
Object.entries(CONTRACT.modes).map(([name, m]) => [
|
|
63
|
+
name,
|
|
64
|
+
{
|
|
65
|
+
phases: modePhases(name),
|
|
66
|
+
local: m.local,
|
|
67
|
+
autopilot: m.autopilot,
|
|
68
|
+
...(m.depth ? { depth: true } : {}),
|
|
69
|
+
},
|
|
70
|
+
]),
|
|
71
|
+
);
|
|
77
72
|
|
|
78
73
|
const spec = MODE ? MODES[MODE] : null;
|
|
79
74
|
if (!spec) {
|
|
@@ -101,7 +96,7 @@ const modeLabel = MODE_LABELS[MODE] ?? `\`${MODE}\``;
|
|
|
101
96
|
const skipNote =
|
|
102
97
|
skippedIds.length > 0
|
|
103
98
|
? `${modeLabel} mode does NOT TaskCreate phases ${skippedIds.join("/")} - those are not part of the ${modeLabel} phase set (\`${spec.phases.join(" ")}\`). Only register tiles for the active set.`
|
|
104
|
-
: `${modeLabel} mode TaskCreates all
|
|
99
|
+
: `${modeLabel} mode TaskCreates all ${PHASES.length} phases (no phase is skipped).`;
|
|
105
100
|
|
|
106
101
|
const orderingNote = `**All TaskCreate calls in a batch fire in strict phase-number order BEFORE any TaskUpdate is applied.** For ${modeLabel} that means: ${spec.depth ? `Phase 0 at Step -1, then the rest in ascending order at Step 7.5 (${phaseSequence} minus whatever the depth answer drops)` : phaseSequence}. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks. Full ordering contract in \`$HOME/.claude/multi-agent-refs/tracker-contract.md\` section "TaskCreate ordering (strict)".`;
|
|
107
102
|
|
|
@@ -119,7 +114,7 @@ const banner = spec.autopilot
|
|
|
119
114
|
// tiles beside the question that decides whether two of them run is the failure this
|
|
120
115
|
// split exists to remove.
|
|
121
116
|
const fullSet = spec.phases.filter((p) => p !== "0:Init");
|
|
122
|
-
const shortSet = fullSet.filter((p) => Number(p.split(":")[0]) >=
|
|
117
|
+
const shortSet = fullSet.filter((p) => Number(p.split(":")[0]) >= 2);
|
|
123
118
|
const loopFor = (list) =>
|
|
124
119
|
`for p in ${list.map((x) => `"${x}"`).join(" ")}; do\n bash $HOME/.claude/scripts/phase-tracker.sh add "\${p%%:*}" "\${p#*:}"\ndone`;
|
|
125
120
|
|