@mmerterden/multi-agent-pipeline 18.0.0 → 19.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +287 -0
- package/README.md +36 -20
- package/README.tr.md +14 -16
- package/docs/adr/0002-instruction-driven-flag.md +1 -0
- package/docs/adr/0005-lazy-phase-docs.md +11 -1
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0010-own-code-graph.md +1 -0
- package/docs/adr/0014-six-phase-consolidation.md +134 -0
- package/docs/adr/README.md +2 -1
- package/docs/architecture.md +37 -38
- package/docs/best-practices.md +1 -1
- package/docs/ecosystem.md +46 -27
- package/docs/engineering.md +1 -1
- package/docs/facts.json +61 -0
- package/docs/features.md +55 -54
- package/docs/performance.md +5 -5
- package/docs/recovery-guide.md +17 -17
- package/docs/token-budget-history.md +3 -1
- package/index.js +2 -2
- package/install/_codex-agents.mjs +1 -1
- package/install/templates/claude-hooks.json +1 -1
- package/install/templates/codex-instructions.md +1 -1
- package/install/templates/copilot-instructions.md +28 -28
- package/manifest.json +234 -216
- package/package.json +2 -2
- package/pipeline/agents/dev-critic.md +7 -7
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/analysis/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
- package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
- package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
- package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
- package/pipeline/commands/multi-agent/review/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
- package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
- package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/status/SKILL.md +5 -5
- package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
- package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
- package/pipeline/lib/credential-inventory.sh +1 -1
- package/pipeline/lib/fetch-fortify.sh +1 -1
- package/pipeline/lib/model-dispatch.sh +140 -0
- package/pipeline/lib/model-rung.sh +142 -0
- package/pipeline/lib/outbound-gate.mjs +14 -0
- package/pipeline/lib/phase-schema.mjs +88 -0
- package/pipeline/lib/plan-todos.sh +5 -5
- package/pipeline/lib/route-state.sh +161 -0
- package/pipeline/lib/run-paths.sh +2 -2
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +6 -6
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/analysis/evidence.md +2 -11
- package/pipeline/multi-agent-refs/analysis/intake.md +7 -7
- package/pipeline/multi-agent-refs/analysis/locked.md +48 -22
- package/pipeline/multi-agent-refs/analysis/redesign.md +1 -1
- package/pipeline/multi-agent-refs/analysis/render.md +10 -10
- package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
- package/pipeline/multi-agent-refs/analysis/review.md +2 -2
- package/pipeline/multi-agent-refs/analysis/synthesis.md +13 -7
- package/pipeline/multi-agent-refs/analysis-template-corporate.md +9 -9
- package/pipeline/multi-agent-refs/analysis-template.md +19 -19
- package/pipeline/multi-agent-refs/android-guide.md +1 -1
- package/pipeline/multi-agent-refs/audit-guide.md +13 -13
- package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
- package/pipeline/multi-agent-refs/channels/jira.md +3 -3
- package/pipeline/multi-agent-refs/channels/pr.md +4 -4
- package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +8 -8
- package/pipeline/multi-agent-refs/conventions-defaults.md +2 -2
- package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
- package/pipeline/multi-agent-refs/features/analysis-jira.md +1 -1
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +4 -4
- package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
- package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
- package/pipeline/multi-agent-refs/features/doctor.md +3 -3
- package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
- package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
- package/pipeline/multi-agent-refs/features/model-fallback.md +41 -5
- package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
- package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
- package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +2 -2
- package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
- package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
- package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
- package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
- package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
- package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
- package/pipeline/multi-agent-refs/knowledge.md +11 -11
- package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
- package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
- package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
- package/pipeline/multi-agent-refs/phases/modes.md +30 -30
- package/pipeline/multi-agent-refs/phases/operations.md +8 -8
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +24 -24
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
- package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
- package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
- package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
- package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
- package/pipeline/multi-agent-refs/phases.md +44 -48
- package/pipeline/multi-agent-refs/picker-contract.md +1 -1
- package/pipeline/multi-agent-refs/progress-contract.md +6 -6
- package/pipeline/multi-agent-refs/readiness-review.md +1 -1
- package/pipeline/multi-agent-refs/rules.md +7 -7
- package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
- package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
- package/pipeline/preferences-template.json +9 -1
- package/pipeline/rules/figma-pipeline.md +8 -8
- package/pipeline/rules/outside-the-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +50 -50
- package/pipeline/schemas/analysis-output.schema.json +3 -3
- package/pipeline/schemas/analysis-spec.schema.json +2 -2
- package/pipeline/schemas/autopilot-config.schema.json +1 -1
- package/pipeline/schemas/code-graph.schema.json +1 -1
- package/pipeline/schemas/criteria-manifest.schema.json +1 -1
- package/pipeline/schemas/dev-critic-output.schema.json +1 -1
- package/pipeline/schemas/diff-risk.schema.json +1 -1
- package/pipeline/schemas/figma-project-config.schema.json +1 -1
- package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
- package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
- package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
- package/pipeline/schemas/phases.json +105 -0
- package/pipeline/schemas/plan-todos.schema.json +5 -5
- package/pipeline/schemas/planning-output.schema.json +1 -1
- package/pipeline/schemas/prefs.schema.json +102 -58
- package/pipeline/schemas/reviewer-output.schema.json +3 -3
- package/pipeline/schemas/route-config.schema.json +74 -0
- package/pipeline/schemas/scope-check.schema.json +1 -1
- package/pipeline/schemas/secret-patterns.json +124 -0
- package/pipeline/schemas/test-gap.schema.json +1 -1
- package/pipeline/schemas/token-budget.json +12 -18
- package/pipeline/schemas/triage-output.schema.json +6 -6
- package/pipeline/scripts/README.md +3 -3
- package/pipeline/scripts/_code-graph.mjs +2 -2
- package/pipeline/scripts/_run-paths.mjs +2 -2
- package/pipeline/scripts/_smoke-root.sh +1 -1
- package/pipeline/scripts/aggregate-metrics.mjs +1 -1
- package/pipeline/scripts/build-references.mjs +2 -2
- package/pipeline/scripts/bulk-read.sh +10 -1
- package/pipeline/scripts/capture-flush.sh +8 -8
- package/pipeline/scripts/capture-resume.sh +3 -3
- package/pipeline/scripts/classify-plan-safety.mjs +1 -1
- package/pipeline/scripts/cost-table.json +8 -1
- package/pipeline/scripts/diff-explain.mjs +1 -1
- package/pipeline/scripts/doctor.mjs +3 -3
- package/pipeline/scripts/gc-abandoned.sh +3 -3
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +1 -1
- package/pipeline/scripts/gen-facts.mjs +280 -0
- package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
- package/pipeline/scripts/gen-ref-toc.mjs +1 -1
- package/pipeline/scripts/graph-report.mjs +1 -1
- package/pipeline/scripts/jira-attach.sh +1 -1
- package/pipeline/scripts/learn-from-transcripts.mjs +1 -1
- package/pipeline/scripts/learning-curve.mjs +2 -2
- package/pipeline/scripts/log-metric.sh +17 -4
- package/pipeline/scripts/memory-save.sh +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +22 -5
- package/pipeline/scripts/phase-banner.sh +20 -20
- package/pipeline/scripts/phase-tracker.sh +12 -12
- package/pipeline/scripts/plan-coverage-gate.mjs +2 -2
- package/pipeline/scripts/pre-commit-check.sh +30 -1
- package/pipeline/scripts/render-agent-log-cost.sh +1 -1
- package/pipeline/scripts/render-work-summary.sh +3 -3
- package/pipeline/scripts/review-file-filter.mjs +1 -1
- package/pipeline/scripts/run-aggregator.mjs +13 -6
- package/pipeline/scripts/run-metrics.mjs +1 -1
- package/pipeline/scripts/runs-index.mjs +11 -1
- package/pipeline/scripts/scan-skills.sh +26 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
- package/pipeline/scripts/smoke-schema-validation.sh +26 -7
- package/pipeline/scripts/token-budget-report.mjs +13 -2
- package/pipeline/scripts/triage-memory.mjs +2 -2
- package/pipeline/scripts/validate-analysis-doc.mjs +274 -43
- package/pipeline/scripts/validate-planning.mjs +1 -1
- package/pipeline/scripts/validate-reviewer.mjs +1 -1
- package/pipeline/scripts/validate-state.mjs +45 -5
- package/pipeline/scripts/validate-triage.mjs +3 -3
- package/pipeline/scripts/verify-citations.mjs +1 -1
- package/pipeline/scripts/worktree-finalize.sh +5 -5
- package/pipeline/scripts/write-state.mjs +32 -0
- package/pipeline/skills/.skill-manifest.json +38 -22
- package/pipeline/skills/.skills-index.json +49 -5
- package/pipeline/skills/shared/README.md +10 -6
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +8 -8
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +81 -82
- package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
- package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
- package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
- package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
- package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
- package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
- package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +1 -1
- package/pipeline/skills/shared/external/signal-community/SKILL.md +8 -1
- package/pipeline/skills/skills-index.md +8 -4
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/route-config.schema.json",
|
|
4
|
+
"title": "Model routing configuration",
|
|
5
|
+
"description": "Policy-driven model selection, OFF by default. Lives under prefs.global.modelRouting. Two layers, and the difference is the whole safety argument. Layer 1 chooses a rung INSIDE the Anthropic ladder and writes to the existing PHASE_MODEL_OVERRIDE seam - no new network path, no new account. Layer 2 sends a call to a non-Anthropic provider, and only at the call sites the pipeline makes itself (bulk-read.sh, research_ask), never at a call site the host owns. `scope` has no `host-session` member and that absence is enforced here rather than written in prose: rewriting the host's base URL would route the user's ENTIRE session, including work that has nothing to do with this pipeline, through a third layer - breaking the subscription's auth model and silently changing which model answers. Shipping this disabled is the same reasoning as modelFallback.fableEnabled: a cost control that is on by default is not a control.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": ["enabled"],
|
|
9
|
+
"properties": {
|
|
10
|
+
"enabled": {
|
|
11
|
+
"type": "boolean",
|
|
12
|
+
"default": false,
|
|
13
|
+
"description": "Master switch, ships false. While false not one line of dispatch behaviour changes: rules are read by route-status and by nothing else. route-off sets this to false and KEEPS the rules, so turning it back on does not re-ask for a configuration the user already gave."
|
|
14
|
+
},
|
|
15
|
+
"strategy": {
|
|
16
|
+
"type": "string",
|
|
17
|
+
"enum": ["manual", "task-fit", "cost-ceiling"],
|
|
18
|
+
"default": "manual",
|
|
19
|
+
"description": "manual: only the explicit rules[] apply, nothing is inferred. task-fit: a rule may match on taskKind and the router picks the cheapest rung that clears it. cost-ceiling: rungs downgrade as the run approaches budgetCeilingUsd."
|
|
20
|
+
},
|
|
21
|
+
"scope": {
|
|
22
|
+
"type": "array",
|
|
23
|
+
"default": ["subagent"],
|
|
24
|
+
"uniqueItems": true,
|
|
25
|
+
"description": "Where routing is allowed to act. Every member names a call site the pipeline itself owns. There is deliberately no `host-session` member - see the title description.",
|
|
26
|
+
"items": {
|
|
27
|
+
"type": "string",
|
|
28
|
+
"enum": ["subagent", "bulk-read", "research"]
|
|
29
|
+
}
|
|
30
|
+
},
|
|
31
|
+
"rules": {
|
|
32
|
+
"type": "array",
|
|
33
|
+
"default": [],
|
|
34
|
+
"description": "Ordered; first match wins. An empty list with enabled:true is valid and means 'routing is armed but nothing matches yet' - which route-status reports as such rather than as an error.",
|
|
35
|
+
"items": {
|
|
36
|
+
"type": "object",
|
|
37
|
+
"additionalProperties": false,
|
|
38
|
+
"required": ["when", "prefer"],
|
|
39
|
+
"properties": {
|
|
40
|
+
"when": {
|
|
41
|
+
"type": "object",
|
|
42
|
+
"additionalProperties": false,
|
|
43
|
+
"minProperties": 1,
|
|
44
|
+
"properties": {
|
|
45
|
+
"persona": { "type": "string" },
|
|
46
|
+
"phase": { "type": "integer", "minimum": 0, "maximum": 5 },
|
|
47
|
+
"taskKind": {
|
|
48
|
+
"type": "string",
|
|
49
|
+
"enum": ["bugfix", "feature", "refactor", "chore", "component"]
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
},
|
|
53
|
+
"prefer": {
|
|
54
|
+
"type": "array",
|
|
55
|
+
"minItems": 1,
|
|
56
|
+
"description": "Rungs in descending preference. Rung NAMES are the contract; model ids are not - they live in cost-table.json and change without a config edit.",
|
|
57
|
+
"items": { "type": "string" }
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
"budgetCeilingUsd": {
|
|
63
|
+
"type": ["number", "null"],
|
|
64
|
+
"default": null,
|
|
65
|
+
"minimum": 0,
|
|
66
|
+
"description": "Only consulted by the cost-ceiling strategy. null means no ceiling, which is not the same as a ceiling of 0."
|
|
67
|
+
},
|
|
68
|
+
"recordDecisions": {
|
|
69
|
+
"type": "boolean",
|
|
70
|
+
"default": true,
|
|
71
|
+
"description": "Write every routing decision to the cost ledger: which rule matched, which rung it chose, and why. On by default because a router whose choices are not recorded cannot be audited after the fact - and the question that always arrives later is 'why did this run cost that'."
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
}
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/scope-check.schema.json",
|
|
4
4
|
"version": "1.0.0",
|
|
5
5
|
"title": "Multi-Agent Pipeline - Phase 3 scope self-check",
|
|
6
|
-
"description": "Written by Phase
|
|
6
|
+
"description": "Written by Phase 2 Step 3.7 to $WORKTREE/.pipeline/scope-check.json before the Phase 4 handoff: one stated reason per file in the diff, the changes deliberately not made, and the code-simplifier rationales. scope-check-gate.mjs compares files[] with the real diff; Phase 4 injects the record as <scope-self-check>; Phase 4 builds the PR Changes bullets and the follow-up list from it.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"required": ["version", "taskId", "files"],
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$comment": "Provider shapes for smoke-secret-parity.sh, which proves both secret surfaces block the same set. Each row is SPLIT into a prefix, a filler character and a length, and the gate joins them at run time. No complete credential-shaped string is stored here, so the scanners this file exists to test do not block the file itself - which is exactly what happened when the samples were stored whole and the commit hook refused them. Adding a provider means adding a row here first; the gate then fails until both surfaces catch it.",
|
|
3
|
+
"providers": [
|
|
4
|
+
{
|
|
5
|
+
"id": "github-pat",
|
|
6
|
+
"prefix": "ghp_",
|
|
7
|
+
"fill": "A",
|
|
8
|
+
"len": 36,
|
|
9
|
+
"suffix": ""
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"id": "github-fine-grained",
|
|
13
|
+
"prefix": "github_pat_",
|
|
14
|
+
"fill": "A",
|
|
15
|
+
"len": 66,
|
|
16
|
+
"suffix": ""
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"id": "slack",
|
|
20
|
+
"prefix": "xoxb-",
|
|
21
|
+
"fill": "A",
|
|
22
|
+
"len": 16,
|
|
23
|
+
"suffix": ""
|
|
24
|
+
},
|
|
25
|
+
{
|
|
26
|
+
"id": "aws",
|
|
27
|
+
"prefix": "AKIA",
|
|
28
|
+
"fill": "A",
|
|
29
|
+
"len": 16,
|
|
30
|
+
"suffix": ""
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"id": "google",
|
|
34
|
+
"prefix": "AIza",
|
|
35
|
+
"fill": "A",
|
|
36
|
+
"len": 35,
|
|
37
|
+
"suffix": ""
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"id": "npm",
|
|
41
|
+
"prefix": "npm_",
|
|
42
|
+
"fill": "A",
|
|
43
|
+
"len": 36,
|
|
44
|
+
"suffix": ""
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"id": "gitlab",
|
|
48
|
+
"prefix": "glpat-",
|
|
49
|
+
"fill": "A",
|
|
50
|
+
"len": 20,
|
|
51
|
+
"suffix": ""
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "stripe",
|
|
55
|
+
"prefix": "sk_live_",
|
|
56
|
+
"fill": "A",
|
|
57
|
+
"len": 20,
|
|
58
|
+
"suffix": ""
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
"id": "anthropic",
|
|
62
|
+
"prefix": "sk-ant-",
|
|
63
|
+
"fill": "A",
|
|
64
|
+
"len": 36,
|
|
65
|
+
"suffix": ""
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
"id": "openai",
|
|
69
|
+
"prefix": "sk-proj-",
|
|
70
|
+
"fill": "A",
|
|
71
|
+
"len": 36,
|
|
72
|
+
"suffix": ""
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
"id": "perplexity",
|
|
76
|
+
"prefix": "pplx-",
|
|
77
|
+
"fill": "A",
|
|
78
|
+
"len": 36,
|
|
79
|
+
"suffix": ""
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
"id": "huggingface",
|
|
83
|
+
"prefix": "hf_",
|
|
84
|
+
"fill": "A",
|
|
85
|
+
"len": 32,
|
|
86
|
+
"suffix": ""
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"id": "figma-pat",
|
|
90
|
+
"prefix": "figd_",
|
|
91
|
+
"fill": "A",
|
|
92
|
+
"len": 24,
|
|
93
|
+
"suffix": ""
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
"id": "figma-mcp",
|
|
97
|
+
"prefix": "figu_",
|
|
98
|
+
"fill": "A",
|
|
99
|
+
"len": 24,
|
|
100
|
+
"suffix": ""
|
|
101
|
+
},
|
|
102
|
+
{
|
|
103
|
+
"id": "service-account",
|
|
104
|
+
"prefix": "\"type\": \"service_",
|
|
105
|
+
"fill": "",
|
|
106
|
+
"len": 0,
|
|
107
|
+
"suffix": "account\""
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
"id": "url-with-credentials",
|
|
111
|
+
"prefix": "https://user:",
|
|
112
|
+
"fill": "s",
|
|
113
|
+
"len": 12,
|
|
114
|
+
"suffix": "@example.com/x.git"
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
"id": "bearer-header",
|
|
118
|
+
"prefix": "Authorization: Bearer ",
|
|
119
|
+
"fill": "A",
|
|
120
|
+
"len": 24,
|
|
121
|
+
"suffix": ""
|
|
122
|
+
}
|
|
123
|
+
]
|
|
124
|
+
}
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/test-gap.schema.json",
|
|
4
4
|
"version": "1.0.0",
|
|
5
|
-
"title": "Multi-Agent Pipeline - Phase
|
|
5
|
+
"title": "Multi-Agent Pipeline - Phase 3 test gap report",
|
|
6
6
|
"description": "Output of test-gap-scan.mjs. Lists symbols added/changed in the diff that have no paired test file or no matching test method. Advisory in default mode; opt-in blocking via prefs.testGap.blockingThreshold.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
@@ -1,32 +1,26 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/token-budget.json",
|
|
4
|
-
"description": "Per-phase token ceilings for the lazy-loaded pipeline docs, enforced by smoke-token-budget.sh. Only the ACTIVE phase is loaded at run time and nothing truncates a phase document, so these numbers govern what may be written, not what a run receives. Two rules, and the split is the point. `max_tokens` and `total_max_tokens` are committed constants a human owns: computing them would end the gate, because a ceiling that is always current x k can never fail. The warn tier is NOT stored - the gate derives it per phase from that phase's own git history (median + 3*MAD of its historical per-commit deltas, which is robust to the one 980-token release that makes sigma meaningless on phase-4), because a soft line maintained by hand rots, and this one rotted twice: it was reset at v13.6.0 when five lines had gone permanently amber, and six of eight were amber again by v17.1.0. A ceiling more than 25% above the measurement is stale and fails, which is the 'only ratchets down' rule the SKILL.md grace list always had and this budget never did. Change history: docs/token-budget-history.md.",
|
|
4
|
+
"description": "Per-phase token ceilings for the lazy-loaded pipeline docs, enforced by smoke-token-budget.sh. Only the ACTIVE phase is loaded at run time and nothing truncates a phase document, so these numbers govern what may be written, not what a run receives. Two rules, and the split is the point. `max_tokens` and `total_max_tokens` are committed constants a human owns: computing them would end the gate, because a ceiling that is always current x k can never fail. The warn tier is NOT stored - the gate derives it per phase from that phase's own git history (median + 3*MAD of its historical per-commit deltas, which is robust to the one 980-token release that makes sigma meaningless on phase-4), because a soft line maintained by hand rots, and this one rotted twice: it was reset at v13.6.0 when five lines had gone permanently amber, and six of eight were amber again by v17.1.0. A ceiling more than 25% above the measurement is stale and fails, which is the 'only ratchets down' rule the SKILL.md grace list always had and this budget never did. Ceilings are DERIVED from schemas/phases.json and re-measured after the v19.0.0 six-phase merge, not summed from the eight old ones: the two-sided rule fails a ceiling more than 25% above the measurement, so a sum would have shipped stale by construction. Change history: docs/token-budget-history.md.",
|
|
5
5
|
"phases": {
|
|
6
6
|
"phase-0-init": {
|
|
7
7
|
"max_tokens": 13400
|
|
8
8
|
},
|
|
9
|
-
"phase-1-
|
|
10
|
-
"max_tokens":
|
|
9
|
+
"phase-1-plan": {
|
|
10
|
+
"max_tokens": 10000
|
|
11
11
|
},
|
|
12
|
-
"phase-2-
|
|
13
|
-
"max_tokens":
|
|
14
|
-
},
|
|
15
|
-
"phase-3-dev": {
|
|
16
|
-
"max_tokens": 9450
|
|
12
|
+
"phase-2-dev": {
|
|
13
|
+
"max_tokens": 10650
|
|
17
14
|
},
|
|
18
|
-
"phase-
|
|
19
|
-
"max_tokens":
|
|
15
|
+
"phase-3-review": {
|
|
16
|
+
"max_tokens": 17200
|
|
20
17
|
},
|
|
21
|
-
"phase-
|
|
22
|
-
"max_tokens":
|
|
23
|
-
},
|
|
24
|
-
"phase-6-commit": {
|
|
25
|
-
"max_tokens": 6550
|
|
18
|
+
"phase-4-commit": {
|
|
19
|
+
"max_tokens": 6500
|
|
26
20
|
},
|
|
27
|
-
"phase-
|
|
28
|
-
"max_tokens":
|
|
21
|
+
"phase-5-report": {
|
|
22
|
+
"max_tokens": 5550
|
|
29
23
|
}
|
|
30
24
|
},
|
|
31
|
-
"total_max_tokens":
|
|
25
|
+
"total_max_tokens": 63300
|
|
32
26
|
}
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/triage-output.schema.json",
|
|
4
4
|
"version": "3.4.0",
|
|
5
|
-
"title": "Multi-Agent Pipeline - Phase
|
|
6
|
-
"description": "Contract for the Opus triage agent's JSON output in Phase 4 Step 3. Triage consumes merged reviewer findings and splits them into accepted/deferred/rejected. Only `accepted` blocking/important items trigger Phase 3 rework. v3.1.0 adds the optional `consensus` block so triage can surface reviewer-agreement risk (false consensus among same-base-model reviewers) instead of silently merging. v3.2.0 adds the optional per-finding `verification` block written by Phase
|
|
5
|
+
"title": "Multi-Agent Pipeline - Phase 3 triage output",
|
|
6
|
+
"description": "Contract for the Opus triage agent's JSON output in Phase 4 Step 3. Triage consumes merged reviewer findings and splits them into accepted/deferred/rejected. Only `accepted` blocking/important items trigger Phase 3 rework. v3.1.0 adds the optional `consensus` block so triage can surface reviewer-agreement risk (false consensus among same-base-model reviewers) instead of silently merging. v3.2.0 adds the optional per-finding `verification` block written by Phase 3 Step 3.7 (verify-by-test): the empirical repro-test outcome for accepted blocking findings. v3.3.0 carries ruleId + criteriaSource through triage so a finding that cites a stable rule ID keeps that citation into Phase 4 and Phase 5, and the lesson loop can key durable learnings by rule. v3.4.0 adds the optional per-finding fingerprint (finding-fingerprint.mjs) so review-delta.mjs can tell still-present, resolved and new findings apart between rounds.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"required": ["accepted", "deferred", "rejected", "approved"],
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
},
|
|
18
18
|
"deferred": {
|
|
19
19
|
"type": "array",
|
|
20
|
-
"description": "Findings that are real but out of current task scope. Surfaced in Phase
|
|
20
|
+
"description": "Findings that are real but out of current task scope. Surfaced in Phase 5 report; not actioned.",
|
|
21
21
|
"items": {
|
|
22
22
|
"type": "object",
|
|
23
23
|
"additionalProperties": false,
|
|
@@ -55,7 +55,7 @@
|
|
|
55
55
|
},
|
|
56
56
|
"approved": {
|
|
57
57
|
"type": "boolean",
|
|
58
|
-
"description": "True if no accepted BLOCKING items remain. The pipeline uses this to gate Phase 5 / Phase
|
|
58
|
+
"description": "True if no accepted BLOCKING items remain. The pipeline uses this to gate Phase 5 / Phase 4 entry."
|
|
59
59
|
},
|
|
60
60
|
"consensus": {
|
|
61
61
|
"$ref": "#/$defs/consensus"
|
|
@@ -111,7 +111,7 @@
|
|
|
111
111
|
},
|
|
112
112
|
"disagreements": {
|
|
113
113
|
"type": "array",
|
|
114
|
-
"description": "Findings where reviewers split (existence or severity). Surfaced verbatim in the Phase
|
|
114
|
+
"description": "Findings where reviewers split (existence or severity). Surfaced verbatim in the Phase 5 report and at the Step 4 user checkpoint.",
|
|
115
115
|
"items": {
|
|
116
116
|
"type": "object",
|
|
117
117
|
"additionalProperties": false,
|
|
@@ -142,7 +142,7 @@
|
|
|
142
142
|
"verification": {
|
|
143
143
|
"type": "object",
|
|
144
144
|
"additionalProperties": false,
|
|
145
|
-
"description": "v3.2.0 verify-by-test outcome (Phase
|
|
145
|
+
"description": "v3.2.0 verify-by-test outcome (Phase 3 Step 3.7, opt-in via prefs.global.verifyByTest). confirmed = repro test failed as the finding predicts (finding stands, test kept as the Phase 3 RED test); not-reproduced = repro test passed under evidence-gate (finding downgraded to deferred); inconclusive = compile error / timeout / not unit-testable (judgment verdict stands).",
|
|
146
146
|
"required": ["result"],
|
|
147
147
|
"properties": {
|
|
148
148
|
"result": {
|
|
@@ -19,10 +19,10 @@ Validate contracts. Each emits `══ <name> smoke: N passed, M failed ══`
|
|
|
19
19
|
|
|
20
20
|
### Phase contracts
|
|
21
21
|
- `smoke-phase-0-multi-repo.sh` - Phase 0 multi-repo mode fetch + worktree atomicity
|
|
22
|
-
- `smoke-phase-6-multi.sh` - Phase
|
|
22
|
+
- `smoke-phase-6-multi.sh` - Phase 4 multi-repo commit/PR cross-linking
|
|
23
23
|
- `smoke-phase-banner.sh` + `smoke-phase-tracker.sh` - Phase UI output contracts
|
|
24
|
-
- `smoke-phase4-triage.sh` - Phase
|
|
25
|
-
- `smoke-verify-by-test.sh` - Phase
|
|
24
|
+
- `smoke-phase4-triage.sh` - Phase 3 reviewer → triage flow
|
|
25
|
+
- `smoke-verify-by-test.sh` - Phase 3 Step 3.7 verify-by-test contract (v10.8.0)
|
|
26
26
|
- `smoke-handoff-contract.sh` - phase-boundary structured handoff + handoff-first resume (v10.8.0)
|
|
27
27
|
- `smoke-update-check.sh` - Phase 0 Step 0.6 update-check + required-floor contract (v10.9.0, floor v15.14.0)
|
|
28
28
|
- `smoke-context-links.sh` - context-link-extractor classification contract, all types (v15.14.0)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @file _code-graph.mjs - deterministic, LLM-free code graph for Phase 1 and Phase
|
|
2
|
+
* @file _code-graph.mjs - deterministic, LLM-free code graph for Phase 1 and Phase 5.
|
|
3
3
|
*
|
|
4
|
-
* Phase 1 narrows its Explore fan-out with this graph, and Phase
|
|
4
|
+
* Phase 1 narrows its Explore fan-out with this graph, and Phase 5 refreshes it
|
|
5
5
|
* instead of hand-writing architecture.md from a single task's window. The
|
|
6
6
|
* design follows graphify (Graphify-Labs/graphify): AST-quality extraction with
|
|
7
7
|
* zero LLM cost, one graph file, an incremental manifest gate, god-node hubs,
|
|
@@ -76,7 +76,7 @@ export function canonicalRunDir(taskId, project) {
|
|
|
76
76
|
}
|
|
77
77
|
|
|
78
78
|
/**
|
|
79
|
-
* Phase
|
|
79
|
+
* Phase 4 removes the worktree and salvages the run's files into an
|
|
80
80
|
* `artifacts/` subdirectory of the same run directory. A finished run therefore
|
|
81
81
|
* keeps its state one level deeper, and a reader that only looks at the top
|
|
82
82
|
* level reports a shipped task as having no state at all.
|
|
@@ -232,7 +232,7 @@ export function resolveRunFile(taskId, filename, project) {
|
|
|
232
232
|
for (const v of taskIdVariants(taskId)) {
|
|
233
233
|
const dir = resolveRunDir(v, project);
|
|
234
234
|
if (!dir) continue;
|
|
235
|
-
// Top level first, then the salvaged copy Phase
|
|
235
|
+
// Top level first, then the salvaged copy Phase 4 leaves behind.
|
|
236
236
|
for (const base of [dir, join(dir, ARTIFACTS_SUBDIR)]) {
|
|
237
237
|
const p = join(base, filename);
|
|
238
238
|
if (existsSync(p)) return p;
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
#
|
|
21
21
|
# Usage (from any script in pipeline/scripts/):
|
|
22
22
|
# . "$(dirname "${BASH_SOURCE[0]}")/_smoke-root.sh"
|
|
23
|
-
# grep -q needle "$MA_REFS/phases/phase-
|
|
23
|
+
# grep -q needle "$MA_REFS/phases/phase-3-review.md"
|
|
24
24
|
#
|
|
25
25
|
# Exports:
|
|
26
26
|
# MA_LAYOUT "repo" | "install"
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
// by --since=<ISO date> / --task-id=<id> / --phase=<N>.
|
|
7
7
|
//
|
|
8
8
|
// Output is plain-text by default; pass --json for machine-readable output
|
|
9
|
-
// (used by Phase
|
|
9
|
+
// (used by Phase 5 to embed a metrics block in the run report).
|
|
10
10
|
//
|
|
11
11
|
// Aggregations:
|
|
12
12
|
// - Tasks completed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
// build-references.mjs - deterministic Section 21 References table for an
|
|
3
|
-
// /multi-agent:analysis document (Locked
|
|
3
|
+
// /multi-agent:analysis document (Locked 33).
|
|
4
4
|
//
|
|
5
5
|
// The references table used to be prose the model filled in. That lists what an
|
|
6
6
|
// author remembers consulting, which is a different set from what the run
|
|
@@ -364,7 +364,7 @@ function main() {
|
|
|
364
364
|
const problems = check(spec, docPath, lang);
|
|
365
365
|
if (problems.length) {
|
|
366
366
|
for (const p of problems) process.stderr.write(`${p}\n`);
|
|
367
|
-
process.stderr.write(`\n${problems.length} references coverage failure(s)\n`);
|
|
367
|
+
process.stderr.write(`\n${problems.length} references coverage failure(s) (Locked 33)\n`);
|
|
368
368
|
process.exitCode = 1;
|
|
369
369
|
return;
|
|
370
370
|
}
|
|
@@ -43,7 +43,7 @@ while [ $# -gt 0 ]; do
|
|
|
43
43
|
--file) FILE="${2:?--file needs a value}"; shift 2 ;;
|
|
44
44
|
--question) QUESTION="${2:?--question needs a value}"; shift 2 ;;
|
|
45
45
|
--phase) PHASE="${2:?--phase needs a value}"; shift 2 ;;
|
|
46
|
-
--model) MODEL="${2:?--model needs a value}"; shift 2 ;;
|
|
46
|
+
--model) MODEL="${2:?--model needs a value}"; MODEL_EXPLICIT=1; shift 2 ;;
|
|
47
47
|
--timeout) TIMEOUT="${2:?--timeout needs a value}"; shift 2 ;;
|
|
48
48
|
--json) AS_JSON=1; shift ;;
|
|
49
49
|
-h|--help) sed -n '2,30p' "$0"; exit 0 ;;
|
|
@@ -90,6 +90,15 @@ if command -v jq >/dev/null 2>&1; then
|
|
|
90
90
|
done
|
|
91
91
|
fi
|
|
92
92
|
[ -n "$MODEL" ] || MODEL="${PREF_MODEL:-haiku}"
|
|
93
|
+
|
|
94
|
+
# bulk-read is one of the two call sites the pipeline makes itself, so routing
|
|
95
|
+
# may send it outside the Anthropic ladder - unlike a subagent, nothing here
|
|
96
|
+
# belongs to the host. An explicit --model still wins: a caller that named a
|
|
97
|
+
# rung asked for that rung.
|
|
98
|
+
if [ -z "${MODEL_EXPLICIT:-}" ] && [ -x "$HERE/../lib/model-dispatch.sh" ]; then
|
|
99
|
+
MODEL=$("$HERE/../lib/model-dispatch.sh" bulk-read \
|
|
100
|
+
--phase "$PHASE" --default "$MODEL" 2>/dev/null) || MODEL="${PREF_MODEL:-haiku}"
|
|
101
|
+
fi
|
|
93
102
|
case "$TIMEOUT" in "" ) TIMEOUT="$PREF_TIMEOUT" ;; esac
|
|
94
103
|
case "$TIMEOUT" in ""|*[!0-9]*|0) TIMEOUT=60 ;; esac
|
|
95
104
|
|
|
@@ -5,14 +5,14 @@
|
|
|
5
5
|
#
|
|
6
6
|
# WHY THIS EXISTS
|
|
7
7
|
#
|
|
8
|
-
# Every persistent write used to live in Phase
|
|
9
|
-
# ledger distill, the knowledge-base append, the code-graph refresh. And Phase
|
|
8
|
+
# Every persistent write used to live in Phase 5: triage ingest, the learnings
|
|
9
|
+
# ledger distill, the knowledge-base append, the code-graph refresh. And Phase 5
|
|
10
10
|
# is, by the pipeline's own admission in features/code-graph.md, the phase a run
|
|
11
11
|
# is LEAST likely to reach. A run killed in Phase 3, a session that hits its
|
|
12
12
|
# context ceiling, a crash after review - each one threw away everything it had
|
|
13
13
|
# established, and the next run on the same repo rediscovered it from scratch.
|
|
14
14
|
#
|
|
15
|
-
# So the writes move here, and Phase
|
|
15
|
+
# So the writes move here, and Phase 5 becomes the LAST flush rather than the
|
|
16
16
|
# only one. Phase boundaries call this, and so does SessionEnd. Nothing about
|
|
17
17
|
# the trigger depends on a model noticing that a moment qualifies: it hangs on
|
|
18
18
|
# a phase transition and on process exit, both objectively visible without any
|
|
@@ -20,15 +20,15 @@
|
|
|
20
20
|
#
|
|
21
21
|
# What it does NOT do: call a model. Everything here is derived from artefacts
|
|
22
22
|
# already on disk (triage-output.json) plus agent-state.json. The parts of
|
|
23
|
-
# Phase
|
|
24
|
-
# per-repo memory synthesis - stay in Phase
|
|
23
|
+
# Phase 5 that genuinely need a model - the knowledge-base extraction, the
|
|
24
|
+
# per-repo memory synthesis - stay in Phase 5, because a hook cannot think.
|
|
25
25
|
#
|
|
26
26
|
# Usage:
|
|
27
27
|
# ./capture-flush.sh [--state <agent-state.json>] [--if-stale] [--json] [--quiet]
|
|
28
28
|
#
|
|
29
29
|
# --state the run to flush. Default: resolved from the newest task dir
|
|
30
30
|
# under $HOME/.claude/logs/multi-agent (see resolve_state).
|
|
31
|
-
# --if-stale flush only when the run did NOT complete Phase
|
|
31
|
+
# --if-stale flush only when the run did NOT complete Phase 5 - the
|
|
32
32
|
# SessionEnd case. A finished run has already flushed.
|
|
33
33
|
# --json machine-readable result for a caller that wants to count rows.
|
|
34
34
|
# --quiet no stdout. Exit status still distinguishes the outcomes.
|
|
@@ -101,13 +101,13 @@ WORKTREE=$(jq -r '.worktreePath // empty' "$STATE" 2>/dev/null)
|
|
|
101
101
|
PHASE7=$(jq -r '[.phases[]? | select((.id // "") == "7") | .status] | first // ""' "$STATE" 2>/dev/null)
|
|
102
102
|
|
|
103
103
|
if [ "$IF_STALE" -eq 1 ] && [ "$PHASE7" = "completed" ]; then
|
|
104
|
-
say "capture-flush: ${TASK_ID:-run} already completed Phase
|
|
104
|
+
say "capture-flush: ${TASK_ID:-run} already completed Phase 5 - nothing stale"
|
|
105
105
|
[ "$JSON" -eq 1 ] && printf '{"status":"noop","reason":"already-flushed","taskId":"%s"}\n' "$TASK_ID"
|
|
106
106
|
exit 0
|
|
107
107
|
fi
|
|
108
108
|
|
|
109
109
|
# The triage artefact is the only input either store needs, and it has two homes:
|
|
110
|
-
# Phase
|
|
110
|
+
# Phase 4 removes the worktree once the PR is open, so the salvaged copy under
|
|
111
111
|
# artifactsPath is tried FIRST. Reading the worktree path first would degrade
|
|
112
112
|
# silently for exactly the runs this script exists to rescue.
|
|
113
113
|
TRIAGE=""
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
# the user learns to skip, which costs more than it saves.
|
|
12
12
|
#
|
|
13
13
|
# It answers two questions:
|
|
14
|
-
# - Is there a pipeline run that stopped before Phase
|
|
14
|
+
# - Is there a pipeline run that stopped before Phase 5? Name it and the resume
|
|
15
15
|
# command, because that run's work is recoverable and its findings are not
|
|
16
16
|
# yet in the durable stores until it flushes.
|
|
17
17
|
# - Is the pipeline's own observation queue stale? (>= REVIEW_DAYS since the
|
|
@@ -32,7 +32,7 @@ REVIEW_DAYS=7
|
|
|
32
32
|
# Three buckets, not one line. Measured on the development machine: 22 runs read
|
|
33
33
|
# `in_progress` and every one was over a day old, but they are not one failure.
|
|
34
34
|
# 13 stopped at Phase 0, which is almost entirely questions - those runs never
|
|
35
|
-
# started. 3 had their PR already open and were waiting at Phase
|
|
35
|
+
# started. 3 had their PR already open and were waiting at Phase 4/7, where the
|
|
36
36
|
# pipeline pauses ON PURPOSE for channel selection. 4 died mid-development.
|
|
37
37
|
#
|
|
38
38
|
# Reporting the newest one of those as "stopped at Phase N" told the truth about
|
|
@@ -71,7 +71,7 @@ if [ -d "$LOGS" ] && command -v jq >/dev/null 2>&1; then
|
|
|
71
71
|
PR=$(jq -r 'if (.pr|type)=="string" then .pr elif (.pr|type)=="object" then (.pr.url // .pr.number // "") else "" end | tostring' "$f" 2>/dev/null)
|
|
72
72
|
|
|
73
73
|
# Waiting for you, not broken: an open PR means the work landed, and
|
|
74
|
-
# Phase
|
|
74
|
+
# Phase 5 pauses for channel selection by design (modes.md).
|
|
75
75
|
if [ "$STATUS" = "awaiting_input" ] || [ -n "$PR" ] ||
|
|
76
76
|
[ "$PHASE" = "6" ] || [ "$PHASE" = "7" ]; then
|
|
77
77
|
AWAITING_N=$((AWAITING_N + 1))
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
// classify-plan-safety.mjs - v7.0.G
|
|
3
3
|
//
|
|
4
|
-
// Heuristic safety classifier for a Phase
|
|
4
|
+
// Heuristic safety classifier for a Phase 1 plan. Autopilot's "zero user
|
|
5
5
|
// interaction" contract is fine for small, predictable tasks but dangerous
|
|
6
6
|
// when a plan touches the security path, deletes files without paired
|
|
7
7
|
// tests, or sprawls across many files. This script inspects the plan and
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
|
|
2
|
+
"_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. `provider` says which layer a rung belongs to: an `anthropic` rung is reachable from every call site, a non-anthropic one only from the call sites the pipeline makes itself (bulk-read, research), which is what model-dispatch.sh enforces. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
|
|
3
3
|
"schemaVersion": "1.1.0",
|
|
4
4
|
"prices": {
|
|
5
5
|
"fable": {
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
"outPerMtok": 50.0,
|
|
8
8
|
"cacheReadPerMtok": 1.0,
|
|
9
9
|
"modelId": "claude-fable-5",
|
|
10
|
+
"provider": "anthropic",
|
|
10
11
|
"note": "Top tier (restored v10.6.0) - architects, Reviewer 1, triage. Verified against Anthropic pricing 2026-07-02."
|
|
11
12
|
},
|
|
12
13
|
"opus": {
|
|
@@ -14,6 +15,7 @@
|
|
|
14
15
|
"outPerMtok": 25.0,
|
|
15
16
|
"cacheReadPerMtok": 0.5,
|
|
16
17
|
"modelId": "claude-opus-5",
|
|
18
|
+
"provider": "anthropic",
|
|
17
19
|
"note": "Second tier - dev phase on a Short run, Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
|
|
18
20
|
},
|
|
19
21
|
"sonnet": {
|
|
@@ -21,6 +23,7 @@
|
|
|
21
23
|
"outPerMtok": 15.0,
|
|
22
24
|
"cacheReadPerMtok": 0.3,
|
|
23
25
|
"modelId": "claude-sonnet-5",
|
|
26
|
+
"provider": "anthropic",
|
|
24
27
|
"note": "Floor tier for Claude-family dispatch - Reviewer 3 on both hosts, and the terminal rung of the fallback ladder. Priced at the standard 3/15 rather than the 2/10 introductory rate that runs through 2026-08-31: over-reporting during the intro window is the safe direction for a cost ledger, and it needs no dated edit when the intro ends."
|
|
25
28
|
},
|
|
26
29
|
"haiku": {
|
|
@@ -28,6 +31,7 @@
|
|
|
28
31
|
"outPerMtok": 5.0,
|
|
29
32
|
"cacheReadPerMtok": 0.1,
|
|
30
33
|
"modelId": "claude-haiku-4-5",
|
|
34
|
+
"provider": "anthropic",
|
|
31
35
|
"note": "Speed tier - task-clarifier and other latency-sensitive dispatches, plus the terminal rung of the fallback ladder. Named by its alias like every other rung here rather than by a dated snapshot, so the four rungs stay comparable at a glance."
|
|
32
36
|
},
|
|
33
37
|
"gpt-5.4": {
|
|
@@ -35,6 +39,7 @@
|
|
|
35
39
|
"outPerMtok": 30.0,
|
|
36
40
|
"cacheReadPerMtok": 1.0,
|
|
37
41
|
"modelId": "gpt-5.4",
|
|
42
|
+
"provider": "openai",
|
|
38
43
|
"note": "Copilot CLI Reviewer 2 and Codex CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
|
|
39
44
|
},
|
|
40
45
|
"gpt-5.6": {
|
|
@@ -42,6 +47,7 @@
|
|
|
42
47
|
"outPerMtok": 30.0,
|
|
43
48
|
"cacheReadPerMtok": 1.0,
|
|
44
49
|
"modelId": "gpt-5.6",
|
|
50
|
+
"provider": "openai",
|
|
45
51
|
"note": "Codex CLI top tier - Reviewer 1 at xhigh effort, Reviewer 3 at medium, triage at max, and the fable/opus rungs of the Codex persona tier map. Reasoning effort changes output volume, not the per-token rate, so one entry covers every effort level. Approximate; verify against OpenAI pricing before relying on totals."
|
|
46
52
|
},
|
|
47
53
|
"gpt-5.6-terra": {
|
|
@@ -49,6 +55,7 @@
|
|
|
49
55
|
"outPerMtok": 5.0,
|
|
50
56
|
"cacheReadPerMtok": 0.1,
|
|
51
57
|
"modelId": "gpt-5.6-terra",
|
|
58
|
+
"provider": "openai",
|
|
52
59
|
"note": "Codex CLI floor tier - speed-optimised, maps the haiku rung of the Codex persona tier map (task-clarifier). Approximate; verify against OpenAI pricing before relying on totals."
|
|
53
60
|
}
|
|
54
61
|
}
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
/**
|
|
4
4
|
* @file diff-explain.mjs - v7.8.0 Paket B
|
|
5
5
|
*
|
|
6
|
-
* Bridges Phase
|
|
6
|
+
* Bridges Phase 3 triage output to the actual git diff. For each accepted /
|
|
7
7
|
* deferred / rejected finding, locates the corresponding hunk in the branch's
|
|
8
8
|
* diff against base and renders an annotated markdown report.
|
|
9
9
|
*
|
|
@@ -658,7 +658,7 @@ function checkMcpRegistration() {
|
|
|
658
658
|
// checkMcpRegistration answers "is OURS registered". This answers the question the
|
|
659
659
|
// user never gets asked: every registered server's tool list is sent with every
|
|
660
660
|
// turn, they are added one at a time, and nobody sees the running total - our own
|
|
661
|
-
// toolkit is
|
|
661
|
+
// toolkit is 115 tools by itself. This check only ever REPORTS. It never disables
|
|
662
662
|
// anything, and it never blocks or warns, because how many servers are worth their
|
|
663
663
|
// context is the user's call and not a health failure.
|
|
664
664
|
//
|
|
@@ -744,7 +744,7 @@ function checkDiskSpace() {
|
|
|
744
744
|
}
|
|
745
745
|
|
|
746
746
|
// A finished task removes its own worktree at PR time, but a run that dies
|
|
747
|
-
// before Phase
|
|
747
|
+
// before Phase 4 never reaches that step and nothing else collects it: the
|
|
748
748
|
// finalizer only runs on success and gc-worktrees only sweeps entries git no
|
|
749
749
|
// longer knows about. So worktrees accumulate silently, and each one is a full
|
|
750
750
|
// second checkout. Measured on one real iOS repo: 15 left behind, 11 GB.
|
|
@@ -788,7 +788,7 @@ function checkWorktreeResidue() {
|
|
|
788
788
|
report(
|
|
789
789
|
"worktree-residue",
|
|
790
790
|
"WARN",
|
|
791
|
-
`${entries.length} worktree(s) left under ${wt}${size}; runs that stopped before Phase
|
|
791
|
+
`${entries.length} worktree(s) left under ${wt}${size}; runs that stopped before Phase 4 are never collected`,
|
|
792
792
|
"run /multi-agent:garbage-collect, or /multi-agent:kill for a task you know is dead",
|
|
793
793
|
);
|
|
794
794
|
return;
|