@mmerterden/multi-agent-pipeline 20.0.0 → 20.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +29 -0
- package/README.md +5 -5
- package/README.tr.md +5 -5
- package/SECURITY.md +3 -3
- package/docs/adr/0011-dormant-ci.md +10 -1
- package/docs/architecture.md +2 -2
- package/docs/ecosystem.md +5 -5
- package/docs/facts.json +7 -6
- package/install/_codex-agents.mjs +1 -1
- package/manifest.json +48 -41
- package/package.json +1 -1
- package/pipeline/agents/code-reviewer.md +2 -2
- package/pipeline/agents/dev-critic.md +5 -5
- package/pipeline/agents/security-auditor.md +80 -72
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/channels/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/scan/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/security-review/SKILL.md +52 -0
- package/pipeline/commands/multi-agent/sync/SKILL.md +3 -3
- package/pipeline/multi-agent-refs/component-dispatch.md +5 -5
- package/pipeline/multi-agent-refs/cross-cli-contract.md +6 -6
- package/pipeline/multi-agent-refs/features/security-audit.md +55 -0
- package/pipeline/multi-agent-refs/phases/modes.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-3-review.md +9 -15
- package/pipeline/multi-agent-refs/phases/phase-5-report.md +1 -1
- package/pipeline/multi-agent-refs/threat-model.md +39 -0
- package/pipeline/schemas/agent-state.schema.json +23 -0
- package/pipeline/schemas/phases.json +1 -2
- package/pipeline/schemas/prefs.schema.json +0 -4
- package/pipeline/schemas/reviewer-output.schema.json +99 -2
- package/pipeline/schemas/security-finding.schema.json +144 -0
- package/pipeline/scripts/_stack-routing.mjs +1 -0
- package/pipeline/scripts/gc-abandoned.sh +16 -9
- package/pipeline/scripts/render-work-summary.sh +7 -4
- package/pipeline/skills/.skill-manifest.json +13 -5
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +3 -4
- package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-security-review/SKILL.md +29 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +3 -3
- package/pipeline/skills/shared/external/security-review/SKILL.md +64 -0
- package/pipeline/skills/shared/external/security-review/references/owasp-mobile-top10-2024.md +53 -0
- package/pipeline/skills/shared/external/security-review/references/owasp-web-api-top10-2021.md +56 -0
- package/pipeline/commands/security-review.md +0 -6
|
@@ -857,6 +857,29 @@
|
|
|
857
857
|
}
|
|
858
858
|
}
|
|
859
859
|
},
|
|
860
|
+
"threatModel": {
|
|
861
|
+
"type": "object",
|
|
862
|
+
"additionalProperties": true,
|
|
863
|
+
"description": "The run-scoped threat model produced at Phase 3 Step 2.7 (or by /multi-agent:security-review), mirrored here so a resume reuses it instead of re-deriving it. Contract: multi-agent-refs/threat-model.md.",
|
|
864
|
+
"properties": {
|
|
865
|
+
"path": {
|
|
866
|
+
"type": "string",
|
|
867
|
+
"description": "Path to .pipeline/threat-model.md in the worktree."
|
|
868
|
+
},
|
|
869
|
+
"sha": {
|
|
870
|
+
"type": "string",
|
|
871
|
+
"description": "Content hash, so a later step can tell whether the model changed."
|
|
872
|
+
},
|
|
873
|
+
"producedAt": {
|
|
874
|
+
"type": "string",
|
|
875
|
+
"format": "date-time"
|
|
876
|
+
},
|
|
877
|
+
"producedBy": {
|
|
878
|
+
"type": "string",
|
|
879
|
+
"description": "Which step or command wrote it (e.g. 'phase-3-step-2.7', 'security-review')."
|
|
880
|
+
}
|
|
881
|
+
}
|
|
882
|
+
},
|
|
860
883
|
"reviewIterations": {
|
|
861
884
|
"type": "array",
|
|
862
885
|
"items": {
|
|
@@ -78,7 +78,6 @@
|
|
|
78
78
|
"thresholds": {
|
|
79
79
|
"waitingFromPhase": 4,
|
|
80
80
|
"mcpAllowedThroughPhase": 1,
|
|
81
|
-
"
|
|
82
|
-
"note": "Phase numbers other code compares against, named here so a renumbering moves them with the contract. waitingFromPhase: runs-index.mjs groups a run as 'waiting on you' from Commit on. mcpAllowedThroughPhase: Figma MCP is reachable only through Plan; smoke-no-mcp-in-dev-phases.sh fails any recorded call at a higher phase. Its literal threshold is unchanged from the eight-phase contract on purpose - Analysis was 1 and is now inside Plan, also 1, so the permitted set {0,1} is identical. shortRunFromPhase: the depth picker's Short set starts here."
|
|
81
|
+
"note": "Phase numbers other code compares against, named here so a renumbering moves them with the contract. waitingFromPhase: runs-index.mjs groups a run as 'waiting on you' from Commit on. mcpAllowedThroughPhase: Figma MCP is reachable only through Plan; smoke-no-mcp-in-dev-phases.sh fails any recorded call at a higher phase. Its literal threshold is unchanged from the eight-phase contract on purpose - Analysis was 1 and is now inside Plan, also 1, so the permitted set {0,1} is identical."
|
|
83
82
|
}
|
|
84
83
|
}
|
|
@@ -2140,10 +2140,6 @@
|
|
|
2140
2140
|
"format": "date-time",
|
|
2141
2141
|
"description": "Timestamp of last task start against this project."
|
|
2142
2142
|
},
|
|
2143
|
-
"componentDevWorkflow": {
|
|
2144
|
-
"type": "boolean",
|
|
2145
|
-
"description": "Project follows the component (Configuration / View / Modifiers) development workflow, so Phase 2 dispatches the component skills instead of the standard TDD flow."
|
|
2146
|
-
},
|
|
2147
2143
|
"figmaConfigPath": {
|
|
2148
2144
|
"type": "string",
|
|
2149
2145
|
"description": "Path to per-project figma-config.json (for Figma pipeline projects)."
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/reviewer-output.schema.json",
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.4.0",
|
|
5
5
|
"title": "Multi-Agent Pipeline - Phase 3 reviewer output",
|
|
6
|
-
"description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Every host dispatches 3 parallel reviewers; the middle slot is CLI-aware: Claude Code (Fable, Opus, Sonnet); Copilot CLI (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them. v1.1.0 adds the rule-ID conformance checklist: when the orchestrator supplies a ${CRITERIA} block (Phase 3 Step 1.78), the reviewer must return one conformance row per selected rule ID. Findings alone cannot answer 'was this applied completely' - a reviewer that opened nothing returns the same empty findings array as one that checked everything. v1.2.0 adds the optional per-finding fingerprint (Phase 3 Step 2.1): the stable id a finding keeps across review rounds.",
|
|
6
|
+
"description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Every host dispatches 3 parallel reviewers; the middle slot is CLI-aware: Claude Code (Fable, Opus, Sonnet); Copilot CLI (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them. v1.1.0 adds the rule-ID conformance checklist: when the orchestrator supplies a ${CRITERIA} block (Phase 3 Step 1.78), the reviewer must return one conformance row per selected rule ID. Findings alone cannot answer 'was this applied completely' - a reviewer that opened nothing returns the same empty findings array as one that checked everything. v1.2.0 adds the optional per-finding fingerprint (Phase 3 Step 2.1): the stable id a finding keeps across review rounds. v1.4.0 adds the optional per-finding `security` envelope (OWASP + CWE + CVSS + evidence + remediation): the security-auditor emits reviewer-output objects whose findings carry it, so a blocking security finding merges at Step 3.0 and blocks Phase 4 like any reviewer blocker. General reviewers omit it. Its shape mirrors security-finding.schema.json, kept in step by smoke-security-schema-parity.sh.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"required": ["findings", "approved"],
|
|
@@ -128,6 +128,103 @@
|
|
|
128
128
|
"type": "string",
|
|
129
129
|
"pattern": "^F:[0-9a-f]{8}$",
|
|
130
130
|
"description": "Stable cross-round identity of the finding, computed by finding-fingerprint.mjs from (file, ruleId or normalized issue text). Never includes the line. On iteration >= 2 a reviewer that recognises an entry from <previous-round-findings> echoes its fingerprint; otherwise leave it unset and the script fills it in."
|
|
131
|
+
},
|
|
132
|
+
"security": {
|
|
133
|
+
"type": "object",
|
|
134
|
+
"additionalProperties": false,
|
|
135
|
+
"required": ["owaspCategory", "cwe", "cvss", "evidence", "confidence", "remediation"],
|
|
136
|
+
"description": "OPTIONAL security envelope. Present when the finding comes from the security-auditor (Phase 3 Step 3.0 merge) or /multi-agent:security-review; absent on a general code-reviewer finding. Its shape is the security block of security-finding.schema.json, kept in step by smoke-security-schema-parity.sh (self-contained, no cross-file $ref). validate-reviewer.mjs ignores it; triage carries it through unchanged so Phase 5 can report CVSS + CWE + remediation.",
|
|
137
|
+
"properties": {
|
|
138
|
+
"owaspCategory": {
|
|
139
|
+
"type": "string",
|
|
140
|
+
"pattern": "^(A(0[1-9]|10):2021|M[1-9]:2024|M10:2024)( .+)?$",
|
|
141
|
+
"description": "OWASP Top 10 2021 id (web/API) or OWASP Mobile Top 10 2024 id, optionally followed by a human title."
|
|
142
|
+
},
|
|
143
|
+
"cwe": {
|
|
144
|
+
"type": "string",
|
|
145
|
+
"pattern": "^CWE-[0-9]{1,5}$",
|
|
146
|
+
"description": "The specific CWE weakness id. One per finding."
|
|
147
|
+
},
|
|
148
|
+
"cve": {
|
|
149
|
+
"type": "string",
|
|
150
|
+
"pattern": "^CVE-[0-9]{4}-[0-9]{4,}$",
|
|
151
|
+
"description": "A published CVE, for a known-vulnerable dependency finding."
|
|
152
|
+
},
|
|
153
|
+
"cvss": {
|
|
154
|
+
"type": "object",
|
|
155
|
+
"additionalProperties": false,
|
|
156
|
+
"required": ["vector", "baseScore", "band"],
|
|
157
|
+
"description": "CVSS 3.1 base metrics; baseScore and band are computed from the vector by security_cvss_score.",
|
|
158
|
+
"properties": {
|
|
159
|
+
"vector": {
|
|
160
|
+
"type": "string",
|
|
161
|
+
"pattern": "^CVSS:3[.]1/AV:[NALP]/AC:[LH]/PR:[NLH]/UI:[NR]/S:[UC]/C:[NLH]/I:[NLH]/A:[NLH]$",
|
|
162
|
+
"description": "Full CVSS 3.1 base vector."
|
|
163
|
+
},
|
|
164
|
+
"baseScore": {
|
|
165
|
+
"type": "number",
|
|
166
|
+
"minimum": 0,
|
|
167
|
+
"maximum": 10,
|
|
168
|
+
"description": "0.0 .. 10.0 base score."
|
|
169
|
+
},
|
|
170
|
+
"band": {
|
|
171
|
+
"type": "string",
|
|
172
|
+
"enum": ["none", "low", "medium", "high", "critical"],
|
|
173
|
+
"description": "Qualitative band; maps to the finding severity (critical/high -> blocking, medium -> important, low/none -> suggestion)."
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
},
|
|
177
|
+
"evidence": {
|
|
178
|
+
"type": "string",
|
|
179
|
+
"minLength": 8,
|
|
180
|
+
"description": "What in the code proves the finding, cited by file:line."
|
|
181
|
+
},
|
|
182
|
+
"counterevidence": {
|
|
183
|
+
"type": "string",
|
|
184
|
+
"description": "What would disprove it, or the condition under which it is a false positive."
|
|
185
|
+
},
|
|
186
|
+
"confidence": {
|
|
187
|
+
"type": "string",
|
|
188
|
+
"enum": ["high", "medium", "low"],
|
|
189
|
+
"description": "How sure the auditor is the finding is real."
|
|
190
|
+
},
|
|
191
|
+
"confidenceRationale": {
|
|
192
|
+
"type": "string",
|
|
193
|
+
"description": "Why the confidence is what it is."
|
|
194
|
+
},
|
|
195
|
+
"severityChangeConditions": {
|
|
196
|
+
"type": "string",
|
|
197
|
+
"description": "The fact that, if learned, would move the severity."
|
|
198
|
+
},
|
|
199
|
+
"remediation": {
|
|
200
|
+
"type": "string",
|
|
201
|
+
"minLength": 8,
|
|
202
|
+
"description": "Remediation steps in prose."
|
|
203
|
+
},
|
|
204
|
+
"remediationDiff": {
|
|
205
|
+
"type": "object",
|
|
206
|
+
"additionalProperties": false,
|
|
207
|
+
"required": ["before", "after"],
|
|
208
|
+
"description": "The fix as a before/after pair.",
|
|
209
|
+
"properties": {
|
|
210
|
+
"before": { "type": "string" },
|
|
211
|
+
"after": { "type": "string" }
|
|
212
|
+
}
|
|
213
|
+
},
|
|
214
|
+
"endpoint": {
|
|
215
|
+
"type": "string",
|
|
216
|
+
"description": "The HTTP route or RPC method for a service-layer finding."
|
|
217
|
+
},
|
|
218
|
+
"method": {
|
|
219
|
+
"type": "string",
|
|
220
|
+
"enum": ["GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS"],
|
|
221
|
+
"description": "HTTP method for an endpoint finding."
|
|
222
|
+
},
|
|
223
|
+
"fixVerification": {
|
|
224
|
+
"type": "string",
|
|
225
|
+
"description": "How to confirm the fix worked - the test to add or check to run."
|
|
226
|
+
}
|
|
227
|
+
}
|
|
131
228
|
}
|
|
132
229
|
}
|
|
133
230
|
}
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/security-finding.schema.json",
|
|
4
|
+
"version": "1.0.0",
|
|
5
|
+
"title": "Multi-Agent Pipeline - security finding",
|
|
6
|
+
"description": "Contract for one finding produced by the security-auditor subagent (Phase 3) and the /multi-agent:security-review command. A security finding IS a reviewer-output.schema.json finding - same core fields (severity, file, line, issue, fix) - carrying one extra `security` block. That is exactly why it works: the auditor emits reviewer-output objects whose findings have the block, they merge into the Phase 3 reviewer set at Step 3.0, and a blocking one blocks Phase 4 like any reviewer blocker. reviewer-output.schema.json declares the same `security` property as OPTIONAL on its finding; here it is REQUIRED, which is the whole difference between a general finding and a security finding. The severity enum is the reviewer's (blocking|important|suggestion), never a private Critical/High/Medium scale: severity is derived from security.cvss.band by the rule on the `severity` property, and a severity that contradicts the band is the one inconsistency this contract forbids. The two schemas are kept in step by smoke-security-schema-parity.sh, not by a cross-file $ref, matching the self-contained style of the other pipeline schemas.",
|
|
7
|
+
"type": "object",
|
|
8
|
+
"additionalProperties": false,
|
|
9
|
+
"required": ["severity", "file", "line", "issue", "fix", "security"],
|
|
10
|
+
"properties": {
|
|
11
|
+
"severity": {
|
|
12
|
+
"type": "string",
|
|
13
|
+
"enum": ["blocking", "important", "suggestion"],
|
|
14
|
+
"description": "The reviewer severity, derived from security.cvss.band so this finding filters through triage like any other: critical or high band -> blocking; medium -> important; low or none -> suggestion. It is not a free choice - a severity that disagrees with the band is the one inconsistency this schema exists to forbid."
|
|
15
|
+
},
|
|
16
|
+
"file": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"minLength": 1,
|
|
19
|
+
"description": "Path relative to repo root. The auditor must not invent a file that is not in the tree."
|
|
20
|
+
},
|
|
21
|
+
"line": {
|
|
22
|
+
"type": "integer",
|
|
23
|
+
"minimum": 0,
|
|
24
|
+
"description": "Line number. 0 = whole-file, configuration-level, or dependency-manifest finding."
|
|
25
|
+
},
|
|
26
|
+
"issue": {
|
|
27
|
+
"type": "string",
|
|
28
|
+
"minLength": 4,
|
|
29
|
+
"description": "What is wrong, in one sentence. The vulnerability, not the fix."
|
|
30
|
+
},
|
|
31
|
+
"fix": {
|
|
32
|
+
"type": "string",
|
|
33
|
+
"minLength": 4,
|
|
34
|
+
"description": "Concrete remediation in one line, for the reviewer-shaped view. The full before/after diff lives in security.remediationDiff."
|
|
35
|
+
},
|
|
36
|
+
"ruleId": {
|
|
37
|
+
"type": "string",
|
|
38
|
+
"minLength": 1,
|
|
39
|
+
"description": "Stable id of a cited rule (e.g. SEC-03) when the finding comes from a standards registry supplied to the auditor. The author can look the rule up and argue with it rather than with the auditor."
|
|
40
|
+
},
|
|
41
|
+
"criteriaSource": {
|
|
42
|
+
"type": "string",
|
|
43
|
+
"minLength": 1,
|
|
44
|
+
"description": "Which source the rule came from: a registry name, a compliance catalog (apple-archive-compliance / google-play-compliance), a reference path, or 'threat-model' when the finding is derived from the run-scoped threat model rather than a fixed rule."
|
|
45
|
+
},
|
|
46
|
+
"security": {
|
|
47
|
+
"type": "object",
|
|
48
|
+
"additionalProperties": false,
|
|
49
|
+
"required": ["owaspCategory", "cwe", "cvss", "evidence", "confidence", "remediation"],
|
|
50
|
+
"description": "The security envelope. Present on every security finding; optional on a general reviewer finding.",
|
|
51
|
+
"properties": {
|
|
52
|
+
"owaspCategory": {
|
|
53
|
+
"type": "string",
|
|
54
|
+
"pattern": "^(A(0[1-9]|10):2021|M[1-9]:2024|M10:2024)( .+)?$",
|
|
55
|
+
"description": "The OWASP category: web/API uses OWASP Top 10 2021 ids (A01:2021 .. A10:2021), mobile uses OWASP Mobile Top 10 2024 ids (M1:2024 .. M10:2024). An optional human title may follow the id, e.g. 'A01:2021 Broken Access Control'."
|
|
56
|
+
},
|
|
57
|
+
"cwe": {
|
|
58
|
+
"type": "string",
|
|
59
|
+
"pattern": "^CWE-[0-9]{1,5}$",
|
|
60
|
+
"description": "The specific weakness, as a validated CWE id (CWE-89, CWE-798, ...). One weakness per finding; if a line has two, it is two findings."
|
|
61
|
+
},
|
|
62
|
+
"cve": {
|
|
63
|
+
"type": "string",
|
|
64
|
+
"pattern": "^CVE-[0-9]{4}-[0-9]{4,}$",
|
|
65
|
+
"description": "A published CVE, when the finding is a known-vulnerable dependency rather than first-party code. Set by the dependency-audit path, not by static code review."
|
|
66
|
+
},
|
|
67
|
+
"cvss": {
|
|
68
|
+
"type": "object",
|
|
69
|
+
"additionalProperties": false,
|
|
70
|
+
"required": ["vector", "baseScore", "band"],
|
|
71
|
+
"description": "CVSS 3.1 base metrics. The auditor writes the vector; security_cvss_score (toolkit) computes baseScore and band from it, so the two cannot drift from the vector by hand.",
|
|
72
|
+
"properties": {
|
|
73
|
+
"vector": {
|
|
74
|
+
"type": "string",
|
|
75
|
+
"pattern": "^CVSS:3[.]1/AV:[NALP]/AC:[LH]/PR:[NLH]/UI:[NR]/S:[UC]/C:[NLH]/I:[NLH]/A:[NLH]$",
|
|
76
|
+
"description": "A full CVSS 3.1 base vector string. All eight base metrics are required; temporal and environmental metrics are out of scope for a static review."
|
|
77
|
+
},
|
|
78
|
+
"baseScore": {
|
|
79
|
+
"type": "number",
|
|
80
|
+
"minimum": 0,
|
|
81
|
+
"maximum": 10,
|
|
82
|
+
"description": "The CVSS 3.1 base score computed from the vector by security_cvss_score. 0.0 .. 10.0, one decimal."
|
|
83
|
+
},
|
|
84
|
+
"band": {
|
|
85
|
+
"type": "string",
|
|
86
|
+
"enum": ["none", "low", "medium", "high", "critical"],
|
|
87
|
+
"description": "The qualitative band of baseScore: none 0.0, low 0.1-3.9, medium 4.0-6.9, high 7.0-8.9, critical 9.0-10.0. This is what maps to `severity`."
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
},
|
|
91
|
+
"evidence": {
|
|
92
|
+
"type": "string",
|
|
93
|
+
"minLength": 8,
|
|
94
|
+
"description": "What in the code proves the finding, quoted or cited by file:line. A static finding with no evidence is a guess; this is what a reader checks before agreeing."
|
|
95
|
+
},
|
|
96
|
+
"counterevidence": {
|
|
97
|
+
"type": "string",
|
|
98
|
+
"description": "What in the code would DISPROVE the finding, or the condition under which it is a false positive - a guard elsewhere, a framework default, an unreachable path. Stating it keeps confidence honest; leaving it blank asserts there is none."
|
|
99
|
+
},
|
|
100
|
+
"confidence": {
|
|
101
|
+
"type": "string",
|
|
102
|
+
"enum": ["high", "medium", "low"],
|
|
103
|
+
"description": "How sure the auditor is that this is real. Low confidence does not mean do-not-report; it means report with the counterevidence and let triage weigh it."
|
|
104
|
+
},
|
|
105
|
+
"confidenceRationale": {
|
|
106
|
+
"type": "string",
|
|
107
|
+
"description": "One line on why the confidence is what it is - what was and was not verifiable from the code alone."
|
|
108
|
+
},
|
|
109
|
+
"severityChangeConditions": {
|
|
110
|
+
"type": "string",
|
|
111
|
+
"description": "The fact that, if learned, would move the severity: 'critical if this endpoint is unauthenticated in production; medium if it is admin-only'. Names the assumption the score rests on."
|
|
112
|
+
},
|
|
113
|
+
"remediation": {
|
|
114
|
+
"type": "string",
|
|
115
|
+
"minLength": 8,
|
|
116
|
+
"description": "The remediation steps in prose - what to change and why it closes the weakness. Actionable, not 'sanitize input'."
|
|
117
|
+
},
|
|
118
|
+
"remediationDiff": {
|
|
119
|
+
"type": "object",
|
|
120
|
+
"additionalProperties": false,
|
|
121
|
+
"required": ["before", "after"],
|
|
122
|
+
"description": "The fix as a before/after pair the author can read as a diff. Optional - some findings are configuration or process changes with no single code hunk - but preferred for first-party code.",
|
|
123
|
+
"properties": {
|
|
124
|
+
"before": { "type": "string", "description": "The vulnerable code as it stands." },
|
|
125
|
+
"after": { "type": "string", "description": "The same region after the fix." }
|
|
126
|
+
}
|
|
127
|
+
},
|
|
128
|
+
"endpoint": {
|
|
129
|
+
"type": "string",
|
|
130
|
+
"description": "The HTTP route or RPC method the finding sits on, when it is a service-layer issue. Empty for non-network findings."
|
|
131
|
+
},
|
|
132
|
+
"method": {
|
|
133
|
+
"type": "string",
|
|
134
|
+
"enum": ["GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS"],
|
|
135
|
+
"description": "The HTTP method for an endpoint finding."
|
|
136
|
+
},
|
|
137
|
+
"fixVerification": {
|
|
138
|
+
"type": "string",
|
|
139
|
+
"description": "How to confirm the fix worked - the test to add or the check to run. Static review cannot fire an exploit, so this is the empirical step a human or a later dynamic pass takes."
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
}
|
|
@@ -11,9 +11,9 @@
|
|
|
11
11
|
# Three things keep this from being a foot-gun, and each is a rule rather than a
|
|
12
12
|
# heuristic:
|
|
13
13
|
#
|
|
14
|
-
# 1. A run WAITING FOR YOU is never reaped. An open PR,
|
|
15
|
-
# `status: awaiting_input` means the work landed and the pipeline
|
|
16
|
-
# holding for an answer by design. That is finished work, not residue.
|
|
14
|
+
# 1. A run WAITING FOR YOU is never reaped. An open PR, the Commit or Report
|
|
15
|
+
# phase, or `status: awaiting_input` means the work landed and the pipeline
|
|
16
|
+
# is holding for an answer by design. That is finished work, not residue.
|
|
17
17
|
# 2. A path that is not strictly inside `<repo>/.worktrees/` is never removed.
|
|
18
18
|
# This is not theoretical: four state files on that machine record
|
|
19
19
|
# `worktreePath` as the REPO ROOT, so a sweep that trusted the field would
|
|
@@ -96,6 +96,13 @@ case "$DAYS$PHASE0_DAYS" in *[!0-9]*) echo "gc-abandoned: --days and --phase0-da
|
|
|
96
96
|
command -v jq >/dev/null 2>&1 || { echo "gc-abandoned: jq is required"; exit 0; }
|
|
97
97
|
[ -d "$LOGS" ] || { echo "gc-abandoned: nothing to do (no state at $LOGS)"; exit 0; }
|
|
98
98
|
|
|
99
|
+
# The phase from which a stopped run is "waiting on you", not residue, is the
|
|
100
|
+
# contract's threshold - Commit onward - not a literal pinned to one numbering.
|
|
101
|
+
# runs-index.mjs groups the same way from the same field.
|
|
102
|
+
SELF_DIR=$(cd "$(dirname "$0")" && pwd)
|
|
103
|
+
WAITING_FROM=$(jq -r '.thresholds.waitingFromPhase' "$SELF_DIR/../schemas/phases.json" 2>/dev/null || echo 4)
|
|
104
|
+
case "$WAITING_FROM" in *[!0-9]*|"") WAITING_FROM=4 ;; esac
|
|
105
|
+
|
|
99
106
|
# GNU form FIRST. `stat -f` is a valid GNU flag (--file-system) that succeeds and
|
|
100
107
|
# prints a mount point, so BSD-first silently returns a number that is not a
|
|
101
108
|
# timestamp. ADR-0012 lists this among the constructs that look cross-platform
|
|
@@ -142,7 +149,7 @@ while IFS= read -r state; do
|
|
|
142
149
|
wt=$(jq -r '.worktreePath // ""' "$state" 2>/dev/null || true)
|
|
143
150
|
|
|
144
151
|
# Rule 1: waiting for you is not residue.
|
|
145
|
-
if [ "$status" = "awaiting_input" ] || [ -n "$pr" ] || [ "$phase"
|
|
152
|
+
if [ "$status" = "awaiting_input" ] || [ -n "$pr" ] || { [ "$phase" != "?" ] && [ "$phase" -ge "$WAITING_FROM" ] 2>/dev/null; }; then
|
|
146
153
|
skipped_waiting=$((skipped_waiting + 1))
|
|
147
154
|
continue
|
|
148
155
|
fi
|
|
@@ -271,10 +278,10 @@ if [ "$STATE_ONLY" -eq 0 ] && [ -d "$REPOS" ]; then
|
|
|
271
278
|
ph=$(printf '%s' "$row" | cut -f5)
|
|
272
279
|
|
|
273
280
|
# Rule 1 again: landed work is not residue, whichever pass finds it. Order
|
|
274
|
-
# matters here - the phase
|
|
275
|
-
# run whose status already says it FINISHED is not waiting for anyone. A
|
|
276
|
-
# complete run sitting at phase
|
|
277
|
-
# which is how three finished worktrees held 5.1 GB indefinitely.
|
|
281
|
+
# matters here - the waiting-phase test is a proxy for "still waiting", and
|
|
282
|
+
# a run whose status already says it FINISHED is not waiting for anyone. A
|
|
283
|
+
# complete run sitting at the report phase was read as waiting and never
|
|
284
|
+
# reaped, which is how three finished worktrees held 5.1 GB indefinitely.
|
|
278
285
|
case "$st" in
|
|
279
286
|
complete | completed | failed) ;;
|
|
280
287
|
awaiting_input)
|
|
@@ -282,7 +289,7 @@ if [ "$STATE_ONLY" -eq 0 ] && [ -d "$REPOS" ]; then
|
|
|
282
289
|
in_progress | paused)
|
|
283
290
|
continue ;; # pass 1 owns these; do not report them twice
|
|
284
291
|
*)
|
|
285
|
-
if [ -n "$pr" ] || [ "$ph"
|
|
292
|
+
if [ -n "$pr" ] || { [ "$ph" != "?" ] && [ "$ph" -ge "$WAITING_FROM" ] 2>/dev/null; }; then
|
|
286
293
|
skipped_waiting=$((skipped_waiting + 1)); continue
|
|
287
294
|
fi ;;
|
|
288
295
|
esac
|
|
@@ -34,7 +34,7 @@
|
|
|
34
34
|
# - <A> accepted · <D> deferred · <R> rejected · approved=<bool>
|
|
35
35
|
#
|
|
36
36
|
# #### Phases
|
|
37
|
-
# - 0 Init done · 1
|
|
37
|
+
# - 0 Init done · 1 Plan done · 2 Dev done · 3 Review done · 4 Commit done · 5 Report active
|
|
38
38
|
#
|
|
39
39
|
# Exit codes: 0 = rendered, 2 = missing state (caller should skip section).
|
|
40
40
|
|
|
@@ -171,7 +171,11 @@ fi
|
|
|
171
171
|
# Phase tick marks
|
|
172
172
|
phase_ticks=""
|
|
173
173
|
if [ -n "$TRACKER_FILE" ]; then
|
|
174
|
-
|
|
174
|
+
# The phase vocabulary is the contract's, not a literal pinned to one
|
|
175
|
+
# numbering: "<id> <name>" per phase, in id order, straight from phases.json.
|
|
176
|
+
PHASES_JSON="$_MA_RP_HERE/../schemas/phases.json"
|
|
177
|
+
LABELS=$(jq -c '[.phases | sort_by(.id) | .[] | "\(.id) \(.name)"]' "$PHASES_JSON" 2>/dev/null || echo '[]')
|
|
178
|
+
phase_ticks=$(jq -r --argjson labels "$LABELS" '
|
|
175
179
|
# Normalize both tracker shapes: phase-tracker.sh writes phases as an
|
|
176
180
|
# array of {id,...}; older fixtures keyed an object by phase id. The
|
|
177
181
|
# string-index below errors on an array and 2>/dev/null would swallow
|
|
@@ -179,8 +183,7 @@ if [ -n "$TRACKER_FILE" ]; then
|
|
|
179
183
|
(if (.phases | type) == "array"
|
|
180
184
|
then (.phases | map({key: (.id | tostring), value: .}) | from_entries)
|
|
181
185
|
else (.phases // {}) end) as $ph |
|
|
182
|
-
[
|
|
183
|
-
[range(0;8) | tostring] as $ids |
|
|
186
|
+
[range(0; ($labels | length)) | tostring] as $ids |
|
|
184
187
|
$ids
|
|
185
188
|
| map(. as $i |
|
|
186
189
|
($labels[$i | tonumber]) as $label |
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schemaVersion": "1.0.0",
|
|
3
|
-
"generatedAt": "2026-09-
|
|
4
|
-
"skillCount":
|
|
3
|
+
"generatedAt": "2026-09-21T16:14:01Z",
|
|
4
|
+
"skillCount": 215,
|
|
5
5
|
"entries": [
|
|
6
6
|
{
|
|
7
7
|
"path": "shared/core/apple-archive-compliance/SKILL.md",
|
|
@@ -177,12 +177,16 @@
|
|
|
177
177
|
},
|
|
178
178
|
{
|
|
179
179
|
"path": "shared/core/multi-agent-scan/SKILL.md",
|
|
180
|
-
"sha256": "
|
|
180
|
+
"sha256": "71cdbf40a40edf80b61985ea4cd6b7d43511e91ad3d7c4121512134a9e94f2a8"
|
|
181
181
|
},
|
|
182
182
|
{
|
|
183
183
|
"path": "shared/core/multi-agent-search/SKILL.md",
|
|
184
184
|
"sha256": "32541ceca20c489814f9d948763d5237d4a71473a05683d9da262b8f6dea815b"
|
|
185
185
|
},
|
|
186
|
+
{
|
|
187
|
+
"path": "shared/core/multi-agent-security-review/SKILL.md",
|
|
188
|
+
"sha256": "a3d4d1e99a8be67f179b08d79ee421108f3e62c8d4190cf4e9af7dd4391a8c11"
|
|
189
|
+
},
|
|
186
190
|
{
|
|
187
191
|
"path": "shared/core/multi-agent-setup/SKILL.md",
|
|
188
192
|
"sha256": "d2c167082f40af3c2a7cbb0be55a8cdf8076a4f06a3df54c89b1318759421a29"
|
|
@@ -205,7 +209,7 @@
|
|
|
205
209
|
},
|
|
206
210
|
{
|
|
207
211
|
"path": "shared/core/multi-agent-sync/SKILL.md",
|
|
208
|
-
"sha256": "
|
|
212
|
+
"sha256": "70b81f275fa430eb7382d0fa16f0a07ec29d92d9476655c496ce38565833fa13"
|
|
209
213
|
},
|
|
210
214
|
{
|
|
211
215
|
"path": "shared/core/multi-agent-test-accessibility/SKILL.md",
|
|
@@ -241,7 +245,7 @@
|
|
|
241
245
|
},
|
|
242
246
|
{
|
|
243
247
|
"path": "shared/core/multi-agent/SKILL.md",
|
|
244
|
-
"sha256": "
|
|
248
|
+
"sha256": "7a8a1e51d7e1f9e6cf00e09a5b1d680db4609d79cb629ab4b80e7a27bee5ef5e"
|
|
245
249
|
},
|
|
246
250
|
{
|
|
247
251
|
"path": "shared/external/accessibility-compliance-accessibility-audit/SKILL.md",
|
|
@@ -651,6 +655,10 @@
|
|
|
651
655
|
"path": "shared/external/search-first/SKILL.md",
|
|
652
656
|
"sha256": "4730d95babebecb3070a23e09cfce57f687467e0d45bbc332ae612cf01066e10"
|
|
653
657
|
},
|
|
658
|
+
{
|
|
659
|
+
"path": "shared/external/security-review/SKILL.md",
|
|
660
|
+
"sha256": "6bf82c336b7067d3e8a874e09916fcdbad4867634881f0a58d721b78fc9e83b2"
|
|
661
|
+
},
|
|
654
662
|
{
|
|
655
663
|
"path": "shared/external/shareplay-activities/SKILL.md",
|
|
656
664
|
"sha256": "508e3123039bc114e4239e66d86f8d550fe94fc259d827e4985e0aed3ff54090"
|
|
@@ -422,9 +422,8 @@ Two halves, one phase, one approval gate (Phases 1 and 2 until 19.0.0).
|
|
|
422
422
|
|
|
423
423
|
**1a - Analysis** (claude-sonnet-5, `explorer` persona)
|
|
424
424
|
1. Launch **explore agents** (parallel) to scan the codebase: related files, existing patterns and conventions, potential impact areas
|
|
425
|
-
2.
|
|
426
|
-
3.
|
|
427
|
-
4. Log: `📊 Phase 1: Plan (analysis) - {N} files identified, {summary}`
|
|
425
|
+
2. Summarize findings
|
|
426
|
+
3. Log: `📊 Phase 1: Plan (analysis) - {N} files identified, {summary}`
|
|
428
427
|
|
|
429
428
|
**1b - Planning** (claude-fable-5)
|
|
430
429
|
1. Create task breakdown → todos with dependencies
|
|
@@ -633,7 +632,7 @@ Sub-agents have embedded skills in their `.agent.md` files. Each agent is pre-lo
|
|
|
633
632
|
|-------|----------------|----------------|
|
|
634
633
|
| **code-reviewer** | Code Review Excellence, Security Audit, Performance Review, TDD Verification, Error Detective | Phase 3: Review |
|
|
635
634
|
| **ios-architect** | Spec-Driven Architecture, Context Management, Multi-Agent Coordination, API Design, Performance Architecture, Design System Governance | Phase 1: Plan |
|
|
636
|
-
| **security-auditor** | iOS Security Deep Dive, App Store Review Gates, Third-Party SDK Risk, Error Detective (Security) | Phase
|
|
635
|
+
| **security-auditor** | iOS Security Deep Dive, App Store Review Gates, Third-Party SDK Risk, Error Detective (Security) | Phase 3: Review |
|
|
637
636
|
|
|
638
637
|
### Phase → Agent → Skills Flow
|
|
639
638
|
|
|
@@ -3,7 +3,7 @@ name: multi-agent-scan
|
|
|
3
3
|
language: en
|
|
4
4
|
description: "Skill security scan: walks local skill directories against a tiered pattern catalog. Use when local skill directories need checking for unsafe or unexpected content."
|
|
5
5
|
user-invocable: true
|
|
6
|
-
argument-hint: "[--strict] [--
|
|
6
|
+
argument-hint: "[--strict] [--root PATH] - optional: --strict enables strict exit codes, --root picks a custom directory"
|
|
7
7
|
---
|
|
8
8
|
|
|
9
9
|
# multi-agent-scan
|
|
@@ -55,7 +55,7 @@ multi-agent-scan --root ~/.copilot/skills
|
|
|
55
55
|
## Integration
|
|
56
56
|
|
|
57
57
|
- **install.js** pre-deploy hook (automatic, warn-only high-threshold)
|
|
58
|
-
- **
|
|
58
|
+
- **Smoke suite** `smoke-skill-scan.sh` runs it in strict mode under `npm test`
|
|
59
59
|
- **Standalone** this skill
|
|
60
60
|
|
|
61
61
|
Cross-CLI parity: `/multi-agent:scan` on Claude Code, `multi-agent-scan` on Copilot CLI. Same source script, same behavior.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: multi-agent-security-review
|
|
3
|
+
language: en
|
|
4
|
+
description: "Run a standalone defensive, static security review of a diff, branch or repo: run-scoped threat model, reviewer-shaped findings joined to OWASP + CWE with CVSS scoring, evidence and before/after fixes, plus an offline dependency inventory. No live target, no payloads. Use when reviewing code for security outside a full pipeline run, auditing dependencies, or preparing a branch for a security sign-off."
|
|
5
|
+
user-invocable: true
|
|
6
|
+
argument-hint: "[#N | repo#N | PR-URL | branch | path] - a PR, a local branch, or a path to scope the review. If omitted: the current branch diff against its base."
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
# multi-agent security-review - standalone defensive static review
|
|
10
|
+
|
|
11
|
+
**Input**: `$ARGUMENTS`
|
|
12
|
+
|
|
13
|
+
The same security audit Phase 3 runs at Step 2.7, invoked on its own. Defensive and static: reads code, config and dependency manifests; never runs the target, fires a payload, or reaches a live host.
|
|
14
|
+
|
|
15
|
+
## Scope
|
|
16
|
+
|
|
17
|
+
Resolve from `$ARGUMENTS`, same shapes as `/multi-agent:review`: `#N` / `repo#N` / PR URL → that PR's diff; a branch → its diff against base; a path → files under it; omitted → the current branch diff. Cap the diff as Phase 3 Step 1.9 does.
|
|
18
|
+
|
|
19
|
+
## Steps
|
|
20
|
+
|
|
21
|
+
1. **Threat model** - produce `.pipeline/threat-model.md` (four sections) if absent, else read it; mirror to `state.threatModel`. Contract: `threat-model.md`.
|
|
22
|
+
2. **Method** - load `ai-common-toolkit:security-review` for the OWASP walk (Web/API Top 10 2021, Mobile Top 10 2024) + CWE join. This command orchestrates, it does not re-derive the method.
|
|
23
|
+
3. **Findings** - dispatch the `security-auditor`; it returns a `reviewer-output.schema.json` object whose `findings[]` carry the `security` envelope (`security-finding.schema.json`). Score every vector with the toolkit `security_cvss_score`; validate with `validate-reviewer.mjs` (one rework then halt).
|
|
24
|
+
4. **Dependencies** - run `security_dep_inventory` on the lockfiles; hand the inventory to `ai-analyst-toolkit:evidence-registry` for known CVEs when it is registered. A vulnerable dependency is an `A06:2021` finding with the advisory CWE + CVE. This command contacts nothing itself.
|
|
25
|
+
5. **Report** - write `.pipeline/security-findings.json` and a human summary (count by severity; each blocking/important finding with OWASP id, CWE, CVSS band, evidence, before/after fix). An empty list with `approved: true` is a clean review.
|
|
26
|
+
|
|
27
|
+
## Boundaries
|
|
28
|
+
|
|
29
|
+
Inside a run the same audit is Phase 3 Step 2.7 (`features/security-audit.md`), triggered by the `security_path` signal; this is the standalone entry point. Not the `store-ready` device pass, not a secret scanner (`pre-commit-check.sh` covers secrets). No writes to Jira / GitHub / Confluence. Autopilot reviews the current branch, writes the report, opens no PR.
|
|
@@ -32,7 +32,7 @@ Run all steps automatically:
|
|
|
32
32
|
```
|
|
33
33
|
Step 0: DOCTOR node $HOME/.claude/scripts/doctor.mjs - exit 2 or 4 STOPS the sync
|
|
34
34
|
Step 1: DETECT Compare timestamps, find stale targets
|
|
35
|
-
Step 2: COPILOT Claude Code -> Copilot CLI (instructions +
|
|
35
|
+
Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 58 sub-command skills)
|
|
36
36
|
Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 57 specs as refs + 8 agent TOML)
|
|
37
37
|
Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub)
|
|
38
38
|
Step 3d: DEV-TOOLKIT Companion MCP server -> detect movement, ship gates, commit + publish
|
|
@@ -228,7 +228,7 @@ When invoked with the `release` argument:
|
|
|
228
228
|
|-------------|-------------|
|
|
229
229
|
| `~/.claude/commands/multi-agent/{cmd}/SKILL.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
|
|
230
230
|
|
|
231
|
-
**
|
|
231
|
+
**58 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
|
|
232
232
|
|
|
233
233
|
```
|
|
234
234
|
analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off,
|
|
@@ -239,7 +239,7 @@ language, log, manual-test, model, prune-logs,
|
|
|
239
239
|
prune-prompts, purge, refactor, resume, review,
|
|
240
240
|
review-analysis, review-issue, review-jira, route-off, route-on,
|
|
241
241
|
route-status, routines, save, scan, search,
|
|
242
|
-
setup, stack, status, steer, store-ready, sync, test, test-accessibility,
|
|
242
|
+
security-review, setup, stack, status, steer, store-ready, sync, test, test-accessibility,
|
|
243
243
|
test-dark-mode, test-dynamic-type, test-screenshots, testflight-validation,
|
|
244
244
|
uninstall, update
|
|
245
245
|
```
|