@mmerterden/multi-agent-pipeline 19.0.0 → 19.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +121 -0
- package/README.md +3 -3
- package/docs/ecosystem.md +12 -4
- package/docs/facts.json +20 -4
- package/docs/features.md +1 -1
- package/docs/recovery-guide.md +8 -8
- package/manifest.json +64 -62
- package/package.json +1 -1
- package/pipeline/agents/dev-critic.md +4 -4
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/analysis/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
- package/pipeline/lib/model-dispatch.sh +140 -0
- package/pipeline/lib/outbound-gate.mjs +14 -0
- package/pipeline/multi-agent-refs/_dev-context.md +5 -5
- package/pipeline/multi-agent-refs/analysis/evidence.md +2 -2
- package/pipeline/multi-agent-refs/analysis/intake.md +6 -6
- package/pipeline/multi-agent-refs/analysis/locked.md +27 -0
- package/pipeline/multi-agent-refs/analysis/redesign.md +1 -1
- package/pipeline/multi-agent-refs/analysis/render.md +9 -9
- package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
- package/pipeline/multi-agent-refs/analysis/review.md +2 -2
- package/pipeline/multi-agent-refs/analysis/synthesis.md +2 -2
- package/pipeline/multi-agent-refs/analysis-template-corporate.md +9 -9
- package/pipeline/multi-agent-refs/analysis-template.md +19 -19
- package/pipeline/multi-agent-refs/component-dispatch.md +5 -5
- package/pipeline/multi-agent-refs/conventions-defaults.md +2 -2
- package/pipeline/multi-agent-refs/features/analysis-jira.md +1 -1
- package/pipeline/multi-agent-refs/features/doctor.md +1 -1
- package/pipeline/multi-agent-refs/features/model-fallback.md +36 -0
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
- package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +2 -2
- package/pipeline/multi-agent-refs/phases/phase-3-review.md +2 -2
- package/pipeline/preferences-template.json +5 -0
- package/pipeline/rules/figma-pipeline.md +8 -8
- package/pipeline/schemas/analysis-output.schema.json +1 -1
- package/pipeline/schemas/analysis-spec.schema.json +2 -2
- package/pipeline/schemas/figma-project-config.schema.json +1 -1
- package/pipeline/schemas/prefs.schema.json +2 -2
- package/pipeline/schemas/secret-patterns.json +124 -0
- package/pipeline/scripts/build-references.mjs +2 -2
- package/pipeline/scripts/bulk-read.sh +10 -1
- package/pipeline/scripts/cost-table.json +8 -1
- package/pipeline/scripts/doctor.mjs +1 -1
- package/pipeline/scripts/gen-facts.mjs +112 -7
- package/pipeline/scripts/phase-tracker.sh +5 -5
- package/pipeline/scripts/pre-commit-check.sh +30 -1
- package/pipeline/scripts/scan-skills.sh +26 -0
- package/pipeline/scripts/validate-analysis-doc.mjs +201 -26
- package/pipeline/scripts/verify-citations.mjs +1 -1
- package/pipeline/scripts/write-state.mjs +32 -0
- package/pipeline/skills/.skill-manifest.json +5 -5
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +6 -6
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +6 -6
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +14 -13
- package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +1 -1
- package/pipeline/skills/shared/external/signal-community/SKILL.md +8 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
|
|
2
|
+
"_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. `provider` says which layer a rung belongs to: an `anthropic` rung is reachable from every call site, a non-anthropic one only from the call sites the pipeline makes itself (bulk-read, research), which is what model-dispatch.sh enforces. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
|
|
3
3
|
"schemaVersion": "1.1.0",
|
|
4
4
|
"prices": {
|
|
5
5
|
"fable": {
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
"outPerMtok": 50.0,
|
|
8
8
|
"cacheReadPerMtok": 1.0,
|
|
9
9
|
"modelId": "claude-fable-5",
|
|
10
|
+
"provider": "anthropic",
|
|
10
11
|
"note": "Top tier (restored v10.6.0) - architects, Reviewer 1, triage. Verified against Anthropic pricing 2026-07-02."
|
|
11
12
|
},
|
|
12
13
|
"opus": {
|
|
@@ -14,6 +15,7 @@
|
|
|
14
15
|
"outPerMtok": 25.0,
|
|
15
16
|
"cacheReadPerMtok": 0.5,
|
|
16
17
|
"modelId": "claude-opus-5",
|
|
18
|
+
"provider": "anthropic",
|
|
17
19
|
"note": "Second tier - dev phase on a Short run, Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
|
|
18
20
|
},
|
|
19
21
|
"sonnet": {
|
|
@@ -21,6 +23,7 @@
|
|
|
21
23
|
"outPerMtok": 15.0,
|
|
22
24
|
"cacheReadPerMtok": 0.3,
|
|
23
25
|
"modelId": "claude-sonnet-5",
|
|
26
|
+
"provider": "anthropic",
|
|
24
27
|
"note": "Floor tier for Claude-family dispatch - Reviewer 3 on both hosts, and the terminal rung of the fallback ladder. Priced at the standard 3/15 rather than the 2/10 introductory rate that runs through 2026-08-31: over-reporting during the intro window is the safe direction for a cost ledger, and it needs no dated edit when the intro ends."
|
|
25
28
|
},
|
|
26
29
|
"haiku": {
|
|
@@ -28,6 +31,7 @@
|
|
|
28
31
|
"outPerMtok": 5.0,
|
|
29
32
|
"cacheReadPerMtok": 0.1,
|
|
30
33
|
"modelId": "claude-haiku-4-5",
|
|
34
|
+
"provider": "anthropic",
|
|
31
35
|
"note": "Speed tier - task-clarifier and other latency-sensitive dispatches, plus the terminal rung of the fallback ladder. Named by its alias like every other rung here rather than by a dated snapshot, so the four rungs stay comparable at a glance."
|
|
32
36
|
},
|
|
33
37
|
"gpt-5.4": {
|
|
@@ -35,6 +39,7 @@
|
|
|
35
39
|
"outPerMtok": 30.0,
|
|
36
40
|
"cacheReadPerMtok": 1.0,
|
|
37
41
|
"modelId": "gpt-5.4",
|
|
42
|
+
"provider": "openai",
|
|
38
43
|
"note": "Copilot CLI Reviewer 2 and Codex CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
|
|
39
44
|
},
|
|
40
45
|
"gpt-5.6": {
|
|
@@ -42,6 +47,7 @@
|
|
|
42
47
|
"outPerMtok": 30.0,
|
|
43
48
|
"cacheReadPerMtok": 1.0,
|
|
44
49
|
"modelId": "gpt-5.6",
|
|
50
|
+
"provider": "openai",
|
|
45
51
|
"note": "Codex CLI top tier - Reviewer 1 at xhigh effort, Reviewer 3 at medium, triage at max, and the fable/opus rungs of the Codex persona tier map. Reasoning effort changes output volume, not the per-token rate, so one entry covers every effort level. Approximate; verify against OpenAI pricing before relying on totals."
|
|
46
52
|
},
|
|
47
53
|
"gpt-5.6-terra": {
|
|
@@ -49,6 +55,7 @@
|
|
|
49
55
|
"outPerMtok": 5.0,
|
|
50
56
|
"cacheReadPerMtok": 0.1,
|
|
51
57
|
"modelId": "gpt-5.6-terra",
|
|
58
|
+
"provider": "openai",
|
|
52
59
|
"note": "Codex CLI floor tier - speed-optimised, maps the haiku rung of the Codex persona tier map (task-clarifier). Approximate; verify against OpenAI pricing before relying on totals."
|
|
53
60
|
}
|
|
54
61
|
}
|
|
@@ -658,7 +658,7 @@ function checkMcpRegistration() {
|
|
|
658
658
|
// checkMcpRegistration answers "is OURS registered". This answers the question the
|
|
659
659
|
// user never gets asked: every registered server's tool list is sent with every
|
|
660
660
|
// turn, they are added one at a time, and nobody sees the running total - our own
|
|
661
|
-
// toolkit is
|
|
661
|
+
// toolkit is 115 tools by itself. This check only ever REPORTS. It never disables
|
|
662
662
|
// anything, and it never blocks or warns, because how many servers are worth their
|
|
663
663
|
// context is the user's call and not a health failure.
|
|
664
664
|
//
|
|
@@ -2,17 +2,21 @@
|
|
|
2
2
|
/**
|
|
3
3
|
* @file gen-facts.mjs - the numbers the website is allowed to state, derived.
|
|
4
4
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* the fix is the same shape: one producer, everything else reads it.
|
|
5
|
+
* A consumer that keeps its own copy of these numbers has nothing guarding it:
|
|
6
|
+
* a count typed into a page or a README is right on the day it is typed and
|
|
7
|
+
* silent afterwards. `smoke-phase-contract.sh` enforces the same shape inside
|
|
8
|
+
* the pipeline - one producer, everything else reads it.
|
|
10
9
|
*
|
|
11
10
|
* Source of truth per field:
|
|
12
11
|
* phases pipeline/schemas/phases.json - the phase contract itself
|
|
13
12
|
* commandCount a count of pipeline/commands/multi-agent/<name>/SKILL.md
|
|
14
13
|
* skillCount a count of pipeline/skills/shared/external/<name>/SKILL.md
|
|
14
|
+
* skillCountAll every SKILL.md under pipeline/skills, at any depth
|
|
15
|
+
* agentCount a count of pipeline/agents/*.md
|
|
16
|
+
* hosts the CLIs install/ has an adapter for
|
|
15
17
|
* toolCount the toolkit's own tools/list response, when reachable
|
|
18
|
+
* toolCategories that same response, grouped by the prefix in each tool name
|
|
19
|
+
* pluginSkillCount every SKILL.md in the plugin marketplace, when checked out
|
|
16
20
|
* version package.json
|
|
17
21
|
*
|
|
18
22
|
* Counts come from the filesystem at the moment of writing, never from a README
|
|
@@ -43,6 +47,24 @@ function die(msg) {
|
|
|
43
47
|
process.exit(2);
|
|
44
48
|
}
|
|
45
49
|
|
|
50
|
+
// Every SKILL.md under a tree, at any depth. `countSkillDirs` only sees the
|
|
51
|
+
// immediate children of one directory, which is the right shape for the
|
|
52
|
+
// external set and the wrong one for the whole tree: the shared/core and
|
|
53
|
+
// per-host mirrors sit a level deeper and were invisible to it.
|
|
54
|
+
function countSkillsDeep(rel) {
|
|
55
|
+
const dir = join(ROOT, rel);
|
|
56
|
+
if (!existsSync(dir)) die(`missing directory: ${rel}`);
|
|
57
|
+
let n = 0;
|
|
58
|
+
const walk = (d) => {
|
|
59
|
+
for (const e of readdirSync(d, { withFileTypes: true })) {
|
|
60
|
+
if (e.isDirectory()) walk(join(d, e.name));
|
|
61
|
+
else if (e.name === "SKILL.md") n += 1;
|
|
62
|
+
}
|
|
63
|
+
};
|
|
64
|
+
walk(dir);
|
|
65
|
+
return n;
|
|
66
|
+
}
|
|
67
|
+
|
|
46
68
|
function countSkillDirs(rel) {
|
|
47
69
|
const dir = join(ROOT, rel);
|
|
48
70
|
if (!existsSync(dir)) die(`missing directory: ${rel}`);
|
|
@@ -70,6 +92,7 @@ try {
|
|
|
70
92
|
// tool count is the problem this file exists to remove, and a made-up one is
|
|
71
93
|
// worse than none.
|
|
72
94
|
let toolCount = null;
|
|
95
|
+
let toolCategories = null;
|
|
73
96
|
let toolkitVersion = null;
|
|
74
97
|
const toolkitDir = join(ROOT, "..", "multi-agent-toolkit-mcp");
|
|
75
98
|
if (existsSync(join(toolkitDir, "package.json"))) {
|
|
@@ -113,10 +136,48 @@ if (existsSync(join(toolkitDir, "package.json"))) {
|
|
|
113
136
|
} catch {
|
|
114
137
|
continue;
|
|
115
138
|
}
|
|
116
|
-
if (msg.id === 2 && Array.isArray(msg.result?.tools))
|
|
139
|
+
if (msg.id === 2 && Array.isArray(msg.result?.tools)) {
|
|
140
|
+
toolCount = msg.result.tools.length;
|
|
141
|
+
// Grouped by the prefix every tool name carries, from the same
|
|
142
|
+
// response the count comes from. Counting families out of the source
|
|
143
|
+
// gives a different and wrong answer for the reason above, and the
|
|
144
|
+
// site shows the per-family breakdown next to the total, so the two
|
|
145
|
+
// have to come from one place or they will disagree.
|
|
146
|
+
const byFamily = {};
|
|
147
|
+
for (const t of msg.result.tools) {
|
|
148
|
+
const family = String(t.name || "").split("_")[0];
|
|
149
|
+
if (family) byFamily[family] = (byFamily[family] || 0) + 1;
|
|
150
|
+
}
|
|
151
|
+
toolCategories = Object.fromEntries(
|
|
152
|
+
Object.entries(byFamily).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])),
|
|
153
|
+
);
|
|
154
|
+
}
|
|
117
155
|
}
|
|
118
156
|
} catch {
|
|
119
157
|
toolCount = null;
|
|
158
|
+
toolCategories = null;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// The plugin marketplace is a third repo, probed the same way and reported as
|
|
163
|
+
// null when it is not checked out beside this one. The site states this number
|
|
164
|
+
// next to the pipeline's own skill counts, and three different skill
|
|
165
|
+
// populations on one page is exactly the shape that drifts.
|
|
166
|
+
let pluginSkillCount = null;
|
|
167
|
+
const pluginsDir = join(ROOT, "..", "multi-agent-plugins", "plugins");
|
|
168
|
+
if (existsSync(pluginsDir)) {
|
|
169
|
+
let n = 0;
|
|
170
|
+
const walk = (d) => {
|
|
171
|
+
for (const e of readdirSync(d, { withFileTypes: true })) {
|
|
172
|
+
if (e.isDirectory()) walk(join(d, e.name));
|
|
173
|
+
else if (e.name === "SKILL.md") n += 1;
|
|
174
|
+
}
|
|
175
|
+
};
|
|
176
|
+
try {
|
|
177
|
+
walk(pluginsDir);
|
|
178
|
+
pluginSkillCount = n;
|
|
179
|
+
} catch {
|
|
180
|
+
pluginSkillCount = null;
|
|
120
181
|
}
|
|
121
182
|
}
|
|
122
183
|
|
|
@@ -130,12 +191,56 @@ const facts = {
|
|
|
130
191
|
phaseCount: contract.phases.length,
|
|
131
192
|
modes: Object.fromEntries(Object.entries(contract.modes).map(([k, v]) => [k, v.phases])),
|
|
132
193
|
commandCount: countSkillDirs("pipeline/commands/multi-agent"),
|
|
194
|
+
// Two populations, and conflating them is how the site came to show a stale
|
|
195
|
+
// "212 Skills" chip next to a correct "216 skills" sentence. `skillCount` is
|
|
196
|
+
// the external set a user installs; `skillCountAll` is every SKILL.md the
|
|
197
|
+
// repo ships, mirrors included. Both are stated on the site, so both are
|
|
198
|
+
// generated here rather than counted by hand at the other end.
|
|
133
199
|
skillCount: countSkillDirs("pipeline/skills/shared/external"),
|
|
200
|
+
skillCountAll: countSkillsDeep("pipeline/skills"),
|
|
201
|
+
// The site shows both of these as headline figures and had typed both by
|
|
202
|
+
// hand: a "9 Sub-agents" that happened to be right and a host list that had
|
|
203
|
+
// gone stale, naming Claude Code and Copilot CLI while install/codex.mjs had
|
|
204
|
+
// been shipping a third target for releases.
|
|
205
|
+
agentCount: readdirSync(join(ROOT, "pipeline", "agents")).filter((f) => f.endsWith(".md")).length,
|
|
206
|
+
hosts: readdirSync(join(ROOT, "install"))
|
|
207
|
+
.filter((f) => f.endsWith(".mjs") && !f.startsWith("_") && f !== "index.mjs")
|
|
208
|
+
.map(
|
|
209
|
+
(f) =>
|
|
210
|
+
({ claude: "Claude Code", copilot: "Copilot CLI", codex: "Codex CLI" })[
|
|
211
|
+
f.replace(".mjs", "")
|
|
212
|
+
],
|
|
213
|
+
)
|
|
214
|
+
.filter(Boolean)
|
|
215
|
+
.sort(),
|
|
134
216
|
toolCount,
|
|
217
|
+
toolCategories,
|
|
218
|
+
pluginSkillCount,
|
|
135
219
|
toolkitVersion,
|
|
136
220
|
};
|
|
137
221
|
|
|
138
|
-
|
|
222
|
+
// JSON.stringify and prettier disagree about short arrays: stringify puts each
|
|
223
|
+
// phase id on its own line, prettier collapses a run that fits. That left this
|
|
224
|
+
// file failing `format:check` after every regeneration, held together only by
|
|
225
|
+
// someone remembering to run prettier by hand - and a generated file that needs
|
|
226
|
+
// a manual step to pass the gate is a generated file that will fail the gate.
|
|
227
|
+
//
|
|
228
|
+
// Importing prettier here is not the fix: this script ships in the published
|
|
229
|
+
// package, where a devDependency is not installed (`n/no-unpublished-import`
|
|
230
|
+
// says so). Collapsing the number runs directly costs nothing and leaves the
|
|
231
|
+
// script dependency-free.
|
|
232
|
+
const collapseShortArrays = (json) =>
|
|
233
|
+
json.replace(
|
|
234
|
+
/\[\n\s+((?:-?\d+|"[^"\n]*")(?:,\n\s+(?:-?\d+|"[^"\n]*"))*)\n\s+\]/g,
|
|
235
|
+
(whole, inner) => {
|
|
236
|
+
const one = `[${inner.split(/,\n\s*/).join(", ")}]`;
|
|
237
|
+
// Prettier only collapses what fits the print width; leave the rest alone
|
|
238
|
+
// rather than guess, so the two can never disagree in the other direction.
|
|
239
|
+
return one.length <= 72 ? one : whole;
|
|
240
|
+
},
|
|
241
|
+
);
|
|
242
|
+
|
|
243
|
+
const body = collapseShortArrays(`${JSON.stringify(facts, null, 2)}\n`);
|
|
139
244
|
|
|
140
245
|
if (process.argv.includes("--stdout")) {
|
|
141
246
|
process.stdout.write(body);
|
|
@@ -613,7 +613,7 @@ tracker_next_hint() {
|
|
|
613
613
|
# default only up to Opus 4.7 / Sonnet 4.6, a default that landed in
|
|
614
614
|
# v2.1.268. Naming the fallback on the same line is what keeps a newer model
|
|
615
615
|
# from advancing six phases in silence.
|
|
616
|
-
# `subjects` now carries Phase
|
|
616
|
+
# `subjects` now carries Phase 1's plan steps as indented rows, and
|
|
617
617
|
# update_plan takes the whole list anyway, so re-reading it is what puts
|
|
618
618
|
# those steps on the Codex plan without a second mechanism. Codex has no
|
|
619
619
|
# dependency concept; the "(bekliyor: ...)" suffix inside the step text is
|
|
@@ -776,9 +776,9 @@ subjects() {
|
|
|
776
776
|
# Sub-phases ride out on the SAME list, indented in the subject string.
|
|
777
777
|
#
|
|
778
778
|
# The card has drawn these since sub-phases existed; the widget never has,
|
|
779
|
-
# and the widget is the surface the user actually looks at. Phase
|
|
779
|
+
# and the widget is the surface the user actually looks at. Phase 1's plan
|
|
780
780
|
# is the case that made the gap matter: the tasks, their order and their
|
|
781
|
-
# dependencies are computed, stored and used to drive Phase
|
|
781
|
+
# dependencies are computed, stored and used to drive Phase 2's picker, and
|
|
782
782
|
# none of it was visible anywhere the user was looking.
|
|
783
783
|
#
|
|
784
784
|
# Indentation is two spaces INSIDE the subject because the widget takes
|
|
@@ -1214,7 +1214,7 @@ GATE
|
|
|
1214
1214
|
;;
|
|
1215
1215
|
|
|
1216
1216
|
plan)
|
|
1217
|
-
# Phase
|
|
1217
|
+
# Phase 1's plan, turned into sub-phases of the phase that will execute it.
|
|
1218
1218
|
#
|
|
1219
1219
|
# The parsing lives here rather than in the phase document for one reason:
|
|
1220
1220
|
# sub-phases are this file's structure, and a jq blob in a phase doc is a
|
|
@@ -1222,7 +1222,7 @@ GATE
|
|
|
1222
1222
|
# keeps the doc to one line, which the aggregate phase-doc budget cares about.
|
|
1223
1223
|
#
|
|
1224
1224
|
# Reads planning-output.schema.json on stdin: tasks[] with id, title and an
|
|
1225
|
-
# optional dependsOn[]. Status is `pending` for all of them - Phase
|
|
1225
|
+
# optional dependsOn[]. Status is `pending` for all of them - Phase 2 moves
|
|
1226
1226
|
# them with `sub`, and pre-marking work as started is the lie the tracker
|
|
1227
1227
|
# exists to avoid.
|
|
1228
1228
|
need_jq
|
|
@@ -148,12 +148,41 @@ scan_file() {
|
|
|
148
148
|
FOUND=1
|
|
149
149
|
fi
|
|
150
150
|
|
|
151
|
-
# High-signal provider token prefixes (low false-positive rate)
|
|
151
|
+
# High-signal provider token prefixes (low false-positive rate).
|
|
152
|
+
#
|
|
153
|
+
# This set and the one in lib/outbound-gate.mjs cover the same providers, and
|
|
154
|
+
# smoke-secret-parity.sh holds them to it by running a fake token of each
|
|
155
|
+
# shape through BOTH.
|
|
152
156
|
if echo "$content" | grep -qE '(ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{36}|github_pat_[A-Za-z0-9_]{60,}|xox[baprs]-[A-Za-z0-9-]{12,}|sk_live_[A-Za-z0-9]{20,}|rk_live_[A-Za-z0-9]{20,}|AIza[0-9A-Za-z_-]{35}|npm_[A-Za-z0-9]{36}|glpat-[A-Za-z0-9_-]{20,}'; then
|
|
153
157
|
echo "BLOCKED: Provider access token in $file" >&2
|
|
154
158
|
FOUND=1
|
|
155
159
|
fi
|
|
156
160
|
|
|
161
|
+
# Model-provider and ML-hub keys. `sk-ant-` and `sk-proj-` are checked before
|
|
162
|
+
# the bare `sk-` form so the message names the provider a reader can revoke.
|
|
163
|
+
if echo "$content" | grep -qE 'sk-ant-[A-Za-z0-9_-]{32,}|sk-proj-[A-Za-z0-9_-]{32,}|pplx-[A-Za-z0-9]{32,}|hf_[A-Za-z0-9]{30,}'; then
|
|
164
|
+
echo "BLOCKED: Model-provider API key in $file" >&2
|
|
165
|
+
FOUND=1
|
|
166
|
+
fi
|
|
167
|
+
|
|
168
|
+
# Figma personal access token (figd_) and the MCP OAuth token (figu_). Same
|
|
169
|
+
# shape, same blast radius: both read every file the account can reach.
|
|
170
|
+
if echo "$content" | grep -qE 'fig[a-z]_[A-Za-z0-9_-]{20,}'; then
|
|
171
|
+
echo "BLOCKED: Figma token in $file" >&2
|
|
172
|
+
FOUND=1
|
|
173
|
+
fi
|
|
174
|
+
|
|
175
|
+
# A URL carrying its own credentials, which is how a `git remote -v` paste
|
|
176
|
+
# leaks, and an Authorization header copied out of a curl trace.
|
|
177
|
+
if echo "$content" | grep -qE '[a-z][a-z0-9+.-]*://[^[:space:]/:@]+:[^[:space:]/@]+@'; then
|
|
178
|
+
echo "BLOCKED: URL with embedded credentials in $file" >&2
|
|
179
|
+
FOUND=1
|
|
180
|
+
fi
|
|
181
|
+
if echo "$content" | grep -qiE 'Authorization:[[:space:]]*(Bearer|Basic)[[:space:]]+[A-Za-z0-9._~+/=-]{16,}'; then
|
|
182
|
+
echo "BLOCKED: Authorization header with a token in $file" >&2
|
|
183
|
+
FOUND=1
|
|
184
|
+
fi
|
|
185
|
+
|
|
157
186
|
# JWT (three base64url segments - header.payload.signature)
|
|
158
187
|
if echo "$content" | grep -qE 'eyJ[A-Za-z0-9_-]{10,}\.eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}'; then
|
|
159
188
|
echo "BLOCKED: JWT in $file" >&2
|
|
@@ -123,6 +123,14 @@ tree_grep() {
|
|
|
123
123
|
return 0
|
|
124
124
|
}
|
|
125
125
|
|
|
126
|
+
# Case-insensitive variant. Prose written to steer a model is written by hand,
|
|
127
|
+
# so its capitalisation is whatever the author felt like, and a case-sensitive
|
|
128
|
+
# pattern would miss "Ignore all previous instructions" by one letter.
|
|
129
|
+
tree_grep_i() {
|
|
130
|
+
tr '\n' '\0' < "$SCAN_LIST" | xargs -0 grep -nHEi -- "$1" 2>/dev/null
|
|
131
|
+
return 0
|
|
132
|
+
}
|
|
133
|
+
|
|
126
134
|
# stdin: "path:line:content" grep hits -> stdout: "fileIdx|path|line|content"
|
|
127
135
|
index_hits() {
|
|
128
136
|
awk -v listfile="$SCAN_LIST" '
|
|
@@ -260,6 +268,24 @@ if [ "$THRESHOLD_RANK" -ge 1 ]; then
|
|
|
260
268
|
seq=$((seq+1))
|
|
261
269
|
emit_raw "$HIT_IDX" 9 "$seq" high "$HIT_FILE" "$HIT_LINE" "chmod-then-exec" "script made executable and immediately invoked"
|
|
262
270
|
done < <(tree_grep 'chmod[[:space:]]+\+x[[:space:]]+[^&;]+[[:space:]]*(&&|;)[[:space:]]*\./' | index_hits)
|
|
271
|
+
|
|
272
|
+
# Prompt injection (OWASP LLM01). The other families ask what a skill makes
|
|
273
|
+
# the MACHINE do; this one asks what it makes the MODEL do. A skill is
|
|
274
|
+
# instructions loaded straight into the context that decides everything after
|
|
275
|
+
# it, so text telling the model to drop its instructions, recite its prompt,
|
|
276
|
+
# or act behind the user's back is an attack delivered as prose - and no
|
|
277
|
+
# pattern above can see it, because nothing is executed.
|
|
278
|
+
#
|
|
279
|
+
# `high`, not `critical`: these phrasings can appear in a skill that DESCRIBES
|
|
280
|
+
# the attack, this scanner's own documentation being the obvious case, so a
|
|
281
|
+
# hit is a line a human reads rather than a verdict.
|
|
282
|
+
seq=0
|
|
283
|
+
while IFS= read -r hit; do
|
|
284
|
+
[ -z "$hit" ] && continue
|
|
285
|
+
parse_hit "$hit"
|
|
286
|
+
seq=$((seq+1))
|
|
287
|
+
emit_raw "$HIT_IDX" 13 "$seq" high "$HIT_FILE" "$HIT_LINE" "prompt-injection" "text instructing the model to override, disclose or hide - OWASP LLM01"
|
|
288
|
+
done < <(tree_grep_i 'ignore (all )?(the )?(previous|prior|above|earlier) (instructions|prompts?|rules?)|disregard (your|the|any) (system prompt|previous instructions|instructions)|(reveal|print|output|repeat|show) (your|the) (system prompt|full instructions)|(do not|don'"'"'t|never) (tell|inform) the (user|operator)|without (telling|informing|asking) the (user|operator)|(always|automatically) (approve|confirm) [^.]{0,30}(without|regardless)|you are now (a|an|the) ' | index_hits)
|
|
263
289
|
fi
|
|
264
290
|
|
|
265
291
|
# --- medium families (rank 2) ---------------------------------------------
|