@mmerterden/multi-agent-pipeline 19.0.0 → 19.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +121 -0
  2. package/README.md +3 -3
  3. package/docs/ecosystem.md +12 -4
  4. package/docs/facts.json +20 -4
  5. package/docs/features.md +1 -1
  6. package/docs/recovery-guide.md +8 -8
  7. package/manifest.json +64 -62
  8. package/package.json +1 -1
  9. package/pipeline/agents/dev-critic.md +4 -4
  10. package/pipeline/commands/multi-agent/SKILL.md +1 -1
  11. package/pipeline/commands/multi-agent/analysis/SKILL.md +6 -6
  12. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -2
  13. package/pipeline/commands/multi-agent/resume-local/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
  15. package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
  16. package/pipeline/lib/model-dispatch.sh +140 -0
  17. package/pipeline/lib/outbound-gate.mjs +14 -0
  18. package/pipeline/multi-agent-refs/_dev-context.md +5 -5
  19. package/pipeline/multi-agent-refs/analysis/evidence.md +2 -2
  20. package/pipeline/multi-agent-refs/analysis/intake.md +6 -6
  21. package/pipeline/multi-agent-refs/analysis/locked.md +27 -0
  22. package/pipeline/multi-agent-refs/analysis/redesign.md +1 -1
  23. package/pipeline/multi-agent-refs/analysis/render.md +9 -9
  24. package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
  25. package/pipeline/multi-agent-refs/analysis/review.md +2 -2
  26. package/pipeline/multi-agent-refs/analysis/synthesis.md +2 -2
  27. package/pipeline/multi-agent-refs/analysis-template-corporate.md +9 -9
  28. package/pipeline/multi-agent-refs/analysis-template.md +19 -19
  29. package/pipeline/multi-agent-refs/component-dispatch.md +5 -5
  30. package/pipeline/multi-agent-refs/conventions-defaults.md +2 -2
  31. package/pipeline/multi-agent-refs/features/analysis-jira.md +1 -1
  32. package/pipeline/multi-agent-refs/features/doctor.md +1 -1
  33. package/pipeline/multi-agent-refs/features/model-fallback.md +36 -0
  34. package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
  35. package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
  36. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +2 -2
  37. package/pipeline/multi-agent-refs/phases/phase-3-review.md +2 -2
  38. package/pipeline/preferences-template.json +5 -0
  39. package/pipeline/rules/figma-pipeline.md +8 -8
  40. package/pipeline/schemas/analysis-output.schema.json +1 -1
  41. package/pipeline/schemas/analysis-spec.schema.json +2 -2
  42. package/pipeline/schemas/figma-project-config.schema.json +1 -1
  43. package/pipeline/schemas/prefs.schema.json +2 -2
  44. package/pipeline/schemas/secret-patterns.json +124 -0
  45. package/pipeline/scripts/build-references.mjs +2 -2
  46. package/pipeline/scripts/bulk-read.sh +10 -1
  47. package/pipeline/scripts/cost-table.json +8 -1
  48. package/pipeline/scripts/doctor.mjs +1 -1
  49. package/pipeline/scripts/gen-facts.mjs +112 -7
  50. package/pipeline/scripts/phase-tracker.sh +5 -5
  51. package/pipeline/scripts/pre-commit-check.sh +30 -1
  52. package/pipeline/scripts/scan-skills.sh +26 -0
  53. package/pipeline/scripts/validate-analysis-doc.mjs +201 -26
  54. package/pipeline/scripts/verify-citations.mjs +1 -1
  55. package/pipeline/scripts/write-state.mjs +32 -0
  56. package/pipeline/skills/.skill-manifest.json +5 -5
  57. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +6 -6
  58. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +6 -6
  59. package/pipeline/skills/shared/core/multi-agent/SKILL.md +14 -13
  60. package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +1 -1
  61. package/pipeline/skills/shared/external/signal-community/SKILL.md +8 -1
@@ -1,5 +1,5 @@
1
1
  {
2
- "_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
2
+ "_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. `provider` says which layer a rung belongs to: an `anthropic` rung is reachable from every call site, a non-anthropic one only from the call sites the pipeline makes itself (bulk-read, research), which is what model-dispatch.sh enforces. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
3
3
  "schemaVersion": "1.1.0",
4
4
  "prices": {
5
5
  "fable": {
@@ -7,6 +7,7 @@
7
7
  "outPerMtok": 50.0,
8
8
  "cacheReadPerMtok": 1.0,
9
9
  "modelId": "claude-fable-5",
10
+ "provider": "anthropic",
10
11
  "note": "Top tier (restored v10.6.0) - architects, Reviewer 1, triage. Verified against Anthropic pricing 2026-07-02."
11
12
  },
12
13
  "opus": {
@@ -14,6 +15,7 @@
14
15
  "outPerMtok": 25.0,
15
16
  "cacheReadPerMtok": 0.5,
16
17
  "modelId": "claude-opus-5",
18
+ "provider": "anthropic",
17
19
  "note": "Second tier - dev phase on a Short run, Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
18
20
  },
19
21
  "sonnet": {
@@ -21,6 +23,7 @@
21
23
  "outPerMtok": 15.0,
22
24
  "cacheReadPerMtok": 0.3,
23
25
  "modelId": "claude-sonnet-5",
26
+ "provider": "anthropic",
24
27
  "note": "Floor tier for Claude-family dispatch - Reviewer 3 on both hosts, and the terminal rung of the fallback ladder. Priced at the standard 3/15 rather than the 2/10 introductory rate that runs through 2026-08-31: over-reporting during the intro window is the safe direction for a cost ledger, and it needs no dated edit when the intro ends."
25
28
  },
26
29
  "haiku": {
@@ -28,6 +31,7 @@
28
31
  "outPerMtok": 5.0,
29
32
  "cacheReadPerMtok": 0.1,
30
33
  "modelId": "claude-haiku-4-5",
34
+ "provider": "anthropic",
31
35
  "note": "Speed tier - task-clarifier and other latency-sensitive dispatches, plus the terminal rung of the fallback ladder. Named by its alias like every other rung here rather than by a dated snapshot, so the four rungs stay comparable at a glance."
32
36
  },
33
37
  "gpt-5.4": {
@@ -35,6 +39,7 @@
35
39
  "outPerMtok": 30.0,
36
40
  "cacheReadPerMtok": 1.0,
37
41
  "modelId": "gpt-5.4",
42
+ "provider": "openai",
38
43
  "note": "Copilot CLI Reviewer 2 and Codex CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
39
44
  },
40
45
  "gpt-5.6": {
@@ -42,6 +47,7 @@
42
47
  "outPerMtok": 30.0,
43
48
  "cacheReadPerMtok": 1.0,
44
49
  "modelId": "gpt-5.6",
50
+ "provider": "openai",
45
51
  "note": "Codex CLI top tier - Reviewer 1 at xhigh effort, Reviewer 3 at medium, triage at max, and the fable/opus rungs of the Codex persona tier map. Reasoning effort changes output volume, not the per-token rate, so one entry covers every effort level. Approximate; verify against OpenAI pricing before relying on totals."
46
52
  },
47
53
  "gpt-5.6-terra": {
@@ -49,6 +55,7 @@
49
55
  "outPerMtok": 5.0,
50
56
  "cacheReadPerMtok": 0.1,
51
57
  "modelId": "gpt-5.6-terra",
58
+ "provider": "openai",
52
59
  "note": "Codex CLI floor tier - speed-optimised, maps the haiku rung of the Codex persona tier map (task-clarifier). Approximate; verify against OpenAI pricing before relying on totals."
53
60
  }
54
61
  }
@@ -658,7 +658,7 @@ function checkMcpRegistration() {
658
658
  // checkMcpRegistration answers "is OURS registered". This answers the question the
659
659
  // user never gets asked: every registered server's tool list is sent with every
660
660
  // turn, they are added one at a time, and nobody sees the running total - our own
661
- // toolkit is 99 tools by itself. This check only ever REPORTS. It never disables
661
+ // toolkit is 115 tools by itself. This check only ever REPORTS. It never disables
662
662
  // anything, and it never blocks or warns, because how many servers are worth their
663
663
  // context is the user's call and not a health failure.
664
664
  //
@@ -2,17 +2,21 @@
2
2
  /**
3
3
  * @file gen-facts.mjs - the numbers the website is allowed to state, derived.
4
4
  *
5
- * The site carried its own copies: `ArchitectureContent.tsx` said "6 faz + 51
6
- * komut" while the repo had six phases and sixty commands. Nobody noticed,
7
- * because a number written into a React component is guarded by nothing. This
8
- * is the same defect `smoke-phase-contract.sh` fixes inside the pipeline, and
9
- * the fix is the same shape: one producer, everything else reads it.
5
+ * A consumer that keeps its own copy of these numbers has nothing guarding it:
6
+ * a count typed into a page or a README is right on the day it is typed and
7
+ * silent afterwards. `smoke-phase-contract.sh` enforces the same shape inside
8
+ * the pipeline - one producer, everything else reads it.
10
9
  *
11
10
  * Source of truth per field:
12
11
  * phases pipeline/schemas/phases.json - the phase contract itself
13
12
  * commandCount a count of pipeline/commands/multi-agent/<name>/SKILL.md
14
13
  * skillCount a count of pipeline/skills/shared/external/<name>/SKILL.md
14
+ * skillCountAll every SKILL.md under pipeline/skills, at any depth
15
+ * agentCount a count of pipeline/agents/*.md
16
+ * hosts the CLIs install/ has an adapter for
15
17
  * toolCount the toolkit's own tools/list response, when reachable
18
+ * toolCategories that same response, grouped by the prefix in each tool name
19
+ * pluginSkillCount every SKILL.md in the plugin marketplace, when checked out
16
20
  * version package.json
17
21
  *
18
22
  * Counts come from the filesystem at the moment of writing, never from a README
@@ -43,6 +47,24 @@ function die(msg) {
43
47
  process.exit(2);
44
48
  }
45
49
 
50
+ // Every SKILL.md under a tree, at any depth. `countSkillDirs` only sees the
51
+ // immediate children of one directory, which is the right shape for the
52
+ // external set and the wrong one for the whole tree: the shared/core and
53
+ // per-host mirrors sit a level deeper and were invisible to it.
54
+ function countSkillsDeep(rel) {
55
+ const dir = join(ROOT, rel);
56
+ if (!existsSync(dir)) die(`missing directory: ${rel}`);
57
+ let n = 0;
58
+ const walk = (d) => {
59
+ for (const e of readdirSync(d, { withFileTypes: true })) {
60
+ if (e.isDirectory()) walk(join(d, e.name));
61
+ else if (e.name === "SKILL.md") n += 1;
62
+ }
63
+ };
64
+ walk(dir);
65
+ return n;
66
+ }
67
+
46
68
  function countSkillDirs(rel) {
47
69
  const dir = join(ROOT, rel);
48
70
  if (!existsSync(dir)) die(`missing directory: ${rel}`);
@@ -70,6 +92,7 @@ try {
70
92
  // tool count is the problem this file exists to remove, and a made-up one is
71
93
  // worse than none.
72
94
  let toolCount = null;
95
+ let toolCategories = null;
73
96
  let toolkitVersion = null;
74
97
  const toolkitDir = join(ROOT, "..", "multi-agent-toolkit-mcp");
75
98
  if (existsSync(join(toolkitDir, "package.json"))) {
@@ -113,10 +136,48 @@ if (existsSync(join(toolkitDir, "package.json"))) {
113
136
  } catch {
114
137
  continue;
115
138
  }
116
- if (msg.id === 2 && Array.isArray(msg.result?.tools)) toolCount = msg.result.tools.length;
139
+ if (msg.id === 2 && Array.isArray(msg.result?.tools)) {
140
+ toolCount = msg.result.tools.length;
141
+ // Grouped by the prefix every tool name carries, from the same
142
+ // response the count comes from. Counting families out of the source
143
+ // gives a different and wrong answer for the reason above, and the
144
+ // site shows the per-family breakdown next to the total, so the two
145
+ // have to come from one place or they will disagree.
146
+ const byFamily = {};
147
+ for (const t of msg.result.tools) {
148
+ const family = String(t.name || "").split("_")[0];
149
+ if (family) byFamily[family] = (byFamily[family] || 0) + 1;
150
+ }
151
+ toolCategories = Object.fromEntries(
152
+ Object.entries(byFamily).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])),
153
+ );
154
+ }
117
155
  }
118
156
  } catch {
119
157
  toolCount = null;
158
+ toolCategories = null;
159
+ }
160
+ }
161
+
162
+ // The plugin marketplace is a third repo, probed the same way and reported as
163
+ // null when it is not checked out beside this one. The site states this number
164
+ // next to the pipeline's own skill counts, and three different skill
165
+ // populations on one page is exactly the shape that drifts.
166
+ let pluginSkillCount = null;
167
+ const pluginsDir = join(ROOT, "..", "multi-agent-plugins", "plugins");
168
+ if (existsSync(pluginsDir)) {
169
+ let n = 0;
170
+ const walk = (d) => {
171
+ for (const e of readdirSync(d, { withFileTypes: true })) {
172
+ if (e.isDirectory()) walk(join(d, e.name));
173
+ else if (e.name === "SKILL.md") n += 1;
174
+ }
175
+ };
176
+ try {
177
+ walk(pluginsDir);
178
+ pluginSkillCount = n;
179
+ } catch {
180
+ pluginSkillCount = null;
120
181
  }
121
182
  }
122
183
 
@@ -130,12 +191,56 @@ const facts = {
130
191
  phaseCount: contract.phases.length,
131
192
  modes: Object.fromEntries(Object.entries(contract.modes).map(([k, v]) => [k, v.phases])),
132
193
  commandCount: countSkillDirs("pipeline/commands/multi-agent"),
194
+ // Two populations, and conflating them is how the site came to show a stale
195
+ // "212 Skills" chip next to a correct "216 skills" sentence. `skillCount` is
196
+ // the external set a user installs; `skillCountAll` is every SKILL.md the
197
+ // repo ships, mirrors included. Both are stated on the site, so both are
198
+ // generated here rather than counted by hand at the other end.
133
199
  skillCount: countSkillDirs("pipeline/skills/shared/external"),
200
+ skillCountAll: countSkillsDeep("pipeline/skills"),
201
+ // The site shows both of these as headline figures and had typed both by
202
+ // hand: a "9 Sub-agents" that happened to be right and a host list that had
203
+ // gone stale, naming Claude Code and Copilot CLI while install/codex.mjs had
204
+ // been shipping a third target for releases.
205
+ agentCount: readdirSync(join(ROOT, "pipeline", "agents")).filter((f) => f.endsWith(".md")).length,
206
+ hosts: readdirSync(join(ROOT, "install"))
207
+ .filter((f) => f.endsWith(".mjs") && !f.startsWith("_") && f !== "index.mjs")
208
+ .map(
209
+ (f) =>
210
+ ({ claude: "Claude Code", copilot: "Copilot CLI", codex: "Codex CLI" })[
211
+ f.replace(".mjs", "")
212
+ ],
213
+ )
214
+ .filter(Boolean)
215
+ .sort(),
134
216
  toolCount,
217
+ toolCategories,
218
+ pluginSkillCount,
135
219
  toolkitVersion,
136
220
  };
137
221
 
138
- const body = `${JSON.stringify(facts, null, 2)}\n`;
222
+ // JSON.stringify and prettier disagree about short arrays: stringify puts each
223
+ // phase id on its own line, prettier collapses a run that fits. That left this
224
+ // file failing `format:check` after every regeneration, held together only by
225
+ // someone remembering to run prettier by hand - and a generated file that needs
226
+ // a manual step to pass the gate is a generated file that will fail the gate.
227
+ //
228
+ // Importing prettier here is not the fix: this script ships in the published
229
+ // package, where a devDependency is not installed (`n/no-unpublished-import`
230
+ // says so). Collapsing the number runs directly costs nothing and leaves the
231
+ // script dependency-free.
232
+ const collapseShortArrays = (json) =>
233
+ json.replace(
234
+ /\[\n\s+((?:-?\d+|"[^"\n]*")(?:,\n\s+(?:-?\d+|"[^"\n]*"))*)\n\s+\]/g,
235
+ (whole, inner) => {
236
+ const one = `[${inner.split(/,\n\s*/).join(", ")}]`;
237
+ // Prettier only collapses what fits the print width; leave the rest alone
238
+ // rather than guess, so the two can never disagree in the other direction.
239
+ return one.length <= 72 ? one : whole;
240
+ },
241
+ );
242
+
243
+ const body = collapseShortArrays(`${JSON.stringify(facts, null, 2)}\n`);
139
244
 
140
245
  if (process.argv.includes("--stdout")) {
141
246
  process.stdout.write(body);
@@ -613,7 +613,7 @@ tracker_next_hint() {
613
613
  # default only up to Opus 4.7 / Sonnet 4.6, a default that landed in
614
614
  # v2.1.268. Naming the fallback on the same line is what keeps a newer model
615
615
  # from advancing six phases in silence.
616
- # `subjects` now carries Phase 2's plan steps as indented rows, and
616
+ # `subjects` now carries Phase 1's plan steps as indented rows, and
617
617
  # update_plan takes the whole list anyway, so re-reading it is what puts
618
618
  # those steps on the Codex plan without a second mechanism. Codex has no
619
619
  # dependency concept; the "(bekliyor: ...)" suffix inside the step text is
@@ -776,9 +776,9 @@ subjects() {
776
776
  # Sub-phases ride out on the SAME list, indented in the subject string.
777
777
  #
778
778
  # The card has drawn these since sub-phases existed; the widget never has,
779
- # and the widget is the surface the user actually looks at. Phase 2's plan
779
+ # and the widget is the surface the user actually looks at. Phase 1's plan
780
780
  # is the case that made the gap matter: the tasks, their order and their
781
- # dependencies are computed, stored and used to drive Phase 3's picker, and
781
+ # dependencies are computed, stored and used to drive Phase 2's picker, and
782
782
  # none of it was visible anywhere the user was looking.
783
783
  #
784
784
  # Indentation is two spaces INSIDE the subject because the widget takes
@@ -1214,7 +1214,7 @@ GATE
1214
1214
  ;;
1215
1215
 
1216
1216
  plan)
1217
- # Phase 2's plan, turned into sub-phases of the phase that will execute it.
1217
+ # Phase 1's plan, turned into sub-phases of the phase that will execute it.
1218
1218
  #
1219
1219
  # The parsing lives here rather than in the phase document for one reason:
1220
1220
  # sub-phases are this file's structure, and a jq blob in a phase doc is a
@@ -1222,7 +1222,7 @@ GATE
1222
1222
  # keeps the doc to one line, which the aggregate phase-doc budget cares about.
1223
1223
  #
1224
1224
  # Reads planning-output.schema.json on stdin: tasks[] with id, title and an
1225
- # optional dependsOn[]. Status is `pending` for all of them - Phase 3 moves
1225
+ # optional dependsOn[]. Status is `pending` for all of them - Phase 2 moves
1226
1226
  # them with `sub`, and pre-marking work as started is the lie the tracker
1227
1227
  # exists to avoid.
1228
1228
  need_jq
@@ -148,12 +148,41 @@ scan_file() {
148
148
  FOUND=1
149
149
  fi
150
150
 
151
- # High-signal provider token prefixes (low false-positive rate)
151
+ # High-signal provider token prefixes (low false-positive rate).
152
+ #
153
+ # This set and the one in lib/outbound-gate.mjs cover the same providers, and
154
+ # smoke-secret-parity.sh holds them to it by running a fake token of each
155
+ # shape through BOTH.
152
156
  if echo "$content" | grep -qE '(ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{36}|github_pat_[A-Za-z0-9_]{60,}|xox[baprs]-[A-Za-z0-9-]{12,}|sk_live_[A-Za-z0-9]{20,}|rk_live_[A-Za-z0-9]{20,}|AIza[0-9A-Za-z_-]{35}|npm_[A-Za-z0-9]{36}|glpat-[A-Za-z0-9_-]{20,}'; then
153
157
  echo "BLOCKED: Provider access token in $file" >&2
154
158
  FOUND=1
155
159
  fi
156
160
 
161
+ # Model-provider and ML-hub keys. `sk-ant-` and `sk-proj-` are checked before
162
+ # the bare `sk-` form so the message names the provider a reader can revoke.
163
+ if echo "$content" | grep -qE 'sk-ant-[A-Za-z0-9_-]{32,}|sk-proj-[A-Za-z0-9_-]{32,}|pplx-[A-Za-z0-9]{32,}|hf_[A-Za-z0-9]{30,}'; then
164
+ echo "BLOCKED: Model-provider API key in $file" >&2
165
+ FOUND=1
166
+ fi
167
+
168
+ # Figma personal access token (figd_) and the MCP OAuth token (figu_). Same
169
+ # shape, same blast radius: both read every file the account can reach.
170
+ if echo "$content" | grep -qE 'fig[a-z]_[A-Za-z0-9_-]{20,}'; then
171
+ echo "BLOCKED: Figma token in $file" >&2
172
+ FOUND=1
173
+ fi
174
+
175
+ # A URL carrying its own credentials, which is how a `git remote -v` paste
176
+ # leaks, and an Authorization header copied out of a curl trace.
177
+ if echo "$content" | grep -qE '[a-z][a-z0-9+.-]*://[^[:space:]/:@]+:[^[:space:]/@]+@'; then
178
+ echo "BLOCKED: URL with embedded credentials in $file" >&2
179
+ FOUND=1
180
+ fi
181
+ if echo "$content" | grep -qiE 'Authorization:[[:space:]]*(Bearer|Basic)[[:space:]]+[A-Za-z0-9._~+/=-]{16,}'; then
182
+ echo "BLOCKED: Authorization header with a token in $file" >&2
183
+ FOUND=1
184
+ fi
185
+
157
186
  # JWT (three base64url segments - header.payload.signature)
158
187
  if echo "$content" | grep -qE 'eyJ[A-Za-z0-9_-]{10,}\.eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}'; then
159
188
  echo "BLOCKED: JWT in $file" >&2
@@ -123,6 +123,14 @@ tree_grep() {
123
123
  return 0
124
124
  }
125
125
 
126
+ # Case-insensitive variant. Prose written to steer a model is written by hand,
127
+ # so its capitalisation is whatever the author felt like, and a case-sensitive
128
+ # pattern would miss "Ignore all previous instructions" by one letter.
129
+ tree_grep_i() {
130
+ tr '\n' '\0' < "$SCAN_LIST" | xargs -0 grep -nHEi -- "$1" 2>/dev/null
131
+ return 0
132
+ }
133
+
126
134
  # stdin: "path:line:content" grep hits -> stdout: "fileIdx|path|line|content"
127
135
  index_hits() {
128
136
  awk -v listfile="$SCAN_LIST" '
@@ -260,6 +268,24 @@ if [ "$THRESHOLD_RANK" -ge 1 ]; then
260
268
  seq=$((seq+1))
261
269
  emit_raw "$HIT_IDX" 9 "$seq" high "$HIT_FILE" "$HIT_LINE" "chmod-then-exec" "script made executable and immediately invoked"
262
270
  done < <(tree_grep 'chmod[[:space:]]+\+x[[:space:]]+[^&;]+[[:space:]]*(&&|;)[[:space:]]*\./' | index_hits)
271
+
272
+ # Prompt injection (OWASP LLM01). The other families ask what a skill makes
273
+ # the MACHINE do; this one asks what it makes the MODEL do. A skill is
274
+ # instructions loaded straight into the context that decides everything after
275
+ # it, so text telling the model to drop its instructions, recite its prompt,
276
+ # or act behind the user's back is an attack delivered as prose - and no
277
+ # pattern above can see it, because nothing is executed.
278
+ #
279
+ # `high`, not `critical`: these phrasings can appear in a skill that DESCRIBES
280
+ # the attack, this scanner's own documentation being the obvious case, so a
281
+ # hit is a line a human reads rather than a verdict.
282
+ seq=0
283
+ while IFS= read -r hit; do
284
+ [ -z "$hit" ] && continue
285
+ parse_hit "$hit"
286
+ seq=$((seq+1))
287
+ emit_raw "$HIT_IDX" 13 "$seq" high "$HIT_FILE" "$HIT_LINE" "prompt-injection" "text instructing the model to override, disclose or hide - OWASP LLM01"
288
+ done < <(tree_grep_i 'ignore (all )?(the )?(previous|prior|above|earlier) (instructions|prompts?|rules?)|disregard (your|the|any) (system prompt|previous instructions|instructions)|(reveal|print|output|repeat|show) (your|the) (system prompt|full instructions)|(do not|don'"'"'t|never) (tell|inform) the (user|operator)|without (telling|informing|asking) the (user|operator)|(always|automatically) (approve|confirm) [^.]{0,30}(without|regardless)|you are now (a|an|the) ' | index_hits)
263
289
  fi
264
290
 
265
291
  # --- medium families (rank 2) ---------------------------------------------