@blamejs/exceptd-skills 0.19.33 → 0.19.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/bin/exceptd.js +895 -2828
  3. package/data/_indexes/_meta.json +2 -2
  4. package/lib/auto-discovery.js +56 -286
  5. package/lib/canonical-eq.js +7 -40
  6. package/lib/citation-resolve.js +22 -70
  7. package/lib/collectors/ai-api.js +170 -76
  8. package/lib/collectors/cicd-pipeline-compromise.js +113 -136
  9. package/lib/collectors/citation-hygiene.js +72 -210
  10. package/lib/collectors/containers.js +41 -130
  11. package/lib/collectors/cred-stores.js +31 -115
  12. package/lib/collectors/crypto-codebase.js +55 -138
  13. package/lib/collectors/crypto.js +24 -54
  14. package/lib/collectors/hardening.js +20 -78
  15. package/lib/collectors/kernel.js +16 -46
  16. package/lib/collectors/library-author.js +198 -211
  17. package/lib/collectors/mcp.js +24 -70
  18. package/lib/collectors/runtime.js +24 -86
  19. package/lib/collectors/sbom.js +130 -118
  20. package/lib/collectors/scan-excludes.js +33 -139
  21. package/lib/collectors/secrets.js +62 -178
  22. package/lib/cross-ref-api.js +39 -123
  23. package/lib/currency-severity.js +8 -27
  24. package/lib/cve-batch.js +13 -21
  25. package/lib/cve-cli.js +13 -20
  26. package/lib/cve-curation.js +72 -239
  27. package/lib/cve-regression-watcher.js +29 -155
  28. package/lib/cvss.js +13 -54
  29. package/lib/doctor-bucketing.js +3 -19
  30. package/lib/exit-codes.js +10 -42
  31. package/lib/flag-suggest.js +7 -25
  32. package/lib/framework-gap.js +39 -113
  33. package/lib/gap-detectors.js +37 -159
  34. package/lib/id-validation.js +9 -30
  35. package/lib/job-queue.js +13 -36
  36. package/lib/lint-skills.js +88 -236
  37. package/lib/playbook-runner.js +759 -2107
  38. package/lib/prefetch.js +101 -376
  39. package/lib/refresh-external.js +199 -633
  40. package/lib/refresh-network.js +78 -311
  41. package/lib/rfc-cli.js +23 -68
  42. package/lib/scoring.js +85 -146
  43. package/lib/sign.js +43 -229
  44. package/lib/source-advisories.js +43 -194
  45. package/lib/source-ghsa.js +37 -120
  46. package/lib/source-osv.js +94 -266
  47. package/lib/ttp-mapper.js +28 -27
  48. package/lib/upstream-check-cli.js +36 -29
  49. package/lib/upstream-check.js +19 -44
  50. package/lib/validate-catalog-meta.js +17 -61
  51. package/lib/validate-cve-catalog.js +52 -121
  52. package/lib/validate-indexes.js +25 -76
  53. package/lib/validate-package.js +16 -62
  54. package/lib/validate-playbooks.js +78 -286
  55. package/lib/validate-vendor.js +16 -49
  56. package/lib/verify.js +56 -286
  57. package/lib/version-pins.js +5 -34
  58. package/lib/worker-pool.js +11 -30
  59. package/lib/xml-tokenizer.js +47 -152
  60. package/manifest.json +53 -53
  61. package/orchestrator/dispatcher.js +17 -68
  62. package/orchestrator/event-bus.js +11 -74
  63. package/orchestrator/index.js +138 -413
  64. package/orchestrator/pipeline.js +28 -85
  65. package/orchestrator/scanner.js +34 -138
  66. package/orchestrator/scheduler.js +20 -84
  67. package/package.json +1 -1
  68. package/sbom.cdx.json +242 -242
  69. package/scripts/audit-catalog-gaps.js +9 -62
  70. package/scripts/audit-cross-skill.js +5 -31
  71. package/scripts/audit-perf.js +29 -28
  72. package/scripts/backfill-theater-test.js +7 -64
  73. package/scripts/bootstrap.js +12 -44
  74. package/scripts/build-indexes.js +40 -154
  75. package/scripts/builders/activity-feed.js +4 -14
  76. package/scripts/builders/catalog-summaries.js +3 -10
  77. package/scripts/builders/currency.js +7 -20
  78. package/scripts/builders/cwe-chains.js +7 -30
  79. package/scripts/builders/did-ladders.js +6 -13
  80. package/scripts/builders/frequency.js +5 -19
  81. package/scripts/builders/jurisdiction-clocks.js +6 -25
  82. package/scripts/builders/recipes.js +6 -14
  83. package/scripts/builders/section-offsets.js +13 -51
  84. package/scripts/builders/stale-content.js +7 -28
  85. package/scripts/builders/summary-cards.js +8 -29
  86. package/scripts/builders/theater-fingerprints.js +21 -31
  87. package/scripts/builders/token-budget.js +4 -31
  88. package/scripts/check-agents-md-collectors.js +26 -57
  89. package/scripts/check-catalog-gap-budget.js +15 -32
  90. package/scripts/check-changelog-extract.js +18 -48
  91. package/scripts/check-codebase-patterns-currency.js +6 -22
  92. package/scripts/check-codebase-patterns.js +63 -143
  93. package/scripts/check-epss-consistency.js +9 -64
  94. package/scripts/check-framework-gap-coverage.js +13 -31
  95. package/scripts/check-manifest-snapshot.js +62 -81
  96. package/scripts/check-sbom-currency.js +44 -142
  97. package/scripts/check-test-count.js +15 -52
  98. package/scripts/check-test-coverage.js +83 -198
  99. package/scripts/check-test-subjects.js +21 -62
  100. package/scripts/check-ttp-references.js +14 -38
  101. package/scripts/check-ttp-upstream.js +8 -40
  102. package/scripts/check-version-bump.js +9 -61
  103. package/scripts/check-version-tags.js +20 -121
  104. package/scripts/predeploy.js +38 -184
  105. package/scripts/refresh-manifest-snapshot.js +16 -38
  106. package/scripts/refresh-mitre-atlas.js +7 -8
  107. package/scripts/refresh-mitre-attack.js +1 -8
  108. package/scripts/refresh-mitre-d3fend.js +3 -9
  109. package/scripts/refresh-mitre-ics-attack.js +7 -8
  110. package/scripts/refresh-reverse-refs.js +27 -94
  111. package/scripts/refresh-rfc-index.js +7 -10
  112. package/scripts/refresh-sbom.js +31 -161
  113. package/scripts/refresh-upstream-catalogs.js +63 -148
  114. package/scripts/release.js +69 -234
  115. package/scripts/run-e2e-scenarios.js +26 -73
  116. package/scripts/sync-manifest-metadata.js +10 -34
  117. package/scripts/sync-package-description.js +8 -17
  118. package/scripts/validate-vendor-online.js +13 -44
  119. package/scripts/verify-shipped-tarball.js +35 -141
@@ -1,18 +1,11 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/did-ladders.js
4
- *
5
- * Builds `data/_indexes/did-ladders.json` — canonical defense-in-depth
6
- * ladders per high-frequency attack class. Each ladder is a layered
7
- * sequence of controls (perimeter → identity → workload → data →
8
- * detection) with a cross-reference to the source skill(s) that
9
- * operationalize each layer and the D3FEND countermeasures backing each
10
- * layer.
11
- *
12
- * Curated content, validated against the manifest: every referenced skill
13
- * and D3FEND id must exist in their respective catalogs or the build
14
- * fails. This is the only place where the project's DiD knowledge is
15
- * laid out flat across attack classes.
3
+ * Builds `data/_indexes/did-ladders.json`: one defense-in-depth ladder per
4
+ * high-frequency attack class, each layer naming the skill that operationalizes
5
+ * it and the D3FEND countermeasures behind it. The ladders are curated here —
6
+ * the only place the project's DiD knowledge is laid out flat across attack
7
+ * classes — and a referenced skill or D3FEND id absent from its catalog fails
8
+ * the build.
16
9
  */
17
10
 
18
11
  const LADDERS = [
@@ -1,21 +1,8 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/frequency.js
4
- *
5
- * Builds `data/_indexes/frequency.json` — citation-count tables. For each
6
- * catalog (CWE, ATLAS, ATT&CK, D3FEND, framework gaps, RFC, DLP), counts
7
- * how many skills cite each entry. Surfaces which entries are load-bearing
8
- * (cited by many) vs. orphan-adjacent (cited by ≤1).
9
- *
10
- * Per-field shape:
11
- * {
12
- * <entry_id>: { count, skills: [name, ...] }
13
- * }
14
- *
15
- * Plus rollups:
16
- * - top_cited: top 10 entries per field
17
- * - orphan_adjacent: entries cited by exactly one skill
18
- * - uncited: catalog entries with zero skill citations (flagged for review)
3
+ * Builds `data/_indexes/frequency.json`: how many skills cite each entry of
4
+ * each catalog, with rollups that separate the load-bearing entries from the
5
+ * ones a single skill cites and the ones nothing cites at all.
19
6
  */
20
7
 
21
8
  function buildFrequency({ skills, catalogs }) {
@@ -51,7 +38,6 @@ function buildFrequency({ skills, catalogs }) {
51
38
  .sort();
52
39
  }
53
40
 
54
- // Uncited: catalog has an entry but zero skill cites it.
55
41
  const uncited = {};
56
42
  const catalogFieldMap = {
57
43
  cwe_refs: catalogs.cwe,
@@ -67,8 +53,8 @@ function buildFrequency({ skills, catalogs }) {
67
53
  uncited[field] = inCatalog.filter((id) => !counts[field][id]).sort();
68
54
  }
69
55
 
70
- // attack_refs has no catalog file (uses MITRE upstream directly), so no
71
- // uncited table for it — only counts.
56
+ // attack_refs is absent from catalogFieldMap: it has no catalog file of its
57
+ // own, so it gets counts but no uncited table.
72
58
 
73
59
  const topCited = {};
74
60
  for (const f of fields) topCited[f] = topN(f);
@@ -2,31 +2,12 @@
2
2
  /**
3
3
  * scripts/builders/jurisdiction-clocks.js
4
4
  *
5
- * Builds `data/_indexes/jurisdiction-clocks.json` — the normalized
6
- * jurisdiction × obligation × clock matrix. Today consumers asking
7
- * "what's the breach-notification clock in jurisdiction X?" have to
8
- * scan `data/global-frameworks.json` and pull `notification_sla` off
9
- * specific framework entries. This index flattens the dimension.
10
- *
11
- * Obligation types covered:
12
- * - breach_notification (hours from awareness)
13
- * - patch_sla (hours from disclosure for Critical/High)
14
- * - incident_reporting (regulator + clock + trigger)
15
- *
16
- * Per-jurisdiction shape:
17
- * {
18
- * jurisdiction_name:
19
- * frameworks: {
20
- * <fwName>: {
21
- * authority,
22
- * breach_notification: { hours, trigger, stages?, source }
23
- * patch_sla: { hours, note?, source }
24
- * incident_reporting: { hours, trigger, source }
25
- * }
26
- * }
27
- * fastest_breach_notification: { hours, framework } // null when none specified
28
- * fastest_patch_sla: { hours, framework }
29
- * }
5
+ * Builds `data/_indexes/jurisdiction-clocks.json` — the normalized jurisdiction
6
+ * × obligation × clock matrix, so a consumer asking "what is the breach-
7
+ * notification clock in jurisdiction X?" need not scan
8
+ * `data/global-frameworks.json` for `notification_sla` on each framework entry.
9
+ * All times are in hours; a `fastest_*` slot is null when no framework in the
10
+ * jurisdiction specifies that clock.
30
11
  */
31
12
 
32
13
  function buildJurisdictionClocks({ globalFrameworks }) {
@@ -1,16 +1,9 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/recipes.js
4
- *
5
- * Builds `data/_indexes/recipes.json` — curated skill sequences for the
6
- * most common operator use cases. Each recipe is a vetted chain of skills
7
- * to invoke, in order, with a brief rationale per step. Saves the
8
- * researcher skill from re-deriving these on every request.
9
- *
10
- * Recipes are content (not derived), so this builder is mostly a
11
- * declarative table. The build step validates that every referenced
12
- * skill exists in the manifest, so a renamed/deleted skill surfaces as a
13
- * build error.
3
+ * Builds `data/_indexes/recipes.json` — curated skill chains for the common
4
+ * operator cases, so the researcher skill does not re-derive them per request.
5
+ * The table below is content, not derived; the build validates every referenced
6
+ * skill against the manifest, so a rename or deletion fails the build.
14
7
  */
15
8
 
16
9
  const RECIPES = [
@@ -138,9 +131,8 @@ function buildRecipes({ skills }) {
138
131
  throw new Error("recipes.js: " + errors.join("; "));
139
132
  }
140
133
 
141
- // Add token-budget hints once we have them — token-budget builder may run
142
- // after this one, so we just emit the per-step list and let consumers
143
- // join to token-budget.json. The skill_count is cheap to include here.
134
+ // No token-budget hints here: that builder may run after this one, so
135
+ // consumers join to token-budget.json themselves.
144
136
  const out = {
145
137
  _meta: {
146
138
  schema_version: "1.0.0",
@@ -1,42 +1,14 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/section-offsets.js
4
- *
5
- * Builds `data/_indexes/section-offsets.json` — for each skill, the byte
6
- * + line offsets of every H2 section header in the body. AI consumers can
7
- * slice a single section (e.g. "Compliance Theater Check") from disk
8
- * without parsing the full skill file.
9
- *
10
- * Per-skill shape:
11
- * {
12
- * path: "skills/<name>/skill.md",
13
- * total_bytes: n,
14
- * total_lines: n,
15
- * frontmatter: { byte_start, byte_end, line_start, line_end },
16
- * sections: [
17
- * {
18
- * name: raw H2 text (e.g. "Threat Context (mid-2026)")
19
- * normalized_name: collapsed for lookup ("threat-context")
20
- * line: 1-based line number of the "## …" header
21
- * byte_start: byte offset of the "## " character
22
- * byte_end: byte offset where the next H2 begins (or EOF)
23
- * bytes: byte_end - byte_start
24
- * h3_count: number of "### " headers contained
25
- * },
26
- * ...
27
- * ]
28
- * }
29
- *
30
- * The normalized_name strips parenthetical qualifiers and common phrasings
31
- * so consumers can request a canonical section name without caring about
32
- * formatting drift.
3
+ * Builds `data/_indexes/section-offsets.json`: per skill, the byte and line
4
+ * offsets of every H2 section in the body, so a consumer can slice one section
5
+ * off disk without parsing the whole file. normalized_name collapses
6
+ * parenthetical qualifiers and phrasing variants onto a canonical name.
33
7
  */
34
8
 
35
9
  const fs = require("fs");
36
10
  const path = require("path");
37
11
 
38
- // Recognized canonical section anchors. Multiple raw H2 phrasings map to one
39
- // normalized name — see grep survey of skills/* for the variant phrasings.
40
12
  const NORMALIZERS = [
41
13
  [/threat\s*context/i, "threat-context"],
42
14
  [/framework\s*lag\s*declaration/i, "framework-lag-declaration"],
@@ -56,7 +28,6 @@ function normalize(headerText) {
56
28
  for (const [re, canonical] of NORMALIZERS) {
57
29
  if (re.test(stripped)) return canonical;
58
30
  }
59
- // Fall back: slug.
60
31
  return stripped
61
32
  .toLowerCase()
62
33
  .replace(/[^a-z0-9]+/g, "-")
@@ -68,14 +39,10 @@ function buildOne(absPath, relPath) {
68
39
  const totalBytes = buf.length;
69
40
  const text = buf.toString("utf8");
70
41
  const lines = text.split(/\r?\n/);
71
- // EOL-aware line-start byte offsets: walk the actual terminator bytes off the
72
- // decoded text rather than assuming a fixed 1-byte newline. On a pure-LF body
73
- // every terminator is 1 byte so the offsets are identical to the old `+ 1`
74
- // accumulator; on a stray CRLF body the \r\n terminator is 2 bytes and the
75
- // offsets stay correct (the old fixed-constant approach undercounted by 1
76
- // byte per line and silently misaligned every token-budget slice). The split
77
- // above discards terminator bytes, so the width can only be recovered by
78
- // reading the terminators off the raw text — done here.
42
+ // Line-start byte offsets are measured off the real terminator bytes: a CRLF
43
+ // terminator is 2 bytes, and assuming a fixed 1-byte newline misaligns every
44
+ // offset in a CRLF body. The split above discards the terminators, so their
45
+ // width can only be recovered from the raw text.
79
46
  const lineByteOffsets = [0];
80
47
  const eolRe = /\r?\n/g;
81
48
  let m;
@@ -103,9 +70,8 @@ function buildOne(absPath, relPath) {
103
70
  }
104
71
  : null;
105
72
 
106
- // H2 headers — only those outside fenced code blocks. Skill bodies often
107
- // contain "## Foo" lines inside ```...``` blocks as output templates; those
108
- // are not real sections.
73
+ // H2 headers outside fenced code blocks only — skill bodies carry "## Foo"
74
+ // lines inside ```...``` blocks as output templates, which are not sections.
109
75
  const h2 = [];
110
76
  let inFence = false;
111
77
  for (let i = 0; i < lines.length; i++) {
@@ -124,10 +90,8 @@ function buildOne(absPath, relPath) {
124
90
  const next = h2[j + 1];
125
91
  const startByte = lineByteOffsets[cur.idx];
126
92
  const endByte = next ? lineByteOffsets[next.idx] : totalBytes;
127
- // Count H3 within this section — fence-aware, the same way the H2 loop
128
- // above is. A section starts and ends on an H2 header, both of which are
129
- // outside any fence, so fence state always begins false here. "### Foo"
130
- // lines inside ```...``` output templates are not real sub-sections.
93
+ // Fence-aware like the H2 loop: a section starts and ends on an H2, both
94
+ // outside any fence, so fence state always begins false here.
131
95
  const endIdx = next ? next.idx : lines.length;
132
96
  let h3Count = 0;
133
97
  let h3InFence = false;
@@ -173,7 +137,5 @@ function buildSectionOffsets({ root, skills }) {
173
137
  };
174
138
  }
175
139
 
176
- // buildOne is exported for regression testing of the EOL-aware byte offsets
177
- // (a CRLF body must still produce byte_start values that point at the real
178
- // "## " byte in the raw file).
140
+ // buildOne is exported for the CRLF byte-offset regression test.
179
141
  module.exports = { buildSectionOffsets, buildOne };
@@ -1,24 +1,9 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/stale-content.js
4
- *
5
- * Builds `data/_indexes/stale-content.json` — surfaces stale or
6
- * drifted references the audit-cross-skill script catches at run time,
7
- * persisted as a JSON artifact so CI / dashboards / downstream tools can
8
- * read the same view without invoking the script.
9
- *
10
- * Checks performed here (subset of audit-cross-skill that's relevant to
11
- * the index layer, deterministic across reruns):
12
- *
13
- * - Skill bodies referencing renamed-skill tokens (e.g. age-gates-minor-*)
14
- * - README badge counts vs. live counts
15
- * - "Researcher routes to N skills" claim vs. live count
16
- * - Skills with last_threat_review older than 180 days from
17
- * manifest.threat_review_date (gives a stale-content snapshot)
18
- * - Catalog _meta.last_verified entries older than freshness_policy.stale_after_days
19
- * - Forward_watch items mentioning dates that have already passed
20
- *
21
- * Each finding is { severity, category, artifact, detail }.
3
+ * Builds `data/_indexes/stale-content.json` — the subset of the
4
+ * audit-cross-skill checks that is deterministic across reruns, persisted so CI
5
+ * and downstream tools read the same view without invoking that script. Each
6
+ * finding is { severity, category, artifact, detail }.
22
7
  */
23
8
 
24
9
  const fs = require("fs");
@@ -45,7 +30,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
45
30
  const findings = [];
46
31
  const refDate = new Date((manifest.threat_review_date || "2026-05-01") + "T00:00:00Z");
47
32
 
48
- // 1. Stale-renamed-skill tokens
49
33
  for (const s of skills) {
50
34
  const body = fs.readFileSync(path.join(root, s.path), "utf8");
51
35
  for (const tok of RENAMED_SKILL_TOKENS) {
@@ -62,7 +46,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
62
46
  }
63
47
  }
64
48
 
65
- // 2. README badge counts vs. live counts
66
49
  const readmePath = path.join(root, "README.md");
67
50
  if (fs.existsSync(readmePath)) {
68
51
  const readme = fs.readFileSync(readmePath, "utf8");
@@ -71,10 +54,9 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
71
54
  const liveJurisdictions = (() => {
72
55
  try {
73
56
  const gf = JSON.parse(fs.readFileSync(path.join(root, "data/global-frameworks.json"), "utf8"));
74
- // Count non-underscore keys (GLOBAL included) — the canonical
75
- // jurisdiction count used by the README badge and catalog-summaries.
76
- // Excluding GLOBAL here uniquely produced 34 and emitted a false
77
- // badge_drift finding against the README's (correct) 35.
57
+ // Non-underscore keys, GLOBAL INCLUDED — the canonical jurisdiction
58
+ // count the README badge and catalog-summaries use. Excluding GLOBAL
59
+ // undercounts by one and emits a false badge_drift.
78
60
  return Object.keys(gf).filter((k) => !k.startsWith("_")).length;
79
61
  } catch {
80
62
  return null;
@@ -98,7 +80,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
98
80
  }
99
81
  }
100
82
 
101
- // 3. Researcher dispatch count claim
102
83
  const researcherPath = path.join(root, "skills/researcher/skill.md");
103
84
  if (fs.existsSync(researcherPath)) {
104
85
  const r = fs.readFileSync(researcherPath, "utf8");
@@ -117,7 +98,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
117
98
  }
118
99
  }
119
100
 
120
- // 4. Skills with > 180 days since review (against reference date)
121
101
  for (const s of skills) {
122
102
  if (!s.last_threat_review) continue;
123
103
  const ageDays = Math.floor(
@@ -133,7 +113,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
133
113
  }
134
114
  }
135
115
 
136
- // 5. Catalog last_verified entries older than freshness_policy.stale_after_days
137
116
  for (const rel of catalogFiles) {
138
117
  const abs = path.join(root, rel);
139
118
  try {
@@ -1,24 +1,8 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/summary-cards.js
4
- *
5
- * Builds `data/_indexes/summary-cards.json` — for each skill, a compact
6
- * abstract that downstream AI consumers (researcher dispatch in particular)
7
- * can render without loading the full skill body.
8
- *
9
- * Card shape per skill:
10
- * {
11
- * description: manifest description
12
- * threat_context_excerpt: first paragraph of Threat Context section
13
- * produces: first paragraph of Output Format section (if present)
14
- * key_xrefs: {
15
- * cwe_refs, d3fend_refs, framework_gaps, atlas_refs,
16
- * attack_refs, rfc_refs, dlp_refs
17
- * }
18
- * trigger_count, atlas_count, attack_count, framework_gap_count,
19
- * last_threat_review, path,
20
- * handoff_targets: skills referenced from this skill's Hand-Off section
21
- * }
3
+ * Builds `data/_indexes/summary-cards.json`: a compact per-skill abstract that
4
+ * downstream AI consumers — researcher dispatch in particular — render without
5
+ * loading the full skill body.
22
6
  */
23
7
 
24
8
  const fs = require("fs");
@@ -47,14 +31,11 @@ function locateHeader(lines, headerRegex) {
47
31
  }
48
32
 
49
33
  function firstParagraphAfterHeader(body, headerRegex) {
50
- // Locate the first real H2 matching the regex, then find the first prose
51
- // paragraph beneath it — skip any H3 / H4 / bold-prefix metadata lines /
52
- // horizontal rules / table separators that often sit at the top of a
53
- // section. Real H2 means outside of fenced code blocks.
34
+ // The first prose paragraph beneath the matching H2, skipping the H3/H4,
35
+ // bold-prefix metadata, rules and table separators that often lead a section.
54
36
  const lines = body.split(/\r?\n/);
55
37
  const hdrIdx = locateHeader(lines, headerRegex);
56
38
  if (hdrIdx < 0) return null;
57
- // Find the next real H2 as the section boundary.
58
39
  const allH2 = findRealH2Indices(lines);
59
40
  const nextH2 = allH2.find((i) => i > hdrIdx);
60
41
  const sectionEnd = nextH2 != null ? nextH2 : lines.length;
@@ -106,11 +87,9 @@ function firstChunkAfterHeader(body, headerRegex, maxChars = 600) {
106
87
  }
107
88
 
108
89
  function handoffTargets(body, allSkillNames, selfName) {
109
- // Look ONLY in the Hand-Off section; backtick-quoted skill names count as a
110
- // target. Bound the scan to [Hand-Off header, next real H2) so unrelated
111
- // later sections are not mis-attributed as hand-off targets. Header
112
- // detection and the section boundary are both fence-aware (a `## ` line
113
- // inside a ```...``` block is not a real H2).
90
+ // Only the Hand-Off section counts, and a backtick-quoted skill name is a
91
+ // target. The scan is bounded to [header, next real H2) so a later section is
92
+ // not mis-attributed, and both bounds are fence-aware.
114
93
  const lines = body.split(/\r?\n/);
115
94
  const h2 = findRealH2Indices(lines);
116
95
  const handoffIdx = h2.find((i) => /^## Hand-?Off/.test(lines[i]));
@@ -1,28 +1,16 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/theater-fingerprints.js
4
- *
5
- * Builds `data/_indexes/theater-fingerprints.json` — for each Compliance
6
- * Theater pattern in the `compliance-theater` skill, a structured record:
7
- * - the claim (what auditors hear)
8
- * - the audit evidence (what passes the audit)
9
- * - the reality (why it's theater)
10
- * - the detection test (operational steps)
11
- * - the controls it spans (NIST 800-53 / ISO 27001 / PCI / SOC 2)
12
- * - the evidence CVE / campaign tying the pattern to the real world
13
- *
14
- * Extracted from `skills/compliance-theater/skill.md`. The compliance-theater
15
- * skill is the source-of-truth — this index just structures the pattern
16
- * library so downstream consumers (audit defense, framework-gap-analysis)
17
- * can join on control IDs without re-parsing the markdown.
3
+ * Builds `data/_indexes/theater-fingerprints.json` from
4
+ * `skills/compliance-theater/skill.md`, which stays the source of truth — the
5
+ * index only structures the pattern library so consumers can join on control
6
+ * IDs without re-parsing the markdown.
18
7
  */
19
8
 
20
9
  const fs = require("fs");
21
10
  const path = require("path");
22
11
 
23
- // Stable mapping of each Pattern → the controls it spans. Manually curated
24
- // from the skill's Framework Lag Declaration table — keep this in lockstep
25
- // with skills/compliance-theater/skill.md.
12
+ // Each pattern → the controls it spans, hand-curated from the skill's Framework
13
+ // Lag Declaration table; keep in lockstep with that table.
26
14
  const PATTERN_CONTROL_MAP = {
27
15
  1: {
28
16
  pattern_name: "Patch Management Theater",
@@ -117,10 +105,9 @@ const PATTERN_CONTROL_MAP = {
117
105
  };
118
106
 
119
107
  function extractPatternBodyFromSkill(skillBody, patternNumber) {
120
- // Find "### Pattern N:" and capture until the next "### Pattern N+1:" OR
121
- // the next ## H2 after the header line itself. We skip the header's own
122
- // line before scanning for an H2 boundary — otherwise the `### Pattern N:`
123
- // line would match the `## ` prefix regex once its leading `#` is sliced.
108
+ // Captures from "### Pattern N:" to the next "### Pattern N+1:" or the next
109
+ // H2. The H2 scan starts past the header line: `### Pattern N:` itself
110
+ // matches the `^## ` prefix once its leading `#` is sliced off.
124
111
  const startRe = new RegExp(`^### Pattern ${patternNumber}:`, "m");
125
112
  const startMatch = skillBody.match(startRe);
126
113
  if (!startMatch) return null;
@@ -142,19 +129,20 @@ function extractPatternBodyFromSkill(skillBody, patternNumber) {
142
129
  }
143
130
 
144
131
  function pullField(body, label) {
145
- // The patterns use a "**Label:** ..." prose convention. Return the line(s)
146
- // after the label until the next "**" or blank line.
132
+ // Patterns write fields as "**Label:** ..."; capture to the next "**" or blank line.
147
133
  const re = new RegExp(`\\*\\*${label.replace(/[-/\\^$*+?.()|[\\]{}]/g, "\\$&")}:?\\*\\*\\s*([\\s\\S]*?)(?=\\n\\n|\\n\\*\\*|$)`);
148
134
  const m = body.match(re);
149
135
  return m ? m[1].trim() : null;
150
136
  }
151
137
 
152
- function buildTheaterFingerprints({ root }) {
138
+ // `patternMap` defaults to the curated map; an explicit one lets a test drive
139
+ // the builder with a pattern set of a different size.
140
+ function buildTheaterFingerprints({ root, patternMap = PATTERN_CONTROL_MAP }) {
153
141
  const skillPath = path.join(root, "skills/compliance-theater/skill.md");
154
142
  const body = fs.readFileSync(skillPath, "utf8");
155
143
 
156
144
  const out = {};
157
- for (const [num, meta] of Object.entries(PATTERN_CONTROL_MAP)) {
145
+ for (const [num, meta] of Object.entries(patternMap)) {
158
146
  const patternBody = extractPatternBodyFromSkill(body, Number(num));
159
147
  out[`pattern-${num}`] = {
160
148
  pattern_number: Number(num),
@@ -173,9 +161,8 @@ function buildTheaterFingerprints({ root }) {
173
161
  };
174
162
  }
175
163
 
176
- // Inverted index: control_id → pattern(s) it spans, so a consumer can ask
177
- // "is this control implicated in a theater pattern?" without scanning all
178
- // seven patterns.
164
+ // Inverted index, framework::control_id → patterns, so a consumer can ask
165
+ // whether a control is implicated without scanning every pattern.
179
166
  const byControl = {};
180
167
  for (const [pid, p] of Object.entries(out)) {
181
168
  for (const c of p.controls) {
@@ -184,11 +171,14 @@ function buildTheaterFingerprints({ root }) {
184
171
  }
185
172
  }
186
173
 
174
+ const patternCount = Object.keys(out).length;
187
175
  return {
188
176
  _meta: {
189
177
  schema_version: "1.0.0",
190
- source: "skills/compliance-theater/skill.md (7 documented patterns)",
191
- pattern_count: Object.keys(out).length,
178
+ // Counted from the built set, not written out, so adding a pattern cannot
179
+ // leave this prose contradicting pattern_count in the shipped index.
180
+ source: `skills/compliance-theater/skill.md (${patternCount} documented patterns)`,
181
+ pattern_count: patternCount,
192
182
  },
193
183
  patterns: out,
194
184
  by_control: byControl,
@@ -1,36 +1,9 @@
1
1
  "use strict";
2
2
  /**
3
- * scripts/builders/token-budget.js
4
- *
5
- * Builds `data/_indexes/token-budget.json` — per-skill approximate token
6
- * counts using a character-density heuristic. Zero-dep (no tiktoken). The
7
- * approximation is documented as such so consumers know to recompute with
8
- * their own tokenizer if precision matters.
9
- *
10
- * Heuristic: 1 token ≈ 4 characters for English prose mixed with technical
11
- * tokens (matches the well-known OpenAI rule-of-thumb). This is an upper
12
- * bound for Claude (Anthropic's tokenizer is more efficient on common
13
- * prose) but is good enough for context-budget planning where consumers
14
- * just need to know "is this load 5K or 50K tokens".
15
- *
16
- * Per-skill shape:
17
- * {
18
- * path: skill file path
19
- * bytes: total file bytes
20
- * chars: total character count
21
- * lines: line count
22
- * approx_tokens: chars / 4 (integer)
23
- * approx_chars_per_token: 4
24
- * sections: {
25
- * <normalized_section_name>: { bytes, approx_tokens }
26
- * }
27
- * }
28
- *
29
- * Corpus totals live under the top-level `_meta` block:
30
- * {
31
- * schema_version, tokenizer_note, approx_chars_per_token,
32
- * total_chars, total_approx_tokens, skill_count
33
- * }
3
+ * Builds `data/_indexes/token-budget.json`: per-skill token counts from a
4
+ * 1-token ≈ 4-characters density heuristic, with no tokenizer dependency. The
5
+ * result is an upper bound for context-budget planning, never a precise count —
6
+ * the caveat travels with the data in `_meta.tokenizer_note`.
34
7
  */
35
8
 
36
9
  const fs = require("fs");