@blamejs/exceptd-skills 0.19.33 → 0.19.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/bin/exceptd.js +896 -2824
  3. package/data/_indexes/_meta.json +2 -2
  4. package/lib/auto-discovery.js +56 -286
  5. package/lib/canonical-eq.js +7 -40
  6. package/lib/citation-resolve.js +22 -70
  7. package/lib/collectors/ai-api.js +20 -54
  8. package/lib/collectors/cicd-pipeline-compromise.js +40 -108
  9. package/lib/collectors/citation-hygiene.js +72 -210
  10. package/lib/collectors/containers.js +41 -130
  11. package/lib/collectors/cred-stores.js +31 -115
  12. package/lib/collectors/crypto-codebase.js +55 -138
  13. package/lib/collectors/crypto.js +24 -54
  14. package/lib/collectors/hardening.js +20 -78
  15. package/lib/collectors/kernel.js +16 -46
  16. package/lib/collectors/library-author.js +57 -206
  17. package/lib/collectors/mcp.js +24 -70
  18. package/lib/collectors/runtime.js +24 -86
  19. package/lib/collectors/sbom.js +34 -106
  20. package/lib/collectors/scan-excludes.js +31 -138
  21. package/lib/collectors/secrets.js +62 -178
  22. package/lib/cross-ref-api.js +39 -123
  23. package/lib/currency-severity.js +8 -27
  24. package/lib/cve-batch.js +13 -21
  25. package/lib/cve-cli.js +13 -20
  26. package/lib/cve-curation.js +72 -239
  27. package/lib/cve-regression-watcher.js +29 -152
  28. package/lib/cvss.js +13 -54
  29. package/lib/doctor-bucketing.js +3 -19
  30. package/lib/exit-codes.js +10 -42
  31. package/lib/flag-suggest.js +7 -25
  32. package/lib/framework-gap.js +35 -114
  33. package/lib/gap-detectors.js +37 -159
  34. package/lib/id-validation.js +9 -30
  35. package/lib/job-queue.js +13 -36
  36. package/lib/lint-skills.js +64 -232
  37. package/lib/playbook-runner.js +693 -2095
  38. package/lib/prefetch.js +100 -376
  39. package/lib/refresh-external.js +199 -627
  40. package/lib/refresh-network.js +75 -307
  41. package/lib/rfc-cli.js +23 -68
  42. package/lib/scoring.js +77 -145
  43. package/lib/sign.js +43 -229
  44. package/lib/source-advisories.js +43 -194
  45. package/lib/source-ghsa.js +37 -120
  46. package/lib/source-osv.js +94 -266
  47. package/lib/ttp-mapper.js +14 -24
  48. package/lib/upstream-check-cli.js +10 -28
  49. package/lib/upstream-check.js +19 -44
  50. package/lib/validate-catalog-meta.js +17 -61
  51. package/lib/validate-cve-catalog.js +43 -119
  52. package/lib/validate-indexes.js +25 -76
  53. package/lib/validate-package.js +16 -62
  54. package/lib/validate-playbooks.js +69 -275
  55. package/lib/validate-vendor.js +16 -49
  56. package/lib/verify.js +56 -286
  57. package/lib/version-pins.js +5 -34
  58. package/lib/worker-pool.js +11 -30
  59. package/lib/xml-tokenizer.js +47 -152
  60. package/manifest.json +53 -53
  61. package/orchestrator/dispatcher.js +17 -68
  62. package/orchestrator/event-bus.js +11 -74
  63. package/orchestrator/index.js +138 -412
  64. package/orchestrator/pipeline.js +28 -85
  65. package/orchestrator/scanner.js +34 -138
  66. package/orchestrator/scheduler.js +20 -84
  67. package/package.json +1 -1
  68. package/sbom.cdx.json +241 -241
  69. package/scripts/audit-catalog-gaps.js +9 -62
  70. package/scripts/audit-cross-skill.js +5 -31
  71. package/scripts/audit-perf.js +6 -16
  72. package/scripts/backfill-theater-test.js +7 -64
  73. package/scripts/bootstrap.js +12 -44
  74. package/scripts/build-indexes.js +40 -154
  75. package/scripts/builders/activity-feed.js +4 -14
  76. package/scripts/builders/catalog-summaries.js +3 -10
  77. package/scripts/builders/currency.js +7 -20
  78. package/scripts/builders/cwe-chains.js +7 -30
  79. package/scripts/builders/did-ladders.js +6 -13
  80. package/scripts/builders/frequency.js +5 -19
  81. package/scripts/builders/jurisdiction-clocks.js +6 -25
  82. package/scripts/builders/recipes.js +6 -14
  83. package/scripts/builders/section-offsets.js +13 -51
  84. package/scripts/builders/stale-content.js +7 -28
  85. package/scripts/builders/summary-cards.js +8 -29
  86. package/scripts/builders/theater-fingerprints.js +12 -27
  87. package/scripts/builders/token-budget.js +4 -31
  88. package/scripts/check-agents-md-collectors.js +11 -54
  89. package/scripts/check-catalog-gap-budget.js +15 -32
  90. package/scripts/check-changelog-extract.js +18 -48
  91. package/scripts/check-codebase-patterns-currency.js +6 -22
  92. package/scripts/check-codebase-patterns.js +50 -143
  93. package/scripts/check-epss-consistency.js +9 -64
  94. package/scripts/check-framework-gap-coverage.js +13 -31
  95. package/scripts/check-manifest-snapshot.js +13 -73
  96. package/scripts/check-sbom-currency.js +44 -142
  97. package/scripts/check-test-count.js +15 -52
  98. package/scripts/check-test-coverage.js +66 -197
  99. package/scripts/check-test-subjects.js +21 -62
  100. package/scripts/check-ttp-references.js +14 -38
  101. package/scripts/check-ttp-upstream.js +8 -40
  102. package/scripts/check-version-bump.js +9 -61
  103. package/scripts/check-version-tags.js +20 -121
  104. package/scripts/predeploy.js +38 -184
  105. package/scripts/refresh-manifest-snapshot.js +16 -38
  106. package/scripts/refresh-mitre-atlas.js +3 -8
  107. package/scripts/refresh-mitre-attack.js +1 -8
  108. package/scripts/refresh-mitre-d3fend.js +3 -9
  109. package/scripts/refresh-mitre-ics-attack.js +3 -8
  110. package/scripts/refresh-reverse-refs.js +27 -94
  111. package/scripts/refresh-rfc-index.js +2 -10
  112. package/scripts/refresh-sbom.js +31 -161
  113. package/scripts/refresh-upstream-catalogs.js +40 -137
  114. package/scripts/release.js +69 -232
  115. package/scripts/run-e2e-scenarios.js +24 -71
  116. package/scripts/sync-manifest-metadata.js +10 -34
  117. package/scripts/sync-package-description.js +8 -17
  118. package/scripts/validate-vendor-online.js +13 -44
  119. package/scripts/verify-shipped-tarball.js +35 -140
@@ -84,38 +84,22 @@ const OPTIMAL = {
84
84
  };
85
85
 
86
86
  /**
87
- * Score how much a given framework lags behind current threat reality.
88
- * Returns 0 (no lag) to 100 (complete theater).
89
- *
90
- * @param {string} frameworkId - Key in framework-control-gaps.json
91
- * @param {object} controlGaps - Parsed framework-control-gaps.json
92
- * @param {object} globalFrameworks - Parsed global-frameworks.json
93
- * @returns {{ score: number, breakdown: object, label: string }}
87
+ * How far a framework lags current threat reality, 0 (no lag) to 100 (complete
88
+ * theater), as { score, label, breakdown }. The catalogs arrive parsed.
94
89
  */
95
90
  function lagScore(frameworkId, controlGaps, globalFrameworks) {
96
91
  const frameworkData = _findFrameworkData(frameworkId, globalFrameworks);
97
92
 
98
- // global-frameworks uses short KEYS (EU_AI_ACT, NCSC_CAF) while
99
- // framework-control-gaps stores human-readable framework strings ("EU
100
- // Artificial Intelligence Act (2024/1689)"). A naive
101
- // `g.framework.includes(frameworkId)` only accidentally matched the few
102
- // frameworks whose key happens to be a substring of the catalog string
103
- // (DORA, GDPR, NIS2); every other framework reported
104
- // framework_specific_gaps:0. Resolve the framework's display name first,
105
- // then match the catalog with the SAME normalized scheme gapReport() and
106
- // the orchestrator use, so the two paths converge. g.framework may be a
107
- // string or an array — iterate either form.
93
+ // global-frameworks is keyed by short id (EU_AI_ACT) while
94
+ // framework-control-gaps stores display strings ("EU Artificial Intelligence
95
+ // Act (2024/1689)"), so the display name resolves first and matching uses the
96
+ // SAME normalized scheme as gapReport(). `g.framework` is a string or an array.
108
97
  const normalize = (s) => String(s).toLowerCase().replace(/[\s_-]/g, '');
109
98
  const idNorm = normalize(frameworkId);
110
99
  const nameNorm = frameworkData?.full_name ? normalize(frameworkData.full_name) : null;
111
- // Data-driven aliases close the naming divergence between a framework's
112
- // global-frameworks full_name and the (often different) labels the control-gap
113
- // catalog uses for it — e.g. ASD_ISM's full_name is "Australian Signals
114
- // Directorate Information Security Manual" but the catalog labels its 5 gaps
115
- // "au-ism" / "ACSC ISM" / "Australian Government Information Security Manual",
116
- // so neither nameNorm nor idNorm matched and lagScore reported 0 gaps. Aliases
117
- // live in data/global-frameworks.json (catalog_aliases) so this generalizes to
118
- // any future framework whose catalog label diverges from its name.
100
+ // catalog_aliases in data/global-frameworks.json bridges a framework's
101
+ // full_name and the labels the control-gap catalog uses: ASD_ISM is also
102
+ // "au-ism" / "ACSC ISM" there, which neither nameNorm nor idNorm matches.
119
103
  const aliasNorms = Array.isArray(frameworkData?.catalog_aliases)
120
104
  ? frameworkData.catalog_aliases.map((a) => normalize(a)).filter(Boolean)
121
105
  : [];
@@ -132,11 +116,8 @@ function lagScore(frameworkId, controlGaps, globalFrameworks) {
132
116
  if (normalize(key).startsWith(idNorm)) return true; // gap-key prefix
133
117
  return false;
134
118
  });
135
- // Observable backstop: a framework that EXISTS in global-frameworks resolving
136
- // to zero catalog gaps is almost always a fresh naming divergence (a new
137
- // catalog label not yet in catalog_aliases), not a genuinely gap-free
138
- // framework. Surface it as a structured flag rather than a silent 0 so a
139
- // regression is greppable in the breakdown instead of invisible.
119
+ // A framework that resolves yet matches zero catalog gaps is almost always a
120
+ // label missing from catalog_aliases, not a gap-free framework.
140
121
  const resolvedButZeroGaps = Boolean(frameworkData && gaps.length === 0);
141
122
 
142
123
  const universalGaps = Object.values(controlGaps).filter(g =>
@@ -168,25 +149,16 @@ function lagScore(frameworkId, controlGaps, globalFrameworks) {
168
149
  pqc_coverage: { coverage: frameworkData?.pqc_coverage ?? 'unknown', score: pqcScore },
169
150
  universal_gaps: { count: universalGaps.length, score: universalGapScore },
170
151
  framework_specific_gaps: gaps.length,
171
- // True only when the framework resolves in global-frameworks yet matched
172
- // zero catalog gaps — a likely naming divergence worth investigating.
173
152
  framework_resolved_but_zero_gaps: resolvedButZeroGaps
174
153
  }
175
154
  };
176
155
  }
177
156
 
178
157
  /**
179
- * Generate a gap report for one or more frameworks vs. a threat scenario.
180
- *
181
- * @param {string[]} frameworkIds - Framework identifiers
182
- * @param {string} threatScenario - Description of the threat or CVE ID
183
- * @param {object} controlGaps - Parsed framework-control-gaps.json
184
- * @param {object} cveCatalog - Parsed cve-catalog.json (optional)
185
- * @param {object} opts - { allFrameworks?: boolean, lessons?: object }
186
- * opts.lessons is the parsed zeroday-lessons.json. Optional: callers
187
- * that omit it get a report without the new-control section rather than
188
- * an error, which keeps the existing call signature working.
189
- * @returns {{ frameworks: object, universal_gaps: object[], new_control_requirements: object[], theater_risks: object[] }}
158
+ * Gap report for one or more frameworks against a threat scenario — a CVE id or
159
+ * free text. Omitting `opts.lessons` (parsed zeroday-lessons.json) yields a
160
+ * report with no new-control section rather than an error. Returns
161
+ * { frameworks, universal_gaps, new_control_requirements, theater_risks, summary }.
190
162
  */
191
163
  function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, opts = {}) {
192
164
  const scenario = threatScenario.toLowerCase();
@@ -207,14 +179,9 @@ function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, o
207
179
 
208
180
  const frameworkResults = {};
209
181
  for (const id of frameworkIds) {
210
- // Match a framework filter ID against catalog entries by:
211
- // - exact match against gap.framework (e.g. "NIST SP 800-53 Rev 5")
212
- // - normalized substring match (strip case + spaces + hyphens, e.g. user
213
- // passing "nist-800-53" matches catalog "NIST SP 800-53 Rev 5")
214
- // - normalized prefix match on the gap KEY (e.g. user "nist-800-53"
215
- // matches keys "NIST-800-53-SI-2", "NIST-800-53-SC-8")
216
- // This makes the named-framework filter behave the same way `all` does
217
- // when extracting per-framework subsets.
182
+ // A filter id matches three ways: exactly against gap.framework, as a
183
+ // normalized substring of it ("nist-800-53" → "NIST SP 800-53 Rev 5"), or as
184
+ // a normalized prefix of the gap KEY ("NIST-800-53-SI-2").
218
185
  const normalize = (s) => String(s).toLowerCase().replace(/[\s_-]/g, '');
219
186
  const idNorm = normalize(id);
220
187
  const frameworkGaps = relevantGaps.filter(([key, g]) => {
@@ -236,13 +203,9 @@ function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, o
236
203
  };
237
204
  }
238
205
 
239
- // Scope the report to what the operator actually requested. With an explicit
240
- // framework filter, only gaps that survived the per-framework filter
241
- // (frameworkResults[*].gaps) belong in the report; with `all`, every
242
- // scenario-relevant gap does. `seen` is the de-duplicated set of surviving
243
- // gap keys (a single gap can match multiple requested frameworks). Both
244
- // theater_risks and the matching count derive from it so the per-framework
245
- // body, the theater-risk list, and the summary footer all agree.
206
+ // With an explicit filter only the gaps that survived it belong in the report;
207
+ // with `all`, every scenario-relevant gap does. theater_risks and the matching
208
+ // count both derive from this, so body, theater list and footer agree.
246
209
  let scopedGaps;
247
210
  if (opts.allFrameworks) {
248
211
  scopedGaps = relevantGaps;
@@ -254,19 +217,9 @@ function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, o
254
217
  scopedGaps = relevantGaps.filter(([key]) => seen.has(key));
255
218
  }
256
219
 
257
- // Previously this filtered on `theater_pattern`
258
- // (a legacy field) but the v0.12.29 backfill added a structured
259
- // `theater_test` block on all 118 entries while leaving most without
260
- // `theater_pattern`. Result: the per-entry badge (line 188 above)
261
- // showed "⚠ THEATER RISK" for every open gap, but the summary
262
- // footer reported "0 theater-risk controls" because nothing matched
263
- // the legacy field. Now: an entry is theater-risk if it's open AND
264
- // carries EITHER `theater_test` OR `theater_pattern`. Footer + badge
265
- // count agree.
266
- //
267
- // theater_risks is built from scopedGaps (not the full relevantGaps) so a
268
- // single-framework request cannot leak or mis-summarize theater controls
269
- // from frameworks the operator never asked about.
220
+ // Theater-risk is open AND EITHER `theater_test` OR `theater_pattern` — most
221
+ // entries have only the structured test, and filtering on `theater_pattern`
222
+ // alone puts the badge and the summary footer at odds.
270
223
  const theaterRisks = scopedGaps
271
224
  .filter(([, g]) => g.status === 'open' && (g.theater_test || g.theater_pattern))
272
225
  .map(([key, g]) => ({
@@ -276,39 +229,20 @@ function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, o
276
229
  theater_test_present: !!g.theater_test,
277
230
  }));
278
231
 
279
- // Summary matching count. With `all` frameworks the summary counts every
280
- // scenario-relevant gap across the whole catalog (relevantGaps). With an
281
- // explicit framework filter the summary must agree with the per-framework
282
- // body the operator actually sees — otherwise `framework-gap nist-800-53
283
- // <cve>` shows e.g. "2 matching control gap(s)" per-framework but "Summary:
284
- // 8 matching gaps" (every framework's hits, pre-filter).
232
+ // The summary counts what the operator sees, so the footer cannot overcount.
285
233
  const matchingGapCount = scopedGaps.length;
286
234
 
287
- // Controls the zero-day lesson for this CVE says no framework carries yet.
288
- //
289
- // `frameworks[].gaps` covers the other half of the same question — existing
290
- // controls that are insufficient — so a report that answered only that half
291
- // told an operator which controls fall short without ever saying what to put
292
- // in their place. The lessons catalog had recorded exactly that for 302 CVEs
293
- // and nothing read the field, so the analysis existed and reached no one.
294
- //
295
- // Keyed by CVE, so a free-text scenario legitimately resolves to none.
235
+ // Controls the zero-day lesson for this CVE says no framework carries yet —
236
+ // the other half of what `frameworks[].gaps` answers. Keyed by CVE, so a
237
+ // free-text scenario legitimately resolves to none.
296
238
  const lessons = (opts && opts.lessons) || {};
297
239
  const lessonEntry = Object.entries(lessons)
298
240
  .find(([id]) => id.toLowerCase() === scenario);
299
241
  const newControls = (lessonEntry && Array.isArray(lessonEntry[1].new_control_requirements))
300
242
  ? lessonEntry[1].new_control_requirements
301
- // Three catalog entries held a bare string where the schema expects a
302
- // control object, which rendered as "- undefined undefined:" the moment
303
- // this section started being shown. Emitting a malformed record is worse
304
- // than omitting it, so drop anything the renderer cannot print.
305
- //
306
- // Every field the renderer interpolates has to be checked, not just the
307
- // identifying one: guarding `id` alone still let `{ id: 'NEW-X' }` reach
308
- // the output as "NEW-X undefined:", so the guard would have covered the
309
- // bare-string case that prompted it and nothing else. The data itself is
310
- // fixed; the catalog test fails on the shape so a malformed record cannot
311
- // hide behind this filter.
243
+ // A record the renderer cannot print is dropped rather than emitted as
244
+ // "- undefined undefined:". EVERY interpolated field is checked — guarding
245
+ // `id` alone still lets `{ id: 'NEW-X' }` through as "NEW-X undefined:".
312
246
  .filter(c =>
313
247
  c && typeof c === 'object' &&
314
248
  typeof c.id === 'string' && c.id.trim() !== '' &&
@@ -320,9 +254,7 @@ function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, o
320
254
  name: c.name,
321
255
  requirement: c.description,
322
256
  evidence: c.evidence,
323
- // Which framework controls this is meant to close. Without it the
324
- // control reads as free-floating advice rather than an answer to a
325
- // named gap above.
257
+ // The framework controls this closes; without them it is free-floating advice.
326
258
  closes: Array.isArray(c.gap_closes) ? c.gap_closes : [],
327
259
  }))
328
260
  : [];
@@ -347,11 +279,8 @@ function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, o
347
279
  }
348
280
 
349
281
  /**
350
- * Run all seven theater pattern checks against an organization's control inventory.
351
- *
352
- * @param {object} controlGaps - Parsed framework-control-gaps.json
353
- * @param {object} cveCatalog - Parsed cve-catalog.json
354
- * @returns {{ findings: object[], theater_score: number, compliant_but_exposed: boolean }}
282
+ * Runs every THEATER_PATTERNS check against a control inventory, returning
283
+ * { findings, theater_score, theater_label, compliant_but_exposed, recommendation }.
355
284
  */
356
285
  function theaterCheck(controlGaps, cveCatalog = {}) {
357
286
  const findings = [];
@@ -385,13 +314,7 @@ function theaterCheck(controlGaps, cveCatalog = {}) {
385
314
  };
386
315
  }
387
316
 
388
- /**
389
- * Compare multiple frameworks by lag score for a dashboard view.
390
- *
391
- * @param {object} controlGaps - Parsed framework-control-gaps.json
392
- * @param {object} globalFrameworks - Parsed global-frameworks.json
393
- * @returns {Array} Sorted by lag score descending
394
- */
317
+ /** Every framework in globalFrameworks with its lag score, sorted descending. */
395
318
  function compareFrameworks(controlGaps, globalFrameworks) {
396
319
  const results = [];
397
320
  const frameworkIds = _extractFrameworkIds(globalFrameworks);
@@ -404,8 +327,6 @@ function compareFrameworks(controlGaps, globalFrameworks) {
404
327
  return results.sort((a, b) => b.score - a.score);
405
328
  }
406
329
 
407
- // --- private helpers ---
408
-
409
330
  function _findFrameworkData(frameworkId, globalFrameworks) {
410
331
  for (const jurisdiction of Object.values(globalFrameworks)) {
411
332
  if (!jurisdiction.frameworks) continue;
@@ -1,51 +1,11 @@
1
1
  "use strict";
2
2
  /**
3
- * lib/gap-detectors.js
4
- *
5
- * v0.13.21 — Catalog gap detection beyond the v0.13.19 missing-context /
6
- * dangling-ref / draft-debt classes. The audit-catalog-gaps detector
7
- * surfaced field-presence holes; this module adds seven cross-cutting
8
- * detection classes the prior detector did not cover.
9
- *
10
- * Each detector is a pure function: takes the loaded catalogs + options,
11
- * returns an array of findings. The audit-catalog-gaps CLI composes them
12
- * into a unified report; the integrity test exercises them against the
13
- * shipped catalogs; --class filters select between them.
14
- *
15
- * Detection classes:
16
- *
17
- * 1. content-quality — fields present but content weak
18
- * (short, placeholder-language, name-as-
19
- * description, KEV-listed but no advisories)
20
- *
21
- * 2. temporal-staleness — last_verified > 180d, last_updated > 365d,
22
- * CISA-KEV due-date passed, EPSS stale
23
- *
24
- * 3. logical-consistency — internal-state contradictions
25
- * (cisa_kev:true + date:null, etc.)
26
- *
27
- * 4. cross-ref-completeness — bidirectional references
28
- * (CVE→CWE present but CWE.evidence_cves
29
- * missing the back-ref)
30
- *
31
- * 5. schema-evolution — required-since-version fields missing
32
- * on older entries
33
- *
34
- * 6. operator-action-sla — auto-imported entries older than the
35
- * curation-SLA without operator action
36
- *
37
- * 7. unused-orphan — catalog entries no skill / playbook /
38
- * CVE references — dead-weight content
39
- *
40
- * Why pure functions: each detector is independently testable against
41
- * synthetic catalog inputs, and the integration is just `Array.concat`
42
- * over the seven results. Composing in audit-catalog-gaps.js stays
43
- * thin.
3
+ * Catalog gap detection. Each detector is a pure function over the loaded
4
+ * catalogs plus options, returning an array of findings; DETECTOR_CLASSES below
5
+ * names the full set.
44
6
  */
45
7
 
46
- // Sentinel strings that indicate placeholder / curation-pending content.
47
- // Adding new sentinels here makes them findable across every text-heavy
48
- // field without changing the call sites.
8
+ // Placeholder / curation-pending sentinels, applied to every text-heavy field.
49
9
  const PLACEHOLDER_SENTINELS = [
50
10
  /pending operator curation/i,
51
11
  /refer to vendor advisory for IOC list/i,
@@ -65,11 +25,7 @@ function hasPlaceholderLanguage(str) {
65
25
  return false;
66
26
  }
67
27
 
68
- // ---------- 1. content-quality ----------
69
- //
70
- // Fields present but content weak. Each rule is per-catalog + per-field
71
- // because the "what's weak" depends on the field's semantic role.
72
-
28
+ // Fields present but weak; what counts as weak is per catalog and field.
73
29
  function contentQualityFindings(loaded) {
74
30
  const out = [];
75
31
  const cve = loaded["cve-catalog"];
@@ -80,9 +36,7 @@ function contentQualityFindings(loaded) {
80
36
  const e = cve[id];
81
37
  if (!e) continue;
82
38
 
83
- // Vector text: < 50 chars or placeholder-language indicates the
84
- // operator didn't actually describe the primitive. Hard Rule #1
85
- // implicit: every CVE needs a real exploitation-vector description.
39
+ // A short or placeholder vector means the exploitation primitive is undescribed.
86
40
  if (typeof e.vector === "string" && e.vector.length > 0 && e.vector.length < 50) {
87
41
  out.push({ class: "content-quality", catalog: "cve-catalog", id,
88
42
  field: "vector", reason: `vector is ${e.vector.length} chars (< 50 threshold) — likely a stub` });
@@ -92,24 +46,18 @@ function contentQualityFindings(loaded) {
92
46
  field: "vector", reason: "vector contains placeholder-language sentinel" });
93
47
  }
94
48
 
95
- // poc_description with placeholder language while poc_available:true
96
- // is a contradiction — the project claims PoC exists but didn't
97
- // document where.
49
+ // poc_available:true with placeholder text claims a PoC without saying where.
98
50
  if (e.poc_available === true && hasPlaceholderLanguage(e.poc_description)) {
99
51
  out.push({ class: "content-quality", catalog: "cve-catalog", id,
100
52
  field: "poc_description", reason: "poc_available:true but description carries placeholder sentinel" });
101
53
  }
102
54
 
103
- // KEV-listed CVEs MUST have vendor_advisories[] non-empty — the
104
- // KEV listing implies CISA has linked vendor advisory metadata.
105
- // Empty vendor_advisories is an operator-curation gap.
55
+ // A KEV listing implies CISA linked advisory metadata, so empty is a curation gap.
106
56
  if (e.cisa_kev === true && (!Array.isArray(e.vendor_advisories) || e.vendor_advisories.length === 0)) {
107
57
  out.push({ class: "content-quality", catalog: "cve-catalog", id,
108
58
  field: "vendor_advisories", reason: "cisa_kev:true but vendor_advisories is empty" });
109
59
  }
110
60
 
111
- // Name reused as description (catalog noise — operator didn't
112
- // write a real description, just echoed the name).
113
61
  if (typeof e.name === "string" && typeof e.description === "string"
114
62
  && e.name === e.description && e.name.length > 0) {
115
63
  out.push({ class: "content-quality", catalog: "cve-catalog", id,
@@ -119,12 +67,7 @@ function contentQualityFindings(loaded) {
119
67
  return out;
120
68
  }
121
69
 
122
- // ---------- 2. temporal-staleness ----------
123
- //
124
- // Time-based decay. Catalog entries get stale as the threat-intelligence
125
- // landscape shifts. Surfacing stale entries gives operators a re-verify
126
- // work-queue.
127
-
70
+ // Time-based decay: stale entries become an operator re-verify work queue.
128
71
  function daysSince(iso, now) {
129
72
  if (typeof iso !== "string" || !/^\d{4}-\d{2}-\d{2}/.test(iso)) return null;
130
73
  const t = Date.parse(iso);
@@ -157,17 +100,11 @@ function temporalStalenessFindings(loaded, opts = {}) {
157
100
  field: "last_updated", reason: `last_updated is ${sinceUpdated}d old (threshold ${STALE_UPDATED_DAYS}d)` });
158
101
  }
159
102
 
160
- // NOTE: a passed CISA KEV due-date is intentionally NOT a temporal-staleness
161
- // finding. The due-date is a fixed external date about an OPERATOR's
162
- // remediation deadline, not about whether this catalog entry's data is
163
- // fresh — every historical KEV entry's due-date passes by calendar and says
164
- // nothing about catalog currency. Counting it made the class grow without
165
- // bound as the catalog ages and as KEV-import drafts get curated (promotion
166
- // would otherwise re-add the finding the draft exemption had removed).
167
- // Catalog-data freshness is measured by maintainer-controllable fields:
168
- // source_verified, last_updated, and epss_date (below).
169
-
170
- // EPSS score has its own currency clock — FIRST recalculates daily.
103
+ // A passed CISA KEV due-date is NOT temporal staleness: it is a fixed
104
+ // external remediation deadline, and every historical entry's passes by
105
+ // calendar while saying nothing about catalog currency.
106
+
107
+ // EPSS has its own currency clock; FIRST recalculates daily.
171
108
  if (typeof e.epss_score === "number" && typeof e.epss_date === "string") {
172
109
  const sinceEpss = daysSince(e.epss_date, now);
173
110
  if (sinceEpss !== null && sinceEpss > STALE_EPSS_DAYS) {
@@ -179,12 +116,7 @@ function temporalStalenessFindings(loaded, opts = {}) {
179
116
  return out;
180
117
  }
181
118
 
182
- // ---------- 3. logical-consistency ----------
183
- //
184
- // Internal-state rules that must hold across multiple fields. These are
185
- // the bugs that pass schema validation (every required field is present)
186
- // but the field combinations don't make sense.
187
-
119
+ // Multi-field rules: combinations that pass schema validation yet contradict.
188
120
  function logicalConsistencyFindings(loaded) {
189
121
  const out = [];
190
122
  const cve = loaded["cve-catalog"];
@@ -195,18 +127,14 @@ function logicalConsistencyFindings(loaded) {
195
127
  const e = cve[id];
196
128
  if (!e) continue;
197
129
 
198
- // cisa_kev:true with null cisa_kev_date — KEV listing has a
199
- // dateAdded field in CISA's authoritative JSON; null means we
200
- // failed to record it at intake time.
130
+ // CISA's JSON carries dateAdded on every listing, so null means intake lost it.
201
131
  if (e.cisa_kev === true && (e.cisa_kev_date == null || e.cisa_kev_date === "")) {
202
132
  out.push({ class: "logical-consistency", catalog: "cve-catalog", id,
203
133
  rule: "cisa_kev_date_present_when_kev_true",
204
134
  reason: "cisa_kev:true requires cisa_kev_date (CISA's dateAdded)" });
205
135
  }
206
136
 
207
- // live_patch_available:true with empty live_patch_tools[] — the
208
- // RWEP live_patch_available factor only fires when tools list
209
- // names a real live-patch path; the boolean alone is a lie.
137
+ // The RWEP deduction only holds when the tools list names a real live-patch path.
210
138
  if (e.live_patch_available === true
211
139
  && (!Array.isArray(e.live_patch_tools) || e.live_patch_tools.length === 0)) {
212
140
  out.push({ class: "logical-consistency", catalog: "cve-catalog", id,
@@ -214,9 +142,7 @@ function logicalConsistencyFindings(loaded) {
214
142
  reason: "live_patch_available:true but live_patch_tools is empty — RWEP factor would mis-fire" });
215
143
  }
216
144
 
217
- // ai_discovered:true requires named AI tool in attribution_note
218
- // (Hard Rule #1 enforcement). The schema-validator catches
219
- // discovery_source==unknown but not the attribution-text absence.
145
+ // The schema validator catches discovery_source==unknown, not a too-short note.
220
146
  if (e.ai_discovered === true) {
221
147
  const note = e.ai_discovery_notes || e.discovery_attribution_note || "";
222
148
  if (typeof note !== "string" || note.length < 30) {
@@ -226,8 +152,6 @@ function logicalConsistencyFindings(loaded) {
226
152
  }
227
153
  }
228
154
 
229
- // active_exploitation:"confirmed" with empty verification_sources
230
- // is a credibility gap — exploitation claims need sourcing.
231
155
  if (e.active_exploitation === "confirmed"
232
156
  && (!Array.isArray(e.verification_sources) || e.verification_sources.length < 2)) {
233
157
  out.push({ class: "logical-consistency", catalog: "cve-catalog", id,
@@ -235,7 +159,6 @@ function logicalConsistencyFindings(loaded) {
235
159
  reason: `active_exploitation:"confirmed" requires >= 2 verification_sources; have ${(e.verification_sources || []).length}` });
236
160
  }
237
161
 
238
- // rwep_score declared but rwep_factors empty — score is unsupported.
239
162
  if (typeof e.rwep_score === "number"
240
163
  && (!e.rwep_factors || Object.keys(e.rwep_factors).length === 0)) {
241
164
  out.push({ class: "logical-consistency", catalog: "cve-catalog", id,
@@ -246,13 +169,8 @@ function logicalConsistencyFindings(loaded) {
246
169
  return out;
247
170
  }
248
171
 
249
- // ---------- 4. cross-ref-completeness ----------
250
- //
251
- // Bidirectional reference checks. Pre-v0.13.21, the dangling-ref class
252
- // only verified the forward direction (CVE.cwe_refs[] resolves into
253
- // cwe-catalog). This class verifies the BACK-reference is present too
254
- // (CWE.evidence_cves[] includes the CVE that cited it).
255
-
172
+ // The dangling-ref class verifies the forward direction; this one verifies the
173
+ // back-reference: the target entry lists the CVE that cited it.
256
174
  function crossRefCompletenessFindings(loaded) {
257
175
  const out = [];
258
176
  const cve = loaded["cve-catalog"];
@@ -269,8 +187,7 @@ function crossRefCompletenessFindings(loaded) {
269
187
  if (cid === "_meta") continue;
270
188
  const e = cve[cid];
271
189
  if (!e) continue;
272
- // Drafts excluded — auto-imported entries don't yet have curated
273
- // refs.
190
+ // Drafts excluded — an auto-imported entry has no curated refs yet.
274
191
  if (e._auto_imported) continue;
275
192
  for (const c of (e.cwe_refs || [])) {
276
193
  if (!cveByCwe.has(c)) cveByCwe.set(c, new Set());
@@ -325,12 +242,7 @@ function crossRefCompletenessFindings(loaded) {
325
242
  return out;
326
243
  }
327
244
 
328
- // ---------- 5. schema-evolution ----------
329
- //
330
- // Required-since-version checks. Fields the schema requires today were
331
- // optional on entries added in older releases. The audit surfaces those
332
- // pre-existing entries so operator-curation can backfill.
333
-
245
+ // Fields the schema requires today that were optional on older entries.
334
246
  const REQUIRED_SINCE = {
335
247
  "cve-catalog": [
336
248
  { field: "ai_discovered", since: "0.12.36", check: (v) => typeof v === "boolean" },
@@ -360,12 +272,7 @@ function schemaEvolutionFindings(loaded) {
360
272
  return out;
361
273
  }
362
274
 
363
- // ---------- 6. operator-action-sla ----------
364
- //
365
- // Auto-imported entries are intake-class events. The catalog allows them
366
- // to ship un-curated (operators add detail later) but past a threshold
367
- // the un-curated state IS the problem.
368
-
275
+ // Past the SLA, an entry's un-curated state is itself the finding.
369
276
  function operatorActionSlaFindings(loaded, opts = {}) {
370
277
  const now = opts.now || new Date();
371
278
  const AUTO_IMPORT_SLA_DAYS = opts.auto_import_sla_days || 60;
@@ -396,35 +303,17 @@ function operatorActionSlaFindings(loaded, opts = {}) {
396
303
  return out;
397
304
  }
398
305
 
399
- // ---------- 7. unused-orphan ----------
400
- //
401
- // Entries that no skill / playbook / CVE references — dead-weight
402
- // content the operator can either repurpose or remove.
403
-
404
- // Build reference sets from skills/*.md frontmatter + body and from
405
- // data/playbooks/*.json content. Pre-v0.13.21 follow-up (codex P1 PR
406
- // #61): unusedOrphanFindings defaulted these to empty sets, which
407
- // flagged D3FEND / CWE / ATT&CK IDs referenced in skill bodies as
408
- // "unused orphans" — false positive. v0.13.21+ builds the reference
409
- // sets internally when the caller doesn't supply them.
410
- //
411
- // The regex is permissive — any CWE-NNN / T1234[.456] / AML.TNNNN /
412
- // D3[AF]-XX / RFC-NNN token in a skill body or playbook JSON counts as a
413
- // reference. We deliberately scan the FULL text, not just structured
414
- // fields, because skill bodies cite IDs in prose ("see CWE-79") as
415
- // often as in frontmatter. The D3FEND alternative covers all three
416
- // namespaces — D3- techniques, D3A- digital artifacts, and D3F-
417
- // fingerprints — with alphanumeric segments; the prior `D3-[A-Z]+`
418
- // pattern matched neither the D3A-/D3F- prefixes nor digit-bearing
419
- // segments, so a D3A-* citation in a skill body went unrecognized and
420
- // the entry it referenced was mis-flagged as an unused orphan.
306
+ // Entries nothing references — dead weight to repurpose or remove.
307
+
308
+ // Any id token in a skill body or playbook JSON counts as a reference, and the
309
+ // full text is scanned rather than the structured fields, because skill bodies
310
+ // cite ids in prose. The D3FEND alternative must cover D3-, D3A- and D3F-: a
311
+ // narrower `D3-[A-Z]+` misses a D3A- citation and mis-flags its entry as orphan.
421
312
  const REFERENCE_TOKEN_RE = /\b(?:CWE-\d+|T\d{4}(?:\.\d{3})?|AML\.T\d{4}(?:\.\d{3})?|D3[AF]?-[A-Z0-9]+(?:-[A-Z0-9]+)*|RFC-\d+)\b/g;
422
313
 
423
314
  function buildExternalRefs(rootPath) {
424
- // Lazy require — `path` + `fs` are already in scope at module level.
425
- // Tolerate the absence of either directory (synthetic-test contexts
426
- // may not have a skills/ tree). Returns { skillRefs, playbookRefs }
427
- // as Sets of stringified IDs.
315
+ // Returns { skillRefs, playbookRefs } as Sets of id strings; a missing
316
+ // skills/ or playbooks/ tree yields an empty set rather than throwing.
428
317
  if (!rootPath) {
429
318
  const path = require("path");
430
319
  rootPath = path.join(__dirname, "..");
@@ -457,9 +346,8 @@ function buildExternalRefs(rootPath) {
457
346
 
458
347
  function unusedOrphanFindings(loaded, opts = {}) {
459
348
  const out = [];
460
- // Auto-populate skill/playbook refs when the caller didn't supply
461
- // them. The composing runAllDetectors() also auto-populates via
462
- // _autoLoadRefs unless tests pin explicit empty sets.
349
+ // Auto-populate the reference sets when the caller supplies neither; a caller
350
+ // wanting genuinely empty sets passes _autoLoadRefs:false.
463
351
  let skillRefs = opts.skillRefs;
464
352
  let playbookRefs = opts.playbookRefs;
465
353
  if (!skillRefs && !playbookRefs && opts._autoLoadRefs !== false) {
@@ -482,10 +370,6 @@ function unusedOrphanFindings(loaded, opts = {}) {
482
370
  }
483
371
  const isReferenced = (id) => skillRefs.has(id) || playbookRefs.has(id) || cveRefIds.has(id);
484
372
 
485
- // CWE / ATT&CK / ATLAS / D3FEND / framework-gap entries that nothing
486
- // references are orphans. Operator-curated entries get a longer
487
- // grace period (intentional forward-looking content); auto-imported
488
- // entries with no reference are clearer waste.
489
373
  for (const catKey of ["cwe-catalog", "attack-techniques", "atlas-ttps", "d3fend-catalog", "framework-control-gaps"]) {
490
374
  const cat = loaded[catKey];
491
375
  if (!cat) continue;
@@ -503,13 +387,9 @@ function unusedOrphanFindings(loaded, opts = {}) {
503
387
  return out;
504
388
  }
505
389
 
506
- // ---------- Composite ----------
507
-
508
390
  function runAllDetectors(loaded, opts = {}) {
509
- // Pre-populate external reference sets ONCE and thread them through
510
- // every detector that needs them. Avoids re-scanning skills/ +
511
- // playbooks/ per detector and keeps the same reference set
512
- // consistent across the composed run.
391
+ // Build the external reference sets once and thread them through, so a composed
392
+ // run does not re-scan per detector or measure against different sets.
513
393
  const orphanOpts = { ...opts };
514
394
  if (!orphanOpts.skillRefs && !orphanOpts.playbookRefs && opts._autoLoadRefs !== false) {
515
395
  const refs = buildExternalRefs(opts._rootPath);
@@ -527,10 +407,8 @@ function runAllDetectors(loaded, opts = {}) {
527
407
  ];
528
408
  }
529
409
 
530
- // Canonical list of detection classes runAllDetectors can emit. The
531
- // budget gate asserts class-set equality against this list so a future
532
- // 8th detector added without a budget entry fails-closed (codex P2
533
- // PR #61).
410
+ // Every class runAllDetectors can emit. The budget gate asserts class-set equality
411
+ // against this list, so a detector added without a budget entry fails closed.
534
412
  const DETECTOR_CLASSES = [
535
413
  "content-quality",
536
414
  "temporal-staleness",