@blamejs/exceptd-skills 0.19.33 → 0.19.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/bin/exceptd.js +896 -2824
  3. package/data/_indexes/_meta.json +2 -2
  4. package/lib/auto-discovery.js +56 -286
  5. package/lib/canonical-eq.js +7 -40
  6. package/lib/citation-resolve.js +22 -70
  7. package/lib/collectors/ai-api.js +20 -54
  8. package/lib/collectors/cicd-pipeline-compromise.js +40 -108
  9. package/lib/collectors/citation-hygiene.js +72 -210
  10. package/lib/collectors/containers.js +41 -130
  11. package/lib/collectors/cred-stores.js +31 -115
  12. package/lib/collectors/crypto-codebase.js +55 -138
  13. package/lib/collectors/crypto.js +24 -54
  14. package/lib/collectors/hardening.js +20 -78
  15. package/lib/collectors/kernel.js +16 -46
  16. package/lib/collectors/library-author.js +57 -206
  17. package/lib/collectors/mcp.js +24 -70
  18. package/lib/collectors/runtime.js +24 -86
  19. package/lib/collectors/sbom.js +34 -106
  20. package/lib/collectors/scan-excludes.js +31 -138
  21. package/lib/collectors/secrets.js +62 -178
  22. package/lib/cross-ref-api.js +39 -123
  23. package/lib/currency-severity.js +8 -27
  24. package/lib/cve-batch.js +13 -21
  25. package/lib/cve-cli.js +13 -20
  26. package/lib/cve-curation.js +72 -239
  27. package/lib/cve-regression-watcher.js +29 -152
  28. package/lib/cvss.js +13 -54
  29. package/lib/doctor-bucketing.js +3 -19
  30. package/lib/exit-codes.js +10 -42
  31. package/lib/flag-suggest.js +7 -25
  32. package/lib/framework-gap.js +35 -114
  33. package/lib/gap-detectors.js +37 -159
  34. package/lib/id-validation.js +9 -30
  35. package/lib/job-queue.js +13 -36
  36. package/lib/lint-skills.js +64 -232
  37. package/lib/playbook-runner.js +693 -2095
  38. package/lib/prefetch.js +100 -376
  39. package/lib/refresh-external.js +199 -627
  40. package/lib/refresh-network.js +75 -307
  41. package/lib/rfc-cli.js +23 -68
  42. package/lib/scoring.js +77 -145
  43. package/lib/sign.js +43 -229
  44. package/lib/source-advisories.js +43 -194
  45. package/lib/source-ghsa.js +37 -120
  46. package/lib/source-osv.js +94 -266
  47. package/lib/ttp-mapper.js +14 -24
  48. package/lib/upstream-check-cli.js +10 -28
  49. package/lib/upstream-check.js +19 -44
  50. package/lib/validate-catalog-meta.js +17 -61
  51. package/lib/validate-cve-catalog.js +43 -119
  52. package/lib/validate-indexes.js +25 -76
  53. package/lib/validate-package.js +16 -62
  54. package/lib/validate-playbooks.js +69 -275
  55. package/lib/validate-vendor.js +16 -49
  56. package/lib/verify.js +56 -286
  57. package/lib/version-pins.js +5 -34
  58. package/lib/worker-pool.js +11 -30
  59. package/lib/xml-tokenizer.js +47 -152
  60. package/manifest.json +53 -53
  61. package/orchestrator/dispatcher.js +17 -68
  62. package/orchestrator/event-bus.js +11 -74
  63. package/orchestrator/index.js +138 -412
  64. package/orchestrator/pipeline.js +28 -85
  65. package/orchestrator/scanner.js +34 -138
  66. package/orchestrator/scheduler.js +20 -84
  67. package/package.json +1 -1
  68. package/sbom.cdx.json +241 -241
  69. package/scripts/audit-catalog-gaps.js +9 -62
  70. package/scripts/audit-cross-skill.js +5 -31
  71. package/scripts/audit-perf.js +6 -16
  72. package/scripts/backfill-theater-test.js +7 -64
  73. package/scripts/bootstrap.js +12 -44
  74. package/scripts/build-indexes.js +40 -154
  75. package/scripts/builders/activity-feed.js +4 -14
  76. package/scripts/builders/catalog-summaries.js +3 -10
  77. package/scripts/builders/currency.js +7 -20
  78. package/scripts/builders/cwe-chains.js +7 -30
  79. package/scripts/builders/did-ladders.js +6 -13
  80. package/scripts/builders/frequency.js +5 -19
  81. package/scripts/builders/jurisdiction-clocks.js +6 -25
  82. package/scripts/builders/recipes.js +6 -14
  83. package/scripts/builders/section-offsets.js +13 -51
  84. package/scripts/builders/stale-content.js +7 -28
  85. package/scripts/builders/summary-cards.js +8 -29
  86. package/scripts/builders/theater-fingerprints.js +12 -27
  87. package/scripts/builders/token-budget.js +4 -31
  88. package/scripts/check-agents-md-collectors.js +11 -54
  89. package/scripts/check-catalog-gap-budget.js +15 -32
  90. package/scripts/check-changelog-extract.js +18 -48
  91. package/scripts/check-codebase-patterns-currency.js +6 -22
  92. package/scripts/check-codebase-patterns.js +50 -143
  93. package/scripts/check-epss-consistency.js +9 -64
  94. package/scripts/check-framework-gap-coverage.js +13 -31
  95. package/scripts/check-manifest-snapshot.js +13 -73
  96. package/scripts/check-sbom-currency.js +44 -142
  97. package/scripts/check-test-count.js +15 -52
  98. package/scripts/check-test-coverage.js +66 -197
  99. package/scripts/check-test-subjects.js +21 -62
  100. package/scripts/check-ttp-references.js +14 -38
  101. package/scripts/check-ttp-upstream.js +8 -40
  102. package/scripts/check-version-bump.js +9 -61
  103. package/scripts/check-version-tags.js +20 -121
  104. package/scripts/predeploy.js +38 -184
  105. package/scripts/refresh-manifest-snapshot.js +16 -38
  106. package/scripts/refresh-mitre-atlas.js +3 -8
  107. package/scripts/refresh-mitre-attack.js +1 -8
  108. package/scripts/refresh-mitre-d3fend.js +3 -9
  109. package/scripts/refresh-mitre-ics-attack.js +3 -8
  110. package/scripts/refresh-reverse-refs.js +27 -94
  111. package/scripts/refresh-rfc-index.js +2 -10
  112. package/scripts/refresh-sbom.js +31 -161
  113. package/scripts/refresh-upstream-catalogs.js +40 -137
  114. package/scripts/release.js +69 -232
  115. package/scripts/run-e2e-scenarios.js +24 -71
  116. package/scripts/sync-manifest-metadata.js +10 -34
  117. package/scripts/sync-package-description.js +8 -17
  118. package/scripts/validate-vendor-online.js +13 -44
  119. package/scripts/verify-shipped-tarball.js +35 -140
@@ -1,10 +1,10 @@
1
1
  {
2
2
  "schema_version": "1.1.0",
3
- "generated_at": "2026-08-23T07:03:41.141Z",
3
+ "generated_at": "2026-08-23T11:56:28.977Z",
4
4
  "generator": "scripts/build-indexes.js",
5
5
  "source_count": 64,
6
6
  "source_hashes": {
7
- "manifest.json": "ab7c042636d0c7d610957171d74bd4e0973d8f5718315c90488333a824444a77",
7
+ "manifest.json": "592a1040e06e609cfc5fa7b690343735f4ad2f52ed3607c526b9ec057f9373eb",
8
8
  "README.md": "bc575c178c21e3d07f1491710e24b26cad3a3cee012c43641b6ea7e83468e64e",
9
9
  "data/atlas-ttps.json": "a14089671efcb4f06dd0f82d688fe49276c1370e167e4483d76e9c0a40ef41eb",
10
10
  "data/attack-techniques.json": "2179122ff8347bc691f245d5fd2601ca81fd1d6f790feae44abca86d3c08f865",
@@ -1,30 +1,7 @@
1
1
  "use strict";
2
2
  /**
3
- * lib/auto-discovery.js
4
- *
5
- * Discovers NEW catalog entries upstream and builds draft entries for
6
- * `refresh-external.js` to include as `op:"add"` diffs in its auto-PR.
7
- *
8
- * Sources covered:
9
- * - KEV: every CVE in the CISA KEV feed that's not in local
10
- * data/cve-catalog.json. NVD + EPSS data is pulled from the same
11
- * prefetch cache the drift-check uses; missing cache entries fall
12
- * through to a draft with null mechanical fields.
13
- * - RFC: every recent IETF RFC published in a working group the
14
- * project's existing rfc-references.json already cites. Queried
15
- * live against Datatracker (small N — typically 1-5 RFCs per
16
- * month across all project-relevant WGs).
17
- *
18
- * Each draft entry carries an `_auto_imported` block with the source,
19
- * import date, and a `curation_needed` list of analytical fields a
20
- * human still needs to fill (framework_control_gaps, atlas_refs,
21
- * attack_refs, type classification, etc.). `validate-cve-catalog.js`
22
- * is tolerant of this annotation; the audit / stale-content index
23
- * surfaces uncurated entries so they don't sit indefinitely.
24
- *
25
- * Both discovery functions accept a `cap` (default 100) so a burst
26
- * upstream addition doesn't generate an unreviewable PR. Items past
27
- * the cap spill to the next run.
3
+ * Discovers new upstream catalog entries (CISA KEV, IETF RFCs) as `op:"add"`
4
+ * diffs for refresh-external.js's auto-PR; items past `cap` spill to the next run.
28
5
  */
29
6
 
30
7
  const fs = require("fs");
@@ -34,16 +11,8 @@ const { scoreCustom, postWeightFactors } = require("./scoring");
34
11
  const { selectNvdCvss } = require("./cvss");
35
12
  const { deriveMechanicalFields } = require("./cve-enrich");
36
13
 
37
- // Stored rwep_factors must reproduce the stored rwep_score.
38
- // `buildScoringInputs` is the single source of truth for both — it captures
39
- // the conservative defaults applied to a freshly-imported KEV draft (CISA
40
- // only lists vulnerabilities with documented exploitation, so we assume a
41
- // public PoC exists; reboot defaults to true because most KEV-listed CVEs
42
- // land in the kernel / hypervisor / vendor firmware where reboot is the
43
- // norm). The same input object is then handed to scoreCustom for the score
44
- // AND mapped into the `rwep_factors` shape stored on the draft. Calling
45
- // scoring.validate() on the post-import catalog will no longer flag every
46
- // auto-imported draft for divergence > 5.
14
+ // Single source of truth for a fresh KEV draft's score and its stored
15
+ // rwep_factors, which must reproduce each other. The defaults read KEV conservatively.
47
16
  function buildScoringInputs(kevEntry /*, nvdPayload */) {
48
17
  void kevEntry;
49
18
  return {
@@ -59,35 +28,13 @@ function buildScoringInputs(kevEntry /*, nvdPayload */) {
59
28
  };
60
29
  }
61
30
 
62
- // cve-catalog.schema.json's `rwep_factors` requires the post-weight
63
- // numeric shape (cisa_kev: 0|25, poc_available: 0|20, ai_factor: 0|15,
64
- // active_exploitation: 0|5|10|20, blast_radius: 0..30, patch_available: 0|-15,
65
- // live_patch_available: 0|-10, reboot_required: 0|5). Pre-fix the auto-discovery
66
- // builder stored the SHAPE-A (boolean + string-ladder) factor bag — semantically
67
- // fine because deriveRwepFromFactors handles either shape, but the strict
68
- // JSON-schema validator (and any downstream tooling that types-checks the
69
- // catalog field) rejected drafts as malformed. `scoring.postWeightFactors` (the
70
- // canonical home for this conversion, also consumed by lib/cve-enrich.js's
71
- // batch-curation path) now does the boolean -> schema-required post-weight
72
- // conversion, so the nightly draft builder and the batch tool can't drift on
73
- // the math.
31
+ // cve-catalog.schema.json requires `rwep_factors` in the post-weight numeric
32
+ // shape, not the boolean + string-ladder shape scoreCustom consumes;
33
+ // `scoring.postWeightFactors` is the one converter.
74
34
 
75
35
  /**
76
- * O — diff severity nuance for KEV-discovered drafts.
77
- *
78
- * Pre-fix every KEV-derived diff carried `severity: "high"`. Operators
79
- * scanning the diff stream had no way to distinguish "patch in 21 days"
80
- * from "active ransomware campaign, patch yesterday." Now:
81
- *
82
- * - ransomware_use === "Known" → "critical" (campaigns observed in the wild)
83
- * - dueDate within 7 days of now → "critical" (CISA escalation window)
84
- * - otherwise → "high" (still actively exploited per KEV listing)
85
- *
86
- * A KEV listing inherently means active exploitation; "low" / "medium"
87
- * never apply here. The split is between "act today" and "act this sprint."
88
- *
89
- * @param {object} kevEntry
90
- * @returns {"critical" | "high"}
36
+ * Severity of a KEV-discovered diff. A KEV listing itself means active
37
+ * exploitation, so "low" and "medium" never apply.
91
38
  */
92
39
  function deriveKevSeverity(kevEntry) {
93
40
  const ransomware = String(kevEntry?.knownRansomwareCampaignUse || "").toLowerCase() === "known";
@@ -97,7 +44,6 @@ function deriveKevSeverity(kevEntry) {
97
44
  const dueMs = Date.parse(due);
98
45
  if (Number.isFinite(dueMs)) {
99
46
  const deltaMs = dueMs - Date.now();
100
- // Within the next 7 days OR already past due → critical.
101
47
  if (deltaMs <= 7 * 86_400_000) return "critical";
102
48
  }
103
49
  }
@@ -109,8 +55,7 @@ const TIMEOUT_MS = 10_000;
109
55
  const USER_AGENT = "exceptd-security/auto-discovery (+https://exceptd.com)";
110
56
  const DEFAULT_CAP = 100;
111
57
 
112
- // IETF Datatracker codes → human-readable status strings used in
113
- // data/rfc-references.json.
58
+ // IETF Datatracker codes → the status strings data/rfc-references.json stores.
114
59
  const RFC_STATUS_MAP = {
115
60
  std: "Internet Standard",
116
61
  ps: "Proposed Standard",
@@ -122,29 +67,11 @@ const RFC_STATUS_MAP = {
122
67
  unkn: "Unknown",
123
68
  };
124
69
 
125
- // Read a per-source cache payload and verify its integrity against the
126
- // signed _index.json before returning it. Values from these files flow into
127
- // auto-imported catalog drafts (CVSS / CWE / EPSS), so an unverified read
128
- // would let an attacker who dropped a forged payload between prefetch and
129
- // refresh inject false mechanical fields. Discovery cannot throw (it returns
130
- // drafts, not errors), so the fail-closed action is to return null — the
131
- // payload is treated as absent and the draft keeps its null mechanical
132
- // fields, the same fallback used when no cache file exists.
133
- //
134
- // Integrity policy, mirroring the hardened drift reader in
135
- // refresh-external.js but never throwing:
136
- // - _index.json present, has a per-entry sha256, payload matches -> return.
137
- // - _index.json present, has a per-entry sha256, payload MISMATCHES ->
138
- // return null. This is the core defense: a freshly-DISCOVERED CVE's
139
- // NVD/EPSS sidecar dropped into a cache that already carries a valid
140
- // signed index has no matching entry, and a tampered legit entry no
141
- // longer hashes — both refuse to populate.
142
- // - _index.json present but no entry for this source/id -> return null
143
- // (the inject-alongside-a-signed-cache vector).
144
- // - _index.json absent entirely -> unverified read. The --from-cache
145
- // signature gate (loadCtx) already refuses an unsigned cache before
146
- // consume unless --force-stale, so an index-less cache only reaches here
147
- // under that explicit operator override or a direct library caller.
70
+ // Reads a per-source cache payload, verified against the signed _index.json.
71
+ // Anything unverifiable returns null — the only fail-closed action available to
72
+ // a function that returns drafts, not errors. A missing index ENTRY returns null
73
+ // too (the inject-alongside-a-signed-cache vector); an absent _index.json is left
74
+ // to loadCtx's --from-cache signature gate.
148
75
  function readCachedJson(cacheDir, source, id) {
149
76
  if (!cacheDir) return null;
150
77
  const safe = String(id).replace(/[^A-Za-z0-9._-]/g, "_");
@@ -159,27 +86,23 @@ function readCachedJson(cacheDir, source, id) {
159
86
  try { idx = JSON.parse(fs.readFileSync(indexPath, "utf8")); } catch { return null; }
160
87
  const meta = idx && idx.entries && idx.entries[`${source}/${id}`];
161
88
  if (!meta || typeof meta.sha256 !== "string") return null;
162
- // The sha256 recorded at prefetch time is computed over JSON.stringify of
163
- // the parsed payload (unindented), so round-trip + re-stringify to compute
164
- // the comparable hash.
89
+ // The recorded sha256 is over JSON.stringify of the parsed payload
90
+ // (unindented), so re-stringify rather than hashing the file bytes.
165
91
  const actual = crypto.createHash("sha256").update(JSON.stringify(parsed)).digest("hex");
166
92
  if (actual !== meta.sha256) return null;
167
93
  return parsed;
168
94
  }
169
95
 
170
96
  function extractNvdMetrics(payload, id) {
171
- // Resolve the NVD vuln by id, not by position. vulnerabilities[0] blindly
172
- // took the first record, so a cache entry keyed under `id` that held another
173
- // CVE's response would attribute that CVE's CVSS/CWE/description to this id.
174
- // When no id is supplied (legacy callers), fall back to the first record.
97
+ // Resolve by id, not by position: a cache entry keyed under `id` that holds
98
+ // another CVE's response would otherwise attribute its CVSS and CWE here.
175
99
  const cves = (payload?.vulnerabilities || []).map((v) => v?.cve).filter(Boolean);
176
100
  const vuln = id
177
101
  ? cves.find((c) => c.id && String(c.id).toUpperCase() === String(id).toUpperCase())
178
102
  : cves[0];
179
103
  if (!vuln) return null;
180
- // Prefer the newest CVSS version (Primary within it) and normalize a bare
181
- // v2 vector to its canonical prefix so an auto-imported draft never carries
182
- // an unprefixed vector that the strict catalog validator would reject.
104
+ // Newest CVSS version, Primary within it, with a bare v2 vector normalized
105
+ // to its canonical prefix — the strict catalog validator rejects unprefixed.
183
106
  const up = selectNvdCvss(vuln.metrics);
184
107
  return {
185
108
  cvss_score: up ? up.baseScore : null,
@@ -190,24 +113,16 @@ function extractNvdMetrics(payload, id) {
190
113
  .map((d) => d.value)
191
114
  .filter((v) => /^CWE-\d+$/.test(v))
192
115
  ),
193
- // Carry NVD's reference list (url + tags) so buildKevDraftEntry can feed
194
- // real "Vendor Advisory"-tagged links into cve-enrich and a nightly
195
- // draft's vendor_advisories reflects genuine advisories (empty otherwise,
196
- // correctly tripping the cisa_kev-but-no-advisory curation-gap detector).
116
+ // The tags are load-bearing: cve-enrich keeps only "Vendor Advisory" links,
117
+ // so an empty result correctly trips the no-advisory curation-gap detector.
197
118
  references: (vuln.references || []).map((r) => ({ url: r.url, tags: r.tags || [] })),
198
119
  };
199
120
  }
200
121
 
201
122
  function extractEpss(payload, id) {
202
123
  const data = Array.isArray(payload?.data) ? payload.data : [];
203
- // Match the requested CVE id only. The prior `|| data[0]` fallback silently
204
- // misattributed a DIFFERENT CVE's EPSS score into the draft whenever the
205
- // requested id was absent from the payload — a single-row response for the
206
- // wrong CVE (or a stale/mismatched sidecar) wrote that CVE's score under the
207
- // discovered id. A non-matching payload now yields null (same
208
- // null-mechanical-field behavior used when no cache exists). The only
209
- // accepted fallback is a single-row, cve-less response shape (the requested
210
- // id is implicit because the payload carries exactly one un-keyed row).
124
+ // Match the requested id only — a `|| data[0]` fallback writes another CVE's
125
+ // score here. The one exception is a single-row, cve-less payload.
211
126
  let row = data.find((r) => r?.cve === id);
212
127
  if (!row && data.length === 1 && data[0]?.cve == null) row = data[0];
213
128
  if (!row) return null;
@@ -218,43 +133,16 @@ function extractEpss(payload, id) {
218
133
  };
219
134
  }
220
135
 
221
- // --- KEV discovery -----------------------------------------------------
222
-
223
136
  /**
224
- * Build a draft CVE catalog entry from a KEV record + optional cached
225
- * NVD/EPSS payloads. Required-schema fields are populated where
226
- * mechanically derivable; analytical fields are nulled and listed in
227
- * `_auto_imported.curation_needed`.
228
- *
229
- * @param {object} kevEntry Single vulnerability from CISA KEV feed
230
- * @param {object|null} nvdPayload Cached NVD 2.0 response (or null)
231
- * @param {object|null} epssPayload Cached EPSS response (or null)
137
+ * Draft catalog entry from one KEV record plus cached NVD 2.0 and EPSS payloads,
138
+ * either of which may be null. Analytical fields stay null.
232
139
  */
233
140
  function buildKevDraftEntry(kevEntry, nvdPayload, epssPayload) {
234
141
  const id = String(kevEntry.cveID);
235
142
  const nvd = nvdPayload ? extractNvdMetrics(nvdPayload, id) : null;
236
143
  const epss = epssPayload ? extractEpss(epssPayload, id) : null;
237
144
 
238
- // Stored rwep_factors and computed rwep_score MUST agree.
239
- // Previously rwep_factors held nulls (for unknown poc/ai/reboot) but
240
- // rwep_score was computed from concrete defaults (poc=true, reboot=true).
241
- // `scoring.validate()` then flagged every auto-imported draft for
242
- // divergence > 5. Now: one canonical input object → both surfaces.
243
- //
244
- // the catalog's JSON-schema for `rwep_factors` requires the
245
- // POST-WEIGHT numeric shape (ai_factor / numeric ladder contributions /
246
- // numeric ±deductions) — not the SHAPE-A boolean + string-ladder shape
247
- // that scoreCustom consumes. Pre-fix the boolean shape was stored
248
- // verbatim, so curate-apply's strict-schema gate rejected KEV-discovered
249
- // drafts as soon as anyone tried to promote them — they were
250
- // permanently unpromotable. `scoring.postWeightFactors` (shared with the
251
- // batch-curation path in lib/cve-enrich.js) does the conversion now.
252
- //
253
- // The curation flow rewrites these once an operator answers the editorial
254
- // questions; until then, the post-weight numeric shape on rwep_factors
255
- // reproduces the score exactly (sum of values === rwep_score, because
256
- // blast_radius weight=30 matches the raw-cap convention documented in
257
- // scoring.js header).
145
+ // One input object feeds both, so the stored factors sum exactly to rwep_score.
258
146
  const scoringInputs = buildScoringInputs(kevEntry, nvdPayload);
259
147
  const rwep_factors = postWeightFactors(scoringInputs);
260
148
  const rwep_score = scoreCustom(scoringInputs);
@@ -263,15 +151,8 @@ function buildKevDraftEntry(kevEntry, nvdPayload, epssPayload) {
263
151
  .filter(Boolean)
264
152
  .join(" ");
265
153
 
266
- // Mechanical fields (name, cvss_score/vector, cwe_refs, cisa_kev +
267
- // cisa_kev_date/due_date, known_ransomware_use, complexity, vector, epss_*,
268
- // vendor_advisories, verification_sources, source_verified/last_updated)
269
- // are derived through the SAME cve-enrich module the `--curate-batch`
270
- // tool uses, so the nightly auto-PR path and the batch tool never drift on
271
- // how a raw KEV/NVD/EPSS fact set becomes a mechanical field. See
272
- // lib/cve-enrich.js:deriveMechanicalFields — this module maps the already-
273
- // extracted `nvd`/`epss` payloads (see extractNvdMetrics/extractEpss above)
274
- // and the raw `kevEntry` into that module's `facts` shape.
154
+ // deriveMechanicalFields is shared with `--curate-batch`, so the nightly path
155
+ // and the batch tool cannot drift. What follows is that module's `facts` shape.
275
156
  const facts = {
276
157
  id,
277
158
  nvd_desc: nvd && nvd.description,
@@ -301,11 +182,8 @@ function buildKevDraftEntry(kevEntry, nvdPayload, epssPayload) {
301
182
  ai_discovered: null,
302
183
  ai_discovery_notes: null,
303
184
  ai_assisted_weaponization: null,
304
- // Conservative pre-curation default — overrides deriveMechanicalFields'
305
- // 'confirmed' (mechanically derived from "this CVE has a KEV listing").
306
- // A nightly draft must not claim CONFIRMED active exploitation before a
307
- // human has reviewed it; 'suspected' is the honest starting posture even
308
- // though KEV listing alone already implies exploitation is real.
185
+ // Overrides deriveMechanicalFields' 'confirmed': a draft must not claim
186
+ // confirmed exploitation before a human has reviewed it.
309
187
  active_exploitation: "suspected",
310
188
  affected: product || "See vendor advisory",
311
189
  affected_versions: [],
@@ -321,12 +199,8 @@ function buildKevDraftEntry(kevEntry, nvdPayload, epssPayload) {
321
199
  rwep_score,
322
200
  rwep_factors,
323
201
  last_verified: TODAY,
324
- // v0.12.15 (D): `_auto_imported` must be the boolean `true`
325
- // for lib/validate-cve-catalog.js's draft-recognition check (strict
326
- // `=== true` comparison). The prior object-shape was non-recognizable
327
- // and the strict validator treated KEV-discovered drafts as
328
- // hard-error entries instead of warning-tier drafts. The provenance
329
- // metadata that used to be inline now lives in `_auto_imported_meta`.
202
+ // Boolean `true`, not an object: lib/validate-cve-catalog.js recognizes a
203
+ // draft by strict `=== true`. Provenance goes in `_auto_imported_meta`.
330
204
  _auto_imported: true,
331
205
  _auto_imported_meta: {
332
206
  source: "KEV discovery",
@@ -348,10 +222,8 @@ function buildKevDraftEntry(kevEntry, nvdPayload, epssPayload) {
348
222
  }
349
223
 
350
224
  /**
351
- * Find KEV entries upstream that are not in local cve-catalog.json.
352
- * Returns an array of { id, op:"add", entry, severity } diffs capped
353
- * at `cap` items. Spill past the cap is logged on the diff object's
354
- * `_spilled` count so the PR body can mention it.
225
+ * KEV entries upstream that are absent from local cve-catalog.json, as
226
+ * { id, op:"add", entry, severity } diffs capped at `cap`; overflow is `spilled`.
355
227
  */
356
228
  function discoverNewKev(ctx, cap = DEFAULT_CAP) {
357
229
  const feed = readCachedJson(ctx.cacheDir, "kev", "known_exploited_vulnerabilities");
@@ -363,8 +235,7 @@ function discoverNewKev(ctx, cap = DEFAULT_CAP) {
363
235
  Object.keys(ctx.cveCatalog).filter((k) => /^CVE-\d{4}-\d{4,7}$/.test(k))
364
236
  );
365
237
 
366
- // Sort by dateAdded descending so the most recent additions are kept
367
- // when the cap clips the list.
238
+ // Newest dateAdded first, so the cap clips the oldest additions.
368
239
  const candidates = feed.vulnerabilities
369
240
  .filter((v) => v && v.cveID && !localCves.has(String(v.cveID)))
370
241
  .sort((a, b) => String(b.dateAdded || "").localeCompare(String(a.dateAdded || "")));
@@ -402,28 +273,14 @@ function discoverNewKev(ctx, cap = DEFAULT_CAP) {
402
273
  };
403
274
  }
404
275
 
405
- // --- RFC discovery -----------------------------------------------------
406
-
407
276
  async function fetchDatatracker(url, ctx) {
408
- // Air-gap refusal — Datatracker is a live IETF service. When the operator
409
- // (or the caller's ctx) declares air-gap, return a structured refusal so
410
- // refresh-external can surface "discovery skipped" rather than logging a
411
- // generic network error. Caller's signature already accepts a nullable
412
- // return; the structured object is distinguishable from a successful
413
- // payload by `ok:false`.
277
+ // A structured `ok:false` the caller tells apart from a payload, so an air-gap
278
+ // refusal reports as a skip rather than a fetch error.
414
279
  if ((ctx && ctx.airGap === true) || process.env.EXCEPTD_AIR_GAP === "1") {
415
280
  return { ok: false, error: "air-gap-blocked", source: "datatracker" };
416
281
  }
417
- // Adopt the vendored blamejs retry primitive, matching source-osv /
418
- // source-ghsa / source-advisories / refresh-network: a transient Datatracker
419
- // failure (HTTP 429/5xx, ECONNRESET/ETIMEDOUT family) backs off with
420
- // exponential jitter instead of being dropped on the first hiccup, which
421
- // silently yielded no discovered working groups. A genuine 4xx (e.g. 404 on
422
- // a renamed WG acronym) carries a non-retryable statusCode, so the classifier
423
- // does not waste attempts on it. The caller contract is unchanged: discovery
424
- // never throws — it returns drafts, not errors — so an exhausted retry budget
425
- // (or any non-retryable failure) collapses to the same `null` the
426
- // single-fetch path returned, which discoverNewRfcs treats as one WG skip.
282
+ // Transient failures (429/5xx, ECONNRESET/ETIMEDOUT) back off with jitter.
283
+ // Discovery never throws, so an exhausted budget collapses to null.
427
284
  const { withRetry } = require("../vendor/blamejs/retry.js");
428
285
  const once = async () => {
429
286
  const ac = new AbortController();
@@ -451,74 +308,8 @@ async function fetchDatatracker(url, ctx) {
451
308
  }
452
309
 
453
310
  /**
454
- * Derive the set of IETF working-group acronyms the project already
455
- * cares about. Reads each entry in data/rfc-references.json, looks up
456
- * its Datatracker doc in the prefetch cache, extracts the group
457
- * acronym, returns the union.
458
- *
459
- * Two layers in the result:
460
- *
461
- * 1. DYNAMICALLY DERIVED — every WG that appears on a project-cited
462
- * RFC's Datatracker record. Grows organically as catalog grows.
463
- *
464
- * 2. SEEDED — a curated baseline of IETF WGs that publish RFCs
465
- * directly relevant to the project's mid-2026 threat model and
466
- * compliance frameworks, even when the catalog doesn't yet cite
467
- * one of their RFCs. Without this, RFC discovery would be blind
468
- * to e.g. SCITT (supply chain) until a SCITT RFC was already
469
- * manually added — defeating the point of discovery.
470
- *
471
- * SEED groups by project area:
472
- *
473
- * Transport / crypto / PKI:
474
- * tls, uta, cfrg, lamps, ipsecme
475
- * HTTP / web / QUIC / API:
476
- * httpbis, quic, ohai, privacypass, httpapi, core
477
- * Identity / auth / SSO / cert mgmt / workload identity / constrained-env auth:
478
- * oauth, gnap, jose, cose, cbor, kitten, emu, secevent, scim,
479
- * acme, wimse, ace
480
- * DNS security + privacy + DNS-based auth:
481
- * dnsop, dprive, add, dance
482
- * Supply chain + attestation + transparency + firmware/TEE:
483
- * scitt, rats, suit, teep, trans
484
- * Threat intel + security automation + operational telemetry:
485
- * mile, sacm, i2nsf, opsawg, opsec
486
- * Messaging + E2E + media:
487
- * mls, moq, sframe
488
- * Network / IoT mgmt + audit-grade time sync:
489
- * anima, drip, iotops, netconf, netmod, ntp
490
- * Data / schema / policy serialization:
491
- * jsonschema
492
- *
493
- * Database protocols themselves (Postgres wire, MongoDB wire, etc.)
494
- * aren't IETF-standardized, so there's no "database" WG. The security
495
- * infrastructure databases USE — TLS for connections (tls/uta/lamps),
496
- * SASL/Kerberos auth (kitten/emu), workload identity (wimse), field
497
- * encryption (cose/cfrg/cbor), audit-trail time (ntp), cert validation
498
- * (lamps/dance/trans), and access-control sync (scim/oauth) — is all
499
- * already covered by the WGs above. jsonschema covers the DB+API+policy
500
- * schema validation layer.
501
- *
502
- * Reasoning for additions over the v0.9.2 seed:
503
- * - wimse (Workload Identity in Multi-System Environments): federal
504
- * zero-trust mandates + cloud-native workload identity are core to
505
- * identity-assurance + sector-federal-government skills.
506
- * - gnap (Grant Negotiation): OAuth successor; identity-assurance
507
- * skill will eventually cite this.
508
- * - ace + core: auth + REST for constrained environments — OT/ICS
509
- * and IoT supply chain.
510
- * - cbor: foundation for COSE, attestation tokens, SCITT receipts.
511
- * Touches MCP trust + supply-chain integrity + RATS attestation.
512
- * - trans (Certificate Transparency): compliance evidence for cert
513
- * issuance; cross-cuts identity + framework-gap analysis.
514
- * - ntp: audit trails need monotonic time. Required for the audit
515
- * skills + breach-notification clocks (DORA, NYDFS, NIS2).
516
- * - opsawg + opsec: operational security guidance and telemetry.
517
- * Touches incident-response + threat-model-currency.
518
- * - dance: DANE Authentication for Named Entities Enhancements —
519
- * adds DNS-anchored TLS trust (complements lamps PKI).
520
- * - netmod: NETCONF YANG models, often security-relevant for OT/ICS
521
- * and network-segmentation policy.
311
+ * Seed WGs, unioned with those on project-cited RFCs: without the seed,
312
+ * discovery stays blind to a WG until one of its RFCs is added by hand.
522
313
  */
523
314
  const SEED_RFC_GROUPS = [
524
315
  // Transport / crypto / PKI
@@ -538,8 +329,7 @@ const SEED_RFC_GROUPS = [
538
329
  "mls", "moq", "sframe",
539
330
  // Network / IoT mgmt + audit-grade time sync
540
331
  "anima", "drip", "iotops", "netconf", "netmod", "ntp",
541
- // Data / schema / policy serialization (DB validation, API schemas,
542
- // security-policy contract languages)
332
+ // Data / schema / policy serialization
543
333
  "jsonschema",
544
334
  ];
545
335
 
@@ -556,35 +346,25 @@ function getProjectRfcGroups(ctx) {
556
346
  const acronym = obj?.group?.acronym || (typeof obj?.group === "string" ? extractAcronymFromGroupUri(obj.group) : null);
557
347
  if (acronym) groups.add(String(acronym).toLowerCase());
558
348
  }
559
- // Always union the seed list — dynamic derivation covers the WGs we
560
- // already cite; the seed covers WGs we SHOULD watch for our skill
561
- // coverage even if no RFC from that WG is in the catalog yet.
562
349
  for (const g of SEED_RFC_GROUPS) groups.add(g);
563
350
  return groups;
564
351
  }
565
352
 
566
353
  function extractAcronymFromGroupUri(uri) {
567
- // Group URIs from Datatracker look like /api/v1/group/group/12345/.
568
- // Group acronym is in the doc object's full record but not in the URI.
569
- // Returning null means we have to live-fetch later.
354
+ // A Datatracker group URI (/api/v1/group/group/12345/) carries no acronym —
355
+ // it is only in the doc object's full record. Null means live-fetch later.
570
356
  void uri;
571
357
  return null;
572
358
  }
573
359
 
574
360
  /**
575
- * Find recent RFCs published in any project-relevant working group
576
- * that aren't already in data/rfc-references.json. Queries Datatracker
577
- * live (small N — runs once per refresh, ~9 WG queries).
578
- *
579
- * @param {object} ctx
580
- * @param {object} opts { cap?: number, sinceDays?: number }
361
+ * Recent RFCs in project-relevant working groups that are not already in
362
+ * data/rfc-references.json. Queries Datatracker live, one call per WG.
363
+ * @param {object} opts { cap = DEFAULT_CAP, sinceDays = 180 }
581
364
  */
582
365
  async function discoverNewRfcs(ctx, opts = {}) {
583
- // Air-gap: new-RFC discovery queries IETF Datatracker live (~one call per
584
- // project working group). Refuse the egress under --air-gap so a fully
585
- // offline run makes no network calls — the other discovery paths guard the
586
- // same way. --from-cache alone (network-available host, e.g. the scheduled
587
- // refresh) still discovers live; add --air-gap for a truly offline run.
366
+ // --air-gap refuses the egress outright; --from-cache alone still queries
367
+ // live, since the scheduled refresh runs on a networked host.
588
368
  if ((ctx && ctx.airGap === true) || process.env.EXCEPTD_AIR_GAP === "1") {
589
369
  return { diffs: [], errors: 0, spilled: 0, summary: "RFC discovery: skipped under air-gap (no live Datatracker query)" };
590
370
  }
@@ -603,18 +383,13 @@ async function discoverNewRfcs(ctx, opts = {}) {
603
383
  let errors = 0;
604
384
 
605
385
  for (const wg of groups) {
606
- // Datatracker filter: RFCs in this WG, time > cutoff. Ordered by time descending.
607
386
  const url =
608
387
  `https://datatracker.ietf.org/api/v1/doc/document/` +
609
388
  `?type=rfc&group__acronym=${encodeURIComponent(wg)}` +
610
389
  `&time__gt=${cutoff}&order_by=-time&limit=20&format=json`;
611
390
  const payload = await fetchDatatracker(url, ctx);
612
- // Treat the air-gap structured refusal as a discovery skip — not an
613
- // error. The existing `!payload || !Array.isArray(payload.objects)`
614
- // path would have classed it as `errors++`, which misreports an
615
- // operator-chosen offline posture as a fault. Short-circuit the whole
616
- // WG loop on first air-gap refusal because every subsequent fetch will
617
- // refuse identically.
391
+ // An air-gap refusal is a skip, not an `errors++` fault, and every later
392
+ // fetch refuses identically — leave the WG loop on the first one.
618
393
  if (payload && payload.ok === false && payload.error === "air-gap-blocked") {
619
394
  return {
620
395
  diffs: [],
@@ -639,8 +414,7 @@ async function discoverNewRfcs(ctx, opts = {}) {
639
414
  }
640
415
  }
641
416
 
642
- // Dedupe across overlapping WG membership (an RFC can list multiple
643
- // groups). Keep the first occurrence (alphabetically first WG match).
417
+ // An RFC can list several groups; keep the first WG match.
644
418
  const seen = new Set();
645
419
  candidates = candidates.filter((c) => {
646
420
  if (seen.has(c.localKey)) return false;
@@ -648,7 +422,7 @@ async function discoverNewRfcs(ctx, opts = {}) {
648
422
  return true;
649
423
  });
650
424
 
651
- // Sort by published time descending so we keep the most recent under the cap.
425
+ // Newest published first, so the cap clips the oldest.
652
426
  candidates.sort((a, b) => String(b.obj.time || "").localeCompare(String(a.obj.time || "")));
653
427
 
654
428
  const total = candidates.length;
@@ -668,11 +442,7 @@ async function discoverNewRfcs(ctx, opts = {}) {
668
442
  skills_referencing: [],
669
443
  errata_count: null,
670
444
  last_verified: TODAY,
671
- // v0.12.15 (D, P3-T): boolean `_auto_imported: true` for
672
- // strict-validator recognition; provenance moved to sibling
673
- // `_auto_imported_meta`. Errata-URL hint converted to a real template
674
- // literal so the rfc number actually interpolates (the previous double-
675
- // quoted string left `${number}` as literal text in operator output).
445
+ // Boolean `true` for strict-validator draft recognition.
676
446
  _auto_imported: true,
677
447
  _auto_imported_meta: {
678
448
  source: `RFC discovery (IETF ${wg} working group)`,
@@ -1,51 +1,20 @@
1
1
  "use strict";
2
2
  /**
3
- * lib/canonical-eq.js
3
+ * Canonical-form deep equality for catalog diff detection.
4
4
  *
5
- * Canonical-form deep equality for catalog diff detection. The diff-
6
- * coverage gate previously compared `JSON.stringify(before.iocs)` vs
7
- * `JSON.stringify(after.iocs)` which is non-canonical: key order,
8
- * trailing whitespace, and numeric format differences all register as
9
- * "different" when the operator made no semantic change.
10
- *
11
- * Pre-v0.13.20 history: the symptom was patched twice with skip rules
12
- * (v0.13.17 _auto_imported skip; v0.13.19 _iocs_stub skip). v0.13.20
13
- * fixes the root cause — canonical recursive equality with sorted-key
14
- * object comparison and array-position-sensitive element comparison.
15
- *
16
- * Contract:
17
- * - Primitives (string / number / boolean / null / undefined) compare
18
- * by strict equality (===).
19
- * - Arrays compare element-by-element in order. [1,2] !== [2,1].
20
- * This matches operator intent — array order in IoCs / attack_refs
21
- * / cwe_refs is meaningful (most-relevant-first convention).
22
- * - Objects compare by key-set equality + per-key recursive equality.
23
- * Key order does NOT matter; { a:1, b:2 } === { b:2, a:1 }.
24
- * - Cycle protection: WeakSet of visited pairs prevents infinite
25
- * recursion on self-referential structures. Cycles compare unequal
26
- * across mismatched topologies; equal across identical topologies.
27
- * - NaN: NaN === NaN under this comparator (deviates from Object.is
28
- * to make the comparator total — useful for catalog data which
29
- * never legitimately contains NaN but might pick one up from a
30
- * buggy upstream).
31
- *
32
- * Helpers:
33
- * - canonicalEqual(a, b): full recursive equality.
34
- * - canonicalStringify(v): sorted-key JSON for hashing / display.
35
- * Produces stable output suitable for SHA-256 etc.
5
+ * Arrays compare element-by-element in order — [1,2] !== [2,1], since position
6
+ * in iocs / attack_refs / cwe_refs carries meaning. Objects compare by key set
7
+ * and per-key value, so key order is irrelevant. NaN equals NaN, which makes
8
+ * the comparator total. A pair already under comparison compares equal, so a
9
+ * cycle leaves the verdict to the non-cyclic structure around it.
36
10
  */
37
11
 
38
12
  function canonicalEqual(a, b, seen = new WeakMap()) {
39
13
  if (a === b) return true;
40
- // NaN === NaN under this comparator.
41
14
  if (typeof a === "number" && typeof b === "number" && Number.isNaN(a) && Number.isNaN(b)) return true;
42
15
  if (a === null || b === null) return a === b;
43
16
  if (typeof a !== "object" || typeof b !== "object") return false;
44
17
 
45
- // Cycle detection — if we've already compared this exact pair, treat
46
- // as equal (assumes the rest of the structure decides). For sibling-
47
- // cycle differences this means the comparator says "equal at the
48
- // cycle point" and lets non-cyclic differences elsewhere decide.
49
18
  const aSeen = seen.get(a);
50
19
  if (aSeen && aSeen.has(b)) return true;
51
20
  if (!aSeen) seen.set(a, new WeakSet([b]));
@@ -63,7 +32,6 @@ function canonicalEqual(a, b, seen = new WeakMap()) {
63
32
  return true;
64
33
  }
65
34
 
66
- // Plain objects — compare key sets + per-key recursive equality.
67
35
  const aKeys = Object.keys(a).sort();
68
36
  const bKeys = Object.keys(b).sort();
69
37
  if (aKeys.length !== bKeys.length) return false;
@@ -76,8 +44,7 @@ function canonicalEqual(a, b, seen = new WeakMap()) {
76
44
  return true;
77
45
  }
78
46
 
79
- // Sorted-key recursive JSON. Stable output for hash digests, diff
80
- // comparison, and human-readable display.
47
+ // Sorted-key recursive JSON — stable bytes for hash digests and diffing.
81
48
  function canonicalStringify(v) {
82
49
  if (v === null || typeof v !== "object") return JSON.stringify(v);
83
50
  if (Array.isArray(v)) return "[" + v.map(canonicalStringify).join(",") + "]";