@blamejs/exceptd-skills 0.18.9 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/CHANGELOG.md +36 -0
  2. package/bin/exceptd.js +204 -118
  3. package/data/_indexes/_meta.json +3 -3
  4. package/data/_indexes/frequency.json +2 -2
  5. package/data/d3fend-catalog.json +6 -6
  6. package/data/playbooks/identity-sso-compromise.json +2 -2
  7. package/data/playbooks/sbom.json +1 -1
  8. package/lib/citation-resolve.js +11 -0
  9. package/lib/collectors/containers.js +13 -0
  10. package/lib/collectors/cred-stores.js +18 -9
  11. package/lib/collectors/secrets.js +4 -2
  12. package/lib/cross-ref-api.js +29 -7
  13. package/lib/cve-regression-watcher.js +47 -15
  14. package/lib/framework-gap.js +52 -19
  15. package/lib/gap-detectors.js +8 -3
  16. package/lib/lint-skills.js +3 -2
  17. package/lib/playbook-runner.js +125 -7
  18. package/lib/refresh-external.js +58 -7
  19. package/lib/refresh-network.js +18 -5
  20. package/lib/rfc-cli.js +113 -18
  21. package/lib/schemas/playbook.schema.json +1 -1
  22. package/lib/scoring.js +71 -8
  23. package/lib/source-advisories.js +58 -9
  24. package/lib/ttp-mapper.js +31 -3
  25. package/lib/upstream-check-cli.js +13 -1
  26. package/lib/validate-catalog-meta.js +51 -7
  27. package/lib/validate-cve-catalog.js +10 -0
  28. package/lib/validate-playbooks.js +19 -1
  29. package/lib/verify.js +35 -34
  30. package/lib/xml-tokenizer.js +187 -25
  31. package/manifest.json +53 -53
  32. package/orchestrator/dispatcher.js +53 -9
  33. package/orchestrator/index.js +9 -7
  34. package/orchestrator/pipeline.js +62 -14
  35. package/orchestrator/scanner.js +60 -9
  36. package/package.json +1 -1
  37. package/sbom.cdx.json +115 -100
  38. package/scripts/build-indexes.js +21 -3
  39. package/scripts/builders/cwe-chains.js +5 -2
  40. package/scripts/builders/section-offsets.js +17 -8
  41. package/scripts/builders/summary-cards.js +12 -4
  42. package/scripts/check-catalog-gap-budget.js +3 -3
  43. package/scripts/check-codebase-patterns-currency.js +1 -0
  44. package/scripts/check-codebase-patterns.js +166 -11
  45. package/scripts/check-sbom-currency.js +69 -3
  46. package/scripts/check-test-count.js +28 -16
  47. package/scripts/check-test-subjects.js +148 -0
  48. package/scripts/check-version-tags.js +24 -5
  49. package/scripts/predeploy.js +32 -8
  50. package/scripts/refresh-upstream-catalogs.js +169 -44
  51. package/scripts/release.js +28 -11
@@ -788,10 +788,12 @@ const { ADVISORIES_SOURCE } = require('./source-advisories');
788
788
  // detection method that surfaces poller-diff historical-CVE references as
789
789
  // candidate silent-regression cases (the MiniPlasma class — a 2026 PoC
790
790
  // drop that re-broke CVE-2020-17103 without any new ID being assigned).
791
- // Report-only; consumes diffs + extracted-CVE-id list from a prior
792
- // advisories run (loadCtx populates ctx.advisoriesDiffs +
793
- // ctx.advisoriesExtractedCveIds when advisories runs alongside the
794
- // watcher, or operators can chain explicitly via the source registry).
791
+ // Report-only; consumes the prior advisories run's output. The main()
792
+ // source loop threads the advisories fetchDiff() result onto
793
+ // ctx.advisoriesObservations (preferred) + ctx.advisoriesDiffs (fallback)
794
+ // after the advisories source resolves and before the watcher runs, so the
795
+ // two must be invoked in that order (advisories first). Under --swarm the
796
+ // watcher runs in a second pass after the parallel batch resolves.
795
797
  const { REGRESSION_WATCHER_SOURCE } = require('./cve-regression-watcher');
796
798
 
797
799
  const ALL_SOURCES = {
@@ -1795,9 +1797,53 @@ async function main() {
1795
1797
  return { src, diff };
1796
1798
  };
1797
1799
 
1798
- const outcomes = opts.swarm
1799
- ? await Promise.all(sources.map(runOne))
1800
- : await sequential(sources, runOne);
1800
+ // NEW-CTRL-074 chaining. cve-regression-watcher consumes the advisories
1801
+ // source's per-feed CVE observations (preferred — includes in-catalog
1802
+ // historical IDs the annotate verdict needs) and falls back to its diffs.
1803
+ // The orchestrator otherwise runs every source independently and only
1804
+ // persists outputs into the report, never back onto ctx — so without this
1805
+ // thread the watcher always sees empty input and emits zero candidates.
1806
+ const threadAdvisoriesIntoCtx = (src, diff) => {
1807
+ if (src && src.name === "advisories" && diff && !diff.air_gap_blocked) {
1808
+ if (Array.isArray(diff.observations)) ctx.advisoriesObservations = diff.observations;
1809
+ if (Array.isArray(diff.diffs)) ctx.advisoriesDiffs = diff.diffs;
1810
+ }
1811
+ };
1812
+
1813
+ let outcomes;
1814
+ if (!opts.swarm) {
1815
+ // Sequential: thread the advisories output onto ctx the instant it
1816
+ // resolves, BEFORE the next source's fetchDiff(ctx) is invoked.
1817
+ outcomes = [];
1818
+ for (const src of sources) {
1819
+ const outcome = await runOne(src);
1820
+ if (!outcome.error) threadAdvisoriesIntoCtx(outcome.src, outcome.diff);
1821
+ outcomes.push(outcome);
1822
+ }
1823
+ } else {
1824
+ // --swarm: advisories and the watcher race when run via the same
1825
+ // Promise.all, so chaining via shared ctx cannot work. Split the
1826
+ // watcher into a second pass: run every other source in parallel,
1827
+ // thread the resolved advisories observations onto ctx, then run the
1828
+ // watcher. If advisories is NOT also selected, the watcher runs in the
1829
+ // first batch like any other source (its empty-input contract is the
1830
+ // operator's choice, not a silent race).
1831
+ const hasAdvisories = sources.some((s) => s.name === "advisories");
1832
+ const watcher = hasAdvisories
1833
+ ? sources.find((s) => s.name === "cve-regression-watcher")
1834
+ : null;
1835
+ const firstBatch = watcher ? sources.filter((s) => s !== watcher) : sources;
1836
+ outcomes = await Promise.all(firstBatch.map(runOne));
1837
+ if (watcher) {
1838
+ for (const o of outcomes) {
1839
+ if (!o.error) threadAdvisoriesIntoCtx(o.src, o.diff);
1840
+ }
1841
+ const watcherOutcome = await runOne(watcher);
1842
+ // Preserve the operator's declared source order in the report.
1843
+ const idx = sources.indexOf(watcher);
1844
+ outcomes.splice(idx, 0, watcherOutcome);
1845
+ }
1846
+ }
1801
1847
 
1802
1848
  // Cache-integrity refusals (sha256 mismatch, missing/partial _index.json,
1803
1849
  // unindexed payload) are thrown by readCachedJson with _exceptd_exit_code=4
@@ -1832,6 +1878,11 @@ async function main() {
1832
1878
  // marker through to the persisted report so stdout-parsing consumers
1833
1879
  // and the regression test can verify the network refusal happened.
1834
1880
  ...(diff.air_gap_blocked ? { air_gap_blocked: true } : {}),
1881
+ // NEW-CTRL-074: persist the per-source _meta (the watcher stamps
1882
+ // input_field_used here so the chaining is observable in the report)
1883
+ // and the advisories observations[] the watcher consumes.
1884
+ ...(diff._meta ? { _meta: diff._meta } : {}),
1885
+ ...(Array.isArray(diff.observations) ? { observations: diff.observations } : {}),
1835
1886
  };
1836
1887
  if (opts.apply && diff.diffs.length > 0 && !src.report_only) {
1837
1888
  const r = await src.applyDiff(ctx, diff.diffs);
@@ -152,9 +152,17 @@ function getJson(url, timeoutMs) {
152
152
  const ALLOWED_TARBALL_HOST = /(?:^|\.)npmjs\.org$|(?:^|\.)npmjs\.com$/;
153
153
 
154
154
  // Exported for in-process tests of the fetch-destination guard.
155
+ // The guard parses the URL once and rejects any non-default port BEFORE the
156
+ // hostname test, so validation and the subsequent https.get({host,path})
157
+ // connect (which reuses u.host — port-inclusive) agree on the same value.
158
+ // Without the port check a `registry.npmjs.org:9999` URL would pass the
159
+ // hostname-only allowlist yet connect to the attacker-chosen port.
155
160
  function isAllowedTarballHost(url) {
156
- try { return ALLOWED_TARBALL_HOST.test(new URL(url).hostname.toLowerCase()); }
157
- catch { return false; }
161
+ try {
162
+ const u = new URL(url);
163
+ if (!(u.port === "" || u.port === "443")) return false;
164
+ return ALLOWED_TARBALL_HOST.test(u.hostname.toLowerCase());
165
+ } catch { return false; }
158
166
  }
159
167
 
160
168
  function getBufferOnce(url, timeoutMs) {
@@ -403,9 +411,14 @@ async function main() {
403
411
 
404
412
  // Air-gap refusal. --network needs egress to registry.npmjs.org for the
405
413
  // /latest metadata + tarball pull. Under air-gap there is no offline
406
- // substitute (the test fixture path remains available for offline tests),
407
- // so refuse before any network attempt and point at the offline workflow.
408
- if (opts.airGap && !process.env.EXCEPTD_REGISTRY_FIXTURE) {
414
+ // substitute, so refuse before any network attempt and point at the
415
+ // offline workflow. The refusal is UNCONDITIONAL w.r.t.
416
+ // EXCEPTD_REGISTRY_FIXTURE: a present metadata fixture only stubs the
417
+ // /latest read, not the subsequent tarball fetch (getBuffer still hits
418
+ // canonicalUrl), so honoring the fixture under air-gap would let egress
419
+ // happen anyway while silently disabling an operator-facing control.
420
+ // Offline tests exercise the metadata+tarball path WITHOUT --air-gap.
421
+ if (opts.airGap) {
409
422
  emit({
410
423
  ok: false,
411
424
  source: "air-gap",
package/lib/rfc-cli.js CHANGED
@@ -13,7 +13,90 @@
13
13
 
14
14
  const { resolveRfc } = require("./citation-resolve.js");
15
15
 
16
- (async () => {
16
+ // Stopwords that don't disambiguate one RFC title from another. A claimed title
17
+ // run preceded by one of these in the index title is still a clean match; a run
18
+ // preceded by a CONTENT word (e.g. "datagram" before "transport layer security")
19
+ // is the tail of a more-specific title and must NOT be accepted as a match.
20
+ const TITLE_STOPWORDS = new Set(["the", "a", "an", "of", "for", "to", "in", "on", "and", "or"]);
21
+
22
+ function normTitle(s) {
23
+ return String(s).toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
24
+ }
25
+
26
+ /**
27
+ * Decide whether a claimed RFC title matches the authoritative index title.
28
+ *
29
+ * Replaces the old lenient bidirectional substring test (`a.includes(b) ||
30
+ * b.includes(a)`), which let "TLS" match the DTLS title (substring of "dtls")
31
+ * and let "Transport Layer Security" match the DTLS title (tail-of-phrase).
32
+ * The comparison is now whole-word and phrase-aware:
33
+ *
34
+ * 1. Every claimed token must appear as a WHOLE word in the index title
35
+ * (so "tls" never matches inside "dtls").
36
+ * 2. The claimed token sequence must appear as a CONTIGUOUS run in the index
37
+ * title, OR the claim must cover enough of the index title (containment
38
+ * ratio floor) to be unambiguous.
39
+ * 3. A contiguous run that is immediately preceded by a distinguishing
40
+ * CONTENT word in the index title is rejected — it is the tail of a
41
+ * more-specific title (the "datagram transport layer security" trap).
42
+ *
43
+ * Returns true / false. Only called when both a claim and an index title exist.
44
+ */
45
+ function titleMatches(claimed, indexTitle) {
46
+ const claimTokens = normTitle(claimed).split(" ").filter(Boolean);
47
+ const titleTokens = normTitle(indexTitle).split(" ").filter(Boolean);
48
+ if (claimTokens.length === 0 || titleTokens.length === 0) return false;
49
+
50
+ // (1) Whole-word containment: every claimed token must be a standalone token
51
+ // in the index title. Kills the tls-inside-dtls substring false positive.
52
+ const titleSet = new Set(titleTokens);
53
+ for (const t of claimTokens) {
54
+ if (!titleSet.has(t)) return false;
55
+ }
56
+
57
+ // Find every contiguous run of the claim inside the index title.
58
+ const runStarts = [];
59
+ for (let i = 0; i + claimTokens.length <= titleTokens.length; i++) {
60
+ let hit = true;
61
+ for (let j = 0; j < claimTokens.length; j++) {
62
+ if (titleTokens[i + j] !== claimTokens[j]) { hit = false; break; }
63
+ }
64
+ if (hit) runStarts.push(i);
65
+ }
66
+
67
+ if (runStarts.length > 0) {
68
+ // A single-token claim that is a whole word in the title is unambiguous on
69
+ // its own — the whole-word check above already excluded the substring trap
70
+ // (e.g. "tls" is NOT a token inside "dtls"), so "TLS" correctly matches the
71
+ // 8446 title (standalone "tls" token) but not the 9147 DTLS title.
72
+ if (claimTokens.length === 1) return true;
73
+ // (3) For a MULTI-token run, accept only if at least one occurrence is NOT
74
+ // preceded by a distinguishing content word — i.e. it begins the title
75
+ // or is preceded only by a stopword. A run preceded solely by a content
76
+ // qualifier (e.g. "datagram" before "transport layer security") is the
77
+ // tail of a more-specific title and must not be accepted as a match.
78
+ for (const start of runStarts) {
79
+ if (start === 0) return true;
80
+ const prev = titleTokens[start - 1];
81
+ if (TITLE_STOPWORDS.has(prev)) return true;
82
+ }
83
+ return false;
84
+ }
85
+
86
+ // No contiguous run, but all tokens present out of order. Accept only when the
87
+ // claim covers a strong majority of the index title's tokens (containment
88
+ // ratio floor) — a few scattered tokens against a long title is ambiguous,
89
+ // not a match. Count DISTINCT claim tokens that appear in the title: counting
90
+ // non-distinct tokens lets a repeated-token claim (e.g. "security security
91
+ // security security") inflate the ratio past the floor and falsely match an
92
+ // unrelated title.
93
+ const distinct = new Set(claimTokens);
94
+ const present = [...distinct].filter((t) => titleSet.has(t)).length;
95
+ const ratio = present / titleTokens.length;
96
+ return ratio >= 0.8;
97
+ }
98
+
99
+ async function main() {
17
100
  const argv = process.argv.slice(2);
18
101
  const flags = new Set(argv.filter((a) => a.startsWith("--")));
19
102
  // Reject unknown flags (same contract as the in-process verbs). `--check`
@@ -29,17 +112,22 @@ const { resolveRfc } = require("./citation-resolve.js");
29
112
  process.exitCode = 1;
30
113
  return;
31
114
  }
32
- const positionals = argv.filter((a) => !a.startsWith("--"));
115
+ // --check "<claimed title>" consumes the FOLLOWING token as its value. Exclude
116
+ // that value token by INDEX from the positional pool before selecting id, so
117
+ // the RFC number resolves correctly regardless of flag order
118
+ // (`rfc --check "Some Title" 9404` reads id=9404, not id="Some Title").
119
+ const checkIdx = argv.indexOf("--check");
120
+ const checkValueIdx = (checkIdx !== -1 && argv[checkIdx + 1] && !argv[checkIdx + 1].startsWith("--")) ? checkIdx + 1 : -1;
121
+ const positionals = argv.filter((a, i) => !a.startsWith("--") && i !== checkValueIdx);
33
122
  const id = positionals[0];
34
123
  const pretty = flags.has("--pretty");
35
124
  const json = flags.has("--json") || pretty;
36
125
 
37
- // --check "<claimed title>" : the next non-flag token after the number.
126
+ // The claimed title is exactly the excluded value token (kept in lockstep with
127
+ // checkValueIdx so the two never diverge); a trailing `--check` with no value
128
+ // leaves it null.
38
129
  let claimedTitle = null;
39
- const checkIdx = argv.indexOf("--check");
40
- if (checkIdx !== -1 && argv[checkIdx + 1] && !argv[checkIdx + 1].startsWith("--")) {
41
- claimedTitle = argv[checkIdx + 1];
42
- }
130
+ if (checkValueIdx !== -1) claimedTitle = argv[checkValueIdx];
43
131
 
44
132
  if (!id) {
45
133
  process.stderr.write(
@@ -53,9 +141,7 @@ const { resolveRfc } = require("./citation-resolve.js");
53
141
 
54
142
  let titleMatch = null;
55
143
  if (claimedTitle && r.title) {
56
- const norm = (s) => s.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
57
- const a = norm(claimedTitle), b = norm(r.title);
58
- titleMatch = a.length > 0 && (b.includes(a) || a.includes(b));
144
+ titleMatch = titleMatches(claimedTitle, r.title);
59
145
  }
60
146
  // Derive `ok` from the resolved status + title-check the same way the exit
61
147
  // code is derived below — a non-zero exit (status nonexistent OR an explicit
@@ -83,11 +169,20 @@ const { resolveRfc } = require("./citation-resolve.js");
83
169
  }
84
170
  // A mismatched or nonexistent citation is a non-zero exit for gates.
85
171
  if (fails) process.exitCode = 2;
86
- })().catch((err) => {
87
- // A corrupt/unreadable RFC index (or any unexpected throw inside the async
88
- // body) becomes a rejected promise. Emit the documented {ok:false,error}
89
- // envelope rather than crashing with a raw stack trace, and signal failure
90
- // via exitCode so the event loop drains stderr before exit.
91
- process.stderr.write(JSON.stringify({ ok: false, verb: "rfc", error: String((err && err.message) || err) }) + "\n");
92
- process.exitCode = 1;
93
- });
172
+ }
173
+
174
+ // Only run the CLI when invoked directly (`exceptd rfc ...`). When required by a
175
+ // test the IIFE must not fire — it would read process.argv and write to stdout —
176
+ // so the pure title-match helper can be exercised in-process.
177
+ if (require.main === module) {
178
+ main().catch((err) => {
179
+ // A corrupt/unreadable RFC index (or any unexpected throw inside the async
180
+ // body) becomes a rejected promise. Emit the documented {ok:false,error}
181
+ // envelope rather than crashing with a raw stack trace, and signal failure
182
+ // via exitCode so the event loop drains stderr before exit.
183
+ process.stderr.write(JSON.stringify({ ok: false, verb: "rfc", error: String((err && err.message) || err) }) + "\n");
184
+ process.exitCode = 1;
185
+ });
186
+ }
187
+
188
+ module.exports = { titleMatches, normTitle, main };
@@ -39,7 +39,7 @@
39
39
  "properties": {
40
40
  "source": {
41
41
  "type": "string",
42
- "pattern": "(https://|http://|gh api|gh release|curl |wget |fetch )"
42
+ "pattern": "(https://|http://|gh api|gh release|curl |wget |fetch |GET /|POST /|PUT /|PATCH /|DELETE /|Graph|Okta|Entra ID|Microsoft Graph)"
43
43
  }
44
44
  },
45
45
  "required": ["source"]
package/lib/scoring.js CHANGED
@@ -151,6 +151,18 @@ const RECOGNISED_FACTOR_KEYS = new Set([
151
151
  'patch_required_reboot',
152
152
  ]);
153
153
 
154
+ // Shape-B (catalog post-weight) keys deriveRwepFromFactors is allowed to sum.
155
+ // The post-weight summation operates on the catalog field names — which include
156
+ // `ai_factor`, the +15 AI weight every Shape-B catalog entry stores. `ai_factor`
157
+ // is deliberately ABSENT from RECOGNISED_FACTOR_KEYS (that set carries the
158
+ // Shape-A boolean inputs `ai_assisted_weapon` / `ai_discovered` /
159
+ // `ai_assisted_weaponization`), so the Shape-B allowlist must add it back — a
160
+ // plain `RECOGNISED_FACTOR_KEYS.has(k)` filter would silently drop the AI weight
161
+ // from every derivation. Any key NOT in this set is a typo or unknown field; it
162
+ // is excluded from the sum AND surfaced (see the Shape-B loop) rather than blindly
163
+ // added, so a sub-5 typo can't corrupt the derived score with no diagnostic.
164
+ const RECOGNISED_POST_WEIGHT_KEYS = new Set([...RECOGNISED_FACTOR_KEYS, 'ai_factor']);
165
+
154
166
  function score(cveId, catalog) {
155
167
  const entry = catalog[cveId];
156
168
  if (!entry) throw new Error(`CVE not in catalog: ${cveId}`);
@@ -355,13 +367,29 @@ function scoreCustom(factors, opts) {
355
367
  */
356
368
  function deriveRwepFromFactors(factors) {
357
369
  if (!factors || typeof factors !== 'object') return 0;
358
- const values = Object.values(factors);
359
- if (values.length === 0) return 0;
370
+ const entries = Object.entries(factors);
371
+ if (entries.length === 0) return 0;
372
+ // A boolean factor OR a string active_exploitation ladder value is Shape-A
373
+ // evidence — scoreCustom reads exactly those. active_exploitation's string
374
+ // form legitimately appears in BOTH shapes (Shape A stores it as the literal
375
+ // ladder string; a Shape B post-weight block can ALSO carry it as a
376
+ // human-readable status alongside its post-weight integers), so it is the
377
+ // hasPostWeightInt guard below — NOT excluding active_exploitation from this
378
+ // check — that disambiguates them. Excluding it here under-scored an
379
+ // active-exploitation-ONLY raw bag (e.g. `{ active_exploitation: 'confirmed',
380
+ // blast_radius: 10 }`): hasBooleanOrLadder went false, the block fell through
381
+ // to the Shape-B sum, and the ladder string was skipped (10 vs scoreCustom 30).
360
382
  const aeAllowed = new Set(['none', 'unknown', 'suspected', 'theoretical', 'confirmed']);
361
- const hasBooleanOrLadder = values.some(
362
- (v) => typeof v === 'boolean' || (typeof v === 'string' && aeAllowed.has(v.trim().toLowerCase())),
383
+ const hasBooleanOrLadder = entries.some(
384
+ ([, v]) => (typeof v === 'boolean' || (typeof v === 'string' && aeAllowed.has(v.trim().toLowerCase()))),
385
+ );
386
+ // A boolean-named key carrying a post-weight integer (>=5) is unambiguous
387
+ // Shape-B evidence. When present, the block is Shape B even if it also carries
388
+ // a string active_exploitation — route to the post-weight sum, not scoreCustom.
389
+ const hasPostWeightInt = entries.some(
390
+ ([k, v]) => k !== 'blast_radius' && typeof v === 'number' && Number.isFinite(v) && Math.abs(v) >= 5,
363
391
  );
364
- if (hasBooleanOrLadder) {
392
+ if (hasBooleanOrLadder && !hasPostWeightInt) {
365
393
  return scoreCustom(factors);
366
394
  }
367
395
  // Shape B: catalog post-weight. Sum + clamp.
@@ -380,6 +408,23 @@ function deriveRwepFromFactors(factors) {
380
408
  let sum = 0;
381
409
  for (const [k, v] of Object.entries(factors)) {
382
410
  if (typeof v !== 'number' || !Number.isFinite(v)) continue;
411
+ // Unrecognised key (a typo such as `cisa_kevv` / `reboot_requiredd`, or a
412
+ // field outside the post-weight vocabulary): do NOT add it to the sum, and
413
+ // surface it. scoreCustom/validateFactors already drop+warn on unknown
414
+ // keys; the Shape-B summation previously added ANY numeric value blindly, so
415
+ // the three scoring surfaces disagreed on what an unknown key means (a sub-5
416
+ // typo silently inflated the derived breakdown). Align them here. The
417
+ // warning mirrors the activeExploitationMultiplier precedent above — an
418
+ // observable diagnostic on the standard Node channel, not a silent skip, so
419
+ // the no-match path surfaces an error instead of defaulting (the file's own
420
+ // "out-of-vocab token -> must surface, not silent-default" rule).
421
+ if (!RECOGNISED_POST_WEIGHT_KEYS.has(k)) {
422
+ process.emitWarning(
423
+ `rwep_factors carries unrecognised key '${k}'; excluded from the derived sum`,
424
+ { type: 'RwepFactorUnrecognised', code: 'RWEP_FACTOR_UNRECOGNISED' },
425
+ );
426
+ continue;
427
+ }
383
428
  // reboot_required and patch_required_reboot are aliases for the SAME
384
429
  // post-weight contribution (scoreCustom collapses them). A block carrying
385
430
  // both must count it once; summing both double-counts the reboot weight,
@@ -476,7 +521,14 @@ function compare(cveId, catalog, opts) {
476
521
  if (entry.poc_available) driving.push('public PoC (+20)');
477
522
  if (entry.ai_discovered || entry.ai_assisted_weaponization) driving.push('AI-discovered (+15 weaponization)');
478
523
  if (String(entry.active_exploitation || '').trim().toLowerCase() === 'confirmed') driving.push('confirmed exploitation (+20)');
479
- if ((entry.reboot_required || entry.patch_required_reboot) && !entry.live_patch_available) driving.push('reboot required (+5)');
524
+ // Mirror scoreCustom's rebootFactor EXACTLY: the +5 reboot weight is added
525
+ // whenever a reboot is required, regardless of live_patch_available (a live
526
+ // patch is a temporary workaround; the full-remediation window still extends
527
+ // — see the RWEP_WEIGHTS header note). Gating this driver on
528
+ // !live_patch_available made the enumerated factors sum to less than the
529
+ // delta on any entry that both requires a reboot AND has a live patch
530
+ // available, hiding a driver the score actually counted.
531
+ if (entry.reboot_required || entry.patch_required_reboot) driving.push('reboot required (+5)');
480
532
  explanation += driving.join(', ');
481
533
  explanation += '. Framework patch SLAs calibrated to CVSS are insufficient for this CVE.';
482
534
  } else if (delta < -10) {
@@ -495,7 +547,7 @@ function compare(cveId, catalog, opts) {
495
547
  cve_id: cveId,
496
548
  cvss: cvss,
497
549
  rwep: rwepValid ? rwep : null,
498
- cvss_framework_sla: timeline(cvssEquivalent),
550
+ cvss_framework_sla: cvssAbsent ? { hours: null, label: 'CVSS unavailable — no framework SLA can be derived' } : timeline(cvssEquivalent),
499
551
  rwep_actual_sla: rwepValid ? timeline(rwep) : { hours: null, label: 'RWEP score unavailable' },
500
552
  delta,
501
553
  explanation,
@@ -540,6 +592,16 @@ function detectFactorShape(factors) {
540
592
  let sawWeightedInt = false;
541
593
  for (const [k, v] of Object.entries(factors)) {
542
594
  if (k === 'blast_radius') continue; // always integer in both shapes
595
+ if (k === 'active_exploitation' && typeof v === 'string') {
596
+ // active_exploitation's string-ladder form is valid in BOTH shapes — a
597
+ // Shape B (post-weight) block can carry it as the human-readable status
598
+ // string alongside its post-weight integers, exactly the way Shape A does.
599
+ // So a string active_exploitation is NOT Shape-A evidence; counting it as
600
+ // sawBool produced a spurious 'mixed' verdict (and a validate() error) on
601
+ // an otherwise-clean Shape B block. Its weight, when summed, is resolved
602
+ // via resolveActiveExploitation in the post-weight path, not here.
603
+ continue;
604
+ }
543
605
  if (typeof v === 'boolean' || v === null) {
544
606
  sawBool = true;
545
607
  } else if (typeof v === 'number' && Math.abs(v) >= 5 && boolFields.includes(k)) {
@@ -550,7 +612,7 @@ function detectFactorShape(factors) {
550
612
  // 0/1 on a boolean-named field could be either shape; ambiguous, ignore.
551
613
  continue;
552
614
  } else if (typeof v === 'string' && boolFields.includes(k)) {
553
- // String values (e.g. active_exploitation: 'confirmed') are Shape A.
615
+ // String values on OTHER boolean-named fields are Shape A.
554
616
  sawBool = true;
555
617
  }
556
618
  }
@@ -735,4 +797,5 @@ module.exports = {
735
797
  RWEP_WEIGHTS,
736
798
  ACTIVE_EXPLOITATION_LADDER,
737
799
  RECOGNISED_FACTOR_KEYS,
800
+ RECOGNISED_POST_WEIGHT_KEYS,
738
801
  };
@@ -204,10 +204,19 @@ function extractCveIds(text) {
204
204
  *
205
205
  * Returns [{ title, link, published, body }, ...].
206
206
  */
207
- const { parseFeed: tokenizerParseFeed } = require('./xml-tokenizer');
207
+ const { parseFeedDetailed: tokenizerParseFeedDetailed } = require('./xml-tokenizer');
208
208
 
209
+ // Parser errors are ALWAYS collected and surfaced — the tokenizer's loud-error
210
+ // contract is no longer opt-in. The optional `errors` array is filled when a
211
+ // caller passes one (so a reachable-but-unparsable feed reads 'partial' in the
212
+ // refresh report instead of '0 new CVEs'). A caller that forgets the array
213
+ // still triggers the always-on collection via parseFeedDetailed.
209
214
  function parseRssAtom(xml, errors = null) {
210
- return tokenizerParseFeed(xml, errors);
215
+ const { items, errors: collected } = tokenizerParseFeedDetailed(xml);
216
+ if (Array.isArray(errors)) {
217
+ for (const e of collected) errors.push(e);
218
+ }
219
+ return items;
211
220
  }
212
221
 
213
222
  /**
@@ -291,9 +300,15 @@ function parseGitHubEvents(body, feed) {
291
300
  * GitHub account was removed — the .atom feed needs no API token and the
292
301
  * existing parseRssAtom tokenizer already handles its XML.
293
302
  */
294
- function parseGitLabActivity(body, feed) {
303
+ function parseGitLabActivity(body, feed, errorsOut = null) {
295
304
  const errors = [];
296
305
  const entries = parseRssAtom(body, errors);
306
+ // Thread the Atom parse errors back to the caller's channel (checkFeed)
307
+ // instead of dropping them — a reachable-but-unparsable GitLab activity feed
308
+ // must read 'partial' in the refresh report, same as the RSS/Atom path.
309
+ if (Array.isArray(errorsOut)) {
310
+ for (const e of errors) errorsOut.push(e);
311
+ }
297
312
  const handle = feed.researcher_handle
298
313
  || (feed.url.match(/gitlab\.com\/([^/.]+)\.atom/) || [])[1]
299
314
  || null;
@@ -376,6 +391,11 @@ async function checkFeed(feed, ctx) {
376
391
  const res = await fetchFeed(feed, ctx);
377
392
  if (!res.ok) return { diffs: [], errors: 1, status: 'unreachable', _why: res.error };
378
393
  let items;
394
+ // Parse errors are collected on the XML-parsing feed kinds so a reachable-
395
+ // but-unparsable feed reads 'partial' in the report instead of silently
396
+ // returning 0 new CVEs (the loud-error contract was opt-in and the live
397
+ // path never opted in).
398
+ const parseErrors = [];
379
399
  if (feed.kind === 'csaf-index') {
380
400
  items = parseCsafIndex(res.body);
381
401
  // Flatten cves_from_filename onto cve_ids field uniformly.
@@ -384,10 +404,10 @@ async function checkFeed(feed, ctx) {
384
404
  items = parseGitHubEvents(res.body, feed);
385
405
  items = items.map((it) => ({ ...it, cve_ids: extractCveIds(`${it.title} ${it.body} ${it.link}`) }));
386
406
  } else if (feed.kind === 'gitlab-activity') {
387
- items = parseGitLabActivity(res.body, feed);
407
+ items = parseGitLabActivity(res.body, feed, parseErrors);
388
408
  items = items.map((it) => ({ ...it, cve_ids: extractCveIds(`${it.title} ${it.body} ${it.link}`) }));
389
409
  } else {
390
- items = parseRssAtom(res.body);
410
+ items = parseRssAtom(res.body, parseErrors);
391
411
  items = items.map((it) => ({ ...it, cve_ids: extractCveIds(`${it.title} ${it.body} ${it.link}`) }));
392
412
  }
393
413
  const diffs = [];
@@ -437,7 +457,18 @@ async function checkFeed(feed, ctx) {
437
457
  });
438
458
  }
439
459
  }
440
- return { diffs, observations, errors: 0, status: 'ok' };
460
+ // Fold reachable-but-unparsable into a 'partial' status via a NEW channel.
461
+ // The integer `errors` field stays the unreachable count (0 here — the feed
462
+ // WAS reached) so the aggregate unreachable===FEEDS.length math and the
463
+ // refresh-* assertions that key off it keep working untouched.
464
+ return {
465
+ diffs,
466
+ observations,
467
+ errors: 0,
468
+ status: parseErrors.length ? 'partial' : 'ok',
469
+ parse_errors: parseErrors.length,
470
+ _parse_errors: parseErrors.slice(0, 5),
471
+ };
441
472
  }
442
473
 
443
474
  /**
@@ -452,10 +483,20 @@ const ADVISORIES_SOURCE = {
452
483
  const allDiffs = [];
453
484
  const allObservations = [];
454
485
  let unreachable = 0;
486
+ let parseErrorFeeds = 0; // feeds reachable but with >=1 parse error
487
+ const parseErrorSamples = []; // bounded sample of {message, position}
455
488
  for (const r of results) {
456
489
  allDiffs.push(...r.diffs);
457
490
  if (Array.isArray(r.observations)) allObservations.push(...r.observations);
458
491
  if (r.status === 'unreachable') unreachable++;
492
+ if (typeof r.parse_errors === 'number' && r.parse_errors > 0) {
493
+ parseErrorFeeds++;
494
+ if (Array.isArray(r._parse_errors)) {
495
+ for (const e of r._parse_errors) {
496
+ if (parseErrorSamples.length < 5) parseErrorSamples.push(e);
497
+ }
498
+ }
499
+ }
459
500
  }
460
501
  // Deduplicate by CVE-ID across feeds — multiple advisories for the
461
502
  // same CVE collapse to one entry with sources[] array of contributing
@@ -503,15 +544,23 @@ const ADVISORIES_SOURCE = {
503
544
  }
504
545
  }
505
546
  const observations = Array.from(obsByCve.values());
547
+ // Status ladder folds reachable-but-unparsable into 'partial'. The integer
548
+ // `errors` field stays the unreachable count (downstream refresh math keys
549
+ // off it); parse errors are surfaced via the separate parse_errors channel.
506
550
  const status =
507
- unreachable === 0 ? 'ok' :
508
- unreachable === FEEDS.length ? 'unreachable' : 'partial';
551
+ unreachable === FEEDS.length ? 'unreachable' :
552
+ (unreachable > 0 || parseErrorFeeds > 0) ? 'partial' :
553
+ 'ok';
554
+ const summary = `${FEEDS.length - unreachable}/${FEEDS.length} feeds reachable; ${diffs.length} new CVE references found, ${observations.length} total CVE observations across primary advisory sources`
555
+ + (parseErrorFeeds > 0 ? `; ${parseErrorFeeds} feed${parseErrorFeeds === 1 ? '' : 's'} returned parse errors` : '');
509
556
  return {
510
557
  status,
511
558
  diffs,
512
559
  observations,
513
560
  errors: unreachable,
514
- summary: `${FEEDS.length - unreachable}/${FEEDS.length} feeds reachable; ${diffs.length} new CVE references found, ${observations.length} total CVE observations across primary advisory sources`,
561
+ parse_errors: parseErrorFeeds,
562
+ _parse_errors: parseErrorSamples,
563
+ summary,
515
564
  };
516
565
  },
517
566
  // Report-only: no applyDiff. Operators route promising CVE IDs through
package/lib/ttp-mapper.js CHANGED
@@ -36,6 +36,17 @@ function gapsFor(attackPattern, gapCatalog, atlasCatalog) {
36
36
  }
37
37
 
38
38
  function coverage(frameworkId, ttpId, gapCatalog, atlasCatalog) {
39
+ // Input guard before any deref — an empty / non-string frameworkId
40
+ // yielded frameworkPrefix='' which matched EVERY control via
41
+ // includes(''), and null/undefined threw on .split(). Match the
42
+ // { found:false } contract already used for an unknown TTP. Surface
43
+ // partially_covered_by / not_covered_by as an explicit null (not absent)
44
+ // so the no-match outcome is observable rather than a silent universal
45
+ // match.
46
+ if (typeof frameworkId !== 'string' || frameworkId.trim() === '') {
47
+ return { ttp_id: ttpId, found: false, error: 'frameworkId required', partially_covered_by: null, not_covered_by: null };
48
+ }
49
+
39
50
  const ttp = atlasCatalog[ttpId];
40
51
  if (!ttp) return { ttp_id: ttpId, found: false };
41
52
 
@@ -45,10 +56,27 @@ function coverage(frameworkId, ttpId, gapCatalog, atlasCatalog) {
45
56
  const gapDetail = ttp.framework_gap_detail || '';
46
57
  const hasFrameworkGap = ttp.framework_gap === true;
47
58
 
48
- // Check if the requested framework has any coverage in the partially-helpful controls
59
+ // Check if the requested framework has any coverage in the partially-helpful
60
+ // controls. Match on the first hyphen-delimited segment of the control id
61
+ // (token-boundary), NOT bare substring containment: a bare includes() let
62
+ // 'IS' match 'NIST' and '' match everything. A control id matches when it
63
+ // begins with the prefix and the next char is a segment boundary (-, .) or
64
+ // end-of-string, so 'soc2' still matches 'soc2-z' but 'is' never matches
65
+ // 'nist-...'.
49
66
  const frameworkPrefix = frameworkId.split('-')[0].toLowerCase();
50
- const partial = partialControls.find(c => c.toLowerCase().includes(frameworkPrefix));
51
- const noHelp = noHelpControls.find(c => c.toLowerCase().includes(frameworkPrefix));
67
+ if (frameworkPrefix.length === 0) {
68
+ // frameworkId is a hyphen-led string (e.g. "-" or "-X") whose first
69
+ // segment is empty — same universal-match hazard, same fail-closed result.
70
+ return { ttp_id: ttpId, found: false, error: 'frameworkId required', partially_covered_by: null, not_covered_by: null };
71
+ }
72
+ const segMatch = (c) => {
73
+ const cl = String(c).toLowerCase();
74
+ if (!cl.startsWith(frameworkPrefix)) return false;
75
+ const next = cl.charAt(frameworkPrefix.length);
76
+ return next === '' || next === '-' || next === '.';
77
+ };
78
+ const partial = partialControls.find(segMatch);
79
+ const noHelp = noHelpControls.find(segMatch);
52
80
 
53
81
  return {
54
82
  ttp_id: ttpId,
@@ -80,4 +80,16 @@ function readManifest() {
80
80
  localManifest: readManifest(),
81
81
  });
82
82
  process.stdout.write(JSON.stringify(report) + "\n");
83
- })();
83
+ })().catch((err) => {
84
+ // Any unexpected throw still yields one parseable JSON line on stdout and a
85
+ // clean exit, consistent with this probe's offline-degradation contract
86
+ // (missing freshness data is not an error for downstream callers, which parse
87
+ // res.stdout for the envelope). String(...) coerces the error to a primitive,
88
+ // so this JSON.stringify itself cannot throw. Exit 0 is the default — no prior
89
+ // non-zero exitCode is set on this path.
90
+ process.stdout.write(JSON.stringify({
91
+ ok: false,
92
+ error: String((err && err.message) || err),
93
+ source: "upstream-check",
94
+ }) + "\n");
95
+ });