@blamejs/exceptd-skills 0.18.8 → 0.18.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/bin/exceptd.js +197 -118
  3. package/data/_indexes/_meta.json +3 -3
  4. package/data/_indexes/frequency.json +2 -2
  5. package/data/d3fend-catalog.json +6 -6
  6. package/data/playbooks/identity-sso-compromise.json +2 -2
  7. package/data/playbooks/sbom.json +1 -1
  8. package/lib/citation-resolve.js +11 -0
  9. package/lib/collectors/containers.js +13 -0
  10. package/lib/cross-ref-api.js +29 -7
  11. package/lib/cve-regression-watcher.js +47 -15
  12. package/lib/framework-gap.js +27 -5
  13. package/lib/gap-detectors.js +8 -3
  14. package/lib/lint-skills.js +3 -2
  15. package/lib/playbook-runner.js +60 -5
  16. package/lib/refresh-external.js +58 -7
  17. package/lib/refresh-network.js +24 -8
  18. package/lib/rfc-cli.js +108 -18
  19. package/lib/schemas/playbook.schema.json +1 -1
  20. package/lib/scoring.js +31 -1
  21. package/lib/source-advisories.js +58 -9
  22. package/lib/ttp-mapper.js +31 -3
  23. package/lib/upstream-check-cli.js +13 -1
  24. package/lib/validate-catalog-meta.js +51 -7
  25. package/lib/validate-cve-catalog.js +10 -0
  26. package/lib/validate-playbooks.js +19 -1
  27. package/lib/xml-tokenizer.js +187 -25
  28. package/manifest.json +53 -53
  29. package/orchestrator/dispatcher.js +45 -9
  30. package/orchestrator/index.js +9 -7
  31. package/orchestrator/pipeline.js +62 -14
  32. package/orchestrator/scanner.js +40 -9
  33. package/package.json +1 -1
  34. package/sbom.cdx.json +105 -90
  35. package/scripts/build-indexes.js +21 -3
  36. package/scripts/builders/section-offsets.js +17 -8
  37. package/scripts/check-catalog-gap-budget.js +3 -3
  38. package/scripts/check-codebase-patterns.js +124 -11
  39. package/scripts/check-sbom-currency.js +69 -3
  40. package/scripts/check-test-count.js +28 -16
  41. package/scripts/check-test-subjects.js +127 -0
  42. package/scripts/check-version-tags.js +24 -5
  43. package/scripts/predeploy.js +13 -0
  44. package/scripts/refresh-upstream-catalogs.js +150 -42
  45. package/scripts/release.js +28 -11
  46. package/scripts/validate-vendor-online.js +12 -9
package/lib/rfc-cli.js CHANGED
@@ -13,7 +13,85 @@
13
13
 
14
14
  const { resolveRfc } = require("./citation-resolve.js");
15
15
 
16
- (async () => {
16
+ // Stopwords that don't disambiguate one RFC title from another. A claimed title
17
+ // run preceded by one of these in the index title is still a clean match; a run
18
+ // preceded by a CONTENT word (e.g. "datagram" before "transport layer security")
19
+ // is the tail of a more-specific title and must NOT be accepted as a match.
20
+ const TITLE_STOPWORDS = new Set(["the", "a", "an", "of", "for", "to", "in", "on", "and", "or"]);
21
+
22
+ function normTitle(s) {
23
+ return String(s).toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
24
+ }
25
+
26
+ /**
27
+ * Decide whether a claimed RFC title matches the authoritative index title.
28
+ *
29
+ * Replaces the old lenient bidirectional substring test (`a.includes(b) ||
30
+ * b.includes(a)`), which let "TLS" match the DTLS title (substring of "dtls")
31
+ * and let "Transport Layer Security" match the DTLS title (tail-of-phrase).
32
+ * The comparison is now whole-word and phrase-aware:
33
+ *
34
+ * 1. Every claimed token must appear as a WHOLE word in the index title
35
+ * (so "tls" never matches inside "dtls").
36
+ * 2. The claimed token sequence must appear as a CONTIGUOUS run in the index
37
+ * title, OR the claim must cover enough of the index title (containment
38
+ * ratio floor) to be unambiguous.
39
+ * 3. A contiguous run that is immediately preceded by a distinguishing
40
+ * CONTENT word in the index title is rejected — it is the tail of a
41
+ * more-specific title (the "datagram transport layer security" trap).
42
+ *
43
+ * Returns true / false. Only called when both a claim and an index title exist.
44
+ */
45
+ function titleMatches(claimed, indexTitle) {
46
+ const claimTokens = normTitle(claimed).split(" ").filter(Boolean);
47
+ const titleTokens = normTitle(indexTitle).split(" ").filter(Boolean);
48
+ if (claimTokens.length === 0 || titleTokens.length === 0) return false;
49
+
50
+ // (1) Whole-word containment: every claimed token must be a standalone token
51
+ // in the index title. Kills the tls-inside-dtls substring false positive.
52
+ const titleSet = new Set(titleTokens);
53
+ for (const t of claimTokens) {
54
+ if (!titleSet.has(t)) return false;
55
+ }
56
+
57
+ // Find every contiguous run of the claim inside the index title.
58
+ const runStarts = [];
59
+ for (let i = 0; i + claimTokens.length <= titleTokens.length; i++) {
60
+ let hit = true;
61
+ for (let j = 0; j < claimTokens.length; j++) {
62
+ if (titleTokens[i + j] !== claimTokens[j]) { hit = false; break; }
63
+ }
64
+ if (hit) runStarts.push(i);
65
+ }
66
+
67
+ if (runStarts.length > 0) {
68
+ // A single-token claim that is a whole word in the title is unambiguous on
69
+ // its own — the whole-word check above already excluded the substring trap
70
+ // (e.g. "tls" is NOT a token inside "dtls"), so "TLS" correctly matches the
71
+ // 8446 title (standalone "tls" token) but not the 9147 DTLS title.
72
+ if (claimTokens.length === 1) return true;
73
+ // (3) For a MULTI-token run, accept only if at least one occurrence is NOT
74
+ // preceded by a distinguishing content word — i.e. it begins the title
75
+ // or is preceded only by a stopword. A run preceded solely by a content
76
+ // qualifier (e.g. "datagram" before "transport layer security") is the
77
+ // tail of a more-specific title and must not be accepted as a match.
78
+ for (const start of runStarts) {
79
+ if (start === 0) return true;
80
+ const prev = titleTokens[start - 1];
81
+ if (TITLE_STOPWORDS.has(prev)) return true;
82
+ }
83
+ return false;
84
+ }
85
+
86
+ // No contiguous run, but all tokens present out of order. Accept only when the
87
+ // claim covers a strong majority of the index title's tokens (containment
88
+ // ratio floor) — a few scattered tokens against a long title is ambiguous,
89
+ // not a match.
90
+ const ratio = claimTokens.length / titleTokens.length;
91
+ return ratio >= 0.8;
92
+ }
93
+
94
+ async function main() {
17
95
  const argv = process.argv.slice(2);
18
96
  const flags = new Set(argv.filter((a) => a.startsWith("--")));
19
97
  // Reject unknown flags (same contract as the in-process verbs). `--check`
@@ -29,17 +107,22 @@ const { resolveRfc } = require("./citation-resolve.js");
29
107
  process.exitCode = 1;
30
108
  return;
31
109
  }
32
- const positionals = argv.filter((a) => !a.startsWith("--"));
110
+ // --check "<claimed title>" consumes the FOLLOWING token as its value. Exclude
111
+ // that value token by INDEX from the positional pool before selecting id, so
112
+ // the RFC number resolves correctly regardless of flag order
113
+ // (`rfc --check "Some Title" 9404` reads id=9404, not id="Some Title").
114
+ const checkIdx = argv.indexOf("--check");
115
+ const checkValueIdx = (checkIdx !== -1 && argv[checkIdx + 1] && !argv[checkIdx + 1].startsWith("--")) ? checkIdx + 1 : -1;
116
+ const positionals = argv.filter((a, i) => !a.startsWith("--") && i !== checkValueIdx);
33
117
  const id = positionals[0];
34
118
  const pretty = flags.has("--pretty");
35
119
  const json = flags.has("--json") || pretty;
36
120
 
37
- // --check "<claimed title>" : the next non-flag token after the number.
121
+ // The claimed title is exactly the excluded value token (kept in lockstep with
122
+ // checkValueIdx so the two never diverge); a trailing `--check` with no value
123
+ // leaves it null.
38
124
  let claimedTitle = null;
39
- const checkIdx = argv.indexOf("--check");
40
- if (checkIdx !== -1 && argv[checkIdx + 1] && !argv[checkIdx + 1].startsWith("--")) {
41
- claimedTitle = argv[checkIdx + 1];
42
- }
125
+ if (checkValueIdx !== -1) claimedTitle = argv[checkValueIdx];
43
126
 
44
127
  if (!id) {
45
128
  process.stderr.write(
@@ -53,9 +136,7 @@ const { resolveRfc } = require("./citation-resolve.js");
53
136
 
54
137
  let titleMatch = null;
55
138
  if (claimedTitle && r.title) {
56
- const norm = (s) => s.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
57
- const a = norm(claimedTitle), b = norm(r.title);
58
- titleMatch = a.length > 0 && (b.includes(a) || a.includes(b));
139
+ titleMatch = titleMatches(claimedTitle, r.title);
59
140
  }
60
141
  // Derive `ok` from the resolved status + title-check the same way the exit
61
142
  // code is derived below — a non-zero exit (status nonexistent OR an explicit
@@ -83,11 +164,20 @@ const { resolveRfc } = require("./citation-resolve.js");
83
164
  }
84
165
  // A mismatched or nonexistent citation is a non-zero exit for gates.
85
166
  if (fails) process.exitCode = 2;
86
- })().catch((err) => {
87
- // A corrupt/unreadable RFC index (or any unexpected throw inside the async
88
- // body) becomes a rejected promise. Emit the documented {ok:false,error}
89
- // envelope rather than crashing with a raw stack trace, and signal failure
90
- // via exitCode so the event loop drains stderr before exit.
91
- process.stderr.write(JSON.stringify({ ok: false, verb: "rfc", error: String((err && err.message) || err) }) + "\n");
92
- process.exitCode = 1;
93
- });
167
+ }
168
+
169
+ // Only run the CLI when invoked directly (`exceptd rfc ...`). When required by a
170
+ // test the IIFE must not fire — it would read process.argv and write to stdout —
171
+ // so the pure title-match helper can be exercised in-process.
172
+ if (require.main === module) {
173
+ main().catch((err) => {
174
+ // A corrupt/unreadable RFC index (or any unexpected throw inside the async
175
+ // body) becomes a rejected promise. Emit the documented {ok:false,error}
176
+ // envelope rather than crashing with a raw stack trace, and signal failure
177
+ // via exitCode so the event loop drains stderr before exit.
178
+ process.stderr.write(JSON.stringify({ ok: false, verb: "rfc", error: String((err && err.message) || err) }) + "\n");
179
+ process.exitCode = 1;
180
+ });
181
+ }
182
+
183
+ module.exports = { titleMatches, normTitle, main };
@@ -39,7 +39,7 @@
39
39
  "properties": {
40
40
  "source": {
41
41
  "type": "string",
42
- "pattern": "(https://|http://|gh api|gh release|curl |wget |fetch )"
42
+ "pattern": "(https://|http://|gh api|gh release|curl |wget |fetch |GET /|POST /|PUT /|PATCH /|DELETE /|Graph|Okta|Entra ID|Microsoft Graph)"
43
43
  }
44
44
  },
45
45
  "required": ["source"]
package/lib/scoring.js CHANGED
@@ -151,6 +151,18 @@ const RECOGNISED_FACTOR_KEYS = new Set([
151
151
  'patch_required_reboot',
152
152
  ]);
153
153
 
154
+ // Shape-B (catalog post-weight) keys deriveRwepFromFactors is allowed to sum.
155
+ // The post-weight summation operates on the catalog field names — which include
156
+ // `ai_factor`, the +15 AI weight every Shape-B catalog entry stores. `ai_factor`
157
+ // is deliberately ABSENT from RECOGNISED_FACTOR_KEYS (that set carries the
158
+ // Shape-A boolean inputs `ai_assisted_weapon` / `ai_discovered` /
159
+ // `ai_assisted_weaponization`), so the Shape-B allowlist must add it back — a
160
+ // plain `RECOGNISED_FACTOR_KEYS.has(k)` filter would silently drop the AI weight
161
+ // from every derivation. Any key NOT in this set is a typo or unknown field; it
162
+ // is excluded from the sum AND surfaced (see the Shape-B loop) rather than blindly
163
+ // added, so a sub-5 typo can't corrupt the derived score with no diagnostic.
164
+ const RECOGNISED_POST_WEIGHT_KEYS = new Set([...RECOGNISED_FACTOR_KEYS, 'ai_factor']);
165
+
154
166
  function score(cveId, catalog) {
155
167
  const entry = catalog[cveId];
156
168
  if (!entry) throw new Error(`CVE not in catalog: ${cveId}`);
@@ -380,6 +392,23 @@ function deriveRwepFromFactors(factors) {
380
392
  let sum = 0;
381
393
  for (const [k, v] of Object.entries(factors)) {
382
394
  if (typeof v !== 'number' || !Number.isFinite(v)) continue;
395
+ // Unrecognised key (a typo such as `cisa_kevv` / `reboot_requiredd`, or a
396
+ // field outside the post-weight vocabulary): do NOT add it to the sum, and
397
+ // surface it. scoreCustom/validateFactors already drop+warn on unknown
398
+ // keys; the Shape-B summation previously added ANY numeric value blindly, so
399
+ // the three scoring surfaces disagreed on what an unknown key means (a sub-5
400
+ // typo silently inflated the derived breakdown). Align them here. The
401
+ // warning mirrors the activeExploitationMultiplier precedent above — an
402
+ // observable diagnostic on the standard Node channel, not a silent skip, so
403
+ // the no-match path surfaces an error instead of defaulting (the file's own
404
+ // "out-of-vocab token -> must surface, not silent-default" rule).
405
+ if (!RECOGNISED_POST_WEIGHT_KEYS.has(k)) {
406
+ process.emitWarning(
407
+ `rwep_factors carries unrecognised key '${k}'; excluded from the derived sum`,
408
+ { type: 'RwepFactorUnrecognised', code: 'RWEP_FACTOR_UNRECOGNISED' },
409
+ );
410
+ continue;
411
+ }
383
412
  // reboot_required and patch_required_reboot are aliases for the SAME
384
413
  // post-weight contribution (scoreCustom collapses them). A block carrying
385
414
  // both must count it once; summing both double-counts the reboot weight,
@@ -495,7 +524,7 @@ function compare(cveId, catalog, opts) {
495
524
  cve_id: cveId,
496
525
  cvss: cvss,
497
526
  rwep: rwepValid ? rwep : null,
498
- cvss_framework_sla: timeline(cvssEquivalent),
527
+ cvss_framework_sla: cvssAbsent ? { hours: null, label: 'CVSS unavailable — no framework SLA can be derived' } : timeline(cvssEquivalent),
499
528
  rwep_actual_sla: rwepValid ? timeline(rwep) : { hours: null, label: 'RWEP score unavailable' },
500
529
  delta,
501
530
  explanation,
@@ -735,4 +764,5 @@ module.exports = {
735
764
  RWEP_WEIGHTS,
736
765
  ACTIVE_EXPLOITATION_LADDER,
737
766
  RECOGNISED_FACTOR_KEYS,
767
+ RECOGNISED_POST_WEIGHT_KEYS,
738
768
  };
@@ -204,10 +204,19 @@ function extractCveIds(text) {
204
204
  *
205
205
  * Returns [{ title, link, published, body }, ...].
206
206
  */
207
- const { parseFeed: tokenizerParseFeed } = require('./xml-tokenizer');
207
+ const { parseFeedDetailed: tokenizerParseFeedDetailed } = require('./xml-tokenizer');
208
208
 
209
+ // Parser errors are ALWAYS collected and surfaced — the tokenizer's loud-error
210
+ // contract is no longer opt-in. The optional `errors` array is filled when a
211
+ // caller passes one (so a reachable-but-unparsable feed reads 'partial' in the
212
+ // refresh report instead of '0 new CVEs'). A caller that forgets the array
213
+ // still triggers the always-on collection via parseFeedDetailed.
209
214
  function parseRssAtom(xml, errors = null) {
210
- return tokenizerParseFeed(xml, errors);
215
+ const { items, errors: collected } = tokenizerParseFeedDetailed(xml);
216
+ if (Array.isArray(errors)) {
217
+ for (const e of collected) errors.push(e);
218
+ }
219
+ return items;
211
220
  }
212
221
 
213
222
  /**
@@ -291,9 +300,15 @@ function parseGitHubEvents(body, feed) {
291
300
  * GitHub account was removed — the .atom feed needs no API token and the
292
301
  * existing parseRssAtom tokenizer already handles its XML.
293
302
  */
294
- function parseGitLabActivity(body, feed) {
303
+ function parseGitLabActivity(body, feed, errorsOut = null) {
295
304
  const errors = [];
296
305
  const entries = parseRssAtom(body, errors);
306
+ // Thread the Atom parse errors back to the caller's channel (checkFeed)
307
+ // instead of dropping them — a reachable-but-unparsable GitLab activity feed
308
+ // must read 'partial' in the refresh report, same as the RSS/Atom path.
309
+ if (Array.isArray(errorsOut)) {
310
+ for (const e of errors) errorsOut.push(e);
311
+ }
297
312
  const handle = feed.researcher_handle
298
313
  || (feed.url.match(/gitlab\.com\/([^/.]+)\.atom/) || [])[1]
299
314
  || null;
@@ -376,6 +391,11 @@ async function checkFeed(feed, ctx) {
376
391
  const res = await fetchFeed(feed, ctx);
377
392
  if (!res.ok) return { diffs: [], errors: 1, status: 'unreachable', _why: res.error };
378
393
  let items;
394
+ // Parse errors are collected on the XML-parsing feed kinds so a reachable-
395
+ // but-unparsable feed reads 'partial' in the report instead of silently
396
+ // returning 0 new CVEs (the loud-error contract was opt-in and the live
397
+ // path never opted in).
398
+ const parseErrors = [];
379
399
  if (feed.kind === 'csaf-index') {
380
400
  items = parseCsafIndex(res.body);
381
401
  // Flatten cves_from_filename onto cve_ids field uniformly.
@@ -384,10 +404,10 @@ async function checkFeed(feed, ctx) {
384
404
  items = parseGitHubEvents(res.body, feed);
385
405
  items = items.map((it) => ({ ...it, cve_ids: extractCveIds(`${it.title} ${it.body} ${it.link}`) }));
386
406
  } else if (feed.kind === 'gitlab-activity') {
387
- items = parseGitLabActivity(res.body, feed);
407
+ items = parseGitLabActivity(res.body, feed, parseErrors);
388
408
  items = items.map((it) => ({ ...it, cve_ids: extractCveIds(`${it.title} ${it.body} ${it.link}`) }));
389
409
  } else {
390
- items = parseRssAtom(res.body);
410
+ items = parseRssAtom(res.body, parseErrors);
391
411
  items = items.map((it) => ({ ...it, cve_ids: extractCveIds(`${it.title} ${it.body} ${it.link}`) }));
392
412
  }
393
413
  const diffs = [];
@@ -437,7 +457,18 @@ async function checkFeed(feed, ctx) {
437
457
  });
438
458
  }
439
459
  }
440
- return { diffs, observations, errors: 0, status: 'ok' };
460
+ // Fold reachable-but-unparsable into a 'partial' status via a NEW channel.
461
+ // The integer `errors` field stays the unreachable count (0 here — the feed
462
+ // WAS reached) so the aggregate unreachable===FEEDS.length math and the
463
+ // refresh-* assertions that key off it keep working untouched.
464
+ return {
465
+ diffs,
466
+ observations,
467
+ errors: 0,
468
+ status: parseErrors.length ? 'partial' : 'ok',
469
+ parse_errors: parseErrors.length,
470
+ _parse_errors: parseErrors.slice(0, 5),
471
+ };
441
472
  }
442
473
 
443
474
  /**
@@ -452,10 +483,20 @@ const ADVISORIES_SOURCE = {
452
483
  const allDiffs = [];
453
484
  const allObservations = [];
454
485
  let unreachable = 0;
486
+ let parseErrorFeeds = 0; // feeds reachable but with >=1 parse error
487
+ const parseErrorSamples = []; // bounded sample of {message, position}
455
488
  for (const r of results) {
456
489
  allDiffs.push(...r.diffs);
457
490
  if (Array.isArray(r.observations)) allObservations.push(...r.observations);
458
491
  if (r.status === 'unreachable') unreachable++;
492
+ if (typeof r.parse_errors === 'number' && r.parse_errors > 0) {
493
+ parseErrorFeeds++;
494
+ if (Array.isArray(r._parse_errors)) {
495
+ for (const e of r._parse_errors) {
496
+ if (parseErrorSamples.length < 5) parseErrorSamples.push(e);
497
+ }
498
+ }
499
+ }
459
500
  }
460
501
  // Deduplicate by CVE-ID across feeds — multiple advisories for the
461
502
  // same CVE collapse to one entry with sources[] array of contributing
@@ -503,15 +544,23 @@ const ADVISORIES_SOURCE = {
503
544
  }
504
545
  }
505
546
  const observations = Array.from(obsByCve.values());
547
+ // Status ladder folds reachable-but-unparsable into 'partial'. The integer
548
+ // `errors` field stays the unreachable count (downstream refresh math keys
549
+ // off it); parse errors are surfaced via the separate parse_errors channel.
506
550
  const status =
507
- unreachable === 0 ? 'ok' :
508
- unreachable === FEEDS.length ? 'unreachable' : 'partial';
551
+ unreachable === FEEDS.length ? 'unreachable' :
552
+ (unreachable > 0 || parseErrorFeeds > 0) ? 'partial' :
553
+ 'ok';
554
+ const summary = `${FEEDS.length - unreachable}/${FEEDS.length} feeds reachable; ${diffs.length} new CVE references found, ${observations.length} total CVE observations across primary advisory sources`
555
+ + (parseErrorFeeds > 0 ? `; ${parseErrorFeeds} feed${parseErrorFeeds === 1 ? '' : 's'} returned parse errors` : '');
509
556
  return {
510
557
  status,
511
558
  diffs,
512
559
  observations,
513
560
  errors: unreachable,
514
- summary: `${FEEDS.length - unreachable}/${FEEDS.length} feeds reachable; ${diffs.length} new CVE references found, ${observations.length} total CVE observations across primary advisory sources`,
561
+ parse_errors: parseErrorFeeds,
562
+ _parse_errors: parseErrorSamples,
563
+ summary,
515
564
  };
516
565
  },
517
566
  // Report-only: no applyDiff. Operators route promising CVE IDs through
package/lib/ttp-mapper.js CHANGED
@@ -36,6 +36,17 @@ function gapsFor(attackPattern, gapCatalog, atlasCatalog) {
36
36
  }
37
37
 
38
38
  function coverage(frameworkId, ttpId, gapCatalog, atlasCatalog) {
39
+ // Input guard before any deref — an empty / non-string frameworkId
40
+ // yielded frameworkPrefix='' which matched EVERY control via
41
+ // includes(''), and null/undefined threw on .split(). Match the
42
+ // { found:false } contract already used for an unknown TTP. Surface
43
+ // partially_covered_by / not_covered_by as an explicit null (not absent)
44
+ // so the no-match outcome is observable rather than a silent universal
45
+ // match.
46
+ if (typeof frameworkId !== 'string' || frameworkId.trim() === '') {
47
+ return { ttp_id: ttpId, found: false, error: 'frameworkId required', partially_covered_by: null, not_covered_by: null };
48
+ }
49
+
39
50
  const ttp = atlasCatalog[ttpId];
40
51
  if (!ttp) return { ttp_id: ttpId, found: false };
41
52
 
@@ -45,10 +56,27 @@ function coverage(frameworkId, ttpId, gapCatalog, atlasCatalog) {
45
56
  const gapDetail = ttp.framework_gap_detail || '';
46
57
  const hasFrameworkGap = ttp.framework_gap === true;
47
58
 
48
- // Check if the requested framework has any coverage in the partially-helpful controls
59
+ // Check if the requested framework has any coverage in the partially-helpful
60
+ // controls. Match on the first hyphen-delimited segment of the control id
61
+ // (token-boundary), NOT bare substring containment: a bare includes() let
62
+ // 'IS' match 'NIST' and '' match everything. A control id matches when it
63
+ // begins with the prefix and the next char is a segment boundary (-, .) or
64
+ // end-of-string, so 'soc2' still matches 'soc2-z' but 'is' never matches
65
+ // 'nist-...'.
49
66
  const frameworkPrefix = frameworkId.split('-')[0].toLowerCase();
50
- const partial = partialControls.find(c => c.toLowerCase().includes(frameworkPrefix));
51
- const noHelp = noHelpControls.find(c => c.toLowerCase().includes(frameworkPrefix));
67
+ if (frameworkPrefix.length === 0) {
68
+ // frameworkId is a hyphen-led string (e.g. "-" or "-X") whose first
69
+ // segment is empty — same universal-match hazard, same fail-closed result.
70
+ return { ttp_id: ttpId, found: false, error: 'frameworkId required', partially_covered_by: null, not_covered_by: null };
71
+ }
72
+ const segMatch = (c) => {
73
+ const cl = String(c).toLowerCase();
74
+ if (!cl.startsWith(frameworkPrefix)) return false;
75
+ const next = cl.charAt(frameworkPrefix.length);
76
+ return next === '' || next === '-' || next === '.';
77
+ };
78
+ const partial = partialControls.find(segMatch);
79
+ const noHelp = noHelpControls.find(segMatch);
52
80
 
53
81
  return {
54
82
  ttp_id: ttpId,
@@ -80,4 +80,16 @@ function readManifest() {
80
80
  localManifest: readManifest(),
81
81
  });
82
82
  process.stdout.write(JSON.stringify(report) + "\n");
83
- })();
83
+ })().catch((err) => {
84
+ // Any unexpected throw still yields one parseable JSON line on stdout and a
85
+ // clean exit, consistent with this probe's offline-degradation contract
86
+ // (missing freshness data is not an error for downstream callers, which parse
87
+ // res.stdout for the envelope). String(...) coerces the error to a primitive,
88
+ // so this JSON.stringify itself cannot throw. Exit 0 is the default — no prior
89
+ // non-zero exitCode is set on this path.
90
+ process.stdout.write(JSON.stringify({
91
+ ok: false,
92
+ error: String((err && err.message) || err),
93
+ source: "upstream-check",
94
+ }) + "\n");
95
+ });
@@ -85,6 +85,29 @@ function containsPlaceholder(s) {
85
85
  return PLACEHOLDER_TOKENS.some((re) => re.test(s));
86
86
  }
87
87
 
88
+ // Round-trip ISO calendar-date check. Returns a Date for a real YYYY-MM-DD
89
+ // calendar date, or null for anything malformed. Unlike a shape-only regex,
90
+ // this rejects impossible dates (2026-13-99, 2026-04-31, 2026-02-29 in a
91
+ // non-leap year): `new Date('2026-02-30T00:00:00Z')` does NOT throw — it rolls
92
+ // over to March 2 with a valid getTime() — so the parsed Y-M-D must round-trip
93
+ // back to the input components. Deliberately carries NO year-floor business
94
+ // rule (a valid-but-old 1900-01-01 stays a valid date so the staleness branch,
95
+ // not the validity branch, reports it).
96
+ function parseIsoDateStrict(v) {
97
+ if (typeof v !== 'string' || !/^\d{4}-\d{2}-\d{2}$/.test(v)) return null;
98
+ const d = new Date(v + 'T00:00:00Z');
99
+ if (Number.isNaN(d.getTime())) return null;
100
+ const [y, m, day] = v.split('-').map(Number);
101
+ if (
102
+ d.getUTCFullYear() !== y ||
103
+ d.getUTCMonth() + 1 !== m ||
104
+ d.getUTCDate() !== day
105
+ ) {
106
+ return null;
107
+ }
108
+ return d;
109
+ }
110
+
88
111
  function validateMeta(catalogPath, opts) {
89
112
  const errors = [];
90
113
  const warnings = [];
@@ -92,7 +115,16 @@ function validateMeta(catalogPath, opts) {
92
115
  const meta = data._meta;
93
116
 
94
117
  if (!meta || typeof meta !== 'object') {
95
- return ['missing _meta block'];
118
+ // Honor both return contracts. The rest of the body dereferences
119
+ // `meta.tlp` / `meta.source_confidence` / `meta.freshness_policy`, so it
120
+ // must NOT run when `meta` is absent or non-object. Return early in the
121
+ // SAME shape the caller asked for: includeWarnings callers (main()) get
122
+ // `{errors, warnings}` so `result.errors` is a real array and the loop
123
+ // reports a clean FAIL + continues; no-opts callers still get a non-empty
124
+ // `string[]`. This still FAILS — it only removes the uncaught TypeError.
125
+ errors.push('missing _meta block');
126
+ if (opts && opts.includeWarnings) return { errors, warnings };
127
+ return errors;
96
128
  }
97
129
 
98
130
  /* tlp */
@@ -176,14 +208,26 @@ function validateMeta(catalogPath, opts) {
176
208
  * the warning posture.
177
209
  */
178
210
  if (
179
- typeof meta.last_updated === 'string' &&
211
+ meta.last_updated !== undefined &&
180
212
  typeof fp.stale_after_days === 'number' &&
181
213
  fp.stale_after_days > 0
182
214
  ) {
183
- const lu = new Date(meta.last_updated + (
184
- /^\d{4}-\d{2}-\d{2}$/.test(meta.last_updated) ? 'T00:00:00Z' : ''
185
- ));
186
- if (!Number.isNaN(lu.getTime())) {
215
+ const lu = parseIsoDateStrict(meta.last_updated);
216
+ if (lu === null) {
217
+ // Fail-closed on a malformed date instead of silently skipping the
218
+ // freshness gate. A NaN/impossible/wrong-shape last_updated is
219
+ // "invalid input" (error under --strict, warning by default), NOT
220
+ // "no opinion" — otherwise the staleness check fails open.
221
+ const msg =
222
+ `_meta.last_updated ${JSON.stringify(meta.last_updated)} is not a valid ISO date ` +
223
+ `(YYYY-MM-DD calendar date) — cannot evaluate freshness. ` +
224
+ `Promoted to an error under --strict.`;
225
+ if (opts && (opts.strict || opts.errorOnStale)) {
226
+ errors.push(msg);
227
+ } else {
228
+ warnings.push(msg);
229
+ }
230
+ } else {
187
231
  const ageDays = Math.floor((Date.now() - lu.getTime()) / 86400000);
188
232
  if (ageDays > fp.stale_after_days) {
189
233
  const msg =
@@ -255,4 +299,4 @@ if (require.main === module) {
255
299
  main();
256
300
  }
257
301
 
258
- module.exports = { validateMeta };
302
+ module.exports = { validateMeta, parseIsoDateStrict };
@@ -244,6 +244,16 @@ function isUsableDate(value) {
244
244
  function additionalChecks(key, entry, ctx) {
245
245
  const warnings = [];
246
246
 
247
+ // A non-object entry has no checkable sub-fields, and validate() already
248
+ // emits the top-level type error for it (lines 127-133). Guarding here turns
249
+ // the uncaught `entry.poc_available` TypeError on a null/array entry into a
250
+ // clean no-op so main() still prints that FAIL and continues to later
251
+ // entries instead of aborting the whole gate. The FAIL is preserved — it
252
+ // originates in validate(), not here.
253
+ if (!entry || typeof entry !== 'object' || Array.isArray(entry)) {
254
+ return [];
255
+ }
256
+
247
257
  // V1 — Hard Rule #14 conditional: poc + public-exploit URL → iocs required.
248
258
  if (entry.poc_available === true) {
249
259
  const sources = Array.isArray(entry.verification_sources)
@@ -322,6 +322,16 @@ function obligationKey(o) {
322
322
 
323
323
  function checkCrossRefs(playbook, ctx, playbookIds) {
324
324
  const findings = [];
325
+ // A null/array/primitive playbook has no cross-refs to check, and validate()
326
+ // already emits the top-level `expected type "object", got null` error for
327
+ // it (main() line ~755). Guarding here turns the uncaught `playbook._meta`
328
+ // TypeError on a literal-null playbook file into a clean no-op so main()
329
+ // still reports the FAIL and continues to the remaining playbooks instead of
330
+ // aborting the whole gate. The FAIL is preserved — it originates in
331
+ // validate(), not here.
332
+ if (!playbook || typeof playbook !== 'object' || Array.isArray(playbook)) {
333
+ return findings;
334
+ }
325
335
  const meta = playbook._meta || {};
326
336
  const phases = playbook.phases || {};
327
337
  const domain = playbook.domain || {};
@@ -533,7 +543,15 @@ function checkCrossRefs(playbook, ctx, playbookIds) {
533
543
  // Case-insensitive + word-bounded so `HTTPS://`, `Curl`, and `fetch(` (no
534
544
  // trailing space) still flag a network source — otherwise an artifact could
535
545
  // ship under air_gap_mode with no offline alternative and run incomplete.
536
- const netSourceRe = /(https?:\/\/|\bgh (?:api|release)\b|\bcurl\b|\bwget\b|\bfetch\b)/i;
546
+ // Network-source detection includes API-verb-phrased sources ("GET
547
+ // /directoryRoles via Graph", "Entra ID", "Okta", "Microsoft Graph") so a
548
+ // REST/Graph endpoint described in prose still flags under air_gap_mode and
549
+ // is not silently collected offline-incomplete. `api/v\d` is deliberately
550
+ // NOT a token — it false-positives on local code-scan artifacts that merely
551
+ // reference an API path. NOTE: lib/schemas/playbook.schema.json carries the
552
+ // same narrow `source` pattern and must be broadened in lockstep with this
553
+ // regex (main-thread item — that file is not edited here).
554
+ const netSourceRe = /(https?:\/\/|\bgh (?:api|release)\b|\bcurl\b|\bwget\b|\bfetch\b|\b(?:GET|POST|PUT|PATCH|DELETE)\s+\/|\bGraph\b|\b(?:Okta|Entra ID|Microsoft Graph)\b)/i;
537
555
  for (const [i, art] of (look.artifacts || []).entries()) {
538
556
  if (!art || typeof art !== 'object') continue;
539
557
  if (typeof art.source === 'string' && netSourceRe.test(art.source)) {