@blamejs/exceptd-skills 0.18.6 → 0.18.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/bin/exceptd.js +261 -56
  3. package/data/_indexes/_meta.json +22 -3
  4. package/data/cve-catalog.json +25 -0
  5. package/data/playbooks/framework.json +2 -2
  6. package/data/playbooks/post-quantum-migration.json +1 -1
  7. package/lib/auto-discovery.js +30 -10
  8. package/lib/collectors/ai-api.js +9 -2
  9. package/lib/collectors/cicd-pipeline-compromise.js +24 -5
  10. package/lib/collectors/cred-stores.js +17 -4
  11. package/lib/collectors/crypto.js +9 -2
  12. package/lib/collectors/hardening.js +9 -2
  13. package/lib/collectors/library-author.js +24 -3
  14. package/lib/collectors/mcp.js +9 -2
  15. package/lib/collectors/runtime.js +9 -2
  16. package/lib/collectors/sbom.js +28 -15
  17. package/lib/collectors/scan-excludes.js +25 -0
  18. package/lib/collectors/secrets.js +40 -4
  19. package/lib/cve-curation.js +84 -8
  20. package/lib/lint-skills.js +75 -3
  21. package/lib/playbook-runner.js +375 -40
  22. package/lib/prefetch.js +6 -1
  23. package/lib/refresh-external.js +32 -1
  24. package/lib/refresh-network.js +201 -24
  25. package/lib/schemas/cve-catalog.schema.json +5 -0
  26. package/lib/scoring.js +106 -13
  27. package/lib/sign.js +107 -29
  28. package/lib/source-advisories.js +23 -5
  29. package/lib/source-ghsa.js +25 -1
  30. package/lib/source-osv.js +26 -1
  31. package/lib/upstream-check.js +1 -1
  32. package/lib/validate-cve-catalog.js +19 -3
  33. package/lib/validate-indexes.js +62 -1
  34. package/lib/validate-playbooks.js +4 -1
  35. package/lib/validate-vendor.js +69 -8
  36. package/manifest.json +53 -53
  37. package/orchestrator/index.js +69 -13
  38. package/package.json +1 -1
  39. package/sbom.cdx.json +124 -124
  40. package/scripts/audit-cross-skill.js +1 -1
  41. package/scripts/bootstrap.js +1 -0
  42. package/scripts/build-indexes.js +58 -4
  43. package/scripts/check-agents-md-collectors.js +41 -13
  44. package/scripts/check-changelog-extract.js +4 -4
  45. package/scripts/check-codebase-patterns.js +19 -5
  46. package/scripts/check-manifest-snapshot.js +74 -30
  47. package/scripts/check-sbom-currency.js +25 -5
  48. package/scripts/check-test-count.js +26 -7
  49. package/scripts/check-test-coverage.js +44 -4
  50. package/scripts/check-version-tags.js +27 -8
  51. package/scripts/predeploy.js +1 -1
  52. package/scripts/refresh-manifest-snapshot.js +14 -4
  53. package/scripts/refresh-reverse-refs.js +7 -1
  54. package/scripts/refresh-sbom.js +1 -1
  55. package/scripts/release.js +3 -3
  56. package/scripts/run-e2e-scenarios.js +18 -8
  57. package/scripts/validate-vendor-online.js +20 -2
  58. package/scripts/verify-shipped-tarball.js +65 -6
  59. package/sources/validators/cve-validator.js +17 -1
  60. package/vendor/blamejs/_PROVENANCE.json +4 -2
@@ -357,19 +357,53 @@ function preflight(playbook, runOpts = {}) {
357
357
  return { ok: true, issues };
358
358
  }
359
359
 
360
- // lockDir lives at a stable global path so two CLI invocations from
360
+ // lockDir lives at a stable per-user path so two CLI invocations from
361
361
  // different working directories still share lock state for cross-process
362
362
  // mutex enforcement. A process.cwd()-relative dir would let invocations
363
363
  // from /tmp and from /home/user/project simultaneously each see an empty
364
- // locks dir and both run unchallenged. The path
365
- // keys on os.platform() so Windows/macOS/Linux locks live under separate
366
- // directories (avoids cross-platform stale-PID confusion when a host is
367
- // shared across OSes via networked FS). Override via EXCEPTD_LOCK_DIR for
368
- // container/CI scenarios that need an explicit shared location.
364
+ // locks dir and both run unchallenged.
365
+ //
366
+ // Resolution order (most-specific first):
367
+ // 1. EXCEPTD_LOCK_DIR explicit container/CI override
368
+ // 2. EXCEPTD_HOME || ~/.exceptd + /locks/<platform> per-user default
369
+ // 3. os.tmpdir()/exceptd-locks-<platform> last-resort fallback
370
+ // when the per-user home is non-writable (read-only home, restricted
371
+ // sandbox/CI runner).
372
+ //
373
+ // The per-user home (mirroring the orchestrator's watch.lock + attestation
374
+ // roots) is both the safer location — a per-user dir is not the
375
+ // world-writable shared OS tempdir a preplant/symlink attack targets — and
376
+ // the better fit for lockDir's stated goal: ~/.exceptd is per-user-stable
377
+ // across working directories, whereas the shared tmpdir is the weaker
378
+ // choice. The path keys on os.platform() so Windows/macOS/Linux locks live
379
+ // under separate directories (avoids cross-platform stale-PID confusion when
380
+ // a host is shared across OSes via networked FS).
381
+ function resolveLockDir() {
382
+ if (process.env.EXCEPTD_LOCK_DIR) return process.env.EXCEPTD_LOCK_DIR;
383
+ const home = process.env.EXCEPTD_HOME || (os.homedir() && path.join(os.homedir(), '.exceptd'));
384
+ if (home) {
385
+ const dir = path.join(home, 'locks', process.platform);
386
+ try {
387
+ fs.mkdirSync(dir, { recursive: true, mode: 0o700 });
388
+ // Probe writability with a marker; remove on success. A home that
389
+ // mkdirs but can't be written (read-only mount, restrictive ACL) must
390
+ // fall through to the tmpdir fallback rather than silently no-op every
391
+ // lock write.
392
+ const probe = path.join(dir, `.write-probe-${process.pid}`);
393
+ fs.writeFileSync(probe, '');
394
+ fs.unlinkSync(probe);
395
+ return dir;
396
+ } catch { /* home non-writable — fall through to tmpdir */ }
397
+ }
398
+ return path.join(os.tmpdir(), `exceptd-locks-${process.platform}`);
399
+ }
400
+
369
401
  function lockDir() {
370
- const dir = process.env.EXCEPTD_LOCK_DIR
371
- || path.join(os.tmpdir(), `exceptd-locks-${process.platform}`);
372
- try { fs.mkdirSync(dir, { recursive: true }); } catch {}
402
+ const dir = resolveLockDir();
403
+ // Owner-only (0700): the mutex lock files inside use O_EXCL ('wx'), but
404
+ // tightening the parent directory keeps another local user from listing or
405
+ // tampering with this user's lock set.
406
+ try { fs.mkdirSync(dir, { recursive: true, mode: 0o700 }); } catch { /* exists / EACCES — non-fatal */ }
373
407
  return dir;
374
408
  }
375
409
 
@@ -388,14 +422,27 @@ function lockFilePath(playbookId) {
388
422
  // lifetime.
389
423
  const STALE_LOCK_MS = 30_000;
390
424
 
425
+ // Create the mutex lock file with an exclusive, owner-only descriptor
426
+ // (O_CREAT | O_EXCL | 0o600 via openSync 'wx'). The lock-file NAME is
427
+ // deliberately predictable — it IS the cross-process mutex, so every contender
428
+ // must agree on it; security comes from the exclusive create (a preplanted file
429
+ // or symlink makes the open throw EEXIST/EPERM, never a silent follow), the
430
+ // 0o700 lock directory, and the 0o600 mode — written through openSync so the
431
+ // secure-create primitive is explicit rather than a bare writeFileSync. Throws
432
+ // EEXIST when the lock is already held, which every caller already handles.
433
+ function writeLockFile(p, playbookId) {
434
+ const fd = fs.openSync(p, 'wx', 0o600);
435
+ try {
436
+ fs.writeSync(fd, JSON.stringify({ pid: process.pid, started_at: new Date().toISOString(), playbook: playbookId }, null, 2));
437
+ } finally {
438
+ fs.closeSync(fd);
439
+ }
440
+ }
441
+
391
442
  function acquireLock(playbookId) {
392
443
  const p = lockFilePath(playbookId);
393
444
  if (!p) return null;
394
- const writePayload = () => fs.writeFileSync(
395
- p,
396
- JSON.stringify({ pid: process.pid, started_at: new Date().toISOString(), playbook: playbookId }, null, 2),
397
- { flag: 'wx' }
398
- );
445
+ const writePayload = () => writeLockFile(p, playbookId);
399
446
  try {
400
447
  writePayload();
401
448
  return p;
@@ -458,9 +505,7 @@ function acquireLockDiagnostic(playbookId) {
458
505
  const p = lockFilePath(playbookId);
459
506
  if (!p) return { ok: false, reason: 'no_lock_path' };
460
507
  try {
461
- fs.writeFileSync(p,
462
- JSON.stringify({ pid: process.pid, started_at: new Date().toISOString(), playbook: playbookId }, null, 2),
463
- { flag: 'wx' });
508
+ writeLockFile(p, playbookId);
464
509
  return { ok: true, path: p };
465
510
  } catch (e) {
466
511
  if (e && (e.code === 'EEXIST' || e.code === 'EPERM')) {
@@ -476,9 +521,7 @@ function acquireLockDiagnostic(playbookId) {
476
521
  if (Number.isInteger(pid) && pid > 0 && pid !== process.pid && !pidAlive(pid)) {
477
522
  try { fs.unlinkSync(p); } catch {}
478
523
  try {
479
- fs.writeFileSync(p,
480
- JSON.stringify({ pid: process.pid, started_at: new Date().toISOString(), playbook: playbookId }, null, 2),
481
- { flag: 'wx' });
524
+ writeLockFile(p, playbookId);
482
525
  return { ok: true, path: p, reclaimed_from_pid: pid };
483
526
  } catch (e2) {
484
527
  return { ok: false, reason: 'reclaim_failed', error: e2.message, lock_path: p, holder_pid: pid };
@@ -494,9 +537,7 @@ function acquireLockDiagnostic(playbookId) {
494
537
  if (mtimeMs !== null && (Date.now() - mtimeMs) > STALE_LOCK_MS) {
495
538
  try { fs.unlinkSync(p); } catch {}
496
539
  try {
497
- fs.writeFileSync(p,
498
- JSON.stringify({ pid: process.pid, started_at: new Date().toISOString(), playbook: playbookId }, null, 2),
499
- { flag: 'wx' });
540
+ writeLockFile(p, playbookId);
500
541
  return { ok: true, path: p, reclaimed_self_stale_pid: true, prior_mtime_ms: mtimeMs };
501
542
  } catch (e3) {
502
543
  return { ok: false, reason: 'reclaim_failed', error: e3.message, lock_path: p, holder_pid: pid };
@@ -1058,6 +1099,14 @@ function analyze(playbookId, directiveId, detectResult, agentSignals = {}, runOp
1058
1099
  const vexStatus = (vexFixed && vexFixed.has(c.cve_id)) ? 'fixed' : null;
1059
1100
  return {
1060
1101
  cve_id: c.cve_id,
1102
+ // attack_class is the coarse chainable taxonomy (kernel-lpe /
1103
+ // mcp-supply-chain / ai-c2 / prompt-injection / container-escape / …)
1104
+ // the sbom -> deep-dive feeds_into rules quantify over
1105
+ // (`any matched_cve.attack_class == 'kernel-lpe'`). It comes from the
1106
+ // catalog entry's explicit attack_class only — no fuzzy inference from
1107
+ // the free-form `type`, so an unclassified CVE stays null and the chain
1108
+ // correctly does not fire (no false escalation) rather than misrouting.
1109
+ attack_class: c.entry?.attack_class ?? null,
1061
1110
  rwep: c.rwep_score,
1062
1111
  cvss_score: c.entry?.cvss_score ?? null,
1063
1112
  cvss_vector: c.entry?.cvss_vector ?? null,
@@ -1408,7 +1457,7 @@ function analyze(playbookId, directiveId, detectResult, agentSignals = {}, runOp
1408
1457
  findingShape = {};
1409
1458
  }
1410
1459
  for (const ec of an.escalation_criteria || []) {
1411
- if (evalCondition(ec.condition, { ...agentSignals, ...evalCtxRoot, rwep: adjustedRwep, blast_radius_score: blastRadiusScore, theater_verdict: theaterVerdict, analyze: result, finding: findingShape }, playbook)) {
1460
+ if (evalCondition(ec.condition, { ...agentSignals, ...evalCtxRoot, rwep: adjustedRwep, blast_radius_score: blastRadiusScore, theater_verdict: theaterVerdict, compliance_theater_check: result.compliance_theater_check, jurisdiction_obligations: (playbook.phases && playbook.phases.govern && playbook.phases.govern.jurisdiction_obligations) || [], analyze: result, matched_cve: result.matched_cves || [], finding: findingShape }, playbook)) {
1412
1461
  escalations.push({ condition: ec.condition, action: ec.action, target_playbook: ec.target_playbook || null });
1413
1462
  }
1414
1463
  }
@@ -1932,7 +1981,32 @@ function close(playbookId, directiveId, analyzeResult, validateResult, agentSign
1932
1981
  // and could suppress a legitimate downstream chain.
1933
1982
  ...agentSignals,
1934
1983
  rwep: analyzeResult.rwep?.adjusted,
1984
+ // Bare-token parity with the escalation context (analyze()): catalog
1985
+ // feeds_into conditions reference the unqualified tokens `blast_radius_score`
1986
+ // and `theater_verdict` (e.g. framework.json's feeds_into into sbom). Without
1987
+ // these top-level keys resolvePath returns null and `null >= 4` is false
1988
+ // regardless of the engine-computed blast radius, so the chain is dead. They
1989
+ // are spread AFTER ...agentSignals so the engine value wins over any
1990
+ // operator-submitted signals.blast_radius_score / signals.theater_verdict.
1991
+ blast_radius_score: analyzeResult.blast_radius_score,
1992
+ theater_verdict: analyzeResult.compliance_theater_check?.verdict,
1993
+ // Bare top-level `compliance_theater_check` so catalog conditions that
1994
+ // reference `compliance_theater_check.verdict` unqualified (framework.json's
1995
+ // feeds_into into sbom) resolve — without it resolvePath returns null and the
1996
+ // clause is dead regardless of the engine verdict. The `analyze.*` alias
1997
+ // below stays for conditions that use the qualified path.
1998
+ compliance_theater_check: analyzeResult.compliance_theater_check,
1999
+ // Govern-phase jurisdiction obligations so feeds_into conditions like
2000
+ // `… AND jurisdiction_obligations contains 'EU'` (framework → sbom) resolve.
2001
+ jurisdiction_obligations: (g && g.jurisdiction_obligations) || (playbook.phases && playbook.phases.govern && playbook.phases.govern.jurisdiction_obligations) || [],
1935
2002
  theater_score: analyzeResult.compliance_theater_check?.verdict === 'theater' ? 0 : 100,
2003
+ // Top-level matched_cve array so the shipped sbom feeds_into quantifiers
2004
+ // (`any matched_cve.attack_class == 'kernel-lpe'`, … IN ['ai-c2', …]) re-root
2005
+ // at each matched CVE. Without it the quantifier head resolves null and the
2006
+ // sbom -> kernel/mcp/ai-api deep-dive chains stay dead even when a matched
2007
+ // CVE carries the attack_class. The `analyze.matched_cves` alias stays for
2008
+ // conditions that use the qualified path.
2009
+ matched_cve: analyzeResult.matched_cves || [],
1936
2010
  analyze: analyzeResult,
1937
2011
  validate: validateResult,
1938
2012
  finding: analyzeFindingShape(analyzeResult),
@@ -2397,11 +2471,12 @@ function sanitizeOperatorText(s) {
2397
2471
  // U+007F, U+0080-009F, private-use, unassigned), so the family pass is a
2398
2472
  // documented-intent superset removal and the result is identical to the
2399
2473
  // single \p{C} strip.
2400
- const familyStripped = normalised
2401
- .replace(codepointClass.BIDI_RE_G, '')
2402
- .replace(codepointClass.C0_CTRL_RE_G, '')
2403
- .replace(codepointClass.ZW_RE_G, '')
2404
- .replace(codepointClass.NULL_RE_G, '');
2474
+ const familyStripped = codepointClass.applyCharStripPolicies(normalised, {
2475
+ bidiPolicy: 'strip',
2476
+ controlPolicy: 'strip',
2477
+ zeroWidthPolicy: 'strip',
2478
+ nullBytePolicy: 'strip',
2479
+ });
2405
2480
  const stripped = familyStripped.replace(/\p{C}/gu, '');
2406
2481
  const trimmed = stripped.trim();
2407
2482
  if (trimmed.length === 0) return null;
@@ -3309,7 +3384,23 @@ function normalizeSubmission(submission, playbook) {
3309
3384
  }
3310
3385
  if (typeof val === "object" && val !== null) {
3311
3386
  const aid = knownArtifacts.has(key) ? key : (val.artifact || key);
3312
- out.artifacts[aid] = { value: val.value, captured: val.captured !== false };
3387
+ // Preserve every evidence-bearing key the observation carried, not just
3388
+ // `value`. An observation whose secret/path lives under a non-`value`
3389
+ // key (e.g. { path, matched, reason } — a natural collector shape) was
3390
+ // previously collapsed to { value: undefined, captured } here, silently
3391
+ // discarding the actual evidence. Two observations capturing DIFFERENT
3392
+ // secrets under path/matched then hashed and diffed byte-identical, so
3393
+ // `attest diff` reported a false "unchanged" and masked real drift.
3394
+ // The reserved control keys (artifact/indicator/result) drive
3395
+ // signal_overrides below and are intentionally not echoed as evidence.
3396
+ const { artifact: _a, indicator: _i, result: _r, captured: _c, value: _v, ...evidence } = val;
3397
+ const normalizedArtifact = { captured: val.captured !== false };
3398
+ if (val.value !== undefined) normalizedArtifact.value = val.value;
3399
+ for (const [ek, ev] of Object.entries(evidence)) {
3400
+ if (ek === "__proto__" || ek === "constructor" || ek === "prototype") continue;
3401
+ normalizedArtifact[ek] = ev;
3402
+ }
3403
+ out.artifacts[aid] = normalizedArtifact;
3313
3404
  if (val.indicator && val.result !== undefined) {
3314
3405
  const newVerdict = canonicalizeOutcome(val.result);
3315
3406
  if (out.signal_overrides[val.indicator] !== undefined && out._signal_origins[val.indicator] !== undefined) {
@@ -3861,6 +3952,15 @@ function extractSubmissionForHash(sub) {
3861
3952
  return pick;
3862
3953
  }
3863
3954
 
3955
+ // Identity fields a `contains`/`includes` membership test targets when the
3956
+ // array holds objects (e.g. jurisdiction_obligations records). Membership is
3957
+ // field-scoped to these so a non-identity field (clock_starts, obligation,
3958
+ // regulation, a free-form tag) that happens to equal the member cannot satisfy
3959
+ // the predicate. `jurisdiction` is the only field the catalog's object-array
3960
+ // `contains` conditions target today; the allowlist is the seam to extend if a
3961
+ // future condition deliberately matches a different identity field.
3962
+ const OBJECT_MEMBERSHIP_FIELDS = ['jurisdiction'];
3963
+
3864
3964
  function evalCondition(expr, ctx, playbook) {
3865
3965
  if (!expr) return false;
3866
3966
  expr = expr.trim();
@@ -3879,6 +3979,91 @@ function evalCondition(expr, ctx, playbook) {
3879
3979
  const andParts = splitAtTopLevel(expr, 'AND');
3880
3980
  if (andParts.length > 1) return andParts.every(s => evalCondition(s, ctx, playbook));
3881
3981
 
3982
+ // Quantifier prefix: catalog conditions write `any <path> <op> <value>`
3983
+ // (e.g. framework.json's feeds_into `any compliance_theater_check.verdict ==
3984
+ // 'theater' …`). An unhandled `any `/`all ` prefix is unparseable and falls
3985
+ // through to `false`, silently disabling the clause it gates. Strip the
3986
+ // quantifier and re-evaluate the inner comparison: when the LHS path resolves
3987
+ // to a scalar (the framework theater_verdict case) the quantifier is prose
3988
+ // emphasis and the scalar comparison is the intended test; when the LHS first
3989
+ // segment resolves to an array, apply the predicate existentially (`any`) /
3990
+ // universally (`all`) across its elements. The inner clause is only the leaf
3991
+ // comparison (the surrounding AND/OR has already been split off above).
3992
+ const quant = expr.match(/^(any|all)\s+(.+)$/);
3993
+ if (quant) {
3994
+ const [, kind, inner] = quant;
3995
+ // `any head.rest <op|keyword> …` → re-root the predicate at each element of
3996
+ // `head` and apply it existentially (`any`) / universally (`all`). The head
3997
+ // is the FIRST dotted-path token of the inner clause; what follows it (`==`,
3998
+ // `IN [...]`, `contains`, `matches /…/`, `>=`, …) is operator-agnostic — the
3999
+ // inner clause is re-evaluated whole against `{ …ctx, [head]: el }`, so every
4000
+ // operator the leaf parser already understands works under a quantifier.
4001
+ // Earlier this branch only re-rooted clauses whose operator was in the
4002
+ // comparison set (`>=|<=|==|=|<|>|!=`); `IN`/`contains`/`matches` were
4003
+ // omitted, so e.g. sbom.json's `any matched_cve.attack_class IN
4004
+ // ['ai-c2','prompt-injection']` (the sbom→ai-api feeds_into trigger) fell
4005
+ // through to a whole-ctx eval where `matched_cve.attack_class` resolved to
4006
+ // undefined on the array and the clause was permanently false, while the
4007
+ // sibling `== 'kernel-lpe'` quantifiers fired. Requiring a `.`-qualified head
4008
+ // keeps the bare-token form (`any <path>`) routed to the non-emptiness branch
4009
+ // below.
4010
+ const headMatch = inner.match(/^([A-Za-z_][\w-]*)(?:\.[A-Za-z_][\w-]*)+(?:[\s.[]|$)/);
4011
+ if (headMatch) {
4012
+ const head = headMatch[1];
4013
+ const arr = resolvePath(ctx, head);
4014
+ if (Array.isArray(arr)) {
4015
+ const test = el => evalCondition(inner, { ...ctx, [head]: el }, playbook);
4016
+ return kind === 'all' ? arr.length > 0 && arr.every(test) : arr.some(test);
4017
+ }
4018
+ }
4019
+ // Bare-quantifier non-emptiness form: `any <path>` / `all <path>` with no
4020
+ // comparison operator is a quantifier OVER the resolved collection itself —
4021
+ // the author's intent is "at least one X exists" (`any`) / "every X is
4022
+ // truthy" (`all`), i.e. a non-emptiness / existence test. sbom.json's
4023
+ // `any actively_exploited_match …` EU CRA Art.14 (24h) notify_legal
4024
+ // escalation is the canonical case. Without this, the inner bare token has
4025
+ // no operator, so no comparison branch parses it and it falls through to
4026
+ // condition_unparsed → false, silently disabling the escalation even when
4027
+ // the array holds entries. Match ONLY a pure dotted-path token (no operator,
4028
+ // keyword, bracket, or space); anything with a comparison/keyword still
4029
+ // routes to the prose fall-through below so a genuinely malformed inner
4030
+ // clause stays observable.
4031
+ if (/^[A-Za-z_][\w-]*(?:\.[A-Za-z_][\w-]*)*$/.test(inner)) {
4032
+ const v = resolvePath(ctx, inner);
4033
+ if (Array.isArray(v)) {
4034
+ return kind === 'all' ? v.length > 0 && v.every(Boolean) : v.length > 0;
4035
+ }
4036
+ // Non-array scalar: existence/truthiness (null/undefined/0/'' → false).
4037
+ return !!v;
4038
+ }
4039
+ // Scalar (or unresolved-array) LHS: the quantifier is prose; evaluate the
4040
+ // bare inner comparison. If it still doesn't parse it falls through to the
4041
+ // condition_unparsed diagnostic below, so a genuinely malformed clause stays
4042
+ // observable rather than silently passing.
4043
+ return evalCondition(inner, ctx, playbook);
4044
+ }
4045
+
4046
+ // Membership clauses (`contains`/`includes`/`IN [...]`) parse fine but resolve
4047
+ // their LHS path to a collection. When that path does NOT exist in ctx (the
4048
+ // resolved value is null/undefined — an authoring typo in the LHS token, or a
4049
+ // wrong-shape ctx that never populated the collection), the branch returns a
4050
+ // silent `false` that disables whatever escalation / feeds_into the clause
4051
+ // gates with no signal. The clause PARSED, so condition_unparsed (a
4052
+ // parse-failure diagnostic) never fires. Surface a distinct, low-severity
4053
+ // condition_path_unresolved so an LHS-path typo is observable — deduped on the
4054
+ // condition string like condition_unparsed. A present-but-empty array (or any
4055
+ // present value that simply isn't a collection) is a LEGITIMATE false, not an
4056
+ // unresolved path, so it does NOT push: only a literally-absent path does.
4057
+ const pushPathUnresolved = () => {
4058
+ const target = (ctx && Array.isArray(ctx._runErrors)) ? ctx._runErrors
4059
+ : (playbook && Array.isArray(playbook._runErrors)) ? playbook._runErrors
4060
+ : null;
4061
+ if (target) {
4062
+ pushRunError(target, { kind: 'condition_path_unresolved', condition: String(expr).slice(0, 200) },
4063
+ { dedupeKey: x => x.condition || '' });
4064
+ }
4065
+ };
4066
+
3882
4067
  // "rwep >= 90". The LHS/RHS path tokens admit hyphens because signal and
3883
4068
  // indicator IDs are canonically hyphenated across the catalog (e.g.
3884
4069
  // `no-security-md`, `kver-in-affected-range`); a `\w`-only token silently
@@ -3921,15 +4106,79 @@ function evalCondition(expr, ctx, playbook) {
3921
4106
  // "scope.targets includes named_remote" — `contains` is accepted as a
3922
4107
  // synonym for `includes` (the catalog uses both); hyphenated paths + members
3923
4108
  // are admitted for the same reason as the comparison branch above.
3924
- m = expr.match(/^([A-Za-z_][\w-]*(?:\.[A-Za-z_][\w-]*)*)\s+(?:includes|contains)\s+([\w-]+)$/);
4109
+ // The member may be a bare token OR a quoted string (the catalog's
4110
+ // `jurisdiction_obligations contains 'EU'` / `contains 'EU/EU CRA Art.14 24h'`
4111
+ // forms). When the array holds objects (jurisdiction_obligations are records
4112
+ // like `{ jurisdiction:'EU', regulation:'NIS2 Art.21', … }`), membership is
4113
+ // field-TARGETED: it matches an identity field of the obligation, not ANY
4114
+ // field value. An unscoped `Object.values(el).includes(member)` over-matched
4115
+ // — `contains 'EU'` would be satisfied by a non-jurisdiction field that
4116
+ // happened to equal 'EU' (e.g. a tag), and `contains 'detect_confirmed'`
4117
+ // would match the unrelated `clock_starts` field that holds that exact value
4118
+ // in the shipped obligations. Scoping to OBJECT_MEMBERSHIP_FIELDS keeps the
4119
+ // intended "the obligation is for jurisdiction X" semantic and prevents a
4120
+ // future field collision (or an operator-/agent-supplied obligations array)
4121
+ // from forcing a notify_legal escalation via a non-jurisdiction field.
4122
+ m = expr.match(/^([A-Za-z_][\w-]*(?:\.[A-Za-z_][\w-]*)*)\s+(?:includes|contains)\s+(?:'([^']+)'|"([^"]+)"|([\w-]+))$/);
3925
4123
  if (m) {
4124
+ const member = m[2] !== undefined ? m[2] : (m[3] !== undefined ? m[3] : m[4]);
3926
4125
  const arr = resolvePath(ctx, m[1]);
3927
- return Array.isArray(arr) && arr.includes(m[2]);
4126
+ if (!Array.isArray(arr)) {
4127
+ if (arr == null) pushPathUnresolved();
4128
+ return false;
4129
+ }
4130
+ return arr.some((el) =>
4131
+ el === member ||
4132
+ (el && typeof el === 'object' &&
4133
+ OBJECT_MEMBERSHIP_FIELDS.some((f) => el[f] === member))
4134
+ );
4135
+ }
4136
+
4137
+ // "matched_cve.attack_class IN ['kernel-lpe', 'rce']" — membership against a
4138
+ // bracketed literal list; members may be quoted or bare. A scalar LHS must be
4139
+ // in the list; an array LHS must intersect it. Unhandled before, so every
4140
+ // `IN [...]` escalation/feeds_into atom fell through to a silent false.
4141
+ // The member list is split quote-aware: a naive `split(',')` is unaware of
4142
+ // quotes, so a quoted member that itself contains a comma (`'EU, US'`) would
4143
+ // be broken into two members (`EU` and `US`), neither of which equals the
4144
+ // author's intended whole member — the clause then evaluated false with no
4145
+ // diagnostic (the regex still matched the bracket, so condition_unparsed never
4146
+ // fired). splitInMembers walks the list tracking single/double quote state and
4147
+ // only splits on commas at quote-depth 0, then strips the surrounding quotes.
4148
+ // The CLOSING `]` is located quote-aware too: a `[^\]]*` capture stops at the
4149
+ // FIRST `]`, so a quoted member containing a literal `]` (`'a]b'`) truncated
4150
+ // the list early and left trailing text the `$` anchor couldn't match — the
4151
+ // whole clause then fell through to condition_unparsed for every input. The
4152
+ // matching bracket is now found at quote-depth 0 (an unquoted `]` is still the
4153
+ // terminator; a `]` inside a quoted member is part of that member). Trailing
4154
+ // text after the closing bracket, or an unterminated bracket, still fails to
4155
+ // match and surfaces as condition_unparsed.
4156
+ m = expr.match(/^([A-Za-z_][\w-]*(?:\.[A-Za-z_][\w-]*)*)\s+IN\s+\[/);
4157
+ if (m) {
4158
+ const body = sliceInBracketBody(expr.slice(m[0].length));
4159
+ if (body !== null) {
4160
+ const members = splitInMembers(body);
4161
+ const lv = resolvePath(ctx, m[1]);
4162
+ if (Array.isArray(lv)) return lv.some((x) => members.includes(String(x)));
4163
+ // lv == null means the LHS path doesn't exist in ctx (typo / wrong-shape) —
4164
+ // surface it as condition_path_unresolved so the dead clause is observable,
4165
+ // mirroring the contains/includes branch. A present scalar that's simply not
4166
+ // in the member list is a legitimate false and pushes nothing.
4167
+ if (lv == null) { pushPathUnresolved(); return false; }
4168
+ return members.includes(String(lv));
4169
+ }
4170
+ // body === null → no quote-depth-0 closing `]` ends the string; fall through
4171
+ // to the condition_unparsed diagnostic so the malformed clause stays visible.
3928
4172
  }
3929
4173
 
3930
- // "matched_cve.vector matches /regex/"
3931
- m = expr.match(/^([A-Za-z_][\w-]*(?:\.[A-Za-z_][\w-]*)*)\s+matches\s+\/(.+)\/$/);
4174
+ // "matched_cve.vector matches /regex/" — both delimiters are accepted: the
4175
+ // slash form `matches /re/` and the quote form `matches 're'` / `matches "re"`.
4176
+ // The catalog authors both (mcp.json's feeds_into uses the quoted form), and a
4177
+ // delimiter-specific parser silently disabled whichever form it didn't match —
4178
+ // the same class as the hyphenated-token gap above.
4179
+ m = expr.match(/^([A-Za-z_][\w-]*(?:\.[A-Za-z_][\w-]*)*)\s+matches\s+(?:\/(.+)\/|'([^']+)'|"([^"]+)")$/);
3932
4180
  if (m) {
4181
+ const pattern = m[2] !== undefined ? m[2] : (m[3] !== undefined ? m[3] : m[4]);
3933
4182
  const val = resolvePath(ctx, m[1]);
3934
4183
  if (typeof val !== 'string') return false;
3935
4184
  // An operator-supplied or playbook-supplied regex with a syntax bug
@@ -3939,9 +4188,9 @@ function evalCondition(expr, ctx, playbook) {
3939
4188
  // analyze() can surface analyze.runtime_errors[] without losing the
3940
4189
  // diagnostic.
3941
4190
  try {
3942
- return new RegExp(m[2], 'i').test(val); // allow:dynamic-regex — m[2] comes from an Ed25519-signed catalog playbook condition (/…/), so the pattern cannot be attacker-controlled without breaking the signature; the try/catch covers construction-time syntax errors only (it does NOT defend against catastrophic backtracking — do not reuse this shape for operator-supplied patterns)
4191
+ return new RegExp(pattern, 'i').test(val); // allow:dynamic-regex — pattern comes from an Ed25519-signed catalog playbook condition (/…/ or '…'), so it cannot be attacker-controlled without breaking the signature; the try/catch covers construction-time syntax errors only (it does NOT defend against catastrophic backtracking — do not reuse this shape for operator-supplied patterns)
3943
4192
  } catch (e) {
3944
- const errorRec = { _regex_eval_error: { source: m[1], expr: m[2], message: e && e.message ? String(e.message) : String(e) } };
4193
+ const errorRec = { _regex_eval_error: { source: m[1], expr: pattern, message: e && e.message ? String(e.message) : String(e) } };
3945
4194
  // Two sites where ctx may carry an accumulator: runOpts._runErrors
3946
4195
  // (threaded from run()) or ctx._runErrors directly. Prefer the runOpts
3947
4196
  // form; fall back to ctx.
@@ -3993,9 +4242,23 @@ function resolvePath(obj, dot) {
3993
4242
  function splitAtTopLevel(expr, sep) {
3994
4243
  const parts = [];
3995
4244
  const needle = ' ' + sep + ' ';
3996
- let depth = 0, buf = '', i = 0;
4245
+ let depth = 0, buf = '', i = 0, quote = null;
3997
4246
  while (i < expr.length) {
3998
4247
  const ch = expr[i];
4248
+ // Inside a quoted string literal, parens and the ` AND `/` OR ` needle are
4249
+ // LITERAL text, not boolean structure. A regex member like `matches 'foo('`
4250
+ // carries an unbalanced `(` that — counted blindly — would leave depth=1 so
4251
+ // a real top-level OR/AND would never split at depth 0 (silently disabling
4252
+ // the disjunct/conjunct), and a member like `contains 'EU AND US'` would be
4253
+ // torn at the inner ` AND ` as if it were an operator. Track quote state and
4254
+ // skip both while inside a quote. An unescaped matching quote closes the
4255
+ // literal; `\'`/`\"` stay in. Mirrors splitInMembers' quote-aware walk.
4256
+ if (quote) {
4257
+ if (ch === '\\' && i + 1 < expr.length) { buf += ch + expr[i + 1]; i += 2; continue; }
4258
+ if (ch === quote) quote = null;
4259
+ buf += ch; i++; continue;
4260
+ }
4261
+ if (ch === "'" || ch === '"') { quote = ch; buf += ch; i++; continue; }
3999
4262
  if (ch === '(') { depth++; buf += ch; i++; continue; }
4000
4263
  if (ch === ')') { depth--; buf += ch; i++; continue; }
4001
4264
  if (depth === 0 && expr.startsWith(needle, i)) {
@@ -4011,18 +4274,90 @@ function splitAtTopLevel(expr, sep) {
4011
4274
  return parts;
4012
4275
  }
4013
4276
 
4277
+ /**
4278
+ * Quote-aware splitter for the body of an `IN [...]` member list. Splits on
4279
+ * commas that are OUTSIDE any quoted run, then strips the surrounding quotes and
4280
+ * trims each member. A naive `.split(',')` is quote-unaware, so a quoted member
4281
+ * that contains a comma (`'EU, US'`) would be torn into two members; this walks
4282
+ * the string tracking single/double quote state so a comma inside a quoted run
4283
+ * is treated as a literal part of that member. Bare (unquoted) members are
4284
+ * supported too — the catalog authors both forms (`'ai-c2', 'prompt-injection'`
4285
+ * and bare `kernel-lpe`). Empty members (e.g. a trailing comma) are dropped.
4286
+ */
4287
+ function splitInMembers(listStr) {
4288
+ const out = [];
4289
+ let buf = '';
4290
+ let quote = null;
4291
+ for (let i = 0; i < listStr.length; i++) {
4292
+ const ch = listStr[i];
4293
+ if (quote) {
4294
+ if (ch === quote) quote = null;
4295
+ else buf += ch;
4296
+ continue;
4297
+ }
4298
+ if (ch === "'" || ch === '"') { quote = ch; continue; }
4299
+ if (ch === ',') { out.push(buf.trim()); buf = ''; continue; }
4300
+ buf += ch;
4301
+ }
4302
+ out.push(buf.trim());
4303
+ return out.filter((s) => s.length);
4304
+ }
4305
+
4306
+ /**
4307
+ * Locate the closing `]` of an `IN [...]` list quote-aware and return the body
4308
+ * between the (already-consumed) `[` and that `]`. `rest` is the text after the
4309
+ * opening bracket. The terminator is the first `]` encountered at quote-depth 0;
4310
+ * a `]` inside a single/double-quoted member (`'a]b'`) is part of the member, not
4311
+ * the terminator. A `[^\]]*` regex capture instead stops at the FIRST `]`, so a
4312
+ * quoted `]` truncated the list and left trailing text the `$` anchor rejected —
4313
+ * the clause then fell through to condition_unparsed for every input.
4314
+ *
4315
+ * Returns the body string on success, or `null` when the list is malformed:
4316
+ * unterminated (no quote-depth-0 `]`) or carrying non-whitespace text after the
4317
+ * closing bracket. `null` lets the caller fall through to the condition_unparsed
4318
+ * diagnostic so a malformed clause stays observable rather than passing silently.
4319
+ */
4320
+ function sliceInBracketBody(rest) {
4321
+ let quote = null;
4322
+ for (let i = 0; i < rest.length; i++) {
4323
+ const ch = rest[i];
4324
+ if (quote) { if (ch === quote) quote = null; continue; }
4325
+ if (ch === "'" || ch === '"') { quote = ch; continue; }
4326
+ if (ch === ']') {
4327
+ // The bracket must be the final structural token: only whitespace may
4328
+ // follow. Anything else (e.g. `IN ['a'] AND …` reaching here, or stray
4329
+ // trailing text) is malformed for this leaf parser.
4330
+ return rest.slice(i + 1).trim() === '' ? rest.slice(0, i) : null;
4331
+ }
4332
+ }
4333
+ return null; // no quote-depth-0 closing bracket → unterminated list
4334
+ }
4335
+
4014
4336
  /**
4015
4337
  * Strip a balanced pair of outer parens, if and only if the very first and last
4016
4338
  * characters are matching parens at the same depth boundary. `(A) AND (B)` keeps
4017
- * its parens; `((A AND B))` peels one layer.
4339
+ * its parens; `((A AND B))` peels one layer. The depth scan is quote-aware: a
4340
+ * paren inside a quoted string literal (e.g. a regex member `matches '(a|b)'` or
4341
+ * an unbalanced `matches 'foo('`) is literal text, not grouping structure, so it
4342
+ * must not move the depth counter — otherwise a quoted `)` could make the outer
4343
+ * pair look unbalanced (skipping a legitimate strip) or an unbalanced quoted `(`
4344
+ * could make a non-wrapping pair look outer-spanning (stripping wrongly).
4018
4345
  */
4019
4346
  function stripOuterParens(expr) {
4020
4347
  while (expr.length >= 2 && expr[0] === '(' && expr[expr.length - 1] === ')') {
4021
4348
  let depth = 0;
4022
4349
  let outerMatches = true;
4350
+ let quote = null;
4023
4351
  for (let i = 0; i < expr.length - 1; i++) {
4024
- if (expr[i] === '(') depth++;
4025
- else if (expr[i] === ')') depth--;
4352
+ const ch = expr[i];
4353
+ if (quote) {
4354
+ if (ch === '\\') { i++; continue; }
4355
+ if (ch === quote) quote = null;
4356
+ continue;
4357
+ }
4358
+ if (ch === "'" || ch === '"') { quote = ch; continue; }
4359
+ if (ch === '(') depth++;
4360
+ else if (ch === ')') depth--;
4026
4361
  if (depth === 0 && i < expr.length - 1) { outerMatches = false; break; }
4027
4362
  }
4028
4363
  if (outerMatches) expr = expr.slice(1, -1).trim();
package/lib/prefetch.js CHANGED
@@ -297,6 +297,11 @@ async function timedFetch(url, headers = {}) {
297
297
  });
298
298
  if (!res.ok) {
299
299
  const err = new Error(`HTTP ${res.status}`);
300
+ // The vendored retry classifier (vendor/blamejs/retry.js isRetryable)
301
+ // keys off err.statusCode — set it so a 429/5xx from KEV/NVD/EPSS/OSV
302
+ // routes through the job-queue backoff instead of being dropped on the
303
+ // first hiccup. err.status kept for callers that read it for messaging.
304
+ err.statusCode = res.status;
300
305
  err.status = res.status;
301
306
  throw err;
302
307
  }
@@ -913,5 +918,5 @@ module.exports = {
913
918
  // Not part of the operator-facing API — internal contract for tests
914
919
  // that need to exercise the lockfile path without spawning the full
915
920
  // prefetch network pipeline.
916
- _internal: { withIndexLock, writeFileAtomic, loadIndex, saveIndex },
921
+ _internal: { withIndexLock, writeFileAtomic, loadIndex, saveIndex, timedFetch },
917
922
  };
@@ -267,11 +267,42 @@ const KEV_SOURCE = {
267
267
  const report = await validateAllCves(ctx.cveCatalog, { concurrency: 4 });
268
268
  const diffs = [];
269
269
  let errors = 0;
270
+ // Mirror the cache path's implausibly-small-feed guard. The live KEV map is
271
+ // fetched once per process and its size is surfaced on every reachable
272
+ // result as fetched.sources.kev.total_entries. A feed that JSON-parses but
273
+ // is far below a real CISA snapshot (a partial CDN response, a momentarily
274
+ // near-empty feed) must not be trusted to de-list curated entries. When the
275
+ // size is known and below the floor we hold ALL de-listings for review,
276
+ // exactly as kevDiffFromCache does; when the size is unknown (no reachable
277
+ // KEV result carried it) we fall back to the per-entry curated-signal guard.
278
+ let liveFeedSize = null;
279
+ for (const r of report.results) {
280
+ const n = r && r.fetched && r.fetched.sources && r.fetched.sources.kev
281
+ && r.fetched.sources.kev.total_entries;
282
+ if (typeof n === "number") { liveFeedSize = n; break; }
283
+ }
284
+ const feedComplete = liveFeedSize === null || liveFeedSize >= KEV_FEED_MIN_PLAUSIBLE;
270
285
  for (const r of report.results) {
271
286
  if (r.status === "unreachable") errors++;
272
287
  for (const d of r.discrepancies || []) {
273
288
  if (d.field === "cisa_kev" || d.field === "cisa_kev_date") {
274
- diffs.push({ id: r.cve_id, field: d.field, before: d.local, after: d.fetched, severity: d.severity });
289
+ const diff = { id: r.cve_id, field: d.field, before: d.local, after: d.fetched, severity: d.severity };
290
+ // Symmetric with the --from-cache path: a LIVE KEV de-listing
291
+ // (true→false) is held for review (applyDiff skips review_only)
292
+ // instead of auto-downgrading the entry when EITHER the entry carries
293
+ // strong human-curated exploitation signal OR the live feed is
294
+ // implausibly small. Without this the live path silently de-listed
295
+ // confirmed-exploitation CVEs the cache path would have held back, and
296
+ // a truncated-but-valid feed could de-list every non-curated entry.
297
+ if (d.field === "cisa_kev" && d.local === true && d.fetched === false &&
298
+ (!feedComplete || hasCuratedExploitSignal(ctx.cveCatalog && ctx.cveCatalog[r.cve_id]))) {
299
+ diff.review_only = true;
300
+ diff.kev_delist_review = true;
301
+ diff.note = !feedComplete
302
+ ? `KEV de-listing held for review: live feed returned only ${liveFeedSize} entries (< ${KEV_FEED_MIN_PLAUSIBLE}), likely incomplete. Confirm against a complete CISA KEV snapshot before de-listing ${r.cve_id}.`
303
+ : `KEV de-listing held for review: ${r.cve_id} carries curated exploitation signal; confirm a genuine CISA removal before downgrading.`;
304
+ }
305
+ diffs.push(diff);
275
306
  }
276
307
  }
277
308
  }