@blamejs/exceptd-skills 0.19.34 → 0.19.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +18 -0
  2. package/bin/exceptd.js +0 -5
  3. package/data/_indexes/_meta.json +2 -2
  4. package/lib/collectors/ai-api.js +150 -22
  5. package/lib/collectors/cicd-pipeline-compromise.js +73 -28
  6. package/lib/collectors/library-author.js +141 -5
  7. package/lib/collectors/sbom.js +96 -12
  8. package/lib/collectors/scan-excludes.js +3 -2
  9. package/lib/cve-regression-watcher.js +0 -3
  10. package/lib/framework-gap.js +5 -0
  11. package/lib/lint-skills.js +24 -4
  12. package/lib/playbook-runner.js +68 -14
  13. package/lib/prefetch.js +3 -2
  14. package/lib/refresh-external.js +0 -6
  15. package/lib/refresh-network.js +3 -4
  16. package/lib/scoring.js +8 -1
  17. package/lib/ttp-mapper.js +14 -3
  18. package/lib/upstream-check-cli.js +26 -1
  19. package/lib/validate-cve-catalog.js +9 -2
  20. package/lib/validate-playbooks.js +9 -11
  21. package/manifest.json +53 -53
  22. package/orchestrator/index.js +0 -1
  23. package/package.json +1 -1
  24. package/sbom.cdx.json +78 -78
  25. package/scripts/audit-perf.js +24 -13
  26. package/scripts/builders/theater-fingerprints.js +9 -4
  27. package/scripts/check-agents-md-collectors.js +15 -3
  28. package/scripts/check-codebase-patterns.js +16 -3
  29. package/scripts/check-manifest-snapshot.js +49 -8
  30. package/scripts/check-test-coverage.js +17 -1
  31. package/scripts/refresh-mitre-atlas.js +5 -1
  32. package/scripts/refresh-mitre-ics-attack.js +5 -1
  33. package/scripts/refresh-rfc-index.js +6 -1
  34. package/scripts/refresh-upstream-catalogs.js +25 -13
  35. package/scripts/release.js +0 -2
  36. package/scripts/run-e2e-scenarios.js +2 -2
  37. package/scripts/verify-shipped-tarball.js +0 -1
@@ -14,7 +14,6 @@ const { codeExcludeSet, isLinkedWorktreeDir, buildEvidenceLocations } = require(
14
14
 
15
15
  const COLLECTOR_ID = "library-author";
16
16
 
17
- const DEFAULT_MAX_DEPTH = 6;
18
17
  // Applied to the vendor-tree walk, so a cache nested under `vendor/` is not
19
18
  // mistaken for vendored provenance state.
20
19
  const DEFAULT_EXCLUDES = codeExcludeSet();
@@ -96,7 +95,104 @@ function stripYamlComments(content) {
96
95
  return content.replace(/#.*$/gm, "");
97
96
  }
98
97
 
99
- function scanPublishWorkflow(content, rel) {
98
+ // Splits comment-stripped workflow text into individual shell commands, so a
99
+ // flag on one install cannot vouch for another. A workflow that hash-pins its
100
+ // build requirements and then resolves its runtime ones must still report the
101
+ // unpinned command, which a whole-file flag test would suppress.
102
+ function shellCommands(code) {
103
+ const out = [];
104
+ const lines = code.split(/\r?\n/);
105
+ for (let i = 0; i < lines.length; i++) {
106
+ // A real `run:` line is never KBs long; skipping overlong ones keeps a
107
+ // crafted whitespace run from driving regex backtracking.
108
+ if (lines[i].length > 4096) continue;
109
+ // A backslash-continued command is ONE command: a `pip install \` whose
110
+ // `-r` sits on the next line is otherwise two fragments that each match
111
+ // nothing. The continuation lines are consumed, never re-emitted.
112
+ const startLine = i + 1;
113
+ let raw = lines[i];
114
+ while (/\\[ \t]*$/.test(raw) && i + 1 < lines.length &&
115
+ lines[i + 1].length <= 4096 && raw.length <= 8192) {
116
+ raw = `${raw.replace(/\\[ \t]*$/, " ")}${lines[i + 1].trim()}`;
117
+ i++;
118
+ }
119
+ for (const seg of raw.split(/&&|\|\||[;|]/)) {
120
+ const cmd = seg.trim();
121
+ // Comment stripping preserves line breaks, so the index still addresses
122
+ // the line the command STARTS on in the file an operator opens.
123
+ if (cmd) out.push({ cmd, line: startLine });
124
+ }
125
+ }
126
+ return out;
127
+ }
128
+
129
+ // Evidence snippets are read in a report, so a pathological one-line script
130
+ // cannot be pasted in whole.
131
+ function snippetOf(cmd, max = 200) {
132
+ return cmd.length > max ? `${cmd.slice(0, max)}…` : cmd;
133
+ }
134
+
135
+ // pip enters hash-checking mode as soon as ONE requirement carries a hash, so
136
+ // `pip install -r <file>` against a `pip-compile --generate-hashes` output is
137
+ // already reproducible. Resolves the named file one level only: a nested `-r`
138
+ // include is not followed, and an unreadable path counts as unhashed.
139
+ // pip accepts the requirement file attached (`-rreq.txt`) as well as separated
140
+ // (`-r req.txt`, `--requirement=req.txt`). One definition for both the detection
141
+ // predicate and the hash lookup: two regexes for one flag drift, and the pair
142
+ // that drifts silently is a command examined by neither.
143
+ // `(?:^|\s)-r` rather than `\b-r` so the `-r` inside `--requirement` is not read
144
+ // as the short form with `equirement...` as its filename.
145
+ const PIP_REQUIREMENT_RE = /(?:^|\s)(?:-r\s*=?\s*|--requirement[=\s]*)['"]?([^'"\s]+)/;
146
+
147
+ function pipRequirementTarget(cmd) {
148
+ const m = cmd.match(PIP_REQUIREMENT_RE);
149
+ return m ? m[1] : null;
150
+ }
151
+
152
+ // `working-directory:`, at job defaults or on a single step, is what a `run:`
153
+ // command resolves its relative paths against. Collected as a set of candidate
154
+ // bases rather than bound to the step that declared it: the scanner is
155
+ // line-based, so attributing one to a specific command would need the step
156
+ // nesting it does not track.
157
+ // Matches the block form and the flow form (`{ run: { working-directory: x } }`)
158
+ // alike. A key-shaped string inside a `run:` command matches too; that only adds
159
+ // a base that does not resolve, and an unresolvable base is skipped.
160
+ function workingDirectories(code) {
161
+ const dirs = [];
162
+ for (const m of code.matchAll(/working-directory:\s*['"]?([^'"\n,}]+?)['"]?\s*(?=$|[,}])/gm)) {
163
+ const d = m[1].trim();
164
+ if (d && d !== "." && !path.isAbsolute(d)) dirs.push(d);
165
+ }
166
+ return dirs;
167
+ }
168
+
169
+ // True only when every candidate that EXISTS is hash-pinned, and at least one
170
+ // does. Requiring all of them keeps a hashed copy in one working directory from
171
+ // vouching for an unhashed copy in another — over-suppression here reports an
172
+ // unpinned release as reproducible, which is the direction that costs something.
173
+ // A target that resolves nowhere stays unproven, so the caller still reports it.
174
+ function requirementsFileIsHashed(cmd, root, extraBases = []) {
175
+ if (!root) return false;
176
+ const target = pipRequirementTarget(cmd);
177
+ if (!target) return false;
178
+ if (path.isAbsolute(target)) return false;
179
+
180
+ let found = 0;
181
+ for (const base of [root, ...extraBases.map((d) => path.resolve(root, d))]) {
182
+ // Never read outside the scanned tree, whatever `../` the workflow names —
183
+ // for the base itself as well as the target resolved against it.
184
+ if (base !== root && !base.startsWith(root + path.sep)) continue;
185
+ const full = path.resolve(base, target);
186
+ if (full !== root && !full.startsWith(root + path.sep)) continue;
187
+ const body = readSafe(full);
188
+ if (body === null) continue;
189
+ found++;
190
+ if (!/--hash=/.test(body)) return false;
191
+ }
192
+ return found > 0;
193
+ }
194
+
195
+ function scanPublishWorkflow(content, rel, root) {
100
196
  // Whole-content probes read `code`; the `uses:` scan below stays on raw lines.
101
197
  const code = stripYamlComments(content);
102
198
  const hits = {
@@ -140,11 +236,51 @@ function scanPublishWorkflow(content, rel) {
140
236
  }
141
237
  }
142
238
 
239
+ // npm alone stays a whole-file test, because its suppressor is a DIFFERENT
240
+ // command rather than a flag: `npm ci` elsewhere in the workflow is the
241
+ // frozen install, and the `npm install` line beside it is usually tooling.
143
242
  if (/\bnpm\s+install\b/.test(code) && !/\bnpm\s+ci\b/.test(code)) {
144
243
  hits["release-workflow-non-frozen-install"].push({ file: rel, line: 0, snippet: "publish workflow uses `npm install` rather than `npm ci` — lockfile is not enforced" });
145
244
  }
146
- if (/\bcargo\s+(?:build|install)\b/.test(code) && !/--locked\b/.test(code) && !/--frozen\b/.test(code)) {
147
- hits["release-workflow-non-frozen-install"].push({ file: rel, line: 0, snippet: "cargo build/install without --locked / --frozen" });
245
+
246
+ // Bundler's frozen mode can be set once for the job rather than per command
247
+ // (`bundle config set frozen/deployment`, or the BUNDLE_ env equivalents), so
248
+ // that suppressor is genuinely file-scoped; the flags below are not.
249
+ // The setting has to be turned ON. `bundle config set frozen false` names the
250
+ // key and disables it, so matching the key alone reads an explicit opt-out as
251
+ // the protection it opts out of — the env form already required a value.
252
+ const bundlerFrozenConfig =
253
+ /\bbundle\s+config\b[^\n]*\b(?:frozen|deployment)\s+['"]?(?:true|1)\b/.test(code) ||
254
+ /\bBUNDLE_(?:FROZEN|DEPLOYMENT)\s*[:=]\s*['"]?(?:true|1)\b/i.test(code);
255
+
256
+ const runBases = workingDirectories(code);
257
+
258
+ // Every remaining ecosystem is probed one command at a time. The playbook
259
+ // indicator names npm / pnpm / pip -r / bundle; cargo is probed on the same
260
+ // --locked / --frozen predicate.
261
+ for (const { cmd, line } of shellCommands(code)) {
262
+ if (/\bcargo\s+(?:build|install)\b/.test(cmd) && !/--locked\b/.test(cmd) && !/--frozen\b/.test(cmd)) {
263
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `cargo build/install without --locked / --frozen: ${snippetOf(cmd)}` });
264
+ }
265
+ // Without --require-hashes the requirement set resolves at release time, so
266
+ // the artifact is not reproducible — unless the named file is hash-pinned,
267
+ // which switches pip into hash-checking mode on its own.
268
+ if (/\b(?:pip3?|python3?\s+-m\s+pip)\s+install\b/.test(cmd) &&
269
+ pipRequirementTarget(cmd) !== null &&
270
+ !/--require-hashes\b/.test(cmd) &&
271
+ !requirementsFileIsHashed(cmd, root, runBases)) {
272
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `pip install -r without --require-hashes — requirements are resolved at release time: ${snippetOf(cmd)}` });
273
+ }
274
+ // pnpm defaults to --frozen-lockfile when CI is set, which GitHub Actions
275
+ // always sets — so a bare `pnpm install` here is already frozen and only an
276
+ // explicit opt-out resolves at release time.
277
+ if (/\bpnpm\s+(?:install|i)\b/.test(cmd) &&
278
+ /--no-frozen-lockfile\b|--frozen-lockfile[=\s]+false\b/.test(cmd)) {
279
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `pnpm install opts out of the frozen lockfile: ${snippetOf(cmd)}` });
280
+ }
281
+ if (/\bbundle\s+install\b/.test(cmd) && !/--deployment\b/.test(cmd) && !/--frozen\b/.test(cmd) && !bundlerFrozenConfig) {
282
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `bundle install without --deployment / --frozen: ${snippetOf(cmd)}` });
283
+ }
148
284
  }
149
285
 
150
286
  if (/runs-on:\s*['"]?(?:self-hosted|\[?\s*self-hosted)/i.test(code)) {
@@ -193,7 +329,7 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
193
329
  // OIDC capability is evaluated per publish workflow, never repo-wide: a sibling
194
330
  // workflow's `id-token: write` gives nothing to a job using a static token.
195
331
  for (const w of publishWorkflows) {
196
- const h = scanPublishWorkflow(w.content, w.rel);
332
+ const h = scanPublishWorkflow(w.content, w.rel, root);
197
333
  for (const [id, list] of Object.entries(h)) workflowHits[id].push(...list);
198
334
  }
199
335
 
@@ -172,13 +172,24 @@ function captureLockfile(p, ecosystem, parser, label) {
172
172
  path: p,
173
173
  size_bytes: Buffer.byteLength(content, "utf8"),
174
174
  ...stats,
175
+ // A parser that could not decode the structure returns `error`; the read
176
+ // itself succeeded, so the two failures carry different kinds downstream.
177
+ ...(stats && stats.error ? { error_kind: "lockfile_parse_failed" } : {}),
175
178
  };
176
179
  } catch (e) {
177
- return { file: label, ecosystem, path: p, error: e.message };
180
+ return { file: label, ecosystem, path: p, error: e.message, error_kind: "lockfile_read_failed" };
178
181
  }
179
182
  }
180
183
 
181
- function findLockfiles(cwd) {
184
+ // `errors` is the collector_errors channel: a directory the walk could not read
185
+ // is a scan gap the operator has to see, not a silent zero. Required, not
186
+ // defaulted: a default array would restore the discard-into-the-void behaviour
187
+ // for any caller that forgets it, and neither this function nor
188
+ // findSbomDocuments is exported, so no test could reach that caller.
189
+ function findLockfiles(cwd, errors) {
190
+ if (!Array.isArray(errors)) {
191
+ throw new TypeError("findLockfiles: `errors` must be the collector_errors array — scan gaps have nowhere else to go");
192
+ }
182
193
  const found = [];
183
194
  for (const lf of LOCKFILES) {
184
195
  const p = path.join(cwd, lf.file);
@@ -196,7 +207,13 @@ function findLockfiles(cwd) {
196
207
  }
197
208
  }
198
209
  }
199
- } catch { /* swallow */ }
210
+ } catch (e) {
211
+ errors.push({
212
+ artifact_id: "lockfile-inventory",
213
+ kind: "readdir_failed",
214
+ reason: `cwd root: ${e.message} — requirements*.txt variants were not enumerated`,
215
+ });
216
+ }
200
217
 
201
218
  for (const sub of SUBDIR_PROBE_PATHS) {
202
219
  const subDir = path.join(cwd, sub);
@@ -204,7 +221,18 @@ function findLockfiles(cwd) {
204
221
  try {
205
222
  if (!fs.statSync(subDir).isDirectory()) continue;
206
223
  entries = fs.readdirSync(subDir, { withFileTypes: true });
207
- } catch { continue; }
224
+ } catch (e) {
225
+ // An absent probe dir is the normal case, not a gap. A permission or I/O
226
+ // failure on one that IS there means the subtree went unscanned.
227
+ if (e.code !== "ENOENT") {
228
+ errors.push({
229
+ artifact_id: "lockfile-inventory",
230
+ kind: "readdir_failed",
231
+ reason: `${sub}/: ${e.message} — subtree not scanned for lockfiles`,
232
+ });
233
+ }
234
+ continue;
235
+ }
208
236
  for (const e of entries) {
209
237
  if (e.isDirectory()) {
210
238
  // packages/* gets one more level for monorepo workspaces, no deeper.
@@ -242,7 +270,11 @@ function findLockfiles(cwd) {
242
270
  return found;
243
271
  }
244
272
 
245
- function findSbomDocuments(cwd) {
273
+ // `errors` is required for the same reason it is on findLockfiles above.
274
+ function findSbomDocuments(cwd, errors) {
275
+ if (!Array.isArray(errors)) {
276
+ throw new TypeError("findSbomDocuments: `errors` must be the collector_errors array — scan gaps have nowhere else to go");
277
+ }
246
278
  const found = [];
247
279
  for (const s of SBOM_FORMATS) {
248
280
  const p = path.join(cwd, s.file);
@@ -250,24 +282,41 @@ function findSbomDocuments(cwd) {
250
282
  // Open once and fstat the descriptor rather than existsSync→statSync→read:
251
283
  // ENOENT still skips, and there is no TOCTOU window.
252
284
  try { fd = fs.openSync(p, "r"); }
253
- catch (e) { if (e.code === "ENOENT") continue; found.push({ file: s.file, format: s.format, error: e.message }); continue; }
285
+ catch (e) { if (e.code === "ENOENT") continue; found.push({ file: s.file, format: s.format, error: e.message, error_kind: "open_failed" }); continue; }
254
286
  try {
255
287
  const stat = fs.fstatSync(fd);
256
288
  let content;
257
289
  // Read from the open descriptor, never by path again. readFileSync(fd) loops
258
290
  // to EOF; a single readSync can return short on a network or FUSE mount,
259
291
  // NUL-padding the buffer and truncating otherwise-valid JSON.
260
- try { content = fs.readFileSync(fd, "utf8"); } catch { content = null; }
292
+ try { content = fs.readFileSync(fd, "utf8"); } catch (e) {
293
+ content = null;
294
+ errors.push({
295
+ artifact_id: "sbom-document",
296
+ kind: "read_failed",
297
+ reason: `${s.file}: ${e.message} — component count not derived`,
298
+ });
299
+ }
261
300
  let component_count = null;
262
301
  if (content && s.format === "cyclonedx-1.x") {
263
302
  try {
264
303
  const j = JSON.parse(content);
265
304
  component_count = (j.components || []).length;
266
- } catch {}
305
+ } catch (e) {
306
+ // A present-but-unparseable SBOM reported `null components` with no
307
+ // explanation, which reads as "no components" rather than "not read".
308
+ errors.push({
309
+ artifact_id: "sbom-document",
310
+ kind: "parse_failed",
311
+ reason: `${s.file}: ${e.message} — component count unavailable`,
312
+ });
313
+ }
267
314
  }
268
315
  found.push({ file: s.file, format: s.format, size_bytes: stat.size, component_count });
269
316
  } catch (e) {
270
- found.push({ file: s.file, format: s.format, error: e.message });
317
+ // The descriptor opened; fstat is what failed here. Reporting it as an
318
+ // open failure sends the operator after the wrong thing.
319
+ found.push({ file: s.file, format: s.format, error: e.message, error_kind: "stat_failed" });
271
320
  } finally {
272
321
  if (fd !== undefined) { try { fs.closeSync(fd); } catch { /* non-fatal */ } }
273
322
  }
@@ -281,10 +330,31 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
281
330
  const root = path.resolve(cwd);
282
331
 
283
332
  // sbom-tool-available: a lockfile or an SBOM document is a sufficient proxy.
284
- const lockfiles = findLockfiles(root);
285
- const sbomDocuments = findSbomDocuments(root);
333
+ const lockfiles = findLockfiles(root, errors);
334
+ const sbomDocuments = findSbomDocuments(root, errors);
286
335
  const hasAnything = lockfiles.length > 0 || sbomDocuments.length > 0;
287
336
 
337
+ // A lockfile that was found but not read contributes no component count, so
338
+ // the inventory under-reports; the operator sees which one and why.
339
+ for (const l of lockfiles) {
340
+ if (!l.error) continue;
341
+ errors.push({
342
+ artifact_id: "lockfile-inventory",
343
+ kind: l.error_kind || "lockfile_read_failed",
344
+ reason: `${l.file}: ${l.error}`,
345
+ });
346
+ }
347
+ for (const s of sbomDocuments) {
348
+ if (!s.error) continue;
349
+ errors.push({
350
+ artifact_id: "sbom-document",
351
+ // Carried from the push site, mirroring the lockfile loop above: an
352
+ // fstat failure and an open failure are different operator actions.
353
+ kind: s.error_kind || "open_failed",
354
+ reason: `${s.file}: ${s.error}`,
355
+ });
356
+ }
357
+
288
358
  const artifacts = {
289
359
  "lockfile-inventory": {
290
360
  value: lockfiles.length
@@ -346,8 +416,13 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
346
416
  }
347
417
  npmLockfile.integrity_present_count = withIntegrity;
348
418
  npmLockfile.integrity_missing_count = withoutIntegrity;
349
- } catch {
419
+ } catch (e) {
350
420
  // A malformed lockfile leaves the indicator unflipped: inconclusive, not miss.
421
+ errors.push({
422
+ artifact_id: "lockfile-inventory",
423
+ kind: "lockfile_parse_failed",
424
+ reason: `${npmLockfile.file}: ${e.message} — lockfile-no-integrity left undecided (inconclusive, not miss)`,
425
+ });
351
426
  }
352
427
  }
353
428
 
@@ -378,6 +453,15 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
378
453
  lockfiles_found: lockfiles.length,
379
454
  sbom_documents_found: sbomDocuments.length,
380
455
  ecosystems_detected: [...new Set(lockfiles.map(l => l.ecosystem))],
456
+ // Magnitude behind the lockfile-no-integrity verdict: "hit" alone does not
457
+ // say whether one entry or nine hundred resolve without a hash. Present
458
+ // only when an npm lockfile was found AND parsed.
459
+ ...(npmLockfile && npmLockfile.integrity_present_count != null
460
+ ? {
461
+ npm_integrity_present_count: npmLockfile.integrity_present_count,
462
+ npm_integrity_missing_count: npmLockfile.integrity_missing_count,
463
+ }
464
+ : {}),
381
465
  },
382
466
  collector_errors: errors,
383
467
  };
@@ -85,8 +85,9 @@ function walkTree(root, opts = {}) {
85
85
  }
86
86
  walk(full, depth + 1);
87
87
  } else if (entry.isSymbolicLink()) {
88
- // A symlink Dirent is neither isDirectory() nor isFile(): never emitted.
89
- try { fs.realpathSync(full); } catch { /* dangling link — ignore */ }
88
+ // A symlink Dirent is neither isDirectory() nor isFile(), so it is
89
+ // never traversed and never emitted — including a dangling one.
90
+ continue;
90
91
  } else if (entry.isFile()) {
91
92
  // A regular file cannot introduce a directory cycle, so no realpath /
92
93
  // `seen` check. Windows path.relative returns backslashes, so rel is
@@ -7,9 +7,6 @@
7
7
  * it never mutates the catalog; each candidate is triage input.
8
8
  */
9
9
 
10
- const path = require('path');
11
- const fs = require('fs');
12
-
13
10
  const CVE_ID_RE = /^CVE-((?:19|20)\d{2})-\d{4,7}$/;
14
11
 
15
12
  /**
@@ -159,6 +159,11 @@ function lagScore(frameworkId, controlGaps, globalFrameworks) {
159
159
  * free text. Omitting `opts.lessons` (parsed zeroday-lessons.json) yields a
160
160
  * report with no new-control section rather than an error. Returns
161
161
  * { frameworks, universal_gaps, new_control_requirements, theater_risks, summary }.
162
+ *
163
+ * `cveCatalog` is positional surface this function does not read — the scenario
164
+ * matches against `controlGaps` alone. It holds the fourth slot so `opts` stays
165
+ * fifth, and mirrors theaterCheck(), which does read its catalog. Dropping the
166
+ * parameter would silently bind every caller's `opts` argument to it instead.
162
167
  */
163
168
  function gapReport(frameworkIds, threatScenario, controlGaps, cveCatalog = {}, opts = {}) {
164
169
  const scenario = threatScenario.toLowerCase();
@@ -100,8 +100,20 @@ function isStrictIsoCalendarDate(s) {
100
100
  const KEBAB_RE = /^[a-z0-9][a-z0-9-]*[a-z0-9]$/;
101
101
  const JSON_FILENAME_RE = /^[A-Za-z0-9._-]+\.json$/;
102
102
 
103
+ // `helpExitCode` is null unless the run is over before linting starts: 0 after
104
+ // --help, 2 after an unknown argument. parseArgs never terminates the process —
105
+ // it returns the code and the caller must honour it and stop.
106
+ //
107
+ // The invariant this keeps: every exit in this file goes through safeExit(), so
108
+ // no path here can write to stdout and then terminate synchronously. That is the
109
+ // `process-exit-after-stdout-write` shape, and it is a convention held file-wide
110
+ // rather than a repair of an observed truncation: Node writes pipe stdout
111
+ // synchronously on Windows and Linux (asynchronously on macOS), and the help
112
+ // block is a few hundred bytes, so process.exit() did not in fact truncate it
113
+ // here. Holding the invariant means a future help block that grows, or a run on
114
+ // a platform with async pipe writes, cannot reintroduce the class.
103
115
  function parseArgs(argv) {
104
- const opts = { skill: null, quiet: false, strict: false };
116
+ const opts = { skill: null, quiet: false, strict: false, helpExitCode: null };
105
117
  for (let i = 2; i < argv.length; i++) {
106
118
  const a = argv[i];
107
119
  if (a === '--skill') {
@@ -115,11 +127,13 @@ function parseArgs(argv) {
115
127
  opts.strict = true;
116
128
  } else if (a === '--help' || a === '-h') {
117
129
  printHelp();
118
- process.exit(0);
130
+ opts.helpExitCode = 0;
131
+ return opts;
119
132
  } else {
120
133
  console.error(`Unknown argument: ${a}`);
121
134
  printHelp();
122
- process.exit(2);
135
+ opts.helpExitCode = 2;
136
+ return opts;
123
137
  }
124
138
  }
125
139
  return opts;
@@ -798,6 +812,10 @@ function lintPlaybookAirGap() {
798
812
 
799
813
  function main() {
800
814
  const opts = parseArgs(process.argv);
815
+ if (opts.helpExitCode !== null) {
816
+ safeExit(opts.helpExitCode);
817
+ return;
818
+ }
801
819
  const manifest = readJson(MANIFEST_PATH);
802
820
 
803
821
  let skills = manifest.skills;
@@ -805,7 +823,8 @@ function main() {
805
823
  skills = skills.filter((s) => s.name === opts.skill);
806
824
  if (skills.length === 0) {
807
825
  console.error(`No skill named "${opts.skill}" in manifest.json`);
808
- process.exit(2);
826
+ safeExit(2);
827
+ return;
809
828
  }
810
829
  }
811
830
 
@@ -888,6 +907,7 @@ function main() {
888
907
 
889
908
  // The frontmatter parser is exported so `watchlist` does not grow a second one.
890
909
  module.exports = {
910
+ parseArgs,
891
911
  parseFrontmatter,
892
912
  extractFrontmatterBlock,
893
913
  unquote,
@@ -1292,6 +1292,60 @@ function close(playbookId, directiveId, analyzeResult, validateResult, agentSign
1292
1292
  analyze_complete: !!(analyzeResult && typeof analyzeResult === 'object'),
1293
1293
  validate_complete: !!(validateResult && typeof validateResult === 'object'),
1294
1294
  };
1295
+
1296
+ // Nothing downstream guards the analyze result's container shape: every
1297
+ // bundle format maps over `analyze.matched_cves` directly, and so does
1298
+ // analyzeFindingShape, so a non-object result or a non-array matched_cves
1299
+ // throws before phase 7 produces any output at all. Normalizing once, here,
1300
+ // is what makes the whole phase survive it — the run still emits an evidence
1301
+ // bundle and the shape failure is readable on runtime_errors. The container
1302
+ // and its elements are both normalized: every bundle builder dereferences an
1303
+ // element, so one null inside an otherwise well-formed array aborts the phase
1304
+ // exactly as a non-array container does.
1305
+ const shapeOf = (v) => (v === null ? 'null' : Array.isArray(v) ? 'array' : typeof v);
1306
+ const analyzeIsObject = !!analyzeResult && typeof analyzeResult === 'object' && !Array.isArray(analyzeResult);
1307
+ if (!analyzeIsObject || !Array.isArray(analyzeResult.matched_cves)) {
1308
+ pushRunError(runOpts._runErrors, {
1309
+ kind: 'analyze_shape',
1310
+ message: analyzeIsObject
1311
+ ? `analyze.matched_cves is ${shapeOf(analyzeResult.matched_cves)}, expected an array; treated as empty`
1312
+ : `analyze result is ${shapeOf(analyzeResult)}, expected an object; treated as empty`,
1313
+ }, { dedupeKey: x => x.message || '' });
1314
+ // A copy, so close() reports the caller's malformed result without
1315
+ // rewriting it underneath them.
1316
+ analyzeResult = { ...(analyzeIsObject ? analyzeResult : {}), matched_cves: [] };
1317
+ } else {
1318
+ // Dropped rather than repaired: a consumer cannot read a CVE id off a null,
1319
+ // and inventing a placeholder entry would put a finding in the evidence
1320
+ // bundle that the analyze phase never produced.
1321
+ const usable = (e) => !!e && typeof e === 'object' && !Array.isArray(e);
1322
+ const dropped = analyzeResult.matched_cves.filter((e) => !usable(e));
1323
+ if (dropped.length) {
1324
+ pushRunError(runOpts._runErrors, {
1325
+ kind: 'analyze_shape',
1326
+ message: `analyze.matched_cves holds ${dropped.length} malformed element(s) (${[...new Set(dropped.map(shapeOf))].join(', ')}); they are dropped and the remaining ${analyzeResult.matched_cves.length - dropped.length} are used`,
1327
+ }, { dedupeKey: x => x.message || '' });
1328
+ analyzeResult = { ...analyzeResult, matched_cves: analyzeResult.matched_cves.filter(usable) };
1329
+ }
1330
+ }
1331
+
1332
+ // The residual guard for the four sites that interpolate the finding shape —
1333
+ // the notification draft, the auditor-ready exception language, the
1334
+ // learning-loop lesson and the feeds_into finding context. The normalization
1335
+ // above removes the shapes analyzeFindingShape is known to throw on; this
1336
+ // catches anything it still throws on, degrading those four fields to
1337
+ // `<MISSING:…>` and putting the throw on runtime_errors rather than aborting
1338
+ // the phase.
1339
+ const safeFindingShape = () => {
1340
+ try { return analyzeFindingShape(analyzeResult); }
1341
+ catch (e) {
1342
+ pushRunError(runOpts._runErrors,
1343
+ { kind: 'analyze_shape', message: (e && e.message) ? String(e.message) : String(e) },
1344
+ { dedupeKey: x => x.message || '' });
1345
+ return {};
1346
+ }
1347
+ };
1348
+
1295
1349
  const enrichNotification = (na) => {
1296
1350
  const obligation = (g.jurisdiction_obligations || []).find(o =>
1297
1351
  o && typeof o === 'object' &&
@@ -1359,16 +1413,9 @@ function close(playbookId, directiveId, analyzeResult, validateResult, agentSign
1359
1413
  // Which template vars failed to resolve; empty when all rendered.
1360
1414
  ...(function () {
1361
1415
  const missing = [];
1362
- // Wrapped so a malformed analyze result cannot bring down close();
1363
- // the failure surfaces through runOpts._runErrors instead.
1364
- let findingShape;
1365
- try { findingShape = analyzeFindingShape(analyzeResult); }
1366
- catch (e) {
1367
- if (Array.isArray(runOpts._runErrors)) {
1368
- pushRunError(runOpts._runErrors, { kind: 'analyze_shape', message: (e && e.message) ? String(e.message) : String(e) }, { dedupeKey: e => e.message || '' });
1369
- }
1370
- findingShape = {};
1371
- }
1416
+ // safeFindingShape, not analyzeFindingShape: a malformed analyze result
1417
+ // must not bring down close(); the failure surfaces on runtime_errors.
1418
+ const findingShape = safeFindingShape();
1372
1419
  const draft = interpolate(
1373
1420
  na.draft_notification,
1374
1421
  { ...agentSignals, ...findingShape },
@@ -1416,7 +1463,11 @@ function close(playbookId, directiveId, analyzeResult, validateResult, agentSign
1416
1463
  // does not supply, so the auditor-ready language routinely renders
1417
1464
  // `<MISSING:…>`. missing_interpolation_vars names which.
1418
1465
  const exMissing = [];
1419
- const findingShape = (() => { try { return analyzeFindingShape(analyzeResult); } catch { return {}; } })();
1466
+ // safeFindingShape, like the other three shape consumers. The feeds_into
1467
+ // context below reports the same throw on every close(), so this site's
1468
+ // own disposition is not separately observable — it is uniform so a
1469
+ // future reader does not read the bare catch as a deliberate exemption.
1470
+ const findingShape = safeFindingShape();
1420
1471
  exception = {
1421
1472
  scope: interpolate(t.scope, { ...agentSignals, ...findingShape }, exMissing),
1422
1473
  duration: t.duration,
@@ -1492,7 +1543,7 @@ function close(playbookId, directiveId, analyzeResult, validateResult, agentSign
1492
1543
 
1493
1544
  const lesson = c.learning_loop?.enabled ? {
1494
1545
  enabled: true,
1495
- attack_vector: interpolate(c.learning_loop.lesson_template.attack_vector, analyzeFindingShape(analyzeResult)),
1546
+ attack_vector: interpolate(c.learning_loop.lesson_template.attack_vector, safeFindingShape()),
1496
1547
  control_gap: c.learning_loop.lesson_template.control_gap,
1497
1548
  framework_gap: c.learning_loop.lesson_template.framework_gap,
1498
1549
  new_control_requirement: c.learning_loop.lesson_template.new_control_requirement,
@@ -1542,7 +1593,7 @@ function close(playbookId, directiveId, analyzeResult, validateResult, agentSign
1542
1593
  analyze: analyzeResult,
1543
1594
  validate: validateResult,
1544
1595
  // Same two-sourced finding.* shape and guards as the escalation context.
1545
- finding: { ...(agentSignals.finding && typeof agentSignals.finding === 'object' && !Array.isArray(agentSignals.finding) ? agentSignals.finding : {}), ...analyzeFindingShape(analyzeResult) },
1596
+ finding: { ...(agentSignals.finding && typeof agentSignals.finding === 'object' && !Array.isArray(agentSignals.finding) ? agentSignals.finding : {}), ...safeFindingShape() },
1546
1597
  // Without the accumulator, a feeds_into condition failure never reaches
1547
1598
  // analyze.runtime_errors[].
1548
1599
  ...(runOpts._runErrors ? { _runErrors: runOpts._runErrors } : {})
@@ -2356,7 +2407,10 @@ function buildEvidenceBundle(format, playbook, analyze, validate, agentSignals,
2356
2407
  '@id': productPurl,
2357
2408
  subcomponents: [{ '@id': productPurl }],
2358
2409
  };
2359
- const remediationId = validate.selected_remediation?.id || (validate.remediation_paths?.[0]?.id) || null;
2410
+ // validate() selects the priority-1 path as its last rung, so a null
2411
+ // selected_remediation means the phase declares no remediation paths at all
2412
+ // and remediation_options_considered is empty too — there is no id to name.
2413
+ const remediationId = validate.selected_remediation?.id || null;
2360
2414
  const remediationDescription = validate.selected_remediation?.description || null;
2361
2415
  const actionStatementFor = (fallback) => {
2362
2416
  if (remediationId && remediationDescription) {
package/lib/prefetch.js CHANGED
@@ -487,8 +487,9 @@ function isFresh(idx, source, id, maxAgeMs) {
487
487
 
488
488
  function authHeadersForSource(source) {
489
489
  if (source === "nvd" && process.env.NVD_API_KEY) return { apiKey: process.env.NVD_API_KEY };
490
- // `pins` is the registered source name; `github` is accepted too, so
491
- // GITHUB_TOKEN reaches the header whichever spelling an operator uses.
490
+ // `pins` is the registered name for the MITRE GitHub releases feed. The
491
+ // `github` alias is defensive: no registered source emits it, and prefetch()
492
+ // refuses `--source github` before any header is built.
492
493
  if ((source === "pins" || source === "github") && process.env.GITHUB_TOKEN) {
493
494
  return { Authorization: `Bearer ${process.env.GITHUB_TOKEN}` };
494
495
  }
@@ -1701,12 +1701,6 @@ async function main() {
1701
1701
  process.exitCode = cacheIntegrityFailure ? 4 : (hadFailure ? 1 : 0);
1702
1702
  }
1703
1703
 
1704
- async function sequential(items, fn) {
1705
- const out = [];
1706
- for (const it of items) out.push(await fn(it));
1707
- return out;
1708
- }
1709
-
1710
1704
  if (require.main === module) {
1711
1705
  main().catch((err) => {
1712
1706
  // A hinted error prints its message plus a JSON line, never a stack trace.
@@ -116,10 +116,9 @@ function getBufferOnce(url, timeoutMs) {
116
116
  if (!isAllowedTarballHost(url)) {
117
117
  return reject(new Error(`refusing to fetch tarball from non-allowlisted host: ${u.hostname}`));
118
118
  }
119
- const cap = (() => {
120
- const env = parseInt(process.env.EXCEPTD_TARBALL_SIZE_CAP_BYTES, 10);
121
- return Number.isFinite(env) && env > 0 ? env : 200 * 1024 * 1024;
122
- })();
119
+ // Same reader as the buffered-length check, so the streaming cap and the
120
+ // post-download cap cannot drift apart when the default changes.
121
+ const cap = tarballSizeCap();
123
122
  const req = https.get({
124
123
  host: u.host, path: u.pathname + u.search,
125
124
  headers: { "User-Agent": "exceptd/refresh-network" },
package/lib/scoring.js CHANGED
@@ -431,10 +431,17 @@ function validate(catalog) {
431
431
  const errors = [];
432
432
  for (const [cveId, entry] of Object.entries(catalog)) {
433
433
  if (cveId.startsWith('_')) continue;
434
+ // A null or non-object entry cannot be walked field by field: `field in
435
+ // entry` throws, taking down validation of the whole catalog. Report it as
436
+ // the error it is, so the remaining entries are still checked.
437
+ if (!entry || typeof entry !== 'object') {
438
+ errors.push(`${cveId}: entry is ${entry === null ? 'null' : typeof entry}, expected an object`);
439
+ continue;
440
+ }
434
441
  // Drafts carry a conservative-default rwep_score beside null-until-curated
435
442
  // factor fields, so the divergence check below fires on every one. They are
436
443
  // reviewed through `_auto_imported_meta.curation_needed` instead.
437
- if (entry && entry._auto_imported === true) continue;
444
+ if (entry._auto_imported === true) continue;
438
445
  for (const field of CVE_SCHEMA_REQUIRED) {
439
446
  if (!(field in entry)) {
440
447
  errors.push(`${cveId}: missing required field '${field}'`);