@blamejs/exceptd-skills 0.19.34 → 0.19.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/bin/exceptd.js +0 -5
  3. package/data/_indexes/_meta.json +4 -4
  4. package/data/cve-catalog.json +24 -24
  5. package/data/zeroday-lessons.json +1249 -1249
  6. package/lib/collectors/ai-api.js +150 -22
  7. package/lib/collectors/cicd-pipeline-compromise.js +73 -28
  8. package/lib/collectors/library-author.js +141 -5
  9. package/lib/collectors/sbom.js +96 -12
  10. package/lib/collectors/scan-excludes.js +3 -2
  11. package/lib/cve-regression-watcher.js +0 -3
  12. package/lib/framework-gap.js +5 -0
  13. package/lib/lint-skills.js +24 -4
  14. package/lib/playbook-runner.js +68 -14
  15. package/lib/prefetch.js +3 -2
  16. package/lib/refresh-external.js +0 -6
  17. package/lib/refresh-network.js +3 -4
  18. package/lib/scoring.js +8 -1
  19. package/lib/ttp-mapper.js +14 -3
  20. package/lib/upstream-check-cli.js +26 -1
  21. package/lib/validate-cve-catalog.js +9 -2
  22. package/lib/validate-playbooks.js +9 -11
  23. package/manifest.json +53 -53
  24. package/orchestrator/index.js +0 -1
  25. package/package.json +1 -1
  26. package/sbom.cdx.json +82 -82
  27. package/scripts/audit-perf.js +24 -13
  28. package/scripts/builders/theater-fingerprints.js +9 -4
  29. package/scripts/check-agents-md-collectors.js +15 -3
  30. package/scripts/check-codebase-patterns.js +16 -3
  31. package/scripts/check-manifest-snapshot.js +49 -8
  32. package/scripts/check-test-coverage.js +17 -1
  33. package/scripts/refresh-mitre-atlas.js +5 -1
  34. package/scripts/refresh-mitre-ics-attack.js +5 -1
  35. package/scripts/refresh-rfc-index.js +6 -1
  36. package/scripts/refresh-upstream-catalogs.js +25 -13
  37. package/scripts/release.js +0 -2
  38. package/scripts/run-e2e-scenarios.js +2 -2
  39. package/scripts/verify-shipped-tarball.js +0 -1
@@ -13,16 +13,40 @@ const os = require("node:os");
13
13
 
14
14
  const COLLECTOR_ID = "ai-api";
15
15
 
16
- function readSafe(full, max = 256 * 1024) {
16
+ // One definition, named at every call site. Written as a literal in both the
17
+ // default and the callers, the two drift apart the moment the cap is raised,
18
+ // and the only symptom is a credential store that quietly stops being scanned
19
+ // at the old size.
20
+ const SCAN_CAP_BYTES = 256 * 1024;
21
+
22
+ // `skip` routes a non-read onto the collector_errors channel:
23
+ // { errors, artifact_id, label }. A null return is indistinguishable between
24
+ // "over the cap" and "unreadable", and both mean the file went unscanned — a
25
+ // clean verdict over an unscanned credential store is the failure mode, so the
26
+ // reason travels with the submission. `label` is home-relative: the absolute
27
+ // path is operator-identifying and collector_meta already carries `home`.
28
+ function readSafe(full, max = SCAN_CAP_BYTES, skip = null) {
29
+ const note = (kind, reason) => {
30
+ if (!skip || !Array.isArray(skip.errors)) return;
31
+ const entry = { kind, reason: `${skip.label || path.basename(full)}: ${reason}` };
32
+ if (skip.artifact_id) entry.artifact_id = skip.artifact_id;
33
+ skip.errors.push(entry);
34
+ };
17
35
  let fd;
18
36
  try {
19
37
  fd = fs.openSync(full, "r");
20
38
  const s = fs.fstatSync(fd);
21
- if (s.size > max) return null;
39
+ if (s.size > max) {
40
+ note("file_too_large_skipped", `${s.size} bytes exceeds ${max}-byte scan limit; not scanned`);
41
+ return null;
42
+ }
22
43
  // readFileSync(fd) loops to EOF; a single readSync can return short on a
23
44
  // network or FUSE fd. Reading the open fd keeps fstat-then-read TOCTOU-free.
24
45
  return fs.readFileSync(fd, "utf8");
25
- } catch { return null; }
46
+ } catch (e) {
47
+ note("read_failed", e.message);
48
+ return null;
49
+ }
26
50
  finally { if (fd !== undefined) { try { fs.closeSync(fd); } catch { /* non-fatal */ } } }
27
51
  }
28
52
 
@@ -30,6 +54,27 @@ function fileExists(full) {
30
54
  try { return fs.statSync(full).isFile(); } catch { return false; }
31
55
  }
32
56
 
57
+ // A credential store is one of three things, and only "absent" is a negative
58
+ // finding. Truthiness collapses the other two: an empty file reads as "" and is
59
+ // a completed scan, not a skipped one, while readSafe signals a skip with null.
60
+ // The distinction is null-vs-string, and it drives the artifact's captured flag
61
+ // and the indicator's verdict together so the two cannot disagree.
62
+ const ABSENT = "absent";
63
+ const UNREAD = "unread";
64
+ const READ = "read";
65
+ function storeState(exists, content) {
66
+ if (!exists) return ABSENT;
67
+ return content === null ? UNREAD : READ;
68
+ }
69
+
70
+ // A store that was never read cannot answer the question its indicator asks.
71
+ // `miss` there is a clean bill of health over an unscanned file; a hit stands
72
+ // on its own evidence and is unaffected by a sibling going unread.
73
+ function verdict(found, undetermined) {
74
+ if (found) return "hit";
75
+ return undetermined ? "inconclusive" : "miss";
76
+ }
77
+
33
78
  // Cleartext key exports: `export VAR=value`, `VAR=value`, fish `set -gx VAR value`.
34
79
  const AI_KEY_PATTERNS = [
35
80
  { id: "openai", re: /(?:^|\n)\s*(?:export\s+|set\s+-gx\s+)?OPENAI_API_KEY\s*[= ]\s*['"]?sk-[A-Za-z0-9_-]{20,}/m },
@@ -129,7 +174,12 @@ function parseGcloudAdc(content) {
129
174
  privateKey: typeof j?.private_key === "string" ? j.private_key : "",
130
175
  clientEmail: typeof j?.client_email === "string" ? j.client_email : "",
131
176
  };
132
- } catch { return { hasServiceAccount: false }; }
177
+ } catch (e) {
178
+ // Unparseable ADC is NOT evidence the file holds no service account; the
179
+ // parse failure travels so `gcp-service-account-json: miss` is not read as
180
+ // a clean bill of health.
181
+ return { hasServiceAccount: false, parse_error: e.message };
182
+ }
133
183
  }
134
184
 
135
185
  function parseKubeStaticToken(content) {
@@ -170,7 +220,17 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
170
220
  for (const e of fs.readdirSync(fishConfD)) {
171
221
  if (e.endsWith(".fish")) shellRcs.push(path.join(fishConfD, e));
172
222
  }
173
- } catch { /* fish not present */ }
223
+ } catch (e) {
224
+ // No fish conf.d is the normal case. A permission or I/O failure on one that
225
+ // IS there means those fragments went unscanned for key exports.
226
+ if (e.code !== "ENOENT") {
227
+ errors.push({
228
+ artifact_id: "shell-rc-files",
229
+ kind: "readdir_failed",
230
+ reason: `.config/fish/conf.d: ${e.message} — fragments not scanned`,
231
+ });
232
+ }
233
+ }
174
234
 
175
235
  const dotfileKeys = [
176
236
  ".openai", ".anthropic",
@@ -180,13 +240,24 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
180
240
  path.join(".config", "azure-openai"),
181
241
  ].map(rel => path.join(home, rel));
182
242
 
183
- const allKeyCarriers = [...shellRcs, ...dotfileKeys];
243
+ // Two artifacts share one scan loop. The carrier's own artifact_id travels
244
+ // with it so a skipped `~/.anthropic` is reported against dotfile-api-keys
245
+ // rather than pointing the operator at the shell-rc row.
246
+ const allKeyCarriers = [
247
+ ...shellRcs.map(p => ({ path: p, artifact_id: "shell-rc-files" })),
248
+ ...dotfileKeys.map(p => ({ path: p, artifact_id: "dotfile-api-keys" })),
249
+ ];
184
250
  const cleartextHitsByFile = {};
251
+ const cleartextUnread = [];
185
252
  let cleartextFp = null;
186
- for (const p of allKeyCarriers) {
253
+ for (const { path: p, artifact_id } of allKeyCarriers) {
187
254
  if (!fileExists(p)) continue;
188
- const c = readSafe(p);
189
- if (c == null) continue;
255
+ const c = readSafe(p, SCAN_CAP_BYTES, {
256
+ errors, artifact_id, label: path.relative(home, p),
257
+ });
258
+ // Present but unread: the file could still hold a key, so it is a gap in
259
+ // the scan rather than a carrier with nothing in it.
260
+ if (c === null) { cleartextUnread.push(path.relative(home, p)); continue; }
190
261
  const hits = scanShellRc(c);
191
262
  if (hits.length > 0) {
192
263
  cleartextHitsByFile[path.relative(home, p)] = hits;
@@ -200,24 +271,64 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
200
271
  const cleartextAnyHit = Object.keys(cleartextHitsByFile).length > 0;
201
272
 
202
273
  const awsCredsPath = path.join(home, ".aws", "credentials");
203
- const awsCredsContent = fileExists(awsCredsPath) ? readSafe(awsCredsPath) : null;
274
+ const awsCredsExists = fileExists(awsCredsPath);
275
+ const awsCredsContent = awsCredsExists
276
+ ? readSafe(awsCredsPath, SCAN_CAP_BYTES, {
277
+ errors, artifact_id: "aws-credentials", label: path.relative(home, awsCredsPath),
278
+ })
279
+ : null;
204
280
  const awsParsed = parseAwsCredentials(awsCredsContent);
205
281
  const longLivedAws = awsParsed.staticProfiles.length > 0;
206
282
 
207
283
  const gcloudAdcPath = path.join(home, ".config", "gcloud", "application_default_credentials.json");
208
- const gcloudContent = fileExists(gcloudAdcPath) ? readSafe(gcloudAdcPath) : null;
284
+ const gcloudAdcExists = fileExists(gcloudAdcPath);
285
+ const gcloudContent = gcloudAdcExists
286
+ ? readSafe(gcloudAdcPath, SCAN_CAP_BYTES, {
287
+ errors, artifact_id: "gcp-credentials", label: path.relative(home, gcloudAdcPath),
288
+ })
289
+ : null;
209
290
  const gcloudParsed = parseGcloudAdc(gcloudContent);
291
+ if (gcloudParsed.parse_error) {
292
+ errors.push({
293
+ artifact_id: "gcp-credentials",
294
+ kind: "parse_failed",
295
+ reason: `${path.relative(home, gcloudAdcPath)}: ${gcloudParsed.parse_error} — service-account presence undetermined, not absent`,
296
+ });
297
+ }
210
298
 
211
299
  const kubeCfgPath = (env && env.KUBECONFIG) || path.join(home, ".kube", "config");
212
- const kubeContent = fileExists(kubeCfgPath) ? readSafe(kubeCfgPath) : null;
300
+ // KUBECONFIG can point outside $HOME, where a home-relative label degrades to
301
+ // a `..` chain; the basename is enough to name the file that went unread.
302
+ // The separator is required: a bare prefix test also matches a SIBLING whose
303
+ // name merely starts with $HOME ("/home/rob" vs "/home/robert-backup"), which
304
+ // produces exactly the `../…` chain this branch exists to avoid — and leaks
305
+ // another account's directory name into the warning.
306
+ const kubeInHome = kubeCfgPath === home || kubeCfgPath.startsWith(home + path.sep);
307
+ const kubeLabel = kubeInHome ? path.relative(home, kubeCfgPath) : path.basename(kubeCfgPath);
308
+ const kubeCfgExists = fileExists(kubeCfgPath);
309
+ const kubeContent = kubeCfgExists
310
+ ? readSafe(kubeCfgPath, SCAN_CAP_BYTES, {
311
+ errors, artifact_id: "kube-config", label: kubeLabel,
312
+ })
313
+ : null;
213
314
  const kubeParsed = parseKubeStaticToken(kubeContent);
214
315
  const kubeStaticToken = kubeParsed.found;
215
316
 
317
+ const awsState = storeState(awsCredsExists, awsCredsContent);
318
+ const gcloudState = storeState(gcloudAdcExists, gcloudContent);
319
+ const kubeState = storeState(kubeCfgExists, kubeContent);
320
+
216
321
  const signal_overrides = {
217
- "cleartext-api-key-in-dotfile": cleartextAnyHit ? "hit" : "miss",
218
- "long-lived-aws-keys": longLivedAws ? "hit" : "miss",
219
- "gcp-service-account-json": gcloudParsed.hasServiceAccount ? "hit" : "miss",
220
- "kubeconfig-with-static-token": kubeStaticToken ? "hit" : "miss",
322
+ "cleartext-api-key-in-dotfile": verdict(cleartextAnyHit, cleartextUnread.length > 0),
323
+ "long-lived-aws-keys": verdict(longLivedAws, awsState === UNREAD),
324
+ // Unread and unparseable are the same answer here: nothing in the file was
325
+ // validated, so its service-account presence is undetermined rather than
326
+ // absent. JSON.parse never runs on an unread store, so the state carries it.
327
+ "gcp-service-account-json": verdict(
328
+ gcloudParsed.hasServiceAccount,
329
+ gcloudState === UNREAD || Boolean(gcloudParsed.parse_error),
330
+ ),
331
+ "kubeconfig-with-static-token": verdict(kubeStaticToken, kubeState === UNREAD),
221
332
  };
222
333
 
223
334
  // Per-indicator __fp_checks attestation. Path checks hold because every store
@@ -271,15 +382,32 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
271
382
  value: dotfileKeys.filter(p => fileExists(p)).map(p => path.relative(home, p)).join(", ") || "no AI vendor dotfile carriers found at the canonical paths",
272
383
  captured: true,
273
384
  },
274
- "aws-credentials": awsCredsContent
385
+ // A null content is two different worlds: the store is not there, or it IS
386
+ // there and went unread (over the scan cap, permissions, I/O). Only the
387
+ // first is an absence. The second is captured:false with the reason on
388
+ // collector_errors — asserting "absent" over a credential file that exists
389
+ // is the same clean-verdict-over-an-unscanned-store failure the error
390
+ // channel was added to close.
391
+ "aws-credentials": awsState === READ
275
392
  ? { value: `${awsParsed.staticProfiles.length} long-lived profile(s): ${awsParsed.staticProfiles.join(", ") || "none"}`, captured: true }
276
- : { value: "~/.aws/credentials absent", captured: true },
277
- "gcp-credentials": gcloudContent
393
+ : awsState === UNREAD
394
+ ? { value: "~/.aws/credentials present but unread — profile inventory undetermined, not absent", captured: false, reason: "read skipped or failed; see collector_errors for the reason" }
395
+ : { value: "~/.aws/credentials absent", captured: true },
396
+ // Read-but-unparseable is a fourth outcome, distinct from read, unread and
397
+ // absent: the bytes arrived and nothing in them was validated. Reporting
398
+ // captured:true there states service_account=false about a file never parsed.
399
+ "gcp-credentials": gcloudState === READ && !gcloudParsed.parse_error
278
400
  ? { value: `application_default_credentials.json present; service_account=${gcloudParsed.hasServiceAccount}`, captured: true }
279
- : { value: "no gcloud ADC at the canonical path", captured: true, reason: "credentials.db / legacy_credentials/*/adc.json inspection deferred (no stdlib SQLite reader)" },
280
- "kube-config": kubeContent
401
+ : gcloudState === READ
402
+ ? { value: "application_default_credentials.json present but unparseable — service-account presence undetermined, not absent", captured: false, reason: "JSON parse failed; see collector_errors for the reason" }
403
+ : gcloudState === UNREAD
404
+ ? { value: "application_default_credentials.json present but unread — service-account presence undetermined, not absent", captured: false, reason: "read skipped or failed; see collector_errors for the reason" }
405
+ : { value: "no gcloud ADC at the canonical path", captured: true, reason: "credentials.db / legacy_credentials/*/adc.json inspection deferred (no stdlib SQLite reader)" },
406
+ "kube-config": kubeState === READ
281
407
  ? { value: `kubeconfig present; static_token=${kubeStaticToken}`, captured: true }
282
- : { value: "no kubeconfig at the canonical path", captured: true },
408
+ : kubeState === UNREAD
409
+ ? { value: "kubeconfig present but unread — static-token presence undetermined, not absent", captured: false, reason: "read skipped or failed; see collector_errors for the reason" }
410
+ : { value: "no kubeconfig at the canonical path", captured: true },
283
411
  "ai-sdk-inventory": {
284
412
  value: "skipped — npm/pip global listing deferred to operator/AI evidence",
285
413
  captured: false,
@@ -17,40 +17,57 @@ const OIDC_WALK_EXCLUDES = codeExcludeSet();
17
17
 
18
18
  const COLLECTOR_ID = "cicd-pipeline-compromise";
19
19
 
20
+ // Returns { text } when the file was read, otherwise { skipped, reason }: a file
21
+ // that is never scanned reads as a "miss" on every indicator it would have
22
+ // flipped, so the caller has to record the skip rather than drop it.
20
23
  function readSafe(p, max = 512 * 1024) {
21
24
  let fd;
22
25
  try {
23
26
  fd = fs.openSync(p, "r");
24
27
  const s = fs.fstatSync(fd);
25
- if (s.size > max) return null;
28
+ if (s.size > max) return { skipped: "file_too_large", reason: `${s.size} bytes exceeds the ${max}-byte scan limit; not scanned` };
26
29
  // readFileSync(fd) loops read() to EOF; a single readSync can return short on
27
30
  // a network or FUSE fd. Reading the open fd keeps fstat-then-read TOCTOU-free.
28
- return fs.readFileSync(fd, "utf8");
29
- } catch { return null; }
31
+ return { text: fs.readFileSync(fd, "utf8") };
32
+ } catch (e) { return { skipped: "read_error", reason: (e && e.message) || String(e) }; }
30
33
  finally { if (fd !== undefined) { try { fs.closeSync(fd); } catch { /* non-fatal */ } } }
31
34
  }
32
35
 
33
- function walkWorkflows(root) {
36
+ function skipEntry(artifact_id, rel, r) {
37
+ return {
38
+ artifact_id,
39
+ kind: r.skipped === "file_too_large" ? "file_too_large_skipped" : "read_failed",
40
+ reason: `${rel}: ${r.reason}`,
41
+ };
42
+ }
43
+
44
+ function walkWorkflows(root, errors = []) {
34
45
  const out = [];
35
46
  const wfDir = path.join(root, ".github", "workflows");
36
47
  if (fs.existsSync(wfDir)) {
37
48
  let entries;
38
49
  try { entries = fs.readdirSync(wfDir, { withFileTypes: true }); }
39
- catch { entries = []; }
50
+ catch (e) {
51
+ entries = [];
52
+ errors.push({ artifact_id: "workflow-yaml-inventory", kind: "readdir_failed", reason: `.github/workflows: ${e.message}` });
53
+ }
40
54
  for (const e of entries) {
41
55
  if (!e.isFile()) continue;
42
56
  if (!/\.(ya?ml)$/i.test(e.name)) continue;
43
57
  const full = path.join(wfDir, e.name);
44
- const content = readSafe(full);
45
- if (content != null) out.push({ full, rel: path.relative(root, full).replace(/\\/g, "/"), content });
58
+ const rel = path.relative(root, full).replace(/\\/g, "/");
59
+ const r = readSafe(full);
60
+ if (r.text != null) out.push({ full, rel, content: r.text });
61
+ else errors.push(skipEntry("workflow-yaml-inventory", rel, r));
46
62
  }
47
63
  }
48
64
  // Also recognise the most common single-file CI YAMLs at repo root.
49
65
  for (const top of [".gitlab-ci.yml", ".circleci/config.yml"]) {
50
66
  const full = path.join(root, top);
51
67
  if (fs.existsSync(full)) {
52
- const content = readSafe(full);
53
- if (content != null) out.push({ full, rel: top, content });
68
+ const r = readSafe(full);
69
+ if (r.text != null) out.push({ full, rel: top, content: r.text });
70
+ else errors.push(skipEntry("workflow-yaml-inventory", top, r));
54
71
  }
55
72
  }
56
73
  return out;
@@ -184,7 +201,7 @@ function scanWorkflow(content, rel) {
184
201
  return hits;
185
202
  }
186
203
 
187
- function scanOidcPolicies(root) {
204
+ function scanOidcPolicies(root, errors = []) {
188
205
  // Walks the infra dirs to depth 4 for *.json naming
189
206
  // token.actions.githubusercontent.com with a wildcarded sub-claim.
190
207
  const rootDirs = ["infra", "terraform", "policies", ".aws", ".github"].map(d => path.join(root, d));
@@ -193,7 +210,14 @@ function scanOidcPolicies(root) {
193
210
  if (depth > 4 || finds.length > 5) return;
194
211
  let entries;
195
212
  try { entries = fs.readdirSync(dir, { withFileTypes: true }); }
196
- catch { return; }
213
+ catch (e) {
214
+ errors.push({
215
+ artifact_id: "oidc-trust-policy-inventory",
216
+ kind: "readdir_failed",
217
+ reason: `${path.relative(root, dir).replace(/\\/g, "/")}: ${e.message}`,
218
+ });
219
+ return;
220
+ }
197
221
  for (const e of entries) {
198
222
  if (OIDC_WALK_EXCLUDES.has(e.name)) continue;
199
223
  const full = path.join(dir, e.name);
@@ -204,15 +228,17 @@ function scanOidcPolicies(root) {
204
228
  continue;
205
229
  }
206
230
  if (!e.isFile() || !/\.json$/i.test(e.name)) continue;
207
- const text = readSafe(full);
208
- if (!text) continue;
231
+ const rel = path.relative(root, full).replace(/\\/g, "/");
232
+ const r = readSafe(full);
233
+ if (r.text == null) { errors.push(skipEntry("oidc-trust-policy-inventory", rel, r)); continue; }
234
+ const text = r.text;
209
235
  // Each pattern is bound to the leading `"` of the JSON key, so a lookalike
210
236
  // issuer like `"eviltoken.actions…"` cannot match.
211
237
  const subWildcard =
212
238
  /"token\.actions\.githubusercontent\.com:sub"\s*:\s*"\*"/.test(text) ||
213
239
  /"token\.actions\.githubusercontent\.com:sub"\s*:\s*"repo:\*[^"]*"/.test(text) ||
214
240
  /"token\.actions\.githubusercontent\.com:sub"\s*:\s*"repo:[^"]*\/\*:[^"]*"/.test(text);
215
- if (subWildcard) finds.push({ file: path.relative(root, full).replace(/\\/g, "/"), snippet: "OIDC sub-claim wildcarded across repos or branches" });
241
+ if (subWildcard) finds.push({ file: rel, snippet: "OIDC sub-claim wildcarded across repos or branches" });
216
242
  }
217
243
  }
218
244
  for (const rd of rootDirs) if (fs.existsSync(rd)) walk(rd, 0);
@@ -245,7 +271,7 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
245
271
  };
246
272
  }
247
273
 
248
- const workflows = walkWorkflows(root);
274
+ const workflows = walkWorkflows(root, errors);
249
275
  const aggregateHits = {
250
276
  "workflow-injection-sink": [],
251
277
  "pull-request-target-with-pr-checkout": [],
@@ -257,14 +283,25 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
257
283
  for (const [k, v] of Object.entries(h)) aggregateHits[k].push(...v);
258
284
  }
259
285
 
260
- const oidcWildcards = scanOidcPolicies(root);
286
+ const oidcWildcards = scanOidcPolicies(root, errors);
287
+
288
+ // A workflow this collector could not read may hold the very thing each
289
+ // indicator looks for, so `miss` over an incomplete inventory is a clean
290
+ // verdict on evidence that was never examined. collector_errors are advisory
291
+ // and change no verdict on their own; the gap has to reach the signal. A hit
292
+ // stands on what WAS read and is unaffected.
293
+ const workflowScanIncomplete = errors.some(
294
+ (e) => e.artifact_id === "workflow-yaml-inventory");
295
+ const oidcScanIncomplete = errors.some(
296
+ (e) => e.artifact_id === "oidc-trust-policy-inventory");
297
+ const verdict = (found, incomplete) => (found ? "hit" : incomplete ? "inconclusive" : "miss");
261
298
 
262
299
  const signal_overrides = {
263
- "workflow-injection-sink": aggregateHits["workflow-injection-sink"].length > 0 ? "hit" : "miss",
264
- "pull-request-target-with-pr-checkout": aggregateHits["pull-request-target-with-pr-checkout"].length > 0 ? "hit" : "miss",
265
- "actions-floating-tag-pin": aggregateHits["actions-floating-tag-pin"].length > 0 ? "hit" : "miss",
266
- "secret-exposed-to-fork-pr": aggregateHits["secret-exposed-to-fork-pr"].length > 0 ? "hit" : "miss",
267
- "wildcarded-oidc-sub-claim": oidcWildcards.length > 0 ? "hit" : "miss",
300
+ "workflow-injection-sink": verdict(aggregateHits["workflow-injection-sink"].length > 0, workflowScanIncomplete),
301
+ "pull-request-target-with-pr-checkout": verdict(aggregateHits["pull-request-target-with-pr-checkout"].length > 0, workflowScanIncomplete),
302
+ "actions-floating-tag-pin": verdict(aggregateHits["actions-floating-tag-pin"].length > 0, workflowScanIncomplete),
303
+ "secret-exposed-to-fork-pr": verdict(aggregateHits["secret-exposed-to-fork-pr"].length > 0, workflowScanIncomplete),
304
+ "wildcarded-oidc-sub-claim": verdict(oidcWildcards.length > 0, oidcScanIncomplete),
268
305
  };
269
306
 
270
307
  // File locations for every indicator flipped to "hit", so a SARIF result points
@@ -280,14 +317,20 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
280
317
 
281
318
  const artifacts = {
282
319
  "workflow-yaml-inventory": {
283
- value: workflows.length ? workflows.map(w => w.rel).join(", ") : "no workflow files found at cwd",
284
- captured: true,
320
+ value: workflowScanIncomplete
321
+ ? `${workflows.length} workflow(s) read, at least one skipped — inventory incomplete, see collector_errors`
322
+ : workflows.length ? workflows.map(w => w.rel).join(", ") : "no workflow files found at cwd",
323
+ captured: !workflowScanIncomplete,
324
+ ...(workflowScanIncomplete ? { reason: "one or more workflow files or the workflow directory could not be read" } : {}),
285
325
  },
286
326
  "oidc-trust-policy-inventory": {
287
- value: oidcWildcards.length
288
- ? `${oidcWildcards.length} wildcarded sub-claim(s): ${oidcWildcards.map(f => f.file).join(", ")}`
289
- : "no wildcarded OIDC sub-claim found in infra / terraform / policies",
290
- captured: true,
327
+ value: oidcScanIncomplete
328
+ ? `${oidcWildcards.length} wildcarded sub-claim(s) found, at least one policy file skipped — inventory incomplete, see collector_errors`
329
+ : oidcWildcards.length
330
+ ? `${oidcWildcards.length} wildcarded sub-claim(s): ${oidcWildcards.map(f => f.file).join(", ")}`
331
+ : "no wildcarded OIDC sub-claim found in infra / terraform / policies",
332
+ captured: !oidcScanIncomplete,
333
+ ...(oidcScanIncomplete ? { reason: "one or more trust-policy files or directories could not be read" } : {}),
291
334
  },
292
335
  "actions-sha-pinning": {
293
336
  value: `${aggregateHits["actions-floating-tag-pin"].length} non-SHA third-party uses: reference(s) across ${workflows.length} workflow(s)`,
@@ -319,7 +362,9 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
319
362
  // authorization scope, and the playbook gates the precondition `on_fail: halt`.
320
363
  precondition_checks: {
321
364
  "cwd-is-repo": true,
322
- "ci-config-readable": true,
365
+ // An attestation that the CI configuration was readable, so it cannot
366
+ // stay true over a file this collector failed to open.
367
+ "ci-config-readable": !workflowScanIncomplete,
323
368
  "operator-owns-ci-fleet": args.attestOwnership === true || args["attest-ownership"] === true,
324
369
  },
325
370
  artifacts,
@@ -14,7 +14,6 @@ const { codeExcludeSet, isLinkedWorktreeDir, buildEvidenceLocations } = require(
14
14
 
15
15
  const COLLECTOR_ID = "library-author";
16
16
 
17
- const DEFAULT_MAX_DEPTH = 6;
18
17
  // Applied to the vendor-tree walk, so a cache nested under `vendor/` is not
19
18
  // mistaken for vendored provenance state.
20
19
  const DEFAULT_EXCLUDES = codeExcludeSet();
@@ -96,7 +95,104 @@ function stripYamlComments(content) {
96
95
  return content.replace(/#.*$/gm, "");
97
96
  }
98
97
 
99
- function scanPublishWorkflow(content, rel) {
98
+ // Splits comment-stripped workflow text into individual shell commands, so a
99
+ // flag on one install cannot vouch for another. A workflow that hash-pins its
100
+ // build requirements and then resolves its runtime ones must still report the
101
+ // unpinned command, which a whole-file flag test would suppress.
102
+ function shellCommands(code) {
103
+ const out = [];
104
+ const lines = code.split(/\r?\n/);
105
+ for (let i = 0; i < lines.length; i++) {
106
+ // A real `run:` line is never KBs long; skipping overlong ones keeps a
107
+ // crafted whitespace run from driving regex backtracking.
108
+ if (lines[i].length > 4096) continue;
109
+ // A backslash-continued command is ONE command: a `pip install \` whose
110
+ // `-r` sits on the next line is otherwise two fragments that each match
111
+ // nothing. The continuation lines are consumed, never re-emitted.
112
+ const startLine = i + 1;
113
+ let raw = lines[i];
114
+ while (/\\[ \t]*$/.test(raw) && i + 1 < lines.length &&
115
+ lines[i + 1].length <= 4096 && raw.length <= 8192) {
116
+ raw = `${raw.replace(/\\[ \t]*$/, " ")}${lines[i + 1].trim()}`;
117
+ i++;
118
+ }
119
+ for (const seg of raw.split(/&&|\|\||[;|]/)) {
120
+ const cmd = seg.trim();
121
+ // Comment stripping preserves line breaks, so the index still addresses
122
+ // the line the command STARTS on in the file an operator opens.
123
+ if (cmd) out.push({ cmd, line: startLine });
124
+ }
125
+ }
126
+ return out;
127
+ }
128
+
129
+ // Evidence snippets are read in a report, so a pathological one-line script
130
+ // cannot be pasted in whole.
131
+ function snippetOf(cmd, max = 200) {
132
+ return cmd.length > max ? `${cmd.slice(0, max)}…` : cmd;
133
+ }
134
+
135
+ // pip enters hash-checking mode as soon as ONE requirement carries a hash, so
136
+ // `pip install -r <file>` against a `pip-compile --generate-hashes` output is
137
+ // already reproducible. Resolves the named file one level only: a nested `-r`
138
+ // include is not followed, and an unreadable path counts as unhashed.
139
+ // pip accepts the requirement file attached (`-rreq.txt`) as well as separated
140
+ // (`-r req.txt`, `--requirement=req.txt`). One definition for both the detection
141
+ // predicate and the hash lookup: two regexes for one flag drift, and the pair
142
+ // that drifts silently is a command examined by neither.
143
+ // `(?:^|\s)-r` rather than `\b-r` so the `-r` inside `--requirement` is not read
144
+ // as the short form with `equirement...` as its filename.
145
+ const PIP_REQUIREMENT_RE = /(?:^|\s)(?:-r\s*=?\s*|--requirement[=\s]*)['"]?([^'"\s]+)/;
146
+
147
+ function pipRequirementTarget(cmd) {
148
+ const m = cmd.match(PIP_REQUIREMENT_RE);
149
+ return m ? m[1] : null;
150
+ }
151
+
152
+ // `working-directory:`, at job defaults or on a single step, is what a `run:`
153
+ // command resolves its relative paths against. Collected as a set of candidate
154
+ // bases rather than bound to the step that declared it: the scanner is
155
+ // line-based, so attributing one to a specific command would need the step
156
+ // nesting it does not track.
157
+ // Matches the block form and the flow form (`{ run: { working-directory: x } }`)
158
+ // alike. A key-shaped string inside a `run:` command matches too; that only adds
159
+ // a base that does not resolve, and an unresolvable base is skipped.
160
+ function workingDirectories(code) {
161
+ const dirs = [];
162
+ for (const m of code.matchAll(/working-directory:\s*['"]?([^'"\n,}]+?)['"]?\s*(?=$|[,}])/gm)) {
163
+ const d = m[1].trim();
164
+ if (d && d !== "." && !path.isAbsolute(d)) dirs.push(d);
165
+ }
166
+ return dirs;
167
+ }
168
+
169
+ // True only when every candidate that EXISTS is hash-pinned, and at least one
170
+ // does. Requiring all of them keeps a hashed copy in one working directory from
171
+ // vouching for an unhashed copy in another — over-suppression here reports an
172
+ // unpinned release as reproducible, which is the direction that costs something.
173
+ // A target that resolves nowhere stays unproven, so the caller still reports it.
174
+ function requirementsFileIsHashed(cmd, root, extraBases = []) {
175
+ if (!root) return false;
176
+ const target = pipRequirementTarget(cmd);
177
+ if (!target) return false;
178
+ if (path.isAbsolute(target)) return false;
179
+
180
+ let found = 0;
181
+ for (const base of [root, ...extraBases.map((d) => path.resolve(root, d))]) {
182
+ // Never read outside the scanned tree, whatever `../` the workflow names —
183
+ // for the base itself as well as the target resolved against it.
184
+ if (base !== root && !base.startsWith(root + path.sep)) continue;
185
+ const full = path.resolve(base, target);
186
+ if (full !== root && !full.startsWith(root + path.sep)) continue;
187
+ const body = readSafe(full);
188
+ if (body === null) continue;
189
+ found++;
190
+ if (!/--hash=/.test(body)) return false;
191
+ }
192
+ return found > 0;
193
+ }
194
+
195
+ function scanPublishWorkflow(content, rel, root) {
100
196
  // Whole-content probes read `code`; the `uses:` scan below stays on raw lines.
101
197
  const code = stripYamlComments(content);
102
198
  const hits = {
@@ -140,11 +236,51 @@ function scanPublishWorkflow(content, rel) {
140
236
  }
141
237
  }
142
238
 
239
+ // npm alone stays a whole-file test, because its suppressor is a DIFFERENT
240
+ // command rather than a flag: `npm ci` elsewhere in the workflow is the
241
+ // frozen install, and the `npm install` line beside it is usually tooling.
143
242
  if (/\bnpm\s+install\b/.test(code) && !/\bnpm\s+ci\b/.test(code)) {
144
243
  hits["release-workflow-non-frozen-install"].push({ file: rel, line: 0, snippet: "publish workflow uses `npm install` rather than `npm ci` — lockfile is not enforced" });
145
244
  }
146
- if (/\bcargo\s+(?:build|install)\b/.test(code) && !/--locked\b/.test(code) && !/--frozen\b/.test(code)) {
147
- hits["release-workflow-non-frozen-install"].push({ file: rel, line: 0, snippet: "cargo build/install without --locked / --frozen" });
245
+
246
+ // Bundler's frozen mode can be set once for the job rather than per command
247
+ // (`bundle config set frozen/deployment`, or the BUNDLE_ env equivalents), so
248
+ // that suppressor is genuinely file-scoped; the flags below are not.
249
+ // The setting has to be turned ON. `bundle config set frozen false` names the
250
+ // key and disables it, so matching the key alone reads an explicit opt-out as
251
+ // the protection it opts out of — the env form already required a value.
252
+ const bundlerFrozenConfig =
253
+ /\bbundle\s+config\b[^\n]*\b(?:frozen|deployment)\s+['"]?(?:true|1)\b/.test(code) ||
254
+ /\bBUNDLE_(?:FROZEN|DEPLOYMENT)\s*[:=]\s*['"]?(?:true|1)\b/i.test(code);
255
+
256
+ const runBases = workingDirectories(code);
257
+
258
+ // Every remaining ecosystem is probed one command at a time. The playbook
259
+ // indicator names npm / pnpm / pip -r / bundle; cargo is probed on the same
260
+ // --locked / --frozen predicate.
261
+ for (const { cmd, line } of shellCommands(code)) {
262
+ if (/\bcargo\s+(?:build|install)\b/.test(cmd) && !/--locked\b/.test(cmd) && !/--frozen\b/.test(cmd)) {
263
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `cargo build/install without --locked / --frozen: ${snippetOf(cmd)}` });
264
+ }
265
+ // Without --require-hashes the requirement set resolves at release time, so
266
+ // the artifact is not reproducible — unless the named file is hash-pinned,
267
+ // which switches pip into hash-checking mode on its own.
268
+ if (/\b(?:pip3?|python3?\s+-m\s+pip)\s+install\b/.test(cmd) &&
269
+ pipRequirementTarget(cmd) !== null &&
270
+ !/--require-hashes\b/.test(cmd) &&
271
+ !requirementsFileIsHashed(cmd, root, runBases)) {
272
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `pip install -r without --require-hashes — requirements are resolved at release time: ${snippetOf(cmd)}` });
273
+ }
274
+ // pnpm defaults to --frozen-lockfile when CI is set, which GitHub Actions
275
+ // always sets — so a bare `pnpm install` here is already frozen and only an
276
+ // explicit opt-out resolves at release time.
277
+ if (/\bpnpm\s+(?:install|i)\b/.test(cmd) &&
278
+ /--no-frozen-lockfile\b|--frozen-lockfile[=\s]+false\b/.test(cmd)) {
279
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `pnpm install opts out of the frozen lockfile: ${snippetOf(cmd)}` });
280
+ }
281
+ if (/\bbundle\s+install\b/.test(cmd) && !/--deployment\b/.test(cmd) && !/--frozen\b/.test(cmd) && !bundlerFrozenConfig) {
282
+ hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `bundle install without --deployment / --frozen: ${snippetOf(cmd)}` });
283
+ }
148
284
  }
149
285
 
150
286
  if (/runs-on:\s*['"]?(?:self-hosted|\[?\s*self-hosted)/i.test(code)) {
@@ -193,7 +329,7 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
193
329
  // OIDC capability is evaluated per publish workflow, never repo-wide: a sibling
194
330
  // workflow's `id-token: write` gives nothing to a job using a static token.
195
331
  for (const w of publishWorkflows) {
196
- const h = scanPublishWorkflow(w.content, w.rel);
332
+ const h = scanPublishWorkflow(w.content, w.rel, root);
197
333
  for (const [id, list] of Object.entries(h)) workflowHits[id].push(...list);
198
334
  }
199
335