@blamejs/exceptd-skills 0.19.34 → 0.19.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/bin/exceptd.js +0 -5
- package/data/_indexes/_meta.json +4 -4
- package/data/cve-catalog.json +24 -24
- package/data/zeroday-lessons.json +1249 -1249
- package/lib/collectors/ai-api.js +150 -22
- package/lib/collectors/cicd-pipeline-compromise.js +73 -28
- package/lib/collectors/library-author.js +141 -5
- package/lib/collectors/sbom.js +96 -12
- package/lib/collectors/scan-excludes.js +3 -2
- package/lib/cve-regression-watcher.js +0 -3
- package/lib/framework-gap.js +5 -0
- package/lib/lint-skills.js +24 -4
- package/lib/playbook-runner.js +68 -14
- package/lib/prefetch.js +3 -2
- package/lib/refresh-external.js +0 -6
- package/lib/refresh-network.js +3 -4
- package/lib/scoring.js +8 -1
- package/lib/ttp-mapper.js +14 -3
- package/lib/upstream-check-cli.js +26 -1
- package/lib/validate-cve-catalog.js +9 -2
- package/lib/validate-playbooks.js +9 -11
- package/manifest.json +53 -53
- package/orchestrator/index.js +0 -1
- package/package.json +1 -1
- package/sbom.cdx.json +82 -82
- package/scripts/audit-perf.js +24 -13
- package/scripts/builders/theater-fingerprints.js +9 -4
- package/scripts/check-agents-md-collectors.js +15 -3
- package/scripts/check-codebase-patterns.js +16 -3
- package/scripts/check-manifest-snapshot.js +49 -8
- package/scripts/check-test-coverage.js +17 -1
- package/scripts/refresh-mitre-atlas.js +5 -1
- package/scripts/refresh-mitre-ics-attack.js +5 -1
- package/scripts/refresh-rfc-index.js +6 -1
- package/scripts/refresh-upstream-catalogs.js +25 -13
- package/scripts/release.js +0 -2
- package/scripts/run-e2e-scenarios.js +2 -2
- package/scripts/verify-shipped-tarball.js +0 -1
package/lib/collectors/ai-api.js
CHANGED
|
@@ -13,16 +13,40 @@ const os = require("node:os");
|
|
|
13
13
|
|
|
14
14
|
const COLLECTOR_ID = "ai-api";
|
|
15
15
|
|
|
16
|
-
|
|
16
|
+
// One definition, named at every call site. Written as a literal in both the
|
|
17
|
+
// default and the callers, the two drift apart the moment the cap is raised,
|
|
18
|
+
// and the only symptom is a credential store that quietly stops being scanned
|
|
19
|
+
// at the old size.
|
|
20
|
+
const SCAN_CAP_BYTES = 256 * 1024;
|
|
21
|
+
|
|
22
|
+
// `skip` routes a non-read onto the collector_errors channel:
|
|
23
|
+
// { errors, artifact_id, label }. A null return is indistinguishable between
|
|
24
|
+
// "over the cap" and "unreadable", and both mean the file went unscanned — a
|
|
25
|
+
// clean verdict over an unscanned credential store is the failure mode, so the
|
|
26
|
+
// reason travels with the submission. `label` is home-relative: the absolute
|
|
27
|
+
// path is operator-identifying and collector_meta already carries `home`.
|
|
28
|
+
function readSafe(full, max = SCAN_CAP_BYTES, skip = null) {
|
|
29
|
+
const note = (kind, reason) => {
|
|
30
|
+
if (!skip || !Array.isArray(skip.errors)) return;
|
|
31
|
+
const entry = { kind, reason: `${skip.label || path.basename(full)}: ${reason}` };
|
|
32
|
+
if (skip.artifact_id) entry.artifact_id = skip.artifact_id;
|
|
33
|
+
skip.errors.push(entry);
|
|
34
|
+
};
|
|
17
35
|
let fd;
|
|
18
36
|
try {
|
|
19
37
|
fd = fs.openSync(full, "r");
|
|
20
38
|
const s = fs.fstatSync(fd);
|
|
21
|
-
if (s.size > max)
|
|
39
|
+
if (s.size > max) {
|
|
40
|
+
note("file_too_large_skipped", `${s.size} bytes exceeds ${max}-byte scan limit; not scanned`);
|
|
41
|
+
return null;
|
|
42
|
+
}
|
|
22
43
|
// readFileSync(fd) loops to EOF; a single readSync can return short on a
|
|
23
44
|
// network or FUSE fd. Reading the open fd keeps fstat-then-read TOCTOU-free.
|
|
24
45
|
return fs.readFileSync(fd, "utf8");
|
|
25
|
-
} catch {
|
|
46
|
+
} catch (e) {
|
|
47
|
+
note("read_failed", e.message);
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
26
50
|
finally { if (fd !== undefined) { try { fs.closeSync(fd); } catch { /* non-fatal */ } } }
|
|
27
51
|
}
|
|
28
52
|
|
|
@@ -30,6 +54,27 @@ function fileExists(full) {
|
|
|
30
54
|
try { return fs.statSync(full).isFile(); } catch { return false; }
|
|
31
55
|
}
|
|
32
56
|
|
|
57
|
+
// A credential store is one of three things, and only "absent" is a negative
|
|
58
|
+
// finding. Truthiness collapses the other two: an empty file reads as "" and is
|
|
59
|
+
// a completed scan, not a skipped one, while readSafe signals a skip with null.
|
|
60
|
+
// The distinction is null-vs-string, and it drives the artifact's captured flag
|
|
61
|
+
// and the indicator's verdict together so the two cannot disagree.
|
|
62
|
+
const ABSENT = "absent";
|
|
63
|
+
const UNREAD = "unread";
|
|
64
|
+
const READ = "read";
|
|
65
|
+
function storeState(exists, content) {
|
|
66
|
+
if (!exists) return ABSENT;
|
|
67
|
+
return content === null ? UNREAD : READ;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// A store that was never read cannot answer the question its indicator asks.
|
|
71
|
+
// `miss` there is a clean bill of health over an unscanned file; a hit stands
|
|
72
|
+
// on its own evidence and is unaffected by a sibling going unread.
|
|
73
|
+
function verdict(found, undetermined) {
|
|
74
|
+
if (found) return "hit";
|
|
75
|
+
return undetermined ? "inconclusive" : "miss";
|
|
76
|
+
}
|
|
77
|
+
|
|
33
78
|
// Cleartext key exports: `export VAR=value`, `VAR=value`, fish `set -gx VAR value`.
|
|
34
79
|
const AI_KEY_PATTERNS = [
|
|
35
80
|
{ id: "openai", re: /(?:^|\n)\s*(?:export\s+|set\s+-gx\s+)?OPENAI_API_KEY\s*[= ]\s*['"]?sk-[A-Za-z0-9_-]{20,}/m },
|
|
@@ -129,7 +174,12 @@ function parseGcloudAdc(content) {
|
|
|
129
174
|
privateKey: typeof j?.private_key === "string" ? j.private_key : "",
|
|
130
175
|
clientEmail: typeof j?.client_email === "string" ? j.client_email : "",
|
|
131
176
|
};
|
|
132
|
-
} catch
|
|
177
|
+
} catch (e) {
|
|
178
|
+
// Unparseable ADC is NOT evidence the file holds no service account; the
|
|
179
|
+
// parse failure travels so `gcp-service-account-json: miss` is not read as
|
|
180
|
+
// a clean bill of health.
|
|
181
|
+
return { hasServiceAccount: false, parse_error: e.message };
|
|
182
|
+
}
|
|
133
183
|
}
|
|
134
184
|
|
|
135
185
|
function parseKubeStaticToken(content) {
|
|
@@ -170,7 +220,17 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
170
220
|
for (const e of fs.readdirSync(fishConfD)) {
|
|
171
221
|
if (e.endsWith(".fish")) shellRcs.push(path.join(fishConfD, e));
|
|
172
222
|
}
|
|
173
|
-
} catch {
|
|
223
|
+
} catch (e) {
|
|
224
|
+
// No fish conf.d is the normal case. A permission or I/O failure on one that
|
|
225
|
+
// IS there means those fragments went unscanned for key exports.
|
|
226
|
+
if (e.code !== "ENOENT") {
|
|
227
|
+
errors.push({
|
|
228
|
+
artifact_id: "shell-rc-files",
|
|
229
|
+
kind: "readdir_failed",
|
|
230
|
+
reason: `.config/fish/conf.d: ${e.message} — fragments not scanned`,
|
|
231
|
+
});
|
|
232
|
+
}
|
|
233
|
+
}
|
|
174
234
|
|
|
175
235
|
const dotfileKeys = [
|
|
176
236
|
".openai", ".anthropic",
|
|
@@ -180,13 +240,24 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
180
240
|
path.join(".config", "azure-openai"),
|
|
181
241
|
].map(rel => path.join(home, rel));
|
|
182
242
|
|
|
183
|
-
|
|
243
|
+
// Two artifacts share one scan loop. The carrier's own artifact_id travels
|
|
244
|
+
// with it so a skipped `~/.anthropic` is reported against dotfile-api-keys
|
|
245
|
+
// rather than pointing the operator at the shell-rc row.
|
|
246
|
+
const allKeyCarriers = [
|
|
247
|
+
...shellRcs.map(p => ({ path: p, artifact_id: "shell-rc-files" })),
|
|
248
|
+
...dotfileKeys.map(p => ({ path: p, artifact_id: "dotfile-api-keys" })),
|
|
249
|
+
];
|
|
184
250
|
const cleartextHitsByFile = {};
|
|
251
|
+
const cleartextUnread = [];
|
|
185
252
|
let cleartextFp = null;
|
|
186
|
-
for (const p of allKeyCarriers) {
|
|
253
|
+
for (const { path: p, artifact_id } of allKeyCarriers) {
|
|
187
254
|
if (!fileExists(p)) continue;
|
|
188
|
-
const c = readSafe(p
|
|
189
|
-
|
|
255
|
+
const c = readSafe(p, SCAN_CAP_BYTES, {
|
|
256
|
+
errors, artifact_id, label: path.relative(home, p),
|
|
257
|
+
});
|
|
258
|
+
// Present but unread: the file could still hold a key, so it is a gap in
|
|
259
|
+
// the scan rather than a carrier with nothing in it.
|
|
260
|
+
if (c === null) { cleartextUnread.push(path.relative(home, p)); continue; }
|
|
190
261
|
const hits = scanShellRc(c);
|
|
191
262
|
if (hits.length > 0) {
|
|
192
263
|
cleartextHitsByFile[path.relative(home, p)] = hits;
|
|
@@ -200,24 +271,64 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
200
271
|
const cleartextAnyHit = Object.keys(cleartextHitsByFile).length > 0;
|
|
201
272
|
|
|
202
273
|
const awsCredsPath = path.join(home, ".aws", "credentials");
|
|
203
|
-
const
|
|
274
|
+
const awsCredsExists = fileExists(awsCredsPath);
|
|
275
|
+
const awsCredsContent = awsCredsExists
|
|
276
|
+
? readSafe(awsCredsPath, SCAN_CAP_BYTES, {
|
|
277
|
+
errors, artifact_id: "aws-credentials", label: path.relative(home, awsCredsPath),
|
|
278
|
+
})
|
|
279
|
+
: null;
|
|
204
280
|
const awsParsed = parseAwsCredentials(awsCredsContent);
|
|
205
281
|
const longLivedAws = awsParsed.staticProfiles.length > 0;
|
|
206
282
|
|
|
207
283
|
const gcloudAdcPath = path.join(home, ".config", "gcloud", "application_default_credentials.json");
|
|
208
|
-
const
|
|
284
|
+
const gcloudAdcExists = fileExists(gcloudAdcPath);
|
|
285
|
+
const gcloudContent = gcloudAdcExists
|
|
286
|
+
? readSafe(gcloudAdcPath, SCAN_CAP_BYTES, {
|
|
287
|
+
errors, artifact_id: "gcp-credentials", label: path.relative(home, gcloudAdcPath),
|
|
288
|
+
})
|
|
289
|
+
: null;
|
|
209
290
|
const gcloudParsed = parseGcloudAdc(gcloudContent);
|
|
291
|
+
if (gcloudParsed.parse_error) {
|
|
292
|
+
errors.push({
|
|
293
|
+
artifact_id: "gcp-credentials",
|
|
294
|
+
kind: "parse_failed",
|
|
295
|
+
reason: `${path.relative(home, gcloudAdcPath)}: ${gcloudParsed.parse_error} — service-account presence undetermined, not absent`,
|
|
296
|
+
});
|
|
297
|
+
}
|
|
210
298
|
|
|
211
299
|
const kubeCfgPath = (env && env.KUBECONFIG) || path.join(home, ".kube", "config");
|
|
212
|
-
|
|
300
|
+
// KUBECONFIG can point outside $HOME, where a home-relative label degrades to
|
|
301
|
+
// a `..` chain; the basename is enough to name the file that went unread.
|
|
302
|
+
// The separator is required: a bare prefix test also matches a SIBLING whose
|
|
303
|
+
// name merely starts with $HOME ("/home/rob" vs "/home/robert-backup"), which
|
|
304
|
+
// produces exactly the `../…` chain this branch exists to avoid — and leaks
|
|
305
|
+
// another account's directory name into the warning.
|
|
306
|
+
const kubeInHome = kubeCfgPath === home || kubeCfgPath.startsWith(home + path.sep);
|
|
307
|
+
const kubeLabel = kubeInHome ? path.relative(home, kubeCfgPath) : path.basename(kubeCfgPath);
|
|
308
|
+
const kubeCfgExists = fileExists(kubeCfgPath);
|
|
309
|
+
const kubeContent = kubeCfgExists
|
|
310
|
+
? readSafe(kubeCfgPath, SCAN_CAP_BYTES, {
|
|
311
|
+
errors, artifact_id: "kube-config", label: kubeLabel,
|
|
312
|
+
})
|
|
313
|
+
: null;
|
|
213
314
|
const kubeParsed = parseKubeStaticToken(kubeContent);
|
|
214
315
|
const kubeStaticToken = kubeParsed.found;
|
|
215
316
|
|
|
317
|
+
const awsState = storeState(awsCredsExists, awsCredsContent);
|
|
318
|
+
const gcloudState = storeState(gcloudAdcExists, gcloudContent);
|
|
319
|
+
const kubeState = storeState(kubeCfgExists, kubeContent);
|
|
320
|
+
|
|
216
321
|
const signal_overrides = {
|
|
217
|
-
"cleartext-api-key-in-dotfile": cleartextAnyHit
|
|
218
|
-
"long-lived-aws-keys": longLivedAws
|
|
219
|
-
|
|
220
|
-
|
|
322
|
+
"cleartext-api-key-in-dotfile": verdict(cleartextAnyHit, cleartextUnread.length > 0),
|
|
323
|
+
"long-lived-aws-keys": verdict(longLivedAws, awsState === UNREAD),
|
|
324
|
+
// Unread and unparseable are the same answer here: nothing in the file was
|
|
325
|
+
// validated, so its service-account presence is undetermined rather than
|
|
326
|
+
// absent. JSON.parse never runs on an unread store, so the state carries it.
|
|
327
|
+
"gcp-service-account-json": verdict(
|
|
328
|
+
gcloudParsed.hasServiceAccount,
|
|
329
|
+
gcloudState === UNREAD || Boolean(gcloudParsed.parse_error),
|
|
330
|
+
),
|
|
331
|
+
"kubeconfig-with-static-token": verdict(kubeStaticToken, kubeState === UNREAD),
|
|
221
332
|
};
|
|
222
333
|
|
|
223
334
|
// Per-indicator __fp_checks attestation. Path checks hold because every store
|
|
@@ -271,15 +382,32 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
271
382
|
value: dotfileKeys.filter(p => fileExists(p)).map(p => path.relative(home, p)).join(", ") || "no AI vendor dotfile carriers found at the canonical paths",
|
|
272
383
|
captured: true,
|
|
273
384
|
},
|
|
274
|
-
|
|
385
|
+
// A null content is two different worlds: the store is not there, or it IS
|
|
386
|
+
// there and went unread (over the scan cap, permissions, I/O). Only the
|
|
387
|
+
// first is an absence. The second is captured:false with the reason on
|
|
388
|
+
// collector_errors — asserting "absent" over a credential file that exists
|
|
389
|
+
// is the same clean-verdict-over-an-unscanned-store failure the error
|
|
390
|
+
// channel was added to close.
|
|
391
|
+
"aws-credentials": awsState === READ
|
|
275
392
|
? { value: `${awsParsed.staticProfiles.length} long-lived profile(s): ${awsParsed.staticProfiles.join(", ") || "none"}`, captured: true }
|
|
276
|
-
:
|
|
277
|
-
|
|
393
|
+
: awsState === UNREAD
|
|
394
|
+
? { value: "~/.aws/credentials present but unread — profile inventory undetermined, not absent", captured: false, reason: "read skipped or failed; see collector_errors for the reason" }
|
|
395
|
+
: { value: "~/.aws/credentials absent", captured: true },
|
|
396
|
+
// Read-but-unparseable is a fourth outcome, distinct from read, unread and
|
|
397
|
+
// absent: the bytes arrived and nothing in them was validated. Reporting
|
|
398
|
+
// captured:true there states service_account=false about a file never parsed.
|
|
399
|
+
"gcp-credentials": gcloudState === READ && !gcloudParsed.parse_error
|
|
278
400
|
? { value: `application_default_credentials.json present; service_account=${gcloudParsed.hasServiceAccount}`, captured: true }
|
|
279
|
-
:
|
|
280
|
-
|
|
401
|
+
: gcloudState === READ
|
|
402
|
+
? { value: "application_default_credentials.json present but unparseable — service-account presence undetermined, not absent", captured: false, reason: "JSON parse failed; see collector_errors for the reason" }
|
|
403
|
+
: gcloudState === UNREAD
|
|
404
|
+
? { value: "application_default_credentials.json present but unread — service-account presence undetermined, not absent", captured: false, reason: "read skipped or failed; see collector_errors for the reason" }
|
|
405
|
+
: { value: "no gcloud ADC at the canonical path", captured: true, reason: "credentials.db / legacy_credentials/*/adc.json inspection deferred (no stdlib SQLite reader)" },
|
|
406
|
+
"kube-config": kubeState === READ
|
|
281
407
|
? { value: `kubeconfig present; static_token=${kubeStaticToken}`, captured: true }
|
|
282
|
-
:
|
|
408
|
+
: kubeState === UNREAD
|
|
409
|
+
? { value: "kubeconfig present but unread — static-token presence undetermined, not absent", captured: false, reason: "read skipped or failed; see collector_errors for the reason" }
|
|
410
|
+
: { value: "no kubeconfig at the canonical path", captured: true },
|
|
283
411
|
"ai-sdk-inventory": {
|
|
284
412
|
value: "skipped — npm/pip global listing deferred to operator/AI evidence",
|
|
285
413
|
captured: false,
|
|
@@ -17,40 +17,57 @@ const OIDC_WALK_EXCLUDES = codeExcludeSet();
|
|
|
17
17
|
|
|
18
18
|
const COLLECTOR_ID = "cicd-pipeline-compromise";
|
|
19
19
|
|
|
20
|
+
// Returns { text } when the file was read, otherwise { skipped, reason }: a file
|
|
21
|
+
// that is never scanned reads as a "miss" on every indicator it would have
|
|
22
|
+
// flipped, so the caller has to record the skip rather than drop it.
|
|
20
23
|
function readSafe(p, max = 512 * 1024) {
|
|
21
24
|
let fd;
|
|
22
25
|
try {
|
|
23
26
|
fd = fs.openSync(p, "r");
|
|
24
27
|
const s = fs.fstatSync(fd);
|
|
25
|
-
if (s.size > max) return
|
|
28
|
+
if (s.size > max) return { skipped: "file_too_large", reason: `${s.size} bytes exceeds the ${max}-byte scan limit; not scanned` };
|
|
26
29
|
// readFileSync(fd) loops read() to EOF; a single readSync can return short on
|
|
27
30
|
// a network or FUSE fd. Reading the open fd keeps fstat-then-read TOCTOU-free.
|
|
28
|
-
return fs.readFileSync(fd, "utf8");
|
|
29
|
-
} catch { return
|
|
31
|
+
return { text: fs.readFileSync(fd, "utf8") };
|
|
32
|
+
} catch (e) { return { skipped: "read_error", reason: (e && e.message) || String(e) }; }
|
|
30
33
|
finally { if (fd !== undefined) { try { fs.closeSync(fd); } catch { /* non-fatal */ } } }
|
|
31
34
|
}
|
|
32
35
|
|
|
33
|
-
function
|
|
36
|
+
function skipEntry(artifact_id, rel, r) {
|
|
37
|
+
return {
|
|
38
|
+
artifact_id,
|
|
39
|
+
kind: r.skipped === "file_too_large" ? "file_too_large_skipped" : "read_failed",
|
|
40
|
+
reason: `${rel}: ${r.reason}`,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function walkWorkflows(root, errors = []) {
|
|
34
45
|
const out = [];
|
|
35
46
|
const wfDir = path.join(root, ".github", "workflows");
|
|
36
47
|
if (fs.existsSync(wfDir)) {
|
|
37
48
|
let entries;
|
|
38
49
|
try { entries = fs.readdirSync(wfDir, { withFileTypes: true }); }
|
|
39
|
-
catch {
|
|
50
|
+
catch (e) {
|
|
51
|
+
entries = [];
|
|
52
|
+
errors.push({ artifact_id: "workflow-yaml-inventory", kind: "readdir_failed", reason: `.github/workflows: ${e.message}` });
|
|
53
|
+
}
|
|
40
54
|
for (const e of entries) {
|
|
41
55
|
if (!e.isFile()) continue;
|
|
42
56
|
if (!/\.(ya?ml)$/i.test(e.name)) continue;
|
|
43
57
|
const full = path.join(wfDir, e.name);
|
|
44
|
-
const
|
|
45
|
-
|
|
58
|
+
const rel = path.relative(root, full).replace(/\\/g, "/");
|
|
59
|
+
const r = readSafe(full);
|
|
60
|
+
if (r.text != null) out.push({ full, rel, content: r.text });
|
|
61
|
+
else errors.push(skipEntry("workflow-yaml-inventory", rel, r));
|
|
46
62
|
}
|
|
47
63
|
}
|
|
48
64
|
// Also recognise the most common single-file CI YAMLs at repo root.
|
|
49
65
|
for (const top of [".gitlab-ci.yml", ".circleci/config.yml"]) {
|
|
50
66
|
const full = path.join(root, top);
|
|
51
67
|
if (fs.existsSync(full)) {
|
|
52
|
-
const
|
|
53
|
-
if (
|
|
68
|
+
const r = readSafe(full);
|
|
69
|
+
if (r.text != null) out.push({ full, rel: top, content: r.text });
|
|
70
|
+
else errors.push(skipEntry("workflow-yaml-inventory", top, r));
|
|
54
71
|
}
|
|
55
72
|
}
|
|
56
73
|
return out;
|
|
@@ -184,7 +201,7 @@ function scanWorkflow(content, rel) {
|
|
|
184
201
|
return hits;
|
|
185
202
|
}
|
|
186
203
|
|
|
187
|
-
function scanOidcPolicies(root) {
|
|
204
|
+
function scanOidcPolicies(root, errors = []) {
|
|
188
205
|
// Walks the infra dirs to depth 4 for *.json naming
|
|
189
206
|
// token.actions.githubusercontent.com with a wildcarded sub-claim.
|
|
190
207
|
const rootDirs = ["infra", "terraform", "policies", ".aws", ".github"].map(d => path.join(root, d));
|
|
@@ -193,7 +210,14 @@ function scanOidcPolicies(root) {
|
|
|
193
210
|
if (depth > 4 || finds.length > 5) return;
|
|
194
211
|
let entries;
|
|
195
212
|
try { entries = fs.readdirSync(dir, { withFileTypes: true }); }
|
|
196
|
-
catch {
|
|
213
|
+
catch (e) {
|
|
214
|
+
errors.push({
|
|
215
|
+
artifact_id: "oidc-trust-policy-inventory",
|
|
216
|
+
kind: "readdir_failed",
|
|
217
|
+
reason: `${path.relative(root, dir).replace(/\\/g, "/")}: ${e.message}`,
|
|
218
|
+
});
|
|
219
|
+
return;
|
|
220
|
+
}
|
|
197
221
|
for (const e of entries) {
|
|
198
222
|
if (OIDC_WALK_EXCLUDES.has(e.name)) continue;
|
|
199
223
|
const full = path.join(dir, e.name);
|
|
@@ -204,15 +228,17 @@ function scanOidcPolicies(root) {
|
|
|
204
228
|
continue;
|
|
205
229
|
}
|
|
206
230
|
if (!e.isFile() || !/\.json$/i.test(e.name)) continue;
|
|
207
|
-
const
|
|
208
|
-
|
|
231
|
+
const rel = path.relative(root, full).replace(/\\/g, "/");
|
|
232
|
+
const r = readSafe(full);
|
|
233
|
+
if (r.text == null) { errors.push(skipEntry("oidc-trust-policy-inventory", rel, r)); continue; }
|
|
234
|
+
const text = r.text;
|
|
209
235
|
// Each pattern is bound to the leading `"` of the JSON key, so a lookalike
|
|
210
236
|
// issuer like `"eviltoken.actions…"` cannot match.
|
|
211
237
|
const subWildcard =
|
|
212
238
|
/"token\.actions\.githubusercontent\.com:sub"\s*:\s*"\*"/.test(text) ||
|
|
213
239
|
/"token\.actions\.githubusercontent\.com:sub"\s*:\s*"repo:\*[^"]*"/.test(text) ||
|
|
214
240
|
/"token\.actions\.githubusercontent\.com:sub"\s*:\s*"repo:[^"]*\/\*:[^"]*"/.test(text);
|
|
215
|
-
if (subWildcard) finds.push({ file:
|
|
241
|
+
if (subWildcard) finds.push({ file: rel, snippet: "OIDC sub-claim wildcarded across repos or branches" });
|
|
216
242
|
}
|
|
217
243
|
}
|
|
218
244
|
for (const rd of rootDirs) if (fs.existsSync(rd)) walk(rd, 0);
|
|
@@ -245,7 +271,7 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
245
271
|
};
|
|
246
272
|
}
|
|
247
273
|
|
|
248
|
-
const workflows = walkWorkflows(root);
|
|
274
|
+
const workflows = walkWorkflows(root, errors);
|
|
249
275
|
const aggregateHits = {
|
|
250
276
|
"workflow-injection-sink": [],
|
|
251
277
|
"pull-request-target-with-pr-checkout": [],
|
|
@@ -257,14 +283,25 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
257
283
|
for (const [k, v] of Object.entries(h)) aggregateHits[k].push(...v);
|
|
258
284
|
}
|
|
259
285
|
|
|
260
|
-
const oidcWildcards = scanOidcPolicies(root);
|
|
286
|
+
const oidcWildcards = scanOidcPolicies(root, errors);
|
|
287
|
+
|
|
288
|
+
// A workflow this collector could not read may hold the very thing each
|
|
289
|
+
// indicator looks for, so `miss` over an incomplete inventory is a clean
|
|
290
|
+
// verdict on evidence that was never examined. collector_errors are advisory
|
|
291
|
+
// and change no verdict on their own; the gap has to reach the signal. A hit
|
|
292
|
+
// stands on what WAS read and is unaffected.
|
|
293
|
+
const workflowScanIncomplete = errors.some(
|
|
294
|
+
(e) => e.artifact_id === "workflow-yaml-inventory");
|
|
295
|
+
const oidcScanIncomplete = errors.some(
|
|
296
|
+
(e) => e.artifact_id === "oidc-trust-policy-inventory");
|
|
297
|
+
const verdict = (found, incomplete) => (found ? "hit" : incomplete ? "inconclusive" : "miss");
|
|
261
298
|
|
|
262
299
|
const signal_overrides = {
|
|
263
|
-
"workflow-injection-sink": aggregateHits["workflow-injection-sink"].length > 0
|
|
264
|
-
"pull-request-target-with-pr-checkout": aggregateHits["pull-request-target-with-pr-checkout"].length > 0
|
|
265
|
-
"actions-floating-tag-pin": aggregateHits["actions-floating-tag-pin"].length > 0
|
|
266
|
-
"secret-exposed-to-fork-pr": aggregateHits["secret-exposed-to-fork-pr"].length > 0
|
|
267
|
-
"wildcarded-oidc-sub-claim": oidcWildcards.length > 0
|
|
300
|
+
"workflow-injection-sink": verdict(aggregateHits["workflow-injection-sink"].length > 0, workflowScanIncomplete),
|
|
301
|
+
"pull-request-target-with-pr-checkout": verdict(aggregateHits["pull-request-target-with-pr-checkout"].length > 0, workflowScanIncomplete),
|
|
302
|
+
"actions-floating-tag-pin": verdict(aggregateHits["actions-floating-tag-pin"].length > 0, workflowScanIncomplete),
|
|
303
|
+
"secret-exposed-to-fork-pr": verdict(aggregateHits["secret-exposed-to-fork-pr"].length > 0, workflowScanIncomplete),
|
|
304
|
+
"wildcarded-oidc-sub-claim": verdict(oidcWildcards.length > 0, oidcScanIncomplete),
|
|
268
305
|
};
|
|
269
306
|
|
|
270
307
|
// File locations for every indicator flipped to "hit", so a SARIF result points
|
|
@@ -280,14 +317,20 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
280
317
|
|
|
281
318
|
const artifacts = {
|
|
282
319
|
"workflow-yaml-inventory": {
|
|
283
|
-
value:
|
|
284
|
-
|
|
320
|
+
value: workflowScanIncomplete
|
|
321
|
+
? `${workflows.length} workflow(s) read, at least one skipped — inventory incomplete, see collector_errors`
|
|
322
|
+
: workflows.length ? workflows.map(w => w.rel).join(", ") : "no workflow files found at cwd",
|
|
323
|
+
captured: !workflowScanIncomplete,
|
|
324
|
+
...(workflowScanIncomplete ? { reason: "one or more workflow files or the workflow directory could not be read" } : {}),
|
|
285
325
|
},
|
|
286
326
|
"oidc-trust-policy-inventory": {
|
|
287
|
-
value:
|
|
288
|
-
? `${oidcWildcards.length} wildcarded sub-claim(s)
|
|
289
|
-
:
|
|
290
|
-
|
|
327
|
+
value: oidcScanIncomplete
|
|
328
|
+
? `${oidcWildcards.length} wildcarded sub-claim(s) found, at least one policy file skipped — inventory incomplete, see collector_errors`
|
|
329
|
+
: oidcWildcards.length
|
|
330
|
+
? `${oidcWildcards.length} wildcarded sub-claim(s): ${oidcWildcards.map(f => f.file).join(", ")}`
|
|
331
|
+
: "no wildcarded OIDC sub-claim found in infra / terraform / policies",
|
|
332
|
+
captured: !oidcScanIncomplete,
|
|
333
|
+
...(oidcScanIncomplete ? { reason: "one or more trust-policy files or directories could not be read" } : {}),
|
|
291
334
|
},
|
|
292
335
|
"actions-sha-pinning": {
|
|
293
336
|
value: `${aggregateHits["actions-floating-tag-pin"].length} non-SHA third-party uses: reference(s) across ${workflows.length} workflow(s)`,
|
|
@@ -319,7 +362,9 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
319
362
|
// authorization scope, and the playbook gates the precondition `on_fail: halt`.
|
|
320
363
|
precondition_checks: {
|
|
321
364
|
"cwd-is-repo": true,
|
|
322
|
-
|
|
365
|
+
// An attestation that the CI configuration was readable, so it cannot
|
|
366
|
+
// stay true over a file this collector failed to open.
|
|
367
|
+
"ci-config-readable": !workflowScanIncomplete,
|
|
323
368
|
"operator-owns-ci-fleet": args.attestOwnership === true || args["attest-ownership"] === true,
|
|
324
369
|
},
|
|
325
370
|
artifacts,
|
|
@@ -14,7 +14,6 @@ const { codeExcludeSet, isLinkedWorktreeDir, buildEvidenceLocations } = require(
|
|
|
14
14
|
|
|
15
15
|
const COLLECTOR_ID = "library-author";
|
|
16
16
|
|
|
17
|
-
const DEFAULT_MAX_DEPTH = 6;
|
|
18
17
|
// Applied to the vendor-tree walk, so a cache nested under `vendor/` is not
|
|
19
18
|
// mistaken for vendored provenance state.
|
|
20
19
|
const DEFAULT_EXCLUDES = codeExcludeSet();
|
|
@@ -96,7 +95,104 @@ function stripYamlComments(content) {
|
|
|
96
95
|
return content.replace(/#.*$/gm, "");
|
|
97
96
|
}
|
|
98
97
|
|
|
99
|
-
|
|
98
|
+
// Splits comment-stripped workflow text into individual shell commands, so a
|
|
99
|
+
// flag on one install cannot vouch for another. A workflow that hash-pins its
|
|
100
|
+
// build requirements and then resolves its runtime ones must still report the
|
|
101
|
+
// unpinned command, which a whole-file flag test would suppress.
|
|
102
|
+
function shellCommands(code) {
|
|
103
|
+
const out = [];
|
|
104
|
+
const lines = code.split(/\r?\n/);
|
|
105
|
+
for (let i = 0; i < lines.length; i++) {
|
|
106
|
+
// A real `run:` line is never KBs long; skipping overlong ones keeps a
|
|
107
|
+
// crafted whitespace run from driving regex backtracking.
|
|
108
|
+
if (lines[i].length > 4096) continue;
|
|
109
|
+
// A backslash-continued command is ONE command: a `pip install \` whose
|
|
110
|
+
// `-r` sits on the next line is otherwise two fragments that each match
|
|
111
|
+
// nothing. The continuation lines are consumed, never re-emitted.
|
|
112
|
+
const startLine = i + 1;
|
|
113
|
+
let raw = lines[i];
|
|
114
|
+
while (/\\[ \t]*$/.test(raw) && i + 1 < lines.length &&
|
|
115
|
+
lines[i + 1].length <= 4096 && raw.length <= 8192) {
|
|
116
|
+
raw = `${raw.replace(/\\[ \t]*$/, " ")}${lines[i + 1].trim()}`;
|
|
117
|
+
i++;
|
|
118
|
+
}
|
|
119
|
+
for (const seg of raw.split(/&&|\|\||[;|]/)) {
|
|
120
|
+
const cmd = seg.trim();
|
|
121
|
+
// Comment stripping preserves line breaks, so the index still addresses
|
|
122
|
+
// the line the command STARTS on in the file an operator opens.
|
|
123
|
+
if (cmd) out.push({ cmd, line: startLine });
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
return out;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// Evidence snippets are read in a report, so a pathological one-line script
|
|
130
|
+
// cannot be pasted in whole.
|
|
131
|
+
function snippetOf(cmd, max = 200) {
|
|
132
|
+
return cmd.length > max ? `${cmd.slice(0, max)}…` : cmd;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// pip enters hash-checking mode as soon as ONE requirement carries a hash, so
|
|
136
|
+
// `pip install -r <file>` against a `pip-compile --generate-hashes` output is
|
|
137
|
+
// already reproducible. Resolves the named file one level only: a nested `-r`
|
|
138
|
+
// include is not followed, and an unreadable path counts as unhashed.
|
|
139
|
+
// pip accepts the requirement file attached (`-rreq.txt`) as well as separated
|
|
140
|
+
// (`-r req.txt`, `--requirement=req.txt`). One definition for both the detection
|
|
141
|
+
// predicate and the hash lookup: two regexes for one flag drift, and the pair
|
|
142
|
+
// that drifts silently is a command examined by neither.
|
|
143
|
+
// `(?:^|\s)-r` rather than `\b-r` so the `-r` inside `--requirement` is not read
|
|
144
|
+
// as the short form with `equirement...` as its filename.
|
|
145
|
+
const PIP_REQUIREMENT_RE = /(?:^|\s)(?:-r\s*=?\s*|--requirement[=\s]*)['"]?([^'"\s]+)/;
|
|
146
|
+
|
|
147
|
+
function pipRequirementTarget(cmd) {
|
|
148
|
+
const m = cmd.match(PIP_REQUIREMENT_RE);
|
|
149
|
+
return m ? m[1] : null;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// `working-directory:`, at job defaults or on a single step, is what a `run:`
|
|
153
|
+
// command resolves its relative paths against. Collected as a set of candidate
|
|
154
|
+
// bases rather than bound to the step that declared it: the scanner is
|
|
155
|
+
// line-based, so attributing one to a specific command would need the step
|
|
156
|
+
// nesting it does not track.
|
|
157
|
+
// Matches the block form and the flow form (`{ run: { working-directory: x } }`)
|
|
158
|
+
// alike. A key-shaped string inside a `run:` command matches too; that only adds
|
|
159
|
+
// a base that does not resolve, and an unresolvable base is skipped.
|
|
160
|
+
function workingDirectories(code) {
|
|
161
|
+
const dirs = [];
|
|
162
|
+
for (const m of code.matchAll(/working-directory:\s*['"]?([^'"\n,}]+?)['"]?\s*(?=$|[,}])/gm)) {
|
|
163
|
+
const d = m[1].trim();
|
|
164
|
+
if (d && d !== "." && !path.isAbsolute(d)) dirs.push(d);
|
|
165
|
+
}
|
|
166
|
+
return dirs;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// True only when every candidate that EXISTS is hash-pinned, and at least one
|
|
170
|
+
// does. Requiring all of them keeps a hashed copy in one working directory from
|
|
171
|
+
// vouching for an unhashed copy in another — over-suppression here reports an
|
|
172
|
+
// unpinned release as reproducible, which is the direction that costs something.
|
|
173
|
+
// A target that resolves nowhere stays unproven, so the caller still reports it.
|
|
174
|
+
function requirementsFileIsHashed(cmd, root, extraBases = []) {
|
|
175
|
+
if (!root) return false;
|
|
176
|
+
const target = pipRequirementTarget(cmd);
|
|
177
|
+
if (!target) return false;
|
|
178
|
+
if (path.isAbsolute(target)) return false;
|
|
179
|
+
|
|
180
|
+
let found = 0;
|
|
181
|
+
for (const base of [root, ...extraBases.map((d) => path.resolve(root, d))]) {
|
|
182
|
+
// Never read outside the scanned tree, whatever `../` the workflow names —
|
|
183
|
+
// for the base itself as well as the target resolved against it.
|
|
184
|
+
if (base !== root && !base.startsWith(root + path.sep)) continue;
|
|
185
|
+
const full = path.resolve(base, target);
|
|
186
|
+
if (full !== root && !full.startsWith(root + path.sep)) continue;
|
|
187
|
+
const body = readSafe(full);
|
|
188
|
+
if (body === null) continue;
|
|
189
|
+
found++;
|
|
190
|
+
if (!/--hash=/.test(body)) return false;
|
|
191
|
+
}
|
|
192
|
+
return found > 0;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function scanPublishWorkflow(content, rel, root) {
|
|
100
196
|
// Whole-content probes read `code`; the `uses:` scan below stays on raw lines.
|
|
101
197
|
const code = stripYamlComments(content);
|
|
102
198
|
const hits = {
|
|
@@ -140,11 +236,51 @@ function scanPublishWorkflow(content, rel) {
|
|
|
140
236
|
}
|
|
141
237
|
}
|
|
142
238
|
|
|
239
|
+
// npm alone stays a whole-file test, because its suppressor is a DIFFERENT
|
|
240
|
+
// command rather than a flag: `npm ci` elsewhere in the workflow is the
|
|
241
|
+
// frozen install, and the `npm install` line beside it is usually tooling.
|
|
143
242
|
if (/\bnpm\s+install\b/.test(code) && !/\bnpm\s+ci\b/.test(code)) {
|
|
144
243
|
hits["release-workflow-non-frozen-install"].push({ file: rel, line: 0, snippet: "publish workflow uses `npm install` rather than `npm ci` — lockfile is not enforced" });
|
|
145
244
|
}
|
|
146
|
-
|
|
147
|
-
|
|
245
|
+
|
|
246
|
+
// Bundler's frozen mode can be set once for the job rather than per command
|
|
247
|
+
// (`bundle config set frozen/deployment`, or the BUNDLE_ env equivalents), so
|
|
248
|
+
// that suppressor is genuinely file-scoped; the flags below are not.
|
|
249
|
+
// The setting has to be turned ON. `bundle config set frozen false` names the
|
|
250
|
+
// key and disables it, so matching the key alone reads an explicit opt-out as
|
|
251
|
+
// the protection it opts out of — the env form already required a value.
|
|
252
|
+
const bundlerFrozenConfig =
|
|
253
|
+
/\bbundle\s+config\b[^\n]*\b(?:frozen|deployment)\s+['"]?(?:true|1)\b/.test(code) ||
|
|
254
|
+
/\bBUNDLE_(?:FROZEN|DEPLOYMENT)\s*[:=]\s*['"]?(?:true|1)\b/i.test(code);
|
|
255
|
+
|
|
256
|
+
const runBases = workingDirectories(code);
|
|
257
|
+
|
|
258
|
+
// Every remaining ecosystem is probed one command at a time. The playbook
|
|
259
|
+
// indicator names npm / pnpm / pip -r / bundle; cargo is probed on the same
|
|
260
|
+
// --locked / --frozen predicate.
|
|
261
|
+
for (const { cmd, line } of shellCommands(code)) {
|
|
262
|
+
if (/\bcargo\s+(?:build|install)\b/.test(cmd) && !/--locked\b/.test(cmd) && !/--frozen\b/.test(cmd)) {
|
|
263
|
+
hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `cargo build/install without --locked / --frozen: ${snippetOf(cmd)}` });
|
|
264
|
+
}
|
|
265
|
+
// Without --require-hashes the requirement set resolves at release time, so
|
|
266
|
+
// the artifact is not reproducible — unless the named file is hash-pinned,
|
|
267
|
+
// which switches pip into hash-checking mode on its own.
|
|
268
|
+
if (/\b(?:pip3?|python3?\s+-m\s+pip)\s+install\b/.test(cmd) &&
|
|
269
|
+
pipRequirementTarget(cmd) !== null &&
|
|
270
|
+
!/--require-hashes\b/.test(cmd) &&
|
|
271
|
+
!requirementsFileIsHashed(cmd, root, runBases)) {
|
|
272
|
+
hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `pip install -r without --require-hashes — requirements are resolved at release time: ${snippetOf(cmd)}` });
|
|
273
|
+
}
|
|
274
|
+
// pnpm defaults to --frozen-lockfile when CI is set, which GitHub Actions
|
|
275
|
+
// always sets — so a bare `pnpm install` here is already frozen and only an
|
|
276
|
+
// explicit opt-out resolves at release time.
|
|
277
|
+
if (/\bpnpm\s+(?:install|i)\b/.test(cmd) &&
|
|
278
|
+
/--no-frozen-lockfile\b|--frozen-lockfile[=\s]+false\b/.test(cmd)) {
|
|
279
|
+
hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `pnpm install opts out of the frozen lockfile: ${snippetOf(cmd)}` });
|
|
280
|
+
}
|
|
281
|
+
if (/\bbundle\s+install\b/.test(cmd) && !/--deployment\b/.test(cmd) && !/--frozen\b/.test(cmd) && !bundlerFrozenConfig) {
|
|
282
|
+
hits["release-workflow-non-frozen-install"].push({ file: rel, line, snippet: `bundle install without --deployment / --frozen: ${snippetOf(cmd)}` });
|
|
283
|
+
}
|
|
148
284
|
}
|
|
149
285
|
|
|
150
286
|
if (/runs-on:\s*['"]?(?:self-hosted|\[?\s*self-hosted)/i.test(code)) {
|
|
@@ -193,7 +329,7 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
193
329
|
// OIDC capability is evaluated per publish workflow, never repo-wide: a sibling
|
|
194
330
|
// workflow's `id-token: write` gives nothing to a job using a static token.
|
|
195
331
|
for (const w of publishWorkflows) {
|
|
196
|
-
const h = scanPublishWorkflow(w.content, w.rel);
|
|
332
|
+
const h = scanPublishWorkflow(w.content, w.rel, root);
|
|
197
333
|
for (const [id, list] of Object.entries(h)) workflowHits[id].push(...list);
|
|
198
334
|
}
|
|
199
335
|
|