@blamejs/exceptd-skills 0.19.33 → 0.19.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/bin/exceptd.js +896 -2824
- package/data/_indexes/_meta.json +2 -2
- package/lib/auto-discovery.js +56 -286
- package/lib/canonical-eq.js +7 -40
- package/lib/citation-resolve.js +22 -70
- package/lib/collectors/ai-api.js +20 -54
- package/lib/collectors/cicd-pipeline-compromise.js +40 -108
- package/lib/collectors/citation-hygiene.js +72 -210
- package/lib/collectors/containers.js +41 -130
- package/lib/collectors/cred-stores.js +31 -115
- package/lib/collectors/crypto-codebase.js +55 -138
- package/lib/collectors/crypto.js +24 -54
- package/lib/collectors/hardening.js +20 -78
- package/lib/collectors/kernel.js +16 -46
- package/lib/collectors/library-author.js +57 -206
- package/lib/collectors/mcp.js +24 -70
- package/lib/collectors/runtime.js +24 -86
- package/lib/collectors/sbom.js +34 -106
- package/lib/collectors/scan-excludes.js +31 -138
- package/lib/collectors/secrets.js +62 -178
- package/lib/cross-ref-api.js +39 -123
- package/lib/currency-severity.js +8 -27
- package/lib/cve-batch.js +13 -21
- package/lib/cve-cli.js +13 -20
- package/lib/cve-curation.js +72 -239
- package/lib/cve-regression-watcher.js +29 -152
- package/lib/cvss.js +13 -54
- package/lib/doctor-bucketing.js +3 -19
- package/lib/exit-codes.js +10 -42
- package/lib/flag-suggest.js +7 -25
- package/lib/framework-gap.js +35 -114
- package/lib/gap-detectors.js +37 -159
- package/lib/id-validation.js +9 -30
- package/lib/job-queue.js +13 -36
- package/lib/lint-skills.js +64 -232
- package/lib/playbook-runner.js +693 -2095
- package/lib/prefetch.js +100 -376
- package/lib/refresh-external.js +199 -627
- package/lib/refresh-network.js +75 -307
- package/lib/rfc-cli.js +23 -68
- package/lib/scoring.js +77 -145
- package/lib/sign.js +43 -229
- package/lib/source-advisories.js +43 -194
- package/lib/source-ghsa.js +37 -120
- package/lib/source-osv.js +94 -266
- package/lib/ttp-mapper.js +14 -24
- package/lib/upstream-check-cli.js +10 -28
- package/lib/upstream-check.js +19 -44
- package/lib/validate-catalog-meta.js +17 -61
- package/lib/validate-cve-catalog.js +43 -119
- package/lib/validate-indexes.js +25 -76
- package/lib/validate-package.js +16 -62
- package/lib/validate-playbooks.js +69 -275
- package/lib/validate-vendor.js +16 -49
- package/lib/verify.js +56 -286
- package/lib/version-pins.js +5 -34
- package/lib/worker-pool.js +11 -30
- package/lib/xml-tokenizer.js +47 -152
- package/manifest.json +53 -53
- package/orchestrator/dispatcher.js +17 -68
- package/orchestrator/event-bus.js +11 -74
- package/orchestrator/index.js +138 -412
- package/orchestrator/pipeline.js +28 -85
- package/orchestrator/scanner.js +34 -138
- package/orchestrator/scheduler.js +20 -84
- package/package.json +1 -1
- package/sbom.cdx.json +241 -241
- package/scripts/audit-catalog-gaps.js +9 -62
- package/scripts/audit-cross-skill.js +5 -31
- package/scripts/audit-perf.js +6 -16
- package/scripts/backfill-theater-test.js +7 -64
- package/scripts/bootstrap.js +12 -44
- package/scripts/build-indexes.js +40 -154
- package/scripts/builders/activity-feed.js +4 -14
- package/scripts/builders/catalog-summaries.js +3 -10
- package/scripts/builders/currency.js +7 -20
- package/scripts/builders/cwe-chains.js +7 -30
- package/scripts/builders/did-ladders.js +6 -13
- package/scripts/builders/frequency.js +5 -19
- package/scripts/builders/jurisdiction-clocks.js +6 -25
- package/scripts/builders/recipes.js +6 -14
- package/scripts/builders/section-offsets.js +13 -51
- package/scripts/builders/stale-content.js +7 -28
- package/scripts/builders/summary-cards.js +8 -29
- package/scripts/builders/theater-fingerprints.js +12 -27
- package/scripts/builders/token-budget.js +4 -31
- package/scripts/check-agents-md-collectors.js +11 -54
- package/scripts/check-catalog-gap-budget.js +15 -32
- package/scripts/check-changelog-extract.js +18 -48
- package/scripts/check-codebase-patterns-currency.js +6 -22
- package/scripts/check-codebase-patterns.js +50 -143
- package/scripts/check-epss-consistency.js +9 -64
- package/scripts/check-framework-gap-coverage.js +13 -31
- package/scripts/check-manifest-snapshot.js +13 -73
- package/scripts/check-sbom-currency.js +44 -142
- package/scripts/check-test-count.js +15 -52
- package/scripts/check-test-coverage.js +66 -197
- package/scripts/check-test-subjects.js +21 -62
- package/scripts/check-ttp-references.js +14 -38
- package/scripts/check-ttp-upstream.js +8 -40
- package/scripts/check-version-bump.js +9 -61
- package/scripts/check-version-tags.js +20 -121
- package/scripts/predeploy.js +38 -184
- package/scripts/refresh-manifest-snapshot.js +16 -38
- package/scripts/refresh-mitre-atlas.js +3 -8
- package/scripts/refresh-mitre-attack.js +1 -8
- package/scripts/refresh-mitre-d3fend.js +3 -9
- package/scripts/refresh-mitre-ics-attack.js +3 -8
- package/scripts/refresh-reverse-refs.js +27 -94
- package/scripts/refresh-rfc-index.js +2 -10
- package/scripts/refresh-sbom.js +31 -161
- package/scripts/refresh-upstream-catalogs.js +40 -137
- package/scripts/release.js +69 -232
- package/scripts/run-e2e-scenarios.js +24 -71
- package/scripts/sync-manifest-metadata.js +10 -34
- package/scripts/sync-package-description.js +8 -17
- package/scripts/validate-vendor-online.js +13 -44
- package/scripts/verify-shipped-tarball.js +35 -140
package/lib/collectors/sbom.js
CHANGED
|
@@ -1,16 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* fingerprint of the cwd (npm / yarn / pnpm / pip / cargo / go / ruby /
|
|
8
|
-
* composer) so the runner can correlate against the SBOM-currency +
|
|
9
|
-
* supply-chain integrity indicators. Counts components per lockfile
|
|
10
|
-
* for a coarse SBOM-presence signal.
|
|
11
|
-
*
|
|
12
|
-
* Scope: any cwd with a recognizable lockfile. Multi-ecosystem repos
|
|
13
|
-
* report every detected lockfile.
|
|
4
|
+
* Companion collector for the `sbom` playbook. Fingerprints the lockfiles in the
|
|
5
|
+
* cwd (npm, yarn, pnpm, pip, cargo, go, ruby, composer) and counts components
|
|
6
|
+
* per lockfile as a coarse SBOM-presence signal.
|
|
14
7
|
*
|
|
15
8
|
* Interface: see lib/collectors/README.md
|
|
16
9
|
*/
|
|
@@ -21,8 +14,8 @@ const { buildEvidenceLocations } = require("./scan-excludes");
|
|
|
21
14
|
|
|
22
15
|
const COLLECTOR_ID = "sbom";
|
|
23
16
|
|
|
24
|
-
//
|
|
25
|
-
//
|
|
17
|
+
// Each parser takes the file content and returns
|
|
18
|
+
// { component_count, top_level_count, lockfile_version } or { error }.
|
|
26
19
|
const LOCKFILES = [
|
|
27
20
|
{
|
|
28
21
|
file: "package-lock.json",
|
|
@@ -123,30 +116,21 @@ const LOCKFILES = [
|
|
|
123
116
|
} catch (e) { return { error: e.message }; }
|
|
124
117
|
},
|
|
125
118
|
},
|
|
126
|
-
//
|
|
127
|
-
// canonical project file for modern Python projects). Counted as
|
|
128
|
-
// a Python ecosystem dependency source so projects with only a
|
|
129
|
-
// pyproject.toml + no requirements.txt are recognized.
|
|
119
|
+
// A manifest, not a lockfile: a pyproject-only project is still recognized.
|
|
130
120
|
{
|
|
131
121
|
file: "pyproject.toml",
|
|
132
122
|
ecosystem: "python",
|
|
133
123
|
parser: (content) => {
|
|
134
|
-
//
|
|
135
|
-
// dependencies.*] / [tool.poetry.dependencies] / [tool.poetry
|
|
136
|
-
// .dev-dependencies]. Coarse line-based count — the TOML parser
|
|
137
|
-
// would pull a dep into the stdlib-only contract.
|
|
124
|
+
// Line-based, not parsed: the collector contract is stdlib-only.
|
|
138
125
|
const depBlocks = content.match(/^\[(?:project\.(?:dependencies|optional-dependencies)|tool\.poetry\.(?:dependencies|dev-dependencies|group\.[a-z0-9_-]+\.dependencies))[^\]]*\][\s\S]*?(?=^\[|$)/gm) || [];
|
|
139
126
|
let count = 0;
|
|
140
127
|
for (const block of depBlocks) {
|
|
141
|
-
// count "name = ..." lines (excluding the block header)
|
|
142
128
|
const lines = block.split(/\r?\n/).slice(1);
|
|
143
129
|
for (const line of lines) {
|
|
144
130
|
if (/^\s*[A-Za-z][A-Za-z0-9._\-]*\s*=/.test(line)) count++;
|
|
145
131
|
}
|
|
146
132
|
}
|
|
147
|
-
//
|
|
148
|
-
// [project]
|
|
149
|
-
// dependencies = [ "a", "b", ... ]
|
|
133
|
+
// PEP 621 array style: `dependencies = [ "a", "b" ]` under [project].
|
|
150
134
|
const arrMatch = content.match(/^\s*dependencies\s*=\s*\[([\s\S]*?)\]/m);
|
|
151
135
|
if (arrMatch) {
|
|
152
136
|
const entries = arrMatch[1].match(/"([^"]+)"|'([^']+)'/g) || [];
|
|
@@ -157,10 +141,7 @@ const LOCKFILES = [
|
|
|
157
141
|
},
|
|
158
142
|
];
|
|
159
143
|
|
|
160
|
-
//
|
|
161
|
-
// .txt, requirements-prod.txt, dev-requirements.txt. The
|
|
162
|
-
// requirements.txt entry above covers the canonical name; this
|
|
163
|
-
// glob extends coverage to the common variants.
|
|
144
|
+
// requirements*.txt variants; the canonical name is a LOCKFILES entry above.
|
|
164
145
|
const REQUIREMENTS_GLOB_RE = /^(?:[a-z0-9_-]+-)?requirements(?:-[a-z0-9_-]+)?\.txt$/i;
|
|
165
146
|
const REQUIREMENTS_LF = {
|
|
166
147
|
ecosystem: "pip",
|
|
@@ -170,11 +151,7 @@ const REQUIREMENTS_LF = {
|
|
|
170
151
|
},
|
|
171
152
|
};
|
|
172
153
|
|
|
173
|
-
//
|
|
174
|
-
// the walk bounded. Covers the common monorepo / docs-subdir / iac
|
|
175
|
-
// layouts: docs/ (requirements.txt for sphinx-style docs builds),
|
|
176
|
-
// packages/* (monorepo workspaces), backend/ + frontend/ +
|
|
177
|
-
// infra/ + iac/ (split-stack repos).
|
|
154
|
+
// Probed one level deep and hand-listed so the walk stays bounded.
|
|
178
155
|
const SUBDIR_PROBE_PATHS = ["docs", "packages", "backend", "frontend", "infra", "iac", "src", "app"];
|
|
179
156
|
|
|
180
157
|
const SBOM_FORMATS = [
|
|
@@ -203,16 +180,12 @@ function captureLockfile(p, ecosystem, parser, label) {
|
|
|
203
180
|
|
|
204
181
|
function findLockfiles(cwd) {
|
|
205
182
|
const found = [];
|
|
206
|
-
// Canonical names at cwd root.
|
|
207
183
|
for (const lf of LOCKFILES) {
|
|
208
184
|
const p = path.join(cwd, lf.file);
|
|
209
185
|
if (fs.existsSync(p)) {
|
|
210
186
|
found.push(captureLockfile(p, lf.ecosystem, lf.parser, lf.file));
|
|
211
187
|
}
|
|
212
188
|
}
|
|
213
|
-
// requirements*.txt glob at cwd root — covers requirements-dev.txt,
|
|
214
|
-
// dev-requirements.txt, etc. The exact-name `requirements.txt`
|
|
215
|
-
// already lands via LOCKFILES; skip it here.
|
|
216
189
|
try {
|
|
217
190
|
for (const entry of fs.readdirSync(cwd)) {
|
|
218
191
|
if (entry === "requirements.txt") continue; // captured above
|
|
@@ -225,11 +198,6 @@ function findLockfiles(cwd) {
|
|
|
225
198
|
}
|
|
226
199
|
} catch { /* swallow */ }
|
|
227
200
|
|
|
228
|
-
// One-level subdirectory probe for canonical-name lockfiles. The
|
|
229
|
-
// common pattern: docs/requirements.txt (sphinx builds),
|
|
230
|
-
// packages/*/package.json (monorepo workspaces), backend/Gemfile
|
|
231
|
-
// .lock (split-stack repos). Capped depth (1 level only) and
|
|
232
|
-
// pre-listed subdirs to keep the walk bounded.
|
|
233
201
|
for (const sub of SUBDIR_PROBE_PATHS) {
|
|
234
202
|
const subDir = path.join(cwd, sub);
|
|
235
203
|
let entries;
|
|
@@ -239,9 +207,7 @@ function findLockfiles(cwd) {
|
|
|
239
207
|
} catch { continue; }
|
|
240
208
|
for (const e of entries) {
|
|
241
209
|
if (e.isDirectory()) {
|
|
242
|
-
//
|
|
243
|
-
// names. (Doesn't recurse further; monorepo workspaces are
|
|
244
|
-
// the only common case.)
|
|
210
|
+
// packages/* gets one more level for monorepo workspaces, no deeper.
|
|
245
211
|
if (sub === "packages") {
|
|
246
212
|
for (const lf of LOCKFILES) {
|
|
247
213
|
const p = path.join(subDir, e.name, lf.file);
|
|
@@ -254,10 +220,7 @@ function findLockfiles(cwd) {
|
|
|
254
220
|
continue;
|
|
255
221
|
}
|
|
256
222
|
if (!e.isFile()) continue;
|
|
257
|
-
//
|
|
258
|
-
// file is captured there. The requirements glob below catches
|
|
259
|
-
// ONLY non-canonical names (e.g. requirements-dev.txt) so
|
|
260
|
-
// exact `requirements.txt` doesn't double-fire.
|
|
223
|
+
// A canonical name wins, so `requirements.txt` cannot double-fire below.
|
|
261
224
|
let captured = false;
|
|
262
225
|
for (const lf of LOCKFILES) {
|
|
263
226
|
if (e.name === lf.file) {
|
|
@@ -284,21 +247,16 @@ function findSbomDocuments(cwd) {
|
|
|
284
247
|
for (const s of SBOM_FORMATS) {
|
|
285
248
|
const p = path.join(cwd, s.file);
|
|
286
249
|
let fd;
|
|
287
|
-
// Open once and fstat the descriptor
|
|
288
|
-
//
|
|
289
|
-
// but there is no TOCTOU window between the check, the size stat, and the read.
|
|
250
|
+
// Open once and fstat the descriptor rather than existsSync→statSync→read:
|
|
251
|
+
// ENOENT still skips, and there is no TOCTOU window.
|
|
290
252
|
try { fd = fs.openSync(p, "r"); }
|
|
291
253
|
catch (e) { if (e.code === "ENOENT") continue; found.push({ file: s.file, format: s.format, error: e.message }); continue; }
|
|
292
254
|
try {
|
|
293
255
|
const stat = fs.fstatSync(fd);
|
|
294
256
|
let content;
|
|
295
|
-
// Read from the
|
|
296
|
-
//
|
|
297
|
-
//
|
|
298
|
-
// NOT — it can return a short read (network/FUSE mount, signal), leaving
|
|
299
|
-
// the buffer NUL-padded and truncating valid JSON, so JSON.parse would
|
|
300
|
-
// throw and component_count silently fall back to null on a present,
|
|
301
|
-
// parseable SBOM. The fd stays open for the finally{} closeSync.
|
|
257
|
+
// Read from the open descriptor, never by path again. readFileSync(fd) loops
|
|
258
|
+
// to EOF; a single readSync can return short on a network or FUSE mount,
|
|
259
|
+
// NUL-padding the buffer and truncating otherwise-valid JSON.
|
|
302
260
|
try { content = fs.readFileSync(fd, "utf8"); } catch { content = null; }
|
|
303
261
|
let component_count = null;
|
|
304
262
|
if (content && s.format === "cyclonedx-1.x") {
|
|
@@ -322,9 +280,7 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
322
280
|
const startTime = Date.now();
|
|
323
281
|
const root = path.resolve(cwd);
|
|
324
282
|
|
|
325
|
-
//
|
|
326
|
-
// this as "an operator has SOME way to produce an SBOM". For the
|
|
327
|
-
// collector, "we found a lockfile or an SBOM" is a sufficient proxy.
|
|
283
|
+
// sbom-tool-available: a lockfile or an SBOM document is a sufficient proxy.
|
|
328
284
|
const lockfiles = findLockfiles(root);
|
|
329
285
|
const sbomDocuments = findSbomDocuments(root);
|
|
330
286
|
const hasAnything = lockfiles.length > 0 || sbomDocuments.length > 0;
|
|
@@ -344,22 +300,9 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
344
300
|
},
|
|
345
301
|
};
|
|
346
302
|
|
|
347
|
-
//
|
|
348
|
-
//
|
|
349
|
-
//
|
|
350
|
-
// catalog cross-referencing the collector does not have — that's
|
|
351
|
-
// the runner's job. The collector's role here is to surface the
|
|
352
|
-
// artifacts (lockfile-inventory + sbom-document) and let the
|
|
353
|
-
// runner evaluate the indicators against them. Emitting
|
|
354
|
-
// signal_overrides for keys that don't exist in the playbook
|
|
355
|
-
// would be silently ignored; surfacing the artifacts honestly
|
|
356
|
-
// is the contract.
|
|
357
|
-
//
|
|
358
|
-
// One indicator the collector CAN decide deterministically:
|
|
359
|
-
// lockfile-no-integrity — true when an npm package-lock.json
|
|
360
|
-
// exists but has zero `integrity` entries (lockfileVersion 1
|
|
361
|
-
// legacy) OR when the dependency list contains entries lacking
|
|
362
|
-
// `integrity` strings.
|
|
303
|
+
// lockfile-no-integrity is the one indicator decidable here: an npm
|
|
304
|
+
// package-lock.json with entries carrying no `integrity`. The rest need catalog
|
|
305
|
+
// cross-referencing, which the runner does against the artifacts above.
|
|
363
306
|
const npmLockfile = lockfiles.find(l => l.file === "package-lock.json");
|
|
364
307
|
const signal_overrides = {};
|
|
365
308
|
if (npmLockfile && !npmLockfile.error) {
|
|
@@ -367,19 +310,15 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
367
310
|
const j = JSON.parse(fs.readFileSync(npmLockfile.path, "utf8"));
|
|
368
311
|
let withIntegrity = 0;
|
|
369
312
|
let withoutIntegrity = 0;
|
|
370
|
-
//
|
|
371
|
-
//
|
|
372
|
-
// have no registry integrity hash. A remote-registry tarball without
|
|
373
|
-
// integrity is the genuine finding.
|
|
313
|
+
// A local-path, workspace or git ref legitimately has no integrity hash, so
|
|
314
|
+
// FP check [0] demotes those; a remote-registry tarball is the finding.
|
|
374
315
|
let withoutIntegrityLocalOnly = true;
|
|
375
316
|
const LOCAL_REF_RE = /^(?:file:|link:|workspace:|git\+ssh:|git\+https:|git:|github:|portal:)/i;
|
|
376
317
|
const walk = (obj) => {
|
|
377
318
|
if (!obj || typeof obj !== "object") return;
|
|
378
|
-
// Only
|
|
379
|
-
//
|
|
380
|
-
// `
|
|
381
|
-
// `integrity`, so keying off `version` would false-positive on
|
|
382
|
-
// every clean lockfile. Mirror library-author.js's guard.
|
|
319
|
+
// Only entries with a `resolved` URL are expected to carry `integrity`.
|
|
320
|
+
// The npm 7+ root entry `"": { name, version }` has neither, so keying off
|
|
321
|
+
// `version` false-positives. library-author.js guards the same way.
|
|
383
322
|
if (obj.resolved != null) {
|
|
384
323
|
if (obj.integrity != null) {
|
|
385
324
|
withIntegrity++;
|
|
@@ -391,15 +330,12 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
391
330
|
for (const v of Object.values(obj)) if (v && typeof v === "object") walk(v);
|
|
392
331
|
};
|
|
393
332
|
walk(j.packages || j.dependencies || {});
|
|
394
|
-
//
|
|
395
|
-
// resolves to a remote tarball — the indicator captures the
|
|
396
|
-
// class, not full coverage.
|
|
333
|
+
// Any single integrity-less resolved entry fires it: the class, not coverage.
|
|
397
334
|
if (withoutIntegrity > 0) {
|
|
398
335
|
signal_overrides["lockfile-no-integrity"] = "hit";
|
|
399
|
-
// __fp_checks attestation
|
|
400
|
-
//
|
|
401
|
-
//
|
|
402
|
-
// build consumes, not a stale copy under archive/ pre-migration/.
|
|
336
|
+
// __fp_checks attestation — [0]: an integrity-less entry is a remote
|
|
337
|
+
// tarball, not only local, workspace or git refs. [1]: this is the root
|
|
338
|
+
// lockfile the build consumes, not a stale archived copy.
|
|
403
339
|
const att = {};
|
|
404
340
|
if (!withoutIntegrityLocalOnly) att["0"] = true;
|
|
405
341
|
const rel = (npmLockfile.path || "").replace(/\\/g, "/");
|
|
@@ -408,19 +344,14 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
408
344
|
} else if (withIntegrity > 0) {
|
|
409
345
|
signal_overrides["lockfile-no-integrity"] = "miss";
|
|
410
346
|
}
|
|
411
|
-
// Stash diagnostic counts on collector_meta further below.
|
|
412
347
|
npmLockfile.integrity_present_count = withIntegrity;
|
|
413
348
|
npmLockfile.integrity_missing_count = withoutIntegrity;
|
|
414
349
|
} catch {
|
|
415
|
-
//
|
|
416
|
-
// runner returns inconclusive rather than a forced miss.
|
|
350
|
+
// A malformed lockfile leaves the indicator unflipped: inconclusive, not miss.
|
|
417
351
|
}
|
|
418
352
|
}
|
|
419
353
|
|
|
420
|
-
//
|
|
421
|
-
// indicator: a lockfile-no-integrity hit points at the npm lockfile that
|
|
422
|
-
// carries integrity-less entries. File-level (the gap is spread across
|
|
423
|
-
// many entries, not one line).
|
|
354
|
+
// File-level: the gap is spread across entries rather than sitting on one line.
|
|
424
355
|
const evidence_locations = {};
|
|
425
356
|
if (signal_overrides["lockfile-no-integrity"] === "hit" && npmLockfile) {
|
|
426
357
|
const locs = buildEvidenceLocations([{ file: npmLockfile.file }]);
|
|
@@ -430,11 +361,8 @@ function collect({ cwd = process.cwd(), env = process.env, args = {} } = {}) {
|
|
|
430
361
|
return {
|
|
431
362
|
precondition_checks: {
|
|
432
363
|
"sbom-tool-available": hasAnything,
|
|
433
|
-
//
|
|
434
|
-
//
|
|
435
|
-
// the scanned --cwd (it sees the run process cwd, not the collected repo),
|
|
436
|
-
// so a lockfile we found here would otherwise surface a spurious
|
|
437
|
-
// precondition_unverified warning on a repo that clearly has one.
|
|
364
|
+
// Attested from what was collected: autoDetectPreconditions sees the run
|
|
365
|
+
// process cwd, not the scanned --cwd, so this would warn as unverified.
|
|
438
366
|
"any-package-manager-present": lockfiles.length > 0,
|
|
439
367
|
},
|
|
440
368
|
artifacts,
|
|
@@ -1,72 +1,33 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* skipped agent/editor worktree copies (`.claude/worktrees/`), so a tree
|
|
10
|
-
* holding N detached worktrees scanned each one as a full repo — inflating
|
|
11
|
-
* hit counts to N+1 duplicates of the same source and forcing manual dedup.
|
|
12
|
-
*
|
|
13
|
-
* Two layers:
|
|
14
|
-
* 1. NAME exclusions — directory basenames never worth descending into
|
|
15
|
-
* (dependency caches, build output, VCS metadata, agent scratch).
|
|
16
|
-
* 2. A LINKED-WORKTREE predicate — a directory that is itself a git
|
|
17
|
-
* worktree distinct from the scan root. A `git worktree add` target
|
|
18
|
-
* carries a `.git` *file* (a gitdir pointer) rather than a `.git`
|
|
19
|
-
* directory; agent tools frequently stamp full repo copies under
|
|
20
|
-
* `.claude/worktrees/`. Skipping these keeps a scan to the one tree
|
|
21
|
-
* the operator actually pointed at.
|
|
22
|
-
*
|
|
23
|
-
* Collectors apply both: spread DEFAULT_CODE_EXCLUDES into their exclude
|
|
24
|
-
* Set, and call isLinkedWorktreeDir(fullPath) before descending into a
|
|
25
|
-
* subdirectory.
|
|
4
|
+
* Shared directory-walk exclusion policy for every code-scope collector: a set of
|
|
5
|
+
* directory basenames never worth descending into, and a predicate for linked git
|
|
6
|
+
* worktrees, which rescan the host tree's source and multiply every hit.
|
|
7
|
+
* Collectors spread DEFAULT_CODE_EXCLUDES into their exclude Set and call
|
|
8
|
+
* isLinkedWorktreeDir(fullPath) before descending.
|
|
26
9
|
*/
|
|
27
10
|
|
|
28
11
|
const fs = require("node:fs");
|
|
29
12
|
const path = require("node:path");
|
|
30
13
|
|
|
31
|
-
// Directory basenames excluded from every code-scope walk. Superset of the
|
|
32
|
-
// per-collector lists that predated this module, plus agent/editor scratch
|
|
33
|
-
// (`.claude`) and additional dependency/build caches that only ever hold
|
|
34
|
-
// generated or third-party content.
|
|
35
14
|
const DEFAULT_CODE_EXCLUDES = Object.freeze([
|
|
36
|
-
// VCS + agent/editor scratch
|
|
37
15
|
".git", ".hg", ".svn", ".claude", ".idea", ".vscode",
|
|
38
|
-
// dependency trees / package caches
|
|
39
16
|
"node_modules", ".pnpm-store", "bower_components",
|
|
40
17
|
".venv", "venv", "__pycache__", ".pytest_cache", ".mypy_cache",
|
|
41
18
|
".tox", ".gradle", ".m2",
|
|
42
|
-
// build output
|
|
43
19
|
"dist", "build", "out", "target", "coverage",
|
|
44
20
|
".next", ".nuxt", ".svelte-kit", ".turbo", ".cache",
|
|
45
21
|
]);
|
|
46
22
|
|
|
47
|
-
/**
|
|
48
|
-
* Build an exclude Set for a collector. Pass any collector-specific extra
|
|
49
|
-
* basenames; they are merged with the shared defaults.
|
|
50
|
-
*
|
|
51
|
-
* @param {Iterable<string>} [extra] additional basenames to exclude
|
|
52
|
-
* @returns {Set<string>}
|
|
53
|
-
*/
|
|
54
23
|
function codeExcludeSet(extra = []) {
|
|
55
24
|
return new Set([...DEFAULT_CODE_EXCLUDES, ...extra]);
|
|
56
25
|
}
|
|
57
26
|
|
|
58
27
|
/**
|
|
59
|
-
* True when `dir` is a git worktree linked
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
* `.claude/worktrees/<id>/`); descending into them rescans unrelated repo
|
|
63
|
-
* state. A normal repo root has a `.git` *directory* and is NOT skipped.
|
|
64
|
-
*
|
|
65
|
-
* Cheap and synchronous: one lstat. Returns false on any error so a walk
|
|
66
|
-
* never aborts on a permission/race issue.
|
|
67
|
-
*
|
|
68
|
-
* @param {string} dir absolute path to a candidate directory
|
|
69
|
-
* @returns {boolean}
|
|
28
|
+
* True when `dir` is a git worktree linked elsewhere — its `.git` entry is a file
|
|
29
|
+
* (a `gitdir: …` pointer), where a normal repo root has a `.git` *directory* and
|
|
30
|
+
* is NOT skipped. Returns false on any error, so a walk never aborts on a race.
|
|
70
31
|
*/
|
|
71
32
|
function isLinkedWorktreeDir(dir) {
|
|
72
33
|
try {
|
|
@@ -79,46 +40,14 @@ function isLinkedWorktreeDir(dir) {
|
|
|
79
40
|
}
|
|
80
41
|
|
|
81
42
|
/**
|
|
82
|
-
* Shared cwd-tree walker for every code-scope collector. Returns one entry
|
|
83
|
-
*
|
|
84
|
-
*
|
|
85
|
-
*
|
|
86
|
-
*
|
|
87
|
-
* Behavior preserved from those walkers:
|
|
88
|
-
* - Directory basenames in `excludes` are never descended into.
|
|
89
|
-
* - A linked git worktree (its `.git` is a gitdir-pointer FILE) is never
|
|
90
|
-
* descended into (isLinkedWorktreeDir).
|
|
91
|
-
* - Recursion follows ONLY real directories (`entry.isDirectory()`).
|
|
92
|
-
* Symlinks are never traversed and symlink targets are never emitted
|
|
93
|
-
* as files — matching the old behavior where a Dirent for a symlink is
|
|
94
|
-
* neither isDirectory() nor isFile(), so it fell through both branches.
|
|
95
|
-
* - Depth is capped at `maxDepth`.
|
|
96
|
-
*
|
|
97
|
-
* Performance: `fs.realpathSync` is called ONLY on directories and symlinks
|
|
98
|
-
* (the only entries that can introduce a traversal cycle), never on regular
|
|
99
|
-
* files. The old walkers realpath'd every file entry, which on a large repo
|
|
100
|
-
* dominated walk time. A regular file cannot create a directory cycle, so
|
|
101
|
-
* dropping its realpath preserves cycle protection exactly while removing
|
|
102
|
-
* the per-file syscall. The symlink-cycle `seen` set still guards every
|
|
103
|
-
* directory and symlink by canonical path.
|
|
104
|
-
*
|
|
105
|
-
* Depth-cap visibility: a subtree pruned because it would exceed `maxDepth`
|
|
106
|
-
* is otherwise an invisible coverage gap — the files inside are never emitted
|
|
107
|
-
* and the caller cannot tell "scanned, nothing there" from "never scanned".
|
|
108
|
-
* Pass `opts.truncations` (an array) to have the walker push one
|
|
109
|
-
* `{ rel, depth }` entry per directory whose contents were skipped for being
|
|
110
|
-
* beyond the cap, so a content collector can surface a `depth_capped` notice
|
|
111
|
-
* symmetric with its per-file size-cap reporting. The return value is
|
|
112
|
-
* unchanged (the file array); callers that don't pass `truncations` are
|
|
113
|
-
* unaffected.
|
|
43
|
+
* Shared cwd-tree walker for every code-scope collector. Returns one entry per
|
|
44
|
+
* regular file as `{ full, rel, name }`, `rel` forward-slashed on every platform.
|
|
45
|
+
* Directory basenames in `excludes` and linked git worktrees are never descended
|
|
46
|
+
* into, recursion follows only real directories — a symlink is neither traversed
|
|
47
|
+
* nor emitted — and depth is capped at `maxDepth`.
|
|
114
48
|
*
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
* @param {number} [opts.maxDepth] max recursion depth (collector-specific)
|
|
118
|
-
* @param {Set<string>} [opts.excludes] directory basenames to skip
|
|
119
|
-
* @param {Array<{rel:string, depth:number}>} [opts.truncations] out-param the
|
|
120
|
-
* walker appends to when a directory is pruned for exceeding maxDepth
|
|
121
|
-
* @returns {Array<{full:string, rel:string, name:string}>}
|
|
49
|
+
* `opts.truncations`, when an array, is an out-param: one `{ rel, depth }` per
|
|
50
|
+
* directory pruned for exceeding the cap. The return value is the file array.
|
|
122
51
|
*/
|
|
123
52
|
function walkTree(root, opts = {}) {
|
|
124
53
|
const maxDepth = opts.maxDepth ?? 6;
|
|
@@ -137,21 +66,14 @@ function walkTree(root, opts = {}) {
|
|
|
137
66
|
const full = path.join(dir, entry.name);
|
|
138
67
|
|
|
139
68
|
if (entry.isDirectory()) {
|
|
140
|
-
// Cycle guard: canonicalize
|
|
141
|
-
//
|
|
142
|
-
// here is bounded by directory count, not file count.
|
|
69
|
+
// Cycle guard: canonicalize so a symlinked or bind-mounted loop is
|
|
70
|
+
// visited at most once.
|
|
143
71
|
let real;
|
|
144
72
|
try { real = fs.realpathSync(full); } catch { continue; }
|
|
145
73
|
if (seen.has(real)) continue;
|
|
146
74
|
seen.add(real);
|
|
147
|
-
// Never descend into a linked git worktree (its `.git` is a gitdir
|
|
148
|
-
// pointer file) — agent tooling stamps full repo copies under
|
|
149
|
-
// `.claude/worktrees/<id>/`; walking them rescans the same source
|
|
150
|
-
// as the host tree and multiplies every hit.
|
|
151
75
|
if (isLinkedWorktreeDir(full)) continue;
|
|
152
|
-
//
|
|
153
|
-
// contents are about to be dropped. Record the prune so the caller
|
|
154
|
-
// can report the unscanned subtree instead of silently missing it.
|
|
76
|
+
// Checked before descending, so the pruned directory can be recorded.
|
|
155
77
|
if (depth + 1 > maxDepth) {
|
|
156
78
|
if (truncations) {
|
|
157
79
|
truncations.push({
|
|
@@ -163,19 +85,12 @@ function walkTree(root, opts = {}) {
|
|
|
163
85
|
}
|
|
164
86
|
walk(full, depth + 1);
|
|
165
87
|
} else if (entry.isSymbolicLink()) {
|
|
166
|
-
// A symlink Dirent is neither isDirectory() nor isFile()
|
|
167
|
-
// walkers realpath'd it (adding the target to `seen`) and then
|
|
168
|
-
// emitted nothing, because neither branch matched. Preserve that:
|
|
169
|
-
// canonicalize for the cycle guard, never emit, never traverse.
|
|
88
|
+
// A symlink Dirent is neither isDirectory() nor isFile(): never emitted.
|
|
170
89
|
try { fs.realpathSync(full); } catch { /* dangling link — ignore */ }
|
|
171
90
|
} else if (entry.isFile()) {
|
|
172
|
-
//
|
|
173
|
-
// `seen` check
|
|
174
|
-
//
|
|
175
|
-
// readdir.
|
|
176
|
-
// Emit forward-slash rel paths on every platform so artifact
|
|
177
|
-
// summaries match the SARIF evidence_locations (which normalize the
|
|
178
|
-
// same way) — on Windows path.relative returns backslash separators.
|
|
91
|
+
// A regular file cannot introduce a directory cycle, so no realpath /
|
|
92
|
+
// `seen` check. Windows path.relative returns backslashes, so rel is
|
|
93
|
+
// normalized to match SARIF locations.
|
|
179
94
|
out.push({ full, rel: path.relative(root, full).split(path.sep).join("/"), name: entry.name });
|
|
180
95
|
}
|
|
181
96
|
}
|
|
@@ -184,27 +99,15 @@ function walkTree(root, opts = {}) {
|
|
|
184
99
|
return out;
|
|
185
100
|
}
|
|
186
101
|
|
|
187
|
-
// Per-indicator cap
|
|
188
|
-
// upload. 50 locations per result is well within GitHub code-scanning's
|
|
189
|
-
// rendering budget while still showing the operator where every distinct
|
|
190
|
-
// finding lives.
|
|
102
|
+
// Per-indicator cap: 50 sits well inside GitHub code-scanning's rendering budget.
|
|
191
103
|
const MAX_EVIDENCE_LOCATIONS_PER_INDICATOR = 50;
|
|
192
104
|
|
|
193
105
|
/**
|
|
194
|
-
*
|
|
195
|
-
*
|
|
196
|
-
* `
|
|
197
|
-
* `
|
|
198
|
-
*
|
|
199
|
-
* present — many collectors store `line: 0` to mean "file-level, no line",
|
|
200
|
-
* which omits the region so SARIF points at the file rather than line 0.
|
|
201
|
-
*
|
|
202
|
-
* Identical `{ uri, startLine }` entries are de-duplicated and the list is
|
|
203
|
-
* capped at MAX_EVIDENCE_LOCATIONS_PER_INDICATOR.
|
|
204
|
-
*
|
|
205
|
-
* @param {Array<object>} hits entries with a `rel` or `file` repo-relative
|
|
206
|
-
* path and an optional `line` number
|
|
207
|
-
* @returns {Array<{uri:string,startLine?:number}>}
|
|
106
|
+
* Turns a collector's per-indicator file hits into `evidence_locations` for SARIF
|
|
107
|
+
* `results[].locations`. Each hit yields `{ uri, startLine? }` from its `rel` (or
|
|
108
|
+
* `file`) repo-relative path; `startLine` is emitted only above zero, because
|
|
109
|
+
* collectors store `line: 0` to mean "file-level, no line". Identical entries
|
|
110
|
+
* de-duplicate; the list caps at MAX_EVIDENCE_LOCATIONS_PER_INDICATOR.
|
|
208
111
|
*/
|
|
209
112
|
function buildEvidenceLocations(hits) {
|
|
210
113
|
if (!Array.isArray(hits) || hits.length === 0) return [];
|
|
@@ -227,19 +130,9 @@ function buildEvidenceLocations(hits) {
|
|
|
227
130
|
}
|
|
228
131
|
|
|
229
132
|
/**
|
|
230
|
-
*
|
|
231
|
-
*
|
|
232
|
-
*
|
|
233
|
-
* with no region. Pairing the offset with this helper lets SARIF point at the
|
|
234
|
-
* exact line.
|
|
235
|
-
*
|
|
236
|
-
* Returns 1 for offset 0 / a non-finite offset (file-level fallback at the
|
|
237
|
-
* first line). Counts `\n` occurrences before the offset; `\r\n` is handled
|
|
238
|
-
* because the `\n` is what increments the line.
|
|
239
|
-
*
|
|
240
|
-
* @param {string} content the file text the offset indexes into
|
|
241
|
-
* @param {number} offset 0-based offset of the match within `content`
|
|
242
|
-
* @returns {number} 1-based line number
|
|
133
|
+
* Converts a 0-based offset into `content` to a 1-based line number, so a
|
|
134
|
+
* content-regex collector can pair a match's `m.index` with a SARIF region.
|
|
135
|
+
* Returns 1 for offset 0 or a non-finite offset. Counts `\n`, covering `\r\n`.
|
|
243
136
|
*/
|
|
244
137
|
function lineFromOffset(content, offset) {
|
|
245
138
|
if (typeof content !== "string") return 1;
|