@blamejs/exceptd-skills 0.19.32 → 0.19.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/bin/exceptd.js +896 -2824
- package/data/_indexes/_meta.json +8 -8
- package/data/_indexes/activity-feed.json +2 -2
- package/data/_indexes/catalog-summaries.json +7 -7
- package/data/_indexes/chains.json +60118 -0
- package/data/attack-techniques.json +267 -7
- package/data/cve-catalog.json +9991 -3
- package/data/cwe-catalog.json +109 -2
- package/data/framework-control-gaps.json +578 -3
- package/data/zeroday-lessons.json +8330 -1
- package/lib/auto-discovery.js +56 -286
- package/lib/canonical-eq.js +7 -40
- package/lib/citation-resolve.js +22 -70
- package/lib/collectors/ai-api.js +20 -54
- package/lib/collectors/cicd-pipeline-compromise.js +40 -108
- package/lib/collectors/citation-hygiene.js +72 -210
- package/lib/collectors/containers.js +41 -130
- package/lib/collectors/cred-stores.js +31 -115
- package/lib/collectors/crypto-codebase.js +55 -138
- package/lib/collectors/crypto.js +24 -54
- package/lib/collectors/hardening.js +20 -78
- package/lib/collectors/kernel.js +16 -46
- package/lib/collectors/library-author.js +57 -206
- package/lib/collectors/mcp.js +24 -70
- package/lib/collectors/runtime.js +24 -86
- package/lib/collectors/sbom.js +34 -106
- package/lib/collectors/scan-excludes.js +31 -138
- package/lib/collectors/secrets.js +62 -178
- package/lib/cross-ref-api.js +39 -123
- package/lib/currency-severity.js +8 -27
- package/lib/cve-batch.js +13 -21
- package/lib/cve-cli.js +13 -20
- package/lib/cve-curation.js +72 -239
- package/lib/cve-regression-watcher.js +29 -152
- package/lib/cvss.js +13 -54
- package/lib/doctor-bucketing.js +3 -19
- package/lib/exit-codes.js +10 -42
- package/lib/flag-suggest.js +7 -25
- package/lib/framework-gap.js +35 -114
- package/lib/gap-detectors.js +37 -159
- package/lib/id-validation.js +9 -30
- package/lib/job-queue.js +13 -36
- package/lib/lint-skills.js +64 -232
- package/lib/playbook-runner.js +693 -2095
- package/lib/prefetch.js +100 -376
- package/lib/refresh-external.js +199 -627
- package/lib/refresh-network.js +75 -307
- package/lib/rfc-cli.js +23 -68
- package/lib/scoring.js +77 -145
- package/lib/sign.js +43 -229
- package/lib/source-advisories.js +43 -194
- package/lib/source-ghsa.js +37 -120
- package/lib/source-osv.js +94 -266
- package/lib/ttp-mapper.js +14 -24
- package/lib/upstream-check-cli.js +10 -28
- package/lib/upstream-check.js +19 -44
- package/lib/validate-catalog-meta.js +17 -61
- package/lib/validate-cve-catalog.js +43 -119
- package/lib/validate-indexes.js +25 -76
- package/lib/validate-package.js +16 -62
- package/lib/validate-playbooks.js +69 -275
- package/lib/validate-vendor.js +16 -49
- package/lib/verify.js +56 -286
- package/lib/version-pins.js +5 -34
- package/lib/worker-pool.js +11 -30
- package/lib/xml-tokenizer.js +47 -152
- package/manifest.json +53 -53
- package/orchestrator/dispatcher.js +17 -68
- package/orchestrator/event-bus.js +11 -74
- package/orchestrator/index.js +138 -412
- package/orchestrator/pipeline.js +28 -85
- package/orchestrator/scanner.js +34 -138
- package/orchestrator/scheduler.js +20 -84
- package/package.json +2 -2
- package/sbom.cdx.json +253 -253
- package/scripts/audit-catalog-gaps.js +9 -62
- package/scripts/audit-cross-skill.js +5 -31
- package/scripts/audit-perf.js +6 -16
- package/scripts/backfill-theater-test.js +7 -64
- package/scripts/bootstrap.js +12 -44
- package/scripts/build-indexes.js +40 -154
- package/scripts/builders/activity-feed.js +4 -14
- package/scripts/builders/catalog-summaries.js +3 -10
- package/scripts/builders/currency.js +7 -20
- package/scripts/builders/cwe-chains.js +7 -30
- package/scripts/builders/did-ladders.js +6 -13
- package/scripts/builders/frequency.js +5 -19
- package/scripts/builders/jurisdiction-clocks.js +6 -25
- package/scripts/builders/recipes.js +6 -14
- package/scripts/builders/section-offsets.js +13 -51
- package/scripts/builders/stale-content.js +7 -28
- package/scripts/builders/summary-cards.js +8 -29
- package/scripts/builders/theater-fingerprints.js +12 -27
- package/scripts/builders/token-budget.js +4 -31
- package/scripts/check-agents-md-collectors.js +11 -54
- package/scripts/check-catalog-gap-budget.js +15 -32
- package/scripts/check-changelog-extract.js +18 -48
- package/scripts/check-codebase-patterns-currency.js +6 -22
- package/scripts/check-codebase-patterns.js +50 -143
- package/scripts/check-epss-consistency.js +9 -64
- package/scripts/check-framework-gap-coverage.js +13 -31
- package/scripts/check-manifest-snapshot.js +13 -73
- package/scripts/check-sbom-currency.js +44 -142
- package/scripts/check-test-count.js +15 -52
- package/scripts/check-test-coverage.js +66 -197
- package/scripts/check-test-subjects.js +21 -62
- package/scripts/check-ttp-references.js +14 -38
- package/scripts/check-ttp-upstream.js +8 -40
- package/scripts/check-version-bump.js +9 -61
- package/scripts/check-version-tags.js +20 -121
- package/scripts/predeploy.js +38 -184
- package/scripts/refresh-manifest-snapshot.js +16 -38
- package/scripts/refresh-mitre-atlas.js +3 -8
- package/scripts/refresh-mitre-attack.js +1 -8
- package/scripts/refresh-mitre-d3fend.js +3 -9
- package/scripts/refresh-mitre-ics-attack.js +3 -8
- package/scripts/refresh-reverse-refs.js +27 -94
- package/scripts/refresh-rfc-index.js +2 -10
- package/scripts/refresh-sbom.js +31 -161
- package/scripts/refresh-upstream-catalogs.js +40 -137
- package/scripts/release.js +69 -232
- package/scripts/run-e2e-scenarios.js +24 -71
- package/scripts/sync-manifest-metadata.js +10 -34
- package/scripts/sync-package-description.js +8 -17
- package/scripts/validate-vendor-online.js +13 -44
- package/scripts/verify-shipped-tarball.js +35 -140
|
@@ -1,18 +1,11 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* operationalize each layer and the D3FEND countermeasures backing each
|
|
10
|
-
* layer.
|
|
11
|
-
*
|
|
12
|
-
* Curated content, validated against the manifest: every referenced skill
|
|
13
|
-
* and D3FEND id must exist in their respective catalogs or the build
|
|
14
|
-
* fails. This is the only place where the project's DiD knowledge is
|
|
15
|
-
* laid out flat across attack classes.
|
|
3
|
+
* Builds `data/_indexes/did-ladders.json`: one defense-in-depth ladder per
|
|
4
|
+
* high-frequency attack class, each layer naming the skill that operationalizes
|
|
5
|
+
* it and the D3FEND countermeasures behind it. The ladders are curated here —
|
|
6
|
+
* the only place the project's DiD knowledge is laid out flat across attack
|
|
7
|
+
* classes — and a referenced skill or D3FEND id absent from its catalog fails
|
|
8
|
+
* the build.
|
|
16
9
|
*/
|
|
17
10
|
|
|
18
11
|
const LADDERS = [
|
|
@@ -1,21 +1,8 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
* catalog (CWE, ATLAS, ATT&CK, D3FEND, framework gaps, RFC, DLP), counts
|
|
7
|
-
* how many skills cite each entry. Surfaces which entries are load-bearing
|
|
8
|
-
* (cited by many) vs. orphan-adjacent (cited by ≤1).
|
|
9
|
-
*
|
|
10
|
-
* Per-field shape:
|
|
11
|
-
* {
|
|
12
|
-
* <entry_id>: { count, skills: [name, ...] }
|
|
13
|
-
* }
|
|
14
|
-
*
|
|
15
|
-
* Plus rollups:
|
|
16
|
-
* - top_cited: top 10 entries per field
|
|
17
|
-
* - orphan_adjacent: entries cited by exactly one skill
|
|
18
|
-
* - uncited: catalog entries with zero skill citations (flagged for review)
|
|
3
|
+
* Builds `data/_indexes/frequency.json`: how many skills cite each entry of
|
|
4
|
+
* each catalog, with rollups that separate the load-bearing entries from the
|
|
5
|
+
* ones a single skill cites and the ones nothing cites at all.
|
|
19
6
|
*/
|
|
20
7
|
|
|
21
8
|
function buildFrequency({ skills, catalogs }) {
|
|
@@ -51,7 +38,6 @@ function buildFrequency({ skills, catalogs }) {
|
|
|
51
38
|
.sort();
|
|
52
39
|
}
|
|
53
40
|
|
|
54
|
-
// Uncited: catalog has an entry but zero skill cites it.
|
|
55
41
|
const uncited = {};
|
|
56
42
|
const catalogFieldMap = {
|
|
57
43
|
cwe_refs: catalogs.cwe,
|
|
@@ -67,8 +53,8 @@ function buildFrequency({ skills, catalogs }) {
|
|
|
67
53
|
uncited[field] = inCatalog.filter((id) => !counts[field][id]).sort();
|
|
68
54
|
}
|
|
69
55
|
|
|
70
|
-
// attack_refs has no catalog file
|
|
71
|
-
//
|
|
56
|
+
// attack_refs is absent from catalogFieldMap: it has no catalog file of its
|
|
57
|
+
// own, so it gets counts but no uncited table.
|
|
72
58
|
|
|
73
59
|
const topCited = {};
|
|
74
60
|
for (const f of fields) topCited[f] = topN(f);
|
|
@@ -2,31 +2,12 @@
|
|
|
2
2
|
/**
|
|
3
3
|
* scripts/builders/jurisdiction-clocks.js
|
|
4
4
|
*
|
|
5
|
-
* Builds `data/_indexes/jurisdiction-clocks.json` — the normalized
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* Obligation types covered:
|
|
12
|
-
* - breach_notification (hours from awareness)
|
|
13
|
-
* - patch_sla (hours from disclosure for Critical/High)
|
|
14
|
-
* - incident_reporting (regulator + clock + trigger)
|
|
15
|
-
*
|
|
16
|
-
* Per-jurisdiction shape:
|
|
17
|
-
* {
|
|
18
|
-
* jurisdiction_name:
|
|
19
|
-
* frameworks: {
|
|
20
|
-
* <fwName>: {
|
|
21
|
-
* authority,
|
|
22
|
-
* breach_notification: { hours, trigger, stages?, source }
|
|
23
|
-
* patch_sla: { hours, note?, source }
|
|
24
|
-
* incident_reporting: { hours, trigger, source }
|
|
25
|
-
* }
|
|
26
|
-
* }
|
|
27
|
-
* fastest_breach_notification: { hours, framework } // null when none specified
|
|
28
|
-
* fastest_patch_sla: { hours, framework }
|
|
29
|
-
* }
|
|
5
|
+
* Builds `data/_indexes/jurisdiction-clocks.json` — the normalized jurisdiction
|
|
6
|
+
* × obligation × clock matrix, so a consumer asking "what is the breach-
|
|
7
|
+
* notification clock in jurisdiction X?" need not scan
|
|
8
|
+
* `data/global-frameworks.json` for `notification_sla` on each framework entry.
|
|
9
|
+
* All times are in hours; a `fastest_*` slot is null when no framework in the
|
|
10
|
+
* jurisdiction specifies that clock.
|
|
30
11
|
*/
|
|
31
12
|
|
|
32
13
|
function buildJurisdictionClocks({ globalFrameworks }) {
|
|
@@ -1,16 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* to invoke, in order, with a brief rationale per step. Saves the
|
|
8
|
-
* researcher skill from re-deriving these on every request.
|
|
9
|
-
*
|
|
10
|
-
* Recipes are content (not derived), so this builder is mostly a
|
|
11
|
-
* declarative table. The build step validates that every referenced
|
|
12
|
-
* skill exists in the manifest, so a renamed/deleted skill surfaces as a
|
|
13
|
-
* build error.
|
|
3
|
+
* Builds `data/_indexes/recipes.json` — curated skill chains for the common
|
|
4
|
+
* operator cases, so the researcher skill does not re-derive them per request.
|
|
5
|
+
* The table below is content, not derived; the build validates every referenced
|
|
6
|
+
* skill against the manifest, so a rename or deletion fails the build.
|
|
14
7
|
*/
|
|
15
8
|
|
|
16
9
|
const RECIPES = [
|
|
@@ -138,9 +131,8 @@ function buildRecipes({ skills }) {
|
|
|
138
131
|
throw new Error("recipes.js: " + errors.join("; "));
|
|
139
132
|
}
|
|
140
133
|
|
|
141
|
-
//
|
|
142
|
-
//
|
|
143
|
-
// join to token-budget.json. The skill_count is cheap to include here.
|
|
134
|
+
// No token-budget hints here: that builder may run after this one, so
|
|
135
|
+
// consumers join to token-budget.json themselves.
|
|
144
136
|
const out = {
|
|
145
137
|
_meta: {
|
|
146
138
|
schema_version: "1.0.0",
|
|
@@ -1,42 +1,14 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* slice a single section (e.g. "Compliance Theater Check") from disk
|
|
8
|
-
* without parsing the full skill file.
|
|
9
|
-
*
|
|
10
|
-
* Per-skill shape:
|
|
11
|
-
* {
|
|
12
|
-
* path: "skills/<name>/skill.md",
|
|
13
|
-
* total_bytes: n,
|
|
14
|
-
* total_lines: n,
|
|
15
|
-
* frontmatter: { byte_start, byte_end, line_start, line_end },
|
|
16
|
-
* sections: [
|
|
17
|
-
* {
|
|
18
|
-
* name: raw H2 text (e.g. "Threat Context (mid-2026)")
|
|
19
|
-
* normalized_name: collapsed for lookup ("threat-context")
|
|
20
|
-
* line: 1-based line number of the "## …" header
|
|
21
|
-
* byte_start: byte offset of the "## " character
|
|
22
|
-
* byte_end: byte offset where the next H2 begins (or EOF)
|
|
23
|
-
* bytes: byte_end - byte_start
|
|
24
|
-
* h3_count: number of "### " headers contained
|
|
25
|
-
* },
|
|
26
|
-
* ...
|
|
27
|
-
* ]
|
|
28
|
-
* }
|
|
29
|
-
*
|
|
30
|
-
* The normalized_name strips parenthetical qualifiers and common phrasings
|
|
31
|
-
* so consumers can request a canonical section name without caring about
|
|
32
|
-
* formatting drift.
|
|
3
|
+
* Builds `data/_indexes/section-offsets.json`: per skill, the byte and line
|
|
4
|
+
* offsets of every H2 section in the body, so a consumer can slice one section
|
|
5
|
+
* off disk without parsing the whole file. normalized_name collapses
|
|
6
|
+
* parenthetical qualifiers and phrasing variants onto a canonical name.
|
|
33
7
|
*/
|
|
34
8
|
|
|
35
9
|
const fs = require("fs");
|
|
36
10
|
const path = require("path");
|
|
37
11
|
|
|
38
|
-
// Recognized canonical section anchors. Multiple raw H2 phrasings map to one
|
|
39
|
-
// normalized name — see grep survey of skills/* for the variant phrasings.
|
|
40
12
|
const NORMALIZERS = [
|
|
41
13
|
[/threat\s*context/i, "threat-context"],
|
|
42
14
|
[/framework\s*lag\s*declaration/i, "framework-lag-declaration"],
|
|
@@ -56,7 +28,6 @@ function normalize(headerText) {
|
|
|
56
28
|
for (const [re, canonical] of NORMALIZERS) {
|
|
57
29
|
if (re.test(stripped)) return canonical;
|
|
58
30
|
}
|
|
59
|
-
// Fall back: slug.
|
|
60
31
|
return stripped
|
|
61
32
|
.toLowerCase()
|
|
62
33
|
.replace(/[^a-z0-9]+/g, "-")
|
|
@@ -68,14 +39,10 @@ function buildOne(absPath, relPath) {
|
|
|
68
39
|
const totalBytes = buf.length;
|
|
69
40
|
const text = buf.toString("utf8");
|
|
70
41
|
const lines = text.split(/\r?\n/);
|
|
71
|
-
//
|
|
72
|
-
//
|
|
73
|
-
//
|
|
74
|
-
//
|
|
75
|
-
// offsets stay correct (the old fixed-constant approach undercounted by 1
|
|
76
|
-
// byte per line and silently misaligned every token-budget slice). The split
|
|
77
|
-
// above discards terminator bytes, so the width can only be recovered by
|
|
78
|
-
// reading the terminators off the raw text — done here.
|
|
42
|
+
// Line-start byte offsets are measured off the real terminator bytes: a CRLF
|
|
43
|
+
// terminator is 2 bytes, and assuming a fixed 1-byte newline misaligns every
|
|
44
|
+
// offset in a CRLF body. The split above discards the terminators, so their
|
|
45
|
+
// width can only be recovered from the raw text.
|
|
79
46
|
const lineByteOffsets = [0];
|
|
80
47
|
const eolRe = /\r?\n/g;
|
|
81
48
|
let m;
|
|
@@ -103,9 +70,8 @@ function buildOne(absPath, relPath) {
|
|
|
103
70
|
}
|
|
104
71
|
: null;
|
|
105
72
|
|
|
106
|
-
// H2 headers
|
|
107
|
-
//
|
|
108
|
-
// are not real sections.
|
|
73
|
+
// H2 headers outside fenced code blocks only — skill bodies carry "## Foo"
|
|
74
|
+
// lines inside ```...``` blocks as output templates, which are not sections.
|
|
109
75
|
const h2 = [];
|
|
110
76
|
let inFence = false;
|
|
111
77
|
for (let i = 0; i < lines.length; i++) {
|
|
@@ -124,10 +90,8 @@ function buildOne(absPath, relPath) {
|
|
|
124
90
|
const next = h2[j + 1];
|
|
125
91
|
const startByte = lineByteOffsets[cur.idx];
|
|
126
92
|
const endByte = next ? lineByteOffsets[next.idx] : totalBytes;
|
|
127
|
-
//
|
|
128
|
-
//
|
|
129
|
-
// outside any fence, so fence state always begins false here. "### Foo"
|
|
130
|
-
// lines inside ```...``` output templates are not real sub-sections.
|
|
93
|
+
// Fence-aware like the H2 loop: a section starts and ends on an H2, both
|
|
94
|
+
// outside any fence, so fence state always begins false here.
|
|
131
95
|
const endIdx = next ? next.idx : lines.length;
|
|
132
96
|
let h3Count = 0;
|
|
133
97
|
let h3InFence = false;
|
|
@@ -173,7 +137,5 @@ function buildSectionOffsets({ root, skills }) {
|
|
|
173
137
|
};
|
|
174
138
|
}
|
|
175
139
|
|
|
176
|
-
// buildOne is exported for
|
|
177
|
-
// (a CRLF body must still produce byte_start values that point at the real
|
|
178
|
-
// "## " byte in the raw file).
|
|
140
|
+
// buildOne is exported for the CRLF byte-offset regression test.
|
|
179
141
|
module.exports = { buildSectionOffsets, buildOne };
|
|
@@ -1,24 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* persisted as a JSON artifact so CI / dashboards / downstream tools can
|
|
8
|
-
* read the same view without invoking the script.
|
|
9
|
-
*
|
|
10
|
-
* Checks performed here (subset of audit-cross-skill that's relevant to
|
|
11
|
-
* the index layer, deterministic across reruns):
|
|
12
|
-
*
|
|
13
|
-
* - Skill bodies referencing renamed-skill tokens (e.g. age-gates-minor-*)
|
|
14
|
-
* - README badge counts vs. live counts
|
|
15
|
-
* - "Researcher routes to N skills" claim vs. live count
|
|
16
|
-
* - Skills with last_threat_review older than 180 days from
|
|
17
|
-
* manifest.threat_review_date (gives a stale-content snapshot)
|
|
18
|
-
* - Catalog _meta.last_verified entries older than freshness_policy.stale_after_days
|
|
19
|
-
* - Forward_watch items mentioning dates that have already passed
|
|
20
|
-
*
|
|
21
|
-
* Each finding is { severity, category, artifact, detail }.
|
|
3
|
+
* Builds `data/_indexes/stale-content.json` — the subset of the
|
|
4
|
+
* audit-cross-skill checks that is deterministic across reruns, persisted so CI
|
|
5
|
+
* and downstream tools read the same view without invoking that script. Each
|
|
6
|
+
* finding is { severity, category, artifact, detail }.
|
|
22
7
|
*/
|
|
23
8
|
|
|
24
9
|
const fs = require("fs");
|
|
@@ -45,7 +30,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
|
|
|
45
30
|
const findings = [];
|
|
46
31
|
const refDate = new Date((manifest.threat_review_date || "2026-05-01") + "T00:00:00Z");
|
|
47
32
|
|
|
48
|
-
// 1. Stale-renamed-skill tokens
|
|
49
33
|
for (const s of skills) {
|
|
50
34
|
const body = fs.readFileSync(path.join(root, s.path), "utf8");
|
|
51
35
|
for (const tok of RENAMED_SKILL_TOKENS) {
|
|
@@ -62,7 +46,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
|
|
|
62
46
|
}
|
|
63
47
|
}
|
|
64
48
|
|
|
65
|
-
// 2. README badge counts vs. live counts
|
|
66
49
|
const readmePath = path.join(root, "README.md");
|
|
67
50
|
if (fs.existsSync(readmePath)) {
|
|
68
51
|
const readme = fs.readFileSync(readmePath, "utf8");
|
|
@@ -71,10 +54,9 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
|
|
|
71
54
|
const liveJurisdictions = (() => {
|
|
72
55
|
try {
|
|
73
56
|
const gf = JSON.parse(fs.readFileSync(path.join(root, "data/global-frameworks.json"), "utf8"));
|
|
74
|
-
//
|
|
75
|
-
//
|
|
76
|
-
//
|
|
77
|
-
// badge_drift finding against the README's (correct) 35.
|
|
57
|
+
// Non-underscore keys, GLOBAL INCLUDED — the canonical jurisdiction
|
|
58
|
+
// count the README badge and catalog-summaries use. Excluding GLOBAL
|
|
59
|
+
// undercounts by one and emits a false badge_drift.
|
|
78
60
|
return Object.keys(gf).filter((k) => !k.startsWith("_")).length;
|
|
79
61
|
} catch {
|
|
80
62
|
return null;
|
|
@@ -98,7 +80,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
|
|
|
98
80
|
}
|
|
99
81
|
}
|
|
100
82
|
|
|
101
|
-
// 3. Researcher dispatch count claim
|
|
102
83
|
const researcherPath = path.join(root, "skills/researcher/skill.md");
|
|
103
84
|
if (fs.existsSync(researcherPath)) {
|
|
104
85
|
const r = fs.readFileSync(researcherPath, "utf8");
|
|
@@ -117,7 +98,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
|
|
|
117
98
|
}
|
|
118
99
|
}
|
|
119
100
|
|
|
120
|
-
// 4. Skills with > 180 days since review (against reference date)
|
|
121
101
|
for (const s of skills) {
|
|
122
102
|
if (!s.last_threat_review) continue;
|
|
123
103
|
const ageDays = Math.floor(
|
|
@@ -133,7 +113,6 @@ function buildStaleContent({ root, manifest, skills, catalogFiles }) {
|
|
|
133
113
|
}
|
|
134
114
|
}
|
|
135
115
|
|
|
136
|
-
// 5. Catalog last_verified entries older than freshness_policy.stale_after_days
|
|
137
116
|
for (const rel of catalogFiles) {
|
|
138
117
|
const abs = path.join(root, rel);
|
|
139
118
|
try {
|
|
@@ -1,24 +1,8 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
* abstract that downstream AI consumers (researcher dispatch in particular)
|
|
7
|
-
* can render without loading the full skill body.
|
|
8
|
-
*
|
|
9
|
-
* Card shape per skill:
|
|
10
|
-
* {
|
|
11
|
-
* description: manifest description
|
|
12
|
-
* threat_context_excerpt: first paragraph of Threat Context section
|
|
13
|
-
* produces: first paragraph of Output Format section (if present)
|
|
14
|
-
* key_xrefs: {
|
|
15
|
-
* cwe_refs, d3fend_refs, framework_gaps, atlas_refs,
|
|
16
|
-
* attack_refs, rfc_refs, dlp_refs
|
|
17
|
-
* }
|
|
18
|
-
* trigger_count, atlas_count, attack_count, framework_gap_count,
|
|
19
|
-
* last_threat_review, path,
|
|
20
|
-
* handoff_targets: skills referenced from this skill's Hand-Off section
|
|
21
|
-
* }
|
|
3
|
+
* Builds `data/_indexes/summary-cards.json`: a compact per-skill abstract that
|
|
4
|
+
* downstream AI consumers — researcher dispatch in particular — render without
|
|
5
|
+
* loading the full skill body.
|
|
22
6
|
*/
|
|
23
7
|
|
|
24
8
|
const fs = require("fs");
|
|
@@ -47,14 +31,11 @@ function locateHeader(lines, headerRegex) {
|
|
|
47
31
|
}
|
|
48
32
|
|
|
49
33
|
function firstParagraphAfterHeader(body, headerRegex) {
|
|
50
|
-
//
|
|
51
|
-
//
|
|
52
|
-
// horizontal rules / table separators that often sit at the top of a
|
|
53
|
-
// section. Real H2 means outside of fenced code blocks.
|
|
34
|
+
// The first prose paragraph beneath the matching H2, skipping the H3/H4,
|
|
35
|
+
// bold-prefix metadata, rules and table separators that often lead a section.
|
|
54
36
|
const lines = body.split(/\r?\n/);
|
|
55
37
|
const hdrIdx = locateHeader(lines, headerRegex);
|
|
56
38
|
if (hdrIdx < 0) return null;
|
|
57
|
-
// Find the next real H2 as the section boundary.
|
|
58
39
|
const allH2 = findRealH2Indices(lines);
|
|
59
40
|
const nextH2 = allH2.find((i) => i > hdrIdx);
|
|
60
41
|
const sectionEnd = nextH2 != null ? nextH2 : lines.length;
|
|
@@ -106,11 +87,9 @@ function firstChunkAfterHeader(body, headerRegex, maxChars = 600) {
|
|
|
106
87
|
}
|
|
107
88
|
|
|
108
89
|
function handoffTargets(body, allSkillNames, selfName) {
|
|
109
|
-
//
|
|
110
|
-
// target.
|
|
111
|
-
//
|
|
112
|
-
// detection and the section boundary are both fence-aware (a `## ` line
|
|
113
|
-
// inside a ```...``` block is not a real H2).
|
|
90
|
+
// Only the Hand-Off section counts, and a backtick-quoted skill name is a
|
|
91
|
+
// target. The scan is bounded to [header, next real H2) so a later section is
|
|
92
|
+
// not mis-attributed, and both bounds are fence-aware.
|
|
114
93
|
const lines = body.split(/\r?\n/);
|
|
115
94
|
const h2 = findRealH2Indices(lines);
|
|
116
95
|
const handoffIdx = h2.find((i) => /^## Hand-?Off/.test(lines[i]));
|
|
@@ -1,28 +1,16 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* - the claim (what auditors hear)
|
|
8
|
-
* - the audit evidence (what passes the audit)
|
|
9
|
-
* - the reality (why it's theater)
|
|
10
|
-
* - the detection test (operational steps)
|
|
11
|
-
* - the controls it spans (NIST 800-53 / ISO 27001 / PCI / SOC 2)
|
|
12
|
-
* - the evidence CVE / campaign tying the pattern to the real world
|
|
13
|
-
*
|
|
14
|
-
* Extracted from `skills/compliance-theater/skill.md`. The compliance-theater
|
|
15
|
-
* skill is the source-of-truth — this index just structures the pattern
|
|
16
|
-
* library so downstream consumers (audit defense, framework-gap-analysis)
|
|
17
|
-
* can join on control IDs without re-parsing the markdown.
|
|
3
|
+
* Builds `data/_indexes/theater-fingerprints.json` from
|
|
4
|
+
* `skills/compliance-theater/skill.md`, which stays the source of truth — the
|
|
5
|
+
* index only structures the pattern library so consumers can join on control
|
|
6
|
+
* IDs without re-parsing the markdown.
|
|
18
7
|
*/
|
|
19
8
|
|
|
20
9
|
const fs = require("fs");
|
|
21
10
|
const path = require("path");
|
|
22
11
|
|
|
23
|
-
//
|
|
24
|
-
//
|
|
25
|
-
// with skills/compliance-theater/skill.md.
|
|
12
|
+
// Each pattern → the controls it spans, hand-curated from the skill's Framework
|
|
13
|
+
// Lag Declaration table; keep in lockstep with that table.
|
|
26
14
|
const PATTERN_CONTROL_MAP = {
|
|
27
15
|
1: {
|
|
28
16
|
pattern_name: "Patch Management Theater",
|
|
@@ -117,10 +105,9 @@ const PATTERN_CONTROL_MAP = {
|
|
|
117
105
|
};
|
|
118
106
|
|
|
119
107
|
function extractPatternBodyFromSkill(skillBody, patternNumber) {
|
|
120
|
-
//
|
|
121
|
-
//
|
|
122
|
-
//
|
|
123
|
-
// line would match the `## ` prefix regex once its leading `#` is sliced.
|
|
108
|
+
// Captures from "### Pattern N:" to the next "### Pattern N+1:" or the next
|
|
109
|
+
// H2. The H2 scan starts past the header line: `### Pattern N:` itself
|
|
110
|
+
// matches the `^## ` prefix once its leading `#` is sliced off.
|
|
124
111
|
const startRe = new RegExp(`^### Pattern ${patternNumber}:`, "m");
|
|
125
112
|
const startMatch = skillBody.match(startRe);
|
|
126
113
|
if (!startMatch) return null;
|
|
@@ -142,8 +129,7 @@ function extractPatternBodyFromSkill(skillBody, patternNumber) {
|
|
|
142
129
|
}
|
|
143
130
|
|
|
144
131
|
function pullField(body, label) {
|
|
145
|
-
//
|
|
146
|
-
// after the label until the next "**" or blank line.
|
|
132
|
+
// Patterns write fields as "**Label:** ..."; capture to the next "**" or blank line.
|
|
147
133
|
const re = new RegExp(`\\*\\*${label.replace(/[-/\\^$*+?.()|[\\]{}]/g, "\\$&")}:?\\*\\*\\s*([\\s\\S]*?)(?=\\n\\n|\\n\\*\\*|$)`);
|
|
148
134
|
const m = body.match(re);
|
|
149
135
|
return m ? m[1].trim() : null;
|
|
@@ -173,9 +159,8 @@ function buildTheaterFingerprints({ root }) {
|
|
|
173
159
|
};
|
|
174
160
|
}
|
|
175
161
|
|
|
176
|
-
// Inverted index
|
|
177
|
-
//
|
|
178
|
-
// seven patterns.
|
|
162
|
+
// Inverted index, framework::control_id → patterns, so a consumer can ask
|
|
163
|
+
// whether a control is implicated without scanning every pattern.
|
|
179
164
|
const byControl = {};
|
|
180
165
|
for (const [pid, p] of Object.entries(out)) {
|
|
181
166
|
for (const c of p.controls) {
|
|
@@ -1,36 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* approximation is documented as such so consumers know to recompute with
|
|
8
|
-
* their own tokenizer if precision matters.
|
|
9
|
-
*
|
|
10
|
-
* Heuristic: 1 token ≈ 4 characters for English prose mixed with technical
|
|
11
|
-
* tokens (matches the well-known OpenAI rule-of-thumb). This is an upper
|
|
12
|
-
* bound for Claude (Anthropic's tokenizer is more efficient on common
|
|
13
|
-
* prose) but is good enough for context-budget planning where consumers
|
|
14
|
-
* just need to know "is this load 5K or 50K tokens".
|
|
15
|
-
*
|
|
16
|
-
* Per-skill shape:
|
|
17
|
-
* {
|
|
18
|
-
* path: skill file path
|
|
19
|
-
* bytes: total file bytes
|
|
20
|
-
* chars: total character count
|
|
21
|
-
* lines: line count
|
|
22
|
-
* approx_tokens: chars / 4 (integer)
|
|
23
|
-
* approx_chars_per_token: 4
|
|
24
|
-
* sections: {
|
|
25
|
-
* <normalized_section_name>: { bytes, approx_tokens }
|
|
26
|
-
* }
|
|
27
|
-
* }
|
|
28
|
-
*
|
|
29
|
-
* Corpus totals live under the top-level `_meta` block:
|
|
30
|
-
* {
|
|
31
|
-
* schema_version, tokenizer_note, approx_chars_per_token,
|
|
32
|
-
* total_chars, total_approx_tokens, skill_count
|
|
33
|
-
* }
|
|
3
|
+
* Builds `data/_indexes/token-budget.json`: per-skill token counts from a
|
|
4
|
+
* 1-token ≈ 4-characters density heuristic, with no tokenizer dependency. The
|
|
5
|
+
* result is an upper bound for context-budget planning, never a precise count —
|
|
6
|
+
* the caveat travels with the data in `_meta.tokenizer_note`.
|
|
34
7
|
*/
|
|
35
8
|
|
|
36
9
|
const fs = require("fs");
|
|
@@ -2,24 +2,9 @@
|
|
|
2
2
|
"use strict";
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* ship today" paragraph stays in sync with the actual contents of
|
|
9
|
-
* lib/collectors/. Drift is silent today - AGENTS.md gets bumped by
|
|
10
|
-
* hand each release; a missed bump produces inaccurate count + stale
|
|
11
|
-
* enumeration that downstream AI consumers parse.
|
|
12
|
-
*
|
|
13
|
-
* Checks:
|
|
14
|
-
* 1. The numeric count word in the paragraph (Eleven / Twelve /
|
|
15
|
-
* Thirteen / ...) matches the actual count of
|
|
16
|
-
* lib/collectors/*.js modules.
|
|
17
|
-
* 2. Every collector named in the parenthesized list exists at
|
|
18
|
-
* lib/collectors/<name>.js.
|
|
19
|
-
* 3. Every lib/collectors/<name>.js module appears in the
|
|
20
|
-
* parenthesized list.
|
|
21
|
-
*
|
|
22
|
-
* Exit codes: 0 ok, 1 drift, 2 parse error.
|
|
5
|
+
* Predeploy gate: AGENTS.md's "<N> reference collectors ship today" paragraph
|
|
6
|
+
* must agree with lib/collectors/ — the count word and the enumeration, both
|
|
7
|
+
* directions. Exit codes: 0 ok, 1 drift, 2 parse error.
|
|
23
8
|
*/
|
|
24
9
|
|
|
25
10
|
const fs = require("node:fs");
|
|
@@ -28,31 +13,20 @@ const path = require("node:path");
|
|
|
28
13
|
const ROOT = path.join(__dirname, "..");
|
|
29
14
|
const AGENTS = path.join(ROOT, "AGENTS.md");
|
|
30
15
|
const REAL_COLLECTOR_DIR = path.join(ROOT, "lib", "collectors");
|
|
31
|
-
// EXCEPTD_COLLECTOR_DIR
|
|
32
|
-
//
|
|
33
|
-
// without writing into the real lib/collectors/). It is honored ONLY when the
|
|
34
|
-
// explicit test-only switch EXCEPTD_COLLECTOR_DIR_TESTONLY=1 is also set, so a
|
|
35
|
-
// stray env var in CI or a developer shell can never point the release gate at
|
|
36
|
-
// an alternate directory and let missing/broken real collectors pass.
|
|
16
|
+
// EXCEPTD_COLLECTOR_DIR is honored ONLY alongside EXCEPTD_COLLECTOR_DIR_TESTONLY=1,
|
|
17
|
+
// so a stray env var cannot aim the release gate at another directory.
|
|
37
18
|
const COLLECTOR_DIR =
|
|
38
19
|
process.env.EXCEPTD_COLLECTOR_DIR_TESTONLY === "1" && process.env.EXCEPTD_COLLECTOR_DIR
|
|
39
20
|
? path.resolve(process.env.EXCEPTD_COLLECTOR_DIR)
|
|
40
21
|
: REAL_COLLECTOR_DIR;
|
|
41
22
|
|
|
42
|
-
// A
|
|
43
|
-
//
|
|
44
|
-
// underscore-prefixed file. So a `__`-prefixed file is never a collector:
|
|
45
|
-
// it is test scaffolding or a stray artifact. classifyCollectors SKIPS it so a
|
|
46
|
-
// leaked fixture cannot poison the count/enumeration/load-error scan; the gate
|
|
47
|
-
// separately FORBIDS it (see findReservedFixtures) so a leaked fixture cannot
|
|
48
|
-
// silently ship in the wholesale-published lib/ tree either.
|
|
23
|
+
// A shipped collector is named [a-z0-9-]+.js, so a `__`-prefixed file is test
|
|
24
|
+
// scaffolding: classifyCollectors skips it, findReservedFixtures forbids it in lib/.
|
|
49
25
|
function isReservedFixture(f) {
|
|
50
26
|
return f.startsWith("__");
|
|
51
27
|
}
|
|
52
28
|
|
|
53
|
-
// Reserved-prefix .js files present in `dir
|
|
54
|
-
// must never ship; the gate fails hard if any are found in the directory it
|
|
55
|
-
// validates.
|
|
29
|
+
// Reserved-prefix .js files present in `dir`; an unreadable dir yields [].
|
|
56
30
|
function findReservedFixtures(dir) {
|
|
57
31
|
try {
|
|
58
32
|
return fs.readdirSync(dir).filter((f) => f.endsWith(".js") && isReservedFixture(f));
|
|
@@ -61,10 +35,8 @@ function findReservedFixtures(dir) {
|
|
|
61
35
|
}
|
|
62
36
|
}
|
|
63
37
|
|
|
64
|
-
// Classify every <dir>/*.js
|
|
65
|
-
//
|
|
66
|
-
// and load-errors (require throws — surfaced, never silently dropped).
|
|
67
|
-
// Exported so the test suite can drive it against a tempdir.
|
|
38
|
+
// Classify every <dir>/*.js: a collector requires cleanly and exports collect();
|
|
39
|
+
// a require that throws becomes a load error rather than a silent omission.
|
|
68
40
|
function classifyCollectors(dir) {
|
|
69
41
|
const jsFiles = fs.readdirSync(dir)
|
|
70
42
|
.filter((f) => f.endsWith(".js") && !isReservedFixture(f))
|
|
@@ -104,10 +76,7 @@ function ok(msg) {
|
|
|
104
76
|
}
|
|
105
77
|
|
|
106
78
|
function main() {
|
|
107
|
-
//
|
|
108
|
-
// `__`-prefixed file in the collectors dir is stray test scaffolding;
|
|
109
|
-
// because lib/ is published wholesale, a leaked one would otherwise ship.
|
|
110
|
-
// Fail hard, naming the file, so the release path deletes it.
|
|
79
|
+
// lib/ is published wholesale, so a stray fixture would ship.
|
|
111
80
|
const stray = findReservedFixtures(COLLECTOR_DIR);
|
|
112
81
|
if (stray.length > 0) {
|
|
113
82
|
console.error(
|
|
@@ -126,18 +95,6 @@ function main() {
|
|
|
126
95
|
return;
|
|
127
96
|
}
|
|
128
97
|
|
|
129
|
-
// Classify every lib/collectors/*.js into exactly one of three buckets:
|
|
130
|
-
// - collector: require() succeeds AND exports a collect() function
|
|
131
|
-
// (counted; must appear in the AGENTS.md enumeration).
|
|
132
|
-
// - helper: require() succeeds but exports no collect() function
|
|
133
|
-
// (e.g. scan-excludes.js, the directory-walk exclusion
|
|
134
|
-
// policy) — legitimately excluded from the count.
|
|
135
|
-
// - load-error: require() THROWS (syntax error, bad top-level require,
|
|
136
|
-
// init-time exception). A broken collector must NOT be
|
|
137
|
-
// silently dropped: doing so excludes it from BOTH the
|
|
138
|
-
// count and the enumeration cross-check, so a file that
|
|
139
|
-
// still ships in the tarball passes the gate undetected.
|
|
140
|
-
// Surface it as a parse error (exit 2) naming the file.
|
|
141
98
|
let collectorFiles, loadErrors;
|
|
142
99
|
try {
|
|
143
100
|
({ collectorFiles, loadErrors } = classifyCollectors(COLLECTOR_DIR));
|