@blamejs/exceptd-skills 0.19.33 → 0.19.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/bin/exceptd.js +896 -2824
- package/data/_indexes/_meta.json +2 -2
- package/lib/auto-discovery.js +56 -286
- package/lib/canonical-eq.js +7 -40
- package/lib/citation-resolve.js +22 -70
- package/lib/collectors/ai-api.js +20 -54
- package/lib/collectors/cicd-pipeline-compromise.js +40 -108
- package/lib/collectors/citation-hygiene.js +72 -210
- package/lib/collectors/containers.js +41 -130
- package/lib/collectors/cred-stores.js +31 -115
- package/lib/collectors/crypto-codebase.js +55 -138
- package/lib/collectors/crypto.js +24 -54
- package/lib/collectors/hardening.js +20 -78
- package/lib/collectors/kernel.js +16 -46
- package/lib/collectors/library-author.js +57 -206
- package/lib/collectors/mcp.js +24 -70
- package/lib/collectors/runtime.js +24 -86
- package/lib/collectors/sbom.js +34 -106
- package/lib/collectors/scan-excludes.js +31 -138
- package/lib/collectors/secrets.js +62 -178
- package/lib/cross-ref-api.js +39 -123
- package/lib/currency-severity.js +8 -27
- package/lib/cve-batch.js +13 -21
- package/lib/cve-cli.js +13 -20
- package/lib/cve-curation.js +72 -239
- package/lib/cve-regression-watcher.js +29 -152
- package/lib/cvss.js +13 -54
- package/lib/doctor-bucketing.js +3 -19
- package/lib/exit-codes.js +10 -42
- package/lib/flag-suggest.js +7 -25
- package/lib/framework-gap.js +35 -114
- package/lib/gap-detectors.js +37 -159
- package/lib/id-validation.js +9 -30
- package/lib/job-queue.js +13 -36
- package/lib/lint-skills.js +64 -232
- package/lib/playbook-runner.js +693 -2095
- package/lib/prefetch.js +100 -376
- package/lib/refresh-external.js +199 -627
- package/lib/refresh-network.js +75 -307
- package/lib/rfc-cli.js +23 -68
- package/lib/scoring.js +77 -145
- package/lib/sign.js +43 -229
- package/lib/source-advisories.js +43 -194
- package/lib/source-ghsa.js +37 -120
- package/lib/source-osv.js +94 -266
- package/lib/ttp-mapper.js +14 -24
- package/lib/upstream-check-cli.js +10 -28
- package/lib/upstream-check.js +19 -44
- package/lib/validate-catalog-meta.js +17 -61
- package/lib/validate-cve-catalog.js +43 -119
- package/lib/validate-indexes.js +25 -76
- package/lib/validate-package.js +16 -62
- package/lib/validate-playbooks.js +69 -275
- package/lib/validate-vendor.js +16 -49
- package/lib/verify.js +56 -286
- package/lib/version-pins.js +5 -34
- package/lib/worker-pool.js +11 -30
- package/lib/xml-tokenizer.js +47 -152
- package/manifest.json +53 -53
- package/orchestrator/dispatcher.js +17 -68
- package/orchestrator/event-bus.js +11 -74
- package/orchestrator/index.js +138 -412
- package/orchestrator/pipeline.js +28 -85
- package/orchestrator/scanner.js +34 -138
- package/orchestrator/scheduler.js +20 -84
- package/package.json +1 -1
- package/sbom.cdx.json +241 -241
- package/scripts/audit-catalog-gaps.js +9 -62
- package/scripts/audit-cross-skill.js +5 -31
- package/scripts/audit-perf.js +6 -16
- package/scripts/backfill-theater-test.js +7 -64
- package/scripts/bootstrap.js +12 -44
- package/scripts/build-indexes.js +40 -154
- package/scripts/builders/activity-feed.js +4 -14
- package/scripts/builders/catalog-summaries.js +3 -10
- package/scripts/builders/currency.js +7 -20
- package/scripts/builders/cwe-chains.js +7 -30
- package/scripts/builders/did-ladders.js +6 -13
- package/scripts/builders/frequency.js +5 -19
- package/scripts/builders/jurisdiction-clocks.js +6 -25
- package/scripts/builders/recipes.js +6 -14
- package/scripts/builders/section-offsets.js +13 -51
- package/scripts/builders/stale-content.js +7 -28
- package/scripts/builders/summary-cards.js +8 -29
- package/scripts/builders/theater-fingerprints.js +12 -27
- package/scripts/builders/token-budget.js +4 -31
- package/scripts/check-agents-md-collectors.js +11 -54
- package/scripts/check-catalog-gap-budget.js +15 -32
- package/scripts/check-changelog-extract.js +18 -48
- package/scripts/check-codebase-patterns-currency.js +6 -22
- package/scripts/check-codebase-patterns.js +50 -143
- package/scripts/check-epss-consistency.js +9 -64
- package/scripts/check-framework-gap-coverage.js +13 -31
- package/scripts/check-manifest-snapshot.js +13 -73
- package/scripts/check-sbom-currency.js +44 -142
- package/scripts/check-test-count.js +15 -52
- package/scripts/check-test-coverage.js +66 -197
- package/scripts/check-test-subjects.js +21 -62
- package/scripts/check-ttp-references.js +14 -38
- package/scripts/check-ttp-upstream.js +8 -40
- package/scripts/check-version-bump.js +9 -61
- package/scripts/check-version-tags.js +20 -121
- package/scripts/predeploy.js +38 -184
- package/scripts/refresh-manifest-snapshot.js +16 -38
- package/scripts/refresh-mitre-atlas.js +3 -8
- package/scripts/refresh-mitre-attack.js +1 -8
- package/scripts/refresh-mitre-d3fend.js +3 -9
- package/scripts/refresh-mitre-ics-attack.js +3 -8
- package/scripts/refresh-reverse-refs.js +27 -94
- package/scripts/refresh-rfc-index.js +2 -10
- package/scripts/refresh-sbom.js +31 -161
- package/scripts/refresh-upstream-catalogs.js +40 -137
- package/scripts/release.js +69 -232
- package/scripts/run-e2e-scenarios.js +24 -71
- package/scripts/sync-manifest-metadata.js +10 -34
- package/scripts/sync-package-description.js +8 -17
- package/scripts/validate-vendor-online.js +13 -44
- package/scripts/verify-shipped-tarball.js +35 -140
package/lib/version-pins.js
CHANGED
|
@@ -1,39 +1,10 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* Pre-v0.13.20 history: ATLAS version was pinned to v5.4.0 in 33+
|
|
9
|
-
* locations (READMEs, AGENTS.md, ARCHITECTURE.md, agent personas,
|
|
10
|
-
* skill bodies, schema descriptions, manifest.json). Bumping required
|
|
11
|
-
* a lockstep regex-replace across all 33 files. v0.13.18 bumped to
|
|
12
|
-
* v5.6.0; the regex sweep accidentally touched dates in unrelated
|
|
13
|
-
* paragraphs and only failed-loudly because the tests asserted
|
|
14
|
-
* version drift. v0.13.20 makes the pin schema-driven:
|
|
15
|
-
*
|
|
16
|
-
* - `data/atlas-ttps.json._meta.atlas_version` is the source of truth.
|
|
17
|
-
* - `data/attack-techniques.json._meta.attack_version` is too.
|
|
18
|
-
* - This module reads both, exposes them via getAtlasVersion() and
|
|
19
|
-
* getAttackVersion() helpers, and is the canonical resolver every
|
|
20
|
-
* consumer (test runner, doc-currency check, lint, skill-body
|
|
21
|
-
* scanner) reaches through.
|
|
22
|
-
*
|
|
23
|
-
* The drift-detection tests in tests/atlas-version-canonical.test.js
|
|
24
|
-
* and tests/attack-version-canonical.test.js now compare every
|
|
25
|
-
* operator-facing mention against the value this module returns.
|
|
26
|
-
* A future bump is `node $(exceptd path)/lib/sign.js sign-all` + this
|
|
27
|
-
* module reads the new value; no lockstep doc edit needed except where
|
|
28
|
-
* the mention is
|
|
29
|
-
* a literal-string semantic ("upgrade from v5.4.0 to v5.6.0") that an
|
|
30
|
-
* operator must read.
|
|
31
|
-
*
|
|
32
|
-
* API:
|
|
33
|
-
* getAtlasVersion() → "2026.06"
|
|
34
|
-
* getAttackVersion() → "19.1"
|
|
35
|
-
* getAtlasReleaseDate() → "2026-05-27"
|
|
36
|
-
* getAllPins() → { atlas_version, atlas_release_date, attack_version, ... }
|
|
3
|
+
* Canonical resolver for the MITRE ATT&CK / ATLAS version pins. The `_meta`
|
|
4
|
+
* blocks of data/atlas-ttps.json and data/attack-techniques.json are the source
|
|
5
|
+
* of truth and every consumer reads them through here, so a pin bump is a data
|
|
6
|
+
* edit. tests/atlas-version-canonical.test.js and its ATT&CK counterpart
|
|
7
|
+
* compare each operator-facing mention against what this module returns.
|
|
37
8
|
*/
|
|
38
9
|
|
|
39
10
|
const fs = require("fs");
|
package/lib/worker-pool.js
CHANGED
|
@@ -1,28 +1,13 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
3
|
+
* Thin wrapper over vendor/blamejs/worker-pool.js, which supplies the bounded
|
|
4
|
+
* concurrency, bounded queue, per-task timeout and worker recycling. Added
|
|
5
|
+
* here: a class around the function-style `create()`, a `runAll()` that owns
|
|
6
|
+
* and terminates its pool, and DEFAULT_SIZE for callers that size manually.
|
|
4
7
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* - bounded concurrency, defaults to max(2, cpus)
|
|
9
|
-
* - bounded in-memory queue (default 1024 depth)
|
|
10
|
-
* - per-task timeout (default 5min)
|
|
11
|
-
* - worker recycle on uncaught error / timeout / exit
|
|
12
|
-
*
|
|
13
|
-
* What this wrapper adds:
|
|
14
|
-
*
|
|
15
|
-
* - WorkerPool class around the function-style `create()` API for
|
|
16
|
-
* callers that prefer an instance
|
|
17
|
-
* - runAll(tasks, opts) helper that runs an array of tasks through a
|
|
18
|
-
* fresh pool and terminates it when done
|
|
19
|
-
* - DEFAULT_SIZE re-export for callers that want to size manually
|
|
20
|
-
*
|
|
21
|
-
* Honest framing: at v0.7.0 corpus size (38 skills, 10 catalogs, ~150ms
|
|
22
|
-
* total build time) worker-thread spawn cost is comparable to the work
|
|
23
|
-
* itself. The pool is here so the architecture scales as the corpus
|
|
24
|
-
* grows and so users can experiment with `--parallel`. Sequential builds
|
|
25
|
-
* remain the default.
|
|
8
|
+
* At current corpus size a worker spawn costs about as much as the work, so
|
|
9
|
+
* sequential builds stay the default; the pool is here for `--parallel` and
|
|
10
|
+
* for the corpus growing.
|
|
26
11
|
*/
|
|
27
12
|
|
|
28
13
|
const os = require("os");
|
|
@@ -32,9 +17,8 @@ const DEFAULT_SIZE = Math.max(1, Math.min(8, os.cpus()?.length || 4));
|
|
|
32
17
|
|
|
33
18
|
class WorkerPool {
|
|
34
19
|
/**
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
* - size, maxQueueDepth, taskTimeoutMs, onExit — forwarded to vendored.create
|
|
20
|
+
* `opts.runnerPath` — absolute path to the worker script — is required; size,
|
|
21
|
+
* maxQueueDepth, taskTimeoutMs and onExit forward to vendored.create.
|
|
38
22
|
*/
|
|
39
23
|
constructor(opts = {}) {
|
|
40
24
|
if (!opts.runnerPath) throw new Error("WorkerPool: runnerPath is required");
|
|
@@ -56,9 +40,7 @@ class WorkerPool {
|
|
|
56
40
|
}
|
|
57
41
|
}
|
|
58
42
|
|
|
59
|
-
/**
|
|
60
|
-
* Run a list of tasks against a fresh pool, await all results, terminate.
|
|
61
|
-
*/
|
|
43
|
+
/** Runs the tasks on a fresh pool, awaits every result, then terminates it. */
|
|
62
44
|
async function runAll(tasks, opts = {}) {
|
|
63
45
|
const pool = new WorkerPool(opts);
|
|
64
46
|
try {
|
|
@@ -72,8 +54,7 @@ module.exports = {
|
|
|
72
54
|
WorkerPool,
|
|
73
55
|
runAll,
|
|
74
56
|
DEFAULT_SIZE,
|
|
75
|
-
//
|
|
76
|
-
// or need the size constants.
|
|
57
|
+
// For callers that prefer the function-style API or need the size constants.
|
|
77
58
|
create: vendored.create,
|
|
78
59
|
MIN_SIZE: vendored.MIN_SIZE,
|
|
79
60
|
MAX_SIZE: vendored.MAX_SIZE,
|
package/lib/xml-tokenizer.js
CHANGED
|
@@ -1,49 +1,16 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
3
|
+
* Minimal XML/RSS/Atom tokenizer with no runtime dependencies.
|
|
4
4
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* failed on:
|
|
8
|
-
* - XML namespaces (`<atom:entry>` vs `<entry>`)
|
|
9
|
-
* - Nested CDATA
|
|
10
|
-
* - Self-closing `<link href="..."/>` vs container `<link>...</link>`
|
|
11
|
-
* - HTML-escaped entities inside titles
|
|
12
|
-
* - Multi-line title content
|
|
5
|
+
* parseFeed(xmlString) → [{title, link, published, body}, ...]
|
|
6
|
+
* tokenize(xml, { onTagOpen, onTagClose, onText, onCData })
|
|
13
7
|
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
* Design constraints:
|
|
19
|
-
* - Zero runtime dependencies (the project ships with no deps).
|
|
20
|
-
* - Streaming-friendly via a callback API (does not buffer the whole
|
|
21
|
-
* DOM — relevant for the 15 MB IETF RFC index).
|
|
22
|
-
* - Namespace-aware via element-localname matching (the local-name
|
|
23
|
-
* of `<atom:entry>` is `entry`).
|
|
24
|
-
* - CDATA-correct: `<![CDATA[...]]>` content is passed through verbatim
|
|
25
|
-
* including unescaped `<` and `&`.
|
|
26
|
-
* - Entity-correct: the five named XML entities (lt, gt, amp, apos,
|
|
27
|
-
* quot) plus numeric character references (`&#NNN;`, `&#xHH;`)
|
|
28
|
-
* are decoded. Other named entities pass through unchanged (HTML
|
|
29
|
-
* entities in RSS bodies are a recoverable variant we tolerate).
|
|
30
|
-
*
|
|
31
|
-
* Not designed for: DTD parsing, XInclude, XSLT, or any external-entity
|
|
32
|
-
* resolution. Feeds that need those features are outside the scope of
|
|
33
|
-
* a security-tooling intake pipeline.
|
|
34
|
-
*
|
|
35
|
-
* API:
|
|
36
|
-
* const { parseFeed } = require("./xml-tokenizer");
|
|
37
|
-
* const items = parseFeed(xmlString); // returns [{title, link, published, body}, ...]
|
|
38
|
-
*
|
|
39
|
-
* For lower-level use, the underlying tokenizer is exported too:
|
|
40
|
-
* const { tokenize } = require("./xml-tokenizer");
|
|
41
|
-
* tokenize(xml, { onTagOpen, onTagClose, onText, onCData });
|
|
8
|
+
* Streams through callbacks rather than a buffered DOM, which matters for the
|
|
9
|
+
* 15 MB IETF RFC index. Elements match on local-name, so `<atom:entry>` and
|
|
10
|
+
* `<entry>` are the same element. No DTD parsing, XInclude, XSLT, or
|
|
11
|
+
* external-entity resolution.
|
|
42
12
|
*/
|
|
43
13
|
|
|
44
|
-
// Decode the five canonical XML entities + numeric character references.
|
|
45
|
-
// Unknown named entities pass through unchanged (we're tolerant of
|
|
46
|
-
// HTML-style entities that legitimately appear in RSS body text).
|
|
47
14
|
function decodeEntities(s) {
|
|
48
15
|
if (typeof s !== "string") return s;
|
|
49
16
|
return s.replace(/&(#x[0-9a-fA-F]+|#[0-9]+|[a-zA-Z]+);/g, (m, ref) => {
|
|
@@ -65,7 +32,6 @@ function decodeEntities(s) {
|
|
|
65
32
|
});
|
|
66
33
|
}
|
|
67
34
|
|
|
68
|
-
// Strip the optional `prefix:` from a namespaced element/attribute name.
|
|
69
35
|
function localName(qname) {
|
|
70
36
|
const idx = qname.indexOf(":");
|
|
71
37
|
return idx === -1 ? qname : qname.slice(idx + 1);
|
|
@@ -74,8 +40,7 @@ function localName(qname) {
|
|
|
74
40
|
function parseAttrs(rawAttrs) {
|
|
75
41
|
const out = {};
|
|
76
42
|
if (!rawAttrs) return out;
|
|
77
|
-
//
|
|
78
|
-
// whitespace without confusing a regex.
|
|
43
|
+
// Character-by-character, so a quoted value may contain `=` and whitespace.
|
|
79
44
|
let i = 0;
|
|
80
45
|
const len = rawAttrs.length;
|
|
81
46
|
while (i < len) {
|
|
@@ -95,7 +60,6 @@ function parseAttrs(rawAttrs) {
|
|
|
95
60
|
while (i < len && /\s/.test(rawAttrs[i])) i++;
|
|
96
61
|
const quote = rawAttrs[i];
|
|
97
62
|
if (quote !== '"' && quote !== "'") {
|
|
98
|
-
// Unquoted value — read until whitespace or end.
|
|
99
63
|
const valStart = i;
|
|
100
64
|
while (i < len && !/\s/.test(rawAttrs[i])) i++;
|
|
101
65
|
out[localName(name)] = decodeEntities(rawAttrs.slice(valStart, i));
|
|
@@ -111,13 +75,9 @@ function parseAttrs(rawAttrs) {
|
|
|
111
75
|
}
|
|
112
76
|
|
|
113
77
|
/**
|
|
114
|
-
*
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
* else onText) exactly like the main loop's CDATA fast-path; all other bytes
|
|
118
|
-
* — including a stray unescaped '<' or inline HTML — are entity-decoded and
|
|
119
|
-
* emitted via onText. Keeping the CDATA contract here means a `<![CDATA[...]]>`
|
|
120
|
-
* title still surfaces only its inner text, not the literal markers.
|
|
78
|
+
* Emits the character data of a raw-text leaf element span `[start, end)`. CDATA
|
|
79
|
+
* inside the span goes out verbatim (onCData, else onText); every other byte —
|
|
80
|
+
* a stray unescaped '<' or inline HTML included — is entity-decoded via onText.
|
|
121
81
|
*/
|
|
122
82
|
function emitRawText(xml, start, end, H) {
|
|
123
83
|
let p = start;
|
|
@@ -133,8 +93,7 @@ function emitRawText(xml, start, end, H) {
|
|
|
133
93
|
if (chunk.length && H.onText) H.onText(decodeEntities(chunk));
|
|
134
94
|
}
|
|
135
95
|
const cend = xml.indexOf("]]>", c + 9);
|
|
136
|
-
// findRawTextEnd
|
|
137
|
-
// tag, so cend is within range; guard anyway.
|
|
96
|
+
// findRawTextEnd guarantees a terminated CDATA here; guarded anyway.
|
|
138
97
|
const inner = cend === -1 || cend > end ? xml.slice(c + 9, end) : xml.slice(c + 9, cend);
|
|
139
98
|
if (H.onCData) H.onCData(inner);
|
|
140
99
|
else if (H.onText) H.onText(inner);
|
|
@@ -143,17 +102,11 @@ function emitRawText(xml, start, end, H) {
|
|
|
143
102
|
}
|
|
144
103
|
|
|
145
104
|
/**
|
|
146
|
-
*
|
|
147
|
-
*
|
|
148
|
-
*
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
* legitimately appears inside `<![CDATA[ ... ]]>` is treated as literal text,
|
|
152
|
-
* not as the element's terminator.
|
|
153
|
-
*
|
|
154
|
-
* Returns { contentEnd, next } on success — `contentEnd` is the index of the
|
|
155
|
-
* close tag's leading `<`, `next` is the index just past its `>`. Returns null
|
|
156
|
-
* when no matching close tag exists before EOF (genuinely truncated input).
|
|
105
|
+
* Finds the close tag of a raw-text leaf element, scanning from `start` (the byte
|
|
106
|
+
* after the open tag's `>`) for the first `</...name>` whose local-name is `name`.
|
|
107
|
+
* CDATA is skipped, so a `</title>` inside `<![CDATA[ ... ]]>` is literal text.
|
|
108
|
+
* Returns `{ contentEnd, next }` — the index of the close tag's `<`, and the index
|
|
109
|
+
* just past its `>` — or null when no close tag exists before EOF.
|
|
157
110
|
*/
|
|
158
111
|
function findRawTextEnd(xml, start, name) {
|
|
159
112
|
const len = xml.length;
|
|
@@ -161,8 +114,6 @@ function findRawTextEnd(xml, start, name) {
|
|
|
161
114
|
while (j < len) {
|
|
162
115
|
const lt = xml.indexOf("<", j);
|
|
163
116
|
if (lt === -1) return null;
|
|
164
|
-
// Step over a CDATA section so its contents (which may contain "</name>"
|
|
165
|
-
// literally) cannot satisfy the close-tag match.
|
|
166
117
|
if (xml.startsWith("<![CDATA[", lt)) {
|
|
167
118
|
const cend = xml.indexOf("]]>", lt + 9);
|
|
168
119
|
if (cend === -1) return null; // unterminated CDATA → truncated
|
|
@@ -179,16 +130,15 @@ function findRawTextEnd(xml, start, name) {
|
|
|
179
130
|
j = gt + 1;
|
|
180
131
|
continue;
|
|
181
132
|
}
|
|
182
|
-
// Any other '<'
|
|
183
|
-
// of the leaf's character data — advance past it without classifying.
|
|
133
|
+
// Any other '<' is part of the leaf's character data — advance past it.
|
|
184
134
|
j = lt + 1;
|
|
185
135
|
}
|
|
186
136
|
return null;
|
|
187
137
|
}
|
|
188
138
|
|
|
189
139
|
/**
|
|
190
|
-
* Streaming tokenizer. Calls handlers in document order. Returns no
|
|
191
|
-
*
|
|
140
|
+
* Streaming tokenizer. Calls handlers in document order. Returns no value —
|
|
141
|
+
* accumulation is the caller's responsibility.
|
|
192
142
|
*
|
|
193
143
|
* Handlers (all optional):
|
|
194
144
|
* onTagOpen(name, attrs, selfClosing)
|
|
@@ -199,16 +149,9 @@ function findRawTextEnd(xml, start, name) {
|
|
|
199
149
|
* onPI(name, content) processing instructions (<?xml-stylesheet?>)
|
|
200
150
|
* onError(message, position)
|
|
201
151
|
*
|
|
202
|
-
*
|
|
203
|
-
*
|
|
204
|
-
*
|
|
205
|
-
* to the matching `</name>` close tag is treated as
|
|
206
|
-
* character data (entities decoded, then onText), exactly
|
|
207
|
-
* like the CDATA fast-path. This makes leaf fields tolerant
|
|
208
|
-
* of stray unescaped '<' (e.g. "affects versions < 5.0"),
|
|
209
|
-
* which real RSS/Atom routinely emits, instead of letting a
|
|
210
|
-
* recoverable lexical glitch silently drop the whole field.
|
|
211
|
-
* Structural parsing stays strict for container elements.
|
|
152
|
+
* `rawTextElements`, a Set of local-names on the same object, marks leaf elements
|
|
153
|
+
* whose content is #PCDATA: every byte up to the matching `</name>` is character
|
|
154
|
+
* data, so a stray unescaped '<' does not drop the field. Containers stay strict.
|
|
212
155
|
*/
|
|
213
156
|
function tokenize(xml, handlers) {
|
|
214
157
|
const H = handlers || {};
|
|
@@ -219,16 +162,11 @@ function tokenize(xml, handlers) {
|
|
|
219
162
|
const rawTextElements = H.rawTextElements instanceof Set ? H.rawTextElements : null;
|
|
220
163
|
const len = xml.length;
|
|
221
164
|
let i = 0;
|
|
222
|
-
//
|
|
223
|
-
// instead of silently dropping the residual content. This is the
|
|
224
|
-
// observability gap the v0.13.17 regex parser had: a malformed feed
|
|
225
|
-
// (truncated mid-element) returned `[]` with no signal that the
|
|
226
|
-
// parser had given up.
|
|
165
|
+
// The open-tag stack turns unclosed-at-EOF into an error, not dropped content.
|
|
227
166
|
const openStack = [];
|
|
228
167
|
while (i < len) {
|
|
229
168
|
const next = xml.indexOf("<", i);
|
|
230
169
|
if (next === -1) {
|
|
231
|
-
// Trailing text — flush.
|
|
232
170
|
const tail = xml.slice(i);
|
|
233
171
|
if (tail.length && H.onText) H.onText(decodeEntities(tail));
|
|
234
172
|
if (openStack.length && H.onError) {
|
|
@@ -240,7 +178,6 @@ function tokenize(xml, handlers) {
|
|
|
240
178
|
const text = xml.slice(i, next);
|
|
241
179
|
if (text.length && H.onText) H.onText(decodeEntities(text));
|
|
242
180
|
}
|
|
243
|
-
// Now at `<` — classify the construct.
|
|
244
181
|
if (xml.startsWith("<!--", next)) {
|
|
245
182
|
const end = xml.indexOf("-->", next + 4);
|
|
246
183
|
if (end === -1) {
|
|
@@ -295,7 +232,6 @@ function tokenize(xml, handlers) {
|
|
|
295
232
|
i = j + 1;
|
|
296
233
|
continue;
|
|
297
234
|
}
|
|
298
|
-
// Element tag — open / close / self-closing.
|
|
299
235
|
const close = xml.indexOf(">", next);
|
|
300
236
|
if (close === -1) {
|
|
301
237
|
if (H.onError) H.onError("unterminated element tag", next);
|
|
@@ -307,7 +243,6 @@ function tokenize(xml, handlers) {
|
|
|
307
243
|
if (inner.startsWith("/")) { isClose = true; inner = inner.slice(1); }
|
|
308
244
|
if (inner.endsWith("/")) { selfClose = true; inner = inner.slice(0, -1); }
|
|
309
245
|
inner = inner.trim();
|
|
310
|
-
// Split name and attrs at the first whitespace.
|
|
311
246
|
const wsAt = inner.search(/\s/);
|
|
312
247
|
const rawName = wsAt === -1 ? inner : inner.slice(0, wsAt);
|
|
313
248
|
const rawAttrs = wsAt === -1 ? "" : inner.slice(wsAt + 1);
|
|
@@ -321,17 +256,11 @@ function tokenize(xml, handlers) {
|
|
|
321
256
|
if (selfClose) {
|
|
322
257
|
if (H.onTagClose) H.onTagClose(name);
|
|
323
258
|
} else if (rawTextElements && rawTextElements.has(name)) {
|
|
324
|
-
//
|
|
325
|
-
//
|
|
326
|
-
// as character data — only the matching close tag terminates it. Any
|
|
327
|
-
// inner '<...>' (a stray unescaped '<', or inline HTML like <b>) is
|
|
328
|
-
// preserved as text and normalized later by stripHtml(), instead of
|
|
329
|
-
// being misclassified as markup and silently dropping the field.
|
|
259
|
+
// A raw-text leaf: only the matching `</name>` terminates it. Inner
|
|
260
|
+
// '<...>' stays text for stripHtml(), rather than being classified as markup.
|
|
330
261
|
const span = findRawTextEnd(xml, close + 1, name);
|
|
331
262
|
if (span === null) {
|
|
332
|
-
//
|
|
333
|
-
// the loud-error contract just like the structural path would, then
|
|
334
|
-
// flush the residual as text and stop.
|
|
263
|
+
// Never closes before EOF: report as the structural path does, flush, stop.
|
|
335
264
|
if (H.onError) H.onError("unterminated element at EOF: " + name, len);
|
|
336
265
|
const tail = xml.slice(close + 1);
|
|
337
266
|
if (tail.length && H.onText) H.onText(decodeEntities(tail));
|
|
@@ -352,39 +281,28 @@ function tokenize(xml, handlers) {
|
|
|
352
281
|
}
|
|
353
282
|
}
|
|
354
283
|
|
|
355
|
-
//
|
|
356
|
-
// one of these opens, the tokenizer runs in raw-text mode until the matching
|
|
357
|
-
// close tag, so a stray unescaped '<' inside the field (e.g. "affects
|
|
358
|
-
// versions < 5.0") is preserved as text instead of being misclassified as
|
|
359
|
-
// markup and silently dropping the whole field.
|
|
284
|
+
// The leaf elements parsed in raw-text mode — see tokenize()'s rawTextElements.
|
|
360
285
|
const LEAF_FIELDS = new Set([
|
|
361
286
|
"title", "link", "pubDate", "published", "updated",
|
|
362
287
|
"description", "content", "summary",
|
|
363
288
|
]);
|
|
364
289
|
|
|
365
290
|
// rel-rank for Atom <link> sibling selection. RFC 4287: a <link> with no rel
|
|
366
|
-
// defaults to rel="alternate",
|
|
367
|
-
// (self / replies / edit / enclosure) should never clobber an alternate.
|
|
291
|
+
// defaults to rel="alternate", so a non-alternate must never clobber one.
|
|
368
292
|
function relRank(rel) {
|
|
369
293
|
if (rel == null || rel === "" || String(rel).toLowerCase() === "alternate") return 2;
|
|
370
294
|
return 1;
|
|
371
295
|
}
|
|
372
296
|
|
|
373
297
|
/**
|
|
374
|
-
*
|
|
375
|
-
*
|
|
376
|
-
*
|
|
377
|
-
* { message, position } records. This is the channel a caller cannot forget
|
|
378
|
-
* to opt into — parseFeed() below is a thin back-compat wrapper.
|
|
379
|
-
*
|
|
380
|
-
* items: [{ title, link, published, body }, ...]
|
|
298
|
+
* Parses an RSS / Atom feed into `{ items, errors }`, where items are
|
|
299
|
+
* `{ title, link, published, body }` and errors are `{ message, position }`.
|
|
300
|
+
* The errors channel cannot be opted out of, so prefer this over parseFeed().
|
|
381
301
|
*/
|
|
382
302
|
function parseFeedDetailed(xml) {
|
|
383
303
|
const items = [];
|
|
384
304
|
const errors = [];
|
|
385
|
-
//
|
|
386
|
-
// <entry>; both nest title / link / pubDate / published / updated /
|
|
387
|
-
// description / content / summary.
|
|
305
|
+
// RSS uses <item>, Atom uses <entry>; both nest the same leaf fields.
|
|
388
306
|
const ITEM_LOCALS = new Set(["item", "entry"]);
|
|
389
307
|
const FIELD_MAP = {
|
|
390
308
|
title: "title",
|
|
@@ -414,23 +332,17 @@ function parseFeedDetailed(xml) {
|
|
|
414
332
|
if (FIELD_MAP[name]) {
|
|
415
333
|
activeField = FIELD_MAP[name];
|
|
416
334
|
buffer = "";
|
|
417
|
-
//
|
|
418
|
-
// ranked the same way the attribute path is — a non-alternate link with
|
|
419
|
-
// text must not clobber a captured rel=alternate.
|
|
335
|
+
// Remembered so the close below ranks the link as this path does.
|
|
420
336
|
if (name === "link") linkRel = attrs ? attrs.rel : null;
|
|
421
|
-
//
|
|
422
|
-
//
|
|
423
|
-
// let a non-alternate rel (self/replies/edit) clobber an alternate
|
|
424
|
-
// already in hand. First-alternate-wins, independent of document
|
|
425
|
-
// order.
|
|
337
|
+
// Only an alternate or rel-absent link upgrades the captured value,
|
|
338
|
+
// whatever the document order.
|
|
426
339
|
if (name === "link" && attrs && attrs.href) {
|
|
427
340
|
const r = relRank(attrs.rel);
|
|
428
341
|
if (r > linkRank) {
|
|
429
342
|
current.link = attrs.href;
|
|
430
343
|
linkRank = r;
|
|
431
344
|
}
|
|
432
|
-
//
|
|
433
|
-
// no element-text close to wait for.
|
|
345
|
+
// Resolved from the attribute — no element-text close to wait for.
|
|
434
346
|
if (selfClosing) activeField = null;
|
|
435
347
|
}
|
|
436
348
|
}
|
|
@@ -447,12 +359,8 @@ function parseFeedDetailed(xml) {
|
|
|
447
359
|
if (!current) return;
|
|
448
360
|
if (FIELD_MAP[name] && activeField === FIELD_MAP[name]) {
|
|
449
361
|
const value = buffer.trim();
|
|
450
|
-
//
|
|
451
|
-
//
|
|
452
|
-
// link with text cannot clobber a captured rel=alternate. RSS links
|
|
453
|
-
// carry no rel, and relRank(undefined) === 2, so RSS element-text links
|
|
454
|
-
// keep their previous authoritative rank. An empty element-text link
|
|
455
|
-
// falls back to whatever attribute capture onTagOpen already recorded.
|
|
362
|
+
// Element text goes through the same rel gate as the attribute path; RSS
|
|
363
|
+
// links carry no rel, so relRank(undefined) is 2 and they keep the top rank.
|
|
456
364
|
if (name === "link") {
|
|
457
365
|
const r = relRank(linkRel);
|
|
458
366
|
if (value && r > linkRank) {
|
|
@@ -461,12 +369,8 @@ function parseFeedDetailed(xml) {
|
|
|
461
369
|
}
|
|
462
370
|
linkRel = null;
|
|
463
371
|
} else if (activeField === "body" || activeField === "title") {
|
|
464
|
-
//
|
|
465
|
-
//
|
|
466
|
-
// emphasis; the operational consumer wants plain text. CDATA
|
|
467
|
-
// content reaches here verbatim, so this also strips HTML
|
|
468
|
-
// that was wrapped in CDATA to dodge entity-encoding. A stray
|
|
469
|
-
// unescaped '<' that survived raw-text mode collapses here too.
|
|
372
|
+
// Feeds embed inline HTML in titles and bodies. CDATA reaches here
|
|
373
|
+
// verbatim, so HTML wrapped in CDATA to dodge encoding is stripped too.
|
|
470
374
|
current[activeField] = stripHtml(value);
|
|
471
375
|
} else {
|
|
472
376
|
current[activeField] = value;
|
|
@@ -490,13 +394,9 @@ function parseFeedDetailed(xml) {
|
|
|
490
394
|
}
|
|
491
395
|
|
|
492
396
|
/**
|
|
493
|
-
*
|
|
494
|
-
*
|
|
495
|
-
*
|
|
496
|
-
* Empty array on parse failure. Errors are ALWAYS collected internally; the
|
|
497
|
-
* optional `errors` array is filled for callers that pass one (back-compat).
|
|
498
|
-
* Prefer parseFeedDetailed(xml) for new callers — it returns the errors
|
|
499
|
-
* channel unconditionally so it cannot be silently dropped.
|
|
397
|
+
* Parses an RSS / Atom feed into `[{ title, link, published, body }, ...]`, or an
|
|
398
|
+
* empty array on parse failure. Errors are copied into the optional `errors` array
|
|
399
|
+
* when a caller passes one and dropped otherwise: prefer parseFeedDetailed().
|
|
500
400
|
*/
|
|
501
401
|
function parseFeed(xml, errors = null) {
|
|
502
402
|
const { items, errors: collected } = parseFeedDetailed(xml);
|
|
@@ -508,14 +408,9 @@ function parseFeed(xml, errors = null) {
|
|
|
508
408
|
|
|
509
409
|
function stripHtml(s) {
|
|
510
410
|
if (typeof s !== "string") return "";
|
|
511
|
-
// Single-pass
|
|
512
|
-
//
|
|
513
|
-
//
|
|
514
|
-
// a ReDoS-class DoS, since parseFeed() runs on network-fetched RSS/Atom
|
|
515
|
-
// bodies). Walk '<'..'>' instead, preserving the prior contract exactly: a
|
|
516
|
-
// real `<...>` becomes a space; a stray '<' with no closing '>' and an empty
|
|
517
|
-
// '<>' survive as literal text (the `[^>]+` quantifier required >=1 char, so
|
|
518
|
-
// neither matched). Then collapse whitespace as before.
|
|
411
|
+
// Single-pass, because a backtracking /<[^>]+>/g is O(n^2) on feed text with many
|
|
412
|
+
// '<' and no '>' — a ReDoS-class DoS, since parseFeed() runs on network-fetched
|
|
413
|
+
// bodies. A stray '<' with no closing '>' and an empty '<>' survive as text.
|
|
519
414
|
let out = "";
|
|
520
415
|
let i = 0;
|
|
521
416
|
while (i < s.length) {
|