@blamejs/exceptd-skills 0.19.33 → 0.19.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +10 -0
  2. package/bin/exceptd.js +896 -2824
  3. package/data/_indexes/_meta.json +2 -2
  4. package/lib/auto-discovery.js +56 -286
  5. package/lib/canonical-eq.js +7 -40
  6. package/lib/citation-resolve.js +22 -70
  7. package/lib/collectors/ai-api.js +20 -54
  8. package/lib/collectors/cicd-pipeline-compromise.js +40 -108
  9. package/lib/collectors/citation-hygiene.js +72 -210
  10. package/lib/collectors/containers.js +41 -130
  11. package/lib/collectors/cred-stores.js +31 -115
  12. package/lib/collectors/crypto-codebase.js +55 -138
  13. package/lib/collectors/crypto.js +24 -54
  14. package/lib/collectors/hardening.js +20 -78
  15. package/lib/collectors/kernel.js +16 -46
  16. package/lib/collectors/library-author.js +57 -206
  17. package/lib/collectors/mcp.js +24 -70
  18. package/lib/collectors/runtime.js +24 -86
  19. package/lib/collectors/sbom.js +34 -106
  20. package/lib/collectors/scan-excludes.js +31 -138
  21. package/lib/collectors/secrets.js +62 -178
  22. package/lib/cross-ref-api.js +39 -123
  23. package/lib/currency-severity.js +8 -27
  24. package/lib/cve-batch.js +13 -21
  25. package/lib/cve-cli.js +13 -20
  26. package/lib/cve-curation.js +72 -239
  27. package/lib/cve-regression-watcher.js +29 -152
  28. package/lib/cvss.js +13 -54
  29. package/lib/doctor-bucketing.js +3 -19
  30. package/lib/exit-codes.js +10 -42
  31. package/lib/flag-suggest.js +7 -25
  32. package/lib/framework-gap.js +35 -114
  33. package/lib/gap-detectors.js +37 -159
  34. package/lib/id-validation.js +9 -30
  35. package/lib/job-queue.js +13 -36
  36. package/lib/lint-skills.js +64 -232
  37. package/lib/playbook-runner.js +693 -2095
  38. package/lib/prefetch.js +100 -376
  39. package/lib/refresh-external.js +199 -627
  40. package/lib/refresh-network.js +75 -307
  41. package/lib/rfc-cli.js +23 -68
  42. package/lib/scoring.js +77 -145
  43. package/lib/sign.js +43 -229
  44. package/lib/source-advisories.js +43 -194
  45. package/lib/source-ghsa.js +37 -120
  46. package/lib/source-osv.js +94 -266
  47. package/lib/ttp-mapper.js +14 -24
  48. package/lib/upstream-check-cli.js +10 -28
  49. package/lib/upstream-check.js +19 -44
  50. package/lib/validate-catalog-meta.js +17 -61
  51. package/lib/validate-cve-catalog.js +43 -119
  52. package/lib/validate-indexes.js +25 -76
  53. package/lib/validate-package.js +16 -62
  54. package/lib/validate-playbooks.js +69 -275
  55. package/lib/validate-vendor.js +16 -49
  56. package/lib/verify.js +56 -286
  57. package/lib/version-pins.js +5 -34
  58. package/lib/worker-pool.js +11 -30
  59. package/lib/xml-tokenizer.js +47 -152
  60. package/manifest.json +53 -53
  61. package/orchestrator/dispatcher.js +17 -68
  62. package/orchestrator/event-bus.js +11 -74
  63. package/orchestrator/index.js +138 -412
  64. package/orchestrator/pipeline.js +28 -85
  65. package/orchestrator/scanner.js +34 -138
  66. package/orchestrator/scheduler.js +20 -84
  67. package/package.json +1 -1
  68. package/sbom.cdx.json +241 -241
  69. package/scripts/audit-catalog-gaps.js +9 -62
  70. package/scripts/audit-cross-skill.js +5 -31
  71. package/scripts/audit-perf.js +6 -16
  72. package/scripts/backfill-theater-test.js +7 -64
  73. package/scripts/bootstrap.js +12 -44
  74. package/scripts/build-indexes.js +40 -154
  75. package/scripts/builders/activity-feed.js +4 -14
  76. package/scripts/builders/catalog-summaries.js +3 -10
  77. package/scripts/builders/currency.js +7 -20
  78. package/scripts/builders/cwe-chains.js +7 -30
  79. package/scripts/builders/did-ladders.js +6 -13
  80. package/scripts/builders/frequency.js +5 -19
  81. package/scripts/builders/jurisdiction-clocks.js +6 -25
  82. package/scripts/builders/recipes.js +6 -14
  83. package/scripts/builders/section-offsets.js +13 -51
  84. package/scripts/builders/stale-content.js +7 -28
  85. package/scripts/builders/summary-cards.js +8 -29
  86. package/scripts/builders/theater-fingerprints.js +12 -27
  87. package/scripts/builders/token-budget.js +4 -31
  88. package/scripts/check-agents-md-collectors.js +11 -54
  89. package/scripts/check-catalog-gap-budget.js +15 -32
  90. package/scripts/check-changelog-extract.js +18 -48
  91. package/scripts/check-codebase-patterns-currency.js +6 -22
  92. package/scripts/check-codebase-patterns.js +50 -143
  93. package/scripts/check-epss-consistency.js +9 -64
  94. package/scripts/check-framework-gap-coverage.js +13 -31
  95. package/scripts/check-manifest-snapshot.js +13 -73
  96. package/scripts/check-sbom-currency.js +44 -142
  97. package/scripts/check-test-count.js +15 -52
  98. package/scripts/check-test-coverage.js +66 -197
  99. package/scripts/check-test-subjects.js +21 -62
  100. package/scripts/check-ttp-references.js +14 -38
  101. package/scripts/check-ttp-upstream.js +8 -40
  102. package/scripts/check-version-bump.js +9 -61
  103. package/scripts/check-version-tags.js +20 -121
  104. package/scripts/predeploy.js +38 -184
  105. package/scripts/refresh-manifest-snapshot.js +16 -38
  106. package/scripts/refresh-mitre-atlas.js +3 -8
  107. package/scripts/refresh-mitre-attack.js +1 -8
  108. package/scripts/refresh-mitre-d3fend.js +3 -9
  109. package/scripts/refresh-mitre-ics-attack.js +3 -8
  110. package/scripts/refresh-reverse-refs.js +27 -94
  111. package/scripts/refresh-rfc-index.js +2 -10
  112. package/scripts/refresh-sbom.js +31 -161
  113. package/scripts/refresh-upstream-catalogs.js +40 -137
  114. package/scripts/release.js +69 -232
  115. package/scripts/run-e2e-scenarios.js +24 -71
  116. package/scripts/sync-manifest-metadata.js +10 -34
  117. package/scripts/sync-package-description.js +8 -17
  118. package/scripts/validate-vendor-online.js +13 -44
  119. package/scripts/verify-shipped-tarball.js +35 -140
@@ -1,39 +1,10 @@
1
1
  "use strict";
2
2
  /**
3
- * lib/version-pins.js
4
- *
5
- * Single source of truth for the canonical MITRE / ATT&CK / ATLAS /
6
- * D3FEND version pins that operator-facing docs reference.
7
- *
8
- * Pre-v0.13.20 history: ATLAS version was pinned to v5.4.0 in 33+
9
- * locations (READMEs, AGENTS.md, ARCHITECTURE.md, agent personas,
10
- * skill bodies, schema descriptions, manifest.json). Bumping required
11
- * a lockstep regex-replace across all 33 files. v0.13.18 bumped to
12
- * v5.6.0; the regex sweep accidentally touched dates in unrelated
13
- * paragraphs and only failed-loudly because the tests asserted
14
- * version drift. v0.13.20 makes the pin schema-driven:
15
- *
16
- * - `data/atlas-ttps.json._meta.atlas_version` is the source of truth.
17
- * - `data/attack-techniques.json._meta.attack_version` is too.
18
- * - This module reads both, exposes them via getAtlasVersion() and
19
- * getAttackVersion() helpers, and is the canonical resolver every
20
- * consumer (test runner, doc-currency check, lint, skill-body
21
- * scanner) reaches through.
22
- *
23
- * The drift-detection tests in tests/atlas-version-canonical.test.js
24
- * and tests/attack-version-canonical.test.js now compare every
25
- * operator-facing mention against the value this module returns.
26
- * A future bump is `node $(exceptd path)/lib/sign.js sign-all` + this
27
- * module reads the new value; no lockstep doc edit needed except where
28
- * the mention is
29
- * a literal-string semantic ("upgrade from v5.4.0 to v5.6.0") that an
30
- * operator must read.
31
- *
32
- * API:
33
- * getAtlasVersion() → "2026.06"
34
- * getAttackVersion() → "19.1"
35
- * getAtlasReleaseDate() → "2026-05-27"
36
- * getAllPins() → { atlas_version, atlas_release_date, attack_version, ... }
3
+ * Canonical resolver for the MITRE ATT&CK / ATLAS version pins. The `_meta`
4
+ * blocks of data/atlas-ttps.json and data/attack-techniques.json are the source
5
+ * of truth and every consumer reads them through here, so a pin bump is a data
6
+ * edit. tests/atlas-version-canonical.test.js and its ATT&CK counterpart
7
+ * compare each operator-facing mention against what this module returns.
37
8
  */
38
9
 
39
10
  const fs = require("fs");
@@ -1,28 +1,13 @@
1
1
  "use strict";
2
2
  /**
3
- * lib/worker-pool.js
3
+ * Thin wrapper over vendor/blamejs/worker-pool.js, which supplies the bounded
4
+ * concurrency, bounded queue, per-task timeout and worker recycling. Added
5
+ * here: a class around the function-style `create()`, a `runAll()` that owns
6
+ * and terminates its pool, and DEFAULT_SIZE for callers that size manually.
4
7
  *
5
- * Thin convenience wrapper over the vendored blamejs worker-pool primitive.
6
- * The vendored module (vendor/blamejs/worker-pool.js) provides:
7
- *
8
- * - bounded concurrency, defaults to max(2, cpus)
9
- * - bounded in-memory queue (default 1024 depth)
10
- * - per-task timeout (default 5min)
11
- * - worker recycle on uncaught error / timeout / exit
12
- *
13
- * What this wrapper adds:
14
- *
15
- * - WorkerPool class around the function-style `create()` API for
16
- * callers that prefer an instance
17
- * - runAll(tasks, opts) helper that runs an array of tasks through a
18
- * fresh pool and terminates it when done
19
- * - DEFAULT_SIZE re-export for callers that want to size manually
20
- *
21
- * Honest framing: at v0.7.0 corpus size (38 skills, 10 catalogs, ~150ms
22
- * total build time) worker-thread spawn cost is comparable to the work
23
- * itself. The pool is here so the architecture scales as the corpus
24
- * grows and so users can experiment with `--parallel`. Sequential builds
25
- * remain the default.
8
+ * At current corpus size a worker spawn costs about as much as the work, so
9
+ * sequential builds stay the default; the pool is here for `--parallel` and
10
+ * for the corpus growing.
26
11
  */
27
12
 
28
13
  const os = require("os");
@@ -32,9 +17,8 @@ const DEFAULT_SIZE = Math.max(1, Math.min(8, os.cpus()?.length || 4));
32
17
 
33
18
  class WorkerPool {
34
19
  /**
35
- * @param {object} opts
36
- * - runnerPath: absolute path to the worker script (required)
37
- * - size, maxQueueDepth, taskTimeoutMs, onExit — forwarded to vendored.create
20
+ * `opts.runnerPath` — absolute path to the worker script — is required; size,
21
+ * maxQueueDepth, taskTimeoutMs and onExit forward to vendored.create.
38
22
  */
39
23
  constructor(opts = {}) {
40
24
  if (!opts.runnerPath) throw new Error("WorkerPool: runnerPath is required");
@@ -56,9 +40,7 @@ class WorkerPool {
56
40
  }
57
41
  }
58
42
 
59
- /**
60
- * Run a list of tasks against a fresh pool, await all results, terminate.
61
- */
43
+ /** Runs the tasks on a fresh pool, awaits every result, then terminates it. */
62
44
  async function runAll(tasks, opts = {}) {
63
45
  const pool = new WorkerPool(opts);
64
46
  try {
@@ -72,8 +54,7 @@ module.exports = {
72
54
  WorkerPool,
73
55
  runAll,
74
56
  DEFAULT_SIZE,
75
- // Re-export vendored primitives for callers that prefer the function-style API
76
- // or need the size constants.
57
+ // For callers that prefer the function-style API or need the size constants.
77
58
  create: vendored.create,
78
59
  MIN_SIZE: vendored.MIN_SIZE,
79
60
  MAX_SIZE: vendored.MAX_SIZE,
@@ -1,49 +1,16 @@
1
1
  "use strict";
2
2
  /**
3
- * lib/xml-tokenizer.js
3
+ * Minimal XML/RSS/Atom tokenizer with no runtime dependencies.
4
4
  *
5
- * Minimal but proper XML/RSS/Atom tokenizer. Replaces the regex-based
6
- * parser in lib/source-advisories.js. The regex approach silently
7
- * failed on:
8
- * - XML namespaces (`<atom:entry>` vs `<entry>`)
9
- * - Nested CDATA
10
- * - Self-closing `<link href="..."/>` vs container `<link>...</link>`
11
- * - HTML-escaped entities inside titles
12
- * - Multi-line title content
5
+ * parseFeed(xmlString) → [{title, link, published, body}, ...]
6
+ * tokenize(xml, { onTagOpen, onTagClose, onText, onCData })
13
7
  *
14
- * Failures returned `[]` silently — operators never saw the parser was
15
- * broken on a given feed. This module fails loudly via tokenizer
16
- * errors so a parser regression is visible in the refresh report.
17
- *
18
- * Design constraints:
19
- * - Zero runtime dependencies (the project ships with no deps).
20
- * - Streaming-friendly via a callback API (does not buffer the whole
21
- * DOM — relevant for the 15 MB IETF RFC index).
22
- * - Namespace-aware via element-localname matching (the local-name
23
- * of `<atom:entry>` is `entry`).
24
- * - CDATA-correct: `<![CDATA[...]]>` content is passed through verbatim
25
- * including unescaped `<` and `&`.
26
- * - Entity-correct: the five named XML entities (lt, gt, amp, apos,
27
- * quot) plus numeric character references (`&#NNN;`, `&#xHH;`)
28
- * are decoded. Other named entities pass through unchanged (HTML
29
- * entities in RSS bodies are a recoverable variant we tolerate).
30
- *
31
- * Not designed for: DTD parsing, XInclude, XSLT, or any external-entity
32
- * resolution. Feeds that need those features are outside the scope of
33
- * a security-tooling intake pipeline.
34
- *
35
- * API:
36
- * const { parseFeed } = require("./xml-tokenizer");
37
- * const items = parseFeed(xmlString); // returns [{title, link, published, body}, ...]
38
- *
39
- * For lower-level use, the underlying tokenizer is exported too:
40
- * const { tokenize } = require("./xml-tokenizer");
41
- * tokenize(xml, { onTagOpen, onTagClose, onText, onCData });
8
+ * Streams through callbacks rather than a buffered DOM, which matters for the
9
+ * 15 MB IETF RFC index. Elements match on local-name, so `<atom:entry>` and
10
+ * `<entry>` are the same element. No DTD parsing, XInclude, XSLT, or
11
+ * external-entity resolution.
42
12
  */
43
13
 
44
- // Decode the five canonical XML entities + numeric character references.
45
- // Unknown named entities pass through unchanged (we're tolerant of
46
- // HTML-style entities that legitimately appear in RSS body text).
47
14
  function decodeEntities(s) {
48
15
  if (typeof s !== "string") return s;
49
16
  return s.replace(/&(#x[0-9a-fA-F]+|#[0-9]+|[a-zA-Z]+);/g, (m, ref) => {
@@ -65,7 +32,6 @@ function decodeEntities(s) {
65
32
  });
66
33
  }
67
34
 
68
- // Strip the optional `prefix:` from a namespaced element/attribute name.
69
35
  function localName(qname) {
70
36
  const idx = qname.indexOf(":");
71
37
  return idx === -1 ? qname : qname.slice(idx + 1);
@@ -74,8 +40,7 @@ function localName(qname) {
74
40
  function parseAttrs(rawAttrs) {
75
41
  const out = {};
76
42
  if (!rawAttrs) return out;
77
- // Walk character-by-character so quoted values can contain `=` and
78
- // whitespace without confusing a regex.
43
+ // Character-by-character, so a quoted value may contain `=` and whitespace.
79
44
  let i = 0;
80
45
  const len = rawAttrs.length;
81
46
  while (i < len) {
@@ -95,7 +60,6 @@ function parseAttrs(rawAttrs) {
95
60
  while (i < len && /\s/.test(rawAttrs[i])) i++;
96
61
  const quote = rawAttrs[i];
97
62
  if (quote !== '"' && quote !== "'") {
98
- // Unquoted value — read until whitespace or end.
99
63
  const valStart = i;
100
64
  while (i < len && !/\s/.test(rawAttrs[i])) i++;
101
65
  out[localName(name)] = decodeEntities(rawAttrs.slice(valStart, i));
@@ -111,13 +75,9 @@ function parseAttrs(rawAttrs) {
111
75
  }
112
76
 
113
77
  /**
114
- * Emit the character-data of a raw-text leaf element span `[start, end)`.
115
- *
116
- * CDATA sections inside the span are emitted verbatim (onCData if present,
117
- * else onText) exactly like the main loop's CDATA fast-path; all other bytes
118
- * — including a stray unescaped '<' or inline HTML — are entity-decoded and
119
- * emitted via onText. Keeping the CDATA contract here means a `<![CDATA[...]]>`
120
- * title still surfaces only its inner text, not the literal markers.
78
+ * Emits the character data of a raw-text leaf element span `[start, end)`. CDATA
79
+ * inside the span goes out verbatim (onCData, else onText); every other byte —
80
+ * a stray unescaped '<' or inline HTML included — is entity-decoded via onText.
121
81
  */
122
82
  function emitRawText(xml, start, end, H) {
123
83
  let p = start;
@@ -133,8 +93,7 @@ function emitRawText(xml, start, end, H) {
133
93
  if (chunk.length && H.onText) H.onText(decodeEntities(chunk));
134
94
  }
135
95
  const cend = xml.indexOf("]]>", c + 9);
136
- // findRawTextEnd already guaranteed a terminated CDATA before the close
137
- // tag, so cend is within range; guard anyway.
96
+ // findRawTextEnd guarantees a terminated CDATA here; guarded anyway.
138
97
  const inner = cend === -1 || cend > end ? xml.slice(c + 9, end) : xml.slice(c + 9, cend);
139
98
  if (H.onCData) H.onCData(inner);
140
99
  else if (H.onText) H.onText(inner);
@@ -143,17 +102,11 @@ function emitRawText(xml, start, end, H) {
143
102
  }
144
103
 
145
104
  /**
146
- * Locate the matching close tag of a raw-text leaf element.
147
- *
148
- * Scans `xml` from `start` (the byte after the leaf open tag's `>`) for the
149
- * first close tag `</...name>` whose local-name (namespace prefix stripped)
150
- * equals `name`. CDATA sections are skipped verbatim so a `</title>` that
151
- * legitimately appears inside `<![CDATA[ ... ]]>` is treated as literal text,
152
- * not as the element's terminator.
153
- *
154
- * Returns { contentEnd, next } on success — `contentEnd` is the index of the
155
- * close tag's leading `<`, `next` is the index just past its `>`. Returns null
156
- * when no matching close tag exists before EOF (genuinely truncated input).
105
+ * Finds the close tag of a raw-text leaf element, scanning from `start` (the byte
106
+ * after the open tag's `>`) for the first `</...name>` whose local-name is `name`.
107
+ * CDATA is skipped, so a `</title>` inside `<![CDATA[ ... ]]>` is literal text.
108
+ * Returns `{ contentEnd, next }` — the index of the close tag's `<`, and the index
109
+ * just past its `>` — or null when no close tag exists before EOF.
157
110
  */
158
111
  function findRawTextEnd(xml, start, name) {
159
112
  const len = xml.length;
@@ -161,8 +114,6 @@ function findRawTextEnd(xml, start, name) {
161
114
  while (j < len) {
162
115
  const lt = xml.indexOf("<", j);
163
116
  if (lt === -1) return null;
164
- // Step over a CDATA section so its contents (which may contain "</name>"
165
- // literally) cannot satisfy the close-tag match.
166
117
  if (xml.startsWith("<![CDATA[", lt)) {
167
118
  const cend = xml.indexOf("]]>", lt + 9);
168
119
  if (cend === -1) return null; // unterminated CDATA → truncated
@@ -179,16 +130,15 @@ function findRawTextEnd(xml, start, name) {
179
130
  j = gt + 1;
180
131
  continue;
181
132
  }
182
- // Any other '<' (stray literal '<', or an inner open tag like <b>) is part
183
- // of the leaf's character data — advance past it without classifying.
133
+ // Any other '<' is part of the leaf's character data — advance past it.
184
134
  j = lt + 1;
185
135
  }
186
136
  return null;
187
137
  }
188
138
 
189
139
  /**
190
- * Streaming tokenizer. Calls handlers in document order. Returns no
191
- * value — accumulation is the caller's responsibility.
140
+ * Streaming tokenizer. Calls handlers in document order. Returns no value —
141
+ * accumulation is the caller's responsibility.
192
142
  *
193
143
  * Handlers (all optional):
194
144
  * onTagOpen(name, attrs, selfClosing)
@@ -199,16 +149,9 @@ function findRawTextEnd(xml, start, name) {
199
149
  * onPI(name, content) processing instructions (<?xml-stylesheet?>)
200
150
  * onError(message, position)
201
151
  *
202
- * Options (second-position fields on the handlers object):
203
- * rawTextElements Set<string> of leaf-element local-names whose content
204
- * is #PCDATA — once such an element opens, every byte up
205
- * to the matching `</name>` close tag is treated as
206
- * character data (entities decoded, then onText), exactly
207
- * like the CDATA fast-path. This makes leaf fields tolerant
208
- * of stray unescaped '<' (e.g. "affects versions < 5.0"),
209
- * which real RSS/Atom routinely emits, instead of letting a
210
- * recoverable lexical glitch silently drop the whole field.
211
- * Structural parsing stays strict for container elements.
152
+ * `rawTextElements`, a Set of local-names on the same object, marks leaf elements
153
+ * whose content is #PCDATA: every byte up to the matching `</name>` is character
154
+ * data, so a stray unescaped '<' does not drop the field. Containers stay strict.
212
155
  */
213
156
  function tokenize(xml, handlers) {
214
157
  const H = handlers || {};
@@ -219,16 +162,11 @@ function tokenize(xml, handlers) {
219
162
  const rawTextElements = H.rawTextElements instanceof Set ? H.rawTextElements : null;
220
163
  const len = xml.length;
221
164
  let i = 0;
222
- // Open-tag stack — surfaces EOF-with-unclosed-elements as an error
223
- // instead of silently dropping the residual content. This is the
224
- // observability gap the v0.13.17 regex parser had: a malformed feed
225
- // (truncated mid-element) returned `[]` with no signal that the
226
- // parser had given up.
165
+ // The open-tag stack turns unclosed-at-EOF into an error, not dropped content.
227
166
  const openStack = [];
228
167
  while (i < len) {
229
168
  const next = xml.indexOf("<", i);
230
169
  if (next === -1) {
231
- // Trailing text — flush.
232
170
  const tail = xml.slice(i);
233
171
  if (tail.length && H.onText) H.onText(decodeEntities(tail));
234
172
  if (openStack.length && H.onError) {
@@ -240,7 +178,6 @@ function tokenize(xml, handlers) {
240
178
  const text = xml.slice(i, next);
241
179
  if (text.length && H.onText) H.onText(decodeEntities(text));
242
180
  }
243
- // Now at `<` — classify the construct.
244
181
  if (xml.startsWith("<!--", next)) {
245
182
  const end = xml.indexOf("-->", next + 4);
246
183
  if (end === -1) {
@@ -295,7 +232,6 @@ function tokenize(xml, handlers) {
295
232
  i = j + 1;
296
233
  continue;
297
234
  }
298
- // Element tag — open / close / self-closing.
299
235
  const close = xml.indexOf(">", next);
300
236
  if (close === -1) {
301
237
  if (H.onError) H.onError("unterminated element tag", next);
@@ -307,7 +243,6 @@ function tokenize(xml, handlers) {
307
243
  if (inner.startsWith("/")) { isClose = true; inner = inner.slice(1); }
308
244
  if (inner.endsWith("/")) { selfClose = true; inner = inner.slice(0, -1); }
309
245
  inner = inner.trim();
310
- // Split name and attrs at the first whitespace.
311
246
  const wsAt = inner.search(/\s/);
312
247
  const rawName = wsAt === -1 ? inner : inner.slice(0, wsAt);
313
248
  const rawAttrs = wsAt === -1 ? "" : inner.slice(wsAt + 1);
@@ -321,17 +256,11 @@ function tokenize(xml, handlers) {
321
256
  if (selfClose) {
322
257
  if (H.onTagClose) H.onTagClose(name);
323
258
  } else if (rawTextElements && rawTextElements.has(name)) {
324
- // Raw-text leaf element (title/summary/description/...). Its content
325
- // is #PCDATA: scan to the matching `</name>` and treat the whole span
326
- // as character data — only the matching close tag terminates it. Any
327
- // inner '<...>' (a stray unescaped '<', or inline HTML like <b>) is
328
- // preserved as text and normalized later by stripHtml(), instead of
329
- // being misclassified as markup and silently dropping the field.
259
+ // A raw-text leaf: only the matching `</name>` terminates it. Inner
260
+ // '<...>' stays text for stripHtml(), rather than being classified as markup.
330
261
  const span = findRawTextEnd(xml, close + 1, name);
331
262
  if (span === null) {
332
- // Genuinely truncated: the leaf never closes before EOF. Surface
333
- // the loud-error contract just like the structural path would, then
334
- // flush the residual as text and stop.
263
+ // Never closes before EOF: report as the structural path does, flush, stop.
335
264
  if (H.onError) H.onError("unterminated element at EOF: " + name, len);
336
265
  const tail = xml.slice(close + 1);
337
266
  if (tail.length && H.onText) H.onText(decodeEntities(tail));
@@ -352,39 +281,28 @@ function tokenize(xml, handlers) {
352
281
  }
353
282
  }
354
283
 
355
- // Leaf-element local-names whose content is character-data (#PCDATA). When
356
- // one of these opens, the tokenizer runs in raw-text mode until the matching
357
- // close tag, so a stray unescaped '<' inside the field (e.g. "affects
358
- // versions < 5.0") is preserved as text instead of being misclassified as
359
- // markup and silently dropping the whole field.
284
+ // The leaf elements parsed in raw-text mode — see tokenize()'s rawTextElements.
360
285
  const LEAF_FIELDS = new Set([
361
286
  "title", "link", "pubDate", "published", "updated",
362
287
  "description", "content", "summary",
363
288
  ]);
364
289
 
365
290
  // rel-rank for Atom <link> sibling selection. RFC 4287: a <link> with no rel
366
- // defaults to rel="alternate", the canonical article URL. A non-alternate rel
367
- // (self / replies / edit / enclosure) should never clobber an alternate.
291
+ // defaults to rel="alternate", so a non-alternate must never clobber one.
368
292
  function relRank(rel) {
369
293
  if (rel == null || rel === "" || String(rel).toLowerCase() === "alternate") return 2;
370
294
  return 1;
371
295
  }
372
296
 
373
297
  /**
374
- * Parse an RSS / Atom feed, always collecting parse errors.
375
- *
376
- * Returns { items, errors } where errors is an array of
377
- * { message, position } records. This is the channel a caller cannot forget
378
- * to opt into — parseFeed() below is a thin back-compat wrapper.
379
- *
380
- * items: [{ title, link, published, body }, ...]
298
+ * Parses an RSS / Atom feed into `{ items, errors }`, where items are
299
+ * `{ title, link, published, body }` and errors are `{ message, position }`.
300
+ * The errors channel cannot be opted out of, so prefer this over parseFeed().
381
301
  */
382
302
  function parseFeedDetailed(xml) {
383
303
  const items = [];
384
304
  const errors = [];
385
- // Stack of "in-progress item" contexts. RSS uses <item>; Atom uses
386
- // <entry>; both nest title / link / pubDate / published / updated /
387
- // description / content / summary.
305
+ // RSS uses <item>, Atom uses <entry>; both nest the same leaf fields.
388
306
  const ITEM_LOCALS = new Set(["item", "entry"]);
389
307
  const FIELD_MAP = {
390
308
  title: "title",
@@ -414,23 +332,17 @@ function parseFeedDetailed(xml) {
414
332
  if (FIELD_MAP[name]) {
415
333
  activeField = FIELD_MAP[name];
416
334
  buffer = "";
417
- // Remember this link element's rel so its element-text close (below) is
418
- // ranked the same way the attribute path is — a non-alternate link with
419
- // text must not clobber a captured rel=alternate.
335
+ // Remembered so the close below ranks the link as this path does.
420
336
  if (name === "link") linkRel = attrs ? attrs.rel : null;
421
- // Atom <link href="..."/> — rel-aware selection. Only let an
422
- // alternate / rel-absent link upgrade the captured value, and never
423
- // let a non-alternate rel (self/replies/edit) clobber an alternate
424
- // already in hand. First-alternate-wins, independent of document
425
- // order.
337
+ // Only an alternate or rel-absent link upgrades the captured value,
338
+ // whatever the document order.
426
339
  if (name === "link" && attrs && attrs.href) {
427
340
  const r = relRank(attrs.rel);
428
341
  if (r > linkRank) {
429
342
  current.link = attrs.href;
430
343
  linkRank = r;
431
344
  }
432
- // The link value has been resolved from the attribute — there is
433
- // no element-text close to wait for.
345
+ // Resolved from the attribute — no element-text close to wait for.
434
346
  if (selfClosing) activeField = null;
435
347
  }
436
348
  }
@@ -447,12 +359,8 @@ function parseFeedDetailed(xml) {
447
359
  if (!current) return;
448
360
  if (FIELD_MAP[name] && activeField === FIELD_MAP[name]) {
449
361
  const value = buffer.trim();
450
- // <link>...</link> element text — ranked by the link's rel, the SAME
451
- // gate the attribute path uses, so a non-alternate (self/replies/edit)
452
- // link with text cannot clobber a captured rel=alternate. RSS links
453
- // carry no rel, and relRank(undefined) === 2, so RSS element-text links
454
- // keep their previous authoritative rank. An empty element-text link
455
- // falls back to whatever attribute capture onTagOpen already recorded.
362
+ // Element text goes through the same rel gate as the attribute path; RSS
363
+ // links carry no rel, so relRank(undefined) is 2 and they keep the top rank.
456
364
  if (name === "link") {
457
365
  const r = relRank(linkRel);
458
366
  if (value && r > linkRank) {
@@ -461,12 +369,8 @@ function parseFeedDetailed(xml) {
461
369
  }
462
370
  linkRel = null;
463
371
  } else if (activeField === "body" || activeField === "title") {
464
- // Strip HTML tags from title + description / content / summary.
465
- // Many feeds embed inline HTML (<b>, <em>, <a>) in titles for
466
- // emphasis; the operational consumer wants plain text. CDATA
467
- // content reaches here verbatim, so this also strips HTML
468
- // that was wrapped in CDATA to dodge entity-encoding. A stray
469
- // unescaped '<' that survived raw-text mode collapses here too.
372
+ // Feeds embed inline HTML in titles and bodies. CDATA reaches here
373
+ // verbatim, so HTML wrapped in CDATA to dodge encoding is stripped too.
470
374
  current[activeField] = stripHtml(value);
471
375
  } else {
472
376
  current[activeField] = value;
@@ -490,13 +394,9 @@ function parseFeedDetailed(xml) {
490
394
  }
491
395
 
492
396
  /**
493
- * Parse an RSS / Atom feed into a flat array of items. Returns:
494
- * [{ title, link, published, body }, ...]
495
- *
496
- * Empty array on parse failure. Errors are ALWAYS collected internally; the
497
- * optional `errors` array is filled for callers that pass one (back-compat).
498
- * Prefer parseFeedDetailed(xml) for new callers — it returns the errors
499
- * channel unconditionally so it cannot be silently dropped.
397
+ * Parses an RSS / Atom feed into `[{ title, link, published, body }, ...]`, or an
398
+ * empty array on parse failure. Errors are copied into the optional `errors` array
399
+ * when a caller passes one and dropped otherwise: prefer parseFeedDetailed().
500
400
  */
501
401
  function parseFeed(xml, errors = null) {
502
402
  const { items, errors: collected } = parseFeedDetailed(xml);
@@ -508,14 +408,9 @@ function parseFeed(xml, errors = null) {
508
408
 
509
409
  function stripHtml(s) {
510
410
  if (typeof s !== "string") return "";
511
- // Single-pass tag strip. A backtracking /<[^>]+>/g is O(n^2) on
512
- // attacker-controlled feed text with many '<' and no '>' (a ~150 KB
513
- // <title>/<description> of bare '<' froze the refresh worker for seconds —
514
- // a ReDoS-class DoS, since parseFeed() runs on network-fetched RSS/Atom
515
- // bodies). Walk '<'..'>' instead, preserving the prior contract exactly: a
516
- // real `<...>` becomes a space; a stray '<' with no closing '>' and an empty
517
- // '<>' survive as literal text (the `[^>]+` quantifier required >=1 char, so
518
- // neither matched). Then collapse whitespace as before.
411
+ // Single-pass, because a backtracking /<[^>]+>/g is O(n^2) on feed text with many
412
+ // '<' and no '>' — a ReDoS-class DoS, since parseFeed() runs on network-fetched
413
+ // bodies. A stray '<' with no closing '>' and an empty '<>' survive as text.
519
414
  let out = "";
520
415
  let i = 0;
521
416
  while (i < s.length) {