@blamejs/exceptd-skills 0.18.9 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/CHANGELOG.md +36 -0
  2. package/bin/exceptd.js +204 -118
  3. package/data/_indexes/_meta.json +3 -3
  4. package/data/_indexes/frequency.json +2 -2
  5. package/data/d3fend-catalog.json +6 -6
  6. package/data/playbooks/identity-sso-compromise.json +2 -2
  7. package/data/playbooks/sbom.json +1 -1
  8. package/lib/citation-resolve.js +11 -0
  9. package/lib/collectors/containers.js +13 -0
  10. package/lib/collectors/cred-stores.js +18 -9
  11. package/lib/collectors/secrets.js +4 -2
  12. package/lib/cross-ref-api.js +29 -7
  13. package/lib/cve-regression-watcher.js +47 -15
  14. package/lib/framework-gap.js +52 -19
  15. package/lib/gap-detectors.js +8 -3
  16. package/lib/lint-skills.js +3 -2
  17. package/lib/playbook-runner.js +125 -7
  18. package/lib/refresh-external.js +58 -7
  19. package/lib/refresh-network.js +18 -5
  20. package/lib/rfc-cli.js +113 -18
  21. package/lib/schemas/playbook.schema.json +1 -1
  22. package/lib/scoring.js +71 -8
  23. package/lib/source-advisories.js +58 -9
  24. package/lib/ttp-mapper.js +31 -3
  25. package/lib/upstream-check-cli.js +13 -1
  26. package/lib/validate-catalog-meta.js +51 -7
  27. package/lib/validate-cve-catalog.js +10 -0
  28. package/lib/validate-playbooks.js +19 -1
  29. package/lib/verify.js +35 -34
  30. package/lib/xml-tokenizer.js +187 -25
  31. package/manifest.json +53 -53
  32. package/orchestrator/dispatcher.js +53 -9
  33. package/orchestrator/index.js +9 -7
  34. package/orchestrator/pipeline.js +62 -14
  35. package/orchestrator/scanner.js +60 -9
  36. package/package.json +1 -1
  37. package/sbom.cdx.json +115 -100
  38. package/scripts/build-indexes.js +21 -3
  39. package/scripts/builders/cwe-chains.js +5 -2
  40. package/scripts/builders/section-offsets.js +17 -8
  41. package/scripts/builders/summary-cards.js +12 -4
  42. package/scripts/check-catalog-gap-budget.js +3 -3
  43. package/scripts/check-codebase-patterns-currency.js +1 -0
  44. package/scripts/check-codebase-patterns.js +166 -11
  45. package/scripts/check-sbom-currency.js +69 -3
  46. package/scripts/check-test-count.js +28 -16
  47. package/scripts/check-test-subjects.js +148 -0
  48. package/scripts/check-version-tags.js +24 -5
  49. package/scripts/predeploy.js +32 -8
  50. package/scripts/refresh-upstream-catalogs.js +169 -44
  51. package/scripts/release.js +28 -11
@@ -85,6 +85,29 @@ function containsPlaceholder(s) {
85
85
  return PLACEHOLDER_TOKENS.some((re) => re.test(s));
86
86
  }
87
87
 
88
+ // Round-trip ISO calendar-date check. Returns a Date for a real YYYY-MM-DD
89
+ // calendar date, or null for anything malformed. Unlike a shape-only regex,
90
+ // this rejects impossible dates (2026-13-99, 2026-04-31, 2026-02-29 in a
91
+ // non-leap year): `new Date('2026-02-30T00:00:00Z')` does NOT throw — it rolls
92
+ // over to March 2 with a valid getTime() — so the parsed Y-M-D must round-trip
93
+ // back to the input components. Deliberately carries NO year-floor business
94
+ // rule (a valid-but-old 1900-01-01 stays a valid date so the staleness branch,
95
+ // not the validity branch, reports it).
96
+ function parseIsoDateStrict(v) {
97
+ if (typeof v !== 'string' || !/^\d{4}-\d{2}-\d{2}$/.test(v)) return null;
98
+ const d = new Date(v + 'T00:00:00Z');
99
+ if (Number.isNaN(d.getTime())) return null;
100
+ const [y, m, day] = v.split('-').map(Number);
101
+ if (
102
+ d.getUTCFullYear() !== y ||
103
+ d.getUTCMonth() + 1 !== m ||
104
+ d.getUTCDate() !== day
105
+ ) {
106
+ return null;
107
+ }
108
+ return d;
109
+ }
110
+
88
111
  function validateMeta(catalogPath, opts) {
89
112
  const errors = [];
90
113
  const warnings = [];
@@ -92,7 +115,16 @@ function validateMeta(catalogPath, opts) {
92
115
  const meta = data._meta;
93
116
 
94
117
  if (!meta || typeof meta !== 'object') {
95
- return ['missing _meta block'];
118
+ // Honor both return contracts. The rest of the body dereferences
119
+ // `meta.tlp` / `meta.source_confidence` / `meta.freshness_policy`, so it
120
+ // must NOT run when `meta` is absent or non-object. Return early in the
121
+ // SAME shape the caller asked for: includeWarnings callers (main()) get
122
+ // `{errors, warnings}` so `result.errors` is a real array and the loop
123
+ // reports a clean FAIL + continues; no-opts callers still get a non-empty
124
+ // `string[]`. This still FAILS — it only removes the uncaught TypeError.
125
+ errors.push('missing _meta block');
126
+ if (opts && opts.includeWarnings) return { errors, warnings };
127
+ return errors;
96
128
  }
97
129
 
98
130
  /* tlp */
@@ -176,14 +208,26 @@ function validateMeta(catalogPath, opts) {
176
208
  * the warning posture.
177
209
  */
178
210
  if (
179
- typeof meta.last_updated === 'string' &&
211
+ meta.last_updated !== undefined &&
180
212
  typeof fp.stale_after_days === 'number' &&
181
213
  fp.stale_after_days > 0
182
214
  ) {
183
- const lu = new Date(meta.last_updated + (
184
- /^\d{4}-\d{2}-\d{2}$/.test(meta.last_updated) ? 'T00:00:00Z' : ''
185
- ));
186
- if (!Number.isNaN(lu.getTime())) {
215
+ const lu = parseIsoDateStrict(meta.last_updated);
216
+ if (lu === null) {
217
+ // Fail-closed on a malformed date instead of silently skipping the
218
+ // freshness gate. A NaN/impossible/wrong-shape last_updated is
219
+ // "invalid input" (error under --strict, warning by default), NOT
220
+ // "no opinion" — otherwise the staleness check fails open.
221
+ const msg =
222
+ `_meta.last_updated ${JSON.stringify(meta.last_updated)} is not a valid ISO date ` +
223
+ `(YYYY-MM-DD calendar date) — cannot evaluate freshness. ` +
224
+ `Promoted to an error under --strict.`;
225
+ if (opts && (opts.strict || opts.errorOnStale)) {
226
+ errors.push(msg);
227
+ } else {
228
+ warnings.push(msg);
229
+ }
230
+ } else {
187
231
  const ageDays = Math.floor((Date.now() - lu.getTime()) / 86400000);
188
232
  if (ageDays > fp.stale_after_days) {
189
233
  const msg =
@@ -255,4 +299,4 @@ if (require.main === module) {
255
299
  main();
256
300
  }
257
301
 
258
- module.exports = { validateMeta };
302
+ module.exports = { validateMeta, parseIsoDateStrict };
@@ -244,6 +244,16 @@ function isUsableDate(value) {
244
244
  function additionalChecks(key, entry, ctx) {
245
245
  const warnings = [];
246
246
 
247
+ // A non-object entry has no checkable sub-fields, and validate() already
248
+ // emits the top-level type error for it (lines 127-133). Guarding here turns
249
+ // the uncaught `entry.poc_available` TypeError on a null/array entry into a
250
+ // clean no-op so main() still prints that FAIL and continues to later
251
+ // entries instead of aborting the whole gate. The FAIL is preserved — it
252
+ // originates in validate(), not here.
253
+ if (!entry || typeof entry !== 'object' || Array.isArray(entry)) {
254
+ return [];
255
+ }
256
+
247
257
  // V1 — Hard Rule #14 conditional: poc + public-exploit URL → iocs required.
248
258
  if (entry.poc_available === true) {
249
259
  const sources = Array.isArray(entry.verification_sources)
@@ -322,6 +322,16 @@ function obligationKey(o) {
322
322
 
323
323
  function checkCrossRefs(playbook, ctx, playbookIds) {
324
324
  const findings = [];
325
+ // A null/array/primitive playbook has no cross-refs to check, and validate()
326
+ // already emits the top-level `expected type "object", got null` error for
327
+ // it (main() line ~755). Guarding here turns the uncaught `playbook._meta`
328
+ // TypeError on a literal-null playbook file into a clean no-op so main()
329
+ // still reports the FAIL and continues to the remaining playbooks instead of
330
+ // aborting the whole gate. The FAIL is preserved — it originates in
331
+ // validate(), not here.
332
+ if (!playbook || typeof playbook !== 'object' || Array.isArray(playbook)) {
333
+ return findings;
334
+ }
325
335
  const meta = playbook._meta || {};
326
336
  const phases = playbook.phases || {};
327
337
  const domain = playbook.domain || {};
@@ -533,7 +543,15 @@ function checkCrossRefs(playbook, ctx, playbookIds) {
533
543
  // Case-insensitive + word-bounded so `HTTPS://`, `Curl`, and `fetch(` (no
534
544
  // trailing space) still flag a network source — otherwise an artifact could
535
545
  // ship under air_gap_mode with no offline alternative and run incomplete.
536
- const netSourceRe = /(https?:\/\/|\bgh (?:api|release)\b|\bcurl\b|\bwget\b|\bfetch\b)/i;
546
+ // Network-source detection includes API-verb-phrased sources ("GET
547
+ // /directoryRoles via Graph", "Entra ID", "Okta", "Microsoft Graph") so a
548
+ // REST/Graph endpoint described in prose still flags under air_gap_mode and
549
+ // is not silently collected offline-incomplete. `api/v\d` is deliberately
550
+ // NOT a token — it false-positives on local code-scan artifacts that merely
551
+ // reference an API path. NOTE: lib/schemas/playbook.schema.json carries the
552
+ // same narrow `source` pattern and must be broadened in lockstep with this
553
+ // regex (main-thread item — that file is not edited here).
554
+ const netSourceRe = /(https?:\/\/|\bgh (?:api|release)\b|\bcurl\b|\bwget\b|\bfetch\b|\b(?:GET|POST|PUT|PATCH|DELETE)\s+\/|\bGraph\b|\b(?:Okta|Entra ID|Microsoft Graph)\b)/i;
537
555
  for (const [i, art] of (look.artifacts || []).entries()) {
538
556
  if (!art || typeof art !== 'object') continue;
539
557
  if (typeof art.source === 'string' && netSourceRe.test(art.source)) {
package/lib/verify.js CHANGED
@@ -342,6 +342,39 @@ function canonicalManifestBytes(manifest) {
342
342
  * @param {object} manifest
343
343
  */
344
344
  function verifyManifestSignature(manifest) {
345
+ // The key-pin fingerprint check runs FIRST — independent of whether a
346
+ // manifest_signature is present — so library callers (refresh-network gate,
347
+ // verify-shipped-tarball gate, tests, downstream `require("lib/verify")`
348
+ // consumers) cannot bypass the pin. Previously the pin only fired AFTER the
349
+ // signature-present check, so a key-substitution attacker who swapped
350
+ // keys/public.pem AND stripped manifest_signature got the early `missing`
351
+ // return and never tripped the pin — authenticating against the attacker key
352
+ // through the library API. Consulting the pin up front closes that on the
353
+ // legacy/missing path too. Honors KEYS_ROTATED=1 for legitimate rotations; a
354
+ // MISSING pin file fails closed (keys/EXPECTED_FINGERPRINT ships in the
355
+ // tarball and is committed, so its absence is the signature of a tamper that
356
+ // stripped the pin to hide a swapped key).
357
+ const publicKey = loadPublicKey();
358
+ if (publicKey) {
359
+ const liveFp = publicKeyFingerprint(publicKey);
360
+ const pinResult = checkExpectedFingerprint(liveFp);
361
+ if (pinResult.status === 'mismatch' && !pinResult.rotationOverride) {
362
+ return {
363
+ status: 'invalid',
364
+ reason: `fingerprint-mismatch: live=${pinResult.actual} pin=${pinResult.expected} — keys/public.pem does not match keys/EXPECTED_FINGERPRINT. If this is an intentional rotation, set KEYS_ROTATED=1 and update the pin.`,
365
+ fingerprint_mismatch: true,
366
+ expected: pinResult.expected,
367
+ actual: pinResult.actual,
368
+ };
369
+ }
370
+ if (pinResult.status === 'no-pin') {
371
+ return {
372
+ status: 'invalid',
373
+ reason: `key-pin absent: keys/EXPECTED_FINGERPRINT is missing, so a swapped keys/public.pem cannot be detected. The pin ships in the tarball and is committed — restore it from the package or version control.`,
374
+ pin_absent: true,
375
+ };
376
+ }
377
+ }
345
378
  const sig = manifest && manifest.manifest_signature;
346
379
  if (!sig || typeof sig !== 'object') return { status: 'missing' };
347
380
  if (typeof sig.signature_base64 !== 'string') {
@@ -359,43 +392,11 @@ function verifyManifestSignature(manifest) {
359
392
  reason: `manifest_signature.algorithm must be exactly 'Ed25519' (got ${JSON.stringify(sig.algorithm)})`,
360
393
  };
361
394
  }
362
- const publicKey = loadPublicKey();
363
395
  if (!publicKey) {
364
396
  return { status: 'no-key', reason: 'public key missing at keys/public.pem' };
365
397
  }
366
- // consult keys/EXPECTED_FINGERPRINT BEFORE crypto.verify so
367
- // library callers (refresh-network gate, verify-shipped-tarball gate, tests,
368
- // downstream consumers via `require("lib/verify")`) cannot bypass the pin.
369
- // Previously the pin only fired at the CLI tail of `node lib/verify.js`,
370
- // letting a coordinated attacker who swapped keys/public.pem authenticate
371
- // against the attacker key without any divergence surfaced through the
372
- // library API. Honors KEYS_ROTATED=1 for legitimate rotations. A MISSING pin
373
- // file is now rejected too: keys/EXPECTED_FINGERPRINT ships in the tarball and
374
- // is committed to the repo, so its absence is not a legacy state — it is the
375
- // signature a key-substitution attack leaves when it strips the pin to hide a
376
- // swapped keys/public.pem.
377
- const liveFp = publicKeyFingerprint(publicKey);
378
- const pinResult = checkExpectedFingerprint(liveFp);
379
- if (pinResult.status === 'mismatch' && !pinResult.rotationOverride) {
380
- return {
381
- status: 'invalid',
382
- reason: `fingerprint-mismatch: live=${pinResult.actual} pin=${pinResult.expected} — keys/public.pem does not match keys/EXPECTED_FINGERPRINT. If this is an intentional rotation, set KEYS_ROTATED=1 and update the pin.`,
383
- fingerprint_mismatch: true,
384
- expected: pinResult.expected,
385
- actual: pinResult.actual,
386
- };
387
- }
388
- if (pinResult.status === 'no-pin') {
389
- // A missing pin fails closed unconditionally — there is nothing to override.
390
- // KEYS_ROTATED only applies to a fingerprint MISMATCH (a new key + a new
391
- // pin); a legitimate rotation updates the pin in place, it never removes it,
392
- // so an absent pin is treated as tampering and restoring it is the only fix.
393
- return {
394
- status: 'invalid',
395
- reason: `key-pin absent: keys/EXPECTED_FINGERPRINT is missing, so a swapped keys/public.pem cannot be detected. The pin ships in the tarball and is committed — restore it from the package or version control.`,
396
- pin_absent: true,
397
- };
398
- }
398
+ // The key-pin (mismatch / no-pin) was already verified up front, before any
399
+ // signature branching — so reaching here means the live key matches the pin.
399
400
  let signatureBytes;
400
401
  try {
401
402
  signatureBytes = Buffer.from(sig.signature_base64, 'base64');
@@ -110,6 +110,82 @@ function parseAttrs(rawAttrs) {
110
110
  return out;
111
111
  }
112
112
 
113
+ /**
114
+ * Emit the character-data of a raw-text leaf element span `[start, end)`.
115
+ *
116
+ * CDATA sections inside the span are emitted verbatim (onCData if present,
117
+ * else onText) exactly like the main loop's CDATA fast-path; all other bytes
118
+ * — including a stray unescaped '<' or inline HTML — are entity-decoded and
119
+ * emitted via onText. Keeping the CDATA contract here means a `<![CDATA[...]]>`
120
+ * title still surfaces only its inner text, not the literal markers.
121
+ */
122
+ function emitRawText(xml, start, end, H) {
123
+ let p = start;
124
+ while (p < end) {
125
+ const c = xml.indexOf("<![CDATA[", p);
126
+ if (c === -1 || c >= end) {
127
+ const chunk = xml.slice(p, end);
128
+ if (chunk.length && H.onText) H.onText(decodeEntities(chunk));
129
+ return;
130
+ }
131
+ if (c > p) {
132
+ const chunk = xml.slice(p, c);
133
+ if (chunk.length && H.onText) H.onText(decodeEntities(chunk));
134
+ }
135
+ const cend = xml.indexOf("]]>", c + 9);
136
+ // findRawTextEnd already guaranteed a terminated CDATA before the close
137
+ // tag, so cend is within range; guard anyway.
138
+ const inner = cend === -1 || cend > end ? xml.slice(c + 9, end) : xml.slice(c + 9, cend);
139
+ if (H.onCData) H.onCData(inner);
140
+ else if (H.onText) H.onText(inner);
141
+ p = cend === -1 || cend > end ? end : cend + 3;
142
+ }
143
+ }
144
+
145
+ /**
146
+ * Locate the matching close tag of a raw-text leaf element.
147
+ *
148
+ * Scans `xml` from `start` (the byte after the leaf open tag's `>`) for the
149
+ * first close tag `</...name>` whose local-name (namespace prefix stripped)
150
+ * equals `name`. CDATA sections are skipped verbatim so a `</title>` that
151
+ * legitimately appears inside `<![CDATA[ ... ]]>` is treated as literal text,
152
+ * not as the element's terminator.
153
+ *
154
+ * Returns { contentEnd, next } on success — `contentEnd` is the index of the
155
+ * close tag's leading `<`, `next` is the index just past its `>`. Returns null
156
+ * when no matching close tag exists before EOF (genuinely truncated input).
157
+ */
158
+ function findRawTextEnd(xml, start, name) {
159
+ const len = xml.length;
160
+ let j = start;
161
+ while (j < len) {
162
+ const lt = xml.indexOf("<", j);
163
+ if (lt === -1) return null;
164
+ // Step over a CDATA section so its contents (which may contain "</name>"
165
+ // literally) cannot satisfy the close-tag match.
166
+ if (xml.startsWith("<![CDATA[", lt)) {
167
+ const cend = xml.indexOf("]]>", lt + 9);
168
+ if (cend === -1) return null; // unterminated CDATA → truncated
169
+ j = cend + 3;
170
+ continue;
171
+ }
172
+ if (xml[lt + 1] === "/") {
173
+ const gt = xml.indexOf(">", lt + 2);
174
+ if (gt === -1) return null;
175
+ const closeName = localName(xml.slice(lt + 2, gt).trim());
176
+ if (closeName === name) {
177
+ return { contentEnd: lt, next: gt + 1 };
178
+ }
179
+ j = gt + 1;
180
+ continue;
181
+ }
182
+ // Any other '<' (stray literal '<', or an inner open tag like <b>) is part
183
+ // of the leaf's character data — advance past it without classifying.
184
+ j = lt + 1;
185
+ }
186
+ return null;
187
+ }
188
+
113
189
  /**
114
190
  * Streaming tokenizer. Calls handlers in document order. Returns no
115
191
  * value — accumulation is the caller's responsibility.
@@ -122,6 +198,17 @@ function parseAttrs(rawAttrs) {
122
198
  * onComment(text)
123
199
  * onPI(name, content) processing instructions (<?xml-stylesheet?>)
124
200
  * onError(message, position)
201
+ *
202
+ * Options (second-position fields on the handlers object):
203
+ * rawTextElements Set<string> of leaf-element local-names whose content
204
+ * is #PCDATA — once such an element opens, every byte up
205
+ * to the matching `</name>` close tag is treated as
206
+ * character data (entities decoded, then onText), exactly
207
+ * like the CDATA fast-path. This makes leaf fields tolerant
208
+ * of stray unescaped '<' (e.g. "affects versions < 5.0"),
209
+ * which real RSS/Atom routinely emits, instead of letting a
210
+ * recoverable lexical glitch silently drop the whole field.
211
+ * Structural parsing stays strict for container elements.
125
212
  */
126
213
  function tokenize(xml, handlers) {
127
214
  const H = handlers || {};
@@ -129,6 +216,7 @@ function tokenize(xml, handlers) {
129
216
  if (H.onError) H.onError("input must be a string", 0);
130
217
  return;
131
218
  }
219
+ const rawTextElements = H.rawTextElements instanceof Set ? H.rawTextElements : null;
132
220
  const len = xml.length;
133
221
  let i = 0;
134
222
  // Open-tag stack — surfaces EOF-with-unclosed-elements as an error
@@ -229,9 +317,33 @@ function tokenize(xml, handlers) {
229
317
  if (H.onTagClose) H.onTagClose(name);
230
318
  } else {
231
319
  const attrs = parseAttrs(rawAttrs);
232
- if (!selfClose) openStack.push(name);
233
320
  if (H.onTagOpen) H.onTagOpen(name, attrs, selfClose);
234
- if (selfClose && H.onTagClose) H.onTagClose(name);
321
+ if (selfClose) {
322
+ if (H.onTagClose) H.onTagClose(name);
323
+ } else if (rawTextElements && rawTextElements.has(name)) {
324
+ // Raw-text leaf element (title/summary/description/...). Its content
325
+ // is #PCDATA: scan to the matching `</name>` and treat the whole span
326
+ // as character data — only the matching close tag terminates it. Any
327
+ // inner '<...>' (a stray unescaped '<', or inline HTML like <b>) is
328
+ // preserved as text and normalized later by stripHtml(), instead of
329
+ // being misclassified as markup and silently dropping the field.
330
+ const span = findRawTextEnd(xml, close + 1, name);
331
+ if (span === null) {
332
+ // Genuinely truncated: the leaf never closes before EOF. Surface
333
+ // the loud-error contract just like the structural path would, then
334
+ // flush the residual as text and stop.
335
+ if (H.onError) H.onError("unterminated element at EOF: " + name, len);
336
+ const tail = xml.slice(close + 1);
337
+ if (tail.length && H.onText) H.onText(decodeEntities(tail));
338
+ return;
339
+ }
340
+ emitRawText(xml, close + 1, span.contentEnd, H);
341
+ if (H.onTagClose) H.onTagClose(name);
342
+ i = span.next;
343
+ continue;
344
+ } else {
345
+ openStack.push(name);
346
+ }
235
347
  }
236
348
  i = close + 1;
237
349
  }
@@ -240,15 +352,36 @@ function tokenize(xml, handlers) {
240
352
  }
241
353
  }
242
354
 
355
+ // Leaf-element local-names whose content is character-data (#PCDATA). When
356
+ // one of these opens, the tokenizer runs in raw-text mode until the matching
357
+ // close tag, so a stray unescaped '<' inside the field (e.g. "affects
358
+ // versions < 5.0") is preserved as text instead of being misclassified as
359
+ // markup and silently dropping the whole field.
360
+ const LEAF_FIELDS = new Set([
361
+ "title", "link", "pubDate", "published", "updated",
362
+ "description", "content", "summary",
363
+ ]);
364
+
365
+ // rel-rank for Atom <link> sibling selection. RFC 4287: a <link> with no rel
366
+ // defaults to rel="alternate", the canonical article URL. A non-alternate rel
367
+ // (self / replies / edit / enclosure) should never clobber an alternate.
368
+ function relRank(rel) {
369
+ if (rel == null || rel === "" || String(rel).toLowerCase() === "alternate") return 2;
370
+ return 1;
371
+ }
372
+
243
373
  /**
244
- * Parse an RSS / Atom feed into a flat array of items. Returns:
245
- * [{ title, link, published, body, raw_attrs: {...} }, ...]
374
+ * Parse an RSS / Atom feed, always collecting parse errors.
375
+ *
376
+ * Returns { items, errors } where errors is an array of
377
+ * { message, position } records. This is the channel a caller cannot forget
378
+ * to opt into — parseFeed() below is a thin back-compat wrapper.
246
379
  *
247
- * Empty array on parse failure. `errors` (out-of-band) captured via
248
- * the optional `errors` array — callers wanting observability pass it.
380
+ * items: [{ title, link, published, body }, ...]
249
381
  */
250
- function parseFeed(xml, errors = null) {
382
+ function parseFeedDetailed(xml) {
251
383
  const items = [];
384
+ const errors = [];
252
385
  // Stack of "in-progress item" contexts. RSS uses <item>; Atom uses
253
386
  // <entry>; both nest title / link / pubDate / published / updated /
254
387
  // description / content / summary.
@@ -266,25 +399,34 @@ function parseFeed(xml, errors = null) {
266
399
  let current = null; // active item context
267
400
  let activeField = null; // active field local-name
268
401
  let buffer = ""; // accumulator for current field text
269
- let linkHref = null; // captured from <link href="..."/> attribute
402
+ let linkRank = 0; // rel-rank of the best <link> captured so far
270
403
 
271
404
  tokenize(xml, {
405
+ rawTextElements: LEAF_FIELDS,
272
406
  onTagOpen(name, attrs, selfClosing) {
273
407
  if (ITEM_LOCALS.has(name)) {
274
408
  current = { title: "", link: "", published: "", body: "" };
409
+ linkRank = 0;
275
410
  return;
276
411
  }
277
412
  if (!current) return;
278
413
  if (FIELD_MAP[name]) {
279
414
  activeField = FIELD_MAP[name];
280
415
  buffer = "";
281
- // Atom <link href="..."/> — capture the href attribute as the
282
- // link value. RSS <link>...</link> uses element text instead.
283
- if (name === "link" && attrs && attrs.href) linkHref = attrs.href;
284
- if (selfClosing && name === "link" && linkHref) {
285
- current.link = linkHref;
286
- linkHref = null;
287
- activeField = null;
416
+ // Atom <link href="..."/> — rel-aware selection. Only let an
417
+ // alternate / rel-absent link upgrade the captured value, and never
418
+ // let a non-alternate rel (self/replies/edit) clobber an alternate
419
+ // already in hand. First-alternate-wins, independent of document
420
+ // order.
421
+ if (name === "link" && attrs && attrs.href) {
422
+ const r = relRank(attrs.rel);
423
+ if (r > linkRank) {
424
+ current.link = attrs.href;
425
+ linkRank = r;
426
+ }
427
+ // The link value has been resolved from the attribute — there is
428
+ // no element-text close to wait for.
429
+ if (selfClosing) activeField = null;
288
430
  }
289
431
  }
290
432
  },
@@ -294,28 +436,31 @@ function parseFeed(xml, errors = null) {
294
436
  current = null;
295
437
  activeField = null;
296
438
  buffer = "";
439
+ linkRank = 0;
297
440
  return;
298
441
  }
299
442
  if (!current) return;
300
443
  if (FIELD_MAP[name] && activeField === FIELD_MAP[name]) {
301
444
  const value = buffer.trim();
302
- // Element-text link overrides the attribute capture when
303
- // both are present.
304
- if (name === "link" && value) {
305
- current.link = value;
306
- } else if (name === "link" && !value && linkHref) {
307
- current.link = linkHref;
445
+ // RSS <link>...</link> — element text is authoritative (rank 2),
446
+ // unconditional. An empty element-text link falls back to whatever
447
+ // attribute capture onTagOpen already recorded.
448
+ if (name === "link") {
449
+ if (value) {
450
+ current.link = value;
451
+ linkRank = 2;
452
+ }
308
453
  } else if (activeField === "body" || activeField === "title") {
309
454
  // Strip HTML tags from title + description / content / summary.
310
455
  // Many feeds embed inline HTML (<b>, <em>, <a>) in titles for
311
456
  // emphasis; the operational consumer wants plain text. CDATA
312
457
  // content reaches here verbatim, so this also strips HTML
313
- // that was wrapped in CDATA to dodge entity-encoding.
458
+ // that was wrapped in CDATA to dodge entity-encoding. A stray
459
+ // unescaped '<' that survived raw-text mode collapses here too.
314
460
  current[activeField] = stripHtml(value);
315
461
  } else {
316
462
  current[activeField] = value;
317
463
  }
318
- linkHref = null;
319
464
  activeField = null;
320
465
  buffer = "";
321
466
  }
@@ -327,10 +472,27 @@ function parseFeed(xml, errors = null) {
327
472
  if (activeField) buffer += text;
328
473
  },
329
474
  onError(msg, pos) {
330
- if (errors) errors.push({ message: msg, position: pos });
475
+ errors.push({ message: msg, position: pos });
331
476
  }
332
477
  });
333
478
 
479
+ return { items, errors };
480
+ }
481
+
482
+ /**
483
+ * Parse an RSS / Atom feed into a flat array of items. Returns:
484
+ * [{ title, link, published, body }, ...]
485
+ *
486
+ * Empty array on parse failure. Errors are ALWAYS collected internally; the
487
+ * optional `errors` array is filled for callers that pass one (back-compat).
488
+ * Prefer parseFeedDetailed(xml) for new callers — it returns the errors
489
+ * channel unconditionally so it cannot be silently dropped.
490
+ */
491
+ function parseFeed(xml, errors = null) {
492
+ const { items, errors: collected } = parseFeedDetailed(xml);
493
+ if (Array.isArray(errors)) {
494
+ for (const e of collected) errors.push(e);
495
+ }
334
496
  return items;
335
497
  }
336
498
 
@@ -341,4 +503,4 @@ function stripHtml(s) {
341
503
  return s.replace(/<[^>]+>/g, " ").replace(/\s+/g, " ").trim();
342
504
  }
343
505
 
344
- module.exports = { tokenize, parseFeed, decodeEntities, localName, parseAttrs, stripHtml };
506
+ module.exports = { tokenize, parseFeed, parseFeedDetailed, decodeEntities, localName, parseAttrs, stripHtml };