@blamejs/exceptd-skills 0.18.8 → 0.18.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/bin/exceptd.js +197 -118
- package/data/_indexes/_meta.json +3 -3
- package/data/_indexes/frequency.json +2 -2
- package/data/d3fend-catalog.json +6 -6
- package/data/playbooks/identity-sso-compromise.json +2 -2
- package/data/playbooks/sbom.json +1 -1
- package/lib/citation-resolve.js +11 -0
- package/lib/collectors/containers.js +13 -0
- package/lib/cross-ref-api.js +29 -7
- package/lib/cve-regression-watcher.js +47 -15
- package/lib/framework-gap.js +27 -5
- package/lib/gap-detectors.js +8 -3
- package/lib/lint-skills.js +3 -2
- package/lib/playbook-runner.js +60 -5
- package/lib/refresh-external.js +58 -7
- package/lib/refresh-network.js +24 -8
- package/lib/rfc-cli.js +108 -18
- package/lib/schemas/playbook.schema.json +1 -1
- package/lib/scoring.js +31 -1
- package/lib/source-advisories.js +58 -9
- package/lib/ttp-mapper.js +31 -3
- package/lib/upstream-check-cli.js +13 -1
- package/lib/validate-catalog-meta.js +51 -7
- package/lib/validate-cve-catalog.js +10 -0
- package/lib/validate-playbooks.js +19 -1
- package/lib/xml-tokenizer.js +187 -25
- package/manifest.json +53 -53
- package/orchestrator/dispatcher.js +45 -9
- package/orchestrator/index.js +9 -7
- package/orchestrator/pipeline.js +62 -14
- package/orchestrator/scanner.js +40 -9
- package/package.json +1 -1
- package/sbom.cdx.json +105 -90
- package/scripts/build-indexes.js +21 -3
- package/scripts/builders/section-offsets.js +17 -8
- package/scripts/check-catalog-gap-budget.js +3 -3
- package/scripts/check-codebase-patterns.js +124 -11
- package/scripts/check-sbom-currency.js +69 -3
- package/scripts/check-test-count.js +28 -16
- package/scripts/check-test-subjects.js +127 -0
- package/scripts/check-version-tags.js +24 -5
- package/scripts/predeploy.js +13 -0
- package/scripts/refresh-upstream-catalogs.js +150 -42
- package/scripts/release.js +28 -11
- package/scripts/validate-vendor-online.js +12 -9
package/lib/xml-tokenizer.js
CHANGED
|
@@ -110,6 +110,82 @@ function parseAttrs(rawAttrs) {
|
|
|
110
110
|
return out;
|
|
111
111
|
}
|
|
112
112
|
|
|
113
|
+
/**
|
|
114
|
+
* Emit the character-data of a raw-text leaf element span `[start, end)`.
|
|
115
|
+
*
|
|
116
|
+
* CDATA sections inside the span are emitted verbatim (onCData if present,
|
|
117
|
+
* else onText) exactly like the main loop's CDATA fast-path; all other bytes
|
|
118
|
+
* — including a stray unescaped '<' or inline HTML — are entity-decoded and
|
|
119
|
+
* emitted via onText. Keeping the CDATA contract here means a `<![CDATA[...]]>`
|
|
120
|
+
* title still surfaces only its inner text, not the literal markers.
|
|
121
|
+
*/
|
|
122
|
+
function emitRawText(xml, start, end, H) {
|
|
123
|
+
let p = start;
|
|
124
|
+
while (p < end) {
|
|
125
|
+
const c = xml.indexOf("<![CDATA[", p);
|
|
126
|
+
if (c === -1 || c >= end) {
|
|
127
|
+
const chunk = xml.slice(p, end);
|
|
128
|
+
if (chunk.length && H.onText) H.onText(decodeEntities(chunk));
|
|
129
|
+
return;
|
|
130
|
+
}
|
|
131
|
+
if (c > p) {
|
|
132
|
+
const chunk = xml.slice(p, c);
|
|
133
|
+
if (chunk.length && H.onText) H.onText(decodeEntities(chunk));
|
|
134
|
+
}
|
|
135
|
+
const cend = xml.indexOf("]]>", c + 9);
|
|
136
|
+
// findRawTextEnd already guaranteed a terminated CDATA before the close
|
|
137
|
+
// tag, so cend is within range; guard anyway.
|
|
138
|
+
const inner = cend === -1 || cend > end ? xml.slice(c + 9, end) : xml.slice(c + 9, cend);
|
|
139
|
+
if (H.onCData) H.onCData(inner);
|
|
140
|
+
else if (H.onText) H.onText(inner);
|
|
141
|
+
p = cend === -1 || cend > end ? end : cend + 3;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Locate the matching close tag of a raw-text leaf element.
|
|
147
|
+
*
|
|
148
|
+
* Scans `xml` from `start` (the byte after the leaf open tag's `>`) for the
|
|
149
|
+
* first close tag `</...name>` whose local-name (namespace prefix stripped)
|
|
150
|
+
* equals `name`. CDATA sections are skipped verbatim so a `</title>` that
|
|
151
|
+
* legitimately appears inside `<![CDATA[ ... ]]>` is treated as literal text,
|
|
152
|
+
* not as the element's terminator.
|
|
153
|
+
*
|
|
154
|
+
* Returns { contentEnd, next } on success — `contentEnd` is the index of the
|
|
155
|
+
* close tag's leading `<`, `next` is the index just past its `>`. Returns null
|
|
156
|
+
* when no matching close tag exists before EOF (genuinely truncated input).
|
|
157
|
+
*/
|
|
158
|
+
function findRawTextEnd(xml, start, name) {
|
|
159
|
+
const len = xml.length;
|
|
160
|
+
let j = start;
|
|
161
|
+
while (j < len) {
|
|
162
|
+
const lt = xml.indexOf("<", j);
|
|
163
|
+
if (lt === -1) return null;
|
|
164
|
+
// Step over a CDATA section so its contents (which may contain "</name>"
|
|
165
|
+
// literally) cannot satisfy the close-tag match.
|
|
166
|
+
if (xml.startsWith("<![CDATA[", lt)) {
|
|
167
|
+
const cend = xml.indexOf("]]>", lt + 9);
|
|
168
|
+
if (cend === -1) return null; // unterminated CDATA → truncated
|
|
169
|
+
j = cend + 3;
|
|
170
|
+
continue;
|
|
171
|
+
}
|
|
172
|
+
if (xml[lt + 1] === "/") {
|
|
173
|
+
const gt = xml.indexOf(">", lt + 2);
|
|
174
|
+
if (gt === -1) return null;
|
|
175
|
+
const closeName = localName(xml.slice(lt + 2, gt).trim());
|
|
176
|
+
if (closeName === name) {
|
|
177
|
+
return { contentEnd: lt, next: gt + 1 };
|
|
178
|
+
}
|
|
179
|
+
j = gt + 1;
|
|
180
|
+
continue;
|
|
181
|
+
}
|
|
182
|
+
// Any other '<' (stray literal '<', or an inner open tag like <b>) is part
|
|
183
|
+
// of the leaf's character data — advance past it without classifying.
|
|
184
|
+
j = lt + 1;
|
|
185
|
+
}
|
|
186
|
+
return null;
|
|
187
|
+
}
|
|
188
|
+
|
|
113
189
|
/**
|
|
114
190
|
* Streaming tokenizer. Calls handlers in document order. Returns no
|
|
115
191
|
* value — accumulation is the caller's responsibility.
|
|
@@ -122,6 +198,17 @@ function parseAttrs(rawAttrs) {
|
|
|
122
198
|
* onComment(text)
|
|
123
199
|
* onPI(name, content) processing instructions (<?xml-stylesheet?>)
|
|
124
200
|
* onError(message, position)
|
|
201
|
+
*
|
|
202
|
+
* Options (second-position fields on the handlers object):
|
|
203
|
+
* rawTextElements Set<string> of leaf-element local-names whose content
|
|
204
|
+
* is #PCDATA — once such an element opens, every byte up
|
|
205
|
+
* to the matching `</name>` close tag is treated as
|
|
206
|
+
* character data (entities decoded, then onText), exactly
|
|
207
|
+
* like the CDATA fast-path. This makes leaf fields tolerant
|
|
208
|
+
* of stray unescaped '<' (e.g. "affects versions < 5.0"),
|
|
209
|
+
* which real RSS/Atom routinely emits, instead of letting a
|
|
210
|
+
* recoverable lexical glitch silently drop the whole field.
|
|
211
|
+
* Structural parsing stays strict for container elements.
|
|
125
212
|
*/
|
|
126
213
|
function tokenize(xml, handlers) {
|
|
127
214
|
const H = handlers || {};
|
|
@@ -129,6 +216,7 @@ function tokenize(xml, handlers) {
|
|
|
129
216
|
if (H.onError) H.onError("input must be a string", 0);
|
|
130
217
|
return;
|
|
131
218
|
}
|
|
219
|
+
const rawTextElements = H.rawTextElements instanceof Set ? H.rawTextElements : null;
|
|
132
220
|
const len = xml.length;
|
|
133
221
|
let i = 0;
|
|
134
222
|
// Open-tag stack — surfaces EOF-with-unclosed-elements as an error
|
|
@@ -229,9 +317,33 @@ function tokenize(xml, handlers) {
|
|
|
229
317
|
if (H.onTagClose) H.onTagClose(name);
|
|
230
318
|
} else {
|
|
231
319
|
const attrs = parseAttrs(rawAttrs);
|
|
232
|
-
if (!selfClose) openStack.push(name);
|
|
233
320
|
if (H.onTagOpen) H.onTagOpen(name, attrs, selfClose);
|
|
234
|
-
if (selfClose
|
|
321
|
+
if (selfClose) {
|
|
322
|
+
if (H.onTagClose) H.onTagClose(name);
|
|
323
|
+
} else if (rawTextElements && rawTextElements.has(name)) {
|
|
324
|
+
// Raw-text leaf element (title/summary/description/...). Its content
|
|
325
|
+
// is #PCDATA: scan to the matching `</name>` and treat the whole span
|
|
326
|
+
// as character data — only the matching close tag terminates it. Any
|
|
327
|
+
// inner '<...>' (a stray unescaped '<', or inline HTML like <b>) is
|
|
328
|
+
// preserved as text and normalized later by stripHtml(), instead of
|
|
329
|
+
// being misclassified as markup and silently dropping the field.
|
|
330
|
+
const span = findRawTextEnd(xml, close + 1, name);
|
|
331
|
+
if (span === null) {
|
|
332
|
+
// Genuinely truncated: the leaf never closes before EOF. Surface
|
|
333
|
+
// the loud-error contract just like the structural path would, then
|
|
334
|
+
// flush the residual as text and stop.
|
|
335
|
+
if (H.onError) H.onError("unterminated element at EOF: " + name, len);
|
|
336
|
+
const tail = xml.slice(close + 1);
|
|
337
|
+
if (tail.length && H.onText) H.onText(decodeEntities(tail));
|
|
338
|
+
return;
|
|
339
|
+
}
|
|
340
|
+
emitRawText(xml, close + 1, span.contentEnd, H);
|
|
341
|
+
if (H.onTagClose) H.onTagClose(name);
|
|
342
|
+
i = span.next;
|
|
343
|
+
continue;
|
|
344
|
+
} else {
|
|
345
|
+
openStack.push(name);
|
|
346
|
+
}
|
|
235
347
|
}
|
|
236
348
|
i = close + 1;
|
|
237
349
|
}
|
|
@@ -240,15 +352,36 @@ function tokenize(xml, handlers) {
|
|
|
240
352
|
}
|
|
241
353
|
}
|
|
242
354
|
|
|
355
|
+
// Leaf-element local-names whose content is character-data (#PCDATA). When
|
|
356
|
+
// one of these opens, the tokenizer runs in raw-text mode until the matching
|
|
357
|
+
// close tag, so a stray unescaped '<' inside the field (e.g. "affects
|
|
358
|
+
// versions < 5.0") is preserved as text instead of being misclassified as
|
|
359
|
+
// markup and silently dropping the whole field.
|
|
360
|
+
const LEAF_FIELDS = new Set([
|
|
361
|
+
"title", "link", "pubDate", "published", "updated",
|
|
362
|
+
"description", "content", "summary",
|
|
363
|
+
]);
|
|
364
|
+
|
|
365
|
+
// rel-rank for Atom <link> sibling selection. RFC 4287: a <link> with no rel
|
|
366
|
+
// defaults to rel="alternate", the canonical article URL. A non-alternate rel
|
|
367
|
+
// (self / replies / edit / enclosure) should never clobber an alternate.
|
|
368
|
+
function relRank(rel) {
|
|
369
|
+
if (rel == null || rel === "" || String(rel).toLowerCase() === "alternate") return 2;
|
|
370
|
+
return 1;
|
|
371
|
+
}
|
|
372
|
+
|
|
243
373
|
/**
|
|
244
|
-
* Parse an RSS / Atom feed
|
|
245
|
-
*
|
|
374
|
+
* Parse an RSS / Atom feed, always collecting parse errors.
|
|
375
|
+
*
|
|
376
|
+
* Returns { items, errors } where errors is an array of
|
|
377
|
+
* { message, position } records. This is the channel a caller cannot forget
|
|
378
|
+
* to opt into — parseFeed() below is a thin back-compat wrapper.
|
|
246
379
|
*
|
|
247
|
-
*
|
|
248
|
-
* the optional `errors` array — callers wanting observability pass it.
|
|
380
|
+
* items: [{ title, link, published, body }, ...]
|
|
249
381
|
*/
|
|
250
|
-
function
|
|
382
|
+
function parseFeedDetailed(xml) {
|
|
251
383
|
const items = [];
|
|
384
|
+
const errors = [];
|
|
252
385
|
// Stack of "in-progress item" contexts. RSS uses <item>; Atom uses
|
|
253
386
|
// <entry>; both nest title / link / pubDate / published / updated /
|
|
254
387
|
// description / content / summary.
|
|
@@ -266,25 +399,34 @@ function parseFeed(xml, errors = null) {
|
|
|
266
399
|
let current = null; // active item context
|
|
267
400
|
let activeField = null; // active field local-name
|
|
268
401
|
let buffer = ""; // accumulator for current field text
|
|
269
|
-
let
|
|
402
|
+
let linkRank = 0; // rel-rank of the best <link> captured so far
|
|
270
403
|
|
|
271
404
|
tokenize(xml, {
|
|
405
|
+
rawTextElements: LEAF_FIELDS,
|
|
272
406
|
onTagOpen(name, attrs, selfClosing) {
|
|
273
407
|
if (ITEM_LOCALS.has(name)) {
|
|
274
408
|
current = { title: "", link: "", published: "", body: "" };
|
|
409
|
+
linkRank = 0;
|
|
275
410
|
return;
|
|
276
411
|
}
|
|
277
412
|
if (!current) return;
|
|
278
413
|
if (FIELD_MAP[name]) {
|
|
279
414
|
activeField = FIELD_MAP[name];
|
|
280
415
|
buffer = "";
|
|
281
|
-
// Atom <link href="..."/> —
|
|
282
|
-
//
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
416
|
+
// Atom <link href="..."/> — rel-aware selection. Only let an
|
|
417
|
+
// alternate / rel-absent link upgrade the captured value, and never
|
|
418
|
+
// let a non-alternate rel (self/replies/edit) clobber an alternate
|
|
419
|
+
// already in hand. First-alternate-wins, independent of document
|
|
420
|
+
// order.
|
|
421
|
+
if (name === "link" && attrs && attrs.href) {
|
|
422
|
+
const r = relRank(attrs.rel);
|
|
423
|
+
if (r > linkRank) {
|
|
424
|
+
current.link = attrs.href;
|
|
425
|
+
linkRank = r;
|
|
426
|
+
}
|
|
427
|
+
// The link value has been resolved from the attribute — there is
|
|
428
|
+
// no element-text close to wait for.
|
|
429
|
+
if (selfClosing) activeField = null;
|
|
288
430
|
}
|
|
289
431
|
}
|
|
290
432
|
},
|
|
@@ -294,28 +436,31 @@ function parseFeed(xml, errors = null) {
|
|
|
294
436
|
current = null;
|
|
295
437
|
activeField = null;
|
|
296
438
|
buffer = "";
|
|
439
|
+
linkRank = 0;
|
|
297
440
|
return;
|
|
298
441
|
}
|
|
299
442
|
if (!current) return;
|
|
300
443
|
if (FIELD_MAP[name] && activeField === FIELD_MAP[name]) {
|
|
301
444
|
const value = buffer.trim();
|
|
302
|
-
//
|
|
303
|
-
//
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
445
|
+
// RSS <link>...</link> — element text is authoritative (rank 2),
|
|
446
|
+
// unconditional. An empty element-text link falls back to whatever
|
|
447
|
+
// attribute capture onTagOpen already recorded.
|
|
448
|
+
if (name === "link") {
|
|
449
|
+
if (value) {
|
|
450
|
+
current.link = value;
|
|
451
|
+
linkRank = 2;
|
|
452
|
+
}
|
|
308
453
|
} else if (activeField === "body" || activeField === "title") {
|
|
309
454
|
// Strip HTML tags from title + description / content / summary.
|
|
310
455
|
// Many feeds embed inline HTML (<b>, <em>, <a>) in titles for
|
|
311
456
|
// emphasis; the operational consumer wants plain text. CDATA
|
|
312
457
|
// content reaches here verbatim, so this also strips HTML
|
|
313
|
-
// that was wrapped in CDATA to dodge entity-encoding.
|
|
458
|
+
// that was wrapped in CDATA to dodge entity-encoding. A stray
|
|
459
|
+
// unescaped '<' that survived raw-text mode collapses here too.
|
|
314
460
|
current[activeField] = stripHtml(value);
|
|
315
461
|
} else {
|
|
316
462
|
current[activeField] = value;
|
|
317
463
|
}
|
|
318
|
-
linkHref = null;
|
|
319
464
|
activeField = null;
|
|
320
465
|
buffer = "";
|
|
321
466
|
}
|
|
@@ -327,10 +472,27 @@ function parseFeed(xml, errors = null) {
|
|
|
327
472
|
if (activeField) buffer += text;
|
|
328
473
|
},
|
|
329
474
|
onError(msg, pos) {
|
|
330
|
-
|
|
475
|
+
errors.push({ message: msg, position: pos });
|
|
331
476
|
}
|
|
332
477
|
});
|
|
333
478
|
|
|
479
|
+
return { items, errors };
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
/**
|
|
483
|
+
* Parse an RSS / Atom feed into a flat array of items. Returns:
|
|
484
|
+
* [{ title, link, published, body }, ...]
|
|
485
|
+
*
|
|
486
|
+
* Empty array on parse failure. Errors are ALWAYS collected internally; the
|
|
487
|
+
* optional `errors` array is filled for callers that pass one (back-compat).
|
|
488
|
+
* Prefer parseFeedDetailed(xml) for new callers — it returns the errors
|
|
489
|
+
* channel unconditionally so it cannot be silently dropped.
|
|
490
|
+
*/
|
|
491
|
+
function parseFeed(xml, errors = null) {
|
|
492
|
+
const { items, errors: collected } = parseFeedDetailed(xml);
|
|
493
|
+
if (Array.isArray(errors)) {
|
|
494
|
+
for (const e of collected) errors.push(e);
|
|
495
|
+
}
|
|
334
496
|
return items;
|
|
335
497
|
}
|
|
336
498
|
|
|
@@ -341,4 +503,4 @@ function stripHtml(s) {
|
|
|
341
503
|
return s.replace(/<[^>]+>/g, " ").replace(/\s+/g, " ").trim();
|
|
342
504
|
}
|
|
343
505
|
|
|
344
|
-
module.exports = { tokenize, parseFeed, decodeEntities, localName, parseAttrs, stripHtml };
|
|
506
|
+
module.exports = { tokenize, parseFeed, parseFeedDetailed, decodeEntities, localName, parseAttrs, stripHtml };
|