bmad-plus 0.21.0 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/README.md +13 -13
- package/SECURITY.md +62 -0
- package/osint-agent-package/skills/bmad-osint-investigate/osint/scripts/_http.py +68 -24
- package/package.json +1 -1
- package/readme-international/README.de.md +13 -13
- package/readme-international/README.es.md +13 -13
- package/readme-international/README.fr.md +13 -13
- package/src/bmad-plus/packs/pack-dev-studio/categories/implementation/code-review.md +3 -1
- package/src/bmad-plus/packs/pack-seo/SKILL.md +3 -1
- package/src/bmad-plus/packs/pack-seo/ref/cwv-thresholds.md +2 -2
- package/src/bmad-plus/packs/pack-seo/requirements.txt +1 -1
- package/src/bmad-plus/packs/pack-seo/scripts/seo_apis.py +72 -30
- package/src/bmad-plus/packs/pack-seo/scripts/seo_report.py +5 -6
- package/src/bmad-plus/packs/pack-seo/scripts/seo_screenshot.py +176 -14
- package/src/bmad-plus/packs/pack-shield/README.md +12 -0
- package/src/bmad-plus/packs/pack-shield/SKILL.md +7 -1
- package/src/bmad-plus/packs/pack-shield/review-rules/access-control.md +10 -0
- package/src/bmad-plus/packs/pack-shield/review-rules/ai-integrations.md +10 -0
- package/src/bmad-plus/packs/pack-shield/review-rules/change-and-supply-chain.md +10 -0
- package/src/bmad-plus/packs/pack-shield/review-rules/cryptography.md +10 -0
- package/src/bmad-plus/packs/pack-shield/review-rules/index.yaml +134 -0
- package/src/bmad-plus/packs/pack-shield/review-rules/logging.md +10 -0
- package/src/bmad-plus/packs/pack-shield/review-rules/personal-data.md +10 -0
- package/src/bmad-plus/packs/pack-shield/shared/ai-processing-register-template.yaml +53 -0
- package/src/bmad-plus/packs/pack-shield/shared/ai-processing-register.md +32 -0
- package/src/bmad-plus/packs/pack-shield/shared/assurance-case-template.yaml +87 -0
- package/src/bmad-plus/packs/pack-shield/shared/assurance-case.md +50 -0
- package/src/bmad-plus/packs/pack-shield/shield-orchestrator.md +24 -1
- package/src/bmad-plus/skills/bmad-plus-uat/SKILL.md +1 -0
- package/src/bmad-plus/skills/bmad-plus-uat/template/page.html +5 -4
- package/tools/build/generate-adapters.js +7 -0
- package/tools/build/generate.js +14 -0
- package/tools/cli/bmad-plus-cli.js +2 -0
- package/tools/cli/commands/ai-register.js +63 -0
- package/tools/cli/commands/assurance.js +162 -0
- package/tools/cli/commands/review.js +10 -3
- package/tools/cli/lib/ai-register.js +393 -0
- package/tools/cli/lib/assurance.js +822 -0
- package/tools/cli/lib/control-refs.js +132 -0
- package/tools/cli/lib/installation-health.js +17 -0
- package/tools/cli/lib/packs.js +60 -2
- package/tools/cli/lib/page-origins.js +582 -0
- package/tools/cli/lib/review-rules.js +92 -24
- package/tools/cli/lib/review.js +28 -1
- package/tools/cli/lib/uat.js +22 -5
|
@@ -0,0 +1,582 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Every address a page template would reach on an origin other than its own.
|
|
3
|
+
*
|
|
4
|
+
* The template is read the way a browser reads it rather than matched as raw text: markup is
|
|
5
|
+
* split into tags, comments and raw text by the HTML tokenizer's own rules, attributes have
|
|
6
|
+
* their character references decoded, style sheets their escapes, script string literals
|
|
7
|
+
* theirs, and a data: document or script is decoded and read in turn. Where the tokenizer's
|
|
8
|
+
* state depends on the tree (a style or script inside svg, math or select), the template is
|
|
9
|
+
* refused rather than guessed at. Each address a browser would load is resolved
|
|
10
|
+
* against the page's own address under every scheme the page is opened from (https in a
|
|
11
|
+
* host, http from the local server, a file), because `https:host`, `\\host` and `/\host`
|
|
12
|
+
* resolve differently from each. What a script assembles from pieces at run time is beyond
|
|
13
|
+
* any reading of the text: the jsdom and Chromium checks watch those requests.
|
|
14
|
+
*/
|
|
15
|
+
'use strict';
|
|
16
|
+
|
|
17
|
+
const { URL } = require('node:url');
|
|
18
|
+
|
|
19
|
+
const BASES = ['https://page.invalid/uat/', 'http://127.0.0.1:4173/uat/', 'file:///uat/'].map(
|
|
20
|
+
(base) => new URL(base)
|
|
21
|
+
);
|
|
22
|
+
/** Schemes whose address never leaves the browser. */
|
|
23
|
+
const LOCAL = new Set(['data:', 'blob:', 'about:']);
|
|
24
|
+
/** Attributes a browser follows by itself when the element is parsed, shown or submitted. */
|
|
25
|
+
const LOADING = new Set([
|
|
26
|
+
'src',
|
|
27
|
+
'href',
|
|
28
|
+
'srcset',
|
|
29
|
+
'imagesrcset',
|
|
30
|
+
'poster',
|
|
31
|
+
'data',
|
|
32
|
+
'action',
|
|
33
|
+
'formaction',
|
|
34
|
+
'background',
|
|
35
|
+
'lowsrc',
|
|
36
|
+
'codebase',
|
|
37
|
+
'manifest',
|
|
38
|
+
'icon',
|
|
39
|
+
'xlink:href',
|
|
40
|
+
'ping',
|
|
41
|
+
'attributionsrc',
|
|
42
|
+
]);
|
|
43
|
+
/** Loading attributes that hold a list of addresses rather than one. */
|
|
44
|
+
const ADDRESS_LISTS = new Set(['srcset', 'imagesrcset', 'ping', 'attributionsrc']);
|
|
45
|
+
/** Elements that only ever show what they load as an image or media, which loads nothing more. */
|
|
46
|
+
const MEDIA = new Set(['img', 'source', 'picture', 'video', 'audio', 'track', 'input', 'image']);
|
|
47
|
+
/** SVG animation elements, whose values can replace the address of the element they animate. */
|
|
48
|
+
const ANIMATION = new Set(['set', 'animate', 'animatecolor', 'animatemotion', 'animatetransform']);
|
|
49
|
+
/** Start tags after which the tokenizer stops reading markup until the matching end tag. */
|
|
50
|
+
const RAW_TEXT = new Set(['style', 'xmp', 'iframe', 'noembed', 'noframes', 'noscript']);
|
|
51
|
+
const ESCAPABLE_RAW_TEXT = new Set(['title', 'textarea']);
|
|
52
|
+
const SWITCHING = new Set([...RAW_TEXT, ...ESCAPABLE_RAW_TEXT, 'script', 'plaintext']);
|
|
53
|
+
/** Foreign elements whose content is HTML again, by namespace. */
|
|
54
|
+
const INTEGRATION = {
|
|
55
|
+
svg: new Set(['foreignobject', 'desc', 'title']),
|
|
56
|
+
math: new Set(['mi', 'mo', 'mn', 'ms', 'mtext']),
|
|
57
|
+
};
|
|
58
|
+
/** HTML start tags that close every open foreign element up to the nearest integration point. */
|
|
59
|
+
const BREAKOUT = new Set(
|
|
60
|
+
(
|
|
61
|
+
'b big blockquote body br center code dd div dl dt em embed h1 h2 h3 h4 h5 h6 head hr i img ' +
|
|
62
|
+
'li listing menu meta nobr ol p pre ruby s small span strong strike sub sup table tt u ul var'
|
|
63
|
+
).split(' ')
|
|
64
|
+
);
|
|
65
|
+
/** Charsets in which every ASCII byte is the ASCII character, so a byte-wise reading holds. */
|
|
66
|
+
const ASCII_CHARSET = /^(?:utf-?8|us-ascii|ascii|iso-8859-1|latin1|windows-125\d)$/i;
|
|
67
|
+
/** Following a link is the reader's act; the ping it carries is not. */
|
|
68
|
+
const NAVIGATION = new Set(['a', 'area']);
|
|
69
|
+
/** Namespace names identify a vocabulary and load nothing. */
|
|
70
|
+
const NAMESPACES = new Set([
|
|
71
|
+
'http://www.w3.org/1999/xhtml',
|
|
72
|
+
'http://www.w3.org/2000/svg',
|
|
73
|
+
'http://www.w3.org/1999/xlink',
|
|
74
|
+
'http://www.w3.org/1998/Math/MathML',
|
|
75
|
+
'http://www.w3.org/XML/1998/namespace',
|
|
76
|
+
'http://www.w3.org/2000/xmlns/',
|
|
77
|
+
]);
|
|
78
|
+
const NAMED_REFERENCES = {
|
|
79
|
+
amp: '&',
|
|
80
|
+
AMP: '&',
|
|
81
|
+
lt: '<',
|
|
82
|
+
LT: '<',
|
|
83
|
+
gt: '>',
|
|
84
|
+
GT: '>',
|
|
85
|
+
quot: '"',
|
|
86
|
+
QUOT: '"',
|
|
87
|
+
apos: "'",
|
|
88
|
+
num: '#',
|
|
89
|
+
quest: '?',
|
|
90
|
+
commat: '@',
|
|
91
|
+
percnt: '%',
|
|
92
|
+
excl: '!',
|
|
93
|
+
equals: '=',
|
|
94
|
+
plus: '+',
|
|
95
|
+
semi: ';',
|
|
96
|
+
lowbar: '_',
|
|
97
|
+
ast: '*',
|
|
98
|
+
lsqb: '[',
|
|
99
|
+
lbrack: '[',
|
|
100
|
+
rsqb: ']',
|
|
101
|
+
rbrack: ']',
|
|
102
|
+
lcub: '{',
|
|
103
|
+
lbrace: '{',
|
|
104
|
+
rcub: '}',
|
|
105
|
+
rbrace: '}',
|
|
106
|
+
verbar: '|',
|
|
107
|
+
vert: '|',
|
|
108
|
+
grave: '`',
|
|
109
|
+
Hat: '^',
|
|
110
|
+
dollar: '$',
|
|
111
|
+
sol: '/',
|
|
112
|
+
bsol: '\\',
|
|
113
|
+
colon: ':',
|
|
114
|
+
period: '.',
|
|
115
|
+
comma: ',',
|
|
116
|
+
lpar: '(',
|
|
117
|
+
rpar: ')',
|
|
118
|
+
Tab: '\t',
|
|
119
|
+
NewLine: '\n',
|
|
120
|
+
};
|
|
121
|
+
const SCRIPT_ESCAPES = { n: '\n', r: '\r', t: '\t', b: '\b', f: '\f', v: '\v', 0: '\0' };
|
|
122
|
+
const REGEX_AFTER = new Set([
|
|
123
|
+
'return',
|
|
124
|
+
'typeof',
|
|
125
|
+
'instanceof',
|
|
126
|
+
'in',
|
|
127
|
+
'of',
|
|
128
|
+
'new',
|
|
129
|
+
'delete',
|
|
130
|
+
'void',
|
|
131
|
+
'throw',
|
|
132
|
+
'case',
|
|
133
|
+
'do',
|
|
134
|
+
'else',
|
|
135
|
+
'yield',
|
|
136
|
+
'await',
|
|
137
|
+
]);
|
|
138
|
+
|
|
139
|
+
const codePoint = (code) =>
|
|
140
|
+
code > 0 && code <= 0x10ffff && (code < 0xd800 || code > 0xdfff)
|
|
141
|
+
? String.fromCodePoint(code)
|
|
142
|
+
: '\uFFFD';
|
|
143
|
+
|
|
144
|
+
function decodeMarkup(value) {
|
|
145
|
+
return value.replace(/&(?:#x([\da-f]+)|#(\d+)|([a-z][a-z\d]*));?/gi, (whole, hex, dec, name) => {
|
|
146
|
+
if (hex || dec) return codePoint(Number.parseInt(hex || dec, hex ? 16 : 10));
|
|
147
|
+
return Object.hasOwn(NAMED_REFERENCES, name) ? NAMED_REFERENCES[name] : whole;
|
|
148
|
+
});
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
function decodeCss(value) {
|
|
152
|
+
return value.replace(/\\(?:([\da-f]{1,6})[ \t\n\r\f]?|\r\n|([\s\S]))/gi, (_, hex, char) => {
|
|
153
|
+
if (hex) return codePoint(Number.parseInt(hex, 16));
|
|
154
|
+
return char === undefined || /[\n\r\f]/.test(char) ? '' : char;
|
|
155
|
+
});
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
function decodeScript(value) {
|
|
159
|
+
return value.replace(
|
|
160
|
+
/\\(?:x([\da-f]{2})|u\{([\da-f]+)\}|u([\da-f]{4})|(\r\n|[\n\r\u2028\u2029])|([\s\S]))/gi,
|
|
161
|
+
(_, hex, braced, unicode, continuation, char) => {
|
|
162
|
+
if (hex || braced || unicode) return codePoint(Number.parseInt(hex || braced || unicode, 16));
|
|
163
|
+
if (continuation) return '';
|
|
164
|
+
return Object.hasOwn(SCRIPT_ESCAPES, char) ? SCRIPT_ESCAPES[char] : char;
|
|
165
|
+
}
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/** True when a browser would fetch `address` from somewhere other than the page's origin. */
|
|
170
|
+
function leavesOrigin(address) {
|
|
171
|
+
for (const base of BASES) {
|
|
172
|
+
let resolved;
|
|
173
|
+
try {
|
|
174
|
+
resolved = new URL(address, base);
|
|
175
|
+
} catch {
|
|
176
|
+
return true; // an address the check cannot read is not one it can vouch for
|
|
177
|
+
}
|
|
178
|
+
if (LOCAL.has(resolved.protocol)) return false;
|
|
179
|
+
if (resolved.protocol !== base.protocol || resolved.host !== base.host) return true;
|
|
180
|
+
}
|
|
181
|
+
return false;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/**
|
|
185
|
+
* The part of a string that names another host, if any: an opening `//` or `\\`, a network
|
|
186
|
+
* scheme (with or without its slashes), or any `scheme://` inside it. Namespace names are not
|
|
187
|
+
* addresses. Used where the text is not yet an address: a script string, a data attribute.
|
|
188
|
+
*/
|
|
189
|
+
function namedHost(text) {
|
|
190
|
+
const value = text.replace(/[\t\n\r]/g, '').replace(/^[\0- ]+/, '');
|
|
191
|
+
if (/^[\\/]{2}|^(?:https?|wss?|ftp|file):/i.test(value) && !NAMESPACES.has(value)) return value;
|
|
192
|
+
for (const [found] of value.matchAll(/[a-z][a-z\d+.-]*:[\\/]{2}[^\s"'`<>()]*/gi))
|
|
193
|
+
if (!NAMESPACES.has(found)) return found;
|
|
194
|
+
return null;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** String literal and template chunk contents of a script, comments and regex literals left out. */
|
|
198
|
+
function scriptStrings(source) {
|
|
199
|
+
const strings = [];
|
|
200
|
+
const substitutions = []; // open `${` of template literals, with the braces opened inside each
|
|
201
|
+
let last = '';
|
|
202
|
+
let i = 0;
|
|
203
|
+
const readTemplate = () => {
|
|
204
|
+
let chunk = '';
|
|
205
|
+
while (i < source.length) {
|
|
206
|
+
const char = source[i];
|
|
207
|
+
if (char === '\\') {
|
|
208
|
+
chunk += source.slice(i, i + 2);
|
|
209
|
+
i += 2;
|
|
210
|
+
} else if (char === '`') {
|
|
211
|
+
i += 1;
|
|
212
|
+
break;
|
|
213
|
+
} else if (char === '$' && source[i + 1] === '{') {
|
|
214
|
+
i += 2;
|
|
215
|
+
substitutions.push(0);
|
|
216
|
+
break;
|
|
217
|
+
} else {
|
|
218
|
+
chunk += char;
|
|
219
|
+
i += 1;
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
strings.push(chunk);
|
|
223
|
+
last = '`';
|
|
224
|
+
};
|
|
225
|
+
while (i < source.length) {
|
|
226
|
+
const char = source[i];
|
|
227
|
+
const next = source[i + 1];
|
|
228
|
+
if (char === '/' && next === '/') {
|
|
229
|
+
const end = source.indexOf('\n', i);
|
|
230
|
+
i = end < 0 ? source.length : end;
|
|
231
|
+
} else if (char === '/' && next === '*') {
|
|
232
|
+
const end = source.indexOf('*/', i + 2);
|
|
233
|
+
i = end < 0 ? source.length : end + 2;
|
|
234
|
+
} else if (char === '"' || char === "'") {
|
|
235
|
+
let chunk = '';
|
|
236
|
+
i += 1;
|
|
237
|
+
while (i < source.length && source[i] !== char && source[i] !== '\n') {
|
|
238
|
+
const step = source[i] === '\\' ? 2 : 1;
|
|
239
|
+
chunk += source.slice(i, i + step);
|
|
240
|
+
i += step;
|
|
241
|
+
}
|
|
242
|
+
i += 1;
|
|
243
|
+
strings.push(chunk);
|
|
244
|
+
last = char;
|
|
245
|
+
} else if (char === '`') {
|
|
246
|
+
i += 1;
|
|
247
|
+
readTemplate();
|
|
248
|
+
} else if (char === '}' && substitutions.length && substitutions.at(-1) === 0) {
|
|
249
|
+
substitutions.pop();
|
|
250
|
+
i += 1;
|
|
251
|
+
readTemplate();
|
|
252
|
+
} else if (
|
|
253
|
+
char === '/' &&
|
|
254
|
+
(last === '' || REGEX_AFTER.has(last) || /^[(,=:[!&|?{};+\-*%<>~^}]$/.test(last))
|
|
255
|
+
) {
|
|
256
|
+
let inClass = false;
|
|
257
|
+
i += 1;
|
|
258
|
+
while (i < source.length && source[i] !== '\n' && (inClass || source[i] !== '/')) {
|
|
259
|
+
if (source[i] === '\\') i += 1;
|
|
260
|
+
else if (source[i] === '[') inClass = true;
|
|
261
|
+
else if (source[i] === ']') inClass = false;
|
|
262
|
+
i += 1;
|
|
263
|
+
}
|
|
264
|
+
i += 1;
|
|
265
|
+
while (/[a-z]/i.test(source[i] || '')) i += 1;
|
|
266
|
+
last = '/re/';
|
|
267
|
+
} else if (/\s/.test(char)) {
|
|
268
|
+
i += 1;
|
|
269
|
+
} else if (/[\w$]/.test(char)) {
|
|
270
|
+
const word = /^[\w$]+/.exec(source.slice(i, i + 64))[0];
|
|
271
|
+
i += word.length;
|
|
272
|
+
last = word;
|
|
273
|
+
} else {
|
|
274
|
+
if (substitutions.length && char === '{') substitutions[substitutions.length - 1] += 1;
|
|
275
|
+
if (substitutions.length && char === '}') substitutions[substitutions.length - 1] -= 1;
|
|
276
|
+
last = char;
|
|
277
|
+
i += 1;
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
return strings.map(decodeScript);
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
function scanScript(where, source, found) {
|
|
284
|
+
for (const literal of scriptStrings(source)) {
|
|
285
|
+
const host = namedHost(literal);
|
|
286
|
+
if (host) found.push(`an address on another origin in ${where}: ${host}`);
|
|
287
|
+
}
|
|
288
|
+
// Behind the reading, the raw text: a scheme with its slashes anywhere, comments included.
|
|
289
|
+
for (const [address] of source.matchAll(/\b[a-z][a-z\d+.-]*:[\\/]{2}[^\s"'`<>()]*/gi))
|
|
290
|
+
if (!NAMESPACES.has(address)) found.push(`an absolute address in ${where}: ${address}`);
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
function scanCss(where, css, found) {
|
|
294
|
+
const text = css.replace(/\/\*[\s\S]*?\*\//g, ' ');
|
|
295
|
+
if (/@import\b/i.test(decodeCss(text))) found.push(`a stylesheet import in ${where}`);
|
|
296
|
+
const escape = String.raw`\\[\da-f]{1,6}[ \t\n\r\f]?|\\[\s\S]`;
|
|
297
|
+
const tokens = new RegExp(
|
|
298
|
+
String.raw`"((?:[^"\\\n]|\\[\s\S])*)"|'((?:[^'\\\n]|\\[\s\S])*)'|((?:[\w-]|${escape})+)\(\s*((?:[^)\\\s"']|${escape})*)\s*\)`,
|
|
299
|
+
'gi'
|
|
300
|
+
);
|
|
301
|
+
for (const [, double, single, name, bare] of text.matchAll(tokens)) {
|
|
302
|
+
// A quoted string may be an address (url("…"), image-set("…"), @import "…") or plain text
|
|
303
|
+
// (content: "Note: "): it is refused when it names a host. A bare url() is an address.
|
|
304
|
+
const quoted = double ?? single;
|
|
305
|
+
const host =
|
|
306
|
+
quoted === undefined
|
|
307
|
+
? /^url$/i.test(decodeCss(name)) && leavesOrigin(decodeCss(bare)) && decodeCss(bare)
|
|
308
|
+
: namedHost(decodeCss(quoted));
|
|
309
|
+
if (host) found.push(`an address on another origin in ${where}: ${host.trim()}`);
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/**
|
|
314
|
+
* The bytes behind a data: address as a browser reads them, or why they cannot be read. The
|
|
315
|
+
* URL parser drops tabs and newlines and ends the body at `#`; the body is percent-decoded,
|
|
316
|
+
* then base64-decoded when the type says so. A charset that is not ASCII-compatible (UTF-16,
|
|
317
|
+
* or its byte order mark) would hide ASCII markup from a byte-wise reading: it is refused.
|
|
318
|
+
*/
|
|
319
|
+
function dataPayload(address) {
|
|
320
|
+
const url = address.replace(/[\t\n\r]/g, '').replace(/^[\0- ]+|[\0- ]+$/g, '');
|
|
321
|
+
const comma = url.indexOf(',');
|
|
322
|
+
if (comma < 0) return { text: '' }; // no body: the browser loads nothing
|
|
323
|
+
const type = url.slice(url.indexOf(':') + 1, comma).trim();
|
|
324
|
+
const raw = Buffer.from(url.slice(comma + 1).split('#')[0], 'utf8');
|
|
325
|
+
const bytes = [];
|
|
326
|
+
for (let i = 0; i < raw.length; i += 1) {
|
|
327
|
+
const hex = raw[i] === 0x25 ? raw.toString('latin1', i + 1, i + 3) : '';
|
|
328
|
+
if (/^[\da-f]{2}$/i.test(hex)) {
|
|
329
|
+
bytes.push(Number.parseInt(hex, 16));
|
|
330
|
+
i += 2;
|
|
331
|
+
} else bytes.push(raw[i]);
|
|
332
|
+
}
|
|
333
|
+
let body = Buffer.from(bytes);
|
|
334
|
+
if (/;\s*base64$/i.test(type)) {
|
|
335
|
+
let text = body.toString('latin1').replace(/[\t\n\f\r ]/g, '');
|
|
336
|
+
if (text.length % 4 === 0) text = text.replace(/==?$/, '');
|
|
337
|
+
if (text.length % 4 === 1 || /[^A-Za-z\d+/]/.test(text)) return { unreadable: 'bad base64' };
|
|
338
|
+
body = Buffer.from(text, 'base64');
|
|
339
|
+
}
|
|
340
|
+
const charset = /;\s*charset="?([^;"]*)/i.exec(type)?.[1].trim();
|
|
341
|
+
if (charset && !ASCII_CHARSET.test(charset)) return { unreadable: `charset ${charset}` };
|
|
342
|
+
if ((body[0] === 0xfe && body[1] === 0xff) || (body[0] === 0xff && body[1] === 0xfe))
|
|
343
|
+
return { unreadable: 'a UTF-16 byte order mark' };
|
|
344
|
+
return { text: body.toString('latin1') };
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* A data: document, stylesheet or script reaches the network like any page: it is decoded and
|
|
349
|
+
* read as what the element that loads it makes of it.
|
|
350
|
+
*/
|
|
351
|
+
function scanData(tag, where, address, found) {
|
|
352
|
+
const payload = dataPayload(address);
|
|
353
|
+
const label = `the data: address of ${where}`;
|
|
354
|
+
if (payload.unreadable) {
|
|
355
|
+
found.push(`${label}, which the check cannot read (${payload.unreadable})`);
|
|
356
|
+
return;
|
|
357
|
+
}
|
|
358
|
+
if (tag === 'script') return scanScript(label, payload.text, found);
|
|
359
|
+
if (tag === 'link') {
|
|
360
|
+
// A stylesheet, or a manifest whose JSON strings name the icons it loads.
|
|
361
|
+
scanCss(label, payload.text, found);
|
|
362
|
+
return scanScript(label, payload.text, found);
|
|
363
|
+
}
|
|
364
|
+
for (const entry of scanMarkup(payload.text, [])) found.push(`${entry}, in ${label}`);
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
const scheme = (address) =>
|
|
368
|
+
/^([a-z][a-z\d+.-]*):/i.exec(address.replace(/[\t\n\r]/g, '').replace(/^[\0- ]+/, ''))?.[1];
|
|
369
|
+
|
|
370
|
+
function scanAttribute(tag, name, value, found) {
|
|
371
|
+
const where = `<${tag} ${name}>`;
|
|
372
|
+
if (name === 'xmlns' || name.startsWith('xmlns:')) return;
|
|
373
|
+
if (name === 'style') return scanCss(where, value, found);
|
|
374
|
+
if (name.startsWith('on')) return scanScript(where, value, found);
|
|
375
|
+
if (name === 'srcdoc') return scanMarkup(value, found);
|
|
376
|
+
if (name === 'href' && NAVIGATION.has(tag)) return;
|
|
377
|
+
if (LOADING.has(name)) {
|
|
378
|
+
const trimmed = value.replace(/^[\0- ]+/, '');
|
|
379
|
+
if (/^javascript:/i.test(trimmed)) return scanScript(where, trimmed.slice(11), found);
|
|
380
|
+
const addresses = ADDRESS_LISTS.has(name) ? trimmed.split(/[\s,]+/).filter(Boolean) : [value];
|
|
381
|
+
for (const address of addresses) {
|
|
382
|
+
if (leavesOrigin(address)) found.push(`an address on another origin in ${where}: ${address}`);
|
|
383
|
+
else if (!MEDIA.has(tag) && scheme(address)?.toLowerCase() === 'data')
|
|
384
|
+
scanData(tag, where, address, found);
|
|
385
|
+
}
|
|
386
|
+
return;
|
|
387
|
+
}
|
|
388
|
+
// An SVG animation sets the attribute it animates, an address included, to these values.
|
|
389
|
+
if (ANIMATION.has(tag) && ['to', 'from', 'by', 'values'].includes(name))
|
|
390
|
+
for (const address of value.split(';'))
|
|
391
|
+
if (leavesOrigin(address.trim()))
|
|
392
|
+
found.push(`an address on another origin in ${where}: ${address.trim()}`);
|
|
393
|
+
// A presentation attribute (mask, filter, cursor, fill…) loads what its url() names.
|
|
394
|
+
if (value.includes('(')) scanCss(where, `x:${value}`, found);
|
|
395
|
+
// Not loaded by the browser, but a script may read it: it must not name another host either.
|
|
396
|
+
const host = namedHost(value);
|
|
397
|
+
if (host) found.push(`an address on another origin in ${where}: ${host}`);
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* The tag that opens at `start`, read by the tokenizer's tag and attribute states: its name,
|
|
402
|
+
* its attributes in order with their references decoded, and where it ends. A repeated name
|
|
403
|
+
* is kept and read, though a browser keeps only the first. Null when the text ends inside it:
|
|
404
|
+
* a browser then drops it.
|
|
405
|
+
*/
|
|
406
|
+
function readTag(html, start) {
|
|
407
|
+
const next = (pattern, index) => {
|
|
408
|
+
pattern.lastIndex = index;
|
|
409
|
+
return pattern.exec(html)[0];
|
|
410
|
+
};
|
|
411
|
+
let i = start + (html[start + 1] === '/' ? 2 : 1);
|
|
412
|
+
const rawName = next(/[^\t\n\f\r />]*/y, i);
|
|
413
|
+
i += rawName.length;
|
|
414
|
+
const attributes = [];
|
|
415
|
+
let selfClosing = false;
|
|
416
|
+
while (i < html.length) {
|
|
417
|
+
const char = html[i];
|
|
418
|
+
if (char === '>') return { name: rawName.toLowerCase(), attributes, selfClosing, end: i + 1 };
|
|
419
|
+
selfClosing = char === '/' && html[i + 1] === '>';
|
|
420
|
+
if (/[\t\n\f\r /]/.test(char)) {
|
|
421
|
+
i += 1;
|
|
422
|
+
continue;
|
|
423
|
+
}
|
|
424
|
+
const key = next(/[\s\S][^\t\n\f\r />=]*/y, i); // a leading `=` belongs to the name
|
|
425
|
+
i += key.length;
|
|
426
|
+
while (/[\t\n\f\r ]/.test(html[i])) i += 1;
|
|
427
|
+
let value = '';
|
|
428
|
+
if (html[i] === '=') {
|
|
429
|
+
i += 1;
|
|
430
|
+
while (/[\t\n\f\r ]/.test(html[i])) i += 1;
|
|
431
|
+
const quote = html[i];
|
|
432
|
+
if (quote === '"' || quote === "'") {
|
|
433
|
+
const close = html.indexOf(quote, i + 1);
|
|
434
|
+
if (close < 0) return null;
|
|
435
|
+
value = html.slice(i + 1, close);
|
|
436
|
+
i = close + 1;
|
|
437
|
+
} else {
|
|
438
|
+
value = next(/[^\t\n\f\r >]*/y, i);
|
|
439
|
+
i += value.length;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
attributes.push([key.toLowerCase(), decodeMarkup(value)]);
|
|
443
|
+
}
|
|
444
|
+
return null;
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
/** Where the raw text of a <style>, <title>… that starts at `from` ends: its end tag. */
|
|
448
|
+
function rawTextEnd(html, from, tag) {
|
|
449
|
+
const end = new RegExp(String.raw`</${tag}[\t\n\f\r />]`, 'gi');
|
|
450
|
+
end.lastIndex = from;
|
|
451
|
+
return end.exec(html)?.index ?? html.length;
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
/**
|
|
455
|
+
* Where a script that starts at `from` ends. Inside `<!--`, a `<script` defers the end until
|
|
456
|
+
* `</script>` has closed it or `-->` has closed the escape: the tokenizer's script data
|
|
457
|
+
* escaped and double escaped states.
|
|
458
|
+
*/
|
|
459
|
+
function scriptEnd(html, from) {
|
|
460
|
+
const token = /<!--|-->|<(\/?)script[\t\n\f\r />]/gi;
|
|
461
|
+
token.lastIndex = from;
|
|
462
|
+
let state = 'data';
|
|
463
|
+
for (let match; (match = token.exec(html));) {
|
|
464
|
+
const [text, closing] = match;
|
|
465
|
+
if (text === '<!--') {
|
|
466
|
+
if (state === 'data') {
|
|
467
|
+
let after = token.lastIndex;
|
|
468
|
+
while (html[after] === '-') after += 1;
|
|
469
|
+
if (html[after] === '>') token.lastIndex = after + 1;
|
|
470
|
+
else state = 'escaped';
|
|
471
|
+
} else token.lastIndex = match.index + 2; // `<!-->` in an escape still ends it
|
|
472
|
+
} else if (text === '-->') state = 'data';
|
|
473
|
+
else if (closing === '/') {
|
|
474
|
+
if (state !== 'double') return match.index;
|
|
475
|
+
state = 'escaped';
|
|
476
|
+
} else if (state === 'escaped') state = 'double';
|
|
477
|
+
}
|
|
478
|
+
return html.length;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
/**
|
|
482
|
+
* Every address in markup, found by the HTML tokenizer's rules: a comment ends at `-->`,
|
|
483
|
+
* `--!>`, or at once on `<!-->` and `<!--->`; a bogus comment at the next `>`; the text of a
|
|
484
|
+
* style, script, title, textarea and the like at its own end tag. Whether a <style> or a
|
|
485
|
+
* <script> switches the tokenizer depends on the tree around it: inside svg or math it does
|
|
486
|
+
* not, inside select or after a frameset a browser may drop the tag. Open svg and math
|
|
487
|
+
* elements and their integration points are followed, closing no sooner than a browser would;
|
|
488
|
+
* where the answer still depends on the tree, the markup is refused.
|
|
489
|
+
*/
|
|
490
|
+
function scanMarkup(html, found) {
|
|
491
|
+
const unreadable = (what) => found.push(`${what}, which the check cannot read as a browser does`);
|
|
492
|
+
const open = []; // open foreign elements and integration points: { name, ns, integration }
|
|
493
|
+
const closeForeign = () => {
|
|
494
|
+
while (open.length && !open.at(-1).integration) open.pop();
|
|
495
|
+
};
|
|
496
|
+
let selects = 0;
|
|
497
|
+
let i = 0;
|
|
498
|
+
while ((i = html.indexOf('<', i)) >= 0) {
|
|
499
|
+
const top = open.at(-1);
|
|
500
|
+
const foreign = Boolean(top && !top.integration);
|
|
501
|
+
const next = html[i + 1] || '';
|
|
502
|
+
if (html.startsWith('<!--', i)) {
|
|
503
|
+
const comment = /<!--(?:>|->|[\s\S]*?(?:--!?>|$))/y;
|
|
504
|
+
comment.lastIndex = i;
|
|
505
|
+
comment.exec(html);
|
|
506
|
+
i = comment.lastIndex;
|
|
507
|
+
continue;
|
|
508
|
+
}
|
|
509
|
+
if (next === '/' && html[i + 2] === '>') {
|
|
510
|
+
i += 3;
|
|
511
|
+
continue;
|
|
512
|
+
}
|
|
513
|
+
if (next === '!' || next === '?' || (next === '/' && !/[a-z]/i.test(html[i + 2] || ''))) {
|
|
514
|
+
if (foreign && html.startsWith('<![CDATA[', i)) unreadable(`a CDATA section in <${top.ns}>`);
|
|
515
|
+
const close = html.indexOf('>', i + 2);
|
|
516
|
+
i = close < 0 ? html.length : close + 1;
|
|
517
|
+
continue;
|
|
518
|
+
}
|
|
519
|
+
if (!/[a-z]/i.test(next === '/' ? html[i + 2] : next)) {
|
|
520
|
+
i += 1;
|
|
521
|
+
continue;
|
|
522
|
+
}
|
|
523
|
+
const tag = readTag(html, i);
|
|
524
|
+
if (!tag) break;
|
|
525
|
+
i = tag.end;
|
|
526
|
+
const { name } = tag;
|
|
527
|
+
if (next === '/') {
|
|
528
|
+
if (name === 'select' && selects) selects -= 1;
|
|
529
|
+
if (foreign && (name === 'br' || name === 'p')) closeForeign();
|
|
530
|
+
else if (foreign) {
|
|
531
|
+
const at = open.findLastIndex((entry) => entry.name === name);
|
|
532
|
+
if (at > open.findLastIndex((entry) => entry.integration)) open.length = at;
|
|
533
|
+
} else if (top && top.name === name) open.pop();
|
|
534
|
+
continue;
|
|
535
|
+
}
|
|
536
|
+
const first = Object.fromEntries([...tag.attributes].reverse());
|
|
537
|
+
for (const [key, value] of tag.attributes) scanAttribute(name, key, value, found);
|
|
538
|
+
if (name === 'base') found.push('a base address <base>');
|
|
539
|
+
if (name === 'meta' && /^\s*(?:refresh|link)\s*$/i.test(first['http-equiv'] || ''))
|
|
540
|
+
found.push(`a ${first['http-equiv'].trim().toLowerCase()} <meta http-equiv>`);
|
|
541
|
+
const opens = (entry) => tag.selfClosing || open.push(entry);
|
|
542
|
+
if (foreign) {
|
|
543
|
+
const html5 = /^(?:text\/html|application\/xhtml\+xml)$/i.test(first.encoding || '');
|
|
544
|
+
if (
|
|
545
|
+
BREAKOUT.has(name) ||
|
|
546
|
+
(name === 'font' && ['color', 'face', 'size'].some((a) => a in first))
|
|
547
|
+
)
|
|
548
|
+
closeForeign();
|
|
549
|
+
else if (
|
|
550
|
+
INTEGRATION[top.ns].has(name) ||
|
|
551
|
+
(top.ns === 'math' && name === 'annotation-xml' && html5)
|
|
552
|
+
)
|
|
553
|
+
opens({ name, ns: top.ns, integration: true });
|
|
554
|
+
else if (SWITCHING.has(name)) unreadable(`a <${name}> inside <${top.ns}>`);
|
|
555
|
+
else if (name === 'svg' || name === 'math') opens({ name, ns: top.ns });
|
|
556
|
+
continue;
|
|
557
|
+
}
|
|
558
|
+
if (top?.ns === 'math' && (name === 'mglyph' || name === 'malignmark'))
|
|
559
|
+
opens({ name, ns: 'math' });
|
|
560
|
+
else if (name === 'svg' || name === 'math') opens({ name, ns: name });
|
|
561
|
+
else if (name === 'frameset') unreadable('a <frameset>');
|
|
562
|
+
else if (name === 'select') selects += 1;
|
|
563
|
+
if (!SWITCHING.has(name)) continue;
|
|
564
|
+
if (selects) {
|
|
565
|
+
unreadable(`a <${name}> inside <select>`);
|
|
566
|
+
continue;
|
|
567
|
+
}
|
|
568
|
+
if (name === 'plaintext') break;
|
|
569
|
+
const end = name === 'script' ? scriptEnd(html, i) : rawTextEnd(html, i, name);
|
|
570
|
+
if (name === 'script') scanScript('<script>', html.slice(i, end), found);
|
|
571
|
+
if (name === 'style') scanCss('<style>', html.slice(i, end), found);
|
|
572
|
+
i = end;
|
|
573
|
+
}
|
|
574
|
+
return found;
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
/** Each address `html` would reach on another origin, named where it sits; empty when none. */
|
|
578
|
+
function foreignAddresses(html) {
|
|
579
|
+
return [...new Set(scanMarkup(html, []))];
|
|
580
|
+
}
|
|
581
|
+
|
|
582
|
+
module.exports = { foreignAddresses, leavesOrigin };
|