@nextcommerce/campaigns-os 1.50.0 → 1.52.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +426 -0
- package/agents/claude/CLAUDE.md +2 -2
- package/agents/codex/AGENTS.md +1 -1
- package/agents/copilot/copilot-instructions.md +1 -1
- package/agents/cursor/campaigns-os.mdc +1 -1
- package/campaign-spec/dist/rules/campaign-metadata.d.ts +5 -1
- package/campaign-spec/dist/rules/campaign-metadata.js +9 -2
- package/campaign-spec/dist/rules/design-source-shape.js +13 -3
- package/campaign-spec/dist/rules/sdk-version.js +2 -1
- package/compatibility.json +1 -1
- package/contracts/commerce-surface-catalog.json +26 -46
- package/contracts/effects.v1.json +81 -2
- package/contracts/release-ledger.json +906 -0
- package/contracts/supported-surface.json +2 -2
- package/contracts/template-brand-contract.shared-commerce.v0.json +2 -2
- package/contracts/template-slot-manifest.shared-content-core.v0.json +24 -0
- package/docs/build-packet.md +93 -9
- package/docs/campaign-build-brief.md +25 -1
- package/docs/effects.md +6 -0
- package/docs/local-setup.md +1 -1
- package/docs/orientation-contract-reference.md +1 -1
- package/docs/polish-evidence.md +10 -0
- package/docs/qa-and-test-orders.md +45 -4
- package/docs/runtime-readiness.md +1 -1
- package/docs/sdk-storage-compatibility.md +1 -1
- package/docs/skills-revision.md +10 -10
- package/package.json +1 -1
- package/skills/campaign-lifecycle-orientation/SKILL.md +3 -3
- package/skills/campaign-readback-classification/SKILL.md +3 -3
- package/skills/campaign-run-evidence/SKILL.md +3 -3
- package/skills/contribution-intake/SKILL.md +3 -3
- package/skills/next-campaigns-build/SKILL.md +3 -3
- package/skills/next-campaigns-os/SKILL.md +4 -4
- package/skills/next-campaigns-os/references/session-intake.md +7 -3
- package/skills/next-campaigns-os-setup/SKILL.md +3 -3
- package/skills/next-campaigns-polish/SKILL.md +3 -3
- package/skills/next-campaigns-qa/SKILL.md +6 -5
- package/skills.json +10 -10
- package/src/adapter-decision-contract.mjs +1 -1
- package/src/brand-theme.mjs +12 -0
- package/src/build-brief.mjs +68 -21
- package/src/built-site-scope.mjs +39 -6
- package/src/built-smoke-qc.mjs +1117 -0
- package/src/campaign-identity.mjs +36 -2
- package/src/cart-placeholders.mjs +730 -0
- package/src/cli.mjs +310 -34
- package/src/commercial-journey.mjs +65 -4
- package/src/commercial-parity.mjs +6 -1
- package/src/doctor/checks.mjs +291 -24
- package/src/doctor/inspect.mjs +53 -2
- package/src/doctor/next-step.mjs +1 -1
- package/src/invocation.mjs +2 -1
- package/src/local-preview-policy.mjs +1 -1
- package/src/local-proof.mjs +4 -1
- package/src/polish-browser.mjs +218 -1
- package/src/polish-capture.mjs +1 -1
- package/src/polish-media-weight.mjs +492 -0
- package/src/polish-node.mjs +96 -4
- package/src/progress-node.mjs +5 -1
- package/src/qa-browser.mjs +308 -96
- package/src/qa-content-params.mjs +889 -0
- package/src/qa-node.mjs +104 -12
- package/src/qa-order-bump.mjs +22 -1
- package/src/qa-policy-links.mjs +1019 -0
- package/src/qa-tracking-params.mjs +1389 -0
- package/src/qa-url-privacy.mjs +168 -0
- package/src/qc-accept.mjs +446 -0
- package/src/qc-check-registry.mjs +83 -0
- package/src/qc-results.mjs +1049 -0
- package/src/sdk-attribute-index.mjs +71 -0
- package/src/sdk-markup.mjs +2 -2
- package/src/sdk-storage-compatibility.mjs +63 -3
- package/src/source-prep.mjs +37 -7
- package/src/stage-record.mjs +56 -17
|
@@ -0,0 +1,1117 @@
|
|
|
1
|
+
// Built-output smoke checks (built_output.smoke_qc).
|
|
2
|
+
//
|
|
3
|
+
// Advisory observations on every built page, each a QC result and never a
|
|
4
|
+
// blocker. A sibling doctor gate, kept out of validateBuiltHtmlStructure
|
|
5
|
+
// (which escalates warnings to errors), so every finding stays a warning.
|
|
6
|
+
//
|
|
7
|
+
// Rules, one result per (page, rule, key); doctor issue code
|
|
8
|
+
// built_output.smoke_qc.<reason_code>:
|
|
9
|
+
// anchor:<target> an in-page link (<a> or <area> href="#x"; SVG <use> is
|
|
10
|
+
// not a link) whose target no element on the page carries
|
|
11
|
+
// as id (or <a name>): warning anchor_target_missing, once
|
|
12
|
+
// every local script the page loads was read; review
|
|
13
|
+
// anchor_target_in_template or
|
|
14
|
+
// anchor_target_possibly_script_created; unexercised
|
|
15
|
+
// script_unreadable when any local script the page loads
|
|
16
|
+
// was missing, outside _site/, over 1 MiB or past the
|
|
17
|
+
// 16-script cap, whatever the scripts read say. `#`,
|
|
18
|
+
// `#top` and the empty fragment are valid. The target is
|
|
19
|
+
// the fragment percent-decoded (UTF-8 bytes, a `%` without
|
|
20
|
+
// two hex digits kept as is) and only that is matched.
|
|
21
|
+
// favicon:link no link[rel~=icon] and no apple-touch-icon link:
|
|
22
|
+
// warning favicon_missing
|
|
23
|
+
// og:title, og:description, og:image
|
|
24
|
+
// warning og_*_missing when the tag is absent
|
|
25
|
+
// og:image_target the og:image destination, only when the page has one:
|
|
26
|
+
// a URL without both a scheme and a host (path-relative,
|
|
27
|
+
// root-relative or scheme-relative `//host/x`) is
|
|
28
|
+
// og_image_not_absolute, or og_image_missing_file when it
|
|
29
|
+
// maps into _site/ and the file is not there (a
|
|
30
|
+
// scheme-relative URL maps only through a deploy base
|
|
31
|
+
// host and path; otherwise its file is never looked for
|
|
32
|
+
// and it reads `relative_present`); an absolute URL under
|
|
33
|
+
// a deploy base (its origin, and a path inside the base's
|
|
34
|
+
// path) passes when its file is in _site/ and is
|
|
35
|
+
// og_image_missing_file otherwise; with no known deploy
|
|
36
|
+
// base it is unexercised og_image_base_unknown, and
|
|
37
|
+
// anywhere else og_image_remote_not_fetched (never
|
|
38
|
+
// fetched)
|
|
39
|
+
// tailwind_cdn:cdn.tailwindcss.com, loopback:loopback
|
|
40
|
+
// decided only for a recorded production build: warning
|
|
41
|
+
// tailwind_cdn_in_production / loopback_url, else pass; a
|
|
42
|
+
// development build reads unexercised development_render
|
|
43
|
+
// and an unknown one build_environment_unknown, on every
|
|
44
|
+
// page whatever it contains
|
|
45
|
+
// asset_host:cdn.29next.store
|
|
46
|
+
// any URL on the primary NEXT asset host, in any
|
|
47
|
+
// environment: warning primary_asset_host
|
|
48
|
+
// The asset-host and loopback rules read every attribute value (data-* and
|
|
49
|
+
// <meta content> included) and <style> text, matching absolute URLs by host.
|
|
50
|
+
// A host is read only from a whole value, never a substring: a URL_ATTRIBUTES
|
|
51
|
+
// value (srcset and imagesrcset split per HTML, ping per whitespace), any
|
|
52
|
+
// other attribute whose whole value is a URL, and the url() and string tokens
|
|
53
|
+
// of a style attribute or <style> text (cssUrlValues), so a URL inside
|
|
54
|
+
// another URL's query names no host. Every host, the Tailwind script's
|
|
55
|
+
// included, is the URL parser's hostname (hostOf). Whether a value is
|
|
56
|
+
// absolute or scheme-relative is read from its own text, never from where it
|
|
57
|
+
// resolves. Markup the HTML parser keeps as text but a visitor can load, the
|
|
58
|
+
// text of a <noscript> and the srcdoc of an <iframe>, is parsed as a document
|
|
59
|
+
// of its own and read by the same two rules. Text nodes and script bodies are
|
|
60
|
+
// not scanned. The per-page candidate cap counts each in-page anchor, each
|
|
61
|
+
// <meta> and <link> (in <template> content too), each URL_ATTRIBUTES value
|
|
62
|
+
// whether relative or absolute, any other attribute value holding an absolute
|
|
63
|
+
// URL, each <style> holding one, and each <noscript> or srcdoc holding markup
|
|
64
|
+
// (whose own candidates count on the page too); that markup together may not
|
|
65
|
+
// exceed the page size cap, and past it the page is capped as well.
|
|
66
|
+
//
|
|
67
|
+
// Every parse goes through the 1.5 gate's bounded parser (parseBounded), so
|
|
68
|
+
// nesting deeper than MAX_ELEMENT_DEPTH (a nested document's depth counting
|
|
69
|
+
// from its host element) stops the parse and reads page_unreadable.
|
|
70
|
+
//
|
|
71
|
+
// Page-level outcomes apply first to every rule key (the bare `anchor` key
|
|
72
|
+
// standing for the anchor rule): page_unreadable, page_too_large and
|
|
73
|
+
// page_cap_reached read unexercised on each key. A page past the candidate cap
|
|
74
|
+
// or with more than 50 anchor results (pass included) is capped: it keeps the
|
|
75
|
+
// warning and review results already observed in full (an anchor's target,
|
|
76
|
+
// read over every id on the page; the og:image destination; a Tailwind,
|
|
77
|
+
// asset-host or loopback URL found) and the first 50 anchor results' findings,
|
|
78
|
+
// and reads unexercised/<cap> on every other key. A finding that rests on
|
|
79
|
+
// something being absent (favicon, og presence) is never kept, nor is any
|
|
80
|
+
// pass. Every result it keeps carries the capped-page member, so none is
|
|
81
|
+
// accept-eligible.
|
|
82
|
+
//
|
|
83
|
+
// Only fragments, the og:image destination (origin and path, or site path),
|
|
84
|
+
// element paths, attribute names and counts are kept; no query string, script
|
|
85
|
+
// body or page text.
|
|
86
|
+
//
|
|
87
|
+
// Callers hand in built HTML and the local scripts each page loads: either
|
|
88
|
+
// `readScripts`, called with the srcs of the scripts the page's own parse
|
|
89
|
+
// lists (doctor's bounded script collector), or a `page.scripts` list
|
|
90
|
+
// ({src, file, content} when read, {src, file, unread} when not); a page with
|
|
91
|
+
// neither reads every local script it loads unread. Every URL reference that
|
|
92
|
+
// names a file under _site/ (the og:image file, each local script) maps
|
|
93
|
+
// through builtFileOf: the URL parser resolves it against the page's own URL
|
|
94
|
+
// under the site root, and each segment of the resulting path is
|
|
95
|
+
// percent-decoded onto `siteRoot`. A path that names no file (`%ZZ`, bytes
|
|
96
|
+
// that are not UTF-8, an empty segment, so a trailing `/`) is never a file.
|
|
97
|
+
// The og:image file is stat-ed here, once per og:image, and counts only when
|
|
98
|
+
// its real path (symlinks followed) lies inside `siteRoot`. No network
|
|
99
|
+
// request.
|
|
100
|
+
|
|
101
|
+
import { realpathSync, statSync } from "node:fs";
|
|
102
|
+
import { dirname, isAbsolute, join, relative, sep } from "node:path";
|
|
103
|
+
|
|
104
|
+
import { CART_PLACEHOLDERS_LIMITS, MAX_ELEMENT_DEPTH, isFileReadFailure, isPageReadFailure, parseBounded, scriptKind } from "./cart-placeholders.mjs";
|
|
105
|
+
import { aggregateQcResults, buildQcResult } from "./qc-results.mjs";
|
|
106
|
+
import { isLoopbackHostname } from "./remit.mjs";
|
|
107
|
+
|
|
108
|
+
export const SMOKE_QC = "built_output.smoke_qc";
|
|
109
|
+
export const SMOKE_QC_CHECK = "smoke_qc";
|
|
110
|
+
|
|
111
|
+
export const PRIMARY_ASSET_HOST = "cdn.29next.store";
|
|
112
|
+
export const TAILWIND_CDN_HOST = "cdn.tailwindcss.com";
|
|
113
|
+
|
|
114
|
+
// Shared doctor bounds, plus the anchor script hint's own.
|
|
115
|
+
export const SMOKE_QC_LIMITS = Object.freeze({
|
|
116
|
+
...CART_PLACEHOLDERS_LIMITS,
|
|
117
|
+
scripts: 16,
|
|
118
|
+
script_bytes: 1024 * 1024,
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
export const SMOKE_QC_REASONS = Object.freeze({
|
|
122
|
+
ANCHOR_TARGET_MISSING: "anchor_target_missing",
|
|
123
|
+
ANCHOR_TARGET_IN_TEMPLATE: "anchor_target_in_template",
|
|
124
|
+
ANCHOR_TARGET_POSSIBLY_SCRIPT_CREATED: "anchor_target_possibly_script_created",
|
|
125
|
+
SCRIPT_UNREADABLE: "script_unreadable",
|
|
126
|
+
FAVICON_MISSING: "favicon_missing",
|
|
127
|
+
OG_TITLE_MISSING: "og_title_missing",
|
|
128
|
+
OG_DESCRIPTION_MISSING: "og_description_missing",
|
|
129
|
+
OG_IMAGE_MISSING: "og_image_missing",
|
|
130
|
+
OG_IMAGE_MISSING_FILE: "og_image_missing_file",
|
|
131
|
+
OG_IMAGE_NOT_ABSOLUTE: "og_image_not_absolute",
|
|
132
|
+
OG_IMAGE_BASE_UNKNOWN: "og_image_base_unknown",
|
|
133
|
+
OG_IMAGE_REMOTE_NOT_FETCHED: "og_image_remote_not_fetched",
|
|
134
|
+
TAILWIND_CDN_IN_PRODUCTION: "tailwind_cdn_in_production",
|
|
135
|
+
PRIMARY_ASSET_HOST: "primary_asset_host",
|
|
136
|
+
LOOPBACK_URL: "loopback_url",
|
|
137
|
+
DEVELOPMENT_RENDER: "development_render",
|
|
138
|
+
BUILD_ENVIRONMENT_UNKNOWN: "build_environment_unknown",
|
|
139
|
+
PAGE_UNREADABLE: "page_unreadable",
|
|
140
|
+
PAGE_TOO_LARGE: "page_too_large",
|
|
141
|
+
PAGE_CAP_REACHED: "page_cap_reached",
|
|
142
|
+
CANDIDATE_CAP_REACHED: "candidate_cap_reached",
|
|
143
|
+
FINDING_CAP_REACHED: "finding_cap_reached",
|
|
144
|
+
});
|
|
145
|
+
const R = SMOKE_QC_REASONS;
|
|
146
|
+
|
|
147
|
+
// The result key of every rule but the per-target anchors. `anchor` is the
|
|
148
|
+
// anchor rule's own row, used only when its targets are not known (a
|
|
149
|
+
// page-level outcome) and as the finding-cap row.
|
|
150
|
+
export const SMOKE_QC_KEYS = Object.freeze({
|
|
151
|
+
anchor: "anchor",
|
|
152
|
+
favicon: "favicon:link",
|
|
153
|
+
og_title: "og:title",
|
|
154
|
+
og_description: "og:description",
|
|
155
|
+
og_image: "og:image",
|
|
156
|
+
og_image_target: "og:image_target",
|
|
157
|
+
tailwind_cdn: `tailwind_cdn:${TAILWIND_CDN_HOST}`,
|
|
158
|
+
asset_host: `asset_host:${PRIMARY_ASSET_HOST}`,
|
|
159
|
+
loopback: "loopback:loopback",
|
|
160
|
+
});
|
|
161
|
+
const K = SMOKE_QC_KEYS;
|
|
162
|
+
const anchorKey = (target) => `anchor:${target}`;
|
|
163
|
+
|
|
164
|
+
const ENVIRONMENTS = new Set(["production", "development"]);
|
|
165
|
+
const ENVIRONMENT_REASON = Object.freeze({ development: R.DEVELOPMENT_RENDER, unknown: R.BUILD_ENVIRONMENT_UNKNOWN });
|
|
166
|
+
const environmentOf = (value) => (ENVIRONMENTS.has(value) ? value : "unknown");
|
|
167
|
+
|
|
168
|
+
const HTML_NAMESPACE = "http://www.w3.org/1999/xhtml";
|
|
169
|
+
const WEB_PROTOCOLS = new Set(["http:", "https:"]);
|
|
170
|
+
const OG_FIELDS = new Map([["og:title", "title"], ["og:description", "description"], ["og:image", "image"]]);
|
|
171
|
+
const ASCII_WHITESPACE = /[\t\n\f\r ]+/;
|
|
172
|
+
|
|
173
|
+
// The origin built pages are read from: the site root (_site/) is its path
|
|
174
|
+
// `/`, so a page's URL is its path under _site/ and every reference on it
|
|
175
|
+
// resolves as a browser resolves it there. Only a path reference resolves
|
|
176
|
+
// against it; no reference is ever judged by whether it lands on this origin,
|
|
177
|
+
// so an absolute URL that names it is just another remote URL.
|
|
178
|
+
const SITE_ORIGIN = "https://built.invalid";
|
|
179
|
+
const SITE_BASE = `${SITE_ORIGIN}/`;
|
|
180
|
+
|
|
181
|
+
const parseUrl = (value, base) => {
|
|
182
|
+
try {
|
|
183
|
+
return new URL(value, base);
|
|
184
|
+
} catch {
|
|
185
|
+
return null;
|
|
186
|
+
}
|
|
187
|
+
};
|
|
188
|
+
|
|
189
|
+
// What a reference is, read from its own text and never from where it
|
|
190
|
+
// resolves: `absolute` when the URL parser reads it with no base (it has a
|
|
191
|
+
// scheme), `scheme_relative` when it starts with two slashes or backslashes
|
|
192
|
+
// (after the leading C0 controls and spaces, and the tabs and newlines, the
|
|
193
|
+
// parser drops), else a `path`. `url` is the parsed URL of an absolute
|
|
194
|
+
// reference, or of a scheme-relative one resolved against `base`.
|
|
195
|
+
function readReference(value, base = SITE_BASE) {
|
|
196
|
+
const absolute = parseUrl(value);
|
|
197
|
+
if (absolute) return { kind: "absolute", url: absolute };
|
|
198
|
+
const text = String(value).replace(/^[\u0000- ]+/, "").replace(/[\t\n\r]/g, "");
|
|
199
|
+
if (/^[/\\]{2}/.test(text)) return { kind: "scheme_relative", url: parseUrl(value, base) };
|
|
200
|
+
return { kind: "path", url: null };
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// The host a candidate names, read by the URL parser (userinfo, port, IPv6
|
|
204
|
+
// brackets, IDN, case and IPv4 forms are its own): an http(s) URL that is
|
|
205
|
+
// absolute or protocol-relative, its hostname without a trailing root dot.
|
|
206
|
+
// Anything else, a relative path included, names no host.
|
|
207
|
+
function hostOf(candidate) {
|
|
208
|
+
const { kind, url } = readReference(candidate);
|
|
209
|
+
if (kind === "path" || !url || !WEB_PROTOCOLS.has(url.protocol)) return null;
|
|
210
|
+
return url.hostname.replace(/\.$/, "") || null;
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const isAsciiWhitespace = (c) => c === "\t" || c === "\n" || c === "\f" || c === "\r" || c === " ";
|
|
214
|
+
const trimAsciiWhitespace = (value) => value.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, "");
|
|
215
|
+
|
|
216
|
+
// The URLs of a srcset or imagesrcset value, split as the HTML "parse a srcset
|
|
217
|
+
// attribute" algorithm splits it: skip ASCII whitespace and commas, take the
|
|
218
|
+
// run up to the next ASCII whitespace as the URL (a comma inside it stays;
|
|
219
|
+
// trailing commas end the entry), then skip its descriptors up to a comma
|
|
220
|
+
// outside parentheses. Descriptors are not validated.
|
|
221
|
+
function srcsetUrls(value) {
|
|
222
|
+
const urls = [];
|
|
223
|
+
let i = 0;
|
|
224
|
+
for (;;) {
|
|
225
|
+
while (i < value.length && (isAsciiWhitespace(value[i]) || value[i] === ",")) i += 1;
|
|
226
|
+
if (i >= value.length) return urls;
|
|
227
|
+
const start = i;
|
|
228
|
+
while (i < value.length && !isAsciiWhitespace(value[i])) i += 1;
|
|
229
|
+
const url = value.slice(start, i);
|
|
230
|
+
if (url.endsWith(",")) {
|
|
231
|
+
urls.push(url.replace(/,+$/, ""));
|
|
232
|
+
continue;
|
|
233
|
+
}
|
|
234
|
+
urls.push(url);
|
|
235
|
+
let inParens = false;
|
|
236
|
+
while (i < value.length) {
|
|
237
|
+
const c = value[i];
|
|
238
|
+
i += 1;
|
|
239
|
+
if (c === "(") inParens = true;
|
|
240
|
+
else if (c === ")") inParens = false;
|
|
241
|
+
else if (c === "," && !inParens) break;
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
// CSS Syntax 3 code point classes (§4.2), on preprocessed input (§3.3).
|
|
247
|
+
const isCssWhitespace = (c) => c === "\n" || c === "\t" || c === " ";
|
|
248
|
+
const isDigit = (c) => c !== undefined && c >= "0" && c <= "9";
|
|
249
|
+
const isHexDigit = (c) => c !== undefined && /[0-9A-Fa-f]/.test(c);
|
|
250
|
+
const isIdentStart = (c) => c !== undefined && (/[A-Za-z_]/.test(c) || c.charCodeAt(0) >= 0x80);
|
|
251
|
+
const isIdentCodePoint = (c) => isIdentStart(c) || isDigit(c) || c === "-";
|
|
252
|
+
const isNonPrintable = (c) => c !== undefined && /[\u0000-\u0008\u000B\u000E-\u001F\u007F]/.test(c);
|
|
253
|
+
const isValidEscape = (a, b) => a === "\\" && b !== "\n";
|
|
254
|
+
const startsIdentSequence = (a, b, c) => {
|
|
255
|
+
if (a === "-") return isIdentStart(b) || b === "-" || isValidEscape(b, c);
|
|
256
|
+
return isIdentStart(a) || isValidEscape(a, b);
|
|
257
|
+
};
|
|
258
|
+
const startsNumber = (a, b, c) => {
|
|
259
|
+
if (a === "+" || a === "-") return isDigit(b) || (b === "." && isDigit(c));
|
|
260
|
+
if (a === ".") return isDigit(b);
|
|
261
|
+
return isDigit(a);
|
|
262
|
+
};
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* The value of every <url-token> and <string-token> in CSS text (a <style>
|
|
266
|
+
* element's text or a style attribute), tokenized per CSS Syntax 3 §4.3:
|
|
267
|
+
* escapes decoded (consume an escaped code point), `url(` recognized only as
|
|
268
|
+
* an ident-like token (so `u\72l(` is one, a `1url(` dimension is not), and
|
|
269
|
+
* `url( "…" )` read as a function whose string token follows. Comments,
|
|
270
|
+
* <bad-url-token>s and <bad-string-token>s give no value. Every other token is
|
|
271
|
+
* consumed only to find where the next one starts.
|
|
272
|
+
*
|
|
273
|
+
* @param {string} text
|
|
274
|
+
* @returns {string[]}
|
|
275
|
+
*/
|
|
276
|
+
export function cssUrlValues(text) {
|
|
277
|
+
const s = String(text).replace(/\r\n?|\f/g, "\n").replace(/\u0000/g, "�");
|
|
278
|
+
const values = [];
|
|
279
|
+
let i = 0;
|
|
280
|
+
|
|
281
|
+
// §4.3.7, the backslash already consumed.
|
|
282
|
+
const escapedCodePoint = () => {
|
|
283
|
+
if (i >= s.length) return "�";
|
|
284
|
+
if (isHexDigit(s[i])) {
|
|
285
|
+
let hex = "";
|
|
286
|
+
while (hex.length < 6 && isHexDigit(s[i])) hex += s[i++];
|
|
287
|
+
if (isCssWhitespace(s[i])) i += 1;
|
|
288
|
+
const value = Number.parseInt(hex, 16);
|
|
289
|
+
return value === 0 || (value >= 0xd800 && value <= 0xdfff) || value > 0x10ffff ? "�" : String.fromCodePoint(value);
|
|
290
|
+
}
|
|
291
|
+
const char = String.fromCodePoint(s.codePointAt(i));
|
|
292
|
+
i += char.length;
|
|
293
|
+
return char;
|
|
294
|
+
};
|
|
295
|
+
// §4.3.12.
|
|
296
|
+
const identSequence = () => {
|
|
297
|
+
let out = "";
|
|
298
|
+
for (;;) {
|
|
299
|
+
if (isIdentCodePoint(s[i])) {
|
|
300
|
+
out += s[i];
|
|
301
|
+
i += 1;
|
|
302
|
+
} else if (isValidEscape(s[i], s[i + 1])) {
|
|
303
|
+
i += 1;
|
|
304
|
+
out += escapedCodePoint();
|
|
305
|
+
} else {
|
|
306
|
+
return out;
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
};
|
|
310
|
+
// §4.3.5, the opening quote already consumed; null for a <bad-string-token>.
|
|
311
|
+
const stringToken = (ending) => {
|
|
312
|
+
let out = "";
|
|
313
|
+
while (i < s.length) {
|
|
314
|
+
const c = s[i];
|
|
315
|
+
if (c === "\n") return null;
|
|
316
|
+
i += 1;
|
|
317
|
+
if (c === ending) return out;
|
|
318
|
+
if (c !== "\\") out += c;
|
|
319
|
+
else if (s[i] === "\n") i += 1;
|
|
320
|
+
else if (i < s.length) out += escapedCodePoint();
|
|
321
|
+
}
|
|
322
|
+
return out;
|
|
323
|
+
};
|
|
324
|
+
// §4.3.14.
|
|
325
|
+
const badUrlRemnants = () => {
|
|
326
|
+
while (i < s.length) {
|
|
327
|
+
const c = s[i];
|
|
328
|
+
i += 1;
|
|
329
|
+
if (c === ")") return;
|
|
330
|
+
if (isValidEscape(c, s[i])) escapedCodePoint();
|
|
331
|
+
}
|
|
332
|
+
};
|
|
333
|
+
// §4.3.6, `url(` already consumed; null for a <bad-url-token>.
|
|
334
|
+
const urlToken = () => {
|
|
335
|
+
while (isCssWhitespace(s[i])) i += 1;
|
|
336
|
+
let out = "";
|
|
337
|
+
while (i < s.length) {
|
|
338
|
+
const c = s[i];
|
|
339
|
+
i += 1;
|
|
340
|
+
if (c === ")") return out;
|
|
341
|
+
if (isCssWhitespace(c)) {
|
|
342
|
+
while (isCssWhitespace(s[i])) i += 1;
|
|
343
|
+
if (i >= s.length) return out;
|
|
344
|
+
if (s[i] === ")") {
|
|
345
|
+
i += 1;
|
|
346
|
+
return out;
|
|
347
|
+
}
|
|
348
|
+
badUrlRemnants();
|
|
349
|
+
return null;
|
|
350
|
+
}
|
|
351
|
+
if (c === "\"" || c === "'" || c === "(" || isNonPrintable(c) || (c === "\\" && !isValidEscape(c, s[i]))) {
|
|
352
|
+
badUrlRemnants();
|
|
353
|
+
return null;
|
|
354
|
+
}
|
|
355
|
+
out += c === "\\" ? escapedCodePoint() : c;
|
|
356
|
+
}
|
|
357
|
+
return out;
|
|
358
|
+
};
|
|
359
|
+
// §4.3.4.
|
|
360
|
+
const identLikeToken = () => {
|
|
361
|
+
const name = identSequence();
|
|
362
|
+
if (s[i] !== "(") return;
|
|
363
|
+
i += 1;
|
|
364
|
+
if (!/^url$/i.test(name)) return;
|
|
365
|
+
while (isCssWhitespace(s[i]) && isCssWhitespace(s[i + 1])) i += 1;
|
|
366
|
+
const next = isCssWhitespace(s[i]) ? s[i + 1] : s[i];
|
|
367
|
+
if (next === "\"" || next === "'") return;
|
|
368
|
+
const value = urlToken();
|
|
369
|
+
if (value != null) values.push(value);
|
|
370
|
+
};
|
|
371
|
+
// §4.3.3.
|
|
372
|
+
const numericToken = () => {
|
|
373
|
+
if (s[i] === "+" || s[i] === "-") i += 1;
|
|
374
|
+
while (isDigit(s[i])) i += 1;
|
|
375
|
+
if (s[i] === "." && isDigit(s[i + 1])) {
|
|
376
|
+
i += 1;
|
|
377
|
+
while (isDigit(s[i])) i += 1;
|
|
378
|
+
}
|
|
379
|
+
if ((s[i] === "e" || s[i] === "E") && (isDigit(s[i + 1]) || ((s[i + 1] === "+" || s[i + 1] === "-") && isDigit(s[i + 2])))) {
|
|
380
|
+
i += 2;
|
|
381
|
+
while (isDigit(s[i])) i += 1;
|
|
382
|
+
}
|
|
383
|
+
if (startsIdentSequence(s[i], s[i + 1], s[i + 2])) identSequence();
|
|
384
|
+
else if (s[i] === "%") i += 1;
|
|
385
|
+
};
|
|
386
|
+
|
|
387
|
+
// §4.3.1 (whitespace and single-code-point tokens advance by one).
|
|
388
|
+
while (i < s.length) {
|
|
389
|
+
const c = s[i];
|
|
390
|
+
if (c === "/" && s[i + 1] === "*") {
|
|
391
|
+
const end = s.indexOf("*/", i + 2);
|
|
392
|
+
i = end === -1 ? s.length : end + 2;
|
|
393
|
+
} else if (c === "\"" || c === "'") {
|
|
394
|
+
i += 1;
|
|
395
|
+
const value = stringToken(c);
|
|
396
|
+
if (value != null) values.push(value);
|
|
397
|
+
} else if (c === "#") {
|
|
398
|
+
i += 1;
|
|
399
|
+
if (isIdentCodePoint(s[i]) || isValidEscape(s[i], s[i + 1])) identSequence();
|
|
400
|
+
} else if (c === "+" || c === ".") {
|
|
401
|
+
if (startsNumber(c, s[i + 1], s[i + 2])) numericToken();
|
|
402
|
+
else i += 1;
|
|
403
|
+
} else if (c === "-") {
|
|
404
|
+
if (startsNumber(c, s[i + 1], s[i + 2])) numericToken();
|
|
405
|
+
else if (s[i + 1] === "-" && s[i + 2] === ">") i += 3;
|
|
406
|
+
else if (startsIdentSequence(c, s[i + 1], s[i + 2])) identLikeToken();
|
|
407
|
+
else i += 1;
|
|
408
|
+
} else if (c === "<") {
|
|
409
|
+
i += s.startsWith("<!--", i) ? 4 : 1;
|
|
410
|
+
} else if (c === "@") {
|
|
411
|
+
i += 1;
|
|
412
|
+
if (startsIdentSequence(s[i], s[i + 1], s[i + 2])) identSequence();
|
|
413
|
+
} else if (c === "\\") {
|
|
414
|
+
if (isValidEscape(c, s[i + 1])) identLikeToken();
|
|
415
|
+
else i += 1;
|
|
416
|
+
} else if (isDigit(c)) {
|
|
417
|
+
numericToken();
|
|
418
|
+
} else if (isIdentStart(c)) {
|
|
419
|
+
identLikeToken();
|
|
420
|
+
} else {
|
|
421
|
+
i += 1;
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
return values;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
// The host candidates of one attribute value, each a whole value: a srcset or
|
|
428
|
+
// imagesrcset URL, a ping URL (split on ASCII whitespace), a url() or string
|
|
429
|
+
// token of a style attribute, or else the whole trimmed value. hostOf reads
|
|
430
|
+
// each one; a value that is not an absolute or protocol-relative URL names no
|
|
431
|
+
// host, so outside URL_ATTRIBUTES only a whole URL counts.
|
|
432
|
+
function attributeCandidates(name, value) {
|
|
433
|
+
if (name === "srcset" || name === "imagesrcset") return srcsetUrls(value);
|
|
434
|
+
if (name === "ping") return value.split(ASCII_WHITESPACE).filter(Boolean);
|
|
435
|
+
if (name === "style") return cssUrlValues(value);
|
|
436
|
+
return [trimAsciiWhitespace(value)];
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
function hostsOf(candidates) {
|
|
440
|
+
const hosts = new Set();
|
|
441
|
+
for (const candidate of candidates) {
|
|
442
|
+
const host = hostOf(candidate);
|
|
443
|
+
if (host) hosts.add(host);
|
|
444
|
+
}
|
|
445
|
+
return [...hosts];
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
const attrsOf = (node) => {
|
|
449
|
+
const map = new Map();
|
|
450
|
+
for (const attr of node.attrs || []) map.set(attr.name.toLowerCase(), attr.value);
|
|
451
|
+
return map;
|
|
452
|
+
};
|
|
453
|
+
const attrName = (attr) => (attr.prefix ? `${attr.prefix}:${attr.name}` : attr.name).toLowerCase();
|
|
454
|
+
const childrenOf = (node) => (node.tagName === "template" && node.content ? node.content.childNodes : node.childNodes) || [];
|
|
455
|
+
const textOf = (node) => (node.childNodes || []).filter((child) => child.nodeName === "#text").map((child) => child.value || "").join("");
|
|
456
|
+
|
|
457
|
+
const isHexByte = (byte) => (byte >= 0x30 && byte <= 0x39) || (byte >= 0x41 && byte <= 0x46) || (byte >= 0x61 && byte <= 0x66);
|
|
458
|
+
const UTF8 = new TextDecoder("utf-8", { ignoreBOM: true });
|
|
459
|
+
const UTF8_STRICT = new TextDecoder("utf-8", { ignoreBOM: true, fatal: true });
|
|
460
|
+
|
|
461
|
+
// The bytes `raw` percent-decodes to over its UTF-8 bytes. A `%` not followed
|
|
462
|
+
// by two hex digits is kept as is, or, with `strict`, makes the whole value
|
|
463
|
+
// undecodable (null).
|
|
464
|
+
function percentBytes(raw, { strict = false } = {}) {
|
|
465
|
+
const input = Buffer.from(raw, "utf8");
|
|
466
|
+
const bytes = [];
|
|
467
|
+
for (let i = 0; i < input.length; i += 1) {
|
|
468
|
+
if (input[i] === 0x25 && i + 2 < input.length && isHexByte(input[i + 1]) && isHexByte(input[i + 2])) {
|
|
469
|
+
bytes.push(Number.parseInt(String.fromCharCode(input[i + 1], input[i + 2]), 16));
|
|
470
|
+
i += 2;
|
|
471
|
+
} else if (input[i] === 0x25 && strict) {
|
|
472
|
+
return null;
|
|
473
|
+
} else {
|
|
474
|
+
bytes.push(input[i]);
|
|
475
|
+
}
|
|
476
|
+
}
|
|
477
|
+
return Uint8Array.from(bytes);
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
// A fragment as the target it names (contract 1.6 "Fragments are
|
|
481
|
+
// percent-decoded"): WHATWG percent-decode over its UTF-8 bytes, a `%` not
|
|
482
|
+
// followed by two hex digits kept as is, then UTF-8 decode (an invalid
|
|
483
|
+
// sequence reads U+FFFD). `#caf%C3%A9` names `café`, never `caf%C3%A9`.
|
|
484
|
+
function decodeFragment(raw) {
|
|
485
|
+
if (!raw.includes("%")) return raw;
|
|
486
|
+
return UTF8.decode(percentBytes(raw));
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
// One URL path segment as the file name it names, or null when it names none:
|
|
490
|
+
// a `%` without two hex digits, bytes that are not UTF-8, or a decoded `/` or
|
|
491
|
+
// NUL (no file name holds either).
|
|
492
|
+
function decodePathSegment(segment) {
|
|
493
|
+
if (!segment.includes("%")) return segment;
|
|
494
|
+
const bytes = percentBytes(segment, { strict: true });
|
|
495
|
+
if (!bytes) return null;
|
|
496
|
+
let decoded;
|
|
497
|
+
try {
|
|
498
|
+
decoded = UTF8_STRICT.decode(bytes);
|
|
499
|
+
} catch {
|
|
500
|
+
return null;
|
|
501
|
+
}
|
|
502
|
+
return /[/\u0000]/.test(decoded) ? null : decoded;
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
// A page's own URL: its path under the site root, each segment
|
|
506
|
+
// percent-encoded; null when the page is not under the site root.
|
|
507
|
+
function pageUrlOf(builtPath, siteRoot) {
|
|
508
|
+
if (!siteRoot || !insideRoot(siteRoot, builtPath)) return null;
|
|
509
|
+
return `${SITE_ORIGIN}/${relative(siteRoot, builtPath).split(sep).map(encodeURIComponent).join("/")}`;
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
/**
|
|
513
|
+
* The one mapping from a URL reference on a built page to a file under
|
|
514
|
+
* `_site/`, used for the og:image file and every local script. The URL parser
|
|
515
|
+
* resolves the reference against the page's own URL (pageUrlOf), as a browser
|
|
516
|
+
* does: backslashes read as `/`, dot segments are removed, a root-relative
|
|
517
|
+
* reference starts at the site root, and the query and fragment are not part
|
|
518
|
+
* of the path. Only a path reference (see readReference) is local, or an
|
|
519
|
+
* absolute or scheme-relative one under one of `bases` (a deploy base: its
|
|
520
|
+
* origin, or its host for a scheme-relative one, and a path starting with the
|
|
521
|
+
* base's path). Each segment of its path is then percent-decoded and joined
|
|
522
|
+
* onto `siteRoot`: `a%23b.png` names `a#b.png`, never a file literally called
|
|
523
|
+
* `a%23b.png`. An empty segment names no file: `og.png/` (or `og.png/%2e`,
|
|
524
|
+
* which the parser reads as `og.png/`) is a directory, and `a//b.png` is not
|
|
525
|
+
* `a/b.png`. Whether the file is inside the site root is the reader's check
|
|
526
|
+
* (its real path).
|
|
527
|
+
*
|
|
528
|
+
* @param {string} reference the attribute value
|
|
529
|
+
* @param {string} builtPath the page's file
|
|
530
|
+
* @param {string|null} siteRoot the built `_site/` directory
|
|
531
|
+
* @param {{ bases?: Array<{ origin: string, host: string, prefix: string }> }} [options]
|
|
532
|
+
* @returns {null|{ path: string, site_path: string }|{ unmappable: true, site_path?: string }}
|
|
533
|
+
* null when the reference is not local (anywhere but a deploy base, data:,
|
|
534
|
+
* or no path at all); unmappable when its path names no file (see
|
|
535
|
+
* decodePathSegment, and an empty segment), or when there is no site root
|
|
536
|
+
* or page URL to resolve it from.
|
|
537
|
+
*/
|
|
538
|
+
export function builtFileOf(reference, builtPath, siteRoot, { bases = [] } = {}) {
|
|
539
|
+
const value = String(reference ?? "").trim();
|
|
540
|
+
// An empty reference, or one with only a query or fragment, names the page
|
|
541
|
+
// itself, not a file it loads.
|
|
542
|
+
if (!value || value.startsWith("?") || value.startsWith("#")) return null;
|
|
543
|
+
const pageUrl = pageUrlOf(builtPath, siteRoot);
|
|
544
|
+
const read = readReference(value, pageUrl ?? SITE_BASE);
|
|
545
|
+
let url;
|
|
546
|
+
if (read.kind === "path") {
|
|
547
|
+
if (!pageUrl) return parseUrl(value, SITE_BASE) ? { unmappable: true } : null;
|
|
548
|
+
url = parseUrl(value, pageUrl);
|
|
549
|
+
if (!url) return null;
|
|
550
|
+
} else {
|
|
551
|
+
url = read.url;
|
|
552
|
+
if (!url || !WEB_PROTOCOLS.has(url.protocol)) return null;
|
|
553
|
+
const same = read.kind === "absolute" ? (base) => base.origin === url.origin : (base) => base.host === url.host;
|
|
554
|
+
if (!bases.some((base) => same(base) && url.pathname.startsWith(base.prefix))) return null;
|
|
555
|
+
if (!siteRoot) return { unmappable: true };
|
|
556
|
+
}
|
|
557
|
+
const segments = url.pathname.split("/").slice(1).map(decodePathSegment);
|
|
558
|
+
if (segments.some((segment) => !segment || segment === "." || segment === "..")) return { unmappable: true, site_path: url.pathname };
|
|
559
|
+
return { path: join(siteRoot, ...segments), site_path: url.pathname };
|
|
560
|
+
}
|
|
561
|
+
|
|
562
|
+
// Nesting past MAX_ELEMENT_DEPTH found after the parse: nodes the parser
|
|
563
|
+
// moved, or a nested document's depth added to its host element's.
|
|
564
|
+
class TooDeep extends Error {}
|
|
565
|
+
|
|
566
|
+
// Whether an error means the page cannot be read or parsed (see
|
|
567
|
+
// isPageReadFailure), nesting past the bound found after the parse included.
|
|
568
|
+
export const isBuiltPageReadFailure = (error) => error instanceof TooDeep || isPageReadFailure(error);
|
|
569
|
+
|
|
570
|
+
// Every element in document order with its path (`html[1]/body[1]/a[2]`),
|
|
571
|
+
// its depth (the root element is 1) and whether it sits in <template>
|
|
572
|
+
// content, without recursion. A nested document starts at its host element's
|
|
573
|
+
// `path` and `depth`. An element deeper than MAX_ELEMENT_DEPTH throws TooDeep.
|
|
574
|
+
function walkElements(document, visit, { path: rootPath = "", depth: rootDepth = 0 } = {}) {
|
|
575
|
+
const stack = [];
|
|
576
|
+
const pushChildren = (node, path, inTemplate, depth) => {
|
|
577
|
+
const counts = new Map();
|
|
578
|
+
const frames = [];
|
|
579
|
+
for (const child of childrenOf(node)) {
|
|
580
|
+
if (!child.tagName) continue;
|
|
581
|
+
const tag = child.tagName.toLowerCase();
|
|
582
|
+
const index = (counts.get(tag) || 0) + 1;
|
|
583
|
+
counts.set(tag, index);
|
|
584
|
+
frames.push({ node: child, tag, path: `${path ? `${path}/` : ""}${tag}[${index}]`, inTemplate, depth: depth + 1 });
|
|
585
|
+
}
|
|
586
|
+
for (let i = frames.length - 1; i >= 0; i -= 1) stack.push(frames[i]);
|
|
587
|
+
};
|
|
588
|
+
pushChildren(document, rootPath, false, rootDepth);
|
|
589
|
+
while (stack.length) {
|
|
590
|
+
const frame = stack.pop();
|
|
591
|
+
if (frame.depth > MAX_ELEMENT_DEPTH) throw new TooDeep();
|
|
592
|
+
visit(frame);
|
|
593
|
+
pushChildren(frame.node, frame.path, frame.inTemplate || frame.tag === "template", frame.depth);
|
|
594
|
+
}
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
class CandidateCap {
|
|
598
|
+
constructor(limit, markupBytes) {
|
|
599
|
+
this.limit = limit;
|
|
600
|
+
this.count = 0;
|
|
601
|
+
this.reached = false;
|
|
602
|
+
this.markupBytes = markupBytes;
|
|
603
|
+
this.markupRead = 0;
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
// Whether one more candidate may be examined.
|
|
607
|
+
take() {
|
|
608
|
+
if (this.reached) return false;
|
|
609
|
+
if (this.count === this.limit) {
|
|
610
|
+
this.reached = true;
|
|
611
|
+
return false;
|
|
612
|
+
}
|
|
613
|
+
this.count += 1;
|
|
614
|
+
return true;
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
// Whether one more nested document of `bytes` may be parsed: one candidate,
|
|
618
|
+
// and the page's nested markup all together within the page size cap.
|
|
619
|
+
takeMarkup(bytes) {
|
|
620
|
+
if (this.reached) return false;
|
|
621
|
+
if (this.markupRead + bytes > this.markupBytes) {
|
|
622
|
+
this.reached = true;
|
|
623
|
+
return false;
|
|
624
|
+
}
|
|
625
|
+
if (!this.take()) return false;
|
|
626
|
+
this.markupRead += bytes;
|
|
627
|
+
return true;
|
|
628
|
+
}
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
// The URL-bearing attributes, a closed list: the HTML Living Standard
|
|
632
|
+
// attribute index entries whose value is a URL, a URL list or a hash-name
|
|
633
|
+
// reference, the obsolete URL attributes, and SVG href and xlink:href. Each
|
|
634
|
+
// non-empty value is one candidate, relative or absolute (a srcset or ping
|
|
635
|
+
// list counts once). Any other attribute (data-*, <meta content>, style, ...)
|
|
636
|
+
// is a candidate only when it holds an absolute URL, the one form the
|
|
637
|
+
// asset-host and loopback rules read from it. Nothing outside the list is
|
|
638
|
+
// added.
|
|
639
|
+
export const URL_ATTRIBUTES = Object.freeze(new Set([
|
|
640
|
+
"href", "src", "srcset", "imagesrcset", "poster", "action", "formaction", "data", "cite",
|
|
641
|
+
"ping", "itemid", "itemtype", "usemap", "background", "longdesc", "manifest", "codebase",
|
|
642
|
+
"classid", "archive", "profile", "lowsrc", "dynsrc", "xlink:href",
|
|
643
|
+
]));
|
|
644
|
+
|
|
645
|
+
// The src of a <script> the page loads, or null: an HTML <script> with a
|
|
646
|
+
// non-empty src whose type runs as JavaScript (classic or module) and that is
|
|
647
|
+
// not in <template> content. <noscript> content parses as text, so a script
|
|
648
|
+
// there is never an element of the page.
|
|
649
|
+
function loadedScriptSrc(node, tag, inTemplate) {
|
|
650
|
+
if (inTemplate || tag !== "script" || node.namespaceURI !== HTML_NAMESPACE) return null;
|
|
651
|
+
const attrs = attrsOf(node);
|
|
652
|
+
const src = attrs.get("src");
|
|
653
|
+
return typeof src === "string" && src.trim() && scriptKind(attrs) ? src : null;
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
/**
|
|
657
|
+
* One built page's parse5 tree, through the 1.5 gate's bounded parser. Throws
|
|
658
|
+
* what it throws (isBuiltPageReadFailure reads it).
|
|
659
|
+
*
|
|
660
|
+
* @param {string} content the page's HTML
|
|
661
|
+
*/
|
|
662
|
+
export const parseBuiltPage = (content) => parseBounded(content);
|
|
663
|
+
|
|
664
|
+
/**
|
|
665
|
+
* The src of every script a built page loads, in document order, read from
|
|
666
|
+
* its parse5 tree (see loadedScriptSrc).
|
|
667
|
+
*
|
|
668
|
+
* @param {object} document the page's tree (parseBuiltPage)
|
|
669
|
+
* @returns {string[]}
|
|
670
|
+
*/
|
|
671
|
+
export function pageScriptSources(document) {
|
|
672
|
+
const sources = [];
|
|
673
|
+
walkElements(document, ({ node, tag, inTemplate }) => {
|
|
674
|
+
const src = loadedScriptSrc(node, tag, inTemplate);
|
|
675
|
+
if (src != null) sources.push(src);
|
|
676
|
+
});
|
|
677
|
+
return sources;
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
// The markup an element holds that the HTML parser keeps as text but a
|
|
681
|
+
// visitor can load, or null: a <noscript>'s text (parse5 parses with
|
|
682
|
+
// scripting on, so its content is one text node) and an <iframe>'s srcdoc.
|
|
683
|
+
// `at` names it in the element paths of what it holds.
|
|
684
|
+
function nestedMarkupOf(node, tag, attrs) {
|
|
685
|
+
if (node.namespaceURI !== HTML_NAMESPACE) return null;
|
|
686
|
+
if (tag === "noscript") return { markup: textOf(node), at: "" };
|
|
687
|
+
if (tag === "iframe" && attrs.has("srcdoc")) return { markup: attrs.get("srcdoc"), at: "@srcdoc/" };
|
|
688
|
+
return null;
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
// One parsed page's raw observation. Candidates past the cap are not
|
|
692
|
+
// examined; ids, names and the page's loaded scripts are read from the whole
|
|
693
|
+
// tree. A nested document (nestedMarkupOf) is read only for the asset-host
|
|
694
|
+
// and loopback rules, its element paths under its host element's.
|
|
695
|
+
function observeDocument(document) {
|
|
696
|
+
const cap = new CandidateCap(SMOKE_QC_LIMITS.candidates, SMOKE_QC_LIMITS.page_bytes);
|
|
697
|
+
const observed = {
|
|
698
|
+
liveTargets: new Set(),
|
|
699
|
+
templateTargets: new Set(),
|
|
700
|
+
anchors: new Map(),
|
|
701
|
+
favicon: [],
|
|
702
|
+
og: { title: null, description: null, image: null },
|
|
703
|
+
scripts: [],
|
|
704
|
+
tailwind: [],
|
|
705
|
+
assetHost: [],
|
|
706
|
+
loopback: [],
|
|
707
|
+
cap,
|
|
708
|
+
};
|
|
709
|
+
const refs = new Map([[observed.assetHost, new Set()], [observed.loopback, new Set()]]);
|
|
710
|
+
const addRef = (list, ref) => {
|
|
711
|
+
const id = `${ref.element_path}\u0000${ref.attr}`;
|
|
712
|
+
if (refs.get(list).has(id)) return;
|
|
713
|
+
refs.get(list).add(id);
|
|
714
|
+
list.push(ref);
|
|
715
|
+
};
|
|
716
|
+
const recordHosts = (hosts, ref) => {
|
|
717
|
+
if (hosts.includes(PRIMARY_ASSET_HOST)) addRef(observed.assetHost, ref);
|
|
718
|
+
if (hosts.some((host) => isLoopbackHostname(host))) addRef(observed.loopback, ref);
|
|
719
|
+
};
|
|
720
|
+
|
|
721
|
+
// The page's own elements: every rule.
|
|
722
|
+
const visitPage = ({ node, tag, path, inTemplate }, attrs) => {
|
|
723
|
+
const html = node.namespaceURI === HTML_NAMESPACE;
|
|
724
|
+
const targets = inTemplate ? observed.templateTargets : observed.liveTargets;
|
|
725
|
+
if (attrs.get("id")) targets.add(attrs.get("id"));
|
|
726
|
+
if (html && tag === "a" && attrs.get("name")) targets.add(attrs.get("name"));
|
|
727
|
+
|
|
728
|
+
// An in-page anchor is one candidate, its href included.
|
|
729
|
+
let anchorHref = false;
|
|
730
|
+
if (!inTemplate) {
|
|
731
|
+
const href = (tag === "a" || tag === "area") && typeof attrs.get("href") === "string" ? attrs.get("href").trim() : null;
|
|
732
|
+
anchorHref = href != null && href.startsWith("#");
|
|
733
|
+
if (anchorHref && cap.take()) {
|
|
734
|
+
const target = decodeFragment(href.slice(1));
|
|
735
|
+
if (!observed.anchors.has(target)) observed.anchors.set(target, { target, element_paths: [] });
|
|
736
|
+
observed.anchors.get(target).element_paths.push(path);
|
|
737
|
+
}
|
|
738
|
+
}
|
|
739
|
+
if (html && tag === "link" && cap.take() && !inTemplate) {
|
|
740
|
+
const rel = String(attrs.get("rel") || "").toLowerCase().split(ASCII_WHITESPACE);
|
|
741
|
+
if (rel.includes("icon") || rel.includes("apple-touch-icon")) observed.favicon.push(path);
|
|
742
|
+
}
|
|
743
|
+
if (html && tag === "meta" && cap.take() && !inTemplate) {
|
|
744
|
+
const content = String(attrs.get("content") || "").trim();
|
|
745
|
+
for (const name of ["property", "name"]) {
|
|
746
|
+
const field = OG_FIELDS.get(String(attrs.get(name) || "").trim().toLowerCase());
|
|
747
|
+
if (field && content && !observed.og[field]) observed.og[field] = { element_path: path, content };
|
|
748
|
+
}
|
|
749
|
+
}
|
|
750
|
+
const scriptSrc = loadedScriptSrc(node, tag, inTemplate);
|
|
751
|
+
if (scriptSrc != null) observed.scripts.push(scriptSrc);
|
|
752
|
+
readUrls(node, tag, path, { skipHref: anchorHref, tailwind: !inTemplate && html && tag === "script" });
|
|
753
|
+
};
|
|
754
|
+
|
|
755
|
+
// Every URL-bearing attribute value is a candidate, relative or absolute;
|
|
756
|
+
// only the absolute ones are matched by host.
|
|
757
|
+
const readUrls = (node, tag, path, { skipHref = false, tailwind = false } = {}) => {
|
|
758
|
+
for (const attr of node.attrs || []) {
|
|
759
|
+
const name = attrName(attr);
|
|
760
|
+
if (skipHref && name === "href") continue;
|
|
761
|
+
const hosts = hostsOf(attributeCandidates(name, attr.value));
|
|
762
|
+
if (!hosts.length && !(URL_ATTRIBUTES.has(name) && attr.value.trim())) continue;
|
|
763
|
+
if (!cap.take() || !hosts.length) continue;
|
|
764
|
+
recordHosts(hosts, { element_path: path, attr: name });
|
|
765
|
+
if (tailwind && name === "src" && hostOf(attr.value) === TAILWIND_CDN_HOST) observed.tailwind.push(path);
|
|
766
|
+
}
|
|
767
|
+
if (tag === "style") {
|
|
768
|
+
const hosts = hostsOf(cssUrlValues(textOf(node)));
|
|
769
|
+
if (hosts.length && cap.take()) recordHosts(hosts, { element_path: path, attr: "style" });
|
|
770
|
+
}
|
|
771
|
+
};
|
|
772
|
+
|
|
773
|
+
// A page or nested document; a nested one is read for its URLs only.
|
|
774
|
+
const walk = (tree, nested, start) => walkElements(tree, (frame) => {
|
|
775
|
+
const attrs = attrsOf(frame.node);
|
|
776
|
+
if (nested) readUrls(frame.node, frame.tag, frame.path);
|
|
777
|
+
else visitPage(frame, attrs);
|
|
778
|
+
const inner = nestedMarkupOf(frame.node, frame.tag, attrs);
|
|
779
|
+
// Markup with no `<` holds no element, so it is not parsed or counted.
|
|
780
|
+
if (!inner || !inner.markup.includes("<")) return;
|
|
781
|
+
if (!cap.takeMarkup(Buffer.byteLength(inner.markup, "utf8"))) return;
|
|
782
|
+
walk(parseBuiltPage(inner.markup), true, { path: `${frame.path}/${inner.at}`.replace(/\/$/, ""), depth: frame.depth });
|
|
783
|
+
}, start);
|
|
784
|
+
walk(document, false, {});
|
|
785
|
+
return observed;
|
|
786
|
+
}
|
|
787
|
+
|
|
788
|
+
// Whether `path` lies strictly inside `root` (both already real paths when
|
|
789
|
+
// the caller compares real paths).
|
|
790
|
+
export const insideRoot = (root, path) => {
|
|
791
|
+
const rel = relative(root, path);
|
|
792
|
+
return rel !== "" && !rel.startsWith(`..${sep}`) && rel !== ".." && !isAbsolute(rel);
|
|
793
|
+
};
|
|
794
|
+
|
|
795
|
+
// A path's real path (every symlink followed), or null when it cannot be
|
|
796
|
+
// resolved. Only a file-system read failure reads null; any other error is a
|
|
797
|
+
// defect and throws.
|
|
798
|
+
export function realPathOf(path) {
|
|
799
|
+
if (!path) return null;
|
|
800
|
+
try {
|
|
801
|
+
return realpathSync(path);
|
|
802
|
+
} catch (error) {
|
|
803
|
+
if (!isFileReadFailure(error)) throw error;
|
|
804
|
+
return null;
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
|
|
808
|
+
// The page's local scripts for the anchor hint: `contents` holds what was
|
|
809
|
+
// read, `complete` whether every local script the page loads was. The list is
|
|
810
|
+
// the caller's (`page.scripts`, or `readScripts` given the srcs the page
|
|
811
|
+
// loads); without one, every local <script src> on the page counts as loaded
|
|
812
|
+
// and unread.
|
|
813
|
+
function pageScripts(page, observed, builtPath, ctx) {
|
|
814
|
+
const listed = Array.isArray(page.scripts)
|
|
815
|
+
? page.scripts
|
|
816
|
+
: ctx.readScripts
|
|
817
|
+
? ctx.readScripts(observed.scripts, builtPath)
|
|
818
|
+
: observed.scripts.filter((src) => builtFileOf(src, builtPath, ctx.siteRoot) != null).map((src) => ({ src, unread: "not_listed" }));
|
|
819
|
+
const contents = listed.filter((script) => typeof script?.content === "string").map((script) => script.content);
|
|
820
|
+
return { contents, loaded: listed.length, read: contents.length, complete: contents.length === listed.length };
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
// Whether a local asset is a file inside the site root, its real path (every
|
|
824
|
+
// symlink followed) compared with the site root's own. A path that cannot be
|
|
825
|
+
// resolved or stat-ed reads as no file.
|
|
826
|
+
function fileExists(realSiteRoot, path) {
|
|
827
|
+
if (!realSiteRoot || path == null) return false;
|
|
828
|
+
const real = realPathOf(path);
|
|
829
|
+
if (!real || !insideRoot(realSiteRoot, real)) return false;
|
|
830
|
+
try {
|
|
831
|
+
return statSync(real).isFile();
|
|
832
|
+
} catch (error) {
|
|
833
|
+
if (!isFileReadFailure(error)) throw error;
|
|
834
|
+
return false;
|
|
835
|
+
}
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
// Where the page's og:image points: { image, image_target }. Absolute means
|
|
839
|
+
// a scheme and a host; a scheme-relative one (`//host/x`, or `\\host/x`) is
|
|
840
|
+
// not absolute, whatever host it names (readReference). Every file it names
|
|
841
|
+
// comes from builtFileOf; a reference whose path names no file is never a
|
|
842
|
+
// file in _site/, so it reads missing.
|
|
843
|
+
function resolveOgImage(content, builtPath, ctx) {
|
|
844
|
+
const value = content.trim();
|
|
845
|
+
const { kind, url } = readReference(value);
|
|
846
|
+
if (kind === "absolute") {
|
|
847
|
+
if (!WEB_PROTOCOLS.has(url.protocol)) return { image: "absolute_remote", image_target: url.protocol };
|
|
848
|
+
const target = `${url.protocol}//${url.host}${url.pathname}`;
|
|
849
|
+
if (!ctx.bases.length) return { image: "absolute_same_base_unmapped", image_target: target };
|
|
850
|
+
const mapped = builtFileOf(value, builtPath, ctx.siteRoot, { bases: ctx.bases });
|
|
851
|
+
if (!mapped) return { image: "absolute_remote", image_target: target };
|
|
852
|
+
return { image: fileExists(ctx.realSiteRoot, mapped.path ?? null) ? "absolute_same_base_present" : "absolute_same_base_missing", image_target: target };
|
|
853
|
+
}
|
|
854
|
+
if (kind === "scheme_relative" && url) return resolveSchemeRelativeOgImage(value, url, builtPath, ctx);
|
|
855
|
+
const mapped = builtFileOf(value, builtPath, ctx.siteRoot);
|
|
856
|
+
const path = mapped?.path ?? null;
|
|
857
|
+
const target = path != null ? `/${relative(ctx.siteRoot, path).split(sep).join("/")}` : mapped?.site_path ?? null;
|
|
858
|
+
return { image: fileExists(ctx.realSiteRoot, path) ? "relative_present" : "relative_missing", image_target: target };
|
|
859
|
+
}
|
|
860
|
+
|
|
861
|
+
// A scheme-relative og:image, `url` as the parser reads it: relative, so never
|
|
862
|
+
// a pass. Under a deploy base (its host and path) its path maps into _site/
|
|
863
|
+
// like an absolute same-base URL; anywhere else, or with no known base, its
|
|
864
|
+
// file is not looked for and it reads relative_present (the observation enum
|
|
865
|
+
// is closed).
|
|
866
|
+
function resolveSchemeRelativeOgImage(value, url, builtPath, ctx) {
|
|
867
|
+
const target = `//${url.host}${url.pathname}`;
|
|
868
|
+
const mapped = builtFileOf(value, builtPath, ctx.siteRoot, { bases: ctx.bases });
|
|
869
|
+
if (!mapped) return { image: "relative_present", image_target: target };
|
|
870
|
+
return { image: fileExists(ctx.realSiteRoot, mapped.path ?? null) ? "relative_present" : "relative_missing", image_target: target };
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
const OG_IMAGE_RESULT = Object.freeze({
|
|
874
|
+
relative_missing: ["warning", R.OG_IMAGE_MISSING_FILE],
|
|
875
|
+
relative_present: ["warning", R.OG_IMAGE_NOT_ABSOLUTE],
|
|
876
|
+
absolute_same_base_present: ["pass", null],
|
|
877
|
+
absolute_same_base_missing: ["warning", R.OG_IMAGE_MISSING_FILE],
|
|
878
|
+
absolute_same_base_unmapped: ["unexercised", R.OG_IMAGE_BASE_UNKNOWN],
|
|
879
|
+
absolute_remote: ["unexercised", R.OG_IMAGE_REMOTE_NOT_FETCHED],
|
|
880
|
+
});
|
|
881
|
+
|
|
882
|
+
// One target's outcome: [result, reason_code, outcome]. Only the decoded
|
|
883
|
+
// target is matched. A target not on the page is unexercised while any local
|
|
884
|
+
// script the page loads went unread, whatever the scripts read contain.
|
|
885
|
+
function anchorOutcome(anchor, observed, scripts) {
|
|
886
|
+
const { target } = anchor;
|
|
887
|
+
if (target === "" || observed.liveTargets.has(target) || target.toLowerCase() === "top") return ["pass", null, "resolved"];
|
|
888
|
+
if (observed.templateTargets.has(target)) return ["review", R.ANCHOR_TARGET_IN_TEMPLATE, "in_template"];
|
|
889
|
+
if (!scripts.complete) return ["unexercised", R.SCRIPT_UNREADABLE, "script_unreadable"];
|
|
890
|
+
if (scripts.contents.some((content) => content.includes(target))) return ["review", R.ANCHOR_TARGET_POSSIBLY_SCRIPT_CREATED, "in_script"];
|
|
891
|
+
return ["warning", R.ANCHOR_TARGET_MISSING, "missing"];
|
|
892
|
+
}
|
|
893
|
+
|
|
894
|
+
const environmentResult = (environment, found, reasonCode) => {
|
|
895
|
+
if (environment !== "production") return ["unexercised", ENVIRONMENT_REASON[environment]];
|
|
896
|
+
return found ? ["warning", reasonCode] : ["pass", null];
|
|
897
|
+
};
|
|
898
|
+
|
|
899
|
+
// The page's results before any cap: { anchors: [...], fixed: [...] }, each
|
|
900
|
+
// { key, result, reason_code, state, observation }, and `absence: true` on a
|
|
901
|
+
// result that rests on something not being on the page.
|
|
902
|
+
function pageFindings(page, file, observed, builtPath, ctx) {
|
|
903
|
+
const scripts = pageScripts(page, observed, builtPath, ctx);
|
|
904
|
+
const anchors = [...observed.anchors.values()].map((anchor) => {
|
|
905
|
+
const [result, reasonCode, outcome] = anchorOutcome(anchor, observed, scripts);
|
|
906
|
+
const read = outcome === "resolved" || outcome === "in_template" ? null : scripts;
|
|
907
|
+
return {
|
|
908
|
+
key: anchorKey(anchor.target),
|
|
909
|
+
result,
|
|
910
|
+
reason_code: reasonCode,
|
|
911
|
+
state: { element_paths: anchor.element_paths },
|
|
912
|
+
observation: { page: file, target: anchor.target, outcome, scripts_read: read ? read.read : null, scripts_loaded: read ? read.loaded : null },
|
|
913
|
+
};
|
|
914
|
+
});
|
|
915
|
+
|
|
916
|
+
const { environment } = ctx;
|
|
917
|
+
const og = observed.og;
|
|
918
|
+
const ogImage = og.image ? resolveOgImage(og.image.content, builtPath, ctx) : { image: "absent", image_target: null };
|
|
919
|
+
const ogObservation = { title: Boolean(og.title), description: Boolean(og.description), image: ogImage.image, image_target: ogImage.image_target };
|
|
920
|
+
const presence = (key, field, reasonCode) => ({
|
|
921
|
+
key,
|
|
922
|
+
result: og[field] ? "pass" : "warning",
|
|
923
|
+
reason_code: og[field] ? null : reasonCode,
|
|
924
|
+
state: { element_paths: og[field] ? [og[field].element_path] : [] },
|
|
925
|
+
observation: { page: file, og: ogObservation },
|
|
926
|
+
absence: true,
|
|
927
|
+
});
|
|
928
|
+
const refsState = (refs) => ({ element_paths: refs.map((ref) => ref.element_path), attributes: refs.map((ref) => ref.attr) });
|
|
929
|
+
const [tailwindResult, tailwindReason] = environmentResult(environment, observed.tailwind.length > 0, R.TAILWIND_CDN_IN_PRODUCTION);
|
|
930
|
+
const [loopbackResult, loopbackReason] = environmentResult(environment, observed.loopback.length > 0, R.LOOPBACK_URL);
|
|
931
|
+
const fixed = [
|
|
932
|
+
{
|
|
933
|
+
key: K.favicon,
|
|
934
|
+
result: observed.favicon.length ? "pass" : "warning",
|
|
935
|
+
reason_code: observed.favicon.length ? null : R.FAVICON_MISSING,
|
|
936
|
+
state: { element_paths: observed.favicon },
|
|
937
|
+
observation: { page: file, favicon: observed.favicon.length > 0 },
|
|
938
|
+
absence: true,
|
|
939
|
+
},
|
|
940
|
+
presence(K.og_title, "title", R.OG_TITLE_MISSING),
|
|
941
|
+
presence(K.og_description, "description", R.OG_DESCRIPTION_MISSING),
|
|
942
|
+
presence(K.og_image, "image", R.OG_IMAGE_MISSING),
|
|
943
|
+
...(og.image ? [{
|
|
944
|
+
key: K.og_image_target,
|
|
945
|
+
result: OG_IMAGE_RESULT[ogImage.image][0],
|
|
946
|
+
reason_code: OG_IMAGE_RESULT[ogImage.image][1],
|
|
947
|
+
state: { element_paths: [og.image.element_path], image_target: ogImage.image_target },
|
|
948
|
+
observation: { page: file, og: ogObservation },
|
|
949
|
+
}] : []),
|
|
950
|
+
{
|
|
951
|
+
key: K.tailwind_cdn,
|
|
952
|
+
result: tailwindResult,
|
|
953
|
+
reason_code: tailwindReason,
|
|
954
|
+
state: { element_paths: observed.tailwind, environment },
|
|
955
|
+
observation: { page: file, tailwind_cdn: observed.tailwind.length, environment },
|
|
956
|
+
},
|
|
957
|
+
{
|
|
958
|
+
key: K.asset_host,
|
|
959
|
+
result: observed.assetHost.length ? "warning" : "pass",
|
|
960
|
+
reason_code: observed.assetHost.length ? R.PRIMARY_ASSET_HOST : null,
|
|
961
|
+
state: refsState(observed.assetHost),
|
|
962
|
+
observation: { page: file, asset_host_refs: observed.assetHost },
|
|
963
|
+
},
|
|
964
|
+
{
|
|
965
|
+
key: K.loopback,
|
|
966
|
+
result: loopbackResult,
|
|
967
|
+
reason_code: loopbackReason,
|
|
968
|
+
state: { ...refsState(observed.loopback), environment },
|
|
969
|
+
observation: { page: file, loopback_refs: observed.loopback, environment },
|
|
970
|
+
},
|
|
971
|
+
];
|
|
972
|
+
return { anchors, fixed };
|
|
973
|
+
}
|
|
974
|
+
|
|
975
|
+
const PAGE_LEVEL_KEYS = Object.freeze(Object.values(K));
|
|
976
|
+
const isFinding = (finding) => finding.result === "warning" || finding.result === "review";
|
|
977
|
+
// What a capped page keeps: a warning or review observed in full, never one
|
|
978
|
+
// that rests on something being absent from a page not seen whole.
|
|
979
|
+
const keptOnCappedPage = (finding) => isFinding(finding) && !finding.absence;
|
|
980
|
+
const capMembersFor = (reasons) => reasons.flatMap((reason) => aggregateQcResults([], { capReason: reason }).members);
|
|
981
|
+
|
|
982
|
+
/**
|
|
983
|
+
* Evaluate the built-output smoke checks.
|
|
984
|
+
*
|
|
985
|
+
* @param {{
|
|
986
|
+
* pages: Array<{ file: string, content?: string, bytes?: number, unreadable?: boolean, scripts?: Array<{ src: string, file?: string, content?: string, unread?: string }> }>,
|
|
987
|
+
* environment?: string|null,
|
|
988
|
+
* siteRoot: string|null,
|
|
989
|
+
* targetDir?: string|null,
|
|
990
|
+
* readScripts?: ((srcs: string[], builtPath: string) => Array<{ src: string, file?: string, content?: string, unread?: string }>)|null,
|
|
991
|
+
* deployBase?: string|string[]|null,
|
|
992
|
+
* measuredAt?: string,
|
|
993
|
+
* }} input `pages` in a stable order, `file` relative to the doctor target
|
|
994
|
+
* ("_site/<slug>/index.html"); pages past the page cap may omit `content`.
|
|
995
|
+
* `scripts` lists every local script the page loads, read or not (see the
|
|
996
|
+
* header); without it `readScripts`, when given, lists them from the srcs
|
|
997
|
+
* the page's parse found, and otherwise the page's local scripts read
|
|
998
|
+
* unread. `environment` is the recorded build environment ("production" or
|
|
999
|
+
* "development"; anything else is unknown). `siteRoot` is the built site
|
|
1000
|
+
* root (the scope's `_site/`, or the directory doctor was pointed at when
|
|
1001
|
+
* it has none): every local reference maps onto it through builtFileOf
|
|
1002
|
+
* (none does without it), and a local asset whose real path lies outside it
|
|
1003
|
+
* reads as missing. Page files resolve against `targetDir`, the doctor
|
|
1004
|
+
* target (by default the site root's parent). `deployBase` lists the deploy
|
|
1005
|
+
* URLs under which an absolute og:image maps into the site root (each
|
|
1006
|
+
* one's origin and path); none means the base is unknown. Every result
|
|
1007
|
+
* carries its own {check, page, key} subject.
|
|
1008
|
+
* @returns {object[]} QC results (src/qc-results.mjs buildQcResult).
|
|
1009
|
+
*/
|
|
1010
|
+
export function evaluateSmokeQc({ pages = [], environment = null, siteRoot = null, targetDir = null, readScripts = null, deployBase = null, measuredAt = new Date().toISOString() } = {}) {
|
|
1011
|
+
const check = SMOKE_QC_CHECK;
|
|
1012
|
+
const ctx = {
|
|
1013
|
+
environment: environmentOf(environment),
|
|
1014
|
+
siteRoot,
|
|
1015
|
+
bases: deployBases(deployBase),
|
|
1016
|
+
realSiteRoot: realPathOf(siteRoot),
|
|
1017
|
+
readScripts,
|
|
1018
|
+
};
|
|
1019
|
+
const pageDir = targetDir ?? (siteRoot ? dirname(siteRoot) : null);
|
|
1020
|
+
const results = [];
|
|
1021
|
+
const row = (page, { key, result, reason_code = null, state = {}, observation = {} }, { members = [], coverage } = {}) => buildQcResult({
|
|
1022
|
+
check,
|
|
1023
|
+
leg: "doctor",
|
|
1024
|
+
subject: { check, page, key },
|
|
1025
|
+
result,
|
|
1026
|
+
reason_code,
|
|
1027
|
+
state: { reason_code: result === "pass" ? null : reason_code, ...state, members },
|
|
1028
|
+
observation,
|
|
1029
|
+
members,
|
|
1030
|
+
coverage: coverage ?? (result === "unexercised" ? { observed: 0, expected: 1, limits: [reason_code] } : { observed: 1, expected: 1, limits: [] }),
|
|
1031
|
+
measured_at: measuredAt,
|
|
1032
|
+
});
|
|
1033
|
+
const pageLevel = (file, reasonCode, extra = {}) => {
|
|
1034
|
+
for (const key of PAGE_LEVEL_KEYS) {
|
|
1035
|
+
results.push(row(file, { key, result: "unexercised", reason_code: reasonCode, observation: { page: file, ...extra } }));
|
|
1036
|
+
}
|
|
1037
|
+
};
|
|
1038
|
+
|
|
1039
|
+
(Array.isArray(pages) ? pages : []).forEach((page, index) => {
|
|
1040
|
+
const file = String(page?.file ?? "");
|
|
1041
|
+
if (index >= SMOKE_QC_LIMITS.pages) {
|
|
1042
|
+
pageLevel(file, R.PAGE_CAP_REACHED);
|
|
1043
|
+
return;
|
|
1044
|
+
}
|
|
1045
|
+
const read = readPage(page || {});
|
|
1046
|
+
if (read.outcome) {
|
|
1047
|
+
pageLevel(file, read.outcome, read.bytes == null ? {} : { bytes: read.bytes });
|
|
1048
|
+
return;
|
|
1049
|
+
}
|
|
1050
|
+
let findings;
|
|
1051
|
+
let observed;
|
|
1052
|
+
try {
|
|
1053
|
+
observed = observeDocument(read.document);
|
|
1054
|
+
findings = pageFindings(page, file, observed, pageDir ? join(pageDir, file) : file, ctx);
|
|
1055
|
+
} catch (error) {
|
|
1056
|
+
if (!isBuiltPageReadFailure(error)) throw error;
|
|
1057
|
+
pageLevel(file, R.PAGE_UNREADABLE, { bytes: read.bytes });
|
|
1058
|
+
return;
|
|
1059
|
+
}
|
|
1060
|
+
|
|
1061
|
+
const caps = [];
|
|
1062
|
+
if (observed.cap.reached) caps.push(R.CANDIDATE_CAP_REACHED);
|
|
1063
|
+
// The result cap counts the anchor rule's results, pass included (the
|
|
1064
|
+
// only rule with more than one result per page).
|
|
1065
|
+
if (findings.anchors.length > SMOKE_QC_LIMITS.results) caps.push(R.FINDING_CAP_REACHED);
|
|
1066
|
+
if (!caps.length) {
|
|
1067
|
+
for (const finding of [...findings.anchors, ...findings.fixed]) results.push(row(file, finding));
|
|
1068
|
+
return;
|
|
1069
|
+
}
|
|
1070
|
+
|
|
1071
|
+
// A capped page: the warning and review results observed in full stay,
|
|
1072
|
+
// each with the capped-page member; every other rule key reads
|
|
1073
|
+
// unexercised for the cap, and nothing passes.
|
|
1074
|
+
const members = capMembersFor(caps);
|
|
1075
|
+
const capped = (finding) => row(file, finding, { members, coverage: { observed: finding.result === "unexercised" ? 0 : 1, expected: null, limits: [...caps] } });
|
|
1076
|
+
const capRow = (key) => capped({ key, result: "unexercised", reason_code: caps[0], observation: { page: file, candidates: observed.cap.count } });
|
|
1077
|
+
for (const finding of findings.anchors.slice(0, SMOKE_QC_LIMITS.results)) {
|
|
1078
|
+
if (keptOnCappedPage(finding)) results.push(capped(finding));
|
|
1079
|
+
}
|
|
1080
|
+
results.push(capRow(K.anchor));
|
|
1081
|
+
for (const key of PAGE_LEVEL_KEYS.filter((value) => value !== K.anchor)) {
|
|
1082
|
+
const finding = findings.fixed.find((item) => item.key === key);
|
|
1083
|
+
results.push(finding && keptOnCappedPage(finding) ? capped(finding) : capRow(key));
|
|
1084
|
+
}
|
|
1085
|
+
});
|
|
1086
|
+
return results;
|
|
1087
|
+
}
|
|
1088
|
+
|
|
1089
|
+
// One page's parse5 tree (parseBuiltPage), or a page-level outcome. A read or
|
|
1090
|
+
// parse failure, nesting past MAX_ELEMENT_DEPTH included, reads the page
|
|
1091
|
+
// unreadable; any other error is a defect and throws.
|
|
1092
|
+
function readPage(page) {
|
|
1093
|
+
if (page.unreadable) return { outcome: R.PAGE_UNREADABLE };
|
|
1094
|
+
const bytes = Number.isFinite(page.bytes) ? page.bytes : typeof page.content === "string" ? Buffer.byteLength(page.content, "utf8") : null;
|
|
1095
|
+
if (bytes != null && bytes > SMOKE_QC_LIMITS.page_bytes) return { outcome: R.PAGE_TOO_LARGE, bytes };
|
|
1096
|
+
if (typeof page.content !== "string") return { outcome: R.PAGE_UNREADABLE };
|
|
1097
|
+
try {
|
|
1098
|
+
return { document: parseBuiltPage(page.content), bytes };
|
|
1099
|
+
} catch (error) {
|
|
1100
|
+
if (!isBuiltPageReadFailure(error)) throw error;
|
|
1101
|
+
return { outcome: R.PAGE_UNREADABLE, bytes };
|
|
1102
|
+
}
|
|
1103
|
+
}
|
|
1104
|
+
|
|
1105
|
+
// Each deploy URL as a base an og:image maps through: its origin, its host
|
|
1106
|
+
// and its path as a directory (`https://deploy.example/s` and `/s/` both
|
|
1107
|
+
// read `/s/`), so only a path under it maps into the site root.
|
|
1108
|
+
function deployBases(deployBase) {
|
|
1109
|
+
const bases = [];
|
|
1110
|
+
for (const value of [deployBase].flat()) {
|
|
1111
|
+
if (typeof value !== "string" || !value.trim()) continue;
|
|
1112
|
+
const url = parseUrl(value.trim());
|
|
1113
|
+
if (!url || !WEB_PROTOCOLS.has(url.protocol)) continue;
|
|
1114
|
+
bases.push({ origin: url.origin, host: url.host, prefix: url.pathname.endsWith("/") ? url.pathname : `${url.pathname}/` });
|
|
1115
|
+
}
|
|
1116
|
+
return bases;
|
|
1117
|
+
}
|