@nextcommerce/campaigns-os 1.50.0 → 1.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/CHANGELOG.md +426 -0
  2. package/agents/claude/CLAUDE.md +2 -2
  3. package/agents/codex/AGENTS.md +1 -1
  4. package/agents/copilot/copilot-instructions.md +1 -1
  5. package/agents/cursor/campaigns-os.mdc +1 -1
  6. package/campaign-spec/dist/rules/campaign-metadata.d.ts +5 -1
  7. package/campaign-spec/dist/rules/campaign-metadata.js +9 -2
  8. package/campaign-spec/dist/rules/design-source-shape.js +13 -3
  9. package/campaign-spec/dist/rules/sdk-version.js +2 -1
  10. package/compatibility.json +1 -1
  11. package/contracts/commerce-surface-catalog.json +26 -46
  12. package/contracts/effects.v1.json +81 -2
  13. package/contracts/release-ledger.json +906 -0
  14. package/contracts/supported-surface.json +2 -2
  15. package/contracts/template-brand-contract.shared-commerce.v0.json +2 -2
  16. package/contracts/template-slot-manifest.shared-content-core.v0.json +24 -0
  17. package/docs/build-packet.md +93 -9
  18. package/docs/campaign-build-brief.md +25 -1
  19. package/docs/effects.md +6 -0
  20. package/docs/local-setup.md +1 -1
  21. package/docs/orientation-contract-reference.md +1 -1
  22. package/docs/polish-evidence.md +10 -0
  23. package/docs/qa-and-test-orders.md +45 -4
  24. package/docs/runtime-readiness.md +1 -1
  25. package/docs/sdk-storage-compatibility.md +1 -1
  26. package/docs/skills-revision.md +10 -10
  27. package/package.json +1 -1
  28. package/skills/campaign-lifecycle-orientation/SKILL.md +3 -3
  29. package/skills/campaign-readback-classification/SKILL.md +3 -3
  30. package/skills/campaign-run-evidence/SKILL.md +3 -3
  31. package/skills/contribution-intake/SKILL.md +3 -3
  32. package/skills/next-campaigns-build/SKILL.md +3 -3
  33. package/skills/next-campaigns-os/SKILL.md +4 -4
  34. package/skills/next-campaigns-os/references/session-intake.md +7 -3
  35. package/skills/next-campaigns-os-setup/SKILL.md +3 -3
  36. package/skills/next-campaigns-polish/SKILL.md +3 -3
  37. package/skills/next-campaigns-qa/SKILL.md +6 -5
  38. package/skills.json +10 -10
  39. package/src/adapter-decision-contract.mjs +1 -1
  40. package/src/brand-theme.mjs +12 -0
  41. package/src/build-brief.mjs +68 -21
  42. package/src/built-site-scope.mjs +39 -6
  43. package/src/built-smoke-qc.mjs +1117 -0
  44. package/src/campaign-identity.mjs +36 -2
  45. package/src/cart-placeholders.mjs +730 -0
  46. package/src/cli.mjs +310 -34
  47. package/src/commercial-journey.mjs +65 -4
  48. package/src/commercial-parity.mjs +6 -1
  49. package/src/doctor/checks.mjs +291 -24
  50. package/src/doctor/inspect.mjs +53 -2
  51. package/src/doctor/next-step.mjs +1 -1
  52. package/src/invocation.mjs +2 -1
  53. package/src/local-preview-policy.mjs +1 -1
  54. package/src/local-proof.mjs +4 -1
  55. package/src/polish-browser.mjs +218 -1
  56. package/src/polish-capture.mjs +1 -1
  57. package/src/polish-media-weight.mjs +492 -0
  58. package/src/polish-node.mjs +96 -4
  59. package/src/progress-node.mjs +5 -1
  60. package/src/qa-browser.mjs +308 -96
  61. package/src/qa-content-params.mjs +889 -0
  62. package/src/qa-node.mjs +104 -12
  63. package/src/qa-order-bump.mjs +22 -1
  64. package/src/qa-policy-links.mjs +1019 -0
  65. package/src/qa-tracking-params.mjs +1389 -0
  66. package/src/qa-url-privacy.mjs +168 -0
  67. package/src/qc-accept.mjs +446 -0
  68. package/src/qc-check-registry.mjs +83 -0
  69. package/src/qc-results.mjs +1049 -0
  70. package/src/sdk-attribute-index.mjs +71 -0
  71. package/src/sdk-markup.mjs +2 -2
  72. package/src/sdk-storage-compatibility.mjs +63 -3
  73. package/src/source-prep.mjs +37 -7
  74. package/src/stage-record.mjs +56 -17
@@ -0,0 +1,1117 @@
1
+ // Built-output smoke checks (built_output.smoke_qc).
2
+ //
3
+ // Advisory observations on every built page, each a QC result and never a
4
+ // blocker. A sibling doctor gate, kept out of validateBuiltHtmlStructure
5
+ // (which escalates warnings to errors), so every finding stays a warning.
6
+ //
7
+ // Rules, one result per (page, rule, key); doctor issue code
8
+ // built_output.smoke_qc.<reason_code>:
9
+ // anchor:<target> an in-page link (<a> or <area> href="#x"; SVG <use> is
10
+ // not a link) whose target no element on the page carries
11
+ // as id (or <a name>): warning anchor_target_missing, once
12
+ // every local script the page loads was read; review
13
+ // anchor_target_in_template or
14
+ // anchor_target_possibly_script_created; unexercised
15
+ // script_unreadable when any local script the page loads
16
+ // was missing, outside _site/, over 1 MiB or past the
17
+ // 16-script cap, whatever the scripts read say. `#`,
18
+ // `#top` and the empty fragment are valid. The target is
19
+ // the fragment percent-decoded (UTF-8 bytes, a `%` without
20
+ // two hex digits kept as is) and only that is matched.
21
+ // favicon:link no link[rel~=icon] and no apple-touch-icon link:
22
+ // warning favicon_missing
23
+ // og:title, og:description, og:image
24
+ // warning og_*_missing when the tag is absent
25
+ // og:image_target the og:image destination, only when the page has one:
26
+ // a URL without both a scheme and a host (path-relative,
27
+ // root-relative or scheme-relative `//host/x`) is
28
+ // og_image_not_absolute, or og_image_missing_file when it
29
+ // maps into _site/ and the file is not there (a
30
+ // scheme-relative URL maps only through a deploy base
31
+ // host and path; otherwise its file is never looked for
32
+ // and it reads `relative_present`); an absolute URL under
33
+ // a deploy base (its origin, and a path inside the base's
34
+ // path) passes when its file is in _site/ and is
35
+ // og_image_missing_file otherwise; with no known deploy
36
+ // base it is unexercised og_image_base_unknown, and
37
+ // anywhere else og_image_remote_not_fetched (never
38
+ // fetched)
39
+ // tailwind_cdn:cdn.tailwindcss.com, loopback:loopback
40
+ // decided only for a recorded production build: warning
41
+ // tailwind_cdn_in_production / loopback_url, else pass; a
42
+ // development build reads unexercised development_render
43
+ // and an unknown one build_environment_unknown, on every
44
+ // page whatever it contains
45
+ // asset_host:cdn.29next.store
46
+ // any URL on the primary NEXT asset host, in any
47
+ // environment: warning primary_asset_host
48
+ // The asset-host and loopback rules read every attribute value (data-* and
49
+ // <meta content> included) and <style> text, matching absolute URLs by host.
50
+ // A host is read only from a whole value, never a substring: a URL_ATTRIBUTES
51
+ // value (srcset and imagesrcset split per HTML, ping per whitespace), any
52
+ // other attribute whose whole value is a URL, and the url() and string tokens
53
+ // of a style attribute or <style> text (cssUrlValues), so a URL inside
54
+ // another URL's query names no host. Every host, the Tailwind script's
55
+ // included, is the URL parser's hostname (hostOf). Whether a value is
56
+ // absolute or scheme-relative is read from its own text, never from where it
57
+ // resolves. Markup the HTML parser keeps as text but a visitor can load, the
58
+ // text of a <noscript> and the srcdoc of an <iframe>, is parsed as a document
59
+ // of its own and read by the same two rules. Text nodes and script bodies are
60
+ // not scanned. The per-page candidate cap counts each in-page anchor, each
61
+ // <meta> and <link> (in <template> content too), each URL_ATTRIBUTES value
62
+ // whether relative or absolute, any other attribute value holding an absolute
63
+ // URL, each <style> holding one, and each <noscript> or srcdoc holding markup
64
+ // (whose own candidates count on the page too); that markup together may not
65
+ // exceed the page size cap, and past it the page is capped as well.
66
+ //
67
+ // Every parse goes through the 1.5 gate's bounded parser (parseBounded), so
68
+ // nesting deeper than MAX_ELEMENT_DEPTH (a nested document's depth counting
69
+ // from its host element) stops the parse and reads page_unreadable.
70
+ //
71
+ // Page-level outcomes apply first to every rule key (the bare `anchor` key
72
+ // standing for the anchor rule): page_unreadable, page_too_large and
73
+ // page_cap_reached read unexercised on each key. A page past the candidate cap
74
+ // or with more than 50 anchor results (pass included) is capped: it keeps the
75
+ // warning and review results already observed in full (an anchor's target,
76
+ // read over every id on the page; the og:image destination; a Tailwind,
77
+ // asset-host or loopback URL found) and the first 50 anchor results' findings,
78
+ // and reads unexercised/<cap> on every other key. A finding that rests on
79
+ // something being absent (favicon, og presence) is never kept, nor is any
80
+ // pass. Every result it keeps carries the capped-page member, so none is
81
+ // accept-eligible.
82
+ //
83
+ // Only fragments, the og:image destination (origin and path, or site path),
84
+ // element paths, attribute names and counts are kept; no query string, script
85
+ // body or page text.
86
+ //
87
+ // Callers hand in built HTML and the local scripts each page loads: either
88
+ // `readScripts`, called with the srcs of the scripts the page's own parse
89
+ // lists (doctor's bounded script collector), or a `page.scripts` list
90
+ // ({src, file, content} when read, {src, file, unread} when not); a page with
91
+ // neither reads every local script it loads unread. Every URL reference that
92
+ // names a file under _site/ (the og:image file, each local script) maps
93
+ // through builtFileOf: the URL parser resolves it against the page's own URL
94
+ // under the site root, and each segment of the resulting path is
95
+ // percent-decoded onto `siteRoot`. A path that names no file (`%ZZ`, bytes
96
+ // that are not UTF-8, an empty segment, so a trailing `/`) is never a file.
97
+ // The og:image file is stat-ed here, once per og:image, and counts only when
98
+ // its real path (symlinks followed) lies inside `siteRoot`. No network
99
+ // request.
100
+
101
+ import { realpathSync, statSync } from "node:fs";
102
+ import { dirname, isAbsolute, join, relative, sep } from "node:path";
103
+
104
+ import { CART_PLACEHOLDERS_LIMITS, MAX_ELEMENT_DEPTH, isFileReadFailure, isPageReadFailure, parseBounded, scriptKind } from "./cart-placeholders.mjs";
105
+ import { aggregateQcResults, buildQcResult } from "./qc-results.mjs";
106
+ import { isLoopbackHostname } from "./remit.mjs";
107
+
108
+ export const SMOKE_QC = "built_output.smoke_qc";
109
+ export const SMOKE_QC_CHECK = "smoke_qc";
110
+
111
+ export const PRIMARY_ASSET_HOST = "cdn.29next.store";
112
+ export const TAILWIND_CDN_HOST = "cdn.tailwindcss.com";
113
+
114
+ // Shared doctor bounds, plus the anchor script hint's own.
115
+ export const SMOKE_QC_LIMITS = Object.freeze({
116
+ ...CART_PLACEHOLDERS_LIMITS,
117
+ scripts: 16,
118
+ script_bytes: 1024 * 1024,
119
+ });
120
+
121
+ export const SMOKE_QC_REASONS = Object.freeze({
122
+ ANCHOR_TARGET_MISSING: "anchor_target_missing",
123
+ ANCHOR_TARGET_IN_TEMPLATE: "anchor_target_in_template",
124
+ ANCHOR_TARGET_POSSIBLY_SCRIPT_CREATED: "anchor_target_possibly_script_created",
125
+ SCRIPT_UNREADABLE: "script_unreadable",
126
+ FAVICON_MISSING: "favicon_missing",
127
+ OG_TITLE_MISSING: "og_title_missing",
128
+ OG_DESCRIPTION_MISSING: "og_description_missing",
129
+ OG_IMAGE_MISSING: "og_image_missing",
130
+ OG_IMAGE_MISSING_FILE: "og_image_missing_file",
131
+ OG_IMAGE_NOT_ABSOLUTE: "og_image_not_absolute",
132
+ OG_IMAGE_BASE_UNKNOWN: "og_image_base_unknown",
133
+ OG_IMAGE_REMOTE_NOT_FETCHED: "og_image_remote_not_fetched",
134
+ TAILWIND_CDN_IN_PRODUCTION: "tailwind_cdn_in_production",
135
+ PRIMARY_ASSET_HOST: "primary_asset_host",
136
+ LOOPBACK_URL: "loopback_url",
137
+ DEVELOPMENT_RENDER: "development_render",
138
+ BUILD_ENVIRONMENT_UNKNOWN: "build_environment_unknown",
139
+ PAGE_UNREADABLE: "page_unreadable",
140
+ PAGE_TOO_LARGE: "page_too_large",
141
+ PAGE_CAP_REACHED: "page_cap_reached",
142
+ CANDIDATE_CAP_REACHED: "candidate_cap_reached",
143
+ FINDING_CAP_REACHED: "finding_cap_reached",
144
+ });
145
+ const R = SMOKE_QC_REASONS;
146
+
147
+ // The result key of every rule but the per-target anchors. `anchor` is the
148
+ // anchor rule's own row, used only when its targets are not known (a
149
+ // page-level outcome) and as the finding-cap row.
150
+ export const SMOKE_QC_KEYS = Object.freeze({
151
+ anchor: "anchor",
152
+ favicon: "favicon:link",
153
+ og_title: "og:title",
154
+ og_description: "og:description",
155
+ og_image: "og:image",
156
+ og_image_target: "og:image_target",
157
+ tailwind_cdn: `tailwind_cdn:${TAILWIND_CDN_HOST}`,
158
+ asset_host: `asset_host:${PRIMARY_ASSET_HOST}`,
159
+ loopback: "loopback:loopback",
160
+ });
161
+ const K = SMOKE_QC_KEYS;
162
+ const anchorKey = (target) => `anchor:${target}`;
163
+
164
+ const ENVIRONMENTS = new Set(["production", "development"]);
165
+ const ENVIRONMENT_REASON = Object.freeze({ development: R.DEVELOPMENT_RENDER, unknown: R.BUILD_ENVIRONMENT_UNKNOWN });
166
+ const environmentOf = (value) => (ENVIRONMENTS.has(value) ? value : "unknown");
167
+
168
+ const HTML_NAMESPACE = "http://www.w3.org/1999/xhtml";
169
+ const WEB_PROTOCOLS = new Set(["http:", "https:"]);
170
+ const OG_FIELDS = new Map([["og:title", "title"], ["og:description", "description"], ["og:image", "image"]]);
171
+ const ASCII_WHITESPACE = /[\t\n\f\r ]+/;
172
+
173
+ // The origin built pages are read from: the site root (_site/) is its path
174
+ // `/`, so a page's URL is its path under _site/ and every reference on it
175
+ // resolves as a browser resolves it there. Only a path reference resolves
176
+ // against it; no reference is ever judged by whether it lands on this origin,
177
+ // so an absolute URL that names it is just another remote URL.
178
+ const SITE_ORIGIN = "https://built.invalid";
179
+ const SITE_BASE = `${SITE_ORIGIN}/`;
180
+
181
+ const parseUrl = (value, base) => {
182
+ try {
183
+ return new URL(value, base);
184
+ } catch {
185
+ return null;
186
+ }
187
+ };
188
+
189
+ // What a reference is, read from its own text and never from where it
190
+ // resolves: `absolute` when the URL parser reads it with no base (it has a
191
+ // scheme), `scheme_relative` when it starts with two slashes or backslashes
192
+ // (after the leading C0 controls and spaces, and the tabs and newlines, the
193
+ // parser drops), else a `path`. `url` is the parsed URL of an absolute
194
+ // reference, or of a scheme-relative one resolved against `base`.
195
+ function readReference(value, base = SITE_BASE) {
196
+ const absolute = parseUrl(value);
197
+ if (absolute) return { kind: "absolute", url: absolute };
198
+ const text = String(value).replace(/^[\u0000- ]+/, "").replace(/[\t\n\r]/g, "");
199
+ if (/^[/\\]{2}/.test(text)) return { kind: "scheme_relative", url: parseUrl(value, base) };
200
+ return { kind: "path", url: null };
201
+ }
202
+
203
+ // The host a candidate names, read by the URL parser (userinfo, port, IPv6
204
+ // brackets, IDN, case and IPv4 forms are its own): an http(s) URL that is
205
+ // absolute or protocol-relative, its hostname without a trailing root dot.
206
+ // Anything else, a relative path included, names no host.
207
+ function hostOf(candidate) {
208
+ const { kind, url } = readReference(candidate);
209
+ if (kind === "path" || !url || !WEB_PROTOCOLS.has(url.protocol)) return null;
210
+ return url.hostname.replace(/\.$/, "") || null;
211
+ }
212
+
213
+ const isAsciiWhitespace = (c) => c === "\t" || c === "\n" || c === "\f" || c === "\r" || c === " ";
214
+ const trimAsciiWhitespace = (value) => value.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, "");
215
+
216
+ // The URLs of a srcset or imagesrcset value, split as the HTML "parse a srcset
217
+ // attribute" algorithm splits it: skip ASCII whitespace and commas, take the
218
+ // run up to the next ASCII whitespace as the URL (a comma inside it stays;
219
+ // trailing commas end the entry), then skip its descriptors up to a comma
220
+ // outside parentheses. Descriptors are not validated.
221
+ function srcsetUrls(value) {
222
+ const urls = [];
223
+ let i = 0;
224
+ for (;;) {
225
+ while (i < value.length && (isAsciiWhitespace(value[i]) || value[i] === ",")) i += 1;
226
+ if (i >= value.length) return urls;
227
+ const start = i;
228
+ while (i < value.length && !isAsciiWhitespace(value[i])) i += 1;
229
+ const url = value.slice(start, i);
230
+ if (url.endsWith(",")) {
231
+ urls.push(url.replace(/,+$/, ""));
232
+ continue;
233
+ }
234
+ urls.push(url);
235
+ let inParens = false;
236
+ while (i < value.length) {
237
+ const c = value[i];
238
+ i += 1;
239
+ if (c === "(") inParens = true;
240
+ else if (c === ")") inParens = false;
241
+ else if (c === "," && !inParens) break;
242
+ }
243
+ }
244
+ }
245
+
246
+ // CSS Syntax 3 code point classes (§4.2), on preprocessed input (§3.3).
247
+ const isCssWhitespace = (c) => c === "\n" || c === "\t" || c === " ";
248
+ const isDigit = (c) => c !== undefined && c >= "0" && c <= "9";
249
+ const isHexDigit = (c) => c !== undefined && /[0-9A-Fa-f]/.test(c);
250
+ const isIdentStart = (c) => c !== undefined && (/[A-Za-z_]/.test(c) || c.charCodeAt(0) >= 0x80);
251
+ const isIdentCodePoint = (c) => isIdentStart(c) || isDigit(c) || c === "-";
252
+ const isNonPrintable = (c) => c !== undefined && /[\u0000-\u0008\u000B\u000E-\u001F\u007F]/.test(c);
253
+ const isValidEscape = (a, b) => a === "\\" && b !== "\n";
254
+ const startsIdentSequence = (a, b, c) => {
255
+ if (a === "-") return isIdentStart(b) || b === "-" || isValidEscape(b, c);
256
+ return isIdentStart(a) || isValidEscape(a, b);
257
+ };
258
+ const startsNumber = (a, b, c) => {
259
+ if (a === "+" || a === "-") return isDigit(b) || (b === "." && isDigit(c));
260
+ if (a === ".") return isDigit(b);
261
+ return isDigit(a);
262
+ };
263
+
264
+ /**
265
+ * The value of every <url-token> and <string-token> in CSS text (a <style>
266
+ * element's text or a style attribute), tokenized per CSS Syntax 3 §4.3:
267
+ * escapes decoded (consume an escaped code point), `url(` recognized only as
268
+ * an ident-like token (so `u\72l(` is one, a `1url(` dimension is not), and
269
+ * `url( "…" )` read as a function whose string token follows. Comments,
270
+ * <bad-url-token>s and <bad-string-token>s give no value. Every other token is
271
+ * consumed only to find where the next one starts.
272
+ *
273
+ * @param {string} text
274
+ * @returns {string[]}
275
+ */
276
+ export function cssUrlValues(text) {
277
+ const s = String(text).replace(/\r\n?|\f/g, "\n").replace(/\u0000/g, "�");
278
+ const values = [];
279
+ let i = 0;
280
+
281
+ // §4.3.7, the backslash already consumed.
282
+ const escapedCodePoint = () => {
283
+ if (i >= s.length) return "�";
284
+ if (isHexDigit(s[i])) {
285
+ let hex = "";
286
+ while (hex.length < 6 && isHexDigit(s[i])) hex += s[i++];
287
+ if (isCssWhitespace(s[i])) i += 1;
288
+ const value = Number.parseInt(hex, 16);
289
+ return value === 0 || (value >= 0xd800 && value <= 0xdfff) || value > 0x10ffff ? "�" : String.fromCodePoint(value);
290
+ }
291
+ const char = String.fromCodePoint(s.codePointAt(i));
292
+ i += char.length;
293
+ return char;
294
+ };
295
+ // §4.3.12.
296
+ const identSequence = () => {
297
+ let out = "";
298
+ for (;;) {
299
+ if (isIdentCodePoint(s[i])) {
300
+ out += s[i];
301
+ i += 1;
302
+ } else if (isValidEscape(s[i], s[i + 1])) {
303
+ i += 1;
304
+ out += escapedCodePoint();
305
+ } else {
306
+ return out;
307
+ }
308
+ }
309
+ };
310
+ // §4.3.5, the opening quote already consumed; null for a <bad-string-token>.
311
+ const stringToken = (ending) => {
312
+ let out = "";
313
+ while (i < s.length) {
314
+ const c = s[i];
315
+ if (c === "\n") return null;
316
+ i += 1;
317
+ if (c === ending) return out;
318
+ if (c !== "\\") out += c;
319
+ else if (s[i] === "\n") i += 1;
320
+ else if (i < s.length) out += escapedCodePoint();
321
+ }
322
+ return out;
323
+ };
324
+ // §4.3.14.
325
+ const badUrlRemnants = () => {
326
+ while (i < s.length) {
327
+ const c = s[i];
328
+ i += 1;
329
+ if (c === ")") return;
330
+ if (isValidEscape(c, s[i])) escapedCodePoint();
331
+ }
332
+ };
333
+ // §4.3.6, `url(` already consumed; null for a <bad-url-token>.
334
+ const urlToken = () => {
335
+ while (isCssWhitespace(s[i])) i += 1;
336
+ let out = "";
337
+ while (i < s.length) {
338
+ const c = s[i];
339
+ i += 1;
340
+ if (c === ")") return out;
341
+ if (isCssWhitespace(c)) {
342
+ while (isCssWhitespace(s[i])) i += 1;
343
+ if (i >= s.length) return out;
344
+ if (s[i] === ")") {
345
+ i += 1;
346
+ return out;
347
+ }
348
+ badUrlRemnants();
349
+ return null;
350
+ }
351
+ if (c === "\"" || c === "'" || c === "(" || isNonPrintable(c) || (c === "\\" && !isValidEscape(c, s[i]))) {
352
+ badUrlRemnants();
353
+ return null;
354
+ }
355
+ out += c === "\\" ? escapedCodePoint() : c;
356
+ }
357
+ return out;
358
+ };
359
+ // §4.3.4.
360
+ const identLikeToken = () => {
361
+ const name = identSequence();
362
+ if (s[i] !== "(") return;
363
+ i += 1;
364
+ if (!/^url$/i.test(name)) return;
365
+ while (isCssWhitespace(s[i]) && isCssWhitespace(s[i + 1])) i += 1;
366
+ const next = isCssWhitespace(s[i]) ? s[i + 1] : s[i];
367
+ if (next === "\"" || next === "'") return;
368
+ const value = urlToken();
369
+ if (value != null) values.push(value);
370
+ };
371
+ // §4.3.3.
372
+ const numericToken = () => {
373
+ if (s[i] === "+" || s[i] === "-") i += 1;
374
+ while (isDigit(s[i])) i += 1;
375
+ if (s[i] === "." && isDigit(s[i + 1])) {
376
+ i += 1;
377
+ while (isDigit(s[i])) i += 1;
378
+ }
379
+ if ((s[i] === "e" || s[i] === "E") && (isDigit(s[i + 1]) || ((s[i + 1] === "+" || s[i + 1] === "-") && isDigit(s[i + 2])))) {
380
+ i += 2;
381
+ while (isDigit(s[i])) i += 1;
382
+ }
383
+ if (startsIdentSequence(s[i], s[i + 1], s[i + 2])) identSequence();
384
+ else if (s[i] === "%") i += 1;
385
+ };
386
+
387
+ // §4.3.1 (whitespace and single-code-point tokens advance by one).
388
+ while (i < s.length) {
389
+ const c = s[i];
390
+ if (c === "/" && s[i + 1] === "*") {
391
+ const end = s.indexOf("*/", i + 2);
392
+ i = end === -1 ? s.length : end + 2;
393
+ } else if (c === "\"" || c === "'") {
394
+ i += 1;
395
+ const value = stringToken(c);
396
+ if (value != null) values.push(value);
397
+ } else if (c === "#") {
398
+ i += 1;
399
+ if (isIdentCodePoint(s[i]) || isValidEscape(s[i], s[i + 1])) identSequence();
400
+ } else if (c === "+" || c === ".") {
401
+ if (startsNumber(c, s[i + 1], s[i + 2])) numericToken();
402
+ else i += 1;
403
+ } else if (c === "-") {
404
+ if (startsNumber(c, s[i + 1], s[i + 2])) numericToken();
405
+ else if (s[i + 1] === "-" && s[i + 2] === ">") i += 3;
406
+ else if (startsIdentSequence(c, s[i + 1], s[i + 2])) identLikeToken();
407
+ else i += 1;
408
+ } else if (c === "<") {
409
+ i += s.startsWith("<!--", i) ? 4 : 1;
410
+ } else if (c === "@") {
411
+ i += 1;
412
+ if (startsIdentSequence(s[i], s[i + 1], s[i + 2])) identSequence();
413
+ } else if (c === "\\") {
414
+ if (isValidEscape(c, s[i + 1])) identLikeToken();
415
+ else i += 1;
416
+ } else if (isDigit(c)) {
417
+ numericToken();
418
+ } else if (isIdentStart(c)) {
419
+ identLikeToken();
420
+ } else {
421
+ i += 1;
422
+ }
423
+ }
424
+ return values;
425
+ }
426
+
427
+ // The host candidates of one attribute value, each a whole value: a srcset or
428
+ // imagesrcset URL, a ping URL (split on ASCII whitespace), a url() or string
429
+ // token of a style attribute, or else the whole trimmed value. hostOf reads
430
+ // each one; a value that is not an absolute or protocol-relative URL names no
431
+ // host, so outside URL_ATTRIBUTES only a whole URL counts.
432
+ function attributeCandidates(name, value) {
433
+ if (name === "srcset" || name === "imagesrcset") return srcsetUrls(value);
434
+ if (name === "ping") return value.split(ASCII_WHITESPACE).filter(Boolean);
435
+ if (name === "style") return cssUrlValues(value);
436
+ return [trimAsciiWhitespace(value)];
437
+ }
438
+
439
+ function hostsOf(candidates) {
440
+ const hosts = new Set();
441
+ for (const candidate of candidates) {
442
+ const host = hostOf(candidate);
443
+ if (host) hosts.add(host);
444
+ }
445
+ return [...hosts];
446
+ }
447
+
448
+ const attrsOf = (node) => {
449
+ const map = new Map();
450
+ for (const attr of node.attrs || []) map.set(attr.name.toLowerCase(), attr.value);
451
+ return map;
452
+ };
453
+ const attrName = (attr) => (attr.prefix ? `${attr.prefix}:${attr.name}` : attr.name).toLowerCase();
454
+ const childrenOf = (node) => (node.tagName === "template" && node.content ? node.content.childNodes : node.childNodes) || [];
455
+ const textOf = (node) => (node.childNodes || []).filter((child) => child.nodeName === "#text").map((child) => child.value || "").join("");
456
+
457
+ const isHexByte = (byte) => (byte >= 0x30 && byte <= 0x39) || (byte >= 0x41 && byte <= 0x46) || (byte >= 0x61 && byte <= 0x66);
458
+ const UTF8 = new TextDecoder("utf-8", { ignoreBOM: true });
459
+ const UTF8_STRICT = new TextDecoder("utf-8", { ignoreBOM: true, fatal: true });
460
+
461
+ // The bytes `raw` percent-decodes to over its UTF-8 bytes. A `%` not followed
462
+ // by two hex digits is kept as is, or, with `strict`, makes the whole value
463
+ // undecodable (null).
464
+ function percentBytes(raw, { strict = false } = {}) {
465
+ const input = Buffer.from(raw, "utf8");
466
+ const bytes = [];
467
+ for (let i = 0; i < input.length; i += 1) {
468
+ if (input[i] === 0x25 && i + 2 < input.length && isHexByte(input[i + 1]) && isHexByte(input[i + 2])) {
469
+ bytes.push(Number.parseInt(String.fromCharCode(input[i + 1], input[i + 2]), 16));
470
+ i += 2;
471
+ } else if (input[i] === 0x25 && strict) {
472
+ return null;
473
+ } else {
474
+ bytes.push(input[i]);
475
+ }
476
+ }
477
+ return Uint8Array.from(bytes);
478
+ }
479
+
480
+ // A fragment as the target it names (contract 1.6 "Fragments are
481
+ // percent-decoded"): WHATWG percent-decode over its UTF-8 bytes, a `%` not
482
+ // followed by two hex digits kept as is, then UTF-8 decode (an invalid
483
+ // sequence reads U+FFFD). `#caf%C3%A9` names `café`, never `caf%C3%A9`.
484
+ function decodeFragment(raw) {
485
+ if (!raw.includes("%")) return raw;
486
+ return UTF8.decode(percentBytes(raw));
487
+ }
488
+
489
+ // One URL path segment as the file name it names, or null when it names none:
490
+ // a `%` without two hex digits, bytes that are not UTF-8, or a decoded `/` or
491
+ // NUL (no file name holds either).
492
+ function decodePathSegment(segment) {
493
+ if (!segment.includes("%")) return segment;
494
+ const bytes = percentBytes(segment, { strict: true });
495
+ if (!bytes) return null;
496
+ let decoded;
497
+ try {
498
+ decoded = UTF8_STRICT.decode(bytes);
499
+ } catch {
500
+ return null;
501
+ }
502
+ return /[/\u0000]/.test(decoded) ? null : decoded;
503
+ }
504
+
505
+ // A page's own URL: its path under the site root, each segment
506
+ // percent-encoded; null when the page is not under the site root.
507
+ function pageUrlOf(builtPath, siteRoot) {
508
+ if (!siteRoot || !insideRoot(siteRoot, builtPath)) return null;
509
+ return `${SITE_ORIGIN}/${relative(siteRoot, builtPath).split(sep).map(encodeURIComponent).join("/")}`;
510
+ }
511
+
512
+ /**
513
+ * The one mapping from a URL reference on a built page to a file under
514
+ * `_site/`, used for the og:image file and every local script. The URL parser
515
+ * resolves the reference against the page's own URL (pageUrlOf), as a browser
516
+ * does: backslashes read as `/`, dot segments are removed, a root-relative
517
+ * reference starts at the site root, and the query and fragment are not part
518
+ * of the path. Only a path reference (see readReference) is local, or an
519
+ * absolute or scheme-relative one under one of `bases` (a deploy base: its
520
+ * origin, or its host for a scheme-relative one, and a path starting with the
521
+ * base's path). Each segment of its path is then percent-decoded and joined
522
+ * onto `siteRoot`: `a%23b.png` names `a#b.png`, never a file literally called
523
+ * `a%23b.png`. An empty segment names no file: `og.png/` (or `og.png/%2e`,
524
+ * which the parser reads as `og.png/`) is a directory, and `a//b.png` is not
525
+ * `a/b.png`. Whether the file is inside the site root is the reader's check
526
+ * (its real path).
527
+ *
528
+ * @param {string} reference the attribute value
529
+ * @param {string} builtPath the page's file
530
+ * @param {string|null} siteRoot the built `_site/` directory
531
+ * @param {{ bases?: Array<{ origin: string, host: string, prefix: string }> }} [options]
532
+ * @returns {null|{ path: string, site_path: string }|{ unmappable: true, site_path?: string }}
533
+ * null when the reference is not local (anywhere but a deploy base, data:,
534
+ * or no path at all); unmappable when its path names no file (see
535
+ * decodePathSegment, and an empty segment), or when there is no site root
536
+ * or page URL to resolve it from.
537
+ */
538
+ export function builtFileOf(reference, builtPath, siteRoot, { bases = [] } = {}) {
539
+ const value = String(reference ?? "").trim();
540
+ // An empty reference, or one with only a query or fragment, names the page
541
+ // itself, not a file it loads.
542
+ if (!value || value.startsWith("?") || value.startsWith("#")) return null;
543
+ const pageUrl = pageUrlOf(builtPath, siteRoot);
544
+ const read = readReference(value, pageUrl ?? SITE_BASE);
545
+ let url;
546
+ if (read.kind === "path") {
547
+ if (!pageUrl) return parseUrl(value, SITE_BASE) ? { unmappable: true } : null;
548
+ url = parseUrl(value, pageUrl);
549
+ if (!url) return null;
550
+ } else {
551
+ url = read.url;
552
+ if (!url || !WEB_PROTOCOLS.has(url.protocol)) return null;
553
+ const same = read.kind === "absolute" ? (base) => base.origin === url.origin : (base) => base.host === url.host;
554
+ if (!bases.some((base) => same(base) && url.pathname.startsWith(base.prefix))) return null;
555
+ if (!siteRoot) return { unmappable: true };
556
+ }
557
+ const segments = url.pathname.split("/").slice(1).map(decodePathSegment);
558
+ if (segments.some((segment) => !segment || segment === "." || segment === "..")) return { unmappable: true, site_path: url.pathname };
559
+ return { path: join(siteRoot, ...segments), site_path: url.pathname };
560
+ }
561
+
562
+ // Nesting past MAX_ELEMENT_DEPTH found after the parse: nodes the parser
563
+ // moved, or a nested document's depth added to its host element's.
564
+ class TooDeep extends Error {}
565
+
566
+ // Whether an error means the page cannot be read or parsed (see
567
+ // isPageReadFailure), nesting past the bound found after the parse included.
568
+ export const isBuiltPageReadFailure = (error) => error instanceof TooDeep || isPageReadFailure(error);
569
+
570
+ // Every element in document order with its path (`html[1]/body[1]/a[2]`),
571
+ // its depth (the root element is 1) and whether it sits in <template>
572
+ // content, without recursion. A nested document starts at its host element's
573
+ // `path` and `depth`. An element deeper than MAX_ELEMENT_DEPTH throws TooDeep.
574
+ function walkElements(document, visit, { path: rootPath = "", depth: rootDepth = 0 } = {}) {
575
+ const stack = [];
576
+ const pushChildren = (node, path, inTemplate, depth) => {
577
+ const counts = new Map();
578
+ const frames = [];
579
+ for (const child of childrenOf(node)) {
580
+ if (!child.tagName) continue;
581
+ const tag = child.tagName.toLowerCase();
582
+ const index = (counts.get(tag) || 0) + 1;
583
+ counts.set(tag, index);
584
+ frames.push({ node: child, tag, path: `${path ? `${path}/` : ""}${tag}[${index}]`, inTemplate, depth: depth + 1 });
585
+ }
586
+ for (let i = frames.length - 1; i >= 0; i -= 1) stack.push(frames[i]);
587
+ };
588
+ pushChildren(document, rootPath, false, rootDepth);
589
+ while (stack.length) {
590
+ const frame = stack.pop();
591
+ if (frame.depth > MAX_ELEMENT_DEPTH) throw new TooDeep();
592
+ visit(frame);
593
+ pushChildren(frame.node, frame.path, frame.inTemplate || frame.tag === "template", frame.depth);
594
+ }
595
+ }
596
+
597
+ class CandidateCap {
598
+ constructor(limit, markupBytes) {
599
+ this.limit = limit;
600
+ this.count = 0;
601
+ this.reached = false;
602
+ this.markupBytes = markupBytes;
603
+ this.markupRead = 0;
604
+ }
605
+
606
+ // Whether one more candidate may be examined.
607
+ take() {
608
+ if (this.reached) return false;
609
+ if (this.count === this.limit) {
610
+ this.reached = true;
611
+ return false;
612
+ }
613
+ this.count += 1;
614
+ return true;
615
+ }
616
+
617
+ // Whether one more nested document of `bytes` may be parsed: one candidate,
618
+ // and the page's nested markup all together within the page size cap.
619
+ takeMarkup(bytes) {
620
+ if (this.reached) return false;
621
+ if (this.markupRead + bytes > this.markupBytes) {
622
+ this.reached = true;
623
+ return false;
624
+ }
625
+ if (!this.take()) return false;
626
+ this.markupRead += bytes;
627
+ return true;
628
+ }
629
+ }
630
+
631
+ // The URL-bearing attributes, a closed list: the HTML Living Standard
632
+ // attribute index entries whose value is a URL, a URL list or a hash-name
633
+ // reference, the obsolete URL attributes, and SVG href and xlink:href. Each
634
+ // non-empty value is one candidate, relative or absolute (a srcset or ping
635
+ // list counts once). Any other attribute (data-*, <meta content>, style, ...)
636
+ // is a candidate only when it holds an absolute URL, the one form the
637
+ // asset-host and loopback rules read from it. Nothing outside the list is
638
+ // added.
639
+ export const URL_ATTRIBUTES = Object.freeze(new Set([
640
+ "href", "src", "srcset", "imagesrcset", "poster", "action", "formaction", "data", "cite",
641
+ "ping", "itemid", "itemtype", "usemap", "background", "longdesc", "manifest", "codebase",
642
+ "classid", "archive", "profile", "lowsrc", "dynsrc", "xlink:href",
643
+ ]));
644
+
645
+ // The src of a <script> the page loads, or null: an HTML <script> with a
646
+ // non-empty src whose type runs as JavaScript (classic or module) and that is
647
+ // not in <template> content. <noscript> content parses as text, so a script
648
+ // there is never an element of the page.
649
+ function loadedScriptSrc(node, tag, inTemplate) {
650
+ if (inTemplate || tag !== "script" || node.namespaceURI !== HTML_NAMESPACE) return null;
651
+ const attrs = attrsOf(node);
652
+ const src = attrs.get("src");
653
+ return typeof src === "string" && src.trim() && scriptKind(attrs) ? src : null;
654
+ }
655
+
656
+ /**
657
+ * One built page's parse5 tree, through the 1.5 gate's bounded parser. Throws
658
+ * what it throws (isBuiltPageReadFailure reads it).
659
+ *
660
+ * @param {string} content the page's HTML
661
+ */
662
+ export const parseBuiltPage = (content) => parseBounded(content);
663
+
664
+ /**
665
+ * The src of every script a built page loads, in document order, read from
666
+ * its parse5 tree (see loadedScriptSrc).
667
+ *
668
+ * @param {object} document the page's tree (parseBuiltPage)
669
+ * @returns {string[]}
670
+ */
671
+ export function pageScriptSources(document) {
672
+ const sources = [];
673
+ walkElements(document, ({ node, tag, inTemplate }) => {
674
+ const src = loadedScriptSrc(node, tag, inTemplate);
675
+ if (src != null) sources.push(src);
676
+ });
677
+ return sources;
678
+ }
679
+
680
+ // The markup an element holds that the HTML parser keeps as text but a
681
+ // visitor can load, or null: a <noscript>'s text (parse5 parses with
682
+ // scripting on, so its content is one text node) and an <iframe>'s srcdoc.
683
+ // `at` names it in the element paths of what it holds.
684
+ function nestedMarkupOf(node, tag, attrs) {
685
+ if (node.namespaceURI !== HTML_NAMESPACE) return null;
686
+ if (tag === "noscript") return { markup: textOf(node), at: "" };
687
+ if (tag === "iframe" && attrs.has("srcdoc")) return { markup: attrs.get("srcdoc"), at: "@srcdoc/" };
688
+ return null;
689
+ }
690
+
691
+ // One parsed page's raw observation. Candidates past the cap are not
692
+ // examined; ids, names and the page's loaded scripts are read from the whole
693
+ // tree. A nested document (nestedMarkupOf) is read only for the asset-host
694
+ // and loopback rules, its element paths under its host element's.
695
+ function observeDocument(document) {
696
+ const cap = new CandidateCap(SMOKE_QC_LIMITS.candidates, SMOKE_QC_LIMITS.page_bytes);
697
+ const observed = {
698
+ liveTargets: new Set(),
699
+ templateTargets: new Set(),
700
+ anchors: new Map(),
701
+ favicon: [],
702
+ og: { title: null, description: null, image: null },
703
+ scripts: [],
704
+ tailwind: [],
705
+ assetHost: [],
706
+ loopback: [],
707
+ cap,
708
+ };
709
+ const refs = new Map([[observed.assetHost, new Set()], [observed.loopback, new Set()]]);
710
+ const addRef = (list, ref) => {
711
+ const id = `${ref.element_path}\u0000${ref.attr}`;
712
+ if (refs.get(list).has(id)) return;
713
+ refs.get(list).add(id);
714
+ list.push(ref);
715
+ };
716
+ const recordHosts = (hosts, ref) => {
717
+ if (hosts.includes(PRIMARY_ASSET_HOST)) addRef(observed.assetHost, ref);
718
+ if (hosts.some((host) => isLoopbackHostname(host))) addRef(observed.loopback, ref);
719
+ };
720
+
721
+ // The page's own elements: every rule.
722
+ const visitPage = ({ node, tag, path, inTemplate }, attrs) => {
723
+ const html = node.namespaceURI === HTML_NAMESPACE;
724
+ const targets = inTemplate ? observed.templateTargets : observed.liveTargets;
725
+ if (attrs.get("id")) targets.add(attrs.get("id"));
726
+ if (html && tag === "a" && attrs.get("name")) targets.add(attrs.get("name"));
727
+
728
+ // An in-page anchor is one candidate, its href included.
729
+ let anchorHref = false;
730
+ if (!inTemplate) {
731
+ const href = (tag === "a" || tag === "area") && typeof attrs.get("href") === "string" ? attrs.get("href").trim() : null;
732
+ anchorHref = href != null && href.startsWith("#");
733
+ if (anchorHref && cap.take()) {
734
+ const target = decodeFragment(href.slice(1));
735
+ if (!observed.anchors.has(target)) observed.anchors.set(target, { target, element_paths: [] });
736
+ observed.anchors.get(target).element_paths.push(path);
737
+ }
738
+ }
739
+ if (html && tag === "link" && cap.take() && !inTemplate) {
740
+ const rel = String(attrs.get("rel") || "").toLowerCase().split(ASCII_WHITESPACE);
741
+ if (rel.includes("icon") || rel.includes("apple-touch-icon")) observed.favicon.push(path);
742
+ }
743
+ if (html && tag === "meta" && cap.take() && !inTemplate) {
744
+ const content = String(attrs.get("content") || "").trim();
745
+ for (const name of ["property", "name"]) {
746
+ const field = OG_FIELDS.get(String(attrs.get(name) || "").trim().toLowerCase());
747
+ if (field && content && !observed.og[field]) observed.og[field] = { element_path: path, content };
748
+ }
749
+ }
750
+ const scriptSrc = loadedScriptSrc(node, tag, inTemplate);
751
+ if (scriptSrc != null) observed.scripts.push(scriptSrc);
752
+ readUrls(node, tag, path, { skipHref: anchorHref, tailwind: !inTemplate && html && tag === "script" });
753
+ };
754
+
755
+ // Every URL-bearing attribute value is a candidate, relative or absolute;
756
+ // only the absolute ones are matched by host.
757
+ const readUrls = (node, tag, path, { skipHref = false, tailwind = false } = {}) => {
758
+ for (const attr of node.attrs || []) {
759
+ const name = attrName(attr);
760
+ if (skipHref && name === "href") continue;
761
+ const hosts = hostsOf(attributeCandidates(name, attr.value));
762
+ if (!hosts.length && !(URL_ATTRIBUTES.has(name) && attr.value.trim())) continue;
763
+ if (!cap.take() || !hosts.length) continue;
764
+ recordHosts(hosts, { element_path: path, attr: name });
765
+ if (tailwind && name === "src" && hostOf(attr.value) === TAILWIND_CDN_HOST) observed.tailwind.push(path);
766
+ }
767
+ if (tag === "style") {
768
+ const hosts = hostsOf(cssUrlValues(textOf(node)));
769
+ if (hosts.length && cap.take()) recordHosts(hosts, { element_path: path, attr: "style" });
770
+ }
771
+ };
772
+
773
+ // A page or nested document; a nested one is read for its URLs only.
774
+ const walk = (tree, nested, start) => walkElements(tree, (frame) => {
775
+ const attrs = attrsOf(frame.node);
776
+ if (nested) readUrls(frame.node, frame.tag, frame.path);
777
+ else visitPage(frame, attrs);
778
+ const inner = nestedMarkupOf(frame.node, frame.tag, attrs);
779
+ // Markup with no `<` holds no element, so it is not parsed or counted.
780
+ if (!inner || !inner.markup.includes("<")) return;
781
+ if (!cap.takeMarkup(Buffer.byteLength(inner.markup, "utf8"))) return;
782
+ walk(parseBuiltPage(inner.markup), true, { path: `${frame.path}/${inner.at}`.replace(/\/$/, ""), depth: frame.depth });
783
+ }, start);
784
+ walk(document, false, {});
785
+ return observed;
786
+ }
787
+
788
+ // Whether `path` lies strictly inside `root` (both already real paths when
789
+ // the caller compares real paths).
790
+ export const insideRoot = (root, path) => {
791
+ const rel = relative(root, path);
792
+ return rel !== "" && !rel.startsWith(`..${sep}`) && rel !== ".." && !isAbsolute(rel);
793
+ };
794
+
795
+ // A path's real path (every symlink followed), or null when it cannot be
796
+ // resolved. Only a file-system read failure reads null; any other error is a
797
+ // defect and throws.
798
+ export function realPathOf(path) {
799
+ if (!path) return null;
800
+ try {
801
+ return realpathSync(path);
802
+ } catch (error) {
803
+ if (!isFileReadFailure(error)) throw error;
804
+ return null;
805
+ }
806
+ }
807
+
808
+ // The page's local scripts for the anchor hint: `contents` holds what was
809
+ // read, `complete` whether every local script the page loads was. The list is
810
+ // the caller's (`page.scripts`, or `readScripts` given the srcs the page
811
+ // loads); without one, every local <script src> on the page counts as loaded
812
+ // and unread.
813
+ function pageScripts(page, observed, builtPath, ctx) {
814
+ const listed = Array.isArray(page.scripts)
815
+ ? page.scripts
816
+ : ctx.readScripts
817
+ ? ctx.readScripts(observed.scripts, builtPath)
818
+ : observed.scripts.filter((src) => builtFileOf(src, builtPath, ctx.siteRoot) != null).map((src) => ({ src, unread: "not_listed" }));
819
+ const contents = listed.filter((script) => typeof script?.content === "string").map((script) => script.content);
820
+ return { contents, loaded: listed.length, read: contents.length, complete: contents.length === listed.length };
821
+ }
822
+
823
+ // Whether a local asset is a file inside the site root, its real path (every
824
+ // symlink followed) compared with the site root's own. A path that cannot be
825
+ // resolved or stat-ed reads as no file.
826
+ function fileExists(realSiteRoot, path) {
827
+ if (!realSiteRoot || path == null) return false;
828
+ const real = realPathOf(path);
829
+ if (!real || !insideRoot(realSiteRoot, real)) return false;
830
+ try {
831
+ return statSync(real).isFile();
832
+ } catch (error) {
833
+ if (!isFileReadFailure(error)) throw error;
834
+ return false;
835
+ }
836
+ }
837
+
838
+ // Where the page's og:image points: { image, image_target }. Absolute means
839
+ // a scheme and a host; a scheme-relative one (`//host/x`, or `\\host/x`) is
840
+ // not absolute, whatever host it names (readReference). Every file it names
841
+ // comes from builtFileOf; a reference whose path names no file is never a
842
+ // file in _site/, so it reads missing.
843
+ function resolveOgImage(content, builtPath, ctx) {
844
+ const value = content.trim();
845
+ const { kind, url } = readReference(value);
846
+ if (kind === "absolute") {
847
+ if (!WEB_PROTOCOLS.has(url.protocol)) return { image: "absolute_remote", image_target: url.protocol };
848
+ const target = `${url.protocol}//${url.host}${url.pathname}`;
849
+ if (!ctx.bases.length) return { image: "absolute_same_base_unmapped", image_target: target };
850
+ const mapped = builtFileOf(value, builtPath, ctx.siteRoot, { bases: ctx.bases });
851
+ if (!mapped) return { image: "absolute_remote", image_target: target };
852
+ return { image: fileExists(ctx.realSiteRoot, mapped.path ?? null) ? "absolute_same_base_present" : "absolute_same_base_missing", image_target: target };
853
+ }
854
+ if (kind === "scheme_relative" && url) return resolveSchemeRelativeOgImage(value, url, builtPath, ctx);
855
+ const mapped = builtFileOf(value, builtPath, ctx.siteRoot);
856
+ const path = mapped?.path ?? null;
857
+ const target = path != null ? `/${relative(ctx.siteRoot, path).split(sep).join("/")}` : mapped?.site_path ?? null;
858
+ return { image: fileExists(ctx.realSiteRoot, path) ? "relative_present" : "relative_missing", image_target: target };
859
+ }
860
+
861
+ // A scheme-relative og:image, `url` as the parser reads it: relative, so never
862
+ // a pass. Under a deploy base (its host and path) its path maps into _site/
863
+ // like an absolute same-base URL; anywhere else, or with no known base, its
864
+ // file is not looked for and it reads relative_present (the observation enum
865
+ // is closed).
866
+ function resolveSchemeRelativeOgImage(value, url, builtPath, ctx) {
867
+ const target = `//${url.host}${url.pathname}`;
868
+ const mapped = builtFileOf(value, builtPath, ctx.siteRoot, { bases: ctx.bases });
869
+ if (!mapped) return { image: "relative_present", image_target: target };
870
+ return { image: fileExists(ctx.realSiteRoot, mapped.path ?? null) ? "relative_present" : "relative_missing", image_target: target };
871
+ }
872
+
873
+ const OG_IMAGE_RESULT = Object.freeze({
874
+ relative_missing: ["warning", R.OG_IMAGE_MISSING_FILE],
875
+ relative_present: ["warning", R.OG_IMAGE_NOT_ABSOLUTE],
876
+ absolute_same_base_present: ["pass", null],
877
+ absolute_same_base_missing: ["warning", R.OG_IMAGE_MISSING_FILE],
878
+ absolute_same_base_unmapped: ["unexercised", R.OG_IMAGE_BASE_UNKNOWN],
879
+ absolute_remote: ["unexercised", R.OG_IMAGE_REMOTE_NOT_FETCHED],
880
+ });
881
+
882
+ // One target's outcome: [result, reason_code, outcome]. Only the decoded
883
+ // target is matched. A target not on the page is unexercised while any local
884
+ // script the page loads went unread, whatever the scripts read contain.
885
+ function anchorOutcome(anchor, observed, scripts) {
886
+ const { target } = anchor;
887
+ if (target === "" || observed.liveTargets.has(target) || target.toLowerCase() === "top") return ["pass", null, "resolved"];
888
+ if (observed.templateTargets.has(target)) return ["review", R.ANCHOR_TARGET_IN_TEMPLATE, "in_template"];
889
+ if (!scripts.complete) return ["unexercised", R.SCRIPT_UNREADABLE, "script_unreadable"];
890
+ if (scripts.contents.some((content) => content.includes(target))) return ["review", R.ANCHOR_TARGET_POSSIBLY_SCRIPT_CREATED, "in_script"];
891
+ return ["warning", R.ANCHOR_TARGET_MISSING, "missing"];
892
+ }
893
+
894
+ const environmentResult = (environment, found, reasonCode) => {
895
+ if (environment !== "production") return ["unexercised", ENVIRONMENT_REASON[environment]];
896
+ return found ? ["warning", reasonCode] : ["pass", null];
897
+ };
898
+
899
+ // The page's results before any cap: { anchors: [...], fixed: [...] }, each
900
+ // { key, result, reason_code, state, observation }, and `absence: true` on a
901
+ // result that rests on something not being on the page.
902
+ function pageFindings(page, file, observed, builtPath, ctx) {
903
+ const scripts = pageScripts(page, observed, builtPath, ctx);
904
+ const anchors = [...observed.anchors.values()].map((anchor) => {
905
+ const [result, reasonCode, outcome] = anchorOutcome(anchor, observed, scripts);
906
+ const read = outcome === "resolved" || outcome === "in_template" ? null : scripts;
907
+ return {
908
+ key: anchorKey(anchor.target),
909
+ result,
910
+ reason_code: reasonCode,
911
+ state: { element_paths: anchor.element_paths },
912
+ observation: { page: file, target: anchor.target, outcome, scripts_read: read ? read.read : null, scripts_loaded: read ? read.loaded : null },
913
+ };
914
+ });
915
+
916
+ const { environment } = ctx;
917
+ const og = observed.og;
918
+ const ogImage = og.image ? resolveOgImage(og.image.content, builtPath, ctx) : { image: "absent", image_target: null };
919
+ const ogObservation = { title: Boolean(og.title), description: Boolean(og.description), image: ogImage.image, image_target: ogImage.image_target };
920
+ const presence = (key, field, reasonCode) => ({
921
+ key,
922
+ result: og[field] ? "pass" : "warning",
923
+ reason_code: og[field] ? null : reasonCode,
924
+ state: { element_paths: og[field] ? [og[field].element_path] : [] },
925
+ observation: { page: file, og: ogObservation },
926
+ absence: true,
927
+ });
928
+ const refsState = (refs) => ({ element_paths: refs.map((ref) => ref.element_path), attributes: refs.map((ref) => ref.attr) });
929
+ const [tailwindResult, tailwindReason] = environmentResult(environment, observed.tailwind.length > 0, R.TAILWIND_CDN_IN_PRODUCTION);
930
+ const [loopbackResult, loopbackReason] = environmentResult(environment, observed.loopback.length > 0, R.LOOPBACK_URL);
931
+ const fixed = [
932
+ {
933
+ key: K.favicon,
934
+ result: observed.favicon.length ? "pass" : "warning",
935
+ reason_code: observed.favicon.length ? null : R.FAVICON_MISSING,
936
+ state: { element_paths: observed.favicon },
937
+ observation: { page: file, favicon: observed.favicon.length > 0 },
938
+ absence: true,
939
+ },
940
+ presence(K.og_title, "title", R.OG_TITLE_MISSING),
941
+ presence(K.og_description, "description", R.OG_DESCRIPTION_MISSING),
942
+ presence(K.og_image, "image", R.OG_IMAGE_MISSING),
943
+ ...(og.image ? [{
944
+ key: K.og_image_target,
945
+ result: OG_IMAGE_RESULT[ogImage.image][0],
946
+ reason_code: OG_IMAGE_RESULT[ogImage.image][1],
947
+ state: { element_paths: [og.image.element_path], image_target: ogImage.image_target },
948
+ observation: { page: file, og: ogObservation },
949
+ }] : []),
950
+ {
951
+ key: K.tailwind_cdn,
952
+ result: tailwindResult,
953
+ reason_code: tailwindReason,
954
+ state: { element_paths: observed.tailwind, environment },
955
+ observation: { page: file, tailwind_cdn: observed.tailwind.length, environment },
956
+ },
957
+ {
958
+ key: K.asset_host,
959
+ result: observed.assetHost.length ? "warning" : "pass",
960
+ reason_code: observed.assetHost.length ? R.PRIMARY_ASSET_HOST : null,
961
+ state: refsState(observed.assetHost),
962
+ observation: { page: file, asset_host_refs: observed.assetHost },
963
+ },
964
+ {
965
+ key: K.loopback,
966
+ result: loopbackResult,
967
+ reason_code: loopbackReason,
968
+ state: { ...refsState(observed.loopback), environment },
969
+ observation: { page: file, loopback_refs: observed.loopback, environment },
970
+ },
971
+ ];
972
+ return { anchors, fixed };
973
+ }
974
+
975
+ const PAGE_LEVEL_KEYS = Object.freeze(Object.values(K));
976
+ const isFinding = (finding) => finding.result === "warning" || finding.result === "review";
977
+ // What a capped page keeps: a warning or review observed in full, never one
978
+ // that rests on something being absent from a page not seen whole.
979
+ const keptOnCappedPage = (finding) => isFinding(finding) && !finding.absence;
980
+ const capMembersFor = (reasons) => reasons.flatMap((reason) => aggregateQcResults([], { capReason: reason }).members);
981
+
982
+ /**
983
+ * Evaluate the built-output smoke checks.
984
+ *
985
+ * @param {{
986
+ * pages: Array<{ file: string, content?: string, bytes?: number, unreadable?: boolean, scripts?: Array<{ src: string, file?: string, content?: string, unread?: string }> }>,
987
+ * environment?: string|null,
988
+ * siteRoot: string|null,
989
+ * targetDir?: string|null,
990
+ * readScripts?: ((srcs: string[], builtPath: string) => Array<{ src: string, file?: string, content?: string, unread?: string }>)|null,
991
+ * deployBase?: string|string[]|null,
992
+ * measuredAt?: string,
993
+ * }} input `pages` in a stable order, `file` relative to the doctor target
994
+ * ("_site/<slug>/index.html"); pages past the page cap may omit `content`.
995
+ * `scripts` lists every local script the page loads, read or not (see the
996
+ * header); without it `readScripts`, when given, lists them from the srcs
997
+ * the page's parse found, and otherwise the page's local scripts read
998
+ * unread. `environment` is the recorded build environment ("production" or
999
+ * "development"; anything else is unknown). `siteRoot` is the built site
1000
+ * root (the scope's `_site/`, or the directory doctor was pointed at when
1001
+ * it has none): every local reference maps onto it through builtFileOf
1002
+ * (none does without it), and a local asset whose real path lies outside it
1003
+ * reads as missing. Page files resolve against `targetDir`, the doctor
1004
+ * target (by default the site root's parent). `deployBase` lists the deploy
1005
+ * URLs under which an absolute og:image maps into the site root (each
1006
+ * one's origin and path); none means the base is unknown. Every result
1007
+ * carries its own {check, page, key} subject.
1008
+ * @returns {object[]} QC results (src/qc-results.mjs buildQcResult).
1009
+ */
1010
+ export function evaluateSmokeQc({ pages = [], environment = null, siteRoot = null, targetDir = null, readScripts = null, deployBase = null, measuredAt = new Date().toISOString() } = {}) {
1011
+ const check = SMOKE_QC_CHECK;
1012
+ const ctx = {
1013
+ environment: environmentOf(environment),
1014
+ siteRoot,
1015
+ bases: deployBases(deployBase),
1016
+ realSiteRoot: realPathOf(siteRoot),
1017
+ readScripts,
1018
+ };
1019
+ const pageDir = targetDir ?? (siteRoot ? dirname(siteRoot) : null);
1020
+ const results = [];
1021
+ const row = (page, { key, result, reason_code = null, state = {}, observation = {} }, { members = [], coverage } = {}) => buildQcResult({
1022
+ check,
1023
+ leg: "doctor",
1024
+ subject: { check, page, key },
1025
+ result,
1026
+ reason_code,
1027
+ state: { reason_code: result === "pass" ? null : reason_code, ...state, members },
1028
+ observation,
1029
+ members,
1030
+ coverage: coverage ?? (result === "unexercised" ? { observed: 0, expected: 1, limits: [reason_code] } : { observed: 1, expected: 1, limits: [] }),
1031
+ measured_at: measuredAt,
1032
+ });
1033
+ const pageLevel = (file, reasonCode, extra = {}) => {
1034
+ for (const key of PAGE_LEVEL_KEYS) {
1035
+ results.push(row(file, { key, result: "unexercised", reason_code: reasonCode, observation: { page: file, ...extra } }));
1036
+ }
1037
+ };
1038
+
1039
+ (Array.isArray(pages) ? pages : []).forEach((page, index) => {
1040
+ const file = String(page?.file ?? "");
1041
+ if (index >= SMOKE_QC_LIMITS.pages) {
1042
+ pageLevel(file, R.PAGE_CAP_REACHED);
1043
+ return;
1044
+ }
1045
+ const read = readPage(page || {});
1046
+ if (read.outcome) {
1047
+ pageLevel(file, read.outcome, read.bytes == null ? {} : { bytes: read.bytes });
1048
+ return;
1049
+ }
1050
+ let findings;
1051
+ let observed;
1052
+ try {
1053
+ observed = observeDocument(read.document);
1054
+ findings = pageFindings(page, file, observed, pageDir ? join(pageDir, file) : file, ctx);
1055
+ } catch (error) {
1056
+ if (!isBuiltPageReadFailure(error)) throw error;
1057
+ pageLevel(file, R.PAGE_UNREADABLE, { bytes: read.bytes });
1058
+ return;
1059
+ }
1060
+
1061
+ const caps = [];
1062
+ if (observed.cap.reached) caps.push(R.CANDIDATE_CAP_REACHED);
1063
+ // The result cap counts the anchor rule's results, pass included (the
1064
+ // only rule with more than one result per page).
1065
+ if (findings.anchors.length > SMOKE_QC_LIMITS.results) caps.push(R.FINDING_CAP_REACHED);
1066
+ if (!caps.length) {
1067
+ for (const finding of [...findings.anchors, ...findings.fixed]) results.push(row(file, finding));
1068
+ return;
1069
+ }
1070
+
1071
+ // A capped page: the warning and review results observed in full stay,
1072
+ // each with the capped-page member; every other rule key reads
1073
+ // unexercised for the cap, and nothing passes.
1074
+ const members = capMembersFor(caps);
1075
+ const capped = (finding) => row(file, finding, { members, coverage: { observed: finding.result === "unexercised" ? 0 : 1, expected: null, limits: [...caps] } });
1076
+ const capRow = (key) => capped({ key, result: "unexercised", reason_code: caps[0], observation: { page: file, candidates: observed.cap.count } });
1077
+ for (const finding of findings.anchors.slice(0, SMOKE_QC_LIMITS.results)) {
1078
+ if (keptOnCappedPage(finding)) results.push(capped(finding));
1079
+ }
1080
+ results.push(capRow(K.anchor));
1081
+ for (const key of PAGE_LEVEL_KEYS.filter((value) => value !== K.anchor)) {
1082
+ const finding = findings.fixed.find((item) => item.key === key);
1083
+ results.push(finding && keptOnCappedPage(finding) ? capped(finding) : capRow(key));
1084
+ }
1085
+ });
1086
+ return results;
1087
+ }
1088
+
1089
+ // One page's parse5 tree (parseBuiltPage), or a page-level outcome. A read or
1090
+ // parse failure, nesting past MAX_ELEMENT_DEPTH included, reads the page
1091
+ // unreadable; any other error is a defect and throws.
1092
+ function readPage(page) {
1093
+ if (page.unreadable) return { outcome: R.PAGE_UNREADABLE };
1094
+ const bytes = Number.isFinite(page.bytes) ? page.bytes : typeof page.content === "string" ? Buffer.byteLength(page.content, "utf8") : null;
1095
+ if (bytes != null && bytes > SMOKE_QC_LIMITS.page_bytes) return { outcome: R.PAGE_TOO_LARGE, bytes };
1096
+ if (typeof page.content !== "string") return { outcome: R.PAGE_UNREADABLE };
1097
+ try {
1098
+ return { document: parseBuiltPage(page.content), bytes };
1099
+ } catch (error) {
1100
+ if (!isBuiltPageReadFailure(error)) throw error;
1101
+ return { outcome: R.PAGE_UNREADABLE, bytes };
1102
+ }
1103
+ }
1104
+
1105
+ // Each deploy URL as a base an og:image maps through: its origin, its host
1106
+ // and its path as a directory (`https://deploy.example/s` and `/s/` both
1107
+ // read `/s/`), so only a path under it maps into the site root.
1108
+ function deployBases(deployBase) {
1109
+ const bases = [];
1110
+ for (const value of [deployBase].flat()) {
1111
+ if (typeof value !== "string" || !value.trim()) continue;
1112
+ const url = parseUrl(value.trim());
1113
+ if (!url || !WEB_PROTOCOLS.has(url.protocol)) continue;
1114
+ bases.push({ origin: url.origin, host: url.host, prefix: url.pathname.endsWith("/") ? url.pathname : `${url.pathname}/` });
1115
+ }
1116
+ return bases;
1117
+ }