mcp-scraper 0.40.1 → 0.40.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/README.md +1 -1
  2. package/dist/bin/api-server.cjs +60220 -0
  3. package/dist/bin/api-server.cjs.map +1 -0
  4. package/dist/bin/api-server.d.cts +1 -0
  5. package/dist/bin/api-server.d.ts +1 -0
  6. package/dist/bin/api-server.js +38 -0
  7. package/dist/bin/api-server.js.map +1 -0
  8. package/dist/bin/mcp-scraper-cli.cjs +2671 -0
  9. package/dist/bin/mcp-scraper-cli.cjs.map +1 -0
  10. package/dist/bin/mcp-scraper-cli.d.cts +1 -0
  11. package/dist/bin/mcp-scraper-cli.d.ts +1 -0
  12. package/dist/bin/mcp-scraper-cli.js +742 -0
  13. package/dist/bin/mcp-scraper-cli.js.map +1 -0
  14. package/dist/bin/mcp-scraper-install.cjs +129 -0
  15. package/dist/bin/mcp-scraper-install.cjs.map +1 -0
  16. package/dist/bin/mcp-scraper-install.d.cts +1 -0
  17. package/dist/bin/mcp-scraper-install.d.ts +1 -0
  18. package/dist/bin/mcp-scraper-install.js +27 -0
  19. package/dist/bin/mcp-scraper-install.js.map +1 -0
  20. package/dist/bin/mcp-stdio-server.cjs +12532 -0
  21. package/dist/bin/mcp-stdio-server.cjs.map +1 -0
  22. package/dist/bin/mcp-stdio-server.d.cts +1 -0
  23. package/dist/bin/mcp-stdio-server.d.ts +1 -0
  24. package/dist/bin/mcp-stdio-server.js +135 -0
  25. package/dist/bin/mcp-stdio-server.js.map +1 -0
  26. package/dist/bin/paa-harvest.cjs +3808 -0
  27. package/dist/bin/paa-harvest.cjs.map +1 -0
  28. package/dist/bin/paa-harvest.d.cts +1 -0
  29. package/dist/bin/paa-harvest.d.ts +1 -0
  30. package/dist/bin/paa-harvest.js +44 -0
  31. package/dist/bin/paa-harvest.js.map +1 -0
  32. package/dist/chunk-345BQXZH.js +712 -0
  33. package/dist/chunk-345BQXZH.js.map +1 -0
  34. package/dist/chunk-3PIWJS6Y.js +276 -0
  35. package/dist/chunk-3PIWJS6Y.js.map +1 -0
  36. package/dist/chunk-44HZLHDV.js +52 -0
  37. package/dist/chunk-44HZLHDV.js.map +1 -0
  38. package/dist/chunk-7XBBFBYY.js +3410 -0
  39. package/dist/chunk-7XBBFBYY.js.map +1 -0
  40. package/dist/chunk-CB5C3BPB.js +135 -0
  41. package/dist/chunk-CB5C3BPB.js.map +1 -0
  42. package/dist/chunk-FQI5PFE7.js +1866 -0
  43. package/dist/chunk-FQI5PFE7.js.map +1 -0
  44. package/dist/chunk-G3P3ZDB4.js +69 -0
  45. package/dist/chunk-G3P3ZDB4.js.map +1 -0
  46. package/dist/chunk-GUVKHCKE.js +3050 -0
  47. package/dist/chunk-GUVKHCKE.js.map +1 -0
  48. package/dist/chunk-K443GQY5.js +24 -0
  49. package/dist/chunk-K443GQY5.js.map +1 -0
  50. package/dist/chunk-LP6E462I.js +11555 -0
  51. package/dist/chunk-LP6E462I.js.map +1 -0
  52. package/dist/chunk-NVXNEOUQ.js +158 -0
  53. package/dist/chunk-NVXNEOUQ.js.map +1 -0
  54. package/dist/chunk-QZXKQB7Y.js +414 -0
  55. package/dist/chunk-QZXKQB7Y.js.map +1 -0
  56. package/dist/chunk-SBLGBZZB.js +647 -0
  57. package/dist/chunk-SBLGBZZB.js.map +1 -0
  58. package/dist/chunk-X2LKCX6H.js +499 -0
  59. package/dist/chunk-X2LKCX6H.js.map +1 -0
  60. package/dist/chunk-XELSA2MS.js +108 -0
  61. package/dist/chunk-XELSA2MS.js.map +1 -0
  62. package/dist/chunk-XORPNO3Z.js +684 -0
  63. package/dist/chunk-XORPNO3Z.js.map +1 -0
  64. package/dist/chunk-ZUJLSICT.js +7 -0
  65. package/dist/chunk-ZUJLSICT.js.map +1 -0
  66. package/dist/db-W3CP562I.js +239 -0
  67. package/dist/db-W3CP562I.js.map +1 -0
  68. package/dist/editorial-reading-room/assets/app.js +335 -0
  69. package/dist/editorial-reading-room/assets/index.html +131 -0
  70. package/dist/editorial-reading-room/assets/styles.css +1052 -0
  71. package/dist/extract-bundle-K4PG3RZJ.js +568 -0
  72. package/dist/extract-bundle-K4PG3RZJ.js.map +1 -0
  73. package/dist/index.cjs +4160 -0
  74. package/dist/index.cjs.map +1 -0
  75. package/dist/index.d.cts +413 -0
  76. package/dist/index.d.ts +413 -0
  77. package/dist/index.js +338 -0
  78. package/dist/index.js.map +1 -0
  79. package/dist/location-data-repository-RLQX6SNM.js +35 -0
  80. package/dist/location-data-repository-RLQX6SNM.js.map +1 -0
  81. package/dist/server-KM3CFWCF.js +34674 -0
  82. package/dist/server-KM3CFWCF.js.map +1 -0
  83. package/dist/site-extract-repository-2SMMFKKL.js +62 -0
  84. package/dist/site-extract-repository-2SMMFKKL.js.map +1 -0
  85. package/dist/worker-FXAGFYOE.js +142 -0
  86. package/dist/worker-FXAGFYOE.js.map +1 -0
  87. package/package.json +4 -2
@@ -0,0 +1,712 @@
1
+ // src/api/url-utils.ts
2
+ import { isIP } from "net";
3
+ import { lookup } from "dns/promises";
4
+ import ipaddr from "ipaddr.js";
5
+ function unbracketIpLiteral(value) {
6
+ return value.startsWith("[") && value.endsWith("]") ? value.slice(1, -1) : value;
7
+ }
8
+ function isPrivateIpAddress(address) {
9
+ const normalized = unbracketIpLiteral(address);
10
+ if (!isIP(normalized)) return false;
11
+ try {
12
+ const parsed = ipaddr.parse(normalized);
13
+ const isExtraNonGlobalIpv4 = (ipv4) => {
14
+ const extraNonGlobal = [
15
+ ["192.0.0.0", 24],
16
+ ["192.0.2.0", 24],
17
+ ["198.18.0.0", 15],
18
+ ["198.51.100.0", 24],
19
+ ["203.0.113.0", 24]
20
+ ];
21
+ return extraNonGlobal.some(([network, bits]) => ipv4.match(ipaddr.parse(network), bits));
22
+ };
23
+ if (parsed instanceof ipaddr.IPv6 && parsed.isIPv4MappedAddress()) {
24
+ const mapped = parsed.toIPv4Address();
25
+ return mapped.range() !== "unicast" || isExtraNonGlobalIpv4(mapped);
26
+ }
27
+ if (parsed instanceof ipaddr.IPv4) {
28
+ if (isExtraNonGlobalIpv4(parsed)) return true;
29
+ }
30
+ return parsed.range() !== "unicast";
31
+ } catch {
32
+ return true;
33
+ }
34
+ }
35
+ async function resolvesToPrivateAddress(hostname) {
36
+ const host = unbracketIpLiteral(hostname.toLowerCase());
37
+ if (host === "localhost" || host.endsWith(".localhost") || host.endsWith(".local")) return true;
38
+ if (isPrivateIpAddress(host)) return true;
39
+ try {
40
+ const addresses = await lookup(host, { all: true, verbatim: true });
41
+ return addresses.length === 0 || addresses.some((entry) => isPrivateIpAddress(entry.address));
42
+ } catch {
43
+ return true;
44
+ }
45
+ }
46
+ async function validatePublicHttpUrl(raw, opts) {
47
+ let parsed;
48
+ try {
49
+ parsed = new URL(raw.trim());
50
+ } catch {
51
+ return { error: `Invalid ${opts.field}` };
52
+ }
53
+ const allowedProtocols = opts.requireHttps ? ["https:"] : ["http:", "https:"];
54
+ if (!allowedProtocols.includes(parsed.protocol)) {
55
+ return { error: opts.requireHttps ? `${opts.field} must use https` : `${opts.field} must use http or https` };
56
+ }
57
+ if (await resolvesToPrivateAddress(parsed.hostname)) {
58
+ return { error: `${opts.field} must resolve to a public internet host` };
59
+ }
60
+ return { parsed };
61
+ }
62
+
63
+ // src/api/image-audit.ts
64
+ var OVER_BYTES = 100 * 1024;
65
+ var MODERN = /* @__PURE__ */ new Set(["webp", "avif", "svg", "svg+xml"]);
66
+ function formatBytes(n) {
67
+ if (n == null) return null;
68
+ if (n < 1024) return `${n} B`;
69
+ if (n < 1024 * 1024) return `${(n / 1024).toFixed(1)} KB`;
70
+ return `${(n / (1024 * 1024)).toFixed(2)} MB`;
71
+ }
72
+ var decodeEntities = (u) => u.replace(/&amp;/g, "&").replace(/&#0?38;/g, "&").replace(/&#x26;/gi, "&").trim();
73
+ var dedupKey = (u) => u.replace(/^https?:\/\//i, "//").replace(/\/$/, "");
74
+ var formatOf = (ct, url) => {
75
+ if (ct) {
76
+ const m = ct.split(";")[0].trim().toLowerCase();
77
+ if (m.startsWith("image/")) return m.replace("image/", "");
78
+ }
79
+ return (url.split("?")[0].match(/\.([a-z0-9]+)$/i)?.[1] || "unknown").toLowerCase();
80
+ };
81
+ function collectUrls(pages) {
82
+ const byKey = /* @__PURE__ */ new Map();
83
+ for (const p of pages) for (const raw of p.imageLinks || []) {
84
+ const u = decodeEntities(raw);
85
+ if (!/^https?:\/\//i.test(u)) continue;
86
+ if (/[{}]|%7[bd]/i.test(u)) continue;
87
+ const k = dedupKey(u);
88
+ const prev = byKey.get(k);
89
+ if (!prev || /^https:/i.test(u) && /^http:/i.test(prev)) byKey.set(k, u);
90
+ }
91
+ return [...byKey.values()];
92
+ }
93
+ async function sizeAndType(url, timeoutMs) {
94
+ const ctrl = new AbortController();
95
+ const t = setTimeout(() => ctrl.abort(), timeoutMs);
96
+ const read = (res) => ({
97
+ len: res.headers.get("content-length"),
98
+ cr: res.headers.get("content-range"),
99
+ ct: res.headers.get("content-type"),
100
+ status: res.status
101
+ });
102
+ const safeFetch = async (method) => {
103
+ let target = url;
104
+ for (let redirects = 0; redirects <= 5; redirects++) {
105
+ const checked = await validatePublicHttpUrl(target, { field: "image URL" });
106
+ if (checked.error || !checked.parsed) throw new Error(checked.error ?? "Image URL was rejected");
107
+ const response = await fetch(checked.parsed.href, {
108
+ method,
109
+ ...method === "GET" ? { headers: { Range: "bytes=0-0" } } : {},
110
+ redirect: "manual",
111
+ signal: ctrl.signal
112
+ });
113
+ if (response.status >= 300 && response.status < 400) {
114
+ const location = response.headers.get("location");
115
+ if (!location) throw new Error(`HTTP ${response.status} redirect did not include Location`);
116
+ target = new URL(location, checked.parsed.href).href;
117
+ continue;
118
+ }
119
+ return response;
120
+ }
121
+ throw new Error("Image request exceeded five redirects");
122
+ };
123
+ try {
124
+ let bytes = null;
125
+ let ct = null;
126
+ let status = null;
127
+ try {
128
+ const h = read(await safeFetch("HEAD"));
129
+ status = h.status;
130
+ ct = h.ct;
131
+ if (h.len) bytes = Number(h.len);
132
+ } catch {
133
+ }
134
+ if (bytes == null) {
135
+ const g = await safeFetch("GET");
136
+ const r = read(g);
137
+ status = status ?? r.status;
138
+ ct = ct ?? r.ct;
139
+ if (r.cr && r.cr.includes("/")) {
140
+ const tail = r.cr.split("/")[1];
141
+ if (tail && tail !== "*") bytes = Number(tail);
142
+ } else if (r.len) bytes = Number(r.len);
143
+ try {
144
+ await g.body?.cancel();
145
+ } catch {
146
+ }
147
+ }
148
+ return { url, status, bytes: Number.isFinite(bytes) ? bytes : null, contentType: ct };
149
+ } catch (e) {
150
+ return { url, status: null, bytes: null, contentType: null, error: String(e.message || e) };
151
+ } finally {
152
+ clearTimeout(t);
153
+ }
154
+ }
155
+ async function pool(items, n, fn) {
156
+ const out = new Array(items.length);
157
+ let i = 0;
158
+ await Promise.all(Array.from({ length: Math.min(n, items.length) }, async () => {
159
+ while (i < items.length) {
160
+ const idx = i++;
161
+ out[idx] = await fn(items[idx]);
162
+ }
163
+ }));
164
+ return out;
165
+ }
166
+ async function auditImages(pages, opts = {}) {
167
+ return auditImageUrls(collectUrls(pages), opts);
168
+ }
169
+ async function auditImageUrls(inputUrls, opts = {}) {
170
+ const concurrency = opts.concurrency ?? 12;
171
+ const timeoutMs = opts.timeoutMs ?? 12e3;
172
+ const max = opts.max ?? 5e3;
173
+ const urls = collectUrls([{ imageLinks: inputUrls }]).slice(0, max);
174
+ const heads = await pool(urls, concurrency, (u) => sizeAndType(u, timeoutMs));
175
+ const rows = heads.map((r) => {
176
+ const format = formatOf(r.contentType, r.url);
177
+ return {
178
+ ...r,
179
+ size: formatBytes(r.bytes),
180
+ format,
181
+ over100kb: r.bytes != null && r.bytes > OVER_BYTES,
182
+ legacyFormat: format !== "unknown" && !MODERN.has(format)
183
+ };
184
+ });
185
+ const sized = rows.filter((r) => r.bytes != null);
186
+ const totalBytes = sized.reduce((a, r) => a + r.bytes, 0);
187
+ const formatCounts = {};
188
+ for (const r of rows) formatCounts[r.format] = (formatCounts[r.format] || 0) + 1;
189
+ return {
190
+ rows: rows.sort((a, b) => (b.bytes ?? 0) - (a.bytes ?? 0)),
191
+ summary: {
192
+ unique: rows.length,
193
+ sized: sized.length,
194
+ totalBytes,
195
+ totalSize: formatBytes(totalBytes) ?? "0 B",
196
+ avgSize: formatBytes(Math.round(totalBytes / (sized.length || 1))) ?? "0 B",
197
+ over100kb: rows.filter((r) => r.over100kb).length,
198
+ legacyFormat: rows.filter((r) => r.legacyFormat).length,
199
+ formatCounts,
200
+ ...opts.sampleTruncated ? { sampleTruncated: true, sampleCap: max } : {}
201
+ }
202
+ };
203
+ }
204
+ function renderImageSection(audit) {
205
+ const s = audit.summary;
206
+ const fmt = Object.entries(s.formatCounts).sort((a, b) => b[1] - a[1]).map(([k, v]) => `${k}:${v}`).join(" ");
207
+ const heaviest = audit.rows.filter((r) => r.over100kb).slice(0, 15);
208
+ const lines = [
209
+ `## Images`,
210
+ `**${s.unique} unique images** \xB7 ${s.totalSize} total \xB7 ${s.avgSize} avg \xB7 ${s.over100kb} over 100 KB \xB7 ${s.legacyFormat} legacy format`,
211
+ `Formats: ${fmt}`
212
+ ];
213
+ if (heaviest.length) {
214
+ lines.push("", "| Size | Format | URL |", "|------|--------|-----|");
215
+ for (const r of heaviest) lines.push(`| ${r.size} | ${r.format} | ${r.url} |`);
216
+ }
217
+ return lines.join("\n");
218
+ }
219
+
220
+ // src/api/seo-link-report.ts
221
+ function registrableDomain(host) {
222
+ const h = host.replace(/^www\./, "").toLowerCase();
223
+ const parts = h.split(".");
224
+ return parts.length <= 2 ? h : parts.slice(-2).join(".");
225
+ }
226
+ function buildLinkReport(edges, metrics, siteUrl) {
227
+ let siteReg = "";
228
+ try {
229
+ siteReg = registrableDomain(new URL(siteUrl).hostname);
230
+ } catch {
231
+ siteReg = "";
232
+ }
233
+ const internalEdges = edges.filter((e) => e.internal);
234
+ const pages = metrics.length;
235
+ const orphans = metrics.filter((m) => m.orphan).length;
236
+ const brokenInternal = internalEdges.filter((e) => e.targetStatus != null && e.targetStatus >= 400).length;
237
+ const sumInlinks = metrics.reduce((a, m) => a + m.inlinks, 0);
238
+ const sumOutlinks = metrics.reduce((a, m) => a + m.outlinksInternal + m.outlinksExternal, 0);
239
+ const distribution = { zero: 0, oneToTwo: 0, threeToTen: 0, elevenPlus: 0 };
240
+ for (const m of metrics) {
241
+ if (m.inlinks === 0) distribution.zero++;
242
+ else if (m.inlinks <= 2) distribution.oneToTwo++;
243
+ else if (m.inlinks <= 10) distribution.threeToTen++;
244
+ else distribution.elevenPlus++;
245
+ }
246
+ const topByInlinks = [...metrics].sort((a, b) => b.inlinks - a.inlinks).slice(0, 20).map((m) => ({ url: m.url, inlinks: m.inlinks, outlinksInternal: m.outlinksInternal, outlinksExternal: m.outlinksExternal }));
247
+ const domMap = /* @__PURE__ */ new Map();
248
+ let externalTotal = 0;
249
+ for (const e of edges) {
250
+ let domain;
251
+ try {
252
+ domain = registrableDomain(new URL(e.to).hostname);
253
+ } catch {
254
+ continue;
255
+ }
256
+ if (!domain || domain === siteReg) continue;
257
+ externalTotal++;
258
+ const d = domMap.get(domain) ?? { links: 0, nofollow: 0, pages: /* @__PURE__ */ new Set() };
259
+ d.links++;
260
+ if (e.nofollow) d.nofollow++;
261
+ d.pages.add(e.from);
262
+ domMap.set(domain, d);
263
+ }
264
+ const externalDomains = [...domMap.entries()].map(([domain, d]) => ({ domain, links: d.links, nofollow: d.nofollow, pages: d.pages.size })).sort((a, b) => b.links - a.links);
265
+ const round = (n) => Math.round(n * 10) / 10;
266
+ return {
267
+ summary: {
268
+ internal: {
269
+ totalLinks: internalEdges.length,
270
+ pages,
271
+ orphans,
272
+ brokenInternal,
273
+ avgInlinks: pages ? round(sumInlinks / pages) : 0,
274
+ avgOutlinks: pages ? round(sumOutlinks / pages) : 0,
275
+ distribution,
276
+ topByInlinks
277
+ },
278
+ external: {
279
+ totalLinks: externalTotal,
280
+ uniqueDomains: externalDomains.length,
281
+ topDomains: externalDomains.slice(0, 20)
282
+ }
283
+ },
284
+ externalDomains
285
+ };
286
+ }
287
+ function renderLinkReport(r) {
288
+ const s = r.summary;
289
+ const lines = [
290
+ `## Link analysis`,
291
+ `**Internal:** ${s.internal.totalLinks} links \xB7 ${s.internal.pages} pages \xB7 avg ${s.internal.avgInlinks} inlinks/page \xB7 ${s.internal.orphans} orphans \xB7 ${s.internal.brokenInternal} broken`,
292
+ `**External:** ${s.external.totalLinks} links to ${s.external.uniqueDomains} domains`,
293
+ ``,
294
+ `### Inlink distribution`,
295
+ `- 0 inlinks (orphans): ${s.internal.distribution.zero}`,
296
+ `- 1\u20132: ${s.internal.distribution.oneToTwo}`,
297
+ `- 3\u201310: ${s.internal.distribution.threeToTen}`,
298
+ `- 11+: ${s.internal.distribution.elevenPlus}`,
299
+ ``,
300
+ `### Top 20 internal pages by inlinks`,
301
+ `| inlinks | out (int/ext) | URL |`,
302
+ `|---|---|---|`,
303
+ ...s.internal.topByInlinks.map((p) => `| ${p.inlinks} | ${p.outlinksInternal}/${p.outlinksExternal} | ${p.url} |`),
304
+ ``,
305
+ `### Top 20 external domains by links`,
306
+ `| links | nofollow | from pages | domain |`,
307
+ `|---|---|---|---|`,
308
+ ...s.external.topDomains.map((d) => `| ${d.links} | ${d.nofollow} | ${d.pages} | ${d.domain} |`)
309
+ ];
310
+ return lines.join("\n");
311
+ }
312
+
313
+ // src/api/seo-issues.ts
314
+ var THIN_WORDS = 200;
315
+ var TITLE_MAX = 60;
316
+ var TITLE_PX_MAX = 561;
317
+ var TITLE_MIN = 30;
318
+ var META_MAX = 155;
319
+ var META_MIN = 70;
320
+ var H1_MAX = 70;
321
+ var URL_MAX = 115;
322
+ function dupes(pages, key) {
323
+ const groups = /* @__PURE__ */ new Map();
324
+ for (const p of pages) {
325
+ const k = key(p);
326
+ if (k === null || k === void 0 || k === "") continue;
327
+ const ks = String(k);
328
+ if (!groups.has(ks)) groups.set(ks, []);
329
+ groups.get(ks).push(p.url);
330
+ }
331
+ return [...groups.values()].filter((g) => g.length > 1).flat();
332
+ }
333
+ function computeIssues(pages, metrics, precomputedBrokenLinkPages) {
334
+ const group = (urls) => ({ count: urls.length, urls: urls.slice(0, 200) });
335
+ const where = (fn) => pages.filter(fn).map((p) => p.url);
336
+ const extracted = (p) => p.extractionStatus !== "failed";
337
+ const ok = (p) => extracted(p) && p.status === 200;
338
+ const extractedUrls = new Set(pages.filter(extracted).map((page) => page.url));
339
+ const pathname = (u) => {
340
+ try {
341
+ return new URL(u).pathname;
342
+ } catch {
343
+ return "";
344
+ }
345
+ };
346
+ const report = {
347
+ "title.missing": group(where((p) => ok(p) && !p.title)),
348
+ "title.duplicate": group(dupes(pages.filter(ok), (p) => p.title)),
349
+ "title.tooLong": group(where((p) => (p.titleLength ?? 0) > TITLE_MAX || (p.titlePixels ?? 0) > TITLE_PX_MAX)),
350
+ "title.tooShort": group(where((p) => p.title != null && (p.titleLength ?? 0) < TITLE_MIN)),
351
+ "title.sameAsH1": group(where((p) => !!p.title && !!p.h1 && p.title.trim() === p.h1.trim())),
352
+ "meta.missing": group(where((p) => ok(p) && !p.metaDescription)),
353
+ "meta.duplicate": group(dupes(pages.filter(ok), (p) => p.metaDescription)),
354
+ "meta.tooLong": group(where((p) => (p.metaDescLength ?? 0) > META_MAX)),
355
+ "meta.tooShort": group(where((p) => p.metaDescription != null && (p.metaDescLength ?? 0) < META_MIN)),
356
+ "h1.missing": group(where((p) => ok(p) && !p.h1)),
357
+ "h1.multiple": group(where((p) => !!p.h1_2)),
358
+ "h1.duplicate": group(dupes(pages.filter(ok), (p) => p.h1)),
359
+ "h1.tooLong": group(where((p) => (p.h1?.length ?? 0) > H1_MAX)),
360
+ "h2.missing": group(where((p) => ok(p) && p.h2Count === 0)),
361
+ // A transport/extraction failure is not an SEO conclusion about the URL.
362
+ // Keep it in a dedicated crawl bucket so empty records cannot become a
363
+ // wall of false missing-title/H1/non-indexable findings.
364
+ "crawl.extractionFailed": group(where((p) => !extracted(p))),
365
+ "indexability.nonIndexable": group(where((p) => extracted(p) && !p.indexable)),
366
+ "indexability.noindex": group(where((p) => p.indexabilityReason === "noindex")),
367
+ "canonical.missing": group(where((p) => ok(p) && !p.canonicalUrl)),
368
+ "canonical.canonicalised": group(where((p) => p.indexabilityReason === "canonicalised")),
369
+ "response.broken4xx": group(where((p) => (p.status ?? 0) >= 400 && (p.status ?? 0) < 500)),
370
+ "response.error5xx": group(where((p) => (p.status ?? 0) >= 500)),
371
+ "response.redirect3xx": group(where((p) => (p.status ?? 0) >= 300 && (p.status ?? 0) < 400)),
372
+ "response.noResponse": group(where((p) => p.status === null)),
373
+ "content.thin": group(where((p) => ok(p) && p.wordCount > 0 && p.wordCount < THIN_WORDS)),
374
+ "content.exactDuplicate": group(dupes(pages.filter((p) => ok(p) && p.contentHash), (p) => p.contentHash)),
375
+ "images.missingAlt": group(where((p) => extracted(p) && p.imagesMissingAlt > 0)),
376
+ "schema.missing": group(where((p) => ok(p) && p.schemaTypes.length === 0)),
377
+ "url.tooLong": group(where((p) => p.url.length > URL_MAX)),
378
+ "url.uppercase": group(where((p) => /[A-Z]/.test(pathname(p.url)))),
379
+ "url.underscores": group(where((p) => pathname(p.url).includes("_"))),
380
+ "links.orphan": group([...metrics.values()].filter((m) => m.orphan && extractedUrls.has(m.url)).map((m) => m.url))
381
+ };
382
+ let brokenLinkPages = precomputedBrokenLinkPages;
383
+ if (!brokenLinkPages) {
384
+ const normUrl = (u) => {
385
+ try {
386
+ const x = new URL(u);
387
+ x.hash = "";
388
+ return (x.origin + x.pathname).replace(/\/+$/, "") + x.search;
389
+ } catch {
390
+ return u.replace(/#.*$/, "").replace(/\/+$/, "");
391
+ }
392
+ };
393
+ const statusByUrl = new Map(pages.map((p) => [normUrl(p.url), p.status]));
394
+ brokenLinkPages = /* @__PURE__ */ new Set();
395
+ for (const p of pages) {
396
+ for (const l of p.outlinks ?? []) {
397
+ if (!l.internal) continue;
398
+ const st = statusByUrl.get(normUrl(l.href));
399
+ if (st != null && st >= 400) brokenLinkPages.add(p.url);
400
+ }
401
+ }
402
+ }
403
+ report["links.brokenInternal"] = group([...brokenLinkPages]);
404
+ return report;
405
+ }
406
+ function renderIssueReport(siteUrl, pages, report, metrics) {
407
+ const total = pages.length;
408
+ const successful = pages.filter((page) => page.extractionStatus !== "failed").length;
409
+ const failed = total - successful;
410
+ const sev = (key) => key.startsWith("response.broken") || key.startsWith("response.error") || key.startsWith("links.broken") ? "\u{1F534}" : key.startsWith("title.missing") || key.startsWith("h1.missing") || key.startsWith("indexability") || key.startsWith("canonical.missing") ? "\u{1F7E0}" : "\u{1F7E1}";
411
+ const rows = Object.entries(report).filter(([, g]) => g.count > 0).sort((a, b) => b[1].count - a[1].count).map(([k, g]) => `| ${sev(k)} | \`${k}\` | ${g.count} | ${g.urls.slice(0, 3).join(" \xB7 ")}${g.urls.length > 3 ? " \u2026" : ""} |`);
412
+ const depths = [...metrics.values()].map((m) => m.crawlDepth).filter((d) => d != null);
413
+ const maxDepth = depths.length ? Math.max(...depths) : 0;
414
+ const extractedUrls = new Set(pages.filter((page) => page.extractionStatus !== "failed").map((page) => page.url));
415
+ const orphans = [...metrics.values()].filter((m) => m.orphan && extractedUrls.has(m.url)).length;
416
+ const topLinked = [...metrics.values()].sort((a, b) => b.inlinks - a.inlinks).slice(0, 5);
417
+ return [
418
+ `# SEO Crawl Report: ${siteUrl}`,
419
+ `**${total} pages attempted** \xB7 ${successful} extracted \xB7 ${failed} failed \xB7 max crawl depth ${maxDepth} \xB7 ${orphans} orphan page(s)`,
420
+ `
421
+ ## Issues
422
+ | | Issue | Count | Examples |
423
+ |---|-------|-------|----------|
424
+ ${rows.join("\n") || "| \u2705 | none | 0 | \u2014 |"}`,
425
+ `
426
+ ## Most-linked pages
427
+ ${topLinked.map((m) => `- ${m.inlinks} inlinks \xB7 depth ${m.crawlDepth ?? "\u2014"} \xB7 ${m.url}`).join("\n")}`,
428
+ `
429
+ _Thresholds: title ${TITLE_MAX}ch/${TITLE_PX_MAX}px, meta ${META_MAX}ch, H1 ${H1_MAX}ch, thin <${THIN_WORDS} words, URL ${URL_MAX}ch. Pixel widths are estimates._`
430
+ ].join("\n");
431
+ }
432
+
433
+ // src/api/private-artifacts.ts
434
+ import { createHash } from "crypto";
435
+ import { mkdir, readFile, writeFile } from "fs/promises";
436
+ import { createWriteStream } from "fs";
437
+ import { homedir } from "os";
438
+ import { dirname, join } from "path";
439
+ import { Transform } from "stream";
440
+ import { pipeline } from "stream/promises";
441
+ function hostedByEnvironment() {
442
+ return process.env.VERCEL === "1" || process.env.NODE_ENV === "production";
443
+ }
444
+ function policyHosted(policy) {
445
+ return policy.hosted ?? hostedByEnvironment();
446
+ }
447
+ function policyLocalBaseDir(policy) {
448
+ return policy.localBaseDir?.trim() || process.env.MCP_SCRAPER_OUTPUT_DIR?.trim() || join(homedir(), "Downloads", "mcp-scraper");
449
+ }
450
+ function normalizedPrefix(prefix) {
451
+ const normalized = prefix.replace(/^\/+/, "").replace(/\/+$/, "");
452
+ if (!normalized || normalized.includes("..") || !/^[a-zA-Z0-9/_-]+$/.test(normalized)) {
453
+ throw new Error("private artifact prefix is invalid");
454
+ }
455
+ return `${normalized}/`;
456
+ }
457
+ function requiredSafeSegment(value, field) {
458
+ const trimmed = value.trim();
459
+ if (!/^[a-zA-Z0-9_-]{1,160}$/.test(trimmed)) {
460
+ throw new Error(`${field} must contain only letters, numbers, underscores, or hyphens`);
461
+ }
462
+ return trimmed;
463
+ }
464
+ function requiredSafeArtifactKey(value) {
465
+ const trimmed = value.trim();
466
+ if (!/^[a-zA-Z0-9_.-]{1,200}$/.test(trimmed) || trimmed === "." || trimmed === "..") {
467
+ throw new Error("private artifact key is invalid");
468
+ }
469
+ return trimmed;
470
+ }
471
+ function safeDownloadFilename(value) {
472
+ const normalized = value.trim().replace(/[^a-zA-Z0-9_.-]+/g, "-").replace(/^-+|-+$/g, "");
473
+ return (normalized || "artifact").slice(0, 160);
474
+ }
475
+ function createdAtMs(value) {
476
+ const normalized = typeof value === "string" && /^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}$/.test(value) ? `${value.replace(" ", "T")}Z` : value;
477
+ const timestamp = normalized instanceof Date ? normalized.getTime() : typeof normalized === "number" ? normalized : new Date(normalized).getTime();
478
+ if (!Number.isSafeInteger(timestamp) || timestamp <= 0) throw new Error("private artifact createdAt is invalid");
479
+ return timestamp;
480
+ }
481
+ function validatePolicy(policy) {
482
+ if (!Number.isSafeInteger(policy.artifactTtlMs) || policy.artifactTtlMs <= 0) {
483
+ throw new Error("private artifact TTL must be a positive integer");
484
+ }
485
+ if (!Number.isSafeInteger(policy.downloadTtlMs) || policy.downloadTtlMs <= 0) {
486
+ throw new Error("private artifact download TTL must be a positive integer");
487
+ }
488
+ return {
489
+ prefix: normalizedPrefix(policy.prefix),
490
+ artifactTtlMs: policy.artifactTtlMs,
491
+ downloadTtlMs: policy.downloadTtlMs
492
+ };
493
+ }
494
+ function artifactTimestamp(artifactId, prefix) {
495
+ const normalized = normalizedPrefix(prefix);
496
+ if (!artifactId.startsWith(normalized)) return null;
497
+ const filename = artifactId.split("/").at(-1) ?? "";
498
+ const match = filename.match(/^(\d{13})-/);
499
+ if (!match) return null;
500
+ const timestamp = Number(match[1]);
501
+ return Number.isSafeInteger(timestamp) && timestamp > 0 ? timestamp : null;
502
+ }
503
+ function privateArtifactOwnerId(artifactId, prefix) {
504
+ const normalized = normalizedPrefix(prefix);
505
+ if (!artifactId.startsWith(normalized) || artifactId.includes("..")) return null;
506
+ const rest = artifactId.slice(normalized.length);
507
+ const segments = rest.split("/");
508
+ if (segments.length !== 2) return null;
509
+ try {
510
+ return requiredSafeSegment(segments[0] ?? "", "ownerId");
511
+ } catch {
512
+ return null;
513
+ }
514
+ }
515
+ function privateArtifactExpiresAt(artifactId, policy) {
516
+ const validated = validatePolicy(policy);
517
+ if (privateArtifactOwnerId(artifactId, validated.prefix) === null) return null;
518
+ const timestamp = artifactTimestamp(artifactId, validated.prefix);
519
+ return timestamp === null ? null : new Date(timestamp + validated.artifactTtlMs);
520
+ }
521
+ async function signedDownloadUrl(pathname, expiresAt, policy) {
522
+ const token = policy.token?.trim();
523
+ if (!token) return null;
524
+ const validUntil = Math.min(Date.now() + policy.downloadTtlMs, expiresAt.getTime());
525
+ if (validUntil <= Date.now()) return null;
526
+ const { issueSignedToken, presignUrl } = await import("@vercel/blob");
527
+ const signedToken = await issueSignedToken({
528
+ token,
529
+ pathname,
530
+ operations: ["get"],
531
+ validUntil
532
+ });
533
+ const { presignedUrl } = await presignUrl(signedToken, {
534
+ access: "private",
535
+ operation: "get",
536
+ pathname,
537
+ validUntil
538
+ });
539
+ return { url: presignedUrl, expiresAt: new Date(validUntil).toISOString() };
540
+ }
541
+ async function createPrivateArtifact(args) {
542
+ const policy = { ...args.policy, ...validatePolicy(args.policy) };
543
+ const ownerId = requiredSafeSegment(args.ownerId, "ownerId");
544
+ const artifactKey = requiredSafeArtifactKey(args.artifactKey);
545
+ const timestamp = createdAtMs(args.createdAt);
546
+ const requestedPathname = `${policy.prefix}${ownerId}/${timestamp}-${artifactKey}`;
547
+ const body = Buffer.isBuffer(args.content) ? args.content : Buffer.from(args.content);
548
+ const bytes = body.length;
549
+ const sha256 = createHash("sha256").update(body).digest("hex");
550
+ const expiresAt = new Date(timestamp + policy.artifactTtlMs);
551
+ const token = policy.token?.trim() || null;
552
+ let artifactId = requestedPathname;
553
+ if (token) {
554
+ const { put } = await import("@vercel/blob");
555
+ const stored = await put(requestedPathname, body, {
556
+ access: "private",
557
+ token,
558
+ contentType: args.contentType,
559
+ addRandomSuffix: false,
560
+ allowOverwrite: true,
561
+ cacheControlMaxAge: 60,
562
+ multipart: bytes > 100 * 1024 * 1024
563
+ });
564
+ artifactId = stored.pathname;
565
+ } else {
566
+ if (policyHosted(policy)) throw new Error("private_artifact_blob_not_configured");
567
+ const path = join(policyLocalBaseDir(policy), "blobs", requestedPathname);
568
+ await mkdir(dirname(path), { recursive: true });
569
+ await writeFile(path, body);
570
+ }
571
+ const download = await signedDownloadUrl(artifactId, expiresAt, policy);
572
+ return {
573
+ artifactId,
574
+ filename: safeDownloadFilename(args.filename),
575
+ contentType: args.contentType,
576
+ bytes,
577
+ sha256,
578
+ expiresAt: expiresAt.toISOString(),
579
+ downloadUrl: download?.url ?? null,
580
+ downloadUrlExpiresAt: download?.expiresAt ?? null
581
+ };
582
+ }
583
+ async function createPrivateArtifactFromStream(args) {
584
+ const policy = { ...args.policy, ...validatePolicy(args.policy) };
585
+ const ownerId = requiredSafeSegment(args.ownerId, "ownerId");
586
+ const artifactKey = requiredSafeArtifactKey(args.artifactKey);
587
+ const timestamp = createdAtMs(args.createdAt);
588
+ const requestedPathname = `${policy.prefix}${ownerId}/${timestamp}-${artifactKey}`;
589
+ const expiresAt = new Date(timestamp + policy.artifactTtlMs);
590
+ const token = policy.token?.trim() || null;
591
+ const hash = createHash("sha256");
592
+ let bytes = 0;
593
+ const meter = new Transform({
594
+ transform(chunk, _encoding, callback) {
595
+ bytes += chunk.length;
596
+ hash.update(chunk);
597
+ callback(null, chunk);
598
+ }
599
+ });
600
+ let artifactId = requestedPathname;
601
+ if (token) {
602
+ const { put } = await import("@vercel/blob");
603
+ const pumping = pipeline(args.content, meter);
604
+ try {
605
+ const stored = await put(requestedPathname, meter, {
606
+ access: "private",
607
+ token,
608
+ contentType: args.contentType,
609
+ addRandomSuffix: false,
610
+ allowOverwrite: true,
611
+ cacheControlMaxAge: 60,
612
+ multipart: true
613
+ });
614
+ await pumping;
615
+ artifactId = stored.pathname;
616
+ } catch (error) {
617
+ args.content.destroy(error instanceof Error ? error : new Error(String(error)));
618
+ meter.destroy(error instanceof Error ? error : new Error(String(error)));
619
+ await pumping.catch(() => void 0);
620
+ throw error;
621
+ }
622
+ } else {
623
+ if (policyHosted(policy)) throw new Error("private_artifact_blob_not_configured");
624
+ const path = join(policyLocalBaseDir(policy), "blobs", requestedPathname);
625
+ await mkdir(dirname(path), { recursive: true });
626
+ await pipeline(args.content, meter, createWriteStream(path));
627
+ }
628
+ const sha256 = hash.digest("hex");
629
+ const download = await signedDownloadUrl(artifactId, expiresAt, policy);
630
+ return {
631
+ artifactId,
632
+ filename: safeDownloadFilename(args.filename),
633
+ contentType: args.contentType,
634
+ bytes,
635
+ sha256,
636
+ expiresAt: expiresAt.toISOString(),
637
+ downloadUrl: download?.url ?? null,
638
+ downloadUrlExpiresAt: download?.expiresAt ?? null
639
+ };
640
+ }
641
+ async function renewPrivateArtifactDownload(args) {
642
+ if (privateArtifactOwnerId(args.artifactId, args.policy.prefix) !== args.ownerId) return null;
643
+ const expiresAt = privateArtifactExpiresAt(args.artifactId, args.policy);
644
+ if (!expiresAt || expiresAt.getTime() <= Date.now()) return null;
645
+ const download = await signedDownloadUrl(args.artifactId, expiresAt, args.policy);
646
+ if (!download) return null;
647
+ return {
648
+ downloadUrl: download.url,
649
+ downloadUrlExpiresAt: download.expiresAt,
650
+ expiresAt: expiresAt.toISOString()
651
+ };
652
+ }
653
+ async function streamToBuffer(stream) {
654
+ const reader = stream.getReader();
655
+ const chunks = [];
656
+ try {
657
+ for (; ; ) {
658
+ const { done, value } = await reader.read();
659
+ if (done) break;
660
+ if (value) chunks.push(Buffer.from(value));
661
+ }
662
+ } finally {
663
+ reader.releaseLock();
664
+ }
665
+ return Buffer.concat(chunks);
666
+ }
667
+ async function readPrivateArtifactWindow(args) {
668
+ const expiresAt = privateArtifactExpiresAt(args.artifactId, args.policy);
669
+ if (!expiresAt || expiresAt.getTime() <= Date.now()) return null;
670
+ if (!Number.isSafeInteger(args.offset) || args.offset < 0) throw new Error("private artifact offset is invalid");
671
+ if (!Number.isSafeInteger(args.maxBytes) || args.maxBytes <= 0) throw new Error("private artifact maxBytes is invalid");
672
+ const token = args.policy.token?.trim() || null;
673
+ let buffer;
674
+ if (token) {
675
+ const { get } = await import("@vercel/blob");
676
+ const result = await get(args.artifactId, { access: "private", token, useCache: false });
677
+ if (!result || result.statusCode !== 200) return null;
678
+ buffer = await streamToBuffer(result.stream);
679
+ } else {
680
+ if (policyHosted(args.policy)) return null;
681
+ try {
682
+ buffer = await readFile(join(policyLocalBaseDir(args.policy), "blobs", args.artifactId));
683
+ } catch {
684
+ return null;
685
+ }
686
+ }
687
+ const totalBytes = buffer.length;
688
+ const end = Math.min(totalBytes, args.offset + args.maxBytes);
689
+ return {
690
+ text: buffer.subarray(args.offset, end).toString("utf8"),
691
+ totalBytes,
692
+ nextOffset: end < totalBytes ? end : null
693
+ };
694
+ }
695
+
696
+ export {
697
+ isPrivateIpAddress,
698
+ validatePublicHttpUrl,
699
+ auditImages,
700
+ auditImageUrls,
701
+ renderImageSection,
702
+ buildLinkReport,
703
+ renderLinkReport,
704
+ computeIssues,
705
+ renderIssueReport,
706
+ privateArtifactOwnerId,
707
+ createPrivateArtifact,
708
+ createPrivateArtifactFromStream,
709
+ renewPrivateArtifactDownload,
710
+ readPrivateArtifactWindow
711
+ };
712
+ //# sourceMappingURL=chunk-345BQXZH.js.map