mcp-scraper 0.88.2 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -15
- package/README.md +6 -5
- package/THIRD_PARTY_NOTICES.html +203 -0
- package/dist/analytics-repository-2BMT5JNE.js +1 -0
- package/dist/bin/api-server.js +2 -41
- package/dist/bin/mcp-scraper-cli.js +39 -756
- package/dist/bin/mcp-scraper-core.js +1 -60
- package/dist/bin/mcp-scraper-install.js +2 -25
- package/dist/bin/mcp-stdio-server.js +1 -19
- package/dist/bin/paa-harvest.js +1 -41
- package/dist/chunk-3GP5CYZX.js +1 -0
- package/dist/chunk-4AI7DOS7.js +59 -0
- package/dist/chunk-4FROKQJN.js +1 -0
- package/dist/chunk-7YGVI5J4.js +21710 -0
- package/dist/chunk-CCYWSJNG.js +16 -0
- package/dist/chunk-CFI6CXIV.js +182 -0
- package/dist/chunk-E2WRWV3A.js +1 -0
- package/dist/chunk-GMWKPIYX.js +1172 -0
- package/dist/chunk-HDPYG3XV.js +102 -0
- package/dist/chunk-HE45FFBU.js +1 -0
- package/dist/chunk-HUV2WTRW.js +1 -0
- package/dist/chunk-KJQXUZ4Y.js +4 -0
- package/dist/chunk-L4CGLFPU.js +4 -0
- package/dist/chunk-M22MM4N4.js +84 -0
- package/dist/chunk-MASR22K4.js +73 -0
- package/dist/chunk-MZN4U5BL.js +1 -0
- package/dist/chunk-PUHFVA7P.js +1280 -0
- package/dist/chunk-QPWPR5XG.js +10 -0
- package/dist/chunk-TMB56NCA.js +1 -0
- package/dist/chunk-TXENITMS.js +20 -0
- package/dist/chunk-W2BVJ7S2.js +13 -0
- package/dist/chunk-WO3N5FH2.js +5 -0
- package/dist/chunk-WSCGYRWA.js +2595 -0
- package/dist/chunk-X54CQLK2.js +1 -0
- package/dist/chunk-XLWNEVUZ.js +27 -0
- package/dist/chunk-XPZVJIZ2.js +100 -0
- package/dist/chunk-YQZGZBB4.js +1 -0
- package/dist/chunk-Z2QGQJS2.js +1 -0
- package/dist/db-F2MX63GI.js +1 -0
- package/dist/extract-bundle-SNUIHM3J.js +26 -0
- package/dist/gmail-service-BZ3H75XC.js +1 -0
- package/dist/index.cjs +21750 -6045
- package/dist/index.d.cts +14 -14
- package/dist/index.d.ts +14 -14
- package/dist/index.js +18 -315
- package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
- package/dist/location-data-repository-OTWHWMV6.js +1 -0
- package/dist/server-RFR2A5UJ.js +7303 -0
- package/dist/site-extract-repository-SE776XDC.js +1 -0
- package/dist/worker-XUDSM3AL.js +1 -0
- package/package.json +17 -124
- package/dist/analytics-repository-GGJJCVVP.js +0 -194
- package/dist/chunk-4QMUF6XM.js +0 -1013
- package/dist/chunk-6HAV7LCE.js +0 -265
- package/dist/chunk-ABF2CGOZ.js +0 -113
- package/dist/chunk-C5Z4OFKW.js +0 -404
- package/dist/chunk-DNM65UCK.js +0 -299
- package/dist/chunk-EQGTEHLZ.js +0 -592
- package/dist/chunk-F5GQJWZU.js +0 -732
- package/dist/chunk-GGZEC22A.js +0 -215
- package/dist/chunk-GXBZXWXB.js +0 -184
- package/dist/chunk-IHXAXYIS.js +0 -843
- package/dist/chunk-K3Z5AQYE.js +0 -683
- package/dist/chunk-K45K75OF.js +0 -6
- package/dist/chunk-LFW2FRPJ.js +0 -224
- package/dist/chunk-MZDNZQWT.js +0 -2078
- package/dist/chunk-NVUKO5NN.js +0 -256
- package/dist/chunk-OM7HVEJ3.js +0 -26
- package/dist/chunk-OPQIGAFB.js +0 -286
- package/dist/chunk-OZJMVCDK.js +0 -16
- package/dist/chunk-P7FWOMU7.js +0 -505
- package/dist/chunk-PGJQDMC2.js +0 -383
- package/dist/chunk-PKZS6SHW.js +0 -33139
- package/dist/chunk-RJ7JVYKU.js +0 -68
- package/dist/chunk-S24LFPL7.js +0 -5262
- package/dist/chunk-T3MZISOF.js +0 -240
- package/dist/chunk-UZPTGUDV.js +0 -1915
- package/dist/chunk-X623GTBV.js +0 -8290
- package/dist/chunk-YXNDOQXN.js +0 -4018
- package/dist/db-Z34LPZNR.js +0 -284
- package/dist/extract-bundle-565SBZCR.js +0 -1003
- package/dist/gmail-service-E6ALS7JG.js +0 -25
- package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
- package/dist/location-data-repository-WPRG62GE.js +0 -34
- package/dist/server-SQZ3A7SY.js +0 -86606
- package/dist/site-extract-repository-VYFZASPU.js +0 -69
- package/dist/worker-LDCAULWL.js +0 -146
package/dist/chunk-PGJQDMC2.js
DELETED
|
@@ -1,383 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
validatePublicHttpUrl
|
|
3
|
-
} from "./chunk-GGZEC22A.js";
|
|
4
|
-
|
|
5
|
-
// src/api/image-audit.ts
|
|
6
|
-
var OVER_BYTES = 100 * 1024;
|
|
7
|
-
var MODERN = /* @__PURE__ */ new Set(["webp", "avif", "svg", "svg+xml"]);
|
|
8
|
-
function formatBytes(n) {
|
|
9
|
-
if (n == null) return null;
|
|
10
|
-
if (n < 1024) return `${n} B`;
|
|
11
|
-
if (n < 1024 * 1024) return `${(n / 1024).toFixed(1)} KB`;
|
|
12
|
-
return `${(n / (1024 * 1024)).toFixed(2)} MB`;
|
|
13
|
-
}
|
|
14
|
-
var decodeEntities = (u) => u.replace(/&/g, "&").replace(/�?38;/g, "&").replace(/&/gi, "&").trim();
|
|
15
|
-
var dedupKey = (u) => u.replace(/^https?:\/\//i, "//").replace(/\/$/, "");
|
|
16
|
-
var formatOf = (ct, url) => {
|
|
17
|
-
if (ct) {
|
|
18
|
-
const m = ct.split(";")[0].trim().toLowerCase();
|
|
19
|
-
if (m.startsWith("image/")) return m.replace("image/", "");
|
|
20
|
-
}
|
|
21
|
-
return (url.split("?")[0].match(/\.([a-z0-9]+)$/i)?.[1] || "unknown").toLowerCase();
|
|
22
|
-
};
|
|
23
|
-
function collectUrls(pages) {
|
|
24
|
-
const byKey = /* @__PURE__ */ new Map();
|
|
25
|
-
for (const p of pages) for (const raw of p.imageLinks || []) {
|
|
26
|
-
const u = decodeEntities(raw);
|
|
27
|
-
if (!/^https?:\/\//i.test(u)) continue;
|
|
28
|
-
if (/[{}]|%7[bd]/i.test(u)) continue;
|
|
29
|
-
const k = dedupKey(u);
|
|
30
|
-
const prev = byKey.get(k);
|
|
31
|
-
if (!prev || /^https:/i.test(u) && /^http:/i.test(prev)) byKey.set(k, u);
|
|
32
|
-
}
|
|
33
|
-
return [...byKey.values()];
|
|
34
|
-
}
|
|
35
|
-
async function sizeAndType(url, timeoutMs) {
|
|
36
|
-
const ctrl = new AbortController();
|
|
37
|
-
const t = setTimeout(() => ctrl.abort(), timeoutMs);
|
|
38
|
-
const read = (res) => ({
|
|
39
|
-
len: res.headers.get("content-length"),
|
|
40
|
-
cr: res.headers.get("content-range"),
|
|
41
|
-
ct: res.headers.get("content-type"),
|
|
42
|
-
status: res.status
|
|
43
|
-
});
|
|
44
|
-
const safeFetch = async (method) => {
|
|
45
|
-
let target = url;
|
|
46
|
-
for (let redirects = 0; redirects <= 5; redirects++) {
|
|
47
|
-
const checked = await validatePublicHttpUrl(target, { field: "image URL" });
|
|
48
|
-
if (checked.error || !checked.parsed) throw new Error(checked.error ?? "Image URL was rejected");
|
|
49
|
-
const response = await fetch(checked.parsed.href, {
|
|
50
|
-
method,
|
|
51
|
-
...method === "GET" ? { headers: { Range: "bytes=0-0" } } : {},
|
|
52
|
-
redirect: "manual",
|
|
53
|
-
signal: ctrl.signal
|
|
54
|
-
});
|
|
55
|
-
if (response.status >= 300 && response.status < 400) {
|
|
56
|
-
const location = response.headers.get("location");
|
|
57
|
-
if (!location) throw new Error(`HTTP ${response.status} redirect did not include Location`);
|
|
58
|
-
target = new URL(location, checked.parsed.href).href;
|
|
59
|
-
continue;
|
|
60
|
-
}
|
|
61
|
-
return response;
|
|
62
|
-
}
|
|
63
|
-
throw new Error("Image request exceeded five redirects");
|
|
64
|
-
};
|
|
65
|
-
try {
|
|
66
|
-
let bytes = null;
|
|
67
|
-
let ct = null;
|
|
68
|
-
let status = null;
|
|
69
|
-
try {
|
|
70
|
-
const h = read(await safeFetch("HEAD"));
|
|
71
|
-
status = h.status;
|
|
72
|
-
ct = h.ct;
|
|
73
|
-
if (h.len) bytes = Number(h.len);
|
|
74
|
-
} catch {
|
|
75
|
-
}
|
|
76
|
-
if (bytes == null) {
|
|
77
|
-
const g = await safeFetch("GET");
|
|
78
|
-
const r = read(g);
|
|
79
|
-
status = status ?? r.status;
|
|
80
|
-
ct = ct ?? r.ct;
|
|
81
|
-
if (r.cr && r.cr.includes("/")) {
|
|
82
|
-
const tail = r.cr.split("/")[1];
|
|
83
|
-
if (tail && tail !== "*") bytes = Number(tail);
|
|
84
|
-
} else if (r.len) bytes = Number(r.len);
|
|
85
|
-
try {
|
|
86
|
-
await g.body?.cancel();
|
|
87
|
-
} catch {
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
return { url, status, bytes: Number.isFinite(bytes) ? bytes : null, contentType: ct };
|
|
91
|
-
} catch (e) {
|
|
92
|
-
return { url, status: null, bytes: null, contentType: null, error: String(e.message || e) };
|
|
93
|
-
} finally {
|
|
94
|
-
clearTimeout(t);
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
async function pool(items, n, fn) {
|
|
98
|
-
const out = new Array(items.length);
|
|
99
|
-
let i = 0;
|
|
100
|
-
await Promise.all(Array.from({ length: Math.min(n, items.length) }, async () => {
|
|
101
|
-
while (i < items.length) {
|
|
102
|
-
const idx = i++;
|
|
103
|
-
out[idx] = await fn(items[idx]);
|
|
104
|
-
}
|
|
105
|
-
}));
|
|
106
|
-
return out;
|
|
107
|
-
}
|
|
108
|
-
async function auditImages(pages, opts = {}) {
|
|
109
|
-
return auditImageUrls(collectUrls(pages), opts);
|
|
110
|
-
}
|
|
111
|
-
async function auditImageUrls(inputUrls, opts = {}) {
|
|
112
|
-
const concurrency = opts.concurrency ?? 12;
|
|
113
|
-
const timeoutMs = opts.timeoutMs ?? 12e3;
|
|
114
|
-
const max = opts.max ?? 5e3;
|
|
115
|
-
const urls = collectUrls([{ imageLinks: inputUrls }]).slice(0, max);
|
|
116
|
-
const heads = await pool(urls, concurrency, (u) => sizeAndType(u, timeoutMs));
|
|
117
|
-
const rows = heads.map((r) => {
|
|
118
|
-
const format = formatOf(r.contentType, r.url);
|
|
119
|
-
return {
|
|
120
|
-
...r,
|
|
121
|
-
size: formatBytes(r.bytes),
|
|
122
|
-
format,
|
|
123
|
-
over100kb: r.bytes != null && r.bytes > OVER_BYTES,
|
|
124
|
-
legacyFormat: format !== "unknown" && !MODERN.has(format)
|
|
125
|
-
};
|
|
126
|
-
});
|
|
127
|
-
const sized = rows.filter((r) => r.bytes != null);
|
|
128
|
-
const totalBytes = sized.reduce((a, r) => a + r.bytes, 0);
|
|
129
|
-
const formatCounts = {};
|
|
130
|
-
for (const r of rows) formatCounts[r.format] = (formatCounts[r.format] || 0) + 1;
|
|
131
|
-
return {
|
|
132
|
-
rows: rows.sort((a, b) => (b.bytes ?? 0) - (a.bytes ?? 0)),
|
|
133
|
-
summary: {
|
|
134
|
-
unique: rows.length,
|
|
135
|
-
sized: sized.length,
|
|
136
|
-
totalBytes,
|
|
137
|
-
totalSize: formatBytes(totalBytes) ?? "0 B",
|
|
138
|
-
avgSize: formatBytes(Math.round(totalBytes / (sized.length || 1))) ?? "0 B",
|
|
139
|
-
over100kb: rows.filter((r) => r.over100kb).length,
|
|
140
|
-
legacyFormat: rows.filter((r) => r.legacyFormat).length,
|
|
141
|
-
formatCounts,
|
|
142
|
-
...opts.sampleTruncated ? { sampleTruncated: true, sampleCap: max } : {}
|
|
143
|
-
}
|
|
144
|
-
};
|
|
145
|
-
}
|
|
146
|
-
function renderImageSection(audit) {
|
|
147
|
-
const s = audit.summary;
|
|
148
|
-
const fmt = Object.entries(s.formatCounts).sort((a, b) => b[1] - a[1]).map(([k, v]) => `${k}:${v}`).join(" ");
|
|
149
|
-
const heaviest = audit.rows.filter((r) => r.over100kb).slice(0, 15);
|
|
150
|
-
const lines = [
|
|
151
|
-
`## Images`,
|
|
152
|
-
`**${s.unique} unique images** \xB7 ${s.totalSize} total \xB7 ${s.avgSize} avg \xB7 ${s.over100kb} over 100 KB \xB7 ${s.legacyFormat} legacy format`,
|
|
153
|
-
`Formats: ${fmt}`
|
|
154
|
-
];
|
|
155
|
-
if (heaviest.length) {
|
|
156
|
-
lines.push("", "| Size | Format | URL |", "|------|--------|-----|");
|
|
157
|
-
for (const r of heaviest) lines.push(`| ${r.size} | ${r.format} | ${r.url} |`);
|
|
158
|
-
}
|
|
159
|
-
return lines.join("\n");
|
|
160
|
-
}
|
|
161
|
-
|
|
162
|
-
// src/api/seo-link-report.ts
|
|
163
|
-
function registrableDomain(host) {
|
|
164
|
-
const h = host.replace(/^www\./, "").toLowerCase();
|
|
165
|
-
const parts = h.split(".");
|
|
166
|
-
return parts.length <= 2 ? h : parts.slice(-2).join(".");
|
|
167
|
-
}
|
|
168
|
-
function buildLinkReport(edges, metrics, siteUrl) {
|
|
169
|
-
let siteReg = "";
|
|
170
|
-
try {
|
|
171
|
-
siteReg = registrableDomain(new URL(siteUrl).hostname);
|
|
172
|
-
} catch {
|
|
173
|
-
siteReg = "";
|
|
174
|
-
}
|
|
175
|
-
const internalEdges = edges.filter((e) => e.internal);
|
|
176
|
-
const pages = metrics.length;
|
|
177
|
-
const orphans = metrics.filter((m) => m.orphan).length;
|
|
178
|
-
const brokenInternal = internalEdges.filter((e) => e.targetStatus != null && e.targetStatus >= 400).length;
|
|
179
|
-
const sumInlinks = metrics.reduce((a, m) => a + m.inlinks, 0);
|
|
180
|
-
const sumOutlinks = metrics.reduce((a, m) => a + m.outlinksInternal + m.outlinksExternal, 0);
|
|
181
|
-
const distribution = { zero: 0, oneToTwo: 0, threeToTen: 0, elevenPlus: 0 };
|
|
182
|
-
for (const m of metrics) {
|
|
183
|
-
if (m.inlinks === 0) distribution.zero++;
|
|
184
|
-
else if (m.inlinks <= 2) distribution.oneToTwo++;
|
|
185
|
-
else if (m.inlinks <= 10) distribution.threeToTen++;
|
|
186
|
-
else distribution.elevenPlus++;
|
|
187
|
-
}
|
|
188
|
-
const topByInlinks = [...metrics].sort((a, b) => b.inlinks - a.inlinks).slice(0, 20).map((m) => ({ url: m.url, inlinks: m.inlinks, outlinksInternal: m.outlinksInternal, outlinksExternal: m.outlinksExternal }));
|
|
189
|
-
const domMap = /* @__PURE__ */ new Map();
|
|
190
|
-
let externalTotal = 0;
|
|
191
|
-
for (const e of edges) {
|
|
192
|
-
let domain;
|
|
193
|
-
try {
|
|
194
|
-
domain = registrableDomain(new URL(e.to).hostname);
|
|
195
|
-
} catch {
|
|
196
|
-
continue;
|
|
197
|
-
}
|
|
198
|
-
if (!domain || domain === siteReg) continue;
|
|
199
|
-
externalTotal++;
|
|
200
|
-
const d = domMap.get(domain) ?? { links: 0, nofollow: 0, pages: /* @__PURE__ */ new Set() };
|
|
201
|
-
d.links++;
|
|
202
|
-
if (e.nofollow) d.nofollow++;
|
|
203
|
-
d.pages.add(e.from);
|
|
204
|
-
domMap.set(domain, d);
|
|
205
|
-
}
|
|
206
|
-
const externalDomains = [...domMap.entries()].map(([domain, d]) => ({ domain, links: d.links, nofollow: d.nofollow, pages: d.pages.size })).sort((a, b) => b.links - a.links);
|
|
207
|
-
const round = (n) => Math.round(n * 10) / 10;
|
|
208
|
-
return {
|
|
209
|
-
summary: {
|
|
210
|
-
internal: {
|
|
211
|
-
totalLinks: internalEdges.length,
|
|
212
|
-
pages,
|
|
213
|
-
orphans,
|
|
214
|
-
brokenInternal,
|
|
215
|
-
avgInlinks: pages ? round(sumInlinks / pages) : 0,
|
|
216
|
-
avgOutlinks: pages ? round(sumOutlinks / pages) : 0,
|
|
217
|
-
distribution,
|
|
218
|
-
topByInlinks
|
|
219
|
-
},
|
|
220
|
-
external: {
|
|
221
|
-
totalLinks: externalTotal,
|
|
222
|
-
uniqueDomains: externalDomains.length,
|
|
223
|
-
topDomains: externalDomains.slice(0, 20)
|
|
224
|
-
}
|
|
225
|
-
},
|
|
226
|
-
externalDomains
|
|
227
|
-
};
|
|
228
|
-
}
|
|
229
|
-
function renderLinkReport(r) {
|
|
230
|
-
const s = r.summary;
|
|
231
|
-
const lines = [
|
|
232
|
-
`## Link analysis`,
|
|
233
|
-
`**Internal:** ${s.internal.totalLinks} links \xB7 ${s.internal.pages} pages \xB7 avg ${s.internal.avgInlinks} inlinks/page \xB7 ${s.internal.orphans} orphans \xB7 ${s.internal.brokenInternal} broken`,
|
|
234
|
-
`**External:** ${s.external.totalLinks} links to ${s.external.uniqueDomains} domains`,
|
|
235
|
-
``,
|
|
236
|
-
`### Inlink distribution`,
|
|
237
|
-
`- 0 inlinks (orphans): ${s.internal.distribution.zero}`,
|
|
238
|
-
`- 1\u20132: ${s.internal.distribution.oneToTwo}`,
|
|
239
|
-
`- 3\u201310: ${s.internal.distribution.threeToTen}`,
|
|
240
|
-
`- 11+: ${s.internal.distribution.elevenPlus}`,
|
|
241
|
-
``,
|
|
242
|
-
`### Top 20 internal pages by inlinks`,
|
|
243
|
-
`| inlinks | out (int/ext) | URL |`,
|
|
244
|
-
`|---|---|---|`,
|
|
245
|
-
...s.internal.topByInlinks.map((p) => `| ${p.inlinks} | ${p.outlinksInternal}/${p.outlinksExternal} | ${p.url} |`),
|
|
246
|
-
``,
|
|
247
|
-
`### Top 20 external domains by links`,
|
|
248
|
-
`| links | nofollow | from pages | domain |`,
|
|
249
|
-
`|---|---|---|---|`,
|
|
250
|
-
...s.external.topDomains.map((d) => `| ${d.links} | ${d.nofollow} | ${d.pages} | ${d.domain} |`)
|
|
251
|
-
];
|
|
252
|
-
return lines.join("\n");
|
|
253
|
-
}
|
|
254
|
-
|
|
255
|
-
// src/api/seo-issues.ts
|
|
256
|
-
var THIN_WORDS = 200;
|
|
257
|
-
var TITLE_MAX = 60;
|
|
258
|
-
var TITLE_PX_MAX = 561;
|
|
259
|
-
var TITLE_MIN = 30;
|
|
260
|
-
var META_MAX = 155;
|
|
261
|
-
var META_MIN = 70;
|
|
262
|
-
var H1_MAX = 70;
|
|
263
|
-
var URL_MAX = 115;
|
|
264
|
-
function dupes(pages, key) {
|
|
265
|
-
const groups = /* @__PURE__ */ new Map();
|
|
266
|
-
for (const p of pages) {
|
|
267
|
-
const k = key(p);
|
|
268
|
-
if (k === null || k === void 0 || k === "") continue;
|
|
269
|
-
const ks = String(k);
|
|
270
|
-
if (!groups.has(ks)) groups.set(ks, []);
|
|
271
|
-
groups.get(ks).push(p.url);
|
|
272
|
-
}
|
|
273
|
-
return [...groups.values()].filter((g) => g.length > 1).flat();
|
|
274
|
-
}
|
|
275
|
-
function computeIssues(pages, metrics, precomputedBrokenLinkPages) {
|
|
276
|
-
const group = (urls) => ({ count: urls.length, urls: urls.slice(0, 200) });
|
|
277
|
-
const where = (fn) => pages.filter(fn).map((p) => p.url);
|
|
278
|
-
const extracted = (p) => p.extractionStatus !== "failed";
|
|
279
|
-
const ok = (p) => extracted(p) && p.status === 200;
|
|
280
|
-
const extractedUrls = new Set(pages.filter(extracted).map((page) => page.url));
|
|
281
|
-
const pathname = (u) => {
|
|
282
|
-
try {
|
|
283
|
-
return new URL(u).pathname;
|
|
284
|
-
} catch {
|
|
285
|
-
return "";
|
|
286
|
-
}
|
|
287
|
-
};
|
|
288
|
-
const report = {
|
|
289
|
-
"title.missing": group(where((p) => ok(p) && !p.title)),
|
|
290
|
-
"title.duplicate": group(dupes(pages.filter(ok), (p) => p.title)),
|
|
291
|
-
"title.tooLong": group(where((p) => (p.titleLength ?? 0) > TITLE_MAX || (p.titlePixels ?? 0) > TITLE_PX_MAX)),
|
|
292
|
-
"title.tooShort": group(where((p) => p.title != null && (p.titleLength ?? 0) < TITLE_MIN)),
|
|
293
|
-
"title.sameAsH1": group(where((p) => !!p.title && !!p.h1 && p.title.trim() === p.h1.trim())),
|
|
294
|
-
"meta.missing": group(where((p) => ok(p) && !p.metaDescription)),
|
|
295
|
-
"meta.duplicate": group(dupes(pages.filter(ok), (p) => p.metaDescription)),
|
|
296
|
-
"meta.tooLong": group(where((p) => (p.metaDescLength ?? 0) > META_MAX)),
|
|
297
|
-
"meta.tooShort": group(where((p) => p.metaDescription != null && (p.metaDescLength ?? 0) < META_MIN)),
|
|
298
|
-
"h1.missing": group(where((p) => ok(p) && !p.h1)),
|
|
299
|
-
"h1.multiple": group(where((p) => !!p.h1_2)),
|
|
300
|
-
"h1.duplicate": group(dupes(pages.filter(ok), (p) => p.h1)),
|
|
301
|
-
"h1.tooLong": group(where((p) => (p.h1?.length ?? 0) > H1_MAX)),
|
|
302
|
-
"h2.missing": group(where((p) => ok(p) && p.h2Count === 0)),
|
|
303
|
-
// A transport/extraction failure is not an SEO conclusion about the URL.
|
|
304
|
-
// Keep it in a dedicated crawl bucket so empty records cannot become a
|
|
305
|
-
// wall of false missing-title/H1/non-indexable findings.
|
|
306
|
-
"crawl.extractionFailed": group(where((p) => !extracted(p))),
|
|
307
|
-
"indexability.nonIndexable": group(where((p) => extracted(p) && !p.indexable)),
|
|
308
|
-
"indexability.noindex": group(where((p) => p.indexabilityReason === "noindex")),
|
|
309
|
-
"canonical.missing": group(where((p) => ok(p) && !p.canonicalUrl)),
|
|
310
|
-
"canonical.canonicalised": group(where((p) => p.indexabilityReason === "canonicalised")),
|
|
311
|
-
"response.broken4xx": group(where((p) => (p.status ?? 0) >= 400 && (p.status ?? 0) < 500)),
|
|
312
|
-
"response.error5xx": group(where((p) => (p.status ?? 0) >= 500)),
|
|
313
|
-
"response.redirect3xx": group(where((p) => (p.status ?? 0) >= 300 && (p.status ?? 0) < 400)),
|
|
314
|
-
"response.noResponse": group(where((p) => p.status === null)),
|
|
315
|
-
"content.thin": group(where((p) => ok(p) && p.wordCount > 0 && p.wordCount < THIN_WORDS)),
|
|
316
|
-
"content.exactDuplicate": group(dupes(pages.filter((p) => ok(p) && p.contentHash), (p) => p.contentHash)),
|
|
317
|
-
"images.missingAlt": group(where((p) => extracted(p) && p.imagesMissingAlt > 0)),
|
|
318
|
-
"schema.missing": group(where((p) => ok(p) && p.schemaTypes.length === 0)),
|
|
319
|
-
"url.tooLong": group(where((p) => p.url.length > URL_MAX)),
|
|
320
|
-
"url.uppercase": group(where((p) => /[A-Z]/.test(pathname(p.url)))),
|
|
321
|
-
"url.underscores": group(where((p) => pathname(p.url).includes("_"))),
|
|
322
|
-
"links.orphan": group([...metrics.values()].filter((m) => m.orphan && extractedUrls.has(m.url)).map((m) => m.url))
|
|
323
|
-
};
|
|
324
|
-
let brokenLinkPages = precomputedBrokenLinkPages;
|
|
325
|
-
if (!brokenLinkPages) {
|
|
326
|
-
const normUrl = (u) => {
|
|
327
|
-
try {
|
|
328
|
-
const x = new URL(u);
|
|
329
|
-
x.hash = "";
|
|
330
|
-
return (x.origin + x.pathname).replace(/\/+$/, "") + x.search;
|
|
331
|
-
} catch {
|
|
332
|
-
return u.replace(/#.*$/, "").replace(/\/+$/, "");
|
|
333
|
-
}
|
|
334
|
-
};
|
|
335
|
-
const statusByUrl = new Map(pages.map((p) => [normUrl(p.url), p.status]));
|
|
336
|
-
brokenLinkPages = /* @__PURE__ */ new Set();
|
|
337
|
-
for (const p of pages) {
|
|
338
|
-
for (const l of p.outlinks ?? []) {
|
|
339
|
-
if (!l.internal) continue;
|
|
340
|
-
const st = statusByUrl.get(normUrl(l.href));
|
|
341
|
-
if (st != null && st >= 400) brokenLinkPages.add(p.url);
|
|
342
|
-
}
|
|
343
|
-
}
|
|
344
|
-
}
|
|
345
|
-
report["links.brokenInternal"] = group([...brokenLinkPages]);
|
|
346
|
-
return report;
|
|
347
|
-
}
|
|
348
|
-
function renderIssueReport(siteUrl, pages, report, metrics) {
|
|
349
|
-
const total = pages.length;
|
|
350
|
-
const successful = pages.filter((page) => page.extractionStatus !== "failed").length;
|
|
351
|
-
const failed = total - successful;
|
|
352
|
-
const sev = (key) => key.startsWith("response.broken") || key.startsWith("response.error") || key.startsWith("links.broken") ? "\u{1F534}" : key.startsWith("title.missing") || key.startsWith("h1.missing") || key.startsWith("indexability") || key.startsWith("canonical.missing") ? "\u{1F7E0}" : "\u{1F7E1}";
|
|
353
|
-
const rows = Object.entries(report).filter(([, g]) => g.count > 0).sort((a, b) => b[1].count - a[1].count).map(([k, g]) => `| ${sev(k)} | \`${k}\` | ${g.count} | ${g.urls.slice(0, 3).join(" \xB7 ")}${g.urls.length > 3 ? " \u2026" : ""} |`);
|
|
354
|
-
const depths = [...metrics.values()].map((m) => m.crawlDepth).filter((d) => d != null);
|
|
355
|
-
const maxDepth = depths.length ? Math.max(...depths) : 0;
|
|
356
|
-
const extractedUrls = new Set(pages.filter((page) => page.extractionStatus !== "failed").map((page) => page.url));
|
|
357
|
-
const orphans = [...metrics.values()].filter((m) => m.orphan && extractedUrls.has(m.url)).length;
|
|
358
|
-
const topLinked = [...metrics.values()].sort((a, b) => b.inlinks - a.inlinks).slice(0, 5);
|
|
359
|
-
return [
|
|
360
|
-
`# SEO Crawl Report: ${siteUrl}`,
|
|
361
|
-
`**${total} pages attempted** \xB7 ${successful} extracted \xB7 ${failed} failed \xB7 max crawl depth ${maxDepth} \xB7 ${orphans} orphan page(s)`,
|
|
362
|
-
`
|
|
363
|
-
## Issues
|
|
364
|
-
| | Issue | Count | Examples |
|
|
365
|
-
|---|-------|-------|----------|
|
|
366
|
-
${rows.join("\n") || "| \u2705 | none | 0 | \u2014 |"}`,
|
|
367
|
-
`
|
|
368
|
-
## Most-linked pages
|
|
369
|
-
${topLinked.map((m) => `- ${m.inlinks} inlinks \xB7 depth ${m.crawlDepth ?? "\u2014"} \xB7 ${m.url}`).join("\n")}`,
|
|
370
|
-
`
|
|
371
|
-
_Thresholds: title ${TITLE_MAX}ch/${TITLE_PX_MAX}px, meta ${META_MAX}ch, H1 ${H1_MAX}ch, thin <${THIN_WORDS} words, URL ${URL_MAX}ch. Pixel widths are estimates._`
|
|
372
|
-
].join("\n");
|
|
373
|
-
}
|
|
374
|
-
|
|
375
|
-
export {
|
|
376
|
-
auditImages,
|
|
377
|
-
auditImageUrls,
|
|
378
|
-
renderImageSection,
|
|
379
|
-
buildLinkReport,
|
|
380
|
-
renderLinkReport,
|
|
381
|
-
computeIssues,
|
|
382
|
-
renderIssueReport
|
|
383
|
-
};
|