mcp-scraper 0.35.1 → 0.37.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -9
- package/dist/bin/api-server.cjs +25071 -19082
- package/dist/bin/api-server.cjs.map +1 -1
- package/dist/bin/api-server.js +3 -3
- package/dist/bin/mcp-scraper-cli.cjs +51 -7
- package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
- package/dist/bin/mcp-scraper-cli.js +48 -5
- package/dist/bin/mcp-scraper-cli.js.map +1 -1
- package/dist/bin/mcp-scraper-install.cjs +2 -2
- package/dist/bin/mcp-scraper-install.cjs.map +1 -1
- package/dist/bin/mcp-scraper-install.js +2 -2
- package/dist/bin/mcp-stdio-server.cjs +995 -222
- package/dist/bin/mcp-stdio-server.cjs.map +1 -1
- package/dist/bin/mcp-stdio-server.js +8 -8
- package/dist/bin/paa-harvest.cjs +125 -70
- package/dist/bin/paa-harvest.cjs.map +1 -1
- package/dist/bin/paa-harvest.js +4 -4
- package/dist/chunk-345BQXZH.js +712 -0
- package/dist/chunk-345BQXZH.js.map +1 -0
- package/dist/chunk-3LWYPAU5.js +7 -0
- package/dist/chunk-3LWYPAU5.js.map +1 -0
- package/dist/{chunk-M2S27J6Z.js → chunk-44HZLHDV.js} +10 -1
- package/dist/chunk-44HZLHDV.js.map +1 -0
- package/dist/{chunk-62DQAWPF.js → chunk-CSCD2HNS.js} +498 -43
- package/dist/chunk-CSCD2HNS.js.map +1 -0
- package/dist/{chunk-ZID3WQID.js → chunk-FQI5PFE7.js} +9 -71
- package/dist/chunk-FQI5PFE7.js.map +1 -0
- package/dist/{chunk-3HBPKR5G.js → chunk-FSAXLDB3.js} +3 -3
- package/dist/chunk-G3P3ZDB4.js +69 -0
- package/dist/chunk-G3P3ZDB4.js.map +1 -0
- package/dist/{chunk-YRGSEY5L.js → chunk-G7KAVJ3F.js} +2 -2
- package/dist/{chunk-YRGSEY5L.js.map → chunk-G7KAVJ3F.js.map} +1 -1
- package/dist/{chunk-NPMW5HUS.js → chunk-JK2FRDAP.js} +930 -244
- package/dist/chunk-JK2FRDAP.js.map +1 -0
- package/dist/{chunk-XVVNKASZ.js → chunk-LOPKN3YL.js} +118 -73
- package/dist/chunk-LOPKN3YL.js.map +1 -0
- package/dist/{chunk-4ZB3X6BQ.js → chunk-MA5JBAUZ.js} +16 -2
- package/dist/{chunk-4ZB3X6BQ.js.map → chunk-MA5JBAUZ.js.map} +1 -1
- package/dist/chunk-PUJFYJXB.js +684 -0
- package/dist/chunk-PUJFYJXB.js.map +1 -0
- package/dist/{chunk-BWXLTWF7.js → chunk-PWPUKR5U.js} +9 -5
- package/dist/chunk-PWPUKR5U.js.map +1 -0
- package/dist/chunk-Q35WZJJK.js +499 -0
- package/dist/chunk-Q35WZJJK.js.map +1 -0
- package/dist/chunk-QZXKQB7Y.js +414 -0
- package/dist/chunk-QZXKQB7Y.js.map +1 -0
- package/dist/{db-YAI5AQOI.js → db-N3YECFWR.js} +12 -2
- package/dist/{extract-bundle-ONWZVV55.js → extract-bundle-346R6MXD.js} +284 -98
- package/dist/extract-bundle-346R6MXD.js.map +1 -0
- package/dist/index.cjs +129 -70
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +11 -0
- package/dist/index.d.ts +11 -0
- package/dist/index.js +4 -4
- package/dist/location-data-repository-O2VII3ON.js +35 -0
- package/dist/{server-GKUTC73B.js → server-QKDDEBVQ.js} +7325 -3762
- package/dist/server-QKDDEBVQ.js.map +1 -0
- package/dist/site-extract-repository-I3VM6WXN.js +62 -0
- package/dist/site-extract-repository-I3VM6WXN.js.map +1 -0
- package/dist/{worker-645BZPEK.js → worker-TTFXPPDK.js} +7 -7
- package/docs/hosted-location-data.md +108 -0
- package/docs/mcp-tool-craft-lint.generated.md +6 -3
- package/docs/mcp-tool-manifest.generated.json +1447 -240
- package/docs/mcp-tool-quality-spec.md +1 -1
- package/docs/specs/connected-services-control-plane-decoupling-spec.md +1044 -0
- package/docs/specs/kernel-stealth-captcha-test-matrix.md +278 -0
- package/docs/specs/multimodal-image-memory-architecture-spec.md +1022 -0
- package/docs/specs/unified-credit-and-scheduled-execution-billing-spec.md +36 -27
- package/package.json +7 -6
- package/dist/chunk-62DQAWPF.js.map +0 -1
- package/dist/chunk-BWXLTWF7.js.map +0 -1
- package/dist/chunk-M2S27J6Z.js.map +0 -1
- package/dist/chunk-NPMW5HUS.js.map +0 -1
- package/dist/chunk-R7EETU7Z.js +0 -419
- package/dist/chunk-R7EETU7Z.js.map +0 -1
- package/dist/chunk-U44TPRST.js +0 -130
- package/dist/chunk-U44TPRST.js.map +0 -1
- package/dist/chunk-XVVNKASZ.js.map +0 -1
- package/dist/chunk-YR4LJ6AQ.js +0 -7
- package/dist/chunk-YR4LJ6AQ.js.map +0 -1
- package/dist/chunk-YV2FUEBX.js +0 -851
- package/dist/chunk-YV2FUEBX.js.map +0 -1
- package/dist/chunk-ZID3WQID.js.map +0 -1
- package/dist/extract-bundle-ONWZVV55.js.map +0 -1
- package/dist/server-GKUTC73B.js.map +0 -1
- package/dist/site-extract-repository-L6BHWVDU.js +0 -30
- /package/dist/{chunk-3HBPKR5G.js.map → chunk-FSAXLDB3.js.map} +0 -0
- /package/dist/{db-YAI5AQOI.js.map → db-N3YECFWR.js.map} +0 -0
- /package/dist/{site-extract-repository-L6BHWVDU.js.map → location-data-repository-O2VII3ON.js.map} +0 -0
- /package/dist/{worker-645BZPEK.js.map → worker-TTFXPPDK.js.map} +0 -0
package/dist/chunk-YV2FUEBX.js
DELETED
|
@@ -1,851 +0,0 @@
|
|
|
1
|
-
// src/lib/media-extractor.ts
|
|
2
|
-
import { createWriteStream, mkdirSync } from "fs";
|
|
3
|
-
import { homedir } from "os";
|
|
4
|
-
import { join, extname, basename } from "path";
|
|
5
|
-
import { pipeline } from "stream/promises";
|
|
6
|
-
import { Readable } from "stream";
|
|
7
|
-
var AD_PATTERNS = [
|
|
8
|
-
"doubleclick.net",
|
|
9
|
-
"googlesyndication.com",
|
|
10
|
-
"googletagmanager.com",
|
|
11
|
-
"google-analytics.com",
|
|
12
|
-
"googletagservices.com",
|
|
13
|
-
"adservice.google",
|
|
14
|
-
"googletag.",
|
|
15
|
-
"pagead2.googlesyndication",
|
|
16
|
-
"facebook.net/tr",
|
|
17
|
-
"connect.facebook.net",
|
|
18
|
-
"fbcdn.net/rsrc",
|
|
19
|
-
"analytics.twitter.com",
|
|
20
|
-
"static.ads-twitter.com",
|
|
21
|
-
"ads.twitter.com",
|
|
22
|
-
"t.co/i/adsct",
|
|
23
|
-
"pixel.advertising.com",
|
|
24
|
-
"hotjar.com",
|
|
25
|
-
"clarity.ms",
|
|
26
|
-
"quantserve.com",
|
|
27
|
-
"scorecardresearch.com",
|
|
28
|
-
"newrelic.com",
|
|
29
|
-
"nr-data.net",
|
|
30
|
-
"segment.io",
|
|
31
|
-
"segment.com",
|
|
32
|
-
"amplitude.com",
|
|
33
|
-
"mixpanel.com",
|
|
34
|
-
"heap.io",
|
|
35
|
-
"fullstory.com",
|
|
36
|
-
"moatads.com",
|
|
37
|
-
"criteo.com",
|
|
38
|
-
"adsrvr.org",
|
|
39
|
-
"rubiconproject.com",
|
|
40
|
-
"pubmatic.com",
|
|
41
|
-
"openx.net",
|
|
42
|
-
"appnexus.com",
|
|
43
|
-
"amazon-adsystem.com",
|
|
44
|
-
"media.net",
|
|
45
|
-
"yieldmo.com",
|
|
46
|
-
"triplelift.com",
|
|
47
|
-
"sharethrough.com",
|
|
48
|
-
"prebid.",
|
|
49
|
-
"smaato.net",
|
|
50
|
-
"indexworm.com",
|
|
51
|
-
"casalemedia.com",
|
|
52
|
-
"outbrain.com",
|
|
53
|
-
"taboola.com",
|
|
54
|
-
"revcontent.com",
|
|
55
|
-
"mgid.com",
|
|
56
|
-
"tawk.to",
|
|
57
|
-
"intercom.io",
|
|
58
|
-
"drift.com",
|
|
59
|
-
"hs-scripts.com",
|
|
60
|
-
"zopim.com",
|
|
61
|
-
"livechatinc.com",
|
|
62
|
-
"userlike.com",
|
|
63
|
-
"onetrust.com",
|
|
64
|
-
"cookielaw.org",
|
|
65
|
-
"cookieinformation.com",
|
|
66
|
-
"trustarc.com",
|
|
67
|
-
"/ads/",
|
|
68
|
-
"/ad/",
|
|
69
|
-
"/banner/",
|
|
70
|
-
"/banners/",
|
|
71
|
-
"/pixel/",
|
|
72
|
-
"/beacon/",
|
|
73
|
-
"/tracking/",
|
|
74
|
-
"/tracker/",
|
|
75
|
-
"/remarketing/",
|
|
76
|
-
"/conversion/",
|
|
77
|
-
"1x1.gif",
|
|
78
|
-
"spacer.gif",
|
|
79
|
-
"blank.gif",
|
|
80
|
-
"transparent.gif"
|
|
81
|
-
];
|
|
82
|
-
var IMAGE_EXTS = /* @__PURE__ */ new Set([".jpg", ".jpeg", ".png", ".webp", ".gif", ".avif", ".svg", ".tiff"]);
|
|
83
|
-
var VIDEO_EXTS = /* @__PURE__ */ new Set([".mp4", ".webm", ".mov", ".avi", ".m4v", ".ogv", ".mkv"]);
|
|
84
|
-
var AUDIO_EXTS = /* @__PURE__ */ new Set([".mp3", ".wav", ".ogg", ".aac", ".m4a", ".flac", ".opus"]);
|
|
85
|
-
function isAdUrl(url) {
|
|
86
|
-
const lower = url.toLowerCase();
|
|
87
|
-
return AD_PATTERNS.some((p) => lower.includes(p));
|
|
88
|
-
}
|
|
89
|
-
function isDataUri(url) {
|
|
90
|
-
return url.startsWith("data:");
|
|
91
|
-
}
|
|
92
|
-
function typeFromUrl(url) {
|
|
93
|
-
try {
|
|
94
|
-
const ext = extname(new URL(url).pathname).toLowerCase();
|
|
95
|
-
if (IMAGE_EXTS.has(ext)) return "image";
|
|
96
|
-
if (VIDEO_EXTS.has(ext)) return "video";
|
|
97
|
-
if (AUDIO_EXTS.has(ext)) return "audio";
|
|
98
|
-
} catch {
|
|
99
|
-
}
|
|
100
|
-
return null;
|
|
101
|
-
}
|
|
102
|
-
function typeFromMime(mime) {
|
|
103
|
-
const lower = mime.toLowerCase();
|
|
104
|
-
if (lower.startsWith("image/")) return "image";
|
|
105
|
-
if (lower.startsWith("video/")) return "video";
|
|
106
|
-
if (lower.startsWith("audio/")) return "audio";
|
|
107
|
-
return null;
|
|
108
|
-
}
|
|
109
|
-
function resolveUrl(raw, base) {
|
|
110
|
-
if (!raw || isDataUri(raw)) return null;
|
|
111
|
-
try {
|
|
112
|
-
return new URL(raw, base).href;
|
|
113
|
-
} catch {
|
|
114
|
-
return null;
|
|
115
|
-
}
|
|
116
|
-
}
|
|
117
|
-
function safeFilename(url, index) {
|
|
118
|
-
try {
|
|
119
|
-
const u = new URL(url);
|
|
120
|
-
const base = basename(u.pathname).replace(/[^a-zA-Z0-9._-]/g, "_").slice(0, 80);
|
|
121
|
-
return base || `asset-${index}`;
|
|
122
|
-
} catch {
|
|
123
|
-
return `asset-${index}`;
|
|
124
|
-
}
|
|
125
|
-
}
|
|
126
|
-
function extractMediaUrls(html, baseUrl) {
|
|
127
|
-
const seen = /* @__PURE__ */ new Set();
|
|
128
|
-
const urls = [];
|
|
129
|
-
const add = (raw) => {
|
|
130
|
-
const resolved = resolveUrl(raw.trim(), baseUrl);
|
|
131
|
-
if (!resolved || seen.has(resolved) || isAdUrl(resolved)) return;
|
|
132
|
-
seen.add(resolved);
|
|
133
|
-
urls.push(resolved);
|
|
134
|
-
};
|
|
135
|
-
for (const m of html.matchAll(/<img\s[^>]*>/gi)) {
|
|
136
|
-
const tag = m[0];
|
|
137
|
-
for (const attr of ["src", "data-src", "data-lazy-src", "data-original"]) {
|
|
138
|
-
const v = (tag.match(new RegExp(`\\b${attr}\\s*=\\s*["']([^"']+)["']`, "i")) ?? [])[1];
|
|
139
|
-
if (v) add(v);
|
|
140
|
-
}
|
|
141
|
-
const srcset = (tag.match(/\bsrcset\s*=\s*["']([^"']+)["']/i) ?? [])[1];
|
|
142
|
-
if (srcset) {
|
|
143
|
-
for (const part of srcset.split(",")) {
|
|
144
|
-
const u = part.trim().split(/\s+/)[0];
|
|
145
|
-
if (u) add(u);
|
|
146
|
-
}
|
|
147
|
-
}
|
|
148
|
-
}
|
|
149
|
-
for (const m of html.matchAll(/<source\s[^>]*>/gi)) {
|
|
150
|
-
const tag = m[0];
|
|
151
|
-
const src = (tag.match(/\bsrc\s*=\s*["']([^"']+)["']/i) ?? [])[1];
|
|
152
|
-
if (src) add(src);
|
|
153
|
-
const srcset = (tag.match(/\bsrcset\s*=\s*["']([^"']+)["']/i) ?? [])[1];
|
|
154
|
-
if (srcset) {
|
|
155
|
-
for (const part of srcset.split(",")) {
|
|
156
|
-
const u = part.trim().split(/\s+/)[0];
|
|
157
|
-
if (u) add(u);
|
|
158
|
-
}
|
|
159
|
-
}
|
|
160
|
-
}
|
|
161
|
-
for (const m of html.matchAll(/<video\s[^>]*>/gi)) {
|
|
162
|
-
const tag = m[0];
|
|
163
|
-
for (const attr of ["src", "poster"]) {
|
|
164
|
-
const v = (tag.match(new RegExp(`\\b${attr}\\s*=\\s*["']([^"']+)["']`, "i")) ?? [])[1];
|
|
165
|
-
if (v) add(v);
|
|
166
|
-
}
|
|
167
|
-
}
|
|
168
|
-
for (const m of html.matchAll(/<audio\s[^>]*>/gi)) {
|
|
169
|
-
const tag = m[0];
|
|
170
|
-
const v = (tag.match(/\bsrc\s*=\s*["']([^"']+)["']/i) ?? [])[1];
|
|
171
|
-
if (v) add(v);
|
|
172
|
-
}
|
|
173
|
-
for (const m of html.matchAll(/<meta\s[^>]*>/gi)) {
|
|
174
|
-
const tag = m[0];
|
|
175
|
-
const prop = (tag.match(/\b(?:property|name)\s*=\s*["']([^"']+)["']/i) ?? [])[1]?.toLowerCase();
|
|
176
|
-
if (prop === "og:image" || prop === "twitter:image") {
|
|
177
|
-
const content = (tag.match(/\bcontent\s*=\s*["']([^"']+)["']/i) ?? [])[1];
|
|
178
|
-
if (content) add(content);
|
|
179
|
-
}
|
|
180
|
-
}
|
|
181
|
-
for (const m of html.matchAll(/background(?:-image)?\s*:\s*url\(\s*["']?([^"')]+)["']?\s*\)/gi)) {
|
|
182
|
-
add(m[1]);
|
|
183
|
-
}
|
|
184
|
-
return urls;
|
|
185
|
-
}
|
|
186
|
-
async function downloadAsset(url, destDir, filename) {
|
|
187
|
-
const res = await fetch(url, {
|
|
188
|
-
headers: { "User-Agent": "Mozilla/5.0 (compatible; ThorbitBot/1.0)" },
|
|
189
|
-
signal: AbortSignal.timeout(15e3)
|
|
190
|
-
});
|
|
191
|
-
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
|
192
|
-
if (!res.body) throw new Error("Empty response body");
|
|
193
|
-
const mimeType = res.headers.get("content-type")?.split(";")[0].trim() ?? null;
|
|
194
|
-
let dest = join(destDir, filename);
|
|
195
|
-
if (mimeType && !extname(filename)) {
|
|
196
|
-
const mimeExt = {
|
|
197
|
-
"image/jpeg": ".jpg",
|
|
198
|
-
"image/png": ".png",
|
|
199
|
-
"image/webp": ".webp",
|
|
200
|
-
"image/gif": ".gif",
|
|
201
|
-
"image/svg+xml": ".svg",
|
|
202
|
-
"image/avif": ".avif",
|
|
203
|
-
"video/mp4": ".mp4",
|
|
204
|
-
"video/webm": ".webm",
|
|
205
|
-
"audio/mpeg": ".mp3",
|
|
206
|
-
"audio/ogg": ".ogg",
|
|
207
|
-
"audio/wav": ".wav"
|
|
208
|
-
};
|
|
209
|
-
const ext = mimeExt[mimeType];
|
|
210
|
-
if (ext) dest = dest + ext;
|
|
211
|
-
}
|
|
212
|
-
const writer = createWriteStream(dest);
|
|
213
|
-
await pipeline(Readable.fromWeb(res.body), writer);
|
|
214
|
-
const { statSync } = await import("fs");
|
|
215
|
-
const sizeBytes = statSync(dest).size;
|
|
216
|
-
return { savedPath: dest, sizeBytes, mimeType };
|
|
217
|
-
}
|
|
218
|
-
async function harvestPageMedia(html, pageUrl, options = {}) {
|
|
219
|
-
const types = options.types ?? ["image", "video", "audio"];
|
|
220
|
-
const typesSet = new Set(types);
|
|
221
|
-
const rawUrls = extractMediaUrls(html, pageUrl);
|
|
222
|
-
const totalFound = rawUrls.length;
|
|
223
|
-
const filteredUrls = [];
|
|
224
|
-
const kept = [];
|
|
225
|
-
for (const url of rawUrls) {
|
|
226
|
-
const type = typeFromUrl(url);
|
|
227
|
-
if (!type || !typesSet.has(type)) {
|
|
228
|
-
filteredUrls.push(url);
|
|
229
|
-
continue;
|
|
230
|
-
}
|
|
231
|
-
kept.push({ url, type });
|
|
232
|
-
}
|
|
233
|
-
if (options.outputDir === null) {
|
|
234
|
-
return {
|
|
235
|
-
pageUrl,
|
|
236
|
-
outputDir: null,
|
|
237
|
-
assets: kept.map(({ url, type }, i) => ({
|
|
238
|
-
url,
|
|
239
|
-
type,
|
|
240
|
-
mimeType: null,
|
|
241
|
-
filename: safeFilename(url, i),
|
|
242
|
-
savedPath: null,
|
|
243
|
-
sizeBytes: null
|
|
244
|
-
})),
|
|
245
|
-
filteredCount: filteredUrls.length,
|
|
246
|
-
totalFound
|
|
247
|
-
};
|
|
248
|
-
}
|
|
249
|
-
const domain = (() => {
|
|
250
|
-
try {
|
|
251
|
-
return new URL(pageUrl).hostname.replace(/^www\./, "");
|
|
252
|
-
} catch {
|
|
253
|
-
return "unknown";
|
|
254
|
-
}
|
|
255
|
-
})();
|
|
256
|
-
const stamp = (/* @__PURE__ */ new Date()).toISOString().replace(/[:.]/g, "-").slice(0, 19);
|
|
257
|
-
const outDir = options.outputDir ?? join(homedir(), "Downloads", "mcp-scraper", "media", `${stamp}-${domain}`);
|
|
258
|
-
mkdirSync(outDir, { recursive: true });
|
|
259
|
-
const assets = [];
|
|
260
|
-
await Promise.allSettled(
|
|
261
|
-
kept.map(async ({ url, type }, i) => {
|
|
262
|
-
const filename = safeFilename(url, i);
|
|
263
|
-
try {
|
|
264
|
-
const { savedPath, sizeBytes, mimeType } = await downloadAsset(url, outDir, filename);
|
|
265
|
-
const resolvedType = mimeType ? typeFromMime(mimeType) ?? type : type;
|
|
266
|
-
assets.push({ url, type: resolvedType, mimeType, filename: basename(savedPath), savedPath, sizeBytes });
|
|
267
|
-
} catch {
|
|
268
|
-
assets.push({ url, type, mimeType: null, filename, savedPath: null, sizeBytes: null });
|
|
269
|
-
}
|
|
270
|
-
})
|
|
271
|
-
);
|
|
272
|
-
assets.sort((a, b) => {
|
|
273
|
-
if (a.savedPath && !b.savedPath) return -1;
|
|
274
|
-
if (!a.savedPath && b.savedPath) return 1;
|
|
275
|
-
return a.url.localeCompare(b.url);
|
|
276
|
-
});
|
|
277
|
-
return { pageUrl, outputDir: outDir, assets, filteredCount: filteredUrls.length, totalFound };
|
|
278
|
-
}
|
|
279
|
-
|
|
280
|
-
// src/api/wayback.ts
|
|
281
|
-
import pLimit from "p-limit";
|
|
282
|
-
var WAYBACK_HOSTS = /* @__PURE__ */ new Set(["web.archive.org"]);
|
|
283
|
-
var MAX_CDX_RESPONSE_BYTES = 5 * 1024 * 1024;
|
|
284
|
-
var MAX_CDX_ROWS = 5e3;
|
|
285
|
-
var MAX_ARCHIVE_URL_CHARS = 4096;
|
|
286
|
-
var MAX_SELECTED_TIMELINE_CELLS = 1e3;
|
|
287
|
-
var TIMELINE_CDX_CONCURRENCY = 4;
|
|
288
|
-
function validTimestamp(value) {
|
|
289
|
-
return /^\d{1,14}$/.test(value);
|
|
290
|
-
}
|
|
291
|
-
function safeHttpUrl(value, base) {
|
|
292
|
-
try {
|
|
293
|
-
const parsed = base ? new URL(value, base) : new URL(value);
|
|
294
|
-
if (!["http:", "https:"].includes(parsed.protocol) || parsed.username || parsed.password) return null;
|
|
295
|
-
parsed.hash = "";
|
|
296
|
-
return parsed.href;
|
|
297
|
-
} catch {
|
|
298
|
-
return null;
|
|
299
|
-
}
|
|
300
|
-
}
|
|
301
|
-
function parseWaybackReplayUrl(value) {
|
|
302
|
-
let parsed;
|
|
303
|
-
try {
|
|
304
|
-
parsed = new URL(value);
|
|
305
|
-
} catch {
|
|
306
|
-
return null;
|
|
307
|
-
}
|
|
308
|
-
if (!WAYBACK_HOSTS.has(parsed.hostname.toLowerCase())) return null;
|
|
309
|
-
const match = parsed.pathname.match(/^\/web\/(\d{1,14})([a-z_]+)?\/(https?:\/\/.+)$/i);
|
|
310
|
-
if (!match || !validTimestamp(match[1])) return null;
|
|
311
|
-
const originalUrl = safeHttpUrl(`${match[3]}${parsed.search}`);
|
|
312
|
-
if (!originalUrl || originalUrl.length > MAX_ARCHIVE_URL_CHARS) return null;
|
|
313
|
-
const timestamp = match[1];
|
|
314
|
-
return {
|
|
315
|
-
timestamp,
|
|
316
|
-
modifier: match[2] ?? null,
|
|
317
|
-
originalUrl,
|
|
318
|
-
replayUrl: `https://web.archive.org/web/${timestamp}/${originalUrl}`,
|
|
319
|
-
rawReplayUrl: `https://web.archive.org/web/${timestamp}id_/${originalUrl}`
|
|
320
|
-
};
|
|
321
|
-
}
|
|
322
|
-
function buildWaybackReplayUrl(timestamp, originalUrl, modifier = "id_") {
|
|
323
|
-
if (!validTimestamp(timestamp)) throw new Error("Wayback timestamp must contain 1-14 digits.");
|
|
324
|
-
const safeOriginal = safeHttpUrl(originalUrl);
|
|
325
|
-
if (!safeOriginal || safeOriginal.length > MAX_ARCHIVE_URL_CHARS) throw new Error("Wayback original URL is invalid.");
|
|
326
|
-
return `https://web.archive.org/web/${timestamp}${modifier}/${safeOriginal}`;
|
|
327
|
-
}
|
|
328
|
-
function timestampMillis(value) {
|
|
329
|
-
const padded = value.padEnd(14, "0");
|
|
330
|
-
const year = Number(padded.slice(0, 4));
|
|
331
|
-
const month = Number(padded.slice(4, 6) || "1");
|
|
332
|
-
const day = Number(padded.slice(6, 8) || "1");
|
|
333
|
-
const hour = Number(padded.slice(8, 10) || "0");
|
|
334
|
-
const minute = Number(padded.slice(10, 12) || "0");
|
|
335
|
-
const second = Number(padded.slice(12, 14) || "0");
|
|
336
|
-
return Date.UTC(year, Math.max(0, month - 1), Math.max(1, day), hour, minute, second);
|
|
337
|
-
}
|
|
338
|
-
function monthAnchor(month) {
|
|
339
|
-
return `${month.replace("-", "")}15120000`;
|
|
340
|
-
}
|
|
341
|
-
function expandWaybackMonths(input) {
|
|
342
|
-
if (input.months?.length) return [...new Set(input.months)].sort();
|
|
343
|
-
if (!input.from || !input.to) return [];
|
|
344
|
-
const [fromYear, fromMonth] = input.from.split("-").map(Number);
|
|
345
|
-
const [toYear, toMonth] = input.to.split("-").map(Number);
|
|
346
|
-
const interval = input.intervalMonths ?? 1;
|
|
347
|
-
const cursor = new Date(Date.UTC(fromYear, fromMonth - 1, 1));
|
|
348
|
-
const end = new Date(Date.UTC(toYear, toMonth - 1, 1));
|
|
349
|
-
const months = [];
|
|
350
|
-
while (cursor <= end && months.length < 60) {
|
|
351
|
-
months.push(`${cursor.getUTCFullYear()}-${String(cursor.getUTCMonth() + 1).padStart(2, "0")}`);
|
|
352
|
-
cursor.setUTCMonth(cursor.getUTCMonth() + interval);
|
|
353
|
-
}
|
|
354
|
-
return months;
|
|
355
|
-
}
|
|
356
|
-
async function readBoundedText(response, maxBytes) {
|
|
357
|
-
const reader = response.body?.getReader();
|
|
358
|
-
if (!reader) return "";
|
|
359
|
-
const chunks = [];
|
|
360
|
-
let bytes = 0;
|
|
361
|
-
for (; ; ) {
|
|
362
|
-
const { done, value } = await reader.read();
|
|
363
|
-
if (done) break;
|
|
364
|
-
bytes += value.byteLength;
|
|
365
|
-
if (bytes > maxBytes) {
|
|
366
|
-
await reader.cancel();
|
|
367
|
-
throw new Error("Wayback CDX response exceeded the safe response limit.");
|
|
368
|
-
}
|
|
369
|
-
chunks.push(value);
|
|
370
|
-
}
|
|
371
|
-
return Buffer.concat(chunks).toString("utf8");
|
|
372
|
-
}
|
|
373
|
-
async function fetchCdxRaw(params) {
|
|
374
|
-
const endpoint = new URL("https://web.archive.org/cdx/search/cdx");
|
|
375
|
-
endpoint.search = params.toString();
|
|
376
|
-
let raw = "";
|
|
377
|
-
let lastError;
|
|
378
|
-
for (let attempt = 0; attempt < 2; attempt++) {
|
|
379
|
-
try {
|
|
380
|
-
const response = await fetch(endpoint, {
|
|
381
|
-
headers: { "User-Agent": "MCP-Scraper/Wayback (+https://mcpscraper.dev)" },
|
|
382
|
-
signal: AbortSignal.timeout(45e3)
|
|
383
|
-
});
|
|
384
|
-
if (!response.ok) {
|
|
385
|
-
const error = new Error(`Wayback CDX returned HTTP ${response.status}.`);
|
|
386
|
-
if (attempt === 0 && (response.status === 429 || response.status >= 500)) {
|
|
387
|
-
lastError = error;
|
|
388
|
-
continue;
|
|
389
|
-
}
|
|
390
|
-
throw error;
|
|
391
|
-
}
|
|
392
|
-
raw = await readBoundedText(response, MAX_CDX_RESPONSE_BYTES);
|
|
393
|
-
lastError = null;
|
|
394
|
-
break;
|
|
395
|
-
} catch (error) {
|
|
396
|
-
lastError = error;
|
|
397
|
-
if (attempt === 0 && (error instanceof DOMException || error instanceof TypeError)) continue;
|
|
398
|
-
throw error;
|
|
399
|
-
}
|
|
400
|
-
}
|
|
401
|
-
if (lastError) throw lastError;
|
|
402
|
-
return raw;
|
|
403
|
-
}
|
|
404
|
-
function parseCdxPage(raw) {
|
|
405
|
-
let rows;
|
|
406
|
-
try {
|
|
407
|
-
rows = JSON.parse(raw);
|
|
408
|
-
} catch {
|
|
409
|
-
throw new Error("Wayback CDX returned invalid JSON.");
|
|
410
|
-
}
|
|
411
|
-
if (!Array.isArray(rows) || rows.length < 1) return { captures: [], resumeKey: null };
|
|
412
|
-
const header = Array.isArray(rows[0]) ? rows[0].map(String) : [];
|
|
413
|
-
const timestampIndex = header.indexOf("timestamp");
|
|
414
|
-
const originalIndex = header.indexOf("original");
|
|
415
|
-
const digestIndex = header.indexOf("digest");
|
|
416
|
-
const statusIndex = header.indexOf("statuscode");
|
|
417
|
-
const mimeIndex = header.indexOf("mimetype");
|
|
418
|
-
const lengthIndex = header.indexOf("length");
|
|
419
|
-
if (timestampIndex < 0 || originalIndex < 0) throw new Error("Wayback CDX response is missing required fields.");
|
|
420
|
-
const captures = [];
|
|
421
|
-
let resumeKey = null;
|
|
422
|
-
let sawResumeSeparator = false;
|
|
423
|
-
for (const row of rows.slice(1)) {
|
|
424
|
-
if (!Array.isArray(row)) continue;
|
|
425
|
-
if (row.length === 0) {
|
|
426
|
-
sawResumeSeparator = true;
|
|
427
|
-
continue;
|
|
428
|
-
}
|
|
429
|
-
if (sawResumeSeparator && row.length === 1 && typeof row[0] === "string") {
|
|
430
|
-
resumeKey = row[0];
|
|
431
|
-
continue;
|
|
432
|
-
}
|
|
433
|
-
const timestamp = String(row[timestampIndex] ?? "");
|
|
434
|
-
const originalUrl = safeHttpUrl(String(row[originalIndex] ?? ""));
|
|
435
|
-
if (!validTimestamp(timestamp) || !originalUrl || originalUrl.length > MAX_ARCHIVE_URL_CHARS) continue;
|
|
436
|
-
const statusValue = statusIndex >= 0 ? Number(row[statusIndex]) : NaN;
|
|
437
|
-
const lengthValue = lengthIndex >= 0 ? Number(row[lengthIndex]) : NaN;
|
|
438
|
-
captures.push({
|
|
439
|
-
timestamp,
|
|
440
|
-
originalUrl,
|
|
441
|
-
digest: digestIndex >= 0 && row[digestIndex] != null ? String(row[digestIndex]) : null,
|
|
442
|
-
rawReplayUrl: buildWaybackReplayUrl(timestamp, originalUrl),
|
|
443
|
-
replayUrl: buildWaybackReplayUrl(timestamp, originalUrl, ""),
|
|
444
|
-
statusCode: Number.isInteger(statusValue) ? statusValue : null,
|
|
445
|
-
mimeType: mimeIndex >= 0 && row[mimeIndex] != null ? String(row[mimeIndex]) : null,
|
|
446
|
-
length: Number.isFinite(lengthValue) && lengthValue >= 0 ? lengthValue : null
|
|
447
|
-
});
|
|
448
|
-
}
|
|
449
|
-
return { captures, resumeKey };
|
|
450
|
-
}
|
|
451
|
-
async function queryCdxPage(baseParams, limit, resumeKey) {
|
|
452
|
-
const params = new URLSearchParams(baseParams);
|
|
453
|
-
params.set("limit", String(Math.max(1, Math.min(MAX_CDX_ROWS, Math.floor(limit)))));
|
|
454
|
-
params.set("showResumeKey", "true");
|
|
455
|
-
if (resumeKey) params.set("resumeKey", resumeKey);
|
|
456
|
-
else params.delete("resumeKey");
|
|
457
|
-
return parseCdxPage(await fetchCdxRaw(params));
|
|
458
|
-
}
|
|
459
|
-
async function queryCdx(params) {
|
|
460
|
-
return (await queryCdxPage(params, Number(params.get("limit") ?? MAX_CDX_ROWS))).captures;
|
|
461
|
-
}
|
|
462
|
-
function siteHost(value) {
|
|
463
|
-
return new URL(value).hostname.replace(/^www\./, "").toLowerCase();
|
|
464
|
-
}
|
|
465
|
-
function pageIdentity(value) {
|
|
466
|
-
const safe = safeHttpUrl(value);
|
|
467
|
-
if (!safe) return null;
|
|
468
|
-
const parsed = new URL(safe);
|
|
469
|
-
parsed.hash = "";
|
|
470
|
-
parsed.hostname = parsed.hostname.replace(/^www\./, "").toLowerCase();
|
|
471
|
-
for (const key of [...parsed.searchParams.keys()]) {
|
|
472
|
-
if (/^(utm_|fbclid$|gclid$|mc_)/i.test(key)) parsed.searchParams.delete(key);
|
|
473
|
-
}
|
|
474
|
-
return parsed.href;
|
|
475
|
-
}
|
|
476
|
-
function isLikelyContentPage(value) {
|
|
477
|
-
const parsed = new URL(value);
|
|
478
|
-
const path = parsed.pathname.toLowerCase();
|
|
479
|
-
if (/(?:^|\/)(?:wp-admin|wp-login\.php|wp-json|xmlrpc\.php)(?:\/|$)/.test(path) || /(?:^|\/)(?:author|tag|category|feed|comments|search)(?:\/|$)/.test(path) || /\.(?:css|js|json|xml|txt|ico|svg|png|jpe?g|gif|webp|avif|mp4|mp3|pdf|zip|woff2?|ttf|eot)$/i.test(path)) return false;
|
|
480
|
-
if ([...parsed.searchParams.keys()].some((key) => /^(?:redirect_to|reauth|action|rest_route|s|replytocom)$/i.test(key))) return false;
|
|
481
|
-
return true;
|
|
482
|
-
}
|
|
483
|
-
function selectClosestCaptures(captures, replay, maxPages) {
|
|
484
|
-
const expectedHost = siteHost(replay.originalUrl);
|
|
485
|
-
const targetTime = timestampMillis(replay.timestamp);
|
|
486
|
-
const byPage = /* @__PURE__ */ new Map();
|
|
487
|
-
for (const capture of captures) {
|
|
488
|
-
if (siteHost(capture.originalUrl) !== expectedHost) continue;
|
|
489
|
-
if (!isLikelyContentPage(capture.originalUrl)) continue;
|
|
490
|
-
const identity = pageIdentity(capture.originalUrl);
|
|
491
|
-
if (!identity) continue;
|
|
492
|
-
const previous = byPage.get(identity);
|
|
493
|
-
if (!previous) {
|
|
494
|
-
byPage.set(identity, capture);
|
|
495
|
-
continue;
|
|
496
|
-
}
|
|
497
|
-
const candidateDistance = Math.abs(timestampMillis(capture.timestamp) - targetTime);
|
|
498
|
-
const previousDistance = Math.abs(timestampMillis(previous.timestamp) - targetTime);
|
|
499
|
-
if (candidateDistance < previousDistance) byPage.set(identity, capture);
|
|
500
|
-
}
|
|
501
|
-
const root = new URL(replay.originalUrl).origin + "/";
|
|
502
|
-
return [...byPage.values()].sort((a, b) => {
|
|
503
|
-
if (a.originalUrl === root) return -1;
|
|
504
|
-
if (b.originalUrl === root) return 1;
|
|
505
|
-
const timeDelta = Math.abs(timestampMillis(a.timestamp) - targetTime) - Math.abs(timestampMillis(b.timestamp) - targetTime);
|
|
506
|
-
return timeDelta || a.originalUrl.localeCompare(b.originalUrl);
|
|
507
|
-
}).slice(0, maxPages);
|
|
508
|
-
}
|
|
509
|
-
function cdxParams(replay, maxRows, from, to) {
|
|
510
|
-
const original = new URL(replay.originalUrl);
|
|
511
|
-
const params = new URLSearchParams({
|
|
512
|
-
url: `${original.hostname}/*`,
|
|
513
|
-
output: "json",
|
|
514
|
-
fl: "timestamp,original,statuscode,mimetype,digest",
|
|
515
|
-
filter: "statuscode:200",
|
|
516
|
-
limit: String(Math.min(MAX_CDX_ROWS, Math.max(maxRows, 100)))
|
|
517
|
-
});
|
|
518
|
-
params.append("filter", "mimetype:text/html");
|
|
519
|
-
if (from) params.set("from", from);
|
|
520
|
-
if (to) params.set("to", to);
|
|
521
|
-
return params;
|
|
522
|
-
}
|
|
523
|
-
function exactCdxParams(originalUrl, month) {
|
|
524
|
-
const params = new URLSearchParams({
|
|
525
|
-
url: originalUrl,
|
|
526
|
-
output: "json",
|
|
527
|
-
fl: "timestamp,original,statuscode,mimetype,digest",
|
|
528
|
-
filter: "statuscode:200",
|
|
529
|
-
from: month.replace("-", ""),
|
|
530
|
-
to: month.replace("-", ""),
|
|
531
|
-
limit: "100"
|
|
532
|
-
});
|
|
533
|
-
params.append("filter", "mimetype:text/html");
|
|
534
|
-
return params;
|
|
535
|
-
}
|
|
536
|
-
function closestCapture(captures, timestamp) {
|
|
537
|
-
const target = timestampMillis(timestamp);
|
|
538
|
-
return captures.reduce((best, capture) => {
|
|
539
|
-
if (!best) return capture;
|
|
540
|
-
return Math.abs(timestampMillis(capture.timestamp) - target) < Math.abs(timestampMillis(best.timestamp) - target) ? capture : best;
|
|
541
|
-
}, null);
|
|
542
|
-
}
|
|
543
|
-
async function resolveWaybackSiteCaptures(replay, maxPages) {
|
|
544
|
-
const safeMaxPages = Math.max(1, Math.min(500, Math.floor(maxPages)));
|
|
545
|
-
const maxRows = Math.min(MAX_CDX_ROWS, safeMaxPages * 25);
|
|
546
|
-
const month = replay.timestamp.slice(0, 6);
|
|
547
|
-
const year = replay.timestamp.slice(0, 4);
|
|
548
|
-
let captures = await queryCdx(cdxParams(replay, maxRows, month, month));
|
|
549
|
-
let selected = selectClosestCaptures(captures, replay, safeMaxPages);
|
|
550
|
-
if (selected.length < Math.min(5, safeMaxPages)) {
|
|
551
|
-
captures = captures.concat(await queryCdx(cdxParams(replay, maxRows, year, year)));
|
|
552
|
-
selected = selectClosestCaptures(captures, replay, safeMaxPages);
|
|
553
|
-
}
|
|
554
|
-
if (selected.length === 0) {
|
|
555
|
-
captures = await queryCdx(cdxParams(replay, maxRows, void 0, replay.timestamp));
|
|
556
|
-
selected = selectClosestCaptures(captures, replay, safeMaxPages);
|
|
557
|
-
}
|
|
558
|
-
const exactPage = {
|
|
559
|
-
timestamp: replay.timestamp,
|
|
560
|
-
originalUrl: replay.originalUrl,
|
|
561
|
-
digest: null,
|
|
562
|
-
rawReplayUrl: replay.rawReplayUrl
|
|
563
|
-
};
|
|
564
|
-
const exactIdentity = pageIdentity(exactPage.originalUrl);
|
|
565
|
-
const withoutDuplicate = selected.filter((capture) => pageIdentity(capture.originalUrl) !== exactIdentity);
|
|
566
|
-
return [exactPage, ...withoutDuplicate].slice(0, safeMaxPages);
|
|
567
|
-
}
|
|
568
|
-
async function resolveWaybackTimelineCaptures(input) {
|
|
569
|
-
const rootUrl = safeHttpUrl(input.rootUrl);
|
|
570
|
-
if (!rootUrl) throw new Error("Wayback timeline root URL is invalid.");
|
|
571
|
-
const rootHost = siteHost(rootUrl);
|
|
572
|
-
const months = expandWaybackMonths(input.timeline);
|
|
573
|
-
if (months.length === 0) throw new Error("At least one Wayback month is required.");
|
|
574
|
-
const maxPagesPerSnapshot = Math.max(1, Math.min(500, Math.floor(input.maxPagesPerSnapshot)));
|
|
575
|
-
const maxCaptures = Math.max(1, Math.min(1e4, Math.floor(input.maxCaptures ?? 1e4)));
|
|
576
|
-
const requestedUrls = input.timeline.urls ? [...new Set(input.timeline.urls.map((url) => safeHttpUrl(url)).filter((url) => Boolean(url)))].slice(0, maxPagesPerSnapshot) : null;
|
|
577
|
-
if (requestedUrls?.some((url) => siteHost(url) !== rootHost)) {
|
|
578
|
-
throw new Error("Every selected Wayback URL must belong to the same site as the root URL.");
|
|
579
|
-
}
|
|
580
|
-
if (requestedUrls && requestedUrls.length * months.length > MAX_SELECTED_TIMELINE_CELLS) {
|
|
581
|
-
throw new Error(`Selected-page Wayback timelines are limited to ${MAX_SELECTED_TIMELINE_CELLS} page-month cells per job.`);
|
|
582
|
-
}
|
|
583
|
-
const captures = [];
|
|
584
|
-
const missing = [];
|
|
585
|
-
if (requestedUrls) {
|
|
586
|
-
const limit = pLimit(TIMELINE_CDX_CONCURRENCY);
|
|
587
|
-
const cells = months.flatMap((requestedMonth) => requestedUrls.map((originalUrl) => ({ requestedMonth, originalUrl }))).slice(0, maxCaptures);
|
|
588
|
-
const resolved = await Promise.all(cells.map((cell) => limit(async () => {
|
|
589
|
-
const rows = await queryCdx(exactCdxParams(cell.originalUrl, cell.requestedMonth));
|
|
590
|
-
const capture = closestCapture(
|
|
591
|
-
rows.filter((row) => pageIdentity(row.originalUrl) === pageIdentity(cell.originalUrl)),
|
|
592
|
-
monthAnchor(cell.requestedMonth)
|
|
593
|
-
);
|
|
594
|
-
return { ...cell, capture };
|
|
595
|
-
})));
|
|
596
|
-
for (const cell of resolved) {
|
|
597
|
-
if (!cell.capture) {
|
|
598
|
-
missing.push({ requestedMonth: cell.requestedMonth, originalUrl: cell.originalUrl });
|
|
599
|
-
continue;
|
|
600
|
-
}
|
|
601
|
-
captures.push({ ...cell.capture, requestedMonth: cell.requestedMonth });
|
|
602
|
-
}
|
|
603
|
-
} else {
|
|
604
|
-
for (const requestedMonth of months) {
|
|
605
|
-
const replay = {
|
|
606
|
-
timestamp: monthAnchor(requestedMonth),
|
|
607
|
-
modifier: null,
|
|
608
|
-
originalUrl: rootUrl,
|
|
609
|
-
replayUrl: buildWaybackReplayUrl(monthAnchor(requestedMonth), rootUrl, ""),
|
|
610
|
-
rawReplayUrl: buildWaybackReplayUrl(monthAnchor(requestedMonth), rootUrl)
|
|
611
|
-
};
|
|
612
|
-
const month = requestedMonth.replace("-", "");
|
|
613
|
-
const rows = await queryCdx(cdxParams(replay, Math.min(MAX_CDX_ROWS, maxPagesPerSnapshot * 25), month, month));
|
|
614
|
-
const selected = selectClosestCaptures(rows, replay, maxPagesPerSnapshot);
|
|
615
|
-
if (selected.length === 0) {
|
|
616
|
-
missing.push({ requestedMonth, originalUrl: rootUrl });
|
|
617
|
-
continue;
|
|
618
|
-
}
|
|
619
|
-
captures.push(...selected.map((capture) => ({ ...capture, requestedMonth })));
|
|
620
|
-
if (captures.length >= maxCaptures) break;
|
|
621
|
-
}
|
|
622
|
-
}
|
|
623
|
-
return {
|
|
624
|
-
months,
|
|
625
|
-
captures: captures.slice(0, maxCaptures),
|
|
626
|
-
requestedUrls,
|
|
627
|
-
missing
|
|
628
|
-
};
|
|
629
|
-
}
|
|
630
|
-
function normalizeCdxRangeValue(value) {
|
|
631
|
-
if (!value) return null;
|
|
632
|
-
const normalized = value.replace(/-/g, "");
|
|
633
|
-
if (!/^(?:\d{4}|\d{6}|\d{8}|\d{14})$/.test(normalized)) {
|
|
634
|
-
throw new Error("Wayback date ranges must use YYYY, YYYY-MM, YYYY-MM-DD, or a 14-digit timestamp.");
|
|
635
|
-
}
|
|
636
|
-
return normalized;
|
|
637
|
-
}
|
|
638
|
-
function monthRange(from, to) {
|
|
639
|
-
if (!from || !to || from.length < 6 || to.length < 6) return [];
|
|
640
|
-
const start = new Date(Date.UTC(Number(from.slice(0, 4)), Number(from.slice(4, 6)) - 1, 1));
|
|
641
|
-
const end = new Date(Date.UTC(Number(to.slice(0, 4)), Number(to.slice(4, 6)) - 1, 1));
|
|
642
|
-
const months = [];
|
|
643
|
-
while (start <= end && months.length < 600) {
|
|
644
|
-
months.push(`${start.getUTCFullYear()}-${String(start.getUTCMonth() + 1).padStart(2, "0")}`);
|
|
645
|
-
start.setUTCMonth(start.getUTCMonth() + 1);
|
|
646
|
-
}
|
|
647
|
-
return months;
|
|
648
|
-
}
|
|
649
|
-
function inventoryCdxParams(input) {
|
|
650
|
-
const parsed = new URL(input.url);
|
|
651
|
-
const target = input.scope === "host" || input.scope === "domain" ? parsed.hostname : parsed.href;
|
|
652
|
-
const params = new URLSearchParams({
|
|
653
|
-
url: target,
|
|
654
|
-
output: "json",
|
|
655
|
-
fl: "timestamp,original,statuscode,mimetype,digest,length"
|
|
656
|
-
});
|
|
657
|
-
if (input.scope !== "exact") params.set("matchType", input.scope);
|
|
658
|
-
if (input.from) params.set("from", input.from);
|
|
659
|
-
if (input.to) params.set("to", input.to);
|
|
660
|
-
if (input.successfulHtmlOnly) {
|
|
661
|
-
params.append("filter", "statuscode:200");
|
|
662
|
-
params.append("filter", "mimetype:text/html");
|
|
663
|
-
}
|
|
664
|
-
return params;
|
|
665
|
-
}
|
|
666
|
-
async function inventoryWaybackSnapshots(input) {
|
|
667
|
-
const startedAt = Date.now();
|
|
668
|
-
const replay = parseWaybackReplayUrl(input.url);
|
|
669
|
-
const rootUrl = safeHttpUrl(replay?.originalUrl ?? input.url);
|
|
670
|
-
if (!rootUrl) throw new Error("Wayback inventory URL is invalid.");
|
|
671
|
-
const scope = input.scope ?? "exact";
|
|
672
|
-
const from = normalizeCdxRangeValue(input.from);
|
|
673
|
-
const to = normalizeCdxRangeValue(input.to);
|
|
674
|
-
if (from && to && from > to) throw new Error("Wayback inventory from must not be after to.");
|
|
675
|
-
const successfulHtmlOnly = input.successfulHtmlOnly !== false;
|
|
676
|
-
const maxCaptures = Math.max(1, Math.min(1e5, Math.floor(input.maxCaptures ?? 1e4)));
|
|
677
|
-
const maxCaptureRows = Math.max(0, Math.min(1e3, Math.floor(input.maxCaptureRows ?? 500)));
|
|
678
|
-
const selectedUrls = input.urls?.length ? [...new Set(input.urls.map((url) => safeHttpUrl(url)).filter((url) => Boolean(url)))] : null;
|
|
679
|
-
const rootHost = siteHost(rootUrl);
|
|
680
|
-
if (selectedUrls?.some((url) => siteHost(url) !== rootHost)) {
|
|
681
|
-
throw new Error("Every selected Wayback inventory URL must belong to the same site as url.");
|
|
682
|
-
}
|
|
683
|
-
const targets = selectedUrls ?? [rootUrl];
|
|
684
|
-
const effectiveScope = selectedUrls ? "exact" : scope;
|
|
685
|
-
const monthly = /* @__PURE__ */ new Map();
|
|
686
|
-
const yearly = /* @__PURE__ */ new Map();
|
|
687
|
-
const digests = /* @__PURE__ */ new Set();
|
|
688
|
-
const perUrlState = /* @__PURE__ */ new Map();
|
|
689
|
-
const captureRows = [];
|
|
690
|
-
let totalCaptures = 0;
|
|
691
|
-
let queryPages = 0;
|
|
692
|
-
let firstCapture = null;
|
|
693
|
-
let lastCapture = null;
|
|
694
|
-
let truncated = false;
|
|
695
|
-
targetLoop:
|
|
696
|
-
for (let targetIndex = 0; targetIndex < targets.length; targetIndex++) {
|
|
697
|
-
const target = targets[targetIndex];
|
|
698
|
-
const params = inventoryCdxParams({
|
|
699
|
-
url: target,
|
|
700
|
-
scope: effectiveScope,
|
|
701
|
-
from,
|
|
702
|
-
to,
|
|
703
|
-
successfulHtmlOnly
|
|
704
|
-
});
|
|
705
|
-
let resumeKey;
|
|
706
|
-
do {
|
|
707
|
-
const remaining = maxCaptures - totalCaptures;
|
|
708
|
-
if (remaining <= 0) {
|
|
709
|
-
truncated = true;
|
|
710
|
-
break targetLoop;
|
|
711
|
-
}
|
|
712
|
-
const page = await queryCdxPage(params, Math.min(MAX_CDX_ROWS, remaining), resumeKey);
|
|
713
|
-
queryPages++;
|
|
714
|
-
for (const capture of page.captures) {
|
|
715
|
-
totalCaptures++;
|
|
716
|
-
const month = `${capture.timestamp.slice(0, 4)}-${capture.timestamp.slice(4, 6)}`;
|
|
717
|
-
const year = capture.timestamp.slice(0, 4);
|
|
718
|
-
monthly.set(month, (monthly.get(month) ?? 0) + 1);
|
|
719
|
-
yearly.set(year, (yearly.get(year) ?? 0) + 1);
|
|
720
|
-
if (capture.digest) digests.add(capture.digest);
|
|
721
|
-
const state = perUrlState.get(capture.originalUrl) ?? {
|
|
722
|
-
captures: 0,
|
|
723
|
-
digests: /* @__PURE__ */ new Set(),
|
|
724
|
-
firstTimestamp: capture.timestamp,
|
|
725
|
-
lastTimestamp: capture.timestamp
|
|
726
|
-
};
|
|
727
|
-
state.captures++;
|
|
728
|
-
if (capture.digest) state.digests.add(capture.digest);
|
|
729
|
-
if (capture.timestamp < state.firstTimestamp) state.firstTimestamp = capture.timestamp;
|
|
730
|
-
if (capture.timestamp > state.lastTimestamp) state.lastTimestamp = capture.timestamp;
|
|
731
|
-
perUrlState.set(capture.originalUrl, state);
|
|
732
|
-
if (!firstCapture || capture.timestamp < firstCapture.timestamp) firstCapture = capture;
|
|
733
|
-
if (!lastCapture || capture.timestamp > lastCapture.timestamp) lastCapture = capture;
|
|
734
|
-
if (input.includeCaptures === true && captureRows.length < maxCaptureRows) captureRows.push(capture);
|
|
735
|
-
}
|
|
736
|
-
resumeKey = page.resumeKey ?? void 0;
|
|
737
|
-
if (resumeKey && totalCaptures >= maxCaptures) {
|
|
738
|
-
truncated = true;
|
|
739
|
-
break targetLoop;
|
|
740
|
-
}
|
|
741
|
-
} while (resumeKey);
|
|
742
|
-
if (targetIndex < targets.length - 1 && totalCaptures >= maxCaptures) {
|
|
743
|
-
truncated = true;
|
|
744
|
-
break;
|
|
745
|
-
}
|
|
746
|
-
}
|
|
747
|
-
const monthlyCounts = [...monthly.entries()].sort(([a], [b]) => a.localeCompare(b)).map(([month, captures]) => ({ month, captures }));
|
|
748
|
-
const yearlyCounts = [...yearly.entries()].sort(([a], [b]) => a.localeCompare(b)).map(([year, captures]) => ({ year, captures }));
|
|
749
|
-
const allPerUrl = [...perUrlState.entries()].map(([url, state]) => ({
|
|
750
|
-
url,
|
|
751
|
-
captures: state.captures,
|
|
752
|
-
uniqueDigests: state.digests.size,
|
|
753
|
-
firstTimestamp: state.firstTimestamp,
|
|
754
|
-
lastTimestamp: state.lastTimestamp
|
|
755
|
-
})).sort((a, b) => b.captures - a.captures || a.url.localeCompare(b.url));
|
|
756
|
-
const availableMonths = new Set(monthly.keys());
|
|
757
|
-
return {
|
|
758
|
-
url: rootUrl,
|
|
759
|
-
scope: effectiveScope,
|
|
760
|
-
selectedUrls,
|
|
761
|
-
from,
|
|
762
|
-
to,
|
|
763
|
-
successfulHtmlOnly,
|
|
764
|
-
totalCaptures,
|
|
765
|
-
countType: truncated ? "lower_bound" : "exact",
|
|
766
|
-
complete: !truncated,
|
|
767
|
-
truncated,
|
|
768
|
-
maxCaptures,
|
|
769
|
-
queryPages,
|
|
770
|
-
uniqueUrls: perUrlState.size,
|
|
771
|
-
uniqueDigests: digests.size,
|
|
772
|
-
firstCapture,
|
|
773
|
-
lastCapture,
|
|
774
|
-
monthlyCounts,
|
|
775
|
-
yearlyCounts,
|
|
776
|
-
missingMonths: monthRange(from, to).filter((month) => !availableMonths.has(month)),
|
|
777
|
-
perUrl: allPerUrl.slice(0, 500),
|
|
778
|
-
perUrlTruncatedCount: Math.max(0, allPerUrl.length - 500),
|
|
779
|
-
captures: captureRows,
|
|
780
|
-
captureRowsTruncatedCount: input.includeCaptures === true ? Math.max(0, totalCaptures - captureRows.length) : totalCaptures,
|
|
781
|
-
durationMs: Date.now() - startedAt
|
|
782
|
-
};
|
|
783
|
-
}
|
|
784
|
-
function firstJsonLdImage(value) {
|
|
785
|
-
if (typeof value === "string") return value;
|
|
786
|
-
if (Array.isArray(value)) {
|
|
787
|
-
for (const item of value) {
|
|
788
|
-
const found = firstJsonLdImage(item);
|
|
789
|
-
if (found) return found;
|
|
790
|
-
}
|
|
791
|
-
return null;
|
|
792
|
-
}
|
|
793
|
-
if (!value || typeof value !== "object") return null;
|
|
794
|
-
const record = value;
|
|
795
|
-
for (const key of ["url", "contentUrl"]) {
|
|
796
|
-
if (typeof record[key] === "string") return record[key];
|
|
797
|
-
}
|
|
798
|
-
for (const key of ["image", "thumbnailUrl", "@graph"]) {
|
|
799
|
-
const found = firstJsonLdImage(record[key]);
|
|
800
|
-
if (found) return found;
|
|
801
|
-
}
|
|
802
|
-
return null;
|
|
803
|
-
}
|
|
804
|
-
function firstContentImage(html) {
|
|
805
|
-
for (const match of html.matchAll(/<img\b[^>]*>/gi)) {
|
|
806
|
-
const tag = match[0];
|
|
807
|
-
const attr = (name) => tag.match(new RegExp(`\\b${name}\\s*=\\s*(?:"([^"]+)"|'([^']+)'|([^\\s>]+))`, "i"))?.slice(1).find(Boolean) ?? null;
|
|
808
|
-
const candidate = attr("data-src") ?? attr("src") ?? attr("data-lazy-src");
|
|
809
|
-
if (!candidate || /^(?:data:|javascript:)/i.test(candidate)) continue;
|
|
810
|
-
const descriptor = `${candidate} ${attr("alt") ?? ""} ${attr("class") ?? ""}`.toLowerCase();
|
|
811
|
-
if (/(?:logo|icon|avatar|sprite|pixel|tracking|spacer)/.test(descriptor)) continue;
|
|
812
|
-
const width = Number(attr("width") ?? 0);
|
|
813
|
-
const height = Number(attr("height") ?? 0);
|
|
814
|
-
if (width > 0 && width < 200 || height > 0 && height < 120) continue;
|
|
815
|
-
return candidate;
|
|
816
|
-
}
|
|
817
|
-
return null;
|
|
818
|
-
}
|
|
819
|
-
function resolveFeaturedImage(input) {
|
|
820
|
-
const replay = parseWaybackReplayUrl(input.pageUrl);
|
|
821
|
-
const baseUrl = replay?.originalUrl ?? input.pageUrl;
|
|
822
|
-
const candidates = [
|
|
823
|
-
{ value: input.meta["og:image"] ?? input.meta["og:image:url"] ?? input.meta["og:image:secure_url"], source: "og:image" },
|
|
824
|
-
{ value: input.meta["twitter:image"] ?? input.meta["twitter:image:src"], source: "twitter:image" },
|
|
825
|
-
{ value: firstJsonLdImage(input.schema), source: "json-ld" },
|
|
826
|
-
{ value: firstContentImage(input.html), source: "content-image" }
|
|
827
|
-
];
|
|
828
|
-
for (const candidate of candidates) {
|
|
829
|
-
if (!candidate.value) continue;
|
|
830
|
-
const url = safeHttpUrl(candidate.value, baseUrl);
|
|
831
|
-
if (!url) continue;
|
|
832
|
-
return {
|
|
833
|
-
url,
|
|
834
|
-
archivedUrl: replay ? buildWaybackReplayUrl(replay.timestamp, url, "im_") : null,
|
|
835
|
-
source: candidate.source
|
|
836
|
-
};
|
|
837
|
-
}
|
|
838
|
-
return null;
|
|
839
|
-
}
|
|
840
|
-
|
|
841
|
-
export {
|
|
842
|
-
downloadAsset,
|
|
843
|
-
harvestPageMedia,
|
|
844
|
-
parseWaybackReplayUrl,
|
|
845
|
-
expandWaybackMonths,
|
|
846
|
-
resolveWaybackSiteCaptures,
|
|
847
|
-
resolveWaybackTimelineCaptures,
|
|
848
|
-
inventoryWaybackSnapshots,
|
|
849
|
-
resolveFeaturedImage
|
|
850
|
-
};
|
|
851
|
-
//# sourceMappingURL=chunk-YV2FUEBX.js.map
|