mcp-scraper 0.88.2 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/CHANGELOG.md +36 -15
  2. package/README.md +6 -5
  3. package/THIRD_PARTY_NOTICES.html +203 -0
  4. package/dist/analytics-repository-2BMT5JNE.js +1 -0
  5. package/dist/bin/api-server.js +2 -41
  6. package/dist/bin/mcp-scraper-cli.js +39 -756
  7. package/dist/bin/mcp-scraper-core.js +1 -60
  8. package/dist/bin/mcp-scraper-install.js +2 -25
  9. package/dist/bin/mcp-stdio-server.js +1 -19
  10. package/dist/bin/paa-harvest.js +1 -41
  11. package/dist/chunk-3GP5CYZX.js +1 -0
  12. package/dist/chunk-4AI7DOS7.js +59 -0
  13. package/dist/chunk-4FROKQJN.js +1 -0
  14. package/dist/chunk-7YGVI5J4.js +21710 -0
  15. package/dist/chunk-CCYWSJNG.js +16 -0
  16. package/dist/chunk-CFI6CXIV.js +182 -0
  17. package/dist/chunk-E2WRWV3A.js +1 -0
  18. package/dist/chunk-GMWKPIYX.js +1172 -0
  19. package/dist/chunk-HDPYG3XV.js +102 -0
  20. package/dist/chunk-HE45FFBU.js +1 -0
  21. package/dist/chunk-HUV2WTRW.js +1 -0
  22. package/dist/chunk-KJQXUZ4Y.js +4 -0
  23. package/dist/chunk-L4CGLFPU.js +4 -0
  24. package/dist/chunk-M22MM4N4.js +84 -0
  25. package/dist/chunk-MASR22K4.js +73 -0
  26. package/dist/chunk-MZN4U5BL.js +1 -0
  27. package/dist/chunk-PUHFVA7P.js +1280 -0
  28. package/dist/chunk-QPWPR5XG.js +10 -0
  29. package/dist/chunk-TMB56NCA.js +1 -0
  30. package/dist/chunk-TXENITMS.js +20 -0
  31. package/dist/chunk-W2BVJ7S2.js +13 -0
  32. package/dist/chunk-WO3N5FH2.js +5 -0
  33. package/dist/chunk-WSCGYRWA.js +2595 -0
  34. package/dist/chunk-X54CQLK2.js +1 -0
  35. package/dist/chunk-XLWNEVUZ.js +27 -0
  36. package/dist/chunk-XPZVJIZ2.js +100 -0
  37. package/dist/chunk-YQZGZBB4.js +1 -0
  38. package/dist/chunk-Z2QGQJS2.js +1 -0
  39. package/dist/db-F2MX63GI.js +1 -0
  40. package/dist/extract-bundle-SNUIHM3J.js +26 -0
  41. package/dist/gmail-service-BZ3H75XC.js +1 -0
  42. package/dist/index.cjs +21750 -6045
  43. package/dist/index.d.cts +14 -14
  44. package/dist/index.d.ts +14 -14
  45. package/dist/index.js +18 -315
  46. package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
  47. package/dist/location-data-repository-OTWHWMV6.js +1 -0
  48. package/dist/server-RFR2A5UJ.js +7303 -0
  49. package/dist/site-extract-repository-SE776XDC.js +1 -0
  50. package/dist/worker-XUDSM3AL.js +1 -0
  51. package/package.json +17 -124
  52. package/dist/analytics-repository-GGJJCVVP.js +0 -194
  53. package/dist/chunk-4QMUF6XM.js +0 -1013
  54. package/dist/chunk-6HAV7LCE.js +0 -265
  55. package/dist/chunk-ABF2CGOZ.js +0 -113
  56. package/dist/chunk-C5Z4OFKW.js +0 -404
  57. package/dist/chunk-DNM65UCK.js +0 -299
  58. package/dist/chunk-EQGTEHLZ.js +0 -592
  59. package/dist/chunk-F5GQJWZU.js +0 -732
  60. package/dist/chunk-GGZEC22A.js +0 -215
  61. package/dist/chunk-GXBZXWXB.js +0 -184
  62. package/dist/chunk-IHXAXYIS.js +0 -843
  63. package/dist/chunk-K3Z5AQYE.js +0 -683
  64. package/dist/chunk-K45K75OF.js +0 -6
  65. package/dist/chunk-LFW2FRPJ.js +0 -224
  66. package/dist/chunk-MZDNZQWT.js +0 -2078
  67. package/dist/chunk-NVUKO5NN.js +0 -256
  68. package/dist/chunk-OM7HVEJ3.js +0 -26
  69. package/dist/chunk-OPQIGAFB.js +0 -286
  70. package/dist/chunk-OZJMVCDK.js +0 -16
  71. package/dist/chunk-P7FWOMU7.js +0 -505
  72. package/dist/chunk-PGJQDMC2.js +0 -383
  73. package/dist/chunk-PKZS6SHW.js +0 -33139
  74. package/dist/chunk-RJ7JVYKU.js +0 -68
  75. package/dist/chunk-S24LFPL7.js +0 -5262
  76. package/dist/chunk-T3MZISOF.js +0 -240
  77. package/dist/chunk-UZPTGUDV.js +0 -1915
  78. package/dist/chunk-X623GTBV.js +0 -8290
  79. package/dist/chunk-YXNDOQXN.js +0 -4018
  80. package/dist/db-Z34LPZNR.js +0 -284
  81. package/dist/extract-bundle-565SBZCR.js +0 -1003
  82. package/dist/gmail-service-E6ALS7JG.js +0 -25
  83. package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
  84. package/dist/location-data-repository-WPRG62GE.js +0 -34
  85. package/dist/server-SQZ3A7SY.js +0 -86606
  86. package/dist/site-extract-repository-VYFZASPU.js +0 -69
  87. package/dist/worker-LDCAULWL.js +0 -146
@@ -1,732 +0,0 @@
1
- import {
2
- isVendorUnavailableError
3
- } from "./chunk-OZJMVCDK.js";
4
- import {
5
- loadHtmlDocument,
6
- validatePublicHttpUrl
7
- } from "./chunk-GGZEC22A.js";
8
- import {
9
- buildPublicErrorEnvelope,
10
- getDb
11
- } from "./chunk-YXNDOQXN.js";
12
-
13
- // src/lib/media-extractor.ts
14
- import { createWriteStream, mkdirSync, rmSync } from "fs";
15
- import { homedir } from "os";
16
- import { join, extname, basename } from "path";
17
- import { pipeline } from "stream/promises";
18
- import { Readable, Transform } from "stream";
19
- var AD_PATTERNS = [
20
- "doubleclick.net",
21
- "googlesyndication.com",
22
- "googletagmanager.com",
23
- "google-analytics.com",
24
- "googletagservices.com",
25
- "adservice.google",
26
- "googletag.",
27
- "pagead2.googlesyndication",
28
- "facebook.net/tr",
29
- "connect.facebook.net",
30
- "fbcdn.net/rsrc",
31
- "analytics.twitter.com",
32
- "static.ads-twitter.com",
33
- "ads.twitter.com",
34
- "t.co/i/adsct",
35
- "pixel.advertising.com",
36
- "hotjar.com",
37
- "clarity.ms",
38
- "quantserve.com",
39
- "scorecardresearch.com",
40
- "newrelic.com",
41
- "nr-data.net",
42
- "segment.io",
43
- "segment.com",
44
- "amplitude.com",
45
- "mixpanel.com",
46
- "heap.io",
47
- "fullstory.com",
48
- "moatads.com",
49
- "criteo.com",
50
- "adsrvr.org",
51
- "rubiconproject.com",
52
- "pubmatic.com",
53
- "openx.net",
54
- "appnexus.com",
55
- "amazon-adsystem.com",
56
- "media.net",
57
- "yieldmo.com",
58
- "triplelift.com",
59
- "sharethrough.com",
60
- "prebid.",
61
- "smaato.net",
62
- "indexworm.com",
63
- "casalemedia.com",
64
- "outbrain.com",
65
- "taboola.com",
66
- "revcontent.com",
67
- "mgid.com",
68
- "tawk.to",
69
- "intercom.io",
70
- "drift.com",
71
- "hs-scripts.com",
72
- "zopim.com",
73
- "livechatinc.com",
74
- "userlike.com",
75
- "onetrust.com",
76
- "cookielaw.org",
77
- "cookieinformation.com",
78
- "trustarc.com",
79
- "/ads/",
80
- "/ad/",
81
- "/banner/",
82
- "/banners/",
83
- "/pixel/",
84
- "/beacon/",
85
- "/tracking/",
86
- "/tracker/",
87
- "/remarketing/",
88
- "/conversion/",
89
- "1x1.gif",
90
- "spacer.gif",
91
- "blank.gif",
92
- "transparent.gif"
93
- ];
94
- var IMAGE_EXTS = /* @__PURE__ */ new Set([".jpg", ".jpeg", ".png", ".webp", ".gif", ".avif", ".svg", ".tiff"]);
95
- var VIDEO_EXTS = /* @__PURE__ */ new Set([".mp4", ".webm", ".mov", ".avi", ".m4v", ".ogv", ".mkv"]);
96
- var AUDIO_EXTS = /* @__PURE__ */ new Set([".mp3", ".wav", ".ogg", ".aac", ".m4a", ".flac", ".opus"]);
97
- function isAdUrl(url) {
98
- const lower = url.toLowerCase();
99
- return AD_PATTERNS.some((p) => lower.includes(p));
100
- }
101
- function isDataUri(url) {
102
- return url.startsWith("data:");
103
- }
104
- function typeFromUrl(url) {
105
- try {
106
- const ext = extname(new URL(url).pathname).toLowerCase();
107
- if (IMAGE_EXTS.has(ext)) return "image";
108
- if (VIDEO_EXTS.has(ext)) return "video";
109
- if (AUDIO_EXTS.has(ext)) return "audio";
110
- } catch {
111
- }
112
- return null;
113
- }
114
- function typeFromMime(mime) {
115
- const lower = mime.toLowerCase();
116
- if (lower.startsWith("image/")) return "image";
117
- if (lower.startsWith("video/")) return "video";
118
- if (lower.startsWith("audio/")) return "audio";
119
- return null;
120
- }
121
- function resolveUrl(raw, base) {
122
- if (!raw || isDataUri(raw)) return null;
123
- try {
124
- return new URL(raw, base).href;
125
- } catch {
126
- return null;
127
- }
128
- }
129
- function safeFilename(url, index) {
130
- try {
131
- const u = new URL(url);
132
- const base = basename(u.pathname).replace(/[^a-zA-Z0-9._-]/g, "_").slice(0, 80);
133
- return base || `asset-${index}`;
134
- } catch {
135
- return `asset-${index}`;
136
- }
137
- }
138
- function boundedText(value, max) {
139
- const normalized = value?.replace(/&(?:amp|#0?38|#x26);/gi, "&").replace(/\s+/g, " ").trim();
140
- return normalized ? normalized.slice(0, max) : null;
141
- }
142
- function dimensionsFromUrl(rawUrl) {
143
- const match = rawUrl.match(/[-_/](\d{2,5})x(\d{2,5})(?=[._/?#-]|$)/i);
144
- return match ? { width: Number(match[1]), height: Number(match[2]) } : null;
145
- }
146
- function mediaFamilyKey(rawUrl) {
147
- try {
148
- const url = new URL(rawUrl);
149
- url.hash = "";
150
- url.pathname = url.pathname.replace(/-(?:\d{2,5}x\d{2,5}|scaled|e\d{6,})(?=\.[a-z0-9]{2,6}$)/i, "");
151
- for (const key of ["w", "width", "h", "height", "resize"]) url.searchParams.delete(key);
152
- url.searchParams.sort();
153
- return url.href;
154
- } catch {
155
- return rawUrl;
156
- }
157
- }
158
- function candidateScore(candidate) {
159
- const area = (candidate.width ?? 0) * (candidate.height ?? 0);
160
- const methodBoost = candidate.discoveryMethods.some((method) => method.includes("lightbox")) ? 1e12 : 0;
161
- const originalBoost = /-(?:\d{2,5}x\d{2,5}|scaled|e\d{6,})(?=\.[a-z0-9]{2,6}(?:[?#]|$))/i.test(candidate.url) ? 0 : 1e11;
162
- return methodBoost + originalBoost + area;
163
- }
164
- function mergeCandidate(target, incoming) {
165
- const existing = target.get(incoming.url);
166
- if (!existing) {
167
- target.set(incoming.url, incoming);
168
- return;
169
- }
170
- existing.discoveryMethods = [.../* @__PURE__ */ new Set([...existing.discoveryMethods, ...incoming.discoveryMethods])];
171
- existing.altTexts = [.../* @__PURE__ */ new Set([...existing.altTexts, ...incoming.altTexts])];
172
- existing.contexts = [.../* @__PURE__ */ new Set([...existing.contexts, ...incoming.contexts])];
173
- if (!existing.type && incoming.type) existing.type = incoming.type;
174
- if ((incoming.width ?? 0) * (incoming.height ?? 0) > (existing.width ?? 0) * (existing.height ?? 0)) {
175
- existing.width = incoming.width;
176
- existing.height = incoming.height;
177
- }
178
- }
179
- function discoverStaticMedia(html, baseUrl) {
180
- const found = /* @__PURE__ */ new Map();
181
- const add = (raw, method, explicitType = null, alt, context, dimensions) => {
182
- const resolved = raw ? resolveUrl(raw.trim().replace(/&/g, "&"), baseUrl) : null;
183
- if (!resolved || isAdUrl(resolved)) return;
184
- const inferredDimensions = dimensions ?? dimensionsFromUrl(resolved);
185
- mergeCandidate(found, {
186
- url: resolved,
187
- type: explicitType ?? typeFromUrl(resolved),
188
- discoveryMethods: [method],
189
- altTexts: alt ? [alt] : [],
190
- contexts: context ? [context] : [],
191
- width: inferredDimensions?.width ?? null,
192
- height: inferredDimensions?.height ?? null
193
- });
194
- };
195
- const scanSrcset = (value, method, type, alt) => {
196
- for (const part of (value ?? "").split(",")) {
197
- const [url, descriptor] = part.trim().split(/\s+/);
198
- if (!url) continue;
199
- const width = descriptor?.match(/^(\d+)w$/)?.[1];
200
- add(url, method, type, alt, null, width ? { width: Number(width), height: 0 } : null);
201
- }
202
- };
203
- const scan = (document2, suffix = "") => {
204
- const $ = document2.$;
205
- for (const image of document2.images) {
206
- const attrs = image.attributes;
207
- const alt = boundedText(image.alt, 500);
208
- const dimensions = image.width && image.height ? { width: image.width, height: image.height } : null;
209
- for (const attr of ["src", "data-src", "data-lazy-src", "data-original", "data-bg", "data-background", "data-bg-url", "data-lazy-bg", "data-echo"]) {
210
- const value = attrs[attr];
211
- add(value, `static-${attr}${suffix}`, "image", alt, null, dimensions);
212
- }
213
- scanSrcset(attrs.srcset, `static-srcset${suffix}`, "image", alt);
214
- scanSrcset(attrs["data-srcset"], `static-data-srcset${suffix}`, "image", alt);
215
- }
216
- $("source").each((_index, element) => {
217
- const node = $(element);
218
- const declared = node.attr("type") ?? "";
219
- const type = declared.startsWith("video/") ? "video" : declared.startsWith("audio/") ? "audio" : "image";
220
- add(node.attr("src"), `static-source${suffix}`, type);
221
- scanSrcset(node.attr("srcset"), `static-source-srcset${suffix}`, type);
222
- });
223
- $("video,audio").each((_index, element) => {
224
- const node = $(element);
225
- const type = element.tagName.toLowerCase();
226
- add(node.attr("src"), `static-${type}${suffix}`, type);
227
- if (type === "video") add(node.attr("poster"), `static-video-poster${suffix}`, "image");
228
- });
229
- $("meta").each((_index, element) => {
230
- const node = $(element);
231
- const prop = (node.attr("property") ?? node.attr("name") ?? "").toLowerCase();
232
- if (prop === "og:image" || prop === "og:image:url" || prop === "twitter:image" || prop === "twitter:image:src") {
233
- add(node.attr("content"), `static-${prop}${suffix}`, "image");
234
- }
235
- });
236
- $("link[href]").each((_index, element) => {
237
- const node = $(element);
238
- const rel = (node.attr("rel") ?? "").toLowerCase().split(/\s+/);
239
- if (rel.some((value) => value === "icon" || value === "apple-touch-icon")) {
240
- add(node.attr("href"), `static-icon${suffix}`, "image");
241
- }
242
- });
243
- $("svg image").each((_index, element) => {
244
- const node = $(element);
245
- add(node.attr("href") ?? node.attr("xlink:href"), `static-svg-image${suffix}`, "image");
246
- });
247
- $("a[href]").has("img").each((_index, element) => {
248
- const node = $(element);
249
- const href = node.attr("href");
250
- if (!href) return;
251
- if (/lightbox|gallery|fancybox|glightbox|swipebox|elementor-open-lightbox/i.test(`${node.attr("class") ?? ""} ${node.attr("data-elementor-open-lightbox") ?? ""}`) || /\.(?:avif|gif|jpe?g|png|svg|webp)(?:[?#]|$)/i.test(href)) {
252
- const alt = boundedText(node.find("img").first().attr("alt"), 500);
253
- add(href, `static-lightbox-target${suffix}`, "image", alt);
254
- }
255
- });
256
- const cssSources = $("style").toArray().map((element) => $(element).text());
257
- $("[style]").each((_index, element) => {
258
- cssSources.push($(element).attr("style") ?? "");
259
- });
260
- for (const match of cssSources.join("\n").matchAll(/(?:background(?:-image)?\s*:\s*)?url\(\s*(?:["']|"|&#0?34;|'|&#0?39;)?([^"')&]+)(?:["']|"|&#0?34;|'|&#0?39;)?\s*\)/gi)) {
261
- add(match[1], `static-css-url${suffix}`, "image");
262
- }
263
- for (const match of document2.sourceHtml.matchAll(/["'](https?:\/\/[^"'\s<>]+\.(?:avif|gif|jpe?g|png|svg|webp|mp4|webm|mp3|ogg)(?:\?[^"'\s<>]*)?)["']/gi)) {
264
- add(match[1], `static-bare-url${suffix}`);
265
- }
266
- };
267
- const document = typeof html === "string" ? loadHtmlDocument(html) : html;
268
- scan(document);
269
- document.$("script").each((_index, element) => {
270
- const source = document.$(element).text();
271
- const unescaped = source.replace(/\\u003[cC]/g, "<").replace(/\\u003[eE]/g, ">").replace(/\\u0026/gi, "&").replace(/\\["']/g, (match) => match.slice(1)).replace(/\\\//g, "/");
272
- if (unescaped !== source && /<(?:img|source|video|audio)\b/i.test(unescaped)) {
273
- scan(loadHtmlDocument(unescaped), "-json-unescaped");
274
- }
275
- });
276
- return [...found.values()];
277
- }
278
- function mergeMediaDiscovery(staticAssets, rendered, maxAssets = 100) {
279
- const exact = /* @__PURE__ */ new Map();
280
- for (const asset of staticAssets) mergeCandidate(exact, asset);
281
- for (const asset of rendered?.assets ?? []) mergeCandidate(exact, { ...asset, type: asset.type });
282
- const families = /* @__PURE__ */ new Map();
283
- for (const asset of exact.values()) {
284
- const key = mediaFamilyKey(asset.url);
285
- const family = families.get(key) ?? [];
286
- family.push(asset);
287
- families.set(key, family);
288
- }
289
- return [...families.values()].map((family) => {
290
- const ordered = [...family].sort((a, b) => candidateScore(b) - candidateScore(a) || a.url.localeCompare(b.url));
291
- const preferred = { ...ordered[0] };
292
- preferred.discoveryMethods = [...new Set(family.flatMap((asset) => asset.discoveryMethods))];
293
- preferred.altTexts = [...new Set(family.flatMap((asset) => asset.altTexts))];
294
- preferred.contexts = [...new Set(family.flatMap((asset) => asset.contexts))];
295
- return { ...preferred, variants: ordered.slice(1).map((asset) => asset.url) };
296
- }).sort((a, b) => candidateScore(b) - candidateScore(a) || a.url.localeCompare(b.url)).slice(0, Math.max(1, maxAssets));
297
- }
298
- async function downloadAsset(url, destDir, filename, options = {}) {
299
- const maxBytes = Math.max(1, Math.min(options.maxBytes ?? 50 * 1024 * 1024, 100 * 1024 * 1024));
300
- let target = url;
301
- let res = null;
302
- for (let redirects = 0; redirects <= 5; redirects++) {
303
- const checked = await validatePublicHttpUrl(target, { field: "media URL" });
304
- if (checked.error || !checked.parsed) throw new Error(checked.error ?? "Media URL was rejected");
305
- res = await fetch(checked.parsed.href, {
306
- signal: AbortSignal.timeout(15e3),
307
- redirect: "manual"
308
- });
309
- if (res.status >= 300 && res.status < 400) {
310
- const location = res.headers.get("location");
311
- if (!location) throw new Error(`HTTP ${res.status} redirect did not include Location`);
312
- target = new URL(location, checked.parsed.href).href;
313
- res = null;
314
- continue;
315
- }
316
- break;
317
- }
318
- if (!res) throw new Error("Media download exceeded five redirects");
319
- if (!res.ok) throw new Error(`HTTP ${res.status}`);
320
- if (!res.body) throw new Error("Empty response body");
321
- const mimeType = res.headers.get("content-type")?.split(";")[0].trim() ?? null;
322
- if (options.expectedType && (!mimeType || typeFromMime(mimeType) !== options.expectedType)) {
323
- throw new Error(`Expected ${options.expectedType} content but received ${mimeType ?? "no content-type"}`);
324
- }
325
- const declaredBytes = Number(res.headers.get("content-length") ?? 0);
326
- if (Number.isFinite(declaredBytes) && declaredBytes > maxBytes) {
327
- throw new Error(`Media asset exceeds ${maxBytes} byte limit`);
328
- }
329
- let dest = join(destDir, filename);
330
- if (mimeType && !extname(filename)) {
331
- const mimeExt = {
332
- "image/jpeg": ".jpg",
333
- "image/png": ".png",
334
- "image/webp": ".webp",
335
- "image/gif": ".gif",
336
- "image/svg+xml": ".svg",
337
- "image/avif": ".avif",
338
- "video/mp4": ".mp4",
339
- "video/webm": ".webm",
340
- "audio/mpeg": ".mp3",
341
- "audio/ogg": ".ogg",
342
- "audio/wav": ".wav"
343
- };
344
- const ext = mimeExt[mimeType];
345
- if (ext) dest = dest + ext;
346
- }
347
- const writer = createWriteStream(dest);
348
- await new Promise((resolve, reject) => {
349
- writer.once("open", () => resolve());
350
- writer.once("error", reject);
351
- });
352
- let streamedBytes = 0;
353
- const limiter = new Transform({
354
- transform(chunk, _encoding, callback) {
355
- const bytes = chunk.length;
356
- if (streamedBytes + bytes > maxBytes) {
357
- callback(new Error(`Media asset exceeds ${maxBytes} byte limit`));
358
- return;
359
- }
360
- if (options.consumeBytes && !options.consumeBytes(bytes)) {
361
- callback(new Error("Media export total byte limit exceeded"));
362
- return;
363
- }
364
- streamedBytes += bytes;
365
- callback(null, chunk);
366
- }
367
- });
368
- try {
369
- await pipeline(Readable.fromWeb(res.body), limiter, writer);
370
- } catch (error) {
371
- writer.destroy();
372
- rmSync(dest, { force: true });
373
- throw error;
374
- }
375
- const { statSync } = await import("fs");
376
- const sizeBytes = statSync(dest).size;
377
- return { savedPath: dest, sizeBytes, mimeType };
378
- }
379
- async function harvestPageMedia(html, pageUrl, options = {}) {
380
- const types = options.types ?? ["image", "video", "audio"];
381
- const typesSet = new Set(types);
382
- const staticAssets = discoverStaticMedia(html, pageUrl);
383
- const maxAssets = Math.max(1, Math.min(options.maxAssets ?? 100, 250));
384
- const allDiscovered = mergeMediaDiscovery(staticAssets, options.rendered, 2e3);
385
- const discovered = allDiscovered.slice(0, maxAssets);
386
- const totalFound = (/* @__PURE__ */ new Set([
387
- ...staticAssets.map((asset) => asset.url),
388
- ...(options.rendered?.assets ?? []).map((asset) => asset.url)
389
- ])).size;
390
- const kept = discovered.flatMap((asset) => {
391
- const type = asset.type ?? typeFromUrl(asset.url);
392
- return type && typesSet.has(type) ? [{ ...asset, type }] : [];
393
- });
394
- const filteredCount = totalFound - kept.length;
395
- const retentionLimitReached = allDiscovered.length > maxAssets;
396
- const exhausted = options.rendered?.exhausted === true && !retentionLimitReached;
397
- const stopReason = retentionLimitReached ? "asset_limit" : options.rendered?.stopReason ?? "render_unavailable";
398
- const warnings = retentionLimitReached ? [`Media discovery found ${allDiscovered.length} responsive families; only the requested ${maxAssets} records were retained.`] : options.rendered ? options.rendered.exhausted ? [] : [`Rendered media discovery stopped with ${options.rendered.stopReason}; the inventory may be incomplete.`] : ["Rendered media discovery was unavailable; this manifest contains static-source evidence only."];
399
- const baseManifest = {
400
- pageUrl,
401
- staticFound: staticAssets.length,
402
- renderedFound: options.rendered?.assets.length ?? 0,
403
- filteredCount,
404
- totalFound,
405
- completeness: exhausted ? "complete" : "partial",
406
- exhausted,
407
- stopReason,
408
- scrollRounds: options.rendered?.scrollRounds ?? 0,
409
- warnings,
410
- artifact: null
411
- };
412
- if (options.outputDir === null) {
413
- return {
414
- ...baseManifest,
415
- outputDir: null,
416
- assets: kept.map((asset, i) => ({
417
- ...asset,
418
- mimeType: null,
419
- filename: safeFilename(asset.url, i),
420
- savedPath: null,
421
- sizeBytes: null,
422
- sha256: null,
423
- downloadStatus: "not_attempted",
424
- downloadError: null,
425
- inlinePreview: null
426
- }))
427
- };
428
- }
429
- const domain = (() => {
430
- try {
431
- return new URL(pageUrl).hostname.replace(/^www\./, "");
432
- } catch {
433
- return "unknown";
434
- }
435
- })();
436
- const stamp = (/* @__PURE__ */ new Date()).toISOString().replace(/[:.]/g, "-").slice(0, 19);
437
- const outDir = options.outputDir ?? join(homedir(), "Downloads", "mcp-scraper", "media", `${stamp}-${domain}`);
438
- mkdirSync(outDir, { recursive: true });
439
- const assets = [];
440
- await Promise.allSettled(
441
- kept.map(async (asset, i) => {
442
- const filename = safeFilename(asset.url, i);
443
- try {
444
- const { savedPath, sizeBytes, mimeType } = await downloadAsset(asset.url, outDir, filename);
445
- const resolvedType = mimeType ? typeFromMime(mimeType) ?? asset.type : asset.type;
446
- assets.push({
447
- ...asset,
448
- type: resolvedType,
449
- mimeType,
450
- filename: basename(savedPath),
451
- savedPath,
452
- sizeBytes,
453
- sha256: null,
454
- downloadStatus: "downloaded",
455
- downloadError: null,
456
- inlinePreview: null
457
- });
458
- } catch (error) {
459
- assets.push({
460
- ...asset,
461
- mimeType: null,
462
- filename,
463
- savedPath: null,
464
- sizeBytes: null,
465
- sha256: null,
466
- downloadStatus: "failed",
467
- downloadError: error instanceof Error ? error.message : "media_download_failed",
468
- inlinePreview: null
469
- });
470
- }
471
- })
472
- );
473
- assets.sort((a, b) => {
474
- if (a.savedPath && !b.savedPath) return -1;
475
- if (!a.savedPath && b.savedPath) return 1;
476
- return a.url.localeCompare(b.url);
477
- });
478
- return { ...baseManifest, outputDir: outDir, assets };
479
- }
480
-
481
- // src/api/extraction-problems.ts
482
- var UNREACHABLE_CODES = /* @__PURE__ */ new Set([
483
- "browser_navigation_blocked",
484
- "browser_navigation_failed",
485
- "browser_timeout",
486
- "browser_session_failed",
487
- "browser_result_missing"
488
- ]);
489
- function errorCode(err) {
490
- return err instanceof Error ? err.code : void 0;
491
- }
492
- function errorMessage(err) {
493
- return err instanceof Error ? err.message : String(err);
494
- }
495
- function httpStatusFromBrowserHttpCode(value) {
496
- const match = value?.match(/^browser_http_(\d+)$/);
497
- return match ? Number(match[1]) : null;
498
- }
499
- function bucketHttpStatus(status) {
500
- if (status === 404) return "page_not_found";
501
- if (status === 403) return "page_forbidden";
502
- if (status === 429) return "page_rate_limited";
503
- if (status >= 500 && status <= 599) return "page_server_error";
504
- return "page_http_error";
505
- }
506
- function classifyExtractionProblem(err) {
507
- const message = errorMessage(err);
508
- const code = errorCode(err);
509
- if (isVendorUnavailableError(err, message)) {
510
- return { errorCode: "vendor_unavailable", httpStatus: null, retryable: true };
511
- }
512
- const httpStatus = httpStatusFromBrowserHttpCode(code) ?? httpStatusFromBrowserHttpCode(message.match(/^(browser_http_\d+)/)?.[1]);
513
- if (httpStatus != null) {
514
- return { errorCode: bucketHttpStatus(httpStatus), httpStatus, retryable: httpStatus === 429 || httpStatus >= 500 };
515
- }
516
- if (code === "browser_challenge_unresolved" || /^browser_challenge_unresolved/.test(message)) {
517
- return { errorCode: "bot_check_unresolved", httpStatus: null, retryable: true };
518
- }
519
- if (code === "browser_response_too_large" || /exceeds \d+ byte limit/i.test(message)) {
520
- return { errorCode: "page_too_large", httpStatus: null, retryable: false };
521
- }
522
- if (/browser (?:has been )?closed|context closed|target page, context or browser has been closed|session closed/i.test(message)) {
523
- return { errorCode: "browser_session_interrupted", httpStatus: null, retryable: true };
524
- }
525
- if (UNREACHABLE_CODES.has(code ?? "") || /timeout|navigation_failed|ERR_/i.test(message)) {
526
- return { errorCode: "page_unreachable", httpStatus: null, retryable: true };
527
- }
528
- return { errorCode: "extraction_failed", httpStatus: null, retryable: false };
529
- }
530
- function publicExtractionErrorMessage(code, httpStatus) {
531
- switch (code) {
532
- case "page_not_found":
533
- return `The page could not be found (HTTP ${httpStatus ?? 404}). It may have been moved or deleted.`;
534
- case "page_forbidden":
535
- return `The site refused the request (HTTP ${httpStatus ?? 403}).`;
536
- case "page_rate_limited":
537
- return "The site is rate-limiting requests right now (HTTP 429). Retrying after a short wait usually works.";
538
- case "page_server_error":
539
- return `The page's own server returned an error (HTTP ${httpStatus}). This is on the target site's side, not something a retry here can fix.`;
540
- case "page_http_error":
541
- return `The page returned an unexpected HTTP ${httpStatus} response.`;
542
- case "bot_check_unresolved":
543
- return "This site has automated-traffic protection (a bot/CAPTCHA check) that could not be resolved in time. Some sites are simply not extractable this way.";
544
- case "page_too_large":
545
- return "The page is larger than we can safely process.";
546
- case "page_unreachable":
547
- return "The page did not respond in time, or the connection could not be completed.";
548
- case "browser_session_interrupted":
549
- return "The browser session closed before extraction completed. A retry uses a fresh browser session.";
550
- case "vendor_unavailable":
551
- return "MCP Scraper's internal browser services require MCP Scraper team attention. Servers or IPs are down until this is fixed \u2014 this is not caused by your request. Please retry in a few minutes.";
552
- case "extraction_failed":
553
- return "The page could not be extracted. Please retry, or contact support if this persists for the same URL.";
554
- }
555
- }
556
- function extractionProblemResponse(problem, context = {}) {
557
- const message = publicExtractionErrorMessage(problem.errorCode, problem.httpStatus);
558
- const envelope = buildPublicErrorEnvelope({
559
- errorCode: problem.errorCode,
560
- retryable: problem.retryable,
561
- chargeStatus: context.chargeStatus,
562
- details: {
563
- ...context.details ?? {},
564
- ...problem.httpStatus != null ? { http_status: problem.httpStatus } : {}
565
- }
566
- });
567
- return { ...envelope, message, error: message };
568
- }
569
- function extractionWireStatus(problem) {
570
- if (problem.httpStatus === 404) return 404;
571
- if (problem.httpStatus === 403) return 403;
572
- if (problem.httpStatus === 429) return 429;
573
- if (problem.httpStatus != null && problem.httpStatus >= 500 && problem.httpStatus <= 599) return 500;
574
- if (problem.httpStatus != null) return 502;
575
- if (problem.errorCode === "vendor_unavailable") return 503;
576
- return 502;
577
- }
578
- var WAYBACK_CAPTURE_MISSING = "wayback_capture_missing";
579
- function publicizeExtractionFailure(failureCode, failureReason) {
580
- if (failureCode == null) return { failureCode: null, failureReason: null };
581
- if (failureCode === WAYBACK_CAPTURE_MISSING) return { failureCode, failureReason: failureReason ?? null };
582
- const err = new Error(failureReason ?? failureCode);
583
- err.code = failureCode;
584
- const problem = classifyExtractionProblem(err);
585
- return { failureCode: problem.errorCode, failureReason: publicExtractionErrorMessage(problem.errorCode, problem.httpStatus) };
586
- }
587
-
588
- // src/api/commons-embeddings.ts
589
- import { neon } from "@neondatabase/serverless";
590
- var COMMONS_EMBED_PROVIDER = "jina";
591
- var _vectorSql = null;
592
- var vectorSchemaReady = false;
593
- function commonsEmbedModel() {
594
- return (process.env.JINA_EMBED_MODEL ?? "jina-embeddings-v5-omni-small").trim();
595
- }
596
- function commonsEmbedDim() {
597
- return Number((process.env.JINA_EMBED_DIM ?? "1024").trim());
598
- }
599
- function commonsSemanticSearchConfigured() {
600
- return Boolean(process.env.JINA_API_KEY?.trim() && process.env.MEMORY_DATABASE_URL?.trim());
601
- }
602
- function vectorSql() {
603
- if (_vectorSql) return _vectorSql;
604
- const url = process.env.MEMORY_DATABASE_URL?.trim();
605
- if (!url) throw new Error("MEMORY_DATABASE_URL is not set; Commons semantic search needs the shared Postgres.");
606
- _vectorSql = neon(url);
607
- return _vectorSql;
608
- }
609
- async function ensureCommonsVectorSchema() {
610
- if (vectorSchemaReady) return;
611
- const dimension = commonsEmbedDim();
612
- await vectorSql().query("CREATE EXTENSION IF NOT EXISTS vector");
613
- await vectorSql().query(`
614
- CREATE TABLE IF NOT EXISTS commons_index_vectors (
615
- document_id TEXT PRIMARY KEY,
616
- entity_id TEXT NOT NULL,
617
- document_type TEXT NOT NULL,
618
- title TEXT NOT NULL,
619
- embedding vector(${dimension}) NOT NULL,
620
- model TEXT NOT NULL,
621
- updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
622
- )
623
- `);
624
- await vectorSql().query("CREATE INDEX IF NOT EXISTS commons_index_vectors_entity ON commons_index_vectors(entity_id)");
625
- vectorSchemaReady = true;
626
- }
627
- async function embedCommonsTexts(texts) {
628
- const apiKey = process.env.JINA_API_KEY?.trim();
629
- if (!apiKey) throw new Error("JINA_API_KEY is not set; Commons semantic search cannot embed.");
630
- if (!texts.length) return [];
631
- const response = await fetch("https://api.jina.ai/v1/embeddings", {
632
- method: "POST",
633
- headers: { authorization: `Bearer ${apiKey}`, "content-type": "application/json" },
634
- body: JSON.stringify({
635
- model: commonsEmbedModel(),
636
- dimensions: commonsEmbedDim(),
637
- input: texts.map((text) => ({ text: text.slice(0, 8e3) }))
638
- }),
639
- signal: AbortSignal.timeout(6e4)
640
- });
641
- if (!response.ok) {
642
- throw new Error(`Jina embedding request failed with HTTP ${response.status}: ${(await response.text()).slice(0, 200)}`);
643
- }
644
- const payload = await response.json();
645
- const vectors = (payload.data ?? []).map((item) => item.embedding);
646
- if (vectors.length !== texts.length) {
647
- throw new Error(`Jina returned ${vectors.length} embeddings for ${texts.length} inputs.`);
648
- }
649
- return vectors;
650
- }
651
- async function embedQueuedCommonsDocuments(limit = 50) {
652
- if (!commonsSemanticSearchConfigured()) return { claimed: 0, embedded: 0, failed: 0, remaining: 0 };
653
- await ensureCommonsVectorSchema();
654
- const bounded = Math.max(1, Math.min(200, Math.floor(limit)));
655
- const queued = await getDb().execute({
656
- sql: `SELECT id, entity_id, document_type, title, text FROM commons_index_documents
657
- WHERE embedding_status IN ('queued', 'failed') ORDER BY updated_at ASC LIMIT ?`,
658
- args: [bounded]
659
- });
660
- const rows = queued.rows;
661
- if (!rows.length) return { claimed: 0, embedded: 0, failed: 0, remaining: await queuedCommonsDocumentCount() };
662
- let embedded = 0;
663
- let failed = 0;
664
- const model = commonsEmbedModel();
665
- try {
666
- const vectors = await embedCommonsTexts(rows.map((row) => `${row.title}
667
-
668
- ${row.text}`));
669
- for (const [index, row] of rows.entries()) {
670
- const literal = `[${vectors[index].join(",")}]`;
671
- await vectorSql().query(
672
- `INSERT INTO commons_index_vectors (document_id, entity_id, document_type, title, embedding, model, updated_at)
673
- VALUES ($1, $2, $3, $4, $5::vector, $6, now())
674
- ON CONFLICT (document_id) DO UPDATE SET entity_id = EXCLUDED.entity_id, document_type = EXCLUDED.document_type,
675
- title = EXCLUDED.title, embedding = EXCLUDED.embedding, model = EXCLUDED.model, updated_at = now()`,
676
- [row.id, row.entity_id, row.document_type, row.title, literal, model]
677
- );
678
- await getDb().execute({
679
- sql: `UPDATE commons_index_documents SET embedding_status = 'indexed', embedding_provider = ?, embedding_model = ?,
680
- vector_ref = ?, indexed_at = ?, error = NULL WHERE id = ?`,
681
- args: [COMMONS_EMBED_PROVIDER, model, row.id, (/* @__PURE__ */ new Date()).toISOString(), row.id]
682
- });
683
- embedded += 1;
684
- }
685
- } catch (error) {
686
- failed = rows.length - embedded;
687
- const message = (error instanceof Error ? error.message : String(error)).slice(0, 500);
688
- for (const row of rows.slice(embedded)) {
689
- await getDb().execute({
690
- sql: `UPDATE commons_index_documents SET embedding_status = 'failed', error = ? WHERE id = ?`,
691
- args: [message, row.id]
692
- }).catch(() => void 0);
693
- }
694
- }
695
- return { claimed: rows.length, embedded, failed, remaining: await queuedCommonsDocumentCount() };
696
- }
697
- async function queuedCommonsDocumentCount() {
698
- const result = await getDb().execute(`SELECT COUNT(*) AS n FROM commons_index_documents WHERE embedding_status IN ('queued', 'failed')`);
699
- return Number(result.rows[0]?.n ?? 0);
700
- }
701
- async function semanticCommonsEntityScores(query, limit = 40) {
702
- const scores = /* @__PURE__ */ new Map();
703
- if (!commonsSemanticSearchConfigured() || !query.trim()) return scores;
704
- await ensureCommonsVectorSchema();
705
- const [vector] = await embedCommonsTexts([query]);
706
- if (!vector) return scores;
707
- const rows = await vectorSql().query(
708
- `SELECT entity_id, MAX(1 - (embedding <=> $1::vector)) AS score
709
- FROM commons_index_vectors GROUP BY entity_id ORDER BY score DESC LIMIT $2`,
710
- [`[${vector.join(",")}]`, Math.max(1, Math.min(100, limit))]
711
- );
712
- for (const row of rows) scores.set(String(row.entity_id), Number(row.score));
713
- return scores;
714
- }
715
-
716
- export {
717
- typeFromMime,
718
- discoverStaticMedia,
719
- downloadAsset,
720
- harvestPageMedia,
721
- classifyExtractionProblem,
722
- extractionProblemResponse,
723
- extractionWireStatus,
724
- publicizeExtractionFailure,
725
- commonsEmbedModel,
726
- commonsEmbedDim,
727
- commonsSemanticSearchConfigured,
728
- embedCommonsTexts,
729
- embedQueuedCommonsDocuments,
730
- queuedCommonsDocumentCount,
731
- semanticCommonsEntityScores
732
- };