mcp-scraper 0.88.2 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/CHANGELOG.md +36 -15
  2. package/README.md +6 -5
  3. package/THIRD_PARTY_NOTICES.html +203 -0
  4. package/dist/analytics-repository-2BMT5JNE.js +1 -0
  5. package/dist/bin/api-server.js +2 -41
  6. package/dist/bin/mcp-scraper-cli.js +39 -756
  7. package/dist/bin/mcp-scraper-core.js +1 -60
  8. package/dist/bin/mcp-scraper-install.js +2 -25
  9. package/dist/bin/mcp-stdio-server.js +1 -19
  10. package/dist/bin/paa-harvest.js +1 -41
  11. package/dist/chunk-3GP5CYZX.js +1 -0
  12. package/dist/chunk-4AI7DOS7.js +59 -0
  13. package/dist/chunk-4FROKQJN.js +1 -0
  14. package/dist/chunk-7YGVI5J4.js +21710 -0
  15. package/dist/chunk-CCYWSJNG.js +16 -0
  16. package/dist/chunk-CFI6CXIV.js +182 -0
  17. package/dist/chunk-E2WRWV3A.js +1 -0
  18. package/dist/chunk-GMWKPIYX.js +1172 -0
  19. package/dist/chunk-HDPYG3XV.js +102 -0
  20. package/dist/chunk-HE45FFBU.js +1 -0
  21. package/dist/chunk-HUV2WTRW.js +1 -0
  22. package/dist/chunk-KJQXUZ4Y.js +4 -0
  23. package/dist/chunk-L4CGLFPU.js +4 -0
  24. package/dist/chunk-M22MM4N4.js +84 -0
  25. package/dist/chunk-MASR22K4.js +73 -0
  26. package/dist/chunk-MZN4U5BL.js +1 -0
  27. package/dist/chunk-PUHFVA7P.js +1280 -0
  28. package/dist/chunk-QPWPR5XG.js +10 -0
  29. package/dist/chunk-TMB56NCA.js +1 -0
  30. package/dist/chunk-TXENITMS.js +20 -0
  31. package/dist/chunk-W2BVJ7S2.js +13 -0
  32. package/dist/chunk-WO3N5FH2.js +5 -0
  33. package/dist/chunk-WSCGYRWA.js +2595 -0
  34. package/dist/chunk-X54CQLK2.js +1 -0
  35. package/dist/chunk-XLWNEVUZ.js +27 -0
  36. package/dist/chunk-XPZVJIZ2.js +100 -0
  37. package/dist/chunk-YQZGZBB4.js +1 -0
  38. package/dist/chunk-Z2QGQJS2.js +1 -0
  39. package/dist/db-F2MX63GI.js +1 -0
  40. package/dist/extract-bundle-SNUIHM3J.js +26 -0
  41. package/dist/gmail-service-BZ3H75XC.js +1 -0
  42. package/dist/index.cjs +21750 -6045
  43. package/dist/index.d.cts +14 -14
  44. package/dist/index.d.ts +14 -14
  45. package/dist/index.js +18 -315
  46. package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
  47. package/dist/location-data-repository-OTWHWMV6.js +1 -0
  48. package/dist/server-RFR2A5UJ.js +7303 -0
  49. package/dist/site-extract-repository-SE776XDC.js +1 -0
  50. package/dist/worker-XUDSM3AL.js +1 -0
  51. package/package.json +17 -124
  52. package/dist/analytics-repository-GGJJCVVP.js +0 -194
  53. package/dist/chunk-4QMUF6XM.js +0 -1013
  54. package/dist/chunk-6HAV7LCE.js +0 -265
  55. package/dist/chunk-ABF2CGOZ.js +0 -113
  56. package/dist/chunk-C5Z4OFKW.js +0 -404
  57. package/dist/chunk-DNM65UCK.js +0 -299
  58. package/dist/chunk-EQGTEHLZ.js +0 -592
  59. package/dist/chunk-F5GQJWZU.js +0 -732
  60. package/dist/chunk-GGZEC22A.js +0 -215
  61. package/dist/chunk-GXBZXWXB.js +0 -184
  62. package/dist/chunk-IHXAXYIS.js +0 -843
  63. package/dist/chunk-K3Z5AQYE.js +0 -683
  64. package/dist/chunk-K45K75OF.js +0 -6
  65. package/dist/chunk-LFW2FRPJ.js +0 -224
  66. package/dist/chunk-MZDNZQWT.js +0 -2078
  67. package/dist/chunk-NVUKO5NN.js +0 -256
  68. package/dist/chunk-OM7HVEJ3.js +0 -26
  69. package/dist/chunk-OPQIGAFB.js +0 -286
  70. package/dist/chunk-OZJMVCDK.js +0 -16
  71. package/dist/chunk-P7FWOMU7.js +0 -505
  72. package/dist/chunk-PGJQDMC2.js +0 -383
  73. package/dist/chunk-PKZS6SHW.js +0 -33139
  74. package/dist/chunk-RJ7JVYKU.js +0 -68
  75. package/dist/chunk-S24LFPL7.js +0 -5262
  76. package/dist/chunk-T3MZISOF.js +0 -240
  77. package/dist/chunk-UZPTGUDV.js +0 -1915
  78. package/dist/chunk-X623GTBV.js +0 -8290
  79. package/dist/chunk-YXNDOQXN.js +0 -4018
  80. package/dist/db-Z34LPZNR.js +0 -284
  81. package/dist/extract-bundle-565SBZCR.js +0 -1003
  82. package/dist/gmail-service-E6ALS7JG.js +0 -25
  83. package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
  84. package/dist/location-data-repository-WPRG62GE.js +0 -34
  85. package/dist/server-SQZ3A7SY.js +0 -86606
  86. package/dist/site-extract-repository-VYFZASPU.js +0 -69
  87. package/dist/worker-LDCAULWL.js +0 -146
@@ -1,1003 +0,0 @@
1
- import {
2
- commonsEmbedDim,
3
- commonsEmbedModel,
4
- downloadAsset,
5
- embedCommonsTexts,
6
- publicizeExtractionFailure
7
- } from "./chunk-F5GQJWZU.js";
8
- import {
9
- createSiteExtractBundleArtifactStream,
10
- createSiteExtractContentReader,
11
- createSiteExtractImageArtifact,
12
- extractJobLimitInfo
13
- } from "./chunk-IHXAXYIS.js";
14
- import "./chunk-OZJMVCDK.js";
15
- import {
16
- computeIssues,
17
- renderImageSection,
18
- renderIssueReport,
19
- renderLinkReport
20
- } from "./chunk-PGJQDMC2.js";
21
- import "./chunk-DNM65UCK.js";
22
- import "./chunk-4QMUF6XM.js";
23
- import "./chunk-GGZEC22A.js";
24
- import {
25
- getDb
26
- } from "./chunk-YXNDOQXN.js";
27
-
28
- // src/api/extract-bundle.ts
29
- import { createWriteStream, mkdirSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "fs";
30
- import { createHash as createHash2 } from "crypto";
31
- import { basename, join } from "path";
32
- import { tmpdir } from "os";
33
- import { createGzip } from "zlib";
34
- import { once } from "events";
35
- import { finished } from "stream/promises";
36
- import { ZipFile } from "yazl";
37
- import pLimit from "p-limit";
38
-
39
- // src/api/site-content-similarity.ts
40
- import { createHash } from "crypto";
41
- var MAX_SIMILARITY_PAGES = 500;
42
- var DEFAULT_SIMILARITY_THRESHOLD = 0.9;
43
- var DEFAULT_SIMILARITY_MAX_PAIRS = 1e4;
44
- var MAX_SIMILARITY_PAIRS = 5e4;
45
- function round(value, places) {
46
- const scale = 10 ** places;
47
- return Math.round(value * scale) / scale;
48
- }
49
- function cosineSimilarity(a, b) {
50
- if (a.length === 0 || a.length !== b.length) throw new Error("Similarity vectors must have the same non-zero dimension.");
51
- let dot = 0;
52
- let normA = 0;
53
- let normB = 0;
54
- for (let index = 0; index < a.length; index++) {
55
- dot += a[index] * b[index];
56
- normA += a[index] * a[index];
57
- normB += b[index] * b[index];
58
- }
59
- if (normA === 0 || normB === 0) return 0;
60
- return Math.max(-1, Math.min(1, dot / (Math.sqrt(normA) * Math.sqrt(normB))));
61
- }
62
- function corpusHash(pages) {
63
- const digest = createHash("sha256");
64
- for (const page of [...pages].sort((a, b) => a.url.localeCompare(b.url))) {
65
- digest.update(page.url);
66
- digest.update("\0");
67
- digest.update(page.contentHash);
68
- digest.update("\0");
69
- }
70
- return digest.digest("hex");
71
- }
72
- function markdownBlockSignature(block) {
73
- const normalized = block.replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1").replace(/\[([^\]]+)\]\([^)]*\)/g, "$1").replace(/https?:\/\/\S+/gi, " ").replace(/&(?:amp|nbsp|quot|#39);/gi, " ").replace(/[#>*_`~|\\-]+/g, " ").replace(/[^\p{L}\p{N}]+/gu, " ").replace(/\s+/g, " ").trim().toLowerCase();
74
- return normalized.length >= 12 ? normalized : null;
75
- }
76
- function prepareEmbeddingTexts(pages) {
77
- const blocksByPage = pages.map((page) => page.bodyMarkdown.split(/\n{2,}/).map((block) => block.trim()).filter(Boolean));
78
- const minimumPageCount = pages.length >= 4 ? Math.max(3, Math.ceil(pages.length * 0.5)) : null;
79
- const pageFrequency = /* @__PURE__ */ new Map();
80
- for (const blocks of blocksByPage) {
81
- const pageSignatures = new Set(blocks.map(markdownBlockSignature).filter((value) => Boolean(value)));
82
- for (const signature of pageSignatures) pageFrequency.set(signature, (pageFrequency.get(signature) ?? 0) + 1);
83
- }
84
- const repeated = /* @__PURE__ */ new Set();
85
- if (minimumPageCount != null) {
86
- for (const [signature, count] of pageFrequency) {
87
- if (count >= minimumPageCount) repeated.add(signature);
88
- }
89
- }
90
- let removedCharacters = 0;
91
- const texts = pages.map((page, pageIndex) => {
92
- const kept = [];
93
- for (const block of blocksByPage[pageIndex]) {
94
- const signature = markdownBlockSignature(block);
95
- if (signature && repeated.has(signature)) {
96
- removedCharacters += block.length;
97
- continue;
98
- }
99
- kept.push(block);
100
- }
101
- const cleaned = kept.join("\n\n").trim();
102
- const content = cleaned.length >= 60 ? cleaned : page.bodyMarkdown;
103
- return `${page.title?.trim() || "Untitled page"}
104
-
105
- ${content}`;
106
- });
107
- return {
108
- texts,
109
- minimumPageCount,
110
- removedBlockSignatures: repeated.size,
111
- removedCharacters
112
- };
113
- }
114
- var DisjointSet = class {
115
- parent;
116
- constructor(size) {
117
- this.parent = Array.from({ length: size }, (_, index) => index);
118
- }
119
- find(value) {
120
- const parent = this.parent[value];
121
- if (parent !== value) this.parent[value] = this.find(parent);
122
- return this.parent[value];
123
- }
124
- union(a, b) {
125
- const rootA = this.find(a);
126
- const rootB = this.find(b);
127
- if (rootA !== rootB) this.parent[rootB] = rootA;
128
- }
129
- };
130
- async function analyzeSiteContentSimilarity(pages, options = {}) {
131
- const eligible = pages.filter((page) => page.bodyMarkdown.trim().length > 0).slice(0, MAX_SIMILARITY_PAGES);
132
- const requestedThreshold = Number.isFinite(options.threshold) ? options.threshold : DEFAULT_SIMILARITY_THRESHOLD;
133
- const requestedMaxPairs = Number.isFinite(options.maxPairs) ? options.maxPairs : DEFAULT_SIMILARITY_MAX_PAIRS;
134
- const threshold = Math.max(0, Math.min(1, requestedThreshold));
135
- const maxPairs = Math.max(1, Math.min(MAX_SIMILARITY_PAIRS, Math.floor(requestedMaxPairs)));
136
- const model = options.model ?? commonsEmbedModel();
137
- const dimensions = options.dimensions ?? commonsEmbedDim();
138
- const embedTexts = options.embedTexts ?? embedCommonsTexts;
139
- const prepared = prepareEmbeddingTexts(eligible);
140
- const uniqueTexts = [];
141
- const textIndexByHash = /* @__PURE__ */ new Map();
142
- const pageTextIndexes = [];
143
- for (const [pageIndex, page] of eligible.entries()) {
144
- const key = page.contentHash || createHash("sha256").update(page.bodyMarkdown).digest("hex");
145
- let textIndex = textIndexByHash.get(key);
146
- if (textIndex == null) {
147
- textIndex = uniqueTexts.length;
148
- textIndexByHash.set(key, textIndex);
149
- uniqueTexts.push(prepared.texts[pageIndex]);
150
- }
151
- pageTextIndexes.push(textIndex);
152
- }
153
- const uniqueVectors = [];
154
- for (let offset = 0; offset < uniqueTexts.length; offset += 64) {
155
- uniqueVectors.push(...await embedTexts(uniqueTexts.slice(offset, offset + 64)));
156
- }
157
- if (uniqueVectors.length !== uniqueTexts.length) {
158
- throw new Error(`Embedding provider returned ${uniqueVectors.length} vectors for ${uniqueTexts.length} unique page bodies.`);
159
- }
160
- const vectors = pageTextIndexes.map((index) => uniqueVectors[index]);
161
- const sets = new DisjointSet(eligible.length);
162
- const allScores = [];
163
- const qualifying = [];
164
- for (let source = 0; source < eligible.length; source++) {
165
- for (let target = source + 1; target < eligible.length; target++) {
166
- const score = cosineSimilarity(vectors[source], vectors[target]);
167
- allScores.push(score);
168
- if (score < threshold) continue;
169
- qualifying.push({ source, target, score });
170
- sets.union(source, target);
171
- }
172
- }
173
- qualifying.sort((a, b) => b.score - a.score || eligible[a.source].url.localeCompare(eligible[b.source].url) || eligible[a.target].url.localeCompare(eligible[b.target].url));
174
- const sortedScores = [...allScores].sort((a, b) => a - b);
175
- const quantile = (value) => {
176
- if (!sortedScores.length) return null;
177
- return round(sortedScores[Math.floor((sortedScores.length - 1) * value)], 6);
178
- };
179
- const membersByRoot = /* @__PURE__ */ new Map();
180
- for (let index = 0; index < eligible.length; index++) {
181
- const root = sets.find(index);
182
- const members = membersByRoot.get(root) ?? [];
183
- members.push(index);
184
- membersByRoot.set(root, members);
185
- }
186
- const clusterIdByPage = /* @__PURE__ */ new Map();
187
- const clusters = [...membersByRoot.values()].filter((members) => members.length > 1).sort((a, b) => b.length - a.length || eligible[a[0]].url.localeCompare(eligible[b[0]].url)).map((members, index) => {
188
- const clusterId = `cluster-${String(index + 1).padStart(3, "0")}`;
189
- for (const member of members) clusterIdByPage.set(member, clusterId);
190
- return {
191
- clusterId,
192
- pageCount: members.length,
193
- urls: members.map((member) => eligible[member].url).sort()
194
- };
195
- });
196
- const rows = qualifying.slice(0, maxPairs).map((pair, pairIndex) => {
197
- const source = eligible[pair.source];
198
- const target = eligible[pair.target];
199
- const similarity = round(pair.score, 6);
200
- return {
201
- sourceUrl: source.url,
202
- sourceTitle: source.title,
203
- targetUrl: target.url,
204
- targetTitle: target.title,
205
- similarity,
206
- similarityPercent: round(similarity * 100, 2),
207
- corpusPercentile: sortedScores.length <= 1 ? 100 : round((1 - pairIndex / (sortedScores.length - 1)) * 100, 2),
208
- exactContentDuplicate: Boolean(source.contentHash && source.contentHash === target.contentHash),
209
- sourceWordCount: source.wordCount,
210
- targetWordCount: target.wordCount,
211
- clusterId: clusterIdByPage.get(pair.source) ?? null
212
- };
213
- });
214
- const corpusSha256 = corpusHash(eligible);
215
- const analysisSha256 = createHash("sha256").update(JSON.stringify({
216
- corpusSha256,
217
- model,
218
- dimensions,
219
- threshold,
220
- maxPairs,
221
- boilerplateRemoval: {
222
- method: "corpus_repeated_markdown_blocks",
223
- minimumPageCount: prepared.minimumPageCount,
224
- removedBlockSignatures: prepared.removedBlockSignatures
225
- }
226
- })).digest("hex");
227
- return {
228
- generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
229
- provider: "jina",
230
- model,
231
- dimensions,
232
- threshold,
233
- requestedMaxPairs: maxPairs,
234
- comparedPages: eligible.length,
235
- possiblePairs: eligible.length * Math.max(0, eligible.length - 1) / 2,
236
- qualifyingPairs: qualifying.length,
237
- returnedPairs: rows.length,
238
- pairsTruncated: qualifying.length > rows.length,
239
- corpusSha256,
240
- analysisSha256,
241
- scoreDistribution: {
242
- min: quantile(0),
243
- p25: quantile(0.25),
244
- median: quantile(0.5),
245
- p75: quantile(0.75),
246
- p90: quantile(0.9),
247
- max: quantile(1)
248
- },
249
- boilerplateRemoval: {
250
- method: "corpus_repeated_markdown_blocks",
251
- minimumPageCount: prepared.minimumPageCount,
252
- removedBlockSignatures: prepared.removedBlockSignatures,
253
- removedCharacters: prepared.removedCharacters
254
- },
255
- rows,
256
- clusters
257
- };
258
- }
259
-
260
- // src/api/extract-bundle.ts
261
- var SITE_EXTRACT_PAGE_CHUNK = 25;
262
- var MAX_IMAGES_PER_PAGE = 20;
263
- var MAX_IMAGES_PER_SITE = 500;
264
- var IMAGE_DOWNLOAD_CONCURRENCY = 8;
265
- var MAX_IMAGE_BYTES = 10 * 1024 * 1024;
266
- var MAX_SITE_IMAGE_BYTES = 100 * 1024 * 1024;
267
- var MAX_ANALYZED_LINK_EDGES = 25e4;
268
- function normalize(u) {
269
- return u.split("#")[0].replace(/\/$/, "");
270
- }
271
- function registrableDomain(host) {
272
- const h = host.replace(/^www\./, "").toLowerCase();
273
- const parts = h.split(".");
274
- return parts.length <= 2 ? h : parts.slice(-2).join(".");
275
- }
276
- function safeImageFilename(url, index) {
277
- try {
278
- const u = new URL(url);
279
- const base = basename(u.pathname).replace(/[^a-zA-Z0-9._-]/g, "_").slice(0, 80);
280
- return base || `image-${index}`;
281
- } catch {
282
- return `image-${index}`;
283
- }
284
- }
285
- function slugFactory() {
286
- const counts = /* @__PURE__ */ new Map();
287
- return (url) => {
288
- const base = url.replace(/^https?:\/\//, "").replace(/[^a-zA-Z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80) || "page";
289
- const n = counts.get(base) ?? 0;
290
- counts.set(base, n + 1);
291
- return n ? `${base}-${n}` : base;
292
- };
293
- }
294
- async function* pageChunks(jobId) {
295
- const db = getDb();
296
- let lastRowId = 0;
297
- for (; ; ) {
298
- const res = await db.execute({
299
- sql: `SELECT rowid AS cursor_id, page
300
- FROM site_extract_pages
301
- WHERE job_id = ? AND rowid > ?
302
- ORDER BY rowid
303
- LIMIT ${SITE_EXTRACT_PAGE_CHUNK}`,
304
- args: [jobId, lastRowId]
305
- });
306
- if (!res.rows.length) return;
307
- lastRowId = Number(res.rows.at(-1)?.cursor_id ?? lastRowId);
308
- if (!Number.isSafeInteger(lastRowId) || lastRowId <= 0) throw new Error("site extract page cursor is invalid");
309
- yield res.rows.map((r) => JSON.parse(String(r.page)));
310
- }
311
- }
312
- function csvCell(value) {
313
- const text = value == null ? "" : String(value);
314
- return /[",\r\n]/.test(text) ? `"${text.replace(/"/g, '""')}"` : text;
315
- }
316
- function similarityTableRow(row) {
317
- return {
318
- source_url: row.sourceUrl,
319
- source_title: row.sourceTitle,
320
- target_url: row.targetUrl,
321
- target_title: row.targetTitle,
322
- similarity: row.similarity,
323
- similarity_percent: row.similarityPercent,
324
- corpus_percentile: row.corpusPercentile,
325
- exact_content_duplicate: row.exactContentDuplicate,
326
- source_word_count: row.sourceWordCount,
327
- target_word_count: row.targetWordCount,
328
- cluster_id: row.clusterId
329
- };
330
- }
331
- async function assembleExtractArtifacts(job, extras = {}) {
332
- const dir = join(tmpdir(), `extract-bundle-${job.id}-${Date.now()}`);
333
- const downloadImages = job.options.downloadImages === true;
334
- if (job.userId == null) throw new Error("site extract artifact owner is missing");
335
- const readPageContent = createSiteExtractContentReader(String(job.userId));
336
- mkdirSync(dir, { recursive: true });
337
- if (downloadImages) mkdirSync(join(dir, "images"), { recursive: true });
338
- const captureRenderedDom = job.options.captureRenderedDom === true;
339
- const semanticSimilarity = job.options.semanticSimilarity === true;
340
- if (captureRenderedDom) mkdirSync(join(dir, "rendered-dom"), { recursive: true });
341
- try {
342
- const metas = [];
343
- const similarityPages = [];
344
- const renderedDomManifest = [];
345
- const domSlug = slugFactory();
346
- const pageExportEntries = [];
347
- const waybackTimeline = job.options.waybackTimeline;
348
- const statusByUrl = /* @__PURE__ */ new Map();
349
- for await (const chunk of pageChunks(job.id)) {
350
- for (const p of chunk) {
351
- statusByUrl.set(normalize(p.url), p.status);
352
- metas.push({
353
- url: p.url,
354
- status: p.status,
355
- title: p.title,
356
- titleLength: p.titleLength,
357
- titlePixels: p.titlePixels,
358
- metaDescription: p.metaDescription,
359
- metaDescLength: p.metaDescLength,
360
- h1: p.h1,
361
- h1_2: p.h1_2,
362
- h2Count: p.h2Count,
363
- extractionStatus: p.extractionStatus,
364
- indexable: p.indexable,
365
- indexabilityReason: p.indexabilityReason,
366
- canonicalUrl: p.canonicalUrl ? p.url : null,
367
- wordCount: p.wordCount,
368
- contentHash: p.contentHash,
369
- imagesMissingAlt: p.imagesMissingAlt,
370
- schemaTypes: p.schemaTypes.length > 0 ? ["present"] : [],
371
- outlinks: []
372
- });
373
- const pageId = p.pageId ?? createHash2("sha256").update(p.archivedUrl ?? p.url).digest("hex");
374
- const pageDir = join(dir, "pages", pageId);
375
- mkdirSync(pageDir, { recursive: true });
376
- const content = p.contentRef ? await readPageContent(p.contentRef) : null;
377
- const html = content?.html ?? null;
378
- const markdown = content?.markdown ?? p.bodyMarkdown ?? null;
379
- if (semanticSimilarity && p.extractionStatus === "successful" && markdown?.trim()) {
380
- similarityPages.push({
381
- url: p.url,
382
- title: p.title,
383
- bodyMarkdown: markdown,
384
- contentHash: p.contentHash,
385
- wordCount: p.wordCount
386
- });
387
- }
388
- if (captureRenderedDom && html != null) {
389
- const relativePath = `rendered-dom/${domSlug(p.url)}.html.txt`;
390
- writeFileSync(join(dir, relativePath), html);
391
- renderedDomManifest.push({
392
- url: p.url,
393
- path: relativePath,
394
- bytes: Buffer.byteLength(html),
395
- truncated: p.renderedDomTruncated === true,
396
- sanitized: p.renderedDomSanitized === true
397
- });
398
- }
399
- const htmlPath = html == null ? null : `pages/${pageId}/page.html`;
400
- const markdownPath = markdown == null ? null : `pages/${pageId}/page.md`;
401
- const legacyMarkdownPath = markdown == null || !p.archiveRequestedMonth ? null : `pages/${p.archiveRequestedMonth}/${pageId}.md`;
402
- if (html != null && htmlPath) writeFileSync(join(dir, htmlPath), html);
403
- if (markdown != null && markdownPath) writeFileSync(join(dir, markdownPath), markdown);
404
- if (markdown != null && legacyMarkdownPath) {
405
- mkdirSync(join(dir, "pages", p.archiveRequestedMonth), { recursive: true });
406
- writeFileSync(join(dir, legacyMarkdownPath), markdown);
407
- }
408
- const { bodyMarkdown: _body, contentRef: _contentRef, discoveryLinks: _discovery, ...metadata } = p;
409
- const pageRecord = {
410
- version: "site-export-page.v1",
411
- ...metadata,
412
- pageId,
413
- content: {
414
- htmlPath,
415
- markdownPath,
416
- htmlBytes: html == null ? 0 : Buffer.byteLength(html),
417
- markdownBytes: markdown == null ? 0 : Buffer.byteLength(markdown),
418
- htmlSha256: html == null ? null : createHash2("sha256").update(html).digest("hex"),
419
- markdownSha256: markdown == null ? null : createHash2("sha256").update(markdown).digest("hex")
420
- }
421
- };
422
- const pageJson = JSON.stringify(pageRecord, null, 2);
423
- const jsonPath = `pages/${pageId}/page.json`;
424
- writeFileSync(join(dir, jsonPath), pageJson);
425
- pageExportEntries.push({
426
- pageId,
427
- url: p.archivedUrl ?? p.url,
428
- jsonPath,
429
- htmlPath,
430
- markdownPath,
431
- legacyMarkdownPath,
432
- jsonSha256: createHash2("sha256").update(pageJson).digest("hex"),
433
- htmlSha256: pageRecord.content.htmlSha256,
434
- markdownSha256: pageRecord.content.markdownSha256,
435
- htmlBytes: pageRecord.content.htmlBytes,
436
- markdownBytes: pageRecord.content.markdownBytes
437
- });
438
- }
439
- }
440
- const gz = createGzip();
441
- const linksOut = createWriteStream(join(dir, "links.jsonl.gz"));
442
- gz.pipe(linksOut);
443
- const gzWrite = async (line) => {
444
- if (!gz.write(line)) await once(gz, "drain");
445
- };
446
- const schemaOut = createWriteStream(join(dir, "schema.jsonl"));
447
- const schemaWrite = async (line) => {
448
- if (!schemaOut.write(line)) await once(schemaOut, "drain");
449
- };
450
- const failuresOut = createWriteStream(join(dir, "failures.jsonl"));
451
- const failureWrite = async (line) => {
452
- if (!failuresOut.write(line)) await once(failuresOut, "drain");
453
- };
454
- const inboundFrom = /* @__PURE__ */ new Map();
455
- const anchorCounts = /* @__PURE__ */ new Map();
456
- const adjacency = /* @__PURE__ */ new Map();
457
- const outCounts = /* @__PURE__ */ new Map();
458
- const domMap = /* @__PURE__ */ new Map();
459
- const brokenLinkPages = /* @__PURE__ */ new Set();
460
- let internalTotal = 0;
461
- let externalTotal = 0;
462
- let brokenInternal = 0;
463
- let streamedLinkEdges = 0;
464
- let analyzedLinkEdges = 0;
465
- let linkAnalysisTruncated = false;
466
- let siteReg = "";
467
- try {
468
- siteReg = registrableDomain(new URL(job.startUrl).hostname);
469
- } catch {
470
- siteReg = "";
471
- }
472
- const imageLimit = pLimit(IMAGE_DOWNLOAD_CONCURRENCY);
473
- const imageDownloads = [];
474
- const imageManifest = [];
475
- const imageArtifacts = [];
476
- let imagesQueued = 0;
477
- let imagesFailed = 0;
478
- let imageBytesDownloaded = 0;
479
- const consumeImageBytes = (bytes) => {
480
- if (imageBytesDownloaded + bytes > MAX_SITE_IMAGE_BYTES) return false;
481
- imageBytesDownloaded += bytes;
482
- return true;
483
- };
484
- for await (const chunk of pageChunks(job.id)) {
485
- for (const p of chunk) {
486
- if (p.schema?.length) await schemaWrite(JSON.stringify({ url: p.url, schema: p.schema }) + "\n");
487
- if (p.extractionStatus === "failed") {
488
- const { failureCode, failureReason } = publicizeExtractionFailure(p.failureCode, p.failureReason);
489
- await failureWrite(JSON.stringify({
490
- url: p.url,
491
- status: p.status,
492
- via: p.via,
493
- failureCode,
494
- failureReason,
495
- fetchAttempts: p.fetchAttempts ?? []
496
- }) + "\n");
497
- }
498
- if (p.imageLinks?.length) {
499
- for (const [idx, imgUrl] of p.imageLinks.entries()) {
500
- const imageId = createHash2("sha256").update(`${p.url}\0${imgUrl}`).digest("hex");
501
- const record = {
502
- imageId,
503
- sourcePage: p.url,
504
- sourceUrl: imgUrl,
505
- path: null,
506
- status: "not_requested",
507
- mimeType: null,
508
- bytes: null,
509
- sha256: null,
510
- failureCode: null,
511
- failureReason: null,
512
- artifactId: null
513
- };
514
- imageManifest.push(record);
515
- if (!downloadImages) continue;
516
- if (idx >= MAX_IMAGES_PER_PAGE) {
517
- record.status = "skipped_page_limit";
518
- record.failureCode = "page_image_limit";
519
- record.failureReason = `Only the first ${MAX_IMAGES_PER_PAGE} image candidates are downloaded per page.`;
520
- continue;
521
- }
522
- if (imagesQueued >= MAX_IMAGES_PER_SITE) {
523
- record.status = "skipped_site_limit";
524
- record.failureCode = "site_image_limit";
525
- record.failureReason = `Only the first ${MAX_IMAGES_PER_SITE} image candidates are downloaded per site.`;
526
- continue;
527
- }
528
- imagesQueued++;
529
- const filename = `${imageId}-${safeImageFilename(imgUrl, idx)}`;
530
- const outputPath = join(dir, "images");
531
- imageDownloads.push(
532
- imageLimit(() => downloadAsset(imgUrl, outputPath, filename, {
533
- expectedType: "image",
534
- maxBytes: MAX_IMAGE_BYTES,
535
- consumeBytes: consumeImageBytes
536
- })).then(async (result) => {
537
- const bytes = readFileSync(result.savedPath);
538
- const artifact = await createSiteExtractImageArtifact({
539
- ownerId: String(job.userId),
540
- jobId: job.id,
541
- imageId,
542
- filename,
543
- contentType: result.mimeType ?? "application/octet-stream",
544
- content: bytes,
545
- sourceUrl: imgUrl,
546
- sourcePage: p.url
547
- });
548
- imageArtifacts.push(artifact);
549
- record.status = "downloaded";
550
- record.path = `images/${filename}`;
551
- record.artifactId = artifact.key;
552
- record.mimeType = result.mimeType;
553
- record.bytes = result.sizeBytes;
554
- record.sha256 = createHash2("sha256").update(bytes).digest("hex");
555
- }, (error) => {
556
- imagesFailed++;
557
- record.status = "failed";
558
- record.failureCode = "image_download_failed";
559
- record.failureReason = (error instanceof Error ? error.message : String(error)).replace(/\s+/g, " ").slice(0, 500);
560
- })
561
- );
562
- }
563
- }
564
- const from = normalize(p.url);
565
- const oc = { int: 0, ext: 0 };
566
- for (const l of p.outlinks ?? []) {
567
- const nofollow = (l.rel ?? "").split(/\s+/).includes("nofollow");
568
- const to = normalize(l.href);
569
- const targetStatus = statusByUrl.get(to) ?? null;
570
- await gzWrite(JSON.stringify({ from: p.url, to: l.href, anchor: l.anchor, rel: l.rel, internal: l.internal, nofollow, targetStatus }) + "\n");
571
- streamedLinkEdges++;
572
- if (analyzedLinkEdges >= MAX_ANALYZED_LINK_EDGES) {
573
- linkAnalysisTruncated = true;
574
- continue;
575
- }
576
- analyzedLinkEdges++;
577
- if (l.internal) {
578
- oc.int++;
579
- internalTotal++;
580
- if (!inboundFrom.has(to)) inboundFrom.set(to, /* @__PURE__ */ new Set());
581
- inboundFrom.get(to).add(from);
582
- if (l.anchor) {
583
- if (!anchorCounts.has(to)) anchorCounts.set(to, /* @__PURE__ */ new Map());
584
- const ac = anchorCounts.get(to);
585
- ac.set(l.anchor, (ac.get(l.anchor) ?? 0) + 1);
586
- }
587
- if (!nofollow && (statusByUrl.get(to) === 200 || !statusByUrl.has(to))) {
588
- if (!adjacency.has(from)) adjacency.set(from, /* @__PURE__ */ new Set());
589
- adjacency.get(from).add(to);
590
- }
591
- if (targetStatus != null && targetStatus >= 400) {
592
- brokenInternal++;
593
- brokenLinkPages.add(p.url);
594
- }
595
- } else {
596
- oc.ext++;
597
- let domain = "";
598
- try {
599
- domain = registrableDomain(new URL(l.href).hostname);
600
- } catch {
601
- continue;
602
- }
603
- if (!domain || domain === siteReg) continue;
604
- externalTotal++;
605
- const d = domMap.get(domain) ?? { links: 0, nofollow: 0, pages: /* @__PURE__ */ new Set() };
606
- d.links++;
607
- if (nofollow) d.nofollow++;
608
- d.pages.add(p.url);
609
- domMap.set(domain, d);
610
- }
611
- }
612
- outCounts.set(p.url, oc);
613
- }
614
- }
615
- const outputFinished = Promise.all([finished(linksOut), finished(schemaOut), finished(failuresOut)]);
616
- gz.end();
617
- schemaOut.end();
618
- failuresOut.end();
619
- await outputFinished;
620
- await Promise.all(imageDownloads);
621
- const depth = /* @__PURE__ */ new Map();
622
- const start = normalize(job.startUrl);
623
- depth.set(start, 0);
624
- const queue = [start];
625
- while (queue.length > 0) {
626
- const cur = queue.shift();
627
- const d = depth.get(cur);
628
- for (const next of adjacency.get(cur) ?? []) {
629
- if (!depth.has(next)) {
630
- depth.set(next, d + 1);
631
- queue.push(next);
632
- }
633
- }
634
- }
635
- const metrics = /* @__PURE__ */ new Map();
636
- for (const p of metas) {
637
- const key = normalize(p.url);
638
- const inbound = inboundFrom.get(key) ?? /* @__PURE__ */ new Set();
639
- const ac = anchorCounts.get(key);
640
- const topAnchors = ac ? [...ac.entries()].sort((a, b) => b[1] - a[1]).slice(0, 3).map((e) => e[0]) : [];
641
- const oc = outCounts.get(p.url) ?? { int: 0, ext: 0 };
642
- metrics.set(p.url, {
643
- url: p.url,
644
- inlinks: inbound.size,
645
- uniqueInlinks: inbound.size,
646
- outlinksInternal: oc.int,
647
- outlinksExternal: oc.ext,
648
- crawlDepth: depth.has(key) ? depth.get(key) : null,
649
- orphan: inbound.size === 0 && key !== start,
650
- topAnchors
651
- });
652
- }
653
- const issues = computeIssues(metas, metrics, brokenLinkPages);
654
- let reportMd = renderIssueReport(job.startUrl, metas, issues, metrics);
655
- const limits = extractJobLimitInfo(job);
656
- const crawlSummary = {
657
- jobId: job.id,
658
- startUrl: job.startUrl,
659
- status: job.successfulUrls === 0 ? "failed" : job.failedUrls > 0 || job.remainingUrls > 0 || limits.creditTruncated ? "partial" : "complete",
660
- ...limits,
661
- discovered: job.totalUrls,
662
- attempted: job.attemptedUrls,
663
- successful: job.successfulUrls,
664
- failed: job.failedUrls,
665
- remaining: job.remainingUrls,
666
- linkAnalysis: {
667
- truncated: linkAnalysisTruncated,
668
- analyzedEdges: analyzedLinkEdges,
669
- streamedEdges: streamedLinkEdges,
670
- edgeCap: MAX_ANALYZED_LINK_EDGES
671
- },
672
- generatedAt: (/* @__PURE__ */ new Date()).toISOString()
673
- };
674
- reportMd = [
675
- `## Crawl delivery`,
676
- `- Discovered: ${crawlSummary.discovered}`,
677
- `- Attempted: ${crawlSummary.attempted}`,
678
- `- Successful: ${crawlSummary.successful}`,
679
- `- Failed: ${crawlSummary.failed}`,
680
- `- Remaining: ${crawlSummary.remaining}`,
681
- `- Requested page cap: ${limits.requestedMaxPages}`,
682
- `- Funded page cap: ${limits.effectiveMaxPages}`,
683
- `- Credit-truncated: ${limits.creditTruncated ? "yes" : "no"}`,
684
- `- Link analysis: ${analyzedLinkEdges}/${streamedLinkEdges} edges${linkAnalysisTruncated ? ` (capped at ${MAX_ANALYZED_LINK_EDGES})` : ""}`,
685
- "",
686
- reportMd
687
- ].join("\n");
688
- if (extras.imageAudit) reportMd += `
689
-
690
- ${renderImageSection(extras.imageAudit)}`;
691
- const metricValues = [...metrics.values()];
692
- const round2 = (n) => Math.round(n * 10) / 10;
693
- const distribution = { zero: 0, oneToTwo: 0, threeToTen: 0, elevenPlus: 0 };
694
- let sumInlinks = 0;
695
- let sumOutlinks = 0;
696
- for (const m of metricValues) {
697
- sumInlinks += m.inlinks;
698
- sumOutlinks += m.outlinksInternal + m.outlinksExternal;
699
- if (m.inlinks === 0) distribution.zero++;
700
- else if (m.inlinks <= 2) distribution.oneToTwo++;
701
- else if (m.inlinks <= 10) distribution.threeToTen++;
702
- else distribution.elevenPlus++;
703
- }
704
- const externalDomains = [...domMap.entries()].map(([domain, d]) => ({ domain, links: d.links, nofollow: d.nofollow, pages: d.pages.size })).sort((a, b) => b.links - a.links);
705
- const linkReport = {
706
- summary: {
707
- internal: {
708
- totalLinks: internalTotal,
709
- pages: metas.length,
710
- orphans: metricValues.filter((m) => m.orphan).length,
711
- brokenInternal,
712
- avgInlinks: metas.length ? round2(sumInlinks / metas.length) : 0,
713
- avgOutlinks: metas.length ? round2(sumOutlinks / metas.length) : 0,
714
- distribution,
715
- topByInlinks: [...metricValues].sort((a, b) => b.inlinks - a.inlinks).slice(0, 20).map((m) => ({ url: m.url, inlinks: m.inlinks, outlinksInternal: m.outlinksInternal, outlinksExternal: m.outlinksExternal }))
716
- },
717
- external: {
718
- totalLinks: externalTotal,
719
- uniqueDomains: externalDomains.length,
720
- topDomains: externalDomains.slice(0, 20)
721
- }
722
- },
723
- externalDomains
724
- };
725
- const pagesOut = createWriteStream(join(dir, "pages.jsonl"));
726
- const pageExportById = new Map(pageExportEntries.map((entry) => [entry.pageId, entry]));
727
- const captureMatrixOut = waybackTimeline ? createWriteStream(join(dir, "capture-matrix.jsonl")) : null;
728
- const capturedCells = /* @__PURE__ */ new Set();
729
- const captureCellKey = (month, url) => {
730
- let normalized = url;
731
- try {
732
- normalized = new URL(url).href;
733
- } catch {
734
- }
735
- return `${month ?? ""}\0${normalized}`;
736
- };
737
- for await (const chunk of pageChunks(job.id)) {
738
- for (const page of chunk) {
739
- const publicFailure = page.extractionStatus === "failed" ? publicizeExtractionFailure(page.failureCode, page.failureReason) : { failureCode: null, failureReason: null };
740
- const { bodyMarkdown: _body, schema: _schema, outlinks: _outlinks, contentRef: _contentRef, ...boundedPage } = page;
741
- const m = metrics.get(page.url);
742
- const exported = page.pageId ? pageExportById.get(page.pageId) : void 0;
743
- const row = {
744
- ...boundedPage,
745
- ...page.extractionStatus === "failed" ? publicFailure : {},
746
- representations: exported ? {
747
- jsonPath: exported.jsonPath,
748
- htmlPath: exported.htmlPath,
749
- markdownPath: exported.markdownPath,
750
- jsonSha256: exported.jsonSha256,
751
- htmlSha256: exported.htmlSha256,
752
- markdownSha256: exported.markdownSha256,
753
- htmlBytes: exported.htmlBytes,
754
- markdownBytes: exported.markdownBytes
755
- } : null,
756
- inlinks: m?.inlinks ?? 0,
757
- crawlDepth: m?.crawlDepth ?? null,
758
- orphan: m?.orphan ?? false
759
- };
760
- if (!pagesOut.write(JSON.stringify(row) + "\n")) await once(pagesOut, "drain");
761
- if (captureMatrixOut) {
762
- const cellKey = captureCellKey(page.archiveRequestedMonth, page.originalUrl ?? page.url);
763
- capturedCells.add(cellKey);
764
- const captureRow = {
765
- requestedMonth: page.archiveRequestedMonth ?? null,
766
- captureTimestamp: page.archiveTimestamp ?? null,
767
- originalUrl: page.originalUrl ?? page.url,
768
- archivedUrl: page.archivedUrl ?? null,
769
- status: page.extractionStatus === "successful" ? "captured" : "failed",
770
- httpStatus: page.status,
771
- contentHash: page.contentHash || null,
772
- wordCount: page.wordCount,
773
- failureCode: publicFailure.failureCode,
774
- failureReason: publicFailure.failureReason
775
- };
776
- if (!captureMatrixOut.write(JSON.stringify(captureRow) + "\n")) await once(captureMatrixOut, "drain");
777
- }
778
- }
779
- }
780
- if (captureMatrixOut && waybackTimeline?.timeline?.urls) {
781
- const timeline = waybackTimeline.timeline;
782
- const requestedUrls = timeline.urls;
783
- const months = (() => {
784
- if (timeline.months?.length) return [...new Set(timeline.months)].sort();
785
- if (!timeline.from || !timeline.to) return [];
786
- const values = [];
787
- const cursor = /* @__PURE__ */ new Date(`${timeline.from}-01T00:00:00Z`);
788
- const end = /* @__PURE__ */ new Date(`${timeline.to}-01T00:00:00Z`);
789
- while (cursor <= end && values.length < 60) {
790
- values.push(`${cursor.getUTCFullYear()}-${String(cursor.getUTCMonth() + 1).padStart(2, "0")}`);
791
- cursor.setUTCMonth(cursor.getUTCMonth() + (timeline.intervalMonths ?? 1));
792
- }
793
- return values;
794
- })();
795
- for (const requestedMonth of months) {
796
- for (const originalUrl of requestedUrls) {
797
- if (capturedCells.has(captureCellKey(requestedMonth, originalUrl))) continue;
798
- if (!captureMatrixOut.write(JSON.stringify({
799
- requestedMonth,
800
- captureTimestamp: null,
801
- originalUrl,
802
- archivedUrl: null,
803
- status: "missing",
804
- httpStatus: null,
805
- contentHash: null,
806
- wordCount: 0,
807
- failureCode: "wayback_capture_missing",
808
- failureReason: "No successful HTML capture was available in the requested month."
809
- }) + "\n")) await once(captureMatrixOut, "drain");
810
- }
811
- }
812
- }
813
- const pagesFinished = finished(pagesOut);
814
- const captureMatrixFinished = captureMatrixOut ? finished(captureMatrixOut) : Promise.resolve();
815
- pagesOut.end();
816
- captureMatrixOut?.end();
817
- await Promise.all([pagesFinished, captureMatrixFinished]);
818
- if (captureRenderedDom) {
819
- writeFileSync(
820
- join(dir, "rendered-dom", "manifest.jsonl"),
821
- renderedDomManifest.map((row) => JSON.stringify(row)).join("\n")
822
- );
823
- }
824
- if (semanticSimilarity) {
825
- const analysis = await analyzeSiteContentSimilarity(similarityPages, {
826
- threshold: Number(job.options.similarityThreshold ?? void 0),
827
- maxPairs: Number(job.options.similarityMaxPairs ?? void 0),
828
- embedTexts: extras.embedSimilarityTexts
829
- });
830
- const tableRows = analysis.rows.map(similarityTableRow);
831
- const columns = tableRows.length > 0 ? Object.keys(tableRows[0]) : [
832
- "source_url",
833
- "source_title",
834
- "target_url",
835
- "target_title",
836
- "similarity",
837
- "similarity_percent",
838
- "corpus_percentile",
839
- "exact_content_duplicate",
840
- "source_word_count",
841
- "target_word_count",
842
- "cluster_id"
843
- ];
844
- const csv = [
845
- columns.join(","),
846
- ...tableRows.map((row) => columns.map((column) => csvCell(row[column] ?? null)).join(","))
847
- ].join("\n") + "\n";
848
- const { rows: _rows, clusters: _clusters, ...summary } = analysis;
849
- writeFileSync(join(dir, "similarity-table.csv"), csv);
850
- writeFileSync(join(dir, "similarity.jsonl"), tableRows.map((row) => JSON.stringify(row)).join("\n"));
851
- writeFileSync(join(dir, "similarity-table-schema.json"), JSON.stringify({
852
- tableNameSuggestion: `site_similarity_${new URL(job.startUrl).hostname.replace(/[^a-z0-9]+/gi, "_").replace(/^_+|_+$/g, "").toLowerCase()}`,
853
- defaultSort: [{ column: "similarity", direction: "desc" }],
854
- columns: {
855
- source_url: "text",
856
- source_title: "text",
857
- target_url: "text",
858
- target_title: "text",
859
- similarity: "number",
860
- similarity_percent: "number",
861
- corpus_percentile: "number",
862
- exact_content_duplicate: "boolean",
863
- source_word_count: "integer",
864
- target_word_count: "integer",
865
- cluster_id: "text"
866
- }
867
- }, null, 2));
868
- writeFileSync(join(dir, "similarity-summary.json"), JSON.stringify(summary, null, 2));
869
- writeFileSync(join(dir, "content-clusters.json"), JSON.stringify(analysis.clusters, null, 2));
870
- reportMd += [
871
- "",
872
- "## Rendered content similarity",
873
- `- Compared pages: ${analysis.comparedPages}`,
874
- `- Raw cosine threshold: ${analysis.threshold}`,
875
- `- Corpus score distribution: min ${analysis.scoreDistribution.min ?? "n/a"} / median ${analysis.scoreDistribution.median ?? "n/a"} / p90 ${analysis.scoreDistribution.p90 ?? "n/a"} / max ${analysis.scoreDistribution.max ?? "n/a"}`,
876
- `- Qualifying pairs: ${analysis.qualifyingPairs}`,
877
- `- Returned table rows: ${analysis.returnedPairs}${analysis.pairsTruncated ? " (capped)" : ""}`,
878
- `- Multi-page clusters: ${analysis.clusters.length}`,
879
- `- Embedding model: ${analysis.model} (${analysis.dimensions} dimensions)`,
880
- `- Corpus boilerplate removed: ${analysis.boilerplateRemoval.removedBlockSignatures} repeated block signatures / ${analysis.boilerplateRemoval.removedCharacters} characters`,
881
- `- Corpus SHA-256: ${analysis.corpusSha256}`
882
- ].join("\n");
883
- }
884
- writeFileSync(join(dir, "report.md"), reportMd);
885
- writeFileSync(join(dir, "crawl-summary.json"), JSON.stringify(crawlSummary, null, 2));
886
- writeFileSync(join(dir, "images-manifest.jsonl"), imageManifest.map((record) => JSON.stringify(record)).join("\n"));
887
- writeFileSync(join(dir, "manifest.json"), JSON.stringify({
888
- version: "site-export.v1",
889
- jobId: job.id,
890
- startUrl: job.startUrl,
891
- generatedAt: crawlSummary.generatedAt,
892
- outputProfile: {
893
- json: true,
894
- html: true,
895
- markdown: true,
896
- imageManifest: true,
897
- imageBytes: downloadImages
898
- },
899
- pages: [...pageExportEntries].sort((a, b) => a.pageId.localeCompare(b.pageId)),
900
- imagesManifestPath: "images-manifest.jsonl",
901
- download: "This ZIP is the complete human export. AI clients should use site_export_read and site_export_image."
902
- }, null, 2));
903
- if (waybackTimeline) {
904
- writeFileSync(join(dir, "wayback-manifest.json"), JSON.stringify({
905
- rootUrl: waybackTimeline.rootUrl ?? job.startUrl,
906
- requested: waybackTimeline.timeline ?? null,
907
- maxPagesPerSnapshot: waybackTimeline.maxPagesPerSnapshot ?? null,
908
- captureMatrix: "capture-matrix.jsonl"
909
- }, null, 2));
910
- }
911
- writeFileSync(join(dir, "issues.json"), JSON.stringify(issues, null, 2));
912
- writeFileSync(join(dir, "link-metrics.jsonl"), metricValues.map((m) => JSON.stringify(m)).join("\n"));
913
- writeFileSync(join(dir, "link-report.md"), renderLinkReport(linkReport));
914
- writeFileSync(join(dir, "links-summary.json"), JSON.stringify(linkReport.summary, null, 2));
915
- writeFileSync(join(dir, "external-domains.json"), JSON.stringify(linkReport.externalDomains, null, 2));
916
- if (extras.imageAudit) {
917
- writeFileSync(join(dir, "images.jsonl"), extras.imageAudit.rows.map((r) => JSON.stringify(r)).join("\n"));
918
- writeFileSync(join(dir, "images-summary.json"), JSON.stringify(extras.imageAudit.summary, null, 2));
919
- }
920
- if (extras.seoAudit) writeFileSync(join(dir, "seo-audit.json"), JSON.stringify(extras.seoAudit, null, 2));
921
- if (extras.branding) writeFileSync(join(dir, "branding.json"), JSON.stringify(extras.branding, null, 2));
922
- if (downloadImages) {
923
- writeFileSync(join(dir, "images-download-summary.json"), JSON.stringify({
924
- queued: imagesQueued,
925
- downloaded: imagesQueued - imagesFailed,
926
- failed: imagesFailed,
927
- downloadedBytes: imageBytesDownloaded,
928
- perPageCap: MAX_IMAGES_PER_PAGE,
929
- perSiteCap: MAX_IMAGES_PER_SITE,
930
- perFileByteCap: MAX_IMAGE_BYTES,
931
- totalByteCap: MAX_SITE_IMAGE_BYTES
932
- }, null, 2));
933
- }
934
- const artifactFiles = [
935
- { name: "manifest.json", type: "application/json" },
936
- { name: "images-manifest.jsonl", type: "application/x-ndjson" },
937
- { name: "report.md", type: "text/markdown" },
938
- { name: "crawl-summary.json", type: "application/json" },
939
- { name: "failures.jsonl", type: "application/x-ndjson" },
940
- { name: "schema.jsonl", type: "application/x-ndjson" },
941
- { name: "issues.json", type: "application/json" },
942
- { name: "pages.jsonl", type: "application/x-ndjson" },
943
- { name: "links.jsonl.gz", type: "application/gzip" },
944
- { name: "link-metrics.jsonl", type: "application/x-ndjson" },
945
- { name: "link-report.md", type: "text/markdown" },
946
- { name: "links-summary.json", type: "application/json" },
947
- { name: "external-domains.json", type: "application/json" }
948
- ];
949
- if (captureRenderedDom) {
950
- artifactFiles.push({ name: "rendered-dom/manifest.jsonl", type: "application/x-ndjson" });
951
- for (const row of renderedDomManifest) artifactFiles.push({ name: row.path, type: "text/plain" });
952
- }
953
- if (semanticSimilarity) {
954
- artifactFiles.push({ name: "similarity-table.csv", type: "text/csv" });
955
- artifactFiles.push({ name: "similarity.jsonl", type: "application/x-ndjson" });
956
- artifactFiles.push({ name: "similarity-table-schema.json", type: "application/json" });
957
- artifactFiles.push({ name: "similarity-summary.json", type: "application/json" });
958
- artifactFiles.push({ name: "content-clusters.json", type: "application/json" });
959
- }
960
- if (waybackTimeline) {
961
- artifactFiles.push({ name: "wayback-manifest.json", type: "application/json" });
962
- artifactFiles.push({ name: "capture-matrix.jsonl", type: "application/x-ndjson" });
963
- }
964
- if (extras.imageAudit) {
965
- artifactFiles.push({ name: "images.jsonl", type: "application/x-ndjson" });
966
- artifactFiles.push({ name: "images-summary.json", type: "application/json" });
967
- }
968
- if (extras.seoAudit) artifactFiles.push({ name: "seo-audit.json", type: "application/json" });
969
- if (extras.branding) artifactFiles.push({ name: "branding.json", type: "application/json" });
970
- if (downloadImages) artifactFiles.push({ name: "images-download-summary.json", type: "application/json" });
971
- const zip = new ZipFile();
972
- const artifactPromise = createSiteExtractBundleArtifactStream({
973
- ownerId: String(job.userId),
974
- jobId: job.id,
975
- // Artifact retention starts when this bundle is assembled, including a
976
- // later support re-finalization, rather than when the crawl was queued.
977
- createdAt: /* @__PURE__ */ new Date(),
978
- content: zip.outputStream
979
- });
980
- for (const f of artifactFiles) zip.addFile(join(dir, f.name), f.name);
981
- const orderedPageEntries = [...pageExportEntries].sort((a, b) => a.pageId.localeCompare(b.pageId));
982
- for (const entry of orderedPageEntries) if (entry.markdownPath) zip.addFile(join(dir, entry.markdownPath), entry.markdownPath);
983
- for (const entry of orderedPageEntries) if (entry.legacyMarkdownPath) zip.addFile(join(dir, entry.legacyMarkdownPath), entry.legacyMarkdownPath);
984
- for (const entry of orderedPageEntries) if (entry.htmlPath) zip.addFile(join(dir, entry.htmlPath), entry.htmlPath);
985
- for (const entry of orderedPageEntries) zip.addFile(join(dir, entry.jsonPath), entry.jsonPath);
986
- if (downloadImages && imagesQueued > imagesFailed) {
987
- const imagesDir = join(dir, "images");
988
- for (const rel of readdirSync(imagesDir, { recursive: true })) {
989
- const full = join(imagesDir, rel);
990
- if (statSync(full).isFile()) zip.addFile(full, `images/${rel.split("\\").join("/")}`);
991
- }
992
- }
993
- zip.end();
994
- return [await artifactPromise, ...imageArtifacts];
995
- } finally {
996
- rmSync(dir, { recursive: true, force: true });
997
- }
998
- }
999
- export {
1000
- MAX_ANALYZED_LINK_EDGES,
1001
- SITE_EXTRACT_PAGE_CHUNK,
1002
- assembleExtractArtifacts
1003
- };