mcp-scraper 0.88.2 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -15
- package/README.md +6 -5
- package/THIRD_PARTY_NOTICES.html +203 -0
- package/dist/analytics-repository-2BMT5JNE.js +1 -0
- package/dist/bin/api-server.js +2 -41
- package/dist/bin/mcp-scraper-cli.js +39 -756
- package/dist/bin/mcp-scraper-core.js +1 -60
- package/dist/bin/mcp-scraper-install.js +2 -25
- package/dist/bin/mcp-stdio-server.js +1 -19
- package/dist/bin/paa-harvest.js +1 -41
- package/dist/chunk-3GP5CYZX.js +1 -0
- package/dist/chunk-4AI7DOS7.js +59 -0
- package/dist/chunk-4FROKQJN.js +1 -0
- package/dist/chunk-7YGVI5J4.js +21710 -0
- package/dist/chunk-CCYWSJNG.js +16 -0
- package/dist/chunk-CFI6CXIV.js +182 -0
- package/dist/chunk-E2WRWV3A.js +1 -0
- package/dist/chunk-GMWKPIYX.js +1172 -0
- package/dist/chunk-HDPYG3XV.js +102 -0
- package/dist/chunk-HE45FFBU.js +1 -0
- package/dist/chunk-HUV2WTRW.js +1 -0
- package/dist/chunk-KJQXUZ4Y.js +4 -0
- package/dist/chunk-L4CGLFPU.js +4 -0
- package/dist/chunk-M22MM4N4.js +84 -0
- package/dist/chunk-MASR22K4.js +73 -0
- package/dist/chunk-MZN4U5BL.js +1 -0
- package/dist/chunk-PUHFVA7P.js +1280 -0
- package/dist/chunk-QPWPR5XG.js +10 -0
- package/dist/chunk-TMB56NCA.js +1 -0
- package/dist/chunk-TXENITMS.js +20 -0
- package/dist/chunk-W2BVJ7S2.js +13 -0
- package/dist/chunk-WO3N5FH2.js +5 -0
- package/dist/chunk-WSCGYRWA.js +2595 -0
- package/dist/chunk-X54CQLK2.js +1 -0
- package/dist/chunk-XLWNEVUZ.js +27 -0
- package/dist/chunk-XPZVJIZ2.js +100 -0
- package/dist/chunk-YQZGZBB4.js +1 -0
- package/dist/chunk-Z2QGQJS2.js +1 -0
- package/dist/db-F2MX63GI.js +1 -0
- package/dist/extract-bundle-SNUIHM3J.js +26 -0
- package/dist/gmail-service-BZ3H75XC.js +1 -0
- package/dist/index.cjs +21750 -6045
- package/dist/index.d.cts +14 -14
- package/dist/index.d.ts +14 -14
- package/dist/index.js +18 -315
- package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
- package/dist/location-data-repository-OTWHWMV6.js +1 -0
- package/dist/server-RFR2A5UJ.js +7303 -0
- package/dist/site-extract-repository-SE776XDC.js +1 -0
- package/dist/worker-XUDSM3AL.js +1 -0
- package/package.json +17 -124
- package/dist/analytics-repository-GGJJCVVP.js +0 -194
- package/dist/chunk-4QMUF6XM.js +0 -1013
- package/dist/chunk-6HAV7LCE.js +0 -265
- package/dist/chunk-ABF2CGOZ.js +0 -113
- package/dist/chunk-C5Z4OFKW.js +0 -404
- package/dist/chunk-DNM65UCK.js +0 -299
- package/dist/chunk-EQGTEHLZ.js +0 -592
- package/dist/chunk-F5GQJWZU.js +0 -732
- package/dist/chunk-GGZEC22A.js +0 -215
- package/dist/chunk-GXBZXWXB.js +0 -184
- package/dist/chunk-IHXAXYIS.js +0 -843
- package/dist/chunk-K3Z5AQYE.js +0 -683
- package/dist/chunk-K45K75OF.js +0 -6
- package/dist/chunk-LFW2FRPJ.js +0 -224
- package/dist/chunk-MZDNZQWT.js +0 -2078
- package/dist/chunk-NVUKO5NN.js +0 -256
- package/dist/chunk-OM7HVEJ3.js +0 -26
- package/dist/chunk-OPQIGAFB.js +0 -286
- package/dist/chunk-OZJMVCDK.js +0 -16
- package/dist/chunk-P7FWOMU7.js +0 -505
- package/dist/chunk-PGJQDMC2.js +0 -383
- package/dist/chunk-PKZS6SHW.js +0 -33139
- package/dist/chunk-RJ7JVYKU.js +0 -68
- package/dist/chunk-S24LFPL7.js +0 -5262
- package/dist/chunk-T3MZISOF.js +0 -240
- package/dist/chunk-UZPTGUDV.js +0 -1915
- package/dist/chunk-X623GTBV.js +0 -8290
- package/dist/chunk-YXNDOQXN.js +0 -4018
- package/dist/db-Z34LPZNR.js +0 -284
- package/dist/extract-bundle-565SBZCR.js +0 -1003
- package/dist/gmail-service-E6ALS7JG.js +0 -25
- package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
- package/dist/location-data-repository-WPRG62GE.js +0 -34
- package/dist/server-SQZ3A7SY.js +0 -86606
- package/dist/site-extract-repository-VYFZASPU.js +0 -69
- package/dist/worker-LDCAULWL.js +0 -146
|
@@ -1,1003 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
commonsEmbedDim,
|
|
3
|
-
commonsEmbedModel,
|
|
4
|
-
downloadAsset,
|
|
5
|
-
embedCommonsTexts,
|
|
6
|
-
publicizeExtractionFailure
|
|
7
|
-
} from "./chunk-F5GQJWZU.js";
|
|
8
|
-
import {
|
|
9
|
-
createSiteExtractBundleArtifactStream,
|
|
10
|
-
createSiteExtractContentReader,
|
|
11
|
-
createSiteExtractImageArtifact,
|
|
12
|
-
extractJobLimitInfo
|
|
13
|
-
} from "./chunk-IHXAXYIS.js";
|
|
14
|
-
import "./chunk-OZJMVCDK.js";
|
|
15
|
-
import {
|
|
16
|
-
computeIssues,
|
|
17
|
-
renderImageSection,
|
|
18
|
-
renderIssueReport,
|
|
19
|
-
renderLinkReport
|
|
20
|
-
} from "./chunk-PGJQDMC2.js";
|
|
21
|
-
import "./chunk-DNM65UCK.js";
|
|
22
|
-
import "./chunk-4QMUF6XM.js";
|
|
23
|
-
import "./chunk-GGZEC22A.js";
|
|
24
|
-
import {
|
|
25
|
-
getDb
|
|
26
|
-
} from "./chunk-YXNDOQXN.js";
|
|
27
|
-
|
|
28
|
-
// src/api/extract-bundle.ts
|
|
29
|
-
import { createWriteStream, mkdirSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "fs";
|
|
30
|
-
import { createHash as createHash2 } from "crypto";
|
|
31
|
-
import { basename, join } from "path";
|
|
32
|
-
import { tmpdir } from "os";
|
|
33
|
-
import { createGzip } from "zlib";
|
|
34
|
-
import { once } from "events";
|
|
35
|
-
import { finished } from "stream/promises";
|
|
36
|
-
import { ZipFile } from "yazl";
|
|
37
|
-
import pLimit from "p-limit";
|
|
38
|
-
|
|
39
|
-
// src/api/site-content-similarity.ts
|
|
40
|
-
import { createHash } from "crypto";
|
|
41
|
-
var MAX_SIMILARITY_PAGES = 500;
|
|
42
|
-
var DEFAULT_SIMILARITY_THRESHOLD = 0.9;
|
|
43
|
-
var DEFAULT_SIMILARITY_MAX_PAIRS = 1e4;
|
|
44
|
-
var MAX_SIMILARITY_PAIRS = 5e4;
|
|
45
|
-
function round(value, places) {
|
|
46
|
-
const scale = 10 ** places;
|
|
47
|
-
return Math.round(value * scale) / scale;
|
|
48
|
-
}
|
|
49
|
-
function cosineSimilarity(a, b) {
|
|
50
|
-
if (a.length === 0 || a.length !== b.length) throw new Error("Similarity vectors must have the same non-zero dimension.");
|
|
51
|
-
let dot = 0;
|
|
52
|
-
let normA = 0;
|
|
53
|
-
let normB = 0;
|
|
54
|
-
for (let index = 0; index < a.length; index++) {
|
|
55
|
-
dot += a[index] * b[index];
|
|
56
|
-
normA += a[index] * a[index];
|
|
57
|
-
normB += b[index] * b[index];
|
|
58
|
-
}
|
|
59
|
-
if (normA === 0 || normB === 0) return 0;
|
|
60
|
-
return Math.max(-1, Math.min(1, dot / (Math.sqrt(normA) * Math.sqrt(normB))));
|
|
61
|
-
}
|
|
62
|
-
function corpusHash(pages) {
|
|
63
|
-
const digest = createHash("sha256");
|
|
64
|
-
for (const page of [...pages].sort((a, b) => a.url.localeCompare(b.url))) {
|
|
65
|
-
digest.update(page.url);
|
|
66
|
-
digest.update("\0");
|
|
67
|
-
digest.update(page.contentHash);
|
|
68
|
-
digest.update("\0");
|
|
69
|
-
}
|
|
70
|
-
return digest.digest("hex");
|
|
71
|
-
}
|
|
72
|
-
function markdownBlockSignature(block) {
|
|
73
|
-
const normalized = block.replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1").replace(/\[([^\]]+)\]\([^)]*\)/g, "$1").replace(/https?:\/\/\S+/gi, " ").replace(/&(?:amp|nbsp|quot|#39);/gi, " ").replace(/[#>*_`~|\\-]+/g, " ").replace(/[^\p{L}\p{N}]+/gu, " ").replace(/\s+/g, " ").trim().toLowerCase();
|
|
74
|
-
return normalized.length >= 12 ? normalized : null;
|
|
75
|
-
}
|
|
76
|
-
function prepareEmbeddingTexts(pages) {
|
|
77
|
-
const blocksByPage = pages.map((page) => page.bodyMarkdown.split(/\n{2,}/).map((block) => block.trim()).filter(Boolean));
|
|
78
|
-
const minimumPageCount = pages.length >= 4 ? Math.max(3, Math.ceil(pages.length * 0.5)) : null;
|
|
79
|
-
const pageFrequency = /* @__PURE__ */ new Map();
|
|
80
|
-
for (const blocks of blocksByPage) {
|
|
81
|
-
const pageSignatures = new Set(blocks.map(markdownBlockSignature).filter((value) => Boolean(value)));
|
|
82
|
-
for (const signature of pageSignatures) pageFrequency.set(signature, (pageFrequency.get(signature) ?? 0) + 1);
|
|
83
|
-
}
|
|
84
|
-
const repeated = /* @__PURE__ */ new Set();
|
|
85
|
-
if (minimumPageCount != null) {
|
|
86
|
-
for (const [signature, count] of pageFrequency) {
|
|
87
|
-
if (count >= minimumPageCount) repeated.add(signature);
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
let removedCharacters = 0;
|
|
91
|
-
const texts = pages.map((page, pageIndex) => {
|
|
92
|
-
const kept = [];
|
|
93
|
-
for (const block of blocksByPage[pageIndex]) {
|
|
94
|
-
const signature = markdownBlockSignature(block);
|
|
95
|
-
if (signature && repeated.has(signature)) {
|
|
96
|
-
removedCharacters += block.length;
|
|
97
|
-
continue;
|
|
98
|
-
}
|
|
99
|
-
kept.push(block);
|
|
100
|
-
}
|
|
101
|
-
const cleaned = kept.join("\n\n").trim();
|
|
102
|
-
const content = cleaned.length >= 60 ? cleaned : page.bodyMarkdown;
|
|
103
|
-
return `${page.title?.trim() || "Untitled page"}
|
|
104
|
-
|
|
105
|
-
${content}`;
|
|
106
|
-
});
|
|
107
|
-
return {
|
|
108
|
-
texts,
|
|
109
|
-
minimumPageCount,
|
|
110
|
-
removedBlockSignatures: repeated.size,
|
|
111
|
-
removedCharacters
|
|
112
|
-
};
|
|
113
|
-
}
|
|
114
|
-
var DisjointSet = class {
|
|
115
|
-
parent;
|
|
116
|
-
constructor(size) {
|
|
117
|
-
this.parent = Array.from({ length: size }, (_, index) => index);
|
|
118
|
-
}
|
|
119
|
-
find(value) {
|
|
120
|
-
const parent = this.parent[value];
|
|
121
|
-
if (parent !== value) this.parent[value] = this.find(parent);
|
|
122
|
-
return this.parent[value];
|
|
123
|
-
}
|
|
124
|
-
union(a, b) {
|
|
125
|
-
const rootA = this.find(a);
|
|
126
|
-
const rootB = this.find(b);
|
|
127
|
-
if (rootA !== rootB) this.parent[rootB] = rootA;
|
|
128
|
-
}
|
|
129
|
-
};
|
|
130
|
-
async function analyzeSiteContentSimilarity(pages, options = {}) {
|
|
131
|
-
const eligible = pages.filter((page) => page.bodyMarkdown.trim().length > 0).slice(0, MAX_SIMILARITY_PAGES);
|
|
132
|
-
const requestedThreshold = Number.isFinite(options.threshold) ? options.threshold : DEFAULT_SIMILARITY_THRESHOLD;
|
|
133
|
-
const requestedMaxPairs = Number.isFinite(options.maxPairs) ? options.maxPairs : DEFAULT_SIMILARITY_MAX_PAIRS;
|
|
134
|
-
const threshold = Math.max(0, Math.min(1, requestedThreshold));
|
|
135
|
-
const maxPairs = Math.max(1, Math.min(MAX_SIMILARITY_PAIRS, Math.floor(requestedMaxPairs)));
|
|
136
|
-
const model = options.model ?? commonsEmbedModel();
|
|
137
|
-
const dimensions = options.dimensions ?? commonsEmbedDim();
|
|
138
|
-
const embedTexts = options.embedTexts ?? embedCommonsTexts;
|
|
139
|
-
const prepared = prepareEmbeddingTexts(eligible);
|
|
140
|
-
const uniqueTexts = [];
|
|
141
|
-
const textIndexByHash = /* @__PURE__ */ new Map();
|
|
142
|
-
const pageTextIndexes = [];
|
|
143
|
-
for (const [pageIndex, page] of eligible.entries()) {
|
|
144
|
-
const key = page.contentHash || createHash("sha256").update(page.bodyMarkdown).digest("hex");
|
|
145
|
-
let textIndex = textIndexByHash.get(key);
|
|
146
|
-
if (textIndex == null) {
|
|
147
|
-
textIndex = uniqueTexts.length;
|
|
148
|
-
textIndexByHash.set(key, textIndex);
|
|
149
|
-
uniqueTexts.push(prepared.texts[pageIndex]);
|
|
150
|
-
}
|
|
151
|
-
pageTextIndexes.push(textIndex);
|
|
152
|
-
}
|
|
153
|
-
const uniqueVectors = [];
|
|
154
|
-
for (let offset = 0; offset < uniqueTexts.length; offset += 64) {
|
|
155
|
-
uniqueVectors.push(...await embedTexts(uniqueTexts.slice(offset, offset + 64)));
|
|
156
|
-
}
|
|
157
|
-
if (uniqueVectors.length !== uniqueTexts.length) {
|
|
158
|
-
throw new Error(`Embedding provider returned ${uniqueVectors.length} vectors for ${uniqueTexts.length} unique page bodies.`);
|
|
159
|
-
}
|
|
160
|
-
const vectors = pageTextIndexes.map((index) => uniqueVectors[index]);
|
|
161
|
-
const sets = new DisjointSet(eligible.length);
|
|
162
|
-
const allScores = [];
|
|
163
|
-
const qualifying = [];
|
|
164
|
-
for (let source = 0; source < eligible.length; source++) {
|
|
165
|
-
for (let target = source + 1; target < eligible.length; target++) {
|
|
166
|
-
const score = cosineSimilarity(vectors[source], vectors[target]);
|
|
167
|
-
allScores.push(score);
|
|
168
|
-
if (score < threshold) continue;
|
|
169
|
-
qualifying.push({ source, target, score });
|
|
170
|
-
sets.union(source, target);
|
|
171
|
-
}
|
|
172
|
-
}
|
|
173
|
-
qualifying.sort((a, b) => b.score - a.score || eligible[a.source].url.localeCompare(eligible[b.source].url) || eligible[a.target].url.localeCompare(eligible[b.target].url));
|
|
174
|
-
const sortedScores = [...allScores].sort((a, b) => a - b);
|
|
175
|
-
const quantile = (value) => {
|
|
176
|
-
if (!sortedScores.length) return null;
|
|
177
|
-
return round(sortedScores[Math.floor((sortedScores.length - 1) * value)], 6);
|
|
178
|
-
};
|
|
179
|
-
const membersByRoot = /* @__PURE__ */ new Map();
|
|
180
|
-
for (let index = 0; index < eligible.length; index++) {
|
|
181
|
-
const root = sets.find(index);
|
|
182
|
-
const members = membersByRoot.get(root) ?? [];
|
|
183
|
-
members.push(index);
|
|
184
|
-
membersByRoot.set(root, members);
|
|
185
|
-
}
|
|
186
|
-
const clusterIdByPage = /* @__PURE__ */ new Map();
|
|
187
|
-
const clusters = [...membersByRoot.values()].filter((members) => members.length > 1).sort((a, b) => b.length - a.length || eligible[a[0]].url.localeCompare(eligible[b[0]].url)).map((members, index) => {
|
|
188
|
-
const clusterId = `cluster-${String(index + 1).padStart(3, "0")}`;
|
|
189
|
-
for (const member of members) clusterIdByPage.set(member, clusterId);
|
|
190
|
-
return {
|
|
191
|
-
clusterId,
|
|
192
|
-
pageCount: members.length,
|
|
193
|
-
urls: members.map((member) => eligible[member].url).sort()
|
|
194
|
-
};
|
|
195
|
-
});
|
|
196
|
-
const rows = qualifying.slice(0, maxPairs).map((pair, pairIndex) => {
|
|
197
|
-
const source = eligible[pair.source];
|
|
198
|
-
const target = eligible[pair.target];
|
|
199
|
-
const similarity = round(pair.score, 6);
|
|
200
|
-
return {
|
|
201
|
-
sourceUrl: source.url,
|
|
202
|
-
sourceTitle: source.title,
|
|
203
|
-
targetUrl: target.url,
|
|
204
|
-
targetTitle: target.title,
|
|
205
|
-
similarity,
|
|
206
|
-
similarityPercent: round(similarity * 100, 2),
|
|
207
|
-
corpusPercentile: sortedScores.length <= 1 ? 100 : round((1 - pairIndex / (sortedScores.length - 1)) * 100, 2),
|
|
208
|
-
exactContentDuplicate: Boolean(source.contentHash && source.contentHash === target.contentHash),
|
|
209
|
-
sourceWordCount: source.wordCount,
|
|
210
|
-
targetWordCount: target.wordCount,
|
|
211
|
-
clusterId: clusterIdByPage.get(pair.source) ?? null
|
|
212
|
-
};
|
|
213
|
-
});
|
|
214
|
-
const corpusSha256 = corpusHash(eligible);
|
|
215
|
-
const analysisSha256 = createHash("sha256").update(JSON.stringify({
|
|
216
|
-
corpusSha256,
|
|
217
|
-
model,
|
|
218
|
-
dimensions,
|
|
219
|
-
threshold,
|
|
220
|
-
maxPairs,
|
|
221
|
-
boilerplateRemoval: {
|
|
222
|
-
method: "corpus_repeated_markdown_blocks",
|
|
223
|
-
minimumPageCount: prepared.minimumPageCount,
|
|
224
|
-
removedBlockSignatures: prepared.removedBlockSignatures
|
|
225
|
-
}
|
|
226
|
-
})).digest("hex");
|
|
227
|
-
return {
|
|
228
|
-
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
229
|
-
provider: "jina",
|
|
230
|
-
model,
|
|
231
|
-
dimensions,
|
|
232
|
-
threshold,
|
|
233
|
-
requestedMaxPairs: maxPairs,
|
|
234
|
-
comparedPages: eligible.length,
|
|
235
|
-
possiblePairs: eligible.length * Math.max(0, eligible.length - 1) / 2,
|
|
236
|
-
qualifyingPairs: qualifying.length,
|
|
237
|
-
returnedPairs: rows.length,
|
|
238
|
-
pairsTruncated: qualifying.length > rows.length,
|
|
239
|
-
corpusSha256,
|
|
240
|
-
analysisSha256,
|
|
241
|
-
scoreDistribution: {
|
|
242
|
-
min: quantile(0),
|
|
243
|
-
p25: quantile(0.25),
|
|
244
|
-
median: quantile(0.5),
|
|
245
|
-
p75: quantile(0.75),
|
|
246
|
-
p90: quantile(0.9),
|
|
247
|
-
max: quantile(1)
|
|
248
|
-
},
|
|
249
|
-
boilerplateRemoval: {
|
|
250
|
-
method: "corpus_repeated_markdown_blocks",
|
|
251
|
-
minimumPageCount: prepared.minimumPageCount,
|
|
252
|
-
removedBlockSignatures: prepared.removedBlockSignatures,
|
|
253
|
-
removedCharacters: prepared.removedCharacters
|
|
254
|
-
},
|
|
255
|
-
rows,
|
|
256
|
-
clusters
|
|
257
|
-
};
|
|
258
|
-
}
|
|
259
|
-
|
|
260
|
-
// src/api/extract-bundle.ts
|
|
261
|
-
var SITE_EXTRACT_PAGE_CHUNK = 25;
|
|
262
|
-
var MAX_IMAGES_PER_PAGE = 20;
|
|
263
|
-
var MAX_IMAGES_PER_SITE = 500;
|
|
264
|
-
var IMAGE_DOWNLOAD_CONCURRENCY = 8;
|
|
265
|
-
var MAX_IMAGE_BYTES = 10 * 1024 * 1024;
|
|
266
|
-
var MAX_SITE_IMAGE_BYTES = 100 * 1024 * 1024;
|
|
267
|
-
var MAX_ANALYZED_LINK_EDGES = 25e4;
|
|
268
|
-
function normalize(u) {
|
|
269
|
-
return u.split("#")[0].replace(/\/$/, "");
|
|
270
|
-
}
|
|
271
|
-
function registrableDomain(host) {
|
|
272
|
-
const h = host.replace(/^www\./, "").toLowerCase();
|
|
273
|
-
const parts = h.split(".");
|
|
274
|
-
return parts.length <= 2 ? h : parts.slice(-2).join(".");
|
|
275
|
-
}
|
|
276
|
-
function safeImageFilename(url, index) {
|
|
277
|
-
try {
|
|
278
|
-
const u = new URL(url);
|
|
279
|
-
const base = basename(u.pathname).replace(/[^a-zA-Z0-9._-]/g, "_").slice(0, 80);
|
|
280
|
-
return base || `image-${index}`;
|
|
281
|
-
} catch {
|
|
282
|
-
return `image-${index}`;
|
|
283
|
-
}
|
|
284
|
-
}
|
|
285
|
-
function slugFactory() {
|
|
286
|
-
const counts = /* @__PURE__ */ new Map();
|
|
287
|
-
return (url) => {
|
|
288
|
-
const base = url.replace(/^https?:\/\//, "").replace(/[^a-zA-Z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80) || "page";
|
|
289
|
-
const n = counts.get(base) ?? 0;
|
|
290
|
-
counts.set(base, n + 1);
|
|
291
|
-
return n ? `${base}-${n}` : base;
|
|
292
|
-
};
|
|
293
|
-
}
|
|
294
|
-
async function* pageChunks(jobId) {
|
|
295
|
-
const db = getDb();
|
|
296
|
-
let lastRowId = 0;
|
|
297
|
-
for (; ; ) {
|
|
298
|
-
const res = await db.execute({
|
|
299
|
-
sql: `SELECT rowid AS cursor_id, page
|
|
300
|
-
FROM site_extract_pages
|
|
301
|
-
WHERE job_id = ? AND rowid > ?
|
|
302
|
-
ORDER BY rowid
|
|
303
|
-
LIMIT ${SITE_EXTRACT_PAGE_CHUNK}`,
|
|
304
|
-
args: [jobId, lastRowId]
|
|
305
|
-
});
|
|
306
|
-
if (!res.rows.length) return;
|
|
307
|
-
lastRowId = Number(res.rows.at(-1)?.cursor_id ?? lastRowId);
|
|
308
|
-
if (!Number.isSafeInteger(lastRowId) || lastRowId <= 0) throw new Error("site extract page cursor is invalid");
|
|
309
|
-
yield res.rows.map((r) => JSON.parse(String(r.page)));
|
|
310
|
-
}
|
|
311
|
-
}
|
|
312
|
-
function csvCell(value) {
|
|
313
|
-
const text = value == null ? "" : String(value);
|
|
314
|
-
return /[",\r\n]/.test(text) ? `"${text.replace(/"/g, '""')}"` : text;
|
|
315
|
-
}
|
|
316
|
-
function similarityTableRow(row) {
|
|
317
|
-
return {
|
|
318
|
-
source_url: row.sourceUrl,
|
|
319
|
-
source_title: row.sourceTitle,
|
|
320
|
-
target_url: row.targetUrl,
|
|
321
|
-
target_title: row.targetTitle,
|
|
322
|
-
similarity: row.similarity,
|
|
323
|
-
similarity_percent: row.similarityPercent,
|
|
324
|
-
corpus_percentile: row.corpusPercentile,
|
|
325
|
-
exact_content_duplicate: row.exactContentDuplicate,
|
|
326
|
-
source_word_count: row.sourceWordCount,
|
|
327
|
-
target_word_count: row.targetWordCount,
|
|
328
|
-
cluster_id: row.clusterId
|
|
329
|
-
};
|
|
330
|
-
}
|
|
331
|
-
async function assembleExtractArtifacts(job, extras = {}) {
|
|
332
|
-
const dir = join(tmpdir(), `extract-bundle-${job.id}-${Date.now()}`);
|
|
333
|
-
const downloadImages = job.options.downloadImages === true;
|
|
334
|
-
if (job.userId == null) throw new Error("site extract artifact owner is missing");
|
|
335
|
-
const readPageContent = createSiteExtractContentReader(String(job.userId));
|
|
336
|
-
mkdirSync(dir, { recursive: true });
|
|
337
|
-
if (downloadImages) mkdirSync(join(dir, "images"), { recursive: true });
|
|
338
|
-
const captureRenderedDom = job.options.captureRenderedDom === true;
|
|
339
|
-
const semanticSimilarity = job.options.semanticSimilarity === true;
|
|
340
|
-
if (captureRenderedDom) mkdirSync(join(dir, "rendered-dom"), { recursive: true });
|
|
341
|
-
try {
|
|
342
|
-
const metas = [];
|
|
343
|
-
const similarityPages = [];
|
|
344
|
-
const renderedDomManifest = [];
|
|
345
|
-
const domSlug = slugFactory();
|
|
346
|
-
const pageExportEntries = [];
|
|
347
|
-
const waybackTimeline = job.options.waybackTimeline;
|
|
348
|
-
const statusByUrl = /* @__PURE__ */ new Map();
|
|
349
|
-
for await (const chunk of pageChunks(job.id)) {
|
|
350
|
-
for (const p of chunk) {
|
|
351
|
-
statusByUrl.set(normalize(p.url), p.status);
|
|
352
|
-
metas.push({
|
|
353
|
-
url: p.url,
|
|
354
|
-
status: p.status,
|
|
355
|
-
title: p.title,
|
|
356
|
-
titleLength: p.titleLength,
|
|
357
|
-
titlePixels: p.titlePixels,
|
|
358
|
-
metaDescription: p.metaDescription,
|
|
359
|
-
metaDescLength: p.metaDescLength,
|
|
360
|
-
h1: p.h1,
|
|
361
|
-
h1_2: p.h1_2,
|
|
362
|
-
h2Count: p.h2Count,
|
|
363
|
-
extractionStatus: p.extractionStatus,
|
|
364
|
-
indexable: p.indexable,
|
|
365
|
-
indexabilityReason: p.indexabilityReason,
|
|
366
|
-
canonicalUrl: p.canonicalUrl ? p.url : null,
|
|
367
|
-
wordCount: p.wordCount,
|
|
368
|
-
contentHash: p.contentHash,
|
|
369
|
-
imagesMissingAlt: p.imagesMissingAlt,
|
|
370
|
-
schemaTypes: p.schemaTypes.length > 0 ? ["present"] : [],
|
|
371
|
-
outlinks: []
|
|
372
|
-
});
|
|
373
|
-
const pageId = p.pageId ?? createHash2("sha256").update(p.archivedUrl ?? p.url).digest("hex");
|
|
374
|
-
const pageDir = join(dir, "pages", pageId);
|
|
375
|
-
mkdirSync(pageDir, { recursive: true });
|
|
376
|
-
const content = p.contentRef ? await readPageContent(p.contentRef) : null;
|
|
377
|
-
const html = content?.html ?? null;
|
|
378
|
-
const markdown = content?.markdown ?? p.bodyMarkdown ?? null;
|
|
379
|
-
if (semanticSimilarity && p.extractionStatus === "successful" && markdown?.trim()) {
|
|
380
|
-
similarityPages.push({
|
|
381
|
-
url: p.url,
|
|
382
|
-
title: p.title,
|
|
383
|
-
bodyMarkdown: markdown,
|
|
384
|
-
contentHash: p.contentHash,
|
|
385
|
-
wordCount: p.wordCount
|
|
386
|
-
});
|
|
387
|
-
}
|
|
388
|
-
if (captureRenderedDom && html != null) {
|
|
389
|
-
const relativePath = `rendered-dom/${domSlug(p.url)}.html.txt`;
|
|
390
|
-
writeFileSync(join(dir, relativePath), html);
|
|
391
|
-
renderedDomManifest.push({
|
|
392
|
-
url: p.url,
|
|
393
|
-
path: relativePath,
|
|
394
|
-
bytes: Buffer.byteLength(html),
|
|
395
|
-
truncated: p.renderedDomTruncated === true,
|
|
396
|
-
sanitized: p.renderedDomSanitized === true
|
|
397
|
-
});
|
|
398
|
-
}
|
|
399
|
-
const htmlPath = html == null ? null : `pages/${pageId}/page.html`;
|
|
400
|
-
const markdownPath = markdown == null ? null : `pages/${pageId}/page.md`;
|
|
401
|
-
const legacyMarkdownPath = markdown == null || !p.archiveRequestedMonth ? null : `pages/${p.archiveRequestedMonth}/${pageId}.md`;
|
|
402
|
-
if (html != null && htmlPath) writeFileSync(join(dir, htmlPath), html);
|
|
403
|
-
if (markdown != null && markdownPath) writeFileSync(join(dir, markdownPath), markdown);
|
|
404
|
-
if (markdown != null && legacyMarkdownPath) {
|
|
405
|
-
mkdirSync(join(dir, "pages", p.archiveRequestedMonth), { recursive: true });
|
|
406
|
-
writeFileSync(join(dir, legacyMarkdownPath), markdown);
|
|
407
|
-
}
|
|
408
|
-
const { bodyMarkdown: _body, contentRef: _contentRef, discoveryLinks: _discovery, ...metadata } = p;
|
|
409
|
-
const pageRecord = {
|
|
410
|
-
version: "site-export-page.v1",
|
|
411
|
-
...metadata,
|
|
412
|
-
pageId,
|
|
413
|
-
content: {
|
|
414
|
-
htmlPath,
|
|
415
|
-
markdownPath,
|
|
416
|
-
htmlBytes: html == null ? 0 : Buffer.byteLength(html),
|
|
417
|
-
markdownBytes: markdown == null ? 0 : Buffer.byteLength(markdown),
|
|
418
|
-
htmlSha256: html == null ? null : createHash2("sha256").update(html).digest("hex"),
|
|
419
|
-
markdownSha256: markdown == null ? null : createHash2("sha256").update(markdown).digest("hex")
|
|
420
|
-
}
|
|
421
|
-
};
|
|
422
|
-
const pageJson = JSON.stringify(pageRecord, null, 2);
|
|
423
|
-
const jsonPath = `pages/${pageId}/page.json`;
|
|
424
|
-
writeFileSync(join(dir, jsonPath), pageJson);
|
|
425
|
-
pageExportEntries.push({
|
|
426
|
-
pageId,
|
|
427
|
-
url: p.archivedUrl ?? p.url,
|
|
428
|
-
jsonPath,
|
|
429
|
-
htmlPath,
|
|
430
|
-
markdownPath,
|
|
431
|
-
legacyMarkdownPath,
|
|
432
|
-
jsonSha256: createHash2("sha256").update(pageJson).digest("hex"),
|
|
433
|
-
htmlSha256: pageRecord.content.htmlSha256,
|
|
434
|
-
markdownSha256: pageRecord.content.markdownSha256,
|
|
435
|
-
htmlBytes: pageRecord.content.htmlBytes,
|
|
436
|
-
markdownBytes: pageRecord.content.markdownBytes
|
|
437
|
-
});
|
|
438
|
-
}
|
|
439
|
-
}
|
|
440
|
-
const gz = createGzip();
|
|
441
|
-
const linksOut = createWriteStream(join(dir, "links.jsonl.gz"));
|
|
442
|
-
gz.pipe(linksOut);
|
|
443
|
-
const gzWrite = async (line) => {
|
|
444
|
-
if (!gz.write(line)) await once(gz, "drain");
|
|
445
|
-
};
|
|
446
|
-
const schemaOut = createWriteStream(join(dir, "schema.jsonl"));
|
|
447
|
-
const schemaWrite = async (line) => {
|
|
448
|
-
if (!schemaOut.write(line)) await once(schemaOut, "drain");
|
|
449
|
-
};
|
|
450
|
-
const failuresOut = createWriteStream(join(dir, "failures.jsonl"));
|
|
451
|
-
const failureWrite = async (line) => {
|
|
452
|
-
if (!failuresOut.write(line)) await once(failuresOut, "drain");
|
|
453
|
-
};
|
|
454
|
-
const inboundFrom = /* @__PURE__ */ new Map();
|
|
455
|
-
const anchorCounts = /* @__PURE__ */ new Map();
|
|
456
|
-
const adjacency = /* @__PURE__ */ new Map();
|
|
457
|
-
const outCounts = /* @__PURE__ */ new Map();
|
|
458
|
-
const domMap = /* @__PURE__ */ new Map();
|
|
459
|
-
const brokenLinkPages = /* @__PURE__ */ new Set();
|
|
460
|
-
let internalTotal = 0;
|
|
461
|
-
let externalTotal = 0;
|
|
462
|
-
let brokenInternal = 0;
|
|
463
|
-
let streamedLinkEdges = 0;
|
|
464
|
-
let analyzedLinkEdges = 0;
|
|
465
|
-
let linkAnalysisTruncated = false;
|
|
466
|
-
let siteReg = "";
|
|
467
|
-
try {
|
|
468
|
-
siteReg = registrableDomain(new URL(job.startUrl).hostname);
|
|
469
|
-
} catch {
|
|
470
|
-
siteReg = "";
|
|
471
|
-
}
|
|
472
|
-
const imageLimit = pLimit(IMAGE_DOWNLOAD_CONCURRENCY);
|
|
473
|
-
const imageDownloads = [];
|
|
474
|
-
const imageManifest = [];
|
|
475
|
-
const imageArtifacts = [];
|
|
476
|
-
let imagesQueued = 0;
|
|
477
|
-
let imagesFailed = 0;
|
|
478
|
-
let imageBytesDownloaded = 0;
|
|
479
|
-
const consumeImageBytes = (bytes) => {
|
|
480
|
-
if (imageBytesDownloaded + bytes > MAX_SITE_IMAGE_BYTES) return false;
|
|
481
|
-
imageBytesDownloaded += bytes;
|
|
482
|
-
return true;
|
|
483
|
-
};
|
|
484
|
-
for await (const chunk of pageChunks(job.id)) {
|
|
485
|
-
for (const p of chunk) {
|
|
486
|
-
if (p.schema?.length) await schemaWrite(JSON.stringify({ url: p.url, schema: p.schema }) + "\n");
|
|
487
|
-
if (p.extractionStatus === "failed") {
|
|
488
|
-
const { failureCode, failureReason } = publicizeExtractionFailure(p.failureCode, p.failureReason);
|
|
489
|
-
await failureWrite(JSON.stringify({
|
|
490
|
-
url: p.url,
|
|
491
|
-
status: p.status,
|
|
492
|
-
via: p.via,
|
|
493
|
-
failureCode,
|
|
494
|
-
failureReason,
|
|
495
|
-
fetchAttempts: p.fetchAttempts ?? []
|
|
496
|
-
}) + "\n");
|
|
497
|
-
}
|
|
498
|
-
if (p.imageLinks?.length) {
|
|
499
|
-
for (const [idx, imgUrl] of p.imageLinks.entries()) {
|
|
500
|
-
const imageId = createHash2("sha256").update(`${p.url}\0${imgUrl}`).digest("hex");
|
|
501
|
-
const record = {
|
|
502
|
-
imageId,
|
|
503
|
-
sourcePage: p.url,
|
|
504
|
-
sourceUrl: imgUrl,
|
|
505
|
-
path: null,
|
|
506
|
-
status: "not_requested",
|
|
507
|
-
mimeType: null,
|
|
508
|
-
bytes: null,
|
|
509
|
-
sha256: null,
|
|
510
|
-
failureCode: null,
|
|
511
|
-
failureReason: null,
|
|
512
|
-
artifactId: null
|
|
513
|
-
};
|
|
514
|
-
imageManifest.push(record);
|
|
515
|
-
if (!downloadImages) continue;
|
|
516
|
-
if (idx >= MAX_IMAGES_PER_PAGE) {
|
|
517
|
-
record.status = "skipped_page_limit";
|
|
518
|
-
record.failureCode = "page_image_limit";
|
|
519
|
-
record.failureReason = `Only the first ${MAX_IMAGES_PER_PAGE} image candidates are downloaded per page.`;
|
|
520
|
-
continue;
|
|
521
|
-
}
|
|
522
|
-
if (imagesQueued >= MAX_IMAGES_PER_SITE) {
|
|
523
|
-
record.status = "skipped_site_limit";
|
|
524
|
-
record.failureCode = "site_image_limit";
|
|
525
|
-
record.failureReason = `Only the first ${MAX_IMAGES_PER_SITE} image candidates are downloaded per site.`;
|
|
526
|
-
continue;
|
|
527
|
-
}
|
|
528
|
-
imagesQueued++;
|
|
529
|
-
const filename = `${imageId}-${safeImageFilename(imgUrl, idx)}`;
|
|
530
|
-
const outputPath = join(dir, "images");
|
|
531
|
-
imageDownloads.push(
|
|
532
|
-
imageLimit(() => downloadAsset(imgUrl, outputPath, filename, {
|
|
533
|
-
expectedType: "image",
|
|
534
|
-
maxBytes: MAX_IMAGE_BYTES,
|
|
535
|
-
consumeBytes: consumeImageBytes
|
|
536
|
-
})).then(async (result) => {
|
|
537
|
-
const bytes = readFileSync(result.savedPath);
|
|
538
|
-
const artifact = await createSiteExtractImageArtifact({
|
|
539
|
-
ownerId: String(job.userId),
|
|
540
|
-
jobId: job.id,
|
|
541
|
-
imageId,
|
|
542
|
-
filename,
|
|
543
|
-
contentType: result.mimeType ?? "application/octet-stream",
|
|
544
|
-
content: bytes,
|
|
545
|
-
sourceUrl: imgUrl,
|
|
546
|
-
sourcePage: p.url
|
|
547
|
-
});
|
|
548
|
-
imageArtifacts.push(artifact);
|
|
549
|
-
record.status = "downloaded";
|
|
550
|
-
record.path = `images/${filename}`;
|
|
551
|
-
record.artifactId = artifact.key;
|
|
552
|
-
record.mimeType = result.mimeType;
|
|
553
|
-
record.bytes = result.sizeBytes;
|
|
554
|
-
record.sha256 = createHash2("sha256").update(bytes).digest("hex");
|
|
555
|
-
}, (error) => {
|
|
556
|
-
imagesFailed++;
|
|
557
|
-
record.status = "failed";
|
|
558
|
-
record.failureCode = "image_download_failed";
|
|
559
|
-
record.failureReason = (error instanceof Error ? error.message : String(error)).replace(/\s+/g, " ").slice(0, 500);
|
|
560
|
-
})
|
|
561
|
-
);
|
|
562
|
-
}
|
|
563
|
-
}
|
|
564
|
-
const from = normalize(p.url);
|
|
565
|
-
const oc = { int: 0, ext: 0 };
|
|
566
|
-
for (const l of p.outlinks ?? []) {
|
|
567
|
-
const nofollow = (l.rel ?? "").split(/\s+/).includes("nofollow");
|
|
568
|
-
const to = normalize(l.href);
|
|
569
|
-
const targetStatus = statusByUrl.get(to) ?? null;
|
|
570
|
-
await gzWrite(JSON.stringify({ from: p.url, to: l.href, anchor: l.anchor, rel: l.rel, internal: l.internal, nofollow, targetStatus }) + "\n");
|
|
571
|
-
streamedLinkEdges++;
|
|
572
|
-
if (analyzedLinkEdges >= MAX_ANALYZED_LINK_EDGES) {
|
|
573
|
-
linkAnalysisTruncated = true;
|
|
574
|
-
continue;
|
|
575
|
-
}
|
|
576
|
-
analyzedLinkEdges++;
|
|
577
|
-
if (l.internal) {
|
|
578
|
-
oc.int++;
|
|
579
|
-
internalTotal++;
|
|
580
|
-
if (!inboundFrom.has(to)) inboundFrom.set(to, /* @__PURE__ */ new Set());
|
|
581
|
-
inboundFrom.get(to).add(from);
|
|
582
|
-
if (l.anchor) {
|
|
583
|
-
if (!anchorCounts.has(to)) anchorCounts.set(to, /* @__PURE__ */ new Map());
|
|
584
|
-
const ac = anchorCounts.get(to);
|
|
585
|
-
ac.set(l.anchor, (ac.get(l.anchor) ?? 0) + 1);
|
|
586
|
-
}
|
|
587
|
-
if (!nofollow && (statusByUrl.get(to) === 200 || !statusByUrl.has(to))) {
|
|
588
|
-
if (!adjacency.has(from)) adjacency.set(from, /* @__PURE__ */ new Set());
|
|
589
|
-
adjacency.get(from).add(to);
|
|
590
|
-
}
|
|
591
|
-
if (targetStatus != null && targetStatus >= 400) {
|
|
592
|
-
brokenInternal++;
|
|
593
|
-
brokenLinkPages.add(p.url);
|
|
594
|
-
}
|
|
595
|
-
} else {
|
|
596
|
-
oc.ext++;
|
|
597
|
-
let domain = "";
|
|
598
|
-
try {
|
|
599
|
-
domain = registrableDomain(new URL(l.href).hostname);
|
|
600
|
-
} catch {
|
|
601
|
-
continue;
|
|
602
|
-
}
|
|
603
|
-
if (!domain || domain === siteReg) continue;
|
|
604
|
-
externalTotal++;
|
|
605
|
-
const d = domMap.get(domain) ?? { links: 0, nofollow: 0, pages: /* @__PURE__ */ new Set() };
|
|
606
|
-
d.links++;
|
|
607
|
-
if (nofollow) d.nofollow++;
|
|
608
|
-
d.pages.add(p.url);
|
|
609
|
-
domMap.set(domain, d);
|
|
610
|
-
}
|
|
611
|
-
}
|
|
612
|
-
outCounts.set(p.url, oc);
|
|
613
|
-
}
|
|
614
|
-
}
|
|
615
|
-
const outputFinished = Promise.all([finished(linksOut), finished(schemaOut), finished(failuresOut)]);
|
|
616
|
-
gz.end();
|
|
617
|
-
schemaOut.end();
|
|
618
|
-
failuresOut.end();
|
|
619
|
-
await outputFinished;
|
|
620
|
-
await Promise.all(imageDownloads);
|
|
621
|
-
const depth = /* @__PURE__ */ new Map();
|
|
622
|
-
const start = normalize(job.startUrl);
|
|
623
|
-
depth.set(start, 0);
|
|
624
|
-
const queue = [start];
|
|
625
|
-
while (queue.length > 0) {
|
|
626
|
-
const cur = queue.shift();
|
|
627
|
-
const d = depth.get(cur);
|
|
628
|
-
for (const next of adjacency.get(cur) ?? []) {
|
|
629
|
-
if (!depth.has(next)) {
|
|
630
|
-
depth.set(next, d + 1);
|
|
631
|
-
queue.push(next);
|
|
632
|
-
}
|
|
633
|
-
}
|
|
634
|
-
}
|
|
635
|
-
const metrics = /* @__PURE__ */ new Map();
|
|
636
|
-
for (const p of metas) {
|
|
637
|
-
const key = normalize(p.url);
|
|
638
|
-
const inbound = inboundFrom.get(key) ?? /* @__PURE__ */ new Set();
|
|
639
|
-
const ac = anchorCounts.get(key);
|
|
640
|
-
const topAnchors = ac ? [...ac.entries()].sort((a, b) => b[1] - a[1]).slice(0, 3).map((e) => e[0]) : [];
|
|
641
|
-
const oc = outCounts.get(p.url) ?? { int: 0, ext: 0 };
|
|
642
|
-
metrics.set(p.url, {
|
|
643
|
-
url: p.url,
|
|
644
|
-
inlinks: inbound.size,
|
|
645
|
-
uniqueInlinks: inbound.size,
|
|
646
|
-
outlinksInternal: oc.int,
|
|
647
|
-
outlinksExternal: oc.ext,
|
|
648
|
-
crawlDepth: depth.has(key) ? depth.get(key) : null,
|
|
649
|
-
orphan: inbound.size === 0 && key !== start,
|
|
650
|
-
topAnchors
|
|
651
|
-
});
|
|
652
|
-
}
|
|
653
|
-
const issues = computeIssues(metas, metrics, brokenLinkPages);
|
|
654
|
-
let reportMd = renderIssueReport(job.startUrl, metas, issues, metrics);
|
|
655
|
-
const limits = extractJobLimitInfo(job);
|
|
656
|
-
const crawlSummary = {
|
|
657
|
-
jobId: job.id,
|
|
658
|
-
startUrl: job.startUrl,
|
|
659
|
-
status: job.successfulUrls === 0 ? "failed" : job.failedUrls > 0 || job.remainingUrls > 0 || limits.creditTruncated ? "partial" : "complete",
|
|
660
|
-
...limits,
|
|
661
|
-
discovered: job.totalUrls,
|
|
662
|
-
attempted: job.attemptedUrls,
|
|
663
|
-
successful: job.successfulUrls,
|
|
664
|
-
failed: job.failedUrls,
|
|
665
|
-
remaining: job.remainingUrls,
|
|
666
|
-
linkAnalysis: {
|
|
667
|
-
truncated: linkAnalysisTruncated,
|
|
668
|
-
analyzedEdges: analyzedLinkEdges,
|
|
669
|
-
streamedEdges: streamedLinkEdges,
|
|
670
|
-
edgeCap: MAX_ANALYZED_LINK_EDGES
|
|
671
|
-
},
|
|
672
|
-
generatedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
673
|
-
};
|
|
674
|
-
reportMd = [
|
|
675
|
-
`## Crawl delivery`,
|
|
676
|
-
`- Discovered: ${crawlSummary.discovered}`,
|
|
677
|
-
`- Attempted: ${crawlSummary.attempted}`,
|
|
678
|
-
`- Successful: ${crawlSummary.successful}`,
|
|
679
|
-
`- Failed: ${crawlSummary.failed}`,
|
|
680
|
-
`- Remaining: ${crawlSummary.remaining}`,
|
|
681
|
-
`- Requested page cap: ${limits.requestedMaxPages}`,
|
|
682
|
-
`- Funded page cap: ${limits.effectiveMaxPages}`,
|
|
683
|
-
`- Credit-truncated: ${limits.creditTruncated ? "yes" : "no"}`,
|
|
684
|
-
`- Link analysis: ${analyzedLinkEdges}/${streamedLinkEdges} edges${linkAnalysisTruncated ? ` (capped at ${MAX_ANALYZED_LINK_EDGES})` : ""}`,
|
|
685
|
-
"",
|
|
686
|
-
reportMd
|
|
687
|
-
].join("\n");
|
|
688
|
-
if (extras.imageAudit) reportMd += `
|
|
689
|
-
|
|
690
|
-
${renderImageSection(extras.imageAudit)}`;
|
|
691
|
-
const metricValues = [...metrics.values()];
|
|
692
|
-
const round2 = (n) => Math.round(n * 10) / 10;
|
|
693
|
-
const distribution = { zero: 0, oneToTwo: 0, threeToTen: 0, elevenPlus: 0 };
|
|
694
|
-
let sumInlinks = 0;
|
|
695
|
-
let sumOutlinks = 0;
|
|
696
|
-
for (const m of metricValues) {
|
|
697
|
-
sumInlinks += m.inlinks;
|
|
698
|
-
sumOutlinks += m.outlinksInternal + m.outlinksExternal;
|
|
699
|
-
if (m.inlinks === 0) distribution.zero++;
|
|
700
|
-
else if (m.inlinks <= 2) distribution.oneToTwo++;
|
|
701
|
-
else if (m.inlinks <= 10) distribution.threeToTen++;
|
|
702
|
-
else distribution.elevenPlus++;
|
|
703
|
-
}
|
|
704
|
-
const externalDomains = [...domMap.entries()].map(([domain, d]) => ({ domain, links: d.links, nofollow: d.nofollow, pages: d.pages.size })).sort((a, b) => b.links - a.links);
|
|
705
|
-
const linkReport = {
|
|
706
|
-
summary: {
|
|
707
|
-
internal: {
|
|
708
|
-
totalLinks: internalTotal,
|
|
709
|
-
pages: metas.length,
|
|
710
|
-
orphans: metricValues.filter((m) => m.orphan).length,
|
|
711
|
-
brokenInternal,
|
|
712
|
-
avgInlinks: metas.length ? round2(sumInlinks / metas.length) : 0,
|
|
713
|
-
avgOutlinks: metas.length ? round2(sumOutlinks / metas.length) : 0,
|
|
714
|
-
distribution,
|
|
715
|
-
topByInlinks: [...metricValues].sort((a, b) => b.inlinks - a.inlinks).slice(0, 20).map((m) => ({ url: m.url, inlinks: m.inlinks, outlinksInternal: m.outlinksInternal, outlinksExternal: m.outlinksExternal }))
|
|
716
|
-
},
|
|
717
|
-
external: {
|
|
718
|
-
totalLinks: externalTotal,
|
|
719
|
-
uniqueDomains: externalDomains.length,
|
|
720
|
-
topDomains: externalDomains.slice(0, 20)
|
|
721
|
-
}
|
|
722
|
-
},
|
|
723
|
-
externalDomains
|
|
724
|
-
};
|
|
725
|
-
const pagesOut = createWriteStream(join(dir, "pages.jsonl"));
|
|
726
|
-
const pageExportById = new Map(pageExportEntries.map((entry) => [entry.pageId, entry]));
|
|
727
|
-
const captureMatrixOut = waybackTimeline ? createWriteStream(join(dir, "capture-matrix.jsonl")) : null;
|
|
728
|
-
const capturedCells = /* @__PURE__ */ new Set();
|
|
729
|
-
const captureCellKey = (month, url) => {
|
|
730
|
-
let normalized = url;
|
|
731
|
-
try {
|
|
732
|
-
normalized = new URL(url).href;
|
|
733
|
-
} catch {
|
|
734
|
-
}
|
|
735
|
-
return `${month ?? ""}\0${normalized}`;
|
|
736
|
-
};
|
|
737
|
-
for await (const chunk of pageChunks(job.id)) {
|
|
738
|
-
for (const page of chunk) {
|
|
739
|
-
const publicFailure = page.extractionStatus === "failed" ? publicizeExtractionFailure(page.failureCode, page.failureReason) : { failureCode: null, failureReason: null };
|
|
740
|
-
const { bodyMarkdown: _body, schema: _schema, outlinks: _outlinks, contentRef: _contentRef, ...boundedPage } = page;
|
|
741
|
-
const m = metrics.get(page.url);
|
|
742
|
-
const exported = page.pageId ? pageExportById.get(page.pageId) : void 0;
|
|
743
|
-
const row = {
|
|
744
|
-
...boundedPage,
|
|
745
|
-
...page.extractionStatus === "failed" ? publicFailure : {},
|
|
746
|
-
representations: exported ? {
|
|
747
|
-
jsonPath: exported.jsonPath,
|
|
748
|
-
htmlPath: exported.htmlPath,
|
|
749
|
-
markdownPath: exported.markdownPath,
|
|
750
|
-
jsonSha256: exported.jsonSha256,
|
|
751
|
-
htmlSha256: exported.htmlSha256,
|
|
752
|
-
markdownSha256: exported.markdownSha256,
|
|
753
|
-
htmlBytes: exported.htmlBytes,
|
|
754
|
-
markdownBytes: exported.markdownBytes
|
|
755
|
-
} : null,
|
|
756
|
-
inlinks: m?.inlinks ?? 0,
|
|
757
|
-
crawlDepth: m?.crawlDepth ?? null,
|
|
758
|
-
orphan: m?.orphan ?? false
|
|
759
|
-
};
|
|
760
|
-
if (!pagesOut.write(JSON.stringify(row) + "\n")) await once(pagesOut, "drain");
|
|
761
|
-
if (captureMatrixOut) {
|
|
762
|
-
const cellKey = captureCellKey(page.archiveRequestedMonth, page.originalUrl ?? page.url);
|
|
763
|
-
capturedCells.add(cellKey);
|
|
764
|
-
const captureRow = {
|
|
765
|
-
requestedMonth: page.archiveRequestedMonth ?? null,
|
|
766
|
-
captureTimestamp: page.archiveTimestamp ?? null,
|
|
767
|
-
originalUrl: page.originalUrl ?? page.url,
|
|
768
|
-
archivedUrl: page.archivedUrl ?? null,
|
|
769
|
-
status: page.extractionStatus === "successful" ? "captured" : "failed",
|
|
770
|
-
httpStatus: page.status,
|
|
771
|
-
contentHash: page.contentHash || null,
|
|
772
|
-
wordCount: page.wordCount,
|
|
773
|
-
failureCode: publicFailure.failureCode,
|
|
774
|
-
failureReason: publicFailure.failureReason
|
|
775
|
-
};
|
|
776
|
-
if (!captureMatrixOut.write(JSON.stringify(captureRow) + "\n")) await once(captureMatrixOut, "drain");
|
|
777
|
-
}
|
|
778
|
-
}
|
|
779
|
-
}
|
|
780
|
-
if (captureMatrixOut && waybackTimeline?.timeline?.urls) {
|
|
781
|
-
const timeline = waybackTimeline.timeline;
|
|
782
|
-
const requestedUrls = timeline.urls;
|
|
783
|
-
const months = (() => {
|
|
784
|
-
if (timeline.months?.length) return [...new Set(timeline.months)].sort();
|
|
785
|
-
if (!timeline.from || !timeline.to) return [];
|
|
786
|
-
const values = [];
|
|
787
|
-
const cursor = /* @__PURE__ */ new Date(`${timeline.from}-01T00:00:00Z`);
|
|
788
|
-
const end = /* @__PURE__ */ new Date(`${timeline.to}-01T00:00:00Z`);
|
|
789
|
-
while (cursor <= end && values.length < 60) {
|
|
790
|
-
values.push(`${cursor.getUTCFullYear()}-${String(cursor.getUTCMonth() + 1).padStart(2, "0")}`);
|
|
791
|
-
cursor.setUTCMonth(cursor.getUTCMonth() + (timeline.intervalMonths ?? 1));
|
|
792
|
-
}
|
|
793
|
-
return values;
|
|
794
|
-
})();
|
|
795
|
-
for (const requestedMonth of months) {
|
|
796
|
-
for (const originalUrl of requestedUrls) {
|
|
797
|
-
if (capturedCells.has(captureCellKey(requestedMonth, originalUrl))) continue;
|
|
798
|
-
if (!captureMatrixOut.write(JSON.stringify({
|
|
799
|
-
requestedMonth,
|
|
800
|
-
captureTimestamp: null,
|
|
801
|
-
originalUrl,
|
|
802
|
-
archivedUrl: null,
|
|
803
|
-
status: "missing",
|
|
804
|
-
httpStatus: null,
|
|
805
|
-
contentHash: null,
|
|
806
|
-
wordCount: 0,
|
|
807
|
-
failureCode: "wayback_capture_missing",
|
|
808
|
-
failureReason: "No successful HTML capture was available in the requested month."
|
|
809
|
-
}) + "\n")) await once(captureMatrixOut, "drain");
|
|
810
|
-
}
|
|
811
|
-
}
|
|
812
|
-
}
|
|
813
|
-
const pagesFinished = finished(pagesOut);
|
|
814
|
-
const captureMatrixFinished = captureMatrixOut ? finished(captureMatrixOut) : Promise.resolve();
|
|
815
|
-
pagesOut.end();
|
|
816
|
-
captureMatrixOut?.end();
|
|
817
|
-
await Promise.all([pagesFinished, captureMatrixFinished]);
|
|
818
|
-
if (captureRenderedDom) {
|
|
819
|
-
writeFileSync(
|
|
820
|
-
join(dir, "rendered-dom", "manifest.jsonl"),
|
|
821
|
-
renderedDomManifest.map((row) => JSON.stringify(row)).join("\n")
|
|
822
|
-
);
|
|
823
|
-
}
|
|
824
|
-
if (semanticSimilarity) {
|
|
825
|
-
const analysis = await analyzeSiteContentSimilarity(similarityPages, {
|
|
826
|
-
threshold: Number(job.options.similarityThreshold ?? void 0),
|
|
827
|
-
maxPairs: Number(job.options.similarityMaxPairs ?? void 0),
|
|
828
|
-
embedTexts: extras.embedSimilarityTexts
|
|
829
|
-
});
|
|
830
|
-
const tableRows = analysis.rows.map(similarityTableRow);
|
|
831
|
-
const columns = tableRows.length > 0 ? Object.keys(tableRows[0]) : [
|
|
832
|
-
"source_url",
|
|
833
|
-
"source_title",
|
|
834
|
-
"target_url",
|
|
835
|
-
"target_title",
|
|
836
|
-
"similarity",
|
|
837
|
-
"similarity_percent",
|
|
838
|
-
"corpus_percentile",
|
|
839
|
-
"exact_content_duplicate",
|
|
840
|
-
"source_word_count",
|
|
841
|
-
"target_word_count",
|
|
842
|
-
"cluster_id"
|
|
843
|
-
];
|
|
844
|
-
const csv = [
|
|
845
|
-
columns.join(","),
|
|
846
|
-
...tableRows.map((row) => columns.map((column) => csvCell(row[column] ?? null)).join(","))
|
|
847
|
-
].join("\n") + "\n";
|
|
848
|
-
const { rows: _rows, clusters: _clusters, ...summary } = analysis;
|
|
849
|
-
writeFileSync(join(dir, "similarity-table.csv"), csv);
|
|
850
|
-
writeFileSync(join(dir, "similarity.jsonl"), tableRows.map((row) => JSON.stringify(row)).join("\n"));
|
|
851
|
-
writeFileSync(join(dir, "similarity-table-schema.json"), JSON.stringify({
|
|
852
|
-
tableNameSuggestion: `site_similarity_${new URL(job.startUrl).hostname.replace(/[^a-z0-9]+/gi, "_").replace(/^_+|_+$/g, "").toLowerCase()}`,
|
|
853
|
-
defaultSort: [{ column: "similarity", direction: "desc" }],
|
|
854
|
-
columns: {
|
|
855
|
-
source_url: "text",
|
|
856
|
-
source_title: "text",
|
|
857
|
-
target_url: "text",
|
|
858
|
-
target_title: "text",
|
|
859
|
-
similarity: "number",
|
|
860
|
-
similarity_percent: "number",
|
|
861
|
-
corpus_percentile: "number",
|
|
862
|
-
exact_content_duplicate: "boolean",
|
|
863
|
-
source_word_count: "integer",
|
|
864
|
-
target_word_count: "integer",
|
|
865
|
-
cluster_id: "text"
|
|
866
|
-
}
|
|
867
|
-
}, null, 2));
|
|
868
|
-
writeFileSync(join(dir, "similarity-summary.json"), JSON.stringify(summary, null, 2));
|
|
869
|
-
writeFileSync(join(dir, "content-clusters.json"), JSON.stringify(analysis.clusters, null, 2));
|
|
870
|
-
reportMd += [
|
|
871
|
-
"",
|
|
872
|
-
"## Rendered content similarity",
|
|
873
|
-
`- Compared pages: ${analysis.comparedPages}`,
|
|
874
|
-
`- Raw cosine threshold: ${analysis.threshold}`,
|
|
875
|
-
`- Corpus score distribution: min ${analysis.scoreDistribution.min ?? "n/a"} / median ${analysis.scoreDistribution.median ?? "n/a"} / p90 ${analysis.scoreDistribution.p90 ?? "n/a"} / max ${analysis.scoreDistribution.max ?? "n/a"}`,
|
|
876
|
-
`- Qualifying pairs: ${analysis.qualifyingPairs}`,
|
|
877
|
-
`- Returned table rows: ${analysis.returnedPairs}${analysis.pairsTruncated ? " (capped)" : ""}`,
|
|
878
|
-
`- Multi-page clusters: ${analysis.clusters.length}`,
|
|
879
|
-
`- Embedding model: ${analysis.model} (${analysis.dimensions} dimensions)`,
|
|
880
|
-
`- Corpus boilerplate removed: ${analysis.boilerplateRemoval.removedBlockSignatures} repeated block signatures / ${analysis.boilerplateRemoval.removedCharacters} characters`,
|
|
881
|
-
`- Corpus SHA-256: ${analysis.corpusSha256}`
|
|
882
|
-
].join("\n");
|
|
883
|
-
}
|
|
884
|
-
writeFileSync(join(dir, "report.md"), reportMd);
|
|
885
|
-
writeFileSync(join(dir, "crawl-summary.json"), JSON.stringify(crawlSummary, null, 2));
|
|
886
|
-
writeFileSync(join(dir, "images-manifest.jsonl"), imageManifest.map((record) => JSON.stringify(record)).join("\n"));
|
|
887
|
-
writeFileSync(join(dir, "manifest.json"), JSON.stringify({
|
|
888
|
-
version: "site-export.v1",
|
|
889
|
-
jobId: job.id,
|
|
890
|
-
startUrl: job.startUrl,
|
|
891
|
-
generatedAt: crawlSummary.generatedAt,
|
|
892
|
-
outputProfile: {
|
|
893
|
-
json: true,
|
|
894
|
-
html: true,
|
|
895
|
-
markdown: true,
|
|
896
|
-
imageManifest: true,
|
|
897
|
-
imageBytes: downloadImages
|
|
898
|
-
},
|
|
899
|
-
pages: [...pageExportEntries].sort((a, b) => a.pageId.localeCompare(b.pageId)),
|
|
900
|
-
imagesManifestPath: "images-manifest.jsonl",
|
|
901
|
-
download: "This ZIP is the complete human export. AI clients should use site_export_read and site_export_image."
|
|
902
|
-
}, null, 2));
|
|
903
|
-
if (waybackTimeline) {
|
|
904
|
-
writeFileSync(join(dir, "wayback-manifest.json"), JSON.stringify({
|
|
905
|
-
rootUrl: waybackTimeline.rootUrl ?? job.startUrl,
|
|
906
|
-
requested: waybackTimeline.timeline ?? null,
|
|
907
|
-
maxPagesPerSnapshot: waybackTimeline.maxPagesPerSnapshot ?? null,
|
|
908
|
-
captureMatrix: "capture-matrix.jsonl"
|
|
909
|
-
}, null, 2));
|
|
910
|
-
}
|
|
911
|
-
writeFileSync(join(dir, "issues.json"), JSON.stringify(issues, null, 2));
|
|
912
|
-
writeFileSync(join(dir, "link-metrics.jsonl"), metricValues.map((m) => JSON.stringify(m)).join("\n"));
|
|
913
|
-
writeFileSync(join(dir, "link-report.md"), renderLinkReport(linkReport));
|
|
914
|
-
writeFileSync(join(dir, "links-summary.json"), JSON.stringify(linkReport.summary, null, 2));
|
|
915
|
-
writeFileSync(join(dir, "external-domains.json"), JSON.stringify(linkReport.externalDomains, null, 2));
|
|
916
|
-
if (extras.imageAudit) {
|
|
917
|
-
writeFileSync(join(dir, "images.jsonl"), extras.imageAudit.rows.map((r) => JSON.stringify(r)).join("\n"));
|
|
918
|
-
writeFileSync(join(dir, "images-summary.json"), JSON.stringify(extras.imageAudit.summary, null, 2));
|
|
919
|
-
}
|
|
920
|
-
if (extras.seoAudit) writeFileSync(join(dir, "seo-audit.json"), JSON.stringify(extras.seoAudit, null, 2));
|
|
921
|
-
if (extras.branding) writeFileSync(join(dir, "branding.json"), JSON.stringify(extras.branding, null, 2));
|
|
922
|
-
if (downloadImages) {
|
|
923
|
-
writeFileSync(join(dir, "images-download-summary.json"), JSON.stringify({
|
|
924
|
-
queued: imagesQueued,
|
|
925
|
-
downloaded: imagesQueued - imagesFailed,
|
|
926
|
-
failed: imagesFailed,
|
|
927
|
-
downloadedBytes: imageBytesDownloaded,
|
|
928
|
-
perPageCap: MAX_IMAGES_PER_PAGE,
|
|
929
|
-
perSiteCap: MAX_IMAGES_PER_SITE,
|
|
930
|
-
perFileByteCap: MAX_IMAGE_BYTES,
|
|
931
|
-
totalByteCap: MAX_SITE_IMAGE_BYTES
|
|
932
|
-
}, null, 2));
|
|
933
|
-
}
|
|
934
|
-
const artifactFiles = [
|
|
935
|
-
{ name: "manifest.json", type: "application/json" },
|
|
936
|
-
{ name: "images-manifest.jsonl", type: "application/x-ndjson" },
|
|
937
|
-
{ name: "report.md", type: "text/markdown" },
|
|
938
|
-
{ name: "crawl-summary.json", type: "application/json" },
|
|
939
|
-
{ name: "failures.jsonl", type: "application/x-ndjson" },
|
|
940
|
-
{ name: "schema.jsonl", type: "application/x-ndjson" },
|
|
941
|
-
{ name: "issues.json", type: "application/json" },
|
|
942
|
-
{ name: "pages.jsonl", type: "application/x-ndjson" },
|
|
943
|
-
{ name: "links.jsonl.gz", type: "application/gzip" },
|
|
944
|
-
{ name: "link-metrics.jsonl", type: "application/x-ndjson" },
|
|
945
|
-
{ name: "link-report.md", type: "text/markdown" },
|
|
946
|
-
{ name: "links-summary.json", type: "application/json" },
|
|
947
|
-
{ name: "external-domains.json", type: "application/json" }
|
|
948
|
-
];
|
|
949
|
-
if (captureRenderedDom) {
|
|
950
|
-
artifactFiles.push({ name: "rendered-dom/manifest.jsonl", type: "application/x-ndjson" });
|
|
951
|
-
for (const row of renderedDomManifest) artifactFiles.push({ name: row.path, type: "text/plain" });
|
|
952
|
-
}
|
|
953
|
-
if (semanticSimilarity) {
|
|
954
|
-
artifactFiles.push({ name: "similarity-table.csv", type: "text/csv" });
|
|
955
|
-
artifactFiles.push({ name: "similarity.jsonl", type: "application/x-ndjson" });
|
|
956
|
-
artifactFiles.push({ name: "similarity-table-schema.json", type: "application/json" });
|
|
957
|
-
artifactFiles.push({ name: "similarity-summary.json", type: "application/json" });
|
|
958
|
-
artifactFiles.push({ name: "content-clusters.json", type: "application/json" });
|
|
959
|
-
}
|
|
960
|
-
if (waybackTimeline) {
|
|
961
|
-
artifactFiles.push({ name: "wayback-manifest.json", type: "application/json" });
|
|
962
|
-
artifactFiles.push({ name: "capture-matrix.jsonl", type: "application/x-ndjson" });
|
|
963
|
-
}
|
|
964
|
-
if (extras.imageAudit) {
|
|
965
|
-
artifactFiles.push({ name: "images.jsonl", type: "application/x-ndjson" });
|
|
966
|
-
artifactFiles.push({ name: "images-summary.json", type: "application/json" });
|
|
967
|
-
}
|
|
968
|
-
if (extras.seoAudit) artifactFiles.push({ name: "seo-audit.json", type: "application/json" });
|
|
969
|
-
if (extras.branding) artifactFiles.push({ name: "branding.json", type: "application/json" });
|
|
970
|
-
if (downloadImages) artifactFiles.push({ name: "images-download-summary.json", type: "application/json" });
|
|
971
|
-
const zip = new ZipFile();
|
|
972
|
-
const artifactPromise = createSiteExtractBundleArtifactStream({
|
|
973
|
-
ownerId: String(job.userId),
|
|
974
|
-
jobId: job.id,
|
|
975
|
-
// Artifact retention starts when this bundle is assembled, including a
|
|
976
|
-
// later support re-finalization, rather than when the crawl was queued.
|
|
977
|
-
createdAt: /* @__PURE__ */ new Date(),
|
|
978
|
-
content: zip.outputStream
|
|
979
|
-
});
|
|
980
|
-
for (const f of artifactFiles) zip.addFile(join(dir, f.name), f.name);
|
|
981
|
-
const orderedPageEntries = [...pageExportEntries].sort((a, b) => a.pageId.localeCompare(b.pageId));
|
|
982
|
-
for (const entry of orderedPageEntries) if (entry.markdownPath) zip.addFile(join(dir, entry.markdownPath), entry.markdownPath);
|
|
983
|
-
for (const entry of orderedPageEntries) if (entry.legacyMarkdownPath) zip.addFile(join(dir, entry.legacyMarkdownPath), entry.legacyMarkdownPath);
|
|
984
|
-
for (const entry of orderedPageEntries) if (entry.htmlPath) zip.addFile(join(dir, entry.htmlPath), entry.htmlPath);
|
|
985
|
-
for (const entry of orderedPageEntries) zip.addFile(join(dir, entry.jsonPath), entry.jsonPath);
|
|
986
|
-
if (downloadImages && imagesQueued > imagesFailed) {
|
|
987
|
-
const imagesDir = join(dir, "images");
|
|
988
|
-
for (const rel of readdirSync(imagesDir, { recursive: true })) {
|
|
989
|
-
const full = join(imagesDir, rel);
|
|
990
|
-
if (statSync(full).isFile()) zip.addFile(full, `images/${rel.split("\\").join("/")}`);
|
|
991
|
-
}
|
|
992
|
-
}
|
|
993
|
-
zip.end();
|
|
994
|
-
return [await artifactPromise, ...imageArtifacts];
|
|
995
|
-
} finally {
|
|
996
|
-
rmSync(dir, { recursive: true, force: true });
|
|
997
|
-
}
|
|
998
|
-
}
|
|
999
|
-
export {
|
|
1000
|
-
MAX_ANALYZED_LINK_EDGES,
|
|
1001
|
-
SITE_EXTRACT_PAGE_CHUNK,
|
|
1002
|
-
assembleExtractArtifacts
|
|
1003
|
-
};
|