mcp-scraper 0.64.0 → 0.65.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +16 -0
  2. package/README.md +2 -2
  3. package/dist/bin/api-server.cjs +1028 -497
  4. package/dist/bin/api-server.cjs.map +1 -1
  5. package/dist/bin/api-server.js +2 -2
  6. package/dist/bin/mcp-scraper-cli.cjs +1 -1
  7. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  8. package/dist/bin/mcp-scraper-cli.js +1 -1
  9. package/dist/bin/mcp-scraper-install.cjs +2 -2
  10. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  11. package/dist/bin/mcp-scraper-install.js +2 -2
  12. package/dist/bin/mcp-stdio-server.cjs +49 -9
  13. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  14. package/dist/bin/mcp-stdio-server.js +4 -4
  15. package/dist/{chunk-TRSZEK2D.js → chunk-2XUXVKT4.js} +3 -3
  16. package/dist/{chunk-TRSZEK2D.js.map → chunk-2XUXVKT4.js.map} +1 -1
  17. package/dist/chunk-5BODYBIP.js +7 -0
  18. package/dist/chunk-5BODYBIP.js.map +1 -0
  19. package/dist/{chunk-BRKCPONM.js → chunk-5PIO7QBG.js} +2 -2
  20. package/dist/{chunk-5QYSY5KI.js → chunk-AMLFKPLL.js} +139 -3
  21. package/dist/chunk-AMLFKPLL.js.map +1 -0
  22. package/dist/{chunk-3NXLGTC6.js → chunk-DCWXVAQT.js} +2 -2
  23. package/dist/{chunk-3NXLGTC6.js.map → chunk-DCWXVAQT.js.map} +1 -1
  24. package/dist/{chunk-EEW3DMI5.js → chunk-PSKRQDGN.js} +50 -10
  25. package/dist/chunk-PSKRQDGN.js.map +1 -0
  26. package/dist/{extract-bundle-FW23CEMG.js → extract-bundle-UWKJT4MU.js} +371 -14
  27. package/dist/extract-bundle-UWKJT4MU.js.map +1 -0
  28. package/dist/{server-TMAWQYZE.js → server-AFEA235I.js} +170 -163
  29. package/dist/server-AFEA235I.js.map +1 -0
  30. package/dist/{site-extract-repository-UXSFD2QM.js → site-extract-repository-J3RH3MOS.js} +3 -3
  31. package/dist/{worker-DD243CON.js → worker-YKRIK5LS.js} +2 -2
  32. package/package.json +3 -1
  33. package/dist/chunk-5QYSY5KI.js.map +0 -1
  34. package/dist/chunk-DYSXI6QU.js +0 -7
  35. package/dist/chunk-DYSXI6QU.js.map +0 -1
  36. package/dist/chunk-EEW3DMI5.js.map +0 -1
  37. package/dist/extract-bundle-FW23CEMG.js.map +0 -1
  38. package/dist/server-TMAWQYZE.js.map +0 -1
  39. /package/dist/{chunk-BRKCPONM.js.map → chunk-5PIO7QBG.js.map} +0 -0
  40. /package/dist/{site-extract-repository-UXSFD2QM.js.map → site-extract-repository-J3RH3MOS.js.map} +0 -0
  41. /package/dist/{worker-DD243CON.js.map → worker-YKRIK5LS.js.map} +0 -0
@@ -1,7 +1,10 @@
1
1
  import {
2
+ commonsEmbedDim,
3
+ commonsEmbedModel,
2
4
  downloadAsset,
5
+ embedCommonsTexts,
3
6
  publicizeExtractionFailure
4
- } from "./chunk-5QYSY5KI.js";
7
+ } from "./chunk-AMLFKPLL.js";
5
8
  import "./chunk-UN4LZVCB.js";
6
9
  import {
7
10
  computeIssues,
@@ -16,16 +19,16 @@ import {
16
19
  createSiteExtractContentReader,
17
20
  createSiteExtractImageArtifact,
18
21
  extractJobLimitInfo
19
- } from "./chunk-BRKCPONM.js";
22
+ } from "./chunk-5PIO7QBG.js";
20
23
  import "./chunk-EKJ65E3Z.js";
21
- import "./chunk-TRSZEK2D.js";
24
+ import "./chunk-2XUXVKT4.js";
22
25
  import {
23
26
  getDb
24
27
  } from "./chunk-4VG3NUUY.js";
25
28
 
26
29
  // src/api/extract-bundle.ts
27
30
  import { createWriteStream, mkdirSync, readFileSync, readdirSync, rmSync, statSync, writeFileSync } from "fs";
28
- import { createHash } from "crypto";
31
+ import { createHash as createHash2 } from "crypto";
29
32
  import { basename, join } from "path";
30
33
  import { tmpdir } from "os";
31
34
  import { createGzip } from "zlib";
@@ -33,6 +36,229 @@ import { once } from "events";
33
36
  import { finished } from "stream/promises";
34
37
  import { ZipFile } from "yazl";
35
38
  import pLimit from "p-limit";
39
+
40
+ // src/api/site-content-similarity.ts
41
+ import { createHash } from "crypto";
42
+ var MAX_SIMILARITY_PAGES = 500;
43
+ var DEFAULT_SIMILARITY_THRESHOLD = 0.9;
44
+ var DEFAULT_SIMILARITY_MAX_PAIRS = 1e4;
45
+ var MAX_SIMILARITY_PAIRS = 5e4;
46
+ function round(value, places) {
47
+ const scale = 10 ** places;
48
+ return Math.round(value * scale) / scale;
49
+ }
50
+ function cosineSimilarity(a, b) {
51
+ if (a.length === 0 || a.length !== b.length) throw new Error("Similarity vectors must have the same non-zero dimension.");
52
+ let dot = 0;
53
+ let normA = 0;
54
+ let normB = 0;
55
+ for (let index = 0; index < a.length; index++) {
56
+ dot += a[index] * b[index];
57
+ normA += a[index] * a[index];
58
+ normB += b[index] * b[index];
59
+ }
60
+ if (normA === 0 || normB === 0) return 0;
61
+ return Math.max(-1, Math.min(1, dot / (Math.sqrt(normA) * Math.sqrt(normB))));
62
+ }
63
+ function corpusHash(pages) {
64
+ const digest = createHash("sha256");
65
+ for (const page of [...pages].sort((a, b) => a.url.localeCompare(b.url))) {
66
+ digest.update(page.url);
67
+ digest.update("\0");
68
+ digest.update(page.contentHash);
69
+ digest.update("\0");
70
+ }
71
+ return digest.digest("hex");
72
+ }
73
+ function markdownBlockSignature(block) {
74
+ const normalized = block.replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1").replace(/\[([^\]]+)\]\([^)]*\)/g, "$1").replace(/https?:\/\/\S+/gi, " ").replace(/&(?:amp|nbsp|quot|#39);/gi, " ").replace(/[#>*_`~|\\-]+/g, " ").replace(/[^\p{L}\p{N}]+/gu, " ").replace(/\s+/g, " ").trim().toLowerCase();
75
+ return normalized.length >= 12 ? normalized : null;
76
+ }
77
+ function prepareEmbeddingTexts(pages) {
78
+ const blocksByPage = pages.map((page) => page.bodyMarkdown.split(/\n{2,}/).map((block) => block.trim()).filter(Boolean));
79
+ const minimumPageCount = pages.length >= 4 ? Math.max(3, Math.ceil(pages.length * 0.5)) : null;
80
+ const pageFrequency = /* @__PURE__ */ new Map();
81
+ for (const blocks of blocksByPage) {
82
+ const pageSignatures = new Set(blocks.map(markdownBlockSignature).filter((value) => Boolean(value)));
83
+ for (const signature of pageSignatures) pageFrequency.set(signature, (pageFrequency.get(signature) ?? 0) + 1);
84
+ }
85
+ const repeated = /* @__PURE__ */ new Set();
86
+ if (minimumPageCount != null) {
87
+ for (const [signature, count] of pageFrequency) {
88
+ if (count >= minimumPageCount) repeated.add(signature);
89
+ }
90
+ }
91
+ let removedCharacters = 0;
92
+ const texts = pages.map((page, pageIndex) => {
93
+ const kept = [];
94
+ for (const block of blocksByPage[pageIndex]) {
95
+ const signature = markdownBlockSignature(block);
96
+ if (signature && repeated.has(signature)) {
97
+ removedCharacters += block.length;
98
+ continue;
99
+ }
100
+ kept.push(block);
101
+ }
102
+ const cleaned = kept.join("\n\n").trim();
103
+ const content = cleaned.length >= 60 ? cleaned : page.bodyMarkdown;
104
+ return `${page.title?.trim() || "Untitled page"}
105
+
106
+ ${content}`;
107
+ });
108
+ return {
109
+ texts,
110
+ minimumPageCount,
111
+ removedBlockSignatures: repeated.size,
112
+ removedCharacters
113
+ };
114
+ }
115
+ var DisjointSet = class {
116
+ parent;
117
+ constructor(size) {
118
+ this.parent = Array.from({ length: size }, (_, index) => index);
119
+ }
120
+ find(value) {
121
+ const parent = this.parent[value];
122
+ if (parent !== value) this.parent[value] = this.find(parent);
123
+ return this.parent[value];
124
+ }
125
+ union(a, b) {
126
+ const rootA = this.find(a);
127
+ const rootB = this.find(b);
128
+ if (rootA !== rootB) this.parent[rootB] = rootA;
129
+ }
130
+ };
131
+ async function analyzeSiteContentSimilarity(pages, options = {}) {
132
+ const eligible = pages.filter((page) => page.bodyMarkdown.trim().length > 0).slice(0, MAX_SIMILARITY_PAGES);
133
+ const requestedThreshold = Number.isFinite(options.threshold) ? options.threshold : DEFAULT_SIMILARITY_THRESHOLD;
134
+ const requestedMaxPairs = Number.isFinite(options.maxPairs) ? options.maxPairs : DEFAULT_SIMILARITY_MAX_PAIRS;
135
+ const threshold = Math.max(0, Math.min(1, requestedThreshold));
136
+ const maxPairs = Math.max(1, Math.min(MAX_SIMILARITY_PAIRS, Math.floor(requestedMaxPairs)));
137
+ const model = options.model ?? commonsEmbedModel();
138
+ const dimensions = options.dimensions ?? commonsEmbedDim();
139
+ const embedTexts = options.embedTexts ?? embedCommonsTexts;
140
+ const prepared = prepareEmbeddingTexts(eligible);
141
+ const uniqueTexts = [];
142
+ const textIndexByHash = /* @__PURE__ */ new Map();
143
+ const pageTextIndexes = [];
144
+ for (const [pageIndex, page] of eligible.entries()) {
145
+ const key = page.contentHash || createHash("sha256").update(page.bodyMarkdown).digest("hex");
146
+ let textIndex = textIndexByHash.get(key);
147
+ if (textIndex == null) {
148
+ textIndex = uniqueTexts.length;
149
+ textIndexByHash.set(key, textIndex);
150
+ uniqueTexts.push(prepared.texts[pageIndex]);
151
+ }
152
+ pageTextIndexes.push(textIndex);
153
+ }
154
+ const uniqueVectors = [];
155
+ for (let offset = 0; offset < uniqueTexts.length; offset += 64) {
156
+ uniqueVectors.push(...await embedTexts(uniqueTexts.slice(offset, offset + 64)));
157
+ }
158
+ if (uniqueVectors.length !== uniqueTexts.length) {
159
+ throw new Error(`Embedding provider returned ${uniqueVectors.length} vectors for ${uniqueTexts.length} unique page bodies.`);
160
+ }
161
+ const vectors = pageTextIndexes.map((index) => uniqueVectors[index]);
162
+ const sets = new DisjointSet(eligible.length);
163
+ const allScores = [];
164
+ const qualifying = [];
165
+ for (let source = 0; source < eligible.length; source++) {
166
+ for (let target = source + 1; target < eligible.length; target++) {
167
+ const score = cosineSimilarity(vectors[source], vectors[target]);
168
+ allScores.push(score);
169
+ if (score < threshold) continue;
170
+ qualifying.push({ source, target, score });
171
+ sets.union(source, target);
172
+ }
173
+ }
174
+ qualifying.sort((a, b) => b.score - a.score || eligible[a.source].url.localeCompare(eligible[b.source].url) || eligible[a.target].url.localeCompare(eligible[b.target].url));
175
+ const sortedScores = [...allScores].sort((a, b) => a - b);
176
+ const quantile = (value) => {
177
+ if (!sortedScores.length) return null;
178
+ return round(sortedScores[Math.floor((sortedScores.length - 1) * value)], 6);
179
+ };
180
+ const membersByRoot = /* @__PURE__ */ new Map();
181
+ for (let index = 0; index < eligible.length; index++) {
182
+ const root = sets.find(index);
183
+ const members = membersByRoot.get(root) ?? [];
184
+ members.push(index);
185
+ membersByRoot.set(root, members);
186
+ }
187
+ const clusterIdByPage = /* @__PURE__ */ new Map();
188
+ const clusters = [...membersByRoot.values()].filter((members) => members.length > 1).sort((a, b) => b.length - a.length || eligible[a[0]].url.localeCompare(eligible[b[0]].url)).map((members, index) => {
189
+ const clusterId = `cluster-${String(index + 1).padStart(3, "0")}`;
190
+ for (const member of members) clusterIdByPage.set(member, clusterId);
191
+ return {
192
+ clusterId,
193
+ pageCount: members.length,
194
+ urls: members.map((member) => eligible[member].url).sort()
195
+ };
196
+ });
197
+ const rows = qualifying.slice(0, maxPairs).map((pair, pairIndex) => {
198
+ const source = eligible[pair.source];
199
+ const target = eligible[pair.target];
200
+ const similarity = round(pair.score, 6);
201
+ return {
202
+ sourceUrl: source.url,
203
+ sourceTitle: source.title,
204
+ targetUrl: target.url,
205
+ targetTitle: target.title,
206
+ similarity,
207
+ similarityPercent: round(similarity * 100, 2),
208
+ corpusPercentile: sortedScores.length <= 1 ? 100 : round((1 - pairIndex / (sortedScores.length - 1)) * 100, 2),
209
+ exactContentDuplicate: Boolean(source.contentHash && source.contentHash === target.contentHash),
210
+ sourceWordCount: source.wordCount,
211
+ targetWordCount: target.wordCount,
212
+ clusterId: clusterIdByPage.get(pair.source) ?? null
213
+ };
214
+ });
215
+ const corpusSha256 = corpusHash(eligible);
216
+ const analysisSha256 = createHash("sha256").update(JSON.stringify({
217
+ corpusSha256,
218
+ model,
219
+ dimensions,
220
+ threshold,
221
+ maxPairs,
222
+ boilerplateRemoval: {
223
+ method: "corpus_repeated_markdown_blocks",
224
+ minimumPageCount: prepared.minimumPageCount,
225
+ removedBlockSignatures: prepared.removedBlockSignatures
226
+ }
227
+ })).digest("hex");
228
+ return {
229
+ generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
230
+ provider: "jina",
231
+ model,
232
+ dimensions,
233
+ threshold,
234
+ requestedMaxPairs: maxPairs,
235
+ comparedPages: eligible.length,
236
+ possiblePairs: eligible.length * Math.max(0, eligible.length - 1) / 2,
237
+ qualifyingPairs: qualifying.length,
238
+ returnedPairs: rows.length,
239
+ pairsTruncated: qualifying.length > rows.length,
240
+ corpusSha256,
241
+ analysisSha256,
242
+ scoreDistribution: {
243
+ min: quantile(0),
244
+ p25: quantile(0.25),
245
+ median: quantile(0.5),
246
+ p75: quantile(0.75),
247
+ p90: quantile(0.9),
248
+ max: quantile(1)
249
+ },
250
+ boilerplateRemoval: {
251
+ method: "corpus_repeated_markdown_blocks",
252
+ minimumPageCount: prepared.minimumPageCount,
253
+ removedBlockSignatures: prepared.removedBlockSignatures,
254
+ removedCharacters: prepared.removedCharacters
255
+ },
256
+ rows,
257
+ clusters
258
+ };
259
+ }
260
+
261
+ // src/api/extract-bundle.ts
36
262
  var SITE_EXTRACT_PAGE_CHUNK = 25;
37
263
  var MAX_IMAGES_PER_PAGE = 20;
38
264
  var MAX_IMAGES_PER_SITE = 500;
@@ -57,6 +283,15 @@ function safeImageFilename(url, index) {
57
283
  return `image-${index}`;
58
284
  }
59
285
  }
286
+ function slugFactory() {
287
+ const counts = /* @__PURE__ */ new Map();
288
+ return (url) => {
289
+ const base = url.replace(/^https?:\/\//, "").replace(/[^a-zA-Z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80) || "page";
290
+ const n = counts.get(base) ?? 0;
291
+ counts.set(base, n + 1);
292
+ return n ? `${base}-${n}` : base;
293
+ };
294
+ }
60
295
  async function* pageChunks(jobId) {
61
296
  const db = getDb();
62
297
  let lastRowId = 0;
@@ -75,6 +310,25 @@ async function* pageChunks(jobId) {
75
310
  yield res.rows.map((r) => JSON.parse(String(r.page)));
76
311
  }
77
312
  }
313
+ function csvCell(value) {
314
+ const text = value == null ? "" : String(value);
315
+ return /[",\r\n]/.test(text) ? `"${text.replace(/"/g, '""')}"` : text;
316
+ }
317
+ function similarityTableRow(row) {
318
+ return {
319
+ source_url: row.sourceUrl,
320
+ source_title: row.sourceTitle,
321
+ target_url: row.targetUrl,
322
+ target_title: row.targetTitle,
323
+ similarity: row.similarity,
324
+ similarity_percent: row.similarityPercent,
325
+ corpus_percentile: row.corpusPercentile,
326
+ exact_content_duplicate: row.exactContentDuplicate,
327
+ source_word_count: row.sourceWordCount,
328
+ target_word_count: row.targetWordCount,
329
+ cluster_id: row.clusterId
330
+ };
331
+ }
78
332
  async function assembleExtractArtifacts(job, extras = {}) {
79
333
  const dir = join(tmpdir(), `extract-bundle-${job.id}-${Date.now()}`);
80
334
  const downloadImages = job.options.downloadImages === true;
@@ -82,8 +336,14 @@ async function assembleExtractArtifacts(job, extras = {}) {
82
336
  const readPageContent = createSiteExtractContentReader(String(job.userId));
83
337
  mkdirSync(dir, { recursive: true });
84
338
  if (downloadImages) mkdirSync(join(dir, "images"), { recursive: true });
339
+ const captureRenderedDom = job.options.captureRenderedDom === true;
340
+ const semanticSimilarity = job.options.semanticSimilarity === true;
341
+ if (captureRenderedDom) mkdirSync(join(dir, "rendered-dom"), { recursive: true });
85
342
  try {
86
343
  const metas = [];
344
+ const similarityPages = [];
345
+ const renderedDomManifest = [];
346
+ const domSlug = slugFactory();
87
347
  const pageExportEntries = [];
88
348
  const waybackTimeline = job.options.waybackTimeline;
89
349
  const statusByUrl = /* @__PURE__ */ new Map();
@@ -111,12 +371,32 @@ async function assembleExtractArtifacts(job, extras = {}) {
111
371
  schemaTypes: p.schemaTypes.length > 0 ? ["present"] : [],
112
372
  outlinks: []
113
373
  });
114
- const pageId = p.pageId ?? createHash("sha256").update(p.archivedUrl ?? p.url).digest("hex");
374
+ const pageId = p.pageId ?? createHash2("sha256").update(p.archivedUrl ?? p.url).digest("hex");
115
375
  const pageDir = join(dir, "pages", pageId);
116
376
  mkdirSync(pageDir, { recursive: true });
117
377
  const content = p.contentRef ? await readPageContent(p.contentRef) : null;
118
378
  const html = content?.html ?? null;
119
379
  const markdown = content?.markdown ?? p.bodyMarkdown ?? null;
380
+ if (semanticSimilarity && p.extractionStatus === "successful" && markdown?.trim()) {
381
+ similarityPages.push({
382
+ url: p.url,
383
+ title: p.title,
384
+ bodyMarkdown: markdown,
385
+ contentHash: p.contentHash,
386
+ wordCount: p.wordCount
387
+ });
388
+ }
389
+ if (captureRenderedDom && html != null) {
390
+ const relativePath = `rendered-dom/${domSlug(p.url)}.html.txt`;
391
+ writeFileSync(join(dir, relativePath), html);
392
+ renderedDomManifest.push({
393
+ url: p.url,
394
+ path: relativePath,
395
+ bytes: Buffer.byteLength(html),
396
+ truncated: p.renderedDomTruncated === true,
397
+ sanitized: p.renderedDomSanitized === true
398
+ });
399
+ }
120
400
  const htmlPath = html == null ? null : `pages/${pageId}/page.html`;
121
401
  const markdownPath = markdown == null ? null : `pages/${pageId}/page.md`;
122
402
  const legacyMarkdownPath = markdown == null || !p.archiveRequestedMonth ? null : `pages/${p.archiveRequestedMonth}/${pageId}.md`;
@@ -136,8 +416,8 @@ async function assembleExtractArtifacts(job, extras = {}) {
136
416
  markdownPath,
137
417
  htmlBytes: html == null ? 0 : Buffer.byteLength(html),
138
418
  markdownBytes: markdown == null ? 0 : Buffer.byteLength(markdown),
139
- htmlSha256: html == null ? null : createHash("sha256").update(html).digest("hex"),
140
- markdownSha256: markdown == null ? null : createHash("sha256").update(markdown).digest("hex")
419
+ htmlSha256: html == null ? null : createHash2("sha256").update(html).digest("hex"),
420
+ markdownSha256: markdown == null ? null : createHash2("sha256").update(markdown).digest("hex")
141
421
  }
142
422
  };
143
423
  const pageJson = JSON.stringify(pageRecord, null, 2);
@@ -150,7 +430,7 @@ async function assembleExtractArtifacts(job, extras = {}) {
150
430
  htmlPath,
151
431
  markdownPath,
152
432
  legacyMarkdownPath,
153
- jsonSha256: createHash("sha256").update(pageJson).digest("hex"),
433
+ jsonSha256: createHash2("sha256").update(pageJson).digest("hex"),
154
434
  htmlSha256: pageRecord.content.htmlSha256,
155
435
  markdownSha256: pageRecord.content.markdownSha256,
156
436
  htmlBytes: pageRecord.content.htmlBytes,
@@ -218,7 +498,7 @@ async function assembleExtractArtifacts(job, extras = {}) {
218
498
  }
219
499
  if (p.imageLinks?.length) {
220
500
  for (const [idx, imgUrl] of p.imageLinks.entries()) {
221
- const imageId = createHash("sha256").update(`${p.url}\0${imgUrl}`).digest("hex");
501
+ const imageId = createHash2("sha256").update(`${p.url}\0${imgUrl}`).digest("hex");
222
502
  const record = {
223
503
  imageId,
224
504
  sourcePage: p.url,
@@ -272,7 +552,7 @@ async function assembleExtractArtifacts(job, extras = {}) {
272
552
  record.artifactId = artifact.key;
273
553
  record.mimeType = result.mimeType;
274
554
  record.bytes = result.sizeBytes;
275
- record.sha256 = createHash("sha256").update(bytes).digest("hex");
555
+ record.sha256 = createHash2("sha256").update(bytes).digest("hex");
276
556
  }, (error) => {
277
557
  imagesFailed++;
278
558
  record.status = "failed";
@@ -410,7 +690,7 @@ async function assembleExtractArtifacts(job, extras = {}) {
410
690
 
411
691
  ${renderImageSection(extras.imageAudit)}`;
412
692
  const metricValues = [...metrics.values()];
413
- const round = (n) => Math.round(n * 10) / 10;
693
+ const round2 = (n) => Math.round(n * 10) / 10;
414
694
  const distribution = { zero: 0, oneToTwo: 0, threeToTen: 0, elevenPlus: 0 };
415
695
  let sumInlinks = 0;
416
696
  let sumOutlinks = 0;
@@ -430,8 +710,8 @@ ${renderImageSection(extras.imageAudit)}`;
430
710
  pages: metas.length,
431
711
  orphans: metricValues.filter((m) => m.orphan).length,
432
712
  brokenInternal,
433
- avgInlinks: metas.length ? round(sumInlinks / metas.length) : 0,
434
- avgOutlinks: metas.length ? round(sumOutlinks / metas.length) : 0,
713
+ avgInlinks: metas.length ? round2(sumInlinks / metas.length) : 0,
714
+ avgOutlinks: metas.length ? round2(sumOutlinks / metas.length) : 0,
435
715
  distribution,
436
716
  topByInlinks: [...metricValues].sort((a, b) => b.inlinks - a.inlinks).slice(0, 20).map((m) => ({ url: m.url, inlinks: m.inlinks, outlinksInternal: m.outlinksInternal, outlinksExternal: m.outlinksExternal }))
437
717
  },
@@ -536,6 +816,72 @@ ${renderImageSection(extras.imageAudit)}`;
536
816
  pagesOut.end();
537
817
  captureMatrixOut?.end();
538
818
  await Promise.all([pagesFinished, captureMatrixFinished]);
819
+ if (captureRenderedDom) {
820
+ writeFileSync(
821
+ join(dir, "rendered-dom", "manifest.jsonl"),
822
+ renderedDomManifest.map((row) => JSON.stringify(row)).join("\n")
823
+ );
824
+ }
825
+ if (semanticSimilarity) {
826
+ const analysis = await analyzeSiteContentSimilarity(similarityPages, {
827
+ threshold: Number(job.options.similarityThreshold ?? void 0),
828
+ maxPairs: Number(job.options.similarityMaxPairs ?? void 0),
829
+ embedTexts: extras.embedSimilarityTexts
830
+ });
831
+ const tableRows = analysis.rows.map(similarityTableRow);
832
+ const columns = tableRows.length > 0 ? Object.keys(tableRows[0]) : [
833
+ "source_url",
834
+ "source_title",
835
+ "target_url",
836
+ "target_title",
837
+ "similarity",
838
+ "similarity_percent",
839
+ "corpus_percentile",
840
+ "exact_content_duplicate",
841
+ "source_word_count",
842
+ "target_word_count",
843
+ "cluster_id"
844
+ ];
845
+ const csv = [
846
+ columns.join(","),
847
+ ...tableRows.map((row) => columns.map((column) => csvCell(row[column] ?? null)).join(","))
848
+ ].join("\n") + "\n";
849
+ const { rows: _rows, clusters: _clusters, ...summary } = analysis;
850
+ writeFileSync(join(dir, "similarity-table.csv"), csv);
851
+ writeFileSync(join(dir, "similarity.jsonl"), tableRows.map((row) => JSON.stringify(row)).join("\n"));
852
+ writeFileSync(join(dir, "similarity-table-schema.json"), JSON.stringify({
853
+ tableNameSuggestion: `site_similarity_${new URL(job.startUrl).hostname.replace(/[^a-z0-9]+/gi, "_").replace(/^_+|_+$/g, "").toLowerCase()}`,
854
+ defaultSort: [{ column: "similarity", direction: "desc" }],
855
+ columns: {
856
+ source_url: "text",
857
+ source_title: "text",
858
+ target_url: "text",
859
+ target_title: "text",
860
+ similarity: "number",
861
+ similarity_percent: "number",
862
+ corpus_percentile: "number",
863
+ exact_content_duplicate: "boolean",
864
+ source_word_count: "integer",
865
+ target_word_count: "integer",
866
+ cluster_id: "text"
867
+ }
868
+ }, null, 2));
869
+ writeFileSync(join(dir, "similarity-summary.json"), JSON.stringify(summary, null, 2));
870
+ writeFileSync(join(dir, "content-clusters.json"), JSON.stringify(analysis.clusters, null, 2));
871
+ reportMd += [
872
+ "",
873
+ "## Rendered content similarity",
874
+ `- Compared pages: ${analysis.comparedPages}`,
875
+ `- Raw cosine threshold: ${analysis.threshold}`,
876
+ `- Corpus score distribution: min ${analysis.scoreDistribution.min ?? "n/a"} / median ${analysis.scoreDistribution.median ?? "n/a"} / p90 ${analysis.scoreDistribution.p90 ?? "n/a"} / max ${analysis.scoreDistribution.max ?? "n/a"}`,
877
+ `- Qualifying pairs: ${analysis.qualifyingPairs}`,
878
+ `- Returned table rows: ${analysis.returnedPairs}${analysis.pairsTruncated ? " (capped)" : ""}`,
879
+ `- Multi-page clusters: ${analysis.clusters.length}`,
880
+ `- Embedding model: ${analysis.model} (${analysis.dimensions} dimensions)`,
881
+ `- Corpus boilerplate removed: ${analysis.boilerplateRemoval.removedBlockSignatures} repeated block signatures / ${analysis.boilerplateRemoval.removedCharacters} characters`,
882
+ `- Corpus SHA-256: ${analysis.corpusSha256}`
883
+ ].join("\n");
884
+ }
539
885
  writeFileSync(join(dir, "report.md"), reportMd);
540
886
  writeFileSync(join(dir, "crawl-summary.json"), JSON.stringify(crawlSummary, null, 2));
541
887
  writeFileSync(join(dir, "images-manifest.jsonl"), imageManifest.map((record) => JSON.stringify(record)).join("\n"));
@@ -601,6 +947,17 @@ ${renderImageSection(extras.imageAudit)}`;
601
947
  { name: "links-summary.json", type: "application/json" },
602
948
  { name: "external-domains.json", type: "application/json" }
603
949
  ];
950
+ if (captureRenderedDom) {
951
+ artifactFiles.push({ name: "rendered-dom/manifest.jsonl", type: "application/x-ndjson" });
952
+ for (const row of renderedDomManifest) artifactFiles.push({ name: row.path, type: "text/plain" });
953
+ }
954
+ if (semanticSimilarity) {
955
+ artifactFiles.push({ name: "similarity-table.csv", type: "text/csv" });
956
+ artifactFiles.push({ name: "similarity.jsonl", type: "application/x-ndjson" });
957
+ artifactFiles.push({ name: "similarity-table-schema.json", type: "application/json" });
958
+ artifactFiles.push({ name: "similarity-summary.json", type: "application/json" });
959
+ artifactFiles.push({ name: "content-clusters.json", type: "application/json" });
960
+ }
604
961
  if (waybackTimeline) {
605
962
  artifactFiles.push({ name: "wayback-manifest.json", type: "application/json" });
606
963
  artifactFiles.push({ name: "capture-matrix.jsonl", type: "application/x-ndjson" });
@@ -645,4 +1002,4 @@ export {
645
1002
  SITE_EXTRACT_PAGE_CHUNK,
646
1003
  assembleExtractArtifacts
647
1004
  };
648
- //# sourceMappingURL=extract-bundle-FW23CEMG.js.map
1005
+ //# sourceMappingURL=extract-bundle-UWKJT4MU.js.map