mcp-scraper 0.4.5 → 0.4.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/dist/bin/api-server.cjs +1866 -1376
  2. package/dist/bin/api-server.cjs.map +1 -1
  3. package/dist/bin/api-server.js +3 -3
  4. package/dist/bin/mcp-scraper-cli.cjs +1 -1
  5. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  6. package/dist/bin/mcp-scraper-cli.js +1 -1
  7. package/dist/bin/mcp-scraper-install.cjs +1 -1
  8. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  9. package/dist/bin/mcp-scraper-install.js +1 -1
  10. package/dist/bin/mcp-stdio-server.cjs +89 -1
  11. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  12. package/dist/bin/mcp-stdio-server.js +2 -2
  13. package/dist/bin/paa-harvest.cjs +13 -4
  14. package/dist/bin/paa-harvest.cjs.map +1 -1
  15. package/dist/bin/paa-harvest.js +2 -2
  16. package/dist/{chunk-NJK5BTUK.js → chunk-2YY46QYT.js} +28 -11
  17. package/dist/chunk-2YY46QYT.js.map +1 -0
  18. package/dist/{chunk-WI4A3KFQ.js → chunk-6EXP6DQG.js} +90 -2
  19. package/dist/chunk-6EXP6DQG.js.map +1 -0
  20. package/dist/{chunk-5LO6EHEH.js → chunk-EMY7ELRU.js} +35 -7
  21. package/dist/chunk-EMY7ELRU.js.map +1 -0
  22. package/dist/{chunk-FUVQR6GK.js → chunk-GL4BW4CP.js} +49 -1
  23. package/dist/chunk-GL4BW4CP.js.map +1 -0
  24. package/dist/{chunk-3UH3BEIZ.js → chunk-HE2LQPJ2.js} +2 -2
  25. package/dist/{chunk-4ZZWFI6L.js → chunk-ONIOF5XW.js} +3 -3
  26. package/dist/chunk-SHXJQQOH.js +7 -0
  27. package/dist/chunk-SHXJQQOH.js.map +1 -0
  28. package/dist/{db-H3S3M6KK.js → db-LIOTIWVN.js} +6 -2
  29. package/dist/{extract-bundle-UKE273TV.js → extract-bundle-COS56ZDO.js} +2 -2
  30. package/dist/index.cjs +59 -4
  31. package/dist/index.cjs.map +1 -1
  32. package/dist/index.js +9 -3
  33. package/dist/index.js.map +1 -1
  34. package/dist/{server-XCDFHHLE.js → server-UNID3SJU.js} +570 -276
  35. package/dist/server-UNID3SJU.js.map +1 -0
  36. package/dist/{site-extract-repository-THBVEXMP.js → site-extract-repository-HTIY52MW.js} +4 -4
  37. package/dist/{worker-YHKUJR5P.js → worker-HTYZAYGB.js} +18 -14
  38. package/dist/worker-HTYZAYGB.js.map +1 -0
  39. package/package.json +2 -2
  40. package/dist/chunk-5LO6EHEH.js.map +0 -1
  41. package/dist/chunk-FUVQR6GK.js.map +0 -1
  42. package/dist/chunk-NJK5BTUK.js.map +0 -1
  43. package/dist/chunk-UGQC2FOX.js +0 -7
  44. package/dist/chunk-UGQC2FOX.js.map +0 -1
  45. package/dist/chunk-WI4A3KFQ.js.map +0 -1
  46. package/dist/server-XCDFHHLE.js.map +0 -1
  47. package/dist/worker-YHKUJR5P.js.map +0 -1
  48. /package/dist/{chunk-3UH3BEIZ.js.map → chunk-HE2LQPJ2.js.map} +0 -0
  49. /package/dist/{chunk-4ZZWFI6L.js.map → chunk-ONIOF5XW.js.map} +0 -0
  50. /package/dist/{db-H3S3M6KK.js.map → db-LIOTIWVN.js.map} +0 -0
  51. /package/dist/{extract-bundle-UKE273TV.js.map → extract-bundle-COS56ZDO.js.map} +0 -0
  52. /package/dist/{site-extract-repository-THBVEXMP.js.map → site-extract-repository-HTIY52MW.js.map} +0 -0
@@ -17,7 +17,7 @@ import {
17
17
  } from "./chunk-XGIPATLV.js";
18
18
  import {
19
19
  PACKAGE_VERSION
20
- } from "./chunk-UGQC2FOX.js";
20
+ } from "./chunk-SHXJQQOH.js";
21
21
  import {
22
22
  sanitizeVendorName
23
23
  } from "./chunk-M2S27J6Z.js";
@@ -1019,6 +1019,56 @@ ${headingSection}${kpoSection}${brandingSection}${bodySectionFull}${memSection}$
1019
1019
  }
1020
1020
  return { ...textResult, structuredContent };
1021
1021
  }
1022
+ var DIFF_PAGE_PREVIEW_HUNKS = 20;
1023
+ function formatDiffPage(raw, input) {
1024
+ const parsed = parseData(raw);
1025
+ if ("error" in parsed) return { content: [{ type: "text", text: parsed.error }], isError: true };
1026
+ const d = parsed.data;
1027
+ const url = d.url ?? input.url;
1028
+ const title = d.title ?? "Untitled";
1029
+ const statusLine = d.status === "baseline" ? d.isReset ? "\u{1F195} **Baseline reset** \u2014 prior history discarded, this check is now the new baseline." : "\u{1F195} **Baseline captured** \u2014 first check for this URL, nothing to compare yet." : d.status === "unchanged" ? `\u2705 **No changes** since ${d.previousCheckedAt ?? "the last check"}.` : `\u{1F504} **Changed** since ${d.previousCheckedAt ?? "the last check"} \u2014 +${d.summary.linesAdded}/-${d.summary.linesRemoved} lines (${d.summary.percentChanged ?? 0}% of content).`;
1030
+ const previewHunks = d.hunks.slice(0, DIFF_PAGE_PREVIEW_HUNKS);
1031
+ const diffBlock = previewHunks.length ? [
1032
+ "\n## Diff",
1033
+ "```diff",
1034
+ ...previewHunks.flatMap((h) => h.lines.map((l) => `${h.type === "added" ? "+" : "-"} ${l}`)),
1035
+ "```",
1036
+ previewHunks.length < d.hunks.length ? `*(showing ${previewHunks.length} of ${d.hunks.length} returned hunks)*` : ""
1037
+ ].filter(Boolean).join("\n") : "";
1038
+ const truncationNotes = [
1039
+ d.contentTruncated ? "\n\u26A0\uFE0F Page content exceeded the storable cap and was truncated before hashing/diffing \u2014 changes past that point are not visible to this comparison." : "",
1040
+ d.hunksTruncatedReason ? `
1041
+ \u26A0\uFE0F ${d.hunksTruncatedReason}` : ""
1042
+ ].filter(Boolean).join("\n");
1043
+ const tips = `
1044
+ ---
1045
+ \u{1F4A1} **Tips**
1046
+ - Call \`diff_page\` again anytime to re-check \u2014 this tool is on-demand, not automatic.
1047
+ - Want the page's current content with nothing to compare? use \`extract_url\`.`;
1048
+ const full = `# Page Diff: ${url}
1049
+ **${title}**
1050
+
1051
+ ${statusLine}${diffBlock}${truncationNotes}${tips}`;
1052
+ return {
1053
+ ...oneBlock(full),
1054
+ structuredContent: {
1055
+ url,
1056
+ title: d.title,
1057
+ status: d.status,
1058
+ isReset: d.isReset,
1059
+ previousCheckedAt: d.previousCheckedAt,
1060
+ currentCheckedAt: d.currentCheckedAt,
1061
+ contentHash: d.contentHash,
1062
+ previousContentHash: d.previousContentHash,
1063
+ summary: d.summary,
1064
+ hunks: d.hunks,
1065
+ contentTruncated: d.contentTruncated,
1066
+ hunksTruncated: d.hunksTruncated,
1067
+ hunksTruncatedReason: d.hunksTruncatedReason,
1068
+ totalChangedLineCount: d.totalChangedLineCount
1069
+ }
1070
+ };
1071
+ }
1022
1072
  async function formatMapSiteUrls(raw, input, ctx) {
1023
1073
  const parsed = parseData(raw);
1024
1074
  if ("error" in parsed) return { content: [{ type: "text", text: parsed.error }], isError: true };
@@ -2776,6 +2826,11 @@ var ExtractUrlInputSchema = {
2776
2826
  depositToVault: z.boolean().default(false).describe("Save the full page content into the user's MCP Memory vault server-side, embedded for semantic recall \u2014 the full body is NOT returned to chat."),
2777
2827
  vaultName: z.string().trim().min(1).max(120).optional().describe("Optional vault to deposit into. Defaults to the user's personal vault.")
2778
2828
  };
2829
+ var DiffPageInputSchema = {
2830
+ url: z.string().url().describe("Public http/https URL to check for changes since the last diff_page call."),
2831
+ allowLocal: z.boolean().default(false).describe("Allow localhost and private-network URLs. Local development only."),
2832
+ resetBaseline: z.boolean().default(false).describe("Discard any previously stored snapshot for this URL and capture the current content as a fresh baseline instead of diffing against history. Use when you deliberately want to restart change tracking.")
2833
+ };
2779
2834
  var MapSiteUrlsInputSchema = {
2780
2835
  url: z.string().url().describe("Public website URL or domain to crawl for internal URLs. Use before extract_site when the user asks to audit/map/crawl a site."),
2781
2836
  maxUrls: z.number().int().min(1).max(1e4).optional().describe("Maximum URLs to discover. Use 100 for normal maps, up to 10000 for a full inventory. Large maps (over 500 URLs) write the complete inventory to a local file and return only a summary plus the file path instead of the full list inline.")
@@ -3169,6 +3224,29 @@ var ExtractUrlOutputSchema = {
3169
3224
  error: z.string().optional()
3170
3225
  }).optional()
3171
3226
  };
3227
+ var DiffPageOutputSchema = {
3228
+ url: z.string(),
3229
+ title: NullableString,
3230
+ status: z.enum(["baseline", "unchanged", "changed"]).describe('"baseline" = first-ever check for this URL, or resetBaseline was used \u2014 nothing to compare against. "unchanged" = content hash matched the stored snapshot. "changed" = a diff was computed.'),
3231
+ isReset: z.boolean().describe("True only when resetBaseline discarded a real prior snapshot \u2014 distinguishes an explicit reset from a URL's true first-ever check."),
3232
+ previousCheckedAt: NullableString.describe("ISO timestamp of the snapshot this was compared against, or null if there was none."),
3233
+ currentCheckedAt: z.string().describe("ISO timestamp of this check, now stored as the new snapshot."),
3234
+ contentHash: z.string().describe("sha256 of the full (untruncated) page content just captured."),
3235
+ previousContentHash: NullableString,
3236
+ summary: z.object({
3237
+ linesAdded: z.number().int().min(0),
3238
+ linesRemoved: z.number().int().min(0),
3239
+ percentChanged: z.number().min(0).max(100).nullable().describe('Proportion of changed lines relative to the larger of the two versions. Null when status is "baseline".')
3240
+ }),
3241
+ hunks: z.array(z.object({
3242
+ type: z.enum(["added", "removed"]).describe("Unchanged context lines are omitted \u2014 only added/removed lines are returned."),
3243
+ lines: z.array(z.string())
3244
+ })).describe("Ordered added/removed line hunks, capped for response size \u2014 see hunksTruncated."),
3245
+ contentTruncated: z.boolean().describe("True if the scraped page exceeded the 250,000-character storable cap and was truncated before hashing/diffing/storing \u2014 changes past that point are invisible to this comparison."),
3246
+ hunksTruncated: z.boolean().describe("True if the hunks list above was capped for response size \u2014 see hunksTruncatedReason."),
3247
+ hunksTruncatedReason: NullableString,
3248
+ totalChangedLineCount: z.number().int().min(0).describe("Total changed lines found before any hunksTruncated capping was applied.")
3249
+ };
3172
3250
  var ExtractSiteOutputSchema = {
3173
3251
  url: z.string(),
3174
3252
  pageCount: z.number().int().min(0),
@@ -4173,6 +4251,13 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
4173
4251
  outputSchema: recordOutputSchema("extract_url", ExtractUrlOutputSchema),
4174
4252
  annotations: liveWebToolAnnotations("Single URL Extract")
4175
4253
  }, async (input) => formatExtractUrl(await executor.extractUrl(input), input));
4254
+ server.registerTool("diff_page", {
4255
+ title: "Page Change Check",
4256
+ description: "Check whether a public URL has changed since you last checked it with this tool: scrapes the current page, diffs it against your last stored snapshot for that URL, and returns what was added or removed (or confirms no change). Stores the new snapshot as the baseline for next time \u2014 on-demand only, no automatic recurring checks. Use extract_url instead when you just want the page's current content with nothing to compare against.",
4257
+ inputSchema: DiffPageInputSchema,
4258
+ outputSchema: recordOutputSchema("diff_page", DiffPageOutputSchema),
4259
+ annotations: liveWebToolAnnotations("Page Change Check")
4260
+ }, async (input) => formatDiffPage(await executor.diffPage(input), input));
4176
4261
  server.registerTool("map_site_urls", {
4177
4262
  title: "Site URL Map",
4178
4263
  description: `Map/crawl a public website for a sitemap, URL inventory, or broken-link scan. Returns internal URLs with HTTP status; ${fileBehavior("maps over 500 URLs are written to a local CSV file instead of inlined.", "large results are stored as a retrievable artifact \u2014 you get an inline summary plus an artifactId for report_artifact_read.")}`,
@@ -4580,6 +4665,9 @@ var HttpMcpToolExecutor = class {
4580
4665
  extractUrl(input) {
4581
4666
  return this.call("/extract-url", input);
4582
4667
  }
4668
+ diffPage(input) {
4669
+ return this.call("/diff-page", input);
4670
+ }
4583
4671
  mapSiteUrls(input) {
4584
4672
  return this.call("/map-urls", input);
4585
4673
  }
@@ -8365,4 +8453,4 @@ export {
8365
8453
  registerMemoryMcpTools,
8366
8454
  MemoryMcpToolExecutor
8367
8455
  };
8368
- //# sourceMappingURL=chunk-WI4A3KFQ.js.map
8456
+ //# sourceMappingURL=chunk-6EXP6DQG.js.map