mcp-scraper 0.51.2 → 0.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +3 -3
  2. package/dist/bin/api-server.cjs +2819 -869
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +2 -2
  5. package/dist/bin/mcp-scraper-cli.cjs +5 -1
  6. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  7. package/dist/bin/mcp-scraper-cli.js +3 -3
  8. package/dist/bin/mcp-scraper-install.cjs +1 -1
  9. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  10. package/dist/bin/mcp-scraper-install.js +1 -1
  11. package/dist/bin/mcp-stdio-server.cjs +355 -28
  12. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  13. package/dist/bin/mcp-stdio-server.js +2 -2
  14. package/dist/bin/paa-harvest.cjs +4 -0
  15. package/dist/bin/paa-harvest.cjs.map +1 -1
  16. package/dist/bin/paa-harvest.js +2 -2
  17. package/dist/{chunk-ZSJHRZY5.js → chunk-J7TH5KU7.js} +2 -2
  18. package/dist/{chunk-FCZ3QCZP.js → chunk-OT2AF7SH.js} +356 -29
  19. package/dist/chunk-OT2AF7SH.js.map +1 -0
  20. package/dist/chunk-SG2PEHR3.js +562 -0
  21. package/dist/chunk-SG2PEHR3.js.map +1 -0
  22. package/dist/{chunk-GOZIG6HD.js → chunk-TVZD37SE.js} +2 -2
  23. package/dist/{chunk-3ZUBQQPQ.js → chunk-V7CEOUBF.js} +5 -1
  24. package/dist/chunk-V7CEOUBF.js.map +1 -0
  25. package/dist/chunk-VIAXQTZK.js +7 -0
  26. package/dist/chunk-VIAXQTZK.js.map +1 -0
  27. package/dist/{extract-bundle-RCTNANCH.js → extract-bundle-GUEUCTAE.js} +2 -2
  28. package/dist/index.cjs +4 -0
  29. package/dist/index.cjs.map +1 -1
  30. package/dist/index.js +2 -2
  31. package/dist/{server-KAEHVDY7.js → server-7EXEDKAU.js} +1857 -401
  32. package/dist/server-7EXEDKAU.js.map +1 -0
  33. package/dist/{worker-KQN673JF.js → worker-JL4TG6IF.js} +3 -3
  34. package/package.json +1 -1
  35. package/dist/chunk-27FMOD6S.js +0 -430
  36. package/dist/chunk-27FMOD6S.js.map +0 -1
  37. package/dist/chunk-3ZUBQQPQ.js.map +0 -1
  38. package/dist/chunk-FCZ3QCZP.js.map +0 -1
  39. package/dist/chunk-JNRSR5ZJ.js +0 -7
  40. package/dist/chunk-JNRSR5ZJ.js.map +0 -1
  41. package/dist/server-KAEHVDY7.js.map +0 -1
  42. /package/dist/{chunk-ZSJHRZY5.js.map → chunk-J7TH5KU7.js.map} +0 -0
  43. /package/dist/{chunk-GOZIG6HD.js.map → chunk-TVZD37SE.js.map} +0 -0
  44. /package/dist/{extract-bundle-RCTNANCH.js.map → extract-bundle-GUEUCTAE.js.map} +0 -0
  45. /package/dist/{worker-KQN673JF.js.map → worker-JL4TG6IF.js.map} +0 -0
@@ -1,11 +1,11 @@
1
1
  #!/usr/bin/env node
2
2
  import {
3
3
  harvest
4
- } from "../chunk-GOZIG6HD.js";
4
+ } from "../chunk-TVZD37SE.js";
5
5
  import {
6
6
  browserServiceApiKey
7
7
  } from "../chunk-QXAY44SA.js";
8
- import "../chunk-3ZUBQQPQ.js";
8
+ import "../chunk-V7CEOUBF.js";
9
9
  import "../chunk-5UN33CGU.js";
10
10
  import "../chunk-2XTYLJQK.js";
11
11
 
@@ -4,7 +4,7 @@ import {
4
4
  import {
5
5
  DEFAULT_MAPS_PROXY_MODE,
6
6
  postToMemoryLibrary
7
- } from "./chunk-3ZUBQQPQ.js";
7
+ } from "./chunk-V7CEOUBF.js";
8
8
 
9
9
  // src/lib/slugify.ts
10
10
  function slugify(s) {
@@ -1861,4 +1861,4 @@ export {
1861
1861
  runWorkflow,
1862
1862
  runWorkflowStep
1863
1863
  };
1864
- //# sourceMappingURL=chunk-ZSJHRZY5.js.map
1864
+ //# sourceMappingURL=chunk-J7TH5KU7.js.map
@@ -21,7 +21,7 @@ import {
21
21
  } from "./chunk-QXAY44SA.js";
22
22
  import {
23
23
  PACKAGE_VERSION
24
- } from "./chunk-JNRSR5ZJ.js";
24
+ } from "./chunk-VIAXQTZK.js";
25
25
  import {
26
26
  MC_PER_CREDIT
27
27
  } from "./chunk-SOEYDWJU.js";
@@ -330,7 +330,7 @@ seam is noted so you can chain them.
330
330
  \`extract_url\` or \`reddit_thread\`.
331
331
 
332
332
  ## Pages & sites
333
- - One page -> **extract_url** (takes a url).
333
+ - One page -> **extract_url** (takes a url). Set \`preserveMedia:true\` only when the user wants the actual page media; it returns static-plus-rendered provenance, bounded image blocks, and an owner-scoped ZIP readable with \`archive_read\`.
334
334
  - Whole site, crawl + SEO report -> **extract_site** (takes a url).
335
335
  - Wayback replay URLs work with the same tools: \`extract_url\` removes playback chrome and can return
336
336
  a featured image; \`extract_site\` batches nearby archived HTML captures for the replayed site.
@@ -417,8 +417,9 @@ seam is noted so you can chain them.
417
417
  placeUrl, cid; set \`includeServices: true\` to enrich each result where available).
418
418
  - One business deep-dive -> **maps_place_intel** (takes businessName + location, NOT an id; returns
419
419
  reviews, full hours, About attributes, entity IDs/CID, and with \`includeServices: true\`, the full
420
- configured services and areas-served lists. Call it directly with a name from a \`maps_search\` result
421
- or from the user).
420
+ configured services and areas-served lists. Set \`includeImages:true\` only when photos are requested;
421
+ use \`imageScope:"owner"\` for listing-owner photos or \`"all"\` for the full evidence-labeled gallery.
422
+ Call it directly with a name from a \`maps_search\` result or from the user).
422
423
 
423
424
  ## YouTube
424
425
  - Find or list videos -> **youtube_harvest** (returns \`videos[].videoId\`).
@@ -1679,6 +1680,7 @@ ${[h1Lines, h2Lines].filter(Boolean).join("\n")}` : "";
1679
1680
  kpo.address ? `- **Address:** ${kpo.address}` : "",
1680
1681
  kpo.phone ? `- **Phone:** ${kpo.phone}` : "",
1681
1682
  kpo.email ? `- **Email:** ${kpo.email}` : "",
1683
+ kpo.logo ? `- **Structured-data logo:** ${kpo.logo}` : "",
1682
1684
  kpo.faqCount ? `- **FAQ items:** ${kpo.faqCount}` : "",
1683
1685
  kpo.sameAs?.length ? `- **sameAs:** ${kpo.sameAs.slice(0, 5).join(", ")}` : "",
1684
1686
  kpo.missingFields?.length ? `
@@ -1713,15 +1715,23 @@ ${mem.error ?? "unknown error"} \u2014 the page content is still in the truncate
1713
1715
  ## Branding`,
1714
1716
  branding.colorScheme ? `- **Color scheme:** ${branding.colorScheme}` : "",
1715
1717
  `- **Colors:**${Object.entries(branding.colors ?? {}).filter(([, v]) => v).map(([k, v]) => ` ${k}=${v}`).join(",") || " (none extracted)"}`,
1718
+ branding.colorEvidence ? `- **Color evidence:**${Object.entries(branding.colorEvidence).filter(([, v]) => v).map(([key, value]) => value ? ` ${key}=${value.source}/${value.confidence}${value.detail ? ` (${value.detail})` : ""}` : "").join(";") || " (none)"}` : "",
1716
1719
  `- **Fonts:**${Object.entries(branding.fonts ?? {}).filter(([, v]) => v).map(([k, v]) => ` ${k}=${v}`).join(",") || " (none extracted)"}`,
1717
- branding.assets?.logo ? `- **Logo:** ${branding.assets.logo}` : "",
1720
+ branding.assets?.logo ? `- **Logo:** ${branding.assets.logo}${branding.assets.logoConfidence ? ` (${branding.assets.logoConfidence} confidence)` : ""}` : "- **Logo:** no candidate cleared the evidence threshold",
1721
+ branding.assets?.logoSelectionReason ? `- **Logo evidence:** ${branding.assets.logoSelectionReason}` : "",
1722
+ branding.assets?.logoVariants?.length ? `- **Logo variants:** ${branding.assets.logoVariants.join(", ")}` : "",
1723
+ branding.assets?.proofImages?.length ? `- **Proof images:** ${branding.assets.proofImages.map((image) => `${image.proofType} (${image.confidence}) \u2014 ${image.url}${image.context ? ` [${image.context}]` : ""}`).join("; ")}` : "",
1718
1724
  branding.assets?.favicon ? `- **Favicon:** ${branding.assets.favicon}` : ""
1719
1725
  ].filter(Boolean).join("\n") : "";
1720
1726
  const mediaSection = media ? [
1721
1727
  `
1722
1728
  ## Media Assets`,
1723
- `- **Found:** ${media.totalFound} total, ${media.filteredCount} filtered (ads/noise), ${media.assets.length} downloaded`,
1724
- media.outputDir ? `- **Saved to:** ${media.outputDir}` : ""
1729
+ `- **Discovery:** ${media.totalFound} candidates (${media.staticFound} static, ${media.renderedFound} rendered); ${media.assets.length} retained after filtering and responsive-variant collapse`,
1730
+ `- **Downloads:** ${media.assets.filter((asset) => asset.downloadStatus === "downloaded").length} succeeded, ${media.assets.filter((asset) => asset.downloadStatus === "failed").length} failed`,
1731
+ `- **Completeness:** ${media.completeness} \u2014 ${media.exhausted ? "rendered page exhausted after stable scrolling" : `stopped with ${media.stopReason}; more media may exist`}`,
1732
+ media.outputDir ? `- **Saved to:** ${media.outputDir}` : "",
1733
+ media.artifact ? `- **ZIP artifact:** \`${String(media.artifact.artifactId ?? "")}\` (use \`archive_read\` for \`summary.json\` or \`media.jsonl\`)` : "",
1734
+ ...media.warnings.map((warning) => `- ${warning}`)
1725
1735
  ].filter(Boolean).join("\n") : "";
1726
1736
  const archiveSection = archive ? `
1727
1737
  ## Wayback Capture
@@ -1748,6 +1758,25 @@ ${archiveSection}${headingSection}${kpoSection}${brandingSection}${featuredImage
1748
1758
  ${archiveSection}${headingSection}${kpoSection}${brandingSection}${featuredImageSection}${bodySectionFull}${memSection}${screenshotSection}${mediaSection}${tips}`;
1749
1759
  const localBlock = oneBlockWithLocalPath(full, diskReport);
1750
1760
  const textResult = localBlock.result;
1761
+ const structuredMediaAssets = media?.assets.map((asset) => {
1762
+ const { inlinePreview: _preview, ...clean } = asset;
1763
+ return { ...clean, contentIndex: null };
1764
+ }) ?? null;
1765
+ const structuredMedia = media ? {
1766
+ pageUrl: url,
1767
+ staticFound: media.staticFound,
1768
+ renderedFound: media.renderedFound,
1769
+ totalFound: media.totalFound,
1770
+ filteredCount: media.filteredCount,
1771
+ retainedCount: media.assets.length,
1772
+ completeness: media.completeness,
1773
+ exhausted: media.exhausted,
1774
+ stopReason: media.stopReason,
1775
+ scrollRounds: media.scrollRounds,
1776
+ warnings: media.warnings,
1777
+ assets: structuredMediaAssets,
1778
+ artifact: media.artifact
1779
+ } : null;
1751
1780
  const structuredContent = {
1752
1781
  url,
1753
1782
  title: d.title ?? null,
@@ -1755,6 +1784,7 @@ ${archiveSection}${headingSection}${kpoSection}${brandingSection}${featuredImage
1755
1784
  schemaBlockCount: schemaCount,
1756
1785
  entityName: kpo?.entityName ?? null,
1757
1786
  entityTypes: kpo?.type ?? [],
1787
+ structuredDataLogo: kpo?.logo ?? null,
1758
1788
  napScore: kpo?.napScore ?? null,
1759
1789
  missingSchemaFields: kpo?.missingFields ?? [],
1760
1790
  screenshotSaved: screenshotPath ?? null,
@@ -1762,12 +1792,44 @@ ${archiveSection}${headingSection}${kpoSection}${brandingSection}${featuredImage
1762
1792
  archive: archive ?? null,
1763
1793
  featuredImage: featuredImage ?? null,
1764
1794
  branding: branding ?? null,
1765
- mediaAssets: media?.assets ?? null,
1795
+ mediaAssets: structuredMediaAssets,
1796
+ media: structuredMedia,
1766
1797
  memory: mem ?? void 0,
1767
1798
  memoryImages: d.memoryImages ?? void 0,
1768
1799
  delivery: d.delivery ?? void 0,
1769
1800
  localPath: localBlock.localPath ?? void 0
1770
1801
  };
1802
+ const attachMedia = (base, baseStructured) => {
1803
+ const content = [...base.content];
1804
+ if (screenshotMeta?.base64) content.push({ type: "image", data: screenshotMeta.base64, mimeType: "image/png" });
1805
+ const assets = structuredMediaAssets?.map((asset) => ({ ...asset })) ?? null;
1806
+ for (let index = 0; index < (media?.assets.length ?? 0); index += 1) {
1807
+ const preview = media?.assets[index]?.inlinePreview;
1808
+ if (!preview?.data || !preview.mimeType) continue;
1809
+ if (assets?.[index]) assets[index].contentIndex = content.length;
1810
+ content.push({ type: "image", data: preview.data, mimeType: preview.mimeType });
1811
+ }
1812
+ const artifact = media?.artifact;
1813
+ if (artifact && typeof artifact.downloadUrl === "string") {
1814
+ content.push({
1815
+ type: "resource_link",
1816
+ name: String(artifact.filename ?? "website-media.zip"),
1817
+ title: `Download media from ${title}`,
1818
+ uri: artifact.downloadUrl,
1819
+ mimeType: "application/zip",
1820
+ size: typeof artifact.bytes === "number" ? artifact.bytes : void 0
1821
+ });
1822
+ }
1823
+ return {
1824
+ ...base,
1825
+ content,
1826
+ structuredContent: {
1827
+ ...baseStructured,
1828
+ mediaAssets: assets,
1829
+ media: structuredMedia ? { ...structuredMedia, assets } : null
1830
+ }
1831
+ };
1832
+ };
1771
1833
  if (input.delivery !== "inline" && input.delivery !== "memory") {
1772
1834
  const offloaded = await maybeOffload(
1773
1835
  "extract_url",
@@ -1787,22 +1849,12 @@ The full extraction is available as an owned artifact.`,
1787
1849
  retained: true,
1788
1850
  nextAction: "Use report_artifact_read with artifact.artifactId."
1789
1851
  };
1790
- return {
1791
- ...offloaded,
1792
- structuredContent: { ...structuredRecord(offloaded.structuredContent), delivery }
1793
- };
1852
+ return attachMedia({
1853
+ ...offloaded
1854
+ }, { ...structuredRecord(offloaded.structuredContent), delivery });
1794
1855
  }
1795
1856
  }
1796
- if (screenshotMeta?.base64) {
1797
- return {
1798
- content: [
1799
- ...textResult.content,
1800
- { type: "image", data: screenshotMeta.base64, mimeType: "image/png" }
1801
- ],
1802
- structuredContent
1803
- };
1804
- }
1805
- return { ...textResult, structuredContent };
1857
+ return attachMedia(textResult, structuredContent);
1806
1858
  }
1807
1859
  var DIFF_PAGE_PREVIEW_HUNKS = 20;
1808
1860
  function formatDiffPage(raw, input) {
@@ -3579,6 +3631,7 @@ function formatMapsPlaceIntel(raw, input) {
3579
3631
  const lat = d.lat;
3580
3632
  const lng = d.lng;
3581
3633
  const durationMs = d.durationMs;
3634
+ const placeUrl = d.placeUrl;
3582
3635
  const histogram = d.reviewHistogram ?? [];
3583
3636
  const topics = d.reviewTopics ?? [];
3584
3637
  const about = d.aboutAttributes ?? [];
@@ -3587,6 +3640,10 @@ function formatMapsPlaceIntel(raw, input) {
3587
3640
  const services = d.services ?? [];
3588
3641
  const areasServed = d.areasServed ?? [];
3589
3642
  const servicesStatus = d.servicesStatus ?? "not_requested";
3643
+ const media = structuredRecord(d.media);
3644
+ const mediaImages = Array.isArray(media.images) ? media.images.map(structuredRecord) : [];
3645
+ const mediaArtifact = media.artifact && typeof media.artifact === "object" && !Array.isArray(media.artifact) ? media.artifact : null;
3646
+ const mediaWarnings = Array.isArray(media.warnings) ? media.warnings.map(String) : [];
3590
3647
  const hoursTable = d.hoursTable ?? [];
3591
3648
  const ratingLine = [rating, reviewCount ? `(${reviewCount} reviews)` : null].filter(Boolean).join(" ");
3592
3649
  const basicLines = [
@@ -3654,6 +3711,35 @@ ${areasServed.map((a) => `- ${a}`).join("\n")}` : null
3654
3711
  return parts.length ? `
3655
3712
  ## Services & Areas Served
3656
3713
  ${parts.join("\n\n")}` : "";
3714
+ })();
3715
+ const mediaSection = (() => {
3716
+ const status = String(media.status ?? "not_requested");
3717
+ if (status === "not_requested") return "";
3718
+ if (status === "unavailable") return "\n## Images\n> The Google Maps photo gallery could not be retrieved in this run.";
3719
+ const counts = [
3720
+ `${Number(media.imagesCollected ?? mediaImages.length)} collected`,
3721
+ `${Number(media.imagesDownloaded ?? 0)} downloaded`,
3722
+ `${Number(media.ownerImagesCollected ?? 0)} owner`,
3723
+ `${Number(media.otherImagesCollected ?? 0)} other`,
3724
+ `${Number(media.unknownOriginImagesCollected ?? 0)} unknown origin`
3725
+ ].join(" \xB7 ");
3726
+ const completion = media.exhausted === true ? "Gallery exhausted after three stable quiescence checks." : `Collection stopped with \`${String(media.stopReason ?? "unknown")}\`; more photos may exist.`;
3727
+ const artifactLines = mediaArtifact ? [
3728
+ `- **ZIP artifact:** \`${String(mediaArtifact.artifactId ?? "")}\``,
3729
+ typeof mediaArtifact.downloadUrl === "string" ? `- **Download:** ${mediaArtifact.downloadUrl}` : null,
3730
+ typeof mediaArtifact.localPath === "string" ? `- **Local file:** \`${mediaArtifact.localPath}\`` : null,
3731
+ "- **AI readback:** call `archive_read` with the artifactId, then read `summary.json` or `images.jsonl`."
3732
+ ].filter(Boolean).join("\n") : "";
3733
+ const provenance = Number(media.otherImagesCollected ?? 0) > 0 || Number(media.unknownOriginImagesCollected ?? 0) > 0 ? "\n> `other` means absent from a fully exhausted **By owner** gallery. It may include customer, review, Street View, Google, or unattributed media; MCP Scraper does not relabel it as review imagery without evidence." : "";
3734
+ return `
3735
+ ## Images
3736
+ **${counts}**
3737
+
3738
+ ${completion}${provenance}${artifactLines ? `
3739
+
3740
+ ${artifactLines}` : ""}${mediaWarnings.length ? `
3741
+
3742
+ ${mediaWarnings.map((warning) => `- ${warning}`).join("\n")}` : ""}`;
3657
3743
  })();
3658
3744
  const full = [
3659
3745
  `# ${name}`,
@@ -3671,14 +3757,49 @@ ${basicLines}` : null,
3671
3757
  ${entitySection}` : null,
3672
3758
  reviewsSection,
3673
3759
  servicesSection,
3760
+ mediaSection,
3674
3761
  durationMs != null ? `
3675
3762
  ---
3676
3763
  *Extracted in ${(durationMs / 1e3).toFixed(1)}s*` : null
3677
3764
  ].filter(Boolean).join("\n");
3765
+ const content = [{ type: "text", text: full }];
3766
+ const structuredImages = mediaImages.map((image) => {
3767
+ const preview = structuredRecord(image.inlinePreview);
3768
+ let contentIndex = null;
3769
+ if (typeof preview.data === "string" && typeof preview.mimeType === "string") {
3770
+ contentIndex = content.length;
3771
+ content.push({ type: "image", data: preview.data, mimeType: preview.mimeType });
3772
+ }
3773
+ return {
3774
+ index: Number(image.index ?? 0),
3775
+ galleryPosition: typeof image.galleryPosition === "number" ? image.galleryPosition : null,
3776
+ sourceUrl: String(image.sourceUrl ?? ""),
3777
+ mediaKey: String(image.mediaKey ?? ""),
3778
+ origin: image.origin === "owner" || image.origin === "other" ? image.origin : "unknown",
3779
+ originConfidence: String(image.originConfidence ?? ""),
3780
+ filename: typeof image.filename === "string" ? image.filename : null,
3781
+ mimeType: typeof image.mimeType === "string" ? image.mimeType : null,
3782
+ bytes: typeof image.bytes === "number" ? image.bytes : null,
3783
+ downloadStatus: typeof image.downloadStatus === "string" ? image.downloadStatus : "not_attempted",
3784
+ downloadError: typeof image.downloadError === "string" ? image.downloadError : null,
3785
+ contentIndex
3786
+ };
3787
+ });
3788
+ if (mediaArtifact && typeof mediaArtifact.downloadUrl === "string") {
3789
+ content.push({
3790
+ type: "resource_link",
3791
+ name: String(mediaArtifact.filename ?? "Google Maps images.zip"),
3792
+ title: `Download ${name} Google Maps images`,
3793
+ uri: mediaArtifact.downloadUrl,
3794
+ mimeType: "application/zip",
3795
+ size: typeof mediaArtifact.bytes === "number" ? mediaArtifact.bytes : void 0
3796
+ });
3797
+ }
3678
3798
  return {
3679
- ...oneBlock(full),
3799
+ content,
3680
3800
  structuredContent: {
3681
3801
  name,
3802
+ placeUrl: placeUrl ?? null,
3682
3803
  rating: rating ?? null,
3683
3804
  reviewCount: reviewCount ?? null,
3684
3805
  category: category ?? null,
@@ -3686,6 +3807,8 @@ ${entitySection}` : null,
3686
3807
  phone: phone ?? null,
3687
3808
  website: website ?? null,
3688
3809
  hoursSummary: hoursSummary ?? null,
3810
+ hoursTable,
3811
+ plusCode: plusCode ?? null,
3689
3812
  bookingUrl: bookingUrl ?? null,
3690
3813
  kgmid: kgmid ?? null,
3691
3814
  cidDecimal: cidDecimal ?? null,
@@ -3694,10 +3817,49 @@ ${entitySection}` : null,
3694
3817
  lng: lng ?? null,
3695
3818
  reviewsStatus,
3696
3819
  reviewsCollected: reviews.length,
3820
+ reviews: reviews.map((review) => ({
3821
+ reviewId: review.reviewId ?? null,
3822
+ author: review.author ?? null,
3823
+ stars: review.stars ?? null,
3824
+ date: review.date ?? null,
3825
+ text: review.text ?? null,
3826
+ ownerResponse: review.ownerResponse ?? null
3827
+ })),
3828
+ reviewHistogram: histogram.map((row) => ({ stars: Number(row.stars), count: String(row.count ?? "") })),
3697
3829
  reviewTopics: topics.map((t) => ({ label: String(t.label ?? ""), count: String(t.count ?? "") })),
3698
3830
  services,
3699
3831
  areasServed,
3700
- servicesStatus
3832
+ servicesStatus,
3833
+ aboutAttributes: about.map((row) => ({ section: String(row.section ?? ""), attribute: String(row.attribute ?? "") })),
3834
+ media: {
3835
+ status: String(media.status ?? "not_requested"),
3836
+ scope: media.scope === "owner" ? "owner" : "all",
3837
+ requestedMaxImages: Number(media.requestedMaxImages ?? 100),
3838
+ imagesCollected: Number(media.imagesCollected ?? structuredImages.length),
3839
+ imagesDownloaded: Number(media.imagesDownloaded ?? 0),
3840
+ ownerImagesCollected: Number(media.ownerImagesCollected ?? 0),
3841
+ otherImagesCollected: Number(media.otherImagesCollected ?? 0),
3842
+ unknownOriginImagesCollected: Number(media.unknownOriginImagesCollected ?? 0),
3843
+ ownerGalleryAvailable: media.ownerGalleryAvailable === true,
3844
+ ownerGalleryExhausted: media.ownerGalleryExhausted === true,
3845
+ ownerPhotosDiscovered: Number(media.ownerPhotosDiscovered ?? 0),
3846
+ allPhotosDiscovered: Number(media.allPhotosDiscovered ?? 0),
3847
+ exhausted: media.exhausted === true,
3848
+ stopReason: String(media.stopReason ?? "not_requested"),
3849
+ images: structuredImages,
3850
+ artifact: mediaArtifact ? {
3851
+ artifactId: String(mediaArtifact.artifactId ?? ""),
3852
+ filename: String(mediaArtifact.filename ?? ""),
3853
+ contentType: String(mediaArtifact.contentType ?? "application/zip"),
3854
+ bytes: Number(mediaArtifact.bytes ?? 0),
3855
+ sha256: String(mediaArtifact.sha256 ?? ""),
3856
+ expiresAt: String(mediaArtifact.expiresAt ?? ""),
3857
+ downloadUrl: typeof mediaArtifact.downloadUrl === "string" ? mediaArtifact.downloadUrl : null,
3858
+ downloadUrlExpiresAt: typeof mediaArtifact.downloadUrlExpiresAt === "string" ? mediaArtifact.downloadUrlExpiresAt : null,
3859
+ localPath: typeof mediaArtifact.localPath === "string" ? mediaArtifact.localPath : null
3860
+ } : null,
3861
+ warnings: mediaWarnings
3862
+ }
3701
3863
  }
3702
3864
  };
3703
3865
  }
@@ -5130,8 +5292,10 @@ var ExtractUrlBaseInputSchema = {
5130
5292
  includeFeaturedImage: z4.boolean().default(false).describe("Return the best featured image from Open Graph, Twitter, JSON-LD, or page content. For Wayback replay URLs, also returns the timestamp-matched archived image URL when available."),
5131
5293
  downloadMedia: z4.boolean().optional().describe("Deprecated alias for preserveMedia. Omit when using preserveMedia; when omitted, media preservation defaults to false."),
5132
5294
  mediaTypes: z4.array(z4.enum(["image", "video", "audio"])).default(["image", "video", "audio"]).describe("Which media types to download. Default all three."),
5295
+ maxMediaAssets: z4.number().int().min(1).max(250).default(100).describe("Maximum media records to retain and attempt to download after filtering and responsive-variant collapse."),
5296
+ maxInlineImages: z4.number().int().min(0).max(5).default(3).describe("Maximum downloaded images to attach as AI-readable image content blocks. All successfully downloaded media remains available in the ZIP."),
5133
5297
  delivery: z4.enum(["auto", "inline", "artifact", "memory"]).default("auto").describe("Where to deliver the result. auto keeps small results inline and offloads large ones; artifact always returns an owned artifact; memory stores the full page in hosted Memory; inline returns a bounded response."),
5134
- preserveMedia: z4.boolean().default(false).describe("Preserve discovered media in the result workflow. This is the preferred replacement for downloadMedia."),
5298
+ preserveMedia: z4.boolean().default(false).describe("Collect media from static source plus a rendered, lazy-loaded page; collapse responsive variants; return provenance and completeness; attach bounded image previews; and create an owner-scoped ZIP readable with archive_read."),
5135
5299
  depositToVault: z4.boolean().default(false).describe("Save the full page content into the user's MCP Memory vault server-side, embedded for semantic recall \u2014 the full body is NOT returned to chat."),
5136
5300
  vaultName: z4.string().trim().min(1).max(120).optional().describe("Optional vault to deposit into. Defaults to the user's personal vault.")
5137
5301
  };
@@ -5289,7 +5453,11 @@ var MapsPlaceIntelInputSchema = {
5289
5453
  hl: z4.string().length(2).default("en").describe("Language inferred from user request."),
5290
5454
  includeReviews: z4.boolean().default(false).describe("Fetch individual review cards \u2014 for reviews, customer pain, complaints, or praise themes."),
5291
5455
  maxReviews: z4.number().int().min(1).max(500).default(50).describe("Max review cards when includeReviews is true. Default 50, maximum 500."),
5292
- includeServices: z4.boolean().default(false).describe("Fetch the business's configured services list and areas-served list, when the profile has them. Adds one extra page visit; not present for every business.")
5456
+ includeServices: z4.boolean().default(false).describe("Fetch the business's configured services list and areas-served list, when the profile has them. Adds one extra page visit; not present for every business."),
5457
+ includeImages: z4.boolean().default(false).describe("Collect Google Maps listing photos, download them, and return an AI-readable manifest plus an owner-scoped ZIP artifact. The gallery is scrolled until quiescent or maxImages is reached."),
5458
+ imageScope: z4.enum(["owner", "all"]).default("all").describe("owner collects only the Google Maps By owner gallery. all collects the full gallery and labels exact owner matches versus other/unknown media."),
5459
+ maxImages: z4.number().int().min(1).max(250).default(100).describe("Maximum photos to collect when includeImages is true. Default 100, maximum 250."),
5460
+ maxInlineImages: z4.number().int().min(0).max(5).default(3).describe("Maximum downloaded photos attached as MCP image blocks for direct AI vision. The ZIP and structured manifest still contain the wider result.")
5293
5461
  };
5294
5462
  var TrustpilotReviewsInputSchema = {
5295
5463
  domain: z4.string().min(1).describe(`The business's domain as it appears in its Trustpilot URL, e.g. "www.bhphotovideo.com" (include the www. if the site uses it \u2014 pass the domain as-is, do not guess).`),
@@ -6046,6 +6214,37 @@ var SearchSerpOutputSchema = {
6046
6214
  aiOverview: AiOverviewOutput,
6047
6215
  entityIds: EntityIdsOutput
6048
6216
  };
6217
+ var PageMediaAssetOutput = z4.object({
6218
+ url: z4.string(),
6219
+ type: z4.enum(["image", "video", "audio"]),
6220
+ mimeType: NullableString,
6221
+ filename: z4.string(),
6222
+ savedPath: NullableString,
6223
+ sizeBytes: z4.number().int().min(0).nullable(),
6224
+ discoveryMethods: z4.array(z4.string()),
6225
+ altTexts: z4.array(z4.string()),
6226
+ contexts: z4.array(z4.string()),
6227
+ width: z4.number().int().min(0).nullable(),
6228
+ height: z4.number().int().min(0).nullable(),
6229
+ variants: z4.array(z4.string()),
6230
+ finalUrl: NullableString.optional(),
6231
+ duplicateOf: NullableString.optional(),
6232
+ sha256: NullableString.optional(),
6233
+ downloadStatus: z4.enum(["downloaded", "failed", "not_attempted"]).optional(),
6234
+ downloadError: NullableString.optional(),
6235
+ contentIndex: z4.number().int().min(0).nullable()
6236
+ });
6237
+ var PageMediaArtifactOutput = z4.object({
6238
+ artifactId: z4.string(),
6239
+ filename: z4.string(),
6240
+ contentType: z4.string(),
6241
+ bytes: z4.number().int().min(0),
6242
+ sha256: z4.string(),
6243
+ expiresAt: z4.string(),
6244
+ downloadUrl: NullableString,
6245
+ downloadUrlExpiresAt: NullableString,
6246
+ localPath: NullableString
6247
+ });
6049
6248
  var ExtractUrlOutputSchema = {
6050
6249
  url: z4.string(),
6051
6250
  title: NullableString,
@@ -6056,6 +6255,7 @@ var ExtractUrlOutputSchema = {
6056
6255
  schemaBlockCount: z4.number().int().min(0),
6057
6256
  entityName: NullableString,
6058
6257
  entityTypes: z4.array(z4.string()),
6258
+ structuredDataLogo: NullableString.describe("Logo declared by the selected Organization or LocalBusiness JSON-LD entity, separate from the rendered branding candidate ranking."),
6059
6259
  napScore: z4.number().nullable(),
6060
6260
  missingSchemaFields: z4.array(z4.string()),
6061
6261
  screenshotSaved: NullableString,
@@ -6080,6 +6280,78 @@ var ExtractUrlOutputSchema = {
6080
6280
  archivedUrl: NullableString,
6081
6281
  source: z4.enum(["og:image", "twitter:image", "json-ld", "content-image"])
6082
6282
  }).nullable(),
6283
+ branding: z4.object({
6284
+ colorScheme: z4.enum(["light", "dark"]).nullable(),
6285
+ colors: z4.object({
6286
+ primary: NullableString,
6287
+ accent: NullableString,
6288
+ background: NullableString,
6289
+ text: NullableString,
6290
+ heading: NullableString
6291
+ }),
6292
+ colorEvidence: z4.object({
6293
+ primary: z4.object({
6294
+ value: z4.string(),
6295
+ source: z4.enum(["css_variable", "theme_color", "visible_cta", "navigation_background", "header_svg", "visible_background", "body_background", "body_text", "heading_text"]),
6296
+ confidence: z4.enum(["high", "medium", "low"]),
6297
+ detail: NullableString
6298
+ }).nullable(),
6299
+ accent: z4.object({
6300
+ value: z4.string(),
6301
+ source: z4.enum(["css_variable", "theme_color", "visible_cta", "navigation_background", "header_svg", "visible_background", "body_background", "body_text", "heading_text"]),
6302
+ confidence: z4.enum(["high", "medium", "low"]),
6303
+ detail: NullableString
6304
+ }).nullable(),
6305
+ background: z4.object({ value: z4.string(), source: z4.literal("body_background"), confidence: z4.enum(["high", "medium", "low"]), detail: NullableString }).nullable(),
6306
+ text: z4.object({ value: z4.string(), source: z4.literal("body_text"), confidence: z4.enum(["high", "medium", "low"]), detail: NullableString }).nullable(),
6307
+ heading: z4.object({ value: z4.string(), source: z4.literal("heading_text"), confidence: z4.enum(["high", "medium", "low"]), detail: NullableString }).nullable()
6308
+ }).describe("Rendered provenance for each selected color so callers can distinguish explicit brand tokens and semantic elements from lower-confidence fallbacks."),
6309
+ fonts: z4.object({ heading: NullableString, body: NullableString }),
6310
+ assets: z4.object({
6311
+ logo: NullableString,
6312
+ favicon: NullableString,
6313
+ logoConfidence: z4.enum(["high", "medium", "low"]).nullable(),
6314
+ logoSelectionReason: NullableString,
6315
+ logoVariants: z4.array(z4.string()).describe("Responsive or original-size files from the same brand-logo family as logo; never partner, certification, award, press, or customer marks."),
6316
+ logoCandidates: z4.array(z4.object({
6317
+ url: z4.string(),
6318
+ score: z4.number(),
6319
+ confidence: z4.enum(["high", "medium", "low"]),
6320
+ region: z4.enum(["json_ld", "header", "nav", "footer", "body", "favicon"]),
6321
+ evidence: z4.array(z4.string()),
6322
+ alt: NullableString,
6323
+ width: z4.number().nullable(),
6324
+ height: z4.number().nullable()
6325
+ })),
6326
+ proofImages: z4.array(z4.object({
6327
+ url: z4.string(),
6328
+ proofType: z4.enum(["certification", "accreditation", "award", "membership", "partner_or_customer", "press_mention", "trust_mark"]),
6329
+ score: z4.number(),
6330
+ confidence: z4.enum(["high", "medium", "low"]),
6331
+ evidence: z4.array(z4.string()),
6332
+ alt: NullableString,
6333
+ context: NullableString,
6334
+ width: z4.number().nullable(),
6335
+ height: z4.number().nullable()
6336
+ })).describe("Prominent body images supported by trust context, kept separate from the site logo. Classification is evidence-ranked rather than a claim that the depicted organization endorses the site.")
6337
+ })
6338
+ }).nullable().describe("Rendered brand and proof evidence. logo is the site identity, logoVariants are the same mark family, and proofImages are separately typed body trust signals. Inspect confidence and evidence rather than treating uncertain relationships as facts."),
6339
+ mediaAssets: z4.array(PageMediaAssetOutput).nullable().describe("Backward-compatible flattened page-media inventory. Use media for completeness, warnings, and artifact delivery."),
6340
+ media: z4.object({
6341
+ pageUrl: z4.string(),
6342
+ staticFound: z4.number().int().min(0),
6343
+ renderedFound: z4.number().int().min(0),
6344
+ totalFound: z4.number().int().min(0),
6345
+ filteredCount: z4.number().int().min(0),
6346
+ retainedCount: z4.number().int().min(0),
6347
+ completeness: z4.enum(["complete", "partial"]),
6348
+ exhausted: z4.boolean(),
6349
+ stopReason: z4.enum(["page_exhausted", "asset_limit", "scroll_round_limit", "render_unavailable"]),
6350
+ scrollRounds: z4.number().int().min(0),
6351
+ warnings: z4.array(z4.string()),
6352
+ assets: z4.array(PageMediaAssetOutput),
6353
+ artifact: PageMediaArtifactOutput.nullable()
6354
+ }).nullable().describe("Static-plus-rendered website media manifest with provenance, bounded image content-block indices, completion state, and owner-scoped ZIP delivery."),
6083
6355
  memory: z4.object({
6084
6356
  deposited: z4.boolean(),
6085
6357
  vault: z4.string().optional(),
@@ -6270,6 +6542,7 @@ var ArchiveReadOutputSchema = {
6270
6542
  };
6271
6543
  var MapsPlaceIntelOutputSchema = {
6272
6544
  name: z4.string(),
6545
+ placeUrl: z4.string().nullable(),
6273
6546
  rating: NullableString,
6274
6547
  reviewCount: NullableString,
6275
6548
  category: NullableString,
@@ -6277,6 +6550,8 @@ var MapsPlaceIntelOutputSchema = {
6277
6550
  phone: NullableString,
6278
6551
  website: NullableString,
6279
6552
  hoursSummary: NullableString,
6553
+ hoursTable: z4.array(z4.object({ day: z4.string(), hours: z4.string() })),
6554
+ plusCode: NullableString,
6280
6555
  bookingUrl: NullableString,
6281
6556
  kgmid: NullableString,
6282
6557
  cidDecimal: NullableString,
@@ -6285,13 +6560,65 @@ var MapsPlaceIntelOutputSchema = {
6285
6560
  lng: z4.number().nullable(),
6286
6561
  reviewsStatus: z4.string(),
6287
6562
  reviewsCollected: z4.number().int().min(0),
6563
+ reviews: z4.array(z4.object({
6564
+ reviewId: NullableString,
6565
+ author: NullableString,
6566
+ stars: NullableString,
6567
+ date: NullableString,
6568
+ text: NullableString,
6569
+ ownerResponse: NullableString
6570
+ })),
6571
+ reviewHistogram: z4.array(z4.object({ stars: z4.number().int().min(1).max(5), count: z4.string() })),
6288
6572
  reviewTopics: z4.array(z4.object({
6289
6573
  label: z4.string(),
6290
6574
  count: z4.string()
6291
6575
  })),
6292
6576
  services: z4.array(z4.string()),
6293
6577
  areasServed: z4.array(z4.string()),
6294
- servicesStatus: z4.string()
6578
+ servicesStatus: z4.string(),
6579
+ aboutAttributes: z4.array(z4.object({ section: z4.string(), attribute: z4.string() })),
6580
+ media: z4.object({
6581
+ status: z4.string(),
6582
+ scope: z4.enum(["owner", "all"]),
6583
+ requestedMaxImages: z4.number().int().min(1),
6584
+ imagesCollected: z4.number().int().min(0),
6585
+ imagesDownloaded: z4.number().int().min(0),
6586
+ ownerImagesCollected: z4.number().int().min(0),
6587
+ otherImagesCollected: z4.number().int().min(0),
6588
+ unknownOriginImagesCollected: z4.number().int().min(0),
6589
+ ownerGalleryAvailable: z4.boolean(),
6590
+ ownerGalleryExhausted: z4.boolean(),
6591
+ ownerPhotosDiscovered: z4.number().int().min(0),
6592
+ allPhotosDiscovered: z4.number().int().min(0),
6593
+ exhausted: z4.boolean(),
6594
+ stopReason: z4.string(),
6595
+ images: z4.array(z4.object({
6596
+ index: z4.number().int().min(1),
6597
+ galleryPosition: z4.number().int().min(1).nullable(),
6598
+ sourceUrl: z4.string(),
6599
+ mediaKey: z4.string(),
6600
+ origin: z4.enum(["owner", "other", "unknown"]),
6601
+ originConfidence: z4.string(),
6602
+ filename: NullableString,
6603
+ mimeType: NullableString,
6604
+ bytes: z4.number().int().min(0).nullable(),
6605
+ downloadStatus: z4.string(),
6606
+ downloadError: NullableString,
6607
+ contentIndex: z4.number().int().min(1).nullable()
6608
+ })),
6609
+ artifact: z4.object({
6610
+ artifactId: z4.string(),
6611
+ filename: z4.string(),
6612
+ contentType: z4.string(),
6613
+ bytes: z4.number().int().min(0),
6614
+ sha256: z4.string(),
6615
+ expiresAt: z4.string(),
6616
+ downloadUrl: NullableString,
6617
+ downloadUrlExpiresAt: NullableString,
6618
+ localPath: NullableString
6619
+ }).nullable(),
6620
+ warnings: z4.array(z4.string())
6621
+ })
6295
6622
  };
6296
6623
  var TrustpilotReviewsOutputSchema = {
6297
6624
  domain: z4.string(),
@@ -8691,7 +9018,7 @@ function registerPaaExtractorMcpTools(server, executor, options = {}) {
8691
9018
  }, async (input) => formatInstagramMediaDownload(await executor.instagramMediaDownload(input), input));
8692
9019
  server.registerTool("maps_place_intel", {
8693
9020
  title: "Google Maps Business Profile Details",
8694
- description: "Deep-dive one known/named Google Business Profile: rating, reviews, category, address, phone, full hours, About attributes, entity IDs/CID, and \u2014 with includeServices: true \u2014 the full configured services and areas-served lists. Not for category searches or multi-business prospect lists; use maps_search for those. Split business name from location.",
9021
+ description: 'Deep-dive one known/named Google Business Profile: rating, reviews, category, address, phone, website, full hours, About attributes, entity IDs/CID, configured services/areas, and optional photos. Set includeImages:true for a provenance-aware manifest, bounded AI image blocks, and an owner-scoped ZIP; choose imageScope:"owner" for listing-owner photos only. Not for category searches or multi-business prospect lists; use maps_search for those. Split business name from location.',
8695
9022
  inputSchema: MapsPlaceIntelInputSchema,
8696
9023
  outputSchema: recordOutputSchema("maps_place_intel", MapsPlaceIntelOutputSchema),
8697
9024
  annotations: liveWebToolAnnotations("Google Maps Business Profile Details")
@@ -14587,4 +14914,4 @@ export {
14587
14914
  ScheduledResultsMcpExecutor,
14588
14915
  registerScheduledResultsMcpTools
14589
14916
  };
14590
- //# sourceMappingURL=chunk-FCZ3QCZP.js.map
14917
+ //# sourceMappingURL=chunk-OT2AF7SH.js.map