mcp-scraper 0.34.0 → 0.35.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/README.md +6 -5
  2. package/dist/bin/api-server.cjs +3788 -2742
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +2 -2
  5. package/dist/bin/mcp-scraper-cli.cjs +1 -1
  6. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  7. package/dist/bin/mcp-scraper-cli.js +1 -1
  8. package/dist/bin/mcp-scraper-install.cjs +2 -2
  9. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  10. package/dist/bin/mcp-scraper-install.js +2 -2
  11. package/dist/bin/mcp-stdio-server.cjs +2129 -1932
  12. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  13. package/dist/bin/mcp-stdio-server.js +5 -5
  14. package/dist/bin/paa-harvest.js +2 -2
  15. package/dist/{chunk-V36LS5YV.js → chunk-4ZB3X6BQ.js} +4 -3
  16. package/dist/chunk-4ZB3X6BQ.js.map +1 -0
  17. package/dist/{chunk-JBSGBSGT.js → chunk-BWXLTWF7.js} +21 -16
  18. package/dist/chunk-BWXLTWF7.js.map +1 -0
  19. package/dist/{chunk-J32XSMJI.js → chunk-NPMW5HUS.js} +2350 -2151
  20. package/dist/chunk-NPMW5HUS.js.map +1 -0
  21. package/dist/{chunk-SUOHUXQS.js → chunk-U44TPRST.js} +3 -3
  22. package/dist/chunk-U44TPRST.js.map +1 -0
  23. package/dist/{chunk-4KQVIYIH.js → chunk-XVVNKASZ.js} +2 -2
  24. package/dist/chunk-YR4LJ6AQ.js +7 -0
  25. package/dist/chunk-YR4LJ6AQ.js.map +1 -0
  26. package/dist/{chunk-2EDFOQD7.js → chunk-YRGSEY5L.js} +2 -2
  27. package/dist/{chunk-2EDFOQD7.js.map → chunk-YRGSEY5L.js.map} +1 -1
  28. package/dist/chunk-YV2FUEBX.js +851 -0
  29. package/dist/chunk-YV2FUEBX.js.map +1 -0
  30. package/dist/{extract-bundle-GPZRLBVY.js → extract-bundle-ONWZVV55.js} +59 -8
  31. package/dist/extract-bundle-ONWZVV55.js.map +1 -0
  32. package/dist/index.js +2 -2
  33. package/dist/{server-LKVPEQFE.js → server-GKUTC73B.js} +278 -52
  34. package/dist/server-GKUTC73B.js.map +1 -0
  35. package/dist/{site-extract-repository-JHMVHENZ.js → site-extract-repository-L6BHWVDU.js} +3 -3
  36. package/dist/{worker-ZZHSYI3Y.js → worker-645BZPEK.js} +4 -4
  37. package/docs/mcp-tool-craft-lint.generated.md +171 -49
  38. package/docs/mcp-tool-manifest.generated.json +508 -10
  39. package/docs/specs/query-fanout-transport-contract-fix.md +45 -0
  40. package/package.json +1 -1
  41. package/dist/chunk-J32XSMJI.js.map +0 -1
  42. package/dist/chunk-JBSGBSGT.js.map +0 -1
  43. package/dist/chunk-JWIE5NCR.js +0 -284
  44. package/dist/chunk-JWIE5NCR.js.map +0 -1
  45. package/dist/chunk-NFMIHEYE.js +0 -7
  46. package/dist/chunk-NFMIHEYE.js.map +0 -1
  47. package/dist/chunk-SUOHUXQS.js.map +0 -1
  48. package/dist/chunk-V36LS5YV.js.map +0 -1
  49. package/dist/extract-bundle-GPZRLBVY.js.map +0 -1
  50. package/dist/server-LKVPEQFE.js.map +0 -1
  51. /package/dist/{chunk-4KQVIYIH.js.map → chunk-XVVNKASZ.js.map} +0 -0
  52. /package/dist/{site-extract-repository-JHMVHENZ.js.map → site-extract-repository-L6BHWVDU.js.map} +0 -0
  53. /package/dist/{worker-ZZHSYI3Y.js.map → worker-645BZPEK.js.map} +0 -0
@@ -1,12 +1,12 @@
1
1
  {
2
- "generatedAt": "2026-07-25T03:25:17.950Z",
2
+ "generatedAt": "2026-07-27T21:59:31.780Z",
3
3
  "generatedFrom": "dist/bin/mcp-stdio-server.js",
4
4
  "serverInfo": {
5
5
  "name": "mcp-scraper",
6
- "version": "0.33.7"
6
+ "version": "0.35.1"
7
7
  },
8
8
  "counts": {
9
- "unified_stdio": 167
9
+ "unified_stdio": 168
10
10
  },
11
11
  "surfaces": {
12
12
  "unified_stdio": [
@@ -105,6 +105,7 @@
105
105
  "list-vaults",
106
106
  "list-webhooks",
107
107
  "map_site_urls",
108
+ "map_wayback_snapshots",
108
109
  "maps_place_intel",
109
110
  "maps_search",
110
111
  "memory-backlinks",
@@ -7296,20 +7297,57 @@
7296
7297
  {
7297
7298
  "name": "extract_site",
7298
7299
  "title": "Multi-Page Site Content Crawl",
7299
- "description": "Crawl a public website and return page CONTENT (Markdown) across multiple pages. Bulk crawls over 25 pages are saved as per-page Markdown files in a local folder instead of inlined. Content only — for a technical SEO audit use audit_site instead.",
7300
+ "description": "Crawl a public website and return page CONTENT (Markdown) across multiple pages. A Wayback replay URL produces one archived site snapshot. The optional wayback plan produces whole-site, single-page, or selected-page timelines across explicit months or a month range, all in one export with a capture matrix. Bulk crawls over 25 pages are saved as per-page Markdown files in a local folder instead of inlined. Content only — for a technical SEO audit use audit_site instead.",
7300
7301
  "inputSchema": {
7301
7302
  "type": "object",
7302
7303
  "properties": {
7303
7304
  "url": {
7304
7305
  "type": "string",
7305
7306
  "format": "uri",
7306
- "description": "Public website URL or domain to crawl for page CONTENT (map + scrape). For a technical SEO audit use audit_site instead — this returns content only, not analysis."
7307
+ "description": "Public website URL or web.archive.org replay URL. Without wayback, this crawls live content or one archived site snapshot. With wayback, it creates a multi-month archive timeline."
7307
7308
  },
7308
7309
  "maxPages": {
7309
7310
  "type": "integer",
7310
7311
  "minimum": 1,
7311
7312
  "maximum": 10000,
7312
- "description": "Maximum pages to extract. Bulk crawls (over 25 pages) switch to folder mode: each page saved as its own Markdown file, with a summary plus folder path returned instead of inlining content."
7313
+ "description": "Maximum pages per Wayback month, or maximum total pages for a normal crawl. Multi-month jobs remain capped at 10,000 total captures and 500 pages per month."
7314
+ },
7315
+ "wayback": {
7316
+ "type": "object",
7317
+ "properties": {
7318
+ "months": {
7319
+ "type": "array",
7320
+ "items": {
7321
+ "type": "string",
7322
+ "pattern": "^\\d{4}-(?:0[1-9]|1[0-2])$"
7323
+ },
7324
+ "minItems": 1,
7325
+ "maxItems": 60
7326
+ },
7327
+ "from": {
7328
+ "$ref": "#/properties/wayback/properties/months/items"
7329
+ },
7330
+ "to": {
7331
+ "$ref": "#/properties/wayback/properties/months/items"
7332
+ },
7333
+ "intervalMonths": {
7334
+ "type": "integer",
7335
+ "minimum": 1,
7336
+ "maximum": 12,
7337
+ "default": 1
7338
+ },
7339
+ "urls": {
7340
+ "type": "array",
7341
+ "items": {
7342
+ "type": "string",
7343
+ "format": "uri"
7344
+ },
7345
+ "minItems": 1,
7346
+ "maxItems": 100
7347
+ }
7348
+ },
7349
+ "additionalProperties": false,
7350
+ "description": "Optional temporal archive plan. Provide explicit YYYY-MM months or a from/to range plus intervalMonths. Omit urls for whole-site monthly snapshots, provide one URL for a single-page timeline, or several URLs for selected-page timelines. All results share one durable export."
7313
7351
  },
7314
7352
  "rotateProxies": {
7315
7353
  "type": "boolean",
@@ -7470,7 +7508,7 @@
7470
7508
  "url": {
7471
7509
  "type": "string",
7472
7510
  "format": "uri",
7473
- "description": "Public http/https URL to extract."
7511
+ "description": "Public http/https URL or web.archive.org replay URL to extract."
7474
7512
  },
7475
7513
  "screenshot": {
7476
7514
  "type": "boolean",
@@ -7491,6 +7529,11 @@
7491
7529
  "default": false,
7492
7530
  "description": "Extract brand colors, fonts, logo, and favicon via a rendered browser session."
7493
7531
  },
7532
+ "includeFeaturedImage": {
7533
+ "type": "boolean",
7534
+ "default": false,
7535
+ "description": "Return the best featured image from Open Graph, Twitter, JSON-LD, or page content. For Wayback replay URLs, also returns the timestamp-matched archived image URL when available."
7536
+ },
7494
7537
  "downloadMedia": {
7495
7538
  "type": "boolean",
7496
7539
  "default": false,
@@ -7595,6 +7638,70 @@
7595
7638
  "screenshotSaved": {
7596
7639
  "$ref": "#/properties/title"
7597
7640
  },
7641
+ "archive": {
7642
+ "anyOf": [
7643
+ {
7644
+ "type": "object",
7645
+ "properties": {
7646
+ "timestamp": {
7647
+ "type": "string"
7648
+ },
7649
+ "originalUrl": {
7650
+ "type": "string"
7651
+ },
7652
+ "replayUrl": {
7653
+ "type": "string"
7654
+ },
7655
+ "rawReplayUrl": {
7656
+ "type": "string"
7657
+ }
7658
+ },
7659
+ "required": [
7660
+ "timestamp",
7661
+ "originalUrl",
7662
+ "replayUrl",
7663
+ "rawReplayUrl"
7664
+ ],
7665
+ "additionalProperties": false
7666
+ },
7667
+ {
7668
+ "type": "null"
7669
+ }
7670
+ ]
7671
+ },
7672
+ "featuredImage": {
7673
+ "anyOf": [
7674
+ {
7675
+ "type": "object",
7676
+ "properties": {
7677
+ "url": {
7678
+ "type": "string"
7679
+ },
7680
+ "archivedUrl": {
7681
+ "$ref": "#/properties/title"
7682
+ },
7683
+ "source": {
7684
+ "type": "string",
7685
+ "enum": [
7686
+ "og:image",
7687
+ "twitter:image",
7688
+ "json-ld",
7689
+ "content-image"
7690
+ ]
7691
+ }
7692
+ },
7693
+ "required": [
7694
+ "url",
7695
+ "archivedUrl",
7696
+ "source"
7697
+ ],
7698
+ "additionalProperties": false
7699
+ },
7700
+ {
7701
+ "type": "null"
7702
+ }
7703
+ ]
7704
+ },
7598
7705
  "memory": {
7599
7706
  "type": "object",
7600
7707
  "properties": {
@@ -7638,7 +7745,9 @@
7638
7745
  "entityTypes",
7639
7746
  "napScore",
7640
7747
  "missingSchemaFields",
7641
- "screenshotSaved"
7748
+ "screenshotSaved",
7749
+ "archive",
7750
+ "featuredImage"
7642
7751
  ],
7643
7752
  "additionalProperties": false,
7644
7753
  "$schema": "http://json-schema.org/draft-07/schema#"
@@ -12277,6 +12386,388 @@
12277
12386
  "taskSupport": "forbidden"
12278
12387
  }
12279
12388
  },
12389
+ {
12390
+ "name": "map_wayback_snapshots",
12391
+ "title": "Wayback Snapshot Inventory",
12392
+ "description": "Inventory Wayback Machine captures without scraping their page content. Counts captures, unique archived URLs, unique content digests, first/last captures, monthly/yearly coverage, missing months, and per-URL history across an inclusive date range. Use exact for one page, prefix for one path tree, host for one hostname, domain for subdomains, or urls for selected pages. Counts are exact unless maxCaptures is reached, in which case countType is lower_bound. Set includeCaptures true only when individual timestamps are needed; use extract_site.wayback afterward to download selected copy.",
12393
+ "inputSchema": {
12394
+ "type": "object",
12395
+ "properties": {
12396
+ "url": {
12397
+ "type": "string",
12398
+ "format": "uri",
12399
+ "description": "Original public page/site URL or a web.archive.org replay URL to inventory."
12400
+ },
12401
+ "scope": {
12402
+ "type": "string",
12403
+ "enum": [
12404
+ "exact",
12405
+ "prefix",
12406
+ "host",
12407
+ "domain"
12408
+ ],
12409
+ "default": "exact",
12410
+ "description": "exact = one page; prefix = one path tree; host = one hostname; domain = the domain plus subdomains. Ignored when urls is provided."
12411
+ },
12412
+ "urls": {
12413
+ "type": "array",
12414
+ "items": {
12415
+ "type": "string",
12416
+ "format": "uri"
12417
+ },
12418
+ "minItems": 1,
12419
+ "maxItems": 100,
12420
+ "description": "Optional selected page URLs to inventory together using exact matching. Every URL must belong to the same site as url."
12421
+ },
12422
+ "from": {
12423
+ "type": "string",
12424
+ "pattern": "^(?:\\d{4}|\\d{4}-(?:0[1-9]|1[0-2])|\\d{4}-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12]\\d|3[01])|\\d{14})$",
12425
+ "description": "Inclusive beginning of the archive range: YYYY, YYYY-MM, YYYY-MM-DD, or a 14-digit Wayback timestamp."
12426
+ },
12427
+ "to": {
12428
+ "$ref": "#/properties/from",
12429
+ "description": "Inclusive end of the archive range: YYYY, YYYY-MM, YYYY-MM-DD, or a 14-digit Wayback timestamp."
12430
+ },
12431
+ "successfulHtmlOnly": {
12432
+ "type": "boolean",
12433
+ "default": true,
12434
+ "description": "Count only HTTP 200 text/html captures. Set false to include redirects, errors, and archived assets."
12435
+ },
12436
+ "maxCaptures": {
12437
+ "type": "integer",
12438
+ "minimum": 1,
12439
+ "maximum": 100000,
12440
+ "default": 10000,
12441
+ "description": "Maximum CDX capture rows to scan. If reached, countType is lower_bound instead of exact. Narrow the range or raise this cap for an exact large inventory."
12442
+ },
12443
+ "includeCaptures": {
12444
+ "type": "boolean",
12445
+ "default": false,
12446
+ "description": "Return individual timestamp rows in addition to aggregate counts. Leave false for a compact count-only inventory."
12447
+ },
12448
+ "maxCaptureRows": {
12449
+ "type": "integer",
12450
+ "minimum": 0,
12451
+ "maximum": 1000,
12452
+ "default": 500,
12453
+ "description": "Maximum individual capture rows returned when includeCaptures is true. Aggregated counts still use every scanned capture."
12454
+ }
12455
+ },
12456
+ "required": [
12457
+ "url"
12458
+ ],
12459
+ "additionalProperties": false,
12460
+ "$schema": "http://json-schema.org/draft-07/schema#"
12461
+ },
12462
+ "outputSchema": {
12463
+ "type": "object",
12464
+ "properties": {
12465
+ "url": {
12466
+ "type": "string"
12467
+ },
12468
+ "scope": {
12469
+ "type": "string",
12470
+ "enum": [
12471
+ "exact",
12472
+ "prefix",
12473
+ "host",
12474
+ "domain"
12475
+ ]
12476
+ },
12477
+ "selectedUrls": {
12478
+ "anyOf": [
12479
+ {
12480
+ "type": "array",
12481
+ "items": {
12482
+ "type": "string"
12483
+ }
12484
+ },
12485
+ {
12486
+ "type": "null"
12487
+ }
12488
+ ]
12489
+ },
12490
+ "from": {
12491
+ "type": [
12492
+ "string",
12493
+ "null"
12494
+ ]
12495
+ },
12496
+ "to": {
12497
+ "$ref": "#/properties/from"
12498
+ },
12499
+ "successfulHtmlOnly": {
12500
+ "type": "boolean"
12501
+ },
12502
+ "totalCaptures": {
12503
+ "type": "integer",
12504
+ "minimum": 0
12505
+ },
12506
+ "countType": {
12507
+ "type": "string",
12508
+ "enum": [
12509
+ "exact",
12510
+ "lower_bound"
12511
+ ]
12512
+ },
12513
+ "complete": {
12514
+ "type": "boolean"
12515
+ },
12516
+ "truncated": {
12517
+ "type": "boolean"
12518
+ },
12519
+ "maxCaptures": {
12520
+ "type": "integer",
12521
+ "minimum": 1
12522
+ },
12523
+ "queryPages": {
12524
+ "type": "integer",
12525
+ "minimum": 0
12526
+ },
12527
+ "uniqueUrls": {
12528
+ "type": "integer",
12529
+ "minimum": 0
12530
+ },
12531
+ "uniqueDigests": {
12532
+ "type": "integer",
12533
+ "minimum": 0
12534
+ },
12535
+ "firstCapture": {
12536
+ "anyOf": [
12537
+ {
12538
+ "type": "object",
12539
+ "properties": {
12540
+ "timestamp": {
12541
+ "type": "string"
12542
+ },
12543
+ "originalUrl": {
12544
+ "type": "string"
12545
+ },
12546
+ "digest": {
12547
+ "$ref": "#/properties/from"
12548
+ },
12549
+ "rawReplayUrl": {
12550
+ "type": "string"
12551
+ },
12552
+ "replayUrl": {
12553
+ "$ref": "#/properties/from"
12554
+ },
12555
+ "statusCode": {
12556
+ "anyOf": [
12557
+ {
12558
+ "type": "integer"
12559
+ },
12560
+ {
12561
+ "type": "null"
12562
+ }
12563
+ ]
12564
+ },
12565
+ "mimeType": {
12566
+ "$ref": "#/properties/from"
12567
+ },
12568
+ "length": {
12569
+ "anyOf": [
12570
+ {
12571
+ "type": "integer",
12572
+ "minimum": 0
12573
+ },
12574
+ {
12575
+ "type": "null"
12576
+ }
12577
+ ]
12578
+ }
12579
+ },
12580
+ "required": [
12581
+ "timestamp",
12582
+ "originalUrl",
12583
+ "digest",
12584
+ "rawReplayUrl",
12585
+ "replayUrl"
12586
+ ],
12587
+ "additionalProperties": false
12588
+ },
12589
+ {
12590
+ "type": "null"
12591
+ }
12592
+ ]
12593
+ },
12594
+ "lastCapture": {
12595
+ "anyOf": [
12596
+ {
12597
+ "$ref": "#/properties/firstCapture/anyOf/0"
12598
+ },
12599
+ {
12600
+ "type": "null"
12601
+ }
12602
+ ]
12603
+ },
12604
+ "monthlyCounts": {
12605
+ "type": "array",
12606
+ "items": {
12607
+ "type": "object",
12608
+ "properties": {
12609
+ "month": {
12610
+ "type": "string"
12611
+ },
12612
+ "captures": {
12613
+ "type": "integer",
12614
+ "minimum": 0
12615
+ }
12616
+ },
12617
+ "required": [
12618
+ "month",
12619
+ "captures"
12620
+ ],
12621
+ "additionalProperties": false
12622
+ }
12623
+ },
12624
+ "yearlyCounts": {
12625
+ "type": "array",
12626
+ "items": {
12627
+ "type": "object",
12628
+ "properties": {
12629
+ "year": {
12630
+ "type": "string"
12631
+ },
12632
+ "captures": {
12633
+ "type": "integer",
12634
+ "minimum": 0
12635
+ }
12636
+ },
12637
+ "required": [
12638
+ "year",
12639
+ "captures"
12640
+ ],
12641
+ "additionalProperties": false
12642
+ }
12643
+ },
12644
+ "missingMonths": {
12645
+ "type": "array",
12646
+ "items": {
12647
+ "type": "string"
12648
+ }
12649
+ },
12650
+ "perUrl": {
12651
+ "type": "array",
12652
+ "items": {
12653
+ "type": "object",
12654
+ "properties": {
12655
+ "url": {
12656
+ "type": "string"
12657
+ },
12658
+ "captures": {
12659
+ "type": "integer",
12660
+ "minimum": 0
12661
+ },
12662
+ "uniqueDigests": {
12663
+ "type": "integer",
12664
+ "minimum": 0
12665
+ },
12666
+ "firstTimestamp": {
12667
+ "type": "string"
12668
+ },
12669
+ "lastTimestamp": {
12670
+ "type": "string"
12671
+ }
12672
+ },
12673
+ "required": [
12674
+ "url",
12675
+ "captures",
12676
+ "uniqueDigests",
12677
+ "firstTimestamp",
12678
+ "lastTimestamp"
12679
+ ],
12680
+ "additionalProperties": false
12681
+ }
12682
+ },
12683
+ "perUrlTruncatedCount": {
12684
+ "type": "integer",
12685
+ "minimum": 0
12686
+ },
12687
+ "captures": {
12688
+ "type": "array",
12689
+ "items": {
12690
+ "$ref": "#/properties/firstCapture/anyOf/0"
12691
+ }
12692
+ },
12693
+ "captureRowsTruncatedCount": {
12694
+ "type": "integer",
12695
+ "minimum": 0
12696
+ },
12697
+ "durationMs": {
12698
+ "type": "number",
12699
+ "minimum": 0
12700
+ },
12701
+ "truncatedCount": {
12702
+ "type": "integer",
12703
+ "minimum": 0
12704
+ },
12705
+ "artifact": {
12706
+ "type": "object",
12707
+ "properties": {
12708
+ "artifactId": {
12709
+ "type": "string"
12710
+ },
12711
+ "bytes": {
12712
+ "type": "integer",
12713
+ "minimum": 0
12714
+ },
12715
+ "expiresAt": {
12716
+ "type": "string"
12717
+ },
12718
+ "preview": {
12719
+ "type": "string"
12720
+ }
12721
+ },
12722
+ "required": [
12723
+ "artifactId",
12724
+ "bytes",
12725
+ "expiresAt",
12726
+ "preview"
12727
+ ],
12728
+ "additionalProperties": false
12729
+ }
12730
+ },
12731
+ "required": [
12732
+ "url",
12733
+ "scope",
12734
+ "selectedUrls",
12735
+ "from",
12736
+ "to",
12737
+ "successfulHtmlOnly",
12738
+ "totalCaptures",
12739
+ "countType",
12740
+ "complete",
12741
+ "truncated",
12742
+ "maxCaptures",
12743
+ "queryPages",
12744
+ "uniqueUrls",
12745
+ "uniqueDigests",
12746
+ "firstCapture",
12747
+ "lastCapture",
12748
+ "monthlyCounts",
12749
+ "yearlyCounts",
12750
+ "missingMonths",
12751
+ "perUrl",
12752
+ "perUrlTruncatedCount",
12753
+ "captures",
12754
+ "captureRowsTruncatedCount",
12755
+ "durationMs"
12756
+ ],
12757
+ "additionalProperties": false,
12758
+ "$schema": "http://json-schema.org/draft-07/schema#"
12759
+ },
12760
+ "annotations": {
12761
+ "title": "Wayback Snapshot Inventory",
12762
+ "readOnlyHint": true,
12763
+ "destructiveHint": false,
12764
+ "idempotentHint": false,
12765
+ "openWorldHint": true
12766
+ },
12767
+ "execution": {
12768
+ "taskSupport": "forbidden"
12769
+ }
12770
+ },
12280
12771
  {
12281
12772
  "name": "maps_place_intel",
12282
12773
  "title": "Google Maps Business Profile Details",
@@ -15606,7 +16097,7 @@
15606
16097
  {
15607
16098
  "name": "query_fanout_workflow",
15608
16099
  "title": "Capture AI Search Fan-Out",
15609
- "description": "Capture the query fan-out behind a ChatGPT or Claude web-search answer for AEO: sub-queries issued, every researched URL split into cited vs browsed-only, and top sourced sites. Returns raw structured data for you to classify and analyze. Set export=true for JSON/CSV/TSV/HTML artifacts. WRITE NOTE: passing prompt submits a real message in the user's logged-in account — only send when the user wants that; omit it to capture a prompt the user just ran. The session must already be open on chatgpt.com or claude.ai (see browser_profile_connect) while the prompt streams. NOT for Google AI Overview — use harvest_paa for that.",
16100
+ "description": "Capture the query fan-out behind a ChatGPT or Claude web-search answer for AEO: sub-queries issued, every researched URL split into cited vs browsed-only, and top sourced sites. Complete structured data is always returned inline for analysis. export=true additionally writes JSON/CSV/TSV/HTML only from an installed local MCP server; hosted OAuth/HTTP clients receive exports=null and use the inline data. A local export failure does not discard a successful capture. WRITE NOTE: passing prompt submits a real message in the user's logged-in account — only send when the user wants that; omit it to capture a prompt the user just ran. The session must already be open on chatgpt.com or claude.ai (see browser_profile_connect) while the prompt streams. NOT for Google AI Overview — use harvest_paa for that.",
15610
16101
  "inputSchema": {
15611
16102
  "type": "object",
15612
16103
  "properties": {
@@ -15636,7 +16127,7 @@
15636
16127
  "export": {
15637
16128
  "type": "boolean",
15638
16129
  "default": false,
15639
- "description": "Write JSON/CSV/TSV/HTML exports to MCP_SCRAPER_OUTPUT_DIR/fanout, returning relative paths."
16130
+ "description": "When using the installed local MCP server, write JSON/CSV/TSV/HTML exports to MCP_SCRAPER_OUTPUT_DIR/fanout. Hosted clients such as ChatGPT always receive the complete structured result inline and leave exports null."
15640
16131
  }
15641
16132
  },
15642
16133
  "required": [
@@ -15914,6 +16405,13 @@
15914
16405
  "null"
15915
16406
  ]
15916
16407
  },
16408
+ "export_error": {
16409
+ "type": [
16410
+ "string",
16411
+ "null"
16412
+ ],
16413
+ "description": "Non-fatal local export failure. The inline capture remains complete when this is present."
16414
+ },
15917
16415
  "exports": {
15918
16416
  "anyOf": [
15919
16417
  {
@@ -0,0 +1,45 @@
1
+ # Query fan-out transport contract fix
2
+
3
+ - **Status:** Released and verified in production
4
+ - **Baseline:** `origin/main` at `e2c90f8db7c22fa2aa48fc62cef10e2e0c799e15` (2026-07-25)
5
+ - **Decision:** [Keep hosted fan-out results inline](../decisions/2026-07-25-hosted-fanout-results-stay-inline.md)
6
+ - **Owner:** MCP Scraper browser-agent surface
7
+
8
+ ## Problem
9
+
10
+ `/mcp` correctly registers browser-agent tools with `savesReportsLocally: false`, but `query_fanout_workflow` previously forwarded `input.export` to the browser service and treated remote export paths as usable. An OAuth call with `export: true` could therefore fail while writing an inaccessible sandbox path.
11
+
12
+ ## Contract
13
+
14
+ | Transport | Capture data | `export: true` behavior | Filesystem contract |
15
+ | --- | --- | --- | --- |
16
+ | Local stdio or MCPB | Always inline structured content | Client writes optional JSON/CSV/TSV/HTML after capture | Relative paths under local `MCP_SCRAPER_OUTPUT_DIR/fanout` |
17
+ | Hosted HTTP/OAuth `/mcp` | Always inline structured content | Accepted for compatibility but does not create files | `exports: null`; no remote or sandbox path is returned |
18
+
19
+ The inline result is canonical: queries, researched/cited URLs, snippets, counts, aggregates, and capture metadata must remain immediately usable by an AI.
20
+
21
+ ## Implementation requirements
22
+
23
+ 1. Send `export: false` to `/agent/sessions/:id/capture-fanout` for every transport.
24
+ 2. Return `res.data.result ?? res.data` inline on success and never return `res.data.exports`.
25
+ 3. Export only when `opts.savesReportsLocally !== false`, the caller asked for it, and the inline result is an enriched fan-out capture.
26
+ 4. Keep `exports` nullable. A local write failure returns the successful inline capture, `exports: null`, and optional `export_error`.
27
+ 5. Keep `export` accepted for compatibility and make docs/manifest distinguish local export from hosted inline delivery.
28
+
29
+ ## Verification
30
+
31
+ Focused tests must prove hosted calls force upstream `export: false` and return `exports: null`; local calls force the same upstream value, then write relative local files; and a local write failure is non-fatal. Preserve the 180,000 ms capture timeout allowance (at least 210,000 ms upstream).
32
+
33
+ Live release acceptance requires an explicitly authorized new web-search prompt: hosted OAuth with `export: true` must return a populated inline result with `exports: null`; stdio with the same option must additionally produce readable local exports.
34
+
35
+ ## Release evidence
36
+
37
+ - PR [#64](https://github.com/VilovietaSEO/mcp-scraper/pull/64) merged as `34a5eefa69b1fcaa2dec1dd645ac0dfc55cf4035`.
38
+ - npm, annotated tag, and GitHub/MCPB release published as `0.34.1` / `v0.34.1`.
39
+ - Vercel production deployment `dpl_CFCSkpAywT6PyqRe8LPDaKvicGrp` reached `READY` and serves `mcpscraper.dev`.
40
+ - Hosted Streamable HTTP with OAuth exposed 167 tools and returned a successful fan-out with 21 researched sources, complete inline data, `exports: null`, and no export error.
41
+ - The exact published npm `0.34.1` stdio package exposed 167 tools, returned a successful fan-out with 30 researched sources, and produced nine readable local export files with no export error.
42
+
43
+ ## Out of scope
44
+
45
+ Public/shared download URLs, capture-parser changes, profile authentication, billing, timeout changes, and recovery of fan-out from pre-capture prompts.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mcp-scraper",
3
- "version": "0.34.0",
3
+ "version": "0.35.1",
4
4
  "description": "MCP server for MCP Scraper web intelligence tools",
5
5
  "type": "module",
6
6
  "main": "./dist/index.cjs",