mcp-scraper 0.34.1 → 0.35.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +4 -3
  2. package/dist/bin/api-server.cjs +3709 -2689
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +1 -1
  5. package/dist/bin/mcp-scraper-cli.cjs +1 -1
  6. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  7. package/dist/bin/mcp-scraper-cli.js +1 -1
  8. package/dist/bin/mcp-scraper-install.cjs +2 -2
  9. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  10. package/dist/bin/mcp-scraper-install.js +2 -2
  11. package/dist/bin/mcp-stdio-server.cjs +2102 -1904
  12. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  13. package/dist/bin/mcp-stdio-server.js +3 -3
  14. package/dist/{chunk-6OPHG76G.js → chunk-NPMW5HUS.js} +2338 -2136
  15. package/dist/chunk-NPMW5HUS.js.map +1 -0
  16. package/dist/{chunk-5PZ6N2QM.js → chunk-U44TPRST.js} +2 -2
  17. package/dist/chunk-U44TPRST.js.map +1 -0
  18. package/dist/chunk-YR4LJ6AQ.js +7 -0
  19. package/dist/chunk-YR4LJ6AQ.js.map +1 -0
  20. package/dist/{chunk-7AYRWAEK.js → chunk-YRGSEY5L.js} +2 -2
  21. package/dist/{chunk-7AYRWAEK.js.map → chunk-YRGSEY5L.js.map} +1 -1
  22. package/dist/chunk-YV2FUEBX.js +851 -0
  23. package/dist/chunk-YV2FUEBX.js.map +1 -0
  24. package/dist/{extract-bundle-GPZRLBVY.js → extract-bundle-ONWZVV55.js} +59 -8
  25. package/dist/extract-bundle-ONWZVV55.js.map +1 -0
  26. package/dist/{server-IWDHTES2.js → server-GKUTC73B.js} +224 -22
  27. package/dist/server-GKUTC73B.js.map +1 -0
  28. package/dist/{site-extract-repository-OLVWMOU2.js → site-extract-repository-L6BHWVDU.js} +2 -2
  29. package/docs/mcp-tool-manifest.generated.json +499 -8
  30. package/docs/specs/query-fanout-transport-contract-fix.md +9 -1
  31. package/package.json +1 -1
  32. package/dist/chunk-5PZ6N2QM.js.map +0 -1
  33. package/dist/chunk-6OPHG76G.js.map +0 -1
  34. package/dist/chunk-7N2KYL4U.js +0 -7
  35. package/dist/chunk-7N2KYL4U.js.map +0 -1
  36. package/dist/chunk-JWIE5NCR.js +0 -284
  37. package/dist/chunk-JWIE5NCR.js.map +0 -1
  38. package/dist/extract-bundle-GPZRLBVY.js.map +0 -1
  39. package/dist/server-IWDHTES2.js.map +0 -1
  40. /package/dist/{site-extract-repository-OLVWMOU2.js.map → site-extract-repository-L6BHWVDU.js.map} +0 -0
@@ -10,7 +10,7 @@ import {
10
10
  saveExtractPages,
11
11
  setExtractJobTotal,
12
12
  settleExtractJob
13
- } from "./chunk-5PZ6N2QM.js";
13
+ } from "./chunk-U44TPRST.js";
14
14
  import "./chunk-BWXLTWF7.js";
15
15
  import "./chunk-M2S27J6Z.js";
16
16
  import "./chunk-62DQAWPF.js";
@@ -27,4 +27,4 @@ export {
27
27
  setExtractJobTotal,
28
28
  settleExtractJob
29
29
  };
30
- //# sourceMappingURL=site-extract-repository-OLVWMOU2.js.map
30
+ //# sourceMappingURL=site-extract-repository-L6BHWVDU.js.map
@@ -1,12 +1,12 @@
1
1
  {
2
- "generatedAt": "2026-07-26T13:48:25.203Z",
2
+ "generatedAt": "2026-07-27T21:59:31.780Z",
3
3
  "generatedFrom": "dist/bin/mcp-stdio-server.js",
4
4
  "serverInfo": {
5
5
  "name": "mcp-scraper",
6
- "version": "0.34.1"
6
+ "version": "0.35.1"
7
7
  },
8
8
  "counts": {
9
- "unified_stdio": 167
9
+ "unified_stdio": 168
10
10
  },
11
11
  "surfaces": {
12
12
  "unified_stdio": [
@@ -105,6 +105,7 @@
105
105
  "list-vaults",
106
106
  "list-webhooks",
107
107
  "map_site_urls",
108
+ "map_wayback_snapshots",
108
109
  "maps_place_intel",
109
110
  "maps_search",
110
111
  "memory-backlinks",
@@ -7296,20 +7297,57 @@
7296
7297
  {
7297
7298
  "name": "extract_site",
7298
7299
  "title": "Multi-Page Site Content Crawl",
7299
- "description": "Crawl a public website and return page CONTENT (Markdown) across multiple pages. Bulk crawls over 25 pages are saved as per-page Markdown files in a local folder instead of inlined. Content only — for a technical SEO audit use audit_site instead.",
7300
+ "description": "Crawl a public website and return page CONTENT (Markdown) across multiple pages. A Wayback replay URL produces one archived site snapshot. The optional wayback plan produces whole-site, single-page, or selected-page timelines across explicit months or a month range, all in one export with a capture matrix. Bulk crawls over 25 pages are saved as per-page Markdown files in a local folder instead of inlined. Content only — for a technical SEO audit use audit_site instead.",
7300
7301
  "inputSchema": {
7301
7302
  "type": "object",
7302
7303
  "properties": {
7303
7304
  "url": {
7304
7305
  "type": "string",
7305
7306
  "format": "uri",
7306
- "description": "Public website URL or domain to crawl for page CONTENT (map + scrape). For a technical SEO audit use audit_site instead — this returns content only, not analysis."
7307
+ "description": "Public website URL or web.archive.org replay URL. Without wayback, this crawls live content or one archived site snapshot. With wayback, it creates a multi-month archive timeline."
7307
7308
  },
7308
7309
  "maxPages": {
7309
7310
  "type": "integer",
7310
7311
  "minimum": 1,
7311
7312
  "maximum": 10000,
7312
- "description": "Maximum pages to extract. Bulk crawls (over 25 pages) switch to folder mode: each page saved as its own Markdown file, with a summary plus folder path returned instead of inlining content."
7313
+ "description": "Maximum pages per Wayback month, or maximum total pages for a normal crawl. Multi-month jobs remain capped at 10,000 total captures and 500 pages per month."
7314
+ },
7315
+ "wayback": {
7316
+ "type": "object",
7317
+ "properties": {
7318
+ "months": {
7319
+ "type": "array",
7320
+ "items": {
7321
+ "type": "string",
7322
+ "pattern": "^\\d{4}-(?:0[1-9]|1[0-2])$"
7323
+ },
7324
+ "minItems": 1,
7325
+ "maxItems": 60
7326
+ },
7327
+ "from": {
7328
+ "$ref": "#/properties/wayback/properties/months/items"
7329
+ },
7330
+ "to": {
7331
+ "$ref": "#/properties/wayback/properties/months/items"
7332
+ },
7333
+ "intervalMonths": {
7334
+ "type": "integer",
7335
+ "minimum": 1,
7336
+ "maximum": 12,
7337
+ "default": 1
7338
+ },
7339
+ "urls": {
7340
+ "type": "array",
7341
+ "items": {
7342
+ "type": "string",
7343
+ "format": "uri"
7344
+ },
7345
+ "minItems": 1,
7346
+ "maxItems": 100
7347
+ }
7348
+ },
7349
+ "additionalProperties": false,
7350
+ "description": "Optional temporal archive plan. Provide explicit YYYY-MM months or a from/to range plus intervalMonths. Omit urls for whole-site monthly snapshots, provide one URL for a single-page timeline, or several URLs for selected-page timelines. All results share one durable export."
7313
7351
  },
7314
7352
  "rotateProxies": {
7315
7353
  "type": "boolean",
@@ -7470,7 +7508,7 @@
7470
7508
  "url": {
7471
7509
  "type": "string",
7472
7510
  "format": "uri",
7473
- "description": "Public http/https URL to extract."
7511
+ "description": "Public http/https URL or web.archive.org replay URL to extract."
7474
7512
  },
7475
7513
  "screenshot": {
7476
7514
  "type": "boolean",
@@ -7491,6 +7529,11 @@
7491
7529
  "default": false,
7492
7530
  "description": "Extract brand colors, fonts, logo, and favicon via a rendered browser session."
7493
7531
  },
7532
+ "includeFeaturedImage": {
7533
+ "type": "boolean",
7534
+ "default": false,
7535
+ "description": "Return the best featured image from Open Graph, Twitter, JSON-LD, or page content. For Wayback replay URLs, also returns the timestamp-matched archived image URL when available."
7536
+ },
7494
7537
  "downloadMedia": {
7495
7538
  "type": "boolean",
7496
7539
  "default": false,
@@ -7595,6 +7638,70 @@
7595
7638
  "screenshotSaved": {
7596
7639
  "$ref": "#/properties/title"
7597
7640
  },
7641
+ "archive": {
7642
+ "anyOf": [
7643
+ {
7644
+ "type": "object",
7645
+ "properties": {
7646
+ "timestamp": {
7647
+ "type": "string"
7648
+ },
7649
+ "originalUrl": {
7650
+ "type": "string"
7651
+ },
7652
+ "replayUrl": {
7653
+ "type": "string"
7654
+ },
7655
+ "rawReplayUrl": {
7656
+ "type": "string"
7657
+ }
7658
+ },
7659
+ "required": [
7660
+ "timestamp",
7661
+ "originalUrl",
7662
+ "replayUrl",
7663
+ "rawReplayUrl"
7664
+ ],
7665
+ "additionalProperties": false
7666
+ },
7667
+ {
7668
+ "type": "null"
7669
+ }
7670
+ ]
7671
+ },
7672
+ "featuredImage": {
7673
+ "anyOf": [
7674
+ {
7675
+ "type": "object",
7676
+ "properties": {
7677
+ "url": {
7678
+ "type": "string"
7679
+ },
7680
+ "archivedUrl": {
7681
+ "$ref": "#/properties/title"
7682
+ },
7683
+ "source": {
7684
+ "type": "string",
7685
+ "enum": [
7686
+ "og:image",
7687
+ "twitter:image",
7688
+ "json-ld",
7689
+ "content-image"
7690
+ ]
7691
+ }
7692
+ },
7693
+ "required": [
7694
+ "url",
7695
+ "archivedUrl",
7696
+ "source"
7697
+ ],
7698
+ "additionalProperties": false
7699
+ },
7700
+ {
7701
+ "type": "null"
7702
+ }
7703
+ ]
7704
+ },
7598
7705
  "memory": {
7599
7706
  "type": "object",
7600
7707
  "properties": {
@@ -7638,7 +7745,9 @@
7638
7745
  "entityTypes",
7639
7746
  "napScore",
7640
7747
  "missingSchemaFields",
7641
- "screenshotSaved"
7748
+ "screenshotSaved",
7749
+ "archive",
7750
+ "featuredImage"
7642
7751
  ],
7643
7752
  "additionalProperties": false,
7644
7753
  "$schema": "http://json-schema.org/draft-07/schema#"
@@ -12277,6 +12386,388 @@
12277
12386
  "taskSupport": "forbidden"
12278
12387
  }
12279
12388
  },
12389
+ {
12390
+ "name": "map_wayback_snapshots",
12391
+ "title": "Wayback Snapshot Inventory",
12392
+ "description": "Inventory Wayback Machine captures without scraping their page content. Counts captures, unique archived URLs, unique content digests, first/last captures, monthly/yearly coverage, missing months, and per-URL history across an inclusive date range. Use exact for one page, prefix for one path tree, host for one hostname, domain for subdomains, or urls for selected pages. Counts are exact unless maxCaptures is reached, in which case countType is lower_bound. Set includeCaptures true only when individual timestamps are needed; use extract_site.wayback afterward to download selected copy.",
12393
+ "inputSchema": {
12394
+ "type": "object",
12395
+ "properties": {
12396
+ "url": {
12397
+ "type": "string",
12398
+ "format": "uri",
12399
+ "description": "Original public page/site URL or a web.archive.org replay URL to inventory."
12400
+ },
12401
+ "scope": {
12402
+ "type": "string",
12403
+ "enum": [
12404
+ "exact",
12405
+ "prefix",
12406
+ "host",
12407
+ "domain"
12408
+ ],
12409
+ "default": "exact",
12410
+ "description": "exact = one page; prefix = one path tree; host = one hostname; domain = the domain plus subdomains. Ignored when urls is provided."
12411
+ },
12412
+ "urls": {
12413
+ "type": "array",
12414
+ "items": {
12415
+ "type": "string",
12416
+ "format": "uri"
12417
+ },
12418
+ "minItems": 1,
12419
+ "maxItems": 100,
12420
+ "description": "Optional selected page URLs to inventory together using exact matching. Every URL must belong to the same site as url."
12421
+ },
12422
+ "from": {
12423
+ "type": "string",
12424
+ "pattern": "^(?:\\d{4}|\\d{4}-(?:0[1-9]|1[0-2])|\\d{4}-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12]\\d|3[01])|\\d{14})$",
12425
+ "description": "Inclusive beginning of the archive range: YYYY, YYYY-MM, YYYY-MM-DD, or a 14-digit Wayback timestamp."
12426
+ },
12427
+ "to": {
12428
+ "$ref": "#/properties/from",
12429
+ "description": "Inclusive end of the archive range: YYYY, YYYY-MM, YYYY-MM-DD, or a 14-digit Wayback timestamp."
12430
+ },
12431
+ "successfulHtmlOnly": {
12432
+ "type": "boolean",
12433
+ "default": true,
12434
+ "description": "Count only HTTP 200 text/html captures. Set false to include redirects, errors, and archived assets."
12435
+ },
12436
+ "maxCaptures": {
12437
+ "type": "integer",
12438
+ "minimum": 1,
12439
+ "maximum": 100000,
12440
+ "default": 10000,
12441
+ "description": "Maximum CDX capture rows to scan. If reached, countType is lower_bound instead of exact. Narrow the range or raise this cap for an exact large inventory."
12442
+ },
12443
+ "includeCaptures": {
12444
+ "type": "boolean",
12445
+ "default": false,
12446
+ "description": "Return individual timestamp rows in addition to aggregate counts. Leave false for a compact count-only inventory."
12447
+ },
12448
+ "maxCaptureRows": {
12449
+ "type": "integer",
12450
+ "minimum": 0,
12451
+ "maximum": 1000,
12452
+ "default": 500,
12453
+ "description": "Maximum individual capture rows returned when includeCaptures is true. Aggregated counts still use every scanned capture."
12454
+ }
12455
+ },
12456
+ "required": [
12457
+ "url"
12458
+ ],
12459
+ "additionalProperties": false,
12460
+ "$schema": "http://json-schema.org/draft-07/schema#"
12461
+ },
12462
+ "outputSchema": {
12463
+ "type": "object",
12464
+ "properties": {
12465
+ "url": {
12466
+ "type": "string"
12467
+ },
12468
+ "scope": {
12469
+ "type": "string",
12470
+ "enum": [
12471
+ "exact",
12472
+ "prefix",
12473
+ "host",
12474
+ "domain"
12475
+ ]
12476
+ },
12477
+ "selectedUrls": {
12478
+ "anyOf": [
12479
+ {
12480
+ "type": "array",
12481
+ "items": {
12482
+ "type": "string"
12483
+ }
12484
+ },
12485
+ {
12486
+ "type": "null"
12487
+ }
12488
+ ]
12489
+ },
12490
+ "from": {
12491
+ "type": [
12492
+ "string",
12493
+ "null"
12494
+ ]
12495
+ },
12496
+ "to": {
12497
+ "$ref": "#/properties/from"
12498
+ },
12499
+ "successfulHtmlOnly": {
12500
+ "type": "boolean"
12501
+ },
12502
+ "totalCaptures": {
12503
+ "type": "integer",
12504
+ "minimum": 0
12505
+ },
12506
+ "countType": {
12507
+ "type": "string",
12508
+ "enum": [
12509
+ "exact",
12510
+ "lower_bound"
12511
+ ]
12512
+ },
12513
+ "complete": {
12514
+ "type": "boolean"
12515
+ },
12516
+ "truncated": {
12517
+ "type": "boolean"
12518
+ },
12519
+ "maxCaptures": {
12520
+ "type": "integer",
12521
+ "minimum": 1
12522
+ },
12523
+ "queryPages": {
12524
+ "type": "integer",
12525
+ "minimum": 0
12526
+ },
12527
+ "uniqueUrls": {
12528
+ "type": "integer",
12529
+ "minimum": 0
12530
+ },
12531
+ "uniqueDigests": {
12532
+ "type": "integer",
12533
+ "minimum": 0
12534
+ },
12535
+ "firstCapture": {
12536
+ "anyOf": [
12537
+ {
12538
+ "type": "object",
12539
+ "properties": {
12540
+ "timestamp": {
12541
+ "type": "string"
12542
+ },
12543
+ "originalUrl": {
12544
+ "type": "string"
12545
+ },
12546
+ "digest": {
12547
+ "$ref": "#/properties/from"
12548
+ },
12549
+ "rawReplayUrl": {
12550
+ "type": "string"
12551
+ },
12552
+ "replayUrl": {
12553
+ "$ref": "#/properties/from"
12554
+ },
12555
+ "statusCode": {
12556
+ "anyOf": [
12557
+ {
12558
+ "type": "integer"
12559
+ },
12560
+ {
12561
+ "type": "null"
12562
+ }
12563
+ ]
12564
+ },
12565
+ "mimeType": {
12566
+ "$ref": "#/properties/from"
12567
+ },
12568
+ "length": {
12569
+ "anyOf": [
12570
+ {
12571
+ "type": "integer",
12572
+ "minimum": 0
12573
+ },
12574
+ {
12575
+ "type": "null"
12576
+ }
12577
+ ]
12578
+ }
12579
+ },
12580
+ "required": [
12581
+ "timestamp",
12582
+ "originalUrl",
12583
+ "digest",
12584
+ "rawReplayUrl",
12585
+ "replayUrl"
12586
+ ],
12587
+ "additionalProperties": false
12588
+ },
12589
+ {
12590
+ "type": "null"
12591
+ }
12592
+ ]
12593
+ },
12594
+ "lastCapture": {
12595
+ "anyOf": [
12596
+ {
12597
+ "$ref": "#/properties/firstCapture/anyOf/0"
12598
+ },
12599
+ {
12600
+ "type": "null"
12601
+ }
12602
+ ]
12603
+ },
12604
+ "monthlyCounts": {
12605
+ "type": "array",
12606
+ "items": {
12607
+ "type": "object",
12608
+ "properties": {
12609
+ "month": {
12610
+ "type": "string"
12611
+ },
12612
+ "captures": {
12613
+ "type": "integer",
12614
+ "minimum": 0
12615
+ }
12616
+ },
12617
+ "required": [
12618
+ "month",
12619
+ "captures"
12620
+ ],
12621
+ "additionalProperties": false
12622
+ }
12623
+ },
12624
+ "yearlyCounts": {
12625
+ "type": "array",
12626
+ "items": {
12627
+ "type": "object",
12628
+ "properties": {
12629
+ "year": {
12630
+ "type": "string"
12631
+ },
12632
+ "captures": {
12633
+ "type": "integer",
12634
+ "minimum": 0
12635
+ }
12636
+ },
12637
+ "required": [
12638
+ "year",
12639
+ "captures"
12640
+ ],
12641
+ "additionalProperties": false
12642
+ }
12643
+ },
12644
+ "missingMonths": {
12645
+ "type": "array",
12646
+ "items": {
12647
+ "type": "string"
12648
+ }
12649
+ },
12650
+ "perUrl": {
12651
+ "type": "array",
12652
+ "items": {
12653
+ "type": "object",
12654
+ "properties": {
12655
+ "url": {
12656
+ "type": "string"
12657
+ },
12658
+ "captures": {
12659
+ "type": "integer",
12660
+ "minimum": 0
12661
+ },
12662
+ "uniqueDigests": {
12663
+ "type": "integer",
12664
+ "minimum": 0
12665
+ },
12666
+ "firstTimestamp": {
12667
+ "type": "string"
12668
+ },
12669
+ "lastTimestamp": {
12670
+ "type": "string"
12671
+ }
12672
+ },
12673
+ "required": [
12674
+ "url",
12675
+ "captures",
12676
+ "uniqueDigests",
12677
+ "firstTimestamp",
12678
+ "lastTimestamp"
12679
+ ],
12680
+ "additionalProperties": false
12681
+ }
12682
+ },
12683
+ "perUrlTruncatedCount": {
12684
+ "type": "integer",
12685
+ "minimum": 0
12686
+ },
12687
+ "captures": {
12688
+ "type": "array",
12689
+ "items": {
12690
+ "$ref": "#/properties/firstCapture/anyOf/0"
12691
+ }
12692
+ },
12693
+ "captureRowsTruncatedCount": {
12694
+ "type": "integer",
12695
+ "minimum": 0
12696
+ },
12697
+ "durationMs": {
12698
+ "type": "number",
12699
+ "minimum": 0
12700
+ },
12701
+ "truncatedCount": {
12702
+ "type": "integer",
12703
+ "minimum": 0
12704
+ },
12705
+ "artifact": {
12706
+ "type": "object",
12707
+ "properties": {
12708
+ "artifactId": {
12709
+ "type": "string"
12710
+ },
12711
+ "bytes": {
12712
+ "type": "integer",
12713
+ "minimum": 0
12714
+ },
12715
+ "expiresAt": {
12716
+ "type": "string"
12717
+ },
12718
+ "preview": {
12719
+ "type": "string"
12720
+ }
12721
+ },
12722
+ "required": [
12723
+ "artifactId",
12724
+ "bytes",
12725
+ "expiresAt",
12726
+ "preview"
12727
+ ],
12728
+ "additionalProperties": false
12729
+ }
12730
+ },
12731
+ "required": [
12732
+ "url",
12733
+ "scope",
12734
+ "selectedUrls",
12735
+ "from",
12736
+ "to",
12737
+ "successfulHtmlOnly",
12738
+ "totalCaptures",
12739
+ "countType",
12740
+ "complete",
12741
+ "truncated",
12742
+ "maxCaptures",
12743
+ "queryPages",
12744
+ "uniqueUrls",
12745
+ "uniqueDigests",
12746
+ "firstCapture",
12747
+ "lastCapture",
12748
+ "monthlyCounts",
12749
+ "yearlyCounts",
12750
+ "missingMonths",
12751
+ "perUrl",
12752
+ "perUrlTruncatedCount",
12753
+ "captures",
12754
+ "captureRowsTruncatedCount",
12755
+ "durationMs"
12756
+ ],
12757
+ "additionalProperties": false,
12758
+ "$schema": "http://json-schema.org/draft-07/schema#"
12759
+ },
12760
+ "annotations": {
12761
+ "title": "Wayback Snapshot Inventory",
12762
+ "readOnlyHint": true,
12763
+ "destructiveHint": false,
12764
+ "idempotentHint": false,
12765
+ "openWorldHint": true
12766
+ },
12767
+ "execution": {
12768
+ "taskSupport": "forbidden"
12769
+ }
12770
+ },
12280
12771
  {
12281
12772
  "name": "maps_place_intel",
12282
12773
  "title": "Google Maps Business Profile Details",
@@ -1,6 +1,6 @@
1
1
  # Query fan-out transport contract fix
2
2
 
3
- - **Status:** Implemented; pending deployed live proof
3
+ - **Status:** Released and verified in production
4
4
  - **Baseline:** `origin/main` at `e2c90f8db7c22fa2aa48fc62cef10e2e0c799e15` (2026-07-25)
5
5
  - **Decision:** [Keep hosted fan-out results inline](../decisions/2026-07-25-hosted-fanout-results-stay-inline.md)
6
6
  - **Owner:** MCP Scraper browser-agent surface
@@ -32,6 +32,14 @@ Focused tests must prove hosted calls force upstream `export: false` and return
32
32
 
33
33
  Live release acceptance requires an explicitly authorized new web-search prompt: hosted OAuth with `export: true` must return a populated inline result with `exports: null`; stdio with the same option must additionally produce readable local exports.
34
34
 
35
+ ## Release evidence
36
+
37
+ - PR [#64](https://github.com/VilovietaSEO/mcp-scraper/pull/64) merged as `34a5eefa69b1fcaa2dec1dd645ac0dfc55cf4035`.
38
+ - npm, annotated tag, and GitHub/MCPB release published as `0.34.1` / `v0.34.1`.
39
+ - Vercel production deployment `dpl_CFCSkpAywT6PyqRe8LPDaKvicGrp` reached `READY` and serves `mcpscraper.dev`.
40
+ - Hosted Streamable HTTP with OAuth exposed 167 tools and returned a successful fan-out with 21 researched sources, complete inline data, `exports: null`, and no export error.
41
+ - The exact published npm `0.34.1` stdio package exposed 167 tools, returned a successful fan-out with 30 researched sources, and produced nine readable local export files with no export error.
42
+
35
43
  ## Out of scope
36
44
 
37
45
  Public/shared download URLs, capture-parser changes, profile authentication, billing, timeout changes, and recovery of fan-out from pre-capture prompts.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mcp-scraper",
3
- "version": "0.34.1",
3
+ "version": "0.35.1",
4
4
  "description": "MCP server for MCP Scraper web intelligence tools",
5
5
  "type": "module",
6
6
  "main": "./dist/index.cjs",
@@ -1 +0,0 @@
1
- {"version":3,"sources":["../src/api/site-extract-repository.ts"],"sourcesContent":["import { getDb } from './db.js'\nimport { sanitizeVendorName } from '../errors.js'\nimport { LedgerOperation } from './rates.js'\nimport type { PageData } from './site-extractor.js'\n\nexport type ExtractJobStatus = 'pending' | 'running' | 'complete' | 'failed'\n\nexport interface ExtractJobRow {\n id: string\n userId: number | null\n status: ExtractJobStatus\n startUrl: string\n options: Record<string, unknown>\n totalUrls: number\n doneUrls: number\n artifacts: Array<{ key: string; url: string; bytes: number; contentType: string }> | null\n error: string | null\n billedMc: number | null\n createdAt: string\n updatedAt: string\n}\n\nfunction rowToJob(r: Record<string, unknown>): ExtractJobRow {\n return {\n id: String(r.id),\n userId: r.user_id != null ? Number(r.user_id) : null,\n status: (r.status != null ? String(r.status) : 'pending') as ExtractJobStatus,\n startUrl: String(r.start_url ?? ''),\n options: r.options ? JSON.parse(String(r.options)) : {},\n totalUrls: Number(r.total_urls ?? 0),\n doneUrls: Number(r.done_urls ?? 0),\n artifacts: r.artifacts ? JSON.parse(String(r.artifacts)) : null,\n error: r.error != null ? String(r.error) : null,\n billedMc: r.billed_mc != null ? Number(r.billed_mc) : null,\n createdAt: String(r.created_at ?? ''),\n updatedAt: String(r.updated_at ?? ''),\n }\n}\n\nexport async function createExtractJob(jobId: string, userId: number, startUrl: string, options: Record<string, unknown>): Promise<void> {\n const db = getDb()\n await db.execute({\n sql: `INSERT INTO site_extract_jobs (id, user_id, status, start_url, options, created_at, updated_at)\n VALUES (?, ?, 'pending', ?, ?, datetime('now'), datetime('now'))`,\n args: [jobId, userId, startUrl, JSON.stringify(options)],\n })\n}\n\nexport async function getExtractJob(jobId: string): Promise<ExtractJobRow | null> {\n const db = getDb()\n const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE id = ?`, args: [jobId] })\n return res.rows[0] ? rowToJob(res.rows[0] as unknown as Record<string, unknown>) : null\n}\n\nexport async function listExtractJobs(userId: number): Promise<ExtractJobRow[]> {\n const db = getDb()\n const res = await db.execute({ sql: `SELECT * FROM site_extract_jobs WHERE user_id = ? ORDER BY created_at DESC LIMIT 50`, args: [userId] })\n return res.rows.map(r => rowToJob(r as unknown as Record<string, unknown>))\n}\n\nexport async function setExtractJobTotal(jobId: string, totalUrls: number): Promise<void> {\n const db = getDb()\n await db.execute({\n sql: `UPDATE site_extract_jobs SET status = 'running', total_urls = ?, updated_at = datetime('now') WHERE id = ?`,\n args: [totalUrls, jobId],\n })\n}\n\nexport async function saveExtractPages(jobId: string, pages: PageData[]): Promise<void> {\n if (pages.length === 0) return\n const db = getDb()\n await db.batch(\n pages.map(p => ({\n sql: `INSERT OR REPLACE INTO site_extract_pages (job_id, url, page) VALUES (?, ?, ?)`,\n args: [jobId, p.url, JSON.stringify({ ...p, bodyMarkdown: '', schema: [] })],\n })),\n )\n await db.execute({\n sql: `UPDATE site_extract_jobs\n SET done_urls = (SELECT COUNT(*) FROM site_extract_pages WHERE job_id = ?), updated_at = datetime('now')\n WHERE id = ?`,\n args: [jobId, jobId],\n })\n}\n\nexport async function getExtractedPages(jobId: string): Promise<PageData[]> {\n const db = getDb()\n const res = await db.execute({ sql: `SELECT page FROM site_extract_pages WHERE job_id = ?`, args: [jobId] })\n return res.rows.map(r => JSON.parse(String((r as unknown as Record<string, unknown>).page)) as PageData)\n}\n\nexport async function countSuccessfulPages(jobId: string): Promise<number> {\n const db = getDb()\n const res = await db.execute({\n sql: `SELECT COUNT(*) AS n FROM site_extract_pages\n WHERE job_id = ? AND json_extract(page, '$.status') = 200 AND json_extract(page, '$.wordCount') > 0`,\n args: [jobId],\n })\n return Number((res.rows[0] as unknown as Record<string, unknown>)?.n ?? 0)\n}\n\nexport async function getExtractedUrls(jobId: string): Promise<Set<string>> {\n const db = getDb()\n const res = await db.execute({ sql: `SELECT url FROM site_extract_pages WHERE job_id = ?`, args: [jobId] })\n return new Set(res.rows.map(r => String((r as unknown as Record<string, unknown>).url)))\n}\n\nexport async function completeExtractJob(jobId: string, artifacts: ExtractJobRow['artifacts']): Promise<void> {\n const db = getDb()\n await db.execute({\n sql: `UPDATE site_extract_jobs SET status = 'complete', artifacts = ?, updated_at = datetime('now') WHERE id = ?`,\n args: [JSON.stringify(artifacts ?? []), jobId],\n })\n}\n\nexport async function failExtractJob(jobId: string, error: string): Promise<void> {\n const db = getDb()\n await db.execute({\n sql: `UPDATE site_extract_jobs SET status = 'failed', error = ?, updated_at = datetime('now') WHERE id = ?`,\n args: [sanitizeVendorName(error).slice(0, 2000), jobId],\n })\n}\n\nexport async function settleExtractJob(jobId: string, userId: number, refundMc: number, netChargeMc: number, reference: string): Promise<void> {\n const db = getDb()\n const job = await getExtractJob(jobId)\n if (!job || job.billedMc != null) return\n if (refundMc <= 0) {\n await db.execute({ sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now') WHERE id = ? AND billed_mc IS NULL`, args: [Math.max(0, netChargeMc), jobId] })\n return\n }\n await db.batch([\n { sql: 'UPDATE users SET balance_mc = balance_mc + ? WHERE id = ?', args: [refundMc, userId] },\n { sql: 'INSERT INTO ledger (user_id, amount_mc, operation, description) VALUES (?, ?, ?, ?)', args: [userId, refundMc, LedgerOperation.EXTRACT_SITE_REFUND, reference] },\n { sql: `UPDATE site_extract_jobs SET billed_mc = ?, updated_at = datetime('now') WHERE id = ? AND billed_mc IS NULL`, args: [Math.max(0, netChargeMc), jobId] },\n ])\n}\n"],"mappings":";;;;;;;;;;;AAsBA,SAAS,SAAS,GAA2C;AAC3D,SAAO;AAAA,IACL,IAAI,OAAO,EAAE,EAAE;AAAA,IACf,QAAQ,EAAE,WAAW,OAAO,OAAO,EAAE,OAAO,IAAI;AAAA,IAChD,QAAS,EAAE,UAAU,OAAO,OAAO,EAAE,MAAM,IAAI;AAAA,IAC/C,UAAU,OAAO,EAAE,aAAa,EAAE;AAAA,IAClC,SAAS,EAAE,UAAU,KAAK,MAAM,OAAO,EAAE,OAAO,CAAC,IAAI,CAAC;AAAA,IACtD,WAAW,OAAO,EAAE,cAAc,CAAC;AAAA,IACnC,UAAU,OAAO,EAAE,aAAa,CAAC;AAAA,IACjC,WAAW,EAAE,YAAY,KAAK,MAAM,OAAO,EAAE,SAAS,CAAC,IAAI;AAAA,IAC3D,OAAO,EAAE,SAAS,OAAO,OAAO,EAAE,KAAK,IAAI;AAAA,IAC3C,UAAU,EAAE,aAAa,OAAO,OAAO,EAAE,SAAS,IAAI;AAAA,IACtD,WAAW,OAAO,EAAE,cAAc,EAAE;AAAA,IACpC,WAAW,OAAO,EAAE,cAAc,EAAE;AAAA,EACtC;AACF;AAEA,eAAsB,iBAAiB,OAAe,QAAgB,UAAkB,SAAiD;AACvI,QAAM,KAAK,MAAM;AACjB,QAAM,GAAG,QAAQ;AAAA,IACf,KAAK;AAAA;AAAA,IAEL,MAAM,CAAC,OAAO,QAAQ,UAAU,KAAK,UAAU,OAAO,CAAC;AAAA,EACzD,CAAC;AACH;AAEA,eAAsB,cAAc,OAA8C;AAChF,QAAM,KAAK,MAAM;AACjB,QAAM,MAAM,MAAM,GAAG,QAAQ,EAAE,KAAK,gDAAgD,MAAM,CAAC,KAAK,EAAE,CAAC;AACnG,SAAO,IAAI,KAAK,CAAC,IAAI,SAAS,IAAI,KAAK,CAAC,CAAuC,IAAI;AACrF;AAEA,eAAsB,gBAAgB,QAA0C;AAC9E,QAAM,KAAK,MAAM;AACjB,QAAM,MAAM,MAAM,GAAG,QAAQ,EAAE,KAAK,uFAAuF,MAAM,CAAC,MAAM,EAAE,CAAC;AAC3I,SAAO,IAAI,KAAK,IAAI,OAAK,SAAS,CAAuC,CAAC;AAC5E;AAEA,eAAsB,mBAAmB,OAAe,WAAkC;AACxF,QAAM,KAAK,MAAM;AACjB,QAAM,GAAG,QAAQ;AAAA,IACf,KAAK;AAAA,IACL,MAAM,CAAC,WAAW,KAAK;AAAA,EACzB,CAAC;AACH;AAEA,eAAsB,iBAAiB,OAAe,OAAkC;AACtF,MAAI,MAAM,WAAW,EAAG;AACxB,QAAM,KAAK,MAAM;AACjB,QAAM,GAAG;AAAA,IACP,MAAM,IAAI,QAAM;AAAA,MACd,KAAK;AAAA,MACL,MAAM,CAAC,OAAO,EAAE,KAAK,KAAK,UAAU,EAAE,GAAG,GAAG,cAAc,IAAI,QAAQ,CAAC,EAAE,CAAC,CAAC;AAAA,IAC7E,EAAE;AAAA,EACJ;AACA,QAAM,GAAG,QAAQ;AAAA,IACf,KAAK;AAAA;AAAA;AAAA,IAGL,MAAM,CAAC,OAAO,KAAK;AAAA,EACrB,CAAC;AACH;AAEA,eAAsB,kBAAkB,OAAoC;AAC1E,QAAM,KAAK,MAAM;AACjB,QAAM,MAAM,MAAM,GAAG,QAAQ,EAAE,KAAK,wDAAwD,MAAM,CAAC,KAAK,EAAE,CAAC;AAC3G,SAAO,IAAI,KAAK,IAAI,OAAK,KAAK,MAAM,OAAQ,EAAyC,IAAI,CAAC,CAAa;AACzG;AAEA,eAAsB,qBAAqB,OAAgC;AACzE,QAAM,KAAK,MAAM;AACjB,QAAM,MAAM,MAAM,GAAG,QAAQ;AAAA,IAC3B,KAAK;AAAA;AAAA,IAEL,MAAM,CAAC,KAAK;AAAA,EACd,CAAC;AACD,SAAO,OAAQ,IAAI,KAAK,CAAC,GAA0C,KAAK,CAAC;AAC3E;AAEA,eAAsB,iBAAiB,OAAqC;AAC1E,QAAM,KAAK,MAAM;AACjB,QAAM,MAAM,MAAM,GAAG,QAAQ,EAAE,KAAK,uDAAuD,MAAM,CAAC,KAAK,EAAE,CAAC;AAC1G,SAAO,IAAI,IAAI,IAAI,KAAK,IAAI,OAAK,OAAQ,EAAyC,GAAG,CAAC,CAAC;AACzF;AAEA,eAAsB,mBAAmB,OAAe,WAAsD;AAC5G,QAAM,KAAK,MAAM;AACjB,QAAM,GAAG,QAAQ;AAAA,IACf,KAAK;AAAA,IACL,MAAM,CAAC,KAAK,UAAU,aAAa,CAAC,CAAC,GAAG,KAAK;AAAA,EAC/C,CAAC;AACH;AAEA,eAAsB,eAAe,OAAe,OAA8B;AAChF,QAAM,KAAK,MAAM;AACjB,QAAM,GAAG,QAAQ;AAAA,IACf,KAAK;AAAA,IACL,MAAM,CAAC,mBAAmB,KAAK,EAAE,MAAM,GAAG,GAAI,GAAG,KAAK;AAAA,EACxD,CAAC;AACH;AAEA,eAAsB,iBAAiB,OAAe,QAAgB,UAAkB,aAAqB,WAAkC;AAC7I,QAAM,KAAK,MAAM;AACjB,QAAM,MAAM,MAAM,cAAc,KAAK;AACrC,MAAI,CAAC,OAAO,IAAI,YAAY,KAAM;AAClC,MAAI,YAAY,GAAG;AACjB,UAAM,GAAG,QAAQ,EAAE,KAAK,+GAA+G,MAAM,CAAC,KAAK,IAAI,GAAG,WAAW,GAAG,KAAK,EAAE,CAAC;AAChL;AAAA,EACF;AACA,QAAM,GAAG,MAAM;AAAA,IACb,EAAE,KAAK,6DAA6D,MAAM,CAAC,UAAU,MAAM,EAAE;AAAA,IAC7F,EAAE,KAAK,uFAAuF,MAAM,CAAC,QAAQ,UAAU,gBAAgB,qBAAqB,SAAS,EAAE;AAAA,IACvK,EAAE,KAAK,+GAA+G,MAAM,CAAC,KAAK,IAAI,GAAG,WAAW,GAAG,KAAK,EAAE;AAAA,EAChK,CAAC;AACH;","names":[]}