mcp-scraper 0.33.6 → 0.34.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +2 -2
  2. package/dist/bin/api-server.cjs +13558 -12516
  3. package/dist/bin/api-server.cjs.map +1 -1
  4. package/dist/bin/api-server.js +3 -3
  5. package/dist/bin/mcp-scraper-cli.cjs +13 -2
  6. package/dist/bin/mcp-scraper-cli.cjs.map +1 -1
  7. package/dist/bin/mcp-scraper-cli.js +14 -4
  8. package/dist/bin/mcp-scraper-cli.js.map +1 -1
  9. package/dist/bin/mcp-scraper-install.cjs +2 -2
  10. package/dist/bin/mcp-scraper-install.cjs.map +1 -1
  11. package/dist/bin/mcp-scraper-install.js +2 -2
  12. package/dist/bin/mcp-stdio-server.cjs +117 -7
  13. package/dist/bin/mcp-stdio-server.cjs.map +1 -1
  14. package/dist/bin/mcp-stdio-server.js +7 -7
  15. package/dist/bin/paa-harvest.cjs +128 -13
  16. package/dist/bin/paa-harvest.cjs.map +1 -1
  17. package/dist/bin/paa-harvest.js +4 -4
  18. package/dist/{chunk-DUS4ENOY.js → chunk-2EDFOQD7.js} +2 -2
  19. package/dist/{chunk-DUS4ENOY.js.map → chunk-2EDFOQD7.js.map} +1 -1
  20. package/dist/{chunk-C2OD4ELT.js → chunk-3HBPKR5G.js} +2 -2
  21. package/dist/{chunk-A54ELKJK.js → chunk-4KQVIYIH.js} +139 -17
  22. package/dist/chunk-4KQVIYIH.js.map +1 -0
  23. package/dist/{chunk-AIK7WQD5.js → chunk-62DQAWPF.js} +141 -3
  24. package/dist/chunk-62DQAWPF.js.map +1 -0
  25. package/dist/{chunk-V73MPRU6.js → chunk-CB5C3BPB.js} +17 -1
  26. package/dist/chunk-CB5C3BPB.js.map +1 -0
  27. package/dist/{chunk-6ZUK4H66.js → chunk-J32XSMJI.js} +117 -8
  28. package/dist/chunk-J32XSMJI.js.map +1 -0
  29. package/dist/{chunk-LKIP7VDN.js → chunk-JBSGBSGT.js} +46 -3
  30. package/dist/chunk-JBSGBSGT.js.map +1 -0
  31. package/dist/chunk-NFMIHEYE.js +7 -0
  32. package/dist/chunk-NFMIHEYE.js.map +1 -0
  33. package/dist/{chunk-XL6YIWJ3.js → chunk-SUOHUXQS.js} +3 -3
  34. package/dist/{chunk-WJJTGRK4.js → chunk-V36LS5YV.js} +2 -2
  35. package/dist/{chunk-AT2PFWHW.js → chunk-ZID3WQID.js} +2 -2
  36. package/dist/{db-5SZQUI7A.js → db-YAI5AQOI.js} +16 -2
  37. package/dist/{extract-bundle-YNXTFCC3.js → extract-bundle-GPZRLBVY.js} +2 -2
  38. package/dist/index.cjs +128 -13
  39. package/dist/index.cjs.map +1 -1
  40. package/dist/index.d.cts +4 -1
  41. package/dist/index.d.ts +4 -1
  42. package/dist/index.js +4 -4
  43. package/dist/{server-SHOYO52X.js → server-LKVPEQFE.js} +1386 -778
  44. package/dist/server-LKVPEQFE.js.map +1 -0
  45. package/dist/{site-extract-repository-ZDTO7ODZ.js → site-extract-repository-JHMVHENZ.js} +4 -4
  46. package/dist/{worker-BJ64GCDE.js → worker-ZZHSYI3Y.js} +7 -7
  47. package/docs/mcp-tool-manifest.generated.json +213 -3
  48. package/package.json +1 -1
  49. package/dist/chunk-6ZUK4H66.js.map +0 -1
  50. package/dist/chunk-A54ELKJK.js.map +0 -1
  51. package/dist/chunk-AIK7WQD5.js.map +0 -1
  52. package/dist/chunk-F65IIUHI.js +0 -7
  53. package/dist/chunk-F65IIUHI.js.map +0 -1
  54. package/dist/chunk-LKIP7VDN.js.map +0 -1
  55. package/dist/chunk-V73MPRU6.js.map +0 -1
  56. package/dist/server-SHOYO52X.js.map +0 -1
  57. /package/dist/{chunk-C2OD4ELT.js.map → chunk-3HBPKR5G.js.map} +0 -0
  58. /package/dist/{chunk-XL6YIWJ3.js.map → chunk-SUOHUXQS.js.map} +0 -0
  59. /package/dist/{chunk-WJJTGRK4.js.map → chunk-V36LS5YV.js.map} +0 -0
  60. /package/dist/{chunk-AT2PFWHW.js.map → chunk-ZID3WQID.js.map} +0 -0
  61. /package/dist/{db-5SZQUI7A.js.map → db-YAI5AQOI.js.map} +0 -0
  62. /package/dist/{extract-bundle-YNXTFCC3.js.map → extract-bundle-GPZRLBVY.js.map} +0 -0
  63. /package/dist/{site-extract-repository-ZDTO7ODZ.js.map → site-extract-repository-JHMVHENZ.js.map} +0 -0
  64. /package/dist/{worker-BJ64GCDE.js.map → worker-ZZHSYI3Y.js.map} +0 -0
@@ -10,10 +10,10 @@ import {
10
10
  saveExtractPages,
11
11
  setExtractJobTotal,
12
12
  settleExtractJob
13
- } from "./chunk-XL6YIWJ3.js";
14
- import "./chunk-LKIP7VDN.js";
13
+ } from "./chunk-SUOHUXQS.js";
14
+ import "./chunk-JBSGBSGT.js";
15
15
  import "./chunk-M2S27J6Z.js";
16
- import "./chunk-AIK7WQD5.js";
16
+ import "./chunk-62DQAWPF.js";
17
17
  export {
18
18
  completeExtractJob,
19
19
  countSuccessfulPages,
@@ -27,4 +27,4 @@ export {
27
27
  setExtractJobTotal,
28
28
  settleExtractJob
29
29
  };
30
- //# sourceMappingURL=site-extract-repository-ZDTO7ODZ.js.map
30
+ //# sourceMappingURL=site-extract-repository-JHMVHENZ.js.map
@@ -3,20 +3,20 @@ import {
3
3
  createHarvestAttemptRecorder,
4
4
  harvestProblemResponse,
5
5
  serializeHarvestProblem
6
- } from "./chunk-C2OD4ELT.js";
6
+ } from "./chunk-3HBPKR5G.js";
7
7
  import {
8
8
  harvest
9
- } from "./chunk-A54ELKJK.js";
9
+ } from "./chunk-4KQVIYIH.js";
10
10
  import {
11
11
  browserServiceApiKey,
12
12
  runWithCostContext
13
- } from "./chunk-WJJTGRK4.js";
13
+ } from "./chunk-V36LS5YV.js";
14
14
  import "./chunk-K443GQY5.js";
15
- import "./chunk-V73MPRU6.js";
15
+ import "./chunk-CB5C3BPB.js";
16
16
  import {
17
17
  MC_COSTS,
18
18
  serpActualCostMc
19
- } from "./chunk-LKIP7VDN.js";
19
+ } from "./chunk-JBSGBSGT.js";
20
20
  import "./chunk-M2S27J6Z.js";
21
21
  import {
22
22
  claimPendingJob,
@@ -25,7 +25,7 @@ import {
25
25
  debitMc,
26
26
  failJob,
27
27
  listHarvestAttempts
28
- } from "./chunk-AIK7WQD5.js";
28
+ } from "./chunk-62DQAWPF.js";
29
29
 
30
30
  // src/api/webhook.ts
31
31
  async function deliverWebhook(url, payload, retries = 3) {
@@ -139,4 +139,4 @@ export {
139
139
  startWorker,
140
140
  tickOnce
141
141
  };
142
- //# sourceMappingURL=worker-BJ64GCDE.js.map
142
+ //# sourceMappingURL=worker-ZZHSYI3Y.js.map
@@ -1,12 +1,12 @@
1
1
  {
2
- "generatedAt": "2026-07-23T22:07:34.402Z",
2
+ "generatedAt": "2026-07-25T03:25:17.950Z",
3
3
  "generatedFrom": "dist/bin/mcp-stdio-server.js",
4
4
  "serverInfo": {
5
5
  "name": "mcp-scraper",
6
- "version": "0.33.6"
6
+ "version": "0.33.7"
7
7
  },
8
8
  "counts": {
9
- "unified_stdio": 166
9
+ "unified_stdio": 167
10
10
  },
11
11
  "surfaces": {
12
12
  "unified_stdio": [
@@ -133,6 +133,7 @@
133
133
  "read_service_connection",
134
134
  "record-fact",
135
135
  "reddit_thread",
136
+ "reddit_trending",
136
137
  "remove-channel-member",
137
138
  "renew_connected_data_download",
138
139
  "reply-message",
@@ -16723,6 +16724,205 @@
16723
16724
  "taskSupport": "forbidden"
16724
16725
  }
16725
16726
  },
16727
+ {
16728
+ "name": "reddit_trending",
16729
+ "title": "Reddit Trending",
16730
+ "description": "Discover the top Reddit conversations about a topic from the last week or month: finds relevant recent threads via a Google site:reddit.com search (optionally scoped to one subreddit), scrapes them for real upvotes, comments, and the questions people asked, and ranks by engagement (upvotes + 2x comments). Scraping runs in parallel across the discovered threads; set includeComments:false for a fast, cheap discovery-only sweep (relevant thread list, no engagement stats, no per-thread billing) and then read the ones you want with reddit_thread. Not for reading one known thread URL — use reddit_thread for that.",
16731
+ "inputSchema": {
16732
+ "type": "object",
16733
+ "properties": {
16734
+ "topic": {
16735
+ "type": "string",
16736
+ "minLength": 1,
16737
+ "description": "Topic to scan, in plain words (e.g. \"crm for small business\"). Not a URL — pass a known thread URL to reddit_thread instead."
16738
+ },
16739
+ "subreddit": {
16740
+ "type": "string",
16741
+ "minLength": 1,
16742
+ "description": "Bare subreddit name to scope the scan to one community, e.g. \"SEO\" (no r/ prefix, no URL). Omit to scan all of Reddit."
16743
+ },
16744
+ "window": {
16745
+ "type": "string",
16746
+ "enum": [
16747
+ "week",
16748
+ "month"
16749
+ ],
16750
+ "default": "month",
16751
+ "description": "How recent the threads must be: \"week\" or \"month\" (default). Applied via a Google time filter over reddit.com, so it reflects genuine recency."
16752
+ },
16753
+ "maxThreads": {
16754
+ "type": "integer",
16755
+ "minimum": 1,
16756
+ "maximum": 40,
16757
+ "default": 20,
16758
+ "description": "How many discovered threads to scrape and rank. Default 20 (scrape-all). Each scraped thread is billed like reddit_thread + its comments, so lower this to cap cost; raise toward 40 for a wider sweep. Scraping runs in parallel and stops early if it nears the request time limit (partial:true in the response)."
16759
+ },
16760
+ "includeComments": {
16761
+ "type": "boolean",
16762
+ "default": true,
16763
+ "description": "Scrape each discovered thread for real upvotes, comments, and the questions people asked, then rank by engagement. Set false for a fast, cheap discovery-only sweep — returns the discovered threads (title + url) in relevance order with NO engagement stats and NO per-thread billing, so you can then call reddit_thread on the ones you want."
16764
+ },
16765
+ "maxCommentsPerThread": {
16766
+ "type": "integer",
16767
+ "minimum": 1,
16768
+ "maximum": 200,
16769
+ "default": 50,
16770
+ "description": "Comments captured per scraped thread when includeComments is true. Default 50. Billed per captured comment."
16771
+ }
16772
+ },
16773
+ "required": [
16774
+ "topic"
16775
+ ],
16776
+ "additionalProperties": false,
16777
+ "$schema": "http://json-schema.org/draft-07/schema#"
16778
+ },
16779
+ "outputSchema": {
16780
+ "type": "object",
16781
+ "properties": {
16782
+ "topic": {
16783
+ "type": "string"
16784
+ },
16785
+ "subreddit": {
16786
+ "type": [
16787
+ "string",
16788
+ "null"
16789
+ ]
16790
+ },
16791
+ "window": {
16792
+ "type": "string"
16793
+ },
16794
+ "totals": {
16795
+ "type": "object",
16796
+ "properties": {
16797
+ "threads": {
16798
+ "type": "integer",
16799
+ "minimum": 0
16800
+ },
16801
+ "upvotes": {
16802
+ "type": "integer",
16803
+ "minimum": 0
16804
+ },
16805
+ "comments": {
16806
+ "type": "integer",
16807
+ "minimum": 0
16808
+ }
16809
+ },
16810
+ "required": [
16811
+ "threads",
16812
+ "upvotes",
16813
+ "comments"
16814
+ ],
16815
+ "additionalProperties": false
16816
+ },
16817
+ "rankedThreads": {
16818
+ "type": "array",
16819
+ "items": {
16820
+ "type": "object",
16821
+ "properties": {
16822
+ "title": {
16823
+ "type": "string"
16824
+ },
16825
+ "url": {
16826
+ "type": "string"
16827
+ },
16828
+ "subreddit": {
16829
+ "type": "string"
16830
+ },
16831
+ "score": {
16832
+ "type": "integer",
16833
+ "minimum": 0
16834
+ },
16835
+ "commentCount": {
16836
+ "type": "integer",
16837
+ "minimum": 0
16838
+ },
16839
+ "engagementScore": {
16840
+ "type": "integer",
16841
+ "minimum": 0
16842
+ },
16843
+ "ageText": {
16844
+ "type": "string"
16845
+ },
16846
+ "topQuestions": {
16847
+ "type": "array",
16848
+ "items": {
16849
+ "type": "string"
16850
+ }
16851
+ }
16852
+ },
16853
+ "required": [
16854
+ "title",
16855
+ "url",
16856
+ "subreddit",
16857
+ "score",
16858
+ "commentCount",
16859
+ "engagementScore",
16860
+ "ageText",
16861
+ "topQuestions"
16862
+ ],
16863
+ "additionalProperties": false
16864
+ }
16865
+ },
16866
+ "questions": {
16867
+ "type": "array",
16868
+ "items": {
16869
+ "type": "object",
16870
+ "properties": {
16871
+ "question": {
16872
+ "type": "string"
16873
+ },
16874
+ "threadUrl": {
16875
+ "type": "string"
16876
+ }
16877
+ },
16878
+ "required": [
16879
+ "question",
16880
+ "threadUrl"
16881
+ ],
16882
+ "additionalProperties": false
16883
+ }
16884
+ },
16885
+ "threadsScraped": {
16886
+ "type": "integer",
16887
+ "minimum": 0
16888
+ },
16889
+ "candidatesFound": {
16890
+ "type": "integer",
16891
+ "minimum": 0
16892
+ },
16893
+ "partial": {
16894
+ "type": "boolean"
16895
+ },
16896
+ "searchQuery": {
16897
+ "type": "string"
16898
+ }
16899
+ },
16900
+ "required": [
16901
+ "topic",
16902
+ "subreddit",
16903
+ "window",
16904
+ "totals",
16905
+ "rankedThreads",
16906
+ "questions",
16907
+ "threadsScraped",
16908
+ "candidatesFound",
16909
+ "partial",
16910
+ "searchQuery"
16911
+ ],
16912
+ "additionalProperties": false,
16913
+ "$schema": "http://json-schema.org/draft-07/schema#"
16914
+ },
16915
+ "annotations": {
16916
+ "title": "Reddit Trending",
16917
+ "readOnlyHint": true,
16918
+ "destructiveHint": false,
16919
+ "idempotentHint": false,
16920
+ "openWorldHint": true
16921
+ },
16922
+ "execution": {
16923
+ "taskSupport": "forbidden"
16924
+ }
16925
+ },
16726
16926
  {
16727
16927
  "name": "remove-channel-member",
16728
16928
  "title": "Remove Channel Member",
@@ -17497,6 +17697,16 @@
17497
17697
  "maximum": 2,
17498
17698
  "default": 1,
17499
17699
  "description": "Number of result pages to fetch (1–2)."
17700
+ },
17701
+ "recency": {
17702
+ "type": "string",
17703
+ "enum": [
17704
+ "day",
17705
+ "week",
17706
+ "month",
17707
+ "year"
17708
+ ],
17709
+ "description": "Restrict results to a recent time window (Google \"past day/week/month/year\" filter). Omit for all-time. Useful for \"what is being said this week\" style queries; pairs well with a site: operator in the query."
17500
17710
  }
17501
17711
  },
17502
17712
  "required": [
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mcp-scraper",
3
- "version": "0.33.6",
3
+ "version": "0.34.0",
4
4
  "description": "MCP server for MCP Scraper web intelligence tools",
5
5
  "type": "module",
6
6
  "main": "./dist/index.cjs",