commoncrawl-mcp 1.0.1 → 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -41,3 +41,22 @@ This project is intentionally narrow. It should be treated as a practical helper
41
41
  ## Try it
42
42
 
43
43
  After building, connect the server through your MCP client. The repository root also contains `smoke-test.mjs` for projects covered by the shared harness. A typical tool call starts with `list_indexes`.
44
+
45
+ <!-- paywall -->
46
+ ## Premium tools
47
+
48
+ These tools need a license key:
49
+
50
+ * `capture_history`
51
+ * `index_stats`
52
+
53
+ Buy a key at https://mcp-marketplace.io/server/io-github-mrfentmen-commoncrawl-mcp and set it in your MCP client config:
54
+
55
+ ```json
56
+ "env": { "MCP_LICENSE_KEY": "mcp_live_..." }
57
+ ```
58
+
59
+ The key is checked against MCP Marketplace, cached for 24 hours, and keeps
60
+ working offline once one check has succeeded. Every other tool on this server
61
+ stays free.
62
+ <!-- /paywall -->
package/dist/api.d.ts CHANGED
@@ -15,4 +15,24 @@ export declare function searchCaptures(urlPattern: string, index: string | undef
15
15
  records: any[];
16
16
  }>;
17
17
  export declare function format(value: unknown): string;
18
+ export interface HistoryArgs {
19
+ url: string;
20
+ index?: string;
21
+ match_type?: string;
22
+ from_year?: number;
23
+ to_year?: number;
24
+ limit?: number;
25
+ }
26
+ export declare function captureHistory(args: HistoryArgs): Promise<string>;
27
+ export interface StatsArgs {
28
+ url: string;
29
+ index?: string;
30
+ match_type?: string;
31
+ sample?: number;
32
+ }
33
+ export declare function indexStats(args: StatsArgs): Promise<string>;
34
+ export interface CollectionArgs {
35
+ id: string;
36
+ }
37
+ export declare function collectionDetails(args: CollectionArgs): Promise<string>;
18
38
  export {};
package/dist/api.js CHANGED
@@ -42,3 +42,170 @@ export async function searchCaptures(urlPattern, index, page, limit) {
42
42
  export function format(value) {
43
43
  return JSON.stringify(value, null, 2).slice(0, 16000);
44
44
  }
45
+ // ---- muxA tools:
46
+ const CC = "https://index.commoncrawl.org";
47
+ function cap(value, fallback, max) {
48
+ return Math.max(1, Math.min(Math.round(value ?? fallback), max));
49
+ }
50
+ function ccLine(raw) {
51
+ try {
52
+ const d = JSON.parse(raw);
53
+ return typeof d?.url === "string" ? d : null;
54
+ }
55
+ catch {
56
+ return null;
57
+ }
58
+ }
59
+ async function cdx(index, params, limit) {
60
+ const query = new URLSearchParams({ ...params, output: "json" });
61
+ const url = `${CC}/${index}-index?${query.toString()}`;
62
+ const res = await fetch(url, {
63
+ headers: { "User-Agent": "mrfentmen-commoncrawl-mcp/1.0", Accept: "application/json" },
64
+ signal: AbortSignal.timeout(45000),
65
+ });
66
+ if (!res.ok) {
67
+ const body = await res.text().catch(() => "");
68
+ throw new Error(`Common Crawl index returned ${res.status}. ${body.slice(0, 200)}`.trim());
69
+ }
70
+ const text = await res.text();
71
+ const rows = [];
72
+ for (const line of text.split("\n")) {
73
+ if (!line.trim())
74
+ continue;
75
+ const d = ccLine(line);
76
+ if (d)
77
+ rows.push(d);
78
+ if (rows.length >= limit)
79
+ break;
80
+ }
81
+ return { rows, pagesFetched: 1 };
82
+ }
83
+ async function pickIndex(id) {
84
+ if (id?.trim())
85
+ return id.trim();
86
+ return latestIndex();
87
+ }
88
+ function stamp(value) {
89
+ const s = String(value ?? "");
90
+ return /^\d{14}$/.test(s) ? `${s.slice(0, 4)}-${s.slice(4, 6)}-${s.slice(6, 8)} ${s.slice(8, 10)}:${s.slice(10, 12)}:${s.slice(12, 14)} UTC` : s || "unknown";
91
+ }
92
+ export async function captureHistory(args) {
93
+ const target = (args.url ?? "").trim();
94
+ if (!target)
95
+ throw new Error("Provide a URL or domain to trace, e.g. example.com or example.com/docs/.");
96
+ const matchType = (args.match_type ?? "prefix").trim().toLowerCase();
97
+ if (!["exact", "prefix", "host", "domain"].includes(matchType)) {
98
+ return `Unknown match_type "${args.match_type}". Use exact, prefix, host or domain.`;
99
+ }
100
+ const index = await pickIndex(args.index);
101
+ const params = { url: target, matchType };
102
+ const from = Number(args.from_year);
103
+ const to = Number(args.to_year);
104
+ if (Number.isFinite(from) && from > 1900)
105
+ params.from = String(Math.round(from));
106
+ if (Number.isFinite(to) && to > 1900)
107
+ params.to = String(Math.round(to));
108
+ const limit = cap(args.limit, 20, 200);
109
+ const { rows } = await cdx(index, params, limit);
110
+ if (!rows.length) {
111
+ return `Common Crawl ${index} holds no ${matchType} capture of ${target}${params.from ? ` from ${params.from}` : ""}${params.to ? ` to ${params.to}` : ""}. The URL may not have been crawled in that crawl.`;
112
+ }
113
+ const times = rows.map((r) => Number(r.timestamp) || 0).filter((t) => t > 0);
114
+ const statuses = new Map();
115
+ const mimes = new Map();
116
+ for (const r of rows) {
117
+ const st = String(r.status ?? "?");
118
+ statuses.set(st, (statuses.get(st) ?? 0) + 1);
119
+ const m = String(r["mime-detected"] ?? r.mime ?? "?");
120
+ mimes.set(m, (mimes.get(m) ?? 0) + 1);
121
+ }
122
+ const distinct = new Set(rows.map((r) => r.url)).size;
123
+ return [
124
+ `Common Crawl ${index} history for ${target} (matchType ${matchType}): ${rows.length} capture record(s) covering ${distinct} distinct URL(s).`,
125
+ `Capture window ${times.length ? `${stamp(Math.min(...times))} to ${stamp(Math.max(...times))}` : "not derivable from the returned records"}`,
126
+ `Status codes: ${[...statuses.entries()].map(([k, v]) => `${k} ${v}`).join(", ")}`,
127
+ `Detected MIME types: ${[...mimes.entries()].sort((a, b) => b[1] - a[1]).map(([k, v]) => `${k} ${v}`).join(", ")}`,
128
+ "",
129
+ ...rows.slice(0, cap(args.limit, 20, 200)).map((r, i) => `${i + 1}. ${stamp(r.timestamp)} | ${r.status ?? "?"} ${r.mime ?? ""} | ${String(r.url).slice(0, 100)}` +
130
+ `${r.digest ? ` | digest ${String(r.digest).slice(0, 16)}` : ""}${r.length ? ` | ${r.length} bytes` : ""}`),
131
+ "The digest is a base32 SHA-1 of the captured payload, so you can tell whether two crawls saw identical content.",
132
+ ].join("\n");
133
+ }
134
+ export async function indexStats(args) {
135
+ const target = (args.url ?? "").trim();
136
+ if (!target)
137
+ throw new Error("Provide a URL or domain to profile, e.g. example.com.");
138
+ const matchType = (args.match_type ?? "prefix").trim().toLowerCase();
139
+ if (!["exact", "prefix", "host", "domain"].includes(matchType)) {
140
+ return `Unknown match_type "${args.match_type}". Use exact, prefix, host or domain.`;
141
+ }
142
+ const index = await pickIndex(args.index);
143
+ const sample = cap(args.sample, 500, 5000);
144
+ const { rows } = await cdx(index, { url: target, matchType }, sample);
145
+ if (!rows.length)
146
+ return `Common Crawl ${index} returned no ${matchType} records for ${target}.`;
147
+ const statuses = new Map();
148
+ const mimes = new Map();
149
+ const depths = new Map();
150
+ const schemes = new Map();
151
+ let bytes = 0;
152
+ let counted = 0;
153
+ for (const r of rows) {
154
+ statuses.set(String(r.status ?? "?"), (statuses.get(String(r.status ?? "?")) ?? 0) + 1);
155
+ mimes.set(String(r["mime-detected"] ?? r.mime ?? "?"), (mimes.get(String(r["mime-detected"] ?? r.mime ?? "?")) ?? 0) + 1);
156
+ schemes.set(String(r.url ?? "").split(":")[0] || "?", (schemes.get(String(r.url ?? "").split(":")[0] || "?") ?? 0) + 1);
157
+ try {
158
+ const u = new URL(String(r.url));
159
+ const depth = u.pathname.split("/").filter(Boolean).length;
160
+ depths.set(String(depth), (depths.get(String(depth)) ?? 0) + 1);
161
+ }
162
+ catch {
163
+ /* ignore unparsable urls */
164
+ }
165
+ const len = Number(r.length);
166
+ if (Number.isFinite(len) && len > 0) {
167
+ bytes += len;
168
+ counted += 1;
169
+ }
170
+ }
171
+ const ok = rows.filter((r) => Number(r.status) >= 200 && Number(r.status) < 300).length;
172
+ return [
173
+ `Common Crawl ${index} profile of ${target} (matchType ${matchType}), ${rows.length} record(s) sampled.`,
174
+ `Status codes: ${[...statuses.entries()].sort((a, b) => b[1] - a[1]).map(([k, v]) => `${k} ${v}`).join(", ")} | 2xx share ${((ok / rows.length) * 100).toFixed(1)}%`,
175
+ `MIME types: ${[...mimes.entries()].sort((a, b) => b[1] - a[1]).slice(0, 8).map(([k, v]) => `${k} ${v}`).join(", ")}`,
176
+ `Schemes: ${[...schemes.entries()].map(([k, v]) => `${k} ${v}`).join(", ")}`,
177
+ `Path depth distribution: ${[...depths.entries()].sort((a, b) => Number(a[0]) - Number(b[0])).map(([k, v]) => `${k}:${v}`).join(" ")}`,
178
+ counted ? `Payload bytes seen across ${counted.toLocaleString("en-US")} record(s): ${(bytes / 1024 / 1024).toFixed(1)} MB, mean ${Math.round(bytes / counted).toLocaleString("en-US")} B` : "No record lengths reported in this sample",
179
+ rows.length >= sample ? `Sample was capped at ${sample} records, so these are proportions from that slice.` : "This is the complete record set the index returned for the query.",
180
+ ].join("\n");
181
+ }
182
+ function crawlWindow(hit) {
183
+ const raw = hit;
184
+ if (raw.from && raw.to)
185
+ return `Crawl window: ${String(raw.from)} to ${String(raw.to)}`;
186
+ return "Crawl window not reported";
187
+ }
188
+ function indexSize(hit) {
189
+ const raw = hit;
190
+ const value = raw["total-index-size"];
191
+ return value != null ? `${String(value)} bytes` : "not reported";
192
+ }
193
+ export async function collectionDetails(args) {
194
+ const id = (args.id ?? "").trim();
195
+ if (!id)
196
+ throw new Error("Provide a collection id, e.g. CC-MAIN-2026-39.");
197
+ const d = (await listCollections());
198
+ const hit = d.find((c) => c.id === id);
199
+ if (!hit) {
200
+ return `"${id}" is not in the Common Crawl collection list. The newest collections are ${d.slice(0, 5).map((c) => c.id).join(", ")}.`;
201
+ }
202
+ return [
203
+ `Common Crawl collection ${String(hit.id)}: ${String(hit.name ?? "unnamed")}`,
204
+ `Timegate: ${hit.timegate ?? "not reported"}`,
205
+ `CDX API: ${hit["cdx-api"] ?? "not reported"}`,
206
+ crawlWindow(hit),
207
+ `Total index size: ${indexSize(hit)}`,
208
+ `This collection is index number ${d.indexOf(hit) + 1} of ${d.length} listed.`,
209
+ "The index only holds metadata and offsets; the captured bytes live in WARC files on S3 and are fetched by range request using the filename and offset in each record.",
210
+ ].join("\n");
211
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Returns null when the caller may run the tool, or a short message telling
3
+ * them how to get a key.
4
+ *
5
+ * A good key is remembered for an hour, so a busy session does not hit the
6
+ * license server on every call and keeps working through a brief outage.
7
+ */
8
+ export declare function premiumRequired(tool: string): Promise<string | null>;
@@ -0,0 +1,79 @@
1
+ // Premium tools on this server need a license key bought from MCP Marketplace.
2
+ // Buyers put the key in their MCP client config as MCP_LICENSE_KEY. Every tool
3
+ // that is not listed in PREMIUM keeps working without a key.
4
+ const SLUG = "commoncrawl-mcp";
5
+ // Tools that need a key. Everything else is free.
6
+ const PREMIUM = new Set([
7
+ "capture_history",
8
+ "index_stats",
9
+ ]);
10
+ const BUY_URL = `https://mcp-marketplace.io/server/io-github-mrfentmen-${SLUG}`;
11
+ // The license check calls the marketplace's verify endpoint directly instead of
12
+ // using @mcp_marketplace/license. That SDK sends no `apikey` header, so every
13
+ // check it makes is rejected by Supabase before it reaches the function. The
14
+ // publishable key below is public by design - it ships in mcp-marketplace.io's
15
+ // own JavaScript and only ever reaches their licence check.
16
+ const DEFAULT_VERIFY_URL = "https://virupvwhtkpkjsiskckg.supabase.co/functions/v1/verify-key";
17
+ const PUBLISHABLE_KEY = "sb_publishable_BuqlW96Ke8C_zzJG-LQv1Q_YHi_r_4h";
18
+ const OK_CACHE_MS = 60 * 60 * 1000;
19
+ const BAD_CACHE_MS = 60 * 1000;
20
+ const seen = new Map();
21
+ const REASONS = {
22
+ missing_key: "no key is set",
23
+ invalid_format: "that key is not a valid MCP Marketplace key",
24
+ not_found: "that key was not found",
25
+ revoked: "that key has been revoked",
26
+ rotated: "that key was rotated, use your new one",
27
+ expired: "that key has expired, renew it",
28
+ rate_limited: "there were too many checks just now, try again shortly",
29
+ network_error: "the license server could not be reached",
30
+ };
31
+ function blocked(tool, reason) {
32
+ const why = REASONS[reason] ?? `the license server answered "${reason}"`;
33
+ return `"${tool}" needs a license key, but ${why}. ` +
34
+ `Set MCP_LICENSE_KEY in your MCP client config. Get a key: ${BUY_URL}`;
35
+ }
36
+ async function verify(key) {
37
+ const res = await fetch(process.env.MCP_LICENSE_VERIFY_URL || DEFAULT_VERIFY_URL, {
38
+ method: "POST",
39
+ headers: {
40
+ apikey: PUBLISHABLE_KEY,
41
+ Authorization: `Bearer ${PUBLISHABLE_KEY}`,
42
+ "Content-Type": "application/json",
43
+ },
44
+ body: JSON.stringify({ key, slug: SLUG }),
45
+ // an unusable key answers 400 with a JSON body, so read the body either way
46
+ signal: AbortSignal.timeout(15000),
47
+ });
48
+ const data = (await res.json());
49
+ if (typeof data?.valid !== "boolean")
50
+ return { valid: false, reason: "unexpected_response" };
51
+ return data;
52
+ }
53
+ /**
54
+ * Returns null when the caller may run the tool, or a short message telling
55
+ * them how to get a key.
56
+ *
57
+ * A good key is remembered for an hour, so a busy session does not hit the
58
+ * license server on every call and keeps working through a brief outage.
59
+ */
60
+ export async function premiumRequired(tool) {
61
+ if (!PREMIUM.has(tool))
62
+ return null;
63
+ const key = process.env.MCP_LICENSE_KEY;
64
+ if (!key)
65
+ return blocked(tool, "missing_key");
66
+ const hit = seen.get(key);
67
+ if (hit && Date.now() - hit.at < (hit.valid ? OK_CACHE_MS : BAD_CACHE_MS)) {
68
+ return hit.valid ? null : blocked(tool, hit.reason ?? "invalid");
69
+ }
70
+ let result;
71
+ try {
72
+ result = await verify(key);
73
+ }
74
+ catch {
75
+ result = { valid: false, reason: "network_error" };
76
+ }
77
+ seen.set(key, { ...result, at: Date.now() });
78
+ return result.valid ? null : blocked(tool, result.reason ?? "invalid");
79
+ }
package/dist/server.js CHANGED
@@ -1,6 +1,7 @@
1
1
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
2
  import { z } from "zod";
3
- import { format, latestIndex, listCollections, searchCaptures } from "./api.js";
3
+ import { format, latestIndex, listCollections, searchCaptures, captureHistory, indexStats, collectionDetails } from "./api.js";
4
+ import { premiumRequired } from "./license.js";
4
5
  const text = (value) => ({ content: [{ type: "text", text: value }] });
5
6
  const textError = (t) => ({ content: [{ type: "text", text: t }], isError: true });
6
7
  const READ_ONLY = { readOnlyHint: true, openWorldHint: true };
@@ -51,5 +52,54 @@ export function createServer() {
51
52
  return errorText(error);
52
53
  }
53
54
  });
55
+ server.registerTool("capture_history", {
56
+ title: "Capture history",
57
+ description: "Trace every Common Crawl capture of a URL or domain in a given index with capture timestamps, status codes, MIME types, payload digests and lengths, restricted to an optional year range.",
58
+ inputSchema: z.object({ url: z.string().describe("URL or domain, e.g. example.com."), index: z.string().describe("Collection id, defaults to the latest.").optional(), match_type: z.string().describe("exact, prefix, host or domain.").optional(), from_year: z.number().describe("Only captures from this year onwards.").optional(), to_year: z.number().describe("Only captures up to this year.").optional(), limit: z.number().describe("Max records.").optional() }),
59
+ annotations: READ_ONLY,
60
+ }, async (args) => {
61
+ // ---- paywall: capture_history ----
62
+ const paywallMessage = await premiumRequired("capture_history");
63
+ if (paywallMessage)
64
+ return { content: [{ type: "text", text: paywallMessage }], isError: true };
65
+ // ---- paywall: end ----
66
+ try {
67
+ return text(await captureHistory(args));
68
+ }
69
+ catch (e) {
70
+ return textError(e instanceof Error ? `Error: ${e.message}` : String(e));
71
+ }
72
+ });
73
+ server.registerTool("index_stats", {
74
+ title: "Index stats",
75
+ description: "Profile a site as Common Crawl sees it: status code and MIME type distributions, 2xx share, scheme mix, URL path depth histogram and mean payload size over a sample of index records.",
76
+ inputSchema: z.object({ url: z.string().describe("URL or domain to profile."), index: z.string().describe("Collection id, defaults to the latest.").optional(), match_type: z.string().describe("exact, prefix, host or domain.").optional(), sample: z.number().describe("Max records to sample.").optional() }),
77
+ annotations: READ_ONLY,
78
+ }, async (args) => {
79
+ // ---- paywall: index_stats ----
80
+ const paywallMessage = await premiumRequired("index_stats");
81
+ if (paywallMessage)
82
+ return { content: [{ type: "text", text: paywallMessage }], isError: true };
83
+ // ---- paywall: end ----
84
+ try {
85
+ return text(await indexStats(args));
86
+ }
87
+ catch (e) {
88
+ return textError(e instanceof Error ? `Error: ${e.message}` : String(e));
89
+ }
90
+ });
91
+ server.registerTool("collection_details", {
92
+ title: "Collection details",
93
+ description: "Look up a specific Common Crawl collection by id with its timegate, CDX API endpoint, crawl window and total index size.",
94
+ inputSchema: z.object({ id: z.string().describe("Collection id, e.g. CC-MAIN-2026-39.") }),
95
+ annotations: READ_ONLY,
96
+ }, async (args) => {
97
+ try {
98
+ return text(await collectionDetails(args));
99
+ }
100
+ catch (e) {
101
+ return textError(e instanceof Error ? `Error: ${e.message}` : String(e));
102
+ }
103
+ });
54
104
  return server;
55
105
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "commoncrawl-mcp",
3
- "version": "1.0.1",
3
+ "version": "1.0.2",
4
4
  "description": "Use this MCP server to Common Crawl index discovery for historical web captures. Tools include list indexes, latest index, search captures",
5
5
  "type": "module",
6
6
  "mcpName": "io.github.mrfentmen/commoncrawl-mcp",