commoncrawl-mcp 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,43 @@
1
+ # commoncrawl-mcp
2
+
3
+ Discover historical web captures through the Common Crawl index. This server is for provenance research, archival discovery, broken-link investigations, and finding the capture metadata needed for a later archive fetch.
4
+
5
+ ## Tools
6
+
7
+ - `list_indexes`: list recent crawl collections.
8
+ - `latest_index`: return the newest collection id.
9
+ - `search_captures`: search URL or wildcard patterns and return capture index metadata such as timestamp, status, digest, filename, offset, and length.
10
+
11
+ This server searches the index only. It does not download page contents, bypass access controls, or guarantee that a capture is complete or legally reusable. Requests and result pages are bounded.
12
+
13
+ ## Run
14
+
15
+ ```bash
16
+ npm install
17
+ npm run build
18
+ node dist/index.js
19
+ ```
20
+
21
+ ## Quick start
22
+
23
+ ```bash
24
+ npm install
25
+ npm run build
26
+ node dist/index.js
27
+ ```
28
+
29
+ The server uses stdio, so it can be connected to Claude Desktop, Cursor, VS Code, MCP Inspector, or another compatible MCP client.
30
+
31
+ ## Tools at a glance
32
+
33
+ - `list_indexes`: List available Common Crawl web crawl index collections, newest first.
34
+ - `latest_index`: Return the identifier of the newest Common Crawl index collection.
35
+ - `search_captures`: Find historical Common Crawl captures matching a URL or wildcard pattern. This returns index metadata, not page contents.
36
+
37
+ ## Limits and privacy
38
+
39
+ This project is intentionally narrow. It should be treated as a practical helper, not a complete certification or security audit. Check the implementation and the returned data before using it with sensitive material. No credentials are required unless the project explicitly says otherwise.
40
+
41
+ ## Try it
42
+
43
+ After building, connect the server through your MCP client. The repository root also contains `smoke-test.mjs` for projects covered by the shared harness. A typical tool call starts with `list_indexes`.
package/dist/api.d.ts ADDED
@@ -0,0 +1,18 @@
1
+ export declare class CommonCrawlError extends Error {
2
+ }
3
+ type Collection = {
4
+ id?: string;
5
+ name?: string;
6
+ timegate?: string;
7
+ cd?: string;
8
+ };
9
+ export declare function listCollections(): Promise<Collection[]>;
10
+ export declare function latestIndex(): Promise<string>;
11
+ export declare function searchCaptures(urlPattern: string, index: string | undefined, page: number, limit: number): Promise<{
12
+ index: string;
13
+ page: number;
14
+ count: number;
15
+ records: any[];
16
+ }>;
17
+ export declare function format(value: unknown): string;
18
+ export {};
package/dist/api.js ADDED
@@ -0,0 +1,44 @@
1
+ const COLLECTIONS = "https://index.commoncrawl.org/collinfo.json";
2
+ const HEADERS = { "User-Agent": "mrfentmen-commoncrawl-mcp/1.0" };
3
+ export class CommonCrawlError extends Error {
4
+ }
5
+ async function getJson(url, timeout = 30000) {
6
+ const response = await fetch(url, { headers: HEADERS, signal: AbortSignal.timeout(timeout) });
7
+ if (!response.ok)
8
+ throw new CommonCrawlError(`Common Crawl error ${response.status}`);
9
+ return response.json();
10
+ }
11
+ export async function listCollections() {
12
+ return (await getJson(new URL(COLLECTIONS)));
13
+ }
14
+ export async function latestIndex() {
15
+ const collections = await listCollections();
16
+ const id = collections[0]?.id;
17
+ if (!id)
18
+ throw new CommonCrawlError("Common Crawl returned no index collections");
19
+ return id;
20
+ }
21
+ export async function searchCaptures(urlPattern, index, page, limit) {
22
+ const selectedIndex = index || await latestIndex();
23
+ const url = new URL(`https://index.commoncrawl.org/${encodeURIComponent(selectedIndex)}-index`);
24
+ url.searchParams.set("url", urlPattern);
25
+ url.searchParams.set("output", "json");
26
+ url.searchParams.set("page", String(page));
27
+ url.searchParams.set("fl", "url,timestamp,status,mime,mime-detected,digest,filename,offset,length");
28
+ const response = await fetch(url, { headers: HEADERS, signal: AbortSignal.timeout(30000) });
29
+ if (!response.ok)
30
+ throw new CommonCrawlError(`Common Crawl index error ${response.status}`);
31
+ const body = await response.text();
32
+ const records = body.split("\\n").filter(Boolean).slice(0, limit).map((line) => {
33
+ try {
34
+ return JSON.parse(line);
35
+ }
36
+ catch {
37
+ return { raw: line.slice(0, 1000) };
38
+ }
39
+ });
40
+ return { index: selectedIndex, page, count: records.length, records };
41
+ }
42
+ export function format(value) {
43
+ return JSON.stringify(value, null, 2).slice(0, 16000);
44
+ }
@@ -0,0 +1 @@
1
+ export {};
package/dist/index.js ADDED
@@ -0,0 +1,12 @@
1
+ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
2
+ import { createServer } from "./server.js";
3
+ async function main() {
4
+ const server = createServer();
5
+ const transport = new StdioServerTransport();
6
+ await server.connect(transport);
7
+ console.error("MCP server running on stdio");
8
+ }
9
+ main().catch((err) => {
10
+ console.error("Fatal error:", err);
11
+ process.exit(1);
12
+ });
@@ -0,0 +1,2 @@
1
+ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
+ export declare function createServer(): McpServer;
package/dist/server.js ADDED
@@ -0,0 +1,55 @@
1
+ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
+ import { z } from "zod";
3
+ import { format, latestIndex, listCollections, searchCaptures } from "./api.js";
4
+ const text = (value) => ({ content: [{ type: "text", text: value }] });
5
+ const textError = (t) => ({ content: [{ type: "text", text: t }], isError: true });
6
+ const READ_ONLY = { readOnlyHint: true, openWorldHint: true };
7
+ const errorText = (error) => text(`Error: ${error instanceof Error ? error.message : String(error)}`);
8
+ export function createServer() {
9
+ const server = new McpServer({ name: "commoncrawl-mcp", version: "1.0.0" });
10
+ server.registerTool("list_indexes", {
11
+ title: "List indexes",
12
+ description: "List available Common Crawl web crawl index collections, newest first.",
13
+ inputSchema: z.object({}),
14
+ annotations: READ_ONLY,
15
+ }, async () => {
16
+ try {
17
+ return text(format((await listCollections()).slice(0, 30)));
18
+ }
19
+ catch (error) {
20
+ return errorText(error);
21
+ }
22
+ });
23
+ server.registerTool("latest_index", {
24
+ title: "Latest index",
25
+ description: "Return the identifier of the newest Common Crawl index collection.",
26
+ inputSchema: z.object({}),
27
+ annotations: READ_ONLY,
28
+ }, async () => {
29
+ try {
30
+ return text(await latestIndex());
31
+ }
32
+ catch (error) {
33
+ return errorText(error);
34
+ }
35
+ });
36
+ server.registerTool("search_captures", {
37
+ title: "Search captures",
38
+ description: "Find historical Common Crawl captures matching a URL or wildcard pattern. This returns index metadata, not page contents.",
39
+ inputSchema: z.object({
40
+ url: z.string().min(1).max(500).describe("URL or wildcard pattern, for example example.com/*"),
41
+ index: z.string().regex(/^[A-Za-z0-9._-]+$/).optional().describe("Optional collection id such as CC-MAIN-2025-30"),
42
+ page: z.number().int().min(1).max(20).default(1),
43
+ limit: z.number().int().min(1).max(100).default(20),
44
+ }),
45
+ annotations: READ_ONLY,
46
+ }, async ({ url, index, page, limit }) => {
47
+ try {
48
+ return text(format(await searchCaptures(url, index, page, limit)));
49
+ }
50
+ catch (error) {
51
+ return errorText(error);
52
+ }
53
+ });
54
+ return server;
55
+ }
package/package.json ADDED
@@ -0,0 +1,42 @@
1
+ {
2
+ "name": "commoncrawl-mcp",
3
+ "version": "1.0.0",
4
+ "description": "Use this MCP server to Common Crawl index discovery for historical web captures. Tools include list indexes, latest index, search captures",
5
+ "type": "module",
6
+ "mcpName": "io.github.mrfentmen/commoncrawl-mcp",
7
+ "repository": {
8
+ "type": "git",
9
+ "url": "https://github.com/mrfentmen/commoncrawl-mcp.git"
10
+ },
11
+ "bin": {
12
+ "commoncrawl-mcp": "./dist/index.js"
13
+ },
14
+ "main": "./dist/index.js",
15
+ "files": [
16
+ "dist",
17
+ "server.json",
18
+ "README.md"
19
+ ],
20
+ "scripts": {
21
+ "build": "tsc -p tsconfig.json",
22
+ "start": "node dist/index.js",
23
+ "dev": "npm run build && node dist/index.js"
24
+ },
25
+ "keywords": [
26
+ "mcp",
27
+ "public-data",
28
+ "commoncrawl"
29
+ ],
30
+ "license": "MIT",
31
+ "dependencies": {
32
+ "@modelcontextprotocol/sdk": "^1.0.4",
33
+ "zod": "^3.23.8"
34
+ },
35
+ "devDependencies": {
36
+ "@types/node": "^22.0.0",
37
+ "typescript": "^5.6.0"
38
+ },
39
+ "engines": {
40
+ "node": ">=20"
41
+ }
42
+ }
package/server.json ADDED
@@ -0,0 +1,20 @@
1
+ {
2
+ "$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
3
+ "name": "io.github.mrfentmen/commoncrawl-mcp",
4
+ "description": "Use this MCP server to Common Crawl index discovery for historical web captures. Tools include...",
5
+ "repository": {
6
+ "url": "https://github.com/mrfentmen/commoncrawl-mcp",
7
+ "source": "github"
8
+ },
9
+ "version": "1.0.0",
10
+ "packages": [
11
+ {
12
+ "registryType": "npm",
13
+ "identifier": "commoncrawl-mcp",
14
+ "version": "1.0.0",
15
+ "transport": {
16
+ "type": "stdio"
17
+ }
18
+ }
19
+ ]
20
+ }