commoncrawl-mcp 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -0
- package/dist/api.d.ts +18 -0
- package/dist/api.js +44 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +12 -0
- package/dist/server.d.ts +2 -0
- package/dist/server.js +55 -0
- package/package.json +42 -0
- package/server.json +20 -0
package/README.md
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# commoncrawl-mcp
|
|
2
|
+
|
|
3
|
+
Discover historical web captures through the Common Crawl index. This server is for provenance research, archival discovery, broken-link investigations, and finding the capture metadata needed for a later archive fetch.
|
|
4
|
+
|
|
5
|
+
## Tools
|
|
6
|
+
|
|
7
|
+
- `list_indexes`: list recent crawl collections.
|
|
8
|
+
- `latest_index`: return the newest collection id.
|
|
9
|
+
- `search_captures`: search URL or wildcard patterns and return capture index metadata such as timestamp, status, digest, filename, offset, and length.
|
|
10
|
+
|
|
11
|
+
This server searches the index only. It does not download page contents, bypass access controls, or guarantee that a capture is complete or legally reusable. Requests and result pages are bounded.
|
|
12
|
+
|
|
13
|
+
## Run
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
npm install
|
|
17
|
+
npm run build
|
|
18
|
+
node dist/index.js
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
## Quick start
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
npm install
|
|
25
|
+
npm run build
|
|
26
|
+
node dist/index.js
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
The server uses stdio, so it can be connected to Claude Desktop, Cursor, VS Code, MCP Inspector, or another compatible MCP client.
|
|
30
|
+
|
|
31
|
+
## Tools at a glance
|
|
32
|
+
|
|
33
|
+
- `list_indexes`: List available Common Crawl web crawl index collections, newest first.
|
|
34
|
+
- `latest_index`: Return the identifier of the newest Common Crawl index collection.
|
|
35
|
+
- `search_captures`: Find historical Common Crawl captures matching a URL or wildcard pattern. This returns index metadata, not page contents.
|
|
36
|
+
|
|
37
|
+
## Limits and privacy
|
|
38
|
+
|
|
39
|
+
This project is intentionally narrow. It should be treated as a practical helper, not a complete certification or security audit. Check the implementation and the returned data before using it with sensitive material. No credentials are required unless the project explicitly says otherwise.
|
|
40
|
+
|
|
41
|
+
## Try it
|
|
42
|
+
|
|
43
|
+
After building, connect the server through your MCP client. The repository root also contains `smoke-test.mjs` for projects covered by the shared harness. A typical tool call starts with `list_indexes`.
|
package/dist/api.d.ts
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
export declare class CommonCrawlError extends Error {
|
|
2
|
+
}
|
|
3
|
+
type Collection = {
|
|
4
|
+
id?: string;
|
|
5
|
+
name?: string;
|
|
6
|
+
timegate?: string;
|
|
7
|
+
cd?: string;
|
|
8
|
+
};
|
|
9
|
+
export declare function listCollections(): Promise<Collection[]>;
|
|
10
|
+
export declare function latestIndex(): Promise<string>;
|
|
11
|
+
export declare function searchCaptures(urlPattern: string, index: string | undefined, page: number, limit: number): Promise<{
|
|
12
|
+
index: string;
|
|
13
|
+
page: number;
|
|
14
|
+
count: number;
|
|
15
|
+
records: any[];
|
|
16
|
+
}>;
|
|
17
|
+
export declare function format(value: unknown): string;
|
|
18
|
+
export {};
|
package/dist/api.js
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
const COLLECTIONS = "https://index.commoncrawl.org/collinfo.json";
|
|
2
|
+
const HEADERS = { "User-Agent": "mrfentmen-commoncrawl-mcp/1.0" };
|
|
3
|
+
export class CommonCrawlError extends Error {
|
|
4
|
+
}
|
|
5
|
+
async function getJson(url, timeout = 30000) {
|
|
6
|
+
const response = await fetch(url, { headers: HEADERS, signal: AbortSignal.timeout(timeout) });
|
|
7
|
+
if (!response.ok)
|
|
8
|
+
throw new CommonCrawlError(`Common Crawl error ${response.status}`);
|
|
9
|
+
return response.json();
|
|
10
|
+
}
|
|
11
|
+
export async function listCollections() {
|
|
12
|
+
return (await getJson(new URL(COLLECTIONS)));
|
|
13
|
+
}
|
|
14
|
+
export async function latestIndex() {
|
|
15
|
+
const collections = await listCollections();
|
|
16
|
+
const id = collections[0]?.id;
|
|
17
|
+
if (!id)
|
|
18
|
+
throw new CommonCrawlError("Common Crawl returned no index collections");
|
|
19
|
+
return id;
|
|
20
|
+
}
|
|
21
|
+
export async function searchCaptures(urlPattern, index, page, limit) {
|
|
22
|
+
const selectedIndex = index || await latestIndex();
|
|
23
|
+
const url = new URL(`https://index.commoncrawl.org/${encodeURIComponent(selectedIndex)}-index`);
|
|
24
|
+
url.searchParams.set("url", urlPattern);
|
|
25
|
+
url.searchParams.set("output", "json");
|
|
26
|
+
url.searchParams.set("page", String(page));
|
|
27
|
+
url.searchParams.set("fl", "url,timestamp,status,mime,mime-detected,digest,filename,offset,length");
|
|
28
|
+
const response = await fetch(url, { headers: HEADERS, signal: AbortSignal.timeout(30000) });
|
|
29
|
+
if (!response.ok)
|
|
30
|
+
throw new CommonCrawlError(`Common Crawl index error ${response.status}`);
|
|
31
|
+
const body = await response.text();
|
|
32
|
+
const records = body.split("\\n").filter(Boolean).slice(0, limit).map((line) => {
|
|
33
|
+
try {
|
|
34
|
+
return JSON.parse(line);
|
|
35
|
+
}
|
|
36
|
+
catch {
|
|
37
|
+
return { raw: line.slice(0, 1000) };
|
|
38
|
+
}
|
|
39
|
+
});
|
|
40
|
+
return { index: selectedIndex, page, count: records.length, records };
|
|
41
|
+
}
|
|
42
|
+
export function format(value) {
|
|
43
|
+
return JSON.stringify(value, null, 2).slice(0, 16000);
|
|
44
|
+
}
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
2
|
+
import { createServer } from "./server.js";
|
|
3
|
+
async function main() {
|
|
4
|
+
const server = createServer();
|
|
5
|
+
const transport = new StdioServerTransport();
|
|
6
|
+
await server.connect(transport);
|
|
7
|
+
console.error("MCP server running on stdio");
|
|
8
|
+
}
|
|
9
|
+
main().catch((err) => {
|
|
10
|
+
console.error("Fatal error:", err);
|
|
11
|
+
process.exit(1);
|
|
12
|
+
});
|
package/dist/server.d.ts
ADDED
package/dist/server.js
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
2
|
+
import { z } from "zod";
|
|
3
|
+
import { format, latestIndex, listCollections, searchCaptures } from "./api.js";
|
|
4
|
+
const text = (value) => ({ content: [{ type: "text", text: value }] });
|
|
5
|
+
const textError = (t) => ({ content: [{ type: "text", text: t }], isError: true });
|
|
6
|
+
const READ_ONLY = { readOnlyHint: true, openWorldHint: true };
|
|
7
|
+
const errorText = (error) => text(`Error: ${error instanceof Error ? error.message : String(error)}`);
|
|
8
|
+
export function createServer() {
|
|
9
|
+
const server = new McpServer({ name: "commoncrawl-mcp", version: "1.0.0" });
|
|
10
|
+
server.registerTool("list_indexes", {
|
|
11
|
+
title: "List indexes",
|
|
12
|
+
description: "List available Common Crawl web crawl index collections, newest first.",
|
|
13
|
+
inputSchema: z.object({}),
|
|
14
|
+
annotations: READ_ONLY,
|
|
15
|
+
}, async () => {
|
|
16
|
+
try {
|
|
17
|
+
return text(format((await listCollections()).slice(0, 30)));
|
|
18
|
+
}
|
|
19
|
+
catch (error) {
|
|
20
|
+
return errorText(error);
|
|
21
|
+
}
|
|
22
|
+
});
|
|
23
|
+
server.registerTool("latest_index", {
|
|
24
|
+
title: "Latest index",
|
|
25
|
+
description: "Return the identifier of the newest Common Crawl index collection.",
|
|
26
|
+
inputSchema: z.object({}),
|
|
27
|
+
annotations: READ_ONLY,
|
|
28
|
+
}, async () => {
|
|
29
|
+
try {
|
|
30
|
+
return text(await latestIndex());
|
|
31
|
+
}
|
|
32
|
+
catch (error) {
|
|
33
|
+
return errorText(error);
|
|
34
|
+
}
|
|
35
|
+
});
|
|
36
|
+
server.registerTool("search_captures", {
|
|
37
|
+
title: "Search captures",
|
|
38
|
+
description: "Find historical Common Crawl captures matching a URL or wildcard pattern. This returns index metadata, not page contents.",
|
|
39
|
+
inputSchema: z.object({
|
|
40
|
+
url: z.string().min(1).max(500).describe("URL or wildcard pattern, for example example.com/*"),
|
|
41
|
+
index: z.string().regex(/^[A-Za-z0-9._-]+$/).optional().describe("Optional collection id such as CC-MAIN-2025-30"),
|
|
42
|
+
page: z.number().int().min(1).max(20).default(1),
|
|
43
|
+
limit: z.number().int().min(1).max(100).default(20),
|
|
44
|
+
}),
|
|
45
|
+
annotations: READ_ONLY,
|
|
46
|
+
}, async ({ url, index, page, limit }) => {
|
|
47
|
+
try {
|
|
48
|
+
return text(format(await searchCaptures(url, index, page, limit)));
|
|
49
|
+
}
|
|
50
|
+
catch (error) {
|
|
51
|
+
return errorText(error);
|
|
52
|
+
}
|
|
53
|
+
});
|
|
54
|
+
return server;
|
|
55
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "commoncrawl-mcp",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "Use this MCP server to Common Crawl index discovery for historical web captures. Tools include list indexes, latest index, search captures",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"mcpName": "io.github.mrfentmen/commoncrawl-mcp",
|
|
7
|
+
"repository": {
|
|
8
|
+
"type": "git",
|
|
9
|
+
"url": "https://github.com/mrfentmen/commoncrawl-mcp.git"
|
|
10
|
+
},
|
|
11
|
+
"bin": {
|
|
12
|
+
"commoncrawl-mcp": "./dist/index.js"
|
|
13
|
+
},
|
|
14
|
+
"main": "./dist/index.js",
|
|
15
|
+
"files": [
|
|
16
|
+
"dist",
|
|
17
|
+
"server.json",
|
|
18
|
+
"README.md"
|
|
19
|
+
],
|
|
20
|
+
"scripts": {
|
|
21
|
+
"build": "tsc -p tsconfig.json",
|
|
22
|
+
"start": "node dist/index.js",
|
|
23
|
+
"dev": "npm run build && node dist/index.js"
|
|
24
|
+
},
|
|
25
|
+
"keywords": [
|
|
26
|
+
"mcp",
|
|
27
|
+
"public-data",
|
|
28
|
+
"commoncrawl"
|
|
29
|
+
],
|
|
30
|
+
"license": "MIT",
|
|
31
|
+
"dependencies": {
|
|
32
|
+
"@modelcontextprotocol/sdk": "^1.0.4",
|
|
33
|
+
"zod": "^3.23.8"
|
|
34
|
+
},
|
|
35
|
+
"devDependencies": {
|
|
36
|
+
"@types/node": "^22.0.0",
|
|
37
|
+
"typescript": "^5.6.0"
|
|
38
|
+
},
|
|
39
|
+
"engines": {
|
|
40
|
+
"node": ">=20"
|
|
41
|
+
}
|
|
42
|
+
}
|
package/server.json
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
|
|
3
|
+
"name": "io.github.mrfentmen/commoncrawl-mcp",
|
|
4
|
+
"description": "Use this MCP server to Common Crawl index discovery for historical web captures. Tools include...",
|
|
5
|
+
"repository": {
|
|
6
|
+
"url": "https://github.com/mrfentmen/commoncrawl-mcp",
|
|
7
|
+
"source": "github"
|
|
8
|
+
},
|
|
9
|
+
"version": "1.0.0",
|
|
10
|
+
"packages": [
|
|
11
|
+
{
|
|
12
|
+
"registryType": "npm",
|
|
13
|
+
"identifier": "commoncrawl-mcp",
|
|
14
|
+
"version": "1.0.0",
|
|
15
|
+
"transport": {
|
|
16
|
+
"type": "stdio"
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
]
|
|
20
|
+
}
|