@pipeworx/mcp-commoncrawl 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +159 -0
- package/bin/cli.js +17 -0
- package/package.json +32 -0
- package/server.json +18 -0
- package/src/index.ts +1036 -0
- package/src/server.ts +45 -0
- package/tsconfig.json +18 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mojibake Inc.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
# @pipeworx/commoncrawl
|
|
2
|
+
|
|
3
|
+
Common Crawl's public web archive — find every time a URL was crawled since 2008
|
|
4
|
+
and read back the exact page bytes that were captured.
|
|
5
|
+
|
|
6
|
+
Part of [Pipeworx](https://pipeworx.io) — an MCP gateway connecting AI agents to 1679+ live data sources.
|
|
7
|
+
|
|
8
|
+
## Tools
|
|
9
|
+
|
|
10
|
+
- `commoncrawl_crawls(limit?, contains_date?)` — the monthly crawl collections
|
|
11
|
+
(`CC-MAIN-2026-34` and friends) with the capture window each one covers.
|
|
12
|
+
`contains_date` answers "which snapshot would have seen my page on that day".
|
|
13
|
+
- `commoncrawl_index_search(url, crawl?, match_type?, limit?, page?, filter?, from?, to?)` —
|
|
14
|
+
CDX index lookup for a URL, host or whole domain. Returns capture timestamp,
|
|
15
|
+
HTTP status, MIME type, detected language, content digest, and the WARC
|
|
16
|
+
`filename` + `offset` + `length` that the next tool needs.
|
|
17
|
+
- `commoncrawl_fetch_record(filename, offset, length, max_body_bytes?)` — reads
|
|
18
|
+
one archived record back by byte range: WARC headers, the captured HTTP
|
|
19
|
+
response headers, and the page body as crawled.
|
|
20
|
+
|
|
21
|
+
## Auth
|
|
22
|
+
|
|
23
|
+
Keyless.
|
|
24
|
+
|
|
25
|
+
## Data sources
|
|
26
|
+
|
|
27
|
+
- <https://index.commoncrawl.org/collinfo.json> — the crawl collection list.
|
|
28
|
+
- <https://index.commoncrawl.org/{crawl}-index> — the CDX index, JSON-lines.
|
|
29
|
+
- <https://data.commoncrawl.org/{warc path}> — WARC records, HTTP range reads.
|
|
30
|
+
|
|
31
|
+
## Things the next person would otherwise rediscover
|
|
32
|
+
|
|
33
|
+
- **The CDX host 502s sporadically under load.** Measured 2026-09-17: the same
|
|
34
|
+
`url=example.com` query alternated between 200 and nginx 502 inside a minute,
|
|
35
|
+
so it is load, not the query. `ccFetch` retries a 5xx twice with a short
|
|
36
|
+
backoff. Without that a caller reads transient nginx noise as "Common Crawl
|
|
37
|
+
has no record of this URL", which is a different and wrong answer.
|
|
38
|
+
- **"No captures" is a 404 with an English sentence, not JSON.** Handled as an
|
|
39
|
+
empty result set with an explicit `note`, so an empty `captures` array is
|
|
40
|
+
never silently indistinguishable from an upstream failure.
|
|
41
|
+
- **Each indexed record is its own gzip member**, so a byte-range read of
|
|
42
|
+
`offset`..`offset+length-1` decompresses standalone — no need to stream the
|
|
43
|
+
whole 1 GB WARC. `DecompressionStream('gzip')` handles it in the Workers
|
|
44
|
+
runtime.
|
|
45
|
+
- **`new TextDecoder('utf-8', { fatal: false })` does not typecheck** against
|
|
46
|
+
`@cloudflare/workers-types`: its `TextDecoderConstructorOptions` requires
|
|
47
|
+
`ignoreBOM` too. Pass no options; non-fatal is the default anyway.
|
|
48
|
+
- A broad `match_type: "domain"` query over a large domain is expensive upstream
|
|
49
|
+
and is the first thing to 502. Narrow with `match_type: "host"` plus a
|
|
50
|
+
`filter` such as `=status:200`.
|
|
51
|
+
|
|
52
|
+
## Related packs
|
|
53
|
+
|
|
54
|
+
`crawlgraph` is our own link graph over sites we crawl. This pack reads the
|
|
55
|
+
Common Crawl Foundation's corpus — different data, different questions.
|
|
56
|
+
|
|
57
|
+
## Quick Start
|
|
58
|
+
|
|
59
|
+
Add to your MCP client (Claude Desktop, Cursor, Windsurf, etc.):
|
|
60
|
+
|
|
61
|
+
```json
|
|
62
|
+
{
|
|
63
|
+
"mcpServers": {
|
|
64
|
+
"commoncrawl": {
|
|
65
|
+
"url": "https://gateway.pipeworx.io/commoncrawl/mcp"
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### What this endpoint actually serves
|
|
72
|
+
|
|
73
|
+
`tools/list` at `https://gateway.pipeworx.io/commoncrawl/mcp` returns the tools in the table
|
|
74
|
+
above **plus the shared Pipeworx meta-tools** — `ask_pipeworx`,
|
|
75
|
+
`discover_tools`, `search_within`, `remember`/`recall` and the rest of the
|
|
76
|
+
gateway-wide set. So the tool count you see is larger than this table: a
|
|
77
|
+
single-pack endpoint currently lists roughly 30 shared tools alongside the
|
|
78
|
+
pack's own. The connection's `initialize` response states its exact scope, and
|
|
79
|
+
is the authoritative answer for a given day.
|
|
80
|
+
|
|
81
|
+
This is deliberate, not multiplexing by accident. The meta-tools are what let a
|
|
82
|
+
scoped connection answer a question this pack does not cover — via
|
|
83
|
+
`ask_pipeworx`, which routes across the whole catalog — without you adding a
|
|
84
|
+
second MCP server. There is currently no way to mount a pack endpoint without
|
|
85
|
+
them; if the extra schemas cost you more context than the routing is worth,
|
|
86
|
+
connect to the full gateway once rather than to several pack endpoints.
|
|
87
|
+
|
|
88
|
+
Or connect to the full Pipeworx gateway to get every pack's tools listed
|
|
89
|
+
directly, instead of just this one's:
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
{
|
|
93
|
+
"mcpServers": {
|
|
94
|
+
"pipeworx": {
|
|
95
|
+
"url": "https://gateway.pipeworx.io/mcp"
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Both URLs reach the same gateway and the same 1679+ data sources. The
|
|
102
|
+
only difference is which pack's tools are listed **directly**; `ask_pipeworx`
|
|
103
|
+
reaches all of them from either one.
|
|
104
|
+
|
|
105
|
+
## No MCP client? Call it over HTTP
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
curl -X POST https://gateway.pipeworx.io/v1/tools/commoncrawl_crawls \
|
|
109
|
+
-H 'Content-Type: application/json' \
|
|
110
|
+
-d '{"limit":3}'
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
No account needed for the first calls. Inspect any tool: `GET https://gateway.pipeworx.io/v1/tools/commoncrawl_crawls`. Find one: `POST https://gateway.pipeworx.io/v1/tools/search_packs` with `{"query":"..."}`.
|
|
114
|
+
|
|
115
|
+
## Standalone (no gateway account)
|
|
116
|
+
|
|
117
|
+
This package also runs as a local stdio MCP server — no Pipeworx account, no
|
|
118
|
+
gateway round-trip:
|
|
119
|
+
|
|
120
|
+
```json
|
|
121
|
+
{
|
|
122
|
+
"mcpServers": {
|
|
123
|
+
"commoncrawl": {
|
|
124
|
+
"command": "npx",
|
|
125
|
+
"args": ["-y", "@pipeworx/mcp-commoncrawl"]
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Or run it directly to confirm it starts:
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
npx -y @pipeworx/mcp-commoncrawl
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
It speaks MCP over stdin/stdout and answers `initialize`/`tools/list`/`tools/call`
|
|
138
|
+
for **only** this pack's tools — none of the shared meta-tools the gateway
|
|
139
|
+
connection above adds. Same source, same tools, no ask_pipeworx routing.
|
|
140
|
+
|
|
141
|
+
## Using with ask_pipeworx
|
|
142
|
+
|
|
143
|
+
Instead of calling tools directly, you can ask questions in plain English —
|
|
144
|
+
this works on the pack endpoint above as well as on the full gateway:
|
|
145
|
+
|
|
146
|
+
```
|
|
147
|
+
ask_pipeworx({ question: "your question about Commoncrawl data" })
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
The gateway picks the right tool and fills the arguments automatically.
|
|
151
|
+
|
|
152
|
+
## More
|
|
153
|
+
|
|
154
|
+
- [Docs and guides](https://pipeworx.io/docs)
|
|
155
|
+
- [pipeworx.io](https://pipeworx.io)
|
|
156
|
+
|
|
157
|
+
## License
|
|
158
|
+
|
|
159
|
+
MIT
|
package/bin/cli.js
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
//
|
|
3
|
+
// Entry point for `npx @pipeworx/mcp-<slug>`.
|
|
4
|
+
//
|
|
5
|
+
// Packs ship as raw TypeScript (no build step — see publish-pack.sh for why:
|
|
6
|
+
// tsx sidesteps every extensionless-import / bare-JSON-import edge case a
|
|
7
|
+
// per-pack tsc build would have to solve one pack at a time). This file
|
|
8
|
+
// registers tsx's ESM loader programmatically, then hands off to src/server.ts,
|
|
9
|
+
// which wraps the pack's {tools, callTool} export in a stdio MCP server.
|
|
10
|
+
//
|
|
11
|
+
// Copied verbatim into every published pack repo by scripts/publish-pack.sh —
|
|
12
|
+
// edit this file, not a per-pack copy.
|
|
13
|
+
import { register } from 'tsx/esm/api';
|
|
14
|
+
|
|
15
|
+
register();
|
|
16
|
+
|
|
17
|
+
await import('../src/server.ts');
|
package/package.json
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@pipeworx/mcp-commoncrawl",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Common Crawl's public web archive — find every time a URL was crawled since 2008 and read the exact page bytes that were captured.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "src/index.ts",
|
|
7
|
+
"types": "src/index.ts",
|
|
8
|
+
"bin": {
|
|
9
|
+
"mcp-commoncrawl": "bin/cli.js"
|
|
10
|
+
},
|
|
11
|
+
"keywords": ["mcp", "mcp-server", "model-context-protocol", "pipeworx", "commoncrawl"],
|
|
12
|
+
"license": "MIT",
|
|
13
|
+
"repository": {
|
|
14
|
+
"type": "git",
|
|
15
|
+
"url": "git+https://github.com/pipeworx-io/mcp-commoncrawl.git"
|
|
16
|
+
},
|
|
17
|
+
"scripts": {
|
|
18
|
+
"typecheck": "tsc --noEmit"
|
|
19
|
+
},
|
|
20
|
+
"dependencies": {
|
|
21
|
+
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
22
|
+
"tsx": "^4.19.0"
|
|
23
|
+
},
|
|
24
|
+
"devDependencies": {
|
|
25
|
+
"typescript": "^5.9.3",
|
|
26
|
+
"@cloudflare/workers-types": "^4.20260405.1"
|
|
27
|
+
},
|
|
28
|
+
"pipeworx": {
|
|
29
|
+
"sourceHash": "v1-893697cf7a7c4f037a1ee7435339472b6070086157a443c3a5f4e0258dd9398b",
|
|
30
|
+
"sourceCommit": "839ab210d969ef7f822cd747ea516f6bb49e9231"
|
|
31
|
+
}
|
|
32
|
+
}
|
package/server.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
|
|
3
|
+
"name": "io.github.pipeworx-io/commoncrawl",
|
|
4
|
+
"title": "Commoncrawl",
|
|
5
|
+
"description": "Common Crawl's public web archive — find every time a URL was crawled since 2008 and read the…",
|
|
6
|
+
"version": "0.1.0",
|
|
7
|
+
"websiteUrl": "https://pipeworx.io/packs/commoncrawl",
|
|
8
|
+
"repository": {
|
|
9
|
+
"url": "https://github.com/pipeworx-io/mcp-commoncrawl",
|
|
10
|
+
"source": "github"
|
|
11
|
+
},
|
|
12
|
+
"remotes": [
|
|
13
|
+
{
|
|
14
|
+
"type": "streamable-http",
|
|
15
|
+
"url": "https://gateway.pipeworx.io/commoncrawl/mcp"
|
|
16
|
+
}
|
|
17
|
+
]
|
|
18
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,1036 @@
|
|
|
1
|
+
interface McpToolDefinition {
|
|
2
|
+
name: string;
|
|
3
|
+
description: string;
|
|
4
|
+
/** Human-facing one-liner (fleet #1967). Optional; consumers fall back to
|
|
5
|
+
* description. Kept in step with shared/src/types.ts — scripts/lib/
|
|
6
|
+
* check-inlined-types.mjs reports drift at publish time. */
|
|
7
|
+
summary?: string;
|
|
8
|
+
inputSchema: {
|
|
9
|
+
type: 'object';
|
|
10
|
+
properties: Record<string, unknown>;
|
|
11
|
+
required?: string[];
|
|
12
|
+
anyOf?: Array<{ required: string[] }>;
|
|
13
|
+
oneOf?: Array<{ required: string[] }>;
|
|
14
|
+
allOf?: Array<{ required: string[] }>;
|
|
15
|
+
};
|
|
16
|
+
outputSchema?: Record<string, unknown>;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
interface McpToolExport {
|
|
20
|
+
tools: McpToolDefinition[];
|
|
21
|
+
callTool: (name: string, args: Record<string, unknown>) => Promise<unknown>;
|
|
22
|
+
meter?: { credits: number };
|
|
23
|
+
cost?: Record<string, unknown>;
|
|
24
|
+
provider?: string;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Was this failure OUR OWN web service? — the other half of `internal-db-class.ts`.
|
|
29
|
+
*
|
|
30
|
+
* fleet #1089 pulled failures from our own Postgres out of `upstream_down` by
|
|
31
|
+
* keying on the SQLSTATE inside PostgREST's four-key error envelope. That
|
|
32
|
+
* covered the majority and structurally could not cover the rest: the rest
|
|
33
|
+
* never reach Postgres, so they carry no SQLSTATE. What was left, measured over
|
|
34
|
+
* the 24h to 2026-09-02T15:00Z (fleet #1096):
|
|
35
|
+
*
|
|
36
|
+
* 5 pipeworx-catalog get_pack_tools Pipeworx catalog error: 522 — error code: 522
|
|
37
|
+
* 3 fleet fleet_list_open … upstream_down: Fleet task queue did not respond within 25s
|
|
38
|
+
*
|
|
39
|
+
* 521/522/523/526 are Cloudflare saying its edge could not reach an ORIGIN, and
|
|
40
|
+
* in both of those rows the origin is ours — `gateway.pipeworx.io` for the
|
|
41
|
+
* catalog pack (it self-fetches when the gateway hasn't injected a manifest),
|
|
42
|
+
* our own Supabase for fleet. There is no third party anywhere in either call.
|
|
43
|
+
* Same defect as #1089: our own outage filed under `upstream_down`, the one
|
|
44
|
+
* class that means "the source is unreachable and there is nothing for us to
|
|
45
|
+
* fix", which is why the problem-tools triage skips it.
|
|
46
|
+
*
|
|
47
|
+
* WHY NOT A WORDING RULE. The obvious fix is to match `fleet db error:` and
|
|
48
|
+
* `Pipeworx catalog error:` in classifyToolError. Each is emitted from exactly
|
|
49
|
+
* one site today, so it would work today. It would also rot the first time
|
|
50
|
+
* somebody rewords a label — silently, and in the direction of hiding our own
|
|
51
|
+
* outage, which is worse than the bug being fixed. Every prose rule in
|
|
52
|
+
* error-class.ts has needed widening as packs invented new wording (#409/#450/
|
|
53
|
+
* #584); that history is most of that file's comment budget.
|
|
54
|
+
*
|
|
55
|
+
* WHAT THIS KEYS ON INSTEAD: **the host the call actually reached.** A URL's
|
|
56
|
+
* hostname is a fact about the call, not a guess about its prose. Two
|
|
57
|
+
* consequences that a pack-level flag could not give us, and the reason the
|
|
58
|
+
* flag was rejected:
|
|
59
|
+
*
|
|
60
|
+
* - It describes the CALL, not the pack. `govcon-intel` fans out to our own
|
|
61
|
+
* Supabase AND to genuine third parties; `court-listener` holds our cache
|
|
62
|
+
* in Supabase and fetches courtlistener.com. An `internallyHosted: true` on
|
|
63
|
+
* either pack would relabel a real third-party outage as ours — inventing
|
|
64
|
+
* work, which is the same class of error in the opposite direction.
|
|
65
|
+
* - It covers every future internal pack for free, instead of one declared
|
|
66
|
+
* slug at a time.
|
|
67
|
+
*
|
|
68
|
+
* WHY IT SURVIVES A REWORD. The marker below is not matched as a literal by two
|
|
69
|
+
* separate files. `markInternalOrigin()` writes it and `internalHostMetricsClass()`
|
|
70
|
+
* reads it, both from the single exported `INTERNAL_ORIGIN_MARKER` constant in
|
|
71
|
+
* this module — so changing the wording changes both sides in the same edit and
|
|
72
|
+
* cannot desynchronise them. The pack's own label (`fleet db error:`,
|
|
73
|
+
* `Pipeworx catalog error:`) is not read at all: reword it freely, the class is
|
|
74
|
+
* unaffected. That is the property `stripClassPrefix` lacked when it drifted
|
|
75
|
+
* from its own classifier three times and needed a CI gate to hold them
|
|
76
|
+
* together.
|
|
77
|
+
*
|
|
78
|
+
* WHERE THE 5xx TEST LIVES. `markInternalOrigin` is called from the places that
|
|
79
|
+
* hold the real `Response` — `httpError`/`httpErrorMessage` and the timeout
|
|
80
|
+
* branch of `fetchWithTimeout` in `shared/src/http.ts` — so "is this an
|
|
81
|
+
* availability failure" is decided from the actual status code, never re-derived
|
|
82
|
+
* by scraping a number out of a sentence. A 404 from our own registry for a slug
|
|
83
|
+
* that does not exist is a caller's bad argument and is deliberately NOT marked.
|
|
84
|
+
*/
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* OUR OWN web service was unreachable — not an upstream, and never `upstream_down`.
|
|
88
|
+
*
|
|
89
|
+
* ONE value, not three, unlike `internal_db_*`. That split existed because a
|
|
90
|
+
* slow query, an exhausted pool and an unknown SQLSTATE have different owners
|
|
91
|
+
* and different fixes. Here there is only one story to tell — an origin we run
|
|
92
|
+
* did not answer the edge — and one owner. A bucket with no distinct owner per
|
|
93
|
+
* value is decoration; #724 is what happens when a class holds several
|
|
94
|
+
* situations, and inventing sub-values ahead of a reason to act on them
|
|
95
|
+
* differently is the same mistake with the sign flipped.
|
|
96
|
+
*
|
|
97
|
+
* METRICS ONLY, exactly like PLATFORM_KEY_ERROR_CLASS and the internal_db
|
|
98
|
+
* values. `classifyToolError` still answers `upstream_down` for the retry and
|
|
99
|
+
* hint paths, which only care whether retrying or a sibling tool might work —
|
|
100
|
+
* and it might. Nothing a caller sees or is charged changes here.
|
|
101
|
+
*
|
|
102
|
+
* READ SIDE: this value is in BROKEN_TOOL_CLASSES, FAULT_CLASSES and
|
|
103
|
+
* ALL_ERROR_CLASSES in `workers/registry-api/src/index.ts`. All three, or it
|
|
104
|
+
* lands on no dashboard — fleet #721 is the warning, where the #719 split
|
|
105
|
+
* worked on the write side and was invisible for weeks.
|
|
106
|
+
*/
|
|
107
|
+
const INTERNAL_SERVICE_UNREACHABLE_CLASS = 'internal_service_unreachable';
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* The token that carries "this origin is ours" from the call site to the
|
|
111
|
+
* classifier.
|
|
112
|
+
*
|
|
113
|
+
* Appended to the error message rather than attached to the Error object,
|
|
114
|
+
* because the object does not survive the trip: 275 packs return `{ error:
|
|
115
|
+
* string }` instead of throwing, the gateway reads `observedError` as a string,
|
|
116
|
+
* and the fleet pack rebuilds its error from a captured status + body across a
|
|
117
|
+
* retry loop. A property on an Error would be dropped by every one of those
|
|
118
|
+
* paths and the class would work in tests and vanish in production.
|
|
119
|
+
*
|
|
120
|
+
* WORDING IS LOAD-BEARING, same rule as labelAge's note in authority.ts. This
|
|
121
|
+
* string is appended to a pack's thrown Error message (shared/src/http.ts),
|
|
122
|
+
* and a thrown Error's message is exactly what the gateway hands back to the
|
|
123
|
+
* caller as `content[0].text` when nothing rewrites it (workers/gateway/src
|
|
124
|
+
* catches the throw and sets `rawResult.message = stripClassPrefix(error)`,
|
|
125
|
+
* which does not touch this suffix) — so the original wording,
|
|
126
|
+
* " [pipeworx-hosted origin — our own service, not a third party]", was not a
|
|
127
|
+
* theoretical leak: it shipped live on pipeworx-catalog's 522s, 7 times in 6
|
|
128
|
+
* hours on 2026-09-02 (see tests/golden-internal-service.test.ts), verbatim
|
|
129
|
+
* naming Pipeworx as the host. check:hosting-claims never caught it because it
|
|
130
|
+
* did not scan shared/ at all (task #2009). Reworded to describe the
|
|
131
|
+
* OBSERVATION (the origin did not answer) without a claim about who runs it —
|
|
132
|
+
* the identical fix labelAge got: drop the possessive, keep the fact.
|
|
133
|
+
*/
|
|
134
|
+
const INTERNAL_ORIGIN_MARKER = ' [origin did not respond — retry before concluding the named source is down]';
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* Supabase's data plane for a project is `<ref>.supabase.co`, where the ref is
|
|
138
|
+
* exactly twenty lowercase letters (ours is `pqauisounztsgdgfkhke`).
|
|
139
|
+
*
|
|
140
|
+
* Matching the shape rather than listing the ref keeps this correct when we add
|
|
141
|
+
* a project — `supabaseEnv` on a pack entry already points some packs at a
|
|
142
|
+
* second one — while still excluding `status.supabase.co`, which is Supabase's
|
|
143
|
+
* own status page and emphatically not our database. Verified 2026-09-02 by
|
|
144
|
+
* `grep -rhoE '[a-z0-9-]+\.supabase\.(co|in)' mcps shared workers scripts`: the
|
|
145
|
+
* only real project ref anywhere in the tree is ours, the rest are doc
|
|
146
|
+
* placeholders (`abc`, `xyz`, `example`) which this pattern also excludes. Same
|
|
147
|
+
* finding internal-db-class.ts relies on for the PostgREST envelope being ours
|
|
148
|
+
* by construction.
|
|
149
|
+
*/
|
|
150
|
+
const SUPABASE_PROJECT_HOST = /^[a-z]{20}\.supabase\.(co|in)$/;
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Is this a host WE run?
|
|
154
|
+
*
|
|
155
|
+
* Deliberately NOT including `*.workers.dev`: plenty of third-party APIs are
|
|
156
|
+
* hosted on workers.dev, so the suffix says where something runs and not who
|
|
157
|
+
* owns it. Every internal call we actually make goes to a `pipeworx.io`
|
|
158
|
+
* hostname or to our Supabase project, both of which are ownership facts.
|
|
159
|
+
*
|
|
160
|
+
* `workers/gateway/src/provenance.ts`'s `OUR_HOSTS` answers the same
|
|
161
|
+
* question and DOES include `workers.dev` — a documented divergence
|
|
162
|
+
* (task #2051), not a bug to converge. That list decides what a response may
|
|
163
|
+
* cite as a data SOURCE, where a false negative (citing our own worker as an
|
|
164
|
+
* external source) is the hosting-disclosure leak this whole file exists to
|
|
165
|
+
* prevent, so it errs broad. This one decides who gets BLAMED for a 5xx in
|
|
166
|
+
* outage metrics read by on-call, where a false positive (crediting our own
|
|
167
|
+
* infra with a third party's outage) hides the real failure, so it errs
|
|
168
|
+
* narrow. Same suffix, opposite direction, because they are never called for
|
|
169
|
+
* the same reason.
|
|
170
|
+
*
|
|
171
|
+
* Returns false on anything unparseable rather than throwing — this runs inside
|
|
172
|
+
* an error path, and an error path that can itself throw turns a diagnosable
|
|
173
|
+
* failure into a mystery.
|
|
174
|
+
*/
|
|
175
|
+
function isPipeworxOrigin(url: string | URL | undefined | null): boolean {
|
|
176
|
+
if (!url) return false;
|
|
177
|
+
let host: string;
|
|
178
|
+
try {
|
|
179
|
+
host = new URL(url instanceof URL ? url.href : url).hostname.toLowerCase();
|
|
180
|
+
} catch {
|
|
181
|
+
return false;
|
|
182
|
+
}
|
|
183
|
+
if (host === 'pipeworx.io' || host.endsWith('.pipeworx.io')) return true;
|
|
184
|
+
return SUPABASE_PROJECT_HOST.test(host);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Append the marker when this failure was OUR origin failing to answer.
|
|
189
|
+
*
|
|
190
|
+
* `status` is the HTTP status when there is one, and omitted for a timeout —
|
|
191
|
+
* where there is no response at all, and "the origin did not answer" is the
|
|
192
|
+
* whole observation. Statuses below 500 are left alone: a 404 from our own
|
|
193
|
+
* registry for a slug that does not exist is the caller's argument, not our
|
|
194
|
+
* outage, and marking it would put ordinary 404s on the incident dashboard.
|
|
195
|
+
*
|
|
196
|
+
* Idempotent, so a message that is wrapped and re-marked on the way up (the
|
|
197
|
+
* fleet pack's retry loop re-throws through two layers) carries the marker once.
|
|
198
|
+
*/
|
|
199
|
+
function markInternalOrigin(
|
|
200
|
+
message: string,
|
|
201
|
+
url: string | URL | undefined | null,
|
|
202
|
+
status?: number,
|
|
203
|
+
): string {
|
|
204
|
+
if (status !== undefined && status < 500) return message;
|
|
205
|
+
if (!isPipeworxOrigin(url)) return message;
|
|
206
|
+
if (message.includes(INTERNAL_ORIGIN_MARKER)) return message;
|
|
207
|
+
return message + INTERNAL_ORIGIN_MARKER;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Which blob4 value a failure from our own web services books as, or undefined
|
|
212
|
+
* if this is not one.
|
|
213
|
+
*
|
|
214
|
+
* Ordered AFTER `internalDbMetricsClass` at the call site: a PostgREST envelope
|
|
215
|
+
* from our own Supabase is a strictly more specific statement about the same
|
|
216
|
+
* row (which of our services, and why), and the two cannot disagree about
|
|
217
|
+
* whether the failure is ours.
|
|
218
|
+
*/
|
|
219
|
+
function internalHostMetricsClass(error: string): string | undefined {
|
|
220
|
+
return error.includes(INTERNAL_ORIGIN_MARKER) ? INTERNAL_SERVICE_UNREACHABLE_CLASS : undefined;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* One place to turn a failed `fetch` into an error a caller can act on.
|
|
226
|
+
*
|
|
227
|
+
* Nearly every pack was written the same way:
|
|
228
|
+
*
|
|
229
|
+
* if (!res.ok) throw new Error(`Unsplash: ${res.status}`);
|
|
230
|
+
*
|
|
231
|
+
* which discards the response body — and the body is usually where the upstream
|
|
232
|
+
* says what was actually wrong ("**symbol** not found: GBP", "parameter `year`
|
|
233
|
+
* out of range", "unknown taxonomy id"). The caller gets a number, cannot
|
|
234
|
+
* self-correct, and retries the same broken call. A 2026-07-31 sweep found this
|
|
235
|
+
* shape in 481 of 1,400 packs, 47 of them PLATFORM-keyed.
|
|
236
|
+
*
|
|
237
|
+
* It also hides bugs one level down. Two of the first three packs audited had a
|
|
238
|
+
* second defect that only existed because of this line: unsplash's rate-limit
|
|
239
|
+
* branch sat BELOW a catch-all and was unreachable, and bea-gov parsed
|
|
240
|
+
* `BEAAPI.Error.APIErrorDescription` below a `!res.ok` throw that made the
|
|
241
|
+
* parsing dead code for every non-200.
|
|
242
|
+
*
|
|
243
|
+
* DELIBERATELY NOT A CLASSIFIER. It does not add `user_error:` /
|
|
244
|
+
* `upstream_down:` prefixes. Those decide which tier a failure lands in, and the
|
|
245
|
+
* `error` tier is what the daily problem-tools list is built from — it means
|
|
246
|
+
* "Pipeworx has a defect". A 400 is genuinely ambiguous: often a caller's bad
|
|
247
|
+
* argument, but sometimes a query WE built wrong (ted-eu comma-joined its CPV
|
|
248
|
+
* values into something TED rejected, and that bug was found only because it sat
|
|
249
|
+
* in `error`). Blanket-classifying 400s as caller mistakes would have hidden it.
|
|
250
|
+
* A pack that KNOWS which it is should keep saying so explicitly; this helper is
|
|
251
|
+
* for the 481 that say nothing at all.
|
|
252
|
+
*/
|
|
253
|
+
|
|
254
|
+
/** Longest upstream explanation we'll pass through. Enough for a real message,
|
|
255
|
+
* short enough that an HTML page or a stack trace can't swamp the error. */
|
|
256
|
+
|
|
257
|
+
const MAX_DETAIL = 300;
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Default bound for `fetchWithTimeout` when a pack doesn't state its own.
|
|
261
|
+
*
|
|
262
|
+
* 25s mirrors the number `epo-ops` landed on after measuring the real failure:
|
|
263
|
+
* a degraded upstream that doesn't error, it just never answers, and a Worker
|
|
264
|
+
* sits in `await fetch()` until ITS OWN execution budget kills the request —
|
|
265
|
+
* which can take minutes, not seconds (epo_ops_search_patents measured 4-8
|
|
266
|
+
* MINUTE hangs before this existed). 25s is short enough that a caller gets a
|
|
267
|
+
* fast, actionable error instead of holding the connection, and long enough
|
|
268
|
+
* that it doesn't false-trip on a merely-slow-but-alive upstream.
|
|
269
|
+
*/
|
|
270
|
+
const DEFAULT_FETCH_TIMEOUT_MS = 25_000;
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Read the body of a failed response and fold it into a throwable Error.
|
|
274
|
+
*
|
|
275
|
+
* Usage — note the `await`, which is the one thing that makes this a mechanical
|
|
276
|
+
* change rather than a drop-in:
|
|
277
|
+
*
|
|
278
|
+
* if (!res.ok) throw await httpError(res, 'Unsplash');
|
|
279
|
+
*
|
|
280
|
+
* Safe to call on any non-ok response: a body that is missing, empty, unreadable
|
|
281
|
+
* or HTML degrades to exactly the old `Name: 404` string rather than throwing
|
|
282
|
+
* something new from inside the error path.
|
|
283
|
+
*/
|
|
284
|
+
async function httpError(res: Response, name: string): Promise<Error> {
|
|
285
|
+
return new Error(await httpErrorMessage(res, name));
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** The message text without constructing an Error — for packs that need to wrap
|
|
289
|
+
* it in their own envelope or add an explicit classification prefix. */
|
|
290
|
+
async function httpErrorMessage(res: Response, name: string): Promise<string> {
|
|
291
|
+
// The one place a 5xx from a host WE run gets stamped as ours. `res.url` is
|
|
292
|
+
// the URL the fetch actually resolved to (after redirects), so this is a fact
|
|
293
|
+
// about the call rather than a guess from the `name` the pack passed in —
|
|
294
|
+
// reword that label freely, the class does not move. See
|
|
295
|
+
// internal-host-class.ts; no-op for every third-party upstream, which is why
|
|
296
|
+
// this touches 481 packs' error text and changes none of it.
|
|
297
|
+
return markInternalOrigin(
|
|
298
|
+
`${name}: ${res.status}${detailSuffix(await readDetail(res))}`,
|
|
299
|
+
res.url,
|
|
300
|
+
res.status,
|
|
301
|
+
);
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* Just the upstream's own explanation — no name, no status.
|
|
306
|
+
*
|
|
307
|
+
* For a pack that has already said both in its own sentence. epo-ops reads
|
|
308
|
+
* `EPO rejected this search as too large (HTTP 413) — ${httpErrorMessage(…)}`,
|
|
309
|
+
* which rendered as `… (HTTP 413) — EPO: 413.` once the XML detail was being
|
|
310
|
+
* dropped: the upstream named twice, the status twice, and the one thing EPO
|
|
311
|
+
* actually said ("Not enough characters before truncation character") nowhere
|
|
312
|
+
* (fleet #712). Returns '' when the body carries nothing readable, so a caller
|
|
313
|
+
* can fall back to its own wording.
|
|
314
|
+
*/
|
|
315
|
+
async function upstreamDetail(res: Response): Promise<string> {
|
|
316
|
+
return readDetail(res);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Read a SUCCESSFUL response as JSON, failing loudly when it isn't JSON.
|
|
321
|
+
*
|
|
322
|
+
* `httpError` above only ever runs on `!res.ok`, which leaves the nastier half
|
|
323
|
+
* of the problem unhandled: an upstream that answers **HTTP 200 with an HTML
|
|
324
|
+
* page**. A bot wall, a login redirect, a maintenance interstitial and a CDN
|
|
325
|
+
* error page are all 200s, so `res.ok` is true, and `res.json()` then throws
|
|
326
|
+
* `Unexpected token '<', "<!DOCTYPE "... is not valid JSON`.
|
|
327
|
+
*
|
|
328
|
+
* That string is the problem. It names no upstream, carries no status, and
|
|
329
|
+
* reads like a parser bug in Pipeworx — so it lands in the `error` tier, which
|
|
330
|
+
* means "we have a defect", and the caller is told nothing they can act on.
|
|
331
|
+
* data.govt.nz sat dead behind an Imperva challenge this way and every
|
|
332
|
+
* status-code health check we own reported it green (7889a845). A zero-length
|
|
333
|
+
* body has the same shape: `Unexpected end of JSON input`, seen this week on
|
|
334
|
+
* uk-gazette (83% of external calls) and census.
|
|
335
|
+
*
|
|
336
|
+
* UNLIKE `httpError`, this one DOES classify, and the asymmetry is deliberate.
|
|
337
|
+
* A 400 is genuinely ambiguous — often the caller's bad argument, sometimes a
|
|
338
|
+
* query we built wrong — so blanket-classifying it would hide our own bugs.
|
|
339
|
+
* There is no such ambiguity here: **no argument a caller can pass makes a JSON
|
|
340
|
+
* API return an HTML page.** It is always the upstream, so `upstream_down:` is
|
|
341
|
+
* a statement of fact rather than a guess, and it keeps these out of the
|
|
342
|
+
* problem-tools list where they crowd out real defects.
|
|
343
|
+
*
|
|
344
|
+
* const data = await parseJson<Feed>(res, 'UK Gazette');
|
|
345
|
+
*
|
|
346
|
+
* Call it only after the `!res.ok` check — on a failed response you want
|
|
347
|
+
* `httpError`, which mines the body for the upstream's own explanation.
|
|
348
|
+
*/
|
|
349
|
+
async function parseJson<T>(res: Response, name: string): Promise<T> {
|
|
350
|
+
let raw: string;
|
|
351
|
+
try {
|
|
352
|
+
raw = await res.text();
|
|
353
|
+
} catch {
|
|
354
|
+
throw new Error(
|
|
355
|
+
`upstream_down: ${name} returned a body that could not be read (HTTP ${res.status}). ` +
|
|
356
|
+
'The connection most likely dropped mid-response; retrying is reasonable.',
|
|
357
|
+
);
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
const type = res.headers.get('content-type') ?? 'no content-type';
|
|
361
|
+
|
|
362
|
+
if (!raw.trim()) {
|
|
363
|
+
throw new Error(
|
|
364
|
+
`upstream_down: ${name} answered HTTP ${res.status} with an EMPTY body where JSON was expected (${type}). ` +
|
|
365
|
+
'Nothing about the request can cause this — it is an upstream fault, and the same call may well work on retry.',
|
|
366
|
+
);
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
// Checked before parsing rather than in the catch, because knowing it is
|
|
370
|
+
// markup is what turns "we failed to parse something" into "they served a
|
|
371
|
+
// web page" — the second is diagnosable, the first is not.
|
|
372
|
+
const head = raw.slice(0, 200).trimStart().toLowerCase();
|
|
373
|
+
if (head.startsWith('<!doctype') || head.startsWith('<html') || head.startsWith('<?xml')) {
|
|
374
|
+
const kind = head.startsWith('<?xml') ? 'an XML document' : 'an HTML page';
|
|
375
|
+
// The summary, not the source. Pasting the first 120 characters of a web
|
|
376
|
+
// page handed the agent `<!DOCTYPE html><html lang="en"…` — the same leak
|
|
377
|
+
// this branch exists to describe (fleet #712).
|
|
378
|
+
throw new Error(
|
|
379
|
+
`upstream_down: ${name} answered HTTP ${res.status} with ${kind} instead of JSON (${type}). ` +
|
|
380
|
+
'That is typically a bot wall, a login redirect or a maintenance page — it is returned as a SUCCESS, ' +
|
|
381
|
+
`so status-code health checks read it as fine. No argument change will get past it. ` +
|
|
382
|
+
`The page says: ${summarizeErrorBody(raw) || 'nothing readable'}`,
|
|
383
|
+
);
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
try {
|
|
387
|
+
return JSON.parse(raw) as T;
|
|
388
|
+
} catch {
|
|
389
|
+
throw new Error(
|
|
390
|
+
`upstream_down: ${name} answered HTTP ${res.status} with a body that is not valid JSON (${type}). ` +
|
|
391
|
+
`It begins: ${stripMarkup(raw).slice(0, 120) || '(unreadable)'}`,
|
|
392
|
+
);
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
/**
|
|
397
|
+
* `fetch`, but bounded — the fix for a systemic gap found 2026-08-30: a grep
|
|
398
|
+
* audit of every pack's `mcps/*\/src/index.ts` found 1,339 of ~1,500 call
|
|
399
|
+
* `fetch()` with NO timeout guard anywhere in the file. Two of those
|
|
400
|
+
* (epo-ops, statcan) were confirmed live-hanging for 4-8 minutes before this
|
|
401
|
+
* existed — every unguarded call carries the same risk, just unconfirmed.
|
|
402
|
+
*
|
|
403
|
+
* Mirrors the `epoFetch` wrapper `mcps/epo-ops/src/index.ts` shipped first:
|
|
404
|
+
* bound the request with `AbortSignal.timeout`, and on a timeout/abort throw
|
|
405
|
+
* an `upstream_down:` error that names the upstream and the bound rather than
|
|
406
|
+
* letting the raw `TimeoutError`/`AbortError` (which names neither) propagate.
|
|
407
|
+
* `upstream_down:` is deliberate, same reasoning as `parseJson` above — no
|
|
408
|
+
* argument a caller passes can make an upstream hang, so it is always the
|
|
409
|
+
* upstream's fault, and marking it that way keeps a slow API off the
|
|
410
|
+
* problem-tools list where it would crowd out our own defects.
|
|
411
|
+
*
|
|
412
|
+
* Usage — a mechanical swap for a bare `fetch(url, init)`:
|
|
413
|
+
*
|
|
414
|
+
* const res = await fetchWithTimeout(url, init, 'Some API');
|
|
415
|
+
*
|
|
416
|
+
* Pass `timeoutMs` as a fourth argument to override the default for a pack
|
|
417
|
+
* with a known-slower upstream; the label should be the same short name you'd
|
|
418
|
+
* pass to `httpError`/`httpErrorMessage` for that call.
|
|
419
|
+
*/
|
|
420
|
+
async function fetchWithTimeout(
|
|
421
|
+
url: string | URL,
|
|
422
|
+
init: RequestInit = {},
|
|
423
|
+
name: string,
|
|
424
|
+
timeoutMs: number = DEFAULT_FETCH_TIMEOUT_MS,
|
|
425
|
+
): Promise<Response> {
|
|
426
|
+
try {
|
|
427
|
+
return await fetch(url, { ...init, signal: AbortSignal.timeout(timeoutMs) });
|
|
428
|
+
} catch (err) {
|
|
429
|
+
if (err instanceof Error && (err.name === 'TimeoutError' || err.name === 'AbortError')) {
|
|
430
|
+
// States the OBSERVATION (no response in N seconds), not a diagnosis.
|
|
431
|
+
// "appears to be degraded" is an inference about the vendor that we have
|
|
432
|
+
// not checked, and it is wrong in a way that misdirects whoever reads it:
|
|
433
|
+
// a timeout from a Worker can equally mean OUR egress is blocked.
|
|
434
|
+
//
|
|
435
|
+
// Measured today (2026-09-01, fleet #1047): every call to
|
|
436
|
+
// mainnet.base.org failed from the x402 facilitator while the identical
|
|
437
|
+
// request from a laptop returned 200. Base was entirely healthy; the
|
|
438
|
+
// public RPC refuses Cloudflare Worker egress. Had this message fired
|
|
439
|
+
// there it would have blamed Base by name, and the next person would have
|
|
440
|
+
// waited for a vendor outage to clear that did not exist.
|
|
441
|
+
// A timeout has no status to test — there is no response at all — so
|
|
442
|
+
// `markInternalOrigin` is called without one: an origin we run that never
|
|
443
|
+
// answered is an availability failure by definition. This is the half of
|
|
444
|
+
// fleet #1096 with neither a SQLSTATE nor a status code to key on.
|
|
445
|
+
throw new Error(
|
|
446
|
+
markInternalOrigin(
|
|
447
|
+
`upstream_down: ${name} did not respond within ${timeoutMs / 1000}s. ` +
|
|
448
|
+
`That can be ${name} being slow or down, or this environment being unable to reach it ` +
|
|
449
|
+
`(some hosts refuse datacenter/Worker egress) — retry shortly, and check reachability ` +
|
|
450
|
+
`from elsewhere before concluding ${name} is down.`,
|
|
451
|
+
url,
|
|
452
|
+
),
|
|
453
|
+
);
|
|
454
|
+
}
|
|
455
|
+
throw err;
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
function detailSuffix(detail: string): string {
|
|
460
|
+
return detail ? ` — ${detail}` : '';
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
async function readDetail(res: Response): Promise<string> {
|
|
464
|
+
let raw: string;
|
|
465
|
+
try {
|
|
466
|
+
raw = await res.text();
|
|
467
|
+
} catch {
|
|
468
|
+
// Body already consumed, or the connection died mid-read. The status alone
|
|
469
|
+
// is still worth throwing — never let the error path throw its own error.
|
|
470
|
+
return '';
|
|
471
|
+
}
|
|
472
|
+
return summarizeErrorBody(raw);
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
/**
|
|
476
|
+
* Turn ANY error body — JSON, HTML, XML or plain text — into one short phrase
|
|
477
|
+
* that never contains markup.
|
|
478
|
+
*
|
|
479
|
+
* This used to just drop an HTML or XML body on the floor, on the reasoning
|
|
480
|
+
* that markup crowds out the status. That was half right. Dropping it loses the
|
|
481
|
+
* one sentence a caller could have acted on: an `Access Denied` title, an SDMX
|
|
482
|
+
* `<message:Error>` text, an OPS fault string. A 2026-08-30 support sweep
|
|
483
|
+
* measured 13 of 291 caller-facing error rows carrying a raw page or document
|
|
484
|
+
* verbatim, across 11 packs, and in every one of them the useful content —
|
|
485
|
+
* "Access Denied", "Invalid country code", "SCRAPE_TIMEOUT" — was in there,
|
|
486
|
+
* buried in markup the agent had to parse out of a string (fleet #712).
|
|
487
|
+
*
|
|
488
|
+
* So: extract the meaning, discard the markup. The output is passed through
|
|
489
|
+
* `stripMarkup` unconditionally, which is what lets `check:error-body-leak`
|
|
490
|
+
* assert mechanically that no caller-facing message can contain `<?xml`,
|
|
491
|
+
* `<!DOCTYPE` or `<html`.
|
|
492
|
+
*/
|
|
493
|
+
function summarizeErrorBody(raw: string): string {
|
|
494
|
+
if (!raw || !raw.trim()) return '';
|
|
495
|
+
|
|
496
|
+
const head = raw.slice(0, 400).trimStart().toLowerCase();
|
|
497
|
+
|
|
498
|
+
// An HTML error page (Cloudflare interstitial, nginx default, a login
|
|
499
|
+
// redirect) says what it is in its <title>, and almost nowhere else.
|
|
500
|
+
if (head.startsWith('<!doctype') || head.startsWith('<html')) {
|
|
501
|
+
const title = htmlTitle(raw);
|
|
502
|
+
return title
|
|
503
|
+
? `${title} (upstream returned an HTML error page, not an API response)`
|
|
504
|
+
: 'upstream returned an HTML error page, not an API response';
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
// XML fault documents — EPO OPS, SDMX (`<message:Error>`), SOAP faults. The
|
|
508
|
+
// human sentence sits in a child element whose tag name says what it is.
|
|
509
|
+
if (head.startsWith('<?xml') || head.startsWith('<')) {
|
|
510
|
+
const fault = xmlFaultText(raw);
|
|
511
|
+
return fault
|
|
512
|
+
? `${stripMarkup(fault).slice(0, MAX_DETAIL)} (from the upstream's XML error document)`
|
|
513
|
+
: 'upstream returned an XML error document with no readable message';
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
// Most JSON error bodies bury one human sentence among ids and echoed request
|
|
517
|
+
// params. Prefer that sentence; fall back to the whole body when the shape is
|
|
518
|
+
// unfamiliar, since an unfamiliar shape is exactly when we can least afford to
|
|
519
|
+
// guess wrong and show nothing.
|
|
520
|
+
const fromJson = messageFromJson(raw);
|
|
521
|
+
return stripMarkup(fromJson ?? raw).slice(0, MAX_DETAIL);
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
/** The `<title>` of an HTML error page, or its first `<h1>` — the two places a
|
|
525
|
+
* bot wall, a 502 and an "Access Denied" all state what happened. */
|
|
526
|
+
function htmlTitle(raw: string): string | null {
|
|
527
|
+
const head = raw.slice(0, 4000);
|
|
528
|
+
for (const re of [/<title[^>]*>([\s\S]*?)<\/title>/i, /<h1[^>]*>([\s\S]*?)<\/h1>/i]) {
|
|
529
|
+
const m = re.exec(head);
|
|
530
|
+
const text = m ? stripMarkup(m[1]) : '';
|
|
531
|
+
if (text) return text.slice(0, 160);
|
|
532
|
+
}
|
|
533
|
+
return null;
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
/** Tag names that carry the explanation in an XML fault document, namespace
|
|
537
|
+
* prefix optional (`<message:Error>`, `<com:Text>`, `<faultstring>`). */
|
|
538
|
+
const XML_FAULT_TAG_RE =
|
|
539
|
+
/<(?:[A-Za-z0-9_.-]+:)?(?:text|message|description|faultstring|reason|detail|title|errormessage|error)\b[^>]*>([^<]{2,400})</i;
|
|
540
|
+
|
|
541
|
+
function xmlFaultText(raw: string): string | null {
|
|
542
|
+
const head = raw.slice(0, 8000);
|
|
543
|
+
const tagged = XML_FAULT_TAG_RE.exec(head);
|
|
544
|
+
if (tagged && tagged[1].trim()) return tagged[1];
|
|
545
|
+
|
|
546
|
+
// Nothing conventionally named — take the longest text node instead. A fault
|
|
547
|
+
// document with one sentence in an oddly named element is still readable;
|
|
548
|
+
// returning nothing at all is not.
|
|
549
|
+
let best = '';
|
|
550
|
+
for (const m of head.matchAll(/>([^<>]{8,400})</g)) {
|
|
551
|
+
const text = m[1].trim();
|
|
552
|
+
if (text.length > best.length) best = text;
|
|
553
|
+
}
|
|
554
|
+
return best || null;
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
/**
|
|
558
|
+
* Remove every tag and stray angle bracket, then collapse whitespace.
|
|
559
|
+
*
|
|
560
|
+
* Applied to everything on the way out, including the JSON and plain-text
|
|
561
|
+
* paths, because an upstream is free to embed markup in a JSON string field —
|
|
562
|
+
* and a leak is a leak regardless of which branch produced it.
|
|
563
|
+
*/
|
|
564
|
+
function stripMarkup(s: string): string {
|
|
565
|
+
return collapse(decodeEntities(s.replace(/<[^>]*>/g, ' ')).replace(/[<>]/g, ' '));
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
/** The handful of entities that show up in error-page titles. Decoded AFTER
|
|
569
|
+
* tags are stripped and BEFORE the angle-bracket sweep, so `<script>`
|
|
570
|
+
* in a title cannot decode into markup that survives — EMBL-EBI's ChEMBL 500
|
|
571
|
+
* page renders as `500 Internal Server Error < EMBL-EBI` otherwise. */
|
|
572
|
+
function decodeEntities(s: string): string {
|
|
573
|
+
return s
|
|
574
|
+
.replace(/&(?:amp|#0*38);/gi, '&')
|
|
575
|
+
.replace(/&(?:lt|#0*60);/gi, '<')
|
|
576
|
+
.replace(/&(?:gt|#0*62);/gi, '>')
|
|
577
|
+
.replace(/&(?:quot|#0*34);/gi, '"')
|
|
578
|
+
.replace(/&(?:#0*39|apos|#x0*27);/gi, "'")
|
|
579
|
+
.replace(/ /gi, ' ');
|
|
580
|
+
}
|
|
581
|
+
|
|
582
|
+
/** The conventional "what went wrong" field, under any of the names upstreams
|
|
583
|
+
* actually use. Checked in order; first non-empty string wins. */
|
|
584
|
+
const MESSAGE_KEYS = [
|
|
585
|
+
'message', 'error_message', 'errorMessage', 'detail', 'details',
|
|
586
|
+
'description', 'error_description', 'reason', 'title', 'fault',
|
|
587
|
+
];
|
|
588
|
+
|
|
589
|
+
function messageFromJson(raw: string): string | null {
|
|
590
|
+
let parsed: unknown;
|
|
591
|
+
try {
|
|
592
|
+
parsed = JSON.parse(raw);
|
|
593
|
+
} catch {
|
|
594
|
+
return null;
|
|
595
|
+
}
|
|
596
|
+
return pickMessage(parsed, 0);
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
function pickMessage(node: unknown, depth: number): string | null {
|
|
600
|
+
// Two levels covers `{error: {message}}` and `{errors: [{detail}]}`, the two
|
|
601
|
+
// shapes that account for nearly all of them, without walking a large payload.
|
|
602
|
+
if (depth > 2 || node == null) return null;
|
|
603
|
+
|
|
604
|
+
if (typeof node === 'string') return node.trim() || null;
|
|
605
|
+
|
|
606
|
+
if (Array.isArray(node)) {
|
|
607
|
+
for (const item of node) {
|
|
608
|
+
const found = pickMessage(item, depth + 1);
|
|
609
|
+
if (found) return found;
|
|
610
|
+
}
|
|
611
|
+
return null;
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
if (typeof node !== 'object') return null;
|
|
615
|
+
const obj = node as Record<string, unknown>;
|
|
616
|
+
|
|
617
|
+
for (const key of MESSAGE_KEYS) {
|
|
618
|
+
const v = obj[key];
|
|
619
|
+
if (typeof v === 'string' && v.trim()) return v.trim();
|
|
620
|
+
}
|
|
621
|
+
// `{error: …}` where error is itself an object or a string — the single most
|
|
622
|
+
// common wrapper, so it is worth descending into by name rather than scanning
|
|
623
|
+
// every key and risking picking up an echoed request parameter.
|
|
624
|
+
for (const key of ['error', 'errors', 'fault', 'Error', 'data']) {
|
|
625
|
+
if (key in obj) {
|
|
626
|
+
const found = pickMessage(obj[key], depth + 1);
|
|
627
|
+
if (found) return found;
|
|
628
|
+
}
|
|
629
|
+
}
|
|
630
|
+
return null;
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
/** Errors are read in a single line of log output; newlines and runs of
|
|
634
|
+
* whitespace make a multi-line body unreadable there. */
|
|
635
|
+
function collapse(s: string): string {
|
|
636
|
+
return s.replace(/\s+/g, ' ').trim();
|
|
637
|
+
}
|
|
638
|
+
/**
|
|
639
|
+
* Common Crawl's public web archive — find every time a URL was crawled since 2008 and read the exact page bytes that were captured.
|
|
640
|
+
*
|
|
641
|
+
* Backed by the Common Crawl Foundation's CDX index at index.commoncrawl.org
|
|
642
|
+
* and the WARC data at data.commoncrawl.org. Keyless. Three tools: list the
|
|
643
|
+
* monthly crawl collections, search one collection's index for a URL, host or
|
|
644
|
+
* domain, and pull a single archived record back by byte range.
|
|
645
|
+
*
|
|
646
|
+
* Distinct from the `crawlgraph` pack, which answers link-graph questions about
|
|
647
|
+
* sites we have crawled ourselves. This one reads the Common Crawl Foundation's
|
|
648
|
+
* corpus.
|
|
649
|
+
*/
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
const INDEX_BASE = 'https://index.commoncrawl.org';
|
|
653
|
+
const DATA_BASE = 'https://data.commoncrawl.org';
|
|
654
|
+
const UA = 'pipeworx-mcp-commoncrawl/1.0 (+https://pipeworx.io)';
|
|
655
|
+
|
|
656
|
+
// Every outbound call is bounded. The CDX index answers a host/domain query by
|
|
657
|
+
// scanning a sharded index and can take tens of seconds for a large domain, so
|
|
658
|
+
// the bound here is deliberately higher than the 25s shared default.
|
|
659
|
+
// The public CDX index is a shared service and returns an nginx 502 sporadically
|
|
660
|
+
// — measured 2026-09-17, the SAME query alternated between 200 and 502 within a
|
|
661
|
+
// minute, so it is load, not the query. Retry a 5xx twice before surfacing it;
|
|
662
|
+
// without this a caller reads transient nginx noise as "Common Crawl has no
|
|
663
|
+
// record of this URL", which is a different and wrong answer.
|
|
664
|
+
async function ccFetch(url: string | URL, init: RequestInit = {}, timeoutMs = 45_000): Promise<Response> {
|
|
665
|
+
const headers = { 'User-Agent': UA, ...(init.headers ?? {}) };
|
|
666
|
+
let last: Response | undefined;
|
|
667
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
668
|
+
if (attempt > 0) await new Promise((r) => setTimeout(r, 750 * attempt));
|
|
669
|
+
last = await fetchWithTimeout(url, { ...init, headers }, 'Common Crawl', timeoutMs);
|
|
670
|
+
if (last.status < 500) return last;
|
|
671
|
+
}
|
|
672
|
+
return last as Response;
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
type CollInfo = {
|
|
676
|
+
id: string;
|
|
677
|
+
name: string;
|
|
678
|
+
timegate: string;
|
|
679
|
+
'cdx-api': string;
|
|
680
|
+
from?: string;
|
|
681
|
+
to?: string;
|
|
682
|
+
};
|
|
683
|
+
|
|
684
|
+
/** Capture row as the CDX index emits it (JSON-lines, all values strings). */
|
|
685
|
+
type CdxRow = Record<string, string>;
|
|
686
|
+
|
|
687
|
+
async function listCrawls(): Promise<CollInfo[]> {
|
|
688
|
+
const res = await ccFetch(`${INDEX_BASE}/collinfo.json`);
|
|
689
|
+
if (!res.ok) {
|
|
690
|
+
throw new Error(
|
|
691
|
+
`Common Crawl collinfo returned HTTP ${res.status}. The crawl list lives at ${INDEX_BASE}/collinfo.json; retry shortly.`,
|
|
692
|
+
);
|
|
693
|
+
}
|
|
694
|
+
return (await res.json()) as CollInfo[];
|
|
695
|
+
}
|
|
696
|
+
|
|
697
|
+
const tools: McpToolExport['tools'] = [
|
|
698
|
+
{
|
|
699
|
+
name: 'commoncrawl_crawls',
|
|
700
|
+
description:
|
|
701
|
+
'List the Common Crawl monthly crawl collections (crawl ids like "CC-MAIN-2026-34") with the date range each one covers. AUTHORITATIVE for "which Common Crawl snapshot covers <date>" — call this first to pick the crawl id that commoncrawl_index_search needs. Newest first. Keyless.',
|
|
702
|
+
inputSchema: {
|
|
703
|
+
type: 'object',
|
|
704
|
+
properties: {
|
|
705
|
+
limit: {
|
|
706
|
+
type: 'number',
|
|
707
|
+
description: 'How many crawls to return, newest first. Default 20, max 120 (the full history back to 2008).',
|
|
708
|
+
},
|
|
709
|
+
contains_date: {
|
|
710
|
+
type: 'string',
|
|
711
|
+
description:
|
|
712
|
+
'Optional ISO date, e.g. "2025-01-15". Returns only the crawl(s) whose capture window covers that date — the answer to "which snapshot would have seen my page then".',
|
|
713
|
+
},
|
|
714
|
+
},
|
|
715
|
+
required: [],
|
|
716
|
+
},
|
|
717
|
+
},
|
|
718
|
+
{
|
|
719
|
+
name: 'commoncrawl_index_search',
|
|
720
|
+
description:
|
|
721
|
+
'Search one Common Crawl collection\'s CDX index for archived captures of a URL, host, or whole domain. PREFER OVER WEB SEARCH when the question is "what did this page look like in <month>", "did Common Crawl ever see this URL", or "list the URLs crawled under this domain" — it returns the crawl record (timestamp, HTTP status, MIME type, detected language, content digest) plus the WARC filename/offset/length that commoncrawl_fetch_record needs to read the page bytes. Keyless.',
|
|
722
|
+
inputSchema: {
|
|
723
|
+
type: 'object',
|
|
724
|
+
properties: {
|
|
725
|
+
url: {
|
|
726
|
+
type: 'string',
|
|
727
|
+
description:
|
|
728
|
+
'URL, host, or domain to look up, e.g. "example.com", "https://www.nasa.gov/news", "*.python.org".',
|
|
729
|
+
},
|
|
730
|
+
crawl: {
|
|
731
|
+
type: 'string',
|
|
732
|
+
description:
|
|
733
|
+
'Crawl collection id from commoncrawl_crawls, e.g. "CC-MAIN-2026-34". Defaults to the most recent crawl.',
|
|
734
|
+
},
|
|
735
|
+
match_type: {
|
|
736
|
+
type: 'string',
|
|
737
|
+
description:
|
|
738
|
+
'How to match `url`: "exact" (that URL only), "prefix" (that path and everything under it), "host" (every URL on that exact host), "domain" (that host and all subdomains). Default "exact".',
|
|
739
|
+
enum: ['exact', 'prefix', 'host', 'domain'],
|
|
740
|
+
},
|
|
741
|
+
limit: {
|
|
742
|
+
type: 'number',
|
|
743
|
+
description: 'Maximum captures to return. Default 20, max 200.',
|
|
744
|
+
},
|
|
745
|
+
page: {
|
|
746
|
+
type: 'number',
|
|
747
|
+
description: 'Zero-based page number for paging through a large domain query. Default 0.',
|
|
748
|
+
},
|
|
749
|
+
filter: {
|
|
750
|
+
type: 'string',
|
|
751
|
+
description:
|
|
752
|
+
'Optional CDX field filter, e.g. "=status:200" (only successful captures), "=mime:text/html", "~url:.*blog.*". Repeatable filters are not supported here — pass one.',
|
|
753
|
+
},
|
|
754
|
+
from: {
|
|
755
|
+
type: 'string',
|
|
756
|
+
description: 'Optional lower bound on capture timestamp, as YYYY, YYYYMM, or YYYYMMDD, e.g. "202501".',
|
|
757
|
+
},
|
|
758
|
+
to: {
|
|
759
|
+
type: 'string',
|
|
760
|
+
description: 'Optional upper bound on capture timestamp, same format as `from`.',
|
|
761
|
+
},
|
|
762
|
+
},
|
|
763
|
+
required: ['url'],
|
|
764
|
+
},
|
|
765
|
+
},
|
|
766
|
+
{
|
|
767
|
+
name: 'commoncrawl_fetch_record',
|
|
768
|
+
description:
|
|
769
|
+
'Read one archived page back out of Common Crawl by byte range — pass the filename, offset and length from a commoncrawl_index_search result and get the WARC headers, the captured HTTP response headers, and the page body as it was crawled. AUTHORITATIVE for "what did this page actually say when it was crawled", including pages that have since changed or gone offline. The body is truncated to a byte cap you control. Keyless.',
|
|
770
|
+
inputSchema: {
|
|
771
|
+
type: 'object',
|
|
772
|
+
properties: {
|
|
773
|
+
filename: {
|
|
774
|
+
type: 'string',
|
|
775
|
+
description:
|
|
776
|
+
'WARC path from a capture row, e.g. "crawl-data/CC-MAIN-2025-05/segments/1736703361969.6/warc/CC-MAIN-20250112194358-20250112224358-00433.warc.gz".',
|
|
777
|
+
},
|
|
778
|
+
offset: {
|
|
779
|
+
type: 'number',
|
|
780
|
+
description: 'Byte offset of the record within that WARC file (the capture row\'s "offset").',
|
|
781
|
+
},
|
|
782
|
+
length: {
|
|
783
|
+
type: 'number',
|
|
784
|
+
description: 'Compressed byte length of the record (the capture row\'s "length"). Max 10485760 (10 MB).',
|
|
785
|
+
},
|
|
786
|
+
max_body_bytes: {
|
|
787
|
+
type: 'number',
|
|
788
|
+
description:
|
|
789
|
+
'Truncate the decoded page body to this many bytes before returning it. Default 20000, max 200000. WARC headers and HTTP headers are always returned in full.',
|
|
790
|
+
},
|
|
791
|
+
},
|
|
792
|
+
required: ['filename', 'offset', 'length'],
|
|
793
|
+
},
|
|
794
|
+
},
|
|
795
|
+
];
|
|
796
|
+
|
|
797
|
+
function clamp(v: unknown, def: number, min: number, max: number): number {
|
|
798
|
+
const n = typeof v === 'number' ? v : Number(v);
|
|
799
|
+
if (!Number.isFinite(n)) return def;
|
|
800
|
+
return Math.min(max, Math.max(min, Math.trunc(n)));
|
|
801
|
+
}
|
|
802
|
+
|
|
803
|
+
/** "20250112195924" -> "2025-01-12T19:59:24Z" */
|
|
804
|
+
function isoFromCdxTimestamp(ts: string): string | null {
|
|
805
|
+
if (!/^\d{14}$/.test(ts)) return null;
|
|
806
|
+
return `${ts.slice(0, 4)}-${ts.slice(4, 6)}-${ts.slice(6, 8)}T${ts.slice(8, 10)}:${ts.slice(10, 12)}:${ts.slice(12, 14)}Z`;
|
|
807
|
+
}
|
|
808
|
+
|
|
809
|
+
async function handleCrawls(args: Record<string, unknown>): Promise<unknown> {
|
|
810
|
+
const limit = clamp(args.limit, 20, 1, 120);
|
|
811
|
+
const containsDate = typeof args.contains_date === 'string' ? args.contains_date : undefined;
|
|
812
|
+
|
|
813
|
+
let crawls = await listCrawls();
|
|
814
|
+
const total = crawls.length;
|
|
815
|
+
|
|
816
|
+
if (containsDate) {
|
|
817
|
+
const t = Date.parse(containsDate);
|
|
818
|
+
if (Number.isNaN(t)) {
|
|
819
|
+
throw new Error(`contains_date "${containsDate}" is not a parseable date — pass something like "2025-01-15".`);
|
|
820
|
+
}
|
|
821
|
+
crawls = crawls.filter((c) => {
|
|
822
|
+
const from = c.from ? Date.parse(c.from) : NaN;
|
|
823
|
+
const to = c.to ? Date.parse(c.to) : NaN;
|
|
824
|
+
return !Number.isNaN(from) && !Number.isNaN(to) && t >= from && t <= to;
|
|
825
|
+
});
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
return {
|
|
829
|
+
source: 'Common Crawl Foundation — index.commoncrawl.org/collinfo.json',
|
|
830
|
+
total_crawls: total,
|
|
831
|
+
matched: crawls.length,
|
|
832
|
+
crawls: crawls.slice(0, limit).map((c) => ({
|
|
833
|
+
crawl: c.id,
|
|
834
|
+
name: c.name,
|
|
835
|
+
first_capture: c.from ?? null,
|
|
836
|
+
last_capture: c.to ?? null,
|
|
837
|
+
cdx_api: c['cdx-api'],
|
|
838
|
+
})),
|
|
839
|
+
};
|
|
840
|
+
}
|
|
841
|
+
|
|
842
|
+
async function handleIndexSearch(args: Record<string, unknown>): Promise<unknown> {
|
|
843
|
+
const url = typeof args.url === 'string' ? args.url.trim() : '';
|
|
844
|
+
if (!url) throw new Error('`url` is required — pass a URL, host, or domain such as "example.com".');
|
|
845
|
+
|
|
846
|
+
let crawl = typeof args.crawl === 'string' ? args.crawl.trim() : '';
|
|
847
|
+
if (!crawl) {
|
|
848
|
+
const crawls = await listCrawls();
|
|
849
|
+
if (crawls.length === 0) {
|
|
850
|
+
throw new Error('Common Crawl returned an empty crawl list, so no default collection could be chosen.');
|
|
851
|
+
}
|
|
852
|
+
crawl = crawls[0].id;
|
|
853
|
+
}
|
|
854
|
+
|
|
855
|
+
const limit = clamp(args.limit, 20, 1, 200);
|
|
856
|
+
const page = clamp(args.page, 0, 0, 10_000);
|
|
857
|
+
|
|
858
|
+
const qs = new URLSearchParams({ url, output: 'json', limit: String(limit), page: String(page) });
|
|
859
|
+
if (typeof args.match_type === 'string' && args.match_type) qs.set('matchType', args.match_type);
|
|
860
|
+
if (typeof args.filter === 'string' && args.filter) qs.set('filter', args.filter);
|
|
861
|
+
if (typeof args.from === 'string' && args.from) qs.set('from', args.from);
|
|
862
|
+
if (typeof args.to === 'string' && args.to) qs.set('to', args.to);
|
|
863
|
+
|
|
864
|
+
const endpoint = `${INDEX_BASE}/${crawl}-index?${qs.toString()}`;
|
|
865
|
+
const res = await ccFetch(endpoint);
|
|
866
|
+
const text = await res.text();
|
|
867
|
+
|
|
868
|
+
// The CDX index answers "nothing matched" with a 404 and a plain-English
|
|
869
|
+
// sentence, not JSON — report that as an empty result set rather than an
|
|
870
|
+
// upstream error, but say so explicitly so a caller never reads a blank
|
|
871
|
+
// array as "this page was never crawled by anyone".
|
|
872
|
+
if (res.status === 404 || /no captures found/i.test(text)) {
|
|
873
|
+
return {
|
|
874
|
+
source: `Common Crawl CDX index — ${crawl}`,
|
|
875
|
+
crawl,
|
|
876
|
+
query: { url, match_type: args.match_type ?? 'exact', filter: args.filter ?? null },
|
|
877
|
+
captures: [],
|
|
878
|
+
count: 0,
|
|
879
|
+
note: `No captures of "${url}" in ${crawl}. Try another crawl from commoncrawl_crawls, or a broader match_type such as "domain".`,
|
|
880
|
+
};
|
|
881
|
+
}
|
|
882
|
+
if (!res.ok) {
|
|
883
|
+
throw new Error(
|
|
884
|
+
`Common Crawl CDX index returned HTTP ${res.status} for ${crawl}. ${summarizeErrorBody(text)}`.trim(),
|
|
885
|
+
);
|
|
886
|
+
}
|
|
887
|
+
|
|
888
|
+
const captures: Array<Record<string, unknown>> = [];
|
|
889
|
+
for (const line of text.split('\n')) {
|
|
890
|
+
const trimmed = line.trim();
|
|
891
|
+
if (!trimmed) continue;
|
|
892
|
+
let row: CdxRow;
|
|
893
|
+
try {
|
|
894
|
+
row = JSON.parse(trimmed) as CdxRow;
|
|
895
|
+
} catch {
|
|
896
|
+
continue;
|
|
897
|
+
}
|
|
898
|
+
captures.push({
|
|
899
|
+
url: row.url,
|
|
900
|
+
urlkey: row.urlkey,
|
|
901
|
+
timestamp: row.timestamp,
|
|
902
|
+
captured_at: row.timestamp ? isoFromCdxTimestamp(row.timestamp) : null,
|
|
903
|
+
status: row.status,
|
|
904
|
+
mime: row.mime,
|
|
905
|
+
mime_detected: row['mime-detected'] ?? null,
|
|
906
|
+
languages: row.languages ?? null,
|
|
907
|
+
encoding: row.encoding ?? null,
|
|
908
|
+
digest: row.digest,
|
|
909
|
+
redirect: row.redirect ?? null,
|
|
910
|
+
// These three are what commoncrawl_fetch_record takes.
|
|
911
|
+
filename: row.filename,
|
|
912
|
+
offset: row.offset ? Number(row.offset) : null,
|
|
913
|
+
length: row.length ? Number(row.length) : null,
|
|
914
|
+
});
|
|
915
|
+
}
|
|
916
|
+
|
|
917
|
+
return {
|
|
918
|
+
source: `Common Crawl CDX index — ${crawl}`,
|
|
919
|
+
crawl,
|
|
920
|
+
query: {
|
|
921
|
+
url,
|
|
922
|
+
match_type: args.match_type ?? 'exact',
|
|
923
|
+
filter: args.filter ?? null,
|
|
924
|
+
from: args.from ?? null,
|
|
925
|
+
to: args.to ?? null,
|
|
926
|
+
page,
|
|
927
|
+
limit,
|
|
928
|
+
},
|
|
929
|
+
count: captures.length,
|
|
930
|
+
captures,
|
|
931
|
+
next_page: captures.length === limit ? page + 1 : null,
|
|
932
|
+
fetch_hint:
|
|
933
|
+
'Pass a capture\'s filename + offset + length to commoncrawl_fetch_record to read the archived page body.',
|
|
934
|
+
};
|
|
935
|
+
}
|
|
936
|
+
|
|
937
|
+
/** Split a raw WARC record into its three parts without allocating a second copy of the body. */
|
|
938
|
+
function splitWarcRecord(raw: string): { warcHeaders: Record<string, string>; httpStatusLine: string | null; httpHeaders: Record<string, string>; body: string } {
|
|
939
|
+
const warcEnd = raw.indexOf('\r\n\r\n');
|
|
940
|
+
const warcBlock = warcEnd === -1 ? raw : raw.slice(0, warcEnd);
|
|
941
|
+
const rest = warcEnd === -1 ? '' : raw.slice(warcEnd + 4);
|
|
942
|
+
|
|
943
|
+
const parseHeaders = (block: string): Record<string, string> => {
|
|
944
|
+
const out: Record<string, string> = {};
|
|
945
|
+
for (const line of block.split('\r\n')) {
|
|
946
|
+
const i = line.indexOf(':');
|
|
947
|
+
if (i > 0) out[line.slice(0, i).trim()] = line.slice(i + 1).trim();
|
|
948
|
+
}
|
|
949
|
+
return out;
|
|
950
|
+
};
|
|
951
|
+
|
|
952
|
+
const warcHeaders = parseHeaders(warcBlock);
|
|
953
|
+
|
|
954
|
+
// For a "response" record the block after the WARC headers is a raw HTTP
|
|
955
|
+
// response: status line, headers, blank line, body.
|
|
956
|
+
const httpEnd = rest.indexOf('\r\n\r\n');
|
|
957
|
+
if (httpEnd === -1 || !/^HTTP\/\d/.test(rest)) {
|
|
958
|
+
return { warcHeaders, httpStatusLine: null, httpHeaders: {}, body: rest };
|
|
959
|
+
}
|
|
960
|
+
const httpBlock = rest.slice(0, httpEnd);
|
|
961
|
+
const body = rest.slice(httpEnd + 4);
|
|
962
|
+
const nl = httpBlock.indexOf('\r\n');
|
|
963
|
+
const httpStatusLine = nl === -1 ? httpBlock : httpBlock.slice(0, nl);
|
|
964
|
+
const httpHeaders = parseHeaders(nl === -1 ? '' : httpBlock.slice(nl + 2));
|
|
965
|
+
return { warcHeaders, httpStatusLine, httpHeaders, body };
|
|
966
|
+
}
|
|
967
|
+
|
|
968
|
+
async function handleFetchRecord(args: Record<string, unknown>): Promise<unknown> {
|
|
969
|
+
const filename = typeof args.filename === 'string' ? args.filename.trim() : '';
|
|
970
|
+
if (!filename) {
|
|
971
|
+
throw new Error('`filename` is required — use the "filename" field from a commoncrawl_index_search capture.');
|
|
972
|
+
}
|
|
973
|
+
if (filename.includes('..') || filename.startsWith('/') || filename.includes('://')) {
|
|
974
|
+
throw new Error('`filename` must be a WARC path relative to data.commoncrawl.org, e.g. "crawl-data/CC-MAIN-.../*.warc.gz".');
|
|
975
|
+
}
|
|
976
|
+
const offset = clamp(args.offset, -1, 0, Number.MAX_SAFE_INTEGER);
|
|
977
|
+
const length = clamp(args.length, -1, 1, 10_485_760);
|
|
978
|
+
if (offset < 0 || length < 1) {
|
|
979
|
+
throw new Error('`offset` and `length` are required and come from the same capture row as `filename`.');
|
|
980
|
+
}
|
|
981
|
+
const maxBody = clamp(args.max_body_bytes, 20_000, 500, 200_000);
|
|
982
|
+
|
|
983
|
+
const recordUrl = `${DATA_BASE}/${filename}`;
|
|
984
|
+
const res = await ccFetch(recordUrl, { headers: { Range: `bytes=${offset}-${offset + length - 1}` } });
|
|
985
|
+
if (res.status !== 206 && res.status !== 200) {
|
|
986
|
+
throw new Error(
|
|
987
|
+
`Common Crawl data store returned HTTP ${res.status} for a range read of ${filename}. ` +
|
|
988
|
+
'Check that filename/offset/length came from the same capture row.',
|
|
989
|
+
);
|
|
990
|
+
}
|
|
991
|
+
if (!res.body) throw new Error('Common Crawl data store returned an empty body for that byte range.');
|
|
992
|
+
|
|
993
|
+
// Each indexed record is its own gzip member, so the range read decompresses
|
|
994
|
+
// on its own. DecompressionStream is available in the Workers runtime.
|
|
995
|
+
const stream = res.body.pipeThrough(new DecompressionStream('gzip'));
|
|
996
|
+
const decompressed = new Uint8Array(await new Response(stream).arrayBuffer());
|
|
997
|
+
const raw = new TextDecoder('utf-8').decode(decompressed);
|
|
998
|
+
|
|
999
|
+
const { warcHeaders, httpStatusLine, httpHeaders, body } = splitWarcRecord(raw);
|
|
1000
|
+
const bodyBytes = new TextEncoder().encode(body).length;
|
|
1001
|
+
const truncated = bodyBytes > maxBody;
|
|
1002
|
+
|
|
1003
|
+
return {
|
|
1004
|
+
source: 'Common Crawl Foundation — data.commoncrawl.org (WARC)',
|
|
1005
|
+
warc_url: recordUrl,
|
|
1006
|
+
byte_range: { offset, length },
|
|
1007
|
+
record_bytes_decompressed: decompressed.length,
|
|
1008
|
+
target_uri: warcHeaders['WARC-Target-URI'] ?? null,
|
|
1009
|
+
warc_type: warcHeaders['WARC-Type'] ?? null,
|
|
1010
|
+
warc_date: warcHeaders['WARC-Date'] ?? null,
|
|
1011
|
+
warc_ip_address: warcHeaders['WARC-IP-Address'] ?? null,
|
|
1012
|
+
payload_digest: warcHeaders['WARC-Payload-Digest'] ?? null,
|
|
1013
|
+
identified_payload_type: warcHeaders['WARC-Identified-Payload-Type'] ?? null,
|
|
1014
|
+
warc_headers: warcHeaders,
|
|
1015
|
+
http_status_line: httpStatusLine,
|
|
1016
|
+
http_headers: httpHeaders,
|
|
1017
|
+
body_bytes: bodyBytes,
|
|
1018
|
+
body_truncated: truncated,
|
|
1019
|
+
body: truncated ? body.slice(0, maxBody) : body,
|
|
1020
|
+
};
|
|
1021
|
+
}
|
|
1022
|
+
|
|
1023
|
+
async function callTool(name: string, args: Record<string, unknown>): Promise<unknown> {
|
|
1024
|
+
switch (name) {
|
|
1025
|
+
case 'commoncrawl_crawls':
|
|
1026
|
+
return handleCrawls(args);
|
|
1027
|
+
case 'commoncrawl_index_search':
|
|
1028
|
+
return handleIndexSearch(args);
|
|
1029
|
+
case 'commoncrawl_fetch_record':
|
|
1030
|
+
return handleFetchRecord(args);
|
|
1031
|
+
default:
|
|
1032
|
+
throw new Error(`Unknown tool: ${name}`);
|
|
1033
|
+
}
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
export default { tools, callTool, meter: { credits: 1 } } satisfies McpToolExport;
|
package/src/server.ts
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Stdio MCP server entry point for @pipeworx/mcp-commoncrawl.
|
|
3
|
+
* Generated by scripts/publish-pack.sh — do not hand-edit in the pack repo;
|
|
4
|
+
* edit scripts/publish-pack.sh (the server.ts heredoc) and republish instead.
|
|
5
|
+
*/
|
|
6
|
+
import { Server } from '@modelcontextprotocol/sdk/server/index.js';
|
|
7
|
+
import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
|
|
8
|
+
import { CallToolRequestSchema, ListToolsRequestSchema } from '@modelcontextprotocol/sdk/types.js';
|
|
9
|
+
import pack from './index.js';
|
|
10
|
+
|
|
11
|
+
const server = new Server(
|
|
12
|
+
{ name: '@pipeworx/mcp-commoncrawl', version: '0.1.0' },
|
|
13
|
+
{ capabilities: { tools: {} } },
|
|
14
|
+
);
|
|
15
|
+
|
|
16
|
+
server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
17
|
+
tools: pack.tools.map((t) => ({
|
|
18
|
+
name: t.name,
|
|
19
|
+
description: t.description,
|
|
20
|
+
inputSchema: t.inputSchema,
|
|
21
|
+
})),
|
|
22
|
+
}));
|
|
23
|
+
|
|
24
|
+
server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
25
|
+
const { name, arguments: args } = request.params;
|
|
26
|
+
try {
|
|
27
|
+
const result = await pack.callTool(name, (args ?? {}) as Record<string, unknown>);
|
|
28
|
+
return { content: [{ type: 'text', text: JSON.stringify(result, null, 2) }] };
|
|
29
|
+
} catch (err) {
|
|
30
|
+
return {
|
|
31
|
+
content: [{ type: 'text', text: err instanceof Error ? err.message : String(err) }],
|
|
32
|
+
isError: true,
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
async function main() {
|
|
38
|
+
const transport = new StdioServerTransport();
|
|
39
|
+
await server.connect(transport);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
main().catch((err) => {
|
|
43
|
+
console.error('Fatal error running server:', err);
|
|
44
|
+
process.exit(1);
|
|
45
|
+
});
|
package/tsconfig.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"compilerOptions": {
|
|
3
|
+
"target": "ES2022",
|
|
4
|
+
"module": "ESNext",
|
|
5
|
+
"moduleResolution": "bundler",
|
|
6
|
+
"lib": ["ES2022"],
|
|
7
|
+
"types": ["@cloudflare/workers-types"],
|
|
8
|
+
"strict": true,
|
|
9
|
+
"esModuleInterop": true,
|
|
10
|
+
"skipLibCheck": true,
|
|
11
|
+
"resolveJsonModule": true,
|
|
12
|
+
"outDir": "dist",
|
|
13
|
+
"rootDir": "src",
|
|
14
|
+
"declaration": true
|
|
15
|
+
},
|
|
16
|
+
"include": ["src"],
|
|
17
|
+
"exclude": ["src/server.ts"]
|
|
18
|
+
}
|