@pipeworx/mcp-legislation-uk 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +1 -1
- package/README.md +99 -8
- package/bin/cli.js +17 -0
- package/package.json +15 -3
- package/server.json +1 -1
- package/src/index.ts +1433 -8
- package/src/server.ts +45 -0
- package/tsconfig.json +5 -1
package/src/index.ts
CHANGED
|
@@ -1,11 +1,19 @@
|
|
|
1
1
|
interface McpToolDefinition {
|
|
2
2
|
name: string;
|
|
3
3
|
description: string;
|
|
4
|
+
/** Human-facing one-liner (fleet #1967). Optional; consumers fall back to
|
|
5
|
+
* description. Kept in step with shared/src/types.ts — scripts/lib/
|
|
6
|
+
* check-inlined-types.mjs reports drift at publish time. */
|
|
7
|
+
summary?: string;
|
|
4
8
|
inputSchema: {
|
|
5
9
|
type: 'object';
|
|
6
10
|
properties: Record<string, unknown>;
|
|
7
11
|
required?: string[];
|
|
12
|
+
anyOf?: Array<{ required: string[] }>;
|
|
13
|
+
oneOf?: Array<{ required: string[] }>;
|
|
14
|
+
allOf?: Array<{ required: string[] }>;
|
|
8
15
|
};
|
|
16
|
+
outputSchema?: Record<string, unknown>;
|
|
9
17
|
}
|
|
10
18
|
|
|
11
19
|
interface McpToolExport {
|
|
@@ -16,6 +24,652 @@ interface McpToolExport {
|
|
|
16
24
|
provider?: string;
|
|
17
25
|
}
|
|
18
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Was this failure OUR OWN web service? — the other half of `internal-db-class.ts`.
|
|
29
|
+
*
|
|
30
|
+
* fleet #1089 pulled failures from our own Postgres out of `upstream_down` by
|
|
31
|
+
* keying on the SQLSTATE inside PostgREST's four-key error envelope. That
|
|
32
|
+
* covered the majority and structurally could not cover the rest: the rest
|
|
33
|
+
* never reach Postgres, so they carry no SQLSTATE. What was left, measured over
|
|
34
|
+
* the 24h to 2026-09-02T15:00Z (fleet #1096):
|
|
35
|
+
*
|
|
36
|
+
* 5 pipeworx-catalog get_pack_tools Pipeworx catalog error: 522 — error code: 522
|
|
37
|
+
* 3 fleet fleet_list_open … upstream_down: Fleet task queue did not respond within 25s
|
|
38
|
+
*
|
|
39
|
+
* 521/522/523/526 are Cloudflare saying its edge could not reach an ORIGIN, and
|
|
40
|
+
* in both of those rows the origin is ours — `gateway.pipeworx.io` for the
|
|
41
|
+
* catalog pack (it self-fetches when the gateway hasn't injected a manifest),
|
|
42
|
+
* our own Supabase for fleet. There is no third party anywhere in either call.
|
|
43
|
+
* Same defect as #1089: our own outage filed under `upstream_down`, the one
|
|
44
|
+
* class that means "the source is unreachable and there is nothing for us to
|
|
45
|
+
* fix", which is why the problem-tools triage skips it.
|
|
46
|
+
*
|
|
47
|
+
* WHY NOT A WORDING RULE. The obvious fix is to match `fleet db error:` and
|
|
48
|
+
* `Pipeworx catalog error:` in classifyToolError. Each is emitted from exactly
|
|
49
|
+
* one site today, so it would work today. It would also rot the first time
|
|
50
|
+
* somebody rewords a label — silently, and in the direction of hiding our own
|
|
51
|
+
* outage, which is worse than the bug being fixed. Every prose rule in
|
|
52
|
+
* error-class.ts has needed widening as packs invented new wording (#409/#450/
|
|
53
|
+
* #584); that history is most of that file's comment budget.
|
|
54
|
+
*
|
|
55
|
+
* WHAT THIS KEYS ON INSTEAD: **the host the call actually reached.** A URL's
|
|
56
|
+
* hostname is a fact about the call, not a guess about its prose. Two
|
|
57
|
+
* consequences that a pack-level flag could not give us, and the reason the
|
|
58
|
+
* flag was rejected:
|
|
59
|
+
*
|
|
60
|
+
* - It describes the CALL, not the pack. `govcon-intel` fans out to our own
|
|
61
|
+
* Supabase AND to genuine third parties; `court-listener` holds our cache
|
|
62
|
+
* in Supabase and fetches courtlistener.com. An `internallyHosted: true` on
|
|
63
|
+
* either pack would relabel a real third-party outage as ours — inventing
|
|
64
|
+
* work, which is the same class of error in the opposite direction.
|
|
65
|
+
* - It covers every future internal pack for free, instead of one declared
|
|
66
|
+
* slug at a time.
|
|
67
|
+
*
|
|
68
|
+
* WHY IT SURVIVES A REWORD. The marker below is not matched as a literal by two
|
|
69
|
+
* separate files. `markInternalOrigin()` writes it and `internalHostMetricsClass()`
|
|
70
|
+
* reads it, both from the single exported `INTERNAL_ORIGIN_MARKER` constant in
|
|
71
|
+
* this module — so changing the wording changes both sides in the same edit and
|
|
72
|
+
* cannot desynchronise them. The pack's own label (`fleet db error:`,
|
|
73
|
+
* `Pipeworx catalog error:`) is not read at all: reword it freely, the class is
|
|
74
|
+
* unaffected. That is the property `stripClassPrefix` lacked when it drifted
|
|
75
|
+
* from its own classifier three times and needed a CI gate to hold them
|
|
76
|
+
* together.
|
|
77
|
+
*
|
|
78
|
+
* WHERE THE 5xx TEST LIVES. `markInternalOrigin` is called from the places that
|
|
79
|
+
* hold the real `Response` — `httpError`/`httpErrorMessage` and the timeout
|
|
80
|
+
* branch of `fetchWithTimeout` in `shared/src/http.ts` — so "is this an
|
|
81
|
+
* availability failure" is decided from the actual status code, never re-derived
|
|
82
|
+
* by scraping a number out of a sentence. A 404 from our own registry for a slug
|
|
83
|
+
* that does not exist is a caller's bad argument and is deliberately NOT marked.
|
|
84
|
+
*/
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* OUR OWN web service was unreachable — not an upstream, and never `upstream_down`.
|
|
88
|
+
*
|
|
89
|
+
* ONE value, not three, unlike `internal_db_*`. That split existed because a
|
|
90
|
+
* slow query, an exhausted pool and an unknown SQLSTATE have different owners
|
|
91
|
+
* and different fixes. Here there is only one story to tell — an origin we run
|
|
92
|
+
* did not answer the edge — and one owner. A bucket with no distinct owner per
|
|
93
|
+
* value is decoration; #724 is what happens when a class holds several
|
|
94
|
+
* situations, and inventing sub-values ahead of a reason to act on them
|
|
95
|
+
* differently is the same mistake with the sign flipped.
|
|
96
|
+
*
|
|
97
|
+
* METRICS ONLY, exactly like PLATFORM_KEY_ERROR_CLASS and the internal_db
|
|
98
|
+
* values. `classifyToolError` still answers `upstream_down` for the retry and
|
|
99
|
+
* hint paths, which only care whether retrying or a sibling tool might work —
|
|
100
|
+
* and it might. Nothing a caller sees or is charged changes here.
|
|
101
|
+
*
|
|
102
|
+
* READ SIDE: this value is in BROKEN_TOOL_CLASSES, FAULT_CLASSES and
|
|
103
|
+
* ALL_ERROR_CLASSES in `workers/registry-api/src/index.ts`. All three, or it
|
|
104
|
+
* lands on no dashboard — fleet #721 is the warning, where the #719 split
|
|
105
|
+
* worked on the write side and was invisible for weeks.
|
|
106
|
+
*/
|
|
107
|
+
const INTERNAL_SERVICE_UNREACHABLE_CLASS = 'internal_service_unreachable';
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* The token that carries "this origin is ours" from the call site to the
|
|
111
|
+
* classifier.
|
|
112
|
+
*
|
|
113
|
+
* Appended to the error message rather than attached to the Error object,
|
|
114
|
+
* because the object does not survive the trip: 275 packs return `{ error:
|
|
115
|
+
* string }` instead of throwing, the gateway reads `observedError` as a string,
|
|
116
|
+
* and the fleet pack rebuilds its error from a captured status + body across a
|
|
117
|
+
* retry loop. A property on an Error would be dropped by every one of those
|
|
118
|
+
* paths and the class would work in tests and vanish in production.
|
|
119
|
+
*
|
|
120
|
+
* WORDING IS LOAD-BEARING, same rule as labelAge's note in authority.ts. This
|
|
121
|
+
* string is appended to a pack's thrown Error message (shared/src/http.ts),
|
|
122
|
+
* and a thrown Error's message is exactly what the gateway hands back to the
|
|
123
|
+
* caller as `content[0].text` when nothing rewrites it (workers/gateway/src
|
|
124
|
+
* catches the throw and sets `rawResult.message = stripClassPrefix(error)`,
|
|
125
|
+
* which does not touch this suffix) — so the original wording,
|
|
126
|
+
* " [pipeworx-hosted origin — our own service, not a third party]", was not a
|
|
127
|
+
* theoretical leak: it shipped live on pipeworx-catalog's 522s, 7 times in 6
|
|
128
|
+
* hours on 2026-09-02 (see tests/golden-internal-service.test.ts), verbatim
|
|
129
|
+
* naming Pipeworx as the host. check:hosting-claims never caught it because it
|
|
130
|
+
* did not scan shared/ at all (task #2009). Reworded to describe the
|
|
131
|
+
* OBSERVATION (the origin did not answer) without a claim about who runs it —
|
|
132
|
+
* the identical fix labelAge got: drop the possessive, keep the fact.
|
|
133
|
+
*/
|
|
134
|
+
const INTERNAL_ORIGIN_MARKER = ' [origin did not respond — retry before concluding the named source is down]';
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* Supabase's data plane for a project is `<ref>.supabase.co`, where the ref is
|
|
138
|
+
* exactly twenty lowercase letters (ours is `pqauisounztsgdgfkhke`).
|
|
139
|
+
*
|
|
140
|
+
* Matching the shape rather than listing the ref keeps this correct when we add
|
|
141
|
+
* a project — `supabaseEnv` on a pack entry already points some packs at a
|
|
142
|
+
* second one — while still excluding `status.supabase.co`, which is Supabase's
|
|
143
|
+
* own status page and emphatically not our database. Verified 2026-09-02 by
|
|
144
|
+
* `grep -rhoE '[a-z0-9-]+\.supabase\.(co|in)' mcps shared workers scripts`: the
|
|
145
|
+
* only real project ref anywhere in the tree is ours, the rest are doc
|
|
146
|
+
* placeholders (`abc`, `xyz`, `example`) which this pattern also excludes. Same
|
|
147
|
+
* finding internal-db-class.ts relies on for the PostgREST envelope being ours
|
|
148
|
+
* by construction.
|
|
149
|
+
*/
|
|
150
|
+
const SUPABASE_PROJECT_HOST = /^[a-z]{20}\.supabase\.(co|in)$/;
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Is this a host WE run?
|
|
154
|
+
*
|
|
155
|
+
* Deliberately NOT including `*.workers.dev`: plenty of third-party APIs are
|
|
156
|
+
* hosted on workers.dev, so the suffix says where something runs and not who
|
|
157
|
+
* owns it. Every internal call we actually make goes to a `pipeworx.io`
|
|
158
|
+
* hostname or to our Supabase project, both of which are ownership facts.
|
|
159
|
+
*
|
|
160
|
+
* `workers/gateway/src/provenance.ts`'s `OUR_HOSTS` answers the same
|
|
161
|
+
* question and DOES include `workers.dev` — a documented divergence
|
|
162
|
+
* (task #2051), not a bug to converge. That list decides what a response may
|
|
163
|
+
* cite as a data SOURCE, where a false negative (citing our own worker as an
|
|
164
|
+
* external source) is the hosting-disclosure leak this whole file exists to
|
|
165
|
+
* prevent, so it errs broad. This one decides who gets BLAMED for a 5xx in
|
|
166
|
+
* outage metrics read by on-call, where a false positive (crediting our own
|
|
167
|
+
* infra with a third party's outage) hides the real failure, so it errs
|
|
168
|
+
* narrow. Same suffix, opposite direction, because they are never called for
|
|
169
|
+
* the same reason.
|
|
170
|
+
*
|
|
171
|
+
* Returns false on anything unparseable rather than throwing — this runs inside
|
|
172
|
+
* an error path, and an error path that can itself throw turns a diagnosable
|
|
173
|
+
* failure into a mystery.
|
|
174
|
+
*/
|
|
175
|
+
function isPipeworxOrigin(url: string | URL | undefined | null): boolean {
|
|
176
|
+
if (!url) return false;
|
|
177
|
+
let host: string;
|
|
178
|
+
try {
|
|
179
|
+
host = new URL(url instanceof URL ? url.href : url).hostname.toLowerCase();
|
|
180
|
+
} catch {
|
|
181
|
+
return false;
|
|
182
|
+
}
|
|
183
|
+
if (host === 'pipeworx.io' || host.endsWith('.pipeworx.io')) return true;
|
|
184
|
+
return SUPABASE_PROJECT_HOST.test(host);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Append the marker when this failure was OUR origin failing to answer.
|
|
189
|
+
*
|
|
190
|
+
* `status` is the HTTP status when there is one, and omitted for a timeout —
|
|
191
|
+
* where there is no response at all, and "the origin did not answer" is the
|
|
192
|
+
* whole observation. Statuses below 500 are left alone: a 404 from our own
|
|
193
|
+
* registry for a slug that does not exist is the caller's argument, not our
|
|
194
|
+
* outage, and marking it would put ordinary 404s on the incident dashboard.
|
|
195
|
+
*
|
|
196
|
+
* Idempotent, so a message that is wrapped and re-marked on the way up (the
|
|
197
|
+
* fleet pack's retry loop re-throws through two layers) carries the marker once.
|
|
198
|
+
*/
|
|
199
|
+
function markInternalOrigin(
|
|
200
|
+
message: string,
|
|
201
|
+
url: string | URL | undefined | null,
|
|
202
|
+
status?: number,
|
|
203
|
+
): string {
|
|
204
|
+
if (status !== undefined && status < 500) return message;
|
|
205
|
+
if (!isPipeworxOrigin(url)) return message;
|
|
206
|
+
if (message.includes(INTERNAL_ORIGIN_MARKER)) return message;
|
|
207
|
+
return message + INTERNAL_ORIGIN_MARKER;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Which blob4 value a failure from our own web services books as, or undefined
|
|
212
|
+
* if this is not one.
|
|
213
|
+
*
|
|
214
|
+
* Ordered AFTER `internalDbMetricsClass` at the call site: a PostgREST envelope
|
|
215
|
+
* from our own Supabase is a strictly more specific statement about the same
|
|
216
|
+
* row (which of our services, and why), and the two cannot disagree about
|
|
217
|
+
* whether the failure is ours.
|
|
218
|
+
*/
|
|
219
|
+
function internalHostMetricsClass(error: string): string | undefined {
|
|
220
|
+
return error.includes(INTERNAL_ORIGIN_MARKER) ? INTERNAL_SERVICE_UNREACHABLE_CLASS : undefined;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* One place to turn a failed `fetch` into an error a caller can act on.
|
|
226
|
+
*
|
|
227
|
+
* Nearly every pack was written the same way:
|
|
228
|
+
*
|
|
229
|
+
* if (!res.ok) throw new Error(`Unsplash: ${res.status}`);
|
|
230
|
+
*
|
|
231
|
+
* which discards the response body — and the body is usually where the upstream
|
|
232
|
+
* says what was actually wrong ("**symbol** not found: GBP", "parameter `year`
|
|
233
|
+
* out of range", "unknown taxonomy id"). The caller gets a number, cannot
|
|
234
|
+
* self-correct, and retries the same broken call. A 2026-07-31 sweep found this
|
|
235
|
+
* shape in 481 of 1,400 packs, 47 of them PLATFORM-keyed.
|
|
236
|
+
*
|
|
237
|
+
* It also hides bugs one level down. Two of the first three packs audited had a
|
|
238
|
+
* second defect that only existed because of this line: unsplash's rate-limit
|
|
239
|
+
* branch sat BELOW a catch-all and was unreachable, and bea-gov parsed
|
|
240
|
+
* `BEAAPI.Error.APIErrorDescription` below a `!res.ok` throw that made the
|
|
241
|
+
* parsing dead code for every non-200.
|
|
242
|
+
*
|
|
243
|
+
* DELIBERATELY NOT A CLASSIFIER. It does not add `user_error:` /
|
|
244
|
+
* `upstream_down:` prefixes. Those decide which tier a failure lands in, and the
|
|
245
|
+
* `error` tier is what the daily problem-tools list is built from — it means
|
|
246
|
+
* "Pipeworx has a defect". A 400 is genuinely ambiguous: often a caller's bad
|
|
247
|
+
* argument, but sometimes a query WE built wrong (ted-eu comma-joined its CPV
|
|
248
|
+
* values into something TED rejected, and that bug was found only because it sat
|
|
249
|
+
* in `error`). Blanket-classifying 400s as caller mistakes would have hidden it.
|
|
250
|
+
* A pack that KNOWS which it is should keep saying so explicitly; this helper is
|
|
251
|
+
* for the 481 that say nothing at all.
|
|
252
|
+
*/
|
|
253
|
+
|
|
254
|
+
/** Longest upstream explanation we'll pass through. Enough for a real message,
|
|
255
|
+
* short enough that an HTML page or a stack trace can't swamp the error. */
|
|
256
|
+
|
|
257
|
+
const MAX_DETAIL = 300;
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Default bound for `fetchWithTimeout` when a pack doesn't state its own.
|
|
261
|
+
*
|
|
262
|
+
* 25s mirrors the number `epo-ops` landed on after measuring the real failure:
|
|
263
|
+
* a degraded upstream that doesn't error, it just never answers, and a Worker
|
|
264
|
+
* sits in `await fetch()` until ITS OWN execution budget kills the request —
|
|
265
|
+
* which can take minutes, not seconds (epo_ops_search_patents measured 4-8
|
|
266
|
+
* MINUTE hangs before this existed). 25s is short enough that a caller gets a
|
|
267
|
+
* fast, actionable error instead of holding the connection, and long enough
|
|
268
|
+
* that it doesn't false-trip on a merely-slow-but-alive upstream.
|
|
269
|
+
*/
|
|
270
|
+
const DEFAULT_FETCH_TIMEOUT_MS = 25_000;
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Read the body of a failed response and fold it into a throwable Error.
|
|
274
|
+
*
|
|
275
|
+
* Usage — note the `await`, which is the one thing that makes this a mechanical
|
|
276
|
+
* change rather than a drop-in:
|
|
277
|
+
*
|
|
278
|
+
* if (!res.ok) throw await httpError(res, 'Unsplash');
|
|
279
|
+
*
|
|
280
|
+
* Safe to call on any non-ok response: a body that is missing, empty, unreadable
|
|
281
|
+
* or HTML degrades to exactly the old `Name: 404` string rather than throwing
|
|
282
|
+
* something new from inside the error path.
|
|
283
|
+
*/
|
|
284
|
+
async function httpError(res: Response, name: string): Promise<Error> {
|
|
285
|
+
return new Error(await httpErrorMessage(res, name));
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** The message text without constructing an Error — for packs that need to wrap
|
|
289
|
+
* it in their own envelope or add an explicit classification prefix. */
|
|
290
|
+
async function httpErrorMessage(res: Response, name: string): Promise<string> {
|
|
291
|
+
// The one place a 5xx from a host WE run gets stamped as ours. `res.url` is
|
|
292
|
+
// the URL the fetch actually resolved to (after redirects), so this is a fact
|
|
293
|
+
// about the call rather than a guess from the `name` the pack passed in —
|
|
294
|
+
// reword that label freely, the class does not move. See
|
|
295
|
+
// internal-host-class.ts; no-op for every third-party upstream, which is why
|
|
296
|
+
// this touches 481 packs' error text and changes none of it.
|
|
297
|
+
return markInternalOrigin(
|
|
298
|
+
`${name}: ${res.status}${detailSuffix(await readDetail(res))}`,
|
|
299
|
+
res.url,
|
|
300
|
+
res.status,
|
|
301
|
+
);
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* Just the upstream's own explanation — no name, no status.
|
|
306
|
+
*
|
|
307
|
+
* For a pack that has already said both in its own sentence. epo-ops reads
|
|
308
|
+
* `EPO rejected this search as too large (HTTP 413) — ${httpErrorMessage(…)}`,
|
|
309
|
+
* which rendered as `… (HTTP 413) — EPO: 413.` once the XML detail was being
|
|
310
|
+
* dropped: the upstream named twice, the status twice, and the one thing EPO
|
|
311
|
+
* actually said ("Not enough characters before truncation character") nowhere
|
|
312
|
+
* (fleet #712). Returns '' when the body carries nothing readable, so a caller
|
|
313
|
+
* can fall back to its own wording.
|
|
314
|
+
*/
|
|
315
|
+
async function upstreamDetail(res: Response): Promise<string> {
|
|
316
|
+
return readDetail(res);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Read a SUCCESSFUL response as JSON, failing loudly when it isn't JSON.
|
|
321
|
+
*
|
|
322
|
+
* `httpError` above only ever runs on `!res.ok`, which leaves the nastier half
|
|
323
|
+
* of the problem unhandled: an upstream that answers **HTTP 200 with an HTML
|
|
324
|
+
* page**. A bot wall, a login redirect, a maintenance interstitial and a CDN
|
|
325
|
+
* error page are all 200s, so `res.ok` is true, and `res.json()` then throws
|
|
326
|
+
* `Unexpected token '<', "<!DOCTYPE "... is not valid JSON`.
|
|
327
|
+
*
|
|
328
|
+
* That string is the problem. It names no upstream, carries no status, and
|
|
329
|
+
* reads like a parser bug in Pipeworx — so it lands in the `error` tier, which
|
|
330
|
+
* means "we have a defect", and the caller is told nothing they can act on.
|
|
331
|
+
* data.govt.nz sat dead behind an Imperva challenge this way and every
|
|
332
|
+
* status-code health check we own reported it green (7889a845). A zero-length
|
|
333
|
+
* body has the same shape: `Unexpected end of JSON input`, seen this week on
|
|
334
|
+
* uk-gazette (83% of external calls) and census.
|
|
335
|
+
*
|
|
336
|
+
* UNLIKE `httpError`, this one DOES classify, and the asymmetry is deliberate.
|
|
337
|
+
* A 400 is genuinely ambiguous — often the caller's bad argument, sometimes a
|
|
338
|
+
* query we built wrong — so blanket-classifying it would hide our own bugs.
|
|
339
|
+
* There is no such ambiguity here: **no argument a caller can pass makes a JSON
|
|
340
|
+
* API return an HTML page.** It is always the upstream, so `upstream_down:` is
|
|
341
|
+
* a statement of fact rather than a guess, and it keeps these out of the
|
|
342
|
+
* problem-tools list where they crowd out real defects.
|
|
343
|
+
*
|
|
344
|
+
* const data = await parseJson<Feed>(res, 'UK Gazette');
|
|
345
|
+
*
|
|
346
|
+
* Call it only after the `!res.ok` check — on a failed response you want
|
|
347
|
+
* `httpError`, which mines the body for the upstream's own explanation.
|
|
348
|
+
*/
|
|
349
|
+
async function parseJson<T>(res: Response, name: string): Promise<T> {
|
|
350
|
+
let raw: string;
|
|
351
|
+
try {
|
|
352
|
+
raw = await res.text();
|
|
353
|
+
} catch {
|
|
354
|
+
throw new Error(
|
|
355
|
+
`upstream_down: ${name} returned a body that could not be read (HTTP ${res.status}). ` +
|
|
356
|
+
'The connection most likely dropped mid-response; retrying is reasonable.',
|
|
357
|
+
);
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
const type = res.headers.get('content-type') ?? 'no content-type';
|
|
361
|
+
|
|
362
|
+
if (!raw.trim()) {
|
|
363
|
+
throw new Error(
|
|
364
|
+
`upstream_down: ${name} answered HTTP ${res.status} with an EMPTY body where JSON was expected (${type}). ` +
|
|
365
|
+
'Nothing about the request can cause this — it is an upstream fault, and the same call may well work on retry.',
|
|
366
|
+
);
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
// Checked before parsing rather than in the catch, because knowing it is
|
|
370
|
+
// markup is what turns "we failed to parse something" into "they served a
|
|
371
|
+
// web page" — the second is diagnosable, the first is not.
|
|
372
|
+
const head = raw.slice(0, 200).trimStart().toLowerCase();
|
|
373
|
+
if (head.startsWith('<!doctype') || head.startsWith('<html') || head.startsWith('<?xml')) {
|
|
374
|
+
const kind = head.startsWith('<?xml') ? 'an XML document' : 'an HTML page';
|
|
375
|
+
// The summary, not the source. Pasting the first 120 characters of a web
|
|
376
|
+
// page handed the agent `<!DOCTYPE html><html lang="en"…` — the same leak
|
|
377
|
+
// this branch exists to describe (fleet #712).
|
|
378
|
+
throw new Error(
|
|
379
|
+
`upstream_down: ${name} answered HTTP ${res.status} with ${kind} instead of JSON (${type}). ` +
|
|
380
|
+
'That is typically a bot wall, a login redirect or a maintenance page — it is returned as a SUCCESS, ' +
|
|
381
|
+
`so status-code health checks read it as fine. No argument change will get past it. ` +
|
|
382
|
+
`The page says: ${summarizeErrorBody(raw) || 'nothing readable'}`,
|
|
383
|
+
);
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
try {
|
|
387
|
+
return JSON.parse(raw) as T;
|
|
388
|
+
} catch {
|
|
389
|
+
throw new Error(
|
|
390
|
+
`upstream_down: ${name} answered HTTP ${res.status} with a body that is not valid JSON (${type}). ` +
|
|
391
|
+
`It begins: ${stripMarkup(raw).slice(0, 120) || '(unreadable)'}`,
|
|
392
|
+
);
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
/**
|
|
397
|
+
* `fetch`, but bounded — the fix for a systemic gap found 2026-08-30: a grep
|
|
398
|
+
* audit of every pack's `mcps/*\/src/index.ts` found 1,339 of ~1,500 call
|
|
399
|
+
* `fetch()` with NO timeout guard anywhere in the file. Two of those
|
|
400
|
+
* (epo-ops, statcan) were confirmed live-hanging for 4-8 minutes before this
|
|
401
|
+
* existed — every unguarded call carries the same risk, just unconfirmed.
|
|
402
|
+
*
|
|
403
|
+
* Mirrors the `epoFetch` wrapper `mcps/epo-ops/src/index.ts` shipped first:
|
|
404
|
+
* bound the request with `AbortSignal.timeout`, and on a timeout/abort throw
|
|
405
|
+
* an `upstream_down:` error that names the upstream and the bound rather than
|
|
406
|
+
* letting the raw `TimeoutError`/`AbortError` (which names neither) propagate.
|
|
407
|
+
* `upstream_down:` is deliberate, same reasoning as `parseJson` above — no
|
|
408
|
+
* argument a caller passes can make an upstream hang, so it is always the
|
|
409
|
+
* upstream's fault, and marking it that way keeps a slow API off the
|
|
410
|
+
* problem-tools list where it would crowd out our own defects.
|
|
411
|
+
*
|
|
412
|
+
* Usage — a mechanical swap for a bare `fetch(url, init)`:
|
|
413
|
+
*
|
|
414
|
+
* const res = await fetchWithTimeout(url, init, 'Some API');
|
|
415
|
+
*
|
|
416
|
+
* Pass `timeoutMs` as a fourth argument to override the default for a pack
|
|
417
|
+
* with a known-slower upstream; the label should be the same short name you'd
|
|
418
|
+
* pass to `httpError`/`httpErrorMessage` for that call.
|
|
419
|
+
*/
|
|
420
|
+
async function fetchWithTimeout(
|
|
421
|
+
url: string | URL,
|
|
422
|
+
init: RequestInit = {},
|
|
423
|
+
name: string,
|
|
424
|
+
timeoutMs: number = DEFAULT_FETCH_TIMEOUT_MS,
|
|
425
|
+
): Promise<Response> {
|
|
426
|
+
try {
|
|
427
|
+
return await fetch(url, { ...init, signal: AbortSignal.timeout(timeoutMs) });
|
|
428
|
+
} catch (err) {
|
|
429
|
+
if (err instanceof Error && (err.name === 'TimeoutError' || err.name === 'AbortError')) {
|
|
430
|
+
// States the OBSERVATION (no response in N seconds), not a diagnosis.
|
|
431
|
+
// "appears to be degraded" is an inference about the vendor that we have
|
|
432
|
+
// not checked, and it is wrong in a way that misdirects whoever reads it:
|
|
433
|
+
// a timeout from a Worker can equally mean OUR egress is blocked.
|
|
434
|
+
//
|
|
435
|
+
// Measured today (2026-09-01, fleet #1047): every call to
|
|
436
|
+
// mainnet.base.org failed from the x402 facilitator while the identical
|
|
437
|
+
// request from a laptop returned 200. Base was entirely healthy; the
|
|
438
|
+
// public RPC refuses Cloudflare Worker egress. Had this message fired
|
|
439
|
+
// there it would have blamed Base by name, and the next person would have
|
|
440
|
+
// waited for a vendor outage to clear that did not exist.
|
|
441
|
+
// A timeout has no status to test — there is no response at all — so
|
|
442
|
+
// `markInternalOrigin` is called without one: an origin we run that never
|
|
443
|
+
// answered is an availability failure by definition. This is the half of
|
|
444
|
+
// fleet #1096 with neither a SQLSTATE nor a status code to key on.
|
|
445
|
+
throw new Error(
|
|
446
|
+
markInternalOrigin(
|
|
447
|
+
`upstream_down: ${name} did not respond within ${timeoutMs / 1000}s. ` +
|
|
448
|
+
`That can be ${name} being slow or down, or this environment being unable to reach it ` +
|
|
449
|
+
`(some hosts refuse datacenter/Worker egress) — retry shortly, and check reachability ` +
|
|
450
|
+
`from elsewhere before concluding ${name} is down.`,
|
|
451
|
+
url,
|
|
452
|
+
),
|
|
453
|
+
);
|
|
454
|
+
}
|
|
455
|
+
// Fleet #2382. Everything that isn't a timeout/abort here is a genuine
|
|
456
|
+
// NETWORK-LEVEL failure — DNS resolution, connection refused, TLS handshake,
|
|
457
|
+
// Cloudflare's own "Network connection lost." — meaning `fetch()` itself
|
|
458
|
+
// threw and no HTTP response of any kind was ever received. Until this fix
|
|
459
|
+
// that raw exception was rethrown VERBATIM: a bare `TypeError: fetch failed`
|
|
460
|
+
// (or the Workers-runtime equivalent) names no upstream, carries no class
|
|
461
|
+
// token, and reads exactly like a defect in OUR code — because it says
|
|
462
|
+
// nothing about the call at all. It landed in `error`, the tier that means
|
|
463
|
+
// "Pipeworx has a defect", for every one of the (at the time of writing)
|
|
464
|
+
// ~470 packs that call this helper directly with no wrapper of their own.
|
|
465
|
+
//
|
|
466
|
+
// `dexscreener` hit this independently (fleet #1579) and fixed it with a
|
|
467
|
+
// bespoke per-pack try/catch around `fetchWithTimeout`. That fix is correct
|
|
468
|
+
// but only covers one pack; every other caller of this shared helper still
|
|
469
|
+
// leaked the raw exception. Moving the same fix HERE — the one place that
|
|
470
|
+
// already carries the timeout case — covers every pack that uses
|
|
471
|
+
// `fetchWithTimeout` without a wrapper, for free, and without widening
|
|
472
|
+
// `classifyToolError`'s regex list: the fix is giving the message a proper
|
|
473
|
+
// `upstream_down:` token at the point the two facts (no response was ever
|
|
474
|
+
// received, and which host we were trying to reach) are actually in hand,
|
|
475
|
+
// not teaching the classifier to guess from prose after the fact.
|
|
476
|
+
//
|
|
477
|
+
// Safe on the same grounds as the timeout branch above: no argument a
|
|
478
|
+
// caller passes can make `fetch()` itself throw a connection-level error,
|
|
479
|
+
// so this is always an availability failure, never a caller mistake. Same
|
|
480
|
+
// `markInternalOrigin` treatment — an origin we run that never answered is
|
|
481
|
+
// still ours, not a third party's outage.
|
|
482
|
+
const raw = err instanceof Error ? err.message : String(err);
|
|
483
|
+
throw new Error(
|
|
484
|
+
markInternalOrigin(
|
|
485
|
+
`upstream_down: could not reach ${name} at all (${raw.slice(0, 160)}). ` +
|
|
486
|
+
`No request reached ${name}, so this says NOTHING about whether the arguments you passed ` +
|
|
487
|
+
'are valid — do not re-check them on the strength of this error. Retry shortly.',
|
|
488
|
+
url,
|
|
489
|
+
),
|
|
490
|
+
);
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
function detailSuffix(detail: string): string {
|
|
495
|
+
return detail ? ` — ${detail}` : '';
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
async function readDetail(res: Response): Promise<string> {
|
|
499
|
+
let raw: string;
|
|
500
|
+
try {
|
|
501
|
+
raw = await res.text();
|
|
502
|
+
} catch {
|
|
503
|
+
// Body already consumed, or the connection died mid-read. The status alone
|
|
504
|
+
// is still worth throwing — never let the error path throw its own error.
|
|
505
|
+
return '';
|
|
506
|
+
}
|
|
507
|
+
return summarizeErrorBody(raw);
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
/**
|
|
511
|
+
* Turn ANY error body — JSON, HTML, XML or plain text — into one short phrase
|
|
512
|
+
* that never contains markup.
|
|
513
|
+
*
|
|
514
|
+
* This used to just drop an HTML or XML body on the floor, on the reasoning
|
|
515
|
+
* that markup crowds out the status. That was half right. Dropping it loses the
|
|
516
|
+
* one sentence a caller could have acted on: an `Access Denied` title, an SDMX
|
|
517
|
+
* `<message:Error>` text, an OPS fault string. A 2026-08-30 support sweep
|
|
518
|
+
* measured 13 of 291 caller-facing error rows carrying a raw page or document
|
|
519
|
+
* verbatim, across 11 packs, and in every one of them the useful content —
|
|
520
|
+
* "Access Denied", "Invalid country code", "SCRAPE_TIMEOUT" — was in there,
|
|
521
|
+
* buried in markup the agent had to parse out of a string (fleet #712).
|
|
522
|
+
*
|
|
523
|
+
* So: extract the meaning, discard the markup. The output is passed through
|
|
524
|
+
* `stripMarkup` unconditionally, which is what lets `check:error-body-leak`
|
|
525
|
+
* assert mechanically that no caller-facing message can contain `<?xml`,
|
|
526
|
+
* `<!DOCTYPE` or `<html`.
|
|
527
|
+
*/
|
|
528
|
+
function summarizeErrorBody(raw: string): string {
|
|
529
|
+
if (!raw || !raw.trim()) return '';
|
|
530
|
+
|
|
531
|
+
const head = raw.slice(0, 400).trimStart().toLowerCase();
|
|
532
|
+
|
|
533
|
+
// An HTML error page (Cloudflare interstitial, nginx default, a login
|
|
534
|
+
// redirect) says what it is in its <title>, and almost nowhere else.
|
|
535
|
+
if (head.startsWith('<!doctype') || head.startsWith('<html')) {
|
|
536
|
+
const title = htmlTitle(raw);
|
|
537
|
+
return title
|
|
538
|
+
? `${title} (upstream returned an HTML error page, not an API response)`
|
|
539
|
+
: 'upstream returned an HTML error page, not an API response';
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
// XML fault documents — EPO OPS, SDMX (`<message:Error>`), SOAP faults. The
|
|
543
|
+
// human sentence sits in a child element whose tag name says what it is.
|
|
544
|
+
if (head.startsWith('<?xml') || head.startsWith('<')) {
|
|
545
|
+
const fault = xmlFaultText(raw);
|
|
546
|
+
return fault
|
|
547
|
+
? `${stripMarkup(fault).slice(0, MAX_DETAIL)} (from the upstream's XML error document)`
|
|
548
|
+
: 'upstream returned an XML error document with no readable message';
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
// Most JSON error bodies bury one human sentence among ids and echoed request
|
|
552
|
+
// params. Prefer that sentence; fall back to the whole body when the shape is
|
|
553
|
+
// unfamiliar, since an unfamiliar shape is exactly when we can least afford to
|
|
554
|
+
// guess wrong and show nothing.
|
|
555
|
+
const fromJson = messageFromJson(raw);
|
|
556
|
+
return stripMarkup(fromJson ?? raw).slice(0, MAX_DETAIL);
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
/** The `<title>` of an HTML error page, or its first `<h1>` — the two places a
|
|
560
|
+
* bot wall, a 502 and an "Access Denied" all state what happened. */
|
|
561
|
+
function htmlTitle(raw: string): string | null {
|
|
562
|
+
const head = raw.slice(0, 4000);
|
|
563
|
+
for (const re of [/<title[^>]*>([\s\S]*?)<\/title>/i, /<h1[^>]*>([\s\S]*?)<\/h1>/i]) {
|
|
564
|
+
const m = re.exec(head);
|
|
565
|
+
const text = m ? stripMarkup(m[1]) : '';
|
|
566
|
+
if (text) return text.slice(0, 160);
|
|
567
|
+
}
|
|
568
|
+
return null;
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
/** Tag names that carry the explanation in an XML fault document, namespace
|
|
572
|
+
* prefix optional (`<message:Error>`, `<com:Text>`, `<faultstring>`). */
|
|
573
|
+
const XML_FAULT_TAG_RE =
|
|
574
|
+
/<(?:[A-Za-z0-9_.-]+:)?(?:text|message|description|faultstring|reason|detail|title|errormessage|error)\b[^>]*>([^<]{2,400})</i;
|
|
575
|
+
|
|
576
|
+
function xmlFaultText(raw: string): string | null {
|
|
577
|
+
const head = raw.slice(0, 8000);
|
|
578
|
+
const tagged = XML_FAULT_TAG_RE.exec(head);
|
|
579
|
+
if (tagged && tagged[1].trim()) return tagged[1];
|
|
580
|
+
|
|
581
|
+
// Nothing conventionally named — take the longest text node instead. A fault
|
|
582
|
+
// document with one sentence in an oddly named element is still readable;
|
|
583
|
+
// returning nothing at all is not.
|
|
584
|
+
let best = '';
|
|
585
|
+
for (const m of head.matchAll(/>([^<>]{8,400})</g)) {
|
|
586
|
+
const text = m[1].trim();
|
|
587
|
+
if (text.length > best.length) best = text;
|
|
588
|
+
}
|
|
589
|
+
return best || null;
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
/**
|
|
593
|
+
* Remove every tag and stray angle bracket, then collapse whitespace.
|
|
594
|
+
*
|
|
595
|
+
* Applied to everything on the way out, including the JSON and plain-text
|
|
596
|
+
* paths, because an upstream is free to embed markup in a JSON string field —
|
|
597
|
+
* and a leak is a leak regardless of which branch produced it.
|
|
598
|
+
*/
|
|
599
|
+
function stripMarkup(s: string): string {
|
|
600
|
+
return collapse(decodeEntities(s.replace(/<[^>]*>/g, ' ')).replace(/[<>]/g, ' '));
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
/** The handful of entities that show up in error-page titles. Decoded AFTER
|
|
604
|
+
* tags are stripped and BEFORE the angle-bracket sweep, so `<script>`
|
|
605
|
+
* in a title cannot decode into markup that survives — EMBL-EBI's ChEMBL 500
|
|
606
|
+
* page renders as `500 Internal Server Error < EMBL-EBI` otherwise. */
|
|
607
|
+
function decodeEntities(s: string): string {
|
|
608
|
+
return s
|
|
609
|
+
.replace(/&(?:amp|#0*38);/gi, '&')
|
|
610
|
+
.replace(/&(?:lt|#0*60);/gi, '<')
|
|
611
|
+
.replace(/&(?:gt|#0*62);/gi, '>')
|
|
612
|
+
.replace(/&(?:quot|#0*34);/gi, '"')
|
|
613
|
+
.replace(/&(?:#0*39|apos|#x0*27);/gi, "'")
|
|
614
|
+
.replace(/ /gi, ' ');
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
/** The conventional "what went wrong" field, under any of the names upstreams
|
|
618
|
+
* actually use. Checked in order; first non-empty string wins. */
|
|
619
|
+
const MESSAGE_KEYS = [
|
|
620
|
+
'message', 'error_message', 'errorMessage', 'detail', 'details',
|
|
621
|
+
'description', 'error_description', 'reason', 'title', 'fault',
|
|
622
|
+
];
|
|
623
|
+
|
|
624
|
+
function messageFromJson(raw: string): string | null {
|
|
625
|
+
let parsed: unknown;
|
|
626
|
+
try {
|
|
627
|
+
parsed = JSON.parse(raw);
|
|
628
|
+
} catch {
|
|
629
|
+
return null;
|
|
630
|
+
}
|
|
631
|
+
return pickMessage(parsed, 0);
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
function pickMessage(node: unknown, depth: number): string | null {
|
|
635
|
+
// Two levels covers `{error: {message}}` and `{errors: [{detail}]}`, the two
|
|
636
|
+
// shapes that account for nearly all of them, without walking a large payload.
|
|
637
|
+
if (depth > 2 || node == null) return null;
|
|
638
|
+
|
|
639
|
+
if (typeof node === 'string') return node.trim() || null;
|
|
640
|
+
|
|
641
|
+
if (Array.isArray(node)) {
|
|
642
|
+
for (const item of node) {
|
|
643
|
+
const found = pickMessage(item, depth + 1);
|
|
644
|
+
if (found) return found;
|
|
645
|
+
}
|
|
646
|
+
return null;
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
if (typeof node !== 'object') return null;
|
|
650
|
+
const obj = node as Record<string, unknown>;
|
|
651
|
+
|
|
652
|
+
for (const key of MESSAGE_KEYS) {
|
|
653
|
+
const v = obj[key];
|
|
654
|
+
if (typeof v === 'string' && v.trim()) return v.trim();
|
|
655
|
+
}
|
|
656
|
+
// `{error: …}` where error is itself an object or a string — the single most
|
|
657
|
+
// common wrapper, so it is worth descending into by name rather than scanning
|
|
658
|
+
// every key and risking picking up an echoed request parameter.
|
|
659
|
+
for (const key of ['error', 'errors', 'fault', 'Error', 'data']) {
|
|
660
|
+
if (key in obj) {
|
|
661
|
+
const found = pickMessage(obj[key], depth + 1);
|
|
662
|
+
if (found) return found;
|
|
663
|
+
}
|
|
664
|
+
}
|
|
665
|
+
return null;
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
/** Errors are read in a single line of log output; newlines and runs of
|
|
669
|
+
* whitespace make a multi-line body unreadable there. */
|
|
670
|
+
function collapse(s: string): string {
|
|
671
|
+
return s.replace(/\s+/g, ' ').trim();
|
|
672
|
+
}
|
|
19
673
|
/**
|
|
20
674
|
* legislation.gov.uk MCP — the UK's official legislation database.
|
|
21
675
|
*
|
|
@@ -27,33 +681,116 @@ interface McpToolExport {
|
|
|
27
681
|
* Document types: ukpga (UK Public General Acts), uksi (UK Statutory
|
|
28
682
|
* Instruments), asp (Acts of the Scottish Parliament), anaw / asc (Wales),
|
|
29
683
|
* nia (Northern Ireland Acts), ukla (UK Local Acts).
|
|
684
|
+
*
|
|
685
|
+
* POINT-IN-TIME IS THE TRAP HERE, and it is a wrong answer with a confident
|
|
686
|
+
* face rather than an error. UK legislation exists in versions: the text AS
|
|
687
|
+
* ENACTED, and the text AS AMENDED at a given date. They are not cosmetically
|
|
688
|
+
* different. Section 5 of the Data Protection Act 2018 reads "Terms used in
|
|
689
|
+
* Chapter 2 of this Part and in the GDPR" as enacted, and "Terms used in ...
|
|
690
|
+
* this Part and in the UK GDPR" as it stands today — post-Brexit amendment
|
|
691
|
+
* changed both the instrument referred to and the scope. Quoting the wrong one
|
|
692
|
+
* misstates current law while looking perfectly authoritative, so every text
|
|
693
|
+
* response here names the version it served and the date that version took
|
|
694
|
+
* effect. Never serve one silently.
|
|
695
|
+
*
|
|
696
|
+
* SIZE IS THE OTHER TRAP. The whole-Act XML for a large statute is enormous —
|
|
697
|
+
* the Data Protection Act 2018 is 5.9 MB, its table of contents alone 1.4 MB —
|
|
698
|
+
* and this runs in a Worker with a hard isolate ceiling. Every fetch here is
|
|
699
|
+
* bounded by bytes read, never by trusting the far end to be reasonable.
|
|
700
|
+
*
|
|
701
|
+
* AMENDMENTS ARE PAGED AT 50 (data.feed), never the total — legislation.gov.uk
|
|
702
|
+
* reports `<openSearch:totalResults>` and `<leg:totalPages>` on every page, and
|
|
703
|
+
* this pack surfaces both rather than letting a caller mistake one page for the
|
|
704
|
+
* whole change history (the Data Protection Act 2018 alone has 2,271+ recorded
|
|
705
|
+
* effects across 46 pages).
|
|
706
|
+
*
|
|
707
|
+
* POINT-IN-TIME REDIRECTS SILENTLY, another version of the same trap as above:
|
|
708
|
+
* ask for a date with no amendment on it and legislation.gov.uk serves the
|
|
709
|
+
* nearest EARLIER version without changing the URL (Content-Location echoes
|
|
710
|
+
* the date you asked for) — the served date is recoverable only from the
|
|
711
|
+
* document's own RestrictStartDate/dct:valid metadata, which can differ from
|
|
712
|
+
* the date you passed. Every as_at response says so explicitly.
|
|
713
|
+
*
|
|
714
|
+
* EXPLANATORY NOTES HAVE NO XML — data.xml 404s; only /notes (HTML) exists,
|
|
715
|
+
* and its template varies by the Act's age: older Acts (e.g. Equality Act
|
|
716
|
+
* 2010) put the whole commentary, section headings included, on one HTML
|
|
717
|
+
* page; newer Acts (e.g. Data Protection Act 2018) serve only a PDF stub at
|
|
718
|
+
* /notes and split the real content into numbered HTML "divisions" under
|
|
719
|
+
* /notes/contents, with per-section commentary living in whichever division
|
|
720
|
+
* is titled "Commentary on provisions of Act". Not every Act has notes at
|
|
721
|
+
* all (Appropriation, Consolidated Fund, Finance and Consolidation Acts never
|
|
722
|
+
* get them) and not every section gets its own paragraph.
|
|
30
723
|
*/
|
|
31
724
|
|
|
32
725
|
|
|
726
|
+
// Bound every fetch() in this pack to a fixed timeout — an upstream that
|
|
727
|
+
// degrades without erroring would otherwise hold the Worker in `await fetch()`
|
|
728
|
+
// until its own execution budget kills the request (minutes, not seconds).
|
|
729
|
+
// Mirrors the epoFetch / usaspending retryFetch pattern (fleet #685).
|
|
730
|
+
async function pwFetch(url: string | URL, init?: RequestInit): Promise<Response> {
|
|
731
|
+
return fetchWithTimeout(url, init ?? {}, 'legislation.gov.uk');
|
|
732
|
+
}
|
|
733
|
+
|
|
33
734
|
const BASE = 'https://www.legislation.gov.uk';
|
|
34
735
|
const UA = 'pipeworx-mcp-legislation-uk/1.0 (+https://pipeworx.io)';
|
|
35
736
|
|
|
737
|
+
const VERSION_HELP =
|
|
738
|
+
'"current" (default — the law as amended and in force today), "enacted" (the original text as passed), or a date YYYY-MM-DD for the text as it stood on that day.';
|
|
739
|
+
|
|
36
740
|
const TYPES =
|
|
37
741
|
'ukpga (UK Public General Acts), uksi (UK Statutory Instruments), asp (Scottish Parliament Acts), anaw/asc (Wales), nia (Northern Ireland), ukla (UK Local Acts)';
|
|
38
742
|
|
|
743
|
+
// An agent without the exact type code otherwise gets `invalid_arguments` —
|
|
744
|
+
// accept a plain-English category as an alternative and map it, defaulting to
|
|
745
|
+
// the most common type (Acts of Parliament) when nothing at all is given.
|
|
746
|
+
const CATEGORY_ALIASES: Record<string, string> = {
|
|
747
|
+
act: 'ukpga',
|
|
748
|
+
acts: 'ukpga',
|
|
749
|
+
si: 'uksi',
|
|
750
|
+
sis: 'uksi',
|
|
751
|
+
'statutory instrument': 'uksi',
|
|
752
|
+
'statutory instruments': 'uksi',
|
|
753
|
+
ssi: 'ssi',
|
|
754
|
+
'scottish statutory instrument': 'ssi',
|
|
755
|
+
'scottish statutory instruments': 'ssi',
|
|
756
|
+
};
|
|
757
|
+
const DEFAULT_TYPE = 'ukpga';
|
|
758
|
+
|
|
759
|
+
function resolveType(v: unknown): { type: string; defaulted: boolean; note?: string } {
|
|
760
|
+
if (typeof v !== 'string' || !v.trim()) {
|
|
761
|
+
return {
|
|
762
|
+
type: DEFAULT_TYPE,
|
|
763
|
+
defaulted: true,
|
|
764
|
+
note: `No "type" given — defaulted to "${DEFAULT_TYPE}" (UK Public General Acts). Pass type to search other document types: ${TYPES}.`,
|
|
765
|
+
};
|
|
766
|
+
}
|
|
767
|
+
const raw = v.trim();
|
|
768
|
+
const alias = CATEGORY_ALIASES[raw.toLowerCase()];
|
|
769
|
+
if (alias) {
|
|
770
|
+
return { type: alias, defaulted: false, note: `"${raw}" mapped to type "${alias}".` };
|
|
771
|
+
}
|
|
772
|
+
return { type: raw, defaulted: false };
|
|
773
|
+
}
|
|
774
|
+
|
|
39
775
|
const tools: McpToolExport['tools'] = [
|
|
40
776
|
{
|
|
41
|
-
name: '
|
|
777
|
+
name: 'search_uk_legislation',
|
|
42
778
|
description:
|
|
43
779
|
'Search UK legislation by title words (and optional year), returning matching Acts/instruments with their full-text URLs. ' +
|
|
44
|
-
`UK legislation only. Source: legislation.gov.uk Atom feed. Document types: ${TYPES}
|
|
780
|
+
`UK legislation only. Source: legislation.gov.uk Atom feed. Document types: ${TYPES}. ` +
|
|
781
|
+
`"type" is optional: give an exact code (e.g. "ukpga"), a plain category ("act", "si", "ssi"), or omit it entirely — omitted defaults to "${DEFAULT_TYPE}" and the response says so.`,
|
|
45
782
|
inputSchema: {
|
|
46
783
|
type: 'object',
|
|
47
784
|
properties: {
|
|
48
785
|
type: {
|
|
49
786
|
type: 'string',
|
|
50
|
-
description: `Document type
|
|
787
|
+
description: `Document type code (e.g. "ukpga") or category ("act", "si", "ssi"). Optional — defaults to "${DEFAULT_TYPE}" if omitted. Full code list: ${TYPES}.`,
|
|
51
788
|
},
|
|
52
789
|
title: { type: 'string', description: 'Words to match in the title, e.g. "equality".' },
|
|
53
790
|
year: { type: 'number', description: 'Optional year to restrict results, e.g. 2010.' },
|
|
54
791
|
page: { type: 'number', description: 'Results page (default 1). 20 results per page.' },
|
|
55
792
|
},
|
|
56
|
-
required: ['
|
|
793
|
+
required: ['title'],
|
|
57
794
|
},
|
|
58
795
|
},
|
|
59
796
|
{
|
|
@@ -72,21 +809,108 @@ const tools: McpToolExport['tools'] = [
|
|
|
72
809
|
required: ['type', 'year', 'number'],
|
|
73
810
|
},
|
|
74
811
|
},
|
|
812
|
+
{
|
|
813
|
+
name: 'get_legislation_section',
|
|
814
|
+
description:
|
|
815
|
+
'Read the actual WORDS of one section of a UK Act or statutory instrument — "what does section 5 of the Data Protection Act 2018 say", "quote section 1 of the Human Rights Act", "text of s.170 DPA 2018". Returns the full text of that section alone, so a specific provision does not cost a whole statute. '
|
|
816
|
+
+ 'Identify the legislation by type + year + number (Data Protection Act 2018 = ukpga/2018/12) and give the section number. '
|
|
817
|
+
+ 'UK law exists in VERSIONS and they differ in substance: pass version to choose ' + VERSION_HELP + ' as_at is an alias for a date version (e.g. as_at: "2019-01-01") — the same date is also accepted via version. '
|
|
818
|
+
+ 'legislation.gov.uk serves the NEAREST EARLIER version for a date with no amendment on it rather than erroring, so every response states the version it actually served (version_valid_from) alongside the date asked for, even when they differ. Source: legislation.gov.uk, the official database. Use search_uk_legislation or get_legislation first if you need the year and number.',
|
|
819
|
+
inputSchema: {
|
|
820
|
+
type: 'object',
|
|
821
|
+
properties: {
|
|
822
|
+
type: { type: 'string', description: `Document type, e.g. "ukpga". One of: ${TYPES}.` },
|
|
823
|
+
year: { type: 'number', description: 'Year, e.g. 2018.' },
|
|
824
|
+
number: { type: 'number', description: 'Item number within that year/type, e.g. 12 for the Data Protection Act 2018.' },
|
|
825
|
+
section: { type: 'string', description: 'Section number, e.g. "5", "170", or "5A" for an inserted section.' },
|
|
826
|
+
version: { type: 'string', description: `Which version of the text: ${VERSION_HELP}` },
|
|
827
|
+
as_at: { type: 'string', description: 'YYYY-MM-DD — the law as it stood on this date. Alias for a date "version"; the response states the actual served version date, which may be earlier than this if nothing changed on the exact date given.' },
|
|
828
|
+
},
|
|
829
|
+
required: ['type', 'year', 'number', 'section'],
|
|
830
|
+
},
|
|
831
|
+
},
|
|
832
|
+
{
|
|
833
|
+
name: 'get_legislation_text',
|
|
834
|
+
description:
|
|
835
|
+
'Read the text of a whole UK Act or statutory instrument — the body of the law rather than a catalogue entry. Use for short instruments and for reading an Act end to end; for one provision prefer get_legislation_section, which returns just that section. '
|
|
836
|
+
+ 'Large Acts run to megabytes, so this reads a bounded amount and says plainly when it stopped short rather than returning a silently clipped statute. '
|
|
837
|
+
+ 'UK law exists in VERSIONS that differ in substance: pass version to choose ' + VERSION_HELP + ' as_at is an alias for a date version (e.g. as_at: "2019-01-01"). '
|
|
838
|
+
+ 'legislation.gov.uk serves the NEAREST EARLIER version for a date with no amendment on it rather than erroring, so every response states the version it actually served (version_valid_from) alongside the date asked for, even when they differ. Source: legislation.gov.uk, the official database.',
|
|
839
|
+
inputSchema: {
|
|
840
|
+
type: 'object',
|
|
841
|
+
properties: {
|
|
842
|
+
type: { type: 'string', description: `Document type, e.g. "ukpga". One of: ${TYPES}.` },
|
|
843
|
+
year: { type: 'number', description: 'Year, e.g. 2018.' },
|
|
844
|
+
number: { type: 'number', description: 'Item number within that year/type, e.g. 12.' },
|
|
845
|
+
version: { type: 'string', description: `Which version of the text: ${VERSION_HELP}` },
|
|
846
|
+
as_at: { type: 'string', description: 'YYYY-MM-DD — the law as it stood on this date. Alias for a date "version"; the response states the actual served version date, which may be earlier than this if nothing changed on the exact date given.' },
|
|
847
|
+
},
|
|
848
|
+
required: ['type', 'year', 'number'],
|
|
849
|
+
},
|
|
850
|
+
},
|
|
851
|
+
{
|
|
852
|
+
name: 'legislation_amendments',
|
|
853
|
+
description:
|
|
854
|
+
'What has amended this UK Act or instrument, and (in the other direction) what this legislation itself amends — "what has changed the Data Protection Act 2018", "has section 13A been amended", "what does the Victims and Prisoners Act 2024 amend". '
|
|
855
|
+
+ 'Each row names the amending/amended instrument, the affected and affecting provisions, the type of change (inserted/substituted/repealed/omitted/amended etc.), whether it has actually been APPLIED (some amendments are made but not yet in force), and the commencement date. '
|
|
856
|
+
+ 'PAGED AT 50 — the response states total_results and total_pages; 50 rows is a page, never the whole change history (some Acts have thousands of recorded effects), pass page for more. '
|
|
857
|
+
+ 'direction "affected" (default) = changes made TO this legislation by others; "affecting" = changes this legislation makes TO others. Source: legislation.gov.uk /changes/ Atom feed.',
|
|
858
|
+
inputSchema: {
|
|
859
|
+
type: 'object',
|
|
860
|
+
properties: {
|
|
861
|
+
type: { type: 'string', description: `Document type, e.g. "ukpga". One of: ${TYPES}.` },
|
|
862
|
+
year: { type: 'number', description: 'Year, e.g. 2018.' },
|
|
863
|
+
number: { type: 'number', description: 'Item number within that year/type, e.g. 12 for the Data Protection Act 2018.' },
|
|
864
|
+
direction: {
|
|
865
|
+
type: 'string',
|
|
866
|
+
description: '"affected" (default) — changes made TO this legislation by other instruments. "affecting" — changes this legislation makes TO other instruments.',
|
|
867
|
+
},
|
|
868
|
+
page: { type: 'number', description: 'Results page (default 1). 50 rows per page — see total_pages in the response.' },
|
|
869
|
+
},
|
|
870
|
+
required: ['type', 'year', 'number'],
|
|
871
|
+
},
|
|
872
|
+
},
|
|
873
|
+
{
|
|
874
|
+
name: 'get_explanatory_notes',
|
|
875
|
+
description:
|
|
876
|
+
'Read the Explanatory Notes for a UK Act — plain-language commentary written by the government department responsible, explaining what a section is meant to do and why. NOT part of the law and not endorsed by Parliament — use for context on intent, never as the text of the law itself (use get_legislation_section for that). '
|
|
877
|
+
+ 'Pass section to get the commentary for just that section; omit it for an overview plus the notes\' own table of contents. '
|
|
878
|
+
+ 'Not every Act has notes (Appropriation, Consolidated Fund, Finance and Consolidation Acts never get them) and not every section gets its own paragraph — some are grouped or unremarked. Notes are written for the Act AS ENACTED and are not updated for later amendments. Source: legislation.gov.uk (HTML only — there is no XML/data feed for notes).',
|
|
879
|
+
inputSchema: {
|
|
880
|
+
type: 'object',
|
|
881
|
+
properties: {
|
|
882
|
+
type: { type: 'string', description: `Document type, e.g. "ukpga". One of: ${TYPES}.` },
|
|
883
|
+
year: { type: 'number', description: 'Year, e.g. 2018.' },
|
|
884
|
+
number: { type: 'number', description: 'Item number within that year/type, e.g. 12 for the Data Protection Act 2018.' },
|
|
885
|
+
section: { type: 'string', description: 'Optional: section number, e.g. "5", to get just that section\'s commentary. Omit for an overview.' },
|
|
886
|
+
},
|
|
887
|
+
required: ['type', 'year', 'number'],
|
|
888
|
+
},
|
|
889
|
+
},
|
|
75
890
|
];
|
|
76
891
|
|
|
77
892
|
async function callTool(name: string, args: Record<string, unknown>): Promise<unknown> {
|
|
78
893
|
switch (name) {
|
|
79
|
-
case '
|
|
894
|
+
case 'search_uk_legislation':
|
|
895
|
+
case 'search_legislation': // pre-rename name; keep resolving (fleet #2256)
|
|
80
896
|
return searchLegislation(args);
|
|
81
897
|
case 'get_legislation':
|
|
82
898
|
return getLegislation(args);
|
|
899
|
+
case 'get_legislation_section':
|
|
900
|
+
return getLegislationSection(args);
|
|
901
|
+
case 'get_legislation_text':
|
|
902
|
+
return getLegislationText(args);
|
|
903
|
+
case 'legislation_amendments':
|
|
904
|
+
return getLegislationAmendments(args);
|
|
905
|
+
case 'get_explanatory_notes':
|
|
906
|
+
return getExplanatoryNotes(args);
|
|
83
907
|
default:
|
|
84
908
|
throw new Error(`Unknown tool: ${name}`);
|
|
85
909
|
}
|
|
86
910
|
}
|
|
87
911
|
|
|
88
912
|
async function searchLegislation(args: Record<string, unknown>): Promise<unknown> {
|
|
89
|
-
const type =
|
|
913
|
+
const { type, defaulted, note: typeNote } = resolveType(args.type);
|
|
90
914
|
const title = reqStr(args, 'title', '"equality"');
|
|
91
915
|
const year = numOrUndef(args.year);
|
|
92
916
|
const page = numOrUndef(args.page) ?? 1;
|
|
@@ -123,7 +947,8 @@ async function searchLegislation(args: Record<string, unknown>): Promise<unknown
|
|
|
123
947
|
|
|
124
948
|
return {
|
|
125
949
|
source: 'legislation.gov.uk',
|
|
126
|
-
note: 'UK legislation only. Parsed best-effort from an Atom feed.',
|
|
950
|
+
note: 'UK legislation only. Parsed best-effort from an Atom feed.' + (typeNote ? ` ${typeNote}` : ''),
|
|
951
|
+
type_defaulted: defaulted,
|
|
127
952
|
query: { type, title, year, page },
|
|
128
953
|
totalResults,
|
|
129
954
|
count: results.length,
|
|
@@ -164,13 +989,578 @@ async function getLegislation(args: Record<string, unknown>): Promise<unknown> {
|
|
|
164
989
|
};
|
|
165
990
|
}
|
|
166
991
|
|
|
992
|
+
/**
|
|
993
|
+
* Read at most maxBytes of a response, then stop.
|
|
994
|
+
*
|
|
995
|
+
* Not a nicety: the whole-Act XML for the Data Protection Act 2018 is 5.9 MB
|
|
996
|
+
* and legislation.gov.uk will happily send all of it. A Worker that
|
|
997
|
+
* materialises that alongside a routing working set is how the isolate ceiling
|
|
998
|
+
* gets hit, so the cap is enforced on OUR side of the socket rather than by
|
|
999
|
+
* asking politely.
|
|
1000
|
+
*/
|
|
1001
|
+
async function getTextBounded(url: string, maxBytes: number): Promise<{ text: string; truncated: boolean; bytes: number }> {
|
|
1002
|
+
const res = await pwFetch(url, { headers: { Accept: 'application/xml', 'User-Agent': UA } });
|
|
1003
|
+
if (!res.ok) {
|
|
1004
|
+
const peek = await res.text();
|
|
1005
|
+
throw new Error(`legislation.gov.uk: ${res.status} ${peek.slice(0, 200)}`);
|
|
1006
|
+
}
|
|
1007
|
+
if (!res.body) {
|
|
1008
|
+
const whole = await res.text();
|
|
1009
|
+
return { text: whole.slice(0, maxBytes), truncated: whole.length > maxBytes, bytes: whole.length };
|
|
1010
|
+
}
|
|
1011
|
+
const reader = res.body.getReader();
|
|
1012
|
+
const chunks: Uint8Array[] = [];
|
|
1013
|
+
let total = 0;
|
|
1014
|
+
let truncated = false;
|
|
1015
|
+
for (;;) {
|
|
1016
|
+
const { done, value } = await reader.read();
|
|
1017
|
+
if (done) break;
|
|
1018
|
+
if (value) {
|
|
1019
|
+
total += value.byteLength;
|
|
1020
|
+
chunks.push(value);
|
|
1021
|
+
if (total >= maxBytes) { truncated = true; await reader.cancel(); break; }
|
|
1022
|
+
}
|
|
1023
|
+
}
|
|
1024
|
+
const buf = new Uint8Array(total);
|
|
1025
|
+
let off = 0;
|
|
1026
|
+
for (const c of chunks) { buf.set(c, off); off += c.byteLength; }
|
|
1027
|
+
return { text: new TextDecoder().decode(buf), truncated, bytes: total };
|
|
1028
|
+
}
|
|
1029
|
+
|
|
1030
|
+
/** CLML marks repealed/omitted material; keep it visible rather than silently dropping it. */
|
|
1031
|
+
function xmlToText(xml: string): string {
|
|
1032
|
+
return xml
|
|
1033
|
+
.replace(/<Commentary[\s\S]*?<\/Commentary>/g, ' ')
|
|
1034
|
+
// Strip the metadata BLOCK by name, with a backreference. The obvious
|
|
1035
|
+
// /<ukm:[\s\S]*?<\/ukm:[A-Za-z]+>/ looks non-greedy and is not safe: ukm:
|
|
1036
|
+
// elements bracket the whole file, so the first opening tag pairs with a
|
|
1037
|
+
// LATER closing tag of a different name and the entire provision
|
|
1038
|
+
// disappears between them. Measured on section 5 DPA 2018: it reduced a
|
|
1039
|
+
// 36 kB document to 69 characters of headings — a confident, empty answer.
|
|
1040
|
+
.replace(/<ukm:Metadata\b[\s\S]*?<\/ukm:Metadata>/g, ' ')
|
|
1041
|
+
.replace(/<ukm:([A-Za-z]+)\b[^>]*\/>/g, ' ')
|
|
1042
|
+
.replace(/<[^>]+>/g, ' ')
|
|
1043
|
+
.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"')
|
|
1044
|
+
.replace(/�?39;|'/g, "'").replace(/&/g, '&')
|
|
1045
|
+
.replace(/[ \t]+/g, ' ')
|
|
1046
|
+
.replace(/\s*\n\s*/g, '\n')
|
|
1047
|
+
.replace(/\n{3,}/g, '\n\n')
|
|
1048
|
+
.replace(/\s+/g, ' ')
|
|
1049
|
+
.trim();
|
|
1050
|
+
}
|
|
1051
|
+
|
|
1052
|
+
|
|
1053
|
+
// ── Version handling ──────────────────────────────────────────────────────
|
|
1054
|
+
//
|
|
1055
|
+
// legislation.gov.uk expresses versions in the PATH:
|
|
1056
|
+
// /ukpga/2018/12/section/5 current, i.e. as amended to date
|
|
1057
|
+
// /ukpga/2018/12/section/5/enacted the text as originally passed
|
|
1058
|
+
// /ukpga/2018/12/section/5/2020-01-01 the text as it stood on that date
|
|
1059
|
+
// The three return materially different words, so which one was asked for has
|
|
1060
|
+
// to survive into the answer.
|
|
1061
|
+
function versionSegment(v: unknown): { seg: string; label: string } {
|
|
1062
|
+
if (typeof v !== 'string' || !v.trim() || v.trim().toLowerCase() === 'current') {
|
|
1063
|
+
return { seg: '', label: 'current (as amended and in force today)' };
|
|
1064
|
+
}
|
|
1065
|
+
const t = v.trim().toLowerCase();
|
|
1066
|
+
if (t === 'enacted' || t === 'as-enacted' || t === 'as enacted') {
|
|
1067
|
+
return { seg: '/enacted', label: 'as enacted (the original text as passed, ignoring later amendments)' };
|
|
1068
|
+
}
|
|
1069
|
+
if (/^\d{4}-\d{2}-\d{2}$/.test(t)) {
|
|
1070
|
+
return { seg: `/${t}`, label: `as it stood on ${t}` };
|
|
1071
|
+
}
|
|
1072
|
+
throw new Error(`Unrecognised "version": ${v}. Use ${VERSION_HELP}`);
|
|
1073
|
+
}
|
|
1074
|
+
|
|
1075
|
+
/**
|
|
1076
|
+
* as_at is a plain-English alias for a date-shaped "version" — accept either,
|
|
1077
|
+
* or both if they agree, and reject the ambiguous case of two DIFFERENT
|
|
1078
|
+
* dates rather than silently picking one.
|
|
1079
|
+
*/
|
|
1080
|
+
function resolveVersion(args: Record<string, unknown>): { seg: string; label: string; requestedDate: string | null } {
|
|
1081
|
+
const asAt = typeof args.as_at === 'string' && args.as_at.trim() ? args.as_at.trim() : undefined;
|
|
1082
|
+
const version = typeof args.version === 'string' && args.version.trim() ? args.version.trim() : undefined;
|
|
1083
|
+
if (asAt && !/^\d{4}-\d{2}-\d{2}$/.test(asAt)) {
|
|
1084
|
+
throw new Error(`Unrecognised "as_at": ${asAt}. Use a date like "2019-01-01".`);
|
|
1085
|
+
}
|
|
1086
|
+
if (asAt && version && version.toLowerCase() !== 'current' && version !== asAt) {
|
|
1087
|
+
throw new Error(`"as_at" (${asAt}) and "version" (${version}) disagree — pass only one.`);
|
|
1088
|
+
}
|
|
1089
|
+
const chosen = asAt ?? version;
|
|
1090
|
+
const { seg, label } = versionSegment(chosen);
|
|
1091
|
+
return { seg, label, requestedDate: asAt ?? (chosen && /^\d{4}-\d{2}-\d{2}$/.test(chosen) ? chosen : null) };
|
|
1092
|
+
}
|
|
1093
|
+
|
|
1094
|
+
/** The date the served version took effect, straight from the document. */
|
|
1095
|
+
function versionValidFrom(xml: string): string | undefined {
|
|
1096
|
+
return attr(xml, /\bRestrictStartDate="([^"]+)"/)
|
|
1097
|
+
?? firstGroup(xml, /<dct:valid>([^<]+)<\/dct:valid>/)?.trim();
|
|
1098
|
+
}
|
|
1099
|
+
|
|
1100
|
+
const SECTION_MAX_BYTES = 400_000;
|
|
1101
|
+
const ACT_MAX_BYTES = 1_200_000;
|
|
1102
|
+
|
|
1103
|
+
async function getLegislationSection(args: Record<string, unknown>): Promise<unknown> {
|
|
1104
|
+
const type = reqStr(args, 'type', '"ukpga"');
|
|
1105
|
+
const year = reqNum(args, 'year', '2018');
|
|
1106
|
+
const number = reqNum(args, 'number', '12');
|
|
1107
|
+
const sectionRaw = args.section;
|
|
1108
|
+
if (sectionRaw === undefined || sectionRaw === null || String(sectionRaw).trim() === '') {
|
|
1109
|
+
throw new Error('Required argument "section" is missing. Pass the section number, e.g. 5 (or "5A").');
|
|
1110
|
+
}
|
|
1111
|
+
const section = String(sectionRaw).trim().replace(/^s(ection)?\.?\s*/i, '');
|
|
1112
|
+
const { seg, label, requestedDate } = resolveVersion(args);
|
|
1113
|
+
|
|
1114
|
+
const docPath = `${encType(type)}/${year}/${number}`;
|
|
1115
|
+
const url = `${BASE}/${docPath}/section/${encodeURIComponent(section)}${seg}/data.xml`;
|
|
1116
|
+
|
|
1117
|
+
let got;
|
|
1118
|
+
try {
|
|
1119
|
+
got = await getTextBounded(url, SECTION_MAX_BYTES);
|
|
1120
|
+
} catch (e) {
|
|
1121
|
+
const msg = e instanceof Error ? e.message : String(e);
|
|
1122
|
+
if (/\b404\b/.test(msg)) {
|
|
1123
|
+
return {
|
|
1124
|
+
found: false,
|
|
1125
|
+
reason: 'section_not_found',
|
|
1126
|
+
hint: `No section ${section} in ${docPath}${seg ? ` for version ${label}` : ''}. Numbering differs between versions — a section inserted by a later amendment does not exist in the "enacted" text. Call get_legislation for the section list.`,
|
|
1127
|
+
source: 'legislation.gov.uk',
|
|
1128
|
+
};
|
|
1129
|
+
}
|
|
1130
|
+
throw e;
|
|
1131
|
+
}
|
|
1132
|
+
|
|
1133
|
+
// Capture groups are load-bearing: firstGroup returns m[1], so a regex
|
|
1134
|
+
// without one silently yields undefined and falls through to the whole
|
|
1135
|
+
// document — which is how the amended text came back as 69 characters of
|
|
1136
|
+
// headings while the as-enacted text, structured differently, looked fine.
|
|
1137
|
+
const body = firstGroup(got.text, /(<P1group[\s\S]*<\/P1group>)/)
|
|
1138
|
+
?? firstGroup(got.text, /(<Pblock[\s\S]*<\/Pblock>)/)
|
|
1139
|
+
?? firstGroup(got.text, /(<Body[\s\S]*<\/Body>)/)
|
|
1140
|
+
?? got.text;
|
|
1141
|
+
const text = xmlToText(body);
|
|
1142
|
+
const servedDate = versionValidFrom(got.text) ?? null;
|
|
1143
|
+
|
|
1144
|
+
return {
|
|
1145
|
+
found: text.length > 0,
|
|
1146
|
+
legislation: docPath,
|
|
1147
|
+
title: decode(firstGroup(got.text, /<dc:title>([\s\S]*?)<\/dc:title>/)),
|
|
1148
|
+
section,
|
|
1149
|
+
// Stated on EVERY response — the same section reads differently across
|
|
1150
|
+
// versions and a quote without its version is not verifiable.
|
|
1151
|
+
version: label,
|
|
1152
|
+
as_at: requestedDate,
|
|
1153
|
+
// The date the SERVED text actually took effect. legislation.gov.uk
|
|
1154
|
+
// silently serves the nearest EARLIER version for a date with no
|
|
1155
|
+
// amendment on it, so this can differ from `as_at` above — surface both
|
|
1156
|
+
// rather than letting a caller assume the exact date they asked for.
|
|
1157
|
+
version_valid_from: servedDate,
|
|
1158
|
+
version_note: requestedDate && servedDate && servedDate !== requestedDate
|
|
1159
|
+
? `Asked for ${requestedDate}; no change took effect exactly then, so legislation.gov.uk served the version in force from ${servedDate} (the nearest earlier version) instead.`
|
|
1160
|
+
: null,
|
|
1161
|
+
text,
|
|
1162
|
+
characters: text.length,
|
|
1163
|
+
truncated: got.truncated,
|
|
1164
|
+
url: `${BASE}/${docPath}/section/${section}${seg}`,
|
|
1165
|
+
data_url: url,
|
|
1166
|
+
source: 'legislation.gov.uk',
|
|
1167
|
+
note: 'Text of one section only. Repealed or omitted words appear as runs of dots in the amended text, exactly as legislation.gov.uk renders them.',
|
|
1168
|
+
};
|
|
1169
|
+
}
|
|
1170
|
+
|
|
1171
|
+
async function getLegislationText(args: Record<string, unknown>): Promise<unknown> {
|
|
1172
|
+
const type = reqStr(args, 'type', '"ukpga"');
|
|
1173
|
+
const year = reqNum(args, 'year', '2018');
|
|
1174
|
+
const number = reqNum(args, 'number', '12');
|
|
1175
|
+
const { seg, label, requestedDate } = resolveVersion(args);
|
|
1176
|
+
const docPath = `${encType(type)}/${year}/${number}`;
|
|
1177
|
+
const url = `${BASE}/${docPath}${seg}/data.xml`;
|
|
1178
|
+
|
|
1179
|
+
const got = await getTextBounded(url, ACT_MAX_BYTES);
|
|
1180
|
+
const text = xmlToText(got.text);
|
|
1181
|
+
const servedDate = versionValidFrom(got.text) ?? null;
|
|
1182
|
+
|
|
1183
|
+
return {
|
|
1184
|
+
found: text.length > 0,
|
|
1185
|
+
legislation: docPath,
|
|
1186
|
+
title: decode(firstGroup(got.text, /<dc:title>([\s\S]*?)<\/dc:title>/)),
|
|
1187
|
+
version: label,
|
|
1188
|
+
as_at: requestedDate,
|
|
1189
|
+
version_valid_from: servedDate,
|
|
1190
|
+
version_note: requestedDate && servedDate && servedDate !== requestedDate
|
|
1191
|
+
? `Asked for ${requestedDate}; no change took effect exactly then, so legislation.gov.uk served the version in force from ${servedDate} (the nearest earlier version) instead.`
|
|
1192
|
+
: null,
|
|
1193
|
+
text,
|
|
1194
|
+
characters: text.length,
|
|
1195
|
+
bytes_read: got.bytes,
|
|
1196
|
+
truncated: got.truncated,
|
|
1197
|
+
// Said plainly rather than left for the caller to infer from a cut-off
|
|
1198
|
+
// sentence. A large Act runs to megabytes and this deliberately does not
|
|
1199
|
+
// return all of it.
|
|
1200
|
+
truncation_note: got.truncated
|
|
1201
|
+
? `This Act is larger than the ${(ACT_MAX_BYTES / 1000).toFixed(0)}kB this tool will read in one call, so the text above stops partway. For a specific provision call get_legislation_section({type, year, number, section}), which returns that section complete.`
|
|
1202
|
+
: null,
|
|
1203
|
+
url: `${BASE}/${docPath}${seg}`,
|
|
1204
|
+
source: 'legislation.gov.uk',
|
|
1205
|
+
};
|
|
1206
|
+
}
|
|
1207
|
+
|
|
167
1208
|
async function getText(url: string): Promise<string> {
|
|
168
|
-
const res = await
|
|
1209
|
+
const res = await pwFetch(url, { headers: { Accept: 'application/atom+xml, application/xml', 'User-Agent': UA } });
|
|
169
1210
|
const body = await res.text();
|
|
170
1211
|
if (!res.ok) throw new Error(`legislation.gov.uk: ${res.status} ${body.slice(0, 200)}`);
|
|
171
1212
|
return body;
|
|
172
1213
|
}
|
|
173
1214
|
|
|
1215
|
+
// ── Amendments ("changes") ──────────────────────────────────────────────────
|
|
1216
|
+
//
|
|
1217
|
+
// /changes/affected/{type}/{year}/{number}/data.feed = changes made TO this
|
|
1218
|
+
// legislation by other instruments (what amended it).
|
|
1219
|
+
// /changes/affecting/{type}/{year}/{number}/data.feed = changes THIS
|
|
1220
|
+
// legislation makes TO other instruments (the reverse direction) — verified
|
|
1221
|
+
// live 2026-09-18; the task's suggested "/changes/made/..." 404s, "affecting"
|
|
1222
|
+
// is the real path.
|
|
1223
|
+
// Each <entry> wraps one <ukm:Effect> element carrying the change as
|
|
1224
|
+
// attributes (Type, Applied, Modified, Comments, the affected/affecting
|
|
1225
|
+
// year+number+URI) plus child elements naming the specific provisions.
|
|
1226
|
+
const AMENDMENTS_PAGE_SIZE = 50;
|
|
1227
|
+
|
|
1228
|
+
async function getLegislationAmendments(args: Record<string, unknown>): Promise<unknown> {
|
|
1229
|
+
const type = reqStr(args, 'type', '"ukpga"');
|
|
1230
|
+
const year = reqNum(args, 'year', '2018');
|
|
1231
|
+
const number = reqNum(args, 'number', '12');
|
|
1232
|
+
const dirRaw = typeof args.direction === 'string' ? args.direction.trim().toLowerCase() : 'affected';
|
|
1233
|
+
const direction = dirRaw === 'affecting' ? 'affecting' : 'affected';
|
|
1234
|
+
const page = Math.max(1, Math.trunc(numOrUndef(args.page) ?? 1));
|
|
1235
|
+
|
|
1236
|
+
const docPath = `${encType(type)}/${year}/${number}`;
|
|
1237
|
+
const params = new URLSearchParams();
|
|
1238
|
+
if (page > 1) params.set('page', String(page));
|
|
1239
|
+
const qs = params.toString();
|
|
1240
|
+
const url = `${BASE}/changes/${direction}/${docPath}/data.feed${qs ? `?${qs}` : ''}`;
|
|
1241
|
+
const xml = await getText(url);
|
|
1242
|
+
|
|
1243
|
+
const totalResults = numAttr(xml, /<openSearch:totalResults>\s*([0-9]+)\s*<\/openSearch:totalResults>/) ?? 0;
|
|
1244
|
+
const totalPages = numAttr(xml, /<leg:totalPages>\s*([0-9]+)\s*<\/leg:totalPages>/) ?? 1;
|
|
1245
|
+
const perPage = numAttr(xml, /<openSearch:itemsPerPage>\s*([0-9]+)\s*<\/openSearch:itemsPerPage>/) ?? AMENDMENTS_PAGE_SIZE;
|
|
1246
|
+
|
|
1247
|
+
const amendments: Array<Record<string, unknown>> = [];
|
|
1248
|
+
for (const entry of matchAll(xml, /<entry>([\s\S]*?)<\/entry>/g)) {
|
|
1249
|
+
const block = entry[1];
|
|
1250
|
+
const entryTitle = decode(firstGroup(block, /<title>([\s\S]*?)<\/title>/));
|
|
1251
|
+
const effectWhole = extractTag(block, 'ukm:Effect');
|
|
1252
|
+
if (!effectWhole) {
|
|
1253
|
+
amendments.push({ title: entryTitle, raw_excerpt: block.slice(0, 400) });
|
|
1254
|
+
continue;
|
|
1255
|
+
}
|
|
1256
|
+
const a = parseAttrs(effectWhole);
|
|
1257
|
+
const inForce = extractTag(effectWhole, 'ukm:InForce');
|
|
1258
|
+
amendments.push({
|
|
1259
|
+
title: entryTitle,
|
|
1260
|
+
type: a.Type ?? null,
|
|
1261
|
+
applied: a.Applied === 'true' ? true : a.Applied === 'false' ? false : null,
|
|
1262
|
+
comments: a.Comments ?? null,
|
|
1263
|
+
modified: a.Modified ?? null,
|
|
1264
|
+
affecting: {
|
|
1265
|
+
title: decode(firstGroup(effectWhole, /<ukm:AffectingTitle>([\s\S]*?)<\/ukm:AffectingTitle>/)) ?? null,
|
|
1266
|
+
year: a.AffectingYear ? Number(a.AffectingYear) : null,
|
|
1267
|
+
number: a.AffectingNumber ? Number(a.AffectingNumber) : null,
|
|
1268
|
+
class: a.AffectingClass ?? null,
|
|
1269
|
+
provisions: sectionRefs(effectWhole, 'AffectingProvisions'),
|
|
1270
|
+
uri: a.AffectingURI ?? null,
|
|
1271
|
+
},
|
|
1272
|
+
affected: {
|
|
1273
|
+
title: decode(firstGroup(effectWhole, /<ukm:AffectedTitle>([\s\S]*?)<\/ukm:AffectedTitle>/)) ?? null,
|
|
1274
|
+
year: a.AffectedYear ? Number(a.AffectedYear) : null,
|
|
1275
|
+
number: a.AffectedNumber ? Number(a.AffectedNumber) : null,
|
|
1276
|
+
class: a.AffectedClass ?? null,
|
|
1277
|
+
provisions: sectionRefs(effectWhole, 'AffectedProvisions'),
|
|
1278
|
+
uri: a.AffectedURI ?? null,
|
|
1279
|
+
},
|
|
1280
|
+
commencement_date: inForce ? (parseAttrs(inForce).Date ?? null) : null,
|
|
1281
|
+
wholly_in_force: inForce ? /wholly in force/i.test(parseAttrs(inForce).Qualification ?? '') : null,
|
|
1282
|
+
});
|
|
1283
|
+
}
|
|
1284
|
+
|
|
1285
|
+
return {
|
|
1286
|
+
source: 'legislation.gov.uk',
|
|
1287
|
+
legislation: docPath,
|
|
1288
|
+
direction,
|
|
1289
|
+
note: direction === 'affected'
|
|
1290
|
+
? 'Changes made TO this legislation by other instruments — what amended it. Pass direction: "affecting" for the reverse (what this legislation amends).'
|
|
1291
|
+
: 'Changes this legislation makes TO other instruments. Pass direction: "affected" (default) for the reverse (what amended this legislation).',
|
|
1292
|
+
page,
|
|
1293
|
+
per_page: perPage,
|
|
1294
|
+
total_results: totalResults,
|
|
1295
|
+
total_pages: totalPages,
|
|
1296
|
+
has_more: page < totalPages,
|
|
1297
|
+
next_page: page < totalPages ? page + 1 : null,
|
|
1298
|
+
// 50 is a PAGE, never the total — stated explicitly so a caller does not
|
|
1299
|
+
// mistake one page's row count for the whole change history.
|
|
1300
|
+
paging_note: `This is page ${page} of ${totalPages} (${totalResults} total recorded changes, ${perPage} per page). Pass page: ${page + 1} for more.`,
|
|
1301
|
+
count: amendments.length,
|
|
1302
|
+
amendments,
|
|
1303
|
+
feed_url: url,
|
|
1304
|
+
};
|
|
1305
|
+
}
|
|
1306
|
+
|
|
1307
|
+
// ── Explanatory Notes ───────────────────────────────────────────────────────
|
|
1308
|
+
//
|
|
1309
|
+
// No XML/data feed exists for notes — data.xml 404s (verified live). Only
|
|
1310
|
+
// HTML, and the template differs by the Act's age:
|
|
1311
|
+
// OLD template (e.g. Equality Act 2010): /notes alone returns the whole
|
|
1312
|
+
// commentary on one page, headings like
|
|
1313
|
+
// <h5 class="...ENCommentaryP1..."><a href="...">Section 5</a>: Title</h5>.
|
|
1314
|
+
// NEW template (e.g. Data Protection Act 2018): /notes is a PDF-only stub;
|
|
1315
|
+
// the real content is under /notes/contents, split into numbered
|
|
1316
|
+
// "divisions" (chapters), with per-section commentary living in whichever
|
|
1317
|
+
// division is titled "Commentary on provisions of Act" — headings like
|
|
1318
|
+
// <h4 class="HSubheading4">Section 5: Title</h4> (no anchor).
|
|
1319
|
+
const NOTES_MAX_BYTES = 1_500_000;
|
|
1320
|
+
|
|
1321
|
+
async function getHtmlBounded(url: string, maxBytes: number): Promise<{ text: string; truncated: boolean; bytes: number; status: number }> {
|
|
1322
|
+
const res = await pwFetch(url, { headers: { Accept: 'text/html,application/xhtml+xml', 'User-Agent': UA } });
|
|
1323
|
+
if (!res.body) {
|
|
1324
|
+
const whole = await res.text();
|
|
1325
|
+
return { text: whole.slice(0, maxBytes), truncated: whole.length > maxBytes, bytes: whole.length, status: res.status };
|
|
1326
|
+
}
|
|
1327
|
+
const reader = res.body.getReader();
|
|
1328
|
+
const chunks: Uint8Array[] = [];
|
|
1329
|
+
let total = 0;
|
|
1330
|
+
let truncated = false;
|
|
1331
|
+
for (;;) {
|
|
1332
|
+
const { done, value } = await reader.read();
|
|
1333
|
+
if (done) break;
|
|
1334
|
+
if (value) {
|
|
1335
|
+
total += value.byteLength;
|
|
1336
|
+
chunks.push(value);
|
|
1337
|
+
if (total >= maxBytes) { truncated = true; await reader.cancel(); break; }
|
|
1338
|
+
}
|
|
1339
|
+
}
|
|
1340
|
+
const buf = new Uint8Array(total);
|
|
1341
|
+
let off = 0;
|
|
1342
|
+
for (const c of chunks) { buf.set(c, off); off += c.byteLength; }
|
|
1343
|
+
return { text: new TextDecoder().decode(buf), truncated, bytes: total, status: res.status };
|
|
1344
|
+
}
|
|
1345
|
+
|
|
1346
|
+
/** Strip tags/scripts/entities from an HTML fragment; keep paragraph breaks. */
|
|
1347
|
+
function htmlToText(html: string): string {
|
|
1348
|
+
return html
|
|
1349
|
+
.replace(/<script[\s\S]*?<\/script>/gi, ' ')
|
|
1350
|
+
.replace(/<style[\s\S]*?<\/style>/gi, ' ')
|
|
1351
|
+
.replace(/<!--[\s\S]*?-->/g, ' ')
|
|
1352
|
+
.replace(/<\/(p|div|li|h[1-6]|tr)>/gi, '\n')
|
|
1353
|
+
.replace(/<br\s*\/?>/gi, '\n')
|
|
1354
|
+
.replace(/<[^>]+>/g, ' ')
|
|
1355
|
+
.replace(/</g, '<').replace(/>/g, '>').replace(/"/g, '"')
|
|
1356
|
+
.replace(/�?39;|'/g, "'").replace(/ /g, ' ')
|
|
1357
|
+
.replace(/&#(\d+);/g, (_m, d: string) => String.fromCharCode(Number(d)))
|
|
1358
|
+
.replace(/&/g, '&')
|
|
1359
|
+
.replace(/[ \t]+/g, ' ')
|
|
1360
|
+
.split('\n').map((l) => l.trim()).filter(Boolean).join('\n')
|
|
1361
|
+
.replace(/\n{3,}/g, '\n\n')
|
|
1362
|
+
.trim();
|
|
1363
|
+
}
|
|
1364
|
+
|
|
1365
|
+
/** Strip any nested tags from a heading's inner HTML, leaving plain text. */
|
|
1366
|
+
function stripTags(html: string): string {
|
|
1367
|
+
return decode(html.replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ')) ?? '';
|
|
1368
|
+
}
|
|
1369
|
+
|
|
1370
|
+
function normSection(s: string): string {
|
|
1371
|
+
return s.trim().toUpperCase().replace(/^S(ECTION)?\.?\s*/, '');
|
|
1372
|
+
}
|
|
1373
|
+
|
|
1374
|
+
/**
|
|
1375
|
+
* Find a "Section N: Title" heading (h1–h6, with or without an anchor
|
|
1376
|
+
* wrapping "Section N") and return the text between it and the next heading.
|
|
1377
|
+
* Covers both the old template (anchor-wrapped) and the new one (plain text).
|
|
1378
|
+
*/
|
|
1379
|
+
function extractSectionNotes(html: string, sectionId: string): { title: string | null; text: string; found: boolean } {
|
|
1380
|
+
const target = normSection(sectionId);
|
|
1381
|
+
const re = /<h[1-6][^>]*>([\s\S]*?)<\/h[1-6]>/gi;
|
|
1382
|
+
const all: Array<{ start: number; end: number; plain: string }> = [];
|
|
1383
|
+
let m: RegExpExecArray | null;
|
|
1384
|
+
while ((m = re.exec(html)) !== null) {
|
|
1385
|
+
all.push({ start: m.index, end: m.index + m[0].length, plain: stripTags(m[1]) });
|
|
1386
|
+
}
|
|
1387
|
+
// Only STRUCTURAL headings (Section/Part/Chapter/Schedule/Annex) bound a
|
|
1388
|
+
// section's commentary — a section's own sub-headings (e.g. "Effect",
|
|
1389
|
+
// "Background") sit at the SAME heading level as "Section N" in some
|
|
1390
|
+
// templates and would otherwise end the extraction one paragraph in.
|
|
1391
|
+
// Measured on Equality Act 2010 s.1: "Section 1: ..." is immediately
|
|
1392
|
+
// followed by a sibling "<h5>Effect</h5>" heading before any body text.
|
|
1393
|
+
const boundaries = all.filter((h) => /^(Section|Part|Chapter|Schedule|Annex)\b/i.test(h.plain));
|
|
1394
|
+
for (let i = 0; i < boundaries.length; i++) {
|
|
1395
|
+
const h = boundaries[i];
|
|
1396
|
+
const mm = /^Section\s+([0-9]+[A-Za-z]{0,3})\s*:?\s*(.*)$/i.exec(h.plain);
|
|
1397
|
+
if (!mm) continue;
|
|
1398
|
+
if (normSection(mm[1]) !== target) continue;
|
|
1399
|
+
const bodyStart = h.end;
|
|
1400
|
+
const bodyEnd = i + 1 < boundaries.length ? boundaries[i + 1].start : Math.min(html.length, bodyStart + 20_000);
|
|
1401
|
+
return { title: mm[2]?.trim() || null, text: htmlToText(html.slice(bodyStart, bodyEnd)), found: true };
|
|
1402
|
+
}
|
|
1403
|
+
return { title: null, text: '', found: false };
|
|
1404
|
+
}
|
|
1405
|
+
|
|
1406
|
+
/**
|
|
1407
|
+
* Every notes-division link on a page, deduped by division number, across
|
|
1408
|
+
* both templates: the new template links siblings RELATIVELY from inside
|
|
1409
|
+
* .../notes/division/1/index.htm (href="../2/index.htm"), the old template
|
|
1410
|
+
* links ABSOLUTELY from .../notes/contents (href="/ukpga/2010/15/notes/division/2").
|
|
1411
|
+
* Neither form necessarily contains the substring "notes/division", so both
|
|
1412
|
+
* shapes are matched explicitly rather than by one substring pattern.
|
|
1413
|
+
*/
|
|
1414
|
+
function extractDivisions(html: string, docPath: string): Array<{ number: string; title: string; url: string }> {
|
|
1415
|
+
const re = /<a\s+href="([^"]+)"[^>]*>([^<]*)<\/a>/gi;
|
|
1416
|
+
const seen = new Set<string>();
|
|
1417
|
+
const out: Array<{ number: string; title: string; url: string }> = [];
|
|
1418
|
+
let m: RegExpExecArray | null;
|
|
1419
|
+
while ((m = re.exec(html)) !== null) {
|
|
1420
|
+
const href = m[1];
|
|
1421
|
+
let number: string | undefined;
|
|
1422
|
+
let url: string | undefined;
|
|
1423
|
+
const rel = /^\.\.\/([0-9]+(?:\/[0-9]+)*)\/index\.htm$/.exec(href);
|
|
1424
|
+
const abs = /\/notes\/division\/([0-9]+(?:\/[0-9]+)*)(?:\/index\.htm)?\/?$/.exec(href);
|
|
1425
|
+
if (rel) {
|
|
1426
|
+
number = rel[1];
|
|
1427
|
+
url = `${BASE}/${docPath}/notes/division/${number}/index.htm`;
|
|
1428
|
+
} else if (abs) {
|
|
1429
|
+
number = abs[1];
|
|
1430
|
+
url = href.startsWith('http') ? href : `${BASE}${href}`;
|
|
1431
|
+
}
|
|
1432
|
+
if (!number || !url || seen.has(number)) continue;
|
|
1433
|
+
seen.add(number);
|
|
1434
|
+
out.push({ number, title: decode(m[2])?.trim() ?? '', url });
|
|
1435
|
+
}
|
|
1436
|
+
return out;
|
|
1437
|
+
}
|
|
1438
|
+
|
|
1439
|
+
async function getExplanatoryNotes(args: Record<string, unknown>): Promise<unknown> {
|
|
1440
|
+
const type = reqStr(args, 'type', '"ukpga"');
|
|
1441
|
+
const year = reqNum(args, 'year', '2018');
|
|
1442
|
+
const number = reqNum(args, 'number', '12');
|
|
1443
|
+
const sectionRaw = args.section;
|
|
1444
|
+
const section = typeof sectionRaw === 'string' && sectionRaw.trim() ? sectionRaw.trim() : undefined;
|
|
1445
|
+
const docPath = `${encType(type)}/${year}/${number}`;
|
|
1446
|
+
const NOT_LAW_NOTE = 'Explanatory Notes are written by the government department responsible for the Act to help a non-lawyer reader. They are NOT part of the law, were not endorsed by Parliament, and are written for the Act as enacted — they are not updated for later amendments.';
|
|
1447
|
+
|
|
1448
|
+
const flatUrl = `${BASE}/${docPath}/notes`;
|
|
1449
|
+
const flat = await getHtmlBounded(flatUrl, NOTES_MAX_BYTES);
|
|
1450
|
+
if (flat.status === 404) {
|
|
1451
|
+
return {
|
|
1452
|
+
found: false,
|
|
1453
|
+
reason: 'no_explanatory_notes',
|
|
1454
|
+
hint: 'legislation.gov.uk has no Explanatory Notes for this item. Appropriation, Consolidated Fund, Finance and Consolidation Acts never get them; older statutory instruments often do not either.',
|
|
1455
|
+
url: flatUrl,
|
|
1456
|
+
source: 'legislation.gov.uk',
|
|
1457
|
+
};
|
|
1458
|
+
}
|
|
1459
|
+
|
|
1460
|
+
// Old-template Acts put the whole commentary, section headings included,
|
|
1461
|
+
// on this one page — try extraction here before fetching anything else.
|
|
1462
|
+
const hasFlatSections = /<h[1-6][^>]*>\s*(?:<a[^>]*>[^<]*<\/a>\s*)?Section\s+[0-9]/i.test(flat.text);
|
|
1463
|
+
if (hasFlatSections) {
|
|
1464
|
+
if (section) {
|
|
1465
|
+
const hit = extractSectionNotes(flat.text, section);
|
|
1466
|
+
if (hit.found) {
|
|
1467
|
+
return {
|
|
1468
|
+
found: true,
|
|
1469
|
+
legislation: docPath,
|
|
1470
|
+
section: normSection(section),
|
|
1471
|
+
section_title: hit.title,
|
|
1472
|
+
text: hit.text,
|
|
1473
|
+
characters: hit.text.length,
|
|
1474
|
+
url: flatUrl,
|
|
1475
|
+
source: 'legislation.gov.uk',
|
|
1476
|
+
note: NOT_LAW_NOTE,
|
|
1477
|
+
};
|
|
1478
|
+
}
|
|
1479
|
+
return {
|
|
1480
|
+
found: false,
|
|
1481
|
+
reason: 'section_not_found_in_notes',
|
|
1482
|
+
hint: `No "Section ${section}" commentary heading found in the Explanatory Notes for ${docPath}. Not every section gets its own paragraph — some are grouped or unremarked.`,
|
|
1483
|
+
url: flatUrl,
|
|
1484
|
+
source: 'legislation.gov.uk',
|
|
1485
|
+
};
|
|
1486
|
+
}
|
|
1487
|
+
const introMatch = /What these notes do[\s\S]{0,4000}/i.exec(flat.text);
|
|
1488
|
+
return {
|
|
1489
|
+
found: true,
|
|
1490
|
+
legislation: docPath,
|
|
1491
|
+
note: `Overview only — pass "section" (e.g. "5") for one section's commentary. ${NOT_LAW_NOTE}`,
|
|
1492
|
+
intro_text: introMatch ? htmlToText(introMatch[0]).slice(0, 3000) : htmlToText(flat.text).slice(0, 3000),
|
|
1493
|
+
url: flatUrl,
|
|
1494
|
+
truncated: flat.truncated,
|
|
1495
|
+
source: 'legislation.gov.uk',
|
|
1496
|
+
};
|
|
1497
|
+
}
|
|
1498
|
+
|
|
1499
|
+
// New-template Acts: /notes is a PDF-only stub. Real content lives under
|
|
1500
|
+
// /notes/contents, split into numbered divisions.
|
|
1501
|
+
const contentsUrl = `${BASE}/${docPath}/notes/contents`;
|
|
1502
|
+
const contents = await getHtmlBounded(contentsUrl, NOTES_MAX_BYTES);
|
|
1503
|
+
if (contents.status === 404) {
|
|
1504
|
+
return {
|
|
1505
|
+
found: false,
|
|
1506
|
+
reason: 'no_explanatory_notes',
|
|
1507
|
+
hint: 'legislation.gov.uk has no Explanatory Notes for this item.',
|
|
1508
|
+
url: contentsUrl,
|
|
1509
|
+
source: 'legislation.gov.uk',
|
|
1510
|
+
};
|
|
1511
|
+
}
|
|
1512
|
+
const divisions = extractDivisions(contents.text, docPath);
|
|
1513
|
+
|
|
1514
|
+
if (!section) {
|
|
1515
|
+
const introMatch = /<article>([\s\S]*?)<\/article>/i.exec(contents.text);
|
|
1516
|
+
return {
|
|
1517
|
+
found: true,
|
|
1518
|
+
legislation: docPath,
|
|
1519
|
+
note: `Overview only — pass "section" (e.g. "5") for one section's commentary. ${NOT_LAW_NOTE}`,
|
|
1520
|
+
intro_text: htmlToText(introMatch ? introMatch[1] : contents.text).slice(0, 3000),
|
|
1521
|
+
divisions,
|
|
1522
|
+
url: contentsUrl,
|
|
1523
|
+
source: 'legislation.gov.uk',
|
|
1524
|
+
};
|
|
1525
|
+
}
|
|
1526
|
+
|
|
1527
|
+
const commentaryDivision = divisions.find((d) => /commentary|provision/i.test(d.title))
|
|
1528
|
+
?? divisions.find((d) => /clause/i.test(d.title));
|
|
1529
|
+
if (!commentaryDivision) {
|
|
1530
|
+
return {
|
|
1531
|
+
found: false,
|
|
1532
|
+
reason: 'commentary_division_not_found',
|
|
1533
|
+
hint: 'Could not identify a "Commentary on provisions" division in the Explanatory Notes table of contents for this item.',
|
|
1534
|
+
divisions,
|
|
1535
|
+
url: contentsUrl,
|
|
1536
|
+
source: 'legislation.gov.uk',
|
|
1537
|
+
};
|
|
1538
|
+
}
|
|
1539
|
+
const divPage = await getHtmlBounded(commentaryDivision.url, NOTES_MAX_BYTES);
|
|
1540
|
+
const hit = extractSectionNotes(divPage.text, section);
|
|
1541
|
+
if (hit.found) {
|
|
1542
|
+
return {
|
|
1543
|
+
found: true,
|
|
1544
|
+
legislation: docPath,
|
|
1545
|
+
section: normSection(section),
|
|
1546
|
+
section_title: hit.title,
|
|
1547
|
+
text: hit.text,
|
|
1548
|
+
characters: hit.text.length,
|
|
1549
|
+
url: commentaryDivision.url,
|
|
1550
|
+
source: 'legislation.gov.uk',
|
|
1551
|
+
note: NOT_LAW_NOTE,
|
|
1552
|
+
};
|
|
1553
|
+
}
|
|
1554
|
+
return {
|
|
1555
|
+
found: false,
|
|
1556
|
+
reason: 'section_not_found_in_notes',
|
|
1557
|
+
hint: `No "Section ${section}" commentary heading found in "${commentaryDivision.title}" (${commentaryDivision.url}). Not every section gets its own paragraph, and notes are not updated for later amendments.`,
|
|
1558
|
+
divisions,
|
|
1559
|
+
url: commentaryDivision.url,
|
|
1560
|
+
source: 'legislation.gov.uk',
|
|
1561
|
+
};
|
|
1562
|
+
}
|
|
1563
|
+
|
|
174
1564
|
// --- XML/Atom helpers (regex-based; no XML libs in the Worker) ---
|
|
175
1565
|
|
|
176
1566
|
function encType(type: string): string {
|
|
@@ -187,6 +1577,41 @@ function firstGroup(s: string, re: RegExp): string | undefined {
|
|
|
187
1577
|
return m ? m[1] : undefined;
|
|
188
1578
|
}
|
|
189
1579
|
|
|
1580
|
+
/** Whole element (self-closing or paired), by local name, e.g. "ukm:Effect". */
|
|
1581
|
+
function extractTag(s: string, tagName: string): string | undefined {
|
|
1582
|
+
const esc = tagName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
1583
|
+
const re = new RegExp(`<${esc}\\b[^>]*\\/>|<${esc}\\b[^>]*>[\\s\\S]*?<\\/${esc}>`, 'i');
|
|
1584
|
+
const m = re.exec(s);
|
|
1585
|
+
return m ? m[0] : undefined;
|
|
1586
|
+
}
|
|
1587
|
+
|
|
1588
|
+
/** key="value" attribute pairs off an element's opening tag. */
|
|
1589
|
+
function parseAttrs(elementXml: string): Record<string, string> {
|
|
1590
|
+
const openTag = firstGroupWhole(elementXml, /^<[A-Za-z0-9:]+\b([^>]*)>/) ?? elementXml;
|
|
1591
|
+
const out: Record<string, string> = {};
|
|
1592
|
+
for (const m of matchAll(openTag, /([A-Za-z][A-Za-z0-9]*)="([^"]*)"/g)) {
|
|
1593
|
+
out[m[1]] = decode(m[2]) ?? m[2];
|
|
1594
|
+
}
|
|
1595
|
+
return out;
|
|
1596
|
+
}
|
|
1597
|
+
|
|
1598
|
+
function firstGroupWhole(s: string, re: RegExp): string | undefined {
|
|
1599
|
+
const m = re.exec(s);
|
|
1600
|
+
return m ? m[1] : undefined;
|
|
1601
|
+
}
|
|
1602
|
+
|
|
1603
|
+
/** Text of every nested <ukm:Section>/<ukm:Part>/etc inside a named wrapper element, e.g. "AffectedProvisions". */
|
|
1604
|
+
function sectionRefs(effectXml: string, wrapperLocalName: string): string[] {
|
|
1605
|
+
const wrapper = extractTag(effectXml, `ukm:${wrapperLocalName}`);
|
|
1606
|
+
if (!wrapper) return [];
|
|
1607
|
+
const out: string[] = [];
|
|
1608
|
+
for (const m of matchAll(wrapper, /<ukm:[A-Za-z]+\b[^>]*>([^<]*)<\/ukm:[A-Za-z]+>/g)) {
|
|
1609
|
+
const t = decode(m[1]);
|
|
1610
|
+
if (t) out.push(t);
|
|
1611
|
+
}
|
|
1612
|
+
return out;
|
|
1613
|
+
}
|
|
1614
|
+
|
|
190
1615
|
function attr(s: string, re: RegExp): string | undefined {
|
|
191
1616
|
const v = firstGroup(s, re);
|
|
192
1617
|
return v ? decode(v) : undefined;
|