@stratta/mcp 0.12.1 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -2
- package/dist/index.js +7 -1
- package/dist/tools/catalog.gen.js +58 -0
- package/dist/tools/site.d.ts +53 -0
- package/dist/tools/site.js +100 -0
- package/package.json +1 -1
- package/scripts/ingest-prepass.py +82 -0
package/README.md
CHANGED
|
@@ -34,7 +34,7 @@ https://stratta.ch/mcp
|
|
|
34
34
|
That is the shorter path, the one the dashboard walks you through for Claude,
|
|
35
35
|
ChatGPT, Claude Code, Codex, Cursor, VS Code, Gemini CLI and Windsurf, and the
|
|
36
36
|
only one that works in an agent running in the cloud (claude.ai, ChatGPT). It
|
|
37
|
-
serves
|
|
37
|
+
serves 35 of the 36 tools below, **ingestion included**: the pre-pass script is
|
|
38
38
|
downloaded from https://stratta.ch/ingest-prepass.py when it is not on disk.
|
|
39
39
|
See https://stratta.ch/docs/en/guides/connect-remote.
|
|
40
40
|
|
|
@@ -176,7 +176,7 @@ named `E2E dossier <timestamp>` and does not delete it.
|
|
|
176
176
|
|
|
177
177
|
## Tools exposed
|
|
178
178
|
|
|
179
|
-
|
|
179
|
+
36 tools, generated from one catalogue shared with the remote connector (which
|
|
180
180
|
serves the same 33 minus `add_attachment`).
|
|
181
181
|
|
|
182
182
|
**Read** (10 tools — query the norms of your workspace):
|
|
@@ -194,6 +194,14 @@ serves the same 33 minus `add_attachment`).
|
|
|
194
194
|
| `get_figure` | Retrieve a figure inline (base64 ImageContent). |
|
|
195
195
|
| `get_cross_refs` | Outgoing cross-refs from a section to other norms. |
|
|
196
196
|
|
|
197
|
+
**Site** (3 tools — what public Swiss registers know about a plot):
|
|
198
|
+
|
|
199
|
+
| Tool | Purpose |
|
|
200
|
+
| ------------------ | ----------------------------------------------------------------------------------------------------------------------------------- |
|
|
201
|
+
| `scan_site` | Collect municipality, parcel, elevation, geology, nearby boreholes, polluted sites, hazards and noise class around a Swiss address. |
|
|
202
|
+
| `get_site_context` | Read what the scan settled, and — separately — what it could not, with the reason. |
|
|
203
|
+
| `get_boreholes` | Read the boreholes nearest the site with their logged strata, SIA 261 ground class, water table, and links to cantonal documents. |
|
|
204
|
+
|
|
197
205
|
**Dossier** (13 tools — keep what was decided on a project):
|
|
198
206
|
|
|
199
207
|
| Tool | Purpose |
|
package/dist/index.js
CHANGED
|
@@ -12,6 +12,7 @@ import { confirmWithUser } from './confirm.js';
|
|
|
12
12
|
import { readTools } from './tools/read.js';
|
|
13
13
|
import { ingestTools } from './tools/ingest.js';
|
|
14
14
|
import { dossierTools } from './tools/dossier.js';
|
|
15
|
+
import { siteTools } from './tools/site.js';
|
|
15
16
|
import { SESSION_ID, usageDigest } from './usage.js';
|
|
16
17
|
import { SERVER_INSTRUCTIONS } from './tools/catalog.gen.js';
|
|
17
18
|
import { registerResources } from './resources.js';
|
|
@@ -102,7 +103,12 @@ async function call(def, args) {
|
|
|
102
103
|
};
|
|
103
104
|
}
|
|
104
105
|
}
|
|
105
|
-
for (const def of [
|
|
106
|
+
for (const def of [
|
|
107
|
+
...readTools,
|
|
108
|
+
...siteTools,
|
|
109
|
+
...dossierTools,
|
|
110
|
+
...ingestTools,
|
|
111
|
+
]) {
|
|
106
112
|
server.registerTool(def.name, {
|
|
107
113
|
title: def.title,
|
|
108
114
|
description: def.description,
|
|
@@ -194,6 +194,64 @@ export const CATALOG = [
|
|
|
194
194
|
"figureId": "Figure identifier from the figures array of get_section."
|
|
195
195
|
}
|
|
196
196
|
},
|
|
197
|
+
{
|
|
198
|
+
"name": "scan_site",
|
|
199
|
+
"title": "Scan a site",
|
|
200
|
+
"description": "Collect what public Swiss registers know about a building site: municipality, parcel, ground elevation, geology, nearby boreholes with their strata, polluted sites, groundwater protection, natural hazards, noise sensitivity. Give a Swiss address; it is geocoded server-side against swisstopo, so do not pass coordinates you inferred. Open the dossier first with open_dossier: one site per dossier. The scan runs in the background and takes a few seconds; read the result with get_site_context. It is a survey of public registers, never a geotechnical study.",
|
|
201
|
+
"annotations": {
|
|
202
|
+
"readOnlyHint": false,
|
|
203
|
+
"destructiveHint": false,
|
|
204
|
+
"idempotentHint": false,
|
|
205
|
+
"openWorldHint": true
|
|
206
|
+
},
|
|
207
|
+
"transports": [
|
|
208
|
+
"stdio",
|
|
209
|
+
"remote"
|
|
210
|
+
],
|
|
211
|
+
"params": {
|
|
212
|
+
"dossierId": "Dossier id returned by open_dossier or list_dossiers.",
|
|
213
|
+
"address": "The street address in Switzerland, as the user wrote it: \"Av. de Rhodanie 58, Lausanne\".",
|
|
214
|
+
"radius": "How far around the site to look for boreholes and constraints, in metres (50 to 500, default 300)."
|
|
215
|
+
}
|
|
216
|
+
},
|
|
217
|
+
{
|
|
218
|
+
"name": "get_boreholes",
|
|
219
|
+
"title": "Read the boreholes of a site",
|
|
220
|
+
"description": "Read the boreholes a site scan found, nearest first, with their logged strata (depth interval, description, period), depth, year, SIA 261 ground class when logged, water table when logged, and links to the cantonal documents. Call it after get_site_context when the question turns on what the ground is actually made of. Each borehole is one observation at one point: quote it with its distance, never interpolate between two of them, and never derive a design value from a layer description.",
|
|
221
|
+
"annotations": {
|
|
222
|
+
"readOnlyHint": true,
|
|
223
|
+
"destructiveHint": false,
|
|
224
|
+
"idempotentHint": true,
|
|
225
|
+
"openWorldHint": false
|
|
226
|
+
},
|
|
227
|
+
"transports": [
|
|
228
|
+
"stdio",
|
|
229
|
+
"remote"
|
|
230
|
+
],
|
|
231
|
+
"params": {
|
|
232
|
+
"siteId": "Site id returned by scan_site.",
|
|
233
|
+
"limit": "How many boreholes to return, nearest first (1 to 40, default 8). Ask for more only when the nearest ones did not answer.",
|
|
234
|
+
"withStrataOnly": "Return only the boreholes whose strata were read. Use it when the layers are the point; the count of what was left out still travels in `total`."
|
|
235
|
+
}
|
|
236
|
+
},
|
|
237
|
+
{
|
|
238
|
+
"name": "get_site_context",
|
|
239
|
+
"title": "Read a site scan",
|
|
240
|
+
"description": "Read what a site scan found, grouped by topic, each value with its source and the date THE DATA was current. Returns two separate lists: what the registers settled, and what they could NOT settle, with the reason. Report the second list to the user in full: it is what a bid has to budget for. Never derive a design value (friction angle, bearing capacity, modulus) from a geological description returned here; state what the register says and that a local investigation is required.",
|
|
241
|
+
"annotations": {
|
|
242
|
+
"readOnlyHint": true,
|
|
243
|
+
"destructiveHint": false,
|
|
244
|
+
"idempotentHint": true,
|
|
245
|
+
"openWorldHint": false
|
|
246
|
+
},
|
|
247
|
+
"transports": [
|
|
248
|
+
"stdio",
|
|
249
|
+
"remote"
|
|
250
|
+
],
|
|
251
|
+
"params": {
|
|
252
|
+
"siteId": "Site id returned by scan_site."
|
|
253
|
+
}
|
|
254
|
+
},
|
|
197
255
|
{
|
|
198
256
|
"name": "list_dossiers",
|
|
199
257
|
"title": "Project dossiers",
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export declare const scanSite: import("./define.js").ToolDef<{
|
|
3
|
+
dossierId: z.ZodString;
|
|
4
|
+
address: z.ZodString;
|
|
5
|
+
/**
|
|
6
|
+
* ⚠️ Bounded here as well as on the server. The server clamps rather than
|
|
7
|
+
* refusing, so a model asking for five kilometres would silently get five
|
|
8
|
+
* hundred metres and describe a radius it did not receive.
|
|
9
|
+
*/
|
|
10
|
+
radius: z.ZodOptional<z.ZodNumber>;
|
|
11
|
+
}>;
|
|
12
|
+
export declare const getSiteContext: import("./define.js").ToolDef<{
|
|
13
|
+
siteId: z.ZodString;
|
|
14
|
+
}>;
|
|
15
|
+
/**
|
|
16
|
+
* The boreholes, with their logs.
|
|
17
|
+
*
|
|
18
|
+
* ⚠️ A separate call rather than a bigger `get_site_context`: a Geneva radius
|
|
19
|
+
* holds a hundred and eight boreholes, and pouring their logs into every read
|
|
20
|
+
* of the fiche would make the cheap call expensive for every question that
|
|
21
|
+
* never looks at a log.
|
|
22
|
+
*/
|
|
23
|
+
export declare const getBoreholes: import("./define.js").ToolDef<{
|
|
24
|
+
siteId: z.ZodString;
|
|
25
|
+
/**
|
|
26
|
+
* ⚠️ Bounded here as well as on the server. The server clamps rather than
|
|
27
|
+
* refusing, so a model asking for two hundred would silently get forty and
|
|
28
|
+
* describe a list it did not receive.
|
|
29
|
+
*/
|
|
30
|
+
limit: z.ZodOptional<z.ZodNumber>;
|
|
31
|
+
withStrataOnly: z.ZodOptional<z.ZodBoolean>;
|
|
32
|
+
}>;
|
|
33
|
+
export declare const siteTools: (import("./define.js").ToolDef<{
|
|
34
|
+
dossierId: z.ZodString;
|
|
35
|
+
address: z.ZodString;
|
|
36
|
+
/**
|
|
37
|
+
* ⚠️ Bounded here as well as on the server. The server clamps rather than
|
|
38
|
+
* refusing, so a model asking for five kilometres would silently get five
|
|
39
|
+
* hundred metres and describe a radius it did not receive.
|
|
40
|
+
*/
|
|
41
|
+
radius: z.ZodOptional<z.ZodNumber>;
|
|
42
|
+
}> | import("./define.js").ToolDef<{
|
|
43
|
+
siteId: z.ZodString;
|
|
44
|
+
}> | import("./define.js").ToolDef<{
|
|
45
|
+
siteId: z.ZodString;
|
|
46
|
+
/**
|
|
47
|
+
* ⚠️ Bounded here as well as on the server. The server clamps rather than
|
|
48
|
+
* refusing, so a model asking for two hundred would silently get forty and
|
|
49
|
+
* describe a list it did not receive.
|
|
50
|
+
*/
|
|
51
|
+
limit: z.ZodOptional<z.ZodNumber>;
|
|
52
|
+
withStrataOnly: z.ZodOptional<z.ZodBoolean>;
|
|
53
|
+
}>)[];
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { api } from '../client.js';
|
|
3
|
+
import { requireApiKey } from '../auth.js';
|
|
4
|
+
import { catalogEntry, param } from './catalog.gen.js';
|
|
5
|
+
import { defineTool } from './define.js';
|
|
6
|
+
/**
|
|
7
|
+
* The three site tools: what public Swiss registers know about a plot.
|
|
8
|
+
*
|
|
9
|
+
* `scan_site` starts the collection, `get_site_context` reads what came back.
|
|
10
|
+
* They are split because the scan queries a dozen federal and cantonal
|
|
11
|
+
* services and takes a few seconds; a single blocking tool would either time
|
|
12
|
+
* out or return half a fiche.
|
|
13
|
+
*
|
|
14
|
+
* ⚠️ `get_site_context` returns what could NOT be established as a separate,
|
|
15
|
+
* first-class list. That list is the point: "no borehole within three hundred
|
|
16
|
+
* metres, molasse likely but unconfirmed, budget for three drillings" is the
|
|
17
|
+
* sentence worth money when pricing a bid, and an agent that only reads the
|
|
18
|
+
* settled facts will present the fiche as complete.
|
|
19
|
+
*/
|
|
20
|
+
const headOf = (name) => {
|
|
21
|
+
const entry = catalogEntry(name);
|
|
22
|
+
return {
|
|
23
|
+
name,
|
|
24
|
+
title: entry.title,
|
|
25
|
+
description: entry.description,
|
|
26
|
+
annotations: entry.annotations,
|
|
27
|
+
};
|
|
28
|
+
};
|
|
29
|
+
export const scanSite = defineTool({
|
|
30
|
+
...headOf('scan_site'),
|
|
31
|
+
inputSchema: {
|
|
32
|
+
dossierId: z.string().min(1).describe(param('scan_site', 'dossierId')),
|
|
33
|
+
address: z.string().min(3).max(200).describe(param('scan_site', 'address')),
|
|
34
|
+
/**
|
|
35
|
+
* ⚠️ Bounded here as well as on the server. The server clamps rather than
|
|
36
|
+
* refusing, so a model asking for five kilometres would silently get five
|
|
37
|
+
* hundred metres and describe a radius it did not receive.
|
|
38
|
+
*/
|
|
39
|
+
radius: z
|
|
40
|
+
.number()
|
|
41
|
+
.int()
|
|
42
|
+
.min(50)
|
|
43
|
+
.max(500)
|
|
44
|
+
.optional()
|
|
45
|
+
.describe(param('scan_site', 'radius')),
|
|
46
|
+
},
|
|
47
|
+
run: (client, args) => client.action(api.sitesApi.scanSite, {
|
|
48
|
+
apiKey: requireApiKey(),
|
|
49
|
+
dossierId: args.dossierId,
|
|
50
|
+
address: args.address,
|
|
51
|
+
radius: args.radius,
|
|
52
|
+
}),
|
|
53
|
+
});
|
|
54
|
+
export const getSiteContext = defineTool({
|
|
55
|
+
...headOf('get_site_context'),
|
|
56
|
+
inputSchema: {
|
|
57
|
+
siteId: z.string().min(1).describe(param('get_site_context', 'siteId')),
|
|
58
|
+
},
|
|
59
|
+
run: (client, args) => client.action(api.sitesApi.getSiteContext, {
|
|
60
|
+
apiKey: requireApiKey(),
|
|
61
|
+
siteId: args.siteId,
|
|
62
|
+
}),
|
|
63
|
+
});
|
|
64
|
+
/**
|
|
65
|
+
* The boreholes, with their logs.
|
|
66
|
+
*
|
|
67
|
+
* ⚠️ A separate call rather than a bigger `get_site_context`: a Geneva radius
|
|
68
|
+
* holds a hundred and eight boreholes, and pouring their logs into every read
|
|
69
|
+
* of the fiche would make the cheap call expensive for every question that
|
|
70
|
+
* never looks at a log.
|
|
71
|
+
*/
|
|
72
|
+
export const getBoreholes = defineTool({
|
|
73
|
+
...headOf('get_boreholes'),
|
|
74
|
+
inputSchema: {
|
|
75
|
+
siteId: z.string().min(1).describe(param('get_boreholes', 'siteId')),
|
|
76
|
+
/**
|
|
77
|
+
* ⚠️ Bounded here as well as on the server. The server clamps rather than
|
|
78
|
+
* refusing, so a model asking for two hundred would silently get forty and
|
|
79
|
+
* describe a list it did not receive.
|
|
80
|
+
*/
|
|
81
|
+
limit: z
|
|
82
|
+
.number()
|
|
83
|
+
.int()
|
|
84
|
+
.min(1)
|
|
85
|
+
.max(40)
|
|
86
|
+
.optional()
|
|
87
|
+
.describe(param('get_boreholes', 'limit')),
|
|
88
|
+
withStrataOnly: z
|
|
89
|
+
.boolean()
|
|
90
|
+
.optional()
|
|
91
|
+
.describe(param('get_boreholes', 'withStrataOnly')),
|
|
92
|
+
},
|
|
93
|
+
run: (client, args) => client.action(api.sitesApi.getBoreholes, {
|
|
94
|
+
apiKey: requireApiKey(),
|
|
95
|
+
siteId: args.siteId,
|
|
96
|
+
limit: args.limit,
|
|
97
|
+
withStrataOnly: args.withStrataOnly,
|
|
98
|
+
}),
|
|
99
|
+
});
|
|
100
|
+
export const siteTools = [scanSite, getSiteContext, getBoreholes];
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stratta/mcp",
|
|
3
3
|
"mcpName": "ch.stratta/mcp",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.13.0",
|
|
5
5
|
"description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
|
|
6
6
|
"license": "UNLICENSED",
|
|
7
7
|
"author": "SmartFlow <hello@stratta.ch>",
|
|
@@ -412,6 +412,14 @@ def extract_chapters(
|
|
|
412
412
|
per_page[meta["pageStart"]] += 1
|
|
413
413
|
return {p - 1 for p, n in per_page.items() if n >= 3}
|
|
414
414
|
|
|
415
|
+
# A règlement (SIA 144) is cut in articles, "Art. 7 Titre", a Eurocode
|
|
416
|
+
# in "Section 7 Titre". The passes below would take table rows for
|
|
417
|
+
# chapters and lose the whole text outside any section; a labelled
|
|
418
|
+
# layout is unambiguous, so it is read first and wins outright.
|
|
419
|
+
labelled = infer_labelled_chapters(doc, skip)
|
|
420
|
+
if len(labelled) >= 3:
|
|
421
|
+
return recover_missing_chapters(doc, prune_chapters(labelled), skip)
|
|
422
|
+
|
|
415
423
|
for pattern in (CHAP_UPPER_RE, CHAP_MIXED_RE):
|
|
416
424
|
chapters = scan(pattern, skip)
|
|
417
425
|
extra = toc_like_pages(chapters)
|
|
@@ -422,10 +430,77 @@ def extract_chapters(
|
|
|
422
430
|
|
|
423
431
|
for num, meta in infer_unnumbered_chapters(doc, skip).items():
|
|
424
432
|
chapters.setdefault(num, meta)
|
|
433
|
+
|
|
425
434
|
chapters = prune_chapters(chapters)
|
|
426
435
|
return recover_missing_chapters(doc, chapters, skip)
|
|
427
436
|
|
|
428
437
|
|
|
438
|
+
LABEL_RE = re.compile(
|
|
439
|
+
r"^(Art\.?|Section|Chapitre|Kapitel|Abschnitt)\s*(\d{1,3})\b[\s:_.\u2013-]*(.*?)(?:\s+(\d{1,3}))?$"
|
|
440
|
+
)
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def infer_labelled_chapters(doc: "fitz.Document", skip: set[int]) -> dict[str, dict]:
|
|
444
|
+
"""Chapters from labelled headings: "Art. 7 Titre" (a règlement, SIA 144)
|
|
445
|
+
or "Section 7 Titre" (a Eurocode adopted as SIA 267.001).
|
|
446
|
+
|
|
447
|
+
Two sources, because neither is complete on its own:
|
|
448
|
+
|
|
449
|
+
- the **contents page** (three or more labelled lines on one page) gives
|
|
450
|
+
clean titles, and for a règlement the printed page number
|
|
451
|
+
("Art. 1 But et modalités de l'appel d'offres 6");
|
|
452
|
+
- the **body** gives where each heading really starts. In a règlement the
|
|
453
|
+
OCR merges the margin column with the text ("Art. 1 1.1 L'enjeu…"), so
|
|
454
|
+
the body knows the page but not the title; in a Eurocode the body line
|
|
455
|
+
is "Section 2 _ Bases du calcul geotechnique", usable but often without
|
|
456
|
+
its accents.
|
|
457
|
+
|
|
458
|
+
A number seen in both takes its title from the contents and its page from
|
|
459
|
+
the body. A number seen only in the contents takes the printed page plus
|
|
460
|
+
the offset the pairs establish; only in the body, the body's title.
|
|
461
|
+
"""
|
|
462
|
+
contents: dict[str, tuple[str, int | None]] = {}
|
|
463
|
+
body: dict[str, tuple[str, int]] = {}
|
|
464
|
+
for pageno in range(doc.page_count):
|
|
465
|
+
if pageno in skip:
|
|
466
|
+
continue
|
|
467
|
+
hits: list[tuple[str, str, int | None]] = []
|
|
468
|
+
for line in (l.strip() for l in page_lines(doc, pageno)):
|
|
469
|
+
m = LABEL_RE.match(line)
|
|
470
|
+
if not m:
|
|
471
|
+
continue
|
|
472
|
+
title = clean(m.group(3))
|
|
473
|
+
printed = int(m.group(4)) if m.group(4) else None
|
|
474
|
+
hits.append((m.group(2), title, printed))
|
|
475
|
+
if len({h[0] for h in hits}) >= 3:
|
|
476
|
+
for num, title, printed in hits:
|
|
477
|
+
if title and plausible_title(title):
|
|
478
|
+
contents.setdefault(num, (title, printed))
|
|
479
|
+
else:
|
|
480
|
+
for num, title, printed in hits:
|
|
481
|
+
body.setdefault(num, (title, pageno + 1))
|
|
482
|
+
|
|
483
|
+
offsets = sorted(
|
|
484
|
+
body[num][1] - printed
|
|
485
|
+
for num, (_t, printed) in contents.items()
|
|
486
|
+
if printed is not None and num in body and body[num][1] > printed
|
|
487
|
+
)
|
|
488
|
+
offset = offsets[len(offsets) // 2] if offsets else None
|
|
489
|
+
|
|
490
|
+
found: dict[str, dict] = {}
|
|
491
|
+
for num in sorted(set(contents) | set(body), key=int):
|
|
492
|
+
c_title, printed = contents.get(num, ("", None))
|
|
493
|
+
b_title, b_page = body.get(num, ("", None))
|
|
494
|
+
title = c_title or b_title
|
|
495
|
+
page = b_page
|
|
496
|
+
if page is None and printed is not None and offset is not None:
|
|
497
|
+
page = printed + offset
|
|
498
|
+
if not title or page is None or not (1 <= page <= doc.page_count):
|
|
499
|
+
continue
|
|
500
|
+
found[num] = {"title": title, "pageStart": page}
|
|
501
|
+
return found
|
|
502
|
+
|
|
503
|
+
|
|
429
504
|
SCOPE_TITLE_RE = re.compile(
|
|
430
505
|
r"^(DOMAINE D.APPLICATION|GELTUNGSBEREICH|CAMPO D.APPLICAZIONE|SCOPE)\b", re.I
|
|
431
506
|
)
|
|
@@ -721,6 +796,13 @@ def locate_heading(lines: list[str], node: dict) -> int:
|
|
|
721
796
|
if line.strip().lower().startswith(needle):
|
|
722
797
|
return i
|
|
723
798
|
return -1
|
|
799
|
+
if path.isdigit():
|
|
800
|
+
labelled = re.compile(
|
|
801
|
+
r"^(?:Art\.?|Section|Chapitre|Kapitel|Abschnitt)\s*" + path + r"\b"
|
|
802
|
+
)
|
|
803
|
+
for i, line in enumerate(lines):
|
|
804
|
+
if labelled.match(line.strip()):
|
|
805
|
+
return i
|
|
724
806
|
prefix = path + " "
|
|
725
807
|
for i, line in enumerate(lines):
|
|
726
808
|
s = line.strip()
|