@stratta/mcp 0.12.1 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -34,7 +34,7 @@ https://stratta.ch/mcp
34
34
  That is the shorter path, the one the dashboard walks you through for Claude,
35
35
  ChatGPT, Claude Code, Codex, Cursor, VS Code, Gemini CLI and Windsurf, and the
36
36
  only one that works in an agent running in the cloud (claude.ai, ChatGPT). It
37
- serves 32 of the 33 tools below, **ingestion included**: the pre-pass script is
37
+ serves 35 of the 36 tools below, **ingestion included**: the pre-pass script is
38
38
  downloaded from https://stratta.ch/ingest-prepass.py when it is not on disk.
39
39
  See https://stratta.ch/docs/en/guides/connect-remote.
40
40
 
@@ -176,7 +176,7 @@ named `E2E dossier <timestamp>` and does not delete it.
176
176
 
177
177
  ## Tools exposed
178
178
 
179
- 33 tools, generated from one catalogue shared with the remote connector (which
179
+ 36 tools, generated from one catalogue shared with the remote connector (which
180
180
  serves the same 33 minus `add_attachment`).
181
181
 
182
182
  **Read** (10 tools — query the norms of your workspace):
@@ -194,6 +194,14 @@ serves the same 33 minus `add_attachment`).
194
194
  | `get_figure` | Retrieve a figure inline (base64 ImageContent). |
195
195
  | `get_cross_refs` | Outgoing cross-refs from a section to other norms. |
196
196
 
197
+ **Site** (3 tools — what public Swiss registers know about a plot):
198
+
199
+ | Tool | Purpose |
200
+ | ------------------ | ----------------------------------------------------------------------------------------------------------------------------------- |
201
+ | `scan_site` | Collect municipality, parcel, elevation, geology, nearby boreholes, polluted sites, hazards and noise class around a Swiss address. |
202
+ | `get_site_context` | Read what the scan settled, and — separately — what it could not, with the reason. |
203
+ | `get_boreholes` | Read the boreholes nearest the site with their logged strata, SIA 261 ground class, water table, and links to cantonal documents. |
204
+
197
205
  **Dossier** (13 tools — keep what was decided on a project):
198
206
 
199
207
  | Tool | Purpose |
package/dist/index.js CHANGED
@@ -12,6 +12,7 @@ import { confirmWithUser } from './confirm.js';
12
12
  import { readTools } from './tools/read.js';
13
13
  import { ingestTools } from './tools/ingest.js';
14
14
  import { dossierTools } from './tools/dossier.js';
15
+ import { siteTools } from './tools/site.js';
15
16
  import { SESSION_ID, usageDigest } from './usage.js';
16
17
  import { SERVER_INSTRUCTIONS } from './tools/catalog.gen.js';
17
18
  import { registerResources } from './resources.js';
@@ -102,7 +103,12 @@ async function call(def, args) {
102
103
  };
103
104
  }
104
105
  }
105
- for (const def of [...readTools, ...dossierTools, ...ingestTools]) {
106
+ for (const def of [
107
+ ...readTools,
108
+ ...siteTools,
109
+ ...dossierTools,
110
+ ...ingestTools,
111
+ ]) {
106
112
  server.registerTool(def.name, {
107
113
  title: def.title,
108
114
  description: def.description,
@@ -194,6 +194,64 @@ export const CATALOG = [
194
194
  "figureId": "Figure identifier from the figures array of get_section."
195
195
  }
196
196
  },
197
+ {
198
+ "name": "scan_site",
199
+ "title": "Scan a site",
200
+ "description": "Collect what public Swiss registers know about a building site: municipality, parcel, ground elevation, geology, nearby boreholes with their strata, polluted sites, groundwater protection, natural hazards, noise sensitivity. Give a Swiss address; it is geocoded server-side against swisstopo, so do not pass coordinates you inferred. Open the dossier first with open_dossier: one site per dossier. The scan runs in the background and takes a few seconds; read the result with get_site_context. It is a survey of public registers, never a geotechnical study.",
201
+ "annotations": {
202
+ "readOnlyHint": false,
203
+ "destructiveHint": false,
204
+ "idempotentHint": false,
205
+ "openWorldHint": true
206
+ },
207
+ "transports": [
208
+ "stdio",
209
+ "remote"
210
+ ],
211
+ "params": {
212
+ "dossierId": "Dossier id returned by open_dossier or list_dossiers.",
213
+ "address": "The street address in Switzerland, as the user wrote it: \"Av. de Rhodanie 58, Lausanne\".",
214
+ "radius": "How far around the site to look for boreholes and constraints, in metres (50 to 500, default 300)."
215
+ }
216
+ },
217
+ {
218
+ "name": "get_boreholes",
219
+ "title": "Read the boreholes of a site",
220
+ "description": "Read the boreholes a site scan found, nearest first, with their logged strata (depth interval, description, period), depth, year, SIA 261 ground class when logged, water table when logged, and links to the cantonal documents. Call it after get_site_context when the question turns on what the ground is actually made of. Each borehole is one observation at one point: quote it with its distance, never interpolate between two of them, and never derive a design value from a layer description.",
221
+ "annotations": {
222
+ "readOnlyHint": true,
223
+ "destructiveHint": false,
224
+ "idempotentHint": true,
225
+ "openWorldHint": false
226
+ },
227
+ "transports": [
228
+ "stdio",
229
+ "remote"
230
+ ],
231
+ "params": {
232
+ "siteId": "Site id returned by scan_site.",
233
+ "limit": "How many boreholes to return, nearest first (1 to 40, default 8). Ask for more only when the nearest ones did not answer.",
234
+ "withStrataOnly": "Return only the boreholes whose strata were read. Use it when the layers are the point; the count of what was left out still travels in `total`."
235
+ }
236
+ },
237
+ {
238
+ "name": "get_site_context",
239
+ "title": "Read a site scan",
240
+ "description": "Read what a site scan found, grouped by topic, each value with its source and the date THE DATA was current. Returns two separate lists: what the registers settled, and what they could NOT settle, with the reason. Report the second list to the user in full: it is what a bid has to budget for. Never derive a design value (friction angle, bearing capacity, modulus) from a geological description returned here; state what the register says and that a local investigation is required.",
241
+ "annotations": {
242
+ "readOnlyHint": true,
243
+ "destructiveHint": false,
244
+ "idempotentHint": true,
245
+ "openWorldHint": false
246
+ },
247
+ "transports": [
248
+ "stdio",
249
+ "remote"
250
+ ],
251
+ "params": {
252
+ "siteId": "Site id returned by scan_site."
253
+ }
254
+ },
197
255
  {
198
256
  "name": "list_dossiers",
199
257
  "title": "Project dossiers",
@@ -0,0 +1,53 @@
1
+ import { z } from 'zod';
2
+ export declare const scanSite: import("./define.js").ToolDef<{
3
+ dossierId: z.ZodString;
4
+ address: z.ZodString;
5
+ /**
6
+ * ⚠️ Bounded here as well as on the server. The server clamps rather than
7
+ * refusing, so a model asking for five kilometres would silently get five
8
+ * hundred metres and describe a radius it did not receive.
9
+ */
10
+ radius: z.ZodOptional<z.ZodNumber>;
11
+ }>;
12
+ export declare const getSiteContext: import("./define.js").ToolDef<{
13
+ siteId: z.ZodString;
14
+ }>;
15
+ /**
16
+ * The boreholes, with their logs.
17
+ *
18
+ * ⚠️ A separate call rather than a bigger `get_site_context`: a Geneva radius
19
+ * holds a hundred and eight boreholes, and pouring their logs into every read
20
+ * of the fiche would make the cheap call expensive for every question that
21
+ * never looks at a log.
22
+ */
23
+ export declare const getBoreholes: import("./define.js").ToolDef<{
24
+ siteId: z.ZodString;
25
+ /**
26
+ * ⚠️ Bounded here as well as on the server. The server clamps rather than
27
+ * refusing, so a model asking for two hundred would silently get forty and
28
+ * describe a list it did not receive.
29
+ */
30
+ limit: z.ZodOptional<z.ZodNumber>;
31
+ withStrataOnly: z.ZodOptional<z.ZodBoolean>;
32
+ }>;
33
+ export declare const siteTools: (import("./define.js").ToolDef<{
34
+ dossierId: z.ZodString;
35
+ address: z.ZodString;
36
+ /**
37
+ * ⚠️ Bounded here as well as on the server. The server clamps rather than
38
+ * refusing, so a model asking for five kilometres would silently get five
39
+ * hundred metres and describe a radius it did not receive.
40
+ */
41
+ radius: z.ZodOptional<z.ZodNumber>;
42
+ }> | import("./define.js").ToolDef<{
43
+ siteId: z.ZodString;
44
+ }> | import("./define.js").ToolDef<{
45
+ siteId: z.ZodString;
46
+ /**
47
+ * ⚠️ Bounded here as well as on the server. The server clamps rather than
48
+ * refusing, so a model asking for two hundred would silently get forty and
49
+ * describe a list it did not receive.
50
+ */
51
+ limit: z.ZodOptional<z.ZodNumber>;
52
+ withStrataOnly: z.ZodOptional<z.ZodBoolean>;
53
+ }>)[];
@@ -0,0 +1,100 @@
1
+ import { z } from 'zod';
2
+ import { api } from '../client.js';
3
+ import { requireApiKey } from '../auth.js';
4
+ import { catalogEntry, param } from './catalog.gen.js';
5
+ import { defineTool } from './define.js';
6
+ /**
7
+ * The three site tools: what public Swiss registers know about a plot.
8
+ *
9
+ * `scan_site` starts the collection, `get_site_context` reads what came back.
10
+ * They are split because the scan queries a dozen federal and cantonal
11
+ * services and takes a few seconds; a single blocking tool would either time
12
+ * out or return half a fiche.
13
+ *
14
+ * ⚠️ `get_site_context` returns what could NOT be established as a separate,
15
+ * first-class list. That list is the point: "no borehole within three hundred
16
+ * metres, molasse likely but unconfirmed, budget for three drillings" is the
17
+ * sentence worth money when pricing a bid, and an agent that only reads the
18
+ * settled facts will present the fiche as complete.
19
+ */
20
+ const headOf = (name) => {
21
+ const entry = catalogEntry(name);
22
+ return {
23
+ name,
24
+ title: entry.title,
25
+ description: entry.description,
26
+ annotations: entry.annotations,
27
+ };
28
+ };
29
+ export const scanSite = defineTool({
30
+ ...headOf('scan_site'),
31
+ inputSchema: {
32
+ dossierId: z.string().min(1).describe(param('scan_site', 'dossierId')),
33
+ address: z.string().min(3).max(200).describe(param('scan_site', 'address')),
34
+ /**
35
+ * ⚠️ Bounded here as well as on the server. The server clamps rather than
36
+ * refusing, so a model asking for five kilometres would silently get five
37
+ * hundred metres and describe a radius it did not receive.
38
+ */
39
+ radius: z
40
+ .number()
41
+ .int()
42
+ .min(50)
43
+ .max(500)
44
+ .optional()
45
+ .describe(param('scan_site', 'radius')),
46
+ },
47
+ run: (client, args) => client.action(api.sitesApi.scanSite, {
48
+ apiKey: requireApiKey(),
49
+ dossierId: args.dossierId,
50
+ address: args.address,
51
+ radius: args.radius,
52
+ }),
53
+ });
54
+ export const getSiteContext = defineTool({
55
+ ...headOf('get_site_context'),
56
+ inputSchema: {
57
+ siteId: z.string().min(1).describe(param('get_site_context', 'siteId')),
58
+ },
59
+ run: (client, args) => client.action(api.sitesApi.getSiteContext, {
60
+ apiKey: requireApiKey(),
61
+ siteId: args.siteId,
62
+ }),
63
+ });
64
+ /**
65
+ * The boreholes, with their logs.
66
+ *
67
+ * ⚠️ A separate call rather than a bigger `get_site_context`: a Geneva radius
68
+ * holds a hundred and eight boreholes, and pouring their logs into every read
69
+ * of the fiche would make the cheap call expensive for every question that
70
+ * never looks at a log.
71
+ */
72
+ export const getBoreholes = defineTool({
73
+ ...headOf('get_boreholes'),
74
+ inputSchema: {
75
+ siteId: z.string().min(1).describe(param('get_boreholes', 'siteId')),
76
+ /**
77
+ * ⚠️ Bounded here as well as on the server. The server clamps rather than
78
+ * refusing, so a model asking for two hundred would silently get forty and
79
+ * describe a list it did not receive.
80
+ */
81
+ limit: z
82
+ .number()
83
+ .int()
84
+ .min(1)
85
+ .max(40)
86
+ .optional()
87
+ .describe(param('get_boreholes', 'limit')),
88
+ withStrataOnly: z
89
+ .boolean()
90
+ .optional()
91
+ .describe(param('get_boreholes', 'withStrataOnly')),
92
+ },
93
+ run: (client, args) => client.action(api.sitesApi.getBoreholes, {
94
+ apiKey: requireApiKey(),
95
+ siteId: args.siteId,
96
+ limit: args.limit,
97
+ withStrataOnly: args.withStrataOnly,
98
+ }),
99
+ });
100
+ export const siteTools = [scanSite, getSiteContext, getBoreholes];
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@stratta/mcp",
3
3
  "mcpName": "ch.stratta/mcp",
4
- "version": "0.12.1",
4
+ "version": "0.13.0",
5
5
  "description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
6
6
  "license": "UNLICENSED",
7
7
  "author": "SmartFlow <hello@stratta.ch>",
@@ -412,6 +412,14 @@ def extract_chapters(
412
412
  per_page[meta["pageStart"]] += 1
413
413
  return {p - 1 for p, n in per_page.items() if n >= 3}
414
414
 
415
+ # A règlement (SIA 144) is cut in articles, "Art. 7 Titre", a Eurocode
416
+ # in "Section 7 Titre". The passes below would take table rows for
417
+ # chapters and lose the whole text outside any section; a labelled
418
+ # layout is unambiguous, so it is read first and wins outright.
419
+ labelled = infer_labelled_chapters(doc, skip)
420
+ if len(labelled) >= 3:
421
+ return recover_missing_chapters(doc, prune_chapters(labelled), skip)
422
+
415
423
  for pattern in (CHAP_UPPER_RE, CHAP_MIXED_RE):
416
424
  chapters = scan(pattern, skip)
417
425
  extra = toc_like_pages(chapters)
@@ -422,10 +430,77 @@ def extract_chapters(
422
430
 
423
431
  for num, meta in infer_unnumbered_chapters(doc, skip).items():
424
432
  chapters.setdefault(num, meta)
433
+
425
434
  chapters = prune_chapters(chapters)
426
435
  return recover_missing_chapters(doc, chapters, skip)
427
436
 
428
437
 
438
+ LABEL_RE = re.compile(
439
+ r"^(Art\.?|Section|Chapitre|Kapitel|Abschnitt)\s*(\d{1,3})\b[\s:_.\u2013-]*(.*?)(?:\s+(\d{1,3}))?$"
440
+ )
441
+
442
+
443
+ def infer_labelled_chapters(doc: "fitz.Document", skip: set[int]) -> dict[str, dict]:
444
+ """Chapters from labelled headings: "Art. 7 Titre" (a règlement, SIA 144)
445
+ or "Section 7 Titre" (a Eurocode adopted as SIA 267.001).
446
+
447
+ Two sources, because neither is complete on its own:
448
+
449
+ - the **contents page** (three or more labelled lines on one page) gives
450
+ clean titles, and for a règlement the printed page number
451
+ ("Art. 1 But et modalités de l'appel d'offres 6");
452
+ - the **body** gives where each heading really starts. In a règlement the
453
+ OCR merges the margin column with the text ("Art. 1 1.1 L'enjeu…"), so
454
+ the body knows the page but not the title; in a Eurocode the body line
455
+ is "Section 2 _ Bases du calcul geotechnique", usable but often without
456
+ its accents.
457
+
458
+ A number seen in both takes its title from the contents and its page from
459
+ the body. A number seen only in the contents takes the printed page plus
460
+ the offset the pairs establish; only in the body, the body's title.
461
+ """
462
+ contents: dict[str, tuple[str, int | None]] = {}
463
+ body: dict[str, tuple[str, int]] = {}
464
+ for pageno in range(doc.page_count):
465
+ if pageno in skip:
466
+ continue
467
+ hits: list[tuple[str, str, int | None]] = []
468
+ for line in (l.strip() for l in page_lines(doc, pageno)):
469
+ m = LABEL_RE.match(line)
470
+ if not m:
471
+ continue
472
+ title = clean(m.group(3))
473
+ printed = int(m.group(4)) if m.group(4) else None
474
+ hits.append((m.group(2), title, printed))
475
+ if len({h[0] for h in hits}) >= 3:
476
+ for num, title, printed in hits:
477
+ if title and plausible_title(title):
478
+ contents.setdefault(num, (title, printed))
479
+ else:
480
+ for num, title, printed in hits:
481
+ body.setdefault(num, (title, pageno + 1))
482
+
483
+ offsets = sorted(
484
+ body[num][1] - printed
485
+ for num, (_t, printed) in contents.items()
486
+ if printed is not None and num in body and body[num][1] > printed
487
+ )
488
+ offset = offsets[len(offsets) // 2] if offsets else None
489
+
490
+ found: dict[str, dict] = {}
491
+ for num in sorted(set(contents) | set(body), key=int):
492
+ c_title, printed = contents.get(num, ("", None))
493
+ b_title, b_page = body.get(num, ("", None))
494
+ title = c_title or b_title
495
+ page = b_page
496
+ if page is None and printed is not None and offset is not None:
497
+ page = printed + offset
498
+ if not title or page is None or not (1 <= page <= doc.page_count):
499
+ continue
500
+ found[num] = {"title": title, "pageStart": page}
501
+ return found
502
+
503
+
429
504
  SCOPE_TITLE_RE = re.compile(
430
505
  r"^(DOMAINE D.APPLICATION|GELTUNGSBEREICH|CAMPO D.APPLICAZIONE|SCOPE)\b", re.I
431
506
  )
@@ -721,6 +796,13 @@ def locate_heading(lines: list[str], node: dict) -> int:
721
796
  if line.strip().lower().startswith(needle):
722
797
  return i
723
798
  return -1
799
+ if path.isdigit():
800
+ labelled = re.compile(
801
+ r"^(?:Art\.?|Section|Chapitre|Kapitel|Abschnitt)\s*" + path + r"\b"
802
+ )
803
+ for i, line in enumerate(lines):
804
+ if labelled.match(line.strip()):
805
+ return i
724
806
  prefix = path + " "
725
807
  for i, line in enumerate(lines):
726
808
  s = line.strip()