@stratta/mcp 0.9.7 → 0.9.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -20,6 +20,25 @@ MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients
20
20
  > a shell environment variable. If a key leaks, revoke it immediately at
21
21
  > https://stratta.ch/api-keys.
22
22
 
23
+ ## Do you need this package?
24
+
25
+ Often not. Stratta also runs as a **remote connector** — one address, a browser
26
+ sign-in, no key and no Node:
27
+
28
+ ```
29
+ https://stratta.ch/mcp
30
+ ```
31
+
32
+ That is the shorter path, and the only one that works in an agent running in the
33
+ cloud (claude.ai). See https://stratta.ch/docs/en/guides/connect-remote.
34
+
35
+ This package is what you want when:
36
+
37
+ - you are **ingesting a norm** — it reads a PDF from your disk and runs a Python
38
+ pre-pass, neither of which a remote connector can reach;
39
+ - your agent **cannot open a browser** — CI, a scheduled task, a server;
40
+ - you would simply rather run the server yourself.
41
+
23
42
  ## Install
24
43
 
25
44
  ### Claude Code (recommended)
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@stratta/mcp",
3
3
  "mcpName": "ch.stratta/mcp",
4
- "version": "0.9.7",
4
+ "version": "0.9.8",
5
5
  "description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
6
6
  "license": "UNLICENSED",
7
7
  "author": "SmartFlow <hello@stratta.ch>",
@@ -77,10 +77,163 @@ def detect_running_text(doc: "fitz.Document") -> set[str]:
77
77
  CHAP_UPPER_RE = re.compile(
78
78
  r"^(\d+)\s+([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{3,120})\s*$", re.MULTILINE
79
79
  )
80
+ # Norms adopted from CEN (SIA 262.6xx, SIA 267.1xx) set their headings in
81
+ # sentence case, which CHAP_UPPER_RE rejects by design. Used only as a second
82
+ # pass, when the uppercase form found almost nothing — on a native SIA norm it
83
+ # would promote body sentences to chapters.
84
+ CHAP_MIXED_RE = re.compile(
85
+ r"^(\d{1,2})\s+([A-ZÀ-Þ][A-Za-zÀ-ÿ][^\n]{2,90})\s*$", re.MULTILINE
86
+ )
80
87
  ANNEX_BODY_RE = re.compile(
81
88
  r"^(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)$", re.MULTILINE
82
89
  )
83
90
 
91
+ # A heading split across two lines: the number alone, the title underneath.
92
+ # Comes from the numbering column of CEN-style layouts, where PyMuPDF reads the
93
+ # column before the text. Left as-is, every regex below misses the heading.
94
+ SPLIT_NUM_RE = re.compile(r"^\s*(\d{1,2}(?:\.\d{1,2}){0,3})\s*$")
95
+ SPLIT_TITLE_RE = re.compile(r"^\s*([A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-][^\n]{2,90})\s*$")
96
+ # Words that open a continuing sentence, never a heading.
97
+ SPLIT_STOP_RE = re.compile(
98
+ r"^(Le |La |Les |Il |Elle |Dans |Pour |Selon |Cette |Ce |Ces |Si |En |Au |Aux |De |Des |Du |Un |Une )",
99
+ )
100
+
101
+
102
+ def join_split_headings(text: str) -> str:
103
+ """Rewrite `12\\nTitre` as `12 Titre` so the heading regexes can see it.
104
+
105
+ Conservative on purpose: the title line must look like a title (starts
106
+ uppercase, no sentence-ending punctuation, not a sentence opener). A false
107
+ join invents a chapter, which is worse than missing one.
108
+ """
109
+ lines = text.splitlines()
110
+ out: list[str] = []
111
+ i = 0
112
+ while i < len(lines):
113
+ m = SPLIT_NUM_RE.match(lines[i])
114
+ if m and i + 1 < len(lines):
115
+ nxt = lines[i + 1]
116
+ t = SPLIT_TITLE_RE.match(nxt)
117
+ if t and not nxt.rstrip().endswith((".", ",", ";", ":")) and not SPLIT_STOP_RE.match(t.group(1)):
118
+ out.append(f"{m.group(1)} {t.group(1).strip()}")
119
+ i += 2
120
+ continue
121
+ out.append(lines[i])
122
+ i += 1
123
+ return "\n".join(out)
124
+
125
+
126
+ def heading_text(doc: "fitz.Document", pageno: int) -> str:
127
+ """Page text prepared for heading detection (never for section content)."""
128
+ return join_split_headings(doc[pageno].get_text("text"))
129
+
130
+
131
+ NOT_A_TITLE_RE = re.compile(
132
+ r"^(?:EN|SN|ISO|DIN|SIA|NOTE|Tableau|Figure|Table|Bild)\b|^\W|\.$", re.IGNORECASE
133
+ )
134
+ # A heading cut mid-phrase by the column break: "Résistance à la flexion au".
135
+ TRUNCATED_RE = re.compile(
136
+ r"\b(?:ou|et|au|aux|de|des|du|le|la|les|un|une|dans|pour|par|sur|avec|sans|selon|entre)$",
137
+ re.IGNORECASE,
138
+ )
139
+
140
+
141
+ def plausible_title(title: str) -> bool:
142
+ """Reject what a numbered line can be other than a heading.
143
+
144
+ A stray table cell ("I ~ ~"), a normative reference ("EN 12063:2024 (F)") or
145
+ a wrapped sentence all match the heading shape. Promoting one invents a
146
+ chapter, and an invented chapter swallows the page range of a real one.
147
+ """
148
+ t = title.strip()
149
+ if not 3 <= len(t) <= 90:
150
+ return False
151
+ # A normative heading is capitalised. A lowercase one is a table row that
152
+ # happens to sit behind a number ("2.3 retrait des obstacles").
153
+ if not (t[0].isupper() or t[0].isdigit()):
154
+ return False
155
+ if NOT_A_TITLE_RE.search(t):
156
+ return False
157
+ if SPLIT_STOP_RE.match(t) or TRUNCATED_RE.search(t):
158
+ return False
159
+ letters = sum(1 for c in t if c.isalpha() or c.isspace())
160
+ return letters / len(t) >= 0.7
161
+
162
+
163
+ def dense_heading_pages(doc: "fitz.Document", toc_pages: set[int]) -> set[int]:
164
+ """Pages listing many numbered headings: a contents page, whatever its
165
+ typography. Detecting it by leader dots alone misses the ones set without
166
+ them, and every heading read there points at the contents page instead of
167
+ the section it names."""
168
+ dense: set[int] = set()
169
+ for i in range(doc.page_count):
170
+ if i in toc_pages:
171
+ continue
172
+ found = {m.group(1) for m in SUB_RE.finditer(heading_text(doc, i))}
173
+ # A page of the body drills into one chapter; a contents page walks
174
+ # across several. Counting headings alone would drop a dense page of
175
+ # definitions (3.1 … 3.8), which is real content.
176
+ if len(found) >= 5 and len({n.split(".", 1)[0] for n in found}) >= 3:
177
+ dense.add(i)
178
+ return dense
179
+
180
+
181
+ def prune_chapters(chapters: dict[str, dict]) -> dict[str, dict]:
182
+ """Keep the run of chapter numbers that reads like a table of contents.
183
+
184
+ Real chapters are consecutive and move forward through the document. A lone
185
+ "25" between chapters 1 and 2, or a chapter that starts thirty pages before
186
+ the one preceding it, is a detection artefact.
187
+ """
188
+ numbered = sorted(((int(k), k) for k in chapters if k.isdigit()))
189
+ if not numbered:
190
+ return chapters
191
+
192
+ # A norm with N detected chapters does not have a chapter 40. Missing a few
193
+ # headings is normal; a number far past the count is a table cell.
194
+ ceiling = 2 * len(numbered) + 3
195
+ numbered = [(n, k) for n, k in numbered if n <= ceiling]
196
+ if not numbered:
197
+ return chapters
198
+
199
+ # Longest run whose pages move forward, so one bad page does not discard
200
+ # every chapter after it.
201
+ best = [1] * len(numbered)
202
+ prev = [-1] * len(numbered)
203
+ for i in range(len(numbered)):
204
+ page_i = chapters[numbered[i][1]]["pageStart"]
205
+ for j in range(i):
206
+ if chapters[numbered[j][1]]["pageStart"] <= page_i and best[j] + 1 > best[i]:
207
+ best[i], prev[i] = best[j] + 1, j
208
+ idx = best.index(max(best))
209
+ chain = []
210
+ while idx != -1:
211
+ chain.append(numbered[idx][1])
212
+ idx = prev[idx]
213
+
214
+ # Missing a heading or two leaves a small gap; doubling (10 → 20 → 30) means
215
+ # the numbers stopped being chapters and started being table rows.
216
+ kept: dict[str, dict] = {}
217
+ previous = None
218
+ for key in reversed(chain):
219
+ n = int(key)
220
+ if previous is not None and n - previous > 3 and n >= 2 * previous:
221
+ break
222
+ kept[key] = chapters[key]
223
+ previous = n
224
+
225
+ # Annexes close a norm, in order. One that lands before the last chapter was
226
+ # read off the contents page, and its page range would swallow the document.
227
+ last_page = max((m["pageStart"] for m in kept.values()), default=0)
228
+ for key, meta in sorted(
229
+ ((k, m) for k, m in chapters.items() if not k.isdigit()), key=lambda kv: kv[0]
230
+ ):
231
+ if meta["pageStart"] < last_page:
232
+ continue
233
+ kept[key] = meta
234
+ last_page = meta["pageStart"]
235
+ return kept
236
+
84
237
 
85
238
  def extract_chapters(
86
239
  doc: "fitz.Document", toc_pages: set[int] | None = None
@@ -105,32 +258,64 @@ def extract_chapters(
105
258
  if chapters:
106
259
  return chapters
107
260
 
108
- # Fallback: no bookmarks → scan body for uppercase chapter headings.
109
- skip = toc_pages or set()
110
- for pageno in range(doc.page_count):
111
- if pageno in skip:
112
- continue
113
- text = doc[pageno].get_text("text")
114
- for m in CHAP_UPPER_RE.finditer(text):
115
- num = m.group(1)
116
- title = clean(m.group(2))
117
- if int(num) > 50:
118
- continue
119
- if num in chapters:
120
- continue
121
- chapters[num] = {"title": title, "pageStart": pageno + 1}
122
- for m in ANNEX_BODY_RE.finditer(text):
123
- letter = m.group(1)
124
- kind = m.group(2) or ""
125
- rest = clean(m.group(3) or "")
126
- path = f"Annexe {letter}"
127
- if path in chapters:
261
+ # Fallback: no bookmarks → scan body for chapter headings.
262
+ skip = set(toc_pages or set())
263
+
264
+ def scan(pattern: "re.Pattern[str]", skip_pages: set[int]) -> dict[str, dict]:
265
+ found: dict[str, dict] = {}
266
+ for pageno in range(doc.page_count):
267
+ if pageno in skip_pages:
128
268
  continue
129
- t = path + (f" ({kind})" if kind else "")
130
- if rest and rest.lower() != path.lower():
131
- t += f" — {rest}"
132
- chapters[path] = {"title": t, "pageStart": pageno + 1}
133
- return chapters
269
+ text = heading_text(doc, pageno)
270
+ for m in pattern.finditer(text):
271
+ num, title = m.group(1), clean(m.group(2))
272
+ if int(num) > 50 or num in found or not plausible_title(title):
273
+ continue
274
+ found[num] = {"title": title, "pageStart": pageno + 1}
275
+ for m in ANNEX_BODY_RE.finditer(text):
276
+ letter, kind = m.group(1), m.group(2) or ""
277
+ rest = clean(m.group(3) or "")
278
+ path = f"Annexe {letter}"
279
+ if path in found:
280
+ continue
281
+ t = path + (f" ({kind})" if kind else "")
282
+ if rest and rest.lower() != path.lower():
283
+ t += f" — {rest}"
284
+ found[path] = {"title": t, "pageStart": pageno + 1}
285
+ return found
286
+
287
+ def toc_like_pages(found: dict[str, dict]) -> set[int]:
288
+ """Pages holding three or more chapter headings are the table of
289
+ contents, whatever the typography. Without this every chapter of such a
290
+ norm starts on the contents page, and every section body is wrong."""
291
+ per_page: dict[int, int] = defaultdict(int)
292
+ for meta in found.values():
293
+ per_page[meta["pageStart"]] += 1
294
+ return {p - 1 for p, n in per_page.items() if n >= 3}
295
+
296
+ for pattern in (CHAP_UPPER_RE, CHAP_MIXED_RE):
297
+ chapters = scan(pattern, skip)
298
+ extra = toc_like_pages(chapters)
299
+ if extra:
300
+ chapters = scan(pattern, skip | extra)
301
+ if len([k for k in chapters if k.isdigit()]) >= 3:
302
+ break
303
+
304
+ return prune_chapters(chapters)
305
+
306
+
307
+ def chapter_page_spans(chapters: dict[str, dict], n_pages: int) -> dict[str, tuple[int, int]]:
308
+ """First and last page of each numbered chapter, from where the next starts."""
309
+ ordered = sorted(
310
+ ((k, v["pageStart"]) for k, v in chapters.items()),
311
+ key=lambda kv: kv[1],
312
+ )
313
+ spans: dict[str, tuple[int, int]] = {}
314
+ for idx, (key, start) in enumerate(ordered):
315
+ end = ordered[idx + 1][1] - 1 if idx + 1 < len(ordered) else n_pages
316
+ if key.isdigit():
317
+ spans[key] = (start, max(start, end))
318
+ return spans
134
319
 
135
320
 
136
321
  def extract_subsections(
@@ -138,13 +323,15 @@ def extract_subsections(
138
323
  chap_prefixes: set[str],
139
324
  toc_pages: set[int],
140
325
  running: set[str],
326
+ chapter_spans: dict[str, tuple[int, int]] | None = None,
141
327
  ) -> dict[str, tuple[str, int]]:
142
328
  """Numeric sub-sections (depths 2-4) detected in body pages."""
143
329
  out: dict[str, tuple[str, int]] = {}
330
+ skip = set(toc_pages) | dense_heading_pages(doc, toc_pages)
144
331
  for i in range(doc.page_count):
145
- if i in toc_pages:
332
+ if i in skip:
146
333
  continue
147
- text = doc[i].get_text("text")
334
+ text = heading_text(doc, i)
148
335
  for m in SUB_RE.finditer(text):
149
336
  num = m.group(1)
150
337
  title = clean(m.group(2))
@@ -153,7 +340,13 @@ def extract_subsections(
153
340
  top = num.split(".", 1)[0]
154
341
  if top not in chap_prefixes:
155
342
  continue
156
- if len(title) < 3 or re.fullmatch(r"[^A-Za-zÀ-ÿ]+", title):
343
+ # A sub-section sits inside its chapter. "1.1" found sixty pages
344
+ # after chapter 1 ended is a numbered line in an annexe, and keeping
345
+ # it hangs an unrelated page under the wrong parent.
346
+ span = chapter_spans.get(top) if chapter_spans else None
347
+ if span and not span[0] <= i + 1 <= span[1]:
348
+ continue
349
+ if not plausible_title(title):
157
350
  continue
158
351
  if num in out:
159
352
  continue
@@ -218,6 +411,44 @@ def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
218
411
  chunks = page_texts[ps - 1 : pe]
219
412
  node["rawText"] = "\n".join(chunks).strip()
220
413
 
414
+ trim_shared_pages(nodes, page_texts)
415
+
416
+
417
+ def trim_shared_pages(nodes: list[dict], page_texts: list[str]) -> None:
418
+ """Cut the opening page where several sections start on it.
419
+
420
+ Page granularity is the unit everywhere else, but two sections beginning on
421
+ the same page would otherwise carry byte-identical text — so a summary reads
422
+ like its neighbour's, and the table of contents misleads before anything is
423
+ even opened.
424
+ """
425
+ by_page: dict[int, list[dict]] = defaultdict(list)
426
+ for node in nodes:
427
+ by_page[node["pageStart"]].append(node)
428
+
429
+ for page, group in by_page.items():
430
+ if len(group) < 2:
431
+ continue
432
+ text = page_texts[page - 1]
433
+ # Where each section's heading sits on that page, in reading order.
434
+ marks: list[tuple[int, dict]] = []
435
+ for node in group:
436
+ title = node["title"].split(" — ")[-1].strip()
437
+ at = text.find(title) if len(title) >= 4 else -1
438
+ if at == -1:
439
+ at = text.find(f"{node['path']} ")
440
+ marks.append((at, node))
441
+ if any(at == -1 for at, _ in marks):
442
+ continue # a heading we cannot locate: leave the whole page in place
443
+ marks.sort(key=lambda m: m[0])
444
+ for i, (at, node) in enumerate(marks):
445
+ end = marks[i + 1][0] if i + 1 < len(marks) else len(text)
446
+ head = text[at:end].strip()
447
+ if not head:
448
+ continue
449
+ rest = "\n".join(page_texts[node["pageStart"] : node["pageEnd"]]).strip()
450
+ node["rawText"] = f"{head}\n{rest}".strip() if rest else head
451
+
221
452
 
222
453
  def extract_figures(
223
454
  doc: "fitz.Document",
@@ -295,7 +526,8 @@ def main() -> None:
295
526
  running = detect_running_text(doc)
296
527
  chapters = extract_chapters(doc, toc_pages)
297
528
  chap_prefixes = {k for k in chapters if k.isdigit()}
298
- subsections = extract_subsections(doc, chap_prefixes, toc_pages, running)
529
+ spans = chapter_page_spans(chapters, doc.page_count)
530
+ subsections = extract_subsections(doc, chap_prefixes, toc_pages, running, spans)
299
531
  nodes = build_tree(chapters, subsections, doc.page_count)
300
532
  extract_section_text(doc, nodes)
301
533
  figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)