@stratta/mcp 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/scripts/ingest-prepass.py +42 -3
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stratta/mcp",
|
|
3
3
|
"mcpName": "io.github.hugogebel-boop/stratta",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.5.0",
|
|
5
5
|
"description": "MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients via Stratta TreeRAG.",
|
|
6
6
|
"license": "UNLICENSED",
|
|
7
7
|
"author": "Hugo Gebel <hugo.gebel@epfl.ch>",
|
|
@@ -74,8 +74,19 @@ def detect_running_text(doc: "fitz.Document") -> set[str]:
|
|
|
74
74
|
return {l for l, c in counter.items() if c >= 5 and len(l) > 8}
|
|
75
75
|
|
|
76
76
|
|
|
77
|
-
|
|
78
|
-
"
|
|
77
|
+
CHAP_UPPER_RE = re.compile(
|
|
78
|
+
r"^(\d+)\s+([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{3,120})\s*$", re.MULTILINE
|
|
79
|
+
)
|
|
80
|
+
ANNEX_BODY_RE = re.compile(
|
|
81
|
+
r"^(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)$", re.MULTILINE
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def extract_chapters(
|
|
86
|
+
doc: "fitz.Document", toc_pages: set[int] | None = None
|
|
87
|
+
) -> dict[str, dict]:
|
|
88
|
+
"""Top-level chapters: prefer PDF bookmarks; fallback to body-text regex
|
|
89
|
+
(uppercase heading on its own line) when the PDF has no bookmarks."""
|
|
79
90
|
chapters: dict[str, dict] = {}
|
|
80
91
|
for _depth, title, page in doc.get_toc(simple=True):
|
|
81
92
|
m = BM_ANNEX_RE.match(title)
|
|
@@ -91,6 +102,34 @@ def extract_chapters(doc: "fitz.Document") -> dict[str, dict]:
|
|
|
91
102
|
if m:
|
|
92
103
|
num, rest = m.group(1), m.group(2).strip()
|
|
93
104
|
chapters[num] = {"title": rest, "pageStart": page}
|
|
105
|
+
if chapters:
|
|
106
|
+
return chapters
|
|
107
|
+
|
|
108
|
+
# Fallback: no bookmarks → scan body for uppercase chapter headings.
|
|
109
|
+
skip = toc_pages or set()
|
|
110
|
+
for pageno in range(doc.page_count):
|
|
111
|
+
if pageno in skip:
|
|
112
|
+
continue
|
|
113
|
+
text = doc[pageno].get_text("text")
|
|
114
|
+
for m in CHAP_UPPER_RE.finditer(text):
|
|
115
|
+
num = m.group(1)
|
|
116
|
+
title = clean(m.group(2))
|
|
117
|
+
if int(num) > 50:
|
|
118
|
+
continue
|
|
119
|
+
if num in chapters:
|
|
120
|
+
continue
|
|
121
|
+
chapters[num] = {"title": title, "pageStart": pageno + 1}
|
|
122
|
+
for m in ANNEX_BODY_RE.finditer(text):
|
|
123
|
+
letter = m.group(1)
|
|
124
|
+
kind = m.group(2) or ""
|
|
125
|
+
rest = clean(m.group(3) or "")
|
|
126
|
+
path = f"Annexe {letter}"
|
|
127
|
+
if path in chapters:
|
|
128
|
+
continue
|
|
129
|
+
t = path + (f" ({kind})" if kind else "")
|
|
130
|
+
if rest and rest.lower() != path.lower():
|
|
131
|
+
t += f" — {rest}"
|
|
132
|
+
chapters[path] = {"title": t, "pageStart": pageno + 1}
|
|
94
133
|
return chapters
|
|
95
134
|
|
|
96
135
|
|
|
@@ -254,7 +293,7 @@ def main() -> None:
|
|
|
254
293
|
doc = fitz.open(str(pdf_path))
|
|
255
294
|
toc_pages = detect_toc_pages(doc)
|
|
256
295
|
running = detect_running_text(doc)
|
|
257
|
-
chapters = extract_chapters(doc)
|
|
296
|
+
chapters = extract_chapters(doc, toc_pages)
|
|
258
297
|
chap_prefixes = {k for k in chapters if k.isdigit()}
|
|
259
298
|
subsections = extract_subsections(doc, chap_prefixes, toc_pages, running)
|
|
260
299
|
nodes = build_tree(chapters, subsections, doc.page_count)
|