@stratta/mcp 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@stratta/mcp",
3
3
  "mcpName": "io.github.hugogebel-boop/stratta",
4
- "version": "0.4.0",
4
+ "version": "0.5.0",
5
5
  "description": "MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients via Stratta TreeRAG.",
6
6
  "license": "UNLICENSED",
7
7
  "author": "Hugo Gebel <hugo.gebel@epfl.ch>",
@@ -74,8 +74,19 @@ def detect_running_text(doc: "fitz.Document") -> set[str]:
74
74
  return {l for l, c in counter.items() if c >= 5 and len(l) > 8}
75
75
 
76
76
 
77
- def extract_chapters(doc: "fitz.Document") -> dict[str, dict]:
78
- """Top-level chapters from PDF bookmarks (authoritative titles + pages)."""
77
+ CHAP_UPPER_RE = re.compile(
78
+ r"^(\d+)\s+([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{3,120})\s*$", re.MULTILINE
79
+ )
80
+ ANNEX_BODY_RE = re.compile(
81
+ r"^(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)$", re.MULTILINE
82
+ )
83
+
84
+
85
+ def extract_chapters(
86
+ doc: "fitz.Document", toc_pages: set[int] | None = None
87
+ ) -> dict[str, dict]:
88
+ """Top-level chapters: prefer PDF bookmarks; fallback to body-text regex
89
+ (uppercase heading on its own line) when the PDF has no bookmarks."""
79
90
  chapters: dict[str, dict] = {}
80
91
  for _depth, title, page in doc.get_toc(simple=True):
81
92
  m = BM_ANNEX_RE.match(title)
@@ -91,6 +102,34 @@ def extract_chapters(doc: "fitz.Document") -> dict[str, dict]:
91
102
  if m:
92
103
  num, rest = m.group(1), m.group(2).strip()
93
104
  chapters[num] = {"title": rest, "pageStart": page}
105
+ if chapters:
106
+ return chapters
107
+
108
+ # Fallback: no bookmarks → scan body for uppercase chapter headings.
109
+ skip = toc_pages or set()
110
+ for pageno in range(doc.page_count):
111
+ if pageno in skip:
112
+ continue
113
+ text = doc[pageno].get_text("text")
114
+ for m in CHAP_UPPER_RE.finditer(text):
115
+ num = m.group(1)
116
+ title = clean(m.group(2))
117
+ if int(num) > 50:
118
+ continue
119
+ if num in chapters:
120
+ continue
121
+ chapters[num] = {"title": title, "pageStart": pageno + 1}
122
+ for m in ANNEX_BODY_RE.finditer(text):
123
+ letter = m.group(1)
124
+ kind = m.group(2) or ""
125
+ rest = clean(m.group(3) or "")
126
+ path = f"Annexe {letter}"
127
+ if path in chapters:
128
+ continue
129
+ t = path + (f" ({kind})" if kind else "")
130
+ if rest and rest.lower() != path.lower():
131
+ t += f" — {rest}"
132
+ chapters[path] = {"title": t, "pageStart": pageno + 1}
94
133
  return chapters
95
134
 
96
135
 
@@ -254,7 +293,7 @@ def main() -> None:
254
293
  doc = fitz.open(str(pdf_path))
255
294
  toc_pages = detect_toc_pages(doc)
256
295
  running = detect_running_text(doc)
257
- chapters = extract_chapters(doc)
296
+ chapters = extract_chapters(doc, toc_pages)
258
297
  chap_prefixes = {k for k in chapters if k.isdigit()}
259
298
  subsections = extract_subsections(doc, chap_prefixes, toc_pages, running)
260
299
  nodes = build_tree(chapters, subsections, doc.page_count)