@stratta/mcp 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -1
- package/scripts/ingest-prepass.py +310 -0
- package/skills/ingest-norm/SKILL.md +128 -75
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stratta/mcp",
|
|
3
3
|
"mcpName": "io.github.hugogebel-boop/stratta",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.4.0",
|
|
5
5
|
"description": "MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients via Stratta TreeRAG.",
|
|
6
6
|
"license": "UNLICENSED",
|
|
7
7
|
"author": "Hugo Gebel <hugo.gebel@epfl.ch>",
|
|
@@ -29,6 +29,7 @@
|
|
|
29
29
|
"files": [
|
|
30
30
|
"dist",
|
|
31
31
|
"skills",
|
|
32
|
+
"scripts",
|
|
32
33
|
"README.md"
|
|
33
34
|
],
|
|
34
35
|
"scripts": {
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Stratta — ingest pre-pass.
|
|
3
|
+
|
|
4
|
+
Reads a norm PDF with PyMuPDF, builds the hierarchical TreeRAG skeleton (chapters
|
|
5
|
+
from PDF bookmarks + sub-sections via heading regex), extracts per-section raw
|
|
6
|
+
text, and rasterizes each page that contains a figure caption. Output is a
|
|
7
|
+
single JSON consumed by the `ingest-norm` skill, which then enriches sections
|
|
8
|
+
(formulas, tables, cross-refs, summaries) and uploads the figures.
|
|
9
|
+
|
|
10
|
+
Usage:
|
|
11
|
+
python scripts/ingest-prepass.py --pdf <path> --output <dir> [--language fr]
|
|
12
|
+
|
|
13
|
+
Output (under <dir>/):
|
|
14
|
+
prepass.json full extracted tree + figure manifest
|
|
15
|
+
figures/figure-<N>.png rasterized full-page renders for each caption
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import argparse
|
|
21
|
+
import json
|
|
22
|
+
import re
|
|
23
|
+
import sys
|
|
24
|
+
from collections import defaultdict
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
try:
|
|
28
|
+
import fitz # PyMuPDF
|
|
29
|
+
except ImportError:
|
|
30
|
+
sys.stderr.write(
|
|
31
|
+
"ERROR: PyMuPDF not installed. Run: pip install --user pymupdf\n"
|
|
32
|
+
)
|
|
33
|
+
sys.exit(2)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
BM_NUM_RE = re.compile(r"^\s*(\d+)\s+(.+?)\s*$")
|
|
37
|
+
BM_ANNEX_RE = re.compile(
|
|
38
|
+
r"^\s*(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)\s*$"
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
SUB_RE = re.compile(
|
|
42
|
+
r"^(\d+(?:\.\d+){1,3})\s+([A-Za-zÀ-ÿ][^\n]{1,140})$", re.MULTILINE
|
|
43
|
+
)
|
|
44
|
+
TOC_DOT_RE = re.compile(r"\.\s*\.\s*\.")
|
|
45
|
+
CLEAN_DOTS = re.compile(r"\s*(\.\s*){3,}.*$")
|
|
46
|
+
|
|
47
|
+
CAPTION_RE = re.compile(
|
|
48
|
+
r"^\s*(?:Figure|Fig\.|Bild|Abbildung|Figura)\s+(\d+[a-z]?)\b[\s:.\-—–]*(.*)$",
|
|
49
|
+
re.IGNORECASE | re.MULTILINE,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def clean(s: str) -> str:
|
|
54
|
+
return re.sub(r"\s+", " ", CLEAN_DOTS.sub("", s)).strip()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def detect_toc_pages(doc: "fitz.Document") -> set[int]:
|
|
58
|
+
out = set()
|
|
59
|
+
for i in range(doc.page_count):
|
|
60
|
+
lines = [l for l in doc[i].get_text("text").splitlines() if l.strip()]
|
|
61
|
+
leader = sum(1 for l in lines if TOC_DOT_RE.search(l))
|
|
62
|
+
if leader >= 5:
|
|
63
|
+
out.add(i)
|
|
64
|
+
return out
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def detect_running_text(doc: "fitz.Document") -> set[str]:
|
|
68
|
+
"""Lines appearing on >=5 pages near the top/bottom (running headers/footers)."""
|
|
69
|
+
counter: dict[str, int] = defaultdict(int)
|
|
70
|
+
for i in range(doc.page_count):
|
|
71
|
+
lines = [l.strip() for l in doc[i].get_text("text").splitlines() if l.strip()]
|
|
72
|
+
for l in lines[:3] + lines[-3:]:
|
|
73
|
+
counter[l] += 1
|
|
74
|
+
return {l for l, c in counter.items() if c >= 5 and len(l) > 8}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def extract_chapters(doc: "fitz.Document") -> dict[str, dict]:
|
|
78
|
+
"""Top-level chapters from PDF bookmarks (authoritative titles + pages)."""
|
|
79
|
+
chapters: dict[str, dict] = {}
|
|
80
|
+
for _depth, title, page in doc.get_toc(simple=True):
|
|
81
|
+
m = BM_ANNEX_RE.match(title)
|
|
82
|
+
if m:
|
|
83
|
+
letter, kind, rest = m.group(1), m.group(2) or "", (m.group(3) or "").strip()
|
|
84
|
+
path = f"Annexe {letter}"
|
|
85
|
+
t = path + (f" ({kind})" if kind else "")
|
|
86
|
+
if rest and rest.lower() != path.lower():
|
|
87
|
+
t += f" — {rest}"
|
|
88
|
+
chapters[path] = {"title": t, "pageStart": page}
|
|
89
|
+
continue
|
|
90
|
+
m = BM_NUM_RE.match(title)
|
|
91
|
+
if m:
|
|
92
|
+
num, rest = m.group(1), m.group(2).strip()
|
|
93
|
+
chapters[num] = {"title": rest, "pageStart": page}
|
|
94
|
+
return chapters
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def extract_subsections(
|
|
98
|
+
doc: "fitz.Document",
|
|
99
|
+
chap_prefixes: set[str],
|
|
100
|
+
toc_pages: set[int],
|
|
101
|
+
running: set[str],
|
|
102
|
+
) -> dict[str, tuple[str, int]]:
|
|
103
|
+
"""Numeric sub-sections (depths 2-4) detected in body pages."""
|
|
104
|
+
out: dict[str, tuple[str, int]] = {}
|
|
105
|
+
for i in range(doc.page_count):
|
|
106
|
+
if i in toc_pages:
|
|
107
|
+
continue
|
|
108
|
+
text = doc[i].get_text("text")
|
|
109
|
+
for m in SUB_RE.finditer(text):
|
|
110
|
+
num = m.group(1)
|
|
111
|
+
title = clean(m.group(2))
|
|
112
|
+
if not title or title in running:
|
|
113
|
+
continue
|
|
114
|
+
top = num.split(".", 1)[0]
|
|
115
|
+
if top not in chap_prefixes:
|
|
116
|
+
continue
|
|
117
|
+
if len(title) < 3 or re.fullmatch(r"[^A-Za-zÀ-ÿ]+", title):
|
|
118
|
+
continue
|
|
119
|
+
if num in out:
|
|
120
|
+
continue
|
|
121
|
+
out[num] = (title, i + 1)
|
|
122
|
+
return out
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def build_tree(chapters: dict, subsections: dict, n_pages: int) -> list[dict]:
|
|
126
|
+
"""Returns flat list of nodes with depth 0-indexed (chapter=0, section=1, ...)."""
|
|
127
|
+
nodes: list[dict] = []
|
|
128
|
+
for path, info in chapters.items():
|
|
129
|
+
nodes.append(
|
|
130
|
+
{"depth": 0, "path": path, "title": info["title"], "pageStart": info["pageStart"]}
|
|
131
|
+
)
|
|
132
|
+
for path, (title, pg) in subsections.items():
|
|
133
|
+
nodes.append(
|
|
134
|
+
{"depth": path.count("."), "path": path, "title": title, "pageStart": pg}
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def sort_key(n):
|
|
138
|
+
p = n["path"]
|
|
139
|
+
if p.startswith("Annexe"):
|
|
140
|
+
return (10**9, ord(p[-1]), [])
|
|
141
|
+
parts = [int(x) for x in p.split(".")]
|
|
142
|
+
return (parts[0], 0, parts)
|
|
143
|
+
|
|
144
|
+
nodes.sort(key=sort_key)
|
|
145
|
+
|
|
146
|
+
# pageEnd via next sibling/ancestor
|
|
147
|
+
for idx, node in enumerate(nodes):
|
|
148
|
+
d, pg = node["depth"], node["pageStart"]
|
|
149
|
+
next_pg = n_pages
|
|
150
|
+
for j in range(idx + 1, len(nodes)):
|
|
151
|
+
if nodes[j]["depth"] <= d:
|
|
152
|
+
next_pg = max(nodes[j]["pageStart"] - 1, pg)
|
|
153
|
+
break
|
|
154
|
+
node["pageEnd"] = next_pg
|
|
155
|
+
|
|
156
|
+
# Parent links + nodeId + orderIndex
|
|
157
|
+
parent_stack: list[tuple[int, str]] = [] # (depth, nodeId)
|
|
158
|
+
for idx, node in enumerate(nodes):
|
|
159
|
+
d = node["depth"]
|
|
160
|
+
while parent_stack and parent_stack[-1][0] >= d:
|
|
161
|
+
parent_stack.pop()
|
|
162
|
+
node["nodeId"] = f"s-{slugify(node['path'])}"
|
|
163
|
+
node["parentNodeId"] = parent_stack[-1][1] if parent_stack else None
|
|
164
|
+
node["orderIndex"] = idx
|
|
165
|
+
parent_stack.append((d, node["nodeId"]))
|
|
166
|
+
|
|
167
|
+
return nodes
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def slugify(path: str) -> str:
|
|
171
|
+
return re.sub(r"[^a-zA-Z0-9]+", "-", path).strip("-").lower() or "root"
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
|
|
175
|
+
"""Attach rawText to each node = concat of pages it spans."""
|
|
176
|
+
page_texts = [doc[i].get_text("text") for i in range(doc.page_count)]
|
|
177
|
+
for node in nodes:
|
|
178
|
+
ps, pe = node["pageStart"], node["pageEnd"]
|
|
179
|
+
chunks = page_texts[ps - 1 : pe]
|
|
180
|
+
node["rawText"] = "\n".join(chunks).strip()
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def extract_figures(
|
|
184
|
+
doc: "fitz.Document",
|
|
185
|
+
toc_pages: set[int],
|
|
186
|
+
out_dir: Path,
|
|
187
|
+
dpi: int = 150,
|
|
188
|
+
) -> list[dict]:
|
|
189
|
+
"""One figure manifest entry per caption detected. Page is rendered to PNG."""
|
|
190
|
+
figs_dir = out_dir / "figures"
|
|
191
|
+
figs_dir.mkdir(parents=True, exist_ok=True)
|
|
192
|
+
seen: dict[str, dict] = {}
|
|
193
|
+
matrix = fitz.Matrix(dpi / 72, dpi / 72)
|
|
194
|
+
for pageno in range(doc.page_count):
|
|
195
|
+
if pageno in toc_pages:
|
|
196
|
+
continue
|
|
197
|
+
text = doc[pageno].get_text("text")
|
|
198
|
+
rendered = False
|
|
199
|
+
page_png: Path | None = None
|
|
200
|
+
for m in CAPTION_RE.finditer(text):
|
|
201
|
+
num = m.group(1)
|
|
202
|
+
label_rest = m.group(2).strip()
|
|
203
|
+
if num in seen:
|
|
204
|
+
continue
|
|
205
|
+
caption = f"Figure {num}" + (f" — {label_rest}" if label_rest else "")
|
|
206
|
+
# render the page once if not done
|
|
207
|
+
if not rendered:
|
|
208
|
+
pix = doc[pageno].get_pixmap(matrix=matrix, alpha=False)
|
|
209
|
+
page_png = figs_dir / f"page-{pageno + 1:03d}.png"
|
|
210
|
+
pix.save(str(page_png))
|
|
211
|
+
rendered = True
|
|
212
|
+
seen[num] = {
|
|
213
|
+
"figureNumber": num,
|
|
214
|
+
"caption": caption[:240],
|
|
215
|
+
"page": pageno + 1,
|
|
216
|
+
"fileName": f"figures/{page_png.name}" if page_png else None,
|
|
217
|
+
"renderDpi": dpi,
|
|
218
|
+
"mimeType": "image/png",
|
|
219
|
+
}
|
|
220
|
+
return list(seen.values())
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def guess_language(doc: "fitz.Document") -> str:
|
|
224
|
+
"""Best-effort language guess based on common French/German/Italian markers."""
|
|
225
|
+
sample = " ".join(doc[i].get_text("text") for i in range(min(5, doc.page_count))).lower()
|
|
226
|
+
scores = {
|
|
227
|
+
"fr": sum(sample.count(w) for w in (" la ", " les ", " sont ", " avec ", " selon ")),
|
|
228
|
+
"de": sum(sample.count(w) for w in (" der ", " die ", " sind ", " mit ", " nach ")),
|
|
229
|
+
"it": sum(sample.count(w) for w in (" la ", " sono ", " con ", " per ", " della ")),
|
|
230
|
+
"en": sum(sample.count(w) for w in (" the ", " are ", " with ", " of ", " shall ")),
|
|
231
|
+
}
|
|
232
|
+
return max(scores, key=scores.get)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def main() -> None:
|
|
236
|
+
ap = argparse.ArgumentParser(description="Stratta ingest pre-pass (PyMuPDF).")
|
|
237
|
+
ap.add_argument("--pdf", required=True, help="Path to the norm PDF.")
|
|
238
|
+
ap.add_argument("--output", required=True, help="Output directory.")
|
|
239
|
+
ap.add_argument(
|
|
240
|
+
"--language",
|
|
241
|
+
choices=["fr", "de", "it", "en"],
|
|
242
|
+
help="Override detected language.",
|
|
243
|
+
)
|
|
244
|
+
ap.add_argument("--dpi", type=int, default=150, help="Figure render DPI.")
|
|
245
|
+
args = ap.parse_args()
|
|
246
|
+
|
|
247
|
+
pdf_path = Path(args.pdf).resolve()
|
|
248
|
+
out_dir = Path(args.output).resolve()
|
|
249
|
+
if not pdf_path.exists():
|
|
250
|
+
sys.stderr.write(f"PDF not found: {pdf_path}\n")
|
|
251
|
+
sys.exit(1)
|
|
252
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
253
|
+
|
|
254
|
+
doc = fitz.open(str(pdf_path))
|
|
255
|
+
toc_pages = detect_toc_pages(doc)
|
|
256
|
+
running = detect_running_text(doc)
|
|
257
|
+
chapters = extract_chapters(doc)
|
|
258
|
+
chap_prefixes = {k for k in chapters if k.isdigit()}
|
|
259
|
+
subsections = extract_subsections(doc, chap_prefixes, toc_pages, running)
|
|
260
|
+
nodes = build_tree(chapters, subsections, doc.page_count)
|
|
261
|
+
extract_section_text(doc, nodes)
|
|
262
|
+
figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)
|
|
263
|
+
|
|
264
|
+
lang = args.language or guess_language(doc)
|
|
265
|
+
|
|
266
|
+
by_depth: dict[int, int] = defaultdict(int)
|
|
267
|
+
for n in nodes:
|
|
268
|
+
by_depth[n["depth"]] += 1
|
|
269
|
+
|
|
270
|
+
manifest = {
|
|
271
|
+
"doc": {
|
|
272
|
+
"sourcePath": str(pdf_path),
|
|
273
|
+
"pageCount": doc.page_count,
|
|
274
|
+
"language": lang,
|
|
275
|
+
"tocSource": "bookmarks+regex" if doc.get_toc() else "regex-only",
|
|
276
|
+
"metadata": {k: v for k, v in doc.metadata.items() if v},
|
|
277
|
+
"tocPagesDetected": sorted(toc_pages),
|
|
278
|
+
"runningHeadersDetected": sorted(running)[:10],
|
|
279
|
+
},
|
|
280
|
+
"stats": {
|
|
281
|
+
"sectionCount": len(nodes),
|
|
282
|
+
"byDepth": dict(by_depth),
|
|
283
|
+
"figureCount": len(figures),
|
|
284
|
+
},
|
|
285
|
+
"sections": nodes,
|
|
286
|
+
"figures": figures,
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
manifest_path = out_dir / "prepass.json"
|
|
290
|
+
manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
291
|
+
|
|
292
|
+
sys.stdout.write(
|
|
293
|
+
json.dumps(
|
|
294
|
+
{
|
|
295
|
+
"ok": True,
|
|
296
|
+
"manifest": str(manifest_path),
|
|
297
|
+
"sectionCount": len(nodes),
|
|
298
|
+
"byDepth": dict(by_depth),
|
|
299
|
+
"figureCount": len(figures),
|
|
300
|
+
"language": lang,
|
|
301
|
+
"pageCount": doc.page_count,
|
|
302
|
+
},
|
|
303
|
+
ensure_ascii=False,
|
|
304
|
+
)
|
|
305
|
+
+ "\n"
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
if __name__ == "__main__":
|
|
310
|
+
main()
|
|
@@ -1,75 +1,128 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: ingest-norm
|
|
3
|
-
description: Use when the user wants to add an engineering norm (SIA, Eurocode, etc.) they are licensed for into THEIR Stratta workspace.
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Ingest a norm into your Stratta workspace
|
|
7
|
-
|
|
8
|
-
This skill turns a norm PDF **you are licensed to use** into a queryable
|
|
9
|
-
inside **your own** Stratta workspace.
|
|
10
|
-
the
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
`
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
- `
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
`
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
1
|
+
---
|
|
2
|
+
name: ingest-norm
|
|
3
|
+
description: Use when the user wants to add an engineering norm (SIA, Eurocode, etc.) they are licensed for into THEIR Stratta workspace. A Python pre-pass (PyMuPDF) extracts the hierarchical tree and rasterizes figures; an agentic pass enriches sections (LaTeX formulas, tables, cross-references, summaries) and writes everything via the MCP `ingest_*` tools — scoped to the user's own organization. Trigger phrases: "ingère cette norme", "/ingest-norm", "ingest SIA", "ajoute la norme X à Stratta".
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Ingest a norm into your Stratta workspace
|
|
7
|
+
|
|
8
|
+
This skill turns a norm PDF **you are licensed to use** into a queryable
|
|
9
|
+
TreeRAG inside **your own** Stratta workspace. A Python pre-pass (PyMuPDF)
|
|
10
|
+
does the deterministic heavy lifting — TOC tree, per-section raw text, figure
|
|
11
|
+
captions and full-page renders. Then an agentic pass enriches sections with
|
|
12
|
+
summaries, LaTeX formulas, structured tables, and cross-references. Final
|
|
13
|
+
writes go through the Stratta MCP `ingest_*` tools, scoped to your org.
|
|
14
|
+
|
|
15
|
+
> ⚠️ **Licence**: only ingest norms your organization holds a valid licence
|
|
16
|
+
> for. You are responsible for your usage rights (see Stratta's Terms).
|
|
17
|
+
|
|
18
|
+
## Prerequisites
|
|
19
|
+
- The Stratta MCP server is installed and your `STRATTA_API_KEY` resolves
|
|
20
|
+
(via env, `~/.stratta/config.json`, or first-call elicitation).
|
|
21
|
+
- **Python ≥ 3.10 with PyMuPDF**. Install once:
|
|
22
|
+
`python -m pip install --user pymupdf` (or `uv pip install pymupdf`).
|
|
23
|
+
- The norm PDF is available locally.
|
|
24
|
+
|
|
25
|
+
## Tools used (all scoped to your workspace)
|
|
26
|
+
`ingest_status` · `ingest_create_document` · `ingest_create_sections` ·
|
|
27
|
+
`ingest_attach_formula` · `ingest_attach_table` · `ingest_attach_cross_ref` ·
|
|
28
|
+
`ingest_upload_figure` · `ingest_normalize_cross_refs` · `ingest_publish` ·
|
|
29
|
+
`ingest_delete`.
|
|
30
|
+
|
|
31
|
+
## Workflow
|
|
32
|
+
|
|
33
|
+
### 1. Locate the pre-pass script
|
|
34
|
+
It ships inside this package at `scripts/ingest-prepass.py`. Resolve its path:
|
|
35
|
+
```bash
|
|
36
|
+
node -e "console.log(require.resolve('@stratta/mcp/package.json'))"
|
|
37
|
+
# → <root>/package.json → <root>/scripts/ingest-prepass.py
|
|
38
|
+
```
|
|
39
|
+
If the user is working in the Stratta monorepo, the script also lives at
|
|
40
|
+
`packages/mcp/scripts/ingest-prepass.py`.
|
|
41
|
+
|
|
42
|
+
### 2. Check for an existing copy
|
|
43
|
+
`ingest_status { code }` (e.g. `"SIA 261"`). To re-ingest, call
|
|
44
|
+
`ingest_delete { documentId }` first.
|
|
45
|
+
|
|
46
|
+
### 3. Run the pre-pass
|
|
47
|
+
```bash
|
|
48
|
+
python <pkg-root>/scripts/ingest-prepass.py \
|
|
49
|
+
--pdf <path-to-pdf> \
|
|
50
|
+
--output .stratta-ingest/<code-slug>
|
|
51
|
+
```
|
|
52
|
+
Output under `.stratta-ingest/<code-slug>/`:
|
|
53
|
+
- `prepass.json` — full manifest (see below).
|
|
54
|
+
- `figures/page-NNN.png` — one PNG per page that contains a `Figure N` caption.
|
|
55
|
+
|
|
56
|
+
`prepass.json` structure:
|
|
57
|
+
- `doc` — `pageCount`, detected `language`, `tocSource`, raw `metadata`.
|
|
58
|
+
- `stats` — `sectionCount`, `byDepth`, `figureCount`.
|
|
59
|
+
- `sections[]` — full hierarchical tree (depth **0 = chapter**, 1+ = sub-sections),
|
|
60
|
+
each with `nodeId`, `parentNodeId`, `path` (`"4.2.1"` or `"Annexe B"`),
|
|
61
|
+
`title`, `depth`, `pageStart`, `pageEnd`, `orderIndex`, `rawText` (concat
|
|
62
|
+
of the pages the node spans).
|
|
63
|
+
- `figures[]` — one entry per `Figure N` caption: `figureNumber`, `caption`,
|
|
64
|
+
`page`, `fileName`, `mimeType`.
|
|
65
|
+
|
|
66
|
+
The pre-pass is **exhaustive** (e.g. ~550 nodes on SIA 261). You decide what
|
|
67
|
+
to keep in the next step.
|
|
68
|
+
|
|
69
|
+
### 4. Create the document
|
|
70
|
+
Read `prepass.json`, then:
|
|
71
|
+
`ingest_create_document { code, year, title, language: <doc.language>, totalPages: <doc.pageCount> }`
|
|
72
|
+
→ returns `documentId`. Keep it for every subsequent call.
|
|
73
|
+
|
|
74
|
+
### 5. Decide section granularity + generate summaries
|
|
75
|
+
Iterate `sections[]` and decide what to keep. Two viable strategies:
|
|
76
|
+
- **Keep all** — most faithful, ~500 sections on a typical SIA norm. Great
|
|
77
|
+
for fine-grained navigation but verbose.
|
|
78
|
+
- **Aggregate trivial leaves** — fold paragraph-level nodes (`6.1.1`...`6.1.11`)
|
|
79
|
+
into their parent (`6.1`), concatenating their `rawText`. Typical result:
|
|
80
|
+
100-150 sections. Recommended unless the user asks for max granularity.
|
|
81
|
+
|
|
82
|
+
For each kept section, prepare:
|
|
83
|
+
- `summary` — 1-3 sentences derived from `rawText` (mention formulas/values).
|
|
84
|
+
- `content` — enriched text with LaTeX inline where the source has math
|
|
85
|
+
(e.g. `$\sigma_d = f_{yd} \cdot \gamma$`). Open the PDF visually for pages
|
|
86
|
+
that contain formulas or multi-column tables — PyMuPDF mangles those.
|
|
87
|
+
- `rawContent` — use the pre-pass `rawText` as-is.
|
|
88
|
+
|
|
89
|
+
Keep the `nodeId` / `parentNodeId` / `path` / `pageStart` / `pageEnd` /
|
|
90
|
+
`orderIndex` / `depth` from the pre-pass — those are deterministic.
|
|
91
|
+
|
|
92
|
+
### 6. Insert sections (batched)
|
|
93
|
+
`ingest_create_sections { documentId, sections: [...] }` in batches of 30-50.
|
|
94
|
+
Parent links resolve via `parentNodeId` within the batch and across prior
|
|
95
|
+
batches. The call returns a `nodeId → sectionId` map — **use those `sectionId`s**
|
|
96
|
+
for every enrichment call below.
|
|
97
|
+
|
|
98
|
+
### 7. Enrich
|
|
99
|
+
- `ingest_attach_formula { sectionId, latex, description, formulaNumber }`
|
|
100
|
+
- `ingest_attach_table { sectionId, data: { headers, rows }, caption, tableNumber }`
|
|
101
|
+
- `ingest_attach_cross_ref { sourceSectionId, targetDocumentCode, targetSectionPath?, refText, refType }`
|
|
102
|
+
|
|
103
|
+
### 8. Upload figures
|
|
104
|
+
For each figure in `prepass.json#figures`:
|
|
105
|
+
- Read `.stratta-ingest/<code>/<fileName>` and base64-encode the bytes.
|
|
106
|
+
- Find the owning section: the kept section whose `pageStart..pageEnd`
|
|
107
|
+
range includes the figure's `page`.
|
|
108
|
+
- `ingest_upload_figure { sectionId, base64, mimeType: "image/png", caption, figureNumber }`
|
|
109
|
+
(≤ 8 MB per image).
|
|
110
|
+
|
|
111
|
+
### 9. Auto cross-references (optional but recommended)
|
|
112
|
+
`ingest_normalize_cross_refs { documentId }` scans every section's text for
|
|
113
|
+
references to other norms (SIA / SN EN / EN / ISO / DIN …) and rebuilds the
|
|
114
|
+
cross-ref index. Idempotent.
|
|
115
|
+
|
|
116
|
+
### 10. Publish
|
|
117
|
+
`ingest_publish { documentId }`. The norm is now queryable in your workspace
|
|
118
|
+
via `list_norms`, `get_toc`, `get_section`, `search_in_norm`, `get_figure`, etc.
|
|
119
|
+
|
|
120
|
+
## Quality bar
|
|
121
|
+
- Trust the pre-pass for `path` + `pageStart`/`pageEnd` — it's deterministic
|
|
122
|
+
and verified against the PDF bookmarks.
|
|
123
|
+
- Summaries drive navigation — be specific (mention key formulas/values).
|
|
124
|
+
- Keep `content` faithful to the source; don't invent values.
|
|
125
|
+
- If PyMuPDF mangled a formula (Greek letters, fractions, exponents broken),
|
|
126
|
+
re-read the relevant PDF page visually and write proper LaTeX.
|
|
127
|
+
- Never skip annexes — they hold key numeric values (zones, coefficients,
|
|
128
|
+
characteristic loads).
|