@stratta/mcp 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@stratta/mcp",
3
3
  "mcpName": "io.github.hugogebel-boop/stratta",
4
- "version": "0.3.0",
4
+ "version": "0.4.0",
5
5
  "description": "MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients via Stratta TreeRAG.",
6
6
  "license": "UNLICENSED",
7
7
  "author": "Hugo Gebel <hugo.gebel@epfl.ch>",
@@ -29,6 +29,7 @@
29
29
  "files": [
30
30
  "dist",
31
31
  "skills",
32
+ "scripts",
32
33
  "README.md"
33
34
  ],
34
35
  "scripts": {
@@ -0,0 +1,310 @@
1
+ #!/usr/bin/env python3
2
+ """Stratta — ingest pre-pass.
3
+
4
+ Reads a norm PDF with PyMuPDF, builds the hierarchical TreeRAG skeleton (chapters
5
+ from PDF bookmarks + sub-sections via heading regex), extracts per-section raw
6
+ text, and rasterizes each page that contains a figure caption. Output is a
7
+ single JSON consumed by the `ingest-norm` skill, which then enriches sections
8
+ (formulas, tables, cross-refs, summaries) and uploads the figures.
9
+
10
+ Usage:
11
+ python scripts/ingest-prepass.py --pdf <path> --output <dir> [--language fr]
12
+
13
+ Output (under <dir>/):
14
+ prepass.json full extracted tree + figure manifest
15
+ figures/figure-<N>.png rasterized full-page renders for each caption
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import argparse
21
+ import json
22
+ import re
23
+ import sys
24
+ from collections import defaultdict
25
+ from pathlib import Path
26
+
27
+ try:
28
+ import fitz # PyMuPDF
29
+ except ImportError:
30
+ sys.stderr.write(
31
+ "ERROR: PyMuPDF not installed. Run: pip install --user pymupdf\n"
32
+ )
33
+ sys.exit(2)
34
+
35
+
36
+ BM_NUM_RE = re.compile(r"^\s*(\d+)\s+(.+?)\s*$")
37
+ BM_ANNEX_RE = re.compile(
38
+ r"^\s*(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)\s*$"
39
+ )
40
+
41
+ SUB_RE = re.compile(
42
+ r"^(\d+(?:\.\d+){1,3})\s+([A-Za-zÀ-ÿ][^\n]{1,140})$", re.MULTILINE
43
+ )
44
+ TOC_DOT_RE = re.compile(r"\.\s*\.\s*\.")
45
+ CLEAN_DOTS = re.compile(r"\s*(\.\s*){3,}.*$")
46
+
47
+ CAPTION_RE = re.compile(
48
+ r"^\s*(?:Figure|Fig\.|Bild|Abbildung|Figura)\s+(\d+[a-z]?)\b[\s:.\-—–]*(.*)$",
49
+ re.IGNORECASE | re.MULTILINE,
50
+ )
51
+
52
+
53
+ def clean(s: str) -> str:
54
+ return re.sub(r"\s+", " ", CLEAN_DOTS.sub("", s)).strip()
55
+
56
+
57
+ def detect_toc_pages(doc: "fitz.Document") -> set[int]:
58
+ out = set()
59
+ for i in range(doc.page_count):
60
+ lines = [l for l in doc[i].get_text("text").splitlines() if l.strip()]
61
+ leader = sum(1 for l in lines if TOC_DOT_RE.search(l))
62
+ if leader >= 5:
63
+ out.add(i)
64
+ return out
65
+
66
+
67
+ def detect_running_text(doc: "fitz.Document") -> set[str]:
68
+ """Lines appearing on >=5 pages near the top/bottom (running headers/footers)."""
69
+ counter: dict[str, int] = defaultdict(int)
70
+ for i in range(doc.page_count):
71
+ lines = [l.strip() for l in doc[i].get_text("text").splitlines() if l.strip()]
72
+ for l in lines[:3] + lines[-3:]:
73
+ counter[l] += 1
74
+ return {l for l, c in counter.items() if c >= 5 and len(l) > 8}
75
+
76
+
77
+ def extract_chapters(doc: "fitz.Document") -> dict[str, dict]:
78
+ """Top-level chapters from PDF bookmarks (authoritative titles + pages)."""
79
+ chapters: dict[str, dict] = {}
80
+ for _depth, title, page in doc.get_toc(simple=True):
81
+ m = BM_ANNEX_RE.match(title)
82
+ if m:
83
+ letter, kind, rest = m.group(1), m.group(2) or "", (m.group(3) or "").strip()
84
+ path = f"Annexe {letter}"
85
+ t = path + (f" ({kind})" if kind else "")
86
+ if rest and rest.lower() != path.lower():
87
+ t += f" — {rest}"
88
+ chapters[path] = {"title": t, "pageStart": page}
89
+ continue
90
+ m = BM_NUM_RE.match(title)
91
+ if m:
92
+ num, rest = m.group(1), m.group(2).strip()
93
+ chapters[num] = {"title": rest, "pageStart": page}
94
+ return chapters
95
+
96
+
97
+ def extract_subsections(
98
+ doc: "fitz.Document",
99
+ chap_prefixes: set[str],
100
+ toc_pages: set[int],
101
+ running: set[str],
102
+ ) -> dict[str, tuple[str, int]]:
103
+ """Numeric sub-sections (depths 2-4) detected in body pages."""
104
+ out: dict[str, tuple[str, int]] = {}
105
+ for i in range(doc.page_count):
106
+ if i in toc_pages:
107
+ continue
108
+ text = doc[i].get_text("text")
109
+ for m in SUB_RE.finditer(text):
110
+ num = m.group(1)
111
+ title = clean(m.group(2))
112
+ if not title or title in running:
113
+ continue
114
+ top = num.split(".", 1)[0]
115
+ if top not in chap_prefixes:
116
+ continue
117
+ if len(title) < 3 or re.fullmatch(r"[^A-Za-zÀ-ÿ]+", title):
118
+ continue
119
+ if num in out:
120
+ continue
121
+ out[num] = (title, i + 1)
122
+ return out
123
+
124
+
125
+ def build_tree(chapters: dict, subsections: dict, n_pages: int) -> list[dict]:
126
+ """Returns flat list of nodes with depth 0-indexed (chapter=0, section=1, ...)."""
127
+ nodes: list[dict] = []
128
+ for path, info in chapters.items():
129
+ nodes.append(
130
+ {"depth": 0, "path": path, "title": info["title"], "pageStart": info["pageStart"]}
131
+ )
132
+ for path, (title, pg) in subsections.items():
133
+ nodes.append(
134
+ {"depth": path.count("."), "path": path, "title": title, "pageStart": pg}
135
+ )
136
+
137
+ def sort_key(n):
138
+ p = n["path"]
139
+ if p.startswith("Annexe"):
140
+ return (10**9, ord(p[-1]), [])
141
+ parts = [int(x) for x in p.split(".")]
142
+ return (parts[0], 0, parts)
143
+
144
+ nodes.sort(key=sort_key)
145
+
146
+ # pageEnd via next sibling/ancestor
147
+ for idx, node in enumerate(nodes):
148
+ d, pg = node["depth"], node["pageStart"]
149
+ next_pg = n_pages
150
+ for j in range(idx + 1, len(nodes)):
151
+ if nodes[j]["depth"] <= d:
152
+ next_pg = max(nodes[j]["pageStart"] - 1, pg)
153
+ break
154
+ node["pageEnd"] = next_pg
155
+
156
+ # Parent links + nodeId + orderIndex
157
+ parent_stack: list[tuple[int, str]] = [] # (depth, nodeId)
158
+ for idx, node in enumerate(nodes):
159
+ d = node["depth"]
160
+ while parent_stack and parent_stack[-1][0] >= d:
161
+ parent_stack.pop()
162
+ node["nodeId"] = f"s-{slugify(node['path'])}"
163
+ node["parentNodeId"] = parent_stack[-1][1] if parent_stack else None
164
+ node["orderIndex"] = idx
165
+ parent_stack.append((d, node["nodeId"]))
166
+
167
+ return nodes
168
+
169
+
170
+ def slugify(path: str) -> str:
171
+ return re.sub(r"[^a-zA-Z0-9]+", "-", path).strip("-").lower() or "root"
172
+
173
+
174
+ def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
175
+ """Attach rawText to each node = concat of pages it spans."""
176
+ page_texts = [doc[i].get_text("text") for i in range(doc.page_count)]
177
+ for node in nodes:
178
+ ps, pe = node["pageStart"], node["pageEnd"]
179
+ chunks = page_texts[ps - 1 : pe]
180
+ node["rawText"] = "\n".join(chunks).strip()
181
+
182
+
183
+ def extract_figures(
184
+ doc: "fitz.Document",
185
+ toc_pages: set[int],
186
+ out_dir: Path,
187
+ dpi: int = 150,
188
+ ) -> list[dict]:
189
+ """One figure manifest entry per caption detected. Page is rendered to PNG."""
190
+ figs_dir = out_dir / "figures"
191
+ figs_dir.mkdir(parents=True, exist_ok=True)
192
+ seen: dict[str, dict] = {}
193
+ matrix = fitz.Matrix(dpi / 72, dpi / 72)
194
+ for pageno in range(doc.page_count):
195
+ if pageno in toc_pages:
196
+ continue
197
+ text = doc[pageno].get_text("text")
198
+ rendered = False
199
+ page_png: Path | None = None
200
+ for m in CAPTION_RE.finditer(text):
201
+ num = m.group(1)
202
+ label_rest = m.group(2).strip()
203
+ if num in seen:
204
+ continue
205
+ caption = f"Figure {num}" + (f" — {label_rest}" if label_rest else "")
206
+ # render the page once if not done
207
+ if not rendered:
208
+ pix = doc[pageno].get_pixmap(matrix=matrix, alpha=False)
209
+ page_png = figs_dir / f"page-{pageno + 1:03d}.png"
210
+ pix.save(str(page_png))
211
+ rendered = True
212
+ seen[num] = {
213
+ "figureNumber": num,
214
+ "caption": caption[:240],
215
+ "page": pageno + 1,
216
+ "fileName": f"figures/{page_png.name}" if page_png else None,
217
+ "renderDpi": dpi,
218
+ "mimeType": "image/png",
219
+ }
220
+ return list(seen.values())
221
+
222
+
223
+ def guess_language(doc: "fitz.Document") -> str:
224
+ """Best-effort language guess based on common French/German/Italian markers."""
225
+ sample = " ".join(doc[i].get_text("text") for i in range(min(5, doc.page_count))).lower()
226
+ scores = {
227
+ "fr": sum(sample.count(w) for w in (" la ", " les ", " sont ", " avec ", " selon ")),
228
+ "de": sum(sample.count(w) for w in (" der ", " die ", " sind ", " mit ", " nach ")),
229
+ "it": sum(sample.count(w) for w in (" la ", " sono ", " con ", " per ", " della ")),
230
+ "en": sum(sample.count(w) for w in (" the ", " are ", " with ", " of ", " shall ")),
231
+ }
232
+ return max(scores, key=scores.get)
233
+
234
+
235
+ def main() -> None:
236
+ ap = argparse.ArgumentParser(description="Stratta ingest pre-pass (PyMuPDF).")
237
+ ap.add_argument("--pdf", required=True, help="Path to the norm PDF.")
238
+ ap.add_argument("--output", required=True, help="Output directory.")
239
+ ap.add_argument(
240
+ "--language",
241
+ choices=["fr", "de", "it", "en"],
242
+ help="Override detected language.",
243
+ )
244
+ ap.add_argument("--dpi", type=int, default=150, help="Figure render DPI.")
245
+ args = ap.parse_args()
246
+
247
+ pdf_path = Path(args.pdf).resolve()
248
+ out_dir = Path(args.output).resolve()
249
+ if not pdf_path.exists():
250
+ sys.stderr.write(f"PDF not found: {pdf_path}\n")
251
+ sys.exit(1)
252
+ out_dir.mkdir(parents=True, exist_ok=True)
253
+
254
+ doc = fitz.open(str(pdf_path))
255
+ toc_pages = detect_toc_pages(doc)
256
+ running = detect_running_text(doc)
257
+ chapters = extract_chapters(doc)
258
+ chap_prefixes = {k for k in chapters if k.isdigit()}
259
+ subsections = extract_subsections(doc, chap_prefixes, toc_pages, running)
260
+ nodes = build_tree(chapters, subsections, doc.page_count)
261
+ extract_section_text(doc, nodes)
262
+ figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)
263
+
264
+ lang = args.language or guess_language(doc)
265
+
266
+ by_depth: dict[int, int] = defaultdict(int)
267
+ for n in nodes:
268
+ by_depth[n["depth"]] += 1
269
+
270
+ manifest = {
271
+ "doc": {
272
+ "sourcePath": str(pdf_path),
273
+ "pageCount": doc.page_count,
274
+ "language": lang,
275
+ "tocSource": "bookmarks+regex" if doc.get_toc() else "regex-only",
276
+ "metadata": {k: v for k, v in doc.metadata.items() if v},
277
+ "tocPagesDetected": sorted(toc_pages),
278
+ "runningHeadersDetected": sorted(running)[:10],
279
+ },
280
+ "stats": {
281
+ "sectionCount": len(nodes),
282
+ "byDepth": dict(by_depth),
283
+ "figureCount": len(figures),
284
+ },
285
+ "sections": nodes,
286
+ "figures": figures,
287
+ }
288
+
289
+ manifest_path = out_dir / "prepass.json"
290
+ manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8")
291
+
292
+ sys.stdout.write(
293
+ json.dumps(
294
+ {
295
+ "ok": True,
296
+ "manifest": str(manifest_path),
297
+ "sectionCount": len(nodes),
298
+ "byDepth": dict(by_depth),
299
+ "figureCount": len(figures),
300
+ "language": lang,
301
+ "pageCount": doc.page_count,
302
+ },
303
+ ensure_ascii=False,
304
+ )
305
+ + "\n"
306
+ )
307
+
308
+
309
+ if __name__ == "__main__":
310
+ main()
@@ -1,75 +1,128 @@
1
- ---
2
- name: ingest-norm
3
- description: Use when the user wants to add an engineering norm (SIA, Eurocode, etc.) they are licensed for into THEIR Stratta workspace. Reads the PDF, builds a hierarchical tree, enriches sections (LaTeX formulas, tables, figures, cross-references), and writes everything to Stratta via the MCP `ingest_*` tools — scoped to the user's own organization. Trigger phrases: "ingère cette norme", "ingest SIA", "ajoute la norme X à Stratta".
4
- ---
5
-
6
- # Ingest a norm into your Stratta workspace
7
-
8
- This skill turns a norm PDF **you are licensed to use** into a queryable TreeRAG
9
- inside **your own** Stratta workspace. Everything runs on your machine through
10
- the Stratta MCP server (authenticated by your `STRATTA_API_KEY`); the norm is
11
- stored privately and scoped to your organization — no one else can see it.
12
-
13
- > ⚠️ **Licence**: only ingest norms your organization holds a valid licence for.
14
- > You are responsible for your usage rights (see Stratta's Terms).
15
-
16
- ## Prerequisites
17
- - The Stratta MCP server is installed and `STRATTA_API_KEY` is set.
18
- - The norm PDF is available locally.
19
- - (Optional, for figures) `poppler` tools (`pdftoppm`, `pdfimages`) to rasterize
20
- pages/extract images.
21
-
22
- ## Tools used (all scoped to your workspace)
23
- `ingest_status` · `ingest_create_document` · `ingest_create_sections` ·
24
- `ingest_attach_formula` · `ingest_attach_table` · `ingest_attach_cross_ref` ·
25
- `ingest_upload_figure` · `ingest_normalize_cross_refs` · `ingest_publish` ·
26
- `ingest_delete`.
27
-
28
- ## Workflow
29
-
30
- ### 1. Check for an existing copy
31
- Call `ingest_status { code }` (e.g. `"SIA 261"`). If it already exists and you
32
- want to re-ingest, call `ingest_delete { documentId }` first.
33
-
34
- ### 2. Read the PDF and plan the tree
35
- Read the PDF natively. Build a PageIndex-style hierarchy: chapters (depth 0),
36
- sections (depth 1), sub-sections (depth 2+), plus annexes. For each node decide
37
- a stable `nodeId`, a human `path` (e.g. `"4.2.1"`, `"Annexe B.1"`), `title`,
38
- a 1-3 sentence `summary` (used for navigation), `pageStart`/`pageEnd`, and an
39
- `orderIndex`. **Never skip annexes** — they hold key numeric values.
40
-
41
- ### 3. Create the document
42
- `ingest_create_document { code, year, title, language, totalPages }` →
43
- returns `documentId`. Keep it for all subsequent calls.
44
-
45
- ### 4. Insert sections (in batches)
46
- `ingest_create_sections { documentId, sections: [...] }`. Each section carries
47
- `nodeId`, optional `parentNodeId` (must already exist in this or a prior batch),
48
- `path`, `title`, `summary`, `depth`, `content` (full enriched text, LaTeX inline
49
- OK), `rawContent`, `pageStart`, `pageEnd`, `orderIndex`. The call returns a
50
- `nodeId → sectionId` map — **use those sectionIds** to attach formulas, tables,
51
- figures and cross-refs. Batch large docs (e.g. 30-50 sections per call).
52
-
53
- ### 5. Enrich sections
54
- - `ingest_attach_formula { sectionId, latex, description, formulaNumber }`
55
- - `ingest_attach_table { sectionId, data: { headers, rows }, caption, tableNumber }`
56
- - `ingest_attach_cross_ref { sourceSectionId, targetDocumentCode, targetSectionPath?, refText, refType }`
57
-
58
- ### 6. Figures
59
- For each relevant figure: rasterize/extract the image, read its bytes, base64-encode,
60
- then `ingest_upload_figure { sectionId, base64, mimeType, caption, figureNumber }`
61
- (png/jpeg/webp, ≤ 8 MB). The tool stores the image and links it in one call.
62
-
63
- ### 7. Auto cross-references (optional)
64
- `ingest_normalize_cross_refs { documentId }` scans every section's text for
65
- references to other norms (SIA / SN EN / EN / ISO / DIN …) and rebuilds the
66
- cross-ref index. Idempotent.
67
-
68
- ### 8. Publish
69
- `ingest_publish { documentId }`. The norm is now queryable in your workspace via
70
- `list_norms`, `get_toc`, `get_section`, `search_in_norm`, `get_figure`, etc.
71
-
72
- ## Quality bar
73
- - Citations depend on accurate `path` + `pageStart/pageEnd` — get them right.
74
- - Summaries drive navigation — make them specific (mention formulas/values when present).
75
- - Keep `content` faithful to the source; don't invent values.
1
+ ---
2
+ name: ingest-norm
3
+ description: Use when the user wants to add an engineering norm (SIA, Eurocode, etc.) they are licensed for into THEIR Stratta workspace. A Python pre-pass (PyMuPDF) extracts the hierarchical tree and rasterizes figures; an agentic pass enriches sections (LaTeX formulas, tables, cross-references, summaries) and writes everything via the MCP `ingest_*` tools — scoped to the user's own organization. Trigger phrases: "ingère cette norme", "/ingest-norm", "ingest SIA", "ajoute la norme X à Stratta".
4
+ ---
5
+
6
+ # Ingest a norm into your Stratta workspace
7
+
8
+ This skill turns a norm PDF **you are licensed to use** into a queryable
9
+ TreeRAG inside **your own** Stratta workspace. A Python pre-pass (PyMuPDF)
10
+ does the deterministic heavy lifting — TOC tree, per-section raw text, figure
11
+ captions and full-page renders. Then an agentic pass enriches sections with
12
+ summaries, LaTeX formulas, structured tables, and cross-references. Final
13
+ writes go through the Stratta MCP `ingest_*` tools, scoped to your org.
14
+
15
+ > ⚠️ **Licence**: only ingest norms your organization holds a valid licence
16
+ > for. You are responsible for your usage rights (see Stratta's Terms).
17
+
18
+ ## Prerequisites
19
+ - The Stratta MCP server is installed and your `STRATTA_API_KEY` resolves
20
+ (via env, `~/.stratta/config.json`, or first-call elicitation).
21
+ - **Python ≥ 3.10 with PyMuPDF**. Install once:
22
+ `python -m pip install --user pymupdf` (or `uv pip install pymupdf`).
23
+ - The norm PDF is available locally.
24
+
25
+ ## Tools used (all scoped to your workspace)
26
+ `ingest_status` · `ingest_create_document` · `ingest_create_sections` ·
27
+ `ingest_attach_formula` · `ingest_attach_table` · `ingest_attach_cross_ref` ·
28
+ `ingest_upload_figure` · `ingest_normalize_cross_refs` · `ingest_publish` ·
29
+ `ingest_delete`.
30
+
31
+ ## Workflow
32
+
33
+ ### 1. Locate the pre-pass script
34
+ It ships inside this package at `scripts/ingest-prepass.py`. Resolve its path:
35
+ ```bash
36
+ node -e "console.log(require.resolve('@stratta/mcp/package.json'))"
37
+ # → <root>/package.json → <root>/scripts/ingest-prepass.py
38
+ ```
39
+ If the user is working in the Stratta monorepo, the script also lives at
40
+ `packages/mcp/scripts/ingest-prepass.py`.
41
+
42
+ ### 2. Check for an existing copy
43
+ `ingest_status { code }` (e.g. `"SIA 261"`). To re-ingest, call
44
+ `ingest_delete { documentId }` first.
45
+
46
+ ### 3. Run the pre-pass
47
+ ```bash
48
+ python <pkg-root>/scripts/ingest-prepass.py \
49
+ --pdf <path-to-pdf> \
50
+ --output .stratta-ingest/<code-slug>
51
+ ```
52
+ Output under `.stratta-ingest/<code-slug>/`:
53
+ - `prepass.json` — full manifest (see below).
54
+ - `figures/page-NNN.png` — one PNG per page that contains a `Figure N` caption.
55
+
56
+ `prepass.json` structure:
57
+ - `doc` — `pageCount`, detected `language`, `tocSource`, raw `metadata`.
58
+ - `stats` — `sectionCount`, `byDepth`, `figureCount`.
59
+ - `sections[]` — full hierarchical tree (depth **0 = chapter**, 1+ = sub-sections),
60
+ each with `nodeId`, `parentNodeId`, `path` (`"4.2.1"` or `"Annexe B"`),
61
+ `title`, `depth`, `pageStart`, `pageEnd`, `orderIndex`, `rawText` (concat
62
+ of the pages the node spans).
63
+ - `figures[]` — one entry per `Figure N` caption: `figureNumber`, `caption`,
64
+ `page`, `fileName`, `mimeType`.
65
+
66
+ The pre-pass is **exhaustive** (e.g. ~550 nodes on SIA 261). You decide what
67
+ to keep in the next step.
68
+
69
+ ### 4. Create the document
70
+ Read `prepass.json`, then:
71
+ `ingest_create_document { code, year, title, language: <doc.language>, totalPages: <doc.pageCount> }`
72
+ → returns `documentId`. Keep it for every subsequent call.
73
+
74
+ ### 5. Decide section granularity + generate summaries
75
+ Iterate `sections[]` and decide what to keep. Two viable strategies:
76
+ - **Keep all** — most faithful, ~500 sections on a typical SIA norm. Great
77
+ for fine-grained navigation but verbose.
78
+ - **Aggregate trivial leaves** — fold paragraph-level nodes (`6.1.1`...`6.1.11`)
79
+ into their parent (`6.1`), concatenating their `rawText`. Typical result:
80
+ 100-150 sections. Recommended unless the user asks for max granularity.
81
+
82
+ For each kept section, prepare:
83
+ - `summary` — 1-3 sentences derived from `rawText` (mention formulas/values).
84
+ - `content` — enriched text with LaTeX inline where the source has math
85
+ (e.g. `$\sigma_d = f_{yd} \cdot \gamma$`). Open the PDF visually for pages
86
+ that contain formulas or multi-column tables — PyMuPDF mangles those.
87
+ - `rawContent` — use the pre-pass `rawText` as-is.
88
+
89
+ Keep the `nodeId` / `parentNodeId` / `path` / `pageStart` / `pageEnd` /
90
+ `orderIndex` / `depth` from the pre-pass — those are deterministic.
91
+
92
+ ### 6. Insert sections (batched)
93
+ `ingest_create_sections { documentId, sections: [...] }` in batches of 30-50.
94
+ Parent links resolve via `parentNodeId` within the batch and across prior
95
+ batches. The call returns a `nodeId → sectionId` map — **use those `sectionId`s**
96
+ for every enrichment call below.
97
+
98
+ ### 7. Enrich
99
+ - `ingest_attach_formula { sectionId, latex, description, formulaNumber }`
100
+ - `ingest_attach_table { sectionId, data: { headers, rows }, caption, tableNumber }`
101
+ - `ingest_attach_cross_ref { sourceSectionId, targetDocumentCode, targetSectionPath?, refText, refType }`
102
+
103
+ ### 8. Upload figures
104
+ For each figure in `prepass.json#figures`:
105
+ - Read `.stratta-ingest/<code>/<fileName>` and base64-encode the bytes.
106
+ - Find the owning section: the kept section whose `pageStart..pageEnd`
107
+ range includes the figure's `page`.
108
+ - `ingest_upload_figure { sectionId, base64, mimeType: "image/png", caption, figureNumber }`
109
+ (≤ 8 MB per image).
110
+
111
+ ### 9. Auto cross-references (optional but recommended)
112
+ `ingest_normalize_cross_refs { documentId }` scans every section's text for
113
+ references to other norms (SIA / SN EN / EN / ISO / DIN …) and rebuilds the
114
+ cross-ref index. Idempotent.
115
+
116
+ ### 10. Publish
117
+ `ingest_publish { documentId }`. The norm is now queryable in your workspace
118
+ via `list_norms`, `get_toc`, `get_section`, `search_in_norm`, `get_figure`, etc.
119
+
120
+ ## Quality bar
121
+ - Trust the pre-pass for `path` + `pageStart`/`pageEnd` — it's deterministic
122
+ and verified against the PDF bookmarks.
123
+ - Summaries drive navigation — be specific (mention key formulas/values).
124
+ - Keep `content` faithful to the source; don't invent values.
125
+ - If PyMuPDF mangled a formula (Greek letters, fractions, exponents broken),
126
+ re-read the relevant PDF page visually and write proper LaTeX.
127
+ - Never skip annexes — they hold key numeric values (zones, coefficients,
128
+ characteristic loads).