@stratta/mcp 0.9.6 → 0.9.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -0
- package/dist/index.js +4 -2
- package/package.json +2 -1
- package/scripts/ingest-prepass.py +261 -29
package/README.md
CHANGED
|
@@ -20,6 +20,25 @@ MCP server exposing Swiss engineering norms (SIA / Eurocodes) to Claude clients
|
|
|
20
20
|
> a shell environment variable. If a key leaks, revoke it immediately at
|
|
21
21
|
> https://stratta.ch/api-keys.
|
|
22
22
|
|
|
23
|
+
## Do you need this package?
|
|
24
|
+
|
|
25
|
+
Often not. Stratta also runs as a **remote connector** — one address, a browser
|
|
26
|
+
sign-in, no key and no Node:
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
https://stratta.ch/mcp
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
That is the shorter path, and the only one that works in an agent running in the
|
|
33
|
+
cloud (claude.ai). See https://stratta.ch/docs/en/guides/connect-remote.
|
|
34
|
+
|
|
35
|
+
This package is what you want when:
|
|
36
|
+
|
|
37
|
+
- you are **ingesting a norm** — it reads a PDF from your disk and runs a Python
|
|
38
|
+
pre-pass, neither of which a remote connector can reach;
|
|
39
|
+
- your agent **cannot open a browser** — CI, a scheduled task, a server;
|
|
40
|
+
- you would simply rather run the server yourself.
|
|
41
|
+
|
|
23
42
|
## Install
|
|
24
43
|
|
|
25
44
|
### Claude Code (recommended)
|
package/dist/index.js
CHANGED
|
@@ -101,10 +101,11 @@ for (const def of [...readTools, ...dossierTools, ...ingestTools]) {
|
|
|
101
101
|
* is the documentation site has no discovery path.
|
|
102
102
|
*/
|
|
103
103
|
function printHelp() {
|
|
104
|
-
console.log(`stratta
|
|
104
|
+
console.log(`stratta ${packageVersion()} — vos normes dans votre agent
|
|
105
105
|
|
|
106
106
|
USAGE
|
|
107
|
-
|
|
107
|
+
stratta <commande> après npm install -g @stratta/mcp
|
|
108
|
+
npx -y @stratta/mcp <commande> sans rien installer
|
|
108
109
|
|
|
109
110
|
COMMANDES
|
|
110
111
|
login Autoriser cette machine depuis votre navigateur
|
|
@@ -120,6 +121,7 @@ Sans commande, le serveur MCP démarre sur stdio : c'est ce que fait votre
|
|
|
120
121
|
agent, vous n'avez pas à le lancer vous-même.
|
|
121
122
|
|
|
122
123
|
Installation claude mcp add stratta --scope user -- npx -y @stratta/mcp
|
|
124
|
+
Raccourci npm install -g @stratta/mcp puis stratta doctor
|
|
123
125
|
Documentation https://stratta.ch/docs`);
|
|
124
126
|
}
|
|
125
127
|
async function main() {
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stratta/mcp",
|
|
3
3
|
"mcpName": "ch.stratta/mcp",
|
|
4
|
-
"version": "0.9.
|
|
4
|
+
"version": "0.9.8",
|
|
5
5
|
"description": "MCP server exposing the engineering norms your firm is licensed for (SIA / Eurocodes) to any MCP client, via Stratta TreeRAG.",
|
|
6
6
|
"license": "UNLICENSED",
|
|
7
7
|
"author": "SmartFlow <hello@stratta.ch>",
|
|
@@ -24,6 +24,7 @@
|
|
|
24
24
|
"type": "module",
|
|
25
25
|
"main": "dist/index.js",
|
|
26
26
|
"bin": {
|
|
27
|
+
"stratta": "dist/index.js",
|
|
27
28
|
"stratta-mcp": "dist/index.js"
|
|
28
29
|
},
|
|
29
30
|
"files": [
|
|
@@ -77,10 +77,163 @@ def detect_running_text(doc: "fitz.Document") -> set[str]:
|
|
|
77
77
|
CHAP_UPPER_RE = re.compile(
|
|
78
78
|
r"^(\d+)\s+([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{3,120})\s*$", re.MULTILINE
|
|
79
79
|
)
|
|
80
|
+
# Norms adopted from CEN (SIA 262.6xx, SIA 267.1xx) set their headings in
|
|
81
|
+
# sentence case, which CHAP_UPPER_RE rejects by design. Used only as a second
|
|
82
|
+
# pass, when the uppercase form found almost nothing — on a native SIA norm it
|
|
83
|
+
# would promote body sentences to chapters.
|
|
84
|
+
CHAP_MIXED_RE = re.compile(
|
|
85
|
+
r"^(\d{1,2})\s+([A-ZÀ-Þ][A-Za-zÀ-ÿ][^\n]{2,90})\s*$", re.MULTILINE
|
|
86
|
+
)
|
|
80
87
|
ANNEX_BODY_RE = re.compile(
|
|
81
88
|
r"^(?:ANNEXE|Annexe)\s+([A-Z])(?:\s*\(([^)]+)\))?\s*(.*?)$", re.MULTILINE
|
|
82
89
|
)
|
|
83
90
|
|
|
91
|
+
# A heading split across two lines: the number alone, the title underneath.
|
|
92
|
+
# Comes from the numbering column of CEN-style layouts, where PyMuPDF reads the
|
|
93
|
+
# column before the text. Left as-is, every regex below misses the heading.
|
|
94
|
+
SPLIT_NUM_RE = re.compile(r"^\s*(\d{1,2}(?:\.\d{1,2}){0,3})\s*$")
|
|
95
|
+
SPLIT_TITLE_RE = re.compile(r"^\s*([A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-][^\n]{2,90})\s*$")
|
|
96
|
+
# Words that open a continuing sentence, never a heading.
|
|
97
|
+
SPLIT_STOP_RE = re.compile(
|
|
98
|
+
r"^(Le |La |Les |Il |Elle |Dans |Pour |Selon |Cette |Ce |Ces |Si |En |Au |Aux |De |Des |Du |Un |Une )",
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def join_split_headings(text: str) -> str:
|
|
103
|
+
"""Rewrite `12\\nTitre` as `12 Titre` so the heading regexes can see it.
|
|
104
|
+
|
|
105
|
+
Conservative on purpose: the title line must look like a title (starts
|
|
106
|
+
uppercase, no sentence-ending punctuation, not a sentence opener). A false
|
|
107
|
+
join invents a chapter, which is worse than missing one.
|
|
108
|
+
"""
|
|
109
|
+
lines = text.splitlines()
|
|
110
|
+
out: list[str] = []
|
|
111
|
+
i = 0
|
|
112
|
+
while i < len(lines):
|
|
113
|
+
m = SPLIT_NUM_RE.match(lines[i])
|
|
114
|
+
if m and i + 1 < len(lines):
|
|
115
|
+
nxt = lines[i + 1]
|
|
116
|
+
t = SPLIT_TITLE_RE.match(nxt)
|
|
117
|
+
if t and not nxt.rstrip().endswith((".", ",", ";", ":")) and not SPLIT_STOP_RE.match(t.group(1)):
|
|
118
|
+
out.append(f"{m.group(1)} {t.group(1).strip()}")
|
|
119
|
+
i += 2
|
|
120
|
+
continue
|
|
121
|
+
out.append(lines[i])
|
|
122
|
+
i += 1
|
|
123
|
+
return "\n".join(out)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def heading_text(doc: "fitz.Document", pageno: int) -> str:
|
|
127
|
+
"""Page text prepared for heading detection (never for section content)."""
|
|
128
|
+
return join_split_headings(doc[pageno].get_text("text"))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
NOT_A_TITLE_RE = re.compile(
|
|
132
|
+
r"^(?:EN|SN|ISO|DIN|SIA|NOTE|Tableau|Figure|Table|Bild)\b|^\W|\.$", re.IGNORECASE
|
|
133
|
+
)
|
|
134
|
+
# A heading cut mid-phrase by the column break: "Résistance à la flexion au".
|
|
135
|
+
TRUNCATED_RE = re.compile(
|
|
136
|
+
r"\b(?:ou|et|au|aux|de|des|du|le|la|les|un|une|dans|pour|par|sur|avec|sans|selon|entre)$",
|
|
137
|
+
re.IGNORECASE,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def plausible_title(title: str) -> bool:
|
|
142
|
+
"""Reject what a numbered line can be other than a heading.
|
|
143
|
+
|
|
144
|
+
A stray table cell ("I ~ ~"), a normative reference ("EN 12063:2024 (F)") or
|
|
145
|
+
a wrapped sentence all match the heading shape. Promoting one invents a
|
|
146
|
+
chapter, and an invented chapter swallows the page range of a real one.
|
|
147
|
+
"""
|
|
148
|
+
t = title.strip()
|
|
149
|
+
if not 3 <= len(t) <= 90:
|
|
150
|
+
return False
|
|
151
|
+
# A normative heading is capitalised. A lowercase one is a table row that
|
|
152
|
+
# happens to sit behind a number ("2.3 retrait des obstacles").
|
|
153
|
+
if not (t[0].isupper() or t[0].isdigit()):
|
|
154
|
+
return False
|
|
155
|
+
if NOT_A_TITLE_RE.search(t):
|
|
156
|
+
return False
|
|
157
|
+
if SPLIT_STOP_RE.match(t) or TRUNCATED_RE.search(t):
|
|
158
|
+
return False
|
|
159
|
+
letters = sum(1 for c in t if c.isalpha() or c.isspace())
|
|
160
|
+
return letters / len(t) >= 0.7
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def dense_heading_pages(doc: "fitz.Document", toc_pages: set[int]) -> set[int]:
|
|
164
|
+
"""Pages listing many numbered headings: a contents page, whatever its
|
|
165
|
+
typography. Detecting it by leader dots alone misses the ones set without
|
|
166
|
+
them, and every heading read there points at the contents page instead of
|
|
167
|
+
the section it names."""
|
|
168
|
+
dense: set[int] = set()
|
|
169
|
+
for i in range(doc.page_count):
|
|
170
|
+
if i in toc_pages:
|
|
171
|
+
continue
|
|
172
|
+
found = {m.group(1) for m in SUB_RE.finditer(heading_text(doc, i))}
|
|
173
|
+
# A page of the body drills into one chapter; a contents page walks
|
|
174
|
+
# across several. Counting headings alone would drop a dense page of
|
|
175
|
+
# definitions (3.1 … 3.8), which is real content.
|
|
176
|
+
if len(found) >= 5 and len({n.split(".", 1)[0] for n in found}) >= 3:
|
|
177
|
+
dense.add(i)
|
|
178
|
+
return dense
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def prune_chapters(chapters: dict[str, dict]) -> dict[str, dict]:
|
|
182
|
+
"""Keep the run of chapter numbers that reads like a table of contents.
|
|
183
|
+
|
|
184
|
+
Real chapters are consecutive and move forward through the document. A lone
|
|
185
|
+
"25" between chapters 1 and 2, or a chapter that starts thirty pages before
|
|
186
|
+
the one preceding it, is a detection artefact.
|
|
187
|
+
"""
|
|
188
|
+
numbered = sorted(((int(k), k) for k in chapters if k.isdigit()))
|
|
189
|
+
if not numbered:
|
|
190
|
+
return chapters
|
|
191
|
+
|
|
192
|
+
# A norm with N detected chapters does not have a chapter 40. Missing a few
|
|
193
|
+
# headings is normal; a number far past the count is a table cell.
|
|
194
|
+
ceiling = 2 * len(numbered) + 3
|
|
195
|
+
numbered = [(n, k) for n, k in numbered if n <= ceiling]
|
|
196
|
+
if not numbered:
|
|
197
|
+
return chapters
|
|
198
|
+
|
|
199
|
+
# Longest run whose pages move forward, so one bad page does not discard
|
|
200
|
+
# every chapter after it.
|
|
201
|
+
best = [1] * len(numbered)
|
|
202
|
+
prev = [-1] * len(numbered)
|
|
203
|
+
for i in range(len(numbered)):
|
|
204
|
+
page_i = chapters[numbered[i][1]]["pageStart"]
|
|
205
|
+
for j in range(i):
|
|
206
|
+
if chapters[numbered[j][1]]["pageStart"] <= page_i and best[j] + 1 > best[i]:
|
|
207
|
+
best[i], prev[i] = best[j] + 1, j
|
|
208
|
+
idx = best.index(max(best))
|
|
209
|
+
chain = []
|
|
210
|
+
while idx != -1:
|
|
211
|
+
chain.append(numbered[idx][1])
|
|
212
|
+
idx = prev[idx]
|
|
213
|
+
|
|
214
|
+
# Missing a heading or two leaves a small gap; doubling (10 → 20 → 30) means
|
|
215
|
+
# the numbers stopped being chapters and started being table rows.
|
|
216
|
+
kept: dict[str, dict] = {}
|
|
217
|
+
previous = None
|
|
218
|
+
for key in reversed(chain):
|
|
219
|
+
n = int(key)
|
|
220
|
+
if previous is not None and n - previous > 3 and n >= 2 * previous:
|
|
221
|
+
break
|
|
222
|
+
kept[key] = chapters[key]
|
|
223
|
+
previous = n
|
|
224
|
+
|
|
225
|
+
# Annexes close a norm, in order. One that lands before the last chapter was
|
|
226
|
+
# read off the contents page, and its page range would swallow the document.
|
|
227
|
+
last_page = max((m["pageStart"] for m in kept.values()), default=0)
|
|
228
|
+
for key, meta in sorted(
|
|
229
|
+
((k, m) for k, m in chapters.items() if not k.isdigit()), key=lambda kv: kv[0]
|
|
230
|
+
):
|
|
231
|
+
if meta["pageStart"] < last_page:
|
|
232
|
+
continue
|
|
233
|
+
kept[key] = meta
|
|
234
|
+
last_page = meta["pageStart"]
|
|
235
|
+
return kept
|
|
236
|
+
|
|
84
237
|
|
|
85
238
|
def extract_chapters(
|
|
86
239
|
doc: "fitz.Document", toc_pages: set[int] | None = None
|
|
@@ -105,32 +258,64 @@ def extract_chapters(
|
|
|
105
258
|
if chapters:
|
|
106
259
|
return chapters
|
|
107
260
|
|
|
108
|
-
# Fallback: no bookmarks → scan body for
|
|
109
|
-
skip = toc_pages or set()
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
num = m.group(1)
|
|
116
|
-
title = clean(m.group(2))
|
|
117
|
-
if int(num) > 50:
|
|
118
|
-
continue
|
|
119
|
-
if num in chapters:
|
|
120
|
-
continue
|
|
121
|
-
chapters[num] = {"title": title, "pageStart": pageno + 1}
|
|
122
|
-
for m in ANNEX_BODY_RE.finditer(text):
|
|
123
|
-
letter = m.group(1)
|
|
124
|
-
kind = m.group(2) or ""
|
|
125
|
-
rest = clean(m.group(3) or "")
|
|
126
|
-
path = f"Annexe {letter}"
|
|
127
|
-
if path in chapters:
|
|
261
|
+
# Fallback: no bookmarks → scan body for chapter headings.
|
|
262
|
+
skip = set(toc_pages or set())
|
|
263
|
+
|
|
264
|
+
def scan(pattern: "re.Pattern[str]", skip_pages: set[int]) -> dict[str, dict]:
|
|
265
|
+
found: dict[str, dict] = {}
|
|
266
|
+
for pageno in range(doc.page_count):
|
|
267
|
+
if pageno in skip_pages:
|
|
128
268
|
continue
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
269
|
+
text = heading_text(doc, pageno)
|
|
270
|
+
for m in pattern.finditer(text):
|
|
271
|
+
num, title = m.group(1), clean(m.group(2))
|
|
272
|
+
if int(num) > 50 or num in found or not plausible_title(title):
|
|
273
|
+
continue
|
|
274
|
+
found[num] = {"title": title, "pageStart": pageno + 1}
|
|
275
|
+
for m in ANNEX_BODY_RE.finditer(text):
|
|
276
|
+
letter, kind = m.group(1), m.group(2) or ""
|
|
277
|
+
rest = clean(m.group(3) or "")
|
|
278
|
+
path = f"Annexe {letter}"
|
|
279
|
+
if path in found:
|
|
280
|
+
continue
|
|
281
|
+
t = path + (f" ({kind})" if kind else "")
|
|
282
|
+
if rest and rest.lower() != path.lower():
|
|
283
|
+
t += f" — {rest}"
|
|
284
|
+
found[path] = {"title": t, "pageStart": pageno + 1}
|
|
285
|
+
return found
|
|
286
|
+
|
|
287
|
+
def toc_like_pages(found: dict[str, dict]) -> set[int]:
|
|
288
|
+
"""Pages holding three or more chapter headings are the table of
|
|
289
|
+
contents, whatever the typography. Without this every chapter of such a
|
|
290
|
+
norm starts on the contents page, and every section body is wrong."""
|
|
291
|
+
per_page: dict[int, int] = defaultdict(int)
|
|
292
|
+
for meta in found.values():
|
|
293
|
+
per_page[meta["pageStart"]] += 1
|
|
294
|
+
return {p - 1 for p, n in per_page.items() if n >= 3}
|
|
295
|
+
|
|
296
|
+
for pattern in (CHAP_UPPER_RE, CHAP_MIXED_RE):
|
|
297
|
+
chapters = scan(pattern, skip)
|
|
298
|
+
extra = toc_like_pages(chapters)
|
|
299
|
+
if extra:
|
|
300
|
+
chapters = scan(pattern, skip | extra)
|
|
301
|
+
if len([k for k in chapters if k.isdigit()]) >= 3:
|
|
302
|
+
break
|
|
303
|
+
|
|
304
|
+
return prune_chapters(chapters)
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def chapter_page_spans(chapters: dict[str, dict], n_pages: int) -> dict[str, tuple[int, int]]:
|
|
308
|
+
"""First and last page of each numbered chapter, from where the next starts."""
|
|
309
|
+
ordered = sorted(
|
|
310
|
+
((k, v["pageStart"]) for k, v in chapters.items()),
|
|
311
|
+
key=lambda kv: kv[1],
|
|
312
|
+
)
|
|
313
|
+
spans: dict[str, tuple[int, int]] = {}
|
|
314
|
+
for idx, (key, start) in enumerate(ordered):
|
|
315
|
+
end = ordered[idx + 1][1] - 1 if idx + 1 < len(ordered) else n_pages
|
|
316
|
+
if key.isdigit():
|
|
317
|
+
spans[key] = (start, max(start, end))
|
|
318
|
+
return spans
|
|
134
319
|
|
|
135
320
|
|
|
136
321
|
def extract_subsections(
|
|
@@ -138,13 +323,15 @@ def extract_subsections(
|
|
|
138
323
|
chap_prefixes: set[str],
|
|
139
324
|
toc_pages: set[int],
|
|
140
325
|
running: set[str],
|
|
326
|
+
chapter_spans: dict[str, tuple[int, int]] | None = None,
|
|
141
327
|
) -> dict[str, tuple[str, int]]:
|
|
142
328
|
"""Numeric sub-sections (depths 2-4) detected in body pages."""
|
|
143
329
|
out: dict[str, tuple[str, int]] = {}
|
|
330
|
+
skip = set(toc_pages) | dense_heading_pages(doc, toc_pages)
|
|
144
331
|
for i in range(doc.page_count):
|
|
145
|
-
if i in
|
|
332
|
+
if i in skip:
|
|
146
333
|
continue
|
|
147
|
-
text = doc
|
|
334
|
+
text = heading_text(doc, i)
|
|
148
335
|
for m in SUB_RE.finditer(text):
|
|
149
336
|
num = m.group(1)
|
|
150
337
|
title = clean(m.group(2))
|
|
@@ -153,7 +340,13 @@ def extract_subsections(
|
|
|
153
340
|
top = num.split(".", 1)[0]
|
|
154
341
|
if top not in chap_prefixes:
|
|
155
342
|
continue
|
|
156
|
-
|
|
343
|
+
# A sub-section sits inside its chapter. "1.1" found sixty pages
|
|
344
|
+
# after chapter 1 ended is a numbered line in an annexe, and keeping
|
|
345
|
+
# it hangs an unrelated page under the wrong parent.
|
|
346
|
+
span = chapter_spans.get(top) if chapter_spans else None
|
|
347
|
+
if span and not span[0] <= i + 1 <= span[1]:
|
|
348
|
+
continue
|
|
349
|
+
if not plausible_title(title):
|
|
157
350
|
continue
|
|
158
351
|
if num in out:
|
|
159
352
|
continue
|
|
@@ -218,6 +411,44 @@ def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
|
|
|
218
411
|
chunks = page_texts[ps - 1 : pe]
|
|
219
412
|
node["rawText"] = "\n".join(chunks).strip()
|
|
220
413
|
|
|
414
|
+
trim_shared_pages(nodes, page_texts)
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def trim_shared_pages(nodes: list[dict], page_texts: list[str]) -> None:
|
|
418
|
+
"""Cut the opening page where several sections start on it.
|
|
419
|
+
|
|
420
|
+
Page granularity is the unit everywhere else, but two sections beginning on
|
|
421
|
+
the same page would otherwise carry byte-identical text — so a summary reads
|
|
422
|
+
like its neighbour's, and the table of contents misleads before anything is
|
|
423
|
+
even opened.
|
|
424
|
+
"""
|
|
425
|
+
by_page: dict[int, list[dict]] = defaultdict(list)
|
|
426
|
+
for node in nodes:
|
|
427
|
+
by_page[node["pageStart"]].append(node)
|
|
428
|
+
|
|
429
|
+
for page, group in by_page.items():
|
|
430
|
+
if len(group) < 2:
|
|
431
|
+
continue
|
|
432
|
+
text = page_texts[page - 1]
|
|
433
|
+
# Where each section's heading sits on that page, in reading order.
|
|
434
|
+
marks: list[tuple[int, dict]] = []
|
|
435
|
+
for node in group:
|
|
436
|
+
title = node["title"].split(" — ")[-1].strip()
|
|
437
|
+
at = text.find(title) if len(title) >= 4 else -1
|
|
438
|
+
if at == -1:
|
|
439
|
+
at = text.find(f"{node['path']} ")
|
|
440
|
+
marks.append((at, node))
|
|
441
|
+
if any(at == -1 for at, _ in marks):
|
|
442
|
+
continue # a heading we cannot locate: leave the whole page in place
|
|
443
|
+
marks.sort(key=lambda m: m[0])
|
|
444
|
+
for i, (at, node) in enumerate(marks):
|
|
445
|
+
end = marks[i + 1][0] if i + 1 < len(marks) else len(text)
|
|
446
|
+
head = text[at:end].strip()
|
|
447
|
+
if not head:
|
|
448
|
+
continue
|
|
449
|
+
rest = "\n".join(page_texts[node["pageStart"] : node["pageEnd"]]).strip()
|
|
450
|
+
node["rawText"] = f"{head}\n{rest}".strip() if rest else head
|
|
451
|
+
|
|
221
452
|
|
|
222
453
|
def extract_figures(
|
|
223
454
|
doc: "fitz.Document",
|
|
@@ -295,7 +526,8 @@ def main() -> None:
|
|
|
295
526
|
running = detect_running_text(doc)
|
|
296
527
|
chapters = extract_chapters(doc, toc_pages)
|
|
297
528
|
chap_prefixes = {k for k in chapters if k.isdigit()}
|
|
298
|
-
|
|
529
|
+
spans = chapter_page_spans(chapters, doc.page_count)
|
|
530
|
+
subsections = extract_subsections(doc, chap_prefixes, toc_pages, running, spans)
|
|
299
531
|
nodes = build_tree(chapters, subsections, doc.page_count)
|
|
300
532
|
extract_section_text(doc, nodes)
|
|
301
533
|
figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)
|