@stratta/mcp 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -3
- package/dist/confirm.d.ts +15 -0
- package/dist/confirm.js +54 -0
- package/dist/errors.js +27 -0
- package/dist/index.js +27 -2
- package/dist/prompts.d.ts +4 -0
- package/dist/prompts.js +40 -0
- package/dist/resources.d.ts +3 -0
- package/dist/resources.js +35 -0
- package/dist/tools/catalog.gen.d.ts +34 -0
- package/dist/tools/catalog.gen.js +724 -0
- package/dist/tools/dossier.d.ts +48 -0
- package/dist/tools/dossier.js +249 -127
- package/dist/tools/ingest.d.ts +8 -0
- package/dist/tools/ingest.js +116 -98
- package/dist/tools/read.d.ts +42 -17
- package/dist/tools/read.js +160 -84
- package/dist/usage.d.ts +3 -0
- package/dist/usage.js +50 -0
- package/package.json +1 -1
- package/scripts/ingest-prepass.py +499 -54
- package/scripts/tests/test_prepass.py +211 -0
- package/skills/consult-stratta/SKILL.md +73 -0
- package/skills/ingest-norm/SKILL.md +99 -44
- package/skills/verification-note/SKILL.md +54 -0
|
@@ -2,10 +2,13 @@
|
|
|
2
2
|
"""Stratta — ingest pre-pass.
|
|
3
3
|
|
|
4
4
|
Reads a norm PDF with PyMuPDF, builds the hierarchical TreeRAG skeleton (chapters
|
|
5
|
-
from PDF bookmarks + sub-sections via heading regex
|
|
6
|
-
text
|
|
7
|
-
|
|
8
|
-
|
|
5
|
+
from PDF bookmarks + sub-sections via heading regex, down to X.Y.Z.W), gives
|
|
6
|
+
each node the text between its heading and the next one on cleaned pages (no
|
|
7
|
+
running headers, no page numbers, hyphenation resolved, clause numbers joined
|
|
8
|
+
to their paragraph), and rasterizes each page that contains a figure caption.
|
|
9
|
+
Output is a single JSON consumed by the `ingest-norm` skill, which then
|
|
10
|
+
enriches sections (formulas, tables, cross-refs, summaries) and uploads the
|
|
11
|
+
figures. `warnings[]` lists what the pre-pass could not decide on its own.
|
|
9
12
|
|
|
10
13
|
Usage:
|
|
11
14
|
python scripts/ingest-prepass.py --pdf <path> --output <dir> [--language fr]
|
|
@@ -54,23 +57,137 @@ def clean(s: str) -> str:
|
|
|
54
57
|
return re.sub(r"\s+", " ", CLEAN_DOTS.sub("", s)).strip()
|
|
55
58
|
|
|
56
59
|
|
|
60
|
+
# --- page lines ----------------------------------------------------------------
|
|
61
|
+
#
|
|
62
|
+
# PyMuPDF's plain text follows the PDF's own block order. On a SIA norm the
|
|
63
|
+
# clause numbers sit in a column of their own, and on an OCR'd copy (the form
|
|
64
|
+
# a bureau actually holds: 60 of the 80 PDFs of the first client corpus are
|
|
65
|
+
# scans) that column is read as one block BEFORE the paragraphs, so a page
|
|
66
|
+
# arrives as "4.4 / 4.4.1 / 4.4.1.1 / ... / Résistances du terrain /
|
|
67
|
+
# Généralités / Sont considérés ..." and no heading regex can see a heading.
|
|
68
|
+
#
|
|
69
|
+
# Lines are therefore rebuilt from the words and their positions: words on the
|
|
70
|
+
# same baseline form a line, left to right. That puts "4.4 Résistances du
|
|
71
|
+
# terrain" back together on both the native and the OCR'd copy. A genuine
|
|
72
|
+
# two-column page (two text columns, not a numbering column) would be
|
|
73
|
+
# scrambled by that rule, so such pages are detected and read in block order.
|
|
74
|
+
|
|
75
|
+
_LINES_CACHE: dict[tuple[int, int], list[str]] = {}
|
|
76
|
+
NUMBER_TOKEN_RE = re.compile(r"^(?:\d{1,2}(?:\.\d{1,3}){0,4}\.?|[A-Z](?:\.\d{1,2}){1,3}|—|-|•)$")
|
|
77
|
+
OCR_DOTTED_NUMBER_RE = re.compile(r"^(\d+(?:\.\d+)*) \.(\d)")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _rows(page: "fitz.Page") -> list[list[tuple]]:
|
|
81
|
+
words = page.get_text("words") # x0, y0, x1, y1, text, block, line, word
|
|
82
|
+
if not words:
|
|
83
|
+
return []
|
|
84
|
+
heights = sorted(w[3] - w[1] for w in words)
|
|
85
|
+
median = heights[len(heights) // 2]
|
|
86
|
+
# A word set vertically (a library watermark such as "Ecole Polytechnique
|
|
87
|
+
# Fédérale de Lausanne" running up the margin) has a box far taller than
|
|
88
|
+
# the text. Left in, its words land on every line they cross.
|
|
89
|
+
words = [w for w in words if (w[3] - w[1]) <= 2.5 * median]
|
|
90
|
+
if not words:
|
|
91
|
+
return []
|
|
92
|
+
tol = max(2.0, 0.45 * median)
|
|
93
|
+
rows: list[list[tuple]] = []
|
|
94
|
+
for w in sorted(words, key=lambda w: (w[1], w[0])):
|
|
95
|
+
if rows and abs(rows[-1][0][1] - w[1]) <= tol:
|
|
96
|
+
rows[-1].append(w)
|
|
97
|
+
else:
|
|
98
|
+
rows.append([w])
|
|
99
|
+
for row in rows:
|
|
100
|
+
row.sort(key=lambda w: w[0])
|
|
101
|
+
return rows
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _segments(row: list[tuple], gap: float) -> list[list[tuple]]:
|
|
105
|
+
"""Split a row where the horizontal gap between words is a column gutter."""
|
|
106
|
+
out: list[list[tuple]] = [[row[0]]]
|
|
107
|
+
for prev, w in zip(row, row[1:]):
|
|
108
|
+
if w[0] - prev[2] > gap:
|
|
109
|
+
out.append([w])
|
|
110
|
+
else:
|
|
111
|
+
out[-1].append(w)
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _two_column(rows: list[list[list[tuple]]]) -> bool:
|
|
116
|
+
"""Two text columns: many rows carry two segments that both read as prose."""
|
|
117
|
+
if len(rows) < 12:
|
|
118
|
+
return False
|
|
119
|
+
both = 0
|
|
120
|
+
for segs in rows:
|
|
121
|
+
texts = [" ".join(w[4] for w in s) for s in segs]
|
|
122
|
+
prose = [t for t in texts if len(t) > 24 and not NUMBER_TOKEN_RE.match(t.split()[0])]
|
|
123
|
+
if len(prose) >= 2:
|
|
124
|
+
both += 1
|
|
125
|
+
return both / len(rows) > 0.3
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def page_lines(doc: "fitz.Document", pageno: int) -> list[str]:
|
|
129
|
+
"""The lines of a page, in reading order, numbers joined to their text."""
|
|
130
|
+
key = (id(doc), pageno)
|
|
131
|
+
cached = _LINES_CACHE.get(key)
|
|
132
|
+
if cached is not None:
|
|
133
|
+
return cached
|
|
134
|
+
page = doc[pageno]
|
|
135
|
+
rows = _rows(page)
|
|
136
|
+
gap = 0.12 * page.rect.width
|
|
137
|
+
segmented = [_segments(row, gap) for row in rows]
|
|
138
|
+
if _two_column(segmented):
|
|
139
|
+
lines = join_split_headings(page.get_text("text")).splitlines()
|
|
140
|
+
else:
|
|
141
|
+
lines = []
|
|
142
|
+
for segs in segmented:
|
|
143
|
+
text = " ".join(" ".join(w[4] for w in s) for s in segs)
|
|
144
|
+
text = OCR_DOTTED_NUMBER_RE.sub(r"\1.\2", text)
|
|
145
|
+
lines.append(text)
|
|
146
|
+
lines = [l.rstrip() for l in lines]
|
|
147
|
+
_LINES_CACHE[key] = lines
|
|
148
|
+
return lines
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def page_text(doc: "fitz.Document", pageno: int) -> str:
|
|
152
|
+
return "\n".join(page_lines(doc, pageno))
|
|
153
|
+
|
|
154
|
+
|
|
57
155
|
def detect_toc_pages(doc: "fitz.Document") -> set[int]:
|
|
58
156
|
out = set()
|
|
59
157
|
for i in range(doc.page_count):
|
|
60
|
-
lines = [l for l in doc
|
|
158
|
+
lines = [l for l in page_lines(doc, i) if l.strip()]
|
|
61
159
|
leader = sum(1 for l in lines if TOC_DOT_RE.search(l))
|
|
62
160
|
if leader >= 5:
|
|
63
161
|
out.add(i)
|
|
64
162
|
return out
|
|
65
163
|
|
|
66
164
|
|
|
165
|
+
# Standalone integers only: the "9" and "1" of a heading "9.1 Délimitation"
|
|
166
|
+
# stay, or every "Généralités" heading would read as one running line.
|
|
167
|
+
FURNITURE_NUMBER_RE = re.compile(r"(?<![\d.])\d{1,4}(?![\d.])")
|
|
168
|
+
FURNITURE_EDGE_RE = re.compile(r"^[\s/|\-–—·.]+|[\s/|\-–—·.]+$")
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def furniture_key(line: str) -> str:
|
|
172
|
+
"""A running line without the page number it carries.
|
|
173
|
+
|
|
174
|
+
`SIA 267, Copyright © 2013 by SIA Zurich 59` on an odd page and
|
|
175
|
+
`64 SIA 267, Copyright © 2013 by SIA Zurich` on an even one are the same
|
|
176
|
+
footer; so are the copies a watermark decorates with a stray slash.
|
|
177
|
+
"""
|
|
178
|
+
s = FURNITURE_NUMBER_RE.sub(" ", line)
|
|
179
|
+
s = re.sub(r"\s+", " ", s).strip()
|
|
180
|
+
return FURNITURE_EDGE_RE.sub("", s).strip()
|
|
181
|
+
|
|
182
|
+
|
|
67
183
|
def detect_running_text(doc: "fitz.Document") -> set[str]:
|
|
68
|
-
"""Lines appearing on >=5 pages near the top/bottom (running headers/footers)
|
|
184
|
+
"""Lines appearing on >=5 pages near the top/bottom (running headers/footers),
|
|
185
|
+
keyed without their page number."""
|
|
69
186
|
counter: dict[str, int] = defaultdict(int)
|
|
70
187
|
for i in range(doc.page_count):
|
|
71
|
-
lines = [l.strip() for l in doc
|
|
188
|
+
lines = [l.strip() for l in page_lines(doc, i) if l.strip()]
|
|
72
189
|
for l in lines[:3] + lines[-3:]:
|
|
73
|
-
counter[l] += 1
|
|
190
|
+
counter[furniture_key(l)] += 1
|
|
74
191
|
return {l for l, c in counter.items() if c >= 5 and len(l) > 8}
|
|
75
192
|
|
|
76
193
|
|
|
@@ -124,8 +241,8 @@ def join_split_headings(text: str) -> str:
|
|
|
124
241
|
|
|
125
242
|
|
|
126
243
|
def heading_text(doc: "fitz.Document", pageno: int) -> str:
|
|
127
|
-
"""Page text prepared for heading detection
|
|
128
|
-
return join_split_headings(doc
|
|
244
|
+
"""Page text prepared for heading detection."""
|
|
245
|
+
return join_split_headings(page_text(doc, pageno))
|
|
129
246
|
|
|
130
247
|
|
|
131
248
|
NOT_A_TITLE_RE = re.compile(
|
|
@@ -212,12 +329,14 @@ def prune_chapters(chapters: dict[str, dict]) -> dict[str, dict]:
|
|
|
212
329
|
idx = prev[idx]
|
|
213
330
|
|
|
214
331
|
# Missing a heading or two leaves a small gap; doubling (10 → 20 → 30) means
|
|
215
|
-
# the numbers stopped being chapters and started being table rows.
|
|
332
|
+
# the numbers stopped being chapters and started being table rows. Only
|
|
333
|
+
# from 5 upwards: an OCR'd copy that lost chapters 3 to 6 jumps from 2 to
|
|
334
|
+
# 7, and that is a hole, not a doubling.
|
|
216
335
|
kept: dict[str, dict] = {}
|
|
217
336
|
previous = None
|
|
218
337
|
for key in reversed(chain):
|
|
219
338
|
n = int(key)
|
|
220
|
-
if previous is not None and n - previous > 3 and n >= 2 * previous:
|
|
339
|
+
if previous is not None and previous >= 5 and n - previous > 3 and n >= 2 * previous:
|
|
221
340
|
break
|
|
222
341
|
kept[key] = chapters[key]
|
|
223
342
|
previous = n
|
|
@@ -301,7 +420,106 @@ def extract_chapters(
|
|
|
301
420
|
if len([k for k in chapters if k.isdigit()]) >= 3:
|
|
302
421
|
break
|
|
303
422
|
|
|
304
|
-
|
|
423
|
+
for num, meta in infer_unnumbered_chapters(doc, skip).items():
|
|
424
|
+
chapters.setdefault(num, meta)
|
|
425
|
+
chapters = prune_chapters(chapters)
|
|
426
|
+
return recover_missing_chapters(doc, chapters, skip)
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
SCOPE_TITLE_RE = re.compile(
|
|
430
|
+
r"^(DOMAINE D.APPLICATION|GELTUNGSBEREICH|CAMPO D.APPLICAZIONE|SCOPE)\b", re.I
|
|
431
|
+
)
|
|
432
|
+
|
|
433
|
+
|
|
434
|
+
def _page_top_titles(doc: "fitz.Document", pages: range, skip: set[int]) -> list[tuple[int, str]]:
|
|
435
|
+
"""Uppercase, plausible titles among the first lines of each page: where a
|
|
436
|
+
chapter of a SIA norm starts. Deeper in a page a capitalised line is a
|
|
437
|
+
table header or a note."""
|
|
438
|
+
out: list[tuple[int, str]] = []
|
|
439
|
+
for pageno in pages:
|
|
440
|
+
if pageno in skip:
|
|
441
|
+
continue
|
|
442
|
+
for line in [l.strip() for l in page_lines(doc, pageno)][:3]:
|
|
443
|
+
m = UNNUMBERED_UPPER_RE.match(line)
|
|
444
|
+
if m and plausible_title(m.group(1)) and not re.match(r"^(TABLEAU|TABELLE|TABELLA|TABLE|FIGURE|FIG\.|BILD|ANNEXE|ANHANG|ANNEX)\b", line):
|
|
445
|
+
out.append((pageno + 1, clean(m.group(1))))
|
|
446
|
+
break
|
|
447
|
+
return out
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def recover_missing_chapters(doc: "fitz.Document", chapters: dict[str, dict], skip: set[int]) -> dict[str, dict]:
|
|
451
|
+
"""Fill holes in the chapter sequence by position.
|
|
452
|
+
|
|
453
|
+
On the OCR'd copies the numbering column is sometimes lost on a whole run
|
|
454
|
+
of pages: the chapter title survives, and so does the first sub-heading,
|
|
455
|
+
but neither carries its number, so `infer_unnumbered_chapters` has nothing
|
|
456
|
+
to read. When the sequence goes 1, 2, 6 and exactly three title-only pages
|
|
457
|
+
lie between chapter 2 and chapter 6, those are chapters 3, 4 and 5, in
|
|
458
|
+
page order. Anything less certain is left as a hole for the agent.
|
|
459
|
+
|
|
460
|
+
A leading "Domaine d'application" before chapter 1 is chapter 0, which is
|
|
461
|
+
how SIA numbers it.
|
|
462
|
+
"""
|
|
463
|
+
numbered = sorted((int(k), k) for k in chapters if k.isdigit())
|
|
464
|
+
if not numbered:
|
|
465
|
+
return chapters
|
|
466
|
+
used = {meta["title"].upper() for meta in chapters.values()}
|
|
467
|
+
recovered: dict[str, dict] = {}
|
|
468
|
+
|
|
469
|
+
for (a, ka), (b, kb) in zip(numbered, numbered[1:]):
|
|
470
|
+
holes = list(range(a + 1, b))
|
|
471
|
+
if not holes:
|
|
472
|
+
continue
|
|
473
|
+
start, end = chapters[ka]["pageStart"], chapters[kb]["pageStart"]
|
|
474
|
+
candidates = [
|
|
475
|
+
(page, title)
|
|
476
|
+
for page, title in _page_top_titles(doc, range(start, end - 1), skip)
|
|
477
|
+
if title.upper() not in used
|
|
478
|
+
]
|
|
479
|
+
if len(candidates) != len(holes):
|
|
480
|
+
continue
|
|
481
|
+
for n, (page, title) in zip(holes, candidates):
|
|
482
|
+
recovered[str(n)] = {"title": title, "pageStart": page, "recovered": True}
|
|
483
|
+
|
|
484
|
+
first_n, first_k = numbered[0]
|
|
485
|
+
if first_n == 1 and "0" not in chapters:
|
|
486
|
+
before = _page_top_titles(doc, range(0, chapters[first_k]["pageStart"] - 1), skip)
|
|
487
|
+
scope = [(p, t) for p, t in before if SCOPE_TITLE_RE.match(t)]
|
|
488
|
+
if len(scope) == 1:
|
|
489
|
+
recovered["0"] = {"title": scope[0][1], "pageStart": scope[0][0], "recovered": True}
|
|
490
|
+
|
|
491
|
+
return {**chapters, **recovered}
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
UNNUMBERED_UPPER_RE = re.compile(r"^([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{5,120})$")
|
|
495
|
+
FIRST_SUB_RE = re.compile(r"^(\d{1,2})\.\d{1,2}\b")
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def infer_unnumbered_chapters(doc: "fitz.Document", skip: set[int]) -> dict[str, dict]:
|
|
499
|
+
"""Chapters whose number the OCR lost.
|
|
500
|
+
|
|
501
|
+
On a scanned copy the large chapter number is often the one glyph the OCR
|
|
502
|
+
does not recognise, so the page reads "FONDATIONS SUR PIEUX" followed by
|
|
503
|
+
"9.1 Délimitation". The number is then taken from the first sub-heading
|
|
504
|
+
under the title. Only fills holes: a chapter found with its number wins.
|
|
505
|
+
"""
|
|
506
|
+
found: dict[str, dict] = {}
|
|
507
|
+
for pageno in range(doc.page_count):
|
|
508
|
+
if pageno in skip:
|
|
509
|
+
continue
|
|
510
|
+
lines = [l.strip() for l in page_lines(doc, pageno)]
|
|
511
|
+
for i, line in enumerate(lines):
|
|
512
|
+
m = UNNUMBERED_UPPER_RE.match(line)
|
|
513
|
+
if not m or not plausible_title(m.group(1)):
|
|
514
|
+
continue
|
|
515
|
+
for follow in lines[i + 1 : i + 9]:
|
|
516
|
+
sub = FIRST_SUB_RE.match(follow)
|
|
517
|
+
if sub:
|
|
518
|
+
num = sub.group(1)
|
|
519
|
+
if num not in found:
|
|
520
|
+
found[num] = {"title": clean(m.group(1)), "pageStart": pageno + 1}
|
|
521
|
+
break
|
|
522
|
+
return found
|
|
305
523
|
|
|
306
524
|
|
|
307
525
|
def chapter_page_spans(chapters: dict[str, dict], n_pages: int) -> dict[str, tuple[int, int]]:
|
|
@@ -340,6 +558,11 @@ def extract_subsections(
|
|
|
340
558
|
top = num.split(".", 1)[0]
|
|
341
559
|
if top not in chap_prefixes:
|
|
342
560
|
continue
|
|
561
|
+
# "9.522" is "9.5.2.2" with a dot the OCR dropped. No norm has a
|
|
562
|
+
# fortieth sub-section; keeping it would hang a page of 9.5 under
|
|
563
|
+
# a sibling of 9.7 and knock the real 9.6 and 9.7 out of sequence.
|
|
564
|
+
if any(int(part) > 40 for part in num.split(".")[1:]):
|
|
565
|
+
continue
|
|
343
566
|
# A sub-section sits inside its chapter. "1.1" found sixty pages
|
|
344
567
|
# after chapter 1 ended is a numbered line in an annexe, and keeping
|
|
345
568
|
# it hangs an unrelated page under the wrong parent.
|
|
@@ -403,51 +626,241 @@ def slugify(path: str) -> str:
|
|
|
403
626
|
return re.sub(r"[^a-zA-Z0-9]+", "-", path).strip("-").lower() or "root"
|
|
404
627
|
|
|
405
628
|
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
629
|
+
# --- section text ------------------------------------------------------------
|
|
630
|
+
#
|
|
631
|
+
# Until 2026-09-04 a node's text was the concatenation of the pages it spanned,
|
|
632
|
+
# which meant three things the audit measured on the production corpus
|
|
633
|
+
# (ADR 22 § 2.1): a chapter carried the full text of all its sub-sections
|
|
634
|
+
# (61 000 characters twice for SIA 267.153 § 11 and § 11.3), the running header
|
|
635
|
+
# and the page number of every page sat in the middle of the prose, and words
|
|
636
|
+
# hyphenated at a line end stayed broken ("ter- rain"). The text of a node is
|
|
637
|
+
# now what lies between its heading and the next heading, on cleaned pages.
|
|
413
638
|
|
|
414
|
-
|
|
639
|
+
PAGE_NUMBER_RE = re.compile(r"^\s*\d{1,4}\s*$")
|
|
640
|
+
CLAUSE_NUMBER_RE = re.compile(r"^\s*(\d{1,2}(?:\.\d{1,3}){1,4})\s*$")
|
|
641
|
+
HYPHEN_BREAK_RE = re.compile(r"([a-zà-ÿ])[-—]\n([a-zà-ÿ])")
|
|
415
642
|
|
|
416
643
|
|
|
417
|
-
def
|
|
418
|
-
"""
|
|
644
|
+
def dehyphenate(text: str) -> str:
|
|
645
|
+
"""Join a word broken at a line end: `tra-\\nvaux` becomes `travaux`.
|
|
419
646
|
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
even opened.
|
|
647
|
+
A compound word split at its hyphen loses it too (`pieux-\\nradier` reads
|
|
648
|
+
`pieuxradier`); syllable breaks outnumber compounds at line ends by far,
|
|
649
|
+
and a broken word is worse for search than a missing hyphen.
|
|
424
650
|
"""
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
651
|
+
return HYPHEN_BREAK_RE.sub(r"\1\2", text)
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def clean_page(text: str, running: set[str], page_count: int) -> str:
|
|
655
|
+
"""Drop what is page furniture rather than norm text.
|
|
428
656
|
|
|
429
|
-
|
|
430
|
-
|
|
657
|
+
- the running header and footer lines `detect_running_text` found (they
|
|
658
|
+
were detected before, and never removed);
|
|
659
|
+
- a bare page number in the three first or three last lines of the page;
|
|
660
|
+
- trailing whitespace on every line.
|
|
661
|
+
"""
|
|
662
|
+
lines = [l.rstrip() for l in text.splitlines()]
|
|
663
|
+
kept: list[str] = []
|
|
664
|
+
n = len(lines)
|
|
665
|
+
for i, line in enumerate(lines):
|
|
666
|
+
s = line.strip()
|
|
667
|
+
if s and (s in running or furniture_key(s) in running):
|
|
431
668
|
continue
|
|
432
|
-
|
|
433
|
-
#
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
669
|
+
# Printed page numbers run a little past the PDF's page count when
|
|
670
|
+
# the front matter is numbered separately; well past it, a bare
|
|
671
|
+
# number in the margin is a table value.
|
|
672
|
+
near_edge = i < 3 or i >= n - 3
|
|
673
|
+
if near_edge and PAGE_NUMBER_RE.match(s) and int(s) <= page_count + 20:
|
|
674
|
+
continue
|
|
675
|
+
kept.append(line)
|
|
676
|
+
return "\n".join(kept)
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
def join_clause_numbers(text: str) -> str:
|
|
680
|
+
"""Rewrite `9.5.1.1\\nLes pieux…` as `9.5.1.1 Les pieux…`.
|
|
681
|
+
|
|
682
|
+
PyMuPDF reads the numbering column before the paragraph, so every numbered
|
|
683
|
+
clause of a SIA norm arrives as a number on its own line. Joined, the
|
|
684
|
+
number becomes an anchor the reader can search for; alone, it is noise.
|
|
685
|
+
Content only: heading detection keeps its stricter rule
|
|
686
|
+
(`join_split_headings`), because a clause number followed by a sentence is
|
|
687
|
+
exactly what must NOT become a heading.
|
|
688
|
+
"""
|
|
689
|
+
lines = text.splitlines()
|
|
690
|
+
out: list[str] = []
|
|
691
|
+
i = 0
|
|
692
|
+
while i < len(lines):
|
|
693
|
+
m = CLAUSE_NUMBER_RE.match(lines[i])
|
|
694
|
+
if m and i + 1 < len(lines) and lines[i + 1].strip():
|
|
695
|
+
out.append(f"{m.group(1)} {lines[i + 1].strip()}")
|
|
696
|
+
i += 2
|
|
697
|
+
continue
|
|
698
|
+
out.append(lines[i])
|
|
699
|
+
i += 1
|
|
700
|
+
return "\n".join(out)
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
def tidy(text: str) -> str:
|
|
704
|
+
"""Collapse runs of blank lines; strip the ends."""
|
|
705
|
+
return re.sub(r"\n{3,}", "\n\n", text).strip()
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
def locate_heading(lines: list[str], node: dict) -> int:
|
|
709
|
+
"""Index of the line carrying this node's heading on its start page, or -1.
|
|
710
|
+
|
|
711
|
+
Matches on the number first (`9.5 ` at the start of a line), then on the
|
|
712
|
+
first words of the title, so a title PyMuPDF wrapped differently from the
|
|
713
|
+
contents page is still found.
|
|
714
|
+
"""
|
|
715
|
+
path = node["path"]
|
|
716
|
+
title = node["title"].split(" — ")[-1].strip()
|
|
717
|
+
stem = re.sub(r"\s+", " ", title[:24]).lower()
|
|
718
|
+
if path.startswith("Annexe"):
|
|
719
|
+
needle = path.lower()
|
|
720
|
+
for i, line in enumerate(lines):
|
|
721
|
+
if line.strip().lower().startswith(needle):
|
|
722
|
+
return i
|
|
723
|
+
return -1
|
|
724
|
+
prefix = path + " "
|
|
725
|
+
for i, line in enumerate(lines):
|
|
726
|
+
s = line.strip()
|
|
727
|
+
if not s.startswith(prefix):
|
|
728
|
+
continue
|
|
729
|
+
rest = re.sub(r"\s+", " ", s[len(prefix):]).lower()
|
|
730
|
+
if not stem or rest.startswith(stem[: min(len(stem), 12)]):
|
|
731
|
+
return i
|
|
732
|
+
for i, line in enumerate(lines):
|
|
733
|
+
if stem and len(stem) >= 8 and re.sub(r"\s+", " ", line.strip()).lower().startswith(stem):
|
|
734
|
+
return i
|
|
735
|
+
return -1
|
|
736
|
+
|
|
737
|
+
|
|
738
|
+
def extract_section_text(
|
|
739
|
+
doc: "fitz.Document", nodes: list[dict], running: set[str]
|
|
740
|
+
) -> list[str]:
|
|
741
|
+
"""Give each node the text between its heading and the next one.
|
|
742
|
+
|
|
743
|
+
Pages are cleaned first (`clean_page`), then read as one stream. Every
|
|
744
|
+
node's heading is located on its start page; the node's text runs from the
|
|
745
|
+
line after its heading to the next located heading, in document order,
|
|
746
|
+
whatever the depth. A heading that cannot be located falls back to the
|
|
747
|
+
start of its page, which is the previous behaviour for that node only, and
|
|
748
|
+
is reported in the warnings so the skill can look at it.
|
|
749
|
+
|
|
750
|
+
Returns the warnings.
|
|
751
|
+
"""
|
|
752
|
+
warnings: list[str] = []
|
|
753
|
+
pages = [
|
|
754
|
+
clean_page(join_split_headings(page_text(doc, i)), running, doc.page_count)
|
|
755
|
+
for i in range(doc.page_count)
|
|
756
|
+
]
|
|
757
|
+
lines_per_page = [p.split("\n") for p in pages]
|
|
758
|
+
|
|
759
|
+
# Absolute line index of the first line of each page in the stream.
|
|
760
|
+
page_offsets: list[int] = []
|
|
761
|
+
total = 0
|
|
762
|
+
for lines in lines_per_page:
|
|
763
|
+
page_offsets.append(total)
|
|
764
|
+
total += len(lines)
|
|
765
|
+
stream = [line for lines in lines_per_page for line in lines]
|
|
766
|
+
|
|
767
|
+
located: list[tuple[int, int, dict]] = [] # (start_line, body_line, node)
|
|
768
|
+
for node in nodes:
|
|
769
|
+
pageno = node["pageStart"] - 1
|
|
770
|
+
at = locate_heading(lines_per_page[pageno], node)
|
|
771
|
+
if at == -1:
|
|
772
|
+
warnings.append(f"heading not located: {node['path']} {node['title'][:40]!r} (p. {node['pageStart']})")
|
|
773
|
+
start = page_offsets[pageno]
|
|
774
|
+
located.append((start, start, node))
|
|
775
|
+
else:
|
|
776
|
+
start = page_offsets[pageno] + at
|
|
777
|
+
located.append((start, start + 1, node))
|
|
778
|
+
|
|
779
|
+
# Document order, not path order: annexes sort last by path but sit last
|
|
780
|
+
# in the document anyway; a heading not located sorts at its page start.
|
|
781
|
+
located.sort(key=lambda t: (t[0], -t[2]["depth"]))
|
|
782
|
+
for idx, (start, body, node) in enumerate(located):
|
|
783
|
+
end = located[idx + 1][0] if idx + 1 < len(located) else len(stream)
|
|
784
|
+
text = "\n".join(stream[body:end])
|
|
785
|
+
node["rawText"] = tidy(dehyphenate(join_clause_numbers(text)))
|
|
786
|
+
node["textSource"] = "heading" if body > start else "page-start"
|
|
787
|
+
|
|
788
|
+
# A chapter whose text lives in its sections is normal; a leaf with
|
|
789
|
+
# nothing under its heading is a heading the text extraction lost.
|
|
790
|
+
parents = {n["parentNodeId"] for n in nodes if n.get("parentNodeId")}
|
|
791
|
+
empty = [n["path"] for n in nodes if len(n["rawText"]) < 40 and n["nodeId"] not in parents]
|
|
792
|
+
if empty:
|
|
793
|
+
warnings.append(f"{len(empty)} leaf node(s) with no text: {', '.join(empty[:8])}")
|
|
794
|
+
return warnings
|
|
795
|
+
|
|
796
|
+
|
|
797
|
+
def missing_chapter_numbers(chapters: dict[str, dict]) -> list[str]:
|
|
798
|
+
"""Holes in the chapter sequence, for the skill to look at the contents page."""
|
|
799
|
+
numbers = sorted(int(k) for k in chapters if k.isdigit())
|
|
800
|
+
if len(numbers) < 2:
|
|
801
|
+
return []
|
|
802
|
+
return [str(n) for n in range(numbers[0], numbers[-1]) if n not in set(numbers)]
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
def prune_subsections(
|
|
806
|
+
subsections: dict[str, tuple[str, int]], chapters: dict[str, dict]
|
|
807
|
+
) -> tuple[dict[str, tuple[str, int]], list[str]]:
|
|
808
|
+
"""Keep, under each parent, the children whose numbers advance with the pages.
|
|
809
|
+
|
|
810
|
+
The same idea as `prune_chapters`, one level down: a titled "4.2" found on
|
|
811
|
+
page 63 inside chapter 9 is a table row, and keeping it would hang a page
|
|
812
|
+
of chapter 9 under chapter 4. Children are checked against their parent's
|
|
813
|
+
first page and against each other's order.
|
|
814
|
+
"""
|
|
815
|
+
dropped: list[str] = []
|
|
816
|
+
by_parent: dict[str, list[str]] = defaultdict(list)
|
|
817
|
+
for path in subsections:
|
|
818
|
+
parent = path.rsplit(".", 1)[0]
|
|
819
|
+
by_parent[parent].append(path)
|
|
820
|
+
|
|
821
|
+
kept: dict[str, tuple[str, int]] = {}
|
|
822
|
+
for parent, children in by_parent.items():
|
|
823
|
+
parent_page = (
|
|
824
|
+
chapters.get(parent, {}).get("pageStart")
|
|
825
|
+
if parent in chapters
|
|
826
|
+
else (subsections[parent][1] if parent in subsections else None)
|
|
827
|
+
) or 0
|
|
828
|
+
ordered = [
|
|
829
|
+
p
|
|
830
|
+
for p in sorted(children, key=lambda p: [int(x) for x in p.split(".")])
|
|
831
|
+
if subsections[p][1] >= parent_page
|
|
832
|
+
]
|
|
833
|
+
for path in children:
|
|
834
|
+
if path not in ordered:
|
|
835
|
+
title, page = subsections[path]
|
|
836
|
+
dropped.append(f"{path} {title[:30]!r} (p. {page}, before its parent)")
|
|
837
|
+
|
|
838
|
+
# Longest run whose pages never go backwards, like `prune_chapters`:
|
|
839
|
+
# a greedy walk would keep the one stray row and drop every real
|
|
840
|
+
# child after it.
|
|
841
|
+
pages = [subsections[p][1] for p in ordered]
|
|
842
|
+
best = [1] * len(ordered)
|
|
843
|
+
prev = [-1] * len(ordered)
|
|
844
|
+
for i in range(len(ordered)):
|
|
845
|
+
for j in range(i):
|
|
846
|
+
if pages[j] <= pages[i] and best[j] + 1 > best[i]:
|
|
847
|
+
best[i], prev[i] = best[j] + 1, j
|
|
848
|
+
chain: set[int] = set()
|
|
849
|
+
if ordered:
|
|
850
|
+
# On a tie, the run that ends earliest in the document: a stray
|
|
851
|
+
# table row sits far past the real children.
|
|
852
|
+
top = max(best)
|
|
853
|
+
idx = min((i for i in range(len(ordered)) if best[i] == top), key=lambda i: pages[i])
|
|
854
|
+
while idx != -1:
|
|
855
|
+
chain.add(idx)
|
|
856
|
+
idx = prev[idx]
|
|
857
|
+
for i, path in enumerate(ordered):
|
|
858
|
+
title, page = subsections[path]
|
|
859
|
+
if i in chain:
|
|
860
|
+
kept[path] = (title, page)
|
|
861
|
+
else:
|
|
862
|
+
dropped.append(f"{path} {title[:30]!r} (p. {page}, out of sequence)")
|
|
863
|
+
return kept, dropped
|
|
451
864
|
|
|
452
865
|
|
|
453
866
|
def extract_figures(
|
|
@@ -464,7 +877,7 @@ def extract_figures(
|
|
|
464
877
|
for pageno in range(doc.page_count):
|
|
465
878
|
if pageno in toc_pages:
|
|
466
879
|
continue
|
|
467
|
-
text = doc
|
|
880
|
+
text = page_text(doc, pageno)
|
|
468
881
|
rendered = False
|
|
469
882
|
page_png: Path | None = None
|
|
470
883
|
for m in CAPTION_RE.finditer(text):
|
|
@@ -492,7 +905,7 @@ def extract_figures(
|
|
|
492
905
|
|
|
493
906
|
def guess_language(doc: "fitz.Document") -> str:
|
|
494
907
|
"""Best-effort language guess based on common French/German/Italian markers."""
|
|
495
|
-
sample = " ".join(doc
|
|
908
|
+
sample = " ".join(page_text(doc, i) for i in range(min(5, doc.page_count))).lower()
|
|
496
909
|
scores = {
|
|
497
910
|
"fr": sum(sample.count(w) for w in (" la ", " les ", " sont ", " avec ", " selon ")),
|
|
498
911
|
"de": sum(sample.count(w) for w in (" der ", " die ", " sind ", " mit ", " nach ")),
|
|
@@ -528,8 +941,22 @@ def main() -> None:
|
|
|
528
941
|
chap_prefixes = {k for k in chapters if k.isdigit()}
|
|
529
942
|
spans = chapter_page_spans(chapters, doc.page_count)
|
|
530
943
|
subsections = extract_subsections(doc, chap_prefixes, toc_pages, running, spans)
|
|
944
|
+
subsections, pruned = prune_subsections(subsections, chapters)
|
|
531
945
|
nodes = build_tree(chapters, subsections, doc.page_count)
|
|
532
|
-
|
|
946
|
+
warnings = [f"sub-section dropped, out of sequence: {p}" for p in pruned]
|
|
947
|
+
holes = missing_chapter_numbers(chapters)
|
|
948
|
+
if holes:
|
|
949
|
+
warnings.append(
|
|
950
|
+
f"chapters missing from the sequence: {', '.join(holes)} (check the contents page; an OCR'd copy often loses the chapter number)"
|
|
951
|
+
)
|
|
952
|
+
recovered = sorted(
|
|
953
|
+
(int(k) for k, meta in chapters.items() if k.isdigit() and meta.get("recovered"))
|
|
954
|
+
)
|
|
955
|
+
if recovered:
|
|
956
|
+
warnings.append(
|
|
957
|
+
f"chapter numbers recovered by position: {', '.join(map(str, recovered))} (their sub-headings are unnumbered in the text; check the contents page)"
|
|
958
|
+
)
|
|
959
|
+
warnings += extract_section_text(doc, nodes, running)
|
|
533
960
|
figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)
|
|
534
961
|
|
|
535
962
|
lang = args.language or guess_language(doc)
|
|
@@ -537,6 +964,13 @@ def main() -> None:
|
|
|
537
964
|
by_depth: dict[int, int] = defaultdict(int)
|
|
538
965
|
for n in nodes:
|
|
539
966
|
by_depth[n["depth"]] += 1
|
|
967
|
+
chars = sum(len(n["rawText"]) for n in nodes)
|
|
968
|
+
parents = {n["parentNodeId"] for n in nodes if n.get("parentNodeId")}
|
|
969
|
+
empty_nodes = sum(
|
|
970
|
+
1 for n in nodes if len(n["rawText"]) < 40 and n["nodeId"] not in parents
|
|
971
|
+
)
|
|
972
|
+
if not chapters:
|
|
973
|
+
warnings.append("no chapter detected: neither bookmarks nor headings")
|
|
540
974
|
|
|
541
975
|
manifest = {
|
|
542
976
|
"doc": {
|
|
@@ -552,7 +986,16 @@ def main() -> None:
|
|
|
552
986
|
"sectionCount": len(nodes),
|
|
553
987
|
"byDepth": dict(by_depth),
|
|
554
988
|
"figureCount": len(figures),
|
|
989
|
+
# What the server's coverage score will see (`corpusQuality.ts`),
|
|
990
|
+
# so the skill can judge the pre-pass before writing anything.
|
|
991
|
+
"chars": chars,
|
|
992
|
+
"charsPerPage": round(chars / max(1, doc.page_count)),
|
|
993
|
+
"emptyLeaves": empty_nodes,
|
|
555
994
|
},
|
|
995
|
+
# Anything the pre-pass could not decide on its own. The skill reads
|
|
996
|
+
# this list and looks at the pages it names; an empty list is the
|
|
997
|
+
# normal case on a native SIA norm.
|
|
998
|
+
"warnings": warnings,
|
|
556
999
|
"sections": nodes,
|
|
557
1000
|
"figures": figures,
|
|
558
1001
|
}
|
|
@@ -568,6 +1011,8 @@ def main() -> None:
|
|
|
568
1011
|
"sectionCount": len(nodes),
|
|
569
1012
|
"byDepth": dict(by_depth),
|
|
570
1013
|
"figureCount": len(figures),
|
|
1014
|
+
"charsPerPage": round(chars / max(1, doc.page_count)),
|
|
1015
|
+
"warnings": len(warnings),
|
|
571
1016
|
"language": lang,
|
|
572
1017
|
"pageCount": doc.page_count,
|
|
573
1018
|
},
|