@stratta/mcp 0.10.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,10 +2,13 @@
2
2
  """Stratta — ingest pre-pass.
3
3
 
4
4
  Reads a norm PDF with PyMuPDF, builds the hierarchical TreeRAG skeleton (chapters
5
- from PDF bookmarks + sub-sections via heading regex), extracts per-section raw
6
- text, and rasterizes each page that contains a figure caption. Output is a
7
- single JSON consumed by the `ingest-norm` skill, which then enriches sections
8
- (formulas, tables, cross-refs, summaries) and uploads the figures.
5
+ from PDF bookmarks + sub-sections via heading regex, down to X.Y.Z.W), gives
6
+ each node the text between its heading and the next one on cleaned pages (no
7
+ running headers, no page numbers, hyphenation resolved, clause numbers joined
8
+ to their paragraph), and rasterizes each page that contains a figure caption.
9
+ Output is a single JSON consumed by the `ingest-norm` skill, which then
10
+ enriches sections (formulas, tables, cross-refs, summaries) and uploads the
11
+ figures. `warnings[]` lists what the pre-pass could not decide on its own.
9
12
 
10
13
  Usage:
11
14
  python scripts/ingest-prepass.py --pdf <path> --output <dir> [--language fr]
@@ -54,23 +57,137 @@ def clean(s: str) -> str:
54
57
  return re.sub(r"\s+", " ", CLEAN_DOTS.sub("", s)).strip()
55
58
 
56
59
 
60
+ # --- page lines ----------------------------------------------------------------
61
+ #
62
+ # PyMuPDF's plain text follows the PDF's own block order. On a SIA norm the
63
+ # clause numbers sit in a column of their own, and on an OCR'd copy (the form
64
+ # a bureau actually holds: 60 of the 80 PDFs of the first client corpus are
65
+ # scans) that column is read as one block BEFORE the paragraphs, so a page
66
+ # arrives as "4.4 / 4.4.1 / 4.4.1.1 / ... / Résistances du terrain /
67
+ # Généralités / Sont considérés ..." and no heading regex can see a heading.
68
+ #
69
+ # Lines are therefore rebuilt from the words and their positions: words on the
70
+ # same baseline form a line, left to right. That puts "4.4 Résistances du
71
+ # terrain" back together on both the native and the OCR'd copy. A genuine
72
+ # two-column page (two text columns, not a numbering column) would be
73
+ # scrambled by that rule, so such pages are detected and read in block order.
74
+
75
+ _LINES_CACHE: dict[tuple[int, int], list[str]] = {}
76
+ NUMBER_TOKEN_RE = re.compile(r"^(?:\d{1,2}(?:\.\d{1,3}){0,4}\.?|[A-Z](?:\.\d{1,2}){1,3}|—|-|•)$")
77
+ OCR_DOTTED_NUMBER_RE = re.compile(r"^(\d+(?:\.\d+)*) \.(\d)")
78
+
79
+
80
+ def _rows(page: "fitz.Page") -> list[list[tuple]]:
81
+ words = page.get_text("words") # x0, y0, x1, y1, text, block, line, word
82
+ if not words:
83
+ return []
84
+ heights = sorted(w[3] - w[1] for w in words)
85
+ median = heights[len(heights) // 2]
86
+ # A word set vertically (a library watermark such as "Ecole Polytechnique
87
+ # Fédérale de Lausanne" running up the margin) has a box far taller than
88
+ # the text. Left in, its words land on every line they cross.
89
+ words = [w for w in words if (w[3] - w[1]) <= 2.5 * median]
90
+ if not words:
91
+ return []
92
+ tol = max(2.0, 0.45 * median)
93
+ rows: list[list[tuple]] = []
94
+ for w in sorted(words, key=lambda w: (w[1], w[0])):
95
+ if rows and abs(rows[-1][0][1] - w[1]) <= tol:
96
+ rows[-1].append(w)
97
+ else:
98
+ rows.append([w])
99
+ for row in rows:
100
+ row.sort(key=lambda w: w[0])
101
+ return rows
102
+
103
+
104
+ def _segments(row: list[tuple], gap: float) -> list[list[tuple]]:
105
+ """Split a row where the horizontal gap between words is a column gutter."""
106
+ out: list[list[tuple]] = [[row[0]]]
107
+ for prev, w in zip(row, row[1:]):
108
+ if w[0] - prev[2] > gap:
109
+ out.append([w])
110
+ else:
111
+ out[-1].append(w)
112
+ return out
113
+
114
+
115
+ def _two_column(rows: list[list[list[tuple]]]) -> bool:
116
+ """Two text columns: many rows carry two segments that both read as prose."""
117
+ if len(rows) < 12:
118
+ return False
119
+ both = 0
120
+ for segs in rows:
121
+ texts = [" ".join(w[4] for w in s) for s in segs]
122
+ prose = [t for t in texts if len(t) > 24 and not NUMBER_TOKEN_RE.match(t.split()[0])]
123
+ if len(prose) >= 2:
124
+ both += 1
125
+ return both / len(rows) > 0.3
126
+
127
+
128
+ def page_lines(doc: "fitz.Document", pageno: int) -> list[str]:
129
+ """The lines of a page, in reading order, numbers joined to their text."""
130
+ key = (id(doc), pageno)
131
+ cached = _LINES_CACHE.get(key)
132
+ if cached is not None:
133
+ return cached
134
+ page = doc[pageno]
135
+ rows = _rows(page)
136
+ gap = 0.12 * page.rect.width
137
+ segmented = [_segments(row, gap) for row in rows]
138
+ if _two_column(segmented):
139
+ lines = join_split_headings(page.get_text("text")).splitlines()
140
+ else:
141
+ lines = []
142
+ for segs in segmented:
143
+ text = " ".join(" ".join(w[4] for w in s) for s in segs)
144
+ text = OCR_DOTTED_NUMBER_RE.sub(r"\1.\2", text)
145
+ lines.append(text)
146
+ lines = [l.rstrip() for l in lines]
147
+ _LINES_CACHE[key] = lines
148
+ return lines
149
+
150
+
151
+ def page_text(doc: "fitz.Document", pageno: int) -> str:
152
+ return "\n".join(page_lines(doc, pageno))
153
+
154
+
57
155
  def detect_toc_pages(doc: "fitz.Document") -> set[int]:
58
156
  out = set()
59
157
  for i in range(doc.page_count):
60
- lines = [l for l in doc[i].get_text("text").splitlines() if l.strip()]
158
+ lines = [l for l in page_lines(doc, i) if l.strip()]
61
159
  leader = sum(1 for l in lines if TOC_DOT_RE.search(l))
62
160
  if leader >= 5:
63
161
  out.add(i)
64
162
  return out
65
163
 
66
164
 
165
+ # Standalone integers only: the "9" and "1" of a heading "9.1 Délimitation"
166
+ # stay, or every "Généralités" heading would read as one running line.
167
+ FURNITURE_NUMBER_RE = re.compile(r"(?<![\d.])\d{1,4}(?![\d.])")
168
+ FURNITURE_EDGE_RE = re.compile(r"^[\s/|\-–—·.]+|[\s/|\-–—·.]+$")
169
+
170
+
171
+ def furniture_key(line: str) -> str:
172
+ """A running line without the page number it carries.
173
+
174
+ `SIA 267, Copyright © 2013 by SIA Zurich 59` on an odd page and
175
+ `64 SIA 267, Copyright © 2013 by SIA Zurich` on an even one are the same
176
+ footer; so are the copies a watermark decorates with a stray slash.
177
+ """
178
+ s = FURNITURE_NUMBER_RE.sub(" ", line)
179
+ s = re.sub(r"\s+", " ", s).strip()
180
+ return FURNITURE_EDGE_RE.sub("", s).strip()
181
+
182
+
67
183
  def detect_running_text(doc: "fitz.Document") -> set[str]:
68
- """Lines appearing on >=5 pages near the top/bottom (running headers/footers)."""
184
+ """Lines appearing on >=5 pages near the top/bottom (running headers/footers),
185
+ keyed without their page number."""
69
186
  counter: dict[str, int] = defaultdict(int)
70
187
  for i in range(doc.page_count):
71
- lines = [l.strip() for l in doc[i].get_text("text").splitlines() if l.strip()]
188
+ lines = [l.strip() for l in page_lines(doc, i) if l.strip()]
72
189
  for l in lines[:3] + lines[-3:]:
73
- counter[l] += 1
190
+ counter[furniture_key(l)] += 1
74
191
  return {l for l, c in counter.items() if c >= 5 and len(l) > 8}
75
192
 
76
193
 
@@ -124,8 +241,8 @@ def join_split_headings(text: str) -> str:
124
241
 
125
242
 
126
243
  def heading_text(doc: "fitz.Document", pageno: int) -> str:
127
- """Page text prepared for heading detection (never for section content)."""
128
- return join_split_headings(doc[pageno].get_text("text"))
244
+ """Page text prepared for heading detection."""
245
+ return join_split_headings(page_text(doc, pageno))
129
246
 
130
247
 
131
248
  NOT_A_TITLE_RE = re.compile(
@@ -212,12 +329,14 @@ def prune_chapters(chapters: dict[str, dict]) -> dict[str, dict]:
212
329
  idx = prev[idx]
213
330
 
214
331
  # Missing a heading or two leaves a small gap; doubling (10 → 20 → 30) means
215
- # the numbers stopped being chapters and started being table rows.
332
+ # the numbers stopped being chapters and started being table rows. Only
333
+ # from 5 upwards: an OCR'd copy that lost chapters 3 to 6 jumps from 2 to
334
+ # 7, and that is a hole, not a doubling.
216
335
  kept: dict[str, dict] = {}
217
336
  previous = None
218
337
  for key in reversed(chain):
219
338
  n = int(key)
220
- if previous is not None and n - previous > 3 and n >= 2 * previous:
339
+ if previous is not None and previous >= 5 and n - previous > 3 and n >= 2 * previous:
221
340
  break
222
341
  kept[key] = chapters[key]
223
342
  previous = n
@@ -301,7 +420,106 @@ def extract_chapters(
301
420
  if len([k for k in chapters if k.isdigit()]) >= 3:
302
421
  break
303
422
 
304
- return prune_chapters(chapters)
423
+ for num, meta in infer_unnumbered_chapters(doc, skip).items():
424
+ chapters.setdefault(num, meta)
425
+ chapters = prune_chapters(chapters)
426
+ return recover_missing_chapters(doc, chapters, skip)
427
+
428
+
429
+ SCOPE_TITLE_RE = re.compile(
430
+ r"^(DOMAINE D.APPLICATION|GELTUNGSBEREICH|CAMPO D.APPLICAZIONE|SCOPE)\b", re.I
431
+ )
432
+
433
+
434
+ def _page_top_titles(doc: "fitz.Document", pages: range, skip: set[int]) -> list[tuple[int, str]]:
435
+ """Uppercase, plausible titles among the first lines of each page: where a
436
+ chapter of a SIA norm starts. Deeper in a page a capitalised line is a
437
+ table header or a note."""
438
+ out: list[tuple[int, str]] = []
439
+ for pageno in pages:
440
+ if pageno in skip:
441
+ continue
442
+ for line in [l.strip() for l in page_lines(doc, pageno)][:3]:
443
+ m = UNNUMBERED_UPPER_RE.match(line)
444
+ if m and plausible_title(m.group(1)) and not re.match(r"^(TABLEAU|TABELLE|TABELLA|TABLE|FIGURE|FIG\.|BILD|ANNEXE|ANHANG|ANNEX)\b", line):
445
+ out.append((pageno + 1, clean(m.group(1))))
446
+ break
447
+ return out
448
+
449
+
450
+ def recover_missing_chapters(doc: "fitz.Document", chapters: dict[str, dict], skip: set[int]) -> dict[str, dict]:
451
+ """Fill holes in the chapter sequence by position.
452
+
453
+ On the OCR'd copies the numbering column is sometimes lost on a whole run
454
+ of pages: the chapter title survives, and so does the first sub-heading,
455
+ but neither carries its number, so `infer_unnumbered_chapters` has nothing
456
+ to read. When the sequence goes 1, 2, 6 and exactly three title-only pages
457
+ lie between chapter 2 and chapter 6, those are chapters 3, 4 and 5, in
458
+ page order. Anything less certain is left as a hole for the agent.
459
+
460
+ A leading "Domaine d'application" before chapter 1 is chapter 0, which is
461
+ how SIA numbers it.
462
+ """
463
+ numbered = sorted((int(k), k) for k in chapters if k.isdigit())
464
+ if not numbered:
465
+ return chapters
466
+ used = {meta["title"].upper() for meta in chapters.values()}
467
+ recovered: dict[str, dict] = {}
468
+
469
+ for (a, ka), (b, kb) in zip(numbered, numbered[1:]):
470
+ holes = list(range(a + 1, b))
471
+ if not holes:
472
+ continue
473
+ start, end = chapters[ka]["pageStart"], chapters[kb]["pageStart"]
474
+ candidates = [
475
+ (page, title)
476
+ for page, title in _page_top_titles(doc, range(start, end - 1), skip)
477
+ if title.upper() not in used
478
+ ]
479
+ if len(candidates) != len(holes):
480
+ continue
481
+ for n, (page, title) in zip(holes, candidates):
482
+ recovered[str(n)] = {"title": title, "pageStart": page, "recovered": True}
483
+
484
+ first_n, first_k = numbered[0]
485
+ if first_n == 1 and "0" not in chapters:
486
+ before = _page_top_titles(doc, range(0, chapters[first_k]["pageStart"] - 1), skip)
487
+ scope = [(p, t) for p, t in before if SCOPE_TITLE_RE.match(t)]
488
+ if len(scope) == 1:
489
+ recovered["0"] = {"title": scope[0][1], "pageStart": scope[0][0], "recovered": True}
490
+
491
+ return {**chapters, **recovered}
492
+
493
+
494
+ UNNUMBERED_UPPER_RE = re.compile(r"^([A-ZÉÈÀÂÔÎÛÇ][^a-z\n]{5,120})$")
495
+ FIRST_SUB_RE = re.compile(r"^(\d{1,2})\.\d{1,2}\b")
496
+
497
+
498
+ def infer_unnumbered_chapters(doc: "fitz.Document", skip: set[int]) -> dict[str, dict]:
499
+ """Chapters whose number the OCR lost.
500
+
501
+ On a scanned copy the large chapter number is often the one glyph the OCR
502
+ does not recognise, so the page reads "FONDATIONS SUR PIEUX" followed by
503
+ "9.1 Délimitation". The number is then taken from the first sub-heading
504
+ under the title. Only fills holes: a chapter found with its number wins.
505
+ """
506
+ found: dict[str, dict] = {}
507
+ for pageno in range(doc.page_count):
508
+ if pageno in skip:
509
+ continue
510
+ lines = [l.strip() for l in page_lines(doc, pageno)]
511
+ for i, line in enumerate(lines):
512
+ m = UNNUMBERED_UPPER_RE.match(line)
513
+ if not m or not plausible_title(m.group(1)):
514
+ continue
515
+ for follow in lines[i + 1 : i + 9]:
516
+ sub = FIRST_SUB_RE.match(follow)
517
+ if sub:
518
+ num = sub.group(1)
519
+ if num not in found:
520
+ found[num] = {"title": clean(m.group(1)), "pageStart": pageno + 1}
521
+ break
522
+ return found
305
523
 
306
524
 
307
525
  def chapter_page_spans(chapters: dict[str, dict], n_pages: int) -> dict[str, tuple[int, int]]:
@@ -340,6 +558,11 @@ def extract_subsections(
340
558
  top = num.split(".", 1)[0]
341
559
  if top not in chap_prefixes:
342
560
  continue
561
+ # "9.522" is "9.5.2.2" with a dot the OCR dropped. No norm has a
562
+ # fortieth sub-section; keeping it would hang a page of 9.5 under
563
+ # a sibling of 9.7 and knock the real 9.6 and 9.7 out of sequence.
564
+ if any(int(part) > 40 for part in num.split(".")[1:]):
565
+ continue
343
566
  # A sub-section sits inside its chapter. "1.1" found sixty pages
344
567
  # after chapter 1 ended is a numbered line in an annexe, and keeping
345
568
  # it hangs an unrelated page under the wrong parent.
@@ -403,51 +626,241 @@ def slugify(path: str) -> str:
403
626
  return re.sub(r"[^a-zA-Z0-9]+", "-", path).strip("-").lower() or "root"
404
627
 
405
628
 
406
- def extract_section_text(doc: "fitz.Document", nodes: list[dict]) -> None:
407
- """Attach rawText to each node = concat of pages it spans."""
408
- page_texts = [doc[i].get_text("text") for i in range(doc.page_count)]
409
- for node in nodes:
410
- ps, pe = node["pageStart"], node["pageEnd"]
411
- chunks = page_texts[ps - 1 : pe]
412
- node["rawText"] = "\n".join(chunks).strip()
629
+ # --- section text ------------------------------------------------------------
630
+ #
631
+ # Until 2026-09-04 a node's text was the concatenation of the pages it spanned,
632
+ # which meant three things the audit measured on the production corpus
633
+ # (ADR 22 § 2.1): a chapter carried the full text of all its sub-sections
634
+ # (61 000 characters twice for SIA 267.153 § 11 and § 11.3), the running header
635
+ # and the page number of every page sat in the middle of the prose, and words
636
+ # hyphenated at a line end stayed broken ("ter- rain"). The text of a node is
637
+ # now what lies between its heading and the next heading, on cleaned pages.
413
638
 
414
- trim_shared_pages(nodes, page_texts)
639
+ PAGE_NUMBER_RE = re.compile(r"^\s*\d{1,4}\s*$")
640
+ CLAUSE_NUMBER_RE = re.compile(r"^\s*(\d{1,2}(?:\.\d{1,3}){1,4})\s*$")
641
+ HYPHEN_BREAK_RE = re.compile(r"([a-zà-ÿ])[-—]\n([a-zà-ÿ])")
415
642
 
416
643
 
417
- def trim_shared_pages(nodes: list[dict], page_texts: list[str]) -> None:
418
- """Cut the opening page where several sections start on it.
644
+ def dehyphenate(text: str) -> str:
645
+ """Join a word broken at a line end: `tra-\\nvaux` becomes `travaux`.
419
646
 
420
- Page granularity is the unit everywhere else, but two sections beginning on
421
- the same page would otherwise carry byte-identical text — so a summary reads
422
- like its neighbour's, and the table of contents misleads before anything is
423
- even opened.
647
+ A compound word split at its hyphen loses it too (`pieux-\\nradier` reads
648
+ `pieuxradier`); syllable breaks outnumber compounds at line ends by far,
649
+ and a broken word is worse for search than a missing hyphen.
424
650
  """
425
- by_page: dict[int, list[dict]] = defaultdict(list)
426
- for node in nodes:
427
- by_page[node["pageStart"]].append(node)
651
+ return HYPHEN_BREAK_RE.sub(r"\1\2", text)
652
+
653
+
654
+ def clean_page(text: str, running: set[str], page_count: int) -> str:
655
+ """Drop what is page furniture rather than norm text.
428
656
 
429
- for page, group in by_page.items():
430
- if len(group) < 2:
657
+ - the running header and footer lines `detect_running_text` found (they
658
+ were detected before, and never removed);
659
+ - a bare page number in the three first or three last lines of the page;
660
+ - trailing whitespace on every line.
661
+ """
662
+ lines = [l.rstrip() for l in text.splitlines()]
663
+ kept: list[str] = []
664
+ n = len(lines)
665
+ for i, line in enumerate(lines):
666
+ s = line.strip()
667
+ if s and (s in running or furniture_key(s) in running):
431
668
  continue
432
- text = page_texts[page - 1]
433
- # Where each section's heading sits on that page, in reading order.
434
- marks: list[tuple[int, dict]] = []
435
- for node in group:
436
- title = node["title"].split(" — ")[-1].strip()
437
- at = text.find(title) if len(title) >= 4 else -1
438
- if at == -1:
439
- at = text.find(f"{node['path']} ")
440
- marks.append((at, node))
441
- if any(at == -1 for at, _ in marks):
442
- continue # a heading we cannot locate: leave the whole page in place
443
- marks.sort(key=lambda m: m[0])
444
- for i, (at, node) in enumerate(marks):
445
- end = marks[i + 1][0] if i + 1 < len(marks) else len(text)
446
- head = text[at:end].strip()
447
- if not head:
448
- continue
449
- rest = "\n".join(page_texts[node["pageStart"] : node["pageEnd"]]).strip()
450
- node["rawText"] = f"{head}\n{rest}".strip() if rest else head
669
+ # Printed page numbers run a little past the PDF's page count when
670
+ # the front matter is numbered separately; well past it, a bare
671
+ # number in the margin is a table value.
672
+ near_edge = i < 3 or i >= n - 3
673
+ if near_edge and PAGE_NUMBER_RE.match(s) and int(s) <= page_count + 20:
674
+ continue
675
+ kept.append(line)
676
+ return "\n".join(kept)
677
+
678
+
679
+ def join_clause_numbers(text: str) -> str:
680
+ """Rewrite `9.5.1.1\\nLes pieux…` as `9.5.1.1 Les pieux…`.
681
+
682
+ PyMuPDF reads the numbering column before the paragraph, so every numbered
683
+ clause of a SIA norm arrives as a number on its own line. Joined, the
684
+ number becomes an anchor the reader can search for; alone, it is noise.
685
+ Content only: heading detection keeps its stricter rule
686
+ (`join_split_headings`), because a clause number followed by a sentence is
687
+ exactly what must NOT become a heading.
688
+ """
689
+ lines = text.splitlines()
690
+ out: list[str] = []
691
+ i = 0
692
+ while i < len(lines):
693
+ m = CLAUSE_NUMBER_RE.match(lines[i])
694
+ if m and i + 1 < len(lines) and lines[i + 1].strip():
695
+ out.append(f"{m.group(1)} {lines[i + 1].strip()}")
696
+ i += 2
697
+ continue
698
+ out.append(lines[i])
699
+ i += 1
700
+ return "\n".join(out)
701
+
702
+
703
+ def tidy(text: str) -> str:
704
+ """Collapse runs of blank lines; strip the ends."""
705
+ return re.sub(r"\n{3,}", "\n\n", text).strip()
706
+
707
+
708
+ def locate_heading(lines: list[str], node: dict) -> int:
709
+ """Index of the line carrying this node's heading on its start page, or -1.
710
+
711
+ Matches on the number first (`9.5 ` at the start of a line), then on the
712
+ first words of the title, so a title PyMuPDF wrapped differently from the
713
+ contents page is still found.
714
+ """
715
+ path = node["path"]
716
+ title = node["title"].split(" — ")[-1].strip()
717
+ stem = re.sub(r"\s+", " ", title[:24]).lower()
718
+ if path.startswith("Annexe"):
719
+ needle = path.lower()
720
+ for i, line in enumerate(lines):
721
+ if line.strip().lower().startswith(needle):
722
+ return i
723
+ return -1
724
+ prefix = path + " "
725
+ for i, line in enumerate(lines):
726
+ s = line.strip()
727
+ if not s.startswith(prefix):
728
+ continue
729
+ rest = re.sub(r"\s+", " ", s[len(prefix):]).lower()
730
+ if not stem or rest.startswith(stem[: min(len(stem), 12)]):
731
+ return i
732
+ for i, line in enumerate(lines):
733
+ if stem and len(stem) >= 8 and re.sub(r"\s+", " ", line.strip()).lower().startswith(stem):
734
+ return i
735
+ return -1
736
+
737
+
738
+ def extract_section_text(
739
+ doc: "fitz.Document", nodes: list[dict], running: set[str]
740
+ ) -> list[str]:
741
+ """Give each node the text between its heading and the next one.
742
+
743
+ Pages are cleaned first (`clean_page`), then read as one stream. Every
744
+ node's heading is located on its start page; the node's text runs from the
745
+ line after its heading to the next located heading, in document order,
746
+ whatever the depth. A heading that cannot be located falls back to the
747
+ start of its page, which is the previous behaviour for that node only, and
748
+ is reported in the warnings so the skill can look at it.
749
+
750
+ Returns the warnings.
751
+ """
752
+ warnings: list[str] = []
753
+ pages = [
754
+ clean_page(join_split_headings(page_text(doc, i)), running, doc.page_count)
755
+ for i in range(doc.page_count)
756
+ ]
757
+ lines_per_page = [p.split("\n") for p in pages]
758
+
759
+ # Absolute line index of the first line of each page in the stream.
760
+ page_offsets: list[int] = []
761
+ total = 0
762
+ for lines in lines_per_page:
763
+ page_offsets.append(total)
764
+ total += len(lines)
765
+ stream = [line for lines in lines_per_page for line in lines]
766
+
767
+ located: list[tuple[int, int, dict]] = [] # (start_line, body_line, node)
768
+ for node in nodes:
769
+ pageno = node["pageStart"] - 1
770
+ at = locate_heading(lines_per_page[pageno], node)
771
+ if at == -1:
772
+ warnings.append(f"heading not located: {node['path']} {node['title'][:40]!r} (p. {node['pageStart']})")
773
+ start = page_offsets[pageno]
774
+ located.append((start, start, node))
775
+ else:
776
+ start = page_offsets[pageno] + at
777
+ located.append((start, start + 1, node))
778
+
779
+ # Document order, not path order: annexes sort last by path but sit last
780
+ # in the document anyway; a heading not located sorts at its page start.
781
+ located.sort(key=lambda t: (t[0], -t[2]["depth"]))
782
+ for idx, (start, body, node) in enumerate(located):
783
+ end = located[idx + 1][0] if idx + 1 < len(located) else len(stream)
784
+ text = "\n".join(stream[body:end])
785
+ node["rawText"] = tidy(dehyphenate(join_clause_numbers(text)))
786
+ node["textSource"] = "heading" if body > start else "page-start"
787
+
788
+ # A chapter whose text lives in its sections is normal; a leaf with
789
+ # nothing under its heading is a heading the text extraction lost.
790
+ parents = {n["parentNodeId"] for n in nodes if n.get("parentNodeId")}
791
+ empty = [n["path"] for n in nodes if len(n["rawText"]) < 40 and n["nodeId"] not in parents]
792
+ if empty:
793
+ warnings.append(f"{len(empty)} leaf node(s) with no text: {', '.join(empty[:8])}")
794
+ return warnings
795
+
796
+
797
+ def missing_chapter_numbers(chapters: dict[str, dict]) -> list[str]:
798
+ """Holes in the chapter sequence, for the skill to look at the contents page."""
799
+ numbers = sorted(int(k) for k in chapters if k.isdigit())
800
+ if len(numbers) < 2:
801
+ return []
802
+ return [str(n) for n in range(numbers[0], numbers[-1]) if n not in set(numbers)]
803
+
804
+
805
+ def prune_subsections(
806
+ subsections: dict[str, tuple[str, int]], chapters: dict[str, dict]
807
+ ) -> tuple[dict[str, tuple[str, int]], list[str]]:
808
+ """Keep, under each parent, the children whose numbers advance with the pages.
809
+
810
+ The same idea as `prune_chapters`, one level down: a titled "4.2" found on
811
+ page 63 inside chapter 9 is a table row, and keeping it would hang a page
812
+ of chapter 9 under chapter 4. Children are checked against their parent's
813
+ first page and against each other's order.
814
+ """
815
+ dropped: list[str] = []
816
+ by_parent: dict[str, list[str]] = defaultdict(list)
817
+ for path in subsections:
818
+ parent = path.rsplit(".", 1)[0]
819
+ by_parent[parent].append(path)
820
+
821
+ kept: dict[str, tuple[str, int]] = {}
822
+ for parent, children in by_parent.items():
823
+ parent_page = (
824
+ chapters.get(parent, {}).get("pageStart")
825
+ if parent in chapters
826
+ else (subsections[parent][1] if parent in subsections else None)
827
+ ) or 0
828
+ ordered = [
829
+ p
830
+ for p in sorted(children, key=lambda p: [int(x) for x in p.split(".")])
831
+ if subsections[p][1] >= parent_page
832
+ ]
833
+ for path in children:
834
+ if path not in ordered:
835
+ title, page = subsections[path]
836
+ dropped.append(f"{path} {title[:30]!r} (p. {page}, before its parent)")
837
+
838
+ # Longest run whose pages never go backwards, like `prune_chapters`:
839
+ # a greedy walk would keep the one stray row and drop every real
840
+ # child after it.
841
+ pages = [subsections[p][1] for p in ordered]
842
+ best = [1] * len(ordered)
843
+ prev = [-1] * len(ordered)
844
+ for i in range(len(ordered)):
845
+ for j in range(i):
846
+ if pages[j] <= pages[i] and best[j] + 1 > best[i]:
847
+ best[i], prev[i] = best[j] + 1, j
848
+ chain: set[int] = set()
849
+ if ordered:
850
+ # On a tie, the run that ends earliest in the document: a stray
851
+ # table row sits far past the real children.
852
+ top = max(best)
853
+ idx = min((i for i in range(len(ordered)) if best[i] == top), key=lambda i: pages[i])
854
+ while idx != -1:
855
+ chain.add(idx)
856
+ idx = prev[idx]
857
+ for i, path in enumerate(ordered):
858
+ title, page = subsections[path]
859
+ if i in chain:
860
+ kept[path] = (title, page)
861
+ else:
862
+ dropped.append(f"{path} {title[:30]!r} (p. {page}, out of sequence)")
863
+ return kept, dropped
451
864
 
452
865
 
453
866
  def extract_figures(
@@ -464,7 +877,7 @@ def extract_figures(
464
877
  for pageno in range(doc.page_count):
465
878
  if pageno in toc_pages:
466
879
  continue
467
- text = doc[pageno].get_text("text")
880
+ text = page_text(doc, pageno)
468
881
  rendered = False
469
882
  page_png: Path | None = None
470
883
  for m in CAPTION_RE.finditer(text):
@@ -492,7 +905,7 @@ def extract_figures(
492
905
 
493
906
  def guess_language(doc: "fitz.Document") -> str:
494
907
  """Best-effort language guess based on common French/German/Italian markers."""
495
- sample = " ".join(doc[i].get_text("text") for i in range(min(5, doc.page_count))).lower()
908
+ sample = " ".join(page_text(doc, i) for i in range(min(5, doc.page_count))).lower()
496
909
  scores = {
497
910
  "fr": sum(sample.count(w) for w in (" la ", " les ", " sont ", " avec ", " selon ")),
498
911
  "de": sum(sample.count(w) for w in (" der ", " die ", " sind ", " mit ", " nach ")),
@@ -528,8 +941,22 @@ def main() -> None:
528
941
  chap_prefixes = {k for k in chapters if k.isdigit()}
529
942
  spans = chapter_page_spans(chapters, doc.page_count)
530
943
  subsections = extract_subsections(doc, chap_prefixes, toc_pages, running, spans)
944
+ subsections, pruned = prune_subsections(subsections, chapters)
531
945
  nodes = build_tree(chapters, subsections, doc.page_count)
532
- extract_section_text(doc, nodes)
946
+ warnings = [f"sub-section dropped, out of sequence: {p}" for p in pruned]
947
+ holes = missing_chapter_numbers(chapters)
948
+ if holes:
949
+ warnings.append(
950
+ f"chapters missing from the sequence: {', '.join(holes)} (check the contents page; an OCR'd copy often loses the chapter number)"
951
+ )
952
+ recovered = sorted(
953
+ (int(k) for k, meta in chapters.items() if k.isdigit() and meta.get("recovered"))
954
+ )
955
+ if recovered:
956
+ warnings.append(
957
+ f"chapter numbers recovered by position: {', '.join(map(str, recovered))} (their sub-headings are unnumbered in the text; check the contents page)"
958
+ )
959
+ warnings += extract_section_text(doc, nodes, running)
533
960
  figures = extract_figures(doc, toc_pages, out_dir, dpi=args.dpi)
534
961
 
535
962
  lang = args.language or guess_language(doc)
@@ -537,6 +964,13 @@ def main() -> None:
537
964
  by_depth: dict[int, int] = defaultdict(int)
538
965
  for n in nodes:
539
966
  by_depth[n["depth"]] += 1
967
+ chars = sum(len(n["rawText"]) for n in nodes)
968
+ parents = {n["parentNodeId"] for n in nodes if n.get("parentNodeId")}
969
+ empty_nodes = sum(
970
+ 1 for n in nodes if len(n["rawText"]) < 40 and n["nodeId"] not in parents
971
+ )
972
+ if not chapters:
973
+ warnings.append("no chapter detected: neither bookmarks nor headings")
540
974
 
541
975
  manifest = {
542
976
  "doc": {
@@ -552,7 +986,16 @@ def main() -> None:
552
986
  "sectionCount": len(nodes),
553
987
  "byDepth": dict(by_depth),
554
988
  "figureCount": len(figures),
989
+ # What the server's coverage score will see (`corpusQuality.ts`),
990
+ # so the skill can judge the pre-pass before writing anything.
991
+ "chars": chars,
992
+ "charsPerPage": round(chars / max(1, doc.page_count)),
993
+ "emptyLeaves": empty_nodes,
555
994
  },
995
+ # Anything the pre-pass could not decide on its own. The skill reads
996
+ # this list and looks at the pages it names; an empty list is the
997
+ # normal case on a native SIA norm.
998
+ "warnings": warnings,
556
999
  "sections": nodes,
557
1000
  "figures": figures,
558
1001
  }
@@ -568,6 +1011,8 @@ def main() -> None:
568
1011
  "sectionCount": len(nodes),
569
1012
  "byDepth": dict(by_depth),
570
1013
  "figureCount": len(figures),
1014
+ "charsPerPage": round(chars / max(1, doc.page_count)),
1015
+ "warnings": len(warnings),
571
1016
  "language": lang,
572
1017
  "pageCount": doc.page_count,
573
1018
  },