simdref 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,116 @@
1
+ """Base PDF section extractor using pdfplumber character-level font metadata.
2
+
3
+ Detects section headings by font size and accumulates body text under each
4
+ heading. ISA-specific modules configure the size thresholds and heading
5
+ patterns.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ LineTuple = tuple[float, float, float, str]
11
+
12
+
13
+ def chars_to_lines(chars: list[dict]) -> list[LineTuple]:
14
+ """Group characters into lines by vertical position.
15
+
16
+ Returns a list of ``(top, size, x0, text)`` tuples sorted by vertical
17
+ position. Characters on the same line (within 2pt vertical tolerance)
18
+ are concatenated. ``x0`` is the left-edge position of the first
19
+ character, which encodes indentation level.
20
+ """
21
+ if not chars:
22
+ return []
23
+ lines: list[tuple[float, list[dict]]] = []
24
+ current_top = chars[0]["top"]
25
+ current_chars: list[dict] = [chars[0]]
26
+
27
+ for c in chars[1:]:
28
+ if abs(c["top"] - current_top) > 2.0:
29
+ lines.append((current_top, current_chars))
30
+ current_top = c["top"]
31
+ current_chars = [c]
32
+ else:
33
+ current_chars.append(c)
34
+
35
+ lines.append((current_top, current_chars))
36
+
37
+ result: list[LineTuple] = []
38
+ for top, line_chars in lines:
39
+ # Build text with gap-based space insertion. When consecutive
40
+ # characters have a large horizontal gap (>10pt) a space is
41
+ # inserted to handle two-column layouts (e.g. "#IS Stack
42
+ # underflow occurred." in Intel SDM exception tables).
43
+ parts: list[str] = []
44
+ prev_right = -1.0
45
+ for c in line_chars:
46
+ if prev_right >= 0 and c["x0"] - prev_right > 10.0:
47
+ if parts and not parts[-1].endswith(" "):
48
+ parts.append(" ")
49
+ parts.append(c["text"])
50
+ prev_right = c["x0"] + c.get("width", 0)
51
+ text = "".join(parts).strip()
52
+ max_size = max(c["size"] for c in line_chars)
53
+ x0 = min(c["x0"] for c in line_chars)
54
+ if text:
55
+ result.append((top, max_size, x0, text))
56
+ return result
57
+
58
+
59
+ def extract_sections_from_lines(
60
+ lines: list[LineTuple],
61
+ heading_min_size: float,
62
+ body_max_size: float,
63
+ known_headings: frozenset[str] | set[str] | None = None,
64
+ ) -> dict[str, list[tuple[float, str]]]:
65
+ """Extract named sections from pre-grouped PDF line tuples."""
66
+ sections: dict[str, list[tuple[float, str]]] = {}
67
+ current_heading: str | None = None
68
+ body_parts: list[tuple[float, str]] = []
69
+
70
+ for _top, size, line_x0, text in lines:
71
+ if known_headings is not None:
72
+ is_heading = text.lower().strip() in known_headings
73
+ else:
74
+ is_heading = size >= heading_min_size
75
+ if is_heading:
76
+ if current_heading is not None and body_parts:
77
+ sections[current_heading] = body_parts
78
+ current_heading = text
79
+ body_parts = []
80
+ elif size <= body_max_size and current_heading is not None:
81
+ body_parts.append((line_x0, text))
82
+
83
+ if current_heading is not None and body_parts:
84
+ sections[current_heading] = body_parts
85
+
86
+ return sections
87
+
88
+
89
+ def extract_sections_from_chars(
90
+ chars: list[dict],
91
+ heading_min_size: float,
92
+ body_max_size: float,
93
+ known_headings: frozenset[str] | set[str] | None = None,
94
+ ) -> dict[str, list[tuple[float, str]]]:
95
+ """Extract named sections from a list of pdfplumber character dicts.
96
+
97
+ Characters with font size >= *heading_min_size* are treated as section
98
+ headings. Characters with font size <= *body_max_size* are accumulated
99
+ as body text under the current heading.
100
+
101
+ If *known_headings* is provided, heading detection switches to
102
+ **whitelist mode**: a line is a heading ONLY if its lowercased text
103
+ matches the known headings set. Lines at heading font size that do
104
+ NOT match are demoted to body text under the current heading. When
105
+ *known_headings* is ``None``, the original font-size-only heuristic
106
+ is used (backward compatibility).
107
+
108
+ Returns a dict mapping heading text to a list of ``(x0, text)`` tuples
109
+ preserving the left-edge position for indentation reconstruction.
110
+ """
111
+ return extract_sections_from_lines(
112
+ chars_to_lines(chars),
113
+ heading_min_size=heading_min_size,
114
+ body_max_size=body_max_size,
115
+ known_headings=known_headings,
116
+ )