simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
simdref/pdfparse/base.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Base PDF section extractor using pdfplumber character-level font metadata.
|
|
2
|
+
|
|
3
|
+
Detects section headings by font size and accumulates body text under each
|
|
4
|
+
heading. ISA-specific modules configure the size thresholds and heading
|
|
5
|
+
patterns.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
LineTuple = tuple[float, float, float, str]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def chars_to_lines(chars: list[dict]) -> list[LineTuple]:
|
|
14
|
+
"""Group characters into lines by vertical position.
|
|
15
|
+
|
|
16
|
+
Returns a list of ``(top, size, x0, text)`` tuples sorted by vertical
|
|
17
|
+
position. Characters on the same line (within 2pt vertical tolerance)
|
|
18
|
+
are concatenated. ``x0`` is the left-edge position of the first
|
|
19
|
+
character, which encodes indentation level.
|
|
20
|
+
"""
|
|
21
|
+
if not chars:
|
|
22
|
+
return []
|
|
23
|
+
lines: list[tuple[float, list[dict]]] = []
|
|
24
|
+
current_top = chars[0]["top"]
|
|
25
|
+
current_chars: list[dict] = [chars[0]]
|
|
26
|
+
|
|
27
|
+
for c in chars[1:]:
|
|
28
|
+
if abs(c["top"] - current_top) > 2.0:
|
|
29
|
+
lines.append((current_top, current_chars))
|
|
30
|
+
current_top = c["top"]
|
|
31
|
+
current_chars = [c]
|
|
32
|
+
else:
|
|
33
|
+
current_chars.append(c)
|
|
34
|
+
|
|
35
|
+
lines.append((current_top, current_chars))
|
|
36
|
+
|
|
37
|
+
result: list[LineTuple] = []
|
|
38
|
+
for top, line_chars in lines:
|
|
39
|
+
# Build text with gap-based space insertion. When consecutive
|
|
40
|
+
# characters have a large horizontal gap (>10pt) a space is
|
|
41
|
+
# inserted to handle two-column layouts (e.g. "#IS Stack
|
|
42
|
+
# underflow occurred." in Intel SDM exception tables).
|
|
43
|
+
parts: list[str] = []
|
|
44
|
+
prev_right = -1.0
|
|
45
|
+
for c in line_chars:
|
|
46
|
+
if prev_right >= 0 and c["x0"] - prev_right > 10.0:
|
|
47
|
+
if parts and not parts[-1].endswith(" "):
|
|
48
|
+
parts.append(" ")
|
|
49
|
+
parts.append(c["text"])
|
|
50
|
+
prev_right = c["x0"] + c.get("width", 0)
|
|
51
|
+
text = "".join(parts).strip()
|
|
52
|
+
max_size = max(c["size"] for c in line_chars)
|
|
53
|
+
x0 = min(c["x0"] for c in line_chars)
|
|
54
|
+
if text:
|
|
55
|
+
result.append((top, max_size, x0, text))
|
|
56
|
+
return result
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def extract_sections_from_lines(
|
|
60
|
+
lines: list[LineTuple],
|
|
61
|
+
heading_min_size: float,
|
|
62
|
+
body_max_size: float,
|
|
63
|
+
known_headings: frozenset[str] | set[str] | None = None,
|
|
64
|
+
) -> dict[str, list[tuple[float, str]]]:
|
|
65
|
+
"""Extract named sections from pre-grouped PDF line tuples."""
|
|
66
|
+
sections: dict[str, list[tuple[float, str]]] = {}
|
|
67
|
+
current_heading: str | None = None
|
|
68
|
+
body_parts: list[tuple[float, str]] = []
|
|
69
|
+
|
|
70
|
+
for _top, size, line_x0, text in lines:
|
|
71
|
+
if known_headings is not None:
|
|
72
|
+
is_heading = text.lower().strip() in known_headings
|
|
73
|
+
else:
|
|
74
|
+
is_heading = size >= heading_min_size
|
|
75
|
+
if is_heading:
|
|
76
|
+
if current_heading is not None and body_parts:
|
|
77
|
+
sections[current_heading] = body_parts
|
|
78
|
+
current_heading = text
|
|
79
|
+
body_parts = []
|
|
80
|
+
elif size <= body_max_size and current_heading is not None:
|
|
81
|
+
body_parts.append((line_x0, text))
|
|
82
|
+
|
|
83
|
+
if current_heading is not None and body_parts:
|
|
84
|
+
sections[current_heading] = body_parts
|
|
85
|
+
|
|
86
|
+
return sections
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def extract_sections_from_chars(
|
|
90
|
+
chars: list[dict],
|
|
91
|
+
heading_min_size: float,
|
|
92
|
+
body_max_size: float,
|
|
93
|
+
known_headings: frozenset[str] | set[str] | None = None,
|
|
94
|
+
) -> dict[str, list[tuple[float, str]]]:
|
|
95
|
+
"""Extract named sections from a list of pdfplumber character dicts.
|
|
96
|
+
|
|
97
|
+
Characters with font size >= *heading_min_size* are treated as section
|
|
98
|
+
headings. Characters with font size <= *body_max_size* are accumulated
|
|
99
|
+
as body text under the current heading.
|
|
100
|
+
|
|
101
|
+
If *known_headings* is provided, heading detection switches to
|
|
102
|
+
**whitelist mode**: a line is a heading ONLY if its lowercased text
|
|
103
|
+
matches the known headings set. Lines at heading font size that do
|
|
104
|
+
NOT match are demoted to body text under the current heading. When
|
|
105
|
+
*known_headings* is ``None``, the original font-size-only heuristic
|
|
106
|
+
is used (backward compatibility).
|
|
107
|
+
|
|
108
|
+
Returns a dict mapping heading text to a list of ``(x0, text)`` tuples
|
|
109
|
+
preserving the left-edge position for indentation reconstruction.
|
|
110
|
+
"""
|
|
111
|
+
return extract_sections_from_lines(
|
|
112
|
+
chars_to_lines(chars),
|
|
113
|
+
heading_min_size=heading_min_size,
|
|
114
|
+
body_max_size=body_max_size,
|
|
115
|
+
known_headings=known_headings,
|
|
116
|
+
)
|