simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,614 @@
|
|
|
1
|
+
"""Intel SDM PDF parser.
|
|
2
|
+
|
|
3
|
+
Extracts per-instruction description sections from the Intel 64 and IA-32
|
|
4
|
+
Architectures Software Developer's Manual (combined volumes PDF).
|
|
5
|
+
|
|
6
|
+
The parser identifies instruction pages by their size-12 title font with
|
|
7
|
+
an all-caps mnemonic before an em-dash, then extracts size-10 section
|
|
8
|
+
headings and size-9 body text within each instruction's page range.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
import os
|
|
15
|
+
import re
|
|
16
|
+
import sys
|
|
17
|
+
from collections import Counter
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Callable
|
|
21
|
+
|
|
22
|
+
import httpx
|
|
23
|
+
|
|
24
|
+
try:
|
|
25
|
+
from pdfminer.pdftypes import resolve1
|
|
26
|
+
except ImportError: # pragma: no cover - exercised in minimal test envs
|
|
27
|
+
def resolve1(value):
|
|
28
|
+
return value
|
|
29
|
+
|
|
30
|
+
from simdref.pdfparse.base import chars_to_lines, extract_sections_from_lines
|
|
31
|
+
from simdref.pdfparse.registry import register_pdf_source
|
|
32
|
+
from simdref.pdfparse.types import PdfDescriptionPayload, PdfEnrichmentResult, PdfSourceSpec
|
|
33
|
+
from simdref.storage import DATA_DIR
|
|
34
|
+
|
|
35
|
+
log = logging.getLogger(__name__)
|
|
36
|
+
|
|
37
|
+
INTEL_SDM_URL = "https://cdrdv2.intel.com/v1/dl/getContent/671200"
|
|
38
|
+
_REPO_ROOT = Path(__file__).resolve().parents[3]
|
|
39
|
+
LOCAL_INTEL_SDM_PDFS = [
|
|
40
|
+
_REPO_ROOT / "vendor" / "intel" / "intel-sdm.pdf",
|
|
41
|
+
]
|
|
42
|
+
INTEL_SDM_CACHE_PATH = DATA_DIR / "intel-sdm-descriptions.msgpack"
|
|
43
|
+
INTEL_SDM_CACHE_VERSION = 1
|
|
44
|
+
INTEL_SDM_SIGNATURE_PATHS = (
|
|
45
|
+
Path(__file__).resolve(),
|
|
46
|
+
Path(__file__).resolve().parent / "base.py",
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# Title pattern: ALL-CAPS mnemonic (with optional / separators) before em-dash.
|
|
50
|
+
_TITLE_RE = re.compile(r"^([A-Z][A-Z0-9/_\s]{0,80})\s*\u2014\s*(.+)")
|
|
51
|
+
|
|
52
|
+
# Words that indicate a chapter/section heading, not an instruction.
|
|
53
|
+
_SKIP_WORDS = frozenset({"CHAPTER", "CONTENTS", "APPENDIX", "VOLUME", "INSTRUCTION SET REFERENCE"})
|
|
54
|
+
|
|
55
|
+
# Font size thresholds (from empirical analysis of Intel SDM).
|
|
56
|
+
_TITLE_MIN_SIZE = 11.5
|
|
57
|
+
_HEADING_MIN_SIZE = 9.8
|
|
58
|
+
_BODY_MAX_SIZE = 9.5
|
|
59
|
+
_BODY_MIN_SIZE = 8.0 # Filter superscripts (®, footnote markers)
|
|
60
|
+
|
|
61
|
+
# Canonical section names and aliases.
|
|
62
|
+
KNOWN_SECTIONS: set[str] = {
|
|
63
|
+
"Description",
|
|
64
|
+
"Operation",
|
|
65
|
+
"Intrinsic Equivalents",
|
|
66
|
+
"Flags Affected",
|
|
67
|
+
"FPU Flags Affected",
|
|
68
|
+
"Exceptions",
|
|
69
|
+
"Numeric Exceptions",
|
|
70
|
+
"SIMD Floating-Point Exceptions",
|
|
71
|
+
"Floating-Point Exceptions",
|
|
72
|
+
"Other Exceptions",
|
|
73
|
+
"Other Mode Exceptions",
|
|
74
|
+
"Protected Mode Exceptions",
|
|
75
|
+
"Real-Address Mode Exceptions",
|
|
76
|
+
"Real Address Mode Exceptions",
|
|
77
|
+
"Virtual-8086 Mode Exceptions",
|
|
78
|
+
"Virtual-8086 Exceptions",
|
|
79
|
+
"Virtual 8086 Mode Exceptions",
|
|
80
|
+
"Compatibility Mode Exceptions",
|
|
81
|
+
"64-Bit Mode Exceptions",
|
|
82
|
+
"x87 FPU and SIMD Floating-Point Exceptions",
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
_SECTION_ALIASES: dict[str, str] = {
|
|
86
|
+
"intel c/c++ compiler intrinsic equivalent": "Intrinsic Equivalents",
|
|
87
|
+
"intel c/c++ compiler intrinsic equivalents": "Intrinsic Equivalents",
|
|
88
|
+
"intel c/c++compiler intrinsic equivalent": "Intrinsic Equivalents",
|
|
89
|
+
"intel c/c++compiler intrinsic equivalents": "Intrinsic Equivalents",
|
|
90
|
+
"c/c++ compiler intrinsic equivalent": "Intrinsic Equivalents",
|
|
91
|
+
"c/c++ compiler intrinsic equivalents": "Intrinsic Equivalents",
|
|
92
|
+
"intrinsic equivalent": "Intrinsic Equivalents",
|
|
93
|
+
"intrinsic equivalents": "Intrinsic Equivalents",
|
|
94
|
+
"instruction operand encoding": "Operand Encoding",
|
|
95
|
+
"fpu flags affected": "FPU Flags Affected",
|
|
96
|
+
"floating-point exceptions": "Floating-Point Exceptions",
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
# Footer pattern: "MNEMONIC—... Vol. 2X N-NN" (space between volume and page optional)
|
|
100
|
+
_FOOTER_RE = re.compile(r"^.+Vol\.\s*2[A-D]?\s*\d+-\d+$")
|
|
101
|
+
|
|
102
|
+
# Sections to discard (tabular data that doesn't render well as text).
|
|
103
|
+
_DISCARD_SECTIONS: frozenset[str] = frozenset({
|
|
104
|
+
"Instruction Operand Encoding",
|
|
105
|
+
"Operand Encoding",
|
|
106
|
+
})
|
|
107
|
+
|
|
108
|
+
# Sections whose text is pseudocode and should preserve indentation from x0.
|
|
109
|
+
_CODE_SECTIONS: frozenset[str] = frozenset({
|
|
110
|
+
"Operation",
|
|
111
|
+
"Intrinsic Equivalents",
|
|
112
|
+
})
|
|
113
|
+
|
|
114
|
+
# All heading names (lowercase) for content-based heading detection.
|
|
115
|
+
# Some Intel SDM pages format section headings at body text size.
|
|
116
|
+
_ALL_HEADING_NAMES: frozenset[str] = frozenset(
|
|
117
|
+
{s.lower() for s in KNOWN_SECTIONS}
|
|
118
|
+
| set(_SECTION_ALIASES.keys())
|
|
119
|
+
| {s.lower() for s in _DISCARD_SECTIONS}
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
# Minimal safety-net for junk lines that appear outside table bounding boxes.
|
|
123
|
+
_JUNK_LINE_RE = re.compile(
|
|
124
|
+
r"^\d+\.\s+See note " # footnote references (1. See note ...)
|
|
125
|
+
r"|^NOTES:\s*$" # trailing NOTES: line
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# Maximum fraction of page area a single table can cover before we
|
|
130
|
+
# consider it a false-positive detection by pdfplumber.
|
|
131
|
+
_TABLE_MAX_PAGE_FRACTION = 0.85
|
|
132
|
+
_TABLE_GRAPHIC_PRIMITIVE_THRESHOLD = 6
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@dataclass(slots=True)
|
|
136
|
+
class _PreparedPage:
|
|
137
|
+
title: tuple[str, str] | None
|
|
138
|
+
body_lines: list[tuple[float, float, float, str]]
|
|
139
|
+
backend: str
|
|
140
|
+
fallback_reason: str | None = None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
_FASTPATH_TABULAR_LINE_RE = re.compile(
|
|
144
|
+
r"^(Opcode|Op/En|64/32-Bit Mode|64-Bit Mode|32-Bit Mode|CPUID Feature Flag)\b",
|
|
145
|
+
re.IGNORECASE,
|
|
146
|
+
)
|
|
147
|
+
_FASTPATH_TABULAR_CAPTION_RE = re.compile(
|
|
148
|
+
r"^(Table\s+\d+-\d+\.|Instruction Operand Encoding\b)",
|
|
149
|
+
re.IGNORECASE,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _resolve_outline_page_number(pdf, dest) -> int | None:
|
|
154
|
+
"""Resolve a pdfminer outline destination to a 1-based page number."""
|
|
155
|
+
if dest is None:
|
|
156
|
+
return None
|
|
157
|
+
try:
|
|
158
|
+
resolved = resolve1(pdf.doc.get_dest(dest) if isinstance(dest, bytes) else dest)
|
|
159
|
+
if isinstance(resolved, dict):
|
|
160
|
+
resolved = resolve1(resolved.get("D"))
|
|
161
|
+
if not isinstance(resolved, list) or not resolved:
|
|
162
|
+
return None
|
|
163
|
+
objid = getattr(resolved[0], "objid", None)
|
|
164
|
+
if objid is None:
|
|
165
|
+
return None
|
|
166
|
+
if not hasattr(pdf, "_simdref_page_map"):
|
|
167
|
+
pdf._simdref_page_map = {page.page_obj.pageid: i for i, page in enumerate(pdf.pages, start=1)}
|
|
168
|
+
return pdf._simdref_page_map.get(objid)
|
|
169
|
+
except Exception:
|
|
170
|
+
return None
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _outline_starts_instruction_range(level: int, title: str) -> bool:
|
|
174
|
+
lowered = title.casefold()
|
|
175
|
+
if title.startswith("Chapter ") and "instruction" in lowered and "reference" in lowered:
|
|
176
|
+
return True
|
|
177
|
+
return "seam instruction reference" in lowered
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _instruction_page_ranges(pdf) -> list[tuple[int, int]]:
|
|
181
|
+
"""Return likely 0-based [start, end) ranges that contain instruction text."""
|
|
182
|
+
total = len(pdf.pages)
|
|
183
|
+
try:
|
|
184
|
+
outlines: list[tuple[int, int, str]] = []
|
|
185
|
+
for level, title, dest, _action, _se in pdf.doc.get_outlines():
|
|
186
|
+
page_number = _resolve_outline_page_number(pdf, dest)
|
|
187
|
+
if page_number is None:
|
|
188
|
+
continue
|
|
189
|
+
outlines.append((page_number, level, title))
|
|
190
|
+
outlines.sort()
|
|
191
|
+
|
|
192
|
+
ranges: list[tuple[int, int]] = []
|
|
193
|
+
for idx, (page_number, level, title) in enumerate(outlines):
|
|
194
|
+
if not _outline_starts_instruction_range(level, title):
|
|
195
|
+
continue
|
|
196
|
+
next_page = total + 1
|
|
197
|
+
for later_page, later_level, _later_title in outlines[idx + 1:]:
|
|
198
|
+
if later_page > page_number and later_level <= level:
|
|
199
|
+
next_page = later_page
|
|
200
|
+
break
|
|
201
|
+
start_idx = max(0, page_number - 1)
|
|
202
|
+
end_idx = max(start_idx + 1, min(total, next_page - 1))
|
|
203
|
+
ranges.append((start_idx, end_idx))
|
|
204
|
+
|
|
205
|
+
if not ranges:
|
|
206
|
+
return [(0, total)]
|
|
207
|
+
|
|
208
|
+
merged: list[tuple[int, int]] = []
|
|
209
|
+
for start_idx, end_idx in sorted(ranges):
|
|
210
|
+
if not merged or start_idx > merged[-1][1]:
|
|
211
|
+
merged.append((start_idx, end_idx))
|
|
212
|
+
else:
|
|
213
|
+
merged[-1] = (merged[-1][0], max(merged[-1][1], end_idx))
|
|
214
|
+
return merged
|
|
215
|
+
except Exception:
|
|
216
|
+
return [(0, total)]
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _page_might_have_tables(page) -> bool:
|
|
220
|
+
"""Cheap precheck before pdfplumber's expensive table finder."""
|
|
221
|
+
primitive_count = len(page.rects) + len(page.lines) + len(page.curves)
|
|
222
|
+
return primitive_count >= _TABLE_GRAPHIC_PRIMITIVE_THRESHOLD
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _table_bboxes_for_page(page) -> list[tuple[float, float, float, float]]:
|
|
226
|
+
if not _page_might_have_tables(page):
|
|
227
|
+
return []
|
|
228
|
+
tables = page.find_tables()
|
|
229
|
+
if not tables:
|
|
230
|
+
return []
|
|
231
|
+
page_area = page.width * page.height
|
|
232
|
+
bboxes_list = []
|
|
233
|
+
for table in tables:
|
|
234
|
+
bx0, by0, bx1, by1 = table.bbox
|
|
235
|
+
table_area = (bx1 - bx0) * (by1 - by0)
|
|
236
|
+
if table_area / page_area < _TABLE_MAX_PAGE_FRACTION:
|
|
237
|
+
bboxes_list.append((bx0, by0, bx1, by1))
|
|
238
|
+
return bboxes_list
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _build_line_text(spans: list[tuple[float, str]]) -> str:
|
|
242
|
+
parts: list[str] = []
|
|
243
|
+
prev_right = -1.0
|
|
244
|
+
for x0, text in spans:
|
|
245
|
+
if prev_right >= 0 and x0 - prev_right > 10.0:
|
|
246
|
+
if parts and not parts[-1].endswith(" "):
|
|
247
|
+
parts.append(" ")
|
|
248
|
+
parts.append(text)
|
|
249
|
+
prev_right = x0 + max(len(text), 1) * 5.0
|
|
250
|
+
return "".join(parts).strip()
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _line_is_tabular_noise(text: str) -> bool:
|
|
254
|
+
stripped = text.strip()
|
|
255
|
+
if not stripped:
|
|
256
|
+
return True
|
|
257
|
+
return bool(
|
|
258
|
+
_FASTPATH_TABULAR_LINE_RE.match(stripped)
|
|
259
|
+
or _FASTPATH_TABULAR_CAPTION_RE.match(stripped)
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _prepare_page_pdfplumber(page) -> _PreparedPage:
|
|
264
|
+
page_chars = page.chars
|
|
265
|
+
title_chars = [c for c in page_chars if c["size"] >= _TITLE_MIN_SIZE]
|
|
266
|
+
parsed_title = None
|
|
267
|
+
if title_chars:
|
|
268
|
+
parsed_title = parse_instruction_title("".join(c["text"] for c in title_chars).strip())
|
|
269
|
+
|
|
270
|
+
bboxes = _table_bboxes_for_page(page)
|
|
271
|
+
body_chars: list[dict] = []
|
|
272
|
+
for char in page_chars:
|
|
273
|
+
if char["size"] >= _TITLE_MIN_SIZE or char["size"] < _BODY_MIN_SIZE:
|
|
274
|
+
continue
|
|
275
|
+
if bboxes and any(
|
|
276
|
+
bbox[0] <= char["x0"] <= bbox[2] and bbox[1] <= char["top"] <= bbox[3]
|
|
277
|
+
for bbox in bboxes
|
|
278
|
+
):
|
|
279
|
+
continue
|
|
280
|
+
body_chars.append(char)
|
|
281
|
+
return _PreparedPage(title=parsed_title, body_lines=chars_to_lines(body_chars), backend="pdfplumber")
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _prepare_page_from_pymupdf_dict(text_dict: dict[str, Any]) -> _PreparedPage:
|
|
285
|
+
title_lines: list[str] = []
|
|
286
|
+
body_lines: list[tuple[float, float, float, str]] = []
|
|
287
|
+
|
|
288
|
+
for block_idx, block in enumerate(text_dict.get("blocks", [])):
|
|
289
|
+
if block.get("type") != 0:
|
|
290
|
+
continue
|
|
291
|
+
for line_idx, line in enumerate(block.get("lines", [])):
|
|
292
|
+
spans: list[tuple[float, str]] = []
|
|
293
|
+
sizes: list[float] = []
|
|
294
|
+
tops: list[float] = []
|
|
295
|
+
left_edges: list[float] = []
|
|
296
|
+
for span in line.get("spans", []):
|
|
297
|
+
text = span.get("text", "")
|
|
298
|
+
if not text.strip():
|
|
299
|
+
continue
|
|
300
|
+
bbox = span.get("bbox", (0.0, 0.0, 0.0, 0.0))
|
|
301
|
+
x0 = float(bbox[0])
|
|
302
|
+
top = float(bbox[1])
|
|
303
|
+
size = float(span.get("size", 0.0))
|
|
304
|
+
spans.append((x0, text))
|
|
305
|
+
sizes.append(size)
|
|
306
|
+
tops.append(top)
|
|
307
|
+
left_edges.append(x0)
|
|
308
|
+
if not spans:
|
|
309
|
+
continue
|
|
310
|
+
|
|
311
|
+
spans.sort(key=lambda item: item[0])
|
|
312
|
+
text = _build_line_text(spans)
|
|
313
|
+
if not text:
|
|
314
|
+
continue
|
|
315
|
+
top = min(tops)
|
|
316
|
+
dominant_size = max(sizes)
|
|
317
|
+
x0 = min(left_edges)
|
|
318
|
+
|
|
319
|
+
if dominant_size >= _TITLE_MIN_SIZE:
|
|
320
|
+
title_lines.append(text)
|
|
321
|
+
continue
|
|
322
|
+
if dominant_size < _BODY_MIN_SIZE:
|
|
323
|
+
continue
|
|
324
|
+
if _line_is_tabular_noise(text):
|
|
325
|
+
continue
|
|
326
|
+
body_lines.append((top, dominant_size, x0, text))
|
|
327
|
+
|
|
328
|
+
title_text = " ".join(title_lines).strip()
|
|
329
|
+
parsed_title = parse_instruction_title(title_text) if title_text else None
|
|
330
|
+
body_lines.sort(key=lambda item: (item[0], item[2], item[1]))
|
|
331
|
+
return _PreparedPage(title=parsed_title, body_lines=body_lines, backend="pymupdf")
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _prepare_page_pymupdf(page) -> _PreparedPage:
|
|
335
|
+
return _prepare_page_from_pymupdf_dict(page.get_text("dict"))
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _prepared_page_needs_fallback(prepared: _PreparedPage) -> str | None:
|
|
339
|
+
if prepared.title is None:
|
|
340
|
+
return None
|
|
341
|
+
if not prepared.body_lines:
|
|
342
|
+
return "empty-fast-path"
|
|
343
|
+
heading_lines = [
|
|
344
|
+
text for _top, _size, _x0, text in prepared.body_lines
|
|
345
|
+
if text.strip().lower() in _ALL_HEADING_NAMES
|
|
346
|
+
]
|
|
347
|
+
if not heading_lines:
|
|
348
|
+
return "missing-heading"
|
|
349
|
+
return None
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def normalize_section_name(raw: str) -> str:
|
|
353
|
+
"""Map a raw heading string to its canonical section name."""
|
|
354
|
+
stripped = raw.strip()
|
|
355
|
+
lowered = stripped.lower()
|
|
356
|
+
if lowered in _SECTION_ALIASES:
|
|
357
|
+
return _SECTION_ALIASES[lowered]
|
|
358
|
+
for known in KNOWN_SECTIONS:
|
|
359
|
+
if lowered == known.lower():
|
|
360
|
+
return known
|
|
361
|
+
return stripped
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def parse_instruction_title(text: str) -> tuple[str, str] | None:
|
|
365
|
+
"""Parse an instruction title into (mnemonic, summary).
|
|
366
|
+
|
|
367
|
+
Returns None if the text is not an instruction title.
|
|
368
|
+
"""
|
|
369
|
+
text = text.strip()
|
|
370
|
+
m = _TITLE_RE.match(text)
|
|
371
|
+
if m is None:
|
|
372
|
+
return None
|
|
373
|
+
mnemonic = m.group(1).strip()
|
|
374
|
+
summary = m.group(2).strip()
|
|
375
|
+
if any(word in mnemonic for word in _SKIP_WORDS):
|
|
376
|
+
return None
|
|
377
|
+
alpha = [c for c in mnemonic if c.isalpha()]
|
|
378
|
+
if not alpha:
|
|
379
|
+
return None
|
|
380
|
+
if sum(1 for c in alpha if c.isupper()) / len(alpha) < 0.9:
|
|
381
|
+
return None
|
|
382
|
+
return mnemonic, summary
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def parse_intel_sdm(
|
|
386
|
+
pdf_path: Path,
|
|
387
|
+
*,
|
|
388
|
+
status: Callable[[str], None] | None = None,
|
|
389
|
+
) -> PdfEnrichmentResult:
|
|
390
|
+
"""Parse the Intel SDM PDF and return per-mnemonic description payloads.
|
|
391
|
+
|
|
392
|
+
Returns a dict mapping uppercase mnemonic to a payload with:
|
|
393
|
+
* ``sections``: dict of section name -> text
|
|
394
|
+
* ``page_start``: 1-based first page in the PDF
|
|
395
|
+
* ``page_end``: 1-based last page in the PDF
|
|
396
|
+
|
|
397
|
+
Mnemonics with ``/`` separators are expanded so each variant maps to the
|
|
398
|
+
same payload.
|
|
399
|
+
"""
|
|
400
|
+
import pdfplumber
|
|
401
|
+
|
|
402
|
+
from rich.progress import BarColumn, MofNCompleteColumn, Progress, SpinnerColumn, TextColumn
|
|
403
|
+
|
|
404
|
+
log.info("parsing Intel SDM: %s", pdf_path)
|
|
405
|
+
try:
|
|
406
|
+
import fitz
|
|
407
|
+
except ImportError:
|
|
408
|
+
fitz = None
|
|
409
|
+
|
|
410
|
+
pdf = pdfplumber.open(pdf_path)
|
|
411
|
+
fitz_doc = fitz.open(pdf_path) if fitz is not None else None
|
|
412
|
+
total = len(pdf.pages)
|
|
413
|
+
log.info("total pages: %d", total)
|
|
414
|
+
page_ranges = _instruction_page_ranges(pdf)
|
|
415
|
+
page_indices = [page_idx for start, end in page_ranges for page_idx in range(start, end)]
|
|
416
|
+
parse_pages = len(page_indices)
|
|
417
|
+
log.info("instruction page ranges: %s (%d pages)", page_ranges, parse_pages)
|
|
418
|
+
|
|
419
|
+
interactive_progress = sys.stderr.isatty() and os.environ.get("GITHUB_ACTIONS") != "true"
|
|
420
|
+
progress = Progress(
|
|
421
|
+
SpinnerColumn(),
|
|
422
|
+
TextColumn("[progress.description]{task.description}"),
|
|
423
|
+
BarColumn(),
|
|
424
|
+
MofNCompleteColumn(),
|
|
425
|
+
transient=True,
|
|
426
|
+
)
|
|
427
|
+
if interactive_progress:
|
|
428
|
+
progress.start()
|
|
429
|
+
if status is not None:
|
|
430
|
+
status(f"Opened Intel SDM PDF with {total} pages")
|
|
431
|
+
if fitz_doc is not None:
|
|
432
|
+
status("Using PyMuPDF fast path for page extraction with pdfplumber fallback")
|
|
433
|
+
else:
|
|
434
|
+
status("PyMuPDF unavailable; using pdfplumber page extraction")
|
|
435
|
+
if parse_pages != total:
|
|
436
|
+
status(
|
|
437
|
+
f"Restricting SDM parse to {len(page_ranges)} outline-derived instruction ranges "
|
|
438
|
+
f"covering {parse_pages} of {total} pages"
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
# Phase 1: preprocess each page once and collect instruction title pages.
|
|
442
|
+
scan_task = progress.add_task("Preprocessing pages", total=parse_pages) if interactive_progress else None
|
|
443
|
+
prepared_pages: dict[int, _PreparedPage] = {}
|
|
444
|
+
title_pages: list[tuple[int, str, str]] = []
|
|
445
|
+
fallback_pages = 0
|
|
446
|
+
for offset, i in enumerate(page_indices):
|
|
447
|
+
prepared = _prepare_page_pymupdf(fitz_doc[i]) if fitz_doc is not None else _prepare_page_pdfplumber(pdf.pages[i])
|
|
448
|
+
fallback_reason = _prepared_page_needs_fallback(prepared)
|
|
449
|
+
if fallback_reason is not None:
|
|
450
|
+
fallback_pages += 1
|
|
451
|
+
prepared = _prepare_page_pdfplumber(pdf.pages[i])
|
|
452
|
+
prepared.fallback_reason = fallback_reason
|
|
453
|
+
prepared_pages[i] = prepared
|
|
454
|
+
if prepared.title is not None:
|
|
455
|
+
title_pages.append((i, prepared.title[0], prepared.title[1]))
|
|
456
|
+
if interactive_progress and scan_task is not None:
|
|
457
|
+
progress.advance(scan_task)
|
|
458
|
+
elif status is not None and ((offset + 1) % 250 == 0 or offset + 1 == parse_pages):
|
|
459
|
+
status(f"Preprocessing SDM pages: {offset + 1}/{parse_pages}")
|
|
460
|
+
|
|
461
|
+
log.info("found %d instruction title pages", len(title_pages))
|
|
462
|
+
if status is not None:
|
|
463
|
+
status(f"Found {len(title_pages)} instruction title pages in Intel SDM")
|
|
464
|
+
if fitz_doc is not None:
|
|
465
|
+
status(f"Fell back to pdfplumber on {fallback_pages} pages")
|
|
466
|
+
|
|
467
|
+
# Phase 2: assemble sections from cached per-page lines.
|
|
468
|
+
extract_task = progress.add_task("Extracting descriptions", total=len(title_pages)) if interactive_progress else None
|
|
469
|
+
result: dict[str, PdfDescriptionPayload] = {}
|
|
470
|
+
for idx, (page_start, mnemonic, _summary) in enumerate(title_pages):
|
|
471
|
+
page_end = title_pages[idx + 1][0] if idx + 1 < len(title_pages) else min(page_start + 10, total)
|
|
472
|
+
|
|
473
|
+
all_lines: list[tuple[float, float, float, str]] = []
|
|
474
|
+
for page_idx in range(page_start, page_end):
|
|
475
|
+
prepared = prepared_pages.get(page_idx)
|
|
476
|
+
if prepared is not None:
|
|
477
|
+
all_lines.extend(prepared.body_lines)
|
|
478
|
+
|
|
479
|
+
raw_sections = extract_sections_from_lines(
|
|
480
|
+
all_lines,
|
|
481
|
+
heading_min_size=_HEADING_MIN_SIZE,
|
|
482
|
+
body_max_size=_BODY_MAX_SIZE,
|
|
483
|
+
known_headings=_ALL_HEADING_NAMES,
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
sections: dict[str, str] = {}
|
|
487
|
+
for heading, line_tuples in raw_sections.items():
|
|
488
|
+
canonical = normalize_section_name(heading)
|
|
489
|
+
if canonical in _DISCARD_SECTIONS:
|
|
490
|
+
continue
|
|
491
|
+
# Filter footer and residual junk lines.
|
|
492
|
+
filtered = [
|
|
493
|
+
(x0, text) for x0, text in line_tuples
|
|
494
|
+
if not _FOOTER_RE.match(text) and not _JUNK_LINE_RE.match(text)
|
|
495
|
+
]
|
|
496
|
+
if not filtered:
|
|
497
|
+
continue
|
|
498
|
+
if canonical in _CODE_SECTIONS:
|
|
499
|
+
# Reconstruct indentation from x0 positions.
|
|
500
|
+
min_x0 = min(x0 for x0, _ in filtered)
|
|
501
|
+
indent_unit = 18.0 # ~18pt per indent level in Intel SDM
|
|
502
|
+
out_lines = []
|
|
503
|
+
for x0, text in filtered:
|
|
504
|
+
level = round((x0 - min_x0) / indent_unit)
|
|
505
|
+
out_lines.append(" " * level + text)
|
|
506
|
+
cleaned = "\n".join(out_lines).strip()
|
|
507
|
+
else:
|
|
508
|
+
# Filter footnote lines between tables that have a
|
|
509
|
+
# significantly different left-edge position (x0).
|
|
510
|
+
if len(filtered) > 3:
|
|
511
|
+
x0_counts: Counter[int] = Counter(
|
|
512
|
+
round(x0) for x0, _ in filtered
|
|
513
|
+
)
|
|
514
|
+
dominant_x0 = x0_counts.most_common(1)[0][0]
|
|
515
|
+
filtered = [
|
|
516
|
+
(x0, t) for x0, t in filtered
|
|
517
|
+
if abs(round(x0) - dominant_x0) <= 3
|
|
518
|
+
]
|
|
519
|
+
prose_lines = [text for _, text in filtered]
|
|
520
|
+
# Join PDF-wrapped lines into paragraphs. A new paragraph
|
|
521
|
+
# starts when the previous line ends with sentence-terminal
|
|
522
|
+
# punctuation. Bullet/numbered list items also start new
|
|
523
|
+
# paragraphs.
|
|
524
|
+
paragraphs: list[str] = []
|
|
525
|
+
for line in prose_lines:
|
|
526
|
+
if not paragraphs:
|
|
527
|
+
paragraphs.append(line)
|
|
528
|
+
elif paragraphs[-1][-1:] in ".):;":
|
|
529
|
+
paragraphs.append(line)
|
|
530
|
+
elif line[:1] in ("\u2022", "\u2013") or re.match(r"^\d+\.\s", line):
|
|
531
|
+
# Bullet points or numbered list items.
|
|
532
|
+
paragraphs.append(line)
|
|
533
|
+
elif paragraphs[-1].endswith("-"):
|
|
534
|
+
# De-hyphenate word breaks (e.g. "indi-\ncate").
|
|
535
|
+
paragraphs[-1] = paragraphs[-1][:-1] + line
|
|
536
|
+
else:
|
|
537
|
+
paragraphs[-1] += " " + line
|
|
538
|
+
cleaned = "\n".join(paragraphs).strip()
|
|
539
|
+
if cleaned:
|
|
540
|
+
sections[canonical] = cleaned
|
|
541
|
+
|
|
542
|
+
payload = PdfDescriptionPayload(
|
|
543
|
+
sections=sections,
|
|
544
|
+
source_url=INTEL_SDM_URL,
|
|
545
|
+
page_start=page_start + 1,
|
|
546
|
+
page_end=page_end,
|
|
547
|
+
)
|
|
548
|
+
for part in mnemonic.split("/"):
|
|
549
|
+
part = part.strip()
|
|
550
|
+
if part:
|
|
551
|
+
result[part.upper()] = payload
|
|
552
|
+
if interactive_progress and extract_task is not None:
|
|
553
|
+
progress.advance(extract_task)
|
|
554
|
+
elif status is not None and ((idx + 1) % 50 == 0 or idx + 1 == len(title_pages)):
|
|
555
|
+
status(f"Extracting SDM descriptions: {idx + 1}/{len(title_pages)} instructions")
|
|
556
|
+
|
|
557
|
+
if interactive_progress:
|
|
558
|
+
progress.stop()
|
|
559
|
+
if fitz_doc is not None:
|
|
560
|
+
fitz_doc.close()
|
|
561
|
+
pdf.close()
|
|
562
|
+
log.info("extracted descriptions for %d mnemonics", len(result))
|
|
563
|
+
if status is not None:
|
|
564
|
+
status(f"Extracted SDM descriptions for {len(result)} mnemonic variants")
|
|
565
|
+
return PdfEnrichmentResult(
|
|
566
|
+
descriptions=result,
|
|
567
|
+
fallback_page_count=fallback_pages,
|
|
568
|
+
stats={"mnemonic_variants": len(result)},
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def find_intel_sdm_pdf() -> Path | None:
|
|
573
|
+
"""Locate or download the Intel SDM PDF."""
|
|
574
|
+
for pdf_path in LOCAL_INTEL_SDM_PDFS:
|
|
575
|
+
if pdf_path.exists():
|
|
576
|
+
return pdf_path
|
|
577
|
+
try:
|
|
578
|
+
from rich.progress import BarColumn, DownloadColumn, Progress, TransferSpeedColumn
|
|
579
|
+
|
|
580
|
+
dest = LOCAL_INTEL_SDM_PDFS[0]
|
|
581
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
582
|
+
with httpx.Client(follow_redirects=True, timeout=120.0) as client:
|
|
583
|
+
with client.stream("GET", INTEL_SDM_URL) as resp:
|
|
584
|
+
resp.raise_for_status()
|
|
585
|
+
total = int(resp.headers.get("content-length", 0))
|
|
586
|
+
with Progress(
|
|
587
|
+
"[progress.description]{task.description}",
|
|
588
|
+
BarColumn(),
|
|
589
|
+
DownloadColumn(),
|
|
590
|
+
TransferSpeedColumn(),
|
|
591
|
+
) as progress:
|
|
592
|
+
task = progress.add_task("Downloading Intel SDM PDF", total=total or None)
|
|
593
|
+
with open(dest, "wb") as fh:
|
|
594
|
+
for chunk in resp.iter_bytes(65536):
|
|
595
|
+
fh.write(chunk)
|
|
596
|
+
progress.advance(task, len(chunk))
|
|
597
|
+
return dest
|
|
598
|
+
except Exception:
|
|
599
|
+
return None
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
INTEL_PDF_SOURCE = PdfSourceSpec(
|
|
603
|
+
source_id="intel-sdm",
|
|
604
|
+
display_name="Intel SDM",
|
|
605
|
+
source_url=INTEL_SDM_URL,
|
|
606
|
+
local_candidates=tuple(LOCAL_INTEL_SDM_PDFS),
|
|
607
|
+
cache_path=INTEL_SDM_CACHE_PATH,
|
|
608
|
+
cache_version=INTEL_SDM_CACHE_VERSION,
|
|
609
|
+
signature_paths=INTEL_SDM_SIGNATURE_PATHS,
|
|
610
|
+
parser=parse_intel_sdm,
|
|
611
|
+
find_source=find_intel_sdm_pdf,
|
|
612
|
+
)
|
|
613
|
+
|
|
614
|
+
register_pdf_source(INTEL_PDF_SOURCE)
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""PDF source registry."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from simdref.pdfparse.types import PdfSourceSpec
|
|
6
|
+
|
|
7
|
+
_PDF_SOURCES: dict[str, PdfSourceSpec] = {}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def register_pdf_source(spec: PdfSourceSpec) -> None:
|
|
11
|
+
_PDF_SOURCES[spec.source_id] = spec
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def get_pdf_source(source_id: str) -> PdfSourceSpec:
|
|
15
|
+
return _PDF_SOURCES[source_id]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def iter_pdf_sources() -> tuple[PdfSourceSpec, ...]:
|
|
19
|
+
return tuple(_PDF_SOURCES.values())
|