simdref 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,614 @@
1
+ """Intel SDM PDF parser.
2
+
3
+ Extracts per-instruction description sections from the Intel 64 and IA-32
4
+ Architectures Software Developer's Manual (combined volumes PDF).
5
+
6
+ The parser identifies instruction pages by their size-12 title font with
7
+ an all-caps mnemonic before an em-dash, then extracts size-10 section
8
+ headings and size-9 body text within each instruction's page range.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import logging
14
+ import os
15
+ import re
16
+ import sys
17
+ from collections import Counter
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+ from typing import Any, Callable
21
+
22
+ import httpx
23
+
24
+ try:
25
+ from pdfminer.pdftypes import resolve1
26
+ except ImportError: # pragma: no cover - exercised in minimal test envs
27
+ def resolve1(value):
28
+ return value
29
+
30
+ from simdref.pdfparse.base import chars_to_lines, extract_sections_from_lines
31
+ from simdref.pdfparse.registry import register_pdf_source
32
+ from simdref.pdfparse.types import PdfDescriptionPayload, PdfEnrichmentResult, PdfSourceSpec
33
+ from simdref.storage import DATA_DIR
34
+
35
+ log = logging.getLogger(__name__)
36
+
37
+ INTEL_SDM_URL = "https://cdrdv2.intel.com/v1/dl/getContent/671200"
38
+ _REPO_ROOT = Path(__file__).resolve().parents[3]
39
+ LOCAL_INTEL_SDM_PDFS = [
40
+ _REPO_ROOT / "vendor" / "intel" / "intel-sdm.pdf",
41
+ ]
42
+ INTEL_SDM_CACHE_PATH = DATA_DIR / "intel-sdm-descriptions.msgpack"
43
+ INTEL_SDM_CACHE_VERSION = 1
44
+ INTEL_SDM_SIGNATURE_PATHS = (
45
+ Path(__file__).resolve(),
46
+ Path(__file__).resolve().parent / "base.py",
47
+ )
48
+
49
+ # Title pattern: ALL-CAPS mnemonic (with optional / separators) before em-dash.
50
+ _TITLE_RE = re.compile(r"^([A-Z][A-Z0-9/_\s]{0,80})\s*\u2014\s*(.+)")
51
+
52
+ # Words that indicate a chapter/section heading, not an instruction.
53
+ _SKIP_WORDS = frozenset({"CHAPTER", "CONTENTS", "APPENDIX", "VOLUME", "INSTRUCTION SET REFERENCE"})
54
+
55
+ # Font size thresholds (from empirical analysis of Intel SDM).
56
+ _TITLE_MIN_SIZE = 11.5
57
+ _HEADING_MIN_SIZE = 9.8
58
+ _BODY_MAX_SIZE = 9.5
59
+ _BODY_MIN_SIZE = 8.0 # Filter superscripts (®, footnote markers)
60
+
61
+ # Canonical section names and aliases.
62
+ KNOWN_SECTIONS: set[str] = {
63
+ "Description",
64
+ "Operation",
65
+ "Intrinsic Equivalents",
66
+ "Flags Affected",
67
+ "FPU Flags Affected",
68
+ "Exceptions",
69
+ "Numeric Exceptions",
70
+ "SIMD Floating-Point Exceptions",
71
+ "Floating-Point Exceptions",
72
+ "Other Exceptions",
73
+ "Other Mode Exceptions",
74
+ "Protected Mode Exceptions",
75
+ "Real-Address Mode Exceptions",
76
+ "Real Address Mode Exceptions",
77
+ "Virtual-8086 Mode Exceptions",
78
+ "Virtual-8086 Exceptions",
79
+ "Virtual 8086 Mode Exceptions",
80
+ "Compatibility Mode Exceptions",
81
+ "64-Bit Mode Exceptions",
82
+ "x87 FPU and SIMD Floating-Point Exceptions",
83
+ }
84
+
85
+ _SECTION_ALIASES: dict[str, str] = {
86
+ "intel c/c++ compiler intrinsic equivalent": "Intrinsic Equivalents",
87
+ "intel c/c++ compiler intrinsic equivalents": "Intrinsic Equivalents",
88
+ "intel c/c++compiler intrinsic equivalent": "Intrinsic Equivalents",
89
+ "intel c/c++compiler intrinsic equivalents": "Intrinsic Equivalents",
90
+ "c/c++ compiler intrinsic equivalent": "Intrinsic Equivalents",
91
+ "c/c++ compiler intrinsic equivalents": "Intrinsic Equivalents",
92
+ "intrinsic equivalent": "Intrinsic Equivalents",
93
+ "intrinsic equivalents": "Intrinsic Equivalents",
94
+ "instruction operand encoding": "Operand Encoding",
95
+ "fpu flags affected": "FPU Flags Affected",
96
+ "floating-point exceptions": "Floating-Point Exceptions",
97
+ }
98
+
99
+ # Footer pattern: "MNEMONIC—... Vol. 2X N-NN" (space between volume and page optional)
100
+ _FOOTER_RE = re.compile(r"^.+Vol\.\s*2[A-D]?\s*\d+-\d+$")
101
+
102
+ # Sections to discard (tabular data that doesn't render well as text).
103
+ _DISCARD_SECTIONS: frozenset[str] = frozenset({
104
+ "Instruction Operand Encoding",
105
+ "Operand Encoding",
106
+ })
107
+
108
+ # Sections whose text is pseudocode and should preserve indentation from x0.
109
+ _CODE_SECTIONS: frozenset[str] = frozenset({
110
+ "Operation",
111
+ "Intrinsic Equivalents",
112
+ })
113
+
114
+ # All heading names (lowercase) for content-based heading detection.
115
+ # Some Intel SDM pages format section headings at body text size.
116
+ _ALL_HEADING_NAMES: frozenset[str] = frozenset(
117
+ {s.lower() for s in KNOWN_SECTIONS}
118
+ | set(_SECTION_ALIASES.keys())
119
+ | {s.lower() for s in _DISCARD_SECTIONS}
120
+ )
121
+
122
+ # Minimal safety-net for junk lines that appear outside table bounding boxes.
123
+ _JUNK_LINE_RE = re.compile(
124
+ r"^\d+\.\s+See note " # footnote references (1. See note ...)
125
+ r"|^NOTES:\s*$" # trailing NOTES: line
126
+ )
127
+
128
+
129
+ # Maximum fraction of page area a single table can cover before we
130
+ # consider it a false-positive detection by pdfplumber.
131
+ _TABLE_MAX_PAGE_FRACTION = 0.85
132
+ _TABLE_GRAPHIC_PRIMITIVE_THRESHOLD = 6
133
+
134
+
135
+ @dataclass(slots=True)
136
+ class _PreparedPage:
137
+ title: tuple[str, str] | None
138
+ body_lines: list[tuple[float, float, float, str]]
139
+ backend: str
140
+ fallback_reason: str | None = None
141
+
142
+
143
+ _FASTPATH_TABULAR_LINE_RE = re.compile(
144
+ r"^(Opcode|Op/En|64/32-Bit Mode|64-Bit Mode|32-Bit Mode|CPUID Feature Flag)\b",
145
+ re.IGNORECASE,
146
+ )
147
+ _FASTPATH_TABULAR_CAPTION_RE = re.compile(
148
+ r"^(Table\s+\d+-\d+\.|Instruction Operand Encoding\b)",
149
+ re.IGNORECASE,
150
+ )
151
+
152
+
153
+ def _resolve_outline_page_number(pdf, dest) -> int | None:
154
+ """Resolve a pdfminer outline destination to a 1-based page number."""
155
+ if dest is None:
156
+ return None
157
+ try:
158
+ resolved = resolve1(pdf.doc.get_dest(dest) if isinstance(dest, bytes) else dest)
159
+ if isinstance(resolved, dict):
160
+ resolved = resolve1(resolved.get("D"))
161
+ if not isinstance(resolved, list) or not resolved:
162
+ return None
163
+ objid = getattr(resolved[0], "objid", None)
164
+ if objid is None:
165
+ return None
166
+ if not hasattr(pdf, "_simdref_page_map"):
167
+ pdf._simdref_page_map = {page.page_obj.pageid: i for i, page in enumerate(pdf.pages, start=1)}
168
+ return pdf._simdref_page_map.get(objid)
169
+ except Exception:
170
+ return None
171
+
172
+
173
+ def _outline_starts_instruction_range(level: int, title: str) -> bool:
174
+ lowered = title.casefold()
175
+ if title.startswith("Chapter ") and "instruction" in lowered and "reference" in lowered:
176
+ return True
177
+ return "seam instruction reference" in lowered
178
+
179
+
180
+ def _instruction_page_ranges(pdf) -> list[tuple[int, int]]:
181
+ """Return likely 0-based [start, end) ranges that contain instruction text."""
182
+ total = len(pdf.pages)
183
+ try:
184
+ outlines: list[tuple[int, int, str]] = []
185
+ for level, title, dest, _action, _se in pdf.doc.get_outlines():
186
+ page_number = _resolve_outline_page_number(pdf, dest)
187
+ if page_number is None:
188
+ continue
189
+ outlines.append((page_number, level, title))
190
+ outlines.sort()
191
+
192
+ ranges: list[tuple[int, int]] = []
193
+ for idx, (page_number, level, title) in enumerate(outlines):
194
+ if not _outline_starts_instruction_range(level, title):
195
+ continue
196
+ next_page = total + 1
197
+ for later_page, later_level, _later_title in outlines[idx + 1:]:
198
+ if later_page > page_number and later_level <= level:
199
+ next_page = later_page
200
+ break
201
+ start_idx = max(0, page_number - 1)
202
+ end_idx = max(start_idx + 1, min(total, next_page - 1))
203
+ ranges.append((start_idx, end_idx))
204
+
205
+ if not ranges:
206
+ return [(0, total)]
207
+
208
+ merged: list[tuple[int, int]] = []
209
+ for start_idx, end_idx in sorted(ranges):
210
+ if not merged or start_idx > merged[-1][1]:
211
+ merged.append((start_idx, end_idx))
212
+ else:
213
+ merged[-1] = (merged[-1][0], max(merged[-1][1], end_idx))
214
+ return merged
215
+ except Exception:
216
+ return [(0, total)]
217
+
218
+
219
+ def _page_might_have_tables(page) -> bool:
220
+ """Cheap precheck before pdfplumber's expensive table finder."""
221
+ primitive_count = len(page.rects) + len(page.lines) + len(page.curves)
222
+ return primitive_count >= _TABLE_GRAPHIC_PRIMITIVE_THRESHOLD
223
+
224
+
225
+ def _table_bboxes_for_page(page) -> list[tuple[float, float, float, float]]:
226
+ if not _page_might_have_tables(page):
227
+ return []
228
+ tables = page.find_tables()
229
+ if not tables:
230
+ return []
231
+ page_area = page.width * page.height
232
+ bboxes_list = []
233
+ for table in tables:
234
+ bx0, by0, bx1, by1 = table.bbox
235
+ table_area = (bx1 - bx0) * (by1 - by0)
236
+ if table_area / page_area < _TABLE_MAX_PAGE_FRACTION:
237
+ bboxes_list.append((bx0, by0, bx1, by1))
238
+ return bboxes_list
239
+
240
+
241
+ def _build_line_text(spans: list[tuple[float, str]]) -> str:
242
+ parts: list[str] = []
243
+ prev_right = -1.0
244
+ for x0, text in spans:
245
+ if prev_right >= 0 and x0 - prev_right > 10.0:
246
+ if parts and not parts[-1].endswith(" "):
247
+ parts.append(" ")
248
+ parts.append(text)
249
+ prev_right = x0 + max(len(text), 1) * 5.0
250
+ return "".join(parts).strip()
251
+
252
+
253
+ def _line_is_tabular_noise(text: str) -> bool:
254
+ stripped = text.strip()
255
+ if not stripped:
256
+ return True
257
+ return bool(
258
+ _FASTPATH_TABULAR_LINE_RE.match(stripped)
259
+ or _FASTPATH_TABULAR_CAPTION_RE.match(stripped)
260
+ )
261
+
262
+
263
+ def _prepare_page_pdfplumber(page) -> _PreparedPage:
264
+ page_chars = page.chars
265
+ title_chars = [c for c in page_chars if c["size"] >= _TITLE_MIN_SIZE]
266
+ parsed_title = None
267
+ if title_chars:
268
+ parsed_title = parse_instruction_title("".join(c["text"] for c in title_chars).strip())
269
+
270
+ bboxes = _table_bboxes_for_page(page)
271
+ body_chars: list[dict] = []
272
+ for char in page_chars:
273
+ if char["size"] >= _TITLE_MIN_SIZE or char["size"] < _BODY_MIN_SIZE:
274
+ continue
275
+ if bboxes and any(
276
+ bbox[0] <= char["x0"] <= bbox[2] and bbox[1] <= char["top"] <= bbox[3]
277
+ for bbox in bboxes
278
+ ):
279
+ continue
280
+ body_chars.append(char)
281
+ return _PreparedPage(title=parsed_title, body_lines=chars_to_lines(body_chars), backend="pdfplumber")
282
+
283
+
284
+ def _prepare_page_from_pymupdf_dict(text_dict: dict[str, Any]) -> _PreparedPage:
285
+ title_lines: list[str] = []
286
+ body_lines: list[tuple[float, float, float, str]] = []
287
+
288
+ for block_idx, block in enumerate(text_dict.get("blocks", [])):
289
+ if block.get("type") != 0:
290
+ continue
291
+ for line_idx, line in enumerate(block.get("lines", [])):
292
+ spans: list[tuple[float, str]] = []
293
+ sizes: list[float] = []
294
+ tops: list[float] = []
295
+ left_edges: list[float] = []
296
+ for span in line.get("spans", []):
297
+ text = span.get("text", "")
298
+ if not text.strip():
299
+ continue
300
+ bbox = span.get("bbox", (0.0, 0.0, 0.0, 0.0))
301
+ x0 = float(bbox[0])
302
+ top = float(bbox[1])
303
+ size = float(span.get("size", 0.0))
304
+ spans.append((x0, text))
305
+ sizes.append(size)
306
+ tops.append(top)
307
+ left_edges.append(x0)
308
+ if not spans:
309
+ continue
310
+
311
+ spans.sort(key=lambda item: item[0])
312
+ text = _build_line_text(spans)
313
+ if not text:
314
+ continue
315
+ top = min(tops)
316
+ dominant_size = max(sizes)
317
+ x0 = min(left_edges)
318
+
319
+ if dominant_size >= _TITLE_MIN_SIZE:
320
+ title_lines.append(text)
321
+ continue
322
+ if dominant_size < _BODY_MIN_SIZE:
323
+ continue
324
+ if _line_is_tabular_noise(text):
325
+ continue
326
+ body_lines.append((top, dominant_size, x0, text))
327
+
328
+ title_text = " ".join(title_lines).strip()
329
+ parsed_title = parse_instruction_title(title_text) if title_text else None
330
+ body_lines.sort(key=lambda item: (item[0], item[2], item[1]))
331
+ return _PreparedPage(title=parsed_title, body_lines=body_lines, backend="pymupdf")
332
+
333
+
334
+ def _prepare_page_pymupdf(page) -> _PreparedPage:
335
+ return _prepare_page_from_pymupdf_dict(page.get_text("dict"))
336
+
337
+
338
+ def _prepared_page_needs_fallback(prepared: _PreparedPage) -> str | None:
339
+ if prepared.title is None:
340
+ return None
341
+ if not prepared.body_lines:
342
+ return "empty-fast-path"
343
+ heading_lines = [
344
+ text for _top, _size, _x0, text in prepared.body_lines
345
+ if text.strip().lower() in _ALL_HEADING_NAMES
346
+ ]
347
+ if not heading_lines:
348
+ return "missing-heading"
349
+ return None
350
+
351
+
352
+ def normalize_section_name(raw: str) -> str:
353
+ """Map a raw heading string to its canonical section name."""
354
+ stripped = raw.strip()
355
+ lowered = stripped.lower()
356
+ if lowered in _SECTION_ALIASES:
357
+ return _SECTION_ALIASES[lowered]
358
+ for known in KNOWN_SECTIONS:
359
+ if lowered == known.lower():
360
+ return known
361
+ return stripped
362
+
363
+
364
+ def parse_instruction_title(text: str) -> tuple[str, str] | None:
365
+ """Parse an instruction title into (mnemonic, summary).
366
+
367
+ Returns None if the text is not an instruction title.
368
+ """
369
+ text = text.strip()
370
+ m = _TITLE_RE.match(text)
371
+ if m is None:
372
+ return None
373
+ mnemonic = m.group(1).strip()
374
+ summary = m.group(2).strip()
375
+ if any(word in mnemonic for word in _SKIP_WORDS):
376
+ return None
377
+ alpha = [c for c in mnemonic if c.isalpha()]
378
+ if not alpha:
379
+ return None
380
+ if sum(1 for c in alpha if c.isupper()) / len(alpha) < 0.9:
381
+ return None
382
+ return mnemonic, summary
383
+
384
+
385
+ def parse_intel_sdm(
386
+ pdf_path: Path,
387
+ *,
388
+ status: Callable[[str], None] | None = None,
389
+ ) -> PdfEnrichmentResult:
390
+ """Parse the Intel SDM PDF and return per-mnemonic description payloads.
391
+
392
+ Returns a dict mapping uppercase mnemonic to a payload with:
393
+ * ``sections``: dict of section name -> text
394
+ * ``page_start``: 1-based first page in the PDF
395
+ * ``page_end``: 1-based last page in the PDF
396
+
397
+ Mnemonics with ``/`` separators are expanded so each variant maps to the
398
+ same payload.
399
+ """
400
+ import pdfplumber
401
+
402
+ from rich.progress import BarColumn, MofNCompleteColumn, Progress, SpinnerColumn, TextColumn
403
+
404
+ log.info("parsing Intel SDM: %s", pdf_path)
405
+ try:
406
+ import fitz
407
+ except ImportError:
408
+ fitz = None
409
+
410
+ pdf = pdfplumber.open(pdf_path)
411
+ fitz_doc = fitz.open(pdf_path) if fitz is not None else None
412
+ total = len(pdf.pages)
413
+ log.info("total pages: %d", total)
414
+ page_ranges = _instruction_page_ranges(pdf)
415
+ page_indices = [page_idx for start, end in page_ranges for page_idx in range(start, end)]
416
+ parse_pages = len(page_indices)
417
+ log.info("instruction page ranges: %s (%d pages)", page_ranges, parse_pages)
418
+
419
+ interactive_progress = sys.stderr.isatty() and os.environ.get("GITHUB_ACTIONS") != "true"
420
+ progress = Progress(
421
+ SpinnerColumn(),
422
+ TextColumn("[progress.description]{task.description}"),
423
+ BarColumn(),
424
+ MofNCompleteColumn(),
425
+ transient=True,
426
+ )
427
+ if interactive_progress:
428
+ progress.start()
429
+ if status is not None:
430
+ status(f"Opened Intel SDM PDF with {total} pages")
431
+ if fitz_doc is not None:
432
+ status("Using PyMuPDF fast path for page extraction with pdfplumber fallback")
433
+ else:
434
+ status("PyMuPDF unavailable; using pdfplumber page extraction")
435
+ if parse_pages != total:
436
+ status(
437
+ f"Restricting SDM parse to {len(page_ranges)} outline-derived instruction ranges "
438
+ f"covering {parse_pages} of {total} pages"
439
+ )
440
+
441
+ # Phase 1: preprocess each page once and collect instruction title pages.
442
+ scan_task = progress.add_task("Preprocessing pages", total=parse_pages) if interactive_progress else None
443
+ prepared_pages: dict[int, _PreparedPage] = {}
444
+ title_pages: list[tuple[int, str, str]] = []
445
+ fallback_pages = 0
446
+ for offset, i in enumerate(page_indices):
447
+ prepared = _prepare_page_pymupdf(fitz_doc[i]) if fitz_doc is not None else _prepare_page_pdfplumber(pdf.pages[i])
448
+ fallback_reason = _prepared_page_needs_fallback(prepared)
449
+ if fallback_reason is not None:
450
+ fallback_pages += 1
451
+ prepared = _prepare_page_pdfplumber(pdf.pages[i])
452
+ prepared.fallback_reason = fallback_reason
453
+ prepared_pages[i] = prepared
454
+ if prepared.title is not None:
455
+ title_pages.append((i, prepared.title[0], prepared.title[1]))
456
+ if interactive_progress and scan_task is not None:
457
+ progress.advance(scan_task)
458
+ elif status is not None and ((offset + 1) % 250 == 0 or offset + 1 == parse_pages):
459
+ status(f"Preprocessing SDM pages: {offset + 1}/{parse_pages}")
460
+
461
+ log.info("found %d instruction title pages", len(title_pages))
462
+ if status is not None:
463
+ status(f"Found {len(title_pages)} instruction title pages in Intel SDM")
464
+ if fitz_doc is not None:
465
+ status(f"Fell back to pdfplumber on {fallback_pages} pages")
466
+
467
+ # Phase 2: assemble sections from cached per-page lines.
468
+ extract_task = progress.add_task("Extracting descriptions", total=len(title_pages)) if interactive_progress else None
469
+ result: dict[str, PdfDescriptionPayload] = {}
470
+ for idx, (page_start, mnemonic, _summary) in enumerate(title_pages):
471
+ page_end = title_pages[idx + 1][0] if idx + 1 < len(title_pages) else min(page_start + 10, total)
472
+
473
+ all_lines: list[tuple[float, float, float, str]] = []
474
+ for page_idx in range(page_start, page_end):
475
+ prepared = prepared_pages.get(page_idx)
476
+ if prepared is not None:
477
+ all_lines.extend(prepared.body_lines)
478
+
479
+ raw_sections = extract_sections_from_lines(
480
+ all_lines,
481
+ heading_min_size=_HEADING_MIN_SIZE,
482
+ body_max_size=_BODY_MAX_SIZE,
483
+ known_headings=_ALL_HEADING_NAMES,
484
+ )
485
+
486
+ sections: dict[str, str] = {}
487
+ for heading, line_tuples in raw_sections.items():
488
+ canonical = normalize_section_name(heading)
489
+ if canonical in _DISCARD_SECTIONS:
490
+ continue
491
+ # Filter footer and residual junk lines.
492
+ filtered = [
493
+ (x0, text) for x0, text in line_tuples
494
+ if not _FOOTER_RE.match(text) and not _JUNK_LINE_RE.match(text)
495
+ ]
496
+ if not filtered:
497
+ continue
498
+ if canonical in _CODE_SECTIONS:
499
+ # Reconstruct indentation from x0 positions.
500
+ min_x0 = min(x0 for x0, _ in filtered)
501
+ indent_unit = 18.0 # ~18pt per indent level in Intel SDM
502
+ out_lines = []
503
+ for x0, text in filtered:
504
+ level = round((x0 - min_x0) / indent_unit)
505
+ out_lines.append(" " * level + text)
506
+ cleaned = "\n".join(out_lines).strip()
507
+ else:
508
+ # Filter footnote lines between tables that have a
509
+ # significantly different left-edge position (x0).
510
+ if len(filtered) > 3:
511
+ x0_counts: Counter[int] = Counter(
512
+ round(x0) for x0, _ in filtered
513
+ )
514
+ dominant_x0 = x0_counts.most_common(1)[0][0]
515
+ filtered = [
516
+ (x0, t) for x0, t in filtered
517
+ if abs(round(x0) - dominant_x0) <= 3
518
+ ]
519
+ prose_lines = [text for _, text in filtered]
520
+ # Join PDF-wrapped lines into paragraphs. A new paragraph
521
+ # starts when the previous line ends with sentence-terminal
522
+ # punctuation. Bullet/numbered list items also start new
523
+ # paragraphs.
524
+ paragraphs: list[str] = []
525
+ for line in prose_lines:
526
+ if not paragraphs:
527
+ paragraphs.append(line)
528
+ elif paragraphs[-1][-1:] in ".):;":
529
+ paragraphs.append(line)
530
+ elif line[:1] in ("\u2022", "\u2013") or re.match(r"^\d+\.\s", line):
531
+ # Bullet points or numbered list items.
532
+ paragraphs.append(line)
533
+ elif paragraphs[-1].endswith("-"):
534
+ # De-hyphenate word breaks (e.g. "indi-\ncate").
535
+ paragraphs[-1] = paragraphs[-1][:-1] + line
536
+ else:
537
+ paragraphs[-1] += " " + line
538
+ cleaned = "\n".join(paragraphs).strip()
539
+ if cleaned:
540
+ sections[canonical] = cleaned
541
+
542
+ payload = PdfDescriptionPayload(
543
+ sections=sections,
544
+ source_url=INTEL_SDM_URL,
545
+ page_start=page_start + 1,
546
+ page_end=page_end,
547
+ )
548
+ for part in mnemonic.split("/"):
549
+ part = part.strip()
550
+ if part:
551
+ result[part.upper()] = payload
552
+ if interactive_progress and extract_task is not None:
553
+ progress.advance(extract_task)
554
+ elif status is not None and ((idx + 1) % 50 == 0 or idx + 1 == len(title_pages)):
555
+ status(f"Extracting SDM descriptions: {idx + 1}/{len(title_pages)} instructions")
556
+
557
+ if interactive_progress:
558
+ progress.stop()
559
+ if fitz_doc is not None:
560
+ fitz_doc.close()
561
+ pdf.close()
562
+ log.info("extracted descriptions for %d mnemonics", len(result))
563
+ if status is not None:
564
+ status(f"Extracted SDM descriptions for {len(result)} mnemonic variants")
565
+ return PdfEnrichmentResult(
566
+ descriptions=result,
567
+ fallback_page_count=fallback_pages,
568
+ stats={"mnemonic_variants": len(result)},
569
+ )
570
+
571
+
572
+ def find_intel_sdm_pdf() -> Path | None:
573
+ """Locate or download the Intel SDM PDF."""
574
+ for pdf_path in LOCAL_INTEL_SDM_PDFS:
575
+ if pdf_path.exists():
576
+ return pdf_path
577
+ try:
578
+ from rich.progress import BarColumn, DownloadColumn, Progress, TransferSpeedColumn
579
+
580
+ dest = LOCAL_INTEL_SDM_PDFS[0]
581
+ dest.parent.mkdir(parents=True, exist_ok=True)
582
+ with httpx.Client(follow_redirects=True, timeout=120.0) as client:
583
+ with client.stream("GET", INTEL_SDM_URL) as resp:
584
+ resp.raise_for_status()
585
+ total = int(resp.headers.get("content-length", 0))
586
+ with Progress(
587
+ "[progress.description]{task.description}",
588
+ BarColumn(),
589
+ DownloadColumn(),
590
+ TransferSpeedColumn(),
591
+ ) as progress:
592
+ task = progress.add_task("Downloading Intel SDM PDF", total=total or None)
593
+ with open(dest, "wb") as fh:
594
+ for chunk in resp.iter_bytes(65536):
595
+ fh.write(chunk)
596
+ progress.advance(task, len(chunk))
597
+ return dest
598
+ except Exception:
599
+ return None
600
+
601
+
602
+ INTEL_PDF_SOURCE = PdfSourceSpec(
603
+ source_id="intel-sdm",
604
+ display_name="Intel SDM",
605
+ source_url=INTEL_SDM_URL,
606
+ local_candidates=tuple(LOCAL_INTEL_SDM_PDFS),
607
+ cache_path=INTEL_SDM_CACHE_PATH,
608
+ cache_version=INTEL_SDM_CACHE_VERSION,
609
+ signature_paths=INTEL_SDM_SIGNATURE_PATHS,
610
+ parser=parse_intel_sdm,
611
+ find_source=find_intel_sdm_pdf,
612
+ )
613
+
614
+ register_pdf_source(INTEL_PDF_SOURCE)
@@ -0,0 +1,19 @@
1
+ """PDF source registry."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from simdref.pdfparse.types import PdfSourceSpec
6
+
7
+ _PDF_SOURCES: dict[str, PdfSourceSpec] = {}
8
+
9
+
10
+ def register_pdf_source(spec: PdfSourceSpec) -> None:
11
+ _PDF_SOURCES[spec.source_id] = spec
12
+
13
+
14
+ def get_pdf_source(source_id: str) -> PdfSourceSpec:
15
+ return _PDF_SOURCES[source_id]
16
+
17
+
18
+ def iter_pdf_sources() -> tuple[PdfSourceSpec, ...]:
19
+ return tuple(_PDF_SOURCES.values())