simdref 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
simdref/ingest_pdf.py ADDED
@@ -0,0 +1,188 @@
1
+ """PDF enrichment acquisition, caching, and merge helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ from functools import lru_cache
7
+ from pathlib import Path
8
+ from typing import Callable
9
+
10
+ import msgpack
11
+
12
+ from simdref.models import InstructionRecord
13
+ from simdref.pdfrefs import apply_legacy_pdf_metadata, normalize_pdf_refs
14
+ from simdref.pdfparse.registry import get_pdf_source
15
+ from simdref.pdfparse.types import PdfEnrichmentResult, PdfSourceSpec
16
+ from simdref.storage import ensure_dir
17
+
18
+
19
+ def _sha256_file(path: Path) -> str:
20
+ hasher = hashlib.sha256()
21
+ with path.open("rb") as fh:
22
+ for chunk in iter(lambda: fh.read(1024 * 1024), b""):
23
+ hasher.update(chunk)
24
+ return hasher.hexdigest()
25
+
26
+
27
+ @lru_cache(maxsize=None)
28
+ def pdf_parser_signature(source_id: str) -> str:
29
+ spec = get_pdf_source(source_id)
30
+ hasher = hashlib.sha256()
31
+ for path in spec.signature_paths:
32
+ hasher.update(path.read_bytes())
33
+ return hasher.hexdigest()
34
+
35
+
36
+ def _load_cached_pdf_source(
37
+ spec: PdfSourceSpec,
38
+ pdf_path: Path,
39
+ *,
40
+ cache_path: Path | None = None,
41
+ status: Callable[[str], None] | None = None,
42
+ ) -> PdfEnrichmentResult | None:
43
+ cache_path = cache_path or spec.cache_path
44
+ if not cache_path.exists():
45
+ return None
46
+ try:
47
+ payload = msgpack.unpackb(cache_path.read_bytes(), raw=False)
48
+ except Exception:
49
+ return None
50
+ if payload.get("cache_version") != spec.cache_version:
51
+ return None
52
+ if payload.get("parser_signature") != pdf_parser_signature(spec.source_id):
53
+ return None
54
+ if payload.get("pdf_url") != spec.source_url:
55
+ return None
56
+ if payload.get("pdf_sha256") != _sha256_file(pdf_path):
57
+ return None
58
+ result_payload = payload.get("result")
59
+ if not isinstance(result_payload, dict):
60
+ return None
61
+ result = PdfEnrichmentResult.from_dict(result_payload)
62
+ if status is not None:
63
+ status(f"Loaded cached {spec.display_name} descriptions for {len(result.descriptions)} mnemonic variants")
64
+ return result
65
+
66
+
67
+ def _save_cached_pdf_source(
68
+ spec: PdfSourceSpec,
69
+ pdf_path: Path,
70
+ result: PdfEnrichmentResult,
71
+ *,
72
+ cache_path: Path | None = None,
73
+ ) -> None:
74
+ cache_path = cache_path or spec.cache_path
75
+ ensure_dir(cache_path.parent)
76
+ payload = {
77
+ "cache_version": spec.cache_version,
78
+ "parser_signature": pdf_parser_signature(spec.source_id),
79
+ "pdf_url": spec.source_url,
80
+ "pdf_sha256": _sha256_file(pdf_path),
81
+ "result": result.to_dict(),
82
+ }
83
+ cache_path.write_bytes(msgpack.packb(payload, use_bin_type=True))
84
+
85
+
86
+ def load_or_parse_pdf_source(
87
+ source_id: str,
88
+ pdf_path: Path,
89
+ *,
90
+ cache_path: Path | None = None,
91
+ status: Callable[[str], None] | None = None,
92
+ ) -> PdfEnrichmentResult:
93
+ spec = get_pdf_source(source_id)
94
+ cached = _load_cached_pdf_source(spec, pdf_path, cache_path=cache_path, status=status)
95
+ if cached is not None:
96
+ return cached
97
+ result = spec.parser(pdf_path, status=status)
98
+ _save_cached_pdf_source(spec, pdf_path, result, cache_path=cache_path)
99
+ if status is not None:
100
+ status(f"Cached {spec.display_name} descriptions for {len(result.descriptions)} mnemonic variants")
101
+ return result
102
+
103
+
104
+ def find_pdf_source_path(source_id: str) -> Path | None:
105
+ spec = get_pdf_source(source_id)
106
+ return spec.find_source()
107
+
108
+
109
+ def merge_pdf_enrichment(
110
+ instructions: list[InstructionRecord],
111
+ source_id: str,
112
+ result: PdfEnrichmentResult,
113
+ ) -> None:
114
+ # Build a list of candidate base mnemonics from a decorated mnemonic.
115
+ _TYPE_SUFFIX_MAP = [
116
+ ("PH", "PD"), ("PH", "PS"),
117
+ ("BF16", "PS"), ("BF8", "PS"), ("BF8S", "PS"),
118
+ ("HF8", "PS"), ("HF8S", "PS"),
119
+ ("IBS", "DQ"), ("IUBS", "UDQ"),
120
+ ]
121
+ _GROUP_SUFFIXES = [
122
+ "F32X8", "F32X4", "F32X2", "F64X4", "F64X2", "F128",
123
+ "I32X8", "I32X4", "I32X2", "I64X4", "I64X2", "I128",
124
+ "MB2Q", "MW2D",
125
+ "BD", "BW", "BQ", "DQ", "WD", "WQ",
126
+ "SD", "SS", "PD", "PS",
127
+ "B", "W", "D", "Q",
128
+ "64", "32", "16", "8",
129
+ ]
130
+ _MIN_GROUP_KEY_LEN = 5
131
+
132
+ def _strip_prefix(mnemonic: str) -> str:
133
+ value = mnemonic
134
+ while value.startswith("{"):
135
+ end = value.find("}")
136
+ if end == -1:
137
+ break
138
+ value = value[end + 1:].lstrip()
139
+ for prefix in ("LOCK ", "REPE ", "REPNE ", "REP ", "REX64 "):
140
+ if value.startswith(prefix):
141
+ return value[len(prefix):]
142
+ return value
143
+
144
+ def _base_candidates(mnemonic: str) -> list[str]:
145
+ candidates: list[str] = []
146
+ bare = _strip_prefix(mnemonic)
147
+ if bare != mnemonic:
148
+ candidates.append(bare)
149
+ if bare.startswith("V") and len(bare) > 1:
150
+ candidates.append(bare[1:])
151
+ for candidate in list(candidates):
152
+ if candidate.startswith("V") and len(candidate) > 1 and candidate[1:] not in candidates:
153
+ candidates.append(candidate[1:])
154
+ all_forms = [mnemonic, bare] + candidates
155
+ for form in list(all_forms):
156
+ if form.endswith("S") and len(form) > 3:
157
+ candidates.append(form[:-1])
158
+ for old_suffix, new_suffix in _TYPE_SUFFIX_MAP:
159
+ if form.endswith(old_suffix):
160
+ candidates.append(form[: -len(old_suffix)] + new_suffix)
161
+ for form in list(all_forms):
162
+ for suffix in _GROUP_SUFFIXES:
163
+ if form.endswith(suffix):
164
+ stem = form[: -len(suffix)]
165
+ if len(stem) >= _MIN_GROUP_KEY_LEN and stem not in candidates:
166
+ candidates.append(stem)
167
+ return candidates
168
+
169
+ for record in instructions:
170
+ mnemonic = record.mnemonic.upper()
171
+ payload = result.descriptions.get(mnemonic)
172
+ if payload is None:
173
+ for candidate in _base_candidates(mnemonic):
174
+ payload = result.descriptions.get(candidate)
175
+ if payload is not None:
176
+ break
177
+ if payload is None:
178
+ continue
179
+ record.description = dict(payload.sections)
180
+ pdf_ref = {
181
+ "source_id": source_id,
182
+ "label": get_pdf_source(source_id).display_name,
183
+ "url": f"{payload.source_url}#page={payload.page_start}" if payload.page_start else payload.source_url,
184
+ "page_start": str(payload.page_start or ""),
185
+ "page_end": str(payload.page_end or ""),
186
+ }
187
+ record.pdf_refs = normalize_pdf_refs([*record.pdf_refs, pdf_ref], record.metadata)
188
+ record.metadata = apply_legacy_pdf_metadata(dict(record.metadata), record.pdf_refs)