simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1172 @@
|
|
|
1
|
+
"""Catalog parsing, linking, and assembly."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import html
|
|
7
|
+
import io
|
|
8
|
+
import json
|
|
9
|
+
import re
|
|
10
|
+
import sys
|
|
11
|
+
import xml.etree.ElementTree as ET
|
|
12
|
+
from functools import lru_cache
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any, Callable
|
|
15
|
+
from urllib.parse import quote, urljoin
|
|
16
|
+
|
|
17
|
+
from simdref.arm_instructions import parse_arm_instruction_payload
|
|
18
|
+
from simdref.ingest_pdf import find_pdf_source_path, load_or_parse_pdf_source, merge_pdf_enrichment
|
|
19
|
+
from simdref.ingest_sources import (
|
|
20
|
+
fetch_arm_a64_data,
|
|
21
|
+
fetch_arm_acle_data,
|
|
22
|
+
fetch_intel_data,
|
|
23
|
+
fetch_riscv_rvv_intrinsics_data,
|
|
24
|
+
fetch_riscv_unified_db_data,
|
|
25
|
+
fetch_uops_xml,
|
|
26
|
+
now_iso,
|
|
27
|
+
)
|
|
28
|
+
from simdref.models import Catalog, InstructionRecord, IntrinsicRecord, SourceVersion
|
|
29
|
+
from simdref.riscv import parse_riscv_instruction_payload, parse_riscv_intrinsics_payload
|
|
30
|
+
|
|
31
|
+
_UOPS_METADATA_KEYS = frozenset({"category", "cpl", "extension", "iclass", "iform", "url", "url-ref"})
|
|
32
|
+
_UOPS_OPERAND_KEYS = ("idx", "r", "w", "type", "width", "xtype", "name")
|
|
33
|
+
_ARM_ACLE_INTRINSIC_BASE_URL = "https://developer.arm.com/architectures/instruction-sets/intrinsics/"
|
|
34
|
+
_ARM_NEON_REFERENCE_URL = "https://arm-software.github.io/acle/neon_intrinsics/advsimd.html"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _iter_xml_elements(source: str | Path, tag: str):
|
|
38
|
+
context = ET.iterparse(io.StringIO(source) if isinstance(source, str) else source, events=("start", "end"))
|
|
39
|
+
root = None
|
|
40
|
+
for event, elem in context:
|
|
41
|
+
if event == "start" and root is None:
|
|
42
|
+
root = elem
|
|
43
|
+
continue
|
|
44
|
+
if event != "end" or elem.tag != tag:
|
|
45
|
+
continue
|
|
46
|
+
yield elem
|
|
47
|
+
elem.clear()
|
|
48
|
+
if root is not None:
|
|
49
|
+
root.clear()
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _intern_small(value: str) -> str:
|
|
53
|
+
return sys.intern(value) if value and len(value) <= 32 else value
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _normalize_isa(value: Any) -> list[str]:
|
|
57
|
+
if isinstance(value, list):
|
|
58
|
+
return [_intern_small(str(item)) for item in value if item]
|
|
59
|
+
if isinstance(value, str):
|
|
60
|
+
return list(_normalize_isa_string(value))
|
|
61
|
+
return []
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@lru_cache(maxsize=512)
|
|
65
|
+
def _normalize_isa_string(value: str) -> tuple[str, ...]:
|
|
66
|
+
parts = re.split(r"[,/|]\s*|\s{2,}", value)
|
|
67
|
+
return tuple(_intern_small(part.strip()) for part in parts if part.strip())
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _canonical_instruction_key(name: str, form: str) -> str:
|
|
71
|
+
instruction_name = name.strip().upper()
|
|
72
|
+
instruction_form = re.sub(r"\s*,\s*", ", ", form.strip())
|
|
73
|
+
instruction_form = re.sub(r"\s+", " ", instruction_form)
|
|
74
|
+
if not instruction_name:
|
|
75
|
+
return ""
|
|
76
|
+
if not instruction_form:
|
|
77
|
+
return instruction_name
|
|
78
|
+
return f"{instruction_name} ({instruction_form.upper()})"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@lru_cache(maxsize=128)
|
|
82
|
+
def _normalize_operand_xtype(value: str) -> str:
|
|
83
|
+
xtype = value.strip()
|
|
84
|
+
if not xtype:
|
|
85
|
+
return xtype
|
|
86
|
+
if xtype == "int":
|
|
87
|
+
return "i32"
|
|
88
|
+
match = re.fullmatch(r"\d+([iu]\d+)", xtype)
|
|
89
|
+
if match:
|
|
90
|
+
return match.group(1)
|
|
91
|
+
return xtype
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _summary_too_terse(summary: str, mnemonic: str) -> bool:
|
|
95
|
+
text = summary.strip().strip(".")
|
|
96
|
+
if not text:
|
|
97
|
+
return True
|
|
98
|
+
words = re.findall(r"[A-Za-z0-9+\-]+", text)
|
|
99
|
+
if len(words) <= 1:
|
|
100
|
+
return True
|
|
101
|
+
if text.casefold() == mnemonic.casefold():
|
|
102
|
+
return True
|
|
103
|
+
if len(words) == 2 and words[0].casefold() == words[1].casefold() == mnemonic.casefold():
|
|
104
|
+
return True
|
|
105
|
+
return False
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _summary_prefix(mnemonic: str, operand_details: list[dict[str, str]], summary: str) -> str:
|
|
109
|
+
upper = mnemonic.upper()
|
|
110
|
+
lowered = summary.casefold()
|
|
111
|
+
has_mask_operand = any(op.get("xtype") == "i1" for op in operand_details)
|
|
112
|
+
if upper.endswith("_Z") and "zero-mask" not in lowered and "zeromask" not in lowered and "zeroing" not in lowered:
|
|
113
|
+
return "Zero-masked "
|
|
114
|
+
if has_mask_operand and "mask" not in lowered:
|
|
115
|
+
return "Masked "
|
|
116
|
+
return ""
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _operand_kind(op: dict[str, str]) -> str:
|
|
120
|
+
kind = op.get("type", "").strip().lower()
|
|
121
|
+
if kind == "reg":
|
|
122
|
+
return "register"
|
|
123
|
+
if kind == "mem":
|
|
124
|
+
return "memory"
|
|
125
|
+
if kind == "imm":
|
|
126
|
+
return "immediate"
|
|
127
|
+
if kind == "agen":
|
|
128
|
+
return "address"
|
|
129
|
+
if kind == "flags":
|
|
130
|
+
return "flags"
|
|
131
|
+
return kind or "operand"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _element_type_phrase(xtype: str, width: str) -> str:
|
|
135
|
+
xtype = xtype.strip().lower()
|
|
136
|
+
mapping = {
|
|
137
|
+
"f16": "FP16",
|
|
138
|
+
"f32": "single-precision floating-point",
|
|
139
|
+
"f64": "double-precision floating-point",
|
|
140
|
+
"i1": "mask",
|
|
141
|
+
"i8": "8-bit integer",
|
|
142
|
+
"u8": "8-bit integer",
|
|
143
|
+
"i16": "16-bit integer",
|
|
144
|
+
"u16": "16-bit integer",
|
|
145
|
+
"i32": "32-bit integer",
|
|
146
|
+
"u32": "32-bit integer",
|
|
147
|
+
"i64": "64-bit integer",
|
|
148
|
+
"u64": "64-bit integer",
|
|
149
|
+
"i128": "128-bit integer",
|
|
150
|
+
"u128": "128-bit integer",
|
|
151
|
+
}
|
|
152
|
+
if xtype in mapping:
|
|
153
|
+
return mapping[xtype]
|
|
154
|
+
if width.isdigit() and xtype.startswith(("i", "u")):
|
|
155
|
+
return f"{width}-bit integer"
|
|
156
|
+
return ""
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _shared_operand_phrase(operand_details: list[dict[str, str]]) -> str:
|
|
160
|
+
semantic_ops = [op for op in operand_details if _operand_kind(op) in {"register", "memory", "immediate"}]
|
|
161
|
+
if not semantic_ops:
|
|
162
|
+
return "operands"
|
|
163
|
+
kinds = {_operand_kind(op) for op in semantic_ops}
|
|
164
|
+
xtypes = {op.get("xtype", "").strip().lower() for op in semantic_ops if op.get("xtype", "").strip().lower() and op.get("xtype", "").strip().lower() != "i1"}
|
|
165
|
+
widths = {op.get("width", "").strip() for op in semantic_ops if op.get("width", "").strip()}
|
|
166
|
+
if len(xtypes) == 1:
|
|
167
|
+
phrase = _element_type_phrase(next(iter(xtypes)), next(iter(widths), ""))
|
|
168
|
+
if phrase:
|
|
169
|
+
return "mask operands" if phrase == "mask" else f"{phrase} operands"
|
|
170
|
+
if len(widths) == 1:
|
|
171
|
+
width = next(iter(widths))
|
|
172
|
+
if width.isdigit():
|
|
173
|
+
return f"{width}-bit operands"
|
|
174
|
+
if kinds == {"register"}:
|
|
175
|
+
return "register operands"
|
|
176
|
+
if kinds == {"memory"}:
|
|
177
|
+
return "memory operands"
|
|
178
|
+
if kinds == {"immediate"}:
|
|
179
|
+
return "immediate operands"
|
|
180
|
+
if kinds == {"register", "memory"}:
|
|
181
|
+
return "register and memory operands"
|
|
182
|
+
if kinds == {"register", "immediate"}:
|
|
183
|
+
return "register and immediate operands"
|
|
184
|
+
if kinds == {"memory", "immediate"}:
|
|
185
|
+
return "memory and immediate operands"
|
|
186
|
+
if kinds == {"register", "memory", "immediate"}:
|
|
187
|
+
return "register, memory, and immediate operands"
|
|
188
|
+
return "operands"
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _verb_for_mnemonic(mnemonic: str, architecture: str = "x86") -> str:
|
|
192
|
+
core = mnemonic.upper()
|
|
193
|
+
for suffix in ("_ER_Z", "_ER", "_Z"):
|
|
194
|
+
if core.endswith(suffix):
|
|
195
|
+
core = core[: -len(suffix)]
|
|
196
|
+
break
|
|
197
|
+
if architecture == "x86" and core.startswith("V") and len(core) > 3:
|
|
198
|
+
core = core[1:]
|
|
199
|
+
for key, verb in [
|
|
200
|
+
("ADC", "Add with carry"), ("ADD", "Add"), ("SUB", "Subtract"), ("SBB", "Subtract with borrow"),
|
|
201
|
+
("MUL", "Multiply"), ("IMUL", "Multiply"), ("DIV", "Divide"), ("IDIV", "Divide"),
|
|
202
|
+
("MOV", "Move"), ("CMP", "Compare"), ("AND", "Bitwise AND"), ("OR", "Bitwise OR"),
|
|
203
|
+
("XOR", "Bitwise XOR"), ("TEST", "Test"), ("MIN", "Compute minimum of"), ("MAX", "Compute maximum of"),
|
|
204
|
+
("BLEND", "Blend"), ("EXPAND", "Expand"), ("LOAD", "Load"), ("STORE", "Store"),
|
|
205
|
+
("SHUFFLE", "Shuffle"), ("PERM", "Permute"),
|
|
206
|
+
]:
|
|
207
|
+
if core.startswith(key):
|
|
208
|
+
return verb
|
|
209
|
+
return core.replace("_", " ").title()
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _generated_instruction_summary(mnemonic: str, operand_details: list[dict[str, str]], architecture: str = "x86") -> str:
|
|
213
|
+
return f"{_verb_for_mnemonic(mnemonic, architecture=architecture)} {_shared_operand_phrase(operand_details)}".strip()
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _instruction_summary(mnemonic: str, raw_summary: str, operand_details: list[dict[str, str]], architecture: str = "x86") -> str:
|
|
217
|
+
base = raw_summary.strip().strip(".")
|
|
218
|
+
prefix = _summary_prefix(mnemonic, operand_details, base)
|
|
219
|
+
if not _summary_too_terse(base, mnemonic):
|
|
220
|
+
return f"{prefix}{base}".strip() + "."
|
|
221
|
+
return f"{prefix}{_generated_instruction_summary(mnemonic, operand_details, architecture=architecture)}.".strip()
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def parse_intel_payload(text: str) -> list[IntrinsicRecord]:
|
|
225
|
+
stripped = text.strip()
|
|
226
|
+
if stripped.startswith("var data_js"):
|
|
227
|
+
match = re.search(r'var\s+data_js\s*=\s*"(?P<body>.*)";\s*$', stripped, re.DOTALL)
|
|
228
|
+
if not match:
|
|
229
|
+
raise ValueError("could not locate Intel XML payload in data.js")
|
|
230
|
+
xml_blob = match.group("body").replace("\\\n", "")
|
|
231
|
+
stripped = bytes(xml_blob, "utf-8").decode("unicode_escape").strip()
|
|
232
|
+
if stripped.startswith("<?xml") or stripped.startswith("<intrinsics_list"):
|
|
233
|
+
records: list[IntrinsicRecord] = []
|
|
234
|
+
for node in _iter_xml_elements(stripped, "intrinsic"):
|
|
235
|
+
name = node.attrib.get("name", "").strip()
|
|
236
|
+
if not name:
|
|
237
|
+
continue
|
|
238
|
+
return_node = node.find("./return")
|
|
239
|
+
ret = node.attrib.get("rettype", "").strip() or (return_node.attrib.get("type", "").strip() if return_node is not None else "") or "void"
|
|
240
|
+
params = []
|
|
241
|
+
instruction_refs: list[dict[str, str]] = []
|
|
242
|
+
for param in node.findall("./parameter"):
|
|
243
|
+
ptype = param.attrib.get("type", "").strip()
|
|
244
|
+
pname = param.attrib.get("varname", "").strip()
|
|
245
|
+
params.append(" ".join(part for part in [ptype, pname] if part))
|
|
246
|
+
description = " ".join((part.text or "").strip() for part in node.findall("./description") if (part.text or "").strip())
|
|
247
|
+
cpuid = [cpuid.text.strip() for cpuid in node.findall("./CPUID") if (cpuid.text or "").strip()]
|
|
248
|
+
notes: list[str] = []
|
|
249
|
+
if node.attrib.get("sequence"):
|
|
250
|
+
notes.append(node.attrib["sequence"].strip())
|
|
251
|
+
for key in ("sequence", "sequence_note"):
|
|
252
|
+
child = node.find(f"./{key}")
|
|
253
|
+
if child is not None and (child.text or "").strip():
|
|
254
|
+
notes.append(child.text.strip())
|
|
255
|
+
instructions: list[str] = []
|
|
256
|
+
for inst in node.findall("./instruction"):
|
|
257
|
+
inst_name = inst.attrib.get("name", "").strip()
|
|
258
|
+
inst_form = inst.attrib.get("form", "").strip()
|
|
259
|
+
inst_xed = inst.attrib.get("xed", "").strip()
|
|
260
|
+
if not inst_name:
|
|
261
|
+
continue
|
|
262
|
+
instruction_refs.append({"name": inst_name, "form": inst_form, "xed": inst_xed})
|
|
263
|
+
instructions.append(_canonical_instruction_key(inst_name, inst_form) or inst_name)
|
|
264
|
+
records.append(
|
|
265
|
+
IntrinsicRecord(
|
|
266
|
+
name=name,
|
|
267
|
+
signature=f"{ret} {name}({', '.join(params)})",
|
|
268
|
+
description=description,
|
|
269
|
+
header=((node.findtext("./header") or "").strip() or node.attrib.get("header", "")),
|
|
270
|
+
architecture="x86",
|
|
271
|
+
isa=_normalize_isa(cpuid or node.attrib.get("isa", "") or node.attrib.get("tech", "")),
|
|
272
|
+
category=((node.findtext("./category") or "").strip() or node.attrib.get("category", "")),
|
|
273
|
+
subcategory=node.attrib.get("tech", "").strip(),
|
|
274
|
+
instructions=instructions,
|
|
275
|
+
instruction_refs=[ref | {"architecture": "x86"} for ref in instruction_refs],
|
|
276
|
+
notes=notes,
|
|
277
|
+
aliases=[],
|
|
278
|
+
)
|
|
279
|
+
)
|
|
280
|
+
return records
|
|
281
|
+
|
|
282
|
+
json_blob = stripped
|
|
283
|
+
if stripped.startswith("var ") or stripped.startswith("window."):
|
|
284
|
+
match = re.search(r"(\{.*\}|\[.*\])", stripped, re.DOTALL)
|
|
285
|
+
if not match:
|
|
286
|
+
raise ValueError("could not locate JSON payload in Intel data")
|
|
287
|
+
json_blob = match.group(1)
|
|
288
|
+
|
|
289
|
+
payload = json.loads(json_blob)
|
|
290
|
+
candidates = payload.get("intrinsics") or payload.get("data") or payload.get("records") or [] if isinstance(payload, dict) else payload
|
|
291
|
+
records: list[IntrinsicRecord] = []
|
|
292
|
+
for item in candidates:
|
|
293
|
+
name = str(item.get("name") or item.get("intrinsic") or "").strip()
|
|
294
|
+
if not name:
|
|
295
|
+
continue
|
|
296
|
+
signature = str(item.get("signature") or item.get("prototype") or "").strip()
|
|
297
|
+
if not signature:
|
|
298
|
+
return_type = str(item.get("returnType") or item.get("rettype") or "void").strip()
|
|
299
|
+
params = item.get("parameters") or item.get("params") or []
|
|
300
|
+
rendered_params = []
|
|
301
|
+
if isinstance(params, list):
|
|
302
|
+
for param in params:
|
|
303
|
+
if isinstance(param, dict):
|
|
304
|
+
ptype = str(param.get("type", "")).strip()
|
|
305
|
+
pname = str(param.get("name") or param.get("varname") or "").strip()
|
|
306
|
+
rendered_params.append(" ".join(part for part in [ptype, pname] if part))
|
|
307
|
+
else:
|
|
308
|
+
rendered_params.append(str(param).strip())
|
|
309
|
+
signature = f"{return_type} {name}({', '.join(p for p in rendered_params if p)})"
|
|
310
|
+
instructions = item.get("instructions") or item.get("instruction") or item.get("Instruction") or []
|
|
311
|
+
if isinstance(instructions, str):
|
|
312
|
+
instructions = [instructions]
|
|
313
|
+
notes = item.get("notes") or item.get("operationNotes") or []
|
|
314
|
+
if isinstance(notes, str):
|
|
315
|
+
notes = [notes]
|
|
316
|
+
aliases = item.get("aliases") or []
|
|
317
|
+
if isinstance(aliases, str):
|
|
318
|
+
aliases = [aliases]
|
|
319
|
+
description = str(item.get("description") or item.get("summary") or item.get("technology", "")).strip()
|
|
320
|
+
records.append(
|
|
321
|
+
IntrinsicRecord(
|
|
322
|
+
name=name,
|
|
323
|
+
signature=signature,
|
|
324
|
+
description=description,
|
|
325
|
+
header=str(item.get("header") or item.get("include") or "").strip(),
|
|
326
|
+
architecture="x86",
|
|
327
|
+
isa=_normalize_isa(item.get("isa") or item.get("tech") or item.get("instructionSet") or []),
|
|
328
|
+
category=str(item.get("category") or "").strip(),
|
|
329
|
+
subcategory=str(item.get("tech") or item.get("subcategory") or "").strip(),
|
|
330
|
+
instructions=[str(value).strip() for value in instructions if str(value).strip()],
|
|
331
|
+
instruction_refs=[{"name": str(value).strip(), "form": "", "xed": "", "architecture": "x86"} for value in instructions if str(value).strip()],
|
|
332
|
+
notes=[str(value).strip() for value in notes if str(value).strip()],
|
|
333
|
+
aliases=[str(value).strip() for value in aliases if str(value).strip()],
|
|
334
|
+
)
|
|
335
|
+
)
|
|
336
|
+
return records
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def parse_arm_intrinsics_payload(text: str) -> list[IntrinsicRecord]:
|
|
340
|
+
stripped = text.strip()
|
|
341
|
+
if stripped.startswith("{"):
|
|
342
|
+
payload = json.loads(text)
|
|
343
|
+
if payload.get("format") == "arm-intrinsics-json-v1":
|
|
344
|
+
return parse_arm_intrinsics_json_bundle(payload)
|
|
345
|
+
if payload.get("format") == "acle-neon-csv-v1":
|
|
346
|
+
return parse_arm_neon_intrinsics_bundle(payload)
|
|
347
|
+
payload = json.loads(text)
|
|
348
|
+
candidates = payload.get("intrinsics") if isinstance(payload, dict) else payload
|
|
349
|
+
records: list[IntrinsicRecord] = []
|
|
350
|
+
for item in candidates or []:
|
|
351
|
+
name = str(item.get("name") or "").strip()
|
|
352
|
+
if not name:
|
|
353
|
+
continue
|
|
354
|
+
instruction_refs = []
|
|
355
|
+
instructions = []
|
|
356
|
+
for ref in item.get("instruction_refs") or item.get("instructions") or []:
|
|
357
|
+
if isinstance(ref, str):
|
|
358
|
+
ref = {"name": ref}
|
|
359
|
+
ref_name = str(ref.get("name") or "").strip()
|
|
360
|
+
ref_form = str(ref.get("form") or "").strip()
|
|
361
|
+
if not ref_name:
|
|
362
|
+
continue
|
|
363
|
+
rendered = _canonical_instruction_key(ref_name, ref_form) or ref_name
|
|
364
|
+
instructions.append(rendered)
|
|
365
|
+
instruction_refs.append({
|
|
366
|
+
"name": ref_name,
|
|
367
|
+
"form": ref_form,
|
|
368
|
+
"architecture": "arm",
|
|
369
|
+
})
|
|
370
|
+
records.append(
|
|
371
|
+
IntrinsicRecord(
|
|
372
|
+
name=name,
|
|
373
|
+
signature=str(item.get("signature") or "").strip(),
|
|
374
|
+
description=str(item.get("description") or "").strip(),
|
|
375
|
+
header=str(item.get("header") or "").strip(),
|
|
376
|
+
url=str(item.get("url") or "").strip(),
|
|
377
|
+
architecture="arm",
|
|
378
|
+
isa=_normalize_isa(item.get("isa") or []),
|
|
379
|
+
category=str(item.get("category") or "").strip(),
|
|
380
|
+
subcategory=str(item.get("subcategory") or item.get("group") or "").strip(),
|
|
381
|
+
instructions=instructions,
|
|
382
|
+
instruction_refs=instruction_refs,
|
|
383
|
+
metadata={
|
|
384
|
+
str(key): str(value).strip()
|
|
385
|
+
for key, value in (item.get("metadata") or {}).items()
|
|
386
|
+
if str(value).strip()
|
|
387
|
+
},
|
|
388
|
+
notes=[str(value).strip() for value in item.get("notes") or [] if str(value).strip()],
|
|
389
|
+
aliases=[str(value).strip() for value in item.get("aliases") or [] if str(value).strip()],
|
|
390
|
+
source="arm-acle",
|
|
391
|
+
)
|
|
392
|
+
)
|
|
393
|
+
return records
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _arm_intrinsic_url(name: str) -> str:
|
|
397
|
+
return urljoin(_ARM_ACLE_INTRINSIC_BASE_URL, quote(name))
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _arm_slug(text: str) -> str:
|
|
401
|
+
slug = re.sub(r"[^a-z0-9]+", "-", text.casefold()).strip("-")
|
|
402
|
+
return slug or "intrinsics"
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _parse_c_signature(signature: str) -> tuple[str, str, list[str]]:
|
|
406
|
+
match = re.match(r"^(?P<ret>.+?)\s+(?P<name>[A-Za-z0-9_]+)\((?P<params>.*)\)$", signature.strip())
|
|
407
|
+
if not match:
|
|
408
|
+
raise ValueError(f"could not parse C signature: {signature}")
|
|
409
|
+
params = [part.strip() for part in match.group("params").split(",") if part.strip() and part.strip() != "void"]
|
|
410
|
+
return match.group("ret").strip(), match.group("name").strip(), params
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _arm_instruction_refs_from_field(value: str) -> list[dict[str, str]]:
|
|
414
|
+
refs: list[dict[str, str]] = []
|
|
415
|
+
for part in [chunk.strip() for chunk in value.split(";") if chunk.strip()]:
|
|
416
|
+
pieces = part.split(None, 1)
|
|
417
|
+
name = pieces[0].strip()
|
|
418
|
+
form = pieces[1].strip() if len(pieces) > 1 else ""
|
|
419
|
+
refs.append({"name": name, "form": form, "architecture": "arm"})
|
|
420
|
+
return refs
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _normalize_arm_intrinsic_name(name: str) -> str:
|
|
424
|
+
return name.strip().replace("[_", "_").replace("]", "")
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def _strip_markdown_html(value: str) -> str:
|
|
428
|
+
text = value.replace("<br>", "\n").replace("<br/>", "\n").replace("<br />", "\n")
|
|
429
|
+
text = re.sub(r"</?(?:code|strong|em|span|p|div)[^>]*>", "", text)
|
|
430
|
+
text = re.sub(r"<a [^>]*>(.*?)</a>", r"\1", text)
|
|
431
|
+
text = html.unescape(re.sub(r" ", " ", text))
|
|
432
|
+
text = re.sub(r"\s+\n", "\n", text)
|
|
433
|
+
return text.strip()
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def _arm_html_to_text(value: str) -> str:
|
|
437
|
+
text = value.replace("<br>", "\n").replace("<br/>", "\n").replace("<br />", "\n")
|
|
438
|
+
text = text.replace("</p>", "\n\n").replace("</pre>", "\n").replace("</h4>", "\n")
|
|
439
|
+
text = re.sub(r"<a [^>]*>(.*?)</a>", r"\1", text)
|
|
440
|
+
text = re.sub(r"</?(?:pre|p|code|strong|em|span|div|h4|ul|ol|li|b|i)[^>]*>", "", text)
|
|
441
|
+
text = html.unescape(text)
|
|
442
|
+
text = re.sub(r"\n{3,}", "\n\n", text)
|
|
443
|
+
return text.strip()
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def _arm_live_instruction_refs(groups: list[dict[str, Any]]) -> list[dict[str, str]]:
|
|
447
|
+
refs: list[dict[str, str]] = []
|
|
448
|
+
for group in groups:
|
|
449
|
+
for entry in group.get("list") or []:
|
|
450
|
+
name = str(entry.get("base_instruction") or "").strip()
|
|
451
|
+
operands = str(entry.get("operands") or "").strip()
|
|
452
|
+
if name:
|
|
453
|
+
refs.append({"name": name, "form": operands, "architecture": "arm"})
|
|
454
|
+
return refs
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def _arm_live_operation_sections(operation_id: str, operations: dict[str, dict[str, Any]]) -> dict[str, str]:
|
|
458
|
+
content = str((operations.get(operation_id) or {}).get("content") or "").strip()
|
|
459
|
+
if not content:
|
|
460
|
+
return {}
|
|
461
|
+
text = _arm_html_to_text(content)
|
|
462
|
+
if text.startswith("Operation\n"):
|
|
463
|
+
text = text[len("Operation\n"):].strip()
|
|
464
|
+
return {"ACLE Operation": text} if text else {}
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _arm_live_notes(item: dict[str, Any]) -> tuple[list[str], dict[str, str]]:
|
|
468
|
+
notes: list[str] = []
|
|
469
|
+
sections: dict[str, str] = {}
|
|
470
|
+
required = item.get("required_streaming_features") or {}
|
|
471
|
+
if required:
|
|
472
|
+
intro = _arm_html_to_text(str(required.get("intro") or ""))
|
|
473
|
+
features = str(required.get("features") or "").strip()
|
|
474
|
+
body = "\n".join(part for part in [intro, f"Features: {features}" if features else ""] if part)
|
|
475
|
+
if body:
|
|
476
|
+
sections[str(required.get("title") or "Required streaming features")] = body
|
|
477
|
+
notes.append("Requires streaming features")
|
|
478
|
+
sme_modes = [str(value).strip() for value in item.get("sme_modes") or [] if str(value).strip()]
|
|
479
|
+
if sme_modes:
|
|
480
|
+
sections["Required keyword attributes"] = "\n".join(sme_modes)
|
|
481
|
+
return notes, sections
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def parse_arm_intrinsics_json_bundle(payload: dict[str, Any]) -> list[IntrinsicRecord]:
|
|
485
|
+
intrinsics = json.loads(str(payload.get("intrinsics_json") or "[]"))
|
|
486
|
+
operations_payload = json.loads(str(payload.get("operations_json") or "[]"))
|
|
487
|
+
operations = {
|
|
488
|
+
str(item.get("item", {}).get("id") or ""): item.get("item", {})
|
|
489
|
+
for item in operations_payload
|
|
490
|
+
if isinstance(item, dict) and item.get("item")
|
|
491
|
+
}
|
|
492
|
+
records: list[IntrinsicRecord] = []
|
|
493
|
+
for item in intrinsics:
|
|
494
|
+
raw_name = str(item.get("name") or "").strip()
|
|
495
|
+
if not raw_name:
|
|
496
|
+
continue
|
|
497
|
+
name = _normalize_arm_intrinsic_name(raw_name)
|
|
498
|
+
args = [str(value).strip() for value in item.get("arguments") or [] if str(value).strip()]
|
|
499
|
+
group_path = [part.strip() for part in str(item.get("instruction_group") or "").split("|") if part.strip()]
|
|
500
|
+
category = group_path[-1] if group_path else ""
|
|
501
|
+
subcategory = " / ".join(group_path[:-1])
|
|
502
|
+
simd_isa = [str(value).strip() for value in item.get("SIMD_ISA") or [] if str(value).strip()]
|
|
503
|
+
isa: list[str] = []
|
|
504
|
+
if "Neon" in simd_isa:
|
|
505
|
+
isa.append("NEON")
|
|
506
|
+
if "SVE" in simd_isa:
|
|
507
|
+
isa.append("SVE")
|
|
508
|
+
if "SVE2" in simd_isa:
|
|
509
|
+
isa.append("SVE2")
|
|
510
|
+
if not isa:
|
|
511
|
+
continue
|
|
512
|
+
instruction_groups = [group for group in (item.get("instructions") or []) if isinstance(group, dict)]
|
|
513
|
+
instruction_refs = _arm_live_instruction_refs(instruction_groups)
|
|
514
|
+
docs: dict[str, str] = {}
|
|
515
|
+
for group in instruction_groups:
|
|
516
|
+
preamble = str(group.get("preamble") or "").strip()
|
|
517
|
+
entries = []
|
|
518
|
+
for entry in group.get("list") or []:
|
|
519
|
+
base_instruction = str(entry.get("base_instruction") or "").strip()
|
|
520
|
+
operands = str(entry.get("operands") or "").strip()
|
|
521
|
+
url = str(entry.get("url") or "").strip()
|
|
522
|
+
rendered = " ".join(part for part in [base_instruction, operands] if part).strip()
|
|
523
|
+
if url:
|
|
524
|
+
rendered = f"{rendered}\nURL: {url}" if rendered else f"URL: {url}"
|
|
525
|
+
if rendered:
|
|
526
|
+
entries.append(rendered)
|
|
527
|
+
if preamble and entries:
|
|
528
|
+
docs[preamble] = "\n\n".join(entries)
|
|
529
|
+
docs.update(_arm_live_operation_sections(str(item.get("Operation") or "").strip(), operations))
|
|
530
|
+
notes, extra_sections = _arm_live_notes(item)
|
|
531
|
+
docs.update(extra_sections)
|
|
532
|
+
arg_prep = item.get("Arguments_Preparation") or {}
|
|
533
|
+
arg_prep_text = ";".join(
|
|
534
|
+
f"{arg} -> {', '.join(f'{key} {value}' for key, value in mapping.items())}"
|
|
535
|
+
for arg, mapping in arg_prep.items()
|
|
536
|
+
if isinstance(mapping, dict)
|
|
537
|
+
)
|
|
538
|
+
result_text = ";".join(f"{key} -> {value}" for row in item.get("results") or [] for key, value in row.items())
|
|
539
|
+
records.append(
|
|
540
|
+
IntrinsicRecord(
|
|
541
|
+
name=name,
|
|
542
|
+
signature=f"{str((item.get('return_type') or {}).get('value') or '').strip()} {name}({', '.join(args)})".strip(),
|
|
543
|
+
description=str(item.get("description") or "").strip(),
|
|
544
|
+
header="arm_neon.h" if "NEON" in isa else "arm_sve.h",
|
|
545
|
+
url=_arm_intrinsic_url(name),
|
|
546
|
+
architecture="arm",
|
|
547
|
+
isa=isa,
|
|
548
|
+
category=category,
|
|
549
|
+
subcategory=subcategory,
|
|
550
|
+
instructions=[_canonical_instruction_key(ref["name"], ref["form"]) or ref["name"] for ref in instruction_refs],
|
|
551
|
+
instruction_refs=instruction_refs,
|
|
552
|
+
metadata={
|
|
553
|
+
"argument_preparation": arg_prep_text,
|
|
554
|
+
"result": result_text,
|
|
555
|
+
"supported_architectures": "/".join(str(value).strip() for value in item.get("Architectures") or [] if str(value).strip()),
|
|
556
|
+
"reference_url": str((instruction_groups[0].get("list") or [{}])[0].get("url") or "").strip() if instruction_groups else "",
|
|
557
|
+
"classification_path": " / ".join(group_path),
|
|
558
|
+
"operation_id": str(item.get("Operation") or "").strip(),
|
|
559
|
+
"simd_isa": ", ".join(simd_isa),
|
|
560
|
+
},
|
|
561
|
+
doc_sections=docs,
|
|
562
|
+
notes=notes,
|
|
563
|
+
source="arm-intrinsics-site",
|
|
564
|
+
)
|
|
565
|
+
)
|
|
566
|
+
return records
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
def _parse_neon_markdown_docs(markdown: str) -> dict[str, dict[str, str]]:
|
|
570
|
+
docs: dict[str, dict[str, str]] = {}
|
|
571
|
+
headings: list[tuple[int, str]] = []
|
|
572
|
+
for raw_line in markdown.splitlines():
|
|
573
|
+
line = raw_line.strip()
|
|
574
|
+
if not line:
|
|
575
|
+
continue
|
|
576
|
+
if line.startswith("#"):
|
|
577
|
+
level = len(line) - len(line.lstrip("#"))
|
|
578
|
+
title = line[level:].strip()
|
|
579
|
+
headings = [(lvl, text) for lvl, text in headings if lvl < level]
|
|
580
|
+
headings.append((level, title))
|
|
581
|
+
continue
|
|
582
|
+
if not line.startswith("| <code>"):
|
|
583
|
+
continue
|
|
584
|
+
cells = [cell.strip() for cell in raw_line.strip().split("|")[1:-1]]
|
|
585
|
+
if len(cells) < 5:
|
|
586
|
+
continue
|
|
587
|
+
name_match = re.search(r">([A-Za-z0-9_]+)</a>", cells[0])
|
|
588
|
+
if not name_match:
|
|
589
|
+
continue
|
|
590
|
+
name = name_match.group(1)
|
|
591
|
+
section_path = " / ".join(text for level, text in headings if level >= 4)
|
|
592
|
+
docs[name] = {
|
|
593
|
+
"ACLE Documentation": "\n".join(
|
|
594
|
+
part
|
|
595
|
+
for part in [
|
|
596
|
+
f"Section: {section_path}" if section_path else "",
|
|
597
|
+
f"Intrinsic: {_strip_markdown_html(cells[0])}",
|
|
598
|
+
f"Argument preparation:\n{_strip_markdown_html(cells[1])}",
|
|
599
|
+
f"AArch64 instruction:\n{_strip_markdown_html(cells[2])}",
|
|
600
|
+
f"Result:\n{_strip_markdown_html(cells[3])}" if _strip_markdown_html(cells[3]) else "",
|
|
601
|
+
f"Supported architectures:\n{_strip_markdown_html(cells[4])}",
|
|
602
|
+
]
|
|
603
|
+
if part
|
|
604
|
+
)
|
|
605
|
+
}
|
|
606
|
+
return docs
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
def _family_stem(name: str) -> str:
|
|
610
|
+
return name.split("_", 1)[0] if "_" in name else name
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
def _parse_sve_markdown_docs(markdown: str) -> dict[str, dict[str, str]]:
|
|
614
|
+
docs: dict[str, dict[str, str]] = {}
|
|
615
|
+
headings: list[tuple[int, str]] = []
|
|
616
|
+
paragraph: list[str] = []
|
|
617
|
+
lines = markdown.splitlines()
|
|
618
|
+
index = 0
|
|
619
|
+
while index < len(lines):
|
|
620
|
+
raw = lines[index]
|
|
621
|
+
line = raw.rstrip()
|
|
622
|
+
stripped = line.strip()
|
|
623
|
+
if stripped.startswith("#"):
|
|
624
|
+
level = len(stripped) - len(stripped.lstrip("#"))
|
|
625
|
+
title = stripped[level:].strip()
|
|
626
|
+
headings = [(lvl, text) for lvl, text in headings if lvl < level]
|
|
627
|
+
headings.append((level, title))
|
|
628
|
+
paragraph = []
|
|
629
|
+
index += 1
|
|
630
|
+
continue
|
|
631
|
+
if stripped.startswith("```"):
|
|
632
|
+
block_lines: list[str] = []
|
|
633
|
+
index += 1
|
|
634
|
+
while index < len(lines) and not lines[index].strip().startswith("```"):
|
|
635
|
+
block_lines.append(lines[index].rstrip())
|
|
636
|
+
index += 1
|
|
637
|
+
block = "\n".join(block_lines).strip()
|
|
638
|
+
section_path = " / ".join(text for level, text in headings if level >= 3)
|
|
639
|
+
notes = "\n".join(line for line in paragraph if line.strip()).strip()
|
|
640
|
+
families = sorted(set(re.findall(r"\b(sv[a-z0-9]+)(?=\[|_|\()", block)))
|
|
641
|
+
for family in families:
|
|
642
|
+
docs.setdefault(family, {})
|
|
643
|
+
if notes:
|
|
644
|
+
docs[family]["ACLE Notes"] = notes
|
|
645
|
+
docs[family]["ACLE Prototypes"] = "\n".join(
|
|
646
|
+
part for part in [f"Section: {section_path}" if section_path else "", block] if part
|
|
647
|
+
)
|
|
648
|
+
paragraph = []
|
|
649
|
+
index += 1
|
|
650
|
+
continue
|
|
651
|
+
if stripped:
|
|
652
|
+
paragraph.append(stripped)
|
|
653
|
+
elif paragraph and paragraph[-1]:
|
|
654
|
+
paragraph.append("")
|
|
655
|
+
index += 1
|
|
656
|
+
return docs
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def parse_arm_neon_intrinsics_bundle(payload: dict[str, Any]) -> list[IntrinsicRecord]:
|
|
660
|
+
intrinsics_csv = str(payload.get("intrinsics_csv") or "")
|
|
661
|
+
classification_csv = str(payload.get("classification_csv") or "")
|
|
662
|
+
neon_markdown = str(payload.get("neon_markdown") or "")
|
|
663
|
+
acle_markdown = str(payload.get("acle_markdown") or "")
|
|
664
|
+
neon_docs = _parse_neon_markdown_docs(neon_markdown) if neon_markdown else {}
|
|
665
|
+
sve_docs = _parse_sve_markdown_docs(acle_markdown) if acle_markdown else {}
|
|
666
|
+
classifications: dict[str, str] = {}
|
|
667
|
+
for row in csv.reader(io.StringIO(classification_csv), delimiter="\t"):
|
|
668
|
+
if not row or row[0].startswith("<"):
|
|
669
|
+
continue
|
|
670
|
+
if len(row) >= 2:
|
|
671
|
+
classifications[row[0].strip()] = row[1].strip()
|
|
672
|
+
|
|
673
|
+
records: list[IntrinsicRecord] = []
|
|
674
|
+
current_section = ""
|
|
675
|
+
current_section_text = ""
|
|
676
|
+
for row in csv.reader(io.StringIO(intrinsics_csv), delimiter="\t"):
|
|
677
|
+
if not row:
|
|
678
|
+
continue
|
|
679
|
+
tag = row[0].strip()
|
|
680
|
+
if tag == "<SECTION>":
|
|
681
|
+
current_section = row[1].strip() if len(row) > 1 else ""
|
|
682
|
+
current_section_text = row[2].strip() if len(row) > 2 else ""
|
|
683
|
+
continue
|
|
684
|
+
if tag.startswith("<"):
|
|
685
|
+
continue
|
|
686
|
+
if len(row) < 5:
|
|
687
|
+
continue
|
|
688
|
+
signature, arg_prep, instruction_field, result_field, supported_arches = (value.strip() for value in row[:5])
|
|
689
|
+
if "A64" not in supported_arches:
|
|
690
|
+
continue
|
|
691
|
+
return_type, name, params = _parse_c_signature(signature)
|
|
692
|
+
path = [current_section] if current_section else []
|
|
693
|
+
if classification := classifications.get(name):
|
|
694
|
+
path.extend(part.strip() for part in classification.split("|") if part.strip())
|
|
695
|
+
category = path[-1] if path else "NEON intrinsics"
|
|
696
|
+
subcategory = " / ".join(path[:-1])
|
|
697
|
+
reference_url = f"{_ARM_NEON_REFERENCE_URL}#{_arm_slug(category)}"
|
|
698
|
+
instruction_refs = _arm_instruction_refs_from_field(instruction_field)
|
|
699
|
+
records.append(
|
|
700
|
+
IntrinsicRecord(
|
|
701
|
+
name=name,
|
|
702
|
+
signature=f"{return_type} {name}({', '.join(params)})",
|
|
703
|
+
description=f"{category}.",
|
|
704
|
+
header="arm_neon.h",
|
|
705
|
+
url=_arm_intrinsic_url(name),
|
|
706
|
+
architecture="arm",
|
|
707
|
+
isa=["NEON"],
|
|
708
|
+
category=category,
|
|
709
|
+
subcategory=subcategory,
|
|
710
|
+
instructions=[_canonical_instruction_key(ref['name'], ref['form']) or ref['name'] for ref in instruction_refs],
|
|
711
|
+
instruction_refs=instruction_refs,
|
|
712
|
+
metadata={
|
|
713
|
+
"argument_preparation": arg_prep,
|
|
714
|
+
"result": result_field,
|
|
715
|
+
"supported_architectures": supported_arches,
|
|
716
|
+
"reference_url": reference_url,
|
|
717
|
+
"section": current_section,
|
|
718
|
+
"section_description": current_section_text,
|
|
719
|
+
"classification_path": " / ".join(path),
|
|
720
|
+
},
|
|
721
|
+
doc_sections=neon_docs.get(name, {}),
|
|
722
|
+
notes=[current_section_text] if current_section_text else [],
|
|
723
|
+
source="arm-acle",
|
|
724
|
+
)
|
|
725
|
+
)
|
|
726
|
+
for item in payload.get("extra_intrinsics") or []:
|
|
727
|
+
name = str(item.get("name") or "").strip()
|
|
728
|
+
if not name:
|
|
729
|
+
continue
|
|
730
|
+
instruction_refs = []
|
|
731
|
+
instructions = []
|
|
732
|
+
for ref in item.get("instruction_refs") or item.get("instructions") or []:
|
|
733
|
+
if isinstance(ref, str):
|
|
734
|
+
ref = {"name": ref}
|
|
735
|
+
ref_name = str(ref.get("name") or "").strip()
|
|
736
|
+
ref_form = str(ref.get("form") or "").strip()
|
|
737
|
+
if not ref_name:
|
|
738
|
+
continue
|
|
739
|
+
instruction_refs.append({"name": ref_name, "form": ref_form, "architecture": "arm"})
|
|
740
|
+
instructions.append(_canonical_instruction_key(ref_name, ref_form) or ref_name)
|
|
741
|
+
records.append(
|
|
742
|
+
IntrinsicRecord(
|
|
743
|
+
name=name,
|
|
744
|
+
signature=str(item.get("signature") or "").strip(),
|
|
745
|
+
description=str(item.get("description") or "").strip(),
|
|
746
|
+
header=str(item.get("header") or "").strip(),
|
|
747
|
+
url=str(item.get("url") or "").strip(),
|
|
748
|
+
architecture="arm",
|
|
749
|
+
isa=_normalize_isa(item.get("isa") or []),
|
|
750
|
+
category=str(item.get("category") or "").strip(),
|
|
751
|
+
subcategory=str(item.get("subcategory") or "").strip(),
|
|
752
|
+
instructions=instructions,
|
|
753
|
+
instruction_refs=instruction_refs,
|
|
754
|
+
metadata={
|
|
755
|
+
str(key): str(value).strip()
|
|
756
|
+
for key, value in (item.get("metadata") or {}).items()
|
|
757
|
+
if str(value).strip()
|
|
758
|
+
},
|
|
759
|
+
doc_sections={
|
|
760
|
+
str(key): str(value).strip()
|
|
761
|
+
for key, value in (
|
|
762
|
+
item.get("doc_sections")
|
|
763
|
+
or sve_docs.get(_family_stem(name))
|
|
764
|
+
or {}
|
|
765
|
+
).items()
|
|
766
|
+
if str(value).strip()
|
|
767
|
+
},
|
|
768
|
+
notes=[str(value).strip() for value in item.get("notes") or [] if str(value).strip()],
|
|
769
|
+
source="arm-acle",
|
|
770
|
+
)
|
|
771
|
+
)
|
|
772
|
+
records.extend(parse_arm_sve_instruction_map(acle_markdown, sve_docs=sve_docs))
|
|
773
|
+
return records
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
def parse_arm_sve_instruction_map(markdown: str, *, sve_docs: dict[str, dict[str, str]] | None = None) -> list[IntrinsicRecord]:
|
|
777
|
+
if "### Mapping of SVE instructions to intrinsics" not in markdown:
|
|
778
|
+
return []
|
|
779
|
+
start = markdown.find("### Mapping of SVE instructions to intrinsics")
|
|
780
|
+
if start < 0:
|
|
781
|
+
return []
|
|
782
|
+
table_start = markdown.find("| **Instruction**", start)
|
|
783
|
+
if table_start < 0:
|
|
784
|
+
return []
|
|
785
|
+
lines = markdown[table_start:].splitlines()
|
|
786
|
+
records: list[IntrinsicRecord] = []
|
|
787
|
+
seen: set[str] = set()
|
|
788
|
+
row_re = re.compile(
|
|
789
|
+
r"^\|\s*(?P<instruction>[^|]+?)\s*\|\s*\[`(?P<name>[^`]+)`\]\((?P<url>[^)]+)\)\s*\|$"
|
|
790
|
+
)
|
|
791
|
+
for line in lines[2:]:
|
|
792
|
+
if not line.startswith("|"):
|
|
793
|
+
break
|
|
794
|
+
match = row_re.match(line.strip())
|
|
795
|
+
if not match:
|
|
796
|
+
continue
|
|
797
|
+
instruction = match.group("instruction").strip()
|
|
798
|
+
name = match.group("name").strip()
|
|
799
|
+
url = match.group("url").strip()
|
|
800
|
+
if name in seen or not name.startswith("sv"):
|
|
801
|
+
continue
|
|
802
|
+
seen.add(name)
|
|
803
|
+
instruction_head, _, instruction_tail = instruction.partition("(")
|
|
804
|
+
instruction_name = instruction_head.strip().split()[0]
|
|
805
|
+
instruction_form = instruction_tail.rsplit(")", 1)[0].strip() if instruction_tail else ""
|
|
806
|
+
isa = "SVE2" if any(token in instruction_name for token in ("ADDB", "ADDT", "HNB", "HNT", "LB", "LT", "WB", "WT")) or instruction_name.startswith("SADD") or instruction_name.startswith("UADD") else "SVE"
|
|
807
|
+
records.append(
|
|
808
|
+
IntrinsicRecord(
|
|
809
|
+
name=name,
|
|
810
|
+
signature=name,
|
|
811
|
+
description=f"{instruction}.",
|
|
812
|
+
header="arm_sve.h",
|
|
813
|
+
url=url,
|
|
814
|
+
architecture="arm",
|
|
815
|
+
isa=[isa],
|
|
816
|
+
category="Instruction mapping",
|
|
817
|
+
subcategory="SVE / Instruction family",
|
|
818
|
+
instructions=[instruction],
|
|
819
|
+
instruction_refs=[{"name": instruction_name, "form": instruction_form, "architecture": "arm"}],
|
|
820
|
+
metadata={
|
|
821
|
+
"reference_url": "https://arm-software.github.io/acle/main/acle.html#mapping-of-sve-instructions-to-intrinsics",
|
|
822
|
+
"mapping_instruction": instruction,
|
|
823
|
+
},
|
|
824
|
+
doc_sections=dict((sve_docs or {}).get(name, {})),
|
|
825
|
+
source="arm-acle",
|
|
826
|
+
)
|
|
827
|
+
)
|
|
828
|
+
return records
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
def parse_uops_xml(source: str | Path) -> list[InstructionRecord]:
|
|
832
|
+
records: list[InstructionRecord] = []
|
|
833
|
+
for node in _iter_xml_elements(source, "instruction"):
|
|
834
|
+
mnemonic = _intern_small((node.attrib.get("asm") or node.attrib.get("name") or "").strip())
|
|
835
|
+
if not mnemonic:
|
|
836
|
+
continue
|
|
837
|
+
form = (node.attrib.get("string") or node.attrib.get("form") or node.attrib.get("cpl") or node.attrib.get("category") or "").strip()
|
|
838
|
+
raw_summary = node.attrib.get("summary", "").strip()
|
|
839
|
+
isa = _normalize_isa(node.attrib.get("isa-set", "") or node.attrib.get("extension", "") or node.attrib.get("isa", ""))
|
|
840
|
+
operand_details: list[dict[str, str]] = []
|
|
841
|
+
metadata = {
|
|
842
|
+
key: (_intern_small(value.strip()) if key in {"category", "cpl", "extension", "iclass"} else value.strip())
|
|
843
|
+
for key, value in node.attrib.items()
|
|
844
|
+
if key in _UOPS_METADATA_KEYS
|
|
845
|
+
}
|
|
846
|
+
if raw_summary:
|
|
847
|
+
metadata["uops_summary"] = raw_summary
|
|
848
|
+
arch_details: dict[str, dict[str, Any]] = {}
|
|
849
|
+
for child in node:
|
|
850
|
+
if child.tag == "operand":
|
|
851
|
+
xtype = _normalize_operand_xtype(child.attrib.get("xtype", "").strip())
|
|
852
|
+
operand_payload = {
|
|
853
|
+
key: (_intern_small(value.strip()) if key in {"type", "width", "name"} else value.strip())
|
|
854
|
+
for key, value in child.attrib.items()
|
|
855
|
+
if key in _UOPS_OPERAND_KEYS
|
|
856
|
+
}
|
|
857
|
+
if xtype:
|
|
858
|
+
operand_payload["xtype"] = _intern_small(xtype)
|
|
859
|
+
operand_details.append(operand_payload)
|
|
860
|
+
elif child.tag == "architecture":
|
|
861
|
+
arch = child.attrib.get("name") or child.attrib.get("uarch") or child.attrib.get("arch")
|
|
862
|
+
if not arch:
|
|
863
|
+
continue
|
|
864
|
+
arch = _intern_small(arch)
|
|
865
|
+
arch_entry: dict[str, Any] = {"measurement": {}, "latencies": [], "doc": {}, "iaca": []}
|
|
866
|
+
for grandchild in child:
|
|
867
|
+
if grandchild.tag == "measurement":
|
|
868
|
+
arch_entry["measurement"] = dict(grandchild.attrib)
|
|
869
|
+
for latency in grandchild.findall("./latency"):
|
|
870
|
+
arch_entry["latencies"].append(dict(latency.attrib))
|
|
871
|
+
elif grandchild.tag == "doc":
|
|
872
|
+
arch_entry["doc"] = dict(grandchild.attrib)
|
|
873
|
+
elif grandchild.tag == "IACA":
|
|
874
|
+
arch_entry["iaca"].append(dict(grandchild.attrib))
|
|
875
|
+
arch_details[arch] = arch_entry
|
|
876
|
+
records.append(
|
|
877
|
+
InstructionRecord(
|
|
878
|
+
mnemonic=mnemonic,
|
|
879
|
+
form=form,
|
|
880
|
+
summary=_instruction_summary(mnemonic, raw_summary, operand_details, architecture="x86"),
|
|
881
|
+
architecture="x86",
|
|
882
|
+
isa=isa,
|
|
883
|
+
operand_details=operand_details,
|
|
884
|
+
metadata=metadata,
|
|
885
|
+
arch_details=arch_details,
|
|
886
|
+
)
|
|
887
|
+
)
|
|
888
|
+
return records
|
|
889
|
+
|
|
890
|
+
|
|
891
|
+
def _instruction_indexes(
|
|
892
|
+
instructions: list[InstructionRecord],
|
|
893
|
+
) -> tuple[
|
|
894
|
+
dict[tuple[str, str], list[InstructionRecord]],
|
|
895
|
+
dict[tuple[str, str], list[InstructionRecord]],
|
|
896
|
+
dict[tuple[str, str], list[InstructionRecord]],
|
|
897
|
+
]:
|
|
898
|
+
by_mnemonic: dict[tuple[str, str], list[InstructionRecord]] = {}
|
|
899
|
+
by_key: dict[tuple[str, str], list[InstructionRecord]] = {}
|
|
900
|
+
by_iform: dict[tuple[str, str], list[InstructionRecord]] = {}
|
|
901
|
+
for record in instructions:
|
|
902
|
+
arch = record.architecture
|
|
903
|
+
by_mnemonic.setdefault((arch, record.mnemonic.casefold()), []).append(record)
|
|
904
|
+
by_key.setdefault((arch, record.key.casefold()), []).append(record)
|
|
905
|
+
if record.form:
|
|
906
|
+
by_mnemonic.setdefault((arch, record.key.casefold()), []).append(record)
|
|
907
|
+
if record.metadata.get("iform"):
|
|
908
|
+
by_iform.setdefault((arch, record.metadata["iform"].casefold()), []).append(record)
|
|
909
|
+
return by_mnemonic, by_key, by_iform
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def _resolve_instruction_ref(
|
|
913
|
+
intrinsic: IntrinsicRecord,
|
|
914
|
+
ref: dict[str, str],
|
|
915
|
+
*,
|
|
916
|
+
by_mnemonic: dict[tuple[str, str], list[InstructionRecord]],
|
|
917
|
+
by_key: dict[tuple[str, str], list[InstructionRecord]],
|
|
918
|
+
by_iform: dict[tuple[str, str], list[InstructionRecord]],
|
|
919
|
+
) -> tuple[list[InstructionRecord], dict[str, str]]:
|
|
920
|
+
matched: list[InstructionRecord] = []
|
|
921
|
+
ref_arch = ref.get("architecture", intrinsic.architecture).strip() or intrinsic.architecture
|
|
922
|
+
xed = ref.get("xed", "").strip()
|
|
923
|
+
name = ref.get("name", "").strip()
|
|
924
|
+
form = ref.get("form", "").strip()
|
|
925
|
+
resolution = "unresolved"
|
|
926
|
+
if xed and ref_arch == "x86":
|
|
927
|
+
matched = by_iform.get((ref_arch, xed.casefold()), [])
|
|
928
|
+
if matched:
|
|
929
|
+
resolution = "xed"
|
|
930
|
+
if not matched and name and form:
|
|
931
|
+
key_candidates = [(_canonical_instruction_key(name, form) or name).casefold()]
|
|
932
|
+
if ref_arch == "riscv":
|
|
933
|
+
key_candidates.insert(0, form.casefold())
|
|
934
|
+
for candidate_key in key_candidates:
|
|
935
|
+
matched = by_key.get((ref_arch, candidate_key), [])
|
|
936
|
+
if matched:
|
|
937
|
+
resolution = "key"
|
|
938
|
+
break
|
|
939
|
+
if not matched and name:
|
|
940
|
+
matched = by_mnemonic.get((ref_arch, name.casefold()), [])
|
|
941
|
+
if matched:
|
|
942
|
+
resolution = "mnemonic"
|
|
943
|
+
if ref_arch == "arm" and len(matched) > 1 and intrinsic.isa:
|
|
944
|
+
intrinsic_isas = {value.casefold() for value in intrinsic.isa}
|
|
945
|
+
narrowed = [
|
|
946
|
+
instruction
|
|
947
|
+
for instruction in matched
|
|
948
|
+
if intrinsic_isas & {value.casefold() for value in instruction.isa}
|
|
949
|
+
]
|
|
950
|
+
if narrowed:
|
|
951
|
+
matched = narrowed
|
|
952
|
+
resolution = "arm-isa"
|
|
953
|
+
if ref_arch == "riscv" and len(matched) > 1:
|
|
954
|
+
ref_isa = {value.casefold() for value in _normalize_isa(ref.get("isa", ""))}
|
|
955
|
+
if ref_isa:
|
|
956
|
+
narrowed = [
|
|
957
|
+
instruction
|
|
958
|
+
for instruction in matched
|
|
959
|
+
if ref_isa & {value.casefold() for value in instruction.isa}
|
|
960
|
+
]
|
|
961
|
+
if narrowed:
|
|
962
|
+
matched = narrowed
|
|
963
|
+
resolution = f"{resolution}-riscv-isa"
|
|
964
|
+
ref_policy = ref.get("policy", "").strip()
|
|
965
|
+
if len(matched) > 1 and ref_policy:
|
|
966
|
+
narrowed = [instruction for instruction in matched if instruction.metadata.get("policy", "").strip() == ref_policy]
|
|
967
|
+
if narrowed:
|
|
968
|
+
matched = narrowed
|
|
969
|
+
resolution = f"{resolution}-riscv-policy"
|
|
970
|
+
ref_tail_policy = ref.get("tail_policy", "").strip()
|
|
971
|
+
if len(matched) > 1 and ref_tail_policy:
|
|
972
|
+
narrowed = [instruction for instruction in matched if instruction.metadata.get("tail_policy", "").strip() == ref_tail_policy]
|
|
973
|
+
if narrowed:
|
|
974
|
+
matched = narrowed
|
|
975
|
+
resolution = f"{resolution}-riscv-tail"
|
|
976
|
+
ref_mask_policy = ref.get("mask_policy", "").strip()
|
|
977
|
+
if len(matched) > 1 and ref_mask_policy:
|
|
978
|
+
narrowed = [instruction for instruction in matched if instruction.metadata.get("mask_policy", "").strip() == ref_mask_policy]
|
|
979
|
+
if narrowed:
|
|
980
|
+
matched = narrowed
|
|
981
|
+
resolution = f"{resolution}-riscv-mask"
|
|
982
|
+
ref_masking = ref.get("masking", "").strip()
|
|
983
|
+
if len(matched) > 1 and ref_masking:
|
|
984
|
+
narrowed = [instruction for instruction in matched if instruction.metadata.get("masking", "").strip() == ref_masking]
|
|
985
|
+
if narrowed:
|
|
986
|
+
matched = narrowed
|
|
987
|
+
resolution = f"{resolution}-riscv-masking"
|
|
988
|
+
if ref_arch == "x86" and len(matched) > 1:
|
|
989
|
+
lowered_name = intrinsic.name.casefold()
|
|
990
|
+
if "_maskz_" in lowered_name:
|
|
991
|
+
narrowed = [instruction for instruction in matched if instruction.key.startswith(f"{instruction.mnemonic}_Z ")]
|
|
992
|
+
if narrowed:
|
|
993
|
+
matched = narrowed
|
|
994
|
+
resolution = f"{resolution}-maskz"
|
|
995
|
+
elif "_mask_" in lowered_name or "_mask2_" in lowered_name:
|
|
996
|
+
narrowed = [
|
|
997
|
+
instruction
|
|
998
|
+
for instruction in matched
|
|
999
|
+
if ", K," in instruction.key and not instruction.key.startswith(f"{instruction.mnemonic}_Z ")
|
|
1000
|
+
]
|
|
1001
|
+
if narrowed:
|
|
1002
|
+
matched = narrowed
|
|
1003
|
+
resolution = f"{resolution}-mask"
|
|
1004
|
+
else:
|
|
1005
|
+
narrowed = [
|
|
1006
|
+
instruction
|
|
1007
|
+
for instruction in matched
|
|
1008
|
+
if ", K," not in instruction.key and not instruction.key.startswith(f"{instruction.mnemonic}_Z ")
|
|
1009
|
+
]
|
|
1010
|
+
if narrowed:
|
|
1011
|
+
matched = narrowed
|
|
1012
|
+
resolution = f"{resolution}-plain"
|
|
1013
|
+
if ref_arch == "x86" and len(matched) > 1:
|
|
1014
|
+
lowered_name = intrinsic.name.casefold()
|
|
1015
|
+
width_markers = (
|
|
1016
|
+
("64", ("R64", "REX64")),
|
|
1017
|
+
("32", ("R32", "Rel32")),
|
|
1018
|
+
("16", ("R16", "Rel16")),
|
|
1019
|
+
("8", ("R8",)),
|
|
1020
|
+
)
|
|
1021
|
+
for width, markers in width_markers:
|
|
1022
|
+
if width not in lowered_name:
|
|
1023
|
+
continue
|
|
1024
|
+
narrowed = [
|
|
1025
|
+
instruction
|
|
1026
|
+
for instruction in matched
|
|
1027
|
+
if any(marker in instruction.key for marker in markers)
|
|
1028
|
+
]
|
|
1029
|
+
if narrowed:
|
|
1030
|
+
matched = narrowed
|
|
1031
|
+
resolution = f"{resolution}-width{width}"
|
|
1032
|
+
break
|
|
1033
|
+
resolved = {
|
|
1034
|
+
"architecture": ref_arch,
|
|
1035
|
+
"name": name,
|
|
1036
|
+
"form": form,
|
|
1037
|
+
"xed": xed,
|
|
1038
|
+
"match_count": str(len(matched)),
|
|
1039
|
+
"resolution": resolution if matched else "unresolved",
|
|
1040
|
+
}
|
|
1041
|
+
return matched, resolved
|
|
1042
|
+
|
|
1043
|
+
|
|
1044
|
+
def link_records(intrinsics: list[IntrinsicRecord], instructions: list[InstructionRecord]) -> None:
|
|
1045
|
+
by_mnemonic, by_key, by_iform = _instruction_indexes(instructions)
|
|
1046
|
+
for intrinsic in intrinsics:
|
|
1047
|
+
linked: list[str] = []
|
|
1048
|
+
resolved_refs: list[dict[str, str]] = []
|
|
1049
|
+
refs = intrinsic.instruction_refs or [{"name": name, "form": "", "xed": ""} for name in intrinsic.instructions]
|
|
1050
|
+
for ref in refs:
|
|
1051
|
+
matched, resolved = _resolve_instruction_ref(
|
|
1052
|
+
intrinsic,
|
|
1053
|
+
ref,
|
|
1054
|
+
by_mnemonic=by_mnemonic,
|
|
1055
|
+
by_key=by_key,
|
|
1056
|
+
by_iform=by_iform,
|
|
1057
|
+
)
|
|
1058
|
+
ref_arch = resolved["architecture"]
|
|
1059
|
+
if not matched:
|
|
1060
|
+
fallback = _canonical_instruction_key(resolved["name"], resolved["form"]) or resolved["name"]
|
|
1061
|
+
if fallback:
|
|
1062
|
+
linked.append(fallback)
|
|
1063
|
+
resolved_refs.append(dict(ref) | resolved)
|
|
1064
|
+
continue
|
|
1065
|
+
for instruction in matched:
|
|
1066
|
+
if intrinsic.name not in instruction.linked_intrinsics:
|
|
1067
|
+
instruction.linked_intrinsics.append(intrinsic.name)
|
|
1068
|
+
linked.append(instruction.key)
|
|
1069
|
+
resolved_refs.append(dict(ref) | resolved | {
|
|
1070
|
+
"architecture": instruction.architecture,
|
|
1071
|
+
"key": instruction.db_key,
|
|
1072
|
+
"display_key": instruction.key,
|
|
1073
|
+
})
|
|
1074
|
+
intrinsic.instructions = list(dict.fromkeys(linked))
|
|
1075
|
+
intrinsic.instruction_refs = resolved_refs
|
|
1076
|
+
|
|
1077
|
+
|
|
1078
|
+
def _ingest_perf_sources(
|
|
1079
|
+
instructions: list[InstructionRecord],
|
|
1080
|
+
*,
|
|
1081
|
+
status: Callable[[str], None],
|
|
1082
|
+
) -> list[SourceVersion]:
|
|
1083
|
+
"""Run the llvm scheduling pipeline and merge rows into *instructions*.
|
|
1084
|
+
|
|
1085
|
+
Any ingester failure propagates so silent breakage (missing LLVM
|
|
1086
|
+
binary, empty exegesis output, scheduler-model mismatch) surfaces at
|
|
1087
|
+
build time instead of producing a silently empty catalog.
|
|
1088
|
+
"""
|
|
1089
|
+
from simdref.perf_sources import (
|
|
1090
|
+
ingest_llvm_mca,
|
|
1091
|
+
merge_perf_rows,
|
|
1092
|
+
)
|
|
1093
|
+
|
|
1094
|
+
versions: list[SourceVersion] = []
|
|
1095
|
+
|
|
1096
|
+
status("Driving llvm-exegesis + llvm-mca across modeled cores")
|
|
1097
|
+
llvm_rows, llvm_version = ingest_llvm_mca(instructions)
|
|
1098
|
+
status(f"Collected {len(llvm_rows)} llvm-mca perf rows")
|
|
1099
|
+
if llvm_rows:
|
|
1100
|
+
merge_perf_rows(instructions, llvm_rows)
|
|
1101
|
+
versions.append(SourceVersion(
|
|
1102
|
+
source="llvm-mca", version=llvm_version or "unknown",
|
|
1103
|
+
fetched_at=now_iso(),
|
|
1104
|
+
url="https://llvm.org/docs/CommandGuide/llvm-mca.html",
|
|
1105
|
+
))
|
|
1106
|
+
|
|
1107
|
+
return versions
|
|
1108
|
+
|
|
1109
|
+
|
|
1110
|
+
def build_catalog(
|
|
1111
|
+
include_sdm: bool = False,
|
|
1112
|
+
*,
|
|
1113
|
+
status: Callable[[str], None] | None = None,
|
|
1114
|
+
) -> Catalog:
|
|
1115
|
+
emit = status or (lambda _msg: None)
|
|
1116
|
+
emit("Fetching Intel intrinsics data")
|
|
1117
|
+
intel_text, intel_source = fetch_intel_data()
|
|
1118
|
+
emit(f"Fetched Intel intrinsics data from {intel_source.url}")
|
|
1119
|
+
emit("Fetching uops.info instruction data")
|
|
1120
|
+
uops_text, uops_source = fetch_uops_xml()
|
|
1121
|
+
emit(f"Fetched uops.info instruction data from {uops_source.url}")
|
|
1122
|
+
emit("Fetching Arm ACLE intrinsic data")
|
|
1123
|
+
arm_acle_text, arm_acle_source = fetch_arm_acle_data()
|
|
1124
|
+
emit(f"Fetched Arm ACLE intrinsic data from {arm_acle_source.url}")
|
|
1125
|
+
emit("Fetching Arm A64 instruction data")
|
|
1126
|
+
arm_a64_text, arm_a64_source = fetch_arm_a64_data()
|
|
1127
|
+
emit(f"Fetched Arm A64 instruction data from {arm_a64_source.url}")
|
|
1128
|
+
emit("Fetching RISC-V RVV intrinsic data")
|
|
1129
|
+
riscv_intrinsics_text, riscv_intrinsics_source = fetch_riscv_rvv_intrinsics_data()
|
|
1130
|
+
emit(f"Fetched RISC-V RVV intrinsic data from {riscv_intrinsics_source.url}")
|
|
1131
|
+
emit("Fetching RISC-V unified-db instruction data")
|
|
1132
|
+
riscv_instructions_text, riscv_instructions_source = fetch_riscv_unified_db_data()
|
|
1133
|
+
emit(f"Fetched RISC-V unified-db instruction data from {riscv_instructions_source.url}")
|
|
1134
|
+
emit("Parsing intrinsic catalog")
|
|
1135
|
+
intrinsics = parse_intel_payload(intel_text)
|
|
1136
|
+
arm_intrinsics = parse_arm_intrinsics_payload(arm_acle_text)
|
|
1137
|
+
intrinsics.extend(arm_intrinsics)
|
|
1138
|
+
intrinsics.extend(parse_riscv_intrinsics_payload(riscv_intrinsics_text))
|
|
1139
|
+
emit(f"Parsed {len(intrinsics)} intrinsics")
|
|
1140
|
+
emit("Parsing instruction catalog")
|
|
1141
|
+
instructions = parse_uops_xml(uops_text)
|
|
1142
|
+
instructions.extend(parse_arm_instruction_payload(arm_a64_text))
|
|
1143
|
+
instructions.extend(parse_riscv_instruction_payload(riscv_instructions_text))
|
|
1144
|
+
emit(f"Parsed {len(instructions)} instructions")
|
|
1145
|
+
emit("Linking intrinsics to instructions")
|
|
1146
|
+
link_records(intrinsics, instructions)
|
|
1147
|
+
emit("Linked intrinsics and instructions")
|
|
1148
|
+
|
|
1149
|
+
perf_sources_version = list(_ingest_perf_sources(instructions, status=emit))
|
|
1150
|
+
|
|
1151
|
+
if include_sdm:
|
|
1152
|
+
sdm_path = find_pdf_source_path("intel-sdm")
|
|
1153
|
+
if sdm_path is not None:
|
|
1154
|
+
try:
|
|
1155
|
+
emit(f"Preparing Intel SDM descriptions from {sdm_path}")
|
|
1156
|
+
result = load_or_parse_pdf_source("intel-sdm", sdm_path, status=status)
|
|
1157
|
+
merge_pdf_enrichment(instructions, "intel-sdm", result)
|
|
1158
|
+
emit("Merged Intel SDM descriptions into instruction records")
|
|
1159
|
+
except Exception:
|
|
1160
|
+
pass
|
|
1161
|
+
|
|
1162
|
+
emit("Assembling final catalog")
|
|
1163
|
+
return Catalog(
|
|
1164
|
+
intrinsics=sorted(intrinsics, key=lambda item: item.name),
|
|
1165
|
+
instructions=sorted(instructions, key=lambda item: (item.architecture, item.mnemonic, item.form)),
|
|
1166
|
+
sources=[
|
|
1167
|
+
intel_source, uops_source, arm_acle_source, arm_a64_source,
|
|
1168
|
+
riscv_intrinsics_source, riscv_instructions_source,
|
|
1169
|
+
*perf_sources_version,
|
|
1170
|
+
],
|
|
1171
|
+
generated_at=now_iso(),
|
|
1172
|
+
)
|