simdref 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,417 @@
1
+ """Arm instruction-source parsing and normalization."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ from typing import Any
8
+
9
+ from simdref.models import InstructionRecord
10
+
11
+
12
+ def _normalize_isa(value: Any) -> list[str]:
13
+ if isinstance(value, list):
14
+ return [str(item).strip() for item in value if str(item).strip()]
15
+ if isinstance(value, str):
16
+ return [part.strip() for part in re.split(r"[,/|]\s*|\s{2,}", value) if part.strip()]
17
+ return []
18
+
19
+
20
+ def _canonical_instruction_key(name: str, form: str) -> str:
21
+ instruction_name = name.strip().upper()
22
+ instruction_form = re.sub(r"\s*,\s*", ", ", form.strip())
23
+ instruction_form = re.sub(r"\s+", " ", instruction_form)
24
+ if not instruction_name:
25
+ return ""
26
+ if not instruction_form:
27
+ return instruction_name
28
+ return f"{instruction_name} ({instruction_form.upper()})"
29
+
30
+
31
+ def _generated_summary(mnemonic: str) -> str:
32
+ core = mnemonic.strip().upper()
33
+ if not core:
34
+ return "Arm instruction."
35
+ if core.startswith("ADD"):
36
+ return "Add operands."
37
+ if core.startswith("SUB"):
38
+ return "Subtract operands."
39
+ if core.startswith("MUL"):
40
+ return "Multiply operands."
41
+ if core.startswith("LD"):
42
+ return "Load data."
43
+ if core.startswith("ST"):
44
+ return "Store data."
45
+ return f"{core.title()} instruction."
46
+
47
+
48
+ def _strip_text(value: Any) -> str:
49
+ if isinstance(value, str):
50
+ return value.strip()
51
+ if isinstance(value, dict):
52
+ for key in ("text", "content", "value", "body", "summary", "description", "brief"):
53
+ text = _strip_text(value.get(key))
54
+ if text:
55
+ return text
56
+ return ""
57
+ if isinstance(value, list):
58
+ parts = [_strip_text(item) for item in value]
59
+ return "\n".join(part for part in parts if part).strip()
60
+ return ""
61
+
62
+
63
+ def _collapse_section_value(value: Any) -> str:
64
+ if isinstance(value, str):
65
+ return value.strip()
66
+ if isinstance(value, list):
67
+ parts = [_collapse_section_value(item) for item in value]
68
+ return "\n\n".join(part for part in parts if part).strip()
69
+ if isinstance(value, dict):
70
+ if "title" in value and any(key in value for key in ("body", "content", "text", "value")):
71
+ title = _strip_text(value.get("title"))
72
+ body = _strip_text(value.get("body") or value.get("content") or value.get("text") or value.get("value"))
73
+ return "\n".join(part for part in (title, body) if part).strip()
74
+ flattened = []
75
+ for item in value.values():
76
+ text = _collapse_section_value(item)
77
+ if text:
78
+ flattened.append(text)
79
+ return "\n\n".join(flattened).strip()
80
+ return ""
81
+
82
+
83
+ def _description_sections(item: dict[str, Any]) -> dict[str, str]:
84
+ sections: dict[str, str] = {}
85
+
86
+ for key in ("description", "descriptions", "doc_sections", "sections"):
87
+ raw = item.get(key)
88
+ if isinstance(raw, dict):
89
+ for title, value in raw.items():
90
+ text = _collapse_section_value(value)
91
+ if text:
92
+ sections[str(title).strip()] = text
93
+ elif isinstance(raw, list):
94
+ for entry in raw:
95
+ if not isinstance(entry, dict):
96
+ continue
97
+ title = _strip_text(entry.get("title") or entry.get("name") or entry.get("heading"))
98
+ body = _collapse_section_value(entry.get("body") or entry.get("content") or entry.get("text") or entry.get("value"))
99
+ if title and body:
100
+ sections[title] = body
101
+
102
+ operation = _collapse_section_value(item.get("operation") or item.get("pseudocode"))
103
+ if operation:
104
+ sections.setdefault("Operation", operation)
105
+
106
+ detail = _strip_text(item.get("detail") or item.get("details"))
107
+ if detail:
108
+ sections.setdefault("Details", detail)
109
+
110
+ return sections
111
+
112
+
113
+ def _infer_arm_isa(item: dict[str, Any]) -> list[str]:
114
+ explicit = _normalize_isa(item.get("isa") or item.get("isas") or item.get("extensions") or item.get("feature_tags"))
115
+ if explicit:
116
+ return explicit
117
+
118
+ haystack_parts = [
119
+ _strip_text(item.get("category")),
120
+ _strip_text(item.get("section")),
121
+ _strip_text(item.get("group")),
122
+ _strip_text(item.get("classification")),
123
+ _strip_text(item.get("url")),
124
+ ]
125
+ haystack = " ".join(part.upper() for part in haystack_parts if part)
126
+
127
+ isa: list[str] = []
128
+ for token in ("SME2", "SME", "SVE2", "SVE", "MVE", "HELIUM", "NEON", "ADVSIMD", "SIMD-FP", "SIMD&FP"):
129
+ if token in haystack:
130
+ if token in {"HELIUM", "MVE"}:
131
+ candidate = "MVE"
132
+ elif token in {"ADVSIMD", "SIMD-FP", "SIMD&FP"}:
133
+ candidate = "NEON"
134
+ else:
135
+ candidate = token
136
+ if candidate not in isa:
137
+ isa.append(candidate)
138
+
139
+ if isa:
140
+ return isa
141
+ return ["A64"]
142
+
143
+
144
+ def _collect_aliases(item: dict[str, Any]) -> list[str]:
145
+ raw = item.get("aliases") or item.get("alias_mnemonics") or []
146
+ if isinstance(raw, str):
147
+ raw = [raw]
148
+ aliases = [str(value).strip() for value in raw if str(value).strip()]
149
+ alias = _strip_text(item.get("alias"))
150
+ if alias:
151
+ aliases.append(alias)
152
+ return list(dict.fromkeys(aliases))
153
+
154
+
155
+ def _collect_metadata(item: dict[str, Any], *, url: str) -> dict[str, str]:
156
+ metadata: dict[str, str] = {}
157
+ for source_key, target_key in (
158
+ ("category", "category"),
159
+ ("section", "section"),
160
+ ("group", "group"),
161
+ ("classification", "classification"),
162
+ ("encoding", "encoding"),
163
+ ("operand_form", "operand_form"),
164
+ ("instruction_class", "instruction_class"),
165
+ ("architecture_state", "architecture_state"),
166
+ ):
167
+ value = _strip_text(item.get(source_key))
168
+ if value:
169
+ metadata[target_key] = value
170
+ if url:
171
+ metadata["url"] = url
172
+ return metadata
173
+
174
+
175
+ def _normalize_instruction_item(item: dict[str, Any]) -> InstructionRecord | None:
176
+ mnemonic = _strip_text(
177
+ item.get("mnemonic")
178
+ or item.get("base_instruction")
179
+ or item.get("instruction")
180
+ or item.get("name")
181
+ or item.get("opcode")
182
+ ).upper()
183
+ if not mnemonic:
184
+ return None
185
+
186
+ operands = _strip_text(item.get("operands") or item.get("operand_form") or item.get("form"))
187
+ form = _canonical_instruction_key(mnemonic, operands) if operands and not operands.upper().startswith(f"{mnemonic} (") else operands
188
+ summary = _strip_text(item.get("summary") or item.get("brief") or item.get("title")) or _generated_summary(mnemonic)
189
+ if not summary.endswith("."):
190
+ summary += "."
191
+ url = _strip_text(item.get("url") or item.get("source_url") or item.get("reference_url"))
192
+ return InstructionRecord(
193
+ mnemonic=mnemonic,
194
+ form=form,
195
+ summary=summary,
196
+ architecture="arm",
197
+ isa=_infer_arm_isa(item),
198
+ metadata=_collect_metadata(item, url=url),
199
+ aliases=_collect_aliases(item),
200
+ description=_description_sections(item),
201
+ source="arm-a64",
202
+ )
203
+
204
+
205
+ def _candidate_instruction_lists(payload: Any) -> list[list[dict[str, Any]]]:
206
+ lists: list[list[dict[str, Any]]] = []
207
+ if isinstance(payload, list):
208
+ if all(isinstance(item, dict) for item in payload):
209
+ lists.append(payload)
210
+ return lists
211
+ if not isinstance(payload, dict):
212
+ return lists
213
+
214
+ for key in (
215
+ "instructions",
216
+ "base_instructions",
217
+ "instruction_set",
218
+ "InstructionSet",
219
+ "items",
220
+ "records",
221
+ ):
222
+ value = payload.get(key)
223
+ if isinstance(value, list) and all(isinstance(item, dict) for item in value):
224
+ lists.append(value)
225
+
226
+ for value in payload.values():
227
+ if isinstance(value, dict):
228
+ lists.extend(_candidate_instruction_lists(value))
229
+
230
+ return lists
231
+
232
+
233
+ def _records_from_payload(payload: Any) -> list[InstructionRecord]:
234
+ if isinstance(payload, dict) and payload.get("format") == "arm-aarchmrs-instructions-v1":
235
+ instructions_json = str(payload.get("instructions_json") or "[]")
236
+ return _records_from_payload(json.loads(instructions_json))
237
+
238
+ if isinstance(payload, dict) and payload.get("format") == "arm-instructions-fixture-v1":
239
+ return _records_from_payload(payload.get("instructions") or [])
240
+
241
+ aarchmrs = _records_from_aarchmrs_tree(payload)
242
+ if aarchmrs:
243
+ return aarchmrs
244
+
245
+ for candidates in _candidate_instruction_lists(payload):
246
+ records = [_normalize_instruction_item(item) for item in candidates]
247
+ normalized = [record for record in records if record is not None]
248
+ if normalized:
249
+ return normalized
250
+ return []
251
+
252
+
253
+ # ---------------------------------------------------------------------------
254
+ # AARCHMRS (official Arm machine-readable spec) tree walker
255
+ # ---------------------------------------------------------------------------
256
+
257
+ _RULE_PLACEHOLDER: dict[str, str] = {
258
+ # Scalar
259
+ "Xd": "x0", "Xn": "x1", "Xm": "x2", "Xa": "x3", "Xt": "x0", "Xt2": "x1",
260
+ "Xs": "x1", "Xt1": "x0",
261
+ "Wd": "w0", "Wn": "w1", "Wm": "w2", "Wa": "w3", "Wt": "w0", "Wt2": "w1",
262
+ "Ws": "w1", "Wt1": "w0",
263
+ # Advanced SIMD vector registers
264
+ "Vd": "v0", "Vn": "v1", "Vm": "v2", "Va": "v3", "Vt": "v0", "Vt2": "v1",
265
+ "Vt3": "v2", "Vt4": "v3",
266
+ "Dd": "d0", "Dn": "d1", "Dm": "d2", "Da": "d3",
267
+ "Sd": "s0", "Sn": "s1", "Sm": "s2", "Sa": "s3",
268
+ "Hd": "h0", "Hn": "h1", "Hm": "h2", "Ha": "h3",
269
+ "Bd": "b0", "Bn": "b1", "Bm": "b2",
270
+ "Qd": "q0", "Qn": "q1", "Qm": "q2", "Qt": "q0", "Qt2": "q1",
271
+ # SVE predicate / vector
272
+ "Zd": "z0", "Zn": "z1", "Zm": "z2", "Za": "z3", "Zdn": "z0", "Zt": "z0",
273
+ "Pd": "p0", "Pn": "p1", "Pm": "p2", "Pg": "p0", "Pt": "p0",
274
+ # SME
275
+ "ZAd": "za0", "ZAn": "za0", "ZAt": "za0",
276
+ # Stack pointer / zero register
277
+ "SP": "sp", "XZR": "xzr", "WZR": "wzr", "XSP": "sp", "WSP": "wsp",
278
+ # Condition codes / misc commonly referenced
279
+ "cond": "eq", "nzcv": "0",
280
+ # Arrangement specifiers and element size hints
281
+ "T": "4S", "Ta": "4S", "Tb": "4S", "Ts": "4S",
282
+ "T__1": "4S", "T__2": "4S", "T__3": "4S", "T__4": "4S",
283
+ "size": "4S", "size__1": "4S",
284
+ }
285
+
286
+ _LITERAL_KEEP = {"COMMA": ",", "SPACE": " ", "LBRACKET": "[", "RBRACKET": "]",
287
+ "LBRACE": "{", "RBRACE": "}", "HASH": "#", "EXCLAM": "!"}
288
+
289
+
290
+ def _render_assembly(assembly: dict[str, Any]) -> str:
291
+ """Render an AARCHMRS ``assembly.symbols`` list to a concrete asm string.
292
+
293
+ Literals are kept verbatim; RuleReferences substitute from
294
+ ``_RULE_PLACEHOLDER`` when known, fall back to ``#0`` for numeric rules
295
+ (imm / off / lsb / width / shift) and to a neutral token otherwise.
296
+ """
297
+ symbols = assembly.get("symbols") if isinstance(assembly, dict) else None
298
+ if not isinstance(symbols, list):
299
+ return ""
300
+ parts: list[str] = []
301
+ for sym in symbols:
302
+ if not isinstance(sym, dict):
303
+ continue
304
+ sym_type = str(sym.get("_type", ""))
305
+ if sym_type.endswith("Literal"):
306
+ parts.append(str(sym.get("value", "")))
307
+ continue
308
+ if sym_type.endswith("RuleReference"):
309
+ rule = str(sym.get("rule_id", ""))
310
+ if rule in _LITERAL_KEEP:
311
+ parts.append(_LITERAL_KEEP[rule])
312
+ continue
313
+ base = re.sub(r"__\d+$", "", rule)
314
+ if rule in _RULE_PLACEHOLDER:
315
+ parts.append(_RULE_PLACEHOLDER[rule])
316
+ elif base in _RULE_PLACEHOLDER:
317
+ parts.append(_RULE_PLACEHOLDER[base])
318
+ elif re.search(r"imm|offset|off|lsb|width|shift|amount|rot", rule, re.IGNORECASE):
319
+ parts.append("#0")
320
+ elif re.search(r"label|addr", rule, re.IGNORECASE):
321
+ parts.append(".")
322
+ else:
323
+ parts.append(f"<{rule}>")
324
+ continue
325
+ return "".join(parts).strip()
326
+
327
+
328
+ def _infer_aarchmrs_isa(group_path: list[str]) -> list[str]:
329
+ joined = " ".join(group_path).lower()
330
+ isa: list[str] = []
331
+ if "sme" in joined:
332
+ isa.append("SME2" if "sme2" in joined else "SME")
333
+ if "sve2" in joined:
334
+ isa.append("SVE2")
335
+ elif "sve" in joined:
336
+ isa.append("SVE")
337
+ if "advsimd" in joined or "asimd" in joined or "neon" in joined or "fp_" in joined or "simd" in joined:
338
+ isa.append("NEON")
339
+ if "mve" in joined:
340
+ isa.append("MVE")
341
+ return isa or ["A64"]
342
+
343
+
344
+ def _records_from_aarchmrs_tree(payload: Any) -> list[InstructionRecord]:
345
+ if not isinstance(payload, dict):
346
+ return []
347
+ top = payload.get("instructions")
348
+ if not isinstance(top, list) or not top:
349
+ return []
350
+ # Detect AARCHMRS: top-level entries are InstructionSet nodes with children.
351
+ if not all(
352
+ isinstance(node, dict) and str(node.get("_type", "")).endswith("InstructionSet")
353
+ for node in top
354
+ ):
355
+ return []
356
+
357
+ records: list[InstructionRecord] = []
358
+ seen_keys: set[tuple[str, str]] = set()
359
+
360
+ def walk(node: dict[str, Any], group_path: list[str]) -> None:
361
+ node_type = str(node.get("_type", ""))
362
+ if node_type.endswith("Instruction"):
363
+ assembly = node.get("assembly")
364
+ if not isinstance(assembly, dict):
365
+ return
366
+ rendered = _render_assembly(assembly)
367
+ if not rendered:
368
+ return
369
+ head, _, tail = rendered.partition(" ")
370
+ mnemonic = head.strip().upper()
371
+ if not mnemonic or not re.match(r"^[A-Z][A-Z0-9]*$", mnemonic):
372
+ return
373
+ operand_form = tail.strip()
374
+ form = _canonical_instruction_key(mnemonic, operand_form) if operand_form else mnemonic
375
+ key = (mnemonic, form.casefold())
376
+ if key in seen_keys:
377
+ return
378
+ seen_keys.add(key)
379
+ summary = _generated_summary(mnemonic)
380
+ if not summary.endswith("."):
381
+ summary += "."
382
+ metadata = {
383
+ "operation_id": str(node.get("operation_id") or "").strip(),
384
+ "instruction_set": group_path[0] if group_path else "A64",
385
+ "group": group_path[-1] if group_path else "",
386
+ }
387
+ metadata = {k: v for k, v in metadata.items() if v}
388
+ records.append(
389
+ InstructionRecord(
390
+ mnemonic=mnemonic,
391
+ form=form,
392
+ summary=summary,
393
+ architecture="arm",
394
+ isa=_infer_aarchmrs_isa(group_path),
395
+ metadata=metadata,
396
+ source="arm-a64",
397
+ )
398
+ )
399
+ return
400
+ if node_type.endswith("InstructionAlias"):
401
+ # Skip aliases — they resolve to another instruction we already parse.
402
+ return
403
+ name = str(node.get("name") or "").strip()
404
+ next_path = group_path + [name] if name else group_path
405
+ for child in node.get("children", []) or []:
406
+ if isinstance(child, dict):
407
+ walk(child, next_path)
408
+
409
+ for top_node in top:
410
+ if isinstance(top_node, dict):
411
+ walk(top_node, [str(top_node.get("name") or "")])
412
+ return records
413
+
414
+
415
+ def parse_arm_instruction_payload(text: str) -> list[InstructionRecord]:
416
+ payload = json.loads(text)
417
+ return _records_from_payload(payload)