simdref 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,625 @@
1
+ """Per-core LLVM scheduling pipeline (llvm-exegesis → llvm-mc → llvm-mca).
2
+
3
+ Three subprocess calls per core enumerate every LLVM-schedulable opcode,
4
+ disassemble its canonical asm form, and measure via ``llvm-mca
5
+ --instruction-tables=full --json``. The result is a list of
6
+ :class:`~simdref.perf_sources.merge.PerfRow` ready for
7
+ :func:`~simdref.perf_sources.merge.merge_perf_rows`.
8
+
9
+ No regex. Input is YAML (llvm-exegesis) and JSON (llvm-mca); output is
10
+ structured data.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import shutil
17
+ import subprocess
18
+ from pathlib import Path
19
+ from typing import Any
20
+
21
+ import yaml
22
+
23
+ from simdref.perf_sources.cores import CoreSpec
24
+ from simdref.perf_sources.llvm_mca import (
25
+ LLVM_MCA_CITATION,
26
+ LLVMMcaError,
27
+ LLVMMcaUnavailable,
28
+ )
29
+ from simdref.perf_sources.merge import PerfRow
30
+
31
+ _REPO_ROOT = Path(__file__).resolve().parents[3]
32
+ DEFAULT_CACHE_ROOT = _REPO_ROOT / "vendor" / "perf-cache"
33
+
34
+ # Minimum number of times a fixed-width chunk must appear inside a
35
+ # ``prepare-and-assemble-snippet`` buffer before we trust it as a real
36
+ # instruction (as opposed to a random slice of prologue / epilogue).
37
+ # llvm-exegesis always emits ≥ 4 repeats for target opcodes.
38
+ _MIN_REPEAT_RUN = 3
39
+
40
+
41
+ class LLVMSchedulingError(LLVMMcaError):
42
+ """Raised when the scheduling pipeline produces unusable output.
43
+
44
+ Subclasses :class:`LLVMMcaError` so existing ``except LLVMMcaError``
45
+ handlers keep working.
46
+ """
47
+
48
+
49
+ def _require(binary: str) -> str:
50
+ path = shutil.which(binary)
51
+ if path is None:
52
+ raise LLVMMcaUnavailable(f"{binary!r} not found on PATH")
53
+ return path
54
+
55
+
56
+ def _parse_exegesis_yaml(text: str) -> list[dict[str, str]]:
57
+ """Reduce llvm-exegesis multi-document YAML to ``[{opcode, snippet}]``.
58
+
59
+ The ``opcode`` is the first whitespace-delimited token of
60
+ ``key.instructions[0]`` (LLVM's MC opcode name, e.g. ``FADDv4f32``).
61
+ The ``snippet`` is the ``assembled_snippet`` hex string verbatim.
62
+ Docs missing either field are dropped.
63
+ """
64
+ entries: list[dict[str, str]] = []
65
+ for doc in yaml.safe_load_all(text):
66
+ if not isinstance(doc, dict):
67
+ continue
68
+ key_block = doc.get("key")
69
+ if not isinstance(key_block, dict):
70
+ continue
71
+ instructions = key_block.get("instructions")
72
+ if not isinstance(instructions, list) or not instructions:
73
+ continue
74
+ first = str(instructions[0] or "").strip()
75
+ if not first:
76
+ continue
77
+ opcode = first.split(None, 1)[0]
78
+ snippet = str(doc.get("assembled_snippet") or "").strip()
79
+ if not opcode or not snippet:
80
+ continue
81
+ entries.append({"opcode": opcode, "snippet": snippet})
82
+ return entries
83
+
84
+
85
+ def _parse_hex_snippet(hex_str: str) -> bytes:
86
+ """Decode a hex-encoded snippet to bytes (uppercase/lowercase agnostic)."""
87
+ try:
88
+ return bytes.fromhex(hex_str)
89
+ except ValueError as exc:
90
+ raise LLVMSchedulingError(f"malformed hex snippet: {hex_str[:40]}...") from exc
91
+
92
+
93
+ def _extract_repeated_chunks(snippet_hex: str, architecture: str) -> list[bytes]:
94
+ """Return every fixed-width chunk that appears ``≥ _MIN_REPEAT_RUN`` times.
95
+
96
+ llvm-exegesis emits one of two snippet shapes:
97
+
98
+ - ``prologue || opcode × N || epilogue`` — the common case;
99
+ - ``prologue || (opcode, breaker) × N || epilogue`` — when the
100
+ opcode depends on its own output, exegesis interleaves a breaker
101
+ instruction to keep the latency-chain isolated.
102
+
103
+ In the second shape neither chunk repeats *consecutively* but both
104
+ repeat by total count, so we count chunk frequency at every valid
105
+ byte alignment and keep everything that repeats often enough. The
106
+ breaker is a real ISA instruction and its scheduling data is just
107
+ as legitimate as the target's, so the caller can disassemble and
108
+ measure both without any additional bookkeeping.
109
+
110
+ AArch64 has 4-byte-aligned, fixed 4-byte instructions. RISC-V mixes
111
+ 2- and 4-byte instructions on 2-byte alignment, so we sweep both
112
+ widths and both starting byte offsets.
113
+ """
114
+ try:
115
+ data = _parse_hex_snippet(snippet_hex)
116
+ except LLVMSchedulingError:
117
+ return []
118
+ # AArch64 is fixed 4-byte on 4-byte boundaries → only offset 0 is
119
+ # meaningful. RISC-V is 2-byte aligned, so 4-byte windows must also
120
+ # be probed at offset 2. Sweeping every byte offset (as the prior
121
+ # revision did) picks up rotated views of the real pattern.
122
+ configurations: tuple[tuple[int, tuple[int, ...]], ...]
123
+ if architecture == "riscv":
124
+ configurations = ((4, (0, 2)), (2, (0,)))
125
+ else:
126
+ configurations = ((4, (0,)),)
127
+ kept: set[bytes] = set()
128
+ for width, byte_offsets in configurations:
129
+ for start_byte in byte_offsets:
130
+ counter: dict[bytes, int] = {}
131
+ pos = start_byte
132
+ while pos + width <= len(data):
133
+ chunk = data[pos : pos + width]
134
+ counter[chunk] = counter.get(chunk, 0) + 1
135
+ pos += width
136
+ for chunk, count in counter.items():
137
+ if count >= _MIN_REPEAT_RUN:
138
+ kept.add(chunk)
139
+ return sorted(kept)
140
+
141
+
142
+ def _hex_to_byte_line(data: bytes) -> str:
143
+ """Turn ``b'\\xef\\xb9 \\x20\\x4e'`` into ``'0xEF 0xB9 0x20 0x4E'``."""
144
+ return " ".join(f"0x{b:02X}" for b in data)
145
+
146
+
147
+ def build_byte_lines(
148
+ entries: list[dict[str, str]], architecture: str
149
+ ) -> list[str]:
150
+ """Map exegesis entries to disassembly-ready hex-byte lines.
151
+
152
+ Dedupes identical byte sequences across opcodes — different MC
153
+ opcodes occasionally encode to the same bytes, and feeding duplicates
154
+ to llvm-mc wastes work in the later stages.
155
+ """
156
+ seen: set[bytes] = set()
157
+ lines: list[str] = []
158
+ for entry in entries:
159
+ chunks = _extract_repeated_chunks(entry.get("snippet", ""), architecture)
160
+ for chunk in chunks:
161
+ if chunk in seen:
162
+ continue
163
+ seen.add(chunk)
164
+ lines.append(_hex_to_byte_line(chunk))
165
+ return lines
166
+
167
+
168
+ def _keep_disassembly_line(line: str) -> bool:
169
+ if not line.startswith("\t"):
170
+ return False
171
+ body = line.lstrip("\t").strip()
172
+ if not body:
173
+ return False
174
+ if body.startswith(".") or body.startswith("#"):
175
+ return False
176
+ return True
177
+
178
+
179
+ def _filter_disassembly(text: str) -> str:
180
+ """Drop ``.text`` directives, comments, and blank lines so llvm-mca
181
+ receives a clean asm stream."""
182
+ body = "\n".join(line for line in text.splitlines() if _keep_disassembly_line(line))
183
+ return body + "\n" if body else ""
184
+
185
+
186
+ def _run_exegesis(core: CoreSpec, output: Path, *, executable: str) -> None:
187
+ _require(executable)
188
+ cmd = [
189
+ executable,
190
+ "--mode=latency",
191
+ "--opcode-index=-1",
192
+ "--benchmark-phase=prepare-and-assemble-snippet",
193
+ f"--mtriple={core.llvm_triple}",
194
+ f"--mcpu={core.llvm_cpu}",
195
+ f"--benchmarks-file={output}",
196
+ ]
197
+ try:
198
+ proc = subprocess.run(
199
+ cmd, check=False, capture_output=True, text=True, timeout=600
200
+ )
201
+ except (subprocess.SubprocessError, OSError) as exc:
202
+ raise LLVMMcaUnavailable(f"{executable} failed: {exc}") from exc
203
+ if proc.returncode != 0 or not output.exists():
204
+ raise LLVMSchedulingError(
205
+ f"{executable} failed for {core.canonical_id} "
206
+ f"(triple={core.llvm_triple}, cpu={core.llvm_cpu}): "
207
+ f"exit={proc.returncode}; stderr={proc.stderr.strip()[:500]}"
208
+ )
209
+
210
+
211
+ def _run_disassemble(
212
+ hex_lines: list[str], core: CoreSpec, *, executable: str
213
+ ) -> str:
214
+ _require(executable)
215
+ stdin = "\n".join(hex_lines) + "\n"
216
+ cmd = [
217
+ executable,
218
+ f"--triple={core.llvm_triple}",
219
+ f"--mcpu={core.llvm_cpu}",
220
+ "--disassemble",
221
+ ]
222
+ try:
223
+ proc = subprocess.run(
224
+ cmd,
225
+ input=stdin,
226
+ check=False,
227
+ capture_output=True,
228
+ text=True,
229
+ timeout=300,
230
+ )
231
+ except (subprocess.SubprocessError, OSError) as exc:
232
+ raise LLVMMcaUnavailable(f"{executable} failed: {exc}") from exc
233
+ if proc.returncode != 0:
234
+ raise LLVMSchedulingError(
235
+ f"{executable} --disassemble failed for {core.canonical_id}: "
236
+ f"exit={proc.returncode}; stderr={proc.stderr.strip()[:500]}"
237
+ )
238
+ return proc.stdout
239
+
240
+
241
+ def _mca_command(core: CoreSpec, executable: str) -> list[str]:
242
+ return [
243
+ executable,
244
+ "--instruction-tables=full",
245
+ "--json",
246
+ # llvm-exegesis legally emits encodings that are architecturally
247
+ # "unpredictable" (LDP with Rt2==Rt, writeback with base in
248
+ # destination, etc.). llvm-mca refuses to schedule them by default
249
+ # — skip them so a handful of exotic corner cases don't fail the
250
+ # whole core.
251
+ "--skip-unsupported-instructions=any",
252
+ f"--mtriple={core.llvm_triple}",
253
+ f"--mcpu={core.llvm_cpu}",
254
+ ]
255
+
256
+
257
+ def _try_mca_once(
258
+ asm_text: str, core: CoreSpec, executable: str, timeout: float
259
+ ) -> tuple[int, str, str]:
260
+ """Run llvm-mca once and return ``(returncode, stdout, stderr)``."""
261
+ _require(executable)
262
+ try:
263
+ proc = subprocess.run(
264
+ _mca_command(core, executable),
265
+ input=asm_text,
266
+ check=False,
267
+ capture_output=True,
268
+ text=True,
269
+ timeout=timeout,
270
+ )
271
+ except (subprocess.SubprocessError, OSError) as exc:
272
+ raise LLVMMcaUnavailable(f"{executable} failed: {exc}") from exc
273
+ return proc.returncode, proc.stdout, proc.stderr
274
+
275
+
276
+ def _merge_mca_payloads(payloads: list[dict[str, Any]]) -> dict[str, Any]:
277
+ """Concatenate ``Instructions`` + ``InstructionInfoView`` across payloads.
278
+
279
+ The scheduling numbers in ``--instruction-tables=full`` are
280
+ per-instruction (no cross-instruction simulation state), so
281
+ concatenating is safe.
282
+ """
283
+ if not payloads:
284
+ return {"CodeRegions": []}
285
+ if len(payloads) == 1:
286
+ return payloads[0]
287
+ merged_instructions: list[Any] = []
288
+ merged_info: list[Any] = []
289
+ merged_pressure: list[dict[str, Any]] = []
290
+ template = payloads[0]
291
+ for payload in payloads:
292
+ regions = payload.get("CodeRegions") or []
293
+ if not regions:
294
+ continue
295
+ region = regions[0]
296
+ offset = len(merged_info)
297
+ merged_instructions.extend(region.get("Instructions") or [])
298
+ merged_info.extend(
299
+ (region.get("InstructionInfoView") or {}).get("InstructionList") or []
300
+ )
301
+ # ``ResourcePressureInfo`` entries carry a per-payload
302
+ # ``InstructionIndex``. Re-index into the merged space so the
303
+ # join in :func:`_pressure_by_index` stays consistent.
304
+ for entry in (region.get("ResourcePressureView") or {}).get("ResourcePressureInfo") or []:
305
+ if not isinstance(entry, dict):
306
+ continue
307
+ shifted = dict(entry)
308
+ if isinstance(shifted.get("InstructionIndex"), int):
309
+ shifted["InstructionIndex"] = shifted["InstructionIndex"] + offset
310
+ merged_pressure.append(shifted)
311
+ first_region = (template.get("CodeRegions") or [{}])[0]
312
+ return {
313
+ **template,
314
+ "CodeRegions": [
315
+ {
316
+ **first_region,
317
+ "Instructions": merged_instructions,
318
+ "InstructionInfoView": {
319
+ **(first_region.get("InstructionInfoView") or {}),
320
+ "InstructionList": merged_info,
321
+ },
322
+ "ResourcePressureView": {
323
+ **(first_region.get("ResourcePressureView") or {}),
324
+ "ResourcePressureInfo": merged_pressure,
325
+ },
326
+ }
327
+ ],
328
+ }
329
+
330
+
331
+ def _run_mca(
332
+ asm_text: str,
333
+ core: CoreSpec,
334
+ *,
335
+ executable: str,
336
+ timeout: float = 600.0,
337
+ ) -> dict[str, Any]:
338
+ """Run llvm-mca, recursively quarantining lines that crash it.
339
+
340
+ llvm-mca (22.x) segfaults on a handful of weird RVV encodings even
341
+ under ``--skip-unsupported-instructions=any`` — a known upstream
342
+ bug. Rather than failing the whole core, we bisect the input on
343
+ each crash, drop the single offending line, and merge the surviving
344
+ halves' payloads.
345
+ """
346
+ lines = [line for line in asm_text.splitlines() if line.strip()]
347
+ if not lines:
348
+ return {"CodeRegions": []}
349
+
350
+ def run_block(block: list[str]) -> dict[str, Any]:
351
+ if not block:
352
+ return {"CodeRegions": []}
353
+ text = "\n".join(block) + "\n"
354
+ rc, stdout, _ = _try_mca_once(text, core, executable, timeout)
355
+ if rc == 0:
356
+ try:
357
+ return json.loads(stdout)
358
+ except json.JSONDecodeError as exc:
359
+ raise LLVMSchedulingError(
360
+ f"{executable} emitted non-JSON for {core.canonical_id}: {exc}"
361
+ ) from exc
362
+ if len(block) == 1:
363
+ # Single line crashed llvm-mca — quarantine and move on. This
364
+ # is where the upstream crash bug lives; we cannot do better.
365
+ return {"CodeRegions": []}
366
+ mid = len(block) // 2
367
+ left = run_block(block[:mid])
368
+ right = run_block(block[mid:])
369
+ # If a clean rerun is possible (both halves survived individually
370
+ # but something about their combination wasn't the problem),
371
+ # prefer the merged payload. Either way we merge what we have.
372
+ return _merge_mca_payloads([left, right])
373
+
374
+ return run_block(lines)
375
+
376
+
377
+ def _asm_mnemonic(line: str) -> str:
378
+ """First whitespace-delimited token of an llvm-mc asm line, uppercased."""
379
+ normalized = line.replace("\t", " ").strip()
380
+ if not normalized:
381
+ return ""
382
+ return normalized.split(None, 1)[0].upper()
383
+
384
+
385
+ def _format_port_name(name: str) -> str:
386
+ """Normalise raw LLVM resource names for a readable ``ports`` column.
387
+
388
+ llvm-mca emits resources shaped like ``N1UnitV0`` or ``N1UnitD.\\x00``
389
+ (the trailing bytes disambiguate sub-units of a replicated resource).
390
+ We strip a short ``<vendor>Unit`` prefix when present, and then drop
391
+ any non-printable / ``.`` suffix so ``N1UnitD.\\x00`` → ``D0``.
392
+
393
+ Names without a recognised prefix (e.g. RISC-V ``SiFive7VA1``) are
394
+ returned unchanged — a slightly longer label is still readable.
395
+ """
396
+ prefix, sep, suffix = name.partition("Unit")
397
+ core = suffix if sep and suffix and prefix and len(prefix) <= 5 and prefix[0].isupper() else name
398
+ if "." in core:
399
+ head, _, tail = core.partition(".")
400
+ # Map ``D.\x00`` / ``D.\x01`` → ``D0`` / ``D1`` so the column stays
401
+ # printable and distinguishes the replicated sub-units.
402
+ index_bytes = [b for b in tail.encode("utf-8", errors="ignore") if b < 32]
403
+ if index_bytes:
404
+ return f"{head}{index_bytes[0]}"
405
+ printable_tail = "".join(ch for ch in tail if ch.isprintable() and ch != "\x00")
406
+ return f"{head}{printable_tail}" if printable_tail else head
407
+ return core
408
+
409
+
410
+ def _pressure_by_index(
411
+ payload: dict[str, Any],
412
+ payload_region: dict[str, Any],
413
+ ) -> dict[int, list[tuple[str, float]]]:
414
+ """Build ``{InstructionIndex: [(port, usage), ...]}`` from llvm-mca.
415
+
416
+ ``TargetInfo.Resources`` lives at the payload top level, while
417
+ ``ResourcePressureView.ResourcePressureInfo`` is per-region. Joins
418
+ the two tables and drops entries whose ``ResourceUsage`` rounds to
419
+ zero at two decimal places.
420
+ """
421
+ target_info = payload.get("TargetInfo") or payload_region.get("TargetInfo") or {}
422
+ resources = target_info.get("Resources") or []
423
+ view = payload_region.get("ResourcePressureView") or {}
424
+ pressure_info = view.get("ResourcePressureInfo") or []
425
+ result: dict[int, list[tuple[str, float]]] = {}
426
+ for entry in pressure_info:
427
+ if not isinstance(entry, dict):
428
+ continue
429
+ idx = entry.get("InstructionIndex")
430
+ res_idx = entry.get("ResourceIndex")
431
+ usage = entry.get("ResourceUsage")
432
+ if not isinstance(idx, int) or not isinstance(res_idx, int):
433
+ continue
434
+ if not isinstance(usage, (int, float)) or round(float(usage), 2) <= 0.0:
435
+ continue
436
+ if res_idx < 0 or res_idx >= len(resources):
437
+ continue
438
+ raw_name = str(resources[res_idx] or "")
439
+ if not raw_name:
440
+ continue
441
+ port = _format_port_name(raw_name)
442
+ result.setdefault(idx, []).append((port, float(usage)))
443
+ return result
444
+
445
+
446
+ def _format_ports(entries: list[tuple[str, float]]) -> str:
447
+ """``[('V0', 0.5), ('V1', 0.5)]`` → ``'0.50*V0 0.50*V1'``.
448
+
449
+ Uses uops.info's ``count*port`` convention so the ARM / RISC-V modeled
450
+ ``ports`` column lines up with the format readers already expect from
451
+ the x86 measured side.
452
+ """
453
+ return " ".join(f"{usage:.2f}*{name}" for name, usage in entries)
454
+
455
+
456
+ def _kind_label(info: dict[str, Any]) -> str:
457
+ """Map ``mayLoad`` / ``mayStore`` flags to a compact label or ``''``."""
458
+ may_load = bool(info.get("mayLoad"))
459
+ may_store = bool(info.get("mayStore"))
460
+ if may_load and may_store:
461
+ return "load+store"
462
+ if may_load:
463
+ return "load"
464
+ if may_store:
465
+ return "store"
466
+ return ""
467
+
468
+
469
+ def _build_perf_rows(
470
+ payload: dict[str, Any],
471
+ asm_lines: list[str],
472
+ *,
473
+ core: CoreSpec,
474
+ mca_version: str,
475
+ ) -> list[PerfRow]:
476
+ """Join an ``llvm-mca --json`` payload to the asm lines that produced it.
477
+
478
+ ``payload`` is the parsed ``--instruction-tables=full --json``
479
+ output. The payload's ``InstructionInfoView.InstructionList`` is
480
+ parallel to the payload's own ``Instructions`` array, which reflects
481
+ any lines that llvm-mca skipped via
482
+ ``--skip-unsupported-instructions``. When the two lists' lengths
483
+ match we prefer the payload's labels as the ground truth; otherwise
484
+ we fall back to the caller-supplied ``asm_lines``.
485
+
486
+ Emits one :class:`PerfRow` per unique mnemonic (first-seen wins —
487
+ same-mnemonic form variants contend for the same
488
+ ``arch_details[core]`` slot anyway in
489
+ :func:`merge_perf_rows`).
490
+ """
491
+ regions = payload.get("CodeRegions") or []
492
+ if not regions:
493
+ return []
494
+ region = regions[0]
495
+ info_list = (region.get("InstructionInfoView") or {}).get("InstructionList") or []
496
+ payload_insts = region.get("Instructions") or []
497
+ labels: list[str]
498
+ if len(payload_insts) == len(info_list) and all(
499
+ isinstance(item, str) for item in payload_insts
500
+ ):
501
+ labels = list(payload_insts)
502
+ else:
503
+ labels = list(asm_lines)
504
+ arch_label = "arm" if core.architecture == "aarch64" else core.architecture
505
+ pressure_by_index = _pressure_by_index(payload, region)
506
+
507
+ rows: list[PerfRow] = []
508
+ seen: set[str] = set()
509
+ for idx, (asm_line, info) in enumerate(zip(labels, info_list)):
510
+ if not isinstance(info, dict):
511
+ continue
512
+ mnemonic = _asm_mnemonic(asm_line)
513
+ if not mnemonic or mnemonic in seen:
514
+ continue
515
+ seen.add(mnemonic)
516
+ latency = info.get("Latency")
517
+ rthroughput = info.get("RThroughput")
518
+ cpi = ""
519
+ if isinstance(rthroughput, (int, float)):
520
+ cpi = f"{float(rthroughput):.3f}".rstrip("0").rstrip(".")
521
+ extra: dict[str, str] = {}
522
+ num_uops = info.get("NumMicroOpcodes")
523
+ if isinstance(num_uops, (int, float)) and num_uops > 0:
524
+ extra["uops"] = str(int(num_uops))
525
+ ports_str = _format_ports(pressure_by_index.get(idx, []))
526
+ if ports_str:
527
+ extra["ports"] = ports_str
528
+ kind = _kind_label(info)
529
+ if kind:
530
+ extra["kind"] = kind
531
+ rows.append(
532
+ PerfRow(
533
+ mnemonic=mnemonic,
534
+ core=core.canonical_id,
535
+ source="llvm-mca",
536
+ source_kind="modeled",
537
+ source_version=mca_version,
538
+ architecture=arch_label,
539
+ latency=str(latency) if latency is not None else "",
540
+ cpi=cpi,
541
+ applies_to="mnemonic",
542
+ citation_url=LLVM_MCA_CITATION,
543
+ extra_measurement=extra,
544
+ )
545
+ )
546
+ return rows
547
+
548
+
549
+ def _disassembly_to_asm_lines(disasm_text: str) -> list[str]:
550
+ """Turn ``_filter_disassembly`` output back into a list of asm lines."""
551
+ return [line for line in disasm_text.splitlines() if line.strip()]
552
+
553
+
554
+ def collect_core_schedule(
555
+ core: CoreSpec,
556
+ *,
557
+ mca_version: str,
558
+ cache_root: Path | None = None,
559
+ executable_exegesis: str = "llvm-exegesis",
560
+ executable_mc: str = "llvm-mc",
561
+ executable_mca: str = "llvm-mca",
562
+ ) -> list[PerfRow]:
563
+ """Run the llvm-exegesis → llvm-mc → llvm-mca pipeline for one core.
564
+
565
+ Returns one :class:`PerfRow` per unique assembly mnemonic that
566
+ llvm-mca's scheduling model produces latency / throughput data for.
567
+
568
+ Cache layout: ``<cache_root>/<triple>/<cpu>/{exegesis.yaml,
569
+ disassembly.s, mca.json}``. Callers pin ``mca_version`` into
570
+ ``cache_root`` themselves if they want LLVM-version isolation.
571
+
572
+ Raises :class:`LLVMMcaUnavailable` when a required binary is missing
573
+ and :class:`LLVMSchedulingError` (a ``LLVMMcaError`` subclass) on
574
+ any subprocess failure, empty result, or malformed output — no
575
+ silent fallback.
576
+ """
577
+ if core.architecture not in {"aarch64", "riscv"}:
578
+ return []
579
+
580
+ root = cache_root if cache_root is not None else DEFAULT_CACHE_ROOT
581
+ cache_dir = root / core.llvm_triple / core.llvm_cpu
582
+ exegesis_path = cache_dir / "exegesis.yaml"
583
+ disasm_path = cache_dir / "disassembly.s"
584
+ mca_path = cache_dir / "mca.json"
585
+ cache_dir.mkdir(parents=True, exist_ok=True)
586
+
587
+ if not exegesis_path.exists():
588
+ _run_exegesis(core, exegesis_path, executable=executable_exegesis)
589
+
590
+ entries = _parse_exegesis_yaml(exegesis_path.read_text())
591
+ if not entries:
592
+ raise LLVMSchedulingError(
593
+ f"{executable_exegesis} produced no entries for {core.canonical_id}"
594
+ )
595
+
596
+ if not disasm_path.exists():
597
+ hex_lines = build_byte_lines(entries, core.architecture)
598
+ if not hex_lines:
599
+ raise LLVMSchedulingError(
600
+ f"could not extract any opcode bytes from {exegesis_path}"
601
+ )
602
+ raw = _run_disassemble(hex_lines, core, executable=executable_mc)
603
+ disasm_path.write_text(_filter_disassembly(raw))
604
+
605
+ disasm_text = disasm_path.read_text()
606
+ asm_lines = _disassembly_to_asm_lines(disasm_text)
607
+ if not asm_lines:
608
+ raise LLVMSchedulingError(
609
+ f"{executable_mc} produced empty disassembly for {core.canonical_id}"
610
+ )
611
+
612
+ if not mca_path.exists():
613
+ payload = _run_mca(disasm_text, core, executable=executable_mca)
614
+ mca_path.write_text(json.dumps(payload))
615
+ else:
616
+ payload = json.loads(mca_path.read_text())
617
+
618
+ rows = _build_perf_rows(
619
+ payload, asm_lines, core=core, mca_version=mca_version
620
+ )
621
+ if not rows:
622
+ raise LLVMSchedulingError(
623
+ f"{executable_mca} produced no schedule rows for {core.canonical_id}"
624
+ )
625
+ return rows