simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,625 @@
|
|
|
1
|
+
"""Per-core LLVM scheduling pipeline (llvm-exegesis → llvm-mc → llvm-mca).
|
|
2
|
+
|
|
3
|
+
Three subprocess calls per core enumerate every LLVM-schedulable opcode,
|
|
4
|
+
disassemble its canonical asm form, and measure via ``llvm-mca
|
|
5
|
+
--instruction-tables=full --json``. The result is a list of
|
|
6
|
+
:class:`~simdref.perf_sources.merge.PerfRow` ready for
|
|
7
|
+
:func:`~simdref.perf_sources.merge.merge_perf_rows`.
|
|
8
|
+
|
|
9
|
+
No regex. Input is YAML (llvm-exegesis) and JSON (llvm-mca); output is
|
|
10
|
+
structured data.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
import yaml
|
|
22
|
+
|
|
23
|
+
from simdref.perf_sources.cores import CoreSpec
|
|
24
|
+
from simdref.perf_sources.llvm_mca import (
|
|
25
|
+
LLVM_MCA_CITATION,
|
|
26
|
+
LLVMMcaError,
|
|
27
|
+
LLVMMcaUnavailable,
|
|
28
|
+
)
|
|
29
|
+
from simdref.perf_sources.merge import PerfRow
|
|
30
|
+
|
|
31
|
+
_REPO_ROOT = Path(__file__).resolve().parents[3]
|
|
32
|
+
DEFAULT_CACHE_ROOT = _REPO_ROOT / "vendor" / "perf-cache"
|
|
33
|
+
|
|
34
|
+
# Minimum number of times a fixed-width chunk must appear inside a
|
|
35
|
+
# ``prepare-and-assemble-snippet`` buffer before we trust it as a real
|
|
36
|
+
# instruction (as opposed to a random slice of prologue / epilogue).
|
|
37
|
+
# llvm-exegesis always emits ≥ 4 repeats for target opcodes.
|
|
38
|
+
_MIN_REPEAT_RUN = 3
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class LLVMSchedulingError(LLVMMcaError):
|
|
42
|
+
"""Raised when the scheduling pipeline produces unusable output.
|
|
43
|
+
|
|
44
|
+
Subclasses :class:`LLVMMcaError` so existing ``except LLVMMcaError``
|
|
45
|
+
handlers keep working.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _require(binary: str) -> str:
|
|
50
|
+
path = shutil.which(binary)
|
|
51
|
+
if path is None:
|
|
52
|
+
raise LLVMMcaUnavailable(f"{binary!r} not found on PATH")
|
|
53
|
+
return path
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _parse_exegesis_yaml(text: str) -> list[dict[str, str]]:
|
|
57
|
+
"""Reduce llvm-exegesis multi-document YAML to ``[{opcode, snippet}]``.
|
|
58
|
+
|
|
59
|
+
The ``opcode`` is the first whitespace-delimited token of
|
|
60
|
+
``key.instructions[0]`` (LLVM's MC opcode name, e.g. ``FADDv4f32``).
|
|
61
|
+
The ``snippet`` is the ``assembled_snippet`` hex string verbatim.
|
|
62
|
+
Docs missing either field are dropped.
|
|
63
|
+
"""
|
|
64
|
+
entries: list[dict[str, str]] = []
|
|
65
|
+
for doc in yaml.safe_load_all(text):
|
|
66
|
+
if not isinstance(doc, dict):
|
|
67
|
+
continue
|
|
68
|
+
key_block = doc.get("key")
|
|
69
|
+
if not isinstance(key_block, dict):
|
|
70
|
+
continue
|
|
71
|
+
instructions = key_block.get("instructions")
|
|
72
|
+
if not isinstance(instructions, list) or not instructions:
|
|
73
|
+
continue
|
|
74
|
+
first = str(instructions[0] or "").strip()
|
|
75
|
+
if not first:
|
|
76
|
+
continue
|
|
77
|
+
opcode = first.split(None, 1)[0]
|
|
78
|
+
snippet = str(doc.get("assembled_snippet") or "").strip()
|
|
79
|
+
if not opcode or not snippet:
|
|
80
|
+
continue
|
|
81
|
+
entries.append({"opcode": opcode, "snippet": snippet})
|
|
82
|
+
return entries
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _parse_hex_snippet(hex_str: str) -> bytes:
|
|
86
|
+
"""Decode a hex-encoded snippet to bytes (uppercase/lowercase agnostic)."""
|
|
87
|
+
try:
|
|
88
|
+
return bytes.fromhex(hex_str)
|
|
89
|
+
except ValueError as exc:
|
|
90
|
+
raise LLVMSchedulingError(f"malformed hex snippet: {hex_str[:40]}...") from exc
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _extract_repeated_chunks(snippet_hex: str, architecture: str) -> list[bytes]:
|
|
94
|
+
"""Return every fixed-width chunk that appears ``≥ _MIN_REPEAT_RUN`` times.
|
|
95
|
+
|
|
96
|
+
llvm-exegesis emits one of two snippet shapes:
|
|
97
|
+
|
|
98
|
+
- ``prologue || opcode × N || epilogue`` — the common case;
|
|
99
|
+
- ``prologue || (opcode, breaker) × N || epilogue`` — when the
|
|
100
|
+
opcode depends on its own output, exegesis interleaves a breaker
|
|
101
|
+
instruction to keep the latency-chain isolated.
|
|
102
|
+
|
|
103
|
+
In the second shape neither chunk repeats *consecutively* but both
|
|
104
|
+
repeat by total count, so we count chunk frequency at every valid
|
|
105
|
+
byte alignment and keep everything that repeats often enough. The
|
|
106
|
+
breaker is a real ISA instruction and its scheduling data is just
|
|
107
|
+
as legitimate as the target's, so the caller can disassemble and
|
|
108
|
+
measure both without any additional bookkeeping.
|
|
109
|
+
|
|
110
|
+
AArch64 has 4-byte-aligned, fixed 4-byte instructions. RISC-V mixes
|
|
111
|
+
2- and 4-byte instructions on 2-byte alignment, so we sweep both
|
|
112
|
+
widths and both starting byte offsets.
|
|
113
|
+
"""
|
|
114
|
+
try:
|
|
115
|
+
data = _parse_hex_snippet(snippet_hex)
|
|
116
|
+
except LLVMSchedulingError:
|
|
117
|
+
return []
|
|
118
|
+
# AArch64 is fixed 4-byte on 4-byte boundaries → only offset 0 is
|
|
119
|
+
# meaningful. RISC-V is 2-byte aligned, so 4-byte windows must also
|
|
120
|
+
# be probed at offset 2. Sweeping every byte offset (as the prior
|
|
121
|
+
# revision did) picks up rotated views of the real pattern.
|
|
122
|
+
configurations: tuple[tuple[int, tuple[int, ...]], ...]
|
|
123
|
+
if architecture == "riscv":
|
|
124
|
+
configurations = ((4, (0, 2)), (2, (0,)))
|
|
125
|
+
else:
|
|
126
|
+
configurations = ((4, (0,)),)
|
|
127
|
+
kept: set[bytes] = set()
|
|
128
|
+
for width, byte_offsets in configurations:
|
|
129
|
+
for start_byte in byte_offsets:
|
|
130
|
+
counter: dict[bytes, int] = {}
|
|
131
|
+
pos = start_byte
|
|
132
|
+
while pos + width <= len(data):
|
|
133
|
+
chunk = data[pos : pos + width]
|
|
134
|
+
counter[chunk] = counter.get(chunk, 0) + 1
|
|
135
|
+
pos += width
|
|
136
|
+
for chunk, count in counter.items():
|
|
137
|
+
if count >= _MIN_REPEAT_RUN:
|
|
138
|
+
kept.add(chunk)
|
|
139
|
+
return sorted(kept)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _hex_to_byte_line(data: bytes) -> str:
|
|
143
|
+
"""Turn ``b'\\xef\\xb9 \\x20\\x4e'`` into ``'0xEF 0xB9 0x20 0x4E'``."""
|
|
144
|
+
return " ".join(f"0x{b:02X}" for b in data)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def build_byte_lines(
|
|
148
|
+
entries: list[dict[str, str]], architecture: str
|
|
149
|
+
) -> list[str]:
|
|
150
|
+
"""Map exegesis entries to disassembly-ready hex-byte lines.
|
|
151
|
+
|
|
152
|
+
Dedupes identical byte sequences across opcodes — different MC
|
|
153
|
+
opcodes occasionally encode to the same bytes, and feeding duplicates
|
|
154
|
+
to llvm-mc wastes work in the later stages.
|
|
155
|
+
"""
|
|
156
|
+
seen: set[bytes] = set()
|
|
157
|
+
lines: list[str] = []
|
|
158
|
+
for entry in entries:
|
|
159
|
+
chunks = _extract_repeated_chunks(entry.get("snippet", ""), architecture)
|
|
160
|
+
for chunk in chunks:
|
|
161
|
+
if chunk in seen:
|
|
162
|
+
continue
|
|
163
|
+
seen.add(chunk)
|
|
164
|
+
lines.append(_hex_to_byte_line(chunk))
|
|
165
|
+
return lines
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _keep_disassembly_line(line: str) -> bool:
|
|
169
|
+
if not line.startswith("\t"):
|
|
170
|
+
return False
|
|
171
|
+
body = line.lstrip("\t").strip()
|
|
172
|
+
if not body:
|
|
173
|
+
return False
|
|
174
|
+
if body.startswith(".") or body.startswith("#"):
|
|
175
|
+
return False
|
|
176
|
+
return True
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _filter_disassembly(text: str) -> str:
|
|
180
|
+
"""Drop ``.text`` directives, comments, and blank lines so llvm-mca
|
|
181
|
+
receives a clean asm stream."""
|
|
182
|
+
body = "\n".join(line for line in text.splitlines() if _keep_disassembly_line(line))
|
|
183
|
+
return body + "\n" if body else ""
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _run_exegesis(core: CoreSpec, output: Path, *, executable: str) -> None:
|
|
187
|
+
_require(executable)
|
|
188
|
+
cmd = [
|
|
189
|
+
executable,
|
|
190
|
+
"--mode=latency",
|
|
191
|
+
"--opcode-index=-1",
|
|
192
|
+
"--benchmark-phase=prepare-and-assemble-snippet",
|
|
193
|
+
f"--mtriple={core.llvm_triple}",
|
|
194
|
+
f"--mcpu={core.llvm_cpu}",
|
|
195
|
+
f"--benchmarks-file={output}",
|
|
196
|
+
]
|
|
197
|
+
try:
|
|
198
|
+
proc = subprocess.run(
|
|
199
|
+
cmd, check=False, capture_output=True, text=True, timeout=600
|
|
200
|
+
)
|
|
201
|
+
except (subprocess.SubprocessError, OSError) as exc:
|
|
202
|
+
raise LLVMMcaUnavailable(f"{executable} failed: {exc}") from exc
|
|
203
|
+
if proc.returncode != 0 or not output.exists():
|
|
204
|
+
raise LLVMSchedulingError(
|
|
205
|
+
f"{executable} failed for {core.canonical_id} "
|
|
206
|
+
f"(triple={core.llvm_triple}, cpu={core.llvm_cpu}): "
|
|
207
|
+
f"exit={proc.returncode}; stderr={proc.stderr.strip()[:500]}"
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _run_disassemble(
|
|
212
|
+
hex_lines: list[str], core: CoreSpec, *, executable: str
|
|
213
|
+
) -> str:
|
|
214
|
+
_require(executable)
|
|
215
|
+
stdin = "\n".join(hex_lines) + "\n"
|
|
216
|
+
cmd = [
|
|
217
|
+
executable,
|
|
218
|
+
f"--triple={core.llvm_triple}",
|
|
219
|
+
f"--mcpu={core.llvm_cpu}",
|
|
220
|
+
"--disassemble",
|
|
221
|
+
]
|
|
222
|
+
try:
|
|
223
|
+
proc = subprocess.run(
|
|
224
|
+
cmd,
|
|
225
|
+
input=stdin,
|
|
226
|
+
check=False,
|
|
227
|
+
capture_output=True,
|
|
228
|
+
text=True,
|
|
229
|
+
timeout=300,
|
|
230
|
+
)
|
|
231
|
+
except (subprocess.SubprocessError, OSError) as exc:
|
|
232
|
+
raise LLVMMcaUnavailable(f"{executable} failed: {exc}") from exc
|
|
233
|
+
if proc.returncode != 0:
|
|
234
|
+
raise LLVMSchedulingError(
|
|
235
|
+
f"{executable} --disassemble failed for {core.canonical_id}: "
|
|
236
|
+
f"exit={proc.returncode}; stderr={proc.stderr.strip()[:500]}"
|
|
237
|
+
)
|
|
238
|
+
return proc.stdout
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _mca_command(core: CoreSpec, executable: str) -> list[str]:
|
|
242
|
+
return [
|
|
243
|
+
executable,
|
|
244
|
+
"--instruction-tables=full",
|
|
245
|
+
"--json",
|
|
246
|
+
# llvm-exegesis legally emits encodings that are architecturally
|
|
247
|
+
# "unpredictable" (LDP with Rt2==Rt, writeback with base in
|
|
248
|
+
# destination, etc.). llvm-mca refuses to schedule them by default
|
|
249
|
+
# — skip them so a handful of exotic corner cases don't fail the
|
|
250
|
+
# whole core.
|
|
251
|
+
"--skip-unsupported-instructions=any",
|
|
252
|
+
f"--mtriple={core.llvm_triple}",
|
|
253
|
+
f"--mcpu={core.llvm_cpu}",
|
|
254
|
+
]
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _try_mca_once(
|
|
258
|
+
asm_text: str, core: CoreSpec, executable: str, timeout: float
|
|
259
|
+
) -> tuple[int, str, str]:
|
|
260
|
+
"""Run llvm-mca once and return ``(returncode, stdout, stderr)``."""
|
|
261
|
+
_require(executable)
|
|
262
|
+
try:
|
|
263
|
+
proc = subprocess.run(
|
|
264
|
+
_mca_command(core, executable),
|
|
265
|
+
input=asm_text,
|
|
266
|
+
check=False,
|
|
267
|
+
capture_output=True,
|
|
268
|
+
text=True,
|
|
269
|
+
timeout=timeout,
|
|
270
|
+
)
|
|
271
|
+
except (subprocess.SubprocessError, OSError) as exc:
|
|
272
|
+
raise LLVMMcaUnavailable(f"{executable} failed: {exc}") from exc
|
|
273
|
+
return proc.returncode, proc.stdout, proc.stderr
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _merge_mca_payloads(payloads: list[dict[str, Any]]) -> dict[str, Any]:
|
|
277
|
+
"""Concatenate ``Instructions`` + ``InstructionInfoView`` across payloads.
|
|
278
|
+
|
|
279
|
+
The scheduling numbers in ``--instruction-tables=full`` are
|
|
280
|
+
per-instruction (no cross-instruction simulation state), so
|
|
281
|
+
concatenating is safe.
|
|
282
|
+
"""
|
|
283
|
+
if not payloads:
|
|
284
|
+
return {"CodeRegions": []}
|
|
285
|
+
if len(payloads) == 1:
|
|
286
|
+
return payloads[0]
|
|
287
|
+
merged_instructions: list[Any] = []
|
|
288
|
+
merged_info: list[Any] = []
|
|
289
|
+
merged_pressure: list[dict[str, Any]] = []
|
|
290
|
+
template = payloads[0]
|
|
291
|
+
for payload in payloads:
|
|
292
|
+
regions = payload.get("CodeRegions") or []
|
|
293
|
+
if not regions:
|
|
294
|
+
continue
|
|
295
|
+
region = regions[0]
|
|
296
|
+
offset = len(merged_info)
|
|
297
|
+
merged_instructions.extend(region.get("Instructions") or [])
|
|
298
|
+
merged_info.extend(
|
|
299
|
+
(region.get("InstructionInfoView") or {}).get("InstructionList") or []
|
|
300
|
+
)
|
|
301
|
+
# ``ResourcePressureInfo`` entries carry a per-payload
|
|
302
|
+
# ``InstructionIndex``. Re-index into the merged space so the
|
|
303
|
+
# join in :func:`_pressure_by_index` stays consistent.
|
|
304
|
+
for entry in (region.get("ResourcePressureView") or {}).get("ResourcePressureInfo") or []:
|
|
305
|
+
if not isinstance(entry, dict):
|
|
306
|
+
continue
|
|
307
|
+
shifted = dict(entry)
|
|
308
|
+
if isinstance(shifted.get("InstructionIndex"), int):
|
|
309
|
+
shifted["InstructionIndex"] = shifted["InstructionIndex"] + offset
|
|
310
|
+
merged_pressure.append(shifted)
|
|
311
|
+
first_region = (template.get("CodeRegions") or [{}])[0]
|
|
312
|
+
return {
|
|
313
|
+
**template,
|
|
314
|
+
"CodeRegions": [
|
|
315
|
+
{
|
|
316
|
+
**first_region,
|
|
317
|
+
"Instructions": merged_instructions,
|
|
318
|
+
"InstructionInfoView": {
|
|
319
|
+
**(first_region.get("InstructionInfoView") or {}),
|
|
320
|
+
"InstructionList": merged_info,
|
|
321
|
+
},
|
|
322
|
+
"ResourcePressureView": {
|
|
323
|
+
**(first_region.get("ResourcePressureView") or {}),
|
|
324
|
+
"ResourcePressureInfo": merged_pressure,
|
|
325
|
+
},
|
|
326
|
+
}
|
|
327
|
+
],
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _run_mca(
|
|
332
|
+
asm_text: str,
|
|
333
|
+
core: CoreSpec,
|
|
334
|
+
*,
|
|
335
|
+
executable: str,
|
|
336
|
+
timeout: float = 600.0,
|
|
337
|
+
) -> dict[str, Any]:
|
|
338
|
+
"""Run llvm-mca, recursively quarantining lines that crash it.
|
|
339
|
+
|
|
340
|
+
llvm-mca (22.x) segfaults on a handful of weird RVV encodings even
|
|
341
|
+
under ``--skip-unsupported-instructions=any`` — a known upstream
|
|
342
|
+
bug. Rather than failing the whole core, we bisect the input on
|
|
343
|
+
each crash, drop the single offending line, and merge the surviving
|
|
344
|
+
halves' payloads.
|
|
345
|
+
"""
|
|
346
|
+
lines = [line for line in asm_text.splitlines() if line.strip()]
|
|
347
|
+
if not lines:
|
|
348
|
+
return {"CodeRegions": []}
|
|
349
|
+
|
|
350
|
+
def run_block(block: list[str]) -> dict[str, Any]:
|
|
351
|
+
if not block:
|
|
352
|
+
return {"CodeRegions": []}
|
|
353
|
+
text = "\n".join(block) + "\n"
|
|
354
|
+
rc, stdout, _ = _try_mca_once(text, core, executable, timeout)
|
|
355
|
+
if rc == 0:
|
|
356
|
+
try:
|
|
357
|
+
return json.loads(stdout)
|
|
358
|
+
except json.JSONDecodeError as exc:
|
|
359
|
+
raise LLVMSchedulingError(
|
|
360
|
+
f"{executable} emitted non-JSON for {core.canonical_id}: {exc}"
|
|
361
|
+
) from exc
|
|
362
|
+
if len(block) == 1:
|
|
363
|
+
# Single line crashed llvm-mca — quarantine and move on. This
|
|
364
|
+
# is where the upstream crash bug lives; we cannot do better.
|
|
365
|
+
return {"CodeRegions": []}
|
|
366
|
+
mid = len(block) // 2
|
|
367
|
+
left = run_block(block[:mid])
|
|
368
|
+
right = run_block(block[mid:])
|
|
369
|
+
# If a clean rerun is possible (both halves survived individually
|
|
370
|
+
# but something about their combination wasn't the problem),
|
|
371
|
+
# prefer the merged payload. Either way we merge what we have.
|
|
372
|
+
return _merge_mca_payloads([left, right])
|
|
373
|
+
|
|
374
|
+
return run_block(lines)
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def _asm_mnemonic(line: str) -> str:
|
|
378
|
+
"""First whitespace-delimited token of an llvm-mc asm line, uppercased."""
|
|
379
|
+
normalized = line.replace("\t", " ").strip()
|
|
380
|
+
if not normalized:
|
|
381
|
+
return ""
|
|
382
|
+
return normalized.split(None, 1)[0].upper()
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _format_port_name(name: str) -> str:
|
|
386
|
+
"""Normalise raw LLVM resource names for a readable ``ports`` column.
|
|
387
|
+
|
|
388
|
+
llvm-mca emits resources shaped like ``N1UnitV0`` or ``N1UnitD.\\x00``
|
|
389
|
+
(the trailing bytes disambiguate sub-units of a replicated resource).
|
|
390
|
+
We strip a short ``<vendor>Unit`` prefix when present, and then drop
|
|
391
|
+
any non-printable / ``.`` suffix so ``N1UnitD.\\x00`` → ``D0``.
|
|
392
|
+
|
|
393
|
+
Names without a recognised prefix (e.g. RISC-V ``SiFive7VA1``) are
|
|
394
|
+
returned unchanged — a slightly longer label is still readable.
|
|
395
|
+
"""
|
|
396
|
+
prefix, sep, suffix = name.partition("Unit")
|
|
397
|
+
core = suffix if sep and suffix and prefix and len(prefix) <= 5 and prefix[0].isupper() else name
|
|
398
|
+
if "." in core:
|
|
399
|
+
head, _, tail = core.partition(".")
|
|
400
|
+
# Map ``D.\x00`` / ``D.\x01`` → ``D0`` / ``D1`` so the column stays
|
|
401
|
+
# printable and distinguishes the replicated sub-units.
|
|
402
|
+
index_bytes = [b for b in tail.encode("utf-8", errors="ignore") if b < 32]
|
|
403
|
+
if index_bytes:
|
|
404
|
+
return f"{head}{index_bytes[0]}"
|
|
405
|
+
printable_tail = "".join(ch for ch in tail if ch.isprintable() and ch != "\x00")
|
|
406
|
+
return f"{head}{printable_tail}" if printable_tail else head
|
|
407
|
+
return core
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _pressure_by_index(
|
|
411
|
+
payload: dict[str, Any],
|
|
412
|
+
payload_region: dict[str, Any],
|
|
413
|
+
) -> dict[int, list[tuple[str, float]]]:
|
|
414
|
+
"""Build ``{InstructionIndex: [(port, usage), ...]}`` from llvm-mca.
|
|
415
|
+
|
|
416
|
+
``TargetInfo.Resources`` lives at the payload top level, while
|
|
417
|
+
``ResourcePressureView.ResourcePressureInfo`` is per-region. Joins
|
|
418
|
+
the two tables and drops entries whose ``ResourceUsage`` rounds to
|
|
419
|
+
zero at two decimal places.
|
|
420
|
+
"""
|
|
421
|
+
target_info = payload.get("TargetInfo") or payload_region.get("TargetInfo") or {}
|
|
422
|
+
resources = target_info.get("Resources") or []
|
|
423
|
+
view = payload_region.get("ResourcePressureView") or {}
|
|
424
|
+
pressure_info = view.get("ResourcePressureInfo") or []
|
|
425
|
+
result: dict[int, list[tuple[str, float]]] = {}
|
|
426
|
+
for entry in pressure_info:
|
|
427
|
+
if not isinstance(entry, dict):
|
|
428
|
+
continue
|
|
429
|
+
idx = entry.get("InstructionIndex")
|
|
430
|
+
res_idx = entry.get("ResourceIndex")
|
|
431
|
+
usage = entry.get("ResourceUsage")
|
|
432
|
+
if not isinstance(idx, int) or not isinstance(res_idx, int):
|
|
433
|
+
continue
|
|
434
|
+
if not isinstance(usage, (int, float)) or round(float(usage), 2) <= 0.0:
|
|
435
|
+
continue
|
|
436
|
+
if res_idx < 0 or res_idx >= len(resources):
|
|
437
|
+
continue
|
|
438
|
+
raw_name = str(resources[res_idx] or "")
|
|
439
|
+
if not raw_name:
|
|
440
|
+
continue
|
|
441
|
+
port = _format_port_name(raw_name)
|
|
442
|
+
result.setdefault(idx, []).append((port, float(usage)))
|
|
443
|
+
return result
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def _format_ports(entries: list[tuple[str, float]]) -> str:
|
|
447
|
+
"""``[('V0', 0.5), ('V1', 0.5)]`` → ``'0.50*V0 0.50*V1'``.
|
|
448
|
+
|
|
449
|
+
Uses uops.info's ``count*port`` convention so the ARM / RISC-V modeled
|
|
450
|
+
``ports`` column lines up with the format readers already expect from
|
|
451
|
+
the x86 measured side.
|
|
452
|
+
"""
|
|
453
|
+
return " ".join(f"{usage:.2f}*{name}" for name, usage in entries)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def _kind_label(info: dict[str, Any]) -> str:
|
|
457
|
+
"""Map ``mayLoad`` / ``mayStore`` flags to a compact label or ``''``."""
|
|
458
|
+
may_load = bool(info.get("mayLoad"))
|
|
459
|
+
may_store = bool(info.get("mayStore"))
|
|
460
|
+
if may_load and may_store:
|
|
461
|
+
return "load+store"
|
|
462
|
+
if may_load:
|
|
463
|
+
return "load"
|
|
464
|
+
if may_store:
|
|
465
|
+
return "store"
|
|
466
|
+
return ""
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def _build_perf_rows(
|
|
470
|
+
payload: dict[str, Any],
|
|
471
|
+
asm_lines: list[str],
|
|
472
|
+
*,
|
|
473
|
+
core: CoreSpec,
|
|
474
|
+
mca_version: str,
|
|
475
|
+
) -> list[PerfRow]:
|
|
476
|
+
"""Join an ``llvm-mca --json`` payload to the asm lines that produced it.
|
|
477
|
+
|
|
478
|
+
``payload`` is the parsed ``--instruction-tables=full --json``
|
|
479
|
+
output. The payload's ``InstructionInfoView.InstructionList`` is
|
|
480
|
+
parallel to the payload's own ``Instructions`` array, which reflects
|
|
481
|
+
any lines that llvm-mca skipped via
|
|
482
|
+
``--skip-unsupported-instructions``. When the two lists' lengths
|
|
483
|
+
match we prefer the payload's labels as the ground truth; otherwise
|
|
484
|
+
we fall back to the caller-supplied ``asm_lines``.
|
|
485
|
+
|
|
486
|
+
Emits one :class:`PerfRow` per unique mnemonic (first-seen wins —
|
|
487
|
+
same-mnemonic form variants contend for the same
|
|
488
|
+
``arch_details[core]`` slot anyway in
|
|
489
|
+
:func:`merge_perf_rows`).
|
|
490
|
+
"""
|
|
491
|
+
regions = payload.get("CodeRegions") or []
|
|
492
|
+
if not regions:
|
|
493
|
+
return []
|
|
494
|
+
region = regions[0]
|
|
495
|
+
info_list = (region.get("InstructionInfoView") or {}).get("InstructionList") or []
|
|
496
|
+
payload_insts = region.get("Instructions") or []
|
|
497
|
+
labels: list[str]
|
|
498
|
+
if len(payload_insts) == len(info_list) and all(
|
|
499
|
+
isinstance(item, str) for item in payload_insts
|
|
500
|
+
):
|
|
501
|
+
labels = list(payload_insts)
|
|
502
|
+
else:
|
|
503
|
+
labels = list(asm_lines)
|
|
504
|
+
arch_label = "arm" if core.architecture == "aarch64" else core.architecture
|
|
505
|
+
pressure_by_index = _pressure_by_index(payload, region)
|
|
506
|
+
|
|
507
|
+
rows: list[PerfRow] = []
|
|
508
|
+
seen: set[str] = set()
|
|
509
|
+
for idx, (asm_line, info) in enumerate(zip(labels, info_list)):
|
|
510
|
+
if not isinstance(info, dict):
|
|
511
|
+
continue
|
|
512
|
+
mnemonic = _asm_mnemonic(asm_line)
|
|
513
|
+
if not mnemonic or mnemonic in seen:
|
|
514
|
+
continue
|
|
515
|
+
seen.add(mnemonic)
|
|
516
|
+
latency = info.get("Latency")
|
|
517
|
+
rthroughput = info.get("RThroughput")
|
|
518
|
+
cpi = ""
|
|
519
|
+
if isinstance(rthroughput, (int, float)):
|
|
520
|
+
cpi = f"{float(rthroughput):.3f}".rstrip("0").rstrip(".")
|
|
521
|
+
extra: dict[str, str] = {}
|
|
522
|
+
num_uops = info.get("NumMicroOpcodes")
|
|
523
|
+
if isinstance(num_uops, (int, float)) and num_uops > 0:
|
|
524
|
+
extra["uops"] = str(int(num_uops))
|
|
525
|
+
ports_str = _format_ports(pressure_by_index.get(idx, []))
|
|
526
|
+
if ports_str:
|
|
527
|
+
extra["ports"] = ports_str
|
|
528
|
+
kind = _kind_label(info)
|
|
529
|
+
if kind:
|
|
530
|
+
extra["kind"] = kind
|
|
531
|
+
rows.append(
|
|
532
|
+
PerfRow(
|
|
533
|
+
mnemonic=mnemonic,
|
|
534
|
+
core=core.canonical_id,
|
|
535
|
+
source="llvm-mca",
|
|
536
|
+
source_kind="modeled",
|
|
537
|
+
source_version=mca_version,
|
|
538
|
+
architecture=arch_label,
|
|
539
|
+
latency=str(latency) if latency is not None else "",
|
|
540
|
+
cpi=cpi,
|
|
541
|
+
applies_to="mnemonic",
|
|
542
|
+
citation_url=LLVM_MCA_CITATION,
|
|
543
|
+
extra_measurement=extra,
|
|
544
|
+
)
|
|
545
|
+
)
|
|
546
|
+
return rows
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
def _disassembly_to_asm_lines(disasm_text: str) -> list[str]:
|
|
550
|
+
"""Turn ``_filter_disassembly`` output back into a list of asm lines."""
|
|
551
|
+
return [line for line in disasm_text.splitlines() if line.strip()]
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def collect_core_schedule(
|
|
555
|
+
core: CoreSpec,
|
|
556
|
+
*,
|
|
557
|
+
mca_version: str,
|
|
558
|
+
cache_root: Path | None = None,
|
|
559
|
+
executable_exegesis: str = "llvm-exegesis",
|
|
560
|
+
executable_mc: str = "llvm-mc",
|
|
561
|
+
executable_mca: str = "llvm-mca",
|
|
562
|
+
) -> list[PerfRow]:
|
|
563
|
+
"""Run the llvm-exegesis → llvm-mc → llvm-mca pipeline for one core.
|
|
564
|
+
|
|
565
|
+
Returns one :class:`PerfRow` per unique assembly mnemonic that
|
|
566
|
+
llvm-mca's scheduling model produces latency / throughput data for.
|
|
567
|
+
|
|
568
|
+
Cache layout: ``<cache_root>/<triple>/<cpu>/{exegesis.yaml,
|
|
569
|
+
disassembly.s, mca.json}``. Callers pin ``mca_version`` into
|
|
570
|
+
``cache_root`` themselves if they want LLVM-version isolation.
|
|
571
|
+
|
|
572
|
+
Raises :class:`LLVMMcaUnavailable` when a required binary is missing
|
|
573
|
+
and :class:`LLVMSchedulingError` (a ``LLVMMcaError`` subclass) on
|
|
574
|
+
any subprocess failure, empty result, or malformed output — no
|
|
575
|
+
silent fallback.
|
|
576
|
+
"""
|
|
577
|
+
if core.architecture not in {"aarch64", "riscv"}:
|
|
578
|
+
return []
|
|
579
|
+
|
|
580
|
+
root = cache_root if cache_root is not None else DEFAULT_CACHE_ROOT
|
|
581
|
+
cache_dir = root / core.llvm_triple / core.llvm_cpu
|
|
582
|
+
exegesis_path = cache_dir / "exegesis.yaml"
|
|
583
|
+
disasm_path = cache_dir / "disassembly.s"
|
|
584
|
+
mca_path = cache_dir / "mca.json"
|
|
585
|
+
cache_dir.mkdir(parents=True, exist_ok=True)
|
|
586
|
+
|
|
587
|
+
if not exegesis_path.exists():
|
|
588
|
+
_run_exegesis(core, exegesis_path, executable=executable_exegesis)
|
|
589
|
+
|
|
590
|
+
entries = _parse_exegesis_yaml(exegesis_path.read_text())
|
|
591
|
+
if not entries:
|
|
592
|
+
raise LLVMSchedulingError(
|
|
593
|
+
f"{executable_exegesis} produced no entries for {core.canonical_id}"
|
|
594
|
+
)
|
|
595
|
+
|
|
596
|
+
if not disasm_path.exists():
|
|
597
|
+
hex_lines = build_byte_lines(entries, core.architecture)
|
|
598
|
+
if not hex_lines:
|
|
599
|
+
raise LLVMSchedulingError(
|
|
600
|
+
f"could not extract any opcode bytes from {exegesis_path}"
|
|
601
|
+
)
|
|
602
|
+
raw = _run_disassemble(hex_lines, core, executable=executable_mc)
|
|
603
|
+
disasm_path.write_text(_filter_disassembly(raw))
|
|
604
|
+
|
|
605
|
+
disasm_text = disasm_path.read_text()
|
|
606
|
+
asm_lines = _disassembly_to_asm_lines(disasm_text)
|
|
607
|
+
if not asm_lines:
|
|
608
|
+
raise LLVMSchedulingError(
|
|
609
|
+
f"{executable_mc} produced empty disassembly for {core.canonical_id}"
|
|
610
|
+
)
|
|
611
|
+
|
|
612
|
+
if not mca_path.exists():
|
|
613
|
+
payload = _run_mca(disasm_text, core, executable=executable_mca)
|
|
614
|
+
mca_path.write_text(json.dumps(payload))
|
|
615
|
+
else:
|
|
616
|
+
payload = json.loads(mca_path.read_text())
|
|
617
|
+
|
|
618
|
+
rows = _build_perf_rows(
|
|
619
|
+
payload, asm_lines, core=core, mca_version=mca_version
|
|
620
|
+
)
|
|
621
|
+
if not rows:
|
|
622
|
+
raise LLVMSchedulingError(
|
|
623
|
+
f"{executable_mca} produced no schedule rows for {core.canonical_id}"
|
|
624
|
+
)
|
|
625
|
+
return rows
|