simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Vendor-neutral PDF enrichment types."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Sequence
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True, slots=True)
|
|
11
|
+
class PdfDescriptionPayload:
|
|
12
|
+
sections: dict[str, str]
|
|
13
|
+
source_url: str
|
|
14
|
+
page_start: int | None = None
|
|
15
|
+
page_end: int | None = None
|
|
16
|
+
|
|
17
|
+
def to_dict(self) -> dict[str, object]:
|
|
18
|
+
return {
|
|
19
|
+
"sections": self.sections,
|
|
20
|
+
"source_url": self.source_url,
|
|
21
|
+
"page_start": self.page_start,
|
|
22
|
+
"page_end": self.page_end,
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class PdfEnrichmentResult:
|
|
28
|
+
descriptions: dict[str, PdfDescriptionPayload]
|
|
29
|
+
fallback_page_count: int = 0
|
|
30
|
+
stats: dict[str, int] = field(default_factory=dict)
|
|
31
|
+
|
|
32
|
+
def to_dict(self) -> dict[str, object]:
|
|
33
|
+
return {
|
|
34
|
+
"descriptions": {key: value.to_dict() for key, value in self.descriptions.items()},
|
|
35
|
+
"fallback_page_count": self.fallback_page_count,
|
|
36
|
+
"stats": dict(self.stats),
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def from_dict(cls, payload: dict[str, object]) -> "PdfEnrichmentResult":
|
|
41
|
+
descriptions: dict[str, PdfDescriptionPayload] = {}
|
|
42
|
+
raw_descriptions = payload.get("descriptions") or {}
|
|
43
|
+
if isinstance(raw_descriptions, dict):
|
|
44
|
+
for key, value in raw_descriptions.items():
|
|
45
|
+
if not isinstance(value, dict):
|
|
46
|
+
continue
|
|
47
|
+
descriptions[str(key)] = PdfDescriptionPayload(
|
|
48
|
+
sections=dict(value.get("sections") or {}),
|
|
49
|
+
source_url=str(value.get("source_url") or ""),
|
|
50
|
+
page_start=value.get("page_start") if isinstance(value.get("page_start"), int) else None,
|
|
51
|
+
page_end=value.get("page_end") if isinstance(value.get("page_end"), int) else None,
|
|
52
|
+
)
|
|
53
|
+
raw_stats = payload.get("stats") or {}
|
|
54
|
+
stats = {
|
|
55
|
+
str(key): int(value)
|
|
56
|
+
for key, value in raw_stats.items()
|
|
57
|
+
if isinstance(value, int)
|
|
58
|
+
} if isinstance(raw_stats, dict) else {}
|
|
59
|
+
fallback_page_count = payload.get("fallback_page_count")
|
|
60
|
+
return cls(
|
|
61
|
+
descriptions=descriptions,
|
|
62
|
+
fallback_page_count=fallback_page_count if isinstance(fallback_page_count, int) else 0,
|
|
63
|
+
stats=stats,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True, slots=True)
|
|
68
|
+
class PdfSourceSpec:
|
|
69
|
+
source_id: str
|
|
70
|
+
display_name: str
|
|
71
|
+
source_url: str
|
|
72
|
+
local_candidates: Sequence[Path]
|
|
73
|
+
cache_path: Path
|
|
74
|
+
cache_version: int
|
|
75
|
+
signature_paths: Sequence[Path]
|
|
76
|
+
parser: Callable[..., PdfEnrichmentResult]
|
|
77
|
+
find_source: Callable[..., Path | None]
|
simdref/pdfrefs.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Shared normalized PDF reference helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
PdfRef = dict[str, str]
|
|
8
|
+
|
|
9
|
+
_LEGACY_INTEL_SOURCE_ID = "intel-sdm"
|
|
10
|
+
_LEGACY_INTEL_LABEL = "Intel SDM"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def normalize_pdf_refs(
|
|
14
|
+
pdf_refs: list[dict[str, Any]] | None,
|
|
15
|
+
metadata: dict[str, str] | None = None,
|
|
16
|
+
) -> list[PdfRef]:
|
|
17
|
+
"""Return normalized PDF references, preserving legacy Intel metadata."""
|
|
18
|
+
normalized: list[PdfRef] = []
|
|
19
|
+
seen: set[tuple[str, str, str, str, str]] = set()
|
|
20
|
+
|
|
21
|
+
for ref in pdf_refs or []:
|
|
22
|
+
candidate = {
|
|
23
|
+
"source_id": str(ref.get("source_id") or "").strip(),
|
|
24
|
+
"label": str(ref.get("label") or "").strip(),
|
|
25
|
+
"url": str(ref.get("url") or "").strip(),
|
|
26
|
+
"page_start": str(ref.get("page_start") or "").strip(),
|
|
27
|
+
"page_end": str(ref.get("page_end") or "").strip(),
|
|
28
|
+
}
|
|
29
|
+
if not candidate["source_id"] or not candidate["label"] or not candidate["url"]:
|
|
30
|
+
continue
|
|
31
|
+
key = (
|
|
32
|
+
candidate["source_id"],
|
|
33
|
+
candidate["label"],
|
|
34
|
+
candidate["url"],
|
|
35
|
+
candidate["page_start"],
|
|
36
|
+
candidate["page_end"],
|
|
37
|
+
)
|
|
38
|
+
if key in seen:
|
|
39
|
+
continue
|
|
40
|
+
seen.add(key)
|
|
41
|
+
normalized.append(candidate)
|
|
42
|
+
|
|
43
|
+
legacy = legacy_intel_pdf_ref(metadata)
|
|
44
|
+
if legacy is not None:
|
|
45
|
+
key = (
|
|
46
|
+
legacy["source_id"],
|
|
47
|
+
legacy["label"],
|
|
48
|
+
legacy["url"],
|
|
49
|
+
legacy["page_start"],
|
|
50
|
+
legacy["page_end"],
|
|
51
|
+
)
|
|
52
|
+
if key not in seen:
|
|
53
|
+
normalized.append(legacy)
|
|
54
|
+
return normalized
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def legacy_intel_pdf_ref(metadata: dict[str, str] | None) -> PdfRef | None:
|
|
58
|
+
"""Build a normalized ref from legacy Intel-specific metadata keys."""
|
|
59
|
+
if not metadata:
|
|
60
|
+
return None
|
|
61
|
+
url = (metadata.get("intel-sdm-url") or "").strip()
|
|
62
|
+
if not url:
|
|
63
|
+
return None
|
|
64
|
+
return {
|
|
65
|
+
"source_id": _LEGACY_INTEL_SOURCE_ID,
|
|
66
|
+
"label": _LEGACY_INTEL_LABEL,
|
|
67
|
+
"url": url,
|
|
68
|
+
"page_start": (metadata.get("intel-sdm-page-start") or "").strip(),
|
|
69
|
+
"page_end": (metadata.get("intel-sdm-page-end") or "").strip(),
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def apply_legacy_pdf_metadata(metadata: dict[str, str], pdf_refs: list[PdfRef]) -> dict[str, str]:
|
|
74
|
+
"""Populate legacy Intel keys for backward-compatible payloads."""
|
|
75
|
+
legacy = next((ref for ref in pdf_refs if ref.get("source_id") == _LEGACY_INTEL_SOURCE_ID), None)
|
|
76
|
+
if legacy is None:
|
|
77
|
+
return metadata
|
|
78
|
+
if legacy.get("url"):
|
|
79
|
+
metadata.setdefault("intel-sdm-url", legacy["url"])
|
|
80
|
+
if legacy.get("page_start"):
|
|
81
|
+
metadata.setdefault("intel-sdm-page-start", legacy["page_start"])
|
|
82
|
+
if legacy.get("page_end"):
|
|
83
|
+
metadata.setdefault("intel-sdm-page-end", legacy["page_end"])
|
|
84
|
+
return metadata
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def pdf_ref_label(ref: PdfRef) -> str:
|
|
88
|
+
label = ref.get("label") or ref.get("source_id") or "PDF"
|
|
89
|
+
page_start = ref.get("page_start") or ""
|
|
90
|
+
page_end = ref.get("page_end") or ""
|
|
91
|
+
if page_start and page_end and page_end != page_start:
|
|
92
|
+
return f"{label} (pages {page_start}-{page_end})"
|
|
93
|
+
if page_start:
|
|
94
|
+
return f"{label} (page {page_start})"
|
|
95
|
+
return label
|
simdref/perf.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""Shared performance-metric extraction helpers.
|
|
2
|
+
|
|
3
|
+
Functions in this module extract latency, throughput (cycles-per-instruction),
|
|
4
|
+
and other microarchitecture-level performance data from the nested
|
|
5
|
+
``arch_details`` dictionaries stored on :class:`~simdref.models.InstructionRecord`.
|
|
6
|
+
|
|
7
|
+
They are consumed by the CLI, LSP, web-export, and man-page modules so
|
|
8
|
+
that the extraction logic lives in exactly one place.
|
|
9
|
+
|
|
10
|
+
Each ``arch_details[core]`` entry may carry provenance keys added by the
|
|
11
|
+
ingesters (``source``, ``source_kind``, ``source_version``, ``applies_to``,
|
|
12
|
+
``citation_url``). ``source_kind`` is either ``"measured"`` or
|
|
13
|
+
``"modeled"``; when absent it defaults to ``"measured"`` for compatibility
|
|
14
|
+
with the original uops.info rows which are measured.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import Any, NamedTuple
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class PerfValue(NamedTuple):
|
|
23
|
+
"""A labeled performance value.
|
|
24
|
+
|
|
25
|
+
``value`` is the cycle-count string (``"-"`` when missing).
|
|
26
|
+
``source_kind`` is ``"measured"`` or ``"modeled"``.
|
|
27
|
+
``core`` is the microarchitecture id the value came from (empty when none).
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
value: str
|
|
31
|
+
source_kind: str
|
|
32
|
+
core: str
|
|
33
|
+
|
|
34
|
+
def __str__(self) -> str:
|
|
35
|
+
return self.value
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
MISSING_PERF = PerfValue("-", "", "")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _is_numeric(value: str) -> bool:
|
|
42
|
+
"""Check if a string is a non-negative number (integer or decimal)."""
|
|
43
|
+
if not value:
|
|
44
|
+
return False
|
|
45
|
+
parts = value.split(".", 1)
|
|
46
|
+
return all(p.isdigit() for p in parts) and parts[0] != ""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _source_kind(details: dict[str, Any]) -> str:
|
|
50
|
+
"""Return the provenance kind for an ``arch_details`` entry.
|
|
51
|
+
|
|
52
|
+
Defaults to ``"measured"`` when unset so legacy uops.info rows carry
|
|
53
|
+
the correct label without migration.
|
|
54
|
+
"""
|
|
55
|
+
kind = details.get("source_kind")
|
|
56
|
+
if kind in ("measured", "modeled"):
|
|
57
|
+
return kind
|
|
58
|
+
return "measured"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def latency_cycle_values(latencies: list[dict[str, Any]]) -> list[str]:
|
|
62
|
+
"""Collect unique cycle-count strings from a list of latency dicts.
|
|
63
|
+
|
|
64
|
+
Each *latency* dict may contain keys like ``cycles``, ``cycles_mem``,
|
|
65
|
+
``cycles_addr``, or ``cycles_addr_index``. Values are collected in
|
|
66
|
+
order, skipping duplicates.
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
De-duplicated list of cycle-count strings (may be empty).
|
|
70
|
+
"""
|
|
71
|
+
values: list[str] = []
|
|
72
|
+
for latency in latencies:
|
|
73
|
+
for key in ("cycles", "cycles_mem", "cycles_addr", "cycles_addr_index"):
|
|
74
|
+
value = latency.get(key)
|
|
75
|
+
if value and value not in values:
|
|
76
|
+
values.append(value)
|
|
77
|
+
return values
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def best_numeric(values: list[str]) -> str:
|
|
81
|
+
"""Return the smallest numeric string from *values*, or ``"-"`` if empty.
|
|
82
|
+
|
|
83
|
+
Non-numeric entries (e.g. ``"variable"``) are ignored when a numeric
|
|
84
|
+
alternative exists. If every entry is non-numeric the first value is
|
|
85
|
+
returned as-is.
|
|
86
|
+
"""
|
|
87
|
+
if not values:
|
|
88
|
+
return "-"
|
|
89
|
+
numeric = [v for v in values if _is_numeric(str(v))]
|
|
90
|
+
if numeric:
|
|
91
|
+
return min(numeric, key=lambda v: float(v))
|
|
92
|
+
return values[0]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _cpi_values(details: dict[str, Any]) -> list[str]:
|
|
96
|
+
measurement = details.get("measurement") or {}
|
|
97
|
+
out: list[str] = []
|
|
98
|
+
for key in ("TP_unrolled", "TP_loop", "TP_ports", "TP"):
|
|
99
|
+
value = measurement.get(key)
|
|
100
|
+
if value and value not in out:
|
|
101
|
+
out.append(value)
|
|
102
|
+
return out
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _best_labeled(
|
|
106
|
+
arch_details: dict[str, dict[str, Any]],
|
|
107
|
+
value_fn,
|
|
108
|
+
kind_filter: str | None = None,
|
|
109
|
+
) -> PerfValue:
|
|
110
|
+
"""Return the lowest numeric value across arch entries, with provenance.
|
|
111
|
+
|
|
112
|
+
*value_fn* is called on each ``details`` dict and must return a list of
|
|
113
|
+
candidate cycle-count strings. When *kind_filter* is ``"measured"`` or
|
|
114
|
+
``"modeled"`` only entries with that ``source_kind`` are considered.
|
|
115
|
+
"""
|
|
116
|
+
best: tuple[float, str, str, str] | None = None
|
|
117
|
+
first_non_numeric: PerfValue | None = None
|
|
118
|
+
for core, details in arch_details.items():
|
|
119
|
+
kind = _source_kind(details)
|
|
120
|
+
if kind_filter is not None and kind != kind_filter:
|
|
121
|
+
continue
|
|
122
|
+
for value in value_fn(details):
|
|
123
|
+
value_str = str(value)
|
|
124
|
+
if _is_numeric(value_str):
|
|
125
|
+
numeric = float(value_str)
|
|
126
|
+
if best is None or numeric < best[0]:
|
|
127
|
+
best = (numeric, value_str, kind, core)
|
|
128
|
+
elif first_non_numeric is None:
|
|
129
|
+
first_non_numeric = PerfValue(value_str, kind, core)
|
|
130
|
+
if best is not None:
|
|
131
|
+
return PerfValue(best[1], best[2], best[3])
|
|
132
|
+
if first_non_numeric is not None:
|
|
133
|
+
return first_non_numeric
|
|
134
|
+
return MISSING_PERF
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def best_latency_labeled(
|
|
138
|
+
arch_details: dict[str, dict[str, Any]],
|
|
139
|
+
kind_filter: str | None = None,
|
|
140
|
+
) -> PerfValue:
|
|
141
|
+
"""Lowest latency across arch entries, as a :class:`PerfValue`.
|
|
142
|
+
|
|
143
|
+
Prefers measured rows: if any measured row exposes a numeric latency
|
|
144
|
+
it is picked over all modeled rows. Pass ``kind_filter="modeled"`` to
|
|
145
|
+
select only modeled data.
|
|
146
|
+
"""
|
|
147
|
+
def lat_values(details: dict[str, Any]) -> list[str]:
|
|
148
|
+
return latency_cycle_values(details.get("latencies") or [])
|
|
149
|
+
|
|
150
|
+
if kind_filter is None:
|
|
151
|
+
measured = _best_labeled(arch_details, lat_values, kind_filter="measured")
|
|
152
|
+
if measured.value != "-":
|
|
153
|
+
return measured
|
|
154
|
+
return _best_labeled(arch_details, lat_values, kind_filter="modeled")
|
|
155
|
+
return _best_labeled(arch_details, lat_values, kind_filter=kind_filter)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def best_cpi_labeled(
|
|
159
|
+
arch_details: dict[str, dict[str, Any]],
|
|
160
|
+
kind_filter: str | None = None,
|
|
161
|
+
) -> PerfValue:
|
|
162
|
+
"""Lowest cycles-per-instruction across arch entries, as a :class:`PerfValue`."""
|
|
163
|
+
if kind_filter is None:
|
|
164
|
+
measured = _best_labeled(arch_details, _cpi_values, kind_filter="measured")
|
|
165
|
+
if measured.value != "-":
|
|
166
|
+
return measured
|
|
167
|
+
return _best_labeled(arch_details, _cpi_values, kind_filter="modeled")
|
|
168
|
+
return _best_labeled(arch_details, _cpi_values, kind_filter=kind_filter)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def best_latency(arch_details: dict[str, dict[str, Any]]) -> str:
|
|
172
|
+
"""Return the best (lowest) latency across all microarchitectures.
|
|
173
|
+
|
|
174
|
+
Prefers measured rows over modeled rows (see :func:`best_latency_labeled`).
|
|
175
|
+
"""
|
|
176
|
+
return best_latency_labeled(arch_details).value
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def best_cpi(arch_details: dict[str, dict[str, Any]]) -> str:
|
|
180
|
+
"""Return the best (lowest) cycles-per-instruction across all microarchitectures.
|
|
181
|
+
|
|
182
|
+
Prefers measured rows over modeled rows (see :func:`best_cpi_labeled`).
|
|
183
|
+
"""
|
|
184
|
+
return best_cpi_labeled(arch_details).value
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def best_latency_measured(arch_details: dict[str, dict[str, Any]]) -> PerfValue:
|
|
188
|
+
"""Lowest latency drawn exclusively from ``source_kind="measured"`` rows."""
|
|
189
|
+
return best_latency_labeled(arch_details, kind_filter="measured")
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def best_latency_modeled(arch_details: dict[str, dict[str, Any]]) -> PerfValue:
|
|
193
|
+
"""Lowest latency drawn exclusively from ``source_kind="modeled"`` rows."""
|
|
194
|
+
return best_latency_labeled(arch_details, kind_filter="modeled")
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def best_cpi_measured(arch_details: dict[str, dict[str, Any]]) -> PerfValue:
|
|
198
|
+
"""Lowest CPI drawn exclusively from ``source_kind="measured"`` rows."""
|
|
199
|
+
return best_cpi_labeled(arch_details, kind_filter="measured")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def best_cpi_modeled(arch_details: dict[str, dict[str, Any]]) -> PerfValue:
|
|
203
|
+
"""Lowest CPI drawn exclusively from ``source_kind="modeled"`` rows."""
|
|
204
|
+
return best_cpi_labeled(arch_details, kind_filter="modeled")
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def variant_perf_summary(arch_details: dict[str, dict[str, Any]]) -> tuple[str, str]:
|
|
208
|
+
"""Return ``(best_latency, best_cpi)`` strings for a single variant."""
|
|
209
|
+
return best_latency(arch_details), best_cpi(arch_details)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def variant_perf_summary_labeled(
|
|
213
|
+
arch_details: dict[str, dict[str, Any]],
|
|
214
|
+
) -> tuple[PerfValue, PerfValue]:
|
|
215
|
+
"""Return labeled ``(latency, cpi)`` for a single variant.
|
|
216
|
+
|
|
217
|
+
Each element carries its ``source_kind`` and originating ``core`` so
|
|
218
|
+
renderers can show "(measured, SKL)" / "(modeled, neoverse-n1)" suffixes.
|
|
219
|
+
"""
|
|
220
|
+
return best_latency_labeled(arch_details), best_cpi_labeled(arch_details)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Ingesters for measured and modeled microarchitectural perf data.
|
|
2
|
+
|
|
3
|
+
Public surface:
|
|
4
|
+
|
|
5
|
+
- :func:`ingest_llvm_mca` — drives the llvm-exegesis → llvm-mc →
|
|
6
|
+
llvm-mca pipeline per canonical core and returns modeled
|
|
7
|
+
:class:`PerfRow` s.
|
|
8
|
+
- :func:`collect_core_schedule` — lower-level single-core entry point.
|
|
9
|
+
- :func:`merge_perf_rows` — attaches produced rows to an existing catalog.
|
|
10
|
+
- :data:`CANONICAL_CORES` — name-map from upstream core ids → stable ids.
|
|
11
|
+
|
|
12
|
+
Per-instruction measured data for RISC-V RVV is not currently available
|
|
13
|
+
from any public upstream: ``rvv-bench-results`` publishes kernel-level
|
|
14
|
+
benchmarks (memcpy, chacha20, etc.), not instruction tables. RISC-V
|
|
15
|
+
per-core rows therefore come from llvm-mca scheduling models only.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from simdref.perf_sources.cores import (
|
|
19
|
+
CANONICAL_CORES,
|
|
20
|
+
canonical_core_id,
|
|
21
|
+
core_architecture,
|
|
22
|
+
)
|
|
23
|
+
from simdref.perf_sources.llvm_mca import (
|
|
24
|
+
LLVM_MCA_MIN_VERSION,
|
|
25
|
+
LLVMMcaError,
|
|
26
|
+
LLVMMcaUnavailable,
|
|
27
|
+
detect_llvm_mca_version,
|
|
28
|
+
ingest_llvm_mca,
|
|
29
|
+
parse_llvm_mca_json,
|
|
30
|
+
)
|
|
31
|
+
from simdref.perf_sources.llvm_scheduling import (
|
|
32
|
+
LLVMSchedulingError,
|
|
33
|
+
collect_core_schedule,
|
|
34
|
+
)
|
|
35
|
+
from simdref.perf_sources.merge import PerfRow, merge_perf_rows
|
|
36
|
+
|
|
37
|
+
__all__ = [
|
|
38
|
+
"CANONICAL_CORES",
|
|
39
|
+
"canonical_core_id",
|
|
40
|
+
"core_architecture",
|
|
41
|
+
"PerfRow",
|
|
42
|
+
"merge_perf_rows",
|
|
43
|
+
"LLVM_MCA_MIN_VERSION",
|
|
44
|
+
"LLVMMcaError",
|
|
45
|
+
"LLVMMcaUnavailable",
|
|
46
|
+
"LLVMSchedulingError",
|
|
47
|
+
"detect_llvm_mca_version",
|
|
48
|
+
"ingest_llvm_mca",
|
|
49
|
+
"parse_llvm_mca_json",
|
|
50
|
+
"collect_core_schedule",
|
|
51
|
+
]
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Canonical core-id table.
|
|
2
|
+
|
|
3
|
+
Upstream sources disagree about core naming: LLVM uses ``neoverse-n1``,
|
|
4
|
+
rvv-bench labels its rows ``c908``/``c910``. This module
|
|
5
|
+
maps every upstream alias to a single canonical id used inside
|
|
6
|
+
``InstructionRecord.arch_details`` so lookups and filters are deterministic.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class CoreSpec:
|
|
16
|
+
"""A canonical microarchitecture entry."""
|
|
17
|
+
|
|
18
|
+
canonical_id: str
|
|
19
|
+
architecture: str # "x86", "aarch64", "riscv"
|
|
20
|
+
llvm_triple: str
|
|
21
|
+
llvm_cpu: str
|
|
22
|
+
aliases: frozenset[str]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# AArch64 cores covered by LLVM scheduling models. Aliases include the
|
|
26
|
+
# exact strings emitted by the upstream sources.
|
|
27
|
+
AARCH64_CORES: tuple[CoreSpec, ...] = (
|
|
28
|
+
CoreSpec("cortex-a72", "aarch64", "aarch64-unknown-linux-gnu", "cortex-a72",
|
|
29
|
+
frozenset({"cortex-a72", "A72"})),
|
|
30
|
+
CoreSpec("cortex-a76", "aarch64", "aarch64-unknown-linux-gnu", "cortex-a76",
|
|
31
|
+
frozenset({"cortex-a76", "A76"})),
|
|
32
|
+
CoreSpec("cortex-a78", "aarch64", "aarch64-unknown-linux-gnu", "cortex-a78",
|
|
33
|
+
frozenset({"cortex-a78", "A78"})),
|
|
34
|
+
CoreSpec("cortex-x1", "aarch64", "aarch64-unknown-linux-gnu", "cortex-x1",
|
|
35
|
+
frozenset({"cortex-x1", "X1"})),
|
|
36
|
+
CoreSpec("cortex-x2", "aarch64", "aarch64-unknown-linux-gnu", "cortex-x2",
|
|
37
|
+
frozenset({"cortex-x2", "X2"})),
|
|
38
|
+
CoreSpec("neoverse-n1", "aarch64", "aarch64-unknown-linux-gnu", "neoverse-n1",
|
|
39
|
+
frozenset({"neoverse-n1", "N1", "Neoverse-N1"})),
|
|
40
|
+
CoreSpec("neoverse-n2", "aarch64", "aarch64-unknown-linux-gnu", "neoverse-n2",
|
|
41
|
+
frozenset({"neoverse-n2", "N2", "Neoverse-N2"})),
|
|
42
|
+
CoreSpec("neoverse-v1", "aarch64", "aarch64-unknown-linux-gnu", "neoverse-v1",
|
|
43
|
+
frozenset({"neoverse-v1", "V1", "Neoverse-V1"})),
|
|
44
|
+
CoreSpec("neoverse-v2", "aarch64", "aarch64-unknown-linux-gnu", "neoverse-v2",
|
|
45
|
+
frozenset({"neoverse-v2", "V2", "Neoverse-V2"})),
|
|
46
|
+
CoreSpec("a64fx", "aarch64", "aarch64-unknown-linux-gnu", "a64fx",
|
|
47
|
+
frozenset({"a64fx", "A64FX"})),
|
|
48
|
+
# Apple cores reuse the Linux AArch64 triple so llvm-exegesis can
|
|
49
|
+
# assemble on a non-Darwin host. The scheduling data we consume is
|
|
50
|
+
# keyed off --mcpu, not --mtriple, so the numbers are identical.
|
|
51
|
+
CoreSpec("apple-m1", "aarch64", "aarch64-unknown-linux-gnu", "apple-m1",
|
|
52
|
+
frozenset({"apple-m1", "M1"})),
|
|
53
|
+
CoreSpec("apple-m2", "aarch64", "aarch64-unknown-linux-gnu", "apple-m2",
|
|
54
|
+
frozenset({"apple-m2", "M2"})),
|
|
55
|
+
CoreSpec("thunderx2t99", "aarch64", "aarch64-unknown-linux-gnu", "thunderx2t99",
|
|
56
|
+
frozenset({"thunderx2t99", "ThunderX2"})),
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# RISC-V cores with measured (rvv-bench) and modeled (LLVM) coverage.
|
|
60
|
+
RISCV_CORES: tuple[CoreSpec, ...] = (
|
|
61
|
+
CoreSpec("sifive-u74", "riscv", "riscv64-unknown-linux-gnu", "sifive-u74",
|
|
62
|
+
frozenset({"sifive-u74", "U74"})),
|
|
63
|
+
CoreSpec("sifive-x280", "riscv", "riscv64-unknown-linux-gnu", "sifive-x280",
|
|
64
|
+
frozenset({"sifive-x280", "X280"})),
|
|
65
|
+
# LLVM 22 exposes the sifive-p400 and sifive-p600 *families* under the
|
|
66
|
+
# specific part names p450 and p670 (the default LLVM CPU names for
|
|
67
|
+
# those families). The canonical id is kept generic so catalog
|
|
68
|
+
# consumers can filter by family rather than part.
|
|
69
|
+
CoreSpec("sifive-p400", "riscv", "riscv64-unknown-linux-gnu", "sifive-p450",
|
|
70
|
+
frozenset({"sifive-p400", "sifive-p450", "P400", "P450"})),
|
|
71
|
+
CoreSpec("sifive-p600", "riscv", "riscv64-unknown-linux-gnu", "sifive-p670",
|
|
72
|
+
frozenset({"sifive-p600", "sifive-p670", "P600", "P670"})),
|
|
73
|
+
CoreSpec("c908", "riscv", "riscv64-unknown-linux-gnu", "xiangshan-nanhu",
|
|
74
|
+
frozenset({"c908", "C908"})),
|
|
75
|
+
CoreSpec("c910", "riscv", "riscv64-unknown-linux-gnu", "xiangshan-nanhu",
|
|
76
|
+
frozenset({"c910", "C910"})),
|
|
77
|
+
CoreSpec("x60", "riscv", "riscv64-unknown-linux-gnu", "sifive-x280",
|
|
78
|
+
frozenset({"x60", "X60", "Spacemit-X60"})),
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
CANONICAL_CORES: tuple[CoreSpec, ...] = AARCH64_CORES + RISCV_CORES
|
|
82
|
+
|
|
83
|
+
_ALIAS_INDEX: dict[str, CoreSpec] = {}
|
|
84
|
+
for _core in CANONICAL_CORES:
|
|
85
|
+
for _alias in _core.aliases:
|
|
86
|
+
_ALIAS_INDEX[_alias.casefold()] = _core
|
|
87
|
+
_ALIAS_INDEX[_core.canonical_id.casefold()] = _core
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def canonical_core_id(name: str) -> str | None:
|
|
91
|
+
"""Return the canonical id for *name* or ``None`` if unknown."""
|
|
92
|
+
if not name:
|
|
93
|
+
return None
|
|
94
|
+
core = _ALIAS_INDEX.get(name.casefold())
|
|
95
|
+
return core.canonical_id if core is not None else None
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def core_architecture(canonical_id: str) -> str | None:
|
|
99
|
+
"""Return the architecture family (x86/aarch64/riscv) for a canonical id."""
|
|
100
|
+
core = _ALIAS_INDEX.get(canonical_id.casefold())
|
|
101
|
+
return core.architecture if core is not None else None
|