simdref 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,121 @@
1
+ """Attach ingested perf rows onto existing :class:`InstructionRecord` data."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from typing import Any, Iterable
7
+
8
+ from simdref.models import InstructionRecord
9
+
10
+
11
+ @dataclass(frozen=True)
12
+ class PerfRow:
13
+ """A single perf observation produced by a perf_sources ingester.
14
+
15
+ Attributes map onto the provenance keys stored in
16
+ ``InstructionRecord.arch_details[core]`` so :func:`merge_perf_rows` can
17
+ drop them in verbatim.
18
+ """
19
+
20
+ mnemonic: str
21
+ core: str
22
+ source: str
23
+ source_kind: str # "measured" | "modeled"
24
+ source_version: str
25
+ architecture: str = ""
26
+ form: str = ""
27
+ latency: str = ""
28
+ cpi: str = ""
29
+ applies_to: str = "mnemonic"
30
+ citation_url: str = ""
31
+ extra_measurement: dict[str, str] = field(default_factory=dict)
32
+
33
+ def as_arch_details_entry(self) -> dict[str, Any]:
34
+ entry: dict[str, Any] = {
35
+ "source": self.source,
36
+ "source_kind": self.source_kind,
37
+ "source_version": self.source_version,
38
+ "applies_to": self.applies_to,
39
+ "citation_url": self.citation_url,
40
+ }
41
+ if self.latency:
42
+ entry["latencies"] = [{"cycles": self.latency}]
43
+ measurement: dict[str, str] = dict(self.extra_measurement)
44
+ if self.cpi:
45
+ # ``TP_loop`` is the column the TUI and CLI measurement tables
46
+ # already render under the "CPI" header, so modeled rows must
47
+ # land there to surface at all. ``TP`` is retained as a
48
+ # duplicate for any downstream consumer that still reads the
49
+ # legacy key.
50
+ measurement.setdefault("TP_loop", self.cpi)
51
+ measurement.setdefault("TP", self.cpi)
52
+ if measurement:
53
+ entry["measurement"] = measurement
54
+ return entry
55
+
56
+
57
+ def _record_key(record: InstructionRecord) -> tuple[str, str, str]:
58
+ return (
59
+ record.architecture.casefold(),
60
+ record.mnemonic.casefold(),
61
+ record.form.strip().casefold(),
62
+ )
63
+
64
+
65
+ def merge_perf_rows(
66
+ records: Iterable[InstructionRecord],
67
+ rows: Iterable[PerfRow],
68
+ *,
69
+ overwrite: bool = False,
70
+ ) -> int:
71
+ """Merge *rows* into matching records in-place.
72
+
73
+ Matching rule: a row attaches to every record with the same architecture
74
+ and mnemonic. If ``row.form`` is non-empty, only records with the
75
+ matching form are updated (case-insensitive). Otherwise the row applies
76
+ to every variant of the mnemonic (``applies_to="mnemonic"``).
77
+
78
+ ``overwrite=False`` (default) preserves existing ``arch_details[core]``
79
+ entries so measured data produced by earlier passes is never clobbered
80
+ by modeled data from a later pass.
81
+
82
+ Returns the number of ``arch_details[core]`` entries newly written.
83
+ """
84
+ records = list(records)
85
+ by_arch_mnemonic: dict[tuple[str, str], list[InstructionRecord]] = {}
86
+ for record in records:
87
+ by_arch_mnemonic.setdefault(
88
+ (record.architecture.casefold(), record.mnemonic.casefold()),
89
+ [],
90
+ ).append(record)
91
+
92
+ written = 0
93
+ for row in rows:
94
+ targets = by_arch_mnemonic.get(
95
+ (row.architecture.casefold() or _arch_guess(row.core), row.mnemonic.casefold()),
96
+ [],
97
+ )
98
+ if not targets:
99
+ continue
100
+ form_key = row.form.strip().casefold()
101
+ matched = [r for r in targets if not form_key or r.form.strip().casefold() == form_key]
102
+ if not matched and form_key:
103
+ matched = targets # graceful fallback: attach to every variant
104
+ entry = row.as_arch_details_entry()
105
+ for record in matched:
106
+ if not overwrite and row.core in record.arch_details:
107
+ continue
108
+ record.arch_details[row.core] = entry
109
+ written += 1
110
+ return written
111
+
112
+
113
+ def _arch_guess(core: str) -> str:
114
+ """Best-effort architecture lookup used when :class:`PerfRow` lacks one."""
115
+ from simdref.perf_sources.cores import core_architecture
116
+ arch = core_architecture(core)
117
+ if arch == "aarch64":
118
+ return "arm"
119
+ if arch == "riscv":
120
+ return "riscv"
121
+ return arch or ""
simdref/queries.py ADDED
@@ -0,0 +1,207 @@
1
+ """Shared record-linking and lookup helpers.
2
+
3
+ Functions here resolve relationships between intrinsics and instructions,
4
+ optionally using a SQLite connection for fast lookups or falling back to
5
+ in-memory catalog scans. They are used by the CLI, LSP, and man-page
6
+ modules.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import sqlite3
12
+ from typing import TYPE_CHECKING
13
+
14
+ from simdref.perf import variant_perf_summary
15
+ from simdref.storage import load_instruction_from_db
16
+
17
+ SOURCE_KINDS: tuple[str, ...] = ("measured", "modeled", "any")
18
+
19
+
20
+ def filter_arch_details_by_source_kind(
21
+ arch_details: dict[str, dict[str, object]],
22
+ source_kind: str,
23
+ ) -> dict[str, dict[str, object]]:
24
+ """Drop ``arch_details`` entries whose provenance does not match.
25
+
26
+ ``"any"`` (or empty) keeps everything; ``"measured"`` / ``"modeled"``
27
+ keep only entries whose ``source_kind`` field matches (legacy entries
28
+ without a ``source_kind`` key are treated as measured).
29
+ """
30
+ if source_kind in ("", "any"):
31
+ return arch_details
32
+ kept: dict[str, dict[str, object]] = {}
33
+ for core, details in arch_details.items():
34
+ kind = details.get("source_kind") or "measured"
35
+ if kind == source_kind:
36
+ kept[core] = details
37
+ return kept
38
+
39
+ if TYPE_CHECKING:
40
+ from simdref.models import Catalog, InstructionRecord, IntrinsicRecord
41
+
42
+
43
+ def linked_instruction_records(
44
+ catalog: Catalog | None,
45
+ intrinsic: IntrinsicRecord,
46
+ conn: sqlite3.Connection | None = None,
47
+ ) -> list[InstructionRecord]:
48
+ """Return instruction records linked to *intrinsic*.
49
+
50
+ When *conn* is provided, performs fast DB lookups by instruction key.
51
+ Otherwise falls back to scanning ``catalog.instructions``.
52
+ """
53
+ ref_keys = [
54
+ ref.get("key", "").strip()
55
+ for ref in intrinsic.instruction_refs
56
+ if isinstance(ref, dict) and ref.get("key", "").strip()
57
+ ]
58
+ if conn is not None:
59
+ linked: list[InstructionRecord] = []
60
+ keys = ref_keys or intrinsic.instructions
61
+ for instruction_key in keys:
62
+ instruction = load_instruction_from_db(conn, instruction_key)
63
+ if instruction is not None:
64
+ linked.append(instruction)
65
+ return linked
66
+ if catalog is None:
67
+ return []
68
+ return [
69
+ instruction
70
+ for instruction in catalog.instructions
71
+ if intrinsic.name in instruction.linked_intrinsics
72
+ ]
73
+
74
+
75
+ def build_intrinsic_instruction_index(
76
+ catalog: Catalog,
77
+ ) -> dict[str, list[InstructionRecord]]:
78
+ """Invert ``catalog.instructions`` into an ``intrinsic name → instructions`` map.
79
+
80
+ One linear pass replaces the quadratic per-intrinsic scan performed by
81
+ :func:`instruction_rows_for_intrinsic`, turning ``O(intrinsics × instructions)``
82
+ into ``O(intrinsics + instructions)``.
83
+ """
84
+ index: dict[str, list[InstructionRecord]] = {}
85
+ for instruction in catalog.instructions:
86
+ for name in instruction.linked_intrinsics:
87
+ index.setdefault(name, []).append(instruction)
88
+ return index
89
+
90
+
91
+ def _rows_from_linked_instructions(
92
+ linked: list[InstructionRecord],
93
+ ) -> list[dict]:
94
+ rows: list[dict] = []
95
+ for instruction in linked:
96
+ if instruction.metrics:
97
+ for arch, values in sorted(instruction.metrics.items()):
98
+ row = {"instruction": instruction.key, "uarch": arch}
99
+ row.update(values)
100
+ rows.append(row)
101
+ else:
102
+ rows.append({
103
+ "instruction": instruction.key,
104
+ "uarch": "-",
105
+ "latency": "-",
106
+ "throughput": "-",
107
+ "uops": "-",
108
+ "ports": "-",
109
+ })
110
+ return rows
111
+
112
+
113
+ def instruction_rows_for_intrinsic_indexed(
114
+ linked: list[InstructionRecord],
115
+ ) -> list[dict]:
116
+ """Variant of :func:`instruction_rows_for_intrinsic` that takes a prebuilt list."""
117
+ return _rows_from_linked_instructions(linked)
118
+
119
+
120
+ def instruction_rows_for_intrinsic(
121
+ catalog: Catalog,
122
+ intrinsic: IntrinsicRecord,
123
+ ) -> list[dict]:
124
+ """Build per-microarchitecture metric rows for an intrinsic's linked instructions.
125
+
126
+ Returns a list of dicts with keys ``instruction``, ``uarch``, and
127
+ whatever metric columns the instruction exposes (``latency``,
128
+ ``throughput``, ``uops``, ``ports``).
129
+ """
130
+ rows: list[dict] = []
131
+ for instruction in catalog.instructions:
132
+ if intrinsic.name not in instruction.linked_intrinsics:
133
+ continue
134
+ if instruction.metrics:
135
+ for arch, values in sorted(instruction.metrics.items()):
136
+ row = {"instruction": instruction.key, "uarch": arch}
137
+ row.update(values)
138
+ rows.append(row)
139
+ else:
140
+ rows.append({
141
+ "instruction": instruction.key,
142
+ "uarch": "-",
143
+ "latency": "-",
144
+ "throughput": "-",
145
+ "uops": "-",
146
+ "ports": "-",
147
+ })
148
+ return rows
149
+
150
+
151
+ def intrinsic_perf_summary(
152
+ catalog: Catalog,
153
+ intrinsic: IntrinsicRecord,
154
+ ) -> tuple[str, str]:
155
+ """Return ``(best_latency, best_cpi)`` across all linked instruction variants.
156
+
157
+ Scans every instruction linked to *intrinsic* in the in-memory catalog.
158
+ """
159
+ from simdref.perf import best_numeric
160
+
161
+ linked = linked_instruction_records(catalog, intrinsic)
162
+ latencies: list[str] = []
163
+ throughput: list[str] = []
164
+ for item in linked:
165
+ lat, cpi = variant_perf_summary(item.arch_details)
166
+ if lat != "-":
167
+ latencies.append(lat)
168
+ if cpi != "-":
169
+ throughput.append(cpi)
170
+ return best_numeric(latencies), best_numeric(throughput)
171
+
172
+
173
+ def intrinsic_perf_summary_runtime(
174
+ conn: sqlite3.Connection,
175
+ intrinsic: IntrinsicRecord,
176
+ instruction_map: dict[str, object],
177
+ ) -> tuple[str, str]:
178
+ """Like :func:`intrinsic_perf_summary` but uses DB + a mutable cache.
179
+
180
+ Resolves linked instructions via *instruction_map* (populated during
181
+ search) and falls back to ``load_instruction_from_db`` for cache misses.
182
+ """
183
+ from simdref.perf import best_numeric
184
+
185
+ ref_keys = [
186
+ ref.get("key", "").strip()
187
+ for ref in intrinsic.instruction_refs
188
+ if isinstance(ref, dict) and ref.get("key", "").strip()
189
+ ]
190
+ linked: list[InstructionRecord] = []
191
+ keys = ref_keys or intrinsic.instructions
192
+ for key in keys:
193
+ instruction = instruction_map.get(key)
194
+ if instruction is None:
195
+ instruction = load_instruction_from_db(conn, key)
196
+ if instruction is not None:
197
+ instruction_map[key] = instruction
198
+ if instruction is not None:
199
+ linked.append(instruction)
200
+ if not linked:
201
+ return "-", "-"
202
+ perf_pairs = [variant_perf_summary(record.arch_details) for record in linked]
203
+ latencies = [float(lat) for lat, _ in perf_pairs if lat not in {"-", ""}]
204
+ cpis = [float(cpi_value) for _, cpi_value in perf_pairs if cpi_value not in {"-", ""}]
205
+ lat = str(min(latencies)).rstrip("0").rstrip(".") if latencies else "-"
206
+ cpi = f"{min(cpis):.2f}" if cpis else "-"
207
+ return lat, cpi