simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
simdref/lsp.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
import sys
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from simdref.perf import best_cpi, best_latency
|
|
9
|
+
from simdref.queries import linked_instruction_records
|
|
10
|
+
from simdref.search import search_records
|
|
11
|
+
from simdref.storage import (
|
|
12
|
+
load_intrinsic_from_db,
|
|
13
|
+
load_instruction_from_db,
|
|
14
|
+
open_db,
|
|
15
|
+
search_intrinsic_candidates_from_db,
|
|
16
|
+
search_instruction_candidates_from_db,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_.]*")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class Session:
|
|
25
|
+
documents: dict[str, str]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _jsonrpc_write(payload: dict) -> None:
|
|
29
|
+
body = json.dumps(payload).encode("utf-8")
|
|
30
|
+
sys.stdout.buffer.write(f"Content-Length: {len(body)}\r\n\r\n".encode("ascii"))
|
|
31
|
+
sys.stdout.buffer.write(body)
|
|
32
|
+
sys.stdout.buffer.flush()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _jsonrpc_read() -> dict | None:
|
|
36
|
+
headers = {}
|
|
37
|
+
while True:
|
|
38
|
+
line = sys.stdin.buffer.readline()
|
|
39
|
+
if not line:
|
|
40
|
+
return None
|
|
41
|
+
if line in (b"\r\n", b"\n"):
|
|
42
|
+
break
|
|
43
|
+
key, value = line.decode("ascii").split(":", 1)
|
|
44
|
+
headers[key.strip().lower()] = value.strip()
|
|
45
|
+
length = int(headers.get("content-length", "0"))
|
|
46
|
+
if length <= 0:
|
|
47
|
+
return None
|
|
48
|
+
body = sys.stdin.buffer.read(length)
|
|
49
|
+
return json.loads(body)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _word_at(text: str, line: int, character: int) -> str | None:
|
|
53
|
+
lines = text.splitlines()
|
|
54
|
+
if line >= len(lines):
|
|
55
|
+
return None
|
|
56
|
+
current = lines[line]
|
|
57
|
+
for match in WORD_RE.finditer(current):
|
|
58
|
+
if match.start() <= character <= match.end():
|
|
59
|
+
return match.group(0)
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _line_prefix(text: str, line: int, character: int) -> str:
|
|
64
|
+
lines = text.splitlines()
|
|
65
|
+
if line >= len(lines):
|
|
66
|
+
return ""
|
|
67
|
+
current = lines[line][:character]
|
|
68
|
+
match = re.search(r"[A-Za-z_][A-Za-z0-9_.]*$", current)
|
|
69
|
+
return match.group(0) if match else ""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _hover_markdown(conn, word: str) -> str | None:
|
|
73
|
+
intrinsic = load_intrinsic_from_db(conn, word)
|
|
74
|
+
if intrinsic is not None:
|
|
75
|
+
lines = [f"```c\n{intrinsic.signature}\n```"]
|
|
76
|
+
if intrinsic.description:
|
|
77
|
+
lines.append(intrinsic.description)
|
|
78
|
+
meta = []
|
|
79
|
+
if intrinsic.header:
|
|
80
|
+
meta.append(f"header `{intrinsic.header}`")
|
|
81
|
+
if intrinsic.isa:
|
|
82
|
+
meta.append(f"ISA {', '.join(intrinsic.isa)}")
|
|
83
|
+
if intrinsic.category:
|
|
84
|
+
meta.append(f"category {intrinsic.category}")
|
|
85
|
+
if intrinsic.url:
|
|
86
|
+
meta.append(f"[source]({intrinsic.url})")
|
|
87
|
+
if meta:
|
|
88
|
+
lines.append(" | ".join(meta))
|
|
89
|
+
if intrinsic.instructions:
|
|
90
|
+
lines.append(f"Instructions: {', '.join(intrinsic.instructions[:6])}")
|
|
91
|
+
linked = linked_instruction_records(None, intrinsic, conn=conn)
|
|
92
|
+
if linked:
|
|
93
|
+
latencies = [best_latency(item.arch_details) for item in linked if best_latency(item.arch_details) != "-"]
|
|
94
|
+
throughputs = [best_cpi(item.arch_details) for item in linked if best_cpi(item.arch_details) != "-"]
|
|
95
|
+
perf = []
|
|
96
|
+
if latencies:
|
|
97
|
+
perf.append(f"best latency {min(latencies, key=lambda value: float(value))} cycles")
|
|
98
|
+
if throughputs:
|
|
99
|
+
perf.append(f"best cycle/instr {min(throughputs, key=lambda value: float(value))}")
|
|
100
|
+
if perf:
|
|
101
|
+
lines.append("Performance: " + ", ".join(perf))
|
|
102
|
+
return "\n\n".join(lines)
|
|
103
|
+
|
|
104
|
+
instruction = load_instruction_from_db(conn, word)
|
|
105
|
+
if instruction is not None:
|
|
106
|
+
lines = [f"```asm\n{instruction.key}\n```"]
|
|
107
|
+
if instruction.summary:
|
|
108
|
+
lines.append(instruction.summary)
|
|
109
|
+
meta = []
|
|
110
|
+
if instruction.isa:
|
|
111
|
+
meta.append(f"ISA {', '.join(instruction.isa)}")
|
|
112
|
+
if instruction.metadata.get("category"):
|
|
113
|
+
meta.append(f"category {instruction.metadata['category']}")
|
|
114
|
+
if meta:
|
|
115
|
+
lines.append(" | ".join(meta))
|
|
116
|
+
if instruction.linked_intrinsics:
|
|
117
|
+
lines.append(f"Intrinsics: {', '.join(instruction.linked_intrinsics[:6])}")
|
|
118
|
+
perf = []
|
|
119
|
+
lat = best_latency(instruction.arch_details)
|
|
120
|
+
cpi = best_cpi(instruction.arch_details)
|
|
121
|
+
if lat != "-":
|
|
122
|
+
perf.append(f"best latency {lat} cycles")
|
|
123
|
+
if cpi != "-":
|
|
124
|
+
perf.append(f"best cycle/instr {cpi}")
|
|
125
|
+
if perf:
|
|
126
|
+
lines.append("Performance: " + ", ".join(perf))
|
|
127
|
+
return "\n\n".join(lines)
|
|
128
|
+
return None
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _completion_candidates(conn, prefix: str, limit: int = 50) -> list[dict]:
|
|
132
|
+
prefix_folded = prefix.casefold()
|
|
133
|
+
emitted: set[tuple[str, str]] = set()
|
|
134
|
+
items: list[dict] = []
|
|
135
|
+
candidate_limit = max(limit * 3, 100)
|
|
136
|
+
intrinsics = search_intrinsic_candidates_from_db(conn, prefix or "_mm", limit=candidate_limit)
|
|
137
|
+
instructions = search_instruction_candidates_from_db(conn, prefix or "_mm", limit=candidate_limit)
|
|
138
|
+
for result in search_records(intrinsics, instructions, prefix or "_mm", limit=candidate_limit):
|
|
139
|
+
label = result.title
|
|
140
|
+
if prefix_folded and not label.casefold().startswith(prefix_folded):
|
|
141
|
+
continue
|
|
142
|
+
key = (result.kind, label)
|
|
143
|
+
if key in emitted:
|
|
144
|
+
continue
|
|
145
|
+
emitted.add(key)
|
|
146
|
+
kind = 3 if result.kind == "intrinsic" else 14
|
|
147
|
+
items.append({"label": label, "kind": kind, "detail": result.subtitle, "insertText": label})
|
|
148
|
+
if len(items) >= limit:
|
|
149
|
+
return items
|
|
150
|
+
return items
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def main() -> int:
|
|
154
|
+
conn = open_db()
|
|
155
|
+
session = Session(documents={})
|
|
156
|
+
while True:
|
|
157
|
+
message = _jsonrpc_read()
|
|
158
|
+
if message is None:
|
|
159
|
+
return 0
|
|
160
|
+
method = message.get("method")
|
|
161
|
+
if method == "initialize":
|
|
162
|
+
_jsonrpc_write(
|
|
163
|
+
{
|
|
164
|
+
"jsonrpc": "2.0",
|
|
165
|
+
"id": message["id"],
|
|
166
|
+
"result": {
|
|
167
|
+
"capabilities": {
|
|
168
|
+
"hoverProvider": True,
|
|
169
|
+
"textDocumentSync": 1,
|
|
170
|
+
"completionProvider": {"resolveProvider": False, "triggerCharacters": ["_", ".", "m", "v"]},
|
|
171
|
+
}
|
|
172
|
+
},
|
|
173
|
+
}
|
|
174
|
+
)
|
|
175
|
+
elif method == "initialized":
|
|
176
|
+
continue
|
|
177
|
+
elif method == "shutdown":
|
|
178
|
+
_jsonrpc_write({"jsonrpc": "2.0", "id": message["id"], "result": None})
|
|
179
|
+
elif method == "exit":
|
|
180
|
+
return 0
|
|
181
|
+
elif method == "textDocument/didOpen":
|
|
182
|
+
params = message["params"]
|
|
183
|
+
session.documents[params["textDocument"]["uri"]] = params["textDocument"]["text"]
|
|
184
|
+
elif method == "textDocument/didChange":
|
|
185
|
+
params = message["params"]
|
|
186
|
+
session.documents[params["textDocument"]["uri"]] = params["contentChanges"][-1]["text"]
|
|
187
|
+
elif method == "textDocument/didClose":
|
|
188
|
+
params = message["params"]
|
|
189
|
+
session.documents.pop(params["textDocument"]["uri"], None)
|
|
190
|
+
elif method == "textDocument/hover":
|
|
191
|
+
params = message["params"]
|
|
192
|
+
uri = params["textDocument"]["uri"]
|
|
193
|
+
text = session.documents.get(uri, "")
|
|
194
|
+
word = _word_at(text, params["position"]["line"], params["position"]["character"])
|
|
195
|
+
contents = None
|
|
196
|
+
if word:
|
|
197
|
+
body = _hover_markdown(conn, word)
|
|
198
|
+
if body:
|
|
199
|
+
contents = {"kind": "markdown", "value": body}
|
|
200
|
+
_jsonrpc_write({"jsonrpc": "2.0", "id": message["id"], "result": {"contents": contents} if contents else None})
|
|
201
|
+
elif method == "textDocument/completion":
|
|
202
|
+
params = message["params"]
|
|
203
|
+
uri = params["textDocument"]["uri"]
|
|
204
|
+
text = session.documents.get(uri, "")
|
|
205
|
+
prefix = _line_prefix(text, params["position"]["line"], params["position"]["character"])
|
|
206
|
+
items = _completion_candidates(conn, prefix)
|
|
207
|
+
_jsonrpc_write({"jsonrpc": "2.0", "id": message["id"], "result": {"isIncomplete": False, "items": items}})
|
|
208
|
+
return 0
|
simdref/manpages.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""Roff manpage generation for intrinsics and instructions.
|
|
2
|
+
|
|
3
|
+
Generates man7 pages that can be viewed with ``man -M share/man <name>``.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import subprocess
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from simdref.display import display_architecture
|
|
13
|
+
from simdref.models import Catalog, InstructionRecord, IntrinsicRecord
|
|
14
|
+
from simdref.queries import (
|
|
15
|
+
build_intrinsic_instruction_index,
|
|
16
|
+
instruction_rows_for_intrinsic,
|
|
17
|
+
instruction_rows_for_intrinsic_indexed,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _roff_escape(text: str) -> str:
|
|
22
|
+
return text.replace("\\", "\\\\").replace("-", "\\-")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _section(title: str, body: str) -> str:
|
|
26
|
+
return f".SH {title}\n{body}\n"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _metric_lines(record: InstructionRecord) -> list[str]:
|
|
30
|
+
lines = []
|
|
31
|
+
for arch, values in sorted(record.metrics.items()):
|
|
32
|
+
rendered = ", ".join(f"{key}={value}" for key, value in sorted(values.items()))
|
|
33
|
+
lines.append(f"{arch}: {rendered}")
|
|
34
|
+
return lines
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _instruction_perf_lines(linked: list[InstructionRecord]) -> list[str]:
|
|
38
|
+
"""Format linked instruction performance data as text lines for man pages."""
|
|
39
|
+
lines = []
|
|
40
|
+
for row in instruction_rows_for_intrinsic_indexed(linked):
|
|
41
|
+
instruction = row.get("instruction", "-")
|
|
42
|
+
uarch = row.get("uarch", "-")
|
|
43
|
+
metrics = {k: v for k, v in row.items() if k not in {"instruction", "uarch"}}
|
|
44
|
+
if metrics and uarch != "-":
|
|
45
|
+
rendered = ", ".join(f"{key}={value}" for key, value in sorted(metrics.items()))
|
|
46
|
+
lines.append(f"{instruction} | {uarch}: {rendered}")
|
|
47
|
+
else:
|
|
48
|
+
lines.append(f"{instruction} | no performance metrics available")
|
|
49
|
+
return lines
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def intrinsic_page(
|
|
53
|
+
record: IntrinsicRecord,
|
|
54
|
+
catalog: Catalog,
|
|
55
|
+
linked_instructions: list[InstructionRecord] | None = None,
|
|
56
|
+
) -> str:
|
|
57
|
+
if linked_instructions is None:
|
|
58
|
+
linked_instructions = [
|
|
59
|
+
instruction
|
|
60
|
+
for instruction in catalog.instructions
|
|
61
|
+
if record.name in instruction.linked_intrinsics
|
|
62
|
+
]
|
|
63
|
+
parts = [f'.TH "{record.name}" "7" "simdref" "simdref" "SIMD Intrinsic Reference"\n']
|
|
64
|
+
parts.append(_section("NAME", f"{_roff_escape(record.name)} \\- {_roff_escape(record.description or 'intrinsic')}"))
|
|
65
|
+
parts.append(_section("SYNOPSIS", f".nf\n{_roff_escape(record.signature)}\n.fi"))
|
|
66
|
+
parts.append(_section("DESCRIPTION", _roff_escape(record.description or "No description available.")))
|
|
67
|
+
parts.append(_section("HEADER", _roff_escape(record.header or "Unknown")))
|
|
68
|
+
if record.url:
|
|
69
|
+
parts.append(_section("SOURCE", _roff_escape(record.url)))
|
|
70
|
+
parts.append(_section("ARCHITECTURE", _roff_escape(display_architecture(record.architecture or "Unknown"))))
|
|
71
|
+
parts.append(_section("ISA", _roff_escape(", ".join(record.isa) or "Unknown")))
|
|
72
|
+
parts.append(_section("CATEGORY", _roff_escape(record.category or "Unknown")))
|
|
73
|
+
parts.append(_section("INSTRUCTIONS", _roff_escape(", ".join(record.instructions) or "None linked")))
|
|
74
|
+
perf_text = _roff_escape("\n".join(_instruction_perf_lines(linked_instructions)) or "No performance metrics available.")
|
|
75
|
+
parts.append(_section("PERFORMANCE SUMMARY", perf_text))
|
|
76
|
+
parts.append(_section("PERFORMANCE DETAILS", perf_text))
|
|
77
|
+
parts.append(_section("NOTES", _roff_escape("; ".join(record.notes) or "None")))
|
|
78
|
+
parts.append(_section("SEE ALSO", _roff_escape(", ".join(record.instructions) or "simdref-search(7)")))
|
|
79
|
+
return "".join(parts)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def instruction_page(record: InstructionRecord) -> str:
|
|
83
|
+
parts = [f'.TH "{record.mnemonic}" "7" "simdref" "simdref" "SIMD Instruction Reference"\n']
|
|
84
|
+
parts.append(_section("NAME", f"{_roff_escape(record.key)} \\- {_roff_escape(record.summary)}"))
|
|
85
|
+
if record.description.get("Description"):
|
|
86
|
+
parts.append(_section("DESCRIPTION", _roff_escape(record.description["Description"])))
|
|
87
|
+
else:
|
|
88
|
+
parts.append(_section("DESCRIPTION", _roff_escape(record.summary)))
|
|
89
|
+
if record.description.get("Operation"):
|
|
90
|
+
parts.append(_section("OPERATION", f".nf\n{_roff_escape(record.description['Operation'])}\n.fi"))
|
|
91
|
+
parts.append(_section("ARCHITECTURE", _roff_escape(display_architecture(record.architecture or "Unknown"))))
|
|
92
|
+
parts.append(_section("ISA", _roff_escape(", ".join(record.isa) or "Unknown")))
|
|
93
|
+
parts.append(_section("OPERANDS", _roff_escape("\n".join(record.operands) or "No operand details available.")))
|
|
94
|
+
parts.append(_section("INTRINSICS", _roff_escape(", ".join(record.linked_intrinsics) or "None linked")))
|
|
95
|
+
parts.append(_section("PERFORMANCE DETAILS", _roff_escape("\n".join(_metric_lines(record)) or "No performance metrics available.")))
|
|
96
|
+
if record.description.get("Flags Affected"):
|
|
97
|
+
parts.append(_section("FLAGS AFFECTED", _roff_escape(record.description["Flags Affected"])))
|
|
98
|
+
for exc_key in ("Exceptions", "SIMD Floating-Point Exceptions", "Numeric Exceptions", "Other Exceptions"):
|
|
99
|
+
if record.description.get(exc_key):
|
|
100
|
+
parts.append(_section(exc_key.upper(), _roff_escape(record.description[exc_key])))
|
|
101
|
+
return "".join(parts)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def write_manpages(
|
|
105
|
+
catalog: Catalog,
|
|
106
|
+
man_dir: Path,
|
|
107
|
+
on_progress: Callable[[int, int], None] | None = None,
|
|
108
|
+
) -> None:
|
|
109
|
+
section_dir = man_dir / "man7"
|
|
110
|
+
section_dir.mkdir(parents=True, exist_ok=True)
|
|
111
|
+
index = build_intrinsic_instruction_index(catalog)
|
|
112
|
+
total = len(catalog.intrinsics) + len(catalog.instructions)
|
|
113
|
+
done = 0
|
|
114
|
+
for intrinsic in catalog.intrinsics:
|
|
115
|
+
linked = index.get(intrinsic.name, [])
|
|
116
|
+
(section_dir / f"{intrinsic.name}.7").write_text(intrinsic_page(intrinsic, catalog, linked))
|
|
117
|
+
done += 1
|
|
118
|
+
if on_progress is not None:
|
|
119
|
+
on_progress(done, total)
|
|
120
|
+
for instruction in catalog.instructions:
|
|
121
|
+
filename = f"instruction-{record_slug(instruction.architecture)}-{record_slug(instruction.key)}.7"
|
|
122
|
+
(section_dir / filename).write_text(instruction_page(instruction))
|
|
123
|
+
if instruction.architecture == "x86":
|
|
124
|
+
(section_dir / f"{instruction.mnemonic}.7").write_text(instruction_page(instruction))
|
|
125
|
+
(section_dir / f"{record_slug(instruction.architecture)}-{instruction.mnemonic}.7").write_text(instruction_page(instruction))
|
|
126
|
+
done += 1
|
|
127
|
+
if on_progress is not None:
|
|
128
|
+
on_progress(done, total)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def record_slug(value: str) -> str:
|
|
132
|
+
return "".join(char.lower() if char.isalnum() else "-" for char in value).strip("-")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def open_manpage(name: str, man_dir: Path) -> int:
|
|
136
|
+
target = man_dir / "man7" / f"{name}.7"
|
|
137
|
+
if not target.exists():
|
|
138
|
+
return 1
|
|
139
|
+
return subprocess.call(["man", "-M", str(man_dir), name])
|
simdref/models.py
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""Core data models for simdref.
|
|
2
|
+
|
|
3
|
+
Defines the four dataclasses that represent the catalog:
|
|
4
|
+
:class:`SourceVersion`, :class:`IntrinsicRecord`, :class:`InstructionRecord`,
|
|
5
|
+
and :class:`Catalog`. All use ``slots=True`` for memory efficiency when
|
|
6
|
+
holding thousands of records.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from simdref.pdfrefs import apply_legacy_pdf_metadata, normalize_pdf_refs
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(slots=True)
|
|
18
|
+
class SourceVersion:
|
|
19
|
+
source: str
|
|
20
|
+
version: str
|
|
21
|
+
fetched_at: str
|
|
22
|
+
url: str
|
|
23
|
+
|
|
24
|
+
def to_dict(self) -> dict[str, Any]:
|
|
25
|
+
return {
|
|
26
|
+
"source": self.source,
|
|
27
|
+
"version": self.version,
|
|
28
|
+
"fetched_at": self.fetched_at,
|
|
29
|
+
"url": self.url,
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(slots=True)
|
|
34
|
+
class IntrinsicRecord:
|
|
35
|
+
name: str
|
|
36
|
+
signature: str
|
|
37
|
+
description: str
|
|
38
|
+
header: str
|
|
39
|
+
url: str = ""
|
|
40
|
+
architecture: str = "x86"
|
|
41
|
+
isa: list[str] = field(default_factory=list)
|
|
42
|
+
category: str = ""
|
|
43
|
+
subcategory: str = ""
|
|
44
|
+
instructions: list[str] = field(default_factory=list)
|
|
45
|
+
instruction_refs: list[dict[str, str]] = field(default_factory=list)
|
|
46
|
+
metadata: dict[str, str] = field(default_factory=dict)
|
|
47
|
+
doc_sections: dict[str, str] = field(default_factory=dict)
|
|
48
|
+
notes: list[str] = field(default_factory=list)
|
|
49
|
+
aliases: list[str] = field(default_factory=list)
|
|
50
|
+
source: str = "intel"
|
|
51
|
+
_search_blob: str = field(default="", repr=False)
|
|
52
|
+
|
|
53
|
+
def __post_init__(self) -> None:
|
|
54
|
+
fields = [
|
|
55
|
+
self.name, self.signature, self.description, self.header, self.url,
|
|
56
|
+
self.architecture,
|
|
57
|
+
self.category, " ".join(self.isa), " ".join(self.instructions),
|
|
58
|
+
" ".join(self.aliases), " ".join(self.metadata.values()), " ".join(self.doc_sections.values()),
|
|
59
|
+
]
|
|
60
|
+
self._search_blob = " ".join(x for x in fields if x)
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def search_blob(self) -> str:
|
|
64
|
+
return self._search_blob
|
|
65
|
+
|
|
66
|
+
def to_dict(self) -> dict[str, Any]:
|
|
67
|
+
return {
|
|
68
|
+
"name": self.name,
|
|
69
|
+
"signature": self.signature,
|
|
70
|
+
"description": self.description,
|
|
71
|
+
"header": self.header,
|
|
72
|
+
"url": self.url,
|
|
73
|
+
"architecture": self.architecture,
|
|
74
|
+
"isa": self.isa,
|
|
75
|
+
"category": self.category,
|
|
76
|
+
"subcategory": self.subcategory,
|
|
77
|
+
"instructions": self.instructions,
|
|
78
|
+
"instruction_refs": self.instruction_refs,
|
|
79
|
+
"metadata": self.metadata,
|
|
80
|
+
"doc_sections": self.doc_sections,
|
|
81
|
+
"notes": self.notes,
|
|
82
|
+
"aliases": self.aliases,
|
|
83
|
+
"source": self.source,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass(slots=True)
|
|
88
|
+
class InstructionRecord:
|
|
89
|
+
mnemonic: str
|
|
90
|
+
form: str
|
|
91
|
+
summary: str
|
|
92
|
+
architecture: str = "x86"
|
|
93
|
+
isa: list[str] = field(default_factory=list)
|
|
94
|
+
operand_details: list[dict[str, str]] = field(default_factory=list)
|
|
95
|
+
metadata: dict[str, str] = field(default_factory=dict)
|
|
96
|
+
arch_details: dict[str, dict[str, Any]] = field(default_factory=dict)
|
|
97
|
+
linked_intrinsics: list[str] = field(default_factory=list)
|
|
98
|
+
aliases: list[str] = field(default_factory=list)
|
|
99
|
+
description: dict[str, str] = field(default_factory=dict)
|
|
100
|
+
pdf_refs: list[dict[str, str]] = field(default_factory=list)
|
|
101
|
+
source: str = "uops.info"
|
|
102
|
+
_search_blob: str = field(default="", repr=False)
|
|
103
|
+
_key: str = field(default="", init=False, repr=False)
|
|
104
|
+
_db_key: str = field(default="", init=False, repr=False)
|
|
105
|
+
|
|
106
|
+
def __post_init__(self) -> None:
|
|
107
|
+
self.pdf_refs = normalize_pdf_refs(self.pdf_refs, self.metadata)
|
|
108
|
+
self.metadata = apply_legacy_pdf_metadata(dict(self.metadata), self.pdf_refs)
|
|
109
|
+
self._key = self.form.strip()
|
|
110
|
+
if not self._key:
|
|
111
|
+
self._key = self.mnemonic
|
|
112
|
+
elif not self._key.casefold().startswith(self.mnemonic.casefold()):
|
|
113
|
+
self._key = f"{self.mnemonic} {self._key}".strip()
|
|
114
|
+
self._db_key = f"{self.architecture}:{self._key.casefold()}"
|
|
115
|
+
fields = [
|
|
116
|
+
self.mnemonic, self.form, self.summary, self.architecture,
|
|
117
|
+
" ".join(self.isa), " ".join(self.operands),
|
|
118
|
+
" ".join(self.linked_intrinsics), " ".join(self.aliases),
|
|
119
|
+
]
|
|
120
|
+
self._search_blob = " ".join(x for x in fields if x)
|
|
121
|
+
|
|
122
|
+
@property
|
|
123
|
+
def key(self) -> str:
|
|
124
|
+
return self._key
|
|
125
|
+
|
|
126
|
+
@property
|
|
127
|
+
def db_key(self) -> str:
|
|
128
|
+
return self._db_key
|
|
129
|
+
|
|
130
|
+
@property
|
|
131
|
+
def operands(self) -> list[str]:
|
|
132
|
+
rendered: list[str] = []
|
|
133
|
+
for operand in self.operand_details:
|
|
134
|
+
rendered_rw = "".join(flag for flag in ("r", "w") if operand.get(flag) == "1")
|
|
135
|
+
idx = operand.get("idx", "")
|
|
136
|
+
text_parts = [
|
|
137
|
+
f"idx={idx}" if idx else "",
|
|
138
|
+
rendered_rw,
|
|
139
|
+
operand.get("type", ""),
|
|
140
|
+
operand.get("width", ""),
|
|
141
|
+
operand.get("xtype", ""),
|
|
142
|
+
operand.get("name", ""),
|
|
143
|
+
]
|
|
144
|
+
text = " ".join(part for part in text_parts if part)
|
|
145
|
+
if text:
|
|
146
|
+
rendered.append(text)
|
|
147
|
+
return rendered
|
|
148
|
+
|
|
149
|
+
@property
|
|
150
|
+
def search_blob(self) -> str:
|
|
151
|
+
return self._search_blob
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def metrics(self) -> dict[str, dict[str, str]]:
|
|
155
|
+
return {
|
|
156
|
+
arch: measurement
|
|
157
|
+
for arch, details in self.arch_details.items()
|
|
158
|
+
if (measurement := details.get("measurement"))
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
def to_dict(self) -> dict[str, Any]:
|
|
162
|
+
metadata = apply_legacy_pdf_metadata(dict(self.metadata), self.pdf_refs)
|
|
163
|
+
return {
|
|
164
|
+
"mnemonic": self.mnemonic,
|
|
165
|
+
"form": self.form,
|
|
166
|
+
"summary": self.summary,
|
|
167
|
+
"architecture": self.architecture,
|
|
168
|
+
"isa": self.isa,
|
|
169
|
+
"operand_details": self.operand_details,
|
|
170
|
+
"metadata": metadata,
|
|
171
|
+
"arch_details": self.arch_details,
|
|
172
|
+
"linked_intrinsics": self.linked_intrinsics,
|
|
173
|
+
"aliases": self.aliases,
|
|
174
|
+
"description": self.description,
|
|
175
|
+
"pdf_refs": self.pdf_refs,
|
|
176
|
+
"source": self.source,
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
@classmethod
|
|
180
|
+
def from_dict(cls, payload: dict[str, Any]) -> "InstructionRecord":
|
|
181
|
+
data = dict(payload)
|
|
182
|
+
data.pop("metrics", None)
|
|
183
|
+
data.pop("operands", None)
|
|
184
|
+
data.setdefault("architecture", "x86")
|
|
185
|
+
data.setdefault("description", {})
|
|
186
|
+
metadata = dict(data.get("metadata") or {})
|
|
187
|
+
data["metadata"] = metadata
|
|
188
|
+
data["pdf_refs"] = normalize_pdf_refs(data.get("pdf_refs"), metadata)
|
|
189
|
+
return cls(**data)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@dataclass(slots=True)
|
|
193
|
+
class Catalog:
|
|
194
|
+
intrinsics: list[IntrinsicRecord]
|
|
195
|
+
instructions: list[InstructionRecord]
|
|
196
|
+
sources: list[SourceVersion]
|
|
197
|
+
generated_at: str
|
|
198
|
+
|
|
199
|
+
def to_dict(self) -> dict[str, Any]:
|
|
200
|
+
return {
|
|
201
|
+
"intrinsics": [item.to_dict() for item in self.intrinsics],
|
|
202
|
+
"instructions": [item.to_dict() for item in self.instructions],
|
|
203
|
+
"sources": [item.to_dict() for item in self.sources],
|
|
204
|
+
"generated_at": self.generated_at,
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
@classmethod
|
|
208
|
+
def from_dict(cls, payload: dict[str, Any]) -> "Catalog":
|
|
209
|
+
return cls(
|
|
210
|
+
intrinsics=[
|
|
211
|
+
IntrinsicRecord(architecture="x86", **item) if "architecture" not in item else IntrinsicRecord(**item)
|
|
212
|
+
for item in payload.get("intrinsics", [])
|
|
213
|
+
],
|
|
214
|
+
instructions=[InstructionRecord.from_dict(item) for item in payload.get("instructions", [])],
|
|
215
|
+
sources=[
|
|
216
|
+
SourceVersion(
|
|
217
|
+
source=item.get("source", ""),
|
|
218
|
+
version=item.get("version", ""),
|
|
219
|
+
fetched_at=item.get("fetched_at", ""),
|
|
220
|
+
url=item.get("url", ""),
|
|
221
|
+
)
|
|
222
|
+
for item in payload.get("sources", [])
|
|
223
|
+
],
|
|
224
|
+
generated_at=payload["generated_at"],
|
|
225
|
+
)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""PDF parsing pipeline for extracting instruction descriptions."""
|
|
2
|
+
|
|
3
|
+
from simdref.pdfparse.registry import get_pdf_source, iter_pdf_sources, register_pdf_source
|
|
4
|
+
from simdref.pdfparse.types import PdfDescriptionPayload, PdfEnrichmentResult, PdfSourceSpec
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"PdfDescriptionPayload",
|
|
8
|
+
"PdfEnrichmentResult",
|
|
9
|
+
"PdfSourceSpec",
|
|
10
|
+
"get_pdf_source",
|
|
11
|
+
"iter_pdf_sources",
|
|
12
|
+
"register_pdf_source",
|
|
13
|
+
]
|