simdref 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simdref/__init__.py +6 -0
- simdref/__main__.py +6 -0
- simdref/annotate.py +448 -0
- simdref/arm_instructions.py +417 -0
- simdref/cli.py +1598 -0
- simdref/display.py +963 -0
- simdref/filters.py +318 -0
- simdref/ingest.py +113 -0
- simdref/ingest_catalog.py +1172 -0
- simdref/ingest_pdf.py +188 -0
- simdref/ingest_sources.py +580 -0
- simdref/lsp.py +208 -0
- simdref/manpages.py +139 -0
- simdref/models.py +225 -0
- simdref/pdfparse/__init__.py +13 -0
- simdref/pdfparse/base.py +116 -0
- simdref/pdfparse/intel.py +614 -0
- simdref/pdfparse/registry.py +19 -0
- simdref/pdfparse/types.py +77 -0
- simdref/pdfrefs.py +95 -0
- simdref/perf.py +220 -0
- simdref/perf_sources/__init__.py +51 -0
- simdref/perf_sources/cores.py +101 -0
- simdref/perf_sources/llvm_mca.py +176 -0
- simdref/perf_sources/llvm_scheduling.py +625 -0
- simdref/perf_sources/merge.py +121 -0
- simdref/queries.py +207 -0
- simdref/riscv.py +446 -0
- simdref/search.py +288 -0
- simdref/storage.py +504 -0
- simdref/templates/__init__.py +0 -0
- simdref/templates/app.js +1590 -0
- simdref/templates/favicon.svg +5 -0
- simdref/templates/index.html +112 -0
- simdref/templates/logo.svg +12 -0
- simdref/templates/style.css +680 -0
- simdref/tui.py +2366 -0
- simdref/web.py +403 -0
- simdref-0.0.0.dist-info/METADATA +240 -0
- simdref-0.0.0.dist-info/RECORD +44 -0
- simdref-0.0.0.dist-info/WHEEL +5 -0
- simdref-0.0.0.dist-info/entry_points.txt +4 -0
- simdref-0.0.0.dist-info/licenses/LICENSE +674 -0
- simdref-0.0.0.dist-info/top_level.txt +1 -0
simdref/cli.py
ADDED
|
@@ -0,0 +1,1598 @@
|
|
|
1
|
+
"""Command-line interface for simdref.
|
|
2
|
+
|
|
3
|
+
This module defines the Typer application, its maintenance/export commands,
|
|
4
|
+
and the smart bare-word lookup that fires when no recognised subcommand is
|
|
5
|
+
given.
|
|
6
|
+
|
|
7
|
+
Display and formatting logic lives in :mod:`simdref.display`.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
import shutil
|
|
15
|
+
import subprocess
|
|
16
|
+
import sys
|
|
17
|
+
from contextlib import contextmanager
|
|
18
|
+
from contextlib import nullcontext as _nullcontext
|
|
19
|
+
from dataclasses import asdict
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
import fnmatch
|
|
23
|
+
|
|
24
|
+
import click
|
|
25
|
+
import httpx
|
|
26
|
+
import typer
|
|
27
|
+
from rich.console import Console
|
|
28
|
+
|
|
29
|
+
# Usage errors (missing args, bad flags) normally exit 2 in Click. We reserve
|
|
30
|
+
# exit code 2 strictly for "query valid but no catalog match" in `simdref llm`,
|
|
31
|
+
# so downgrade Click usage errors to exit 1 at the CLI boundary.
|
|
32
|
+
click.exceptions.UsageError.exit_code = 1
|
|
33
|
+
from rich.progress import (
|
|
34
|
+
BarColumn,
|
|
35
|
+
DownloadColumn,
|
|
36
|
+
MofNCompleteColumn,
|
|
37
|
+
Progress,
|
|
38
|
+
SpinnerColumn,
|
|
39
|
+
TaskProgressColumn,
|
|
40
|
+
TextColumn,
|
|
41
|
+
TimeElapsedColumn,
|
|
42
|
+
TimeRemainingColumn,
|
|
43
|
+
TransferSpeedColumn,
|
|
44
|
+
)
|
|
45
|
+
from typer.core import TyperGroup
|
|
46
|
+
|
|
47
|
+
# Stderr-only Console for bootstrap/download status. Must not share stdout with
|
|
48
|
+
# `simdref llm` payloads, otherwise callers that json.loads(stdout) break.
|
|
49
|
+
err_console = Console(stderr=True)
|
|
50
|
+
|
|
51
|
+
from simdref.display import (
|
|
52
|
+
console,
|
|
53
|
+
display_architecture,
|
|
54
|
+
display_isa,
|
|
55
|
+
instruction_query_text,
|
|
56
|
+
instruction_variant_items,
|
|
57
|
+
isa_sort_key,
|
|
58
|
+
isa_visible,
|
|
59
|
+
normalize_instruction_query,
|
|
60
|
+
render_intrinsic,
|
|
61
|
+
render_instruction,
|
|
62
|
+
render_instruction_variants,
|
|
63
|
+
render_search_results,
|
|
64
|
+
)
|
|
65
|
+
from simdref import __version__
|
|
66
|
+
from simdref.ingest import build_catalog
|
|
67
|
+
from simdref.ingest_sources import (
|
|
68
|
+
ARM_A64_ARCHIVE_CACHE,
|
|
69
|
+
refresh_local_arm_a64_archive,
|
|
70
|
+
refresh_local_arm_intrinsics_bundle,
|
|
71
|
+
)
|
|
72
|
+
from simdref.manpages import open_manpage, write_manpages
|
|
73
|
+
from simdref.perf import variant_perf_summary
|
|
74
|
+
from simdref.queries import intrinsic_perf_summary_runtime, instruction_rows_for_intrinsic
|
|
75
|
+
from simdref.search import SearchResult, find_intrinsic, find_instructions, search_catalog, search_records
|
|
76
|
+
from simdref.storage import (
|
|
77
|
+
CATALOG_PATH,
|
|
78
|
+
DATA_DIR,
|
|
79
|
+
DEFAULT_MAN_DIR,
|
|
80
|
+
SQLITE_PATH,
|
|
81
|
+
WEB_DIR,
|
|
82
|
+
build_sqlite,
|
|
83
|
+
load_catalog,
|
|
84
|
+
load_instruction_from_db,
|
|
85
|
+
load_intrinsic_from_db,
|
|
86
|
+
load_instructions_by_mnemonic_from_db,
|
|
87
|
+
load_instructions_by_mnemonic_prefix_from_db,
|
|
88
|
+
open_db,
|
|
89
|
+
save_catalog,
|
|
90
|
+
search_instruction_candidates_from_db,
|
|
91
|
+
search_intrinsic_candidates_from_db,
|
|
92
|
+
sqlite_schema_is_current,
|
|
93
|
+
)
|
|
94
|
+
from simdref.web import export_web
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _run_tui(
|
|
98
|
+
*,
|
|
99
|
+
initial_query: str = "",
|
|
100
|
+
initial_preset: str | None = None,
|
|
101
|
+
initial_view: str = "search",
|
|
102
|
+
initial_asm: str = "",
|
|
103
|
+
):
|
|
104
|
+
from simdref.tui import run_tui
|
|
105
|
+
|
|
106
|
+
return run_tui(
|
|
107
|
+
initial_query=initial_query,
|
|
108
|
+
initial_preset=initial_preset,
|
|
109
|
+
initial_view=initial_view,
|
|
110
|
+
initial_asm=initial_asm,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class SimdrefGroup(TyperGroup):
|
|
115
|
+
"""Top-level CLI group with usage that reflects bare-query mode."""
|
|
116
|
+
|
|
117
|
+
def collect_usage_pieces(self, ctx): # type: ignore[override]
|
|
118
|
+
return ["[OPTIONS] [QUERY] | COMMAND [ARGS]..."]
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
app = typer.Typer(
|
|
122
|
+
cls=SimdrefGroup,
|
|
123
|
+
add_completion=False,
|
|
124
|
+
help=(
|
|
125
|
+
"Local SIMD reference across Intel intrinsics, instruction data, performance measurements, and SDM-derived descriptions.\n\n"
|
|
126
|
+
"Run without arguments to open the TUI. Pass a bare query to search or open matching results directly.\n\n"
|
|
127
|
+
"Installed under two names — 'isa' (short) and 'simdref' (explicit) — both accept every subcommand.\n\n"
|
|
128
|
+
"Common commands:\n"
|
|
129
|
+
" isa update Download the pre-built release catalog (no llvm-mca required).\n"
|
|
130
|
+
" isa build Full local rebuild from upstream sources (requires llvm-mca).\n"
|
|
131
|
+
" isa completion install Install shell completion into your shell profile."
|
|
132
|
+
),
|
|
133
|
+
context_settings={"help_option_names": ["-h", "--help"]},
|
|
134
|
+
)
|
|
135
|
+
SHOW_FP16_ISAS = False
|
|
136
|
+
SHORT_MODE = False
|
|
137
|
+
FULL_MODE = False
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@contextmanager
|
|
141
|
+
def _pager_context():
|
|
142
|
+
"""Pipe Rich output through a pager that handles ANSI colors."""
|
|
143
|
+
pager_cmd = os.environ.get("PAGER", "")
|
|
144
|
+
less = shutil.which("less")
|
|
145
|
+
if less and ("less" in pager_cmd or not pager_cmd):
|
|
146
|
+
# Use less -RFX: Raw ANSI, quit-if-one-screen, no-init
|
|
147
|
+
from rich.pager import Pager
|
|
148
|
+
|
|
149
|
+
class _LessPager(Pager):
|
|
150
|
+
def show(self, content: str) -> None:
|
|
151
|
+
proc = subprocess.Popen(
|
|
152
|
+
[less, "-RFX"],
|
|
153
|
+
stdin=subprocess.PIPE,
|
|
154
|
+
encoding="utf-8",
|
|
155
|
+
errors="replace",
|
|
156
|
+
)
|
|
157
|
+
try:
|
|
158
|
+
proc.communicate(input=content)
|
|
159
|
+
except KeyboardInterrupt:
|
|
160
|
+
proc.kill()
|
|
161
|
+
|
|
162
|
+
yield console.pager(pager=_LessPager(), styles=True)
|
|
163
|
+
else:
|
|
164
|
+
# Fallback: Rich's default pager without styles (safe)
|
|
165
|
+
yield console.pager(styles=False)
|
|
166
|
+
|
|
167
|
+
GITHUB_REPO = "DiamonDinoia/simdref"
|
|
168
|
+
RELEASE_TAG = "data-latest"
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# ---------------------------------------------------------------------------
|
|
172
|
+
# Release download helpers
|
|
173
|
+
# ---------------------------------------------------------------------------
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _release_tag_candidates() -> list[str]:
|
|
177
|
+
return [f"data-v{__version__}", RELEASE_TAG]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _release_asset_url(tag: str, asset_name: str) -> str:
|
|
181
|
+
return f"https://github.com/{GITHUB_REPO}/releases/download/{tag}/{asset_name}"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _download_from_release() -> None:
|
|
185
|
+
"""Download pre-built catalog and database from GitHub Release."""
|
|
186
|
+
from simdref.storage import ensure_dir
|
|
187
|
+
|
|
188
|
+
ensure_dir(DATA_DIR)
|
|
189
|
+
|
|
190
|
+
for asset in ("catalog.msgpack", "catalog.db"):
|
|
191
|
+
dest = DATA_DIR / asset
|
|
192
|
+
for tag in _release_tag_candidates():
|
|
193
|
+
url = _release_asset_url(tag, asset)
|
|
194
|
+
err_console.print(f"downloading {asset} from {tag}...", style="dim")
|
|
195
|
+
try:
|
|
196
|
+
with httpx.stream("GET", url, follow_redirects=True, timeout=120) as resp:
|
|
197
|
+
resp.raise_for_status()
|
|
198
|
+
with open(dest, "wb") as f:
|
|
199
|
+
for chunk in resp.iter_bytes(chunk_size=1024 * 64):
|
|
200
|
+
f.write(chunk)
|
|
201
|
+
break
|
|
202
|
+
except httpx.HTTPStatusError as exc:
|
|
203
|
+
if exc.response.status_code == 404:
|
|
204
|
+
continue
|
|
205
|
+
err_console.print(f"failed to download {asset}: {exc.response.status_code}", style="red")
|
|
206
|
+
raise typer.Exit(code=1) from exc
|
|
207
|
+
else:
|
|
208
|
+
err_console.print(f"failed to download {asset}: no compatible release asset found", style="red")
|
|
209
|
+
err_console.print("try 'simdref build' to build locally", style="yellow")
|
|
210
|
+
raise typer.Exit(code=1)
|
|
211
|
+
err_console.print("download complete", style="green")
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _build_runtime_locally(*, man_dir: Path, include_sdm: bool = False) -> None:
|
|
215
|
+
"""Build catalog, SQLite, manpages, and web bundle locally.
|
|
216
|
+
|
|
217
|
+
Renders a single rich.progress.Progress that shows a per-phase ETA.
|
|
218
|
+
Download phases render bytes + transfer speed; processing phases
|
|
219
|
+
render item counts + remaining time.
|
|
220
|
+
"""
|
|
221
|
+
interactive_progress = console.is_terminal and os.environ.get("GITHUB_ACTIONS") != "true"
|
|
222
|
+
|
|
223
|
+
if not interactive_progress:
|
|
224
|
+
def _status(msg: str) -> None:
|
|
225
|
+
err_console.print(msg, style="dim")
|
|
226
|
+
|
|
227
|
+
_status("Refreshing local Arm intrinsics cache")
|
|
228
|
+
try:
|
|
229
|
+
written = refresh_local_arm_intrinsics_bundle()
|
|
230
|
+
except Exception as exc:
|
|
231
|
+
err_console.print(f"Arm intrinsics download failed: {exc}", style="red")
|
|
232
|
+
raise typer.Exit(code=1) from exc
|
|
233
|
+
_status(f"Refreshed {len(written)} Arm JSON files in {written[0].parent}")
|
|
234
|
+
_status("Fetching Arm A64 AARCHMRS archive (large, one-time download)")
|
|
235
|
+
try:
|
|
236
|
+
archive_path = refresh_local_arm_a64_archive()
|
|
237
|
+
except Exception as exc:
|
|
238
|
+
err_console.print(f"AARCHMRS download failed: {exc}", style="red")
|
|
239
|
+
raise typer.Exit(code=1) from exc
|
|
240
|
+
_status(f"AARCHMRS archive ready at {archive_path}")
|
|
241
|
+
_status("Building local catalog")
|
|
242
|
+
catalog = build_catalog(include_sdm=include_sdm, status=_status)
|
|
243
|
+
_status("Saving catalog snapshot")
|
|
244
|
+
save_catalog(catalog)
|
|
245
|
+
_status("Building SQLite search database")
|
|
246
|
+
build_sqlite(catalog)
|
|
247
|
+
_status("Writing manpages")
|
|
248
|
+
write_manpages(catalog, man_dir)
|
|
249
|
+
_status("Exporting static web bundle")
|
|
250
|
+
export_web(catalog, WEB_DIR)
|
|
251
|
+
err_console.print(
|
|
252
|
+
f"updated catalog with {len(catalog.intrinsics)} intrinsics and {len(catalog.instructions)} instructions",
|
|
253
|
+
style="green",
|
|
254
|
+
)
|
|
255
|
+
return
|
|
256
|
+
|
|
257
|
+
download_progress = Progress(
|
|
258
|
+
SpinnerColumn(),
|
|
259
|
+
TextColumn("[progress.description]{task.description}"),
|
|
260
|
+
BarColumn(),
|
|
261
|
+
DownloadColumn(),
|
|
262
|
+
TransferSpeedColumn(),
|
|
263
|
+
TimeRemainingColumn(),
|
|
264
|
+
console=err_console,
|
|
265
|
+
transient=True,
|
|
266
|
+
)
|
|
267
|
+
count_progress = Progress(
|
|
268
|
+
SpinnerColumn(),
|
|
269
|
+
TextColumn("[progress.description]{task.description}"),
|
|
270
|
+
BarColumn(),
|
|
271
|
+
MofNCompleteColumn(),
|
|
272
|
+
TaskProgressColumn(),
|
|
273
|
+
TimeElapsedColumn(),
|
|
274
|
+
TimeRemainingColumn(),
|
|
275
|
+
console=err_console,
|
|
276
|
+
transient=True,
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
from rich.live import Live
|
|
280
|
+
from rich.console import Group
|
|
281
|
+
|
|
282
|
+
with Live(Group(download_progress, count_progress), console=err_console, refresh_per_second=10):
|
|
283
|
+
arm_task = count_progress.add_task("Arm intrinsics JSON bundle", total=3)
|
|
284
|
+
try:
|
|
285
|
+
refresh_local_arm_intrinsics_bundle(
|
|
286
|
+
on_progress=lambda done, total: count_progress.update(arm_task, completed=done, total=total),
|
|
287
|
+
)
|
|
288
|
+
except Exception as exc:
|
|
289
|
+
count_progress.update(arm_task, description=f"Arm intrinsics download failed: {exc}")
|
|
290
|
+
raise typer.Exit(code=1) from exc
|
|
291
|
+
count_progress.update(arm_task, description="Arm intrinsics JSON bundle \u2713")
|
|
292
|
+
|
|
293
|
+
if ARM_A64_ARCHIVE_CACHE.exists() and ARM_A64_ARCHIVE_CACHE.stat().st_size > 0:
|
|
294
|
+
cached_task = count_progress.add_task(
|
|
295
|
+
f"Arm A64 AARCHMRS archive \u2713 ({ARM_A64_ARCHIVE_CACHE.name}, cached)",
|
|
296
|
+
total=1,
|
|
297
|
+
)
|
|
298
|
+
count_progress.update(cached_task, completed=1)
|
|
299
|
+
else:
|
|
300
|
+
a64_task = download_progress.add_task("Arm A64 AARCHMRS archive", total=None)
|
|
301
|
+
|
|
302
|
+
def _a64_progress(done: int, total: int | None) -> None:
|
|
303
|
+
download_progress.update(a64_task, completed=done, total=total)
|
|
304
|
+
|
|
305
|
+
try:
|
|
306
|
+
archive_path = refresh_local_arm_a64_archive(on_progress=_a64_progress)
|
|
307
|
+
except Exception as exc:
|
|
308
|
+
download_progress.update(a64_task, description=f"AARCHMRS download failed: {exc}")
|
|
309
|
+
raise typer.Exit(code=1) from exc
|
|
310
|
+
download_progress.update(a64_task, description=f"Arm A64 AARCHMRS archive \u2713 ({archive_path.name})")
|
|
311
|
+
|
|
312
|
+
build_task = count_progress.add_task("Building catalog from sources", total=None)
|
|
313
|
+
|
|
314
|
+
def _build_status(msg: str) -> None:
|
|
315
|
+
count_progress.update(build_task, description=f"Building catalog: {msg}")
|
|
316
|
+
|
|
317
|
+
catalog = build_catalog(include_sdm=include_sdm, status=_build_status)
|
|
318
|
+
count_progress.update(build_task, description="Building catalog \u2713", completed=1, total=1)
|
|
319
|
+
|
|
320
|
+
save_task = count_progress.add_task("Saving catalog snapshot", total=1)
|
|
321
|
+
save_catalog(catalog)
|
|
322
|
+
count_progress.update(save_task, completed=1, description="Saving catalog snapshot \u2713")
|
|
323
|
+
|
|
324
|
+
sqlite_task = count_progress.add_task("Building SQLite search database", total=1)
|
|
325
|
+
build_sqlite(catalog)
|
|
326
|
+
count_progress.update(sqlite_task, completed=1, description="Building SQLite search database \u2713")
|
|
327
|
+
|
|
328
|
+
man_total = len(catalog.intrinsics) + len(catalog.instructions)
|
|
329
|
+
man_task = count_progress.add_task("Writing manpages", total=man_total)
|
|
330
|
+
write_manpages(
|
|
331
|
+
catalog,
|
|
332
|
+
man_dir,
|
|
333
|
+
on_progress=lambda done, total: count_progress.update(man_task, completed=done, total=total),
|
|
334
|
+
)
|
|
335
|
+
count_progress.update(man_task, description="Writing manpages \u2713")
|
|
336
|
+
|
|
337
|
+
web_task = count_progress.add_task("Exporting static web bundle", total=1)
|
|
338
|
+
export_web(catalog, WEB_DIR)
|
|
339
|
+
count_progress.update(web_task, completed=1, description="Exporting static web bundle \u2713")
|
|
340
|
+
|
|
341
|
+
err_console.print(
|
|
342
|
+
f"updated catalog with {len(catalog.intrinsics)} intrinsics and {len(catalog.instructions)} instructions",
|
|
343
|
+
style="green",
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _refresh_runtime_from_existing_catalog(*, man_dir: Path) -> None:
|
|
348
|
+
"""Rebuild derived runtime artifacts from the local msgpack snapshot.
|
|
349
|
+
|
|
350
|
+
This is substantially cheaper than a full local source rebuild and is
|
|
351
|
+
sufficient when the catalog is already present but the SQLite schema,
|
|
352
|
+
manpages, or web export need to be refreshed.
|
|
353
|
+
"""
|
|
354
|
+
if not CATALOG_PATH.exists():
|
|
355
|
+
raise typer.Exit(code=1)
|
|
356
|
+
catalog = load_catalog()
|
|
357
|
+
build_sqlite(catalog)
|
|
358
|
+
write_manpages(catalog, man_dir)
|
|
359
|
+
export_web(catalog, WEB_DIR)
|
|
360
|
+
err_console.print(
|
|
361
|
+
f"refreshed runtime from existing catalog with {len(catalog.intrinsics)} intrinsics and {len(catalog.instructions)} instructions",
|
|
362
|
+
style="green",
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _finalize_runtime_from_download(*, man_dir: Path) -> None:
|
|
367
|
+
"""Refresh local derived artifacts after downloading release assets."""
|
|
368
|
+
if not CATALOG_PATH.exists():
|
|
369
|
+
raise typer.Exit(code=1)
|
|
370
|
+
catalog = load_catalog()
|
|
371
|
+
write_manpages(catalog, man_dir)
|
|
372
|
+
export_web(catalog, WEB_DIR)
|
|
373
|
+
err_console.print(
|
|
374
|
+
f"refreshed local web/man assets from downloaded catalog with {len(catalog.intrinsics)} intrinsics and {len(catalog.instructions)} instructions",
|
|
375
|
+
style="green",
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _download_release_or_fallback(*, man_dir: Path) -> None:
|
|
380
|
+
"""Prefer pre-built assets. Fall back to the existing on-disk catalog.
|
|
381
|
+
|
|
382
|
+
When the download fails and no catalog is cached, the caller must
|
|
383
|
+
run ``simdref update --build`` — there is no longer a bundled-fixture
|
|
384
|
+
fallback.
|
|
385
|
+
"""
|
|
386
|
+
try:
|
|
387
|
+
_download_from_release()
|
|
388
|
+
if sqlite_schema_is_current():
|
|
389
|
+
_finalize_runtime_from_download(man_dir=man_dir)
|
|
390
|
+
return
|
|
391
|
+
if CATALOG_PATH.exists():
|
|
392
|
+
err_console.print("downloaded catalog is usable but SQLite is stale; rebuilding runtime locally from the downloaded catalog", style="yellow")
|
|
393
|
+
_refresh_runtime_from_existing_catalog(man_dir=man_dir)
|
|
394
|
+
return
|
|
395
|
+
err_console.print("[bold red]downloaded runtime schema is not current[/bold red] and no local catalog exists", style="yellow")
|
|
396
|
+
err_console.print("run `simdref update --build` to build from upstream sources (requires llvm-mca)")
|
|
397
|
+
raise typer.Exit(code=1)
|
|
398
|
+
except typer.Exit:
|
|
399
|
+
if CATALOG_PATH.exists():
|
|
400
|
+
err_console.print("download failed; refreshing runtime from the existing local catalog", style="yellow")
|
|
401
|
+
_refresh_runtime_from_existing_catalog(man_dir=man_dir)
|
|
402
|
+
return
|
|
403
|
+
err_console.print("[bold red]no pre-built catalog available and no local cache[/bold red]", style="yellow")
|
|
404
|
+
err_console.print("run `simdref update --build` to build from upstream sources (requires llvm-mca)")
|
|
405
|
+
raise
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
# ---------------------------------------------------------------------------
|
|
409
|
+
# Catalog / runtime helpers
|
|
410
|
+
# ---------------------------------------------------------------------------
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _bootstrap_interactive() -> None:
|
|
414
|
+
"""Bootstrap runtime data with a lightweight default path."""
|
|
415
|
+
err_console.print("\n[bold]No catalog found.[/bold] Downloading pre-built data if available...\n")
|
|
416
|
+
_download_release_or_fallback(man_dir=DEFAULT_MAN_DIR)
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def ensure_catalog():
|
|
420
|
+
"""Load (or bootstrap) the in-memory catalog."""
|
|
421
|
+
if not CATALOG_PATH.exists():
|
|
422
|
+
_bootstrap_interactive()
|
|
423
|
+
return load_catalog()
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def ensure_runtime() -> None:
|
|
427
|
+
"""Ensure catalog + SQLite are present and current."""
|
|
428
|
+
if not CATALOG_PATH.exists():
|
|
429
|
+
_bootstrap_interactive()
|
|
430
|
+
return
|
|
431
|
+
if not sqlite_schema_is_current():
|
|
432
|
+
err_console.print("runtime schema is missing or out of date; rebuilding derived runtime artifacts from the local catalog", style="yellow")
|
|
433
|
+
_refresh_runtime_from_existing_catalog(man_dir=DEFAULT_MAN_DIR)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def _catalog_meta(catalog) -> dict:
|
|
437
|
+
return {
|
|
438
|
+
"generated_at": catalog.generated_at,
|
|
439
|
+
"source_versions": [asdict(source) for source in catalog.sources],
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _search_runtime(conn, query: str, limit: int = 20) -> tuple[list[SearchResult], dict[str, object], dict[str, object]]:
|
|
445
|
+
candidate_limit = max(limit * 6, 60)
|
|
446
|
+
intrinsics = search_intrinsic_candidates_from_db(conn, query, limit=candidate_limit)
|
|
447
|
+
instructions = search_instruction_candidates_from_db(conn, query, limit=candidate_limit)
|
|
448
|
+
results = search_records(intrinsics, instructions, query, limit=limit)
|
|
449
|
+
intrinsic_map = {item.name: item for item in intrinsics}
|
|
450
|
+
instruction_map = {item.db_key: item for item in instructions}
|
|
451
|
+
return results, intrinsic_map, instruction_map
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
# ---------------------------------------------------------------------------
|
|
455
|
+
# Instruction lookup helpers
|
|
456
|
+
# ---------------------------------------------------------------------------
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def _select_instruction_variant(catalog, query: str, items):
|
|
460
|
+
parts = query.split()
|
|
461
|
+
if len(parts) < 2 or not parts[-1].isdigit():
|
|
462
|
+
return None
|
|
463
|
+
base_query = " ".join(parts[:-1]).strip()
|
|
464
|
+
if not base_query:
|
|
465
|
+
return None
|
|
466
|
+
index = int(parts[-1])
|
|
467
|
+
if index < 1:
|
|
468
|
+
return None
|
|
469
|
+
if items:
|
|
470
|
+
variants = instruction_variant_items(items)
|
|
471
|
+
elif catalog is not None:
|
|
472
|
+
variants = instruction_variant_items(find_instructions(catalog, base_query))
|
|
473
|
+
else:
|
|
474
|
+
variants = instruction_variant_items(_find_instructions_fast(base_query))
|
|
475
|
+
if 1 <= index <= len(variants):
|
|
476
|
+
return variants[index - 1]
|
|
477
|
+
return None
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _find_instructions_fast(query: str):
|
|
481
|
+
ensure_runtime()
|
|
482
|
+
with open_db() as conn:
|
|
483
|
+
exact = load_instruction_from_db(conn, query)
|
|
484
|
+
if exact is not None:
|
|
485
|
+
return [exact]
|
|
486
|
+
parts = query.split()
|
|
487
|
+
mnemonic = parts[0] if parts else query
|
|
488
|
+
candidates = load_instructions_by_mnemonic_from_db(conn, mnemonic)
|
|
489
|
+
if not candidates:
|
|
490
|
+
return []
|
|
491
|
+
normalized_query = normalize_instruction_query(query)
|
|
492
|
+
exact_candidates = [
|
|
493
|
+
item
|
|
494
|
+
for item in candidates
|
|
495
|
+
if normalize_instruction_query(item.key) == normalized_query
|
|
496
|
+
or normalize_instruction_query(instruction_query_text(item)) == normalized_query
|
|
497
|
+
or item.mnemonic.casefold() == query.casefold()
|
|
498
|
+
]
|
|
499
|
+
if exact_candidates:
|
|
500
|
+
return exact_candidates
|
|
501
|
+
if mnemonic.casefold() == query.casefold():
|
|
502
|
+
return candidates
|
|
503
|
+
return []
|
|
504
|
+
|
|
505
|
+
|
|
506
|
+
def _find_instruction_family_fast(query: str):
|
|
507
|
+
ensure_runtime()
|
|
508
|
+
token = (query.split()[0] if query.split() else query).strip()
|
|
509
|
+
if not token:
|
|
510
|
+
return []
|
|
511
|
+
with open_db() as conn:
|
|
512
|
+
candidates = load_instructions_by_mnemonic_prefix_from_db(conn, token)
|
|
513
|
+
exact_mnemonic = {item.mnemonic.casefold() for item in candidates}
|
|
514
|
+
if token.casefold() in exact_mnemonic:
|
|
515
|
+
return []
|
|
516
|
+
return candidates
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
# ---------------------------------------------------------------------------
|
|
520
|
+
# LLM / JSON payload builders
|
|
521
|
+
# ---------------------------------------------------------------------------
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def _resolve_query_payload(catalog, query: str, limit: int = 8) -> dict:
|
|
525
|
+
intrinsic = find_intrinsic(catalog, query)
|
|
526
|
+
if intrinsic is not None:
|
|
527
|
+
return {
|
|
528
|
+
"query": query,
|
|
529
|
+
"mode": "exact",
|
|
530
|
+
"match_kind": "intrinsic",
|
|
531
|
+
"intrinsic": asdict(intrinsic),
|
|
532
|
+
"performance": instruction_rows_for_intrinsic(catalog, intrinsic),
|
|
533
|
+
**_catalog_meta(catalog),
|
|
534
|
+
}
|
|
535
|
+
instructions = find_instructions(catalog, query)
|
|
536
|
+
if instructions:
|
|
537
|
+
return {
|
|
538
|
+
"query": query,
|
|
539
|
+
"mode": "exact",
|
|
540
|
+
"match_kind": "instruction",
|
|
541
|
+
"instructions": [asdict(item) | {"key": item.key} for item in instructions],
|
|
542
|
+
**_catalog_meta(catalog),
|
|
543
|
+
}
|
|
544
|
+
return {
|
|
545
|
+
"query": query,
|
|
546
|
+
"mode": "search",
|
|
547
|
+
"match_kind": None,
|
|
548
|
+
"results": [asdict(result) for result in search_catalog(catalog, query, limit=limit)],
|
|
549
|
+
**_catalog_meta(catalog),
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _llm_result_payload(conn, result: SearchResult, intrinsic_map: dict[str, object], instruction_map: dict[str, object]) -> dict:
|
|
554
|
+
if result.kind == "intrinsic":
|
|
555
|
+
item = intrinsic_map.get(result.key)
|
|
556
|
+
if item is None:
|
|
557
|
+
item = load_intrinsic_from_db(conn, result.key)
|
|
558
|
+
if item is not None:
|
|
559
|
+
intrinsic_map[result.key] = item
|
|
560
|
+
if item is not None:
|
|
561
|
+
lat, cpi = intrinsic_perf_summary_runtime(conn, item, instruction_map)
|
|
562
|
+
return {
|
|
563
|
+
"query": item.name,
|
|
564
|
+
"intrinsic": item.name,
|
|
565
|
+
"signature": item.signature,
|
|
566
|
+
"instructions": item.instructions,
|
|
567
|
+
"instruction_refs": item.instruction_refs,
|
|
568
|
+
"summary": item.description,
|
|
569
|
+
"isa": item.isa,
|
|
570
|
+
"lat": lat,
|
|
571
|
+
"cpi": cpi,
|
|
572
|
+
}
|
|
573
|
+
item = instruction_map.get(result.key)
|
|
574
|
+
if item is None:
|
|
575
|
+
item = load_instruction_from_db(conn, result.key)
|
|
576
|
+
if item is not None:
|
|
577
|
+
instruction_map[result.key] = item
|
|
578
|
+
if item is not None:
|
|
579
|
+
lat, cpi = variant_perf_summary(item.arch_details)
|
|
580
|
+
return {
|
|
581
|
+
"query": item.key,
|
|
582
|
+
"intrinsic": item.linked_intrinsics,
|
|
583
|
+
"summary": item.summary,
|
|
584
|
+
"isa": item.isa,
|
|
585
|
+
"lat": lat,
|
|
586
|
+
"cpi": cpi,
|
|
587
|
+
}
|
|
588
|
+
return {"query": result.title, "intrinsic": [], "summary": result.subtitle, "isa": [], "lat": "-", "cpi": "-"}
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def _llm_intrinsic_payload(conn, intrinsic) -> dict:
|
|
592
|
+
instruction_map: dict[str, object] = {}
|
|
593
|
+
lat, cpi = intrinsic_perf_summary_runtime(conn, intrinsic, instruction_map)
|
|
594
|
+
return {
|
|
595
|
+
"query": intrinsic.name,
|
|
596
|
+
"intrinsic": intrinsic.name,
|
|
597
|
+
"signature": intrinsic.signature,
|
|
598
|
+
"url": intrinsic.url,
|
|
599
|
+
"instructions": intrinsic.instructions,
|
|
600
|
+
"instruction_refs": intrinsic.instruction_refs,
|
|
601
|
+
"isa": intrinsic.isa,
|
|
602
|
+
"lat": lat,
|
|
603
|
+
"cpi": cpi,
|
|
604
|
+
"summary": intrinsic.description,
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _llm_instruction_payload(item) -> dict:
|
|
609
|
+
lat, cpi = variant_perf_summary(item.arch_details)
|
|
610
|
+
return {
|
|
611
|
+
"query": item.key,
|
|
612
|
+
"intrinsic": item.linked_intrinsics,
|
|
613
|
+
"isa": item.isa,
|
|
614
|
+
"lat": lat,
|
|
615
|
+
"cpi": cpi,
|
|
616
|
+
"summary": item.summary,
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
# ---------------------------------------------------------------------------
|
|
621
|
+
# Search results
|
|
622
|
+
# ---------------------------------------------------------------------------
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _print_search_results_runtime(conn, query: str, limit: int = 20) -> None:
|
|
626
|
+
results, intrinsic_map, instruction_map = _search_runtime(conn, query, limit=limit)
|
|
627
|
+
prepared_rows = []
|
|
628
|
+
for result in results:
|
|
629
|
+
arch = "-"
|
|
630
|
+
isa = "-"
|
|
631
|
+
lat = "-"
|
|
632
|
+
cpi = "-"
|
|
633
|
+
isa_sort = (99, "-")
|
|
634
|
+
if result.kind == "instruction":
|
|
635
|
+
item = instruction_map.get(result.key)
|
|
636
|
+
if item is not None:
|
|
637
|
+
if not isa_visible(item.isa, show_fp16=SHOW_FP16_ISAS):
|
|
638
|
+
continue
|
|
639
|
+
arch = display_architecture(item.architecture)
|
|
640
|
+
isa = display_isa(item.isa)
|
|
641
|
+
isa_sort = isa_sort_key(item.isa)
|
|
642
|
+
lat, cpi = variant_perf_summary(item.arch_details)
|
|
643
|
+
elif result.kind == "intrinsic":
|
|
644
|
+
item = intrinsic_map.get(result.key)
|
|
645
|
+
if item is not None:
|
|
646
|
+
if not isa_visible(item.isa, show_fp16=SHOW_FP16_ISAS):
|
|
647
|
+
continue
|
|
648
|
+
arch = display_architecture(item.architecture)
|
|
649
|
+
isa = display_isa(item.isa)
|
|
650
|
+
isa_sort = isa_sort_key(item.isa)
|
|
651
|
+
lat, cpi = intrinsic_perf_summary_runtime(conn, item, instruction_map)
|
|
652
|
+
prepared_rows.append((result, arch, isa, lat, cpi, isa_sort))
|
|
653
|
+
prepared_rows.sort(key=lambda row: (row[0].kind != "instruction", row[5], row[0].title.casefold(), row[0].key.casefold()))
|
|
654
|
+
render_search_results([(r, arch, isa, lat, cpi) for r, arch, isa, lat, cpi, _ in prepared_rows])
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
# ---------------------------------------------------------------------------
|
|
658
|
+
# Smart lookup (bare-word query)
|
|
659
|
+
# ---------------------------------------------------------------------------
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
def _smart_lookup(query: str, preset: str | None = None) -> int:
|
|
663
|
+
"""Open the TUI pre-filled with the given query."""
|
|
664
|
+
ensure_runtime()
|
|
665
|
+
return _run_tui(initial_query=query, initial_preset=preset)
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def _is_completion_invocation(env: dict[str, str] | None = None) -> bool:
|
|
669
|
+
env = env or os.environ
|
|
670
|
+
for key, value in env.items():
|
|
671
|
+
if not key.endswith("_COMPLETE"):
|
|
672
|
+
continue
|
|
673
|
+
upper = key.upper()
|
|
674
|
+
if "SIMDREF" not in upper and upper != "_ISA_COMPLETE":
|
|
675
|
+
continue
|
|
676
|
+
if value:
|
|
677
|
+
return True
|
|
678
|
+
return False
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
# ---------------------------------------------------------------------------
|
|
682
|
+
# Typer commands
|
|
683
|
+
# ---------------------------------------------------------------------------
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
@app.command(rich_help_panel="Commands")
|
|
687
|
+
def annotate(
|
|
688
|
+
input_path: Path | None = typer.Argument(None, help="Input .s assembly file, or '-' for stdin. Omit to open the TUI annotate tab."),
|
|
689
|
+
output: Path = typer.Option(None, "-o", "--output", help="Output .sa path (default: <input>.sa, or '-' for stdout)."),
|
|
690
|
+
performance: bool = typer.Option(True, "--performance/--no-performance", help="Include latency/CPI annotations."),
|
|
691
|
+
docs: bool = typer.Option(True, "--docs/--no-docs", help="Include human-readable instruction summaries."),
|
|
692
|
+
arch: str | None = typer.Option(None, "--arch", help="Pin annotations to a specific microarch (e.g. skylake-x, zen4)."),
|
|
693
|
+
agg: str = typer.Option("avg", "--agg", help="Aggregation across archs when --arch is not set: avg|median|best|worst."),
|
|
694
|
+
include_modeled: bool = typer.Option(False, "--include-modeled", help="Fall back to modeled perf data when no arch has measured data."),
|
|
695
|
+
block: bool = typer.Option(False, "--block/--inline", help="Emit annotation as a comment block above each instruction (default: inline trailing)."),
|
|
696
|
+
unknown: str = typer.Option("mark", "--unknown", help="Handling of unknown mnemonics: keep|drop|mark."),
|
|
697
|
+
fmt: str = typer.Option("sa", "--format", help="Output format: sa|md|json."),
|
|
698
|
+
) -> None:
|
|
699
|
+
"""Annotate a ``.s`` assembly file with instruction summaries and perf data.
|
|
700
|
+
|
|
701
|
+
With no positional argument, launches the TUI on the Annotate tab."""
|
|
702
|
+
from simdref.annotate import AnnotateOptions, annotate_stream
|
|
703
|
+
|
|
704
|
+
if input_path is None:
|
|
705
|
+
ensure_runtime()
|
|
706
|
+
raise typer.Exit(code=_run_tui(initial_view="annotate"))
|
|
707
|
+
|
|
708
|
+
if agg not in {"avg", "median", "best", "worst"}:
|
|
709
|
+
err_console.print(f"invalid --agg value: {agg}", style="red")
|
|
710
|
+
raise typer.Exit(code=1)
|
|
711
|
+
if unknown not in {"keep", "drop", "mark"}:
|
|
712
|
+
err_console.print(f"invalid --unknown value: {unknown}", style="red")
|
|
713
|
+
raise typer.Exit(code=1)
|
|
714
|
+
if fmt not in {"sa", "md", "json"}:
|
|
715
|
+
err_console.print(f"invalid --format value: {fmt}", style="red")
|
|
716
|
+
raise typer.Exit(code=1)
|
|
717
|
+
|
|
718
|
+
ensure_runtime()
|
|
719
|
+
|
|
720
|
+
input_is_stdin = str(input_path) == "-"
|
|
721
|
+
if input_is_stdin:
|
|
722
|
+
source_lines: list[str] = sys.stdin.readlines()
|
|
723
|
+
default_out = Path("-")
|
|
724
|
+
else:
|
|
725
|
+
if not input_path.exists():
|
|
726
|
+
err_console.print(f"input not found: {input_path}", style="red")
|
|
727
|
+
raise typer.Exit(code=1)
|
|
728
|
+
source_lines = input_path.read_text().splitlines(keepends=True)
|
|
729
|
+
default_out = input_path.with_suffix(input_path.suffix + "a") if input_path.suffix == ".s" else input_path.with_suffix(".sa")
|
|
730
|
+
|
|
731
|
+
out_path = output if output is not None else default_out
|
|
732
|
+
opts = AnnotateOptions(
|
|
733
|
+
performance=performance,
|
|
734
|
+
docs=docs,
|
|
735
|
+
arch=arch,
|
|
736
|
+
agg=agg,
|
|
737
|
+
include_modeled=include_modeled,
|
|
738
|
+
block=block,
|
|
739
|
+
unknown=unknown,
|
|
740
|
+
fmt=fmt,
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
with open_db(SQLITE_PATH) as conn:
|
|
744
|
+
rendered = "".join(annotate_stream(source_lines, opts=opts, conn=conn))
|
|
745
|
+
|
|
746
|
+
if str(out_path) == "-":
|
|
747
|
+
sys.stdout.write(rendered)
|
|
748
|
+
else:
|
|
749
|
+
out_path.write_text(rendered)
|
|
750
|
+
err_console.print(f"wrote {out_path}", style="green")
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
@app.command(rich_help_panel="Commands")
|
|
754
|
+
def update(
|
|
755
|
+
from_release: bool = typer.Option(False, "--from-release", help="Download pre-built data from GitHub Release."),
|
|
756
|
+
man_dir: Path = typer.Option(DEFAULT_MAN_DIR, help="Target man root directory."),
|
|
757
|
+
) -> None:
|
|
758
|
+
"""Download the pre-built release catalog (no llvm-mca required)."""
|
|
759
|
+
if from_release:
|
|
760
|
+
_download_from_release()
|
|
761
|
+
_finalize_runtime_from_download(man_dir=man_dir)
|
|
762
|
+
return
|
|
763
|
+
|
|
764
|
+
_download_release_or_fallback(man_dir=man_dir)
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
@app.command(rich_help_panel="Dev commands")
|
|
768
|
+
def build(
|
|
769
|
+
man_dir: Path = typer.Option(DEFAULT_MAN_DIR, help="Target man root directory."),
|
|
770
|
+
) -> None:
|
|
771
|
+
"""Full local rebuild from upstream sources, including Intel SDM parsing (requires llvm-mca on PATH)."""
|
|
772
|
+
_require_llvm_mca_or_hint()
|
|
773
|
+
_build_runtime_locally(man_dir=man_dir, include_sdm=True)
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
def _require_llvm_mca_or_hint() -> None:
|
|
777
|
+
"""Abort with an install hint when ``llvm-mca`` is missing on PATH.
|
|
778
|
+
|
|
779
|
+
``--build`` needs it to generate modeled ARM/RISC-V perf rows.
|
|
780
|
+
Users who only want pre-built data can drop the flag.
|
|
781
|
+
"""
|
|
782
|
+
from simdref.perf_sources.llvm_mca import LLVMMcaUnavailable, detect_llvm_mca_version
|
|
783
|
+
try:
|
|
784
|
+
detect_llvm_mca_version()
|
|
785
|
+
except LLVMMcaUnavailable as exc:
|
|
786
|
+
err_console.print(
|
|
787
|
+
f"[bold red]llvm-mca is required for --build[/bold red]: {exc}",
|
|
788
|
+
)
|
|
789
|
+
err_console.print(LLVMMcaUnavailable.install_hint)
|
|
790
|
+
raise typer.Exit(code=1) from exc
|
|
791
|
+
|
|
792
|
+
|
|
793
|
+
LLM_EXIT_MATCH = 0
|
|
794
|
+
LLM_EXIT_USAGE = 1
|
|
795
|
+
LLM_EXIT_NO_MATCH = 2
|
|
796
|
+
LLM_EXIT_AMBIGUOUS = 3
|
|
797
|
+
LLM_EXIT_INTERNAL = 10
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _resolve_preset_filters(preset: str | None) -> tuple[list[str] | None, list[str] | None]:
|
|
801
|
+
"""Translate a preset name into (isa_families, categories) overrides.
|
|
802
|
+
|
|
803
|
+
Presets supply ISA-family + sub-ISA facets; we map them to the coarse
|
|
804
|
+
ISA-family list the llm filter uses. Categories are not implied by a
|
|
805
|
+
preset (they come from --filter / --category).
|
|
806
|
+
"""
|
|
807
|
+
if not preset:
|
|
808
|
+
return None, None
|
|
809
|
+
from simdref.filters import ARCH_PRESETS
|
|
810
|
+
spec = ARCH_PRESETS.get(preset)
|
|
811
|
+
if spec is None:
|
|
812
|
+
return None, None
|
|
813
|
+
return sorted(spec.families), None
|
|
814
|
+
|
|
815
|
+
|
|
816
|
+
def _llm_filter_records(
|
|
817
|
+
records: list[dict],
|
|
818
|
+
isa: list[str] | None,
|
|
819
|
+
category: list[str] | None,
|
|
820
|
+
source_kind: str | None = None,
|
|
821
|
+
) -> list[dict]:
|
|
822
|
+
"""Filter llm payload dicts by ISA family, category, and source-kind."""
|
|
823
|
+
from simdref.display import isa_family as _isa_family
|
|
824
|
+
source_kind = (source_kind or "").strip().lower()
|
|
825
|
+
if source_kind in ("", "any"):
|
|
826
|
+
source_kind = ""
|
|
827
|
+
if not isa and not category and not source_kind:
|
|
828
|
+
return records
|
|
829
|
+
isa_set = {f.strip() for f in (isa or []) if f and f.strip()}
|
|
830
|
+
cat_set = {c.strip() for c in (category or []) if c and c.strip()}
|
|
831
|
+
kept: list[dict] = []
|
|
832
|
+
for rec in records:
|
|
833
|
+
if isa_set:
|
|
834
|
+
rec_isa = rec.get("isa") or []
|
|
835
|
+
if isinstance(rec_isa, str):
|
|
836
|
+
rec_isa = [rec_isa]
|
|
837
|
+
families = {_isa_family(v) for v in rec_isa}
|
|
838
|
+
if not families & isa_set:
|
|
839
|
+
continue
|
|
840
|
+
if cat_set:
|
|
841
|
+
rec_cat = rec.get("category", "")
|
|
842
|
+
if rec_cat not in cat_set:
|
|
843
|
+
continue
|
|
844
|
+
if source_kind:
|
|
845
|
+
if not _record_has_source_kind(rec, source_kind):
|
|
846
|
+
continue
|
|
847
|
+
kept.append(rec)
|
|
848
|
+
return kept
|
|
849
|
+
|
|
850
|
+
|
|
851
|
+
def _record_has_source_kind(rec: dict, wanted: str) -> bool:
|
|
852
|
+
"""Check whether an llm payload dict carries at least one entry with *wanted* provenance."""
|
|
853
|
+
arch_details = rec.get("arch_details") or {}
|
|
854
|
+
if isinstance(arch_details, dict):
|
|
855
|
+
for details in arch_details.values():
|
|
856
|
+
if isinstance(details, dict):
|
|
857
|
+
kind = details.get("source_kind") or "measured"
|
|
858
|
+
if kind == wanted:
|
|
859
|
+
return True
|
|
860
|
+
for nested_key in ("instruction", "instructions", "results"):
|
|
861
|
+
nested = rec.get(nested_key)
|
|
862
|
+
if isinstance(nested, dict):
|
|
863
|
+
if _record_has_source_kind(nested, wanted):
|
|
864
|
+
return True
|
|
865
|
+
elif isinstance(nested, list):
|
|
866
|
+
if any(_record_has_source_kind(n, wanted) for n in nested if isinstance(n, dict)):
|
|
867
|
+
return True
|
|
868
|
+
return False
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
def _llm_format_markdown(payload: dict) -> str:
|
|
872
|
+
"""Render an llm payload as prompt-friendly markdown."""
|
|
873
|
+
mode = payload.get("mode", "search")
|
|
874
|
+
query = payload.get("query", "")
|
|
875
|
+
lines: list[str] = [f"# simdref: {query}", ""]
|
|
876
|
+
if mode == "exact" and payload.get("match_kind") == "intrinsic":
|
|
877
|
+
rec = payload.get("result", {})
|
|
878
|
+
lines.append(f"**Intrinsic:** `{rec.get('intrinsic', '')}`")
|
|
879
|
+
if rec.get("signature"):
|
|
880
|
+
lines.append(f"**Signature:** `{rec['signature']}`")
|
|
881
|
+
if rec.get("isa"):
|
|
882
|
+
lines.append(f"**ISA:** {', '.join(rec['isa'])}")
|
|
883
|
+
if rec.get("instructions"):
|
|
884
|
+
lines.append(f"**Instruction:** `{rec['instructions'][0]}`")
|
|
885
|
+
if rec.get("lat") and rec["lat"] != "-":
|
|
886
|
+
lines.append(f"**Latency:** {rec['lat']} • **CPI:** {rec.get('cpi', '-')}")
|
|
887
|
+
if rec.get("summary"):
|
|
888
|
+
lines += ["", rec["summary"]]
|
|
889
|
+
return "\n".join(lines)
|
|
890
|
+
items = payload.get("results", [])
|
|
891
|
+
if mode == "exact":
|
|
892
|
+
lines.append(f"**{len(items)} instruction match(es)**")
|
|
893
|
+
else:
|
|
894
|
+
lines.append(f"**{len(items)} search result(s)**")
|
|
895
|
+
lines.append("")
|
|
896
|
+
for r in items:
|
|
897
|
+
title = r.get("intrinsic") or r.get("query") or ""
|
|
898
|
+
if isinstance(title, list):
|
|
899
|
+
title = ", ".join(title)
|
|
900
|
+
summary = r.get("summary", "")
|
|
901
|
+
isa = ", ".join(r.get("isa") or [])
|
|
902
|
+
lines.append(f"- **{title}** `{isa}` — {summary}")
|
|
903
|
+
return "\n".join(lines)
|
|
904
|
+
|
|
905
|
+
|
|
906
|
+
def _emit_llm_payload(payload: dict, fmt: str) -> None:
|
|
907
|
+
if fmt == "ndjson":
|
|
908
|
+
mode = payload.get("mode")
|
|
909
|
+
if mode == "exact" and "result" in payload:
|
|
910
|
+
typer.echo(json.dumps(payload["result"], sort_keys=True))
|
|
911
|
+
return
|
|
912
|
+
for item in payload.get("results") or []:
|
|
913
|
+
typer.echo(json.dumps(item, sort_keys=True))
|
|
914
|
+
return
|
|
915
|
+
if fmt == "markdown":
|
|
916
|
+
typer.echo(_llm_format_markdown(payload))
|
|
917
|
+
return
|
|
918
|
+
typer.echo(json.dumps(payload, sort_keys=True, indent=2))
|
|
919
|
+
|
|
920
|
+
|
|
921
|
+
def _llm_exit_code(payload: dict) -> int:
|
|
922
|
+
mode = payload.get("mode")
|
|
923
|
+
if mode == "exact":
|
|
924
|
+
if "result" in payload:
|
|
925
|
+
return LLM_EXIT_MATCH
|
|
926
|
+
results = payload.get("results") or []
|
|
927
|
+
if len(results) > 1 and payload.get("match_kind") == "instruction":
|
|
928
|
+
exact_name_hits = sum(1 for r in results if r.get("query", "").casefold() == payload.get("query", "").casefold())
|
|
929
|
+
if exact_name_hits > 1:
|
|
930
|
+
return LLM_EXIT_AMBIGUOUS
|
|
931
|
+
return LLM_EXIT_MATCH if results else LLM_EXIT_NO_MATCH
|
|
932
|
+
return LLM_EXIT_MATCH if payload.get("results") else LLM_EXIT_NO_MATCH
|
|
933
|
+
|
|
934
|
+
|
|
935
|
+
def _llm_schema_payload() -> dict:
|
|
936
|
+
"""Approximate JSON Schema for llm payloads (stable for tool consumers)."""
|
|
937
|
+
return {
|
|
938
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
939
|
+
"title": "simdref.llm",
|
|
940
|
+
"type": "object",
|
|
941
|
+
"properties": {
|
|
942
|
+
"query": {"type": "string"},
|
|
943
|
+
"mode": {"type": "string", "enum": ["exact", "search"]},
|
|
944
|
+
"match_kind": {"type": ["string", "null"], "enum": ["intrinsic", "instruction", None]},
|
|
945
|
+
"generated_at": {
|
|
946
|
+
"type": "string",
|
|
947
|
+
"description": "ISO-8601 timestamp of the catalog build the answer was derived from.",
|
|
948
|
+
},
|
|
949
|
+
"source_versions": {
|
|
950
|
+
"type": "array",
|
|
951
|
+
"description": "Upstream source descriptors (name, version, url) pinned by this catalog.",
|
|
952
|
+
"items": {
|
|
953
|
+
"type": "object",
|
|
954
|
+
"properties": {
|
|
955
|
+
"source": {"type": "string"},
|
|
956
|
+
"version": {"type": "string"},
|
|
957
|
+
"url": {"type": "string"},
|
|
958
|
+
},
|
|
959
|
+
},
|
|
960
|
+
},
|
|
961
|
+
"result": {
|
|
962
|
+
"type": "object",
|
|
963
|
+
"properties": {
|
|
964
|
+
"query": {"type": "string"},
|
|
965
|
+
"intrinsic": {"type": ["string", "array"]},
|
|
966
|
+
"signature": {"type": "string"},
|
|
967
|
+
"instructions": {"type": "array", "items": {"type": "string"}},
|
|
968
|
+
"instruction_refs": {
|
|
969
|
+
"type": "array",
|
|
970
|
+
"description": "Resolved instruction references when known.",
|
|
971
|
+
"items": {
|
|
972
|
+
"type": "object",
|
|
973
|
+
"properties": {
|
|
974
|
+
"key": {"type": "string"},
|
|
975
|
+
"name": {"type": "string"},
|
|
976
|
+
"form": {"type": "string"},
|
|
977
|
+
"architecture": {"type": "string"},
|
|
978
|
+
"xed": {"type": "string"},
|
|
979
|
+
"resolution": {"type": "string"},
|
|
980
|
+
"match_count": {"type": "integer"},
|
|
981
|
+
},
|
|
982
|
+
},
|
|
983
|
+
},
|
|
984
|
+
"isa": {"type": "array", "items": {"type": "string"}},
|
|
985
|
+
"lat": {"type": "string"},
|
|
986
|
+
"cpi": {"type": "string"},
|
|
987
|
+
"summary": {"type": "string"},
|
|
988
|
+
},
|
|
989
|
+
},
|
|
990
|
+
"results": {"type": "array", "items": {"$ref": "#/properties/result"}},
|
|
991
|
+
},
|
|
992
|
+
"required": ["query", "mode"],
|
|
993
|
+
}
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
llm_app = typer.Typer(help="Structured output for LLM/tool consumption.", invoke_without_command=False)
|
|
997
|
+
_LLM_HELP_PANEL = "Commands"
|
|
998
|
+
|
|
999
|
+
|
|
1000
|
+
def _build_llm_payload(
|
|
1001
|
+
conn,
|
|
1002
|
+
query_str: str,
|
|
1003
|
+
limit: int,
|
|
1004
|
+
isa: list[str] | None,
|
|
1005
|
+
category: list[str] | None,
|
|
1006
|
+
source_kind: str | None,
|
|
1007
|
+
) -> dict:
|
|
1008
|
+
"""Build the llm payload for *query_str* against an open DB connection.
|
|
1009
|
+
|
|
1010
|
+
Kept free of I/O and exit logic so that ``simdref llm batch`` can call it
|
|
1011
|
+
in a loop without re-opening the catalog per query.
|
|
1012
|
+
"""
|
|
1013
|
+
intrinsic = load_intrinsic_from_db(conn, query_str)
|
|
1014
|
+
if intrinsic is not None:
|
|
1015
|
+
result = _llm_intrinsic_payload(conn, intrinsic)
|
|
1016
|
+
kept = _llm_filter_records([result], isa, category, source_kind=source_kind)
|
|
1017
|
+
return {
|
|
1018
|
+
"query": query_str, "mode": "exact",
|
|
1019
|
+
"match_kind": "intrinsic" if kept else None,
|
|
1020
|
+
**({"result": kept[0]} if kept else {"results": []}),
|
|
1021
|
+
}
|
|
1022
|
+
instructions = _find_instructions_fast(query_str)
|
|
1023
|
+
if instructions:
|
|
1024
|
+
items = [_llm_instruction_payload(item) for item in instructions]
|
|
1025
|
+
items = _llm_filter_records(items, isa, category, source_kind=source_kind)
|
|
1026
|
+
return {
|
|
1027
|
+
"query": query_str, "mode": "exact",
|
|
1028
|
+
"match_kind": "instruction",
|
|
1029
|
+
"results": items,
|
|
1030
|
+
}
|
|
1031
|
+
results, intrinsic_map, instruction_map = _search_runtime(conn, query_str, limit=limit)
|
|
1032
|
+
items = [_llm_result_payload(conn, r, intrinsic_map, instruction_map) for r in results]
|
|
1033
|
+
items = _llm_filter_records(items, isa, category, source_kind=source_kind)
|
|
1034
|
+
return {
|
|
1035
|
+
"query": query_str, "mode": "search",
|
|
1036
|
+
"match_kind": None,
|
|
1037
|
+
"results": items,
|
|
1038
|
+
}
|
|
1039
|
+
|
|
1040
|
+
|
|
1041
|
+
def _normalize_fmt(fmt: str, allowed: set[str]) -> str:
|
|
1042
|
+
fmt_lower = (fmt or "json").lower()
|
|
1043
|
+
if fmt_lower not in allowed:
|
|
1044
|
+
typer.echo(
|
|
1045
|
+
f"error: unknown --format '{fmt}' (expected {'|'.join(sorted(allowed))})",
|
|
1046
|
+
err=True,
|
|
1047
|
+
)
|
|
1048
|
+
raise typer.Exit(code=LLM_EXIT_USAGE)
|
|
1049
|
+
return fmt_lower
|
|
1050
|
+
|
|
1051
|
+
|
|
1052
|
+
def _resolve_preset_or_exit(preset: str | None, isa: list[str] | None) -> list[str] | None:
|
|
1053
|
+
if not preset:
|
|
1054
|
+
return isa
|
|
1055
|
+
from simdref.filters import ARCH_PRESETS
|
|
1056
|
+
if preset not in ARCH_PRESETS:
|
|
1057
|
+
known = ", ".join(sorted(ARCH_PRESETS))
|
|
1058
|
+
typer.echo(f"error: unknown --preset '{preset}' (known: {known})", err=True)
|
|
1059
|
+
raise typer.Exit(code=LLM_EXIT_USAGE)
|
|
1060
|
+
preset_isa, _ = _resolve_preset_filters(preset)
|
|
1061
|
+
if preset_isa and not isa:
|
|
1062
|
+
return preset_isa
|
|
1063
|
+
return isa
|
|
1064
|
+
|
|
1065
|
+
|
|
1066
|
+
def _llm_query_impl(
|
|
1067
|
+
query_tokens: list[str],
|
|
1068
|
+
limit: int,
|
|
1069
|
+
fmt: str,
|
|
1070
|
+
isa: list[str] | None,
|
|
1071
|
+
category: list[str] | None,
|
|
1072
|
+
preset: str | None = None,
|
|
1073
|
+
source_kind: str | None = None,
|
|
1074
|
+
) -> None:
|
|
1075
|
+
fmt_lower = _normalize_fmt(fmt, {"json", "ndjson", "markdown"})
|
|
1076
|
+
isa = _resolve_preset_or_exit(preset, isa)
|
|
1077
|
+
if not query_tokens:
|
|
1078
|
+
typer.echo("error: query required (or use `simdref llm list` / `simdref llm schema`)", err=True)
|
|
1079
|
+
raise typer.Exit(code=LLM_EXIT_USAGE)
|
|
1080
|
+
query_str = " ".join(query_tokens)
|
|
1081
|
+
ensure_runtime()
|
|
1082
|
+
try:
|
|
1083
|
+
with open_db() as conn:
|
|
1084
|
+
payload = _build_llm_payload(conn, query_str, limit, isa, category, source_kind)
|
|
1085
|
+
except typer.Exit:
|
|
1086
|
+
raise
|
|
1087
|
+
except Exception as exc: # pragma: no cover - internal error path
|
|
1088
|
+
typer.echo(f"internal error: {exc}", err=True)
|
|
1089
|
+
raise typer.Exit(code=LLM_EXIT_INTERNAL)
|
|
1090
|
+
_emit_llm_payload(payload, fmt_lower)
|
|
1091
|
+
raise typer.Exit(code=_llm_exit_code(payload))
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
@llm_app.command("query")
|
|
1095
|
+
def llm_query(
|
|
1096
|
+
query: list[str] = typer.Argument(..., help="Search query (multiple tokens allowed)."),
|
|
1097
|
+
limit: int = typer.Option(8, help="Maximum number of search results in search mode."),
|
|
1098
|
+
fmt: str = typer.Option("json", "--format", "-F", help="Output format: json, ndjson, or markdown."),
|
|
1099
|
+
isa: list[str] = typer.Option(None, "--isa", help="Filter by ISA family (repeatable)."),
|
|
1100
|
+
preset: str = typer.Option(None, "--preset", help="Apply a named preset (default, intel, arm32, arm64, riscv, none, all)."),
|
|
1101
|
+
source_kind: str = typer.Option("any", "--source-kind", help="Filter perf rows by provenance: measured, modeled, or any."),
|
|
1102
|
+
) -> None:
|
|
1103
|
+
"""Resolve a query and emit an LLM-friendly payload.
|
|
1104
|
+
|
|
1105
|
+
Exit codes: 0 match, 2 no-match, 3 ambiguous, 1 usage error, 10 internal.
|
|
1106
|
+
"""
|
|
1107
|
+
_llm_query_impl(query, limit, fmt, isa, None, preset=preset, source_kind=source_kind)
|
|
1108
|
+
|
|
1109
|
+
|
|
1110
|
+
def _emit_filtered_names(
|
|
1111
|
+
conn,
|
|
1112
|
+
pattern: str,
|
|
1113
|
+
isa: list[str] | None,
|
|
1114
|
+
) -> int:
|
|
1115
|
+
"""Stream NDJSON records matching *pattern* filtered by ISA family.
|
|
1116
|
+
|
|
1117
|
+
Iterates the SQLite catalog directly so the caller avoids loading the
|
|
1118
|
+
full msgpack payload for every candidate. Returns number of records
|
|
1119
|
+
emitted (caller uses this to decide the exit code).
|
|
1120
|
+
"""
|
|
1121
|
+
from simdref.display import isa_family as _isa_family
|
|
1122
|
+
glob_pat = pattern
|
|
1123
|
+
isa_set = {f.strip() for f in (isa or []) if f and f.strip()}
|
|
1124
|
+
emitted = 0
|
|
1125
|
+
|
|
1126
|
+
intrinsic_rows = conn.execute(
|
|
1127
|
+
"SELECT name, isa, category FROM intrinsics_data ORDER BY name"
|
|
1128
|
+
).fetchall()
|
|
1129
|
+
for row in intrinsic_rows:
|
|
1130
|
+
name = row["name"]
|
|
1131
|
+
if not fnmatch.fnmatchcase(name, glob_pat) and not fnmatch.fnmatch(name.casefold(), glob_pat.casefold()):
|
|
1132
|
+
continue
|
|
1133
|
+
isas = [s.strip() for s in (row["isa"] or "").split(",") if s.strip()]
|
|
1134
|
+
if isa_set:
|
|
1135
|
+
families = {_isa_family(s) for s in isas}
|
|
1136
|
+
if not families & isa_set:
|
|
1137
|
+
continue
|
|
1138
|
+
typer.echo(json.dumps(
|
|
1139
|
+
{"name": name, "kind": "intrinsic", "isa": isas, "category": row["category"] or ""},
|
|
1140
|
+
sort_keys=True,
|
|
1141
|
+
))
|
|
1142
|
+
emitted += 1
|
|
1143
|
+
|
|
1144
|
+
instruction_rows = conn.execute(
|
|
1145
|
+
"SELECT key, db_key, isa, category FROM instructions_data ORDER BY key"
|
|
1146
|
+
).fetchall()
|
|
1147
|
+
for row in instruction_rows:
|
|
1148
|
+
key = row["key"]
|
|
1149
|
+
db_key = row["db_key"]
|
|
1150
|
+
if (
|
|
1151
|
+
not fnmatch.fnmatchcase(key, glob_pat)
|
|
1152
|
+
and not fnmatch.fnmatch(key.casefold(), glob_pat.casefold())
|
|
1153
|
+
and not fnmatch.fnmatchcase(db_key, glob_pat)
|
|
1154
|
+
):
|
|
1155
|
+
continue
|
|
1156
|
+
isas = [s.strip() for s in (row["isa"] or "").split(",") if s.strip()]
|
|
1157
|
+
if isa_set:
|
|
1158
|
+
families = {_isa_family(s) for s in isas}
|
|
1159
|
+
if not families & isa_set:
|
|
1160
|
+
continue
|
|
1161
|
+
typer.echo(json.dumps(
|
|
1162
|
+
{"name": key, "kind": "instruction", "isa": isas, "category": row["category"] or ""},
|
|
1163
|
+
sort_keys=True,
|
|
1164
|
+
))
|
|
1165
|
+
emitted += 1
|
|
1166
|
+
return emitted
|
|
1167
|
+
|
|
1168
|
+
|
|
1169
|
+
@llm_app.command("list")
|
|
1170
|
+
def llm_list(
|
|
1171
|
+
fmt: str = typer.Option("json", "--format", "-F", help="Output format: json or markdown (ignored when --pattern is given)."),
|
|
1172
|
+
pattern: str = typer.Option(None, "--pattern", help="Glob filter over intrinsic/instruction names. When set, the command emits NDJSON {name, kind, isa, category} records instead of the FilterSpec."),
|
|
1173
|
+
isa: list[str] = typer.Option(None, "--isa", help="Restrict --pattern output to the given ISA family (repeatable)."),
|
|
1174
|
+
) -> None:
|
|
1175
|
+
"""Emit the FilterSpec or stream matching catalog entries.
|
|
1176
|
+
|
|
1177
|
+
Without ``--pattern`` this emits the full :class:`FilterSpec` describing
|
|
1178
|
+
ISA families, sub-ISAs, and categories. With ``--pattern GLOB`` it emits
|
|
1179
|
+
NDJSON records for each matching intrinsic/instruction — useful for a
|
|
1180
|
+
Claude skill that wants "all AVX-512 *gather* intrinsics" without
|
|
1181
|
+
calling ``query`` per name.
|
|
1182
|
+
"""
|
|
1183
|
+
ensure_runtime()
|
|
1184
|
+
if pattern:
|
|
1185
|
+
with open_db() as conn:
|
|
1186
|
+
emitted = _emit_filtered_names(conn, pattern, isa)
|
|
1187
|
+
raise typer.Exit(code=LLM_EXIT_MATCH if emitted else LLM_EXIT_NO_MATCH)
|
|
1188
|
+
|
|
1189
|
+
from simdref.filters import build_filter_spec
|
|
1190
|
+
with open_db() as conn:
|
|
1191
|
+
spec = build_filter_spec(conn)
|
|
1192
|
+
payload = spec.to_json()
|
|
1193
|
+
if (fmt or "json").lower() == "markdown":
|
|
1194
|
+
lines = ["# simdref filter spec", "", "## ISA families"]
|
|
1195
|
+
for fam in payload["default_enabled"]:
|
|
1196
|
+
lines.append(f"- **{fam}** (default)")
|
|
1197
|
+
for fam in payload["family_order"]:
|
|
1198
|
+
if fam not in payload["default_enabled"]:
|
|
1199
|
+
lines.append(f"- {fam}")
|
|
1200
|
+
lines += ["", "## Categories"]
|
|
1201
|
+
for cat in payload["categories"]:
|
|
1202
|
+
lines.append(f"- {cat['family']} / {cat['category']} ({cat['count']})")
|
|
1203
|
+
typer.echo("\n".join(lines))
|
|
1204
|
+
return
|
|
1205
|
+
typer.echo(json.dumps(payload, sort_keys=True, indent=2))
|
|
1206
|
+
|
|
1207
|
+
|
|
1208
|
+
@llm_app.command("batch")
|
|
1209
|
+
def llm_batch(
|
|
1210
|
+
limit: int = typer.Option(8, help="Maximum number of search results per query in search mode."),
|
|
1211
|
+
isa: list[str] = typer.Option(None, "--isa", help="Filter results by ISA family (repeatable)."),
|
|
1212
|
+
preset: str = typer.Option(None, "--preset", help="Apply a named preset (default, intel, arm32, arm64, riscv, none, all)."),
|
|
1213
|
+
source_kind: str = typer.Option("any", "--source-kind", help="Filter perf rows by provenance: measured, modeled, or any."),
|
|
1214
|
+
) -> None:
|
|
1215
|
+
"""Resolve queries from stdin (one per line); emit NDJSON records.
|
|
1216
|
+
|
|
1217
|
+
Each output line is ``{"query": ..., "status": "match|no_match|ambiguous|error",
|
|
1218
|
+
"payload": {...}}``. Amortizes catalog load across hundreds of lookups — useful
|
|
1219
|
+
when a Claude skill resolves every mnemonic in a disassembly.
|
|
1220
|
+
"""
|
|
1221
|
+
isa = _resolve_preset_or_exit(preset, isa)
|
|
1222
|
+
ensure_runtime()
|
|
1223
|
+
with open_db() as conn:
|
|
1224
|
+
for raw_line in sys.stdin:
|
|
1225
|
+
query = raw_line.strip()
|
|
1226
|
+
if not query or query.startswith("#"):
|
|
1227
|
+
continue
|
|
1228
|
+
try:
|
|
1229
|
+
payload = _build_llm_payload(conn, query, limit, isa, None, source_kind)
|
|
1230
|
+
exit_code = _llm_exit_code(payload)
|
|
1231
|
+
if exit_code == LLM_EXIT_MATCH:
|
|
1232
|
+
status = "match"
|
|
1233
|
+
elif exit_code == LLM_EXIT_AMBIGUOUS:
|
|
1234
|
+
status = "ambiguous"
|
|
1235
|
+
else:
|
|
1236
|
+
status = "no_match"
|
|
1237
|
+
typer.echo(json.dumps(
|
|
1238
|
+
{"query": query, "status": status, "payload": payload},
|
|
1239
|
+
sort_keys=True,
|
|
1240
|
+
))
|
|
1241
|
+
except Exception as exc: # pragma: no cover - defensive
|
|
1242
|
+
typer.echo(json.dumps(
|
|
1243
|
+
{"query": query, "status": "error", "error": str(exc)},
|
|
1244
|
+
sort_keys=True,
|
|
1245
|
+
))
|
|
1246
|
+
|
|
1247
|
+
|
|
1248
|
+
@llm_app.command("schema")
|
|
1249
|
+
def llm_schema() -> None:
|
|
1250
|
+
"""Emit the JSON Schema for `simdref llm` payloads."""
|
|
1251
|
+
typer.echo(json.dumps(_llm_schema_payload(), sort_keys=True, indent=2))
|
|
1252
|
+
|
|
1253
|
+
|
|
1254
|
+
app.add_typer(llm_app, name="llm", rich_help_panel=_LLM_HELP_PANEL)
|
|
1255
|
+
|
|
1256
|
+
|
|
1257
|
+
# ---------------------------------------------------------------------------
|
|
1258
|
+
# Shell completion (opt-in subcommand; replaces Typer's default
|
|
1259
|
+
# --install-completion / --show-completion options)
|
|
1260
|
+
# ---------------------------------------------------------------------------
|
|
1261
|
+
|
|
1262
|
+
|
|
1263
|
+
completion_app = typer.Typer(help="Shell completion helpers.", no_args_is_help=True)
|
|
1264
|
+
|
|
1265
|
+
_COMPLETION_SHELLS = ("bash", "zsh", "fish", "powershell", "pwsh")
|
|
1266
|
+
|
|
1267
|
+
|
|
1268
|
+
def _resolve_completion_shell(shell: str | None) -> str:
|
|
1269
|
+
if shell:
|
|
1270
|
+
shell = shell.strip().lower()
|
|
1271
|
+
else:
|
|
1272
|
+
shell_env = os.environ.get("SHELL", "")
|
|
1273
|
+
shell = Path(shell_env).name.lower() if shell_env else ""
|
|
1274
|
+
if shell not in _COMPLETION_SHELLS:
|
|
1275
|
+
err_console.print(
|
|
1276
|
+
f"error: unsupported or undetected shell '{shell}'; pass one of {', '.join(_COMPLETION_SHELLS)}",
|
|
1277
|
+
style="red",
|
|
1278
|
+
)
|
|
1279
|
+
raise typer.Exit(code=1)
|
|
1280
|
+
return shell
|
|
1281
|
+
|
|
1282
|
+
|
|
1283
|
+
def _completion_prog_name() -> str:
|
|
1284
|
+
prog = Path(sys.argv[0]).name if sys.argv and sys.argv[0] else "simdref"
|
|
1285
|
+
# Strip a stray ``__main__.py`` when invoked via ``python -m simdref``.
|
|
1286
|
+
if prog in {"", "__main__.py"}:
|
|
1287
|
+
prog = "simdref"
|
|
1288
|
+
return prog
|
|
1289
|
+
|
|
1290
|
+
|
|
1291
|
+
@completion_app.command("show")
|
|
1292
|
+
def completion_show(
|
|
1293
|
+
shell: str = typer.Argument(None, help="Shell: bash, zsh, fish, or powershell. Detected from $SHELL when omitted."),
|
|
1294
|
+
) -> None:
|
|
1295
|
+
"""Print a shell completion script to stdout."""
|
|
1296
|
+
shell = _resolve_completion_shell(shell)
|
|
1297
|
+
from typer._completion_shared import get_completion_script
|
|
1298
|
+
prog_name = _completion_prog_name()
|
|
1299
|
+
complete_var = f"_{prog_name.upper().replace('-', '_')}_COMPLETE"
|
|
1300
|
+
typer.echo(get_completion_script(prog_name=prog_name, complete_var=complete_var, shell=shell))
|
|
1301
|
+
|
|
1302
|
+
|
|
1303
|
+
@completion_app.command("install")
|
|
1304
|
+
def completion_install(
|
|
1305
|
+
shell: str = typer.Argument(None, help="Shell: bash, zsh, fish, or powershell. Detected from $SHELL when omitted."),
|
|
1306
|
+
) -> None:
|
|
1307
|
+
"""Install shell completion into the user's shell profile."""
|
|
1308
|
+
shell = _resolve_completion_shell(shell)
|
|
1309
|
+
from typer._completion_shared import install as _install_completion
|
|
1310
|
+
prog_name = _completion_prog_name()
|
|
1311
|
+
complete_var = f"_{prog_name.upper().replace('-', '_')}_COMPLETE"
|
|
1312
|
+
try:
|
|
1313
|
+
shell_detected, path = _install_completion(shell=shell, prog_name=prog_name, complete_var=complete_var)
|
|
1314
|
+
except Exception as exc:
|
|
1315
|
+
err_console.print(f"error: completion install failed: {exc}", style="red")
|
|
1316
|
+
raise typer.Exit(code=1) from exc
|
|
1317
|
+
err_console.print(f"installed {shell_detected} completion for {prog_name} at {path}", style="green")
|
|
1318
|
+
|
|
1319
|
+
|
|
1320
|
+
app.add_typer(completion_app, name="completion", rich_help_panel="Dev commands")
|
|
1321
|
+
|
|
1322
|
+
|
|
1323
|
+
def _registered_command_names() -> set[str]:
|
|
1324
|
+
"""Return the set of Typer commands + subcommand groups the dispatcher knows about.
|
|
1325
|
+
|
|
1326
|
+
Kept as introspection so the bare-word dispatcher in ``main()`` never drifts
|
|
1327
|
+
from the real command surface.
|
|
1328
|
+
"""
|
|
1329
|
+
names: set[str] = set()
|
|
1330
|
+
for info in getattr(app, "registered_commands", []):
|
|
1331
|
+
if info.name:
|
|
1332
|
+
names.add(info.name)
|
|
1333
|
+
elif info.callback is not None:
|
|
1334
|
+
names.add(info.callback.__name__.replace("_", "-"))
|
|
1335
|
+
for info in getattr(app, "registered_groups", []):
|
|
1336
|
+
if info.name:
|
|
1337
|
+
names.add(info.name)
|
|
1338
|
+
names.update({"--help", "-h"})
|
|
1339
|
+
return names
|
|
1340
|
+
|
|
1341
|
+
|
|
1342
|
+
@app.command(rich_help_panel="Commands")
|
|
1343
|
+
def doctor() -> None:
|
|
1344
|
+
"""Check the installation and report pass/fail for each component.
|
|
1345
|
+
|
|
1346
|
+
Exits with a non-zero status when any required check fails so this
|
|
1347
|
+
command is usable from scripts and CI.
|
|
1348
|
+
"""
|
|
1349
|
+
from rich.table import Table
|
|
1350
|
+
|
|
1351
|
+
ok_icon = "[green]✓[/]"
|
|
1352
|
+
fail_icon = "[red]✗[/]"
|
|
1353
|
+
warn_icon = "[yellow]![/]"
|
|
1354
|
+
failures = 0
|
|
1355
|
+
warnings = 0
|
|
1356
|
+
|
|
1357
|
+
table = Table(show_header=True, header_style="bold", box=None, pad_edge=False)
|
|
1358
|
+
table.add_column("", width=2)
|
|
1359
|
+
table.add_column("check", style="cyan", no_wrap=True)
|
|
1360
|
+
table.add_column("status")
|
|
1361
|
+
table.add_column("detail", style="dim")
|
|
1362
|
+
|
|
1363
|
+
# Catalog file
|
|
1364
|
+
if CATALOG_PATH.exists():
|
|
1365
|
+
try:
|
|
1366
|
+
catalog = load_catalog()
|
|
1367
|
+
except Exception as exc:
|
|
1368
|
+
table.add_row(fail_icon, "catalog", "[red]unreadable[/]", f"{CATALOG_PATH}: {exc}")
|
|
1369
|
+
failures += 1
|
|
1370
|
+
console.print(table)
|
|
1371
|
+
console.print(f"\n[red]{failures} check failed — run[/] [cyan]simdref ingest[/] [red]to rebuild.[/]")
|
|
1372
|
+
raise typer.Exit(1)
|
|
1373
|
+
table.add_row(ok_icon, "catalog", "[green]present[/]", str(CATALOG_PATH))
|
|
1374
|
+
else:
|
|
1375
|
+
table.add_row(fail_icon, "catalog", "[red]missing[/]", f"{CATALOG_PATH} — run `simdref ingest`")
|
|
1376
|
+
failures += 1
|
|
1377
|
+
console.print(table)
|
|
1378
|
+
console.print(f"\n[red]{failures} check failed.[/]")
|
|
1379
|
+
raise typer.Exit(1)
|
|
1380
|
+
|
|
1381
|
+
# SQLite index
|
|
1382
|
+
if not SQLITE_PATH.exists():
|
|
1383
|
+
table.add_row(fail_icon, "sqlite index", "[red]missing[/]", f"{SQLITE_PATH} — run `simdref ingest`")
|
|
1384
|
+
failures += 1
|
|
1385
|
+
elif not sqlite_schema_is_current():
|
|
1386
|
+
table.add_row(warn_icon, "sqlite index", "[yellow]outdated schema[/]", "rebuild with `simdref ingest`")
|
|
1387
|
+
warnings += 1
|
|
1388
|
+
else:
|
|
1389
|
+
table.add_row(ok_icon, "sqlite index", "[green]current[/]", str(SQLITE_PATH))
|
|
1390
|
+
|
|
1391
|
+
# Catalog counts
|
|
1392
|
+
n_intr = len(catalog.intrinsics)
|
|
1393
|
+
n_instr = len(catalog.instructions)
|
|
1394
|
+
if n_intr > 0 and n_instr > 0:
|
|
1395
|
+
table.add_row(ok_icon, "catalog data", "[green]populated[/]", f"{n_intr:,} intrinsics · {n_instr:,} instructions")
|
|
1396
|
+
else:
|
|
1397
|
+
table.add_row(fail_icon, "catalog data", "[red]empty[/]", f"{n_intr} intrinsics · {n_instr} instructions")
|
|
1398
|
+
failures += 1
|
|
1399
|
+
|
|
1400
|
+
# Sources
|
|
1401
|
+
if catalog.sources:
|
|
1402
|
+
table.add_row(ok_icon, "sources", "[green]recorded[/]", f"{len(catalog.sources)} source(s)")
|
|
1403
|
+
for source in catalog.sources:
|
|
1404
|
+
table.add_row("", f" {source.source}", "", f"version={source.version}")
|
|
1405
|
+
else:
|
|
1406
|
+
table.add_row(warn_icon, "sources", "[yellow]none recorded[/]", "catalog has no provenance entries")
|
|
1407
|
+
warnings += 1
|
|
1408
|
+
|
|
1409
|
+
# FTS smoke test
|
|
1410
|
+
if SQLITE_PATH.exists() and sqlite_schema_is_current():
|
|
1411
|
+
try:
|
|
1412
|
+
from simdref.storage import open_db
|
|
1413
|
+
with open_db() as conn:
|
|
1414
|
+
row = conn.execute(
|
|
1415
|
+
"SELECT count(*) AS c FROM intrinsics_fts WHERE intrinsics_fts MATCH ?",
|
|
1416
|
+
("add",),
|
|
1417
|
+
).fetchone()
|
|
1418
|
+
hits = row["c"] if row else 0
|
|
1419
|
+
if hits > 0:
|
|
1420
|
+
table.add_row(ok_icon, "fts search", "[green]working[/]", f"query 'add' -> {hits} hits")
|
|
1421
|
+
else:
|
|
1422
|
+
table.add_row(warn_icon, "fts search", "[yellow]no hits[/]", "query 'add' returned 0 hits")
|
|
1423
|
+
warnings += 1
|
|
1424
|
+
except Exception as exc:
|
|
1425
|
+
table.add_row(fail_icon, "fts search", "[red]error[/]", str(exc))
|
|
1426
|
+
failures += 1
|
|
1427
|
+
|
|
1428
|
+
# Man page directory (informational — missing is fine)
|
|
1429
|
+
man_present = DEFAULT_MAN_DIR.exists() and any(DEFAULT_MAN_DIR.rglob("*"))
|
|
1430
|
+
if man_present:
|
|
1431
|
+
table.add_row(ok_icon, "man pages", "[green]present[/]", str(DEFAULT_MAN_DIR))
|
|
1432
|
+
else:
|
|
1433
|
+
table.add_row(warn_icon, "man pages", "[dim]not installed[/]", f"{DEFAULT_MAN_DIR} — optional; install with `simdref install-manpages`")
|
|
1434
|
+
|
|
1435
|
+
console.print(table)
|
|
1436
|
+
|
|
1437
|
+
if failures:
|
|
1438
|
+
console.print(f"\n[red]{failures} failed[/], [yellow]{warnings} warnings[/] — simdref is not ready.")
|
|
1439
|
+
raise typer.Exit(1)
|
|
1440
|
+
if warnings:
|
|
1441
|
+
console.print(f"\n[yellow]OK with {warnings} warning(s)[/] — simdref will run but consider the notes above.")
|
|
1442
|
+
return
|
|
1443
|
+
console.print("\n[bold green]All checks passed.[/] simdref is ready.")
|
|
1444
|
+
|
|
1445
|
+
|
|
1446
|
+
def _export_web_impl(web_dir: Path) -> None:
|
|
1447
|
+
catalog = ensure_catalog()
|
|
1448
|
+
export_web(catalog, web_dir)
|
|
1449
|
+
console.print(f"exported static web app to {web_dir}", style="green")
|
|
1450
|
+
|
|
1451
|
+
|
|
1452
|
+
@app.command("web", rich_help_panel="Dev commands")
|
|
1453
|
+
def web_command(web_dir: Path = typer.Option(WEB_DIR, help="Output directory for static assets.")) -> None:
|
|
1454
|
+
"""Export static web app."""
|
|
1455
|
+
_export_web_impl(web_dir)
|
|
1456
|
+
|
|
1457
|
+
|
|
1458
|
+
@app.command("serve", rich_help_panel="Dev commands")
|
|
1459
|
+
def serve_command(
|
|
1460
|
+
web_dir: Path = typer.Option(WEB_DIR, help="Directory to serve (usually the export dir)."),
|
|
1461
|
+
host: str = typer.Option("127.0.0.1"),
|
|
1462
|
+
port: int = typer.Option(8765),
|
|
1463
|
+
preset: str = typer.Option(None, "--preset", help="Open URL with ?preset=NAME so the web UI applies it on load."),
|
|
1464
|
+
) -> None:
|
|
1465
|
+
"""Serve the exported web app with gzip support.
|
|
1466
|
+
|
|
1467
|
+
Prefers pre-compressed ``*.json.gz`` sidecars written by ``simdref web``
|
|
1468
|
+
when the client sends ``Accept-Encoding: gzip``; falls back to plain files.
|
|
1469
|
+
"""
|
|
1470
|
+
import http.server
|
|
1471
|
+
import os
|
|
1472
|
+
import socketserver
|
|
1473
|
+
|
|
1474
|
+
web_dir = Path(web_dir).resolve()
|
|
1475
|
+
if not web_dir.is_dir():
|
|
1476
|
+
console.print(f"[red]directory not found: {web_dir}[/red]")
|
|
1477
|
+
raise typer.Exit(1)
|
|
1478
|
+
|
|
1479
|
+
class Handler(http.server.SimpleHTTPRequestHandler):
|
|
1480
|
+
def __init__(self, *args, **kwargs):
|
|
1481
|
+
super().__init__(*args, directory=str(web_dir), **kwargs)
|
|
1482
|
+
|
|
1483
|
+
def do_GET(self) -> None: # noqa: N802
|
|
1484
|
+
accepts_gzip = "gzip" in (self.headers.get("Accept-Encoding") or "")
|
|
1485
|
+
url_path = self.path.split("?", 1)[0].split("#", 1)[0]
|
|
1486
|
+
rel = url_path.lstrip("/")
|
|
1487
|
+
target = (web_dir / rel).resolve()
|
|
1488
|
+
# Containment check.
|
|
1489
|
+
try:
|
|
1490
|
+
target.relative_to(web_dir)
|
|
1491
|
+
except ValueError:
|
|
1492
|
+
self.send_error(403)
|
|
1493
|
+
return
|
|
1494
|
+
if target.is_dir():
|
|
1495
|
+
target = target / "index.html"
|
|
1496
|
+
gz_candidate = Path(str(target) + ".gz")
|
|
1497
|
+
if accepts_gzip and target.suffix == ".json" and gz_candidate.is_file():
|
|
1498
|
+
try:
|
|
1499
|
+
data = gz_candidate.read_bytes()
|
|
1500
|
+
except OSError:
|
|
1501
|
+
super().do_GET()
|
|
1502
|
+
return
|
|
1503
|
+
self.send_response(200)
|
|
1504
|
+
self.send_header("Content-Type", "application/json")
|
|
1505
|
+
self.send_header("Content-Encoding", "gzip")
|
|
1506
|
+
self.send_header("Content-Length", str(len(data)))
|
|
1507
|
+
self.send_header("Cache-Control", "public, max-age=60")
|
|
1508
|
+
self.end_headers()
|
|
1509
|
+
self.wfile.write(data)
|
|
1510
|
+
return
|
|
1511
|
+
super().do_GET()
|
|
1512
|
+
|
|
1513
|
+
os.chdir(web_dir)
|
|
1514
|
+
|
|
1515
|
+
class _Server(socketserver.ThreadingTCPServer):
|
|
1516
|
+
allow_reuse_address = True
|
|
1517
|
+
|
|
1518
|
+
with _Server((host, port), Handler) as srv:
|
|
1519
|
+
query_suffix = ""
|
|
1520
|
+
if preset:
|
|
1521
|
+
from urllib.parse import quote
|
|
1522
|
+
query_suffix = f"?preset={quote(preset)}"
|
|
1523
|
+
console.print(
|
|
1524
|
+
f"serving [cyan]{web_dir}[/cyan] at [cyan]http://{host}:{port}/{query_suffix}[/cyan] (gzip-aware)"
|
|
1525
|
+
)
|
|
1526
|
+
try:
|
|
1527
|
+
srv.serve_forever()
|
|
1528
|
+
except KeyboardInterrupt:
|
|
1529
|
+
pass
|
|
1530
|
+
|
|
1531
|
+
|
|
1532
|
+
# ---------------------------------------------------------------------------
|
|
1533
|
+
# Entry point
|
|
1534
|
+
# ---------------------------------------------------------------------------
|
|
1535
|
+
|
|
1536
|
+
|
|
1537
|
+
def main() -> int:
|
|
1538
|
+
"""CLI entry point — dispatches to subcommand or smart lookup."""
|
|
1539
|
+
global SHOW_FP16_ISAS, SHORT_MODE, FULL_MODE
|
|
1540
|
+
argv = sys.argv[1:]
|
|
1541
|
+
if any(arg in ("--version", "-V") for arg in argv):
|
|
1542
|
+
print(__version__)
|
|
1543
|
+
return 0
|
|
1544
|
+
if _is_completion_invocation():
|
|
1545
|
+
app()
|
|
1546
|
+
return 0
|
|
1547
|
+
if "--fp16" in argv:
|
|
1548
|
+
SHOW_FP16_ISAS = True
|
|
1549
|
+
argv = [arg for arg in argv if arg != "--fp16"]
|
|
1550
|
+
if "--short" in argv or "-s" in argv:
|
|
1551
|
+
SHORT_MODE = True
|
|
1552
|
+
argv = [arg for arg in argv if arg not in ("--short", "-s")]
|
|
1553
|
+
if "--full" in argv or "-f" in argv:
|
|
1554
|
+
FULL_MODE = True
|
|
1555
|
+
argv = [arg for arg in argv if arg not in ("--full", "-f")]
|
|
1556
|
+
# Pre-parse top-level --preset NAME / --preset=NAME for bare-query TUI mode.
|
|
1557
|
+
# Subcommands (llm, etc.) handle their own --preset via Typer, so only
|
|
1558
|
+
# strip it here when it would otherwise reach the smart-lookup dispatch.
|
|
1559
|
+
initial_preset: str | None = None
|
|
1560
|
+
_cleaned: list[str] = []
|
|
1561
|
+
_i = 0
|
|
1562
|
+
while _i < len(argv):
|
|
1563
|
+
arg = argv[_i]
|
|
1564
|
+
if arg == "--preset" and _i + 1 < len(argv):
|
|
1565
|
+
initial_preset = argv[_i + 1]
|
|
1566
|
+
_i += 2
|
|
1567
|
+
continue
|
|
1568
|
+
if arg.startswith("--preset="):
|
|
1569
|
+
initial_preset = arg.split("=", 1)[1]
|
|
1570
|
+
_i += 1
|
|
1571
|
+
continue
|
|
1572
|
+
_cleaned.append(arg)
|
|
1573
|
+
_i += 1
|
|
1574
|
+
# Only consume --preset at the top level when the remainder is a bare
|
|
1575
|
+
# query or empty; otherwise leave it for the subcommand (e.g. `llm query`).
|
|
1576
|
+
subcommand_consumers = {"llm"}
|
|
1577
|
+
if _cleaned and _cleaned[0] in subcommand_consumers:
|
|
1578
|
+
# Restore; let Typer subcommand parse it.
|
|
1579
|
+
pass
|
|
1580
|
+
else:
|
|
1581
|
+
argv = _cleaned
|
|
1582
|
+
# Rewrite `llm <bare-query>` to `llm query <bare-query>` so Typer's
|
|
1583
|
+
# subcommand dispatch (list/batch/schema/query) works without stealing
|
|
1584
|
+
# bare queries. Derived from Typer introspection so new subcommands
|
|
1585
|
+
# automatically become recognised.
|
|
1586
|
+
llm_subcommands = {info.name for info in getattr(llm_app, "registered_commands", []) if info.name}
|
|
1587
|
+
llm_subcommands |= {"--help", "-h"}
|
|
1588
|
+
if argv and argv[0] == "llm" and len(argv) >= 2 and argv[1] not in llm_subcommands and not argv[1].startswith("-"):
|
|
1589
|
+
argv = ["llm", "query", *argv[1:]]
|
|
1590
|
+
sys.argv = [sys.argv[0], *argv]
|
|
1591
|
+
commands = _registered_command_names()
|
|
1592
|
+
if argv and argv[0] not in commands and not argv[0].startswith("-"):
|
|
1593
|
+
return _smart_lookup(" ".join(argv), preset=initial_preset)
|
|
1594
|
+
if not argv:
|
|
1595
|
+
ensure_runtime()
|
|
1596
|
+
return _run_tui(initial_preset=initial_preset or "intel")
|
|
1597
|
+
app()
|
|
1598
|
+
return 0
|