physlint 0.1.0a1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- physlint/__init__.py +6 -0
- physlint/_version.py +3 -0
- physlint/adapters/__init__.py +5 -0
- physlint/adapters/base.py +9 -0
- physlint/adapters/lerobot.py +323 -0
- physlint/api.py +29 -0
- physlint/cli.py +174 -0
- physlint/config.py +89 -0
- physlint/corruptions.py +51 -0
- physlint/engine/__init__.py +1 -0
- physlint/engine/discovery.py +33 -0
- physlint/engine/planner.py +86 -0
- physlint/engine/runner.py +141 -0
- physlint/models/__init__.py +15 -0
- physlint/models/dataset.py +112 -0
- physlint/models/finding.py +95 -0
- physlint/models/rule.py +35 -0
- physlint/py.typed +1 -0
- physlint/reporters/__init__.py +1 -0
- physlint/reporters/json.py +28 -0
- physlint/reporters/terminal.py +92 -0
- physlint/rules/__init__.py +11 -0
- physlint/rules/common.py +71 -0
- physlint/rules/manifest.py +292 -0
- physlint/rules/numeric.py +182 -0
- physlint/rules/temporal.py +312 -0
- physlint/rules/video.py +285 -0
- physlint-0.1.0a1.dist-info/METADATA +350 -0
- physlint-0.1.0a1.dist-info/RECORD +32 -0
- physlint-0.1.0a1.dist-info/WHEEL +4 -0
- physlint-0.1.0a1.dist-info/entry_points.txt +2 -0
- physlint-0.1.0a1.dist-info/licenses/LICENSE +21 -0
physlint/__init__.py
ADDED
physlint/_version.py
ADDED
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
"""A dependency-light, read-only adapter for local LeRobot Dataset v3.0."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections.abc import Iterator
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
import pyarrow as pa
|
|
12
|
+
import pyarrow.parquet as pq
|
|
13
|
+
|
|
14
|
+
from physlint.adapters.base import AdapterError
|
|
15
|
+
from physlint.models.dataset import (
|
|
16
|
+
DatasetInventory,
|
|
17
|
+
Episode,
|
|
18
|
+
SampleBatch,
|
|
19
|
+
Stream,
|
|
20
|
+
VideoAnalysis,
|
|
21
|
+
VideoFrame,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
DEFAULT_DATA_PATH = "data/chunk-{chunk_index:03d}/file-{file_index:03d}.parquet"
|
|
25
|
+
DEFAULT_VIDEO_PATH = "videos/{video_key}/chunk-{chunk_index:03d}/file-{file_index:03d}.mp4"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class LeRobotAdapter:
|
|
29
|
+
"""Expose LeRobot v3 metadata and samples through the canonical lazy view."""
|
|
30
|
+
|
|
31
|
+
name = "lerobot"
|
|
32
|
+
version = "3.0"
|
|
33
|
+
|
|
34
|
+
def __init__(self, root: Path):
|
|
35
|
+
self.root = root
|
|
36
|
+
self._info = self._read_info()
|
|
37
|
+
self._features = self._read_features()
|
|
38
|
+
self._episodes = self._read_episodes()
|
|
39
|
+
self._schema_cache: dict[Path, pa.Schema] = {}
|
|
40
|
+
self._video_analysis_cache: dict[tuple[int, str], VideoAnalysis] = {}
|
|
41
|
+
inferred_name, source_revision = _hugging_face_identity(root)
|
|
42
|
+
capabilities = {"metadata", "episodes", "timestamps", "numeric"}
|
|
43
|
+
if any(stream.kind == "video" for stream in self._features):
|
|
44
|
+
capabilities.add("video")
|
|
45
|
+
if any(stream.key.endswith(".timestamp") for stream in self._features):
|
|
46
|
+
capabilities.add("stream_timestamps")
|
|
47
|
+
self.inventory = DatasetInventory(
|
|
48
|
+
name=str(self._info.get("repo_id") or inferred_name or root.name),
|
|
49
|
+
source_revision=source_revision,
|
|
50
|
+
path=str(root),
|
|
51
|
+
adapter=self.name,
|
|
52
|
+
format_version=str(self._info.get("codebase_version", "unknown")),
|
|
53
|
+
robot_type=self._info.get("robot_type"),
|
|
54
|
+
fps=float(self._info["fps"]),
|
|
55
|
+
total_frames=int(self._info.get("total_frames", sum(ep.length for ep in self._episodes))),
|
|
56
|
+
streams=self._features,
|
|
57
|
+
episodes=self._episodes,
|
|
58
|
+
capabilities=frozenset(capabilities),
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
def _read_info(self) -> dict[str, Any]:
|
|
62
|
+
path = self.root / "meta" / "info.json"
|
|
63
|
+
try:
|
|
64
|
+
payload = json.loads(path.read_text(encoding="utf-8"))
|
|
65
|
+
except FileNotFoundError as exc:
|
|
66
|
+
raise AdapterError(f"missing LeRobot metadata: {path}") from exc
|
|
67
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
68
|
+
raise AdapterError(f"unreadable LeRobot metadata {path}: {exc}") from exc
|
|
69
|
+
if not isinstance(payload, dict):
|
|
70
|
+
raise AdapterError(f"LeRobot info must be an object: {path}")
|
|
71
|
+
version = str(payload.get("codebase_version", ""))
|
|
72
|
+
if not version.startswith("v3"):
|
|
73
|
+
raise AdapterError(f"unsupported LeRobot version {version or 'missing'}; Physlint supports v3.x")
|
|
74
|
+
if not isinstance(payload.get("features"), dict) or "fps" not in payload:
|
|
75
|
+
raise AdapterError("LeRobot info.json is missing features or fps")
|
|
76
|
+
return payload
|
|
77
|
+
|
|
78
|
+
def _read_features(self) -> list[Stream]:
|
|
79
|
+
streams: list[Stream] = []
|
|
80
|
+
for key, raw in self._info["features"].items():
|
|
81
|
+
if not isinstance(raw, dict):
|
|
82
|
+
raise AdapterError(f"feature declaration for {key!r} must be an object")
|
|
83
|
+
dtype = str(raw.get("dtype", "unknown"))
|
|
84
|
+
shape = tuple(int(value) for value in raw.get("shape", ()))
|
|
85
|
+
kind = "video" if dtype == "video" else "image" if dtype == "image" else "numeric"
|
|
86
|
+
streams.append(
|
|
87
|
+
Stream(
|
|
88
|
+
key=key,
|
|
89
|
+
dtype=dtype,
|
|
90
|
+
shape=shape,
|
|
91
|
+
kind=kind,
|
|
92
|
+
names=raw.get("names"),
|
|
93
|
+
units=raw.get("unit") or raw.get("units"),
|
|
94
|
+
)
|
|
95
|
+
)
|
|
96
|
+
return streams
|
|
97
|
+
|
|
98
|
+
def _read_episodes(self) -> list[Episode]:
|
|
99
|
+
paths = sorted((self.root / "meta" / "episodes").glob("**/*.parquet"))
|
|
100
|
+
if not paths:
|
|
101
|
+
raise AdapterError("missing LeRobot v3 episode metadata parquet files")
|
|
102
|
+
records: list[dict[str, Any]] = []
|
|
103
|
+
try:
|
|
104
|
+
for path in paths:
|
|
105
|
+
table = pq.read_table(path)
|
|
106
|
+
records.extend(table.to_pylist())
|
|
107
|
+
except (OSError, pa.ArrowException) as exc:
|
|
108
|
+
if "Repetition level histogram size mismatch" in str(exc):
|
|
109
|
+
raise AdapterError(
|
|
110
|
+
"cannot read episode metadata because PyArrow 19.0.0 has a Parquet repetition-level "
|
|
111
|
+
"reader bug; upgrade PyArrow to >=19.0.1 and rerun the validation"
|
|
112
|
+
) from exc
|
|
113
|
+
raise AdapterError(f"cannot read episode metadata: {exc}") from exc
|
|
114
|
+
episodes = [self._episode_from_record(record) for record in records]
|
|
115
|
+
return sorted(episodes, key=lambda episode: episode.index)
|
|
116
|
+
|
|
117
|
+
def _episode_from_record(self, record: dict[str, Any]) -> Episode:
|
|
118
|
+
try:
|
|
119
|
+
index = int(record["episode_index"])
|
|
120
|
+
length = int(record["length"])
|
|
121
|
+
except (KeyError, TypeError, ValueError) as exc:
|
|
122
|
+
raise AdapterError("episode metadata lacks a valid episode_index or length") from exc
|
|
123
|
+
data_file: str | None = None
|
|
124
|
+
if "data/chunk_index" in record and "data/file_index" in record:
|
|
125
|
+
data_file = self._format_path(
|
|
126
|
+
str(self._info.get("data_path", DEFAULT_DATA_PATH)),
|
|
127
|
+
chunk_index=int(record["data/chunk_index"]),
|
|
128
|
+
file_index=int(record["data/file_index"]),
|
|
129
|
+
)
|
|
130
|
+
video_files: dict[str, str] = {}
|
|
131
|
+
video_ranges: dict[str, tuple[float, float]] = {}
|
|
132
|
+
for stream in self._features:
|
|
133
|
+
if stream.kind != "video":
|
|
134
|
+
continue
|
|
135
|
+
prefix = f"videos/{stream.key}"
|
|
136
|
+
if f"{prefix}/chunk_index" in record and f"{prefix}/file_index" in record:
|
|
137
|
+
video_files[stream.key] = self._format_path(
|
|
138
|
+
str(self._info.get("video_path", DEFAULT_VIDEO_PATH)),
|
|
139
|
+
video_key=stream.key,
|
|
140
|
+
chunk_index=int(record[f"{prefix}/chunk_index"]),
|
|
141
|
+
file_index=int(record[f"{prefix}/file_index"]),
|
|
142
|
+
)
|
|
143
|
+
start = float(record.get(f"{prefix}/from_timestamp", 0.0))
|
|
144
|
+
end = float(record.get(f"{prefix}/to_timestamp", start + length / self._info["fps"]))
|
|
145
|
+
video_ranges[stream.key] = (start, end)
|
|
146
|
+
tasks = record.get("tasks") or []
|
|
147
|
+
if isinstance(tasks, str):
|
|
148
|
+
tasks = [tasks]
|
|
149
|
+
return Episode(
|
|
150
|
+
index=index,
|
|
151
|
+
identifier=str(record.get("episode_id", f"episode_{index:06d}")),
|
|
152
|
+
length=length,
|
|
153
|
+
data_file=data_file,
|
|
154
|
+
from_index=_optional_int(record.get("dataset_from_index")),
|
|
155
|
+
to_index=_optional_int(record.get("dataset_to_index")),
|
|
156
|
+
tasks=[str(task) for task in tasks],
|
|
157
|
+
video_files=video_files,
|
|
158
|
+
video_ranges=video_ranges,
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
@staticmethod
|
|
162
|
+
def _format_path(template: str, **values: Any) -> str:
|
|
163
|
+
try:
|
|
164
|
+
return template.format(**values)
|
|
165
|
+
except (KeyError, ValueError) as exc:
|
|
166
|
+
raise AdapterError(f"invalid path template in info.json: {template}") from exc
|
|
167
|
+
|
|
168
|
+
def _data_path(self, episode: Episode) -> Path:
|
|
169
|
+
if episode.data_file is None:
|
|
170
|
+
raise AdapterError(f"episode {episode.identifier} has no data-file reference")
|
|
171
|
+
path = self.root / episode.data_file
|
|
172
|
+
if not path.is_file():
|
|
173
|
+
raise AdapterError(f"episode data file is missing: {path}")
|
|
174
|
+
return path
|
|
175
|
+
|
|
176
|
+
def parquet_schema(self, episode: Episode) -> dict[str, Any]:
|
|
177
|
+
path = self._data_path(episode)
|
|
178
|
+
if path not in self._schema_cache:
|
|
179
|
+
try:
|
|
180
|
+
self._schema_cache[path] = pq.read_schema(path)
|
|
181
|
+
except (OSError, pa.ArrowException) as exc:
|
|
182
|
+
raise AdapterError(f"cannot read parquet schema {path}: {exc}") from exc
|
|
183
|
+
return {field.name: field.type for field in self._schema_cache[path]}
|
|
184
|
+
|
|
185
|
+
def iter_batches(
|
|
186
|
+
self,
|
|
187
|
+
episode: Episode,
|
|
188
|
+
columns: tuple[str, ...] | None = None,
|
|
189
|
+
batch_size: int = 65_536,
|
|
190
|
+
) -> Iterator[SampleBatch]:
|
|
191
|
+
path = self._data_path(episode)
|
|
192
|
+
try:
|
|
193
|
+
parquet = pq.ParquetFile(path)
|
|
194
|
+
available = set(parquet.schema_arrow.names)
|
|
195
|
+
requested = list(columns or tuple(available))
|
|
196
|
+
selector = "episode_index" if "episode_index" in available else "index"
|
|
197
|
+
read_columns = list(dict.fromkeys([*requested, selector]))
|
|
198
|
+
missing = set(read_columns) - available
|
|
199
|
+
if missing:
|
|
200
|
+
raise AdapterError(f"columns missing from {path}: {', '.join(sorted(missing))}")
|
|
201
|
+
for raw in parquet.iter_batches(batch_size=batch_size, columns=read_columns):
|
|
202
|
+
selection = raw.column(raw.schema.get_field_index(selector)).to_numpy(zero_copy_only=False)
|
|
203
|
+
if selector == "episode_index":
|
|
204
|
+
mask = selection == episode.index
|
|
205
|
+
elif episode.from_index is not None and episode.to_index is not None:
|
|
206
|
+
mask = (selection >= episode.from_index) & (selection < episode.to_index)
|
|
207
|
+
else:
|
|
208
|
+
mask = np.ones(len(selection), dtype=bool)
|
|
209
|
+
if not np.any(mask):
|
|
210
|
+
continue
|
|
211
|
+
converted: dict[str, np.ndarray] = {}
|
|
212
|
+
for name in requested:
|
|
213
|
+
array = raw.column(raw.schema.get_field_index(name))
|
|
214
|
+
converted[name] = _arrow_to_numpy(array)[mask]
|
|
215
|
+
yield SampleBatch(episode=episode, columns=converted)
|
|
216
|
+
except AdapterError:
|
|
217
|
+
raise
|
|
218
|
+
except (OSError, pa.ArrowException) as exc:
|
|
219
|
+
raise AdapterError(f"cannot stream parquet data {path}: {exc}") from exc
|
|
220
|
+
|
|
221
|
+
def iter_video_frames(self, episode: Episode, stream: str, stride: int = 1) -> Iterator[VideoFrame]:
|
|
222
|
+
try:
|
|
223
|
+
import cv2
|
|
224
|
+
except ImportError as exc: # pragma: no cover - environment dependent
|
|
225
|
+
raise AdapterError("video checks require the 'video' optional dependency") from exc
|
|
226
|
+
if stream not in episode.video_files:
|
|
227
|
+
raise AdapterError(f"episode {episode.identifier} has no {stream} video reference")
|
|
228
|
+
path = self.root / episode.video_files[stream]
|
|
229
|
+
capture = cv2.VideoCapture(str(path))
|
|
230
|
+
if not capture.isOpened():
|
|
231
|
+
capture.release()
|
|
232
|
+
raise AdapterError(f"cannot decode video: {path}")
|
|
233
|
+
fps = float(capture.get(cv2.CAP_PROP_FPS)) or self.inventory.fps
|
|
234
|
+
start, end = episode.video_ranges.get(stream, (0.0, episode.length / fps))
|
|
235
|
+
start_frame = max(0, int(round(start * fps)))
|
|
236
|
+
end_frame = max(start_frame, int(round(end * fps)))
|
|
237
|
+
capture.set(cv2.CAP_PROP_POS_FRAMES, start_frame)
|
|
238
|
+
try:
|
|
239
|
+
for frame_index in range(start_frame, end_frame):
|
|
240
|
+
ok, image = capture.read()
|
|
241
|
+
if not ok:
|
|
242
|
+
break
|
|
243
|
+
relative = frame_index - start_frame
|
|
244
|
+
if relative % max(1, stride):
|
|
245
|
+
continue
|
|
246
|
+
yield VideoFrame(
|
|
247
|
+
episode=episode,
|
|
248
|
+
stream=stream,
|
|
249
|
+
frame_index=relative,
|
|
250
|
+
timestamp=relative / fps,
|
|
251
|
+
image=image,
|
|
252
|
+
)
|
|
253
|
+
finally:
|
|
254
|
+
capture.release()
|
|
255
|
+
|
|
256
|
+
def video_analysis(self, episode: Episode, stream: str) -> VideoAnalysis:
|
|
257
|
+
"""Decode once and cache small, non-image statistics for every video rule."""
|
|
258
|
+
key = (episode.index, stream)
|
|
259
|
+
cached = self._video_analysis_cache.get(key)
|
|
260
|
+
if cached is not None:
|
|
261
|
+
return cached
|
|
262
|
+
|
|
263
|
+
frame_indices: list[int] = []
|
|
264
|
+
timestamps: list[float] = []
|
|
265
|
+
means: list[float] = []
|
|
266
|
+
stddevs: list[float] = []
|
|
267
|
+
differences: list[float] = []
|
|
268
|
+
previous: np.ndarray | None = None
|
|
269
|
+
for frame in self.iter_video_frames(episode, stream):
|
|
270
|
+
small = _small_grayscale(frame.image)
|
|
271
|
+
frame_indices.append(frame.frame_index)
|
|
272
|
+
timestamps.append(frame.timestamp)
|
|
273
|
+
means.append(float(np.mean(small)))
|
|
274
|
+
stddevs.append(float(np.std(small)))
|
|
275
|
+
if previous is not None:
|
|
276
|
+
differences.append(float(np.mean(np.abs(small - previous))))
|
|
277
|
+
previous = small
|
|
278
|
+
analysis = VideoAnalysis(
|
|
279
|
+
frame_indices=np.asarray(frame_indices, dtype=np.int64),
|
|
280
|
+
timestamps=np.asarray(timestamps, dtype=float),
|
|
281
|
+
mean_intensities=np.asarray(means, dtype=float),
|
|
282
|
+
stddevs=np.asarray(stddevs, dtype=float),
|
|
283
|
+
mean_absolute_differences=np.asarray(differences, dtype=float),
|
|
284
|
+
)
|
|
285
|
+
self._video_analysis_cache[key] = analysis
|
|
286
|
+
return analysis
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _optional_int(value: Any) -> int | None:
|
|
290
|
+
return None if value is None else int(value)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _small_grayscale(image: np.ndarray) -> np.ndarray:
|
|
294
|
+
"""Spatially sample first, avoiding full-resolution color reductions."""
|
|
295
|
+
row_stride = max(1, image.shape[0] // 64)
|
|
296
|
+
col_stride = max(1, image.shape[1] // 64)
|
|
297
|
+
small = image[::row_stride, ::col_stride]
|
|
298
|
+
gray = np.mean(small, axis=2) if small.ndim == 3 else small
|
|
299
|
+
return np.asarray(gray, dtype=np.float32)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _hugging_face_identity(root: Path) -> tuple[str | None, str | None]:
|
|
303
|
+
"""Recover repo/revision from a Hugging Face snapshot cache path."""
|
|
304
|
+
parts = root.resolve().parts
|
|
305
|
+
for index, part in enumerate(parts):
|
|
306
|
+
if not part.startswith("datasets--") or index + 2 >= len(parts):
|
|
307
|
+
continue
|
|
308
|
+
if parts[index + 1] != "snapshots":
|
|
309
|
+
continue
|
|
310
|
+
encoded = part.removeprefix("datasets--")
|
|
311
|
+
namespace, separator, repository = encoded.partition("--")
|
|
312
|
+
if separator and namespace and repository:
|
|
313
|
+
return f"{namespace}/{repository}", parts[index + 2]
|
|
314
|
+
return None, None
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def _arrow_to_numpy(array: pa.Array) -> np.ndarray:
|
|
318
|
+
"""Preserve vector columns as a dense array when their shape is regular."""
|
|
319
|
+
values = array.to_pylist()
|
|
320
|
+
try:
|
|
321
|
+
return np.asarray(values)
|
|
322
|
+
except ValueError:
|
|
323
|
+
return np.asarray(values, dtype=object)
|
physlint/api.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Stable library entry points independent of terminal behavior."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from physlint.config import Config, load_config
|
|
8
|
+
from physlint.engine.discovery import discover
|
|
9
|
+
from physlint.engine.runner import run_validation
|
|
10
|
+
from physlint.models.dataset import DatasetInventory
|
|
11
|
+
from physlint.models.finding import Report
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def inspect_dataset(path: str | Path, *, adapter: str = "auto") -> DatasetInventory:
|
|
15
|
+
return discover(path, adapter).inventory
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def check_dataset(
|
|
19
|
+
path: str | Path,
|
|
20
|
+
*,
|
|
21
|
+
config: Config | None = None,
|
|
22
|
+
config_path: str | Path | None = None,
|
|
23
|
+
) -> Report:
|
|
24
|
+
root = Path(path).expanduser().resolve()
|
|
25
|
+
resolved_config = config or load_config(
|
|
26
|
+
Path(config_path).expanduser().resolve() if config_path is not None else None, root
|
|
27
|
+
)
|
|
28
|
+
dataset = discover(root, resolved_config.adapter)
|
|
29
|
+
return run_validation(dataset, resolved_config)
|
physlint/cli.py
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""Physlint command-line interface and frozen exit codes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from datetime import UTC
|
|
7
|
+
from enum import StrEnum
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Annotated
|
|
10
|
+
|
|
11
|
+
import typer
|
|
12
|
+
from rich.console import Console
|
|
13
|
+
|
|
14
|
+
from physlint import __version__
|
|
15
|
+
from physlint.adapters.base import AdapterError
|
|
16
|
+
from physlint.config import DEFAULT_CONFIG, ConfigurationError, load_config
|
|
17
|
+
from physlint.engine.discovery import discover
|
|
18
|
+
from physlint.engine.runner import run_validation
|
|
19
|
+
from physlint.reporters.json import write_json_report
|
|
20
|
+
from physlint.reporters.terminal import (
|
|
21
|
+
render_explanation,
|
|
22
|
+
render_inventory,
|
|
23
|
+
render_report,
|
|
24
|
+
render_rules,
|
|
25
|
+
)
|
|
26
|
+
from physlint.rules import BUILTIN_RULES
|
|
27
|
+
|
|
28
|
+
app = typer.Typer(
|
|
29
|
+
name="physlint",
|
|
30
|
+
help="Find concrete integrity defects in physical-AI datasets.",
|
|
31
|
+
no_args_is_help=True,
|
|
32
|
+
pretty_exceptions_enable=False,
|
|
33
|
+
)
|
|
34
|
+
console = Console()
|
|
35
|
+
error_console = Console(stderr=True)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class OutputFormat(StrEnum):
|
|
39
|
+
TERMINAL = "terminal"
|
|
40
|
+
JSON = "json"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _version_callback(value: bool) -> None:
|
|
44
|
+
if value:
|
|
45
|
+
typer.echo(__version__)
|
|
46
|
+
raise typer.Exit(0)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@app.callback()
|
|
50
|
+
def main(
|
|
51
|
+
version: Annotated[
|
|
52
|
+
bool,
|
|
53
|
+
typer.Option("--version", callback=_version_callback, is_eager=True, help="Show version."),
|
|
54
|
+
] = False,
|
|
55
|
+
) -> None:
|
|
56
|
+
"""Validate robot-learning data before it reaches training."""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@app.command()
|
|
60
|
+
def check(
|
|
61
|
+
path: Annotated[Path, typer.Argument(help="Local LeRobot v3 dataset directory.")],
|
|
62
|
+
config: Annotated[Path | None, typer.Option("--config", "-c", help="Quality contract YAML file.")] = None,
|
|
63
|
+
output: Annotated[
|
|
64
|
+
OutputFormat, typer.Option("--output", "-o", help="Terminal or JSON stdout output.")
|
|
65
|
+
] = OutputFormat.TERMINAL,
|
|
66
|
+
json_output: Annotated[Path | None, typer.Option("--json-output", help="Exact JSON report destination.")] = None,
|
|
67
|
+
) -> None:
|
|
68
|
+
"""Validate a dataset and return a CI-safe pass/fail exit code."""
|
|
69
|
+
try:
|
|
70
|
+
root = path.expanduser().resolve()
|
|
71
|
+
settings = load_config(config.expanduser().resolve() if config else None, root)
|
|
72
|
+
dataset = discover(root, settings.adapter)
|
|
73
|
+
report = run_validation(dataset, settings)
|
|
74
|
+
report_path: Path | None = None
|
|
75
|
+
if json_output is not None:
|
|
76
|
+
report_path = write_json_report(report, json_output)
|
|
77
|
+
elif settings.reports.json_enabled:
|
|
78
|
+
timestamp = report.started_at.astimezone(UTC).strftime("%Y-%m-%dT%H%M%SZ")
|
|
79
|
+
report_path = write_json_report(report, Path(settings.reports.output_dir) / f"{timestamp}.json")
|
|
80
|
+
if output == OutputFormat.JSON:
|
|
81
|
+
typer.echo(report.model_dump_json(indent=2))
|
|
82
|
+
else:
|
|
83
|
+
render_report(report, report_path, console)
|
|
84
|
+
adapter_errors = any(result.error_kind == "adapter" for result in report.results)
|
|
85
|
+
rule_errors = any(result.error_kind == "rule" for result in report.results)
|
|
86
|
+
if adapter_errors:
|
|
87
|
+
raise typer.Exit(3)
|
|
88
|
+
if rule_errors:
|
|
89
|
+
raise typer.Exit(4)
|
|
90
|
+
if report.status == "failed":
|
|
91
|
+
raise typer.Exit(1)
|
|
92
|
+
except KeyboardInterrupt as exc:
|
|
93
|
+
raise typer.Exit(130) from exc
|
|
94
|
+
except ConfigurationError as exc:
|
|
95
|
+
error_console.print(f"[red]Configuration error:[/red] {exc}")
|
|
96
|
+
raise typer.Exit(2) from exc
|
|
97
|
+
except AdapterError as exc:
|
|
98
|
+
error_console.print(f"[red]Dataset error:[/red] {exc}")
|
|
99
|
+
raise typer.Exit(3) from exc
|
|
100
|
+
except typer.Exit:
|
|
101
|
+
raise
|
|
102
|
+
except Exception as exc: # noqa: BLE001 - CLI must preserve the documented code
|
|
103
|
+
error_console.print(f"[red]Internal Physlint error:[/red] {type(exc).__name__}: {exc}")
|
|
104
|
+
raise typer.Exit(4) from exc
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@app.command()
|
|
108
|
+
def inspect(
|
|
109
|
+
path: Annotated[Path, typer.Argument(help="Local LeRobot v3 dataset directory.")],
|
|
110
|
+
output_json: Annotated[bool, typer.Option("--json", help="Print the inventory as JSON.")] = False,
|
|
111
|
+
) -> None:
|
|
112
|
+
"""Show streams, schemas, rates, and episode inventory."""
|
|
113
|
+
try:
|
|
114
|
+
inventory = discover(path).inventory
|
|
115
|
+
if output_json:
|
|
116
|
+
typer.echo(inventory.model_dump_json(indent=2))
|
|
117
|
+
else:
|
|
118
|
+
render_inventory(inventory, console)
|
|
119
|
+
except AdapterError as exc:
|
|
120
|
+
error_console.print(f"[red]Dataset error:[/red] {exc}")
|
|
121
|
+
raise typer.Exit(3) from exc
|
|
122
|
+
except KeyboardInterrupt as exc:
|
|
123
|
+
raise typer.Exit(130) from exc
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@app.command("rules")
|
|
127
|
+
def list_rules(
|
|
128
|
+
output_json: Annotated[bool, typer.Option("--json", help="Print rule metadata as JSON.")] = False,
|
|
129
|
+
) -> None:
|
|
130
|
+
"""List built-in rules and their requirements."""
|
|
131
|
+
metadata = [rule.metadata for rule in BUILTIN_RULES]
|
|
132
|
+
if output_json:
|
|
133
|
+
typer.echo(
|
|
134
|
+
json.dumps(
|
|
135
|
+
[
|
|
136
|
+
{
|
|
137
|
+
**item.__dict__,
|
|
138
|
+
"severity": item.severity.value,
|
|
139
|
+
"required_capabilities": sorted(item.required_capabilities),
|
|
140
|
+
"required_streams": sorted(item.required_streams),
|
|
141
|
+
}
|
|
142
|
+
for item in metadata
|
|
143
|
+
],
|
|
144
|
+
indent=2,
|
|
145
|
+
)
|
|
146
|
+
)
|
|
147
|
+
else:
|
|
148
|
+
render_rules(metadata, console)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@app.command()
|
|
152
|
+
def explain(rule_id: Annotated[str, typer.Argument(help="Stable rule ID.")]) -> None:
|
|
153
|
+
"""Explain one rule, including remediation and limitations."""
|
|
154
|
+
for rule in BUILTIN_RULES:
|
|
155
|
+
if rule.metadata.id == rule_id:
|
|
156
|
+
render_explanation(rule.metadata, console)
|
|
157
|
+
return
|
|
158
|
+
error_console.print(f"[red]Unknown rule ID:[/red] {rule_id}")
|
|
159
|
+
raise typer.Exit(2)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
@app.command()
|
|
163
|
+
def init(
|
|
164
|
+
path: Annotated[Path, typer.Option("--path", "-p", help="Configuration file to create.")] = Path("physlint.yaml"),
|
|
165
|
+
force: Annotated[bool, typer.Option("--force", help="Overwrite an existing file.")] = False,
|
|
166
|
+
) -> None:
|
|
167
|
+
"""Generate a documented quality-contract configuration."""
|
|
168
|
+
destination = path.expanduser().resolve()
|
|
169
|
+
if destination.exists() and not force:
|
|
170
|
+
error_console.print(f"[red]Refusing to overwrite existing file:[/red] {destination}")
|
|
171
|
+
raise typer.Exit(2)
|
|
172
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
173
|
+
destination.write_text(DEFAULT_CONFIG, encoding="utf-8")
|
|
174
|
+
console.print(f"Created {destination}")
|
physlint/config.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Strict project configuration and quality-contract loading."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any, Literal
|
|
9
|
+
|
|
10
|
+
import yaml
|
|
11
|
+
from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class RuleSettings(BaseModel):
|
|
15
|
+
model_config = ConfigDict(extra="forbid")
|
|
16
|
+
|
|
17
|
+
enabled: bool = True
|
|
18
|
+
severity: Literal["critical", "error", "warning", "notice"] | None = None
|
|
19
|
+
options: dict[str, Any] = Field(default_factory=dict)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ReportSettings(BaseModel):
|
|
23
|
+
model_config = ConfigDict(extra="forbid", populate_by_name=True)
|
|
24
|
+
|
|
25
|
+
json_enabled: bool = Field(default=True, alias="json")
|
|
26
|
+
output_dir: str = ".physlint/reports"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Config(BaseModel):
|
|
30
|
+
model_config = ConfigDict(extra="forbid")
|
|
31
|
+
|
|
32
|
+
config_version: Literal[1] = 1
|
|
33
|
+
adapter: Literal["auto", "lerobot"] = "auto"
|
|
34
|
+
required_streams: list[str] = Field(default_factory=lambda: ["observation.state", "action"])
|
|
35
|
+
fail_on: Literal["critical", "error", "warning", "notice"] = "error"
|
|
36
|
+
rules: dict[str, RuleSettings] = Field(default_factory=dict)
|
|
37
|
+
reports: ReportSettings = Field(default_factory=ReportSettings)
|
|
38
|
+
|
|
39
|
+
@field_validator("required_streams")
|
|
40
|
+
@classmethod
|
|
41
|
+
def unique_streams(cls, value: list[str]) -> list[str]:
|
|
42
|
+
if len(value) != len(set(value)):
|
|
43
|
+
raise ValueError("required_streams must not contain duplicates")
|
|
44
|
+
return value
|
|
45
|
+
|
|
46
|
+
def digest(self) -> str:
|
|
47
|
+
data = json.dumps(self.model_dump(mode="json", by_alias=True), sort_keys=True, separators=(",", ":"))
|
|
48
|
+
return hashlib.sha256(data.encode()).hexdigest()
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class ConfigurationError(ValueError):
|
|
52
|
+
"""Invalid or unreadable user configuration."""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def load_config(path: Path | None, dataset_path: Path | None = None) -> Config:
|
|
56
|
+
candidate = path
|
|
57
|
+
if candidate is None and dataset_path is not None:
|
|
58
|
+
local = dataset_path / "physlint.yaml"
|
|
59
|
+
cwd = Path.cwd() / "physlint.yaml"
|
|
60
|
+
candidate = local if local.is_file() else cwd if cwd.is_file() else None
|
|
61
|
+
if candidate is None:
|
|
62
|
+
return Config()
|
|
63
|
+
try:
|
|
64
|
+
payload = yaml.safe_load(candidate.read_text(encoding="utf-8")) or {}
|
|
65
|
+
if not isinstance(payload, dict):
|
|
66
|
+
raise ConfigurationError("configuration root must be a mapping")
|
|
67
|
+
return Config.model_validate(payload)
|
|
68
|
+
except (OSError, yaml.YAMLError, ValidationError) as exc:
|
|
69
|
+
raise ConfigurationError(f"invalid configuration {candidate}: {exc}") from exc
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
DEFAULT_CONFIG = """# Physlint quality contract
|
|
73
|
+
config_version: 1
|
|
74
|
+
adapter: auto
|
|
75
|
+
required_streams:
|
|
76
|
+
- observation.state
|
|
77
|
+
- action
|
|
78
|
+
fail_on: error
|
|
79
|
+
rules:
|
|
80
|
+
temporal.max_gap:
|
|
81
|
+
options:
|
|
82
|
+
max_gap_multiplier: 2.0
|
|
83
|
+
video.frozen_frames:
|
|
84
|
+
options:
|
|
85
|
+
max_consecutive_frames: 5
|
|
86
|
+
reports:
|
|
87
|
+
json: true
|
|
88
|
+
output_dir: .physlint/reports
|
|
89
|
+
"""
|