physlint 0.1.0a1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
physlint/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """Physlint public package."""
2
+
3
+ from physlint._version import __version__ as __version__
4
+ from physlint.api import check_dataset, inspect_dataset
5
+
6
+ __all__ = ["check_dataset", "inspect_dataset"]
physlint/_version.py ADDED
@@ -0,0 +1,3 @@
1
+ """Single package version source."""
2
+
3
+ __version__ = "0.1.0a1"
@@ -0,0 +1,5 @@
1
+ """Dataset adapter registry."""
2
+
3
+ from physlint.adapters.lerobot import LeRobotAdapter
4
+
5
+ __all__ = ["LeRobotAdapter"]
@@ -0,0 +1,9 @@
1
+ """Adapter errors shared by discovery and the CLI."""
2
+
3
+
4
+ class AdapterError(RuntimeError):
5
+ """The selected adapter could not safely read the dataset."""
6
+
7
+
8
+ class UnsupportedDatasetError(AdapterError):
9
+ """No installed adapter recognizes the source."""
@@ -0,0 +1,323 @@
1
+ """A dependency-light, read-only adapter for local LeRobot Dataset v3.0."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from collections.abc import Iterator
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ import numpy as np
11
+ import pyarrow as pa
12
+ import pyarrow.parquet as pq
13
+
14
+ from physlint.adapters.base import AdapterError
15
+ from physlint.models.dataset import (
16
+ DatasetInventory,
17
+ Episode,
18
+ SampleBatch,
19
+ Stream,
20
+ VideoAnalysis,
21
+ VideoFrame,
22
+ )
23
+
24
+ DEFAULT_DATA_PATH = "data/chunk-{chunk_index:03d}/file-{file_index:03d}.parquet"
25
+ DEFAULT_VIDEO_PATH = "videos/{video_key}/chunk-{chunk_index:03d}/file-{file_index:03d}.mp4"
26
+
27
+
28
+ class LeRobotAdapter:
29
+ """Expose LeRobot v3 metadata and samples through the canonical lazy view."""
30
+
31
+ name = "lerobot"
32
+ version = "3.0"
33
+
34
+ def __init__(self, root: Path):
35
+ self.root = root
36
+ self._info = self._read_info()
37
+ self._features = self._read_features()
38
+ self._episodes = self._read_episodes()
39
+ self._schema_cache: dict[Path, pa.Schema] = {}
40
+ self._video_analysis_cache: dict[tuple[int, str], VideoAnalysis] = {}
41
+ inferred_name, source_revision = _hugging_face_identity(root)
42
+ capabilities = {"metadata", "episodes", "timestamps", "numeric"}
43
+ if any(stream.kind == "video" for stream in self._features):
44
+ capabilities.add("video")
45
+ if any(stream.key.endswith(".timestamp") for stream in self._features):
46
+ capabilities.add("stream_timestamps")
47
+ self.inventory = DatasetInventory(
48
+ name=str(self._info.get("repo_id") or inferred_name or root.name),
49
+ source_revision=source_revision,
50
+ path=str(root),
51
+ adapter=self.name,
52
+ format_version=str(self._info.get("codebase_version", "unknown")),
53
+ robot_type=self._info.get("robot_type"),
54
+ fps=float(self._info["fps"]),
55
+ total_frames=int(self._info.get("total_frames", sum(ep.length for ep in self._episodes))),
56
+ streams=self._features,
57
+ episodes=self._episodes,
58
+ capabilities=frozenset(capabilities),
59
+ )
60
+
61
+ def _read_info(self) -> dict[str, Any]:
62
+ path = self.root / "meta" / "info.json"
63
+ try:
64
+ payload = json.loads(path.read_text(encoding="utf-8"))
65
+ except FileNotFoundError as exc:
66
+ raise AdapterError(f"missing LeRobot metadata: {path}") from exc
67
+ except (OSError, json.JSONDecodeError) as exc:
68
+ raise AdapterError(f"unreadable LeRobot metadata {path}: {exc}") from exc
69
+ if not isinstance(payload, dict):
70
+ raise AdapterError(f"LeRobot info must be an object: {path}")
71
+ version = str(payload.get("codebase_version", ""))
72
+ if not version.startswith("v3"):
73
+ raise AdapterError(f"unsupported LeRobot version {version or 'missing'}; Physlint supports v3.x")
74
+ if not isinstance(payload.get("features"), dict) or "fps" not in payload:
75
+ raise AdapterError("LeRobot info.json is missing features or fps")
76
+ return payload
77
+
78
+ def _read_features(self) -> list[Stream]:
79
+ streams: list[Stream] = []
80
+ for key, raw in self._info["features"].items():
81
+ if not isinstance(raw, dict):
82
+ raise AdapterError(f"feature declaration for {key!r} must be an object")
83
+ dtype = str(raw.get("dtype", "unknown"))
84
+ shape = tuple(int(value) for value in raw.get("shape", ()))
85
+ kind = "video" if dtype == "video" else "image" if dtype == "image" else "numeric"
86
+ streams.append(
87
+ Stream(
88
+ key=key,
89
+ dtype=dtype,
90
+ shape=shape,
91
+ kind=kind,
92
+ names=raw.get("names"),
93
+ units=raw.get("unit") or raw.get("units"),
94
+ )
95
+ )
96
+ return streams
97
+
98
+ def _read_episodes(self) -> list[Episode]:
99
+ paths = sorted((self.root / "meta" / "episodes").glob("**/*.parquet"))
100
+ if not paths:
101
+ raise AdapterError("missing LeRobot v3 episode metadata parquet files")
102
+ records: list[dict[str, Any]] = []
103
+ try:
104
+ for path in paths:
105
+ table = pq.read_table(path)
106
+ records.extend(table.to_pylist())
107
+ except (OSError, pa.ArrowException) as exc:
108
+ if "Repetition level histogram size mismatch" in str(exc):
109
+ raise AdapterError(
110
+ "cannot read episode metadata because PyArrow 19.0.0 has a Parquet repetition-level "
111
+ "reader bug; upgrade PyArrow to >=19.0.1 and rerun the validation"
112
+ ) from exc
113
+ raise AdapterError(f"cannot read episode metadata: {exc}") from exc
114
+ episodes = [self._episode_from_record(record) for record in records]
115
+ return sorted(episodes, key=lambda episode: episode.index)
116
+
117
+ def _episode_from_record(self, record: dict[str, Any]) -> Episode:
118
+ try:
119
+ index = int(record["episode_index"])
120
+ length = int(record["length"])
121
+ except (KeyError, TypeError, ValueError) as exc:
122
+ raise AdapterError("episode metadata lacks a valid episode_index or length") from exc
123
+ data_file: str | None = None
124
+ if "data/chunk_index" in record and "data/file_index" in record:
125
+ data_file = self._format_path(
126
+ str(self._info.get("data_path", DEFAULT_DATA_PATH)),
127
+ chunk_index=int(record["data/chunk_index"]),
128
+ file_index=int(record["data/file_index"]),
129
+ )
130
+ video_files: dict[str, str] = {}
131
+ video_ranges: dict[str, tuple[float, float]] = {}
132
+ for stream in self._features:
133
+ if stream.kind != "video":
134
+ continue
135
+ prefix = f"videos/{stream.key}"
136
+ if f"{prefix}/chunk_index" in record and f"{prefix}/file_index" in record:
137
+ video_files[stream.key] = self._format_path(
138
+ str(self._info.get("video_path", DEFAULT_VIDEO_PATH)),
139
+ video_key=stream.key,
140
+ chunk_index=int(record[f"{prefix}/chunk_index"]),
141
+ file_index=int(record[f"{prefix}/file_index"]),
142
+ )
143
+ start = float(record.get(f"{prefix}/from_timestamp", 0.0))
144
+ end = float(record.get(f"{prefix}/to_timestamp", start + length / self._info["fps"]))
145
+ video_ranges[stream.key] = (start, end)
146
+ tasks = record.get("tasks") or []
147
+ if isinstance(tasks, str):
148
+ tasks = [tasks]
149
+ return Episode(
150
+ index=index,
151
+ identifier=str(record.get("episode_id", f"episode_{index:06d}")),
152
+ length=length,
153
+ data_file=data_file,
154
+ from_index=_optional_int(record.get("dataset_from_index")),
155
+ to_index=_optional_int(record.get("dataset_to_index")),
156
+ tasks=[str(task) for task in tasks],
157
+ video_files=video_files,
158
+ video_ranges=video_ranges,
159
+ )
160
+
161
+ @staticmethod
162
+ def _format_path(template: str, **values: Any) -> str:
163
+ try:
164
+ return template.format(**values)
165
+ except (KeyError, ValueError) as exc:
166
+ raise AdapterError(f"invalid path template in info.json: {template}") from exc
167
+
168
+ def _data_path(self, episode: Episode) -> Path:
169
+ if episode.data_file is None:
170
+ raise AdapterError(f"episode {episode.identifier} has no data-file reference")
171
+ path = self.root / episode.data_file
172
+ if not path.is_file():
173
+ raise AdapterError(f"episode data file is missing: {path}")
174
+ return path
175
+
176
+ def parquet_schema(self, episode: Episode) -> dict[str, Any]:
177
+ path = self._data_path(episode)
178
+ if path not in self._schema_cache:
179
+ try:
180
+ self._schema_cache[path] = pq.read_schema(path)
181
+ except (OSError, pa.ArrowException) as exc:
182
+ raise AdapterError(f"cannot read parquet schema {path}: {exc}") from exc
183
+ return {field.name: field.type for field in self._schema_cache[path]}
184
+
185
+ def iter_batches(
186
+ self,
187
+ episode: Episode,
188
+ columns: tuple[str, ...] | None = None,
189
+ batch_size: int = 65_536,
190
+ ) -> Iterator[SampleBatch]:
191
+ path = self._data_path(episode)
192
+ try:
193
+ parquet = pq.ParquetFile(path)
194
+ available = set(parquet.schema_arrow.names)
195
+ requested = list(columns or tuple(available))
196
+ selector = "episode_index" if "episode_index" in available else "index"
197
+ read_columns = list(dict.fromkeys([*requested, selector]))
198
+ missing = set(read_columns) - available
199
+ if missing:
200
+ raise AdapterError(f"columns missing from {path}: {', '.join(sorted(missing))}")
201
+ for raw in parquet.iter_batches(batch_size=batch_size, columns=read_columns):
202
+ selection = raw.column(raw.schema.get_field_index(selector)).to_numpy(zero_copy_only=False)
203
+ if selector == "episode_index":
204
+ mask = selection == episode.index
205
+ elif episode.from_index is not None and episode.to_index is not None:
206
+ mask = (selection >= episode.from_index) & (selection < episode.to_index)
207
+ else:
208
+ mask = np.ones(len(selection), dtype=bool)
209
+ if not np.any(mask):
210
+ continue
211
+ converted: dict[str, np.ndarray] = {}
212
+ for name in requested:
213
+ array = raw.column(raw.schema.get_field_index(name))
214
+ converted[name] = _arrow_to_numpy(array)[mask]
215
+ yield SampleBatch(episode=episode, columns=converted)
216
+ except AdapterError:
217
+ raise
218
+ except (OSError, pa.ArrowException) as exc:
219
+ raise AdapterError(f"cannot stream parquet data {path}: {exc}") from exc
220
+
221
+ def iter_video_frames(self, episode: Episode, stream: str, stride: int = 1) -> Iterator[VideoFrame]:
222
+ try:
223
+ import cv2
224
+ except ImportError as exc: # pragma: no cover - environment dependent
225
+ raise AdapterError("video checks require the 'video' optional dependency") from exc
226
+ if stream not in episode.video_files:
227
+ raise AdapterError(f"episode {episode.identifier} has no {stream} video reference")
228
+ path = self.root / episode.video_files[stream]
229
+ capture = cv2.VideoCapture(str(path))
230
+ if not capture.isOpened():
231
+ capture.release()
232
+ raise AdapterError(f"cannot decode video: {path}")
233
+ fps = float(capture.get(cv2.CAP_PROP_FPS)) or self.inventory.fps
234
+ start, end = episode.video_ranges.get(stream, (0.0, episode.length / fps))
235
+ start_frame = max(0, int(round(start * fps)))
236
+ end_frame = max(start_frame, int(round(end * fps)))
237
+ capture.set(cv2.CAP_PROP_POS_FRAMES, start_frame)
238
+ try:
239
+ for frame_index in range(start_frame, end_frame):
240
+ ok, image = capture.read()
241
+ if not ok:
242
+ break
243
+ relative = frame_index - start_frame
244
+ if relative % max(1, stride):
245
+ continue
246
+ yield VideoFrame(
247
+ episode=episode,
248
+ stream=stream,
249
+ frame_index=relative,
250
+ timestamp=relative / fps,
251
+ image=image,
252
+ )
253
+ finally:
254
+ capture.release()
255
+
256
+ def video_analysis(self, episode: Episode, stream: str) -> VideoAnalysis:
257
+ """Decode once and cache small, non-image statistics for every video rule."""
258
+ key = (episode.index, stream)
259
+ cached = self._video_analysis_cache.get(key)
260
+ if cached is not None:
261
+ return cached
262
+
263
+ frame_indices: list[int] = []
264
+ timestamps: list[float] = []
265
+ means: list[float] = []
266
+ stddevs: list[float] = []
267
+ differences: list[float] = []
268
+ previous: np.ndarray | None = None
269
+ for frame in self.iter_video_frames(episode, stream):
270
+ small = _small_grayscale(frame.image)
271
+ frame_indices.append(frame.frame_index)
272
+ timestamps.append(frame.timestamp)
273
+ means.append(float(np.mean(small)))
274
+ stddevs.append(float(np.std(small)))
275
+ if previous is not None:
276
+ differences.append(float(np.mean(np.abs(small - previous))))
277
+ previous = small
278
+ analysis = VideoAnalysis(
279
+ frame_indices=np.asarray(frame_indices, dtype=np.int64),
280
+ timestamps=np.asarray(timestamps, dtype=float),
281
+ mean_intensities=np.asarray(means, dtype=float),
282
+ stddevs=np.asarray(stddevs, dtype=float),
283
+ mean_absolute_differences=np.asarray(differences, dtype=float),
284
+ )
285
+ self._video_analysis_cache[key] = analysis
286
+ return analysis
287
+
288
+
289
+ def _optional_int(value: Any) -> int | None:
290
+ return None if value is None else int(value)
291
+
292
+
293
+ def _small_grayscale(image: np.ndarray) -> np.ndarray:
294
+ """Spatially sample first, avoiding full-resolution color reductions."""
295
+ row_stride = max(1, image.shape[0] // 64)
296
+ col_stride = max(1, image.shape[1] // 64)
297
+ small = image[::row_stride, ::col_stride]
298
+ gray = np.mean(small, axis=2) if small.ndim == 3 else small
299
+ return np.asarray(gray, dtype=np.float32)
300
+
301
+
302
+ def _hugging_face_identity(root: Path) -> tuple[str | None, str | None]:
303
+ """Recover repo/revision from a Hugging Face snapshot cache path."""
304
+ parts = root.resolve().parts
305
+ for index, part in enumerate(parts):
306
+ if not part.startswith("datasets--") or index + 2 >= len(parts):
307
+ continue
308
+ if parts[index + 1] != "snapshots":
309
+ continue
310
+ encoded = part.removeprefix("datasets--")
311
+ namespace, separator, repository = encoded.partition("--")
312
+ if separator and namespace and repository:
313
+ return f"{namespace}/{repository}", parts[index + 2]
314
+ return None, None
315
+
316
+
317
+ def _arrow_to_numpy(array: pa.Array) -> np.ndarray:
318
+ """Preserve vector columns as a dense array when their shape is regular."""
319
+ values = array.to_pylist()
320
+ try:
321
+ return np.asarray(values)
322
+ except ValueError:
323
+ return np.asarray(values, dtype=object)
physlint/api.py ADDED
@@ -0,0 +1,29 @@
1
+ """Stable library entry points independent of terminal behavior."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+
7
+ from physlint.config import Config, load_config
8
+ from physlint.engine.discovery import discover
9
+ from physlint.engine.runner import run_validation
10
+ from physlint.models.dataset import DatasetInventory
11
+ from physlint.models.finding import Report
12
+
13
+
14
+ def inspect_dataset(path: str | Path, *, adapter: str = "auto") -> DatasetInventory:
15
+ return discover(path, adapter).inventory
16
+
17
+
18
+ def check_dataset(
19
+ path: str | Path,
20
+ *,
21
+ config: Config | None = None,
22
+ config_path: str | Path | None = None,
23
+ ) -> Report:
24
+ root = Path(path).expanduser().resolve()
25
+ resolved_config = config or load_config(
26
+ Path(config_path).expanduser().resolve() if config_path is not None else None, root
27
+ )
28
+ dataset = discover(root, resolved_config.adapter)
29
+ return run_validation(dataset, resolved_config)
physlint/cli.py ADDED
@@ -0,0 +1,174 @@
1
+ """Physlint command-line interface and frozen exit codes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from datetime import UTC
7
+ from enum import StrEnum
8
+ from pathlib import Path
9
+ from typing import Annotated
10
+
11
+ import typer
12
+ from rich.console import Console
13
+
14
+ from physlint import __version__
15
+ from physlint.adapters.base import AdapterError
16
+ from physlint.config import DEFAULT_CONFIG, ConfigurationError, load_config
17
+ from physlint.engine.discovery import discover
18
+ from physlint.engine.runner import run_validation
19
+ from physlint.reporters.json import write_json_report
20
+ from physlint.reporters.terminal import (
21
+ render_explanation,
22
+ render_inventory,
23
+ render_report,
24
+ render_rules,
25
+ )
26
+ from physlint.rules import BUILTIN_RULES
27
+
28
+ app = typer.Typer(
29
+ name="physlint",
30
+ help="Find concrete integrity defects in physical-AI datasets.",
31
+ no_args_is_help=True,
32
+ pretty_exceptions_enable=False,
33
+ )
34
+ console = Console()
35
+ error_console = Console(stderr=True)
36
+
37
+
38
+ class OutputFormat(StrEnum):
39
+ TERMINAL = "terminal"
40
+ JSON = "json"
41
+
42
+
43
+ def _version_callback(value: bool) -> None:
44
+ if value:
45
+ typer.echo(__version__)
46
+ raise typer.Exit(0)
47
+
48
+
49
+ @app.callback()
50
+ def main(
51
+ version: Annotated[
52
+ bool,
53
+ typer.Option("--version", callback=_version_callback, is_eager=True, help="Show version."),
54
+ ] = False,
55
+ ) -> None:
56
+ """Validate robot-learning data before it reaches training."""
57
+
58
+
59
+ @app.command()
60
+ def check(
61
+ path: Annotated[Path, typer.Argument(help="Local LeRobot v3 dataset directory.")],
62
+ config: Annotated[Path | None, typer.Option("--config", "-c", help="Quality contract YAML file.")] = None,
63
+ output: Annotated[
64
+ OutputFormat, typer.Option("--output", "-o", help="Terminal or JSON stdout output.")
65
+ ] = OutputFormat.TERMINAL,
66
+ json_output: Annotated[Path | None, typer.Option("--json-output", help="Exact JSON report destination.")] = None,
67
+ ) -> None:
68
+ """Validate a dataset and return a CI-safe pass/fail exit code."""
69
+ try:
70
+ root = path.expanduser().resolve()
71
+ settings = load_config(config.expanduser().resolve() if config else None, root)
72
+ dataset = discover(root, settings.adapter)
73
+ report = run_validation(dataset, settings)
74
+ report_path: Path | None = None
75
+ if json_output is not None:
76
+ report_path = write_json_report(report, json_output)
77
+ elif settings.reports.json_enabled:
78
+ timestamp = report.started_at.astimezone(UTC).strftime("%Y-%m-%dT%H%M%SZ")
79
+ report_path = write_json_report(report, Path(settings.reports.output_dir) / f"{timestamp}.json")
80
+ if output == OutputFormat.JSON:
81
+ typer.echo(report.model_dump_json(indent=2))
82
+ else:
83
+ render_report(report, report_path, console)
84
+ adapter_errors = any(result.error_kind == "adapter" for result in report.results)
85
+ rule_errors = any(result.error_kind == "rule" for result in report.results)
86
+ if adapter_errors:
87
+ raise typer.Exit(3)
88
+ if rule_errors:
89
+ raise typer.Exit(4)
90
+ if report.status == "failed":
91
+ raise typer.Exit(1)
92
+ except KeyboardInterrupt as exc:
93
+ raise typer.Exit(130) from exc
94
+ except ConfigurationError as exc:
95
+ error_console.print(f"[red]Configuration error:[/red] {exc}")
96
+ raise typer.Exit(2) from exc
97
+ except AdapterError as exc:
98
+ error_console.print(f"[red]Dataset error:[/red] {exc}")
99
+ raise typer.Exit(3) from exc
100
+ except typer.Exit:
101
+ raise
102
+ except Exception as exc: # noqa: BLE001 - CLI must preserve the documented code
103
+ error_console.print(f"[red]Internal Physlint error:[/red] {type(exc).__name__}: {exc}")
104
+ raise typer.Exit(4) from exc
105
+
106
+
107
+ @app.command()
108
+ def inspect(
109
+ path: Annotated[Path, typer.Argument(help="Local LeRobot v3 dataset directory.")],
110
+ output_json: Annotated[bool, typer.Option("--json", help="Print the inventory as JSON.")] = False,
111
+ ) -> None:
112
+ """Show streams, schemas, rates, and episode inventory."""
113
+ try:
114
+ inventory = discover(path).inventory
115
+ if output_json:
116
+ typer.echo(inventory.model_dump_json(indent=2))
117
+ else:
118
+ render_inventory(inventory, console)
119
+ except AdapterError as exc:
120
+ error_console.print(f"[red]Dataset error:[/red] {exc}")
121
+ raise typer.Exit(3) from exc
122
+ except KeyboardInterrupt as exc:
123
+ raise typer.Exit(130) from exc
124
+
125
+
126
+ @app.command("rules")
127
+ def list_rules(
128
+ output_json: Annotated[bool, typer.Option("--json", help="Print rule metadata as JSON.")] = False,
129
+ ) -> None:
130
+ """List built-in rules and their requirements."""
131
+ metadata = [rule.metadata for rule in BUILTIN_RULES]
132
+ if output_json:
133
+ typer.echo(
134
+ json.dumps(
135
+ [
136
+ {
137
+ **item.__dict__,
138
+ "severity": item.severity.value,
139
+ "required_capabilities": sorted(item.required_capabilities),
140
+ "required_streams": sorted(item.required_streams),
141
+ }
142
+ for item in metadata
143
+ ],
144
+ indent=2,
145
+ )
146
+ )
147
+ else:
148
+ render_rules(metadata, console)
149
+
150
+
151
+ @app.command()
152
+ def explain(rule_id: Annotated[str, typer.Argument(help="Stable rule ID.")]) -> None:
153
+ """Explain one rule, including remediation and limitations."""
154
+ for rule in BUILTIN_RULES:
155
+ if rule.metadata.id == rule_id:
156
+ render_explanation(rule.metadata, console)
157
+ return
158
+ error_console.print(f"[red]Unknown rule ID:[/red] {rule_id}")
159
+ raise typer.Exit(2)
160
+
161
+
162
+ @app.command()
163
+ def init(
164
+ path: Annotated[Path, typer.Option("--path", "-p", help="Configuration file to create.")] = Path("physlint.yaml"),
165
+ force: Annotated[bool, typer.Option("--force", help="Overwrite an existing file.")] = False,
166
+ ) -> None:
167
+ """Generate a documented quality-contract configuration."""
168
+ destination = path.expanduser().resolve()
169
+ if destination.exists() and not force:
170
+ error_console.print(f"[red]Refusing to overwrite existing file:[/red] {destination}")
171
+ raise typer.Exit(2)
172
+ destination.parent.mkdir(parents=True, exist_ok=True)
173
+ destination.write_text(DEFAULT_CONFIG, encoding="utf-8")
174
+ console.print(f"Created {destination}")
physlint/config.py ADDED
@@ -0,0 +1,89 @@
1
+ """Strict project configuration and quality-contract loading."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from pathlib import Path
8
+ from typing import Any, Literal
9
+
10
+ import yaml
11
+ from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator
12
+
13
+
14
+ class RuleSettings(BaseModel):
15
+ model_config = ConfigDict(extra="forbid")
16
+
17
+ enabled: bool = True
18
+ severity: Literal["critical", "error", "warning", "notice"] | None = None
19
+ options: dict[str, Any] = Field(default_factory=dict)
20
+
21
+
22
+ class ReportSettings(BaseModel):
23
+ model_config = ConfigDict(extra="forbid", populate_by_name=True)
24
+
25
+ json_enabled: bool = Field(default=True, alias="json")
26
+ output_dir: str = ".physlint/reports"
27
+
28
+
29
+ class Config(BaseModel):
30
+ model_config = ConfigDict(extra="forbid")
31
+
32
+ config_version: Literal[1] = 1
33
+ adapter: Literal["auto", "lerobot"] = "auto"
34
+ required_streams: list[str] = Field(default_factory=lambda: ["observation.state", "action"])
35
+ fail_on: Literal["critical", "error", "warning", "notice"] = "error"
36
+ rules: dict[str, RuleSettings] = Field(default_factory=dict)
37
+ reports: ReportSettings = Field(default_factory=ReportSettings)
38
+
39
+ @field_validator("required_streams")
40
+ @classmethod
41
+ def unique_streams(cls, value: list[str]) -> list[str]:
42
+ if len(value) != len(set(value)):
43
+ raise ValueError("required_streams must not contain duplicates")
44
+ return value
45
+
46
+ def digest(self) -> str:
47
+ data = json.dumps(self.model_dump(mode="json", by_alias=True), sort_keys=True, separators=(",", ":"))
48
+ return hashlib.sha256(data.encode()).hexdigest()
49
+
50
+
51
+ class ConfigurationError(ValueError):
52
+ """Invalid or unreadable user configuration."""
53
+
54
+
55
+ def load_config(path: Path | None, dataset_path: Path | None = None) -> Config:
56
+ candidate = path
57
+ if candidate is None and dataset_path is not None:
58
+ local = dataset_path / "physlint.yaml"
59
+ cwd = Path.cwd() / "physlint.yaml"
60
+ candidate = local if local.is_file() else cwd if cwd.is_file() else None
61
+ if candidate is None:
62
+ return Config()
63
+ try:
64
+ payload = yaml.safe_load(candidate.read_text(encoding="utf-8")) or {}
65
+ if not isinstance(payload, dict):
66
+ raise ConfigurationError("configuration root must be a mapping")
67
+ return Config.model_validate(payload)
68
+ except (OSError, yaml.YAMLError, ValidationError) as exc:
69
+ raise ConfigurationError(f"invalid configuration {candidate}: {exc}") from exc
70
+
71
+
72
+ DEFAULT_CONFIG = """# Physlint quality contract
73
+ config_version: 1
74
+ adapter: auto
75
+ required_streams:
76
+ - observation.state
77
+ - action
78
+ fail_on: error
79
+ rules:
80
+ temporal.max_gap:
81
+ options:
82
+ max_gap_multiplier: 2.0
83
+ video.frozen_frames:
84
+ options:
85
+ max_consecutive_frames: 5
86
+ reports:
87
+ json: true
88
+ output_dir: .physlint/reports
89
+ """