hflow 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {hflow-0.2.2 → hflow-0.2.4}/PKG-INFO +37 -16
  2. {hflow-0.2.2 → hflow-0.2.4}/README.md +32 -11
  3. {hflow-0.2.2 → hflow-0.2.4}/pyproject.toml +21 -13
  4. {hflow-0.2.2 → hflow-0.2.4}/pyproject.toml.orig +24 -16
  5. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/__init__.py +16 -1
  6. hflow-0.2.4/src/hflow/__main__.py +3 -0
  7. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_grouped_mcap_writer/_writer.py +16 -0
  8. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_video_measurements/_camera_motion.py +3 -0
  9. hflow-0.2.4/src/hflow/_video_measurements/_field_guards.py +30 -0
  10. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_video_measurements/_frame_statistics.py +16 -1
  11. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/app.py +215 -44
  12. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/batching.py +29 -4
  13. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/behavior.py +11 -1
  14. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/catalog.py +164 -13
  15. hflow-0.2.4/src/hflow/catalog_ui.py +152 -0
  16. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/checks.py +703 -291
  17. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/cli.py +187 -2
  18. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/curation.py +144 -8
  19. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/doctor.py +115 -20
  20. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/episode.py +200 -2
  21. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/ffmpeg/_binary.py +1 -1
  22. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/ffmpeg/_contact_sheet.py +11 -3
  23. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/format.py +7 -0
  24. hflow-0.2.4/src/hflow/importers/__init__.py +5 -0
  25. hflow-0.2.4/src/hflow/importers/lerobot.py +926 -0
  26. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/reader.py +3 -1
  27. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_bundle.py +25 -9
  28. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_client.py +2 -0
  29. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_templates.py +89 -10
  30. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/snapshot.py +16 -0
  31. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/stage_execution.py +26 -0
  32. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/stage_planning.py +128 -13
  33. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/steps.py +39 -2
  34. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/storage.py +8 -1
  35. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/testing.py +82 -0
  36. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/transform.py +142 -29
  37. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/video.py +120 -10
  38. {hflow-0.2.2 → hflow-0.2.4}/LICENSE +0 -0
  39. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_grouped_mcap_writer/__init__.py +0 -0
  40. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_pinned_asset.py +0 -0
  41. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_video_measurement_toolchain.py +0 -0
  42. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_video_measurements/__init__.py +0 -0
  43. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_video_measurements/_raw_frames.py +0 -0
  44. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/_video_measurements/_toolchain.py +0 -0
  45. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/dataset.py +0 -0
  46. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/ffmpeg/__init__.py +0 -0
  47. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/ingest_ledger.py +0 -0
  48. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/manifest.py +0 -0
  49. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/mediapipe_hands.py +0 -0
  50. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/project.py +0 -0
  51. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/providers.py +0 -0
  52. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/py.typed +0 -0
  53. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/resample.py +0 -0
  54. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/__init__.py +0 -0
  55. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_compose.py +0 -0
  56. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_deploy.py +0 -0
  57. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_endpoint.py +0 -0
  58. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_lifecycle.py +0 -0
  59. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/runtime/_topology.py +0 -0
  60. {hflow-0.2.2 → hflow-0.2.4}/src/hflow/workspace.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hflow
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: Open source SDK for building Physical AI data pipelines
5
5
  Keywords: airflow,data-pipeline,data-quality,dataset-curation,mcap,physical-ai,robotics
6
6
  Author: Hebbian Robotics, Kingston Kuan, Brandon Ong
@@ -14,10 +14,10 @@ Requires-Dist: mcap-ros2-support>=0.5.7
14
14
  Requires-Dist: numpy>=2.4.6
15
15
  Requires-Dist: zstandard>=0.25.0
16
16
  Requires-Dist: pyarrow>=25.0.1 ; extra == 'arrow'
17
- Requires-Dist: obstore>=0.10.0 ; extra == 'bucket'
18
- Requires-Dist: mediapipe>=1.0.0 ; extra == 'mediapipe'
19
- Requires-Dist: opencv-python-headless>=4.10 ; extra == 'motion'
20
- Requires-Dist: openai>=3.0.0 ; extra == 'openai'
17
+ Requires-Dist: obstore>=0.11.1 ; extra == 'bucket'
18
+ Requires-Dist: mediapipe>=1.0.1 ; extra == 'mediapipe'
19
+ Requires-Dist: opencv-python-headless>=5.0.0.93 ; extra == 'motion'
20
+ Requires-Dist: openai>=3.3.1 ; extra == 'openai'
21
21
  Requires-Python: >=3.11
22
22
  Project-URL: Repository, https://github.com/Hebbian-Robotics/hflow
23
23
  Project-URL: Documentation, https://github.com/Hebbian-Robotics/hflow/tree/main/docs
@@ -87,7 +87,7 @@ episode.
87
87
 
88
88
  | | HFlow's boundary |
89
89
  | --- | --- |
90
- | **Input** | One multimodal episode per standard MCAP file (`hflow doctor` says whether yours qualifies) |
90
+ | **Input** | Supported standard MCAP episodes directly; LeRobot Dataset v3 repositories through `hflow import lerobot` |
91
91
  | **Processing** | Your Python transforms, checks, labels, and enrichments |
92
92
  | **Execution** | In-process for development; generated Airflow 3 DAGs for scheduled runs |
93
93
  | **Durable output** | Canonical MCAP episodes, provenance, artifacts, and a Parquet catalog |
@@ -115,6 +115,13 @@ collection --> ingestion ---------------> curation ------> delivery
115
115
  - **Quality checks produce reusable evidence.** Accessors extract the inputs existing processing code expects (numpy arrays, MP4 paths, JPEG frames), and results land as queryable measurements rather than hardcoded verdicts. Different datasets can apply different thresholds without processing the media again.
116
116
  - **Query the corpus without loading the recordings.** Metadata, quality measurements, tags, version stamps, and artifact locations live in the Parquet catalog. [DuckDB](https://duckdb.org/) can answer corpus-wide questions and build manifests without opening the underlying MCAP files.
117
117
 
118
+ Open DuckDB's browser over the catalog at any time, including before the first
119
+ run starts:
120
+
121
+ ```bash
122
+ hflow catalog ui
123
+ ```
124
+
118
125
  ## Hosting and scale
119
126
 
120
127
  The open-source deployment is built to be easy to own: run one single-tenant
@@ -135,10 +142,9 @@ the trust model, and the current limits.
135
142
 
136
143
  ## Community and hosted interest
137
144
 
138
- <!-- Add the Google Form link to the entry below once it is live. -->
139
-
140
- - **Hosted version interest:** Google Form coming soon.
145
+ - **Hosted platform waitlist:** [tell us about your workflow](https://forms.gle/EZpQpGGF3eJomx498).
141
146
  - **Community Discord:** [join us](https://discord.gg/vacepQvjmg) for questions, feedback, and contribution discussion.
147
+ - **Code of conduct:** review our [community standards](./CODE_OF_CONDUCT.md) and report concerns privately.
142
148
 
143
149
  For reproducible bugs and scoped feature requests, use
144
150
  [GitHub issues](https://github.com/Hebbian-Robotics/hflow/issues).
@@ -179,6 +185,19 @@ and contributors should start with [CONTRIBUTING.md](https://github.com/Hebbian-
179
185
  the [examples catalog](https://github.com/Hebbian-Robotics/hflow/blob/main/examples/README.md) for the egocentric-corpus and
180
186
  OpenAI vision paths.
181
187
 
188
+ To import a LeRobot Dataset v3 episode into the same canonical MCAP boundary:
189
+
190
+ ```bash
191
+ uv run hflow import lerobot \
192
+ --repo lerobot/pusht --revision main \
193
+ --camera observation.image --episode-index 0 \
194
+ --output-dir ./data/lerobot_pusht
195
+ ```
196
+
197
+ The importer resolves `main` to an immutable source commit and records it as
198
+ episode provenance. See the [LeRobot import guide](./docs/how-to/import-lerobot-v3.md)
199
+ for the supported feature subset and a multi-camera example.
200
+
182
201
  ## What it looks like
183
202
 
184
203
  Get started in six lines of code. This fuller example uses a robot
@@ -233,19 +252,13 @@ WHERE task = 'fold_napkin'
233
252
  AND pipeline_version = 'a41c9f27b3d8' -- pin one reprocessing generation
234
253
  ```
235
254
 
236
- ## Design tenets
255
+ ## Design principles
237
256
 
238
257
  1. **Democratize the architecture, defer the optimizations.** Preserve the useful workflow and standard interfaces at small scale, and label each production-scale mechanism honestly as implemented, simplified, deferred, or out of scope.
239
258
  2. **Evidence, not verdicts.** Checks record measurements with coverage; pass/fail policy belongs to the consumer, at curation time. Quality tags route episodes; they never delete data.
240
259
  3. **Standard formats at every boundary.** MCAP episodes, Parquet catalogs, Airflow DAGs. Our code exists only where the format forces bridging or a pitfall is genuinely non-obvious.
241
260
  4. **Your code stays your code.** Existing transforms, checks, and enrichments plug in through small adapters instead of being rewritten.
242
261
 
243
- ## Non-goals
244
-
245
- - **Training.** The pipeline ends at curated, quality-tagged, version-stamped episodes and a manifest. Many users filter data to deliver or sell it, not to train on it. (Converters to training formats such as [LeRobot](https://github.com/huggingface/lerobot) are planned as a separate, standalone package.)
246
- - **Maximum flexibility.** Robotics/physical-AI data is the narrative and the constraint budget: one canonical episode format, coarse-grained steps, and opinionated defaults are features.
247
- - **Million-hour throughput.** The [benchmark report](https://github.com/Hebbian-Robotics/hflow/blob/main/docs/BENCHMARKS.md) documents honestly what the simple version achieves and where it falls over.
248
-
249
262
  ## Requirements
250
263
 
251
264
  - Python ≥ 3.11
@@ -278,6 +291,14 @@ WHERE task = 'fold_napkin'
278
291
  - [FFmpeg](https://ffmpeg.org/)
279
292
  - [Pareto](https://github.com/Hebbian-Robotics/pareto), Hebbian Robotics' robotics data curation platform.
280
293
 
294
+ ## Contributing
295
+
296
+ Thank you to all our contributors for making HFlow awesome! See [CONTRIBUTING.md](https://github.com/Hebbian-Robotics/hflow/blob/main/CONTRIBUTING.md) to join our community.
297
+
298
+ <a href="https://github.com/Hebbian-Robotics/hflow/graphs/contributors">
299
+ <img src="https://contrib.rocks/image?repo=Hebbian-Robotics/hflow&max=48&columns=12" alt="HFlow contributors" />
300
+ </a>
301
+
281
302
  ## License
282
303
 
283
304
  [Apache-2.0](https://github.com/Hebbian-Robotics/hflow/blob/main/LICENSE).
@@ -56,7 +56,7 @@ episode.
56
56
 
57
57
  | | HFlow's boundary |
58
58
  | --- | --- |
59
- | **Input** | One multimodal episode per standard MCAP file (`hflow doctor` says whether yours qualifies) |
59
+ | **Input** | Supported standard MCAP episodes directly; LeRobot Dataset v3 repositories through `hflow import lerobot` |
60
60
  | **Processing** | Your Python transforms, checks, labels, and enrichments |
61
61
  | **Execution** | In-process for development; generated Airflow 3 DAGs for scheduled runs |
62
62
  | **Durable output** | Canonical MCAP episodes, provenance, artifacts, and a Parquet catalog |
@@ -84,6 +84,13 @@ collection --> ingestion ---------------> curation ------> delivery
84
84
  - **Quality checks produce reusable evidence.** Accessors extract the inputs existing processing code expects (numpy arrays, MP4 paths, JPEG frames), and results land as queryable measurements rather than hardcoded verdicts. Different datasets can apply different thresholds without processing the media again.
85
85
  - **Query the corpus without loading the recordings.** Metadata, quality measurements, tags, version stamps, and artifact locations live in the Parquet catalog. [DuckDB](https://duckdb.org/) can answer corpus-wide questions and build manifests without opening the underlying MCAP files.
86
86
 
87
+ Open DuckDB's browser over the catalog at any time, including before the first
88
+ run starts:
89
+
90
+ ```bash
91
+ hflow catalog ui
92
+ ```
93
+
87
94
  ## Hosting and scale
88
95
 
89
96
  The open-source deployment is built to be easy to own: run one single-tenant
@@ -104,10 +111,9 @@ the trust model, and the current limits.
104
111
 
105
112
  ## Community and hosted interest
106
113
 
107
- <!-- Add the Google Form link to the entry below once it is live. -->
108
-
109
- - **Hosted version interest:** Google Form coming soon.
114
+ - **Hosted platform waitlist:** [tell us about your workflow](https://forms.gle/EZpQpGGF3eJomx498).
110
115
  - **Community Discord:** [join us](https://discord.gg/vacepQvjmg) for questions, feedback, and contribution discussion.
116
+ - **Code of conduct:** review our [community standards](./CODE_OF_CONDUCT.md) and report concerns privately.
111
117
 
112
118
  For reproducible bugs and scoped feature requests, use
113
119
  [GitHub issues](https://github.com/Hebbian-Robotics/hflow/issues).
@@ -148,6 +154,19 @@ and contributors should start with [CONTRIBUTING.md](https://github.com/Hebbian-
148
154
  the [examples catalog](https://github.com/Hebbian-Robotics/hflow/blob/main/examples/README.md) for the egocentric-corpus and
149
155
  OpenAI vision paths.
150
156
 
157
+ To import a LeRobot Dataset v3 episode into the same canonical MCAP boundary:
158
+
159
+ ```bash
160
+ uv run hflow import lerobot \
161
+ --repo lerobot/pusht --revision main \
162
+ --camera observation.image --episode-index 0 \
163
+ --output-dir ./data/lerobot_pusht
164
+ ```
165
+
166
+ The importer resolves `main` to an immutable source commit and records it as
167
+ episode provenance. See the [LeRobot import guide](./docs/how-to/import-lerobot-v3.md)
168
+ for the supported feature subset and a multi-camera example.
169
+
151
170
  ## What it looks like
152
171
 
153
172
  Get started in six lines of code. This fuller example uses a robot
@@ -202,19 +221,13 @@ WHERE task = 'fold_napkin'
202
221
  AND pipeline_version = 'a41c9f27b3d8' -- pin one reprocessing generation
203
222
  ```
204
223
 
205
- ## Design tenets
224
+ ## Design principles
206
225
 
207
226
  1. **Democratize the architecture, defer the optimizations.** Preserve the useful workflow and standard interfaces at small scale, and label each production-scale mechanism honestly as implemented, simplified, deferred, or out of scope.
208
227
  2. **Evidence, not verdicts.** Checks record measurements with coverage; pass/fail policy belongs to the consumer, at curation time. Quality tags route episodes; they never delete data.
209
228
  3. **Standard formats at every boundary.** MCAP episodes, Parquet catalogs, Airflow DAGs. Our code exists only where the format forces bridging or a pitfall is genuinely non-obvious.
210
229
  4. **Your code stays your code.** Existing transforms, checks, and enrichments plug in through small adapters instead of being rewritten.
211
230
 
212
- ## Non-goals
213
-
214
- - **Training.** The pipeline ends at curated, quality-tagged, version-stamped episodes and a manifest. Many users filter data to deliver or sell it, not to train on it. (Converters to training formats such as [LeRobot](https://github.com/huggingface/lerobot) are planned as a separate, standalone package.)
215
- - **Maximum flexibility.** Robotics/physical-AI data is the narrative and the constraint budget: one canonical episode format, coarse-grained steps, and opinionated defaults are features.
216
- - **Million-hour throughput.** The [benchmark report](https://github.com/Hebbian-Robotics/hflow/blob/main/docs/BENCHMARKS.md) documents honestly what the simple version achieves and where it falls over.
217
-
218
231
  ## Requirements
219
232
 
220
233
  - Python ≥ 3.11
@@ -247,6 +260,14 @@ WHERE task = 'fold_napkin'
247
260
  - [FFmpeg](https://ffmpeg.org/)
248
261
  - [Pareto](https://github.com/Hebbian-Robotics/pareto), Hebbian Robotics' robotics data curation platform.
249
262
 
263
+ ## Contributing
264
+
265
+ Thank you to all our contributors for making HFlow awesome! See [CONTRIBUTING.md](https://github.com/Hebbian-Robotics/hflow/blob/main/CONTRIBUTING.md) to join our community.
266
+
267
+ <a href="https://github.com/Hebbian-Robotics/hflow/graphs/contributors">
268
+ <img src="https://contrib.rocks/image?repo=Hebbian-Robotics/hflow&max=48&columns=12" alt="HFlow contributors" />
269
+ </a>
270
+
250
271
  ## License
251
272
 
252
273
  [Apache-2.0](https://github.com/Hebbian-Robotics/hflow/blob/main/LICENSE).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "hflow"
3
- version = "0.2.2"
3
+ version = "0.2.4"
4
4
  description = "Open source SDK for building Physical AI data pipelines"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -36,10 +36,10 @@ name = "Brandon Ong"
36
36
 
37
37
  [project.optional-dependencies]
38
38
  arrow = ["pyarrow>=25.0.1"]
39
- bucket = ["obstore>=0.10.0"]
40
- mediapipe = ["mediapipe>=1.0.0"]
41
- motion = ["opencv-python-headless>=4.10"]
42
- openai = ["openai>=3.0.0"]
39
+ bucket = ["obstore>=0.11.1"]
40
+ mediapipe = ["mediapipe>=1.0.1"]
41
+ motion = ["opencv-python-headless>=5.0.0.93"]
42
+ openai = ["openai>=3.3.1"]
43
43
 
44
44
  [project.urls]
45
45
  Repository = "https://github.com/Hebbian-Robotics/hflow"
@@ -52,25 +52,29 @@ hflow = "hflow.cli:main"
52
52
  [dependency-groups]
53
53
  dev = [
54
54
  "hflow-server",
55
- "httpx2>=2.10",
56
- "obstore>=0.10.0",
57
- "opencv-python-headless>=4.10",
55
+ "httpx2>=2.12.0",
56
+ "obstore>=0.11.1",
57
+ "opencv-python-headless>=5.0.0.93",
58
58
  "pyarrow>=25.0.1",
59
59
  "pytest>=9.1.1",
60
60
  "pyyaml>=6.0.3",
61
- "ruff>=0.16.2",
62
- "ty>=0.0.71",
61
+ "ruff>=0.16.4",
62
+ "ty>=0.0.74",
63
63
  ]
64
64
 
65
65
  [build-system]
66
- requires = ["uv_build>=0.11.33,<0.12"]
66
+ requires = ["uv_build>=0.12.5,<0.13"]
67
67
  build-backend = "uv_build"
68
68
 
69
69
  [tool.uv]
70
70
  exclude-newer = "5 days"
71
71
 
72
72
  [tool.uv.workspace]
73
- members = ["packages/*"]
73
+ members = [
74
+ "packages/*",
75
+ "examples/build_ai_evaluation",
76
+ "examples/egosuite_evaluation",
77
+ ]
74
78
 
75
79
  [tool.uv.sources.hflow-server]
76
80
  workspace = true
@@ -120,4 +124,8 @@ unused-ignore-comment = "error"
120
124
  possibly-unresolved-reference = "error"
121
125
 
122
126
  [tool.ty.src]
123
- exclude = ["examples/openai_vision/"]
127
+ exclude = [
128
+ "examples/build_ai_evaluation/",
129
+ "examples/egosuite_evaluation/",
130
+ "examples/openai_vision/",
131
+ ]
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "hflow"
3
- version = "0.2.2"
3
+ version = "0.2.4"
4
4
  description = "Open source SDK for building Physical AI data pipelines"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -35,21 +35,21 @@ arrow = ["pyarrow>=25.0.1"]
35
35
  # Native object-store data roots (gs://, s3://, az://): obstore is the Rust
36
36
  # object_store crate as a single dependency-free wheel; local-path roots
37
37
  # never import it.
38
- bucket = ["obstore>=0.10.0"]
38
+ bucket = ["obstore>=0.11.1"]
39
39
  # Hand detection with MediaPipe's Hand Landmarker (hflow.mediapipe_hands).
40
40
  # A MODEL, not a signal statistic, so it is opt-in twice over: outside the core
41
41
  # install, and outside the dev group as well (unlike motion and bucket, which
42
42
  # are mirrored into dev so the suite exercises them). It adds ~430 MB and its
43
43
  # own OpenCV; a dev environment opts in with `--extra mediapipe`, exactly as
44
44
  # the openai example does.
45
- mediapipe = ["mediapipe>=1.0.0"]
45
+ mediapipe = ["mediapipe>=1.0.1"]
46
46
  # Camera-shake measurement: pyramidal Lucas-Kanade optical flow and a RANSAC
47
47
  # similarity fit. No numpy-only equivalent preserves the instrument, and
48
48
  # reimplementing them would make it a different, unvalidated one -- so the
49
49
  # dependency is real, and confined to the one check that needs it. Headless
50
50
  # because a data pipeline runs where there is no display.
51
- motion = ["opencv-python-headless>=4.10"]
52
- openai = ["openai>=3.0.0"]
51
+ motion = ["opencv-python-headless>=5.0.0.93"]
52
+ openai = ["openai>=3.3.1"]
53
53
 
54
54
  [project.urls]
55
55
  Repository = "https://github.com/Hebbian-Robotics/hflow"
@@ -64,20 +64,20 @@ dev = [
64
64
  "hflow-server", # the workspace API package, present in dev so the suite tests it
65
65
  # Starlette's TestClient transport for the hflow-server suite. httpx2, not
66
66
  # httpx: starlette 1.6 deprecates the httpx transport in its favour.
67
- "httpx2>=2.10",
68
- "obstore>=0.10.0", # the bucket extra, present in dev so the suite tests it
67
+ "httpx2>=2.12.0",
68
+ "obstore>=0.11.1", # the bucket extra, present in dev so the suite tests it
69
69
  # the motion extra, present in dev so the suite tests it rather than
70
70
  # skipping the one check that cannot be verified any other way
71
- "opencv-python-headless>=4.10",
71
+ "opencv-python-headless>=5.0.0.93",
72
72
  "pyarrow>=25.0.1", # the arrow extra, present in dev so the suite tests it
73
73
  "pytest>=9.1.1",
74
74
  "pyyaml>=6.0.3",
75
- "ruff>=0.16.2",
76
- "ty>=0.0.71",
75
+ "ruff>=0.16.4",
76
+ "ty>=0.0.74",
77
77
  ]
78
78
 
79
79
  [build-system]
80
- requires = ["uv_build>=0.11.33,<0.12"]
80
+ requires = ["uv_build>=0.12.5,<0.13"]
81
81
  build-backend = "uv_build"
82
82
 
83
83
  [tool.uv]
@@ -87,7 +87,11 @@ build-backend = "uv_build"
87
87
  exclude-newer = "5 days"
88
88
 
89
89
  [tool.uv.workspace]
90
- members = ["packages/*"]
90
+ members = [
91
+ "packages/*",
92
+ "examples/build_ai_evaluation",
93
+ "examples/egosuite_evaluation",
94
+ ]
91
95
 
92
96
  [tool.uv.sources]
93
97
  hflow-server = { workspace = true }
@@ -128,7 +132,11 @@ unused-ignore-comment = "error"
128
132
  possibly-unresolved-reference = "error"
129
133
 
130
134
  [tool.ty.src]
131
- # examples/openai_vision imports the user's own `openai` client: bundling a
132
- # VLM client is a non-goal (examples ARE the docs), so the package is
133
- # deliberately not a dependency and the example is excluded from typechecking.
134
- exclude = ["examples/openai_vision/"]
135
+ # Workspace examples typecheck themselves with their own dependencies.
136
+ # openai_vision imports the user's optional client, so the root project does
137
+ # not typecheck it either.
138
+ exclude = [
139
+ "examples/build_ai_evaluation/",
140
+ "examples/egosuite_evaluation/",
141
+ "examples/openai_vision/",
142
+ ]
@@ -20,6 +20,12 @@ from hflow.app import (
20
20
  )
21
21
  from hflow.batching import PlannedBatch, plan_batches, plan_batches_from_files
22
22
  from hflow.catalog import AppendResult, Catalog, CheckRunRow
23
+ from hflow.catalog_ui import (
24
+ DEFAULT_CATALOG_UI_PORT,
25
+ CatalogUiSettings,
26
+ CatalogUiStartupError,
27
+ serve_catalog_ui,
28
+ )
23
29
  from hflow.curation import (
24
30
  CheckCoverage,
25
31
  CurationReport,
@@ -29,8 +35,9 @@ from hflow.curation import (
29
35
  stale_episodes,
30
36
  )
31
37
  from hflow.doctor import DiagnosticLevel, DoctorReport, Finding, diagnose
32
- from hflow.episode import ChannelData, Episode, ExtractedFrame
38
+ from hflow.episode import ChannelData, DecodedMessageBatch, Episode, ExtractedFrame
33
39
  from hflow.format import GopPreset
40
+ from hflow.importers import import_lerobot_dataset
34
41
  from hflow.manifest import (
35
42
  DerivedChannelManifest,
36
43
  PipelineManifest,
@@ -66,6 +73,7 @@ from hflow.steps import (
66
73
  IngestMode,
67
74
  Interval,
68
75
  MeasurementValue,
76
+ Observation,
69
77
  RegisteredCheck,
70
78
  RegisteredEnrichment,
71
79
  Stage,
@@ -92,12 +100,15 @@ except PackageNotFoundError: # running from a source tree without an install
92
100
  __all__ = [
93
101
  "DATASET_SNAPSHOT_FORMAT_NAME",
94
102
  "DATASET_SNAPSHOT_FORMAT_VERSION",
103
+ "DEFAULT_CATALOG_UI_PORT",
95
104
  "RUN_PROFILES",
96
105
  "Aggregation",
97
106
  "App",
98
107
  "AppendResult",
99
108
  "BucketStorageRoot",
100
109
  "Catalog",
110
+ "CatalogUiSettings",
111
+ "CatalogUiStartupError",
101
112
  "ChannelData",
102
113
  "CheckCoverage",
103
114
  "CheckResult",
@@ -107,6 +118,7 @@ __all__ = [
107
118
  "Comparison",
108
119
  "CurationReport",
109
120
  "DatasetSnapshotReport",
121
+ "DecodedMessageBatch",
110
122
  "DerivedChannel",
111
123
  "DerivedChannelManifest",
112
124
  "DerivedSeries",
@@ -128,6 +140,7 @@ __all__ = [
128
140
  "LocalStorageRoot",
129
141
  "MeasurementValue",
130
142
  "MessageBatch",
143
+ "Observation",
131
144
  "PipelineManifest",
132
145
  "PlannedBatch",
133
146
  "PythonMcapEpisodeReader",
@@ -158,6 +171,7 @@ __all__ = [
158
171
  "export_dataset_snapshot",
159
172
  "fetch_uri",
160
173
  "ffmpeg",
174
+ "import_lerobot_dataset",
161
175
  "import_pipeline_application",
162
176
  "is_bucket_url",
163
177
  "open_catalog_connection",
@@ -166,6 +180,7 @@ __all__ = [
166
180
  "plan_batches",
167
181
  "plan_batches_from_files",
168
182
  "providers",
183
+ "serve_catalog_ui",
169
184
  "stages_for_profile",
170
185
  "stale_episodes",
171
186
  "testing",
@@ -0,0 +1,3 @@
1
+ from hflow.cli import main
2
+
3
+ raise SystemExit(main())
@@ -236,6 +236,7 @@ class GroupedMcapWriter:
236
236
  self._flush()
237
237
 
238
238
  def register_schema(self, name: str, encoding: str, data: bytes) -> SchemaId:
239
+ """Register a schema and return its id for :meth:`register_channel`."""
239
240
  self._raise_if_not_active("register a schema")
240
241
  schema_id = SchemaId(len(self._schemas_by_id) + 1)
241
242
  schema = Schema(id=schema_id, data=data, encoding=encoding, name=name)
@@ -255,6 +256,11 @@ class GroupedMcapWriter:
255
256
  group: str,
256
257
  metadata: dict[str, str] | None = None,
257
258
  ) -> ChannelId:
259
+ """Register a topic on a named group and return its id for :meth:`write_message`.
260
+
261
+ ``schema_id`` must be a prior :meth:`register_schema` return value, or
262
+ ``NO_SCHEMA_ID`` for schemaless channels.
263
+ """
258
264
  self._raise_if_not_active("register a channel")
259
265
  # schema_id 0 is the MCAP spec's "no schema" sentinel.
260
266
  if schema_id != NO_SCHEMA_ID and schema_id not in self._schemas_by_id:
@@ -288,6 +294,11 @@ class GroupedMcapWriter:
288
294
  publish_time: int | None = None,
289
295
  sequence: int = 0,
290
296
  ) -> None:
297
+ """Write one message on a registered channel.
298
+
299
+ ``log_time`` and ``publish_time`` are nanoseconds since epoch; when
300
+ ``publish_time`` is omitted it defaults to ``log_time``.
301
+ """
291
302
  self._raise_if_not_active("write a message")
292
303
  group_name = self._group_name_by_channel_id.get(channel_id)
293
304
  if group_name is None:
@@ -318,6 +329,7 @@ class GroupedMcapWriter:
318
329
  self._finalize_group_chunk(group_name)
319
330
 
320
331
  def add_metadata(self, name: str, data: dict[str, str]) -> None:
332
+ """Append an MCAP Metadata record (e.g. episode or provenance stamps)."""
321
333
  self._raise_if_not_active("add metadata")
322
334
  self._flush()
323
335
  offset = self._stream.tell()
@@ -337,6 +349,7 @@ class GroupedMcapWriter:
337
349
  log_time: int = 0,
338
350
  create_time: int = 0,
339
351
  ) -> None:
352
+ """Append an MCAP Attachment record; times are nanoseconds since epoch."""
340
353
  self._raise_if_not_active("add an attachment")
341
354
  self._flush()
342
355
  offset = self._stream.tell()
@@ -363,6 +376,9 @@ class GroupedMcapWriter:
363
376
  self._flush()
364
377
 
365
378
  def finish(self) -> None:
379
+ """Finalize the MCAP file. Must be the last mutating call; safe to call
380
+ again (idempotent). Further writes raise :class:`RuntimeError`.
381
+ """
366
382
  if self._state is _WriterState.FINISHED:
367
383
  return
368
384
  self._raise_if_not_active("finish")
@@ -7,6 +7,7 @@ from types import ModuleType
7
7
 
8
8
  import numpy as np
9
9
 
10
+ from ._field_guards import require_float
10
11
  from ._raw_frames import LUMA_FRAME_FILTER_GRAPH, luma_frames
11
12
  from ._toolchain import VideoMeasurementToolchain
12
13
 
@@ -51,6 +52,8 @@ class CameraMotionSettings:
51
52
  horizontal_field_of_view_degrees: float = DEFAULT_HORIZONTAL_FIELD_OF_VIEW_DEGREES
52
53
 
53
54
  def __post_init__(self) -> None:
55
+ require_float(self.frames_per_second, "frames_per_second")
56
+ require_float(self.horizontal_field_of_view_degrees, "horizontal_field_of_view_degrees")
54
57
  if not math.isfinite(self.frames_per_second) or self.frames_per_second <= 0:
55
58
  raise ValueError(
56
59
  f"frames_per_second must be finite and positive, got {self.frames_per_second}"
@@ -0,0 +1,30 @@
1
+ """Shared numeric-field validation for the video-measurement settings dataclasses.
2
+
3
+ Range checks alone (``0 <= x <= 100``) do not reject the wrong *type*: ``bool``
4
+ subclasses ``int``, so ``True``/``False`` satisfy every numeric comparison and
5
+ range test the settings below already run, and a ``str``/``None`` would raise
6
+ a bare ``TypeError`` from the comparison itself instead of a clear message
7
+ naming the field. This closes that hole the same way ``catalog.py`` already
8
+ does for interval bounds and measurement values.
9
+
10
+ These settings are caller-constructed configuration, not measurement-pipeline
11
+ output, so unlike ``catalog.py``'s NumPy-scalar coercion, no NumPy handling is
12
+ added here: a caller building a ``np.float64`` threshold can call ``.item()``
13
+ itself, the same way any other non-native-Python value would need to.
14
+ """
15
+
16
+
17
+ def require_int(value: object, name: str) -> None:
18
+ """Refuse anything but a plain ``int``, ``bool`` included."""
19
+ if isinstance(value, bool) or not isinstance(value, int):
20
+ raise ValueError(f"{name} must be an int, got {type(value).__name__}")
21
+
22
+
23
+ def require_float(value: object, name: str) -> None:
24
+ """Refuse anything but a plain ``int`` or ``float``, ``bool`` included.
25
+
26
+ An ``int`` is accepted for a float-declared field: it is a perfectly good
27
+ float value (``0`` is a real, falsy value that must still pass).
28
+ """
29
+ if isinstance(value, bool) or not isinstance(value, int | float):
30
+ raise ValueError(f"{name} must be an int or float, got {type(value).__name__}")
@@ -14,6 +14,7 @@ from io import StringIO
14
14
  from pathlib import Path
15
15
  from typing import Protocol
16
16
 
17
+ from ._field_guards import require_float, require_int
17
18
  from ._toolchain import VideoMeasurementToolchain
18
19
 
19
20
  FRAME_STATISTICS_DEFINITION_VERSION = "video-frame-statistics/v1"
@@ -54,6 +55,8 @@ class VideoTimeInterval:
54
55
  end_seconds: float
55
56
 
56
57
  def __post_init__(self) -> None:
58
+ require_float(self.start_seconds, "start_seconds")
59
+ require_float(self.end_seconds, "end_seconds")
57
60
  if not math.isfinite(self.start_seconds) or self.start_seconds < 0:
58
61
  raise ValueError(f"start_seconds must be finite and nonnegative: {self.start_seconds}")
59
62
  if not math.isfinite(self.end_seconds) or self.end_seconds < self.start_seconds:
@@ -74,6 +77,14 @@ class FrameStatisticsSettings:
74
77
  overexposed_average_luma_threshold: float = 235.0
75
78
 
76
79
  def __post_init__(self) -> None:
80
+ require_int(
81
+ self.black_frame_minimum_pixel_share_percent,
82
+ "black_frame_minimum_pixel_share_percent",
83
+ )
84
+ require_int(self.black_pixel_luma_threshold, "black_pixel_luma_threshold")
85
+ require_float(self.freeze_noise_tolerance_decibels, "freeze_noise_tolerance_decibels")
86
+ require_float(self.freeze_minimum_duration_seconds, "freeze_minimum_duration_seconds")
87
+ require_float(self.overexposed_average_luma_threshold, "overexposed_average_luma_threshold")
77
88
  if not 0 <= self.black_frame_minimum_pixel_share_percent <= 100:
78
89
  raise ValueError("black_frame_minimum_pixel_share_percent must be between 0 and 100")
79
90
  if not 0 <= self.black_pixel_luma_threshold <= 255:
@@ -371,7 +382,11 @@ class _FrameStatisticsAccumulator:
371
382
  overexposed_frame_percent=100.0 * self.overexposed_frame_count / frame_count,
372
383
  freeze_intervals=tuple(self.freeze_intervals),
373
384
  freeze_total_seconds=sum(
374
- interval.end_seconds - interval.start_seconds for interval in self.freeze_intervals
385
+ (
386
+ interval.end_seconds - interval.start_seconds
387
+ for interval in self.freeze_intervals
388
+ ),
389
+ 0.0,
375
390
  ),
376
391
  average_luma_mean=self.average_luma.mean(),
377
392
  average_luma_minimum=self.average_luma.minimum(),