pyroparse 0.3.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyroparse-0.3.6/.agents/skills/pyroparse/SKILL.md +180 -0
- pyroparse-0.3.6/.agents/skills/pyroparse/batch-and-convert.md +161 -0
- pyroparse-0.3.6/.agents/skills/pyroparse/integrations.md +123 -0
- pyroparse-0.3.6/.agents/skills/pyroparse/metadata.md +142 -0
- pyroparse-0.3.6/.agents/skills/pyroparse/raw-messages.md +112 -0
- pyroparse-0.3.6/.agents/skills/pyroparse/schema.md +94 -0
- pyroparse-0.3.6/.dockerignore +23 -0
- pyroparse-0.3.6/.github/workflows/ci.yml +142 -0
- pyroparse-0.3.6/.gitignore +35 -0
- pyroparse-0.3.6/.python-version +1 -0
- pyroparse-0.3.6/BENCHMARK.md +100 -0
- pyroparse-0.3.6/BENCHMARK_HTTP.md +55 -0
- pyroparse-0.3.6/CHANGELOG.md +99 -0
- pyroparse-0.3.6/Cargo.lock +1331 -0
- pyroparse-0.3.6/Cargo.toml +24 -0
- pyroparse-0.3.6/DEVELOPING.md +179 -0
- pyroparse-0.3.6/Dockerfile +29 -0
- pyroparse-0.3.6/Makefile +38 -0
- pyroparse-0.3.6/PKG-INFO +447 -0
- pyroparse-0.3.6/README.md +423 -0
- pyroparse-0.3.6/docs/FIT-FORMAT.md +1054 -0
- pyroparse-0.3.6/docs/bench_column_projection.png +0 -0
- pyroparse-0.3.6/docs/bench_compression.png +0 -0
- pyroparse-0.3.6/docs/bench_load_vs_duration.png +0 -0
- pyroparse-0.3.6/docs/bench_load_vs_size.png +0 -0
- pyroparse-0.3.6/docs/bench_meta_vs_size.png +0 -0
- pyroparse-0.3.6/pyproject.toml +55 -0
- pyroparse-0.3.6/scripts/benchmark.py +462 -0
- pyroparse-0.3.6/scripts/benchmark_http.py +369 -0
- pyroparse-0.3.6/scripts/convert_to_parquet.py +126 -0
- pyroparse-0.3.6/scripts/demo_duckdb.py +330 -0
- pyroparse-0.3.6/scripts/demo_polars.py +316 -0
- pyroparse-0.3.6/scripts/download_fit_files.py +62 -0
- pyroparse-0.3.6/scripts/generate_profile.py +598 -0
- pyroparse-0.3.6/scripts/generate_sport.py +216 -0
- pyroparse-0.3.6/scripts/profile.toml +52 -0
- pyroparse-0.3.6/server/app.py +155 -0
- pyroparse-0.3.6/server/index.html +287 -0
- pyroparse-0.3.6/src/fields.rs +160 -0
- pyroparse-0.3.6/src/fit/binary.rs +975 -0
- pyroparse-0.3.6/src/fit/decode.rs +1485 -0
- pyroparse-0.3.6/src/fit/mod.rs +14 -0
- pyroparse-0.3.6/src/fit/profile.rs +6305 -0
- pyroparse-0.3.6/src/lib.rs +1361 -0
- pyroparse-0.3.6/src/pyroparse/__init__.py +83 -0
- pyroparse-0.3.6/src/pyroparse/__main__.py +207 -0
- pyroparse-0.3.6/src/pyroparse/_activity.py +265 -0
- pyroparse-0.3.6/src/pyroparse/_batch.py +263 -0
- pyroparse-0.3.6/src/pyroparse/_convert.py +158 -0
- pyroparse-0.3.6/src/pyroparse/_core.pyi +5 -0
- pyroparse-0.3.6/src/pyroparse/_course.py +139 -0
- pyroparse-0.3.6/src/pyroparse/_csv.py +97 -0
- pyroparse-0.3.6/src/pyroparse/_errors.py +34 -0
- pyroparse-0.3.6/src/pyroparse/_messages.py +25 -0
- pyroparse-0.3.6/src/pyroparse/_metadata.py +268 -0
- pyroparse-0.3.6/src/pyroparse/_parquet.py +143 -0
- pyroparse-0.3.6/src/pyroparse/_schema.py +119 -0
- pyroparse-0.3.6/src/pyroparse/_session.py +84 -0
- pyroparse-0.3.6/src/pyroparse/_sport.py +247 -0
- pyroparse-0.3.6/src/pyroparse/_sport_categories.py +74 -0
- pyroparse-0.3.6/src/pyroparse/_types.py +9 -0
- pyroparse-0.3.6/src/pyroparse/duckdb.py +112 -0
- pyroparse-0.3.6/src/pyroparse/polars.py +76 -0
- pyroparse-0.3.6/src/pyroparse/py.typed +0 -0
- pyroparse-0.3.6/src/reference.rs +60 -0
- pyroparse-0.3.6/src/types.rs +305 -0
- pyroparse-0.3.6/src/values.rs +222 -0
- pyroparse-0.3.6/tests/conftest.py +85 -0
- pyroparse-0.3.6/tests/fixtures/course.fit +0 -0
- pyroparse-0.3.6/tests/fixtures/cycling-rowing-cycling-rowing.fit +0 -0
- pyroparse-0.3.6/tests/fixtures/test.fit +0 -0
- pyroparse-0.3.6/tests/fixtures/with-developer-fields.fit +0 -0
- pyroparse-0.3.6/tests/test_activity.py +192 -0
- pyroparse-0.3.6/tests/test_batch.py +181 -0
- pyroparse-0.3.6/tests/test_convert.py +221 -0
- pyroparse-0.3.6/tests/test_course.py +197 -0
- pyroparse-0.3.6/tests/test_csv.py +51 -0
- pyroparse-0.3.6/tests/test_devices.py +283 -0
- pyroparse-0.3.6/tests/test_duckdb.py +46 -0
- pyroparse-0.3.6/tests/test_input.py +67 -0
- pyroparse-0.3.6/tests/test_laps.py +147 -0
- pyroparse-0.3.6/tests/test_messages.py +190 -0
- pyroparse-0.3.6/tests/test_parquet.py +40 -0
- pyroparse-0.3.6/tests/test_parser_parity.py +156 -0
- pyroparse-0.3.6/tests/test_polars.py +46 -0
- pyroparse-0.3.6/tests/test_schema.py +303 -0
- pyroparse-0.3.6/tests/test_sport.py +137 -0
- pyroparse-0.3.6/uv.lock +1888 -0
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: pyroparse
|
|
3
|
+
description: >
|
|
4
|
+
Parse FIT files (activities, courses) into typed PyArrow tables with structured
|
|
5
|
+
metadata. Covers read_fit, Activity/Session/Course classes, all_messages(),
|
|
6
|
+
batch operations, Parquet round-trips, column selection, and the pyroparse CLI.
|
|
7
|
+
Use when writing Python code that reads FIT files, processes activity/workout/
|
|
8
|
+
route data, or builds fitness data pipelines — even if the user just says
|
|
9
|
+
"parse FIT" or "activity data" without naming the library.
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
# Pyroparse
|
|
13
|
+
|
|
14
|
+
Rust-backed FIT file parser with Python bindings. Reads FIT files into typed
|
|
15
|
+
PyArrow tables with structured metadata. Normalizes manufacturer-specific
|
|
16
|
+
field names into a consistent schema. Round-trips to Parquet with metadata
|
|
17
|
+
preserved. Zero-copy into Polars, DuckDB, and pandas.
|
|
18
|
+
|
|
19
|
+
**Install:** `uv add pyroparse` or `pip install pyroparse`
|
|
20
|
+
|
|
21
|
+
## Core patterns
|
|
22
|
+
|
|
23
|
+
### Read a FIT file
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
import pyroparse as pp
|
|
27
|
+
|
|
28
|
+
# Table only (no metadata access)
|
|
29
|
+
table = pp.read_fit("ride.fit") # -> pa.Table
|
|
30
|
+
|
|
31
|
+
# With metadata
|
|
32
|
+
activity = pp.Activity.load_fit("ride.fit")
|
|
33
|
+
activity.data # -> pa.Table
|
|
34
|
+
activity.metadata # -> ActivityMetadata
|
|
35
|
+
activity.metadata.sport # "cycling.road"
|
|
36
|
+
activity.metadata.devices # [Device(...), ...]
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### Column selection
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
# Default: 11 standard columns
|
|
43
|
+
table = pp.read_fit("ride.fit")
|
|
44
|
+
|
|
45
|
+
# All columns (standard + extras like core_temperature, smo2, form_power)
|
|
46
|
+
table = pp.read_fit("ride.fit", columns="all")
|
|
47
|
+
|
|
48
|
+
# Explicit list
|
|
49
|
+
table = pp.read_fit("ride.fit", columns=["timestamp", "power", "heart_rate"])
|
|
50
|
+
|
|
51
|
+
# Standard + specific extras
|
|
52
|
+
table = pp.read_fit("ride.fit", extra_columns=["core_temperature"])
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
### Parquet round-trip
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
activity = pp.Activity.load_fit("ride.fit")
|
|
59
|
+
activity.to_parquet("ride.parquet") # ZSTD, metadata preserved
|
|
60
|
+
|
|
61
|
+
loaded = pp.Activity.load_parquet("ride.parquet")
|
|
62
|
+
loaded.metadata.sport # "cycling.road" — survived
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
### Multi-activity FIT files
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
# Activity.load_fit raises MultipleActivitiesError for multi-session files
|
|
69
|
+
session = pp.Session.load_fit("triathlon.fit")
|
|
70
|
+
session.activities[0].metadata.sport # "swimming"
|
|
71
|
+
session.activities[1].metadata.sport # "cycling"
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
### Course files (planned routes)
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
course = pp.Course.load_fit("stage3.fit")
|
|
78
|
+
course.track # -> pa.Table (lat, lon, alt, distance)
|
|
79
|
+
course.metadata.name # "Volta Ciclista Stage 3"
|
|
80
|
+
course.metadata.waypoints # -> list[Waypoint] (turns, climbs, sprints)
|
|
81
|
+
course.metadata.waypoints[0].name # "km 0"
|
|
82
|
+
course.metadata.waypoints[0].type # "generic"
|
|
83
|
+
course.to_parquet("stage3.parquet") # single file, waypoints in metadata
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
### Raw FIT messages (escape hatch)
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
msgs = pp.all_messages("ride.fit")
|
|
90
|
+
# Returns list[dict] — every message, no normalization, fitparser-native format.
|
|
91
|
+
# Each dict: {"kind": "record", "fields": [{"name": ..., "value": ..., "units": ...}, ...]}
|
|
92
|
+
|
|
93
|
+
events = [m["fields"] for m in msgs if m["kind"] == "event"]
|
|
94
|
+
zones = [m["fields"] for m in msgs if m["kind"] == "hr_zone"]
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### Batch operations
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
# Scan metadata only (fast)
|
|
101
|
+
catalog = pp.scan_fit("~/activities/") # -> pa.Table (one row per file)
|
|
102
|
+
|
|
103
|
+
# Load timeseries from multiple files
|
|
104
|
+
data = pp.load_fit_batch(paths, columns=["timestamp", "power"])
|
|
105
|
+
|
|
106
|
+
# Batch FIT -> Parquet conversion
|
|
107
|
+
result = pp.convert_fit_tree("~/fit/", "~/parquet/", workers=-1, progress=True)
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
### CLI
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
pyroparse convert ride.fit # -> ride.parquet
|
|
114
|
+
pyroparse convert ./activities/ -w -1 # batch, all cores
|
|
115
|
+
pyroparse dump ride.fit # raw JSON to stdout
|
|
116
|
+
pyroparse dump ride.fit --kind event,hr_zone # filter by message type
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## API surface
|
|
120
|
+
|
|
121
|
+
| Function / Class | Description |
|
|
122
|
+
|---|---|
|
|
123
|
+
| `pp.read_fit(source, ...)` | FIT -> `pa.Table` (convenience, no metadata) |
|
|
124
|
+
| `pp.read_parquet(source, ...)` | Parquet -> `pa.Table` |
|
|
125
|
+
| `pp.read_csv(source, ...)` | CSV -> `pa.Table` |
|
|
126
|
+
| `pp.all_messages(source)` | FIT -> `list[dict]` (raw, no normalization) |
|
|
127
|
+
| `pp.Activity.load_fit(source, ...)` | FIT -> `Activity` (data + metadata) |
|
|
128
|
+
| `pp.Activity.load_parquet(source, ...)` | Parquet -> `Activity` |
|
|
129
|
+
| `pp.Activity.load_csv(source, ...)` | CSV -> `Activity` |
|
|
130
|
+
| `pp.Activity.open_fit(path, ...)` | Lazy: metadata now, data on `.data` access |
|
|
131
|
+
| `pp.Activity.open_parquet(path, ...)` | Lazy Parquet loader |
|
|
132
|
+
| `pp.Course.load_fit(source)` | Course FIT -> `Course` (track + waypoints) |
|
|
133
|
+
| `pp.Course.load_parquet(path)` | Parquet -> `Course` |
|
|
134
|
+
| `pp.Session.load_fit(source, ...)` | Multi-activity FIT -> `Session` |
|
|
135
|
+
| `pp.Session.open_fit(path, ...)` | Lazy multi-activity loader |
|
|
136
|
+
| `pp.scan_fit(path, ...)` | Directory -> catalog `pa.Table` (metadata only) |
|
|
137
|
+
| `pp.scan_parquet(path, ...)` | Same for Parquet directories |
|
|
138
|
+
| `pp.load_fit_batch(paths, ...)` | Multiple FIT files -> concatenated `pa.Table` |
|
|
139
|
+
| `pp.convert_fit_file(src, dst)` | Single FIT -> Parquet |
|
|
140
|
+
| `pp.convert_fit_tree(src, dst, ...)` | Batch FIT -> Parquet with directory mirroring |
|
|
141
|
+
| `pp.classify_sport(sport, sub_sport, has_gps)` | -> `Sport` enum value |
|
|
142
|
+
| `pp.STANDARD_COLUMNS` | The 11 default column names |
|
|
143
|
+
|
|
144
|
+
## Reference
|
|
145
|
+
|
|
146
|
+
**Schema & columns** — standard columns, types, extras, column selection. Read [schema.md](schema.md)
|
|
147
|
+
|
|
148
|
+
**Metadata & devices** — ActivityMetadata, Device, Sport enum. Read [metadata.md](metadata.md)
|
|
149
|
+
|
|
150
|
+
**Batch & convert** — scan, batch load, conversion, CLI commands. Read [batch-and-convert.md](batch-and-convert.md)
|
|
151
|
+
|
|
152
|
+
**Raw messages** — all_messages() format and dump CLI. Read [raw-messages.md](raw-messages.md)
|
|
153
|
+
|
|
154
|
+
**Integrations** — Polars, DuckDB, CSV, Parquet metadata queries. Read [integrations.md](integrations.md)
|
|
155
|
+
|
|
156
|
+
## Gotchas
|
|
157
|
+
|
|
158
|
+
- `Activity.load_fit()` raises `FileTypeMismatchError` for non-activity FIT
|
|
159
|
+
files (e.g. course files). Use `Course.load_fit()` for course/route files.
|
|
160
|
+
- `Activity.load_fit()` raises `MultipleActivitiesError` for multi-session
|
|
161
|
+
FIT files (triathlon, multisport). Use `Session.load_fit()` instead.
|
|
162
|
+
- `read_fit()` returns a bare `pa.Table` with no metadata access. Use
|
|
163
|
+
`Activity.load_fit()` when you need sport, duration, devices, etc.
|
|
164
|
+
- Default columns are 11 standard columns only. Use `columns="all"` to get
|
|
165
|
+
extras like `core_temperature`, `smo2`, `form_power`, `stance_time`.
|
|
166
|
+
- `extra_columns` cannot be combined with `columns="all"` or an explicit
|
|
167
|
+
column list. It only works with the default (standard) column set.
|
|
168
|
+
- GPS coordinates (`latitude`, `longitude`) are degrees (already converted
|
|
169
|
+
from FIT semicircles). Do NOT convert them again.
|
|
170
|
+
- All timestamps are `Timestamp(us, UTC)` — microsecond precision, UTC.
|
|
171
|
+
`start_time_local` on metadata is naive (no timezone).
|
|
172
|
+
- `all_messages()` returns raw FIT profile names with no normalization.
|
|
173
|
+
Field names will differ from the standard pyroparse schema (e.g.
|
|
174
|
+
`enhanced_speed` instead of `speed`, `position_lat` instead of `latitude`).
|
|
175
|
+
- `open_fit()` uses an experimental binary scanner for metadata. Values
|
|
176
|
+
should be validated against `load_fit()` for critical workflows.
|
|
177
|
+
- `Source` type accepts `str`, `PathLike`, `bytes`, or `BinaryIO` (file-like
|
|
178
|
+
object opened in binary mode).
|
|
179
|
+
- `missing="ignore"` fills absent columns with typed nulls instead of raising.
|
|
180
|
+
Useful for batch loading files with different sensor configurations.
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# Batch Operations & Conversion
|
|
2
|
+
|
|
3
|
+
## scan_fit
|
|
4
|
+
|
|
5
|
+
Scan a directory for FIT files. Returns metadata only — no timeseries parsed.
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
catalog = pp.scan_fit("~/activities/", recursive=True, errors="warn")
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Returns a `pa.Table` with one row per file:
|
|
12
|
+
|
|
13
|
+
| Column | Type | Description |
|
|
14
|
+
|---|---|---|
|
|
15
|
+
| `file_path` | `Utf8` | Absolute path |
|
|
16
|
+
| `sport` | `Utf8` | Classified sport string |
|
|
17
|
+
| `name` | `Utf8` | Activity name (if set) |
|
|
18
|
+
| `start_time` | `Timestamp(us, UTC)` | UTC start |
|
|
19
|
+
| `start_time_local` | `Timestamp(us)` | Naive local time |
|
|
20
|
+
| `duration` | `Float64` | Seconds |
|
|
21
|
+
| `distance` | `Float64` | Meters |
|
|
22
|
+
| `metrics` | `List(Utf8)` | Available metrics |
|
|
23
|
+
| `device_name` | `Utf8` | Creator device name |
|
|
24
|
+
| `device_type` | `Utf8` | Creator device type |
|
|
25
|
+
|
|
26
|
+
Parameters:
|
|
27
|
+
- `path: str` — directory to scan
|
|
28
|
+
- `recursive: bool = True` — search subdirectories
|
|
29
|
+
- `errors: str = "warn"` — `"warn"` skips corrupt files, `"raise"` fails immediately
|
|
30
|
+
|
|
31
|
+
Multi-activity files are skipped with a warning (use `Session.load_fit()`).
|
|
32
|
+
|
|
33
|
+
## scan_parquet
|
|
34
|
+
|
|
35
|
+
Same interface and schema as `scan_fit`, but reads Parquet schema footers.
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
catalog = pp.scan_parquet("~/parquet/", recursive=True, errors="warn")
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## load_fit_batch
|
|
42
|
+
|
|
43
|
+
Parse multiple FIT files into a single concatenated table.
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
data = pp.load_fit_batch(
|
|
47
|
+
paths,
|
|
48
|
+
columns=["timestamp", "power"],
|
|
49
|
+
errors="warn",
|
|
50
|
+
)
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Returns a `pa.Table` with a `file_path` column prepended. Supports the same
|
|
54
|
+
`columns`, `extra_columns`, `missing` parameters as `read_fit()`.
|
|
55
|
+
|
|
56
|
+
Parameters:
|
|
57
|
+
- `paths: list[str]` — file paths to load
|
|
58
|
+
- `columns`, `extra_columns`, `missing` — same as `read_fit()`
|
|
59
|
+
- `errors: str = "warn"` — `"warn"` skips corrupt files, `"raise"` fails
|
|
60
|
+
|
|
61
|
+
Uses `ThreadPoolExecutor` for concurrent parsing.
|
|
62
|
+
|
|
63
|
+
## convert_fit_file
|
|
64
|
+
|
|
65
|
+
Convert a single FIT file to Parquet.
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
result = pp.convert_fit_file("ride.fit", "ride.parquet")
|
|
69
|
+
# -> Path("ride.parquet")
|
|
70
|
+
|
|
71
|
+
# Multi-activity files produce indexed outputs:
|
|
72
|
+
result = pp.convert_fit_file("triathlon.fit", "triathlon.parquet")
|
|
73
|
+
# -> [Path("triathlon_0.parquet"), Path("triathlon_1.parquet"), ...]
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Returns `Path` for single-activity files, `list[Path]` for multi-activity.
|
|
77
|
+
|
|
78
|
+
## convert_fit_tree
|
|
79
|
+
|
|
80
|
+
Batch-convert a directory of FIT files to Parquet.
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
result = pp.convert_fit_tree(
|
|
84
|
+
"~/fit/",
|
|
85
|
+
"~/parquet/", # None = in-place (next to source)
|
|
86
|
+
glob="**/*.[fF][iI][tT]", # default, case-insensitive
|
|
87
|
+
overwrite=False, # skip existing (idempotent re-runs)
|
|
88
|
+
workers=-1, # -1 = all CPU cores
|
|
89
|
+
progress=True, # tqdm bar
|
|
90
|
+
)
|
|
91
|
+
result.converted # list[Path]
|
|
92
|
+
result.errors # list[tuple[Path, Exception]]
|
|
93
|
+
result.failed # bool
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Parameters:
|
|
97
|
+
- `src` — single FIT file or directory
|
|
98
|
+
- `dst` — output directory, `None` for in-place
|
|
99
|
+
- `glob: str` — file discovery pattern (default: case-insensitive `**/*.fit`)
|
|
100
|
+
- `overwrite: bool = False` — re-convert existing files
|
|
101
|
+
- `workers: int = 1` — parallel processes, `-1` for all cores
|
|
102
|
+
- `progress: bool = False` — show tqdm progress bar
|
|
103
|
+
|
|
104
|
+
### ConvertResult
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
@dataclass
|
|
108
|
+
class ConvertResult:
|
|
109
|
+
converted: list[Path]
|
|
110
|
+
errors: list[tuple[Path, Exception]]
|
|
111
|
+
failed: bool # property: True if any errors
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## CLI
|
|
115
|
+
|
|
116
|
+
### pyroparse convert
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
pyroparse convert <src> [-o <dst>] [flags]
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
| Flag | Description |
|
|
123
|
+
|---|---|
|
|
124
|
+
| `-o, --output PATH` | Output file or directory (default: `.parquet` next to source) |
|
|
125
|
+
| `--overwrite` | Re-convert files whose output already exists |
|
|
126
|
+
| `--glob PATTERN` | File discovery pattern (default: `**/*.[fF][iI][tT]`) |
|
|
127
|
+
| `-w, --workers N` | Parallel workers; -1 = all cores (default: 1) |
|
|
128
|
+
| `--no-progress` | Disable progress bar |
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
pyroparse convert ride.fit # -> ride.parquet
|
|
132
|
+
pyroparse convert ride.fit -o /tmp/ride.parquet # explicit output
|
|
133
|
+
pyroparse convert ./activities/ -w -1 # batch, all cores
|
|
134
|
+
pyroparse convert ./activities/ -o /tmp/parquet/ # mirror tree
|
|
135
|
+
pyroparse convert ./activities/ --overwrite # force re-convert
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
### pyroparse dump
|
|
139
|
+
|
|
140
|
+
```
|
|
141
|
+
pyroparse dump <src> [-o <file>] [flags]
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
| Flag | Description |
|
|
145
|
+
|---|---|
|
|
146
|
+
| `-o, --output FILE` | Write to file instead of stdout |
|
|
147
|
+
| `--kind TYPE[,TYPE,...]` | Only include these message types |
|
|
148
|
+
| `--exclude TYPE[,TYPE,...]` | Exclude these message types |
|
|
149
|
+
| `--compact` | Single-line JSON (default: pretty-printed) |
|
|
150
|
+
|
|
151
|
+
`--kind` and `--exclude` are mutually exclusive.
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
pyroparse dump ride.fit # all messages, pretty JSON
|
|
155
|
+
pyroparse dump ride.fit --kind event,hr_zone # filter by type
|
|
156
|
+
pyroparse dump ride.fit --exclude record # skip record messages
|
|
157
|
+
pyroparse dump ride.fit --compact -o out.json # compact, to file
|
|
158
|
+
pyroparse dump ride.fit | jq '.[] | select(.kind == "session")'
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Single file only — no batch mode. For batch: `for f in *.fit; do pyroparse dump "$f" -o "${f%.fit}.json"; done`
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
# Integrations
|
|
2
|
+
|
|
3
|
+
## Polars
|
|
4
|
+
|
|
5
|
+
Requires `polars` installed separately.
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
import polars as pl
|
|
9
|
+
import pyroparse.polars as ppl
|
|
10
|
+
|
|
11
|
+
# Scan directory -> Polars DataFrame catalog
|
|
12
|
+
catalog = ppl.scan_fit("~/data/")
|
|
13
|
+
|
|
14
|
+
# Filter and load timeseries
|
|
15
|
+
catalog.filter(pl.col("sport") == "cycling.road") \
|
|
16
|
+
.fit.load_data(columns=["timestamp", "power"])
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
### API
|
|
20
|
+
|
|
21
|
+
```python
|
|
22
|
+
ppl.scan_fit(path, *, recursive=True, errors="warn") -> pl.DataFrame
|
|
23
|
+
ppl.scan_parquet(path, *, recursive=True, errors="warn") -> pl.DataFrame
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
The `.fit` namespace is registered on all Polars DataFrames:
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
df.fit.load_data(*, columns=None, errors="warn") -> pl.DataFrame
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Reads the `file_path` column, loads FIT files, returns concatenated DataFrame.
|
|
33
|
+
|
|
34
|
+
### Zero-copy from PyArrow
|
|
35
|
+
|
|
36
|
+
For single files, use `pl.from_arrow()` directly:
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
import polars as pl
|
|
40
|
+
import pyroparse as pp
|
|
41
|
+
|
|
42
|
+
df = pl.from_arrow(pp.read_fit("ride.fit"))
|
|
43
|
+
df.group_by("lap").agg(pl.col("power").mean())
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## DuckDB
|
|
47
|
+
|
|
48
|
+
Requires `duckdb` installed separately.
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
import pyroparse.duckdb as ppdb
|
|
52
|
+
|
|
53
|
+
# Scan -> DuckDB relation
|
|
54
|
+
catalog = ppdb.scan_fit("~/data/")
|
|
55
|
+
catalog.filter("sport = 'cycling.road'").fetchdf()
|
|
56
|
+
|
|
57
|
+
# Load timeseries -> DuckDB relation
|
|
58
|
+
paths = catalog.filter("sport = 'cycling.road'").fetchnumpy()["file_path"].tolist()
|
|
59
|
+
data = ppdb.load_fit(paths, columns=["timestamp", "power"])
|
|
60
|
+
data.filter("power > 300").fetchdf()
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
### API
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
ppdb.scan_fit(path, *, recursive=True, errors="warn", con=None) -> DuckDBPyRelation
|
|
67
|
+
ppdb.scan_parquet(path, *, recursive=True, errors="warn", con=None) -> DuckDBPyRelation
|
|
68
|
+
ppdb.load_fit(paths, *, columns=None, errors="warn", con=None) -> DuckDBPyRelation
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
All accept an optional `con` parameter for a specific DuckDB connection.
|
|
72
|
+
Defaults to `duckdb.default_connection`.
|
|
73
|
+
|
|
74
|
+
### Direct Arrow scan
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
import duckdb
|
|
78
|
+
import pyroparse as pp
|
|
79
|
+
|
|
80
|
+
activity = pp.Activity.load_fit("ride.fit")
|
|
81
|
+
duckdb.from_arrow(activity.data).filter("power > 300").fetchdf()
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### Parquet metadata queries
|
|
85
|
+
|
|
86
|
+
Pyroparse stores metadata in Parquet schema under the `b"pyroparse"` key.
|
|
87
|
+
Query it with DuckDB without reading row data:
|
|
88
|
+
|
|
89
|
+
```sql
|
|
90
|
+
SELECT filename, json_extract_string(value, '$.sport') AS sport
|
|
91
|
+
FROM parquet_kv_metadata('activities/*.parquet')
|
|
92
|
+
WHERE key = 'pyroparse'
|
|
93
|
+
AND json_extract_string(value, '$.sport') = 'cycling.road';
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## CSV
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
activity = pp.Activity.load_csv("export.csv", metadata={"sport": "cycling"})
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
CSV loading infers:
|
|
103
|
+
- Timestamps from common column names
|
|
104
|
+
- Duration from first/last timestamp
|
|
105
|
+
- Available metrics from column names
|
|
106
|
+
- Constant string columns are promoted to metadata
|
|
107
|
+
|
|
108
|
+
No special dependencies. Use `metadata={}` override for sport and other
|
|
109
|
+
values CSV cannot express.
|
|
110
|
+
|
|
111
|
+
## pandas
|
|
112
|
+
|
|
113
|
+
Use PyArrow's built-in conversion:
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
import pyroparse as pp
|
|
117
|
+
|
|
118
|
+
df = pp.read_fit("ride.fit").to_pandas()
|
|
119
|
+
|
|
120
|
+
# Or from an Activity
|
|
121
|
+
activity = pp.Activity.load_fit("ride.fit")
|
|
122
|
+
df = activity.data.to_pandas()
|
|
123
|
+
```
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# Metadata, Devices & Sport
|
|
2
|
+
|
|
3
|
+
## ActivityMetadata
|
|
4
|
+
|
|
5
|
+
Dataclass extracted from FIT Session and DeviceInfo messages.
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
@dataclass
|
|
9
|
+
class ActivityMetadata:
|
|
10
|
+
sport: str | None # "cycling.road", "running.trail", etc.
|
|
11
|
+
name: str | None # User-given activity name
|
|
12
|
+
start_time: datetime | None # UTC, timezone-aware
|
|
13
|
+
start_time_local: datetime | None # Naive, local wall-clock time (no tz)
|
|
14
|
+
duration: float | None # Seconds
|
|
15
|
+
distance: float | None # Meters
|
|
16
|
+
metrics: set[str] # {"heart_rate", "power", "speed", "cadence", "gps"}
|
|
17
|
+
devices: list[Device] # Head unit + connected sensors
|
|
18
|
+
extra: dict # {"sub_sport": "road", ...}
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
### Methods
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
meta.column_source("power") # -> Device that produced the column, or None
|
|
25
|
+
meta.to_dict() # -> JSON-serializable dict
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
### Metadata override
|
|
29
|
+
|
|
30
|
+
All loaders accept `metadata={}` to override file-native values:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
activity = pp.Activity.load_fit("ride.fit", metadata={"sport": "gravel"})
|
|
34
|
+
activity.metadata.sport # "gravel" (overridden)
|
|
35
|
+
activity.metadata.duration # 3842.7 (preserved from FIT)
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Override keys must match `ActivityMetadata` field names. Overrides merge on
|
|
39
|
+
top — unspecified fields keep their file-native values.
|
|
40
|
+
|
|
41
|
+
## Device
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
@dataclass
|
|
45
|
+
class Device:
|
|
46
|
+
name: str | None # "garmin edge_540", "stryd Stryd"
|
|
47
|
+
manufacturer: str | None # "garmin", "stryd", "wahoo_fitness"
|
|
48
|
+
product: str | None # "edge_540", "Stryd"
|
|
49
|
+
serial_number: str | None # String (may be numeric but stored as str)
|
|
50
|
+
device_type: str | None # "creator", "sensor", or "developer"
|
|
51
|
+
columns: list[str] # ["power", "cadence"] — columns this device produced
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
- `"creator"` = head unit (device_index 0 in FIT)
|
|
55
|
+
- `"sensor"` = hardware sensor (ANT+/BLE)
|
|
56
|
+
- `"developer"` = CIQ app (e.g. Stryd, CORE)
|
|
57
|
+
|
|
58
|
+
`columns` lists the data columns attributed to this device. Pyroparse uses
|
|
59
|
+
ANT+ device type and known manufacturer tables to attribute columns. For
|
|
60
|
+
developer fields, it detects CIQ apps by UUID (Stryd, CORE, etc.).
|
|
61
|
+
|
|
62
|
+
After column selection, `device.columns` is filtered to only include columns
|
|
63
|
+
present in the final table.
|
|
64
|
+
|
|
65
|
+
## CourseMetadata
|
|
66
|
+
|
|
67
|
+
Dataclass for course/route files. Waypoints are embedded in metadata (not a
|
|
68
|
+
separate table) since they're small, sparse annotations.
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
@dataclass
|
|
72
|
+
class CourseMetadata:
|
|
73
|
+
name: str | None # Course name
|
|
74
|
+
distance: float | None # Total distance in meters
|
|
75
|
+
ascent: float | None # Total ascent in meters
|
|
76
|
+
descent: float | None # Total descent in meters
|
|
77
|
+
waypoints: list[Waypoint] # Turns, climbs, sprints, etc.
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Waypoint
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
@dataclass
|
|
84
|
+
class Waypoint:
|
|
85
|
+
name: str | None # "km 0", "Sprint 1", "Road works"
|
|
86
|
+
type: str | None # "generic", "sharp_left", "first_category", etc.
|
|
87
|
+
latitude: float | None # Degrees
|
|
88
|
+
longitude: float | None # Degrees
|
|
89
|
+
distance: float | None # Meters along route
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Types include: `generic`, `summit`, `valley`, `water`, `food`, `danger`,
|
|
93
|
+
`left`, `right`, `sharp_left`, `sharp_right`, `slight_left`, `slight_right`,
|
|
94
|
+
`sprint`, `first_category`, `second_category`, `third_category`,
|
|
95
|
+
`fourth_category`, `hors_category`, and more.
|
|
96
|
+
|
|
97
|
+
## Sport enum
|
|
98
|
+
|
|
99
|
+
Hierarchical enum with dot-notation values.
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
from pyroparse import Sport, classify_sport
|
|
103
|
+
|
|
104
|
+
Sport.CYCLING # "cycling"
|
|
105
|
+
Sport.CYCLING_ROAD # "cycling.road"
|
|
106
|
+
Sport.CYCLING_TRACK_250M # "cycling.track.250m"
|
|
107
|
+
|
|
108
|
+
sport = classify_sport("cycling", "road", has_gps=True)
|
|
109
|
+
# -> Sport.CYCLING_ROAD
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
### Methods
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
sport.parent_sport() # Sport.CYCLING (or None for root sports)
|
|
116
|
+
sport.root_sport() # Sport.CYCLING (walks to root)
|
|
117
|
+
sport.is_root_sport() # False
|
|
118
|
+
sport.display_name() # "Cycling > Road"
|
|
119
|
+
|
|
120
|
+
sport.is_sub_sport_of(Sport.CYCLING) # True
|
|
121
|
+
sport.is_sub_sport_of([Sport.CYCLING, Sport.RUNNING]) # True (any match)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### classify_sport()
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
pp.classify_sport(sport: str | None, sub_sport: str | None, has_gps: bool) -> Sport
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Maps FIT `sport` + `sub_sport` strings to a `Sport` enum value. Uses `has_gps`
|
|
131
|
+
to distinguish indoor/outdoor variants (e.g. cycling with GPS -> `cycling.road`,
|
|
132
|
+
without -> `cycling.trainer`).
|
|
133
|
+
|
|
134
|
+
Returns `Sport.UNKNOWN` for unrecognized combinations.
|
|
135
|
+
|
|
136
|
+
### Available sports
|
|
137
|
+
|
|
138
|
+
Root sports: `cycling`, `running`, `walking`, `swimming`, `rowing`,
|
|
139
|
+
`cross_country_skiing`, `generic`, `unknown`.
|
|
140
|
+
|
|
141
|
+
Each has sub-sports (e.g. `cycling.road`, `cycling.trainer`,
|
|
142
|
+
`running.trail`, `swimming.pool.25m`). Up to 3 levels deep.
|