pyroparse 0.3.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. pyroparse-0.3.6/.agents/skills/pyroparse/SKILL.md +180 -0
  2. pyroparse-0.3.6/.agents/skills/pyroparse/batch-and-convert.md +161 -0
  3. pyroparse-0.3.6/.agents/skills/pyroparse/integrations.md +123 -0
  4. pyroparse-0.3.6/.agents/skills/pyroparse/metadata.md +142 -0
  5. pyroparse-0.3.6/.agents/skills/pyroparse/raw-messages.md +112 -0
  6. pyroparse-0.3.6/.agents/skills/pyroparse/schema.md +94 -0
  7. pyroparse-0.3.6/.dockerignore +23 -0
  8. pyroparse-0.3.6/.github/workflows/ci.yml +142 -0
  9. pyroparse-0.3.6/.gitignore +35 -0
  10. pyroparse-0.3.6/.python-version +1 -0
  11. pyroparse-0.3.6/BENCHMARK.md +100 -0
  12. pyroparse-0.3.6/BENCHMARK_HTTP.md +55 -0
  13. pyroparse-0.3.6/CHANGELOG.md +99 -0
  14. pyroparse-0.3.6/Cargo.lock +1331 -0
  15. pyroparse-0.3.6/Cargo.toml +24 -0
  16. pyroparse-0.3.6/DEVELOPING.md +179 -0
  17. pyroparse-0.3.6/Dockerfile +29 -0
  18. pyroparse-0.3.6/Makefile +38 -0
  19. pyroparse-0.3.6/PKG-INFO +447 -0
  20. pyroparse-0.3.6/README.md +423 -0
  21. pyroparse-0.3.6/docs/FIT-FORMAT.md +1054 -0
  22. pyroparse-0.3.6/docs/bench_column_projection.png +0 -0
  23. pyroparse-0.3.6/docs/bench_compression.png +0 -0
  24. pyroparse-0.3.6/docs/bench_load_vs_duration.png +0 -0
  25. pyroparse-0.3.6/docs/bench_load_vs_size.png +0 -0
  26. pyroparse-0.3.6/docs/bench_meta_vs_size.png +0 -0
  27. pyroparse-0.3.6/pyproject.toml +55 -0
  28. pyroparse-0.3.6/scripts/benchmark.py +462 -0
  29. pyroparse-0.3.6/scripts/benchmark_http.py +369 -0
  30. pyroparse-0.3.6/scripts/convert_to_parquet.py +126 -0
  31. pyroparse-0.3.6/scripts/demo_duckdb.py +330 -0
  32. pyroparse-0.3.6/scripts/demo_polars.py +316 -0
  33. pyroparse-0.3.6/scripts/download_fit_files.py +62 -0
  34. pyroparse-0.3.6/scripts/generate_profile.py +598 -0
  35. pyroparse-0.3.6/scripts/generate_sport.py +216 -0
  36. pyroparse-0.3.6/scripts/profile.toml +52 -0
  37. pyroparse-0.3.6/server/app.py +155 -0
  38. pyroparse-0.3.6/server/index.html +287 -0
  39. pyroparse-0.3.6/src/fields.rs +160 -0
  40. pyroparse-0.3.6/src/fit/binary.rs +975 -0
  41. pyroparse-0.3.6/src/fit/decode.rs +1485 -0
  42. pyroparse-0.3.6/src/fit/mod.rs +14 -0
  43. pyroparse-0.3.6/src/fit/profile.rs +6305 -0
  44. pyroparse-0.3.6/src/lib.rs +1361 -0
  45. pyroparse-0.3.6/src/pyroparse/__init__.py +83 -0
  46. pyroparse-0.3.6/src/pyroparse/__main__.py +207 -0
  47. pyroparse-0.3.6/src/pyroparse/_activity.py +265 -0
  48. pyroparse-0.3.6/src/pyroparse/_batch.py +263 -0
  49. pyroparse-0.3.6/src/pyroparse/_convert.py +158 -0
  50. pyroparse-0.3.6/src/pyroparse/_core.pyi +5 -0
  51. pyroparse-0.3.6/src/pyroparse/_course.py +139 -0
  52. pyroparse-0.3.6/src/pyroparse/_csv.py +97 -0
  53. pyroparse-0.3.6/src/pyroparse/_errors.py +34 -0
  54. pyroparse-0.3.6/src/pyroparse/_messages.py +25 -0
  55. pyroparse-0.3.6/src/pyroparse/_metadata.py +268 -0
  56. pyroparse-0.3.6/src/pyroparse/_parquet.py +143 -0
  57. pyroparse-0.3.6/src/pyroparse/_schema.py +119 -0
  58. pyroparse-0.3.6/src/pyroparse/_session.py +84 -0
  59. pyroparse-0.3.6/src/pyroparse/_sport.py +247 -0
  60. pyroparse-0.3.6/src/pyroparse/_sport_categories.py +74 -0
  61. pyroparse-0.3.6/src/pyroparse/_types.py +9 -0
  62. pyroparse-0.3.6/src/pyroparse/duckdb.py +112 -0
  63. pyroparse-0.3.6/src/pyroparse/polars.py +76 -0
  64. pyroparse-0.3.6/src/pyroparse/py.typed +0 -0
  65. pyroparse-0.3.6/src/reference.rs +60 -0
  66. pyroparse-0.3.6/src/types.rs +305 -0
  67. pyroparse-0.3.6/src/values.rs +222 -0
  68. pyroparse-0.3.6/tests/conftest.py +85 -0
  69. pyroparse-0.3.6/tests/fixtures/course.fit +0 -0
  70. pyroparse-0.3.6/tests/fixtures/cycling-rowing-cycling-rowing.fit +0 -0
  71. pyroparse-0.3.6/tests/fixtures/test.fit +0 -0
  72. pyroparse-0.3.6/tests/fixtures/with-developer-fields.fit +0 -0
  73. pyroparse-0.3.6/tests/test_activity.py +192 -0
  74. pyroparse-0.3.6/tests/test_batch.py +181 -0
  75. pyroparse-0.3.6/tests/test_convert.py +221 -0
  76. pyroparse-0.3.6/tests/test_course.py +197 -0
  77. pyroparse-0.3.6/tests/test_csv.py +51 -0
  78. pyroparse-0.3.6/tests/test_devices.py +283 -0
  79. pyroparse-0.3.6/tests/test_duckdb.py +46 -0
  80. pyroparse-0.3.6/tests/test_input.py +67 -0
  81. pyroparse-0.3.6/tests/test_laps.py +147 -0
  82. pyroparse-0.3.6/tests/test_messages.py +190 -0
  83. pyroparse-0.3.6/tests/test_parquet.py +40 -0
  84. pyroparse-0.3.6/tests/test_parser_parity.py +156 -0
  85. pyroparse-0.3.6/tests/test_polars.py +46 -0
  86. pyroparse-0.3.6/tests/test_schema.py +303 -0
  87. pyroparse-0.3.6/tests/test_sport.py +137 -0
  88. pyroparse-0.3.6/uv.lock +1888 -0
@@ -0,0 +1,180 @@
1
+ ---
2
+ name: pyroparse
3
+ description: >
4
+ Parse FIT files (activities, courses) into typed PyArrow tables with structured
5
+ metadata. Covers read_fit, Activity/Session/Course classes, all_messages(),
6
+ batch operations, Parquet round-trips, column selection, and the pyroparse CLI.
7
+ Use when writing Python code that reads FIT files, processes activity/workout/
8
+ route data, or builds fitness data pipelines — even if the user just says
9
+ "parse FIT" or "activity data" without naming the library.
10
+ ---
11
+
12
+ # Pyroparse
13
+
14
+ Rust-backed FIT file parser with Python bindings. Reads FIT files into typed
15
+ PyArrow tables with structured metadata. Normalizes manufacturer-specific
16
+ field names into a consistent schema. Round-trips to Parquet with metadata
17
+ preserved. Zero-copy into Polars, DuckDB, and pandas.
18
+
19
+ **Install:** `uv add pyroparse` or `pip install pyroparse`
20
+
21
+ ## Core patterns
22
+
23
+ ### Read a FIT file
24
+
25
+ ```python
26
+ import pyroparse as pp
27
+
28
+ # Table only (no metadata access)
29
+ table = pp.read_fit("ride.fit") # -> pa.Table
30
+
31
+ # With metadata
32
+ activity = pp.Activity.load_fit("ride.fit")
33
+ activity.data # -> pa.Table
34
+ activity.metadata # -> ActivityMetadata
35
+ activity.metadata.sport # "cycling.road"
36
+ activity.metadata.devices # [Device(...), ...]
37
+ ```
38
+
39
+ ### Column selection
40
+
41
+ ```python
42
+ # Default: 11 standard columns
43
+ table = pp.read_fit("ride.fit")
44
+
45
+ # All columns (standard + extras like core_temperature, smo2, form_power)
46
+ table = pp.read_fit("ride.fit", columns="all")
47
+
48
+ # Explicit list
49
+ table = pp.read_fit("ride.fit", columns=["timestamp", "power", "heart_rate"])
50
+
51
+ # Standard + specific extras
52
+ table = pp.read_fit("ride.fit", extra_columns=["core_temperature"])
53
+ ```
54
+
55
+ ### Parquet round-trip
56
+
57
+ ```python
58
+ activity = pp.Activity.load_fit("ride.fit")
59
+ activity.to_parquet("ride.parquet") # ZSTD, metadata preserved
60
+
61
+ loaded = pp.Activity.load_parquet("ride.parquet")
62
+ loaded.metadata.sport # "cycling.road" — survived
63
+ ```
64
+
65
+ ### Multi-activity FIT files
66
+
67
+ ```python
68
+ # Activity.load_fit raises MultipleActivitiesError for multi-session files
69
+ session = pp.Session.load_fit("triathlon.fit")
70
+ session.activities[0].metadata.sport # "swimming"
71
+ session.activities[1].metadata.sport # "cycling"
72
+ ```
73
+
74
+ ### Course files (planned routes)
75
+
76
+ ```python
77
+ course = pp.Course.load_fit("stage3.fit")
78
+ course.track # -> pa.Table (lat, lon, alt, distance)
79
+ course.metadata.name # "Volta Ciclista Stage 3"
80
+ course.metadata.waypoints # -> list[Waypoint] (turns, climbs, sprints)
81
+ course.metadata.waypoints[0].name # "km 0"
82
+ course.metadata.waypoints[0].type # "generic"
83
+ course.to_parquet("stage3.parquet") # single file, waypoints in metadata
84
+ ```
85
+
86
+ ### Raw FIT messages (escape hatch)
87
+
88
+ ```python
89
+ msgs = pp.all_messages("ride.fit")
90
+ # Returns list[dict] — every message, no normalization, fitparser-native format.
91
+ # Each dict: {"kind": "record", "fields": [{"name": ..., "value": ..., "units": ...}, ...]}
92
+
93
+ events = [m["fields"] for m in msgs if m["kind"] == "event"]
94
+ zones = [m["fields"] for m in msgs if m["kind"] == "hr_zone"]
95
+ ```
96
+
97
+ ### Batch operations
98
+
99
+ ```python
100
+ # Scan metadata only (fast)
101
+ catalog = pp.scan_fit("~/activities/") # -> pa.Table (one row per file)
102
+
103
+ # Load timeseries from multiple files
104
+ data = pp.load_fit_batch(paths, columns=["timestamp", "power"])
105
+
106
+ # Batch FIT -> Parquet conversion
107
+ result = pp.convert_fit_tree("~/fit/", "~/parquet/", workers=-1, progress=True)
108
+ ```
109
+
110
+ ### CLI
111
+
112
+ ```bash
113
+ pyroparse convert ride.fit # -> ride.parquet
114
+ pyroparse convert ./activities/ -w -1 # batch, all cores
115
+ pyroparse dump ride.fit # raw JSON to stdout
116
+ pyroparse dump ride.fit --kind event,hr_zone # filter by message type
117
+ ```
118
+
119
+ ## API surface
120
+
121
+ | Function / Class | Description |
122
+ |---|---|
123
+ | `pp.read_fit(source, ...)` | FIT -> `pa.Table` (convenience, no metadata) |
124
+ | `pp.read_parquet(source, ...)` | Parquet -> `pa.Table` |
125
+ | `pp.read_csv(source, ...)` | CSV -> `pa.Table` |
126
+ | `pp.all_messages(source)` | FIT -> `list[dict]` (raw, no normalization) |
127
+ | `pp.Activity.load_fit(source, ...)` | FIT -> `Activity` (data + metadata) |
128
+ | `pp.Activity.load_parquet(source, ...)` | Parquet -> `Activity` |
129
+ | `pp.Activity.load_csv(source, ...)` | CSV -> `Activity` |
130
+ | `pp.Activity.open_fit(path, ...)` | Lazy: metadata now, data on `.data` access |
131
+ | `pp.Activity.open_parquet(path, ...)` | Lazy Parquet loader |
132
+ | `pp.Course.load_fit(source)` | Course FIT -> `Course` (track + waypoints) |
133
+ | `pp.Course.load_parquet(path)` | Parquet -> `Course` |
134
+ | `pp.Session.load_fit(source, ...)` | Multi-activity FIT -> `Session` |
135
+ | `pp.Session.open_fit(path, ...)` | Lazy multi-activity loader |
136
+ | `pp.scan_fit(path, ...)` | Directory -> catalog `pa.Table` (metadata only) |
137
+ | `pp.scan_parquet(path, ...)` | Same for Parquet directories |
138
+ | `pp.load_fit_batch(paths, ...)` | Multiple FIT files -> concatenated `pa.Table` |
139
+ | `pp.convert_fit_file(src, dst)` | Single FIT -> Parquet |
140
+ | `pp.convert_fit_tree(src, dst, ...)` | Batch FIT -> Parquet with directory mirroring |
141
+ | `pp.classify_sport(sport, sub_sport, has_gps)` | -> `Sport` enum value |
142
+ | `pp.STANDARD_COLUMNS` | The 11 default column names |
143
+
144
+ ## Reference
145
+
146
+ **Schema & columns** — standard columns, types, extras, column selection. Read [schema.md](schema.md)
147
+
148
+ **Metadata & devices** — ActivityMetadata, Device, Sport enum. Read [metadata.md](metadata.md)
149
+
150
+ **Batch & convert** — scan, batch load, conversion, CLI commands. Read [batch-and-convert.md](batch-and-convert.md)
151
+
152
+ **Raw messages** — all_messages() format and dump CLI. Read [raw-messages.md](raw-messages.md)
153
+
154
+ **Integrations** — Polars, DuckDB, CSV, Parquet metadata queries. Read [integrations.md](integrations.md)
155
+
156
+ ## Gotchas
157
+
158
+ - `Activity.load_fit()` raises `FileTypeMismatchError` for non-activity FIT
159
+ files (e.g. course files). Use `Course.load_fit()` for course/route files.
160
+ - `Activity.load_fit()` raises `MultipleActivitiesError` for multi-session
161
+ FIT files (triathlon, multisport). Use `Session.load_fit()` instead.
162
+ - `read_fit()` returns a bare `pa.Table` with no metadata access. Use
163
+ `Activity.load_fit()` when you need sport, duration, devices, etc.
164
+ - Default columns are 11 standard columns only. Use `columns="all"` to get
165
+ extras like `core_temperature`, `smo2`, `form_power`, `stance_time`.
166
+ - `extra_columns` cannot be combined with `columns="all"` or an explicit
167
+ column list. It only works with the default (standard) column set.
168
+ - GPS coordinates (`latitude`, `longitude`) are degrees (already converted
169
+ from FIT semicircles). Do NOT convert them again.
170
+ - All timestamps are `Timestamp(us, UTC)` — microsecond precision, UTC.
171
+ `start_time_local` on metadata is naive (no timezone).
172
+ - `all_messages()` returns raw FIT profile names with no normalization.
173
+ Field names will differ from the standard pyroparse schema (e.g.
174
+ `enhanced_speed` instead of `speed`, `position_lat` instead of `latitude`).
175
+ - `open_fit()` uses an experimental binary scanner for metadata. Values
176
+ should be validated against `load_fit()` for critical workflows.
177
+ - `Source` type accepts `str`, `PathLike`, `bytes`, or `BinaryIO` (file-like
178
+ object opened in binary mode).
179
+ - `missing="ignore"` fills absent columns with typed nulls instead of raising.
180
+ Useful for batch loading files with different sensor configurations.
@@ -0,0 +1,161 @@
1
+ # Batch Operations & Conversion
2
+
3
+ ## scan_fit
4
+
5
+ Scan a directory for FIT files. Returns metadata only — no timeseries parsed.
6
+
7
+ ```python
8
+ catalog = pp.scan_fit("~/activities/", recursive=True, errors="warn")
9
+ ```
10
+
11
+ Returns a `pa.Table` with one row per file:
12
+
13
+ | Column | Type | Description |
14
+ |---|---|---|
15
+ | `file_path` | `Utf8` | Absolute path |
16
+ | `sport` | `Utf8` | Classified sport string |
17
+ | `name` | `Utf8` | Activity name (if set) |
18
+ | `start_time` | `Timestamp(us, UTC)` | UTC start |
19
+ | `start_time_local` | `Timestamp(us)` | Naive local time |
20
+ | `duration` | `Float64` | Seconds |
21
+ | `distance` | `Float64` | Meters |
22
+ | `metrics` | `List(Utf8)` | Available metrics |
23
+ | `device_name` | `Utf8` | Creator device name |
24
+ | `device_type` | `Utf8` | Creator device type |
25
+
26
+ Parameters:
27
+ - `path: str` — directory to scan
28
+ - `recursive: bool = True` — search subdirectories
29
+ - `errors: str = "warn"` — `"warn"` skips corrupt files, `"raise"` fails immediately
30
+
31
+ Multi-activity files are skipped with a warning (use `Session.load_fit()`).
32
+
33
+ ## scan_parquet
34
+
35
+ Same interface and schema as `scan_fit`, but reads Parquet schema footers.
36
+
37
+ ```python
38
+ catalog = pp.scan_parquet("~/parquet/", recursive=True, errors="warn")
39
+ ```
40
+
41
+ ## load_fit_batch
42
+
43
+ Parse multiple FIT files into a single concatenated table.
44
+
45
+ ```python
46
+ data = pp.load_fit_batch(
47
+ paths,
48
+ columns=["timestamp", "power"],
49
+ errors="warn",
50
+ )
51
+ ```
52
+
53
+ Returns a `pa.Table` with a `file_path` column prepended. Supports the same
54
+ `columns`, `extra_columns`, `missing` parameters as `read_fit()`.
55
+
56
+ Parameters:
57
+ - `paths: list[str]` — file paths to load
58
+ - `columns`, `extra_columns`, `missing` — same as `read_fit()`
59
+ - `errors: str = "warn"` — `"warn"` skips corrupt files, `"raise"` fails
60
+
61
+ Uses `ThreadPoolExecutor` for concurrent parsing.
62
+
63
+ ## convert_fit_file
64
+
65
+ Convert a single FIT file to Parquet.
66
+
67
+ ```python
68
+ result = pp.convert_fit_file("ride.fit", "ride.parquet")
69
+ # -> Path("ride.parquet")
70
+
71
+ # Multi-activity files produce indexed outputs:
72
+ result = pp.convert_fit_file("triathlon.fit", "triathlon.parquet")
73
+ # -> [Path("triathlon_0.parquet"), Path("triathlon_1.parquet"), ...]
74
+ ```
75
+
76
+ Returns `Path` for single-activity files, `list[Path]` for multi-activity.
77
+
78
+ ## convert_fit_tree
79
+
80
+ Batch-convert a directory of FIT files to Parquet.
81
+
82
+ ```python
83
+ result = pp.convert_fit_tree(
84
+ "~/fit/",
85
+ "~/parquet/", # None = in-place (next to source)
86
+ glob="**/*.[fF][iI][tT]", # default, case-insensitive
87
+ overwrite=False, # skip existing (idempotent re-runs)
88
+ workers=-1, # -1 = all CPU cores
89
+ progress=True, # tqdm bar
90
+ )
91
+ result.converted # list[Path]
92
+ result.errors # list[tuple[Path, Exception]]
93
+ result.failed # bool
94
+ ```
95
+
96
+ Parameters:
97
+ - `src` — single FIT file or directory
98
+ - `dst` — output directory, `None` for in-place
99
+ - `glob: str` — file discovery pattern (default: case-insensitive `**/*.fit`)
100
+ - `overwrite: bool = False` — re-convert existing files
101
+ - `workers: int = 1` — parallel processes, `-1` for all cores
102
+ - `progress: bool = False` — show tqdm progress bar
103
+
104
+ ### ConvertResult
105
+
106
+ ```python
107
+ @dataclass
108
+ class ConvertResult:
109
+ converted: list[Path]
110
+ errors: list[tuple[Path, Exception]]
111
+ failed: bool # property: True if any errors
112
+ ```
113
+
114
+ ## CLI
115
+
116
+ ### pyroparse convert
117
+
118
+ ```
119
+ pyroparse convert <src> [-o <dst>] [flags]
120
+ ```
121
+
122
+ | Flag | Description |
123
+ |---|---|
124
+ | `-o, --output PATH` | Output file or directory (default: `.parquet` next to source) |
125
+ | `--overwrite` | Re-convert files whose output already exists |
126
+ | `--glob PATTERN` | File discovery pattern (default: `**/*.[fF][iI][tT]`) |
127
+ | `-w, --workers N` | Parallel workers; -1 = all cores (default: 1) |
128
+ | `--no-progress` | Disable progress bar |
129
+
130
+ ```bash
131
+ pyroparse convert ride.fit # -> ride.parquet
132
+ pyroparse convert ride.fit -o /tmp/ride.parquet # explicit output
133
+ pyroparse convert ./activities/ -w -1 # batch, all cores
134
+ pyroparse convert ./activities/ -o /tmp/parquet/ # mirror tree
135
+ pyroparse convert ./activities/ --overwrite # force re-convert
136
+ ```
137
+
138
+ ### pyroparse dump
139
+
140
+ ```
141
+ pyroparse dump <src> [-o <file>] [flags]
142
+ ```
143
+
144
+ | Flag | Description |
145
+ |---|---|
146
+ | `-o, --output FILE` | Write to file instead of stdout |
147
+ | `--kind TYPE[,TYPE,...]` | Only include these message types |
148
+ | `--exclude TYPE[,TYPE,...]` | Exclude these message types |
149
+ | `--compact` | Single-line JSON (default: pretty-printed) |
150
+
151
+ `--kind` and `--exclude` are mutually exclusive.
152
+
153
+ ```bash
154
+ pyroparse dump ride.fit # all messages, pretty JSON
155
+ pyroparse dump ride.fit --kind event,hr_zone # filter by type
156
+ pyroparse dump ride.fit --exclude record # skip record messages
157
+ pyroparse dump ride.fit --compact -o out.json # compact, to file
158
+ pyroparse dump ride.fit | jq '.[] | select(.kind == "session")'
159
+ ```
160
+
161
+ Single file only — no batch mode. For batch: `for f in *.fit; do pyroparse dump "$f" -o "${f%.fit}.json"; done`
@@ -0,0 +1,123 @@
1
+ # Integrations
2
+
3
+ ## Polars
4
+
5
+ Requires `polars` installed separately.
6
+
7
+ ```python
8
+ import polars as pl
9
+ import pyroparse.polars as ppl
10
+
11
+ # Scan directory -> Polars DataFrame catalog
12
+ catalog = ppl.scan_fit("~/data/")
13
+
14
+ # Filter and load timeseries
15
+ catalog.filter(pl.col("sport") == "cycling.road") \
16
+ .fit.load_data(columns=["timestamp", "power"])
17
+ ```
18
+
19
+ ### API
20
+
21
+ ```python
22
+ ppl.scan_fit(path, *, recursive=True, errors="warn") -> pl.DataFrame
23
+ ppl.scan_parquet(path, *, recursive=True, errors="warn") -> pl.DataFrame
24
+ ```
25
+
26
+ The `.fit` namespace is registered on all Polars DataFrames:
27
+
28
+ ```python
29
+ df.fit.load_data(*, columns=None, errors="warn") -> pl.DataFrame
30
+ ```
31
+
32
+ Reads the `file_path` column, loads FIT files, returns concatenated DataFrame.
33
+
34
+ ### Zero-copy from PyArrow
35
+
36
+ For single files, use `pl.from_arrow()` directly:
37
+
38
+ ```python
39
+ import polars as pl
40
+ import pyroparse as pp
41
+
42
+ df = pl.from_arrow(pp.read_fit("ride.fit"))
43
+ df.group_by("lap").agg(pl.col("power").mean())
44
+ ```
45
+
46
+ ## DuckDB
47
+
48
+ Requires `duckdb` installed separately.
49
+
50
+ ```python
51
+ import pyroparse.duckdb as ppdb
52
+
53
+ # Scan -> DuckDB relation
54
+ catalog = ppdb.scan_fit("~/data/")
55
+ catalog.filter("sport = 'cycling.road'").fetchdf()
56
+
57
+ # Load timeseries -> DuckDB relation
58
+ paths = catalog.filter("sport = 'cycling.road'").fetchnumpy()["file_path"].tolist()
59
+ data = ppdb.load_fit(paths, columns=["timestamp", "power"])
60
+ data.filter("power > 300").fetchdf()
61
+ ```
62
+
63
+ ### API
64
+
65
+ ```python
66
+ ppdb.scan_fit(path, *, recursive=True, errors="warn", con=None) -> DuckDBPyRelation
67
+ ppdb.scan_parquet(path, *, recursive=True, errors="warn", con=None) -> DuckDBPyRelation
68
+ ppdb.load_fit(paths, *, columns=None, errors="warn", con=None) -> DuckDBPyRelation
69
+ ```
70
+
71
+ All accept an optional `con` parameter for a specific DuckDB connection.
72
+ Defaults to `duckdb.default_connection`.
73
+
74
+ ### Direct Arrow scan
75
+
76
+ ```python
77
+ import duckdb
78
+ import pyroparse as pp
79
+
80
+ activity = pp.Activity.load_fit("ride.fit")
81
+ duckdb.from_arrow(activity.data).filter("power > 300").fetchdf()
82
+ ```
83
+
84
+ ### Parquet metadata queries
85
+
86
+ Pyroparse stores metadata in Parquet schema under the `b"pyroparse"` key.
87
+ Query it with DuckDB without reading row data:
88
+
89
+ ```sql
90
+ SELECT filename, json_extract_string(value, '$.sport') AS sport
91
+ FROM parquet_kv_metadata('activities/*.parquet')
92
+ WHERE key = 'pyroparse'
93
+ AND json_extract_string(value, '$.sport') = 'cycling.road';
94
+ ```
95
+
96
+ ## CSV
97
+
98
+ ```python
99
+ activity = pp.Activity.load_csv("export.csv", metadata={"sport": "cycling"})
100
+ ```
101
+
102
+ CSV loading infers:
103
+ - Timestamps from common column names
104
+ - Duration from first/last timestamp
105
+ - Available metrics from column names
106
+ - Constant string columns are promoted to metadata
107
+
108
+ No special dependencies. Use `metadata={}` override for sport and other
109
+ values CSV cannot express.
110
+
111
+ ## pandas
112
+
113
+ Use PyArrow's built-in conversion:
114
+
115
+ ```python
116
+ import pyroparse as pp
117
+
118
+ df = pp.read_fit("ride.fit").to_pandas()
119
+
120
+ # Or from an Activity
121
+ activity = pp.Activity.load_fit("ride.fit")
122
+ df = activity.data.to_pandas()
123
+ ```
@@ -0,0 +1,142 @@
1
+ # Metadata, Devices & Sport
2
+
3
+ ## ActivityMetadata
4
+
5
+ Dataclass extracted from FIT Session and DeviceInfo messages.
6
+
7
+ ```python
8
+ @dataclass
9
+ class ActivityMetadata:
10
+ sport: str | None # "cycling.road", "running.trail", etc.
11
+ name: str | None # User-given activity name
12
+ start_time: datetime | None # UTC, timezone-aware
13
+ start_time_local: datetime | None # Naive, local wall-clock time (no tz)
14
+ duration: float | None # Seconds
15
+ distance: float | None # Meters
16
+ metrics: set[str] # {"heart_rate", "power", "speed", "cadence", "gps"}
17
+ devices: list[Device] # Head unit + connected sensors
18
+ extra: dict # {"sub_sport": "road", ...}
19
+ ```
20
+
21
+ ### Methods
22
+
23
+ ```python
24
+ meta.column_source("power") # -> Device that produced the column, or None
25
+ meta.to_dict() # -> JSON-serializable dict
26
+ ```
27
+
28
+ ### Metadata override
29
+
30
+ All loaders accept `metadata={}` to override file-native values:
31
+
32
+ ```python
33
+ activity = pp.Activity.load_fit("ride.fit", metadata={"sport": "gravel"})
34
+ activity.metadata.sport # "gravel" (overridden)
35
+ activity.metadata.duration # 3842.7 (preserved from FIT)
36
+ ```
37
+
38
+ Override keys must match `ActivityMetadata` field names. Overrides merge on
39
+ top — unspecified fields keep their file-native values.
40
+
41
+ ## Device
42
+
43
+ ```python
44
+ @dataclass
45
+ class Device:
46
+ name: str | None # "garmin edge_540", "stryd Stryd"
47
+ manufacturer: str | None # "garmin", "stryd", "wahoo_fitness"
48
+ product: str | None # "edge_540", "Stryd"
49
+ serial_number: str | None # String (may be numeric but stored as str)
50
+ device_type: str | None # "creator", "sensor", or "developer"
51
+ columns: list[str] # ["power", "cadence"] — columns this device produced
52
+ ```
53
+
54
+ - `"creator"` = head unit (device_index 0 in FIT)
55
+ - `"sensor"` = hardware sensor (ANT+/BLE)
56
+ - `"developer"` = CIQ app (e.g. Stryd, CORE)
57
+
58
+ `columns` lists the data columns attributed to this device. Pyroparse uses
59
+ ANT+ device type and known manufacturer tables to attribute columns. For
60
+ developer fields, it detects CIQ apps by UUID (Stryd, CORE, etc.).
61
+
62
+ After column selection, `device.columns` is filtered to only include columns
63
+ present in the final table.
64
+
65
+ ## CourseMetadata
66
+
67
+ Dataclass for course/route files. Waypoints are embedded in metadata (not a
68
+ separate table) since they're small, sparse annotations.
69
+
70
+ ```python
71
+ @dataclass
72
+ class CourseMetadata:
73
+ name: str | None # Course name
74
+ distance: float | None # Total distance in meters
75
+ ascent: float | None # Total ascent in meters
76
+ descent: float | None # Total descent in meters
77
+ waypoints: list[Waypoint] # Turns, climbs, sprints, etc.
78
+ ```
79
+
80
+ ## Waypoint
81
+
82
+ ```python
83
+ @dataclass
84
+ class Waypoint:
85
+ name: str | None # "km 0", "Sprint 1", "Road works"
86
+ type: str | None # "generic", "sharp_left", "first_category", etc.
87
+ latitude: float | None # Degrees
88
+ longitude: float | None # Degrees
89
+ distance: float | None # Meters along route
90
+ ```
91
+
92
+ Types include: `generic`, `summit`, `valley`, `water`, `food`, `danger`,
93
+ `left`, `right`, `sharp_left`, `sharp_right`, `slight_left`, `slight_right`,
94
+ `sprint`, `first_category`, `second_category`, `third_category`,
95
+ `fourth_category`, `hors_category`, and more.
96
+
97
+ ## Sport enum
98
+
99
+ Hierarchical enum with dot-notation values.
100
+
101
+ ```python
102
+ from pyroparse import Sport, classify_sport
103
+
104
+ Sport.CYCLING # "cycling"
105
+ Sport.CYCLING_ROAD # "cycling.road"
106
+ Sport.CYCLING_TRACK_250M # "cycling.track.250m"
107
+
108
+ sport = classify_sport("cycling", "road", has_gps=True)
109
+ # -> Sport.CYCLING_ROAD
110
+ ```
111
+
112
+ ### Methods
113
+
114
+ ```python
115
+ sport.parent_sport() # Sport.CYCLING (or None for root sports)
116
+ sport.root_sport() # Sport.CYCLING (walks to root)
117
+ sport.is_root_sport() # False
118
+ sport.display_name() # "Cycling > Road"
119
+
120
+ sport.is_sub_sport_of(Sport.CYCLING) # True
121
+ sport.is_sub_sport_of([Sport.CYCLING, Sport.RUNNING]) # True (any match)
122
+ ```
123
+
124
+ ### classify_sport()
125
+
126
+ ```python
127
+ pp.classify_sport(sport: str | None, sub_sport: str | None, has_gps: bool) -> Sport
128
+ ```
129
+
130
+ Maps FIT `sport` + `sub_sport` strings to a `Sport` enum value. Uses `has_gps`
131
+ to distinguish indoor/outdoor variants (e.g. cycling with GPS -> `cycling.road`,
132
+ without -> `cycling.trainer`).
133
+
134
+ Returns `Sport.UNKNOWN` for unrecognized combinations.
135
+
136
+ ### Available sports
137
+
138
+ Root sports: `cycling`, `running`, `walking`, `swimming`, `rowing`,
139
+ `cross_country_skiing`, `generic`, `unknown`.
140
+
141
+ Each has sub-sports (e.g. `cycling.road`, `cycling.trainer`,
142
+ `running.trail`, `swimming.pool.25m`). Up to 3 levels deep.