microsegments 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {microsegments-0.1.0 → microsegments-0.2.0}/PKG-INFO +15 -3
- {microsegments-0.1.0 → microsegments-0.2.0}/README.md +14 -2
- microsegments-0.2.0/examples/stib1_platform.toml +23 -0
- microsegments-0.2.0/examples/stib55_full.toml +32 -0
- microsegments-0.2.0/examples/stib71_platform.toml +23 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/__init__.py +3 -2
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/_version.py +2 -2
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/aggregate.py +29 -8
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/cli.py +102 -8
- microsegments-0.2.0/src/microsegments/compare.py +512 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/config.py +14 -2
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/contract.py +27 -2
- microsegments-0.2.0/src/microsegments/hotspots.py +572 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/template.html +327 -52
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/metrics.py +134 -14
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/calendar.py +20 -2
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/patterns.py +51 -11
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/pipeline.py +28 -12
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/report.py +71 -7
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/schema.py +1 -1
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/tune.py +111 -14
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_aggregate.py +13 -1
- microsegments-0.2.0/tests/test_compare.py +83 -0
- microsegments-0.2.0/tests/test_hotspots_criteria.py +92 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_pipeline_cli.py +36 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_tune.py +3 -1
- microsegments-0.2.0/tests/test_versions_merge.py +60 -0
- microsegments-0.1.0/src/microsegments/hotspots.py +0 -337
- {microsegments-0.1.0 → microsegments-0.2.0}/.github/workflows/publish.yml +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/.github/workflows/test.yml +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/.gitignore +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/LICENSE +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/docs/report.jpg +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/examples/gtfsrt_csv.toml +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/examples/make_synth.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/examples/stib55.toml +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/examples/synth.toml +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/pyproject.toml +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/__init__.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/export.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/leaflet.min.css +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/__init__.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/coverage.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/events.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/gtfsrt.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/ids.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/stib.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/tabular.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/timeutil.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/__init__.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/common.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/linear.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/mapmatch.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/resample.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/__init__.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/geometry.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/keys.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/plot.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/py.typed +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/segments.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/simulate.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/tracks.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/buckets.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/drops.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/flen.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/obs_day.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/obs_dow.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/passages.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/summary.json +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/validation.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/agency.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/calendar.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/calendar_dates.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/routes.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/shapes.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/stop_times.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/stops.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/trips.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/punctuality/route55.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/speed55_ref.json +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250318.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250318.snaps.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250319.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250319.snaps.parquet +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/gtfs_fixtures.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/regression/make_golden.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/t4_fixtures.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_coverage.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_hotspots.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_events.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_gtfsrt.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_stib.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_tabular.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_locate_linear.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_locate_mapmatch.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_metrics.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_network_geometry.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_network_patterns.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_regression_stib55.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_report_html.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_scaffold.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_segments.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_simulate.py +0 -0
- {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_tracks.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: microsegments
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Where do transit vehicles linger? Count vehicle position observations per micro-segment of a line, from GTFS + AVL / GTFS-RT positions.
|
|
5
5
|
Project-URL: Homepage, https://github.com/GaspardMerten/microsegments
|
|
6
6
|
Project-URL: Repository, https://github.com/GaspardMerten/microsegments
|
|
@@ -132,7 +132,9 @@ Stages are usable on their own: `read_observations`, `build_network`, `segment`,
|
|
|
132
132
|
microsegments run CONFIG -o OUT [--lang en] [--segment-m 20] [--dates A..B] [--source "credit"]
|
|
133
133
|
microsegments tune CONFIG [--lengths 10,20,30,50] [--sensitivity] [-o OUT]
|
|
134
134
|
microsegments hotspots CONFIG [-o hotspots.geojson|.csv|.parquet]
|
|
135
|
-
microsegments inspect CONFIG # coverage per day, GTFS versions, placement
|
|
135
|
+
microsegments inspect CONFIG # coverage per day, GTFS versions, placement, evening vs line reference
|
|
136
|
+
microsegments compare CONFIG --a 2025-02-17..2025-03-31 --b 2025-04-01..2025-05-15 -o OUT
|
|
137
|
+
microsegments sensitivity-table CONFIG [CONFIG ...] -o OUT # one tidy table + markdown over several lines
|
|
136
138
|
```
|
|
137
139
|
|
|
138
140
|
## What the numbers mean
|
|
@@ -169,7 +171,8 @@ fixed per metric (p98 over 6–21 h), so hours and weekdays compare directly.
|
|
|
169
171
|
split-half reliability, hotspot localisation spread and Jaccard) and picks the smallest L with reliability
|
|
170
172
|
≥ 0.8, spread ≤ 30 m and deviance within one standard error of the minimum. 30 m is a good default for
|
|
171
173
|
20 s polling at urban speeds; go shorter only with dense GTFS-RT fixes and many days. `--sensitivity`
|
|
172
|
-
re-runs phase offsets, gap caps, stop zones, references
|
|
174
|
+
re-runs phase offsets, gap caps (30, 40, 60, ∞ s), stop zones, references, passage sources and segment
|
|
175
|
+
lengths 15 / 30 / 60 m around the baseline.
|
|
173
176
|
|
|
174
177
|
## Hotspots
|
|
175
178
|
|
|
@@ -179,6 +182,15 @@ Adjacent bins merge, never across a stop / running border. Classes: *infrastruct
|
|
|
179
182
|
*congestion* (peak only), *mixed*. Needs at least 5 included days. Export as GeoJSON with
|
|
180
183
|
`microsegments hotspots CONFIG -o hotspots.geojson`.
|
|
181
184
|
|
|
185
|
+
## Comparing two periods
|
|
186
|
+
|
|
187
|
+
`microsegments compare` (or `ms.compare(analysis_a, analysis_b)`) compares observations per passage
|
|
188
|
+
segment by segment and hour by hour, each period with its own days and references, with a day bootstrap
|
|
189
|
+
of the difference. Segments are matched by key (or by stop pair when a new GTFS version moved the shape a
|
|
190
|
+
little); the others are listed and left out. Adjacent significant segments form stretches, ranked by the
|
|
191
|
+
change in vehicle time per day; stretches whose frequency changed by more than 20 % are flagged, since a
|
|
192
|
+
timetable change can explain part of the difference. The HTML page gets a "compare two periods" view.
|
|
193
|
+
|
|
182
194
|
## License
|
|
183
195
|
|
|
184
196
|
MIT. The bundled Leaflet CSS is BSD-2-Clause.
|
|
@@ -94,7 +94,9 @@ Stages are usable on their own: `read_observations`, `build_network`, `segment`,
|
|
|
94
94
|
microsegments run CONFIG -o OUT [--lang en] [--segment-m 20] [--dates A..B] [--source "credit"]
|
|
95
95
|
microsegments tune CONFIG [--lengths 10,20,30,50] [--sensitivity] [-o OUT]
|
|
96
96
|
microsegments hotspots CONFIG [-o hotspots.geojson|.csv|.parquet]
|
|
97
|
-
microsegments inspect CONFIG # coverage per day, GTFS versions, placement
|
|
97
|
+
microsegments inspect CONFIG # coverage per day, GTFS versions, placement, evening vs line reference
|
|
98
|
+
microsegments compare CONFIG --a 2025-02-17..2025-03-31 --b 2025-04-01..2025-05-15 -o OUT
|
|
99
|
+
microsegments sensitivity-table CONFIG [CONFIG ...] -o OUT # one tidy table + markdown over several lines
|
|
98
100
|
```
|
|
99
101
|
|
|
100
102
|
## What the numbers mean
|
|
@@ -131,7 +133,8 @@ fixed per metric (p98 over 6–21 h), so hours and weekdays compare directly.
|
|
|
131
133
|
split-half reliability, hotspot localisation spread and Jaccard) and picks the smallest L with reliability
|
|
132
134
|
≥ 0.8, spread ≤ 30 m and deviance within one standard error of the minimum. 30 m is a good default for
|
|
133
135
|
20 s polling at urban speeds; go shorter only with dense GTFS-RT fixes and many days. `--sensitivity`
|
|
134
|
-
re-runs phase offsets, gap caps, stop zones, references
|
|
136
|
+
re-runs phase offsets, gap caps (30, 40, 60, ∞ s), stop zones, references, passage sources and segment
|
|
137
|
+
lengths 15 / 30 / 60 m around the baseline.
|
|
135
138
|
|
|
136
139
|
## Hotspots
|
|
137
140
|
|
|
@@ -141,6 +144,15 @@ Adjacent bins merge, never across a stop / running border. Classes: *infrastruct
|
|
|
141
144
|
*congestion* (peak only), *mixed*. Needs at least 5 included days. Export as GeoJSON with
|
|
142
145
|
`microsegments hotspots CONFIG -o hotspots.geojson`.
|
|
143
146
|
|
|
147
|
+
## Comparing two periods
|
|
148
|
+
|
|
149
|
+
`microsegments compare` (or `ms.compare(analysis_a, analysis_b)`) compares observations per passage
|
|
150
|
+
segment by segment and hour by hour, each period with its own days and references, with a day bootstrap
|
|
151
|
+
of the difference. Segments are matched by key (or by stop pair when a new GTFS version moved the shape a
|
|
152
|
+
little); the others are listed and left out. Adjacent significant segments form stretches, ranked by the
|
|
153
|
+
change in vehicle time per day; stretches whose frequency changed by more than 20 % are flagged, since a
|
|
154
|
+
timetable change can explain part of the difference. The HTML page gets a "compare two periods" view.
|
|
155
|
+
|
|
144
156
|
## License
|
|
145
157
|
|
|
146
158
|
MIT. The bundled Leaflet CSS is BSD-2-Clause.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# LOCAL EXAMPLE (not shipped data): STIB line 1 from the StibMicrosegments platform's local ingest
|
|
2
|
+
# (raw/vd/date=.../{vd,snaps}.parquet with a line column, raw/punctuality, gtfs/index.parquet + gtfs/feeds/<sha>).
|
|
3
|
+
# microsegments sensitivity-table examples/stib55_full.toml examples/stib71_platform.toml examples/stib1_platform.toml -o out/sens
|
|
4
|
+
|
|
5
|
+
[input]
|
|
6
|
+
paths = ["/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/vd/date=2025-0[2-5]-*/*.parquet"]
|
|
7
|
+
kind = "stib"
|
|
8
|
+
events = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/punctuality/date=2025-0[2-5]-*/p.parquet"
|
|
9
|
+
|
|
10
|
+
[gtfs]
|
|
11
|
+
index = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/gtfs/index.parquet"
|
|
12
|
+
|
|
13
|
+
[select]
|
|
14
|
+
route = "1"
|
|
15
|
+
dates = "2025-03-03..2025-03-28"
|
|
16
|
+
weekdays = [0, 1, 2, 3, 4]
|
|
17
|
+
|
|
18
|
+
[params]
|
|
19
|
+
segment_m = 30
|
|
20
|
+
bootstrap = 200
|
|
21
|
+
|
|
22
|
+
[report]
|
|
23
|
+
names = "title"
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# LOCAL EXAMPLE (not shipped data): STIB tram 55, weekdays 17 Feb - 15 May 2025, on the author's machine.
|
|
2
|
+
# Paths point at ~/Documents/Dev/CoDE/BrusselsTransitStuck/data2025 (MobilityTwin stib/vehicle-distance as
|
|
3
|
+
# compact daily parquet, STIB punctuality, one GTFS feed); adapt them to your copy of the data.
|
|
4
|
+
# microsegments run examples/stib55_full.toml -o out/stib55_full
|
|
5
|
+
# microsegments tune examples/stib55_full.toml --sensitivity
|
|
6
|
+
|
|
7
|
+
[input]
|
|
8
|
+
paths = ["/home/gaspard/Documents/Dev/CoDE/BrusselsTransitStuck/data2025/stib/vd55/*.parquet"]
|
|
9
|
+
kind = "stib"
|
|
10
|
+
events = "/home/gaspard/Documents/Dev/CoDE/BrusselsTransitStuck/data2025/stib/punctuality/2025-0[2-5]-*.parquet"
|
|
11
|
+
|
|
12
|
+
[input.linear]
|
|
13
|
+
direction_kind = "terminus_stop"
|
|
14
|
+
id_normaliser = "leading_digits"
|
|
15
|
+
feed_length = "auto"
|
|
16
|
+
|
|
17
|
+
[gtfs]
|
|
18
|
+
# one feed (calendar 14 Apr - 11 May 2025); other dates take the patterns of the same weekday
|
|
19
|
+
path = "/home/gaspard/Documents/Dev/CoDE/BrusselsTransitStuck/data2025/gtfs"
|
|
20
|
+
|
|
21
|
+
[select]
|
|
22
|
+
route = "55"
|
|
23
|
+
dates = "2025-02-17..2025-05-15"
|
|
24
|
+
weekdays = [0, 1, 2, 3, 4]
|
|
25
|
+
exclude_dates = ["2025-04-21", "2025-05-01"] # Easter Monday, Labour Day
|
|
26
|
+
|
|
27
|
+
[params]
|
|
28
|
+
segment_m = 30
|
|
29
|
+
bootstrap = 200
|
|
30
|
+
|
|
31
|
+
[report]
|
|
32
|
+
names = "title" # "GARE DU NORD" -> "Gare du Nord"
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# LOCAL EXAMPLE (not shipped data): STIB line 71 from the StibMicrosegments platform's local ingest
|
|
2
|
+
# (raw/vd/date=.../{vd,snaps}.parquet with a line column, raw/punctuality, gtfs/index.parquet + gtfs/feeds/<sha>).
|
|
3
|
+
# microsegments sensitivity-table examples/stib55_full.toml examples/stib71_platform.toml examples/stib1_platform.toml -o out/sens
|
|
4
|
+
|
|
5
|
+
[input]
|
|
6
|
+
paths = ["/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/vd/date=2025-0[2-5]-*/*.parquet"]
|
|
7
|
+
kind = "stib"
|
|
8
|
+
events = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/punctuality/date=2025-0[2-5]-*/p.parquet"
|
|
9
|
+
|
|
10
|
+
[gtfs]
|
|
11
|
+
index = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/gtfs/index.parquet"
|
|
12
|
+
|
|
13
|
+
[select]
|
|
14
|
+
route = "71"
|
|
15
|
+
dates = "2025-03-03..2025-03-28"
|
|
16
|
+
weekdays = [0, 1, 2, 3, 4]
|
|
17
|
+
|
|
18
|
+
[params]
|
|
19
|
+
segment_m = 30
|
|
20
|
+
bootstrap = 200
|
|
21
|
+
|
|
22
|
+
[report]
|
|
23
|
+
names = "title"
|
|
@@ -15,6 +15,7 @@ except ImportError: # pragma: no cover
|
|
|
15
15
|
__version__ = "0.0.0"
|
|
16
16
|
|
|
17
17
|
from .aggregate import count
|
|
18
|
+
from .compare import Comparison, compare, run_compare
|
|
18
19
|
from .config import Config
|
|
19
20
|
from .hotspots import hotspots
|
|
20
21
|
from .html import export
|
|
@@ -28,8 +29,8 @@ from .segments import segment
|
|
|
28
29
|
from .simulate import simulate
|
|
29
30
|
from . import tune
|
|
30
31
|
|
|
31
|
-
__all__ = ["Analysis", "Config", "Flag", "RunResult", "__version__", "analyse", "build_network", "count", "export",
|
|
32
|
-
"hotspots", "plot", "prepare", "read_observations", "run", "segment", "simulate", "to_contract", "tune"]
|
|
32
|
+
__all__ = ["Analysis", "Comparison", "Config", "Flag", "RunResult", "__version__", "analyse", "build_network", "compare", "count", "export",
|
|
33
|
+
"hotspots", "plot", "prepare", "read_observations", "run", "run_compare", "segment", "simulate", "to_contract", "tune"]
|
|
33
34
|
|
|
34
35
|
|
|
35
36
|
def __getattr__(name: str):
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.
|
|
22
|
-
__version_tuple__ = version_tuple = (0,
|
|
21
|
+
__version__ = version = '0.2.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 0)
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
|
@@ -6,9 +6,11 @@ upper bound closed, as in the prototype: a vehicle standing at stop B is at the
|
|
|
6
6
|
A row at ``pos_m <= 0`` of link i > 0 goes to the last segment of link i - 1, i.e. the end of the
|
|
7
7
|
link arriving at that stop. Positions beyond the link end are clipped to its last segment.
|
|
8
8
|
|
|
9
|
-
Per-vehicle feeds (GTFS-RT with irregular / dense fixes) are
|
|
10
|
-
|
|
11
|
-
|
|
9
|
+
Per-vehicle feeds (GTFS-RT with irregular / dense fixes) are put on a ``tick_s`` grid so counts stay
|
|
10
|
+
comparable with a snapshot feed polled every ``tick_s``: by default ``locate.resample.resample`` (each
|
|
11
|
+
track sampled at the grid instants between its fixes, nearer fix; unbiased at track ends and for
|
|
12
|
+
sparse fixes), or ``method="thin"``: at most one observation per ``(track_id, floor(ts / tick_s))``
|
|
13
|
+
(the last fix of each tick; under-counts when fixes are sparser than the tick).
|
|
12
14
|
"""
|
|
13
15
|
from __future__ import annotations
|
|
14
16
|
|
|
@@ -74,15 +76,34 @@ def assign(placed: pl.DataFrame, segments: pl.DataFrame) -> pl.DataFrame:
|
|
|
74
76
|
)
|
|
75
77
|
|
|
76
78
|
|
|
79
|
+
def per_vehicle_grid(placed: pl.DataFrame, tick_s: float = 20.0, method: str = "resample") -> pl.DataFrame:
|
|
80
|
+
"""Counted rows of a per-vehicle feed on the ``tick_s`` grid (``method`` "resample" or "thin")."""
|
|
81
|
+
if method == "thin":
|
|
82
|
+
return thin(placed.filter(pl.col("count").fill_null(True)), tick_s)
|
|
83
|
+
if method != "resample":
|
|
84
|
+
raise ValueError(f"unknown per-vehicle method {method!r}")
|
|
85
|
+
from .locate.resample import resample
|
|
86
|
+
df = placed
|
|
87
|
+
if "s_m" not in df.columns:
|
|
88
|
+
df = df.with_columns(pl.col("pos_m").alias("s_m"))
|
|
89
|
+
elif "pos_m" in df.columns and df["s_m"].null_count():
|
|
90
|
+
df = df.with_columns(pl.coalesce("s_m", pl.col("pos_m").cast(df.schema["s_m"])).alias("s_m"))
|
|
91
|
+
if "count" not in df.columns:
|
|
92
|
+
df = df.with_columns(pl.lit(True).alias("count"))
|
|
93
|
+
# resample first (non-counted fixes still bound the intervals), then keep counted instants
|
|
94
|
+
return resample(df, tick_s).filter(pl.col("count").fill_null(True))
|
|
95
|
+
|
|
96
|
+
|
|
77
97
|
def count(placed: pl.DataFrame, segments: pl.DataFrame, tick_s: float = 20.0,
|
|
78
|
-
per_vehicle: bool = False) -> pl.DataFrame:
|
|
98
|
+
per_vehicle: bool = False, method: str = "resample") -> pl.DataFrame:
|
|
79
99
|
"""Placed -> Cube (service_date, hour, pattern_uid, seg_key, obs).
|
|
80
100
|
|
|
81
|
-
Σ obs equals the number of counted (
|
|
82
|
-
segments."""
|
|
83
|
-
df = placed.filter(pl.col("count").fill_null(True))
|
|
101
|
+
Σ obs equals the number of counted (for per-vehicle feeds: resampled, or thinned with
|
|
102
|
+
``method="thin"``) rows whose link has segments."""
|
|
84
103
|
if per_vehicle:
|
|
85
|
-
df =
|
|
104
|
+
df = per_vehicle_grid(placed, tick_s, method)
|
|
105
|
+
else:
|
|
106
|
+
df = placed.filter(pl.col("count").fill_null(True))
|
|
86
107
|
df = assign(df.select("ts", "service_date", "hour", "pattern_uid", "link_idx", "pos_m", "track_id"), segments)
|
|
87
108
|
cube = (df.filter(pl.col("seg_key").is_not_null())
|
|
88
109
|
.group_by("service_date", "hour", "pattern_uid", "seg_key")
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
"""Command line: ``microsegments run | tune | hotspots | inspect CONFIG
|
|
1
|
+
"""Command line: ``microsegments run | tune | hotspots | inspect | compare CONFIG``, ``microsegments
|
|
2
|
+
sensitivity-table CONFIG...``."""
|
|
2
3
|
from __future__ import annotations
|
|
3
4
|
|
|
4
5
|
import argparse
|
|
@@ -25,13 +26,16 @@ def _log(args):
|
|
|
25
26
|
return (lambda s: print(s, file=sys.stderr)) if not getattr(args, "quiet", False) else None
|
|
26
27
|
|
|
27
28
|
|
|
28
|
-
def _hotspot_table(hs: pl.DataFrame, tick_s: float) -> pl.DataFrame:
|
|
29
|
+
def _hotspot_table(hs: pl.DataFrame, tick_s: float, names: dict[str, str] | None = None) -> pl.DataFrame:
|
|
29
30
|
if hs is None or hs.height == 0:
|
|
30
31
|
return pl.DataFrame()
|
|
32
|
+
nm = names or {}
|
|
33
|
+
fix = lambda c: pl.col(c).replace(nm) if nm else pl.col(c) # noqa: E731
|
|
34
|
+
crit = pl.col("criterion") if "criterion" in hs.columns else pl.lit("peak")
|
|
31
35
|
return hs.select(
|
|
32
36
|
"direction_id", "rank",
|
|
33
|
-
pl.concat_str([
|
|
34
|
-
pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone", "kind",
|
|
37
|
+
pl.concat_str([fix("from_stop_name"), pl.lit(" -> "), fix("to_stop_name")]).alias("stretch"),
|
|
38
|
+
pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone", "kind", crit.alias("criterion"),
|
|
35
39
|
pl.col("hours").cast(pl.List(pl.Utf8)).list.join(",").alias("hours"),
|
|
36
40
|
pl.col("excess_per_passage").round(2).alias("excess_obs_per_veh"),
|
|
37
41
|
(pl.col("excess_per_passage") * tick_s).round(0).alias("approx_s_per_veh"),
|
|
@@ -57,9 +61,10 @@ def cmd_hotspots(args) -> int:
|
|
|
57
61
|
from .pipeline import run
|
|
58
62
|
cfg = _cfg(args)
|
|
59
63
|
res = run(cfg, log=_log(args))
|
|
60
|
-
|
|
64
|
+
from .report import name_map
|
|
65
|
+
t = _hotspot_table(res.hotspots, cfg.params.tick_s, name_map(cfg.report.names, res.network))
|
|
61
66
|
if t.height == 0:
|
|
62
|
-
print("no hotspot")
|
|
67
|
+
print(f"no hotspot ({res.hotspots_status})")
|
|
63
68
|
return 0
|
|
64
69
|
with pl.Config(tbl_rows=200, tbl_cols=20, fmt_str_lengths=60, tbl_width_chars=200):
|
|
65
70
|
print(t)
|
|
@@ -95,7 +100,9 @@ def cmd_tune(args) -> int:
|
|
|
95
100
|
tr.table.write_parquet(out / "tune.parquet")
|
|
96
101
|
sr = None
|
|
97
102
|
if args.sensitivity:
|
|
98
|
-
|
|
103
|
+
kw["params"] = replace(cfg.params, segment_m=float(tr.recommended))
|
|
104
|
+
sr = tune.sensitivity(pre.placed, pre.segment_fn(), tr.recommended, B=args.bootstrap,
|
|
105
|
+
coverage_fn=pre.coverage_fn(), **kw)
|
|
99
106
|
with pl.Config(tbl_rows=50, tbl_cols=20, tbl_width_chars=200, float_precision=3):
|
|
100
107
|
print(sr.summary)
|
|
101
108
|
print(sr.hotspots)
|
|
@@ -152,6 +159,70 @@ def cmd_inspect(args) -> int:
|
|
|
152
159
|
p = pre.passages
|
|
153
160
|
print(f"\npassages: {p['n'].sum():,.0f} link passages"
|
|
154
161
|
+ (f" (feed {p['n_feed'].sum():,.0f}, events {p['n_events'].sum():,.0f})" if p["n_events"].null_count() < p.height else ""))
|
|
162
|
+
if not args.no_agreement:
|
|
163
|
+
from .metrics import reference_agreement
|
|
164
|
+
from .pipeline import run
|
|
165
|
+
res = run(cfg, prepared=pre, hotspot_kw={"B": min(cfg.params.bootstrap, 100)})
|
|
166
|
+
ag = reference_agreement(res.analysis, res.hotspots)
|
|
167
|
+
print(f"\nreferences: evening {cfg.params.reference_hours[0]}-{cfg.params.reference_hours[1]} h vs line level "
|
|
168
|
+
"(median running obs / passage / m, 6-23 h); day-band excess profiles:")
|
|
169
|
+
with pl.Config(tbl_rows=10, tbl_cols=20, tbl_width_chars=200, float_precision=3):
|
|
170
|
+
print(ag)
|
|
171
|
+
return 0
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _period(s: str) -> str:
|
|
175
|
+
a, b = s.split("..")
|
|
176
|
+
import datetime as dt
|
|
177
|
+
dt.date.fromisoformat(a), dt.date.fromisoformat(b)
|
|
178
|
+
return s
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def cmd_compare(args) -> int:
|
|
182
|
+
from .compare import run_compare
|
|
183
|
+
cfg = _cfg(args)
|
|
184
|
+
res = run_compare(cfg, args.a, args.b, B=args.bootstrap, log=_log(args))
|
|
185
|
+
paths = res.save(args.out, html=not args.no_html, lang=args.lang, title=args.title, source=args.source)
|
|
186
|
+
st = res.comparison.stretches
|
|
187
|
+
if st.height:
|
|
188
|
+
from .report import name_map
|
|
189
|
+
nm = name_map(cfg.report.names, res.prepared.network)
|
|
190
|
+
fix = lambda c: pl.col(c).replace(nm) if nm else pl.col(c) # noqa: E731
|
|
191
|
+
t = st.select("direction_id", "rank",
|
|
192
|
+
pl.concat_str([fix("from_stop_name"), pl.lit(" -> "), fix("to_stop_name")]).alias("stretch"),
|
|
193
|
+
pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone",
|
|
194
|
+
pl.col("hours").cast(pl.List(pl.Utf8)).list.join(",").alias("hours"),
|
|
195
|
+
pl.col("delta_s").round(0).alias("delta_s_per_veh"),
|
|
196
|
+
(pl.col("veh_time_s_per_day") / 60).round(1).alias("veh_min_per_day"),
|
|
197
|
+
pl.col("passages_per_h_a").round(1).alias("veh_h_a"), pl.col("passages_per_h_b").round(1).alias("veh_h_b"),
|
|
198
|
+
"frequency_changed")
|
|
199
|
+
with pl.Config(tbl_rows=60, tbl_cols=20, fmt_str_lengths=60, tbl_width_chars=220):
|
|
200
|
+
print(t.head(args.top) if args.top else t)
|
|
201
|
+
else:
|
|
202
|
+
print("no significant change")
|
|
203
|
+
for p in paths.values():
|
|
204
|
+
print(p)
|
|
205
|
+
return 0
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def cmd_sensitivity_table(args) -> int:
|
|
209
|
+
from . import tune
|
|
210
|
+
configs = [Config.from_toml(c) for c in args.configs]
|
|
211
|
+
if args.segment_m:
|
|
212
|
+
configs = [replace(c, params=replace(c.params, segment_m=float(args.segment_m))) for c in configs]
|
|
213
|
+
table, md = tune.sensitivity_table(configs, B=args.bootstrap, dates=args.dates, log=_log(args))
|
|
214
|
+
with pl.Config(tbl_rows=200, tbl_cols=20, tbl_width_chars=200, float_precision=3):
|
|
215
|
+
print(table)
|
|
216
|
+
if args.out:
|
|
217
|
+
out = Path(args.out)
|
|
218
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
219
|
+
table.write_parquet(out / "sensitivity_table.parquet")
|
|
220
|
+
table.write_csv(out / "sensitivity_table.csv")
|
|
221
|
+
(out / "sensitivity_table.md").write_text(md)
|
|
222
|
+
print(out / "sensitivity_table.parquet")
|
|
223
|
+
print(out / "sensitivity_table.md")
|
|
224
|
+
else:
|
|
225
|
+
print(md)
|
|
155
226
|
return 0
|
|
156
227
|
|
|
157
228
|
|
|
@@ -189,10 +260,33 @@ def main(argv=None) -> int:
|
|
|
189
260
|
p.add_argument("-o", "--out", help="write .parquet, .csv or .geojson")
|
|
190
261
|
p.set_defaults(fn=cmd_hotspots)
|
|
191
262
|
|
|
192
|
-
p = sub.add_parser("inspect", help="coverage, GTFS versions, placement / drop counts")
|
|
263
|
+
p = sub.add_parser("inspect", help="coverage, GTFS versions, placement / drop counts, reference agreement")
|
|
193
264
|
common(p)
|
|
265
|
+
p.add_argument("--no-agreement", action="store_true", help="skip the analysis (evening vs line-level reference)")
|
|
194
266
|
p.set_defaults(fn=cmd_inspect)
|
|
195
267
|
|
|
268
|
+
p = sub.add_parser("compare", help="compare two periods of the same line: parquet + JSON + HTML")
|
|
269
|
+
common(p)
|
|
270
|
+
p.add_argument("--a", required=True, type=_period, help="period A, before (YYYY-MM-DD..YYYY-MM-DD)")
|
|
271
|
+
p.add_argument("--b", required=True, type=_period, help="period B, after (YYYY-MM-DD..YYYY-MM-DD)")
|
|
272
|
+
p.add_argument("-o", "--out", default="out_compare", help="output directory (default: out_compare)")
|
|
273
|
+
p.add_argument("--bootstrap", type=int, help="bootstrap draws (default: params.bootstrap)")
|
|
274
|
+
p.add_argument("--top", type=int, default=20, help="changes printed (0: all)")
|
|
275
|
+
p.add_argument("--lang", default="fr", choices=["fr", "en"])
|
|
276
|
+
p.add_argument("--title")
|
|
277
|
+
p.add_argument("--source", help="data credit shown in the page footer")
|
|
278
|
+
p.add_argument("--no-html", action="store_true")
|
|
279
|
+
p.set_defaults(fn=cmd_compare)
|
|
280
|
+
|
|
281
|
+
p = sub.add_parser("sensitivity-table", help="sensitivity suite over several configs -> one tidy table + markdown")
|
|
282
|
+
p.add_argument("configs", nargs="+", help="TOML configurations (one per line)")
|
|
283
|
+
p.add_argument("--segment-m", type=float, help="baseline segment length (default: each config's params.segment_m)")
|
|
284
|
+
p.add_argument("--dates", help="override select.dates for every config")
|
|
285
|
+
p.add_argument("--bootstrap", type=int, default=100)
|
|
286
|
+
p.add_argument("-o", "--out", help="directory for sensitivity_table.{parquet,csv,md}")
|
|
287
|
+
p.add_argument("-q", "--quiet", action="store_true")
|
|
288
|
+
p.set_defaults(fn=cmd_sensitivity_table)
|
|
289
|
+
|
|
196
290
|
args = ap.parse_args(argv)
|
|
197
291
|
return args.fn(args)
|
|
198
292
|
|