microsegments 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. {microsegments-0.1.0 → microsegments-0.2.0}/PKG-INFO +15 -3
  2. {microsegments-0.1.0 → microsegments-0.2.0}/README.md +14 -2
  3. microsegments-0.2.0/examples/stib1_platform.toml +23 -0
  4. microsegments-0.2.0/examples/stib55_full.toml +32 -0
  5. microsegments-0.2.0/examples/stib71_platform.toml +23 -0
  6. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/__init__.py +3 -2
  7. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/_version.py +2 -2
  8. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/aggregate.py +29 -8
  9. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/cli.py +102 -8
  10. microsegments-0.2.0/src/microsegments/compare.py +512 -0
  11. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/config.py +14 -2
  12. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/contract.py +27 -2
  13. microsegments-0.2.0/src/microsegments/hotspots.py +572 -0
  14. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/template.html +327 -52
  15. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/metrics.py +134 -14
  16. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/calendar.py +20 -2
  17. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/patterns.py +51 -11
  18. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/pipeline.py +28 -12
  19. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/report.py +71 -7
  20. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/schema.py +1 -1
  21. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/tune.py +111 -14
  22. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_aggregate.py +13 -1
  23. microsegments-0.2.0/tests/test_compare.py +83 -0
  24. microsegments-0.2.0/tests/test_hotspots_criteria.py +92 -0
  25. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_pipeline_cli.py +36 -0
  26. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_tune.py +3 -1
  27. microsegments-0.2.0/tests/test_versions_merge.py +60 -0
  28. microsegments-0.1.0/src/microsegments/hotspots.py +0 -337
  29. {microsegments-0.1.0 → microsegments-0.2.0}/.github/workflows/publish.yml +0 -0
  30. {microsegments-0.1.0 → microsegments-0.2.0}/.github/workflows/test.yml +0 -0
  31. {microsegments-0.1.0 → microsegments-0.2.0}/.gitignore +0 -0
  32. {microsegments-0.1.0 → microsegments-0.2.0}/LICENSE +0 -0
  33. {microsegments-0.1.0 → microsegments-0.2.0}/docs/report.jpg +0 -0
  34. {microsegments-0.1.0 → microsegments-0.2.0}/examples/gtfsrt_csv.toml +0 -0
  35. {microsegments-0.1.0 → microsegments-0.2.0}/examples/make_synth.py +0 -0
  36. {microsegments-0.1.0 → microsegments-0.2.0}/examples/stib55.toml +0 -0
  37. {microsegments-0.1.0 → microsegments-0.2.0}/examples/synth.toml +0 -0
  38. {microsegments-0.1.0 → microsegments-0.2.0}/pyproject.toml +0 -0
  39. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/__init__.py +0 -0
  40. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/export.py +0 -0
  41. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/html/leaflet.min.css +0 -0
  42. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/__init__.py +0 -0
  43. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/coverage.py +0 -0
  44. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/events.py +0 -0
  45. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/gtfsrt.py +0 -0
  46. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/ids.py +0 -0
  47. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/stib.py +0 -0
  48. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/tabular.py +0 -0
  49. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/io/timeutil.py +0 -0
  50. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/__init__.py +0 -0
  51. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/common.py +0 -0
  52. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/linear.py +0 -0
  53. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/mapmatch.py +0 -0
  54. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/locate/resample.py +0 -0
  55. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/__init__.py +0 -0
  56. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/geometry.py +0 -0
  57. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/network/keys.py +0 -0
  58. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/plot.py +0 -0
  59. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/py.typed +0 -0
  60. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/segments.py +0 -0
  61. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/simulate.py +0 -0
  62. {microsegments-0.1.0 → microsegments-0.2.0}/src/microsegments/tracks.py +0 -0
  63. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/buckets.parquet +0 -0
  64. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/drops.parquet +0 -0
  65. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/flen.parquet +0 -0
  66. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/obs_day.parquet +0 -0
  67. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/obs_dow.parquet +0 -0
  68. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/passages.parquet +0 -0
  69. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/summary.json +0 -0
  70. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/golden/validation.parquet +0 -0
  71. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/agency.parquet +0 -0
  72. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/calendar.parquet +0 -0
  73. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/calendar_dates.parquet +0 -0
  74. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/routes.parquet +0 -0
  75. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/shapes.parquet +0 -0
  76. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/stop_times.parquet +0 -0
  77. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/stops.parquet +0 -0
  78. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/gtfs/trips.parquet +0 -0
  79. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/punctuality/route55.parquet +0 -0
  80. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/speed55_ref.json +0 -0
  81. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250318.parquet +0 -0
  82. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250318.snaps.parquet +0 -0
  83. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250319.parquet +0 -0
  84. {microsegments-0.1.0 → microsegments-0.2.0}/tests/fixtures/stib55/vd55/20250319.snaps.parquet +0 -0
  85. {microsegments-0.1.0 → microsegments-0.2.0}/tests/gtfs_fixtures.py +0 -0
  86. {microsegments-0.1.0 → microsegments-0.2.0}/tests/regression/make_golden.py +0 -0
  87. {microsegments-0.1.0 → microsegments-0.2.0}/tests/t4_fixtures.py +0 -0
  88. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_coverage.py +0 -0
  89. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_hotspots.py +0 -0
  90. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_events.py +0 -0
  91. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_gtfsrt.py +0 -0
  92. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_stib.py +0 -0
  93. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_io_tabular.py +0 -0
  94. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_locate_linear.py +0 -0
  95. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_locate_mapmatch.py +0 -0
  96. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_metrics.py +0 -0
  97. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_network_geometry.py +0 -0
  98. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_network_patterns.py +0 -0
  99. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_regression_stib55.py +0 -0
  100. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_report_html.py +0 -0
  101. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_scaffold.py +0 -0
  102. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_segments.py +0 -0
  103. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_simulate.py +0 -0
  104. {microsegments-0.1.0 → microsegments-0.2.0}/tests/test_tracks.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: microsegments
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: Where do transit vehicles linger? Count vehicle position observations per micro-segment of a line, from GTFS + AVL / GTFS-RT positions.
5
5
  Project-URL: Homepage, https://github.com/GaspardMerten/microsegments
6
6
  Project-URL: Repository, https://github.com/GaspardMerten/microsegments
@@ -132,7 +132,9 @@ Stages are usable on their own: `read_observations`, `build_network`, `segment`,
132
132
  microsegments run CONFIG -o OUT [--lang en] [--segment-m 20] [--dates A..B] [--source "credit"]
133
133
  microsegments tune CONFIG [--lengths 10,20,30,50] [--sensitivity] [-o OUT]
134
134
  microsegments hotspots CONFIG [-o hotspots.geojson|.csv|.parquet]
135
- microsegments inspect CONFIG # coverage per day, GTFS versions, placement / drop counts
135
+ microsegments inspect CONFIG # coverage per day, GTFS versions, placement, evening vs line reference
136
+ microsegments compare CONFIG --a 2025-02-17..2025-03-31 --b 2025-04-01..2025-05-15 -o OUT
137
+ microsegments sensitivity-table CONFIG [CONFIG ...] -o OUT # one tidy table + markdown over several lines
136
138
  ```
137
139
 
138
140
  ## What the numbers mean
@@ -169,7 +171,8 @@ fixed per metric (p98 over 6–21 h), so hours and weekdays compare directly.
169
171
  split-half reliability, hotspot localisation spread and Jaccard) and picks the smallest L with reliability
170
172
  ≥ 0.8, spread ≤ 30 m and deviance within one standard error of the minimum. 30 m is a good default for
171
173
  20 s polling at urban speeds; go shorter only with dense GTFS-RT fixes and many days. `--sensitivity`
172
- re-runs phase offsets, gap caps, stop zones, references and passage sources.
174
+ re-runs phase offsets, gap caps (30, 40, 60, ∞ s), stop zones, references, passage sources and segment
175
+ lengths 15 / 30 / 60 m around the baseline.
173
176
 
174
177
  ## Hotspots
175
178
 
@@ -179,6 +182,15 @@ Adjacent bins merge, never across a stop / running border. Classes: *infrastruct
179
182
  *congestion* (peak only), *mixed*. Needs at least 5 included days. Export as GeoJSON with
180
183
  `microsegments hotspots CONFIG -o hotspots.geojson`.
181
184
 
185
+ ## Comparing two periods
186
+
187
+ `microsegments compare` (or `ms.compare(analysis_a, analysis_b)`) compares observations per passage
188
+ segment by segment and hour by hour, each period with its own days and references, with a day bootstrap
189
+ of the difference. Segments are matched by key (or by stop pair when a new GTFS version moved the shape a
190
+ little); the others are listed and left out. Adjacent significant segments form stretches, ranked by the
191
+ change in vehicle time per day; stretches whose frequency changed by more than 20 % are flagged, since a
192
+ timetable change can explain part of the difference. The HTML page gets a "compare two periods" view.
193
+
182
194
  ## License
183
195
 
184
196
  MIT. The bundled Leaflet CSS is BSD-2-Clause.
@@ -94,7 +94,9 @@ Stages are usable on their own: `read_observations`, `build_network`, `segment`,
94
94
  microsegments run CONFIG -o OUT [--lang en] [--segment-m 20] [--dates A..B] [--source "credit"]
95
95
  microsegments tune CONFIG [--lengths 10,20,30,50] [--sensitivity] [-o OUT]
96
96
  microsegments hotspots CONFIG [-o hotspots.geojson|.csv|.parquet]
97
- microsegments inspect CONFIG # coverage per day, GTFS versions, placement / drop counts
97
+ microsegments inspect CONFIG # coverage per day, GTFS versions, placement, evening vs line reference
98
+ microsegments compare CONFIG --a 2025-02-17..2025-03-31 --b 2025-04-01..2025-05-15 -o OUT
99
+ microsegments sensitivity-table CONFIG [CONFIG ...] -o OUT # one tidy table + markdown over several lines
98
100
  ```
99
101
 
100
102
  ## What the numbers mean
@@ -131,7 +133,8 @@ fixed per metric (p98 over 6–21 h), so hours and weekdays compare directly.
131
133
  split-half reliability, hotspot localisation spread and Jaccard) and picks the smallest L with reliability
132
134
  ≥ 0.8, spread ≤ 30 m and deviance within one standard error of the minimum. 30 m is a good default for
133
135
  20 s polling at urban speeds; go shorter only with dense GTFS-RT fixes and many days. `--sensitivity`
134
- re-runs phase offsets, gap caps, stop zones, references and passage sources.
136
+ re-runs phase offsets, gap caps (30, 40, 60, ∞ s), stop zones, references, passage sources and segment
137
+ lengths 15 / 30 / 60 m around the baseline.
135
138
 
136
139
  ## Hotspots
137
140
 
@@ -141,6 +144,15 @@ Adjacent bins merge, never across a stop / running border. Classes: *infrastruct
141
144
  *congestion* (peak only), *mixed*. Needs at least 5 included days. Export as GeoJSON with
142
145
  `microsegments hotspots CONFIG -o hotspots.geojson`.
143
146
 
147
+ ## Comparing two periods
148
+
149
+ `microsegments compare` (or `ms.compare(analysis_a, analysis_b)`) compares observations per passage
150
+ segment by segment and hour by hour, each period with its own days and references, with a day bootstrap
151
+ of the difference. Segments are matched by key (or by stop pair when a new GTFS version moved the shape a
152
+ little); the others are listed and left out. Adjacent significant segments form stretches, ranked by the
153
+ change in vehicle time per day; stretches whose frequency changed by more than 20 % are flagged, since a
154
+ timetable change can explain part of the difference. The HTML page gets a "compare two periods" view.
155
+
144
156
  ## License
145
157
 
146
158
  MIT. The bundled Leaflet CSS is BSD-2-Clause.
@@ -0,0 +1,23 @@
1
+ # LOCAL EXAMPLE (not shipped data): STIB line 1 from the StibMicrosegments platform's local ingest
2
+ # (raw/vd/date=.../{vd,snaps}.parquet with a line column, raw/punctuality, gtfs/index.parquet + gtfs/feeds/<sha>).
3
+ # microsegments sensitivity-table examples/stib55_full.toml examples/stib71_platform.toml examples/stib1_platform.toml -o out/sens
4
+
5
+ [input]
6
+ paths = ["/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/vd/date=2025-0[2-5]-*/*.parquet"]
7
+ kind = "stib"
8
+ events = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/punctuality/date=2025-0[2-5]-*/p.parquet"
9
+
10
+ [gtfs]
11
+ index = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/gtfs/index.parquet"
12
+
13
+ [select]
14
+ route = "1"
15
+ dates = "2025-03-03..2025-03-28"
16
+ weekdays = [0, 1, 2, 3, 4]
17
+
18
+ [params]
19
+ segment_m = 30
20
+ bootstrap = 200
21
+
22
+ [report]
23
+ names = "title"
@@ -0,0 +1,32 @@
1
+ # LOCAL EXAMPLE (not shipped data): STIB tram 55, weekdays 17 Feb - 15 May 2025, on the author's machine.
2
+ # Paths point at ~/Documents/Dev/CoDE/BrusselsTransitStuck/data2025 (MobilityTwin stib/vehicle-distance as
3
+ # compact daily parquet, STIB punctuality, one GTFS feed); adapt them to your copy of the data.
4
+ # microsegments run examples/stib55_full.toml -o out/stib55_full
5
+ # microsegments tune examples/stib55_full.toml --sensitivity
6
+
7
+ [input]
8
+ paths = ["/home/gaspard/Documents/Dev/CoDE/BrusselsTransitStuck/data2025/stib/vd55/*.parquet"]
9
+ kind = "stib"
10
+ events = "/home/gaspard/Documents/Dev/CoDE/BrusselsTransitStuck/data2025/stib/punctuality/2025-0[2-5]-*.parquet"
11
+
12
+ [input.linear]
13
+ direction_kind = "terminus_stop"
14
+ id_normaliser = "leading_digits"
15
+ feed_length = "auto"
16
+
17
+ [gtfs]
18
+ # one feed (calendar 14 Apr - 11 May 2025); other dates take the patterns of the same weekday
19
+ path = "/home/gaspard/Documents/Dev/CoDE/BrusselsTransitStuck/data2025/gtfs"
20
+
21
+ [select]
22
+ route = "55"
23
+ dates = "2025-02-17..2025-05-15"
24
+ weekdays = [0, 1, 2, 3, 4]
25
+ exclude_dates = ["2025-04-21", "2025-05-01"] # Easter Monday, Labour Day
26
+
27
+ [params]
28
+ segment_m = 30
29
+ bootstrap = 200
30
+
31
+ [report]
32
+ names = "title" # "GARE DU NORD" -> "Gare du Nord"
@@ -0,0 +1,23 @@
1
+ # LOCAL EXAMPLE (not shipped data): STIB line 71 from the StibMicrosegments platform's local ingest
2
+ # (raw/vd/date=.../{vd,snaps}.parquet with a line column, raw/punctuality, gtfs/index.parquet + gtfs/feeds/<sha>).
3
+ # microsegments sensitivity-table examples/stib55_full.toml examples/stib71_platform.toml examples/stib1_platform.toml -o out/sens
4
+
5
+ [input]
6
+ paths = ["/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/vd/date=2025-0[2-5]-*/*.parquet"]
7
+ kind = "stib"
8
+ events = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/raw/punctuality/date=2025-0[2-5]-*/p.parquet"
9
+
10
+ [gtfs]
11
+ index = "/home/gaspard/Documents/Dev/CoDE/StibMicrosegments/data/ms/gtfs/index.parquet"
12
+
13
+ [select]
14
+ route = "71"
15
+ dates = "2025-03-03..2025-03-28"
16
+ weekdays = [0, 1, 2, 3, 4]
17
+
18
+ [params]
19
+ segment_m = 30
20
+ bootstrap = 200
21
+
22
+ [report]
23
+ names = "title"
@@ -15,6 +15,7 @@ except ImportError: # pragma: no cover
15
15
  __version__ = "0.0.0"
16
16
 
17
17
  from .aggregate import count
18
+ from .compare import Comparison, compare, run_compare
18
19
  from .config import Config
19
20
  from .hotspots import hotspots
20
21
  from .html import export
@@ -28,8 +29,8 @@ from .segments import segment
28
29
  from .simulate import simulate
29
30
  from . import tune
30
31
 
31
- __all__ = ["Analysis", "Config", "Flag", "RunResult", "__version__", "analyse", "build_network", "count", "export",
32
- "hotspots", "plot", "prepare", "read_observations", "run", "segment", "simulate", "to_contract", "tune"]
32
+ __all__ = ["Analysis", "Comparison", "Config", "Flag", "RunResult", "__version__", "analyse", "build_network", "compare", "count", "export",
33
+ "hotspots", "plot", "prepare", "read_observations", "run", "run_compare", "segment", "simulate", "to_contract", "tune"]
33
34
 
34
35
 
35
36
  def __getattr__(name: str):
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.1.0'
22
- __version_tuple__ = version_tuple = (0, 1, 0)
21
+ __version__ = version = '0.2.0'
22
+ __version_tuple__ = version_tuple = (0, 2, 0)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -6,9 +6,11 @@ upper bound closed, as in the prototype: a vehicle standing at stop B is at the
6
6
  A row at ``pos_m <= 0`` of link i > 0 goes to the last segment of link i - 1, i.e. the end of the
7
7
  link arriving at that stop. Positions beyond the link end are clipped to its last segment.
8
8
 
9
- Per-vehicle feeds (GTFS-RT with irregular / dense fixes) are thinned to at most one observation
10
- per ``(track_id, floor(ts / tick_s))`` (the last fix of each tick), so counts stay comparable with
11
- a snapshot feed polled every ``tick_s``.
9
+ Per-vehicle feeds (GTFS-RT with irregular / dense fixes) are put on a ``tick_s`` grid so counts stay
10
+ comparable with a snapshot feed polled every ``tick_s``: by default ``locate.resample.resample`` (each
11
+ track sampled at the grid instants between its fixes, nearer fix; unbiased at track ends and for
12
+ sparse fixes), or ``method="thin"``: at most one observation per ``(track_id, floor(ts / tick_s))``
13
+ (the last fix of each tick; under-counts when fixes are sparser than the tick).
12
14
  """
13
15
  from __future__ import annotations
14
16
 
@@ -74,15 +76,34 @@ def assign(placed: pl.DataFrame, segments: pl.DataFrame) -> pl.DataFrame:
74
76
  )
75
77
 
76
78
 
79
+ def per_vehicle_grid(placed: pl.DataFrame, tick_s: float = 20.0, method: str = "resample") -> pl.DataFrame:
80
+ """Counted rows of a per-vehicle feed on the ``tick_s`` grid (``method`` "resample" or "thin")."""
81
+ if method == "thin":
82
+ return thin(placed.filter(pl.col("count").fill_null(True)), tick_s)
83
+ if method != "resample":
84
+ raise ValueError(f"unknown per-vehicle method {method!r}")
85
+ from .locate.resample import resample
86
+ df = placed
87
+ if "s_m" not in df.columns:
88
+ df = df.with_columns(pl.col("pos_m").alias("s_m"))
89
+ elif "pos_m" in df.columns and df["s_m"].null_count():
90
+ df = df.with_columns(pl.coalesce("s_m", pl.col("pos_m").cast(df.schema["s_m"])).alias("s_m"))
91
+ if "count" not in df.columns:
92
+ df = df.with_columns(pl.lit(True).alias("count"))
93
+ # resample first (non-counted fixes still bound the intervals), then keep counted instants
94
+ return resample(df, tick_s).filter(pl.col("count").fill_null(True))
95
+
96
+
77
97
  def count(placed: pl.DataFrame, segments: pl.DataFrame, tick_s: float = 20.0,
78
- per_vehicle: bool = False) -> pl.DataFrame:
98
+ per_vehicle: bool = False, method: str = "resample") -> pl.DataFrame:
79
99
  """Placed -> Cube (service_date, hour, pattern_uid, seg_key, obs).
80
100
 
81
- Σ obs equals the number of counted (and, for per-vehicle feeds, thinned) rows whose link has
82
- segments."""
83
- df = placed.filter(pl.col("count").fill_null(True))
101
+ Σ obs equals the number of counted (for per-vehicle feeds: resampled, or thinned with
102
+ ``method="thin"``) rows whose link has segments."""
84
103
  if per_vehicle:
85
- df = thin(df, tick_s)
104
+ df = per_vehicle_grid(placed, tick_s, method)
105
+ else:
106
+ df = placed.filter(pl.col("count").fill_null(True))
86
107
  df = assign(df.select("ts", "service_date", "hour", "pattern_uid", "link_idx", "pos_m", "track_id"), segments)
87
108
  cube = (df.filter(pl.col("seg_key").is_not_null())
88
109
  .group_by("service_date", "hour", "pattern_uid", "seg_key")
@@ -1,4 +1,5 @@
1
- """Command line: ``microsegments run | tune | hotspots | inspect CONFIG``."""
1
+ """Command line: ``microsegments run | tune | hotspots | inspect | compare CONFIG``, ``microsegments
2
+ sensitivity-table CONFIG...``."""
2
3
  from __future__ import annotations
3
4
 
4
5
  import argparse
@@ -25,13 +26,16 @@ def _log(args):
25
26
  return (lambda s: print(s, file=sys.stderr)) if not getattr(args, "quiet", False) else None
26
27
 
27
28
 
28
- def _hotspot_table(hs: pl.DataFrame, tick_s: float) -> pl.DataFrame:
29
+ def _hotspot_table(hs: pl.DataFrame, tick_s: float, names: dict[str, str] | None = None) -> pl.DataFrame:
29
30
  if hs is None or hs.height == 0:
30
31
  return pl.DataFrame()
32
+ nm = names or {}
33
+ fix = lambda c: pl.col(c).replace(nm) if nm else pl.col(c) # noqa: E731
34
+ crit = pl.col("criterion") if "criterion" in hs.columns else pl.lit("peak")
31
35
  return hs.select(
32
36
  "direction_id", "rank",
33
- pl.concat_str([pl.col("from_stop_name"), pl.lit(" -> "), pl.col("to_stop_name")]).alias("stretch"),
34
- pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone", "kind",
37
+ pl.concat_str([fix("from_stop_name"), pl.lit(" -> "), fix("to_stop_name")]).alias("stretch"),
38
+ pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone", "kind", crit.alias("criterion"),
35
39
  pl.col("hours").cast(pl.List(pl.Utf8)).list.join(",").alias("hours"),
36
40
  pl.col("excess_per_passage").round(2).alias("excess_obs_per_veh"),
37
41
  (pl.col("excess_per_passage") * tick_s).round(0).alias("approx_s_per_veh"),
@@ -57,9 +61,10 @@ def cmd_hotspots(args) -> int:
57
61
  from .pipeline import run
58
62
  cfg = _cfg(args)
59
63
  res = run(cfg, log=_log(args))
60
- t = _hotspot_table(res.hotspots, cfg.params.tick_s)
64
+ from .report import name_map
65
+ t = _hotspot_table(res.hotspots, cfg.params.tick_s, name_map(cfg.report.names, res.network))
61
66
  if t.height == 0:
62
- print("no hotspot")
67
+ print(f"no hotspot ({res.hotspots_status})")
63
68
  return 0
64
69
  with pl.Config(tbl_rows=200, tbl_cols=20, fmt_str_lengths=60, tbl_width_chars=200):
65
70
  print(t)
@@ -95,7 +100,9 @@ def cmd_tune(args) -> int:
95
100
  tr.table.write_parquet(out / "tune.parquet")
96
101
  sr = None
97
102
  if args.sensitivity:
98
- sr = tune.sensitivity(pre.placed, pre.segment_fn(), tr.recommended, B=args.bootstrap, **kw)
103
+ kw["params"] = replace(cfg.params, segment_m=float(tr.recommended))
104
+ sr = tune.sensitivity(pre.placed, pre.segment_fn(), tr.recommended, B=args.bootstrap,
105
+ coverage_fn=pre.coverage_fn(), **kw)
99
106
  with pl.Config(tbl_rows=50, tbl_cols=20, tbl_width_chars=200, float_precision=3):
100
107
  print(sr.summary)
101
108
  print(sr.hotspots)
@@ -152,6 +159,70 @@ def cmd_inspect(args) -> int:
152
159
  p = pre.passages
153
160
  print(f"\npassages: {p['n'].sum():,.0f} link passages"
154
161
  + (f" (feed {p['n_feed'].sum():,.0f}, events {p['n_events'].sum():,.0f})" if p["n_events"].null_count() < p.height else ""))
162
+ if not args.no_agreement:
163
+ from .metrics import reference_agreement
164
+ from .pipeline import run
165
+ res = run(cfg, prepared=pre, hotspot_kw={"B": min(cfg.params.bootstrap, 100)})
166
+ ag = reference_agreement(res.analysis, res.hotspots)
167
+ print(f"\nreferences: evening {cfg.params.reference_hours[0]}-{cfg.params.reference_hours[1]} h vs line level "
168
+ "(median running obs / passage / m, 6-23 h); day-band excess profiles:")
169
+ with pl.Config(tbl_rows=10, tbl_cols=20, tbl_width_chars=200, float_precision=3):
170
+ print(ag)
171
+ return 0
172
+
173
+
174
+ def _period(s: str) -> str:
175
+ a, b = s.split("..")
176
+ import datetime as dt
177
+ dt.date.fromisoformat(a), dt.date.fromisoformat(b)
178
+ return s
179
+
180
+
181
+ def cmd_compare(args) -> int:
182
+ from .compare import run_compare
183
+ cfg = _cfg(args)
184
+ res = run_compare(cfg, args.a, args.b, B=args.bootstrap, log=_log(args))
185
+ paths = res.save(args.out, html=not args.no_html, lang=args.lang, title=args.title, source=args.source)
186
+ st = res.comparison.stretches
187
+ if st.height:
188
+ from .report import name_map
189
+ nm = name_map(cfg.report.names, res.prepared.network)
190
+ fix = lambda c: pl.col(c).replace(nm) if nm else pl.col(c) # noqa: E731
191
+ t = st.select("direction_id", "rank",
192
+ pl.concat_str([fix("from_stop_name"), pl.lit(" -> "), fix("to_stop_name")]).alias("stretch"),
193
+ pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone",
194
+ pl.col("hours").cast(pl.List(pl.Utf8)).list.join(",").alias("hours"),
195
+ pl.col("delta_s").round(0).alias("delta_s_per_veh"),
196
+ (pl.col("veh_time_s_per_day") / 60).round(1).alias("veh_min_per_day"),
197
+ pl.col("passages_per_h_a").round(1).alias("veh_h_a"), pl.col("passages_per_h_b").round(1).alias("veh_h_b"),
198
+ "frequency_changed")
199
+ with pl.Config(tbl_rows=60, tbl_cols=20, fmt_str_lengths=60, tbl_width_chars=220):
200
+ print(t.head(args.top) if args.top else t)
201
+ else:
202
+ print("no significant change")
203
+ for p in paths.values():
204
+ print(p)
205
+ return 0
206
+
207
+
208
+ def cmd_sensitivity_table(args) -> int:
209
+ from . import tune
210
+ configs = [Config.from_toml(c) for c in args.configs]
211
+ if args.segment_m:
212
+ configs = [replace(c, params=replace(c.params, segment_m=float(args.segment_m))) for c in configs]
213
+ table, md = tune.sensitivity_table(configs, B=args.bootstrap, dates=args.dates, log=_log(args))
214
+ with pl.Config(tbl_rows=200, tbl_cols=20, tbl_width_chars=200, float_precision=3):
215
+ print(table)
216
+ if args.out:
217
+ out = Path(args.out)
218
+ out.mkdir(parents=True, exist_ok=True)
219
+ table.write_parquet(out / "sensitivity_table.parquet")
220
+ table.write_csv(out / "sensitivity_table.csv")
221
+ (out / "sensitivity_table.md").write_text(md)
222
+ print(out / "sensitivity_table.parquet")
223
+ print(out / "sensitivity_table.md")
224
+ else:
225
+ print(md)
155
226
  return 0
156
227
 
157
228
 
@@ -189,10 +260,33 @@ def main(argv=None) -> int:
189
260
  p.add_argument("-o", "--out", help="write .parquet, .csv or .geojson")
190
261
  p.set_defaults(fn=cmd_hotspots)
191
262
 
192
- p = sub.add_parser("inspect", help="coverage, GTFS versions, placement / drop counts")
263
+ p = sub.add_parser("inspect", help="coverage, GTFS versions, placement / drop counts, reference agreement")
193
264
  common(p)
265
+ p.add_argument("--no-agreement", action="store_true", help="skip the analysis (evening vs line-level reference)")
194
266
  p.set_defaults(fn=cmd_inspect)
195
267
 
268
+ p = sub.add_parser("compare", help="compare two periods of the same line: parquet + JSON + HTML")
269
+ common(p)
270
+ p.add_argument("--a", required=True, type=_period, help="period A, before (YYYY-MM-DD..YYYY-MM-DD)")
271
+ p.add_argument("--b", required=True, type=_period, help="period B, after (YYYY-MM-DD..YYYY-MM-DD)")
272
+ p.add_argument("-o", "--out", default="out_compare", help="output directory (default: out_compare)")
273
+ p.add_argument("--bootstrap", type=int, help="bootstrap draws (default: params.bootstrap)")
274
+ p.add_argument("--top", type=int, default=20, help="changes printed (0: all)")
275
+ p.add_argument("--lang", default="fr", choices=["fr", "en"])
276
+ p.add_argument("--title")
277
+ p.add_argument("--source", help="data credit shown in the page footer")
278
+ p.add_argument("--no-html", action="store_true")
279
+ p.set_defaults(fn=cmd_compare)
280
+
281
+ p = sub.add_parser("sensitivity-table", help="sensitivity suite over several configs -> one tidy table + markdown")
282
+ p.add_argument("configs", nargs="+", help="TOML configurations (one per line)")
283
+ p.add_argument("--segment-m", type=float, help="baseline segment length (default: each config's params.segment_m)")
284
+ p.add_argument("--dates", help="override select.dates for every config")
285
+ p.add_argument("--bootstrap", type=int, default=100)
286
+ p.add_argument("-o", "--out", help="directory for sensitivity_table.{parquet,csv,md}")
287
+ p.add_argument("-q", "--quiet", action="store_true")
288
+ p.set_defaults(fn=cmd_sensitivity_table)
289
+
196
290
  args = ap.parse_args(argv)
197
291
  return args.fn(args)
198
292