microsegments 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- microsegments/__init__.py +42 -0
- microsegments/_version.py +24 -0
- microsegments/aggregate.py +101 -0
- microsegments/cli.py +201 -0
- microsegments/config.py +142 -0
- microsegments/contract.py +42 -0
- microsegments/hotspots.py +337 -0
- microsegments/html/__init__.py +6 -0
- microsegments/html/export.py +96 -0
- microsegments/html/leaflet.min.css +2 -0
- microsegments/html/template.html +863 -0
- microsegments/io/__init__.py +9 -0
- microsegments/io/coverage.py +114 -0
- microsegments/io/events.py +98 -0
- microsegments/io/gtfsrt.py +168 -0
- microsegments/io/ids.py +34 -0
- microsegments/io/stib.py +217 -0
- microsegments/io/tabular.py +188 -0
- microsegments/io/timeutil.py +84 -0
- microsegments/locate/__init__.py +47 -0
- microsegments/locate/common.py +149 -0
- microsegments/locate/linear.py +317 -0
- microsegments/locate/mapmatch.py +390 -0
- microsegments/locate/resample.py +71 -0
- microsegments/metrics.py +520 -0
- microsegments/network/__init__.py +10 -0
- microsegments/network/calendar.py +218 -0
- microsegments/network/geometry.py +224 -0
- microsegments/network/keys.py +97 -0
- microsegments/network/patterns.py +328 -0
- microsegments/pipeline.py +308 -0
- microsegments/plot.py +476 -0
- microsegments/py.typed +0 -0
- microsegments/report.py +258 -0
- microsegments/schema.py +257 -0
- microsegments/segments.py +253 -0
- microsegments/simulate.py +612 -0
- microsegments/tracks.py +391 -0
- microsegments/tune.py +518 -0
- microsegments-0.1.0.dist-info/METADATA +184 -0
- microsegments-0.1.0.dist-info/RECORD +44 -0
- microsegments-0.1.0.dist-info/WHEEL +4 -0
- microsegments-0.1.0.dist-info/entry_points.txt +2 -0
- microsegments-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""microsegments: count where transit vehicles are observed along a line, per micro-segment.
|
|
2
|
+
|
|
3
|
+
Quick start::
|
|
4
|
+
|
|
5
|
+
import microsegments as ms
|
|
6
|
+
res = ms.run(ms.Config.from_toml("ms.toml"))
|
|
7
|
+
ms.export(res, "report.html")
|
|
8
|
+
ms.plot.matrix(res, direction_id=0)
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
try:
|
|
13
|
+
from ._version import __version__
|
|
14
|
+
except ImportError: # pragma: no cover
|
|
15
|
+
__version__ = "0.0.0"
|
|
16
|
+
|
|
17
|
+
from .aggregate import count
|
|
18
|
+
from .config import Config
|
|
19
|
+
from .hotspots import hotspots
|
|
20
|
+
from .html import export
|
|
21
|
+
from .io import read_observations
|
|
22
|
+
from .metrics import Analysis, analyse
|
|
23
|
+
from .network import build_network
|
|
24
|
+
from .pipeline import RunResult, prepare, run
|
|
25
|
+
from .report import to_contract
|
|
26
|
+
from .schema import Flag
|
|
27
|
+
from .segments import segment
|
|
28
|
+
from .simulate import simulate
|
|
29
|
+
from . import tune
|
|
30
|
+
|
|
31
|
+
__all__ = ["Analysis", "Config", "Flag", "RunResult", "__version__", "analyse", "build_network", "count", "export",
|
|
32
|
+
"hotspots", "plot", "prepare", "read_observations", "run", "segment", "simulate", "to_contract", "tune"]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def __getattr__(name: str):
|
|
36
|
+
# matplotlib is optional: import the plot module on first use
|
|
37
|
+
if name == "plot":
|
|
38
|
+
import importlib
|
|
39
|
+
mod = importlib.import_module(".plot", __name__)
|
|
40
|
+
globals()["plot"] = mod
|
|
41
|
+
return mod
|
|
42
|
+
raise AttributeError(f"module 'microsegments' has no attribute {name!r}")
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.1.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 0)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = None
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Placed observations -> Cube: how many times a vehicle was observed in each micro-segment.
|
|
2
|
+
|
|
3
|
+
Only rows with ``count`` True (or null) are counted. A row is assigned to the segment whose
|
|
4
|
+
link-relative interval ``(offset_m, offset_m + len_m]`` contains its ``pos_m`` (lower bound open,
|
|
5
|
+
upper bound closed, as in the prototype: a vehicle standing at stop B is at the end of link A->B).
|
|
6
|
+
A row at ``pos_m <= 0`` of link i > 0 goes to the last segment of link i - 1, i.e. the end of the
|
|
7
|
+
link arriving at that stop. Positions beyond the link end are clipped to its last segment.
|
|
8
|
+
|
|
9
|
+
Per-vehicle feeds (GTFS-RT with irregular / dense fixes) are thinned to at most one observation
|
|
10
|
+
per ``(track_id, floor(ts / tick_s))`` (the last fix of each tick), so counts stay comparable with
|
|
11
|
+
a snapshot feed polled every ``tick_s``.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
import polars as pl
|
|
17
|
+
|
|
18
|
+
from .schema import CUBE, conform
|
|
19
|
+
|
|
20
|
+
_SCALE = 1e7 # metres; composite sort key = group * _SCALE + position (links are far shorter)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def thin(placed: pl.DataFrame, tick_s: float = 20.0) -> pl.DataFrame:
|
|
24
|
+
"""Keep the last fix per (track_id, tick bucket). Rows without track_id are kept as they are."""
|
|
25
|
+
tick_ms = max(int(round(tick_s * 1000)), 1)
|
|
26
|
+
df = placed.with_columns((pl.col("ts").dt.epoch("ms") // tick_ms).alias("_tick"))
|
|
27
|
+
has = df.filter(pl.col("track_id").is_not_null())
|
|
28
|
+
none = df.filter(pl.col("track_id").is_null())
|
|
29
|
+
has = has.sort("ts", maintain_order=True).unique(subset=["track_id", "_tick"], keep="last", maintain_order=True)
|
|
30
|
+
return pl.concat([has, none], how="vertical").drop("_tick")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def assign(placed: pl.DataFrame, segments: pl.DataFrame) -> pl.DataFrame:
|
|
34
|
+
"""Add ``seg_idx`` and ``seg_key`` to placed rows (null when the row's link has no segment).
|
|
35
|
+
|
|
36
|
+
Vectorised: one searchsorted over all (pattern, link) groups at once."""
|
|
37
|
+
seg = (segments.select("pattern_uid", "link_idx", "seg_idx", "seg_key", "offset_m", "len_m")
|
|
38
|
+
.with_columns(pl.col("link_idx").cast(pl.Int16))
|
|
39
|
+
.sort("pattern_uid", "link_idx", "offset_m"))
|
|
40
|
+
groups = (seg.group_by("pattern_uid", "link_idx", maintain_order=True)
|
|
41
|
+
.agg(pl.len().alias("_n"))
|
|
42
|
+
.with_row_index("_gid"))
|
|
43
|
+
gn = groups["_n"].to_numpy().astype(np.int64)
|
|
44
|
+
g_last = np.cumsum(gn) - 1
|
|
45
|
+
g_first = g_last - gn + 1
|
|
46
|
+
seg = seg.join(groups.select("pattern_uid", "link_idx", "_gid"), on=["pattern_uid", "link_idx"], how="left")
|
|
47
|
+
seg_comp = seg["_gid"].to_numpy().astype(np.float64) * _SCALE + (seg["offset_m"] + seg["len_m"]).to_numpy()
|
|
48
|
+
|
|
49
|
+
# pos <= 0 on link i > 0 belongs to the end of link i - 1.
|
|
50
|
+
back = (pl.col("pos_m") <= 0) & (pl.col("link_idx") > 0)
|
|
51
|
+
df = placed.with_columns(
|
|
52
|
+
pl.when(back).then(pl.col("link_idx") - 1).otherwise(pl.col("link_idx")).cast(pl.Int16).alias("_li"),
|
|
53
|
+
pl.when(back).then(pl.lit(np.inf)).otherwise(pl.col("pos_m").cast(pl.Float64)).alias("_pos"),
|
|
54
|
+
)
|
|
55
|
+
df = df.join(groups.select(pl.col("pattern_uid"), pl.col("link_idx").alias("_li"), "_gid"),
|
|
56
|
+
on=["pattern_uid", "_li"], how="left", maintain_order="left")
|
|
57
|
+
ok = df["_gid"].is_not_null().to_numpy()
|
|
58
|
+
gid = df["_gid"].fill_null(0).to_numpy().astype(np.int64)
|
|
59
|
+
pos = df["_pos"].fill_null(0.0).to_numpy().astype(np.float64)
|
|
60
|
+
pos = np.clip(np.nan_to_num(pos, nan=0.0, posinf=_SCALE / 2), -1.0, _SCALE / 2)
|
|
61
|
+
n = len(df)
|
|
62
|
+
if len(seg_comp) == 0:
|
|
63
|
+
ok = np.zeros(n, bool)
|
|
64
|
+
idx = np.zeros(n, np.int64)
|
|
65
|
+
else:
|
|
66
|
+
idx = np.searchsorted(seg_comp, gid * _SCALE + pos, side="left")
|
|
67
|
+
idx = np.clip(idx, g_first[gid], g_last[gid])
|
|
68
|
+
seg_idx = seg["seg_idx"].to_numpy()[idx] if len(seg_comp) else np.zeros(n, np.int32)
|
|
69
|
+
seg_key = seg["seg_key"].gather(idx) if len(seg_comp) else pl.Series([None] * n, dtype=pl.Utf8)
|
|
70
|
+
ok_s = pl.Series(ok)
|
|
71
|
+
return df.drop("_li", "_pos", "_gid").with_columns(
|
|
72
|
+
pl.when(ok_s).then(pl.Series(seg_idx, dtype=pl.Int32)).otherwise(None).alias("seg_idx"),
|
|
73
|
+
pl.when(ok_s).then(seg_key).otherwise(None).alias("seg_key"),
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def count(placed: pl.DataFrame, segments: pl.DataFrame, tick_s: float = 20.0,
|
|
78
|
+
per_vehicle: bool = False) -> pl.DataFrame:
|
|
79
|
+
"""Placed -> Cube (service_date, hour, pattern_uid, seg_key, obs).
|
|
80
|
+
|
|
81
|
+
Σ obs equals the number of counted (and, for per-vehicle feeds, thinned) rows whose link has
|
|
82
|
+
segments."""
|
|
83
|
+
df = placed.filter(pl.col("count").fill_null(True))
|
|
84
|
+
if per_vehicle:
|
|
85
|
+
df = thin(df, tick_s)
|
|
86
|
+
df = assign(df.select("ts", "service_date", "hour", "pattern_uid", "link_idx", "pos_m", "track_id"), segments)
|
|
87
|
+
cube = (df.filter(pl.col("seg_key").is_not_null())
|
|
88
|
+
.group_by("service_date", "hour", "pattern_uid", "seg_key")
|
|
89
|
+
.agg(pl.len().alias("obs"))
|
|
90
|
+
.sort("service_date", "hour", "pattern_uid", "seg_key"))
|
|
91
|
+
return conform(cube, CUBE)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def link_totals(cube: pl.DataFrame, segments: pl.DataFrame,
|
|
95
|
+
by: tuple[str, ...] = ("service_date", "hour")) -> pl.DataFrame:
|
|
96
|
+
"""Sum the cube per link: (pattern_uid, link_key, *by, obs). Invariant to segment length and phase."""
|
|
97
|
+
keys = segments.select("pattern_uid", "seg_key", "link_key").unique(["pattern_uid", "seg_key"])
|
|
98
|
+
return (cube.join(keys, on=["pattern_uid", "seg_key"], how="left")
|
|
99
|
+
.group_by("pattern_uid", "link_key", *by)
|
|
100
|
+
.agg(pl.col("obs").sum().cast(pl.UInt64))
|
|
101
|
+
.sort("pattern_uid", "link_key", *by))
|
microsegments/cli.py
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Command line: ``microsegments run | tune | hotspots | inspect CONFIG``."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
from dataclasses import replace
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import polars as pl
|
|
11
|
+
|
|
12
|
+
from .config import Config
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _cfg(args) -> Config:
|
|
16
|
+
cfg = Config.from_toml(args.config)
|
|
17
|
+
if getattr(args, "segment_m", None):
|
|
18
|
+
cfg.params = replace(cfg.params, segment_m=float(args.segment_m))
|
|
19
|
+
if getattr(args, "dates", None):
|
|
20
|
+
cfg.select = replace(cfg.select, dates=args.dates)
|
|
21
|
+
return cfg
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _log(args):
|
|
25
|
+
return (lambda s: print(s, file=sys.stderr)) if not getattr(args, "quiet", False) else None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _hotspot_table(hs: pl.DataFrame, tick_s: float) -> pl.DataFrame:
|
|
29
|
+
if hs is None or hs.height == 0:
|
|
30
|
+
return pl.DataFrame()
|
|
31
|
+
return hs.select(
|
|
32
|
+
"direction_id", "rank",
|
|
33
|
+
pl.concat_str([pl.col("from_stop_name"), pl.lit(" -> "), pl.col("to_stop_name")]).alias("stretch"),
|
|
34
|
+
pl.col("x0_m").round(0), pl.col("x1_m").round(0), "zone", "kind",
|
|
35
|
+
pl.col("hours").cast(pl.List(pl.Utf8)).list.join(",").alias("hours"),
|
|
36
|
+
pl.col("excess_per_passage").round(2).alias("excess_obs_per_veh"),
|
|
37
|
+
(pl.col("excess_per_passage") * tick_s).round(0).alias("approx_s_per_veh"),
|
|
38
|
+
pl.col("persistence").round(2))
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def cmd_run(args) -> int:
|
|
42
|
+
from .pipeline import run
|
|
43
|
+
cfg = _cfg(args)
|
|
44
|
+
res = run(cfg, log=_log(args), hotspots=not args.no_hotspots)
|
|
45
|
+
paths = res.save(args.out, html=not args.no_html, png=not args.no_png, lang=args.lang, title=args.title,
|
|
46
|
+
source=args.source)
|
|
47
|
+
for name, p in paths.items():
|
|
48
|
+
print(p)
|
|
49
|
+
if not args.quiet:
|
|
50
|
+
t = " ".join(f"{k} {v:.1f}s" for k, v in res.timings.items())
|
|
51
|
+
print(f"done: {len(res.analysis.days)} days, {res.segments.height} segments, "
|
|
52
|
+
f"{0 if res.hotspots is None else res.hotspots.height} hotspots ({t})", file=sys.stderr)
|
|
53
|
+
return 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def cmd_hotspots(args) -> int:
|
|
57
|
+
from .pipeline import run
|
|
58
|
+
cfg = _cfg(args)
|
|
59
|
+
res = run(cfg, log=_log(args))
|
|
60
|
+
t = _hotspot_table(res.hotspots, cfg.params.tick_s)
|
|
61
|
+
if t.height == 0:
|
|
62
|
+
print("no hotspot")
|
|
63
|
+
return 0
|
|
64
|
+
with pl.Config(tbl_rows=200, tbl_cols=20, fmt_str_lengths=60, tbl_width_chars=200):
|
|
65
|
+
print(t)
|
|
66
|
+
if args.out:
|
|
67
|
+
out = Path(args.out)
|
|
68
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
69
|
+
if out.suffix == ".geojson":
|
|
70
|
+
from .hotspots import to_geojson
|
|
71
|
+
out.write_text(json.dumps(to_geojson(res.hotspots, res.segments), default=str))
|
|
72
|
+
elif out.suffix == ".csv":
|
|
73
|
+
t.write_csv(out)
|
|
74
|
+
else:
|
|
75
|
+
res.hotspots.write_parquet(out)
|
|
76
|
+
print(out)
|
|
77
|
+
return 0
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def cmd_tune(args) -> int:
|
|
81
|
+
from . import tune
|
|
82
|
+
from .pipeline import prepare
|
|
83
|
+
cfg = _cfg(args)
|
|
84
|
+
pre = prepare(cfg, log=_log(args))
|
|
85
|
+
lengths = [float(x) for x in args.lengths.split(",")]
|
|
86
|
+
kw = dict(coverage=pre.coverage, passages=pre.passages, pattern_days=pre.network.pattern_days,
|
|
87
|
+
select=cfg.select, params=cfg.params, quality=cfg.quality, per_vehicle=pre.per_vehicle)
|
|
88
|
+
tr = tune.segment_length(pre.placed, pre.segment_fn(), lengths, B=args.bootstrap, **kw)
|
|
89
|
+
with pl.Config(tbl_rows=50, tbl_cols=20, tbl_width_chars=200, float_precision=3):
|
|
90
|
+
print(tr.table)
|
|
91
|
+
print(f"recommended L = {tr.recommended:g} m ({tr.rule})")
|
|
92
|
+
out = Path(args.out) if args.out else None
|
|
93
|
+
if out:
|
|
94
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
95
|
+
tr.table.write_parquet(out / "tune.parquet")
|
|
96
|
+
sr = None
|
|
97
|
+
if args.sensitivity:
|
|
98
|
+
sr = tune.sensitivity(pre.placed, pre.segment_fn(), tr.recommended, B=args.bootstrap, **kw)
|
|
99
|
+
with pl.Config(tbl_rows=50, tbl_cols=20, tbl_width_chars=200, float_precision=3):
|
|
100
|
+
print(sr.summary)
|
|
101
|
+
print(sr.hotspots)
|
|
102
|
+
print("stable" if sr.stable else "NOT stable")
|
|
103
|
+
if out:
|
|
104
|
+
sr.table.write_parquet(out / "sensitivity_profiles.parquet")
|
|
105
|
+
sr.summary.write_parquet(out / "sensitivity.parquet")
|
|
106
|
+
if out:
|
|
107
|
+
try:
|
|
108
|
+
import matplotlib
|
|
109
|
+
matplotlib.use("Agg")
|
|
110
|
+
from . import plot
|
|
111
|
+
plot.save(plot.tune_curves(tr), out / "tune.png")
|
|
112
|
+
if sr is not None:
|
|
113
|
+
plot.save(plot.sensitivity(sr), out / "sensitivity.png")
|
|
114
|
+
except ImportError:
|
|
115
|
+
pass
|
|
116
|
+
print(out)
|
|
117
|
+
return 0
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def cmd_inspect(args) -> int:
|
|
121
|
+
from .pipeline import prepare
|
|
122
|
+
cfg = _cfg(args)
|
|
123
|
+
pre = prepare(cfg, log=_log(args))
|
|
124
|
+
cov = pre.coverage
|
|
125
|
+
q = cfg.quality
|
|
126
|
+
print(f"route {cfg.select.route}: {pre.observations.height:,} observations, {pre.snapshots.height:,} polls")
|
|
127
|
+
day = (cov.filter(pl.col("hour").is_between(6, 20))
|
|
128
|
+
.group_by("service_date").agg(pl.col("coverage").mean().alias("mean_cov"),
|
|
129
|
+
(pl.col("coverage") < q.min_hour_coverage).sum().alias("low_hours"),
|
|
130
|
+
(pl.col("frozen_s") > 0).sum().alias("frozen_hours"))
|
|
131
|
+
.sort("service_date"))
|
|
132
|
+
print("\ncoverage per service date (hours 6-20):")
|
|
133
|
+
with pl.Config(tbl_rows=400, float_precision=2):
|
|
134
|
+
print(day)
|
|
135
|
+
print("\nGTFS versions (main pattern runs):")
|
|
136
|
+
with pl.Config(tbl_rows=50, tbl_cols=12, fmt_str_lengths=40, tbl_width_chars=200):
|
|
137
|
+
print(pre.network.versions().with_columns(pl.col("links_added").list.len(), pl.col("links_removed").list.len()))
|
|
138
|
+
if pre.network.missing_dates:
|
|
139
|
+
print("dates without GTFS service:", ", ".join(d.isoformat() for d in pre.network.missing_dates))
|
|
140
|
+
print("\nplacement:")
|
|
141
|
+
for k, v in (pre.place_report or {}).items():
|
|
142
|
+
print(f" {k:<36} {v:>12,}" if isinstance(v, (int, float)) else f" {k:<36} {v}")
|
|
143
|
+
from .schema import Flag
|
|
144
|
+
if "flags" in pre.placed.columns and pre.placed.height:
|
|
145
|
+
fl = pre.placed["flags"].fill_null(0).to_numpy()
|
|
146
|
+
for name in ("LAYOVER", "SHORT_WORKING", "OFF_PATTERN", "AMBIGUOUS", "AT_STOP"):
|
|
147
|
+
n = int(((fl & getattr(Flag, name)) != 0).sum())
|
|
148
|
+
if n:
|
|
149
|
+
print(f" flag {name.lower():<31} {n:>12,}")
|
|
150
|
+
print(f" counted rows {int(pre.placed['count'].fill_null(True).sum()):>12,}")
|
|
151
|
+
if pre.passages is not None and pre.passages.height:
|
|
152
|
+
p = pre.passages
|
|
153
|
+
print(f"\npassages: {p['n'].sum():,.0f} link passages"
|
|
154
|
+
+ (f" (feed {p['n_feed'].sum():,.0f}, events {p['n_events'].sum():,.0f})" if p["n_events"].null_count() < p.height else ""))
|
|
155
|
+
return 0
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def main(argv=None) -> int:
|
|
159
|
+
ap = argparse.ArgumentParser(prog="microsegments", description="Vehicle observations per micro-segment of a transit line.")
|
|
160
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
161
|
+
|
|
162
|
+
def common(p):
|
|
163
|
+
p.add_argument("config", help="TOML configuration")
|
|
164
|
+
p.add_argument("--segment-m", type=float, help="override params.segment_m")
|
|
165
|
+
p.add_argument("--dates", help="override select.dates (YYYY-MM-DD..YYYY-MM-DD)")
|
|
166
|
+
p.add_argument("-q", "--quiet", action="store_true")
|
|
167
|
+
|
|
168
|
+
p = sub.add_parser("run", help="full pipeline: parquet + JSON + HTML report + PNG matrices")
|
|
169
|
+
common(p)
|
|
170
|
+
p.add_argument("-o", "--out", default="out", help="output directory (default: out)")
|
|
171
|
+
p.add_argument("--lang", default="fr", choices=["fr", "en"])
|
|
172
|
+
p.add_argument("--title")
|
|
173
|
+
p.add_argument("--source", help="data credit shown in the page footer")
|
|
174
|
+
p.add_argument("--no-html", action="store_true")
|
|
175
|
+
p.add_argument("--no-png", action="store_true")
|
|
176
|
+
p.add_argument("--no-hotspots", action="store_true")
|
|
177
|
+
p.set_defaults(fn=cmd_run)
|
|
178
|
+
|
|
179
|
+
p = sub.add_parser("tune", help="choose the segment length (and optionally run the sensitivity suite)")
|
|
180
|
+
common(p)
|
|
181
|
+
p.add_argument("--lengths", default="10,15,20,30,40,50,75,100")
|
|
182
|
+
p.add_argument("--bootstrap", type=int, default=100)
|
|
183
|
+
p.add_argument("--sensitivity", action="store_true")
|
|
184
|
+
p.add_argument("-o", "--out", help="directory for tune.parquet / tune.png")
|
|
185
|
+
p.set_defaults(fn=cmd_tune)
|
|
186
|
+
|
|
187
|
+
p = sub.add_parser("hotspots", help="print the hotspot table")
|
|
188
|
+
common(p)
|
|
189
|
+
p.add_argument("-o", "--out", help="write .parquet, .csv or .geojson")
|
|
190
|
+
p.set_defaults(fn=cmd_hotspots)
|
|
191
|
+
|
|
192
|
+
p = sub.add_parser("inspect", help="coverage, GTFS versions, placement / drop counts")
|
|
193
|
+
common(p)
|
|
194
|
+
p.set_defaults(fn=cmd_inspect)
|
|
195
|
+
|
|
196
|
+
args = ap.parse_args(argv)
|
|
197
|
+
return args.fn(args)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
if __name__ == "__main__": # pragma: no cover
|
|
201
|
+
sys.exit(main())
|
microsegments/config.py
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""Run configuration: where the data is, how its columns map to ``schema.OBSERVATION``, what to select
|
|
2
|
+
and the method parameters. Load from TOML (``Config.from_toml``) or build in Python.
|
|
3
|
+
|
|
4
|
+
Example (lat/lon GTFS-RT positions exported to CSV)::
|
|
5
|
+
|
|
6
|
+
[input]
|
|
7
|
+
paths = ["positions/*.csv"]
|
|
8
|
+
kind = "latlon"
|
|
9
|
+
[input.columns]
|
|
10
|
+
ts = "timestamp"
|
|
11
|
+
route_id = "route_id"
|
|
12
|
+
vehicle_id = "vehicle_id"
|
|
13
|
+
lat = "latitude"
|
|
14
|
+
lon = "longitude"
|
|
15
|
+
[gtfs]
|
|
16
|
+
path = "gtfs.zip"
|
|
17
|
+
[select]
|
|
18
|
+
route = "55"
|
|
19
|
+
dates = "2025-02-17..2025-05-16"
|
|
20
|
+
weekdays = [0, 1, 2, 3, 4]
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import datetime as dt
|
|
25
|
+
import tomllib
|
|
26
|
+
from dataclasses import dataclass, field, fields, is_dataclass
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Any, Literal
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class LinearInput:
|
|
33
|
+
"""Linear-referenced feeds: 'last stop passed + metres since' (STIB vehicle-distance)."""
|
|
34
|
+
direction_kind: Literal["terminus_stop", "direction_id"] = "terminus_stop"
|
|
35
|
+
id_normaliser: str = "leading_digits" # "none" | "leading_digits" | a regex with one group
|
|
36
|
+
feed_length: Literal["auto", "gtfs"] = "auto" # auto: rescale feed metres to GTFS link length (p99.5)
|
|
37
|
+
terminus_aliases: dict[str, str] = field(default_factory=dict) # code -> stop_id, else learned
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class Input:
|
|
42
|
+
paths: list[str] = field(default_factory=list) # csv / parquet / json globs
|
|
43
|
+
kind: Literal["latlon", "linear", "stib"] = "latlon"
|
|
44
|
+
format: Literal["auto", "csv", "parquet", "stib_json", "gtfsrt_pb"] = "auto"
|
|
45
|
+
timezone: str = "Europe/Brussels"
|
|
46
|
+
service_day_start: str = "04:00"
|
|
47
|
+
ts_unit: Literal["s", "ms", "us", "iso"] = "s"
|
|
48
|
+
# canonical name -> source column name (only the ones that differ / exist)
|
|
49
|
+
columns: dict[str, str] = field(default_factory=dict)
|
|
50
|
+
linear: LinearInput = field(default_factory=LinearInput)
|
|
51
|
+
snapshots: str | None = None # optional path(s) to a Snapshot table
|
|
52
|
+
events: str | None = None # optional stop events (StopEvent schema or STIB punctuality)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class Gtfs:
|
|
57
|
+
path: str | None = None # zip / directory / gtfs-parquet directory
|
|
58
|
+
dated: str | None = None # template with {date} (YYYY-MM-DD) for one feed per day
|
|
59
|
+
route_key: Literal["route_short_name", "route_id"] = "route_short_name"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class Select:
|
|
64
|
+
route: str | None = None
|
|
65
|
+
dates: str | None = None # "YYYY-MM-DD..YYYY-MM-DD"
|
|
66
|
+
weekdays: list[int] = field(default_factory=lambda: [0, 1, 2, 3, 4])
|
|
67
|
+
exclude_dates: list[str] = field(default_factory=list)
|
|
68
|
+
hours: tuple[int, int] = (5, 24) # [start, end)
|
|
69
|
+
directions: list[int] | None = None
|
|
70
|
+
|
|
71
|
+
def date_range(self) -> tuple[dt.date, dt.date] | None:
|
|
72
|
+
if not self.dates:
|
|
73
|
+
return None
|
|
74
|
+
a, b = self.dates.split("..")
|
|
75
|
+
return dt.date.fromisoformat(a), dt.date.fromisoformat(b)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@dataclass
|
|
79
|
+
class Params:
|
|
80
|
+
segment_m: float = 30.0
|
|
81
|
+
phase_m: float = 0.0
|
|
82
|
+
grid: Literal["equal", "fixed"] = "equal" # equal: L' = len/round(len/L) per link; fixed: L with a short last bin
|
|
83
|
+
tick_s: float = 20.0 # nominal poll interval; per-vehicle feeds are thinned to 1 obs / tick
|
|
84
|
+
gap_cap_s: float = 40.0 # a poll covers at most this long
|
|
85
|
+
stop_zone: tuple[float, float] = (30.0, 60.0) # metres before / after a stop masked as "stop" zone
|
|
86
|
+
reference_hours: tuple[int, int] = (20, 23) # [start, end) "normal" evening reference
|
|
87
|
+
reference: Literal["evening", "freeflow"] = "evening"
|
|
88
|
+
layover_m: float = 10.0 # at the first stop of a pattern closer than this = terminus layover
|
|
89
|
+
max_lateral_m: float = 40.0 # map matching: reject fixes farther from the shape
|
|
90
|
+
bootstrap: int = 200
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass
|
|
94
|
+
class Quality:
|
|
95
|
+
min_hour_coverage: float = 0.8
|
|
96
|
+
min_day_coverage: float = 0.75 # share of 6-21 h day-hours that must pass
|
|
97
|
+
max_frozen_s: float = 300.0
|
|
98
|
+
vehicle_ratio: tuple[float, float] = (0.7, 1.3)
|
|
99
|
+
max_off_pattern: float = 0.2
|
|
100
|
+
min_days: int = 5
|
|
101
|
+
min_passages_per_day: float = 0.5
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@dataclass
|
|
105
|
+
class Config:
|
|
106
|
+
input: Input = field(default_factory=Input)
|
|
107
|
+
gtfs: Gtfs = field(default_factory=Gtfs)
|
|
108
|
+
select: Select = field(default_factory=Select)
|
|
109
|
+
params: Params = field(default_factory=Params)
|
|
110
|
+
quality: Quality = field(default_factory=Quality)
|
|
111
|
+
|
|
112
|
+
@classmethod
|
|
113
|
+
def from_dict(cls, d: dict[str, Any]) -> "Config":
|
|
114
|
+
return _build(cls, d)
|
|
115
|
+
|
|
116
|
+
@classmethod
|
|
117
|
+
def from_toml(cls, path: str | Path) -> "Config":
|
|
118
|
+
path = Path(path)
|
|
119
|
+
cfg = cls.from_dict(tomllib.loads(path.read_text()))
|
|
120
|
+
base = path.parent
|
|
121
|
+
cfg.input.paths = [p if Path(p).is_absolute() else str(base / p) for p in cfg.input.paths]
|
|
122
|
+
for obj, attr in ((cfg.gtfs, "path"), (cfg.gtfs, "dated"), (cfg.input, "snapshots"), (cfg.input, "events")):
|
|
123
|
+
v = getattr(obj, attr)
|
|
124
|
+
if v and not Path(v).is_absolute():
|
|
125
|
+
setattr(obj, attr, str(base / v))
|
|
126
|
+
return cfg
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _build(cls, d: dict[str, Any]):
|
|
130
|
+
kw = {}
|
|
131
|
+
names = {f.name: f for f in fields(cls)}
|
|
132
|
+
for k, v in (d or {}).items():
|
|
133
|
+
if k not in names:
|
|
134
|
+
raise ValueError(f"unknown config key {cls.__name__}.{k}")
|
|
135
|
+
f = names[k]
|
|
136
|
+
default = f.default_factory() if callable(f.default_factory) else f.default # type: ignore[misc]
|
|
137
|
+
if is_dataclass(default) and isinstance(v, dict):
|
|
138
|
+
v = _build(type(default), v)
|
|
139
|
+
elif isinstance(default, tuple) and isinstance(v, list):
|
|
140
|
+
v = tuple(v)
|
|
141
|
+
kw[k] = v
|
|
142
|
+
return cls(**kw)
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""JSON exchanged with the HTML page (``html.export``) and the platform (``GET /api/analysis``).
|
|
2
|
+
|
|
3
|
+
Totals are kept per weekday so the page can recombine any set of weekdays client-side
|
|
4
|
+
(sum the selected rows, then divide). Nothing is pre-divided except geometry.
|
|
5
|
+
|
|
6
|
+
{
|
|
7
|
+
"v": 1,
|
|
8
|
+
"line": "55", "route_id": "...", "mode": "tram",
|
|
9
|
+
"params": {segment_m, phase_m, grid, tick_s, gap_cap_s, stop_zone: [up, down], reference_hours: [a, b]},
|
|
10
|
+
"period": {"first": "2025-02-17", "last": "2025-05-16",
|
|
11
|
+
"days": 60, # included service days
|
|
12
|
+
"per_dow": [11, 12, 13, 12, 12, 0, 0],
|
|
13
|
+
"excluded": [{"date": "2025-03-04", "reason": "low_coverage"}]},
|
|
14
|
+
"hours": [5, ..., 23], # hour index h used below
|
|
15
|
+
"coverage": {"dates": ["2025-02-17", ...], "c": [[0.98, ...] per date][per hour]},
|
|
16
|
+
"versions": [{"dir": 0, "pattern_uid": "...", "first": "...", "last": "...", "n_days": 54,
|
|
17
|
+
"links_added": [...], "links_removed": [...], "display": true}],
|
|
18
|
+
"dirs": [{
|
|
19
|
+
"dir": 0, "pattern_uid": "...", "from": "Da Vinci", "to": "Rogier",
|
|
20
|
+
"stops": [{"id": "...", "name": "...", "ll": [lon, lat], "x": 0.0}],
|
|
21
|
+
"links": [{"key": "...", "from": "...", "to": "...", "len": 252.7, "n_days": 60, "flags": 0}],
|
|
22
|
+
"seg": {"key": [...], "link": [...], "x0": [...], "len": [...], "zone": ["stop"|"running"],
|
|
23
|
+
"c": [[[lon, lat], ...] per segment], "flags": [...]},
|
|
24
|
+
"wk": {"obs": [dow][h][seg] int observation counts,
|
|
25
|
+
"cov": [dow][h][link] float covered hours (sum of coverage over included days where the link is valid),
|
|
26
|
+
"p": [dow][h][link] float passages,
|
|
27
|
+
"days": [dow][link] int included days where the link is valid}
|
|
28
|
+
}],
|
|
29
|
+
"hotspots": [HOTSPOT rows as dicts],
|
|
30
|
+
"ctx": [[[lon, lat], ...]] # optional background polylines
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
Client-side metrics for a weekday set W and hour h, segment s in link l:
|
|
34
|
+
obs_per_h = sum_W obs / sum_W cov
|
|
35
|
+
obs_per_passage = sum_W obs / sum_W p
|
|
36
|
+
ref = sum_{W, h in ref} obs / sum_{W, h in ref} p
|
|
37
|
+
excess = obs_per_passage - ref
|
|
38
|
+
"""
|
|
39
|
+
from __future__ import annotations
|
|
40
|
+
|
|
41
|
+
CONTRACT_VERSION = 1
|
|
42
|
+
DOW_FR = ["Lu", "Ma", "Me", "Je", "Ve", "Sa", "Di"]
|