aperta 0.1.0a0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
aperta/__init__.py ADDED
@@ -0,0 +1,70 @@
1
+ """
2
+ aperta — cross-modal accessibility analysis on transport networks.
3
+
4
+ The library is organized around a six-phase workflow:
5
+
6
+ 1. Load and prepare data — networks (per mode), land use, topography.
7
+ 2. Map data to units — `cells → zones` aggregation hierarchy
8
+ (`geo_mapping`, `network_processing.snap_to_network_nodes`,
9
+ `network_processing.assign_to_eligible_centroid`).
10
+ 3. Build sparse OD pairs — `od_pairs.get_pairs` returns a `TieredODNodePairs`
11
+ (three distance tiers: cells_to_cells,
12
+ cells_to_zones, zones_to_zones). Lift to
13
+ `TieredODGeoPairs` via
14
+ `od_pairs.reindex_by_geo_unit` for cross-modal
15
+ alignment / geo-unit-keyed overheads.
16
+ 4. Estimate traffic flows — `traffic_flows.nested_node_sample` +
17
+ `network_processing.get_*_betweenness*`.
18
+ 5. Estimate travel costs — `routing.tiered_path_costs` /
19
+ `routing.tiered_path_aggregate` (Dijkstra on any
20
+ networkx graph; the latter also aggregates
21
+ per-edge / per-node features along realised
22
+ routes via `PathAggregation` / `NodeAggregation`)
23
+ + the `overhead` module
24
+ (`add_node_overheads` for node-keyed,
25
+ `add_geo_overheads` / `add_origin_cell_overhead`
26
+ for geo-keyed). `utility.route_utility` +
27
+ `add_endpoint_utility` for utility-based costs.
28
+ `routing.aggregate_along_paths` is the path-walker
29
+ primitive when you have a pre-computed list of
30
+ paths rather than a `TieredODPairs`.
31
+ 6. Calculate accessibility — `accessibility.cumulative_opportunities` (cumulative),
32
+ `accessibility.gravity`, `accessibility.nearest_k`.
33
+ Cross-modal: combine per-mode `TieredODGeoPairs`
34
+ with `od_pairs.aggregate_across_modes` first.
35
+
36
+ All algorithm modules (`od_pairs`, `routing`, `overhead`, `accessibility`,
37
+ `utility`, `traffic_flows`, `geo_processing`, `geo_mapping`,
38
+ `network_processing`, `visualization`, `osm_helpers`, `calibration`,
39
+ `topography`, `errors`) operate on plain numpy / pandas / networkx inputs — no
40
+ filesystem assumptions, no opinionated project structure.
41
+ See `tests/test_workflow.py` for the ~150-line end-to-end toy-world
42
+ example, `examples/minimal/accessibility.ipynb` for a ~50-line OSM
43
+ quickstart, `examples/walkthrough/accessibility.ipynb` for the full
44
+ guided tour, and `examples/extended/` for a multi-notebook showcase
45
+ with published-paper calibration (Bern + 40 km).
46
+
47
+ Key types:
48
+ - `od_pairs.TieredODNodePairs` — three-tier OD dict-of-arrays keyed by network
49
+ node IDs. Output of routing.
50
+ - `od_pairs.TieredODGeoPairs` — three-tier OD dict-of-arrays keyed by
51
+ geo-unit IDs (cells / zones).
52
+ Mode-agnostic; required for cross-modal
53
+ accessibility and geo-unit-keyed overhead.
54
+ - `od_pairs.TieredODPairs` — abstract base of the two above; use as a
55
+ type hint when key space doesn't matter.
56
+ - `accessibility.Bin` — half-open cost bin for `cumulative_opportunities`.
57
+ - `accessibility.Decay` — named cost-decay callable for `gravity`.
58
+ - `utility.Utility` — linear utility spec (constant + cost + route
59
+ + origin + destination feature coefficients).
60
+ """
61
+
62
+ from importlib.metadata import PackageNotFoundError
63
+ from importlib.metadata import version as _pkg_version
64
+
65
+ try:
66
+ __version__ = _pkg_version("aperta")
67
+ except PackageNotFoundError: # running from a source tree without install
68
+ __version__ = "0.0.0+unknown"
69
+
70
+ __all__ = ["__version__"]
@@ -0,0 +1,481 @@
1
+ """
2
+ Accessibility metrics computed against tiered OD tables.
3
+
4
+ Inputs are always:
5
+ 1. A cost `TieredODPairs` (subclass) — see "Key space" below.
6
+ 2. One or more pre-aggregated property weights, position-aligned with the
7
+ cost ODM at each tier — a `dict[name -> TieredODPairs]`.
8
+ 3. A `cell_to_zone` mapping that gives each cell-tier origin its parent
9
+ zone-tier key (for stitching the `zones_to_zones` far tier — the
10
+ `cells_to_cells` and `cells_to_zones` tiers are already cell-keyed
11
+ and don't need the indirection).
12
+
13
+ The per-origin stitching of `cells_to_cells / cells_to_zones / zones_to_zones`
14
+ tiers is done once, then reused across every (parameter × property) combination
15
+ — so adding more bins, decays, or k values is essentially free relative to a
16
+ single-parameter call.
17
+
18
+ **Key space — node-keyed vs geo-keyed.**
19
+
20
+ Two valid input shapes, distinguished by the costs/weights subclass:
21
+
22
+ - **`TieredODNodePairs`** (node-keyed): origins and dests are network node IDs.
23
+ Each origin row in the output is a network node. Per-cell origin overhead
24
+ cannot be applied here — the function returns per-NODE accessibilities. Build
25
+ the `cell_to_zone` map from `od_pairs.build_cell_to_zone_node_map(cells,
26
+ zones, node_column)` (cell-tier node → zone-tier node). Weights from
27
+ `od_pairs.dest_values` (per-node sums).
28
+
29
+ - **`TieredODGeoPairs`** (geo-keyed): origins and dests are geo-unit IDs
30
+ (cells_to_cells → cell_id; zones_to_zones → zone_id; etc.). Each origin row
31
+ in the output is a cell. Per-cell origin overhead should be baked into the
32
+ ODM *before* calling this function via `overhead.add_origin_cell_overhead`.
33
+ Build the `cell_to_zone` map directly: `cells['zone_id'].to_dict()`. Weights
34
+ from `od_pairs.dest_values_geo` (per-cell direct lookup, no implicit summing).
35
+
36
+ The same three functions (`cumulative_opportunities`, `gravity`, `nearest_k`) accept
37
+ either shape; output index name (`'node'` vs `'cell'`) reflects the input.
38
+
39
+ For cross-modal accessibility ("destinations within X min by ANY mode",
40
+ cross-modal logsum), combine per-mode `TieredODGeoPairs` cost ODMs with
41
+ `od_pairs.aggregate_across_modes` before passing here. Node-keyed cross-modal
42
+ is not supported — different modes live on different graphs.
43
+
44
+ For gravity in particular, the intrazonal-cost issue (cell-tier self-pairs
45
+ route at cost 0, which sends exp(0)=1 to maximum weight) is addressed by
46
+ calling `routing.floor_intrazonal_costs` on the cost ODM before passing here.
47
+
48
+ Provides:
49
+ - `Bin` namedtuple — half-open `[lo, hi)` cost bin with a name.
50
+ - `Decay` namedtuple — named callable for gravity-style cost decay.
51
+ - `exp_decay`, `power_decay` — convenience constructors for common families.
52
+ - `cumulative_opportunities` — sum each property's weights over destinations within each
53
+ cost bin (cumulative-opportunity accessibility).
54
+ - `gravity` — sum each property's weights weighted by f(cost), over all
55
+ destinations, for one or more decay specs.
56
+ - `nearest_k` — mean cost (or cost-at-k) to the k nearest weight-units,
57
+ for one or more k values. Lower is better; canonical "mean travel time
58
+ to the nearest k opportunities" formulation.
59
+ """
60
+
61
+ from typing import Callable, NamedTuple
62
+
63
+ import numpy as np
64
+ import pandas as pd
65
+
66
+ from aperta.od_pairs import TieredODGeoPairs, TieredODPairs
67
+
68
+
69
+ def _require_cell_tier(costs: TieredODPairs) -> dict:
70
+ """Narrow `costs.cells_to_cells` to a concrete `dict`, raising if None.
71
+
72
+ Accessibility metrics require the cell-tier to be populated (origins are
73
+ the cell-tier dict keys); raise a clear error if it isn't.
74
+ """
75
+ if costs.cells_to_cells is None:
76
+ raise ValueError("`costs.cells_to_cells` is None; cell-tier is required.")
77
+ return costs.cells_to_cells
78
+
79
+
80
+ class Bin(NamedTuple):
81
+ """Half-open cost bin: `lo <= cost < hi`. `name` labels the output column.
82
+
83
+ Bins should be mutually exclusive (the function does *not* check); a
84
+ destination falling in multiple bins would be counted multiple times.
85
+ """
86
+
87
+ name: str
88
+ lo: float
89
+ hi: float
90
+
91
+
92
+ class Decay(NamedTuple):
93
+ """Named cost-decay specification for `gravity`.
94
+
95
+ `fn` is a vectorised callable mapping a cost array to a weight array; `name`
96
+ labels the corresponding output column. Use `exp_decay` / `power_decay` for
97
+ the common families, or construct directly with any user-defined callable.
98
+
99
+ Multiple `Decay` specs can be passed to a single `gravity` call; the per-OD
100
+ stitching is then amortised across all of them.
101
+ """
102
+
103
+ name: str
104
+ fn: Callable[[np.ndarray], np.ndarray]
105
+
106
+
107
+ def exp_decay(name: str, beta: float) -> Decay:
108
+ """Exponential decay: `f(c) = exp(-beta * c)`. `beta` > 0 for sensible decay."""
109
+ return Decay(name, lambda c: np.exp(-beta * c))
110
+
111
+
112
+ def power_decay(name: str, beta: float) -> Decay:
113
+ """Power-law decay: `f(c) = c ** (-beta)`. `beta` > 0; c = 0 yields `inf`,
114
+ so callers should apply `routing.add_intrazonal_cost` first to replace
115
+ self-pair cost-0 entries with a finite intrazonal cost.
116
+ """
117
+ return Decay(name, lambda c: np.power(c, -beta))
118
+
119
+
120
+ def _stitched_for(
121
+ origin: int | str, costs: TieredODPairs, cell_to_zone_node: dict
122
+ ) -> tuple[np.ndarray, slice, slice, slice]:
123
+ """Per-origin stitched cost array + per-tier slices into it.
124
+
125
+ The slices let callers stitch *other* per-tier arrays (e.g. each property's
126
+ weights) into the same position-aligned 1-D layout without re-deriving the
127
+ boundaries.
128
+
129
+ Three tiers in order: cells_to_cells (cell-keyed, direct), cells_to_zones
130
+ (cell-keyed, direct), zones_to_zones (zone-keyed, needs origin's parent
131
+ zone). The returned slices have the same order.
132
+ """
133
+ cell_arr = _require_cell_tier(costs)[origin]
134
+ c2z_arr = costs.cells_to_zones.get(origin) if costs.cells_to_zones is not None else None
135
+ zone_node = cell_to_zone_node.get(origin)
136
+ zone_arr = costs.zones_to_zones.get(zone_node) if costs.zones_to_zones is not None else None
137
+ parts = [cell_arr]
138
+ if c2z_arr is not None:
139
+ parts.append(c2z_arr)
140
+ if zone_arr is not None:
141
+ parts.append(zone_arr)
142
+ stitched = np.concatenate(parts) if len(parts) > 1 else cell_arr
143
+ n_cell = len(cell_arr)
144
+ n_c2z = len(c2z_arr) if c2z_arr is not None else 0
145
+ n_zone = len(zone_arr) if zone_arr is not None else 0
146
+ return (
147
+ stitched,
148
+ slice(0, n_cell),
149
+ slice(n_cell, n_cell + n_c2z),
150
+ slice(n_cell + n_c2z, n_cell + n_c2z + n_zone),
151
+ )
152
+
153
+
154
+ def _stitched_weights(
155
+ origin: int | str,
156
+ zone_node: int | str | None,
157
+ weights: TieredODPairs,
158
+ n_cell: int,
159
+ n_c2z: int,
160
+ n_zone: int,
161
+ dtype: np.dtype | type = np.float32,
162
+ ) -> np.ndarray:
163
+ """Stitch a single property's three-tier value arrays for one origin.
164
+
165
+ Tiers that are `None` (or where the origin / zone has no entry) contribute
166
+ zeros so the result is positionally aligned with the cost stitching from
167
+ `_stitched_for`.
168
+ """
169
+ total = n_cell + n_c2z + n_zone
170
+ out = np.zeros(total, dtype=dtype)
171
+ cell_w = weights.cells_to_cells.get(origin) if weights.cells_to_cells is not None else None
172
+ if cell_w is not None and n_cell:
173
+ out[:n_cell] = cell_w
174
+ if n_c2z and weights.cells_to_zones is not None:
175
+ c2z_w = weights.cells_to_zones.get(origin)
176
+ if c2z_w is not None:
177
+ out[n_cell : n_cell + n_c2z] = c2z_w
178
+ if n_zone and weights.zones_to_zones is not None:
179
+ zw = weights.zones_to_zones.get(zone_node)
180
+ if zw is not None:
181
+ out[n_cell + n_c2z :] = zw
182
+ return out
183
+
184
+
185
+ def _origin_index_name(costs: TieredODPairs) -> str:
186
+ """Output-DataFrame index name based on the input ODM's key space."""
187
+ return "cell" if isinstance(costs, TieredODGeoPairs) else "node"
188
+
189
+
190
+ def _sniff_dtype(costs: TieredODPairs) -> np.dtype:
191
+ """Return the dtype of the first non-empty per-tier cost array.
192
+
193
+ Used to make accessibility output dtype follow the input costs dtype
194
+ (FP32 by default, FP64 if the caller opted in upstream) without an
195
+ explicit kwarg on every accessibility function.
196
+ """
197
+ for tier in (costs.cells_to_cells, costs.cells_to_zones, costs.zones_to_zones):
198
+ if tier is None:
199
+ continue
200
+ for arr in tier.values():
201
+ return np.asarray(arr).dtype
202
+ return np.dtype(np.float32)
203
+
204
+
205
+ def cumulative_opportunities(
206
+ costs: TieredODPairs,
207
+ weights: dict[str, TieredODPairs],
208
+ cell_to_zone: dict,
209
+ bins: list[Bin],
210
+ ) -> pd.DataFrame:
211
+ """Sum each property's weights over destinations whose cost falls in each bin.
212
+
213
+ Args:
214
+ costs: tiered travel costs. Subclass determines output indexing:
215
+ `TieredODNodePairs` → per-node output; `TieredODGeoPairs` → per-cell
216
+ output. Non-finite entries (`np.inf`, `np.nan`) won't match any
217
+ finite bin and are silently dropped.
218
+ weights: `{property_name -> TieredODPairs}`, position-aligned with
219
+ `costs` per tier. Must share the costs' key space (node-keyed
220
+ weights for node-keyed costs; geo-keyed for geo-keyed). Build via
221
+ `od_pairs.dest_values` (node-keyed) or `od_pairs.dest_values_geo`
222
+ (geo-keyed). Missing origins / tiers contribute zeros, not errors.
223
+ cell_to_zone: `{cell_tier_key -> zone_tier_key}` map for tier
224
+ stitching. Build from `od_pairs.build_cell_to_zone_node_map`
225
+ (node-keyed: cell_node → zone_node) or directly from
226
+ `cells['zone_id'].to_dict()` (geo-keyed: cell_id → zone_id).
227
+ bins: half-open `[lo, hi)` cost bins. Should be mutually exclusive
228
+ (not checked).
229
+
230
+ Returns:
231
+ DataFrame indexed by origin key with `(bin_name, property_name)`
232
+ MultiIndex on columns. Order: bins outer, properties inner. Dtype
233
+ follows the input `costs` ODM (FP32 by default, FP64 if the caller
234
+ opted in upstream).
235
+
236
+ Per-cell overhead: for `TieredODGeoPairs` inputs, bake per-cell origin
237
+ overhead into the cost ODM upfront via `overhead.add_origin_cell_overhead`.
238
+ """
239
+ prop_names = list(weights.keys())
240
+ origins = list(_require_cell_tier(costs).keys())
241
+ columns = pd.MultiIndex.from_product(
242
+ [[b.name for b in bins], prop_names], names=["bin", "property"]
243
+ )
244
+ dtype = _sniff_dtype(costs)
245
+ out = np.zeros((len(origins), len(bins) * len(prop_names)), dtype=dtype)
246
+
247
+ for i, origin in enumerate(origins):
248
+ stitched_cost, cell_sl, c2z_sl, zone_sl = _stitched_for(origin, costs, cell_to_zone)
249
+ n_cell = cell_sl.stop - cell_sl.start
250
+ n_c2z = c2z_sl.stop - c2z_sl.start
251
+ n_zone = zone_sl.stop - zone_sl.start
252
+ zone_key = cell_to_zone.get(origin)
253
+
254
+ prop_weights = np.empty((len(prop_names), len(stitched_cost)), dtype=dtype)
255
+ for p, name in enumerate(prop_names):
256
+ prop_weights[p] = _stitched_weights(
257
+ origin, zone_key, weights[name], n_cell, n_c2z, n_zone, dtype=dtype
258
+ )
259
+
260
+ for b, bin_ in enumerate(bins):
261
+ mask = (stitched_cost >= bin_.lo) & (stitched_cost < bin_.hi)
262
+ if not mask.any():
263
+ continue
264
+ out[i, b * len(prop_names) : (b + 1) * len(prop_names)] = prop_weights[:, mask].sum(
265
+ axis=1
266
+ )
267
+
268
+ return pd.DataFrame(
269
+ out,
270
+ index=pd.Index(origins, name=_origin_index_name(costs)),
271
+ columns=columns,
272
+ )
273
+
274
+
275
+ def gravity(
276
+ costs: TieredODPairs,
277
+ weights: dict[str, TieredODPairs],
278
+ cell_to_zone: dict,
279
+ decays: Decay | list[Decay],
280
+ ) -> pd.DataFrame:
281
+ """Gravity-based accessibility: sum each property's weights, weighted by
282
+ `f(cost)`, over all destinations — for one or more decay specs in a single
283
+ call.
284
+
285
+ For each origin `i`, each property `w`, and each decay spec `f`::
286
+
287
+ A_i^{f,w} = Σ_j w_j · f(cost_ij)
288
+
289
+ Multiple decay specs share the per-OD stitching, so calling with a list of
290
+ `Decay` specs is much cheaper than calling once per spec — useful for sensitivity
291
+ analyses across decay-coefficient ranges.
292
+
293
+ For utility-based accessibility, pass the per-OD utility values as the cost
294
+ ODM and an exponential decay with the desired scale (β=1 gives the
295
+ standard `Σ_j w_j · exp(-U_ij)` form, on which logsum accessibility is
296
+ `ln(...)` of the same sum).
297
+
298
+ Args:
299
+ costs, weights, cell_to_zone: see `cumulative_opportunities`.
300
+ decays: a single `Decay` or list of `Decay` specs. Output columns are
301
+ MultiIndex `(decay_name, property_name)` with decay names outer.
302
+
303
+ Returns:
304
+ DataFrame indexed by origin key with MultiIndex columns
305
+ `(decay, property)`. Dtype follows the input `costs` ODM
306
+ (FP32 by default, FP64 if the caller opted in upstream).
307
+ """
308
+ if isinstance(decays, Decay):
309
+ decays = [decays]
310
+ if not decays:
311
+ raise ValueError("`decays` must be a non-empty list of `Decay` specs.")
312
+
313
+ prop_names = list(weights.keys())
314
+ decay_names = [d.name for d in decays]
315
+ origins = list(_require_cell_tier(costs).keys())
316
+ columns = pd.MultiIndex.from_product([decay_names, prop_names], names=["decay", "property"])
317
+ n_props = len(prop_names)
318
+ dtype = _sniff_dtype(costs)
319
+ out = np.zeros((len(origins), len(decays) * n_props), dtype=dtype)
320
+
321
+ for i, origin in enumerate(origins):
322
+ stitched_cost, cell_sl, c2z_sl, zone_sl = _stitched_for(origin, costs, cell_to_zone)
323
+ n_cell = cell_sl.stop - cell_sl.start
324
+ n_c2z = c2z_sl.stop - c2z_sl.start
325
+ n_zone = zone_sl.stop - zone_sl.start
326
+ zone_key = cell_to_zone.get(origin)
327
+
328
+ prop_weights = np.empty((n_props, len(stitched_cost)), dtype=dtype)
329
+ for p, name in enumerate(prop_names):
330
+ prop_weights[p] = _stitched_weights(
331
+ origin, zone_key, weights[name], n_cell, n_c2z, n_zone, dtype=dtype
332
+ )
333
+
334
+ finite_mask = np.isfinite(stitched_cost)
335
+ if not finite_mask.any():
336
+ continue
337
+ cost_finite = stitched_cost[finite_mask]
338
+ w_finite = prop_weights[:, finite_mask]
339
+ for d, decay in enumerate(decays):
340
+ decayed = decay.fn(cost_finite)
341
+ # Defensive: a decay that produces non-finite at finite cost
342
+ # (e.g. power with c=0 if intrazonal cost wasn't applied) should
343
+ # not silently corrupt the sum. Drop those entries.
344
+ if not np.all(np.isfinite(decayed)):
345
+ decay_finite = np.isfinite(decayed)
346
+ decayed = decayed[decay_finite]
347
+ w_used = w_finite[:, decay_finite]
348
+ else:
349
+ w_used = w_finite
350
+ out[i, d * n_props : (d + 1) * n_props] = (w_used * decayed).sum(axis=1)
351
+
352
+ return pd.DataFrame(
353
+ out,
354
+ index=pd.Index(origins, name=_origin_index_name(costs)),
355
+ columns=columns,
356
+ )
357
+
358
+
359
+ def nearest_k(
360
+ costs: TieredODPairs,
361
+ weights: dict[str, TieredODPairs],
362
+ cell_to_zone: dict,
363
+ ks: int | float | list[int | float],
364
+ *,
365
+ aggregator: str = "cost_mean",
366
+ ) -> pd.DataFrame:
367
+ """Nearest-`k` accessibility: cost (mean, or at-`k`) over the `k` nearest
368
+ weight-units.
369
+
370
+ Each destination is treated as carrying `weight_j` opportunities at cost
371
+ `cost_ij`. Destinations are sorted ascending by cost; the first `k`
372
+ weight-units (with fractional contribution at the boundary) define the
373
+ "nearest `k` opportunities". The aggregator decides what to return:
374
+
375
+ - **`'cost_mean'`** (default): the mean cost over the first `k` weight-units,
376
+ ``A_i^{k,w} = (Σ cost_j · weight_j contributed, fractional at the boundary) / k``.
377
+ The canonical "mean travel cost to the nearest `k` opportunities"
378
+ formulation — directly comparable across `k` values (k=3 and k=5 are on
379
+ the same scale, in cost units).
380
+
381
+ - **`'cost_at_k'`**: the cost of the `k`-th weight-unit — i.e., the cost
382
+ at which the cumulative weight first reaches `k`. Answers
383
+ "how far is the `k`-th nearest opportunity?".
384
+
385
+ Both aggregators return a value in the same units as `costs`, with **lower
386
+ values = better accessibility**. NaN is returned where the total available
387
+ (finite-cost, positive-weight) opportunities at an origin is less than `k`
388
+ — i.e., the `k`-th opportunity is unreachable in finite cost.
389
+
390
+ Multiple `k` values share the per-OD sort, so a multi-`k` call is much
391
+ cheaper than `k` individual calls.
392
+
393
+ Args:
394
+ costs, weights, cell_to_zone: see `cumulative_opportunities`.
395
+ ks: a single `k` or list of `k`s; positive values, integer or float.
396
+ Output columns are MultiIndex `(k, property_name)` with `k` outer.
397
+ aggregator: `'cost_mean'` (default) or `'cost_at_k'`.
398
+
399
+ Returns:
400
+ DataFrame indexed by origin key with MultiIndex columns `(k, property)`.
401
+ Dtype follows the input `costs` ODM (FP32 by default, FP64 if the
402
+ caller opted in upstream). NaN where the `k`-th opportunity is
403
+ unreachable.
404
+ """
405
+ if aggregator not in ("cost_mean", "cost_at_k"):
406
+ raise ValueError(f"Unknown aggregator {aggregator!r}; expected 'cost_mean' or 'cost_at_k'.")
407
+ if isinstance(ks, (int, float)):
408
+ ks = [ks]
409
+ if not ks:
410
+ raise ValueError("`ks` must be a non-empty list of positive values.")
411
+ if any(k <= 0 for k in ks):
412
+ raise ValueError(f"All `k` values must be > 0; got {ks!r}.")
413
+
414
+ prop_names = list(weights.keys())
415
+ origins = list(_require_cell_tier(costs).keys())
416
+ columns = pd.MultiIndex.from_product([ks, prop_names], names=["k", "property"])
417
+ n_props = len(prop_names)
418
+ dtype = _sniff_dtype(costs)
419
+ out = np.full((len(origins), len(ks) * n_props), np.nan, dtype=dtype)
420
+ ks_arr = np.asarray(ks, dtype=dtype)
421
+
422
+ for i, origin in enumerate(origins):
423
+ stitched_cost, cell_sl, c2z_sl, zone_sl = _stitched_for(origin, costs, cell_to_zone)
424
+ n_cell = cell_sl.stop - cell_sl.start
425
+ n_c2z = c2z_sl.stop - c2z_sl.start
426
+ n_zone = zone_sl.stop - zone_sl.start
427
+ zone_key = cell_to_zone.get(origin)
428
+
429
+ prop_weights = np.empty((n_props, len(stitched_cost)), dtype=dtype)
430
+ for p, name in enumerate(prop_names):
431
+ prop_weights[p] = _stitched_weights(
432
+ origin, zone_key, weights[name], n_cell, n_c2z, n_zone, dtype=dtype
433
+ )
434
+
435
+ finite_mask = np.isfinite(stitched_cost)
436
+ if not finite_mask.any():
437
+ continue
438
+ costs_f = stitched_cost[finite_mask]
439
+ sort_idx = np.argsort(costs_f)
440
+ sorted_costs = costs_f[sort_idx]
441
+ for p in range(n_props):
442
+ w = prop_weights[p][finite_mask][sort_idx]
443
+ # Non-finite or non-positive weights contribute nothing.
444
+ w = np.where(np.isfinite(w) & (w > 0), w, 0.0)
445
+ cum_w = np.cumsum(w)
446
+ if cum_w.size == 0 or cum_w[-1] == 0.0:
447
+ continue
448
+ cum_cw = np.cumsum(sorted_costs * w) if aggregator == "cost_mean" else None
449
+ total = cum_w[-1]
450
+ for ki, k in enumerate(ks_arr):
451
+ if total < k:
452
+ continue
453
+ idx = int(np.searchsorted(cum_w, k, side="left"))
454
+ if aggregator == "cost_at_k":
455
+ out[i, ki * n_props + p] = sorted_costs[idx]
456
+ else: # cost_mean
457
+ assert cum_cw is not None # guaranteed by aggregator branch above
458
+ if idx == 0:
459
+ out[i, ki * n_props + p] = sorted_costs[0]
460
+ else:
461
+ full_cw = cum_cw[idx - 1]
462
+ partial = sorted_costs[idx] * (k - cum_w[idx - 1])
463
+ out[i, ki * n_props + p] = (full_cw + partial) / k
464
+
465
+ return pd.DataFrame(
466
+ out,
467
+ index=pd.Index(origins, name=_origin_index_name(costs)),
468
+ columns=columns,
469
+ )
470
+
471
+
472
+ def flatten_index(df: pd.DataFrame) -> pd.DataFrame:
473
+ """Collapse a 2-level column MultiIndex into single strings joined by `__`.
474
+
475
+ Convenience for the accessibility outputs (which carry `(bin, property)`
476
+ or `(decay, property)` MultiIndex columns) when downstream code prefers
477
+ flat single-string column names (e.g. for CSV export). Mutates in place
478
+ and also returns `df`.
479
+ """
480
+ df.columns = ["__".join(col).strip() for col in df.columns.values]
481
+ return df