osmsg 1.2.4__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {osmsg-1.2.4 → osmsg-1.3.0}/PKG-INFO +4 -3
  2. {osmsg-1.2.4 → osmsg-1.3.0}/README.md +3 -2
  3. osmsg-1.3.0/osmsg/__version__.py +1 -0
  4. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/_tick.py +25 -7
  5. osmsg-1.3.0/osmsg/catalog.py +212 -0
  6. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/cli.py +17 -5
  7. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/duckdb_schema.py +2 -1
  8. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/ingest.py +26 -7
  9. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/queries.py +31 -54
  10. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/schema.py +28 -1
  11. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/markdown.py +9 -9
  12. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/parquet.py +11 -11
  13. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/psql.py +95 -9
  14. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/gui.py +27 -12
  15. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/history.py +46 -30
  16. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/cli.py +20 -1
  17. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/convert.py +7 -10
  18. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/month.py +16 -7
  19. osmsg-1.3.0/osmsg/maintain/rollup.py +151 -0
  20. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/models.py +11 -13
  21. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/pg_schema.py +19 -6
  22. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/pipeline.py +129 -67
  23. osmsg-1.3.0/osmsg/prune.py +50 -0
  24. osmsg-1.3.0/osmsg/query.py +462 -0
  25. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/replication.py +1 -1
  26. {osmsg-1.2.4 → osmsg-1.3.0}/pyproject.toml +1 -1
  27. osmsg-1.2.4/osmsg/__version__.py +0 -1
  28. {osmsg-1.2.4 → osmsg-1.3.0}/LICENSE +0 -0
  29. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/__init__.py +0 -0
  30. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/_http.py +0 -0
  31. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/auth.py +0 -0
  32. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/boundary.py +0 -0
  33. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/__init__.py +0 -0
  34. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/exceptions.py +0 -0
  35. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/__init__.py +0 -0
  36. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/csv.py +0 -0
  37. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/json.py +0 -0
  38. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/fetch.py +0 -0
  39. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/geofabrik.py +0 -0
  40. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/handlers.py +0 -0
  41. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/__init__.py +0 -0
  42. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/manifest.py +0 -0
  43. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/parquet.py +0 -0
  44. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/pbf_split.py +0 -0
  45. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/py.typed +0 -0
  46. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/tm.py +0 -0
  47. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/ui.py +0 -0
  48. {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/workers.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: osmsg
3
- Version: 1.2.4
3
+ Version: 1.3.0
4
4
  Summary: OpenStreetMap Stats Generator: Commandline
5
5
  Keywords: osm,stats,commandline,openstreetmap
6
6
  Author: Kshitij Raj Sharma
@@ -178,7 +178,8 @@ litestar --app api.app:app run --host 0.0.0.0 --port 8000
178
178
 
179
179
  ```text
180
180
  GET /health
181
- GET /api/v1/user-stats?start=2026-05-01T00:00:00Z&end=2026-05-02T00:00:00Z
181
+ GET /api/v2/hashtag/hotosm/summary
182
+ GET /api/v2/hashtag/hotosm/leaderboard
182
183
  GET /docs
183
184
  ```
184
185
 
@@ -230,7 +231,7 @@ docker-compose `environment:` block all reach the same setting. CLI flag wins ov
230
231
  | `--country` | `OSMSG_COUNTRY` | unset | Geofabrik region id(s). Comma-separated when set via env. |
231
232
  | `--boundary` | `OSMSG_BOUNDARY` | unset | GeoJSON path or inline GeoJSON. |
232
233
  | `--url` | `OSMSG_URL` | `minute` | `minute`/`hour`/`day` shortcut or full URL. Comma-separated when set via env. |
233
- | `--workers` | `OSMSG_WORKERS` | cpu count | Parallel workers. |
234
+ | `--workers` | `OSMSG_WORKERS` | cpu count | Parallel parse workers. |
234
235
  | `--cache-dir` | `OSMSG_CACHE_DIR` | platform cache | Where downloaded OSM files are kept across runs. |
235
236
  | `--output-dir` | `OSMSG_OUTPUT_DIR` | `.` | Where `<name>.duckdb` and exports are written. |
236
237
  | `--format` / `-f` | `OSMSG_FORMAT` | `parquet` | Repeat for multiple. Comma-separated when set via env. |
@@ -146,7 +146,8 @@ litestar --app api.app:app run --host 0.0.0.0 --port 8000
146
146
 
147
147
  ```text
148
148
  GET /health
149
- GET /api/v1/user-stats?start=2026-05-01T00:00:00Z&end=2026-05-02T00:00:00Z
149
+ GET /api/v2/hashtag/hotosm/summary
150
+ GET /api/v2/hashtag/hotosm/leaderboard
150
151
  GET /docs
151
152
  ```
152
153
 
@@ -198,7 +199,7 @@ docker-compose `environment:` block all reach the same setting. CLI flag wins ov
198
199
  | `--country` | `OSMSG_COUNTRY` | unset | Geofabrik region id(s). Comma-separated when set via env. |
199
200
  | `--boundary` | `OSMSG_BOUNDARY` | unset | GeoJSON path or inline GeoJSON. |
200
201
  | `--url` | `OSMSG_URL` | `minute` | `minute`/`hour`/`day` shortcut or full URL. Comma-separated when set via env. |
201
- | `--workers` | `OSMSG_WORKERS` | cpu count | Parallel workers. |
202
+ | `--workers` | `OSMSG_WORKERS` | cpu count | Parallel parse workers. |
202
203
  | `--cache-dir` | `OSMSG_CACHE_DIR` | platform cache | Where downloaded OSM files are kept across runs. |
203
204
  | `--output-dir` | `OSMSG_OUTPUT_DIR` | `.` | Where `<name>.duckdb` and exports are written. |
204
205
  | `--format` / `-f` | `OSMSG_FORMAT` | `parquet` | Repeat for multiple. Comma-separated when set via env. |
@@ -0,0 +1 @@
1
+ __version__ = "1.3.0"
@@ -22,6 +22,20 @@ def _has_state(db_path: Path, source_url: str) -> bool:
22
22
  return result
23
23
 
24
24
 
25
+ def _has_any_state(db_path: Path) -> bool:
26
+ """True if the store has a resume position for ANY source. A `--insert --seed-only` seeds the source
27
+ at the granularity matched to the gap (e.g. day), which differs from this tick's default source
28
+ (minute); `--update` reads the store's own tracked source and continues/auto-refines it, so any
29
+ seeded state means we continue rather than bootstrap to now."""
30
+ if not db_path.exists():
31
+ return False
32
+ conn = connect(str(db_path))
33
+ create_tables(conn)
34
+ row = conn.execute("SELECT count(*) FROM state").fetchone()
35
+ conn.close()
36
+ return bool(row and row[0])
37
+
38
+
25
39
  def _parse_arg(args: list[str], flag: str) -> str | None:
26
40
  for i, arg in enumerate(args):
27
41
  if arg == flag and i + 1 < len(args):
@@ -31,13 +45,12 @@ def _parse_arg(args: list[str], flag: str) -> str | None:
31
45
 
32
46
  def main() -> int:
33
47
  extra_args = shlex.split(os.environ.get("OSMSG_EXTRA_ARGS", ""))
34
- bootstrap = os.environ.get("OSMSG_BOOTSTRAP", "hour")
35
- bootstrap_days = os.environ.get("OSMSG_BOOTSTRAP_DAYS")
48
+ bootstrap_days = os.environ.get("OSMSG_BOOTSTRAP_DAYS", "1")
36
49
  name = _parse_arg(extra_args, "--name") or "stats"
37
50
  out = Path(_parse_arg(extra_args, "--output-dir") or "/var/lib/osmsg")
38
51
  country = _parse_arg(extra_args, "--country")
39
52
  explicit_url = _parse_arg(extra_args, "--url")
40
- url = explicit_url or "minute"
53
+ url = explicit_url or "day"
41
54
 
42
55
  out.mkdir(parents=True, exist_ok=True)
43
56
 
@@ -61,12 +74,17 @@ def main() -> int:
61
74
  if not (extra_set & {"--all", "--keys"}):
62
75
  cmd.append("--all")
63
76
 
64
- if _has_state(db_path, source_url):
77
+ # A country run tracks one specific geofabrik source; the planet run continues whatever the store
78
+ # was seeded with (possibly a coarser source than this tick's default), so accept any state.
79
+ has_state = _has_state(db_path, source_url) if country else _has_any_state(db_path)
80
+ if has_state:
65
81
  cmd.append("--update")
66
- elif bootstrap_days:
67
- cmd.extend(["--days", bootstrap_days])
68
82
  else:
69
- cmd.extend(["--last", bootstrap])
83
+ # Cold start at day granularity (coarse, few files); --update then refines day->hour->minute
84
+ # as the backlog shrinks. A normal deployment seeds the same way via `--insert`.
85
+ if explicit_url is None and not country:
86
+ cmd.extend(["--url", "day"])
87
+ cmd.extend(["--days", bootstrap_days])
70
88
 
71
89
  print(f"[osmsg-tick] {' '.join(cmd)}", flush=True)
72
90
  return subprocess.call(cmd)
@@ -0,0 +1,212 @@
1
+ """Assemble query sources into one relation so `stats.py` runs a single query regardless of where
2
+ data lives. History (older than the published frontier) comes from the `hashtag_changeset` rollup;
3
+ the recent tail (from the frontier on) is derived on the fly from the base tables, filtered by the
4
+ hashtag first so only matching changesets are read. The two are unioned and split at the frontier, so
5
+ nothing is counted twice and nothing is dropped, and the recent side is always as fresh as the base.
6
+
7
+ The base is just relations: a local DuckDB store's `changeset_stats`/`changesets`, or an attached
8
+ Postgres's `pg.changeset_stats`/`pg.changesets`. Either way the query runs in DuckDB.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import datetime as dt
14
+
15
+ from .stats import COUNT_COLS
16
+
17
+ _PROJECTION = f"changeset_id, uid, editor, created_at, {', '.join(COUNT_COLS)}, tags"
18
+
19
+
20
+ def _window_clause(start: dt.datetime | None, end: dt.datetime | None) -> tuple[str, list[object]]:
21
+ """Optional `created_at` bounds as SQL fragments plus their params, in order. Half-open [start, end):
22
+ inclusive lower, exclusive upper, matching the frontier split and the v1 API's window semantics."""
23
+ parts: list[str] = []
24
+ params: list[object] = []
25
+ if start is not None:
26
+ parts.append(" AND created_at >= ?")
27
+ params.append(start)
28
+ if end is not None:
29
+ parts.append(" AND created_at < ?")
30
+ params.append(end)
31
+ return "".join(parts), params
32
+
33
+
34
+ def _recent_from_base(stats_rel: str, changesets_rel: str, window_sql: str, prefix_pred: str) -> str:
35
+ """Per-changeset rows built on the fly from the base tables for changesets whose hashtags match any
36
+ requested prefix and are on/after the frontier. Counts are summed across a changeset's seq rows and
37
+ the native `tags` lists merged, matching the rollup shape exactly (one tag representation, one
38
+ breakdown path). The changeset filter runs first, so only matching changesets are read. `prefix_pred`
39
+ is the OR of `lower(h) >= ? AND lower(h) < ?` range tests on the lambda var. Placeholders, in order:
40
+ frontier, (window bounds), (prefix bounds)."""
41
+ sums = ", ".join(f"SUM(s.{c}) AS {c}" for c in COUNT_COLS)
42
+ payload = ", ".join(f"p.{c}" for c in COUNT_COLS)
43
+ return f"""
44
+ WITH matched AS (
45
+ SELECT changeset_id, uid, editor, created_at
46
+ FROM {changesets_rel}
47
+ WHERE created_at >= ?{window_sql} AND len(list_filter(hashtags, h -> {prefix_pred})) > 0
48
+ ),
49
+ per_cs AS (
50
+ SELECT s.changeset_id, {sums}
51
+ FROM {stats_rel} s JOIN matched USING (changeset_id) GROUP BY s.changeset_id
52
+ ),
53
+ tag_rows AS (
54
+ SELECT changeset_id, t.k AS k, t.v AS v,
55
+ SUM(t.c) AS c, SUM(t.m) AS m, SUM(t.len_m) AS len_m
56
+ FROM (
57
+ SELECT s.changeset_id, UNNEST(s.tags) AS t
58
+ FROM {stats_rel} s JOIN matched USING (changeset_id)
59
+ WHERE s.tags IS NOT NULL AND len(s.tags) > 0
60
+ )
61
+ GROUP BY changeset_id, t.k, t.v
62
+ ),
63
+ tags AS (
64
+ SELECT changeset_id, list(struct_pack(k := k, v := v, c := c, m := m, len_m := len_m)) AS tags
65
+ FROM tag_rows GROUP BY changeset_id
66
+ )
67
+ SELECT m.changeset_id, m.uid, m.editor, m.created_at, {payload}, COALESCE(t.tags, []) AS tags
68
+ FROM matched m JOIN per_cs p USING (changeset_id) LEFT JOIN tags t USING (changeset_id)
69
+ """
70
+
71
+
72
+ def hashtag_scope(
73
+ history_rel: str,
74
+ recent_stats_rel: str,
75
+ recent_changesets_rel: str,
76
+ *,
77
+ prefixes: list[tuple[str, str]],
78
+ frontier: dt.datetime,
79
+ start: dt.datetime | None = None,
80
+ end: dt.datetime | None = None,
81
+ ) -> tuple[str, list[object]]:
82
+ """One deduped per-changeset relation for one or more hashtag prefix ranges [lo, hi): the rollup for
83
+ history (created_at < frontier, hashtag range prunes row groups) unioned with the recent tail
84
+ derived from the base (created_at >= frontier). A changeset matching more than one prefix (or
85
+ carrying two matching hashtags) is deduped by changeset_id, so it counts once. An optional half-open
86
+ [start, end) window bounds created_at on both sides, so it intersects the frontier split cleanly (a
87
+ window entirely before the frontier reads only history; entirely after, only the base). `prefixes`
88
+ must be non-empty. Returns (sql, params)."""
89
+ if not prefixes:
90
+ raise ValueError("prefixes must be non-empty")
91
+ window_sql, window_params = _window_clause(start, end)
92
+ prefix_params = [bound for pair in prefixes for bound in pair]
93
+ hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
94
+ recent_pred = " OR ".join("(lower(h) >= ? AND lower(h) < ?)" for _ in prefixes)
95
+ recent = _recent_from_base(recent_stats_rel, recent_changesets_rel, window_sql, recent_pred)
96
+ sql = f"""
97
+ SELECT DISTINCT ON (changeset_id) {_PROJECTION} FROM (
98
+ SELECT {_PROJECTION} FROM {history_rel}
99
+ WHERE ({hist_pred}) AND created_at < ?{window_sql}
100
+ UNION ALL
101
+ ({recent})
102
+ )
103
+ """
104
+ params: list[object] = [
105
+ *prefix_params,
106
+ frontier,
107
+ *window_params,
108
+ frontier,
109
+ *window_params,
110
+ *prefix_params,
111
+ ]
112
+ return sql, params
113
+
114
+
115
+ def history_dedup_scope(
116
+ history_rel: str,
117
+ *,
118
+ prefixes: list[tuple[str, str]],
119
+ frontier: dt.datetime,
120
+ start: dt.datetime | None = None,
121
+ end: dt.datetime | None = None,
122
+ ) -> tuple[str, list[object]]:
123
+ """The deduped per-changeset HISTORY relation alone (created_at < frontier), as its own SELECT so a
124
+ caller can aggregate it and the recent side SEPARATELY and combine the small results. This avoids
125
+ DISTINCT ON over the history+recent UNION, which forces DuckDB to materialize the whole 14M-row
126
+ intermediate for a big hashtag; aggregating each side first keeps the history dedup streaming.
127
+ `prefixes` must be non-empty. Returns (sql, params)."""
128
+ if not prefixes:
129
+ raise ValueError("prefixes must be non-empty")
130
+ window_sql, window_params = _window_clause(start, end)
131
+ prefix_params = [bound for pair in prefixes for bound in pair]
132
+ hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
133
+ sql = (
134
+ f"SELECT DISTINCT ON (changeset_id) {_PROJECTION} FROM {history_rel} "
135
+ f"WHERE ({hist_pred}) AND created_at < ?{window_sql}"
136
+ )
137
+ return sql, [*prefix_params, frontier, *window_params]
138
+
139
+
140
+ def history_scope_count(
141
+ history_rel: str,
142
+ *,
143
+ prefixes: list[tuple[str, str]],
144
+ frontier: dt.datetime,
145
+ start: dt.datetime | None = None,
146
+ end: dt.datetime | None = None,
147
+ ) -> tuple[str, list[object]]:
148
+ """A CHEAP raw `count(*)` of history rows matching the hashtag scope + window (no dedup, hashtag range
149
+ prunes row groups: ~sub-second even for a 14M-changeset hashtag). Used to decide whether a per-user
150
+ tag breakdown is affordable before paying for it. Returns (sql, params)."""
151
+ if not prefixes:
152
+ raise ValueError("prefixes must be non-empty")
153
+ window_sql, window_params = _window_clause(start, end)
154
+ prefix_params = [bound for pair in prefixes for bound in pair]
155
+ hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
156
+ sql = f"SELECT count(*) FROM {history_rel} WHERE ({hist_pred}) AND created_at < ?{window_sql}"
157
+ return sql, [*prefix_params, frontier, *window_params]
158
+
159
+
160
+ def recent_scope(
161
+ recent_stats_rel: str,
162
+ recent_changesets_rel: str,
163
+ *,
164
+ prefixes: list[tuple[str, str]],
165
+ frontier: dt.datetime,
166
+ start: dt.datetime | None = None,
167
+ end: dt.datetime | None = None,
168
+ ) -> tuple[str, list[object]]:
169
+ """The recent per-changeset relation alone (created_at >= frontier), already one row per changeset.
170
+ Companion to `history_dedup_scope` for the aggregate-then-combine path. Returns (sql, params)."""
171
+ if not prefixes:
172
+ raise ValueError("prefixes must be non-empty")
173
+ window_sql, window_params = _window_clause(start, end)
174
+ prefix_params = [bound for pair in prefixes for bound in pair]
175
+ recent_pred = " OR ".join("(lower(h) >= ? AND lower(h) < ?)" for _ in prefixes)
176
+ sql = "(" + _recent_from_base(recent_stats_rel, recent_changesets_rel, window_sql, recent_pred) + ")"
177
+ return sql, [frontier, *window_params, *prefix_params]
178
+
179
+
180
+ def map_scope(
181
+ history_rollup_rel: str,
182
+ recent_changesets_rel: str,
183
+ *,
184
+ prefixes: list[tuple[str, str]],
185
+ frontier: dt.datetime,
186
+ start: dt.datetime | None = None,
187
+ end: dt.datetime | None = None,
188
+ ) -> tuple[str, list[object]]:
189
+ """Deduped changeset centroids `(changeset_id, uid, lon, lat)` for the hashtag union, for the map.
190
+ History centroids come from the rollup's `lon`/`lat`; recent centroids from the base changesets bbox
191
+ midpoint (`recent_changesets_rel` must expose min_lon/min_lat/max_lon/max_lat, e.g. the published
192
+ changesets dataset or the Postgres `changesets` table). Changesets with no bbox are dropped. Same
193
+ frontier split and optional [start, end) window as `hashtag_scope`. `prefixes` must be non-empty."""
194
+ if not prefixes:
195
+ raise ValueError("prefixes must be non-empty")
196
+ window_sql, window_params = _window_clause(start, end)
197
+ prefix_params = [bound for pair in prefixes for bound in pair]
198
+ hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
199
+ recent_pred = " OR ".join("(lower(h) >= ? AND lower(h) < ?)" for _ in prefixes)
200
+ sql = f"""
201
+ SELECT DISTINCT ON (changeset_id) changeset_id, uid, lon, lat FROM (
202
+ SELECT changeset_id, uid, lon, lat FROM {history_rollup_rel}
203
+ WHERE ({hist_pred}) AND created_at < ?{window_sql} AND lon IS NOT NULL
204
+ UNION ALL
205
+ SELECT changeset_id, uid, (min_lon + max_lon) / 2.0 AS lon, (min_lat + max_lat) / 2.0 AS lat
206
+ FROM {recent_changesets_rel}
207
+ WHERE created_at >= ?{window_sql} AND min_lon IS NOT NULL
208
+ AND len(list_filter(hashtags, h -> {recent_pred})) > 0
209
+ )
210
+ """
211
+ params: list[object] = [*prefix_params, frontier, *window_params, frontier, *window_params, *prefix_params]
212
+ return sql, params
@@ -150,7 +150,7 @@ def main(
150
150
  ] = None,
151
151
  workers: Annotated[
152
152
  int | None,
153
- typer.Option(envvar="OSMSG_WORKERS", help="Parallel workers (default: cpu count)."),
153
+ typer.Option(envvar="OSMSG_WORKERS", help="Parallel parse workers (default: cpu count)."),
154
154
  ] = None,
155
155
  rows: Annotated[
156
156
  int | None,
@@ -263,6 +263,14 @@ def main(
263
263
  "whole published history; --start/--end loads a slice. Follow with --update to catch up.",
264
264
  ),
265
265
  ] = False,
266
+ seed_only: Annotated[
267
+ bool,
268
+ typer.Option(
269
+ "--seed-only",
270
+ help="With --insert: seed resume state at the published frontier without ingesting any "
271
+ "history (it stays remote), so --update only ever fetches the live tail. The lightest setup.",
272
+ ),
273
+ ] = False,
266
274
  osh_file: Annotated[
267
275
  str | None,
268
276
  typer.Option("--osh-file", help="Insert from a local .osh.pbf instead of the published dataset."),
@@ -296,6 +304,9 @@ def main(
296
304
  if insert and update:
297
305
  error("--insert and --update are mutually exclusive; insert first, then update.")
298
306
  raise typer.Exit(code=2)
307
+ if seed_only and not insert:
308
+ error("--seed-only is only valid with --insert.")
309
+ raise typer.Exit(code=2)
299
310
  if insert and (last is not None or days is not None):
300
311
  error("--insert takes --start/--end (or no window), not --last/--days.")
301
312
  raise typer.Exit(code=2)
@@ -344,6 +355,7 @@ def main(
344
355
  history_mode="auto" if history else "off",
345
356
  history_url=history_url,
346
357
  insert=insert,
358
+ seed_only=seed_only,
347
359
  osh_file=osh_file,
348
360
  changeset_file=changeset_file,
349
361
  overwrite=overwrite,
@@ -395,10 +407,10 @@ def main(
395
407
  "name",
396
408
  "changesets",
397
409
  "map_changes",
398
- "nodes_create",
399
- "ways_create",
400
- "rels_create",
401
- "poi_create",
410
+ "nodes_created",
411
+ "ways_created",
412
+ "rels_created",
413
+ "poi_created",
402
414
  "hashtags",
403
415
  ),
404
416
  title=f"Top users (showing {display_n} of {result['rows']})",
@@ -1,4 +1,5 @@
1
1
  # No FKs: DuckDB rejects UPDATE on FK-referenced LIST/GEOMETRY columns, which would block changeset upgrades.
2
+ # `tags` is the native tag breakdown (osmsg.stats.TAG_STRUCT_DDL); kept in sync with that constant.
2
3
  DUCKDB_SCHEMA = """
3
4
  CREATE TABLE IF NOT EXISTS users (
4
5
  uid BIGINT PRIMARY KEY,
@@ -28,7 +29,7 @@ CREATE TABLE IF NOT EXISTS changeset_stats (
28
29
  rels_deleted INTEGER DEFAULT 0,
29
30
  poi_created INTEGER DEFAULT 0,
30
31
  poi_modified INTEGER DEFAULT 0,
31
- tag_stats JSON,
32
+ tags STRUCT(k VARCHAR, v VARCHAR, c BIGINT, m BIGINT, len_m DOUBLE)[],
32
33
  PRIMARY KEY (seq_id, changeset_id)
33
34
  );
34
35
  CREATE INDEX IF NOT EXISTS idx_changeset_stats_uid ON changeset_stats(uid);
@@ -10,6 +10,20 @@ import duckdb
10
10
  import pyarrow as pa
11
11
  import pyarrow.parquet as pq
12
12
 
13
+ # Native shard column for the per-changeset tag breakdown, matching the store's
14
+ # STRUCT(k VARCHAR, v VARCHAR, c BIGINT, m BIGINT, len_m DOUBLE)[] so ingest is a direct copy.
15
+ _TAG_PA_TYPE = pa.list_(
16
+ pa.struct(
17
+ [
18
+ pa.field("k", pa.string()),
19
+ pa.field("v", pa.string()),
20
+ pa.field("c", pa.int64()),
21
+ pa.field("m", pa.int64()),
22
+ pa.field("len_m", pa.float64()),
23
+ ]
24
+ )
25
+ )
26
+
13
27
 
14
28
  def _quarantine_corrupt(parquet_dir: Path) -> None:
15
29
  """Rename unreadable parquet shards out of the way so the bulk read doesn't abort."""
@@ -59,7 +73,7 @@ CHANGESET_STATS_SCHEMA = pa.schema(
59
73
  pa.field("rels_deleted", pa.int32()),
60
74
  pa.field("poi_created", pa.int32()),
61
75
  pa.field("poi_modified", pa.int32()),
62
- pa.field("tag_stats", pa.string()),
76
+ pa.field("tags", _TAG_PA_TYPE),
63
77
  ]
64
78
  )
65
79
 
@@ -109,7 +123,12 @@ def merge_parquet_files(conn: duckdb.DuckDBPyConnection, parquet_dir: Path, *, c
109
123
  # read_parquet() takes a literal, escape so quoted paths can't break out.
110
124
  return _sql_escape((parquet_dir / f"temp_*_{name}_*.parquet").as_posix())
111
125
 
112
- conn.execute("BEGIN")
126
+ # Each step auto-commits (no enclosing transaction): DuckDB keeps an uncommitted transaction's changes
127
+ # and the non-spillable PK index in memory until COMMIT, so wrapping a whole multi-day window in one
128
+ # transaction exhausts a constrained memory_limit. Every statement here is idempotent (INSERT OR IGNORE,
129
+ # COALESCE update), so committing per step - and re-merging leftover shards after a crash - is safe.
130
+ # preserve_insertion_order=false lets the merges stream instead of buffering rows to preserve order.
131
+ conn.execute("SET preserve_insertion_order = false")
113
132
  try:
114
133
  if any(parquet_dir.glob("temp_*_users_*.parquet")):
115
134
  conn.execute(f"INSERT OR IGNORE INTO users SELECT uid, username FROM read_parquet('{pattern('users')}')")
@@ -153,6 +172,8 @@ def merge_parquet_files(conn: duckdb.DuckDBPyConnection, parquet_dir: Path, *, c
153
172
  """
154
173
  )
155
174
  if any(parquet_dir.glob("temp_*_changeset_stats_*.parquet")):
175
+ # The shard stores `tags` as a native LIST<STRUCT> (built in the handler), so ingest is a
176
+ # direct column copy.
156
177
  conn.execute(
157
178
  f"""
158
179
  INSERT OR IGNORE INTO changeset_stats
@@ -161,14 +182,12 @@ def merge_parquet_files(conn: duckdb.DuckDBPyConnection, parquet_dir: Path, *, c
161
182
  ways_created, ways_modified, ways_deleted,
162
183
  rels_created, rels_modified, rels_deleted,
163
184
  poi_created, poi_modified,
164
- tag_stats::JSON
185
+ tags
165
186
  FROM read_parquet('{pattern("changeset_stats")}')
166
187
  """
167
188
  )
168
- conn.execute("COMMIT")
169
- except Exception:
170
- conn.execute("ROLLBACK")
171
- raise
189
+ finally:
190
+ conn.execute("SET preserve_insertion_order = true")
172
191
 
173
192
  if cleanup:
174
193
  shutil.rmtree(parquet_dir, ignore_errors=True)
@@ -2,42 +2,41 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
- import json
6
5
  from typing import Any
7
6
 
8
7
  import duckdb
9
8
 
9
+ from ..stats import map_changes_sum, sum_cols
10
+
10
11
 
11
12
  def _rows(result) -> list[dict[str, Any]]:
12
13
  cols = [d[0] for d in result.description]
13
14
  return [dict(zip(cols, r, strict=True)) for r in result.fetchall()]
14
15
 
15
16
 
17
+ def _tags_to_nested(tags: list[dict[str, Any]] | None) -> dict[str, dict[str, dict[str, Any]]]:
18
+ """The native `tags` list (list of {k, v, c, m, len_m}) as the nested {key: {value: {c, m, len}}}
19
+ shape `_accumulate_tags` sums over. len is omitted when absent."""
20
+ out: dict[str, dict[str, dict[str, Any]]] = {}
21
+ for t in tags or []:
22
+ entry: dict[str, Any] = {"c": t["c"], "m": t["m"]}
23
+ if t["len_m"] is not None:
24
+ entry["len"] = t["len_m"]
25
+ out.setdefault(t["k"], {})[t["v"]] = entry
26
+ return out
27
+
28
+
16
29
  def user_stats(conn: duckdb.DuckDBPyConnection, top_n: int | None = None) -> list[dict[str, Any]]:
17
- """One row per user, ranked by total map changes."""
30
+ """One row per user, ranked by total map changes. Counts come from the shared stats vocabulary."""
18
31
  rows = _rows(
19
32
  conn.execute(
20
- """
33
+ f"""
21
34
  SELECT
22
35
  u.uid,
23
36
  u.username AS name,
24
37
  COUNT(DISTINCT cs.changeset_id) AS changesets,
25
- SUM(cs.nodes_created) AS nodes_create,
26
- SUM(cs.nodes_modified) AS nodes_modify,
27
- SUM(cs.nodes_deleted) AS nodes_delete,
28
- SUM(cs.ways_created) AS ways_create,
29
- SUM(cs.ways_modified) AS ways_modify,
30
- SUM(cs.ways_deleted) AS ways_delete,
31
- SUM(cs.rels_created) AS rels_create,
32
- SUM(cs.rels_modified) AS rels_modify,
33
- SUM(cs.rels_deleted) AS rels_delete,
34
- SUM(cs.poi_created) AS poi_create,
35
- SUM(cs.poi_modified) AS poi_modify,
36
- SUM(
37
- cs.nodes_created + cs.nodes_modified + cs.nodes_deleted +
38
- cs.ways_created + cs.ways_modified + cs.ways_deleted +
39
- cs.rels_created + cs.rels_modified + cs.rels_deleted
40
- ) AS map_changes
38
+ {sum_cols("cs")},
39
+ {map_changes_sum("cs")}
41
40
  FROM users u
42
41
  JOIN changeset_stats cs ON u.uid = cs.uid
43
42
  GROUP BY u.uid, u.username
@@ -121,7 +120,7 @@ def attach_tag_stats(
121
120
  tag_mode: str = "none",
122
121
  length_tags: list[str] | None = None,
123
122
  ) -> None:
124
- """In-place: parse the JSON tag_stats column once per row, then aggregate per user."""
123
+ """In-place: read the native `tags` list column once per row, then aggregate per user."""
125
124
  if not rows:
126
125
  return
127
126
  if not (additional_tags or tag_mode != "none" or length_tags):
@@ -138,18 +137,14 @@ def attach_tag_stats(
138
137
  for k in length_tags or []:
139
138
  r.setdefault(f"{k}_len_m", 0)
140
139
 
141
- for uid, tag_stats_json in conn.execute(
142
- "SELECT uid, tag_stats FROM changeset_stats WHERE tag_stats IS NOT NULL"
140
+ for uid, tags in conn.execute(
141
+ "SELECT uid, tags FROM changeset_stats WHERE tags IS NOT NULL AND len(tags) > 0"
143
142
  ).fetchall():
144
- if uid not in by_uid or not tag_stats_json:
145
- continue
146
- try:
147
- payload = json.loads(tag_stats_json) if isinstance(tag_stats_json, str) else tag_stats_json
148
- except (json.JSONDecodeError, TypeError):
143
+ if uid not in by_uid or not tags:
149
144
  continue
150
145
  _accumulate_tags(
151
146
  by_uid[uid],
152
- payload,
147
+ _tags_to_nested(tags),
153
148
  additional_tags=additional_tags,
154
149
  tag_mode=tag_mode,
155
150
  length_tags=length_tags,
@@ -171,27 +166,13 @@ def daily_summary(
171
166
  """One row per UTC day. Requires `changesets` populated (--changeset / --hashtags)."""
172
167
  rows = _rows(
173
168
  conn.execute(
174
- """
169
+ f"""
175
170
  SELECT
176
171
  CAST(DATE_TRUNC('day', cs.created_at) AS DATE)::VARCHAR AS date,
177
172
  COUNT(DISTINCT cs.changeset_id) AS changesets,
178
173
  COUNT(DISTINCT cs.uid) AS users,
179
- SUM(st.nodes_created) AS nodes_create,
180
- SUM(st.nodes_modified) AS nodes_modify,
181
- SUM(st.nodes_deleted) AS nodes_delete,
182
- SUM(st.ways_created) AS ways_create,
183
- SUM(st.ways_modified) AS ways_modify,
184
- SUM(st.ways_deleted) AS ways_delete,
185
- SUM(st.rels_created) AS rels_create,
186
- SUM(st.rels_modified) AS rels_modify,
187
- SUM(st.rels_deleted) AS rels_delete,
188
- SUM(st.poi_created) AS poi_create,
189
- SUM(st.poi_modified) AS poi_modify,
190
- SUM(
191
- st.nodes_created + st.nodes_modified + st.nodes_deleted +
192
- st.ways_created + st.ways_modified + st.ways_deleted +
193
- st.rels_created + st.rels_modified + st.rels_deleted
194
- ) AS map_changes
174
+ {sum_cols("st")},
175
+ {map_changes_sum("st")}
195
176
  FROM changesets cs
196
177
  JOIN changeset_stats st ON cs.changeset_id = st.changeset_id
197
178
  GROUP BY DATE_TRUNC('day', cs.created_at)
@@ -227,22 +208,18 @@ def daily_summary(
227
208
  for k in length_tags or []:
228
209
  r.setdefault(f"{k}_len_m", 0)
229
210
 
230
- for date, tag_stats_json in conn.execute(
211
+ for date, tags in conn.execute(
231
212
  """
232
- SELECT CAST(DATE_TRUNC('day', cs.created_at) AS DATE)::VARCHAR, st.tag_stats
213
+ SELECT CAST(DATE_TRUNC('day', cs.created_at) AS DATE)::VARCHAR, st.tags
233
214
  FROM changesets cs JOIN changeset_stats st ON cs.changeset_id = st.changeset_id
234
- WHERE st.tag_stats IS NOT NULL
215
+ WHERE st.tags IS NOT NULL AND len(st.tags) > 0
235
216
  """
236
217
  ).fetchall():
237
- if date not in by_date or not tag_stats_json:
238
- continue
239
- try:
240
- payload = json.loads(tag_stats_json) if isinstance(tag_stats_json, str) else tag_stats_json
241
- except (json.JSONDecodeError, TypeError):
218
+ if date not in by_date or not tags:
242
219
  continue
243
220
  _accumulate_tags(
244
221
  by_date[date],
245
- payload,
222
+ _tags_to_nested(tags),
246
223
  additional_tags=additional_tags,
247
224
  tag_mode=tag_mode,
248
225
  length_tags=length_tags,
@@ -1,14 +1,41 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import os
4
+ import re
3
5
  from typing import Any
4
6
 
5
7
  import duckdb
6
8
 
7
9
  from .duckdb_schema import DUCKDB_SCHEMA
8
10
 
11
+ _MEMORY_LIMIT_RE = re.compile(r"^\d+(\.\d+)?\s?(B|KB|MB|GB|TB|KiB|MiB|GiB|TiB)$", re.IGNORECASE)
12
+
13
+
14
+ def _apply_runtime_pragmas(conn: duckdb.DuckDBPyConnection) -> None:
15
+ """Bound DuckDB memory and point spilling at a roomy disk, from operator env.
16
+
17
+ Unset means DuckDB defaults, so library and test behaviour is unchanged. On a
18
+ memory-capped host these keep a large merge/aggregation spilling to disk instead
19
+ of aborting with an out-of-memory error.
20
+ """
21
+ memory_limit = os.environ.get("OSMSG_DUCKDB_MEMORY_LIMIT")
22
+ if memory_limit:
23
+ if not _MEMORY_LIMIT_RE.match(memory_limit):
24
+ raise ValueError(f"OSMSG_DUCKDB_MEMORY_LIMIT must be like '1GB', got {memory_limit!r}")
25
+ conn.execute(f"SET memory_limit='{memory_limit}'")
26
+ threads = os.environ.get("OSMSG_DUCKDB_THREADS")
27
+ if threads:
28
+ conn.execute(f"SET threads={int(threads)}")
29
+ temp_directory = os.environ.get("OSMSG_DUCKDB_TEMP_DIR")
30
+ if temp_directory:
31
+ os.makedirs(temp_directory, exist_ok=True)
32
+ conn.execute(f"SET temp_directory='{temp_directory.replace(chr(39), chr(39) * 2)}'")
33
+
9
34
 
10
35
  def connect(db_path: str) -> duckdb.DuckDBPyConnection:
11
- return duckdb.connect(db_path)
36
+ conn = duckdb.connect(db_path)
37
+ _apply_runtime_pragmas(conn)
38
+ return conn
12
39
 
13
40
 
14
41
  def close(conn: duckdb.DuckDBPyConnection) -> None: