osmsg 1.2.4__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {osmsg-1.2.4 → osmsg-1.3.0}/PKG-INFO +4 -3
- {osmsg-1.2.4 → osmsg-1.3.0}/README.md +3 -2
- osmsg-1.3.0/osmsg/__version__.py +1 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/_tick.py +25 -7
- osmsg-1.3.0/osmsg/catalog.py +212 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/cli.py +17 -5
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/duckdb_schema.py +2 -1
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/ingest.py +26 -7
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/queries.py +31 -54
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/schema.py +28 -1
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/markdown.py +9 -9
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/parquet.py +11 -11
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/psql.py +95 -9
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/gui.py +27 -12
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/history.py +46 -30
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/cli.py +20 -1
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/convert.py +7 -10
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/month.py +16 -7
- osmsg-1.3.0/osmsg/maintain/rollup.py +151 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/models.py +11 -13
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/pg_schema.py +19 -6
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/pipeline.py +129 -67
- osmsg-1.3.0/osmsg/prune.py +50 -0
- osmsg-1.3.0/osmsg/query.py +462 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/replication.py +1 -1
- {osmsg-1.2.4 → osmsg-1.3.0}/pyproject.toml +1 -1
- osmsg-1.2.4/osmsg/__version__.py +0 -1
- {osmsg-1.2.4 → osmsg-1.3.0}/LICENSE +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/__init__.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/_http.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/auth.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/boundary.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/db/__init__.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/exceptions.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/__init__.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/csv.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/export/json.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/fetch.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/geofabrik.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/handlers.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/__init__.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/manifest.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/parquet.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/maintain/pbf_split.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/py.typed +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/tm.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/ui.py +0 -0
- {osmsg-1.2.4 → osmsg-1.3.0}/osmsg/workers.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: osmsg
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: OpenStreetMap Stats Generator: Commandline
|
|
5
5
|
Keywords: osm,stats,commandline,openstreetmap
|
|
6
6
|
Author: Kshitij Raj Sharma
|
|
@@ -178,7 +178,8 @@ litestar --app api.app:app run --host 0.0.0.0 --port 8000
|
|
|
178
178
|
|
|
179
179
|
```text
|
|
180
180
|
GET /health
|
|
181
|
-
GET /api/
|
|
181
|
+
GET /api/v2/hashtag/hotosm/summary
|
|
182
|
+
GET /api/v2/hashtag/hotosm/leaderboard
|
|
182
183
|
GET /docs
|
|
183
184
|
```
|
|
184
185
|
|
|
@@ -230,7 +231,7 @@ docker-compose `environment:` block all reach the same setting. CLI flag wins ov
|
|
|
230
231
|
| `--country` | `OSMSG_COUNTRY` | unset | Geofabrik region id(s). Comma-separated when set via env. |
|
|
231
232
|
| `--boundary` | `OSMSG_BOUNDARY` | unset | GeoJSON path or inline GeoJSON. |
|
|
232
233
|
| `--url` | `OSMSG_URL` | `minute` | `minute`/`hour`/`day` shortcut or full URL. Comma-separated when set via env. |
|
|
233
|
-
| `--workers` | `OSMSG_WORKERS` | cpu count | Parallel workers. |
|
|
234
|
+
| `--workers` | `OSMSG_WORKERS` | cpu count | Parallel parse workers. |
|
|
234
235
|
| `--cache-dir` | `OSMSG_CACHE_DIR` | platform cache | Where downloaded OSM files are kept across runs. |
|
|
235
236
|
| `--output-dir` | `OSMSG_OUTPUT_DIR` | `.` | Where `<name>.duckdb` and exports are written. |
|
|
236
237
|
| `--format` / `-f` | `OSMSG_FORMAT` | `parquet` | Repeat for multiple. Comma-separated when set via env. |
|
|
@@ -146,7 +146,8 @@ litestar --app api.app:app run --host 0.0.0.0 --port 8000
|
|
|
146
146
|
|
|
147
147
|
```text
|
|
148
148
|
GET /health
|
|
149
|
-
GET /api/
|
|
149
|
+
GET /api/v2/hashtag/hotosm/summary
|
|
150
|
+
GET /api/v2/hashtag/hotosm/leaderboard
|
|
150
151
|
GET /docs
|
|
151
152
|
```
|
|
152
153
|
|
|
@@ -198,7 +199,7 @@ docker-compose `environment:` block all reach the same setting. CLI flag wins ov
|
|
|
198
199
|
| `--country` | `OSMSG_COUNTRY` | unset | Geofabrik region id(s). Comma-separated when set via env. |
|
|
199
200
|
| `--boundary` | `OSMSG_BOUNDARY` | unset | GeoJSON path or inline GeoJSON. |
|
|
200
201
|
| `--url` | `OSMSG_URL` | `minute` | `minute`/`hour`/`day` shortcut or full URL. Comma-separated when set via env. |
|
|
201
|
-
| `--workers` | `OSMSG_WORKERS` | cpu count | Parallel workers. |
|
|
202
|
+
| `--workers` | `OSMSG_WORKERS` | cpu count | Parallel parse workers. |
|
|
202
203
|
| `--cache-dir` | `OSMSG_CACHE_DIR` | platform cache | Where downloaded OSM files are kept across runs. |
|
|
203
204
|
| `--output-dir` | `OSMSG_OUTPUT_DIR` | `.` | Where `<name>.duckdb` and exports are written. |
|
|
204
205
|
| `--format` / `-f` | `OSMSG_FORMAT` | `parquet` | Repeat for multiple. Comma-separated when set via env. |
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.3.0"
|
|
@@ -22,6 +22,20 @@ def _has_state(db_path: Path, source_url: str) -> bool:
|
|
|
22
22
|
return result
|
|
23
23
|
|
|
24
24
|
|
|
25
|
+
def _has_any_state(db_path: Path) -> bool:
|
|
26
|
+
"""True if the store has a resume position for ANY source. A `--insert --seed-only` seeds the source
|
|
27
|
+
at the granularity matched to the gap (e.g. day), which differs from this tick's default source
|
|
28
|
+
(minute); `--update` reads the store's own tracked source and continues/auto-refines it, so any
|
|
29
|
+
seeded state means we continue rather than bootstrap to now."""
|
|
30
|
+
if not db_path.exists():
|
|
31
|
+
return False
|
|
32
|
+
conn = connect(str(db_path))
|
|
33
|
+
create_tables(conn)
|
|
34
|
+
row = conn.execute("SELECT count(*) FROM state").fetchone()
|
|
35
|
+
conn.close()
|
|
36
|
+
return bool(row and row[0])
|
|
37
|
+
|
|
38
|
+
|
|
25
39
|
def _parse_arg(args: list[str], flag: str) -> str | None:
|
|
26
40
|
for i, arg in enumerate(args):
|
|
27
41
|
if arg == flag and i + 1 < len(args):
|
|
@@ -31,13 +45,12 @@ def _parse_arg(args: list[str], flag: str) -> str | None:
|
|
|
31
45
|
|
|
32
46
|
def main() -> int:
|
|
33
47
|
extra_args = shlex.split(os.environ.get("OSMSG_EXTRA_ARGS", ""))
|
|
34
|
-
|
|
35
|
-
bootstrap_days = os.environ.get("OSMSG_BOOTSTRAP_DAYS")
|
|
48
|
+
bootstrap_days = os.environ.get("OSMSG_BOOTSTRAP_DAYS", "1")
|
|
36
49
|
name = _parse_arg(extra_args, "--name") or "stats"
|
|
37
50
|
out = Path(_parse_arg(extra_args, "--output-dir") or "/var/lib/osmsg")
|
|
38
51
|
country = _parse_arg(extra_args, "--country")
|
|
39
52
|
explicit_url = _parse_arg(extra_args, "--url")
|
|
40
|
-
url = explicit_url or "
|
|
53
|
+
url = explicit_url or "day"
|
|
41
54
|
|
|
42
55
|
out.mkdir(parents=True, exist_ok=True)
|
|
43
56
|
|
|
@@ -61,12 +74,17 @@ def main() -> int:
|
|
|
61
74
|
if not (extra_set & {"--all", "--keys"}):
|
|
62
75
|
cmd.append("--all")
|
|
63
76
|
|
|
64
|
-
|
|
77
|
+
# A country run tracks one specific geofabrik source; the planet run continues whatever the store
|
|
78
|
+
# was seeded with (possibly a coarser source than this tick's default), so accept any state.
|
|
79
|
+
has_state = _has_state(db_path, source_url) if country else _has_any_state(db_path)
|
|
80
|
+
if has_state:
|
|
65
81
|
cmd.append("--update")
|
|
66
|
-
elif bootstrap_days:
|
|
67
|
-
cmd.extend(["--days", bootstrap_days])
|
|
68
82
|
else:
|
|
69
|
-
|
|
83
|
+
# Cold start at day granularity (coarse, few files); --update then refines day->hour->minute
|
|
84
|
+
# as the backlog shrinks. A normal deployment seeds the same way via `--insert`.
|
|
85
|
+
if explicit_url is None and not country:
|
|
86
|
+
cmd.extend(["--url", "day"])
|
|
87
|
+
cmd.extend(["--days", bootstrap_days])
|
|
70
88
|
|
|
71
89
|
print(f"[osmsg-tick] {' '.join(cmd)}", flush=True)
|
|
72
90
|
return subprocess.call(cmd)
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Assemble query sources into one relation so `stats.py` runs a single query regardless of where
|
|
2
|
+
data lives. History (older than the published frontier) comes from the `hashtag_changeset` rollup;
|
|
3
|
+
the recent tail (from the frontier on) is derived on the fly from the base tables, filtered by the
|
|
4
|
+
hashtag first so only matching changesets are read. The two are unioned and split at the frontier, so
|
|
5
|
+
nothing is counted twice and nothing is dropped, and the recent side is always as fresh as the base.
|
|
6
|
+
|
|
7
|
+
The base is just relations: a local DuckDB store's `changeset_stats`/`changesets`, or an attached
|
|
8
|
+
Postgres's `pg.changeset_stats`/`pg.changesets`. Either way the query runs in DuckDB.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import datetime as dt
|
|
14
|
+
|
|
15
|
+
from .stats import COUNT_COLS
|
|
16
|
+
|
|
17
|
+
_PROJECTION = f"changeset_id, uid, editor, created_at, {', '.join(COUNT_COLS)}, tags"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _window_clause(start: dt.datetime | None, end: dt.datetime | None) -> tuple[str, list[object]]:
|
|
21
|
+
"""Optional `created_at` bounds as SQL fragments plus their params, in order. Half-open [start, end):
|
|
22
|
+
inclusive lower, exclusive upper, matching the frontier split and the v1 API's window semantics."""
|
|
23
|
+
parts: list[str] = []
|
|
24
|
+
params: list[object] = []
|
|
25
|
+
if start is not None:
|
|
26
|
+
parts.append(" AND created_at >= ?")
|
|
27
|
+
params.append(start)
|
|
28
|
+
if end is not None:
|
|
29
|
+
parts.append(" AND created_at < ?")
|
|
30
|
+
params.append(end)
|
|
31
|
+
return "".join(parts), params
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _recent_from_base(stats_rel: str, changesets_rel: str, window_sql: str, prefix_pred: str) -> str:
|
|
35
|
+
"""Per-changeset rows built on the fly from the base tables for changesets whose hashtags match any
|
|
36
|
+
requested prefix and are on/after the frontier. Counts are summed across a changeset's seq rows and
|
|
37
|
+
the native `tags` lists merged, matching the rollup shape exactly (one tag representation, one
|
|
38
|
+
breakdown path). The changeset filter runs first, so only matching changesets are read. `prefix_pred`
|
|
39
|
+
is the OR of `lower(h) >= ? AND lower(h) < ?` range tests on the lambda var. Placeholders, in order:
|
|
40
|
+
frontier, (window bounds), (prefix bounds)."""
|
|
41
|
+
sums = ", ".join(f"SUM(s.{c}) AS {c}" for c in COUNT_COLS)
|
|
42
|
+
payload = ", ".join(f"p.{c}" for c in COUNT_COLS)
|
|
43
|
+
return f"""
|
|
44
|
+
WITH matched AS (
|
|
45
|
+
SELECT changeset_id, uid, editor, created_at
|
|
46
|
+
FROM {changesets_rel}
|
|
47
|
+
WHERE created_at >= ?{window_sql} AND len(list_filter(hashtags, h -> {prefix_pred})) > 0
|
|
48
|
+
),
|
|
49
|
+
per_cs AS (
|
|
50
|
+
SELECT s.changeset_id, {sums}
|
|
51
|
+
FROM {stats_rel} s JOIN matched USING (changeset_id) GROUP BY s.changeset_id
|
|
52
|
+
),
|
|
53
|
+
tag_rows AS (
|
|
54
|
+
SELECT changeset_id, t.k AS k, t.v AS v,
|
|
55
|
+
SUM(t.c) AS c, SUM(t.m) AS m, SUM(t.len_m) AS len_m
|
|
56
|
+
FROM (
|
|
57
|
+
SELECT s.changeset_id, UNNEST(s.tags) AS t
|
|
58
|
+
FROM {stats_rel} s JOIN matched USING (changeset_id)
|
|
59
|
+
WHERE s.tags IS NOT NULL AND len(s.tags) > 0
|
|
60
|
+
)
|
|
61
|
+
GROUP BY changeset_id, t.k, t.v
|
|
62
|
+
),
|
|
63
|
+
tags AS (
|
|
64
|
+
SELECT changeset_id, list(struct_pack(k := k, v := v, c := c, m := m, len_m := len_m)) AS tags
|
|
65
|
+
FROM tag_rows GROUP BY changeset_id
|
|
66
|
+
)
|
|
67
|
+
SELECT m.changeset_id, m.uid, m.editor, m.created_at, {payload}, COALESCE(t.tags, []) AS tags
|
|
68
|
+
FROM matched m JOIN per_cs p USING (changeset_id) LEFT JOIN tags t USING (changeset_id)
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def hashtag_scope(
|
|
73
|
+
history_rel: str,
|
|
74
|
+
recent_stats_rel: str,
|
|
75
|
+
recent_changesets_rel: str,
|
|
76
|
+
*,
|
|
77
|
+
prefixes: list[tuple[str, str]],
|
|
78
|
+
frontier: dt.datetime,
|
|
79
|
+
start: dt.datetime | None = None,
|
|
80
|
+
end: dt.datetime | None = None,
|
|
81
|
+
) -> tuple[str, list[object]]:
|
|
82
|
+
"""One deduped per-changeset relation for one or more hashtag prefix ranges [lo, hi): the rollup for
|
|
83
|
+
history (created_at < frontier, hashtag range prunes row groups) unioned with the recent tail
|
|
84
|
+
derived from the base (created_at >= frontier). A changeset matching more than one prefix (or
|
|
85
|
+
carrying two matching hashtags) is deduped by changeset_id, so it counts once. An optional half-open
|
|
86
|
+
[start, end) window bounds created_at on both sides, so it intersects the frontier split cleanly (a
|
|
87
|
+
window entirely before the frontier reads only history; entirely after, only the base). `prefixes`
|
|
88
|
+
must be non-empty. Returns (sql, params)."""
|
|
89
|
+
if not prefixes:
|
|
90
|
+
raise ValueError("prefixes must be non-empty")
|
|
91
|
+
window_sql, window_params = _window_clause(start, end)
|
|
92
|
+
prefix_params = [bound for pair in prefixes for bound in pair]
|
|
93
|
+
hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
|
|
94
|
+
recent_pred = " OR ".join("(lower(h) >= ? AND lower(h) < ?)" for _ in prefixes)
|
|
95
|
+
recent = _recent_from_base(recent_stats_rel, recent_changesets_rel, window_sql, recent_pred)
|
|
96
|
+
sql = f"""
|
|
97
|
+
SELECT DISTINCT ON (changeset_id) {_PROJECTION} FROM (
|
|
98
|
+
SELECT {_PROJECTION} FROM {history_rel}
|
|
99
|
+
WHERE ({hist_pred}) AND created_at < ?{window_sql}
|
|
100
|
+
UNION ALL
|
|
101
|
+
({recent})
|
|
102
|
+
)
|
|
103
|
+
"""
|
|
104
|
+
params: list[object] = [
|
|
105
|
+
*prefix_params,
|
|
106
|
+
frontier,
|
|
107
|
+
*window_params,
|
|
108
|
+
frontier,
|
|
109
|
+
*window_params,
|
|
110
|
+
*prefix_params,
|
|
111
|
+
]
|
|
112
|
+
return sql, params
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def history_dedup_scope(
|
|
116
|
+
history_rel: str,
|
|
117
|
+
*,
|
|
118
|
+
prefixes: list[tuple[str, str]],
|
|
119
|
+
frontier: dt.datetime,
|
|
120
|
+
start: dt.datetime | None = None,
|
|
121
|
+
end: dt.datetime | None = None,
|
|
122
|
+
) -> tuple[str, list[object]]:
|
|
123
|
+
"""The deduped per-changeset HISTORY relation alone (created_at < frontier), as its own SELECT so a
|
|
124
|
+
caller can aggregate it and the recent side SEPARATELY and combine the small results. This avoids
|
|
125
|
+
DISTINCT ON over the history+recent UNION, which forces DuckDB to materialize the whole 14M-row
|
|
126
|
+
intermediate for a big hashtag; aggregating each side first keeps the history dedup streaming.
|
|
127
|
+
`prefixes` must be non-empty. Returns (sql, params)."""
|
|
128
|
+
if not prefixes:
|
|
129
|
+
raise ValueError("prefixes must be non-empty")
|
|
130
|
+
window_sql, window_params = _window_clause(start, end)
|
|
131
|
+
prefix_params = [bound for pair in prefixes for bound in pair]
|
|
132
|
+
hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
|
|
133
|
+
sql = (
|
|
134
|
+
f"SELECT DISTINCT ON (changeset_id) {_PROJECTION} FROM {history_rel} "
|
|
135
|
+
f"WHERE ({hist_pred}) AND created_at < ?{window_sql}"
|
|
136
|
+
)
|
|
137
|
+
return sql, [*prefix_params, frontier, *window_params]
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def history_scope_count(
|
|
141
|
+
history_rel: str,
|
|
142
|
+
*,
|
|
143
|
+
prefixes: list[tuple[str, str]],
|
|
144
|
+
frontier: dt.datetime,
|
|
145
|
+
start: dt.datetime | None = None,
|
|
146
|
+
end: dt.datetime | None = None,
|
|
147
|
+
) -> tuple[str, list[object]]:
|
|
148
|
+
"""A CHEAP raw `count(*)` of history rows matching the hashtag scope + window (no dedup, hashtag range
|
|
149
|
+
prunes row groups: ~sub-second even for a 14M-changeset hashtag). Used to decide whether a per-user
|
|
150
|
+
tag breakdown is affordable before paying for it. Returns (sql, params)."""
|
|
151
|
+
if not prefixes:
|
|
152
|
+
raise ValueError("prefixes must be non-empty")
|
|
153
|
+
window_sql, window_params = _window_clause(start, end)
|
|
154
|
+
prefix_params = [bound for pair in prefixes for bound in pair]
|
|
155
|
+
hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
|
|
156
|
+
sql = f"SELECT count(*) FROM {history_rel} WHERE ({hist_pred}) AND created_at < ?{window_sql}"
|
|
157
|
+
return sql, [*prefix_params, frontier, *window_params]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def recent_scope(
|
|
161
|
+
recent_stats_rel: str,
|
|
162
|
+
recent_changesets_rel: str,
|
|
163
|
+
*,
|
|
164
|
+
prefixes: list[tuple[str, str]],
|
|
165
|
+
frontier: dt.datetime,
|
|
166
|
+
start: dt.datetime | None = None,
|
|
167
|
+
end: dt.datetime | None = None,
|
|
168
|
+
) -> tuple[str, list[object]]:
|
|
169
|
+
"""The recent per-changeset relation alone (created_at >= frontier), already one row per changeset.
|
|
170
|
+
Companion to `history_dedup_scope` for the aggregate-then-combine path. Returns (sql, params)."""
|
|
171
|
+
if not prefixes:
|
|
172
|
+
raise ValueError("prefixes must be non-empty")
|
|
173
|
+
window_sql, window_params = _window_clause(start, end)
|
|
174
|
+
prefix_params = [bound for pair in prefixes for bound in pair]
|
|
175
|
+
recent_pred = " OR ".join("(lower(h) >= ? AND lower(h) < ?)" for _ in prefixes)
|
|
176
|
+
sql = "(" + _recent_from_base(recent_stats_rel, recent_changesets_rel, window_sql, recent_pred) + ")"
|
|
177
|
+
return sql, [frontier, *window_params, *prefix_params]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def map_scope(
|
|
181
|
+
history_rollup_rel: str,
|
|
182
|
+
recent_changesets_rel: str,
|
|
183
|
+
*,
|
|
184
|
+
prefixes: list[tuple[str, str]],
|
|
185
|
+
frontier: dt.datetime,
|
|
186
|
+
start: dt.datetime | None = None,
|
|
187
|
+
end: dt.datetime | None = None,
|
|
188
|
+
) -> tuple[str, list[object]]:
|
|
189
|
+
"""Deduped changeset centroids `(changeset_id, uid, lon, lat)` for the hashtag union, for the map.
|
|
190
|
+
History centroids come from the rollup's `lon`/`lat`; recent centroids from the base changesets bbox
|
|
191
|
+
midpoint (`recent_changesets_rel` must expose min_lon/min_lat/max_lon/max_lat, e.g. the published
|
|
192
|
+
changesets dataset or the Postgres `changesets` table). Changesets with no bbox are dropped. Same
|
|
193
|
+
frontier split and optional [start, end) window as `hashtag_scope`. `prefixes` must be non-empty."""
|
|
194
|
+
if not prefixes:
|
|
195
|
+
raise ValueError("prefixes must be non-empty")
|
|
196
|
+
window_sql, window_params = _window_clause(start, end)
|
|
197
|
+
prefix_params = [bound for pair in prefixes for bound in pair]
|
|
198
|
+
hist_pred = " OR ".join("(hashtag >= ? AND hashtag < ?)" for _ in prefixes)
|
|
199
|
+
recent_pred = " OR ".join("(lower(h) >= ? AND lower(h) < ?)" for _ in prefixes)
|
|
200
|
+
sql = f"""
|
|
201
|
+
SELECT DISTINCT ON (changeset_id) changeset_id, uid, lon, lat FROM (
|
|
202
|
+
SELECT changeset_id, uid, lon, lat FROM {history_rollup_rel}
|
|
203
|
+
WHERE ({hist_pred}) AND created_at < ?{window_sql} AND lon IS NOT NULL
|
|
204
|
+
UNION ALL
|
|
205
|
+
SELECT changeset_id, uid, (min_lon + max_lon) / 2.0 AS lon, (min_lat + max_lat) / 2.0 AS lat
|
|
206
|
+
FROM {recent_changesets_rel}
|
|
207
|
+
WHERE created_at >= ?{window_sql} AND min_lon IS NOT NULL
|
|
208
|
+
AND len(list_filter(hashtags, h -> {recent_pred})) > 0
|
|
209
|
+
)
|
|
210
|
+
"""
|
|
211
|
+
params: list[object] = [*prefix_params, frontier, *window_params, frontier, *window_params, *prefix_params]
|
|
212
|
+
return sql, params
|
|
@@ -150,7 +150,7 @@ def main(
|
|
|
150
150
|
] = None,
|
|
151
151
|
workers: Annotated[
|
|
152
152
|
int | None,
|
|
153
|
-
typer.Option(envvar="OSMSG_WORKERS", help="Parallel workers (default: cpu count)."),
|
|
153
|
+
typer.Option(envvar="OSMSG_WORKERS", help="Parallel parse workers (default: cpu count)."),
|
|
154
154
|
] = None,
|
|
155
155
|
rows: Annotated[
|
|
156
156
|
int | None,
|
|
@@ -263,6 +263,14 @@ def main(
|
|
|
263
263
|
"whole published history; --start/--end loads a slice. Follow with --update to catch up.",
|
|
264
264
|
),
|
|
265
265
|
] = False,
|
|
266
|
+
seed_only: Annotated[
|
|
267
|
+
bool,
|
|
268
|
+
typer.Option(
|
|
269
|
+
"--seed-only",
|
|
270
|
+
help="With --insert: seed resume state at the published frontier without ingesting any "
|
|
271
|
+
"history (it stays remote), so --update only ever fetches the live tail. The lightest setup.",
|
|
272
|
+
),
|
|
273
|
+
] = False,
|
|
266
274
|
osh_file: Annotated[
|
|
267
275
|
str | None,
|
|
268
276
|
typer.Option("--osh-file", help="Insert from a local .osh.pbf instead of the published dataset."),
|
|
@@ -296,6 +304,9 @@ def main(
|
|
|
296
304
|
if insert and update:
|
|
297
305
|
error("--insert and --update are mutually exclusive; insert first, then update.")
|
|
298
306
|
raise typer.Exit(code=2)
|
|
307
|
+
if seed_only and not insert:
|
|
308
|
+
error("--seed-only is only valid with --insert.")
|
|
309
|
+
raise typer.Exit(code=2)
|
|
299
310
|
if insert and (last is not None or days is not None):
|
|
300
311
|
error("--insert takes --start/--end (or no window), not --last/--days.")
|
|
301
312
|
raise typer.Exit(code=2)
|
|
@@ -344,6 +355,7 @@ def main(
|
|
|
344
355
|
history_mode="auto" if history else "off",
|
|
345
356
|
history_url=history_url,
|
|
346
357
|
insert=insert,
|
|
358
|
+
seed_only=seed_only,
|
|
347
359
|
osh_file=osh_file,
|
|
348
360
|
changeset_file=changeset_file,
|
|
349
361
|
overwrite=overwrite,
|
|
@@ -395,10 +407,10 @@ def main(
|
|
|
395
407
|
"name",
|
|
396
408
|
"changesets",
|
|
397
409
|
"map_changes",
|
|
398
|
-
"
|
|
399
|
-
"
|
|
400
|
-
"
|
|
401
|
-
"
|
|
410
|
+
"nodes_created",
|
|
411
|
+
"ways_created",
|
|
412
|
+
"rels_created",
|
|
413
|
+
"poi_created",
|
|
402
414
|
"hashtags",
|
|
403
415
|
),
|
|
404
416
|
title=f"Top users (showing {display_n} of {result['rows']})",
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
# No FKs: DuckDB rejects UPDATE on FK-referenced LIST/GEOMETRY columns, which would block changeset upgrades.
|
|
2
|
+
# `tags` is the native tag breakdown (osmsg.stats.TAG_STRUCT_DDL); kept in sync with that constant.
|
|
2
3
|
DUCKDB_SCHEMA = """
|
|
3
4
|
CREATE TABLE IF NOT EXISTS users (
|
|
4
5
|
uid BIGINT PRIMARY KEY,
|
|
@@ -28,7 +29,7 @@ CREATE TABLE IF NOT EXISTS changeset_stats (
|
|
|
28
29
|
rels_deleted INTEGER DEFAULT 0,
|
|
29
30
|
poi_created INTEGER DEFAULT 0,
|
|
30
31
|
poi_modified INTEGER DEFAULT 0,
|
|
31
|
-
|
|
32
|
+
tags STRUCT(k VARCHAR, v VARCHAR, c BIGINT, m BIGINT, len_m DOUBLE)[],
|
|
32
33
|
PRIMARY KEY (seq_id, changeset_id)
|
|
33
34
|
);
|
|
34
35
|
CREATE INDEX IF NOT EXISTS idx_changeset_stats_uid ON changeset_stats(uid);
|
|
@@ -10,6 +10,20 @@ import duckdb
|
|
|
10
10
|
import pyarrow as pa
|
|
11
11
|
import pyarrow.parquet as pq
|
|
12
12
|
|
|
13
|
+
# Native shard column for the per-changeset tag breakdown, matching the store's
|
|
14
|
+
# STRUCT(k VARCHAR, v VARCHAR, c BIGINT, m BIGINT, len_m DOUBLE)[] so ingest is a direct copy.
|
|
15
|
+
_TAG_PA_TYPE = pa.list_(
|
|
16
|
+
pa.struct(
|
|
17
|
+
[
|
|
18
|
+
pa.field("k", pa.string()),
|
|
19
|
+
pa.field("v", pa.string()),
|
|
20
|
+
pa.field("c", pa.int64()),
|
|
21
|
+
pa.field("m", pa.int64()),
|
|
22
|
+
pa.field("len_m", pa.float64()),
|
|
23
|
+
]
|
|
24
|
+
)
|
|
25
|
+
)
|
|
26
|
+
|
|
13
27
|
|
|
14
28
|
def _quarantine_corrupt(parquet_dir: Path) -> None:
|
|
15
29
|
"""Rename unreadable parquet shards out of the way so the bulk read doesn't abort."""
|
|
@@ -59,7 +73,7 @@ CHANGESET_STATS_SCHEMA = pa.schema(
|
|
|
59
73
|
pa.field("rels_deleted", pa.int32()),
|
|
60
74
|
pa.field("poi_created", pa.int32()),
|
|
61
75
|
pa.field("poi_modified", pa.int32()),
|
|
62
|
-
pa.field("
|
|
76
|
+
pa.field("tags", _TAG_PA_TYPE),
|
|
63
77
|
]
|
|
64
78
|
)
|
|
65
79
|
|
|
@@ -109,7 +123,12 @@ def merge_parquet_files(conn: duckdb.DuckDBPyConnection, parquet_dir: Path, *, c
|
|
|
109
123
|
# read_parquet() takes a literal, escape so quoted paths can't break out.
|
|
110
124
|
return _sql_escape((parquet_dir / f"temp_*_{name}_*.parquet").as_posix())
|
|
111
125
|
|
|
112
|
-
|
|
126
|
+
# Each step auto-commits (no enclosing transaction): DuckDB keeps an uncommitted transaction's changes
|
|
127
|
+
# and the non-spillable PK index in memory until COMMIT, so wrapping a whole multi-day window in one
|
|
128
|
+
# transaction exhausts a constrained memory_limit. Every statement here is idempotent (INSERT OR IGNORE,
|
|
129
|
+
# COALESCE update), so committing per step - and re-merging leftover shards after a crash - is safe.
|
|
130
|
+
# preserve_insertion_order=false lets the merges stream instead of buffering rows to preserve order.
|
|
131
|
+
conn.execute("SET preserve_insertion_order = false")
|
|
113
132
|
try:
|
|
114
133
|
if any(parquet_dir.glob("temp_*_users_*.parquet")):
|
|
115
134
|
conn.execute(f"INSERT OR IGNORE INTO users SELECT uid, username FROM read_parquet('{pattern('users')}')")
|
|
@@ -153,6 +172,8 @@ def merge_parquet_files(conn: duckdb.DuckDBPyConnection, parquet_dir: Path, *, c
|
|
|
153
172
|
"""
|
|
154
173
|
)
|
|
155
174
|
if any(parquet_dir.glob("temp_*_changeset_stats_*.parquet")):
|
|
175
|
+
# The shard stores `tags` as a native LIST<STRUCT> (built in the handler), so ingest is a
|
|
176
|
+
# direct column copy.
|
|
156
177
|
conn.execute(
|
|
157
178
|
f"""
|
|
158
179
|
INSERT OR IGNORE INTO changeset_stats
|
|
@@ -161,14 +182,12 @@ def merge_parquet_files(conn: duckdb.DuckDBPyConnection, parquet_dir: Path, *, c
|
|
|
161
182
|
ways_created, ways_modified, ways_deleted,
|
|
162
183
|
rels_created, rels_modified, rels_deleted,
|
|
163
184
|
poi_created, poi_modified,
|
|
164
|
-
|
|
185
|
+
tags
|
|
165
186
|
FROM read_parquet('{pattern("changeset_stats")}')
|
|
166
187
|
"""
|
|
167
188
|
)
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
conn.execute("ROLLBACK")
|
|
171
|
-
raise
|
|
189
|
+
finally:
|
|
190
|
+
conn.execute("SET preserve_insertion_order = true")
|
|
172
191
|
|
|
173
192
|
if cleanup:
|
|
174
193
|
shutil.rmtree(parquet_dir, ignore_errors=True)
|
|
@@ -2,42 +2,41 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
-
import json
|
|
6
5
|
from typing import Any
|
|
7
6
|
|
|
8
7
|
import duckdb
|
|
9
8
|
|
|
9
|
+
from ..stats import map_changes_sum, sum_cols
|
|
10
|
+
|
|
10
11
|
|
|
11
12
|
def _rows(result) -> list[dict[str, Any]]:
|
|
12
13
|
cols = [d[0] for d in result.description]
|
|
13
14
|
return [dict(zip(cols, r, strict=True)) for r in result.fetchall()]
|
|
14
15
|
|
|
15
16
|
|
|
17
|
+
def _tags_to_nested(tags: list[dict[str, Any]] | None) -> dict[str, dict[str, dict[str, Any]]]:
|
|
18
|
+
"""The native `tags` list (list of {k, v, c, m, len_m}) as the nested {key: {value: {c, m, len}}}
|
|
19
|
+
shape `_accumulate_tags` sums over. len is omitted when absent."""
|
|
20
|
+
out: dict[str, dict[str, dict[str, Any]]] = {}
|
|
21
|
+
for t in tags or []:
|
|
22
|
+
entry: dict[str, Any] = {"c": t["c"], "m": t["m"]}
|
|
23
|
+
if t["len_m"] is not None:
|
|
24
|
+
entry["len"] = t["len_m"]
|
|
25
|
+
out.setdefault(t["k"], {})[t["v"]] = entry
|
|
26
|
+
return out
|
|
27
|
+
|
|
28
|
+
|
|
16
29
|
def user_stats(conn: duckdb.DuckDBPyConnection, top_n: int | None = None) -> list[dict[str, Any]]:
|
|
17
|
-
"""One row per user, ranked by total map changes."""
|
|
30
|
+
"""One row per user, ranked by total map changes. Counts come from the shared stats vocabulary."""
|
|
18
31
|
rows = _rows(
|
|
19
32
|
conn.execute(
|
|
20
|
-
"""
|
|
33
|
+
f"""
|
|
21
34
|
SELECT
|
|
22
35
|
u.uid,
|
|
23
36
|
u.username AS name,
|
|
24
37
|
COUNT(DISTINCT cs.changeset_id) AS changesets,
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
SUM(cs.nodes_deleted) AS nodes_delete,
|
|
28
|
-
SUM(cs.ways_created) AS ways_create,
|
|
29
|
-
SUM(cs.ways_modified) AS ways_modify,
|
|
30
|
-
SUM(cs.ways_deleted) AS ways_delete,
|
|
31
|
-
SUM(cs.rels_created) AS rels_create,
|
|
32
|
-
SUM(cs.rels_modified) AS rels_modify,
|
|
33
|
-
SUM(cs.rels_deleted) AS rels_delete,
|
|
34
|
-
SUM(cs.poi_created) AS poi_create,
|
|
35
|
-
SUM(cs.poi_modified) AS poi_modify,
|
|
36
|
-
SUM(
|
|
37
|
-
cs.nodes_created + cs.nodes_modified + cs.nodes_deleted +
|
|
38
|
-
cs.ways_created + cs.ways_modified + cs.ways_deleted +
|
|
39
|
-
cs.rels_created + cs.rels_modified + cs.rels_deleted
|
|
40
|
-
) AS map_changes
|
|
38
|
+
{sum_cols("cs")},
|
|
39
|
+
{map_changes_sum("cs")}
|
|
41
40
|
FROM users u
|
|
42
41
|
JOIN changeset_stats cs ON u.uid = cs.uid
|
|
43
42
|
GROUP BY u.uid, u.username
|
|
@@ -121,7 +120,7 @@ def attach_tag_stats(
|
|
|
121
120
|
tag_mode: str = "none",
|
|
122
121
|
length_tags: list[str] | None = None,
|
|
123
122
|
) -> None:
|
|
124
|
-
"""In-place:
|
|
123
|
+
"""In-place: read the native `tags` list column once per row, then aggregate per user."""
|
|
125
124
|
if not rows:
|
|
126
125
|
return
|
|
127
126
|
if not (additional_tags or tag_mode != "none" or length_tags):
|
|
@@ -138,18 +137,14 @@ def attach_tag_stats(
|
|
|
138
137
|
for k in length_tags or []:
|
|
139
138
|
r.setdefault(f"{k}_len_m", 0)
|
|
140
139
|
|
|
141
|
-
for uid,
|
|
142
|
-
"SELECT uid,
|
|
140
|
+
for uid, tags in conn.execute(
|
|
141
|
+
"SELECT uid, tags FROM changeset_stats WHERE tags IS NOT NULL AND len(tags) > 0"
|
|
143
142
|
).fetchall():
|
|
144
|
-
if uid not in by_uid or not
|
|
145
|
-
continue
|
|
146
|
-
try:
|
|
147
|
-
payload = json.loads(tag_stats_json) if isinstance(tag_stats_json, str) else tag_stats_json
|
|
148
|
-
except (json.JSONDecodeError, TypeError):
|
|
143
|
+
if uid not in by_uid or not tags:
|
|
149
144
|
continue
|
|
150
145
|
_accumulate_tags(
|
|
151
146
|
by_uid[uid],
|
|
152
|
-
|
|
147
|
+
_tags_to_nested(tags),
|
|
153
148
|
additional_tags=additional_tags,
|
|
154
149
|
tag_mode=tag_mode,
|
|
155
150
|
length_tags=length_tags,
|
|
@@ -171,27 +166,13 @@ def daily_summary(
|
|
|
171
166
|
"""One row per UTC day. Requires `changesets` populated (--changeset / --hashtags)."""
|
|
172
167
|
rows = _rows(
|
|
173
168
|
conn.execute(
|
|
174
|
-
"""
|
|
169
|
+
f"""
|
|
175
170
|
SELECT
|
|
176
171
|
CAST(DATE_TRUNC('day', cs.created_at) AS DATE)::VARCHAR AS date,
|
|
177
172
|
COUNT(DISTINCT cs.changeset_id) AS changesets,
|
|
178
173
|
COUNT(DISTINCT cs.uid) AS users,
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
SUM(st.nodes_deleted) AS nodes_delete,
|
|
182
|
-
SUM(st.ways_created) AS ways_create,
|
|
183
|
-
SUM(st.ways_modified) AS ways_modify,
|
|
184
|
-
SUM(st.ways_deleted) AS ways_delete,
|
|
185
|
-
SUM(st.rels_created) AS rels_create,
|
|
186
|
-
SUM(st.rels_modified) AS rels_modify,
|
|
187
|
-
SUM(st.rels_deleted) AS rels_delete,
|
|
188
|
-
SUM(st.poi_created) AS poi_create,
|
|
189
|
-
SUM(st.poi_modified) AS poi_modify,
|
|
190
|
-
SUM(
|
|
191
|
-
st.nodes_created + st.nodes_modified + st.nodes_deleted +
|
|
192
|
-
st.ways_created + st.ways_modified + st.ways_deleted +
|
|
193
|
-
st.rels_created + st.rels_modified + st.rels_deleted
|
|
194
|
-
) AS map_changes
|
|
174
|
+
{sum_cols("st")},
|
|
175
|
+
{map_changes_sum("st")}
|
|
195
176
|
FROM changesets cs
|
|
196
177
|
JOIN changeset_stats st ON cs.changeset_id = st.changeset_id
|
|
197
178
|
GROUP BY DATE_TRUNC('day', cs.created_at)
|
|
@@ -227,22 +208,18 @@ def daily_summary(
|
|
|
227
208
|
for k in length_tags or []:
|
|
228
209
|
r.setdefault(f"{k}_len_m", 0)
|
|
229
210
|
|
|
230
|
-
for date,
|
|
211
|
+
for date, tags in conn.execute(
|
|
231
212
|
"""
|
|
232
|
-
SELECT CAST(DATE_TRUNC('day', cs.created_at) AS DATE)::VARCHAR, st.
|
|
213
|
+
SELECT CAST(DATE_TRUNC('day', cs.created_at) AS DATE)::VARCHAR, st.tags
|
|
233
214
|
FROM changesets cs JOIN changeset_stats st ON cs.changeset_id = st.changeset_id
|
|
234
|
-
WHERE st.
|
|
215
|
+
WHERE st.tags IS NOT NULL AND len(st.tags) > 0
|
|
235
216
|
"""
|
|
236
217
|
).fetchall():
|
|
237
|
-
if date not in by_date or not
|
|
238
|
-
continue
|
|
239
|
-
try:
|
|
240
|
-
payload = json.loads(tag_stats_json) if isinstance(tag_stats_json, str) else tag_stats_json
|
|
241
|
-
except (json.JSONDecodeError, TypeError):
|
|
218
|
+
if date not in by_date or not tags:
|
|
242
219
|
continue
|
|
243
220
|
_accumulate_tags(
|
|
244
221
|
by_date[date],
|
|
245
|
-
|
|
222
|
+
_tags_to_nested(tags),
|
|
246
223
|
additional_tags=additional_tags,
|
|
247
224
|
tag_mode=tag_mode,
|
|
248
225
|
length_tags=length_tags,
|
|
@@ -1,14 +1,41 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
3
5
|
from typing import Any
|
|
4
6
|
|
|
5
7
|
import duckdb
|
|
6
8
|
|
|
7
9
|
from .duckdb_schema import DUCKDB_SCHEMA
|
|
8
10
|
|
|
11
|
+
_MEMORY_LIMIT_RE = re.compile(r"^\d+(\.\d+)?\s?(B|KB|MB|GB|TB|KiB|MiB|GiB|TiB)$", re.IGNORECASE)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _apply_runtime_pragmas(conn: duckdb.DuckDBPyConnection) -> None:
|
|
15
|
+
"""Bound DuckDB memory and point spilling at a roomy disk, from operator env.
|
|
16
|
+
|
|
17
|
+
Unset means DuckDB defaults, so library and test behaviour is unchanged. On a
|
|
18
|
+
memory-capped host these keep a large merge/aggregation spilling to disk instead
|
|
19
|
+
of aborting with an out-of-memory error.
|
|
20
|
+
"""
|
|
21
|
+
memory_limit = os.environ.get("OSMSG_DUCKDB_MEMORY_LIMIT")
|
|
22
|
+
if memory_limit:
|
|
23
|
+
if not _MEMORY_LIMIT_RE.match(memory_limit):
|
|
24
|
+
raise ValueError(f"OSMSG_DUCKDB_MEMORY_LIMIT must be like '1GB', got {memory_limit!r}")
|
|
25
|
+
conn.execute(f"SET memory_limit='{memory_limit}'")
|
|
26
|
+
threads = os.environ.get("OSMSG_DUCKDB_THREADS")
|
|
27
|
+
if threads:
|
|
28
|
+
conn.execute(f"SET threads={int(threads)}")
|
|
29
|
+
temp_directory = os.environ.get("OSMSG_DUCKDB_TEMP_DIR")
|
|
30
|
+
if temp_directory:
|
|
31
|
+
os.makedirs(temp_directory, exist_ok=True)
|
|
32
|
+
conn.execute(f"SET temp_directory='{temp_directory.replace(chr(39), chr(39) * 2)}'")
|
|
33
|
+
|
|
9
34
|
|
|
10
35
|
def connect(db_path: str) -> duckdb.DuckDBPyConnection:
|
|
11
|
-
|
|
36
|
+
conn = duckdb.connect(db_path)
|
|
37
|
+
_apply_runtime_pragmas(conn)
|
|
38
|
+
return conn
|
|
12
39
|
|
|
13
40
|
|
|
14
41
|
def close(conn: duckdb.DuckDBPyConnection) -> None:
|