routemap-engine 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- routemap_engine/__about__.py +12 -0
- routemap_engine/__init__.py +28 -0
- routemap_engine/atlas.py +207 -0
- routemap_engine/cache.py +139 -0
- routemap_engine/cities.py +239 -0
- routemap_engine/data/README.md +93 -0
- routemap_engine/data/cities.tsv +34155 -0
- routemap_engine/data/site_codes.tsv +123 -0
- routemap_engine/geo.py +832 -0
- routemap_engine/hoiho.py +269 -0
- routemap_engine/logsafe.py +43 -0
- routemap_engine/model.py +176 -0
- routemap_engine/netaddr.py +51 -0
- routemap_engine/parse.py +548 -0
- routemap_engine/progressive.py +152 -0
- routemap_engine/route.schema.json +371 -0
- routemap_engine/runner.py +272 -0
- routemap_engine/sitecodes.py +173 -0
- routemap_engine/sitegen.py +237 -0
- routemap_engine/target.py +63 -0
- routemap_engine/whereami.py +79 -0
- routemap_engine-0.2.0.dist-info/METADATA +84 -0
- routemap_engine-0.2.0.dist-info/RECORD +25 -0
- routemap_engine-0.2.0.dist-info/WHEEL +4 -0
- routemap_engine-0.2.0.dist-info/licenses/LICENSE +662 -0
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""The engine's own name and version. The desktop app and FalconEye name themselves."""
|
|
2
|
+
|
|
3
|
+
NAME = "routemap-engine"
|
|
4
|
+
REPO_SLUG = "osintph/routemap-engine"
|
|
5
|
+
REPO_URL = f"https://github.com/{REPO_SLUG}"
|
|
6
|
+
|
|
7
|
+
__version__ = "0.2.0"
|
|
8
|
+
|
|
9
|
+
# Upstreams see this product token unless the caller passes its own User-Agent,
|
|
10
|
+
# so a complaint about traffic reaches the project rather than nobody.
|
|
11
|
+
USER_AGENT_PRODUCT = f"{NAME}/{__version__}"
|
|
12
|
+
USER_AGENT = f"{USER_AGENT_PRODUCT} (+{REPO_URL})"
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The Route Map engine. Pure Python: nothing in this package imports Qt.
|
|
3
|
+
|
|
4
|
+
from routemap_engine import analyse_sync, run_trace, TraceOptions
|
|
5
|
+
|
|
6
|
+
result = run_trace("heise.de", TraceOptions(on_line=print))
|
|
7
|
+
route = analyse_sync(result.text, origin=(14.6, 121.0))
|
|
8
|
+
route.to_dict() # the JSON route model, see route.schema.json
|
|
9
|
+
|
|
10
|
+
FalconEye imports this package for its Route Map tab, so anything added here
|
|
11
|
+
ships to a web server as well as to the desktop app.
|
|
12
|
+
"""
|
|
13
|
+
from routemap_engine.cache import MemoryCache, NullCache, SqliteCache
|
|
14
|
+
from routemap_engine.geo import OFFLINE, Sources, default_sources
|
|
15
|
+
from routemap_engine.model import (Route, analyse, analyse_sync, normalise_origin,
|
|
16
|
+
origin_block, schema)
|
|
17
|
+
from routemap_engine.parse import Hop, ParsedTrace, TraceParseError, parse_trace
|
|
18
|
+
from routemap_engine.runner import (TraceOptions, TraceResult, TraceToolMissing,
|
|
19
|
+
available_tools, install_hint, run_trace)
|
|
20
|
+
from routemap_engine.target import InvalidTarget, validate_target
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"Hop", "InvalidTarget", "MemoryCache", "NullCache", "OFFLINE", "ParsedTrace",
|
|
24
|
+
"Route", "Sources", "SqliteCache", "TraceOptions", "TraceParseError",
|
|
25
|
+
"TraceResult", "TraceToolMissing", "analyse", "analyse_sync",
|
|
26
|
+
"available_tools", "default_sources", "install_hint", "normalise_origin",
|
|
27
|
+
"origin_block", "parse_trace", "run_trace", "schema", "validate_target",
|
|
28
|
+
]
|
routemap_engine/atlas.py
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""
|
|
2
|
+
RIPE Atlas, with the user's own key: one traceroute from a probe near them.
|
|
3
|
+
|
|
4
|
+
The desktop counterpart of FalconEye's Atlas client (app/routemap/atlas.py),
|
|
5
|
+
with the server-only parts left there: no instance credit cap, no shared probe
|
|
6
|
+
cache, no configuration read from the environment. The key is passed in.
|
|
7
|
+
|
|
8
|
+
WHAT IS SENT TO RIPE, AND WHAT IS NOT
|
|
9
|
+
-------------------------------------
|
|
10
|
+
Sent: the target, in the measurement definition, and the probe selection
|
|
11
|
+
criteria (an AS number, or a two-letter country code). The user's coordinates
|
|
12
|
+
are never sent: there is deliberately no ``radius=`` probe filter, and the
|
|
13
|
+
"distance from you" figure is computed here from the probe's own published
|
|
14
|
+
coordinates.
|
|
15
|
+
|
|
16
|
+
A measurement is PUBLIC. RIPE Atlas publishes one-off measurements, including
|
|
17
|
+
the target, the probe and the result. The desktop app asks the user to
|
|
18
|
+
acknowledge that before the first Atlas trace, and nothing here can make it
|
|
19
|
+
untrue.
|
|
20
|
+
|
|
21
|
+
RESULTS GO THROUGH THE PARSER
|
|
22
|
+
-----------------------------
|
|
23
|
+
A result is rendered as Unix traceroute text (:func:`to_trace_text`) and parsed
|
|
24
|
+
like a pasted trace: one parser, one set of fixtures, one place for a format bug.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import asyncio
|
|
29
|
+
import logging
|
|
30
|
+
import time
|
|
31
|
+
from typing import Callable
|
|
32
|
+
|
|
33
|
+
import httpx
|
|
34
|
+
|
|
35
|
+
from routemap_engine import geo
|
|
36
|
+
from routemap_engine.logsafe import tag
|
|
37
|
+
|
|
38
|
+
log = logging.getLogger("routemap_engine.atlas")
|
|
39
|
+
|
|
40
|
+
BASE_URL = "https://atlas.ripe.net/api/v2"
|
|
41
|
+
TRACEROUTE_CREDITS = 30
|
|
42
|
+
DEFAULT_TIMEOUT_SECONDS = 150.0
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AtlasUnavailable(Exception):
|
|
46
|
+
"""Atlas cannot run this trace. ``kind``: auth | credits | noprobe | failed."""
|
|
47
|
+
|
|
48
|
+
def __init__(self, kind: str, message: str):
|
|
49
|
+
super().__init__(message)
|
|
50
|
+
self.kind = kind
|
|
51
|
+
self.message = message
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def to_trace_text(result: dict) -> str:
|
|
55
|
+
"""Render one Atlas traceroute result as Unix traceroute output."""
|
|
56
|
+
target = result.get("dst_name") or result.get("dst_addr") or "target"
|
|
57
|
+
dst = result.get("dst_addr") or ""
|
|
58
|
+
lines = [f"traceroute to {target} ({dst}), 30 hops max, 60 byte packets"]
|
|
59
|
+
for hop in result.get("result") or []:
|
|
60
|
+
number = hop.get("hop")
|
|
61
|
+
if number is None:
|
|
62
|
+
continue
|
|
63
|
+
parts = []
|
|
64
|
+
for probe in hop.get("result") or []:
|
|
65
|
+
if "x" in probe:
|
|
66
|
+
parts.append("*")
|
|
67
|
+
continue
|
|
68
|
+
rtt, addr = probe.get("rtt"), probe.get("from")
|
|
69
|
+
if rtt is None or not addr:
|
|
70
|
+
parts.append("*")
|
|
71
|
+
continue
|
|
72
|
+
name = probe.get("name")
|
|
73
|
+
label = f"{name} ({addr})" if name and name != addr else addr
|
|
74
|
+
parts.append(f"{label} {float(rtt):.3f} ms")
|
|
75
|
+
lines.append(f"{number:>2} " + " ".join(parts) if parts else f"{number:>2} * * *")
|
|
76
|
+
return "\n".join(lines) + "\n"
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Atlas:
|
|
80
|
+
def __init__(self, key: str, *, user_agent: str = geo.DEFAULT_USER_AGENT,
|
|
81
|
+
base_url: str = BASE_URL, description: str = "routemap-engine traceroute"):
|
|
82
|
+
if not key or not key.strip():
|
|
83
|
+
raise AtlasUnavailable("auth", "No RIPE Atlas API key is set. Add one in Settings.")
|
|
84
|
+
self.key = key.strip()
|
|
85
|
+
self.user_agent = user_agent
|
|
86
|
+
self.base_url = base_url.rstrip("/")
|
|
87
|
+
# Published by RIPE with the measurement, so it names the tool, not the user.
|
|
88
|
+
self.description = description
|
|
89
|
+
|
|
90
|
+
def _client(self, authenticated: bool = True) -> httpx.AsyncClient:
|
|
91
|
+
headers = {"User-Agent": self.user_agent, "Accept": "application/json"}
|
|
92
|
+
if authenticated:
|
|
93
|
+
headers["Authorization"] = f"Key {self.key}"
|
|
94
|
+
return httpx.AsyncClient(base_url=self.base_url, headers=headers, timeout=20.0)
|
|
95
|
+
|
|
96
|
+
async def probes(self, params: dict) -> list[dict]:
|
|
97
|
+
# Probe lists are public; the key is not sent where it is not needed.
|
|
98
|
+
async with self._client(authenticated=False) as client:
|
|
99
|
+
response = await client.get("/probes/", params=params)
|
|
100
|
+
if response.status_code != 200:
|
|
101
|
+
return []
|
|
102
|
+
out = []
|
|
103
|
+
for item in (response.json() or {}).get("results") or []:
|
|
104
|
+
coords = (item.get("geometry") or {}).get("coordinates") or []
|
|
105
|
+
lat = lon = None
|
|
106
|
+
if len(coords) == 2:
|
|
107
|
+
try:
|
|
108
|
+
lon, lat = float(coords[0]), float(coords[1])
|
|
109
|
+
except (TypeError, ValueError):
|
|
110
|
+
lat = lon = None
|
|
111
|
+
out.append({"id": item.get("id"), "asn": item.get("asn_v4") or item.get("asn_v6"),
|
|
112
|
+
"country": item.get("country_code"), "lat": lat, "lon": lon})
|
|
113
|
+
return [p for p in out if p.get("id")]
|
|
114
|
+
|
|
115
|
+
async def select_probe(self, asn: int | None, country: str | None,
|
|
116
|
+
origin: tuple[float, float] | None) -> dict:
|
|
117
|
+
"""The user's own network first, their country next; nearest first.
|
|
118
|
+
|
|
119
|
+
*origin* only ranks candidates, here. It is never sent.
|
|
120
|
+
"""
|
|
121
|
+
candidates: list[dict] = []
|
|
122
|
+
if asn:
|
|
123
|
+
candidates = await self.probes({"asn_v4": asn, "status": 1, "page_size": 100})
|
|
124
|
+
if not candidates and country:
|
|
125
|
+
candidates = await self.probes({"country_code": country.upper(), "status": 1,
|
|
126
|
+
"page_size": 100})
|
|
127
|
+
if not candidates:
|
|
128
|
+
raise AtlasUnavailable(
|
|
129
|
+
"noprobe", "No connected RIPE Atlas probe was found on your network or in "
|
|
130
|
+
"your country, so an Atlas trace would not describe your path.")
|
|
131
|
+
if origin is not None:
|
|
132
|
+
for probe in candidates:
|
|
133
|
+
probe["distance_km"] = (None if probe["lat"] is None else round(
|
|
134
|
+
geo.haversine_km(origin[0], origin[1], probe["lat"], probe["lon"]), 1))
|
|
135
|
+
candidates.sort(key=lambda p: (p["distance_km"] is None, p["distance_km"] or 0.0))
|
|
136
|
+
return candidates[0]
|
|
137
|
+
|
|
138
|
+
async def balance(self) -> int | None:
|
|
139
|
+
try:
|
|
140
|
+
async with self._client() as client:
|
|
141
|
+
response = await client.get("/credits/")
|
|
142
|
+
if response.status_code != 200:
|
|
143
|
+
return None
|
|
144
|
+
return int((response.json() or {}).get("current_balance"))
|
|
145
|
+
except Exception: # noqa: BLE001
|
|
146
|
+
return None
|
|
147
|
+
|
|
148
|
+
async def create(self, target: str, probe_id: int, af: int = 4) -> int:
|
|
149
|
+
body = {
|
|
150
|
+
"definitions": [{
|
|
151
|
+
"type": "traceroute", "af": af, "target": target,
|
|
152
|
+
"description": self.description, "protocol": "ICMP",
|
|
153
|
+
"resolve_on_probe": True, "paris": 0, "first_hop": 1, "max_hops": 30,
|
|
154
|
+
"packets": 3,
|
|
155
|
+
}],
|
|
156
|
+
"probes": [{"type": "probes", "value": str(probe_id), "requested": 1}],
|
|
157
|
+
"is_oneoff": True,
|
|
158
|
+
}
|
|
159
|
+
async with self._client() as client:
|
|
160
|
+
response = await client.post("/measurements/", json=body)
|
|
161
|
+
if response.status_code in (200, 201):
|
|
162
|
+
try:
|
|
163
|
+
measurement = int((response.json() or {}).get("measurements")[0])
|
|
164
|
+
except Exception as exc:
|
|
165
|
+
raise AtlasUnavailable("failed", "RIPE Atlas did not return a measurement id") from exc
|
|
166
|
+
log.info("event=atlas_measurement id=%s probe=%s target=%s",
|
|
167
|
+
measurement, probe_id, tag(target))
|
|
168
|
+
return measurement
|
|
169
|
+
detail = ""
|
|
170
|
+
try:
|
|
171
|
+
detail = str((response.json() or {}).get("error") or "")[:200]
|
|
172
|
+
except Exception: # noqa: BLE001
|
|
173
|
+
pass
|
|
174
|
+
lowered = detail.lower()
|
|
175
|
+
if response.status_code in (401, 403) and "credit" not in lowered and "balance" not in lowered:
|
|
176
|
+
raise AtlasUnavailable(
|
|
177
|
+
"auth", "RIPE Atlas refused the key. Check it in Settings: it needs the "
|
|
178
|
+
"\"schedule a new measurement\" permission and must not have expired.")
|
|
179
|
+
if response.status_code in (402, 403):
|
|
180
|
+
raise AtlasUnavailable("credits", "RIPE Atlas refused the measurement; the account "
|
|
181
|
+
"is probably out of credits.")
|
|
182
|
+
raise AtlasUnavailable("failed", f"RIPE Atlas refused the measurement "
|
|
183
|
+
f"(HTTP {response.status_code}). {detail}".strip())
|
|
184
|
+
|
|
185
|
+
async def wait(self, measurement_id: int, timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
|
186
|
+
on_wait: Callable[[float], None] | None = None) -> str:
|
|
187
|
+
"""Poll until the measurement has a result; return it as trace text."""
|
|
188
|
+
deadline = time.monotonic() + timeout
|
|
189
|
+
started = time.monotonic()
|
|
190
|
+
delay = 3.0
|
|
191
|
+
async with self._client(authenticated=False) as client:
|
|
192
|
+
while time.monotonic() < deadline:
|
|
193
|
+
await asyncio.sleep(delay)
|
|
194
|
+
delay = min(delay * 1.4, 10.0)
|
|
195
|
+
if on_wait is not None:
|
|
196
|
+
try:
|
|
197
|
+
on_wait(time.monotonic() - started)
|
|
198
|
+
except Exception: # noqa: BLE001
|
|
199
|
+
pass
|
|
200
|
+
try:
|
|
201
|
+
response = await client.get(f"/measurements/{measurement_id}/results/")
|
|
202
|
+
results = response.json() if response.status_code == 200 else []
|
|
203
|
+
except Exception: # noqa: BLE001
|
|
204
|
+
continue
|
|
205
|
+
if results:
|
|
206
|
+
return to_trace_text(results[0])
|
|
207
|
+
raise AtlasUnavailable("failed", "The Atlas measurement did not return a result in time.")
|
routemap_engine/cache.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Where the engine keeps answers it may reuse.
|
|
3
|
+
|
|
4
|
+
Only one kind of answer is cached: what CAIDA Hoiho said about a router
|
|
5
|
+
hostname. A hostname is cached whether or not it matched, because the misses
|
|
6
|
+
are the common case and re-asking for them on every trace would be most of the
|
|
7
|
+
traffic sent to CAIDA for no new information. Nothing else from a trace is
|
|
8
|
+
stored by the engine: not the text, not the addresses, not the origin.
|
|
9
|
+
|
|
10
|
+
The engine does not decide where the cache lives. A caller passes any object
|
|
11
|
+
with ``get(key)`` and ``set(key, value)``; the TTL is the cache's business.
|
|
12
|
+
Three are provided:
|
|
13
|
+
|
|
14
|
+
NullCache nothing is kept (the default when no cache is passed)
|
|
15
|
+
MemoryCache for one process, with a TTL
|
|
16
|
+
SqliteCache one file on disk, with a TTL, for the desktop app's config dir
|
|
17
|
+
|
|
18
|
+
FalconEye passes its own, backed by the application database, so its existing
|
|
19
|
+
``route_map_hostname_cache`` table keeps working unchanged.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import sqlite3
|
|
25
|
+
import threading
|
|
26
|
+
import time
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Protocol
|
|
29
|
+
|
|
30
|
+
# Thirty days, the window Hoiho answers are trusted for. The ruleset itself is
|
|
31
|
+
# dated (2024-08 at the time of writing) and changes rarely.
|
|
32
|
+
DEFAULT_TTL_SECONDS = 30 * 24 * 3600
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Cache(Protocol):
|
|
36
|
+
def get(self, key: str) -> dict | None: ...
|
|
37
|
+
|
|
38
|
+
def set(self, key: str, value: dict) -> None: ...
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class NullCache:
|
|
42
|
+
"""Remembers nothing."""
|
|
43
|
+
|
|
44
|
+
def get(self, key: str) -> dict | None:
|
|
45
|
+
return None
|
|
46
|
+
|
|
47
|
+
def set(self, key: str, value: dict) -> None:
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class MemoryCache:
|
|
52
|
+
"""A dict with a TTL. Thread-safe, per instance, never global."""
|
|
53
|
+
|
|
54
|
+
def __init__(self, ttl_seconds: float = DEFAULT_TTL_SECONDS):
|
|
55
|
+
self.ttl_seconds = float(ttl_seconds)
|
|
56
|
+
self._lock = threading.Lock()
|
|
57
|
+
self._data: dict[str, tuple[float, str]] = {}
|
|
58
|
+
|
|
59
|
+
def get(self, key: str) -> dict | None:
|
|
60
|
+
with self._lock:
|
|
61
|
+
hit = self._data.get(key)
|
|
62
|
+
if hit is None:
|
|
63
|
+
return None
|
|
64
|
+
stored_at, blob = hit
|
|
65
|
+
if time.time() - stored_at > self.ttl_seconds:
|
|
66
|
+
del self._data[key]
|
|
67
|
+
return None
|
|
68
|
+
return json.loads(blob)
|
|
69
|
+
|
|
70
|
+
def set(self, key: str, value: dict) -> None:
|
|
71
|
+
# Stored serialised, so a caller mutating what it got back cannot
|
|
72
|
+
# change what the next caller gets.
|
|
73
|
+
blob = json.dumps(value)
|
|
74
|
+
with self._lock:
|
|
75
|
+
self._data[key] = (time.time(), blob)
|
|
76
|
+
|
|
77
|
+
def clear(self) -> None:
|
|
78
|
+
with self._lock:
|
|
79
|
+
self._data.clear()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class SqliteCache:
|
|
83
|
+
"""One SQLite file, one table, a TTL enforced on read.
|
|
84
|
+
|
|
85
|
+
A connection per call: the desktop app reads this from a worker thread and
|
|
86
|
+
the CLI from the main one, and SQLite connections are not shareable across
|
|
87
|
+
threads by default.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
TABLE = "hostname_cache"
|
|
91
|
+
|
|
92
|
+
def __init__(self, path: str | Path, ttl_seconds: float = DEFAULT_TTL_SECONDS):
|
|
93
|
+
self.path = Path(path)
|
|
94
|
+
self.ttl_seconds = float(ttl_seconds)
|
|
95
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
96
|
+
with self._connect() as conn:
|
|
97
|
+
conn.execute(
|
|
98
|
+
f"CREATE TABLE IF NOT EXISTS {self.TABLE} ("
|
|
99
|
+
"key TEXT PRIMARY KEY, value TEXT NOT NULL, stored_at REAL NOT NULL)")
|
|
100
|
+
|
|
101
|
+
def _connect(self) -> sqlite3.Connection:
|
|
102
|
+
return sqlite3.connect(self.path, timeout=5.0)
|
|
103
|
+
|
|
104
|
+
def get(self, key: str) -> dict | None:
|
|
105
|
+
try:
|
|
106
|
+
with self._connect() as conn:
|
|
107
|
+
row = conn.execute(
|
|
108
|
+
f"SELECT value, stored_at FROM {self.TABLE} WHERE key = ?",
|
|
109
|
+
(key,)).fetchone()
|
|
110
|
+
except sqlite3.Error:
|
|
111
|
+
return None
|
|
112
|
+
if row is None or time.time() - row[1] > self.ttl_seconds:
|
|
113
|
+
return None
|
|
114
|
+
try:
|
|
115
|
+
return json.loads(row[0])
|
|
116
|
+
except ValueError:
|
|
117
|
+
return None
|
|
118
|
+
|
|
119
|
+
def set(self, key: str, value: dict) -> None:
|
|
120
|
+
try:
|
|
121
|
+
with self._connect() as conn:
|
|
122
|
+
conn.execute(
|
|
123
|
+
f"INSERT OR REPLACE INTO {self.TABLE} (key, value, stored_at) "
|
|
124
|
+
"VALUES (?, ?, ?)", (key, json.dumps(value), time.time()))
|
|
125
|
+
except sqlite3.Error:
|
|
126
|
+
# A cache that cannot be written is a cache miss next time, not a
|
|
127
|
+
# failed trace.
|
|
128
|
+
return None
|
|
129
|
+
|
|
130
|
+
def clear(self) -> int:
|
|
131
|
+
"""Delete every entry. Returns how many there were."""
|
|
132
|
+
with self._connect() as conn:
|
|
133
|
+
count = conn.execute(f"SELECT COUNT(*) FROM {self.TABLE}").fetchone()[0]
|
|
134
|
+
conn.execute(f"DELETE FROM {self.TABLE}")
|
|
135
|
+
return int(count)
|
|
136
|
+
|
|
137
|
+
def __len__(self) -> int:
|
|
138
|
+
with self._connect() as conn:
|
|
139
|
+
return int(conn.execute(f"SELECT COUNT(*) FROM {self.TABLE}").fetchone()[0])
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The bundled offline city list, for the Route Map origin picker.
|
|
3
|
+
|
|
4
|
+
WHAT IT IS AND WHERE IT CAME FROM
|
|
5
|
+
---------------------------------
|
|
6
|
+
``routemap_engine/data/cities.tsv`` is GeoNames' ``cities15000`` dump (every
|
|
7
|
+
populated place above 15,000 people, 34,152 rows), trimmed to the seven columns
|
|
8
|
+
this feature uses and with the coordinates rounded to two decimal places.
|
|
9
|
+
|
|
10
|
+
Licence, checked against https://download.geonames.org/export/dump/readme.txt
|
|
11
|
+
on 2026-10-03, which states: "This work is licensed under a Creative Commons
|
|
12
|
+
Attribution 4.0 License, see https://creativecommons.org/licenses/by/4.0/".
|
|
13
|
+
CC BY 4.0 permits redistribution and modification, including commercially, on
|
|
14
|
+
condition of attribution. The attribution is carried in three places so it
|
|
15
|
+
cannot be lost by touching only one of them: the header of the data file, the
|
|
16
|
+
data-sources strip in the page footer, and the Acknowledgments section of the
|
|
17
|
+
README.
|
|
18
|
+
|
|
19
|
+
WHY IT IS BUNDLED AND NOT A SERVICE
|
|
20
|
+
-----------------------------------
|
|
21
|
+
The thing being resolved is the visitor's own location. Typing "Manila" into a
|
|
22
|
+
geocoding API to find out where Manila is would mean telling a third party where
|
|
23
|
+
the visitor is, to answer a question that has a fixed answer. A file in the
|
|
24
|
+
repository answers it with no request leaving the box.
|
|
25
|
+
|
|
26
|
+
MEMORY, AND WHY THE LOAD IS LAZY
|
|
27
|
+
--------------------------------
|
|
28
|
+
Parsed, the table is roughly 9 MB of Python objects per worker, which on a
|
|
29
|
+
1 GB box with three workers is not free. It is therefore loaded on the first
|
|
30
|
+
search and not at import: a worker that never serves the origin picker never
|
|
31
|
+
pays for it, which on an instance where nobody opens the Route Map tab is every
|
|
32
|
+
worker. :func:`loaded` exists so a test can assert that property rather than
|
|
33
|
+
trust this comment.
|
|
34
|
+
"""
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import pathlib
|
|
38
|
+
import re
|
|
39
|
+
import threading
|
|
40
|
+
import unicodedata
|
|
41
|
+
|
|
42
|
+
DATA_FILE = pathlib.Path(__file__).resolve().parent / "data" / "cities.tsv"
|
|
43
|
+
|
|
44
|
+
ATTRIBUTION = ("City list from GeoNames (cities15000), used under "
|
|
45
|
+
"Creative Commons Attribution 4.0.")
|
|
46
|
+
ATTRIBUTION_URL = "https://www.geonames.org/"
|
|
47
|
+
LICENCE_URL = "https://creativecommons.org/licenses/by/4.0/"
|
|
48
|
+
|
|
49
|
+
# How many rows one search may return. The picker shows a short list; a query
|
|
50
|
+
# like "san" matches hundreds and nobody scrolls them.
|
|
51
|
+
MAX_RESULTS = 12
|
|
52
|
+
|
|
53
|
+
_lock = threading.Lock()
|
|
54
|
+
_rows: list[tuple] | None = None
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _fold(text: str) -> str:
|
|
58
|
+
"""Match key: accents stripped, lowercased, non-alphanumerics removed.
|
|
59
|
+
|
|
60
|
+
So "Sao Paulo" finds "Sao Paulo", and "ho chi minh" finds
|
|
61
|
+
"Ho Chi Minh City" through the prefix rule below.
|
|
62
|
+
"""
|
|
63
|
+
decomposed = unicodedata.normalize("NFKD", text or "")
|
|
64
|
+
stripped = "".join(ch for ch in decomposed if not unicodedata.combining(ch))
|
|
65
|
+
return re.sub(r"[^a-z0-9]+", "", stripped.lower())
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _load() -> list[tuple]:
|
|
69
|
+
"""Parse the TSV into (key, name, cc, admin1, lat, lon, population).
|
|
70
|
+
|
|
71
|
+
The file is sorted by population descending when it is generated, so the
|
|
72
|
+
search can keep first-seen order and get "most likely city first" without
|
|
73
|
+
sorting 34,000 rows per query.
|
|
74
|
+
"""
|
|
75
|
+
rows: list[tuple] = []
|
|
76
|
+
with DATA_FILE.open(encoding="utf-8") as handle:
|
|
77
|
+
for line in handle:
|
|
78
|
+
if line.startswith("#") or not line.strip():
|
|
79
|
+
continue
|
|
80
|
+
parts = line.rstrip("\n").split("\t")
|
|
81
|
+
if len(parts) != 7:
|
|
82
|
+
continue
|
|
83
|
+
name, ascii_name, cc, admin1, lat, lon, population = parts
|
|
84
|
+
try:
|
|
85
|
+
lat_f, lon_f, pop_i = float(lat), float(lon), int(population or 0)
|
|
86
|
+
except ValueError:
|
|
87
|
+
continue
|
|
88
|
+
key = _fold(name)
|
|
89
|
+
rows.append((key, name, cc, admin1, lat_f, lon_f, pop_i))
|
|
90
|
+
# A separate key for the ASCII spelling, where GeoNames gives one
|
|
91
|
+
# that differs, so a keyboard without diacritics still finds the
|
|
92
|
+
# city ("sao paulo" -> "Sao Paulo" -> "Sao Paulo"). Note this is
|
|
93
|
+
# transliteration only: the dump's alternate-names column, which is
|
|
94
|
+
# where an exonym like "Kiev" for "Kyiv" lives, is not bundled.
|
|
95
|
+
ascii_key = _fold(ascii_name) if ascii_name else ""
|
|
96
|
+
if ascii_key and ascii_key != key:
|
|
97
|
+
rows.append((ascii_key, name, cc, admin1, lat_f, lon_f, pop_i))
|
|
98
|
+
return rows
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _table() -> list[tuple]:
|
|
102
|
+
global _rows
|
|
103
|
+
if _rows is None:
|
|
104
|
+
with _lock:
|
|
105
|
+
if _rows is None:
|
|
106
|
+
_rows = _load()
|
|
107
|
+
return _rows
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def loaded() -> bool:
|
|
111
|
+
"""Whether the table is resident in this process yet."""
|
|
112
|
+
return _rows is not None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def display(name: str, cc: str, admin1: str) -> str:
|
|
116
|
+
"""How a city is labelled in the UI: "San Jose, CA, US", "Manila, PH"."""
|
|
117
|
+
parts = [name]
|
|
118
|
+
if admin1:
|
|
119
|
+
parts.append(admin1)
|
|
120
|
+
if cc:
|
|
121
|
+
parts.append(cc)
|
|
122
|
+
return ", ".join(parts)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _as_dict(row: tuple) -> dict:
|
|
126
|
+
_key, name, cc, admin1, lat, lon, population = row
|
|
127
|
+
return {"name": name, "cc": cc, "admin1": admin1, "lat": lat, "lon": lon,
|
|
128
|
+
"population": population, "display": display(name, cc, admin1)}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def search(query: str, limit: int = MAX_RESULTS) -> list[dict]:
|
|
132
|
+
"""Cities matching *query*, best first, de-duplicated by display name.
|
|
133
|
+
|
|
134
|
+
Ranked exact match, then prefix, then substring. Within a rank the file's
|
|
135
|
+
own population order decides, so "san jose" offers the Californian one
|
|
136
|
+
before the three in the Philippines.
|
|
137
|
+
"""
|
|
138
|
+
key = _fold(query)
|
|
139
|
+
if len(key) < 2:
|
|
140
|
+
return []
|
|
141
|
+
limit = max(1, min(int(limit or MAX_RESULTS), MAX_RESULTS))
|
|
142
|
+
|
|
143
|
+
buckets: tuple[list[dict], list[dict], list[dict]] = ([], [], [])
|
|
144
|
+
seen: set[str] = set()
|
|
145
|
+
for row in _table():
|
|
146
|
+
row_key = row[0]
|
|
147
|
+
if key == row_key:
|
|
148
|
+
rank = 0
|
|
149
|
+
elif row_key.startswith(key):
|
|
150
|
+
rank = 1
|
|
151
|
+
elif key in row_key:
|
|
152
|
+
rank = 2
|
|
153
|
+
else:
|
|
154
|
+
continue
|
|
155
|
+
entry = _as_dict(row)
|
|
156
|
+
if entry["display"] in seen:
|
|
157
|
+
continue
|
|
158
|
+
seen.add(entry["display"])
|
|
159
|
+
buckets[rank].append(entry)
|
|
160
|
+
# Enough exact matches to fill the answer means nothing weaker can
|
|
161
|
+
# displace them, so the scan can stop.
|
|
162
|
+
if rank == 0 and len(buckets[0]) >= limit:
|
|
163
|
+
break
|
|
164
|
+
|
|
165
|
+
return (buckets[0] + buckets[1] + buckets[2])[:limit]
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def lookup(name: str) -> dict | None:
|
|
169
|
+
"""The single best city for a typed name, or None.
|
|
170
|
+
|
|
171
|
+
Used when the origin arrives as a city string (the MCP tool, a scripted
|
|
172
|
+
call) rather than through the picker, which sends coordinates.
|
|
173
|
+
"""
|
|
174
|
+
results = search(name, limit=1)
|
|
175
|
+
return results[0] if results else None
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def nearest(lat: float, lon: float, max_km: float = 250.0) -> dict | None:
|
|
179
|
+
"""The bundled city that best labels (lat, lon), or None if nothing is near.
|
|
180
|
+
|
|
181
|
+
This is what turns a coordinate into the label the UI shows ("Manila, PH"
|
|
182
|
+
rather than "14.6, 121.0"). Coordinates stay as the secondary text, because
|
|
183
|
+
the city is a convenience and the coordinates are the actual origin.
|
|
184
|
+
|
|
185
|
+
Not simply the closest city. GeoNames' cities15000 contains city districts
|
|
186
|
+
as rows of their own, so the nearest row to central Manila is Paco and the
|
|
187
|
+
nearest to central Hannover is Nordstadt: both correct, both useless as a
|
|
188
|
+
label for where somebody is. The pick is therefore the most *significant*
|
|
189
|
+
nearby place, scoring population against distance:
|
|
190
|
+
|
|
191
|
+
score = population / (1 + distance_km) ** 1.5
|
|
192
|
+
|
|
193
|
+
which puts Manila (1.6M, 2 km) ahead of both Paco next door and Quezon City
|
|
194
|
+
(2.9M, 8 km), and is the answer a person would give.
|
|
195
|
+
|
|
196
|
+
``max_km`` exists so a point in the middle of an ocean is reported honestly
|
|
197
|
+
as coordinates rather than attached to the nearest populated place 900 km
|
|
198
|
+
away. A full scan of 34,000 rows is a few milliseconds and happens once per
|
|
199
|
+
origin, so there is no index to keep correct.
|
|
200
|
+
"""
|
|
201
|
+
import math
|
|
202
|
+
|
|
203
|
+
try:
|
|
204
|
+
lat, lon = float(lat), float(lon)
|
|
205
|
+
except (TypeError, ValueError):
|
|
206
|
+
return None
|
|
207
|
+
if not (-90.0 <= lat <= 90.0 and -180.0 <= lon <= 180.0):
|
|
208
|
+
return None
|
|
209
|
+
|
|
210
|
+
phi1 = math.radians(lat)
|
|
211
|
+
cos_phi1 = math.cos(phi1)
|
|
212
|
+
best = best_nearest = None
|
|
213
|
+
best_score, best_km = -1.0, float("inf")
|
|
214
|
+
|
|
215
|
+
for row in _table():
|
|
216
|
+
_key, name, cc, admin1, row_lat, row_lon, population = row
|
|
217
|
+
# Equirectangular approximation: accurate well inside the distances
|
|
218
|
+
# that matter here, and avoids 34,000 haversines per call.
|
|
219
|
+
dx = math.radians(row_lon - lon) * cos_phi1
|
|
220
|
+
dy = math.radians(row_lat - lat)
|
|
221
|
+
km = 6371.0088 * math.hypot(dx, dy)
|
|
222
|
+
if km > max_km:
|
|
223
|
+
continue
|
|
224
|
+
if km < best_km:
|
|
225
|
+
best_km, best_nearest = km, (name, cc, admin1, row_lat, row_lon, km)
|
|
226
|
+
score = population / (1.0 + km) ** 1.5
|
|
227
|
+
if score > best_score:
|
|
228
|
+
best_score, best = score, (name, cc, admin1, row_lat, row_lon, km)
|
|
229
|
+
|
|
230
|
+
# Every candidate had population 0, so significance cannot decide; fall
|
|
231
|
+
# back to plain proximity.
|
|
232
|
+
chosen = best if best_score > 0 else best_nearest
|
|
233
|
+
if chosen is None:
|
|
234
|
+
return None
|
|
235
|
+
name, cc, admin1, row_lat, row_lon, km = chosen
|
|
236
|
+
return {"name": name, "cc": cc, "admin1": admin1,
|
|
237
|
+
"lat": row_lat, "lon": row_lon,
|
|
238
|
+
"display": display(name, cc, admin1),
|
|
239
|
+
"distance_km": round(km, 1)}
|