tlf-geo 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tlf_geo-0.1.0/PKG-INFO +293 -0
- tlf_geo-0.1.0/README.md +268 -0
- tlf_geo-0.1.0/pyproject.toml +42 -0
- tlf_geo-0.1.0/setup.cfg +4 -0
- tlf_geo-0.1.0/src/tlf_geo/__init__.py +4 -0
- tlf_geo-0.1.0/src/tlf_geo/data/codes.yaml +11254 -0
- tlf_geo-0.1.0/src/tlf_geo/data/places.yaml +4510 -0
- tlf_geo-0.1.0/src/tlf_geo/exceptions.py +66 -0
- tlf_geo-0.1.0/src/tlf_geo/geo_resolver.py +614 -0
- tlf_geo-0.1.0/src/tlf_geo/test_resolver.py +14 -0
- tlf_geo-0.1.0/src/tlf_geo.egg-info/PKG-INFO +293 -0
- tlf_geo-0.1.0/src/tlf_geo.egg-info/SOURCES.txt +14 -0
- tlf_geo-0.1.0/src/tlf_geo.egg-info/dependency_links.txt +1 -0
- tlf_geo-0.1.0/src/tlf_geo.egg-info/requires.txt +8 -0
- tlf_geo-0.1.0/src/tlf_geo.egg-info/top_level.txt +1 -0
- tlf_geo-0.1.0/tests/test_geo_resolver.py +342 -0
tlf_geo-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,293 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tlf-geo
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Fuzzy-matched Nepal administrative boundary resolver for TLF
|
|
5
|
+
Author-email: Pujan Pandey <your.email@example.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/PujanPandey07/tlf-monorepo
|
|
8
|
+
Project-URL: Issues, https://github.com/PujanPandey07/tlf-monorepo/issues
|
|
9
|
+
Keywords: nepal,civic-data,geography,administrative-boundaries,fuzzy-matching
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
Requires-Dist: pandas
|
|
19
|
+
Requires-Dist: pyyaml
|
|
20
|
+
Requires-Dist: rapidfuzz
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest; extra == "dev"
|
|
23
|
+
Requires-Dist: build; extra == "dev"
|
|
24
|
+
Requires-Dist: twine; extra == "dev"
|
|
25
|
+
|
|
26
|
+
# tlf-geo
|
|
27
|
+
|
|
28
|
+
**One official code, however many ways someone spelled the place.**
|
|
29
|
+
|
|
30
|
+
`tlf-geo` resolves messy real-world Nepali place names — provinces,
|
|
31
|
+
districts, and all 753 local government units (gaunpalika, municipality,
|
|
32
|
+
sub-metropolitan city, metropolitan city) — to their official NSO codes,
|
|
33
|
+
handling spelling variants, script mixing (Roman/Devanagari), common
|
|
34
|
+
administrative-suffix noise, and the ~30 place names that legitimately exist
|
|
35
|
+
in more than one district.
|
|
36
|
+
|
|
37
|
+
Part of **TLF (The Living Fact)**, an initiative of Corpola Tech and the
|
|
38
|
+
Open Tech Community for making Nepal's civic and census data interoperable.
|
|
39
|
+
Full project story, data sources, and architecture reasoning: [github.com/PujanPandey07/TLF-The-Living-Fact-](https://github.com/PujanPandey07/TLF-The-Living-Fact-)
|
|
40
|
+
|
|
41
|
+
> **Status:** Alpha (v0.1.0). Built on real crosswalk data from Nepal's
|
|
42
|
+
> local-level code registry; see [Contributing](#contributing) for how to
|
|
43
|
+
> report a missing or incorrect place.
|
|
44
|
+
|
|
45
|
+
---
|
|
46
|
+
|
|
47
|
+
## Installation
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install tlf-geo
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## Quick start
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
from tlf_geo import GeoResolver
|
|
59
|
+
|
|
60
|
+
resolver = GeoResolver()
|
|
61
|
+
|
|
62
|
+
# Resolve a place name, disambiguated by district
|
|
63
|
+
result = resolver.resolve("Kalika", district="Rasuwa")
|
|
64
|
+
# {
|
|
65
|
+
# "code": "32902", "canonical_key": "kalika", "level": "gaunpalika",
|
|
66
|
+
# "name": "Kalika", "name_ne": "कालिका गाउँपालिका",
|
|
67
|
+
# "district": "rasuwa", "district_code": 29,
|
|
68
|
+
# "province_code": 3, "wards": "5",
|
|
69
|
+
# }
|
|
70
|
+
|
|
71
|
+
# Same name without a district — genuinely ambiguous, raises instead of guessing
|
|
72
|
+
resolver.resolve("Kalika")
|
|
73
|
+
# -> AmbiguityError: 3 candidates (Rasuwa/gaunpalika, Kalikot/gaunpalika, Chitawan/municipality)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Resolving a whole DataFrame column at once, without letting one bad row kill
|
|
77
|
+
the batch:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
import pandas as pd
|
|
81
|
+
|
|
82
|
+
df = pd.read_csv("survey.csv")
|
|
83
|
+
resolved = resolver.resolve_df(
|
|
84
|
+
df,
|
|
85
|
+
name_col="place_name",
|
|
86
|
+
district_col="district_name",
|
|
87
|
+
)
|
|
88
|
+
# adds `geo_status` (resolved / ambiguous / not_found / invalid_district /
|
|
89
|
+
# level_mismatch) and `geo_error_reason` columns, plus resolved code/name/
|
|
90
|
+
# level/etc. columns for rows that succeeded
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## Why resolution can fail — and why that's on purpose
|
|
96
|
+
|
|
97
|
+
`resolve()` never guesses. If a name doesn't map cleanly to exactly one
|
|
98
|
+
place, it raises one of three exceptions instead of silently picking a
|
|
99
|
+
"probably right" answer:
|
|
100
|
+
|
|
101
|
+
| Exception | Raised when | Carries |
|
|
102
|
+
| ----------------- | ---------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- |
|
|
103
|
+
| `NotFoundError` | No candidate matches at all, or a given `district`/`district_code` rules out every candidate for that name | A message describing what was searched |
|
|
104
|
+
| `AmbiguityError` | More than one real place matches (same name, multiple districts, no district given to disambiguate) | Full candidate dicts, one per real place |
|
|
105
|
+
| `FuzzyMatchError` | Nothing matched exactly or after suffix-stripping, so the name fell through to fuzzy matching | Top 3 `(matched_key, score)` suggestions, no minimum-similarity cutoff — you decide what counts as close enough |
|
|
106
|
+
|
|
107
|
+
This mirrors `tlf-core`'s "never guess, never drop" rule: unresolved data
|
|
108
|
+
surfaces with enough information to resolve it by hand, rather than being
|
|
109
|
+
silently miscategorized.
|
|
110
|
+
|
|
111
|
+
---
|
|
112
|
+
|
|
113
|
+
## API Reference
|
|
114
|
+
|
|
115
|
+
Everything below is available from `from tlf_geo import GeoResolver` (the
|
|
116
|
+
exceptions are importable from `tlf_geo.exceptions` or the top-level
|
|
117
|
+
package).
|
|
118
|
+
|
|
119
|
+
### `GeoResolver()`
|
|
120
|
+
|
|
121
|
+
Loads the bundled `places.yaml` (alias/fuzzy data) and `codes.yaml`
|
|
122
|
+
(official codes + district disambiguation) — no arguments, no external
|
|
123
|
+
files needed.
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
### `resolver.resolve(name, district=None, district_code=None, level=None) -> dict`
|
|
128
|
+
|
|
129
|
+
Resolves one place name to a single official record.
|
|
130
|
+
|
|
131
|
+
- `district` (name string) or `district_code` narrows candidates to that
|
|
132
|
+
district; if both are given, `district_code` takes priority.
|
|
133
|
+
- `level` (`"province"`, `"district"`, `"gaunpalika"`, `"municipality"`,
|
|
134
|
+
`"sub_metropolitan_city"`, `"metropolitan_city"`) narrows candidates to
|
|
135
|
+
that administrative level.
|
|
136
|
+
- Returns a dict: `code, canonical_key, level, name, name_ne, district,
|
|
137
|
+
district_code, province_code, wards`.
|
|
138
|
+
- Raises `NotFoundError`, `AmbiguityError`, or `FuzzyMatchError` — see
|
|
139
|
+
above — if it can't resolve to exactly one place.
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
resolver.resolve("Kalika", district="Rasuwa", level="gaunpalika")
|
|
143
|
+
resolver.resolve("Kalka") # typo -> FuzzyMatchError, suggests kalika/kakani/malika
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
> **Note:** protected areas (national parks, wildlife reserves) are not
|
|
147
|
+
> resolvable by name through `resolve()` — that's a deliberate scope
|
|
148
|
+
> decision, not a bug. Use `protected_areas()` to list them instead.
|
|
149
|
+
|
|
150
|
+
---
|
|
151
|
+
|
|
152
|
+
### `resolver.search(name, limit=5) -> list[tuple[str, float]]`
|
|
153
|
+
|
|
154
|
+
The non-raising sibling of `resolve()`, meant for interactive use —
|
|
155
|
+
autocomplete, a "did you mean...?" prompt, a search box — where you want to
|
|
156
|
+
_see options_, not get exactly one answer or an exception.
|
|
157
|
+
|
|
158
|
+
Runs the same 3-tier lookup as `resolve()`, but always returns a list of
|
|
159
|
+
`(label, score)` pairs instead of raising:
|
|
160
|
+
|
|
161
|
+
- An exact/suffix-stripped match returns `"canonical_key (level)"` labels
|
|
162
|
+
at a score of `100.0` — genuinely distinct places sharing a bare name
|
|
163
|
+
(e.g. Kalika gaunpalika vs. Kalika municipality) show up as separate
|
|
164
|
+
entries rather than being collapsed into one.
|
|
165
|
+
- A fuzzy match returns the same `(matched_key, score)` suggestions
|
|
166
|
+
`FuzzyMatchError` would have carried, just returned instead of raised.
|
|
167
|
+
- No match at all returns `[]`.
|
|
168
|
+
|
|
169
|
+
`search()` never filters by district or level the way `resolve()` does —
|
|
170
|
+
it's purely "what places even loosely match this string."
|
|
171
|
+
|
|
172
|
+
```python
|
|
173
|
+
resolver.search("Kalika")
|
|
174
|
+
# [("kalika (gaunpalika)", 100.0), ("kalika (municipality)", 100.0)]
|
|
175
|
+
|
|
176
|
+
resolver.search("Kalka") # typo
|
|
177
|
+
# [("kalika", 91.0), ("kakani", 73.0), ("malika", 73.0)]
|
|
178
|
+
|
|
179
|
+
resolver.search("zzzznotaplace")
|
|
180
|
+
# []
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
---
|
|
184
|
+
|
|
185
|
+
### `resolver.resolve_df(df, name_col, name_ne_col=None, district_col=None, district_code_col=None, level_col=None) -> DataFrame`
|
|
186
|
+
|
|
187
|
+
Batch version of `resolve()` for a whole DataFrame. Never raises — every
|
|
188
|
+
row's outcome is captured instead of stopping the batch:
|
|
189
|
+
|
|
190
|
+
| `geo_status` | Meaning |
|
|
191
|
+
| ------------------ | ----------------------------------------------------------- |
|
|
192
|
+
| `resolved` | Matched exactly one place; resolved fields filled in |
|
|
193
|
+
| `ambiguous` | Multiple real candidates, no district given to disambiguate |
|
|
194
|
+
| `not_found` | No candidate matched at all |
|
|
195
|
+
| `invalid_district` | Name matched, but not in the given district |
|
|
196
|
+
| `level_mismatch` | Name matched, but not at the given level |
|
|
197
|
+
|
|
198
|
+
Rows that don't resolve get their resolved-field columns left blank and a
|
|
199
|
+
human-readable `geo_error_reason` instead of raising.
|
|
200
|
+
|
|
201
|
+
---
|
|
202
|
+
|
|
203
|
+
### Listing & filtering — return pandas DataFrames by default
|
|
204
|
+
|
|
205
|
+
Every listing method takes an `as_dict=False` param — pass `as_dict=True`
|
|
206
|
+
to get a plain `list[dict]` back instead of a DataFrame, for callers who
|
|
207
|
+
aren't otherwise using pandas (a script, a JSON API response).
|
|
208
|
+
|
|
209
|
+
| Function | Returns |
|
|
210
|
+
| --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- |
|
|
211
|
+
| `resolver.provinces(as_dict=False)` | All 7 provinces |
|
|
212
|
+
| `resolver.districts(province=None, province_code=None, as_dict=False)` | All districts, optionally filtered by province |
|
|
213
|
+
| `resolver.local_levels(level=None, district=None, district_code=None, province_code=None, as_dict=False)` | Local government units, filterable by level, district, and/or province |
|
|
214
|
+
| `resolver.protected_areas(as_dict=False)` | National parks, wildlife reserves, etc. — the listing method, not name-resolvable via `resolve()` |
|
|
215
|
+
|
|
216
|
+
```python
|
|
217
|
+
resolver.provinces(as_dict=True)
|
|
218
|
+
# [{"code": 1, "canonical_key": "koshi", "name": "Koshi", "name_ne": "कोशी"}, ...]
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
---
|
|
222
|
+
|
|
223
|
+
### Code lookups — null-safe, `.apply()`-safe, never raise
|
|
224
|
+
|
|
225
|
+
| Function | Does |
|
|
226
|
+
| ------------------------------ | --------------------------------- |
|
|
227
|
+
| `resolver.get_by_code(code)` | Full record for an official code |
|
|
228
|
+
| `resolver.get_name(code)` | Canonical English name for a code |
|
|
229
|
+
| `resolver.get_canonical(code)` | Canonical key for a code |
|
|
230
|
+
| `resolver.get_wards(code)` | Ward count for a local level |
|
|
231
|
+
| `resolver.get_parent(code)` | Parent district/province code |
|
|
232
|
+
|
|
233
|
+
Each is safe to call directly inside `df["code"].apply(resolver.get_name)` —
|
|
234
|
+
unknown codes return `None` rather than raising.
|
|
235
|
+
|
|
236
|
+
---
|
|
237
|
+
|
|
238
|
+
### Validation
|
|
239
|
+
|
|
240
|
+
- **`resolver.is_valid_code(code) -> bool`**
|
|
241
|
+
- **`resolver.is_valid_name(name) -> bool`**
|
|
242
|
+
|
|
243
|
+
---
|
|
244
|
+
|
|
245
|
+
### Django integration
|
|
246
|
+
|
|
247
|
+
**`resolver.to_choices(level, province_code=None, district_code=None) -> list[tuple[str, str]]`**
|
|
248
|
+
|
|
249
|
+
Returns `(code, name)` pairs ready for a Django `ChoiceField`/model field's
|
|
250
|
+
`choices=`, optionally scoped to a province or district.
|
|
251
|
+
|
|
252
|
+
```python
|
|
253
|
+
# models.py
|
|
254
|
+
district_code = models.CharField(
|
|
255
|
+
max_length=5,
|
|
256
|
+
choices=resolver.to_choices("municipality", province_code=3),
|
|
257
|
+
)
|
|
258
|
+
```
|
|
259
|
+
|
|
260
|
+
---
|
|
261
|
+
|
|
262
|
+
## What this does NOT do
|
|
263
|
+
|
|
264
|
+
- No name-based resolution of protected areas (listing via
|
|
265
|
+
`protected_areas()` works fine — `resolve()` by name doesn't)
|
|
266
|
+
- No fuzzy match cutoff — `FuzzyMatchError`'s top-3 suggestions are always
|
|
267
|
+
returned, however weak, so you judge what's close enough
|
|
268
|
+
- No BS↔AD calendar handling or general value normalization — that's
|
|
269
|
+
`tlf-core`'s job, not this package's
|
|
270
|
+
|
|
271
|
+
---
|
|
272
|
+
|
|
273
|
+
## Contributing
|
|
274
|
+
|
|
275
|
+
`places.yaml` and `codes.yaml` are built from Nepal's official local-level
|
|
276
|
+
code crosswalk. If you find a place that doesn't resolve, resolves
|
|
277
|
+
incorrectly, or is missing an alias:
|
|
278
|
+
|
|
279
|
+
1. [Open an issue](https://github.com/PujanPandey07/TLF-The-Living-Fact-/issues)
|
|
280
|
+
with the raw name, the district it belongs to, and what you expected —
|
|
281
|
+
or
|
|
282
|
+
2. Clone this repo, add the alias/entry directly to `places.yaml` (and
|
|
283
|
+
`codes.yaml` if it's a new place rather than a spelling variant), and
|
|
284
|
+
open a PR.
|
|
285
|
+
|
|
286
|
+
For the project's background, data sources, and architecture decisions, see
|
|
287
|
+
the [main repo](https://github.com/PujanPandey07/TLF-The-Living-Fact-).
|
|
288
|
+
|
|
289
|
+
---
|
|
290
|
+
|
|
291
|
+
## License
|
|
292
|
+
|
|
293
|
+
MIT
|
tlf_geo-0.1.0/README.md
ADDED
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
# tlf-geo
|
|
2
|
+
|
|
3
|
+
**One official code, however many ways someone spelled the place.**
|
|
4
|
+
|
|
5
|
+
`tlf-geo` resolves messy real-world Nepali place names — provinces,
|
|
6
|
+
districts, and all 753 local government units (gaunpalika, municipality,
|
|
7
|
+
sub-metropolitan city, metropolitan city) — to their official NSO codes,
|
|
8
|
+
handling spelling variants, script mixing (Roman/Devanagari), common
|
|
9
|
+
administrative-suffix noise, and the ~30 place names that legitimately exist
|
|
10
|
+
in more than one district.
|
|
11
|
+
|
|
12
|
+
Part of **TLF (The Living Fact)**, an initiative of Corpola Tech and the
|
|
13
|
+
Open Tech Community for making Nepal's civic and census data interoperable.
|
|
14
|
+
Full project story, data sources, and architecture reasoning: [github.com/PujanPandey07/TLF-The-Living-Fact-](https://github.com/PujanPandey07/TLF-The-Living-Fact-)
|
|
15
|
+
|
|
16
|
+
> **Status:** Alpha (v0.1.0). Built on real crosswalk data from Nepal's
|
|
17
|
+
> local-level code registry; see [Contributing](#contributing) for how to
|
|
18
|
+
> report a missing or incorrect place.
|
|
19
|
+
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
## Installation
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install tlf-geo
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
## Quick start
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from tlf_geo import GeoResolver
|
|
34
|
+
|
|
35
|
+
resolver = GeoResolver()
|
|
36
|
+
|
|
37
|
+
# Resolve a place name, disambiguated by district
|
|
38
|
+
result = resolver.resolve("Kalika", district="Rasuwa")
|
|
39
|
+
# {
|
|
40
|
+
# "code": "32902", "canonical_key": "kalika", "level": "gaunpalika",
|
|
41
|
+
# "name": "Kalika", "name_ne": "कालिका गाउँपालिका",
|
|
42
|
+
# "district": "rasuwa", "district_code": 29,
|
|
43
|
+
# "province_code": 3, "wards": "5",
|
|
44
|
+
# }
|
|
45
|
+
|
|
46
|
+
# Same name without a district — genuinely ambiguous, raises instead of guessing
|
|
47
|
+
resolver.resolve("Kalika")
|
|
48
|
+
# -> AmbiguityError: 3 candidates (Rasuwa/gaunpalika, Kalikot/gaunpalika, Chitawan/municipality)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Resolving a whole DataFrame column at once, without letting one bad row kill
|
|
52
|
+
the batch:
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
import pandas as pd
|
|
56
|
+
|
|
57
|
+
df = pd.read_csv("survey.csv")
|
|
58
|
+
resolved = resolver.resolve_df(
|
|
59
|
+
df,
|
|
60
|
+
name_col="place_name",
|
|
61
|
+
district_col="district_name",
|
|
62
|
+
)
|
|
63
|
+
# adds `geo_status` (resolved / ambiguous / not_found / invalid_district /
|
|
64
|
+
# level_mismatch) and `geo_error_reason` columns, plus resolved code/name/
|
|
65
|
+
# level/etc. columns for rows that succeeded
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
---
|
|
69
|
+
|
|
70
|
+
## Why resolution can fail — and why that's on purpose
|
|
71
|
+
|
|
72
|
+
`resolve()` never guesses. If a name doesn't map cleanly to exactly one
|
|
73
|
+
place, it raises one of three exceptions instead of silently picking a
|
|
74
|
+
"probably right" answer:
|
|
75
|
+
|
|
76
|
+
| Exception | Raised when | Carries |
|
|
77
|
+
| ----------------- | ---------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- |
|
|
78
|
+
| `NotFoundError` | No candidate matches at all, or a given `district`/`district_code` rules out every candidate for that name | A message describing what was searched |
|
|
79
|
+
| `AmbiguityError` | More than one real place matches (same name, multiple districts, no district given to disambiguate) | Full candidate dicts, one per real place |
|
|
80
|
+
| `FuzzyMatchError` | Nothing matched exactly or after suffix-stripping, so the name fell through to fuzzy matching | Top 3 `(matched_key, score)` suggestions, no minimum-similarity cutoff — you decide what counts as close enough |
|
|
81
|
+
|
|
82
|
+
This mirrors `tlf-core`'s "never guess, never drop" rule: unresolved data
|
|
83
|
+
surfaces with enough information to resolve it by hand, rather than being
|
|
84
|
+
silently miscategorized.
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## API Reference
|
|
89
|
+
|
|
90
|
+
Everything below is available from `from tlf_geo import GeoResolver` (the
|
|
91
|
+
exceptions are importable from `tlf_geo.exceptions` or the top-level
|
|
92
|
+
package).
|
|
93
|
+
|
|
94
|
+
### `GeoResolver()`
|
|
95
|
+
|
|
96
|
+
Loads the bundled `places.yaml` (alias/fuzzy data) and `codes.yaml`
|
|
97
|
+
(official codes + district disambiguation) — no arguments, no external
|
|
98
|
+
files needed.
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
### `resolver.resolve(name, district=None, district_code=None, level=None) -> dict`
|
|
103
|
+
|
|
104
|
+
Resolves one place name to a single official record.
|
|
105
|
+
|
|
106
|
+
- `district` (name string) or `district_code` narrows candidates to that
|
|
107
|
+
district; if both are given, `district_code` takes priority.
|
|
108
|
+
- `level` (`"province"`, `"district"`, `"gaunpalika"`, `"municipality"`,
|
|
109
|
+
`"sub_metropolitan_city"`, `"metropolitan_city"`) narrows candidates to
|
|
110
|
+
that administrative level.
|
|
111
|
+
- Returns a dict: `code, canonical_key, level, name, name_ne, district,
|
|
112
|
+
district_code, province_code, wards`.
|
|
113
|
+
- Raises `NotFoundError`, `AmbiguityError`, or `FuzzyMatchError` — see
|
|
114
|
+
above — if it can't resolve to exactly one place.
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
resolver.resolve("Kalika", district="Rasuwa", level="gaunpalika")
|
|
118
|
+
resolver.resolve("Kalka") # typo -> FuzzyMatchError, suggests kalika/kakani/malika
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
> **Note:** protected areas (national parks, wildlife reserves) are not
|
|
122
|
+
> resolvable by name through `resolve()` — that's a deliberate scope
|
|
123
|
+
> decision, not a bug. Use `protected_areas()` to list them instead.
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
### `resolver.search(name, limit=5) -> list[tuple[str, float]]`
|
|
128
|
+
|
|
129
|
+
The non-raising sibling of `resolve()`, meant for interactive use —
|
|
130
|
+
autocomplete, a "did you mean...?" prompt, a search box — where you want to
|
|
131
|
+
_see options_, not get exactly one answer or an exception.
|
|
132
|
+
|
|
133
|
+
Runs the same 3-tier lookup as `resolve()`, but always returns a list of
|
|
134
|
+
`(label, score)` pairs instead of raising:
|
|
135
|
+
|
|
136
|
+
- An exact/suffix-stripped match returns `"canonical_key (level)"` labels
|
|
137
|
+
at a score of `100.0` — genuinely distinct places sharing a bare name
|
|
138
|
+
(e.g. Kalika gaunpalika vs. Kalika municipality) show up as separate
|
|
139
|
+
entries rather than being collapsed into one.
|
|
140
|
+
- A fuzzy match returns the same `(matched_key, score)` suggestions
|
|
141
|
+
`FuzzyMatchError` would have carried, just returned instead of raised.
|
|
142
|
+
- No match at all returns `[]`.
|
|
143
|
+
|
|
144
|
+
`search()` never filters by district or level the way `resolve()` does —
|
|
145
|
+
it's purely "what places even loosely match this string."
|
|
146
|
+
|
|
147
|
+
```python
|
|
148
|
+
resolver.search("Kalika")
|
|
149
|
+
# [("kalika (gaunpalika)", 100.0), ("kalika (municipality)", 100.0)]
|
|
150
|
+
|
|
151
|
+
resolver.search("Kalka") # typo
|
|
152
|
+
# [("kalika", 91.0), ("kakani", 73.0), ("malika", 73.0)]
|
|
153
|
+
|
|
154
|
+
resolver.search("zzzznotaplace")
|
|
155
|
+
# []
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
160
|
+
### `resolver.resolve_df(df, name_col, name_ne_col=None, district_col=None, district_code_col=None, level_col=None) -> DataFrame`
|
|
161
|
+
|
|
162
|
+
Batch version of `resolve()` for a whole DataFrame. Never raises — every
|
|
163
|
+
row's outcome is captured instead of stopping the batch:
|
|
164
|
+
|
|
165
|
+
| `geo_status` | Meaning |
|
|
166
|
+
| ------------------ | ----------------------------------------------------------- |
|
|
167
|
+
| `resolved` | Matched exactly one place; resolved fields filled in |
|
|
168
|
+
| `ambiguous` | Multiple real candidates, no district given to disambiguate |
|
|
169
|
+
| `not_found` | No candidate matched at all |
|
|
170
|
+
| `invalid_district` | Name matched, but not in the given district |
|
|
171
|
+
| `level_mismatch` | Name matched, but not at the given level |
|
|
172
|
+
|
|
173
|
+
Rows that don't resolve get their resolved-field columns left blank and a
|
|
174
|
+
human-readable `geo_error_reason` instead of raising.
|
|
175
|
+
|
|
176
|
+
---
|
|
177
|
+
|
|
178
|
+
### Listing & filtering — return pandas DataFrames by default
|
|
179
|
+
|
|
180
|
+
Every listing method takes an `as_dict=False` param — pass `as_dict=True`
|
|
181
|
+
to get a plain `list[dict]` back instead of a DataFrame, for callers who
|
|
182
|
+
aren't otherwise using pandas (a script, a JSON API response).
|
|
183
|
+
|
|
184
|
+
| Function | Returns |
|
|
185
|
+
| --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- |
|
|
186
|
+
| `resolver.provinces(as_dict=False)` | All 7 provinces |
|
|
187
|
+
| `resolver.districts(province=None, province_code=None, as_dict=False)` | All districts, optionally filtered by province |
|
|
188
|
+
| `resolver.local_levels(level=None, district=None, district_code=None, province_code=None, as_dict=False)` | Local government units, filterable by level, district, and/or province |
|
|
189
|
+
| `resolver.protected_areas(as_dict=False)` | National parks, wildlife reserves, etc. — the listing method, not name-resolvable via `resolve()` |
|
|
190
|
+
|
|
191
|
+
```python
|
|
192
|
+
resolver.provinces(as_dict=True)
|
|
193
|
+
# [{"code": 1, "canonical_key": "koshi", "name": "Koshi", "name_ne": "कोशी"}, ...]
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
---
|
|
197
|
+
|
|
198
|
+
### Code lookups — null-safe, `.apply()`-safe, never raise
|
|
199
|
+
|
|
200
|
+
| Function | Does |
|
|
201
|
+
| ------------------------------ | --------------------------------- |
|
|
202
|
+
| `resolver.get_by_code(code)` | Full record for an official code |
|
|
203
|
+
| `resolver.get_name(code)` | Canonical English name for a code |
|
|
204
|
+
| `resolver.get_canonical(code)` | Canonical key for a code |
|
|
205
|
+
| `resolver.get_wards(code)` | Ward count for a local level |
|
|
206
|
+
| `resolver.get_parent(code)` | Parent district/province code |
|
|
207
|
+
|
|
208
|
+
Each is safe to call directly inside `df["code"].apply(resolver.get_name)` —
|
|
209
|
+
unknown codes return `None` rather than raising.
|
|
210
|
+
|
|
211
|
+
---
|
|
212
|
+
|
|
213
|
+
### Validation
|
|
214
|
+
|
|
215
|
+
- **`resolver.is_valid_code(code) -> bool`**
|
|
216
|
+
- **`resolver.is_valid_name(name) -> bool`**
|
|
217
|
+
|
|
218
|
+
---
|
|
219
|
+
|
|
220
|
+
### Django integration
|
|
221
|
+
|
|
222
|
+
**`resolver.to_choices(level, province_code=None, district_code=None) -> list[tuple[str, str]]`**
|
|
223
|
+
|
|
224
|
+
Returns `(code, name)` pairs ready for a Django `ChoiceField`/model field's
|
|
225
|
+
`choices=`, optionally scoped to a province or district.
|
|
226
|
+
|
|
227
|
+
```python
|
|
228
|
+
# models.py
|
|
229
|
+
district_code = models.CharField(
|
|
230
|
+
max_length=5,
|
|
231
|
+
choices=resolver.to_choices("municipality", province_code=3),
|
|
232
|
+
)
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
---
|
|
236
|
+
|
|
237
|
+
## What this does NOT do
|
|
238
|
+
|
|
239
|
+
- No name-based resolution of protected areas (listing via
|
|
240
|
+
`protected_areas()` works fine — `resolve()` by name doesn't)
|
|
241
|
+
- No fuzzy match cutoff — `FuzzyMatchError`'s top-3 suggestions are always
|
|
242
|
+
returned, however weak, so you judge what's close enough
|
|
243
|
+
- No BS↔AD calendar handling or general value normalization — that's
|
|
244
|
+
`tlf-core`'s job, not this package's
|
|
245
|
+
|
|
246
|
+
---
|
|
247
|
+
|
|
248
|
+
## Contributing
|
|
249
|
+
|
|
250
|
+
`places.yaml` and `codes.yaml` are built from Nepal's official local-level
|
|
251
|
+
code crosswalk. If you find a place that doesn't resolve, resolves
|
|
252
|
+
incorrectly, or is missing an alias:
|
|
253
|
+
|
|
254
|
+
1. [Open an issue](https://github.com/PujanPandey07/TLF-The-Living-Fact-/issues)
|
|
255
|
+
with the raw name, the district it belongs to, and what you expected —
|
|
256
|
+
or
|
|
257
|
+
2. Clone this repo, add the alias/entry directly to `places.yaml` (and
|
|
258
|
+
`codes.yaml` if it's a new place rather than a spelling variant), and
|
|
259
|
+
open a PR.
|
|
260
|
+
|
|
261
|
+
For the project's background, data sources, and architecture decisions, see
|
|
262
|
+
the [main repo](https://github.com/PujanPandey07/TLF-The-Living-Fact-).
|
|
263
|
+
|
|
264
|
+
---
|
|
265
|
+
|
|
266
|
+
## License
|
|
267
|
+
|
|
268
|
+
MIT
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "tlf-geo"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Fuzzy-matched Nepal administrative boundary resolver for TLF"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [
|
|
14
|
+
{name = "Pujan Pandey", email = "your.email@example.com"}
|
|
15
|
+
]
|
|
16
|
+
keywords = ["nepal", "civic-data", "geography", "administrative-boundaries", "fuzzy-matching"]
|
|
17
|
+
classifiers = [
|
|
18
|
+
"Development Status :: 3 - Alpha",
|
|
19
|
+
"Intended Audience :: Developers",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
]
|
|
25
|
+
dependencies = [
|
|
26
|
+
"pandas",
|
|
27
|
+
"pyyaml",
|
|
28
|
+
"rapidfuzz",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
dev = ["pytest", "build", "twine"]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://github.com/PujanPandey07/tlf-monorepo"
|
|
36
|
+
Issues = "https://github.com/PujanPandey07/tlf-monorepo/issues"
|
|
37
|
+
|
|
38
|
+
[tool.setuptools.packages.find]
|
|
39
|
+
where = ["src"]
|
|
40
|
+
|
|
41
|
+
[tool.setuptools.package-data]
|
|
42
|
+
tlf_geo = ["data/*.yaml"]
|
tlf_geo-0.1.0/setup.cfg
ADDED