syscli-algeria 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syscli_algeria/__init__.py +199 -0
- syscli_algeria/_normalize.py +69 -0
- syscli_algeria/data/communes.csv +1542 -0
- syscli_algeria/data/wilayas.csv +70 -0
- syscli_algeria/py.typed +0 -0
- syscli_algeria-0.1.0.dist-info/METADATA +58 -0
- syscli_algeria-0.1.0.dist-info/RECORD +9 -0
- syscli_algeria-0.1.0.dist-info/WHEEL +4 -0
- syscli_algeria-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Algeria's 69 wilayas and 1541 communes, from the Journal officiel.
|
|
2
|
+
|
|
3
|
+
The same data and search as the npm package @syscli/algeria.
|
|
4
|
+
|
|
5
|
+
>>> from syscli_algeria import wilaya, commune, search
|
|
6
|
+
>>> wilaya(16).name_fr
|
|
7
|
+
'Alger'
|
|
8
|
+
>>> search("bou saada")[0].wilaya.code
|
|
9
|
+
'68'
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import csv
|
|
15
|
+
import io
|
|
16
|
+
import re
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from importlib import resources
|
|
19
|
+
from typing import Literal, Optional, Union
|
|
20
|
+
|
|
21
|
+
from ._normalize import distance, normalize
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"Commune",
|
|
25
|
+
"SearchResult",
|
|
26
|
+
"Wilaya",
|
|
27
|
+
"WilayaChange",
|
|
28
|
+
"changes",
|
|
29
|
+
"commune",
|
|
30
|
+
"communes",
|
|
31
|
+
"normalize",
|
|
32
|
+
"search",
|
|
33
|
+
"wilaya",
|
|
34
|
+
"wilayas",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
__version__ = "0.1.0"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class Wilaya:
|
|
42
|
+
code: str
|
|
43
|
+
"""Two-digit code, "01" to "69". Also the vehicle plate number."""
|
|
44
|
+
name_fr: str
|
|
45
|
+
name_ar: str
|
|
46
|
+
created_by: str
|
|
47
|
+
"""The law that created the wilaya: "loi 84-09", "loi 19-12" or "loi 26-06"."""
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class Commune:
|
|
52
|
+
code: str
|
|
53
|
+
"""Four-digit ONS code, e.g. "1601"."""
|
|
54
|
+
wilaya_code: str
|
|
55
|
+
name_fr: str
|
|
56
|
+
name_ar: str
|
|
57
|
+
daira_1991: Optional[str]
|
|
58
|
+
"""Seat of the daira the commune belonged to in decret 91-306 (1991)."""
|
|
59
|
+
wilaya_code_58: str
|
|
60
|
+
"""Wilaya under the 58-wilaya map (2019 to 2026)."""
|
|
61
|
+
wilaya_code_48: str
|
|
62
|
+
"""Wilaya under the 48-wilaya map (1984 to 2019)."""
|
|
63
|
+
aliases: tuple[str, ...]
|
|
64
|
+
"""Older official names and other common spellings."""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True)
|
|
68
|
+
class SearchResult:
|
|
69
|
+
commune: Commune
|
|
70
|
+
wilaya: Wilaya
|
|
71
|
+
score: float
|
|
72
|
+
"""0 is an exact match; higher is a weaker match."""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True)
|
|
76
|
+
class WilayaChange:
|
|
77
|
+
commune: Commune
|
|
78
|
+
from_: str
|
|
79
|
+
to: str
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _rows(name: str) -> list[dict[str, str]]:
|
|
83
|
+
text = resources.files(__package__).joinpath("data", name).read_text(encoding="utf-8-sig")
|
|
84
|
+
return list(csv.DictReader(io.StringIO(text)))
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
_WILAYAS: tuple[Wilaya, ...] = tuple(
|
|
88
|
+
Wilaya(r["code"], r["name_fr"], r["name_ar"], r["created_by"]) for r in _rows("wilayas.csv")
|
|
89
|
+
)
|
|
90
|
+
_COMMUNES: tuple[Commune, ...] = tuple(
|
|
91
|
+
Commune(
|
|
92
|
+
code=r["code"],
|
|
93
|
+
wilaya_code=r["wilaya_code"],
|
|
94
|
+
name_fr=r["name_fr"],
|
|
95
|
+
name_ar=r["name_ar"],
|
|
96
|
+
daira_1991=r["daira_1991"] or None,
|
|
97
|
+
wilaya_code_58=r["wilaya_code_58"],
|
|
98
|
+
wilaya_code_48=r["wilaya_code_48"],
|
|
99
|
+
aliases=tuple(r["aliases"].split("|")) if r["aliases"] else (),
|
|
100
|
+
)
|
|
101
|
+
for r in _rows("communes.csv")
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
_BY_WILAYA = {w.code: w for w in _WILAYAS}
|
|
105
|
+
_BY_COMMUNE = {c.code: c for c in _COMMUNES}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _keys(c: Commune) -> list[str]:
|
|
109
|
+
seen: dict[str, None] = {}
|
|
110
|
+
for name in (c.name_fr, c.name_ar, *c.aliases):
|
|
111
|
+
k = normalize(name)
|
|
112
|
+
if k:
|
|
113
|
+
seen[k] = None
|
|
114
|
+
return list(seen)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
_INDEX = [(c, _keys(c)) for c in _COMMUNES]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
_TWO_DIGITS = re.compile("[0-9]{1,2}")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _code(value: Union[str, int], width: int) -> str:
|
|
124
|
+
return str(value).strip().zfill(width)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def wilayas() -> tuple[Wilaya, ...]:
|
|
128
|
+
"""All 69 wilayas, ordered by code."""
|
|
129
|
+
return _WILAYAS
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def wilaya(query: Union[str, int]) -> Optional[Wilaya]:
|
|
133
|
+
"""A wilaya by code (5, "05") or by name in French or Arabic."""
|
|
134
|
+
if isinstance(query, int) or _TWO_DIGITS.fullmatch(query.strip()):
|
|
135
|
+
return _BY_WILAYA.get(_code(query, 2))
|
|
136
|
+
q = normalize(query)
|
|
137
|
+
for w in _WILAYAS:
|
|
138
|
+
if normalize(w.name_fr) == q or normalize(w.name_ar) == q:
|
|
139
|
+
return w
|
|
140
|
+
return None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def communes(wilaya_code: Union[str, int, None] = None) -> tuple[Commune, ...]:
|
|
144
|
+
"""All communes, or the communes of one wilaya."""
|
|
145
|
+
if wilaya_code is None:
|
|
146
|
+
return _COMMUNES
|
|
147
|
+
w = _code(wilaya_code, 2)
|
|
148
|
+
return tuple(c for c in _COMMUNES if c.wilaya_code == w)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def commune(ons_code: Union[str, int]) -> Optional[Commune]:
|
|
152
|
+
"""A commune by its four-digit ONS code."""
|
|
153
|
+
return _BY_COMMUNE.get(_code(ons_code, 4))
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def search(query: str, limit: int = 10, wilaya: Union[str, int, None] = None) -> list[SearchResult]:
|
|
157
|
+
"""Finds communes by name.
|
|
158
|
+
|
|
159
|
+
Accepts French, Arabic, old official names and loose spellings ("setif"
|
|
160
|
+
finds Setif with or without the accent), and tolerates small typos.
|
|
161
|
+
"""
|
|
162
|
+
q = normalize(query)
|
|
163
|
+
if not q:
|
|
164
|
+
return []
|
|
165
|
+
only = None if wilaya is None else _code(wilaya, 2)
|
|
166
|
+
results: list[SearchResult] = []
|
|
167
|
+
for c, keys in _INDEX:
|
|
168
|
+
if only and c.wilaya_code != only:
|
|
169
|
+
continue
|
|
170
|
+
best = float("inf")
|
|
171
|
+
for k in keys:
|
|
172
|
+
if k == q:
|
|
173
|
+
s: float = 0
|
|
174
|
+
elif k.startswith(q):
|
|
175
|
+
s = 1 + (len(k) - len(q)) / 100
|
|
176
|
+
elif q in k:
|
|
177
|
+
s = 2 + (len(k) - len(q)) / 100
|
|
178
|
+
else:
|
|
179
|
+
d = distance(q, k)
|
|
180
|
+
s = 3 + d if len(q) >= 4 and d <= max(1, len(q) // 4) else float("inf")
|
|
181
|
+
if s < best:
|
|
182
|
+
best = s
|
|
183
|
+
if best != float("inf"):
|
|
184
|
+
results.append(SearchResult(c, _BY_WILAYA[c.wilaya_code], best))
|
|
185
|
+
results.sort(key=lambda r: (r.score, r.commune.code))
|
|
186
|
+
return results[:limit]
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def changes(reform: Literal[2019, 2026] = 2026) -> list[WilayaChange]:
|
|
190
|
+
"""Communes whose wilaya changed with a reform.
|
|
191
|
+
|
|
192
|
+
2026: 58 to 69 wilayas (loi 26-06). 2019: 48 to 58 wilayas (loi 19-12).
|
|
193
|
+
"""
|
|
194
|
+
out = []
|
|
195
|
+
for c in _COMMUNES:
|
|
196
|
+
before, after = (c.wilaya_code_58, c.wilaya_code) if reform == 2026 else (c.wilaya_code_48, c.wilaya_code_58)
|
|
197
|
+
if before != after:
|
|
198
|
+
out.append(WilayaChange(c, before, after))
|
|
199
|
+
return out
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Folds the many ways a place name gets written into one comparable key.
|
|
2
|
+
|
|
3
|
+
A port of src/normalize.ts; tests/test_parity.py checks that both give the
|
|
4
|
+
same keys for every commune name.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
import unicodedata
|
|
9
|
+
|
|
10
|
+
_LATIN = [
|
|
11
|
+
(re.compile(r"\b(el|al|les?|la)\b"), " "),
|
|
12
|
+
(re.compile(r"\bbeni\b|\bbni\b"), "bn"),
|
|
13
|
+
(re.compile(r"\bain\b"), "an"),
|
|
14
|
+
(re.compile(r"dj"), "j"),
|
|
15
|
+
(re.compile(r"ou"), "u"),
|
|
16
|
+
(re.compile(r"ch|sh"), "c"),
|
|
17
|
+
(re.compile(r"kh"), "k"),
|
|
18
|
+
(re.compile(r"gh"), "g"),
|
|
19
|
+
(re.compile(r"th"), "t"),
|
|
20
|
+
(re.compile(r"dh"), "d"),
|
|
21
|
+
(re.compile(r"q"), "k"),
|
|
22
|
+
(re.compile(r"y"), "i"),
|
|
23
|
+
(re.compile(r"e"), "a"),
|
|
24
|
+
(re.compile(r"([a-z])\1+"), r"\1"),
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
_ARABIC = [
|
|
28
|
+
(re.compile("[\u064b-\u065f\u0670\u0640]"), ""),
|
|
29
|
+
(re.compile("[\u0622\u0623\u0625\u0671]"), "\u0627"),
|
|
30
|
+
(re.compile("\u0649"), "\u064a"),
|
|
31
|
+
(re.compile("\u0629"), "\u0647"),
|
|
32
|
+
(re.compile("\u0624"), "\u0648"),
|
|
33
|
+
(re.compile("\u0626"), "\u064a"),
|
|
34
|
+
(re.compile("(^|\\s)\u0627\u0644"), "\\1"),
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
_HAS_ARABIC = re.compile("[\u0600-\u06ff]")
|
|
38
|
+
_NOT_ARABIC = re.compile("[^\u0600-\u06ff]")
|
|
39
|
+
_MARKS = re.compile("[\u0300-\u036f]")
|
|
40
|
+
_NOT_LATIN = re.compile("[^a-z]+")
|
|
41
|
+
_SPACES = re.compile(r"\s+")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def normalize(text: str) -> str:
|
|
45
|
+
"""Returns the comparison key of a place name, in French or Arabic."""
|
|
46
|
+
s = unicodedata.normalize("NFKD", text).lower()
|
|
47
|
+
if _HAS_ARABIC.search(s):
|
|
48
|
+
for pattern, repl in _ARABIC:
|
|
49
|
+
s = pattern.sub(repl, s)
|
|
50
|
+
return _NOT_ARABIC.sub("", s)
|
|
51
|
+
s = _NOT_LATIN.sub(" ", _MARKS.sub("", s)).strip()
|
|
52
|
+
for pattern, repl in _LATIN:
|
|
53
|
+
s = pattern.sub(repl, s)
|
|
54
|
+
return _SPACES.sub("", s)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def distance(a: str, b: str) -> int:
|
|
58
|
+
"""Levenshtein distance, or 99 when the lengths differ by more than 3."""
|
|
59
|
+
if abs(len(a) - len(b)) > 3:
|
|
60
|
+
return 99
|
|
61
|
+
row = list(range(len(b) + 1))
|
|
62
|
+
for i in range(1, len(a) + 1):
|
|
63
|
+
prev = row[0]
|
|
64
|
+
row[0] = i
|
|
65
|
+
for j in range(1, len(b) + 1):
|
|
66
|
+
tmp = row[j]
|
|
67
|
+
row[j] = min(row[j] + 1, row[j - 1] + 1, prev + (0 if a[i - 1] == b[j - 1] else 1))
|
|
68
|
+
prev = tmp
|
|
69
|
+
return row[len(b)]
|