@mailwoman/resolver-wof-sqlite 8.0.0 → 8.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (219) hide show
  1. package/address-point-interpolation.ts +9 -3
  2. package/address-point-schema.ts +32 -10
  3. package/address-point.ts +3 -0
  4. package/ancestry-backfill.ts +18 -5
  5. package/ancestry.ts +7 -2
  6. package/build-candidate.ts +29 -6
  7. package/build-slim.ts +40 -11
  8. package/candidate-fts.ts +1 -0
  9. package/candidate-lookup.ts +49 -15
  10. package/candidate-schema.ts +35 -11
  11. package/coincident-roles.ts +28 -6
  12. package/convention.ts +3 -1
  13. package/coverage-manifest-schema.ts +243 -0
  14. package/fst-autocomplete.ts +15 -9
  15. package/fst-builder.ts +65 -9
  16. package/fst-deserialize-web.ts +46 -9
  17. package/fst-matcher.ts +12 -5
  18. package/fst-serialize.ts +83 -9
  19. package/fst-types.ts +43 -0
  20. package/fts-query.ts +84 -0
  21. package/fts.ts +35 -9
  22. package/geo.ts +9 -3
  23. package/geonames-aliases.ts +116 -79
  24. package/geonames-postal.ts +25 -5
  25. package/index.ts +19 -0
  26. package/interpolation.ts +59 -55
  27. package/lookup.ts +103 -292
  28. package/name-score.ts +76 -0
  29. package/out/address-point-interpolation.d.ts.map +1 -1
  30. package/out/address-point-interpolation.js +4 -2
  31. package/out/address-point-interpolation.js.map +1 -1
  32. package/out/address-point-schema.d.ts +30 -10
  33. package/out/address-point-schema.d.ts.map +1 -1
  34. package/out/address-point-schema.js +6 -2
  35. package/out/address-point-schema.js.map +1 -1
  36. package/out/address-point.d.ts.map +1 -1
  37. package/out/address-point.js.map +1 -1
  38. package/out/ancestry-backfill.d.ts +6 -2
  39. package/out/ancestry-backfill.d.ts.map +1 -1
  40. package/out/ancestry-backfill.js +7 -3
  41. package/out/ancestry-backfill.js.map +1 -1
  42. package/out/ancestry.d.ts +6 -2
  43. package/out/ancestry.d.ts.map +1 -1
  44. package/out/ancestry.js +3 -1
  45. package/out/ancestry.js.map +1 -1
  46. package/out/build-candidate.d.ts +9 -3
  47. package/out/build-candidate.d.ts.map +1 -1
  48. package/out/build-candidate.js +5 -3
  49. package/out/build-candidate.js.map +1 -1
  50. package/out/build-slim.d.ts +15 -5
  51. package/out/build-slim.d.ts.map +1 -1
  52. package/out/build-slim.js +11 -5
  53. package/out/build-slim.js.map +1 -1
  54. package/out/candidate-fts.d.ts.map +1 -1
  55. package/out/candidate-fts.js.map +1 -1
  56. package/out/candidate-lookup.d.ts +16 -3
  57. package/out/candidate-lookup.d.ts.map +1 -1
  58. package/out/candidate-lookup.js +27 -11
  59. package/out/candidate-lookup.js.map +1 -1
  60. package/out/candidate-schema.d.ts +33 -11
  61. package/out/candidate-schema.d.ts.map +1 -1
  62. package/out/candidate-schema.js.map +1 -1
  63. package/out/coincident-roles.d.ts +16 -4
  64. package/out/coincident-roles.d.ts.map +1 -1
  65. package/out/coincident-roles.js +9 -3
  66. package/out/coincident-roles.js.map +1 -1
  67. package/out/convention.d.ts +3 -1
  68. package/out/convention.d.ts.map +1 -1
  69. package/out/convention.js.map +1 -1
  70. package/out/coverage-manifest-schema.d.ts +112 -0
  71. package/out/coverage-manifest-schema.d.ts.map +1 -0
  72. package/out/coverage-manifest-schema.js +154 -0
  73. package/out/coverage-manifest-schema.js.map +1 -0
  74. package/out/fst-autocomplete.d.ts +1 -1
  75. package/out/fst-autocomplete.d.ts.map +1 -1
  76. package/out/fst-autocomplete.js +11 -9
  77. package/out/fst-autocomplete.js.map +1 -1
  78. package/out/fst-builder.d.ts.map +1 -1
  79. package/out/fst-builder.js +42 -9
  80. package/out/fst-builder.js.map +1 -1
  81. package/out/fst-deserialize-web.d.ts.map +1 -1
  82. package/out/fst-deserialize-web.js +34 -9
  83. package/out/fst-deserialize-web.js.map +1 -1
  84. package/out/fst-matcher.d.ts +6 -2
  85. package/out/fst-matcher.d.ts.map +1 -1
  86. package/out/fst-matcher.js +9 -5
  87. package/out/fst-matcher.js.map +1 -1
  88. package/out/fst-serialize.d.ts.map +1 -1
  89. package/out/fst-serialize.js +62 -9
  90. package/out/fst-serialize.js.map +1 -1
  91. package/out/fst-types.d.ts +43 -0
  92. package/out/fst-types.d.ts.map +1 -1
  93. package/out/fts-query.d.ts +41 -0
  94. package/out/fts-query.d.ts.map +1 -0
  95. package/out/fts-query.js +75 -0
  96. package/out/fts-query.js.map +1 -0
  97. package/out/fts.d.ts +21 -7
  98. package/out/fts.d.ts.map +1 -1
  99. package/out/fts.js +10 -4
  100. package/out/fts.js.map +1 -1
  101. package/out/geo.d.ts +6 -2
  102. package/out/geo.d.ts.map +1 -1
  103. package/out/geo.js +3 -1
  104. package/out/geo.js.map +1 -1
  105. package/out/geonames-aliases.d.ts +12 -4
  106. package/out/geonames-aliases.d.ts.map +1 -1
  107. package/out/geonames-aliases.js +72 -67
  108. package/out/geonames-aliases.js.map +1 -1
  109. package/out/geonames-postal.d.ts +9 -3
  110. package/out/geonames-postal.d.ts.map +1 -1
  111. package/out/geonames-postal.js +7 -2
  112. package/out/geonames-postal.js.map +1 -1
  113. package/out/index.d.ts +2 -0
  114. package/out/index.d.ts.map +1 -1
  115. package/out/index.js +1 -0
  116. package/out/index.js.map +1 -1
  117. package/out/interpolation.d.ts +24 -6
  118. package/out/interpolation.d.ts.map +1 -1
  119. package/out/interpolation.js +32 -40
  120. package/out/interpolation.js.map +1 -1
  121. package/out/lookup.d.ts +3 -97
  122. package/out/lookup.d.ts.map +1 -1
  123. package/out/lookup.js +52 -184
  124. package/out/lookup.js.map +1 -1
  125. package/out/name-score.d.ts +28 -0
  126. package/out/name-score.d.ts.map +1 -0
  127. package/out/name-score.js +67 -0
  128. package/out/name-score.js.map +1 -0
  129. package/out/poi-lookup.d.ts +24 -8
  130. package/out/poi-lookup.d.ts.map +1 -1
  131. package/out/poi-lookup.js +27 -13
  132. package/out/poi-lookup.js.map +1 -1
  133. package/out/poi-schema.d.ts +42 -13
  134. package/out/poi-schema.d.ts.map +1 -1
  135. package/out/poi-schema.js +12 -3
  136. package/out/poi-schema.js.map +1 -1
  137. package/out/postal-city-alias-lookup.d.ts +18 -6
  138. package/out/postal-city-alias-lookup.d.ts.map +1 -1
  139. package/out/postal-city-alias-lookup.js.map +1 -1
  140. package/out/postal-city-alias-schema.d.ts +27 -9
  141. package/out/postal-city-alias-schema.d.ts.map +1 -1
  142. package/out/postal-city-alias-schema.js +3 -1
  143. package/out/postal-city-alias-schema.js.map +1 -1
  144. package/out/postal-city-candidate-schema.d.ts +15 -5
  145. package/out/postal-city-candidate-schema.d.ts.map +1 -1
  146. package/out/postal-city-candidate-schema.js +3 -1
  147. package/out/postal-city-candidate-schema.js.map +1 -1
  148. package/out/postcode-point-lookup.d.ts +6 -2
  149. package/out/postcode-point-lookup.d.ts.map +1 -1
  150. package/out/postcode-point-lookup.js +6 -2
  151. package/out/postcode-point-lookup.js.map +1 -1
  152. package/out/ranking-weights.d.ts +118 -0
  153. package/out/ranking-weights.d.ts.map +1 -0
  154. package/out/ranking-weights.js +44 -0
  155. package/out/ranking-weights.js.map +1 -0
  156. package/out/reverse.d.ts +9 -3
  157. package/out/reverse.d.ts.map +1 -1
  158. package/out/reverse.js +20 -6
  159. package/out/reverse.js.map +1 -1
  160. package/out/sharding.d.ts +3 -1
  161. package/out/sharding.d.ts.map +1 -1
  162. package/out/sharding.js +7 -5
  163. package/out/sharding.js.map +1 -1
  164. package/out/sqlite-convention-source.d.ts.map +1 -1
  165. package/out/sqlite-convention-source.js +3 -1
  166. package/out/sqlite-convention-source.js.map +1 -1
  167. package/out/street-centroid-schema.d.ts +33 -11
  168. package/out/street-centroid-schema.d.ts.map +1 -1
  169. package/out/street-centroid-schema.js +3 -1
  170. package/out/street-centroid-schema.js.map +1 -1
  171. package/out/street-centroid.d.ts.map +1 -1
  172. package/out/street-centroid.js +6 -2
  173. package/out/street-centroid.js.map +1 -1
  174. package/out/street-morphology-fst-builder.d.ts +6 -2
  175. package/out/street-morphology-fst-builder.d.ts.map +1 -1
  176. package/out/street-morphology-fst-builder.js +8 -7
  177. package/out/street-morphology-fst-builder.js.map +1 -1
  178. package/out/street-morphology-fst-loader.d.ts +67 -0
  179. package/out/street-morphology-fst-loader.d.ts.map +1 -0
  180. package/out/street-morphology-fst-loader.js +59 -0
  181. package/out/street-morphology-fst-loader.js.map +1 -0
  182. package/out/street-name-lookup.d.ts +9 -3
  183. package/out/street-name-lookup.d.ts.map +1 -1
  184. package/out/street-name-lookup.js +9 -7
  185. package/out/street-name-lookup.js.map +1 -1
  186. package/out/street-normalize.d.ts +3 -1
  187. package/out/street-normalize.d.ts.map +1 -1
  188. package/out/street-normalize.js +23 -13
  189. package/out/street-normalize.js.map +1 -1
  190. package/out/street-segment-schema.d.ts +68 -13
  191. package/out/street-segment-schema.d.ts.map +1 -1
  192. package/out/street-segment-schema.js +21 -2
  193. package/out/street-segment-schema.js.map +1 -1
  194. package/out/types.d.ts +18 -6
  195. package/out/types.d.ts.map +1 -1
  196. package/out/unified-schema.d.ts +1 -1
  197. package/out/unified-schema.d.ts.map +1 -1
  198. package/out/unified-schema.js +2 -2
  199. package/out/unified-schema.js.map +1 -1
  200. package/package.json +13 -5
  201. package/poi-lookup.ts +53 -21
  202. package/poi-schema.ts +43 -13
  203. package/postal-city-alias-lookup.ts +20 -6
  204. package/postal-city-alias-schema.ts +28 -9
  205. package/postal-city-candidate-schema.ts +15 -5
  206. package/postcode-point-lookup.ts +6 -2
  207. package/ranking-weights.ts +148 -0
  208. package/reverse.ts +47 -10
  209. package/sharding.ts +13 -6
  210. package/sqlite-convention-source.ts +4 -1
  211. package/street-centroid-schema.ts +35 -11
  212. package/street-centroid.ts +10 -3
  213. package/street-morphology-fst-builder.ts +25 -9
  214. package/street-morphology-fst-loader.ts +103 -0
  215. package/street-name-lookup.ts +19 -7
  216. package/street-normalize.ts +28 -13
  217. package/street-segment-schema.ts +83 -13
  218. package/types.ts +18 -6
  219. package/unified-schema.ts +11 -2
package/fst-types.ts CHANGED
@@ -16,6 +16,13 @@ export interface PlaceEntry {
16
16
  importance: number
17
17
  lat: number
18
18
  lon: number
19
+ /**
20
+ * Surface-ambiguity class (survey #4): how many DISTINCT countries carry a place with THIS entry's accepting surface,
21
+ * counted over the whole admin DB at build time (clamped to 255). A property of the surface, not the place — the same
22
+ * place reached via different alias surfaces reports each surface's own count. `undefined` = built without ambiguity
23
+ * data (pre-2026-07-27 artifacts) — NEVER conflate with 1 (the unambiguous case); the meaning-of-zero rule.
24
+ */
25
+ crossCountryBranches?: number
19
26
  }
20
27
 
21
28
  export type PlacetypeID =
@@ -60,6 +67,14 @@ export interface FSTProvenance {
60
67
  importanceMatches: number
61
68
  sourceDB?: string
62
69
  modelCardVersion?: string
70
+ /**
71
+ * Degenerate-surface curation policy applied at build time (absent = uncurated build).
72
+ */
73
+ exclusionPolicy?: string
74
+ /**
75
+ * Name insertions refused by the curation policy.
76
+ */
77
+ excludedInsertions?: number
63
78
  }
64
79
 
65
80
  export interface BuildFSTOpts {
@@ -67,6 +82,34 @@ export interface BuildFSTOpts {
67
82
  countries?: string[]
68
83
  placetypes?: PlacetypeID[]
69
84
  languages?: string[]
85
+ /**
86
+ * Degenerate-surface curation (build-time; the ASR-contextual-biasing "prune the bias list" discipline). A name whose
87
+ * FULL normalized token sequence joins to a member of this set is never inserted — the surface carries no
88
+ * discriminative value as a bias key (bare function words: "la"; bare street-type words: "boulevard"). The FST is a
89
+ * bias list, not the gazetteer of record — the resolver's candidate tables are untouched, so excluded places stay
90
+ * findable; they just stop nudging the decoder on degenerate keys. Keys must be `normalizeTokens(...).join(" ")`.
91
+ */
92
+ excludeSurfaces?: ReadonlySet<string>
93
+ /**
94
+ * Compositional clause of the same policy: refuse a name whose EVERY normalized token is a member (e.g. "de la") — a
95
+ * surface made entirely of function words cannot be discriminative. Source this from stopwords only, never
96
+ * street-type words ("Avenue Road" is a real name; "de la" is not).
97
+ */
98
+ excludeAllTokensOf?: ReadonlySet<string>
99
+ /**
100
+ * Recorded verbatim into provenance when either exclusion set is supplied.
101
+ */
102
+ exclusionPolicy?: string
103
+ /**
104
+ * Surface-ambiguity classes (survey #4, 2026-07-27): normalized-join surface → the number of DISTINCT countries
105
+ * (across the WHOLE admin DB, not just this build's country scope) with a place carrying that surface. When supplied,
106
+ * every inserted place row records the count for ITS accepting surface (`PlaceEntry.crossCountryBranches`) — an entry
107
+ * accessible under several surfaces records each surface's own count. Serialized into the place row's former `_pad`
108
+ * byte with presence signaled by header flags bit0, so VERSION stays put and pre-ambiguity artifacts read as "no
109
+ * data" (never "0 branches" — the meaning-of-zero rule). No decoder consumes it yet; consumers (FST-prior tempering,
110
+ * the Option-A evidence channel) arrive behind their own measured gates.
111
+ */
112
+ surfaceCountryCounts?: ReadonlyMap<string, number>
70
113
  onProgress?: (phase: string, detail?: string) => void
71
114
  }
72
115
 
package/fts-query.ts ADDED
@@ -0,0 +1,84 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Query shaping for the FTS5 lookup: placetype normalization and the MATCH-expression sanitizer.
7
+ * Both turn a caller's loose input into something SQLite's FTS5 parser accepts without throwing —
8
+ * an unescaped quote or a bare `*` is a syntax error, not an empty result.
9
+ */
10
+
11
+ import type { FindPlaceQuery, WOFPlacetype } from "./types.ts"
12
+
13
+ export function normalizePlacetypes(p: FindPlaceQuery["placetype"]): WOFPlacetype[] | null {
14
+ if (!p) return null
15
+
16
+ return Array.isArray(p) ? p : [p]
17
+ }
18
+
19
+ /**
20
+ * Make an arbitrary user-typed string safe for FTS5 MATCH.
21
+ *
22
+ * FTS5 has its own query syntax (`"phrase"`, `term1 OR term2`, `prefix*`, NEAR/N, etc.). Letting raw user input through
23
+ * means a user typing `Paris's` or `St. (Petersburg)` causes a syntax error.
24
+ *
25
+ * Per-token rules:
26
+ *
27
+ * - Strip all punctuation except trailing `*` from each whitespace-separated token.
28
+ * - **Trailing `*`** is preserved as FTS5 **prefix syntax** — `627*` becomes the literal `627*` (unquoted). The caller
29
+ * signaled they want a prefix; respect that.
30
+ * - All other tokens are wrapped in `"..."` as a single-word phrase. Conservative — handles apostrophes, parens, accented
31
+ * input, etc. safely.
32
+ * - Multiple tokens join with implicit AND.
33
+ *
34
+ * Examples:
35
+ *
36
+ * - `"Paris"` → `"Paris"` (phrase)
37
+ * - `"627*"` → `627*` (prefix)
38
+ * - `"St. (Petersburg)"` → `"St" "Petersburg"` (two phrases, AND-joined)
39
+ * - `"Thiron-Gardais"` → `"Thiron" "Gardais"` (intra-token punctuation SPLITS — #945; fusing to `ThironGardais` matched
40
+ * nothing because the FTS doc tokenizes the hyphenated name as two terms)
41
+ * - `"110 00"` with `fuseTokens` (postcode-typed) → `"110" "00"` per-token fused — the #920 name law
42
+ * - `"Pari* TX"` → `Pari* "TX"` (mixed prefix + phrase)
43
+ * - `"*"` alone → `""` (no body → drop)
44
+ */
45
+ export function sanitizeFTSQuery(text: string, opts?: { fuseTokens?: boolean }): string {
46
+ const out: string[] = []
47
+
48
+ for (const rawToken of text.normalize("NFKC").split(/\s+/u)) {
49
+ const trimmed = rawToken.trim()
50
+
51
+ if (!trimmed) continue
52
+ const hasPrefixStar = trimmed.endsWith("*")
53
+
54
+ // #920 name law (postcode-typed queries ONLY): delete intra-token punctuation and FUSE the
55
+ // remainder — postal names are stored in this collapsed shape ("SW1A" stays one term).
56
+ if (opts?.fuseTokens) {
57
+ const body = trimmed.replaceAll(/[^\p{L}\p{N}]/gu, "")
58
+
59
+ if (!body) continue
60
+ out.push(hasPrefixStar ? `${body}*` : `"${body.replaceAll('"', '""')}"`)
61
+
62
+ continue
63
+ }
64
+
65
+ // Everything else SPLITS on intra-token punctuation — the behavior the docstring always
66
+ // promised ("St. (Petersburg)" → two phrases). The old code DELETED punctuation instead,
67
+ // fusing "Thiron-Gardais" into the unmatchable single term `ThironGardais` while the FTS
68
+ // doc holds two terms (#945 — the entire hyphenated-name class missed at the raw lookup;
69
+ // masked for years because pre-splice tokenizers never emitted hyphen-preserved values).
70
+ const parts = trimmed.split(/[^\p{L}\p{N}]+/u).filter(Boolean)
71
+
72
+ if (!parts.length) continue
73
+
74
+ for (let i = 0; i < parts.length; i++) {
75
+ const body = parts[i]!.replaceAll("*", "")
76
+
77
+ if (!body) continue
78
+ // The caller's trailing `*` applies to the FINAL part ("Thiron-Gard*" → "Thiron" Gard*).
79
+ out.push(hasPrefixStar && i === parts.length - 1 ? `${body}*` : `"${body.replaceAll('"', '""')}"`)
80
+ }
81
+ }
82
+
83
+ return out.join(" ")
84
+ }
package/fts.ts CHANGED
@@ -66,7 +66,9 @@ export const PLACE_SEARCH_TABLE = "place_search"
66
66
  */
67
67
  export const ALIAS_SEPARATOR = "\uE000"
68
68
 
69
- /** `char()` argument for {@link ALIAS_SEPARATOR} in SQL — keeps the SQL text plain ASCII. */
69
+ /**
70
+ * `char()` argument for {@link ALIAS_SEPARATOR} in SQL — keeps the SQL text plain ASCII.
71
+ */
70
72
  const ALIAS_SEPARATOR_CODEPOINT = ALIAS_SEPARATOR.codePointAt(0) as number
71
73
 
72
74
  /**
@@ -93,7 +95,7 @@ const ALIAS_SEPARATOR_CODEPOINT = ALIAS_SEPARATOR.codePointAt(0) as number
93
95
  */
94
96
  export function aliasBagExactMatch(altNames: string | null, normalizedQuery: string, anyStrictExact: boolean): boolean {
95
97
  if (altNames === null || altNames === "" || !normalizedQuery) return false
96
- const norm = (s: string): string => s.toLowerCase().trim().replace(/\s+/g, " ")
98
+ const norm = (s: string): string => s.toLowerCase().trim().replaceAll(/\s+/g, " ")
97
99
 
98
100
  if (altNames.includes(ALIAS_SEPARATOR)) {
99
101
  return altNames.split(ALIAS_SEPARATOR).some((alias) => norm(alias) === normalizedQuery)
@@ -126,15 +128,25 @@ export const PLACE_POPULATION_TABLE = "place_population"
126
128
  * user.
127
129
  */
128
130
  export interface BuildPlaceSearchFTSResult {
129
- /** Whether the FTS5 index was created (true) or already existed and was left alone (false). */
131
+ /**
132
+ * Whether the FTS5 index was created (true) or already existed and was left alone (false).
133
+ */
130
134
  created: boolean
131
- /** Number of rows in the `place_search` table after the call. */
135
+ /**
136
+ * Number of rows in the `place_search` table after the call.
137
+ */
132
138
  indexedRows: number
133
- /** Whether the R*Tree bbox index was created (true) or already existed (false). */
139
+ /**
140
+ * Whether the R*Tree bbox index was created (true) or already existed (false).
141
+ */
134
142
  bboxCreated: boolean
135
- /** Number of rows in the `place_bbox` R*Tree after the call. */
143
+ /**
144
+ * Number of rows in the `place_bbox` R*Tree after the call.
145
+ */
136
146
  bboxIndexedRows: number
137
- /** Wall-clock duration of the build step, in milliseconds. */
147
+ /**
148
+ * Wall-clock duration of the build step, in milliseconds.
149
+ */
138
150
  durationMs: number
139
151
  }
140
152
 
@@ -183,6 +195,7 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
183
195
 
184
196
  if (!ftsExisting || opts.drop) {
185
197
  onProgress("creating")
198
+
186
199
  db.exec(`
187
200
  CREATE VIRTUAL TABLE ${PLACE_SEARCH_TABLE} USING fts5(
188
201
  wof_id UNINDEXED,
@@ -191,7 +204,9 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
191
204
  tokenize = 'unicode61 remove_diacritics 2'
192
205
  );
193
206
  `)
207
+
194
208
  onProgress("populating")
209
+
195
210
  // Excludes only definitively-not-current places. WOF's `is_current` carries TWO conventions:
196
211
  // `-1` (modern Who's On First) and `1` (legacy Mapzen-era), both meaning "currently valid".
197
212
  // Only `0` means "no longer current". Filtering on `= -1` strict (as Phase 4.2 did) excluded
@@ -220,8 +235,10 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
220
235
  AND spr.is_deprecated = 0
221
236
  AND spr.name IS NOT NULL;
222
237
  `)
238
+
223
239
  ftsCreated = true
224
240
  }
241
+
225
242
  const ftsCountRow = db.prepare(`SELECT COUNT(*) AS n FROM ${PLACE_SEARCH_TABLE}`).get() as { n: number }
226
243
 
227
244
  // ─── R*Tree phase ────────────────────────────────────────────────
@@ -234,6 +251,7 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
234
251
 
235
252
  if (!bboxExisting || opts.drop) {
236
253
  onProgress("creating-bbox")
254
+
237
255
  // R*Tree requires INTEGER PRIMARY KEY (id) + paired min/max for each indexed dimension.
238
256
  // `rtree` (not `rtree_i32`) keeps coordinates as REAL — what we want for WGS-84.
239
257
  db.exec(`
@@ -243,7 +261,9 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
243
261
  min_lon, max_lon
244
262
  );
245
263
  `)
264
+
246
265
  onProgress("populating-bbox")
266
+
247
267
  // Only index places that have non-zero coordinates AND a real bbox. WOF stores both the
248
268
  // centroid (latitude/longitude) and the bounding box (min_*/max_*). A subset of rows have
249
269
  // all-zero coordinates — likely placeholders for deprecated / unmapped entries; the
@@ -266,8 +286,10 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
266
286
  AND NOT (spr.min_latitude = 0 AND spr.max_latitude = 0
267
287
  AND spr.min_longitude = 0 AND spr.max_longitude = 0);
268
288
  `)
289
+
269
290
  bboxCreated = true
270
291
  }
292
+
271
293
  const bboxCountRow = db.prepare(`SELECT COUNT(*) AS n FROM ${PLACE_BBOX_TABLE}`).get() as { n: number }
272
294
 
273
295
  // NOTE: `place_population` is NOT built here. `scripts/build-unified-wof.ts` extracts
@@ -299,12 +321,16 @@ export function placeSearchFTSExists(db: DatabaseSync): boolean {
299
321
  return tableExists(db, PLACE_SEARCH_TABLE)
300
322
  }
301
323
 
302
- /** Returns true iff the `place_bbox` R*Tree table exists. Used for opt-in proximity lookup checks. */
324
+ /**
325
+ * Returns true iff the `place_bbox` R*Tree table exists. Used for opt-in proximity lookup checks.
326
+ */
303
327
  export function placeBboxExists(db: DatabaseSync): boolean {
304
328
  return tableExists(db, PLACE_BBOX_TABLE)
305
329
  }
306
330
 
307
- /** Returns true iff the `place_population` table exists. Used for opt-in population-ranking checks. */
331
+ /**
332
+ * Returns true iff the `place_population` table exists. Used for opt-in population-ranking checks.
333
+ */
308
334
  export function placePopulationExists(db: DatabaseSync): boolean {
309
335
  return tableExists(db, PLACE_POPULATION_TABLE)
310
336
  }
package/geo.ts CHANGED
@@ -22,7 +22,9 @@
22
22
  // readers keep importing it from "./geo.ts" (the spatial dep is transitive via @mailwoman/resolver).
23
23
  export { haversineKm } from "@mailwoman/spatial"
24
24
 
25
- /** WGS-84 degrees → radians. */
25
+ /**
26
+ * WGS-84 degrees → radians.
27
+ */
26
28
  function toRad(deg: number): number {
27
29
  return (deg * Math.PI) / 180
28
30
  }
@@ -61,10 +63,14 @@ export function bboxAround(lat: number, lon: number, radiusKm: number): Bbox {
61
63
  */
62
64
  export type GeojsonPosition = [number, number, ...number[]]
63
65
 
64
- /** The two areal GeoJSON geometry types PIP can test, plus an open fallback (Point etc.). */
66
+ /**
67
+ * The two areal GeoJSON geometry types PIP can test, plus an open fallback (Point etc.).
68
+ */
65
69
  export interface GeojsonPolygon {
66
70
  type: "Polygon"
67
- /** `[outerRing, hole1, hole2, …]` — each ring a closed list of positions. */
71
+ /**
72
+ * `[outerRing, hole1, hole2, …]` — each ring a closed list of positions.
73
+ */
68
74
  coordinates: GeojsonPosition[][]
69
75
  }
70
76
 
@@ -36,16 +36,115 @@ import { isOfficialLanguage } from "@mailwoman/codex/country"
36
36
  */
37
37
  export const GEONAMES_ID_BASE = 9_000_000_000_000
38
38
 
39
- /** Per-country progress for the ingest — one event per country dump processed (or skipped). */
39
+ /**
40
+ * Per-country progress for the ingest — one event per country dump processed (or skipped).
41
+ */
40
42
  export interface GeonamesIngestProgress {
41
- /** ISO 3166-1 alpha-2 code. */
43
+ /**
44
+ * ISO 3166-1 alpha-2 code.
45
+ */
42
46
  country: string
43
- /** Populated places ingested from this country's dump (0 when skipped). */
47
+ /**
48
+ * Populated places ingested from this country's dump (0 when skipped).
49
+ */
44
50
  places: number
45
- /** True when the country's `<CC>.txt` dump was missing — the country is skipped, not fatal. */
51
+ /**
52
+ * True when the country's `<CC>.txt` dump was missing — the country is skipped, not fatal.
53
+ */
46
54
  skipped: boolean
47
55
  }
48
56
 
57
+ /**
58
+ * One spelling's language attribution, as reconciled across every alternate-names row that carries it.
59
+ */
60
+ interface V2Alias {
61
+ language: string
62
+ privateuse: string
63
+ official: number
64
+ }
65
+
66
+ /**
67
+ * Parse a country's alternateNamesV2 file into `geonameid -> spelling -> attribution`, restricted to the populated
68
+ * places (`P`) present in `lines`.
69
+ */
70
+ function parseAlternateNamesV2(v2File: string, cc: string, lines: string[]): Map<number, Map<string, V2Alias>> {
71
+ const wanted = new Set<number>()
72
+
73
+ for (const line of lines) {
74
+ const f = line.split("\t")
75
+
76
+ if (f[6] === "P") {
77
+ wanted.add(Number(f[0]))
78
+ }
79
+ }
80
+
81
+ const v2 = new Map<number, Map<string, V2Alias>>()
82
+
83
+ // V2 columns (0-indexed): 1 geonameid, 2 isolanguage, 3 name, 4 isPreferredName, 5 isShortName,
84
+ // 6 isColloquial, 7 isHistoric, 8 from, 9 to.
85
+ //
86
+ // Two passes, because historic-ness is a fact about the NAME, not the row: GeoNames splits one
87
+ // spelling across rows — Malabo carries "Santa Isabel" as (es, unflagged) AND as (no-language,
88
+ // isHistoric=1, to=1973). Officialness must see the flags from EVERY row for the spelling, or the
89
+ // colonial-era name sails through on the language-tagged row (the #936 review's Malabo finding).
90
+ // Do NOT gate on isPreferredName instead — it's sparse annotation, not a signal (Turku's sv "Åbo"
91
+ // is unflagged; FI has 1,746 flags across the whole dump).
92
+ const v2Lines = readFileSync(v2File, "utf8").split("\n")
93
+ const historicNames = new Set<string>()
94
+
95
+ for (const line of v2Lines) {
96
+ if (!line) continue
97
+ const f = line.split("\t")
98
+
99
+ if (f[6] === "1" || f[7] === "1" || (f[9] ?? "").trim() !== "") {
100
+ const alt = (f[3] ?? "").trim()
101
+
102
+ if (alt && wanted.has(Number(f[1]))) {
103
+ historicNames.add(`${f[1]}|${alt}`)
104
+ }
105
+ }
106
+ }
107
+
108
+ for (const line of v2Lines) {
109
+ if (!line) continue
110
+ const f = line.split("\t")
111
+ const gid = Number(f[1])
112
+
113
+ if (!wanted.has(gid)) continue
114
+ const lang = f[2] ?? ""
115
+
116
+ // ISO 639 codes are 2-3 letters; GeoNames' pseudo-codes (post, link, iata, wkdt, …) are 4+.
117
+ if (!/^[a-z]{2,3}$/.test(lang)) continue
118
+ const alt = (f[3] ?? "").trim()
119
+
120
+ if (!alt) continue
121
+ const preferred = f[4] === "1"
122
+ const official = !historicNames.has(`${gid}|${alt}`) && isOfficialLanguage(cc, lang) ? 1 : 0
123
+ let byName = v2.get(gid)
124
+
125
+ if (!byName) {
126
+ v2.set(gid, (byName = new Map()))
127
+ }
128
+
129
+ const prev = byName.get(alt)
130
+
131
+ if (!prev) {
132
+ byName.set(alt, { language: lang, privateuse: preferred ? "preferred" : "", official })
133
+ } else {
134
+ if (official && !prev.official) {
135
+ prev.language = lang
136
+ prev.official = 1
137
+ }
138
+
139
+ if (preferred && !prev.privateuse) {
140
+ prev.privateuse = "preferred"
141
+ }
142
+ }
143
+ }
144
+
145
+ return v2
146
+ }
147
+
49
148
  /**
50
149
  * Fold the GeoNames `P`-class places (+ their Latin alt-names) for `countries` into `db`'s `spr` / `names` /
51
150
  * `place_population` tables. Returns the total places ingested.
@@ -82,18 +181,23 @@ export function ingestGeonamesAliases(
82
181
  // Latin-only, no bracket/paren noise GeoNames packs into `alternatenames` ("(( Karis Landskommun ))",
83
182
  // airport codes), 2–60 chars, at least one letter (drops bare postcodes/numbers).
84
183
  const LATIN_NAME = /^[\p{Script=Latin}\p{M}\s\-'.]{2,60}$/u
184
+
85
185
  const clean = (s: string): string | null => {
86
186
  const t = s.trim()
87
187
 
88
188
  return t && LATIN_NAME.test(t) && /\p{L}/u.test(t) ? t : null
89
189
  }
190
+
90
191
  const sprInsert = db.prepare(
91
192
  `INSERT OR REPLACE INTO spr (id, parent_id, name, placetype, country, latitude, longitude, min_latitude, min_longitude, max_latitude, max_longitude, is_current, is_deprecated, is_ceased, is_superseded, is_superseding, lastmodified) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`
92
193
  )
194
+
93
195
  const namesInsert = db.prepare(
94
196
  `INSERT INTO names (id, name, placetype, country, language, privateuse, official, lastmodified) VALUES (?, ?, ?, ?, ?, ?, ?, ?)`
95
197
  )
198
+
96
199
  const populationInsert = db.prepare(`INSERT OR REPLACE INTO place_population (id, population) VALUES (?, ?)`)
200
+
97
201
  // #267 admin linkage: ancestor rows (locality→region→country) so parentID scoping + adminCoherence reach
98
202
  // the gap countries. Only used for a country in opts.adminForCountries.
99
203
  const ancestorInsert = db.prepare(
@@ -123,8 +227,10 @@ export function ingestGeonamesAliases(
123
227
 
124
228
  if (!existsSync(file)) {
125
229
  report({ country: cc, places: 0, skipped: true }, file)
230
+
126
231
  continue
127
232
  }
233
+
128
234
  let nc = 0
129
235
  // #267: add A-class admin + ancestry only for the gap countries this country is in (never the EU set).
130
236
  const addAdmin = opts?.adminForCountries?.has(cc) ?? false
@@ -136,81 +242,7 @@ export function ingestGeonamesAliases(
136
242
  // dump repeats one spelling under several languages ("Åbo" sv/da/no); the merged tag is official /
137
243
  // preferred if ANY qualifying row is.
138
244
  const v2File = opts?.alternateDir ? join(opts.alternateDir, `${cc}.txt`) : undefined
139
- let v2: Map<number, Map<string, { language: string; privateuse: string; official: number }>> | undefined
140
-
141
- if (v2File && existsSync(v2File)) {
142
- const wanted = new Set<number>()
143
-
144
- for (const line of lines) {
145
- const f = line.split("\t")
146
-
147
- if (f[6] === "P") {
148
- wanted.add(Number(f[0]))
149
- }
150
- }
151
- v2 = new Map()
152
-
153
- // V2 columns (0-indexed): 1 geonameid, 2 isolanguage, 3 name, 4 isPreferredName, 5 isShortName,
154
- // 6 isColloquial, 7 isHistoric, 8 from, 9 to.
155
- //
156
- // Two passes, because historic-ness is a fact about the NAME, not the row: GeoNames splits one
157
- // spelling across rows — Malabo carries "Santa Isabel" as (es, unflagged) AND as (no-language,
158
- // isHistoric=1, to=1973). Officialness must see the flags from EVERY row for the spelling, or the
159
- // colonial-era name sails through on the language-tagged row (the #936 review's Malabo finding).
160
- // Do NOT gate on isPreferredName instead — it's sparse annotation, not a signal (Turku's sv "Åbo"
161
- // is unflagged; FI has 1,746 flags across the whole dump).
162
- const v2Lines = readFileSync(v2File, "utf8").split("\n")
163
- const historicNames = new Set<string>()
164
-
165
- for (const line of v2Lines) {
166
- if (!line) continue
167
- const f = line.split("\t")
168
-
169
- if (f[6] === "1" || f[7] === "1" || (f[9] ?? "").trim() !== "") {
170
- const alt = (f[3] ?? "").trim()
171
-
172
- if (alt && wanted.has(Number(f[1]))) {
173
- historicNames.add(`${f[1]}|${alt}`)
174
- }
175
- }
176
- }
177
-
178
- for (const line of v2Lines) {
179
- if (!line) continue
180
- const f = line.split("\t")
181
- const gid = Number(f[1])
182
-
183
- if (!wanted.has(gid)) continue
184
- const lang = f[2] ?? ""
185
-
186
- // ISO 639 codes are 2-3 letters; GeoNames' pseudo-codes (post, link, iata, wkdt, …) are 4+.
187
- if (!/^[a-z]{2,3}$/.test(lang)) continue
188
- const alt = (f[3] ?? "").trim()
189
-
190
- if (!alt) continue
191
- const preferred = f[4] === "1"
192
- const official = !historicNames.has(`${gid}|${alt}`) && isOfficialLanguage(cc, lang) ? 1 : 0
193
- let byName = v2.get(gid)
194
-
195
- if (!byName) {
196
- v2.set(gid, (byName = new Map()))
197
- }
198
- const prev = byName.get(alt)
199
-
200
- if (!prev) {
201
- byName.set(alt, { language: lang, privateuse: preferred ? "preferred" : "", official })
202
- } else {
203
- if (official && !prev.official) {
204
- prev.language = lang
205
- prev.official = 1
206
- }
207
-
208
- if (preferred && !prev.privateuse) {
209
- prev.privateuse = "preferred"
210
- }
211
- }
212
- }
213
- }
245
+ const v2 = v2File && existsSync(v2File) ? parseAlternateNamesV2(v2File, cc, lines) : undefined
214
246
 
215
247
  // #267 admin pre-pass (gap countries): fold the country (PCLI) + regions (ADM1), self+ancestry them, and
216
248
  // build the admin1→region map the localities link through. Point bbox (GeoNames gives a centroid only).
@@ -287,6 +319,7 @@ export function ingestGeonamesAliases(
287
319
  ancestorInsert.run(nid, countryID, "country")
288
320
  }
289
321
  }
322
+
290
323
  const seen = new Set([name])
291
324
 
292
325
  const tags = v2?.get(Number(f[0]))
@@ -301,16 +334,20 @@ export function ingestGeonamesAliases(
301
334
  namesInsert.run(nid, alt, "locality", cc, tag?.language ?? "", tag?.privateuse ?? "", tag?.official ?? 0, 0)
302
335
  }
303
336
  }
337
+
304
338
  const pop = Number(f[14]) || 0
305
339
 
306
340
  if (pop > 0) {
307
341
  populationInsert.run(nid, pop)
308
342
  }
343
+
309
344
  nc++
310
345
  }
346
+
311
347
  report({ country: cc, places: nc, skipped: false })
312
348
  total += nc
313
349
  }
350
+
314
351
  db.exec("COMMIT")
315
352
 
316
353
  return total
@@ -38,6 +38,12 @@ import { existsSync, readFileSync } from "node:fs"
38
38
  import { join } from "node:path"
39
39
  import type { DatabaseSync } from "node:sqlite"
40
40
 
41
+ /**
42
+ * Column count of a GeoNames postal-code TSV row. Short rows are truncated or blank and are skipped. See the
43
+ * allCountries.zip readme for the field list.
44
+ */
45
+ const GEONAMES_POSTAL_COLUMNS = 11
46
+
41
47
  /**
42
48
  * Synthetic id base for GeoNames-POSTAL rows — its own namespace above the alias fold's {@link GEONAMES_ID_BASE} (9e12)
43
49
  * allocation so all four sources (WOF, Overture, GeoNames-alias, GeoNames-postal) coexist collision-free in a combined
@@ -51,15 +57,21 @@ export const GEONAMES_POSTAL_ID_BASE = 9_500_000_000_000
51
57
  * `"11-041"` → `"11041"`, `"AD500"` → `"AD500"`.
52
58
  */
53
59
  export function normalizePostcodeName(raw: string): string {
54
- return raw.replace(/[^\p{L}\p{N}]/gu, "")
60
+ return raw.replaceAll(/[^\p{L}\p{N}]/gu, "")
55
61
  }
56
62
 
57
63
  export interface GeonamesPostalIngestResult {
58
- /** Distinct postcodes inserted across all countries. */
64
+ /**
65
+ * Distinct postcodes inserted across all countries.
66
+ */
59
67
  inserted: number
60
- /** Per-country distinct-postcode counts. */
68
+ /**
69
+ * Per-country distinct-postcode counts.
70
+ */
61
71
  byCountry: Record<string, number>
62
- /** Countries whose `<CC>.txt` was missing under the postal dir (skipped, reported). */
72
+ /**
73
+ * Countries whose `<CC>.txt` was missing under the postal dir (skipped, reported).
74
+ */
63
75
  missing: string[]
64
76
  }
65
77
 
@@ -77,6 +89,7 @@ export function ingestGeonamesPostal(
77
89
  const sprInsert = db.prepare(
78
90
  `INSERT OR REPLACE INTO spr (id, parent_id, name, placetype, country, latitude, longitude, min_latitude, min_longitude, max_latitude, max_longitude, is_current, is_deprecated, is_ceased, is_superseded, is_superseding, lastmodified) VALUES (?, -1, ?, 'postalcode', ?, ?, ?, ?, ?, ?, ?, 1, 0, 0, 0, 0, 0)`
79
91
  )
92
+
80
93
  const namesInsert = db.prepare(
81
94
  `INSERT INTO names (id, name, placetype, country, language, lastmodified) VALUES (?, ?, 'postalcode', ?, '', 0)`
82
95
  )
@@ -92,18 +105,21 @@ export function ingestGeonamesPostal(
92
105
 
93
106
  if (!existsSync(file)) {
94
107
  missing.push(cc)
108
+
95
109
  console.error(
96
110
  ` GeoNames postal ${cc}: ${file} missing — download from download.geonames.org/export/zip/${cc}.zip; skipped`
97
111
  )
112
+
98
113
  continue
99
114
  }
115
+
100
116
  // Group member settlement points per NORMALIZED code; remember one display form.
101
117
  const members = new Map<string, { display: string; pts: Array<[number, number]> }>()
102
118
 
103
119
  for (const line of readFileSync(file, "utf8").split("\n")) {
104
120
  const cols = line.split("\t")
105
121
 
106
- if (cols.length < 11) continue
122
+ if (cols.length < GEONAMES_POSTAL_COLUMNS) continue
107
123
  const display = cols[1]!.trim()
108
124
  const name = normalizePostcodeName(display)
109
125
  const lat = Number(cols[9])
@@ -132,6 +148,7 @@ export function ingestGeonamesPostal(
132
148
  best = p
133
149
  }
134
150
  }
151
+
135
152
  const id = nextID++
136
153
  sprInsert.run(id, name, cc, best[0], best[1], best[0], best[1], best[0], best[1])
137
154
  namesInsert.run(id, name, cc)
@@ -139,10 +156,13 @@ export function ingestGeonamesPostal(
139
156
  if (m.display !== name) {
140
157
  namesInsert.run(id, m.display, cc)
141
158
  }
159
+
142
160
  inserted++
143
161
  }
162
+
144
163
  db.exec("COMMIT")
145
164
  byCountry[cc] = members.size
165
+
146
166
  console.error(` GeoNames postal ${cc}: ${members.size.toLocaleString()} distinct codes (medoid centroids)`)
147
167
  }
148
168