@mailwoman/resolver-wof-sqlite 7.2.0 → 7.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/address-point-interpolation.ts +207 -0
  2. package/address-point-schema.ts +107 -0
  3. package/address-point.ts +122 -0
  4. package/ancestry-backfill.ts +205 -0
  5. package/ancestry.ts +70 -0
  6. package/build-candidate.ts +351 -0
  7. package/build-slim.ts +394 -0
  8. package/candidate-fts.ts +43 -0
  9. package/candidate-lookup.ts +382 -0
  10. package/candidate-schema.ts +166 -0
  11. package/coincident-roles.ts +240 -0
  12. package/convention.ts +152 -0
  13. package/fst-autocomplete.ts +187 -0
  14. package/fst-builder.ts +291 -0
  15. package/fst-deserialize-web.ts +164 -0
  16. package/fst-matcher.ts +150 -0
  17. package/fst-serialize.ts +311 -0
  18. package/fst-types.ts +78 -0
  19. package/fts.ts +318 -0
  20. package/geo.ts +140 -0
  21. package/geonames-aliases.ts +317 -0
  22. package/geonames-postal.ts +150 -0
  23. package/index.ts +117 -0
  24. package/interpolation.ts +232 -0
  25. package/lookup.ts +1498 -0
  26. package/out/poi-lookup.d.ts +14 -2
  27. package/out/poi-lookup.d.ts.map +1 -1
  28. package/out/poi-lookup.js +55 -21
  29. package/out/poi-lookup.js.map +1 -1
  30. package/out/poi-schema.d.ts +9 -0
  31. package/out/poi-schema.d.ts.map +1 -1
  32. package/out/poi-schema.js +16 -0
  33. package/out/poi-schema.js.map +1 -1
  34. package/out/reverse.d.ts +8 -1
  35. package/out/reverse.d.ts.map +1 -1
  36. package/out/reverse.js +10 -1
  37. package/out/reverse.js.map +1 -1
  38. package/package.json +168 -82
  39. package/poi-lookup.ts +375 -0
  40. package/poi-schema.ts +164 -0
  41. package/postal-city-alias-lookup.ts +89 -0
  42. package/postal-city-alias-schema.ts +75 -0
  43. package/postal-city-candidate-schema.ts +81 -0
  44. package/postcode-point-lookup.ts +64 -0
  45. package/reverse.ts +439 -0
  46. package/schema.ts +176 -0
  47. package/sharding.ts +235 -0
  48. package/sqlite-convention-source.ts +61 -0
  49. package/sqlite-utils.ts +25 -0
  50. package/street-centroid-schema.ts +124 -0
  51. package/street-centroid.ts +124 -0
  52. package/street-morphology-fst-builder.ts +230 -0
  53. package/street-name-lookup.ts +101 -0
  54. package/street-normalize.ts +302 -0
  55. package/street-segment-schema.ts +104 -0
  56. package/types.ts +164 -0
  57. package/unified-schema.ts +171 -0
@@ -0,0 +1,302 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * THE street normalizer for the address-point tier (#476). One function, used by BOTH the shard
7
+ * builder (`scripts/build-address-point-shard.ts`) and the lookup tier (`address-point.ts`) —
8
+ * never two implementations (the PLACETYPE_ORDER lesson: parallel copies silently corrupt).
9
+ *
10
+ * Normalization contract (deliberately aggressive — both sides apply the same function, so
11
+ * collisions only need to be _consistent_, not linguistically perfect):
12
+ *
13
+ * 1. Lowercase, NFKD-fold diacritics, collapse whitespace, strip punctuation (periods, commas,
14
+ * apostrophes).
15
+ * 2. Expand USPS directional abbreviations at the FIRST and LAST token position (`n` → `north`, `se` →
16
+ * `southeast`) — Overture sources abbreviate inconsistently.
17
+ * 3. Canonicalize a trailing USPS street-type token via the codex suffix table to its canonical full
18
+ * form (`st`/`str`/`street` → `street`).
19
+ *
20
+ * Numbered streets are left as digits (`5th` stays `5th`); a SPELLED ordinal before a street suffix
21
+ * folds to its digit form (`tenth street` → `10th street`, #723) so the grid-city ordinal
22
+ * cross-streets the source data spells with digits become reachable.
23
+ */
24
+
25
+ import { AbbreviationToDirectional, US_STREET_SUFFIX_LOOKUP } from "@mailwoman/codex/us"
26
+
27
+ /**
28
+ * Spelled ordinal street names → their digit-ordinal form ("tenth" → "10th"), applied ONLY when a street-type suffix
29
+ * follows (#723 admin-tail) — so the ordinal cross-streets common in grid cities ("Tenth Street", "Fifth Avenue") match
30
+ * the shards' digit keys, WITHOUT rewriting ordinal-WORD names where the next token is not a suffix ("First National
31
+ * Bank Rd" stays "first national …"). Digit-source shards are unaffected (a digit token isn't in this map), so the
32
+ * existing keys need no rebuild; a future rebuild folds any spelled-source key the same way (the one-function
33
+ * discipline).
34
+ */
35
+ const SPELLED_ORDINAL_TO_DIGIT = new Map<string, string>([
36
+ ["first", "1st"],
37
+ ["second", "2nd"],
38
+ ["third", "3rd"],
39
+ ["fourth", "4th"],
40
+ ["fifth", "5th"],
41
+ ["sixth", "6th"],
42
+ ["seventh", "7th"],
43
+ ["eighth", "8th"],
44
+ ["ninth", "9th"],
45
+ ["tenth", "10th"],
46
+ ["eleventh", "11th"],
47
+ ["twelfth", "12th"],
48
+ ["thirteenth", "13th"],
49
+ ["fourteenth", "14th"],
50
+ ["fifteenth", "15th"],
51
+ ["sixteenth", "16th"],
52
+ ["seventeenth", "17th"],
53
+ ["eighteenth", "18th"],
54
+ ["nineteenth", "19th"],
55
+ ["twentieth", "20th"],
56
+ ["thirtieth", "30th"],
57
+ ["fortieth", "40th"],
58
+ ["fiftieth", "50th"],
59
+ ["sixtieth", "60th"],
60
+ ["seventieth", "70th"],
61
+ ["eightieth", "80th"],
62
+ ["ninetieth", "90th"],
63
+ ["hundredth", "100th"],
64
+ ])
65
+
66
+ /** Lowercase + diacritic-fold + punctuation strip + whitespace collapse. */
67
+ function fold(input: string): string {
68
+ return input
69
+ .normalize("NFKD")
70
+ .replace(/[̀-ͯ]/g, "")
71
+ .toLowerCase()
72
+ .replace(/[.,'’]/g, "")
73
+ .replace(/\s+/g, " ")
74
+ .trim()
75
+ }
76
+
77
+ /**
78
+ * Normalize a street name for address-point keying. Same function at build time and lookup time — see module docstring
79
+ * for the contract.
80
+ */
81
+ export function normalizeStreetForKey(street: string): string {
82
+ const tokens = fold(street).split(" ")
83
+
84
+ if (tokens.length === 0) return ""
85
+
86
+ // Spelled-ordinal street names → digit form when a street suffix follows ("Tenth Street" →
87
+ // "10th street", #723). Gated on the next token being a suffix so ordinal-WORD names are untouched.
88
+ for (let i = 0; i < tokens.length - 1; i++) {
89
+ const digit = SPELLED_ORDINAL_TO_DIGIT.get(tokens[i]!)
90
+
91
+ if (digit && US_STREET_SUFFIX_LOOKUP.has(tokens[i + 1]!)) {
92
+ tokens[i] = digit
93
+ }
94
+ }
95
+
96
+ // Directional expansion at the edges only ("N Main St" / "Main St N" — never interior
97
+ // tokens, where "W" may be an initial in a person-named street). The codex expands
98
+ // compounds to two words ("SE" → "SOUTH EAST"); we key on the spaceless form
99
+ // ("southeast"), and also merge an already-written two-token pair ("South East …").
100
+ const edgeDirectional = (raw: string) =>
101
+ AbbreviationToDirectional.get(raw.toUpperCase())?.toLowerCase().replace(" ", "")
102
+ const mergePair = (a?: string, b?: string) =>
103
+ a && b && /^(north|south)$/.test(a) && /^(east|west)$/.test(b) ? a + b : undefined
104
+
105
+ const leadPair = mergePair(tokens[0], tokens[1])
106
+
107
+ if (leadPair && tokens.length > 2) {
108
+ tokens.splice(0, 2, leadPair)
109
+ }
110
+ const first = edgeDirectional(tokens[0]!)
111
+
112
+ if (first && tokens.length > 1) {
113
+ tokens[0] = first
114
+ }
115
+
116
+ const tailPair = mergePair(tokens[tokens.length - 2], tokens[tokens.length - 1])
117
+
118
+ if (tailPair && tokens.length > 3) {
119
+ tokens.splice(tokens.length - 2, 2, tailPair)
120
+ }
121
+
122
+ if (tokens.length > 2) {
123
+ const last = edgeDirectional(tokens[tokens.length - 1]!)
124
+
125
+ if (last) {
126
+ tokens[tokens.length - 1] = last
127
+ }
128
+ }
129
+
130
+ // Street-type canonicalization via the codex table (lowercase keys, UPPER canonical
131
+ // values). The suffix is usually the last token, but sits second-to-last when a trailing
132
+ // directional follows ("Main St N") — check both positions, canonicalize the first hit.
133
+ for (const at of [tokens.length - 1, tokens.length - 2]) {
134
+ if (at < 1) continue // never canonicalize the only/first token ("Street Road" exists)
135
+ const canonical = US_STREET_SUFFIX_LOOKUP.get(tokens[at]!)
136
+
137
+ if (canonical) {
138
+ tokens[at] = canonical.toLowerCase()
139
+ break
140
+ }
141
+ }
142
+
143
+ return tokens.join(" ")
144
+ }
145
+
146
+ /**
147
+ * Street-name locale for the address-point key. The US path is the full USPS pipeline ({@link normalizeStreetForKey});
148
+ * the international paths fold + apply a SMALL, consistent per-locale type-token canonicalization. Same discipline as
149
+ * the US normalizer: build side and probe side call the identical function, so the key only needs to be CONSISTENT, not
150
+ * linguistically perfect — a folded "rue du chevaleret" keys the same on both sides whether or not we reorder the
151
+ * article, so no salient-token / multi-key index is built yet (deferred until probing shows the normalizer can't absorb
152
+ * the false-negatives).
153
+ */
154
+ export type StreetLocale = "us" | "fr" | "de" | "nl"
155
+
156
+ /**
157
+ * French street-type abbreviations → canonical full form, applied per token after {@link fold}. French address types
158
+ * LEAD the name ("Av. de…", "Bd …", "Pl. …") and "St"/"Ste" abbreviate Saint/Sainte inside names ("Rue St-Honoré" →
159
+ * "rue saint honore"). fold() has already stripped the trailing period, so the keys are point-free ("av", "bd").
160
+ */
161
+ const FR_STREET_ABBREV = new Map<string, string>([
162
+ ["av", "avenue"],
163
+ ["ave", "avenue"],
164
+ ["bd", "boulevard"],
165
+ ["bld", "boulevard"],
166
+ ["bvd", "boulevard"],
167
+ ["boul", "boulevard"],
168
+ ["pl", "place"],
169
+ ["imp", "impasse"],
170
+ ["all", "allee"],
171
+ ["ch", "chemin"],
172
+ ["che", "chemin"],
173
+ ["sq", "square"],
174
+ ["pas", "passage"],
175
+ ["fg", "faubourg"],
176
+ ["fbg", "faubourg"],
177
+ ["rte", "route"],
178
+ ["st", "saint"],
179
+ ["ste", "sainte"],
180
+ ["sts", "saints"],
181
+ ])
182
+
183
+ /**
184
+ * Normalize a street name for the address-point key in a non-US locale. Same function build-side and probe-side (the
185
+ * one-function discipline). US delegates to {@link normalizeStreetForKey}.
186
+ *
187
+ * - **fr** — fold + expand leading type abbreviations and Saint/Sainte (token map).
188
+ * - **de** — fold + ß→ss + canonicalize the GLUED `-str(.)` suffix to `-strasse` ("Lindenstr." → "lindenstrasse",
189
+ * "Lindenstraße" → "lindenstrasse"); an already-full "-strasse" is left intact.
190
+ * - **nl** — fold + canonicalize the glued `-str` suffix to `-straat` ("Kerkstr." → "kerkstraat").
191
+ */
192
+ export function normalizeStreetForKeyLocale(street: string, locale: StreetLocale): string {
193
+ if (locale === "us") return normalizeStreetForKey(street)
194
+
195
+ // Hyphen → space so a compound name keys the same whether the source or the query writes the
196
+ // hyphen ("Champs-Élysées", "St-Honoré") or a space — both sides fold identically, so this is pure
197
+ // robustness. It also splits a hyphenated abbreviation ("St-Honoré" → "st honore") into tokens the
198
+ // per-locale type/Saint map can see.
199
+ const tokens = fold(street).replace(/ß/g, "ss").replace(/-/g, " ").split(/\s+/).filter(Boolean)
200
+
201
+ if (tokens.length === 0) return ""
202
+
203
+ switch (locale) {
204
+ case "fr":
205
+ for (let i = 0; i < tokens.length; i++) {
206
+ tokens[i] = FR_STREET_ABBREV.get(tokens[i]!) ?? tokens[i]!
207
+ }
208
+ break
209
+ case "de":
210
+ for (let i = 0; i < tokens.length; i++) {
211
+ const t = tokens[i]!
212
+
213
+ if (t.endsWith("str") && !t.endsWith("strasse")) {
214
+ tokens[i] = t.replace(/str$/, "strasse")
215
+ }
216
+ }
217
+ break
218
+ case "nl":
219
+ for (let i = 0; i < tokens.length; i++) {
220
+ const t = tokens[i]!
221
+
222
+ if (t.endsWith("str") && !t.endsWith("straat")) {
223
+ tokens[i] = t.replace(/str$/, "straat")
224
+ }
225
+ }
226
+ break
227
+ }
228
+
229
+ return tokens.join(" ")
230
+ }
231
+
232
+ /** Normalize a locality name for address-point keying (fold only — no street semantics). */
233
+ export function normalizeLocalityForKey(locality: string): string {
234
+ return fold(locality)
235
+ }
236
+
237
+ /**
238
+ * Strip a trailing French arrondissement designator from a FOLDED commune key ("paris 8e arrondissement" → "paris",
239
+ * "lyon 1er arrondissement" → "lyon", "marseille 10e arrondissement" → "marseille"). Paris, Lyon and Marseille are the
240
+ * only French communes subdivided into _arrondissements municipaux_; a national register (BAN) names each row per
241
+ * arrondissement, but a query names the base commune ("Place Bellecour, Lyon", never "…, Lyon 2e"). Applied on BOTH
242
+ * sides of the #1042 street-centroid key — build-side (deriving the `locality_base` column) and query-side (folding the
243
+ * probe commune) — so the two agree by construction (the one-function discipline). Input must already be folded
244
+ * (lower-case, diacritic-stripped); a no-op for every other commune. Returns the input unchanged if the strip would
245
+ * empty it.
246
+ */
247
+ export function stripArrondissement(localityNorm: string): string {
248
+ const stripped = localityNorm.replace(/\s+\d+(?:er|e)\s+arrondissement$/, "").trim()
249
+
250
+ return stripped || localityNorm
251
+ }
252
+
253
+ /**
254
+ * Strip a locality QUALIFIER for a query-side fallback — when an OA locality's exact normalized name misses the
255
+ * gazetteer's canonical name, retry with the qualifier removed. OA address data carries disambiguating qualifiers the
256
+ * gazetteer's canonical name omits: Austrian `Kraubath/Mur` and `Hart b.Graz` → `Hart`; Swiss `Lenk im Simmental` →
257
+ * `Lenk`, `Roche VD` → `Roche`; Danish `Odense S`, `Hurup Thy`. A FALLBACK ONLY — the exact name is tried first, and
258
+ * the region-bbox disambiguation resolves any base-name ambiguity downstream. The candidate table is unchanged (this is
259
+ * purely query-side); feed the result back through {@link normalizeLocalityForKey}. Returns "" when nothing was stripped
260
+ * (no point re-probing the identical key).
261
+ *
262
+ * Measured (`scripts/eval/candidate-recall.ts --strip-fallback`, EU OA holdouts): recovers AT 74.1→88.2% (+14.1pp), DK
263
+ * 91.5→96.2%, CH 90.4→92.6%; +1.3pp overall (diluted by the already-100% locales). Conservative by design — only the
264
+ * qualifier forms above; FI/PT/SI misses are untouched.
265
+ */
266
+ export function stripLocalityQualifier(locality: string): string {
267
+ let s = locality.trim()
268
+
269
+ if (s.includes("/")) {
270
+ s = s.split("/")[0]!.trim()
271
+ } // "Kraubath/Mur", "St.Kanzian/Klopeiner See"
272
+ s = s.replace(/\s+[a-zà-ÿ]\.\s*\S.*$/iu, "") // abbreviated " b.Graz" / " o.Bleiburg" / " a.d. …"
273
+ s = s.replace(/\s+(im|an der|ob|bei|in der|unter|vor)\s+\S.*$/iu, "") // " im Simmental", " bei Graz"
274
+ s = s.replace(/\s+(S|N|E|W|V|Ø|Sø|Fyn|Thy|Sjælland|Jylland|[A-ZÅÄÖ]{2})$/u, "") // " S", " VD", " Thy"
275
+ s = s.trim()
276
+
277
+ return s === locality.trim() ? "" : s
278
+ }
279
+
280
+ /**
281
+ * Fold numbered-route designators to a canonical key, applied AFTER {@link normalizeStreetForKey}. Sources disagree
282
+ * systematically on how they spell a route: TIGER says `State Rte 100` / `US Hwy 5` where E911/Overture say `VT ROUTE
283
+ * 100` / `US ROUTE 5` — the dominant street-name miss class in the #483 interpolation eval (rural addresses live on
284
+ * routes). `us <designator> N…` folds to `us route N…`; `state <designator> N…` and `<2-letter-prefix> <designator> N…`
285
+ * (the state abbreviation form) fold to `state route N…`. Only digit-leading route numbers fold — `State Street` and
286
+ * friends never match.
287
+ *
288
+ * Used by BOTH the segment-shard builder (`scripts/build-interpolation-shard.ts`) and the interpolation lookup — same
289
+ * one-function discipline as {@link normalizeStreetForKey}. The address-point tier (#476) does NOT apply it yet:
290
+ * adopting it there requires a shard rebuild (noted on #483).
291
+ *
292
+ * A same-numbered US and state route stay DISTINCT keys (`us route 5` vs `state route 5`); only the BARE `route N` form
293
+ * is ambiguous (designator unknown) and it stays unfolded — a bare-route query therefore misses rather than guessing a
294
+ * designator.
295
+ */
296
+ export function canonicalizeRouteKey(streetNorm: string): string {
297
+ const match = /^(us|state|[a-z]{2}) (?:route|rte|rt|highway|hwy) (\d.*)$/.exec(streetNorm)
298
+
299
+ if (!match) return streetNorm
300
+
301
+ return `${match[1] === "us" ? "us" : "state"} route ${match[2]}`
302
+ }
@@ -0,0 +1,104 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Typed schema for the TIGER STREET-SEGMENT interpolation shards (`street-segments-<cc>-<st>.db`,
7
+ * built by `scripts/build-interpolation-shard.ts` from TIGER EDGES) — the #483 Method-3 fallback
8
+ * the resolver drops to when the address-point tier (Method 2) can't bracket. Single source of
9
+ * truth for the columns the BUILDER writes and the READER ({@link StreetInterpolator}) probes, so
10
+ * a column rename in one is a compile error in the other.
11
+ *
12
+ * The builder reads geometry from shapefiles via DuckDB's spatial extension (raw `ST_Read` — see
13
+ * AGENTS.md "Database / inline SQL") and writes here through `node:sqlite`. The hot positional
14
+ * INSERT (a county's worth of edges) stays raw; its column list is derived from
15
+ * {@link STREET_SEGMENT_COLUMNS} so it can't drift from the DDL.
16
+ */
17
+
18
+ import type { Kysely } from "kysely"
19
+
20
+ /**
21
+ * One TIGER street-segment edge: a `(from_hn, to_hn)` house-number range on one `side` of a named street, with the
22
+ * geometry the interpolator walks. `min_hn`/`max_hn` are the sorted bounds (the probe filters on them); `parity` is
23
+ * `odd`/`even`/`mixed`.
24
+ */
25
+ export interface StreetSegmentTable {
26
+ /** Shared {@link normalizeStreetForKey} of the street — the build/query-consistent probe key. */
27
+ street_norm: string
28
+ /** `L` or `R` — the TIGER side the address range sits on. */
29
+ side: string
30
+ from_hn: number
31
+ to_hn: number
32
+ /** Sorted lower bound of `(from_hn, to_hn)` — the probe filters `min_hn <= n <= max_hn`. */
33
+ min_hn: number
34
+ /** Sorted upper bound of `(from_hn, to_hn)`. */
35
+ max_hn: number
36
+ /** `odd` | `even` | `mixed` — the house-number parity along the range. */
37
+ parity: string
38
+ postcode: string | null
39
+ /** 5-digit state+county FIPS the edge came from. */
40
+ county_fips: string
41
+ /** The street as it appeared in TIGER (kept for display / debugging). */
42
+ street_raw: string
43
+ /** GeoJSON LineString text (no SpatiaLite — read back with `JSON.parse`). */
44
+ geometry: string
45
+ /** Provenance: the dataset this edge came from (e.g. `tiger:edges`). */
46
+ source: string
47
+ /** The pinned TIGER release the edge was ingested from. */
48
+ release: string
49
+ }
50
+
51
+ /** The street-segment database schema for `new DatabaseClient<StreetSegmentDatabase>(...)`. */
52
+ export interface StreetSegmentDatabase {
53
+ street_segment: StreetSegmentTable
54
+ }
55
+
56
+ /**
57
+ * The `street_segment` columns in INSERT order. The builder's positional prepared statement derives its placeholder
58
+ * list from this, so the positional order can't drift from the DDL / the reader.
59
+ */
60
+ export const STREET_SEGMENT_COLUMNS = [
61
+ "street_norm",
62
+ "side",
63
+ "from_hn",
64
+ "to_hn",
65
+ "min_hn",
66
+ "max_hn",
67
+ "parity",
68
+ "postcode",
69
+ "county_fips",
70
+ "street_raw",
71
+ "geometry",
72
+ "source",
73
+ "release",
74
+ ] as const
75
+
76
+ /** Create the `street_segment` table — called before the streaming bulk load. */
77
+ export async function createStreetSegmentTable(db: Kysely<StreetSegmentDatabase>): Promise<void> {
78
+ await db.schema
79
+ .createTable("street_segment")
80
+ .addColumn("street_norm", "text", (c) => c.notNull())
81
+ .addColumn("side", "text", (c) => c.notNull())
82
+ .addColumn("from_hn", "integer", (c) => c.notNull())
83
+ .addColumn("to_hn", "integer", (c) => c.notNull())
84
+ .addColumn("min_hn", "integer", (c) => c.notNull())
85
+ .addColumn("max_hn", "integer", (c) => c.notNull())
86
+ .addColumn("parity", "text", (c) => c.notNull())
87
+ .addColumn("postcode", "text")
88
+ .addColumn("county_fips", "text", (c) => c.notNull())
89
+ .addColumn("street_raw", "text", (c) => c.notNull())
90
+ .addColumn("geometry", "text", (c) => c.notNull())
91
+ .addColumn("source", "text", (c) => c.notNull())
92
+ .addColumn("release", "text", (c) => c.notNull())
93
+ .execute()
94
+ }
95
+
96
+ /** Create the two probe indexes the reader relies on (postcode-scope, street-scope). */
97
+ export async function createStreetSegmentIndexes(db: Kysely<StreetSegmentDatabase>): Promise<void> {
98
+ await db.schema
99
+ .createIndex("idx_seg_postcode")
100
+ .on("street_segment")
101
+ .columns(["postcode", "street_norm", "min_hn"])
102
+ .execute()
103
+ await db.schema.createIndex("idx_seg_street").on("street_segment").columns(["street_norm", "min_hn"]).execute()
104
+ }
package/types.ts ADDED
@@ -0,0 +1,164 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Public surface for the WOF SQLite resolver — types only, no runtime.
7
+ *
8
+ * These mirror the conceptual model described in `docs/plan/phases/PHASE_4_2_wof_sqlite.md`. Phase
9
+ * 4.3 will extend `PlaceCandidate` with the resolver-decorated fields that flow into
10
+ * `AddressNode.source` / `sourceID` (e.g. an explicit `wofURI: "wof-admin:101751113"` form).
11
+ */
12
+
13
+ /**
14
+ * The placetype taxonomy used by Who's On First. Ordered roughly from coarsest (country) to finest (address). See
15
+ * https://github.com/whosonfirst/whosonfirst-placetypes for the authoritative definitions of each.
16
+ *
17
+ * Phase 4.2 only emits the ones we actually look up; the union is open enough to extend later.
18
+ */
19
+ export type WOFPlacetype =
20
+ | "country"
21
+ | "macroregion"
22
+ | "region"
23
+ | "macrocounty"
24
+ | "county"
25
+ | "localadmin"
26
+ | "locality"
27
+ | "borough"
28
+ | "neighbourhood"
29
+ | "microhood"
30
+ | "postalcode"
31
+ | "venue"
32
+ | "campus"
33
+ | "address"
34
+
35
+ /**
36
+ * One candidate match for a place lookup.
37
+ *
38
+ * `score` is the post-boost ranking number — higher is better, but the scale is implementation- defined. Callers should
39
+ * treat it as ordinal, not absolute.
40
+ *
41
+ * `id` is the WOF place id. It's named generically (not `wof_id`) so the shape stays structurally compatible with
42
+ * `@mailwoman/resolver`'s `ResolvedPlace` — `WOFSqlitePlaceLookup` satisfies the generic `ResolverBackend` contract
43
+ * without an adapter shim.
44
+ *
45
+ * `distanceKm` is populated only when the query carried `near` (and the place has a centroid). Useful for downstream
46
+ * UIs that want to show "X km from you" alongside the result.
47
+ */
48
+ export interface PlaceCandidate {
49
+ id: number
50
+ name: string
51
+ placetype: WOFPlacetype
52
+ /** ISO 3166-1 alpha-2 country code. */
53
+ country: string
54
+ lat: number
55
+ lon: number
56
+ parent_id?: number
57
+ score: number
58
+ distanceKm?: number
59
+ /**
60
+ * True when this candidate's name OR an alias EXACTLY equals the query (the exact-match tier from
61
+ * {@link RankingWeights.exactMatchTiering}). Surfaced so a downstream country re-rank (#369's postcode anchor in
62
+ * `resolveTree`) can pin the country without crossing the tier — see the `exactMatch` field on `@mailwoman/core`'s
63
+ * `ResolvedPlace`.
64
+ */
65
+ exactMatch?: boolean
66
+ /**
67
+ * Combined prominence (population term + best proximity-bias term, same additive units) — populated by the FTS
68
+ * lookup; the exact-tier sort orders by THIS instead of raw population when the query carried proximity hints
69
+ * (`near`/`bias`).
70
+ */
71
+ prominence?: number
72
+ /**
73
+ * Population from WOF's `wof:population` property. Only present when the candidate has it on record — WOF carries
74
+ * population for ~15% of localities (mostly larger ones). Absent does NOT mean zero, just unknown.
75
+ */
76
+ population?: number
77
+ /**
78
+ * Bounding box from WOF's `spr.{min,max}_{latitude,longitude}` columns. Coarse outline for the place — a city's bbox
79
+ * is the city's full extent, a postcode's is roughly the postcode polygon's envelope. Optional because not all
80
+ * callers ask for it; implementations are free to omit when the underlying schema lacks the columns.
81
+ */
82
+ bbox?: GeoBbox
83
+ /**
84
+ * Set by the coordinate-first path when the chosen locality and the sibling postcode's containing locality are
85
+ * geographically far apart — the postcode and the parsed city name disagree (a transposed / wrong-for-the-city
86
+ * postcode). The candidate is still returned (the name wins for the locality), but the flag lets callers lower
87
+ * confidence / surface the conflict rather than silently mislocate. A retrieval/BM25 geocoder can't raise this — it's
88
+ * the falsehood-detection differentiator.
89
+ */
90
+ mismatch?: boolean
91
+ }
92
+
93
+ /**
94
+ * A WGS-84 lat/lon point. Used as a proximity hint for `FindPlaceQuery.near`.
95
+ */
96
+ export interface GeoPoint {
97
+ lat: number
98
+ lon: number
99
+ }
100
+
101
+ /**
102
+ * A WGS-84 bounding box. Used as a hard filter via `FindPlaceQuery.bbox`.
103
+ */
104
+ export interface GeoBbox {
105
+ minLat: number
106
+ maxLat: number
107
+ minLon: number
108
+ maxLon: number
109
+ }
110
+
111
+ /**
112
+ * Query against the resolver.
113
+ *
114
+ * `text` is the only required field; everything else narrows the search. When `country` and `parentID` are both set,
115
+ * `parentID` wins (it's more specific).
116
+ *
117
+ * `near` and `bbox` are independent. `near` is a soft signal — candidates close to the point get a ranking boost but
118
+ * distant candidates aren't dropped. `bbox` is a hard filter — only candidates whose bbox intersects the query bbox are
119
+ * returned (uses the package-built R*Tree index when present; if the index is missing the option is silently ignored to
120
+ * preserve backwards compatibility).
121
+ *
122
+ * `near` may carry `maxDistanceKm` to escalate from a boost to a hard filter — candidates further than that distance
123
+ * from the point are dropped at the SQL level via an R*Tree pre-filter.
124
+ */
125
+ export interface FindPlaceQuery {
126
+ text: string
127
+ placetype?: WOFPlacetype | WOFPlacetype[]
128
+ /** ISO 3166-1 alpha-2 — narrows to one country. */
129
+ country?: string
130
+ /** WOF place id — narrows to descendants of this place. */
131
+ parentID?: number
132
+ /**
133
+ * Sibling postcode. When set on a `locality` query AND a `postcode_locality` table is present, triggers the
134
+ * coordinate-first soft-score path: postcode→candidate localities are injected and scored `0.6·S_pc + 0.3·S_name +
135
+ * 0.1·S_pop` against the FTS name-match set, recovering small localities the name-match alone misses. Ignored when no
136
+ * postcode_locality shard is present.
137
+ */
138
+ postcode?: string
139
+ /** Proximity hint — candidates close to this point get a ranking boost. */
140
+ near?: GeoPoint & { maxDistanceKm?: number }
141
+ /**
142
+ * Ordered proximity-bias points (viewport center, user location, …), each optionally weighted (default 1.0, first
143
+ * entry strongest by convention). SOFT — a re-rank signal, never a filter: with bias present, exact-tier candidates
144
+ * order by combined prominence (population + the best decayed-distance term over these points) instead of population
145
+ * alone, which is how an ambiguous bare postcode ("48026": Fraser MI vs Russi IT) follows the map view / the user.
146
+ * Absent (and no `near`) → ranking is byte-identical to today. `near` is treated as a weight-1.0 bias point for
147
+ * back-compat.
148
+ */
149
+ bias?: Array<GeoPoint & { weight?: number }>
150
+ /** Bounding-box filter — only candidates whose bbox intersects this box are returned. */
151
+ bbox?: GeoBbox
152
+ /** Default 10. */
153
+ limit?: number
154
+ }
155
+
156
+ /**
157
+ * The pull-based lookup surface. Implementations resolve a `FindPlaceQuery` to a ranked list of `PlaceCandidate`s. The
158
+ * interface is async even though `node:sqlite` is sync — leaves room for `Worker`-backed implementations later without
159
+ * a public API break.
160
+ */
161
+ export interface PlaceLookup {
162
+ findPlace(query: FindPlaceQuery): Promise<PlaceCandidate[]>
163
+ close(): void
164
+ }