@mailwoman/resolver-wof-sqlite 9.0.0 → 9.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -9
- package/address-point-interpolation.ts +18 -8
- package/address-point-schema.ts +18 -6
- package/address-point.ts +111 -18
- package/ancestry.ts +9 -6
- package/build-candidate.ts +287 -157
- package/build-slim.ts +3 -3
- package/candidate/alias-bags.ts +54 -0
- package/candidate/ancestors-sidecar.ts +206 -0
- package/candidate/country-display-names.ts +79 -0
- package/candidate/name-roles.ts +237 -0
- package/candidate/own-name.ts +146 -0
- package/candidate/place-attrs.ts +44 -0
- package/candidate/shard-fold.ts +137 -0
- package/candidate-ancestors-schema.ts +195 -0
- package/candidate-fts.ts +4 -2
- package/candidate-importance.ts +228 -0
- package/candidate-lookup.ts +564 -174
- package/candidate-schema.ts +60 -3
- package/candidate-scoring.ts +268 -0
- package/capital-schema.ts +90 -0
- package/capitals.ts +148 -0
- package/coincident-roles.ts +69 -10
- package/convention-schema.ts +72 -0
- package/convention.ts +2 -2
- package/coverage-manifest-schema.ts +7 -7
- package/currency-backfill.ts +249 -0
- package/exact-match.ts +104 -0
- package/fst-autocomplete.ts +105 -122
- package/fst-builder.ts +39 -47
- package/fst-deserialize-web.ts +43 -7
- package/fst-freshness.ts +2 -2
- package/fst-serialize.ts +68 -12
- package/fst-types.ts +35 -1
- package/fts-query.ts +1 -1
- package/fts.ts +16 -4
- package/geonames-postal.ts +2 -2
- package/index.ts +26 -14
- package/interpolation.ts +113 -19
- package/lookup.ts +118 -560
- package/name-score.ts +6 -4
- package/out/address-point-interpolation.d.ts.map +1 -1
- package/out/address-point-interpolation.js +13 -7
- package/out/address-point-interpolation.js.map +1 -1
- package/out/address-point-schema.d.ts +16 -6
- package/out/address-point-schema.d.ts.map +1 -1
- package/out/address-point-schema.js.map +1 -1
- package/out/address-point.d.ts.map +1 -1
- package/out/address-point.js +70 -14
- package/out/address-point.js.map +1 -1
- package/out/ancestry.d.ts +2 -2
- package/out/ancestry.d.ts.map +1 -1
- package/out/ancestry.js +5 -6
- package/out/ancestry.js.map +1 -1
- package/out/build-candidate.d.ts +108 -0
- package/out/build-candidate.d.ts.map +1 -1
- package/out/build-candidate.js +151 -120
- package/out/build-candidate.js.map +1 -1
- package/out/build-slim.d.ts +1 -1
- package/out/build-slim.js +3 -3
- package/out/build-slim.js.map +1 -1
- package/out/candidate/alias-bags.d.ts +17 -0
- package/out/candidate/alias-bags.d.ts.map +1 -0
- package/out/candidate/alias-bags.js +39 -0
- package/out/candidate/alias-bags.js.map +1 -0
- package/out/candidate/ancestors-sidecar.d.ts +33 -0
- package/out/candidate/ancestors-sidecar.d.ts.map +1 -0
- package/out/candidate/ancestors-sidecar.js +140 -0
- package/out/candidate/ancestors-sidecar.js.map +1 -0
- package/out/candidate/country-display-names.d.ts +35 -0
- package/out/candidate/country-display-names.d.ts.map +1 -0
- package/out/candidate/country-display-names.js +59 -0
- package/out/candidate/country-display-names.js.map +1 -0
- package/out/candidate/name-roles.d.ts +55 -0
- package/out/candidate/name-roles.d.ts.map +1 -0
- package/out/candidate/name-roles.js +165 -0
- package/out/candidate/name-roles.js.map +1 -0
- package/out/candidate/own-name.d.ts +50 -0
- package/out/candidate/own-name.d.ts.map +1 -0
- package/out/candidate/own-name.js +132 -0
- package/out/candidate/own-name.js.map +1 -0
- package/out/candidate/place-attrs.d.ts +43 -0
- package/out/candidate/place-attrs.d.ts.map +1 -0
- package/out/candidate/place-attrs.js +15 -0
- package/out/candidate/place-attrs.js.map +1 -0
- package/out/candidate/shard-fold.d.ts +31 -0
- package/out/candidate/shard-fold.d.ts.map +1 -0
- package/out/candidate/shard-fold.js +104 -0
- package/out/candidate/shard-fold.js.map +1 -0
- package/out/candidate-ancestors-schema.d.ts +150 -0
- package/out/candidate-ancestors-schema.d.ts.map +1 -0
- package/out/candidate-ancestors-schema.js +123 -0
- package/out/candidate-ancestors-schema.js.map +1 -0
- package/out/candidate-fts.d.ts +4 -2
- package/out/candidate-fts.d.ts.map +1 -1
- package/out/candidate-fts.js +4 -2
- package/out/candidate-fts.js.map +1 -1
- package/out/candidate-importance.d.ts +132 -0
- package/out/candidate-importance.d.ts.map +1 -0
- package/out/candidate-importance.js +174 -0
- package/out/candidate-importance.js.map +1 -0
- package/out/candidate-lookup.d.ts +22 -37
- package/out/candidate-lookup.d.ts.map +1 -1
- package/out/candidate-lookup.js +446 -132
- package/out/candidate-lookup.js.map +1 -1
- package/out/candidate-schema.d.ts +52 -4
- package/out/candidate-schema.d.ts.map +1 -1
- package/out/candidate-schema.js +8 -0
- package/out/candidate-schema.js.map +1 -1
- package/out/candidate-scoring.d.ts +34 -0
- package/out/candidate-scoring.d.ts.map +1 -0
- package/out/candidate-scoring.js +200 -0
- package/out/candidate-scoring.js.map +1 -0
- package/out/capital-schema.d.ts +51 -0
- package/out/capital-schema.d.ts.map +1 -0
- package/out/capital-schema.js +63 -0
- package/out/capital-schema.js.map +1 -0
- package/out/capitals.d.ts +69 -0
- package/out/capitals.d.ts.map +1 -0
- package/out/capitals.js +98 -0
- package/out/capitals.js.map +1 -0
- package/out/coincident-roles.d.ts +7 -0
- package/out/coincident-roles.d.ts.map +1 -1
- package/out/coincident-roles.js +42 -8
- package/out/coincident-roles.js.map +1 -1
- package/out/convention-schema.d.ts +51 -0
- package/out/convention-schema.d.ts.map +1 -0
- package/out/convention-schema.js +34 -0
- package/out/convention-schema.js.map +1 -0
- package/out/convention.d.ts +1 -1
- package/out/convention.js +2 -2
- package/out/coverage-manifest-schema.js +3 -7
- package/out/coverage-manifest-schema.js.map +1 -1
- package/out/currency-backfill.d.ts +46 -0
- package/out/currency-backfill.d.ts.map +1 -0
- package/out/currency-backfill.js +180 -0
- package/out/currency-backfill.js.map +1 -0
- package/out/exact-match.d.ts +25 -0
- package/out/exact-match.d.ts.map +1 -0
- package/out/exact-match.js +89 -0
- package/out/exact-match.js.map +1 -0
- package/out/fst-autocomplete.d.ts +24 -14
- package/out/fst-autocomplete.d.ts.map +1 -1
- package/out/fst-autocomplete.js +84 -100
- package/out/fst-autocomplete.js.map +1 -1
- package/out/fst-builder.d.ts.map +1 -1
- package/out/fst-builder.js +32 -40
- package/out/fst-builder.js.map +1 -1
- package/out/fst-deserialize-web.d.ts.map +1 -1
- package/out/fst-deserialize-web.js +36 -7
- package/out/fst-deserialize-web.js.map +1 -1
- package/out/fst-freshness.d.ts +2 -2
- package/out/fst-freshness.js +2 -2
- package/out/fst-serialize.d.ts +14 -4
- package/out/fst-serialize.d.ts.map +1 -1
- package/out/fst-serialize.js +60 -12
- package/out/fst-serialize.js.map +1 -1
- package/out/fst-types.d.ts +35 -1
- package/out/fst-types.d.ts.map +1 -1
- package/out/fts-query.js +1 -1
- package/out/fts-query.js.map +1 -1
- package/out/fts.d.ts +15 -4
- package/out/fts.d.ts.map +1 -1
- package/out/fts.js +15 -4
- package/out/fts.js.map +1 -1
- package/out/geonames-postal.d.ts +2 -2
- package/out/geonames-postal.js +2 -2
- package/out/index.d.ts +4 -2
- package/out/index.d.ts.map +1 -1
- package/out/index.js +3 -2
- package/out/index.js.map +1 -1
- package/out/interpolation.d.ts +8 -0
- package/out/interpolation.d.ts.map +1 -1
- package/out/interpolation.js +91 -19
- package/out/interpolation.js.map +1 -1
- package/out/lookup.d.ts +4 -5
- package/out/lookup.d.ts.map +1 -1
- package/out/lookup.js +102 -444
- package/out/lookup.js.map +1 -1
- package/out/name-score.d.ts +0 -10
- package/out/name-score.d.ts.map +1 -1
- package/out/name-score.js +6 -4
- package/out/name-score.js.map +1 -1
- package/out/place-importance-schema.d.ts +226 -0
- package/out/place-importance-schema.d.ts.map +1 -0
- package/out/place-importance-schema.js +288 -0
- package/out/place-importance-schema.js.map +1 -0
- package/out/poi-lookup.d.ts +1 -1
- package/out/poi-lookup.d.ts.map +1 -1
- package/out/poi-lookup.js +12 -13
- package/out/poi-lookup.js.map +1 -1
- package/out/poi-schema.d.ts +7 -3
- package/out/poi-schema.d.ts.map +1 -1
- package/out/poi-schema.js.map +1 -1
- package/out/polygon-schema.d.ts +37 -0
- package/out/polygon-schema.d.ts.map +1 -0
- package/out/polygon-schema.js +23 -0
- package/out/polygon-schema.js.map +1 -0
- package/out/postal-city-alias-lookup.d.ts +1 -1
- package/out/postal-city-alias-lookup.js +1 -1
- package/out/postal-city-candidate-schema.d.ts +2 -1
- package/out/postal-city-candidate-schema.d.ts.map +1 -1
- package/out/postal-city-candidate-schema.js.map +1 -1
- package/out/postcode-point-lookup.d.ts +1 -1
- package/out/postcode-point-lookup.js +1 -1
- package/out/primary-preference.d.ts +125 -0
- package/out/primary-preference.d.ts.map +1 -0
- package/out/primary-preference.js +138 -0
- package/out/primary-preference.js.map +1 -0
- package/out/proximity-rerank.d.ts +77 -0
- package/out/proximity-rerank.d.ts.map +1 -0
- package/out/proximity-rerank.js +86 -0
- package/out/proximity-rerank.js.map +1 -0
- package/out/region-keys.d.ts +47 -0
- package/out/region-keys.d.ts.map +1 -0
- package/out/region-keys.js +121 -0
- package/out/region-keys.js.map +1 -0
- package/out/reverse.d.ts.map +1 -1
- package/out/reverse.js +6 -9
- package/out/reverse.js.map +1 -1
- package/out/schema.d.ts +1 -1
- package/out/search-fetch.d.ts +57 -0
- package/out/search-fetch.d.ts.map +1 -0
- package/out/search-fetch.js +183 -0
- package/out/search-fetch.js.map +1 -0
- package/out/sharding.d.ts +3 -3
- package/out/sharding.js +1 -1
- package/out/sqlite-convention-source.d.ts +1 -1
- package/out/sqlite-convention-source.js +1 -1
- package/out/sqlite-utils.d.ts +31 -1
- package/out/sqlite-utils.d.ts.map +1 -1
- package/out/sqlite-utils.js +38 -0
- package/out/sqlite-utils.js.map +1 -1
- package/out/street-centroid-schema.d.ts +7 -2
- package/out/street-centroid-schema.d.ts.map +1 -1
- package/out/street-centroid-schema.js.map +1 -1
- package/out/street-centroid.d.ts.map +1 -1
- package/out/street-centroid.js +7 -7
- package/out/street-centroid.js.map +1 -1
- package/out/street-morphology-fst-builder.d.ts.map +1 -1
- package/out/street-morphology-fst-builder.js +5 -4
- package/out/street-morphology-fst-builder.js.map +1 -1
- package/out/street-normalize.d.ts +83 -9
- package/out/street-normalize.d.ts.map +1 -1
- package/out/street-normalize.js +177 -10
- package/out/street-normalize.js.map +1 -1
- package/out/street-segment-schema.d.ts +6 -2
- package/out/street-segment-schema.d.ts.map +1 -1
- package/out/street-segment-schema.js.map +1 -1
- package/out/types.d.ts +74 -1
- package/out/types.d.ts.map +1 -1
- package/out/unified-schema.d.ts +1 -1
- package/out/unified-schema.js +1 -1
- package/out/uprn-lookup.d.ts +85 -0
- package/out/uprn-lookup.d.ts.map +1 -0
- package/out/uprn-lookup.js +152 -0
- package/out/uprn-lookup.js.map +1 -0
- package/out/uprn-schema.d.ts +93 -0
- package/out/uprn-schema.d.ts.map +1 -0
- package/out/uprn-schema.js +78 -0
- package/out/uprn-schema.js.map +1 -0
- package/out/weights-overlay-linker.d.ts +141 -0
- package/out/weights-overlay-linker.d.ts.map +1 -0
- package/out/weights-overlay-linker.js +259 -0
- package/out/weights-overlay-linker.js.map +1 -0
- package/package.json +296 -16
- package/place-importance-schema.ts +402 -0
- package/poi-lookup.ts +12 -13
- package/poi-schema.ts +8 -3
- package/polygon-schema.ts +47 -0
- package/postal-city-alias-lookup.ts +1 -1
- package/postal-city-candidate-schema.ts +3 -1
- package/postcode-point-lookup.ts +1 -1
- package/primary-preference.ts +207 -0
- package/proximity-rerank.ts +120 -0
- package/region-keys.ts +144 -0
- package/reverse.ts +17 -16
- package/schema.ts +1 -1
- package/search-fetch.ts +256 -0
- package/sharding.ts +3 -3
- package/sqlite-convention-source.ts +1 -1
- package/sqlite-utils.ts +63 -1
- package/street-centroid-schema.ts +8 -2
- package/street-centroid.ts +13 -8
- package/street-morphology-fst-builder.ts +5 -4
- package/street-normalize.ts +254 -24
- package/street-segment-schema.ts +7 -2
- package/types.ts +74 -1
- package/unified-schema.ts +1 -1
- package/uprn-lookup.ts +210 -0
- package/uprn-schema.ts +124 -0
- package/weights-overlay-linker.ts +377 -0
- package/geo.ts +0 -121
- package/out/geo.d.ts +0 -74
- package/out/geo.d.ts.map +0 -1
- package/out/geo.js +0 -71
- package/out/geo.js.map +0 -1
package/street-normalize.ts
CHANGED
|
@@ -23,6 +23,37 @@
|
|
|
23
23
|
*/
|
|
24
24
|
|
|
25
25
|
import { AbbreviationToDirectional, US_STREET_SUFFIX_LOOKUP } from "@mailwoman/codex/us"
|
|
26
|
+
import type { Tagged } from "type-fest"
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* A place name folded by {@link normalizeLocalityForKey} — the value stored in, and required to probe, every
|
|
30
|
+
* `name_key`-style column: the candidate gazetteer, the postal-city side-index, the street-centroid shard, an ancestor
|
|
31
|
+
* chain's `parent_name_key`, and the address-point locality scope.
|
|
32
|
+
*
|
|
33
|
+
* The brand is here because the fold is applied at BUILD time and is therefore mandatory at QUERY time, while a
|
|
34
|
+
* near-miss approximation of it (`toLowerCase()`, `trim()`) is still a `string`: it binds to the parameter, returns
|
|
35
|
+
* fewer rows, and the shortfall reads as a coverage gap in the data rather than a defect in the probe. Requiring the
|
|
36
|
+
* brand at the seam turns that silent under-match into a compile error. Mint one only by calling the fold.
|
|
37
|
+
*/
|
|
38
|
+
export type NameKey = Tagged<string, "NameKey">
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* A street name folded by {@link normalizeStreetForKey} / {@link normalizeStreetForKeyLocale} — the value stored in a
|
|
42
|
+
* `street_norm` address-point column and required to probe it.
|
|
43
|
+
*
|
|
44
|
+
* Distinct from {@link RouteKey}: the route fold is a FURTHER canonicalization applied to some columns and not others,
|
|
45
|
+
* so the two are not interchangeable even though both are folded streets.
|
|
46
|
+
*/
|
|
47
|
+
export type StreetKey = Tagged<string, "StreetKey">
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* A street key that has additionally passed {@link canonicalizeRouteKey} — the numbered-route canonical form.
|
|
51
|
+
*
|
|
52
|
+
* A separate brand from {@link StreetKey} because it identifies a DIFFERENT set of columns, and the two folds disagree
|
|
53
|
+
* exactly on the rows that motivate the route fold: binding a plain street key to a route-folded column silently misses
|
|
54
|
+
* every row whose source spelled the route differently, which is the miss class the fold exists to close.
|
|
55
|
+
*/
|
|
56
|
+
export type RouteKey = Tagged<string, "RouteKey">
|
|
26
57
|
|
|
27
58
|
/**
|
|
28
59
|
* Token count a street must exceed before its trailing pair is merged. At or below it the pair IS the whole street
|
|
@@ -86,10 +117,10 @@ function fold(input: string): string {
|
|
|
86
117
|
* Normalize a street name for address-point keying. Same function at build time and lookup time — see module docstring
|
|
87
118
|
* for the contract.
|
|
88
119
|
*/
|
|
89
|
-
export function normalizeStreetForKey(street: string):
|
|
120
|
+
export function normalizeStreetForKey(street: string): StreetKey {
|
|
90
121
|
const tokens = fold(street).split(" ")
|
|
91
122
|
|
|
92
|
-
if (!tokens.length) return ""
|
|
123
|
+
if (!tokens.length) return "" as StreetKey
|
|
93
124
|
|
|
94
125
|
// Spelled-ordinal street names → digit form when a street suffix follows ("Tenth Street" →
|
|
95
126
|
// "10th street", #723). Gated on the next token being a suffix so ordinal-WORD names are untouched.
|
|
@@ -151,7 +182,7 @@ export function normalizeStreetForKey(street: string): string {
|
|
|
151
182
|
}
|
|
152
183
|
}
|
|
153
184
|
|
|
154
|
-
return tokens.join(" ")
|
|
185
|
+
return tokens.join(" ") as StreetKey
|
|
155
186
|
}
|
|
156
187
|
|
|
157
188
|
/**
|
|
@@ -162,7 +193,7 @@ export function normalizeStreetForKey(street: string): string {
|
|
|
162
193
|
* article, so no salient-token / multi-key index is built yet (deferred until probing shows the normalizer can't absorb
|
|
163
194
|
* the false-negatives).
|
|
164
195
|
*/
|
|
165
|
-
export type StreetLocale = "us" | "fr" | "de" | "nl"
|
|
196
|
+
export type StreetLocale = "us" | "en" | "fr" | "de" | "nl" | "pl" | "vn" | "id"
|
|
166
197
|
|
|
167
198
|
/**
|
|
168
199
|
* French street-type abbreviations → canonical full form, applied per token after {@link fold}. French address types
|
|
@@ -191,25 +222,60 @@ const FR_STREET_ABBREV = new Map<string, string>([
|
|
|
191
222
|
["sts", "saints"],
|
|
192
223
|
])
|
|
193
224
|
|
|
225
|
+
/**
|
|
226
|
+
* Polish leading street-type tokens, STRIPPED rather than expanded. Measured on the 5.56M-row PL OSM shard
|
|
227
|
+
* (2026-08-19): OSM Poland tags `addr:street` BARE — `ulica%` covers 22 rows and `ul.%` eight, so an EXPANSION rule
|
|
228
|
+
* makes a typed query ("ul. Świętokrzyska" → "ulica swietokrzyska") miss the 3,846 bare "swietokrzyska" rows. Stripping
|
|
229
|
+
* the leading type on BOTH sides keys typed and bare surfaces identically. The full words are stripped too: the
|
|
230
|
+
* aleja/plac/osiedle populations (48,597 / 26,815 / 39,552 rows) spell the word out, and "Plac Zamkowy" must key the
|
|
231
|
+
* same as a query's "plac zamkowy" or bare "Zamkowy". Never stripped when it is the only token.
|
|
232
|
+
*/
|
|
233
|
+
const PL_LEADING_TYPE = new Set(["ul", "ulica", "al", "aleja", "aleje", "pl", "plac", "os", "osiedle"])
|
|
234
|
+
|
|
235
|
+
/**
|
|
236
|
+
* Indonesian street-type abbreviations → canonical full form, leading position ("Jl. Thamrin", "Gg. Waru").
|
|
237
|
+
*/
|
|
238
|
+
const ID_STREET_ABBREV = new Map<string, string>([
|
|
239
|
+
["jl", "jalan"],
|
|
240
|
+
["jln", "jalan"],
|
|
241
|
+
["gg", "gang"],
|
|
242
|
+
])
|
|
243
|
+
|
|
194
244
|
/**
|
|
195
245
|
* Normalize a street name for the address-point key in a non-US locale. Same function build-side and probe-side (the
|
|
196
246
|
* one-function discipline). US delegates to {@link normalizeStreetForKey}.
|
|
197
247
|
*
|
|
248
|
+
* - **en** — the shared English address-key pipeline (directionals + suffix aliases), currently identical to US.
|
|
198
249
|
* - **fr** — fold + expand leading type abbreviations and Saint/Sainte (token map).
|
|
199
250
|
* - **de** — fold + ß→ss + canonicalize the GLUED `-str(.)` suffix to `-strasse` ("Lindenstr." → "lindenstrasse",
|
|
200
251
|
* "Lindenstraße" → "lindenstrasse"); an already-full "-strasse" is left intact.
|
|
201
252
|
* - **nl** — fold + canonicalize the glued `-str` suffix to `-straat` ("Kerkstr." → "kerkstraat").
|
|
253
|
+
* - **pl** — fold + ł→l (Ł does not NFKD-decompose, so the generic fold keeps it and an undiacritized query would miss;
|
|
254
|
+
* every other Polish diacritic is a combining form the fold already strips) + STRIP the leading type token — see
|
|
255
|
+
* {@link PL_LEADING_TYPE} for the measured reason expansion was wrong for this source.
|
|
256
|
+
* - **vn** — fold + đ→d AND ð→d (both non-decomposing, and OSM mixes the two codepoints inside single values — see the
|
|
257
|
+
* branch comment). Deliberately NO type-abbreviation map yet: the common abbreviation is the single letter "Đ." for
|
|
258
|
+
* Đường, and expanding a bare folded "d" token would rewrite initials — measure the miss rate on the built shard
|
|
259
|
+
* before adding anything.
|
|
260
|
+
* - **id** — fold + expand leading type abbreviations (jl/jln→jalan, gg→gang); Indonesian street surfaces are otherwise
|
|
261
|
+
* ASCII-clean.
|
|
202
262
|
*/
|
|
203
|
-
export function normalizeStreetForKeyLocale(street: string, locale: StreetLocale):
|
|
204
|
-
if (locale === "us") return normalizeStreetForKey(street)
|
|
263
|
+
export function normalizeStreetForKeyLocale(street: string, locale: StreetLocale): StreetKey {
|
|
264
|
+
if (locale === "us" || locale === "en") return normalizeStreetForKey(street)
|
|
205
265
|
|
|
206
266
|
// Hyphen → space so a compound name keys the same whether the source or the query writes the
|
|
207
267
|
// hyphen ("Champs-Élysées", "St-Honoré") or a space — both sides fold identically, so this is pure
|
|
208
268
|
// robustness. It also splits a hyphenated abbreviation ("St-Honoré" → "st honore") into tokens the
|
|
209
|
-
// per-locale type/Saint map can see.
|
|
210
|
-
|
|
269
|
+
// per-locale type/Saint map can see. Letter maps for non-decomposing letters (ß, ł, đ) live in the
|
|
270
|
+
// per-locale branches, NEVER here: widening the shared pipeline would silently change keys under
|
|
271
|
+
// every already-built shard of the other locales.
|
|
272
|
+
const tokens = fold(street)
|
|
273
|
+
.replaceAll("ß", "ss")
|
|
274
|
+
.replaceAll("-", " ")
|
|
275
|
+
.split(/\s+/)
|
|
276
|
+
.filter((value) => value.length > 0)
|
|
211
277
|
|
|
212
|
-
if (!tokens.length) return ""
|
|
278
|
+
if (!tokens.length) return "" as StreetKey
|
|
213
279
|
|
|
214
280
|
switch (locale) {
|
|
215
281
|
case "fr":
|
|
@@ -235,16 +301,69 @@ export function normalizeStreetForKeyLocale(street: string, locale: StreetLocale
|
|
|
235
301
|
}
|
|
236
302
|
}
|
|
237
303
|
break
|
|
304
|
+
case "pl":
|
|
305
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
306
|
+
tokens[i] = tokens[i]!.replaceAll("ł", "l")
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (tokens.length > 1 && PL_LEADING_TYPE.has(tokens[0]!)) {
|
|
310
|
+
tokens.shift()
|
|
311
|
+
}
|
|
312
|
+
break
|
|
313
|
+
case "id":
|
|
314
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
315
|
+
tokens[i] = ID_STREET_ABBREV.get(tokens[i]!) ?? tokens[i]!
|
|
316
|
+
}
|
|
317
|
+
break
|
|
318
|
+
case "vn":
|
|
319
|
+
// đ (U+0111, d-with-stroke) AND ð (U+00F0, eth): OSM Vietnamese data mixes the two visually
|
|
320
|
+
// identical codepoints inside single values — the 2026-08-19 shard measured `Đường Trần Hưng
|
|
321
|
+
// Đạo` carrying d-with-stroke at the front and ETH in `Đạo`, splitting the country's most
|
|
322
|
+
// common street into two keys with the MAJORITY variant (630 vs 421 rows) unreachable by a
|
|
323
|
+
// correctly-typed query. No abbreviation map — see the locale table.
|
|
324
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
325
|
+
tokens[i] = tokens[i]!.replaceAll("đ", "d").replaceAll("ð", "d")
|
|
326
|
+
}
|
|
327
|
+
break
|
|
238
328
|
}
|
|
239
329
|
|
|
240
|
-
return tokens.join(" ")
|
|
330
|
+
return tokens.join(" ") as StreetKey
|
|
241
331
|
}
|
|
242
332
|
|
|
243
333
|
/**
|
|
244
334
|
* Normalize a locality name for address-point keying (fold only — no street semantics).
|
|
245
335
|
*/
|
|
246
|
-
export function normalizeLocalityForKey(locality: string):
|
|
247
|
-
return fold(locality)
|
|
336
|
+
export function normalizeLocalityForKey(locality: string): NameKey {
|
|
337
|
+
return fold(locality) as NameKey
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* Leading French street-type token. French types LEAD the name ("avenue du Parc", "boul Saint-Laurent") while English
|
|
342
|
+
* types TRAIL ("Fifth Avenue", "Grosvenor Place"), so a leading type-token is the discriminating signal — "1 Avenue NE"
|
|
343
|
+
* starts with a digit and never matches. The separator lookahead is explicit rather than `\b` because a word boundary
|
|
344
|
+
* after an accented final letter ("carré") is not one to an ASCII-word `\b`.
|
|
345
|
+
*/
|
|
346
|
+
const FRENCH_LEAD_TYPE =
|
|
347
|
+
/^(?:rue|ruelle|av|ave|avenue|boul|bd|boulevard|ch|che|chemin|all[ée]e|imp|impasse|mont[ée]e|c[ôo]te|pl|place|prom|promenade|rang|rte|route|autoroute|carr[eé]|croissant|terrasse|sentier)(?=[\s.-]|$)/i
|
|
348
|
+
|
|
349
|
+
/**
|
|
350
|
+
* Surface-driven street-locale ROUTER for bilingual shards — the Québec finishing move on the CA rooftop shard.
|
|
351
|
+
*
|
|
352
|
+
* A country registers ONE street locale, but Canada's street surfaces are two languages: under the `en` rules a French
|
|
353
|
+
* surface passes through mostly unchanged (both sides fold identically, so those rows stay reachable), and what breaks
|
|
354
|
+
* is abbreviation variance — the `en` rules cannot fold "boul"/"Ste-" to the full French word, so an abbreviated query
|
|
355
|
+
* misses a full-word row and vice versa. Routing on the SURFACE (not the province) also carries bilingual NB and the
|
|
356
|
+
* French street names outside Québec for free.
|
|
357
|
+
*
|
|
358
|
+
* ONE function, called by the shard BUILDER and the query PROBE alike — the #861 discipline: routing is part of the
|
|
359
|
+
* fold contract, and two transcriptions of this predicate would diverge exactly where it matters. Only an `en` base
|
|
360
|
+
* re-routes: a `fr`/`de`/`nl` shard already speaks its own rules, and the US pipeline stays untouched.
|
|
361
|
+
*
|
|
362
|
+
* Measured basis (CA shard, 2026-08-19): 183,963 distinct surfaces, 43,762 French-lead; 29,682 fold differently under
|
|
363
|
+
* fr-vs-en (888,265 rows), and the non-French half of those are English surfaces this router keeps on `en` unchanged.
|
|
364
|
+
*/
|
|
365
|
+
export function streetLocaleForSurface(street: string, base: StreetLocale): StreetLocale {
|
|
366
|
+
return base === "en" && FRENCH_LEAD_TYPE.test(street.trimStart()) ? "fr" : base
|
|
248
367
|
}
|
|
249
368
|
|
|
250
369
|
/**
|
|
@@ -253,12 +372,13 @@ export function normalizeLocalityForKey(locality: string): string {
|
|
|
253
372
|
* only French communes subdivided into _arrondissements municipaux_; a national register (BAN) names each row per
|
|
254
373
|
* arrondissement, but a query names the base commune ("Place Bellecour, Lyon", never "…, Lyon 2e"). Applied on BOTH
|
|
255
374
|
* sides of the #1042 street-centroid key — build-side (deriving the `locality_base` column) and query-side (folding the
|
|
256
|
-
* probe commune) — so the two agree by construction (the one-function discipline).
|
|
257
|
-
*
|
|
258
|
-
*
|
|
375
|
+
* probe commune) — so the two agree by construction (the one-function discipline). The {@link NameKey} parameter is the
|
|
376
|
+
* enforcement of "must already be folded": stripping an arrondissement off an unfolded surface yields a key neither
|
|
377
|
+
* side stores. A no-op for every other commune, and the strip leaves a fold fixed point, so the result is still a
|
|
378
|
+
* {@link NameKey}. Returns the input unchanged if the strip would empty it.
|
|
259
379
|
*/
|
|
260
|
-
export function stripArrondissement(localityNorm:
|
|
261
|
-
const stripped = localityNorm.replace(/\s+\d+(?:er|e)\s+arrondissement$/, "").trim()
|
|
380
|
+
export function stripArrondissement(localityNorm: NameKey): NameKey {
|
|
381
|
+
const stripped = localityNorm.replace(/\s+\d+(?:er|e)\s+arrondissement$/, "").trim() as NameKey
|
|
262
382
|
|
|
263
383
|
return stripped || localityNorm
|
|
264
384
|
}
|
|
@@ -283,10 +403,43 @@ export function stripLocalityQualifier(locality: string): string {
|
|
|
283
403
|
s = s.split("/")[0]!.trim()
|
|
284
404
|
}
|
|
285
405
|
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
406
|
+
const words = [...s.matchAll(/\S+/gu)].map((match) => ({ value: match[0], index: match.index }))
|
|
407
|
+
let qualifierStart: number | undefined
|
|
408
|
+
|
|
409
|
+
for (let index = 1; index < words.length; index++) {
|
|
410
|
+
const word = words[index]!
|
|
411
|
+
const first = word.value[0] ?? ""
|
|
412
|
+
const abbreviated = /[a-zà-ÿ]/iu.test(first) && word.value[1] === "." && word.value.length > 2
|
|
413
|
+
const oneWordQualifier = ["im", "ob", "bei", "unter", "vor"].includes(word.value.toLowerCase())
|
|
414
|
+
|
|
415
|
+
const twoWordQualifier =
|
|
416
|
+
["an", "in"].includes(word.value.toLowerCase()) && words[index + 1]?.value.toLowerCase() === "der"
|
|
417
|
+
|
|
418
|
+
const hasQualifierValue = oneWordQualifier ? index + 1 < words.length : index + 2 < words.length
|
|
419
|
+
|
|
420
|
+
if (abbreviated || ((oneWordQualifier || twoWordQualifier) && hasQualifierValue)) {
|
|
421
|
+
qualifierStart = word.index
|
|
422
|
+
|
|
423
|
+
break
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
const suffix = words.at(-1)
|
|
428
|
+
|
|
429
|
+
if (
|
|
430
|
+
qualifierStart === undefined &&
|
|
431
|
+
suffix &&
|
|
432
|
+
words.length > 1 &&
|
|
433
|
+
(["S", "N", "E", "W", "V", "Ø", "Sø", "Fyn", "Thy", "Sjælland", "Jylland"].includes(suffix.value) ||
|
|
434
|
+
/^[A-ZÅÄÖ]{2}$/u.test(suffix.value))
|
|
435
|
+
) {
|
|
436
|
+
qualifierStart = suffix.index
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
if (qualifierStart !== undefined) {
|
|
440
|
+
s = s.slice(0, qualifierStart).trimEnd()
|
|
441
|
+
}
|
|
442
|
+
|
|
290
443
|
s = s.trim()
|
|
291
444
|
|
|
292
445
|
return s === locality.trim() ? "" : s
|
|
@@ -308,10 +461,87 @@ export function stripLocalityQualifier(locality: string): string {
|
|
|
308
461
|
* is ambiguous (designator unknown) and it stays unfolded — a bare-route query therefore misses rather than guessing a
|
|
309
462
|
* designator.
|
|
310
463
|
*/
|
|
311
|
-
export function canonicalizeRouteKey(streetNorm:
|
|
464
|
+
export function canonicalizeRouteKey(streetNorm: StreetKey): RouteKey {
|
|
312
465
|
const match = /^(us|state|[a-z]{2}) (?:route|rte|rt|highway|hwy) (\d.*)$/.exec(streetNorm)
|
|
313
466
|
|
|
314
|
-
|
|
467
|
+
// A street naming no route is already its own route key. {@link StreetKey} and {@link RouteKey} are
|
|
468
|
+
// SIBLING brands rather than nested ones — precisely so neither column accepts the other's key — so
|
|
469
|
+
// re-minting here is an explicit hop through the unbranded string.
|
|
470
|
+
if (!match) return streetNorm as string as RouteKey
|
|
471
|
+
|
|
472
|
+
return `${match[1] === "us" ? "us" : "state"} route ${match[2]}` as RouteKey
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
/**
|
|
476
|
+
* The canonical street-type words (lowercase) — the VALUE side of the codex suffix table. Membership means "this
|
|
477
|
+
* normalized token is a fully-spelled street type" ("place", "street", "road" …), which is how the doubled-type
|
|
478
|
+
* collapse below recognizes its shape without any parse-tree knowledge.
|
|
479
|
+
*/
|
|
480
|
+
const CANONICAL_TYPE_WORDS: ReadonlySet<string> = new Set(
|
|
481
|
+
[...US_STREET_SUFFIX_LOOKUP.values()].map((v) => v.toLowerCase())
|
|
482
|
+
)
|
|
483
|
+
|
|
484
|
+
/**
|
|
485
|
+
* Ordered lookup-key variants for a US/EN street span — the primary normalized key first, then the null-only recovery
|
|
486
|
+
* forms a reader may probe when the primary misses. Two register mismatches motivate them, both measured on live
|
|
487
|
+
* queries (2026-08-14):
|
|
488
|
+
*
|
|
489
|
+
* - **Doubled type** — a user types the type twice ("Saint Pauls PL St"), and the normalizer canonicalizes only the LAST
|
|
490
|
+
* type token, leaving `saint pauls pl street`, which matches nothing anywhere. The signature is visible in the key
|
|
491
|
+
* itself: a canonical type word in last position DIRECTLY after an uncanonicalized type abbreviation. The variant
|
|
492
|
+
* drops the trailing word and canonicalizes what remains (`saint pauls place`). A street genuinely named with two
|
|
493
|
+
* types keys identically on both sides and is caught by the primary probe first.
|
|
494
|
+
* - **Saint↔St register split** — the artifacts preserve each SOURCE's spelling (NYC situs keys `st pauls place`, Nassau
|
|
495
|
+
* keys `saint pauls place`), and a query arrives in whichever register the user typed. A leading `saint` or `st`
|
|
496
|
+
* token is swapped for its sibling; leading position only — a leading `st` is always the hagionym in US street names,
|
|
497
|
+
* and interior tokens ("Mount Saint Helens Dr") are out of scope until measured.
|
|
498
|
+
*
|
|
499
|
+
* Deduplicated and ordered most-literal-first, so probing the list in order preserves the primary key's precedence.
|
|
500
|
+
*/
|
|
501
|
+
export function streetKeyVariants(street: string, locale: StreetLocale = "us"): StreetKey[] {
|
|
502
|
+
const primary = locale === "us" ? normalizeStreetForKey(street) : normalizeStreetForKeyLocale(street, locale)
|
|
503
|
+
const variants: StreetKey[] = primary ? [primary] : []
|
|
504
|
+
|
|
505
|
+
if (!primary || (locale !== "us" && locale !== "en")) return variants
|
|
506
|
+
|
|
507
|
+
const tokens = primary.split(" ")
|
|
508
|
+
const last = tokens.at(-1)
|
|
509
|
+
const secondLast = tokens.at(-2)
|
|
510
|
+
|
|
511
|
+
if (
|
|
512
|
+
tokens.length > 2 &&
|
|
513
|
+
last &&
|
|
514
|
+
secondLast &&
|
|
515
|
+
CANONICAL_TYPE_WORDS.has(last) &&
|
|
516
|
+
!CANONICAL_TYPE_WORDS.has(secondLast) &&
|
|
517
|
+
US_STREET_SUFFIX_LOOKUP.has(secondLast)
|
|
518
|
+
) {
|
|
519
|
+
const collapsed = normalizeStreetForKey(tokens.slice(0, -1).join(" "))
|
|
520
|
+
|
|
521
|
+
if (collapsed && !variants.includes(collapsed)) {
|
|
522
|
+
variants.push(collapsed)
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
// Snapshot before the swap pass — it appends while reading, and iterating the live array would
|
|
527
|
+
// re-visit its own additions.
|
|
528
|
+
const preSwap = variants.slice()
|
|
529
|
+
|
|
530
|
+
for (const variant of preSwap) {
|
|
531
|
+
const [head, ...rest] = variant.split(" ")
|
|
532
|
+
const swapped = head === "saint" ? "st" : head === "st" ? "saint" : null
|
|
533
|
+
|
|
534
|
+
if (swapped && rest.length) {
|
|
535
|
+
// Substituting a leading hagionym token in an already-normalized key leaves a fold fixed point —
|
|
536
|
+
// the suffix pass never rewrites position 0 — so the swap yields a {@link StreetKey} without
|
|
537
|
+
// re-folding.
|
|
538
|
+
const candidate = [swapped, ...rest].join(" ") as StreetKey
|
|
539
|
+
|
|
540
|
+
if (!variants.includes(candidate)) {
|
|
541
|
+
variants.push(candidate)
|
|
542
|
+
}
|
|
543
|
+
}
|
|
544
|
+
}
|
|
315
545
|
|
|
316
|
-
return
|
|
546
|
+
return variants
|
|
317
547
|
}
|
package/street-segment-schema.ts
CHANGED
|
@@ -17,6 +17,8 @@
|
|
|
17
17
|
|
|
18
18
|
import type { Kysely } from "kysely"
|
|
19
19
|
|
|
20
|
+
import type { RouteKey } from "./street-normalize.ts"
|
|
21
|
+
|
|
20
22
|
/**
|
|
21
23
|
* One TIGER street-segment edge: a `(from_hn, to_hn)` house-number range on one `side` of a named street, with the
|
|
22
24
|
* geometry the interpolator walks. `min_hn`/`max_hn` are the sorted bounds (the probe filters on them); `parity` is
|
|
@@ -24,9 +26,12 @@ import type { Kysely } from "kysely"
|
|
|
24
26
|
*/
|
|
25
27
|
export interface StreetSegmentTable {
|
|
26
28
|
/**
|
|
27
|
-
*
|
|
29
|
+
* `canonicalizeRouteKey(normalizeStreetForKey(street))` — the build/query-consistent probe key. The column NAME says
|
|
30
|
+
* `street_norm`, but the value carries the route fold on top of the street fold, which is why the brand is
|
|
31
|
+
* {@link RouteKey}: builder and probe both apply both folds, and a plain street key bound here misses every
|
|
32
|
+
* numbered-route row.
|
|
28
33
|
*/
|
|
29
|
-
street_norm:
|
|
34
|
+
street_norm: RouteKey
|
|
30
35
|
/**
|
|
31
36
|
* `L` or `R` — the TIGER side the address range sits on.
|
|
32
37
|
*/
|
package/types.ts
CHANGED
|
@@ -39,7 +39,7 @@ export type WOFPlacetype =
|
|
|
39
39
|
* treat it as ordinal, not absolute.
|
|
40
40
|
*
|
|
41
41
|
* `id` is the WOF place id. It's named generically (not `wof_id`) so the shape stays structurally compatible with
|
|
42
|
-
* `@mailwoman/resolver`'s `ResolvedPlace` — `
|
|
42
|
+
* `@mailwoman/resolver`'s `ResolvedPlace` — `WOFSQLitePlaceLookup` satisfies the generic `ResolverBackend` contract
|
|
43
43
|
* without an adapter shim.
|
|
44
44
|
*
|
|
45
45
|
* `distanceKm` is populated only when the query carried `near` (and the place has a centroid). Useful for downstream
|
|
@@ -76,6 +76,38 @@ export interface PlaceCandidate {
|
|
|
76
76
|
* population for ~15% of localities (mostly larger ones). Absent does NOT mean zero, just unknown.
|
|
77
77
|
*/
|
|
78
78
|
population?: number
|
|
79
|
+
/**
|
|
80
|
+
* REFERENTIAL likelihood in [0, 1] — `referentialFromPopulation(population)`, the named form of the prominence key
|
|
81
|
+
* this resolver has always ranked namesakes by (ROAD_TO_V9 §2, ratified 2026-08-06).
|
|
82
|
+
*
|
|
83
|
+
* It is a strictly-increasing function of {@link PlaceCandidate.population} below `REFERENTIAL_SATURATION_POPULATION`
|
|
84
|
+
* and constant above it, so ordering by it — via `compareReferential`, which restores the megacity order with a
|
|
85
|
+
* population tiebreak — is the SAME ORDER as ordering by population. That equivalence is the point: naming the
|
|
86
|
+
* ranking key costs nothing at the ranking.
|
|
87
|
+
*
|
|
88
|
+
* Absent when the candidate has no population on record, exactly as {@link PlaceCandidate.population} is.
|
|
89
|
+
*/
|
|
90
|
+
referential?: number
|
|
91
|
+
/**
|
|
92
|
+
* STRICT encyclopedia-evidence importance in [0, 1], fan-out-guarded per #1497 — CARRIED, NEVER RANKED ON.
|
|
93
|
+
*
|
|
94
|
+
* RESERVED SLOT awaiting a strict-channel source: present only when a shard's `place_importance` table carries the
|
|
95
|
+
* split columns, and no shipped shard does — the FTS lookup's clauses emit NULL for everything today. `undefined`
|
|
96
|
+
* means either "no encyclopedia entry for this place" or "this gazetteer predates the split"; both are absence, and
|
|
97
|
+
* neither is 0. The BLENDED prior the ranking reads is {@link PlaceCandidate.importance}.
|
|
98
|
+
*
|
|
99
|
+
* Saint-Denis is why this is not a ranking key: the Seine-Saint-Denis suburb (pop 96,128) scores 0.1173 while the
|
|
100
|
+
* Aude hamlet (pop 418) scores 0.5683. Consumers that want to display salience read this; the ranking never does.
|
|
101
|
+
*/
|
|
102
|
+
encyclopedic?: number
|
|
103
|
+
/**
|
|
104
|
+
* BLENDED global toponym prior in [0, 1] (#28) — `candidate.importance` surfaced verbatim: the score source's legacy
|
|
105
|
+
* blended importance (encyclopedia-derived where the concordance matched, a population-derived proxy elsewhere).
|
|
106
|
+
* Emitted only by the candidate-table backend, and only when the artifact measured this place — absent is UNMEASURED,
|
|
107
|
+
* never zero. Consumed by `rankByImportance` (`resolver/toponym-prior.ts`) for the bare-toponym class. See
|
|
108
|
+
* `candidate-schema.ts` → `CandidateTable.importance` for why the blend, not the strict channel, is what ships.
|
|
109
|
+
*/
|
|
110
|
+
importance?: number
|
|
79
111
|
/**
|
|
80
112
|
* Bounding box from WOF's `spr.{min,max}_{latitude,longitude}` columns. Coarse outline for the place — a city's bbox
|
|
81
113
|
* is the city's full extent, a postcode's is roughly the postcode polygon's envelope. Optional because not all
|
|
@@ -90,6 +122,14 @@ export interface PlaceCandidate {
|
|
|
90
122
|
* the falsehood-detection differentiator.
|
|
91
123
|
*/
|
|
92
124
|
mismatch?: boolean
|
|
125
|
+
/**
|
|
126
|
+
* Admin-containment stamp (#1717 stage 2) — TRI-STATE, mirroring `ResolvedPlace.containedByQualifier` in
|
|
127
|
+
* `@mailwoman/core`: `true` = the ancestors sidecar vouches this candidate sits under the query's
|
|
128
|
+
* {@link FindPlaceQuery.regionQualifier}; `false` = evaluated and not vouched for; absent = never evaluated (no
|
|
129
|
+
* qualifier on the query, or an artifact without the sidecar). Absence is load-bearing — the resolver walk reads it
|
|
130
|
+
* as `unavailable`, never as "not contained".
|
|
131
|
+
*/
|
|
132
|
+
containedByQualifier?: boolean
|
|
93
133
|
}
|
|
94
134
|
|
|
95
135
|
/**
|
|
@@ -131,6 +171,24 @@ export interface FindPlaceQuery {
|
|
|
131
171
|
* ISO 3166-1 alpha-2 — narrows to one country.
|
|
132
172
|
*/
|
|
133
173
|
country?: string
|
|
174
|
+
/**
|
|
175
|
+
* ISO 3166-1 alpha-2 — narrows the TYPO-FUZZY tier only (#1585). Exact and qualifier-strip probes stay worldwide (a
|
|
176
|
+
* locale hint is a prior, never a hard filter on exact matches), but a typo CORRECTION into a different country's
|
|
177
|
+
* namespace is nearly always a scrape, so the corrected-key probes honor this scope and a scoped-empty ABSTAINS
|
|
178
|
+
* rather than falling through to a world-fuzzy candidate. Ignored when `country` is set (already narrower).
|
|
179
|
+
*/
|
|
180
|
+
fuzzyCountry?: string
|
|
181
|
+
/**
|
|
182
|
+
* Restrict name matching to PRIMARY-keyed rows (#1632) — set by probes whose surface is a RE-READING (a token cut out
|
|
183
|
+
* of a longer classified span), which never named an alias. See the ResolverBackend contract in
|
|
184
|
+
* `@mailwoman/core/resolver`.
|
|
185
|
+
*/
|
|
186
|
+
primaryOnly?: boolean
|
|
187
|
+
/**
|
|
188
|
+
* Alias-row NAME ROLES the probe refuses to answer through (#1730) — the bare-toponym side races pass `abbr`/`gloss`.
|
|
189
|
+
* Role-NULL alias rows (the exonym tier) stay open; backends/artifacts without a role column ignore it.
|
|
190
|
+
*/
|
|
191
|
+
excludeNameRoles?: readonly string[]
|
|
134
192
|
/**
|
|
135
193
|
* WOF place id — narrows to descendants of this place.
|
|
136
194
|
*/
|
|
@@ -142,6 +200,21 @@ export interface FindPlaceQuery {
|
|
|
142
200
|
* postcode_locality shard is present.
|
|
143
201
|
*/
|
|
144
202
|
postcode?: string
|
|
203
|
+
/**
|
|
204
|
+
* Postcode-containment coherence (#31, Mechanism 2) — when true on a locality query that also carries `postcode`,
|
|
205
|
+
* candidate rows within `POSTCODE_CONTAINMENT_GATE_KM` of the postcode's own centroid sort by distance first, the
|
|
206
|
+
* rest appended in their original order. Set by the resolver from `ResolveOpts.postcodeContainmentCoherence`; absent
|
|
207
|
+
* → the population-first order is untouched (byte-identical).
|
|
208
|
+
*/
|
|
209
|
+
postcodeContainmentCoherence?: boolean
|
|
210
|
+
/**
|
|
211
|
+
* The tree's parsed REGION qualifier (#1717 stage 2) — set by the resolver on locality lookups when
|
|
212
|
+
* `ResolveOpts.adminContainmentRerank` is on. See the `ResolverBackend` contract in `@mailwoman/core/resolver`: a
|
|
213
|
+
* capable backend stamps `containedByQualifier`, ranks contained candidates first, and may ADD contained same-key
|
|
214
|
+
* candidates a country scope hid — additive only, never a filter. Ignored on an artifact without the ancestors
|
|
215
|
+
* sidecar (candidates then carry no stamp).
|
|
216
|
+
*/
|
|
217
|
+
regionQualifier?: string
|
|
145
218
|
/**
|
|
146
219
|
* Proximity hint — candidates close to this point get a ranking boost.
|
|
147
220
|
*/
|
package/unified-schema.ts
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* (`scripts/build-unified-wof.ts`). This is the CANONICAL gazetteer — we never use the
|
|
8
8
|
* off-the-shelf geocode.earth prebuilt dumps (they assign different WOF ids to the same place;
|
|
9
9
|
* see the `feedback-custom-wof-db-only` memory). The table/column names match the resolver's
|
|
10
|
-
* expectations (`lookup.ts`) so `
|
|
10
|
+
* expectations (`lookup.ts`) so `WOFSQLitePlaceLookup` works unchanged, INCLUDING the `ancestors`
|
|
11
11
|
* table (which lookup.ts's parent-constraint subquery needs) — see `populateAncestors`. The
|
|
12
12
|
* `place_search` FTS5 + `place_bbox` R*Tree are built separately by `build-fts` (fts.ts).
|
|
13
13
|
*/
|