@mailwoman/resolver-wof-sqlite 7.2.0 → 7.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/address-point-interpolation.ts +207 -0
  2. package/address-point-schema.ts +107 -0
  3. package/address-point.ts +122 -0
  4. package/ancestry-backfill.ts +205 -0
  5. package/ancestry.ts +70 -0
  6. package/build-candidate.ts +351 -0
  7. package/build-slim.ts +394 -0
  8. package/candidate-fts.ts +43 -0
  9. package/candidate-lookup.ts +382 -0
  10. package/candidate-schema.ts +166 -0
  11. package/coincident-roles.ts +240 -0
  12. package/convention.ts +152 -0
  13. package/fst-autocomplete.ts +187 -0
  14. package/fst-builder.ts +291 -0
  15. package/fst-deserialize-web.ts +164 -0
  16. package/fst-matcher.ts +150 -0
  17. package/fst-serialize.ts +311 -0
  18. package/fst-types.ts +78 -0
  19. package/fts.ts +318 -0
  20. package/geo.ts +140 -0
  21. package/geonames-aliases.ts +317 -0
  22. package/geonames-postal.ts +150 -0
  23. package/index.ts +117 -0
  24. package/interpolation.ts +232 -0
  25. package/lookup.ts +1498 -0
  26. package/package.json +168 -82
  27. package/poi-lookup.ts +319 -0
  28. package/poi-schema.ts +147 -0
  29. package/postal-city-alias-lookup.ts +89 -0
  30. package/postal-city-alias-schema.ts +75 -0
  31. package/postal-city-candidate-schema.ts +81 -0
  32. package/postcode-point-lookup.ts +64 -0
  33. package/reverse.ts +429 -0
  34. package/schema.ts +176 -0
  35. package/sharding.ts +235 -0
  36. package/sqlite-convention-source.ts +61 -0
  37. package/sqlite-utils.ts +25 -0
  38. package/street-centroid-schema.ts +124 -0
  39. package/street-centroid.ts +124 -0
  40. package/street-morphology-fst-builder.ts +230 -0
  41. package/street-name-lookup.ts +101 -0
  42. package/street-normalize.ts +302 -0
  43. package/street-segment-schema.ts +104 -0
  44. package/types.ts +164 -0
  45. package/unified-schema.ts +171 -0
@@ -0,0 +1,230 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Build a street-morphology FST from libpostal's street_types dictionaries. The morphology FST maps
7
+ * street-typing affixes (Street/Avenue/rue/Calle/Straße/...) to a single synthetic placetype
8
+ * `"street_affix"` — distinct from the admin FST in source data, intent, and binary artifact.
9
+ *
10
+ * The morphology FST closes the inference-time vacuum identified by the v0.6.1 postmortem: street
11
+ * tokens have no admin-FST anchor, so synth-street training pushed the model toward over-emitting
12
+ * `dependent_locality` on subcomponents. With the morphology FST, the neural decoder gets
13
+ * positive evidence for street-typing affixes and the adjacent name tokens, plus negative
14
+ * evidence away from `dependent_locality` on the same neighbours.
15
+ *
16
+ * Design rationale + the four-layer street-supplement architecture lives in
17
+ * `docs/articles/concepts/street-supplement-architecture.md`.
18
+ *
19
+ * Source: `core/data/libpostal/dictionaries/{locale}/street_types.txt`. Each line is pipe-delimited
20
+ * surface forms with the canonical form first: avenue|av|ave|aven|avenu|avn|avnu|avnue
21
+ *
22
+ * Output: an `FSTMatcher` ready to serialize via `serializeFST` to e.g.
23
+ * `fst-street-morphology.bin`.
24
+ */
25
+
26
+ import { readdirSync, readFileSync, statSync } from "node:fs"
27
+ import { join } from "node:path"
28
+
29
+ import type { FSTNode } from "./fst-matcher.ts"
30
+ import { FSTMatcher, normalizeTokens } from "./fst-matcher.ts"
31
+ import type { FSTProvenance, PlaceEntry } from "./fst-types.ts"
32
+
33
+ /**
34
+ * Reserved synthetic wofID base for street-morphology entries. 32-bit unsigned, well above any realistic WOF
35
+ * allocation. Reusing the same base across rebuilds keeps IDs stable for any consumer that caches them. See
36
+ * [[project-schema-storage-decision]] for the reserved range policy.
37
+ */
38
+ const STREET_AFFIX_WOFID_BASE = 1_900_000_000
39
+
40
+ const STREET_TYPES_FILENAME = "street_types.txt"
41
+
42
+ export interface BuildStreetMorphologyFSTOpts {
43
+ /** Path to the `core/data/libpostal/dictionaries` directory containing per-locale subfolders. */
44
+ dictionariesDir: string
45
+ /**
46
+ * Optional locale filter — only ingest these locale subfolders. Defaults to all that have a `street_types.txt`.
47
+ */
48
+ locales?: string[]
49
+ /**
50
+ * Minimum length (in characters, post-normalization) of variant surface forms to insert into the trie. Defaults to 3.
51
+ *
52
+ * Rationale: libpostal's street_types dictionaries contain 1-2 character abbreviations (`a`, `b`, `av`, `bd`, `br`,
53
+ * ...) that collide with non-affix tokens at parse time — notably US state abbreviations (`OR`, `CA`, `ND`, `NY`),
54
+ * single-letter unit designators, and arbitrary short tokens. Empirically these collisions push the morphology prior
55
+ * to mis-tag state abbreviations as `street_suffix`. A minimum length of 3 retains useful forms (`ave`, `blvd`,
56
+ * `rue`, `str`) while filtering out the noise.
57
+ */
58
+ minVariantLength?: number
59
+ /** Optional progress callback. */
60
+ onProgress?: (phase: string, detail?: string) => void
61
+ }
62
+
63
+ export interface BuildStreetMorphologyFSTResult {
64
+ matcher: FSTMatcher
65
+ provenance: FSTProvenance
66
+ canonicalCount: number
67
+ variantCount: number
68
+ insertCount: number
69
+ locales: string[]
70
+ }
71
+
72
+ /**
73
+ * Parse one `street_types.txt` line into `{ canonical, variants }`. Canonical is the first token (pre-`|`); variants
74
+ * are all whitespace-stripped non-empty tokens including the canonical.
75
+ *
76
+ * Lines with no `|` are treated as a single-form entry where canonical == variant.
77
+ */
78
+ function parseLine(line: string): { canonical: string; variants: string[] } | null {
79
+ const trimmed = line.trim()
80
+
81
+ if (trimmed.length === 0 || trimmed.startsWith("#")) return null
82
+ const parts = trimmed
83
+ .split("|")
84
+ .map((s) => s.trim())
85
+ .filter((s) => s.length > 0)
86
+
87
+ if (parts.length === 0) return null
88
+
89
+ return { canonical: parts[0]!, variants: parts }
90
+ }
91
+
92
+ export function buildStreetMorphologyFST(opts: BuildStreetMorphologyFSTOpts): BuildStreetMorphologyFSTResult {
93
+ const progress = opts.onProgress ?? (() => {})
94
+ const minVariantLength = opts.minVariantLength ?? 3
95
+
96
+ // Discover locales — either provided explicitly, or all directories containing street_types.txt.
97
+ let locales: string[]
98
+
99
+ if (opts.locales && opts.locales.length > 0) {
100
+ locales = opts.locales
101
+ } else {
102
+ locales = readdirSync(opts.dictionariesDir).filter((entry) => {
103
+ const localePath = join(opts.dictionariesDir, entry)
104
+
105
+ if (!statSync(localePath).isDirectory()) return false
106
+
107
+ try {
108
+ statSync(join(localePath, STREET_TYPES_FILENAME))
109
+
110
+ return true
111
+ } catch {
112
+ return false
113
+ }
114
+ })
115
+ }
116
+ progress("discover", `Found ${locales.length} locales with ${STREET_TYPES_FILENAME}`)
117
+
118
+ // Collect canonical → set-of-variants across all locales. Same canonical form may appear in
119
+ // multiple locales (e.g. "avenue" in en/fr); we union the variant sets.
120
+ const canonicalToVariants = new Map<string, Set<string>>()
121
+
122
+ for (const locale of locales) {
123
+ const filePath = join(opts.dictionariesDir, locale, STREET_TYPES_FILENAME)
124
+ const content = readFileSync(filePath, "utf8")
125
+
126
+ for (const line of content.split("\n")) {
127
+ const parsed = parseLine(line)
128
+
129
+ if (!parsed) continue
130
+ const existing = canonicalToVariants.get(parsed.canonical) ?? new Set<string>()
131
+
132
+ for (const variant of parsed.variants) {
133
+ existing.add(variant)
134
+ }
135
+ canonicalToVariants.set(parsed.canonical, existing)
136
+ }
137
+ }
138
+ progress("collect", `Collected ${canonicalToVariants.size} canonical affixes`)
139
+
140
+ // Assign stable synthetic wofIDs. Sort canonicals for determinism.
141
+ const sortedCanonicals = [...canonicalToVariants.keys()].sort()
142
+ const canonicalToWOFID = new Map<string, number>()
143
+
144
+ for (let i = 0; i < sortedCanonicals.length; i++) {
145
+ canonicalToWOFID.set(sortedCanonicals[i]!, STREET_AFFIX_WOFID_BASE + i)
146
+ }
147
+
148
+ // Build the trie. Each variant is inserted as a token sequence pointing to its canonical's
149
+ // PlaceEntry — so all variants of "avenue" (av/ave/aven/...) lead to the same terminal entry.
150
+ const nodes: FSTNode[] = [{ edges: new Map(), places: [] }]
151
+
152
+ function insertName(tokens: string[], entry: PlaceEntry): void {
153
+ if (tokens.length === 0) return
154
+ let stateID = 0
155
+
156
+ for (const t of tokens) {
157
+ const node = nodes[stateID]!
158
+ let next = node.edges.get(t)
159
+
160
+ if (next === undefined) {
161
+ next = nodes.length
162
+ nodes.push({ edges: new Map(), places: [] })
163
+ node.edges.set(t, next)
164
+ }
165
+ stateID = next
166
+ }
167
+ const existing = nodes[stateID]!.places
168
+
169
+ if (!existing.some((p) => p.wofID === entry.wofID && p.placetype === entry.placetype)) {
170
+ existing.push(entry)
171
+ }
172
+ }
173
+
174
+ let insertCount = 0
175
+ let variantCount = 0
176
+
177
+ for (const canonical of sortedCanonicals) {
178
+ const variants = canonicalToVariants.get(canonical)!
179
+ const wofID = canonicalToWOFID.get(canonical)!
180
+ const entry: PlaceEntry = {
181
+ wofID,
182
+ placetype: "street_affix",
183
+ name: canonical,
184
+ parentChain: [],
185
+ // Fixed importance: street affixes are structurally unambiguous (Avenue is almost never
186
+ // anything but street-typing). The morphology prior caps bias separately; this value
187
+ // just feeds the cap formula `importance * cap`.
188
+ importance: 1.0,
189
+ lat: 0,
190
+ lon: 0,
191
+ }
192
+
193
+ for (const variant of variants) {
194
+ const tokens = normalizeTokens(variant)
195
+
196
+ if (tokens.length === 0) continue
197
+ // Filter out collision-prone short surface forms — see `minVariantLength` docstring.
198
+ // We measure against the joined token form (no spaces) since FST keys are token sequences.
199
+ const joined = tokens.join("")
200
+
201
+ if (joined.length < minVariantLength) continue
202
+ insertName(tokens, entry)
203
+ insertCount++
204
+ variantCount++
205
+ }
206
+ }
207
+ progress("trie", `Built trie: ${nodes.length} states, ${insertCount} variant insertions`)
208
+
209
+ const edgeCount = nodes.reduce((sum, n) => sum + n.edges.size, 0)
210
+ const matcher = FSTMatcher.fromNodes(nodes)
211
+ const provenance: FSTProvenance = {
212
+ builtAt: new Date().toISOString(),
213
+ countries: locales, // Reuse `countries` slot for locale provenance — semantics differ from admin FST.
214
+ stateCount: nodes.length,
215
+ placeCount: sortedCanonicals.length,
216
+ edgeCount,
217
+ nameInsertions: insertCount,
218
+ importanceMatches: 0, // No importance scoring for morphology — fixed at 1.0.
219
+ sourceDB: opts.dictionariesDir,
220
+ }
221
+
222
+ return {
223
+ matcher,
224
+ provenance,
225
+ canonicalCount: sortedCanonicals.length,
226
+ variantCount,
227
+ insertCount,
228
+ locales,
229
+ }
230
+ }
@@ -0,0 +1,101 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * #727 stage-2 phase 4c — the SQLite backend for {@link StreetLocalityEvidence}.
7
+ *
8
+ * Reads a street-name index (the FR instance = BAN `street-centroids-fr.db`, a `street_centroid`
9
+ * table of `street_norm × locality_base × postcode` rows) and answers "does this street surface
10
+ * exist as a name" for the k-best rerank. Sync-by-interface, `readOnly`, prepared statements,
11
+ * graceful-degrade on a tableless shard — the same reader discipline as `AddressPointSqliteLookup`.
12
+ *
13
+ * THE FOLD CONTRACT: the surface is folded with {@link foldStreetSurface} (the shared function),
14
+ * and the DB's `street_norm` column MUST have been built with that SAME fold or every hyphenated /
15
+ * apostrophe'd street silently misses. The current `street-centroids-fr.db` predates the contract
16
+ * fold (it folded without hyphen/apostrophe normalization); it must be REBUILT with
17
+ * `foldStreetSurface` + a `street_norm` index before this backend is wired in production. Until
18
+ * then this class is correct-by-construction against a fixture built with the contract fold, and
19
+ * the production rebuild is a tracked BAN-sdk follow-up.
20
+ */
21
+
22
+ import { DatabaseSync } from "node:sqlite"
23
+
24
+ import { foldStreetSurface, type StreetEvidenceScope, type StreetLocalityEvidence } from "@mailwoman/resolver"
25
+
26
+ function hasTable(db: DatabaseSync, table: string): boolean {
27
+ const row = db.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ? LIMIT 1").get(table)
28
+
29
+ return row !== undefined
30
+ }
31
+
32
+ function hasColumn(db: DatabaseSync, table: string, column: string): boolean {
33
+ // `table` is a caller-controlled identifier (default `street_centroid`), not user input — safe to interpolate.
34
+ for (const row of db.prepare(`PRAGMA table_info(${table})`).all() as Array<{ name: string }>) {
35
+ if (row.name === column) return true
36
+ }
37
+
38
+ return false
39
+ }
40
+
41
+ export interface SQLiteStreetNameLookupOpts {
42
+ /** ISO-2 (upper-case) countries this index answers for. Default `["FR"]` (the BAN street-centroids instance). */
43
+ countries?: Iterable<string>
44
+ /** Table name. Default `street_centroid`. */
45
+ table?: string
46
+ }
47
+
48
+ /**
49
+ * A {@link StreetLocalityEvidence} backed by a street-name SQLite index. Positive evidence only: any doubt (missing
50
+ * table, read miss) returns `false`, so the rerank fails open to the model's ranking.
51
+ */
52
+ export class SQLiteStreetNameLookup implements StreetLocalityEvidence {
53
+ readonly countries: ReadonlySet<string>
54
+ readonly #db: DatabaseSync
55
+ readonly #byName: ReturnType<DatabaseSync["prepare"]> | undefined
56
+ readonly #byNameLocality: ReturnType<DatabaseSync["prepare"]> | undefined
57
+ readonly #byNamePostcode: ReturnType<DatabaseSync["prepare"]> | undefined
58
+
59
+ constructor(dbPath: string, opts: SQLiteStreetNameLookupOpts = {}) {
60
+ this.countries = new Set([...(opts.countries ?? ["FR"])].map((c) => c.toUpperCase()))
61
+ this.#db = new DatabaseSync(dbPath, { readOnly: true })
62
+ const table = opts.table ?? "street_centroid"
63
+
64
+ // Degrade gracefully on an empty/tableless shard — a no-op miss, never a crash (#568 discipline).
65
+ if (hasTable(this.#db, table)) {
66
+ // Prefer the #727 phase-4c `name_key` column (foldStreetSurface, indexed by `idx_sc_name` for a direct seek);
67
+ // fall back to `street_norm` on a pre-rebuild shard (a skip-scan, but correct). The fold used to build
68
+ // `name_key` MUST match `foldStreetSurface` here (the fold-parity contract).
69
+ const keyCol = hasColumn(this.#db, table, "name_key") ? "name_key" : "street_norm"
70
+ this.#byName = this.#db.prepare(`SELECT 1 FROM ${table} WHERE ${keyCol} = ? LIMIT 1`)
71
+ this.#byNameLocality = this.#db.prepare(
72
+ `SELECT 1 FROM ${table} WHERE ${keyCol} = ? AND locality_base = ? LIMIT 1`
73
+ )
74
+ this.#byNamePostcode = this.#db.prepare(`SELECT 1 FROM ${table} WHERE ${keyCol} = ? AND postcode = ? LIMIT 1`)
75
+ }
76
+ }
77
+
78
+ hasStreetName(streetSurface: string, scope?: StreetEvidenceScope): boolean {
79
+ if (!this.#byName) return false
80
+ const norm = foldStreetSurface(streetSurface)
81
+
82
+ if (!norm) return false
83
+
84
+ // Scoped lookups tighten precision when the hypothesis carries a locality/postcode; a scoped MISS falls back to the
85
+ // unscoped probe (index incompleteness in the scope column is not evidence of absence — positive-evidence rule).
86
+ if (scope?.locality && this.#byNameLocality) {
87
+ if (this.#byNameLocality.get(norm, foldStreetSurface(scope.locality)) !== undefined) return true
88
+ }
89
+
90
+ if (scope?.postcode && this.#byNamePostcode) {
91
+ if (this.#byNamePostcode.get(norm, scope.postcode) !== undefined) return true
92
+ }
93
+
94
+ return this.#byName.get(norm) !== undefined
95
+ }
96
+
97
+ /** Close the underlying handle. */
98
+ close(): void {
99
+ this.#db.close()
100
+ }
101
+ }
@@ -0,0 +1,302 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * THE street normalizer for the address-point tier (#476). One function, used by BOTH the shard
7
+ * builder (`scripts/build-address-point-shard.ts`) and the lookup tier (`address-point.ts`) —
8
+ * never two implementations (the PLACETYPE_ORDER lesson: parallel copies silently corrupt).
9
+ *
10
+ * Normalization contract (deliberately aggressive — both sides apply the same function, so
11
+ * collisions only need to be _consistent_, not linguistically perfect):
12
+ *
13
+ * 1. Lowercase, NFKD-fold diacritics, collapse whitespace, strip punctuation (periods, commas,
14
+ * apostrophes).
15
+ * 2. Expand USPS directional abbreviations at the FIRST and LAST token position (`n` → `north`, `se` →
16
+ * `southeast`) — Overture sources abbreviate inconsistently.
17
+ * 3. Canonicalize a trailing USPS street-type token via the codex suffix table to its canonical full
18
+ * form (`st`/`str`/`street` → `street`).
19
+ *
20
+ * Numbered streets are left as digits (`5th` stays `5th`); a SPELLED ordinal before a street suffix
21
+ * folds to its digit form (`tenth street` → `10th street`, #723) so the grid-city ordinal
22
+ * cross-streets the source data spells with digits become reachable.
23
+ */
24
+
25
+ import { AbbreviationToDirectional, US_STREET_SUFFIX_LOOKUP } from "@mailwoman/codex/us"
26
+
27
+ /**
28
+ * Spelled ordinal street names → their digit-ordinal form ("tenth" → "10th"), applied ONLY when a street-type suffix
29
+ * follows (#723 admin-tail) — so the ordinal cross-streets common in grid cities ("Tenth Street", "Fifth Avenue") match
30
+ * the shards' digit keys, WITHOUT rewriting ordinal-WORD names where the next token is not a suffix ("First National
31
+ * Bank Rd" stays "first national …"). Digit-source shards are unaffected (a digit token isn't in this map), so the
32
+ * existing keys need no rebuild; a future rebuild folds any spelled-source key the same way (the one-function
33
+ * discipline).
34
+ */
35
+ const SPELLED_ORDINAL_TO_DIGIT = new Map<string, string>([
36
+ ["first", "1st"],
37
+ ["second", "2nd"],
38
+ ["third", "3rd"],
39
+ ["fourth", "4th"],
40
+ ["fifth", "5th"],
41
+ ["sixth", "6th"],
42
+ ["seventh", "7th"],
43
+ ["eighth", "8th"],
44
+ ["ninth", "9th"],
45
+ ["tenth", "10th"],
46
+ ["eleventh", "11th"],
47
+ ["twelfth", "12th"],
48
+ ["thirteenth", "13th"],
49
+ ["fourteenth", "14th"],
50
+ ["fifteenth", "15th"],
51
+ ["sixteenth", "16th"],
52
+ ["seventeenth", "17th"],
53
+ ["eighteenth", "18th"],
54
+ ["nineteenth", "19th"],
55
+ ["twentieth", "20th"],
56
+ ["thirtieth", "30th"],
57
+ ["fortieth", "40th"],
58
+ ["fiftieth", "50th"],
59
+ ["sixtieth", "60th"],
60
+ ["seventieth", "70th"],
61
+ ["eightieth", "80th"],
62
+ ["ninetieth", "90th"],
63
+ ["hundredth", "100th"],
64
+ ])
65
+
66
+ /** Lowercase + diacritic-fold + punctuation strip + whitespace collapse. */
67
+ function fold(input: string): string {
68
+ return input
69
+ .normalize("NFKD")
70
+ .replace(/[̀-ͯ]/g, "")
71
+ .toLowerCase()
72
+ .replace(/[.,'’]/g, "")
73
+ .replace(/\s+/g, " ")
74
+ .trim()
75
+ }
76
+
77
+ /**
78
+ * Normalize a street name for address-point keying. Same function at build time and lookup time — see module docstring
79
+ * for the contract.
80
+ */
81
+ export function normalizeStreetForKey(street: string): string {
82
+ const tokens = fold(street).split(" ")
83
+
84
+ if (tokens.length === 0) return ""
85
+
86
+ // Spelled-ordinal street names → digit form when a street suffix follows ("Tenth Street" →
87
+ // "10th street", #723). Gated on the next token being a suffix so ordinal-WORD names are untouched.
88
+ for (let i = 0; i < tokens.length - 1; i++) {
89
+ const digit = SPELLED_ORDINAL_TO_DIGIT.get(tokens[i]!)
90
+
91
+ if (digit && US_STREET_SUFFIX_LOOKUP.has(tokens[i + 1]!)) {
92
+ tokens[i] = digit
93
+ }
94
+ }
95
+
96
+ // Directional expansion at the edges only ("N Main St" / "Main St N" — never interior
97
+ // tokens, where "W" may be an initial in a person-named street). The codex expands
98
+ // compounds to two words ("SE" → "SOUTH EAST"); we key on the spaceless form
99
+ // ("southeast"), and also merge an already-written two-token pair ("South East …").
100
+ const edgeDirectional = (raw: string) =>
101
+ AbbreviationToDirectional.get(raw.toUpperCase())?.toLowerCase().replace(" ", "")
102
+ const mergePair = (a?: string, b?: string) =>
103
+ a && b && /^(north|south)$/.test(a) && /^(east|west)$/.test(b) ? a + b : undefined
104
+
105
+ const leadPair = mergePair(tokens[0], tokens[1])
106
+
107
+ if (leadPair && tokens.length > 2) {
108
+ tokens.splice(0, 2, leadPair)
109
+ }
110
+ const first = edgeDirectional(tokens[0]!)
111
+
112
+ if (first && tokens.length > 1) {
113
+ tokens[0] = first
114
+ }
115
+
116
+ const tailPair = mergePair(tokens[tokens.length - 2], tokens[tokens.length - 1])
117
+
118
+ if (tailPair && tokens.length > 3) {
119
+ tokens.splice(tokens.length - 2, 2, tailPair)
120
+ }
121
+
122
+ if (tokens.length > 2) {
123
+ const last = edgeDirectional(tokens[tokens.length - 1]!)
124
+
125
+ if (last) {
126
+ tokens[tokens.length - 1] = last
127
+ }
128
+ }
129
+
130
+ // Street-type canonicalization via the codex table (lowercase keys, UPPER canonical
131
+ // values). The suffix is usually the last token, but sits second-to-last when a trailing
132
+ // directional follows ("Main St N") — check both positions, canonicalize the first hit.
133
+ for (const at of [tokens.length - 1, tokens.length - 2]) {
134
+ if (at < 1) continue // never canonicalize the only/first token ("Street Road" exists)
135
+ const canonical = US_STREET_SUFFIX_LOOKUP.get(tokens[at]!)
136
+
137
+ if (canonical) {
138
+ tokens[at] = canonical.toLowerCase()
139
+ break
140
+ }
141
+ }
142
+
143
+ return tokens.join(" ")
144
+ }
145
+
146
+ /**
147
+ * Street-name locale for the address-point key. The US path is the full USPS pipeline ({@link normalizeStreetForKey});
148
+ * the international paths fold + apply a SMALL, consistent per-locale type-token canonicalization. Same discipline as
149
+ * the US normalizer: build side and probe side call the identical function, so the key only needs to be CONSISTENT, not
150
+ * linguistically perfect — a folded "rue du chevaleret" keys the same on both sides whether or not we reorder the
151
+ * article, so no salient-token / multi-key index is built yet (deferred until probing shows the normalizer can't absorb
152
+ * the false-negatives).
153
+ */
154
+ export type StreetLocale = "us" | "fr" | "de" | "nl"
155
+
156
+ /**
157
+ * French street-type abbreviations → canonical full form, applied per token after {@link fold}. French address types
158
+ * LEAD the name ("Av. de…", "Bd …", "Pl. …") and "St"/"Ste" abbreviate Saint/Sainte inside names ("Rue St-Honoré" →
159
+ * "rue saint honore"). fold() has already stripped the trailing period, so the keys are point-free ("av", "bd").
160
+ */
161
+ const FR_STREET_ABBREV = new Map<string, string>([
162
+ ["av", "avenue"],
163
+ ["ave", "avenue"],
164
+ ["bd", "boulevard"],
165
+ ["bld", "boulevard"],
166
+ ["bvd", "boulevard"],
167
+ ["boul", "boulevard"],
168
+ ["pl", "place"],
169
+ ["imp", "impasse"],
170
+ ["all", "allee"],
171
+ ["ch", "chemin"],
172
+ ["che", "chemin"],
173
+ ["sq", "square"],
174
+ ["pas", "passage"],
175
+ ["fg", "faubourg"],
176
+ ["fbg", "faubourg"],
177
+ ["rte", "route"],
178
+ ["st", "saint"],
179
+ ["ste", "sainte"],
180
+ ["sts", "saints"],
181
+ ])
182
+
183
+ /**
184
+ * Normalize a street name for the address-point key in a non-US locale. Same function build-side and probe-side (the
185
+ * one-function discipline). US delegates to {@link normalizeStreetForKey}.
186
+ *
187
+ * - **fr** — fold + expand leading type abbreviations and Saint/Sainte (token map).
188
+ * - **de** — fold + ß→ss + canonicalize the GLUED `-str(.)` suffix to `-strasse` ("Lindenstr." → "lindenstrasse",
189
+ * "Lindenstraße" → "lindenstrasse"); an already-full "-strasse" is left intact.
190
+ * - **nl** — fold + canonicalize the glued `-str` suffix to `-straat` ("Kerkstr." → "kerkstraat").
191
+ */
192
+ export function normalizeStreetForKeyLocale(street: string, locale: StreetLocale): string {
193
+ if (locale === "us") return normalizeStreetForKey(street)
194
+
195
+ // Hyphen → space so a compound name keys the same whether the source or the query writes the
196
+ // hyphen ("Champs-Élysées", "St-Honoré") or a space — both sides fold identically, so this is pure
197
+ // robustness. It also splits a hyphenated abbreviation ("St-Honoré" → "st honore") into tokens the
198
+ // per-locale type/Saint map can see.
199
+ const tokens = fold(street).replace(/ß/g, "ss").replace(/-/g, " ").split(/\s+/).filter(Boolean)
200
+
201
+ if (tokens.length === 0) return ""
202
+
203
+ switch (locale) {
204
+ case "fr":
205
+ for (let i = 0; i < tokens.length; i++) {
206
+ tokens[i] = FR_STREET_ABBREV.get(tokens[i]!) ?? tokens[i]!
207
+ }
208
+ break
209
+ case "de":
210
+ for (let i = 0; i < tokens.length; i++) {
211
+ const t = tokens[i]!
212
+
213
+ if (t.endsWith("str") && !t.endsWith("strasse")) {
214
+ tokens[i] = t.replace(/str$/, "strasse")
215
+ }
216
+ }
217
+ break
218
+ case "nl":
219
+ for (let i = 0; i < tokens.length; i++) {
220
+ const t = tokens[i]!
221
+
222
+ if (t.endsWith("str") && !t.endsWith("straat")) {
223
+ tokens[i] = t.replace(/str$/, "straat")
224
+ }
225
+ }
226
+ break
227
+ }
228
+
229
+ return tokens.join(" ")
230
+ }
231
+
232
+ /** Normalize a locality name for address-point keying (fold only — no street semantics). */
233
+ export function normalizeLocalityForKey(locality: string): string {
234
+ return fold(locality)
235
+ }
236
+
237
+ /**
238
+ * Strip a trailing French arrondissement designator from a FOLDED commune key ("paris 8e arrondissement" → "paris",
239
+ * "lyon 1er arrondissement" → "lyon", "marseille 10e arrondissement" → "marseille"). Paris, Lyon and Marseille are the
240
+ * only French communes subdivided into _arrondissements municipaux_; a national register (BAN) names each row per
241
+ * arrondissement, but a query names the base commune ("Place Bellecour, Lyon", never "…, Lyon 2e"). Applied on BOTH
242
+ * sides of the #1042 street-centroid key — build-side (deriving the `locality_base` column) and query-side (folding the
243
+ * probe commune) — so the two agree by construction (the one-function discipline). Input must already be folded
244
+ * (lower-case, diacritic-stripped); a no-op for every other commune. Returns the input unchanged if the strip would
245
+ * empty it.
246
+ */
247
+ export function stripArrondissement(localityNorm: string): string {
248
+ const stripped = localityNorm.replace(/\s+\d+(?:er|e)\s+arrondissement$/, "").trim()
249
+
250
+ return stripped || localityNorm
251
+ }
252
+
253
+ /**
254
+ * Strip a locality QUALIFIER for a query-side fallback — when an OA locality's exact normalized name misses the
255
+ * gazetteer's canonical name, retry with the qualifier removed. OA address data carries disambiguating qualifiers the
256
+ * gazetteer's canonical name omits: Austrian `Kraubath/Mur` and `Hart b.Graz` → `Hart`; Swiss `Lenk im Simmental` →
257
+ * `Lenk`, `Roche VD` → `Roche`; Danish `Odense S`, `Hurup Thy`. A FALLBACK ONLY — the exact name is tried first, and
258
+ * the region-bbox disambiguation resolves any base-name ambiguity downstream. The candidate table is unchanged (this is
259
+ * purely query-side); feed the result back through {@link normalizeLocalityForKey}. Returns "" when nothing was stripped
260
+ * (no point re-probing the identical key).
261
+ *
262
+ * Measured (`scripts/eval/candidate-recall.ts --strip-fallback`, EU OA holdouts): recovers AT 74.1→88.2% (+14.1pp), DK
263
+ * 91.5→96.2%, CH 90.4→92.6%; +1.3pp overall (diluted by the already-100% locales). Conservative by design — only the
264
+ * qualifier forms above; FI/PT/SI misses are untouched.
265
+ */
266
+ export function stripLocalityQualifier(locality: string): string {
267
+ let s = locality.trim()
268
+
269
+ if (s.includes("/")) {
270
+ s = s.split("/")[0]!.trim()
271
+ } // "Kraubath/Mur", "St.Kanzian/Klopeiner See"
272
+ s = s.replace(/\s+[a-zà-ÿ]\.\s*\S.*$/iu, "") // abbreviated " b.Graz" / " o.Bleiburg" / " a.d. …"
273
+ s = s.replace(/\s+(im|an der|ob|bei|in der|unter|vor)\s+\S.*$/iu, "") // " im Simmental", " bei Graz"
274
+ s = s.replace(/\s+(S|N|E|W|V|Ø|Sø|Fyn|Thy|Sjælland|Jylland|[A-ZÅÄÖ]{2})$/u, "") // " S", " VD", " Thy"
275
+ s = s.trim()
276
+
277
+ return s === locality.trim() ? "" : s
278
+ }
279
+
280
+ /**
281
+ * Fold numbered-route designators to a canonical key, applied AFTER {@link normalizeStreetForKey}. Sources disagree
282
+ * systematically on how they spell a route: TIGER says `State Rte 100` / `US Hwy 5` where E911/Overture say `VT ROUTE
283
+ * 100` / `US ROUTE 5` — the dominant street-name miss class in the #483 interpolation eval (rural addresses live on
284
+ * routes). `us <designator> N…` folds to `us route N…`; `state <designator> N…` and `<2-letter-prefix> <designator> N…`
285
+ * (the state abbreviation form) fold to `state route N…`. Only digit-leading route numbers fold — `State Street` and
286
+ * friends never match.
287
+ *
288
+ * Used by BOTH the segment-shard builder (`scripts/build-interpolation-shard.ts`) and the interpolation lookup — same
289
+ * one-function discipline as {@link normalizeStreetForKey}. The address-point tier (#476) does NOT apply it yet:
290
+ * adopting it there requires a shard rebuild (noted on #483).
291
+ *
292
+ * A same-numbered US and state route stay DISTINCT keys (`us route 5` vs `state route 5`); only the BARE `route N` form
293
+ * is ambiguous (designator unknown) and it stays unfolded — a bare-route query therefore misses rather than guessing a
294
+ * designator.
295
+ */
296
+ export function canonicalizeRouteKey(streetNorm: string): string {
297
+ const match = /^(us|state|[a-z]{2}) (?:route|rte|rt|highway|hwy) (\d.*)$/.exec(streetNorm)
298
+
299
+ if (!match) return streetNorm
300
+
301
+ return `${match[1] === "us" ? "us" : "state"} route ${match[2]}`
302
+ }