@mailwoman/resolver-wof-sqlite 7.2.0 → 7.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/address-point-interpolation.ts +207 -0
- package/address-point-schema.ts +107 -0
- package/address-point.ts +122 -0
- package/ancestry-backfill.ts +205 -0
- package/ancestry.ts +70 -0
- package/build-candidate.ts +351 -0
- package/build-slim.ts +394 -0
- package/candidate-fts.ts +43 -0
- package/candidate-lookup.ts +382 -0
- package/candidate-schema.ts +166 -0
- package/coincident-roles.ts +240 -0
- package/convention.ts +152 -0
- package/fst-autocomplete.ts +187 -0
- package/fst-builder.ts +291 -0
- package/fst-deserialize-web.ts +164 -0
- package/fst-matcher.ts +150 -0
- package/fst-serialize.ts +311 -0
- package/fst-types.ts +78 -0
- package/fts.ts +318 -0
- package/geo.ts +140 -0
- package/geonames-aliases.ts +317 -0
- package/geonames-postal.ts +150 -0
- package/index.ts +117 -0
- package/interpolation.ts +232 -0
- package/lookup.ts +1498 -0
- package/package.json +168 -82
- package/poi-lookup.ts +319 -0
- package/poi-schema.ts +147 -0
- package/postal-city-alias-lookup.ts +89 -0
- package/postal-city-alias-schema.ts +75 -0
- package/postal-city-candidate-schema.ts +81 -0
- package/postcode-point-lookup.ts +64 -0
- package/reverse.ts +429 -0
- package/schema.ts +176 -0
- package/sharding.ts +235 -0
- package/sqlite-convention-source.ts +61 -0
- package/sqlite-utils.ts +25 -0
- package/street-centroid-schema.ts +124 -0
- package/street-centroid.ts +124 -0
- package/street-morphology-fst-builder.ts +230 -0
- package/street-name-lookup.ts +101 -0
- package/street-normalize.ts +302 -0
- package/street-segment-schema.ts +104 -0
- package/types.ts +164 -0
- package/unified-schema.ts +171 -0
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* Build a street-morphology FST from libpostal's street_types dictionaries. The morphology FST maps
|
|
7
|
+
* street-typing affixes (Street/Avenue/rue/Calle/Straße/...) to a single synthetic placetype
|
|
8
|
+
* `"street_affix"` — distinct from the admin FST in source data, intent, and binary artifact.
|
|
9
|
+
*
|
|
10
|
+
* The morphology FST closes the inference-time vacuum identified by the v0.6.1 postmortem: street
|
|
11
|
+
* tokens have no admin-FST anchor, so synth-street training pushed the model toward over-emitting
|
|
12
|
+
* `dependent_locality` on subcomponents. With the morphology FST, the neural decoder gets
|
|
13
|
+
* positive evidence for street-typing affixes and the adjacent name tokens, plus negative
|
|
14
|
+
* evidence away from `dependent_locality` on the same neighbours.
|
|
15
|
+
*
|
|
16
|
+
* Design rationale + the four-layer street-supplement architecture lives in
|
|
17
|
+
* `docs/articles/concepts/street-supplement-architecture.md`.
|
|
18
|
+
*
|
|
19
|
+
* Source: `core/data/libpostal/dictionaries/{locale}/street_types.txt`. Each line is pipe-delimited
|
|
20
|
+
* surface forms with the canonical form first: avenue|av|ave|aven|avenu|avn|avnu|avnue
|
|
21
|
+
*
|
|
22
|
+
* Output: an `FSTMatcher` ready to serialize via `serializeFST` to e.g.
|
|
23
|
+
* `fst-street-morphology.bin`.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { readdirSync, readFileSync, statSync } from "node:fs"
|
|
27
|
+
import { join } from "node:path"
|
|
28
|
+
|
|
29
|
+
import type { FSTNode } from "./fst-matcher.ts"
|
|
30
|
+
import { FSTMatcher, normalizeTokens } from "./fst-matcher.ts"
|
|
31
|
+
import type { FSTProvenance, PlaceEntry } from "./fst-types.ts"
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Reserved synthetic wofID base for street-morphology entries. 32-bit unsigned, well above any realistic WOF
|
|
35
|
+
* allocation. Reusing the same base across rebuilds keeps IDs stable for any consumer that caches them. See
|
|
36
|
+
* [[project-schema-storage-decision]] for the reserved range policy.
|
|
37
|
+
*/
|
|
38
|
+
const STREET_AFFIX_WOFID_BASE = 1_900_000_000
|
|
39
|
+
|
|
40
|
+
const STREET_TYPES_FILENAME = "street_types.txt"
|
|
41
|
+
|
|
42
|
+
export interface BuildStreetMorphologyFSTOpts {
|
|
43
|
+
/** Path to the `core/data/libpostal/dictionaries` directory containing per-locale subfolders. */
|
|
44
|
+
dictionariesDir: string
|
|
45
|
+
/**
|
|
46
|
+
* Optional locale filter — only ingest these locale subfolders. Defaults to all that have a `street_types.txt`.
|
|
47
|
+
*/
|
|
48
|
+
locales?: string[]
|
|
49
|
+
/**
|
|
50
|
+
* Minimum length (in characters, post-normalization) of variant surface forms to insert into the trie. Defaults to 3.
|
|
51
|
+
*
|
|
52
|
+
* Rationale: libpostal's street_types dictionaries contain 1-2 character abbreviations (`a`, `b`, `av`, `bd`, `br`,
|
|
53
|
+
* ...) that collide with non-affix tokens at parse time — notably US state abbreviations (`OR`, `CA`, `ND`, `NY`),
|
|
54
|
+
* single-letter unit designators, and arbitrary short tokens. Empirically these collisions push the morphology prior
|
|
55
|
+
* to mis-tag state abbreviations as `street_suffix`. A minimum length of 3 retains useful forms (`ave`, `blvd`,
|
|
56
|
+
* `rue`, `str`) while filtering out the noise.
|
|
57
|
+
*/
|
|
58
|
+
minVariantLength?: number
|
|
59
|
+
/** Optional progress callback. */
|
|
60
|
+
onProgress?: (phase: string, detail?: string) => void
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export interface BuildStreetMorphologyFSTResult {
|
|
64
|
+
matcher: FSTMatcher
|
|
65
|
+
provenance: FSTProvenance
|
|
66
|
+
canonicalCount: number
|
|
67
|
+
variantCount: number
|
|
68
|
+
insertCount: number
|
|
69
|
+
locales: string[]
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Parse one `street_types.txt` line into `{ canonical, variants }`. Canonical is the first token (pre-`|`); variants
|
|
74
|
+
* are all whitespace-stripped non-empty tokens including the canonical.
|
|
75
|
+
*
|
|
76
|
+
* Lines with no `|` are treated as a single-form entry where canonical == variant.
|
|
77
|
+
*/
|
|
78
|
+
function parseLine(line: string): { canonical: string; variants: string[] } | null {
|
|
79
|
+
const trimmed = line.trim()
|
|
80
|
+
|
|
81
|
+
if (trimmed.length === 0 || trimmed.startsWith("#")) return null
|
|
82
|
+
const parts = trimmed
|
|
83
|
+
.split("|")
|
|
84
|
+
.map((s) => s.trim())
|
|
85
|
+
.filter((s) => s.length > 0)
|
|
86
|
+
|
|
87
|
+
if (parts.length === 0) return null
|
|
88
|
+
|
|
89
|
+
return { canonical: parts[0]!, variants: parts }
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export function buildStreetMorphologyFST(opts: BuildStreetMorphologyFSTOpts): BuildStreetMorphologyFSTResult {
|
|
93
|
+
const progress = opts.onProgress ?? (() => {})
|
|
94
|
+
const minVariantLength = opts.minVariantLength ?? 3
|
|
95
|
+
|
|
96
|
+
// Discover locales — either provided explicitly, or all directories containing street_types.txt.
|
|
97
|
+
let locales: string[]
|
|
98
|
+
|
|
99
|
+
if (opts.locales && opts.locales.length > 0) {
|
|
100
|
+
locales = opts.locales
|
|
101
|
+
} else {
|
|
102
|
+
locales = readdirSync(opts.dictionariesDir).filter((entry) => {
|
|
103
|
+
const localePath = join(opts.dictionariesDir, entry)
|
|
104
|
+
|
|
105
|
+
if (!statSync(localePath).isDirectory()) return false
|
|
106
|
+
|
|
107
|
+
try {
|
|
108
|
+
statSync(join(localePath, STREET_TYPES_FILENAME))
|
|
109
|
+
|
|
110
|
+
return true
|
|
111
|
+
} catch {
|
|
112
|
+
return false
|
|
113
|
+
}
|
|
114
|
+
})
|
|
115
|
+
}
|
|
116
|
+
progress("discover", `Found ${locales.length} locales with ${STREET_TYPES_FILENAME}`)
|
|
117
|
+
|
|
118
|
+
// Collect canonical → set-of-variants across all locales. Same canonical form may appear in
|
|
119
|
+
// multiple locales (e.g. "avenue" in en/fr); we union the variant sets.
|
|
120
|
+
const canonicalToVariants = new Map<string, Set<string>>()
|
|
121
|
+
|
|
122
|
+
for (const locale of locales) {
|
|
123
|
+
const filePath = join(opts.dictionariesDir, locale, STREET_TYPES_FILENAME)
|
|
124
|
+
const content = readFileSync(filePath, "utf8")
|
|
125
|
+
|
|
126
|
+
for (const line of content.split("\n")) {
|
|
127
|
+
const parsed = parseLine(line)
|
|
128
|
+
|
|
129
|
+
if (!parsed) continue
|
|
130
|
+
const existing = canonicalToVariants.get(parsed.canonical) ?? new Set<string>()
|
|
131
|
+
|
|
132
|
+
for (const variant of parsed.variants) {
|
|
133
|
+
existing.add(variant)
|
|
134
|
+
}
|
|
135
|
+
canonicalToVariants.set(parsed.canonical, existing)
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
progress("collect", `Collected ${canonicalToVariants.size} canonical affixes`)
|
|
139
|
+
|
|
140
|
+
// Assign stable synthetic wofIDs. Sort canonicals for determinism.
|
|
141
|
+
const sortedCanonicals = [...canonicalToVariants.keys()].sort()
|
|
142
|
+
const canonicalToWOFID = new Map<string, number>()
|
|
143
|
+
|
|
144
|
+
for (let i = 0; i < sortedCanonicals.length; i++) {
|
|
145
|
+
canonicalToWOFID.set(sortedCanonicals[i]!, STREET_AFFIX_WOFID_BASE + i)
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// Build the trie. Each variant is inserted as a token sequence pointing to its canonical's
|
|
149
|
+
// PlaceEntry — so all variants of "avenue" (av/ave/aven/...) lead to the same terminal entry.
|
|
150
|
+
const nodes: FSTNode[] = [{ edges: new Map(), places: [] }]
|
|
151
|
+
|
|
152
|
+
function insertName(tokens: string[], entry: PlaceEntry): void {
|
|
153
|
+
if (tokens.length === 0) return
|
|
154
|
+
let stateID = 0
|
|
155
|
+
|
|
156
|
+
for (const t of tokens) {
|
|
157
|
+
const node = nodes[stateID]!
|
|
158
|
+
let next = node.edges.get(t)
|
|
159
|
+
|
|
160
|
+
if (next === undefined) {
|
|
161
|
+
next = nodes.length
|
|
162
|
+
nodes.push({ edges: new Map(), places: [] })
|
|
163
|
+
node.edges.set(t, next)
|
|
164
|
+
}
|
|
165
|
+
stateID = next
|
|
166
|
+
}
|
|
167
|
+
const existing = nodes[stateID]!.places
|
|
168
|
+
|
|
169
|
+
if (!existing.some((p) => p.wofID === entry.wofID && p.placetype === entry.placetype)) {
|
|
170
|
+
existing.push(entry)
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
let insertCount = 0
|
|
175
|
+
let variantCount = 0
|
|
176
|
+
|
|
177
|
+
for (const canonical of sortedCanonicals) {
|
|
178
|
+
const variants = canonicalToVariants.get(canonical)!
|
|
179
|
+
const wofID = canonicalToWOFID.get(canonical)!
|
|
180
|
+
const entry: PlaceEntry = {
|
|
181
|
+
wofID,
|
|
182
|
+
placetype: "street_affix",
|
|
183
|
+
name: canonical,
|
|
184
|
+
parentChain: [],
|
|
185
|
+
// Fixed importance: street affixes are structurally unambiguous (Avenue is almost never
|
|
186
|
+
// anything but street-typing). The morphology prior caps bias separately; this value
|
|
187
|
+
// just feeds the cap formula `importance * cap`.
|
|
188
|
+
importance: 1.0,
|
|
189
|
+
lat: 0,
|
|
190
|
+
lon: 0,
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
for (const variant of variants) {
|
|
194
|
+
const tokens = normalizeTokens(variant)
|
|
195
|
+
|
|
196
|
+
if (tokens.length === 0) continue
|
|
197
|
+
// Filter out collision-prone short surface forms — see `minVariantLength` docstring.
|
|
198
|
+
// We measure against the joined token form (no spaces) since FST keys are token sequences.
|
|
199
|
+
const joined = tokens.join("")
|
|
200
|
+
|
|
201
|
+
if (joined.length < minVariantLength) continue
|
|
202
|
+
insertName(tokens, entry)
|
|
203
|
+
insertCount++
|
|
204
|
+
variantCount++
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
progress("trie", `Built trie: ${nodes.length} states, ${insertCount} variant insertions`)
|
|
208
|
+
|
|
209
|
+
const edgeCount = nodes.reduce((sum, n) => sum + n.edges.size, 0)
|
|
210
|
+
const matcher = FSTMatcher.fromNodes(nodes)
|
|
211
|
+
const provenance: FSTProvenance = {
|
|
212
|
+
builtAt: new Date().toISOString(),
|
|
213
|
+
countries: locales, // Reuse `countries` slot for locale provenance — semantics differ from admin FST.
|
|
214
|
+
stateCount: nodes.length,
|
|
215
|
+
placeCount: sortedCanonicals.length,
|
|
216
|
+
edgeCount,
|
|
217
|
+
nameInsertions: insertCount,
|
|
218
|
+
importanceMatches: 0, // No importance scoring for morphology — fixed at 1.0.
|
|
219
|
+
sourceDB: opts.dictionariesDir,
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
return {
|
|
223
|
+
matcher,
|
|
224
|
+
provenance,
|
|
225
|
+
canonicalCount: sortedCanonicals.length,
|
|
226
|
+
variantCount,
|
|
227
|
+
insertCount,
|
|
228
|
+
locales,
|
|
229
|
+
}
|
|
230
|
+
}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* #727 stage-2 phase 4c — the SQLite backend for {@link StreetLocalityEvidence}.
|
|
7
|
+
*
|
|
8
|
+
* Reads a street-name index (the FR instance = BAN `street-centroids-fr.db`, a `street_centroid`
|
|
9
|
+
* table of `street_norm × locality_base × postcode` rows) and answers "does this street surface
|
|
10
|
+
* exist as a name" for the k-best rerank. Sync-by-interface, `readOnly`, prepared statements,
|
|
11
|
+
* graceful-degrade on a tableless shard — the same reader discipline as `AddressPointSqliteLookup`.
|
|
12
|
+
*
|
|
13
|
+
* THE FOLD CONTRACT: the surface is folded with {@link foldStreetSurface} (the shared function),
|
|
14
|
+
* and the DB's `street_norm` column MUST have been built with that SAME fold or every hyphenated /
|
|
15
|
+
* apostrophe'd street silently misses. The current `street-centroids-fr.db` predates the contract
|
|
16
|
+
* fold (it folded without hyphen/apostrophe normalization); it must be REBUILT with
|
|
17
|
+
* `foldStreetSurface` + a `street_norm` index before this backend is wired in production. Until
|
|
18
|
+
* then this class is correct-by-construction against a fixture built with the contract fold, and
|
|
19
|
+
* the production rebuild is a tracked BAN-sdk follow-up.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { DatabaseSync } from "node:sqlite"
|
|
23
|
+
|
|
24
|
+
import { foldStreetSurface, type StreetEvidenceScope, type StreetLocalityEvidence } from "@mailwoman/resolver"
|
|
25
|
+
|
|
26
|
+
function hasTable(db: DatabaseSync, table: string): boolean {
|
|
27
|
+
const row = db.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ? LIMIT 1").get(table)
|
|
28
|
+
|
|
29
|
+
return row !== undefined
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function hasColumn(db: DatabaseSync, table: string, column: string): boolean {
|
|
33
|
+
// `table` is a caller-controlled identifier (default `street_centroid`), not user input — safe to interpolate.
|
|
34
|
+
for (const row of db.prepare(`PRAGMA table_info(${table})`).all() as Array<{ name: string }>) {
|
|
35
|
+
if (row.name === column) return true
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
return false
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export interface SQLiteStreetNameLookupOpts {
|
|
42
|
+
/** ISO-2 (upper-case) countries this index answers for. Default `["FR"]` (the BAN street-centroids instance). */
|
|
43
|
+
countries?: Iterable<string>
|
|
44
|
+
/** Table name. Default `street_centroid`. */
|
|
45
|
+
table?: string
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* A {@link StreetLocalityEvidence} backed by a street-name SQLite index. Positive evidence only: any doubt (missing
|
|
50
|
+
* table, read miss) returns `false`, so the rerank fails open to the model's ranking.
|
|
51
|
+
*/
|
|
52
|
+
export class SQLiteStreetNameLookup implements StreetLocalityEvidence {
|
|
53
|
+
readonly countries: ReadonlySet<string>
|
|
54
|
+
readonly #db: DatabaseSync
|
|
55
|
+
readonly #byName: ReturnType<DatabaseSync["prepare"]> | undefined
|
|
56
|
+
readonly #byNameLocality: ReturnType<DatabaseSync["prepare"]> | undefined
|
|
57
|
+
readonly #byNamePostcode: ReturnType<DatabaseSync["prepare"]> | undefined
|
|
58
|
+
|
|
59
|
+
constructor(dbPath: string, opts: SQLiteStreetNameLookupOpts = {}) {
|
|
60
|
+
this.countries = new Set([...(opts.countries ?? ["FR"])].map((c) => c.toUpperCase()))
|
|
61
|
+
this.#db = new DatabaseSync(dbPath, { readOnly: true })
|
|
62
|
+
const table = opts.table ?? "street_centroid"
|
|
63
|
+
|
|
64
|
+
// Degrade gracefully on an empty/tableless shard — a no-op miss, never a crash (#568 discipline).
|
|
65
|
+
if (hasTable(this.#db, table)) {
|
|
66
|
+
// Prefer the #727 phase-4c `name_key` column (foldStreetSurface, indexed by `idx_sc_name` for a direct seek);
|
|
67
|
+
// fall back to `street_norm` on a pre-rebuild shard (a skip-scan, but correct). The fold used to build
|
|
68
|
+
// `name_key` MUST match `foldStreetSurface` here (the fold-parity contract).
|
|
69
|
+
const keyCol = hasColumn(this.#db, table, "name_key") ? "name_key" : "street_norm"
|
|
70
|
+
this.#byName = this.#db.prepare(`SELECT 1 FROM ${table} WHERE ${keyCol} = ? LIMIT 1`)
|
|
71
|
+
this.#byNameLocality = this.#db.prepare(
|
|
72
|
+
`SELECT 1 FROM ${table} WHERE ${keyCol} = ? AND locality_base = ? LIMIT 1`
|
|
73
|
+
)
|
|
74
|
+
this.#byNamePostcode = this.#db.prepare(`SELECT 1 FROM ${table} WHERE ${keyCol} = ? AND postcode = ? LIMIT 1`)
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
hasStreetName(streetSurface: string, scope?: StreetEvidenceScope): boolean {
|
|
79
|
+
if (!this.#byName) return false
|
|
80
|
+
const norm = foldStreetSurface(streetSurface)
|
|
81
|
+
|
|
82
|
+
if (!norm) return false
|
|
83
|
+
|
|
84
|
+
// Scoped lookups tighten precision when the hypothesis carries a locality/postcode; a scoped MISS falls back to the
|
|
85
|
+
// unscoped probe (index incompleteness in the scope column is not evidence of absence — positive-evidence rule).
|
|
86
|
+
if (scope?.locality && this.#byNameLocality) {
|
|
87
|
+
if (this.#byNameLocality.get(norm, foldStreetSurface(scope.locality)) !== undefined) return true
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
if (scope?.postcode && this.#byNamePostcode) {
|
|
91
|
+
if (this.#byNamePostcode.get(norm, scope.postcode) !== undefined) return true
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
return this.#byName.get(norm) !== undefined
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/** Close the underlying handle. */
|
|
98
|
+
close(): void {
|
|
99
|
+
this.#db.close()
|
|
100
|
+
}
|
|
101
|
+
}
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* THE street normalizer for the address-point tier (#476). One function, used by BOTH the shard
|
|
7
|
+
* builder (`scripts/build-address-point-shard.ts`) and the lookup tier (`address-point.ts`) —
|
|
8
|
+
* never two implementations (the PLACETYPE_ORDER lesson: parallel copies silently corrupt).
|
|
9
|
+
*
|
|
10
|
+
* Normalization contract (deliberately aggressive — both sides apply the same function, so
|
|
11
|
+
* collisions only need to be _consistent_, not linguistically perfect):
|
|
12
|
+
*
|
|
13
|
+
* 1. Lowercase, NFKD-fold diacritics, collapse whitespace, strip punctuation (periods, commas,
|
|
14
|
+
* apostrophes).
|
|
15
|
+
* 2. Expand USPS directional abbreviations at the FIRST and LAST token position (`n` → `north`, `se` →
|
|
16
|
+
* `southeast`) — Overture sources abbreviate inconsistently.
|
|
17
|
+
* 3. Canonicalize a trailing USPS street-type token via the codex suffix table to its canonical full
|
|
18
|
+
* form (`st`/`str`/`street` → `street`).
|
|
19
|
+
*
|
|
20
|
+
* Numbered streets are left as digits (`5th` stays `5th`); a SPELLED ordinal before a street suffix
|
|
21
|
+
* folds to its digit form (`tenth street` → `10th street`, #723) so the grid-city ordinal
|
|
22
|
+
* cross-streets the source data spells with digits become reachable.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
import { AbbreviationToDirectional, US_STREET_SUFFIX_LOOKUP } from "@mailwoman/codex/us"
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Spelled ordinal street names → their digit-ordinal form ("tenth" → "10th"), applied ONLY when a street-type suffix
|
|
29
|
+
* follows (#723 admin-tail) — so the ordinal cross-streets common in grid cities ("Tenth Street", "Fifth Avenue") match
|
|
30
|
+
* the shards' digit keys, WITHOUT rewriting ordinal-WORD names where the next token is not a suffix ("First National
|
|
31
|
+
* Bank Rd" stays "first national …"). Digit-source shards are unaffected (a digit token isn't in this map), so the
|
|
32
|
+
* existing keys need no rebuild; a future rebuild folds any spelled-source key the same way (the one-function
|
|
33
|
+
* discipline).
|
|
34
|
+
*/
|
|
35
|
+
const SPELLED_ORDINAL_TO_DIGIT = new Map<string, string>([
|
|
36
|
+
["first", "1st"],
|
|
37
|
+
["second", "2nd"],
|
|
38
|
+
["third", "3rd"],
|
|
39
|
+
["fourth", "4th"],
|
|
40
|
+
["fifth", "5th"],
|
|
41
|
+
["sixth", "6th"],
|
|
42
|
+
["seventh", "7th"],
|
|
43
|
+
["eighth", "8th"],
|
|
44
|
+
["ninth", "9th"],
|
|
45
|
+
["tenth", "10th"],
|
|
46
|
+
["eleventh", "11th"],
|
|
47
|
+
["twelfth", "12th"],
|
|
48
|
+
["thirteenth", "13th"],
|
|
49
|
+
["fourteenth", "14th"],
|
|
50
|
+
["fifteenth", "15th"],
|
|
51
|
+
["sixteenth", "16th"],
|
|
52
|
+
["seventeenth", "17th"],
|
|
53
|
+
["eighteenth", "18th"],
|
|
54
|
+
["nineteenth", "19th"],
|
|
55
|
+
["twentieth", "20th"],
|
|
56
|
+
["thirtieth", "30th"],
|
|
57
|
+
["fortieth", "40th"],
|
|
58
|
+
["fiftieth", "50th"],
|
|
59
|
+
["sixtieth", "60th"],
|
|
60
|
+
["seventieth", "70th"],
|
|
61
|
+
["eightieth", "80th"],
|
|
62
|
+
["ninetieth", "90th"],
|
|
63
|
+
["hundredth", "100th"],
|
|
64
|
+
])
|
|
65
|
+
|
|
66
|
+
/** Lowercase + diacritic-fold + punctuation strip + whitespace collapse. */
|
|
67
|
+
function fold(input: string): string {
|
|
68
|
+
return input
|
|
69
|
+
.normalize("NFKD")
|
|
70
|
+
.replace(/[̀-ͯ]/g, "")
|
|
71
|
+
.toLowerCase()
|
|
72
|
+
.replace(/[.,'’]/g, "")
|
|
73
|
+
.replace(/\s+/g, " ")
|
|
74
|
+
.trim()
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Normalize a street name for address-point keying. Same function at build time and lookup time — see module docstring
|
|
79
|
+
* for the contract.
|
|
80
|
+
*/
|
|
81
|
+
export function normalizeStreetForKey(street: string): string {
|
|
82
|
+
const tokens = fold(street).split(" ")
|
|
83
|
+
|
|
84
|
+
if (tokens.length === 0) return ""
|
|
85
|
+
|
|
86
|
+
// Spelled-ordinal street names → digit form when a street suffix follows ("Tenth Street" →
|
|
87
|
+
// "10th street", #723). Gated on the next token being a suffix so ordinal-WORD names are untouched.
|
|
88
|
+
for (let i = 0; i < tokens.length - 1; i++) {
|
|
89
|
+
const digit = SPELLED_ORDINAL_TO_DIGIT.get(tokens[i]!)
|
|
90
|
+
|
|
91
|
+
if (digit && US_STREET_SUFFIX_LOOKUP.has(tokens[i + 1]!)) {
|
|
92
|
+
tokens[i] = digit
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// Directional expansion at the edges only ("N Main St" / "Main St N" — never interior
|
|
97
|
+
// tokens, where "W" may be an initial in a person-named street). The codex expands
|
|
98
|
+
// compounds to two words ("SE" → "SOUTH EAST"); we key on the spaceless form
|
|
99
|
+
// ("southeast"), and also merge an already-written two-token pair ("South East …").
|
|
100
|
+
const edgeDirectional = (raw: string) =>
|
|
101
|
+
AbbreviationToDirectional.get(raw.toUpperCase())?.toLowerCase().replace(" ", "")
|
|
102
|
+
const mergePair = (a?: string, b?: string) =>
|
|
103
|
+
a && b && /^(north|south)$/.test(a) && /^(east|west)$/.test(b) ? a + b : undefined
|
|
104
|
+
|
|
105
|
+
const leadPair = mergePair(tokens[0], tokens[1])
|
|
106
|
+
|
|
107
|
+
if (leadPair && tokens.length > 2) {
|
|
108
|
+
tokens.splice(0, 2, leadPair)
|
|
109
|
+
}
|
|
110
|
+
const first = edgeDirectional(tokens[0]!)
|
|
111
|
+
|
|
112
|
+
if (first && tokens.length > 1) {
|
|
113
|
+
tokens[0] = first
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
const tailPair = mergePair(tokens[tokens.length - 2], tokens[tokens.length - 1])
|
|
117
|
+
|
|
118
|
+
if (tailPair && tokens.length > 3) {
|
|
119
|
+
tokens.splice(tokens.length - 2, 2, tailPair)
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
if (tokens.length > 2) {
|
|
123
|
+
const last = edgeDirectional(tokens[tokens.length - 1]!)
|
|
124
|
+
|
|
125
|
+
if (last) {
|
|
126
|
+
tokens[tokens.length - 1] = last
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// Street-type canonicalization via the codex table (lowercase keys, UPPER canonical
|
|
131
|
+
// values). The suffix is usually the last token, but sits second-to-last when a trailing
|
|
132
|
+
// directional follows ("Main St N") — check both positions, canonicalize the first hit.
|
|
133
|
+
for (const at of [tokens.length - 1, tokens.length - 2]) {
|
|
134
|
+
if (at < 1) continue // never canonicalize the only/first token ("Street Road" exists)
|
|
135
|
+
const canonical = US_STREET_SUFFIX_LOOKUP.get(tokens[at]!)
|
|
136
|
+
|
|
137
|
+
if (canonical) {
|
|
138
|
+
tokens[at] = canonical.toLowerCase()
|
|
139
|
+
break
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
return tokens.join(" ")
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Street-name locale for the address-point key. The US path is the full USPS pipeline ({@link normalizeStreetForKey});
|
|
148
|
+
* the international paths fold + apply a SMALL, consistent per-locale type-token canonicalization. Same discipline as
|
|
149
|
+
* the US normalizer: build side and probe side call the identical function, so the key only needs to be CONSISTENT, not
|
|
150
|
+
* linguistically perfect — a folded "rue du chevaleret" keys the same on both sides whether or not we reorder the
|
|
151
|
+
* article, so no salient-token / multi-key index is built yet (deferred until probing shows the normalizer can't absorb
|
|
152
|
+
* the false-negatives).
|
|
153
|
+
*/
|
|
154
|
+
export type StreetLocale = "us" | "fr" | "de" | "nl"
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* French street-type abbreviations → canonical full form, applied per token after {@link fold}. French address types
|
|
158
|
+
* LEAD the name ("Av. de…", "Bd …", "Pl. …") and "St"/"Ste" abbreviate Saint/Sainte inside names ("Rue St-Honoré" →
|
|
159
|
+
* "rue saint honore"). fold() has already stripped the trailing period, so the keys are point-free ("av", "bd").
|
|
160
|
+
*/
|
|
161
|
+
const FR_STREET_ABBREV = new Map<string, string>([
|
|
162
|
+
["av", "avenue"],
|
|
163
|
+
["ave", "avenue"],
|
|
164
|
+
["bd", "boulevard"],
|
|
165
|
+
["bld", "boulevard"],
|
|
166
|
+
["bvd", "boulevard"],
|
|
167
|
+
["boul", "boulevard"],
|
|
168
|
+
["pl", "place"],
|
|
169
|
+
["imp", "impasse"],
|
|
170
|
+
["all", "allee"],
|
|
171
|
+
["ch", "chemin"],
|
|
172
|
+
["che", "chemin"],
|
|
173
|
+
["sq", "square"],
|
|
174
|
+
["pas", "passage"],
|
|
175
|
+
["fg", "faubourg"],
|
|
176
|
+
["fbg", "faubourg"],
|
|
177
|
+
["rte", "route"],
|
|
178
|
+
["st", "saint"],
|
|
179
|
+
["ste", "sainte"],
|
|
180
|
+
["sts", "saints"],
|
|
181
|
+
])
|
|
182
|
+
|
|
183
|
+
/**
|
|
184
|
+
* Normalize a street name for the address-point key in a non-US locale. Same function build-side and probe-side (the
|
|
185
|
+
* one-function discipline). US delegates to {@link normalizeStreetForKey}.
|
|
186
|
+
*
|
|
187
|
+
* - **fr** — fold + expand leading type abbreviations and Saint/Sainte (token map).
|
|
188
|
+
* - **de** — fold + ß→ss + canonicalize the GLUED `-str(.)` suffix to `-strasse` ("Lindenstr." → "lindenstrasse",
|
|
189
|
+
* "Lindenstraße" → "lindenstrasse"); an already-full "-strasse" is left intact.
|
|
190
|
+
* - **nl** — fold + canonicalize the glued `-str` suffix to `-straat` ("Kerkstr." → "kerkstraat").
|
|
191
|
+
*/
|
|
192
|
+
export function normalizeStreetForKeyLocale(street: string, locale: StreetLocale): string {
|
|
193
|
+
if (locale === "us") return normalizeStreetForKey(street)
|
|
194
|
+
|
|
195
|
+
// Hyphen → space so a compound name keys the same whether the source or the query writes the
|
|
196
|
+
// hyphen ("Champs-Élysées", "St-Honoré") or a space — both sides fold identically, so this is pure
|
|
197
|
+
// robustness. It also splits a hyphenated abbreviation ("St-Honoré" → "st honore") into tokens the
|
|
198
|
+
// per-locale type/Saint map can see.
|
|
199
|
+
const tokens = fold(street).replace(/ß/g, "ss").replace(/-/g, " ").split(/\s+/).filter(Boolean)
|
|
200
|
+
|
|
201
|
+
if (tokens.length === 0) return ""
|
|
202
|
+
|
|
203
|
+
switch (locale) {
|
|
204
|
+
case "fr":
|
|
205
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
206
|
+
tokens[i] = FR_STREET_ABBREV.get(tokens[i]!) ?? tokens[i]!
|
|
207
|
+
}
|
|
208
|
+
break
|
|
209
|
+
case "de":
|
|
210
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
211
|
+
const t = tokens[i]!
|
|
212
|
+
|
|
213
|
+
if (t.endsWith("str") && !t.endsWith("strasse")) {
|
|
214
|
+
tokens[i] = t.replace(/str$/, "strasse")
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
break
|
|
218
|
+
case "nl":
|
|
219
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
220
|
+
const t = tokens[i]!
|
|
221
|
+
|
|
222
|
+
if (t.endsWith("str") && !t.endsWith("straat")) {
|
|
223
|
+
tokens[i] = t.replace(/str$/, "straat")
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
break
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
return tokens.join(" ")
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/** Normalize a locality name for address-point keying (fold only — no street semantics). */
|
|
233
|
+
export function normalizeLocalityForKey(locality: string): string {
|
|
234
|
+
return fold(locality)
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* Strip a trailing French arrondissement designator from a FOLDED commune key ("paris 8e arrondissement" → "paris",
|
|
239
|
+
* "lyon 1er arrondissement" → "lyon", "marseille 10e arrondissement" → "marseille"). Paris, Lyon and Marseille are the
|
|
240
|
+
* only French communes subdivided into _arrondissements municipaux_; a national register (BAN) names each row per
|
|
241
|
+
* arrondissement, but a query names the base commune ("Place Bellecour, Lyon", never "…, Lyon 2e"). Applied on BOTH
|
|
242
|
+
* sides of the #1042 street-centroid key — build-side (deriving the `locality_base` column) and query-side (folding the
|
|
243
|
+
* probe commune) — so the two agree by construction (the one-function discipline). Input must already be folded
|
|
244
|
+
* (lower-case, diacritic-stripped); a no-op for every other commune. Returns the input unchanged if the strip would
|
|
245
|
+
* empty it.
|
|
246
|
+
*/
|
|
247
|
+
export function stripArrondissement(localityNorm: string): string {
|
|
248
|
+
const stripped = localityNorm.replace(/\s+\d+(?:er|e)\s+arrondissement$/, "").trim()
|
|
249
|
+
|
|
250
|
+
return stripped || localityNorm
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Strip a locality QUALIFIER for a query-side fallback — when an OA locality's exact normalized name misses the
|
|
255
|
+
* gazetteer's canonical name, retry with the qualifier removed. OA address data carries disambiguating qualifiers the
|
|
256
|
+
* gazetteer's canonical name omits: Austrian `Kraubath/Mur` and `Hart b.Graz` → `Hart`; Swiss `Lenk im Simmental` →
|
|
257
|
+
* `Lenk`, `Roche VD` → `Roche`; Danish `Odense S`, `Hurup Thy`. A FALLBACK ONLY — the exact name is tried first, and
|
|
258
|
+
* the region-bbox disambiguation resolves any base-name ambiguity downstream. The candidate table is unchanged (this is
|
|
259
|
+
* purely query-side); feed the result back through {@link normalizeLocalityForKey}. Returns "" when nothing was stripped
|
|
260
|
+
* (no point re-probing the identical key).
|
|
261
|
+
*
|
|
262
|
+
* Measured (`scripts/eval/candidate-recall.ts --strip-fallback`, EU OA holdouts): recovers AT 74.1→88.2% (+14.1pp), DK
|
|
263
|
+
* 91.5→96.2%, CH 90.4→92.6%; +1.3pp overall (diluted by the already-100% locales). Conservative by design — only the
|
|
264
|
+
* qualifier forms above; FI/PT/SI misses are untouched.
|
|
265
|
+
*/
|
|
266
|
+
export function stripLocalityQualifier(locality: string): string {
|
|
267
|
+
let s = locality.trim()
|
|
268
|
+
|
|
269
|
+
if (s.includes("/")) {
|
|
270
|
+
s = s.split("/")[0]!.trim()
|
|
271
|
+
} // "Kraubath/Mur", "St.Kanzian/Klopeiner See"
|
|
272
|
+
s = s.replace(/\s+[a-zà-ÿ]\.\s*\S.*$/iu, "") // abbreviated " b.Graz" / " o.Bleiburg" / " a.d. …"
|
|
273
|
+
s = s.replace(/\s+(im|an der|ob|bei|in der|unter|vor)\s+\S.*$/iu, "") // " im Simmental", " bei Graz"
|
|
274
|
+
s = s.replace(/\s+(S|N|E|W|V|Ø|Sø|Fyn|Thy|Sjælland|Jylland|[A-ZÅÄÖ]{2})$/u, "") // " S", " VD", " Thy"
|
|
275
|
+
s = s.trim()
|
|
276
|
+
|
|
277
|
+
return s === locality.trim() ? "" : s
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
/**
|
|
281
|
+
* Fold numbered-route designators to a canonical key, applied AFTER {@link normalizeStreetForKey}. Sources disagree
|
|
282
|
+
* systematically on how they spell a route: TIGER says `State Rte 100` / `US Hwy 5` where E911/Overture say `VT ROUTE
|
|
283
|
+
* 100` / `US ROUTE 5` — the dominant street-name miss class in the #483 interpolation eval (rural addresses live on
|
|
284
|
+
* routes). `us <designator> N…` folds to `us route N…`; `state <designator> N…` and `<2-letter-prefix> <designator> N…`
|
|
285
|
+
* (the state abbreviation form) fold to `state route N…`. Only digit-leading route numbers fold — `State Street` and
|
|
286
|
+
* friends never match.
|
|
287
|
+
*
|
|
288
|
+
* Used by BOTH the segment-shard builder (`scripts/build-interpolation-shard.ts`) and the interpolation lookup — same
|
|
289
|
+
* one-function discipline as {@link normalizeStreetForKey}. The address-point tier (#476) does NOT apply it yet:
|
|
290
|
+
* adopting it there requires a shard rebuild (noted on #483).
|
|
291
|
+
*
|
|
292
|
+
* A same-numbered US and state route stay DISTINCT keys (`us route 5` vs `state route 5`); only the BARE `route N` form
|
|
293
|
+
* is ambiguous (designator unknown) and it stays unfolded — a bare-route query therefore misses rather than guessing a
|
|
294
|
+
* designator.
|
|
295
|
+
*/
|
|
296
|
+
export function canonicalizeRouteKey(streetNorm: string): string {
|
|
297
|
+
const match = /^(us|state|[a-z]{2}) (?:route|rte|rt|highway|hwy) (\d.*)$/.exec(streetNorm)
|
|
298
|
+
|
|
299
|
+
if (!match) return streetNorm
|
|
300
|
+
|
|
301
|
+
return `${match[1] === "us" ? "us" : "state"} route ${match[2]}`
|
|
302
|
+
}
|