@mailwoman/resolver-wof-sqlite 9.1.0 → 9.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -9
- package/address-point-interpolation.ts +18 -8
- package/address-point-schema.ts +18 -6
- package/address-point.ts +111 -18
- package/ancestry.ts +9 -6
- package/build-candidate.ts +200 -163
- package/build-slim.ts +3 -3
- package/candidate/alias-bags.ts +54 -0
- package/candidate/ancestors-sidecar.ts +206 -0
- package/candidate/country-display-names.ts +79 -0
- package/candidate/name-roles.ts +237 -0
- package/candidate/own-name.ts +146 -0
- package/candidate/place-attrs.ts +44 -0
- package/candidate/shard-fold.ts +137 -0
- package/candidate-ancestors-schema.ts +195 -0
- package/candidate-fts.ts +4 -2
- package/candidate-importance.ts +2 -1
- package/candidate-lookup.ts +439 -186
- package/candidate-schema.ts +33 -5
- package/candidate-scoring.ts +268 -0
- package/capital-schema.ts +90 -0
- package/capitals.ts +148 -0
- package/coincident-roles.ts +69 -10
- package/convention-schema.ts +72 -0
- package/convention.ts +2 -2
- package/coverage-manifest-schema.ts +7 -7
- package/currency-backfill.ts +249 -0
- package/exact-match.ts +104 -0
- package/fst-autocomplete.ts +91 -119
- package/fst-builder.ts +14 -12
- package/fst-freshness.ts +2 -2
- package/fts-query.ts +1 -1
- package/fts.ts +4 -4
- package/geonames-postal.ts +2 -2
- package/index.ts +18 -14
- package/interpolation.ts +113 -19
- package/lookup.ts +110 -591
- package/name-score.ts +6 -4
- package/out/address-point-interpolation.d.ts.map +1 -1
- package/out/address-point-interpolation.js +13 -7
- package/out/address-point-interpolation.js.map +1 -1
- package/out/address-point-schema.d.ts +16 -6
- package/out/address-point-schema.d.ts.map +1 -1
- package/out/address-point-schema.js.map +1 -1
- package/out/address-point.d.ts.map +1 -1
- package/out/address-point.js +70 -14
- package/out/address-point.js.map +1 -1
- package/out/ancestry.d.ts +2 -2
- package/out/ancestry.d.ts.map +1 -1
- package/out/ancestry.js +5 -6
- package/out/ancestry.js.map +1 -1
- package/out/build-candidate.d.ts +75 -0
- package/out/build-candidate.d.ts.map +1 -1
- package/out/build-candidate.js +120 -122
- package/out/build-candidate.js.map +1 -1
- package/out/build-slim.d.ts +1 -1
- package/out/build-slim.js +3 -3
- package/out/build-slim.js.map +1 -1
- package/out/candidate/alias-bags.d.ts +17 -0
- package/out/candidate/alias-bags.d.ts.map +1 -0
- package/out/candidate/alias-bags.js +39 -0
- package/out/candidate/alias-bags.js.map +1 -0
- package/out/candidate/ancestors-sidecar.d.ts +33 -0
- package/out/candidate/ancestors-sidecar.d.ts.map +1 -0
- package/out/candidate/ancestors-sidecar.js +140 -0
- package/out/candidate/ancestors-sidecar.js.map +1 -0
- package/out/candidate/country-display-names.d.ts +35 -0
- package/out/candidate/country-display-names.d.ts.map +1 -0
- package/out/candidate/country-display-names.js +59 -0
- package/out/candidate/country-display-names.js.map +1 -0
- package/out/candidate/name-roles.d.ts +55 -0
- package/out/candidate/name-roles.d.ts.map +1 -0
- package/out/candidate/name-roles.js +165 -0
- package/out/candidate/name-roles.js.map +1 -0
- package/out/candidate/own-name.d.ts +50 -0
- package/out/candidate/own-name.d.ts.map +1 -0
- package/out/candidate/own-name.js +132 -0
- package/out/candidate/own-name.js.map +1 -0
- package/out/candidate/place-attrs.d.ts +43 -0
- package/out/candidate/place-attrs.d.ts.map +1 -0
- package/out/candidate/place-attrs.js +15 -0
- package/out/candidate/place-attrs.js.map +1 -0
- package/out/candidate/shard-fold.d.ts +31 -0
- package/out/candidate/shard-fold.d.ts.map +1 -0
- package/out/candidate/shard-fold.js +104 -0
- package/out/candidate/shard-fold.js.map +1 -0
- package/out/candidate-ancestors-schema.d.ts +150 -0
- package/out/candidate-ancestors-schema.d.ts.map +1 -0
- package/out/candidate-ancestors-schema.js +123 -0
- package/out/candidate-ancestors-schema.js.map +1 -0
- package/out/candidate-fts.d.ts +4 -2
- package/out/candidate-fts.d.ts.map +1 -1
- package/out/candidate-fts.js +4 -2
- package/out/candidate-fts.js.map +1 -1
- package/out/candidate-importance.d.ts.map +1 -1
- package/out/candidate-importance.js +1 -1
- package/out/candidate-importance.js.map +1 -1
- package/out/candidate-lookup.d.ts +22 -45
- package/out/candidate-lookup.d.ts.map +1 -1
- package/out/candidate-lookup.js +340 -135
- package/out/candidate-lookup.js.map +1 -1
- package/out/candidate-schema.d.ts +30 -6
- package/out/candidate-schema.d.ts.map +1 -1
- package/out/candidate-schema.js +3 -0
- package/out/candidate-schema.js.map +1 -1
- package/out/candidate-scoring.d.ts +34 -0
- package/out/candidate-scoring.d.ts.map +1 -0
- package/out/candidate-scoring.js +200 -0
- package/out/candidate-scoring.js.map +1 -0
- package/out/capital-schema.d.ts +51 -0
- package/out/capital-schema.d.ts.map +1 -0
- package/out/capital-schema.js +63 -0
- package/out/capital-schema.js.map +1 -0
- package/out/capitals.d.ts +69 -0
- package/out/capitals.d.ts.map +1 -0
- package/out/capitals.js +98 -0
- package/out/capitals.js.map +1 -0
- package/out/coincident-roles.d.ts +7 -0
- package/out/coincident-roles.d.ts.map +1 -1
- package/out/coincident-roles.js +42 -8
- package/out/coincident-roles.js.map +1 -1
- package/out/convention-schema.d.ts +51 -0
- package/out/convention-schema.d.ts.map +1 -0
- package/out/convention-schema.js +34 -0
- package/out/convention-schema.js.map +1 -0
- package/out/convention.d.ts +1 -1
- package/out/convention.js +2 -2
- package/out/coverage-manifest-schema.js +3 -7
- package/out/coverage-manifest-schema.js.map +1 -1
- package/out/currency-backfill.d.ts +46 -0
- package/out/currency-backfill.d.ts.map +1 -0
- package/out/currency-backfill.js +180 -0
- package/out/currency-backfill.js.map +1 -0
- package/out/exact-match.d.ts +25 -0
- package/out/exact-match.d.ts.map +1 -0
- package/out/exact-match.js +89 -0
- package/out/exact-match.js.map +1 -0
- package/out/fst-autocomplete.d.ts +11 -11
- package/out/fst-autocomplete.d.ts.map +1 -1
- package/out/fst-autocomplete.js +82 -99
- package/out/fst-autocomplete.js.map +1 -1
- package/out/fst-builder.d.ts.map +1 -1
- package/out/fst-builder.js +11 -12
- package/out/fst-builder.js.map +1 -1
- package/out/fst-freshness.d.ts +2 -2
- package/out/fst-freshness.js +2 -2
- package/out/fts-query.js +1 -1
- package/out/fts-query.js.map +1 -1
- package/out/fts.d.ts +4 -4
- package/out/fts.js +4 -4
- package/out/geonames-postal.d.ts +2 -2
- package/out/geonames-postal.js +2 -2
- package/out/index.d.ts +3 -2
- package/out/index.d.ts.map +1 -1
- package/out/index.js +2 -2
- package/out/index.js.map +1 -1
- package/out/interpolation.d.ts +8 -0
- package/out/interpolation.d.ts.map +1 -1
- package/out/interpolation.js +91 -19
- package/out/interpolation.js.map +1 -1
- package/out/lookup.d.ts +4 -5
- package/out/lookup.d.ts.map +1 -1
- package/out/lookup.js +94 -468
- package/out/lookup.js.map +1 -1
- package/out/name-score.d.ts +0 -10
- package/out/name-score.d.ts.map +1 -1
- package/out/name-score.js +6 -4
- package/out/name-score.js.map +1 -1
- package/out/place-importance-schema.d.ts +42 -5
- package/out/place-importance-schema.d.ts.map +1 -1
- package/out/place-importance-schema.js +54 -8
- package/out/place-importance-schema.js.map +1 -1
- package/out/poi-lookup.d.ts +1 -1
- package/out/poi-lookup.d.ts.map +1 -1
- package/out/poi-lookup.js +12 -13
- package/out/poi-lookup.js.map +1 -1
- package/out/poi-schema.d.ts +7 -3
- package/out/poi-schema.d.ts.map +1 -1
- package/out/poi-schema.js.map +1 -1
- package/out/polygon-schema.d.ts +37 -0
- package/out/polygon-schema.d.ts.map +1 -0
- package/out/polygon-schema.js +23 -0
- package/out/polygon-schema.js.map +1 -0
- package/out/postal-city-alias-lookup.d.ts +1 -1
- package/out/postal-city-alias-lookup.js +1 -1
- package/out/postal-city-candidate-schema.d.ts +2 -1
- package/out/postal-city-candidate-schema.d.ts.map +1 -1
- package/out/postal-city-candidate-schema.js.map +1 -1
- package/out/postcode-point-lookup.d.ts +1 -1
- package/out/postcode-point-lookup.js +1 -1
- package/out/primary-preference.d.ts +125 -0
- package/out/primary-preference.d.ts.map +1 -0
- package/out/primary-preference.js +138 -0
- package/out/primary-preference.js.map +1 -0
- package/out/proximity-rerank.d.ts +77 -0
- package/out/proximity-rerank.d.ts.map +1 -0
- package/out/proximity-rerank.js +86 -0
- package/out/proximity-rerank.js.map +1 -0
- package/out/region-keys.d.ts +47 -0
- package/out/region-keys.d.ts.map +1 -0
- package/out/region-keys.js +121 -0
- package/out/region-keys.js.map +1 -0
- package/out/reverse.d.ts.map +1 -1
- package/out/reverse.js +6 -9
- package/out/reverse.js.map +1 -1
- package/out/schema.d.ts +1 -1
- package/out/search-fetch.d.ts +57 -0
- package/out/search-fetch.d.ts.map +1 -0
- package/out/search-fetch.js +183 -0
- package/out/search-fetch.js.map +1 -0
- package/out/sharding.d.ts +3 -3
- package/out/sharding.js +1 -1
- package/out/sqlite-convention-source.d.ts +1 -1
- package/out/sqlite-convention-source.js +1 -1
- package/out/sqlite-utils.d.ts +19 -1
- package/out/sqlite-utils.d.ts.map +1 -1
- package/out/sqlite-utils.js +19 -1
- package/out/sqlite-utils.js.map +1 -1
- package/out/street-centroid-schema.d.ts +7 -2
- package/out/street-centroid-schema.d.ts.map +1 -1
- package/out/street-centroid-schema.js.map +1 -1
- package/out/street-centroid.d.ts.map +1 -1
- package/out/street-centroid.js +7 -7
- package/out/street-centroid.js.map +1 -1
- package/out/street-normalize.d.ts +82 -9
- package/out/street-normalize.d.ts.map +1 -1
- package/out/street-normalize.js +175 -9
- package/out/street-normalize.js.map +1 -1
- package/out/street-segment-schema.d.ts +6 -2
- package/out/street-segment-schema.d.ts.map +1 -1
- package/out/street-segment-schema.js.map +1 -1
- package/out/types.d.ts +35 -1
- package/out/types.d.ts.map +1 -1
- package/out/unified-schema.d.ts +1 -1
- package/out/unified-schema.js +1 -1
- package/out/uprn-lookup.d.ts +85 -0
- package/out/uprn-lookup.d.ts.map +1 -0
- package/out/uprn-lookup.js +152 -0
- package/out/uprn-lookup.js.map +1 -0
- package/out/uprn-schema.d.ts +93 -0
- package/out/uprn-schema.d.ts.map +1 -0
- package/out/uprn-schema.js +78 -0
- package/out/uprn-schema.js.map +1 -0
- package/out/weights-overlay-linker.d.ts +141 -0
- package/out/weights-overlay-linker.d.ts.map +1 -0
- package/out/weights-overlay-linker.js +259 -0
- package/out/weights-overlay-linker.js.map +1 -0
- package/package.json +288 -16
- package/place-importance-schema.ts +64 -15
- package/poi-lookup.ts +12 -13
- package/poi-schema.ts +8 -3
- package/polygon-schema.ts +47 -0
- package/postal-city-alias-lookup.ts +1 -1
- package/postal-city-candidate-schema.ts +3 -1
- package/postcode-point-lookup.ts +1 -1
- package/primary-preference.ts +207 -0
- package/proximity-rerank.ts +120 -0
- package/region-keys.ts +144 -0
- package/reverse.ts +17 -16
- package/schema.ts +1 -1
- package/search-fetch.ts +256 -0
- package/sharding.ts +3 -3
- package/sqlite-convention-source.ts +1 -1
- package/sqlite-utils.ts +43 -2
- package/street-centroid-schema.ts +8 -2
- package/street-centroid.ts +13 -8
- package/street-normalize.ts +252 -23
- package/street-segment-schema.ts +7 -2
- package/types.ts +35 -1
- package/unified-schema.ts +1 -1
- package/uprn-lookup.ts +210 -0
- package/uprn-schema.ts +124 -0
- package/weights-overlay-linker.ts +377 -0
- package/geo.ts +0 -121
- package/out/geo.d.ts +0 -74
- package/out/geo.d.ts.map +0 -1
- package/out/geo.js +0 -71
- package/out/geo.js.map +0 -1
package/fst-autocomplete.ts
CHANGED
|
@@ -3,20 +3,29 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
* FST-based autocomplete
|
|
7
|
-
*
|
|
6
|
+
* FST-based autocomplete — mailwoman vocabulary over `@mailwoman/ancestrie`'s generic algorithm
|
|
7
|
+
* (#1728 phase 2). The #587 behavior — prefix walk + BFS expansion, partial-last-token completion,
|
|
8
|
+
* per-branch capping, dedupe — lives in ancestrie's `autocomplete`; this module contributes only
|
|
9
|
+
* the storage adapter ({@link FSTMatcher} → `AncestrieReaderLike`) and the mapping back to
|
|
10
|
+
* mailwoman's suggestion shape (name, placetype, referential/encyclopedic, WOF ids).
|
|
8
11
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* -
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
* walk the complete prefix, then complete the partial token by prefix-filtering the
|
|
16
|
-
* continuation edges (`token.startsWith(partial)`). This is what a char-level typeahead
|
|
17
|
-
* needs; without it "new yor" returns nothing useful. (#587)
|
|
12
|
+
* THE BYTES DO NOT MIGRATE. The shipped artifacts are `FST\0` v1–v5 (`fst-serialize.ts`), not
|
|
13
|
+
* ancestrie's `ANCT`: ancestrie entries are id-keyed with one record per id, while an FST place row
|
|
14
|
+
* is per-(surface, place) — `crossCountryBranches` is a property of the SURFACE, so the same wofID
|
|
15
|
+
* legitimately carries different values under different aliases and cannot be represented id-keyed.
|
|
16
|
+
* The matcher, both deserializers, and the serializer therefore stay here; what migrated is the
|
|
17
|
+
* ALGORITHM, which is the half that drifts (the #861 share-the-function rule).
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
|
+
import type {
|
|
21
|
+
AncestrieContinuation,
|
|
22
|
+
AncestrieMatch,
|
|
23
|
+
AncestrieReaderLike,
|
|
24
|
+
AncestrieRecord,
|
|
25
|
+
AncestrieSuggestion,
|
|
26
|
+
} from "@mailwoman/ancestrie"
|
|
27
|
+
import { autocomplete as ancestrieAutocomplete } from "@mailwoman/ancestrie"
|
|
28
|
+
|
|
20
29
|
import type { FSTMatcher } from "./fst-matcher.ts"
|
|
21
30
|
import { normalizeTokens } from "./fst-matcher.ts"
|
|
22
31
|
import type { PlaceEntry } from "./fst-types.ts"
|
|
@@ -59,19 +68,15 @@ export interface AutocompleteOpts {
|
|
|
59
68
|
dedupeByName?: boolean
|
|
60
69
|
}
|
|
61
70
|
|
|
62
|
-
interface BfsItem {
|
|
63
|
-
stateID: number
|
|
64
|
-
depth: number
|
|
65
|
-
tokens: string[]
|
|
66
|
-
}
|
|
67
|
-
|
|
68
71
|
/**
|
|
69
72
|
* Max accepting entries collected per BFS branch — keeps one dense branch from starving the search.
|
|
70
73
|
*/
|
|
71
74
|
const PER_BRANCH = 4
|
|
72
75
|
|
|
73
76
|
/**
|
|
74
|
-
* The top-`k` entries by REFERENTIAL likelihood (descending). Avoids sorting/allocating when `entries` is small
|
|
77
|
+
* The top-`k` entries by REFERENTIAL likelihood (descending). Avoids sorting/allocating when `entries` is small — and
|
|
78
|
+
* that shortcut is part of the observable contract: at or under `k` the INSERTION order is served, which decides
|
|
79
|
+
* suggestion order among referential ties.
|
|
75
80
|
*/
|
|
76
81
|
function topByReferential(entries: readonly PlaceEntry[], k: number): PlaceEntry[] {
|
|
77
82
|
if (entries.length <= k) return [...entries]
|
|
@@ -80,125 +85,92 @@ function topByReferential(entries: readonly PlaceEntry[], k: number): PlaceEntry
|
|
|
80
85
|
}
|
|
81
86
|
|
|
82
87
|
/**
|
|
83
|
-
*
|
|
88
|
+
* {@link FSTMatcher} presented through ancestrie's storage seam. Records carry the {@link PlaceEntry} itself as the
|
|
89
|
+
* payload, so the entry that WINS the algorithm's shallowest-depth rule is the entry whose fields the suggestion
|
|
90
|
+
* reports — a side lookup keyed on id could pick a different surface's row (`crossCountryBranches` differs per
|
|
91
|
+
* surface).
|
|
84
92
|
*/
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
const maxExpansionDepth = opts.maxExpansionDepth ?? 2
|
|
88
|
-
const normalizedTokens = normalizeTokens(query)
|
|
89
|
-
|
|
90
|
-
if (!normalizedTokens.length) {
|
|
91
|
-
return { query, normalizedTokens: [], depth: 0, suggestions: [] }
|
|
92
|
-
}
|
|
93
|
-
|
|
94
|
-
const seen = new Map<number, AutocompleteSuggestion>()
|
|
95
|
-
const queue: BfsItem[] = []
|
|
96
|
-
let depth: number
|
|
97
|
-
|
|
98
|
-
const match = fst.walk(normalizedTokens)
|
|
99
|
-
|
|
100
|
-
if (match) {
|
|
101
|
-
// COMPLETE-token prefix landed on a state. Seed at the match state (accepting + continuations).
|
|
102
|
-
depth = match.depth
|
|
103
|
-
|
|
104
|
-
for (const entry of fst.accepting(match.stateID)) {
|
|
105
|
-
addSuggestion(seen, entry, match.depth, [])
|
|
106
|
-
}
|
|
107
|
-
|
|
108
|
-
for (const cont of fst.continuations(match.stateID)) {
|
|
109
|
-
queue.push({ stateID: cont.targetState, depth: 1, tokens: [cont.token] })
|
|
110
|
-
}
|
|
111
|
-
} else {
|
|
112
|
-
// PARTIAL last token — walk the complete prefix, complete the partial by prefix-filtering edges.
|
|
113
|
-
const complete = normalizedTokens.slice(0, -1)
|
|
114
|
-
const partial = normalizedTokens.at(-1)!
|
|
115
|
-
const prefixState = !complete.length ? 0 : (fst.walk(complete)?.stateID ?? undefined)
|
|
116
|
-
|
|
117
|
-
if (prefixState === undefined) {
|
|
118
|
-
return { query, normalizedTokens, depth: 0, suggestions: [] }
|
|
119
|
-
}
|
|
120
|
-
|
|
121
|
-
depth = complete.length
|
|
93
|
+
class FSTReader implements AncestrieReaderLike<PlaceEntry> {
|
|
94
|
+
readonly #fst: FSTMatcher
|
|
122
95
|
|
|
123
|
-
|
|
124
|
-
|
|
96
|
+
/**
|
|
97
|
+
* Parent chains of the entries this reader has served, id-keyed. A place's chain is identical across its surfaces (it
|
|
98
|
+
* is place-row data), so last-write-wins is safe. The algorithm asks {@link FSTReader.ancestorsOf} only for ids it
|
|
99
|
+
* just received from {@link FSTReader.entriesAt}, so serving from this memo answers every real call without an
|
|
100
|
+
* artifact-wide id index.
|
|
101
|
+
*/
|
|
102
|
+
readonly #chains = new Map<number, number[]>()
|
|
125
103
|
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
}
|
|
104
|
+
constructor(fst: FSTMatcher) {
|
|
105
|
+
this.#fst = fst
|
|
106
|
+
}
|
|
130
107
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
}
|
|
108
|
+
walk(tokens: readonly string[]): AncestrieMatch | null {
|
|
109
|
+
return this.#fst.walk([...tokens])
|
|
134
110
|
}
|
|
135
111
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
112
|
+
continuations(stateID: number): AncestrieContinuation[] {
|
|
113
|
+
// Insertion order, verbatim — BFS visit order under the suggestion budget depends on it.
|
|
114
|
+
return this.#fst.continuations(stateID).map((c) => ({
|
|
115
|
+
token: c.token,
|
|
116
|
+
targetState: c.targetState,
|
|
117
|
+
entryCount: c.acceptingCount,
|
|
118
|
+
}))
|
|
119
|
+
}
|
|
142
120
|
|
|
143
|
-
|
|
121
|
+
entriesAt(stateID: number, limit?: number): AncestrieRecord<PlaceEntry>[] {
|
|
122
|
+
const places = this.#fst.accepting(stateID)
|
|
123
|
+
const selected = limit === undefined ? places : topByReferential(places, limit)
|
|
144
124
|
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
}
|
|
125
|
+
return selected.map((entry) => {
|
|
126
|
+
this.#chains.set(entry.wofID, entry.parentChain)
|
|
148
127
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
128
|
+
return {
|
|
129
|
+
id: entry.wofID,
|
|
130
|
+
rank: entry.referential,
|
|
131
|
+
parentIDs: entry.parentChain,
|
|
132
|
+
payload: entry,
|
|
152
133
|
}
|
|
153
|
-
}
|
|
134
|
+
})
|
|
154
135
|
}
|
|
155
136
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
if (opts.dedupeByName) {
|
|
159
|
-
suggestions = dedupeByName(suggestions)
|
|
137
|
+
ancestorsOf(id: number): number[] {
|
|
138
|
+
return this.#chains.get(id) ?? []
|
|
160
139
|
}
|
|
161
|
-
|
|
162
|
-
return { query, normalizedTokens, depth, suggestions: suggestions.slice(0, maxSuggestions) }
|
|
163
|
-
}
|
|
164
|
-
|
|
165
|
-
function addSuggestion(
|
|
166
|
-
seen: Map<number, AutocompleteSuggestion>,
|
|
167
|
-
entry: PlaceEntry,
|
|
168
|
-
matchDepth: number,
|
|
169
|
-
completionTokens: string[]
|
|
170
|
-
): void {
|
|
171
|
-
const existing = seen.get(entry.wofID)
|
|
172
|
-
|
|
173
|
-
if (existing && existing.matchDepth <= matchDepth) return
|
|
174
|
-
|
|
175
|
-
seen.set(entry.wofID, {
|
|
176
|
-
name: entry.name,
|
|
177
|
-
placetype: entry.placetype,
|
|
178
|
-
referential: entry.referential,
|
|
179
|
-
...(entry.encyclopedic === undefined ? {} : { encyclopedic: entry.encyclopedic }),
|
|
180
|
-
wofID: entry.wofID,
|
|
181
|
-
parentChain: entry.parentChain,
|
|
182
|
-
matchDepth,
|
|
183
|
-
completionTokens: [...completionTokens],
|
|
184
|
-
})
|
|
185
140
|
}
|
|
186
141
|
|
|
187
142
|
/**
|
|
188
|
-
*
|
|
189
|
-
* per name wins; order is preserved.
|
|
143
|
+
* Autocomplete from the current prefix. Returns suggestions ranked referential-descending.
|
|
190
144
|
*/
|
|
191
|
-
function
|
|
192
|
-
const
|
|
193
|
-
const out: AutocompleteSuggestion[] = []
|
|
145
|
+
export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteOpts = {}): AutocompleteResult {
|
|
146
|
+
const normalizedTokens = normalizeTokens(query)
|
|
194
147
|
|
|
195
|
-
|
|
196
|
-
|
|
148
|
+
const result = ancestrieAutocomplete<PlaceEntry>(new FSTReader(fst), normalizedTokens, {
|
|
149
|
+
...(opts.maxSuggestions === undefined ? {} : { maxSuggestions: opts.maxSuggestions }),
|
|
150
|
+
...(opts.maxExpansionDepth === undefined ? {} : { maxExpansionDepth: opts.maxExpansionDepth }),
|
|
151
|
+
perBranchLimit: PER_BRANCH,
|
|
152
|
+
// The dedupe key is the DISPLAY name, not the token path: two surfaces of one name must still collapse. (#587)
|
|
153
|
+
...(opts.dedupeByName ? { dedupe: (s: AncestrieSuggestion<PlaceEntry>) => s.payload!.name.toLowerCase() } : {}),
|
|
154
|
+
})
|
|
197
155
|
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
156
|
+
return {
|
|
157
|
+
query,
|
|
158
|
+
normalizedTokens,
|
|
159
|
+
depth: result.depth,
|
|
160
|
+
suggestions: result.suggestions.map((s) => {
|
|
161
|
+
// Every record this adapter serves carries its entry; the assertion documents the invariant.
|
|
162
|
+
const entry = s.payload!
|
|
163
|
+
|
|
164
|
+
return {
|
|
165
|
+
name: entry.name,
|
|
166
|
+
placetype: entry.placetype,
|
|
167
|
+
referential: entry.referential,
|
|
168
|
+
...(entry.encyclopedic === undefined ? {} : { encyclopedic: entry.encyclopedic }),
|
|
169
|
+
wofID: s.id,
|
|
170
|
+
parentChain: s.parentIDs,
|
|
171
|
+
matchDepth: s.matchDepth,
|
|
172
|
+
completionTokens: s.completionTokens,
|
|
173
|
+
}
|
|
174
|
+
}),
|
|
201
175
|
}
|
|
202
|
-
|
|
203
|
-
return out
|
|
204
176
|
}
|
package/fst-builder.ts
CHANGED
|
@@ -18,6 +18,7 @@ import type { FSTNode } from "./fst-matcher.ts"
|
|
|
18
18
|
import { FSTMatcher, normalizeTokens } from "./fst-matcher.ts"
|
|
19
19
|
import type { BuildFSTOpts, BuildFSTResult, FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
|
|
20
20
|
import { loadImportanceSplit } from "./place-importance-schema.ts"
|
|
21
|
+
import { allRows, getRow } from "./sqlite-utils.ts"
|
|
21
22
|
|
|
22
23
|
const DEFAULT_PLACETYPES: PlacetypeID[] = [
|
|
23
24
|
"country",
|
|
@@ -78,7 +79,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
78
79
|
AND placetype IN (${placeholders(placetypes)})`
|
|
79
80
|
)
|
|
80
81
|
|
|
81
|
-
const sprRows = sprStmt
|
|
82
|
+
const sprRows = allRows<SprRow>(sprStmt, ...countries, ...placetypes)
|
|
82
83
|
progress("spr", `Loaded ${sprRows.length} places`)
|
|
83
84
|
|
|
84
85
|
// Phase 2: Build a lookup for parent chain resolution.
|
|
@@ -103,8 +104,8 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
103
104
|
for (let i = 0; i < orphanIDs.length; i += ANCESTOR_CHUNK) {
|
|
104
105
|
const chunk = orphanIDs.slice(i, i + ANCESTOR_CHUNK)
|
|
105
106
|
|
|
106
|
-
const rows =
|
|
107
|
-
.prepare(
|
|
107
|
+
const rows = allRows<{ id: number; ancestor_id: number }>(
|
|
108
|
+
db.prepare(
|
|
108
109
|
`SELECT DISTINCT id, ancestor_id FROM ancestors
|
|
109
110
|
WHERE id IN (${chunk.map(() => "?").join(",")}) AND ancestor_placetype IN ('country', 'region', 'county')
|
|
110
111
|
ORDER BY id, CASE ancestor_placetype
|
|
@@ -112,8 +113,9 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
112
113
|
WHEN 'region' THEN 2
|
|
113
114
|
WHEN 'country' THEN 3
|
|
114
115
|
END`
|
|
115
|
-
)
|
|
116
|
-
|
|
116
|
+
),
|
|
117
|
+
...chunk
|
|
118
|
+
)
|
|
117
119
|
|
|
118
120
|
for (const row of rows) {
|
|
119
121
|
let chain = ancestorsByID.get(row.id)
|
|
@@ -150,7 +152,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
150
152
|
let parentRow = sprByID.get(current)
|
|
151
153
|
|
|
152
154
|
if (!parentRow) {
|
|
153
|
-
const fetched = parentStmt
|
|
155
|
+
const fetched = getRow<SprRow>(parentStmt, current)
|
|
154
156
|
|
|
155
157
|
if (!fetched) break
|
|
156
158
|
parentRow = fetched
|
|
@@ -186,13 +188,13 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
186
188
|
|
|
187
189
|
// Phase 4: Load names for matching places.
|
|
188
190
|
progress("names", "Loading name variants")
|
|
189
|
-
const
|
|
191
|
+
const placeIDs = sprRows.map((r) => r.id)
|
|
190
192
|
const namesByPlace = new Map<number, string[]>()
|
|
191
193
|
|
|
192
194
|
const allLanguages = languages.includes("*")
|
|
193
195
|
|
|
194
|
-
for (let i = 0; i <
|
|
195
|
-
const chunk =
|
|
196
|
+
for (let i = 0; i < placeIDs.length; i += 500) {
|
|
197
|
+
const chunk = placeIDs.slice(i, i + 500)
|
|
196
198
|
const idPlaceholders = chunk.map(() => "?").join(",")
|
|
197
199
|
|
|
198
200
|
const nameStmt = allLanguages
|
|
@@ -201,9 +203,9 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
201
203
|
`SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders}) AND language IN (${languages.map(() => "?").join(",")})`
|
|
202
204
|
)
|
|
203
205
|
|
|
204
|
-
const nameRows =
|
|
205
|
-
? nameStmt
|
|
206
|
-
: nameStmt
|
|
206
|
+
const nameRows = allLanguages
|
|
207
|
+
? allRows<NameRow>(nameStmt, ...chunk)
|
|
208
|
+
: allRows<NameRow>(nameStmt, ...chunk, ...languages)
|
|
207
209
|
|
|
208
210
|
for (const row of nameRows) {
|
|
209
211
|
const existing = namesByPlace.get(row.id) ?? []
|
package/fst-freshness.ts
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* `FSTProvenance` already recorded `sourceDB` — the source's PATH, which is exactly the field that
|
|
16
16
|
* cannot change when the bytes behind it do. So the stamp gains the source's IDENTITY (md5 + byte
|
|
17
17
|
* size) and this module compares it. The shape deliberately mirrors
|
|
18
|
-
*
|
|
18
|
+
* `@mailwoman/resolver-wof-sqlite/weights-overlay-linker`'s `pairIndexStaleReason`: one function returning a reason
|
|
19
19
|
* string or `undefined`, so a fact added to the stamp cannot be checked by some callers and not
|
|
20
20
|
* others — which is how three of the four base linkers ended up unable to notice a PIX1 schema bump.
|
|
21
21
|
*
|
|
@@ -162,7 +162,7 @@ export function peekFSTStampFields(path: string): FSTStampFields | undefined {
|
|
|
162
162
|
*
|
|
163
163
|
* The async `md5File` in `@mailwoman/core/utils` is the one to reach for anywhere else. This exists because the FST
|
|
164
164
|
* builder and its whole call chain are synchronous by design (`buildFSTFromWOF` → `buildLocaleFSTs`), and making them
|
|
165
|
-
* async to stamp a checksum would cascade through
|
|
165
|
+
* async to stamp a checksum would cascade through command callers and tests for one hash. It reads in
|
|
166
166
|
* {@link MD5_CHUNK_BYTES} chunks rather than `readFileSync` — the source is a multi-gigabyte database.
|
|
167
167
|
*/
|
|
168
168
|
export function md5FileSync(path: string): string {
|
package/fts-query.ts
CHANGED
|
@@ -67,7 +67,7 @@ export function sanitizeFTSQuery(text: string, opts?: { fuseTokens?: boolean }):
|
|
|
67
67
|
// fusing "Thiron-Gardais" into the unmatchable single term `ThironGardais` while the FTS
|
|
68
68
|
// doc holds two terms (#945 — the entire hyphenated-name class missed at the raw lookup;
|
|
69
69
|
// masked for years because pre-splice tokenizers never emitted hyphen-preserved values).
|
|
70
|
-
const parts = trimmed.split(/[^\p{L}\p{N}]+/u).filter(
|
|
70
|
+
const parts = trimmed.split(/[^\p{L}\p{N}]+/u).filter((part) => part.length > 0)
|
|
71
71
|
|
|
72
72
|
if (!parts.length) continue
|
|
73
73
|
|
package/fts.ts
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*
|
|
6
6
|
* FTS5 index lifecycle for the WOF SQLite distribution.
|
|
7
7
|
*
|
|
8
|
-
* Shared by `
|
|
8
|
+
* Shared by `WOFSQLitePlaceLookup` (lazy build via `buildFTS: true`) and the operator-side
|
|
9
9
|
* `mailwoman gazetteer build fts` CLI (ahead-of-time build to avoid first-open latency in production).
|
|
10
10
|
*
|
|
11
11
|
* Upstream WOF SQLite distributions do NOT ship FTS5. The index lives in a `place_search` virtual
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
import type { DatabaseSync } from "node:sqlite"
|
|
18
18
|
|
|
19
19
|
/**
|
|
20
|
-
* Name of the FTS5 virtual table this module owns. Centralized so `
|
|
20
|
+
* Name of the FTS5 virtual table this module owns. Centralized so `WOFSQLitePlaceLookup` and the CLI can't drift apart.
|
|
21
21
|
*/
|
|
22
22
|
export const PLACE_SEARCH_TABLE = "place_search"
|
|
23
23
|
|
|
@@ -73,7 +73,7 @@ const ALIAS_SEPARATOR_CODEPOINT = ALIAS_SEPARATOR.codePointAt(0) as number
|
|
|
73
73
|
|
|
74
74
|
/**
|
|
75
75
|
* Does any alias in an `alt_names` bag exactly equal the (already-normalized) query? The single shared implementation
|
|
76
|
-
* of the exact-tier alias check for every consumer of the bag — the Node resolver's `#
|
|
76
|
+
* of the exact-tier alias check for every consumer of the bag — the Node resolver's `#exactMatchIDs` fallback, the WASM
|
|
77
77
|
* resolver, and the demo's httpvfs resolver — so the bag format and its parsers can't drift.
|
|
78
78
|
*
|
|
79
79
|
* Two formats exist in the wild:
|
|
@@ -326,7 +326,7 @@ export function buildPlaceSearchFTS(db: DatabaseSync, opts: BuildPlaceSearchFTSO
|
|
|
326
326
|
}
|
|
327
327
|
|
|
328
328
|
/**
|
|
329
|
-
* Returns true iff the `place_search` table exists in the connected DB. Used by `
|
|
329
|
+
* Returns true iff the `place_search` table exists in the connected DB. Used by `WOFSQLitePlaceLookup` for its "FTS
|
|
330
330
|
* missing — pass buildFTS:true or run the CLI" guard.
|
|
331
331
|
*/
|
|
332
332
|
export function placeSearchFTSExists(db: DatabaseSync): boolean {
|
package/geonames-postal.ts
CHANGED
|
@@ -54,7 +54,7 @@ const GEONAMES_POSTAL_COLUMNS = 11
|
|
|
54
54
|
export const GEONAMES_POSTAL_ID_BASE = 9_500_000_000_000
|
|
55
55
|
|
|
56
56
|
/**
|
|
57
|
-
* The #920 name law: reduce a
|
|
57
|
+
* The #920 name law: reduce a postcode to the sanitized-query token shape — strip every non-letter/number — so the
|
|
58
58
|
* stored name matches what `sanitizeFTSQuery` produces from the parsed postcode token. `"110 00"` → `"11000"`,
|
|
59
59
|
* `"11-041"` → `"11041"`, `"AD500"` → `"AD500"`.
|
|
60
60
|
*/
|
|
@@ -132,7 +132,7 @@ export interface GeonamesPostalIngestResult {
|
|
|
132
132
|
}
|
|
133
133
|
|
|
134
134
|
/**
|
|
135
|
-
* Fold GeoNames
|
|
135
|
+
* Fold GeoNames postcodes for `countries` into an open unified/postcode ingest DB: one `spr` row per distinct
|
|
136
136
|
* normalized postcode (placetype `postalcode`, medoid centroid, degenerate bbox), the normalized form as `name`, and
|
|
137
137
|
* the display form as an extra `names` row when it differs. The caller owns the FTS rebuild (rows ride the standard
|
|
138
138
|
* freeze phase).
|
package/index.ts
CHANGED
|
@@ -19,7 +19,24 @@ export type {
|
|
|
19
19
|
WOFDatabase,
|
|
20
20
|
} from "./schema.ts"
|
|
21
21
|
|
|
22
|
-
export {
|
|
22
|
+
export { WOFSQLitePlaceLookup, type RankingWeights, type WOFSQLitePlaceLookupOpts } from "./lookup.ts"
|
|
23
|
+
|
|
24
|
+
export {
|
|
25
|
+
CANDIDATE_ANCESTOR_COLUMNS,
|
|
26
|
+
CANDIDATE_ANCESTOR_TABLE,
|
|
27
|
+
CANDIDATE_INTERVAL_TABLE,
|
|
28
|
+
createCandidateAncestorTable,
|
|
29
|
+
createCandidateIntervalTable,
|
|
30
|
+
intervalContains,
|
|
31
|
+
MAX_ANCESTOR_DEPTH,
|
|
32
|
+
} from "./candidate-ancestors-schema.ts"
|
|
33
|
+
|
|
34
|
+
export type {
|
|
35
|
+
CandidateAncestorsDatabase,
|
|
36
|
+
CandidateAncestorTable,
|
|
37
|
+
CandidateIntervalTable,
|
|
38
|
+
IntervalLabel,
|
|
39
|
+
} from "./candidate-ancestors-schema.ts"
|
|
23
40
|
|
|
24
41
|
export { CANDIDATE_FTS_TABLE, createCandidateFTS } from "./candidate-fts.ts"
|
|
25
42
|
|
|
@@ -100,19 +117,6 @@ export {
|
|
|
100
117
|
type BuildPlaceSearchFTSResult,
|
|
101
118
|
} from "./fts.ts"
|
|
102
119
|
|
|
103
|
-
export {
|
|
104
|
-
bboxAround,
|
|
105
|
-
geometryContains,
|
|
106
|
-
haversineKm,
|
|
107
|
-
pointInPolygonRings,
|
|
108
|
-
pointInRing,
|
|
109
|
-
type Bbox,
|
|
110
|
-
type GeojsonGeometry,
|
|
111
|
-
type GeojsonMultiPolygon,
|
|
112
|
-
type GeojsonPolygon,
|
|
113
|
-
type GeojsonPosition,
|
|
114
|
-
} from "./geo.ts"
|
|
115
|
-
|
|
116
120
|
export { PLACETYPE_DEPTH, ancestorLineage, placetypeDepth, type AncestorPlaceRow } from "./ancestry.ts"
|
|
117
121
|
|
|
118
122
|
export {
|
package/interpolation.ts
CHANGED
|
@@ -30,11 +30,10 @@ import { DatabaseSync } from "node:sqlite"
|
|
|
30
30
|
|
|
31
31
|
import { parseJSONStrict } from "@mailwoman/core/objects"
|
|
32
32
|
import type { InterpolationLookup } from "@mailwoman/resolver"
|
|
33
|
-
import { clampFraction, pointAlong } from "@mailwoman/spatial"
|
|
33
|
+
import { clampFraction, haversineKm, pointAlong } from "@mailwoman/spatial"
|
|
34
34
|
|
|
35
|
-
import {
|
|
36
|
-
import {
|
|
37
|
-
import { canonicalizeRouteKey, normalizeStreetForKey } from "./street-normalize.ts"
|
|
35
|
+
import { hasTable, prepareAll, type PreparedAll } from "./sqlite-utils.ts"
|
|
36
|
+
import { canonicalizeRouteKey, type RouteKey, streetKeyVariants } from "./street-normalize.ts"
|
|
38
37
|
|
|
39
38
|
/**
|
|
40
39
|
* How an interpolated answer was computed (#483 Method 2):
|
|
@@ -92,8 +91,27 @@ export interface InterpolationQuery {
|
|
|
92
91
|
* ZIP scope — strongly preferred; without it common street names abstain (see module doc).
|
|
93
92
|
*/
|
|
94
93
|
postcode?: string
|
|
94
|
+
/**
|
|
95
|
+
* The resolved locality's coordinate — the tie-breaker when no postcode was given and the parity-preferred covering
|
|
96
|
+
* ranges still span several postcodes. See {@link NEAR_MAX_KM} for the acceptance geometry.
|
|
97
|
+
*/
|
|
98
|
+
near?: { lat: number; lon: number }
|
|
95
99
|
}
|
|
96
100
|
|
|
101
|
+
/**
|
|
102
|
+
* Acceptance geometry for the `near` tie-break: the winning postcode group's closest segment must sit within this many
|
|
103
|
+
* kilometres of `near`, AND the runner-up group must be at least {@link NEAR_DOMINANCE} times farther. Both measured on
|
|
104
|
+
* the two live failures: Brooklyn's `st pauls place` 11226 segment is ~2 km from the Brooklyn centroid with Great
|
|
105
|
+
* Neck's 11021 at ~24 km (12×); Fraser's `east 13 mile road` 48026 is ~2 km with Mecosta's namesake ~190 km away. A
|
|
106
|
+
* near-tie between groups is genuine ambiguity and stays an abstention.
|
|
107
|
+
*/
|
|
108
|
+
const NEAR_MAX_KM = 25
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* See {@link NEAR_MAX_KM}.
|
|
112
|
+
*/
|
|
113
|
+
const NEAR_DOMINANCE = 2
|
|
114
|
+
|
|
97
115
|
interface SegmentRow {
|
|
98
116
|
from_hn: number
|
|
99
117
|
to_hn: number
|
|
@@ -106,11 +124,55 @@ interface SegmentRow {
|
|
|
106
124
|
release: string
|
|
107
125
|
}
|
|
108
126
|
|
|
127
|
+
/**
|
|
128
|
+
* The postcode group nearest `near`, under the {@link NEAR_MAX_KM} dominance geometry — or null when no group qualifies
|
|
129
|
+
* (out of range, or the runner-up is too close to call). A group's distance is its closest segment's first polyline
|
|
130
|
+
* vertex; a segment whose geometry fails to parse prices as unreachable rather than aborting the tie-break.
|
|
131
|
+
*/
|
|
132
|
+
function nearestPostcodeGroup(pool: readonly SegmentRow[], near: { lat: number; lon: number }): SegmentRow[] | null {
|
|
133
|
+
const groups = new Map<string, { rows: SegmentRow[]; km: number }>()
|
|
134
|
+
|
|
135
|
+
for (const row of pool) {
|
|
136
|
+
const key = row.postcode ?? ""
|
|
137
|
+
let km = Number.POSITIVE_INFINITY
|
|
138
|
+
|
|
139
|
+
try {
|
|
140
|
+
const [firstVertex] = parseJSONStrict<[number, number][]>(row.geometry)
|
|
141
|
+
|
|
142
|
+
if (firstVertex) {
|
|
143
|
+
km = haversineKm(near.lat, near.lon, firstVertex[1], firstVertex[0])
|
|
144
|
+
}
|
|
145
|
+
} catch {
|
|
146
|
+
// Unparseable geometry: this row cannot be sited, so it cannot win the tie-break.
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
const group = groups.get(key)
|
|
150
|
+
|
|
151
|
+
if (group) {
|
|
152
|
+
group.rows.push(row)
|
|
153
|
+
group.km = Math.min(group.km, km)
|
|
154
|
+
} else {
|
|
155
|
+
groups.set(key, { rows: [row], km })
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const ranked = [...groups.values()].toSorted((a, b) => a.km - b.km)
|
|
160
|
+
const [winner, runnerUp] = ranked
|
|
161
|
+
|
|
162
|
+
if (!winner || winner.km > NEAR_MAX_KM) return null
|
|
163
|
+
|
|
164
|
+
if (runnerUp && runnerUp.km < winner.km * NEAR_DOMINANCE) return null
|
|
165
|
+
|
|
166
|
+
return winner.rows
|
|
167
|
+
}
|
|
168
|
+
|
|
109
169
|
export class StreetInterpolator implements InterpolationLookup {
|
|
110
170
|
readonly #db: DatabaseSync
|
|
111
171
|
readonly #ownsDB: boolean
|
|
112
|
-
readonly #byPostcode:
|
|
113
|
-
|
|
172
|
+
readonly #byPostcode:
|
|
173
|
+
| PreparedAll<[postcode: string, street: RouteKey, minNumber: number, maxNumber: number], SegmentRow>
|
|
174
|
+
| undefined
|
|
175
|
+
readonly #byStreet: PreparedAll<[street: RouteKey, minNumber: number, maxNumber: number], SegmentRow> | undefined
|
|
114
176
|
readonly #radiusCalibration: number | undefined
|
|
115
177
|
|
|
116
178
|
constructor(opts: { dbPath?: string; database?: DatabaseSync }) {
|
|
@@ -129,12 +191,14 @@ export class StreetInterpolator implements InterpolationLookup {
|
|
|
129
191
|
if (hasTable(this.#db, "street_segment")) {
|
|
130
192
|
const columns = `from_hn, to_hn, min_hn, max_hn, parity, postcode, geometry, source, release`
|
|
131
193
|
|
|
132
|
-
this.#byPostcode =
|
|
194
|
+
this.#byPostcode = prepareAll(
|
|
195
|
+
this.#db,
|
|
133
196
|
`SELECT ${columns} FROM street_segment
|
|
134
197
|
WHERE postcode = ? AND street_norm = ? AND min_hn <= ? AND max_hn >= ?`
|
|
135
198
|
)
|
|
136
199
|
|
|
137
|
-
this.#byStreet =
|
|
200
|
+
this.#byStreet = prepareAll(
|
|
201
|
+
this.#db,
|
|
138
202
|
`SELECT ${columns} FROM street_segment
|
|
139
203
|
WHERE street_norm = ? AND min_hn <= ? AND max_hn >= ?`
|
|
140
204
|
)
|
|
@@ -169,29 +233,41 @@ export class StreetInterpolator implements InterpolationLookup {
|
|
|
169
233
|
|
|
170
234
|
find(query: InterpolationQuery): InterpolatedHit | null {
|
|
171
235
|
if (!this.#byPostcode || !this.#byStreet) return null
|
|
172
|
-
const streetNorm = canonicalizeRouteKey(normalizeStreetForKey(query.street))
|
|
173
236
|
const numberRaw = query.number.trim()
|
|
174
237
|
|
|
175
238
|
// Strictly-numeric house numbers only — this tier estimates, it doesn't guess at
|
|
176
239
|
// hyphenated/alphanumeric schemes the ranges don't model.
|
|
177
|
-
if (
|
|
240
|
+
if (!/^\d+$/.test(numberRaw)) return null
|
|
178
241
|
const n = Number(numberRaw)
|
|
179
242
|
|
|
180
|
-
|
|
243
|
+
// Key-variant ladder (see `streetKeyVariants`): the literal key first, then the doubled-type
|
|
244
|
+
// collapse and the saint↔st register swap. A variant advances the ladder when it produces no
|
|
245
|
+
// ANSWER, not merely no rows — a wrong-register key can cover the number in far-away towns and
|
|
246
|
+
// then fail the ambiguity gate ("saint pauls place" reaches Nassau's rows; the Brooklyn answer
|
|
247
|
+
// lives under "st pauls place"), and stopping at rows would eclipse the right variant.
|
|
248
|
+
for (const variant of streetKeyVariants(query.street)) {
|
|
249
|
+
const streetNorm = canonicalizeRouteKey(variant)
|
|
181
250
|
|
|
182
|
-
if (query.postcode) {
|
|
183
251
|
// A given ZIP that scopes to nothing is a MISS, not a statewide guess: the retry was
|
|
184
252
|
// measured (2026-06-11 VT eval) at +2.3pp coverage for a poisoned tail (p99 1.0 → 20.8
|
|
185
253
|
// km, max 204 km — a unique name statewide can live in a far-away town).
|
|
186
|
-
rows =
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
const
|
|
254
|
+
const rows = query.postcode
|
|
255
|
+
? this.#byPostcode(query.postcode.trim(), streetNorm, n, n)
|
|
256
|
+
: this.#byStreet(streetNorm, n, n)
|
|
257
|
+
|
|
258
|
+
const hit = this.#answerFromRows(rows, n, query)
|
|
191
259
|
|
|
192
|
-
if (
|
|
260
|
+
if (hit) return hit
|
|
193
261
|
}
|
|
194
262
|
|
|
263
|
+
return null
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/**
|
|
267
|
+
* Resolve one key variant's covering rows to an answer, or null when they cannot honestly give one — the
|
|
268
|
+
* parity/ambiguity/tightest-range pipeline the module doc describes.
|
|
269
|
+
*/
|
|
270
|
+
#answerFromRows(rows: SegmentRow[], n: number, query: InterpolationQuery): InterpolatedHit | null {
|
|
195
271
|
if (!rows.length) return null
|
|
196
272
|
|
|
197
273
|
// Parity preference: exact side first, then 'mixed' (matches either), then the
|
|
@@ -200,9 +276,27 @@ export class StreetInterpolator implements InterpolationLookup {
|
|
|
200
276
|
const exact = rows.filter((r) => r.parity === (wantOdd ? "odd" : "even"))
|
|
201
277
|
const mixed = rows.filter((r) => r.parity === "mixed")
|
|
202
278
|
const preferred = exact.length ? exact : mixed
|
|
203
|
-
|
|
279
|
+
let pool = preferred.length ? preferred : rows
|
|
204
280
|
const parityMatched = preferred.length > 0
|
|
205
281
|
|
|
282
|
+
// No scope given: the covering ranges must agree on ONE postcode or the lookup abstains — a
|
|
283
|
+
// name spanning towns is ambiguity, not an answer. Counted over the PARITY pool, not all
|
|
284
|
+
// rows: a section-line boundary road carries a different ZIP per side ("east 13 mile road"
|
|
285
|
+
// is Fraser 48026 odd / Roseville 48066 even), and the opposite side can never hold the
|
|
286
|
+
// number it would otherwise veto. When several postcodes survive parity, the caller's
|
|
287
|
+
// resolved-locality coordinate breaks the tie by segment proximity under the dominance
|
|
288
|
+
// geometry of {@link NEAR_MAX_KM} — a near-tie stays an abstention.
|
|
289
|
+
if (!query.postcode) {
|
|
290
|
+
const postcodes = new Set(pool.map((r) => r.postcode ?? ""))
|
|
291
|
+
|
|
292
|
+
if (postcodes.size > 1) {
|
|
293
|
+
const scoped = query.near ? nearestPostcodeGroup(pool, query.near) : null
|
|
294
|
+
|
|
295
|
+
if (!scoped) return null
|
|
296
|
+
pool = scoped
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
|
|
206
300
|
// Tightest range wins — the most specific claim about where this number lives.
|
|
207
301
|
let best = pool[0]!
|
|
208
302
|
|