@mailwoman/resolver-wof-sqlite 8.0.0 → 8.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (219) hide show
  1. package/address-point-interpolation.ts +9 -3
  2. package/address-point-schema.ts +32 -10
  3. package/address-point.ts +3 -0
  4. package/ancestry-backfill.ts +18 -5
  5. package/ancestry.ts +7 -2
  6. package/build-candidate.ts +29 -6
  7. package/build-slim.ts +40 -11
  8. package/candidate-fts.ts +1 -0
  9. package/candidate-lookup.ts +49 -15
  10. package/candidate-schema.ts +35 -11
  11. package/coincident-roles.ts +28 -6
  12. package/convention.ts +3 -1
  13. package/coverage-manifest-schema.ts +243 -0
  14. package/fst-autocomplete.ts +15 -9
  15. package/fst-builder.ts +65 -9
  16. package/fst-deserialize-web.ts +46 -9
  17. package/fst-matcher.ts +12 -5
  18. package/fst-serialize.ts +83 -9
  19. package/fst-types.ts +43 -0
  20. package/fts-query.ts +84 -0
  21. package/fts.ts +35 -9
  22. package/geo.ts +9 -3
  23. package/geonames-aliases.ts +116 -79
  24. package/geonames-postal.ts +25 -5
  25. package/index.ts +19 -0
  26. package/interpolation.ts +59 -55
  27. package/lookup.ts +103 -292
  28. package/name-score.ts +76 -0
  29. package/out/address-point-interpolation.d.ts.map +1 -1
  30. package/out/address-point-interpolation.js +4 -2
  31. package/out/address-point-interpolation.js.map +1 -1
  32. package/out/address-point-schema.d.ts +30 -10
  33. package/out/address-point-schema.d.ts.map +1 -1
  34. package/out/address-point-schema.js +6 -2
  35. package/out/address-point-schema.js.map +1 -1
  36. package/out/address-point.d.ts.map +1 -1
  37. package/out/address-point.js.map +1 -1
  38. package/out/ancestry-backfill.d.ts +6 -2
  39. package/out/ancestry-backfill.d.ts.map +1 -1
  40. package/out/ancestry-backfill.js +7 -3
  41. package/out/ancestry-backfill.js.map +1 -1
  42. package/out/ancestry.d.ts +6 -2
  43. package/out/ancestry.d.ts.map +1 -1
  44. package/out/ancestry.js +3 -1
  45. package/out/ancestry.js.map +1 -1
  46. package/out/build-candidate.d.ts +9 -3
  47. package/out/build-candidate.d.ts.map +1 -1
  48. package/out/build-candidate.js +5 -3
  49. package/out/build-candidate.js.map +1 -1
  50. package/out/build-slim.d.ts +15 -5
  51. package/out/build-slim.d.ts.map +1 -1
  52. package/out/build-slim.js +11 -5
  53. package/out/build-slim.js.map +1 -1
  54. package/out/candidate-fts.d.ts.map +1 -1
  55. package/out/candidate-fts.js.map +1 -1
  56. package/out/candidate-lookup.d.ts +16 -3
  57. package/out/candidate-lookup.d.ts.map +1 -1
  58. package/out/candidate-lookup.js +27 -11
  59. package/out/candidate-lookup.js.map +1 -1
  60. package/out/candidate-schema.d.ts +33 -11
  61. package/out/candidate-schema.d.ts.map +1 -1
  62. package/out/candidate-schema.js.map +1 -1
  63. package/out/coincident-roles.d.ts +16 -4
  64. package/out/coincident-roles.d.ts.map +1 -1
  65. package/out/coincident-roles.js +9 -3
  66. package/out/coincident-roles.js.map +1 -1
  67. package/out/convention.d.ts +3 -1
  68. package/out/convention.d.ts.map +1 -1
  69. package/out/convention.js.map +1 -1
  70. package/out/coverage-manifest-schema.d.ts +112 -0
  71. package/out/coverage-manifest-schema.d.ts.map +1 -0
  72. package/out/coverage-manifest-schema.js +154 -0
  73. package/out/coverage-manifest-schema.js.map +1 -0
  74. package/out/fst-autocomplete.d.ts +1 -1
  75. package/out/fst-autocomplete.d.ts.map +1 -1
  76. package/out/fst-autocomplete.js +11 -9
  77. package/out/fst-autocomplete.js.map +1 -1
  78. package/out/fst-builder.d.ts.map +1 -1
  79. package/out/fst-builder.js +42 -9
  80. package/out/fst-builder.js.map +1 -1
  81. package/out/fst-deserialize-web.d.ts.map +1 -1
  82. package/out/fst-deserialize-web.js +34 -9
  83. package/out/fst-deserialize-web.js.map +1 -1
  84. package/out/fst-matcher.d.ts +6 -2
  85. package/out/fst-matcher.d.ts.map +1 -1
  86. package/out/fst-matcher.js +9 -5
  87. package/out/fst-matcher.js.map +1 -1
  88. package/out/fst-serialize.d.ts.map +1 -1
  89. package/out/fst-serialize.js +62 -9
  90. package/out/fst-serialize.js.map +1 -1
  91. package/out/fst-types.d.ts +43 -0
  92. package/out/fst-types.d.ts.map +1 -1
  93. package/out/fts-query.d.ts +41 -0
  94. package/out/fts-query.d.ts.map +1 -0
  95. package/out/fts-query.js +75 -0
  96. package/out/fts-query.js.map +1 -0
  97. package/out/fts.d.ts +21 -7
  98. package/out/fts.d.ts.map +1 -1
  99. package/out/fts.js +10 -4
  100. package/out/fts.js.map +1 -1
  101. package/out/geo.d.ts +6 -2
  102. package/out/geo.d.ts.map +1 -1
  103. package/out/geo.js +3 -1
  104. package/out/geo.js.map +1 -1
  105. package/out/geonames-aliases.d.ts +12 -4
  106. package/out/geonames-aliases.d.ts.map +1 -1
  107. package/out/geonames-aliases.js +72 -67
  108. package/out/geonames-aliases.js.map +1 -1
  109. package/out/geonames-postal.d.ts +9 -3
  110. package/out/geonames-postal.d.ts.map +1 -1
  111. package/out/geonames-postal.js +7 -2
  112. package/out/geonames-postal.js.map +1 -1
  113. package/out/index.d.ts +2 -0
  114. package/out/index.d.ts.map +1 -1
  115. package/out/index.js +1 -0
  116. package/out/index.js.map +1 -1
  117. package/out/interpolation.d.ts +24 -6
  118. package/out/interpolation.d.ts.map +1 -1
  119. package/out/interpolation.js +32 -40
  120. package/out/interpolation.js.map +1 -1
  121. package/out/lookup.d.ts +3 -97
  122. package/out/lookup.d.ts.map +1 -1
  123. package/out/lookup.js +52 -184
  124. package/out/lookup.js.map +1 -1
  125. package/out/name-score.d.ts +28 -0
  126. package/out/name-score.d.ts.map +1 -0
  127. package/out/name-score.js +67 -0
  128. package/out/name-score.js.map +1 -0
  129. package/out/poi-lookup.d.ts +24 -8
  130. package/out/poi-lookup.d.ts.map +1 -1
  131. package/out/poi-lookup.js +27 -13
  132. package/out/poi-lookup.js.map +1 -1
  133. package/out/poi-schema.d.ts +42 -13
  134. package/out/poi-schema.d.ts.map +1 -1
  135. package/out/poi-schema.js +12 -3
  136. package/out/poi-schema.js.map +1 -1
  137. package/out/postal-city-alias-lookup.d.ts +18 -6
  138. package/out/postal-city-alias-lookup.d.ts.map +1 -1
  139. package/out/postal-city-alias-lookup.js.map +1 -1
  140. package/out/postal-city-alias-schema.d.ts +27 -9
  141. package/out/postal-city-alias-schema.d.ts.map +1 -1
  142. package/out/postal-city-alias-schema.js +3 -1
  143. package/out/postal-city-alias-schema.js.map +1 -1
  144. package/out/postal-city-candidate-schema.d.ts +15 -5
  145. package/out/postal-city-candidate-schema.d.ts.map +1 -1
  146. package/out/postal-city-candidate-schema.js +3 -1
  147. package/out/postal-city-candidate-schema.js.map +1 -1
  148. package/out/postcode-point-lookup.d.ts +6 -2
  149. package/out/postcode-point-lookup.d.ts.map +1 -1
  150. package/out/postcode-point-lookup.js +6 -2
  151. package/out/postcode-point-lookup.js.map +1 -1
  152. package/out/ranking-weights.d.ts +118 -0
  153. package/out/ranking-weights.d.ts.map +1 -0
  154. package/out/ranking-weights.js +44 -0
  155. package/out/ranking-weights.js.map +1 -0
  156. package/out/reverse.d.ts +9 -3
  157. package/out/reverse.d.ts.map +1 -1
  158. package/out/reverse.js +20 -6
  159. package/out/reverse.js.map +1 -1
  160. package/out/sharding.d.ts +3 -1
  161. package/out/sharding.d.ts.map +1 -1
  162. package/out/sharding.js +7 -5
  163. package/out/sharding.js.map +1 -1
  164. package/out/sqlite-convention-source.d.ts.map +1 -1
  165. package/out/sqlite-convention-source.js +3 -1
  166. package/out/sqlite-convention-source.js.map +1 -1
  167. package/out/street-centroid-schema.d.ts +33 -11
  168. package/out/street-centroid-schema.d.ts.map +1 -1
  169. package/out/street-centroid-schema.js +3 -1
  170. package/out/street-centroid-schema.js.map +1 -1
  171. package/out/street-centroid.d.ts.map +1 -1
  172. package/out/street-centroid.js +6 -2
  173. package/out/street-centroid.js.map +1 -1
  174. package/out/street-morphology-fst-builder.d.ts +6 -2
  175. package/out/street-morphology-fst-builder.d.ts.map +1 -1
  176. package/out/street-morphology-fst-builder.js +8 -7
  177. package/out/street-morphology-fst-builder.js.map +1 -1
  178. package/out/street-morphology-fst-loader.d.ts +67 -0
  179. package/out/street-morphology-fst-loader.d.ts.map +1 -0
  180. package/out/street-morphology-fst-loader.js +59 -0
  181. package/out/street-morphology-fst-loader.js.map +1 -0
  182. package/out/street-name-lookup.d.ts +9 -3
  183. package/out/street-name-lookup.d.ts.map +1 -1
  184. package/out/street-name-lookup.js +9 -7
  185. package/out/street-name-lookup.js.map +1 -1
  186. package/out/street-normalize.d.ts +3 -1
  187. package/out/street-normalize.d.ts.map +1 -1
  188. package/out/street-normalize.js +23 -13
  189. package/out/street-normalize.js.map +1 -1
  190. package/out/street-segment-schema.d.ts +68 -13
  191. package/out/street-segment-schema.d.ts.map +1 -1
  192. package/out/street-segment-schema.js +21 -2
  193. package/out/street-segment-schema.js.map +1 -1
  194. package/out/types.d.ts +18 -6
  195. package/out/types.d.ts.map +1 -1
  196. package/out/unified-schema.d.ts +1 -1
  197. package/out/unified-schema.d.ts.map +1 -1
  198. package/out/unified-schema.js +2 -2
  199. package/out/unified-schema.js.map +1 -1
  200. package/package.json +13 -5
  201. package/poi-lookup.ts +53 -21
  202. package/poi-schema.ts +43 -13
  203. package/postal-city-alias-lookup.ts +20 -6
  204. package/postal-city-alias-schema.ts +28 -9
  205. package/postal-city-candidate-schema.ts +15 -5
  206. package/postcode-point-lookup.ts +6 -2
  207. package/ranking-weights.ts +148 -0
  208. package/reverse.ts +47 -10
  209. package/sharding.ts +13 -6
  210. package/sqlite-convention-source.ts +4 -1
  211. package/street-centroid-schema.ts +35 -11
  212. package/street-centroid.ts +10 -3
  213. package/street-morphology-fst-builder.ts +25 -9
  214. package/street-morphology-fst-loader.ts +103 -0
  215. package/street-name-lookup.ts +19 -7
  216. package/street-normalize.ts +28 -13
  217. package/street-segment-schema.ts +83 -13
  218. package/types.ts +18 -6
  219. package/unified-schema.ts +11 -2
@@ -17,7 +17,8 @@
17
17
  * needs; without it "new yor" returns nothing useful. (#587)
18
18
  */
19
19
 
20
- import { FSTMatcher, normalizeTokens } from "./fst-matcher.ts"
20
+ import type { FSTMatcher } from "./fst-matcher.ts"
21
+ import { normalizeTokens } from "./fst-matcher.ts"
21
22
  import type { PlaceEntry } from "./fst-types.ts"
22
23
 
23
24
  export interface AutocompleteResult {
@@ -54,7 +55,9 @@ interface BfsItem {
54
55
  tokens: string[]
55
56
  }
56
57
 
57
- /** Max accepting entries collected per BFS branch — keeps one dense branch from starving the search. */
58
+ /**
59
+ * Max accepting entries collected per BFS branch — keeps one dense branch from starving the search.
60
+ */
58
61
  const PER_BRANCH = 4
59
62
 
60
63
  /**
@@ -63,7 +66,7 @@ const PER_BRANCH = 4
63
66
  function topByImportance(entries: readonly PlaceEntry[], k: number): PlaceEntry[] {
64
67
  if (entries.length <= k) return [...entries]
65
68
 
66
- return [...entries].sort((a, b) => b.importance - a.importance).slice(0, k)
69
+ return [...entries].toSorted((a, b) => b.importance - a.importance).slice(0, k)
67
70
  }
68
71
 
69
72
  /**
@@ -74,13 +77,13 @@ export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteO
74
77
  const maxExpansionDepth = opts.maxExpansionDepth ?? 2
75
78
  const normalizedTokens = normalizeTokens(query)
76
79
 
77
- if (normalizedTokens.length === 0) {
80
+ if (!normalizedTokens.length) {
78
81
  return { query, normalizedTokens: [], depth: 0, suggestions: [] }
79
82
  }
80
83
 
81
84
  const seen = new Map<number, AutocompleteSuggestion>()
82
85
  const queue: BfsItem[] = []
83
- let depth = 0
86
+ let depth: number
84
87
 
85
88
  const match = fst.walk(normalizedTokens)
86
89
 
@@ -98,12 +101,13 @@ export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteO
98
101
  } else {
99
102
  // PARTIAL last token — walk the complete prefix, complete the partial by prefix-filtering edges.
100
103
  const complete = normalizedTokens.slice(0, -1)
101
- const partial = normalizedTokens[normalizedTokens.length - 1]!
102
- const prefixState = complete.length === 0 ? 0 : (fst.walk(complete)?.stateID ?? undefined)
104
+ const partial = normalizedTokens.at(-1)!
105
+ const prefixState = !complete.length ? 0 : (fst.walk(complete)?.stateID ?? undefined)
103
106
 
104
107
  if (prefixState === undefined) {
105
108
  return { query, normalizedTokens, depth: 0, suggestions: [] }
106
109
  }
110
+
107
111
  depth = complete.length
108
112
 
109
113
  for (const cont of fst.continuations(prefixState)) {
@@ -113,6 +117,7 @@ export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteO
113
117
  for (const entry of topByImportance(fst.accepting(cont.targetState), PER_BRANCH)) {
114
118
  addSuggestion(seen, entry, complete.length + 1, [cont.token])
115
119
  }
120
+
116
121
  // BFS a little past it too (multi-token completions: "new yor" → "New York Mills").
117
122
  queue.push({ stateID: cont.targetState, depth: 1, tokens: [cont.token] })
118
123
  }
@@ -122,7 +127,7 @@ export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteO
122
127
  // branch contributes only its top PER_BRANCH places: a state like "new london" has dozens of
123
128
  // accepting entries and would otherwise blow the budget before the BFS ever reaches "new york"
124
129
  // (the "new" state has 311 continuations). Per-branch capping keeps the search broad. (#587)
125
- while (queue.length > 0 && seen.size < maxSuggestions * 4) {
130
+ while (queue.length && seen.size < maxSuggestions * 4) {
126
131
  const item = queue.shift()!
127
132
 
128
133
  if (item.depth > maxExpansionDepth) continue
@@ -138,7 +143,7 @@ export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteO
138
143
  }
139
144
  }
140
145
 
141
- let suggestions = [...seen.values()].sort((a, b) => b.importance - a.importance)
146
+ let suggestions = [...seen.values()].toSorted((a, b) => b.importance - a.importance)
142
147
 
143
148
  if (opts.dedupeByName) {
144
149
  suggestions = dedupeByName(suggestions)
@@ -156,6 +161,7 @@ function addSuggestion(
156
161
  const existing = seen.get(entry.wofID)
157
162
 
158
163
  if (existing && existing.matchDepth <= matchDepth) return
164
+
159
165
  seen.set(entry.wofID, {
160
166
  name: entry.name,
161
167
  placetype: entry.placetype,
package/fst-builder.ts CHANGED
@@ -26,6 +26,7 @@ const DEFAULT_PLACETYPES: PlacetypeID[] = [
26
26
  "borough",
27
27
  "neighbourhood",
28
28
  ]
29
+
29
30
  const DEFAULT_COUNTRIES = ["US"]
30
31
  const DEFAULT_LANGUAGES = ["eng", ""]
31
32
 
@@ -66,6 +67,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
66
67
  // Phase 1: Load all matching SPR rows.
67
68
  progress("spr", `Loading places for countries=[${countries}], placetypes=[${placetypes}]`)
68
69
  const placeholders = (arr: string[]) => arr.map(() => "?").join(",")
70
+
69
71
  const sprStmt = db.prepare(
70
72
  `SELECT id, name, placetype, parent_id, latitude, longitude
71
73
  FROM spr
@@ -73,6 +75,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
73
75
  AND country IN (${placeholders(countries)})
74
76
  AND placetype IN (${placeholders(placetypes)})`
75
77
  )
78
+
76
79
  const sprRows = sprStmt.all(...countries, ...placetypes) as unknown as SprRow[]
77
80
  progress("spr", `Loaded ${sprRows.length} places`)
78
81
 
@@ -155,6 +158,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
155
158
  for (const row of impRows) {
156
159
  importanceMap.set(row.id, row.importance)
157
160
  }
161
+
158
162
  progress("importance", `Loaded ${importanceMap.size} importance scores`)
159
163
  } catch {
160
164
  progress("importance", "No place_importance table — falling back to population")
@@ -164,7 +168,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
164
168
  const popRows = popStmt.all() as unknown as PopulationRow[]
165
169
 
166
170
  for (const row of popRows) {
167
- const normalized = row.population > 0 ? Math.min(1.0, Math.log2(1 + row.population / 1000) / 14) : 0
171
+ const normalized = row.population > 0 ? Math.min(1, Math.log2(1 + row.population / 1000) / 14) : 0
168
172
  importanceMap.set(row.id, normalized)
169
173
  }
170
174
  } catch {
@@ -182,11 +186,13 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
182
186
  for (let i = 0; i < placeIds.length; i += 500) {
183
187
  const chunk = placeIds.slice(i, i + 500)
184
188
  const idPlaceholders = chunk.map(() => "?").join(",")
189
+
185
190
  const nameStmt = allLanguages
186
191
  ? db.prepare(`SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders})`)
187
192
  : db.prepare(
188
193
  `SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders}) AND language IN (${languages.map(() => "?").join(",")})`
189
194
  )
195
+
190
196
  const nameRows = (allLanguages
191
197
  ? nameStmt.all(...chunk)
192
198
  : nameStmt.all(...chunk, ...languages)) as unknown as NameRow[]
@@ -197,17 +203,47 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
197
203
  if (!existing.includes(row.name)) {
198
204
  existing.push(row.name)
199
205
  }
206
+
200
207
  namesByPlace.set(row.id, existing)
201
208
  }
202
209
  }
210
+
203
211
  progress("names", `Loaded names for ${namesByPlace.size} places`)
204
212
 
205
213
  // Phase 5: Build the trie.
206
214
  progress("trie", "Building trie")
207
215
  const nodes: FSTNode[] = [{ edges: new Map(), places: [] }]
208
216
 
209
- function insertName(tokens: string[], entry: PlaceEntry): void {
210
- if (tokens.length === 0) return
217
+ // Degenerate-surface curation (see BuildFSTOpts.excludeSurfaces). Applied to the WHOLE normalized
218
+ // surface only — a multi-token name containing a function word ("los angeles") is never affected.
219
+ const excludeSurfaces = opts.excludeSurfaces
220
+ const excludeAllTokensOf = opts.excludeAllTokensOf
221
+ let excludedCount = 0
222
+
223
+ function isDegenerate(tokens: string[]): boolean {
224
+ if (!tokens.length) return false
225
+
226
+ if (excludeSurfaces?.has(tokens.join(" "))) return true
227
+
228
+ if (excludeAllTokensOf !== undefined && tokens.every((t) => excludeAllTokensOf.has(t))) return true
229
+
230
+ return false
231
+ }
232
+
233
+ // Surface-ambiguity classes (survey #4): a per-SURFACE fact, so the entry is cloned per insertion
234
+ // with its accepting surface's count attached (the same place under "nyc" and "new york city"
235
+ // records each surface's own ambiguity). Absent map → entries carry no count (back-compat bytes).
236
+ const surfaceCountryCounts = opts.surfaceCountryCounts
237
+
238
+ function insertName(tokens: string[], entry: PlaceEntry): boolean {
239
+ if (!tokens.length) return false
240
+
241
+ if (isDegenerate(tokens)) {
242
+ excludedCount++
243
+
244
+ return false
245
+ }
246
+
211
247
  let stateID = 0
212
248
 
213
249
  for (const t of tokens) {
@@ -219,20 +255,30 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
219
255
  nodes.push({ edges: new Map(), places: [] })
220
256
  node.edges.set(t, next)
221
257
  }
258
+
222
259
  stateID = next
223
260
  }
261
+
224
262
  // Deduplicate: don't add the same wofID twice at the same state.
225
263
  const existing = nodes[stateID]!.places
226
264
 
227
265
  if (!existing.some((p) => p.wofID === entry.wofID && p.placetype === entry.placetype)) {
228
- existing.push(entry)
266
+ if (surfaceCountryCounts !== undefined) {
267
+ const count = surfaceCountryCounts.get(tokens.join(" "))
268
+ existing.push({ ...entry, crossCountryBranches: Math.min(count ?? 1, 255) })
269
+ } else {
270
+ existing.push(entry)
271
+ }
229
272
  }
273
+
274
+ return true
230
275
  }
231
276
 
232
277
  let insertCount = 0
233
278
 
234
279
  for (const row of sprRows) {
235
280
  const parentChain = resolveParentChain(row.id)
281
+
236
282
  const entry: PlaceEntry = {
237
283
  wofID: row.id,
238
284
  placetype: row.placetype as PlacetypeID,
@@ -245,8 +291,10 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
245
291
 
246
292
  // Insert the primary name from spr.
247
293
  const primaryTokens = normalizeTokens(row.name)
248
- insertName(primaryTokens, entry)
249
- insertCount++
294
+
295
+ if (insertName(primaryTokens, entry)) {
296
+ insertCount++
297
+ }
250
298
 
251
299
  // Insert alt names from the names table.
252
300
  const altNames = namesByPlace.get(row.id) ?? []
@@ -255,18 +303,23 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
255
303
  if (altName === row.name) continue
256
304
  const altTokens = normalizeTokens(altName)
257
305
 
258
- if (altTokens.length > 0 && altTokens.join(" ") !== primaryTokens.join(" ")) {
259
- insertName(altTokens, entry)
306
+ if (altTokens.length && altTokens.join(" ") !== primaryTokens.join(" ") && insertName(altTokens, entry)) {
260
307
  insertCount++
261
308
  }
262
309
  }
263
310
  }
264
311
 
265
312
  db.close()
266
- progress("done", `Built trie: ${nodes.length} states, ${insertCount} name insertions`)
313
+
314
+ progress(
315
+ "done",
316
+ `Built trie: ${nodes.length} states, ${insertCount} name insertions` +
317
+ (excludedCount > 0 ? ` (${excludedCount} degenerate surfaces excluded)` : "")
318
+ )
267
319
 
268
320
  const edgeCount = nodes.reduce((sum, n) => sum + n.edges.size, 0)
269
321
  const matcher = FSTMatcher.fromNodes(nodes)
322
+
270
323
  const provenance: FSTProvenance = {
271
324
  builtAt: new Date().toISOString(),
272
325
  countries,
@@ -276,6 +329,9 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
276
329
  nameInsertions: insertCount,
277
330
  importanceMatches: importanceMap.size,
278
331
  sourceDB: opts.dbPath,
332
+ ...(excludeSurfaces !== undefined || excludeAllTokensOf !== undefined
333
+ ? { exclusionPolicy: opts.exclusionPolicy ?? "unspecified", excludedInsertions: excludedCount }
334
+ : {}),
279
335
  }
280
336
 
281
337
  return {
@@ -14,13 +14,39 @@ import type { FSTNode } from "./fst-matcher.ts"
14
14
  import { FSTMatcher } from "./fst-matcher.ts"
15
15
  import type { FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
16
16
 
17
+ /**
18
+ * Format version that widened the per-state edge and place counters from 16 to 32 bits, growing the state entry from 12
19
+ * to 16 bytes. Readers branch on it to stay backward-compatible with v2/v3 files.
20
+ */
21
+ const VERSION_WIDE_STATE_COUNTERS = 4
22
+
23
+ /**
24
+ * State-table entry size in bytes at or above {@link VERSION_WIDE_STATE_COUNTERS}.
25
+ */
26
+ const WIDE_STATE_ENTRY_SIZE = 16
27
+
28
+ /**
29
+ * State-table entry size in bytes below {@link VERSION_WIDE_STATE_COUNTERS}.
30
+ */
31
+ const NARROW_STATE_ENTRY_SIZE = 12
32
+
33
+ /**
34
+ * First format version carrying the trailing metadata block; older files simply have none.
35
+ */
36
+ const VERSION_WITH_METADATA = 3
37
+
17
38
  const HEADER_SIZE = 32
18
39
  const EDGE_ENTRY_SIZE = 8
19
40
  const PLACE_ENTRY_SIZE = 56
20
- const MAGIC_BYTES = [0x46, 0x53, 0x54, 0x00] // "FST\0"
21
- // Must track the serializer's VERSION (fst-serialize.ts, currently 4). The v3 provenance + v4
22
- // 16-byte-state/u32-count layout logic below already matches the Node deserializer; only this gate
23
- // was left stale at 2, so the browser FST loader rejected every real (v4) artifact.
41
+ /**
42
+ * "FST\0".
43
+ */
44
+ const MAGIC_BYTES = [0x46, 0x53, 0x54, 0x00]
45
+ /**
46
+ * Must track the serializer's VERSION (fst-serialize.ts, currently 4). The v3 provenance + v4 16-byte-state/u32-count
47
+ * layout logic below already matches the Node deserializer; only this gate was left stale at 2, so the browser FST
48
+ * loader rejected every real (v4) artifact.
49
+ */
24
50
  const MAX_VERSION = 4
25
51
 
26
52
  const PLACETYPE_ORDER: readonly PlacetypeID[] = [
@@ -58,7 +84,10 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
58
84
  if (version < 1 || version > MAX_VERSION) {
59
85
  throw new Error(`FST version ${version} unsupported (expected 1..${MAX_VERSION})`)
60
86
  }
87
+
61
88
  const isV2 = version >= 2
89
+ // flags bit0 (survey #4, mirrors fst-serialize.ts): place rows carry surface-ambiguity data.
90
+ const hasAmbiguity = (view.getUint16(6, true) & 1) === 1
62
91
 
63
92
  const stateCount = view.getUint32(8, true)
64
93
  const edgeCount = view.getUint32(12, true)
@@ -75,6 +104,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
75
104
  strOffsets[i] = view.getUint32(pos, true)
76
105
  pos += 4
77
106
  }
107
+
78
108
  const strDataStart = pos
79
109
  const strings: string[] = new Array(stringCount)
80
110
 
@@ -83,10 +113,11 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
83
113
  const end = strDataStart + strOffsets[i + 1]!
84
114
  strings[i] = decoder.decode(bytes.subarray(start, end))
85
115
  }
116
+
86
117
  pos += stringBytes
87
118
 
88
119
  // --- State table ---
89
- const stateEntrySize = version >= 4 ? 16 : 12
120
+ const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
90
121
  const stateTableStart = pos
91
122
  const edgeTableStart = stateTableStart + stateCount * stateEntrySize
92
123
  const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
@@ -97,8 +128,12 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
97
128
  const sp = stateTableStart + si * stateEntrySize
98
129
  const edgeStart = view.getUint32(sp, true)
99
130
  const placeStart = view.getUint32(sp + 4, true)
100
- const edgeCountForState = version >= 4 ? view.getUint32(sp + 8, true) : view.getUint16(sp + 8, true)
101
- const placeCountForState = version >= 4 ? view.getUint32(sp + 12, true) : view.getUint16(sp + 10, true)
131
+
132
+ const edgeCountForState =
133
+ version >= VERSION_WIDE_STATE_COUNTERS ? view.getUint32(sp + 8, true) : view.getUint16(sp + 8, true)
134
+
135
+ const placeCountForState =
136
+ version >= VERSION_WIDE_STATE_COUNTERS ? view.getUint32(sp + 12, true) : view.getUint16(sp + 10, true)
102
137
 
103
138
  const edges = new Map<string, number>()
104
139
 
@@ -119,9 +154,10 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
119
154
  for (let ci = 0; ci < chainLen; ci++) {
120
155
  parentChain.push(view.getUint32(pp + 24 + ci * 4, true))
121
156
  }
157
+
122
158
  const rawImportance = isV2
123
159
  ? view.getFloat32(pp + 12, true)
124
- : Math.min(1.0, Math.log2(1 + view.getUint32(pp + 12, true) / 1000) / 14)
160
+ : Math.min(1, Math.log2(1 + view.getUint32(pp + 12, true) / 1000) / 14)
125
161
 
126
162
  places[pi] = {
127
163
  wofID: view.getUint32(pp, true),
@@ -131,6 +167,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
131
167
  lat: view.getFloat32(pp + 16, true),
132
168
  lon: view.getFloat32(pp + 20, true),
133
169
  parentChain,
170
+ ...(hasAmbiguity ? { crossCountryBranches: view.getUint8(pp + 6) } : {}),
134
171
  }
135
172
  }
136
173
 
@@ -148,7 +185,7 @@ export function readFSTProvenanceWeb(input: ArrayBuffer | Uint8Array): FSTProven
148
185
  if (bytes.byteLength < HEADER_SIZE) return undefined
149
186
  const version = view.getUint16(4, true)
150
187
 
151
- if (version < 3) return undefined
188
+ if (version < VERSION_WITH_METADATA) return undefined
152
189
  const provenanceOffset = view.getUint32(28, true)
153
190
 
154
191
  if (provenanceOffset === 0 || provenanceOffset >= bytes.byteLength) return undefined
package/fst-matcher.ts CHANGED
@@ -39,15 +39,16 @@ export class FSTMatcher {
39
39
  walk(tokens: string[]): FSTMatchResult | null {
40
40
  let stateID = 0
41
41
 
42
- for (let i = 0; i < tokens.length; i++) {
42
+ for (const token of tokens) {
43
43
  const node = this.nodes[stateID]
44
44
 
45
45
  if (!node) return null
46
- const next = node.edges.get(tokens[i]!)
46
+ const next = node.edges.get(token)
47
47
 
48
48
  if (next === undefined) return null
49
49
  stateID = next
50
50
  }
51
+
51
52
  const node = this.nodes[stateID]!
52
53
 
53
54
  return { stateID, accepted: node.places.length > 0, depth: tokens.length }
@@ -77,6 +78,7 @@ export class FSTMatcher {
77
78
 
78
79
  for (const [token, targetID] of node.edges) {
79
80
  const target = this.nodes[targetID]!
81
+
80
82
  result.push({
81
83
  token,
82
84
  targetState: targetID,
@@ -104,6 +106,7 @@ export class FSTMatcher {
104
106
 
105
107
  if (next === undefined) break
106
108
  stateID = next
109
+
107
110
  depth++
108
111
  }
109
112
 
@@ -127,7 +130,9 @@ export class FSTMatcher {
127
130
  return this.nodes.length
128
131
  }
129
132
 
130
- /** Expose the internal node array for serialization. */
133
+ /**
134
+ * Expose the internal node array for serialization.
135
+ */
131
136
  toNodes(): readonly FSTNode[] {
132
137
  return this.nodes
133
138
  }
@@ -137,12 +142,14 @@ export class FSTMatcher {
137
142
  }
138
143
  }
139
144
 
140
- /** Normalize text into FST tokens: lowercase, NFKC, strip punctuation, split on whitespace. */
145
+ /**
146
+ * Normalize text into FST tokens: lowercase, NFKC, strip punctuation, split on whitespace.
147
+ */
141
148
  export function normalizeTokens(text: string): string[] {
142
149
  return text
143
150
  .normalize("NFKC")
144
151
  .toLowerCase()
145
- .replace(/[\p{P}\p{S}]/gu, "")
152
+ .replaceAll(/[\p{P}\p{S}]/gu, "")
146
153
  .split(/\s+/)
147
154
  .filter((t) => t.length > 0)
148
155
  }
package/fst-serialize.ts CHANGED
@@ -27,14 +27,67 @@ import type { FSTNode } from "./fst-matcher.ts"
27
27
  import { FSTMatcher } from "./fst-matcher.ts"
28
28
  import type { FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
29
29
 
30
+ /**
31
+ * Format version that widened the per-state edge and place counters from 16 to 32 bits, growing the state entry from 12
32
+ * to 16 bytes. Readers branch on it to stay backward-compatible with v2/v3 files.
33
+ */
34
+ const VERSION_WIDE_STATE_COUNTERS = 4
35
+
36
+ /**
37
+ * State-table entry size in bytes at or above {@link VERSION_WIDE_STATE_COUNTERS}.
38
+ */
39
+ const WIDE_STATE_ENTRY_SIZE = 16
40
+
41
+ /**
42
+ * State-table entry size in bytes below {@link VERSION_WIDE_STATE_COUNTERS}.
43
+ */
44
+ const NARROW_STATE_ENTRY_SIZE = 12
45
+
46
+ /**
47
+ * First format version carrying the trailing metadata block; older files simply have none.
48
+ */
49
+ const VERSION_WITH_METADATA = 3
50
+
51
+ /**
52
+ * File magic. A reader rejects anything not starting with these four bytes before parsing further.
53
+ */
30
54
  const MAGIC = Buffer.from("FST\0", "ascii")
55
+
56
+ /**
57
+ * Format version this serializer emits. See {@link VERSION_WIDE_STATE_COUNTERS} for what each bump changed.
58
+ */
31
59
  const VERSION = 4
60
+
61
+ /**
62
+ * Fixed header size in bytes: magic, version, and the section offsets that follow it.
63
+ */
32
64
  const HEADER_SIZE = 32
65
+
66
+ /**
67
+ * State-table entry: edge offset, place offset, and the two 32-bit counters (v4 widths).
68
+ */
33
69
  const STATE_ENTRY_SIZE = 16
70
+
71
+ /**
72
+ * Edge-table entry: the transition label and the target state index.
73
+ */
34
74
  const EDGE_ENTRY_SIZE = 8
75
+
76
+ /**
77
+ * Place-table entry: the place id, its placetype, coordinates, and importance.
78
+ */
35
79
  const PLACE_ENTRY_SIZE = 56
80
+
81
+ /**
82
+ * Longest ancestry chain stored per place. Deeper hierarchies are truncated at the leaf end, since the specific end of
83
+ * the chain is what disambiguates and the country end is recoverable anyway.
84
+ */
36
85
  const MAX_CHAIN_LEN = 8
37
86
 
87
+ /**
88
+ * Placetypes in hierarchy order, largest first. The index into this array is what gets written into a place entry, so
89
+ * REORDERING IT BREAKS EVERY EXISTING FILE — append instead, and bump the version.
90
+ */
38
91
  const PLACETYPE_ORDER: readonly PlacetypeID[] = [
39
92
  "country",
40
93
  "region",
@@ -109,11 +162,15 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
109
162
  let pos = 0
110
163
 
111
164
  // --- Header ---
165
+ // flags bit0 (survey #4, 2026-07-27): place rows carry surface-ambiguity data in the former _pad
166
+ // byte (pp+6 = crossCountryBranches u8, pp+7 reserved). Presence-signaled here so VERSION stays
167
+ // put: pre-ambiguity artifacts read flags=0 → readers expose `undefined`, never a fake 0.
168
+ const hasAmbiguity = nodes.some((n) => n.places.some((p) => p.crossCountryBranches !== undefined))
112
169
  MAGIC.copy(buf, pos)
113
170
  pos += 4
114
171
  buf.writeUInt16LE(VERSION, pos)
115
172
  pos += 2
116
- buf.writeUInt16LE(0, pos)
173
+ buf.writeUInt16LE(hasAmbiguity ? 1 : 0, pos)
117
174
  pos += 2
118
175
  buf.writeUInt32LE(nodes.length, pos)
119
176
  pos += 4
@@ -131,11 +188,12 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
131
188
  // --- String table ---
132
189
  let strOffset = 0
133
190
 
134
- for (let i = 0; i < encodedStrings.length; i++) {
191
+ for (const encoded of encodedStrings) {
135
192
  buf.writeUInt32LE(strOffset, pos)
136
193
  pos += 4
137
- strOffset += encodedStrings[i]!.length
194
+ strOffset += encoded.length
138
195
  }
196
+
139
197
  buf.writeUInt32LE(strOffset, pos)
140
198
  pos += 4
141
199
 
@@ -167,6 +225,7 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
167
225
  const ep = edgeTableStart + edgeIdx * EDGE_ENTRY_SIZE
168
226
  buf.writeUInt32LE(intern(token), ep)
169
227
  buf.writeUInt32LE(target, ep + 4)
228
+
170
229
  edgeIdx++
171
230
  }
172
231
 
@@ -178,7 +237,9 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
178
237
  buf.writeUInt32LE(place.wofID, pp)
179
238
  buf.writeUInt8(placetypeToIdx.get(place.placetype) ?? 0, pp + 4)
180
239
  buf.writeUInt8(chainLen, pp + 5)
181
- buf.writeUInt16LE(0, pp + 6) // pad
240
+ // Former _pad: byte 0 = crossCountryBranches (header flags bit0 gates the read), byte 1 reserved.
241
+ buf.writeUInt8(hasAmbiguity ? Math.min(place.crossCountryBranches ?? 0, 255) : 0, pp + 6)
242
+ buf.writeUInt8(0, pp + 7)
182
243
  buf.writeUInt32LE(intern(place.name), pp + 8)
183
244
  buf.writeFloatLE(place.importance, pp + 12)
184
245
  buf.writeFloatLE(place.lat, pp + 16)
@@ -187,6 +248,7 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
187
248
  for (let ci = 0; ci < MAX_CHAIN_LEN; ci++) {
188
249
  buf.writeUInt32LE(ci < chainLen ? validChain[ci]! : 0, pp + 24 + ci * 4)
189
250
  }
251
+
190
252
  placeIdx++
191
253
  }
192
254
  }
@@ -209,6 +271,8 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
209
271
 
210
272
  if (version < 1 || version > VERSION) throw new Error(`FST version ${version} unsupported (expected 1..${VERSION})`)
211
273
  const isV2 = version >= 2
274
+ // flags bit0 (survey #4): place rows carry surface-ambiguity data in the former _pad byte.
275
+ const hasAmbiguity = (buf.readUInt16LE(6) & 1) === 1
212
276
 
213
277
  const stateCount = buf.readUInt32LE(8)
214
278
  const edgeCount = buf.readUInt32LE(12)
@@ -225,6 +289,7 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
225
289
  strOffsets[i] = buf.readUInt32LE(pos)
226
290
  pos += 4
227
291
  }
292
+
228
293
  const strDataStart = pos
229
294
  const strings: string[] = new Array(stringCount)
230
295
 
@@ -233,10 +298,11 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
233
298
  const end = strDataStart + strOffsets[i + 1]!
234
299
  strings[i] = buf.toString("utf8", start, end)
235
300
  }
301
+
236
302
  pos += stringBytes
237
303
 
238
304
  // --- State table ---
239
- const stateEntrySize = version >= 4 ? 16 : 12
305
+ const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
240
306
  const stateTableStart = pos
241
307
  const edgeTableStart = stateTableStart + stateCount * stateEntrySize
242
308
  const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
@@ -247,8 +313,12 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
247
313
  const sp = stateTableStart + si * stateEntrySize
248
314
  const edgeStart = buf.readUInt32LE(sp)
249
315
  const placeStart = buf.readUInt32LE(sp + 4)
250
- const edgeCountForState = version >= 4 ? buf.readUInt32LE(sp + 8) : buf.readUInt16LE(sp + 8)
251
- const placeCountForState = version >= 4 ? buf.readUInt32LE(sp + 12) : buf.readUInt16LE(sp + 10)
316
+
317
+ const edgeCountForState =
318
+ version >= VERSION_WIDE_STATE_COUNTERS ? buf.readUInt32LE(sp + 8) : buf.readUInt16LE(sp + 8)
319
+
320
+ const placeCountForState =
321
+ version >= VERSION_WIDE_STATE_COUNTERS ? buf.readUInt32LE(sp + 12) : buf.readUInt16LE(sp + 10)
252
322
 
253
323
  const edges = new Map<string, number>()
254
324
 
@@ -269,9 +339,11 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
269
339
  for (let ci = 0; ci < chainLen; ci++) {
270
340
  parentChain.push(buf.readUInt32LE(pp + 24 + ci * 4))
271
341
  }
342
+
272
343
  const rawImportance = isV2
273
344
  ? buf.readFloatLE(pp + 12)
274
- : Math.min(1.0, Math.log2(1 + buf.readUInt32LE(pp + 12) / 1000) / 14)
345
+ : Math.min(1, Math.log2(1 + buf.readUInt32LE(pp + 12) / 1000) / 14)
346
+
275
347
  places[pi] = {
276
348
  wofID: buf.readUInt32LE(pp),
277
349
  placetype: PLACETYPE_ORDER[buf.readUInt8(pp + 4)] ?? "locality",
@@ -280,6 +352,8 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
280
352
  lat: buf.readFloatLE(pp + 16),
281
353
  lon: buf.readFloatLE(pp + 20),
282
354
  parentChain,
355
+ // Header flags bit0 gates the read (survey #4): pre-ambiguity artifacts expose undefined.
356
+ ...(hasAmbiguity ? { crossCountryBranches: buf.readUInt8(pp + 6) } : {}),
283
357
  }
284
358
  }
285
359
 
@@ -295,7 +369,7 @@ export function readFSTProvenance(buf: Buffer): FSTProvenance | undefined {
295
369
  if (!buf.subarray(0, 4).equals(MAGIC)) return undefined
296
370
  const version = buf.readUInt16LE(4)
297
371
 
298
- if (version < 3) return undefined
372
+ if (version < VERSION_WITH_METADATA) return undefined
299
373
  const provenanceOffset = buf.readUInt32LE(28)
300
374
 
301
375
  if (provenanceOffset === 0 || provenanceOffset >= buf.length) return undefined