@mailwoman/resolver-wof-sqlite 8.1.0 → 8.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (217) hide show
  1. package/address-point-interpolation.ts +9 -3
  2. package/address-point-schema.ts +32 -10
  3. package/address-point.ts +3 -0
  4. package/ancestry-backfill.ts +18 -5
  5. package/ancestry.ts +7 -2
  6. package/build-candidate.ts +29 -6
  7. package/build-slim.ts +40 -11
  8. package/candidate-fts.ts +1 -0
  9. package/candidate-lookup.ts +36 -14
  10. package/candidate-schema.ts +35 -11
  11. package/coincident-roles.ts +28 -6
  12. package/convention.ts +3 -1
  13. package/coverage-manifest-schema.ts +49 -16
  14. package/fst-autocomplete.ts +15 -9
  15. package/fst-builder.ts +29 -5
  16. package/fst-deserialize-web.ts +46 -9
  17. package/fst-matcher.ts +12 -5
  18. package/fst-serialize.ts +83 -9
  19. package/fst-types.ts +26 -3
  20. package/fts-query.ts +84 -0
  21. package/fts.ts +35 -9
  22. package/geo.ts +9 -3
  23. package/geonames-aliases.ts +116 -79
  24. package/geonames-postal.ts +25 -5
  25. package/index.ts +10 -0
  26. package/interpolation.ts +33 -55
  27. package/lookup.ts +103 -292
  28. package/name-score.ts +76 -0
  29. package/out/address-point-interpolation.d.ts.map +1 -1
  30. package/out/address-point-interpolation.js +4 -2
  31. package/out/address-point-interpolation.js.map +1 -1
  32. package/out/address-point-schema.d.ts +30 -10
  33. package/out/address-point-schema.d.ts.map +1 -1
  34. package/out/address-point-schema.js +6 -2
  35. package/out/address-point-schema.js.map +1 -1
  36. package/out/address-point.d.ts.map +1 -1
  37. package/out/address-point.js.map +1 -1
  38. package/out/ancestry-backfill.d.ts +6 -2
  39. package/out/ancestry-backfill.d.ts.map +1 -1
  40. package/out/ancestry-backfill.js +7 -3
  41. package/out/ancestry-backfill.js.map +1 -1
  42. package/out/ancestry.d.ts +6 -2
  43. package/out/ancestry.d.ts.map +1 -1
  44. package/out/ancestry.js +3 -1
  45. package/out/ancestry.js.map +1 -1
  46. package/out/build-candidate.d.ts +9 -3
  47. package/out/build-candidate.d.ts.map +1 -1
  48. package/out/build-candidate.js +5 -3
  49. package/out/build-candidate.js.map +1 -1
  50. package/out/build-slim.d.ts +15 -5
  51. package/out/build-slim.d.ts.map +1 -1
  52. package/out/build-slim.js +11 -5
  53. package/out/build-slim.js.map +1 -1
  54. package/out/candidate-fts.d.ts.map +1 -1
  55. package/out/candidate-fts.js.map +1 -1
  56. package/out/candidate-lookup.d.ts +9 -3
  57. package/out/candidate-lookup.d.ts.map +1 -1
  58. package/out/candidate-lookup.js +16 -11
  59. package/out/candidate-lookup.js.map +1 -1
  60. package/out/candidate-schema.d.ts +33 -11
  61. package/out/candidate-schema.d.ts.map +1 -1
  62. package/out/candidate-schema.js.map +1 -1
  63. package/out/coincident-roles.d.ts +16 -4
  64. package/out/coincident-roles.d.ts.map +1 -1
  65. package/out/coincident-roles.js +9 -3
  66. package/out/coincident-roles.js.map +1 -1
  67. package/out/convention.d.ts +3 -1
  68. package/out/convention.d.ts.map +1 -1
  69. package/out/convention.js.map +1 -1
  70. package/out/coverage-manifest-schema.d.ts +45 -14
  71. package/out/coverage-manifest-schema.d.ts.map +1 -1
  72. package/out/coverage-manifest-schema.js +14 -5
  73. package/out/coverage-manifest-schema.js.map +1 -1
  74. package/out/fst-autocomplete.d.ts +1 -1
  75. package/out/fst-autocomplete.d.ts.map +1 -1
  76. package/out/fst-autocomplete.js +11 -9
  77. package/out/fst-autocomplete.js.map +1 -1
  78. package/out/fst-builder.d.ts.map +1 -1
  79. package/out/fst-builder.js +15 -5
  80. package/out/fst-builder.js.map +1 -1
  81. package/out/fst-deserialize-web.d.ts.map +1 -1
  82. package/out/fst-deserialize-web.js +34 -9
  83. package/out/fst-deserialize-web.js.map +1 -1
  84. package/out/fst-matcher.d.ts +6 -2
  85. package/out/fst-matcher.d.ts.map +1 -1
  86. package/out/fst-matcher.js +9 -5
  87. package/out/fst-matcher.js.map +1 -1
  88. package/out/fst-serialize.d.ts.map +1 -1
  89. package/out/fst-serialize.js +62 -9
  90. package/out/fst-serialize.js.map +1 -1
  91. package/out/fst-types.d.ts +26 -3
  92. package/out/fst-types.d.ts.map +1 -1
  93. package/out/fts-query.d.ts +41 -0
  94. package/out/fts-query.d.ts.map +1 -0
  95. package/out/fts-query.js +75 -0
  96. package/out/fts-query.js.map +1 -0
  97. package/out/fts.d.ts +21 -7
  98. package/out/fts.d.ts.map +1 -1
  99. package/out/fts.js +10 -4
  100. package/out/fts.js.map +1 -1
  101. package/out/geo.d.ts +6 -2
  102. package/out/geo.d.ts.map +1 -1
  103. package/out/geo.js +3 -1
  104. package/out/geo.js.map +1 -1
  105. package/out/geonames-aliases.d.ts +12 -4
  106. package/out/geonames-aliases.d.ts.map +1 -1
  107. package/out/geonames-aliases.js +72 -67
  108. package/out/geonames-aliases.js.map +1 -1
  109. package/out/geonames-postal.d.ts +9 -3
  110. package/out/geonames-postal.d.ts.map +1 -1
  111. package/out/geonames-postal.js +7 -2
  112. package/out/geonames-postal.js.map +1 -1
  113. package/out/index.d.ts.map +1 -1
  114. package/out/index.js.map +1 -1
  115. package/out/interpolation.d.ts +18 -6
  116. package/out/interpolation.d.ts.map +1 -1
  117. package/out/interpolation.js +11 -40
  118. package/out/interpolation.js.map +1 -1
  119. package/out/lookup.d.ts +3 -97
  120. package/out/lookup.d.ts.map +1 -1
  121. package/out/lookup.js +52 -184
  122. package/out/lookup.js.map +1 -1
  123. package/out/name-score.d.ts +28 -0
  124. package/out/name-score.d.ts.map +1 -0
  125. package/out/name-score.js +67 -0
  126. package/out/name-score.js.map +1 -0
  127. package/out/poi-lookup.d.ts +24 -8
  128. package/out/poi-lookup.d.ts.map +1 -1
  129. package/out/poi-lookup.js +27 -13
  130. package/out/poi-lookup.js.map +1 -1
  131. package/out/poi-schema.d.ts +42 -13
  132. package/out/poi-schema.d.ts.map +1 -1
  133. package/out/poi-schema.js +12 -3
  134. package/out/poi-schema.js.map +1 -1
  135. package/out/postal-city-alias-lookup.d.ts +18 -6
  136. package/out/postal-city-alias-lookup.d.ts.map +1 -1
  137. package/out/postal-city-alias-lookup.js.map +1 -1
  138. package/out/postal-city-alias-schema.d.ts +27 -9
  139. package/out/postal-city-alias-schema.d.ts.map +1 -1
  140. package/out/postal-city-alias-schema.js +3 -1
  141. package/out/postal-city-alias-schema.js.map +1 -1
  142. package/out/postal-city-candidate-schema.d.ts +15 -5
  143. package/out/postal-city-candidate-schema.d.ts.map +1 -1
  144. package/out/postal-city-candidate-schema.js +3 -1
  145. package/out/postal-city-candidate-schema.js.map +1 -1
  146. package/out/postcode-point-lookup.d.ts +6 -2
  147. package/out/postcode-point-lookup.d.ts.map +1 -1
  148. package/out/postcode-point-lookup.js +6 -2
  149. package/out/postcode-point-lookup.js.map +1 -1
  150. package/out/ranking-weights.d.ts +118 -0
  151. package/out/ranking-weights.d.ts.map +1 -0
  152. package/out/ranking-weights.js +44 -0
  153. package/out/ranking-weights.js.map +1 -0
  154. package/out/reverse.d.ts +9 -3
  155. package/out/reverse.d.ts.map +1 -1
  156. package/out/reverse.js +20 -6
  157. package/out/reverse.js.map +1 -1
  158. package/out/sharding.d.ts +3 -1
  159. package/out/sharding.d.ts.map +1 -1
  160. package/out/sharding.js +7 -5
  161. package/out/sharding.js.map +1 -1
  162. package/out/sqlite-convention-source.d.ts.map +1 -1
  163. package/out/sqlite-convention-source.js +3 -1
  164. package/out/sqlite-convention-source.js.map +1 -1
  165. package/out/street-centroid-schema.d.ts +33 -11
  166. package/out/street-centroid-schema.d.ts.map +1 -1
  167. package/out/street-centroid-schema.js +3 -1
  168. package/out/street-centroid-schema.js.map +1 -1
  169. package/out/street-centroid.d.ts.map +1 -1
  170. package/out/street-centroid.js +6 -2
  171. package/out/street-centroid.js.map +1 -1
  172. package/out/street-morphology-fst-builder.d.ts +6 -2
  173. package/out/street-morphology-fst-builder.d.ts.map +1 -1
  174. package/out/street-morphology-fst-builder.js +8 -7
  175. package/out/street-morphology-fst-builder.js.map +1 -1
  176. package/out/street-morphology-fst-loader.d.ts +24 -8
  177. package/out/street-morphology-fst-loader.d.ts.map +1 -1
  178. package/out/street-morphology-fst-loader.js +11 -5
  179. package/out/street-morphology-fst-loader.js.map +1 -1
  180. package/out/street-name-lookup.d.ts +9 -3
  181. package/out/street-name-lookup.d.ts.map +1 -1
  182. package/out/street-name-lookup.js +9 -7
  183. package/out/street-name-lookup.js.map +1 -1
  184. package/out/street-normalize.d.ts +3 -1
  185. package/out/street-normalize.d.ts.map +1 -1
  186. package/out/street-normalize.js +23 -13
  187. package/out/street-normalize.js.map +1 -1
  188. package/out/street-segment-schema.d.ts +48 -16
  189. package/out/street-segment-schema.d.ts.map +1 -1
  190. package/out/street-segment-schema.js +6 -2
  191. package/out/street-segment-schema.js.map +1 -1
  192. package/out/types.d.ts +18 -6
  193. package/out/types.d.ts.map +1 -1
  194. package/out/unified-schema.d.ts +1 -1
  195. package/out/unified-schema.d.ts.map +1 -1
  196. package/out/unified-schema.js +2 -2
  197. package/out/unified-schema.js.map +1 -1
  198. package/package.json +5 -5
  199. package/poi-lookup.ts +53 -21
  200. package/poi-schema.ts +43 -13
  201. package/postal-city-alias-lookup.ts +20 -6
  202. package/postal-city-alias-schema.ts +28 -9
  203. package/postal-city-candidate-schema.ts +15 -5
  204. package/postcode-point-lookup.ts +6 -2
  205. package/ranking-weights.ts +148 -0
  206. package/reverse.ts +47 -10
  207. package/sharding.ts +13 -6
  208. package/sqlite-convention-source.ts +4 -1
  209. package/street-centroid-schema.ts +35 -11
  210. package/street-centroid.ts +10 -3
  211. package/street-morphology-fst-builder.ts +25 -9
  212. package/street-morphology-fst-loader.ts +26 -10
  213. package/street-name-lookup.ts +19 -7
  214. package/street-normalize.ts +28 -13
  215. package/street-segment-schema.ts +50 -16
  216. package/types.ts +18 -6
  217. package/unified-schema.ts +11 -2
package/fst-builder.ts CHANGED
@@ -26,6 +26,7 @@ const DEFAULT_PLACETYPES: PlacetypeID[] = [
26
26
  "borough",
27
27
  "neighbourhood",
28
28
  ]
29
+
29
30
  const DEFAULT_COUNTRIES = ["US"]
30
31
  const DEFAULT_LANGUAGES = ["eng", ""]
31
32
 
@@ -66,6 +67,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
66
67
  // Phase 1: Load all matching SPR rows.
67
68
  progress("spr", `Loading places for countries=[${countries}], placetypes=[${placetypes}]`)
68
69
  const placeholders = (arr: string[]) => arr.map(() => "?").join(",")
70
+
69
71
  const sprStmt = db.prepare(
70
72
  `SELECT id, name, placetype, parent_id, latitude, longitude
71
73
  FROM spr
@@ -73,6 +75,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
73
75
  AND country IN (${placeholders(countries)})
74
76
  AND placetype IN (${placeholders(placetypes)})`
75
77
  )
78
+
76
79
  const sprRows = sprStmt.all(...countries, ...placetypes) as unknown as SprRow[]
77
80
  progress("spr", `Loaded ${sprRows.length} places`)
78
81
 
@@ -155,6 +158,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
155
158
  for (const row of impRows) {
156
159
  importanceMap.set(row.id, row.importance)
157
160
  }
161
+
158
162
  progress("importance", `Loaded ${importanceMap.size} importance scores`)
159
163
  } catch {
160
164
  progress("importance", "No place_importance table — falling back to population")
@@ -164,7 +168,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
164
168
  const popRows = popStmt.all() as unknown as PopulationRow[]
165
169
 
166
170
  for (const row of popRows) {
167
- const normalized = row.population > 0 ? Math.min(1.0, Math.log2(1 + row.population / 1000) / 14) : 0
171
+ const normalized = row.population > 0 ? Math.min(1, Math.log2(1 + row.population / 1000) / 14) : 0
168
172
  importanceMap.set(row.id, normalized)
169
173
  }
170
174
  } catch {
@@ -182,11 +186,13 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
182
186
  for (let i = 0; i < placeIds.length; i += 500) {
183
187
  const chunk = placeIds.slice(i, i + 500)
184
188
  const idPlaceholders = chunk.map(() => "?").join(",")
189
+
185
190
  const nameStmt = allLanguages
186
191
  ? db.prepare(`SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders})`)
187
192
  : db.prepare(
188
193
  `SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders}) AND language IN (${languages.map(() => "?").join(",")})`
189
194
  )
195
+
190
196
  const nameRows = (allLanguages
191
197
  ? nameStmt.all(...chunk)
192
198
  : nameStmt.all(...chunk, ...languages)) as unknown as NameRow[]
@@ -197,9 +203,11 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
197
203
  if (!existing.includes(row.name)) {
198
204
  existing.push(row.name)
199
205
  }
206
+
200
207
  namesByPlace.set(row.id, existing)
201
208
  }
202
209
  }
210
+
203
211
  progress("names", `Loaded names for ${namesByPlace.size} places`)
204
212
 
205
213
  // Phase 5: Build the trie.
@@ -213,7 +221,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
213
221
  let excludedCount = 0
214
222
 
215
223
  function isDegenerate(tokens: string[]): boolean {
216
- if (tokens.length === 0) return false
224
+ if (!tokens.length) return false
217
225
 
218
226
  if (excludeSurfaces?.has(tokens.join(" "))) return true
219
227
 
@@ -222,14 +230,20 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
222
230
  return false
223
231
  }
224
232
 
233
+ // Surface-ambiguity classes (survey #4): a per-SURFACE fact, so the entry is cloned per insertion
234
+ // with its accepting surface's count attached (the same place under "nyc" and "new york city"
235
+ // records each surface's own ambiguity). Absent map → entries carry no count (back-compat bytes).
236
+ const surfaceCountryCounts = opts.surfaceCountryCounts
237
+
225
238
  function insertName(tokens: string[], entry: PlaceEntry): boolean {
226
- if (tokens.length === 0) return false
239
+ if (!tokens.length) return false
227
240
 
228
241
  if (isDegenerate(tokens)) {
229
242
  excludedCount++
230
243
 
231
244
  return false
232
245
  }
246
+
233
247
  let stateID = 0
234
248
 
235
249
  for (const t of tokens) {
@@ -241,13 +255,20 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
241
255
  nodes.push({ edges: new Map(), places: [] })
242
256
  node.edges.set(t, next)
243
257
  }
258
+
244
259
  stateID = next
245
260
  }
261
+
246
262
  // Deduplicate: don't add the same wofID twice at the same state.
247
263
  const existing = nodes[stateID]!.places
248
264
 
249
265
  if (!existing.some((p) => p.wofID === entry.wofID && p.placetype === entry.placetype)) {
250
- existing.push(entry)
266
+ if (surfaceCountryCounts !== undefined) {
267
+ const count = surfaceCountryCounts.get(tokens.join(" "))
268
+ existing.push({ ...entry, crossCountryBranches: Math.min(count ?? 1, 255) })
269
+ } else {
270
+ existing.push(entry)
271
+ }
251
272
  }
252
273
 
253
274
  return true
@@ -257,6 +278,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
257
278
 
258
279
  for (const row of sprRows) {
259
280
  const parentChain = resolveParentChain(row.id)
281
+
260
282
  const entry: PlaceEntry = {
261
283
  wofID: row.id,
262
284
  placetype: row.placetype as PlacetypeID,
@@ -281,13 +303,14 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
281
303
  if (altName === row.name) continue
282
304
  const altTokens = normalizeTokens(altName)
283
305
 
284
- if (altTokens.length > 0 && altTokens.join(" ") !== primaryTokens.join(" ") && insertName(altTokens, entry)) {
306
+ if (altTokens.length && altTokens.join(" ") !== primaryTokens.join(" ") && insertName(altTokens, entry)) {
285
307
  insertCount++
286
308
  }
287
309
  }
288
310
  }
289
311
 
290
312
  db.close()
313
+
291
314
  progress(
292
315
  "done",
293
316
  `Built trie: ${nodes.length} states, ${insertCount} name insertions` +
@@ -296,6 +319,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
296
319
 
297
320
  const edgeCount = nodes.reduce((sum, n) => sum + n.edges.size, 0)
298
321
  const matcher = FSTMatcher.fromNodes(nodes)
322
+
299
323
  const provenance: FSTProvenance = {
300
324
  builtAt: new Date().toISOString(),
301
325
  countries,
@@ -14,13 +14,39 @@ import type { FSTNode } from "./fst-matcher.ts"
14
14
  import { FSTMatcher } from "./fst-matcher.ts"
15
15
  import type { FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
16
16
 
17
+ /**
18
+ * Format version that widened the per-state edge and place counters from 16 to 32 bits, growing the state entry from 12
19
+ * to 16 bytes. Readers branch on it to stay backward-compatible with v2/v3 files.
20
+ */
21
+ const VERSION_WIDE_STATE_COUNTERS = 4
22
+
23
+ /**
24
+ * State-table entry size in bytes at or above {@link VERSION_WIDE_STATE_COUNTERS}.
25
+ */
26
+ const WIDE_STATE_ENTRY_SIZE = 16
27
+
28
+ /**
29
+ * State-table entry size in bytes below {@link VERSION_WIDE_STATE_COUNTERS}.
30
+ */
31
+ const NARROW_STATE_ENTRY_SIZE = 12
32
+
33
+ /**
34
+ * First format version carrying the trailing metadata block; older files simply have none.
35
+ */
36
+ const VERSION_WITH_METADATA = 3
37
+
17
38
  const HEADER_SIZE = 32
18
39
  const EDGE_ENTRY_SIZE = 8
19
40
  const PLACE_ENTRY_SIZE = 56
20
- const MAGIC_BYTES = [0x46, 0x53, 0x54, 0x00] // "FST\0"
21
- // Must track the serializer's VERSION (fst-serialize.ts, currently 4). The v3 provenance + v4
22
- // 16-byte-state/u32-count layout logic below already matches the Node deserializer; only this gate
23
- // was left stale at 2, so the browser FST loader rejected every real (v4) artifact.
41
+ /**
42
+ * "FST\0".
43
+ */
44
+ const MAGIC_BYTES = [0x46, 0x53, 0x54, 0x00]
45
+ /**
46
+ * Must track the serializer's VERSION (fst-serialize.ts, currently 4). The v3 provenance + v4 16-byte-state/u32-count
47
+ * layout logic below already matches the Node deserializer; only this gate was left stale at 2, so the browser FST
48
+ * loader rejected every real (v4) artifact.
49
+ */
24
50
  const MAX_VERSION = 4
25
51
 
26
52
  const PLACETYPE_ORDER: readonly PlacetypeID[] = [
@@ -58,7 +84,10 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
58
84
  if (version < 1 || version > MAX_VERSION) {
59
85
  throw new Error(`FST version ${version} unsupported (expected 1..${MAX_VERSION})`)
60
86
  }
87
+
61
88
  const isV2 = version >= 2
89
+ // flags bit0 (survey #4, mirrors fst-serialize.ts): place rows carry surface-ambiguity data.
90
+ const hasAmbiguity = (view.getUint16(6, true) & 1) === 1
62
91
 
63
92
  const stateCount = view.getUint32(8, true)
64
93
  const edgeCount = view.getUint32(12, true)
@@ -75,6 +104,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
75
104
  strOffsets[i] = view.getUint32(pos, true)
76
105
  pos += 4
77
106
  }
107
+
78
108
  const strDataStart = pos
79
109
  const strings: string[] = new Array(stringCount)
80
110
 
@@ -83,10 +113,11 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
83
113
  const end = strDataStart + strOffsets[i + 1]!
84
114
  strings[i] = decoder.decode(bytes.subarray(start, end))
85
115
  }
116
+
86
117
  pos += stringBytes
87
118
 
88
119
  // --- State table ---
89
- const stateEntrySize = version >= 4 ? 16 : 12
120
+ const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
90
121
  const stateTableStart = pos
91
122
  const edgeTableStart = stateTableStart + stateCount * stateEntrySize
92
123
  const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
@@ -97,8 +128,12 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
97
128
  const sp = stateTableStart + si * stateEntrySize
98
129
  const edgeStart = view.getUint32(sp, true)
99
130
  const placeStart = view.getUint32(sp + 4, true)
100
- const edgeCountForState = version >= 4 ? view.getUint32(sp + 8, true) : view.getUint16(sp + 8, true)
101
- const placeCountForState = version >= 4 ? view.getUint32(sp + 12, true) : view.getUint16(sp + 10, true)
131
+
132
+ const edgeCountForState =
133
+ version >= VERSION_WIDE_STATE_COUNTERS ? view.getUint32(sp + 8, true) : view.getUint16(sp + 8, true)
134
+
135
+ const placeCountForState =
136
+ version >= VERSION_WIDE_STATE_COUNTERS ? view.getUint32(sp + 12, true) : view.getUint16(sp + 10, true)
102
137
 
103
138
  const edges = new Map<string, number>()
104
139
 
@@ -119,9 +154,10 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
119
154
  for (let ci = 0; ci < chainLen; ci++) {
120
155
  parentChain.push(view.getUint32(pp + 24 + ci * 4, true))
121
156
  }
157
+
122
158
  const rawImportance = isV2
123
159
  ? view.getFloat32(pp + 12, true)
124
- : Math.min(1.0, Math.log2(1 + view.getUint32(pp + 12, true) / 1000) / 14)
160
+ : Math.min(1, Math.log2(1 + view.getUint32(pp + 12, true) / 1000) / 14)
125
161
 
126
162
  places[pi] = {
127
163
  wofID: view.getUint32(pp, true),
@@ -131,6 +167,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
131
167
  lat: view.getFloat32(pp + 16, true),
132
168
  lon: view.getFloat32(pp + 20, true),
133
169
  parentChain,
170
+ ...(hasAmbiguity ? { crossCountryBranches: view.getUint8(pp + 6) } : {}),
134
171
  }
135
172
  }
136
173
 
@@ -148,7 +185,7 @@ export function readFSTProvenanceWeb(input: ArrayBuffer | Uint8Array): FSTProven
148
185
  if (bytes.byteLength < HEADER_SIZE) return undefined
149
186
  const version = view.getUint16(4, true)
150
187
 
151
- if (version < 3) return undefined
188
+ if (version < VERSION_WITH_METADATA) return undefined
152
189
  const provenanceOffset = view.getUint32(28, true)
153
190
 
154
191
  if (provenanceOffset === 0 || provenanceOffset >= bytes.byteLength) return undefined
package/fst-matcher.ts CHANGED
@@ -39,15 +39,16 @@ export class FSTMatcher {
39
39
  walk(tokens: string[]): FSTMatchResult | null {
40
40
  let stateID = 0
41
41
 
42
- for (let i = 0; i < tokens.length; i++) {
42
+ for (const token of tokens) {
43
43
  const node = this.nodes[stateID]
44
44
 
45
45
  if (!node) return null
46
- const next = node.edges.get(tokens[i]!)
46
+ const next = node.edges.get(token)
47
47
 
48
48
  if (next === undefined) return null
49
49
  stateID = next
50
50
  }
51
+
51
52
  const node = this.nodes[stateID]!
52
53
 
53
54
  return { stateID, accepted: node.places.length > 0, depth: tokens.length }
@@ -77,6 +78,7 @@ export class FSTMatcher {
77
78
 
78
79
  for (const [token, targetID] of node.edges) {
79
80
  const target = this.nodes[targetID]!
81
+
80
82
  result.push({
81
83
  token,
82
84
  targetState: targetID,
@@ -104,6 +106,7 @@ export class FSTMatcher {
104
106
 
105
107
  if (next === undefined) break
106
108
  stateID = next
109
+
107
110
  depth++
108
111
  }
109
112
 
@@ -127,7 +130,9 @@ export class FSTMatcher {
127
130
  return this.nodes.length
128
131
  }
129
132
 
130
- /** Expose the internal node array for serialization. */
133
+ /**
134
+ * Expose the internal node array for serialization.
135
+ */
131
136
  toNodes(): readonly FSTNode[] {
132
137
  return this.nodes
133
138
  }
@@ -137,12 +142,14 @@ export class FSTMatcher {
137
142
  }
138
143
  }
139
144
 
140
- /** Normalize text into FST tokens: lowercase, NFKC, strip punctuation, split on whitespace. */
145
+ /**
146
+ * Normalize text into FST tokens: lowercase, NFKC, strip punctuation, split on whitespace.
147
+ */
141
148
  export function normalizeTokens(text: string): string[] {
142
149
  return text
143
150
  .normalize("NFKC")
144
151
  .toLowerCase()
145
- .replace(/[\p{P}\p{S}]/gu, "")
152
+ .replaceAll(/[\p{P}\p{S}]/gu, "")
146
153
  .split(/\s+/)
147
154
  .filter((t) => t.length > 0)
148
155
  }
package/fst-serialize.ts CHANGED
@@ -27,14 +27,67 @@ import type { FSTNode } from "./fst-matcher.ts"
27
27
  import { FSTMatcher } from "./fst-matcher.ts"
28
28
  import type { FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
29
29
 
30
+ /**
31
+ * Format version that widened the per-state edge and place counters from 16 to 32 bits, growing the state entry from 12
32
+ * to 16 bytes. Readers branch on it to stay backward-compatible with v2/v3 files.
33
+ */
34
+ const VERSION_WIDE_STATE_COUNTERS = 4
35
+
36
+ /**
37
+ * State-table entry size in bytes at or above {@link VERSION_WIDE_STATE_COUNTERS}.
38
+ */
39
+ const WIDE_STATE_ENTRY_SIZE = 16
40
+
41
+ /**
42
+ * State-table entry size in bytes below {@link VERSION_WIDE_STATE_COUNTERS}.
43
+ */
44
+ const NARROW_STATE_ENTRY_SIZE = 12
45
+
46
+ /**
47
+ * First format version carrying the trailing metadata block; older files simply have none.
48
+ */
49
+ const VERSION_WITH_METADATA = 3
50
+
51
+ /**
52
+ * File magic. A reader rejects anything not starting with these four bytes before parsing further.
53
+ */
30
54
  const MAGIC = Buffer.from("FST\0", "ascii")
55
+
56
+ /**
57
+ * Format version this serializer emits. See {@link VERSION_WIDE_STATE_COUNTERS} for what each bump changed.
58
+ */
31
59
  const VERSION = 4
60
+
61
+ /**
62
+ * Fixed header size in bytes: magic, version, and the section offsets that follow it.
63
+ */
32
64
  const HEADER_SIZE = 32
65
+
66
+ /**
67
+ * State-table entry: edge offset, place offset, and the two 32-bit counters (v4 widths).
68
+ */
33
69
  const STATE_ENTRY_SIZE = 16
70
+
71
+ /**
72
+ * Edge-table entry: the transition label and the target state index.
73
+ */
34
74
  const EDGE_ENTRY_SIZE = 8
75
+
76
+ /**
77
+ * Place-table entry: the place id, its placetype, coordinates, and importance.
78
+ */
35
79
  const PLACE_ENTRY_SIZE = 56
80
+
81
+ /**
82
+ * Longest ancestry chain stored per place. Deeper hierarchies are truncated at the leaf end, since the specific end of
83
+ * the chain is what disambiguates and the country end is recoverable anyway.
84
+ */
36
85
  const MAX_CHAIN_LEN = 8
37
86
 
87
+ /**
88
+ * Placetypes in hierarchy order, largest first. The index into this array is what gets written into a place entry, so
89
+ * REORDERING IT BREAKS EVERY EXISTING FILE — append instead, and bump the version.
90
+ */
38
91
  const PLACETYPE_ORDER: readonly PlacetypeID[] = [
39
92
  "country",
40
93
  "region",
@@ -109,11 +162,15 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
109
162
  let pos = 0
110
163
 
111
164
  // --- Header ---
165
+ // flags bit0 (survey #4, 2026-07-27): place rows carry surface-ambiguity data in the former _pad
166
+ // byte (pp+6 = crossCountryBranches u8, pp+7 reserved). Presence-signaled here so VERSION stays
167
+ // put: pre-ambiguity artifacts read flags=0 → readers expose `undefined`, never a fake 0.
168
+ const hasAmbiguity = nodes.some((n) => n.places.some((p) => p.crossCountryBranches !== undefined))
112
169
  MAGIC.copy(buf, pos)
113
170
  pos += 4
114
171
  buf.writeUInt16LE(VERSION, pos)
115
172
  pos += 2
116
- buf.writeUInt16LE(0, pos)
173
+ buf.writeUInt16LE(hasAmbiguity ? 1 : 0, pos)
117
174
  pos += 2
118
175
  buf.writeUInt32LE(nodes.length, pos)
119
176
  pos += 4
@@ -131,11 +188,12 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
131
188
  // --- String table ---
132
189
  let strOffset = 0
133
190
 
134
- for (let i = 0; i < encodedStrings.length; i++) {
191
+ for (const encoded of encodedStrings) {
135
192
  buf.writeUInt32LE(strOffset, pos)
136
193
  pos += 4
137
- strOffset += encodedStrings[i]!.length
194
+ strOffset += encoded.length
138
195
  }
196
+
139
197
  buf.writeUInt32LE(strOffset, pos)
140
198
  pos += 4
141
199
 
@@ -167,6 +225,7 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
167
225
  const ep = edgeTableStart + edgeIdx * EDGE_ENTRY_SIZE
168
226
  buf.writeUInt32LE(intern(token), ep)
169
227
  buf.writeUInt32LE(target, ep + 4)
228
+
170
229
  edgeIdx++
171
230
  }
172
231
 
@@ -178,7 +237,9 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
178
237
  buf.writeUInt32LE(place.wofID, pp)
179
238
  buf.writeUInt8(placetypeToIdx.get(place.placetype) ?? 0, pp + 4)
180
239
  buf.writeUInt8(chainLen, pp + 5)
181
- buf.writeUInt16LE(0, pp + 6) // pad
240
+ // Former _pad: byte 0 = crossCountryBranches (header flags bit0 gates the read), byte 1 reserved.
241
+ buf.writeUInt8(hasAmbiguity ? Math.min(place.crossCountryBranches ?? 0, 255) : 0, pp + 6)
242
+ buf.writeUInt8(0, pp + 7)
182
243
  buf.writeUInt32LE(intern(place.name), pp + 8)
183
244
  buf.writeFloatLE(place.importance, pp + 12)
184
245
  buf.writeFloatLE(place.lat, pp + 16)
@@ -187,6 +248,7 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
187
248
  for (let ci = 0; ci < MAX_CHAIN_LEN; ci++) {
188
249
  buf.writeUInt32LE(ci < chainLen ? validChain[ci]! : 0, pp + 24 + ci * 4)
189
250
  }
251
+
190
252
  placeIdx++
191
253
  }
192
254
  }
@@ -209,6 +271,8 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
209
271
 
210
272
  if (version < 1 || version > VERSION) throw new Error(`FST version ${version} unsupported (expected 1..${VERSION})`)
211
273
  const isV2 = version >= 2
274
+ // flags bit0 (survey #4): place rows carry surface-ambiguity data in the former _pad byte.
275
+ const hasAmbiguity = (buf.readUInt16LE(6) & 1) === 1
212
276
 
213
277
  const stateCount = buf.readUInt32LE(8)
214
278
  const edgeCount = buf.readUInt32LE(12)
@@ -225,6 +289,7 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
225
289
  strOffsets[i] = buf.readUInt32LE(pos)
226
290
  pos += 4
227
291
  }
292
+
228
293
  const strDataStart = pos
229
294
  const strings: string[] = new Array(stringCount)
230
295
 
@@ -233,10 +298,11 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
233
298
  const end = strDataStart + strOffsets[i + 1]!
234
299
  strings[i] = buf.toString("utf8", start, end)
235
300
  }
301
+
236
302
  pos += stringBytes
237
303
 
238
304
  // --- State table ---
239
- const stateEntrySize = version >= 4 ? 16 : 12
305
+ const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
240
306
  const stateTableStart = pos
241
307
  const edgeTableStart = stateTableStart + stateCount * stateEntrySize
242
308
  const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
@@ -247,8 +313,12 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
247
313
  const sp = stateTableStart + si * stateEntrySize
248
314
  const edgeStart = buf.readUInt32LE(sp)
249
315
  const placeStart = buf.readUInt32LE(sp + 4)
250
- const edgeCountForState = version >= 4 ? buf.readUInt32LE(sp + 8) : buf.readUInt16LE(sp + 8)
251
- const placeCountForState = version >= 4 ? buf.readUInt32LE(sp + 12) : buf.readUInt16LE(sp + 10)
316
+
317
+ const edgeCountForState =
318
+ version >= VERSION_WIDE_STATE_COUNTERS ? buf.readUInt32LE(sp + 8) : buf.readUInt16LE(sp + 8)
319
+
320
+ const placeCountForState =
321
+ version >= VERSION_WIDE_STATE_COUNTERS ? buf.readUInt32LE(sp + 12) : buf.readUInt16LE(sp + 10)
252
322
 
253
323
  const edges = new Map<string, number>()
254
324
 
@@ -269,9 +339,11 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
269
339
  for (let ci = 0; ci < chainLen; ci++) {
270
340
  parentChain.push(buf.readUInt32LE(pp + 24 + ci * 4))
271
341
  }
342
+
272
343
  const rawImportance = isV2
273
344
  ? buf.readFloatLE(pp + 12)
274
- : Math.min(1.0, Math.log2(1 + buf.readUInt32LE(pp + 12) / 1000) / 14)
345
+ : Math.min(1, Math.log2(1 + buf.readUInt32LE(pp + 12) / 1000) / 14)
346
+
275
347
  places[pi] = {
276
348
  wofID: buf.readUInt32LE(pp),
277
349
  placetype: PLACETYPE_ORDER[buf.readUInt8(pp + 4)] ?? "locality",
@@ -280,6 +352,8 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
280
352
  lat: buf.readFloatLE(pp + 16),
281
353
  lon: buf.readFloatLE(pp + 20),
282
354
  parentChain,
355
+ // Header flags bit0 gates the read (survey #4): pre-ambiguity artifacts expose undefined.
356
+ ...(hasAmbiguity ? { crossCountryBranches: buf.readUInt8(pp + 6) } : {}),
283
357
  }
284
358
  }
285
359
 
@@ -295,7 +369,7 @@ export function readFSTProvenance(buf: Buffer): FSTProvenance | undefined {
295
369
  if (!buf.subarray(0, 4).equals(MAGIC)) return undefined
296
370
  const version = buf.readUInt16LE(4)
297
371
 
298
- if (version < 3) return undefined
372
+ if (version < VERSION_WITH_METADATA) return undefined
299
373
  const provenanceOffset = buf.readUInt32LE(28)
300
374
 
301
375
  if (provenanceOffset === 0 || provenanceOffset >= buf.length) return undefined
package/fst-types.ts CHANGED
@@ -16,6 +16,13 @@ export interface PlaceEntry {
16
16
  importance: number
17
17
  lat: number
18
18
  lon: number
19
+ /**
20
+ * Surface-ambiguity class (survey #4): how many DISTINCT countries carry a place with THIS entry's accepting surface,
21
+ * counted over the whole admin DB at build time (clamped to 255). A property of the surface, not the place — the same
22
+ * place reached via different alias surfaces reports each surface's own count. `undefined` = built without ambiguity
23
+ * data (pre-2026-07-27 artifacts) — NEVER conflate with 1 (the unambiguous case); the meaning-of-zero rule.
24
+ */
25
+ crossCountryBranches?: number
19
26
  }
20
27
 
21
28
  export type PlacetypeID =
@@ -60,9 +67,13 @@ export interface FSTProvenance {
60
67
  importanceMatches: number
61
68
  sourceDB?: string
62
69
  modelCardVersion?: string
63
- /** Degenerate-surface curation policy applied at build time (absent = uncurated build). */
70
+ /**
71
+ * Degenerate-surface curation policy applied at build time (absent = uncurated build).
72
+ */
64
73
  exclusionPolicy?: string
65
- /** Name insertions refused by the curation policy. */
74
+ /**
75
+ * Name insertions refused by the curation policy.
76
+ */
66
77
  excludedInsertions?: number
67
78
  }
68
79
 
@@ -85,8 +96,20 @@ export interface BuildFSTOpts {
85
96
  * street-type words ("Avenue Road" is a real name; "de la" is not).
86
97
  */
87
98
  excludeAllTokensOf?: ReadonlySet<string>
88
- /** Recorded verbatim into provenance when either exclusion set is supplied. */
99
+ /**
100
+ * Recorded verbatim into provenance when either exclusion set is supplied.
101
+ */
89
102
  exclusionPolicy?: string
103
+ /**
104
+ * Surface-ambiguity classes (survey #4, 2026-07-27): normalized-join surface → the number of DISTINCT countries
105
+ * (across the WHOLE admin DB, not just this build's country scope) with a place carrying that surface. When supplied,
106
+ * every inserted place row records the count for ITS accepting surface (`PlaceEntry.crossCountryBranches`) — an entry
107
+ * accessible under several surfaces records each surface's own count. Serialized into the place row's former `_pad`
108
+ * byte with presence signaled by header flags bit0, so VERSION stays put and pre-ambiguity artifacts read as "no
109
+ * data" (never "0 branches" — the meaning-of-zero rule). No decoder consumes it yet; consumers (FST-prior tempering,
110
+ * the Option-A evidence channel) arrive behind their own measured gates.
111
+ */
112
+ surfaceCountryCounts?: ReadonlyMap<string, number>
90
113
  onProgress?: (phase: string, detail?: string) => void
91
114
  }
92
115
 
package/fts-query.ts ADDED
@@ -0,0 +1,84 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Query shaping for the FTS5 lookup: placetype normalization and the MATCH-expression sanitizer.
7
+ * Both turn a caller's loose input into something SQLite's FTS5 parser accepts without throwing —
8
+ * an unescaped quote or a bare `*` is a syntax error, not an empty result.
9
+ */
10
+
11
+ import type { FindPlaceQuery, WOFPlacetype } from "./types.ts"
12
+
13
+ export function normalizePlacetypes(p: FindPlaceQuery["placetype"]): WOFPlacetype[] | null {
14
+ if (!p) return null
15
+
16
+ return Array.isArray(p) ? p : [p]
17
+ }
18
+
19
+ /**
20
+ * Make an arbitrary user-typed string safe for FTS5 MATCH.
21
+ *
22
+ * FTS5 has its own query syntax (`"phrase"`, `term1 OR term2`, `prefix*`, NEAR/N, etc.). Letting raw user input through
23
+ * means a user typing `Paris's` or `St. (Petersburg)` causes a syntax error.
24
+ *
25
+ * Per-token rules:
26
+ *
27
+ * - Strip all punctuation except trailing `*` from each whitespace-separated token.
28
+ * - **Trailing `*`** is preserved as FTS5 **prefix syntax** — `627*` becomes the literal `627*` (unquoted). The caller
29
+ * signaled they want a prefix; respect that.
30
+ * - All other tokens are wrapped in `"..."` as a single-word phrase. Conservative — handles apostrophes, parens, accented
31
+ * input, etc. safely.
32
+ * - Multiple tokens join with implicit AND.
33
+ *
34
+ * Examples:
35
+ *
36
+ * - `"Paris"` → `"Paris"` (phrase)
37
+ * - `"627*"` → `627*` (prefix)
38
+ * - `"St. (Petersburg)"` → `"St" "Petersburg"` (two phrases, AND-joined)
39
+ * - `"Thiron-Gardais"` → `"Thiron" "Gardais"` (intra-token punctuation SPLITS — #945; fusing to `ThironGardais` matched
40
+ * nothing because the FTS doc tokenizes the hyphenated name as two terms)
41
+ * - `"110 00"` with `fuseTokens` (postcode-typed) → `"110" "00"` per-token fused — the #920 name law
42
+ * - `"Pari* TX"` → `Pari* "TX"` (mixed prefix + phrase)
43
+ * - `"*"` alone → `""` (no body → drop)
44
+ */
45
+ export function sanitizeFTSQuery(text: string, opts?: { fuseTokens?: boolean }): string {
46
+ const out: string[] = []
47
+
48
+ for (const rawToken of text.normalize("NFKC").split(/\s+/u)) {
49
+ const trimmed = rawToken.trim()
50
+
51
+ if (!trimmed) continue
52
+ const hasPrefixStar = trimmed.endsWith("*")
53
+
54
+ // #920 name law (postcode-typed queries ONLY): delete intra-token punctuation and FUSE the
55
+ // remainder — postal names are stored in this collapsed shape ("SW1A" stays one term).
56
+ if (opts?.fuseTokens) {
57
+ const body = trimmed.replaceAll(/[^\p{L}\p{N}]/gu, "")
58
+
59
+ if (!body) continue
60
+ out.push(hasPrefixStar ? `${body}*` : `"${body.replaceAll('"', '""')}"`)
61
+
62
+ continue
63
+ }
64
+
65
+ // Everything else SPLITS on intra-token punctuation — the behavior the docstring always
66
+ // promised ("St. (Petersburg)" → two phrases). The old code DELETED punctuation instead,
67
+ // fusing "Thiron-Gardais" into the unmatchable single term `ThironGardais` while the FTS
68
+ // doc holds two terms (#945 — the entire hyphenated-name class missed at the raw lookup;
69
+ // masked for years because pre-splice tokenizers never emitted hyphen-preserved values).
70
+ const parts = trimmed.split(/[^\p{L}\p{N}]+/u).filter(Boolean)
71
+
72
+ if (!parts.length) continue
73
+
74
+ for (let i = 0; i < parts.length; i++) {
75
+ const body = parts[i]!.replaceAll("*", "")
76
+
77
+ if (!body) continue
78
+ // The caller's trailing `*` applies to the FINAL part ("Thiron-Gard*" → "Thiron" Gard*).
79
+ out.push(hasPrefixStar && i === parts.length - 1 ? `${body}*` : `"${body.replaceAll('"', '""')}"`)
80
+ }
81
+ }
82
+
83
+ return out.join(" ")
84
+ }