@mailwoman/resolver-wof-sqlite 9.0.0 → 9.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (297) hide show
  1. package/README.md +28 -9
  2. package/address-point-interpolation.ts +18 -8
  3. package/address-point-schema.ts +18 -6
  4. package/address-point.ts +111 -18
  5. package/ancestry.ts +9 -6
  6. package/build-candidate.ts +287 -157
  7. package/build-slim.ts +3 -3
  8. package/candidate/alias-bags.ts +54 -0
  9. package/candidate/ancestors-sidecar.ts +206 -0
  10. package/candidate/country-display-names.ts +79 -0
  11. package/candidate/name-roles.ts +237 -0
  12. package/candidate/own-name.ts +146 -0
  13. package/candidate/place-attrs.ts +44 -0
  14. package/candidate/shard-fold.ts +137 -0
  15. package/candidate-ancestors-schema.ts +195 -0
  16. package/candidate-fts.ts +4 -2
  17. package/candidate-importance.ts +228 -0
  18. package/candidate-lookup.ts +564 -174
  19. package/candidate-schema.ts +60 -3
  20. package/candidate-scoring.ts +268 -0
  21. package/capital-schema.ts +90 -0
  22. package/capitals.ts +148 -0
  23. package/coincident-roles.ts +69 -10
  24. package/convention-schema.ts +72 -0
  25. package/convention.ts +2 -2
  26. package/coverage-manifest-schema.ts +7 -7
  27. package/currency-backfill.ts +249 -0
  28. package/exact-match.ts +104 -0
  29. package/fst-autocomplete.ts +105 -122
  30. package/fst-builder.ts +39 -47
  31. package/fst-deserialize-web.ts +43 -7
  32. package/fst-freshness.ts +2 -2
  33. package/fst-serialize.ts +68 -12
  34. package/fst-types.ts +35 -1
  35. package/fts-query.ts +1 -1
  36. package/fts.ts +16 -4
  37. package/geonames-postal.ts +2 -2
  38. package/index.ts +26 -14
  39. package/interpolation.ts +113 -19
  40. package/lookup.ts +118 -560
  41. package/name-score.ts +6 -4
  42. package/out/address-point-interpolation.d.ts.map +1 -1
  43. package/out/address-point-interpolation.js +13 -7
  44. package/out/address-point-interpolation.js.map +1 -1
  45. package/out/address-point-schema.d.ts +16 -6
  46. package/out/address-point-schema.d.ts.map +1 -1
  47. package/out/address-point-schema.js.map +1 -1
  48. package/out/address-point.d.ts.map +1 -1
  49. package/out/address-point.js +70 -14
  50. package/out/address-point.js.map +1 -1
  51. package/out/ancestry.d.ts +2 -2
  52. package/out/ancestry.d.ts.map +1 -1
  53. package/out/ancestry.js +5 -6
  54. package/out/ancestry.js.map +1 -1
  55. package/out/build-candidate.d.ts +108 -0
  56. package/out/build-candidate.d.ts.map +1 -1
  57. package/out/build-candidate.js +151 -120
  58. package/out/build-candidate.js.map +1 -1
  59. package/out/build-slim.d.ts +1 -1
  60. package/out/build-slim.js +3 -3
  61. package/out/build-slim.js.map +1 -1
  62. package/out/candidate/alias-bags.d.ts +17 -0
  63. package/out/candidate/alias-bags.d.ts.map +1 -0
  64. package/out/candidate/alias-bags.js +39 -0
  65. package/out/candidate/alias-bags.js.map +1 -0
  66. package/out/candidate/ancestors-sidecar.d.ts +33 -0
  67. package/out/candidate/ancestors-sidecar.d.ts.map +1 -0
  68. package/out/candidate/ancestors-sidecar.js +140 -0
  69. package/out/candidate/ancestors-sidecar.js.map +1 -0
  70. package/out/candidate/country-display-names.d.ts +35 -0
  71. package/out/candidate/country-display-names.d.ts.map +1 -0
  72. package/out/candidate/country-display-names.js +59 -0
  73. package/out/candidate/country-display-names.js.map +1 -0
  74. package/out/candidate/name-roles.d.ts +55 -0
  75. package/out/candidate/name-roles.d.ts.map +1 -0
  76. package/out/candidate/name-roles.js +165 -0
  77. package/out/candidate/name-roles.js.map +1 -0
  78. package/out/candidate/own-name.d.ts +50 -0
  79. package/out/candidate/own-name.d.ts.map +1 -0
  80. package/out/candidate/own-name.js +132 -0
  81. package/out/candidate/own-name.js.map +1 -0
  82. package/out/candidate/place-attrs.d.ts +43 -0
  83. package/out/candidate/place-attrs.d.ts.map +1 -0
  84. package/out/candidate/place-attrs.js +15 -0
  85. package/out/candidate/place-attrs.js.map +1 -0
  86. package/out/candidate/shard-fold.d.ts +31 -0
  87. package/out/candidate/shard-fold.d.ts.map +1 -0
  88. package/out/candidate/shard-fold.js +104 -0
  89. package/out/candidate/shard-fold.js.map +1 -0
  90. package/out/candidate-ancestors-schema.d.ts +150 -0
  91. package/out/candidate-ancestors-schema.d.ts.map +1 -0
  92. package/out/candidate-ancestors-schema.js +123 -0
  93. package/out/candidate-ancestors-schema.js.map +1 -0
  94. package/out/candidate-fts.d.ts +4 -2
  95. package/out/candidate-fts.d.ts.map +1 -1
  96. package/out/candidate-fts.js +4 -2
  97. package/out/candidate-fts.js.map +1 -1
  98. package/out/candidate-importance.d.ts +132 -0
  99. package/out/candidate-importance.d.ts.map +1 -0
  100. package/out/candidate-importance.js +174 -0
  101. package/out/candidate-importance.js.map +1 -0
  102. package/out/candidate-lookup.d.ts +22 -37
  103. package/out/candidate-lookup.d.ts.map +1 -1
  104. package/out/candidate-lookup.js +446 -132
  105. package/out/candidate-lookup.js.map +1 -1
  106. package/out/candidate-schema.d.ts +52 -4
  107. package/out/candidate-schema.d.ts.map +1 -1
  108. package/out/candidate-schema.js +8 -0
  109. package/out/candidate-schema.js.map +1 -1
  110. package/out/candidate-scoring.d.ts +34 -0
  111. package/out/candidate-scoring.d.ts.map +1 -0
  112. package/out/candidate-scoring.js +200 -0
  113. package/out/candidate-scoring.js.map +1 -0
  114. package/out/capital-schema.d.ts +51 -0
  115. package/out/capital-schema.d.ts.map +1 -0
  116. package/out/capital-schema.js +63 -0
  117. package/out/capital-schema.js.map +1 -0
  118. package/out/capitals.d.ts +69 -0
  119. package/out/capitals.d.ts.map +1 -0
  120. package/out/capitals.js +98 -0
  121. package/out/capitals.js.map +1 -0
  122. package/out/coincident-roles.d.ts +7 -0
  123. package/out/coincident-roles.d.ts.map +1 -1
  124. package/out/coincident-roles.js +42 -8
  125. package/out/coincident-roles.js.map +1 -1
  126. package/out/convention-schema.d.ts +51 -0
  127. package/out/convention-schema.d.ts.map +1 -0
  128. package/out/convention-schema.js +34 -0
  129. package/out/convention-schema.js.map +1 -0
  130. package/out/convention.d.ts +1 -1
  131. package/out/convention.js +2 -2
  132. package/out/coverage-manifest-schema.js +3 -7
  133. package/out/coverage-manifest-schema.js.map +1 -1
  134. package/out/currency-backfill.d.ts +46 -0
  135. package/out/currency-backfill.d.ts.map +1 -0
  136. package/out/currency-backfill.js +180 -0
  137. package/out/currency-backfill.js.map +1 -0
  138. package/out/exact-match.d.ts +25 -0
  139. package/out/exact-match.d.ts.map +1 -0
  140. package/out/exact-match.js +89 -0
  141. package/out/exact-match.js.map +1 -0
  142. package/out/fst-autocomplete.d.ts +24 -14
  143. package/out/fst-autocomplete.d.ts.map +1 -1
  144. package/out/fst-autocomplete.js +84 -100
  145. package/out/fst-autocomplete.js.map +1 -1
  146. package/out/fst-builder.d.ts.map +1 -1
  147. package/out/fst-builder.js +32 -40
  148. package/out/fst-builder.js.map +1 -1
  149. package/out/fst-deserialize-web.d.ts.map +1 -1
  150. package/out/fst-deserialize-web.js +36 -7
  151. package/out/fst-deserialize-web.js.map +1 -1
  152. package/out/fst-freshness.d.ts +2 -2
  153. package/out/fst-freshness.js +2 -2
  154. package/out/fst-serialize.d.ts +14 -4
  155. package/out/fst-serialize.d.ts.map +1 -1
  156. package/out/fst-serialize.js +60 -12
  157. package/out/fst-serialize.js.map +1 -1
  158. package/out/fst-types.d.ts +35 -1
  159. package/out/fst-types.d.ts.map +1 -1
  160. package/out/fts-query.js +1 -1
  161. package/out/fts-query.js.map +1 -1
  162. package/out/fts.d.ts +15 -4
  163. package/out/fts.d.ts.map +1 -1
  164. package/out/fts.js +15 -4
  165. package/out/fts.js.map +1 -1
  166. package/out/geonames-postal.d.ts +2 -2
  167. package/out/geonames-postal.js +2 -2
  168. package/out/index.d.ts +4 -2
  169. package/out/index.d.ts.map +1 -1
  170. package/out/index.js +3 -2
  171. package/out/index.js.map +1 -1
  172. package/out/interpolation.d.ts +8 -0
  173. package/out/interpolation.d.ts.map +1 -1
  174. package/out/interpolation.js +91 -19
  175. package/out/interpolation.js.map +1 -1
  176. package/out/lookup.d.ts +4 -5
  177. package/out/lookup.d.ts.map +1 -1
  178. package/out/lookup.js +102 -444
  179. package/out/lookup.js.map +1 -1
  180. package/out/name-score.d.ts +0 -10
  181. package/out/name-score.d.ts.map +1 -1
  182. package/out/name-score.js +6 -4
  183. package/out/name-score.js.map +1 -1
  184. package/out/place-importance-schema.d.ts +226 -0
  185. package/out/place-importance-schema.d.ts.map +1 -0
  186. package/out/place-importance-schema.js +288 -0
  187. package/out/place-importance-schema.js.map +1 -0
  188. package/out/poi-lookup.d.ts +1 -1
  189. package/out/poi-lookup.d.ts.map +1 -1
  190. package/out/poi-lookup.js +12 -13
  191. package/out/poi-lookup.js.map +1 -1
  192. package/out/poi-schema.d.ts +7 -3
  193. package/out/poi-schema.d.ts.map +1 -1
  194. package/out/poi-schema.js.map +1 -1
  195. package/out/polygon-schema.d.ts +37 -0
  196. package/out/polygon-schema.d.ts.map +1 -0
  197. package/out/polygon-schema.js +23 -0
  198. package/out/polygon-schema.js.map +1 -0
  199. package/out/postal-city-alias-lookup.d.ts +1 -1
  200. package/out/postal-city-alias-lookup.js +1 -1
  201. package/out/postal-city-candidate-schema.d.ts +2 -1
  202. package/out/postal-city-candidate-schema.d.ts.map +1 -1
  203. package/out/postal-city-candidate-schema.js.map +1 -1
  204. package/out/postcode-point-lookup.d.ts +1 -1
  205. package/out/postcode-point-lookup.js +1 -1
  206. package/out/primary-preference.d.ts +125 -0
  207. package/out/primary-preference.d.ts.map +1 -0
  208. package/out/primary-preference.js +138 -0
  209. package/out/primary-preference.js.map +1 -0
  210. package/out/proximity-rerank.d.ts +77 -0
  211. package/out/proximity-rerank.d.ts.map +1 -0
  212. package/out/proximity-rerank.js +86 -0
  213. package/out/proximity-rerank.js.map +1 -0
  214. package/out/region-keys.d.ts +47 -0
  215. package/out/region-keys.d.ts.map +1 -0
  216. package/out/region-keys.js +121 -0
  217. package/out/region-keys.js.map +1 -0
  218. package/out/reverse.d.ts.map +1 -1
  219. package/out/reverse.js +6 -9
  220. package/out/reverse.js.map +1 -1
  221. package/out/schema.d.ts +1 -1
  222. package/out/search-fetch.d.ts +57 -0
  223. package/out/search-fetch.d.ts.map +1 -0
  224. package/out/search-fetch.js +183 -0
  225. package/out/search-fetch.js.map +1 -0
  226. package/out/sharding.d.ts +3 -3
  227. package/out/sharding.js +1 -1
  228. package/out/sqlite-convention-source.d.ts +1 -1
  229. package/out/sqlite-convention-source.js +1 -1
  230. package/out/sqlite-utils.d.ts +31 -1
  231. package/out/sqlite-utils.d.ts.map +1 -1
  232. package/out/sqlite-utils.js +38 -0
  233. package/out/sqlite-utils.js.map +1 -1
  234. package/out/street-centroid-schema.d.ts +7 -2
  235. package/out/street-centroid-schema.d.ts.map +1 -1
  236. package/out/street-centroid-schema.js.map +1 -1
  237. package/out/street-centroid.d.ts.map +1 -1
  238. package/out/street-centroid.js +7 -7
  239. package/out/street-centroid.js.map +1 -1
  240. package/out/street-morphology-fst-builder.d.ts.map +1 -1
  241. package/out/street-morphology-fst-builder.js +5 -4
  242. package/out/street-morphology-fst-builder.js.map +1 -1
  243. package/out/street-normalize.d.ts +83 -9
  244. package/out/street-normalize.d.ts.map +1 -1
  245. package/out/street-normalize.js +177 -10
  246. package/out/street-normalize.js.map +1 -1
  247. package/out/street-segment-schema.d.ts +6 -2
  248. package/out/street-segment-schema.d.ts.map +1 -1
  249. package/out/street-segment-schema.js.map +1 -1
  250. package/out/types.d.ts +74 -1
  251. package/out/types.d.ts.map +1 -1
  252. package/out/unified-schema.d.ts +1 -1
  253. package/out/unified-schema.js +1 -1
  254. package/out/uprn-lookup.d.ts +85 -0
  255. package/out/uprn-lookup.d.ts.map +1 -0
  256. package/out/uprn-lookup.js +152 -0
  257. package/out/uprn-lookup.js.map +1 -0
  258. package/out/uprn-schema.d.ts +93 -0
  259. package/out/uprn-schema.d.ts.map +1 -0
  260. package/out/uprn-schema.js +78 -0
  261. package/out/uprn-schema.js.map +1 -0
  262. package/out/weights-overlay-linker.d.ts +141 -0
  263. package/out/weights-overlay-linker.d.ts.map +1 -0
  264. package/out/weights-overlay-linker.js +259 -0
  265. package/out/weights-overlay-linker.js.map +1 -0
  266. package/package.json +296 -16
  267. package/place-importance-schema.ts +402 -0
  268. package/poi-lookup.ts +12 -13
  269. package/poi-schema.ts +8 -3
  270. package/polygon-schema.ts +47 -0
  271. package/postal-city-alias-lookup.ts +1 -1
  272. package/postal-city-candidate-schema.ts +3 -1
  273. package/postcode-point-lookup.ts +1 -1
  274. package/primary-preference.ts +207 -0
  275. package/proximity-rerank.ts +120 -0
  276. package/region-keys.ts +144 -0
  277. package/reverse.ts +17 -16
  278. package/schema.ts +1 -1
  279. package/search-fetch.ts +256 -0
  280. package/sharding.ts +3 -3
  281. package/sqlite-convention-source.ts +1 -1
  282. package/sqlite-utils.ts +63 -1
  283. package/street-centroid-schema.ts +8 -2
  284. package/street-centroid.ts +13 -8
  285. package/street-morphology-fst-builder.ts +5 -4
  286. package/street-normalize.ts +254 -24
  287. package/street-segment-schema.ts +7 -2
  288. package/types.ts +74 -1
  289. package/unified-schema.ts +1 -1
  290. package/uprn-lookup.ts +210 -0
  291. package/uprn-schema.ts +124 -0
  292. package/weights-overlay-linker.ts +377 -0
  293. package/geo.ts +0 -121
  294. package/out/geo.d.ts +0 -74
  295. package/out/geo.d.ts.map +0 -1
  296. package/out/geo.js +0 -71
  297. package/out/geo.js.map +0 -1
@@ -3,20 +3,29 @@
3
3
  * @license AGPL-3.0
4
4
  * @author Teffen Ellis, et al.
5
5
  *
6
- * FST-based autocomplete. Prefix walk + BFS expansion to collect ranked place suggestions. O(depth
7
- * × branching) — the FST IS the autocomplete index.
6
+ * FST-based autocomplete — mailwoman vocabulary over `@mailwoman/ancestrie`'s generic algorithm
7
+ * (#1728 phase 2). The #587 behavior — prefix walk + BFS expansion, partial-last-token completion,
8
+ * per-branch capping, dedupe — lives in ancestrie's `autocomplete`; this module contributes only
9
+ * the storage adapter ({@link FSTMatcher} → `AncestrieReaderLike`) and the mapping back to
10
+ * mailwoman's suggestion shape (name, placetype, referential/encyclopedic, WOF ids).
8
11
  *
9
- * Two query shapes are handled (the FST is a trie over normalized WORD tokens):
10
- *
11
- * - COMPLETE tokens ("new york") — `walk` lands on a state; collect its accepting entries + BFS a
12
- * couple tokens past it for nearby completions. This is the CLI's "complete a place word"
13
- * path.
14
- * - A PARTIAL last token ("new yor", "chic") — `walk` fails (there is no "yor" edge, only "york"). So
15
- * walk the complete prefix, then complete the partial token by prefix-filtering the
16
- * continuation edges (`token.startsWith(partial)`). This is what a char-level typeahead
17
- * needs; without it "new yor" returns nothing useful. (#587)
12
+ * THE BYTES DO NOT MIGRATE. The shipped artifacts are `FST\0` v1–v5 (`fst-serialize.ts`), not
13
+ * ancestrie's `ANCT`: ancestrie entries are id-keyed with one record per id, while an FST place row
14
+ * is per-(surface, place) — `crossCountryBranches` is a property of the SURFACE, so the same wofID
15
+ * legitimately carries different values under different aliases and cannot be represented id-keyed.
16
+ * The matcher, both deserializers, and the serializer therefore stay here; what migrated is the
17
+ * ALGORITHM, which is the half that drifts (the #861 share-the-function rule).
18
18
  */
19
19
 
20
+ import type {
21
+ AncestrieContinuation,
22
+ AncestrieMatch,
23
+ AncestrieReaderLike,
24
+ AncestrieRecord,
25
+ AncestrieSuggestion,
26
+ } from "@mailwoman/ancestrie"
27
+ import { autocomplete as ancestrieAutocomplete } from "@mailwoman/ancestrie"
28
+
20
29
  import type { FSTMatcher } from "./fst-matcher.ts"
21
30
  import { normalizeTokens } from "./fst-matcher.ts"
22
31
  import type { PlaceEntry } from "./fst-types.ts"
@@ -31,7 +40,17 @@ export interface AutocompleteResult {
31
40
  export interface AutocompleteSuggestion {
32
41
  name: string
33
42
  placetype: string
34
- importance: number
43
+ /**
44
+ * The REFERENTIAL likelihood the suggestion is ranked by (ROAD_TO_V9 §2). Autocomplete answers "which place does the
45
+ * user mean", so it ranks referentially like everything else; encyclopedic importance rides along on
46
+ * {@link AutocompleteSuggestion.encyclopedic} for display and never enters the order.
47
+ */
48
+ referential: number
49
+ /**
50
+ * Encyclopedic (Wikipedia) importance, when the FST artifact carries one for this place. `undefined` = no article, or
51
+ * a pre-v5 binary — never 0.
52
+ */
53
+ encyclopedic?: number
35
54
  wofID: number
36
55
  parentChain: number[]
37
56
  matchDepth: number
@@ -42,152 +61,116 @@ export interface AutocompleteOpts {
42
61
  maxSuggestions?: number
43
62
  maxExpansionDepth?: number
44
63
  /**
45
- * Collapse same-name suggestions to the single highest-importance one. Off by default (the CLI surfaces distinct
64
+ * Collapse same-name suggestions to the single highest-referential one. Off by default (the CLI surfaces distinct
46
65
  * same-name places — New York the city vs the county); a typeahead wants it ON so the dropdown isn't four "New
47
66
  * London"s. (#587)
48
67
  */
49
68
  dedupeByName?: boolean
50
69
  }
51
70
 
52
- interface BfsItem {
53
- stateID: number
54
- depth: number
55
- tokens: string[]
56
- }
57
-
58
71
  /**
59
72
  * Max accepting entries collected per BFS branch — keeps one dense branch from starving the search.
60
73
  */
61
74
  const PER_BRANCH = 4
62
75
 
63
76
  /**
64
- * The top-`k` entries by importance (descending). Avoids sorting/allocating when `entries` is small.
77
+ * The top-`k` entries by REFERENTIAL likelihood (descending). Avoids sorting/allocating when `entries` is small — and
78
+ * that shortcut is part of the observable contract: at or under `k` the INSERTION order is served, which decides
79
+ * suggestion order among referential ties.
65
80
  */
66
- function topByImportance(entries: readonly PlaceEntry[], k: number): PlaceEntry[] {
81
+ function topByReferential(entries: readonly PlaceEntry[], k: number): PlaceEntry[] {
67
82
  if (entries.length <= k) return [...entries]
68
83
 
69
- return [...entries].toSorted((a, b) => b.importance - a.importance).slice(0, k)
84
+ return [...entries].toSorted((a, b) => b.referential - a.referential).slice(0, k)
70
85
  }
71
86
 
72
87
  /**
73
- * Autocomplete from the current prefix. Returns suggestions ranked importance-descending.
88
+ * {@link FSTMatcher} presented through ancestrie's storage seam. Records carry the {@link PlaceEntry} itself as the
89
+ * payload, so the entry that WINS the algorithm's shallowest-depth rule is the entry whose fields the suggestion
90
+ * reports — a side lookup keyed on id could pick a different surface's row (`crossCountryBranches` differs per
91
+ * surface).
74
92
  */
75
- export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteOpts = {}): AutocompleteResult {
76
- const maxSuggestions = opts.maxSuggestions ?? 10
77
- const maxExpansionDepth = opts.maxExpansionDepth ?? 2
78
- const normalizedTokens = normalizeTokens(query)
79
-
80
- if (!normalizedTokens.length) {
81
- return { query, normalizedTokens: [], depth: 0, suggestions: [] }
82
- }
83
-
84
- const seen = new Map<number, AutocompleteSuggestion>()
85
- const queue: BfsItem[] = []
86
- let depth: number
87
-
88
- const match = fst.walk(normalizedTokens)
89
-
90
- if (match) {
91
- // COMPLETE-token prefix landed on a state. Seed at the match state (accepting + continuations).
92
- depth = match.depth
93
-
94
- for (const entry of fst.accepting(match.stateID)) {
95
- addSuggestion(seen, entry, match.depth, [])
96
- }
97
-
98
- for (const cont of fst.continuations(match.stateID)) {
99
- queue.push({ stateID: cont.targetState, depth: 1, tokens: [cont.token] })
100
- }
101
- } else {
102
- // PARTIAL last token — walk the complete prefix, complete the partial by prefix-filtering edges.
103
- const complete = normalizedTokens.slice(0, -1)
104
- const partial = normalizedTokens.at(-1)!
105
- const prefixState = !complete.length ? 0 : (fst.walk(complete)?.stateID ?? undefined)
93
+ class FSTReader implements AncestrieReaderLike<PlaceEntry> {
94
+ readonly #fst: FSTMatcher
106
95
 
107
- if (prefixState === undefined) {
108
- return { query, normalizedTokens, depth: 0, suggestions: [] }
109
- }
110
-
111
- depth = complete.length
112
-
113
- for (const cont of fst.continuations(prefixState)) {
114
- if (!cont.token.startsWith(partial)) continue
96
+ /**
97
+ * Parent chains of the entries this reader has served, id-keyed. A place's chain is identical across its surfaces (it
98
+ * is place-row data), so last-write-wins is safe. The algorithm asks {@link FSTReader.ancestorsOf} only for ids it
99
+ * just received from {@link FSTReader.entriesAt}, so serving from this memo answers every real call without an
100
+ * artifact-wide id index.
101
+ */
102
+ readonly #chains = new Map<number, number[]>()
115
103
 
116
- // This edge completes the typed partial token — its target is a real match at depth+1.
117
- for (const entry of topByImportance(fst.accepting(cont.targetState), PER_BRANCH)) {
118
- addSuggestion(seen, entry, complete.length + 1, [cont.token])
119
- }
104
+ constructor(fst: FSTMatcher) {
105
+ this.#fst = fst
106
+ }
120
107
 
121
- // BFS a little past it too (multi-token completions: "new yor" → "New York Mills").
122
- queue.push({ stateID: cont.targetState, depth: 1, tokens: [cont.token] })
123
- }
108
+ walk(tokens: readonly string[]): AncestrieMatch | null {
109
+ return this.#fst.walk([...tokens])
124
110
  }
125
111
 
126
- // BFS expansion (shared by both paths) — find nearby completions up to maxExpansionDepth. Each
127
- // branch contributes only its top PER_BRANCH places: a state like "new london" has dozens of
128
- // accepting entries and would otherwise blow the budget before the BFS ever reaches "new york"
129
- // (the "new" state has 311 continuations). Per-branch capping keeps the search broad. (#587)
130
- while (queue.length && seen.size < maxSuggestions * 4) {
131
- const item = queue.shift()!
112
+ continuations(stateID: number): AncestrieContinuation[] {
113
+ // Insertion order, verbatim — BFS visit order under the suggestion budget depends on it.
114
+ return this.#fst.continuations(stateID).map((c) => ({
115
+ token: c.token,
116
+ targetState: c.targetState,
117
+ entryCount: c.acceptingCount,
118
+ }))
119
+ }
132
120
 
133
- if (item.depth > maxExpansionDepth) continue
121
+ entriesAt(stateID: number, limit?: number): AncestrieRecord<PlaceEntry>[] {
122
+ const places = this.#fst.accepting(stateID)
123
+ const selected = limit === undefined ? places : topByReferential(places, limit)
134
124
 
135
- for (const entry of topByImportance(fst.accepting(item.stateID), PER_BRANCH)) {
136
- addSuggestion(seen, entry, depth + item.depth, item.tokens)
137
- }
125
+ return selected.map((entry) => {
126
+ this.#chains.set(entry.wofID, entry.parentChain)
138
127
 
139
- if (item.depth < maxExpansionDepth) {
140
- for (const cont of fst.continuations(item.stateID)) {
141
- queue.push({ stateID: cont.targetState, depth: item.depth + 1, tokens: [...item.tokens, cont.token] })
128
+ return {
129
+ id: entry.wofID,
130
+ rank: entry.referential,
131
+ parentIDs: entry.parentChain,
132
+ payload: entry,
142
133
  }
143
- }
134
+ })
144
135
  }
145
136
 
146
- let suggestions = [...seen.values()].toSorted((a, b) => b.importance - a.importance)
147
-
148
- if (opts.dedupeByName) {
149
- suggestions = dedupeByName(suggestions)
137
+ ancestorsOf(id: number): number[] {
138
+ return this.#chains.get(id) ?? []
150
139
  }
151
-
152
- return { query, normalizedTokens, depth, suggestions: suggestions.slice(0, maxSuggestions) }
153
- }
154
-
155
- function addSuggestion(
156
- seen: Map<number, AutocompleteSuggestion>,
157
- entry: PlaceEntry,
158
- matchDepth: number,
159
- completionTokens: string[]
160
- ): void {
161
- const existing = seen.get(entry.wofID)
162
-
163
- if (existing && existing.matchDepth <= matchDepth) return
164
-
165
- seen.set(entry.wofID, {
166
- name: entry.name,
167
- placetype: entry.placetype,
168
- importance: entry.importance,
169
- wofID: entry.wofID,
170
- parentChain: entry.parentChain,
171
- matchDepth,
172
- completionTokens: [...completionTokens],
173
- })
174
140
  }
175
141
 
176
142
  /**
177
- * Keep one suggestion per name — the highest-importance. Input is already importance-sorted, so the first occurrence
178
- * per name wins; order is preserved.
143
+ * Autocomplete from the current prefix. Returns suggestions ranked referential-descending.
179
144
  */
180
- function dedupeByName(suggestions: AutocompleteSuggestion[]): AutocompleteSuggestion[] {
181
- const seenNames = new Set<string>()
182
- const out: AutocompleteSuggestion[] = []
145
+ export function autocomplete(fst: FSTMatcher, query: string, opts: AutocompleteOpts = {}): AutocompleteResult {
146
+ const normalizedTokens = normalizeTokens(query)
183
147
 
184
- for (const s of suggestions) {
185
- const key = s.name.toLowerCase()
148
+ const result = ancestrieAutocomplete<PlaceEntry>(new FSTReader(fst), normalizedTokens, {
149
+ ...(opts.maxSuggestions === undefined ? {} : { maxSuggestions: opts.maxSuggestions }),
150
+ ...(opts.maxExpansionDepth === undefined ? {} : { maxExpansionDepth: opts.maxExpansionDepth }),
151
+ perBranchLimit: PER_BRANCH,
152
+ // The dedupe key is the DISPLAY name, not the token path: two surfaces of one name must still collapse. (#587)
153
+ ...(opts.dedupeByName ? { dedupe: (s: AncestrieSuggestion<PlaceEntry>) => s.payload!.name.toLowerCase() } : {}),
154
+ })
186
155
 
187
- if (seenNames.has(key)) continue
188
- seenNames.add(key)
189
- out.push(s)
156
+ return {
157
+ query,
158
+ normalizedTokens,
159
+ depth: result.depth,
160
+ suggestions: result.suggestions.map((s) => {
161
+ // Every record this adapter serves carries its entry; the assertion documents the invariant.
162
+ const entry = s.payload!
163
+
164
+ return {
165
+ name: entry.name,
166
+ placetype: entry.placetype,
167
+ referential: entry.referential,
168
+ ...(entry.encyclopedic === undefined ? {} : { encyclopedic: entry.encyclopedic }),
169
+ wofID: s.id,
170
+ parentChain: s.parentIDs,
171
+ matchDepth: s.matchDepth,
172
+ completionTokens: s.completionTokens,
173
+ }
174
+ }),
190
175
  }
191
-
192
- return out
193
176
  }
package/fst-builder.ts CHANGED
@@ -17,6 +17,8 @@ import { readWOFSourceIdentity } from "./fst-freshness.ts"
17
17
  import type { FSTNode } from "./fst-matcher.ts"
18
18
  import { FSTMatcher, normalizeTokens } from "./fst-matcher.ts"
19
19
  import type { BuildFSTOpts, BuildFSTResult, FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
20
+ import { loadImportanceSplit } from "./place-importance-schema.ts"
21
+ import { allRows, getRow } from "./sqlite-utils.ts"
20
22
 
21
23
  const DEFAULT_PLACETYPES: PlacetypeID[] = [
22
24
  "country",
@@ -52,11 +54,6 @@ interface NameRow {
52
54
  privateuse: string
53
55
  }
54
56
 
55
- interface PopulationRow {
56
- id: number
57
- population: number
58
- }
59
-
60
57
  export function buildFSTFromWOF(opts: BuildFSTOpts): {
61
58
  matcher: FSTMatcher
62
59
  provenance: FSTProvenance
@@ -82,7 +79,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
82
79
  AND placetype IN (${placeholders(placetypes)})`
83
80
  )
84
81
 
85
- const sprRows = sprStmt.all(...countries, ...placetypes) as unknown as SprRow[]
82
+ const sprRows = allRows<SprRow>(sprStmt, ...countries, ...placetypes)
86
83
  progress("spr", `Loaded ${sprRows.length} places`)
87
84
 
88
85
  // Phase 2: Build a lookup for parent chain resolution.
@@ -107,8 +104,8 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
107
104
  for (let i = 0; i < orphanIDs.length; i += ANCESTOR_CHUNK) {
108
105
  const chunk = orphanIDs.slice(i, i + ANCESTOR_CHUNK)
109
106
 
110
- const rows = db
111
- .prepare(
107
+ const rows = allRows<{ id: number; ancestor_id: number }>(
108
+ db.prepare(
112
109
  `SELECT DISTINCT id, ancestor_id FROM ancestors
113
110
  WHERE id IN (${chunk.map(() => "?").join(",")}) AND ancestor_placetype IN ('country', 'region', 'county')
114
111
  ORDER BY id, CASE ancestor_placetype
@@ -116,8 +113,9 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
116
113
  WHEN 'region' THEN 2
117
114
  WHEN 'country' THEN 3
118
115
  END`
119
- )
120
- .all(...chunk) as unknown as Array<{ id: number; ancestor_id: number }>
116
+ ),
117
+ ...chunk
118
+ )
121
119
 
122
120
  for (const row of rows) {
123
121
  let chain = ancestorsByID.get(row.id)
@@ -154,7 +152,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
154
152
  let parentRow = sprByID.get(current)
155
153
 
156
154
  if (!parentRow) {
157
- const fetched = parentStmt.get(current) as unknown as SprRow | undefined
155
+ const fetched = getRow<SprRow>(parentStmt, current)
158
156
 
159
157
  if (!fetched) break
160
158
  parentRow = fetched
@@ -171,45 +169,32 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
171
169
  return chain
172
170
  }
173
171
 
174
- // Phase 3: Load importance data (Wikipedia-based, falls back to population-scaled).
175
- // See docs/articles/concepts/importance-vs-population.md for the two-signal contract.
176
- progress("importance", "Loading importance data")
177
- const importanceMap = new Map<number, number>()
172
+ // Phase 3: Load BOTH scores (ROAD_TO_V9 §2 R1, the two-score split).
173
+ //
174
+ // Referential is ALWAYS population-anchored and never read out of a legacy `place_importance`
175
+ // column, because a legacy row that got a Wikipedia score overwrote whatever population would have
176
+ // said and the two are indistinguishable afterwards. Encyclopedic rides along for consumers and is
177
+ // never handed to the decoder. `loadImportanceSplit` handles all four schema generations; the
178
+ // source it reports is stamped into provenance so an artifact says which one it read.
179
+ progress("importance", "Loading referential + encyclopedic scores")
180
+ const split = loadImportanceSplit(db)
178
181
 
179
- try {
180
- const impStmt = db.prepare("SELECT id, importance FROM place_importance")
181
- const impRows = impStmt.all() as unknown as Array<{ id: number; importance: number }>
182
-
183
- for (const row of impRows) {
184
- importanceMap.set(row.id, row.importance)
185
- }
186
-
187
- progress("importance", `Loaded ${importanceMap.size} importance scores`)
188
- } catch {
189
- progress("importance", "No place_importance table — falling back to population")
190
-
191
- try {
192
- const popStmt = db.prepare("SELECT id, population FROM place_population")
193
- const popRows = popStmt.all() as unknown as PopulationRow[]
194
-
195
- for (const row of popRows) {
196
- const normalized = row.population > 0 ? Math.min(1, Math.log2(1 + row.population / 1000) / 14) : 0
197
- importanceMap.set(row.id, normalized)
198
- }
199
- } catch {
200
- progress("importance", "No place_population either — using 0 for all")
201
- }
202
- }
182
+ progress(
183
+ "importance",
184
+ `${split.referential.size} referential, ${split.encyclopedic.size} encyclopedic (source: ${split.source}` +
185
+ (split.legacyFallbackRows ? `, ${split.legacyFallbackRows} legacy rows attributed to the population pass` : "") +
186
+ ")"
187
+ )
203
188
 
204
189
  // Phase 4: Load names for matching places.
205
190
  progress("names", "Loading name variants")
206
- const placeIds = sprRows.map((r) => r.id)
191
+ const placeIDs = sprRows.map((r) => r.id)
207
192
  const namesByPlace = new Map<number, string[]>()
208
193
 
209
194
  const allLanguages = languages.includes("*")
210
195
 
211
- for (let i = 0; i < placeIds.length; i += 500) {
212
- const chunk = placeIds.slice(i, i + 500)
196
+ for (let i = 0; i < placeIDs.length; i += 500) {
197
+ const chunk = placeIDs.slice(i, i + 500)
213
198
  const idPlaceholders = chunk.map(() => "?").join(",")
214
199
 
215
200
  const nameStmt = allLanguages
@@ -218,9 +203,9 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
218
203
  `SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders}) AND language IN (${languages.map(() => "?").join(",")})`
219
204
  )
220
205
 
221
- const nameRows = (allLanguages
222
- ? nameStmt.all(...chunk)
223
- : nameStmt.all(...chunk, ...languages)) as unknown as NameRow[]
206
+ const nameRows = allLanguages
207
+ ? allRows<NameRow>(nameStmt, ...chunk)
208
+ : allRows<NameRow>(nameStmt, ...chunk, ...languages)
224
209
 
225
210
  for (const row of nameRows) {
226
211
  const existing = namesByPlace.get(row.id) ?? []
@@ -304,12 +289,17 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
304
289
  for (const row of sprRows) {
305
290
  const parentChain = resolveParentChain(row.id)
306
291
 
292
+ const encyclopedic = split.encyclopedic.get(row.id)
293
+
307
294
  const entry: PlaceEntry = {
308
295
  wofID: row.id,
309
296
  placetype: row.placetype as PlacetypeID,
310
297
  name: row.name,
311
298
  parentChain,
312
- importance: importanceMap.get(row.id) ?? 0,
299
+ referential: split.referential.get(row.id) ?? 0,
300
+ // Spread rather than assigned: a place with no Wikipedia article must carry NO field, not a
301
+ // zero. The serializer's per-place presence bit reads `!== undefined`.
302
+ ...(encyclopedic === undefined ? {} : { encyclopedic }),
313
303
  lat: row.latitude,
314
304
  lon: row.longitude,
315
305
  }
@@ -361,7 +351,9 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
361
351
  placeCount: sprRows.length,
362
352
  edgeCount,
363
353
  nameInsertions: insertCount,
364
- importanceMatches: importanceMap.size,
354
+ importanceMatches: split.referential.size,
355
+ encyclopedicMatches: split.encyclopedic.size,
356
+ importanceSource: split.source,
365
357
  sourceDB: opts.dbPath,
366
358
  sourceDBMD5: source.md5,
367
359
  sourceDBBytes: source.bytes,
@@ -39,17 +39,44 @@ const VERSION_WITH_METADATA = 3
39
39
 
40
40
  const HEADER_SIZE = 32
41
41
  const EDGE_ENTRY_SIZE = 8
42
- const PLACE_ENTRY_SIZE = 56
42
+
43
+ /**
44
+ * Format version that split the single `importance` float into `referential` + `encyclopedic` (ROAD_TO_V9 §2 R1),
45
+ * growing the place entry from 56 to 60 bytes. Mirrors `fst-serialize.ts`'s constant of the same name.
46
+ */
47
+ const VERSION_TWO_SCORE_SPLIT = 5
48
+
49
+ /**
50
+ * Place-entry size in bytes at or above {@link VERSION_TWO_SCORE_SPLIT}.
51
+ */
52
+ const SPLIT_PLACE_ENTRY_SIZE = 60
53
+
54
+ /**
55
+ * Place-entry size in bytes below {@link VERSION_TWO_SCORE_SPLIT}.
56
+ */
57
+ const LEGACY_PLACE_ENTRY_SIZE = 56
58
+
59
+ /**
60
+ * Byte offset of the encyclopedic float inside a v5 place entry — immediately after the 8-slot parent chain.
61
+ */
62
+ const ENCYCLOPEDIC_OFFSET = 56
63
+
64
+ /**
65
+ * `placeFlags` bit 0 (byte `pp+7`, v5+): this place carries an encyclopedic score. Per-place because absence is the
66
+ * common case; see `fst-serialize.ts`.
67
+ */
68
+ const PLACE_FLAG_HAS_ENCYCLOPEDIC = 1
43
69
  /**
44
70
  * "FST\0".
45
71
  */
46
72
  const MAGIC_BYTES = [0x46, 0x53, 0x54, 0x00]
47
73
  /**
48
- * Must track the serializer's VERSION (fst-serialize.ts, currently 4). The v3 provenance + v4 16-byte-state/u32-count
74
+ * Must track the serializer's VERSION (fst-serialize.ts, currently 5). The v3 provenance + v4 16-byte-state/u32-count
49
75
  * layout logic below already matches the Node deserializer; only this gate was left stale at 2, so the browser FST
50
- * loader rejected every real (v4) artifact.
76
+ * loader rejected every real (v4) artifact. It was left stale again at 4 by the v5 two-score split until this line
77
+ * moved with it — the gate is a SEPARATE number from the layout branches, which is exactly why it keeps drifting.
51
78
  */
52
- const MAX_VERSION = 4
79
+ const MAX_VERSION = 5
53
80
 
54
81
  const PLACETYPE_ORDER: readonly PlacetypeID[] = [
55
82
  "country",
@@ -88,6 +115,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
88
115
  }
89
116
 
90
117
  const isV2 = version >= 2
118
+ const isSplit = version >= VERSION_TWO_SCORE_SPLIT
91
119
  // flags bit0 (survey #4, mirrors fst-serialize.ts): place rows carry surface-ambiguity data.
92
120
  const hasAmbiguity = (view.getUint16(6, true) & 1) === 1
93
121
 
@@ -120,6 +148,8 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
120
148
 
121
149
  // --- State table ---
122
150
  const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
151
+ // v5 grew the place entry by the encyclopedic float; v4-and-below files are read at the old stride.
152
+ const placeEntrySize = isSplit ? SPLIT_PLACE_ENTRY_SIZE : LEGACY_PLACE_ENTRY_SIZE
123
153
  const stateTableStart = pos
124
154
  const edgeTableStart = stateTableStart + stateCount * stateEntrySize
125
155
  const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
@@ -149,7 +179,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
149
179
  const places: PlaceEntry[] = new Array(placeCountForState)
150
180
 
151
181
  for (let pi = 0; pi < placeCountForState; pi++) {
152
- const pp = placeTableStart + (placeStart + pi) * PLACE_ENTRY_SIZE
182
+ const pp = placeTableStart + (placeStart + pi) * placeEntrySize
153
183
  const chainLen = view.getUint8(pp + 5)
154
184
  const parentChain: number[] = []
155
185
 
@@ -157,19 +187,25 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
157
187
  parentChain.push(view.getUint32(pp + 24 + ci * 4, true))
158
188
  }
159
189
 
160
- const rawImportance = isV2
190
+ // v1 stored a raw population u32 here; v2-v4 the conflated `importance` float; v5 the
191
+ // referential score. See the Node deserializer for why a v1 value is genuinely referential.
192
+ const referential = isV2
161
193
  ? view.getFloat32(pp + 12, true)
162
194
  : Math.min(1, Math.log2(1 + view.getUint32(pp + 12, true) / 1000) / 14)
163
195
 
196
+ const hasEncyclopedic =
197
+ isSplit && (view.getUint8(pp + 7) & PLACE_FLAG_HAS_ENCYCLOPEDIC) === PLACE_FLAG_HAS_ENCYCLOPEDIC
198
+
164
199
  places[pi] = {
165
200
  wofID: view.getUint32(pp, true),
166
201
  placetype: PLACETYPE_ORDER[view.getUint8(pp + 4)] ?? "locality",
167
202
  name: strings[view.getUint32(pp + 8, true)]!,
168
- importance: rawImportance,
203
+ referential,
169
204
  lat: view.getFloat32(pp + 16, true),
170
205
  lon: view.getFloat32(pp + 20, true),
171
206
  parentChain,
172
207
  ...(hasAmbiguity ? { crossCountryBranches: view.getUint8(pp + 6) } : {}),
208
+ ...(hasEncyclopedic ? { encyclopedic: view.getFloat32(pp + ENCYCLOPEDIC_OFFSET, true) } : {}),
173
209
  }
174
210
  }
175
211
 
package/fst-freshness.ts CHANGED
@@ -15,7 +15,7 @@
15
15
  * `FSTProvenance` already recorded `sourceDB` — the source's PATH, which is exactly the field that
16
16
  * cannot change when the bytes behind it do. So the stamp gains the source's IDENTITY (md5 + byte
17
17
  * size) and this module compares it. The shape deliberately mirrors
18
- * `scripts/weights-overlay-linker.ts`'s `pairIndexStaleReason`: one function returning a reason
18
+ * `@mailwoman/resolver-wof-sqlite/weights-overlay-linker`'s `pairIndexStaleReason`: one function returning a reason
19
19
  * string or `undefined`, so a fact added to the stamp cannot be checked by some callers and not
20
20
  * others — which is how three of the four base linkers ended up unable to notice a PIX1 schema bump.
21
21
  *
@@ -162,7 +162,7 @@ export function peekFSTStampFields(path: string): FSTStampFields | undefined {
162
162
  *
163
163
  * The async `md5File` in `@mailwoman/core/utils` is the one to reach for anywhere else. This exists because the FST
164
164
  * builder and its whole call chain are synchronous by design (`buildFSTFromWOF` → `buildLocaleFSTs`), and making them
165
- * async to stamp a checksum would cascade through the Pastel commands and the tests for one hash. It reads in
165
+ * async to stamp a checksum would cascade through command callers and tests for one hash. It reads in
166
166
  * {@link MD5_CHUNK_BYTES} chunks rather than `readFileSync` — the source is a multi-gigabyte database.
167
167
  */
168
168
  export function md5FileSync(path: string): string {