@mailwoman/resolver-wof-sqlite 8.1.0 → 8.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/address-point-interpolation.ts +9 -3
- package/address-point-schema.ts +32 -10
- package/address-point.ts +3 -0
- package/ancestry-backfill.ts +18 -5
- package/ancestry.ts +7 -2
- package/build-candidate.ts +29 -6
- package/build-slim.ts +40 -11
- package/candidate-fts.ts +1 -0
- package/candidate-lookup.ts +36 -14
- package/candidate-schema.ts +35 -11
- package/coincident-roles.ts +28 -6
- package/convention.ts +3 -1
- package/coverage-manifest-schema.ts +49 -16
- package/fst-autocomplete.ts +15 -9
- package/fst-builder.ts +29 -5
- package/fst-deserialize-web.ts +46 -9
- package/fst-matcher.ts +12 -5
- package/fst-serialize.ts +83 -9
- package/fst-types.ts +26 -3
- package/fts-query.ts +84 -0
- package/fts.ts +35 -9
- package/geo.ts +9 -3
- package/geonames-aliases.ts +116 -79
- package/geonames-postal.ts +25 -5
- package/index.ts +10 -0
- package/interpolation.ts +33 -55
- package/lookup.ts +103 -292
- package/name-score.ts +76 -0
- package/out/address-point-interpolation.d.ts.map +1 -1
- package/out/address-point-interpolation.js +4 -2
- package/out/address-point-interpolation.js.map +1 -1
- package/out/address-point-schema.d.ts +30 -10
- package/out/address-point-schema.d.ts.map +1 -1
- package/out/address-point-schema.js +6 -2
- package/out/address-point-schema.js.map +1 -1
- package/out/address-point.d.ts.map +1 -1
- package/out/address-point.js.map +1 -1
- package/out/ancestry-backfill.d.ts +6 -2
- package/out/ancestry-backfill.d.ts.map +1 -1
- package/out/ancestry-backfill.js +7 -3
- package/out/ancestry-backfill.js.map +1 -1
- package/out/ancestry.d.ts +6 -2
- package/out/ancestry.d.ts.map +1 -1
- package/out/ancestry.js +3 -1
- package/out/ancestry.js.map +1 -1
- package/out/build-candidate.d.ts +9 -3
- package/out/build-candidate.d.ts.map +1 -1
- package/out/build-candidate.js +5 -3
- package/out/build-candidate.js.map +1 -1
- package/out/build-slim.d.ts +15 -5
- package/out/build-slim.d.ts.map +1 -1
- package/out/build-slim.js +11 -5
- package/out/build-slim.js.map +1 -1
- package/out/candidate-fts.d.ts.map +1 -1
- package/out/candidate-fts.js.map +1 -1
- package/out/candidate-lookup.d.ts +9 -3
- package/out/candidate-lookup.d.ts.map +1 -1
- package/out/candidate-lookup.js +16 -11
- package/out/candidate-lookup.js.map +1 -1
- package/out/candidate-schema.d.ts +33 -11
- package/out/candidate-schema.d.ts.map +1 -1
- package/out/candidate-schema.js.map +1 -1
- package/out/coincident-roles.d.ts +16 -4
- package/out/coincident-roles.d.ts.map +1 -1
- package/out/coincident-roles.js +9 -3
- package/out/coincident-roles.js.map +1 -1
- package/out/convention.d.ts +3 -1
- package/out/convention.d.ts.map +1 -1
- package/out/convention.js.map +1 -1
- package/out/coverage-manifest-schema.d.ts +45 -14
- package/out/coverage-manifest-schema.d.ts.map +1 -1
- package/out/coverage-manifest-schema.js +14 -5
- package/out/coverage-manifest-schema.js.map +1 -1
- package/out/fst-autocomplete.d.ts +1 -1
- package/out/fst-autocomplete.d.ts.map +1 -1
- package/out/fst-autocomplete.js +11 -9
- package/out/fst-autocomplete.js.map +1 -1
- package/out/fst-builder.d.ts.map +1 -1
- package/out/fst-builder.js +15 -5
- package/out/fst-builder.js.map +1 -1
- package/out/fst-deserialize-web.d.ts.map +1 -1
- package/out/fst-deserialize-web.js +34 -9
- package/out/fst-deserialize-web.js.map +1 -1
- package/out/fst-matcher.d.ts +6 -2
- package/out/fst-matcher.d.ts.map +1 -1
- package/out/fst-matcher.js +9 -5
- package/out/fst-matcher.js.map +1 -1
- package/out/fst-serialize.d.ts.map +1 -1
- package/out/fst-serialize.js +62 -9
- package/out/fst-serialize.js.map +1 -1
- package/out/fst-types.d.ts +26 -3
- package/out/fst-types.d.ts.map +1 -1
- package/out/fts-query.d.ts +41 -0
- package/out/fts-query.d.ts.map +1 -0
- package/out/fts-query.js +75 -0
- package/out/fts-query.js.map +1 -0
- package/out/fts.d.ts +21 -7
- package/out/fts.d.ts.map +1 -1
- package/out/fts.js +10 -4
- package/out/fts.js.map +1 -1
- package/out/geo.d.ts +6 -2
- package/out/geo.d.ts.map +1 -1
- package/out/geo.js +3 -1
- package/out/geo.js.map +1 -1
- package/out/geonames-aliases.d.ts +12 -4
- package/out/geonames-aliases.d.ts.map +1 -1
- package/out/geonames-aliases.js +72 -67
- package/out/geonames-aliases.js.map +1 -1
- package/out/geonames-postal.d.ts +9 -3
- package/out/geonames-postal.d.ts.map +1 -1
- package/out/geonames-postal.js +7 -2
- package/out/geonames-postal.js.map +1 -1
- package/out/index.d.ts.map +1 -1
- package/out/index.js.map +1 -1
- package/out/interpolation.d.ts +18 -6
- package/out/interpolation.d.ts.map +1 -1
- package/out/interpolation.js +11 -40
- package/out/interpolation.js.map +1 -1
- package/out/lookup.d.ts +3 -97
- package/out/lookup.d.ts.map +1 -1
- package/out/lookup.js +52 -184
- package/out/lookup.js.map +1 -1
- package/out/name-score.d.ts +28 -0
- package/out/name-score.d.ts.map +1 -0
- package/out/name-score.js +67 -0
- package/out/name-score.js.map +1 -0
- package/out/poi-lookup.d.ts +24 -8
- package/out/poi-lookup.d.ts.map +1 -1
- package/out/poi-lookup.js +27 -13
- package/out/poi-lookup.js.map +1 -1
- package/out/poi-schema.d.ts +42 -13
- package/out/poi-schema.d.ts.map +1 -1
- package/out/poi-schema.js +12 -3
- package/out/poi-schema.js.map +1 -1
- package/out/postal-city-alias-lookup.d.ts +18 -6
- package/out/postal-city-alias-lookup.d.ts.map +1 -1
- package/out/postal-city-alias-lookup.js.map +1 -1
- package/out/postal-city-alias-schema.d.ts +27 -9
- package/out/postal-city-alias-schema.d.ts.map +1 -1
- package/out/postal-city-alias-schema.js +3 -1
- package/out/postal-city-alias-schema.js.map +1 -1
- package/out/postal-city-candidate-schema.d.ts +15 -5
- package/out/postal-city-candidate-schema.d.ts.map +1 -1
- package/out/postal-city-candidate-schema.js +3 -1
- package/out/postal-city-candidate-schema.js.map +1 -1
- package/out/postcode-point-lookup.d.ts +6 -2
- package/out/postcode-point-lookup.d.ts.map +1 -1
- package/out/postcode-point-lookup.js +6 -2
- package/out/postcode-point-lookup.js.map +1 -1
- package/out/ranking-weights.d.ts +118 -0
- package/out/ranking-weights.d.ts.map +1 -0
- package/out/ranking-weights.js +44 -0
- package/out/ranking-weights.js.map +1 -0
- package/out/reverse.d.ts +9 -3
- package/out/reverse.d.ts.map +1 -1
- package/out/reverse.js +20 -6
- package/out/reverse.js.map +1 -1
- package/out/sharding.d.ts +3 -1
- package/out/sharding.d.ts.map +1 -1
- package/out/sharding.js +7 -5
- package/out/sharding.js.map +1 -1
- package/out/sqlite-convention-source.d.ts.map +1 -1
- package/out/sqlite-convention-source.js +3 -1
- package/out/sqlite-convention-source.js.map +1 -1
- package/out/street-centroid-schema.d.ts +33 -11
- package/out/street-centroid-schema.d.ts.map +1 -1
- package/out/street-centroid-schema.js +3 -1
- package/out/street-centroid-schema.js.map +1 -1
- package/out/street-centroid.d.ts.map +1 -1
- package/out/street-centroid.js +6 -2
- package/out/street-centroid.js.map +1 -1
- package/out/street-morphology-fst-builder.d.ts +6 -2
- package/out/street-morphology-fst-builder.d.ts.map +1 -1
- package/out/street-morphology-fst-builder.js +8 -7
- package/out/street-morphology-fst-builder.js.map +1 -1
- package/out/street-morphology-fst-loader.d.ts +24 -8
- package/out/street-morphology-fst-loader.d.ts.map +1 -1
- package/out/street-morphology-fst-loader.js +11 -5
- package/out/street-morphology-fst-loader.js.map +1 -1
- package/out/street-name-lookup.d.ts +9 -3
- package/out/street-name-lookup.d.ts.map +1 -1
- package/out/street-name-lookup.js +9 -7
- package/out/street-name-lookup.js.map +1 -1
- package/out/street-normalize.d.ts +3 -1
- package/out/street-normalize.d.ts.map +1 -1
- package/out/street-normalize.js +23 -13
- package/out/street-normalize.js.map +1 -1
- package/out/street-segment-schema.d.ts +48 -16
- package/out/street-segment-schema.d.ts.map +1 -1
- package/out/street-segment-schema.js +6 -2
- package/out/street-segment-schema.js.map +1 -1
- package/out/types.d.ts +18 -6
- package/out/types.d.ts.map +1 -1
- package/out/unified-schema.d.ts +1 -1
- package/out/unified-schema.d.ts.map +1 -1
- package/out/unified-schema.js +2 -2
- package/out/unified-schema.js.map +1 -1
- package/package.json +5 -5
- package/poi-lookup.ts +53 -21
- package/poi-schema.ts +43 -13
- package/postal-city-alias-lookup.ts +20 -6
- package/postal-city-alias-schema.ts +28 -9
- package/postal-city-candidate-schema.ts +15 -5
- package/postcode-point-lookup.ts +6 -2
- package/ranking-weights.ts +148 -0
- package/reverse.ts +47 -10
- package/sharding.ts +13 -6
- package/sqlite-convention-source.ts +4 -1
- package/street-centroid-schema.ts +35 -11
- package/street-centroid.ts +10 -3
- package/street-morphology-fst-builder.ts +25 -9
- package/street-morphology-fst-loader.ts +26 -10
- package/street-name-lookup.ts +19 -7
- package/street-normalize.ts +28 -13
- package/street-segment-schema.ts +50 -16
- package/types.ts +18 -6
- package/unified-schema.ts +11 -2
package/fst-builder.ts
CHANGED
|
@@ -26,6 +26,7 @@ const DEFAULT_PLACETYPES: PlacetypeID[] = [
|
|
|
26
26
|
"borough",
|
|
27
27
|
"neighbourhood",
|
|
28
28
|
]
|
|
29
|
+
|
|
29
30
|
const DEFAULT_COUNTRIES = ["US"]
|
|
30
31
|
const DEFAULT_LANGUAGES = ["eng", ""]
|
|
31
32
|
|
|
@@ -66,6 +67,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
66
67
|
// Phase 1: Load all matching SPR rows.
|
|
67
68
|
progress("spr", `Loading places for countries=[${countries}], placetypes=[${placetypes}]`)
|
|
68
69
|
const placeholders = (arr: string[]) => arr.map(() => "?").join(",")
|
|
70
|
+
|
|
69
71
|
const sprStmt = db.prepare(
|
|
70
72
|
`SELECT id, name, placetype, parent_id, latitude, longitude
|
|
71
73
|
FROM spr
|
|
@@ -73,6 +75,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
73
75
|
AND country IN (${placeholders(countries)})
|
|
74
76
|
AND placetype IN (${placeholders(placetypes)})`
|
|
75
77
|
)
|
|
78
|
+
|
|
76
79
|
const sprRows = sprStmt.all(...countries, ...placetypes) as unknown as SprRow[]
|
|
77
80
|
progress("spr", `Loaded ${sprRows.length} places`)
|
|
78
81
|
|
|
@@ -155,6 +158,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
155
158
|
for (const row of impRows) {
|
|
156
159
|
importanceMap.set(row.id, row.importance)
|
|
157
160
|
}
|
|
161
|
+
|
|
158
162
|
progress("importance", `Loaded ${importanceMap.size} importance scores`)
|
|
159
163
|
} catch {
|
|
160
164
|
progress("importance", "No place_importance table — falling back to population")
|
|
@@ -164,7 +168,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
164
168
|
const popRows = popStmt.all() as unknown as PopulationRow[]
|
|
165
169
|
|
|
166
170
|
for (const row of popRows) {
|
|
167
|
-
const normalized = row.population > 0 ? Math.min(1
|
|
171
|
+
const normalized = row.population > 0 ? Math.min(1, Math.log2(1 + row.population / 1000) / 14) : 0
|
|
168
172
|
importanceMap.set(row.id, normalized)
|
|
169
173
|
}
|
|
170
174
|
} catch {
|
|
@@ -182,11 +186,13 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
182
186
|
for (let i = 0; i < placeIds.length; i += 500) {
|
|
183
187
|
const chunk = placeIds.slice(i, i + 500)
|
|
184
188
|
const idPlaceholders = chunk.map(() => "?").join(",")
|
|
189
|
+
|
|
185
190
|
const nameStmt = allLanguages
|
|
186
191
|
? db.prepare(`SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders})`)
|
|
187
192
|
: db.prepare(
|
|
188
193
|
`SELECT id, name, language, privateuse FROM names WHERE id IN (${idPlaceholders}) AND language IN (${languages.map(() => "?").join(",")})`
|
|
189
194
|
)
|
|
195
|
+
|
|
190
196
|
const nameRows = (allLanguages
|
|
191
197
|
? nameStmt.all(...chunk)
|
|
192
198
|
: nameStmt.all(...chunk, ...languages)) as unknown as NameRow[]
|
|
@@ -197,9 +203,11 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
197
203
|
if (!existing.includes(row.name)) {
|
|
198
204
|
existing.push(row.name)
|
|
199
205
|
}
|
|
206
|
+
|
|
200
207
|
namesByPlace.set(row.id, existing)
|
|
201
208
|
}
|
|
202
209
|
}
|
|
210
|
+
|
|
203
211
|
progress("names", `Loaded names for ${namesByPlace.size} places`)
|
|
204
212
|
|
|
205
213
|
// Phase 5: Build the trie.
|
|
@@ -213,7 +221,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
213
221
|
let excludedCount = 0
|
|
214
222
|
|
|
215
223
|
function isDegenerate(tokens: string[]): boolean {
|
|
216
|
-
if (tokens.length
|
|
224
|
+
if (!tokens.length) return false
|
|
217
225
|
|
|
218
226
|
if (excludeSurfaces?.has(tokens.join(" "))) return true
|
|
219
227
|
|
|
@@ -222,14 +230,20 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
222
230
|
return false
|
|
223
231
|
}
|
|
224
232
|
|
|
233
|
+
// Surface-ambiguity classes (survey #4): a per-SURFACE fact, so the entry is cloned per insertion
|
|
234
|
+
// with its accepting surface's count attached (the same place under "nyc" and "new york city"
|
|
235
|
+
// records each surface's own ambiguity). Absent map → entries carry no count (back-compat bytes).
|
|
236
|
+
const surfaceCountryCounts = opts.surfaceCountryCounts
|
|
237
|
+
|
|
225
238
|
function insertName(tokens: string[], entry: PlaceEntry): boolean {
|
|
226
|
-
if (tokens.length
|
|
239
|
+
if (!tokens.length) return false
|
|
227
240
|
|
|
228
241
|
if (isDegenerate(tokens)) {
|
|
229
242
|
excludedCount++
|
|
230
243
|
|
|
231
244
|
return false
|
|
232
245
|
}
|
|
246
|
+
|
|
233
247
|
let stateID = 0
|
|
234
248
|
|
|
235
249
|
for (const t of tokens) {
|
|
@@ -241,13 +255,20 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
241
255
|
nodes.push({ edges: new Map(), places: [] })
|
|
242
256
|
node.edges.set(t, next)
|
|
243
257
|
}
|
|
258
|
+
|
|
244
259
|
stateID = next
|
|
245
260
|
}
|
|
261
|
+
|
|
246
262
|
// Deduplicate: don't add the same wofID twice at the same state.
|
|
247
263
|
const existing = nodes[stateID]!.places
|
|
248
264
|
|
|
249
265
|
if (!existing.some((p) => p.wofID === entry.wofID && p.placetype === entry.placetype)) {
|
|
250
|
-
|
|
266
|
+
if (surfaceCountryCounts !== undefined) {
|
|
267
|
+
const count = surfaceCountryCounts.get(tokens.join(" "))
|
|
268
|
+
existing.push({ ...entry, crossCountryBranches: Math.min(count ?? 1, 255) })
|
|
269
|
+
} else {
|
|
270
|
+
existing.push(entry)
|
|
271
|
+
}
|
|
251
272
|
}
|
|
252
273
|
|
|
253
274
|
return true
|
|
@@ -257,6 +278,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
257
278
|
|
|
258
279
|
for (const row of sprRows) {
|
|
259
280
|
const parentChain = resolveParentChain(row.id)
|
|
281
|
+
|
|
260
282
|
const entry: PlaceEntry = {
|
|
261
283
|
wofID: row.id,
|
|
262
284
|
placetype: row.placetype as PlacetypeID,
|
|
@@ -281,13 +303,14 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
281
303
|
if (altName === row.name) continue
|
|
282
304
|
const altTokens = normalizeTokens(altName)
|
|
283
305
|
|
|
284
|
-
if (altTokens.length
|
|
306
|
+
if (altTokens.length && altTokens.join(" ") !== primaryTokens.join(" ") && insertName(altTokens, entry)) {
|
|
285
307
|
insertCount++
|
|
286
308
|
}
|
|
287
309
|
}
|
|
288
310
|
}
|
|
289
311
|
|
|
290
312
|
db.close()
|
|
313
|
+
|
|
291
314
|
progress(
|
|
292
315
|
"done",
|
|
293
316
|
`Built trie: ${nodes.length} states, ${insertCount} name insertions` +
|
|
@@ -296,6 +319,7 @@ export function buildFSTFromWOF(opts: BuildFSTOpts): {
|
|
|
296
319
|
|
|
297
320
|
const edgeCount = nodes.reduce((sum, n) => sum + n.edges.size, 0)
|
|
298
321
|
const matcher = FSTMatcher.fromNodes(nodes)
|
|
322
|
+
|
|
299
323
|
const provenance: FSTProvenance = {
|
|
300
324
|
builtAt: new Date().toISOString(),
|
|
301
325
|
countries,
|
package/fst-deserialize-web.ts
CHANGED
|
@@ -14,13 +14,39 @@ import type { FSTNode } from "./fst-matcher.ts"
|
|
|
14
14
|
import { FSTMatcher } from "./fst-matcher.ts"
|
|
15
15
|
import type { FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
|
|
16
16
|
|
|
17
|
+
/**
|
|
18
|
+
* Format version that widened the per-state edge and place counters from 16 to 32 bits, growing the state entry from 12
|
|
19
|
+
* to 16 bytes. Readers branch on it to stay backward-compatible with v2/v3 files.
|
|
20
|
+
*/
|
|
21
|
+
const VERSION_WIDE_STATE_COUNTERS = 4
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* State-table entry size in bytes at or above {@link VERSION_WIDE_STATE_COUNTERS}.
|
|
25
|
+
*/
|
|
26
|
+
const WIDE_STATE_ENTRY_SIZE = 16
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* State-table entry size in bytes below {@link VERSION_WIDE_STATE_COUNTERS}.
|
|
30
|
+
*/
|
|
31
|
+
const NARROW_STATE_ENTRY_SIZE = 12
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* First format version carrying the trailing metadata block; older files simply have none.
|
|
35
|
+
*/
|
|
36
|
+
const VERSION_WITH_METADATA = 3
|
|
37
|
+
|
|
17
38
|
const HEADER_SIZE = 32
|
|
18
39
|
const EDGE_ENTRY_SIZE = 8
|
|
19
40
|
const PLACE_ENTRY_SIZE = 56
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
41
|
+
/**
|
|
42
|
+
* "FST\0".
|
|
43
|
+
*/
|
|
44
|
+
const MAGIC_BYTES = [0x46, 0x53, 0x54, 0x00]
|
|
45
|
+
/**
|
|
46
|
+
* Must track the serializer's VERSION (fst-serialize.ts, currently 4). The v3 provenance + v4 16-byte-state/u32-count
|
|
47
|
+
* layout logic below already matches the Node deserializer; only this gate was left stale at 2, so the browser FST
|
|
48
|
+
* loader rejected every real (v4) artifact.
|
|
49
|
+
*/
|
|
24
50
|
const MAX_VERSION = 4
|
|
25
51
|
|
|
26
52
|
const PLACETYPE_ORDER: readonly PlacetypeID[] = [
|
|
@@ -58,7 +84,10 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
|
|
|
58
84
|
if (version < 1 || version > MAX_VERSION) {
|
|
59
85
|
throw new Error(`FST version ${version} unsupported (expected 1..${MAX_VERSION})`)
|
|
60
86
|
}
|
|
87
|
+
|
|
61
88
|
const isV2 = version >= 2
|
|
89
|
+
// flags bit0 (survey #4, mirrors fst-serialize.ts): place rows carry surface-ambiguity data.
|
|
90
|
+
const hasAmbiguity = (view.getUint16(6, true) & 1) === 1
|
|
62
91
|
|
|
63
92
|
const stateCount = view.getUint32(8, true)
|
|
64
93
|
const edgeCount = view.getUint32(12, true)
|
|
@@ -75,6 +104,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
|
|
|
75
104
|
strOffsets[i] = view.getUint32(pos, true)
|
|
76
105
|
pos += 4
|
|
77
106
|
}
|
|
107
|
+
|
|
78
108
|
const strDataStart = pos
|
|
79
109
|
const strings: string[] = new Array(stringCount)
|
|
80
110
|
|
|
@@ -83,10 +113,11 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
|
|
|
83
113
|
const end = strDataStart + strOffsets[i + 1]!
|
|
84
114
|
strings[i] = decoder.decode(bytes.subarray(start, end))
|
|
85
115
|
}
|
|
116
|
+
|
|
86
117
|
pos += stringBytes
|
|
87
118
|
|
|
88
119
|
// --- State table ---
|
|
89
|
-
const stateEntrySize = version >=
|
|
120
|
+
const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
|
|
90
121
|
const stateTableStart = pos
|
|
91
122
|
const edgeTableStart = stateTableStart + stateCount * stateEntrySize
|
|
92
123
|
const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
|
|
@@ -97,8 +128,12 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
|
|
|
97
128
|
const sp = stateTableStart + si * stateEntrySize
|
|
98
129
|
const edgeStart = view.getUint32(sp, true)
|
|
99
130
|
const placeStart = view.getUint32(sp + 4, true)
|
|
100
|
-
|
|
101
|
-
const
|
|
131
|
+
|
|
132
|
+
const edgeCountForState =
|
|
133
|
+
version >= VERSION_WIDE_STATE_COUNTERS ? view.getUint32(sp + 8, true) : view.getUint16(sp + 8, true)
|
|
134
|
+
|
|
135
|
+
const placeCountForState =
|
|
136
|
+
version >= VERSION_WIDE_STATE_COUNTERS ? view.getUint32(sp + 12, true) : view.getUint16(sp + 10, true)
|
|
102
137
|
|
|
103
138
|
const edges = new Map<string, number>()
|
|
104
139
|
|
|
@@ -119,9 +154,10 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
|
|
|
119
154
|
for (let ci = 0; ci < chainLen; ci++) {
|
|
120
155
|
parentChain.push(view.getUint32(pp + 24 + ci * 4, true))
|
|
121
156
|
}
|
|
157
|
+
|
|
122
158
|
const rawImportance = isV2
|
|
123
159
|
? view.getFloat32(pp + 12, true)
|
|
124
|
-
: Math.min(1
|
|
160
|
+
: Math.min(1, Math.log2(1 + view.getUint32(pp + 12, true) / 1000) / 14)
|
|
125
161
|
|
|
126
162
|
places[pi] = {
|
|
127
163
|
wofID: view.getUint32(pp, true),
|
|
@@ -131,6 +167,7 @@ export function deserializeFSTWeb(input: ArrayBuffer | Uint8Array): FSTMatcher {
|
|
|
131
167
|
lat: view.getFloat32(pp + 16, true),
|
|
132
168
|
lon: view.getFloat32(pp + 20, true),
|
|
133
169
|
parentChain,
|
|
170
|
+
...(hasAmbiguity ? { crossCountryBranches: view.getUint8(pp + 6) } : {}),
|
|
134
171
|
}
|
|
135
172
|
}
|
|
136
173
|
|
|
@@ -148,7 +185,7 @@ export function readFSTProvenanceWeb(input: ArrayBuffer | Uint8Array): FSTProven
|
|
|
148
185
|
if (bytes.byteLength < HEADER_SIZE) return undefined
|
|
149
186
|
const version = view.getUint16(4, true)
|
|
150
187
|
|
|
151
|
-
if (version <
|
|
188
|
+
if (version < VERSION_WITH_METADATA) return undefined
|
|
152
189
|
const provenanceOffset = view.getUint32(28, true)
|
|
153
190
|
|
|
154
191
|
if (provenanceOffset === 0 || provenanceOffset >= bytes.byteLength) return undefined
|
package/fst-matcher.ts
CHANGED
|
@@ -39,15 +39,16 @@ export class FSTMatcher {
|
|
|
39
39
|
walk(tokens: string[]): FSTMatchResult | null {
|
|
40
40
|
let stateID = 0
|
|
41
41
|
|
|
42
|
-
for (
|
|
42
|
+
for (const token of tokens) {
|
|
43
43
|
const node = this.nodes[stateID]
|
|
44
44
|
|
|
45
45
|
if (!node) return null
|
|
46
|
-
const next = node.edges.get(
|
|
46
|
+
const next = node.edges.get(token)
|
|
47
47
|
|
|
48
48
|
if (next === undefined) return null
|
|
49
49
|
stateID = next
|
|
50
50
|
}
|
|
51
|
+
|
|
51
52
|
const node = this.nodes[stateID]!
|
|
52
53
|
|
|
53
54
|
return { stateID, accepted: node.places.length > 0, depth: tokens.length }
|
|
@@ -77,6 +78,7 @@ export class FSTMatcher {
|
|
|
77
78
|
|
|
78
79
|
for (const [token, targetID] of node.edges) {
|
|
79
80
|
const target = this.nodes[targetID]!
|
|
81
|
+
|
|
80
82
|
result.push({
|
|
81
83
|
token,
|
|
82
84
|
targetState: targetID,
|
|
@@ -104,6 +106,7 @@ export class FSTMatcher {
|
|
|
104
106
|
|
|
105
107
|
if (next === undefined) break
|
|
106
108
|
stateID = next
|
|
109
|
+
|
|
107
110
|
depth++
|
|
108
111
|
}
|
|
109
112
|
|
|
@@ -127,7 +130,9 @@ export class FSTMatcher {
|
|
|
127
130
|
return this.nodes.length
|
|
128
131
|
}
|
|
129
132
|
|
|
130
|
-
/**
|
|
133
|
+
/**
|
|
134
|
+
* Expose the internal node array for serialization.
|
|
135
|
+
*/
|
|
131
136
|
toNodes(): readonly FSTNode[] {
|
|
132
137
|
return this.nodes
|
|
133
138
|
}
|
|
@@ -137,12 +142,14 @@ export class FSTMatcher {
|
|
|
137
142
|
}
|
|
138
143
|
}
|
|
139
144
|
|
|
140
|
-
/**
|
|
145
|
+
/**
|
|
146
|
+
* Normalize text into FST tokens: lowercase, NFKC, strip punctuation, split on whitespace.
|
|
147
|
+
*/
|
|
141
148
|
export function normalizeTokens(text: string): string[] {
|
|
142
149
|
return text
|
|
143
150
|
.normalize("NFKC")
|
|
144
151
|
.toLowerCase()
|
|
145
|
-
.
|
|
152
|
+
.replaceAll(/[\p{P}\p{S}]/gu, "")
|
|
146
153
|
.split(/\s+/)
|
|
147
154
|
.filter((t) => t.length > 0)
|
|
148
155
|
}
|
package/fst-serialize.ts
CHANGED
|
@@ -27,14 +27,67 @@ import type { FSTNode } from "./fst-matcher.ts"
|
|
|
27
27
|
import { FSTMatcher } from "./fst-matcher.ts"
|
|
28
28
|
import type { FSTProvenance, PlaceEntry, PlacetypeID } from "./fst-types.ts"
|
|
29
29
|
|
|
30
|
+
/**
|
|
31
|
+
* Format version that widened the per-state edge and place counters from 16 to 32 bits, growing the state entry from 12
|
|
32
|
+
* to 16 bytes. Readers branch on it to stay backward-compatible with v2/v3 files.
|
|
33
|
+
*/
|
|
34
|
+
const VERSION_WIDE_STATE_COUNTERS = 4
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* State-table entry size in bytes at or above {@link VERSION_WIDE_STATE_COUNTERS}.
|
|
38
|
+
*/
|
|
39
|
+
const WIDE_STATE_ENTRY_SIZE = 16
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* State-table entry size in bytes below {@link VERSION_WIDE_STATE_COUNTERS}.
|
|
43
|
+
*/
|
|
44
|
+
const NARROW_STATE_ENTRY_SIZE = 12
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* First format version carrying the trailing metadata block; older files simply have none.
|
|
48
|
+
*/
|
|
49
|
+
const VERSION_WITH_METADATA = 3
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* File magic. A reader rejects anything not starting with these four bytes before parsing further.
|
|
53
|
+
*/
|
|
30
54
|
const MAGIC = Buffer.from("FST\0", "ascii")
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Format version this serializer emits. See {@link VERSION_WIDE_STATE_COUNTERS} for what each bump changed.
|
|
58
|
+
*/
|
|
31
59
|
const VERSION = 4
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Fixed header size in bytes: magic, version, and the section offsets that follow it.
|
|
63
|
+
*/
|
|
32
64
|
const HEADER_SIZE = 32
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* State-table entry: edge offset, place offset, and the two 32-bit counters (v4 widths).
|
|
68
|
+
*/
|
|
33
69
|
const STATE_ENTRY_SIZE = 16
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Edge-table entry: the transition label and the target state index.
|
|
73
|
+
*/
|
|
34
74
|
const EDGE_ENTRY_SIZE = 8
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Place-table entry: the place id, its placetype, coordinates, and importance.
|
|
78
|
+
*/
|
|
35
79
|
const PLACE_ENTRY_SIZE = 56
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Longest ancestry chain stored per place. Deeper hierarchies are truncated at the leaf end, since the specific end of
|
|
83
|
+
* the chain is what disambiguates and the country end is recoverable anyway.
|
|
84
|
+
*/
|
|
36
85
|
const MAX_CHAIN_LEN = 8
|
|
37
86
|
|
|
87
|
+
/**
|
|
88
|
+
* Placetypes in hierarchy order, largest first. The index into this array is what gets written into a place entry, so
|
|
89
|
+
* REORDERING IT BREAKS EVERY EXISTING FILE — append instead, and bump the version.
|
|
90
|
+
*/
|
|
38
91
|
const PLACETYPE_ORDER: readonly PlacetypeID[] = [
|
|
39
92
|
"country",
|
|
40
93
|
"region",
|
|
@@ -109,11 +162,15 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
|
|
|
109
162
|
let pos = 0
|
|
110
163
|
|
|
111
164
|
// --- Header ---
|
|
165
|
+
// flags bit0 (survey #4, 2026-07-27): place rows carry surface-ambiguity data in the former _pad
|
|
166
|
+
// byte (pp+6 = crossCountryBranches u8, pp+7 reserved). Presence-signaled here so VERSION stays
|
|
167
|
+
// put: pre-ambiguity artifacts read flags=0 → readers expose `undefined`, never a fake 0.
|
|
168
|
+
const hasAmbiguity = nodes.some((n) => n.places.some((p) => p.crossCountryBranches !== undefined))
|
|
112
169
|
MAGIC.copy(buf, pos)
|
|
113
170
|
pos += 4
|
|
114
171
|
buf.writeUInt16LE(VERSION, pos)
|
|
115
172
|
pos += 2
|
|
116
|
-
buf.writeUInt16LE(0, pos)
|
|
173
|
+
buf.writeUInt16LE(hasAmbiguity ? 1 : 0, pos)
|
|
117
174
|
pos += 2
|
|
118
175
|
buf.writeUInt32LE(nodes.length, pos)
|
|
119
176
|
pos += 4
|
|
@@ -131,11 +188,12 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
|
|
|
131
188
|
// --- String table ---
|
|
132
189
|
let strOffset = 0
|
|
133
190
|
|
|
134
|
-
for (
|
|
191
|
+
for (const encoded of encodedStrings) {
|
|
135
192
|
buf.writeUInt32LE(strOffset, pos)
|
|
136
193
|
pos += 4
|
|
137
|
-
strOffset +=
|
|
194
|
+
strOffset += encoded.length
|
|
138
195
|
}
|
|
196
|
+
|
|
139
197
|
buf.writeUInt32LE(strOffset, pos)
|
|
140
198
|
pos += 4
|
|
141
199
|
|
|
@@ -167,6 +225,7 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
|
|
|
167
225
|
const ep = edgeTableStart + edgeIdx * EDGE_ENTRY_SIZE
|
|
168
226
|
buf.writeUInt32LE(intern(token), ep)
|
|
169
227
|
buf.writeUInt32LE(target, ep + 4)
|
|
228
|
+
|
|
170
229
|
edgeIdx++
|
|
171
230
|
}
|
|
172
231
|
|
|
@@ -178,7 +237,9 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
|
|
|
178
237
|
buf.writeUInt32LE(place.wofID, pp)
|
|
179
238
|
buf.writeUInt8(placetypeToIdx.get(place.placetype) ?? 0, pp + 4)
|
|
180
239
|
buf.writeUInt8(chainLen, pp + 5)
|
|
181
|
-
|
|
240
|
+
// Former _pad: byte 0 = crossCountryBranches (header flags bit0 gates the read), byte 1 reserved.
|
|
241
|
+
buf.writeUInt8(hasAmbiguity ? Math.min(place.crossCountryBranches ?? 0, 255) : 0, pp + 6)
|
|
242
|
+
buf.writeUInt8(0, pp + 7)
|
|
182
243
|
buf.writeUInt32LE(intern(place.name), pp + 8)
|
|
183
244
|
buf.writeFloatLE(place.importance, pp + 12)
|
|
184
245
|
buf.writeFloatLE(place.lat, pp + 16)
|
|
@@ -187,6 +248,7 @@ export function serializeFST(matcher: FSTMatcher, provenance?: FSTProvenance): B
|
|
|
187
248
|
for (let ci = 0; ci < MAX_CHAIN_LEN; ci++) {
|
|
188
249
|
buf.writeUInt32LE(ci < chainLen ? validChain[ci]! : 0, pp + 24 + ci * 4)
|
|
189
250
|
}
|
|
251
|
+
|
|
190
252
|
placeIdx++
|
|
191
253
|
}
|
|
192
254
|
}
|
|
@@ -209,6 +271,8 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
|
|
|
209
271
|
|
|
210
272
|
if (version < 1 || version > VERSION) throw new Error(`FST version ${version} unsupported (expected 1..${VERSION})`)
|
|
211
273
|
const isV2 = version >= 2
|
|
274
|
+
// flags bit0 (survey #4): place rows carry surface-ambiguity data in the former _pad byte.
|
|
275
|
+
const hasAmbiguity = (buf.readUInt16LE(6) & 1) === 1
|
|
212
276
|
|
|
213
277
|
const stateCount = buf.readUInt32LE(8)
|
|
214
278
|
const edgeCount = buf.readUInt32LE(12)
|
|
@@ -225,6 +289,7 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
|
|
|
225
289
|
strOffsets[i] = buf.readUInt32LE(pos)
|
|
226
290
|
pos += 4
|
|
227
291
|
}
|
|
292
|
+
|
|
228
293
|
const strDataStart = pos
|
|
229
294
|
const strings: string[] = new Array(stringCount)
|
|
230
295
|
|
|
@@ -233,10 +298,11 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
|
|
|
233
298
|
const end = strDataStart + strOffsets[i + 1]!
|
|
234
299
|
strings[i] = buf.toString("utf8", start, end)
|
|
235
300
|
}
|
|
301
|
+
|
|
236
302
|
pos += stringBytes
|
|
237
303
|
|
|
238
304
|
// --- State table ---
|
|
239
|
-
const stateEntrySize = version >=
|
|
305
|
+
const stateEntrySize = version >= VERSION_WIDE_STATE_COUNTERS ? WIDE_STATE_ENTRY_SIZE : NARROW_STATE_ENTRY_SIZE
|
|
240
306
|
const stateTableStart = pos
|
|
241
307
|
const edgeTableStart = stateTableStart + stateCount * stateEntrySize
|
|
242
308
|
const placeTableStart = edgeTableStart + edgeCount * EDGE_ENTRY_SIZE
|
|
@@ -247,8 +313,12 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
|
|
|
247
313
|
const sp = stateTableStart + si * stateEntrySize
|
|
248
314
|
const edgeStart = buf.readUInt32LE(sp)
|
|
249
315
|
const placeStart = buf.readUInt32LE(sp + 4)
|
|
250
|
-
|
|
251
|
-
const
|
|
316
|
+
|
|
317
|
+
const edgeCountForState =
|
|
318
|
+
version >= VERSION_WIDE_STATE_COUNTERS ? buf.readUInt32LE(sp + 8) : buf.readUInt16LE(sp + 8)
|
|
319
|
+
|
|
320
|
+
const placeCountForState =
|
|
321
|
+
version >= VERSION_WIDE_STATE_COUNTERS ? buf.readUInt32LE(sp + 12) : buf.readUInt16LE(sp + 10)
|
|
252
322
|
|
|
253
323
|
const edges = new Map<string, number>()
|
|
254
324
|
|
|
@@ -269,9 +339,11 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
|
|
|
269
339
|
for (let ci = 0; ci < chainLen; ci++) {
|
|
270
340
|
parentChain.push(buf.readUInt32LE(pp + 24 + ci * 4))
|
|
271
341
|
}
|
|
342
|
+
|
|
272
343
|
const rawImportance = isV2
|
|
273
344
|
? buf.readFloatLE(pp + 12)
|
|
274
|
-
: Math.min(1
|
|
345
|
+
: Math.min(1, Math.log2(1 + buf.readUInt32LE(pp + 12) / 1000) / 14)
|
|
346
|
+
|
|
275
347
|
places[pi] = {
|
|
276
348
|
wofID: buf.readUInt32LE(pp),
|
|
277
349
|
placetype: PLACETYPE_ORDER[buf.readUInt8(pp + 4)] ?? "locality",
|
|
@@ -280,6 +352,8 @@ export function deserializeFST(buf: Buffer): FSTMatcher {
|
|
|
280
352
|
lat: buf.readFloatLE(pp + 16),
|
|
281
353
|
lon: buf.readFloatLE(pp + 20),
|
|
282
354
|
parentChain,
|
|
355
|
+
// Header flags bit0 gates the read (survey #4): pre-ambiguity artifacts expose undefined.
|
|
356
|
+
...(hasAmbiguity ? { crossCountryBranches: buf.readUInt8(pp + 6) } : {}),
|
|
283
357
|
}
|
|
284
358
|
}
|
|
285
359
|
|
|
@@ -295,7 +369,7 @@ export function readFSTProvenance(buf: Buffer): FSTProvenance | undefined {
|
|
|
295
369
|
if (!buf.subarray(0, 4).equals(MAGIC)) return undefined
|
|
296
370
|
const version = buf.readUInt16LE(4)
|
|
297
371
|
|
|
298
|
-
if (version <
|
|
372
|
+
if (version < VERSION_WITH_METADATA) return undefined
|
|
299
373
|
const provenanceOffset = buf.readUInt32LE(28)
|
|
300
374
|
|
|
301
375
|
if (provenanceOffset === 0 || provenanceOffset >= buf.length) return undefined
|
package/fst-types.ts
CHANGED
|
@@ -16,6 +16,13 @@ export interface PlaceEntry {
|
|
|
16
16
|
importance: number
|
|
17
17
|
lat: number
|
|
18
18
|
lon: number
|
|
19
|
+
/**
|
|
20
|
+
* Surface-ambiguity class (survey #4): how many DISTINCT countries carry a place with THIS entry's accepting surface,
|
|
21
|
+
* counted over the whole admin DB at build time (clamped to 255). A property of the surface, not the place — the same
|
|
22
|
+
* place reached via different alias surfaces reports each surface's own count. `undefined` = built without ambiguity
|
|
23
|
+
* data (pre-2026-07-27 artifacts) — NEVER conflate with 1 (the unambiguous case); the meaning-of-zero rule.
|
|
24
|
+
*/
|
|
25
|
+
crossCountryBranches?: number
|
|
19
26
|
}
|
|
20
27
|
|
|
21
28
|
export type PlacetypeID =
|
|
@@ -60,9 +67,13 @@ export interface FSTProvenance {
|
|
|
60
67
|
importanceMatches: number
|
|
61
68
|
sourceDB?: string
|
|
62
69
|
modelCardVersion?: string
|
|
63
|
-
/**
|
|
70
|
+
/**
|
|
71
|
+
* Degenerate-surface curation policy applied at build time (absent = uncurated build).
|
|
72
|
+
*/
|
|
64
73
|
exclusionPolicy?: string
|
|
65
|
-
/**
|
|
74
|
+
/**
|
|
75
|
+
* Name insertions refused by the curation policy.
|
|
76
|
+
*/
|
|
66
77
|
excludedInsertions?: number
|
|
67
78
|
}
|
|
68
79
|
|
|
@@ -85,8 +96,20 @@ export interface BuildFSTOpts {
|
|
|
85
96
|
* street-type words ("Avenue Road" is a real name; "de la" is not).
|
|
86
97
|
*/
|
|
87
98
|
excludeAllTokensOf?: ReadonlySet<string>
|
|
88
|
-
/**
|
|
99
|
+
/**
|
|
100
|
+
* Recorded verbatim into provenance when either exclusion set is supplied.
|
|
101
|
+
*/
|
|
89
102
|
exclusionPolicy?: string
|
|
103
|
+
/**
|
|
104
|
+
* Surface-ambiguity classes (survey #4, 2026-07-27): normalized-join surface → the number of DISTINCT countries
|
|
105
|
+
* (across the WHOLE admin DB, not just this build's country scope) with a place carrying that surface. When supplied,
|
|
106
|
+
* every inserted place row records the count for ITS accepting surface (`PlaceEntry.crossCountryBranches`) — an entry
|
|
107
|
+
* accessible under several surfaces records each surface's own count. Serialized into the place row's former `_pad`
|
|
108
|
+
* byte with presence signaled by header flags bit0, so VERSION stays put and pre-ambiguity artifacts read as "no
|
|
109
|
+
* data" (never "0 branches" — the meaning-of-zero rule). No decoder consumes it yet; consumers (FST-prior tempering,
|
|
110
|
+
* the Option-A evidence channel) arrive behind their own measured gates.
|
|
111
|
+
*/
|
|
112
|
+
surfaceCountryCounts?: ReadonlyMap<string, number>
|
|
90
113
|
onProgress?: (phase: string, detail?: string) => void
|
|
91
114
|
}
|
|
92
115
|
|
package/fts-query.ts
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* Query shaping for the FTS5 lookup: placetype normalization and the MATCH-expression sanitizer.
|
|
7
|
+
* Both turn a caller's loose input into something SQLite's FTS5 parser accepts without throwing —
|
|
8
|
+
* an unescaped quote or a bare `*` is a syntax error, not an empty result.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import type { FindPlaceQuery, WOFPlacetype } from "./types.ts"
|
|
12
|
+
|
|
13
|
+
export function normalizePlacetypes(p: FindPlaceQuery["placetype"]): WOFPlacetype[] | null {
|
|
14
|
+
if (!p) return null
|
|
15
|
+
|
|
16
|
+
return Array.isArray(p) ? p : [p]
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Make an arbitrary user-typed string safe for FTS5 MATCH.
|
|
21
|
+
*
|
|
22
|
+
* FTS5 has its own query syntax (`"phrase"`, `term1 OR term2`, `prefix*`, NEAR/N, etc.). Letting raw user input through
|
|
23
|
+
* means a user typing `Paris's` or `St. (Petersburg)` causes a syntax error.
|
|
24
|
+
*
|
|
25
|
+
* Per-token rules:
|
|
26
|
+
*
|
|
27
|
+
* - Strip all punctuation except trailing `*` from each whitespace-separated token.
|
|
28
|
+
* - **Trailing `*`** is preserved as FTS5 **prefix syntax** — `627*` becomes the literal `627*` (unquoted). The caller
|
|
29
|
+
* signaled they want a prefix; respect that.
|
|
30
|
+
* - All other tokens are wrapped in `"..."` as a single-word phrase. Conservative — handles apostrophes, parens, accented
|
|
31
|
+
* input, etc. safely.
|
|
32
|
+
* - Multiple tokens join with implicit AND.
|
|
33
|
+
*
|
|
34
|
+
* Examples:
|
|
35
|
+
*
|
|
36
|
+
* - `"Paris"` → `"Paris"` (phrase)
|
|
37
|
+
* - `"627*"` → `627*` (prefix)
|
|
38
|
+
* - `"St. (Petersburg)"` → `"St" "Petersburg"` (two phrases, AND-joined)
|
|
39
|
+
* - `"Thiron-Gardais"` → `"Thiron" "Gardais"` (intra-token punctuation SPLITS — #945; fusing to `ThironGardais` matched
|
|
40
|
+
* nothing because the FTS doc tokenizes the hyphenated name as two terms)
|
|
41
|
+
* - `"110 00"` with `fuseTokens` (postcode-typed) → `"110" "00"` per-token fused — the #920 name law
|
|
42
|
+
* - `"Pari* TX"` → `Pari* "TX"` (mixed prefix + phrase)
|
|
43
|
+
* - `"*"` alone → `""` (no body → drop)
|
|
44
|
+
*/
|
|
45
|
+
export function sanitizeFTSQuery(text: string, opts?: { fuseTokens?: boolean }): string {
|
|
46
|
+
const out: string[] = []
|
|
47
|
+
|
|
48
|
+
for (const rawToken of text.normalize("NFKC").split(/\s+/u)) {
|
|
49
|
+
const trimmed = rawToken.trim()
|
|
50
|
+
|
|
51
|
+
if (!trimmed) continue
|
|
52
|
+
const hasPrefixStar = trimmed.endsWith("*")
|
|
53
|
+
|
|
54
|
+
// #920 name law (postcode-typed queries ONLY): delete intra-token punctuation and FUSE the
|
|
55
|
+
// remainder — postal names are stored in this collapsed shape ("SW1A" stays one term).
|
|
56
|
+
if (opts?.fuseTokens) {
|
|
57
|
+
const body = trimmed.replaceAll(/[^\p{L}\p{N}]/gu, "")
|
|
58
|
+
|
|
59
|
+
if (!body) continue
|
|
60
|
+
out.push(hasPrefixStar ? `${body}*` : `"${body.replaceAll('"', '""')}"`)
|
|
61
|
+
|
|
62
|
+
continue
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// Everything else SPLITS on intra-token punctuation — the behavior the docstring always
|
|
66
|
+
// promised ("St. (Petersburg)" → two phrases). The old code DELETED punctuation instead,
|
|
67
|
+
// fusing "Thiron-Gardais" into the unmatchable single term `ThironGardais` while the FTS
|
|
68
|
+
// doc holds two terms (#945 — the entire hyphenated-name class missed at the raw lookup;
|
|
69
|
+
// masked for years because pre-splice tokenizers never emitted hyphen-preserved values).
|
|
70
|
+
const parts = trimmed.split(/[^\p{L}\p{N}]+/u).filter(Boolean)
|
|
71
|
+
|
|
72
|
+
if (!parts.length) continue
|
|
73
|
+
|
|
74
|
+
for (let i = 0; i < parts.length; i++) {
|
|
75
|
+
const body = parts[i]!.replaceAll("*", "")
|
|
76
|
+
|
|
77
|
+
if (!body) continue
|
|
78
|
+
// The caller's trailing `*` applies to the FINAL part ("Thiron-Gard*" → "Thiron" Gard*).
|
|
79
|
+
out.push(hasPrefixStar && i === parts.length - 1 ? `${body}*` : `"${body.replaceAll('"', '""')}"`)
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
return out.join(" ")
|
|
84
|
+
}
|