@mailwoman/resolver-wof-sqlite 7.2.0 → 7.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/address-point-interpolation.ts +207 -0
- package/address-point-schema.ts +107 -0
- package/address-point.ts +122 -0
- package/ancestry-backfill.ts +205 -0
- package/ancestry.ts +70 -0
- package/build-candidate.ts +351 -0
- package/build-slim.ts +394 -0
- package/candidate-fts.ts +43 -0
- package/candidate-lookup.ts +382 -0
- package/candidate-schema.ts +166 -0
- package/coincident-roles.ts +240 -0
- package/convention.ts +152 -0
- package/fst-autocomplete.ts +187 -0
- package/fst-builder.ts +291 -0
- package/fst-deserialize-web.ts +164 -0
- package/fst-matcher.ts +150 -0
- package/fst-serialize.ts +311 -0
- package/fst-types.ts +78 -0
- package/fts.ts +318 -0
- package/geo.ts +140 -0
- package/geonames-aliases.ts +317 -0
- package/geonames-postal.ts +150 -0
- package/index.ts +117 -0
- package/interpolation.ts +232 -0
- package/lookup.ts +1498 -0
- package/package.json +168 -82
- package/poi-lookup.ts +319 -0
- package/poi-schema.ts +147 -0
- package/postal-city-alias-lookup.ts +89 -0
- package/postal-city-alias-schema.ts +75 -0
- package/postal-city-candidate-schema.ts +81 -0
- package/postcode-point-lookup.ts +64 -0
- package/reverse.ts +429 -0
- package/schema.ts +176 -0
- package/sharding.ts +235 -0
- package/sqlite-convention-source.ts +61 -0
- package/sqlite-utils.ts +25 -0
- package/street-centroid-schema.ts +124 -0
- package/street-centroid.ts +124 -0
- package/street-morphology-fst-builder.ts +230 -0
- package/street-name-lookup.ts +101 -0
- package/street-normalize.ts +302 -0
- package/street-segment-schema.ts +104 -0
- package/types.ts +164 -0
- package/unified-schema.ts +171 -0
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* Typed schema for the TIGER STREET-SEGMENT interpolation shards (`street-segments-<cc>-<st>.db`,
|
|
7
|
+
* built by `scripts/build-interpolation-shard.ts` from TIGER EDGES) — the #483 Method-3 fallback
|
|
8
|
+
* the resolver drops to when the address-point tier (Method 2) can't bracket. Single source of
|
|
9
|
+
* truth for the columns the BUILDER writes and the READER ({@link StreetInterpolator}) probes, so
|
|
10
|
+
* a column rename in one is a compile error in the other.
|
|
11
|
+
*
|
|
12
|
+
* The builder reads geometry from shapefiles via DuckDB's spatial extension (raw `ST_Read` — see
|
|
13
|
+
* AGENTS.md "Database / inline SQL") and writes here through `node:sqlite`. The hot positional
|
|
14
|
+
* INSERT (a county's worth of edges) stays raw; its column list is derived from
|
|
15
|
+
* {@link STREET_SEGMENT_COLUMNS} so it can't drift from the DDL.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import type { Kysely } from "kysely"
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* One TIGER street-segment edge: a `(from_hn, to_hn)` house-number range on one `side` of a named street, with the
|
|
22
|
+
* geometry the interpolator walks. `min_hn`/`max_hn` are the sorted bounds (the probe filters on them); `parity` is
|
|
23
|
+
* `odd`/`even`/`mixed`.
|
|
24
|
+
*/
|
|
25
|
+
export interface StreetSegmentTable {
|
|
26
|
+
/** Shared {@link normalizeStreetForKey} of the street — the build/query-consistent probe key. */
|
|
27
|
+
street_norm: string
|
|
28
|
+
/** `L` or `R` — the TIGER side the address range sits on. */
|
|
29
|
+
side: string
|
|
30
|
+
from_hn: number
|
|
31
|
+
to_hn: number
|
|
32
|
+
/** Sorted lower bound of `(from_hn, to_hn)` — the probe filters `min_hn <= n <= max_hn`. */
|
|
33
|
+
min_hn: number
|
|
34
|
+
/** Sorted upper bound of `(from_hn, to_hn)`. */
|
|
35
|
+
max_hn: number
|
|
36
|
+
/** `odd` | `even` | `mixed` — the house-number parity along the range. */
|
|
37
|
+
parity: string
|
|
38
|
+
postcode: string | null
|
|
39
|
+
/** 5-digit state+county FIPS the edge came from. */
|
|
40
|
+
county_fips: string
|
|
41
|
+
/** The street as it appeared in TIGER (kept for display / debugging). */
|
|
42
|
+
street_raw: string
|
|
43
|
+
/** GeoJSON LineString text (no SpatiaLite — read back with `JSON.parse`). */
|
|
44
|
+
geometry: string
|
|
45
|
+
/** Provenance: the dataset this edge came from (e.g. `tiger:edges`). */
|
|
46
|
+
source: string
|
|
47
|
+
/** The pinned TIGER release the edge was ingested from. */
|
|
48
|
+
release: string
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** The street-segment database schema for `new DatabaseClient<StreetSegmentDatabase>(...)`. */
|
|
52
|
+
export interface StreetSegmentDatabase {
|
|
53
|
+
street_segment: StreetSegmentTable
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* The `street_segment` columns in INSERT order. The builder's positional prepared statement derives its placeholder
|
|
58
|
+
* list from this, so the positional order can't drift from the DDL / the reader.
|
|
59
|
+
*/
|
|
60
|
+
export const STREET_SEGMENT_COLUMNS = [
|
|
61
|
+
"street_norm",
|
|
62
|
+
"side",
|
|
63
|
+
"from_hn",
|
|
64
|
+
"to_hn",
|
|
65
|
+
"min_hn",
|
|
66
|
+
"max_hn",
|
|
67
|
+
"parity",
|
|
68
|
+
"postcode",
|
|
69
|
+
"county_fips",
|
|
70
|
+
"street_raw",
|
|
71
|
+
"geometry",
|
|
72
|
+
"source",
|
|
73
|
+
"release",
|
|
74
|
+
] as const
|
|
75
|
+
|
|
76
|
+
/** Create the `street_segment` table — called before the streaming bulk load. */
|
|
77
|
+
export async function createStreetSegmentTable(db: Kysely<StreetSegmentDatabase>): Promise<void> {
|
|
78
|
+
await db.schema
|
|
79
|
+
.createTable("street_segment")
|
|
80
|
+
.addColumn("street_norm", "text", (c) => c.notNull())
|
|
81
|
+
.addColumn("side", "text", (c) => c.notNull())
|
|
82
|
+
.addColumn("from_hn", "integer", (c) => c.notNull())
|
|
83
|
+
.addColumn("to_hn", "integer", (c) => c.notNull())
|
|
84
|
+
.addColumn("min_hn", "integer", (c) => c.notNull())
|
|
85
|
+
.addColumn("max_hn", "integer", (c) => c.notNull())
|
|
86
|
+
.addColumn("parity", "text", (c) => c.notNull())
|
|
87
|
+
.addColumn("postcode", "text")
|
|
88
|
+
.addColumn("county_fips", "text", (c) => c.notNull())
|
|
89
|
+
.addColumn("street_raw", "text", (c) => c.notNull())
|
|
90
|
+
.addColumn("geometry", "text", (c) => c.notNull())
|
|
91
|
+
.addColumn("source", "text", (c) => c.notNull())
|
|
92
|
+
.addColumn("release", "text", (c) => c.notNull())
|
|
93
|
+
.execute()
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** Create the two probe indexes the reader relies on (postcode-scope, street-scope). */
|
|
97
|
+
export async function createStreetSegmentIndexes(db: Kysely<StreetSegmentDatabase>): Promise<void> {
|
|
98
|
+
await db.schema
|
|
99
|
+
.createIndex("idx_seg_postcode")
|
|
100
|
+
.on("street_segment")
|
|
101
|
+
.columns(["postcode", "street_norm", "min_hn"])
|
|
102
|
+
.execute()
|
|
103
|
+
await db.schema.createIndex("idx_seg_street").on("street_segment").columns(["street_norm", "min_hn"]).execute()
|
|
104
|
+
}
|
package/types.ts
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* Public surface for the WOF SQLite resolver — types only, no runtime.
|
|
7
|
+
*
|
|
8
|
+
* These mirror the conceptual model described in `docs/plan/phases/PHASE_4_2_wof_sqlite.md`. Phase
|
|
9
|
+
* 4.3 will extend `PlaceCandidate` with the resolver-decorated fields that flow into
|
|
10
|
+
* `AddressNode.source` / `sourceID` (e.g. an explicit `wofURI: "wof-admin:101751113"` form).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* The placetype taxonomy used by Who's On First. Ordered roughly from coarsest (country) to finest (address). See
|
|
15
|
+
* https://github.com/whosonfirst/whosonfirst-placetypes for the authoritative definitions of each.
|
|
16
|
+
*
|
|
17
|
+
* Phase 4.2 only emits the ones we actually look up; the union is open enough to extend later.
|
|
18
|
+
*/
|
|
19
|
+
export type WOFPlacetype =
|
|
20
|
+
| "country"
|
|
21
|
+
| "macroregion"
|
|
22
|
+
| "region"
|
|
23
|
+
| "macrocounty"
|
|
24
|
+
| "county"
|
|
25
|
+
| "localadmin"
|
|
26
|
+
| "locality"
|
|
27
|
+
| "borough"
|
|
28
|
+
| "neighbourhood"
|
|
29
|
+
| "microhood"
|
|
30
|
+
| "postalcode"
|
|
31
|
+
| "venue"
|
|
32
|
+
| "campus"
|
|
33
|
+
| "address"
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* One candidate match for a place lookup.
|
|
37
|
+
*
|
|
38
|
+
* `score` is the post-boost ranking number — higher is better, but the scale is implementation- defined. Callers should
|
|
39
|
+
* treat it as ordinal, not absolute.
|
|
40
|
+
*
|
|
41
|
+
* `id` is the WOF place id. It's named generically (not `wof_id`) so the shape stays structurally compatible with
|
|
42
|
+
* `@mailwoman/resolver`'s `ResolvedPlace` — `WOFSqlitePlaceLookup` satisfies the generic `ResolverBackend` contract
|
|
43
|
+
* without an adapter shim.
|
|
44
|
+
*
|
|
45
|
+
* `distanceKm` is populated only when the query carried `near` (and the place has a centroid). Useful for downstream
|
|
46
|
+
* UIs that want to show "X km from you" alongside the result.
|
|
47
|
+
*/
|
|
48
|
+
export interface PlaceCandidate {
|
|
49
|
+
id: number
|
|
50
|
+
name: string
|
|
51
|
+
placetype: WOFPlacetype
|
|
52
|
+
/** ISO 3166-1 alpha-2 country code. */
|
|
53
|
+
country: string
|
|
54
|
+
lat: number
|
|
55
|
+
lon: number
|
|
56
|
+
parent_id?: number
|
|
57
|
+
score: number
|
|
58
|
+
distanceKm?: number
|
|
59
|
+
/**
|
|
60
|
+
* True when this candidate's name OR an alias EXACTLY equals the query (the exact-match tier from
|
|
61
|
+
* {@link RankingWeights.exactMatchTiering}). Surfaced so a downstream country re-rank (#369's postcode anchor in
|
|
62
|
+
* `resolveTree`) can pin the country without crossing the tier — see the `exactMatch` field on `@mailwoman/core`'s
|
|
63
|
+
* `ResolvedPlace`.
|
|
64
|
+
*/
|
|
65
|
+
exactMatch?: boolean
|
|
66
|
+
/**
|
|
67
|
+
* Combined prominence (population term + best proximity-bias term, same additive units) — populated by the FTS
|
|
68
|
+
* lookup; the exact-tier sort orders by THIS instead of raw population when the query carried proximity hints
|
|
69
|
+
* (`near`/`bias`).
|
|
70
|
+
*/
|
|
71
|
+
prominence?: number
|
|
72
|
+
/**
|
|
73
|
+
* Population from WOF's `wof:population` property. Only present when the candidate has it on record — WOF carries
|
|
74
|
+
* population for ~15% of localities (mostly larger ones). Absent does NOT mean zero, just unknown.
|
|
75
|
+
*/
|
|
76
|
+
population?: number
|
|
77
|
+
/**
|
|
78
|
+
* Bounding box from WOF's `spr.{min,max}_{latitude,longitude}` columns. Coarse outline for the place — a city's bbox
|
|
79
|
+
* is the city's full extent, a postcode's is roughly the postcode polygon's envelope. Optional because not all
|
|
80
|
+
* callers ask for it; implementations are free to omit when the underlying schema lacks the columns.
|
|
81
|
+
*/
|
|
82
|
+
bbox?: GeoBbox
|
|
83
|
+
/**
|
|
84
|
+
* Set by the coordinate-first path when the chosen locality and the sibling postcode's containing locality are
|
|
85
|
+
* geographically far apart — the postcode and the parsed city name disagree (a transposed / wrong-for-the-city
|
|
86
|
+
* postcode). The candidate is still returned (the name wins for the locality), but the flag lets callers lower
|
|
87
|
+
* confidence / surface the conflict rather than silently mislocate. A retrieval/BM25 geocoder can't raise this — it's
|
|
88
|
+
* the falsehood-detection differentiator.
|
|
89
|
+
*/
|
|
90
|
+
mismatch?: boolean
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* A WGS-84 lat/lon point. Used as a proximity hint for `FindPlaceQuery.near`.
|
|
95
|
+
*/
|
|
96
|
+
export interface GeoPoint {
|
|
97
|
+
lat: number
|
|
98
|
+
lon: number
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* A WGS-84 bounding box. Used as a hard filter via `FindPlaceQuery.bbox`.
|
|
103
|
+
*/
|
|
104
|
+
export interface GeoBbox {
|
|
105
|
+
minLat: number
|
|
106
|
+
maxLat: number
|
|
107
|
+
minLon: number
|
|
108
|
+
maxLon: number
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Query against the resolver.
|
|
113
|
+
*
|
|
114
|
+
* `text` is the only required field; everything else narrows the search. When `country` and `parentID` are both set,
|
|
115
|
+
* `parentID` wins (it's more specific).
|
|
116
|
+
*
|
|
117
|
+
* `near` and `bbox` are independent. `near` is a soft signal — candidates close to the point get a ranking boost but
|
|
118
|
+
* distant candidates aren't dropped. `bbox` is a hard filter — only candidates whose bbox intersects the query bbox are
|
|
119
|
+
* returned (uses the package-built R*Tree index when present; if the index is missing the option is silently ignored to
|
|
120
|
+
* preserve backwards compatibility).
|
|
121
|
+
*
|
|
122
|
+
* `near` may carry `maxDistanceKm` to escalate from a boost to a hard filter — candidates further than that distance
|
|
123
|
+
* from the point are dropped at the SQL level via an R*Tree pre-filter.
|
|
124
|
+
*/
|
|
125
|
+
export interface FindPlaceQuery {
|
|
126
|
+
text: string
|
|
127
|
+
placetype?: WOFPlacetype | WOFPlacetype[]
|
|
128
|
+
/** ISO 3166-1 alpha-2 — narrows to one country. */
|
|
129
|
+
country?: string
|
|
130
|
+
/** WOF place id — narrows to descendants of this place. */
|
|
131
|
+
parentID?: number
|
|
132
|
+
/**
|
|
133
|
+
* Sibling postcode. When set on a `locality` query AND a `postcode_locality` table is present, triggers the
|
|
134
|
+
* coordinate-first soft-score path: postcode→candidate localities are injected and scored `0.6·S_pc + 0.3·S_name +
|
|
135
|
+
* 0.1·S_pop` against the FTS name-match set, recovering small localities the name-match alone misses. Ignored when no
|
|
136
|
+
* postcode_locality shard is present.
|
|
137
|
+
*/
|
|
138
|
+
postcode?: string
|
|
139
|
+
/** Proximity hint — candidates close to this point get a ranking boost. */
|
|
140
|
+
near?: GeoPoint & { maxDistanceKm?: number }
|
|
141
|
+
/**
|
|
142
|
+
* Ordered proximity-bias points (viewport center, user location, …), each optionally weighted (default 1.0, first
|
|
143
|
+
* entry strongest by convention). SOFT — a re-rank signal, never a filter: with bias present, exact-tier candidates
|
|
144
|
+
* order by combined prominence (population + the best decayed-distance term over these points) instead of population
|
|
145
|
+
* alone, which is how an ambiguous bare postcode ("48026": Fraser MI vs Russi IT) follows the map view / the user.
|
|
146
|
+
* Absent (and no `near`) → ranking is byte-identical to today. `near` is treated as a weight-1.0 bias point for
|
|
147
|
+
* back-compat.
|
|
148
|
+
*/
|
|
149
|
+
bias?: Array<GeoPoint & { weight?: number }>
|
|
150
|
+
/** Bounding-box filter — only candidates whose bbox intersects this box are returned. */
|
|
151
|
+
bbox?: GeoBbox
|
|
152
|
+
/** Default 10. */
|
|
153
|
+
limit?: number
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* The pull-based lookup surface. Implementations resolve a `FindPlaceQuery` to a ranked list of `PlaceCandidate`s. The
|
|
158
|
+
* interface is async even though `node:sqlite` is sync — leaves room for `Worker`-backed implementations later without
|
|
159
|
+
* a public API break.
|
|
160
|
+
*/
|
|
161
|
+
export interface PlaceLookup {
|
|
162
|
+
findPlace(query: FindPlaceQuery): Promise<PlaceCandidate[]>
|
|
163
|
+
close(): void
|
|
164
|
+
}
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* Schema for the unified WOF SQLite database we build from cloned WOF GeoJSON repos
|
|
7
|
+
* (`scripts/build-unified-wof.ts`). This is the CANONICAL gazetteer — we never use the
|
|
8
|
+
* off-the-shelf geocode.earth prebuilt dumps (they assign different WOF ids to the same place;
|
|
9
|
+
* see the `feedback-custom-wof-db-only` memory). The table/column names match the resolver's
|
|
10
|
+
* expectations (`lookup.ts`) so `WOFSqlitePlaceLookup` works unchanged, INCLUDING the `ancestors`
|
|
11
|
+
* table (which lookup.ts's parent-constraint subquery needs) — see `populateAncestors`. The
|
|
12
|
+
* `place_search` FTS5 + `place_bbox` R*Tree are built separately by `build-fts` (fts.ts).
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { DatabaseSync } from "node:sqlite"
|
|
16
|
+
|
|
17
|
+
import { DatabaseClient } from "@mailwoman/core/kysley/client"
|
|
18
|
+
|
|
19
|
+
import type { WOFDatabase } from "./schema.ts"
|
|
20
|
+
|
|
21
|
+
export async function createUnifiedSchema(db: DatabaseSync): Promise<void> {
|
|
22
|
+
// PRAGMAs stay raw — not Kysely-modelled, and these tune the bulk build.
|
|
23
|
+
db.exec("PRAGMA journal_mode = WAL")
|
|
24
|
+
db.exec("PRAGMA busy_timeout = 10000")
|
|
25
|
+
db.exec("PRAGMA synchronous = OFF")
|
|
26
|
+
|
|
27
|
+
// `kdb` wraps `db` for the DDL (the house idiom); the caller owns `db`'s lifecycle, so we don't
|
|
28
|
+
// destroy it here. The bulk INSERTs (populateAncestors + build-unified-wof) stay on the raw handle.
|
|
29
|
+
const kdb = new DatabaseClient<WOFDatabase>({ database: db })
|
|
30
|
+
|
|
31
|
+
await kdb.schema
|
|
32
|
+
.createTable("spr")
|
|
33
|
+
.ifNotExists()
|
|
34
|
+
.addColumn("id", "integer", (c) => c.primaryKey())
|
|
35
|
+
.addColumn("parent_id", "integer", (c) => c.notNull().defaultTo(-1))
|
|
36
|
+
.addColumn("name", "text", (c) => c.notNull().defaultTo(""))
|
|
37
|
+
.addColumn("placetype", "text", (c) => c.notNull().defaultTo(""))
|
|
38
|
+
.addColumn("country", "text", (c) => c.notNull().defaultTo(""))
|
|
39
|
+
.addColumn("latitude", "real", (c) => c.notNull().defaultTo(0))
|
|
40
|
+
.addColumn("longitude", "real", (c) => c.notNull().defaultTo(0))
|
|
41
|
+
.addColumn("min_latitude", "real", (c) => c.notNull().defaultTo(0))
|
|
42
|
+
.addColumn("min_longitude", "real", (c) => c.notNull().defaultTo(0))
|
|
43
|
+
.addColumn("max_latitude", "real", (c) => c.notNull().defaultTo(0))
|
|
44
|
+
.addColumn("max_longitude", "real", (c) => c.notNull().defaultTo(0))
|
|
45
|
+
.addColumn("is_current", "integer", (c) => c.notNull().defaultTo(1))
|
|
46
|
+
.addColumn("is_deprecated", "integer", (c) => c.notNull().defaultTo(0))
|
|
47
|
+
.addColumn("is_ceased", "integer", (c) => c.notNull().defaultTo(0))
|
|
48
|
+
.addColumn("is_superseded", "integer", (c) => c.notNull().defaultTo(0))
|
|
49
|
+
.addColumn("is_superseding", "integer", (c) => c.notNull().defaultTo(0))
|
|
50
|
+
.addColumn("lastmodified", "integer", (c) => c.notNull().defaultTo(0))
|
|
51
|
+
.execute()
|
|
52
|
+
|
|
53
|
+
// `privateuse` carries WOF's x_<variant> kind (preferred | variant) / GeoNames' isPreferredName
|
|
54
|
+
// ("preferred" | ""). `official` is the #936 ingest bit: 1 when the row's language is an official
|
|
55
|
+
// language of the place's country (codex OFFICIAL_LANGUAGES) AND the row is a preferred form —
|
|
56
|
+
// x_variant rows tagged with an official language ("MSP", "Frisco") stay 0. Primary-name mirror
|
|
57
|
+
// rows stay 0 too: the name-exact tier already consults spr.name; `official` only marks the
|
|
58
|
+
// ALIASES eligible to join it. Both are ingest-time facts, never computed at query time.
|
|
59
|
+
await kdb.schema
|
|
60
|
+
.createTable("names")
|
|
61
|
+
.ifNotExists()
|
|
62
|
+
.addColumn("id", "integer", (c) => c.notNull())
|
|
63
|
+
.addColumn("name", "text", (c) => c.notNull())
|
|
64
|
+
.addColumn("placetype", "text", (c) => c.notNull().defaultTo(""))
|
|
65
|
+
.addColumn("country", "text", (c) => c.notNull().defaultTo(""))
|
|
66
|
+
.addColumn("language", "text", (c) => c.notNull().defaultTo(""))
|
|
67
|
+
.addColumn("privateuse", "text", (c) => c.notNull().defaultTo(""))
|
|
68
|
+
.addColumn("official", "integer", (c) => c.notNull().defaultTo(0))
|
|
69
|
+
.addColumn("lastmodified", "integer", (c) => c.notNull().defaultTo(0))
|
|
70
|
+
.execute()
|
|
71
|
+
|
|
72
|
+
await kdb.schema
|
|
73
|
+
.createTable("concordances")
|
|
74
|
+
.ifNotExists()
|
|
75
|
+
.addColumn("id", "integer", (c) => c.notNull())
|
|
76
|
+
.addColumn("other_id", "text", (c) => c.notNull())
|
|
77
|
+
.addColumn("other_source", "text", (c) => c.notNull())
|
|
78
|
+
.addColumn("lastmodified", "integer", (c) => c.notNull().defaultTo(0))
|
|
79
|
+
.execute()
|
|
80
|
+
|
|
81
|
+
await kdb.schema
|
|
82
|
+
.createTable("place_population")
|
|
83
|
+
.ifNotExists()
|
|
84
|
+
.addColumn("id", "integer", (c) => c.primaryKey())
|
|
85
|
+
.addColumn("population", "integer", (c) => c.notNull().defaultTo(0))
|
|
86
|
+
.execute()
|
|
87
|
+
|
|
88
|
+
// `ancestors` maps each place to every place above it in the hierarchy (and itself). The
|
|
89
|
+
// resolver's parent-constraint scopes a child lookup to a parent's descendants via
|
|
90
|
+
// `spr.id IN (SELECT id FROM ancestors WHERE ancestor_id = ?)`. The off-the-shelf WOF dumps
|
|
91
|
+
// ship this table; our build derives it from the parent_id chain (see populateAncestors) since
|
|
92
|
+
// we don't capture `wof:hierarchy`.
|
|
93
|
+
await kdb.schema
|
|
94
|
+
.createTable("ancestors")
|
|
95
|
+
.ifNotExists()
|
|
96
|
+
.addColumn("id", "integer", (c) => c.notNull())
|
|
97
|
+
.addColumn("ancestor_id", "integer", (c) => c.notNull())
|
|
98
|
+
.addColumn("ancestor_placetype", "text", (c) => c.notNull().defaultTo(""))
|
|
99
|
+
.addColumn("lastmodified", "integer", (c) => c.notNull().defaultTo(0))
|
|
100
|
+
.execute()
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Populate the `ancestors` table by walking each place's `parent_id` chain in `spr` (transitive closure, including the
|
|
105
|
+
* place itself). Idempotent: drops + rebuilds the table contents. Returns the row count. Run after `spr` is fully
|
|
106
|
+
* ingested (build-unified-wof freeze phase) or standalone on an existing unified DB (`scripts/add-ancestors.ts`).
|
|
107
|
+
* Sentinel/negative parent_ids and cycles terminate the walk. ~4 rows/place average; a transaction keeps the ~5M
|
|
108
|
+
* inserts fast.
|
|
109
|
+
*/
|
|
110
|
+
export function populateAncestors(db: DatabaseSync): number {
|
|
111
|
+
db.exec("DELETE FROM ancestors")
|
|
112
|
+
const rows = db.prepare("SELECT id, parent_id, placetype FROM spr").all() as Array<{
|
|
113
|
+
id: number
|
|
114
|
+
parent_id: number
|
|
115
|
+
placetype: string
|
|
116
|
+
}>
|
|
117
|
+
const byID = new Map<number, { parent: number; placetype: string }>()
|
|
118
|
+
|
|
119
|
+
for (const r of rows) {
|
|
120
|
+
byID.set(r.id, { parent: r.parent_id, placetype: r.placetype })
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const insert = db.prepare("INSERT INTO ancestors (id, ancestor_id, ancestor_placetype) VALUES (?, ?, ?)")
|
|
124
|
+
db.exec("BEGIN")
|
|
125
|
+
let count = 0
|
|
126
|
+
|
|
127
|
+
for (const r of rows) {
|
|
128
|
+
insert.run(r.id, r.id, r.placetype) // self
|
|
129
|
+
count++
|
|
130
|
+
const seen = new Set<number>([r.id])
|
|
131
|
+
let cur = r.parent_id
|
|
132
|
+
|
|
133
|
+
while (cur > 0 && !seen.has(cur)) {
|
|
134
|
+
const node = byID.get(cur)
|
|
135
|
+
|
|
136
|
+
if (!node) break
|
|
137
|
+
insert.run(r.id, cur, node.placetype)
|
|
138
|
+
count++
|
|
139
|
+
seen.add(cur)
|
|
140
|
+
cur = node.parent
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
db.exec("COMMIT")
|
|
144
|
+
|
|
145
|
+
return count
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export async function createUnifiedIndexes(db: DatabaseSync): Promise<void> {
|
|
149
|
+
const kdb = new DatabaseClient<WOFDatabase>({ database: db })
|
|
150
|
+
await kdb.schema.createIndex("spr_by_placetype").ifNotExists().on("spr").column("placetype").execute()
|
|
151
|
+
await kdb.schema.createIndex("spr_by_country").ifNotExists().on("spr").column("country").execute()
|
|
152
|
+
await kdb.schema.createIndex("spr_by_parent").ifNotExists().on("spr").column("parent_id").execute()
|
|
153
|
+
await kdb.schema.createIndex("names_by_id").ifNotExists().on("names").column("id").execute()
|
|
154
|
+
await kdb.schema.createIndex("names_by_name").ifNotExists().on("names").column("name").execute()
|
|
155
|
+
await kdb.schema
|
|
156
|
+
.createIndex("concordances_by_id")
|
|
157
|
+
.ifNotExists()
|
|
158
|
+
.on("concordances")
|
|
159
|
+
.columns(["id", "lastmodified"])
|
|
160
|
+
.execute()
|
|
161
|
+
await kdb.schema
|
|
162
|
+
.createIndex("concordances_by_other_id")
|
|
163
|
+
.ifNotExists()
|
|
164
|
+
.on("concordances")
|
|
165
|
+
.columns(["other_source", "other_id"])
|
|
166
|
+
.execute()
|
|
167
|
+
// ancestor_id is the hot column (parent-constraint queries `WHERE ancestor_id = ?`); id supports
|
|
168
|
+
// the reverse lookup.
|
|
169
|
+
await kdb.schema.createIndex("ancestors_by_ancestor").ifNotExists().on("ancestors").column("ancestor_id").execute()
|
|
170
|
+
await kdb.schema.createIndex("ancestors_by_id").ifNotExists().on("ancestors").column("id").execute()
|
|
171
|
+
}
|