@mailwoman/corpus 8.6.0 → 9.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (310) hide show
  1. package/README.md +1 -1
  2. package/out/src/adapter.d.ts +41 -0
  3. package/out/src/adapter.d.ts.map +1 -1
  4. package/out/src/adapter.js +46 -1
  5. package/out/src/adapter.js.map +1 -1
  6. package/out/src/adapters/ban/adapter.d.ts +4 -3
  7. package/out/src/adapters/ban/adapter.d.ts.map +1 -1
  8. package/out/src/adapters/ban/adapter.js +60 -67
  9. package/out/src/adapters/ban/adapter.js.map +1 -1
  10. package/out/src/adapters/ban/street-decompose.d.ts.map +1 -1
  11. package/out/src/adapters/ban/street-decompose.js +18 -25
  12. package/out/src/adapters/ban/street-decompose.js.map +1 -1
  13. package/out/src/adapters/fcc-bdc/adapter.d.ts +0 -6
  14. package/out/src/adapters/fcc-bdc/adapter.d.ts.map +1 -1
  15. package/out/src/adapters/fcc-bdc/adapter.js +2 -24
  16. package/out/src/adapters/fcc-bdc/adapter.js.map +1 -1
  17. package/out/src/adapters/geonames/adapter.d.ts.map +1 -1
  18. package/out/src/adapters/geonames/adapter.js +70 -76
  19. package/out/src/adapters/geonames/adapter.js.map +1 -1
  20. package/out/src/adapters/geonames-postal/adapter.d.ts.map +1 -1
  21. package/out/src/adapters/geonames-postal/adapter.js +44 -49
  22. package/out/src/adapters/geonames-postal/adapter.js.map +1 -1
  23. package/out/src/adapters/gnaf/adapter.d.ts.map +1 -1
  24. package/out/src/adapters/gnaf/adapter.js +4 -7
  25. package/out/src/adapters/gnaf/adapter.js.map +1 -1
  26. package/out/src/adapters/gnaf/assemble.d.ts.map +1 -1
  27. package/out/src/adapters/gnaf/assemble.js +6 -11
  28. package/out/src/adapters/gnaf/assemble.js.map +1 -1
  29. package/out/src/adapters/openaddresses/adapter.d.ts.map +1 -1
  30. package/out/src/adapters/openaddresses/adapter.js +2 -7
  31. package/out/src/adapters/openaddresses/adapter.js.map +1 -1
  32. package/out/src/adapters/overture/adapter.d.ts.map +1 -1
  33. package/out/src/adapters/overture/adapter.js +3 -7
  34. package/out/src/adapters/overture/adapter.js.map +1 -1
  35. package/out/src/adapters/state-hi-schools/adapter.d.ts.map +1 -1
  36. package/out/src/adapters/state-hi-schools/adapter.js +55 -73
  37. package/out/src/adapters/state-hi-schools/adapter.js.map +1 -1
  38. package/out/src/adapters/state-ia-contractors/adapter.d.ts.map +1 -1
  39. package/out/src/adapters/state-ia-contractors/adapter.js +58 -76
  40. package/out/src/adapters/state-ia-contractors/adapter.js.map +1 -1
  41. package/out/src/adapters/state-ny-notaries/adapter.d.ts.map +1 -1
  42. package/out/src/adapters/state-ny-notaries/adapter.js +67 -85
  43. package/out/src/adapters/state-ny-notaries/adapter.js.map +1 -1
  44. package/out/src/adapters/state-tx-notaries/adapter.d.ts.map +1 -1
  45. package/out/src/adapters/state-tx-notaries/adapter.js +69 -87
  46. package/out/src/adapters/state-tx-notaries/adapter.js.map +1 -1
  47. package/out/src/adapters/synth-po-box/adapter.d.ts.map +1 -1
  48. package/out/src/adapters/synth-po-box/adapter.js +7 -15
  49. package/out/src/adapters/synth-po-box/adapter.js.map +1 -1
  50. package/out/src/adapters/tiger/street-decompose.d.ts.map +1 -1
  51. package/out/src/adapters/tiger/street-decompose.js +18 -27
  52. package/out/src/adapters/tiger/street-decompose.js.map +1 -1
  53. package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts +3 -3
  54. package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts.map +1 -1
  55. package/out/src/adapters/usgov-hrsa-fqhc/adapter.js +62 -80
  56. package/out/src/adapters/usgov-hrsa-fqhc/adapter.js.map +1 -1
  57. package/out/src/adapters/usgov-imls-pls/adapter.d.ts.map +1 -1
  58. package/out/src/adapters/usgov-imls-pls/adapter.js +60 -82
  59. package/out/src/adapters/usgov-imls-pls/adapter.js.map +1 -1
  60. package/out/src/adapters/usgov-irs-bmf/adapter.d.ts.map +1 -1
  61. package/out/src/adapters/usgov-irs-bmf/adapter.js +58 -65
  62. package/out/src/adapters/usgov-irs-bmf/adapter.js.map +1 -1
  63. package/out/src/adapters/usgov-nad/adapter.d.ts.map +1 -1
  64. package/out/src/adapters/usgov-nad/adapter.js +5 -8
  65. package/out/src/adapters/usgov-nad/adapter.js.map +1 -1
  66. package/out/src/adapters/usgov-nppes/adapter.d.ts.map +1 -1
  67. package/out/src/adapters/usgov-nppes/adapter.js +58 -76
  68. package/out/src/adapters/usgov-nppes/adapter.js.map +1 -1
  69. package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.d.ts.map +1 -1
  70. package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js +51 -71
  71. package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js.map +1 -1
  72. package/out/src/adapters/wof-admin-jp/adapter.d.ts.map +1 -1
  73. package/out/src/adapters/wof-admin-jp/adapter.js +24 -6
  74. package/out/src/adapters/wof-admin-jp/adapter.js.map +1 -1
  75. package/out/src/build.d.ts.map +1 -1
  76. package/out/src/build.js +2 -1
  77. package/out/src/build.js.map +1 -1
  78. package/out/src/golden.d.ts +5 -0
  79. package/out/src/golden.d.ts.map +1 -1
  80. package/out/src/golden.js +23 -9
  81. package/out/src/golden.js.map +1 -1
  82. package/out/src/parquet-wrapper/reader.d.ts +8 -0
  83. package/out/src/parquet-wrapper/reader.d.ts.map +1 -1
  84. package/out/src/parquet-wrapper/reader.js +16 -0
  85. package/out/src/parquet-wrapper/reader.js.map +1 -1
  86. package/out/src/parquet-wrapper/schema.d.ts +4 -0
  87. package/out/src/parquet-wrapper/schema.d.ts.map +1 -1
  88. package/out/src/parquet-wrapper/schema.js.map +1 -1
  89. package/out/src/parquet.d.ts.map +1 -1
  90. package/out/src/parquet.js +2 -13
  91. package/out/src/parquet.js.map +1 -1
  92. package/out/src/shard-recipes/anchor-absorption.d.ts.map +1 -1
  93. package/out/src/shard-recipes/anchor-absorption.js +2 -1
  94. package/out/src/shard-recipes/anchor-absorption.js.map +1 -1
  95. package/out/src/shard-recipes/country-balanced.d.ts.map +1 -1
  96. package/out/src/shard-recipes/country-balanced.js +68 -60
  97. package/out/src/shard-recipes/country-balanced.js.map +1 -1
  98. package/out/src/shard-recipes/fr-fragment.d.ts.map +1 -1
  99. package/out/src/shard-recipes/fr-fragment.js +8 -5
  100. package/out/src/shard-recipes/fr-fragment.js.map +1 -1
  101. package/out/src/shard-recipes/fr-order.d.ts.map +1 -1
  102. package/out/src/shard-recipes/fr-order.js +14 -58
  103. package/out/src/shard-recipes/fr-order.js.map +1 -1
  104. package/out/src/shard-recipes/german.d.ts.map +1 -1
  105. package/out/src/shard-recipes/german.js +11 -58
  106. package/out/src/shard-recipes/german.js.map +1 -1
  107. package/out/src/shard-recipes/index.d.ts.map +1 -1
  108. package/out/src/shard-recipes/index.js +2 -0
  109. package/out/src/shard-recipes/index.js.map +1 -1
  110. package/out/src/shard-recipes/intersection.d.ts +2 -2
  111. package/out/src/shard-recipes/intersection.d.ts.map +1 -1
  112. package/out/src/shard-recipes/intersection.js +24 -62
  113. package/out/src/shard-recipes/intersection.js.map +1 -1
  114. package/out/src/shard-recipes/locale.d.ts.map +1 -1
  115. package/out/src/shard-recipes/locale.js +14 -16
  116. package/out/src/shard-recipes/locale.js.map +1 -1
  117. package/out/src/shard-recipes/no-fragment.d.ts.map +1 -1
  118. package/out/src/shard-recipes/no-fragment.js +8 -5
  119. package/out/src/shard-recipes/no-fragment.js.map +1 -1
  120. package/out/src/shard-recipes/no-street-led.d.ts.map +1 -1
  121. package/out/src/shard-recipes/no-street-led.js +8 -5
  122. package/out/src/shard-recipes/no-street-led.js.map +1 -1
  123. package/out/src/shard-recipes/po-box-cedex.d.ts +3 -3
  124. package/out/src/shard-recipes/po-box-cedex.d.ts.map +1 -1
  125. package/out/src/shard-recipes/po-box-cedex.js +51 -94
  126. package/out/src/shard-recipes/po-box-cedex.js.map +1 -1
  127. package/out/src/shard-recipes/scaffold.d.ts +59 -9
  128. package/out/src/shard-recipes/scaffold.d.ts.map +1 -1
  129. package/out/src/shard-recipes/scaffold.js +51 -30
  130. package/out/src/shard-recipes/scaffold.js.map +1 -1
  131. package/out/src/shard-recipes/street-affix.d.ts.map +1 -1
  132. package/out/src/shard-recipes/street-affix.js +56 -82
  133. package/out/src/shard-recipes/street-affix.js.map +1 -1
  134. package/out/src/shard-recipes/sub-venue-sources.d.ts +231 -0
  135. package/out/src/shard-recipes/sub-venue-sources.d.ts.map +1 -0
  136. package/out/src/shard-recipes/sub-venue-sources.js +669 -0
  137. package/out/src/shard-recipes/sub-venue-sources.js.map +1 -0
  138. package/out/src/shard-recipes/sub-venue.d.ts +269 -0
  139. package/out/src/shard-recipes/sub-venue.d.ts.map +1 -0
  140. package/out/src/shard-recipes/sub-venue.js +709 -0
  141. package/out/src/shard-recipes/sub-venue.js.map +1 -0
  142. package/out/src/shard-recipes/unit.d.ts +1 -1
  143. package/out/src/shard-recipes/unit.d.ts.map +1 -1
  144. package/out/src/shard-recipes/unit.js +22 -64
  145. package/out/src/shard-recipes/unit.js.map +1 -1
  146. package/out/src/split.d.ts +7 -0
  147. package/out/src/split.d.ts.map +1 -1
  148. package/out/src/split.js +7 -0
  149. package/out/src/split.js.map +1 -1
  150. package/out/src/synthesize.d.ts.map +1 -1
  151. package/out/src/synthesize.js +1 -12
  152. package/out/src/synthesize.js.map +1 -1
  153. package/out/src/tools/align-shard.d.ts.map +1 -1
  154. package/out/src/tools/align-shard.js +3 -7
  155. package/out/src/tools/align-shard.js.map +1 -1
  156. package/out/src/tools/audit.d.ts.map +1 -1
  157. package/out/src/tools/audit.js +4 -4
  158. package/out/src/tools/audit.js.map +1 -1
  159. package/out/src/tools/corpus-stats.d.ts +2 -2
  160. package/out/src/tools/corpus-stats.d.ts.map +1 -1
  161. package/out/src/tools/corpus-stats.js +18 -31
  162. package/out/src/tools/corpus-stats.js.map +1 -1
  163. package/out/src/tools/fetch/download.d.ts.map +1 -1
  164. package/out/src/tools/fetch/download.js +5 -6
  165. package/out/src/tools/fetch/download.js.map +1 -1
  166. package/out/src/tools/fetch/imls-pls.d.ts.map +1 -1
  167. package/out/src/tools/fetch/imls-pls.js +2 -2
  168. package/out/src/tools/fetch/imls-pls.js.map +1 -1
  169. package/out/src/tools/fetch/index.d.ts +14 -1
  170. package/out/src/tools/fetch/index.d.ts.map +1 -1
  171. package/out/src/tools/fetch/index.js +14 -1
  172. package/out/src/tools/fetch/index.js.map +1 -1
  173. package/out/src/tools/fetch/nad.d.ts +1 -1
  174. package/out/src/tools/fetch/nad.js +1 -1
  175. package/out/src/tools/fetch/nppes.d.ts.map +1 -1
  176. package/out/src/tools/fetch/nppes.js +2 -1
  177. package/out/src/tools/fetch/nppes.js.map +1 -1
  178. package/out/src/tools/fetch/openaddresses.d.ts +1 -1
  179. package/out/src/tools/fetch/openaddresses.js +1 -1
  180. package/out/src/tools/fetch/ourairports.d.ts +45 -0
  181. package/out/src/tools/fetch/ourairports.d.ts.map +1 -0
  182. package/out/src/tools/fetch/ourairports.js +123 -0
  183. package/out/src/tools/fetch/ourairports.js.map +1 -0
  184. package/out/src/tools/fetch/wikidata-subvenue.d.ts +171 -0
  185. package/out/src/tools/fetch/wikidata-subvenue.d.ts.map +1 -0
  186. package/out/src/tools/fetch/wikidata-subvenue.js +275 -0
  187. package/out/src/tools/fetch/wikidata-subvenue.js.map +1 -0
  188. package/out/src/tools/golden-expand.d.ts.map +1 -1
  189. package/out/src/tools/golden-expand.js +18 -16
  190. package/out/src/tools/golden-expand.js.map +1 -1
  191. package/out/src/tools/golden-promote.d.ts.map +1 -1
  192. package/out/src/tools/golden-promote.js +12 -6
  193. package/out/src/tools/golden-promote.js.map +1 -1
  194. package/out/src/tools/golden-relabel-street.d.ts +196 -0
  195. package/out/src/tools/golden-relabel-street.d.ts.map +1 -0
  196. package/out/src/tools/golden-relabel-street.js +513 -0
  197. package/out/src/tools/golden-relabel-street.js.map +1 -0
  198. package/out/src/tools/index.d.ts +4 -0
  199. package/out/src/tools/index.d.ts.map +1 -1
  200. package/out/src/tools/index.js +4 -0
  201. package/out/src/tools/index.js.map +1 -1
  202. package/out/src/tools/jsonl-to-parquet.d.ts.map +1 -1
  203. package/out/src/tools/jsonl-to-parquet.js +3 -2
  204. package/out/src/tools/jsonl-to-parquet.js.map +1 -1
  205. package/out/src/tools/lint-shard-vocab.d.ts.map +1 -1
  206. package/out/src/tools/lint-shard-vocab.js +3 -2
  207. package/out/src/tools/lint-shard-vocab.js.map +1 -1
  208. package/out/src/tools/lint-shard.d.ts +3 -3
  209. package/out/src/tools/lint-shard.d.ts.map +1 -1
  210. package/out/src/tools/lint-shard.js +23 -32
  211. package/out/src/tools/lint-shard.js.map +1 -1
  212. package/out/src/tools/overlay-manifest.d.ts.map +1 -1
  213. package/out/src/tools/overlay-manifest.js +2 -1
  214. package/out/src/tools/overlay-manifest.js.map +1 -1
  215. package/out/src/tools/overture-subvenue.d.ts +112 -0
  216. package/out/src/tools/overture-subvenue.d.ts.map +1 -0
  217. package/out/src/tools/overture-subvenue.js +144 -0
  218. package/out/src/tools/overture-subvenue.js.map +1 -0
  219. package/out/src/tools/shard-kryptonite.d.ts +1 -1
  220. package/out/src/tools/shard-kryptonite.d.ts.map +1 -1
  221. package/out/src/tools/shard-kryptonite.js +5 -4
  222. package/out/src/tools/shard-kryptonite.js.map +1 -1
  223. package/out/src/tools/shard-translit.d.ts +6 -3
  224. package/out/src/tools/shard-translit.d.ts.map +1 -1
  225. package/out/src/tools/shard-translit.js +8 -6
  226. package/out/src/tools/shard-translit.js.map +1 -1
  227. package/out/src/tools/sub-venue-lexicon.d.ts +507 -0
  228. package/out/src/tools/sub-venue-lexicon.d.ts.map +1 -0
  229. package/out/src/tools/sub-venue-lexicon.js +817 -0
  230. package/out/src/tools/sub-venue-lexicon.js.map +1 -0
  231. package/out/src/tools/sub-venue-promotions.d.ts +94 -0
  232. package/out/src/tools/sub-venue-promotions.d.ts.map +1 -0
  233. package/out/src/tools/sub-venue-promotions.js +266 -0
  234. package/out/src/tools/sub-venue-promotions.js.map +1 -0
  235. package/out/src/wof-json.d.ts +2 -2
  236. package/out/src/wof-json.d.ts.map +1 -1
  237. package/out/src/wof-json.js +5 -8
  238. package/out/src/wof-json.js.map +1 -1
  239. package/package.json +8 -8
  240. package/src/adapter.ts +58 -1
  241. package/src/adapters/ban/adapter.ts +55 -65
  242. package/src/adapters/ban/street-decompose.ts +17 -25
  243. package/src/adapters/fcc-bdc/adapter.ts +2 -33
  244. package/src/adapters/geonames/adapter.ts +64 -72
  245. package/src/adapters/geonames-postal/adapter.ts +42 -50
  246. package/src/adapters/gnaf/adapter.ts +4 -7
  247. package/src/adapters/gnaf/assemble.ts +6 -10
  248. package/src/adapters/openaddresses/adapter.ts +2 -7
  249. package/src/adapters/overture/adapter.ts +3 -6
  250. package/src/adapters/state-hi-schools/adapter.ts +50 -73
  251. package/src/adapters/state-ia-contractors/adapter.ts +52 -76
  252. package/src/adapters/state-ny-notaries/adapter.ts +61 -85
  253. package/src/adapters/state-tx-notaries/adapter.ts +60 -84
  254. package/src/adapters/synth-po-box/adapter.ts +7 -17
  255. package/src/adapters/tiger/street-decompose.ts +21 -31
  256. package/src/adapters/usgov-hrsa-fqhc/adapter.ts +61 -84
  257. package/src/adapters/usgov-imls-pls/adapter.ts +54 -82
  258. package/src/adapters/usgov-irs-bmf/adapter.ts +53 -63
  259. package/src/adapters/usgov-nad/adapter.ts +5 -8
  260. package/src/adapters/usgov-nppes/adapter.ts +51 -75
  261. package/src/adapters/usgov-samhsa-treatment-locator/adapter.ts +45 -71
  262. package/src/adapters/wof-admin-jp/adapter.ts +34 -8
  263. package/src/build.ts +2 -1
  264. package/src/golden.ts +24 -9
  265. package/src/parquet-wrapper/reader.ts +19 -0
  266. package/src/parquet-wrapper/schema.ts +4 -0
  267. package/src/parquet.ts +3 -16
  268. package/src/shard-recipes/anchor-absorption.ts +2 -1
  269. package/src/shard-recipes/country-balanced.ts +69 -70
  270. package/src/shard-recipes/fr-fragment.ts +10 -7
  271. package/src/shard-recipes/fr-order.ts +13 -63
  272. package/src/shard-recipes/german.ts +12 -65
  273. package/src/shard-recipes/index.ts +2 -0
  274. package/src/shard-recipes/intersection.ts +32 -60
  275. package/src/shard-recipes/locale.ts +14 -16
  276. package/src/shard-recipes/no-fragment.ts +10 -7
  277. package/src/shard-recipes/no-street-led.ts +10 -7
  278. package/src/shard-recipes/po-box-cedex.ts +68 -119
  279. package/src/shard-recipes/scaffold.ts +82 -28
  280. package/src/shard-recipes/street-affix.ts +58 -95
  281. package/src/shard-recipes/sub-venue-sources.ts +895 -0
  282. package/src/shard-recipes/sub-venue.ts +982 -0
  283. package/src/shard-recipes/unit.ts +22 -70
  284. package/src/split.ts +7 -0
  285. package/src/synthesize.ts +1 -15
  286. package/src/tools/align-shard.ts +3 -6
  287. package/src/tools/audit.ts +6 -5
  288. package/src/tools/corpus-stats.ts +25 -33
  289. package/src/tools/fetch/download.ts +7 -5
  290. package/src/tools/fetch/imls-pls.ts +2 -2
  291. package/src/tools/fetch/index.ts +14 -1
  292. package/src/tools/fetch/nad.ts +1 -1
  293. package/src/tools/fetch/nppes.ts +2 -1
  294. package/src/tools/fetch/openaddresses.ts +1 -1
  295. package/src/tools/fetch/ourairports.ts +166 -0
  296. package/src/tools/fetch/wikidata-subvenue.ts +386 -0
  297. package/src/tools/golden-expand.ts +18 -13
  298. package/src/tools/golden-promote.ts +15 -6
  299. package/src/tools/golden-relabel-street.ts +748 -0
  300. package/src/tools/index.ts +4 -0
  301. package/src/tools/jsonl-to-parquet.ts +3 -2
  302. package/src/tools/lint-shard-vocab.ts +3 -2
  303. package/src/tools/lint-shard.ts +33 -36
  304. package/src/tools/overlay-manifest.ts +2 -1
  305. package/src/tools/overture-subvenue.ts +215 -0
  306. package/src/tools/shard-kryptonite.ts +5 -4
  307. package/src/tools/shard-translit.ts +12 -7
  308. package/src/tools/sub-venue-lexicon.ts +1250 -0
  309. package/src/tools/sub-venue-promotions.ts +330 -0
  310. package/src/wof-json.ts +5 -8
@@ -0,0 +1,1250 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Build the sub-venue designator lexicon (#35) — the vocabulary a corpus shard and, eventually, the
7
+ * span proposer read to recognize `Terminal 5`, `North Terminal`, `Concourse B`, `ターミナル1` as
8
+ * venue-INTERIOR structure.
9
+ *
10
+ * Reads the fetch outputs (`mailwoman corpus fetch wikidata-subvenue`, a JSONL of
11
+ * `@mailwoman/osm/sdk`'s `SubVenueSourceRow`s per region, and the Overture slice of `poi.db` via
12
+ * `overture-subvenue.ts`) and emits one committed JSON table. The shape follows
13
+ * `@mailwoman/poi-taxonomy`'s `taxonomy.json` idiom exactly — typed records plus a FLAT phrase array
14
+ * keyed back to a record id, which is what makes a longest-match phrase index cheap to build over it.
15
+ * {@link SubVenueSurface} is this table's `SynonymEntry`.
16
+ *
17
+ * ── Determinism ──────────────────────────────────────────────────────────────────────────────────
18
+ * {@link buildSubVenueLexicon} is a PURE function of its inputs with a stable sort on every array, so
19
+ * a regenerate against the same fetch outputs is byte-identical. No timestamp is emitted for the same
20
+ * reason `taxonomy.json` carries none — a clock in the artifact makes every regenerate a diff.
21
+ * Vintages live in `sources[]`, taken from the fetch manifests.
22
+ *
23
+ * ── The findings this build is SHAPED BY, all measured ───────────────────────────────────────────
24
+ *
25
+ * **1. A feature's `name` is usually the VENUE's name, not a sub-venue phrase.** Measured on the
26
+ * Berlin extract (2,060 matched features, 2026-08-04): a `railway=platform` is named `Stendaler
27
+ * Straße`, a `railway=station` is named `Bellevue`, an `amenity=university` is named `Hertie School`.
28
+ * Across 250,116 named Great Britain features only 6,003 (2.40%) contain a designator token at all.
29
+ * So names are NOT harvested wholesale — {@link extractAttestedPhrases} keeps a name only when it
30
+ * CONTAINS a known designator surface, which is what makes `Terminal E (Untere Ebene)` evidence and
31
+ * `Otto Lilienthal Flughafen Berlin Tegel` not.
32
+ *
33
+ * **2. The identifier lives in `ref`, not in `name`.** Every one of Berlin's 26 `aeroway=gate`
34
+ * features is unnamed and carries only a `ref`: `13`, `6`, `0/1`, `14/15`, `16-18`. That means
35
+ * `Gate A12` is a RENDERING (`<designator> <ref>`) rather than a string anyone has written down, and a
36
+ * shard that wants to generate the designator+identifier form needs the identifier DISTRIBUTION, not
37
+ * a list of phrases. {@link IdentifierShape} is that distribution, and it is why the artifact has a
38
+ * section for it at all.
39
+ *
40
+ * **3. A matched phrase belongs to the record the PHRASE names, not to the record the ROW carries.**
41
+ * Wave 1 attributed every hit to `row.designatorID`, which is the rule that matched the FEATURE. On
42
+ * the GB extract that produced `west → platform`, `hall → platform`, `biggin → platform` — 108 of 133
43
+ * OSM-derived surfaces had a `phrase` that named a different record than the one they pointed at,
44
+ * because a bus stop tagged `public_transport=platform` is named "Village Hall" or "West Kensington".
45
+ * {@link extractAttestedPhrases} now takes a phrase → record INDEX and attributes by phrase; the row's
46
+ * own designator is kept as `context`, which is exactly the axis a confound board needs (a `hall` seen
47
+ * on a platform is a confound; a `hall` seen on a terminal is evidence).
48
+ *
49
+ * ── What `curated: false` means, and how a surface stops being it ────────────────────────────────
50
+ * Wikidata gives a CONCEPT NAME per language, not a designator as addressed. Q849706's Spanish label
51
+ * is `terminal aeroportuaria`; the addressed form is `Terminal`. Q240854 (`hall`) is `sala` in
52
+ * Italian, which names an ordinary room and would fire on half of Italy. So every machine-derived
53
+ * surface lands `curated: false`, and {@link SubVenueLexiconTable} consumers that gate parsing MUST
54
+ * filter to `curated: true`.
55
+ *
56
+ * A surface becomes curated ONLY by matching a {@link SubVenuePromotion} in
57
+ * `sub-venue-promotions.ts` — a per-designator, per-LOCALE decision carrying the census that backs
58
+ * it. Promotion is per-locale because the same token is a designator in one language and a disaster
59
+ * in another: `hall` is `Halle 2` at Frankfurt and `Village Hall` at 3,205 British bus stops.
60
+ *
61
+ * ── The seed DUPLICATES `neural/venue-structure.ts`, knowingly ───────────────────────────────────
62
+ * `@mailwoman/corpus` does not depend on `@mailwoman/neural` (the dependency runs the other way for
63
+ * the training path, and pulling onnxruntime into a corpus build to read three string arrays would be
64
+ * absurd), so the shipped vocabulary is re-declared below. That is a drift surface and it is stated
65
+ * rather than hidden: `sub-venue-lexicon.test.ts` pins the seed's contents literally, so a change in
66
+ * either place fails a test rather than passing silently.
67
+ */
68
+
69
+ import { readFileSync, statSync, writeFileSync } from "node:fs"
70
+ import { basename, join } from "node:path"
71
+
72
+ import { parseJSONStrict } from "@mailwoman/core/objects"
73
+ import { TextSpliterator } from "spliterator"
74
+
75
+ import { SUBVENUE_PROMOTIONS, type SubVenuePromotion } from "./sub-venue-promotions.ts"
76
+
77
+ /**
78
+ * This table's own data version. Bump when the source vintages or the build semantics change.
79
+ *
80
+ * `0.2.0` — wave 2: per-region attestation, the phrase-attribution fix (finding 3 above), derived head nouns, the
81
+ * Overture source, and the `promotions[]` receipts section.
82
+ */
83
+ export const SUBVENUE_LEXICON_VERSION = "0.2.0"
84
+
85
+ /**
86
+ * Which side of the containment relation a designator names. Mirrors `@mailwoman/osm/sdk`'s `SubVenueTier`; re-declared
87
+ * for the same dependency-direction reason as the seed.
88
+ */
89
+ export const LexiconTier = {
90
+ SubVenue: "subvenue",
91
+ Venue: "venue",
92
+ } as const
93
+
94
+ export type LexiconTier = (typeof LexiconTier)[keyof typeof LexiconTier]
95
+
96
+ /**
97
+ * One designator record — a venue-interior (or containing-venue) structural noun.
98
+ */
99
+ export interface SubVenueDesignator {
100
+ /**
101
+ * Canonical id, lowercase English. Matches `neural/venue-structure.ts`'s `VENUE_STRUCTURE_DESIGNATORS` wherever the
102
+ * two overlap.
103
+ */
104
+ id: string
105
+ tier: LexiconTier
106
+ /**
107
+ * Whether this designator may be preceded by a {@link SubVenueModifier} — the `North Terminal` shape.
108
+ *
109
+ * A SUBSET, and the exclusions are load-bearing: `gate` and `building` form ordinary STREET names in exactly this
110
+ * shape ("East Gate" is a real GB street, "Building Society Place" is a real street), so admitting them turns a
111
+ * correct street parse into a sub-venue one. Setting this true means claiming no street is named `<modifier> <id>`.
112
+ * Check before you do.
113
+ */
114
+ modifierEligible: boolean
115
+ /**
116
+ * Whether the shipped span proposer already recognizes this designator. `false` means the lexicon proposes it and
117
+ * nothing consumes it yet.
118
+ */
119
+ shipped: boolean
120
+ /**
121
+ * Where the term comes from, one entry per attesting source: `wof:placetype`, `osm:aeroway=terminal`,
122
+ * `wikidata:Q849706`, `overture:airport_terminal`. Sorted, so a regenerate is stable.
123
+ */
124
+ provenance: string[]
125
+ }
126
+
127
+ /**
128
+ * One positional modifier — the `North`/`Upper`/`Main` half of `North Terminal`.
129
+ */
130
+ export interface SubVenueModifier {
131
+ id: string
132
+ shipped: boolean
133
+ provenance: string[]
134
+ }
135
+
136
+ /**
137
+ * One surface form: a phrase, the record it names, and where it was attested.
138
+ *
139
+ * This is the table's `SynonymEntry` — the flat array a phrase index is built over.
140
+ */
141
+ export interface SubVenueSurface {
142
+ /**
143
+ * The phrase, lowercased for Latin-script languages and left as written otherwise (case-folding is meaningless for
144
+ * Japanese, and `toLowerCase` on Turkish `I` is actively wrong).
145
+ */
146
+ phrase: string
147
+ /**
148
+ * The {@link SubVenueDesignator.id} or {@link SubVenueModifier.id} this phrase is a surface of.
149
+ */
150
+ recordID: string
151
+ /**
152
+ * Which record table `recordID` points into.
153
+ */
154
+ recordKind: "designator" | "modifier"
155
+ /**
156
+ * BCP-47-ish language subtag as the source wrote it (`en`, `ja`, `zh-Hant`, `pt-BR`), or `und` when the source gave
157
+ * an untagged default name.
158
+ */
159
+ lang: string
160
+ /**
161
+ * ISO 3166-1 alpha-2 of the DATA the phrase was attested in, `""` for vocabulary sources that attest a term's
162
+ * existence rather than its use anywhere. This is the axis promotion is decided on: `hall` is attested 3,274 times in
163
+ * `GB` and every promotion of it lives or dies on a per-region census, never a global one.
164
+ */
165
+ region: string
166
+ /**
167
+ * `wikidata:label`, `wikidata:alt`, `osm:name`, `osm:name:<lang>`, `overture:name`, `derived:head-noun`, or `seed`.
168
+ */
169
+ source: string
170
+ /**
171
+ * Whether a human has approved this surface for parsing use IN ITS REGION. Everything machine-derived starts `false`
172
+ * and is flipped only by a matching {@link SubVenuePromotion}. A consumer that gates a parse MUST filter on this — see
173
+ * the module docstring.
174
+ */
175
+ curated: boolean
176
+ /**
177
+ * How many source features attested this exact phrase, when the source counts (OSM, Overture). `0` for vocabulary
178
+ * sources, which attest a term's EXISTENCE rather than its frequency.
179
+ */
180
+ observations: number
181
+ /**
182
+ * The rule-assigned designator of the FEATURES that carried this phrase, with a count each — `platform:3205
183
+ * campus:49` for GB's `hall`. Empty for vocabulary sources.
184
+ *
185
+ * This is the confound axis. A `hall` on a `platform` row is a British bus stop named after a village hall; a `hall`
186
+ * on a `terminal` row is a real German departure hall. Without it, a surface's `observations` count is a magnitude
187
+ * with no sign — see the repo's "meaning of zero" rule, which applies just as hard to a large number.
188
+ */
189
+ context: Record<string, number>
190
+ }
191
+
192
+ /**
193
+ * The measured shape of a designator's identifier half — what follows `Gate`/`Terminal` in real data.
194
+ *
195
+ * Derived from OSM `ref` values, not from names. See the module docstring's finding 2.
196
+ */
197
+ export interface IdentifierShape {
198
+ designatorID: string
199
+ /**
200
+ * ISO 3166-1 alpha-2 of the extract this distribution was measured in. Per-region because the shapes differ: GB gates
201
+ * are 70% bare digits, Japanese platform refs are overwhelmingly bare digits with a different range, and a shard that
202
+ * generates `Gate <ref>` for a French address should sample France's distribution.
203
+ */
204
+ region: string
205
+ /**
206
+ * A coarse class: `digit` (`5`), `letter` (`B`), `letter-digit` (`A12`), `digit-letter` (`2F`), `range` (`16-18`,
207
+ * `0/1`), or `other`.
208
+ */
209
+ shape: string
210
+ observations: number
211
+ /**
212
+ * Up to eight real values, sorted, so a shard author can see what the class actually contains.
213
+ */
214
+ examples: string[]
215
+ }
216
+
217
+ /**
218
+ * One input source's provenance, copied off its fetch manifest.
219
+ */
220
+ export interface SubVenueLexiconSource {
221
+ id: string
222
+ origin: string
223
+ license: string
224
+ retrieved: string
225
+ rows: number
226
+ }
227
+
228
+ /**
229
+ * The committed table.
230
+ */
231
+ export interface SubVenueLexiconTable {
232
+ version: string
233
+ sources: SubVenueLexiconSource[]
234
+ designators: SubVenueDesignator[]
235
+ modifiers: SubVenueModifier[]
236
+ surfaces: SubVenueSurface[]
237
+ identifierShapes: IdentifierShape[]
238
+ /**
239
+ * Every curation decision taken against this table, promotion AND rejection, each with the census that backs it. A
240
+ * rejection is as load-bearing as a promotion: it is what stops the next reader re-proposing `hall` for en-GB.
241
+ */
242
+ promotions: SubVenuePromotion[]
243
+ }
244
+
245
+ /**
246
+ * The vocabulary that already ships in `neural/venue-structure.ts`, re-declared. See the module docstring for why this
247
+ * duplication exists.
248
+ *
249
+ * `tier` is added here (the shipped list has no such field): the seven WOF placetypes plus `terminal`/`gate` are all
250
+ * venue-INTERIOR, except `campus` and `building`, which name a whole venue as often as a part of one. They are marked
251
+ * `subvenue` anyway, because that is the role the span proposer uses them in — `Building 43, Googleplex` is a unit
252
+ * inside a venue.
253
+ */
254
+ export const SHIPPED_DESIGNATOR_SEED: ReadonlyArray<{
255
+ id: string
256
+ modifierEligible: boolean
257
+ provenance: string[]
258
+ }> = [
259
+ { id: "arcade", modifierEligible: true, provenance: ["wof:placetype"] },
260
+ { id: "building", modifierEligible: false, provenance: ["wof:placetype"] },
261
+ { id: "campus", modifierEligible: true, provenance: ["wof:placetype"] },
262
+ { id: "concourse", modifierEligible: true, provenance: ["wof:placetype"] },
263
+ { id: "enclosure", modifierEligible: false, provenance: ["wof:placetype"] },
264
+ { id: "gate", modifierEligible: false, provenance: ["osm:aeroway=gate"] },
265
+ { id: "installation", modifierEligible: false, provenance: ["wof:placetype"] },
266
+ { id: "terminal", modifierEligible: true, provenance: ["osm:aeroway=terminal"] },
267
+ { id: "wing", modifierEligible: true, provenance: ["wof:placetype"] },
268
+ ]
269
+
270
+ /**
271
+ * The shipped positional modifiers, re-declared from `neural/venue-structure.ts`'s `VENUE_STRUCTURE_MODIFIERS`.
272
+ */
273
+ export const SHIPPED_MODIFIER_SEED: readonly string[] = [
274
+ "central",
275
+ "east",
276
+ "front",
277
+ "inner",
278
+ "lower",
279
+ "main",
280
+ "north",
281
+ "outer",
282
+ "rear",
283
+ "south",
284
+ "upper",
285
+ "west",
286
+ ]
287
+
288
+ /**
289
+ * Designators the lexicon ADDS beyond what ships, each with the source that attests it.
290
+ *
291
+ * `platform`, `station` and `airport` come from the OSM extractor's rule table and are the rail/aviation venue-side
292
+ * vocabulary the corpus line needs. `hall` and `satellite` come from Wikidata concepts and from
293
+ * `wof-osm-placetype-map.mdx`'s own "plausible additions" note, which lists `hall` explicitly. `pier` joins them in
294
+ * wave 2 on 282 Overture attestations in the `pier` category plus 162 in the GB extract — the corpus task names `Pier
295
+ * C` as a target shape, so the record has to exist before a shard can generate it.
296
+ *
297
+ * None is `modifierEligible`: that claim needs a confound board per term AND per locale, and `sub-venue-promotions.ts`
298
+ * is where those live. A promotion marks a SURFACE usable; it does not widen the modifier grammar.
299
+ */
300
+ export const PROPOSED_DESIGNATORS: ReadonlyArray<{
301
+ id: string
302
+ tier: LexiconTier
303
+ provenance: string[]
304
+ }> = [
305
+ { id: "airport", tier: LexiconTier.Venue, provenance: ["osm:aeroway=aerodrome"] },
306
+ { id: "hall", tier: LexiconTier.SubVenue, provenance: ["wikidata:Q240854"] },
307
+ { id: "pier", tier: LexiconTier.SubVenue, provenance: ["overture:pier"] },
308
+ { id: "platform", tier: LexiconTier.SubVenue, provenance: ["osm:public_transport=platform", "osm:railway=platform"] },
309
+ { id: "satellite", tier: LexiconTier.SubVenue, provenance: ["wikidata:Q15990706"] },
310
+ { id: "station", tier: LexiconTier.Venue, provenance: ["osm:railway=station"] },
311
+ ]
312
+
313
+ /**
314
+ * `designatorID` → Wikidata QID, mirroring `fetch/wikidata-subvenue.ts`'s `SUBVENUE_CONCEPTS`. Re-declared here so the
315
+ * builder stays a pure function over PARSED input rather than reaching into a fetch module for a constant; the test
316
+ * pins the two against each other.
317
+ */
318
+ export const CONCEPT_QIDS: Readonly<Record<string, string>> = {
319
+ terminal: "Q849706",
320
+ gate: "Q247739",
321
+ concourse: "Q862212",
322
+ campus: "Q209465",
323
+ building: "Q41176",
324
+ arcade: "Q186637",
325
+ hall: "Q240854",
326
+ satellite: "Q15990706",
327
+ }
328
+
329
+ /**
330
+ * How many real `ref` values each {@link IdentifierShape} keeps.
331
+ *
332
+ * Eight, not "all" and not one. The field exists so a shard author can see what a class actually CONTAINS — GB's
333
+ * `other` class turned out to be semicolon multi-values (`1;2;3`, `13;14`), which one example would have hidden and
334
+ * which the class name does not say. Eight fits a terminal line and covers the variety inside every class the GB
335
+ * extract produced. The COUNT lives in `observations`; this is a sample, not a census.
336
+ */
337
+ const IDENTIFIER_EXAMPLES_PER_SHAPE = 8
338
+
339
+ /**
340
+ * Scripts whose case is meaningful to fold. Everything else is left as written — see {@link SubVenueSurface.phrase}.
341
+ */
342
+ const CASE_FOLDING_SCRIPT = /^[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\d\s\p{P}]+$/u
343
+
344
+ /**
345
+ * Scripts written without spaces between words, where a token split cannot find a designator and a SUBSTRING test is
346
+ * the correct operator. Han, Hiragana, Katakana; Hangul is excluded because Korean does space its words.
347
+ *
348
+ * The Germanic-compound argument that keeps {@link nameContainsSurface} token-bounded for Latin script does not transfer
349
+ * here — there is no `-gate`/`-hall` street-name suffix class in Japanese, and `第1ターミナル` is unreachable by any token
350
+ * split. Measured on the Japan extract: see the harvest counts in `corpus/data/PROVENANCE.md`.
351
+ */
352
+ const NON_SPACING_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}]/u
353
+
354
+ /**
355
+ * Normalize a surface for the table: trim, collapse internal whitespace, and lowercase ONLY when the string is entirely
356
+ * in a bicameral script. `ターミナルビル` and `航站楼` pass through untouched; `Flughafenterminal` folds.
357
+ */
358
+ export function normalizeSurface(text: string): string {
359
+ const trimmed = text.trim().replaceAll(/\s+/gu, " ")
360
+
361
+ return CASE_FOLDING_SCRIPT.test(trimmed) ? trimmed.toLowerCase() : trimmed
362
+ }
363
+
364
+ /**
365
+ * The SPARQL results envelope, narrowed to the columns the designator-label query produces.
366
+ */
367
+ interface SPARQLBinding {
368
+ item?: { value: string }
369
+ lang?: { value: string }
370
+ label?: { value: string }
371
+ kind?: { value: string }
372
+ }
373
+
374
+ interface SPARQLEnvelope {
375
+ results?: { bindings?: SPARQLBinding[] }
376
+ }
377
+
378
+ /**
379
+ * Turn the Wikidata designator-label payload into surfaces.
380
+ *
381
+ * A row is dropped when its language tag is empty (an untagged literal, which Wikidata occasionally carries), when the
382
+ * QID maps to no designator in {@link CONCEPT_QIDS}, or when the normalized phrase is empty. Everything that survives
383
+ * lands `curated: false` — see the module docstring.
384
+ */
385
+ export function surfacesFromWikidata(
386
+ payload: unknown,
387
+ conceptQIDs: Readonly<Record<string, string>> = CONCEPT_QIDS
388
+ ): SubVenueSurface[] {
389
+ const byQID = new Map(Object.entries(conceptQIDs).map(([id, qid]) => [qid, id]))
390
+ const envelope = payload as SPARQLEnvelope
391
+ const seen = new Set<string>()
392
+ const out: SubVenueSurface[] = []
393
+
394
+ for (const binding of envelope.results?.bindings ?? []) {
395
+ const qid = binding.item?.value?.split("/").pop()
396
+ const recordID = qid ? byQID.get(qid) : undefined
397
+ const lang = binding.lang?.value
398
+ const raw = binding.label?.value
399
+
400
+ if (!recordID || !lang || !raw) continue
401
+
402
+ const phrase = normalizeSurface(raw)
403
+
404
+ if (!phrase) continue
405
+
406
+ const source = binding.kind?.value === "alt" ? "wikidata:alt" : "wikidata:label"
407
+ // A concept can carry the same string as both a label and an alias, and across dialect subtags
408
+ // (`zh`, `zh-cn`, `zh-hans` all say 航站楼). Key the dedupe on the tuple that identifies a row.
409
+ const key = `${phrase}${recordID}${lang}${source}`
410
+
411
+ if (seen.has(key)) continue
412
+ seen.add(key)
413
+
414
+ out.push({
415
+ phrase,
416
+ recordID,
417
+ recordKind: "designator",
418
+ lang,
419
+ region: "",
420
+ source,
421
+ curated: false,
422
+ observations: 0,
423
+ context: {},
424
+ })
425
+ }
426
+
427
+ return out
428
+ }
429
+
430
+ /**
431
+ * One row of a harvestable source, as read back off JSONL or out of a layer database.
432
+ *
433
+ * SOURCE-NEUTRAL by design, and verified so in wave 2: an Overture Places row from the `airport_terminal` category is
434
+ * `{ designatorID, name }` and fits unchanged. What did NOT fit was the harvest function's hardcoded `osm:name` source
435
+ * stamp — see `overture-subvenue.ts`'s docstring. Declared locally so the builder does not import `@mailwoman/osm`
436
+ * (which `@mailwoman/corpus` does not depend on) just to name a shape it reads from a file.
437
+ */
438
+ export interface SubVenueHarvestRow {
439
+ /**
440
+ * The designator the SOURCE's rule assigned to the FEATURE. Not necessarily the record a matched phrase names — see
441
+ * the module docstring's finding 3. Carried into {@link SubVenueSurface.context}.
442
+ */
443
+ designatorID: string
444
+ name?: string | null
445
+ ref?: string | null
446
+ localizedNames?: Record<string, string>
447
+ }
448
+
449
+ /**
450
+ * Classify an OSM `ref` into an {@link IdentifierShape} class.
451
+ *
452
+ * The classes are the ones Berlin's gates actually produced, plus the two aviation forms the corpus task names
453
+ * (`Terminal 2F` is digit-letter, `Concourse B` is letter). `range` covers both separators OSM uses for a gate serving
454
+ * more than one stand: `16-18` and `0/1`.
455
+ */
456
+ export function classifyIdentifier(ref: string): string {
457
+ const value = ref.trim()
458
+
459
+ if (/^[0-9]+$/.test(value)) return "digit"
460
+
461
+ if (/^[A-Za-z]$/.test(value)) return "letter"
462
+
463
+ if (/^[A-Za-z]+[0-9]+$/.test(value)) return "letter-digit"
464
+
465
+ if (/^[0-9]+[A-Za-z]+$/.test(value)) return "digit-letter"
466
+
467
+ if (/^[0-9A-Za-z]+\s*[-/]\s*[0-9A-Za-z]+$/.test(value)) return "range"
468
+
469
+ return "other"
470
+ }
471
+
472
+ /**
473
+ * A phrase → record index, keyed on the normalized phrase. Built by {@link buildSurfaceIndex} from the surfaces present
474
+ * before the harvest runs, and the reason a matched phrase can be attributed to the record it actually names.
475
+ */
476
+ export type SurfaceIndex = ReadonlyMap<string, { recordID: string; recordKind: "designator" | "modifier" }>
477
+
478
+ /**
479
+ * Index the surfaces accumulated so far by phrase. First writer wins, so a seed record beats a Wikidata alias that
480
+ * happens to collide — `terminal` stays the `terminal` designator even though it is also an Italian alias for it.
481
+ */
482
+ export function buildSurfaceIndex(surfaces: readonly SubVenueSurface[]): SurfaceIndex {
483
+ const index = new Map<string, { recordID: string; recordKind: "designator" | "modifier" }>()
484
+
485
+ for (const surface of surfaces) {
486
+ if (index.has(surface.phrase)) continue
487
+ index.set(surface.phrase, { recordID: surface.recordID, recordKind: surface.recordKind })
488
+ }
489
+
490
+ return index
491
+ }
492
+
493
+ /**
494
+ * Every known phrase found in `name`, as whole-token runs for spacing scripts and as substrings for non-spacing ones.
495
+ *
496
+ * Token-boundary matching for Latin script, not substring: `Nordterminal` is a real German compound in which `terminal`
497
+ * is a suffix, and a substring test would also fire on `Terminalstraße`. The compound case is a genuine miss and it is
498
+ * the right miss — admitting suffix matches would fire on every `-hall`/`-gate` compound in Germanic and Nordic street
499
+ * naming, which is exactly the confound class `Briggate`/`Kirkgate` represents.
500
+ *
501
+ * For Han/Kana names that rule finds nothing at all, because the script has no word boundaries: `第1ターミナル` splits into
502
+ * one token that matches no surface. There the LONGEST known substring is the correct operator, and the compound
503
+ * objection does not transfer — Japanese has no `-gate` street-name suffix class.
504
+ */
505
+ export function nameContainsSurfaces(name: string, index: SurfaceIndex): string[] {
506
+ const normalized = normalizeSurface(name)
507
+ const hits = new Set<string>()
508
+
509
+ for (const token of normalized.split(/[\s,()/]+/u)) {
510
+ const stripped = token.replaceAll(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "")
511
+
512
+ if (stripped && index.has(stripped)) {
513
+ hits.add(stripped)
514
+ }
515
+ }
516
+
517
+ if (NON_SPACING_SCRIPT.test(normalized)) {
518
+ for (const [phrase] of index) {
519
+ if (NON_SPACING_SCRIPT.test(phrase) && normalized.includes(phrase)) {
520
+ hits.add(phrase)
521
+ }
522
+ }
523
+ }
524
+
525
+ return [...hits]
526
+ }
527
+
528
+ /**
529
+ * Options for one harvest pass — which source stamp its surfaces carry and which region they were attested in.
530
+ *
531
+ * Both default to the OSM/unknown-region values wave 1 hardcoded, so an existing caller is unchanged.
532
+ */
533
+ export interface HarvestOptions {
534
+ /**
535
+ * Source family: `osm` yields `osm:name` / `osm:name:<lang>`, `overture` yields `overture:name`.
536
+ */
537
+ source?: string
538
+ /**
539
+ * ISO 3166-1 alpha-2 of the extract or partition. `""` when unknown.
540
+ */
541
+ region?: string
542
+ }
543
+
544
+ /**
545
+ * Harvest attested phrases and identifier shapes out of a source's rows.
546
+ *
547
+ * `index` gates the name harvest — a name contributes only when it CONTAINS a phrase already in the table. That filter
548
+ * is the whole reason this function is safe to run over raw OSM: see the module docstring's finding 1 for the Berlin
549
+ * measurement that motivated it. The index also decides ATTRIBUTION (finding 3): a hit is a surface of the record the
550
+ * PHRASE names, and the row's own designator is recorded as `context`.
551
+ *
552
+ * Returns surfaces with real `observations` counts, so the lexicon can rank `terminal` above a phrase attested once.
553
+ */
554
+ export function extractAttestedPhrases(
555
+ rows: Iterable<SubVenueHarvestRow>,
556
+ index: SurfaceIndex,
557
+ options: HarvestOptions = {}
558
+ ): { surfaces: SubVenueSurface[]; identifierShapes: IdentifierShape[] } {
559
+ const source = options.source ?? "osm"
560
+ const region = options.region ?? ""
561
+ /**
562
+ * `phrase\0lang\0source` → { count, context }.
563
+ */
564
+ const surfaceCounts = new Map<string, { count: number; context: Map<string, number> }>()
565
+ /**
566
+ * `designatorID\0shape` → { count, examples }.
567
+ */
568
+ const shapes = new Map<string, { count: number; examples: Set<string> }>()
569
+
570
+ const note = (phrase: string, lang: string, sourceTag: string, context: string): void => {
571
+ const key = `${phrase}${lang}${sourceTag}`
572
+ const entry = surfaceCounts.get(key) ?? { count: 0, context: new Map<string, number>() }
573
+
574
+ entry.count++
575
+ entry.context.set(context, (entry.context.get(context) ?? 0) + 1)
576
+ surfaceCounts.set(key, entry)
577
+ }
578
+
579
+ for (const row of rows) {
580
+ if (row.name) {
581
+ for (const hit of nameContainsSurfaces(row.name, index)) {
582
+ // `und` — the default `name` tag carries no language. Overture's `name` is the same: a
583
+ // primary name in whatever language the place uses, untagged.
584
+ note(hit, "und", `${source}:name`, row.designatorID)
585
+ }
586
+ }
587
+
588
+ for (const [lang, localized] of Object.entries(row.localizedNames ?? {})) {
589
+ for (const hit of nameContainsSurfaces(localized, index)) {
590
+ note(hit, lang, `${source}:name:${lang}`, row.designatorID)
591
+ }
592
+ }
593
+
594
+ if (row.ref) {
595
+ const shape = classifyIdentifier(row.ref)
596
+ const key = `${row.designatorID}${shape}`
597
+ const entry = shapes.get(key) ?? { count: 0, examples: new Set<string>() }
598
+
599
+ entry.count++
600
+
601
+ if (entry.examples.size < IDENTIFIER_EXAMPLES_PER_SHAPE) {
602
+ entry.examples.add(row.ref.trim())
603
+ }
604
+
605
+ shapes.set(key, entry)
606
+ }
607
+ }
608
+
609
+ const surfaces: SubVenueSurface[] = [...surfaceCounts].map(([key, entry]) => {
610
+ const [phrase, lang, sourceTag] = key.split("") as [string, string, string]
611
+ const record = index.get(phrase)!
612
+
613
+ return {
614
+ phrase,
615
+ recordID: record.recordID,
616
+ recordKind: record.recordKind,
617
+ lang,
618
+ region,
619
+ source: sourceTag,
620
+ curated: false,
621
+ observations: entry.count,
622
+ context: Object.fromEntries([...entry.context].toSorted((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))),
623
+ }
624
+ })
625
+
626
+ const identifierShapes: IdentifierShape[] = [...shapes].map(([key, entry]) => {
627
+ const [designatorID, shape] = key.split("") as [string, string]
628
+
629
+ return {
630
+ designatorID,
631
+ region,
632
+ shape,
633
+ observations: entry.count,
634
+ examples: [...entry.examples].toSorted((a, b) => a.localeCompare(b)),
635
+ }
636
+ })
637
+
638
+ return { surfaces, identifierShapes }
639
+ }
640
+
641
+ /**
642
+ * Diacritic-flattened ASCII fold, for comparing a Slavic or Turkish inflection against its Latin root.
643
+ */
644
+ function asciiFold(text: string): string {
645
+ return text
646
+ .normalize("NFD")
647
+ .replaceAll(/\p{Diacritic}/gu, "")
648
+ .toLowerCase()
649
+ }
650
+
651
+ /**
652
+ * How many leading characters two ASCII-folded forms must share for one to count as the other's inflection.
653
+ *
654
+ * Five, or the id's own length when that is shorter (`hall`, `gate`, `wing`, `pier` are four). Measured against the
655
+ * committed Wikidata pull: at five, `terminal`/`terminál`/`terminale`/`terminali`/`terminála`/`terminalo` are all
656
+ * accepted for `terminal` while `campo` and `campws` are both rejected for `campus` (they share four). At six the
657
+ * Spanish `satélite` is lost; at four, Italian `campo` is admitted and it means FIELD.
658
+ */
659
+ const HEAD_NOUN_PREFIX_FLOOR = 5
660
+
661
+ /**
662
+ * The shortest substring a non-Latin head-noun candidate may be. Two: `航站` and `터미널` are both real, `楼` alone is
663
+ * "building" and would fire on every Chinese building name.
664
+ */
665
+ const NON_LATIN_HEAD_MIN_LENGTH = 2
666
+
667
+ /**
668
+ * How many head-noun candidates one non-Latin record+language group may contribute. Six — enough to carry `ターミナル`,
669
+ * `ターミナルビル` and `旅客ターミナル` together, capped because the substring lattice of a nine-character label is large and, ranked
670
+ * by attesting-surface count, nothing past the sixth has more than the minimum two.
671
+ */
672
+ const NON_LATIN_HEAD_CANDIDATE_CAP = 6
673
+
674
+ /**
675
+ * Latin-script test — the scripts an ASCII-folded prefix comparison against a Latin designator id can work on.
676
+ */
677
+ const LATIN_PHRASE = /^[\p{Script=Latin}\d\s\p{P}]+$/u
678
+
679
+ /**
680
+ * The scripts the shared-substring derivation is allowed to run on: Han, Hiragana, Katakana, Hangul.
681
+ *
682
+ * NARROWER than "not Latin", and the narrowing was earned. Run over every non-Latin phrase in the table, the derivation
683
+ * produced 90 fragments of Cyrillic, Greek, Arabic, Thai, Burmese and Tamil words — `сгра`, `град`, `κτίρ`,
684
+ * `ิ่งก่อสร้า` — because those languages have exactly one surface per concept and the only substrings shared inside a
685
+ * group are pieces of one word. Every one of them was unusable, and none could ever be counted: `poi.db` is four
686
+ * countries and this wave's extracts are GB, DE, FR, ES and JP, so nothing in reach attests a Thai or Burmese surface.
687
+ * Deriving a candidate no available source can confirm is not a hypothesis, it is table weight.
688
+ */
689
+ const SHARED_SUBSTRING_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u
690
+
691
+ /**
692
+ * Derive the HEAD NOUN of every multi-part surface, so `terminal aeroportuaria` contributes the form anyone actually
693
+ * writes on an envelope.
694
+ *
695
+ * The problem this solves is the whole reason wave 1 shipped 1,014 uncurated surfaces: Wikidata's label for a concept
696
+ * is the ENCYCLOPAEDIC name (`terminal aeroportuaria`, `letištní terminál`, `havalimanı terminali`), while the
697
+ * addressed form is the bare head (`Terminal`, `Terminál`, `Terminali`). Nothing can promote the encyclopaedic form, so
698
+ * the head has to be extracted before the curation pass has anything to decide about.
699
+ *
700
+ * Two derivations, because the table holds two kinds of writing:
701
+ *
702
+ * - **Latin script — the COGNATE test.** A token is the head when its ASCII fold shares {@link HEAD_NOUN_PREFIX_FLOOR}
703
+ * leading characters with the designator's own canonical id. Nothing subtler survived contact with the data: an
704
+ * earlier version matched a token against any SINGLE-TOKEN surface of the record, and because Dutch `universiteit` is
705
+ * a one-token surface of `campus`, it derived `universitario`, `universitaire`, `üniversite` and twenty more as head
706
+ * nouns of `campus`. Those are the MODIFIER half of the label, and admitting them would have taught the harvest to
707
+ * read "Ciudad Universitaria" as sub-venue structure.
708
+ * - **Non-Latin script — the SHARED-SUBSTRING test.** The cognate test cannot reach a script the id is not written in,
709
+ * and for Han and Kana a token split finds nothing at all. So every substring of length ≥
710
+ * {@link NON_LATIN_HEAD_MIN_LENGTH} occurring in at least two DISTINCT surfaces of the same record and primary
711
+ * language becomes a candidate, ranked by how many surfaces carry it. Japanese yields `ターミナル` (in all five `ja`
712
+ * terminal labels) ahead of `ターミナルビル` (three); Chinese yields `航站`, `航站楼`, `航站樓`. Where the script DOES space its
713
+ * words (Korean, Greek, Cyrillic) a candidate must be a whole token, so `공항 터미널` ∩ `공항터미널` gives `터미널` and never a
714
+ * fragment.
715
+ *
716
+ * The non-Latin branch deliberately emits SEVERAL candidates instead of picking one. Choosing between `航站` and `航站楼`
717
+ * from Wikidata alone is guesswork; the Japan extract answers it by counting, and the promotion ledger records which
718
+ * count won. Everything derived lands `curated: false` — the derivation is a hypothesis about what the addressed form
719
+ * is, and a locale's own data is what confirms or kills it.
720
+ */
721
+ export function deriveHeadNounSurfaces(surfaces: readonly SubVenueSurface[]): SubVenueSurface[] {
722
+ const derived = new Map<string, SubVenueSurface>()
723
+ const seen = new Set(surfaces.map((s) => `${s.phrase}${s.recordID}${s.lang}`))
724
+
725
+ const emit = (phrase: string, from: SubVenueSurface): void => {
726
+ if (phrase === from.phrase) return
727
+
728
+ const key = `${phrase}${from.recordID}${from.lang}`
729
+
730
+ if (seen.has(key) || derived.has(key)) return
731
+
732
+ derived.set(key, {
733
+ phrase,
734
+ recordID: from.recordID,
735
+ recordKind: from.recordKind,
736
+ lang: from.lang,
737
+ region: "",
738
+ source: "derived:head-noun",
739
+ curated: false,
740
+ observations: 0,
741
+ context: {},
742
+ })
743
+ }
744
+
745
+ // ── Spacing scripts: prefix-match a token against a single-token surface of the same record ──────
746
+ // Latin script: a token that is a cognate of the designator's own canonical id.
747
+ for (const surface of surfaces) {
748
+ if (!LATIN_PHRASE.test(surface.phrase)) continue
749
+
750
+ const parts = surface.phrase.split(/[^\p{L}\p{N}]+/u).filter(Boolean)
751
+
752
+ if (parts.length < 2) continue
753
+
754
+ const root = asciiFold(surface.recordID)
755
+ const floor = Math.min(HEAD_NOUN_PREFIX_FLOOR, root.length)
756
+
757
+ for (const part of parts) {
758
+ const folded = asciiFold(part)
759
+
760
+ if (folded.length >= floor && commonPrefixLength(folded, root) >= floor) {
761
+ emit(part, surface)
762
+ }
763
+ }
764
+ }
765
+
766
+ // Non-Latin script: substrings shared by two or more surfaces of the same record + language.
767
+ const groups = new Map<string, Set<string>>()
768
+
769
+ for (const surface of surfaces) {
770
+ if (!SHARED_SUBSTRING_SCRIPT.test(surface.phrase)) continue
771
+
772
+ // Group `zh`, `zh-cn`, `zh-hant` together: they are writing systems for one vocabulary, and the
773
+ // simplified/traditional pair is exactly the evidence a shared substring needs.
774
+ const key = `${surface.recordID} ${surface.lang.split(/[-_]/u)[0]!}`
775
+ const pool = groups.get(key) ?? new Set<string>()
776
+ pool.add(surface.phrase)
777
+ groups.set(key, pool)
778
+ }
779
+
780
+ const candidatesByGroup = new Map<string, string[]>()
781
+
782
+ for (const [key, pool] of groups) {
783
+ if (pool.size < 2) continue
784
+ candidatesByGroup.set(key, sharedSubstringCandidates(pool))
785
+ }
786
+
787
+ for (const surface of surfaces) {
788
+ if (!SHARED_SUBSTRING_SCRIPT.test(surface.phrase)) continue
789
+
790
+ const key = `${surface.recordID} ${surface.lang.split(/[-_]/u)[0]!}`
791
+
792
+ for (const candidate of candidatesByGroup.get(key) ?? []) {
793
+ if (surface.phrase.includes(candidate)) {
794
+ emit(candidate, surface)
795
+ }
796
+ }
797
+ }
798
+
799
+ return [...derived.values()]
800
+ }
801
+
802
+ /**
803
+ * Length of the shared leading run of two strings.
804
+ */
805
+ function commonPrefixLength(a: string, b: string): number {
806
+ const limit = Math.min(a.length, b.length)
807
+ let i = 0
808
+
809
+ while (i < limit && a[i] === b[i]) {
810
+ i++
811
+ }
812
+
813
+ return i
814
+ }
815
+
816
+ /**
817
+ * Substrings occurring in at least two DISTINCT members of `pool`, ranked by that count and then by length, capped at
818
+ * {@link NON_LATIN_HEAD_CANDIDATE_CAP}.
819
+ *
820
+ * A candidate never spans whitespace, and in a pool whose members contain whitespace a candidate must be a whole token
821
+ * of some member. That is what keeps Korean `공항 터미널` from contributing a fragment straddling the space.
822
+ *
823
+ * MAXIMAL candidates only: one contained in a longer candidate carried by the SAME number of surfaces is dropped, since
824
+ * counting can never separate the two. Every one of `ターミナル`'s five ja labels also contains `ターミ`, `ターミナ` and `ミナル`, so
825
+ * without this the group contributes four indistinguishable candidates and the Japan harvest returns four identical
826
+ * counts. `航站` survives next to `航站楼` because six surfaces carry it against that one's two.
827
+ */
828
+ function sharedSubstringCandidates(pool: ReadonlySet<string>): string[] {
829
+ const phrases = [...pool]
830
+ const spaced = phrases.some((phrase) => /\s/u.test(phrase))
831
+ const tokens = spaced ? new Set(phrases.flatMap((phrase) => phrase.split(/\s+/u).filter(Boolean))) : null
832
+ const counts = new Map<string, number>()
833
+
834
+ for (const phrase of phrases) {
835
+ const local = new Set<string>()
836
+
837
+ for (let length = NON_LATIN_HEAD_MIN_LENGTH; length <= phrase.length; length++) {
838
+ for (let start = 0; start + length <= phrase.length; start++) {
839
+ const candidate = phrase.slice(start, start + length)
840
+
841
+ if (/\s/u.test(candidate)) continue
842
+ local.add(candidate)
843
+ }
844
+ }
845
+
846
+ for (const candidate of local) {
847
+ counts.set(candidate, (counts.get(candidate) ?? 0) + 1)
848
+ }
849
+ }
850
+
851
+ const kept = [...counts].filter(([candidate, count]) => count >= 2 && (!tokens || tokens.has(candidate)))
852
+
853
+ return kept
854
+ .filter(([candidate, count]) =>
855
+ kept.every(([other, otherCount]) => other === candidate || otherCount !== count || !other.includes(candidate))
856
+ )
857
+ .toSorted((a, b) => b[1] - a[1] || b[0].length - a[0].length || a[0].localeCompare(b[0]))
858
+ .slice(0, NON_LATIN_HEAD_CANDIDATE_CAP)
859
+ .map(([candidate]) => candidate)
860
+ }
861
+
862
+ /**
863
+ * Apply the curation decisions to a surface list, IN PLACE on a copy.
864
+ *
865
+ * A promotion binds `(designatorID, phrase, locale)`. A surface matches when it names the same record with the same
866
+ * phrase and its language is the locale's language OR the untagged `und` — the default `name` tag carries no language,
867
+ * and a German extract's untagged `Halle 2` is German.
868
+ *
869
+ * REGION is the subtle half. A surface attested in an extract carries that extract's region and matches only its own
870
+ * locale. A surface with `region: ""` is region-FREE — a Wikidata label or a derived head noun — and a promotion
871
+ * reaches it only when no REJECTION exists for the same designator, phrase and language anywhere else. That guard is
872
+ * not decoration: `pier` is promoted for en-GB and rejected for en-US, and without it the en-GB decision would curate
873
+ * the region-free English surface and hand `Pier 1 Imports` the promotion en-US was refused. Where no rejection
874
+ * competes — `terminal` in `es`, `ターミナル` in `ja` — the region-free surface is the whole point, since a language's
875
+ * designator does not stop at a border.
876
+ *
877
+ * Rejections mark nothing themselves. They exist in `promotions[]` as the record of a decision taken, so the next
878
+ * reader meets en-GB `hall`'s 3,204 bus stops before re-proposing it, not after.
879
+ */
880
+ export function applyPromotions(
881
+ surfaces: readonly SubVenueSurface[],
882
+ promotions: readonly SubVenuePromotion[]
883
+ ): SubVenueSurface[] {
884
+ const language = (locale: string): string => locale.split("-")[0]!
885
+ const region = (locale: string): string => locale.split("-")[1] ?? ""
886
+
887
+ const rejectedLanguages = new Set(
888
+ promotions
889
+ .filter((promotion) => promotion.decision === "reject")
890
+ .map((promotion) => `${promotion.designatorID} ${promotion.phrase} ${language(promotion.locale)}`)
891
+ )
892
+
893
+ const promoted = promotions.filter((promotion) => promotion.decision === "promote")
894
+
895
+ return surfaces.map((surface) => {
896
+ if (surface.curated) return surface
897
+
898
+ const hit = promoted.some((promotion) => {
899
+ if (promotion.designatorID !== surface.recordID || promotion.phrase !== surface.phrase) return false
900
+
901
+ const lang = language(promotion.locale)
902
+
903
+ if (surface.lang !== lang && surface.lang !== "und") return false
904
+
905
+ if (surface.region === region(promotion.locale)) return true
906
+
907
+ return surface.region === "" && !rejectedLanguages.has(`${promotion.designatorID} ${promotion.phrase} ${lang}`)
908
+ })
909
+
910
+ return hit ? { ...surface, curated: true } : surface
911
+ })
912
+ }
913
+
914
+ /**
915
+ * One harvestable input: rows plus the stamp they carry into the table.
916
+ */
917
+ export interface SubVenueHarvest {
918
+ rows: readonly SubVenueHarvestRow[]
919
+ source?: string
920
+ region?: string
921
+ }
922
+
923
+ /**
924
+ * Everything {@link buildSubVenueLexicon} needs, already parsed. Keeping the builder off the filesystem is what makes it
925
+ * deterministic and testable without fixtures on disk.
926
+ */
927
+ export interface BuildSubVenueLexiconInput {
928
+ /**
929
+ * The raw `designator-labels.json` SPARQL envelope, or `null` to build the seed-only table.
930
+ */
931
+ wikidata: unknown | null
932
+ /**
933
+ * Every harvestable source, in the order they should contribute. Order matters only for the surface INDEX: a source
934
+ * can match a phrase an earlier source introduced, never a later one.
935
+ */
936
+ harvests: readonly SubVenueHarvest[]
937
+ /**
938
+ * Provenance rows, copied off the fetch manifests by the caller.
939
+ */
940
+ sources: readonly SubVenueLexiconSource[]
941
+ /**
942
+ * Curation decisions. Defaults to the committed {@link SUBVENUE_PROMOTIONS}; pass an empty array to build the
943
+ * pre-curation table (which is what the promotion census itself is taken against).
944
+ */
945
+ promotions?: readonly SubVenuePromotion[]
946
+ }
947
+
948
+ /**
949
+ * Build the lexicon table. PURE and deterministic — same inputs, byte-identical output.
950
+ *
951
+ * Order of operations is load-bearing in three places:
952
+ *
953
+ * 1. Seed surfaces are inserted before anything else, so `terminal` indexes to the `terminal` designator rather than to
954
+ * whichever Wikidata alias sorts first.
955
+ * 2. Head nouns are derived AFTER Wikidata and BEFORE the harvests, because `ターミナル` has to exist as a surface before a
956
+ * Japanese extract can be searched for it. That ordering is the entire reason the Japan harvest finds anything — see
957
+ * `PROVENANCE.md`.
958
+ * 3. Promotions are applied LAST, over the union, so a decision can promote a surface whichever source produced it.
959
+ */
960
+ export function buildSubVenueLexicon(input: BuildSubVenueLexiconInput): SubVenueLexiconTable {
961
+ const designators: SubVenueDesignator[] = SHIPPED_DESIGNATOR_SEED.map((seed) => ({
962
+ id: seed.id,
963
+ tier: LexiconTier.SubVenue,
964
+ modifierEligible: seed.modifierEligible,
965
+ shipped: true,
966
+ provenance: [...seed.provenance],
967
+ }))
968
+
969
+ const byID = new Map(designators.map((d) => [d.id, d]))
970
+
971
+ for (const proposed of PROPOSED_DESIGNATORS) {
972
+ const existing = byID.get(proposed.id)
973
+
974
+ if (existing) {
975
+ existing.provenance = [...new Set([...existing.provenance, ...proposed.provenance])]
976
+
977
+ continue
978
+ }
979
+
980
+ const record: SubVenueDesignator = {
981
+ id: proposed.id,
982
+ tier: proposed.tier,
983
+ modifierEligible: false,
984
+ shipped: false,
985
+ provenance: [...proposed.provenance],
986
+ }
987
+
988
+ designators.push(record)
989
+ byID.set(record.id, record)
990
+ }
991
+
992
+ // A Wikidata concept id is provenance for the designator it names, whether or not the concept
993
+ // contributed a usable surface.
994
+ for (const [id, qid] of Object.entries(CONCEPT_QIDS)) {
995
+ const record = byID.get(id)
996
+
997
+ if (record) {
998
+ record.provenance = [...new Set([...record.provenance, `wikidata:${qid}`])]
999
+ }
1000
+ }
1001
+
1002
+ const modifiers: SubVenueModifier[] = SHIPPED_MODIFIER_SEED.map((id) => ({
1003
+ id,
1004
+ shipped: true,
1005
+ provenance: ["codex:directionals"],
1006
+ }))
1007
+
1008
+ const surfaces: SubVenueSurface[] = [
1009
+ ...designators.map(
1010
+ (d): SubVenueSurface => ({
1011
+ phrase: d.id,
1012
+ recordID: d.id,
1013
+ recordKind: "designator",
1014
+ lang: "en",
1015
+ region: "",
1016
+ source: "seed",
1017
+ // The English designator IS the shipped vocabulary — curated by construction.
1018
+ curated: d.shipped,
1019
+ observations: 0,
1020
+ context: {},
1021
+ })
1022
+ ),
1023
+ ...modifiers.map(
1024
+ (m): SubVenueSurface => ({
1025
+ phrase: m.id,
1026
+ recordID: m.id,
1027
+ recordKind: "modifier",
1028
+ lang: "en",
1029
+ region: "",
1030
+ source: "seed",
1031
+ curated: true,
1032
+ observations: 0,
1033
+ context: {},
1034
+ })
1035
+ ),
1036
+ ]
1037
+
1038
+ if (input.wikidata) {
1039
+ surfaces.push(...surfacesFromWikidata(input.wikidata))
1040
+ }
1041
+
1042
+ surfaces.push(...deriveHeadNounSurfaces(surfaces))
1043
+
1044
+ const identifierShapes: IdentifierShape[] = []
1045
+
1046
+ for (const harvest of input.harvests) {
1047
+ const attested = extractAttestedPhrases(harvest.rows, buildSurfaceIndex(surfaces), {
1048
+ source: harvest.source,
1049
+ region: harvest.region,
1050
+ })
1051
+
1052
+ surfaces.push(...attested.surfaces)
1053
+ identifierShapes.push(...attested.identifierShapes)
1054
+ }
1055
+
1056
+ const promotions = [...(input.promotions ?? SUBVENUE_PROMOTIONS)]
1057
+ const curated = applyPromotions(surfaces, promotions)
1058
+
1059
+ // Deterministic order everywhere. `localeCompare` matches the tie-break discipline
1060
+ // `generate-taxonomy.ts` and `build-brands.ts` already use.
1061
+ designators.sort((a, b) => a.id.localeCompare(b.id))
1062
+ modifiers.sort((a, b) => a.id.localeCompare(b.id))
1063
+
1064
+ curated.sort(
1065
+ (a, b) =>
1066
+ a.phrase.localeCompare(b.phrase) ||
1067
+ a.recordID.localeCompare(b.recordID) ||
1068
+ a.lang.localeCompare(b.lang) ||
1069
+ a.region.localeCompare(b.region) ||
1070
+ a.source.localeCompare(b.source)
1071
+ )
1072
+
1073
+ return {
1074
+ version: SUBVENUE_LEXICON_VERSION,
1075
+ sources: input.sources.toSorted((a, b) => a.id.localeCompare(b.id)),
1076
+ designators,
1077
+ modifiers,
1078
+ surfaces: curated,
1079
+ identifierShapes: identifierShapes.toSorted(
1080
+ (a, b) =>
1081
+ a.designatorID.localeCompare(b.designatorID) ||
1082
+ a.region.localeCompare(b.region) ||
1083
+ a.shape.localeCompare(b.shape)
1084
+ ),
1085
+ promotions: promotions.toSorted(
1086
+ (a, b) =>
1087
+ a.designatorID.localeCompare(b.designatorID) ||
1088
+ a.locale.localeCompare(b.locale) ||
1089
+ a.phrase.localeCompare(b.phrase)
1090
+ ),
1091
+ }
1092
+ }
1093
+
1094
+ /**
1095
+ * Serialize the table the way the committed artifact stores it: pretty-printed, trailing newline. Run `oxfmt` over the
1096
+ * result before committing — repo law is that committed JSON is oxfmt-clean, which `JSON.stringify` cannot reproduce.
1097
+ */
1098
+ export function serializeSubVenueLexicon(table: SubVenueLexiconTable): string {
1099
+ return JSON.stringify(table, null, 2) + "\n"
1100
+ }
1101
+
1102
+ /**
1103
+ * Read a JSONL file of {@link SubVenueHarvestRow}s. Blank lines and unparseable rows are skipped rather than fatal — an
1104
+ * extract is a build output, and one malformed line should not cost the whole lexicon.
1105
+ */
1106
+ export function readSubVenueJSONL(path: string): SubVenueHarvestRow[] {
1107
+ const out: SubVenueHarvestRow[] = []
1108
+
1109
+ // `TextSpliterator` rather than `split("\n")` — a whole-country extract runs to 250,000 lines
1110
+ // (52 MB for Great Britain), and materializing every segment before reading the first is exactly
1111
+ // what the repo lint rule exists to prevent.
1112
+ for (const line of TextSpliterator.from(readFileSync(path, "utf8"))) {
1113
+ const trimmed = line.trim()
1114
+
1115
+ if (!trimmed) continue
1116
+
1117
+ try {
1118
+ out.push(parseJSONStrict<SubVenueHarvestRow>(trimmed))
1119
+ } catch {
1120
+ continue
1121
+ }
1122
+ }
1123
+
1124
+ return out
1125
+ }
1126
+
1127
+ /**
1128
+ * The Wikidata fetch manifest's shape, narrowed to the fields the lexicon copies into `sources[]`.
1129
+ */
1130
+ interface WikidataFetchManifest {
1131
+ endpoint?: string
1132
+ license?: string
1133
+ downloaded_at?: string
1134
+ files?: Array<{ filename?: string; rows?: number }>
1135
+ }
1136
+
1137
+ /**
1138
+ * One OSM extract to harvest: the JSONL path plus the ISO country its rows describe.
1139
+ */
1140
+ export interface SubVenueExtractInput {
1141
+ path: string
1142
+ region: string
1143
+ }
1144
+
1145
+ export interface GenerateSubVenueLexiconOptions {
1146
+ /**
1147
+ * Directory holding the `mailwoman corpus fetch wikidata-subvenue` output. Omit to build the seed-only table.
1148
+ */
1149
+ wikidataDir?: string
1150
+ /**
1151
+ * OSM extract JSONLs, one per region.
1152
+ */
1153
+ extracts?: readonly SubVenueExtractInput[]
1154
+ /**
1155
+ * Already-read Overture rows (`readOvertureSubVenues`), grouped by the caller. Kept as a parameter rather than a path
1156
+ * so this function stays free of a 3.9 GB database dependency — the CLI opens `poi.db`, this assembles.
1157
+ */
1158
+ overtureRows?: readonly (SubVenueHarvestRow & { country: string })[]
1159
+ /**
1160
+ * `poi.db`'s layer vintage, for `sources[]`. Only read when `overtureRows` is non-empty.
1161
+ */
1162
+ overtureVintage?: string
1163
+ /**
1164
+ * Where the table is written.
1165
+ */
1166
+ outPath: string
1167
+ }
1168
+
1169
+ /**
1170
+ * Read the fetch outputs, build the table, and write it.
1171
+ *
1172
+ * The IO half only — every decision lives in {@link buildSubVenueLexicon}, which is pure. Run `oxfmt` over `outPath`
1173
+ * afterwards; repo law is that committed JSON is oxfmt-clean.
1174
+ */
1175
+ export function generateSubVenueLexicon(options: GenerateSubVenueLexiconOptions): SubVenueLexiconTable {
1176
+ const sources: SubVenueLexiconSource[] = []
1177
+ const harvests: SubVenueHarvest[] = []
1178
+ let wikidata: unknown = null
1179
+
1180
+ if (options.wikidataDir) {
1181
+ wikidata = parseJSONStrict<unknown>(readFileSync(join(options.wikidataDir, "designator-labels.json"), "utf8"))
1182
+
1183
+ const manifest = parseJSONStrict<WikidataFetchManifest>(
1184
+ readFileSync(join(options.wikidataDir, "MANIFEST.json"), "utf8")
1185
+ )
1186
+
1187
+ const labelFile = manifest.files?.find((f) => f.filename === "designator-labels.json")
1188
+
1189
+ sources.push({
1190
+ id: "wikidata",
1191
+ origin: manifest.endpoint ?? "https://query.wikidata.org/sparql",
1192
+ license: manifest.license ?? "CC0",
1193
+ // The DATE only. A full ISO timestamp would make every re-fetch a diff in the committed artifact for
1194
+ // no information a reader of a vocabulary table can act on.
1195
+ retrieved: (manifest.downloaded_at ?? "").slice(0, 10),
1196
+ rows: labelFile?.rows ?? 0,
1197
+ })
1198
+ }
1199
+
1200
+ for (const extract of options.extracts ?? []) {
1201
+ const rows = readSubVenueJSONL(extract.path)
1202
+
1203
+ harvests.push({ rows, source: "osm", region: extract.region })
1204
+
1205
+ sources.push({
1206
+ id: `osm:${extract.region.toLowerCase()}`,
1207
+ // The extract's NAME, never its path. `AGENTS.md` forbids re-hardcoding the lab data root
1208
+ // anywhere, and a committed artifact carrying `/mnt/playpen/...` would do exactly that while
1209
+ // telling a reader on another machine nothing. `great-britain` identifies the Geofabrik region,
1210
+ // which is the fact that matters.
1211
+ origin: `OpenStreetMap via Geofabrik (${basename(extract.path, ".jsonl")})`,
1212
+ license: "ODbL (OpenStreetMap)",
1213
+ // The extract's mtime — when the rows were produced. `corpus/AGENTS.md`'s standing warning that
1214
+ // a file's mtime is not its DATA's vintage applies to a downloaded archive; this file is a build
1215
+ // output of ours, so its mtime is exactly the right number.
1216
+ retrieved: statSync(extract.path).mtime.toISOString().slice(0, 10),
1217
+ rows: rows.length,
1218
+ })
1219
+ }
1220
+
1221
+ if (options.overtureRows?.length) {
1222
+ // Overture rows carry their own country, so they are harvested per REGION rather than in one pass —
1223
+ // a `region` on the surface is the axis promotion is decided on and a mixed-country bucket would
1224
+ // make it meaningless.
1225
+ const byCountry = new Map<string, SubVenueHarvestRow[]>()
1226
+
1227
+ for (const row of options.overtureRows) {
1228
+ const bucket = byCountry.get(row.country) ?? []
1229
+ bucket.push(row)
1230
+ byCountry.set(row.country, bucket)
1231
+ }
1232
+
1233
+ for (const [country, rows] of [...byCountry].toSorted((a, b) => a[0].localeCompare(b[0]))) {
1234
+ harvests.push({ rows, source: "overture", region: country })
1235
+ }
1236
+
1237
+ sources.push({
1238
+ id: "overture",
1239
+ origin: "Overture Maps Foundation places, via the poi.db spatial layer",
1240
+ license: "CDLA-Permissive-2.0",
1241
+ retrieved: options.overtureVintage ?? "",
1242
+ rows: options.overtureRows.length,
1243
+ })
1244
+ }
1245
+
1246
+ const table = buildSubVenueLexicon({ wikidata, harvests, sources })
1247
+ writeFileSync(options.outPath, serializeSubVenueLexicon(table))
1248
+
1249
+ return table
1250
+ }