@mailwoman/corpus 8.5.0 → 9.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (310) hide show
  1. package/README.md +1 -1
  2. package/out/src/adapter.d.ts +41 -0
  3. package/out/src/adapter.d.ts.map +1 -1
  4. package/out/src/adapter.js +46 -1
  5. package/out/src/adapter.js.map +1 -1
  6. package/out/src/adapters/ban/adapter.d.ts +4 -3
  7. package/out/src/adapters/ban/adapter.d.ts.map +1 -1
  8. package/out/src/adapters/ban/adapter.js +60 -67
  9. package/out/src/adapters/ban/adapter.js.map +1 -1
  10. package/out/src/adapters/ban/street-decompose.d.ts.map +1 -1
  11. package/out/src/adapters/ban/street-decompose.js +18 -25
  12. package/out/src/adapters/ban/street-decompose.js.map +1 -1
  13. package/out/src/adapters/fcc-bdc/adapter.d.ts +0 -6
  14. package/out/src/adapters/fcc-bdc/adapter.d.ts.map +1 -1
  15. package/out/src/adapters/fcc-bdc/adapter.js +2 -24
  16. package/out/src/adapters/fcc-bdc/adapter.js.map +1 -1
  17. package/out/src/adapters/geonames/adapter.d.ts.map +1 -1
  18. package/out/src/adapters/geonames/adapter.js +70 -76
  19. package/out/src/adapters/geonames/adapter.js.map +1 -1
  20. package/out/src/adapters/geonames-postal/adapter.d.ts.map +1 -1
  21. package/out/src/adapters/geonames-postal/adapter.js +44 -49
  22. package/out/src/adapters/geonames-postal/adapter.js.map +1 -1
  23. package/out/src/adapters/gnaf/adapter.d.ts.map +1 -1
  24. package/out/src/adapters/gnaf/adapter.js +4 -7
  25. package/out/src/adapters/gnaf/adapter.js.map +1 -1
  26. package/out/src/adapters/gnaf/assemble.d.ts.map +1 -1
  27. package/out/src/adapters/gnaf/assemble.js +6 -11
  28. package/out/src/adapters/gnaf/assemble.js.map +1 -1
  29. package/out/src/adapters/openaddresses/adapter.d.ts.map +1 -1
  30. package/out/src/adapters/openaddresses/adapter.js +2 -7
  31. package/out/src/adapters/openaddresses/adapter.js.map +1 -1
  32. package/out/src/adapters/overture/adapter.d.ts.map +1 -1
  33. package/out/src/adapters/overture/adapter.js +3 -7
  34. package/out/src/adapters/overture/adapter.js.map +1 -1
  35. package/out/src/adapters/state-hi-schools/adapter.d.ts.map +1 -1
  36. package/out/src/adapters/state-hi-schools/adapter.js +55 -73
  37. package/out/src/adapters/state-hi-schools/adapter.js.map +1 -1
  38. package/out/src/adapters/state-ia-contractors/adapter.d.ts.map +1 -1
  39. package/out/src/adapters/state-ia-contractors/adapter.js +58 -76
  40. package/out/src/adapters/state-ia-contractors/adapter.js.map +1 -1
  41. package/out/src/adapters/state-ny-notaries/adapter.d.ts.map +1 -1
  42. package/out/src/adapters/state-ny-notaries/adapter.js +67 -85
  43. package/out/src/adapters/state-ny-notaries/adapter.js.map +1 -1
  44. package/out/src/adapters/state-tx-notaries/adapter.d.ts.map +1 -1
  45. package/out/src/adapters/state-tx-notaries/adapter.js +69 -87
  46. package/out/src/adapters/state-tx-notaries/adapter.js.map +1 -1
  47. package/out/src/adapters/synth-po-box/adapter.d.ts.map +1 -1
  48. package/out/src/adapters/synth-po-box/adapter.js +7 -15
  49. package/out/src/adapters/synth-po-box/adapter.js.map +1 -1
  50. package/out/src/adapters/tiger/street-decompose.d.ts.map +1 -1
  51. package/out/src/adapters/tiger/street-decompose.js +18 -27
  52. package/out/src/adapters/tiger/street-decompose.js.map +1 -1
  53. package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts +3 -3
  54. package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts.map +1 -1
  55. package/out/src/adapters/usgov-hrsa-fqhc/adapter.js +62 -80
  56. package/out/src/adapters/usgov-hrsa-fqhc/adapter.js.map +1 -1
  57. package/out/src/adapters/usgov-imls-pls/adapter.d.ts.map +1 -1
  58. package/out/src/adapters/usgov-imls-pls/adapter.js +60 -82
  59. package/out/src/adapters/usgov-imls-pls/adapter.js.map +1 -1
  60. package/out/src/adapters/usgov-irs-bmf/adapter.d.ts.map +1 -1
  61. package/out/src/adapters/usgov-irs-bmf/adapter.js +58 -65
  62. package/out/src/adapters/usgov-irs-bmf/adapter.js.map +1 -1
  63. package/out/src/adapters/usgov-nad/adapter.d.ts.map +1 -1
  64. package/out/src/adapters/usgov-nad/adapter.js +5 -8
  65. package/out/src/adapters/usgov-nad/adapter.js.map +1 -1
  66. package/out/src/adapters/usgov-nppes/adapter.d.ts.map +1 -1
  67. package/out/src/adapters/usgov-nppes/adapter.js +58 -76
  68. package/out/src/adapters/usgov-nppes/adapter.js.map +1 -1
  69. package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.d.ts.map +1 -1
  70. package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js +51 -71
  71. package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js.map +1 -1
  72. package/out/src/adapters/wof-admin-jp/adapter.d.ts.map +1 -1
  73. package/out/src/adapters/wof-admin-jp/adapter.js +24 -6
  74. package/out/src/adapters/wof-admin-jp/adapter.js.map +1 -1
  75. package/out/src/build.d.ts.map +1 -1
  76. package/out/src/build.js +2 -1
  77. package/out/src/build.js.map +1 -1
  78. package/out/src/golden.d.ts +5 -0
  79. package/out/src/golden.d.ts.map +1 -1
  80. package/out/src/golden.js +23 -9
  81. package/out/src/golden.js.map +1 -1
  82. package/out/src/parquet-wrapper/reader.d.ts +8 -0
  83. package/out/src/parquet-wrapper/reader.d.ts.map +1 -1
  84. package/out/src/parquet-wrapper/reader.js +16 -0
  85. package/out/src/parquet-wrapper/reader.js.map +1 -1
  86. package/out/src/parquet-wrapper/schema.d.ts +4 -0
  87. package/out/src/parquet-wrapper/schema.d.ts.map +1 -1
  88. package/out/src/parquet-wrapper/schema.js.map +1 -1
  89. package/out/src/parquet.d.ts.map +1 -1
  90. package/out/src/parquet.js +2 -13
  91. package/out/src/parquet.js.map +1 -1
  92. package/out/src/shard-recipes/anchor-absorption.d.ts.map +1 -1
  93. package/out/src/shard-recipes/anchor-absorption.js +2 -1
  94. package/out/src/shard-recipes/anchor-absorption.js.map +1 -1
  95. package/out/src/shard-recipes/country-balanced.d.ts.map +1 -1
  96. package/out/src/shard-recipes/country-balanced.js +68 -60
  97. package/out/src/shard-recipes/country-balanced.js.map +1 -1
  98. package/out/src/shard-recipes/fr-fragment.d.ts.map +1 -1
  99. package/out/src/shard-recipes/fr-fragment.js +8 -5
  100. package/out/src/shard-recipes/fr-fragment.js.map +1 -1
  101. package/out/src/shard-recipes/fr-order.d.ts.map +1 -1
  102. package/out/src/shard-recipes/fr-order.js +14 -58
  103. package/out/src/shard-recipes/fr-order.js.map +1 -1
  104. package/out/src/shard-recipes/german.d.ts.map +1 -1
  105. package/out/src/shard-recipes/german.js +11 -58
  106. package/out/src/shard-recipes/german.js.map +1 -1
  107. package/out/src/shard-recipes/index.d.ts.map +1 -1
  108. package/out/src/shard-recipes/index.js +2 -0
  109. package/out/src/shard-recipes/index.js.map +1 -1
  110. package/out/src/shard-recipes/intersection.d.ts +2 -2
  111. package/out/src/shard-recipes/intersection.d.ts.map +1 -1
  112. package/out/src/shard-recipes/intersection.js +24 -62
  113. package/out/src/shard-recipes/intersection.js.map +1 -1
  114. package/out/src/shard-recipes/locale.d.ts.map +1 -1
  115. package/out/src/shard-recipes/locale.js +14 -16
  116. package/out/src/shard-recipes/locale.js.map +1 -1
  117. package/out/src/shard-recipes/no-fragment.d.ts.map +1 -1
  118. package/out/src/shard-recipes/no-fragment.js +8 -5
  119. package/out/src/shard-recipes/no-fragment.js.map +1 -1
  120. package/out/src/shard-recipes/no-street-led.d.ts.map +1 -1
  121. package/out/src/shard-recipes/no-street-led.js +8 -5
  122. package/out/src/shard-recipes/no-street-led.js.map +1 -1
  123. package/out/src/shard-recipes/po-box-cedex.d.ts +3 -3
  124. package/out/src/shard-recipes/po-box-cedex.d.ts.map +1 -1
  125. package/out/src/shard-recipes/po-box-cedex.js +51 -94
  126. package/out/src/shard-recipes/po-box-cedex.js.map +1 -1
  127. package/out/src/shard-recipes/scaffold.d.ts +59 -9
  128. package/out/src/shard-recipes/scaffold.d.ts.map +1 -1
  129. package/out/src/shard-recipes/scaffold.js +51 -30
  130. package/out/src/shard-recipes/scaffold.js.map +1 -1
  131. package/out/src/shard-recipes/street-affix.d.ts.map +1 -1
  132. package/out/src/shard-recipes/street-affix.js +56 -82
  133. package/out/src/shard-recipes/street-affix.js.map +1 -1
  134. package/out/src/shard-recipes/sub-venue-sources.d.ts +231 -0
  135. package/out/src/shard-recipes/sub-venue-sources.d.ts.map +1 -0
  136. package/out/src/shard-recipes/sub-venue-sources.js +669 -0
  137. package/out/src/shard-recipes/sub-venue-sources.js.map +1 -0
  138. package/out/src/shard-recipes/sub-venue.d.ts +269 -0
  139. package/out/src/shard-recipes/sub-venue.d.ts.map +1 -0
  140. package/out/src/shard-recipes/sub-venue.js +709 -0
  141. package/out/src/shard-recipes/sub-venue.js.map +1 -0
  142. package/out/src/shard-recipes/unit.d.ts +1 -1
  143. package/out/src/shard-recipes/unit.d.ts.map +1 -1
  144. package/out/src/shard-recipes/unit.js +22 -64
  145. package/out/src/shard-recipes/unit.js.map +1 -1
  146. package/out/src/split.d.ts +7 -0
  147. package/out/src/split.d.ts.map +1 -1
  148. package/out/src/split.js +7 -0
  149. package/out/src/split.js.map +1 -1
  150. package/out/src/synthesize.d.ts.map +1 -1
  151. package/out/src/synthesize.js +1 -12
  152. package/out/src/synthesize.js.map +1 -1
  153. package/out/src/tools/align-shard.d.ts.map +1 -1
  154. package/out/src/tools/align-shard.js +3 -7
  155. package/out/src/tools/align-shard.js.map +1 -1
  156. package/out/src/tools/audit.d.ts.map +1 -1
  157. package/out/src/tools/audit.js +4 -4
  158. package/out/src/tools/audit.js.map +1 -1
  159. package/out/src/tools/corpus-stats.d.ts +2 -2
  160. package/out/src/tools/corpus-stats.d.ts.map +1 -1
  161. package/out/src/tools/corpus-stats.js +18 -31
  162. package/out/src/tools/corpus-stats.js.map +1 -1
  163. package/out/src/tools/fetch/download.d.ts.map +1 -1
  164. package/out/src/tools/fetch/download.js +5 -6
  165. package/out/src/tools/fetch/download.js.map +1 -1
  166. package/out/src/tools/fetch/imls-pls.d.ts.map +1 -1
  167. package/out/src/tools/fetch/imls-pls.js +2 -2
  168. package/out/src/tools/fetch/imls-pls.js.map +1 -1
  169. package/out/src/tools/fetch/index.d.ts +14 -1
  170. package/out/src/tools/fetch/index.d.ts.map +1 -1
  171. package/out/src/tools/fetch/index.js +14 -1
  172. package/out/src/tools/fetch/index.js.map +1 -1
  173. package/out/src/tools/fetch/nad.d.ts +1 -1
  174. package/out/src/tools/fetch/nad.js +1 -1
  175. package/out/src/tools/fetch/nppes.d.ts.map +1 -1
  176. package/out/src/tools/fetch/nppes.js +2 -1
  177. package/out/src/tools/fetch/nppes.js.map +1 -1
  178. package/out/src/tools/fetch/openaddresses.d.ts +1 -1
  179. package/out/src/tools/fetch/openaddresses.js +1 -1
  180. package/out/src/tools/fetch/ourairports.d.ts +45 -0
  181. package/out/src/tools/fetch/ourairports.d.ts.map +1 -0
  182. package/out/src/tools/fetch/ourairports.js +123 -0
  183. package/out/src/tools/fetch/ourairports.js.map +1 -0
  184. package/out/src/tools/fetch/wikidata-subvenue.d.ts +171 -0
  185. package/out/src/tools/fetch/wikidata-subvenue.d.ts.map +1 -0
  186. package/out/src/tools/fetch/wikidata-subvenue.js +275 -0
  187. package/out/src/tools/fetch/wikidata-subvenue.js.map +1 -0
  188. package/out/src/tools/golden-expand.d.ts.map +1 -1
  189. package/out/src/tools/golden-expand.js +18 -16
  190. package/out/src/tools/golden-expand.js.map +1 -1
  191. package/out/src/tools/golden-promote.d.ts.map +1 -1
  192. package/out/src/tools/golden-promote.js +12 -6
  193. package/out/src/tools/golden-promote.js.map +1 -1
  194. package/out/src/tools/golden-relabel-street.d.ts +196 -0
  195. package/out/src/tools/golden-relabel-street.d.ts.map +1 -0
  196. package/out/src/tools/golden-relabel-street.js +513 -0
  197. package/out/src/tools/golden-relabel-street.js.map +1 -0
  198. package/out/src/tools/index.d.ts +4 -0
  199. package/out/src/tools/index.d.ts.map +1 -1
  200. package/out/src/tools/index.js +4 -0
  201. package/out/src/tools/index.js.map +1 -1
  202. package/out/src/tools/jsonl-to-parquet.d.ts.map +1 -1
  203. package/out/src/tools/jsonl-to-parquet.js +3 -2
  204. package/out/src/tools/jsonl-to-parquet.js.map +1 -1
  205. package/out/src/tools/lint-shard-vocab.d.ts.map +1 -1
  206. package/out/src/tools/lint-shard-vocab.js +3 -2
  207. package/out/src/tools/lint-shard-vocab.js.map +1 -1
  208. package/out/src/tools/lint-shard.d.ts +3 -3
  209. package/out/src/tools/lint-shard.d.ts.map +1 -1
  210. package/out/src/tools/lint-shard.js +23 -32
  211. package/out/src/tools/lint-shard.js.map +1 -1
  212. package/out/src/tools/overlay-manifest.d.ts.map +1 -1
  213. package/out/src/tools/overlay-manifest.js +2 -1
  214. package/out/src/tools/overlay-manifest.js.map +1 -1
  215. package/out/src/tools/overture-subvenue.d.ts +112 -0
  216. package/out/src/tools/overture-subvenue.d.ts.map +1 -0
  217. package/out/src/tools/overture-subvenue.js +144 -0
  218. package/out/src/tools/overture-subvenue.js.map +1 -0
  219. package/out/src/tools/shard-kryptonite.d.ts +1 -1
  220. package/out/src/tools/shard-kryptonite.d.ts.map +1 -1
  221. package/out/src/tools/shard-kryptonite.js +5 -4
  222. package/out/src/tools/shard-kryptonite.js.map +1 -1
  223. package/out/src/tools/shard-translit.d.ts +6 -3
  224. package/out/src/tools/shard-translit.d.ts.map +1 -1
  225. package/out/src/tools/shard-translit.js +8 -6
  226. package/out/src/tools/shard-translit.js.map +1 -1
  227. package/out/src/tools/sub-venue-lexicon.d.ts +507 -0
  228. package/out/src/tools/sub-venue-lexicon.d.ts.map +1 -0
  229. package/out/src/tools/sub-venue-lexicon.js +817 -0
  230. package/out/src/tools/sub-venue-lexicon.js.map +1 -0
  231. package/out/src/tools/sub-venue-promotions.d.ts +94 -0
  232. package/out/src/tools/sub-venue-promotions.d.ts.map +1 -0
  233. package/out/src/tools/sub-venue-promotions.js +266 -0
  234. package/out/src/tools/sub-venue-promotions.js.map +1 -0
  235. package/out/src/wof-json.d.ts +2 -2
  236. package/out/src/wof-json.d.ts.map +1 -1
  237. package/out/src/wof-json.js +5 -8
  238. package/out/src/wof-json.js.map +1 -1
  239. package/package.json +8 -8
  240. package/src/adapter.ts +58 -1
  241. package/src/adapters/ban/adapter.ts +55 -65
  242. package/src/adapters/ban/street-decompose.ts +17 -25
  243. package/src/adapters/fcc-bdc/adapter.ts +2 -33
  244. package/src/adapters/geonames/adapter.ts +64 -72
  245. package/src/adapters/geonames-postal/adapter.ts +42 -50
  246. package/src/adapters/gnaf/adapter.ts +4 -7
  247. package/src/adapters/gnaf/assemble.ts +6 -10
  248. package/src/adapters/openaddresses/adapter.ts +2 -7
  249. package/src/adapters/overture/adapter.ts +3 -6
  250. package/src/adapters/state-hi-schools/adapter.ts +50 -73
  251. package/src/adapters/state-ia-contractors/adapter.ts +52 -76
  252. package/src/adapters/state-ny-notaries/adapter.ts +61 -85
  253. package/src/adapters/state-tx-notaries/adapter.ts +60 -84
  254. package/src/adapters/synth-po-box/adapter.ts +7 -17
  255. package/src/adapters/tiger/street-decompose.ts +21 -31
  256. package/src/adapters/usgov-hrsa-fqhc/adapter.ts +61 -84
  257. package/src/adapters/usgov-imls-pls/adapter.ts +54 -82
  258. package/src/adapters/usgov-irs-bmf/adapter.ts +53 -63
  259. package/src/adapters/usgov-nad/adapter.ts +5 -8
  260. package/src/adapters/usgov-nppes/adapter.ts +51 -75
  261. package/src/adapters/usgov-samhsa-treatment-locator/adapter.ts +45 -71
  262. package/src/adapters/wof-admin-jp/adapter.ts +34 -8
  263. package/src/build.ts +2 -1
  264. package/src/golden.ts +24 -9
  265. package/src/parquet-wrapper/reader.ts +19 -0
  266. package/src/parquet-wrapper/schema.ts +4 -0
  267. package/src/parquet.ts +3 -16
  268. package/src/shard-recipes/anchor-absorption.ts +2 -1
  269. package/src/shard-recipes/country-balanced.ts +69 -70
  270. package/src/shard-recipes/fr-fragment.ts +10 -7
  271. package/src/shard-recipes/fr-order.ts +13 -63
  272. package/src/shard-recipes/german.ts +12 -65
  273. package/src/shard-recipes/index.ts +2 -0
  274. package/src/shard-recipes/intersection.ts +32 -60
  275. package/src/shard-recipes/locale.ts +14 -16
  276. package/src/shard-recipes/no-fragment.ts +10 -7
  277. package/src/shard-recipes/no-street-led.ts +10 -7
  278. package/src/shard-recipes/po-box-cedex.ts +68 -119
  279. package/src/shard-recipes/scaffold.ts +82 -28
  280. package/src/shard-recipes/street-affix.ts +58 -95
  281. package/src/shard-recipes/sub-venue-sources.ts +895 -0
  282. package/src/shard-recipes/sub-venue.ts +982 -0
  283. package/src/shard-recipes/unit.ts +22 -70
  284. package/src/split.ts +7 -0
  285. package/src/synthesize.ts +1 -15
  286. package/src/tools/align-shard.ts +3 -6
  287. package/src/tools/audit.ts +6 -5
  288. package/src/tools/corpus-stats.ts +25 -33
  289. package/src/tools/fetch/download.ts +7 -5
  290. package/src/tools/fetch/imls-pls.ts +2 -2
  291. package/src/tools/fetch/index.ts +14 -1
  292. package/src/tools/fetch/nad.ts +1 -1
  293. package/src/tools/fetch/nppes.ts +2 -1
  294. package/src/tools/fetch/openaddresses.ts +1 -1
  295. package/src/tools/fetch/ourairports.ts +166 -0
  296. package/src/tools/fetch/wikidata-subvenue.ts +386 -0
  297. package/src/tools/golden-expand.ts +18 -13
  298. package/src/tools/golden-promote.ts +15 -6
  299. package/src/tools/golden-relabel-street.ts +748 -0
  300. package/src/tools/index.ts +4 -0
  301. package/src/tools/jsonl-to-parquet.ts +3 -2
  302. package/src/tools/lint-shard-vocab.ts +3 -2
  303. package/src/tools/lint-shard.ts +33 -36
  304. package/src/tools/overlay-manifest.ts +2 -1
  305. package/src/tools/overture-subvenue.ts +215 -0
  306. package/src/tools/shard-kryptonite.ts +5 -4
  307. package/src/tools/shard-translit.ts +12 -7
  308. package/src/tools/sub-venue-lexicon.ts +1250 -0
  309. package/src/tools/sub-venue-promotions.ts +330 -0
  310. package/src/wof-json.ts +5 -8
@@ -0,0 +1,817 @@
1
+ /**
2
+ * @copyright Sister Software
3
+ * @license AGPL-3.0
4
+ * @author Teffen Ellis, et al.
5
+ *
6
+ * Build the sub-venue designator lexicon (#35) — the vocabulary a corpus shard and, eventually, the
7
+ * span proposer read to recognize `Terminal 5`, `North Terminal`, `Concourse B`, `ターミナル1` as
8
+ * venue-INTERIOR structure.
9
+ *
10
+ * Reads the fetch outputs (`mailwoman corpus fetch wikidata-subvenue`, a JSONL of
11
+ * `@mailwoman/osm/sdk`'s `SubVenueSourceRow`s per region, and the Overture slice of `poi.db` via
12
+ * `overture-subvenue.ts`) and emits one committed JSON table. The shape follows
13
+ * `@mailwoman/poi-taxonomy`'s `taxonomy.json` idiom exactly — typed records plus a FLAT phrase array
14
+ * keyed back to a record id, which is what makes a longest-match phrase index cheap to build over it.
15
+ * {@link SubVenueSurface} is this table's `SynonymEntry`.
16
+ *
17
+ * ── Determinism ──────────────────────────────────────────────────────────────────────────────────
18
+ * {@link buildSubVenueLexicon} is a PURE function of its inputs with a stable sort on every array, so
19
+ * a regenerate against the same fetch outputs is byte-identical. No timestamp is emitted for the same
20
+ * reason `taxonomy.json` carries none — a clock in the artifact makes every regenerate a diff.
21
+ * Vintages live in `sources[]`, taken from the fetch manifests.
22
+ *
23
+ * ── The findings this build is SHAPED BY, all measured ───────────────────────────────────────────
24
+ *
25
+ * **1. A feature's `name` is usually the VENUE's name, not a sub-venue phrase.** Measured on the
26
+ * Berlin extract (2,060 matched features, 2026-08-04): a `railway=platform` is named `Stendaler
27
+ * Straße`, a `railway=station` is named `Bellevue`, an `amenity=university` is named `Hertie School`.
28
+ * Across 250,116 named Great Britain features only 6,003 (2.40%) contain a designator token at all.
29
+ * So names are NOT harvested wholesale — {@link extractAttestedPhrases} keeps a name only when it
30
+ * CONTAINS a known designator surface, which is what makes `Terminal E (Untere Ebene)` evidence and
31
+ * `Otto Lilienthal Flughafen Berlin Tegel` not.
32
+ *
33
+ * **2. The identifier lives in `ref`, not in `name`.** Every one of Berlin's 26 `aeroway=gate`
34
+ * features is unnamed and carries only a `ref`: `13`, `6`, `0/1`, `14/15`, `16-18`. That means
35
+ * `Gate A12` is a RENDERING (`<designator> <ref>`) rather than a string anyone has written down, and a
36
+ * shard that wants to generate the designator+identifier form needs the identifier DISTRIBUTION, not
37
+ * a list of phrases. {@link IdentifierShape} is that distribution, and it is why the artifact has a
38
+ * section for it at all.
39
+ *
40
+ * **3. A matched phrase belongs to the record the PHRASE names, not to the record the ROW carries.**
41
+ * Wave 1 attributed every hit to `row.designatorID`, which is the rule that matched the FEATURE. On
42
+ * the GB extract that produced `west → platform`, `hall → platform`, `biggin → platform` — 108 of 133
43
+ * OSM-derived surfaces had a `phrase` that named a different record than the one they pointed at,
44
+ * because a bus stop tagged `public_transport=platform` is named "Village Hall" or "West Kensington".
45
+ * {@link extractAttestedPhrases} now takes a phrase → record INDEX and attributes by phrase; the row's
46
+ * own designator is kept as `context`, which is exactly the axis a confound board needs (a `hall` seen
47
+ * on a platform is a confound; a `hall` seen on a terminal is evidence).
48
+ *
49
+ * ── What `curated: false` means, and how a surface stops being it ────────────────────────────────
50
+ * Wikidata gives a CONCEPT NAME per language, not a designator as addressed. Q849706's Spanish label
51
+ * is `terminal aeroportuaria`; the addressed form is `Terminal`. Q240854 (`hall`) is `sala` in
52
+ * Italian, which names an ordinary room and would fire on half of Italy. So every machine-derived
53
+ * surface lands `curated: false`, and {@link SubVenueLexiconTable} consumers that gate parsing MUST
54
+ * filter to `curated: true`.
55
+ *
56
+ * A surface becomes curated ONLY by matching a {@link SubVenuePromotion} in
57
+ * `sub-venue-promotions.ts` — a per-designator, per-LOCALE decision carrying the census that backs
58
+ * it. Promotion is per-locale because the same token is a designator in one language and a disaster
59
+ * in another: `hall` is `Halle 2` at Frankfurt and `Village Hall` at 3,205 British bus stops.
60
+ *
61
+ * ── The seed DUPLICATES `neural/venue-structure.ts`, knowingly ───────────────────────────────────
62
+ * `@mailwoman/corpus` does not depend on `@mailwoman/neural` (the dependency runs the other way for
63
+ * the training path, and pulling onnxruntime into a corpus build to read three string arrays would be
64
+ * absurd), so the shipped vocabulary is re-declared below. That is a drift surface and it is stated
65
+ * rather than hidden: `sub-venue-lexicon.test.ts` pins the seed's contents literally, so a change in
66
+ * either place fails a test rather than passing silently.
67
+ */
68
+ import { readFileSync, statSync, writeFileSync } from "node:fs";
69
+ import { basename, join } from "node:path";
70
+ import { parseJSONStrict } from "@mailwoman/core/objects";
71
+ import { TextSpliterator } from "spliterator";
72
+ import { SUBVENUE_PROMOTIONS } from "./sub-venue-promotions.js";
73
+ /**
74
+ * This table's own data version. Bump when the source vintages or the build semantics change.
75
+ *
76
+ * `0.2.0` — wave 2: per-region attestation, the phrase-attribution fix (finding 3 above), derived head nouns, the
77
+ * Overture source, and the `promotions[]` receipts section.
78
+ */
79
+ export const SUBVENUE_LEXICON_VERSION = "0.2.0";
80
+ /**
81
+ * Which side of the containment relation a designator names. Mirrors `@mailwoman/osm/sdk`'s `SubVenueTier`; re-declared
82
+ * for the same dependency-direction reason as the seed.
83
+ */
84
+ export const LexiconTier = {
85
+ SubVenue: "subvenue",
86
+ Venue: "venue",
87
+ };
88
+ /**
89
+ * The vocabulary that already ships in `neural/venue-structure.ts`, re-declared. See the module docstring for why this
90
+ * duplication exists.
91
+ *
92
+ * `tier` is added here (the shipped list has no such field): the seven WOF placetypes plus `terminal`/`gate` are all
93
+ * venue-INTERIOR, except `campus` and `building`, which name a whole venue as often as a part of one. They are marked
94
+ * `subvenue` anyway, because that is the role the span proposer uses them in — `Building 43, Googleplex` is a unit
95
+ * inside a venue.
96
+ */
97
+ export const SHIPPED_DESIGNATOR_SEED = [
98
+ { id: "arcade", modifierEligible: true, provenance: ["wof:placetype"] },
99
+ { id: "building", modifierEligible: false, provenance: ["wof:placetype"] },
100
+ { id: "campus", modifierEligible: true, provenance: ["wof:placetype"] },
101
+ { id: "concourse", modifierEligible: true, provenance: ["wof:placetype"] },
102
+ { id: "enclosure", modifierEligible: false, provenance: ["wof:placetype"] },
103
+ { id: "gate", modifierEligible: false, provenance: ["osm:aeroway=gate"] },
104
+ { id: "installation", modifierEligible: false, provenance: ["wof:placetype"] },
105
+ { id: "terminal", modifierEligible: true, provenance: ["osm:aeroway=terminal"] },
106
+ { id: "wing", modifierEligible: true, provenance: ["wof:placetype"] },
107
+ ];
108
+ /**
109
+ * The shipped positional modifiers, re-declared from `neural/venue-structure.ts`'s `VENUE_STRUCTURE_MODIFIERS`.
110
+ */
111
+ export const SHIPPED_MODIFIER_SEED = [
112
+ "central",
113
+ "east",
114
+ "front",
115
+ "inner",
116
+ "lower",
117
+ "main",
118
+ "north",
119
+ "outer",
120
+ "rear",
121
+ "south",
122
+ "upper",
123
+ "west",
124
+ ];
125
+ /**
126
+ * Designators the lexicon ADDS beyond what ships, each with the source that attests it.
127
+ *
128
+ * `platform`, `station` and `airport` come from the OSM extractor's rule table and are the rail/aviation venue-side
129
+ * vocabulary the corpus line needs. `hall` and `satellite` come from Wikidata concepts and from
130
+ * `wof-osm-placetype-map.mdx`'s own "plausible additions" note, which lists `hall` explicitly. `pier` joins them in
131
+ * wave 2 on 282 Overture attestations in the `pier` category plus 162 in the GB extract — the corpus task names `Pier
132
+ * C` as a target shape, so the record has to exist before a shard can generate it.
133
+ *
134
+ * None is `modifierEligible`: that claim needs a confound board per term AND per locale, and `sub-venue-promotions.ts`
135
+ * is where those live. A promotion marks a SURFACE usable; it does not widen the modifier grammar.
136
+ */
137
+ export const PROPOSED_DESIGNATORS = [
138
+ { id: "airport", tier: LexiconTier.Venue, provenance: ["osm:aeroway=aerodrome"] },
139
+ { id: "hall", tier: LexiconTier.SubVenue, provenance: ["wikidata:Q240854"] },
140
+ { id: "pier", tier: LexiconTier.SubVenue, provenance: ["overture:pier"] },
141
+ { id: "platform", tier: LexiconTier.SubVenue, provenance: ["osm:public_transport=platform", "osm:railway=platform"] },
142
+ { id: "satellite", tier: LexiconTier.SubVenue, provenance: ["wikidata:Q15990706"] },
143
+ { id: "station", tier: LexiconTier.Venue, provenance: ["osm:railway=station"] },
144
+ ];
145
+ /**
146
+ * `designatorID` → Wikidata QID, mirroring `fetch/wikidata-subvenue.ts`'s `SUBVENUE_CONCEPTS`. Re-declared here so the
147
+ * builder stays a pure function over PARSED input rather than reaching into a fetch module for a constant; the test
148
+ * pins the two against each other.
149
+ */
150
+ export const CONCEPT_QIDS = {
151
+ terminal: "Q849706",
152
+ gate: "Q247739",
153
+ concourse: "Q862212",
154
+ campus: "Q209465",
155
+ building: "Q41176",
156
+ arcade: "Q186637",
157
+ hall: "Q240854",
158
+ satellite: "Q15990706",
159
+ };
160
+ /**
161
+ * How many real `ref` values each {@link IdentifierShape} keeps.
162
+ *
163
+ * Eight, not "all" and not one. The field exists so a shard author can see what a class actually CONTAINS — GB's
164
+ * `other` class turned out to be semicolon multi-values (`1;2;3`, `13;14`), which one example would have hidden and
165
+ * which the class name does not say. Eight fits a terminal line and covers the variety inside every class the GB
166
+ * extract produced. The COUNT lives in `observations`; this is a sample, not a census.
167
+ */
168
+ const IDENTIFIER_EXAMPLES_PER_SHAPE = 8;
169
+ /**
170
+ * Scripts whose case is meaningful to fold. Everything else is left as written — see {@link SubVenueSurface.phrase}.
171
+ */
172
+ const CASE_FOLDING_SCRIPT = /^[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\d\s\p{P}]+$/u;
173
+ /**
174
+ * Scripts written without spaces between words, where a token split cannot find a designator and a SUBSTRING test is
175
+ * the correct operator. Han, Hiragana, Katakana; Hangul is excluded because Korean does space its words.
176
+ *
177
+ * The Germanic-compound argument that keeps {@link nameContainsSurface} token-bounded for Latin script does not transfer
178
+ * here — there is no `-gate`/`-hall` street-name suffix class in Japanese, and `第1ターミナル` is unreachable by any token
179
+ * split. Measured on the Japan extract: see the harvest counts in `corpus/data/PROVENANCE.md`.
180
+ */
181
+ const NON_SPACING_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}]/u;
182
+ /**
183
+ * Normalize a surface for the table: trim, collapse internal whitespace, and lowercase ONLY when the string is entirely
184
+ * in a bicameral script. `ターミナルビル` and `航站楼` pass through untouched; `Flughafenterminal` folds.
185
+ */
186
+ export function normalizeSurface(text) {
187
+ const trimmed = text.trim().replaceAll(/\s+/gu, " ");
188
+ return CASE_FOLDING_SCRIPT.test(trimmed) ? trimmed.toLowerCase() : trimmed;
189
+ }
190
+ /**
191
+ * Turn the Wikidata designator-label payload into surfaces.
192
+ *
193
+ * A row is dropped when its language tag is empty (an untagged literal, which Wikidata occasionally carries), when the
194
+ * QID maps to no designator in {@link CONCEPT_QIDS}, or when the normalized phrase is empty. Everything that survives
195
+ * lands `curated: false` — see the module docstring.
196
+ */
197
+ export function surfacesFromWikidata(payload, conceptQIDs = CONCEPT_QIDS) {
198
+ const byQID = new Map(Object.entries(conceptQIDs).map(([id, qid]) => [qid, id]));
199
+ const envelope = payload;
200
+ const seen = new Set();
201
+ const out = [];
202
+ for (const binding of envelope.results?.bindings ?? []) {
203
+ const qid = binding.item?.value?.split("/").pop();
204
+ const recordID = qid ? byQID.get(qid) : undefined;
205
+ const lang = binding.lang?.value;
206
+ const raw = binding.label?.value;
207
+ if (!recordID || !lang || !raw)
208
+ continue;
209
+ const phrase = normalizeSurface(raw);
210
+ if (!phrase)
211
+ continue;
212
+ const source = binding.kind?.value === "alt" ? "wikidata:alt" : "wikidata:label";
213
+ // A concept can carry the same string as both a label and an alias, and across dialect subtags
214
+ // (`zh`, `zh-cn`, `zh-hans` all say 航站楼). Key the dedupe on the tuple that identifies a row.
215
+ const key = `${phrase}${recordID}${lang}${source}`;
216
+ if (seen.has(key))
217
+ continue;
218
+ seen.add(key);
219
+ out.push({
220
+ phrase,
221
+ recordID,
222
+ recordKind: "designator",
223
+ lang,
224
+ region: "",
225
+ source,
226
+ curated: false,
227
+ observations: 0,
228
+ context: {},
229
+ });
230
+ }
231
+ return out;
232
+ }
233
+ /**
234
+ * Classify an OSM `ref` into an {@link IdentifierShape} class.
235
+ *
236
+ * The classes are the ones Berlin's gates actually produced, plus the two aviation forms the corpus task names
237
+ * (`Terminal 2F` is digit-letter, `Concourse B` is letter). `range` covers both separators OSM uses for a gate serving
238
+ * more than one stand: `16-18` and `0/1`.
239
+ */
240
+ export function classifyIdentifier(ref) {
241
+ const value = ref.trim();
242
+ if (/^[0-9]+$/.test(value))
243
+ return "digit";
244
+ if (/^[A-Za-z]$/.test(value))
245
+ return "letter";
246
+ if (/^[A-Za-z]+[0-9]+$/.test(value))
247
+ return "letter-digit";
248
+ if (/^[0-9]+[A-Za-z]+$/.test(value))
249
+ return "digit-letter";
250
+ if (/^[0-9A-Za-z]+\s*[-/]\s*[0-9A-Za-z]+$/.test(value))
251
+ return "range";
252
+ return "other";
253
+ }
254
+ /**
255
+ * Index the surfaces accumulated so far by phrase. First writer wins, so a seed record beats a Wikidata alias that
256
+ * happens to collide — `terminal` stays the `terminal` designator even though it is also an Italian alias for it.
257
+ */
258
+ export function buildSurfaceIndex(surfaces) {
259
+ const index = new Map();
260
+ for (const surface of surfaces) {
261
+ if (index.has(surface.phrase))
262
+ continue;
263
+ index.set(surface.phrase, { recordID: surface.recordID, recordKind: surface.recordKind });
264
+ }
265
+ return index;
266
+ }
267
+ /**
268
+ * Every known phrase found in `name`, as whole-token runs for spacing scripts and as substrings for non-spacing ones.
269
+ *
270
+ * Token-boundary matching for Latin script, not substring: `Nordterminal` is a real German compound in which `terminal`
271
+ * is a suffix, and a substring test would also fire on `Terminalstraße`. The compound case is a genuine miss and it is
272
+ * the right miss — admitting suffix matches would fire on every `-hall`/`-gate` compound in Germanic and Nordic street
273
+ * naming, which is exactly the confound class `Briggate`/`Kirkgate` represents.
274
+ *
275
+ * For Han/Kana names that rule finds nothing at all, because the script has no word boundaries: `第1ターミナル` splits into
276
+ * one token that matches no surface. There the LONGEST known substring is the correct operator, and the compound
277
+ * objection does not transfer — Japanese has no `-gate` street-name suffix class.
278
+ */
279
+ export function nameContainsSurfaces(name, index) {
280
+ const normalized = normalizeSurface(name);
281
+ const hits = new Set();
282
+ for (const token of normalized.split(/[\s,()/]+/u)) {
283
+ const stripped = token.replaceAll(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "");
284
+ if (stripped && index.has(stripped)) {
285
+ hits.add(stripped);
286
+ }
287
+ }
288
+ if (NON_SPACING_SCRIPT.test(normalized)) {
289
+ for (const [phrase] of index) {
290
+ if (NON_SPACING_SCRIPT.test(phrase) && normalized.includes(phrase)) {
291
+ hits.add(phrase);
292
+ }
293
+ }
294
+ }
295
+ return [...hits];
296
+ }
297
+ /**
298
+ * Harvest attested phrases and identifier shapes out of a source's rows.
299
+ *
300
+ * `index` gates the name harvest — a name contributes only when it CONTAINS a phrase already in the table. That filter
301
+ * is the whole reason this function is safe to run over raw OSM: see the module docstring's finding 1 for the Berlin
302
+ * measurement that motivated it. The index also decides ATTRIBUTION (finding 3): a hit is a surface of the record the
303
+ * PHRASE names, and the row's own designator is recorded as `context`.
304
+ *
305
+ * Returns surfaces with real `observations` counts, so the lexicon can rank `terminal` above a phrase attested once.
306
+ */
307
+ export function extractAttestedPhrases(rows, index, options = {}) {
308
+ const source = options.source ?? "osm";
309
+ const region = options.region ?? "";
310
+ /**
311
+ * `phrase\0lang\0source` → { count, context }.
312
+ */
313
+ const surfaceCounts = new Map();
314
+ /**
315
+ * `designatorID\0shape` → { count, examples }.
316
+ */
317
+ const shapes = new Map();
318
+ const note = (phrase, lang, sourceTag, context) => {
319
+ const key = `${phrase}${lang}${sourceTag}`;
320
+ const entry = surfaceCounts.get(key) ?? { count: 0, context: new Map() };
321
+ entry.count++;
322
+ entry.context.set(context, (entry.context.get(context) ?? 0) + 1);
323
+ surfaceCounts.set(key, entry);
324
+ };
325
+ for (const row of rows) {
326
+ if (row.name) {
327
+ for (const hit of nameContainsSurfaces(row.name, index)) {
328
+ // `und` — the default `name` tag carries no language. Overture's `name` is the same: a
329
+ // primary name in whatever language the place uses, untagged.
330
+ note(hit, "und", `${source}:name`, row.designatorID);
331
+ }
332
+ }
333
+ for (const [lang, localized] of Object.entries(row.localizedNames ?? {})) {
334
+ for (const hit of nameContainsSurfaces(localized, index)) {
335
+ note(hit, lang, `${source}:name:${lang}`, row.designatorID);
336
+ }
337
+ }
338
+ if (row.ref) {
339
+ const shape = classifyIdentifier(row.ref);
340
+ const key = `${row.designatorID}${shape}`;
341
+ const entry = shapes.get(key) ?? { count: 0, examples: new Set() };
342
+ entry.count++;
343
+ if (entry.examples.size < IDENTIFIER_EXAMPLES_PER_SHAPE) {
344
+ entry.examples.add(row.ref.trim());
345
+ }
346
+ shapes.set(key, entry);
347
+ }
348
+ }
349
+ const surfaces = [...surfaceCounts].map(([key, entry]) => {
350
+ const [phrase, lang, sourceTag] = key.split("");
351
+ const record = index.get(phrase);
352
+ return {
353
+ phrase,
354
+ recordID: record.recordID,
355
+ recordKind: record.recordKind,
356
+ lang,
357
+ region,
358
+ source: sourceTag,
359
+ curated: false,
360
+ observations: entry.count,
361
+ context: Object.fromEntries([...entry.context].toSorted((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))),
362
+ };
363
+ });
364
+ const identifierShapes = [...shapes].map(([key, entry]) => {
365
+ const [designatorID, shape] = key.split("");
366
+ return {
367
+ designatorID,
368
+ region,
369
+ shape,
370
+ observations: entry.count,
371
+ examples: [...entry.examples].toSorted((a, b) => a.localeCompare(b)),
372
+ };
373
+ });
374
+ return { surfaces, identifierShapes };
375
+ }
376
+ /**
377
+ * Diacritic-flattened ASCII fold, for comparing a Slavic or Turkish inflection against its Latin root.
378
+ */
379
+ function asciiFold(text) {
380
+ return text
381
+ .normalize("NFD")
382
+ .replaceAll(/\p{Diacritic}/gu, "")
383
+ .toLowerCase();
384
+ }
385
+ /**
386
+ * How many leading characters two ASCII-folded forms must share for one to count as the other's inflection.
387
+ *
388
+ * Five, or the id's own length when that is shorter (`hall`, `gate`, `wing`, `pier` are four). Measured against the
389
+ * committed Wikidata pull: at five, `terminal`/`terminál`/`terminale`/`terminali`/`terminála`/`terminalo` are all
390
+ * accepted for `terminal` while `campo` and `campws` are both rejected for `campus` (they share four). At six the
391
+ * Spanish `satélite` is lost; at four, Italian `campo` is admitted and it means FIELD.
392
+ */
393
+ const HEAD_NOUN_PREFIX_FLOOR = 5;
394
+ /**
395
+ * The shortest substring a non-Latin head-noun candidate may be. Two: `航站` and `터미널` are both real, `楼` alone is
396
+ * "building" and would fire on every Chinese building name.
397
+ */
398
+ const NON_LATIN_HEAD_MIN_LENGTH = 2;
399
+ /**
400
+ * How many head-noun candidates one non-Latin record+language group may contribute. Six — enough to carry `ターミナル`,
401
+ * `ターミナルビル` and `旅客ターミナル` together, capped because the substring lattice of a nine-character label is large and, ranked
402
+ * by attesting-surface count, nothing past the sixth has more than the minimum two.
403
+ */
404
+ const NON_LATIN_HEAD_CANDIDATE_CAP = 6;
405
+ /**
406
+ * Latin-script test — the scripts an ASCII-folded prefix comparison against a Latin designator id can work on.
407
+ */
408
+ const LATIN_PHRASE = /^[\p{Script=Latin}\d\s\p{P}]+$/u;
409
+ /**
410
+ * The scripts the shared-substring derivation is allowed to run on: Han, Hiragana, Katakana, Hangul.
411
+ *
412
+ * NARROWER than "not Latin", and the narrowing was earned. Run over every non-Latin phrase in the table, the derivation
413
+ * produced 90 fragments of Cyrillic, Greek, Arabic, Thai, Burmese and Tamil words — `сгра`, `град`, `κτίρ`,
414
+ * `ิ่งก่อสร้า` — because those languages have exactly one surface per concept and the only substrings shared inside a
415
+ * group are pieces of one word. Every one of them was unusable, and none could ever be counted: `poi.db` is four
416
+ * countries and this wave's extracts are GB, DE, FR, ES and JP, so nothing in reach attests a Thai or Burmese surface.
417
+ * Deriving a candidate no available source can confirm is not a hypothesis, it is table weight.
418
+ */
419
+ const SHARED_SUBSTRING_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u;
420
+ /**
421
+ * Derive the HEAD NOUN of every multi-part surface, so `terminal aeroportuaria` contributes the form anyone actually
422
+ * writes on an envelope.
423
+ *
424
+ * The problem this solves is the whole reason wave 1 shipped 1,014 uncurated surfaces: Wikidata's label for a concept
425
+ * is the ENCYCLOPAEDIC name (`terminal aeroportuaria`, `letištní terminál`, `havalimanı terminali`), while the
426
+ * addressed form is the bare head (`Terminal`, `Terminál`, `Terminali`). Nothing can promote the encyclopaedic form, so
427
+ * the head has to be extracted before the curation pass has anything to decide about.
428
+ *
429
+ * Two derivations, because the table holds two kinds of writing:
430
+ *
431
+ * - **Latin script — the COGNATE test.** A token is the head when its ASCII fold shares {@link HEAD_NOUN_PREFIX_FLOOR}
432
+ * leading characters with the designator's own canonical id. Nothing subtler survived contact with the data: an
433
+ * earlier version matched a token against any SINGLE-TOKEN surface of the record, and because Dutch `universiteit` is
434
+ * a one-token surface of `campus`, it derived `universitario`, `universitaire`, `üniversite` and twenty more as head
435
+ * nouns of `campus`. Those are the MODIFIER half of the label, and admitting them would have taught the harvest to
436
+ * read "Ciudad Universitaria" as sub-venue structure.
437
+ * - **Non-Latin script — the SHARED-SUBSTRING test.** The cognate test cannot reach a script the id is not written in,
438
+ * and for Han and Kana a token split finds nothing at all. So every substring of length ≥
439
+ * {@link NON_LATIN_HEAD_MIN_LENGTH} occurring in at least two DISTINCT surfaces of the same record and primary
440
+ * language becomes a candidate, ranked by how many surfaces carry it. Japanese yields `ターミナル` (in all five `ja`
441
+ * terminal labels) ahead of `ターミナルビル` (three); Chinese yields `航站`, `航站楼`, `航站樓`. Where the script DOES space its
442
+ * words (Korean, Greek, Cyrillic) a candidate must be a whole token, so `공항 터미널` ∩ `공항터미널` gives `터미널` and never a
443
+ * fragment.
444
+ *
445
+ * The non-Latin branch deliberately emits SEVERAL candidates instead of picking one. Choosing between `航站` and `航站楼`
446
+ * from Wikidata alone is guesswork; the Japan extract answers it by counting, and the promotion ledger records which
447
+ * count won. Everything derived lands `curated: false` — the derivation is a hypothesis about what the addressed form
448
+ * is, and a locale's own data is what confirms or kills it.
449
+ */
450
+ export function deriveHeadNounSurfaces(surfaces) {
451
+ const derived = new Map();
452
+ const seen = new Set(surfaces.map((s) => `${s.phrase}${s.recordID}${s.lang}`));
453
+ const emit = (phrase, from) => {
454
+ if (phrase === from.phrase)
455
+ return;
456
+ const key = `${phrase}${from.recordID}${from.lang}`;
457
+ if (seen.has(key) || derived.has(key))
458
+ return;
459
+ derived.set(key, {
460
+ phrase,
461
+ recordID: from.recordID,
462
+ recordKind: from.recordKind,
463
+ lang: from.lang,
464
+ region: "",
465
+ source: "derived:head-noun",
466
+ curated: false,
467
+ observations: 0,
468
+ context: {},
469
+ });
470
+ };
471
+ // ── Spacing scripts: prefix-match a token against a single-token surface of the same record ──────
472
+ // Latin script: a token that is a cognate of the designator's own canonical id.
473
+ for (const surface of surfaces) {
474
+ if (!LATIN_PHRASE.test(surface.phrase))
475
+ continue;
476
+ const parts = surface.phrase.split(/[^\p{L}\p{N}]+/u).filter(Boolean);
477
+ if (parts.length < 2)
478
+ continue;
479
+ const root = asciiFold(surface.recordID);
480
+ const floor = Math.min(HEAD_NOUN_PREFIX_FLOOR, root.length);
481
+ for (const part of parts) {
482
+ const folded = asciiFold(part);
483
+ if (folded.length >= floor && commonPrefixLength(folded, root) >= floor) {
484
+ emit(part, surface);
485
+ }
486
+ }
487
+ }
488
+ // Non-Latin script: substrings shared by two or more surfaces of the same record + language.
489
+ const groups = new Map();
490
+ for (const surface of surfaces) {
491
+ if (!SHARED_SUBSTRING_SCRIPT.test(surface.phrase))
492
+ continue;
493
+ // Group `zh`, `zh-cn`, `zh-hant` together: they are writing systems for one vocabulary, and the
494
+ // simplified/traditional pair is exactly the evidence a shared substring needs.
495
+ const key = `${surface.recordID} ${surface.lang.split(/[-_]/u)[0]}`;
496
+ const pool = groups.get(key) ?? new Set();
497
+ pool.add(surface.phrase);
498
+ groups.set(key, pool);
499
+ }
500
+ const candidatesByGroup = new Map();
501
+ for (const [key, pool] of groups) {
502
+ if (pool.size < 2)
503
+ continue;
504
+ candidatesByGroup.set(key, sharedSubstringCandidates(pool));
505
+ }
506
+ for (const surface of surfaces) {
507
+ if (!SHARED_SUBSTRING_SCRIPT.test(surface.phrase))
508
+ continue;
509
+ const key = `${surface.recordID} ${surface.lang.split(/[-_]/u)[0]}`;
510
+ for (const candidate of candidatesByGroup.get(key) ?? []) {
511
+ if (surface.phrase.includes(candidate)) {
512
+ emit(candidate, surface);
513
+ }
514
+ }
515
+ }
516
+ return [...derived.values()];
517
+ }
518
+ /**
519
+ * Length of the shared leading run of two strings.
520
+ */
521
+ function commonPrefixLength(a, b) {
522
+ const limit = Math.min(a.length, b.length);
523
+ let i = 0;
524
+ while (i < limit && a[i] === b[i]) {
525
+ i++;
526
+ }
527
+ return i;
528
+ }
529
+ /**
530
+ * Substrings occurring in at least two DISTINCT members of `pool`, ranked by that count and then by length, capped at
531
+ * {@link NON_LATIN_HEAD_CANDIDATE_CAP}.
532
+ *
533
+ * A candidate never spans whitespace, and in a pool whose members contain whitespace a candidate must be a whole token
534
+ * of some member. That is what keeps Korean `공항 터미널` from contributing a fragment straddling the space.
535
+ *
536
+ * MAXIMAL candidates only: one contained in a longer candidate carried by the SAME number of surfaces is dropped, since
537
+ * counting can never separate the two. Every one of `ターミナル`'s five ja labels also contains `ターミ`, `ターミナ` and `ミナル`, so
538
+ * without this the group contributes four indistinguishable candidates and the Japan harvest returns four identical
539
+ * counts. `航站` survives next to `航站楼` because six surfaces carry it against that one's two.
540
+ */
541
+ function sharedSubstringCandidates(pool) {
542
+ const phrases = [...pool];
543
+ const spaced = phrases.some((phrase) => /\s/u.test(phrase));
544
+ const tokens = spaced ? new Set(phrases.flatMap((phrase) => phrase.split(/\s+/u).filter(Boolean))) : null;
545
+ const counts = new Map();
546
+ for (const phrase of phrases) {
547
+ const local = new Set();
548
+ for (let length = NON_LATIN_HEAD_MIN_LENGTH; length <= phrase.length; length++) {
549
+ for (let start = 0; start + length <= phrase.length; start++) {
550
+ const candidate = phrase.slice(start, start + length);
551
+ if (/\s/u.test(candidate))
552
+ continue;
553
+ local.add(candidate);
554
+ }
555
+ }
556
+ for (const candidate of local) {
557
+ counts.set(candidate, (counts.get(candidate) ?? 0) + 1);
558
+ }
559
+ }
560
+ const kept = [...counts].filter(([candidate, count]) => count >= 2 && (!tokens || tokens.has(candidate)));
561
+ return kept
562
+ .filter(([candidate, count]) => kept.every(([other, otherCount]) => other === candidate || otherCount !== count || !other.includes(candidate)))
563
+ .toSorted((a, b) => b[1] - a[1] || b[0].length - a[0].length || a[0].localeCompare(b[0]))
564
+ .slice(0, NON_LATIN_HEAD_CANDIDATE_CAP)
565
+ .map(([candidate]) => candidate);
566
+ }
567
+ /**
568
+ * Apply the curation decisions to a surface list, IN PLACE on a copy.
569
+ *
570
+ * A promotion binds `(designatorID, phrase, locale)`. A surface matches when it names the same record with the same
571
+ * phrase and its language is the locale's language OR the untagged `und` — the default `name` tag carries no language,
572
+ * and a German extract's untagged `Halle 2` is German.
573
+ *
574
+ * REGION is the subtle half. A surface attested in an extract carries that extract's region and matches only its own
575
+ * locale. A surface with `region: ""` is region-FREE — a Wikidata label or a derived head noun — and a promotion
576
+ * reaches it only when no REJECTION exists for the same designator, phrase and language anywhere else. That guard is
577
+ * not decoration: `pier` is promoted for en-GB and rejected for en-US, and without it the en-GB decision would curate
578
+ * the region-free English surface and hand `Pier 1 Imports` the promotion en-US was refused. Where no rejection
579
+ * competes — `terminal` in `es`, `ターミナル` in `ja` — the region-free surface is the whole point, since a language's
580
+ * designator does not stop at a border.
581
+ *
582
+ * Rejections mark nothing themselves. They exist in `promotions[]` as the record of a decision taken, so the next
583
+ * reader meets en-GB `hall`'s 3,204 bus stops before re-proposing it, not after.
584
+ */
585
+ export function applyPromotions(surfaces, promotions) {
586
+ const language = (locale) => locale.split("-")[0];
587
+ const region = (locale) => locale.split("-")[1] ?? "";
588
+ const rejectedLanguages = new Set(promotions
589
+ .filter((promotion) => promotion.decision === "reject")
590
+ .map((promotion) => `${promotion.designatorID} ${promotion.phrase} ${language(promotion.locale)}`));
591
+ const promoted = promotions.filter((promotion) => promotion.decision === "promote");
592
+ return surfaces.map((surface) => {
593
+ if (surface.curated)
594
+ return surface;
595
+ const hit = promoted.some((promotion) => {
596
+ if (promotion.designatorID !== surface.recordID || promotion.phrase !== surface.phrase)
597
+ return false;
598
+ const lang = language(promotion.locale);
599
+ if (surface.lang !== lang && surface.lang !== "und")
600
+ return false;
601
+ if (surface.region === region(promotion.locale))
602
+ return true;
603
+ return surface.region === "" && !rejectedLanguages.has(`${promotion.designatorID} ${promotion.phrase} ${lang}`);
604
+ });
605
+ return hit ? { ...surface, curated: true } : surface;
606
+ });
607
+ }
608
+ /**
609
+ * Build the lexicon table. PURE and deterministic — same inputs, byte-identical output.
610
+ *
611
+ * Order of operations is load-bearing in three places:
612
+ *
613
+ * 1. Seed surfaces are inserted before anything else, so `terminal` indexes to the `terminal` designator rather than to
614
+ * whichever Wikidata alias sorts first.
615
+ * 2. Head nouns are derived AFTER Wikidata and BEFORE the harvests, because `ターミナル` has to exist as a surface before a
616
+ * Japanese extract can be searched for it. That ordering is the entire reason the Japan harvest finds anything — see
617
+ * `PROVENANCE.md`.
618
+ * 3. Promotions are applied LAST, over the union, so a decision can promote a surface whichever source produced it.
619
+ */
620
+ export function buildSubVenueLexicon(input) {
621
+ const designators = SHIPPED_DESIGNATOR_SEED.map((seed) => ({
622
+ id: seed.id,
623
+ tier: LexiconTier.SubVenue,
624
+ modifierEligible: seed.modifierEligible,
625
+ shipped: true,
626
+ provenance: [...seed.provenance],
627
+ }));
628
+ const byID = new Map(designators.map((d) => [d.id, d]));
629
+ for (const proposed of PROPOSED_DESIGNATORS) {
630
+ const existing = byID.get(proposed.id);
631
+ if (existing) {
632
+ existing.provenance = [...new Set([...existing.provenance, ...proposed.provenance])];
633
+ continue;
634
+ }
635
+ const record = {
636
+ id: proposed.id,
637
+ tier: proposed.tier,
638
+ modifierEligible: false,
639
+ shipped: false,
640
+ provenance: [...proposed.provenance],
641
+ };
642
+ designators.push(record);
643
+ byID.set(record.id, record);
644
+ }
645
+ // A Wikidata concept id is provenance for the designator it names, whether or not the concept
646
+ // contributed a usable surface.
647
+ for (const [id, qid] of Object.entries(CONCEPT_QIDS)) {
648
+ const record = byID.get(id);
649
+ if (record) {
650
+ record.provenance = [...new Set([...record.provenance, `wikidata:${qid}`])];
651
+ }
652
+ }
653
+ const modifiers = SHIPPED_MODIFIER_SEED.map((id) => ({
654
+ id,
655
+ shipped: true,
656
+ provenance: ["codex:directionals"],
657
+ }));
658
+ const surfaces = [
659
+ ...designators.map((d) => ({
660
+ phrase: d.id,
661
+ recordID: d.id,
662
+ recordKind: "designator",
663
+ lang: "en",
664
+ region: "",
665
+ source: "seed",
666
+ // The English designator IS the shipped vocabulary — curated by construction.
667
+ curated: d.shipped,
668
+ observations: 0,
669
+ context: {},
670
+ })),
671
+ ...modifiers.map((m) => ({
672
+ phrase: m.id,
673
+ recordID: m.id,
674
+ recordKind: "modifier",
675
+ lang: "en",
676
+ region: "",
677
+ source: "seed",
678
+ curated: true,
679
+ observations: 0,
680
+ context: {},
681
+ })),
682
+ ];
683
+ if (input.wikidata) {
684
+ surfaces.push(...surfacesFromWikidata(input.wikidata));
685
+ }
686
+ surfaces.push(...deriveHeadNounSurfaces(surfaces));
687
+ const identifierShapes = [];
688
+ for (const harvest of input.harvests) {
689
+ const attested = extractAttestedPhrases(harvest.rows, buildSurfaceIndex(surfaces), {
690
+ source: harvest.source,
691
+ region: harvest.region,
692
+ });
693
+ surfaces.push(...attested.surfaces);
694
+ identifierShapes.push(...attested.identifierShapes);
695
+ }
696
+ const promotions = [...(input.promotions ?? SUBVENUE_PROMOTIONS)];
697
+ const curated = applyPromotions(surfaces, promotions);
698
+ // Deterministic order everywhere. `localeCompare` matches the tie-break discipline
699
+ // `generate-taxonomy.ts` and `build-brands.ts` already use.
700
+ designators.sort((a, b) => a.id.localeCompare(b.id));
701
+ modifiers.sort((a, b) => a.id.localeCompare(b.id));
702
+ curated.sort((a, b) => a.phrase.localeCompare(b.phrase) ||
703
+ a.recordID.localeCompare(b.recordID) ||
704
+ a.lang.localeCompare(b.lang) ||
705
+ a.region.localeCompare(b.region) ||
706
+ a.source.localeCompare(b.source));
707
+ return {
708
+ version: SUBVENUE_LEXICON_VERSION,
709
+ sources: input.sources.toSorted((a, b) => a.id.localeCompare(b.id)),
710
+ designators,
711
+ modifiers,
712
+ surfaces: curated,
713
+ identifierShapes: identifierShapes.toSorted((a, b) => a.designatorID.localeCompare(b.designatorID) ||
714
+ a.region.localeCompare(b.region) ||
715
+ a.shape.localeCompare(b.shape)),
716
+ promotions: promotions.toSorted((a, b) => a.designatorID.localeCompare(b.designatorID) ||
717
+ a.locale.localeCompare(b.locale) ||
718
+ a.phrase.localeCompare(b.phrase)),
719
+ };
720
+ }
721
+ /**
722
+ * Serialize the table the way the committed artifact stores it: pretty-printed, trailing newline. Run `oxfmt` over the
723
+ * result before committing — repo law is that committed JSON is oxfmt-clean, which `JSON.stringify` cannot reproduce.
724
+ */
725
+ export function serializeSubVenueLexicon(table) {
726
+ return JSON.stringify(table, null, 2) + "\n";
727
+ }
728
+ /**
729
+ * Read a JSONL file of {@link SubVenueHarvestRow}s. Blank lines and unparseable rows are skipped rather than fatal — an
730
+ * extract is a build output, and one malformed line should not cost the whole lexicon.
731
+ */
732
+ export function readSubVenueJSONL(path) {
733
+ const out = [];
734
+ // `TextSpliterator` rather than `split("\n")` — a whole-country extract runs to 250,000 lines
735
+ // (52 MB for Great Britain), and materializing every segment before reading the first is exactly
736
+ // what the repo lint rule exists to prevent.
737
+ for (const line of TextSpliterator.from(readFileSync(path, "utf8"))) {
738
+ const trimmed = line.trim();
739
+ if (!trimmed)
740
+ continue;
741
+ try {
742
+ out.push(parseJSONStrict(trimmed));
743
+ }
744
+ catch {
745
+ continue;
746
+ }
747
+ }
748
+ return out;
749
+ }
750
+ /**
751
+ * Read the fetch outputs, build the table, and write it.
752
+ *
753
+ * The IO half only — every decision lives in {@link buildSubVenueLexicon}, which is pure. Run `oxfmt` over `outPath`
754
+ * afterwards; repo law is that committed JSON is oxfmt-clean.
755
+ */
756
+ export function generateSubVenueLexicon(options) {
757
+ const sources = [];
758
+ const harvests = [];
759
+ let wikidata = null;
760
+ if (options.wikidataDir) {
761
+ wikidata = parseJSONStrict(readFileSync(join(options.wikidataDir, "designator-labels.json"), "utf8"));
762
+ const manifest = parseJSONStrict(readFileSync(join(options.wikidataDir, "MANIFEST.json"), "utf8"));
763
+ const labelFile = manifest.files?.find((f) => f.filename === "designator-labels.json");
764
+ sources.push({
765
+ id: "wikidata",
766
+ origin: manifest.endpoint ?? "https://query.wikidata.org/sparql",
767
+ license: manifest.license ?? "CC0",
768
+ // The DATE only. A full ISO timestamp would make every re-fetch a diff in the committed artifact for
769
+ // no information a reader of a vocabulary table can act on.
770
+ retrieved: (manifest.downloaded_at ?? "").slice(0, 10),
771
+ rows: labelFile?.rows ?? 0,
772
+ });
773
+ }
774
+ for (const extract of options.extracts ?? []) {
775
+ const rows = readSubVenueJSONL(extract.path);
776
+ harvests.push({ rows, source: "osm", region: extract.region });
777
+ sources.push({
778
+ id: `osm:${extract.region.toLowerCase()}`,
779
+ // The extract's NAME, never its path. `AGENTS.md` forbids re-hardcoding the lab data root
780
+ // anywhere, and a committed artifact carrying `/mnt/playpen/...` would do exactly that while
781
+ // telling a reader on another machine nothing. `great-britain` identifies the Geofabrik region,
782
+ // which is the fact that matters.
783
+ origin: `OpenStreetMap via Geofabrik (${basename(extract.path, ".jsonl")})`,
784
+ license: "ODbL (OpenStreetMap)",
785
+ // The extract's mtime — when the rows were produced. `corpus/AGENTS.md`'s standing warning that
786
+ // a file's mtime is not its DATA's vintage applies to a downloaded archive; this file is a build
787
+ // output of ours, so its mtime is exactly the right number.
788
+ retrieved: statSync(extract.path).mtime.toISOString().slice(0, 10),
789
+ rows: rows.length,
790
+ });
791
+ }
792
+ if (options.overtureRows?.length) {
793
+ // Overture rows carry their own country, so they are harvested per REGION rather than in one pass —
794
+ // a `region` on the surface is the axis promotion is decided on and a mixed-country bucket would
795
+ // make it meaningless.
796
+ const byCountry = new Map();
797
+ for (const row of options.overtureRows) {
798
+ const bucket = byCountry.get(row.country) ?? [];
799
+ bucket.push(row);
800
+ byCountry.set(row.country, bucket);
801
+ }
802
+ for (const [country, rows] of [...byCountry].toSorted((a, b) => a[0].localeCompare(b[0]))) {
803
+ harvests.push({ rows, source: "overture", region: country });
804
+ }
805
+ sources.push({
806
+ id: "overture",
807
+ origin: "Overture Maps Foundation places, via the poi.db spatial layer",
808
+ license: "CDLA-Permissive-2.0",
809
+ retrieved: options.overtureVintage ?? "",
810
+ rows: options.overtureRows.length,
811
+ });
812
+ }
813
+ const table = buildSubVenueLexicon({ wikidata, harvests, sources });
814
+ writeFileSync(options.outPath, serializeSubVenueLexicon(table));
815
+ return table;
816
+ }
817
+ //# sourceMappingURL=sub-venue-lexicon.js.map