@mailwoman/corpus 9.0.0 → 9.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/data/PROVENANCE.md +227 -0
- package/data/reviewed-ve-postcode-tuples.json +74 -0
- package/data/sub-venue-lexicon.json +18269 -0
- package/out/src/adapters/ban/adapter.d.ts +1 -1
- package/out/src/adapters/ban/adapter.d.ts.map +1 -1
- package/out/src/adapters/ban/adapter.js +5 -4
- package/out/src/adapters/ban/adapter.js.map +1 -1
- package/out/src/adapters/fcc-bdc/adapter.d.ts +1 -1
- package/out/src/adapters/fcc-bdc/adapter.d.ts.map +1 -1
- package/out/src/adapters/fcc-bdc/adapter.js +3 -3
- package/out/src/adapters/fcc-bdc/adapter.js.map +1 -1
- package/out/src/adapters/geonames/adapter.d.ts +1 -1
- package/out/src/adapters/geonames/adapter.d.ts.map +1 -1
- package/out/src/adapters/geonames/adapter.js +2 -2
- package/out/src/adapters/geonames/adapter.js.map +1 -1
- package/out/src/adapters/geonames-postal/adapter.d.ts +1 -1
- package/out/src/adapters/geonames-postal/adapter.d.ts.map +1 -1
- package/out/src/adapters/geonames-postal/adapter.js +3 -3
- package/out/src/adapters/geonames-postal/adapter.js.map +1 -1
- package/out/src/adapters/gnaf/adapter.d.ts +1 -1
- package/out/src/adapters/gnaf/adapter.d.ts.map +1 -1
- package/out/src/adapters/gnaf/adapter.js +2 -2
- package/out/src/adapters/gnaf/adapter.js.map +1 -1
- package/out/src/adapters/gnaf/assemble.js +1 -1
- package/out/src/adapters/gnaf/assemble.js.map +1 -1
- package/out/src/adapters/index.d.ts +1 -1
- package/out/src/adapters/index.d.ts.map +1 -1
- package/out/src/adapters/index.js +1 -1
- package/out/src/adapters/index.js.map +1 -1
- package/out/src/adapters/openaddresses/adapter.d.ts +1 -1
- package/out/src/adapters/openaddresses/adapter.d.ts.map +1 -1
- package/out/src/adapters/openaddresses/adapter.js +3 -3
- package/out/src/adapters/openaddresses/adapter.js.map +1 -1
- package/out/src/adapters/overture/adapter.d.ts +1 -1
- package/out/src/adapters/overture/adapter.d.ts.map +1 -1
- package/out/src/adapters/overture/adapter.js +2 -2
- package/out/src/adapters/overture/adapter.js.map +1 -1
- package/out/src/adapters/state-hi-schools/adapter.d.ts +1 -1
- package/out/src/adapters/state-hi-schools/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-hi-schools/adapter.js +7 -6
- package/out/src/adapters/state-hi-schools/adapter.js.map +1 -1
- package/out/src/adapters/state-ia-contractors/adapter.d.ts +1 -1
- package/out/src/adapters/state-ia-contractors/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-ia-contractors/adapter.js +12 -7
- package/out/src/adapters/state-ia-contractors/adapter.js.map +1 -1
- package/out/src/adapters/state-ny-notaries/adapter.d.ts +1 -1
- package/out/src/adapters/state-ny-notaries/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-ny-notaries/adapter.js +12 -18
- package/out/src/adapters/state-ny-notaries/adapter.js.map +1 -1
- package/out/src/adapters/state-tx-notaries/adapter.d.ts +1 -1
- package/out/src/adapters/state-tx-notaries/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-tx-notaries/adapter.js +12 -7
- package/out/src/adapters/state-tx-notaries/adapter.js.map +1 -1
- package/out/src/adapters/synth-po-box/adapter.d.ts +3 -3
- package/out/src/adapters/synth-po-box/adapter.d.ts.map +1 -1
- package/out/src/adapters/synth-po-box/adapter.js +3 -3
- package/out/src/adapters/synth-po-box/adapter.js.map +1 -1
- package/out/src/adapters/tiger/adapter.d.ts +1 -1
- package/out/src/adapters/tiger/adapter.d.ts.map +1 -1
- package/out/src/adapters/tiger/adapter.js +2 -2
- package/out/src/adapters/tiger/adapter.js.map +1 -1
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts +1 -1
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.js +7 -6
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.js.map +1 -1
- package/out/src/adapters/usgov-imls-pls/adapter.d.ts +1 -1
- package/out/src/adapters/usgov-imls-pls/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-imls-pls/adapter.js +11 -7
- package/out/src/adapters/usgov-imls-pls/adapter.js.map +1 -1
- package/out/src/adapters/usgov-irs-bmf/adapter.d.ts +1 -1
- package/out/src/adapters/usgov-irs-bmf/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-irs-bmf/adapter.js +6 -5
- package/out/src/adapters/usgov-irs-bmf/adapter.js.map +1 -1
- package/out/src/adapters/usgov-nad/adapter.d.ts +1 -1
- package/out/src/adapters/usgov-nad/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-nad/adapter.js +8 -8
- package/out/src/adapters/usgov-nad/adapter.js.map +1 -1
- package/out/src/adapters/usgov-nppes/adapter.d.ts +1 -1
- package/out/src/adapters/usgov-nppes/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-nppes/adapter.js +10 -10
- package/out/src/adapters/usgov-nppes/adapter.js.map +1 -1
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.d.ts +1 -1
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js +7 -6
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js.map +1 -1
- package/out/src/{adapter.d.ts → adapters/utils/index.d.ts} +8 -2
- package/out/src/adapters/utils/index.d.ts.map +1 -0
- package/out/src/{adapter.js → adapters/utils/index.js} +11 -3
- package/out/src/adapters/utils/index.js.map +1 -0
- package/out/src/adapters/wof-admin-jp/adapter.d.ts +1 -1
- package/out/src/adapters/wof-admin-jp/adapter.d.ts.map +1 -1
- package/out/src/adapters/wof-admin-jp/adapter.js +5 -5
- package/out/src/adapters/wof-admin-jp/adapter.js.map +1 -1
- package/out/src/adapters/wof-admin-json/adapter.d.ts +4 -15
- package/out/src/adapters/wof-admin-json/adapter.d.ts.map +1 -1
- package/out/src/adapters/wof-admin-json/adapter.js +11 -35
- package/out/src/adapters/wof-admin-json/adapter.js.map +1 -1
- package/out/src/adapters/wof-json-rows.d.ts +32 -0
- package/out/src/adapters/wof-json-rows.d.ts.map +1 -0
- package/out/src/adapters/wof-json-rows.js +45 -0
- package/out/src/adapters/wof-json-rows.js.map +1 -0
- package/out/src/adapters/wof-postalcode-json/adapter.d.ts +4 -9
- package/out/src/adapters/wof-postalcode-json/adapter.d.ts.map +1 -1
- package/out/src/adapters/wof-postalcode-json/adapter.js +12 -37
- package/out/src/adapters/wof-postalcode-json/adapter.js.map +1 -1
- package/out/src/build.d.ts +4 -4
- package/out/src/build.d.ts.map +1 -1
- package/out/src/build.js +7 -7
- package/out/src/build.js.map +1 -1
- package/out/src/index.d.ts +3 -17
- package/out/src/index.d.ts.map +1 -1
- package/out/src/index.js +3 -17
- package/out/src/index.js.map +1 -1
- package/out/src/name-prone-us-suffixes.d.ts +12 -0
- package/out/src/name-prone-us-suffixes.d.ts.map +1 -0
- package/out/src/name-prone-us-suffixes.js +12 -0
- package/out/src/name-prone-us-suffixes.js.map +1 -0
- package/out/src/parquet-wrapper/reader.d.ts.map +1 -1
- package/out/src/parquet-wrapper/reader.js +5 -0
- package/out/src/parquet-wrapper/reader.js.map +1 -1
- package/out/src/runner.d.ts +2 -2
- package/out/src/runner.d.ts.map +1 -1
- package/out/src/runner.js +1 -1
- package/out/src/runner.js.map +1 -1
- package/out/src/shard-recipes/anchor-absorption.d.ts.map +1 -1
- package/out/src/shard-recipes/anchor-absorption.js +4 -5
- package/out/src/shard-recipes/anchor-absorption.js.map +1 -1
- package/out/src/shard-recipes/bare-country.d.ts +25 -0
- package/out/src/shard-recipes/bare-country.d.ts.map +1 -0
- package/out/src/shard-recipes/bare-country.js +87 -0
- package/out/src/shard-recipes/bare-country.js.map +1 -0
- package/out/src/shard-recipes/boundary-stress.d.ts.map +1 -1
- package/out/src/shard-recipes/boundary-stress.js +2 -2
- package/out/src/shard-recipes/boundary-stress.js.map +1 -1
- package/out/src/shard-recipes/country-balanced.d.ts.map +1 -1
- package/out/src/shard-recipes/country-balanced.js +14 -21
- package/out/src/shard-recipes/country-balanced.js.map +1 -1
- package/out/src/shard-recipes/cz-pcfirst-preposition.d.ts.map +1 -1
- package/out/src/shard-recipes/cz-pcfirst-preposition.js +28 -10
- package/out/src/shard-recipes/cz-pcfirst-preposition.js.map +1 -1
- package/out/src/shard-recipes/fr-admin-split.d.ts.map +1 -1
- package/out/src/shard-recipes/fr-admin-split.js +4 -3
- package/out/src/shard-recipes/fr-admin-split.js.map +1 -1
- package/out/src/shard-recipes/fr-bare-street.d.ts.map +1 -1
- package/out/src/shard-recipes/fr-bare-street.js +72 -11
- package/out/src/shard-recipes/fr-bare-street.js.map +1 -1
- package/out/src/shard-recipes/fr-lieudit.d.ts.map +1 -1
- package/out/src/shard-recipes/fr-lieudit.js +4 -3
- package/out/src/shard-recipes/fr-lieudit.js.map +1 -1
- package/out/src/shard-recipes/fr-order.d.ts.map +1 -1
- package/out/src/shard-recipes/fr-order.js +11 -20
- package/out/src/shard-recipes/fr-order.js.map +1 -1
- package/out/src/shard-recipes/german.d.ts.map +1 -1
- package/out/src/shard-recipes/german.js +8 -14
- package/out/src/shard-recipes/german.js.map +1 -1
- package/out/src/shard-recipes/house-venue.d.ts.map +1 -1
- package/out/src/shard-recipes/house-venue.js +1 -1
- package/out/src/shard-recipes/house-venue.js.map +1 -1
- package/out/src/shard-recipes/index.d.ts.map +1 -1
- package/out/src/shard-recipes/index.js +8 -1
- package/out/src/shard-recipes/index.js.map +1 -1
- package/out/src/shard-recipes/intersection.d.ts.map +1 -1
- package/out/src/shard-recipes/intersection.js +7 -13
- package/out/src/shard-recipes/intersection.js.map +1 -1
- package/out/src/shard-recipes/locale.d.ts +5 -5
- package/out/src/shard-recipes/locale.d.ts.map +1 -1
- package/out/src/shard-recipes/locale.js +13 -20
- package/out/src/shard-recipes/locale.js.map +1 -1
- package/out/src/shard-recipes/no-street.d.ts.map +1 -1
- package/out/src/shard-recipes/no-street.js +1 -1
- package/out/src/shard-recipes/no-street.js.map +1 -1
- package/out/src/shard-recipes/po-box-cedex.d.ts.map +1 -1
- package/out/src/shard-recipes/po-box-cedex.js +60 -43
- package/out/src/shard-recipes/po-box-cedex.js.map +1 -1
- package/out/src/shard-recipes/po-box.d.ts.map +1 -1
- package/out/src/shard-recipes/po-box.js +1 -1
- package/out/src/shard-recipes/po-box.js.map +1 -1
- package/out/src/shard-recipes/reviewed-postcode-tail.d.ts +47 -0
- package/out/src/shard-recipes/reviewed-postcode-tail.d.ts.map +1 -0
- package/out/src/shard-recipes/reviewed-postcode-tail.js +125 -0
- package/out/src/shard-recipes/reviewed-postcode-tail.js.map +1 -0
- package/out/src/shard-recipes/scaffold.d.ts +51 -23
- package/out/src/shard-recipes/scaffold.d.ts.map +1 -1
- package/out/src/shard-recipes/scaffold.js +59 -23
- package/out/src/shard-recipes/scaffold.js.map +1 -1
- package/out/src/shard-recipes/street-affix.d.ts +54 -0
- package/out/src/shard-recipes/street-affix.d.ts.map +1 -1
- package/out/src/shard-recipes/street-affix.js +203 -35
- package/out/src/shard-recipes/street-affix.js.map +1 -1
- package/out/src/shard-recipes/street-bare.d.ts.map +1 -1
- package/out/src/shard-recipes/street-bare.js +3 -3
- package/out/src/shard-recipes/street-bare.js.map +1 -1
- package/out/src/shard-recipes/street.d.ts.map +1 -1
- package/out/src/shard-recipes/street.js +2 -2
- package/out/src/shard-recipes/street.js.map +1 -1
- package/out/src/shard-recipes/sub-venue-sources.d.ts +3 -6
- package/out/src/shard-recipes/sub-venue-sources.d.ts.map +1 -1
- package/out/src/shard-recipes/sub-venue-sources.js +4 -9
- package/out/src/shard-recipes/sub-venue-sources.js.map +1 -1
- package/out/src/shard-recipes/sub-venue.d.ts +2 -2
- package/out/src/shard-recipes/sub-venue.d.ts.map +1 -1
- package/out/src/shard-recipes/sub-venue.js +7 -6
- package/out/src/shard-recipes/sub-venue.js.map +1 -1
- package/out/src/shard-recipes/trailing-region.d.ts +74 -0
- package/out/src/shard-recipes/trailing-region.d.ts.map +1 -0
- package/out/src/shard-recipes/trailing-region.js +157 -0
- package/out/src/shard-recipes/trailing-region.js.map +1 -0
- package/out/src/shard-recipes/unit.d.ts.map +1 -1
- package/out/src/shard-recipes/unit.js +6 -12
- package/out/src/shard-recipes/unit.js.map +1 -1
- package/out/src/{synthesize-anchor-absorption.d.ts → synthesizers/anchor-absorption.d.ts} +1 -1
- package/out/src/synthesizers/anchor-absorption.d.ts.map +1 -0
- package/out/src/{synthesize-anchor-absorption.js → synthesizers/anchor-absorption.js} +1 -1
- package/out/src/synthesizers/anchor-absorption.js.map +1 -0
- package/out/src/{synthesize-boundary-stress.d.ts → synthesizers/boundary-stress.d.ts} +3 -3
- package/out/src/synthesizers/boundary-stress.d.ts.map +1 -0
- package/out/src/{synthesize-boundary-stress.js → synthesizers/boundary-stress.js} +2 -2
- package/out/src/synthesizers/boundary-stress.js.map +1 -0
- package/out/src/{synthesize-german.d.ts → synthesizers/german.d.ts} +2 -2
- package/out/src/synthesizers/german.d.ts.map +1 -0
- package/out/src/{synthesize-german.js → synthesizers/german.js} +2 -2
- package/out/src/synthesizers/german.js.map +1 -0
- package/out/src/{synthesize-house-venue.d.ts → synthesizers/house-venue.d.ts} +4 -4
- package/out/src/synthesizers/house-venue.d.ts.map +1 -0
- package/out/src/{synthesize-house-venue.js → synthesizers/house-venue.js} +20 -6
- package/out/src/synthesizers/house-venue.js.map +1 -0
- package/out/src/{synthesize-intersection.d.ts → synthesizers/intersection.d.ts} +2 -2
- package/out/src/synthesizers/intersection.d.ts.map +1 -0
- package/out/src/{synthesize-intersection.js → synthesizers/intersection.js} +1 -1
- package/out/src/synthesizers/intersection.js.map +1 -0
- package/out/src/{synthesize-no-street.d.ts → synthesizers/no-street.d.ts} +3 -3
- package/out/src/synthesizers/no-street.d.ts.map +1 -0
- package/out/src/{synthesize-no-street.js → synthesizers/no-street.js} +2 -2
- package/out/src/synthesizers/no-street.js.map +1 -0
- package/out/src/{synthesize-po-box.d.ts → synthesizers/po-box.d.ts} +2 -2
- package/out/src/synthesizers/po-box.d.ts.map +1 -0
- package/out/src/{synthesize-po-box.js → synthesizers/po-box.js} +1 -1
- package/out/src/synthesizers/po-box.js.map +1 -0
- package/out/src/{synthesize-street.d.ts → synthesizers/street.d.ts} +2 -2
- package/out/src/synthesizers/street.d.ts.map +1 -0
- package/out/src/{synthesize-street.js → synthesizers/street.js} +4 -3
- package/out/src/synthesizers/street.js.map +1 -0
- package/out/src/{synthesize.d.ts → synthesizers/utils.d.ts} +3 -3
- package/out/src/synthesizers/utils.d.ts.map +1 -0
- package/out/src/{synthesize.js → synthesizers/utils.js} +5 -4
- package/out/src/synthesizers/utils.js.map +1 -0
- package/out/src/tools/align-shard.d.ts.map +1 -1
- package/out/src/tools/align-shard.js +2 -2
- package/out/src/tools/align-shard.js.map +1 -1
- package/out/src/tools/fetch/ban.d.ts.map +1 -1
- package/out/src/tools/fetch/ban.js +2 -18
- package/out/src/tools/fetch/ban.js.map +1 -1
- package/out/src/tools/fetch/download.d.ts +12 -3
- package/out/src/tools/fetch/download.d.ts.map +1 -1
- package/out/src/tools/fetch/download.js +20 -4
- package/out/src/tools/fetch/download.js.map +1 -1
- package/out/src/tools/fetch/geonames-dump.d.ts +87 -0
- package/out/src/tools/fetch/geonames-dump.d.ts.map +1 -0
- package/out/src/tools/fetch/geonames-dump.js +178 -0
- package/out/src/tools/fetch/geonames-dump.js.map +1 -0
- package/out/src/tools/fetch/geonames-postal.d.ts +76 -0
- package/out/src/tools/fetch/geonames-postal.d.ts.map +1 -0
- package/out/src/tools/fetch/geonames-postal.js +125 -0
- package/out/src/tools/fetch/geonames-postal.js.map +1 -0
- package/out/src/tools/fetch/imls-pls.d.ts.map +1 -1
- package/out/src/tools/fetch/imls-pls.js +5 -17
- package/out/src/tools/fetch/imls-pls.js.map +1 -1
- package/out/src/tools/fetch/index.d.ts +11 -0
- package/out/src/tools/fetch/index.d.ts.map +1 -1
- package/out/src/tools/fetch/index.js +11 -0
- package/out/src/tools/fetch/index.js.map +1 -1
- package/out/src/tools/fetch/nad.d.ts.map +1 -1
- package/out/src/tools/fetch/nad.js +20 -11
- package/out/src/tools/fetch/nad.js.map +1 -1
- package/out/src/tools/fetch/nppes.d.ts.map +1 -1
- package/out/src/tools/fetch/nppes.js +17 -22
- package/out/src/tools/fetch/nppes.js.map +1 -1
- package/out/src/tools/fetch/openaddresses.d.ts.map +1 -1
- package/out/src/tools/fetch/openaddresses.js +18 -17
- package/out/src/tools/fetch/openaddresses.js.map +1 -1
- package/out/src/tools/fetch/ourairports.d.ts.map +1 -1
- package/out/src/tools/fetch/ourairports.js +7 -2
- package/out/src/tools/fetch/ourairports.js.map +1 -1
- package/out/src/tools/fetch/ppd.d.ts +0 -4
- package/out/src/tools/fetch/ppd.d.ts.map +1 -1
- package/out/src/tools/fetch/ppd.js +1 -1
- package/out/src/tools/fetch/ppd.js.map +1 -1
- package/out/src/tools/fetch/state-hi-schools.d.ts.map +1 -1
- package/out/src/tools/fetch/state-hi-schools.js +3 -19
- package/out/src/tools/fetch/state-hi-schools.js.map +1 -1
- package/out/src/tools/fetch/state-sources.d.ts +4 -0
- package/out/src/tools/fetch/state-sources.d.ts.map +1 -1
- package/out/src/tools/fetch/state-sources.js +1 -5
- package/out/src/tools/fetch/state-sources.js.map +1 -1
- package/out/src/tools/fetch/tiger-full.d.ts.map +1 -1
- package/out/src/tools/fetch/tiger-full.js +15 -21
- package/out/src/tools/fetch/tiger-full.js.map +1 -1
- package/out/src/tools/golden-expand.d.ts.map +1 -1
- package/out/src/tools/golden-expand.js +34 -25
- package/out/src/tools/golden-expand.js.map +1 -1
- package/out/src/tools/golden-relabel-street.d.ts.map +1 -1
- package/out/src/tools/golden-relabel-street.js +4 -77
- package/out/src/tools/golden-relabel-street.js.map +1 -1
- package/out/src/tools/index.d.ts +2 -2
- package/out/src/tools/index.d.ts.map +1 -1
- package/out/src/tools/index.js +2 -2
- package/out/src/tools/index.js.map +1 -1
- package/out/src/tools/ingest-csv.d.ts.map +1 -1
- package/out/src/tools/ingest-csv.js +0 -11
- package/out/src/tools/ingest-csv.js.map +1 -1
- package/out/src/tools/overlay-manifest.d.ts +4 -0
- package/out/src/tools/overlay-manifest.d.ts.map +1 -1
- package/out/src/tools/overlay-manifest.js +16 -6
- package/out/src/tools/overlay-manifest.js.map +1 -1
- package/out/src/tools/postcode-triples.d.ts +175 -0
- package/out/src/tools/postcode-triples.d.ts.map +1 -0
- package/out/src/tools/postcode-triples.js +304 -0
- package/out/src/tools/postcode-triples.js.map +1 -0
- package/out/src/tools/shard-kryptonite.d.ts.map +1 -1
- package/out/src/tools/shard-kryptonite.js +1 -3
- package/out/src/tools/shard-kryptonite.js.map +1 -1
- package/out/src/tools/shard-translit.d.ts.map +1 -1
- package/out/src/tools/shard-translit.js +4 -27
- package/out/src/tools/shard-translit.js.map +1 -1
- package/out/src/tools/sub-venue/harvest.d.ts +100 -0
- package/out/src/tools/sub-venue/harvest.d.ts.map +1 -0
- package/out/src/tools/sub-venue/harvest.js +168 -0
- package/out/src/tools/sub-venue/harvest.js.map +1 -0
- package/out/src/tools/sub-venue/head-nouns.d.ts +50 -0
- package/out/src/tools/sub-venue/head-nouns.d.ts.map +1 -0
- package/out/src/tools/sub-venue/head-nouns.js +210 -0
- package/out/src/tools/sub-venue/head-nouns.js.map +1 -0
- package/out/src/tools/sub-venue/surfaces.d.ts +45 -0
- package/out/src/tools/sub-venue/surfaces.d.ts.map +1 -0
- package/out/src/tools/sub-venue/surfaces.js +77 -0
- package/out/src/tools/sub-venue/surfaces.js.map +1 -0
- package/out/src/tools/sub-venue/table.d.ts +229 -0
- package/out/src/tools/sub-venue/table.d.ts.map +1 -0
- package/out/src/tools/sub-venue/table.js +108 -0
- package/out/src/tools/sub-venue/table.js.map +1 -0
- package/out/src/tools/sub-venue/wikidata.d.ts +26 -0
- package/out/src/tools/sub-venue/wikidata.d.ts.map +1 -0
- package/out/src/tools/sub-venue/wikidata.js +62 -0
- package/out/src/tools/sub-venue/wikidata.js.map +1 -0
- package/out/src/tools/sub-venue-lexicon.d.ts +27 -378
- package/out/src/tools/sub-venue-lexicon.d.ts.map +1 -1
- package/out/src/tools/sub-venue-lexicon.js +30 -565
- package/out/src/tools/sub-venue-lexicon.js.map +1 -1
- package/out/src/{align.d.ts → utils/align.d.ts} +1 -1
- package/out/src/utils/align.d.ts.map +1 -0
- package/out/src/utils/align.js.map +1 -0
- package/out/src/utils/golden.d.ts.map +1 -0
- package/out/src/{golden.js → utils/golden.js} +2 -2
- package/out/src/utils/golden.js.map +1 -0
- package/out/src/utils/index.d.ts +14 -0
- package/out/src/utils/index.d.ts.map +1 -0
- package/out/src/utils/index.js +14 -0
- package/out/src/utils/index.js.map +1 -0
- package/out/src/utils/license.d.ts.map +1 -0
- package/out/src/{license.js → utils/license.js} +1 -1
- package/out/src/utils/license.js.map +1 -0
- package/out/src/{parquet.d.ts → utils/parquet.d.ts} +24 -5
- package/out/src/utils/parquet.d.ts.map +1 -0
- package/out/src/{parquet.js → utils/parquet.js} +6 -3
- package/out/src/utils/parquet.js.map +1 -0
- package/out/src/{split.d.ts → utils/split.d.ts} +1 -1
- package/out/src/utils/split.d.ts.map +1 -0
- package/out/src/{split.js → utils/split.js} +2 -2
- package/out/src/utils/split.js.map +1 -0
- package/out/src/utils/tokenize.d.ts.map +1 -0
- package/out/src/utils/tokenize.js.map +1 -0
- package/out/src/utils/wof-json.d.ts.map +1 -0
- package/out/src/utils/wof-json.js.map +1 -0
- package/package.json +287 -21
- package/src/adapters/ban/adapter.ts +7 -5
- package/src/adapters/fcc-bdc/adapter.ts +5 -4
- package/src/adapters/geonames/adapter.ts +3 -3
- package/src/adapters/geonames-postal/adapter.ts +4 -4
- package/src/adapters/gnaf/adapter.ts +3 -3
- package/src/adapters/gnaf/assemble.ts +1 -1
- package/src/adapters/index.ts +3 -2
- package/src/adapters/openaddresses/adapter.ts +4 -4
- package/src/adapters/overture/adapter.ts +3 -3
- package/src/adapters/state-hi-schools/adapter.ts +8 -7
- package/src/adapters/state-ia-contractors/adapter.ts +13 -8
- package/src/adapters/state-ny-notaries/adapter.ts +13 -32
- package/src/adapters/state-tx-notaries/adapter.ts +13 -8
- package/src/adapters/synth-po-box/adapter.ts +4 -4
- package/src/adapters/tiger/adapter.ts +5 -3
- package/src/adapters/usgov-hrsa-fqhc/adapter.ts +8 -7
- package/src/adapters/usgov-imls-pls/adapter.ts +12 -8
- package/src/adapters/usgov-irs-bmf/adapter.ts +7 -6
- package/src/adapters/usgov-nad/adapter.ts +9 -9
- package/src/adapters/usgov-nppes/adapter.ts +13 -12
- package/src/adapters/usgov-samhsa-treatment-locator/adapter.ts +8 -7
- package/src/{adapter.ts → adapters/utils/index.ts} +15 -3
- package/src/adapters/wof-admin-jp/adapter.ts +9 -12
- package/src/adapters/wof-admin-json/adapter.ts +15 -54
- package/src/adapters/wof-json-rows.ts +79 -0
- package/src/adapters/wof-postalcode-json/adapter.ts +16 -50
- package/src/build.ts +8 -8
- package/src/index.ts +3 -17
- package/src/name-prone-us-suffixes.ts +12 -0
- package/src/parquet-wrapper/reader.ts +9 -0
- package/src/runner.ts +2 -2
- package/src/shard-recipes/anchor-absorption.ts +5 -5
- package/src/shard-recipes/bare-country.ts +95 -0
- package/src/shard-recipes/boundary-stress.ts +3 -2
- package/src/shard-recipes/country-balanced.ts +16 -27
- package/src/shard-recipes/cz-pcfirst-preposition.ts +29 -10
- package/src/shard-recipes/fr-admin-split.ts +6 -4
- package/src/shard-recipes/fr-bare-street.ts +85 -10
- package/src/shard-recipes/fr-lieudit.ts +6 -4
- package/src/shard-recipes/fr-order.ts +13 -26
- package/src/shard-recipes/german.ts +9 -18
- package/src/shard-recipes/house-venue.ts +2 -1
- package/src/shard-recipes/index.ts +8 -1
- package/src/shard-recipes/intersection.ts +9 -17
- package/src/shard-recipes/locale.ts +16 -24
- package/src/shard-recipes/no-street.ts +2 -1
- package/src/shard-recipes/po-box-cedex.ts +76 -58
- package/src/shard-recipes/po-box.ts +2 -1
- package/src/shard-recipes/reviewed-postcode-tail.ts +189 -0
- package/src/shard-recipes/scaffold.ts +99 -24
- package/src/shard-recipes/street-affix.ts +284 -45
- package/src/shard-recipes/street-bare.ts +4 -3
- package/src/shard-recipes/street.ts +3 -2
- package/src/shard-recipes/sub-venue-sources.ts +13 -23
- package/src/shard-recipes/sub-venue.ts +11 -9
- package/src/shard-recipes/trailing-region.ts +183 -0
- package/src/shard-recipes/unit.ts +8 -17
- package/src/{synthesize-boundary-stress.ts → synthesizers/boundary-stress.ts} +2 -2
- package/src/{synthesize-german.ts → synthesizers/german.ts} +3 -2
- package/src/{synthesize-house-venue.ts → synthesizers/house-venue.ts} +20 -6
- package/src/{synthesize-intersection.ts → synthesizers/intersection.ts} +1 -1
- package/src/{synthesize-no-street.ts → synthesizers/no-street.ts} +2 -2
- package/src/{synthesize-po-box.ts → synthesizers/po-box.ts} +1 -1
- package/src/{synthesize-street.ts → synthesizers/street.ts} +5 -3
- package/src/{synthesize.ts → synthesizers/utils.ts} +6 -4
- package/src/tools/align-shard.ts +3 -2
- package/src/tools/fetch/ban.ts +2 -22
- package/src/tools/fetch/download.ts +22 -4
- package/src/tools/fetch/geonames-dump.ts +267 -0
- package/src/tools/fetch/geonames-postal.ts +180 -0
- package/src/tools/fetch/imls-pls.ts +6 -20
- package/src/tools/fetch/index.ts +11 -0
- package/src/tools/fetch/nad.ts +21 -10
- package/src/tools/fetch/nppes.ts +17 -23
- package/src/tools/fetch/openaddresses.ts +18 -22
- package/src/tools/fetch/ourairports.ts +7 -2
- package/src/tools/fetch/ppd.ts +1 -1
- package/src/tools/fetch/state-hi-schools.ts +3 -23
- package/src/tools/fetch/state-sources.ts +1 -1
- package/src/tools/fetch/tiger-full.ts +17 -24
- package/src/tools/golden-expand.ts +44 -33
- package/src/tools/golden-relabel-street.ts +5 -68
- package/src/tools/index.ts +2 -2
- package/src/tools/ingest-csv.ts +0 -13
- package/src/tools/overlay-manifest.ts +20 -8
- package/src/tools/postcode-triples.ts +375 -0
- package/src/tools/shard-kryptonite.ts +4 -5
- package/src/tools/shard-translit.ts +16 -35
- package/src/tools/sub-venue/harvest.ts +235 -0
- package/src/tools/sub-venue/head-nouns.ts +242 -0
- package/src/tools/sub-venue/surfaces.ts +95 -0
- package/src/tools/sub-venue/table.ts +281 -0
- package/src/tools/sub-venue/wikidata.ts +85 -0
- package/src/tools/sub-venue-lexicon.ts +67 -886
- package/src/{align.ts → utils/align.ts} +2 -1
- package/src/{golden.ts → utils/golden.ts} +2 -3
- package/src/utils/index.ts +14 -0
- package/src/{license.ts → utils/license.ts} +1 -1
- package/src/{parquet.ts → utils/parquet.ts} +22 -9
- package/src/{split.ts → utils/split.ts} +3 -3
- package/out/src/adapter.d.ts.map +0 -1
- package/out/src/adapter.js.map +0 -1
- package/out/src/align.d.ts.map +0 -1
- package/out/src/align.js.map +0 -1
- package/out/src/format.d.ts +0 -14
- package/out/src/format.d.ts.map +0 -1
- package/out/src/format.js +0 -14
- package/out/src/format.js.map +0 -1
- package/out/src/golden.d.ts.map +0 -1
- package/out/src/golden.js.map +0 -1
- package/out/src/license.d.ts.map +0 -1
- package/out/src/license.js.map +0 -1
- package/out/src/parquet.d.ts.map +0 -1
- package/out/src/parquet.js.map +0 -1
- package/out/src/split.d.ts.map +0 -1
- package/out/src/split.js.map +0 -1
- package/out/src/synthesize-anchor-absorption.d.ts.map +0 -1
- package/out/src/synthesize-anchor-absorption.js.map +0 -1
- package/out/src/synthesize-boundary-stress.d.ts.map +0 -1
- package/out/src/synthesize-boundary-stress.js.map +0 -1
- package/out/src/synthesize-german.d.ts.map +0 -1
- package/out/src/synthesize-german.js.map +0 -1
- package/out/src/synthesize-house-venue.d.ts.map +0 -1
- package/out/src/synthesize-house-venue.js.map +0 -1
- package/out/src/synthesize-intersection.d.ts.map +0 -1
- package/out/src/synthesize-intersection.js.map +0 -1
- package/out/src/synthesize-no-street.d.ts.map +0 -1
- package/out/src/synthesize-no-street.js.map +0 -1
- package/out/src/synthesize-po-box.d.ts.map +0 -1
- package/out/src/synthesize-po-box.js.map +0 -1
- package/out/src/synthesize-street.d.ts.map +0 -1
- package/out/src/synthesize-street.js.map +0 -1
- package/out/src/synthesize.d.ts.map +0 -1
- package/out/src/synthesize.js.map +0 -1
- package/out/src/tokenize.d.ts.map +0 -1
- package/out/src/tokenize.js.map +0 -1
- package/out/src/wof-json.d.ts.map +0 -1
- package/out/src/wof-json.js.map +0 -1
- package/src/format.ts +0 -14
- /package/out/src/{align.js → utils/align.js} +0 -0
- /package/out/src/{golden.d.ts → utils/golden.d.ts} +0 -0
- /package/out/src/{license.d.ts → utils/license.d.ts} +0 -0
- /package/out/src/{tokenize.d.ts → utils/tokenize.d.ts} +0 -0
- /package/out/src/{tokenize.js → utils/tokenize.js} +0 -0
- /package/out/src/{wof-json.d.ts → utils/wof-json.d.ts} +0 -0
- /package/out/src/{wof-json.js → utils/wof-json.js} +0 -0
- /package/src/{synthesize-anchor-absorption.ts → synthesizers/anchor-absorption.ts} +0 -0
- /package/src/{tokenize.ts → utils/tokenize.ts} +0 -0
- /package/src/{wof-json.ts → utils/wof-json.ts} +0 -0
|
@@ -3,16 +3,14 @@
|
|
|
3
3
|
* @license AGPL-3.0
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
|
-
* Build the sub-venue designator lexicon (#35) — the vocabulary a corpus shard and, eventually,
|
|
7
|
-
* span proposer read to recognize `Terminal 5`, `North Terminal`, `Concourse B`, `ターミナル1` as
|
|
8
|
-
* venue-INTERIOR structure.
|
|
6
|
+
* @file Build the sub-venue designator lexicon (#35) — the vocabulary a corpus shard and, eventually,
|
|
7
|
+
* the span proposer read to recognize `Terminal 5`, `North Terminal`, `Concourse B`, `ターミナル1` as
|
|
8
|
+
* venue-INTERIOR structure. This is the assembly: the record schema lives in `sub-venue/table.ts`, the
|
|
9
|
+
* per-stage machinery in its siblings, and the curation decisions in `sub-venue-promotions.ts`.
|
|
9
10
|
*
|
|
10
11
|
* Reads the fetch outputs (`mailwoman corpus fetch wikidata-subvenue`, a JSONL of
|
|
11
12
|
* `@mailwoman/osm/sdk`'s `SubVenueSourceRow`s per region, and the Overture slice of `poi.db` via
|
|
12
|
-
* `overture-subvenue.ts`) and emits one committed JSON table.
|
|
13
|
-
* `@mailwoman/poi-taxonomy`'s `taxonomy.json` idiom exactly — typed records plus a FLAT phrase array
|
|
14
|
-
* keyed back to a record id, which is what makes a longest-match phrase index cheap to build over it.
|
|
15
|
-
* {@link SubVenueSurface} is this table's `SynonymEntry`.
|
|
13
|
+
* `overture-subvenue.ts`) and emits one committed JSON table.
|
|
16
14
|
*
|
|
17
15
|
* ── Determinism ──────────────────────────────────────────────────────────────────────────────────
|
|
18
16
|
* {@link buildSubVenueLexicon} is a PURE function of its inputs with a stable sort on every array, so
|
|
@@ -20,844 +18,56 @@
|
|
|
20
18
|
* reason `taxonomy.json` carries none — a clock in the artifact makes every regenerate a diff.
|
|
21
19
|
* Vintages live in `sources[]`, taken from the fetch manifests.
|
|
22
20
|
*
|
|
23
|
-
* ──
|
|
21
|
+
* ── Where the stages live ────────────────────────────────────────────────────────────────────────
|
|
22
|
+
* Each stage carries the measurements that shaped it; the order they run in is
|
|
23
|
+
* {@link buildSubVenueLexicon}'s own docstring, and it is load-bearing.
|
|
24
24
|
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
* `Otto Lilienthal Flughafen Berlin Tegel` not.
|
|
32
|
-
*
|
|
33
|
-
* **2. The identifier lives in `ref`, not in `name`.** Every one of Berlin's 26 `aeroway=gate`
|
|
34
|
-
* features is unnamed and carries only a `ref`: `13`, `6`, `0/1`, `14/15`, `16-18`. That means
|
|
35
|
-
* `Gate A12` is a RENDERING (`<designator> <ref>`) rather than a string anyone has written down, and a
|
|
36
|
-
* shard that wants to generate the designator+identifier form needs the identifier DISTRIBUTION, not
|
|
37
|
-
* a list of phrases. {@link IdentifierShape} is that distribution, and it is why the artifact has a
|
|
38
|
-
* section for it at all.
|
|
39
|
-
*
|
|
40
|
-
* **3. A matched phrase belongs to the record the PHRASE names, not to the record the ROW carries.**
|
|
41
|
-
* Wave 1 attributed every hit to `row.designatorID`, which is the rule that matched the FEATURE. On
|
|
42
|
-
* the GB extract that produced `west → platform`, `hall → platform`, `biggin → platform` — 108 of 133
|
|
43
|
-
* OSM-derived surfaces had a `phrase` that named a different record than the one they pointed at,
|
|
44
|
-
* because a bus stop tagged `public_transport=platform` is named "Village Hall" or "West Kensington".
|
|
45
|
-
* {@link extractAttestedPhrases} now takes a phrase → record INDEX and attributes by phrase; the row's
|
|
46
|
-
* own designator is kept as `context`, which is exactly the axis a confound board needs (a `hall` seen
|
|
47
|
-
* on a platform is a confound; a `hall` seen on a terminal is evidence).
|
|
25
|
+
* - `sub-venue/table.ts` — the emitted record schema plus the shipped seed vocabulary.
|
|
26
|
+
* - `sub-venue/surfaces.ts` — phrase normalization, the phrase → record index, and the name-match
|
|
27
|
+
* operator that gates the harvest.
|
|
28
|
+
* - `sub-venue/wikidata.ts` — the designator-label SPARQL payload turned into surfaces.
|
|
29
|
+
* - `sub-venue/head-nouns.ts` — the addressed form derived from an encyclopaedic label.
|
|
30
|
+
* - `sub-venue/harvest.ts` — the harvestable row shape, its JSONL reader, and the attestation pass.
|
|
48
31
|
*
|
|
49
32
|
* ── What `curated: false` means, and how a surface stops being it ────────────────────────────────
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
*
|
|
55
|
-
*
|
|
56
|
-
* A surface becomes curated ONLY by matching a {@link SubVenuePromotion} in
|
|
57
|
-
* `sub-venue-promotions.ts` — a per-designator, per-LOCALE decision carrying the census that backs
|
|
58
|
-
* it. Promotion is per-locale because the same token is a designator in one language and a disaster
|
|
59
|
-
* in another: `hall` is `Halle 2` at Frankfurt and `Village Hall` at 3,205 British bus stops.
|
|
60
|
-
*
|
|
61
|
-
* ── The seed DUPLICATES `neural/venue-structure.ts`, knowingly ───────────────────────────────────
|
|
62
|
-
* `@mailwoman/corpus` does not depend on `@mailwoman/neural` (the dependency runs the other way for
|
|
63
|
-
* the training path, and pulling onnxruntime into a corpus build to read three string arrays would be
|
|
64
|
-
* absurd), so the shipped vocabulary is re-declared below. That is a drift surface and it is stated
|
|
65
|
-
* rather than hidden: `sub-venue-lexicon.test.ts` pins the seed's contents literally, so a change in
|
|
66
|
-
* either place fails a test rather than passing silently.
|
|
33
|
+
* Every machine-derived surface lands `curated: false`, and {@link SubVenueLexiconTable} consumers
|
|
34
|
+
* that gate parsing MUST filter to `curated: true`. A surface becomes curated ONLY by matching a
|
|
35
|
+
* {@link SubVenuePromotion} in `sub-venue-promotions.ts` — a per-designator, per-LOCALE decision
|
|
36
|
+
* carrying the census that backs it. Promotion is per-locale because the same token is a designator in
|
|
37
|
+
* one language and a disaster in another: `hall` is `Halle 2` at Frankfurt and `Village Hall` at 3,205
|
|
38
|
+
* British bus stops.
|
|
67
39
|
*/
|
|
68
40
|
|
|
69
41
|
import { readFileSync, statSync, writeFileSync } from "node:fs"
|
|
70
42
|
import { basename, join } from "node:path"
|
|
71
43
|
|
|
72
44
|
import { parseJSONStrict } from "@mailwoman/core/objects"
|
|
73
|
-
import { TextSpliterator } from "spliterator"
|
|
74
45
|
|
|
75
46
|
import { SUBVENUE_PROMOTIONS, type SubVenuePromotion } from "./sub-venue-promotions.ts"
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
}
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
*
|
|
98
|
-
|
|
99
|
-
export
|
|
100
|
-
/**
|
|
101
|
-
* Canonical id, lowercase English. Matches `neural/venue-structure.ts`'s `VENUE_STRUCTURE_DESIGNATORS` wherever the
|
|
102
|
-
* two overlap.
|
|
103
|
-
*/
|
|
104
|
-
id: string
|
|
105
|
-
tier: LexiconTier
|
|
106
|
-
/**
|
|
107
|
-
* Whether this designator may be preceded by a {@link SubVenueModifier} — the `North Terminal` shape.
|
|
108
|
-
*
|
|
109
|
-
* A SUBSET, and the exclusions are load-bearing: `gate` and `building` form ordinary STREET names in exactly this
|
|
110
|
-
* shape ("East Gate" is a real GB street, "Building Society Place" is a real street), so admitting them turns a
|
|
111
|
-
* correct street parse into a sub-venue one. Setting this true means claiming no street is named `<modifier> <id>`.
|
|
112
|
-
* Check before you do.
|
|
113
|
-
*/
|
|
114
|
-
modifierEligible: boolean
|
|
115
|
-
/**
|
|
116
|
-
* Whether the shipped span proposer already recognizes this designator. `false` means the lexicon proposes it and
|
|
117
|
-
* nothing consumes it yet.
|
|
118
|
-
*/
|
|
119
|
-
shipped: boolean
|
|
120
|
-
/**
|
|
121
|
-
* Where the term comes from, one entry per attesting source: `wof:placetype`, `osm:aeroway=terminal`,
|
|
122
|
-
* `wikidata:Q849706`, `overture:airport_terminal`. Sorted, so a regenerate is stable.
|
|
123
|
-
*/
|
|
124
|
-
provenance: string[]
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
/**
|
|
128
|
-
* One positional modifier — the `North`/`Upper`/`Main` half of `North Terminal`.
|
|
129
|
-
*/
|
|
130
|
-
export interface SubVenueModifier {
|
|
131
|
-
id: string
|
|
132
|
-
shipped: boolean
|
|
133
|
-
provenance: string[]
|
|
134
|
-
}
|
|
135
|
-
|
|
136
|
-
/**
|
|
137
|
-
* One surface form: a phrase, the record it names, and where it was attested.
|
|
138
|
-
*
|
|
139
|
-
* This is the table's `SynonymEntry` — the flat array a phrase index is built over.
|
|
140
|
-
*/
|
|
141
|
-
export interface SubVenueSurface {
|
|
142
|
-
/**
|
|
143
|
-
* The phrase, lowercased for Latin-script languages and left as written otherwise (case-folding is meaningless for
|
|
144
|
-
* Japanese, and `toLowerCase` on Turkish `I` is actively wrong).
|
|
145
|
-
*/
|
|
146
|
-
phrase: string
|
|
147
|
-
/**
|
|
148
|
-
* The {@link SubVenueDesignator.id} or {@link SubVenueModifier.id} this phrase is a surface of.
|
|
149
|
-
*/
|
|
150
|
-
recordID: string
|
|
151
|
-
/**
|
|
152
|
-
* Which record table `recordID` points into.
|
|
153
|
-
*/
|
|
154
|
-
recordKind: "designator" | "modifier"
|
|
155
|
-
/**
|
|
156
|
-
* BCP-47-ish language subtag as the source wrote it (`en`, `ja`, `zh-Hant`, `pt-BR`), or `und` when the source gave
|
|
157
|
-
* an untagged default name.
|
|
158
|
-
*/
|
|
159
|
-
lang: string
|
|
160
|
-
/**
|
|
161
|
-
* ISO 3166-1 alpha-2 of the DATA the phrase was attested in, `""` for vocabulary sources that attest a term's
|
|
162
|
-
* existence rather than its use anywhere. This is the axis promotion is decided on: `hall` is attested 3,274 times in
|
|
163
|
-
* `GB` and every promotion of it lives or dies on a per-region census, never a global one.
|
|
164
|
-
*/
|
|
165
|
-
region: string
|
|
166
|
-
/**
|
|
167
|
-
* `wikidata:label`, `wikidata:alt`, `osm:name`, `osm:name:<lang>`, `overture:name`, `derived:head-noun`, or `seed`.
|
|
168
|
-
*/
|
|
169
|
-
source: string
|
|
170
|
-
/**
|
|
171
|
-
* Whether a human has approved this surface for parsing use IN ITS REGION. Everything machine-derived starts `false`
|
|
172
|
-
* and is flipped only by a matching {@link SubVenuePromotion}. A consumer that gates a parse MUST filter on this — see
|
|
173
|
-
* the module docstring.
|
|
174
|
-
*/
|
|
175
|
-
curated: boolean
|
|
176
|
-
/**
|
|
177
|
-
* How many source features attested this exact phrase, when the source counts (OSM, Overture). `0` for vocabulary
|
|
178
|
-
* sources, which attest a term's EXISTENCE rather than its frequency.
|
|
179
|
-
*/
|
|
180
|
-
observations: number
|
|
181
|
-
/**
|
|
182
|
-
* The rule-assigned designator of the FEATURES that carried this phrase, with a count each — `platform:3205
|
|
183
|
-
* campus:49` for GB's `hall`. Empty for vocabulary sources.
|
|
184
|
-
*
|
|
185
|
-
* This is the confound axis. A `hall` on a `platform` row is a British bus stop named after a village hall; a `hall`
|
|
186
|
-
* on a `terminal` row is a real German departure hall. Without it, a surface's `observations` count is a magnitude
|
|
187
|
-
* with no sign — see the repo's "meaning of zero" rule, which applies just as hard to a large number.
|
|
188
|
-
*/
|
|
189
|
-
context: Record<string, number>
|
|
190
|
-
}
|
|
191
|
-
|
|
192
|
-
/**
|
|
193
|
-
* The measured shape of a designator's identifier half — what follows `Gate`/`Terminal` in real data.
|
|
194
|
-
*
|
|
195
|
-
* Derived from OSM `ref` values, not from names. See the module docstring's finding 2.
|
|
196
|
-
*/
|
|
197
|
-
export interface IdentifierShape {
|
|
198
|
-
designatorID: string
|
|
199
|
-
/**
|
|
200
|
-
* ISO 3166-1 alpha-2 of the extract this distribution was measured in. Per-region because the shapes differ: GB gates
|
|
201
|
-
* are 70% bare digits, Japanese platform refs are overwhelmingly bare digits with a different range, and a shard that
|
|
202
|
-
* generates `Gate <ref>` for a French address should sample France's distribution.
|
|
203
|
-
*/
|
|
204
|
-
region: string
|
|
205
|
-
/**
|
|
206
|
-
* A coarse class: `digit` (`5`), `letter` (`B`), `letter-digit` (`A12`), `digit-letter` (`2F`), `range` (`16-18`,
|
|
207
|
-
* `0/1`), or `other`.
|
|
208
|
-
*/
|
|
209
|
-
shape: string
|
|
210
|
-
observations: number
|
|
211
|
-
/**
|
|
212
|
-
* Up to eight real values, sorted, so a shard author can see what the class actually contains.
|
|
213
|
-
*/
|
|
214
|
-
examples: string[]
|
|
215
|
-
}
|
|
216
|
-
|
|
217
|
-
/**
|
|
218
|
-
* One input source's provenance, copied off its fetch manifest.
|
|
219
|
-
*/
|
|
220
|
-
export interface SubVenueLexiconSource {
|
|
221
|
-
id: string
|
|
222
|
-
origin: string
|
|
223
|
-
license: string
|
|
224
|
-
retrieved: string
|
|
225
|
-
rows: number
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
/**
|
|
229
|
-
* The committed table.
|
|
230
|
-
*/
|
|
231
|
-
export interface SubVenueLexiconTable {
|
|
232
|
-
version: string
|
|
233
|
-
sources: SubVenueLexiconSource[]
|
|
234
|
-
designators: SubVenueDesignator[]
|
|
235
|
-
modifiers: SubVenueModifier[]
|
|
236
|
-
surfaces: SubVenueSurface[]
|
|
237
|
-
identifierShapes: IdentifierShape[]
|
|
238
|
-
/**
|
|
239
|
-
* Every curation decision taken against this table, promotion AND rejection, each with the census that backs it. A
|
|
240
|
-
* rejection is as load-bearing as a promotion: it is what stops the next reader re-proposing `hall` for en-GB.
|
|
241
|
-
*/
|
|
242
|
-
promotions: SubVenuePromotion[]
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
/**
|
|
246
|
-
* The vocabulary that already ships in `neural/venue-structure.ts`, re-declared. See the module docstring for why this
|
|
247
|
-
* duplication exists.
|
|
248
|
-
*
|
|
249
|
-
* `tier` is added here (the shipped list has no such field): the seven WOF placetypes plus `terminal`/`gate` are all
|
|
250
|
-
* venue-INTERIOR, except `campus` and `building`, which name a whole venue as often as a part of one. They are marked
|
|
251
|
-
* `subvenue` anyway, because that is the role the span proposer uses them in — `Building 43, Googleplex` is a unit
|
|
252
|
-
* inside a venue.
|
|
253
|
-
*/
|
|
254
|
-
export const SHIPPED_DESIGNATOR_SEED: ReadonlyArray<{
|
|
255
|
-
id: string
|
|
256
|
-
modifierEligible: boolean
|
|
257
|
-
provenance: string[]
|
|
258
|
-
}> = [
|
|
259
|
-
{ id: "arcade", modifierEligible: true, provenance: ["wof:placetype"] },
|
|
260
|
-
{ id: "building", modifierEligible: false, provenance: ["wof:placetype"] },
|
|
261
|
-
{ id: "campus", modifierEligible: true, provenance: ["wof:placetype"] },
|
|
262
|
-
{ id: "concourse", modifierEligible: true, provenance: ["wof:placetype"] },
|
|
263
|
-
{ id: "enclosure", modifierEligible: false, provenance: ["wof:placetype"] },
|
|
264
|
-
{ id: "gate", modifierEligible: false, provenance: ["osm:aeroway=gate"] },
|
|
265
|
-
{ id: "installation", modifierEligible: false, provenance: ["wof:placetype"] },
|
|
266
|
-
{ id: "terminal", modifierEligible: true, provenance: ["osm:aeroway=terminal"] },
|
|
267
|
-
{ id: "wing", modifierEligible: true, provenance: ["wof:placetype"] },
|
|
268
|
-
]
|
|
269
|
-
|
|
270
|
-
/**
|
|
271
|
-
* The shipped positional modifiers, re-declared from `neural/venue-structure.ts`'s `VENUE_STRUCTURE_MODIFIERS`.
|
|
272
|
-
*/
|
|
273
|
-
export const SHIPPED_MODIFIER_SEED: readonly string[] = [
|
|
274
|
-
"central",
|
|
275
|
-
"east",
|
|
276
|
-
"front",
|
|
277
|
-
"inner",
|
|
278
|
-
"lower",
|
|
279
|
-
"main",
|
|
280
|
-
"north",
|
|
281
|
-
"outer",
|
|
282
|
-
"rear",
|
|
283
|
-
"south",
|
|
284
|
-
"upper",
|
|
285
|
-
"west",
|
|
286
|
-
]
|
|
287
|
-
|
|
288
|
-
/**
|
|
289
|
-
* Designators the lexicon ADDS beyond what ships, each with the source that attests it.
|
|
290
|
-
*
|
|
291
|
-
* `platform`, `station` and `airport` come from the OSM extractor's rule table and are the rail/aviation venue-side
|
|
292
|
-
* vocabulary the corpus line needs. `hall` and `satellite` come from Wikidata concepts and from
|
|
293
|
-
* `wof-osm-placetype-map.mdx`'s own "plausible additions" note, which lists `hall` explicitly. `pier` joins them in
|
|
294
|
-
* wave 2 on 282 Overture attestations in the `pier` category plus 162 in the GB extract — the corpus task names `Pier
|
|
295
|
-
* C` as a target shape, so the record has to exist before a shard can generate it.
|
|
296
|
-
*
|
|
297
|
-
* None is `modifierEligible`: that claim needs a confound board per term AND per locale, and `sub-venue-promotions.ts`
|
|
298
|
-
* is where those live. A promotion marks a SURFACE usable; it does not widen the modifier grammar.
|
|
299
|
-
*/
|
|
300
|
-
export const PROPOSED_DESIGNATORS: ReadonlyArray<{
|
|
301
|
-
id: string
|
|
302
|
-
tier: LexiconTier
|
|
303
|
-
provenance: string[]
|
|
304
|
-
}> = [
|
|
305
|
-
{ id: "airport", tier: LexiconTier.Venue, provenance: ["osm:aeroway=aerodrome"] },
|
|
306
|
-
{ id: "hall", tier: LexiconTier.SubVenue, provenance: ["wikidata:Q240854"] },
|
|
307
|
-
{ id: "pier", tier: LexiconTier.SubVenue, provenance: ["overture:pier"] },
|
|
308
|
-
{ id: "platform", tier: LexiconTier.SubVenue, provenance: ["osm:public_transport=platform", "osm:railway=platform"] },
|
|
309
|
-
{ id: "satellite", tier: LexiconTier.SubVenue, provenance: ["wikidata:Q15990706"] },
|
|
310
|
-
{ id: "station", tier: LexiconTier.Venue, provenance: ["osm:railway=station"] },
|
|
311
|
-
]
|
|
312
|
-
|
|
313
|
-
/**
|
|
314
|
-
* `designatorID` → Wikidata QID, mirroring `fetch/wikidata-subvenue.ts`'s `SUBVENUE_CONCEPTS`. Re-declared here so the
|
|
315
|
-
* builder stays a pure function over PARSED input rather than reaching into a fetch module for a constant; the test
|
|
316
|
-
* pins the two against each other.
|
|
317
|
-
*/
|
|
318
|
-
export const CONCEPT_QIDS: Readonly<Record<string, string>> = {
|
|
319
|
-
terminal: "Q849706",
|
|
320
|
-
gate: "Q247739",
|
|
321
|
-
concourse: "Q862212",
|
|
322
|
-
campus: "Q209465",
|
|
323
|
-
building: "Q41176",
|
|
324
|
-
arcade: "Q186637",
|
|
325
|
-
hall: "Q240854",
|
|
326
|
-
satellite: "Q15990706",
|
|
327
|
-
}
|
|
328
|
-
|
|
329
|
-
/**
|
|
330
|
-
* How many real `ref` values each {@link IdentifierShape} keeps.
|
|
331
|
-
*
|
|
332
|
-
* Eight, not "all" and not one. The field exists so a shard author can see what a class actually CONTAINS — GB's
|
|
333
|
-
* `other` class turned out to be semicolon multi-values (`1;2;3`, `13;14`), which one example would have hidden and
|
|
334
|
-
* which the class name does not say. Eight fits a terminal line and covers the variety inside every class the GB
|
|
335
|
-
* extract produced. The COUNT lives in `observations`; this is a sample, not a census.
|
|
336
|
-
*/
|
|
337
|
-
const IDENTIFIER_EXAMPLES_PER_SHAPE = 8
|
|
338
|
-
|
|
339
|
-
/**
|
|
340
|
-
* Scripts whose case is meaningful to fold. Everything else is left as written — see {@link SubVenueSurface.phrase}.
|
|
341
|
-
*/
|
|
342
|
-
const CASE_FOLDING_SCRIPT = /^[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\d\s\p{P}]+$/u
|
|
343
|
-
|
|
344
|
-
/**
|
|
345
|
-
* Scripts written without spaces between words, where a token split cannot find a designator and a SUBSTRING test is
|
|
346
|
-
* the correct operator. Han, Hiragana, Katakana; Hangul is excluded because Korean does space its words.
|
|
347
|
-
*
|
|
348
|
-
* The Germanic-compound argument that keeps {@link nameContainsSurface} token-bounded for Latin script does not transfer
|
|
349
|
-
* here — there is no `-gate`/`-hall` street-name suffix class in Japanese, and `第1ターミナル` is unreachable by any token
|
|
350
|
-
* split. Measured on the Japan extract: see the harvest counts in `corpus/data/PROVENANCE.md`.
|
|
351
|
-
*/
|
|
352
|
-
const NON_SPACING_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}]/u
|
|
353
|
-
|
|
354
|
-
/**
|
|
355
|
-
* Normalize a surface for the table: trim, collapse internal whitespace, and lowercase ONLY when the string is entirely
|
|
356
|
-
* in a bicameral script. `ターミナルビル` and `航站楼` pass through untouched; `Flughafenterminal` folds.
|
|
357
|
-
*/
|
|
358
|
-
export function normalizeSurface(text: string): string {
|
|
359
|
-
const trimmed = text.trim().replaceAll(/\s+/gu, " ")
|
|
360
|
-
|
|
361
|
-
return CASE_FOLDING_SCRIPT.test(trimmed) ? trimmed.toLowerCase() : trimmed
|
|
362
|
-
}
|
|
363
|
-
|
|
364
|
-
/**
|
|
365
|
-
* The SPARQL results envelope, narrowed to the columns the designator-label query produces.
|
|
366
|
-
*/
|
|
367
|
-
interface SPARQLBinding {
|
|
368
|
-
item?: { value: string }
|
|
369
|
-
lang?: { value: string }
|
|
370
|
-
label?: { value: string }
|
|
371
|
-
kind?: { value: string }
|
|
372
|
-
}
|
|
373
|
-
|
|
374
|
-
interface SPARQLEnvelope {
|
|
375
|
-
results?: { bindings?: SPARQLBinding[] }
|
|
376
|
-
}
|
|
377
|
-
|
|
378
|
-
/**
|
|
379
|
-
* Turn the Wikidata designator-label payload into surfaces.
|
|
380
|
-
*
|
|
381
|
-
* A row is dropped when its language tag is empty (an untagged literal, which Wikidata occasionally carries), when the
|
|
382
|
-
* QID maps to no designator in {@link CONCEPT_QIDS}, or when the normalized phrase is empty. Everything that survives
|
|
383
|
-
* lands `curated: false` — see the module docstring.
|
|
384
|
-
*/
|
|
385
|
-
export function surfacesFromWikidata(
|
|
386
|
-
payload: unknown,
|
|
387
|
-
conceptQIDs: Readonly<Record<string, string>> = CONCEPT_QIDS
|
|
388
|
-
): SubVenueSurface[] {
|
|
389
|
-
const byQID = new Map(Object.entries(conceptQIDs).map(([id, qid]) => [qid, id]))
|
|
390
|
-
const envelope = payload as SPARQLEnvelope
|
|
391
|
-
const seen = new Set<string>()
|
|
392
|
-
const out: SubVenueSurface[] = []
|
|
393
|
-
|
|
394
|
-
for (const binding of envelope.results?.bindings ?? []) {
|
|
395
|
-
const qid = binding.item?.value?.split("/").pop()
|
|
396
|
-
const recordID = qid ? byQID.get(qid) : undefined
|
|
397
|
-
const lang = binding.lang?.value
|
|
398
|
-
const raw = binding.label?.value
|
|
399
|
-
|
|
400
|
-
if (!recordID || !lang || !raw) continue
|
|
401
|
-
|
|
402
|
-
const phrase = normalizeSurface(raw)
|
|
403
|
-
|
|
404
|
-
if (!phrase) continue
|
|
405
|
-
|
|
406
|
-
const source = binding.kind?.value === "alt" ? "wikidata:alt" : "wikidata:label"
|
|
407
|
-
// A concept can carry the same string as both a label and an alias, and across dialect subtags
|
|
408
|
-
// (`zh`, `zh-cn`, `zh-hans` all say 航站楼). Key the dedupe on the tuple that identifies a row.
|
|
409
|
-
const key = `${phrase}${recordID}${lang}${source}`
|
|
410
|
-
|
|
411
|
-
if (seen.has(key)) continue
|
|
412
|
-
seen.add(key)
|
|
413
|
-
|
|
414
|
-
out.push({
|
|
415
|
-
phrase,
|
|
416
|
-
recordID,
|
|
417
|
-
recordKind: "designator",
|
|
418
|
-
lang,
|
|
419
|
-
region: "",
|
|
420
|
-
source,
|
|
421
|
-
curated: false,
|
|
422
|
-
observations: 0,
|
|
423
|
-
context: {},
|
|
424
|
-
})
|
|
425
|
-
}
|
|
426
|
-
|
|
427
|
-
return out
|
|
428
|
-
}
|
|
429
|
-
|
|
430
|
-
/**
|
|
431
|
-
* One row of a harvestable source, as read back off JSONL or out of a layer database.
|
|
432
|
-
*
|
|
433
|
-
* SOURCE-NEUTRAL by design, and verified so in wave 2: an Overture Places row from the `airport_terminal` category is
|
|
434
|
-
* `{ designatorID, name }` and fits unchanged. What did NOT fit was the harvest function's hardcoded `osm:name` source
|
|
435
|
-
* stamp — see `overture-subvenue.ts`'s docstring. Declared locally so the builder does not import `@mailwoman/osm`
|
|
436
|
-
* (which `@mailwoman/corpus` does not depend on) just to name a shape it reads from a file.
|
|
437
|
-
*/
|
|
438
|
-
export interface SubVenueHarvestRow {
|
|
439
|
-
/**
|
|
440
|
-
* The designator the SOURCE's rule assigned to the FEATURE. Not necessarily the record a matched phrase names — see
|
|
441
|
-
* the module docstring's finding 3. Carried into {@link SubVenueSurface.context}.
|
|
442
|
-
*/
|
|
443
|
-
designatorID: string
|
|
444
|
-
name?: string | null
|
|
445
|
-
ref?: string | null
|
|
446
|
-
localizedNames?: Record<string, string>
|
|
447
|
-
}
|
|
448
|
-
|
|
449
|
-
/**
|
|
450
|
-
* Classify an OSM `ref` into an {@link IdentifierShape} class.
|
|
451
|
-
*
|
|
452
|
-
* The classes are the ones Berlin's gates actually produced, plus the two aviation forms the corpus task names
|
|
453
|
-
* (`Terminal 2F` is digit-letter, `Concourse B` is letter). `range` covers both separators OSM uses for a gate serving
|
|
454
|
-
* more than one stand: `16-18` and `0/1`.
|
|
455
|
-
*/
|
|
456
|
-
export function classifyIdentifier(ref: string): string {
|
|
457
|
-
const value = ref.trim()
|
|
458
|
-
|
|
459
|
-
if (/^[0-9]+$/.test(value)) return "digit"
|
|
460
|
-
|
|
461
|
-
if (/^[A-Za-z]$/.test(value)) return "letter"
|
|
462
|
-
|
|
463
|
-
if (/^[A-Za-z]+[0-9]+$/.test(value)) return "letter-digit"
|
|
464
|
-
|
|
465
|
-
if (/^[0-9]+[A-Za-z]+$/.test(value)) return "digit-letter"
|
|
466
|
-
|
|
467
|
-
if (/^[0-9A-Za-z]+\s*[-/]\s*[0-9A-Za-z]+$/.test(value)) return "range"
|
|
468
|
-
|
|
469
|
-
return "other"
|
|
470
|
-
}
|
|
471
|
-
|
|
472
|
-
/**
|
|
473
|
-
* A phrase → record index, keyed on the normalized phrase. Built by {@link buildSurfaceIndex} from the surfaces present
|
|
474
|
-
* before the harvest runs, and the reason a matched phrase can be attributed to the record it actually names.
|
|
475
|
-
*/
|
|
476
|
-
export type SurfaceIndex = ReadonlyMap<string, { recordID: string; recordKind: "designator" | "modifier" }>
|
|
477
|
-
|
|
478
|
-
/**
|
|
479
|
-
* Index the surfaces accumulated so far by phrase. First writer wins, so a seed record beats a Wikidata alias that
|
|
480
|
-
* happens to collide — `terminal` stays the `terminal` designator even though it is also an Italian alias for it.
|
|
481
|
-
*/
|
|
482
|
-
export function buildSurfaceIndex(surfaces: readonly SubVenueSurface[]): SurfaceIndex {
|
|
483
|
-
const index = new Map<string, { recordID: string; recordKind: "designator" | "modifier" }>()
|
|
484
|
-
|
|
485
|
-
for (const surface of surfaces) {
|
|
486
|
-
if (index.has(surface.phrase)) continue
|
|
487
|
-
index.set(surface.phrase, { recordID: surface.recordID, recordKind: surface.recordKind })
|
|
488
|
-
}
|
|
489
|
-
|
|
490
|
-
return index
|
|
491
|
-
}
|
|
492
|
-
|
|
493
|
-
/**
|
|
494
|
-
* Every known phrase found in `name`, as whole-token runs for spacing scripts and as substrings for non-spacing ones.
|
|
495
|
-
*
|
|
496
|
-
* Token-boundary matching for Latin script, not substring: `Nordterminal` is a real German compound in which `terminal`
|
|
497
|
-
* is a suffix, and a substring test would also fire on `Terminalstraße`. The compound case is a genuine miss and it is
|
|
498
|
-
* the right miss — admitting suffix matches would fire on every `-hall`/`-gate` compound in Germanic and Nordic street
|
|
499
|
-
* naming, which is exactly the confound class `Briggate`/`Kirkgate` represents.
|
|
500
|
-
*
|
|
501
|
-
* For Han/Kana names that rule finds nothing at all, because the script has no word boundaries: `第1ターミナル` splits into
|
|
502
|
-
* one token that matches no surface. There the LONGEST known substring is the correct operator, and the compound
|
|
503
|
-
* objection does not transfer — Japanese has no `-gate` street-name suffix class.
|
|
504
|
-
*/
|
|
505
|
-
export function nameContainsSurfaces(name: string, index: SurfaceIndex): string[] {
|
|
506
|
-
const normalized = normalizeSurface(name)
|
|
507
|
-
const hits = new Set<string>()
|
|
508
|
-
|
|
509
|
-
for (const token of normalized.split(/[\s,()/]+/u)) {
|
|
510
|
-
const stripped = token.replaceAll(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "")
|
|
511
|
-
|
|
512
|
-
if (stripped && index.has(stripped)) {
|
|
513
|
-
hits.add(stripped)
|
|
514
|
-
}
|
|
515
|
-
}
|
|
516
|
-
|
|
517
|
-
if (NON_SPACING_SCRIPT.test(normalized)) {
|
|
518
|
-
for (const [phrase] of index) {
|
|
519
|
-
if (NON_SPACING_SCRIPT.test(phrase) && normalized.includes(phrase)) {
|
|
520
|
-
hits.add(phrase)
|
|
521
|
-
}
|
|
522
|
-
}
|
|
523
|
-
}
|
|
524
|
-
|
|
525
|
-
return [...hits]
|
|
526
|
-
}
|
|
527
|
-
|
|
528
|
-
/**
|
|
529
|
-
* Options for one harvest pass — which source stamp its surfaces carry and which region they were attested in.
|
|
530
|
-
*
|
|
531
|
-
* Both default to the OSM/unknown-region values wave 1 hardcoded, so an existing caller is unchanged.
|
|
532
|
-
*/
|
|
533
|
-
export interface HarvestOptions {
|
|
534
|
-
/**
|
|
535
|
-
* Source family: `osm` yields `osm:name` / `osm:name:<lang>`, `overture` yields `overture:name`.
|
|
536
|
-
*/
|
|
537
|
-
source?: string
|
|
538
|
-
/**
|
|
539
|
-
* ISO 3166-1 alpha-2 of the extract or partition. `""` when unknown.
|
|
540
|
-
*/
|
|
541
|
-
region?: string
|
|
542
|
-
}
|
|
543
|
-
|
|
544
|
-
/**
|
|
545
|
-
* Harvest attested phrases and identifier shapes out of a source's rows.
|
|
546
|
-
*
|
|
547
|
-
* `index` gates the name harvest — a name contributes only when it CONTAINS a phrase already in the table. That filter
|
|
548
|
-
* is the whole reason this function is safe to run over raw OSM: see the module docstring's finding 1 for the Berlin
|
|
549
|
-
* measurement that motivated it. The index also decides ATTRIBUTION (finding 3): a hit is a surface of the record the
|
|
550
|
-
* PHRASE names, and the row's own designator is recorded as `context`.
|
|
551
|
-
*
|
|
552
|
-
* Returns surfaces with real `observations` counts, so the lexicon can rank `terminal` above a phrase attested once.
|
|
553
|
-
*/
|
|
554
|
-
export function extractAttestedPhrases(
|
|
555
|
-
rows: Iterable<SubVenueHarvestRow>,
|
|
556
|
-
index: SurfaceIndex,
|
|
557
|
-
options: HarvestOptions = {}
|
|
558
|
-
): { surfaces: SubVenueSurface[]; identifierShapes: IdentifierShape[] } {
|
|
559
|
-
const source = options.source ?? "osm"
|
|
560
|
-
const region = options.region ?? ""
|
|
561
|
-
/**
|
|
562
|
-
* `phrase\0lang\0source` → { count, context }.
|
|
563
|
-
*/
|
|
564
|
-
const surfaceCounts = new Map<string, { count: number; context: Map<string, number> }>()
|
|
565
|
-
/**
|
|
566
|
-
* `designatorID\0shape` → { count, examples }.
|
|
567
|
-
*/
|
|
568
|
-
const shapes = new Map<string, { count: number; examples: Set<string> }>()
|
|
569
|
-
|
|
570
|
-
const note = (phrase: string, lang: string, sourceTag: string, context: string): void => {
|
|
571
|
-
const key = `${phrase}${lang}${sourceTag}`
|
|
572
|
-
const entry = surfaceCounts.get(key) ?? { count: 0, context: new Map<string, number>() }
|
|
573
|
-
|
|
574
|
-
entry.count++
|
|
575
|
-
entry.context.set(context, (entry.context.get(context) ?? 0) + 1)
|
|
576
|
-
surfaceCounts.set(key, entry)
|
|
577
|
-
}
|
|
578
|
-
|
|
579
|
-
for (const row of rows) {
|
|
580
|
-
if (row.name) {
|
|
581
|
-
for (const hit of nameContainsSurfaces(row.name, index)) {
|
|
582
|
-
// `und` — the default `name` tag carries no language. Overture's `name` is the same: a
|
|
583
|
-
// primary name in whatever language the place uses, untagged.
|
|
584
|
-
note(hit, "und", `${source}:name`, row.designatorID)
|
|
585
|
-
}
|
|
586
|
-
}
|
|
587
|
-
|
|
588
|
-
for (const [lang, localized] of Object.entries(row.localizedNames ?? {})) {
|
|
589
|
-
for (const hit of nameContainsSurfaces(localized, index)) {
|
|
590
|
-
note(hit, lang, `${source}:name:${lang}`, row.designatorID)
|
|
591
|
-
}
|
|
592
|
-
}
|
|
593
|
-
|
|
594
|
-
if (row.ref) {
|
|
595
|
-
const shape = classifyIdentifier(row.ref)
|
|
596
|
-
const key = `${row.designatorID}${shape}`
|
|
597
|
-
const entry = shapes.get(key) ?? { count: 0, examples: new Set<string>() }
|
|
598
|
-
|
|
599
|
-
entry.count++
|
|
600
|
-
|
|
601
|
-
if (entry.examples.size < IDENTIFIER_EXAMPLES_PER_SHAPE) {
|
|
602
|
-
entry.examples.add(row.ref.trim())
|
|
603
|
-
}
|
|
604
|
-
|
|
605
|
-
shapes.set(key, entry)
|
|
606
|
-
}
|
|
607
|
-
}
|
|
608
|
-
|
|
609
|
-
const surfaces: SubVenueSurface[] = [...surfaceCounts].map(([key, entry]) => {
|
|
610
|
-
const [phrase, lang, sourceTag] = key.split("") as [string, string, string]
|
|
611
|
-
const record = index.get(phrase)!
|
|
612
|
-
|
|
613
|
-
return {
|
|
614
|
-
phrase,
|
|
615
|
-
recordID: record.recordID,
|
|
616
|
-
recordKind: record.recordKind,
|
|
617
|
-
lang,
|
|
618
|
-
region,
|
|
619
|
-
source: sourceTag,
|
|
620
|
-
curated: false,
|
|
621
|
-
observations: entry.count,
|
|
622
|
-
context: Object.fromEntries([...entry.context].toSorted((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))),
|
|
623
|
-
}
|
|
624
|
-
})
|
|
625
|
-
|
|
626
|
-
const identifierShapes: IdentifierShape[] = [...shapes].map(([key, entry]) => {
|
|
627
|
-
const [designatorID, shape] = key.split("") as [string, string]
|
|
628
|
-
|
|
629
|
-
return {
|
|
630
|
-
designatorID,
|
|
631
|
-
region,
|
|
632
|
-
shape,
|
|
633
|
-
observations: entry.count,
|
|
634
|
-
examples: [...entry.examples].toSorted((a, b) => a.localeCompare(b)),
|
|
635
|
-
}
|
|
636
|
-
})
|
|
637
|
-
|
|
638
|
-
return { surfaces, identifierShapes }
|
|
639
|
-
}
|
|
640
|
-
|
|
641
|
-
/**
|
|
642
|
-
* Diacritic-flattened ASCII fold, for comparing a Slavic or Turkish inflection against its Latin root.
|
|
643
|
-
*/
|
|
644
|
-
function asciiFold(text: string): string {
|
|
645
|
-
return text
|
|
646
|
-
.normalize("NFD")
|
|
647
|
-
.replaceAll(/\p{Diacritic}/gu, "")
|
|
648
|
-
.toLowerCase()
|
|
649
|
-
}
|
|
650
|
-
|
|
651
|
-
/**
|
|
652
|
-
* How many leading characters two ASCII-folded forms must share for one to count as the other's inflection.
|
|
653
|
-
*
|
|
654
|
-
* Five, or the id's own length when that is shorter (`hall`, `gate`, `wing`, `pier` are four). Measured against the
|
|
655
|
-
* committed Wikidata pull: at five, `terminal`/`terminál`/`terminale`/`terminali`/`terminála`/`terminalo` are all
|
|
656
|
-
* accepted for `terminal` while `campo` and `campws` are both rejected for `campus` (they share four). At six the
|
|
657
|
-
* Spanish `satélite` is lost; at four, Italian `campo` is admitted and it means FIELD.
|
|
658
|
-
*/
|
|
659
|
-
const HEAD_NOUN_PREFIX_FLOOR = 5
|
|
660
|
-
|
|
661
|
-
/**
|
|
662
|
-
* The shortest substring a non-Latin head-noun candidate may be. Two: `航站` and `터미널` are both real, `楼` alone is
|
|
663
|
-
* "building" and would fire on every Chinese building name.
|
|
664
|
-
*/
|
|
665
|
-
const NON_LATIN_HEAD_MIN_LENGTH = 2
|
|
666
|
-
|
|
667
|
-
/**
|
|
668
|
-
* How many head-noun candidates one non-Latin record+language group may contribute. Six — enough to carry `ターミナル`,
|
|
669
|
-
* `ターミナルビル` and `旅客ターミナル` together, capped because the substring lattice of a nine-character label is large and, ranked
|
|
670
|
-
* by attesting-surface count, nothing past the sixth has more than the minimum two.
|
|
671
|
-
*/
|
|
672
|
-
const NON_LATIN_HEAD_CANDIDATE_CAP = 6
|
|
673
|
-
|
|
674
|
-
/**
|
|
675
|
-
* Latin-script test — the scripts an ASCII-folded prefix comparison against a Latin designator id can work on.
|
|
676
|
-
*/
|
|
677
|
-
const LATIN_PHRASE = /^[\p{Script=Latin}\d\s\p{P}]+$/u
|
|
678
|
-
|
|
679
|
-
/**
|
|
680
|
-
* The scripts the shared-substring derivation is allowed to run on: Han, Hiragana, Katakana, Hangul.
|
|
681
|
-
*
|
|
682
|
-
* NARROWER than "not Latin", and the narrowing was earned. Run over every non-Latin phrase in the table, the derivation
|
|
683
|
-
* produced 90 fragments of Cyrillic, Greek, Arabic, Thai, Burmese and Tamil words — `сгра`, `град`, `κτίρ`,
|
|
684
|
-
* `ิ่งก่อสร้า` — because those languages have exactly one surface per concept and the only substrings shared inside a
|
|
685
|
-
* group are pieces of one word. Every one of them was unusable, and none could ever be counted: `poi.db` is four
|
|
686
|
-
* countries and this wave's extracts are GB, DE, FR, ES and JP, so nothing in reach attests a Thai or Burmese surface.
|
|
687
|
-
* Deriving a candidate no available source can confirm is not a hypothesis, it is table weight.
|
|
688
|
-
*/
|
|
689
|
-
const SHARED_SUBSTRING_SCRIPT = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u
|
|
690
|
-
|
|
691
|
-
/**
|
|
692
|
-
* Derive the HEAD NOUN of every multi-part surface, so `terminal aeroportuaria` contributes the form anyone actually
|
|
693
|
-
* writes on an envelope.
|
|
694
|
-
*
|
|
695
|
-
* The problem this solves is the whole reason wave 1 shipped 1,014 uncurated surfaces: Wikidata's label for a concept
|
|
696
|
-
* is the ENCYCLOPAEDIC name (`terminal aeroportuaria`, `letištní terminál`, `havalimanı terminali`), while the
|
|
697
|
-
* addressed form is the bare head (`Terminal`, `Terminál`, `Terminali`). Nothing can promote the encyclopaedic form, so
|
|
698
|
-
* the head has to be extracted before the curation pass has anything to decide about.
|
|
699
|
-
*
|
|
700
|
-
* Two derivations, because the table holds two kinds of writing:
|
|
701
|
-
*
|
|
702
|
-
* - **Latin script — the COGNATE test.** A token is the head when its ASCII fold shares {@link HEAD_NOUN_PREFIX_FLOOR}
|
|
703
|
-
* leading characters with the designator's own canonical id. Nothing subtler survived contact with the data: an
|
|
704
|
-
* earlier version matched a token against any SINGLE-TOKEN surface of the record, and because Dutch `universiteit` is
|
|
705
|
-
* a one-token surface of `campus`, it derived `universitario`, `universitaire`, `üniversite` and twenty more as head
|
|
706
|
-
* nouns of `campus`. Those are the MODIFIER half of the label, and admitting them would have taught the harvest to
|
|
707
|
-
* read "Ciudad Universitaria" as sub-venue structure.
|
|
708
|
-
* - **Non-Latin script — the SHARED-SUBSTRING test.** The cognate test cannot reach a script the id is not written in,
|
|
709
|
-
* and for Han and Kana a token split finds nothing at all. So every substring of length ≥
|
|
710
|
-
* {@link NON_LATIN_HEAD_MIN_LENGTH} occurring in at least two DISTINCT surfaces of the same record and primary
|
|
711
|
-
* language becomes a candidate, ranked by how many surfaces carry it. Japanese yields `ターミナル` (in all five `ja`
|
|
712
|
-
* terminal labels) ahead of `ターミナルビル` (three); Chinese yields `航站`, `航站楼`, `航站樓`. Where the script DOES space its
|
|
713
|
-
* words (Korean, Greek, Cyrillic) a candidate must be a whole token, so `공항 터미널` ∩ `공항터미널` gives `터미널` and never a
|
|
714
|
-
* fragment.
|
|
715
|
-
*
|
|
716
|
-
* The non-Latin branch deliberately emits SEVERAL candidates instead of picking one. Choosing between `航站` and `航站楼`
|
|
717
|
-
* from Wikidata alone is guesswork; the Japan extract answers it by counting, and the promotion ledger records which
|
|
718
|
-
* count won. Everything derived lands `curated: false` — the derivation is a hypothesis about what the addressed form
|
|
719
|
-
* is, and a locale's own data is what confirms or kills it.
|
|
720
|
-
*/
|
|
721
|
-
export function deriveHeadNounSurfaces(surfaces: readonly SubVenueSurface[]): SubVenueSurface[] {
|
|
722
|
-
const derived = new Map<string, SubVenueSurface>()
|
|
723
|
-
const seen = new Set(surfaces.map((s) => `${s.phrase}${s.recordID}${s.lang}`))
|
|
724
|
-
|
|
725
|
-
const emit = (phrase: string, from: SubVenueSurface): void => {
|
|
726
|
-
if (phrase === from.phrase) return
|
|
727
|
-
|
|
728
|
-
const key = `${phrase}${from.recordID}${from.lang}`
|
|
729
|
-
|
|
730
|
-
if (seen.has(key) || derived.has(key)) return
|
|
731
|
-
|
|
732
|
-
derived.set(key, {
|
|
733
|
-
phrase,
|
|
734
|
-
recordID: from.recordID,
|
|
735
|
-
recordKind: from.recordKind,
|
|
736
|
-
lang: from.lang,
|
|
737
|
-
region: "",
|
|
738
|
-
source: "derived:head-noun",
|
|
739
|
-
curated: false,
|
|
740
|
-
observations: 0,
|
|
741
|
-
context: {},
|
|
742
|
-
})
|
|
743
|
-
}
|
|
744
|
-
|
|
745
|
-
// ── Spacing scripts: prefix-match a token against a single-token surface of the same record ──────
|
|
746
|
-
// Latin script: a token that is a cognate of the designator's own canonical id.
|
|
747
|
-
for (const surface of surfaces) {
|
|
748
|
-
if (!LATIN_PHRASE.test(surface.phrase)) continue
|
|
749
|
-
|
|
750
|
-
const parts = surface.phrase.split(/[^\p{L}\p{N}]+/u).filter(Boolean)
|
|
751
|
-
|
|
752
|
-
if (parts.length < 2) continue
|
|
753
|
-
|
|
754
|
-
const root = asciiFold(surface.recordID)
|
|
755
|
-
const floor = Math.min(HEAD_NOUN_PREFIX_FLOOR, root.length)
|
|
756
|
-
|
|
757
|
-
for (const part of parts) {
|
|
758
|
-
const folded = asciiFold(part)
|
|
759
|
-
|
|
760
|
-
if (folded.length >= floor && commonPrefixLength(folded, root) >= floor) {
|
|
761
|
-
emit(part, surface)
|
|
762
|
-
}
|
|
763
|
-
}
|
|
764
|
-
}
|
|
765
|
-
|
|
766
|
-
// Non-Latin script: substrings shared by two or more surfaces of the same record + language.
|
|
767
|
-
const groups = new Map<string, Set<string>>()
|
|
768
|
-
|
|
769
|
-
for (const surface of surfaces) {
|
|
770
|
-
if (!SHARED_SUBSTRING_SCRIPT.test(surface.phrase)) continue
|
|
771
|
-
|
|
772
|
-
// Group `zh`, `zh-cn`, `zh-hant` together: they are writing systems for one vocabulary, and the
|
|
773
|
-
// simplified/traditional pair is exactly the evidence a shared substring needs.
|
|
774
|
-
const key = `${surface.recordID} ${surface.lang.split(/[-_]/u)[0]!}`
|
|
775
|
-
const pool = groups.get(key) ?? new Set<string>()
|
|
776
|
-
pool.add(surface.phrase)
|
|
777
|
-
groups.set(key, pool)
|
|
778
|
-
}
|
|
779
|
-
|
|
780
|
-
const candidatesByGroup = new Map<string, string[]>()
|
|
781
|
-
|
|
782
|
-
for (const [key, pool] of groups) {
|
|
783
|
-
if (pool.size < 2) continue
|
|
784
|
-
candidatesByGroup.set(key, sharedSubstringCandidates(pool))
|
|
785
|
-
}
|
|
786
|
-
|
|
787
|
-
for (const surface of surfaces) {
|
|
788
|
-
if (!SHARED_SUBSTRING_SCRIPT.test(surface.phrase)) continue
|
|
789
|
-
|
|
790
|
-
const key = `${surface.recordID} ${surface.lang.split(/[-_]/u)[0]!}`
|
|
791
|
-
|
|
792
|
-
for (const candidate of candidatesByGroup.get(key) ?? []) {
|
|
793
|
-
if (surface.phrase.includes(candidate)) {
|
|
794
|
-
emit(candidate, surface)
|
|
795
|
-
}
|
|
796
|
-
}
|
|
797
|
-
}
|
|
798
|
-
|
|
799
|
-
return [...derived.values()]
|
|
800
|
-
}
|
|
801
|
-
|
|
802
|
-
/**
|
|
803
|
-
* Length of the shared leading run of two strings.
|
|
804
|
-
*/
|
|
805
|
-
function commonPrefixLength(a: string, b: string): number {
|
|
806
|
-
const limit = Math.min(a.length, b.length)
|
|
807
|
-
let i = 0
|
|
808
|
-
|
|
809
|
-
while (i < limit && a[i] === b[i]) {
|
|
810
|
-
i++
|
|
811
|
-
}
|
|
812
|
-
|
|
813
|
-
return i
|
|
814
|
-
}
|
|
815
|
-
|
|
816
|
-
/**
|
|
817
|
-
* Substrings occurring in at least two DISTINCT members of `pool`, ranked by that count and then by length, capped at
|
|
818
|
-
* {@link NON_LATIN_HEAD_CANDIDATE_CAP}.
|
|
819
|
-
*
|
|
820
|
-
* A candidate never spans whitespace, and in a pool whose members contain whitespace a candidate must be a whole token
|
|
821
|
-
* of some member. That is what keeps Korean `공항 터미널` from contributing a fragment straddling the space.
|
|
822
|
-
*
|
|
823
|
-
* MAXIMAL candidates only: one contained in a longer candidate carried by the SAME number of surfaces is dropped, since
|
|
824
|
-
* counting can never separate the two. Every one of `ターミナル`'s five ja labels also contains `ターミ`, `ターミナ` and `ミナル`, so
|
|
825
|
-
* without this the group contributes four indistinguishable candidates and the Japan harvest returns four identical
|
|
826
|
-
* counts. `航站` survives next to `航站楼` because six surfaces carry it against that one's two.
|
|
827
|
-
*/
|
|
828
|
-
function sharedSubstringCandidates(pool: ReadonlySet<string>): string[] {
|
|
829
|
-
const phrases = [...pool]
|
|
830
|
-
const spaced = phrases.some((phrase) => /\s/u.test(phrase))
|
|
831
|
-
const tokens = spaced ? new Set(phrases.flatMap((phrase) => phrase.split(/\s+/u).filter(Boolean))) : null
|
|
832
|
-
const counts = new Map<string, number>()
|
|
833
|
-
|
|
834
|
-
for (const phrase of phrases) {
|
|
835
|
-
const local = new Set<string>()
|
|
836
|
-
|
|
837
|
-
for (let length = NON_LATIN_HEAD_MIN_LENGTH; length <= phrase.length; length++) {
|
|
838
|
-
for (let start = 0; start + length <= phrase.length; start++) {
|
|
839
|
-
const candidate = phrase.slice(start, start + length)
|
|
840
|
-
|
|
841
|
-
if (/\s/u.test(candidate)) continue
|
|
842
|
-
local.add(candidate)
|
|
843
|
-
}
|
|
844
|
-
}
|
|
845
|
-
|
|
846
|
-
for (const candidate of local) {
|
|
847
|
-
counts.set(candidate, (counts.get(candidate) ?? 0) + 1)
|
|
848
|
-
}
|
|
849
|
-
}
|
|
850
|
-
|
|
851
|
-
const kept = [...counts].filter(([candidate, count]) => count >= 2 && (!tokens || tokens.has(candidate)))
|
|
852
|
-
|
|
853
|
-
return kept
|
|
854
|
-
.filter(([candidate, count]) =>
|
|
855
|
-
kept.every(([other, otherCount]) => other === candidate || otherCount !== count || !other.includes(candidate))
|
|
856
|
-
)
|
|
857
|
-
.toSorted((a, b) => b[1] - a[1] || b[0].length - a[0].length || a[0].localeCompare(b[0]))
|
|
858
|
-
.slice(0, NON_LATIN_HEAD_CANDIDATE_CAP)
|
|
859
|
-
.map(([candidate]) => candidate)
|
|
860
|
-
}
|
|
47
|
+
import { extractAttestedPhrases, readSubVenueJSONL, type SubVenueHarvestRow } from "./sub-venue/harvest.ts"
|
|
48
|
+
import { deriveHeadNounSurfaces } from "./sub-venue/head-nouns.ts"
|
|
49
|
+
import { buildSurfaceIndex } from "./sub-venue/surfaces.ts"
|
|
50
|
+
import {
|
|
51
|
+
CONCEPT_QIDS,
|
|
52
|
+
type IdentifierShape,
|
|
53
|
+
LexiconTier,
|
|
54
|
+
PROPOSED_DESIGNATORS,
|
|
55
|
+
SHIPPED_DESIGNATOR_SEED,
|
|
56
|
+
SHIPPED_MODIFIER_SEED,
|
|
57
|
+
type SubVenueDesignator,
|
|
58
|
+
type SubVenueLexiconSource,
|
|
59
|
+
type SubVenueLexiconTable,
|
|
60
|
+
type SubVenueModifier,
|
|
61
|
+
type SubVenueSurface,
|
|
62
|
+
SUBVENUE_LEXICON_VERSION,
|
|
63
|
+
} from "./sub-venue/table.ts"
|
|
64
|
+
import { surfacesFromWikidata } from "./sub-venue/wikidata.ts"
|
|
65
|
+
|
|
66
|
+
export * from "./sub-venue/harvest.ts"
|
|
67
|
+
export * from "./sub-venue/head-nouns.ts"
|
|
68
|
+
export * from "./sub-venue/surfaces.ts"
|
|
69
|
+
export * from "./sub-venue/table.ts"
|
|
70
|
+
export * from "./sub-venue/wikidata.ts"
|
|
861
71
|
|
|
862
72
|
/**
|
|
863
73
|
* Apply the curation decisions to a surface list, IN PLACE on a copy.
|
|
@@ -1006,33 +216,29 @@ export function buildSubVenueLexicon(input: BuildSubVenueLexiconInput): SubVenue
|
|
|
1006
216
|
}))
|
|
1007
217
|
|
|
1008
218
|
const surfaces: SubVenueSurface[] = [
|
|
1009
|
-
...designators.map(
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
observations: 0,
|
|
1033
|
-
context: {},
|
|
1034
|
-
})
|
|
1035
|
-
),
|
|
219
|
+
...designators.map((d): SubVenueSurface => ({
|
|
220
|
+
phrase: d.id,
|
|
221
|
+
recordID: d.id,
|
|
222
|
+
recordKind: "designator",
|
|
223
|
+
lang: "en",
|
|
224
|
+
region: "",
|
|
225
|
+
source: "seed",
|
|
226
|
+
// The English designator IS the shipped vocabulary — curated by construction.
|
|
227
|
+
curated: d.shipped,
|
|
228
|
+
observations: 0,
|
|
229
|
+
context: {},
|
|
230
|
+
})),
|
|
231
|
+
...modifiers.map((m): SubVenueSurface => ({
|
|
232
|
+
phrase: m.id,
|
|
233
|
+
recordID: m.id,
|
|
234
|
+
recordKind: "modifier",
|
|
235
|
+
lang: "en",
|
|
236
|
+
region: "",
|
|
237
|
+
source: "seed",
|
|
238
|
+
curated: true,
|
|
239
|
+
observations: 0,
|
|
240
|
+
context: {},
|
|
241
|
+
})),
|
|
1036
242
|
]
|
|
1037
243
|
|
|
1038
244
|
if (input.wikidata) {
|
|
@@ -1099,31 +305,6 @@ export function serializeSubVenueLexicon(table: SubVenueLexiconTable): string {
|
|
|
1099
305
|
return JSON.stringify(table, null, 2) + "\n"
|
|
1100
306
|
}
|
|
1101
307
|
|
|
1102
|
-
/**
|
|
1103
|
-
* Read a JSONL file of {@link SubVenueHarvestRow}s. Blank lines and unparseable rows are skipped rather than fatal — an
|
|
1104
|
-
* extract is a build output, and one malformed line should not cost the whole lexicon.
|
|
1105
|
-
*/
|
|
1106
|
-
export function readSubVenueJSONL(path: string): SubVenueHarvestRow[] {
|
|
1107
|
-
const out: SubVenueHarvestRow[] = []
|
|
1108
|
-
|
|
1109
|
-
// `TextSpliterator` rather than `split("\n")` — a whole-country extract runs to 250,000 lines
|
|
1110
|
-
// (52 MB for Great Britain), and materializing every segment before reading the first is exactly
|
|
1111
|
-
// what the repo lint rule exists to prevent.
|
|
1112
|
-
for (const line of TextSpliterator.from(readFileSync(path, "utf8"))) {
|
|
1113
|
-
const trimmed = line.trim()
|
|
1114
|
-
|
|
1115
|
-
if (!trimmed) continue
|
|
1116
|
-
|
|
1117
|
-
try {
|
|
1118
|
-
out.push(parseJSONStrict<SubVenueHarvestRow>(trimmed))
|
|
1119
|
-
} catch {
|
|
1120
|
-
continue
|
|
1121
|
-
}
|
|
1122
|
-
}
|
|
1123
|
-
|
|
1124
|
-
return out
|
|
1125
|
-
}
|
|
1126
|
-
|
|
1127
308
|
/**
|
|
1128
309
|
* The Wikidata fetch manifest's shape, narrowed to the fields the lexicon copies into `sources[]`.
|
|
1129
310
|
*/
|