@mailwoman/corpus 8.6.0 → 9.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/out/src/adapter.d.ts +41 -0
- package/out/src/adapter.d.ts.map +1 -1
- package/out/src/adapter.js +46 -1
- package/out/src/adapter.js.map +1 -1
- package/out/src/adapters/ban/adapter.d.ts +4 -3
- package/out/src/adapters/ban/adapter.d.ts.map +1 -1
- package/out/src/adapters/ban/adapter.js +60 -67
- package/out/src/adapters/ban/adapter.js.map +1 -1
- package/out/src/adapters/ban/street-decompose.d.ts.map +1 -1
- package/out/src/adapters/ban/street-decompose.js +18 -25
- package/out/src/adapters/ban/street-decompose.js.map +1 -1
- package/out/src/adapters/fcc-bdc/adapter.d.ts +0 -6
- package/out/src/adapters/fcc-bdc/adapter.d.ts.map +1 -1
- package/out/src/adapters/fcc-bdc/adapter.js +2 -24
- package/out/src/adapters/fcc-bdc/adapter.js.map +1 -1
- package/out/src/adapters/geonames/adapter.d.ts.map +1 -1
- package/out/src/adapters/geonames/adapter.js +70 -76
- package/out/src/adapters/geonames/adapter.js.map +1 -1
- package/out/src/adapters/geonames-postal/adapter.d.ts.map +1 -1
- package/out/src/adapters/geonames-postal/adapter.js +44 -49
- package/out/src/adapters/geonames-postal/adapter.js.map +1 -1
- package/out/src/adapters/gnaf/adapter.d.ts.map +1 -1
- package/out/src/adapters/gnaf/adapter.js +4 -7
- package/out/src/adapters/gnaf/adapter.js.map +1 -1
- package/out/src/adapters/gnaf/assemble.d.ts.map +1 -1
- package/out/src/adapters/gnaf/assemble.js +6 -11
- package/out/src/adapters/gnaf/assemble.js.map +1 -1
- package/out/src/adapters/openaddresses/adapter.d.ts.map +1 -1
- package/out/src/adapters/openaddresses/adapter.js +2 -7
- package/out/src/adapters/openaddresses/adapter.js.map +1 -1
- package/out/src/adapters/overture/adapter.d.ts.map +1 -1
- package/out/src/adapters/overture/adapter.js +3 -7
- package/out/src/adapters/overture/adapter.js.map +1 -1
- package/out/src/adapters/state-hi-schools/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-hi-schools/adapter.js +55 -73
- package/out/src/adapters/state-hi-schools/adapter.js.map +1 -1
- package/out/src/adapters/state-ia-contractors/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-ia-contractors/adapter.js +58 -76
- package/out/src/adapters/state-ia-contractors/adapter.js.map +1 -1
- package/out/src/adapters/state-ny-notaries/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-ny-notaries/adapter.js +67 -85
- package/out/src/adapters/state-ny-notaries/adapter.js.map +1 -1
- package/out/src/adapters/state-tx-notaries/adapter.d.ts.map +1 -1
- package/out/src/adapters/state-tx-notaries/adapter.js +69 -87
- package/out/src/adapters/state-tx-notaries/adapter.js.map +1 -1
- package/out/src/adapters/synth-po-box/adapter.d.ts.map +1 -1
- package/out/src/adapters/synth-po-box/adapter.js +7 -15
- package/out/src/adapters/synth-po-box/adapter.js.map +1 -1
- package/out/src/adapters/tiger/street-decompose.d.ts.map +1 -1
- package/out/src/adapters/tiger/street-decompose.js +18 -27
- package/out/src/adapters/tiger/street-decompose.js.map +1 -1
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts +3 -3
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.js +62 -80
- package/out/src/adapters/usgov-hrsa-fqhc/adapter.js.map +1 -1
- package/out/src/adapters/usgov-imls-pls/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-imls-pls/adapter.js +60 -82
- package/out/src/adapters/usgov-imls-pls/adapter.js.map +1 -1
- package/out/src/adapters/usgov-irs-bmf/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-irs-bmf/adapter.js +58 -65
- package/out/src/adapters/usgov-irs-bmf/adapter.js.map +1 -1
- package/out/src/adapters/usgov-nad/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-nad/adapter.js +5 -8
- package/out/src/adapters/usgov-nad/adapter.js.map +1 -1
- package/out/src/adapters/usgov-nppes/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-nppes/adapter.js +58 -76
- package/out/src/adapters/usgov-nppes/adapter.js.map +1 -1
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.d.ts.map +1 -1
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js +51 -71
- package/out/src/adapters/usgov-samhsa-treatment-locator/adapter.js.map +1 -1
- package/out/src/adapters/wof-admin-jp/adapter.d.ts.map +1 -1
- package/out/src/adapters/wof-admin-jp/adapter.js +24 -6
- package/out/src/adapters/wof-admin-jp/adapter.js.map +1 -1
- package/out/src/build.d.ts.map +1 -1
- package/out/src/build.js +2 -1
- package/out/src/build.js.map +1 -1
- package/out/src/golden.d.ts +5 -0
- package/out/src/golden.d.ts.map +1 -1
- package/out/src/golden.js +23 -9
- package/out/src/golden.js.map +1 -1
- package/out/src/parquet-wrapper/reader.d.ts +8 -0
- package/out/src/parquet-wrapper/reader.d.ts.map +1 -1
- package/out/src/parquet-wrapper/reader.js +16 -0
- package/out/src/parquet-wrapper/reader.js.map +1 -1
- package/out/src/parquet-wrapper/schema.d.ts +4 -0
- package/out/src/parquet-wrapper/schema.d.ts.map +1 -1
- package/out/src/parquet-wrapper/schema.js.map +1 -1
- package/out/src/parquet.d.ts.map +1 -1
- package/out/src/parquet.js +2 -13
- package/out/src/parquet.js.map +1 -1
- package/out/src/shard-recipes/anchor-absorption.d.ts.map +1 -1
- package/out/src/shard-recipes/anchor-absorption.js +2 -1
- package/out/src/shard-recipes/anchor-absorption.js.map +1 -1
- package/out/src/shard-recipes/country-balanced.d.ts.map +1 -1
- package/out/src/shard-recipes/country-balanced.js +68 -60
- package/out/src/shard-recipes/country-balanced.js.map +1 -1
- package/out/src/shard-recipes/fr-fragment.d.ts.map +1 -1
- package/out/src/shard-recipes/fr-fragment.js +8 -5
- package/out/src/shard-recipes/fr-fragment.js.map +1 -1
- package/out/src/shard-recipes/fr-order.d.ts.map +1 -1
- package/out/src/shard-recipes/fr-order.js +14 -58
- package/out/src/shard-recipes/fr-order.js.map +1 -1
- package/out/src/shard-recipes/german.d.ts.map +1 -1
- package/out/src/shard-recipes/german.js +11 -58
- package/out/src/shard-recipes/german.js.map +1 -1
- package/out/src/shard-recipes/index.d.ts.map +1 -1
- package/out/src/shard-recipes/index.js +2 -0
- package/out/src/shard-recipes/index.js.map +1 -1
- package/out/src/shard-recipes/intersection.d.ts +2 -2
- package/out/src/shard-recipes/intersection.d.ts.map +1 -1
- package/out/src/shard-recipes/intersection.js +24 -62
- package/out/src/shard-recipes/intersection.js.map +1 -1
- package/out/src/shard-recipes/locale.d.ts.map +1 -1
- package/out/src/shard-recipes/locale.js +14 -16
- package/out/src/shard-recipes/locale.js.map +1 -1
- package/out/src/shard-recipes/no-fragment.d.ts.map +1 -1
- package/out/src/shard-recipes/no-fragment.js +8 -5
- package/out/src/shard-recipes/no-fragment.js.map +1 -1
- package/out/src/shard-recipes/no-street-led.d.ts.map +1 -1
- package/out/src/shard-recipes/no-street-led.js +8 -5
- package/out/src/shard-recipes/no-street-led.js.map +1 -1
- package/out/src/shard-recipes/po-box-cedex.d.ts +3 -3
- package/out/src/shard-recipes/po-box-cedex.d.ts.map +1 -1
- package/out/src/shard-recipes/po-box-cedex.js +51 -94
- package/out/src/shard-recipes/po-box-cedex.js.map +1 -1
- package/out/src/shard-recipes/scaffold.d.ts +59 -9
- package/out/src/shard-recipes/scaffold.d.ts.map +1 -1
- package/out/src/shard-recipes/scaffold.js +51 -30
- package/out/src/shard-recipes/scaffold.js.map +1 -1
- package/out/src/shard-recipes/street-affix.d.ts.map +1 -1
- package/out/src/shard-recipes/street-affix.js +56 -82
- package/out/src/shard-recipes/street-affix.js.map +1 -1
- package/out/src/shard-recipes/sub-venue-sources.d.ts +231 -0
- package/out/src/shard-recipes/sub-venue-sources.d.ts.map +1 -0
- package/out/src/shard-recipes/sub-venue-sources.js +669 -0
- package/out/src/shard-recipes/sub-venue-sources.js.map +1 -0
- package/out/src/shard-recipes/sub-venue.d.ts +269 -0
- package/out/src/shard-recipes/sub-venue.d.ts.map +1 -0
- package/out/src/shard-recipes/sub-venue.js +709 -0
- package/out/src/shard-recipes/sub-venue.js.map +1 -0
- package/out/src/shard-recipes/unit.d.ts +1 -1
- package/out/src/shard-recipes/unit.d.ts.map +1 -1
- package/out/src/shard-recipes/unit.js +22 -64
- package/out/src/shard-recipes/unit.js.map +1 -1
- package/out/src/split.d.ts +7 -0
- package/out/src/split.d.ts.map +1 -1
- package/out/src/split.js +7 -0
- package/out/src/split.js.map +1 -1
- package/out/src/synthesize.d.ts.map +1 -1
- package/out/src/synthesize.js +1 -12
- package/out/src/synthesize.js.map +1 -1
- package/out/src/tools/align-shard.d.ts.map +1 -1
- package/out/src/tools/align-shard.js +3 -7
- package/out/src/tools/align-shard.js.map +1 -1
- package/out/src/tools/audit.d.ts.map +1 -1
- package/out/src/tools/audit.js +4 -4
- package/out/src/tools/audit.js.map +1 -1
- package/out/src/tools/corpus-stats.d.ts +2 -2
- package/out/src/tools/corpus-stats.d.ts.map +1 -1
- package/out/src/tools/corpus-stats.js +18 -31
- package/out/src/tools/corpus-stats.js.map +1 -1
- package/out/src/tools/fetch/download.d.ts.map +1 -1
- package/out/src/tools/fetch/download.js +5 -6
- package/out/src/tools/fetch/download.js.map +1 -1
- package/out/src/tools/fetch/imls-pls.d.ts.map +1 -1
- package/out/src/tools/fetch/imls-pls.js +2 -2
- package/out/src/tools/fetch/imls-pls.js.map +1 -1
- package/out/src/tools/fetch/index.d.ts +14 -1
- package/out/src/tools/fetch/index.d.ts.map +1 -1
- package/out/src/tools/fetch/index.js +14 -1
- package/out/src/tools/fetch/index.js.map +1 -1
- package/out/src/tools/fetch/nad.d.ts +1 -1
- package/out/src/tools/fetch/nad.js +1 -1
- package/out/src/tools/fetch/nppes.d.ts.map +1 -1
- package/out/src/tools/fetch/nppes.js +2 -1
- package/out/src/tools/fetch/nppes.js.map +1 -1
- package/out/src/tools/fetch/openaddresses.d.ts +1 -1
- package/out/src/tools/fetch/openaddresses.js +1 -1
- package/out/src/tools/fetch/ourairports.d.ts +45 -0
- package/out/src/tools/fetch/ourairports.d.ts.map +1 -0
- package/out/src/tools/fetch/ourairports.js +123 -0
- package/out/src/tools/fetch/ourairports.js.map +1 -0
- package/out/src/tools/fetch/wikidata-subvenue.d.ts +171 -0
- package/out/src/tools/fetch/wikidata-subvenue.d.ts.map +1 -0
- package/out/src/tools/fetch/wikidata-subvenue.js +275 -0
- package/out/src/tools/fetch/wikidata-subvenue.js.map +1 -0
- package/out/src/tools/golden-expand.d.ts.map +1 -1
- package/out/src/tools/golden-expand.js +18 -16
- package/out/src/tools/golden-expand.js.map +1 -1
- package/out/src/tools/golden-promote.d.ts.map +1 -1
- package/out/src/tools/golden-promote.js +12 -6
- package/out/src/tools/golden-promote.js.map +1 -1
- package/out/src/tools/golden-relabel-street.d.ts +196 -0
- package/out/src/tools/golden-relabel-street.d.ts.map +1 -0
- package/out/src/tools/golden-relabel-street.js +513 -0
- package/out/src/tools/golden-relabel-street.js.map +1 -0
- package/out/src/tools/index.d.ts +4 -0
- package/out/src/tools/index.d.ts.map +1 -1
- package/out/src/tools/index.js +4 -0
- package/out/src/tools/index.js.map +1 -1
- package/out/src/tools/jsonl-to-parquet.d.ts.map +1 -1
- package/out/src/tools/jsonl-to-parquet.js +3 -2
- package/out/src/tools/jsonl-to-parquet.js.map +1 -1
- package/out/src/tools/lint-shard-vocab.d.ts.map +1 -1
- package/out/src/tools/lint-shard-vocab.js +3 -2
- package/out/src/tools/lint-shard-vocab.js.map +1 -1
- package/out/src/tools/lint-shard.d.ts +3 -3
- package/out/src/tools/lint-shard.d.ts.map +1 -1
- package/out/src/tools/lint-shard.js +23 -32
- package/out/src/tools/lint-shard.js.map +1 -1
- package/out/src/tools/overlay-manifest.d.ts.map +1 -1
- package/out/src/tools/overlay-manifest.js +2 -1
- package/out/src/tools/overlay-manifest.js.map +1 -1
- package/out/src/tools/overture-subvenue.d.ts +112 -0
- package/out/src/tools/overture-subvenue.d.ts.map +1 -0
- package/out/src/tools/overture-subvenue.js +144 -0
- package/out/src/tools/overture-subvenue.js.map +1 -0
- package/out/src/tools/shard-kryptonite.d.ts +1 -1
- package/out/src/tools/shard-kryptonite.d.ts.map +1 -1
- package/out/src/tools/shard-kryptonite.js +5 -4
- package/out/src/tools/shard-kryptonite.js.map +1 -1
- package/out/src/tools/shard-translit.d.ts +6 -3
- package/out/src/tools/shard-translit.d.ts.map +1 -1
- package/out/src/tools/shard-translit.js +8 -6
- package/out/src/tools/shard-translit.js.map +1 -1
- package/out/src/tools/sub-venue-lexicon.d.ts +507 -0
- package/out/src/tools/sub-venue-lexicon.d.ts.map +1 -0
- package/out/src/tools/sub-venue-lexicon.js +817 -0
- package/out/src/tools/sub-venue-lexicon.js.map +1 -0
- package/out/src/tools/sub-venue-promotions.d.ts +94 -0
- package/out/src/tools/sub-venue-promotions.d.ts.map +1 -0
- package/out/src/tools/sub-venue-promotions.js +266 -0
- package/out/src/tools/sub-venue-promotions.js.map +1 -0
- package/out/src/wof-json.d.ts +2 -2
- package/out/src/wof-json.d.ts.map +1 -1
- package/out/src/wof-json.js +5 -8
- package/out/src/wof-json.js.map +1 -1
- package/package.json +8 -8
- package/src/adapter.ts +58 -1
- package/src/adapters/ban/adapter.ts +55 -65
- package/src/adapters/ban/street-decompose.ts +17 -25
- package/src/adapters/fcc-bdc/adapter.ts +2 -33
- package/src/adapters/geonames/adapter.ts +64 -72
- package/src/adapters/geonames-postal/adapter.ts +42 -50
- package/src/adapters/gnaf/adapter.ts +4 -7
- package/src/adapters/gnaf/assemble.ts +6 -10
- package/src/adapters/openaddresses/adapter.ts +2 -7
- package/src/adapters/overture/adapter.ts +3 -6
- package/src/adapters/state-hi-schools/adapter.ts +50 -73
- package/src/adapters/state-ia-contractors/adapter.ts +52 -76
- package/src/adapters/state-ny-notaries/adapter.ts +61 -85
- package/src/adapters/state-tx-notaries/adapter.ts +60 -84
- package/src/adapters/synth-po-box/adapter.ts +7 -17
- package/src/adapters/tiger/street-decompose.ts +21 -31
- package/src/adapters/usgov-hrsa-fqhc/adapter.ts +61 -84
- package/src/adapters/usgov-imls-pls/adapter.ts +54 -82
- package/src/adapters/usgov-irs-bmf/adapter.ts +53 -63
- package/src/adapters/usgov-nad/adapter.ts +5 -8
- package/src/adapters/usgov-nppes/adapter.ts +51 -75
- package/src/adapters/usgov-samhsa-treatment-locator/adapter.ts +45 -71
- package/src/adapters/wof-admin-jp/adapter.ts +34 -8
- package/src/build.ts +2 -1
- package/src/golden.ts +24 -9
- package/src/parquet-wrapper/reader.ts +19 -0
- package/src/parquet-wrapper/schema.ts +4 -0
- package/src/parquet.ts +3 -16
- package/src/shard-recipes/anchor-absorption.ts +2 -1
- package/src/shard-recipes/country-balanced.ts +69 -70
- package/src/shard-recipes/fr-fragment.ts +10 -7
- package/src/shard-recipes/fr-order.ts +13 -63
- package/src/shard-recipes/german.ts +12 -65
- package/src/shard-recipes/index.ts +2 -0
- package/src/shard-recipes/intersection.ts +32 -60
- package/src/shard-recipes/locale.ts +14 -16
- package/src/shard-recipes/no-fragment.ts +10 -7
- package/src/shard-recipes/no-street-led.ts +10 -7
- package/src/shard-recipes/po-box-cedex.ts +68 -119
- package/src/shard-recipes/scaffold.ts +82 -28
- package/src/shard-recipes/street-affix.ts +58 -95
- package/src/shard-recipes/sub-venue-sources.ts +895 -0
- package/src/shard-recipes/sub-venue.ts +982 -0
- package/src/shard-recipes/unit.ts +22 -70
- package/src/split.ts +7 -0
- package/src/synthesize.ts +1 -15
- package/src/tools/align-shard.ts +3 -6
- package/src/tools/audit.ts +6 -5
- package/src/tools/corpus-stats.ts +25 -33
- package/src/tools/fetch/download.ts +7 -5
- package/src/tools/fetch/imls-pls.ts +2 -2
- package/src/tools/fetch/index.ts +14 -1
- package/src/tools/fetch/nad.ts +1 -1
- package/src/tools/fetch/nppes.ts +2 -1
- package/src/tools/fetch/openaddresses.ts +1 -1
- package/src/tools/fetch/ourairports.ts +166 -0
- package/src/tools/fetch/wikidata-subvenue.ts +386 -0
- package/src/tools/golden-expand.ts +18 -13
- package/src/tools/golden-promote.ts +15 -6
- package/src/tools/golden-relabel-street.ts +748 -0
- package/src/tools/index.ts +4 -0
- package/src/tools/jsonl-to-parquet.ts +3 -2
- package/src/tools/lint-shard-vocab.ts +3 -2
- package/src/tools/lint-shard.ts +33 -36
- package/src/tools/overlay-manifest.ts +2 -1
- package/src/tools/overture-subvenue.ts +215 -0
- package/src/tools/shard-kryptonite.ts +5 -4
- package/src/tools/shard-translit.ts +12 -7
- package/src/tools/sub-venue-lexicon.ts +1250 -0
- package/src/tools/sub-venue-promotions.ts +330 -0
- package/src/wof-json.ts +5 -8
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* @author Teffen Ellis, et al.
|
|
5
5
|
*
|
|
6
6
|
* `unit` shard recipe — US secondary-unit coverage (#451, the v0-parity `unit` gap). Onto REAL US
|
|
7
|
-
* OpenAddresses skeletons (cached zips under
|
|
7
|
+
* OpenAddresses skeletons (cached zips under `$MAILWOMAN_DATA_ROOT/oa-cache`) it INJECTS a USPS Pub-28 Appendix
|
|
8
8
|
* C2 secondary-unit designator (the `@mailwoman/codex/us` table), varying the surface form
|
|
9
9
|
* (canonical "Apartment" vs approved "Apt") AND the unit's POSITION (after-street / unit-first /
|
|
10
10
|
* bare / venue-prefixed) per row, so the model learns to RECOGNIZE the designator wherever it
|
|
@@ -25,11 +25,12 @@ import { spawnSync } from "node:child_process"
|
|
|
25
25
|
|
|
26
26
|
import { US_UNIT_DESIGNATOR_PREFERRED_ABBR, type USUnitDesignator } from "@mailwoman/codex/us"
|
|
27
27
|
import type { ComponentTag } from "@mailwoman/core/types"
|
|
28
|
+
import { dataRootPath } from "@mailwoman/core/utils"
|
|
28
29
|
|
|
29
30
|
import { stableSourceID } from "../adapter.ts"
|
|
30
31
|
import { alignRow } from "../align.ts"
|
|
31
32
|
import type { CanonicalRow } from "../types.ts"
|
|
32
|
-
import { makeMulberry32, type ShardRecipe } from "./scaffold.ts"
|
|
33
|
+
import { makeMulberry32, readCSVRecords, type ShardRecipe } from "./scaffold.ts"
|
|
33
34
|
|
|
34
35
|
/**
|
|
35
36
|
* A cached OpenAddresses extract: the zip, the CSV member, and the implied (file-level) region.
|
|
@@ -58,16 +59,20 @@ interface UnitSource {
|
|
|
58
59
|
* state cached; eval is Vermont only (the corpus holdout).
|
|
59
60
|
*/
|
|
60
61
|
const TRAIN_SOURCES: readonly UnitSource[] = [
|
|
61
|
-
{ zip: "
|
|
62
|
-
{ zip: "
|
|
63
|
-
{ zip: "
|
|
64
|
-
{ zip: "
|
|
65
|
-
{ zip: "
|
|
66
|
-
{ zip: "
|
|
67
|
-
{ zip: "
|
|
62
|
+
{ zip: dataRootPath("oa-cache", "us__ca__berkeley.zip"), csv: "us/ca/berkeley.csv", region: "CA" },
|
|
63
|
+
{ zip: dataRootPath("oa-cache", "us__ca__marin.zip"), csv: "us/ca/marin.csv", region: "CA" },
|
|
64
|
+
{ zip: dataRootPath("oa-cache", "us__dc__statewide.zip"), csv: "us/dc/statewide.csv", region: "DC" },
|
|
65
|
+
{ zip: dataRootPath("oa-cache", "us__ia__statewide.zip"), csv: "us/ia/statewide.csv", region: "IA" },
|
|
66
|
+
{ zip: dataRootPath("oa-cache", "us__il__cook.zip"), csv: "us/il/cook.csv", region: "IL" },
|
|
67
|
+
{ zip: dataRootPath("oa-cache", "us__mt__statewide.zip"), csv: "us/mt/statewide.csv", region: "MT" },
|
|
68
|
+
{ zip: dataRootPath("oa-cache", "us__sd__statewide.zip"), csv: "us/sd/statewide.csv", region: "SD" },
|
|
68
69
|
]
|
|
69
70
|
|
|
70
|
-
const EVAL_SOURCE: UnitSource = {
|
|
71
|
+
const EVAL_SOURCE: UnitSource = {
|
|
72
|
+
zip: dataRootPath("oa-cache", "us__vt__statewide.zip"),
|
|
73
|
+
csv: "us/vt/statewide.csv",
|
|
74
|
+
region: "VT",
|
|
75
|
+
}
|
|
71
76
|
|
|
72
77
|
/**
|
|
73
78
|
* USPS Pub-28 C2 designators that take a secondary identifier ("Apt 4B"). Weighted toward the common ones the v0-parity
|
|
@@ -114,44 +119,6 @@ interface UnitTuple {
|
|
|
114
119
|
oaUnit: string
|
|
115
120
|
}
|
|
116
121
|
|
|
117
|
-
/**
|
|
118
|
-
* Minimal RFC-4180-ish splitter (handles quoted fields).
|
|
119
|
-
*/
|
|
120
|
-
function splitCSV(line: string): string[] {
|
|
121
|
-
const out: string[] = []
|
|
122
|
-
let cur = ""
|
|
123
|
-
let inQ = false
|
|
124
|
-
|
|
125
|
-
for (let i = 0; i < line.length; i++) {
|
|
126
|
-
const c = line[i]
|
|
127
|
-
|
|
128
|
-
if (inQ) {
|
|
129
|
-
if (c === '"') {
|
|
130
|
-
if (line[i + 1] === '"') {
|
|
131
|
-
cur += '"'
|
|
132
|
-
|
|
133
|
-
i++
|
|
134
|
-
} else {
|
|
135
|
-
inQ = false
|
|
136
|
-
}
|
|
137
|
-
} else {
|
|
138
|
-
cur += c
|
|
139
|
-
}
|
|
140
|
-
} else if (c === '"') {
|
|
141
|
-
inQ = true
|
|
142
|
-
} else if (c === ",") {
|
|
143
|
-
out.push(cur)
|
|
144
|
-
cur = ""
|
|
145
|
-
} else {
|
|
146
|
-
cur += c
|
|
147
|
-
}
|
|
148
|
-
}
|
|
149
|
-
|
|
150
|
-
out.push(cur)
|
|
151
|
-
|
|
152
|
-
return out
|
|
153
|
-
}
|
|
154
|
-
|
|
155
122
|
/**
|
|
156
123
|
* Stream real US tuples (number/street/city/postcode + the bare OA unit id) out of a cached OA zip.
|
|
157
124
|
*/
|
|
@@ -164,28 +131,13 @@ function readTuples(source: UnitSource): UnitTuple[] {
|
|
|
164
131
|
return []
|
|
165
132
|
}
|
|
166
133
|
|
|
167
|
-
const lines = r.stdout.toString("utf8").split(/\r?\n/)
|
|
168
|
-
|
|
169
|
-
if (lines.length < 2) return []
|
|
170
|
-
const header = splitCSV(lines[0]!).map((h) => h.trim().toLowerCase())
|
|
171
|
-
const idx = (name: string): number => header.indexOf(name)
|
|
172
|
-
|
|
173
|
-
const iNum = idx("number"),
|
|
174
|
-
iStreet = idx("street"),
|
|
175
|
-
iUnit = idx("unit"),
|
|
176
|
-
iCity = idx("city"),
|
|
177
|
-
iPost = idx("postcode")
|
|
178
|
-
|
|
179
|
-
const get = (cells: string[], i: number): string => (i >= 0 && i < cells.length ? (cells[i] ?? "").trim() : "")
|
|
180
134
|
const tuples: UnitTuple[] = []
|
|
181
135
|
const seen = new Set<string>()
|
|
182
136
|
|
|
183
|
-
for (
|
|
184
|
-
|
|
185
|
-
const
|
|
186
|
-
const
|
|
187
|
-
const locality = get(cells, iCity)
|
|
188
|
-
const house_number = get(cells, iNum)
|
|
137
|
+
for (const row of readCSVRecords(r.stdout)) {
|
|
138
|
+
const street = row.street ?? ""
|
|
139
|
+
const locality = row.city ?? ""
|
|
140
|
+
const house_number = row.number ?? ""
|
|
189
141
|
|
|
190
142
|
if (!street || !locality || !house_number) continue
|
|
191
143
|
const key = `${house_number}|${street}|${locality}`.toLowerCase()
|
|
@@ -198,8 +150,8 @@ function readTuples(source: UnitSource): UnitTuple[] {
|
|
|
198
150
|
street,
|
|
199
151
|
locality,
|
|
200
152
|
region: source.region,
|
|
201
|
-
postcode:
|
|
202
|
-
oaUnit:
|
|
153
|
+
postcode: row.postcode ?? "",
|
|
154
|
+
oaUnit: row.unit ?? "",
|
|
203
155
|
})
|
|
204
156
|
}
|
|
205
157
|
|
|
@@ -321,7 +273,7 @@ export const unitRecipe: ShardRecipe = {
|
|
|
321
273
|
}
|
|
322
274
|
|
|
323
275
|
if (!pool.length) {
|
|
324
|
-
throw new Error(
|
|
276
|
+
throw new Error(`No US tuples found — are the cached OA zips present in ${dataRootPath("oa-cache")}?`)
|
|
325
277
|
}
|
|
326
278
|
|
|
327
279
|
let emitted = 0
|
package/src/split.ts
CHANGED
|
@@ -152,6 +152,13 @@ export function splitRows(rows: Iterable<SplitInputRow>, opts: SplitOptions = {}
|
|
|
152
152
|
/**
|
|
153
153
|
* Lightweight deterministic 0..(n-1) bucket from a string id.
|
|
154
154
|
*/
|
|
155
|
+
/**
|
|
156
|
+
* Deterministic bucket for a stable id.
|
|
157
|
+
*
|
|
158
|
+
* Stays on raw `createHash` rather than `sha256Hex` from `@mailwoman/core/utils`: it needs the digest BYTES, and the
|
|
159
|
+
* shared helper returns hex. Re-parsing hex back into bytes to reach the same four octets would cost more than the one
|
|
160
|
+
* line it saves.
|
|
161
|
+
*/
|
|
155
162
|
export function hashBucket(id: string, n: number): number {
|
|
156
163
|
const digest = createHash("sha256").update(id).digest()
|
|
157
164
|
// Read 4 bytes as uint32 to avoid bigint overhead.
|
package/src/synthesize.ts
CHANGED
|
@@ -34,6 +34,7 @@ import {
|
|
|
34
34
|
matchTrailingSuffix,
|
|
35
35
|
} from "@mailwoman/codex/us"
|
|
36
36
|
import type { BIOLabel, ComponentTag } from "@mailwoman/core/types"
|
|
37
|
+
import { mulberry32 } from "@mailwoman/core/utils"
|
|
37
38
|
|
|
38
39
|
import { alignRow, assertSpanInvariants, type ComponentSpan } from "./align.ts"
|
|
39
40
|
import { whitespaceTokenizer, type Tokenizer } from "./tokenize.ts"
|
|
@@ -213,21 +214,6 @@ function hashString(s: string): number {
|
|
|
213
214
|
return h >>> 0
|
|
214
215
|
}
|
|
215
216
|
|
|
216
|
-
/**
|
|
217
|
-
* Mulberry32 — a tiny seeded PRNG. Same seed → same stream → reproducible typos.
|
|
218
|
-
*/
|
|
219
|
-
function mulberry32(seed: number): () => number {
|
|
220
|
-
let a = seed >>> 0
|
|
221
|
-
|
|
222
|
-
return () => {
|
|
223
|
-
a = (a + 0x6d_2b_79_f5) | 0
|
|
224
|
-
let t = Math.imul(a ^ (a >>> 15), 1 | a)
|
|
225
|
-
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t
|
|
226
|
-
|
|
227
|
-
return ((t ^ (t >>> 14)) >>> 0) / 4_294_967_296
|
|
228
|
-
}
|
|
229
|
-
}
|
|
230
|
-
|
|
231
217
|
/**
|
|
232
218
|
* Inject ONE realistic typo — an adjacent-QWERTY-key substitution OR an adjacent-character transposition — into a
|
|
233
219
|
* single alpha name component (street/locality/region…), teaching the model to recover from real-world misspellings
|
package/src/tools/align-shard.ts
CHANGED
|
@@ -27,7 +27,7 @@ import { createWriteStream } from "node:fs"
|
|
|
27
27
|
* --corpus-version 0.5.0
|
|
28
28
|
*/
|
|
29
29
|
import { alignRow } from "@mailwoman/corpus"
|
|
30
|
-
import {
|
|
30
|
+
import { JSONSpliterator } from "spliterator"
|
|
31
31
|
|
|
32
32
|
export interface AlignShardOptions {
|
|
33
33
|
input: string
|
|
@@ -36,16 +36,13 @@ export interface AlignShardOptions {
|
|
|
36
36
|
}
|
|
37
37
|
|
|
38
38
|
export async function alignCanonicalShard(args: AlignShardOptions): Promise<void> {
|
|
39
|
-
// Read phase only — the write path stays on createWriteStream.
|
|
40
|
-
// original tolerance: the `!line.trim()` guard skips blank lines and a trailing CR is valid JSON whitespace.
|
|
39
|
+
// Read phase only — the write path stays on createWriteStream.
|
|
41
40
|
const outStream = createWriteStream(args.output, { encoding: "utf8" })
|
|
42
41
|
let labeled = 0
|
|
43
42
|
let quarantined = 0
|
|
44
43
|
const quarantineReasons: Record<string, number> = {}
|
|
45
44
|
|
|
46
|
-
for await (const
|
|
47
|
-
if (!line.trim()) continue
|
|
48
|
-
const canonical = JSON.parse(line) as Parameters<typeof alignRow>[0]
|
|
45
|
+
for await (const canonical of JSONSpliterator.fromAsync<Parameters<typeof alignRow>[0]>(args.input)) {
|
|
49
46
|
// Stamp the target corpus version so the emitted row's provenance matches the run it joins.
|
|
50
47
|
canonical.corpus_version = args.corpusVersion
|
|
51
48
|
const result = alignRow(canonical)
|
package/src/tools/audit.ts
CHANGED
|
@@ -18,6 +18,9 @@
|
|
|
18
18
|
import { existsSync, readFileSync, readdirSync } from "node:fs"
|
|
19
19
|
import { basename, join } from "node:path"
|
|
20
20
|
|
|
21
|
+
import { parseJSONStrict } from "@mailwoman/core/objects"
|
|
22
|
+
import { TextSpliterator } from "spliterator"
|
|
23
|
+
|
|
21
24
|
/**
|
|
22
25
|
* Share of a shard one source may hold before the mix is flagged as dominated by it.
|
|
23
26
|
*/
|
|
@@ -66,13 +69,11 @@ interface ParsedConfig {
|
|
|
66
69
|
*/
|
|
67
70
|
function parseConfig(configPath: string): ParsedConfig | null {
|
|
68
71
|
if (!existsSync(configPath)) return null
|
|
69
|
-
const text = readFileSync(configPath, "utf8")
|
|
70
|
-
const lines = text.split("\n")
|
|
71
72
|
const weights: Record<string, number> = {}
|
|
72
73
|
let inBlock = false
|
|
73
74
|
let blockIndent = -1
|
|
74
75
|
|
|
75
|
-
for (const raw of
|
|
76
|
+
for (const raw of TextSpliterator.from(readFileSync(configPath))) {
|
|
76
77
|
const sourceWeightsMatch = raw.match(/^([\t ]*)source_weights:\s*$/)
|
|
77
78
|
|
|
78
79
|
if (sourceWeightsMatch) {
|
|
@@ -217,9 +218,9 @@ function manifestScan(corpusDir: string, knownPrefixes: readonly string[]): Shar
|
|
|
217
218
|
|
|
218
219
|
if (!existsSync(manifestPath)) return null
|
|
219
220
|
|
|
220
|
-
const manifest =
|
|
221
|
+
const manifest = parseJSONStrict<{
|
|
221
222
|
shards?: Array<{ split: string; source?: string | null; first_source_id?: string | null }>
|
|
222
|
-
}
|
|
223
|
+
}>(readFileSync(manifestPath, "utf8"))
|
|
223
224
|
|
|
224
225
|
if (!Array.isArray(manifest.shards)) return null
|
|
225
226
|
const bySplit: Record<string, Record<string, number>> = {}
|
|
@@ -32,14 +32,15 @@
|
|
|
32
32
|
*
|
|
33
33
|
* For a quick local-corpus baseline (limited but useful for linter testing): node
|
|
34
34
|
* scripts/build-corpus-stats.ts\
|
|
35
|
-
* --shards /
|
|
35
|
+
* --shards $MAILWOMAN_DATA_ROOT/corpus/versioned/v0.4.0/corpus-v0.4.0/train/\
|
|
36
36
|
* --output /tmp/corpus-stats-local.json
|
|
37
37
|
*/
|
|
38
38
|
|
|
39
|
-
import { execSync } from "node:child_process"
|
|
40
39
|
import { readdirSync, statSync, writeFileSync } from "node:fs"
|
|
41
40
|
import { join } from "node:path"
|
|
42
41
|
|
|
42
|
+
import { ParquetReader } from "../parquet-wrapper/index.ts"
|
|
43
|
+
|
|
43
44
|
const SEP = ""
|
|
44
45
|
const MIN_BIGRAM_COUNT = 2
|
|
45
46
|
|
|
@@ -65,36 +66,28 @@ function discoverShards(shardsArg: string): string[] {
|
|
|
65
66
|
}
|
|
66
67
|
|
|
67
68
|
/**
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
-
*
|
|
69
|
+
* Stream a shard's `tokens`/`labels` columns.
|
|
70
|
+
*
|
|
71
|
+
* `limit` stops the iteration rather than filtering afterwards, so a capped run reads only the row groups it needs.
|
|
71
72
|
*/
|
|
72
|
-
function streamShardRows(
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
tokens_col = t['tokens'].to_pylist()
|
|
80
|
-
labels_col = t['labels'].to_pylist()
|
|
81
|
-
n = min(len(tokens_col), ${limit ?? "len(tokens_col)"})
|
|
82
|
-
for i in range(n):
|
|
83
|
-
sys.stdout.write(json.dumps({"tokens": tokens_col[i], "labels": labels_col[i]}) + "\\n")
|
|
84
|
-
`
|
|
85
|
-
|
|
86
|
-
const buf = execSync(`python3`, { input: py, maxBuffer: 1024 * 1024 * 1024 })
|
|
87
|
-
const rows: Array<{ tokens: string[]; labels: string[] }> = []
|
|
88
|
-
|
|
89
|
-
for (const line of buf.toString("utf8").split("\n")) {
|
|
90
|
-
if (!line) continue
|
|
91
|
-
rows.push(JSON.parse(line))
|
|
92
|
-
}
|
|
73
|
+
async function* streamShardRows(
|
|
74
|
+
shardPath: string,
|
|
75
|
+
limit?: number
|
|
76
|
+
): AsyncIterable<{ tokens: string[]; labels: string[] }> {
|
|
77
|
+
await using reader = await ParquetReader.openFile<{ tokens: string[]; labels: string[] }>(shardPath)
|
|
78
|
+
|
|
79
|
+
let emitted = 0
|
|
93
80
|
|
|
94
|
-
|
|
81
|
+
for await (const row of reader.project("tokens", "labels")) {
|
|
82
|
+
if (limit !== undefined && emitted >= limit) break
|
|
83
|
+
|
|
84
|
+
yield row
|
|
85
|
+
|
|
86
|
+
emitted++
|
|
87
|
+
}
|
|
95
88
|
}
|
|
96
89
|
|
|
97
|
-
export function buildCorpusStats(args: CorpusStatsOptions): void {
|
|
90
|
+
export async function buildCorpusStats(args: CorpusStatsOptions): Promise<void> {
|
|
98
91
|
const shardPaths = discoverShards(args.shardsArg)
|
|
99
92
|
|
|
100
93
|
console.error(`Discovered ${shardPaths.length} parquet shard(s)`)
|
|
@@ -106,11 +99,10 @@ export function buildCorpusStats(args: CorpusStatsOptions): void {
|
|
|
106
99
|
for (const path of shardPaths) {
|
|
107
100
|
console.error(`Reading ${path}...`)
|
|
108
101
|
|
|
109
|
-
const
|
|
110
|
-
totalRows += rows.length
|
|
102
|
+
const before = totalRows
|
|
111
103
|
|
|
112
|
-
for (const
|
|
113
|
-
|
|
104
|
+
for await (const { tokens, labels } of streamShardRows(path, args.limitPerShard)) {
|
|
105
|
+
totalRows++
|
|
114
106
|
|
|
115
107
|
if (tokens.length !== labels.length) continue
|
|
116
108
|
|
|
@@ -143,7 +135,7 @@ export function buildCorpusStats(args: CorpusStatsOptions): void {
|
|
|
143
135
|
}
|
|
144
136
|
|
|
145
137
|
console.error(
|
|
146
|
-
` ${
|
|
138
|
+
` ${totalRows - before} rows; running totals: ${tokenStats.size} unique tokens, ${bigramStats.size} unique bigrams`
|
|
147
139
|
)
|
|
148
140
|
}
|
|
149
141
|
|
|
@@ -12,6 +12,8 @@ import { existsSync } from "node:fs"
|
|
|
12
12
|
import { readFile, writeFile } from "node:fs/promises"
|
|
13
13
|
import { setTimeout as sleep } from "node:timers/promises"
|
|
14
14
|
|
|
15
|
+
import { tryParsingJSON } from "@mailwoman/core/objects"
|
|
16
|
+
|
|
15
17
|
/**
|
|
16
18
|
* Rate limited — retryable, the server is asking us to back off.
|
|
17
19
|
*/
|
|
@@ -127,11 +129,11 @@ export async function downloadToFile(options: DownloadOptions): Promise<{ bytes:
|
|
|
127
129
|
export async function readManifest<T>(path: string): Promise<T | null> {
|
|
128
130
|
if (!existsSync(path)) return null
|
|
129
131
|
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
132
|
+
// A read failure (e.g. the file vanished after the existsSync probe) maps to null like corrupt
|
|
133
|
+
// JSON does; tryParsingJSON returns null for the non-string sentinel.
|
|
134
|
+
const text = await readFile(path, "utf8").catch(() => null)
|
|
135
|
+
|
|
136
|
+
return tryParsingJSON<T>(text)
|
|
135
137
|
}
|
|
136
138
|
|
|
137
139
|
/**
|
|
@@ -30,6 +30,7 @@ import { basename, join } from "node:path"
|
|
|
30
30
|
import { promisify } from "node:util"
|
|
31
31
|
|
|
32
32
|
import { sha256File } from "@mailwoman/core/utils"
|
|
33
|
+
import { TextSpliterator } from "spliterator"
|
|
33
34
|
|
|
34
35
|
import type { BaseFetchOptions, FetchSummary } from "./download.ts"
|
|
35
36
|
import { downloadToFile, readManifest, writeManifest } from "./download.ts"
|
|
@@ -64,8 +65,7 @@ interface SourceManifest {
|
|
|
64
65
|
async function listZipEntries(zipPath: string): Promise<string[]> {
|
|
65
66
|
const listing = await execFileAsync("unzip", ["-l", zipPath])
|
|
66
67
|
|
|
67
|
-
return listing.stdout
|
|
68
|
-
.split("\n")
|
|
68
|
+
return [...TextSpliterator.from(listing.stdout)]
|
|
69
69
|
.map((line) => line.trim().split(/\s+/).pop() ?? "")
|
|
70
70
|
.filter((name) => name.length > 0)
|
|
71
71
|
}
|
package/src/tools/fetch/index.ts
CHANGED
|
@@ -40,6 +40,13 @@
|
|
|
40
40
|
* Tier A (US PD).
|
|
41
41
|
* - `openaddresses` — OpenAddresses country collections (default: Canada / `ca`). Tier B/C mixed
|
|
42
42
|
* — per-row filter.
|
|
43
|
+
* - `ourairports` — OurAirports global airport CSVs (~83K airports + the country/region/runway
|
|
44
|
+
* joins). Tier A (public domain). The VENUE half of the sub-venue arc (#35); it carries no
|
|
45
|
+
* interior structure, which comes from `@mailwoman/osm/sdk`'s `extractOSMSubVenues`.
|
|
46
|
+
* - `wikidata-subvenue` — the multilingual sub-venue DESIGNATOR vocabulary, pulled as the labels
|
|
47
|
+
* and aliases of eight Wikidata concepts (terminal, gate, concourse, campus, …) plus the
|
|
48
|
+
* airport-terminal instance labels as attested usage. Tier A (CC0). The only module in this
|
|
49
|
+
* family built on `APIClient` — see its docstring for why.
|
|
43
50
|
* - `state-sources` — NY/TX/DE/OR notaries, IA contractors, WA health providers, HI lobbyists.
|
|
44
51
|
* Tier A (state PD-equivalent).
|
|
45
52
|
* - `state-hi-schools` — Hawaii DOE school directory (XLSX → CSV via openpyxl). Tier A (state
|
|
@@ -65,7 +72,7 @@
|
|
|
65
72
|
*
|
|
66
73
|
* # Download Canada (~2 GiB compressed, ~7 GiB uncompressed)
|
|
67
74
|
* mailwoman corpus fetch openaddresses --country ca \
|
|
68
|
-
* --out-root /
|
|
75
|
+
* --out-root $MAILWOMAN_DATA_ROOT/corpus/sources
|
|
69
76
|
*
|
|
70
77
|
* # Or any other OA country code
|
|
71
78
|
* mailwoman corpus fetch openaddresses --country fr
|
|
@@ -91,9 +98,11 @@ import { fetchIMLSPLS } from "./imls-pls.ts"
|
|
|
91
98
|
import { fetchNAD } from "./nad.ts"
|
|
92
99
|
import { fetchNPPES } from "./nppes.ts"
|
|
93
100
|
import { fetchOpenAddresses } from "./openaddresses.ts"
|
|
101
|
+
import { fetchOurAirports } from "./ourairports.ts"
|
|
94
102
|
import { fetchStateHISchools } from "./state-hi-schools.ts"
|
|
95
103
|
import { fetchStateSources } from "./state-sources.ts"
|
|
96
104
|
import { fetchTigerFull } from "./tiger-full.ts"
|
|
105
|
+
import { fetchWikidataSubVenue } from "./wikidata-subvenue.ts"
|
|
97
106
|
|
|
98
107
|
export * from "./ban.ts"
|
|
99
108
|
export * from "./hrsa.ts"
|
|
@@ -101,9 +110,11 @@ export * from "./imls-pls.ts"
|
|
|
101
110
|
export * from "./nad.ts"
|
|
102
111
|
export * from "./nppes.ts"
|
|
103
112
|
export * from "./openaddresses.ts"
|
|
113
|
+
export * from "./ourairports.ts"
|
|
104
114
|
export * from "./state-hi-schools.ts"
|
|
105
115
|
export * from "./state-sources.ts"
|
|
106
116
|
export * from "./tiger-full.ts"
|
|
117
|
+
export * from "./wikidata-subvenue.ts"
|
|
107
118
|
|
|
108
119
|
/**
|
|
109
120
|
* The fetch-source registry: id → module entry point. Each entry point takes its own options interface.
|
|
@@ -115,9 +126,11 @@ export const FETCH_SOURCES = {
|
|
|
115
126
|
"imls-pls": fetchIMLSPLS,
|
|
116
127
|
nppes: fetchNPPES,
|
|
117
128
|
openaddresses: fetchOpenAddresses,
|
|
129
|
+
ourairports: fetchOurAirports,
|
|
118
130
|
"state-sources": fetchStateSources,
|
|
119
131
|
"state-hi-schools": fetchStateHISchools,
|
|
120
132
|
"tiger-full": fetchTigerFull,
|
|
133
|
+
"wikidata-subvenue": fetchWikidataSubVenue,
|
|
121
134
|
} as const
|
|
122
135
|
|
|
123
136
|
export type FetchSourceID = keyof typeof FETCH_SOURCES
|
package/src/tools/fetch/nad.ts
CHANGED
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
* ## Usage
|
|
27
27
|
*
|
|
28
28
|
* ```sh
|
|
29
|
-
* mailwoman corpus fetch nad --out-root /
|
|
29
|
+
* mailwoman corpus fetch nad --out-root $MAILWOMAN_DATA_ROOT/corpus/sources
|
|
30
30
|
*
|
|
31
31
|
* # Resume from an OID
|
|
32
32
|
* mailwoman corpus fetch nad --start-oid 34400001
|
package/src/tools/fetch/nppes.ts
CHANGED
|
@@ -31,6 +31,7 @@ import { join } from "node:path"
|
|
|
31
31
|
import { promisify } from "node:util"
|
|
32
32
|
|
|
33
33
|
import { sha256File } from "@mailwoman/core/utils"
|
|
34
|
+
import { TextSpliterator } from "spliterator"
|
|
34
35
|
|
|
35
36
|
import type { BaseFetchOptions, FetchSummary } from "./download.ts"
|
|
36
37
|
import { downloadToFile, readManifest, writeManifest } from "./download.ts"
|
|
@@ -79,7 +80,7 @@ async function discoverLatestZip(): Promise<string | undefined> {
|
|
|
79
80
|
async function findNpidataCSV(zipPath: string): Promise<string | undefined> {
|
|
80
81
|
const listing = await execFileAsync("unzip", ["-l", zipPath])
|
|
81
82
|
|
|
82
|
-
for (const line of listing.stdout
|
|
83
|
+
for (const line of TextSpliterator.from(listing.stdout)) {
|
|
83
84
|
const match = /npidata_pfile\S+\.csv/i.exec(line)
|
|
84
85
|
|
|
85
86
|
if (match?.[0]) return match[0]
|
|
@@ -44,7 +44,7 @@
|
|
|
44
44
|
* ```sh
|
|
45
45
|
* # With token (preferred). Default country: ca. Supports any OA country code (us-west, fr, …)
|
|
46
46
|
* OA_BATCH_TOKEN=<token> mailwoman corpus fetch openaddresses --country ca \
|
|
47
|
-
* --out-root /
|
|
47
|
+
* --out-root $MAILWOMAN_DATA_ROOT/corpus/sources
|
|
48
48
|
*
|
|
49
49
|
* # Without token (will detect + print instructions, then report the failure):
|
|
50
50
|
* mailwoman corpus fetch openaddresses --country ca
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @copyright Sister Software
|
|
3
|
+
* @license AGPL-3.0
|
|
4
|
+
* @author Teffen Ellis, et al.
|
|
5
|
+
*
|
|
6
|
+
* Fetch the OurAirports CSV dumps — the VENUE side of the sub-venue corpus arc (#35).
|
|
7
|
+
*
|
|
8
|
+
* Source : https://davidmegginson.github.io/ourairports-data/ (the project's own GitHub Pages
|
|
9
|
+
* mirror of the nightly export; `ourairports.com/data/` redirects here).
|
|
10
|
+
* License: PUBLIC DOMAIN. OurAirports places its data in the public domain and asks only for a
|
|
11
|
+
* courtesy credit — no attribution obligation rides on a derived shard, which makes this
|
|
12
|
+
* the one transport source in the arc with no licensing question at all. Tier A.
|
|
13
|
+
*
|
|
14
|
+
* ## What it is good for, and what it is not
|
|
15
|
+
*
|
|
16
|
+
* `airports.csv` is every airport on earth with ICAO/IATA codes, coordinates, `municipality`, and
|
|
17
|
+
* `iso_country` — 12.7 MB, ~83,000 rows as of 2026-08-04. That is the CONTAINING VENUE for a
|
|
18
|
+
* `<sub-venue>, <venue>, <street>, <locality>, <postcode>` corpus line, and it is better at that job
|
|
19
|
+
* than OSM: every row is named, the name is canonical, and `municipality` gives the locality without
|
|
20
|
+
* a spatial join.
|
|
21
|
+
*
|
|
22
|
+
* It carries NO interior structure. There is no terminal, concourse, gate or pier table — the
|
|
23
|
+
* corpus task says as much ("Good for the venue side of each pair, weaker on interior structure")
|
|
24
|
+
* and a row-level read confirms it. Pair it with the OSM `aeroway` extractor
|
|
25
|
+
* (`@mailwoman/osm/sdk`'s `extractOSMSubVenues`), which is where the sub-venue half comes from.
|
|
26
|
+
*
|
|
27
|
+
* ## Why `downloadToFile` and not `APIClient`
|
|
28
|
+
*
|
|
29
|
+
* `AGENTS.md` routes HTTP through `APIClient`, and that rule is about API REQUESTS — small bodies,
|
|
30
|
+
* repeated calls, rate-limited hosts. This is four static file transfers against a GitHub Pages CDN
|
|
31
|
+
* with no rate limit and nothing to pace, run once per refresh. It uses the same `downloadToFile`
|
|
32
|
+
* every other module in this `fetch/` family uses, which is where the retry and timeout live.
|
|
33
|
+
* The Wikidata sibling (`wikidata-subvenue.ts`) IS an API client and is built on `APIClient`
|
|
34
|
+
* accordingly; the split between the two is the one `AGENTS.md` draws.
|
|
35
|
+
*
|
|
36
|
+
* Invoke via `mailwoman corpus fetch ourairports --out-root <path>`.
|
|
37
|
+
*/
|
|
38
|
+
|
|
39
|
+
import { mkdirSync } from "node:fs"
|
|
40
|
+
import { join } from "node:path"
|
|
41
|
+
|
|
42
|
+
import { sha256File } from "@mailwoman/core/utils"
|
|
43
|
+
|
|
44
|
+
import type { BaseFetchOptions, FetchSummary } from "./download.ts"
|
|
45
|
+
import { downloadToFile, writeManifest } from "./download.ts"
|
|
46
|
+
|
|
47
|
+
const SLUG = "ourairports"
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* The GitHub Pages mirror the project itself publishes. `ourairports.com/data/*.csv` 302s here, so pointing at the
|
|
51
|
+
* mirror directly saves a redirect and is the URL the project's own README gives.
|
|
52
|
+
*/
|
|
53
|
+
const BASE_URL = "https://davidmegginson.github.io/ourairports-data"
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* The files worth having, and why each one.
|
|
57
|
+
*
|
|
58
|
+
* `airports.csv` is the payload. The other three are small joins that turn its codes into text: `countries.csv` and
|
|
59
|
+
* `regions.csv` expand `iso_country`/`iso_region` into names (a corpus line needs "Germany", not "DE"), and
|
|
60
|
+
* `runways.csv` is the only file carrying per-airport sub-structure of any kind — runway designators, which are NOT
|
|
61
|
+
* sub-venue designators (nobody addresses mail to a runway) but are worth having on disk as the negative class if the
|
|
62
|
+
* shard ever needs one.
|
|
63
|
+
*/
|
|
64
|
+
const FILES = ["airports.csv", "countries.csv", "regions.csv", "runways.csv"] as const
|
|
65
|
+
|
|
66
|
+
export type FetchOurAirportsOptions = BaseFetchOptions
|
|
67
|
+
|
|
68
|
+
interface OurAirportsFileEntry {
|
|
69
|
+
filename: string
|
|
70
|
+
source_url: string
|
|
71
|
+
sha256: string
|
|
72
|
+
bytes: number
|
|
73
|
+
/**
|
|
74
|
+
* The upstream `Last-Modified`, when the CDN gave one. This is the DATA's vintage; `downloaded_at` is only when we
|
|
75
|
+
* asked. `corpus/AGENTS.md` has the standing warning that a file's mtime is not its data's vintage — recording the
|
|
76
|
+
* upstream header is how a later refresh decision gets made on the right number.
|
|
77
|
+
*/
|
|
78
|
+
last_modified: string | null
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
interface OurAirportsManifest {
|
|
82
|
+
source: string
|
|
83
|
+
base_url: string
|
|
84
|
+
license: string
|
|
85
|
+
downloaded_at: string
|
|
86
|
+
files: OurAirportsFileEntry[]
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Read the upstream `Last-Modified` with a HEAD. Returns `null` on any failure — provenance metadata is nice to have
|
|
91
|
+
* and must never fail a download that otherwise succeeded.
|
|
92
|
+
*/
|
|
93
|
+
async function readLastModified(url: string): Promise<string | null> {
|
|
94
|
+
try {
|
|
95
|
+
const res = await fetch(url, { method: "HEAD", signal: AbortSignal.timeout(30_000) })
|
|
96
|
+
|
|
97
|
+
return res.ok ? res.headers.get("last-modified") : null
|
|
98
|
+
} catch {
|
|
99
|
+
return null
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Download the OurAirports CSVs into `<outRoot>/ourairports/`, with a sibling `MANIFEST.json` carrying each file's
|
|
105
|
+
* origin URL, sha256, byte count and upstream `Last-Modified`.
|
|
106
|
+
*/
|
|
107
|
+
export async function fetchOurAirports(
|
|
108
|
+
options: FetchOurAirportsOptions,
|
|
109
|
+
report?: (line: string) => void
|
|
110
|
+
): Promise<FetchSummary> {
|
|
111
|
+
const destDir = join(options.outRoot, SLUG)
|
|
112
|
+
mkdirSync(destDir, { recursive: true })
|
|
113
|
+
|
|
114
|
+
const entries: OurAirportsFileEntry[] = []
|
|
115
|
+
const failedCodes: string[] = []
|
|
116
|
+
let fetched = 0
|
|
117
|
+
let failed = 0
|
|
118
|
+
|
|
119
|
+
for (const filename of FILES) {
|
|
120
|
+
const url = `${BASE_URL}/${filename}`
|
|
121
|
+
const dest = join(destDir, filename)
|
|
122
|
+
|
|
123
|
+
report?.(`=== ${SLUG} / ${filename}`)
|
|
124
|
+
|
|
125
|
+
try {
|
|
126
|
+
const [{ bytes }, lastModified] = await Promise.all([
|
|
127
|
+
downloadToFile({
|
|
128
|
+
url,
|
|
129
|
+
dest,
|
|
130
|
+
timeoutMs: 600_000,
|
|
131
|
+
retries: 2,
|
|
132
|
+
headers: { "Accept-Encoding": "gzip, br" },
|
|
133
|
+
report,
|
|
134
|
+
}),
|
|
135
|
+
readLastModified(url),
|
|
136
|
+
])
|
|
137
|
+
|
|
138
|
+
entries.push({
|
|
139
|
+
filename,
|
|
140
|
+
source_url: url,
|
|
141
|
+
sha256: await sha256File(dest),
|
|
142
|
+
bytes,
|
|
143
|
+
last_modified: lastModified,
|
|
144
|
+
})
|
|
145
|
+
|
|
146
|
+
fetched++
|
|
147
|
+
} catch (error) {
|
|
148
|
+
report?.(`✗ ${filename}: ${error instanceof Error ? error.message : String(error)}`)
|
|
149
|
+
failedCodes.push(filename)
|
|
150
|
+
|
|
151
|
+
failed++
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const manifest: OurAirportsManifest = {
|
|
156
|
+
source: "OurAirports",
|
|
157
|
+
base_url: BASE_URL,
|
|
158
|
+
license: "public domain (courtesy credit requested)",
|
|
159
|
+
downloaded_at: new Date().toISOString(),
|
|
160
|
+
files: entries,
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
await writeManifest(join(destDir, "MANIFEST.json"), manifest)
|
|
164
|
+
|
|
165
|
+
return { fetched, skipped: 0, failed, failedCodes }
|
|
166
|
+
}
|