@finbheara/names 0.14.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,40 +1,86 @@
1
1
  {
2
- "about": "Bloom filters the phrase classifier consults. Each table: s = bits, h = hashes, d = seed, c = base64 bit content, n = items. Keys are lower-cased, accent-stripped, apostrophe-free. Built 2026-01-03, target false-positive rate 0.01. A table holds hashes of its source list, not the list; sources are recorded for rebuilding.",
2
+ "about": "Bloom filters the phrase classifier consults. Each table: s = bits, h = hashes, d = seed, c = base64 bit content, n = items. A table holds hashes of its source lists, not the lists; the lists are third-party files, recorded here for rebuilding. The four name tables are rebuilt by scripts/build-blooms.ts from the files named in their inputs; locations, us-locations and schools were built 2026-01-03 by a process that is not recorded, and are not rebuilt.",
3
+ "build": {
4
+ "script": "scripts/build-blooms.ts <source-dir> [--check]",
5
+ "seed": 78187493520,
6
+ "fpp": 0.01,
7
+ "sizing": "BloomFilter.create(n, fpp) from bloom-filters: s = ceil(-n ln(fpp) / ln(2)^2), h = ceil(s/n ln 2); n is the number of distinct keys",
8
+ "key": "normalize() in src/classifier/bf/filters.ts: lower-case, NFD, combining marks U+0300-U+036F removed, the ASCII apostrophe removed (U+2019 is kept). An empty key is dropped; a key repeated within or across inputs counts once",
9
+ "read": "a leading U+FEFF is stripped; lines split on LF or CRLF; format csv takes cell `column` after splitting the line on every comma (no quoting); header: the first line is skipped; trim: surrounding whitespace is removed before the key is taken"
10
+ },
3
11
  "tables": {
4
12
  "surnames.json": {
5
13
  "items": 151671,
6
- "source": "FiveThirtyEight most-common-name (surnames)",
7
- "url": "https://github.com/fivethirtyeight/data/tree/master/most-common-name"
14
+ "inputs": [
15
+ {
16
+ "file": "most-common-name_surnames.csv",
17
+ "format": "csv",
18
+ "column": 0,
19
+ "header": true,
20
+ "source": "FiveThirtyEight most-common-name (surnames)",
21
+ "url": "https://github.com/fivethirtyeight/data/tree/master/most-common-name"
22
+ }
23
+ ]
8
24
  },
9
25
  "human-names.json": {
10
26
  "items": 197226,
11
- "source": "human-names-list.txt (upstream not recorded)",
12
- "url": null
27
+ "inputs": [
28
+ {
29
+ "file": "human-names-list.txt",
30
+ "format": "lines",
31
+ "source": "human-names-list.txt (upstream not recorded)",
32
+ "url": null
33
+ }
34
+ ]
13
35
  },
14
36
  "first-names.json": {
15
- "items": 133749,
16
- "source": "name_gender_dataset.csv (upstream not recorded)",
17
- "url": null
37
+ "items": 134399,
38
+ "inputs": [
39
+ {
40
+ "file": "name_gender_dataset.csv",
41
+ "format": "csv",
42
+ "column": 0,
43
+ "header": true,
44
+ "source": "Gender by Name, UCI Machine Learning Repository dataset 591 (doi:10.24432/C55G7X), CC BY 4.0; the file inside gender+by+name.zip",
45
+ "url": "https://archive.ics.uci.edu/dataset/591/gender+by+name"
46
+ },
47
+ {
48
+ "file": "diminutive-names.txt",
49
+ "format": "lines",
50
+ "source": "given-name diminutives and familiar forms, 3,511 spellings as printed, one per line, merged from: this package's given-diminutive table; diminutives.db (data public domain) at commit caae757f1f91bce2175a49942a6c466846ade50c; Behind the Name names tagged \"diminutives\", fetched 2026-10-05; the familiar-form pairs mined in this repository's issue #59. Non-western names are kept: the table answers only whether a word is a given name",
51
+ "url": null
52
+ }
53
+ ]
18
54
  },
19
55
  "irish-surnames.json": {
20
56
  "items": 1624,
21
- "source": "Gaois sloinnte (Irish surnames database)",
22
- "url": "https://github.com/gaois/sloinnte"
57
+ "inputs": [
58
+ {
59
+ "file": "irish_surnames.txt",
60
+ "format": "lines",
61
+ "trim": true,
62
+ "source": "Gaois sloinnte (Irish surnames database)",
63
+ "url": "https://github.com/gaois/sloinnte"
64
+ }
65
+ ]
23
66
  },
24
67
  "locations.json": {
25
68
  "items": 8527,
26
69
  "source": "ISO 3166-1 and 3166-2; UN population centres; Wikipedia US cities; urban areas; FIPS; names.txt G entries; common Irish/UK/AU/NZ terms",
27
- "url": null
70
+ "url": null,
71
+ "rebuild": "not reproducible: most of its source lists and its key rules are not recorded. Checked 2026-10-06: country names, official names and ISO alpha-2/alpha-3 codes from a public country list (geo_country.csv) are held (249/250, 248/250, 241/243, 242/243 distinct keys), but these are a small part of 8,527 keys"
28
72
  },
29
73
  "us-locations.json": {
30
74
  "items": 23922,
31
75
  "source": "CDC PLACES county and place data; US states and territories; common US regions",
32
- "url": "https://www.cdc.gov/places/"
76
+ "url": "https://www.cdc.gov/places/",
77
+ "rebuild": "not reproducible: the states, territories and regions lists are not recorded. Checked 2026-10-06 against PLACES__County_Data__GIS_Friendly_Format___2020_release.csv and PLACES__Place_Data__GIS_Friendly_Format___2025_release.csv: every distinct key is held (1,839 county names, the same with ' County', 19,837 place names (LocationName), 51 state names, 51 state abbreviations), together 22,347 of the table's 23,922 keys; the other 1,575 are unaccounted for"
33
78
  },
34
79
  "schools.json": {
35
80
  "items": 17342,
36
81
  "source": "Irish dance school names: a curated school list and the school cells of a results corpus",
37
- "url": null
82
+ "url": null,
83
+ "rebuild": "not reproducible: the sources are not public"
38
84
  }
39
85
  }
40
86
  }