@finbheara/names 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/data/bloom/MANIFEST.json +59 -13
- package/dist/chunks/{index-r5fp3znj.js → index-2565cn5x.js} +5 -5
- package/dist/chunks/{index-r5fp3znj.js.map → index-2565cn5x.js.map} +3 -3
- package/dist/chunks/{index-ztpxgcs1.js → index-a15g2gp6.js} +2 -2
- package/dist/chunks/{index-yf904dam.js → index-ka8nzreg.js} +6 -2
- package/dist/chunks/{index-yf904dam.js.map → index-ka8nzreg.js.map} +3 -3
- package/dist/chunks/{index-wq6kn9jq.js → index-kw69ry3j.js} +213 -42
- package/dist/chunks/{index-wq6kn9jq.js.map → index-kw69ry3j.js.map} +4 -4
- package/dist/chunks/{phrase-672g18kw.js → phrase-2nmvkrc5.js} +2 -2
- package/dist/classifier/index.js +1 -1
- package/dist/cli/census.js +3 -3
- package/dist/index.js +4 -4
- package/dist/measure/index.js +2 -2
- package/dist/normalize/index.js +2 -2
- package/dist/types/classifier/bf/filters.d.ts +8 -3
- package/dist/types/normalize/given-diminutives.d.ts +29 -18
- package/dist/types/normalize/name-kinship.d.ts +6 -0
- package/docs/NAME-NORMALIZATION.md +31 -6
- package/package.json +2 -1
- /package/dist/chunks/{index-ztpxgcs1.js.map → index-a15g2gp6.js.map} +0 -0
- /package/dist/chunks/{phrase-672g18kw.js.map → phrase-2nmvkrc5.js.map} +0 -0
package/README.md
CHANGED
|
@@ -154,12 +154,15 @@ nameKinship(personName("Mary Larson"), personName("Mary Conley-Larson")).via; /
|
|
|
154
154
|
nameKinship(personName("Mary Larson"), personName("Mary Carlson")).reason; // "surname-differs"
|
|
155
155
|
nameKinship(personName("Kate M Smith"), personName("Katherine Conley-Smith")).via;
|
|
156
156
|
// ["surname-part", "given-family", "middle-absent"]
|
|
157
|
+
nameKinship(personName("Connor Smith"), personName("Conor Smith")).via; // ["given-variant"]
|
|
158
|
+
nameKinship(personName("Evie Smith"), personName("Genevieve Smith"));
|
|
159
|
+
// { relation: "candidate", via: ["given-family-not-gate"], reason: "middles-equal", gateSafe: false }
|
|
157
160
|
|
|
158
161
|
// the given-name tolerance is a list of providers, asked in order
|
|
159
162
|
const withMaybe = createNameKinship({
|
|
160
|
-
given: [givenFamilyTable, (a, b) => (a === "
|
|
163
|
+
given: [givenFamilyTable, (a, b) => (a === "NAIMH" && b === "NIAMH" ? "given-maybe" : null)],
|
|
161
164
|
});
|
|
162
|
-
withMaybe(personName("
|
|
165
|
+
withMaybe(personName("Naimh Smith"), personName("Niamh Smith"));
|
|
163
166
|
// { relation: "candidate", via: ["given-maybe"], reason: "middles-equal", gateSafe: false }
|
|
164
167
|
|
|
165
168
|
const index = createNameIndex<number>("nn3"); // a NameMap with a tolerant lookup
|
package/data/bloom/MANIFEST.json
CHANGED
|
@@ -1,40 +1,86 @@
|
|
|
1
1
|
{
|
|
2
|
-
"about": "Bloom filters the phrase classifier consults. Each table: s = bits, h = hashes, d = seed, c = base64 bit content, n = items.
|
|
2
|
+
"about": "Bloom filters the phrase classifier consults. Each table: s = bits, h = hashes, d = seed, c = base64 bit content, n = items. A table holds hashes of its source lists, not the lists; the lists are third-party files, recorded here for rebuilding. The four name tables are rebuilt by scripts/build-blooms.ts from the files named in their inputs; locations, us-locations and schools were built 2026-01-03 by a process that is not recorded, and are not rebuilt.",
|
|
3
|
+
"build": {
|
|
4
|
+
"script": "scripts/build-blooms.ts <source-dir> [--check]",
|
|
5
|
+
"seed": 78187493520,
|
|
6
|
+
"fpp": 0.01,
|
|
7
|
+
"sizing": "BloomFilter.create(n, fpp) from bloom-filters: s = ceil(-n ln(fpp) / ln(2)^2), h = ceil(s/n ln 2); n is the number of distinct keys",
|
|
8
|
+
"key": "normalize() in src/classifier/bf/filters.ts: lower-case, NFD, combining marks U+0300-U+036F removed, the ASCII apostrophe removed (U+2019 is kept). An empty key is dropped; a key repeated within or across inputs counts once",
|
|
9
|
+
"read": "a leading U+FEFF is stripped; lines split on LF or CRLF; format csv takes cell `column` after splitting the line on every comma (no quoting); header: the first line is skipped; trim: surrounding whitespace is removed before the key is taken"
|
|
10
|
+
},
|
|
3
11
|
"tables": {
|
|
4
12
|
"surnames.json": {
|
|
5
13
|
"items": 151671,
|
|
6
|
-
"
|
|
7
|
-
|
|
14
|
+
"inputs": [
|
|
15
|
+
{
|
|
16
|
+
"file": "most-common-name_surnames.csv",
|
|
17
|
+
"format": "csv",
|
|
18
|
+
"column": 0,
|
|
19
|
+
"header": true,
|
|
20
|
+
"source": "FiveThirtyEight most-common-name (surnames)",
|
|
21
|
+
"url": "https://github.com/fivethirtyeight/data/tree/master/most-common-name"
|
|
22
|
+
}
|
|
23
|
+
]
|
|
8
24
|
},
|
|
9
25
|
"human-names.json": {
|
|
10
26
|
"items": 197226,
|
|
11
|
-
"
|
|
12
|
-
|
|
27
|
+
"inputs": [
|
|
28
|
+
{
|
|
29
|
+
"file": "human-names-list.txt",
|
|
30
|
+
"format": "lines",
|
|
31
|
+
"source": "human-names-list.txt (upstream not recorded)",
|
|
32
|
+
"url": null
|
|
33
|
+
}
|
|
34
|
+
]
|
|
13
35
|
},
|
|
14
36
|
"first-names.json": {
|
|
15
|
-
"items":
|
|
16
|
-
"
|
|
17
|
-
|
|
37
|
+
"items": 134399,
|
|
38
|
+
"inputs": [
|
|
39
|
+
{
|
|
40
|
+
"file": "name_gender_dataset.csv",
|
|
41
|
+
"format": "csv",
|
|
42
|
+
"column": 0,
|
|
43
|
+
"header": true,
|
|
44
|
+
"source": "Gender by Name, UCI Machine Learning Repository dataset 591 (doi:10.24432/C55G7X), CC BY 4.0; the file inside gender+by+name.zip",
|
|
45
|
+
"url": "https://archive.ics.uci.edu/dataset/591/gender+by+name"
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"file": "diminutive-names.txt",
|
|
49
|
+
"format": "lines",
|
|
50
|
+
"source": "given-name diminutives and familiar forms, 3,511 spellings as printed, one per line, merged from: this package's given-diminutive table; diminutives.db (data public domain) at commit caae757f1f91bce2175a49942a6c466846ade50c; Behind the Name names tagged \"diminutives\", fetched 2026-10-05; the familiar-form pairs mined in this repository's issue #59. Non-western names are kept: the table answers only whether a word is a given name",
|
|
51
|
+
"url": null
|
|
52
|
+
}
|
|
53
|
+
]
|
|
18
54
|
},
|
|
19
55
|
"irish-surnames.json": {
|
|
20
56
|
"items": 1624,
|
|
21
|
-
"
|
|
22
|
-
|
|
57
|
+
"inputs": [
|
|
58
|
+
{
|
|
59
|
+
"file": "irish_surnames.txt",
|
|
60
|
+
"format": "lines",
|
|
61
|
+
"trim": true,
|
|
62
|
+
"source": "Gaois sloinnte (Irish surnames database)",
|
|
63
|
+
"url": "https://github.com/gaois/sloinnte"
|
|
64
|
+
}
|
|
65
|
+
]
|
|
23
66
|
},
|
|
24
67
|
"locations.json": {
|
|
25
68
|
"items": 8527,
|
|
26
69
|
"source": "ISO 3166-1 and 3166-2; UN population centres; Wikipedia US cities; urban areas; FIPS; names.txt G entries; common Irish/UK/AU/NZ terms",
|
|
27
|
-
"url": null
|
|
70
|
+
"url": null,
|
|
71
|
+
"rebuild": "not reproducible: most of its source lists and its key rules are not recorded. Checked 2026-10-06: country names, official names and ISO alpha-2/alpha-3 codes from a public country list (geo_country.csv) are held (249/250, 248/250, 241/243, 242/243 distinct keys), but these are a small part of 8,527 keys"
|
|
28
72
|
},
|
|
29
73
|
"us-locations.json": {
|
|
30
74
|
"items": 23922,
|
|
31
75
|
"source": "CDC PLACES county and place data; US states and territories; common US regions",
|
|
32
|
-
"url": "https://www.cdc.gov/places/"
|
|
76
|
+
"url": "https://www.cdc.gov/places/",
|
|
77
|
+
"rebuild": "not reproducible: the states, territories and regions lists are not recorded. Checked 2026-10-06 against PLACES__County_Data__GIS_Friendly_Format___2020_release.csv and PLACES__Place_Data__GIS_Friendly_Format___2025_release.csv: every distinct key is held (1,839 county names, the same with ' County', 19,837 place names (LocationName), 51 state names, 51 state abbreviations), together 22,347 of the table's 23,922 keys; the other 1,575 are unaccounted for"
|
|
33
78
|
},
|
|
34
79
|
"schools.json": {
|
|
35
80
|
"items": 17342,
|
|
36
81
|
"source": "Irish dance school names: a curated school list and the school cells of a results corpus",
|
|
37
|
-
"url": null
|
|
82
|
+
"url": null,
|
|
83
|
+
"rebuild": "not reproducible: the sources are not public"
|
|
38
84
|
}
|
|
39
85
|
}
|
|
40
86
|
}
|