@finbheara/names 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +12 -6
  2. package/data/bloom/MANIFEST.json +59 -13
  3. package/dist/chunks/{index-7vvt5hhy.js → index-0wpvbdgh.js} +2631 -1395
  4. package/dist/chunks/{index-7vvt5hhy.js.map → index-0wpvbdgh.js.map} +4 -4
  5. package/dist/chunks/{index-kw69ry3j.js → index-74fyytnh.js} +1333 -20
  6. package/dist/chunks/{index-kw69ry3j.js.map → index-74fyytnh.js.map} +5 -4
  7. package/dist/chunks/{index-pncn9y2r.js → index-a6f2y4aa.js} +2 -2
  8. package/dist/chunks/{index-r5fp3znj.js → index-ky754nsf.js} +7 -7
  9. package/dist/chunks/{index-r5fp3znj.js.map → index-ky754nsf.js.map} +3 -3
  10. package/dist/chunks/{index-ka8nzreg.js → index-tk01chxw.js} +2 -2
  11. package/dist/chunks/{index-kw6gbrnq.js → index-tqdbxd92.js} +2 -2
  12. package/dist/chunks/{index-a15g2gp6.js → index-ye6j60jn.js} +3 -3
  13. package/dist/chunks/phrase-2hj26asc.js +11 -0
  14. package/dist/classifier/index.js +3 -3
  15. package/dist/cli/census.js +4 -4
  16. package/dist/index.js +14 -8
  17. package/dist/index.js.map +1 -1
  18. package/dist/lexicon/index.js +4 -2
  19. package/dist/lexicon/index.js.map +1 -1
  20. package/dist/measure/index.js +4 -4
  21. package/dist/normalize/index.js +9 -5
  22. package/dist/normalize/index.js.map +1 -1
  23. package/dist/types/classifier/bf/filters.d.ts +8 -3
  24. package/dist/types/lexicon/flags.d.ts +13 -5
  25. package/dist/types/normalize/given-families.d.ts +48 -0
  26. package/dist/types/normalize/index.d.ts +1 -0
  27. package/dist/types/normalize/name-kinship.d.ts +12 -8
  28. package/docs/NAME-NORMALIZATION.md +36 -7
  29. package/names.txt +2624 -1389
  30. package/package.json +4 -2
  31. package/src/normalize/given-families.MANIFEST.json +46 -0
  32. package/dist/chunks/phrase-672g18kw.js +0 -11
  33. /package/dist/chunks/{index-pncn9y2r.js.map → index-a6f2y4aa.js.map} +0 -0
  34. /package/dist/chunks/{index-ka8nzreg.js.map → index-tk01chxw.js.map} +0 -0
  35. /package/dist/chunks/{index-kw6gbrnq.js.map → index-tqdbxd92.js.map} +0 -0
  36. /package/dist/chunks/{index-a15g2gp6.js.map → index-ye6j60jn.js.map} +0 -0
  37. /package/dist/chunks/{phrase-672g18kw.js.map → phrase-2hj26asc.js.map} +0 -0
package/README.md CHANGED
@@ -145,8 +145,10 @@ name cells, one per line. The specification is `docs/NAME-NORMALIZATION.md`.
145
145
  ### How two names are related
146
146
 
147
147
  ```ts
148
- import { createNameIndex, createNameKinship, givenFamilyTable, nameKinship, personName }
149
- from "@finbheara/names/normalize";
148
+ import {
149
+ createGivenFamilyList, createNameIndex, createNameKinship, givenFamilyTable, nameKinship,
150
+ personName,
151
+ } from "@finbheara/names/normalize";
150
152
 
151
153
  nameKinship(personName("Kate Smith"), personName("Katherine Smith"));
152
154
  // { relation: "candidate", via: ["given-family"], reason: "middles-equal", gateSafe: true }
@@ -157,13 +159,16 @@ nameKinship(personName("Kate M Smith"), personName("Katherine Conley-Smith")).vi
157
159
  nameKinship(personName("Connor Smith"), personName("Conor Smith")).via; // ["given-variant"]
158
160
  nameKinship(personName("Evie Smith"), personName("Genevieve Smith"));
159
161
  // { relation: "candidate", via: ["given-family-not-gate"], reason: "middles-equal", gateSafe: false }
162
+ nameKinship(personName("Beth Smith"), personName("Bethany Smith")); // the extended family list
163
+ // { relation: "candidate", via: ["given-maybe"], reason: "middles-equal", gateSafe: false }
160
164
 
161
165
  // the given-name tolerance is a list of providers, asked in order
162
- const withMaybe = createNameKinship({
163
- given: [givenFamilyTable, (a, b) => (a === "NAIMH" && b === "NIAMH" ? "given-maybe" : null)],
166
+ const tableOnly = createNameKinship({ given: [givenFamilyTable] });
167
+ tableOnly(personName("Beth Smith"), personName("Bethany Smith")).reason; // "given-differs"
168
+ const common = createNameKinship({
169
+ given: [givenFamilyTable, createGivenFamilyList({ maxTier: 1 })], // common forms only
164
170
  });
165
- withMaybe(personName("Naimh Smith"), personName("Niamh Smith"));
166
- // { relation: "candidate", via: ["given-maybe"], reason: "middles-equal", gateSafe: false }
171
+ common(personName("Maggie Smith"), personName("Marjorie Smith")).reason; // "given-differs"
167
172
 
168
173
  const index = createNameIndex<number>("nn3"); // a NameMap with a tolerant lookup
169
174
  index.set(personName("Mary Conley-Larson"), 1).set(personName("Mary Carlson"), 2);
@@ -242,6 +247,7 @@ given name. Flags are case-sensitive and combine (`Kelly:AF`, `New York:BG`).
242
247
  | `B` | multi-word entry (bigram or trigram) |
243
248
  | `S` | a known typo, kept so it is recognised |
244
249
  | `X` | excluded from the display renderer's round-trip check |
250
+ | `D` | a given name a diminutive table holds (`Kate:FD`); a mark, like `S` and `X` |
245
251
 
246
252
  A digit after `L` or `A` is a **tradition footnote**: `1` Armenian, `2` Germanic,
247
253
  `3` Iberian, `4` Irish/Gaelic, `5` Nordic, `6` Slavic (`Kowalski:L6`). An index is
@@ -1,40 +1,86 @@
1
1
  {
2
- "about": "Bloom filters the phrase classifier consults. Each table: s = bits, h = hashes, d = seed, c = base64 bit content, n = items. Keys are lower-cased, accent-stripped, apostrophe-free. Built 2026-01-03, target false-positive rate 0.01. A table holds hashes of its source list, not the list; sources are recorded for rebuilding.",
2
+ "about": "Bloom filters the phrase classifier consults. Each table: s = bits, h = hashes, d = seed, c = base64 bit content, n = items. A table holds hashes of its source lists, not the lists; the lists are third-party files, recorded here for rebuilding. The four name tables are rebuilt by scripts/build-blooms.ts from the files named in their inputs; locations, us-locations and schools were built 2026-01-03 by a process that is not recorded, and are not rebuilt.",
3
+ "build": {
4
+ "script": "scripts/build-blooms.ts <source-dir> [--check]",
5
+ "seed": 78187493520,
6
+ "fpp": 0.01,
7
+ "sizing": "BloomFilter.create(n, fpp) from bloom-filters: s = ceil(-n ln(fpp) / ln(2)^2), h = ceil(s/n ln 2); n is the number of distinct keys",
8
+ "key": "normalize() in src/classifier/bf/filters.ts: lower-case, NFD, combining marks U+0300-U+036F removed, the ASCII apostrophe removed (U+2019 is kept). An empty key is dropped; a key repeated within or across inputs counts once",
9
+ "read": "a leading U+FEFF is stripped; lines split on LF or CRLF; format csv takes cell `column` after splitting the line on every comma (no quoting); header: the first line is skipped; trim: surrounding whitespace is removed before the key is taken"
10
+ },
3
11
  "tables": {
4
12
  "surnames.json": {
5
13
  "items": 151671,
6
- "source": "FiveThirtyEight most-common-name (surnames)",
7
- "url": "https://github.com/fivethirtyeight/data/tree/master/most-common-name"
14
+ "inputs": [
15
+ {
16
+ "file": "most-common-name_surnames.csv",
17
+ "format": "csv",
18
+ "column": 0,
19
+ "header": true,
20
+ "source": "FiveThirtyEight most-common-name (surnames)",
21
+ "url": "https://github.com/fivethirtyeight/data/tree/master/most-common-name"
22
+ }
23
+ ]
8
24
  },
9
25
  "human-names.json": {
10
26
  "items": 197226,
11
- "source": "human-names-list.txt (upstream not recorded)",
12
- "url": null
27
+ "inputs": [
28
+ {
29
+ "file": "human-names-list.txt",
30
+ "format": "lines",
31
+ "source": "human-names-list.txt (upstream not recorded)",
32
+ "url": null
33
+ }
34
+ ]
13
35
  },
14
36
  "first-names.json": {
15
- "items": 133749,
16
- "source": "name_gender_dataset.csv (upstream not recorded)",
17
- "url": null
37
+ "items": 134399,
38
+ "inputs": [
39
+ {
40
+ "file": "name_gender_dataset.csv",
41
+ "format": "csv",
42
+ "column": 0,
43
+ "header": true,
44
+ "source": "Gender by Name, UCI Machine Learning Repository dataset 591 (doi:10.24432/C55G7X), CC BY 4.0; the file inside gender+by+name.zip",
45
+ "url": "https://archive.ics.uci.edu/dataset/591/gender+by+name"
46
+ },
47
+ {
48
+ "file": "diminutive-names.txt",
49
+ "format": "lines",
50
+ "source": "given-name diminutives and familiar forms, 3,511 spellings as printed, one per line, merged from: this package's given-diminutive table; diminutives.db (data public domain) at commit caae757f1f91bce2175a49942a6c466846ade50c; Behind the Name names tagged \"diminutives\", fetched 2026-10-05; the familiar-form pairs mined in this repository's issue #59. Non-western names are kept: the table answers only whether a word is a given name",
51
+ "url": null
52
+ }
53
+ ]
18
54
  },
19
55
  "irish-surnames.json": {
20
56
  "items": 1624,
21
- "source": "Gaois sloinnte (Irish surnames database)",
22
- "url": "https://github.com/gaois/sloinnte"
57
+ "inputs": [
58
+ {
59
+ "file": "irish_surnames.txt",
60
+ "format": "lines",
61
+ "trim": true,
62
+ "source": "Gaois sloinnte (Irish surnames database)",
63
+ "url": "https://github.com/gaois/sloinnte"
64
+ }
65
+ ]
23
66
  },
24
67
  "locations.json": {
25
68
  "items": 8527,
26
69
  "source": "ISO 3166-1 and 3166-2; UN population centres; Wikipedia US cities; urban areas; FIPS; names.txt G entries; common Irish/UK/AU/NZ terms",
27
- "url": null
70
+ "url": null,
71
+ "rebuild": "not reproducible: most of its source lists and its key rules are not recorded. Checked 2026-10-06: country names, official names and ISO alpha-2/alpha-3 codes from a public country list (geo_country.csv) are held (249/250, 248/250, 241/243, 242/243 distinct keys), but these are a small part of 8,527 keys"
28
72
  },
29
73
  "us-locations.json": {
30
74
  "items": 23922,
31
75
  "source": "CDC PLACES county and place data; US states and territories; common US regions",
32
- "url": "https://www.cdc.gov/places/"
76
+ "url": "https://www.cdc.gov/places/",
77
+ "rebuild": "not reproducible: the states, territories and regions lists are not recorded. Checked 2026-10-06 against PLACES__County_Data__GIS_Friendly_Format___2020_release.csv and PLACES__Place_Data__GIS_Friendly_Format___2025_release.csv: every distinct key is held (1,839 county names, the same with ' County', 19,837 place names (LocationName), 51 state names, 51 state abbreviations), together 22,347 of the table's 23,922 keys; the other 1,575 are unaccounted for"
33
78
  },
34
79
  "schools.json": {
35
80
  "items": 17342,
36
81
  "source": "Irish dance school names: a curated school list and the school cells of a results corpus",
37
- "url": null
82
+ "url": null,
83
+ "rebuild": "not reproducible: the sources are not public"
38
84
  }
39
85
  }
40
86
  }