@finbheara/names 0.9.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -1
- package/dist/chunks/{index-b2k3er3h.js → index-7sppf1vh.js} +90 -8
- package/dist/chunks/{index-b2k3er3h.js.map → index-7sppf1vh.js.map} +5 -3
- package/dist/chunks/{index-0qptxnwh.js → index-7vvt5hhy.js} +13 -2
- package/dist/chunks/{index-0qptxnwh.js.map → index-7vvt5hhy.js.map} +2 -2
- package/dist/chunks/index-ebebeamh.js +446 -0
- package/dist/chunks/index-ebebeamh.js.map +10 -0
- package/dist/chunks/{index-fzzq5txh.js → index-khxa4qjy.js} +2 -2
- package/dist/chunks/{index-2fsdvhaf.js → index-kw6gbrnq.js} +2 -2
- package/dist/chunks/{index-d4wr471g.js → index-pncn9y2r.js} +2 -2
- package/dist/chunks/index-qsznkk4s.js +3 -0
- package/dist/chunks/index-qsznkk4s.js.map +9 -0
- package/dist/chunks/{index-852w8w1s.js → index-r5fp3znj.js} +3 -3
- package/dist/chunks/{index-0vg56t5v.js → index-yq8hsd1f.js} +3 -3
- package/dist/chunks/phrase-672g18kw.js +11 -0
- package/dist/classifier/index.js +3 -3
- package/dist/cli/census.js +4 -4
- package/dist/index.js +25 -8
- package/dist/index.js.map +1 -1
- package/dist/lexicon/index.js +1 -1
- package/dist/measure/index.js +4 -4
- package/dist/normalize/index.js +17 -5
- package/dist/normalize/index.js.map +1 -1
- package/dist/phonetic/index.js +10 -0
- package/dist/phonetic/index.js.map +9 -0
- package/dist/types/index.d.ts +2 -0
- package/dist/types/normalize/block-keys.d.ts +53 -0
- package/dist/types/normalize/index.d.ts +2 -0
- package/dist/types/normalize/name-columns.d.ts +51 -0
- package/dist/types/phonetic/double-metaphone.d.ts +25 -0
- package/dist/types/phonetic/index.d.ts +11 -0
- package/names.txt +11 -0
- package/package.json +8 -2
- package/dist/chunks/phrase-dczgr4vs.js +0 -11
- /package/dist/chunks/{index-fzzq5txh.js.map → index-khxa4qjy.js.map} +0 -0
- /package/dist/chunks/{index-2fsdvhaf.js.map → index-kw6gbrnq.js.map} +0 -0
- /package/dist/chunks/{index-d4wr471g.js.map → index-pncn9y2r.js.map} +0 -0
- /package/dist/chunks/{index-852w8w1s.js.map → index-r5fp3znj.js.map} +0 -0
- /package/dist/chunks/{index-0vg56t5v.js.map → index-yq8hsd1f.js.map} +0 -0
- /package/dist/chunks/{phrase-dczgr4vs.js.map → phrase-672g18kw.js.map} +0 -0
package/README.md
CHANGED
|
@@ -26,7 +26,7 @@ ESM, Node ≥ 22 or Bun ≥ 1.4.2. MIT.
|
|
|
26
26
|
|
|
27
27
|
## Entry points
|
|
28
28
|
|
|
29
|
-
Import only what you use; `/lexicon`, `/particles` and `/
|
|
29
|
+
Import only what you use; `/lexicon`, `/particles`, `/normalize` and `/phonetic` never load the
|
|
30
30
|
classifier's bloom tables.
|
|
31
31
|
|
|
32
32
|
| import | what |
|
|
@@ -35,6 +35,7 @@ classifier's bloom tables.
|
|
|
35
35
|
| `@finbheara/names/lexicon` | `names.txt` parsed: flags, multi-word phrases, spellings, roles |
|
|
36
36
|
| `@finbheara/names/particles` | the particle vocabulary |
|
|
37
37
|
| `@finbheara/names/classifier` | person / school / location classification and segmentation |
|
|
38
|
+
| `@finbheara/names/phonetic` | Double Metaphone: how a word sounds, as a key |
|
|
38
39
|
| `@finbheara/names/measure` | the census and the classifier evaluation |
|
|
39
40
|
| `@finbheara/names` | all of the above |
|
|
40
41
|
| `@finbheara/names/names.txt` | the raw lexicon |
|
|
@@ -141,6 +142,42 @@ The verdict is a declaration the caller makes about a document; grouping rows in
|
|
|
141
142
|
documents is the caller's. `bun run names order <file>` prints the same for a file of
|
|
142
143
|
name cells, one per line. The specification is `docs/NAME-NORMALIZATION.md`.
|
|
143
144
|
|
|
145
|
+
## Phonetic and blocking keys
|
|
146
|
+
|
|
147
|
+
```ts
|
|
148
|
+
import { doubleMetaphone } from "@finbheara/names/phonetic";
|
|
149
|
+
import { initialsBlock, metaphoneBlocks, personName, surnameBlock } from "@finbheara/names/normalize";
|
|
150
|
+
|
|
151
|
+
doubleMetaphone("SMITH"); // ["SM0", "XMT"] primary, secondary
|
|
152
|
+
doubleMetaphone("KAVANAGH"); // ["KFNK", "KFNK"]
|
|
153
|
+
|
|
154
|
+
const n = personName("Moira O'Rervton");
|
|
155
|
+
surnameBlock(n.surnameKey()); // "ORERVTON" A-Z letters only
|
|
156
|
+
metaphoneBlocks(n.surnameKey()); // ["ARRF"]
|
|
157
|
+
initialsBlock(n.givenFields()[0] ?? "", n.surnameKey()); // "MO"
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
`doubleMetaphone` is a general phonetic key: a printed name and a transcribed spoken
|
|
161
|
+
word are asked the same question. The three block keys bucket names that are worth
|
|
162
|
+
comparing and decide nothing; each answers `null` (or `[]`) where there is nothing to
|
|
163
|
+
bucket on, so a self-join never pools the blanks.
|
|
164
|
+
|
|
165
|
+
### A name as database columns
|
|
166
|
+
|
|
167
|
+
```ts
|
|
168
|
+
import { NAME_COLUMNS, nameColumns, personName } from "@finbheara/names/normalize";
|
|
169
|
+
|
|
170
|
+
NAME_COLUMNS; // ["nn0", …, "nn4", "surname_key", "given_key", "blk_surname", "blk_initials",
|
|
171
|
+
// "blk_metaphone", "blk_metaphone_alt"]
|
|
172
|
+
nameColumns(personName("Moira O'Rervton"));
|
|
173
|
+
// { nn0: "Moira O'Rervton", …, nn3: "ORERVTON<<MOIRA", surname_key: "ORERVTON",
|
|
174
|
+
// given_key: "MOIRA", blk_surname: "ORERVTON", blk_initials: "MO", blk_metaphone: "ARRF", … }
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
One encoder for every key a store indexes, with no database dependency. A level column
|
|
178
|
+
is the slot `createNameMap` files a name under, `null` for a print that names no person,
|
|
179
|
+
so a map lookup at a level is one equality on one column.
|
|
180
|
+
|
|
144
181
|
## The lexicon
|
|
145
182
|
|
|
146
183
|
```ts
|
|
@@ -3,7 +3,13 @@ import {
|
|
|
3
3
|
isGiven2,
|
|
4
4
|
isAmbiguous2,
|
|
5
5
|
Lexicon2
|
|
6
|
-
} from "./index-
|
|
6
|
+
} from "./index-7vvt5hhy.js";
|
|
7
|
+
import {
|
|
8
|
+
isPlaceWordFlags
|
|
9
|
+
} from "./index-kw6gbrnq.js";
|
|
10
|
+
import {
|
|
11
|
+
doubleMetaphone2
|
|
12
|
+
} from "./index-ebebeamh.js";
|
|
7
13
|
import {
|
|
8
14
|
nameFold2,
|
|
9
15
|
isPlaceholderPrint2,
|
|
@@ -13,10 +19,7 @@ import {
|
|
|
13
19
|
commaReading2,
|
|
14
20
|
levelOf2,
|
|
15
21
|
defaultPersonNameFactory2
|
|
16
|
-
} from "./index-
|
|
17
|
-
import {
|
|
18
|
-
isPlaceWordFlags
|
|
19
|
-
} from "./index-2fsdvhaf.js";
|
|
22
|
+
} from "./index-khxa4qjy.js";
|
|
20
23
|
import {
|
|
21
24
|
isTeamPrint2
|
|
22
25
|
} from "./index-0phgwy52.js";
|
|
@@ -94,11 +97,14 @@ var given_diminutives_default = {
|
|
|
94
97
|
"the two names."
|
|
95
98
|
],
|
|
96
99
|
families: {
|
|
100
|
+
ABIGAIL: ["ABBY", "ABBIE"],
|
|
97
101
|
AIDAN: ["AIDEN", "AODHAN", "AIDY"],
|
|
98
102
|
ALEXANDER: ["ALEX", "ALEC", "SANDY", "XANDER", "ALEXANDRA", "ALEXA", "LEXI", "LEXIE"],
|
|
103
|
+
AMANDA: ["MANDY"],
|
|
99
104
|
ANDREW: ["ANDY"],
|
|
100
105
|
ANNE: ["ANN", "ANNA", "ANNIE", "NANCY", "NAN", "ANYA"],
|
|
101
106
|
ANTHONY: ["TONY"],
|
|
107
|
+
BARBARA: ["BARB"],
|
|
102
108
|
BRIDGET: ["BRIDIE", "BRID", "BRIGID", "BRIGHID", "BREDA", "BREEDA", "BIDDY"],
|
|
103
109
|
CAITLIN: ["CAITLYN", "KAITLIN", "KAITLYN", "KATELYN", "CAIT"],
|
|
104
110
|
CATHERINE: [
|
|
@@ -119,13 +125,18 @@ var given_diminutives_default = {
|
|
|
119
125
|
CHRISTOPHER: ["CHRIS", "KIT", "TOPHER"],
|
|
120
126
|
CIARA: ["KIERA", "KEIRA", "CIARRA"],
|
|
121
127
|
CIARAN: ["KIERAN", "KEIRAN"],
|
|
128
|
+
CYNTHIA: ["CINDY"],
|
|
122
129
|
DANIEL: ["DAN", "DANNY", "DANIELLE", "DANI"],
|
|
130
|
+
DEBORAH: ["DEBRA", "DEBBIE", "DEB"],
|
|
123
131
|
DEIRDRE: ["DEE", "DIDI"],
|
|
132
|
+
DOROTHY: ["DOT", "DOTTIE"],
|
|
124
133
|
EDWARD: ["ED", "EDDIE", "TED", "TEDDY", "NED"],
|
|
125
134
|
EILEEN: ["EILISH", "EILIS", "AILEEN"],
|
|
126
135
|
ELIZABETH: ["LIZ", "LIZZIE", "BETH", "BETSY", "BETTY", "ELIZA", "LIBBY", "ELSIE", "ELISE"],
|
|
127
136
|
EOIN: ["OWEN", "EOGHAN"],
|
|
137
|
+
FIONNUALA: ["FIONNULA", "FINOLA", "NUALA"],
|
|
128
138
|
FRANCIS: ["FRANK", "FRANKIE", "FRAN", "FRANCES"],
|
|
139
|
+
GILLIAN: ["GILL"],
|
|
129
140
|
GRAINNE: ["GRANIA", "GRACE", "GRAINE"],
|
|
130
141
|
JACQUELINE: ["JACKIE"],
|
|
131
142
|
JAMES: ["JIM", "JIMMY", "JAMIE", "SEAMUS", "JAY"],
|
|
@@ -133,6 +144,7 @@ var given_diminutives_default = {
|
|
|
133
144
|
JOHN: ["JACK", "JOHNNY", "SEAN", "SHANE", "JOHNNIE", "EOIN"],
|
|
134
145
|
JOSEPH: ["JOE", "JOEY", "JOSEPHINE", "JO"],
|
|
135
146
|
JUDITH: ["JUDY"],
|
|
147
|
+
KIMBERLY: ["KIMBERLEY", "KIM"],
|
|
136
148
|
MARGARET: [
|
|
137
149
|
"MAGGIE",
|
|
138
150
|
"PEGGY",
|
|
@@ -150,8 +162,10 @@ var given_diminutives_default = {
|
|
|
150
162
|
MICHAEL: ["MIKE", "MICK", "MICKEY", "MICHEAL", "MICHELLE", "MICHAELA"],
|
|
151
163
|
NICHOLAS: ["NICK", "NICKY", "NICOLE", "NICOLA", "NIKKI"],
|
|
152
164
|
NIAMH: ["NEVE", "NEEVE"],
|
|
165
|
+
PAMELA: ["PAM"],
|
|
153
166
|
PATRICK: ["PAT", "PADDY", "PADRAIG", "PATSY", "PATRICIA", "TRICIA", "TRISH", "PATTY"],
|
|
154
167
|
PETER: ["PETE", "PEADAR"],
|
|
168
|
+
PHILIPPA: ["PIPPA"],
|
|
155
169
|
RAYMOND: ["RAY"],
|
|
156
170
|
REBECCA: ["BECKY", "BECCA", "BEX"],
|
|
157
171
|
RICHARD: ["RICK", "RICKY", "DICK", "RICH", "RICHIE"],
|
|
@@ -160,11 +174,15 @@ var given_diminutives_default = {
|
|
|
160
174
|
SIOBHAN: ["SHIVAUN", "CHEVONNE", "SIOBHAIN"],
|
|
161
175
|
STEPHEN: ["STEVEN", "STEVE", "STEPHANIE", "STEPH"],
|
|
162
176
|
SUSAN: ["SUE", "SUSIE", "SUZANNE", "SUZY"],
|
|
177
|
+
TERESA: ["THERESA", "TESS"],
|
|
163
178
|
THOMAS: ["TOM", "TOMMY", "TOMAS", "THOM"],
|
|
179
|
+
VICTORIA: ["VICKY", "VICKI"],
|
|
164
180
|
VINCENT: ["VINNIE"],
|
|
165
181
|
WILLIAM: ["WILL", "BILL", "BILLY", "WILLIE", "LIAM"]
|
|
166
182
|
},
|
|
167
183
|
notAtGate: [
|
|
184
|
+
"ABBIE",
|
|
185
|
+
"ABBY",
|
|
168
186
|
"ALEXA",
|
|
169
187
|
"ALEXANDRA",
|
|
170
188
|
"ANNA",
|
|
@@ -187,6 +205,7 @@ var given_diminutives_default = {
|
|
|
187
205
|
"JO",
|
|
188
206
|
"JOSEPHINE",
|
|
189
207
|
"KATHLEEN",
|
|
208
|
+
"KIM",
|
|
190
209
|
"LEXI",
|
|
191
210
|
"LEXIE",
|
|
192
211
|
"LIAM",
|
|
@@ -206,6 +225,7 @@ var given_diminutives_default = {
|
|
|
206
225
|
"NICOLA",
|
|
207
226
|
"NICOLE",
|
|
208
227
|
"NIKKI",
|
|
228
|
+
"NUALA",
|
|
209
229
|
"OWEN",
|
|
210
230
|
"PADRAIG",
|
|
211
231
|
"PATRICIA",
|
|
@@ -221,6 +241,7 @@ var given_diminutives_default = {
|
|
|
221
241
|
"STEPH",
|
|
222
242
|
"STEPHANIE",
|
|
223
243
|
"SUZANNE",
|
|
244
|
+
"TESS",
|
|
224
245
|
"TOMAS",
|
|
225
246
|
"TRICIA",
|
|
226
247
|
"TRISH"
|
|
@@ -951,6 +972,67 @@ function personList2(print, factory = defaultPersonNameFactory2()) {
|
|
|
951
972
|
}
|
|
952
973
|
return out;
|
|
953
974
|
}
|
|
975
|
+
// src/normalize/block-keys.ts
|
|
976
|
+
function blockLetters(surname) {
|
|
977
|
+
return surname.toUpperCase().replace(/[^A-Z]/g, "");
|
|
978
|
+
}
|
|
979
|
+
function firstLetter(s) {
|
|
980
|
+
const m = s.match(/[A-Za-z]/);
|
|
981
|
+
return m ? m[0].toUpperCase() : null;
|
|
982
|
+
}
|
|
983
|
+
function surnameBlock2(surname) {
|
|
984
|
+
return blockLetters(surname) || null;
|
|
985
|
+
}
|
|
986
|
+
function initialsBlock2(given, surname) {
|
|
987
|
+
const first = firstLetter(given);
|
|
988
|
+
const last = firstLetter(blockLetters(surname));
|
|
989
|
+
return first && last ? first + last : null;
|
|
990
|
+
}
|
|
991
|
+
function metaphoneBlocks2(surname) {
|
|
992
|
+
const letters = blockLetters(surname);
|
|
993
|
+
if (!letters)
|
|
994
|
+
return [];
|
|
995
|
+
const [primary, secondary] = doubleMetaphone2(letters);
|
|
996
|
+
if (!primary)
|
|
997
|
+
return [];
|
|
998
|
+
return secondary && secondary !== primary ? [primary, secondary] : [primary];
|
|
999
|
+
}
|
|
1000
|
+
// src/normalize/name-columns.ts
|
|
1001
|
+
var NAME_LEVEL_COLUMNS2 = [
|
|
1002
|
+
"nn0",
|
|
1003
|
+
"nn1",
|
|
1004
|
+
"nn2",
|
|
1005
|
+
"nn3",
|
|
1006
|
+
"nn4"
|
|
1007
|
+
];
|
|
1008
|
+
var NAME_COLUMNS2 = [
|
|
1009
|
+
...NAME_LEVEL_COLUMNS2,
|
|
1010
|
+
"surname_key",
|
|
1011
|
+
"given_key",
|
|
1012
|
+
"blk_surname",
|
|
1013
|
+
"blk_initials",
|
|
1014
|
+
"blk_metaphone",
|
|
1015
|
+
"blk_metaphone_alt"
|
|
1016
|
+
];
|
|
1017
|
+
function nameColumns2(name) {
|
|
1018
|
+
const person = name.isPerson();
|
|
1019
|
+
const level = (dop) => person ? name[dop] : null;
|
|
1020
|
+
const surname = person ? name.surnameKey() : "";
|
|
1021
|
+
const [metaphone, metaphoneAlt] = metaphoneBlocks2(surname);
|
|
1022
|
+
return {
|
|
1023
|
+
nn0: level("nn0"),
|
|
1024
|
+
nn1: level("nn1"),
|
|
1025
|
+
nn2: level("nn2"),
|
|
1026
|
+
nn3: level("nn3"),
|
|
1027
|
+
nn4: level("nn4"),
|
|
1028
|
+
surname_key: surname || null,
|
|
1029
|
+
given_key: person && name.givenName() || null,
|
|
1030
|
+
blk_surname: surnameBlock2(surname),
|
|
1031
|
+
blk_initials: person ? initialsBlock2(name.givenFields()[0] ?? "", surname) : null,
|
|
1032
|
+
blk_metaphone: metaphone ?? null,
|
|
1033
|
+
blk_metaphone_alt: metaphoneAlt ?? null
|
|
1034
|
+
};
|
|
1035
|
+
}
|
|
954
1036
|
// src/normalize/name-format.ts
|
|
955
1037
|
var NAME_FORMATS2 = ["First Last", "Last, First", "Last First"];
|
|
956
1038
|
function nameFormatOrder2(verdict) {
|
|
@@ -1338,7 +1420,7 @@ function deriveNameFormat2(prints, options = {}) {
|
|
|
1338
1420
|
return first;
|
|
1339
1421
|
return scoreReadings(cells.map((c) => readNameCell2(c, { ...options, placeColumn: true })), NAME_FORMAT_MODEL, threshold, schoolRead);
|
|
1340
1422
|
}
|
|
1341
|
-
export { repairMojibake2, sameGivenFamily2, comparePosition2, nameRelation2, nameCompare2, probablySame2, createNameSet2, createNameMap2, createNameClusters2, MAX_READINGS2, surnameParts2, printAttestsOrder2, nameReadings2, meetReading2, refinesReading2, sameReading2, chooseReadingPair2, composeReadings2, composeNames2, meetReadings2, nameMeet2, showReading2, blockKeysOfReadings2, nameBlockKeys2, namesFeasible2, INFEASIBLE_CLASSES2, editDistance2, nearSpelling2, compoundGiven2, middleForm2, infeasibleClass2, isListJoiner2, personList2, NAME_FORMATS2, nameFormatOrder2, ORDER_READINGS2, readNameCell2, DEFAULT_NAME_FORMAT_THRESHOLD2, deriveNameFormat2 };
|
|
1423
|
+
export { repairMojibake2, sameGivenFamily2, comparePosition2, nameRelation2, nameCompare2, probablySame2, createNameSet2, createNameMap2, createNameClusters2, MAX_READINGS2, surnameParts2, printAttestsOrder2, nameReadings2, meetReading2, refinesReading2, sameReading2, chooseReadingPair2, composeReadings2, composeNames2, meetReadings2, nameMeet2, showReading2, blockKeysOfReadings2, nameBlockKeys2, namesFeasible2, INFEASIBLE_CLASSES2, editDistance2, nearSpelling2, compoundGiven2, middleForm2, infeasibleClass2, isListJoiner2, personList2, surnameBlock2, initialsBlock2, metaphoneBlocks2, NAME_LEVEL_COLUMNS2, NAME_COLUMNS2, nameColumns2, NAME_FORMATS2, nameFormatOrder2, ORDER_READINGS2, readNameCell2, DEFAULT_NAME_FORMAT_THRESHOLD2, deriveNameFormat2 };
|
|
1342
1424
|
|
|
1343
|
-
//# debugId=
|
|
1344
|
-
//# sourceMappingURL=index-
|
|
1425
|
+
//# debugId=A39684B9543A024164756E2164756E21
|
|
1426
|
+
//# sourceMappingURL=index-7sppf1vh.js.map
|