@finbheara/names 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -0
- package/dist/chunks/{index-ye6j60jn.js → index-17zzg7f3.js} +5 -5
- package/dist/chunks/{index-ye6j60jn.js.map → index-17zzg7f3.js.map} +2 -2
- package/dist/chunks/index-18z8akw8.js +42 -0
- package/dist/chunks/{index-tqdbxd92.js.map → index-18z8akw8.js.map} +3 -3
- package/dist/chunks/{index-ky754nsf.js → index-7j32khen.js} +9 -7
- package/dist/chunks/{index-ky754nsf.js.map → index-7j32khen.js.map} +4 -4
- package/dist/chunks/{index-tk01chxw.js → index-gxw36cg3.js} +27 -16
- package/dist/chunks/{index-tk01chxw.js.map → index-gxw36cg3.js.map} +2 -2
- package/dist/chunks/{index-74fyytnh.js → index-qmn87bdh.js} +21 -1411
- package/dist/chunks/index-qmn87bdh.js.map +11 -0
- package/dist/chunks/index-xtwbdzy1.js +1418 -0
- package/dist/chunks/{index-74fyytnh.js.map → index-xtwbdzy1.js.map} +3 -5
- package/dist/chunks/{phrase-2hj26asc.js → phrase-hze7av6y.js} +3 -3
- package/dist/classifier/index.js +2 -2
- package/dist/cli/bin.js +109 -0
- package/dist/cli/bin.js.map +11 -0
- package/dist/cli/census.js +3 -3
- package/dist/index.js +24 -10
- package/dist/index.js.map +1 -1
- package/dist/measure/index.js +2 -2
- package/dist/normalize/index.js +22 -8
- package/dist/normalize/index.js.map +1 -1
- package/dist/types/classifier/features.d.ts +2 -2
- package/dist/types/cli/bin.d.ts +10 -0
- package/dist/types/cli/tables.d.ts +13 -0
- package/dist/types/lexicon/place.d.ts +16 -0
- package/dist/types/normalize/given-diminutives.d.ts +15 -6
- package/dist/types/normalize/given-families.d.ts +5 -0
- package/dist/types/normalize/index.d.ts +2 -2
- package/package.json +2 -1
- package/dist/chunks/index-tqdbxd92.js +0 -27
- /package/dist/chunks/{phrase-2hj26asc.js.map → phrase-hze7av6y.js.map} +0 -0
package/README.md
CHANGED
|
@@ -180,6 +180,39 @@ index.related(personName("Mary Larson")); // [{ name: Conley-Larson<<Mary
|
|
|
180
180
|
tolerance the pair used, and `gateSafe` is false when any of them is one a gate must not
|
|
181
181
|
admit on its own.
|
|
182
182
|
|
|
183
|
+
### The given-name tables, for another language
|
|
184
|
+
|
|
185
|
+
The tables behind `givenFamilyTable` and `givenFamilyList` print as JSON, for a consumer
|
|
186
|
+
that is not JavaScript:
|
|
187
|
+
|
|
188
|
+
```sh
|
|
189
|
+
npx @finbheara/names tables given-name > given-name-tables.json # or bunx
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
`--output json` and `--schema 1` are the defaults. The PRINTED form is the contract: a
|
|
193
|
+
schema keeps its shape for as long as the package offers it, and a new shape is a new
|
|
194
|
+
schema, so a command written today keeps working. The files the tables are stored in are
|
|
195
|
+
not part of it. Schema 1 is
|
|
196
|
+
|
|
197
|
+
```jsonc
|
|
198
|
+
{
|
|
199
|
+
"$comment": [ /* how to read it */ ],
|
|
200
|
+
"table": "given-name", "schema": 1,
|
|
201
|
+
"variants": { "CONNOR": ["CONOR"], … }, // canonical -> its other spellings
|
|
202
|
+
"families": { "CATHERINE": ["KATE", …], … }, // the curated gate tier: root -> forms
|
|
203
|
+
"notAtGate": ["MEGAN", …], // forms a gate must not admit as forms
|
|
204
|
+
"extended": { "BETHANY": { "BETH": 1 }, … } // root -> form -> tier (1 common .. 3 rare)
|
|
205
|
+
}
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
Read it as the library does, and as `$comment` says: read each name as its canonical
|
|
209
|
+
spelling first; two names relate only when they share a root, a form meets EVERY root it
|
|
210
|
+
sits under, and nothing is transitive (`EVIE` meets `EVE` and `EVA`, which do not meet each
|
|
211
|
+
other). A shared root is `given-family` unless one name reaches it only as a `notAtGate`
|
|
212
|
+
form; a name that is the root itself always passes. An `extended` pair's tier is its rarer
|
|
213
|
+
member's, the root counting as 1. To avoid comparing every pair, index each name under
|
|
214
|
+
every root it has and compare within a root, never on one root per name.
|
|
215
|
+
|
|
183
216
|
## Phonetic and blocking keys
|
|
184
217
|
|
|
185
218
|
```ts
|
|
@@ -3,9 +3,9 @@ import {
|
|
|
3
3
|
Lexicon2
|
|
4
4
|
} from "./index-0wpvbdgh.js";
|
|
5
5
|
import {
|
|
6
|
-
|
|
6
|
+
lexiconNameOrder2,
|
|
7
7
|
defaultPersonNameFactory2
|
|
8
|
-
} from "./index-
|
|
8
|
+
} from "./index-gxw36cg3.js";
|
|
9
9
|
import {
|
|
10
10
|
isNameParticle2
|
|
11
11
|
} from "./index-23rzmwe2.js";
|
|
@@ -73,7 +73,7 @@ function createNameCensus2(options = {}) {
|
|
|
73
73
|
return;
|
|
74
74
|
}
|
|
75
75
|
persons += times;
|
|
76
|
-
if (order === "unknown" && !print.includes(",") &&
|
|
76
|
+
if (order === "unknown" && !print.includes(",") && lexiconNameOrder2(print, roles) === undefined) {
|
|
77
77
|
orderUndecided.add("order undecided", print, times);
|
|
78
78
|
}
|
|
79
79
|
for (const run of runs(name)) {
|
|
@@ -133,5 +133,5 @@ function createNameCensus2(options = {}) {
|
|
|
133
133
|
|
|
134
134
|
export { createNameCensus2 };
|
|
135
135
|
|
|
136
|
-
//# debugId=
|
|
137
|
-
//# sourceMappingURL=index-
|
|
136
|
+
//# debugId=C23DFDD520B2D39C64756E2164756E21
|
|
137
|
+
//# sourceMappingURL=index-17zzg7f3.js.map
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
"sourcesContent": [
|
|
5
5
|
"/**\n * A census of name-shaped strings: run a stream of printed names through the\n * library and report what it could not read, and why.\n *\n * The report is the maintenance loop. An UNKNOWN TOKEN is a word no `names.txt`\n * entry knows; an UNKNOWN PAIR is two adjacent unknown words, the usual shape of\n * a school or place phrase the lexicon should carry as a `B` entry. A misread is\n * fixed by making the name known in `names.txt`, so the unknown tokens, ranked\n * by how often they print, are the candidate additions.\n *\n * Nothing here reads a file or keeps global state; a census holds only what was\n * added to it.\n */\n\nimport type { PhraseClassifier } from \"../classifier/phrase.ts\";\nimport { Lexicon } from \"../lexicon/index.ts\";\nimport { lookupKey } from \"../lexicon/keys.ts\";\nimport type { PersonName, PersonNameFactory } from \"../normalize/factory.ts\";\nimport { defaultPersonNameFactory } from \"../normalize/factory.ts\";\nimport { type NameOrder, lexiconNameOrder } from \"../normalize/name-split.ts\";\nimport { isNameParticle } from \"../particles/particle-vocabulary.ts\";\n\nexport interface NameCensusOptions {\n\t/** Parses each print. Default: `defaultPersonNameFactory()`. */\n\treadonly factory?: PersonNameFactory;\n\t/** What counts as known. Default: `Lexicon.default()`. */\n\treadonly lexicon?: Lexicon;\n\t/**\n\t * Also classify each distinct print as person / school / location and report\n\t * the low-confidence ones. Off unless given, since it loads the bloom tables.\n\t */\n\treadonly classifier?: PhraseClassifier;\n\t/** Below this `classifyNameNormalized` confidence a print is reported. Default 0.3. */\n\treadonly lowConfidence?: number;\n\t/** Examples kept per reported item. Default 5. */\n\treadonly examples?: number;\n}\n\n/** One reported item: how often it printed, and a few prints it came from. */\nexport interface CensusItem {\n\treadonly key: string;\n\treadonly count: number;\n\treadonly examples: readonly string[];\n}\n\nexport interface NameCensusReport {\n\t/** prints added, counting repeats */\n\treadonly total: number;\n\t/** distinct (print, order) pairs */\n\treadonly distinct: number;\n\t/** parsed as a person */\n\treadonly persons: number;\n\t/** read as a team designation (`Team A`, `Foireann B`) */\n\treadonly teams: number;\n\t/** empty cells */\n\treadonly empty: number;\n\t/** refused as no person's name, by rule (`refuse-digit`, `unsplittable`, …) */\n\treadonly refused: readonly CensusItem[];\n\t/** words of person names that no lexicon entry knows, most frequent first */\n\treadonly unknownTokens: readonly CensusItem[];\n\t/** adjacent unknown words, most frequent first: candidate `B` entries */\n\treadonly unknownPairs: readonly CensusItem[];\n\t/** comma-less prints of undeclared order the lexicon could not orient */\n\treadonly orderUndecided: CensusItem;\n\t/** low-confidence classifications, when a classifier was given */\n\treadonly lowConfidence: readonly CensusItem[];\n}\n\nexport interface NameCensus {\n\t/** Count one printed cell. `order` is what the source declared, if anything. */\n\tadd(print: string, order?: NameOrder): void;\n\treport(options?: { readonly limit?: number }): NameCensusReport;\n}\n\nclass Tally {\n\treadonly items = new Map<string, { count: number; examples: string[] }>();\n\tconstructor(private readonly keep: number) {}\n\tadd(key: string, example: string, times = 1): void {\n\t\tlet item = this.items.get(key);\n\t\tif (item === undefined) this.items.set(key, (item = { count: 0, examples: [] }));\n\t\titem.count += times;\n\t\tif (item.examples.length < this.keep && !item.examples.includes(example)) {\n\t\t\titem.examples.push(example);\n\t\t}\n\t}\n\ttop(limit: number): CensusItem[] {\n\t\treturn [...this.items]\n\t\t\t.map(([key, { count, examples }]) => ({ key, count, examples }))\n\t\t\t.sort((a, b) => b.count - a.count || (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))\n\t\t\t.slice(0, limit);\n\t}\n}\n\n/**\n * Could this word be a missing names.txt entry? Not an initial (`M`, `M.`), not\n * a surname particle (`O`, `de`), and not empty after folding.\n */\nfunction isNameWord(word: string, key: string): boolean {\n\tif (key === \"\") return false;\n\tif (word.replace(/\\./gu, \"\").length <= 1) return false;\n\treturn !isNameParticle(word);\n}\n\n/** A census over the default factory and lexicon, or the ones given. */\nexport function createNameCensus(options: NameCensusOptions = {}): NameCensus {\n\tconst factory = options.factory ?? defaultPersonNameFactory();\n\tconst lexicon = options.lexicon ?? Lexicon.default();\n\tconst classifier = options.classifier;\n\tconst threshold = options.lowConfidence ?? 0.3;\n\tconst keep = options.examples ?? 5;\n\tconst roles = lexicon.roles();\n\n\tconst seen = new Map<string, number>();\n\tlet total = 0;\n\tlet persons = 0;\n\tlet teams = 0;\n\tlet empty = 0;\n\tconst refused = new Tally(keep);\n\tconst unknownTokens = new Tally(keep);\n\tconst unknownPairs = new Tally(keep);\n\tconst orderUndecided = new Tally(keep);\n\tconst lowConfidence = new Tally(keep);\n\n\t/**\n\t * The words of the scrubbed print IN PRINT ORDER, one run per comma-separated\n\t * field, so a pair is two words the page printed side by side.\n\t */\n\tfunction runs(name: PersonName): string[][] {\n\t\tconst text = name.verdict.kind === \"name\" ? name.verdict.scrubbed : name.nn0;\n\t\treturn text.split(\",\").map((f) => f.split(/[\\s-]+/u).filter((w) => w !== \"\"));\n\t}\n\n\tfunction measure(print: string, order: NameOrder, times: number): void {\n\t\tconst name = factory.parse(print, order);\n\t\tif (name.verdict.kind === \"empty\") {\n\t\t\tempty += times;\n\t\t\treturn;\n\t\t}\n\t\tif (name.isTeam()) {\n\t\t\tteams += times;\n\t\t\treturn;\n\t\t}\n\t\tif (!name.isPerson()) {\n\t\t\trefused.add(\n\t\t\t\tname.verdict.kind === \"refused\" ? name.verdict.refusal.rule : \"unsplittable\",\n\t\t\t\tprint,\n\t\t\t\ttimes,\n\t\t\t);\n\t\t\treturn;\n\t\t}\n\t\tpersons += times;\n\t\tif (\n\t\t\torder === \"unknown\" &&\n\t\t\t!print.includes(\",\") &&\n\t\t\tlexiconNameOrder(print, roles) === undefined\n\t\t) {\n\t\t\torderUndecided.add(\"order undecided\", print, times);\n\t\t}\n\t\tfor (const run of runs(name)) {\n\t\t\tlet previousUnknown: string | undefined;\n\t\t\tfor (const w of run) {\n\t\t\t\tconst key = lookupKey(w);\n\t\t\t\tconst unknown = isNameWord(w, key) && lexicon.getNameFlags(w) === null;\n\t\t\t\tif (unknown) {\n\t\t\t\t\tunknownTokens.add(key, print, times);\n\t\t\t\t\tif (previousUnknown !== undefined) {\n\t\t\t\t\t\tunknownPairs.add(`${previousUnknown} ${key}`, print, times);\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t\tpreviousUnknown = unknown ? key : undefined;\n\t\t\t}\n\t\t}\n\t\tif (classifier !== undefined) {\n\t\t\tconst r = classifier.classifyNameNormalized(print);\n\t\t\tif (r.confidence < threshold) lowConfidence.add(r.category, print, times);\n\t\t}\n\t}\n\n\treturn {\n\t\tadd(print, order = \"unknown\") {\n\t\t\ttotal++;\n\t\t\tconst k = `${order}\\u0000${print}`;\n\t\t\tseen.set(k, (seen.get(k) ?? 0) + 1);\n\t\t},\n\t\treport({ limit = 50 } = {}) {\n\t\t\tpersons = teams = empty = 0;\n\t\t\tfor (const t of [refused, unknownTokens, unknownPairs, orderUndecided, lowConfidence]) {\n\t\t\t\tt.items.clear();\n\t\t\t}\n\t\t\tfor (const [k, times] of seen) {\n\t\t\t\tconst at = k.indexOf(\"\\u0000\");\n\t\t\t\tmeasure(k.slice(at + 1), k.slice(0, at) as NameOrder, times);\n\t\t\t}\n\t\t\treturn {\n\t\t\t\ttotal,\n\t\t\t\tdistinct: seen.size,\n\t\t\t\tpersons,\n\t\t\t\tteams,\n\t\t\t\tempty,\n\t\t\t\trefused: refused.top(limit),\n\t\t\t\tunknownTokens: unknownTokens.top(limit),\n\t\t\t\tunknownPairs: unknownPairs.top(limit),\n\t\t\t\torderUndecided: orderUndecided.top(1)[0] ?? {\n\t\t\t\t\tkey: \"order undecided\",\n\t\t\t\t\tcount: 0,\n\t\t\t\t\texamples: [],\n\t\t\t\t},\n\t\t\t\tlowConfidence: lowConfidence.top(limit),\n\t\t\t};\n\t\t},\n\t};\n}\n"
|
|
6
6
|
],
|
|
7
|
-
"mappings": ";;;;;;;;;;;;;AA0EA,MAAM,MAAM;AAAA,EAEkB;AAAA,EADpB,QAAQ,IAAI;AAAA,EACrB,WAAW,CAAkB,MAAc;AAAA,IAAd;AAAA;AAAA,EAC7B,GAAG,CAAC,KAAa,SAAiB,QAAQ,GAAS;AAAA,IAClD,IAAI,OAAO,KAAK,MAAM,IAAI,GAAG;AAAA,IAC7B,IAAI,SAAS;AAAA,MAAW,KAAK,MAAM,IAAI,KAAM,OAAO,EAAE,OAAO,GAAG,UAAU,CAAC,EAAE,CAAE;AAAA,IAC/E,KAAK,SAAS;AAAA,IACd,IAAI,KAAK,SAAS,SAAS,KAAK,QAAQ,CAAC,KAAK,SAAS,SAAS,OAAO,GAAG;AAAA,MACzE,KAAK,SAAS,KAAK,OAAO;AAAA,IAC3B;AAAA;AAAA,EAED,GAAG,CAAC,OAA6B;AAAA,IAChC,OAAO,CAAC,GAAG,KAAK,KAAK,EACnB,IAAI,EAAE,OAAO,OAAO,iBAAiB,EAAE,KAAK,OAAO,SAAS,EAAE,EAC9D,KAAK,CAAC,GAAG,MAAM,EAAE,QAAQ,EAAE,UAAU,EAAE,MAAM,EAAE,MAAM,KAAK,EAAE,MAAM,EAAE,MAAM,IAAI,EAAE,EAChF,MAAM,GAAG,KAAK;AAAA;AAElB;AAMA,SAAS,UAAU,CAAC,MAAc,KAAsB;AAAA,EACvD,IAAI,QAAQ;AAAA,IAAI,OAAO;AAAA,EACvB,IAAI,KAAK,QAAQ,QAAQ,EAAE,EAAE,UAAU;AAAA,IAAG,OAAO;AAAA,EACjD,OAAO,CAAC,gBAAe,IAAI;AAAA;AAIrB,SAAS,iBAAgB,CAAC,UAA6B,CAAC,GAAe;AAAA,EAC7E,MAAM,UAAU,QAAQ,WAAW,0BAAyB;AAAA,EAC5D,MAAM,UAAU,QAAQ,WAAW,SAAQ,QAAQ;AAAA,EACnD,MAAM,aAAa,QAAQ;AAAA,EAC3B,MAAM,YAAY,QAAQ,iBAAiB;AAAA,EAC3C,MAAM,OAAO,QAAQ,YAAY;AAAA,EACjC,MAAM,QAAQ,QAAQ,MAAM;AAAA,EAE5B,MAAM,OAAO,IAAI;AAAA,EACjB,IAAI,QAAQ;AAAA,EACZ,IAAI,UAAU;AAAA,EACd,IAAI,QAAQ;AAAA,EACZ,IAAI,QAAQ;AAAA,EACZ,MAAM,UAAU,IAAI,MAAM,IAAI;AAAA,EAC9B,MAAM,gBAAgB,IAAI,MAAM,IAAI;AAAA,EACpC,MAAM,eAAe,IAAI,MAAM,IAAI;AAAA,EACnC,MAAM,iBAAiB,IAAI,MAAM,IAAI;AAAA,EACrC,MAAM,gBAAgB,IAAI,MAAM,IAAI;AAAA,EAMpC,SAAS,IAAI,CAAC,MAA8B;AAAA,IAC3C,MAAM,OAAO,KAAK,QAAQ,SAAS,SAAS,KAAK,QAAQ,WAAW,KAAK;AAAA,IACzE,OAAO,KAAK,MAAM,GAAG,EAAE,IAAI,CAAC,MAAM,EAAE,MAAM,SAAS,EAAE,OAAO,CAAC,MAAM,MAAM,EAAE,CAAC;AAAA;AAAA,EAG7E,SAAS,OAAO,CAAC,OAAe,OAAkB,OAAqB;AAAA,IACtE,MAAM,OAAO,QAAQ,MAAM,OAAO,KAAK;AAAA,IACvC,IAAI,KAAK,QAAQ,SAAS,SAAS;AAAA,MAClC,SAAS;AAAA,MACT;AAAA,IACD;AAAA,IACA,IAAI,KAAK,OAAO,GAAG;AAAA,MAClB,SAAS;AAAA,MACT;AAAA,IACD;AAAA,IACA,IAAI,CAAC,KAAK,SAAS,GAAG;AAAA,MACrB,QAAQ,IACP,KAAK,QAAQ,SAAS,YAAY,KAAK,QAAQ,QAAQ,OAAO,gBAC9D,OACA,KACD;AAAA,MACA;AAAA,IACD;AAAA,IACA,WAAW;AAAA,IACX,IACC,UAAU,aACV,CAAC,MAAM,SAAS,GAAG,KACnB,
|
|
8
|
-
"debugId": "
|
|
7
|
+
"mappings": ";;;;;;;;;;;;;AA0EA,MAAM,MAAM;AAAA,EAEkB;AAAA,EADpB,QAAQ,IAAI;AAAA,EACrB,WAAW,CAAkB,MAAc;AAAA,IAAd;AAAA;AAAA,EAC7B,GAAG,CAAC,KAAa,SAAiB,QAAQ,GAAS;AAAA,IAClD,IAAI,OAAO,KAAK,MAAM,IAAI,GAAG;AAAA,IAC7B,IAAI,SAAS;AAAA,MAAW,KAAK,MAAM,IAAI,KAAM,OAAO,EAAE,OAAO,GAAG,UAAU,CAAC,EAAE,CAAE;AAAA,IAC/E,KAAK,SAAS;AAAA,IACd,IAAI,KAAK,SAAS,SAAS,KAAK,QAAQ,CAAC,KAAK,SAAS,SAAS,OAAO,GAAG;AAAA,MACzE,KAAK,SAAS,KAAK,OAAO;AAAA,IAC3B;AAAA;AAAA,EAED,GAAG,CAAC,OAA6B;AAAA,IAChC,OAAO,CAAC,GAAG,KAAK,KAAK,EACnB,IAAI,EAAE,OAAO,OAAO,iBAAiB,EAAE,KAAK,OAAO,SAAS,EAAE,EAC9D,KAAK,CAAC,GAAG,MAAM,EAAE,QAAQ,EAAE,UAAU,EAAE,MAAM,EAAE,MAAM,KAAK,EAAE,MAAM,EAAE,MAAM,IAAI,EAAE,EAChF,MAAM,GAAG,KAAK;AAAA;AAElB;AAMA,SAAS,UAAU,CAAC,MAAc,KAAsB;AAAA,EACvD,IAAI,QAAQ;AAAA,IAAI,OAAO;AAAA,EACvB,IAAI,KAAK,QAAQ,QAAQ,EAAE,EAAE,UAAU;AAAA,IAAG,OAAO;AAAA,EACjD,OAAO,CAAC,gBAAe,IAAI;AAAA;AAIrB,SAAS,iBAAgB,CAAC,UAA6B,CAAC,GAAe;AAAA,EAC7E,MAAM,UAAU,QAAQ,WAAW,0BAAyB;AAAA,EAC5D,MAAM,UAAU,QAAQ,WAAW,SAAQ,QAAQ;AAAA,EACnD,MAAM,aAAa,QAAQ;AAAA,EAC3B,MAAM,YAAY,QAAQ,iBAAiB;AAAA,EAC3C,MAAM,OAAO,QAAQ,YAAY;AAAA,EACjC,MAAM,QAAQ,QAAQ,MAAM;AAAA,EAE5B,MAAM,OAAO,IAAI;AAAA,EACjB,IAAI,QAAQ;AAAA,EACZ,IAAI,UAAU;AAAA,EACd,IAAI,QAAQ;AAAA,EACZ,IAAI,QAAQ;AAAA,EACZ,MAAM,UAAU,IAAI,MAAM,IAAI;AAAA,EAC9B,MAAM,gBAAgB,IAAI,MAAM,IAAI;AAAA,EACpC,MAAM,eAAe,IAAI,MAAM,IAAI;AAAA,EACnC,MAAM,iBAAiB,IAAI,MAAM,IAAI;AAAA,EACrC,MAAM,gBAAgB,IAAI,MAAM,IAAI;AAAA,EAMpC,SAAS,IAAI,CAAC,MAA8B;AAAA,IAC3C,MAAM,OAAO,KAAK,QAAQ,SAAS,SAAS,KAAK,QAAQ,WAAW,KAAK;AAAA,IACzE,OAAO,KAAK,MAAM,GAAG,EAAE,IAAI,CAAC,MAAM,EAAE,MAAM,SAAS,EAAE,OAAO,CAAC,MAAM,MAAM,EAAE,CAAC;AAAA;AAAA,EAG7E,SAAS,OAAO,CAAC,OAAe,OAAkB,OAAqB;AAAA,IACtE,MAAM,OAAO,QAAQ,MAAM,OAAO,KAAK;AAAA,IACvC,IAAI,KAAK,QAAQ,SAAS,SAAS;AAAA,MAClC,SAAS;AAAA,MACT;AAAA,IACD;AAAA,IACA,IAAI,KAAK,OAAO,GAAG;AAAA,MAClB,SAAS;AAAA,MACT;AAAA,IACD;AAAA,IACA,IAAI,CAAC,KAAK,SAAS,GAAG;AAAA,MACrB,QAAQ,IACP,KAAK,QAAQ,SAAS,YAAY,KAAK,QAAQ,QAAQ,OAAO,gBAC9D,OACA,KACD;AAAA,MACA;AAAA,IACD;AAAA,IACA,WAAW;AAAA,IACX,IACC,UAAU,aACV,CAAC,MAAM,SAAS,GAAG,KACnB,kBAAiB,OAAO,KAAK,MAAM,WAClC;AAAA,MACD,eAAe,IAAI,mBAAmB,OAAO,KAAK;AAAA,IACnD;AAAA,IACA,WAAW,OAAO,KAAK,IAAI,GAAG;AAAA,MAC7B,IAAI;AAAA,MACJ,WAAW,KAAK,KAAK;AAAA,QACpB,MAAM,MAAM,WAAU,CAAC;AAAA,QACvB,MAAM,UAAU,WAAW,GAAG,GAAG,KAAK,QAAQ,aAAa,CAAC,MAAM;AAAA,QAClE,IAAI,SAAS;AAAA,UACZ,cAAc,IAAI,KAAK,OAAO,KAAK;AAAA,UACnC,IAAI,oBAAoB,WAAW;AAAA,YAClC,aAAa,IAAI,GAAG,mBAAmB,OAAO,OAAO,KAAK;AAAA,UAC3D;AAAA,QACD;AAAA,QACA,kBAAkB,UAAU,MAAM;AAAA,MACnC;AAAA,IACD;AAAA,IACA,IAAI,eAAe,WAAW;AAAA,MAC7B,MAAM,IAAI,WAAW,uBAAuB,KAAK;AAAA,MACjD,IAAI,EAAE,aAAa;AAAA,QAAW,cAAc,IAAI,EAAE,UAAU,OAAO,KAAK;AAAA,IACzE;AAAA;AAAA,EAGD,OAAO;AAAA,IACN,GAAG,CAAC,OAAO,QAAQ,WAAW;AAAA,MAC7B;AAAA,MACA,MAAM,IAAI,GAAG,YAAc;AAAA,MAC3B,KAAK,IAAI,IAAI,KAAK,IAAI,CAAC,KAAK,KAAK,CAAC;AAAA;AAAA,IAEnC,MAAM,GAAG,QAAQ,OAAO,CAAC,GAAG;AAAA,MAC3B,UAAU,QAAQ,QAAQ;AAAA,MAC1B,WAAW,KAAK,CAAC,SAAS,eAAe,cAAc,gBAAgB,aAAa,GAAG;AAAA,QACtF,EAAE,MAAM,MAAM;AAAA,MACf;AAAA,MACA,YAAY,GAAG,UAAU,MAAM;AAAA,QAC9B,MAAM,KAAK,EAAE,QAAQ,MAAQ;AAAA,QAC7B,QAAQ,EAAE,MAAM,KAAK,CAAC,GAAG,EAAE,MAAM,GAAG,EAAE,GAAgB,KAAK;AAAA,MAC5D;AAAA,MACA,OAAO;AAAA,QACN;AAAA,QACA,UAAU,KAAK;AAAA,QACf;AAAA,QACA;AAAA,QACA;AAAA,QACA,SAAS,QAAQ,IAAI,KAAK;AAAA,QAC1B,eAAe,cAAc,IAAI,KAAK;AAAA,QACtC,cAAc,aAAa,IAAI,KAAK;AAAA,QACpC,gBAAgB,eAAe,IAAI,CAAC,EAAE,MAAM;AAAA,UAC3C,KAAK;AAAA,UACL,OAAO;AAAA,UACP,UAAU,CAAC;AAAA,QACZ;AAAA,QACA,eAAe,cAAc,IAAI,KAAK;AAAA,MACvC;AAAA;AAAA,EAEF;AAAA;",
|
|
8
|
+
"debugId": "C23DFDD520B2D39C64756E2164756E21",
|
|
9
9
|
"names": []
|
|
10
10
|
}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
import {
|
|
2
|
+
isGiven2,
|
|
3
|
+
isAmbiguous2,
|
|
4
|
+
isPlace2,
|
|
5
|
+
isTradename2
|
|
6
|
+
} from "./index-0wpvbdgh.js";
|
|
7
|
+
|
|
8
|
+
// src/lexicon/place.ts
|
|
9
|
+
var JOINING_COMMA = /(?<=[^\s,]),(?=[^\s,])/g;
|
|
10
|
+
function splitWords(s) {
|
|
11
|
+
return s.replace(JOINING_COMMA, (comma, at) => /\d/.test(s[at - 1]) && /\d/.test(s[at + 1]) ? comma : ", ").split(/\s+/).filter(Boolean);
|
|
12
|
+
}
|
|
13
|
+
var FIELD_END = /[,-]+$/;
|
|
14
|
+
function isPlaceWordFlags(flags) {
|
|
15
|
+
return isPlace2(flags) && !isGiven2(flags) && !isAmbiguous2(flags) && !isTradename2(flags);
|
|
16
|
+
}
|
|
17
|
+
function isPlacePhrase(phrase, getFlags) {
|
|
18
|
+
const words = splitWords(phrase).map((w) => w.replace(FIELD_END, "")).filter((w) => w !== "");
|
|
19
|
+
if (words.length < 2)
|
|
20
|
+
return false;
|
|
21
|
+
return words.every((w) => isPlaceWordFlags(getFlags(w) ?? ""));
|
|
22
|
+
}
|
|
23
|
+
function isCommaPlacePhrase(phrase, getFlags) {
|
|
24
|
+
const printed = splitWords(phrase);
|
|
25
|
+
let cut = -1;
|
|
26
|
+
for (let i = 0;i < printed.length - 1; i++)
|
|
27
|
+
if (printed[i].endsWith(","))
|
|
28
|
+
cut = i;
|
|
29
|
+
if (cut < 0)
|
|
30
|
+
return false;
|
|
31
|
+
const bare = (ws) => ws.map((w) => w.replace(FIELD_END, "")).filter((w) => w !== "");
|
|
32
|
+
const words = bare(printed);
|
|
33
|
+
const after = bare(printed.slice(cut + 1));
|
|
34
|
+
if (words.length < 2 || after.length === 0)
|
|
35
|
+
return false;
|
|
36
|
+
return words.every((w) => isPlace2(getFlags(w) ?? "")) && after.every((w) => isPlaceWordFlags(getFlags(w) ?? ""));
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export { splitWords, FIELD_END, isPlaceWordFlags, isPlacePhrase, isCommaPlacePhrase };
|
|
40
|
+
|
|
41
|
+
//# debugId=56C39740150DE0D964756E2164756E21
|
|
42
|
+
//# sourceMappingURL=index-18z8akw8.js.map
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
"version": 3,
|
|
3
3
|
"sources": ["../src/lexicon/place.ts"],
|
|
4
4
|
"sourcesContent": [
|
|
5
|
-
"/**\n * Printed words, and which of them name a place, read off the lexicon's flags alone.\n *\n * Moved here from the classifier so that `/normalize` can ask the same question the\n * classifier asks (is this field a place?) without importing the bloom tables. The\n * classifier re-exports both functions from their old modules.\n */\n\nimport { isAmbiguous, isGiven, isPlace, isTradename } from \"./flags.ts\";\n\n/** A comma with a printed character, not a space or another comma, on each side. */\nconst JOINING_COMMA = /(?<=[^\\s,]),(?=[^\\s,])/g;\n\n/**\n * The printed words of a string, as the classifier, the segmenter and their trainers\n * all read them: split on whitespace, and at a comma that joins two words with no\n * space after it, which separates them as `, ` does (`MUNSTER,EIRE` is `MUNSTER,`\n * `EIRE`; `Quorrish,Zelvine` is `Quorrish,` `Zelvine`). The comma stays on the word before it,\n * where a reader that cares (the classifier's phrase joining) sees the field end.\n * A comma between two digits (`1,000`) is part of the number.\n */\nexport function splitWords(s: string): string[] {\n\treturn s\n\t\t.replace(JOINING_COMMA, (comma, at: number) =>\n\t\t\t/\\d/.test(s[at - 1]!) && /\\d/.test(s[at + 1]!) ? comma : \", \",\n\t\t)\n\t\t.split(/\\s+/)\n\t\t.filter(Boolean);\n}\n\n/** A printed word's trailing `,` or `-`: it ends a field, and no phrase continues through it. */\nexport const FIELD_END = /[,-]+$/;\n\n/**\n * One word's flags read as a place: a place (`G`) and not a given name (`F`, `M`, `N`),\n * an ambiguous name (`A`) or a school's name (`T`). An unknown word (`\"\"`, which\n * `classesOf` reads as `F`) is never a place.\n */\nexport function isPlaceWordFlags(flags: string): boolean {\n\treturn isPlace(flags) && !isGiven(flags) && !isAmbiguous(flags) && !isTradename(flags);\n}\n\n/**\n * A place phrase: two or more printed words (a field-ending `,` or `-` set aside), every\n * one a place (`G`) and none a given name (`F`, `M`, `N`, `A`) or a school's name (`T`).\n * `Varrow Pellinor` (`Varrow:SLG Pellinor:G`) is one; `Lirabel Pellinor` (`Lirabel:FG`, a\n * given name) and `Corrandel Pellinor` (`Corrandel:TG`, a school's name) are not. A surname that is also a place\n * does not make a person: a person's name has a given name. The runtime answers a place\n * phrase `location` without the regression, as it answers one word flagged G.\n */\nexport function isPlacePhrase(phrase: string, getFlags: (name: string) => string | null): boolean {\n\tconst words = splitWords(phrase)\n\t\t.map((w) => w.replace(FIELD_END, \"\"))\n\t\t.filter((w) => w !== \"\");\n\tif (words.length < 2) return false;\n\treturn words.every((w) => isPlaceWordFlags(getFlags(w) ?? \"\"));\n}\n"
|
|
5
|
+
"/**\n * Printed words, and which of them name a place, read off the lexicon's flags alone.\n *\n * Moved here from the classifier so that `/normalize` can ask the same question the\n * classifier asks (is this field a place?) without importing the bloom tables. The\n * classifier re-exports both functions from their old modules.\n */\n\nimport { isAmbiguous, isGiven, isPlace, isTradename } from \"./flags.ts\";\n\n/** A comma with a printed character, not a space or another comma, on each side. */\nconst JOINING_COMMA = /(?<=[^\\s,]),(?=[^\\s,])/g;\n\n/**\n * The printed words of a string, as the classifier, the segmenter and their trainers\n * all read them: split on whitespace, and at a comma that joins two words with no\n * space after it, which separates them as `, ` does (`MUNSTER,EIRE` is `MUNSTER,`\n * `EIRE`; `Quorrish,Zelvine` is `Quorrish,` `Zelvine`). The comma stays on the word before it,\n * where a reader that cares (the classifier's phrase joining) sees the field end.\n * A comma between two digits (`1,000`) is part of the number.\n */\nexport function splitWords(s: string): string[] {\n\treturn s\n\t\t.replace(JOINING_COMMA, (comma, at: number) =>\n\t\t\t/\\d/.test(s[at - 1]!) && /\\d/.test(s[at + 1]!) ? comma : \", \",\n\t\t)\n\t\t.split(/\\s+/)\n\t\t.filter(Boolean);\n}\n\n/** A printed word's trailing `,` or `-`: it ends a field, and no phrase continues through it. */\nexport const FIELD_END = /[,-]+$/;\n\n/**\n * One word's flags read as a place: a place (`G`) and not a given name (`F`, `M`, `N`),\n * an ambiguous name (`A`) or a school's name (`T`). An unknown word (`\"\"`, which\n * `classesOf` reads as `F`) is never a place.\n */\nexport function isPlaceWordFlags(flags: string): boolean {\n\treturn isPlace(flags) && !isGiven(flags) && !isAmbiguous(flags) && !isTradename(flags);\n}\n\n/**\n * A place phrase: two or more printed words (a field-ending `,` or `-` set aside), every\n * one a place (`G`) and none a given name (`F`, `M`, `N`, `A`) or a school's name (`T`).\n * `Varrow Pellinor` (`Varrow:SLG Pellinor:G`) is one; `Lirabel Pellinor` (`Lirabel:FG`, a\n * given name) and `Corrandel Pellinor` (`Corrandel:TG`, a school's name) are not. A surname that is also a place\n * does not make a person: a person's name has a given name. The runtime answers a place\n * phrase `location` without the regression, as it answers one word flagged G.\n */\nexport function isPlacePhrase(phrase: string, getFlags: (name: string) => string | null): boolean {\n\tconst words = splitWords(phrase)\n\t\t.map((w) => w.replace(FIELD_END, \"\"))\n\t\t.filter((w) => w !== \"\");\n\tif (words.length < 2) return false;\n\treturn words.every((w) => isPlaceWordFlags(getFlags(w) ?? \"\"));\n}\n\n/**\n * A place cut by a comma: two or more printed words, every one a place (`G`), at least one\n * before the last ending its field with `,`, and every word after the last such comma a\n * place and nothing else (`isPlaceWordFlags`). The words before the comma may also be a\n * given name or a school's name: `Victoria, Australia` (`Victoria:FG`, `Australia:G`) and\n * `Sydney, Australia` (`Sydney:FGT`) are ones; `Victoria Australia` (no comma) and\n * `Australia, Victoria` (a given name after the comma) are not.\n *\n * The comma is the signal only where the document prints given names first. A person is\n * printed `Surname, Given`, and under a surname-first or an undeclared order a given name\n * that is also a place may follow the comma; a school is printed `School, State`\n * (`Claddagh, Utah`) in surname-first documents. So the runtime answers this phrase\n * `location` only for a document that declares `given-first`; `isPlacePhrase` answers\n * under every order.\n */\nexport function isCommaPlacePhrase(\n\tphrase: string,\n\tgetFlags: (name: string) => string | null,\n): boolean {\n\tconst printed = splitWords(phrase);\n\tlet cut = -1;\n\tfor (let i = 0; i < printed.length - 1; i++) if (printed[i]!.endsWith(\",\")) cut = i;\n\tif (cut < 0) return false;\n\tconst bare = (ws: string[]) => ws.map((w) => w.replace(FIELD_END, \"\")).filter((w) => w !== \"\");\n\tconst words = bare(printed);\n\tconst after = bare(printed.slice(cut + 1));\n\tif (words.length < 2 || after.length === 0) return false;\n\treturn (\n\t\twords.every((w) => isPlace(getFlags(w) ?? \"\")) &&\n\t\tafter.every((w) => isPlaceWordFlags(getFlags(w) ?? \"\"))\n\t);\n}\n"
|
|
6
6
|
],
|
|
7
|
-
"mappings": ";;;;;;;;AAWA,IAAM,gBAAgB;AAUf,SAAS,UAAU,CAAC,GAAqB;AAAA,EAC/C,OAAO,EACL,QAAQ,eAAe,CAAC,OAAO,OAC/B,KAAK,KAAK,EAAE,KAAK,EAAG,KAAK,KAAK,KAAK,EAAE,KAAK,EAAG,IAAI,QAAQ,IAC1D,EACC,MAAM,KAAK,EACX,OAAO,OAAO;AAAA;AAIV,IAAM,YAAY;AAOlB,SAAS,gBAAgB,CAAC,OAAwB;AAAA,EACxD,OAAO,SAAQ,KAAK,KAAK,CAAC,SAAQ,KAAK,KAAK,CAAC,aAAY,KAAK,KAAK,CAAC,aAAY,KAAK;AAAA;AAW/E,SAAS,aAAa,CAAC,QAAgB,UAAoD;AAAA,EACjG,MAAM,QAAQ,WAAW,MAAM,EAC7B,IAAI,CAAC,MAAM,EAAE,QAAQ,WAAW,EAAE,CAAC,EACnC,OAAO,CAAC,MAAM,MAAM,EAAE;AAAA,EACxB,IAAI,MAAM,SAAS;AAAA,IAAG,OAAO;AAAA,EAC7B,OAAO,MAAM,MAAM,CAAC,MAAM,iBAAiB,SAAS,CAAC,KAAK,EAAE,CAAC;AAAA;",
|
|
8
|
-
"debugId": "
|
|
7
|
+
"mappings": ";;;;;;;;AAWA,IAAM,gBAAgB;AAUf,SAAS,UAAU,CAAC,GAAqB;AAAA,EAC/C,OAAO,EACL,QAAQ,eAAe,CAAC,OAAO,OAC/B,KAAK,KAAK,EAAE,KAAK,EAAG,KAAK,KAAK,KAAK,EAAE,KAAK,EAAG,IAAI,QAAQ,IAC1D,EACC,MAAM,KAAK,EACX,OAAO,OAAO;AAAA;AAIV,IAAM,YAAY;AAOlB,SAAS,gBAAgB,CAAC,OAAwB;AAAA,EACxD,OAAO,SAAQ,KAAK,KAAK,CAAC,SAAQ,KAAK,KAAK,CAAC,aAAY,KAAK,KAAK,CAAC,aAAY,KAAK;AAAA;AAW/E,SAAS,aAAa,CAAC,QAAgB,UAAoD;AAAA,EACjG,MAAM,QAAQ,WAAW,MAAM,EAC7B,IAAI,CAAC,MAAM,EAAE,QAAQ,WAAW,EAAE,CAAC,EACnC,OAAO,CAAC,MAAM,MAAM,EAAE;AAAA,EACxB,IAAI,MAAM,SAAS;AAAA,IAAG,OAAO;AAAA,EAC7B,OAAO,MAAM,MAAM,CAAC,MAAM,iBAAiB,SAAS,CAAC,KAAK,EAAE,CAAC;AAAA;AAkBvD,SAAS,kBAAkB,CACjC,QACA,UACU;AAAA,EACV,MAAM,UAAU,WAAW,MAAM;AAAA,EACjC,IAAI,MAAM;AAAA,EACV,SAAS,IAAI,EAAG,IAAI,QAAQ,SAAS,GAAG;AAAA,IAAK,IAAI,QAAQ,GAAI,SAAS,GAAG;AAAA,MAAG,MAAM;AAAA,EAClF,IAAI,MAAM;AAAA,IAAG,OAAO;AAAA,EACpB,MAAM,OAAO,CAAC,OAAiB,GAAG,IAAI,CAAC,MAAM,EAAE,QAAQ,WAAW,EAAE,CAAC,EAAE,OAAO,CAAC,MAAM,MAAM,EAAE;AAAA,EAC7F,MAAM,QAAQ,KAAK,OAAO;AAAA,EAC1B,MAAM,QAAQ,KAAK,QAAQ,MAAM,MAAM,CAAC,CAAC;AAAA,EACzC,IAAI,MAAM,SAAS,KAAK,MAAM,WAAW;AAAA,IAAG,OAAO;AAAA,EACnD,OACC,MAAM,MAAM,CAAC,MAAM,SAAQ,SAAS,CAAC,KAAK,EAAE,CAAC,KAC7C,MAAM,MAAM,CAAC,MAAM,iBAAiB,SAAS,CAAC,KAAK,EAAE,CAAC;AAAA;",
|
|
8
|
+
"debugId": "56C39740150DE0D964756E2164756E21",
|
|
9
9
|
"names": []
|
|
10
10
|
}
|
|
@@ -6,8 +6,9 @@ import {
|
|
|
6
6
|
import {
|
|
7
7
|
splitWords,
|
|
8
8
|
FIELD_END,
|
|
9
|
-
isPlacePhrase
|
|
10
|
-
|
|
9
|
+
isPlacePhrase,
|
|
10
|
+
isCommaPlacePhrase
|
|
11
|
+
} from "./index-18z8akw8.js";
|
|
11
12
|
import {
|
|
12
13
|
TEAM_DESIGNATION_WORDS2,
|
|
13
14
|
partitionTeamSuffix2,
|
|
@@ -2744,10 +2745,11 @@ function phraseClassifier2(names = Lexicon2.default(), options = {}) {
|
|
|
2744
2745
|
const model = (order = "unknown") => byOrder[order] ?? classifier;
|
|
2745
2746
|
const segmenter = createSegmenter2(namesData, options.weights?.segmenter);
|
|
2746
2747
|
const PLACE = { person: 0.05, school: 0.05, location: 0.9 };
|
|
2747
|
-
function placeOverride(phrase) {
|
|
2748
|
+
function placeOverride(phrase, order = "unknown") {
|
|
2748
2749
|
const trimmed = phrase.trim();
|
|
2750
|
+
const flagsOf = (n) => names.getNameFlags(n);
|
|
2749
2751
|
if (splitWords(trimmed).length !== 1)
|
|
2750
|
-
return isPlacePhrase(trimmed,
|
|
2752
|
+
return isPlacePhrase(trimmed, flagsOf) || order === "given-first" && isCommaPlacePhrase(trimmed, flagsOf) ? { ...PLACE } : null;
|
|
2751
2753
|
const flags = names.getNameFlags(trimmed);
|
|
2752
2754
|
if (!flags)
|
|
2753
2755
|
return null;
|
|
@@ -2759,7 +2761,7 @@ function phraseClassifier2(names = Lexicon2.default(), options = {}) {
|
|
|
2759
2761
|
if (isGarbage(phrase))
|
|
2760
2762
|
return { ...ZERO };
|
|
2761
2763
|
const s = subject(phrase);
|
|
2762
|
-
return placeOverride(s) ?? model(order).classifyName(s);
|
|
2764
|
+
return placeOverride(s, order) ?? model(order).classifyName(s);
|
|
2763
2765
|
}
|
|
2764
2766
|
function category(scores) {
|
|
2765
2767
|
const max = Math.max(scores.person, scores.school, scores.location);
|
|
@@ -2797,5 +2799,5 @@ function phraseClassifier2(names = Lexicon2.default(), options = {}) {
|
|
|
2797
2799
|
|
|
2798
2800
|
export { CONNECTORS2, SCHOOL_KEYWORDS2, isStopword2, createSegmenter2, createClassifier2, phraseClassifier2 };
|
|
2799
2801
|
|
|
2800
|
-
//# debugId=
|
|
2801
|
-
//# sourceMappingURL=index-
|
|
2802
|
+
//# debugId=DEAAEE822B81CADB64756E2164756E21
|
|
2803
|
+
//# sourceMappingURL=index-7j32khen.js.map
|