nameparser 2.0.0rc1__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {nameparser-2.0.0rc1/nameparser.egg-info → nameparser-2.1.0}/PKG-INFO +48 -8
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/README.rst +45 -7
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/__init__.py +8 -2
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_config_shim.py +74 -14
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_facade.py +24 -1
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_lexicon.py +134 -14
- nameparser-2.1.0/nameparser/_parser.py +306 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/__init__.py +4 -3
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_assemble.py +25 -7
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_assign.py +84 -6
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_classify.py +3 -1
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_extract.py +9 -1
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_group.py +72 -5
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_post_rules.py +2 -1
- nameparser-2.1.0/nameparser/_pipeline/_script_segment.py +767 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_segment.py +22 -51
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_state.py +24 -10
- nameparser-2.1.0/nameparser/_pipeline/_tokenize.py +208 -0
- nameparser-2.1.0/nameparser/_pipeline/_vocab.py +356 -0
- nameparser-2.1.0/nameparser/_policy.py +916 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_render.py +12 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_types.py +199 -40
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_version.py +2 -2
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/conjunctions.py +5 -1
- nameparser-2.1.0/nameparser/config/maiden_markers.py +70 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/suffixes.py +118 -7
- nameparser-2.1.0/nameparser/config/surnames.py +49 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/titles.py +3 -1
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/locales/__init__.py +22 -0
- nameparser-2.1.0/nameparser/locales/ja.py +228 -0
- nameparser-2.1.0/nameparser/locales/zh.py +113 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0/nameparser.egg-info}/PKG-INFO +48 -8
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser.egg-info/SOURCES.txt +7 -0
- nameparser-2.1.0/nameparser.egg-info/requires.txt +3 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/pyproject.toml +19 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_nicknames.py +39 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_titles.py +13 -0
- nameparser-2.1.0/tests/v2/cases.py +1807 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/conftest.py +10 -15
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_assemble.py +7 -4
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_assign.py +68 -1
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_classify.py +21 -1
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_group.py +92 -1
- nameparser-2.1.0/tests/v2/pipeline/test_script_segment.py +815 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_segment.py +14 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_state.py +15 -2
- nameparser-2.1.0/tests/v2/pipeline/test_tokenize.py +228 -0
- nameparser-2.1.0/tests/v2/pipeline/test_vocab.py +373 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_benchmark.py +94 -12
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_cli.py +13 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_config_shim.py +59 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_contracts.py +28 -8
- nameparser-2.1.0/tests/v2/test_differential.py +696 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_facade.py +69 -0
- nameparser-2.1.0/tests/v2/test_facade_cases.py +153 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_layering.py +18 -8
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_lexicon.py +155 -2
- nameparser-2.1.0/tests/v2/test_locales.py +1202 -0
- nameparser-2.1.0/tests/v2/test_parser.py +813 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_policy.py +338 -5
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_properties.py +219 -17
- nameparser-2.1.0/tests/v2/test_regex_sync.py +366 -0
- nameparser-2.1.0/tests/v2/test_reprs.py +161 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_types.py +144 -2
- nameparser-2.0.0rc1/nameparser/_parser.py +0 -130
- nameparser-2.0.0rc1/nameparser/_pipeline/_tokenize.py +0 -125
- nameparser-2.0.0rc1/nameparser/_pipeline/_vocab.py +0 -126
- nameparser-2.0.0rc1/nameparser/_policy.py +0 -456
- nameparser-2.0.0rc1/nameparser/config/maiden_markers.py +0 -42
- nameparser-2.0.0rc1/tests/v2/cases.py +0 -469
- nameparser-2.0.0rc1/tests/v2/pipeline/test_tokenize.py +0 -91
- nameparser-2.0.0rc1/tests/v2/pipeline/test_vocab.py +0 -40
- nameparser-2.0.0rc1/tests/v2/test_facade_cases.py +0 -75
- nameparser-2.0.0rc1/tests/v2/test_locales.py +0 -522
- nameparser-2.0.0rc1/tests/v2/test_parser.py +0 -317
- nameparser-2.0.0rc1/tests/v2/test_regex_sync.py +0 -151
- nameparser-2.0.0rc1/tests/v2/test_reprs.py +0 -68
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/AUTHORS +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/LICENSE +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/MANIFEST.in +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/__main__.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_locale.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/__init__.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/_invariants.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/bound_first_names.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/capitalization.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/prefixes.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/regexes.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/locales/ru.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/locales/tr_az.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/parser.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/py.typed +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/util.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser.egg-info/dependency_links.txt +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser.egg-info/top_level.txt +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/setup.cfg +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/__init__.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/base.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/conftest.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_bound_first_names.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_brute_force.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_capitalization.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_comma_variants.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_conjunctions.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_constants.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_east_slavic_patronymic_order.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_first_name.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_initials.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_middle_name_as_last.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_output_format.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_prefixes.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_python_api.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_suffixes.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_turkic_patronymic_order.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_variations.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/__init__.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/__init__.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_extract.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_post_rules.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_cases.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_locale.py +0 -0
- {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_render.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: nameparser
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: A simple Python module for parsing human names into their individual components.
|
|
5
5
|
Author-email: Derek Gulbranson <derek73@gmail.com>
|
|
6
6
|
License: LGPL
|
|
@@ -22,6 +22,8 @@ Requires-Python: >=3.11
|
|
|
22
22
|
Description-Content-Type: text/x-rst
|
|
23
23
|
License-File: LICENSE
|
|
24
24
|
License-File: AUTHORS
|
|
25
|
+
Provides-Extra: ja
|
|
26
|
+
Requires-Dist: namedivider-python>=0.4; extra == "ja"
|
|
25
27
|
Dynamic: license-file
|
|
26
28
|
|
|
27
29
|
Name Parser
|
|
@@ -33,6 +35,19 @@ nameparser parses human names into seven fields — title, given, middle,
|
|
|
33
35
|
family, suffix, nickname, maiden. Results are immutable, configuration is
|
|
34
36
|
composable, and locale packs are opt-in.
|
|
35
37
|
|
|
38
|
+
📣 **nameparser 2.0 is out.** Existing ``HumanName`` code keeps working
|
|
39
|
+
through 2.x, and most 1.x code needs no changes. The `migration guide
|
|
40
|
+
<https://nameparser.readthedocs.io/en/latest/migrate.html>`__ has the
|
|
41
|
+
field-by-field map. Please `open an issue
|
|
42
|
+
<https://github.com/derek73/python-nameparser/issues>`__ for anything that
|
|
43
|
+
parses wrong.
|
|
44
|
+
|
|
45
|
+
**2.1 adds East Asian name support.** Chinese, Japanese and Korean names
|
|
46
|
+
written in their own scripts are read family-first, unspaced Korean names
|
|
47
|
+
are split against the census surname list, and CJK honorifics are
|
|
48
|
+
recognized. See `East Asian names
|
|
49
|
+
<https://nameparser.readthedocs.io/en/latest/usage.html#east-asian-names>`__.
|
|
50
|
+
|
|
36
51
|
Installation
|
|
37
52
|
------------
|
|
38
53
|
|
|
@@ -48,16 +63,41 @@ Quick Start Example
|
|
|
48
63
|
.. code-block:: python
|
|
49
64
|
|
|
50
65
|
>>> from nameparser import parse
|
|
51
|
-
>>> name = parse("Dr. Juan Q. Xavier de la Vega III")
|
|
52
|
-
>>> name
|
|
53
|
-
|
|
66
|
+
>>> name = parse("Dr. Juan Q. Xavier de la Vega III (Doc Vega)")
|
|
67
|
+
>>> name
|
|
68
|
+
<ParsedName: [
|
|
69
|
+
title: 'Dr.'
|
|
70
|
+
given: 'Juan'
|
|
71
|
+
middle: 'Q. Xavier'
|
|
72
|
+
family: 'de la Vega'
|
|
73
|
+
suffix: 'III'
|
|
74
|
+
nickname: 'Doc Vega'
|
|
75
|
+
]>
|
|
76
|
+
>>> name.family_base, name.family_particles
|
|
77
|
+
('Vega', 'de la')
|
|
54
78
|
>>> name.render("{family}, {given}")
|
|
55
79
|
'de la Vega, Juan'
|
|
56
80
|
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
81
|
+
>>> parse("김민준").family # Korean: unspaced, split on the census list
|
|
82
|
+
'김'
|
|
83
|
+
>>> parse("高橋 みなみ").family # Japanese: kanji with kana, family first
|
|
84
|
+
'高橋'
|
|
85
|
+
>>> parse("김민준씨").suffix # an honorific written against the name
|
|
86
|
+
'씨'
|
|
87
|
+
>>> parse("г-н Иван Петров").title # Cyrillic title
|
|
88
|
+
'г-н'
|
|
89
|
+
>>> parse("محمد بن سلمان").family # Arabic: بن chains onto the family name
|
|
90
|
+
'بن سلمان'
|
|
91
|
+
|
|
92
|
+
>>> from nameparser import locales, parser_for
|
|
93
|
+
>>> chinese = parser_for(locales.ZH) # Han text does not say which language
|
|
94
|
+
>>> chinese.parse("毛泽东").family # so splitting it is opt-in
|
|
95
|
+
'毛'
|
|
96
|
+
>>> russian = parser_for(locales.RU)
|
|
97
|
+
>>> russian.parse("Сидоров Иван Петрович").family
|
|
98
|
+
'Сидоров'
|
|
99
|
+
>>> locales.available()
|
|
100
|
+
('ja', 'ru', 'tr_az', 'zh')
|
|
61
101
|
|
|
62
102
|
Learn more
|
|
63
103
|
----------
|
|
@@ -7,6 +7,19 @@ nameparser parses human names into seven fields — title, given, middle,
|
|
|
7
7
|
family, suffix, nickname, maiden. Results are immutable, configuration is
|
|
8
8
|
composable, and locale packs are opt-in.
|
|
9
9
|
|
|
10
|
+
📣 **nameparser 2.0 is out.** Existing ``HumanName`` code keeps working
|
|
11
|
+
through 2.x, and most 1.x code needs no changes. The `migration guide
|
|
12
|
+
<https://nameparser.readthedocs.io/en/latest/migrate.html>`__ has the
|
|
13
|
+
field-by-field map. Please `open an issue
|
|
14
|
+
<https://github.com/derek73/python-nameparser/issues>`__ for anything that
|
|
15
|
+
parses wrong.
|
|
16
|
+
|
|
17
|
+
**2.1 adds East Asian name support.** Chinese, Japanese and Korean names
|
|
18
|
+
written in their own scripts are read family-first, unspaced Korean names
|
|
19
|
+
are split against the census surname list, and CJK honorifics are
|
|
20
|
+
recognized. See `East Asian names
|
|
21
|
+
<https://nameparser.readthedocs.io/en/latest/usage.html#east-asian-names>`__.
|
|
22
|
+
|
|
10
23
|
Installation
|
|
11
24
|
------------
|
|
12
25
|
|
|
@@ -22,16 +35,41 @@ Quick Start Example
|
|
|
22
35
|
.. code-block:: python
|
|
23
36
|
|
|
24
37
|
>>> from nameparser import parse
|
|
25
|
-
>>> name = parse("Dr. Juan Q. Xavier de la Vega III")
|
|
26
|
-
>>> name
|
|
27
|
-
|
|
38
|
+
>>> name = parse("Dr. Juan Q. Xavier de la Vega III (Doc Vega)")
|
|
39
|
+
>>> name
|
|
40
|
+
<ParsedName: [
|
|
41
|
+
title: 'Dr.'
|
|
42
|
+
given: 'Juan'
|
|
43
|
+
middle: 'Q. Xavier'
|
|
44
|
+
family: 'de la Vega'
|
|
45
|
+
suffix: 'III'
|
|
46
|
+
nickname: 'Doc Vega'
|
|
47
|
+
]>
|
|
48
|
+
>>> name.family_base, name.family_particles
|
|
49
|
+
('Vega', 'de la')
|
|
28
50
|
>>> name.render("{family}, {given}")
|
|
29
51
|
'de la Vega, Juan'
|
|
30
52
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
53
|
+
>>> parse("김민준").family # Korean: unspaced, split on the census list
|
|
54
|
+
'김'
|
|
55
|
+
>>> parse("高橋 みなみ").family # Japanese: kanji with kana, family first
|
|
56
|
+
'高橋'
|
|
57
|
+
>>> parse("김민준씨").suffix # an honorific written against the name
|
|
58
|
+
'씨'
|
|
59
|
+
>>> parse("г-н Иван Петров").title # Cyrillic title
|
|
60
|
+
'г-н'
|
|
61
|
+
>>> parse("محمد بن سلمان").family # Arabic: بن chains onto the family name
|
|
62
|
+
'بن سلمان'
|
|
63
|
+
|
|
64
|
+
>>> from nameparser import locales, parser_for
|
|
65
|
+
>>> chinese = parser_for(locales.ZH) # Han text does not say which language
|
|
66
|
+
>>> chinese.parse("毛泽东").family # so splitting it is opt-in
|
|
67
|
+
'毛'
|
|
68
|
+
>>> russian = parser_for(locales.RU)
|
|
69
|
+
>>> russian.parse("Сидоров Иван Петрович").family
|
|
70
|
+
'Сидоров'
|
|
71
|
+
>>> locales.available()
|
|
72
|
+
('ja', 'ru', 'tr_az', 'zh')
|
|
35
73
|
|
|
36
74
|
Learn more
|
|
37
75
|
----------
|
|
@@ -11,6 +11,7 @@ from nameparser._locale import Locale
|
|
|
11
11
|
from nameparser._parser import Parser, parse, parser_for
|
|
12
12
|
from nameparser._policy import (
|
|
13
13
|
DEFAULT_NICKNAME_DELIMITERS,
|
|
14
|
+
DEFAULT_SCRIPT_ORDERS,
|
|
14
15
|
FAMILY_FIRST,
|
|
15
16
|
FAMILY_FIRST_GIVEN_LAST,
|
|
16
17
|
GIVEN_FIRST,
|
|
@@ -18,12 +19,16 @@ from nameparser._policy import (
|
|
|
18
19
|
PatronymicRule,
|
|
19
20
|
Policy,
|
|
20
21
|
PolicyPatch,
|
|
22
|
+
Script,
|
|
21
23
|
)
|
|
22
24
|
from nameparser._types import (
|
|
25
|
+
STABLE_TAGS,
|
|
23
26
|
Ambiguity,
|
|
24
27
|
AmbiguityKind,
|
|
25
28
|
ParsedName,
|
|
26
29
|
Role,
|
|
30
|
+
Segmentation,
|
|
31
|
+
Segmenter,
|
|
27
32
|
Span,
|
|
28
33
|
Token,
|
|
29
34
|
)
|
|
@@ -33,8 +38,9 @@ __all__ = [
|
|
|
33
38
|
"HumanName",
|
|
34
39
|
# v2 core
|
|
35
40
|
"Span", "Role", "Token", "Ambiguity", "AmbiguityKind", "ParsedName",
|
|
36
|
-
"
|
|
41
|
+
"STABLE_TAGS", "Segmentation", "Segmenter",
|
|
42
|
+
"Lexicon", "Policy", "PolicyPatch", "PatronymicRule", "Script", "UNSET",
|
|
37
43
|
"GIVEN_FIRST", "FAMILY_FIRST", "FAMILY_FIRST_GIVEN_LAST",
|
|
38
|
-
"DEFAULT_NICKNAME_DELIMITERS", "Locale",
|
|
44
|
+
"DEFAULT_NICKNAME_DELIMITERS", "DEFAULT_SCRIPT_ORDERS", "Locale",
|
|
39
45
|
"Parser", "parse", "parser_for",
|
|
40
46
|
]
|
|
@@ -33,6 +33,28 @@ from nameparser._policy import PatronymicRule, Policy
|
|
|
33
33
|
from nameparser.util import lc
|
|
34
34
|
|
|
35
35
|
|
|
36
|
+
#: The eight multi-word entries the pre-2.0 DEFAULT vocabulary shipped.
|
|
37
|
+
#: Provably inert in every release (they can never match; see the
|
|
38
|
+
#: 2.0.0 release log), so dropping them from a restored legacy pickle
|
|
39
|
+
#: changes no parse -- and keeps the multi-word warning from firing
|
|
40
|
+
#: eight times, with wrong advice, at library-internal lines, on the
|
|
41
|
+
#: first parse after a supported 1.3/1.4 pickle upgrade.
|
|
42
|
+
#:
|
|
43
|
+
#: Gated on ALL EIGHT being present across the two fields -- the
|
|
44
|
+
#: signature of a pre-2.0 blob, which froze the complete shipped set.
|
|
45
|
+
#: A round-trip of 2.0-era state that carries FEWER than all eight
|
|
46
|
+
#: keeps them (a user who deliberately re-added one or two does not
|
|
47
|
+
#: match the signature, and .copy() never subtracts either). The trade:
|
|
48
|
+
#: a 2.0 user who re-adds ALL eight exact strings is indistinguishable
|
|
49
|
+
#: from a legacy blob and loses them on the next unpickle.
|
|
50
|
+
_LEGACY_DEAD_ENTRIES = {
|
|
51
|
+
"titles": frozenset({"chargé d'affaires"}),
|
|
52
|
+
"suffix_acronyms": frozenset({
|
|
53
|
+
"leed ap", "nicet i", "nicet ii", "nicet iii", "nicet iv",
|
|
54
|
+
"psm i", "psm ii"}),
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
36
58
|
def _reject_bare_str_or_bytes(value: object, expected: str) -> None:
|
|
37
59
|
# A bare string is an iterable of its characters, so e.g. SetManager('dr')
|
|
38
60
|
# would silently shred it into {'d', 'r'} instead of raising -- shared by
|
|
@@ -964,10 +986,23 @@ class Constants:
|
|
|
964
986
|
removal path.
|
|
965
987
|
"""
|
|
966
988
|
from nameparser.config.maiden_markers import MAIDEN_MARKERS
|
|
989
|
+
from nameparser.config.suffixes import GLUED_HONORIFICS
|
|
990
|
+
from nameparser.config.surnames import KOREAN_SURNAMES
|
|
967
991
|
acronyms = frozenset(self.suffix_acronyms)
|
|
968
992
|
particles = frozenset(self.prefixes)
|
|
969
993
|
bound = frozenset(self.bound_first_names)
|
|
970
994
|
ambiguous_acronyms = frozenset(self.suffix_acronyms_ambiguous) & acronyms
|
|
995
|
+
# Drop any ambiguous acronym from the word set rather than the
|
|
996
|
+
# other way round. Lexicon forbids the overlap because the word
|
|
997
|
+
# branch bypasses the period gate, and adding an ambiguous
|
|
998
|
+
# acronym to suffix_not_acronyms is INERT in v1 anyway:
|
|
999
|
+
# is_suffix already accepts it via the acronym branch, and
|
|
1000
|
+
# reserve_last keeps it as the surname. So ignoring the
|
|
1001
|
+
# addition reproduces v1 ("Jack Ma" keeps last='Ma'), where
|
|
1002
|
+
# dropping it from the AMBIGUOUS set instead ungated the word
|
|
1003
|
+
# and lost the family name -- a silent misparse worse than the
|
|
1004
|
+
# raise it avoided.
|
|
1005
|
+
suffix_words = frozenset(self.suffix_not_acronyms) - ambiguous_acronyms
|
|
971
1006
|
# keep in sync with _lexicon._default_lexicon() (pinned by
|
|
972
1007
|
# tests/v2/test_config_shim.py::test_snapshot_field_translation)
|
|
973
1008
|
lexicon = Lexicon(
|
|
@@ -994,18 +1029,7 @@ class Constants:
|
|
|
994
1029
|
if e == " ".join(e.split())
|
|
995
1030
|
) if t),
|
|
996
1031
|
suffix_acronyms=acronyms,
|
|
997
|
-
|
|
998
|
-
# the other way round. Lexicon forbids the overlap because
|
|
999
|
-
# the word branch bypasses the period gate, and adding an
|
|
1000
|
-
# ambiguous acronym to suffix_not_acronyms is INERT in v1
|
|
1001
|
-
# anyway: is_suffix already accepts it via the acronym
|
|
1002
|
-
# branch, and reserve_last keeps it as the surname. So
|
|
1003
|
-
# ignoring the addition reproduces v1 ("Jack Ma" keeps
|
|
1004
|
-
# last='Ma'), where dropping it from the AMBIGUOUS set
|
|
1005
|
-
# instead ungated the word and lost the family name --
|
|
1006
|
-
# a silent misparse worse than the raise it avoided.
|
|
1007
|
-
suffix_words=frozenset(
|
|
1008
|
-
self.suffix_not_acronyms) - ambiguous_acronyms,
|
|
1032
|
+
suffix_words=suffix_words,
|
|
1009
1033
|
# Intersect with acronyms: Lexicon enforces ambiguous <=
|
|
1010
1034
|
# acronyms; v1 behaves the same when an acronym is deleted
|
|
1011
1035
|
# but its ambiguous entry lingers (the entry stops
|
|
@@ -1039,6 +1063,22 @@ class Constants:
|
|
|
1039
1063
|
# v1 Constants has no manager for these (#274 is 2.0
|
|
1040
1064
|
# behavior); the data module is the only source
|
|
1041
1065
|
maiden_markers=frozenset(MAIDEN_MARKERS),
|
|
1066
|
+
# likewise no v1 manager: the unspaced-name segmentation
|
|
1067
|
+
# vocabulary is 2.0 behavior (#271), so it rides in the
|
|
1068
|
+
# snapshot only -- v1's Constants surface stays frozen.
|
|
1069
|
+
# Unwrapped where maiden_markers above is wrapped: this
|
|
1070
|
+
# module is born frozen (#293), so no wrap
|
|
1071
|
+
surnames=KOREAN_SURNAMES,
|
|
1072
|
+
# likewise no v1 manager: the glued-honorific tail set is
|
|
1073
|
+
# 2.1 behavior (#308), so it rides in the snapshot only.
|
|
1074
|
+
# Wrapped, unlike surnames above: suffixes.py is still a
|
|
1075
|
+
# mutable v1 module, not born-frozen like surnames.py
|
|
1076
|
+
# (#293). Intersect with the word set: Lexicon enforces
|
|
1077
|
+
# tails <= suffix_words, and v1 semantics are that deleting
|
|
1078
|
+
# a suffix word turns the behavior off -- a lingering tail
|
|
1079
|
+
# simply stops mattering, the same rule ambiguous_acronyms
|
|
1080
|
+
# gets against suffix_acronyms above.
|
|
1081
|
+
honorific_tails=frozenset(GLUED_HONORIFICS) & suffix_words,
|
|
1042
1082
|
# TupleManager is dict[str, object] (v1 parity: values were
|
|
1043
1083
|
# never statically str-typed); every real entry is a str,
|
|
1044
1084
|
# same assumption _DelimiterManager's sentinel lookup makes
|
|
@@ -1103,10 +1143,30 @@ class Constants:
|
|
|
1103
1143
|
self.__init__() # type: ignore[misc] # defaults, then overlay
|
|
1104
1144
|
# (managers re-wrapped below so _on_change points at THIS
|
|
1105
1145
|
# instance, not whatever produced the incoming state)
|
|
1146
|
+
managers: dict[str, SetManager] = {}
|
|
1106
1147
|
for name in _SET_FIELDS:
|
|
1107
1148
|
if name in state:
|
|
1108
|
-
|
|
1109
|
-
state[name], _on_change=self._bump)
|
|
1149
|
+
managers[name] = SetManager(
|
|
1150
|
+
state[name], _on_change=self._bump) # type: ignore[arg-type]
|
|
1151
|
+
# SetManager normalized on construction, so the frozen 1.3/1.4
|
|
1152
|
+
# vocabulary's dead entries are matchable in their normalized
|
|
1153
|
+
# spelling here. Subtract only when ALL EIGHT are present --
|
|
1154
|
+
# the pre-2.0 signature; a 2.0 user who re-added one or two
|
|
1155
|
+
# keeps them through a round-trip (see _LEGACY_DEAD_ENTRIES).
|
|
1156
|
+
legacy = all(
|
|
1157
|
+
name in managers and entry in managers[name]
|
|
1158
|
+
for name, entries in _LEGACY_DEAD_ENTRIES.items()
|
|
1159
|
+
for entry in entries
|
|
1160
|
+
)
|
|
1161
|
+
for name, manager in managers.items():
|
|
1162
|
+
if legacy:
|
|
1163
|
+
# Reach past the public discard() deliberately: this is
|
|
1164
|
+
# part of restoring the state, not a mutation of it, and
|
|
1165
|
+
# must not bump the generation of an instance that is
|
|
1166
|
+
# still being built.
|
|
1167
|
+
manager._elements -= _LEGACY_DEAD_ENTRIES.get(
|
|
1168
|
+
name, frozenset())
|
|
1169
|
+
object.__setattr__(self, name, manager)
|
|
1110
1170
|
if "capitalization_exceptions" in state:
|
|
1111
1171
|
object.__setattr__(
|
|
1112
1172
|
self, "capitalization_exceptions", TupleManager(
|
|
@@ -335,6 +335,8 @@ class HumanName:
|
|
|
335
335
|
else:
|
|
336
336
|
raise TypeError(
|
|
337
337
|
f"{member} must be a str, list, or None, got {value!r}")
|
|
338
|
+
# v1 setters stay on replace(): revise()'s vocabulary tags would
|
|
339
|
+
# change v1 parity
|
|
338
340
|
self._parsed = self._parsed.replace(
|
|
339
341
|
**{_V2_FIELD.get(member, member): joined})
|
|
340
342
|
|
|
@@ -616,7 +618,28 @@ class HumanName:
|
|
|
616
618
|
"slicing a HumanName was removed in 2.0 (#258); access "
|
|
617
619
|
"the named attributes instead"
|
|
618
620
|
)
|
|
619
|
-
|
|
621
|
+
# Role is a StrEnum, so Role members (and the plain 'given'/
|
|
622
|
+
# 'family' strings) reach here too -- translate to the v1
|
|
623
|
+
# spelling the facade actually exposes as attributes.
|
|
624
|
+
return getattr(self, _V1_SPELLING.get(key, key))
|
|
625
|
+
|
|
626
|
+
def __setattr__(self, name: str, value: object) -> None:
|
|
627
|
+
# "given"/"family" are the 2.0 spellings of first/last; the
|
|
628
|
+
# facade has no such attributes, so plain assignment creates a
|
|
629
|
+
# stray instance attribute while the parse (and .first/.last)
|
|
630
|
+
# keeps the old value -- a silently forked name. Warn but
|
|
631
|
+
# still set: ad-hoc attribute stashing is a legal v1 pattern,
|
|
632
|
+
# so any code that worked keeps working. Only these two names
|
|
633
|
+
# warn -- the other five 2.0 field names are real properties
|
|
634
|
+
# whose setters work, and Role members reach here as their
|
|
635
|
+
# string values (StrEnum).
|
|
636
|
+
if name in _V1_SPELLING:
|
|
637
|
+
warnings.warn(
|
|
638
|
+
f"assigning HumanName.{name} creates an inert attribute; "
|
|
639
|
+
f"the parse is unchanged -- use .{_V1_SPELLING[name]} "
|
|
640
|
+
f"(the v1 spelling) to update the name",
|
|
641
|
+
UserWarning, stacklevel=2)
|
|
642
|
+
super().__setattr__(name, value)
|
|
620
643
|
|
|
621
644
|
def as_dict(self, include_empty: bool = True) -> dict[str, str]:
|
|
622
645
|
"""The seven v1-named components as a dict; include_empty=False
|
|
@@ -9,9 +9,11 @@ from __future__ import annotations
|
|
|
9
9
|
|
|
10
10
|
import dataclasses
|
|
11
11
|
import functools
|
|
12
|
+
import sys
|
|
13
|
+
import warnings
|
|
12
14
|
from collections.abc import Iterable, Mapping
|
|
13
15
|
from dataclasses import dataclass, field
|
|
14
|
-
from types import MappingProxyType
|
|
16
|
+
from types import FrameType, MappingProxyType
|
|
15
17
|
from typing import cast
|
|
16
18
|
|
|
17
19
|
#: Vocabulary set fields, in declaration order. add()/remove() operate
|
|
@@ -21,20 +23,44 @@ from typing import cast
|
|
|
21
23
|
_VOCAB_FIELDS = (
|
|
22
24
|
"titles", "given_name_titles", "suffix_acronyms", "suffix_words",
|
|
23
25
|
"suffix_acronyms_ambiguous", "particles", "particles_ambiguous",
|
|
24
|
-
"conjunctions", "bound_given_names", "maiden_markers",
|
|
26
|
+
"conjunctions", "bound_given_names", "maiden_markers", "surnames",
|
|
27
|
+
"honorific_tails",
|
|
25
28
|
)
|
|
26
29
|
|
|
27
|
-
#: (marker, base, why) triples. Each marker
|
|
28
|
-
#: base vocabulary are read and carries no vocabulary of its own,
|
|
29
|
-
#: entry outside the base is a configuration mistake -- but the
|
|
30
|
-
#: differs per pair, and the reason is recorded here rather
|
|
31
|
-
#: generalized, because an orphan is NOT simply inert
|
|
30
|
+
#: (marker, base, why) triples. Each marker QUALIFIES how entries of
|
|
31
|
+
#: its base vocabulary are read and carries no vocabulary of its own,
|
|
32
|
+
#: so an entry outside the base is a configuration mistake -- but the
|
|
33
|
+
#: mistake differs per pair, and the reason is recorded here rather
|
|
34
|
+
#: than generalized, because an orphan is NOT simply inert. Nor is the
|
|
35
|
+
#: qualification one-directional: the first two NARROW their base (an
|
|
36
|
+
#: entry is read as vocabulary in fewer places), while honorific_tails
|
|
37
|
+
#: WIDENS it, granting a suffix word the glued position on top of the
|
|
38
|
+
#: whole-token match every suffix word already gets.
|
|
32
39
|
#:
|
|
33
40
|
#: * particles_ambiguous: _assign keys on the tag alone, so an orphan
|
|
34
41
|
#: makes the parse emit a spurious particle-or-given ambiguity.
|
|
35
42
|
#: * suffix_acronyms_ambiguous: _vocab returns True on the ambiguous
|
|
36
43
|
#: set before testing suffix_acronyms, so an orphan silently turns a
|
|
37
44
|
#: word into a period-gated suffix.
|
|
45
|
+
#: * honorific_tails: script_segment peels the tail into its own token
|
|
46
|
+
#: before classify ever runs, so an orphan splits the name and leaves
|
|
47
|
+
#: the fragment stranded inside it -- worse than not peeling at all.
|
|
48
|
+
#: Its base is deliberately NARROWER than what actually claims the
|
|
49
|
+
#: peeled piece: suffix_as_written ORs suffix_words with the
|
|
50
|
+
#: non-ambiguous suffix_acronyms, so a tail listed only as an acronym
|
|
51
|
+
#: would classify fine yet is rejected here. Accepted, and a decision
|
|
52
|
+
#: rather than an oversight -- the three-term predicate is easy to
|
|
53
|
+
#: get wrong in the dangerous direction (an ambiguous acronym admitted
|
|
54
|
+
#: as a tail would peel a period-gated word off a real name), and
|
|
55
|
+
#: nothing needs the acronym half: the shipped tails are CJK
|
|
56
|
+
#: honorifics, which are words.
|
|
57
|
+
#: The same relation is asserted a second time in config/suffixes.py,
|
|
58
|
+
#: over the raw GLUED_HONORIFICS/SUFFIX_NOT_ACRONYMS constants at
|
|
59
|
+
#: import. The two are not redundant in the way they look: that one
|
|
60
|
+
#: is an `assert`, stripped under `python -O`, while the check here
|
|
61
|
+
#: raises unconditionally -- so under -O this is what still holds the
|
|
62
|
+
#: SHIPPED vocabulary to the invariant, as it is the only thing that
|
|
63
|
+
#: ever held a caller's own.
|
|
38
64
|
#:
|
|
39
65
|
#: given_name_titles is deliberately NOT here and has no check of its
|
|
40
66
|
#: own -- see the note in __post_init__ for why every attempt at one
|
|
@@ -44,6 +70,8 @@ _SUBSET_FIELDS = (
|
|
|
44
70
|
"an orphan emits a spurious particle-or-given ambiguity"),
|
|
45
71
|
("suffix_acronyms_ambiguous", "suffix_acronyms",
|
|
46
72
|
"an orphan silently becomes a period-gated suffix"),
|
|
73
|
+
("honorific_tails", "suffix_words",
|
|
74
|
+
"an orphan splits the name and leaves the tail inside it"),
|
|
47
75
|
)
|
|
48
76
|
|
|
49
77
|
|
|
@@ -112,7 +140,27 @@ def _reject_buffer(value: object, label: str, plural: str) -> None:
|
|
|
112
140
|
)
|
|
113
141
|
|
|
114
142
|
|
|
115
|
-
def
|
|
143
|
+
def _warn_dead_entry(message: str) -> None:
|
|
144
|
+
# A fixed stacklevel always lands on library internals: the call
|
|
145
|
+
# depth differs per entry point (Lexicon(), add(), unpickle,
|
|
146
|
+
# dataclasses.replace, and the v1 shim's lazy snapshot -- built on
|
|
147
|
+
# the first parse after a Constants mutation, several facade frames
|
|
148
|
+
# below the user's own add()). Walk out of this module (and
|
|
149
|
+
# dataclasses' replace frames, and the facade layer that builds
|
|
150
|
+
# lexicons on the caller's behalf) so the warning points at the
|
|
151
|
+
# caller's own line.
|
|
152
|
+
level = 2
|
|
153
|
+
frame: FrameType | None = sys._getframe(1)
|
|
154
|
+
while frame is not None and frame.f_globals.get("__name__") in (
|
|
155
|
+
__name__, "dataclasses",
|
|
156
|
+
"nameparser._config_shim", "nameparser._facade"):
|
|
157
|
+
frame, level = frame.f_back, level + 1
|
|
158
|
+
warnings.warn(message, UserWarning, stacklevel=level)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _normset(
|
|
162
|
+
entries: Iterable[str], field_name: str, warn: bool = True,
|
|
163
|
+
) -> frozenset[str]:
|
|
116
164
|
# Reject a bare str before iterating: iterating "dr" would silently
|
|
117
165
|
# yield the single characters {'d', 'r'} -- the set(str) footgun on
|
|
118
166
|
# the primary customization surface.
|
|
@@ -159,6 +207,25 @@ def _normset(entries: Iterable[str], field_name: str) -> frozenset[str]:
|
|
|
159
207
|
f"Lexicon.{field_name} entry {w!r} normalizes to empty "
|
|
160
208
|
f"(lowercase + strip periods/whitespace leaves nothing)"
|
|
161
209
|
)
|
|
210
|
+
# Every field but given_name_titles is matched one word at a
|
|
211
|
+
# time, so a multi-word entry can never match -- the library
|
|
212
|
+
# itself shipped eight such dead entries for years (repaired
|
|
213
|
+
# 2026-07-26). Warn, never raise: an inert entry produces
|
|
214
|
+
# nothing, and the given_name_titles precedent says a raise
|
|
215
|
+
# here costs working configurations (see __post_init__).
|
|
216
|
+
# warn=False is _edit()'s pass (both ops): add() warns via the
|
|
217
|
+
# new instance's __post_init__; remove() stores nothing, so
|
|
218
|
+
# warning there would name entries the caller is trying to get
|
|
219
|
+
# RID of, with "split it" advice that makes no sense for a
|
|
220
|
+
# no-op.
|
|
221
|
+
if (warn and field_name != "given_name_titles"
|
|
222
|
+
# interior whitespace test; split() covers all Unicode
|
|
223
|
+
# whitespace
|
|
224
|
+
and n != "".join(n.split())):
|
|
225
|
+
_warn_dead_entry(
|
|
226
|
+
f"Lexicon.{field_name} entries are matched one word at "
|
|
227
|
+
f"a time; multi-word entry {w!r} can never match. "
|
|
228
|
+
f"Split it into separate entries")
|
|
162
229
|
normalized.add(n)
|
|
163
230
|
return frozenset(normalized)
|
|
164
231
|
|
|
@@ -214,6 +281,14 @@ def _normpairs(
|
|
|
214
281
|
f"empty (lowercase + strip periods/whitespace leaves "
|
|
215
282
|
f"nothing)"
|
|
216
283
|
)
|
|
284
|
+
# capitalized() looks words up one at a time (the _WORD regex
|
|
285
|
+
# never yields spaces), so a multi-word key is unreachable.
|
|
286
|
+
# interior whitespace test; split() covers all Unicode whitespace
|
|
287
|
+
if normalized_key != "".join(normalized_key.split()):
|
|
288
|
+
_warn_dead_entry(
|
|
289
|
+
f"capitalization_exceptions keys are matched one word "
|
|
290
|
+
f"at a time; multi-word key {k!r} can never match. "
|
|
291
|
+
f"Split it into per-word entries")
|
|
217
292
|
deduped[normalized_key] = v
|
|
218
293
|
return tuple(sorted(deduped.items()))
|
|
219
294
|
|
|
@@ -226,9 +301,12 @@ class Lexicon:
|
|
|
226
301
|
:meth:`empty`, derive variants with :meth:`add` / :meth:`remove` /
|
|
227
302
|
``|`` (union), and pass the result to ``Parser(lexicon=...)``.
|
|
228
303
|
Entries are normalized at construction -- lowercased, edge periods
|
|
229
|
-
stripped -- so matching is case-insensitive.
|
|
230
|
-
|
|
231
|
-
|
|
304
|
+
stripped -- so matching is case-insensitive. Vocabulary entries are
|
|
305
|
+
single words -- a multi-word entry warns at construction and can
|
|
306
|
+
never match (``given_name_titles``, matched as a space-joined run,
|
|
307
|
+
is the one exception). Field docs below show examples, not full
|
|
308
|
+
contents; inspect any field's shipped vocabulary directly, e.g.
|
|
309
|
+
``Lexicon.default().conjunctions``."""
|
|
232
310
|
|
|
233
311
|
#: Pre-nominal titles ("dr", "sir", "capt", ...). Full default
|
|
234
312
|
#: list: :data:`~nameparser.config.titles.TITLES`.
|
|
@@ -256,7 +334,8 @@ class Lexicon:
|
|
|
256
334
|
particles: frozenset[str] = frozenset()
|
|
257
335
|
#: Subset of particles that can also BE a given name: a leading
|
|
258
336
|
#: one reads as given and records a particle-or-given ambiguity
|
|
259
|
-
#: ("Van Johnson"). No constant of its own
|
|
337
|
+
#: ("Van Johnson", but also "Van Buren"). No constant of its own
|
|
338
|
+
#: -- the default derives
|
|
260
339
|
#: as particles minus
|
|
261
340
|
#: :data:`~nameparser.config.prefixes.NON_FIRST_NAME_PREFIXES`
|
|
262
341
|
#: (which marks the opposite, never-given subset).
|
|
@@ -274,6 +353,31 @@ class Lexicon:
|
|
|
274
353
|
#: field ("née", "geb.", "roz.", ...). Full default list:
|
|
275
354
|
#: :data:`~nameparser.config.maiden_markers.MAIDEN_MARKERS`.
|
|
276
355
|
maiden_markers: frozenset[str] = frozenset()
|
|
356
|
+
#: Family names for the unspaced-name segmentation stage (#271),
|
|
357
|
+
#: matched longest-first against the start of the FIRST token
|
|
358
|
+
#: written wholly in a script :attr:`Policy.segment_scripts
|
|
359
|
+
#: <nameparser.Policy.segment_scripts>` activates. The default
|
|
360
|
+
#: carries the Korean census list
|
|
361
|
+
#: (:data:`~nameparser.config.surnames.KOREAN_SURNAMES`); Chinese
|
|
362
|
+
#: surnames ship in locales.ZH because Han segmentation is opt-in.
|
|
363
|
+
surnames: frozenset[str] = frozenset()
|
|
364
|
+
#: Honorifics that may be peeled off the END of a name token
|
|
365
|
+
#: (#308), matched longest-first: 田中さん splits into 田中 and さん
|
|
366
|
+
#: before the tokens are classified. Every entry must also be a
|
|
367
|
+
#: :attr:`suffix_words` entry -- the peeled tail is claimed by
|
|
368
|
+
#: suffix classification like any other post-nominal. Deliberately
|
|
369
|
+
#: NOT gated on :attr:`Policy.segment_scripts
|
|
370
|
+
#: <nameparser.Policy.segment_scripts>` (unlike :attr:`surnames`
|
|
371
|
+
#: above): 田中さん peels under the default policy, where HAN is in
|
|
372
|
+
#: no activation set, because a tail entry carries its own license
|
|
373
|
+
#: to fire. Entries are matched against the RAW token text, and
|
|
374
|
+
#: only within a name containing a non-ASCII character, so an ASCII
|
|
375
|
+
#: or mixed-case entry is at best conditionally active -- a ``"Jr"``
|
|
376
|
+
#: entry is stored ``"jr"`` and matches only lowercase text. The
|
|
377
|
+
#: field is effectively CJK-scoped in 2.1, which is what the shipped
|
|
378
|
+
#: vocabulary is. Full default list:
|
|
379
|
+
#: :data:`~nameparser.config.suffixes.GLUED_HONORIFICS`.
|
|
380
|
+
honorific_tails: frozenset[str] = frozenset()
|
|
277
381
|
#: Lowercase word -> exact-cased replacement used by capitalized()
|
|
278
382
|
#: ("phd" -> "Ph.D."). Pair-valued: change it with
|
|
279
383
|
#: dataclasses.replace(), not add()/remove(); read it as a mapping
|
|
@@ -479,7 +583,17 @@ class Lexicon:
|
|
|
479
583
|
f"{', '.join(_VOCAB_FIELDS)}"
|
|
480
584
|
)
|
|
481
585
|
current: frozenset[str] = getattr(self, name)
|
|
482
|
-
|
|
586
|
+
# warn=False: this pass only computes the new set membership,
|
|
587
|
+
# never stores it directly. add() still warns exactly once,
|
|
588
|
+
# from the replaced instance's own __post_init__ -> _normset;
|
|
589
|
+
# remove() never reaches __post_init__ with the dead entry
|
|
590
|
+
# (it is subtracted out here), so it stays silent -- correct,
|
|
591
|
+
# since a removal stores nothing a warning could be about.
|
|
592
|
+
# That silence covers the entry BEING REMOVED only: a
|
|
593
|
+
# different multi-word entry still stored re-warns from the
|
|
594
|
+
# derived instance's __post_init__, since the warning is
|
|
595
|
+
# per-construction by design.
|
|
596
|
+
normalized = _normset(words, name, warn=False)
|
|
483
597
|
updates[name] = (current | normalized if op == "add"
|
|
484
598
|
else current - normalized)
|
|
485
599
|
# mypy's dataclasses.replace() typing checks a **dict's single
|
|
@@ -506,8 +620,10 @@ def _default_lexicon() -> Lexicon:
|
|
|
506
620
|
from nameparser.config.maiden_markers import MAIDEN_MARKERS
|
|
507
621
|
from nameparser.config.prefixes import NON_FIRST_NAME_PREFIXES, PREFIXES
|
|
508
622
|
from nameparser.config.suffixes import (
|
|
509
|
-
SUFFIX_ACRONYMS, SUFFIX_ACRONYMS_AMBIGUOUS,
|
|
623
|
+
GLUED_HONORIFICS, SUFFIX_ACRONYMS, SUFFIX_ACRONYMS_AMBIGUOUS,
|
|
624
|
+
SUFFIX_NOT_ACRONYMS,
|
|
510
625
|
)
|
|
626
|
+
from nameparser.config.surnames import KOREAN_SURNAMES
|
|
511
627
|
from nameparser.config.titles import FIRST_NAME_TITLES, TITLES
|
|
512
628
|
|
|
513
629
|
# v1 data modules export plain `set[str]`; wrap each at this call site
|
|
@@ -527,6 +643,10 @@ def _default_lexicon() -> Lexicon:
|
|
|
527
643
|
conjunctions=frozenset(CONJUNCTIONS),
|
|
528
644
|
bound_given_names=frozenset(BOUND_FIRST_NAMES),
|
|
529
645
|
maiden_markers=frozenset(MAIDEN_MARKERS),
|
|
646
|
+
# surnames.py is born frozen (#293) -- no call-site wrap needed,
|
|
647
|
+
# unlike the v1 modules above (their wraps drop when #293 lands)
|
|
648
|
+
surnames=KOREAN_SURNAMES,
|
|
649
|
+
honorific_tails=frozenset(GLUED_HONORIFICS),
|
|
530
650
|
# pass canonical pair-tuples so this strictly-typed call site never
|
|
531
651
|
# feeds a Mapping to the tuple-annotated field; __post_init__
|
|
532
652
|
# still tolerates a Mapping at runtime for interactive use
|