nameparser 2.0.0rc1__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. {nameparser-2.0.0rc1/nameparser.egg-info → nameparser-2.1.0}/PKG-INFO +48 -8
  2. {nameparser-2.0.0rc1 → nameparser-2.1.0}/README.rst +45 -7
  3. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/__init__.py +8 -2
  4. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_config_shim.py +74 -14
  5. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_facade.py +24 -1
  6. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_lexicon.py +134 -14
  7. nameparser-2.1.0/nameparser/_parser.py +306 -0
  8. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/__init__.py +4 -3
  9. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_assemble.py +25 -7
  10. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_assign.py +84 -6
  11. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_classify.py +3 -1
  12. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_extract.py +9 -1
  13. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_group.py +72 -5
  14. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_post_rules.py +2 -1
  15. nameparser-2.1.0/nameparser/_pipeline/_script_segment.py +767 -0
  16. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_segment.py +22 -51
  17. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_pipeline/_state.py +24 -10
  18. nameparser-2.1.0/nameparser/_pipeline/_tokenize.py +208 -0
  19. nameparser-2.1.0/nameparser/_pipeline/_vocab.py +356 -0
  20. nameparser-2.1.0/nameparser/_policy.py +916 -0
  21. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_render.py +12 -0
  22. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_types.py +199 -40
  23. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_version.py +2 -2
  24. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/conjunctions.py +5 -1
  25. nameparser-2.1.0/nameparser/config/maiden_markers.py +70 -0
  26. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/suffixes.py +118 -7
  27. nameparser-2.1.0/nameparser/config/surnames.py +49 -0
  28. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/titles.py +3 -1
  29. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/locales/__init__.py +22 -0
  30. nameparser-2.1.0/nameparser/locales/ja.py +228 -0
  31. nameparser-2.1.0/nameparser/locales/zh.py +113 -0
  32. {nameparser-2.0.0rc1 → nameparser-2.1.0/nameparser.egg-info}/PKG-INFO +48 -8
  33. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser.egg-info/SOURCES.txt +7 -0
  34. nameparser-2.1.0/nameparser.egg-info/requires.txt +3 -0
  35. {nameparser-2.0.0rc1 → nameparser-2.1.0}/pyproject.toml +19 -0
  36. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_nicknames.py +39 -0
  37. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_titles.py +13 -0
  38. nameparser-2.1.0/tests/v2/cases.py +1807 -0
  39. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/conftest.py +10 -15
  40. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_assemble.py +7 -4
  41. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_assign.py +68 -1
  42. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_classify.py +21 -1
  43. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_group.py +92 -1
  44. nameparser-2.1.0/tests/v2/pipeline/test_script_segment.py +815 -0
  45. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_segment.py +14 -0
  46. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_state.py +15 -2
  47. nameparser-2.1.0/tests/v2/pipeline/test_tokenize.py +228 -0
  48. nameparser-2.1.0/tests/v2/pipeline/test_vocab.py +373 -0
  49. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_benchmark.py +94 -12
  50. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_cli.py +13 -0
  51. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_config_shim.py +59 -0
  52. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_contracts.py +28 -8
  53. nameparser-2.1.0/tests/v2/test_differential.py +696 -0
  54. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_facade.py +69 -0
  55. nameparser-2.1.0/tests/v2/test_facade_cases.py +153 -0
  56. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_layering.py +18 -8
  57. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_lexicon.py +155 -2
  58. nameparser-2.1.0/tests/v2/test_locales.py +1202 -0
  59. nameparser-2.1.0/tests/v2/test_parser.py +813 -0
  60. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_policy.py +338 -5
  61. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_properties.py +219 -17
  62. nameparser-2.1.0/tests/v2/test_regex_sync.py +366 -0
  63. nameparser-2.1.0/tests/v2/test_reprs.py +161 -0
  64. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_types.py +144 -2
  65. nameparser-2.0.0rc1/nameparser/_parser.py +0 -130
  66. nameparser-2.0.0rc1/nameparser/_pipeline/_tokenize.py +0 -125
  67. nameparser-2.0.0rc1/nameparser/_pipeline/_vocab.py +0 -126
  68. nameparser-2.0.0rc1/nameparser/_policy.py +0 -456
  69. nameparser-2.0.0rc1/nameparser/config/maiden_markers.py +0 -42
  70. nameparser-2.0.0rc1/tests/v2/cases.py +0 -469
  71. nameparser-2.0.0rc1/tests/v2/pipeline/test_tokenize.py +0 -91
  72. nameparser-2.0.0rc1/tests/v2/pipeline/test_vocab.py +0 -40
  73. nameparser-2.0.0rc1/tests/v2/test_facade_cases.py +0 -75
  74. nameparser-2.0.0rc1/tests/v2/test_locales.py +0 -522
  75. nameparser-2.0.0rc1/tests/v2/test_parser.py +0 -317
  76. nameparser-2.0.0rc1/tests/v2/test_regex_sync.py +0 -151
  77. nameparser-2.0.0rc1/tests/v2/test_reprs.py +0 -68
  78. {nameparser-2.0.0rc1 → nameparser-2.1.0}/AUTHORS +0 -0
  79. {nameparser-2.0.0rc1 → nameparser-2.1.0}/LICENSE +0 -0
  80. {nameparser-2.0.0rc1 → nameparser-2.1.0}/MANIFEST.in +0 -0
  81. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/__main__.py +0 -0
  82. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/_locale.py +0 -0
  83. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/__init__.py +0 -0
  84. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/_invariants.py +0 -0
  85. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/bound_first_names.py +0 -0
  86. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/capitalization.py +0 -0
  87. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/prefixes.py +0 -0
  88. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/config/regexes.py +0 -0
  89. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/locales/ru.py +0 -0
  90. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/locales/tr_az.py +0 -0
  91. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/parser.py +0 -0
  92. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/py.typed +0 -0
  93. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser/util.py +0 -0
  94. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser.egg-info/dependency_links.txt +0 -0
  95. {nameparser-2.0.0rc1 → nameparser-2.1.0}/nameparser.egg-info/top_level.txt +0 -0
  96. {nameparser-2.0.0rc1 → nameparser-2.1.0}/setup.cfg +0 -0
  97. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/__init__.py +0 -0
  98. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/base.py +0 -0
  99. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/conftest.py +0 -0
  100. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_bound_first_names.py +0 -0
  101. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_brute_force.py +0 -0
  102. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_capitalization.py +0 -0
  103. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_comma_variants.py +0 -0
  104. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_conjunctions.py +0 -0
  105. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_constants.py +0 -0
  106. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_east_slavic_patronymic_order.py +0 -0
  107. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_first_name.py +0 -0
  108. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_initials.py +0 -0
  109. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_middle_name_as_last.py +0 -0
  110. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_output_format.py +0 -0
  111. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_prefixes.py +0 -0
  112. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_python_api.py +0 -0
  113. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_suffixes.py +0 -0
  114. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_turkic_patronymic_order.py +0 -0
  115. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/test_variations.py +0 -0
  116. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/__init__.py +0 -0
  117. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/__init__.py +0 -0
  118. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_extract.py +0 -0
  119. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/pipeline/test_post_rules.py +0 -0
  120. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_cases.py +0 -0
  121. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_locale.py +0 -0
  122. {nameparser-2.0.0rc1 → nameparser-2.1.0}/tests/v2/test_render.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: nameparser
3
- Version: 2.0.0rc1
3
+ Version: 2.1.0
4
4
  Summary: A simple Python module for parsing human names into their individual components.
5
5
  Author-email: Derek Gulbranson <derek73@gmail.com>
6
6
  License: LGPL
@@ -22,6 +22,8 @@ Requires-Python: >=3.11
22
22
  Description-Content-Type: text/x-rst
23
23
  License-File: LICENSE
24
24
  License-File: AUTHORS
25
+ Provides-Extra: ja
26
+ Requires-Dist: namedivider-python>=0.4; extra == "ja"
25
27
  Dynamic: license-file
26
28
 
27
29
  Name Parser
@@ -33,6 +35,19 @@ nameparser parses human names into seven fields — title, given, middle,
33
35
  family, suffix, nickname, maiden. Results are immutable, configuration is
34
36
  composable, and locale packs are opt-in.
35
37
 
38
+ 📣 **nameparser 2.0 is out.** Existing ``HumanName`` code keeps working
39
+ through 2.x, and most 1.x code needs no changes. The `migration guide
40
+ <https://nameparser.readthedocs.io/en/latest/migrate.html>`__ has the
41
+ field-by-field map. Please `open an issue
42
+ <https://github.com/derek73/python-nameparser/issues>`__ for anything that
43
+ parses wrong.
44
+
45
+ **2.1 adds East Asian name support.** Chinese, Japanese and Korean names
46
+ written in their own scripts are read family-first, unspaced Korean names
47
+ are split against the census surname list, and CJK honorifics are
48
+ recognized. See `East Asian names
49
+ <https://nameparser.readthedocs.io/en/latest/usage.html#east-asian-names>`__.
50
+
36
51
  Installation
37
52
  ------------
38
53
 
@@ -48,16 +63,41 @@ Quick Start Example
48
63
  .. code-block:: python
49
64
 
50
65
  >>> from nameparser import parse
51
- >>> name = parse("Dr. Juan Q. Xavier de la Vega III")
52
- >>> name.given, name.family
53
- ('Juan', 'de la Vega')
66
+ >>> name = parse("Dr. Juan Q. Xavier de la Vega III (Doc Vega)")
67
+ >>> name
68
+ <ParsedName: [
69
+ title: 'Dr.'
70
+ given: 'Juan'
71
+ middle: 'Q. Xavier'
72
+ family: 'de la Vega'
73
+ suffix: 'III'
74
+ nickname: 'Doc Vega'
75
+ ]>
76
+ >>> name.family_base, name.family_particles
77
+ ('Vega', 'de la')
54
78
  >>> name.render("{family}, {given}")
55
79
  'de la Vega, Juan'
56
80
 
57
- Those seven fields are ``title``, ``given``, ``middle``, ``family``,
58
- ``suffix``, ``nickname``, and ``maiden`` — plus aggregate views like
59
- ``given_names``, ``surnames``, ``family_base``, and ``family_particles``
60
- for combining or splitting them further.
81
+ >>> parse("김민준").family # Korean: unspaced, split on the census list
82
+ '김'
83
+ >>> parse("高橋 みなみ").family # Japanese: kanji with kana, family first
84
+ '高橋'
85
+ >>> parse("김민준씨").suffix # an honorific written against the name
86
+ '씨'
87
+ >>> parse("г-н Иван Петров").title # Cyrillic title
88
+ 'г-н'
89
+ >>> parse("محمد بن سلمان").family # Arabic: بن chains onto the family name
90
+ 'بن سلمان'
91
+
92
+ >>> from nameparser import locales, parser_for
93
+ >>> chinese = parser_for(locales.ZH) # Han text does not say which language
94
+ >>> chinese.parse("毛泽东").family # so splitting it is opt-in
95
+ '毛'
96
+ >>> russian = parser_for(locales.RU)
97
+ >>> russian.parse("Сидоров Иван Петрович").family
98
+ 'Сидоров'
99
+ >>> locales.available()
100
+ ('ja', 'ru', 'tr_az', 'zh')
61
101
 
62
102
  Learn more
63
103
  ----------
@@ -7,6 +7,19 @@ nameparser parses human names into seven fields — title, given, middle,
7
7
  family, suffix, nickname, maiden. Results are immutable, configuration is
8
8
  composable, and locale packs are opt-in.
9
9
 
10
+ 📣 **nameparser 2.0 is out.** Existing ``HumanName`` code keeps working
11
+ through 2.x, and most 1.x code needs no changes. The `migration guide
12
+ <https://nameparser.readthedocs.io/en/latest/migrate.html>`__ has the
13
+ field-by-field map. Please `open an issue
14
+ <https://github.com/derek73/python-nameparser/issues>`__ for anything that
15
+ parses wrong.
16
+
17
+ **2.1 adds East Asian name support.** Chinese, Japanese and Korean names
18
+ written in their own scripts are read family-first, unspaced Korean names
19
+ are split against the census surname list, and CJK honorifics are
20
+ recognized. See `East Asian names
21
+ <https://nameparser.readthedocs.io/en/latest/usage.html#east-asian-names>`__.
22
+
10
23
  Installation
11
24
  ------------
12
25
 
@@ -22,16 +35,41 @@ Quick Start Example
22
35
  .. code-block:: python
23
36
 
24
37
  >>> from nameparser import parse
25
- >>> name = parse("Dr. Juan Q. Xavier de la Vega III")
26
- >>> name.given, name.family
27
- ('Juan', 'de la Vega')
38
+ >>> name = parse("Dr. Juan Q. Xavier de la Vega III (Doc Vega)")
39
+ >>> name
40
+ <ParsedName: [
41
+ title: 'Dr.'
42
+ given: 'Juan'
43
+ middle: 'Q. Xavier'
44
+ family: 'de la Vega'
45
+ suffix: 'III'
46
+ nickname: 'Doc Vega'
47
+ ]>
48
+ >>> name.family_base, name.family_particles
49
+ ('Vega', 'de la')
28
50
  >>> name.render("{family}, {given}")
29
51
  'de la Vega, Juan'
30
52
 
31
- Those seven fields are ``title``, ``given``, ``middle``, ``family``,
32
- ``suffix``, ``nickname``, and ``maiden`` — plus aggregate views like
33
- ``given_names``, ``surnames``, ``family_base``, and ``family_particles``
34
- for combining or splitting them further.
53
+ >>> parse("김민준").family # Korean: unspaced, split on the census list
54
+ '김'
55
+ >>> parse("高橋 みなみ").family # Japanese: kanji with kana, family first
56
+ '高橋'
57
+ >>> parse("김민준씨").suffix # an honorific written against the name
58
+ '씨'
59
+ >>> parse("г-н Иван Петров").title # Cyrillic title
60
+ 'г-н'
61
+ >>> parse("محمد بن سلمان").family # Arabic: بن chains onto the family name
62
+ 'بن سلمان'
63
+
64
+ >>> from nameparser import locales, parser_for
65
+ >>> chinese = parser_for(locales.ZH) # Han text does not say which language
66
+ >>> chinese.parse("毛泽东").family # so splitting it is opt-in
67
+ '毛'
68
+ >>> russian = parser_for(locales.RU)
69
+ >>> russian.parse("Сидоров Иван Петрович").family
70
+ 'Сидоров'
71
+ >>> locales.available()
72
+ ('ja', 'ru', 'tr_az', 'zh')
35
73
 
36
74
  Learn more
37
75
  ----------
@@ -11,6 +11,7 @@ from nameparser._locale import Locale
11
11
  from nameparser._parser import Parser, parse, parser_for
12
12
  from nameparser._policy import (
13
13
  DEFAULT_NICKNAME_DELIMITERS,
14
+ DEFAULT_SCRIPT_ORDERS,
14
15
  FAMILY_FIRST,
15
16
  FAMILY_FIRST_GIVEN_LAST,
16
17
  GIVEN_FIRST,
@@ -18,12 +19,16 @@ from nameparser._policy import (
18
19
  PatronymicRule,
19
20
  Policy,
20
21
  PolicyPatch,
22
+ Script,
21
23
  )
22
24
  from nameparser._types import (
25
+ STABLE_TAGS,
23
26
  Ambiguity,
24
27
  AmbiguityKind,
25
28
  ParsedName,
26
29
  Role,
30
+ Segmentation,
31
+ Segmenter,
27
32
  Span,
28
33
  Token,
29
34
  )
@@ -33,8 +38,9 @@ __all__ = [
33
38
  "HumanName",
34
39
  # v2 core
35
40
  "Span", "Role", "Token", "Ambiguity", "AmbiguityKind", "ParsedName",
36
- "Lexicon", "Policy", "PolicyPatch", "PatronymicRule", "UNSET",
41
+ "STABLE_TAGS", "Segmentation", "Segmenter",
42
+ "Lexicon", "Policy", "PolicyPatch", "PatronymicRule", "Script", "UNSET",
37
43
  "GIVEN_FIRST", "FAMILY_FIRST", "FAMILY_FIRST_GIVEN_LAST",
38
- "DEFAULT_NICKNAME_DELIMITERS", "Locale",
44
+ "DEFAULT_NICKNAME_DELIMITERS", "DEFAULT_SCRIPT_ORDERS", "Locale",
39
45
  "Parser", "parse", "parser_for",
40
46
  ]
@@ -33,6 +33,28 @@ from nameparser._policy import PatronymicRule, Policy
33
33
  from nameparser.util import lc
34
34
 
35
35
 
36
+ #: The eight multi-word entries the pre-2.0 DEFAULT vocabulary shipped.
37
+ #: Provably inert in every release (they can never match; see the
38
+ #: 2.0.0 release log), so dropping them from a restored legacy pickle
39
+ #: changes no parse -- and keeps the multi-word warning from firing
40
+ #: eight times, with wrong advice, at library-internal lines, on the
41
+ #: first parse after a supported 1.3/1.4 pickle upgrade.
42
+ #:
43
+ #: Gated on ALL EIGHT being present across the two fields -- the
44
+ #: signature of a pre-2.0 blob, which froze the complete shipped set.
45
+ #: A round-trip of 2.0-era state that carries FEWER than all eight
46
+ #: keeps them (a user who deliberately re-added one or two does not
47
+ #: match the signature, and .copy() never subtracts either). The trade:
48
+ #: a 2.0 user who re-adds ALL eight exact strings is indistinguishable
49
+ #: from a legacy blob and loses them on the next unpickle.
50
+ _LEGACY_DEAD_ENTRIES = {
51
+ "titles": frozenset({"chargé d'affaires"}),
52
+ "suffix_acronyms": frozenset({
53
+ "leed ap", "nicet i", "nicet ii", "nicet iii", "nicet iv",
54
+ "psm i", "psm ii"}),
55
+ }
56
+
57
+
36
58
  def _reject_bare_str_or_bytes(value: object, expected: str) -> None:
37
59
  # A bare string is an iterable of its characters, so e.g. SetManager('dr')
38
60
  # would silently shred it into {'d', 'r'} instead of raising -- shared by
@@ -964,10 +986,23 @@ class Constants:
964
986
  removal path.
965
987
  """
966
988
  from nameparser.config.maiden_markers import MAIDEN_MARKERS
989
+ from nameparser.config.suffixes import GLUED_HONORIFICS
990
+ from nameparser.config.surnames import KOREAN_SURNAMES
967
991
  acronyms = frozenset(self.suffix_acronyms)
968
992
  particles = frozenset(self.prefixes)
969
993
  bound = frozenset(self.bound_first_names)
970
994
  ambiguous_acronyms = frozenset(self.suffix_acronyms_ambiguous) & acronyms
995
+ # Drop any ambiguous acronym from the word set rather than the
996
+ # other way round. Lexicon forbids the overlap because the word
997
+ # branch bypasses the period gate, and adding an ambiguous
998
+ # acronym to suffix_not_acronyms is INERT in v1 anyway:
999
+ # is_suffix already accepts it via the acronym branch, and
1000
+ # reserve_last keeps it as the surname. So ignoring the
1001
+ # addition reproduces v1 ("Jack Ma" keeps last='Ma'), where
1002
+ # dropping it from the AMBIGUOUS set instead ungated the word
1003
+ # and lost the family name -- a silent misparse worse than the
1004
+ # raise it avoided.
1005
+ suffix_words = frozenset(self.suffix_not_acronyms) - ambiguous_acronyms
971
1006
  # keep in sync with _lexicon._default_lexicon() (pinned by
972
1007
  # tests/v2/test_config_shim.py::test_snapshot_field_translation)
973
1008
  lexicon = Lexicon(
@@ -994,18 +1029,7 @@ class Constants:
994
1029
  if e == " ".join(e.split())
995
1030
  ) if t),
996
1031
  suffix_acronyms=acronyms,
997
- # Drop any ambiguous acronym from the word set rather than
998
- # the other way round. Lexicon forbids the overlap because
999
- # the word branch bypasses the period gate, and adding an
1000
- # ambiguous acronym to suffix_not_acronyms is INERT in v1
1001
- # anyway: is_suffix already accepts it via the acronym
1002
- # branch, and reserve_last keeps it as the surname. So
1003
- # ignoring the addition reproduces v1 ("Jack Ma" keeps
1004
- # last='Ma'), where dropping it from the AMBIGUOUS set
1005
- # instead ungated the word and lost the family name --
1006
- # a silent misparse worse than the raise it avoided.
1007
- suffix_words=frozenset(
1008
- self.suffix_not_acronyms) - ambiguous_acronyms,
1032
+ suffix_words=suffix_words,
1009
1033
  # Intersect with acronyms: Lexicon enforces ambiguous <=
1010
1034
  # acronyms; v1 behaves the same when an acronym is deleted
1011
1035
  # but its ambiguous entry lingers (the entry stops
@@ -1039,6 +1063,22 @@ class Constants:
1039
1063
  # v1 Constants has no manager for these (#274 is 2.0
1040
1064
  # behavior); the data module is the only source
1041
1065
  maiden_markers=frozenset(MAIDEN_MARKERS),
1066
+ # likewise no v1 manager: the unspaced-name segmentation
1067
+ # vocabulary is 2.0 behavior (#271), so it rides in the
1068
+ # snapshot only -- v1's Constants surface stays frozen.
1069
+ # Unwrapped where maiden_markers above is wrapped: this
1070
+ # module is born frozen (#293), so no wrap
1071
+ surnames=KOREAN_SURNAMES,
1072
+ # likewise no v1 manager: the glued-honorific tail set is
1073
+ # 2.1 behavior (#308), so it rides in the snapshot only.
1074
+ # Wrapped, unlike surnames above: suffixes.py is still a
1075
+ # mutable v1 module, not born-frozen like surnames.py
1076
+ # (#293). Intersect with the word set: Lexicon enforces
1077
+ # tails <= suffix_words, and v1 semantics are that deleting
1078
+ # a suffix word turns the behavior off -- a lingering tail
1079
+ # simply stops mattering, the same rule ambiguous_acronyms
1080
+ # gets against suffix_acronyms above.
1081
+ honorific_tails=frozenset(GLUED_HONORIFICS) & suffix_words,
1042
1082
  # TupleManager is dict[str, object] (v1 parity: values were
1043
1083
  # never statically str-typed); every real entry is a str,
1044
1084
  # same assumption _DelimiterManager's sentinel lookup makes
@@ -1103,10 +1143,30 @@ class Constants:
1103
1143
  self.__init__() # type: ignore[misc] # defaults, then overlay
1104
1144
  # (managers re-wrapped below so _on_change points at THIS
1105
1145
  # instance, not whatever produced the incoming state)
1146
+ managers: dict[str, SetManager] = {}
1106
1147
  for name in _SET_FIELDS:
1107
1148
  if name in state:
1108
- object.__setattr__(self, name, SetManager(
1109
- state[name], _on_change=self._bump)) # type: ignore[arg-type]
1149
+ managers[name] = SetManager(
1150
+ state[name], _on_change=self._bump) # type: ignore[arg-type]
1151
+ # SetManager normalized on construction, so the frozen 1.3/1.4
1152
+ # vocabulary's dead entries are matchable in their normalized
1153
+ # spelling here. Subtract only when ALL EIGHT are present --
1154
+ # the pre-2.0 signature; a 2.0 user who re-added one or two
1155
+ # keeps them through a round-trip (see _LEGACY_DEAD_ENTRIES).
1156
+ legacy = all(
1157
+ name in managers and entry in managers[name]
1158
+ for name, entries in _LEGACY_DEAD_ENTRIES.items()
1159
+ for entry in entries
1160
+ )
1161
+ for name, manager in managers.items():
1162
+ if legacy:
1163
+ # Reach past the public discard() deliberately: this is
1164
+ # part of restoring the state, not a mutation of it, and
1165
+ # must not bump the generation of an instance that is
1166
+ # still being built.
1167
+ manager._elements -= _LEGACY_DEAD_ENTRIES.get(
1168
+ name, frozenset())
1169
+ object.__setattr__(self, name, manager)
1110
1170
  if "capitalization_exceptions" in state:
1111
1171
  object.__setattr__(
1112
1172
  self, "capitalization_exceptions", TupleManager(
@@ -335,6 +335,8 @@ class HumanName:
335
335
  else:
336
336
  raise TypeError(
337
337
  f"{member} must be a str, list, or None, got {value!r}")
338
+ # v1 setters stay on replace(): revise()'s vocabulary tags would
339
+ # change v1 parity
338
340
  self._parsed = self._parsed.replace(
339
341
  **{_V2_FIELD.get(member, member): joined})
340
342
 
@@ -616,7 +618,28 @@ class HumanName:
616
618
  "slicing a HumanName was removed in 2.0 (#258); access "
617
619
  "the named attributes instead"
618
620
  )
619
- return getattr(self, key)
621
+ # Role is a StrEnum, so Role members (and the plain 'given'/
622
+ # 'family' strings) reach here too -- translate to the v1
623
+ # spelling the facade actually exposes as attributes.
624
+ return getattr(self, _V1_SPELLING.get(key, key))
625
+
626
+ def __setattr__(self, name: str, value: object) -> None:
627
+ # "given"/"family" are the 2.0 spellings of first/last; the
628
+ # facade has no such attributes, so plain assignment creates a
629
+ # stray instance attribute while the parse (and .first/.last)
630
+ # keeps the old value -- a silently forked name. Warn but
631
+ # still set: ad-hoc attribute stashing is a legal v1 pattern,
632
+ # so any code that worked keeps working. Only these two names
633
+ # warn -- the other five 2.0 field names are real properties
634
+ # whose setters work, and Role members reach here as their
635
+ # string values (StrEnum).
636
+ if name in _V1_SPELLING:
637
+ warnings.warn(
638
+ f"assigning HumanName.{name} creates an inert attribute; "
639
+ f"the parse is unchanged -- use .{_V1_SPELLING[name]} "
640
+ f"(the v1 spelling) to update the name",
641
+ UserWarning, stacklevel=2)
642
+ super().__setattr__(name, value)
620
643
 
621
644
  def as_dict(self, include_empty: bool = True) -> dict[str, str]:
622
645
  """The seven v1-named components as a dict; include_empty=False
@@ -9,9 +9,11 @@ from __future__ import annotations
9
9
 
10
10
  import dataclasses
11
11
  import functools
12
+ import sys
13
+ import warnings
12
14
  from collections.abc import Iterable, Mapping
13
15
  from dataclasses import dataclass, field
14
- from types import MappingProxyType
16
+ from types import FrameType, MappingProxyType
15
17
  from typing import cast
16
18
 
17
19
  #: Vocabulary set fields, in declaration order. add()/remove() operate
@@ -21,20 +23,44 @@ from typing import cast
21
23
  _VOCAB_FIELDS = (
22
24
  "titles", "given_name_titles", "suffix_acronyms", "suffix_words",
23
25
  "suffix_acronyms_ambiguous", "particles", "particles_ambiguous",
24
- "conjunctions", "bound_given_names", "maiden_markers",
26
+ "conjunctions", "bound_given_names", "maiden_markers", "surnames",
27
+ "honorific_tails",
25
28
  )
26
29
 
27
- #: (marker, base, why) triples. Each marker narrows how entries of its
28
- #: base vocabulary are read and carries no vocabulary of its own, so an
29
- #: entry outside the base is a configuration mistake -- but the mistake
30
- #: differs per pair, and the reason is recorded here rather than
31
- #: generalized, because an orphan is NOT simply inert:
30
+ #: (marker, base, why) triples. Each marker QUALIFIES how entries of
31
+ #: its base vocabulary are read and carries no vocabulary of its own,
32
+ #: so an entry outside the base is a configuration mistake -- but the
33
+ #: mistake differs per pair, and the reason is recorded here rather
34
+ #: than generalized, because an orphan is NOT simply inert. Nor is the
35
+ #: qualification one-directional: the first two NARROW their base (an
36
+ #: entry is read as vocabulary in fewer places), while honorific_tails
37
+ #: WIDENS it, granting a suffix word the glued position on top of the
38
+ #: whole-token match every suffix word already gets.
32
39
  #:
33
40
  #: * particles_ambiguous: _assign keys on the tag alone, so an orphan
34
41
  #: makes the parse emit a spurious particle-or-given ambiguity.
35
42
  #: * suffix_acronyms_ambiguous: _vocab returns True on the ambiguous
36
43
  #: set before testing suffix_acronyms, so an orphan silently turns a
37
44
  #: word into a period-gated suffix.
45
+ #: * honorific_tails: script_segment peels the tail into its own token
46
+ #: before classify ever runs, so an orphan splits the name and leaves
47
+ #: the fragment stranded inside it -- worse than not peeling at all.
48
+ #: Its base is deliberately NARROWER than what actually claims the
49
+ #: peeled piece: suffix_as_written ORs suffix_words with the
50
+ #: non-ambiguous suffix_acronyms, so a tail listed only as an acronym
51
+ #: would classify fine yet is rejected here. Accepted, and a decision
52
+ #: rather than an oversight -- the three-term predicate is easy to
53
+ #: get wrong in the dangerous direction (an ambiguous acronym admitted
54
+ #: as a tail would peel a period-gated word off a real name), and
55
+ #: nothing needs the acronym half: the shipped tails are CJK
56
+ #: honorifics, which are words.
57
+ #: The same relation is asserted a second time in config/suffixes.py,
58
+ #: over the raw GLUED_HONORIFICS/SUFFIX_NOT_ACRONYMS constants at
59
+ #: import. The two are not redundant in the way they look: that one
60
+ #: is an `assert`, stripped under `python -O`, while the check here
61
+ #: raises unconditionally -- so under -O this is what still holds the
62
+ #: SHIPPED vocabulary to the invariant, as it is the only thing that
63
+ #: ever held a caller's own.
38
64
  #:
39
65
  #: given_name_titles is deliberately NOT here and has no check of its
40
66
  #: own -- see the note in __post_init__ for why every attempt at one
@@ -44,6 +70,8 @@ _SUBSET_FIELDS = (
44
70
  "an orphan emits a spurious particle-or-given ambiguity"),
45
71
  ("suffix_acronyms_ambiguous", "suffix_acronyms",
46
72
  "an orphan silently becomes a period-gated suffix"),
73
+ ("honorific_tails", "suffix_words",
74
+ "an orphan splits the name and leaves the tail inside it"),
47
75
  )
48
76
 
49
77
 
@@ -112,7 +140,27 @@ def _reject_buffer(value: object, label: str, plural: str) -> None:
112
140
  )
113
141
 
114
142
 
115
- def _normset(entries: Iterable[str], field_name: str) -> frozenset[str]:
143
+ def _warn_dead_entry(message: str) -> None:
144
+ # A fixed stacklevel always lands on library internals: the call
145
+ # depth differs per entry point (Lexicon(), add(), unpickle,
146
+ # dataclasses.replace, and the v1 shim's lazy snapshot -- built on
147
+ # the first parse after a Constants mutation, several facade frames
148
+ # below the user's own add()). Walk out of this module (and
149
+ # dataclasses' replace frames, and the facade layer that builds
150
+ # lexicons on the caller's behalf) so the warning points at the
151
+ # caller's own line.
152
+ level = 2
153
+ frame: FrameType | None = sys._getframe(1)
154
+ while frame is not None and frame.f_globals.get("__name__") in (
155
+ __name__, "dataclasses",
156
+ "nameparser._config_shim", "nameparser._facade"):
157
+ frame, level = frame.f_back, level + 1
158
+ warnings.warn(message, UserWarning, stacklevel=level)
159
+
160
+
161
+ def _normset(
162
+ entries: Iterable[str], field_name: str, warn: bool = True,
163
+ ) -> frozenset[str]:
116
164
  # Reject a bare str before iterating: iterating "dr" would silently
117
165
  # yield the single characters {'d', 'r'} -- the set(str) footgun on
118
166
  # the primary customization surface.
@@ -159,6 +207,25 @@ def _normset(entries: Iterable[str], field_name: str) -> frozenset[str]:
159
207
  f"Lexicon.{field_name} entry {w!r} normalizes to empty "
160
208
  f"(lowercase + strip periods/whitespace leaves nothing)"
161
209
  )
210
+ # Every field but given_name_titles is matched one word at a
211
+ # time, so a multi-word entry can never match -- the library
212
+ # itself shipped eight such dead entries for years (repaired
213
+ # 2026-07-26). Warn, never raise: an inert entry produces
214
+ # nothing, and the given_name_titles precedent says a raise
215
+ # here costs working configurations (see __post_init__).
216
+ # warn=False is _edit()'s pass (both ops): add() warns via the
217
+ # new instance's __post_init__; remove() stores nothing, so
218
+ # warning there would name entries the caller is trying to get
219
+ # RID of, with "split it" advice that makes no sense for a
220
+ # no-op.
221
+ if (warn and field_name != "given_name_titles"
222
+ # interior whitespace test; split() covers all Unicode
223
+ # whitespace
224
+ and n != "".join(n.split())):
225
+ _warn_dead_entry(
226
+ f"Lexicon.{field_name} entries are matched one word at "
227
+ f"a time; multi-word entry {w!r} can never match. "
228
+ f"Split it into separate entries")
162
229
  normalized.add(n)
163
230
  return frozenset(normalized)
164
231
 
@@ -214,6 +281,14 @@ def _normpairs(
214
281
  f"empty (lowercase + strip periods/whitespace leaves "
215
282
  f"nothing)"
216
283
  )
284
+ # capitalized() looks words up one at a time (the _WORD regex
285
+ # never yields spaces), so a multi-word key is unreachable.
286
+ # interior whitespace test; split() covers all Unicode whitespace
287
+ if normalized_key != "".join(normalized_key.split()):
288
+ _warn_dead_entry(
289
+ f"capitalization_exceptions keys are matched one word "
290
+ f"at a time; multi-word key {k!r} can never match. "
291
+ f"Split it into per-word entries")
217
292
  deduped[normalized_key] = v
218
293
  return tuple(sorted(deduped.items()))
219
294
 
@@ -226,9 +301,12 @@ class Lexicon:
226
301
  :meth:`empty`, derive variants with :meth:`add` / :meth:`remove` /
227
302
  ``|`` (union), and pass the result to ``Parser(lexicon=...)``.
228
303
  Entries are normalized at construction -- lowercased, edge periods
229
- stripped -- so matching is case-insensitive. Field docs below show
230
- examples, not full contents; inspect any field's shipped vocabulary
231
- directly, e.g. ``Lexicon.default().conjunctions``."""
304
+ stripped -- so matching is case-insensitive. Vocabulary entries are
305
+ single words -- a multi-word entry warns at construction and can
306
+ never match (``given_name_titles``, matched as a space-joined run,
307
+ is the one exception). Field docs below show examples, not full
308
+ contents; inspect any field's shipped vocabulary directly, e.g.
309
+ ``Lexicon.default().conjunctions``."""
232
310
 
233
311
  #: Pre-nominal titles ("dr", "sir", "capt", ...). Full default
234
312
  #: list: :data:`~nameparser.config.titles.TITLES`.
@@ -256,7 +334,8 @@ class Lexicon:
256
334
  particles: frozenset[str] = frozenset()
257
335
  #: Subset of particles that can also BE a given name: a leading
258
336
  #: one reads as given and records a particle-or-given ambiguity
259
- #: ("Van Johnson"). No constant of its own -- the default derives
337
+ #: ("Van Johnson", but also "Van Buren"). No constant of its own
338
+ #: -- the default derives
260
339
  #: as particles minus
261
340
  #: :data:`~nameparser.config.prefixes.NON_FIRST_NAME_PREFIXES`
262
341
  #: (which marks the opposite, never-given subset).
@@ -274,6 +353,31 @@ class Lexicon:
274
353
  #: field ("née", "geb.", "roz.", ...). Full default list:
275
354
  #: :data:`~nameparser.config.maiden_markers.MAIDEN_MARKERS`.
276
355
  maiden_markers: frozenset[str] = frozenset()
356
+ #: Family names for the unspaced-name segmentation stage (#271),
357
+ #: matched longest-first against the start of the FIRST token
358
+ #: written wholly in a script :attr:`Policy.segment_scripts
359
+ #: <nameparser.Policy.segment_scripts>` activates. The default
360
+ #: carries the Korean census list
361
+ #: (:data:`~nameparser.config.surnames.KOREAN_SURNAMES`); Chinese
362
+ #: surnames ship in locales.ZH because Han segmentation is opt-in.
363
+ surnames: frozenset[str] = frozenset()
364
+ #: Honorifics that may be peeled off the END of a name token
365
+ #: (#308), matched longest-first: 田中さん splits into 田中 and さん
366
+ #: before the tokens are classified. Every entry must also be a
367
+ #: :attr:`suffix_words` entry -- the peeled tail is claimed by
368
+ #: suffix classification like any other post-nominal. Deliberately
369
+ #: NOT gated on :attr:`Policy.segment_scripts
370
+ #: <nameparser.Policy.segment_scripts>` (unlike :attr:`surnames`
371
+ #: above): 田中さん peels under the default policy, where HAN is in
372
+ #: no activation set, because a tail entry carries its own license
373
+ #: to fire. Entries are matched against the RAW token text, and
374
+ #: only within a name containing a non-ASCII character, so an ASCII
375
+ #: or mixed-case entry is at best conditionally active -- a ``"Jr"``
376
+ #: entry is stored ``"jr"`` and matches only lowercase text. The
377
+ #: field is effectively CJK-scoped in 2.1, which is what the shipped
378
+ #: vocabulary is. Full default list:
379
+ #: :data:`~nameparser.config.suffixes.GLUED_HONORIFICS`.
380
+ honorific_tails: frozenset[str] = frozenset()
277
381
  #: Lowercase word -> exact-cased replacement used by capitalized()
278
382
  #: ("phd" -> "Ph.D."). Pair-valued: change it with
279
383
  #: dataclasses.replace(), not add()/remove(); read it as a mapping
@@ -479,7 +583,17 @@ class Lexicon:
479
583
  f"{', '.join(_VOCAB_FIELDS)}"
480
584
  )
481
585
  current: frozenset[str] = getattr(self, name)
482
- normalized = _normset(words, name)
586
+ # warn=False: this pass only computes the new set membership,
587
+ # never stores it directly. add() still warns exactly once,
588
+ # from the replaced instance's own __post_init__ -> _normset;
589
+ # remove() never reaches __post_init__ with the dead entry
590
+ # (it is subtracted out here), so it stays silent -- correct,
591
+ # since a removal stores nothing a warning could be about.
592
+ # That silence covers the entry BEING REMOVED only: a
593
+ # different multi-word entry still stored re-warns from the
594
+ # derived instance's __post_init__, since the warning is
595
+ # per-construction by design.
596
+ normalized = _normset(words, name, warn=False)
483
597
  updates[name] = (current | normalized if op == "add"
484
598
  else current - normalized)
485
599
  # mypy's dataclasses.replace() typing checks a **dict's single
@@ -506,8 +620,10 @@ def _default_lexicon() -> Lexicon:
506
620
  from nameparser.config.maiden_markers import MAIDEN_MARKERS
507
621
  from nameparser.config.prefixes import NON_FIRST_NAME_PREFIXES, PREFIXES
508
622
  from nameparser.config.suffixes import (
509
- SUFFIX_ACRONYMS, SUFFIX_ACRONYMS_AMBIGUOUS, SUFFIX_NOT_ACRONYMS,
623
+ GLUED_HONORIFICS, SUFFIX_ACRONYMS, SUFFIX_ACRONYMS_AMBIGUOUS,
624
+ SUFFIX_NOT_ACRONYMS,
510
625
  )
626
+ from nameparser.config.surnames import KOREAN_SURNAMES
511
627
  from nameparser.config.titles import FIRST_NAME_TITLES, TITLES
512
628
 
513
629
  # v1 data modules export plain `set[str]`; wrap each at this call site
@@ -527,6 +643,10 @@ def _default_lexicon() -> Lexicon:
527
643
  conjunctions=frozenset(CONJUNCTIONS),
528
644
  bound_given_names=frozenset(BOUND_FIRST_NAMES),
529
645
  maiden_markers=frozenset(MAIDEN_MARKERS),
646
+ # surnames.py is born frozen (#293) -- no call-site wrap needed,
647
+ # unlike the v1 modules above (their wraps drop when #293 lands)
648
+ surnames=KOREAN_SURNAMES,
649
+ honorific_tails=frozenset(GLUED_HONORIFICS),
530
650
  # pass canonical pair-tuples so this strictly-typed call site never
531
651
  # feeds a Mapping to the tuple-annotated field; __post_init__
532
652
  # still tolerates a Mapping at runtime for interactive use