eKoNLPy 2.2.1__tar.gz → 2.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CHANGELOG.md +23 -0
  2. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/PKG-INFO +1 -1
  3. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/pyproject.toml +2 -1
  4. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/__cli__.py +7 -5
  5. ekonlpy-2.2.2/src/ekonlpy/_version.py +1 -0
  6. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/base/base.py +7 -7
  7. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/__init__.py +2 -0
  8. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/tagset.py +0 -1
  9. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/etag/_template.py +35 -26
  10. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/mecab/_mecab.py +19 -19
  11. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/mecab/_userdic.py +55 -23
  12. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/base.py +75 -37
  13. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/euko.py +12 -6
  14. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/hiv4.py +9 -3
  15. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/kosac.py +55 -40
  16. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/lm.py +9 -3
  17. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/mpck.py +151 -102
  18. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/mpko.py +9 -6
  19. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/utils.py +77 -59
  20. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/tag/_mecab.py +51 -44
  21. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/tag/_postprocess.py +17 -12
  22. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/utils/dictionary.py +9 -14
  23. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/utils/io.py +56 -55
  24. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/demo_breakdown_api.py +2 -2
  25. ekonlpy-2.2.2/tests/ekonlpy/test_io.py +66 -0
  26. ekonlpy-2.2.2/tests/ekonlpy/test_sentiment_regressions.py +152 -0
  27. ekonlpy-2.2.2/tests/ekonlpy/test_userdic.py +90 -0
  28. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/uv.lock +26 -26
  29. ekonlpy-2.2.1/src/ekonlpy/_version.py +0 -1
  30. ekonlpy-2.2.1/tests/ekonlpy/test_sentiment_regressions.py +0 -57
  31. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.copier-config.poetry.yaml +0 -0
  32. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.copier-config.yaml +0 -0
  33. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.copierignore +0 -0
  34. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.editorconfig +0 -0
  35. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.envrc +0 -0
  36. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.gitattributes +0 -0
  37. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
  38. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
  39. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/dependabot.yaml +0 -0
  40. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/labels.yaml +0 -0
  41. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/release.yaml +0 -0
  42. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/deploy-docs.yaml +0 -0
  43. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/lint_and_test.yaml +0 -0
  44. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/release-test.yaml +0 -0
  45. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/release.yaml +0 -0
  46. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.gitignore +0 -0
  47. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.pre-commit-config.yaml +0 -0
  48. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.python-version +0 -0
  49. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.readthedocs.yaml +0 -0
  50. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.tasks-extra.toml +0 -0
  51. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.tasks.toml +0 -0
  52. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CITATION.cff +0 -0
  53. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CLAUDE.md +0 -0
  54. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CODE_OF_CONDUCT.md +0 -0
  55. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CONTRIBUTING.md +0 -0
  56. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/LICENSE +0 -0
  57. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/Makefile +0 -0
  58. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/README.md +0 -0
  59. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/REVIEW.md +0 -0
  60. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/codecov.yaml +0 -0
  61. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/codecov.yml +0 -0
  62. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/CNAME +0 -0
  63. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/cli.md +0 -0
  64. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/contributing.md +0 -0
  65. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/dictionaries.md +0 -0
  66. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/eKoNLPy.pdf +0 -0
  67. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/index.md +0 -0
  68. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/installation.md +0 -0
  69. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/javascripts/mathjax.js +0 -0
  70. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/requirements.txt +0 -0
  71. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/sentiment.md +0 -0
  72. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/tagging.md +0 -0
  73. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/troubleshooting.md +0 -0
  74. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/mkdocs.yaml +0 -0
  75. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/__init__.py +1 -1
  76. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/base/__init__.py +0 -0
  77. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ADJECTIVES.txt +0 -0
  78. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ADVERBS.txt +0 -0
  79. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/COUNTRY.txt +0 -0
  80. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/CURRENCY.txt +0 -0
  81. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ECON_PHRASES.txt +0 -0
  82. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ECON_TERMS.txt +0 -0
  83. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ENTITY.txt +0 -0
  84. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/FOREIGN_TERMS.txt +0 -0
  85. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/GENERIC.txt +0 -0
  86. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/INDUSTRY_TERMS.txt +0 -0
  87. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/INSTITUTION.txt +0 -0
  88. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/LEMMA.txt +0 -0
  89. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/NAMES.txt +0 -0
  90. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/NOUNS.txt +0 -0
  91. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/POLARITY_PHRASES.txt +0 -0
  92. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/PROPER_NOUNS.txt +0 -0
  93. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SECTOR.txt +0 -0
  94. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS.txt +0 -0
  95. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS_CUST.txt +0 -0
  96. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS_EN.txt +0 -0
  97. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS_KOR.txt +0 -0
  98. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM.txt +0 -0
  99. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_MAG.txt +0 -0
  100. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_PHRASES.txt +0 -0
  101. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_TERMS.txt +0 -0
  102. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_VA.txt +0 -0
  103. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/UNIT.txt +0 -0
  104. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/VERBS.txt +0 -0
  105. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Currencies.txt +0 -0
  106. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/DatesandNumbers.txt +0 -0
  107. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Generic.txt +0 -0
  108. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Geographic.txt +0 -0
  109. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/HIV-4.csv +0 -0
  110. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/LM.csv +0 -0
  111. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Names.txt +0 -0
  112. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/euko/mp_uncertainty_lexicon_lex.csv +0 -0
  113. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/euko/mp_uncertainty_lexicon_mkt.csv +0 -0
  114. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/kosac/expressive-type.csv +0 -0
  115. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/kosac/intensity.csv +0 -0
  116. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/kosac/polarity.csv +0 -0
  117. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_lex.csv +0 -0
  118. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_mkt.csv +0 -0
  119. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_mkt_n3.csv +0 -0
  120. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_mkt_n7.csv +0 -0
  121. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_vocab.txt +0 -0
  122. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_wordset.txt +0 -0
  123. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/model/MPKC.nbc +0 -0
  124. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/etag/__init__.py +0 -0
  125. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/mecab/__init__.py +0 -0
  126. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/py.typed +0 -0
  127. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/__init__.py +0 -0
  128. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/tag/__init__.py +0 -0
  129. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/utils/__init__.py +0 -0
  130. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_fugashi.py +0 -0
  131. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_init.py +0 -0
  132. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_make_gate.py +0 -0
  133. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_sentiment.py +0 -0
  134. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_tagger.py +0 -0
  135. {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_tagger_regressions.py +0 -0
@@ -1,5 +1,28 @@
1
1
  <!--next-version-placeholder-->
2
2
 
3
+ ## v2.2.2 (2026-09-17)
4
+
5
+ ### Bug Fixes
6
+
7
+ - Harden data loading, correct classifier metric bugs, and modernize typing
8
+ ([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
9
+ [`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
10
+
11
+ - Resolve fugashi-build-dict via PATH fallback and tighten loader consistency
12
+ ([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
13
+ [`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
14
+
15
+ - Use forward slashes for dictionary compiler paths on Windows
16
+ ([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
17
+ [`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
18
+
19
+ ### Testing
20
+
21
+ - Expect normalized dictionary compiler paths in build_userdic assertions
22
+ ([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
23
+ [`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
24
+
25
+
3
26
  ## v2.2.1 (2026-09-16)
4
27
 
5
28
  ### Bug Fixes
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: eKoNLPy
3
- Version: 2.2.1
3
+ Version: 2.2.2
4
4
  Summary: A Korean natural language processing toolkit for economic analysis
5
5
  Project-URL: Homepage, https://ekonlpy.entelecheia.ai
6
6
  Project-URL: Repository, https://github.com/entelecheia/eKoNLPy
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "eKoNLPy"
3
- version = "2.2.1"
3
+ version = "2.2.2"
4
4
  description = "A Korean natural language processing toolkit for economic analysis"
5
5
  authors = [{ name = "Young Joon Lee", email = "yj.lee@chu.ac.kr" }]
6
6
  license = "MIT"
@@ -101,6 +101,7 @@ skip = [
101
101
  'docs',
102
102
  'tests',
103
103
  'venv',
104
+ '.venv',
104
105
  '.copier-template',
105
106
  '.refs',
106
107
  ]
@@ -2,11 +2,13 @@
2
2
 
3
3
  # Importing the libraries
4
4
 
5
+ from typing import Optional
6
+
5
7
  import click
6
8
 
7
9
  from ._version import __version__
8
10
 
9
- CONTEXT_SETTINGS = dict(help_option_names=["-h", "--help"])
11
+ CONTEXT_SETTINGS = {"help_option_names": ["-h", "--help"]}
10
12
 
11
13
 
12
14
  @click.command(context_settings=CONTEXT_SETTINGS)
@@ -18,16 +20,16 @@ CONTEXT_SETTINGS = dict(help_option_names=["-h", "--help"])
18
20
  default="ekonlpy",
19
21
  help="The tagger to use. [ekonlpy|mecab]]",
20
22
  )
21
- @click.option("--input", "-i", help="The input text to tag.")
23
+ @click.option("--input", "-i", "text", help="The input text to tag.")
22
24
  @click.pass_context
23
- def main(ctk, tagger, input):
25
+ def main(ctk: click.Context, tagger: str, text: Optional[str]) -> None:
24
26
  """This is the command line interface for eKoNLPy.
25
27
 
26
28
  It is used to tag Korean text with a Korean morphological analyzer.
27
29
  """
28
30
  # Print a message to the user.
29
- if input:
30
- click.echo(tag(tagger, input))
31
+ if text:
32
+ click.echo(tag(tagger, text))
31
33
  else:
32
34
  # Print usage message to the user.
33
35
  click.echo(ctk.get_help())
@@ -0,0 +1 @@
1
+ __version__ = "2.2.2"
@@ -1,5 +1,5 @@
1
1
  import logging
2
- from typing import List, Optional, Tuple
2
+ from typing import Optional
3
3
 
4
4
  logger = logging.getLogger(__name__)
5
5
 
@@ -10,13 +10,13 @@ class BaseMecab:
10
10
  def parse(
11
11
  self,
12
12
  text: str,
13
- ) -> List[Tuple[str, str]]:
13
+ ) -> list[tuple[str, str]]:
14
14
  raise NotImplementedError
15
15
 
16
16
  def pos(
17
17
  self,
18
18
  text: str,
19
- ) -> List[Tuple[str, str]]:
19
+ ) -> list[tuple[str, str]]:
20
20
  return self.parse(text)
21
21
 
22
22
  def tokenize(
@@ -24,7 +24,7 @@ class BaseMecab:
24
24
  text: str,
25
25
  strip_pos: bool = False,
26
26
  postag_delim: str = "/",
27
- ) -> List[str]:
27
+ ) -> list[str]:
28
28
  tokens = self.parse(text)
29
29
 
30
30
  return [
@@ -32,15 +32,15 @@ class BaseMecab:
32
32
  for token_pos in tokens
33
33
  ]
34
34
 
35
- def morphs(self, text: str) -> List[str]:
35
+ def morphs(self, text: str) -> list[str]:
36
36
  return self.tokenize(text, strip_pos=True)
37
37
 
38
38
  def nouns(
39
39
  self,
40
40
  text: str,
41
41
  flatten: bool = True,
42
- noun_pos: Optional[List[str]] = None,
43
- ) -> List[str]:
42
+ noun_pos: Optional[list[str]] = None,
43
+ ) -> list[str]:
44
44
  if not noun_pos:
45
45
  noun_pos = []
46
46
  return [surface for surface, pos in self.pos(text) if pos in noun_pos]
@@ -1 +1,3 @@
1
1
  from .tagset import mecab_tags
2
+
3
+ __all__ = ["mecab_tags"]
@@ -124,7 +124,6 @@ skip_chk_tags = {
124
124
  ("MM", "SC", "NNG", "SC", "NR"): "NNG",
125
125
  ("NNG", "SC", "NNG"): "NNG",
126
126
  ("NNG", "SN", "NNG"): "NNG",
127
- ("NNG", "SN", "NNG"): "NNG",
128
127
  ("NNG", "SN", "NNBC"): "NNG",
129
128
  ("NNG", "SN", "NNBC", "NNG"): "NNG",
130
129
  ("NNG", "SY", "NNBC", "JX"): "NNG",
@@ -1,4 +1,4 @@
1
- from typing import Dict, Set
1
+ from typing import Union
2
2
 
3
3
  from ekonlpy.data.tagset import nouns_tags, pass_tags, skip_chk_tags, skip_tags
4
4
  from ekonlpy.utils.dictionary import TermDictionary
@@ -7,10 +7,10 @@ from ekonlpy.utils.dictionary import TermDictionary
7
7
  class ExtTagger:
8
8
  dictionary: TermDictionary
9
9
  max_ngram: int
10
- skip_chk_tags: Dict[str, str]
11
- skip_tags: Set[str]
12
- nouns_tags: Set[str]
13
- pass_tags: Set[str]
10
+ skip_chk_tags: dict[tuple[str, ...], str]
11
+ skip_tags: set[str]
12
+ nouns_tags: set[str]
13
+ pass_tags: set[tuple[str, str]]
14
14
 
15
15
  def __init__(
16
16
  self,
@@ -24,27 +24,27 @@ class ExtTagger:
24
24
  self.nouns_tags = set(nouns_tags)
25
25
  self.pass_tags = set(pass_tags)
26
26
 
27
- def add_skip_chk_tags(self, template):
27
+ def add_skip_chk_tags(self, template: dict[tuple[str, ...], str]) -> None:
28
28
  if isinstance(template, dict):
29
29
  self.skip_chk_tags.update(template)
30
30
 
31
- def add_skip_tags(self, tags):
31
+ def add_skip_tags(self, tags: Union[list[str], set[str]]) -> None:
32
32
  if isinstance(tags, (list, set)):
33
33
  self.skip_tags.update(tags)
34
34
 
35
- def pos(self, tokens):
36
- def ctagger(
37
- ctokens,
38
- max_ngram,
39
- cnouns_tags,
40
- cpass_tags,
41
- cskip_chk_tags,
42
- cskip_tags,
43
- cdictionary,
44
- ):
35
+ def pos(self, tokens: list[tuple[str, str]]) -> list[tuple[str, str]]: # noqa: C901
36
+ def ctagger( # noqa: C901
37
+ ctokens: list[tuple[str, str]],
38
+ max_ngram: int,
39
+ cnouns_tags: set[str],
40
+ cpass_tags: set[tuple[str, str]],
41
+ cskip_chk_tags: dict[tuple[str, ...], str],
42
+ cskip_tags: set[str],
43
+ cdictionary: TermDictionary,
44
+ ) -> list[tuple[str, str]]:
45
45
  tokens_org = ctokens
46
46
  num_tokens = len(ctokens)
47
- tokens_new = []
47
+ tokens_new: list[tuple[str, str]] = []
48
48
  ipos = 0
49
49
 
50
50
  while ipos < num_tokens:
@@ -53,17 +53,19 @@ class ExtTagger:
53
53
  # if found a word from the dictionary, skip for loop
54
54
  if word_found or ipos + ngram > num_tokens:
55
55
  continue
56
- if any(word.isspace() for word, _ in tokens_org[ipos:ipos + ngram]):
56
+ if any(
57
+ word.isspace() for word, _ in tokens_org[ipos : ipos + ngram]
58
+ ):
57
59
  continue
58
60
 
59
- tmp_tags = []
60
- for j in range(ngram):
61
- tmp_tags.append(
61
+ tmp_tags = tuple(
62
+ (
62
63
  "NNG"
63
64
  if tokens_org[ipos + j][1] in cnouns_tags
64
65
  else tokens_org[ipos + j][1]
65
66
  )
66
- tmp_tags = tuple(tmp_tags)
67
+ for j in range(ngram)
68
+ )
67
69
 
68
70
  if tmp_tags not in cpass_tags:
69
71
  new_word = ""
@@ -75,7 +77,7 @@ class ExtTagger:
75
77
  ipos += ngram
76
78
  word_found = True
77
79
 
78
- if not word_found and tmp_tags in cskip_chk_tags.keys():
80
+ if not word_found and tmp_tags in cskip_chk_tags:
79
81
  new_word = ""
80
82
  num_word = ""
81
83
  for j in range(ngram):
@@ -107,7 +109,11 @@ class ExtTagger:
107
109
  return tokens_new
108
110
 
109
111
  tokens = [
110
- (w, t) if w.isspace() else (w.strip(), self.dictionary.check_tag(w.strip(), t))
112
+ (
113
+ (w, t)
114
+ if w.isspace()
115
+ else (w.strip(), self.dictionary.check_tag(w.strip(), t))
116
+ )
111
117
  for w, t in tokens
112
118
  ]
113
119
 
@@ -130,6 +136,9 @@ class ExtTagger:
130
136
  self.dictionary,
131
137
  )
132
138
 
133
- tokens = [(w, t if w.isspace() else self.dictionary.check_tag(w, t)) for w, t in tokens]
139
+ tokens = [
140
+ (w, t if w.isspace() else self.dictionary.check_tag(w, t))
141
+ for w, t in tokens
142
+ ]
134
143
 
135
144
  return tokens
@@ -1,7 +1,8 @@
1
1
  import logging
2
2
  import os
3
3
  from collections import namedtuple
4
- from typing import Any, List, Optional, Sequence, Tuple, Union
4
+ from collections.abc import Sequence
5
+ from typing import Optional, Union
5
6
 
6
7
  import fugashi as _mecab
7
8
 
@@ -29,14 +30,14 @@ Feature = namedtuple(
29
30
  # expression='하/XSA/*+ᄇ니다/EF/*')),
30
31
 
31
32
 
32
- def _extract_feature(values: Sequence[Any]) -> Feature:
33
+ def _extract_feature(values: Sequence[object]) -> Feature:
33
34
  # Reference:
34
35
  # - http://taku910.github.io/mecab/learn.html
35
36
  # - https://docs.google.com/spreadsheets/d/1-9blXKjtjeKZqsf4NzHeYJCrr49-nXeRF6D80udfcwY
36
37
  # - https://bitbucket.org/eunjeon/mecab-ko-dic/src/master/utils/dictionary/lexicon.py
37
38
 
38
39
  # feature = <pos>,<semantic>,<has_jongseong>,<reading>,<type>,<start_pos>,<end_pos>,<expression>
39
- assert len(values) == 8
40
+ assert len(values) == 8 # noqa: S101
40
41
 
41
42
  feature_values = [value if value != "*" else None for value in values]
42
43
  feature = dict(zip(Feature._fields, feature_values))
@@ -54,14 +55,14 @@ class Mecab(BaseMecab):
54
55
  backend = "fugashi"
55
56
  verbose: bool = False
56
57
 
57
- _tagger: _mecab.GenericTagger # type: ignore
58
+ _tagger: _mecab.GenericTagger # type: ignore[no-any-unimported]
58
59
 
59
60
  def __init__(
60
61
  self,
61
62
  dicdir: Optional[str] = None,
62
63
  userdic_path: Optional[str] = None,
63
64
  verbose: bool = False,
64
- **kwargs,
65
+ **kwargs: object,
65
66
  ):
66
67
  import mecab_ko_dic
67
68
 
@@ -85,22 +86,21 @@ class Mecab(BaseMecab):
85
86
  userdic_path,
86
87
  )
87
88
  try:
88
- self._tagger = _mecab.GenericTagger(MECAB_ARGS) # type: ignore
89
+ self._tagger = _mecab.GenericTagger(MECAB_ARGS)
89
90
  if self.verbose:
90
91
  dictionary_info = self._tagger.dictionary_info
91
92
  sysdic_path = dictionary_info[0]["filename"]
92
93
  logger.debug("Mecab is loaded from %s", sysdic_path)
93
94
  except RuntimeError as e:
94
- raise MeCabError(
95
- 'The MeCab dictionary does not exist at "%s". Is the dictionary correctly installed?\nYou can also try entering the dictionary path when initializing the MeCab class: "MeCab(\'/some/dic/path\')"'
96
- % dicdir
95
+ raise MeCabError( # noqa: TRY003
96
+ f'The MeCab dictionary does not exist at "{dicdir}". Is the dictionary correctly installed?\nYou can also try entering the dictionary path when initializing the MeCab class: "MeCab(\'/some/dic/path\')"'
97
97
  ) from e
98
98
  except NameError as e:
99
- raise MeCabError(
99
+ raise MeCabError( # noqa: TRY003
100
100
  "The fugashi package is not installed. Please install fugashi with: pip install fugashi"
101
101
  ) from e
102
102
 
103
- def _parse(self, text: str) -> List[Tuple[str, Feature]]:
103
+ def _parse(self, text: str) -> list[tuple[str, Feature]]:
104
104
  return [
105
105
  (node.surface, _extract_feature(node.feature))
106
106
  for node in self._tagger(text)
@@ -111,7 +111,7 @@ class Mecab(BaseMecab):
111
111
  text: str,
112
112
  flatten: bool = True,
113
113
  include_whitespace_token: bool = False,
114
- ) -> List[Tuple[str, str]]:
114
+ ) -> list[tuple[str, str]]:
115
115
  """Tag text, optionally preserving original whitespace runs as SP tokens."""
116
116
  res = []
117
117
  cursor = 0
@@ -138,19 +138,19 @@ class Mecab(BaseMecab):
138
138
  text: str,
139
139
  flatten: bool = True,
140
140
  include_whitespace_token: bool = False,
141
- ) -> List[Tuple[str, str]]:
141
+ ) -> list[tuple[str, str]]:
142
142
  return self.parse(
143
143
  text, flatten=flatten, include_whitespace_token=include_whitespace_token
144
144
  )
145
145
 
146
- def tokenize(
146
+ def tokenize( # type: ignore[override]
147
147
  self,
148
148
  text: str,
149
149
  flatten: bool = True,
150
150
  include_whitespace_token: bool = False,
151
151
  strip_pos: bool = False,
152
152
  postag_delim: str = "/",
153
- ) -> List[str]:
153
+ ) -> list[str]:
154
154
  tokens = self.parse(
155
155
  text, flatten=flatten, include_whitespace_token=include_whitespace_token
156
156
  )
@@ -160,17 +160,17 @@ class Mecab(BaseMecab):
160
160
  for token_pos in tokens
161
161
  ]
162
162
 
163
- def morphs(self, text: str, flatten: bool = True) -> List[str]:
163
+ def morphs(self, text: str, flatten: bool = True) -> list[str]:
164
164
  return self.tokenize(
165
165
  text, flatten=flatten, strip_pos=True, include_whitespace_token=False
166
166
  )
167
167
 
168
168
  def nouns(
169
169
  self,
170
- text: Union[str, List[Tuple[str, str]]],
170
+ text: Union[str, list[tuple[str, str]]],
171
171
  flatten: bool = True,
172
- noun_pos: Optional[List[str]] = None,
173
- ) -> List[str]:
172
+ noun_pos: Optional[list[str]] = None,
173
+ ) -> list[str]:
174
174
  if not noun_pos:
175
175
  noun_pos = ["NNG", "NNP", "XSN", "SL", "XR", "NNB", "NR"]
176
176
  tagged = (
@@ -1,9 +1,12 @@
1
1
  import logging
2
2
  import os
3
+ import shutil
3
4
  import subprocess
5
+ import sys
4
6
  from collections import namedtuple
7
+ from collections.abc import Iterator
5
8
  from pathlib import Path
6
- from typing import Any, Dict, List, Optional, Sequence, Tuple
9
+ from typing import Optional
7
10
 
8
11
  import mecab_ko_dic
9
12
  import pandas as pd
@@ -46,23 +49,26 @@ ContextEntry = namedtuple(
46
49
  )
47
50
 
48
51
 
49
- def iternamedtuples(df):
50
- Row = namedtuple("Row", df.columns)
52
+ def iternamedtuples( # type: ignore[no-any-unimported]
53
+ df: pd.DataFrame,
54
+ ) -> Iterator[DicEntry]:
51
55
  for row in df.itertuples():
52
- yield Row(*row[1:])
56
+ yield DicEntry(*row[1:])
53
57
 
54
58
 
55
- def has_jongseong(c):
59
+ def has_jongseong(c: str) -> bool:
56
60
  return int((ord(c[-1]) - 0xAC00) % 28) != 0
57
61
 
58
62
 
59
63
  class MecabDicConfig:
60
- userdic: Dict[str, DicEntry] = {}
64
+ userdic: dict[str, DicEntry]
61
65
  dicdir: str = mecab_ko_dic.DICDIR
62
- left_ids: List[ContextEntry] = []
63
- right_ids: List[ContextEntry] = []
66
+ left_ids: list[ContextEntry]
67
+ right_ids: list[ContextEntry]
68
+ userdic_path: Optional[str] = None
64
69
 
65
70
  def __init__(self, userdic_path: Optional[str] = None):
71
+ self.userdic_path = userdic_path
66
72
  if userdic_path:
67
73
  self.load_userdic(userdic_path)
68
74
  else:
@@ -71,13 +77,13 @@ class MecabDicConfig:
71
77
  self.left_ids = self.load_context_ids("left-id.def")
72
78
  self.right_ids = self.load_context_ids("right-id.def")
73
79
 
74
- def load_context_ids(self, id_file: str) -> List[ContextEntry]:
80
+ def load_context_ids(self, id_file: str) -> list[ContextEntry]:
75
81
  id_file = os.path.join(self.dicdir, id_file)
76
82
  context_ids = []
77
- with open(id_file, "r", encoding="utf-8") as f:
83
+ with open(id_file, encoding="utf-8") as f:
78
84
  for line in f:
79
- id, vals = line.split()
80
- entry = ContextEntry(id, *vals.split(","))
85
+ entry_id, vals = line.split()
86
+ entry = ContextEntry(entry_id, *vals.split(","))
81
87
  context_ids.append(entry)
82
88
  return context_ids
83
89
 
@@ -85,6 +91,7 @@ class MecabDicConfig:
85
91
  for entry in self.left_ids:
86
92
  if entry.pos == search.pos and entry.semantic == search.semantic:
87
93
  return entry.id
94
+ return None
88
95
 
89
96
  def find_right_context_id(self, search: DicEntry) -> Optional[str]:
90
97
  for entry in self.right_ids:
@@ -94,19 +101,20 @@ class MecabDicConfig:
94
101
  and entry.has_jongseong == search.has_jongseong
95
102
  ):
96
103
  return entry.id
104
+ return None
97
105
 
98
- def load_userdic(self, userdic_path: str):
106
+ def load_userdic(self, userdic_path: str) -> None:
99
107
  userdic_path_ = Path(userdic_path)
100
108
 
101
109
  if userdic_path_.is_dir():
102
110
  self.userdic = {}
103
111
  for f in userdic_path_.glob("*.csv"):
104
112
  df = pd.read_csv(f, names=DicEntry._fields)
105
- dic = {e.surface: DicEntry(*e) for e in iternamedtuples(df)}
113
+ dic = {e.surface: e for e in iternamedtuples(df)}
106
114
  self.userdic = {**self.userdic, **dic}
107
115
  else:
108
116
  df = pd.read_csv(userdic_path_, names=DicEntry._fields)
109
- self.userdic = {e.surface: DicEntry(*e) for e in iternamedtuples(df)}
117
+ self.userdic = {e.surface: e for e in iternamedtuples(df)}
110
118
  logger.info("No. of user dictionary entires loaded: %d", len(self.userdic))
111
119
 
112
120
  def add_entry_to_userdic(
@@ -116,7 +124,7 @@ class MecabDicConfig:
116
124
  semantic: str = "*",
117
125
  reading: Optional[str] = None,
118
126
  cost: int = 1000,
119
- ):
127
+ ) -> None:
120
128
  entry = DicEntry(
121
129
  surface=surface,
122
130
  cost=cost,
@@ -131,7 +139,7 @@ class MecabDicConfig:
131
139
  )
132
140
  self.userdic[surface] = entry
133
141
 
134
- def adjust_context_ids(self):
142
+ def adjust_context_ids(self) -> None:
135
143
  for entry in self.userdic.values():
136
144
  entry = entry._replace(
137
145
  left_id=self.find_left_context_id(entry),
@@ -139,11 +147,11 @@ class MecabDicConfig:
139
147
  )
140
148
  self.userdic[entry.surface] = entry
141
149
 
142
- def adjust_costs(self, cost: int = 1000):
150
+ def adjust_costs(self, cost: int = 1000) -> None:
143
151
  for surface, entry in self.userdic.items():
144
152
  self.userdic[surface] = entry._replace(cost=cost)
145
153
 
146
- def save_userdic(self, save_path: str):
154
+ def save_userdic(self, save_path: str) -> None:
147
155
  if len(self.userdic) > 0:
148
156
  df = pd.DataFrame(self.userdic.values())
149
157
  df.to_csv(save_path, header=False, index=False)
@@ -153,11 +161,35 @@ class MecabDicConfig:
153
161
  else:
154
162
  logger.warning("No user dictionary entries to save.")
155
163
 
164
+ @staticmethod
165
+ def _build_dict_executable() -> str:
166
+ executable = shutil.which("fugashi-build-dict")
167
+ if executable:
168
+ return executable
169
+ binary = "fugashi-build-dict.exe" if os.name == "nt" else "fugashi-build-dict"
170
+ return os.path.join(os.path.dirname(sys.executable), binary)
171
+
172
+ @staticmethod
173
+ def _build_dict_path(path: str) -> str:
174
+ # The MeCab dictionary compiler eats backslashes as escape characters
175
+ # in its arguments, so Windows paths must use forward slashes.
176
+ return path.replace("\\", "/") if os.name == "nt" else path
177
+
156
178
  def build_userdic(
157
179
  self, built_userdic_path: str, userdic_path: Optional[str] = None
158
- ):
180
+ ) -> None:
159
181
  if userdic_path:
160
182
  self.userdic_path = userdic_path
161
- args = f'-d "{self.dicdir}" -u "{built_userdic_path}" {self.userdic_path}'
162
- # print(args)
163
- subprocess.run(["fugashi-build-dict", args])
183
+ if not self.userdic_path:
184
+ raise ValueError( # noqa: TRY003
185
+ "userdic_path is not set; call save_userdic() first or pass userdic_path."
186
+ )
187
+ cmd = [
188
+ self._build_dict_executable(),
189
+ "-d",
190
+ self._build_dict_path(self.dicdir),
191
+ "-u",
192
+ self._build_dict_path(built_userdic_path),
193
+ self._build_dict_path(self.userdic_path),
194
+ ]
195
+ subprocess.run(cmd, check=True) # noqa: S603