eKoNLPy 2.2.1__tar.gz → 2.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CHANGELOG.md +23 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/PKG-INFO +1 -1
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/pyproject.toml +2 -1
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/__cli__.py +7 -5
- ekonlpy-2.2.2/src/ekonlpy/_version.py +1 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/base/base.py +7 -7
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/__init__.py +2 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/tagset.py +0 -1
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/etag/_template.py +35 -26
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/mecab/_mecab.py +19 -19
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/mecab/_userdic.py +55 -23
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/base.py +75 -37
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/euko.py +12 -6
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/hiv4.py +9 -3
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/kosac.py +55 -40
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/lm.py +9 -3
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/mpck.py +151 -102
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/mpko.py +9 -6
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/utils.py +77 -59
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/tag/_mecab.py +51 -44
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/tag/_postprocess.py +17 -12
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/utils/dictionary.py +9 -14
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/utils/io.py +56 -55
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/demo_breakdown_api.py +2 -2
- ekonlpy-2.2.2/tests/ekonlpy/test_io.py +66 -0
- ekonlpy-2.2.2/tests/ekonlpy/test_sentiment_regressions.py +152 -0
- ekonlpy-2.2.2/tests/ekonlpy/test_userdic.py +90 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/uv.lock +26 -26
- ekonlpy-2.2.1/src/ekonlpy/_version.py +0 -1
- ekonlpy-2.2.1/tests/ekonlpy/test_sentiment_regressions.py +0 -57
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.copier-config.poetry.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.copier-config.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.copierignore +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.editorconfig +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.envrc +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.gitattributes +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/ISSUE_TEMPLATE/feature_request.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/dependabot.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/labels.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/release.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/deploy-docs.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/lint_and_test.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/release-test.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.github/workflows/release.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.gitignore +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.pre-commit-config.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.python-version +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.readthedocs.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.tasks-extra.toml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/.tasks.toml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CITATION.cff +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CLAUDE.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CODE_OF_CONDUCT.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/CONTRIBUTING.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/LICENSE +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/Makefile +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/README.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/REVIEW.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/codecov.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/codecov.yml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/CNAME +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/cli.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/contributing.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/dictionaries.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/eKoNLPy.pdf +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/index.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/installation.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/javascripts/mathjax.js +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/requirements.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/sentiment.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/tagging.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/docs/troubleshooting.md +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/mkdocs.yaml +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/__init__.py +1 -1
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/base/__init__.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ADJECTIVES.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ADVERBS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/COUNTRY.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/CURRENCY.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ECON_PHRASES.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ECON_TERMS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/ENTITY.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/FOREIGN_TERMS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/GENERIC.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/INDUSTRY_TERMS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/INSTITUTION.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/LEMMA.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/NAMES.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/NOUNS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/POLARITY_PHRASES.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/PROPER_NOUNS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SECTOR.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS_CUST.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS_EN.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/STOPWORDS_KOR.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_MAG.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_PHRASES.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_TERMS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/SYNONYM_VA.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/UNIT.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/dictionary/VERBS.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Currencies.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/DatesandNumbers.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Generic.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Geographic.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/HIV-4.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/LM.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/Names.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/euko/mp_uncertainty_lexicon_lex.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/euko/mp_uncertainty_lexicon_mkt.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/kosac/expressive-type.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/kosac/intensity.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/kosac/polarity.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_lex.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_mkt.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_mkt_n3.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_lexicon_mkt_n7.csv +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_vocab.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/lexicon/mpko/mp_polarity_wordset.txt +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/data/model/MPKC.nbc +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/etag/__init__.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/mecab/__init__.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/py.typed +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/sentiment/__init__.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/tag/__init__.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/src/ekonlpy/utils/__init__.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_fugashi.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_init.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_make_gate.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_sentiment.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_tagger.py +0 -0
- {ekonlpy-2.2.1 → ekonlpy-2.2.2}/tests/ekonlpy/test_tagger_regressions.py +0 -0
|
@@ -1,5 +1,28 @@
|
|
|
1
1
|
<!--next-version-placeholder-->
|
|
2
2
|
|
|
3
|
+
## v2.2.2 (2026-09-17)
|
|
4
|
+
|
|
5
|
+
### Bug Fixes
|
|
6
|
+
|
|
7
|
+
- Harden data loading, correct classifier metric bugs, and modernize typing
|
|
8
|
+
([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
|
|
9
|
+
[`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
|
|
10
|
+
|
|
11
|
+
- Resolve fugashi-build-dict via PATH fallback and tighten loader consistency
|
|
12
|
+
([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
|
|
13
|
+
[`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
|
|
14
|
+
|
|
15
|
+
- Use forward slashes for dictionary compiler paths on Windows
|
|
16
|
+
([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
|
|
17
|
+
[`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
|
|
18
|
+
|
|
19
|
+
### Testing
|
|
20
|
+
|
|
21
|
+
- Expect normalized dictionary compiler paths in build_userdic assertions
|
|
22
|
+
([#125](https://github.com/entelecheia/eKoNLPy/pull/125),
|
|
23
|
+
[`a6144e1`](https://github.com/entelecheia/eKoNLPy/commit/a6144e18f09f680874648dd3e5f1f152e34b8833))
|
|
24
|
+
|
|
25
|
+
|
|
3
26
|
## v2.2.1 (2026-09-16)
|
|
4
27
|
|
|
5
28
|
### Bug Fixes
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: eKoNLPy
|
|
3
|
-
Version: 2.2.
|
|
3
|
+
Version: 2.2.2
|
|
4
4
|
Summary: A Korean natural language processing toolkit for economic analysis
|
|
5
5
|
Project-URL: Homepage, https://ekonlpy.entelecheia.ai
|
|
6
6
|
Project-URL: Repository, https://github.com/entelecheia/eKoNLPy
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "eKoNLPy"
|
|
3
|
-
version = "2.2.
|
|
3
|
+
version = "2.2.2"
|
|
4
4
|
description = "A Korean natural language processing toolkit for economic analysis"
|
|
5
5
|
authors = [{ name = "Young Joon Lee", email = "yj.lee@chu.ac.kr" }]
|
|
6
6
|
license = "MIT"
|
|
@@ -101,6 +101,7 @@ skip = [
|
|
|
101
101
|
'docs',
|
|
102
102
|
'tests',
|
|
103
103
|
'venv',
|
|
104
|
+
'.venv',
|
|
104
105
|
'.copier-template',
|
|
105
106
|
'.refs',
|
|
106
107
|
]
|
|
@@ -2,11 +2,13 @@
|
|
|
2
2
|
|
|
3
3
|
# Importing the libraries
|
|
4
4
|
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
5
7
|
import click
|
|
6
8
|
|
|
7
9
|
from ._version import __version__
|
|
8
10
|
|
|
9
|
-
CONTEXT_SETTINGS =
|
|
11
|
+
CONTEXT_SETTINGS = {"help_option_names": ["-h", "--help"]}
|
|
10
12
|
|
|
11
13
|
|
|
12
14
|
@click.command(context_settings=CONTEXT_SETTINGS)
|
|
@@ -18,16 +20,16 @@ CONTEXT_SETTINGS = dict(help_option_names=["-h", "--help"])
|
|
|
18
20
|
default="ekonlpy",
|
|
19
21
|
help="The tagger to use. [ekonlpy|mecab]]",
|
|
20
22
|
)
|
|
21
|
-
@click.option("--input", "-i", help="The input text to tag.")
|
|
23
|
+
@click.option("--input", "-i", "text", help="The input text to tag.")
|
|
22
24
|
@click.pass_context
|
|
23
|
-
def main(ctk, tagger,
|
|
25
|
+
def main(ctk: click.Context, tagger: str, text: Optional[str]) -> None:
|
|
24
26
|
"""This is the command line interface for eKoNLPy.
|
|
25
27
|
|
|
26
28
|
It is used to tag Korean text with a Korean morphological analyzer.
|
|
27
29
|
"""
|
|
28
30
|
# Print a message to the user.
|
|
29
|
-
if
|
|
30
|
-
click.echo(tag(tagger,
|
|
31
|
+
if text:
|
|
32
|
+
click.echo(tag(tagger, text))
|
|
31
33
|
else:
|
|
32
34
|
# Print usage message to the user.
|
|
33
35
|
click.echo(ctk.get_help())
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "2.2.2"
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import logging
|
|
2
|
-
from typing import
|
|
2
|
+
from typing import Optional
|
|
3
3
|
|
|
4
4
|
logger = logging.getLogger(__name__)
|
|
5
5
|
|
|
@@ -10,13 +10,13 @@ class BaseMecab:
|
|
|
10
10
|
def parse(
|
|
11
11
|
self,
|
|
12
12
|
text: str,
|
|
13
|
-
) ->
|
|
13
|
+
) -> list[tuple[str, str]]:
|
|
14
14
|
raise NotImplementedError
|
|
15
15
|
|
|
16
16
|
def pos(
|
|
17
17
|
self,
|
|
18
18
|
text: str,
|
|
19
|
-
) ->
|
|
19
|
+
) -> list[tuple[str, str]]:
|
|
20
20
|
return self.parse(text)
|
|
21
21
|
|
|
22
22
|
def tokenize(
|
|
@@ -24,7 +24,7 @@ class BaseMecab:
|
|
|
24
24
|
text: str,
|
|
25
25
|
strip_pos: bool = False,
|
|
26
26
|
postag_delim: str = "/",
|
|
27
|
-
) ->
|
|
27
|
+
) -> list[str]:
|
|
28
28
|
tokens = self.parse(text)
|
|
29
29
|
|
|
30
30
|
return [
|
|
@@ -32,15 +32,15 @@ class BaseMecab:
|
|
|
32
32
|
for token_pos in tokens
|
|
33
33
|
]
|
|
34
34
|
|
|
35
|
-
def morphs(self, text: str) ->
|
|
35
|
+
def morphs(self, text: str) -> list[str]:
|
|
36
36
|
return self.tokenize(text, strip_pos=True)
|
|
37
37
|
|
|
38
38
|
def nouns(
|
|
39
39
|
self,
|
|
40
40
|
text: str,
|
|
41
41
|
flatten: bool = True,
|
|
42
|
-
noun_pos: Optional[
|
|
43
|
-
) ->
|
|
42
|
+
noun_pos: Optional[list[str]] = None,
|
|
43
|
+
) -> list[str]:
|
|
44
44
|
if not noun_pos:
|
|
45
45
|
noun_pos = []
|
|
46
46
|
return [surface for surface, pos in self.pos(text) if pos in noun_pos]
|
|
@@ -124,7 +124,6 @@ skip_chk_tags = {
|
|
|
124
124
|
("MM", "SC", "NNG", "SC", "NR"): "NNG",
|
|
125
125
|
("NNG", "SC", "NNG"): "NNG",
|
|
126
126
|
("NNG", "SN", "NNG"): "NNG",
|
|
127
|
-
("NNG", "SN", "NNG"): "NNG",
|
|
128
127
|
("NNG", "SN", "NNBC"): "NNG",
|
|
129
128
|
("NNG", "SN", "NNBC", "NNG"): "NNG",
|
|
130
129
|
("NNG", "SY", "NNBC", "JX"): "NNG",
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
from typing import
|
|
1
|
+
from typing import Union
|
|
2
2
|
|
|
3
3
|
from ekonlpy.data.tagset import nouns_tags, pass_tags, skip_chk_tags, skip_tags
|
|
4
4
|
from ekonlpy.utils.dictionary import TermDictionary
|
|
@@ -7,10 +7,10 @@ from ekonlpy.utils.dictionary import TermDictionary
|
|
|
7
7
|
class ExtTagger:
|
|
8
8
|
dictionary: TermDictionary
|
|
9
9
|
max_ngram: int
|
|
10
|
-
skip_chk_tags:
|
|
11
|
-
skip_tags:
|
|
12
|
-
nouns_tags:
|
|
13
|
-
pass_tags:
|
|
10
|
+
skip_chk_tags: dict[tuple[str, ...], str]
|
|
11
|
+
skip_tags: set[str]
|
|
12
|
+
nouns_tags: set[str]
|
|
13
|
+
pass_tags: set[tuple[str, str]]
|
|
14
14
|
|
|
15
15
|
def __init__(
|
|
16
16
|
self,
|
|
@@ -24,27 +24,27 @@ class ExtTagger:
|
|
|
24
24
|
self.nouns_tags = set(nouns_tags)
|
|
25
25
|
self.pass_tags = set(pass_tags)
|
|
26
26
|
|
|
27
|
-
def add_skip_chk_tags(self, template):
|
|
27
|
+
def add_skip_chk_tags(self, template: dict[tuple[str, ...], str]) -> None:
|
|
28
28
|
if isinstance(template, dict):
|
|
29
29
|
self.skip_chk_tags.update(template)
|
|
30
30
|
|
|
31
|
-
def add_skip_tags(self, tags):
|
|
31
|
+
def add_skip_tags(self, tags: Union[list[str], set[str]]) -> None:
|
|
32
32
|
if isinstance(tags, (list, set)):
|
|
33
33
|
self.skip_tags.update(tags)
|
|
34
34
|
|
|
35
|
-
def pos(self, tokens):
|
|
36
|
-
def ctagger(
|
|
37
|
-
ctokens,
|
|
38
|
-
max_ngram,
|
|
39
|
-
cnouns_tags,
|
|
40
|
-
cpass_tags,
|
|
41
|
-
cskip_chk_tags,
|
|
42
|
-
cskip_tags,
|
|
43
|
-
cdictionary,
|
|
44
|
-
):
|
|
35
|
+
def pos(self, tokens: list[tuple[str, str]]) -> list[tuple[str, str]]: # noqa: C901
|
|
36
|
+
def ctagger( # noqa: C901
|
|
37
|
+
ctokens: list[tuple[str, str]],
|
|
38
|
+
max_ngram: int,
|
|
39
|
+
cnouns_tags: set[str],
|
|
40
|
+
cpass_tags: set[tuple[str, str]],
|
|
41
|
+
cskip_chk_tags: dict[tuple[str, ...], str],
|
|
42
|
+
cskip_tags: set[str],
|
|
43
|
+
cdictionary: TermDictionary,
|
|
44
|
+
) -> list[tuple[str, str]]:
|
|
45
45
|
tokens_org = ctokens
|
|
46
46
|
num_tokens = len(ctokens)
|
|
47
|
-
tokens_new = []
|
|
47
|
+
tokens_new: list[tuple[str, str]] = []
|
|
48
48
|
ipos = 0
|
|
49
49
|
|
|
50
50
|
while ipos < num_tokens:
|
|
@@ -53,17 +53,19 @@ class ExtTagger:
|
|
|
53
53
|
# if found a word from the dictionary, skip for loop
|
|
54
54
|
if word_found or ipos + ngram > num_tokens:
|
|
55
55
|
continue
|
|
56
|
-
if any(
|
|
56
|
+
if any(
|
|
57
|
+
word.isspace() for word, _ in tokens_org[ipos : ipos + ngram]
|
|
58
|
+
):
|
|
57
59
|
continue
|
|
58
60
|
|
|
59
|
-
tmp_tags =
|
|
60
|
-
|
|
61
|
-
tmp_tags.append(
|
|
61
|
+
tmp_tags = tuple(
|
|
62
|
+
(
|
|
62
63
|
"NNG"
|
|
63
64
|
if tokens_org[ipos + j][1] in cnouns_tags
|
|
64
65
|
else tokens_org[ipos + j][1]
|
|
65
66
|
)
|
|
66
|
-
|
|
67
|
+
for j in range(ngram)
|
|
68
|
+
)
|
|
67
69
|
|
|
68
70
|
if tmp_tags not in cpass_tags:
|
|
69
71
|
new_word = ""
|
|
@@ -75,7 +77,7 @@ class ExtTagger:
|
|
|
75
77
|
ipos += ngram
|
|
76
78
|
word_found = True
|
|
77
79
|
|
|
78
|
-
if not word_found and tmp_tags in cskip_chk_tags
|
|
80
|
+
if not word_found and tmp_tags in cskip_chk_tags:
|
|
79
81
|
new_word = ""
|
|
80
82
|
num_word = ""
|
|
81
83
|
for j in range(ngram):
|
|
@@ -107,7 +109,11 @@ class ExtTagger:
|
|
|
107
109
|
return tokens_new
|
|
108
110
|
|
|
109
111
|
tokens = [
|
|
110
|
-
(
|
|
112
|
+
(
|
|
113
|
+
(w, t)
|
|
114
|
+
if w.isspace()
|
|
115
|
+
else (w.strip(), self.dictionary.check_tag(w.strip(), t))
|
|
116
|
+
)
|
|
111
117
|
for w, t in tokens
|
|
112
118
|
]
|
|
113
119
|
|
|
@@ -130,6 +136,9 @@ class ExtTagger:
|
|
|
130
136
|
self.dictionary,
|
|
131
137
|
)
|
|
132
138
|
|
|
133
|
-
tokens = [
|
|
139
|
+
tokens = [
|
|
140
|
+
(w, t if w.isspace() else self.dictionary.check_tag(w, t))
|
|
141
|
+
for w, t in tokens
|
|
142
|
+
]
|
|
134
143
|
|
|
135
144
|
return tokens
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import logging
|
|
2
2
|
import os
|
|
3
3
|
from collections import namedtuple
|
|
4
|
-
from
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from typing import Optional, Union
|
|
5
6
|
|
|
6
7
|
import fugashi as _mecab
|
|
7
8
|
|
|
@@ -29,14 +30,14 @@ Feature = namedtuple(
|
|
|
29
30
|
# expression='하/XSA/*+ᄇ니다/EF/*')),
|
|
30
31
|
|
|
31
32
|
|
|
32
|
-
def _extract_feature(values: Sequence[
|
|
33
|
+
def _extract_feature(values: Sequence[object]) -> Feature:
|
|
33
34
|
# Reference:
|
|
34
35
|
# - http://taku910.github.io/mecab/learn.html
|
|
35
36
|
# - https://docs.google.com/spreadsheets/d/1-9blXKjtjeKZqsf4NzHeYJCrr49-nXeRF6D80udfcwY
|
|
36
37
|
# - https://bitbucket.org/eunjeon/mecab-ko-dic/src/master/utils/dictionary/lexicon.py
|
|
37
38
|
|
|
38
39
|
# feature = <pos>,<semantic>,<has_jongseong>,<reading>,<type>,<start_pos>,<end_pos>,<expression>
|
|
39
|
-
assert len(values) == 8
|
|
40
|
+
assert len(values) == 8 # noqa: S101
|
|
40
41
|
|
|
41
42
|
feature_values = [value if value != "*" else None for value in values]
|
|
42
43
|
feature = dict(zip(Feature._fields, feature_values))
|
|
@@ -54,14 +55,14 @@ class Mecab(BaseMecab):
|
|
|
54
55
|
backend = "fugashi"
|
|
55
56
|
verbose: bool = False
|
|
56
57
|
|
|
57
|
-
_tagger: _mecab.GenericTagger # type: ignore
|
|
58
|
+
_tagger: _mecab.GenericTagger # type: ignore[no-any-unimported]
|
|
58
59
|
|
|
59
60
|
def __init__(
|
|
60
61
|
self,
|
|
61
62
|
dicdir: Optional[str] = None,
|
|
62
63
|
userdic_path: Optional[str] = None,
|
|
63
64
|
verbose: bool = False,
|
|
64
|
-
**kwargs,
|
|
65
|
+
**kwargs: object,
|
|
65
66
|
):
|
|
66
67
|
import mecab_ko_dic
|
|
67
68
|
|
|
@@ -85,22 +86,21 @@ class Mecab(BaseMecab):
|
|
|
85
86
|
userdic_path,
|
|
86
87
|
)
|
|
87
88
|
try:
|
|
88
|
-
self._tagger = _mecab.GenericTagger(MECAB_ARGS)
|
|
89
|
+
self._tagger = _mecab.GenericTagger(MECAB_ARGS)
|
|
89
90
|
if self.verbose:
|
|
90
91
|
dictionary_info = self._tagger.dictionary_info
|
|
91
92
|
sysdic_path = dictionary_info[0]["filename"]
|
|
92
93
|
logger.debug("Mecab is loaded from %s", sysdic_path)
|
|
93
94
|
except RuntimeError as e:
|
|
94
|
-
raise MeCabError(
|
|
95
|
-
'The MeCab dictionary does not exist at "
|
|
96
|
-
% dicdir
|
|
95
|
+
raise MeCabError( # noqa: TRY003
|
|
96
|
+
f'The MeCab dictionary does not exist at "{dicdir}". Is the dictionary correctly installed?\nYou can also try entering the dictionary path when initializing the MeCab class: "MeCab(\'/some/dic/path\')"'
|
|
97
97
|
) from e
|
|
98
98
|
except NameError as e:
|
|
99
|
-
raise MeCabError(
|
|
99
|
+
raise MeCabError( # noqa: TRY003
|
|
100
100
|
"The fugashi package is not installed. Please install fugashi with: pip install fugashi"
|
|
101
101
|
) from e
|
|
102
102
|
|
|
103
|
-
def _parse(self, text: str) ->
|
|
103
|
+
def _parse(self, text: str) -> list[tuple[str, Feature]]:
|
|
104
104
|
return [
|
|
105
105
|
(node.surface, _extract_feature(node.feature))
|
|
106
106
|
for node in self._tagger(text)
|
|
@@ -111,7 +111,7 @@ class Mecab(BaseMecab):
|
|
|
111
111
|
text: str,
|
|
112
112
|
flatten: bool = True,
|
|
113
113
|
include_whitespace_token: bool = False,
|
|
114
|
-
) ->
|
|
114
|
+
) -> list[tuple[str, str]]:
|
|
115
115
|
"""Tag text, optionally preserving original whitespace runs as SP tokens."""
|
|
116
116
|
res = []
|
|
117
117
|
cursor = 0
|
|
@@ -138,19 +138,19 @@ class Mecab(BaseMecab):
|
|
|
138
138
|
text: str,
|
|
139
139
|
flatten: bool = True,
|
|
140
140
|
include_whitespace_token: bool = False,
|
|
141
|
-
) ->
|
|
141
|
+
) -> list[tuple[str, str]]:
|
|
142
142
|
return self.parse(
|
|
143
143
|
text, flatten=flatten, include_whitespace_token=include_whitespace_token
|
|
144
144
|
)
|
|
145
145
|
|
|
146
|
-
def tokenize(
|
|
146
|
+
def tokenize( # type: ignore[override]
|
|
147
147
|
self,
|
|
148
148
|
text: str,
|
|
149
149
|
flatten: bool = True,
|
|
150
150
|
include_whitespace_token: bool = False,
|
|
151
151
|
strip_pos: bool = False,
|
|
152
152
|
postag_delim: str = "/",
|
|
153
|
-
) ->
|
|
153
|
+
) -> list[str]:
|
|
154
154
|
tokens = self.parse(
|
|
155
155
|
text, flatten=flatten, include_whitespace_token=include_whitespace_token
|
|
156
156
|
)
|
|
@@ -160,17 +160,17 @@ class Mecab(BaseMecab):
|
|
|
160
160
|
for token_pos in tokens
|
|
161
161
|
]
|
|
162
162
|
|
|
163
|
-
def morphs(self, text: str, flatten: bool = True) ->
|
|
163
|
+
def morphs(self, text: str, flatten: bool = True) -> list[str]:
|
|
164
164
|
return self.tokenize(
|
|
165
165
|
text, flatten=flatten, strip_pos=True, include_whitespace_token=False
|
|
166
166
|
)
|
|
167
167
|
|
|
168
168
|
def nouns(
|
|
169
169
|
self,
|
|
170
|
-
text: Union[str,
|
|
170
|
+
text: Union[str, list[tuple[str, str]]],
|
|
171
171
|
flatten: bool = True,
|
|
172
|
-
noun_pos: Optional[
|
|
173
|
-
) ->
|
|
172
|
+
noun_pos: Optional[list[str]] = None,
|
|
173
|
+
) -> list[str]:
|
|
174
174
|
if not noun_pos:
|
|
175
175
|
noun_pos = ["NNG", "NNP", "XSN", "SL", "XR", "NNB", "NR"]
|
|
176
176
|
tagged = (
|
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
import logging
|
|
2
2
|
import os
|
|
3
|
+
import shutil
|
|
3
4
|
import subprocess
|
|
5
|
+
import sys
|
|
4
6
|
from collections import namedtuple
|
|
7
|
+
from collections.abc import Iterator
|
|
5
8
|
from pathlib import Path
|
|
6
|
-
from typing import
|
|
9
|
+
from typing import Optional
|
|
7
10
|
|
|
8
11
|
import mecab_ko_dic
|
|
9
12
|
import pandas as pd
|
|
@@ -46,23 +49,26 @@ ContextEntry = namedtuple(
|
|
|
46
49
|
)
|
|
47
50
|
|
|
48
51
|
|
|
49
|
-
def iternamedtuples(
|
|
50
|
-
|
|
52
|
+
def iternamedtuples( # type: ignore[no-any-unimported]
|
|
53
|
+
df: pd.DataFrame,
|
|
54
|
+
) -> Iterator[DicEntry]:
|
|
51
55
|
for row in df.itertuples():
|
|
52
|
-
yield
|
|
56
|
+
yield DicEntry(*row[1:])
|
|
53
57
|
|
|
54
58
|
|
|
55
|
-
def has_jongseong(c):
|
|
59
|
+
def has_jongseong(c: str) -> bool:
|
|
56
60
|
return int((ord(c[-1]) - 0xAC00) % 28) != 0
|
|
57
61
|
|
|
58
62
|
|
|
59
63
|
class MecabDicConfig:
|
|
60
|
-
userdic:
|
|
64
|
+
userdic: dict[str, DicEntry]
|
|
61
65
|
dicdir: str = mecab_ko_dic.DICDIR
|
|
62
|
-
left_ids:
|
|
63
|
-
right_ids:
|
|
66
|
+
left_ids: list[ContextEntry]
|
|
67
|
+
right_ids: list[ContextEntry]
|
|
68
|
+
userdic_path: Optional[str] = None
|
|
64
69
|
|
|
65
70
|
def __init__(self, userdic_path: Optional[str] = None):
|
|
71
|
+
self.userdic_path = userdic_path
|
|
66
72
|
if userdic_path:
|
|
67
73
|
self.load_userdic(userdic_path)
|
|
68
74
|
else:
|
|
@@ -71,13 +77,13 @@ class MecabDicConfig:
|
|
|
71
77
|
self.left_ids = self.load_context_ids("left-id.def")
|
|
72
78
|
self.right_ids = self.load_context_ids("right-id.def")
|
|
73
79
|
|
|
74
|
-
def load_context_ids(self, id_file: str) ->
|
|
80
|
+
def load_context_ids(self, id_file: str) -> list[ContextEntry]:
|
|
75
81
|
id_file = os.path.join(self.dicdir, id_file)
|
|
76
82
|
context_ids = []
|
|
77
|
-
with open(id_file,
|
|
83
|
+
with open(id_file, encoding="utf-8") as f:
|
|
78
84
|
for line in f:
|
|
79
|
-
|
|
80
|
-
entry = ContextEntry(
|
|
85
|
+
entry_id, vals = line.split()
|
|
86
|
+
entry = ContextEntry(entry_id, *vals.split(","))
|
|
81
87
|
context_ids.append(entry)
|
|
82
88
|
return context_ids
|
|
83
89
|
|
|
@@ -85,6 +91,7 @@ class MecabDicConfig:
|
|
|
85
91
|
for entry in self.left_ids:
|
|
86
92
|
if entry.pos == search.pos and entry.semantic == search.semantic:
|
|
87
93
|
return entry.id
|
|
94
|
+
return None
|
|
88
95
|
|
|
89
96
|
def find_right_context_id(self, search: DicEntry) -> Optional[str]:
|
|
90
97
|
for entry in self.right_ids:
|
|
@@ -94,19 +101,20 @@ class MecabDicConfig:
|
|
|
94
101
|
and entry.has_jongseong == search.has_jongseong
|
|
95
102
|
):
|
|
96
103
|
return entry.id
|
|
104
|
+
return None
|
|
97
105
|
|
|
98
|
-
def load_userdic(self, userdic_path: str):
|
|
106
|
+
def load_userdic(self, userdic_path: str) -> None:
|
|
99
107
|
userdic_path_ = Path(userdic_path)
|
|
100
108
|
|
|
101
109
|
if userdic_path_.is_dir():
|
|
102
110
|
self.userdic = {}
|
|
103
111
|
for f in userdic_path_.glob("*.csv"):
|
|
104
112
|
df = pd.read_csv(f, names=DicEntry._fields)
|
|
105
|
-
dic = {e.surface:
|
|
113
|
+
dic = {e.surface: e for e in iternamedtuples(df)}
|
|
106
114
|
self.userdic = {**self.userdic, **dic}
|
|
107
115
|
else:
|
|
108
116
|
df = pd.read_csv(userdic_path_, names=DicEntry._fields)
|
|
109
|
-
self.userdic = {e.surface:
|
|
117
|
+
self.userdic = {e.surface: e for e in iternamedtuples(df)}
|
|
110
118
|
logger.info("No. of user dictionary entires loaded: %d", len(self.userdic))
|
|
111
119
|
|
|
112
120
|
def add_entry_to_userdic(
|
|
@@ -116,7 +124,7 @@ class MecabDicConfig:
|
|
|
116
124
|
semantic: str = "*",
|
|
117
125
|
reading: Optional[str] = None,
|
|
118
126
|
cost: int = 1000,
|
|
119
|
-
):
|
|
127
|
+
) -> None:
|
|
120
128
|
entry = DicEntry(
|
|
121
129
|
surface=surface,
|
|
122
130
|
cost=cost,
|
|
@@ -131,7 +139,7 @@ class MecabDicConfig:
|
|
|
131
139
|
)
|
|
132
140
|
self.userdic[surface] = entry
|
|
133
141
|
|
|
134
|
-
def adjust_context_ids(self):
|
|
142
|
+
def adjust_context_ids(self) -> None:
|
|
135
143
|
for entry in self.userdic.values():
|
|
136
144
|
entry = entry._replace(
|
|
137
145
|
left_id=self.find_left_context_id(entry),
|
|
@@ -139,11 +147,11 @@ class MecabDicConfig:
|
|
|
139
147
|
)
|
|
140
148
|
self.userdic[entry.surface] = entry
|
|
141
149
|
|
|
142
|
-
def adjust_costs(self, cost: int = 1000):
|
|
150
|
+
def adjust_costs(self, cost: int = 1000) -> None:
|
|
143
151
|
for surface, entry in self.userdic.items():
|
|
144
152
|
self.userdic[surface] = entry._replace(cost=cost)
|
|
145
153
|
|
|
146
|
-
def save_userdic(self, save_path: str):
|
|
154
|
+
def save_userdic(self, save_path: str) -> None:
|
|
147
155
|
if len(self.userdic) > 0:
|
|
148
156
|
df = pd.DataFrame(self.userdic.values())
|
|
149
157
|
df.to_csv(save_path, header=False, index=False)
|
|
@@ -153,11 +161,35 @@ class MecabDicConfig:
|
|
|
153
161
|
else:
|
|
154
162
|
logger.warning("No user dictionary entries to save.")
|
|
155
163
|
|
|
164
|
+
@staticmethod
|
|
165
|
+
def _build_dict_executable() -> str:
|
|
166
|
+
executable = shutil.which("fugashi-build-dict")
|
|
167
|
+
if executable:
|
|
168
|
+
return executable
|
|
169
|
+
binary = "fugashi-build-dict.exe" if os.name == "nt" else "fugashi-build-dict"
|
|
170
|
+
return os.path.join(os.path.dirname(sys.executable), binary)
|
|
171
|
+
|
|
172
|
+
@staticmethod
|
|
173
|
+
def _build_dict_path(path: str) -> str:
|
|
174
|
+
# The MeCab dictionary compiler eats backslashes as escape characters
|
|
175
|
+
# in its arguments, so Windows paths must use forward slashes.
|
|
176
|
+
return path.replace("\\", "/") if os.name == "nt" else path
|
|
177
|
+
|
|
156
178
|
def build_userdic(
|
|
157
179
|
self, built_userdic_path: str, userdic_path: Optional[str] = None
|
|
158
|
-
):
|
|
180
|
+
) -> None:
|
|
159
181
|
if userdic_path:
|
|
160
182
|
self.userdic_path = userdic_path
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
183
|
+
if not self.userdic_path:
|
|
184
|
+
raise ValueError( # noqa: TRY003
|
|
185
|
+
"userdic_path is not set; call save_userdic() first or pass userdic_path."
|
|
186
|
+
)
|
|
187
|
+
cmd = [
|
|
188
|
+
self._build_dict_executable(),
|
|
189
|
+
"-d",
|
|
190
|
+
self._build_dict_path(self.dicdir),
|
|
191
|
+
"-u",
|
|
192
|
+
self._build_dict_path(built_userdic_path),
|
|
193
|
+
self._build_dict_path(self.userdic_path),
|
|
194
|
+
]
|
|
195
|
+
subprocess.run(cmd, check=True) # noqa: S603
|