korean-datetime 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- korean_datetime/__init__.py +58 -0
- korean_datetime/__main__.py +61 -0
- korean_datetime/core/__init__.py +29 -0
- korean_datetime/core/clock.py +50 -0
- korean_datetime/core/evaluation.py +189 -0
- korean_datetime/core/numerals.py +92 -0
- korean_datetime/core/scanner.py +195 -0
- korean_datetime/core/types.py +24 -0
- korean_datetime/py.typed +0 -0
- korean_datetime/temporal/__init__.py +37 -0
- korean_datetime/temporal/ambiguity.py +34 -0
- korean_datetime/temporal/calendar_math.py +71 -0
- korean_datetime/temporal/context_hour.py +111 -0
- korean_datetime/temporal/data/__init__.py +0 -0
- korean_datetime/temporal/data/holidays.json +34 -0
- korean_datetime/temporal/evaluation.py +156 -0
- korean_datetime/temporal/expectation.py +730 -0
- korean_datetime/temporal/frame.py +363 -0
- korean_datetime/temporal/holiday_calendar.py +141 -0
- korean_datetime/temporal/holidays.py +84 -0
- korean_datetime/temporal/lexicon.py +324 -0
- korean_datetime/temporal/lunar.py +216 -0
- korean_datetime/temporal/model.py +95 -0
- korean_datetime/temporal/options.py +96 -0
- korean_datetime/temporal/parser.py +131 -0
- korean_datetime/temporal/postprocess.py +119 -0
- korean_datetime/temporal/ranges.py +145 -0
- korean_datetime/temporal/relative.py +91 -0
- korean_datetime/temporal/resolve.py +184 -0
- korean_datetime/temporal/resolve_date.py +333 -0
- korean_datetime/temporal/resolve_time.py +140 -0
- korean_datetime/temporal/rules.py +448 -0
- korean_datetime/temporal/tokens.py +95 -0
- korean_datetime-1.0.0.dist-info/METADATA +459 -0
- korean_datetime-1.0.0.dist-info/RECORD +38 -0
- korean_datetime-1.0.0.dist-info/WHEEL +4 -0
- korean_datetime-1.0.0.dist-info/entry_points.txt +2 -0
- korean_datetime-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""
|
|
2
|
+
korean_datetime: 한국어 날짜·시간·혼합 표현 추출·정규화 (표준 라이브러리만 사용)
|
|
3
|
+
|
|
4
|
+
>>> from datetime import datetime
|
|
5
|
+
>>> from korean_datetime import parse
|
|
6
|
+
>>> parse("다음주 월요일 저녁 7시 반", now=datetime(2026, 9, 28, 14, 30)).start
|
|
7
|
+
datetime.datetime(2026, 10, 5, 19, 30)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .core import current_reference, reference_time
|
|
11
|
+
from .temporal import (
|
|
12
|
+
Ambiguity,
|
|
13
|
+
AmbiguousHour,
|
|
14
|
+
BuiltinHolidays,
|
|
15
|
+
ChainedHolidays,
|
|
16
|
+
Cycle,
|
|
17
|
+
DateTableHolidays,
|
|
18
|
+
Direction,
|
|
19
|
+
Grain,
|
|
20
|
+
HolidayCalendar,
|
|
21
|
+
Kind,
|
|
22
|
+
ParseOptions,
|
|
23
|
+
TemporalExpression,
|
|
24
|
+
TemporalParser,
|
|
25
|
+
lunar_to_solar,
|
|
26
|
+
parse,
|
|
27
|
+
parse_all,
|
|
28
|
+
parse_date,
|
|
29
|
+
parse_datetime,
|
|
30
|
+
parse_time,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
__version__ = "1.0.0"
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"Ambiguity",
|
|
37
|
+
"AmbiguousHour",
|
|
38
|
+
"BuiltinHolidays",
|
|
39
|
+
"ChainedHolidays",
|
|
40
|
+
"Cycle",
|
|
41
|
+
"DateTableHolidays",
|
|
42
|
+
"Direction",
|
|
43
|
+
"Grain",
|
|
44
|
+
"HolidayCalendar",
|
|
45
|
+
"Kind",
|
|
46
|
+
"ParseOptions",
|
|
47
|
+
"TemporalExpression",
|
|
48
|
+
"TemporalParser",
|
|
49
|
+
"__version__",
|
|
50
|
+
"current_reference",
|
|
51
|
+
"lunar_to_solar",
|
|
52
|
+
"parse",
|
|
53
|
+
"parse_all",
|
|
54
|
+
"parse_date",
|
|
55
|
+
"parse_datetime",
|
|
56
|
+
"parse_time",
|
|
57
|
+
"reference_time",
|
|
58
|
+
]
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""명령행: 날짜/시간 표현을 JSON Lines로 출력
|
|
2
|
+
|
|
3
|
+
$ python -m korean_datetime "내일 저녁 7시" --now 2026-09-28T14:30
|
|
4
|
+
{"text": "내일 저녁 7시", ..., "value": "2026-09-29T19:00:00", ...}
|
|
5
|
+
|
|
6
|
+
인식 결과가 없으면 종료 코드 1.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import json
|
|
13
|
+
import sys
|
|
14
|
+
from collections.abc import Sequence
|
|
15
|
+
from datetime import datetime
|
|
16
|
+
|
|
17
|
+
from .temporal import AmbiguousHour, Cycle, ParseOptions, TemporalParser
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _build_arg_parser() -> argparse.ArgumentParser:
|
|
21
|
+
parser = argparse.ArgumentParser(prog="korean-datetime", description="한국어 날짜/시간 표현 정규화")
|
|
22
|
+
parser.add_argument("text", nargs="?", help="분석할 텍스트 (생략하면 표준 입력)")
|
|
23
|
+
parser.add_argument("--now", help="기준 시각 (ISO 8601, 예: 2026-09-28T14:30). 기본: 현재 시각")
|
|
24
|
+
parser.add_argument("--all", action="store_true", help="모든 표현 출력 (기본: 첫 표현만)")
|
|
25
|
+
parser.add_argument(
|
|
26
|
+
"--ambiguous-hour", choices=[p.value for p in AmbiguousHour], default="nearest_future"
|
|
27
|
+
)
|
|
28
|
+
parser.add_argument(
|
|
29
|
+
"--cycle",
|
|
30
|
+
choices=[c.value for c in Cycle],
|
|
31
|
+
default=Cycle.FUTURE.value,
|
|
32
|
+
help="생략된 연/월/날짜의 주기",
|
|
33
|
+
)
|
|
34
|
+
parser.add_argument("--compact-dates", action="store_true", help="1015, 261015 같은 숫자를 날짜로 인식")
|
|
35
|
+
parser.add_argument("--vague", action="store_true", help="'최근', '향후' 같은 막연한 때도 인식")
|
|
36
|
+
return parser
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
40
|
+
arg_parser = _build_arg_parser()
|
|
41
|
+
args = arg_parser.parse_args(argv)
|
|
42
|
+
try:
|
|
43
|
+
now = datetime.fromisoformat(args.now) if args.now else None
|
|
44
|
+
except ValueError:
|
|
45
|
+
arg_parser.error(f"--now 형식이 올바르지 않습니다: {args.now!r}")
|
|
46
|
+
options = ParseOptions(
|
|
47
|
+
cycle=Cycle(args.cycle),
|
|
48
|
+
ambiguous_hour=AmbiguousHour(args.ambiguous_hour),
|
|
49
|
+
compact_dates=args.compact_dates,
|
|
50
|
+
vague=args.vague,
|
|
51
|
+
)
|
|
52
|
+
text = args.text if args.text is not None else sys.stdin.read()
|
|
53
|
+
parser = TemporalParser(options)
|
|
54
|
+
results = parser.parse_all(text, now) if args.all else [r for r in [parser.parse(text, now)] if r]
|
|
55
|
+
for result in results:
|
|
56
|
+
print(json.dumps(result.to_dict(), ensure_ascii=False))
|
|
57
|
+
return 0 if results else 1
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
if __name__ == "__main__":
|
|
61
|
+
sys.exit(main())
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""날짜/시간 파서가 쓰는 내부 기반
|
|
2
|
+
|
|
3
|
+
- scanner: 한국어 단어 경계·조사를 인식하는 정규식 토큰 스캐너
|
|
4
|
+
- numerals: 한글 수사 파서 ('이십삼', '스물세')
|
|
5
|
+
- clock: 기준 시각 (요청 단위 reference_time)
|
|
6
|
+
- evaluation: 정답셋 기반 정량 평가
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from .clock import current_reference, reference_time
|
|
10
|
+
from .evaluation import EvaluationReport, GoldCase, Metrics, evaluate, load_gold
|
|
11
|
+
from .numerals import parse_korean_number, parse_native, parse_sino
|
|
12
|
+
from .scanner import Scanner, Split, Token, TokenRule
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"EvaluationReport",
|
|
16
|
+
"GoldCase",
|
|
17
|
+
"Metrics",
|
|
18
|
+
"Scanner",
|
|
19
|
+
"Split",
|
|
20
|
+
"Token",
|
|
21
|
+
"TokenRule",
|
|
22
|
+
"current_reference",
|
|
23
|
+
"evaluate",
|
|
24
|
+
"load_gold",
|
|
25
|
+
"parse_korean_number",
|
|
26
|
+
"parse_native",
|
|
27
|
+
"parse_sino",
|
|
28
|
+
"reference_time",
|
|
29
|
+
]
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""기준 시각 — 상대 표현('내일', '3시', '1시간 후')을 계산하는 기준
|
|
2
|
+
|
|
3
|
+
우선순위:
|
|
4
|
+
1. 함수에 넘긴 now
|
|
5
|
+
2. reference_time()으로 정한 요청 단위 기준 시각
|
|
6
|
+
3. 서버 시각 (시간대를 지정했으면 그 시간대로)
|
|
7
|
+
|
|
8
|
+
웹 서버에서는 요청마다 한 번 정해 두면, 그 요청 안의 모든 호출이 now 없이 같은 기준을 씁니다.
|
|
9
|
+
|
|
10
|
+
@app.middleware("http")
|
|
11
|
+
async def set_request_time(request, call_next):
|
|
12
|
+
with reference_time(datetime.now(KST)):
|
|
13
|
+
return await call_next(request)
|
|
14
|
+
|
|
15
|
+
ContextVar 기반이라 스레드·비동기 요청끼리 섞이지 않습니다.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from collections.abc import Iterator
|
|
21
|
+
from contextlib import contextmanager
|
|
22
|
+
from contextvars import ContextVar
|
|
23
|
+
from datetime import date, datetime, tzinfo
|
|
24
|
+
|
|
25
|
+
_REFERENCE: ContextVar[datetime | None] = ContextVar("korean_datetime_reference_time", default=None)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _as_datetime(value: datetime | date) -> datetime:
|
|
29
|
+
if isinstance(value, datetime):
|
|
30
|
+
return value
|
|
31
|
+
if isinstance(value, date):
|
|
32
|
+
return datetime(value.year, value.month, value.day)
|
|
33
|
+
raise TypeError(f"기준 시각은 datetime 또는 date여야 합니다: {type(value).__name__}")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@contextmanager
|
|
37
|
+
def reference_time(value: datetime | date) -> Iterator[datetime]:
|
|
38
|
+
"""블록 안에서 now를 생략한 호출이 쓸 기준 시각을 정합니다. 블록을 나오면 이전 값으로 돌아갑니다."""
|
|
39
|
+
moment = _as_datetime(value)
|
|
40
|
+
token = _REFERENCE.set(moment)
|
|
41
|
+
try:
|
|
42
|
+
yield moment
|
|
43
|
+
finally:
|
|
44
|
+
_REFERENCE.reset(token)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def current_reference(tz: tzinfo | None = None) -> datetime:
|
|
48
|
+
"""reference_time()으로 정한 기준 시각. 없으면 서버 시각 (tz가 있으면 그 시간대)."""
|
|
49
|
+
fixed = _REFERENCE.get()
|
|
50
|
+
return fixed if fixed is not None else datetime.now(tz)
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
"""정답셋(gold) 기반 정량 평가 (값 하나를 예측하는 파서 공용)
|
|
2
|
+
|
|
3
|
+
정답셋은 JSONL입니다. 첫 줄에 `{"_meta": {"now": "..."}}`로 기본 기준 시각을 둘 수 있고,
|
|
4
|
+
각 줄은 `{"category": str, "text": str, "expected": {...} | null, "now"?: str}` 형식입니다.
|
|
5
|
+
expected가 null이면 "아무것도 인식하지 않아야 함"(음성 케이스)입니다.
|
|
6
|
+
|
|
7
|
+
지표:
|
|
8
|
+
- 검출: TP(정답 있음·예측 있음), FN(정답 있음·예측 없음), FP(정답 없음·예측 있음), TN
|
|
9
|
+
- precision = TP/(TP+FP), recall = TP/(TP+FN), F1
|
|
10
|
+
- value_accuracy = 값까지 맞은 TP / TP (검출한 것 중 정규화가 맞은 비율)
|
|
11
|
+
- accuracy = (값까지 맞은 TP + TN) / 전체
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
from collections.abc import Callable, Iterable, Mapping, Sequence
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from datetime import datetime
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
Expected = Mapping[str, Any]
|
|
24
|
+
Prediction = Any # 예측 결과 (예: TemporalExpression). None이면 인식하지 않음
|
|
25
|
+
Predict = Callable[[str, datetime], Prediction | None]
|
|
26
|
+
Match = Callable[[Prediction, Expected], bool]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True, slots=True)
|
|
30
|
+
class GoldCase:
|
|
31
|
+
text: str
|
|
32
|
+
expected: Expected | None
|
|
33
|
+
category: str
|
|
34
|
+
now: datetime
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True, slots=True)
|
|
38
|
+
class Metrics:
|
|
39
|
+
tp: int = 0
|
|
40
|
+
fp: int = 0
|
|
41
|
+
fn: int = 0
|
|
42
|
+
tn: int = 0
|
|
43
|
+
correct: int = 0
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def total(self) -> int:
|
|
47
|
+
return self.tp + self.fp + self.fn + self.tn
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def precision(self) -> float:
|
|
51
|
+
return self.tp / (self.tp + self.fp) if self.tp + self.fp else 1.0
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def recall(self) -> float:
|
|
55
|
+
return self.tp / (self.tp + self.fn) if self.tp + self.fn else 1.0
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def f1(self) -> float:
|
|
59
|
+
p, r = self.precision, self.recall
|
|
60
|
+
return 2 * p * r / (p + r) if p + r else 0.0
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def value_accuracy(self) -> float:
|
|
64
|
+
"""검출한 것(TP) 중 값까지 맞은 비율 — 정규화 품질"""
|
|
65
|
+
return (self.correct - self.tn) / self.tp if self.tp else 1.0
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def accuracy(self) -> float:
|
|
69
|
+
return self.correct / self.total if self.total else 1.0
|
|
70
|
+
|
|
71
|
+
def add(self, result: CaseResult) -> Metrics:
|
|
72
|
+
expected, predicted = result.case.expected is not None, result.predicted is not None
|
|
73
|
+
return Metrics(
|
|
74
|
+
tp=self.tp + (expected and predicted),
|
|
75
|
+
fp=self.fp + (not expected and predicted),
|
|
76
|
+
fn=self.fn + (expected and not predicted),
|
|
77
|
+
tn=self.tn + (not expected and not predicted),
|
|
78
|
+
correct=self.correct + result.correct,
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True, slots=True)
|
|
83
|
+
class CaseResult:
|
|
84
|
+
case: GoldCase
|
|
85
|
+
predicted: Prediction | None
|
|
86
|
+
correct: bool
|
|
87
|
+
detail: str = ""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass(frozen=True)
|
|
91
|
+
class EvaluationReport:
|
|
92
|
+
results: tuple[CaseResult, ...]
|
|
93
|
+
overall: Metrics
|
|
94
|
+
by_category: Mapping[str, Metrics] = field(default_factory=dict)
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def failures(self) -> list[CaseResult]:
|
|
98
|
+
return [r for r in self.results if not r.correct]
|
|
99
|
+
|
|
100
|
+
def format(self, max_failures: int = 20) -> str:
|
|
101
|
+
columns = ("precision", 11), ("recall", 8), ("f1", 7), ("value_acc", 11), ("accuracy", 10)
|
|
102
|
+
header = f"{'category':<18}{'n':>7}" + "".join(f"{name:>{width}}" for name, width in columns)
|
|
103
|
+
rows = [header, "-" * len(header)]
|
|
104
|
+
items = [*sorted(self.by_category.items()), ("TOTAL", self.overall)]
|
|
105
|
+
for name, m in items:
|
|
106
|
+
rows.append(
|
|
107
|
+
f"{name:<18}{m.total:>7}{m.precision:>11.3f}{m.recall:>8.3f}{m.f1:>7.3f}"
|
|
108
|
+
f"{m.value_accuracy:>11.3f}{m.accuracy:>10.3f}"
|
|
109
|
+
)
|
|
110
|
+
failures = self.failures
|
|
111
|
+
rows.append(f"실패 {len(failures)}건")
|
|
112
|
+
rows.extend(
|
|
113
|
+
f" [{r.case.category}] {r.case.text!r} @ {r.case.now:%Y-%m-%d %H:%M}: {r.detail}"
|
|
114
|
+
for r in failures[:max_failures]
|
|
115
|
+
)
|
|
116
|
+
return "\n".join(rows)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def evaluate(cases: Iterable[GoldCase], predict: Predict, match: Match) -> EvaluationReport:
|
|
120
|
+
"""cases마다 predict(text, now)를 실행하고 match(예측, 정답)로 값을 비교합니다."""
|
|
121
|
+
results = tuple(_run_case(case, predict, match) for case in cases)
|
|
122
|
+
overall = Metrics()
|
|
123
|
+
by_category: dict[str, Metrics] = {}
|
|
124
|
+
for result in results:
|
|
125
|
+
overall = overall.add(result)
|
|
126
|
+
by_category = {
|
|
127
|
+
**by_category,
|
|
128
|
+
result.case.category: by_category.get(result.case.category, Metrics()).add(result),
|
|
129
|
+
}
|
|
130
|
+
return EvaluationReport(results=results, overall=overall, by_category=by_category)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _run_case(case: GoldCase, predict: Predict, match: Match) -> CaseResult:
|
|
134
|
+
try:
|
|
135
|
+
predicted = predict(case.text, case.now)
|
|
136
|
+
except Exception as error: # 한 케이스의 예외가 전체 평가를 멈추지 않도록 실패로 기록
|
|
137
|
+
return CaseResult(case, None, False, f"예외 {type(error).__name__}: {error}")
|
|
138
|
+
if case.expected is None:
|
|
139
|
+
ok = predicted is None
|
|
140
|
+
return CaseResult(
|
|
141
|
+
case, predicted, ok, "" if ok else f"인식하지 않아야 함, 예측={_describe(predicted)}"
|
|
142
|
+
)
|
|
143
|
+
if predicted is None:
|
|
144
|
+
return CaseResult(case, None, False, f"인식 실패, 정답={dict(case.expected)}")
|
|
145
|
+
ok = match(predicted, case.expected)
|
|
146
|
+
return CaseResult(
|
|
147
|
+
case, predicted, ok, "" if ok else f"예측={_describe(predicted)} 정답={dict(case.expected)}"
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _describe(prediction: Prediction | None) -> str:
|
|
152
|
+
if prediction is None:
|
|
153
|
+
return "없음"
|
|
154
|
+
to_dict = getattr(prediction, "to_dict", None)
|
|
155
|
+
return str(to_dict()) if callable(to_dict) else f"{prediction.text!r}={prediction.value!r}"
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def load_gold(path: str | Path) -> list[GoldCase]:
|
|
159
|
+
"""JSONL 정답셋을 읽습니다. 형식 오류는 줄 번호와 함께 ValueError로 알립니다."""
|
|
160
|
+
default_now: datetime | None = None
|
|
161
|
+
cases: list[GoldCase] = []
|
|
162
|
+
lines = Path(path).read_text(encoding="utf-8").splitlines()
|
|
163
|
+
for number, line in enumerate(lines, start=1):
|
|
164
|
+
if not line.strip():
|
|
165
|
+
continue
|
|
166
|
+
try:
|
|
167
|
+
row = json.loads(line)
|
|
168
|
+
if "_meta" in row:
|
|
169
|
+
default_now = datetime.fromisoformat(row["_meta"]["now"])
|
|
170
|
+
continue
|
|
171
|
+
cases.append(_to_case(row, default_now))
|
|
172
|
+
except (KeyError, TypeError, ValueError) as error:
|
|
173
|
+
raise ValueError(f"{path}:{number} 정답셋 형식 오류: {error}") from error
|
|
174
|
+
return cases
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _to_case(row: Mapping[str, Any], default_now: datetime | None) -> GoldCase:
|
|
178
|
+
now = datetime.fromisoformat(row["now"]) if "now" in row else default_now
|
|
179
|
+
if now is None:
|
|
180
|
+
raise ValueError("now가 없습니다 (_meta.now 또는 케이스의 now 필요)")
|
|
181
|
+
expected = row["expected"]
|
|
182
|
+
if expected is not None and not isinstance(expected, Mapping):
|
|
183
|
+
raise TypeError("expected는 객체 또는 null이어야 합니다")
|
|
184
|
+
return GoldCase(
|
|
185
|
+
text=str(row["text"]), expected=expected, category=str(row.get("category", "default")), now=now
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
__all__: Sequence[str] = ["CaseResult", "EvaluationReport", "GoldCase", "Metrics", "evaluate", "load_gold"]
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""한글 수사 파서 (한자어 수사 '이십삼', 고유어 수사 '스물세')
|
|
2
|
+
|
|
3
|
+
정규식 조각(SINO, NATIVE)은 다른 규칙에 끼워 넣어 쓰고, 매칭된 문자열은 parse_* 함수로 정수로 바꿉니다.
|
|
4
|
+
목록을 나열하지 않고 구조로 파싱하므로 '열한시 십오분', '삼십일일' 같은 조합도 정확히 처리합니다.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
|
|
11
|
+
_SINO_DIGITS = {"일": 1, "이": 2, "삼": 3, "사": 4, "오": 5, "육": 6, "칠": 7, "팔": 8, "구": 9}
|
|
12
|
+
_SINO_BODY = "(?:[이삼사오육칠팔구]?백)?(?:[이삼사오육칠팔구]?십)?[일이삼사오육칠팔구]?"
|
|
13
|
+
# 1~999. lookahead로 빈 문자열 매칭을 막는다.
|
|
14
|
+
SINO = f"(?=[일이삼사오육칠팔구십백]){_SINO_BODY}"
|
|
15
|
+
|
|
16
|
+
_NATIVE_UNITS = {
|
|
17
|
+
"하나": 1,
|
|
18
|
+
"한": 1,
|
|
19
|
+
"둘": 2,
|
|
20
|
+
"두": 2,
|
|
21
|
+
"셋": 3,
|
|
22
|
+
"세": 3,
|
|
23
|
+
"석": 3,
|
|
24
|
+
"넷": 4,
|
|
25
|
+
"네": 4,
|
|
26
|
+
"넉": 4,
|
|
27
|
+
"다섯": 5,
|
|
28
|
+
"여섯": 6,
|
|
29
|
+
"일곱": 7,
|
|
30
|
+
"여덟": 8,
|
|
31
|
+
"아홉": 9,
|
|
32
|
+
}
|
|
33
|
+
_NATIVE_TENS = {
|
|
34
|
+
"열": 10,
|
|
35
|
+
"스물": 20,
|
|
36
|
+
"서른": 30,
|
|
37
|
+
"마흔": 40,
|
|
38
|
+
"쉰": 50,
|
|
39
|
+
"예순": 60,
|
|
40
|
+
"일흔": 70,
|
|
41
|
+
"여든": 80,
|
|
42
|
+
"아흔": 90,
|
|
43
|
+
}
|
|
44
|
+
_TENS_ALT = "|".join(sorted(_NATIVE_TENS, key=len, reverse=True))
|
|
45
|
+
_UNITS_ALT = "|".join(sorted(_NATIVE_UNITS, key=len, reverse=True))
|
|
46
|
+
# 1~99. '스무'는 단위명사 앞에서만 쓰는 20.
|
|
47
|
+
NATIVE = f"(?:(?:{_TENS_ALT})(?:\\s?(?:{_UNITS_ALT}))?|스무|(?:{_UNITS_ALT}))"
|
|
48
|
+
|
|
49
|
+
_SINO_FULL = re.compile(SINO)
|
|
50
|
+
_NATIVE_FULL = re.compile(
|
|
51
|
+
f"(?:(?P<tens>{_TENS_ALT})\\s?(?P<unit>{_UNITS_ALT})?|(?P<twenty>스무)|(?P<only>{_UNITS_ALT}))"
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def parse_sino(text: str) -> int | None:
|
|
56
|
+
"""한자어 수사를 정수로 바꿉니다. 형식이 틀리면 None ('십십', '일일')."""
|
|
57
|
+
if not text or _SINO_FULL.fullmatch(text) is None:
|
|
58
|
+
return None
|
|
59
|
+
total, current = 0, 0
|
|
60
|
+
for char in text:
|
|
61
|
+
if char == "백":
|
|
62
|
+
total, current = total + (current or 1) * 100, 0
|
|
63
|
+
elif char == "십":
|
|
64
|
+
total, current = total + (current or 1) * 10, 0
|
|
65
|
+
else:
|
|
66
|
+
current = _SINO_DIGITS[char]
|
|
67
|
+
return total + current
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def parse_native(text: str) -> int | None:
|
|
71
|
+
"""고유어 수사를 정수로 바꿉니다. 형식이 틀리면 None ('하나둘')."""
|
|
72
|
+
match = _NATIVE_FULL.fullmatch(text.strip()) if text else None
|
|
73
|
+
if match is None:
|
|
74
|
+
return None
|
|
75
|
+
if match["twenty"]:
|
|
76
|
+
return 20
|
|
77
|
+
if match["only"]:
|
|
78
|
+
return _NATIVE_UNITS[match["only"]]
|
|
79
|
+
return _NATIVE_TENS[match["tens"]] + (_NATIVE_UNITS[match["unit"]] if match["unit"] else 0)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def parse_korean_number(text: str) -> int | None:
|
|
83
|
+
"""아라비아 숫자, 한자어 수사, 고유어 수사 중 무엇이든 정수로 바꿉니다."""
|
|
84
|
+
stripped = text.strip()
|
|
85
|
+
if stripped.isdecimal():
|
|
86
|
+
return int(stripped)
|
|
87
|
+
sino = parse_sino(stripped)
|
|
88
|
+
return sino if sino is not None else parse_native(stripped)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def syllable_count(text: str) -> int:
|
|
92
|
+
return sum(1 for char in text if not char.isspace())
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""정규식 규칙 기반 토큰 스캐너 (한국어 단어 경계·조사 인식)
|
|
2
|
+
|
|
3
|
+
날짜·시간 규칙이 쓰는 저수준 엔진입니다. 규칙(TokenRule) 목록을 받아 텍스트를
|
|
4
|
+
왼쪽부터 훑으며, 각 위치에서 **가장 긴 매칭**(동률이면 먼저 등록된 규칙)을 토큰으로 채택합니다.
|
|
5
|
+
|
|
6
|
+
경계 규칙:
|
|
7
|
+
- 왼쪽: 한글로 시작하는 매칭은 앞 글자가 한글이면 거부합니다(단, 직전 토큰이 끝난 자리는 허용).
|
|
8
|
+
"그내일"의 "내일"은 거부, "오늘저녁"의 "저녁"은 허용됩니다.
|
|
9
|
+
- 오른쪽(right_boundary=True인 규칙만): 뒤에 한글이 이어지면 조사이거나 다른 토큰의 시작이어야 합니다.
|
|
10
|
+
"오일에"는 허용, "오일교환"은 거부됩니다.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from collections.abc import Callable, Iterable, Sequence
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
PARTICLES: tuple[str, ...] = (
|
|
21
|
+
"에서",
|
|
22
|
+
"에게",
|
|
23
|
+
"에는",
|
|
24
|
+
"에도",
|
|
25
|
+
"에",
|
|
26
|
+
"엔",
|
|
27
|
+
"의",
|
|
28
|
+
"쯤",
|
|
29
|
+
"경",
|
|
30
|
+
"께",
|
|
31
|
+
"즈음",
|
|
32
|
+
"부터",
|
|
33
|
+
"까지",
|
|
34
|
+
"이나",
|
|
35
|
+
"나",
|
|
36
|
+
"은",
|
|
37
|
+
"는",
|
|
38
|
+
"이",
|
|
39
|
+
"가",
|
|
40
|
+
"도",
|
|
41
|
+
"로",
|
|
42
|
+
"으로",
|
|
43
|
+
"만",
|
|
44
|
+
"요",
|
|
45
|
+
"이요",
|
|
46
|
+
"야",
|
|
47
|
+
"이야",
|
|
48
|
+
"랑",
|
|
49
|
+
"이랑",
|
|
50
|
+
"하고",
|
|
51
|
+
# 비교·한정·서술 ('어제만큼', '내일부턴', '오늘부터다')
|
|
52
|
+
"만큼",
|
|
53
|
+
"부턴",
|
|
54
|
+
"까진",
|
|
55
|
+
"보단",
|
|
56
|
+
"처럼",
|
|
57
|
+
"이다",
|
|
58
|
+
"다",
|
|
59
|
+
# 비교 기준으로 붙는 말 ('전년대비', '전년동기', '전월비')
|
|
60
|
+
"대비",
|
|
61
|
+
"동기",
|
|
62
|
+
"동월",
|
|
63
|
+
"동기간",
|
|
64
|
+
"비",
|
|
65
|
+
"과",
|
|
66
|
+
"와",
|
|
67
|
+
"인",
|
|
68
|
+
"이서",
|
|
69
|
+
"을",
|
|
70
|
+
"를",
|
|
71
|
+
"처럼",
|
|
72
|
+
"보다",
|
|
73
|
+
"마다",
|
|
74
|
+
"씩",
|
|
75
|
+
"이면",
|
|
76
|
+
"면",
|
|
77
|
+
"이라",
|
|
78
|
+
"라",
|
|
79
|
+
"이에요",
|
|
80
|
+
"예요",
|
|
81
|
+
"입니다",
|
|
82
|
+
)
|
|
83
|
+
_PARTICLE_RUN = re.compile("(?:" + "|".join(sorted(PARTICLES, key=len, reverse=True)) + ")*(?![가-힣])")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def is_hangul(char: str) -> bool:
|
|
87
|
+
return "가" <= char <= "힣"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def ends_word(text: str, end: int) -> bool:
|
|
91
|
+
"""end 위치가 단어의 끝인지: 문장 끝, 한글이 아님, 또는 조사만 이어짐 ('서울에서' ○, '서울시' ×)"""
|
|
92
|
+
return end >= len(text) or not is_hangul(text[end]) or _PARTICLE_RUN.match(text, end) is not None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass(frozen=True, slots=True)
|
|
96
|
+
class Token:
|
|
97
|
+
"""스캐너가 만든 토큰. value는 규칙의 builder가 결정합니다."""
|
|
98
|
+
|
|
99
|
+
kind: str
|
|
100
|
+
value: Any
|
|
101
|
+
start: int
|
|
102
|
+
end: int
|
|
103
|
+
text: str
|
|
104
|
+
weak: bool = False
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class Split(tuple[Token, ...]):
|
|
108
|
+
"""builder가 하나의 매칭을 여러 토큰으로 나눠 내보낼 때 반환합니다 (예: '다음주말' → 주 + 주말)."""
|
|
109
|
+
|
|
110
|
+
__slots__ = ()
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
Builder = Callable[["re.Match[str]"], Any]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass(frozen=True, slots=True)
|
|
117
|
+
class TokenRule:
|
|
118
|
+
"""
|
|
119
|
+
Attributes:
|
|
120
|
+
kind: 토큰 종류
|
|
121
|
+
pattern: 현재 위치에서 match()로 시도할 정규식
|
|
122
|
+
build: 매칭 → 토큰 값. None을 반환하면 매칭을 거부하고, Split을 반환하면 그 토큰들을 그대로 사용
|
|
123
|
+
right_boundary: True면 오른쪽 경계 규칙을 적용
|
|
124
|
+
weak: 문맥이 있어야 의미가 있는 토큰 표시 (해석 단계에서 사용)
|
|
125
|
+
attachable: True면 앞말에 붙어 있어도 매칭 (왼쪽 경계 예외). 패턴 자체가 충분히 구별될 때만
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
kind: str
|
|
129
|
+
pattern: re.Pattern[str]
|
|
130
|
+
build: Builder
|
|
131
|
+
right_boundary: bool = False
|
|
132
|
+
weak: bool = False
|
|
133
|
+
attachable: bool = False
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class Scanner:
|
|
137
|
+
"""TokenRule 목록으로 텍스트를 토큰화합니다. 규칙이 불변이므로 인스턴스를 재사용해도 안전합니다."""
|
|
138
|
+
|
|
139
|
+
def __init__(self, rules: Iterable[TokenRule]) -> None:
|
|
140
|
+
self._rules: tuple[TokenRule, ...] = tuple(rules)
|
|
141
|
+
if not all(isinstance(rule, TokenRule) for rule in self._rules):
|
|
142
|
+
raise TypeError("rules에는 TokenRule만 넣을 수 있습니다")
|
|
143
|
+
self._has_attachable = any(rule.attachable for rule in self._rules)
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def rules(self) -> tuple[TokenRule, ...]:
|
|
147
|
+
return self._rules
|
|
148
|
+
|
|
149
|
+
def scan(self, text: str) -> list[Token]:
|
|
150
|
+
tokens: list[Token] = []
|
|
151
|
+
pos, last_end = 0, 0
|
|
152
|
+
while pos < len(text):
|
|
153
|
+
attached = not self._left_ok(text, pos, last_end)
|
|
154
|
+
if attached and not self._has_attachable:
|
|
155
|
+
pos += 1
|
|
156
|
+
continue
|
|
157
|
+
best = self._best_match(text, pos, attached)
|
|
158
|
+
if best is None:
|
|
159
|
+
pos += 1
|
|
160
|
+
continue
|
|
161
|
+
tokens.extend(best)
|
|
162
|
+
pos = last_end = best[-1].end
|
|
163
|
+
return tokens
|
|
164
|
+
|
|
165
|
+
def _best_match(self, text: str, pos: int, attached: bool = False) -> Sequence[Token] | None:
|
|
166
|
+
"""attached: 앞말에 붙은 위치라 attachable 규칙만 시도"""
|
|
167
|
+
best: Sequence[Token] | None = None
|
|
168
|
+
best_end = pos
|
|
169
|
+
for rule in self._rules:
|
|
170
|
+
if attached and not rule.attachable:
|
|
171
|
+
continue
|
|
172
|
+
match = rule.pattern.match(text, pos)
|
|
173
|
+
if match is None or match.end() <= best_end:
|
|
174
|
+
continue
|
|
175
|
+
if rule.right_boundary and not self._right_ok(text, match.end()):
|
|
176
|
+
continue
|
|
177
|
+
value = rule.build(match)
|
|
178
|
+
if value is None:
|
|
179
|
+
continue
|
|
180
|
+
best = value if isinstance(value, Split) else (self._token(rule, match, value),)
|
|
181
|
+
best_end = match.end()
|
|
182
|
+
return best
|
|
183
|
+
|
|
184
|
+
@staticmethod
|
|
185
|
+
def _token(rule: TokenRule, match: re.Match[str], value: Any) -> Token:
|
|
186
|
+
return Token(rule.kind, value, match.start(), match.end(), match.group(), rule.weak)
|
|
187
|
+
|
|
188
|
+
@staticmethod
|
|
189
|
+
def _left_ok(text: str, pos: int, last_end: int) -> bool:
|
|
190
|
+
if pos == 0 or pos == last_end or not is_hangul(text[pos]):
|
|
191
|
+
return True
|
|
192
|
+
return not is_hangul(text[pos - 1])
|
|
193
|
+
|
|
194
|
+
def _right_ok(self, text: str, end: int) -> bool:
|
|
195
|
+
return ends_word(text, end) or any(rule.pattern.match(text, end) for rule in self._rules)
|