lipla-jp 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lipla/__init__.py +5 -0
- lipla/configs/characters.yml +208 -0
- lipla/core/fixed_field_recognizer.py +286 -0
- lipla/core/image_processing.py +164 -0
- lipla/core/license_plate_recognizer.py +385 -0
- lipla/core/ocr_result_parser.py +373 -0
- lipla/core/plate_normalizer.py +110 -0
- lipla/core/pose_postprocessor.py +123 -0
- lipla/core/recognition_result.py +313 -0
- lipla/inferencers/ec_pose.py +326 -0
- lipla/inferencers/execution_provider.py +91 -0
- lipla/inferencers/model_loader.py +52 -0
- lipla/inferencers/ppocr.py +679 -0
- lipla_jp-0.2.0.dist-info/METADATA +90 -0
- lipla_jp-0.2.0.dist-info/RECORD +18 -0
- lipla_jp-0.2.0.dist-info/WHEEL +5 -0
- lipla_jp-0.2.0.dist-info/licenses/LICENSE +21 -0
- lipla_jp-0.2.0.dist-info/top_level.txt +1 -0
lipla/__init__.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
areas:
|
|
2
|
+
- 札幌
|
|
3
|
+
- 函館
|
|
4
|
+
- 旭川
|
|
5
|
+
- 室蘭
|
|
6
|
+
- 苫小牧
|
|
7
|
+
- 釧路
|
|
8
|
+
- 知床
|
|
9
|
+
- 帯広
|
|
10
|
+
- 十勝
|
|
11
|
+
- 北見
|
|
12
|
+
- 青森
|
|
13
|
+
- 弘前
|
|
14
|
+
- 八戸
|
|
15
|
+
- 盛岡
|
|
16
|
+
- 岩手
|
|
17
|
+
- 平泉
|
|
18
|
+
- 仙台
|
|
19
|
+
- 宮城
|
|
20
|
+
- 秋田
|
|
21
|
+
- 山形
|
|
22
|
+
- 庄内
|
|
23
|
+
- 福島
|
|
24
|
+
- 会津
|
|
25
|
+
- 郡山
|
|
26
|
+
- 白河
|
|
27
|
+
- いわき
|
|
28
|
+
- 水戸
|
|
29
|
+
- 土浦
|
|
30
|
+
- つくば
|
|
31
|
+
- 宇都宮
|
|
32
|
+
- 日光
|
|
33
|
+
- 那須
|
|
34
|
+
- とちぎ
|
|
35
|
+
- 前橋
|
|
36
|
+
- 高崎
|
|
37
|
+
- 群馬
|
|
38
|
+
- 大宮
|
|
39
|
+
- 川口
|
|
40
|
+
- 川越
|
|
41
|
+
- 所沢
|
|
42
|
+
- 熊谷
|
|
43
|
+
- 春日部
|
|
44
|
+
- 越谷
|
|
45
|
+
- 千葉
|
|
46
|
+
- 成田
|
|
47
|
+
- 市川
|
|
48
|
+
- 船橋
|
|
49
|
+
- 習志野
|
|
50
|
+
- 袖ケ浦
|
|
51
|
+
- 市原
|
|
52
|
+
- 松戸
|
|
53
|
+
- 野田
|
|
54
|
+
- 柏
|
|
55
|
+
- 品川
|
|
56
|
+
- 世田谷
|
|
57
|
+
- 練馬
|
|
58
|
+
- 杉並
|
|
59
|
+
- 板橋
|
|
60
|
+
- 足立
|
|
61
|
+
- 江東
|
|
62
|
+
- 葛飾
|
|
63
|
+
- 江戸川
|
|
64
|
+
- 八王子
|
|
65
|
+
- 多摩
|
|
66
|
+
- 横浜
|
|
67
|
+
- 川崎
|
|
68
|
+
- 湘南
|
|
69
|
+
- 相模
|
|
70
|
+
- 山梨
|
|
71
|
+
- 富士山
|
|
72
|
+
- 新潟
|
|
73
|
+
- 長岡
|
|
74
|
+
- 上越
|
|
75
|
+
- 富山
|
|
76
|
+
- 金沢
|
|
77
|
+
- 石川
|
|
78
|
+
- 福井
|
|
79
|
+
- 長野
|
|
80
|
+
- 松本
|
|
81
|
+
- 諏訪
|
|
82
|
+
- 南信州
|
|
83
|
+
- 安曇野
|
|
84
|
+
- 岐阜
|
|
85
|
+
- 飛騨
|
|
86
|
+
- 静岡
|
|
87
|
+
- 浜松
|
|
88
|
+
- 沼津
|
|
89
|
+
- 伊豆
|
|
90
|
+
- 名古屋
|
|
91
|
+
- 豊橋
|
|
92
|
+
- 岡崎
|
|
93
|
+
- 三河
|
|
94
|
+
- 豊田
|
|
95
|
+
- 一宮
|
|
96
|
+
- 尾張小牧
|
|
97
|
+
- 春日井
|
|
98
|
+
- 三重
|
|
99
|
+
- 四日市
|
|
100
|
+
- 伊勢志摩
|
|
101
|
+
- 鈴鹿
|
|
102
|
+
- 滋賀
|
|
103
|
+
- 京都
|
|
104
|
+
- 大阪
|
|
105
|
+
- なにわ
|
|
106
|
+
- 堺
|
|
107
|
+
- 和泉
|
|
108
|
+
- 神戸
|
|
109
|
+
- 姫路
|
|
110
|
+
- 奈良
|
|
111
|
+
- 飛鳥
|
|
112
|
+
- 和歌山
|
|
113
|
+
- 鳥取
|
|
114
|
+
- 島根
|
|
115
|
+
- 出雲
|
|
116
|
+
- 岡山
|
|
117
|
+
- 倉敷
|
|
118
|
+
- 広島
|
|
119
|
+
- 福山
|
|
120
|
+
- 下関
|
|
121
|
+
- 山口
|
|
122
|
+
- 徳島
|
|
123
|
+
- 高松
|
|
124
|
+
- 香川
|
|
125
|
+
- 愛媛
|
|
126
|
+
- 高知
|
|
127
|
+
- 福岡
|
|
128
|
+
- 北九州
|
|
129
|
+
- 久留米
|
|
130
|
+
- 筑豊
|
|
131
|
+
- 佐賀
|
|
132
|
+
- 長崎
|
|
133
|
+
- 佐世保
|
|
134
|
+
- 熊本
|
|
135
|
+
- 大分
|
|
136
|
+
- 宮崎
|
|
137
|
+
- 鹿児島
|
|
138
|
+
- 奄美
|
|
139
|
+
- 沖縄
|
|
140
|
+
hiragana:
|
|
141
|
+
- あ
|
|
142
|
+
- い
|
|
143
|
+
- う
|
|
144
|
+
- え
|
|
145
|
+
- か
|
|
146
|
+
- き
|
|
147
|
+
- く
|
|
148
|
+
- け
|
|
149
|
+
- こ
|
|
150
|
+
- さ
|
|
151
|
+
- す
|
|
152
|
+
- せ
|
|
153
|
+
- そ
|
|
154
|
+
- た
|
|
155
|
+
- ち
|
|
156
|
+
- つ
|
|
157
|
+
- て
|
|
158
|
+
- と
|
|
159
|
+
- な
|
|
160
|
+
- に
|
|
161
|
+
- ぬ
|
|
162
|
+
- ね
|
|
163
|
+
- の
|
|
164
|
+
- は
|
|
165
|
+
- ひ
|
|
166
|
+
- ふ
|
|
167
|
+
- ほ
|
|
168
|
+
- ま
|
|
169
|
+
- み
|
|
170
|
+
- む
|
|
171
|
+
- め
|
|
172
|
+
- も
|
|
173
|
+
- や
|
|
174
|
+
- ゆ
|
|
175
|
+
- よ
|
|
176
|
+
- ら
|
|
177
|
+
- り
|
|
178
|
+
- る
|
|
179
|
+
- れ
|
|
180
|
+
- ろ
|
|
181
|
+
- わ
|
|
182
|
+
- を
|
|
183
|
+
alphabet:
|
|
184
|
+
- A
|
|
185
|
+
- C
|
|
186
|
+
- F
|
|
187
|
+
- H
|
|
188
|
+
- K
|
|
189
|
+
- L
|
|
190
|
+
- M
|
|
191
|
+
- P
|
|
192
|
+
- X
|
|
193
|
+
- Y
|
|
194
|
+
numbers:
|
|
195
|
+
- "0"
|
|
196
|
+
- "1"
|
|
197
|
+
- "2"
|
|
198
|
+
- "3"
|
|
199
|
+
- "4"
|
|
200
|
+
- "5"
|
|
201
|
+
- "6"
|
|
202
|
+
- "7"
|
|
203
|
+
- "8"
|
|
204
|
+
- "9"
|
|
205
|
+
symbols:
|
|
206
|
+
- "-"
|
|
207
|
+
- "·"
|
|
208
|
+
- "・"
|
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""ナンバープレートの固定領域に特化した文字認識。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from typing import Final, Protocol
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
from lipla.inferencers.ppocr import preprocess_rec
|
|
11
|
+
|
|
12
|
+
from .image_processing import validate_bgr_image
|
|
13
|
+
from .ocr_result_parser import (
|
|
14
|
+
CandidateMap,
|
|
15
|
+
OCRResultParser,
|
|
16
|
+
TextCandidate,
|
|
17
|
+
create_candidate_map,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
FIELD_RECTS: Final = {
|
|
21
|
+
"area": (0.194, 0.00, 0.519, 0.325),
|
|
22
|
+
"class_number": (0.51, 0.00, 0.81, 0.36),
|
|
23
|
+
"number": (0.15, 0.34, 1.00, 1.00),
|
|
24
|
+
}
|
|
25
|
+
KANA_RECTS: Final = (
|
|
26
|
+
((0.009, 0.438, 0.197, 1.00), 1.0),
|
|
27
|
+
((0.016, 0.469, 0.197, 1.00), 0.7),
|
|
28
|
+
((0.000, 0.375, 0.219, 1.00), 0.7),
|
|
29
|
+
)
|
|
30
|
+
FIELD_FILTERS: Final = {
|
|
31
|
+
"class_number": ["numbers", "alphabet"],
|
|
32
|
+
"number": ["numbers", "symbols"],
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class RecognitionSession(Protocol):
|
|
37
|
+
"""文字認識セッションに必要なインターフェース。"""
|
|
38
|
+
|
|
39
|
+
def run(
|
|
40
|
+
self,
|
|
41
|
+
output_names: None,
|
|
42
|
+
input_feed: dict[str, np.ndarray],
|
|
43
|
+
) -> list[np.ndarray]: ...
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class RecognitionDecoder(Protocol):
|
|
47
|
+
"""文字認識デコーダーに必要なインターフェース。"""
|
|
48
|
+
|
|
49
|
+
character: list[str]
|
|
50
|
+
|
|
51
|
+
def decode(
|
|
52
|
+
self,
|
|
53
|
+
predictions: np.ndarray,
|
|
54
|
+
filters: Iterable[str] | None = None,
|
|
55
|
+
) -> tuple[str, float]: ...
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class RecognitionModel(Protocol):
|
|
59
|
+
"""固定領域OCRに必要なモデルインターフェース。"""
|
|
60
|
+
|
|
61
|
+
rec_session: RecognitionSession
|
|
62
|
+
rec_input_name: str
|
|
63
|
+
rec_target_height: int
|
|
64
|
+
decoder: RecognitionDecoder
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def logsumexp(values: np.ndarray) -> float:
|
|
68
|
+
"""オーバーフローを避けながら指数の和の対数を計算する。
|
|
69
|
+
|
|
70
|
+
Args:
|
|
71
|
+
values: 対数空間の値。
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
``log(sum(exp(values)))`` の計算結果。
|
|
75
|
+
"""
|
|
76
|
+
maximum = float(np.max(values))
|
|
77
|
+
if not np.isfinite(maximum):
|
|
78
|
+
return -np.inf
|
|
79
|
+
return maximum + float(np.log(np.exp(values - maximum).sum()))
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def ctc_log_probability(
|
|
83
|
+
log_probabilities: np.ndarray, tokens: tuple[int, ...]
|
|
84
|
+
) -> float:
|
|
85
|
+
"""1つの語彙に対するCTC前向き確率を対数空間で計算する。
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
log_probabilities: 各時刻・各文字の対数確率。Blankは添字0とする。
|
|
89
|
+
tokens: 評価する文字列の文字インデックス。
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
指定した文字列を生成する全CTCパスの対数確率。
|
|
93
|
+
|
|
94
|
+
Raises:
|
|
95
|
+
ValueError: ``tokens`` が空、または時間ステップが不足している場合。
|
|
96
|
+
"""
|
|
97
|
+
if not tokens:
|
|
98
|
+
raise ValueError("tokens must not be empty")
|
|
99
|
+
if len(log_probabilities) == 0:
|
|
100
|
+
raise ValueError("log_probabilities must contain at least one timestep")
|
|
101
|
+
|
|
102
|
+
extended = [0]
|
|
103
|
+
for token in tokens:
|
|
104
|
+
extended.extend((token, 0))
|
|
105
|
+
|
|
106
|
+
previous = np.full(len(extended), -np.inf, dtype=np.float64)
|
|
107
|
+
previous[0] = log_probabilities[0, 0]
|
|
108
|
+
previous[1] = log_probabilities[0, extended[1]]
|
|
109
|
+
for timestep in range(1, len(log_probabilities)):
|
|
110
|
+
current = np.full_like(previous, -np.inf)
|
|
111
|
+
for state, token in enumerate(extended):
|
|
112
|
+
incoming = [previous[state]]
|
|
113
|
+
if state > 0:
|
|
114
|
+
incoming.append(previous[state - 1])
|
|
115
|
+
if state > 1 and token != 0 and token != extended[state - 2]:
|
|
116
|
+
incoming.append(previous[state - 2])
|
|
117
|
+
current[state] = logsumexp(np.asarray(incoming))
|
|
118
|
+
current[state] += log_probabilities[timestep, token]
|
|
119
|
+
previous = current
|
|
120
|
+
return logsumexp(previous[-2:])
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class FixedFieldRecognizer:
|
|
124
|
+
"""プレートの既知レイアウトから4項目を個別認識する。
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
ocr_model: PP-OCRの文字認識セッションとデコーダーを持つオブジェクト。
|
|
128
|
+
parser: 地名とひらがなの語彙を保持するOCR結果パーサー。
|
|
129
|
+
"""
|
|
130
|
+
|
|
131
|
+
def __init__(self, ocr_model: RecognitionModel, parser: OCRResultParser) -> None:
|
|
132
|
+
self.ocr_model = ocr_model
|
|
133
|
+
self.parser = parser
|
|
134
|
+
decoder_characters = self.ocr_model.decoder.character
|
|
135
|
+
self.area_tokens = tuple(
|
|
136
|
+
(
|
|
137
|
+
area,
|
|
138
|
+
tuple(decoder_characters.index(character) for character in area),
|
|
139
|
+
)
|
|
140
|
+
for area in parser.areas
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
def recognize(self, image: np.ndarray) -> CandidateMap:
|
|
144
|
+
"""固定領域を切り出して項目別の文字列候補を生成する。
|
|
145
|
+
|
|
146
|
+
必要な文字認識インターフェースをOCRモデルが持たない場合は、全項目が
|
|
147
|
+
空の候補辞書を返す。これにより、文字検出だけを実装した差し替えモデル
|
|
148
|
+
でも空間OCR解析を利用できる。
|
|
149
|
+
|
|
150
|
+
Args:
|
|
151
|
+
image: 射影・色補正済みのBGRプレート画像。
|
|
152
|
+
|
|
153
|
+
Returns:
|
|
154
|
+
項目別の文字列候補。
|
|
155
|
+
|
|
156
|
+
Raises:
|
|
157
|
+
TypeError: 画像の型またはデータ型が不正な場合。
|
|
158
|
+
ValueError: 画像が空、またはBGR画像ではない場合。
|
|
159
|
+
"""
|
|
160
|
+
validate_bgr_image(image)
|
|
161
|
+
candidates = create_candidate_map()
|
|
162
|
+
required = (
|
|
163
|
+
"rec_session",
|
|
164
|
+
"rec_input_name",
|
|
165
|
+
"rec_target_height",
|
|
166
|
+
"decoder",
|
|
167
|
+
)
|
|
168
|
+
if not all(hasattr(self.ocr_model, name) for name in required):
|
|
169
|
+
return candidates
|
|
170
|
+
|
|
171
|
+
height, width = image.shape[:2]
|
|
172
|
+
for field_name, rect in FIELD_RECTS.items():
|
|
173
|
+
crop = self._crop(image, rect)
|
|
174
|
+
if field_name == "area":
|
|
175
|
+
candidate = self._recognize_area(crop)
|
|
176
|
+
elif field_name == "class_number":
|
|
177
|
+
candidate = self._recognize_class_number(crop)
|
|
178
|
+
else:
|
|
179
|
+
candidate = self._recognize_crop(crop, FIELD_FILTERS[field_name])
|
|
180
|
+
if candidate.text:
|
|
181
|
+
candidates[field_name].append(candidate)
|
|
182
|
+
|
|
183
|
+
kana_candidate = self._recognize_kana(image)
|
|
184
|
+
if kana_candidate.text:
|
|
185
|
+
candidates["kana"].append(kana_candidate)
|
|
186
|
+
|
|
187
|
+
top_row = image[
|
|
188
|
+
: max(1, int(round(height * 0.41))),
|
|
189
|
+
int(round(width * 0.14)) : int(round(width * 0.85)),
|
|
190
|
+
]
|
|
191
|
+
self.parser.split_top_text(self._recognize_crop(top_row, None), candidates)
|
|
192
|
+
return candidates
|
|
193
|
+
|
|
194
|
+
@staticmethod
|
|
195
|
+
def _crop(image: np.ndarray, rect: tuple[float, float, float, float]) -> np.ndarray:
|
|
196
|
+
height, width = image.shape[:2]
|
|
197
|
+
x1, y1, x2, y2 = rect
|
|
198
|
+
left = max(0, min(width - 1, int(round(x1 * width))))
|
|
199
|
+
top = max(0, min(height - 1, int(round(y1 * height))))
|
|
200
|
+
right = max(left + 1, min(width, int(round(x2 * width))))
|
|
201
|
+
bottom = max(top + 1, min(height, int(round(y2 * height))))
|
|
202
|
+
return image[top:bottom, left:right]
|
|
203
|
+
|
|
204
|
+
def _recognize_crop(
|
|
205
|
+
self, crop: np.ndarray, filters: list[str] | None
|
|
206
|
+
) -> TextCandidate:
|
|
207
|
+
if crop.size == 0:
|
|
208
|
+
return TextCandidate("", 0.0)
|
|
209
|
+
rec_input = preprocess_rec(crop, target_height=self.ocr_model.rec_target_height)
|
|
210
|
+
rec_outputs = self.ocr_model.rec_session.run(
|
|
211
|
+
None, {self.ocr_model.rec_input_name: rec_input}
|
|
212
|
+
)
|
|
213
|
+
text, score = self.ocr_model.decoder.decode(rec_outputs[0], filters=filters)
|
|
214
|
+
return TextCandidate(str(text).strip(), float(score))
|
|
215
|
+
|
|
216
|
+
def _recognize_area(self, crop: np.ndarray) -> TextCandidate:
|
|
217
|
+
if crop.size == 0:
|
|
218
|
+
return TextCandidate("", 0.0)
|
|
219
|
+
rec_input = preprocess_rec(crop, target_height=self.ocr_model.rec_target_height)
|
|
220
|
+
rec_outputs = self.ocr_model.rec_session.run(
|
|
221
|
+
None, {self.ocr_model.rec_input_name: rec_input}
|
|
222
|
+
)[0]
|
|
223
|
+
probabilities = np.asarray(rec_outputs, dtype=np.float64)[0]
|
|
224
|
+
log_probabilities = np.log(np.clip(probabilities, 1e-30, 1.0))
|
|
225
|
+
scores = np.asarray(
|
|
226
|
+
[
|
|
227
|
+
ctc_log_probability(log_probabilities, tokens) / len(tokens)
|
|
228
|
+
for _, tokens in self.area_tokens
|
|
229
|
+
],
|
|
230
|
+
dtype=np.float64,
|
|
231
|
+
)
|
|
232
|
+
best_index = int(np.argmax(scores))
|
|
233
|
+
normalizer = logsumexp(scores)
|
|
234
|
+
confidence = float(np.exp(scores[best_index] - normalizer))
|
|
235
|
+
return TextCandidate(self.area_tokens[best_index][0], confidence)
|
|
236
|
+
|
|
237
|
+
def _recognize_class_number(self, crop: np.ndarray) -> TextCandidate:
|
|
238
|
+
if crop.size == 0:
|
|
239
|
+
return TextCandidate("", 0.0)
|
|
240
|
+
rec_input = preprocess_rec(crop, target_height=self.ocr_model.rec_target_height)
|
|
241
|
+
rec_outputs = self.ocr_model.rec_session.run(
|
|
242
|
+
None, {self.ocr_model.rec_input_name: rec_input}
|
|
243
|
+
)[0]
|
|
244
|
+
text, score = self.ocr_model.decoder.decode(
|
|
245
|
+
rec_outputs, filters=FIELD_FILTERS["class_number"]
|
|
246
|
+
)
|
|
247
|
+
return TextCandidate(str(text).strip().upper(), float(score))
|
|
248
|
+
|
|
249
|
+
def _single_character_scores(
|
|
250
|
+
self, crop: np.ndarray, allowed_characters: Iterable[str]
|
|
251
|
+
) -> dict[str, float]:
|
|
252
|
+
if crop.size == 0:
|
|
253
|
+
return {}
|
|
254
|
+
rec_input = preprocess_rec(crop, target_height=self.ocr_model.rec_target_height)
|
|
255
|
+
rec_outputs = self.ocr_model.rec_session.run(
|
|
256
|
+
None, {self.ocr_model.rec_input_name: rec_input}
|
|
257
|
+
)[0]
|
|
258
|
+
probabilities = np.asarray(rec_outputs)[0]
|
|
259
|
+
character_indices = {
|
|
260
|
+
character: self.ocr_model.decoder.character.index(character)
|
|
261
|
+
for character in allowed_characters
|
|
262
|
+
if character in self.ocr_model.decoder.character
|
|
263
|
+
}
|
|
264
|
+
return {
|
|
265
|
+
character: max(0.0, float(np.max(probabilities[:, index])))
|
|
266
|
+
for character, index in character_indices.items()
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
def _recognize_kana(self, image: np.ndarray) -> TextCandidate:
|
|
270
|
+
combined = {character: 0.0 for character in self.parser.kana_characters}
|
|
271
|
+
for rect, weight in KANA_RECTS:
|
|
272
|
+
scores = self._single_character_scores(
|
|
273
|
+
self._crop(image, rect), self.parser.kana_characters
|
|
274
|
+
)
|
|
275
|
+
score_sum = sum(scores.values())
|
|
276
|
+
if score_sum <= 0.0:
|
|
277
|
+
continue
|
|
278
|
+
for character, score in scores.items():
|
|
279
|
+
combined[character] += weight * score / score_sum
|
|
280
|
+
|
|
281
|
+
if not combined:
|
|
282
|
+
return TextCandidate("", 0.0)
|
|
283
|
+
character = max(combined, key=combined.get)
|
|
284
|
+
score_sum = sum(combined.values())
|
|
285
|
+
confidence = combined[character] / score_sum if score_sum > 0.0 else 0.0
|
|
286
|
+
return TextCandidate(character, float(confidence))
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""ナンバープレート認識で使用する画像の検証と前処理。"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import cv2
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
LOCAL_BLACK_VALUE_MAX = 50
|
|
11
|
+
LOCAL_GREEN_HUE_MIN = 30
|
|
12
|
+
LOCAL_GREEN_HUE_MAX = 95
|
|
13
|
+
LOCAL_GREEN_SATURATION_MIN = 25
|
|
14
|
+
LOCAL_GREEN_VALUE_MAX = 130
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def validate_bgr_image(image: np.ndarray) -> None:
|
|
18
|
+
"""BGR画像として利用できる配列か検証する。
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
image: 検証対象の画像。
|
|
22
|
+
|
|
23
|
+
Raises:
|
|
24
|
+
TypeError: ``image`` が ``numpy.ndarray`` ではない場合、または
|
|
25
|
+
データ型が ``uint8`` ではない場合。
|
|
26
|
+
ValueError: ``image`` が空、または形状が ``(H, W, 3)`` ではない場合。
|
|
27
|
+
"""
|
|
28
|
+
if not isinstance(image, np.ndarray):
|
|
29
|
+
raise TypeError("image must be a numpy.ndarray")
|
|
30
|
+
if image.ndim != 3 or image.shape[2] != 3:
|
|
31
|
+
raise ValueError("image must be a BGR image with shape (H, W, 3)")
|
|
32
|
+
if image.shape[0] == 0 or image.shape[1] == 0:
|
|
33
|
+
raise ValueError("image must not be empty")
|
|
34
|
+
if image.dtype != np.uint8:
|
|
35
|
+
raise TypeError("image must have dtype uint8")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def read_bgr_image(path: str | Path) -> np.ndarray:
|
|
39
|
+
"""ファイルパスからBGR画像を読み込む。
|
|
40
|
+
|
|
41
|
+
OpenCVのパス処理に依存せずバイト列を復号するため、日本語を含むパスも
|
|
42
|
+
使用できる。
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
path: 読み込む画像ファイルのパス。
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
``uint8`` のBGR画像。
|
|
49
|
+
|
|
50
|
+
Raises:
|
|
51
|
+
OSError: ファイルを読み込めない場合。
|
|
52
|
+
ValueError: ファイルを画像として復号できない場合。
|
|
53
|
+
"""
|
|
54
|
+
image_path = Path(path)
|
|
55
|
+
encoded = np.frombuffer(image_path.read_bytes(), dtype=np.uint8)
|
|
56
|
+
image = cv2.imdecode(encoded, cv2.IMREAD_COLOR) if encoded.size > 0 else None
|
|
57
|
+
if image is None:
|
|
58
|
+
raise ValueError(f"Could not decode image file: {image_path}")
|
|
59
|
+
return image
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def as_bgr(image: np.ndarray) -> np.ndarray:
|
|
63
|
+
"""正規化後の画像をBGR形式へ揃える。
|
|
64
|
+
|
|
65
|
+
Args:
|
|
66
|
+
image: グレースケール画像、またはBGR画像。
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
BGR形式の画像。入力がBGRの場合は同じ配列を返す。
|
|
70
|
+
|
|
71
|
+
Raises:
|
|
72
|
+
TypeError: 入力が配列ではない場合、またはデータ型が ``uint8`` では
|
|
73
|
+
ない場合。
|
|
74
|
+
ValueError: 入力が空、または対応していない形状の場合。
|
|
75
|
+
"""
|
|
76
|
+
if not isinstance(image, np.ndarray):
|
|
77
|
+
raise TypeError("normalized image must be a numpy.ndarray")
|
|
78
|
+
if image.ndim == 2:
|
|
79
|
+
if image.size == 0:
|
|
80
|
+
raise ValueError("normalized image must not be empty")
|
|
81
|
+
if image.dtype != np.uint8:
|
|
82
|
+
raise TypeError("normalized image must have dtype uint8")
|
|
83
|
+
return cv2.cvtColor(image, cv2.COLOR_GRAY2BGR)
|
|
84
|
+
if image.ndim == 3 and image.shape[2] == 3:
|
|
85
|
+
validate_bgr_image(image)
|
|
86
|
+
return image
|
|
87
|
+
raise ValueError(f"normalized image has an invalid shape: {image.shape}")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def preprocess_local_plate(image: np.ndarray) -> np.ndarray:
|
|
91
|
+
"""図柄入りナンバープレートから文字色以外を白くする。
|
|
92
|
+
|
|
93
|
+
黒色から濃緑色までの画素を残し、OCRを妨げる背景の図柄を除去する。
|
|
94
|
+
|
|
95
|
+
Args:
|
|
96
|
+
image: ``uint8`` のBGR画像。
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
背景の図柄を白色に置換したBGR画像。
|
|
100
|
+
|
|
101
|
+
Raises:
|
|
102
|
+
TypeError: 画像の型またはデータ型が不正な場合。
|
|
103
|
+
ValueError: 画像が空、またはBGR画像ではない場合。
|
|
104
|
+
"""
|
|
105
|
+
validate_bgr_image(image)
|
|
106
|
+
hsv = cv2.cvtColor(image, cv2.COLOR_BGR2HSV)
|
|
107
|
+
hue, saturation, value = cv2.split(hsv)
|
|
108
|
+
black = value <= LOCAL_BLACK_VALUE_MAX
|
|
109
|
+
dark_green = (
|
|
110
|
+
(hue >= LOCAL_GREEN_HUE_MIN)
|
|
111
|
+
& (hue <= LOCAL_GREEN_HUE_MAX)
|
|
112
|
+
& (saturation >= LOCAL_GREEN_SATURATION_MIN)
|
|
113
|
+
& (value <= LOCAL_GREEN_VALUE_MAX)
|
|
114
|
+
)
|
|
115
|
+
filtered = np.full_like(image, 255)
|
|
116
|
+
keep = black | dark_green
|
|
117
|
+
filtered[keep] = image[keep]
|
|
118
|
+
return filtered
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def pad_for_ocr(image: np.ndarray) -> tuple[np.ndarray, int]:
|
|
122
|
+
"""OCRの文字検出用に画像の周囲へ余白を追加する。
|
|
123
|
+
|
|
124
|
+
Args:
|
|
125
|
+
image: 余白を追加するBGR画像。
|
|
126
|
+
|
|
127
|
+
Returns:
|
|
128
|
+
余白を追加した画像と、上下左右に追加したピクセル数。
|
|
129
|
+
|
|
130
|
+
Raises:
|
|
131
|
+
TypeError: 画像の型またはデータ型が不正な場合。
|
|
132
|
+
ValueError: 画像が空、またはBGR画像ではない場合。
|
|
133
|
+
"""
|
|
134
|
+
validate_bgr_image(image)
|
|
135
|
+
padding = max(8, int(round(image.shape[0] * 0.10)))
|
|
136
|
+
padded = cv2.copyMakeBorder(
|
|
137
|
+
image,
|
|
138
|
+
padding,
|
|
139
|
+
padding,
|
|
140
|
+
padding,
|
|
141
|
+
padding,
|
|
142
|
+
cv2.BORDER_CONSTANT,
|
|
143
|
+
value=(127, 127, 127),
|
|
144
|
+
)
|
|
145
|
+
return padded, padding
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def remove_padding(image: np.ndarray, padding: int) -> np.ndarray:
|
|
149
|
+
"""画像から上下左右の余白を取り除く。
|
|
150
|
+
|
|
151
|
+
余白がゼロ以下、または画像に対して大きすぎる場合は入力をそのまま返す。
|
|
152
|
+
|
|
153
|
+
Args:
|
|
154
|
+
image: 余白を含む画像。
|
|
155
|
+
padding: 上下左右から取り除くピクセル数。
|
|
156
|
+
|
|
157
|
+
Returns:
|
|
158
|
+
余白を取り除いた画像、または元の画像。
|
|
159
|
+
"""
|
|
160
|
+
if padding <= 0:
|
|
161
|
+
return image
|
|
162
|
+
if image.shape[0] <= padding * 2 or image.shape[1] <= padding * 2:
|
|
163
|
+
return image
|
|
164
|
+
return image[padding:-padding, padding:-padding]
|