comic-ocr-reader 0.0.5__tar.gz → 1.0.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/PKG-INFO +3 -3
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/README.md +2 -2
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/pyproject.toml +1 -1
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/__main__.py +34 -15
- comic_ocr_reader-1.0.3/src/comic_ocr_reader/functions/model_init.py +92 -0
- comic_ocr_reader-1.0.3/tests/test_main_flow.py +46 -0
- comic_ocr_reader-1.0.3/tests/test_model_init.py +48 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.github/workflows/python-publish.yml +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.gitignore +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/comic-ocr.iml +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/inspectionProfiles/Project_Default.xml +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/inspectionProfiles/profiles_settings.xml +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/misc.xml +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/vcs.xml +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/LICENSE.md +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/requirements.txt +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/__init__.py +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/filepaths.py +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/functions/__init__.py +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/html_utils/__init__.py +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/html_utils/html_template.html +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/img_utils/__init__.py +0 -0
- {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/tests/test.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: comic_ocr_reader
|
|
3
|
-
Version:
|
|
3
|
+
Version: 1.0.3
|
|
4
4
|
Summary: A Python package made for processing Japanese manga into a format usable with web-based on-screen dictionaries like Yomitan.
|
|
5
5
|
Project-URL: Homepage, https://github.com/HagantaRG/comic-ocr
|
|
6
6
|
Author-email: HRG <rhaganta@gmail.com>
|
|
@@ -27,11 +27,11 @@ pip install comic-ocr-reader
|
|
|
27
27
|
```
|
|
28
28
|
Once that it is installed (it may take some time, there are large dependencies in there) you should then run:
|
|
29
29
|
```
|
|
30
|
-
|
|
30
|
+
comic_ocr_reader
|
|
31
31
|
```
|
|
32
32
|
Or if that doesn't work:
|
|
33
33
|
```
|
|
34
|
-
python3 -m
|
|
34
|
+
python3 -m comic_ocr_reader
|
|
35
35
|
```
|
|
36
36
|
## Usage
|
|
37
37
|
1. Prepare a folder containing the images you would like to process. As of v.0.0.1 the images in this folder must be named the page number you would like that image to be.
|
|
@@ -8,11 +8,11 @@ pip install comic-ocr-reader
|
|
|
8
8
|
```
|
|
9
9
|
Once that it is installed (it may take some time, there are large dependencies in there) you should then run:
|
|
10
10
|
```
|
|
11
|
-
|
|
11
|
+
comic_ocr_reader
|
|
12
12
|
```
|
|
13
13
|
Or if that doesn't work:
|
|
14
14
|
```
|
|
15
|
-
python3 -m
|
|
15
|
+
python3 -m comic_ocr_reader
|
|
16
16
|
```
|
|
17
17
|
## Usage
|
|
18
18
|
1. Prepare a folder containing the images you would like to process. As of v.0.0.1 the images in this folder must be named the page number you would like that image to be.
|
|
@@ -9,6 +9,7 @@ from manga_ocr import MangaOcr
|
|
|
9
9
|
|
|
10
10
|
from comic_ocr_reader.html_utils import Page, make_html_file
|
|
11
11
|
from comic_ocr_reader.functions import process_page
|
|
12
|
+
from comic_ocr_reader.functions.model_init import ensure_models_initialised
|
|
12
13
|
|
|
13
14
|
VALID_EXTENSIONS: Final[list[str]] = [
|
|
14
15
|
"jpg",
|
|
@@ -19,6 +20,14 @@ VALID_EXTENSIONS: Final[list[str]] = [
|
|
|
19
20
|
]
|
|
20
21
|
logger.disable("manga_ocr")
|
|
21
22
|
|
|
23
|
+
|
|
24
|
+
def print_main_menu() -> None:
|
|
25
|
+
print(
|
|
26
|
+
'Hello! Please type in "help" to see a list of commands. '
|
|
27
|
+
'Otherwise, please type in a valid command.'
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
22
31
|
def main(
|
|
23
32
|
folder_path: str,
|
|
24
33
|
manga_name: str = ...
|
|
@@ -45,12 +54,19 @@ def main(
|
|
|
45
54
|
print("One of the image files in the folder you have entered has a non-integer name. (e.g. 123abc.jpeg instead of 123.jpeg)\n"
|
|
46
55
|
"All image files within the folder *must* have an integer name for ordering purposes.")
|
|
47
56
|
return False
|
|
57
|
+
|
|
58
|
+
print("=== Initialisation ===")
|
|
59
|
+
ensure_models_initialised()
|
|
60
|
+
|
|
61
|
+
print("=== OCR setup ===")
|
|
48
62
|
detector = Reader(
|
|
49
63
|
lang_list=['ja'],
|
|
50
64
|
recognizer=False,
|
|
51
65
|
gpu=True
|
|
52
66
|
)
|
|
53
67
|
recogniser = MangaOcr()
|
|
68
|
+
|
|
69
|
+
print("=== OCR processing ===")
|
|
54
70
|
for file in tqdm(images):
|
|
55
71
|
page_num += 1
|
|
56
72
|
process_page(
|
|
@@ -67,38 +83,41 @@ def main(
|
|
|
67
83
|
f.flush()
|
|
68
84
|
return True
|
|
69
85
|
|
|
70
|
-
|
|
86
|
+
|
|
87
|
+
def run() -> None:
|
|
71
88
|
while True:
|
|
72
89
|
try:
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
)
|
|
76
|
-
user_input = user_input.strip()
|
|
90
|
+
print_main_menu()
|
|
91
|
+
user_input: str = input().strip().lower()
|
|
77
92
|
match user_input:
|
|
78
93
|
case "help":
|
|
79
94
|
print(
|
|
80
95
|
"""
|
|
81
96
|
The valid commands are:
|
|
82
|
-
- folder
|
|
97
|
+
- folder
|
|
83
98
|
Allows you to specify a folder containing the image files for the manga you want to process.
|
|
84
99
|
Also optionally allows you to give the manga a name. Otherwise the folder name will be used.
|
|
85
|
-
Supported file formats for images in folder: .jpg, .jpeg, .png, .bmp, .tiff
|
|
86
|
-
And of course,
|
|
87
|
-
- help
|
|
88
|
-
|
|
100
|
+
Supported file formats for images in folder: .jpg, .jpeg, .png, .bmp, .tiff
|
|
101
|
+
And of course,
|
|
102
|
+
- help
|
|
103
|
+
Which you are currently using.
|
|
89
104
|
"""
|
|
90
105
|
)
|
|
91
106
|
case "folder":
|
|
92
107
|
target_folder: str = input("Please input a target folder.\n")
|
|
93
108
|
if os.path.isdir(target_folder):
|
|
94
109
|
if main(target_folder):
|
|
95
|
-
print("Success!
|
|
110
|
+
print("Success! Returning to the main menu.")
|
|
96
111
|
else:
|
|
97
|
-
print("Something went wrong.")
|
|
112
|
+
print("Something went wrong. Returning to the main menu.")
|
|
98
113
|
else:
|
|
99
|
-
print("Please input a valid folder.")
|
|
114
|
+
print("Please input a valid folder. Returning to the main menu.")
|
|
100
115
|
case _:
|
|
101
|
-
print("That was not a valid command.")
|
|
116
|
+
print("That was not a valid command. Returning to the main menu.")
|
|
102
117
|
except KeyboardInterrupt:
|
|
103
118
|
print("\nGoodbye!")
|
|
104
|
-
|
|
119
|
+
return
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
if __name__ == "__main__":
|
|
123
|
+
run()
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Callable, Optional
|
|
6
|
+
|
|
7
|
+
from easyocr import Reader
|
|
8
|
+
from easyocr.config import MODULE_PATH as EASYOCR_MODULE_PATH, detection_models as EASYOCR_DETECTION_MODELS
|
|
9
|
+
from manga_ocr import MangaOcr
|
|
10
|
+
|
|
11
|
+
EASYOCR_DETECTOR_NAME = "craft"
|
|
12
|
+
EASYOCR_DETECTOR_MODEL = Path(EASYOCR_MODULE_PATH) / "model" / EASYOCR_DETECTION_MODELS[EASYOCR_DETECTOR_NAME]["filename"]
|
|
13
|
+
|
|
14
|
+
MANGA_OCR_REPO = (
|
|
15
|
+
Path.home()
|
|
16
|
+
/ ".cache"
|
|
17
|
+
/ "huggingface"
|
|
18
|
+
/ "hub"
|
|
19
|
+
/ "models--kha-white--manga-ocr-base"
|
|
20
|
+
)
|
|
21
|
+
MANGA_OCR_REQUIRED_FILES = (
|
|
22
|
+
"config.json",
|
|
23
|
+
"preprocessor_config.json",
|
|
24
|
+
"pytorch_model.bin",
|
|
25
|
+
"special_tokens_map.json",
|
|
26
|
+
"tokenizer_config.json",
|
|
27
|
+
"vocab.txt",
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _file_md5(path: Path) -> str:
|
|
32
|
+
digest = hashlib.md5()
|
|
33
|
+
with path.open("rb") as file_handle:
|
|
34
|
+
for chunk in iter(lambda: file_handle.read(1024 * 1024), b""):
|
|
35
|
+
digest.update(chunk)
|
|
36
|
+
return digest.hexdigest()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def easyocr_detector_is_initialised(model_path: Path = EASYOCR_DETECTOR_MODEL) -> bool:
|
|
40
|
+
expected_md5 = EASYOCR_DETECTION_MODELS[EASYOCR_DETECTOR_NAME]["md5sum"]
|
|
41
|
+
return model_path.is_file() and _file_md5(model_path) == expected_md5
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def manga_ocr_is_initialised(repo_path: Path = MANGA_OCR_REPO) -> bool:
|
|
45
|
+
refs_main = repo_path / "refs" / "main"
|
|
46
|
+
if not refs_main.is_file():
|
|
47
|
+
return False
|
|
48
|
+
|
|
49
|
+
snapshot_hash = refs_main.read_text(encoding="utf-8").strip()
|
|
50
|
+
if not snapshot_hash:
|
|
51
|
+
return False
|
|
52
|
+
|
|
53
|
+
snapshot_dir = repo_path / "snapshots" / snapshot_hash
|
|
54
|
+
if not snapshot_dir.is_dir():
|
|
55
|
+
return False
|
|
56
|
+
|
|
57
|
+
return all((snapshot_dir / filename).is_file() for filename in MANGA_OCR_REQUIRED_FILES)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def models_are_initialised(
|
|
61
|
+
easyocr_model_path: Path = EASYOCR_DETECTOR_MODEL,
|
|
62
|
+
manga_repo_path: Path = MANGA_OCR_REPO,
|
|
63
|
+
) -> bool:
|
|
64
|
+
return easyocr_detector_is_initialised(easyocr_model_path) and manga_ocr_is_initialised(manga_repo_path)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _download_easyocr_detector() -> None:
|
|
68
|
+
Reader(lang_list=["ja"], recognizer=False, gpu=True, detect_network=EASYOCR_DETECTOR_NAME)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _download_manga_ocr() -> None:
|
|
72
|
+
MangaOcr()
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def ensure_models_initialised(
|
|
76
|
+
easyocr_model_path: Path = EASYOCR_DETECTOR_MODEL,
|
|
77
|
+
manga_repo_path: Path = MANGA_OCR_REPO,
|
|
78
|
+
easyocr_factory: Optional[Callable[[], None]] = None,
|
|
79
|
+
manga_factory: Optional[Callable[[], None]] = None,
|
|
80
|
+
) -> bool:
|
|
81
|
+
if models_are_initialised(easyocr_model_path, manga_repo_path):
|
|
82
|
+
print("Not required! You have already initialised.")
|
|
83
|
+
return False
|
|
84
|
+
|
|
85
|
+
if not easyocr_detector_is_initialised(easyocr_model_path):
|
|
86
|
+
(easyocr_factory or _download_easyocr_detector)()
|
|
87
|
+
|
|
88
|
+
if not manga_ocr_is_initialised(manga_repo_path):
|
|
89
|
+
(manga_factory or _download_manga_ocr)()
|
|
90
|
+
|
|
91
|
+
print("Initialisation complete.")
|
|
92
|
+
return True
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import builtins
|
|
2
|
+
import contextlib
|
|
3
|
+
import io
|
|
4
|
+
import unittest
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from unittest.mock import mock_open, patch
|
|
7
|
+
|
|
8
|
+
from comic_ocr_reader import __main__ as cli
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class MainFlowTests(unittest.TestCase):
|
|
12
|
+
def test_main_runs_initialisation_before_ocr_processing(self):
|
|
13
|
+
folder = r"C:\fake\folder"
|
|
14
|
+
output = io.StringIO()
|
|
15
|
+
|
|
16
|
+
with patch.object(cli, "ensure_models_initialised") as init_mock:
|
|
17
|
+
with patch.object(cli, "Reader") as reader_mock:
|
|
18
|
+
with patch.object(cli, "MangaOcr") as manga_mock:
|
|
19
|
+
with patch.object(cli, "process_page") as process_page_mock:
|
|
20
|
+
with patch.object(cli, "make_html_file", return_value="<html></html>"):
|
|
21
|
+
with patch.object(cli, "tqdm", side_effect=lambda items: items):
|
|
22
|
+
with patch.object(cli.os.path, "isdir", return_value=True):
|
|
23
|
+
with patch.object(cli.os, "listdir", return_value=["1.jpg"]):
|
|
24
|
+
with patch.object(Path, "is_file", return_value=True):
|
|
25
|
+
with patch.object(builtins, "open", mock_open()):
|
|
26
|
+
reader_mock.return_value = object()
|
|
27
|
+
manga_mock.return_value = object()
|
|
28
|
+
with contextlib.redirect_stdout(output):
|
|
29
|
+
result = cli.main(folder)
|
|
30
|
+
|
|
31
|
+
self.assertTrue(result)
|
|
32
|
+
init_mock.assert_called_once()
|
|
33
|
+
reader_mock.assert_called_once()
|
|
34
|
+
manga_mock.assert_called_once()
|
|
35
|
+
process_page_mock.assert_called_once()
|
|
36
|
+
|
|
37
|
+
rendered = output.getvalue()
|
|
38
|
+
self.assertIn("=== Initialisation ===", rendered)
|
|
39
|
+
self.assertIn("=== OCR setup ===", rendered)
|
|
40
|
+
self.assertIn("=== OCR processing ===", rendered)
|
|
41
|
+
self.assertLess(rendered.index("=== Initialisation ==="), rendered.index("=== OCR setup ==="))
|
|
42
|
+
self.assertLess(rendered.index("=== OCR setup ==="), rendered.index("=== OCR processing ==="))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
if __name__ == "__main__":
|
|
46
|
+
unittest.main()
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import contextlib
|
|
2
|
+
import io
|
|
3
|
+
import unittest
|
|
4
|
+
from unittest.mock import patch
|
|
5
|
+
|
|
6
|
+
from comic_ocr_reader.functions import model_init
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ModelInitTests(unittest.TestCase):
|
|
10
|
+
def test_ensure_models_initialised_prints_noop_message_when_ready(self):
|
|
11
|
+
with patch.object(model_init, "models_are_initialised", return_value=True):
|
|
12
|
+
with patch.object(model_init, "_download_easyocr_detector") as easyocr_factory:
|
|
13
|
+
with patch.object(model_init, "_download_manga_ocr") as manga_factory:
|
|
14
|
+
buffer = io.StringIO()
|
|
15
|
+
with contextlib.redirect_stdout(buffer):
|
|
16
|
+
result = model_init.ensure_models_initialised()
|
|
17
|
+
|
|
18
|
+
self.assertFalse(result)
|
|
19
|
+
self.assertEqual(buffer.getvalue(), "Not required! You have already initialised.\n")
|
|
20
|
+
easyocr_factory.assert_not_called()
|
|
21
|
+
manga_factory.assert_not_called()
|
|
22
|
+
|
|
23
|
+
def test_ensure_models_initialised_calls_missing_factories(self):
|
|
24
|
+
with patch.object(model_init, "models_are_initialised", return_value=False):
|
|
25
|
+
with patch.object(model_init, "easyocr_detector_is_initialised", return_value=False):
|
|
26
|
+
with patch.object(model_init, "manga_ocr_is_initialised", return_value=False):
|
|
27
|
+
calls = []
|
|
28
|
+
|
|
29
|
+
def easyocr_factory():
|
|
30
|
+
calls.append("easyocr")
|
|
31
|
+
|
|
32
|
+
def manga_factory():
|
|
33
|
+
calls.append("manga")
|
|
34
|
+
|
|
35
|
+
buffer = io.StringIO()
|
|
36
|
+
with contextlib.redirect_stdout(buffer):
|
|
37
|
+
result = model_init.ensure_models_initialised(
|
|
38
|
+
easyocr_factory=easyocr_factory,
|
|
39
|
+
manga_factory=manga_factory,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
self.assertTrue(result)
|
|
43
|
+
self.assertEqual(calls, ["easyocr", "manga"])
|
|
44
|
+
self.assertEqual(buffer.getvalue(), "Initialisation complete.\n")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
if __name__ == "__main__":
|
|
48
|
+
unittest.main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/inspectionProfiles/Project_Default.xml
RENAMED
|
File without changes
|
{comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/inspectionProfiles/profiles_settings.xml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/functions/__init__.py
RENAMED
|
File without changes
|
{comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/html_utils/__init__.py
RENAMED
|
File without changes
|
{comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/html_utils/html_template.html
RENAMED
|
File without changes
|
{comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/img_utils/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|