comic-ocr-reader 0.0.5__tar.gz → 1.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/PKG-INFO +3 -3
  2. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/README.md +2 -2
  3. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/pyproject.toml +1 -1
  4. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/__main__.py +34 -15
  5. comic_ocr_reader-1.0.3/src/comic_ocr_reader/functions/model_init.py +92 -0
  6. comic_ocr_reader-1.0.3/tests/test_main_flow.py +46 -0
  7. comic_ocr_reader-1.0.3/tests/test_model_init.py +48 -0
  8. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.github/workflows/python-publish.yml +0 -0
  9. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.gitignore +0 -0
  10. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/comic-ocr.iml +0 -0
  11. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/inspectionProfiles/Project_Default.xml +0 -0
  12. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/inspectionProfiles/profiles_settings.xml +0 -0
  13. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/misc.xml +0 -0
  14. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/.idea/vcs.xml +0 -0
  15. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/LICENSE.md +0 -0
  16. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/requirements.txt +0 -0
  17. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/__init__.py +0 -0
  18. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/filepaths.py +0 -0
  19. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/functions/__init__.py +0 -0
  20. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/html_utils/__init__.py +0 -0
  21. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/html_utils/html_template.html +0 -0
  22. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/src/comic_ocr_reader/img_utils/__init__.py +0 -0
  23. {comic_ocr_reader-0.0.5 → comic_ocr_reader-1.0.3}/tests/test.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: comic_ocr_reader
3
- Version: 0.0.5
3
+ Version: 1.0.3
4
4
  Summary: A Python package made for processing Japanese manga into a format usable with web-based on-screen dictionaries like Yomitan.
5
5
  Project-URL: Homepage, https://github.com/HagantaRG/comic-ocr
6
6
  Author-email: HRG <rhaganta@gmail.com>
@@ -27,11 +27,11 @@ pip install comic-ocr-reader
27
27
  ```
28
28
  Once that it is installed (it may take some time, there are large dependencies in there) you should then run:
29
29
  ```
30
- comic-ocr-reader
30
+ comic_ocr_reader
31
31
  ```
32
32
  Or if that doesn't work:
33
33
  ```
34
- python3 -m comic-ocr-reader
34
+ python3 -m comic_ocr_reader
35
35
  ```
36
36
  ## Usage
37
37
  1. Prepare a folder containing the images you would like to process. As of v.0.0.1 the images in this folder must be named the page number you would like that image to be.
@@ -8,11 +8,11 @@ pip install comic-ocr-reader
8
8
  ```
9
9
  Once that it is installed (it may take some time, there are large dependencies in there) you should then run:
10
10
  ```
11
- comic-ocr-reader
11
+ comic_ocr_reader
12
12
  ```
13
13
  Or if that doesn't work:
14
14
  ```
15
- python3 -m comic-ocr-reader
15
+ python3 -m comic_ocr_reader
16
16
  ```
17
17
  ## Usage
18
18
  1. Prepare a folder containing the images you would like to process. As of v.0.0.1 the images in this folder must be named the page number you would like that image to be.
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "comic_ocr_reader"
7
- version = "0.0.5"
7
+ version = "1.0.3"
8
8
  authors = [
9
9
  { name="HRG", email="rhaganta@gmail.com" },
10
10
  ]
@@ -9,6 +9,7 @@ from manga_ocr import MangaOcr
9
9
 
10
10
  from comic_ocr_reader.html_utils import Page, make_html_file
11
11
  from comic_ocr_reader.functions import process_page
12
+ from comic_ocr_reader.functions.model_init import ensure_models_initialised
12
13
 
13
14
  VALID_EXTENSIONS: Final[list[str]] = [
14
15
  "jpg",
@@ -19,6 +20,14 @@ VALID_EXTENSIONS: Final[list[str]] = [
19
20
  ]
20
21
  logger.disable("manga_ocr")
21
22
 
23
+
24
+ def print_main_menu() -> None:
25
+ print(
26
+ 'Hello! Please type in "help" to see a list of commands. '
27
+ 'Otherwise, please type in a valid command.'
28
+ )
29
+
30
+
22
31
  def main(
23
32
  folder_path: str,
24
33
  manga_name: str = ...
@@ -45,12 +54,19 @@ def main(
45
54
  print("One of the image files in the folder you have entered has a non-integer name. (e.g. 123abc.jpeg instead of 123.jpeg)\n"
46
55
  "All image files within the folder *must* have an integer name for ordering purposes.")
47
56
  return False
57
+
58
+ print("=== Initialisation ===")
59
+ ensure_models_initialised()
60
+
61
+ print("=== OCR setup ===")
48
62
  detector = Reader(
49
63
  lang_list=['ja'],
50
64
  recognizer=False,
51
65
  gpu=True
52
66
  )
53
67
  recogniser = MangaOcr()
68
+
69
+ print("=== OCR processing ===")
54
70
  for file in tqdm(images):
55
71
  page_num += 1
56
72
  process_page(
@@ -67,38 +83,41 @@ def main(
67
83
  f.flush()
68
84
  return True
69
85
 
70
- if __name__ == "__main__":
86
+
87
+ def run() -> None:
71
88
  while True:
72
89
  try:
73
- user_input: str = input(
74
- f"Hello! Please type in \"help\" to see a list of commands. Otherwise, please type in a valid command.\n"
75
- )
76
- user_input = user_input.strip()
90
+ print_main_menu()
91
+ user_input: str = input().strip().lower()
77
92
  match user_input:
78
93
  case "help":
79
94
  print(
80
95
  """
81
96
  The valid commands are:
82
- - folder
97
+ - folder
83
98
  Allows you to specify a folder containing the image files for the manga you want to process.
84
99
  Also optionally allows you to give the manga a name. Otherwise the folder name will be used.
85
- Supported file formats for images in folder: .jpg, .jpeg, .png, .bmp, .tiff
86
- And of course,
87
- - help
88
- Which you are currently using.
100
+ Supported file formats for images in folder: .jpg, .jpeg, .png, .bmp, .tiff
101
+ And of course,
102
+ - help
103
+ Which you are currently using.
89
104
  """
90
105
  )
91
106
  case "folder":
92
107
  target_folder: str = input("Please input a target folder.\n")
93
108
  if os.path.isdir(target_folder):
94
109
  if main(target_folder):
95
- print("Success! Please press Ctrl+C to exit.")
110
+ print("Success! Returning to the main menu.")
96
111
  else:
97
- print("Something went wrong.")
112
+ print("Something went wrong. Returning to the main menu.")
98
113
  else:
99
- print("Please input a valid folder.")
114
+ print("Please input a valid folder. Returning to the main menu.")
100
115
  case _:
101
- print("That was not a valid command.")
116
+ print("That was not a valid command. Returning to the main menu.")
102
117
  except KeyboardInterrupt:
103
118
  print("\nGoodbye!")
104
- exit(0)
119
+ return
120
+
121
+
122
+ if __name__ == "__main__":
123
+ run()
@@ -0,0 +1,92 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ from pathlib import Path
5
+ from typing import Callable, Optional
6
+
7
+ from easyocr import Reader
8
+ from easyocr.config import MODULE_PATH as EASYOCR_MODULE_PATH, detection_models as EASYOCR_DETECTION_MODELS
9
+ from manga_ocr import MangaOcr
10
+
11
+ EASYOCR_DETECTOR_NAME = "craft"
12
+ EASYOCR_DETECTOR_MODEL = Path(EASYOCR_MODULE_PATH) / "model" / EASYOCR_DETECTION_MODELS[EASYOCR_DETECTOR_NAME]["filename"]
13
+
14
+ MANGA_OCR_REPO = (
15
+ Path.home()
16
+ / ".cache"
17
+ / "huggingface"
18
+ / "hub"
19
+ / "models--kha-white--manga-ocr-base"
20
+ )
21
+ MANGA_OCR_REQUIRED_FILES = (
22
+ "config.json",
23
+ "preprocessor_config.json",
24
+ "pytorch_model.bin",
25
+ "special_tokens_map.json",
26
+ "tokenizer_config.json",
27
+ "vocab.txt",
28
+ )
29
+
30
+
31
+ def _file_md5(path: Path) -> str:
32
+ digest = hashlib.md5()
33
+ with path.open("rb") as file_handle:
34
+ for chunk in iter(lambda: file_handle.read(1024 * 1024), b""):
35
+ digest.update(chunk)
36
+ return digest.hexdigest()
37
+
38
+
39
+ def easyocr_detector_is_initialised(model_path: Path = EASYOCR_DETECTOR_MODEL) -> bool:
40
+ expected_md5 = EASYOCR_DETECTION_MODELS[EASYOCR_DETECTOR_NAME]["md5sum"]
41
+ return model_path.is_file() and _file_md5(model_path) == expected_md5
42
+
43
+
44
+ def manga_ocr_is_initialised(repo_path: Path = MANGA_OCR_REPO) -> bool:
45
+ refs_main = repo_path / "refs" / "main"
46
+ if not refs_main.is_file():
47
+ return False
48
+
49
+ snapshot_hash = refs_main.read_text(encoding="utf-8").strip()
50
+ if not snapshot_hash:
51
+ return False
52
+
53
+ snapshot_dir = repo_path / "snapshots" / snapshot_hash
54
+ if not snapshot_dir.is_dir():
55
+ return False
56
+
57
+ return all((snapshot_dir / filename).is_file() for filename in MANGA_OCR_REQUIRED_FILES)
58
+
59
+
60
+ def models_are_initialised(
61
+ easyocr_model_path: Path = EASYOCR_DETECTOR_MODEL,
62
+ manga_repo_path: Path = MANGA_OCR_REPO,
63
+ ) -> bool:
64
+ return easyocr_detector_is_initialised(easyocr_model_path) and manga_ocr_is_initialised(manga_repo_path)
65
+
66
+
67
+ def _download_easyocr_detector() -> None:
68
+ Reader(lang_list=["ja"], recognizer=False, gpu=True, detect_network=EASYOCR_DETECTOR_NAME)
69
+
70
+
71
+ def _download_manga_ocr() -> None:
72
+ MangaOcr()
73
+
74
+
75
+ def ensure_models_initialised(
76
+ easyocr_model_path: Path = EASYOCR_DETECTOR_MODEL,
77
+ manga_repo_path: Path = MANGA_OCR_REPO,
78
+ easyocr_factory: Optional[Callable[[], None]] = None,
79
+ manga_factory: Optional[Callable[[], None]] = None,
80
+ ) -> bool:
81
+ if models_are_initialised(easyocr_model_path, manga_repo_path):
82
+ print("Not required! You have already initialised.")
83
+ return False
84
+
85
+ if not easyocr_detector_is_initialised(easyocr_model_path):
86
+ (easyocr_factory or _download_easyocr_detector)()
87
+
88
+ if not manga_ocr_is_initialised(manga_repo_path):
89
+ (manga_factory or _download_manga_ocr)()
90
+
91
+ print("Initialisation complete.")
92
+ return True
@@ -0,0 +1,46 @@
1
+ import builtins
2
+ import contextlib
3
+ import io
4
+ import unittest
5
+ from pathlib import Path
6
+ from unittest.mock import mock_open, patch
7
+
8
+ from comic_ocr_reader import __main__ as cli
9
+
10
+
11
+ class MainFlowTests(unittest.TestCase):
12
+ def test_main_runs_initialisation_before_ocr_processing(self):
13
+ folder = r"C:\fake\folder"
14
+ output = io.StringIO()
15
+
16
+ with patch.object(cli, "ensure_models_initialised") as init_mock:
17
+ with patch.object(cli, "Reader") as reader_mock:
18
+ with patch.object(cli, "MangaOcr") as manga_mock:
19
+ with patch.object(cli, "process_page") as process_page_mock:
20
+ with patch.object(cli, "make_html_file", return_value="<html></html>"):
21
+ with patch.object(cli, "tqdm", side_effect=lambda items: items):
22
+ with patch.object(cli.os.path, "isdir", return_value=True):
23
+ with patch.object(cli.os, "listdir", return_value=["1.jpg"]):
24
+ with patch.object(Path, "is_file", return_value=True):
25
+ with patch.object(builtins, "open", mock_open()):
26
+ reader_mock.return_value = object()
27
+ manga_mock.return_value = object()
28
+ with contextlib.redirect_stdout(output):
29
+ result = cli.main(folder)
30
+
31
+ self.assertTrue(result)
32
+ init_mock.assert_called_once()
33
+ reader_mock.assert_called_once()
34
+ manga_mock.assert_called_once()
35
+ process_page_mock.assert_called_once()
36
+
37
+ rendered = output.getvalue()
38
+ self.assertIn("=== Initialisation ===", rendered)
39
+ self.assertIn("=== OCR setup ===", rendered)
40
+ self.assertIn("=== OCR processing ===", rendered)
41
+ self.assertLess(rendered.index("=== Initialisation ==="), rendered.index("=== OCR setup ==="))
42
+ self.assertLess(rendered.index("=== OCR setup ==="), rendered.index("=== OCR processing ==="))
43
+
44
+
45
+ if __name__ == "__main__":
46
+ unittest.main()
@@ -0,0 +1,48 @@
1
+ import contextlib
2
+ import io
3
+ import unittest
4
+ from unittest.mock import patch
5
+
6
+ from comic_ocr_reader.functions import model_init
7
+
8
+
9
+ class ModelInitTests(unittest.TestCase):
10
+ def test_ensure_models_initialised_prints_noop_message_when_ready(self):
11
+ with patch.object(model_init, "models_are_initialised", return_value=True):
12
+ with patch.object(model_init, "_download_easyocr_detector") as easyocr_factory:
13
+ with patch.object(model_init, "_download_manga_ocr") as manga_factory:
14
+ buffer = io.StringIO()
15
+ with contextlib.redirect_stdout(buffer):
16
+ result = model_init.ensure_models_initialised()
17
+
18
+ self.assertFalse(result)
19
+ self.assertEqual(buffer.getvalue(), "Not required! You have already initialised.\n")
20
+ easyocr_factory.assert_not_called()
21
+ manga_factory.assert_not_called()
22
+
23
+ def test_ensure_models_initialised_calls_missing_factories(self):
24
+ with patch.object(model_init, "models_are_initialised", return_value=False):
25
+ with patch.object(model_init, "easyocr_detector_is_initialised", return_value=False):
26
+ with patch.object(model_init, "manga_ocr_is_initialised", return_value=False):
27
+ calls = []
28
+
29
+ def easyocr_factory():
30
+ calls.append("easyocr")
31
+
32
+ def manga_factory():
33
+ calls.append("manga")
34
+
35
+ buffer = io.StringIO()
36
+ with contextlib.redirect_stdout(buffer):
37
+ result = model_init.ensure_models_initialised(
38
+ easyocr_factory=easyocr_factory,
39
+ manga_factory=manga_factory,
40
+ )
41
+
42
+ self.assertTrue(result)
43
+ self.assertEqual(calls, ["easyocr", "manga"])
44
+ self.assertEqual(buffer.getvalue(), "Initialisation complete.\n")
45
+
46
+
47
+ if __name__ == "__main__":
48
+ unittest.main()