mokuro-bridge 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mokuro_bridge-0.7.0/.env.example +96 -0
- mokuro_bridge-0.7.0/LICENSE +21 -0
- mokuro_bridge-0.7.0/MANIFEST.in +41 -0
- mokuro_bridge-0.7.0/PKG-INFO +327 -0
- mokuro_bridge-0.7.0/README.md +272 -0
- mokuro_bridge-0.7.0/com.mokuro-bridge.plist +46 -0
- mokuro_bridge-0.7.0/docs/architecture.md +210 -0
- mokuro_bridge-0.7.0/install-launchd.sh +78 -0
- mokuro_bridge-0.7.0/mokuro_bridge/__init__.py +2 -0
- mokuro_bridge-0.7.0/mokuro_bridge/__main__.py +13 -0
- mokuro_bridge-0.7.0/mokuro_bridge/accounts.py +425 -0
- mokuro_bridge-0.7.0/mokuro_bridge/api.py +2848 -0
- mokuro_bridge-0.7.0/mokuro_bridge/cli.py +486 -0
- mokuro_bridge-0.7.0/mokuro_bridge/config.py +224 -0
- mokuro_bridge-0.7.0/mokuro_bridge/creds.py +296 -0
- mokuro_bridge-0.7.0/mokuro_bridge/fetchproxy.py +379 -0
- mokuro_bridge-0.7.0/mokuro_bridge/log.py +94 -0
- mokuro_bridge-0.7.0/mokuro_bridge/ocr.py +1285 -0
- mokuro_bridge-0.7.0/mokuro_bridge/ocr_folder.py +228 -0
- mokuro_bridge-0.7.0/mokuro_bridge/providers/__init__.py +416 -0
- mokuro_bridge-0.7.0/mokuro_bridge/providers/drive.py +568 -0
- mokuro_bridge-0.7.0/mokuro_bridge/providers/mega.py +476 -0
- mokuro_bridge-0.7.0/mokuro_bridge/providers/onedrive.py +481 -0
- mokuro_bridge-0.7.0/mokuro_bridge/sessions.py +382 -0
- mokuro_bridge-0.7.0/mokuro_bridge/update.py +369 -0
- mokuro_bridge-0.7.0/mokuro_bridge/util.py +114 -0
- mokuro_bridge-0.7.0/mokuro_bridge.egg-info/PKG-INFO +327 -0
- mokuro_bridge-0.7.0/mokuro_bridge.egg-info/SOURCES.txt +54 -0
- mokuro_bridge-0.7.0/mokuro_bridge.egg-info/dependency_links.txt +1 -0
- mokuro_bridge-0.7.0/mokuro_bridge.egg-info/entry_points.txt +3 -0
- mokuro_bridge-0.7.0/mokuro_bridge.egg-info/requires.txt +32 -0
- mokuro_bridge-0.7.0/mokuro_bridge.egg-info/top_level.txt +1 -0
- mokuro_bridge-0.7.0/ocr_folder.py +17 -0
- mokuro_bridge-0.7.0/pyproject.toml +113 -0
- mokuro_bridge-0.7.0/pytest.ini +4 -0
- mokuro_bridge-0.7.0/requirements-dev.txt +10 -0
- mokuro_bridge-0.7.0/requirements-drive.txt +14 -0
- mokuro_bridge-0.7.0/requirements-ocr.txt +19 -0
- mokuro_bridge-0.7.0/requirements-onedrive.txt +8 -0
- mokuro_bridge-0.7.0/requirements.txt +23 -0
- mokuro_bridge-0.7.0/run.sh +19 -0
- mokuro_bridge-0.7.0/server.py +24 -0
- mokuro_bridge-0.7.0/setup-keychain.sh +45 -0
- mokuro_bridge-0.7.0/setup.cfg +4 -0
- mokuro_bridge-0.7.0/tests/test_accounts.py +151 -0
- mokuro_bridge-0.7.0/tests/test_api_hardening.py +128 -0
- mokuro_bridge-0.7.0/tests/test_endpoints.py +358 -0
- mokuro_bridge-0.7.0/tests/test_fetchproxy.py +377 -0
- mokuro_bridge-0.7.0/tests/test_imports.py +56 -0
- mokuro_bridge-0.7.0/tests/test_mega_creds.py +149 -0
- mokuro_bridge-0.7.0/tests/test_ocr_folder.py +87 -0
- mokuro_bridge-0.7.0/tests/test_ocr_scheduling.py +199 -0
- mokuro_bridge-0.7.0/tests/test_optional_ocr.py +92 -0
- mokuro_bridge-0.7.0/tests/test_packaging.py +216 -0
- mokuro_bridge-0.7.0/tests/test_update.py +335 -0
- mokuro_bridge-0.7.0/tests/test_upload_methods.py +155 -0
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# mokuro-bridge configuration - copy to .env and adapt, or export these.
|
|
2
|
+
# The server does NOT read .env automatically; either `set -a && source .env
|
|
3
|
+
# && set +a` before ./run.sh, or export the values in your launcher/shell.
|
|
4
|
+
|
|
5
|
+
# ── Server ─────────────────────────────────────────────────────────────
|
|
6
|
+
# Keep the default 127.0.0.1 - the bridge is unauthenticated, so loopback is
|
|
7
|
+
# the only access control it has.
|
|
8
|
+
#MOKURO_BRIDGE_HOST=127.0.0.1
|
|
9
|
+
#MOKURO_BRIDGE_PORT=62642 # 62642 spells "MANGA" on a phone keypad
|
|
10
|
+
|
|
11
|
+
# Origins allowed to POST from browser userscripts (comma separated).
|
|
12
|
+
# Default: * (any origin). The bridge is loopback-only and sends no cookies,
|
|
13
|
+
# so it is unauthenticated by design; this variable is the opt-in way to
|
|
14
|
+
# narrow it back to a list. Uncomment to allow only the viewers you use:
|
|
15
|
+
#CORS_ORIGINS=https://viewer.bookwalker.jp,https://viewer-trial.bookwalker.jp
|
|
16
|
+
|
|
17
|
+
# ── Paths ───────────────────────────────────────────────────────────────
|
|
18
|
+
# Scratch space for pages + OCR JSON while a volume is in progress.
|
|
19
|
+
#MOKURO_BRIDGE_WORK_DIR=/Users/you/mokuro-input
|
|
20
|
+
|
|
21
|
+
# Where finished <series>/<volume>.{cbz,mokuro,webp} land when not uploading.
|
|
22
|
+
#MOKURO_BRIDGE_OUTPUT_DIR=/Users/you/mokuro-bridge/output
|
|
23
|
+
|
|
24
|
+
# Base URL `mokuro-bridge-ocr` (the folder-OCR companion) uses to reach a
|
|
25
|
+
# running bridge. Match it to MOKURO_BRIDGE_HOST/PORT above.
|
|
26
|
+
#MOKURO_BRIDGE_URL=http://127.0.0.1:62642
|
|
27
|
+
|
|
28
|
+
# ── MEGA upload (optional - off by default) ─────────────────────────────
|
|
29
|
+
#MOKURO_BRIDGE_UPLOAD_DEFAULT=true # finalize uploads unless told otherwise
|
|
30
|
+
# (also accepts an account id: mega:work)
|
|
31
|
+
#MEGA_LIBRARY_ROOT=/Root/mokuro-reader
|
|
32
|
+
#MEGA_EMAIL=you@example.com # credentials (alternative to
|
|
33
|
+
#MEGA_PASSWORD=super-secret # `python3 server.py --setup-upload mega`)
|
|
34
|
+
#MEGA_CREDS_FILE=/Users/you/.config/mokuro-bridge/credentials.env
|
|
35
|
+
|
|
36
|
+
# ── Extra accounts (optional) ───────────────────────────────────────────
|
|
37
|
+
# Every provider can hold several accounts: `python3 server.py --setup-upload
|
|
38
|
+
# mega --name work [--root /Root/other-library] [--label "Work"]` adds one,
|
|
39
|
+
# `--list-uploads` shows them, `--remove-upload mega:work` deletes one. The
|
|
40
|
+
# bare ids (mega/drive/onedrive) stay the default account; the extras are
|
|
41
|
+
# addressed as `<provider>:<name>`. Non-secret metadata lives in this
|
|
42
|
+
# directory (secrets stay in the keychain / their own 0600 files):
|
|
43
|
+
#MOKURO_BRIDGE_ACCOUNTS_DIR=/Users/you/.config/mokuro-bridge/accounts
|
|
44
|
+
|
|
45
|
+
# ── Google Drive upload (optional; needs pip install -r requirements-drive.txt)
|
|
46
|
+
# Run `python3 server.py --setup-upload drive` - it walks you through creating
|
|
47
|
+
# a free Google Cloud "Desktop app" OAuth client and pasting its Client ID +
|
|
48
|
+
# Client secret (Google requires the secret at token exchange; it is never
|
|
49
|
+
# stored). Optional overrides:
|
|
50
|
+
#DRIVE_ROOT_NAME=mokuro-reader # library folder at My Drive root
|
|
51
|
+
#DRIVE_CREDS_FILE=/Users/you/.config/mokuro-bridge/drive_credentials.json
|
|
52
|
+
#DRIVE_CLIENT_ID=1234-....apps.googleusercontent.com # skip the paste step
|
|
53
|
+
#DRIVE_CLIENT_SECRET=GOCSPX-.... # ...and the secret
|
|
54
|
+
#DRIVE_CLIENT_SECRET_FILE=/Users/you/client_secret.json # or a full client_secrets.json
|
|
55
|
+
|
|
56
|
+
# ── OneDrive upload (optional; needs pip install -r requirements-onedrive.txt)
|
|
57
|
+
#ONEDRIVE_CLIENT_ID=00000000-0000-0000-0000-000000000000 # Azure app (public client)
|
|
58
|
+
#ONEDRIVE_ROOT_NAME=mokuro-reader # library folder at your OneDrive root
|
|
59
|
+
#ONEDRIVE_TOKEN_FILE=/Users/you/.config/mokuro-bridge/onedrive_token.json
|
|
60
|
+
|
|
61
|
+
# ── OCR engine ──────────────────────────────────────────────────────────
|
|
62
|
+
# Path to a mokuro checkout (optimized fork) instead of the pip package.
|
|
63
|
+
#MOKURO_REPO=/Users/you/mokuro
|
|
64
|
+
#OCR_CHUNK_SIZE=8
|
|
65
|
+
#OCR_IDLE_FLUSH_S=1.5
|
|
66
|
+
# Round-robin mixed-volume batches; set 0 for legacy FIFO/finalize priority.
|
|
67
|
+
#MOKURO_BRIDGE_OCR_FAIR_SCHEDULING=true
|
|
68
|
+
# Fraction of a mixed batch reserved for sessions waiting on finalize (0-1).
|
|
69
|
+
#MOKURO_BRIDGE_OCR_FINALIZE_PRIORITY=0.5
|
|
70
|
+
#MIN_PAGES_FOR_MEGA=10
|
|
71
|
+
|
|
72
|
+
# ── Page-fetch accelerator ─────────────────────────────────────────────
|
|
73
|
+
# Extra localhost ports the downloader can fetch pages through. Chrome allows 6
|
|
74
|
+
# concurrent connections per origin, and an origin includes the port, so each
|
|
75
|
+
# port here is worth 6 more sockets to the browser. 0 disables the feature.
|
|
76
|
+
#MOKURO_BRIDGE_FETCH_PORTS=48
|
|
77
|
+
#MOKURO_BRIDGE_FETCH_CONCURRENCY=0 # 0 = no cap on simultaneous fetches
|
|
78
|
+
#MOKURO_BRIDGE_FETCH_UPSTREAM=https://bw-bv-epubs.bookwalker.jp
|
|
79
|
+
# Comma-separated host patterns the accelerator may forward to. Empty/unset
|
|
80
|
+
# means "*", i.e. any public host: the real gate is the address check, which
|
|
81
|
+
# refuses loopback, private, link-local, multicast and reserved targets (so LAN
|
|
82
|
+
# hosts and 169.254.169.254 are blocked) but allows the public internet. Set it
|
|
83
|
+
# to pin the accelerator to specific CDNs.
|
|
84
|
+
#MOKURO_BRIDGE_FETCH_ALLOWED_HOSTS=bw-bv-epubs.bookwalker.jp
|
|
85
|
+
|
|
86
|
+
# ── Update check ────────────────────────────────────────────────────────
|
|
87
|
+
# The bridge asks the public GitHub releases API whether a newer version
|
|
88
|
+
# exists, and reports it in /health and via `--check-update`. Nothing is ever
|
|
89
|
+
# downloaded or installed automatically. Set 0 on an offline machine; a
|
|
90
|
+
# `--check-update` you run by hand still works then.
|
|
91
|
+
#MOKURO_BRIDGE_UPDATE_CHECK=true
|
|
92
|
+
#MOKURO_BRIDGE_UPDATE_TTL_S=21600 # reuse a result for 6 hours
|
|
93
|
+
#MOKURO_BRIDGE_UPDATE_TIMEOUT_S=5
|
|
94
|
+
|
|
95
|
+
# ── Dev only ────────────────────────────────────────────────────────────
|
|
96
|
+
#UVICORN_RELOAD=1
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 GolyBidoof
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# The wheel is one install path; the git checkout is the other, and the README
|
|
2
|
+
# documents the checkout one in detail (run.sh, the launchd agent, the keychain
|
|
3
|
+
# helper). setuptools builds an sdist from this file, and without it that sdist
|
|
4
|
+
# carries none of those files: the install it describes cannot be reproduced
|
|
5
|
+
# from a published release. Everything here is a file a checkout install needs.
|
|
6
|
+
|
|
7
|
+
# The checkout entry points and their helpers.
|
|
8
|
+
include server.py
|
|
9
|
+
include ocr_folder.py
|
|
10
|
+
include run.sh
|
|
11
|
+
include install-launchd.sh
|
|
12
|
+
include setup-keychain.sh
|
|
13
|
+
include com.mokuro-bridge.plist
|
|
14
|
+
include pytest.ini
|
|
15
|
+
include .env.example
|
|
16
|
+
|
|
17
|
+
# The requirements lists. pyproject is a build input and is included anyway,
|
|
18
|
+
# but the tests assert the two stay in step, so the sdist has to carry them.
|
|
19
|
+
include requirements.txt
|
|
20
|
+
include requirements-dev.txt
|
|
21
|
+
include requirements-ocr.txt
|
|
22
|
+
include requirements-drive.txt
|
|
23
|
+
include requirements-onedrive.txt
|
|
24
|
+
|
|
25
|
+
# The test suite, so `python -m pytest` works from an unpacked sdist.
|
|
26
|
+
recursive-include tests *.py
|
|
27
|
+
|
|
28
|
+
# The docs the README links to. PyPI renders the README but does not rewrite a
|
|
29
|
+
# relative link, so a reader who follows one from the project page needs these
|
|
30
|
+
# to be in the sdist they downloaded.
|
|
31
|
+
recursive-include docs *.md
|
|
32
|
+
|
|
33
|
+
# Never ship the full mokuro fork checkout at ./mokuro. It is gitignored and
|
|
34
|
+
# not part of this distribution, but it is a real directory on a maintainer's
|
|
35
|
+
# disk, and pruning it keeps an sdist built by hand from sweeping it in.
|
|
36
|
+
prune mokuro
|
|
37
|
+
prune output
|
|
38
|
+
prune build
|
|
39
|
+
prune dist
|
|
40
|
+
|
|
41
|
+
global-exclude __pycache__ *.py[cod] .DS_Store
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mokuro-bridge
|
|
3
|
+
Version: 0.7.0
|
|
4
|
+
Summary: Local OCR bridge for manga page captures, producing CBZ + .mokuro + cover for reader.mokuro.app
|
|
5
|
+
Author: GolyBidoof
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/GolyBidoof/mokuro-bridge
|
|
8
|
+
Project-URL: Repository, https://github.com/GolyBidoof/mokuro-bridge
|
|
9
|
+
Project-URL: Issues, https://github.com/GolyBidoof/mokuro-bridge/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/GolyBidoof/mokuro-bridge/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: Releases, https://github.com/GolyBidoof/mokuro-bridge/releases
|
|
12
|
+
Keywords: manga,ocr,mokuro,bookwalker,cbz
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Environment :: Console
|
|
15
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
16
|
+
Classifier: Operating System :: MacOS
|
|
17
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
18
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
25
|
+
Classifier: Topic :: Multimedia :: Graphics :: Capture
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Requires-Dist: fastapi>=0.115.0
|
|
30
|
+
Requires-Dist: uvicorn[standard]>=0.32.0
|
|
31
|
+
Requires-Dist: python-multipart>=0.0.12
|
|
32
|
+
Requires-Dist: httpx>=0.27.0
|
|
33
|
+
Requires-Dist: keyring>=23.13
|
|
34
|
+
Provides-Extra: ocr
|
|
35
|
+
Requires-Dist: mokuro>=0.2.0; extra == "ocr"
|
|
36
|
+
Provides-Extra: drive
|
|
37
|
+
Requires-Dist: google-api-python-client>=2.100.0; extra == "drive"
|
|
38
|
+
Requires-Dist: google-auth-oauthlib>=1.0.0; extra == "drive"
|
|
39
|
+
Requires-Dist: google-auth>=2.0.0; extra == "drive"
|
|
40
|
+
Requires-Dist: requests>=2.20.0; extra == "drive"
|
|
41
|
+
Provides-Extra: onedrive
|
|
42
|
+
Requires-Dist: msal>=1.24.0; extra == "onedrive"
|
|
43
|
+
Requires-Dist: requests>=2.31; extra == "onedrive"
|
|
44
|
+
Provides-Extra: cloud
|
|
45
|
+
Requires-Dist: google-api-python-client>=2.100.0; extra == "cloud"
|
|
46
|
+
Requires-Dist: google-auth-oauthlib>=1.0.0; extra == "cloud"
|
|
47
|
+
Requires-Dist: google-auth>=2.0.0; extra == "cloud"
|
|
48
|
+
Requires-Dist: msal>=1.24.0; extra == "cloud"
|
|
49
|
+
Requires-Dist: requests>=2.31; extra == "cloud"
|
|
50
|
+
Provides-Extra: dev
|
|
51
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
52
|
+
Requires-Dist: tomli>=2.0; python_version < "3.11" and extra == "dev"
|
|
53
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
54
|
+
Dynamic: license-file
|
|
55
|
+
|
|
56
|
+
# mokuro-bridge
|
|
57
|
+
|
|
58
|
+
**It makes your browser downloader up to 48x faster, and then OCRs what it caught.**
|
|
59
|
+
|
|
60
|
+
Two jobs, and you can take either without the other.
|
|
61
|
+
|
|
62
|
+
**1. A capture accelerator for browser clients.** A browser will not open more than
|
|
63
|
+
six connections to one origin, and an origin is scheme + host + **port**. A store's
|
|
64
|
+
page CDN is a single host that refuses HTTP/2, so an in-browser download is pinned
|
|
65
|
+
to six sockets no matter how fast the link is. This bridge opens a range of extra
|
|
66
|
+
localhost ports, each serving the same small proxy. The browser sees each as a fresh
|
|
67
|
+
origin and gets six more. Out of the box that is **6 sockets becomes 288**, with
|
|
68
|
+
nothing to configure.
|
|
69
|
+
|
|
70
|
+
**2. An OCR and delivery back end.** Point it at a folder of page images, or POST
|
|
71
|
+
pages from a capture script, and it runs [mokuro](https://github.com/kha-white/mokuro)
|
|
72
|
+
over them and produces the three files [reader.mokuro.app](https://reader.mokuro.app/) reads:
|
|
73
|
+
|
|
74
|
+
```
|
|
75
|
+
<output>/<series>/
|
|
76
|
+
<volume>.cbz page images
|
|
77
|
+
<volume>.mokuro OCR text + block data
|
|
78
|
+
<volume>.webp cover
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Everything runs on your machine. Output stays local by default; **MEGA, Google Drive
|
|
82
|
+
or OneDrive** are optional. Works on macOS, Windows and Linux, and is built for
|
|
83
|
+
Japanese manga, because the model reads Japanese text.
|
|
84
|
+
|
|
85
|
+
## Why it exists
|
|
86
|
+
|
|
87
|
+
OCR for manga is heavy. The model wants PyTorch, which is several gigabytes, and
|
|
88
|
+
mokuro's own scripts are shaped for a person at a terminal who already has a folder
|
|
89
|
+
of pages. What is missing is the bit in between: something that accepts pages *as a
|
|
90
|
+
download produces them*, recognises them while they are still arriving, and puts
|
|
91
|
+
the result wherever you actually read it.
|
|
92
|
+
|
|
93
|
+
That is the whole job. Hand it pages, get a `.mokuro` back.
|
|
94
|
+
|
|
95
|
+
**It is the OCR half of a capture pipeline, and only that half.** The bridge never
|
|
96
|
+
touches a storefront: it holds no store account, no cookie, no store protocol and no
|
|
97
|
+
store code. A client gets the images out however it likes and POSTs them here. That
|
|
98
|
+
boundary is deliberate -- it is why the bridge does not break when a store changes,
|
|
99
|
+
and why any client that can produce a JPEG can use it.
|
|
100
|
+
|
|
101
|
+
Two working clients:
|
|
102
|
+
|
|
103
|
+
- **[bookwalker-ebookjapan-cmoa-native-downloader](https://github.com/GolyBidoof/bookwalker-ebookjapan-cmoa-native-downloader)**
|
|
104
|
+
-- a userscript that captures pages in the browser and streams them here as it
|
|
105
|
+
descrambles them. It uses **both** halves: the extra ports take its download from
|
|
106
|
+
six sockets to 288, and the bridge OCRs and delivers what it caught.
|
|
107
|
+
- **[dokuha-cli](https://github.com/GolyBidoof/dokuha-cli)** -- a browserless CLI for
|
|
108
|
+
BookWalker, CMOA, ebookjapan, Kindle and k-manga. No browser, so no six-socket
|
|
109
|
+
ceiling to work around; it uses the OCR and delivery half, as two independent
|
|
110
|
+
stages, so a volume waiting on a slow upload never stalls the network behind it.
|
|
111
|
+
|
|
112
|
+
## Six sockets to 288
|
|
113
|
+
|
|
114
|
+
A browser allows six concurrent HTTP/1.1 connections per **origin**, and an origin is
|
|
115
|
+
scheme + host + **port**. A store's page CDN is one host and refuses to negotiate
|
|
116
|
+
HTTP/2, so a viewer download is pinned to six sockets however fast the connection
|
|
117
|
+
is. No amount of page-level concurrency in the downloader gets past that; it is a
|
|
118
|
+
browser rule.
|
|
119
|
+
|
|
120
|
+
Because the port is part of the origin, the bridge opens a range of extra localhost
|
|
121
|
+
ports, each serving the same small proxy. The browser treats every port as a fresh
|
|
122
|
+
origin and so gets six more sockets per port, while the bridge itself does the
|
|
123
|
+
fetching under no browser limit at all.
|
|
124
|
+
|
|
125
|
+
```
|
|
126
|
+
mokuro-bridge v0.7.0 on http://127.0.0.1:62642
|
|
127
|
+
fetch proxy: 48 extra port(s) 63443-63490 -> 288 browser sockets for the downloader
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
There is nothing to configure. The bridge binds what it can and advertises the range
|
|
131
|
+
on `/health` as `fetchProxyPorts`, and the client picks it up on its own. 48 is
|
|
132
|
+
chosen against Chrome's own ceiling: a profile gets roughly 300 sockets in total, so
|
|
133
|
+
48 ports leaves room for the page itself, the userscript and the downloader's
|
|
134
|
+
fallback lanes. `MOKURO_BRIDGE_FETCH_PORTS` raises or lowers it; `0` turns it off.
|
|
135
|
+
|
|
136
|
+
**You do not need the OCR engine to use this.** The accelerator is part of the base
|
|
137
|
+
server and is started regardless of whether mokuro is installed. A client that only
|
|
138
|
+
wants its browser download to stop being connection-bound can run the bridge and
|
|
139
|
+
ignore everything else.
|
|
140
|
+
|
|
141
|
+
**And downloading never depends on the bridge.** With it stopped, a client falls
|
|
142
|
+
back to its own page and background-context lanes. This is an accelerator, not a
|
|
143
|
+
dependency: it makes capture much faster, and nothing breaks without it.
|
|
144
|
+
|
|
145
|
+
One caveat the maintainer should know: each port costs a file descriptor on each
|
|
146
|
+
side of the bridge, so 48 ports at six sockets is roughly 600 at peak. The bridge
|
|
147
|
+
raises `RLIMIT_NOFILE` at startup to cover that, which matters on macOS, where a
|
|
148
|
+
launchd agent starts with a soft limit of 256 and exhausting it shows up as
|
|
149
|
+
`OSError: Too many open files` on `accept()` and ECONNRESET in the browser, with
|
|
150
|
+
nothing pointing at a ulimit. Lower `MOKURO_BRIDGE_FETCH_PORTS` instead if you would
|
|
151
|
+
rather not raise it.
|
|
152
|
+
|
|
153
|
+
Separately, pages are recognised **as they arrive** in chunked batches, so capture
|
|
154
|
+
and OCR overlap rather than running back to back: a 250-page volume is not a
|
|
155
|
+
250-page wait followed by a recognition pass.
|
|
156
|
+
|
|
157
|
+
## Quickstart
|
|
158
|
+
|
|
159
|
+
**1. Install.** A prebuilt wheel with [pipx](https://pipx.pypa.io/) (or `uv tool
|
|
160
|
+
install`) is the shortest route, and keeps the bridge in its own environment:
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
pipx install mokuro-bridge
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
From a checkout instead:
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
git clone https://github.com/GolyBidoof/mokuro-bridge && cd mokuro-bridge
|
|
170
|
+
python3 -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
171
|
+
pip install -r requirements.txt
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
Python 3.10 or newer. The base install is small: `fastapi`, `uvicorn`,
|
|
175
|
+
`python-multipart`, `httpx` and `keyring`. Cloud libraries are opt-in per provider
|
|
176
|
+
and are not installed unless you ask for them.
|
|
177
|
+
|
|
178
|
+
**2. Start it.**
|
|
179
|
+
|
|
180
|
+
```bash
|
|
181
|
+
mokuro-bridge # pipx / uv install
|
|
182
|
+
./run.sh # macOS / Linux, from a checkout
|
|
183
|
+
python server.py # Windows, from a checkout
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
```
|
|
187
|
+
mokuro-bridge v0.7.0 on http://127.0.0.1:62642
|
|
188
|
+
fetch proxy: 48 extra port(s) 63443-63490 -> 288 browser sockets for the downloader
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
**3. OCR pages you already have.** In a second terminal:
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
mokuro-bridge-ocr ./my-volume/ --title 'Volume title'
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
Done. It writes the `.cbz`, the `.mokuro` and the cover, arranged per series, with
|
|
198
|
+
edition suffixes stripped so one series does not split across four folders.
|
|
199
|
+
|
|
200
|
+
To go straight from a download instead, let the client do it:
|
|
201
|
+
|
|
202
|
+
```sh
|
|
203
|
+
dokuha --mokuro 'https://bookwalker.jp/de00000000-0000-4000-8000-000000000001/'
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## The OCR engine: install the fork
|
|
207
|
+
|
|
208
|
+
`mokuro` depends on PyTorch, which is several GB, so it is not in the base
|
|
209
|
+
requirements. The bridge runs without it -- `/health` reports
|
|
210
|
+
`mokuro_installed: false` and only OCR is unavailable. When you want OCR:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
pip install "mokuro @ git+https://github.com/GolyBidoof/mokuro"
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
**That is the recommended engine, and it is one line.** It is
|
|
217
|
+
[GolyBidoof's fork of mokuro](https://github.com/GolyBidoof/mokuro), which adds a
|
|
218
|
+
batched recognition API. On a measured workload it is worth about **1.8x** -- about
|
|
219
|
+
23.5 ms per crop against 43 at one beam versus four, on MPS. Recognition is the
|
|
220
|
+
slowest part of a long volume, so this is the single largest speedup available to
|
|
221
|
+
you after the port trick.
|
|
222
|
+
|
|
223
|
+
Nothing else is needed. The bridge detects the fork's batch API by introspection
|
|
224
|
+
rather than by configuration, so installing it is enough -- there is no environment
|
|
225
|
+
variable to set and no flag to pass. `/health` will report the engine as installed
|
|
226
|
+
and the batch path active.
|
|
227
|
+
|
|
228
|
+
To develop on the fork itself, or to run an uninstalled checkout, set `MOKURO_REPO`
|
|
229
|
+
to its path and the bridge will put it ahead of anything installed:
|
|
230
|
+
|
|
231
|
+
```bash
|
|
232
|
+
git clone https://github.com/GolyBidoof/mokuro
|
|
233
|
+
export MOKURO_REPO="$PWD/mokuro"
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
**Stock mokuro from PyPI still works**, and is the fallback if you would rather not
|
|
237
|
+
track a git dependency:
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
pip install -r requirements-ocr.txt
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
The bridge uses the slower single-crop path with it. Nothing breaks; it is just
|
|
244
|
+
slower.
|
|
245
|
+
|
|
246
|
+
## Deliver it wherever you read
|
|
247
|
+
|
|
248
|
+
Local disk by default. MEGA, Google Drive and OneDrive are opt-in:
|
|
249
|
+
|
|
250
|
+
```bash
|
|
251
|
+
mokuro-bridge --setup-upload mega
|
|
252
|
+
mokuro-bridge --list-uploads
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
**More than one account per provider.** `--setup-upload mega --name work` adds a
|
|
256
|
+
second, addressed as `mega:work`, each with its own remote root and credentials. A
|
|
257
|
+
bare provider id still means the default, so nothing existing breaks.
|
|
258
|
+
|
|
259
|
+
Credentials go to the OS keychain where one exists (macOS Keychain, Windows
|
|
260
|
+
Credential Manager, Linux Secret Service), or to 0600 files under
|
|
261
|
+
`~/.config/mokuro-bridge/`. They are never written to this repository.
|
|
262
|
+
|
|
263
|
+
## Runs unattended
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
./install-launchd.sh # macOS: a launchd agent that starts on login
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
`KeepAlive` with a throttle interval, so it comes back if it dies without
|
|
270
|
+
crash-looping. Nothing is uploaded or updated without you asking: `--check-update`
|
|
271
|
+
reports a newer release and prints the command for how *this* copy was installed,
|
|
272
|
+
and applying it stays your call, because a restart mid-OCR loses work.
|
|
273
|
+
|
|
274
|
+
## Being straight about where it is
|
|
275
|
+
|
|
276
|
+
- **The bridge is unauthenticated.** No token, no login. It binds to `127.0.0.1` and
|
|
277
|
+
that is the entire access control. The intended deployment is loopback-only: do
|
|
278
|
+
not bind it to a LAN address, because anyone who can reach the port can start a
|
|
279
|
+
volume and have the result written or uploaded to your account.
|
|
280
|
+
- **CORS is a wildcard by default**, deliberately, since the client runs on whichever
|
|
281
|
+
storefront you happen to be reading. `CORS_ORIGINS` narrows it.
|
|
282
|
+
- **There is no path traversal.** Page names are validated, the ingest resolver is
|
|
283
|
+
confined to your home and the system temp locations, and uploads are
|
|
284
|
+
extension-checked before anything is read. A caller can drive the bridge; it cannot
|
|
285
|
+
read arbitrary files off your disk.
|
|
286
|
+
- **The page-fetch proxy is not restricted to one CDN by default.** It refuses
|
|
287
|
+
loopback, private and link-local targets, so it cannot reach your network -- but
|
|
288
|
+
any public host is fetchable. `MOKURO_BRIDGE_FETCH_ALLOWED_HOSTS` closes it to one.
|
|
289
|
+
- **The OCR engine lags Python.** PyTorch wheels often trail new releases.
|
|
290
|
+
|
|
291
|
+
## API
|
|
292
|
+
|
|
293
|
+
It is a small HTTP API, and anything that can POST an image can use it:
|
|
294
|
+
|
|
295
|
+
| | |
|
|
296
|
+
| --- | --- |
|
|
297
|
+
| `POST /session/start` | create a session for a volume title |
|
|
298
|
+
| `POST /session/{id}/page` | one page image |
|
|
299
|
+
| `POST /session/{id}/finalize` | run OCR, package, deliver; answers a progress stream |
|
|
300
|
+
| `POST /session/resume` | hand over a folder that already exists on disk |
|
|
301
|
+
| `GET /session/{id}/status` | poll progress |
|
|
302
|
+
| `GET /health` | readiness, engine state, bound ports, update status |
|
|
303
|
+
|
|
304
|
+
`finalize` streams newline-delimited JSON rather than returning one result, so a
|
|
305
|
+
client reads progress off the same connection doing the work. The full surface,
|
|
306
|
+
including the progress frame format and the upload methods, is in
|
|
307
|
+
[docs/architecture.md](https://github.com/GolyBidoof/mokuro-bridge/blob/main/docs/architecture.md).
|
|
308
|
+
|
|
309
|
+
## Development
|
|
310
|
+
|
|
311
|
+
```bash
|
|
312
|
+
pip install -r requirements-dev.txt
|
|
313
|
+
python3 -m pytest # 195 tests, ~10s, hermetic
|
|
314
|
+
python3 -m ruff check .
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
No network, no credentials and no model are needed to run the suite. There is a
|
|
318
|
+
packaging contract test that fails if `pyproject.toml` and the `requirements*.txt`
|
|
319
|
+
files drift apart.
|
|
320
|
+
|
|
321
|
+
## Credits and licence
|
|
322
|
+
|
|
323
|
+
MIT. [LICENSE](https://github.com/GolyBidoof/mokuro-bridge/blob/main/LICENSE)
|
|
324
|
+
|
|
325
|
+
OCR is done by [mokuro](https://github.com/kha-white/mokuro), which carries its own
|
|
326
|
+
licence and is installed separately. Cloud delivery uses the MEGA, Google Drive and
|
|
327
|
+
OneDrive clients, each under its own terms.
|