mwoffliner 1.17.4 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +355 -62
- package/docs/functional_architecture.md +34 -38
- package/extensions/wiktionary_fr.js +18 -17
- package/jest.config.cjs +2 -2
- package/lib/DOMUtils.js +0 -1
- package/lib/DOMUtils.js.map +1 -1
- package/lib/Downloader.d.ts +39 -31
- package/lib/Downloader.js +418 -349
- package/lib/Downloader.js.map +1 -1
- package/lib/Dump.d.ts +24 -10
- package/lib/Dump.js +95 -50
- package/lib/Dump.js.map +1 -1
- package/lib/Gadgets.d.ts +2 -1
- package/lib/Gadgets.js +7 -5
- package/lib/Gadgets.js.map +1 -1
- package/lib/Logger.d.ts +6 -6
- package/lib/Logger.js +33 -33
- package/lib/Logger.js.map +1 -1
- package/lib/MediaWiki.d.ts +14 -31
- package/lib/MediaWiki.js +119 -172
- package/lib/MediaWiki.js.map +1 -1
- package/lib/RedisStore.d.ts +3 -3
- package/lib/RedisStore.js +21 -23
- package/lib/RedisStore.js.map +1 -1
- package/lib/S3.js +1 -1
- package/lib/S3.js.map +1 -1
- package/lib/Templates.d.ts +3 -10
- package/lib/Templates.js +5 -14
- package/lib/Templates.js.map +1 -1
- package/lib/cli.js +7 -6
- package/lib/cli.js.map +1 -1
- package/lib/config.d.ts +10 -16
- package/lib/config.js +33 -55
- package/lib/config.js.map +1 -1
- package/lib/error.manager.d.ts +2 -2
- package/lib/error.manager.js +42 -49
- package/lib/error.manager.js.map +1 -1
- package/lib/i18n.d.ts +2 -0
- package/lib/i18n.js +53 -0
- package/lib/i18n.js.map +1 -0
- package/lib/mutex.d.ts +2 -1
- package/lib/mutex.js +2 -1
- package/lib/mutex.js.map +1 -1
- package/lib/mwoffliner.lib.js +370 -222
- package/lib/mwoffliner.lib.js.map +1 -1
- package/lib/parameterList.d.ts +20 -8
- package/lib/parameterList.js +30 -18
- package/lib/parameterList.js.map +1 -1
- package/lib/renderers/abstract.renderer.d.ts +48 -55
- package/lib/renderers/abstract.renderer.js +377 -98
- package/lib/renderers/abstract.renderer.js.map +1 -1
- package/lib/renderers/action-parse.renderer.d.ts +22 -3
- package/lib/renderers/action-parse.renderer.js +123 -81
- package/lib/renderers/action-parse.renderer.js.map +1 -1
- package/lib/renderers/renderer.builder.js +6 -69
- package/lib/renderers/renderer.builder.js.map +1 -1
- package/lib/renderers/rendering.context.d.ts +2 -3
- package/lib/renderers/rendering.context.js +6 -14
- package/lib/renderers/rendering.context.js.map +1 -1
- package/lib/sanitize-argument.d.ts +7 -4
- package/lib/sanitize-argument.js +99 -41
- package/lib/sanitize-argument.js.map +1 -1
- package/lib/util/FileManager.d.ts +36 -0
- package/lib/util/FileManager.js +246 -0
- package/lib/util/FileManager.js.map +1 -0
- package/lib/util/RedisKvs.js +4 -4
- package/lib/util/RedisKvs.js.map +1 -1
- package/lib/util/RedisQueue.js +1 -1
- package/lib/util/RedisQueue.js.map +1 -1
- package/lib/util/builders/url/action-parse.director.d.ts +2 -2
- package/lib/util/builders/url/action-parse.director.js +7 -7
- package/lib/util/builders/url/action-parse.director.js.map +1 -1
- package/lib/util/builders/url/api.director.d.ts +2 -4
- package/lib/util/builders/url/api.director.js +10 -17
- package/lib/util/builders/url/api.director.js.map +1 -1
- package/lib/util/builders/url/base.director.d.ts +0 -4
- package/lib/util/builders/url/base.director.js +0 -25
- package/lib/util/builders/url/base.director.js.map +1 -1
- package/lib/util/builders/url/basic.director.d.ts +1 -1
- package/lib/util/builders/url/basic.director.js +1 -1
- package/lib/util/builders/url/web.director.d.ts +1 -1
- package/lib/util/builders/url/web.director.js +2 -2
- package/lib/util/builders/url/web.director.js.map +1 -1
- package/lib/util/categories.d.ts +5 -5
- package/lib/util/categories.js +175 -174
- package/lib/util/categories.js.map +1 -1
- package/lib/util/const.d.ts +4 -8
- package/lib/util/const.js +9 -17
- package/lib/util/const.js.map +1 -1
- package/lib/util/customCssJs.d.ts +6 -0
- package/lib/util/customCssJs.js +70 -0
- package/lib/util/customCssJs.js.map +1 -0
- package/lib/util/dump.d.ts +15 -4
- package/lib/util/dump.js +282 -86
- package/lib/util/dump.js.map +1 -1
- package/lib/util/index.d.ts +1 -1
- package/lib/util/index.js +1 -1
- package/lib/util/index.js.map +1 -1
- package/lib/util/metaData.js.map +1 -1
- package/lib/util/misc.d.ts +23 -13
- package/lib/util/misc.js +90 -59
- package/lib/util/misc.js.map +1 -1
- package/lib/util/mw-api.d.ts +7 -5
- package/lib/util/mw-api.js +167 -221
- package/lib/util/mw-api.js.map +1 -1
- package/lib/util/pageListMainPage.d.ts +3 -0
- package/lib/util/pageListMainPage.js +10 -0
- package/lib/util/pageListMainPage.js.map +1 -0
- package/lib/util/pages.d.ts +30 -0
- package/lib/util/pages.js +146 -0
- package/lib/util/pages.js.map +1 -0
- package/lib/util/rewriteUrls.d.ts +2 -2
- package/lib/util/rewriteUrls.js +33 -31
- package/lib/util/rewriteUrls.js.map +1 -1
- package/lib/util/savePages.d.ts +8 -0
- package/lib/util/savePages.js +182 -0
- package/lib/util/savePages.js.map +1 -0
- package/lib/version.d.ts +1 -1
- package/lib/version.js +1 -1
- package/lib/version.js.map +1 -1
- package/offliner-definition.json +178 -90
- package/package.json +50 -49
- package/res/page_list_home.js +20 -0
- package/res/{article_not_found.svg → page_not_found.svg} +1 -1
- package/res/script.js +30 -45
- package/res/style.css +52 -68
- package/res/templates/download_error_placeholder.html +7 -8
- package/res/templates/javaScript.html +19 -0
- package/res/templates/pageFallback.html +9 -7
- package/res/templates/pageVector2022.html +9 -7
- package/res/templates/pageVectorLegacy.html +9 -7
- package/res/templates/{article_list_home.html → page_list_home.html} +1 -2
- package/translation/ar.json +21 -21
- package/translation/bn.json +1 -1
- package/translation/de.json +46 -19
- package/translation/en.json +54 -21
- package/translation/es.json +17 -19
- package/translation/fi.json +23 -3
- package/translation/fr.json +57 -21
- package/translation/he.json +8 -6
- package/translation/ia.json +19 -19
- package/translation/id.json +13 -13
- package/translation/it.json +13 -8
- package/translation/ko.json +55 -3
- package/translation/lb.json +5 -2
- package/translation/mk.json +5 -5
- package/translation/nl.json +57 -22
- package/translation/or.json +2 -2
- package/translation/pt.json +2 -2
- package/translation/qqq.json +55 -22
- package/translation/ru.json +2 -2
- package/translation/sc.json +2 -2
- package/translation/sk.json +62 -0
- package/translation/sl.json +54 -21
- package/translation/sv.json +20 -19
- package/translation/sw.json +45 -2
- package/translation/zh-hans.json +56 -22
- package/translation/zh-hant.json +26 -19
- package/lib/renderers/abstractDesktop.render.d.ts +0 -12
- package/lib/renderers/abstractDesktop.render.js +0 -67
- package/lib/renderers/abstractDesktop.render.js.map +0 -1
- package/lib/renderers/abstractMobile.render.d.ts +0 -12
- package/lib/renderers/abstractMobile.render.js +0 -54
- package/lib/renderers/abstractMobile.render.js.map +0 -1
- package/lib/renderers/rest-api.renderer.d.ts +0 -7
- package/lib/renderers/rest-api.renderer.js +0 -68
- package/lib/renderers/rest-api.renderer.js.map +0 -1
- package/lib/renderers/visual-editor.renderer.d.ts +0 -7
- package/lib/renderers/visual-editor.renderer.js +0 -76
- package/lib/renderers/visual-editor.renderer.js.map +0 -1
- package/lib/renderers/wikimedia-desktop.renderer.d.ts +0 -4
- package/lib/renderers/wikimedia-desktop.renderer.js +0 -7
- package/lib/renderers/wikimedia-desktop.renderer.js.map +0 -1
- package/lib/renderers/wikimedia-mobile.renderer.d.ts +0 -19
- package/lib/renderers/wikimedia-mobile.renderer.js +0 -243
- package/lib/renderers/wikimedia-mobile.renderer.js.map +0 -1
- package/lib/util/articleListMainPage.d.ts +0 -3
- package/lib/util/articleListMainPage.js +0 -10
- package/lib/util/articleListMainPage.js.map +0 -1
- package/lib/util/articles.d.ts +0 -26
- package/lib/util/articles.js +0 -102
- package/lib/util/articles.js.map +0 -1
- package/lib/util/builders/url/desktop.director.d.ts +0 -8
- package/lib/util/builders/url/desktop.director.js +0 -15
- package/lib/util/builders/url/desktop.director.js.map +0 -1
- package/lib/util/builders/url/mobile.director.d.ts +0 -8
- package/lib/util/builders/url/mobile.director.js +0 -15
- package/lib/util/builders/url/mobile.director.js.map +0 -1
- package/lib/util/builders/url/rest-api.director.d.ts +0 -8
- package/lib/util/builders/url/rest-api.director.js +0 -18
- package/lib/util/builders/url/rest-api.director.js.map +0 -1
- package/lib/util/builders/url/visual-editor.director.d.ts +0 -9
- package/lib/util/builders/url/visual-editor.director.js +0 -18
- package/lib/util/builders/url/visual-editor.director.js.map +0 -1
- package/lib/util/saveArticles.d.ts +0 -8
- package/lib/util/saveArticles.js +0 -411
- package/lib/util/saveArticles.js.map +0 -1
- package/res/article_list_home.js +0 -24
- package/res/content.parsoid.css +0 -196
- package/res/inserted_style.css +0 -45
- package/res/templates/categories.html +0 -11
- package/res/templates/lead_section_wrapper.html +0 -6
- package/res/templates/pageWikimediaDesktop.html +0 -24
- package/res/templates/pageWikimediaMobile.html +0 -24
- package/res/templates/section_wrapper.html +0 -5
- package/res/templates/subcategories.html +0 -23
- package/res/templates/subpages.html +0 -17
- package/res/templates/subsection_wrapper.html +0 -5
- package/res/webpHandler.js +0 -165
- package/res/wm_mobile_override_script.js +0 -15
- package/res/wm_mobile_override_style.css +0 -20
- package/translation/br.json +0 -8
- package/translation/dag.json +0 -9
- package/translation/es-formal.json +0 -29
- package/translation/ha.json +0 -10
- package/translation/hi.json +0 -9
- package/translation/ig.json +0 -10
- package/translation/kaa.json +0 -9
- package/translation/nb.json +0 -9
- package/translation/nqo.json +0 -8
- package/translation/pt-br.json +0 -10
- package/translation/ro.json +0 -9
- package/translation/scn.json +0 -8
- package/translation/sq.json +0 -9
- package/translation/te.json +0 -9
- package/translation/tn.json +0 -9
- package/translation/tr.json +0 -10
package/README.md
CHANGED
|
@@ -1,19 +1,10 @@
|
|
|
1
1
|
# MWoffliner
|
|
2
2
|
|
|
3
|
-
MWoffliner is a tool for
|
|
4
|
-
online [MediaWiki](https://mediawiki.org) instance. It goes through
|
|
5
|
-
all online articles (or a selection if specified) and create the
|
|
6
|
-
corresponding [ZIM](https://openzim.org) file. It has mainly been
|
|
7
|
-
tested against Wikimedia projects like
|
|
8
|
-
[Wikipedia](https://wikipedia.org) and
|
|
9
|
-
[Wiktionary](https://wiktionary.org) --- but it should also work for
|
|
10
|
-
any recent MediaWiki.
|
|
3
|
+
MWoffliner is a tool for creating a local offline HTML snapshot of any online [MediaWiki](https://mediawiki.org) instance. It scrapes all pages (or a selection if specified) and creates the corresponding [ZIM](https://openzim.org) file. While primarily targeted for Wikimedia projects like [Wikipedia](https://wikipedia.org) and [Wiktionary](https://wiktionary.org), MWoffliner also supports any recent MediaWiki instance (version 1.27+), though instances with custom skins or highly unusual configurations may have limitations.
|
|
11
4
|
|
|
12
|
-
Read [CONTRIBUTING.md](./CONTRIBUTING.md) to
|
|
13
|
-
MWoffliner development.
|
|
5
|
+
Read [CONTRIBUTING.md](./CONTRIBUTING.md) to learn more about MWoffliner development.
|
|
14
6
|
|
|
15
|
-
User
|
|
16
|
-
[FAQ](https://github.com/openzim/mwoffliner/wiki/Frequently-Asked-Questions).
|
|
7
|
+
User help is available in the [FAQ](https://github.com/openzim/mwoffliner/wiki/Frequently-Asked-Questions).
|
|
17
8
|
|
|
18
9
|
[](https://www.npmjs.com/package/mwoffliner)
|
|
19
10
|
|
|
@@ -28,53 +19,282 @@ User Help is available in the for a a
|
|
|
28
19
|
|
|
29
20
|
## Features
|
|
30
21
|
|
|
31
|
-
- Scrape with or without image
|
|
22
|
+
- Scrape with or without image thumbnails
|
|
32
23
|
- Scrape with or without audio/video multimedia content
|
|
33
24
|
- S3 cache (optional)
|
|
34
|
-
- Image size
|
|
35
|
-
- Scrape all
|
|
25
|
+
- Image size optimization and WebP conversion
|
|
26
|
+
- Scrape all pages in namespaces or title list based
|
|
36
27
|
- Specify additional/non-main namespaces to scrape
|
|
37
28
|
|
|
38
|
-
Run `mwoffliner --help` to
|
|
29
|
+
Run `mwoffliner --help` to see all available options.
|
|
39
30
|
|
|
40
|
-
##
|
|
31
|
+
## Quick Start
|
|
41
32
|
|
|
42
|
-
|
|
43
|
-
- [Redis](https://redis.io/)
|
|
44
|
-
- [NodeJS](https://nodejs.org/en/) version 24 (we support only one single Node.JS version, other versions might work or not)
|
|
45
|
-
- [Libzim](https://github.com/openzim/libzim) (On GNU/Linux & macOS we automatically download it)
|
|
46
|
-
- Various build tools which are probably already installed on your
|
|
47
|
-
machine (packages `libjpeg-dev`, `libglu1`, `autoconf`, `automake`, `gcc` on
|
|
48
|
-
Debian/Ubuntu)
|
|
33
|
+
### Prerequisites
|
|
49
34
|
|
|
50
|
-
|
|
35
|
+
- [Docker](https://docs.docker.com/engine/install/) (or Docker-based engine)
|
|
36
|
+
- amd64 or arm64 architecture
|
|
51
37
|
|
|
52
|
-
|
|
38
|
+
To turn the Bambara Wikipedia into an offline ZIM:
|
|
53
39
|
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
40
|
+
```sh
|
|
41
|
+
mkdir -p output
|
|
42
|
+
docker run -v $(pwd)/output:/output ghcr.io/openzim/mwoffliner \
|
|
43
|
+
mwoffliner --mwUrl=https://bm.wikipedia.org --adminEmail=you@example.com \
|
|
44
|
+
--outputDirectory=/output
|
|
57
45
|
```
|
|
58
46
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
how your `npm` / OS is configured. `npm` permission checking can be a bit annoying for a
|
|
62
|
-
newcomer. Please read the documentation carefully if you hit problems: https://docs.npmjs.com/cli/v7/using-npm/scripts#user
|
|
47
|
+
**NOTE**: In order to avoid [429 responses](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/429), be sure to use an
|
|
48
|
+
actual email address as dummy addresses like `you@example.com` will be flagged by the Wiki servers.
|
|
63
49
|
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
50
|
+
That's it. When it finishes you'll find a `.zim` file in the `output/` directory. Open it with
|
|
51
|
+
[Kiwix Reader](https://get.kiwix.org/en/solutions/applications/kiwix-reader/).
|
|
52
|
+
|
|
53
|
+
### The Four Decisions
|
|
54
|
+
|
|
55
|
+
Every scrape comes down to four choices:
|
|
56
|
+
|
|
57
|
+
#### 1. Which wiki?
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
--mwUrl=https://en.wikipedia.org # English Wikipedia
|
|
61
|
+
--mwUrl=https://fr.wiktionary.org # French Wiktionary
|
|
62
|
+
--mwUrl=https://terraria.wiki.gg # Terraria Wiki
|
|
63
|
+
--mwUrl=https://proofwiki.org # ProofWiki
|
|
64
|
+
--mwUrl=https://wiki.my-org.com # Any MediaWiki site
|
|
67
65
|
```
|
|
68
66
|
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
67
|
+
#### 2. Which API path?
|
|
68
|
+
|
|
69
|
+
Most MediaWiki sites use the default API path `/w/api.php`, but many
|
|
70
|
+
don't. Check by visiting `<wiki-url>/w/api.php` in your browser. If that returns a 404, you need to set `--mwActionApiPath` to the correct path.
|
|
71
|
+
|
|
72
|
+
> [!TIP]
|
|
73
|
+
> Try `https://terraria.wiki.gg/api.php` to see a wiki where
|
|
74
|
+
> `--mwActionApiPath=/api.php` is required.
|
|
75
|
+
|
|
76
|
+
#### 3. What content?
|
|
77
|
+
|
|
78
|
+
| If you want… | Add this |
|
|
79
|
+
| --------------------------------------- | ------------------------------- |
|
|
80
|
+
| Everything (text, images, audio, video) | (nothing — this is the default) |
|
|
81
|
+
| Everything except video and audio | `--format=novid:maxi` |
|
|
82
|
+
| No pictures, no video, no audio | `--format=nopic:nopic` |
|
|
83
|
+
| Head paragraphs only, no media at all | `--format=nodet,nopic:mini` |
|
|
84
|
+
|
|
85
|
+
#### 4. Where does it go?
|
|
86
|
+
|
|
87
|
+
The ZIM is written to `/output` inside the container. Map that to a folder on your machine:
|
|
88
|
+
|
|
89
|
+
```sh
|
|
90
|
+
docker run -v /path/on/my/machine:/output ghcr.io/openzim/mwoffliner ...
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
**NOTE**: `--adminEmail=you@example.com` is also required. It is included in the HTTP
|
|
94
|
+
User-Agent so wiki operators know who is scraping.
|
|
95
|
+
|
|
96
|
+
### A Few More Things You Might Want
|
|
97
|
+
|
|
98
|
+
#### Scrape only specific pages
|
|
99
|
+
|
|
100
|
+
Pass a comma-separated list of page titles directly:
|
|
101
|
+
|
|
102
|
+
```sh
|
|
103
|
+
--pageList="Main Page,Earth,Albert Einstein"
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Or put one page title per line in a text file and point to it:
|
|
107
|
+
|
|
108
|
+
```sh
|
|
109
|
+
--pageList=./my-pages.txt
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
#### Scrape a private wiki
|
|
113
|
+
|
|
114
|
+
Provide credentials with `--mwUsername` and `--mwPassword`:
|
|
115
|
+
|
|
116
|
+
```sh
|
|
117
|
+
--mwUsername=jdoe --mwPassword=s3cret
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Preferably use a [bot password](https://www.mediawiki.org/wiki/Manual:Bot_passwords)
|
|
121
|
+
rather than a regular user account.
|
|
122
|
+
|
|
123
|
+
If authentication requires a separate domain, also pass `--mwDomain`:
|
|
124
|
+
|
|
125
|
+
```sh
|
|
126
|
+
--mwDomain=corp --mwUsername=jdoe --mwPassword=s3cret
|
|
73
127
|
```
|
|
74
128
|
|
|
129
|
+
#### Customising the Result
|
|
130
|
+
|
|
131
|
+
Want your ZIM to have a specific title, description, or icon?
|
|
132
|
+
|
|
133
|
+
```sh
|
|
134
|
+
--customZimTitle="My Offline Wiki"
|
|
135
|
+
--customZimDescription="A hand-picked selection of articles"
|
|
136
|
+
--customZimFavicon=https://example.com/icon.png
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
#### Adjusting the scrape speed
|
|
140
|
+
|
|
141
|
+
Since version 2.0.0, the default request rate (speed `1`) is fine for most
|
|
142
|
+
wikis. The `--speed` option controls the climb rate — how aggressively
|
|
143
|
+
mwoffliner ramps up its request concurrency:
|
|
144
|
+
|
|
145
|
+
If you see lots of HTTP errors (e.g. 429 Too Many Requests) in the logs,
|
|
146
|
+
try lowering the speed e.g `--speed=0.5` can help prevent
|
|
147
|
+
the wiki from rate-limiting you.
|
|
148
|
+
|
|
149
|
+
#### Going Further
|
|
150
|
+
|
|
151
|
+
These and all other options are listed in `mwoffliner --help`.
|
|
152
|
+
|
|
153
|
+
Also, see the [FAQ](https://github.com/openzim/mwoffliner/wiki/Frequently-Asked-Questions) for detailed explanations of
|
|
154
|
+
the command line options and common issues.
|
|
155
|
+
|
|
156
|
+
Need help? [](https://slack.kiwix.org)
|
|
157
|
+
|
|
158
|
+
## Installation
|
|
159
|
+
|
|
160
|
+
The recommended way to install and run `mwoffliner` is using the pre-built Docker container:
|
|
161
|
+
|
|
162
|
+
```sh
|
|
163
|
+
docker pull ghcr.io/openzim/mwoffliner
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
<details>
|
|
167
|
+
<summary>Run software locally / Build from source</summary>
|
|
168
|
+
|
|
169
|
+
### Prerequisites for local execution
|
|
170
|
+
|
|
171
|
+
- \*NIX Operating System (GNU/Linux, macOS, etc.)
|
|
172
|
+
- [Redis](https://redis.io/) — in-memory data store
|
|
173
|
+
- [Node.js](https://nodejs.org/en/) version 24 (we support only one single Node.js version; other versions might work or might not)
|
|
174
|
+
- [Libzim](https://github.com/openzim/libzim) — C++ library for creating ZIM files (automatically downloaded on GNU/Linux & macOS)
|
|
175
|
+
- Various build tools which are probably already installed on your machine:
|
|
176
|
+
- `libjpeg-dev` — JPEG image processing
|
|
177
|
+
- `libglu1` — OpenGL utility library
|
|
178
|
+
- `autoconf` — automatic configuration system
|
|
179
|
+
- `automake` — Makefile generator
|
|
180
|
+
- `gcc` — C compiler
|
|
181
|
+
|
|
182
|
+
(These packages are for Debian/Ubuntu systems)
|
|
183
|
+
|
|
184
|
+
An online [MediaWiki](https://mediawiki.org) instance with its API available.
|
|
185
|
+
|
|
186
|
+
### Installation methods
|
|
187
|
+
|
|
188
|
+
#### Build your own container
|
|
189
|
+
|
|
190
|
+
1. Clone the repository locally:
|
|
191
|
+
|
|
192
|
+
```sh
|
|
193
|
+
git clone https://github.com/openzim/mwoffliner.git && cd mwoffliner
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
1. Build the image:
|
|
197
|
+
|
|
198
|
+
```sh
|
|
199
|
+
docker build . -f docker/Dockerfile -t ghcr.io/openzim/mwoffliner
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
#### Run the software locally using NPM
|
|
203
|
+
|
|
204
|
+
> [!WARNING]
|
|
205
|
+
> Local installation requires several system dependencies (see above). Using the Docker image is strongly recommended to avoid setup issues.
|
|
206
|
+
|
|
207
|
+
Setting up MWoffliner locally for development can be tricky due to several dependencies and version requirements. Follow these steps carefully to avoid common errors.
|
|
208
|
+
|
|
209
|
+
##### 1. Node.js Version
|
|
210
|
+
|
|
211
|
+
MWoffliner requires Node.js 24 (other versions may fail).
|
|
212
|
+
|
|
213
|
+
Compatible Node 24 ranges: `>=24 <24.6` or `>=24.7 <25`.
|
|
214
|
+
|
|
215
|
+
Check your version:
|
|
216
|
+
|
|
217
|
+
```sh
|
|
218
|
+
node -v
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
If your version does not match, use [nvm](https://github.com/nvm-sh/nvm) to install the correct Node.js version.
|
|
222
|
+
|
|
223
|
+
##### 2. libzim Dependency
|
|
224
|
+
|
|
225
|
+
MWoffliner depends on [`@openzim/libzim`](https://github.com/openzim/libzim), which requires the C++ libzim library.
|
|
226
|
+
|
|
227
|
+
- On Linux/macOS, MWoffliner can download libzim automatically.
|
|
228
|
+
- On Windows, you must install libzim manually because there are no prebuilt binaries. See the [libzim installation guide](https://github.com/openzim/libzim) for details.
|
|
229
|
+
|
|
230
|
+
##### 3. Compiler Requirements (Windows)
|
|
231
|
+
|
|
232
|
+
Node 24 on Windows officially supports [Visual Studio 2019 (v16)](https://visualstudio.microsoft.com/vs/older-downloads/) or [Visual Studio 2022 (v17)](https://visualstudio.microsoft.com/downloads/).
|
|
233
|
+
|
|
234
|
+
Ensure C++ build tools are installed and environment variables are set correctly. See [Windows Setup for node-gyp](https://github.com/nodejs/node-gyp#on-windows) for detailed instructions.
|
|
235
|
+
|
|
236
|
+
##### 4. Node-gyp
|
|
237
|
+
|
|
238
|
+
MWoffliner uses [node-gyp](https://github.com/nodejs/node-gyp), which enforces strict checks for Node and compiler versions. Make sure you have:
|
|
239
|
+
|
|
240
|
+
- Proper Visual Studio version (Windows) — see [Visual Studio versions](https://visualstudio.microsoft.com/downloads/)
|
|
241
|
+
- Required C++ headers, e.g., `zim/archive.h` — see [libzim documentation](https://github.com/openzim/libzim)
|
|
242
|
+
- [Python 3.10+](https://www.python.org/downloads/) (required by node-gyp; a recent version is preferred for compatibility)
|
|
243
|
+
|
|
244
|
+
##### Additional troubleshooting steps if errors persist:
|
|
245
|
+
|
|
246
|
+
1. **Clear npm cache** — a corrupted cache can cause cryptic install failures:
|
|
247
|
+
|
|
248
|
+
```sh
|
|
249
|
+
npm cache clean --force
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
2. **Delete node_modules and reinstall** — stale or partially installed dependencies are a common source of errors:
|
|
253
|
+
|
|
254
|
+
```sh
|
|
255
|
+
rm -rf node_modules package-lock.json
|
|
256
|
+
npm install
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
3. **Check that all environment variables are set** — especially on Windows, `PATH`, `INCLUDE`, and `LIB` must point to the correct Visual Studio and libzim directories. Reopen your terminal after installing new tools.
|
|
260
|
+
|
|
261
|
+
4. **Verify Redis is running before starting MWoffliner** — MWoffliner will fail immediately if it cannot connect to Redis:
|
|
262
|
+
|
|
263
|
+
```sh
|
|
264
|
+
redis-cli ping # expected output: PONG
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
5. **Run npm install with verbose logging** to see exactly where it fails:
|
|
268
|
+
```sh
|
|
269
|
+
npm install --verbose
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
##### 5. Common Errors & Troubleshooting
|
|
273
|
+
|
|
274
|
+
| Error | Cause | Solution |
|
|
275
|
+
| ---------------------------------- | ------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
276
|
+
| Node.js version error | Node.js version incompatible | Install [Node 24 with nvm](https://github.com/nvm-sh/nvm) |
|
|
277
|
+
| Cannot find module @openzim/libzim | libzim not installed | Follow [libzim installation guide](https://github.com/openzim/libzim); Windows users must install manually |
|
|
278
|
+
| node-gyp rebuild failed | Wrong Node or compiler version | Check [Node.js version](https://nodejs.org/en/), [Visual Studio version](https://visualstudio.microsoft.com/downloads/), [Python 3.x](https://www.python.org/downloads/) |
|
|
279
|
+
| zim/archive.h not found | C++ headers missing | Install [libzim](https://github.com/openzim/libzim) system-wide, verify include paths |
|
|
280
|
+
|
|
281
|
+
> [!NOTE]
|
|
282
|
+
> Even with these steps, other setup errors may occur. Using Docker is strongly recommended for a smoother experience.
|
|
283
|
+
|
|
284
|
+
##### Installation via NPM
|
|
285
|
+
|
|
286
|
+
```sh
|
|
287
|
+
npm i -g mwoffliner
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
> [!WARNING]
|
|
291
|
+
> You might need to run this command with the `sudo` command, depending on how your `npm` / OS is configured. `npm` permission checking can be a bit annoying for newcomers. Please read the [npm script documentation](https://docs.npmjs.com/cli/v7/using-npm/scripts#user) if you encounter issues.
|
|
292
|
+
|
|
293
|
+
</details>
|
|
294
|
+
|
|
75
295
|
## Contribute
|
|
76
296
|
|
|
77
|
-
If you've retrieved
|
|
297
|
+
If you've retrieved the MWoffliner source code (e.g., via a git clone), you can install and run it locally with your modifications:
|
|
78
298
|
|
|
79
299
|
```bash
|
|
80
300
|
npm i
|
|
@@ -85,35 +305,108 @@ Detailed [contribution documentation and guidelines](CONTRIBUTING.md) are availa
|
|
|
85
305
|
|
|
86
306
|
## API
|
|
87
307
|
|
|
88
|
-
MWoffliner provides
|
|
89
|
-
|
|
308
|
+
MWoffliner provides an API and can be used as a Node.js library. Here's a stub example for your `index.mjs` file:
|
|
309
|
+
|
|
90
310
|
```javascript
|
|
91
|
-
import * as mwoffliner from 'mwoffliner'
|
|
311
|
+
import * as mwoffliner from 'mwoffliner'
|
|
92
312
|
|
|
93
313
|
const parameters = {
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
}
|
|
100
|
-
|
|
314
|
+
mwUrl: 'https://es.wikipedia.org',
|
|
315
|
+
adminEmail: 'foo@bar.net',
|
|
316
|
+
verbose: true,
|
|
317
|
+
format: 'nopic',
|
|
318
|
+
pageList: './pageList',
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
mwoffliner.execute(parameters) // returns a Promise
|
|
322
|
+
```
|
|
323
|
+
|
|
324
|
+
## MathJax support
|
|
325
|
+
|
|
326
|
+
> [!WARNING]
|
|
327
|
+
> MathJax support is UNSTABLE: the `--mathJax*` CLI parameters described below may change, even in a minor release. Setting them up also requires wiki-specific technical preparation, so this is aimed at developers rather than end users.
|
|
328
|
+
|
|
329
|
+
MWoffliner can bundle [MathJax](https://www.mathjax.org) into the ZIM so that math formulas keep rendering offline, on wikis that rely on it (e.g. via the `SimpleMathJax` extension).
|
|
330
|
+
|
|
331
|
+
MathJax 2, 3 and 4 are very different beasts: each ships its own set of files and, more importantly, each requires its own incompatible configuration format (which defines things like math delimiters and custom macros). Because of this, MWoffliner cannot auto-detect and configure MathJax for you — you have to supply a matching MathJax build and configuration yourself, via three CLI parameters:
|
|
332
|
+
|
|
333
|
+
- `--mathJaxSource`: local path or HTTP(S) URL to a ZIP archive of a compiled MathJax distribution. Its content is extracted and pushed to the ZIM (under an internal `_mathjax_/` namespace).
|
|
334
|
+
- `--mathJaxConfig`: local path or HTTP(S) URL to an HTML file containing a single `<script>` tag with the MathJax configuration (this must be copied from the wiki, see below). Its content is injected inline, before the MathJax library, on every page that needs it. If the configuration needs to reference a path inside the MathJax archive (e.g. `MathJax.Ajax.config.path[...]`), it cannot use an absolute path since ZIMs have no fixed root URL. Write `__MATHJAX_ROOT__` instead of the leading slash and mwoffliner will replace it, on every page, with the correct relative path to the root of the MathJax archive you provided. For instance use `MathJax.Ajax.config.path["Contrib"] = "__MATHJAX_ROOT__/MathJaxExtensions/legacy";` if the resources are located in a `MathJaxExtensions/legacy` folder in the archive (inside the root folder of the archive if there is a single folder as usual).
|
|
335
|
+
- `--mathJaxEntryPoint`: path, relative to the root of the extracted archive, to the MathJax script to load (e.g. `es5/tex-chtml.js` for MathJax 3). Defaults to `MathJax.js` (MathJax 2 entry point).
|
|
336
|
+
- `--mathJaxAllPages`: inject the config/entry-point `<script>` tags on every page instead of only on pages detected to need MathJax (see below). Some wikis don't list MathJax in their page JS modules, which defeats the automatic detection; this flag works around that at the cost of adding the scripts to every page.
|
|
337
|
+
|
|
338
|
+
`--mathJaxConfig`, `--mathJaxEntryPoint` and `--mathJaxAllPages` all require `--mathJaxSource` to also be set. A page is considered to "need" MathJax, and only then gets the config/entry-point `<script>` tags injected, when one of the JS modules MediaWiki reports for that page matches `mathjax` (case-insensitive), unless `--mathJaxAllPages` is set, in which case every page gets them. The extracted library files themselves are always pushed to the ZIM as soon as `--mathJaxSource` is set, regardless of which pages use them.
|
|
339
|
+
|
|
340
|
+
### Preparing the parameters for a given wiki
|
|
341
|
+
|
|
342
|
+
1. **Find the live MathJax version and configuration.** Open a wiki page that renders math formulas, open your browser's developer console and type `MathJax.version` to get the exact version. Then find the configuration, typically a `<script>` block setting `window.MathJax = {...}` (MathJax 3/4) or `MathJax.Hub.Config({...})` (MathJax 2) in the page source — copy it as-is into a local file, e.g. `mathjax-config.html`.
|
|
343
|
+
2. **Find the exact entry point.** In your browser's network tab, find the request loading the MathJax library itself (typically named `MathJax.js` for MathJax 2, or a `tex-chtml.js`/`tex-svg.js`/... for MathJax 3/4) and note its full path, including any query string.
|
|
344
|
+
3. **Build a matching MathJax ZIP**, the exact steps depend on the major version in use (see below).
|
|
345
|
+
4. **Run the scraper** with the three parameters, e.g.:
|
|
346
|
+
```sh
|
|
347
|
+
mwoffliner --mwUrl=https://your.wiki --adminEmail=foo@bar.net \
|
|
348
|
+
--mathJaxSource=./mathjax-source.zip \
|
|
349
|
+
--mathJaxConfig=./mathjax-config.html \
|
|
350
|
+
--mathJaxEntryPoint=es5/tex-chtml.js
|
|
351
|
+
```
|
|
352
|
+
|
|
353
|
+
### Building the ZIP — MathJax 2
|
|
354
|
+
|
|
355
|
+
MathJax 2 is not published to npm as an installable package; it is only distributed as source on GitHub. Download a release archive directly from the [MathJax releases page](https://github.com/mathjax/MathJax/releases) (e.g. `2.7.9`) and use it as-is as `--mathJaxSource`, no re-zipping needed:
|
|
356
|
+
|
|
357
|
+
```sh
|
|
358
|
+
curl -Lo mathjax-source.zip https://github.com/mathjax/MathJax/archive/refs/tags/2.7.9.zip
|
|
101
359
|
```
|
|
102
360
|
|
|
361
|
+
The archive has a single top-level folder (e.g. `MathJax-2.7.9/`); mwoffliner strips it automatically when extracting.
|
|
362
|
+
|
|
363
|
+
Unlike MathJax 3/4, MathJax 2 loads its extensions/output-processor via a `config=` query parameter on the `MathJax.js` request itself (e.g. `MathJax.js?config=TeX-MML-AM_CHTML`) rather than solely through the injected configuration script — this is the URL you captured in step 2 above. Set `--mathJaxEntryPoint` to that same path and query string, e.g.:
|
|
364
|
+
|
|
365
|
+
```sh
|
|
366
|
+
--mathJaxEntryPoint="MathJax.js?config=TeX-MML-AM_CHTML"
|
|
367
|
+
```
|
|
368
|
+
|
|
369
|
+
The referenced combined-configuration file (here `config/TeX-MML-AM_CHTML.js`) is part of the standard MathJax 2 distribution, so it is already included in the ZIP from the release archive.
|
|
370
|
+
|
|
371
|
+
### Building the ZIP — MathJax 3
|
|
372
|
+
|
|
373
|
+
MathJax 3 is published to npm as `mathjax-full`, which bundles both the compiled runtime (under `es5/`) and the TypeScript sources (which mwoffliner automatically ignores when an `es5/` folder is present):
|
|
374
|
+
|
|
375
|
+
```sh
|
|
376
|
+
npm install mathjax-full@3.2.2
|
|
377
|
+
cd node_modules/mathjax-full && zip -r ../../mathjax-source.zip . && cd ../..
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
Entry point example: `--mathJaxEntryPoint=es5/tex-chtml.js`.
|
|
381
|
+
|
|
382
|
+
### Building the ZIP — MathJax 4
|
|
383
|
+
|
|
384
|
+
MathJax 4 moved to scoped npm packages. Use the plain `mathjax` package (deployment-ready bundle), **not** `@mathjax/src` (the TypeScript source package meant for building MathJax itself, which requires separately installing font packages):
|
|
385
|
+
|
|
386
|
+
```sh
|
|
387
|
+
npm install mathjax@4
|
|
388
|
+
cd node_modules/mathjax && zip -r ../../mathjax-source.zip . && cd ../..
|
|
389
|
+
```
|
|
390
|
+
|
|
391
|
+
Unlike MathJax 3, entry-point files live at the root of the package (no `es5/` folder), e.g. `--mathJaxEntryPoint=tex-chtml.js`.
|
|
392
|
+
|
|
103
393
|
## Background
|
|
104
394
|
|
|
105
395
|
Complementary information about MWoffliner:
|
|
106
396
|
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
397
|
+
- **MediaWiki software** is used by thousands of wikis, the most famous ones being the Wikimedia ones, including [Wikipedia](https://wikipedia.org).
|
|
398
|
+
- **MediaWiki** is a PHP wiki runtime engine.
|
|
399
|
+
- **Wikitext** is the markup language that MediaWiki uses.
|
|
400
|
+
- **MediaWiki parser** converts Wikitext to HTML, which displays in your browser.
|
|
401
|
+
- Read the [scraper functional architecture](docs/functional_architecture.md) for more details.
|
|
402
|
+
|
|
403
|
+
## License
|
|
404
|
+
|
|
405
|
+
[GPLv3](https://www.gnu.org/licenses/gpl-3.0) or later, see [LICENSE](LICENSE) for more details.
|
|
406
|
+
|
|
407
|
+
## Acknowledgements
|
|
114
408
|
|
|
115
|
-
|
|
116
|
-
-------
|
|
409
|
+
This project received funding through [NGI Zero Core](https://nlnet.nl/core), a fund established by [NLnet](https://nlnet.nl/) with financial support from the European Commission's [Next Generation Internet](https://ngi.eu/) program. Learn more at the [NLnet project page](https://nlnet.nl/project/MWOffliner).
|
|
117
410
|
|
|
118
|
-
[
|
|
119
|
-
[
|
|
411
|
+
[<img width="20%" alt="NLnet foundation logo" src="https://github.com/user-attachments/assets/22233242-ec49-4540-a0af-b70725cedbee" />](https://nlnet.nl/)
|
|
412
|
+
[<img width="20%" alt="NGI Zero Logo" src="https://github.com/user-attachments/assets/1bbbda57-dc6f-4902-ae29-236e5e89228f" />](https://nlnet.nl/core)
|
|
@@ -5,11 +5,11 @@ This document describes a high-level overview of how mwoffliner scraper works.
|
|
|
5
5
|
At a high level, mwoffliner is divided into following sequence of actions.
|
|
6
6
|
|
|
7
7
|
- retrieve Mediawiki info
|
|
8
|
-
- retrieve list of
|
|
9
|
-
- for every
|
|
10
|
-
- retrieve its parsed HTML (Wikitext transformed into HTML) and JS/CSS
|
|
8
|
+
- retrieve list of pages to include and their metadata
|
|
9
|
+
- for every page:
|
|
10
|
+
- retrieve its parsed HTML (Wikitext transformed into HTML) and JS/CSS dependencies
|
|
11
11
|
- adapt / render it for proper operation within the ZIM file (includes detection of media dependencies)
|
|
12
|
-
- save rendered
|
|
12
|
+
- save rendered page HTML into the ZIM
|
|
13
13
|
- for every file dependency (JS/CSS/media)
|
|
14
14
|
- if its an image, download it either from S3 cache (images only) or from online and recompress when possible
|
|
15
15
|
- otherwise download it from online
|
|
@@ -17,7 +17,17 @@ At a high level, mwoffliner is divided into following sequence of actions.
|
|
|
17
17
|
|
|
18
18
|
The scraper supports flavours, which are variants of the ZIM (e.g. without images, with images but without videos, ...).
|
|
19
19
|
|
|
20
|
-
For now, retrieval of
|
|
20
|
+
For now, retrieval of pages and files dependencies is repeated for every flavour requested (even if content probably didn't changed).
|
|
21
|
+
|
|
22
|
+
## Lexicography
|
|
23
|
+
|
|
24
|
+
> [!NOTE]
|
|
25
|
+
> All these concepts have been clarified in mwoffliner 2.0.0.
|
|
26
|
+
|
|
27
|
+
- Page: Base object of Mediawikis. Could be an article. Since this scraper is capable to process non-main namespaces, it processes pages, not only articles (see [The_difference_between_articles_and_page](https://en.wikipedia.org/wiki/Wikipedia:The_difference_between_articles_and_pages)).
|
|
28
|
+
- Page Title : as in [Mediawiki](https://www.mediawiki.org/wiki/Manual:Page_title), title of the page with spaces (not underscores) but namespace. E.g. 'Escherichia coli O157:H7' or 'Category:Escherichia coli'.
|
|
29
|
+
- Page Display Title : as in [Mediawiki](https://www.mediawiki.org/wiki/Display_title), preferred title for display. May contain HTML code.
|
|
30
|
+
- Page ZIM Title : Title of the page in the ZIM
|
|
21
31
|
|
|
22
32
|
## Retrieve Mediawiki info
|
|
23
33
|
|
|
@@ -28,68 +38,54 @@ Currently, scrape uses this call to retrieve:
|
|
|
28
38
|
- from `siteinfo`: `general` info (language, title, mainPage, site name, logo, text direction, ...), `skins` (to detect default skin), `rightsinfo` (to extract the license), `namespaces` and `namespacealiases` to build the list of namespaces
|
|
29
39
|
- from `allmessages`: the `tagline` (subtitle)
|
|
30
40
|
|
|
31
|
-
##
|
|
32
|
-
|
|
33
|
-
First, the scraper needs a list of article IDs and their details (redirects, ...).
|
|
41
|
+
## Retrieve list of pages to include and their metadata
|
|
34
42
|
|
|
35
|
-
|
|
43
|
+
First, the scraper needs a list of page titles and their details (redirects, ...).
|
|
36
44
|
|
|
37
|
-
|
|
45
|
+
"Page title" refers to the page title with spaces (not underscores), even if it is possible to use underscores in many places since Mediawiki APIs + scraper are permissive when possible. It is not the numeric page ID.
|
|
38
46
|
|
|
39
|
-
|
|
47
|
+
If user specified the exact list of pages to retrieve, then scraper simply request details about every pages in the list, in batches of about 50 pages (with `action=query&titles=titleX|titleY|...` ; batch sizes may vary due to constraints on query parameters size).
|
|
40
48
|
|
|
41
|
-
|
|
49
|
+
Otherwise, the scraper enumerates pages in given namespaces (by default, all content namespaces) with the `allpages` generator, requesting one namespace content at a time (`action=query&generator=allpages&gapnamespace=xx`).
|
|
42
50
|
|
|
43
|
-
|
|
51
|
+
The details we retrieve at this stage about every pages are their title, subtitle, revisions, redirects, thumbnail, categories, coordinates (when it applies), text language, text direction and contentmodel (to consider only on wikitext ones).
|
|
44
52
|
|
|
45
|
-
|
|
53
|
+
## Retrieving page HTML and rendering
|
|
46
54
|
|
|
47
|
-
|
|
55
|
+
In order to retrieve page HTML and render it, multiple solutions have been identified.
|
|
48
56
|
|
|
49
|
-
|
|
50
|
-
- WikimediaMobile
|
|
51
|
-
- RestApi
|
|
52
|
-
- VisualEditor
|
|
53
|
-
- ActionParse
|
|
57
|
+
As of today, 1 renderer (way to download page + render it to ZIM compatible HTML) is left implemented: ActionParse. Other renderers have been dropped in mwoffliner 2.0 because they were not used anymore, not implementing skin support and not providing added value compared to ActionParse render.
|
|
54
58
|
|
|
55
|
-
|
|
59
|
+
ActionParse API is available in Mediawiki since 1.16.0 (2010).
|
|
56
60
|
|
|
57
|
-
|
|
61
|
+
ActionParse renderer implements a thorough skin support (see below about skin) and needs only one HTTP query per page.
|
|
58
62
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
Only ActionParse (most recent renderer at mwoffliner level) implements a thorough skin support (see below about skin).
|
|
62
|
-
|
|
63
|
-
All renderers but ActionParse needs two HTTP queries: one to retrieve the article HTML and one to retrieve its metadata (to a 'simplified' ActionParse URL in fact).
|
|
64
|
-
|
|
65
|
-
The article metadata we retrieve at this stage is the article are:
|
|
63
|
+
The page metadata we retrieve at this stage are:
|
|
66
64
|
|
|
67
65
|
- `displaytitle` and `subtitle`
|
|
68
66
|
- `headhtml`: used to extract proper CSS classes we have to set on `<html>` and `<body>` tags + some JS configuration variables which are part of the HTML head
|
|
69
67
|
- `jsconfigvars`: other JS configuration variables which comes from other parts of the Mediawiki codebase
|
|
70
|
-
- `modules`: list of JS and CSS modules to apply on current
|
|
71
|
-
|
|
72
|
-
Renderer is automatically selected based on its availability and mwoffliner own preference. ActionParse is the preferred renderer since 1.15.0 due to its general availability and support of skins.
|
|
68
|
+
- `modules`: list of JS and CSS modules to apply on current page
|
|
73
69
|
|
|
74
70
|
### ActionParse parser
|
|
75
71
|
|
|
76
|
-
When using ActionParse renderer, we pass `usearticle=1`, which means that we ask the Mediawiki to use the parser configured for this
|
|
72
|
+
When using ActionParse renderer, we pass `usearticle=1`, which means that we ask the Mediawiki to use the parser configured for this page. This allows the scraper to retrieve page `text` (inner of the HTML containing the page itself) that is as close as possible to what is used online, either Parsoid or legacy parser (see https://www.mediawiki.org/wiki/Parsoid). This is mandatory because if we use a parser different than the one used online, we will de-facto get bugs which are "normal", either because parser still has a bug, or an extension is not compatible, or because Wikitext contains some workaround understandable only by a given parser. This should be avoided at all price if we want to have a versatile scraper capable of processing any Mediawiki. Even focusing only on Wikimedia wikis, not all of them have already transitioned to Parsoid for instance.
|
|
77
73
|
|
|
78
74
|
### Skins
|
|
79
75
|
|
|
80
|
-
In Mediawikis, rendering of Wikitext into HTML works around a concept of skin. A skin is a mix of HTML template and CSS+JS dependencies. It defines both the visual
|
|
76
|
+
In Mediawikis, rendering of Wikitext into HTML works around a concept of skin. A skin is a mix of HTML template and CSS+JS dependencies. It defines both the visual appearance of the rendered Wikitext but also everything "around it".
|
|
81
77
|
|
|
82
|
-
Since most wikis have adapted their content to their skin (and vice versa), it is mostly mandatory to use the skin inside the ZIM, both for proper rendering and for a visual
|
|
78
|
+
Since most wikis have adapted their content to their skin (and vice versa), it is mostly mandatory to use the skin inside the ZIM, both for proper rendering and for a visual appearance similar to online website (users don't mind about technical details, they want the wiki to be the same inside the ZIM than online).
|
|
83
79
|
|
|
84
80
|
Skin detection is automated in mwoffliner for now (see https://github.com/openzim/mwoffliner/issues/2213).
|
|
85
81
|
|
|
86
|
-
For now, only `vector` (legacy) and `vector-2022` are supported, and only with ActionParse renderer. Only `vector-2022` is the
|
|
82
|
+
For now, only `vector` (legacy) and `vector-2022` are supported, and only with ActionParse renderer. Only `vector-2022` is the truly responsive skin, providing ultimate rendering on mostly all screen sizes.
|
|
87
83
|
|
|
88
|
-
With ActionParse renderer, other skins have a `fallback` skin implemented in the scraper. This means that the scraper will render the
|
|
84
|
+
With ActionParse renderer, other skins have a `fallback` skin implemented in the scraper. This means that the scraper will render the page HTML inside an HTML structure which looks like an expected structure (which comes from `vector` legacy). The consequence is that if the skins uses only page HTML structure to attach CSS rules or JS code, then everything will render fine. If the skins uses HTML structures coming from the "surrounding HTML" (headers, ...) then things will not apply 100% correctly.
|
|
89
85
|
|
|
90
86
|
### JS / CSS dependencies
|
|
91
87
|
|
|
92
|
-
ActionParse API is returning the list of JS and CSS dependencies for a given
|
|
88
|
+
ActionParse API is returning the list of JS and CSS dependencies for a given page, by inspecting what Wikitext is using.
|
|
93
89
|
|
|
94
90
|
The special `startup` JS module is missing from results because always used anyway.
|
|
95
91
|
|
|
@@ -1,21 +1,22 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
1
|
+
export default class WiktionaryFR {
|
|
2
|
+
// implements CustomProcessor
|
|
3
|
+
async shouldKeepPage(pageTitle, doc) {
|
|
4
|
+
const frenchTitle = doc.querySelector(`#fr.sectionlangue`)
|
|
5
|
+
return !!frenchTitle
|
|
6
|
+
}
|
|
7
|
+
async preProcessPage(pageTitle, doc) {
|
|
8
|
+
const nonFrenchTitles = Array.from(doc.querySelectorAll(`.sectionlangue:not(#fr)`))
|
|
9
|
+
for (const title of nonFrenchTitles) {
|
|
10
|
+
title.closest('details').remove()
|
|
5
11
|
}
|
|
6
|
-
async preProcessArticle(articleId, doc) {
|
|
7
|
-
const nonFrenchTitles = Array.from(doc.querySelectorAll(`.sectionlangue:not(#fr)`));
|
|
8
|
-
for (const title of nonFrenchTitles) {
|
|
9
|
-
title.closest('details').remove();
|
|
10
|
-
}
|
|
11
12
|
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
}
|
|
16
|
-
//Remove h2 summary title
|
|
17
|
-
doc.querySelector('h2').closest('summary').setAttribute('style', 'display:none! important')
|
|
18
|
-
|
|
19
|
-
return doc;
|
|
13
|
+
const h4titles = Array.from(doc.querySelectorAll(`h4`))
|
|
14
|
+
for (const h4title of h4titles) {
|
|
15
|
+
h4title.closest('details').remove()
|
|
20
16
|
}
|
|
17
|
+
//Remove h2 summary title
|
|
18
|
+
doc.querySelector('h2').closest('summary').setAttribute('style', 'display:none! important')
|
|
19
|
+
|
|
20
|
+
return doc
|
|
21
|
+
}
|
|
21
22
|
}
|