mwoffliner 1.17.4 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. package/README.md +355 -62
  2. package/docs/functional_architecture.md +34 -38
  3. package/extensions/wiktionary_fr.js +18 -17
  4. package/jest.config.cjs +2 -2
  5. package/lib/DOMUtils.js +0 -1
  6. package/lib/DOMUtils.js.map +1 -1
  7. package/lib/Downloader.d.ts +39 -31
  8. package/lib/Downloader.js +418 -349
  9. package/lib/Downloader.js.map +1 -1
  10. package/lib/Dump.d.ts +24 -10
  11. package/lib/Dump.js +95 -50
  12. package/lib/Dump.js.map +1 -1
  13. package/lib/Gadgets.d.ts +2 -1
  14. package/lib/Gadgets.js +7 -5
  15. package/lib/Gadgets.js.map +1 -1
  16. package/lib/Logger.d.ts +6 -6
  17. package/lib/Logger.js +33 -33
  18. package/lib/Logger.js.map +1 -1
  19. package/lib/MediaWiki.d.ts +14 -31
  20. package/lib/MediaWiki.js +119 -172
  21. package/lib/MediaWiki.js.map +1 -1
  22. package/lib/RedisStore.d.ts +3 -3
  23. package/lib/RedisStore.js +21 -23
  24. package/lib/RedisStore.js.map +1 -1
  25. package/lib/S3.js +1 -1
  26. package/lib/S3.js.map +1 -1
  27. package/lib/Templates.d.ts +3 -10
  28. package/lib/Templates.js +5 -14
  29. package/lib/Templates.js.map +1 -1
  30. package/lib/cli.js +7 -6
  31. package/lib/cli.js.map +1 -1
  32. package/lib/config.d.ts +10 -16
  33. package/lib/config.js +33 -55
  34. package/lib/config.js.map +1 -1
  35. package/lib/error.manager.d.ts +2 -2
  36. package/lib/error.manager.js +42 -49
  37. package/lib/error.manager.js.map +1 -1
  38. package/lib/i18n.d.ts +2 -0
  39. package/lib/i18n.js +53 -0
  40. package/lib/i18n.js.map +1 -0
  41. package/lib/mutex.d.ts +2 -1
  42. package/lib/mutex.js +2 -1
  43. package/lib/mutex.js.map +1 -1
  44. package/lib/mwoffliner.lib.js +370 -222
  45. package/lib/mwoffliner.lib.js.map +1 -1
  46. package/lib/parameterList.d.ts +20 -8
  47. package/lib/parameterList.js +30 -18
  48. package/lib/parameterList.js.map +1 -1
  49. package/lib/renderers/abstract.renderer.d.ts +48 -55
  50. package/lib/renderers/abstract.renderer.js +377 -98
  51. package/lib/renderers/abstract.renderer.js.map +1 -1
  52. package/lib/renderers/action-parse.renderer.d.ts +22 -3
  53. package/lib/renderers/action-parse.renderer.js +123 -81
  54. package/lib/renderers/action-parse.renderer.js.map +1 -1
  55. package/lib/renderers/renderer.builder.js +6 -69
  56. package/lib/renderers/renderer.builder.js.map +1 -1
  57. package/lib/renderers/rendering.context.d.ts +2 -3
  58. package/lib/renderers/rendering.context.js +6 -14
  59. package/lib/renderers/rendering.context.js.map +1 -1
  60. package/lib/sanitize-argument.d.ts +7 -4
  61. package/lib/sanitize-argument.js +99 -41
  62. package/lib/sanitize-argument.js.map +1 -1
  63. package/lib/util/FileManager.d.ts +36 -0
  64. package/lib/util/FileManager.js +246 -0
  65. package/lib/util/FileManager.js.map +1 -0
  66. package/lib/util/RedisKvs.js +4 -4
  67. package/lib/util/RedisKvs.js.map +1 -1
  68. package/lib/util/RedisQueue.js +1 -1
  69. package/lib/util/RedisQueue.js.map +1 -1
  70. package/lib/util/builders/url/action-parse.director.d.ts +2 -2
  71. package/lib/util/builders/url/action-parse.director.js +7 -7
  72. package/lib/util/builders/url/action-parse.director.js.map +1 -1
  73. package/lib/util/builders/url/api.director.d.ts +2 -4
  74. package/lib/util/builders/url/api.director.js +10 -17
  75. package/lib/util/builders/url/api.director.js.map +1 -1
  76. package/lib/util/builders/url/base.director.d.ts +0 -4
  77. package/lib/util/builders/url/base.director.js +0 -25
  78. package/lib/util/builders/url/base.director.js.map +1 -1
  79. package/lib/util/builders/url/basic.director.d.ts +1 -1
  80. package/lib/util/builders/url/basic.director.js +1 -1
  81. package/lib/util/builders/url/web.director.d.ts +1 -1
  82. package/lib/util/builders/url/web.director.js +2 -2
  83. package/lib/util/builders/url/web.director.js.map +1 -1
  84. package/lib/util/categories.d.ts +5 -5
  85. package/lib/util/categories.js +175 -174
  86. package/lib/util/categories.js.map +1 -1
  87. package/lib/util/const.d.ts +4 -8
  88. package/lib/util/const.js +9 -17
  89. package/lib/util/const.js.map +1 -1
  90. package/lib/util/customCssJs.d.ts +6 -0
  91. package/lib/util/customCssJs.js +70 -0
  92. package/lib/util/customCssJs.js.map +1 -0
  93. package/lib/util/dump.d.ts +15 -4
  94. package/lib/util/dump.js +282 -86
  95. package/lib/util/dump.js.map +1 -1
  96. package/lib/util/index.d.ts +1 -1
  97. package/lib/util/index.js +1 -1
  98. package/lib/util/index.js.map +1 -1
  99. package/lib/util/metaData.js.map +1 -1
  100. package/lib/util/misc.d.ts +23 -13
  101. package/lib/util/misc.js +90 -59
  102. package/lib/util/misc.js.map +1 -1
  103. package/lib/util/mw-api.d.ts +7 -5
  104. package/lib/util/mw-api.js +167 -221
  105. package/lib/util/mw-api.js.map +1 -1
  106. package/lib/util/pageListMainPage.d.ts +3 -0
  107. package/lib/util/pageListMainPage.js +10 -0
  108. package/lib/util/pageListMainPage.js.map +1 -0
  109. package/lib/util/pages.d.ts +30 -0
  110. package/lib/util/pages.js +146 -0
  111. package/lib/util/pages.js.map +1 -0
  112. package/lib/util/rewriteUrls.d.ts +2 -2
  113. package/lib/util/rewriteUrls.js +33 -31
  114. package/lib/util/rewriteUrls.js.map +1 -1
  115. package/lib/util/savePages.d.ts +8 -0
  116. package/lib/util/savePages.js +182 -0
  117. package/lib/util/savePages.js.map +1 -0
  118. package/lib/version.d.ts +1 -1
  119. package/lib/version.js +1 -1
  120. package/lib/version.js.map +1 -1
  121. package/offliner-definition.json +178 -90
  122. package/package.json +50 -49
  123. package/res/page_list_home.js +20 -0
  124. package/res/{article_not_found.svg → page_not_found.svg} +1 -1
  125. package/res/script.js +30 -45
  126. package/res/style.css +52 -68
  127. package/res/templates/download_error_placeholder.html +7 -8
  128. package/res/templates/javaScript.html +19 -0
  129. package/res/templates/pageFallback.html +9 -7
  130. package/res/templates/pageVector2022.html +9 -7
  131. package/res/templates/pageVectorLegacy.html +9 -7
  132. package/res/templates/{article_list_home.html → page_list_home.html} +1 -2
  133. package/translation/ar.json +21 -21
  134. package/translation/bn.json +1 -1
  135. package/translation/de.json +46 -19
  136. package/translation/en.json +54 -21
  137. package/translation/es.json +17 -19
  138. package/translation/fi.json +23 -3
  139. package/translation/fr.json +57 -21
  140. package/translation/he.json +8 -6
  141. package/translation/ia.json +19 -19
  142. package/translation/id.json +13 -13
  143. package/translation/it.json +13 -8
  144. package/translation/ko.json +55 -3
  145. package/translation/lb.json +5 -2
  146. package/translation/mk.json +5 -5
  147. package/translation/nl.json +57 -22
  148. package/translation/or.json +2 -2
  149. package/translation/pt.json +2 -2
  150. package/translation/qqq.json +55 -22
  151. package/translation/ru.json +2 -2
  152. package/translation/sc.json +2 -2
  153. package/translation/sk.json +62 -0
  154. package/translation/sl.json +54 -21
  155. package/translation/sv.json +20 -19
  156. package/translation/sw.json +45 -2
  157. package/translation/zh-hans.json +56 -22
  158. package/translation/zh-hant.json +26 -19
  159. package/lib/renderers/abstractDesktop.render.d.ts +0 -12
  160. package/lib/renderers/abstractDesktop.render.js +0 -67
  161. package/lib/renderers/abstractDesktop.render.js.map +0 -1
  162. package/lib/renderers/abstractMobile.render.d.ts +0 -12
  163. package/lib/renderers/abstractMobile.render.js +0 -54
  164. package/lib/renderers/abstractMobile.render.js.map +0 -1
  165. package/lib/renderers/rest-api.renderer.d.ts +0 -7
  166. package/lib/renderers/rest-api.renderer.js +0 -68
  167. package/lib/renderers/rest-api.renderer.js.map +0 -1
  168. package/lib/renderers/visual-editor.renderer.d.ts +0 -7
  169. package/lib/renderers/visual-editor.renderer.js +0 -76
  170. package/lib/renderers/visual-editor.renderer.js.map +0 -1
  171. package/lib/renderers/wikimedia-desktop.renderer.d.ts +0 -4
  172. package/lib/renderers/wikimedia-desktop.renderer.js +0 -7
  173. package/lib/renderers/wikimedia-desktop.renderer.js.map +0 -1
  174. package/lib/renderers/wikimedia-mobile.renderer.d.ts +0 -19
  175. package/lib/renderers/wikimedia-mobile.renderer.js +0 -243
  176. package/lib/renderers/wikimedia-mobile.renderer.js.map +0 -1
  177. package/lib/util/articleListMainPage.d.ts +0 -3
  178. package/lib/util/articleListMainPage.js +0 -10
  179. package/lib/util/articleListMainPage.js.map +0 -1
  180. package/lib/util/articles.d.ts +0 -26
  181. package/lib/util/articles.js +0 -102
  182. package/lib/util/articles.js.map +0 -1
  183. package/lib/util/builders/url/desktop.director.d.ts +0 -8
  184. package/lib/util/builders/url/desktop.director.js +0 -15
  185. package/lib/util/builders/url/desktop.director.js.map +0 -1
  186. package/lib/util/builders/url/mobile.director.d.ts +0 -8
  187. package/lib/util/builders/url/mobile.director.js +0 -15
  188. package/lib/util/builders/url/mobile.director.js.map +0 -1
  189. package/lib/util/builders/url/rest-api.director.d.ts +0 -8
  190. package/lib/util/builders/url/rest-api.director.js +0 -18
  191. package/lib/util/builders/url/rest-api.director.js.map +0 -1
  192. package/lib/util/builders/url/visual-editor.director.d.ts +0 -9
  193. package/lib/util/builders/url/visual-editor.director.js +0 -18
  194. package/lib/util/builders/url/visual-editor.director.js.map +0 -1
  195. package/lib/util/saveArticles.d.ts +0 -8
  196. package/lib/util/saveArticles.js +0 -411
  197. package/lib/util/saveArticles.js.map +0 -1
  198. package/res/article_list_home.js +0 -24
  199. package/res/content.parsoid.css +0 -196
  200. package/res/inserted_style.css +0 -45
  201. package/res/templates/categories.html +0 -11
  202. package/res/templates/lead_section_wrapper.html +0 -6
  203. package/res/templates/pageWikimediaDesktop.html +0 -24
  204. package/res/templates/pageWikimediaMobile.html +0 -24
  205. package/res/templates/section_wrapper.html +0 -5
  206. package/res/templates/subcategories.html +0 -23
  207. package/res/templates/subpages.html +0 -17
  208. package/res/templates/subsection_wrapper.html +0 -5
  209. package/res/webpHandler.js +0 -165
  210. package/res/wm_mobile_override_script.js +0 -15
  211. package/res/wm_mobile_override_style.css +0 -20
  212. package/translation/br.json +0 -8
  213. package/translation/dag.json +0 -9
  214. package/translation/es-formal.json +0 -29
  215. package/translation/ha.json +0 -10
  216. package/translation/hi.json +0 -9
  217. package/translation/ig.json +0 -10
  218. package/translation/kaa.json +0 -9
  219. package/translation/nb.json +0 -9
  220. package/translation/nqo.json +0 -8
  221. package/translation/pt-br.json +0 -10
  222. package/translation/ro.json +0 -9
  223. package/translation/scn.json +0 -8
  224. package/translation/sq.json +0 -9
  225. package/translation/te.json +0 -9
  226. package/translation/tn.json +0 -9
  227. package/translation/tr.json +0 -10
package/README.md CHANGED
@@ -1,19 +1,10 @@
1
1
  # MWoffliner
2
2
 
3
- MWoffliner is a tool for making a local offline HTML snapshot of any
4
- online [MediaWiki](https://mediawiki.org) instance. It goes through
5
- all online articles (or a selection if specified) and create the
6
- corresponding [ZIM](https://openzim.org) file. It has mainly been
7
- tested against Wikimedia projects like
8
- [Wikipedia](https://wikipedia.org) and
9
- [Wiktionary](https://wiktionary.org) --- but it should also work for
10
- any recent MediaWiki.
3
+ MWoffliner is a tool for creating a local offline HTML snapshot of any online [MediaWiki](https://mediawiki.org) instance. It scrapes all pages (or a selection if specified) and creates the corresponding [ZIM](https://openzim.org) file. While primarily targeted for Wikimedia projects like [Wikipedia](https://wikipedia.org) and [Wiktionary](https://wiktionary.org), MWoffliner also supports any recent MediaWiki instance (version 1.27+), though instances with custom skins or highly unusual configurations may have limitations.
11
4
 
12
- Read [CONTRIBUTING.md](./CONTRIBUTING.md) to know more about
13
- MWoffliner development.
5
+ Read [CONTRIBUTING.md](./CONTRIBUTING.md) to learn more about MWoffliner development.
14
6
 
15
- User Help is available in the for a a
16
- [FAQ](https://github.com/openzim/mwoffliner/wiki/Frequently-Asked-Questions).
7
+ User help is available in the [FAQ](https://github.com/openzim/mwoffliner/wiki/Frequently-Asked-Questions).
17
8
 
18
9
  [![NPM](https://nodei.co/npm/mwoffliner.png)](https://www.npmjs.com/package/mwoffliner)
19
10
 
@@ -28,53 +19,282 @@ User Help is available in the for a a
28
19
 
29
20
  ## Features
30
21
 
31
- - Scrape with or without image thumbnail
22
+ - Scrape with or without image thumbnails
32
23
  - Scrape with or without audio/video multimedia content
33
24
  - S3 cache (optional)
34
- - Image size optimiser / Webp converter
35
- - Scrape all articles in namespaces or title list based
25
+ - Image size optimization and WebP conversion
26
+ - Scrape all pages in namespaces or title list based
36
27
  - Specify additional/non-main namespaces to scrape
37
28
 
38
- Run `mwoffliner --help` to get all the possible options.
29
+ Run `mwoffliner --help` to see all available options.
39
30
 
40
- ## Prerequisites
31
+ ## Quick Start
41
32
 
42
- - *NIX Operating System (GNU/Linux, macOS, ...)
43
- - [Redis](https://redis.io/)
44
- - [NodeJS](https://nodejs.org/en/) version 24 (we support only one single Node.JS version, other versions might work or not)
45
- - [Libzim](https://github.com/openzim/libzim) (On GNU/Linux & macOS we automatically download it)
46
- - Various build tools which are probably already installed on your
47
- machine (packages `libjpeg-dev`, `libglu1`, `autoconf`, `automake`, `gcc` on
48
- Debian/Ubuntu)
33
+ ### Prerequisites
49
34
 
50
- ... and an online MediaWiki with its API available.
35
+ - [Docker](https://docs.docker.com/engine/install/) (or Docker-based engine)
36
+ - amd64 or arm64 architecture
51
37
 
52
- ## Usage
38
+ To turn the Bambara Wikipedia into an offline ZIM:
53
39
 
54
- To install latest released MWoffliner version from NPM repo (use `-g` to install globally, not only in current folder):
55
- ```bash
56
- npm i -g mwoffliner
40
+ ```sh
41
+ mkdir -p output
42
+ docker run -v $(pwd)/output:/output ghcr.io/openzim/mwoffliner \
43
+ mwoffliner --mwUrl=https://bm.wikipedia.org --adminEmail=you@example.com \
44
+ --outputDirectory=/output
57
45
  ```
58
46
 
59
- > [!WARNING]
60
- > Note that you might need to run this command with the `sudo` command, depending
61
- how your `npm` / OS is configured. `npm` permission checking can be a bit annoying for a
62
- newcomer. Please read the documentation carefully if you hit problems: https://docs.npmjs.com/cli/v7/using-npm/scripts#user
47
+ **NOTE**: In order to avoid [429 responses](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/429), be sure to use an
48
+ actual email address as dummy addresses like `you@example.com` will be flagged by the Wiki servers.
63
49
 
64
- Then you can run the scraper:
65
- ```bash
66
- mwoffliner --help
50
+ That's it. When it finishes you'll find a `.zim` file in the `output/` directory. Open it with
51
+ [Kiwix Reader](https://get.kiwix.org/en/solutions/applications/kiwix-reader/).
52
+
53
+ ### The Four Decisions
54
+
55
+ Every scrape comes down to four choices:
56
+
57
+ #### 1. Which wiki?
58
+
59
+ ```sh
60
+ --mwUrl=https://en.wikipedia.org # English Wikipedia
61
+ --mwUrl=https://fr.wiktionary.org # French Wiktionary
62
+ --mwUrl=https://terraria.wiki.gg # Terraria Wiki
63
+ --mwUrl=https://proofwiki.org # ProofWiki
64
+ --mwUrl=https://wiki.my-org.com # Any MediaWiki site
67
65
  ```
68
66
 
69
- To use MWoffliner with a S3 cache, you should provide a S3 URL like
70
- this:
71
- ```bash
72
- --optimisationCacheUrl="https://wasabisys.com/?bucketName=my-bucket&keyId=my-key-id&secretAccessKey=my-sac"
67
+ #### 2. Which API path?
68
+
69
+ Most MediaWiki sites use the default API path `/w/api.php`, but many
70
+ don't. Check by visiting `<wiki-url>/w/api.php` in your browser. If that returns a 404, you need to set `--mwActionApiPath` to the correct path.
71
+
72
+ > [!TIP]
73
+ > Try `https://terraria.wiki.gg/api.php` to see a wiki where
74
+ > `--mwActionApiPath=/api.php` is required.
75
+
76
+ #### 3. What content?
77
+
78
+ | If you want… | Add this |
79
+ | --------------------------------------- | ------------------------------- |
80
+ | Everything (text, images, audio, video) | (nothing — this is the default) |
81
+ | Everything except video and audio | `--format=novid:maxi` |
82
+ | No pictures, no video, no audio | `--format=nopic:nopic` |
83
+ | Head paragraphs only, no media at all | `--format=nodet,nopic:mini` |
84
+
85
+ #### 4. Where does it go?
86
+
87
+ The ZIM is written to `/output` inside the container. Map that to a folder on your machine:
88
+
89
+ ```sh
90
+ docker run -v /path/on/my/machine:/output ghcr.io/openzim/mwoffliner ...
91
+ ```
92
+
93
+ **NOTE**: `--adminEmail=you@example.com` is also required. It is included in the HTTP
94
+ User-Agent so wiki operators know who is scraping.
95
+
96
+ ### A Few More Things You Might Want
97
+
98
+ #### Scrape only specific pages
99
+
100
+ Pass a comma-separated list of page titles directly:
101
+
102
+ ```sh
103
+ --pageList="Main Page,Earth,Albert Einstein"
104
+ ```
105
+
106
+ Or put one page title per line in a text file and point to it:
107
+
108
+ ```sh
109
+ --pageList=./my-pages.txt
110
+ ```
111
+
112
+ #### Scrape a private wiki
113
+
114
+ Provide credentials with `--mwUsername` and `--mwPassword`:
115
+
116
+ ```sh
117
+ --mwUsername=jdoe --mwPassword=s3cret
118
+ ```
119
+
120
+ Preferably use a [bot password](https://www.mediawiki.org/wiki/Manual:Bot_passwords)
121
+ rather than a regular user account.
122
+
123
+ If authentication requires a separate domain, also pass `--mwDomain`:
124
+
125
+ ```sh
126
+ --mwDomain=corp --mwUsername=jdoe --mwPassword=s3cret
73
127
  ```
74
128
 
129
+ #### Customising the Result
130
+
131
+ Want your ZIM to have a specific title, description, or icon?
132
+
133
+ ```sh
134
+ --customZimTitle="My Offline Wiki"
135
+ --customZimDescription="A hand-picked selection of articles"
136
+ --customZimFavicon=https://example.com/icon.png
137
+ ```
138
+
139
+ #### Adjusting the scrape speed
140
+
141
+ Since version 2.0.0, the default request rate (speed `1`) is fine for most
142
+ wikis. The `--speed` option controls the climb rate — how aggressively
143
+ mwoffliner ramps up its request concurrency:
144
+
145
+ If you see lots of HTTP errors (e.g. 429 Too Many Requests) in the logs,
146
+ try lowering the speed e.g `--speed=0.5` can help prevent
147
+ the wiki from rate-limiting you.
148
+
149
+ #### Going Further
150
+
151
+ These and all other options are listed in `mwoffliner --help`.
152
+
153
+ Also, see the [FAQ](https://github.com/openzim/mwoffliner/wiki/Frequently-Asked-Questions) for detailed explanations of
154
+ the command line options and common issues.
155
+
156
+ Need help? [![Join Slack](https://img.shields.io/badge/Join%20us%20on%20Slack%20%23mwoffliner-2EB67D)](https://slack.kiwix.org)
157
+
158
+ ## Installation
159
+
160
+ The recommended way to install and run `mwoffliner` is using the pre-built Docker container:
161
+
162
+ ```sh
163
+ docker pull ghcr.io/openzim/mwoffliner
164
+ ```
165
+
166
+ <details>
167
+ <summary>Run software locally / Build from source</summary>
168
+
169
+ ### Prerequisites for local execution
170
+
171
+ - \*NIX Operating System (GNU/Linux, macOS, etc.)
172
+ - [Redis](https://redis.io/) — in-memory data store
173
+ - [Node.js](https://nodejs.org/en/) version 24 (we support only one single Node.js version; other versions might work or might not)
174
+ - [Libzim](https://github.com/openzim/libzim) — C++ library for creating ZIM files (automatically downloaded on GNU/Linux & macOS)
175
+ - Various build tools which are probably already installed on your machine:
176
+ - `libjpeg-dev` — JPEG image processing
177
+ - `libglu1` — OpenGL utility library
178
+ - `autoconf` — automatic configuration system
179
+ - `automake` — Makefile generator
180
+ - `gcc` — C compiler
181
+
182
+ (These packages are for Debian/Ubuntu systems)
183
+
184
+ An online [MediaWiki](https://mediawiki.org) instance with its API available.
185
+
186
+ ### Installation methods
187
+
188
+ #### Build your own container
189
+
190
+ 1. Clone the repository locally:
191
+
192
+ ```sh
193
+ git clone https://github.com/openzim/mwoffliner.git && cd mwoffliner
194
+ ```
195
+
196
+ 1. Build the image:
197
+
198
+ ```sh
199
+ docker build . -f docker/Dockerfile -t ghcr.io/openzim/mwoffliner
200
+ ```
201
+
202
+ #### Run the software locally using NPM
203
+
204
+ > [!WARNING]
205
+ > Local installation requires several system dependencies (see above). Using the Docker image is strongly recommended to avoid setup issues.
206
+
207
+ Setting up MWoffliner locally for development can be tricky due to several dependencies and version requirements. Follow these steps carefully to avoid common errors.
208
+
209
+ ##### 1. Node.js Version
210
+
211
+ MWoffliner requires Node.js 24 (other versions may fail).
212
+
213
+ Compatible Node 24 ranges: `>=24 <24.6` or `>=24.7 <25`.
214
+
215
+ Check your version:
216
+
217
+ ```sh
218
+ node -v
219
+ ```
220
+
221
+ If your version does not match, use [nvm](https://github.com/nvm-sh/nvm) to install the correct Node.js version.
222
+
223
+ ##### 2. libzim Dependency
224
+
225
+ MWoffliner depends on [`@openzim/libzim`](https://github.com/openzim/libzim), which requires the C++ libzim library.
226
+
227
+ - On Linux/macOS, MWoffliner can download libzim automatically.
228
+ - On Windows, you must install libzim manually because there are no prebuilt binaries. See the [libzim installation guide](https://github.com/openzim/libzim) for details.
229
+
230
+ ##### 3. Compiler Requirements (Windows)
231
+
232
+ Node 24 on Windows officially supports [Visual Studio 2019 (v16)](https://visualstudio.microsoft.com/vs/older-downloads/) or [Visual Studio 2022 (v17)](https://visualstudio.microsoft.com/downloads/).
233
+
234
+ Ensure C++ build tools are installed and environment variables are set correctly. See [Windows Setup for node-gyp](https://github.com/nodejs/node-gyp#on-windows) for detailed instructions.
235
+
236
+ ##### 4. Node-gyp
237
+
238
+ MWoffliner uses [node-gyp](https://github.com/nodejs/node-gyp), which enforces strict checks for Node and compiler versions. Make sure you have:
239
+
240
+ - Proper Visual Studio version (Windows) — see [Visual Studio versions](https://visualstudio.microsoft.com/downloads/)
241
+ - Required C++ headers, e.g., `zim/archive.h` — see [libzim documentation](https://github.com/openzim/libzim)
242
+ - [Python 3.10+](https://www.python.org/downloads/) (required by node-gyp; a recent version is preferred for compatibility)
243
+
244
+ ##### Additional troubleshooting steps if errors persist:
245
+
246
+ 1. **Clear npm cache** — a corrupted cache can cause cryptic install failures:
247
+
248
+ ```sh
249
+ npm cache clean --force
250
+ ```
251
+
252
+ 2. **Delete node_modules and reinstall** — stale or partially installed dependencies are a common source of errors:
253
+
254
+ ```sh
255
+ rm -rf node_modules package-lock.json
256
+ npm install
257
+ ```
258
+
259
+ 3. **Check that all environment variables are set** — especially on Windows, `PATH`, `INCLUDE`, and `LIB` must point to the correct Visual Studio and libzim directories. Reopen your terminal after installing new tools.
260
+
261
+ 4. **Verify Redis is running before starting MWoffliner** — MWoffliner will fail immediately if it cannot connect to Redis:
262
+
263
+ ```sh
264
+ redis-cli ping # expected output: PONG
265
+ ```
266
+
267
+ 5. **Run npm install with verbose logging** to see exactly where it fails:
268
+ ```sh
269
+ npm install --verbose
270
+ ```
271
+
272
+ ##### 5. Common Errors & Troubleshooting
273
+
274
+ | Error | Cause | Solution |
275
+ | ---------------------------------- | ------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
276
+ | Node.js version error | Node.js version incompatible | Install [Node 24 with nvm](https://github.com/nvm-sh/nvm) |
277
+ | Cannot find module @openzim/libzim | libzim not installed | Follow [libzim installation guide](https://github.com/openzim/libzim); Windows users must install manually |
278
+ | node-gyp rebuild failed | Wrong Node or compiler version | Check [Node.js version](https://nodejs.org/en/), [Visual Studio version](https://visualstudio.microsoft.com/downloads/), [Python 3.x](https://www.python.org/downloads/) |
279
+ | zim/archive.h not found | C++ headers missing | Install [libzim](https://github.com/openzim/libzim) system-wide, verify include paths |
280
+
281
+ > [!NOTE]
282
+ > Even with these steps, other setup errors may occur. Using Docker is strongly recommended for a smoother experience.
283
+
284
+ ##### Installation via NPM
285
+
286
+ ```sh
287
+ npm i -g mwoffliner
288
+ ```
289
+
290
+ > [!WARNING]
291
+ > You might need to run this command with the `sudo` command, depending on how your `npm` / OS is configured. `npm` permission checking can be a bit annoying for newcomers. Please read the [npm script documentation](https://docs.npmjs.com/cli/v7/using-npm/scripts#user) if you encounter issues.
292
+
293
+ </details>
294
+
75
295
  ## Contribute
76
296
 
77
- If you've retrieved mwoffliner source code (e.g. with a git clone of our repo), you can then install and run it locally (including with your local modifications):
297
+ If you've retrieved the MWoffliner source code (e.g., via a git clone), you can install and run it locally with your modifications:
78
298
 
79
299
  ```bash
80
300
  npm i
@@ -85,35 +305,108 @@ Detailed [contribution documentation and guidelines](CONTRIBUTING.md) are availa
85
305
 
86
306
  ## API
87
307
 
88
- MWoffliner provides also an API and therefore can be used as a NodeJS
89
- library. Here a stub example that could go in your index.mjs file:
308
+ MWoffliner provides an API and can be used as a Node.js library. Here's a stub example for your `index.mjs` file:
309
+
90
310
  ```javascript
91
- import * as mwoffliner from 'mwoffliner';
311
+ import * as mwoffliner from 'mwoffliner'
92
312
 
93
313
  const parameters = {
94
- mwUrl: "https://es.wikipedia.org",
95
- adminEmail: "foo@bar.net",
96
- verbose: true,
97
- format: "nopic",
98
- articleList: "./articleList"
99
- };
100
- mwoffliner.execute(parameters); // returns a Promise
314
+ mwUrl: 'https://es.wikipedia.org',
315
+ adminEmail: 'foo@bar.net',
316
+ verbose: true,
317
+ format: 'nopic',
318
+ pageList: './pageList',
319
+ }
320
+
321
+ mwoffliner.execute(parameters) // returns a Promise
322
+ ```
323
+
324
+ ## MathJax support
325
+
326
+ > [!WARNING]
327
+ > MathJax support is UNSTABLE: the `--mathJax*` CLI parameters described below may change, even in a minor release. Setting them up also requires wiki-specific technical preparation, so this is aimed at developers rather than end users.
328
+
329
+ MWoffliner can bundle [MathJax](https://www.mathjax.org) into the ZIM so that math formulas keep rendering offline, on wikis that rely on it (e.g. via the `SimpleMathJax` extension).
330
+
331
+ MathJax 2, 3 and 4 are very different beasts: each ships its own set of files and, more importantly, each requires its own incompatible configuration format (which defines things like math delimiters and custom macros). Because of this, MWoffliner cannot auto-detect and configure MathJax for you — you have to supply a matching MathJax build and configuration yourself, via three CLI parameters:
332
+
333
+ - `--mathJaxSource`: local path or HTTP(S) URL to a ZIP archive of a compiled MathJax distribution. Its content is extracted and pushed to the ZIM (under an internal `_mathjax_/` namespace).
334
+ - `--mathJaxConfig`: local path or HTTP(S) URL to an HTML file containing a single `<script>` tag with the MathJax configuration (this must be copied from the wiki, see below). Its content is injected inline, before the MathJax library, on every page that needs it. If the configuration needs to reference a path inside the MathJax archive (e.g. `MathJax.Ajax.config.path[...]`), it cannot use an absolute path since ZIMs have no fixed root URL. Write `__MATHJAX_ROOT__` instead of the leading slash and mwoffliner will replace it, on every page, with the correct relative path to the root of the MathJax archive you provided. For instance use `MathJax.Ajax.config.path["Contrib"] = "__MATHJAX_ROOT__/MathJaxExtensions/legacy";` if the resources are located in a `MathJaxExtensions/legacy` folder in the archive (inside the root folder of the archive if there is a single folder as usual).
335
+ - `--mathJaxEntryPoint`: path, relative to the root of the extracted archive, to the MathJax script to load (e.g. `es5/tex-chtml.js` for MathJax 3). Defaults to `MathJax.js` (MathJax 2 entry point).
336
+ - `--mathJaxAllPages`: inject the config/entry-point `<script>` tags on every page instead of only on pages detected to need MathJax (see below). Some wikis don't list MathJax in their page JS modules, which defeats the automatic detection; this flag works around that at the cost of adding the scripts to every page.
337
+
338
+ `--mathJaxConfig`, `--mathJaxEntryPoint` and `--mathJaxAllPages` all require `--mathJaxSource` to also be set. A page is considered to "need" MathJax, and only then gets the config/entry-point `<script>` tags injected, when one of the JS modules MediaWiki reports for that page matches `mathjax` (case-insensitive), unless `--mathJaxAllPages` is set, in which case every page gets them. The extracted library files themselves are always pushed to the ZIM as soon as `--mathJaxSource` is set, regardless of which pages use them.
339
+
340
+ ### Preparing the parameters for a given wiki
341
+
342
+ 1. **Find the live MathJax version and configuration.** Open a wiki page that renders math formulas, open your browser's developer console and type `MathJax.version` to get the exact version. Then find the configuration, typically a `<script>` block setting `window.MathJax = {...}` (MathJax 3/4) or `MathJax.Hub.Config({...})` (MathJax 2) in the page source — copy it as-is into a local file, e.g. `mathjax-config.html`.
343
+ 2. **Find the exact entry point.** In your browser's network tab, find the request loading the MathJax library itself (typically named `MathJax.js` for MathJax 2, or a `tex-chtml.js`/`tex-svg.js`/... for MathJax 3/4) and note its full path, including any query string.
344
+ 3. **Build a matching MathJax ZIP**, the exact steps depend on the major version in use (see below).
345
+ 4. **Run the scraper** with the three parameters, e.g.:
346
+ ```sh
347
+ mwoffliner --mwUrl=https://your.wiki --adminEmail=foo@bar.net \
348
+ --mathJaxSource=./mathjax-source.zip \
349
+ --mathJaxConfig=./mathjax-config.html \
350
+ --mathJaxEntryPoint=es5/tex-chtml.js
351
+ ```
352
+
353
+ ### Building the ZIP — MathJax 2
354
+
355
+ MathJax 2 is not published to npm as an installable package; it is only distributed as source on GitHub. Download a release archive directly from the [MathJax releases page](https://github.com/mathjax/MathJax/releases) (e.g. `2.7.9`) and use it as-is as `--mathJaxSource`, no re-zipping needed:
356
+
357
+ ```sh
358
+ curl -Lo mathjax-source.zip https://github.com/mathjax/MathJax/archive/refs/tags/2.7.9.zip
101
359
  ```
102
360
 
361
+ The archive has a single top-level folder (e.g. `MathJax-2.7.9/`); mwoffliner strips it automatically when extracting.
362
+
363
+ Unlike MathJax 3/4, MathJax 2 loads its extensions/output-processor via a `config=` query parameter on the `MathJax.js` request itself (e.g. `MathJax.js?config=TeX-MML-AM_CHTML`) rather than solely through the injected configuration script — this is the URL you captured in step 2 above. Set `--mathJaxEntryPoint` to that same path and query string, e.g.:
364
+
365
+ ```sh
366
+ --mathJaxEntryPoint="MathJax.js?config=TeX-MML-AM_CHTML"
367
+ ```
368
+
369
+ The referenced combined-configuration file (here `config/TeX-MML-AM_CHTML.js`) is part of the standard MathJax 2 distribution, so it is already included in the ZIP from the release archive.
370
+
371
+ ### Building the ZIP — MathJax 3
372
+
373
+ MathJax 3 is published to npm as `mathjax-full`, which bundles both the compiled runtime (under `es5/`) and the TypeScript sources (which mwoffliner automatically ignores when an `es5/` folder is present):
374
+
375
+ ```sh
376
+ npm install mathjax-full@3.2.2
377
+ cd node_modules/mathjax-full && zip -r ../../mathjax-source.zip . && cd ../..
378
+ ```
379
+
380
+ Entry point example: `--mathJaxEntryPoint=es5/tex-chtml.js`.
381
+
382
+ ### Building the ZIP — MathJax 4
383
+
384
+ MathJax 4 moved to scoped npm packages. Use the plain `mathjax` package (deployment-ready bundle), **not** `@mathjax/src` (the TypeScript source package meant for building MathJax itself, which requires separately installing font packages):
385
+
386
+ ```sh
387
+ npm install mathjax@4
388
+ cd node_modules/mathjax && zip -r ../../mathjax-source.zip . && cd ../..
389
+ ```
390
+
391
+ Unlike MathJax 3, entry-point files live at the root of the package (no `es5/` folder), e.g. `--mathJaxEntryPoint=tex-chtml.js`.
392
+
103
393
  ## Background
104
394
 
105
395
  Complementary information about MWoffliner:
106
396
 
107
- * MediaWiki software is used by thousands of wikis, the most
108
- famous ones being the Wikimedia ones, including [Wikipedia](https://wikipedia.org).
109
- * MediaWiki is a PHP wiki runtime engine.
110
- * Wikitext is the name of the markup language that MediaWiki uses.
111
- * MediaWiki includes a parser for WikiText into HTML, and this
112
- parser creates the HTML pages displayed in your browser.
113
- * Have a look at the scraper [functional architecture](docs/functional_architecture.md)
397
+ - **MediaWiki software** is used by thousands of wikis, the most famous ones being the Wikimedia ones, including [Wikipedia](https://wikipedia.org).
398
+ - **MediaWiki** is a PHP wiki runtime engine.
399
+ - **Wikitext** is the markup language that MediaWiki uses.
400
+ - **MediaWiki parser** converts Wikitext to HTML, which displays in your browser.
401
+ - Read the [scraper functional architecture](docs/functional_architecture.md) for more details.
402
+
403
+ ## License
404
+
405
+ [GPLv3](https://www.gnu.org/licenses/gpl-3.0) or later, see [LICENSE](LICENSE) for more details.
406
+
407
+ ## Acknowledgements
114
408
 
115
- License
116
- -------
409
+ This project received funding through [NGI Zero Core](https://nlnet.nl/core), a fund established by [NLnet](https://nlnet.nl/) with financial support from the European Commission's [Next Generation Internet](https://ngi.eu/) program. Learn more at the [NLnet project page](https://nlnet.nl/project/MWOffliner).
117
410
 
118
- [GPLv3](https://www.gnu.org/licenses/gpl-3.0) or later, see
119
- [LICENSE](LICENSE) for more details.
411
+ [<img width="20%" alt="NLnet foundation logo" src="https://github.com/user-attachments/assets/22233242-ec49-4540-a0af-b70725cedbee" />](https://nlnet.nl/)
412
+ [<img width="20%" alt="NGI Zero Logo" src="https://github.com/user-attachments/assets/1bbbda57-dc6f-4902-ae29-236e5e89228f" />](https://nlnet.nl/core)
@@ -5,11 +5,11 @@ This document describes a high-level overview of how mwoffliner scraper works.
5
5
  At a high level, mwoffliner is divided into following sequence of actions.
6
6
 
7
7
  - retrieve Mediawiki info
8
- - retrieve list of articles to include and their metadata
9
- - for every article:
10
- - retrieve its parsed HTML (Wikitext transformed into HTML) and JS/CSS dependencie
8
+ - retrieve list of pages to include and their metadata
9
+ - for every page:
10
+ - retrieve its parsed HTML (Wikitext transformed into HTML) and JS/CSS dependencies
11
11
  - adapt / render it for proper operation within the ZIM file (includes detection of media dependencies)
12
- - save rendered article HTML into the ZIM
12
+ - save rendered page HTML into the ZIM
13
13
  - for every file dependency (JS/CSS/media)
14
14
  - if its an image, download it either from S3 cache (images only) or from online and recompress when possible
15
15
  - otherwise download it from online
@@ -17,7 +17,17 @@ At a high level, mwoffliner is divided into following sequence of actions.
17
17
 
18
18
  The scraper supports flavours, which are variants of the ZIM (e.g. without images, with images but without videos, ...).
19
19
 
20
- For now, retrieval of articles and files dependencies is repeated for every flavour requested (even if content probably didn't changed).
20
+ For now, retrieval of pages and files dependencies is repeated for every flavour requested (even if content probably didn't changed).
21
+
22
+ ## Lexicography
23
+
24
+ > [!NOTE]
25
+ > All these concepts have been clarified in mwoffliner 2.0.0.
26
+
27
+ - Page: Base object of Mediawikis. Could be an article. Since this scraper is capable to process non-main namespaces, it processes pages, not only articles (see [The_difference_between_articles_and_page](https://en.wikipedia.org/wiki/Wikipedia:The_difference_between_articles_and_pages)).
28
+ - Page Title : as in [Mediawiki](https://www.mediawiki.org/wiki/Manual:Page_title), title of the page with spaces (not underscores) but namespace. E.g. 'Escherichia coli O157:H7' or 'Category:Escherichia coli'.
29
+ - Page Display Title : as in [Mediawiki](https://www.mediawiki.org/wiki/Display_title), preferred title for display. May contain HTML code.
30
+ - Page ZIM Title : Title of the page in the ZIM
21
31
 
22
32
  ## Retrieve Mediawiki info
23
33
 
@@ -28,68 +38,54 @@ Currently, scrape uses this call to retrieve:
28
38
  - from `siteinfo`: `general` info (language, title, mainPage, site name, logo, text direction, ...), `skins` (to detect default skin), `rightsinfo` (to extract the license), `namespaces` and `namespacealiases` to build the list of namespaces
29
39
  - from `allmessages`: the `tagline` (subtitle)
30
40
 
31
- ## Retrive list of articles to include and their metadata
32
-
33
- First, the scraper needs a list of article IDs and their details (redirects, ...).
41
+ ## Retrieve list of pages to include and their metadata
34
42
 
35
- "Article ID" refers to the page title with underscores instead of spaces, not the numeric page ID.
43
+ First, the scraper needs a list of page titles and their details (redirects, ...).
36
44
 
37
- If user specified the exact list of articles to retrieve, then scraper simply request details about every articles in the list, in batches of about 50 articles (with `action=query&titles=titleX|titleY|...` ; batch sizes may vary due to constraints on query parameters size).
45
+ "Page title" refers to the page title with spaces (not underscores), even if it is possible to use underscores in many places since Mediawiki APIs + scraper are permissive when possible. It is not the numeric page ID.
38
46
 
39
- Otherwise, the scraper enumerates articles in given namespaces (by default, all content namespaces) with the `allpages` generator, requesting one namespace content at a time (`action=query&generator=allpages&gapnamespace=xx`).
47
+ If user specified the exact list of pages to retrieve, then scraper simply request details about every pages in the list, in batches of about 50 pages (with `action=query&titles=titleX|titleY|...` ; batch sizes may vary due to constraints on query parameters size).
40
48
 
41
- The details we retrieve at this stage about every articles are their title, subtitle, revisions, redirects, thumbnail, categories, coordinates (when it applies), text language, text direction and contentmodel (to consider only on wikitext ones).
49
+ Otherwise, the scraper enumerates pages in given namespaces (by default, all content namespaces) with the `allpages` generator, requesting one namespace content at a time (`action=query&generator=allpages&gapnamespace=xx`).
42
50
 
43
- ## Retrieving article HTML and rendering
51
+ The details we retrieve at this stage about every pages are their title, subtitle, revisions, redirects, thumbnail, categories, coordinates (when it applies), text language, text direction and contentmodel (to consider only on wikitext ones).
44
52
 
45
- In order to retrieve article HTML and render it, multiple solutions have been identified.
53
+ ## Retrieving page HTML and rendering
46
54
 
47
- As of today, 5 renderers (way to download article + render it to ZIM compatible HTML) are implemented:
55
+ In order to retrieve page HTML and render it, multiple solutions have been identified.
48
56
 
49
- - WikimediaDesktop
50
- - WikimediaMobile
51
- - RestApi
52
- - VisualEditor
53
- - ActionParse
57
+ As of today, 1 renderer (way to download page + render it to ZIM compatible HTML) is left implemented: ActionParse. Other renderers have been dropped in mwoffliner 2.0 because they were not used anymore, not implementing skin support and not providing added value compared to ActionParse render.
54
58
 
55
- WikimediaDesktop and WikimediaMobile are only available on Wikimedia Mediawikis.
59
+ ActionParse API is available in Mediawiki since 1.16.0 (2010).
56
60
 
57
- Availability of RestApi and VisualEditor is subject to Mediawiki admin decision to support it or not. RestApi is available by default but might be blocked by admin. VisualEditor is an extension which might be installed or not.
61
+ ActionParse renderer implements a thorough skin support (see below about skin) and needs only one HTTP query per page.
58
62
 
59
- ActionParse is available since 1.16.0 (2010) and is anyway a requirement for other APIs.
60
-
61
- Only ActionParse (most recent renderer at mwoffliner level) implements a thorough skin support (see below about skin).
62
-
63
- All renderers but ActionParse needs two HTTP queries: one to retrieve the article HTML and one to retrieve its metadata (to a 'simplified' ActionParse URL in fact).
64
-
65
- The article metadata we retrieve at this stage is the article are:
63
+ The page metadata we retrieve at this stage are:
66
64
 
67
65
  - `displaytitle` and `subtitle`
68
66
  - `headhtml`: used to extract proper CSS classes we have to set on `<html>` and `<body>` tags + some JS configuration variables which are part of the HTML head
69
67
  - `jsconfigvars`: other JS configuration variables which comes from other parts of the Mediawiki codebase
70
- - `modules`: list of JS and CSS modules to apply on current article
71
-
72
- Renderer is automatically selected based on its availability and mwoffliner own preference. ActionParse is the preferred renderer since 1.15.0 due to its general availability and support of skins.
68
+ - `modules`: list of JS and CSS modules to apply on current page
73
69
 
74
70
  ### ActionParse parser
75
71
 
76
- When using ActionParse renderer, we pass `usearticle=1`, which means that we ask the Mediawiki to use the parser configured for this article. This allows the scraper to retrieve article `text` (inner of the HTML containing the article itself) that is as close as possible to what is used online, either Parsoid or legacy parser (see https://www.mediawiki.org/wiki/Parsoid). This is mandatory because if we use a parser different than the one used online, we will de-facto get bugs which are "normal", either because parser still has a bug, or an extension is not compatible, or because Wikitext contains some workaround understandable only by a given parser. This should be avoided at all price if we want to have a versatile scraper capable of processing any Mediawiki. Even focusing only on Wikimedia wikis, not all of them have already transitioned to Parsoid for instance.
72
+ When using ActionParse renderer, we pass `usearticle=1`, which means that we ask the Mediawiki to use the parser configured for this page. This allows the scraper to retrieve page `text` (inner of the HTML containing the page itself) that is as close as possible to what is used online, either Parsoid or legacy parser (see https://www.mediawiki.org/wiki/Parsoid). This is mandatory because if we use a parser different than the one used online, we will de-facto get bugs which are "normal", either because parser still has a bug, or an extension is not compatible, or because Wikitext contains some workaround understandable only by a given parser. This should be avoided at all price if we want to have a versatile scraper capable of processing any Mediawiki. Even focusing only on Wikimedia wikis, not all of them have already transitioned to Parsoid for instance.
77
73
 
78
74
  ### Skins
79
75
 
80
- In Mediawikis, rendering of Wikitext into HTML works around a concept of skin. A skin is a mix of HTML template and CSS+JS dependencies. It defines both the visual appareance of the rendered Wikitext but also everything "around it".
76
+ In Mediawikis, rendering of Wikitext into HTML works around a concept of skin. A skin is a mix of HTML template and CSS+JS dependencies. It defines both the visual appearance of the rendered Wikitext but also everything "around it".
81
77
 
82
- Since most wikis have adapted their content to their skin (and vice versa), it is mostly mandatory to use the skin inside the ZIM, both for proper rendering and for a visual appareance similar to online website (users don't mind about technical details, they want the wiki to be the same inside the ZIM than online).
78
+ Since most wikis have adapted their content to their skin (and vice versa), it is mostly mandatory to use the skin inside the ZIM, both for proper rendering and for a visual appearance similar to online website (users don't mind about technical details, they want the wiki to be the same inside the ZIM than online).
83
79
 
84
80
  Skin detection is automated in mwoffliner for now (see https://github.com/openzim/mwoffliner/issues/2213).
85
81
 
86
- For now, only `vector` (legacy) and `vector-2022` are supported, and only with ActionParse renderer. Only `vector-2022` is the truely responsive skin, providing ultimate rendering on mostly all screen sizes.
82
+ For now, only `vector` (legacy) and `vector-2022` are supported, and only with ActionParse renderer. Only `vector-2022` is the truly responsive skin, providing ultimate rendering on mostly all screen sizes.
87
83
 
88
- With ActionParse renderer, other skins have a `fallback` skin implemented in the scraper. This means that the scraper will render the article HTML inside an HTML structure which looks like an expected structure (which comes from `vector` legacy). The consequence is that if the skins uses only article HTML structure to attach CSS rules or JS code, then everything will render fine. If the skins uses HTML structures coming from the "surrounding HTML" (headers, ...) then things will not apply 100% correctly.
84
+ With ActionParse renderer, other skins have a `fallback` skin implemented in the scraper. This means that the scraper will render the page HTML inside an HTML structure which looks like an expected structure (which comes from `vector` legacy). The consequence is that if the skins uses only page HTML structure to attach CSS rules or JS code, then everything will render fine. If the skins uses HTML structures coming from the "surrounding HTML" (headers, ...) then things will not apply 100% correctly.
89
85
 
90
86
  ### JS / CSS dependencies
91
87
 
92
- ActionParse API is returning the list of JS and CSS dependencies for a given article, by inspecting what Wikitext is using.
88
+ ActionParse API is returning the list of JS and CSS dependencies for a given page, by inspecting what Wikitext is using.
93
89
 
94
90
  The special `startup` JS module is missing from results because always used anyway.
95
91
 
@@ -1,21 +1,22 @@
1
- module.exports = class WiktionaryFR { // implements CustomProcessor
2
- async shouldKeepArticle(articleId, doc) {
3
- const frenchTitle = doc.querySelector(`#fr.sectionlangue`);
4
- return !!frenchTitle;
1
+ export default class WiktionaryFR {
2
+ // implements CustomProcessor
3
+ async shouldKeepPage(pageTitle, doc) {
4
+ const frenchTitle = doc.querySelector(`#fr.sectionlangue`)
5
+ return !!frenchTitle
6
+ }
7
+ async preProcessPage(pageTitle, doc) {
8
+ const nonFrenchTitles = Array.from(doc.querySelectorAll(`.sectionlangue:not(#fr)`))
9
+ for (const title of nonFrenchTitles) {
10
+ title.closest('details').remove()
5
11
  }
6
- async preProcessArticle(articleId, doc) {
7
- const nonFrenchTitles = Array.from(doc.querySelectorAll(`.sectionlangue:not(#fr)`));
8
- for (const title of nonFrenchTitles) {
9
- title.closest('details').remove();
10
- }
11
12
 
12
- const h4titles = Array.from(doc.querySelectorAll(`h4`));
13
- for (const h4title of h4titles) {
14
- h4title.closest('details').remove();
15
- }
16
- //Remove h2 summary title
17
- doc.querySelector('h2').closest('summary').setAttribute('style', 'display:none! important')
18
-
19
- return doc;
13
+ const h4titles = Array.from(doc.querySelectorAll(`h4`))
14
+ for (const h4title of h4titles) {
15
+ h4title.closest('details').remove()
20
16
  }
17
+ //Remove h2 summary title
18
+ doc.querySelector('h2').closest('summary').setAttribute('style', 'display:none! important')
19
+
20
+ return doc
21
+ }
21
22
  }