sedd 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sedd/__init__.py +0 -0
- sedd/__main__.py +1 -0
- sedd/cli.py +69 -0
- sedd/driver.py +62 -0
- sedd/main.py +263 -0
- sedd/utils.py +85 -0
- sedd-0.0.0.dist-info/METADATA +91 -0
- sedd-0.0.0.dist-info/RECORD +11 -0
- sedd-0.0.0.dist-info/WHEEL +5 -0
- sedd-0.0.0.dist-info/licenses/LICENSE +32 -0
- sedd-0.0.0.dist-info/top_level.txt +1 -0
sedd/__init__.py
ADDED
|
File without changes
|
sedd/__main__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .main import *
|
sedd/cli.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
|
|
3
|
+
from os import getcwd
|
|
4
|
+
from os.path import join
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class SEDDCLIArgs(argparse.Namespace):
|
|
8
|
+
skip_loaded: bool
|
|
9
|
+
keep_consent: bool
|
|
10
|
+
output_dir: str
|
|
11
|
+
dry_run: bool
|
|
12
|
+
disable_undetected: bool
|
|
13
|
+
verbose: bool
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
parser = argparse.ArgumentParser(
|
|
17
|
+
prog="sedd",
|
|
18
|
+
description="Automatic (unofficial) SE data dump downloader for the anti-community data dump format",
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
parser.add_argument(
|
|
22
|
+
"-s", "--skip-loaded",
|
|
23
|
+
required=False,
|
|
24
|
+
default=False,
|
|
25
|
+
action="store_true",
|
|
26
|
+
dest="skip_loaded"
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
parser.add_argument(
|
|
30
|
+
"-k", "--keep-consent",
|
|
31
|
+
required=False,
|
|
32
|
+
dest="keep_consent",
|
|
33
|
+
action="store_true",
|
|
34
|
+
default=False
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
parser.add_argument(
|
|
38
|
+
"-o", "--outputDir",
|
|
39
|
+
required=False,
|
|
40
|
+
dest="output_dir",
|
|
41
|
+
default=join(getcwd(), "downloads")
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
parser.add_argument(
|
|
45
|
+
"-g", "--disable-undetected-geckodriver",
|
|
46
|
+
required=False,
|
|
47
|
+
dest="disable_undetected",
|
|
48
|
+
action="store_true",
|
|
49
|
+
default=False
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
parser.add_argument(
|
|
53
|
+
"--dry-run",
|
|
54
|
+
required=False,
|
|
55
|
+
default=False,
|
|
56
|
+
action="store_true",
|
|
57
|
+
dest="dry_run"
|
|
58
|
+
)
|
|
59
|
+
parser.add_argument(
|
|
60
|
+
"-v",
|
|
61
|
+
required=False,
|
|
62
|
+
default=False,
|
|
63
|
+
action="store_true",
|
|
64
|
+
dest="verbose"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def parse_cli_args() -> SEDDCLIArgs:
|
|
69
|
+
return parser.parse_args()
|
sedd/driver.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
from os import path, makedirs
|
|
2
|
+
from urllib import request
|
|
3
|
+
from json import dumps
|
|
4
|
+
from uuid import uuid4
|
|
5
|
+
import platform
|
|
6
|
+
|
|
7
|
+
from selenium import webdriver
|
|
8
|
+
from selenium.webdriver.firefox.options import Options
|
|
9
|
+
from undetected_geckodriver import Firefox as UFirefox
|
|
10
|
+
|
|
11
|
+
from .config import SEDDConfig
|
|
12
|
+
from .ubo import init_ubo_settings
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def init_output_dir(output_dir: str):
|
|
16
|
+
if not path.exists(output_dir):
|
|
17
|
+
makedirs(output_dir)
|
|
18
|
+
|
|
19
|
+
print(output_dir)
|
|
20
|
+
|
|
21
|
+
return output_dir
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def init_firefox_driver(config: SEDDConfig, disable_undetected: bool, output_dir: str):
|
|
25
|
+
options = Options()
|
|
26
|
+
options.enable_downloads = True
|
|
27
|
+
options.set_preference("browser.download.folderList", 2)
|
|
28
|
+
options.set_preference("browser.download.manager.showWhenStarting", False)
|
|
29
|
+
options.set_preference("browser.download.dir", output_dir)
|
|
30
|
+
options.set_preference(
|
|
31
|
+
"browser.helperApps.neverAsk.saveToDisk", "application/x-gzip"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
# our own uuid for uBO so as we don't need to do the dance of inspecing internals
|
|
35
|
+
ubo_internal_uuid = f"{uuid4()}"
|
|
36
|
+
|
|
37
|
+
options.set_preference("extensions.webextensions.uuids", dumps(
|
|
38
|
+
{"uBlock0@raymondhill.net": ubo_internal_uuid}))
|
|
39
|
+
|
|
40
|
+
is_apple = platform.system() == "Darwin"
|
|
41
|
+
use_undetected = not disable_undetected and not is_apple
|
|
42
|
+
if use_undetected:
|
|
43
|
+
print("Using undetected-geckodriver")
|
|
44
|
+
browser = UFirefox(options = options)
|
|
45
|
+
else:
|
|
46
|
+
print("Warning: using standard geckodriver. Cloudflare may perpetually block you")
|
|
47
|
+
if is_apple:
|
|
48
|
+
print("This option is forced on macOS. For undetected_geckodriver, "
|
|
49
|
+
"run the downloader in a Linux or Windows environment.")
|
|
50
|
+
browser = webdriver.Firefox(options=options)
|
|
51
|
+
|
|
52
|
+
ubo_download_url = config.get_ubo_download_url()
|
|
53
|
+
|
|
54
|
+
if not path.exists("ubo.xpi"):
|
|
55
|
+
print(f"Downloading uBO from: {ubo_download_url}")
|
|
56
|
+
request.urlretrieve(ubo_download_url, "ubo.xpi")
|
|
57
|
+
|
|
58
|
+
ubo_id = browser.install_addon("ubo.xpi", temporary=True)
|
|
59
|
+
|
|
60
|
+
init_ubo_settings(browser, config, ubo_internal_uuid)
|
|
61
|
+
|
|
62
|
+
return browser, ubo_id
|
sedd/main.py
ADDED
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
from selenium.webdriver.common.by import By
|
|
2
|
+
from selenium.webdriver.firefox.webdriver import WebDriver
|
|
3
|
+
from selenium.common.exceptions import NoSuchElementException
|
|
4
|
+
from typing import Dict
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
from time import sleep
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
import sys
|
|
11
|
+
from traceback import print_exception
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
from .cli import parse_cli_args
|
|
15
|
+
from .config import load_sedd_config
|
|
16
|
+
from .data import sites
|
|
17
|
+
from .meta import notifications
|
|
18
|
+
from .watcher.observer import register_pending_downloads_observer
|
|
19
|
+
from . import utils
|
|
20
|
+
|
|
21
|
+
from .driver import init_output_dir, init_firefox_driver
|
|
22
|
+
import logging
|
|
23
|
+
|
|
24
|
+
args = parse_cli_args()
|
|
25
|
+
|
|
26
|
+
if args.verbose:
|
|
27
|
+
logging.basicConfig(level=logging.DEBUG)
|
|
28
|
+
|
|
29
|
+
sedd_config = load_sedd_config()
|
|
30
|
+
|
|
31
|
+
output_dir = init_output_dir(args.output_dir)
|
|
32
|
+
|
|
33
|
+
browser, ubo_id = init_firefox_driver(
|
|
34
|
+
sedd_config,
|
|
35
|
+
args.disable_undetected,
|
|
36
|
+
output_dir
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def kill_cookie_shit(browser: WebDriver):
|
|
41
|
+
sleep(3)
|
|
42
|
+
browser.execute_script(
|
|
43
|
+
"""let elem = document.getElementById("onetrust-banner-sdk"); if (elem) { elem.parentNode.removeChild(elem); }""")
|
|
44
|
+
sleep(1)
|
|
45
|
+
|
|
46
|
+
def check_cloudflare_intercept(browser: WebDriver):
|
|
47
|
+
if browser.title == "Just a moment...":
|
|
48
|
+
print("CF verification hit. Trying soft workaround")
|
|
49
|
+
sleep(15)
|
|
50
|
+
|
|
51
|
+
if (browser.title == "Just a moment..."):
|
|
52
|
+
print("Irrecoverable state suspected; captcha solving likely required")
|
|
53
|
+
notifications.notify("CloudFlare verification hit; auto-verification failed. Please complete the captcha", sedd_config)
|
|
54
|
+
else:
|
|
55
|
+
print("Auto-recovered from CF wall")
|
|
56
|
+
return
|
|
57
|
+
|
|
58
|
+
while browser.title == "Just a moment...":
|
|
59
|
+
print("Still stuck on CF verification. Waiting for 10 seconds")
|
|
60
|
+
sleep(10)
|
|
61
|
+
|
|
62
|
+
def is_logged_in(browser: WebDriver, site: str):
|
|
63
|
+
url = f"{site}/users/current"
|
|
64
|
+
browser.get(url)
|
|
65
|
+
sleep(1)
|
|
66
|
+
check_cloudflare_intercept(browser)
|
|
67
|
+
|
|
68
|
+
return "/users/" in browser.current_url
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def login_or_create(browser: WebDriver, site: str):
|
|
72
|
+
if is_logged_in(browser, site):
|
|
73
|
+
print("Already logged in")
|
|
74
|
+
else:
|
|
75
|
+
print("Not logged in and/or not registered. Logging in now")
|
|
76
|
+
while True:
|
|
77
|
+
browser.get(f"{site}/users/login")
|
|
78
|
+
check_cloudflare_intercept(browser)
|
|
79
|
+
|
|
80
|
+
if "?newreg" in browser.current_url:
|
|
81
|
+
print(f"Auto-created {site} without login needed")
|
|
82
|
+
break
|
|
83
|
+
|
|
84
|
+
email_elem = browser.find_element(By.ID, "email")
|
|
85
|
+
password_elem = browser.find_element(By.ID, "password")
|
|
86
|
+
email_elem.send_keys(sedd_config.email)
|
|
87
|
+
password_elem.send_keys(sedd_config.password)
|
|
88
|
+
retryLogin = False
|
|
89
|
+
|
|
90
|
+
curr_url = browser.current_url
|
|
91
|
+
browser.find_element(By.ID, "submit-button").click()
|
|
92
|
+
|
|
93
|
+
try:
|
|
94
|
+
elem = browser.find_element(By.CSS_SELECTOR, "#login-form > .js-error-message")
|
|
95
|
+
if elem is not None:
|
|
96
|
+
print("Login failed quietly. Retrying")
|
|
97
|
+
continue
|
|
98
|
+
except:
|
|
99
|
+
# No error element
|
|
100
|
+
pass
|
|
101
|
+
|
|
102
|
+
check_cloudflare_intercept(browser)
|
|
103
|
+
|
|
104
|
+
while browser.current_url == curr_url:
|
|
105
|
+
sleep(3)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
captcha_walled = False
|
|
109
|
+
while "/nocaptcha" in browser.current_url:
|
|
110
|
+
if not captcha_walled:
|
|
111
|
+
captcha_walled = True
|
|
112
|
+
|
|
113
|
+
notifications.notify(
|
|
114
|
+
"Captcha wall hit during login", sedd_config
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
sleep(10)
|
|
118
|
+
|
|
119
|
+
if captcha_walled or retryLogin:
|
|
120
|
+
continue
|
|
121
|
+
|
|
122
|
+
if not is_logged_in(browser, site):
|
|
123
|
+
raise RuntimeError("Login failed")
|
|
124
|
+
|
|
125
|
+
break
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def download_data_dump(browser: WebDriver, site: str, meta_url: str, etags: Dict[str, str]):
|
|
129
|
+
print(f"Downloading data dump from {site}")
|
|
130
|
+
|
|
131
|
+
def _exec_download(browser: WebDriver):
|
|
132
|
+
if args.keep_consent:
|
|
133
|
+
print('Consent dialog will not be auto-removed')
|
|
134
|
+
else:
|
|
135
|
+
kill_cookie_shit(browser)
|
|
136
|
+
|
|
137
|
+
try:
|
|
138
|
+
checkbox = browser.find_element(By.ID, "datadump-agree-checkbox")
|
|
139
|
+
btn = browser.find_element(By.ID, "datadump-download-button")
|
|
140
|
+
except NoSuchElementException:
|
|
141
|
+
raise RuntimeError(f"Bad site: {site}")
|
|
142
|
+
|
|
143
|
+
if args.dry_run:
|
|
144
|
+
return
|
|
145
|
+
|
|
146
|
+
browser.execute_script("""
|
|
147
|
+
(function() {
|
|
148
|
+
let oldFetch = window.fetch;
|
|
149
|
+
window.fetch = (url, opts) => {
|
|
150
|
+
let promise = oldFetch(url, opts);
|
|
151
|
+
|
|
152
|
+
if (url.includes("/link")) {
|
|
153
|
+
promise.then(res => {
|
|
154
|
+
res.clone().json().then(json => {
|
|
155
|
+
window.extractedUrl = json["url"];
|
|
156
|
+
console.log(extractedUrl);
|
|
157
|
+
});
|
|
158
|
+
return res;
|
|
159
|
+
});
|
|
160
|
+
return new Promise(resolve => setTimeout(resolve, 4000))
|
|
161
|
+
.then(_ => promise);
|
|
162
|
+
}
|
|
163
|
+
return promise;
|
|
164
|
+
};
|
|
165
|
+
})();
|
|
166
|
+
""")
|
|
167
|
+
|
|
168
|
+
checkbox.click()
|
|
169
|
+
sleep(1)
|
|
170
|
+
btn.click()
|
|
171
|
+
check_cloudflare_intercept(browser)
|
|
172
|
+
sleep(2)
|
|
173
|
+
url = browser.execute_script("return window.extractedUrl;")
|
|
174
|
+
utils.extract_etag(url, etags)
|
|
175
|
+
|
|
176
|
+
sleep(5)
|
|
177
|
+
|
|
178
|
+
main_loaded = utils.is_file_downloaded(args.output_dir, site)
|
|
179
|
+
meta_loaded = utils.is_file_downloaded(args.output_dir, meta_url)
|
|
180
|
+
|
|
181
|
+
if not args.skip_loaded or not main_loaded or not meta_loaded:
|
|
182
|
+
if args.skip_loaded and main_loaded:
|
|
183
|
+
pass
|
|
184
|
+
else:
|
|
185
|
+
browser.get(f"{site}/users/data-dump-access/current")
|
|
186
|
+
check_cloudflare_intercept(browser)
|
|
187
|
+
|
|
188
|
+
if not args.dry_run:
|
|
189
|
+
utils.archive_file(args.output_dir, site)
|
|
190
|
+
|
|
191
|
+
_exec_download(browser)
|
|
192
|
+
|
|
193
|
+
if args.skip_loaded and meta_loaded:
|
|
194
|
+
pass
|
|
195
|
+
else:
|
|
196
|
+
browser.get(f"{meta_url}/users/data-dump-access/current")
|
|
197
|
+
check_cloudflare_intercept(browser)
|
|
198
|
+
|
|
199
|
+
if not args.dry_run:
|
|
200
|
+
utils.archive_file(args.output_dir, meta_url)
|
|
201
|
+
|
|
202
|
+
_exec_download(browser)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
etags: Dict[str, str] = {}
|
|
206
|
+
|
|
207
|
+
try:
|
|
208
|
+
state, observer = register_pending_downloads_observer(args.output_dir)
|
|
209
|
+
|
|
210
|
+
for site in sites.sites:
|
|
211
|
+
if site not in ["https://meta.stackexchange.com", "https://stackapps.com"]:
|
|
212
|
+
# https://regex101.com/r/kG6nTN/1
|
|
213
|
+
meta_url = re.sub(
|
|
214
|
+
r"(https://(?:[^.]+\.(?=stackexchange))?)", r"\1meta.", site)
|
|
215
|
+
|
|
216
|
+
main_loaded = utils.is_file_downloaded(args.output_dir, site)
|
|
217
|
+
meta_loaded = utils.is_file_downloaded(args.output_dir, meta_url)
|
|
218
|
+
|
|
219
|
+
if args.skip_loaded and main_loaded and meta_loaded:
|
|
220
|
+
pass
|
|
221
|
+
else:
|
|
222
|
+
print(f"Extracting from {site}...")
|
|
223
|
+
|
|
224
|
+
login_or_create(browser, site)
|
|
225
|
+
download_data_dump(
|
|
226
|
+
browser,
|
|
227
|
+
site,
|
|
228
|
+
meta_url,
|
|
229
|
+
etags
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
if observer:
|
|
233
|
+
pending = state.size()
|
|
234
|
+
|
|
235
|
+
print(f"Waiting for {pending} download{'s'[:pending^1]} to complete")
|
|
236
|
+
|
|
237
|
+
while True:
|
|
238
|
+
if state.empty():
|
|
239
|
+
observer.stop()
|
|
240
|
+
browser.quit()
|
|
241
|
+
|
|
242
|
+
utils.cleanup_archive(args.output_dir)
|
|
243
|
+
break
|
|
244
|
+
else:
|
|
245
|
+
sleep(1)
|
|
246
|
+
|
|
247
|
+
except KeyboardInterrupt:
|
|
248
|
+
pass
|
|
249
|
+
|
|
250
|
+
except:
|
|
251
|
+
exception = sys.exc_info()
|
|
252
|
+
|
|
253
|
+
try:
|
|
254
|
+
print_exception(exception)
|
|
255
|
+
except:
|
|
256
|
+
print(exception)
|
|
257
|
+
|
|
258
|
+
browser.quit()
|
|
259
|
+
finally:
|
|
260
|
+
# TODO: replace with validation once downloading is verified done
|
|
261
|
+
# (or export for separate, later verification)
|
|
262
|
+
# Though keeping it here, removing files and re-running downloads feels like a better idea
|
|
263
|
+
print(etags)
|
sedd/utils.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
from typing import Dict
|
|
2
|
+
import requests as r
|
|
3
|
+
from urllib.parse import urlparse
|
|
4
|
+
import os.path
|
|
5
|
+
import re
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
from .data.files_map import files_map, inverse_files_map
|
|
9
|
+
from .data.sites import sites
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def extract_etag(url: str, etags: Dict[str, str]):
|
|
13
|
+
res = r.get(
|
|
14
|
+
url,
|
|
15
|
+
stream=True
|
|
16
|
+
)
|
|
17
|
+
if res.status_code != 200:
|
|
18
|
+
raise RuntimeError(f"Panic: failed to get {url}: {res.status_code}")
|
|
19
|
+
|
|
20
|
+
etag = res.headers["ETag"]
|
|
21
|
+
res.close()
|
|
22
|
+
|
|
23
|
+
parsed_url = urlparse(url)
|
|
24
|
+
path = parsed_url.path
|
|
25
|
+
filename = os.path.basename(path)
|
|
26
|
+
|
|
27
|
+
etags[filename] = etag
|
|
28
|
+
|
|
29
|
+
print(f"ETag for {filename}: {etag}")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def get_file_name(site_or_url: str) -> str:
|
|
33
|
+
domain = re.sub(r'https://', '', site_or_url)
|
|
34
|
+
|
|
35
|
+
try:
|
|
36
|
+
file_name = files_map[domain]
|
|
37
|
+
return f'{file_name}.7z'
|
|
38
|
+
except KeyError:
|
|
39
|
+
return f'{domain}.7z'
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def is_dump_file(file_name: str) -> bool:
|
|
43
|
+
file_name = re.sub(r'\.7z$', '', file_name)
|
|
44
|
+
|
|
45
|
+
try:
|
|
46
|
+
inverse_files_map[file_name]
|
|
47
|
+
except KeyError:
|
|
48
|
+
origin = f'https://{file_name}'
|
|
49
|
+
return origin in sites
|
|
50
|
+
|
|
51
|
+
return True
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def check_file(base_path: str, file_name: str) -> bool:
|
|
55
|
+
try:
|
|
56
|
+
res = os.stat(os.path.join(base_path, file_name))
|
|
57
|
+
return res.st_size > 0
|
|
58
|
+
except FileNotFoundError:
|
|
59
|
+
return False
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def archive_file(base_path: str, site_or_url: str) -> None:
|
|
63
|
+
try:
|
|
64
|
+
file_name = get_file_name(site_or_url)
|
|
65
|
+
file_path = os.path.join(base_path, file_name)
|
|
66
|
+
os.rename(file_path, f"{file_path}.old")
|
|
67
|
+
except FileNotFoundError:
|
|
68
|
+
pass
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def cleanup_archive(base_path: str) -> None:
|
|
72
|
+
try:
|
|
73
|
+
file_entries = os.listdir(base_path)
|
|
74
|
+
|
|
75
|
+
for entry in file_entries:
|
|
76
|
+
if entry.endswith('.old'):
|
|
77
|
+
entry_path = os.path.join(base_path, entry)
|
|
78
|
+
os.remove(entry_path)
|
|
79
|
+
except:
|
|
80
|
+
print(sys.exc_info())
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def is_file_downloaded(base_path: str, site_or_url: str) -> bool:
|
|
84
|
+
file_name = get_file_name(site_or_url)
|
|
85
|
+
return check_file(base_path, file_name)
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sedd
|
|
3
|
+
Version: 0.0.0
|
|
4
|
+
Summary: Unofficial, community-made tool for downloading the Stack Exchange data dumps
|
|
5
|
+
Author-email: LunarWatcher <oliviawolfie@pm.me>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/LunarWatcher/se-data-dump-transformer
|
|
8
|
+
Project-URL: Documentation, https://github.com/LunarWatcher/se-data-dump-transformer
|
|
9
|
+
Project-URL: Repository, https://github.com/LunarWatcher/se-data-dump-transformer.git
|
|
10
|
+
Project-URL: Issues, https://github.com/LunarWatcher/se-data-dump-transformer/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/LunarWatcher/se-data-dump-transformer/releases
|
|
12
|
+
Keywords: Stack Exchange,data dump,Stack Exchange data dump,downloader
|
|
13
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
14
|
+
Classifier: Environment :: Console
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Topic :: System :: Archiving
|
|
17
|
+
Classifier: Topic :: Utilities
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
23
|
+
Classifier: Operating System :: MacOS
|
|
24
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
25
|
+
Requires-Python: >=3.10
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: selenium==4.32.0
|
|
29
|
+
Requires-Dist: undetected-geckodriver-lw>=2.1.0
|
|
30
|
+
Requires-Dist: desktop-notifier==5.0.1
|
|
31
|
+
Requires-Dist: watchdog==4.0.2
|
|
32
|
+
Requires-Dist: requests
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# SE Data Dump Downloader
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
For more comprehensive information, please read the [main README](https://github.com/LunarWatcher/se-data-dump-transformer/tree/master) on GitHub. This README contains an abridged version of the main README specifically aimed at Pypi users.
|
|
39
|
+
|
|
40
|
+
For usage problems not listed in this readme, see the main README. If no information exists, please open an issue on GitHub - keeping the tool accessible to everyone is a priority.
|
|
41
|
+
|
|
42
|
+
---
|
|
43
|
+
|
|
44
|
+
The SE Data Dump Downloader (abbreviated `sedd`) is a command line Selenium-based utility for downloading the entire Stack Exchange data dump in their new [anti-community format](https://stackoverflow.com/help/data-dumps), since they decided not to bother providing an official "download all" button. It's one of two components that operate on the data dump in the second project, the other being the (non-python-based) SE data dump transformer - a project that converts the data dump from the not-so-useful official `.xml` format to some other formats. The pypi package is exclusively for the downloader, and does not ship with a copy of the transformer. See the main README if you're looking for the transformer.
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
For the pypi version, you can download it with:
|
|
48
|
+
```python3
|
|
49
|
+
pip3 install sedd
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Note that there are some additional steps before you can start using it, that are detailed in this README.
|
|
53
|
+
|
|
54
|
+
## Configuration
|
|
55
|
+
|
|
56
|
+
`sedd` requires a special `config.json` file in the current working directory. There's a template available [on GitHub](https://github.com/LunarWatcher/se-data-dump-transformer/blob/master/config.example.json).
|
|
57
|
+
|
|
58
|
+
The only two fields you _need_ to fill out in the template is the email and password fields with credentials for a Stack Exchange account. You need to be logged in to download the data dumps, so the downloader needs the credentials to log in on your behalf. It doesn't matter if you're logged into SE elsewhere, as Selenium automatically creates a blank profile every time it starts, which won't include any cookies from SE, which means login is required.
|
|
59
|
+
|
|
60
|
+
> [!tip]
|
|
61
|
+
>
|
|
62
|
+
> The downloader can automatically create new accounts in the network for you, if you don't have all 180-whatever accounts on every site in the network already. You can also create these by hand if you prefer for some reason, but you are not required to have all 180+ accounts before using the downloader.
|
|
63
|
+
|
|
64
|
+
## System requirements and pitfalls
|
|
65
|
+
|
|
66
|
+
`sedd` is exclusively Firefox-based, due to Chromium completely gutting support for uBlock Origin and custom filters. You need Firefox installed on your system to use `sedd`.
|
|
67
|
+
|
|
68
|
+
> [!note]
|
|
69
|
+
> On Linux and Windows-based systems, geckodriver is [slightly modified](https://pypi.org/project/undetected-geckodriver-lw/). This is an anti-anti-bot measure meant to prevent Cloudflare loops. If you're on macOS and get sent in a captcha loop, it's recommended you switch to Windows or Linux - a Linux VM is also an option if you have no way out of Apple's closed-down ecosystem.
|
|
70
|
+
|
|
71
|
+
Note that Ubuntu users, or other people who (for whatever reason) choose to use the Snap version of Firefox, have to jump through some extra hoops. The native version of Firefox is strongly encouraged, but if you run into problems with the snap version of Firefox and can't or won't switch, you need to define `export SE_GECKODRIVER=/snap/bin/geckodriver`. Selenium can and will find the snap version of `geckodriver` on its own, but for reasons I simply don't understand, it will still fail with several arbitrary errors.
|
|
72
|
+
|
|
73
|
+
### Cloudflare issues or download issues.
|
|
74
|
+
|
|
75
|
+
Stack Exchange has configured Cloudflare to be _highly_ aggressive, especially to certain countries. You will almost certainly run into captchas, and the downloader is designed to deal with this. After an initial attempt to solve the captcha on its own, you'll be notified (provided you don't disable the notification provider in `config.json`) and asked to solve it manually.
|
|
76
|
+
|
|
77
|
+
If, at this point, it appears to succeed, but you're redirected back to a full-screen Cloudflare captcha wall, you've likely run into a Cloudflare loop. See [the main README](https://github.com/LunarWatcher/se-data-dump-transformer/tree/master?tab=readme-ov-file#cloudflare-loops) for further help. If this doesn't help, please open an issue.
|
|
78
|
+
|
|
79
|
+
If the downloads start fine, but later suddenly fail for no good reason, you're likely running into general download instability. This especially applies to `stackoverflow.com.7z`, as its massive size simply increases the chance you wait for it long enough that it flakes out. See [the main README](https://github.com/LunarWatcher/se-data-dump-transformer/tree/master?tab=readme-ov-file#download-instability-particularly-of-stackoverflowcom7z) for further help.
|
|
80
|
+
|
|
81
|
+
The "Warnings" section in the README may contain additional information about other failure modes not listed here in the future.
|
|
82
|
+
|
|
83
|
+
## Using the downloader
|
|
84
|
+
|
|
85
|
+
With `./config.json` in the current working directory and Firefox installed, you can now run the downloader with:
|
|
86
|
+
```python3
|
|
87
|
+
sedd
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
For command line flags, see `sedd --help`, or [the main readme](https://github.com/LunarWatcher/se-data-dump-transformer/tree/master?tab=readme-ov-file#cli-options).
|
|
91
|
+
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
sedd/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
sedd/__main__.py,sha256=8D5a8oKgqd6WA1RUkiKCn4l_PVemtyuckxQut0vDHXM,20
|
|
3
|
+
sedd/cli.py,sha256=PRAwTyrY5csFHsPvBgriPxziGAhGqwSj0XGmnm4XRfQ,1280
|
|
4
|
+
sedd/driver.py,sha256=jfxck8Qy8l1JjkvS55hHXTLsX0ouQjGGLZCA7atUWXI,2096
|
|
5
|
+
sedd/main.py,sha256=Dy0jmHOgt-qr759Kq8lnTmNrKdgPP1ONabeLn7mKqCg,7851
|
|
6
|
+
sedd/utils.py,sha256=niSYc5yDmnPWzFsUu_c3UB16zeQVAIjYX9HzrM17Lrg,2055
|
|
7
|
+
sedd-0.0.0.dist-info/licenses/LICENSE,sha256=1jbxcNcOnucv3gZNjgsGmAvleNJnhH5LkVUpYPqYSOA,1487
|
|
8
|
+
sedd-0.0.0.dist-info/METADATA,sha256=toTxrEAmBjabji1AqPKWhxXRj3SbzIKiHGM2D6rkw-0,6724
|
|
9
|
+
sedd-0.0.0.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
10
|
+
sedd-0.0.0.dist-info/top_level.txt,sha256=LW8YTI3wZnjHrWtSdboYe4QHk2pHgkRE22bje4NKxhU,5
|
|
11
|
+
sedd-0.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
Note that the MIT license only applies to the software. Any downloaded
|
|
2
|
+
or generated data dumps are under various versions of CC-By-SA:
|
|
3
|
+
|
|
4
|
+
* CC-By-SA 2.5: https://creativecommons.org/licenses/by-sa/2.5/
|
|
5
|
+
* CC-By-SA 3.0: https://creativecommons.org/licenses/by-sa/3.0/
|
|
6
|
+
* CC-By-SA 4.0: https://creativecommons.org/licenses/by-sa/4.0/
|
|
7
|
+
|
|
8
|
+
For more details, and date ranges for the licenses, see
|
|
9
|
+
https://stackoverflow.com/help/licensing
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
Copyright © 2024 Olivia
|
|
14
|
+
|
|
15
|
+
Permission is hereby granted, free of charge, to any person obtaining
|
|
16
|
+
a copy of this software and associated documentation files (the "Software"),
|
|
17
|
+
to deal in the Software without restriction, including without limitation
|
|
18
|
+
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
|
19
|
+
and/or sell copies of the Software, and to permit persons to whom the
|
|
20
|
+
Software is furnished to do so, subject to the following conditions:
|
|
21
|
+
|
|
22
|
+
The above copyright notice and this permission notice shall be included
|
|
23
|
+
in all copies or substantial portions of the Software.
|
|
24
|
+
|
|
25
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
|
26
|
+
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES
|
|
27
|
+
OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
|
28
|
+
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM,
|
|
29
|
+
DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
|
30
|
+
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE
|
|
31
|
+
OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
32
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
sedd
|