sedd 2.2.2__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sedd-2.2.2/sedd.egg-info → sedd-2.4.0}/PKG-INFO +1 -1
- {sedd-2.2.2 → sedd-2.4.0}/README.md +4 -1
- {sedd-2.2.2 → sedd-2.4.0}/sedd/cli.py +18 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/main.py +146 -26
- {sedd-2.2.2 → sedd-2.4.0}/sedd/ubo/ubo.py +1 -1
- {sedd-2.2.2 → sedd-2.4.0}/sedd/utils.py +3 -1
- sedd-2.4.0/sedd/watcher/recovery.py +6 -0
- {sedd-2.2.2 → sedd-2.4.0/sedd.egg-info}/PKG-INFO +1 -1
- {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/SOURCES.txt +1 -0
- {sedd-2.2.2 → sedd-2.4.0}/LICENSE +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/README-python.md +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/pyproject.toml +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/__init__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/__main__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/config/__init__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/config/config.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/config/defaults.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/config/typings.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/data/__init__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/data/files_map.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/data/sites.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/driver.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/meta/__init__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/meta/notifications.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/pipscript.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/ubo/__init__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/ubo/utils.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/__init__.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/handler.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/observer.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/state.py +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/dependency_links.txt +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/entry_points.txt +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/requires.txt +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/top_level.txt +0 -0
- {sedd-2.2.2 → sedd-2.4.0}/setup.cfg +0 -0
|
@@ -169,6 +169,7 @@ Exractor CLI supports the following configuration options:
|
|
|
169
169
|
| `-v` | - | Optional | `false` | Whether or not to enable verbose logging. You do not want to enable this unless you're diagnosing a problem with sedd - the verbose logging includes low-level Selenium output. |
|
|
170
170
|
| - | `--detect <last upload date>` | Optional | - | Enables a mode where sedd checks for the upload timestamp to change, and only then begins the download. This has several caveats. Please read [docs/Automated downloads](docs/Automated downloads.md) for more information before using this feature. |
|
|
171
171
|
| `-u` | `--unsupervised` | Optional | false | Rather than notify you about failures and asking you to resolve them, hard failures that would require manual supervision throw an exception instead. |
|
|
172
|
+
| `-N` | `--no-wipe-part-files` | Optional | false | Don't wipe `.part` files in the download directory on boot. Only ever useful if you're doing cursed shit. Disables automatic download error recovery, and may result in the download completion watcher getting stuck |
|
|
172
173
|
|
|
173
174
|
#### Captchas and other misc. barriers
|
|
174
175
|
|
|
@@ -299,7 +300,9 @@ cmake --build . -j 8 --config Release
|
|
|
299
300
|
.\sedd-transformer.exe -i ..\..\downloads -t [formatter type]
|
|
300
301
|
```
|
|
301
302
|
|
|
302
|
-
Pass `--help` to see the available formatters for your current version of the data dump transformer.
|
|
303
|
+
Pass `--help` to see the available formatters for your current version of the data dump transformer. There's also some filters available for use, that change the data.
|
|
304
|
+
|
|
305
|
+
If you only want to filter the data, but keep the XML format, you can use `-t xml` to convert back to XML.
|
|
303
306
|
|
|
304
307
|
### Supported transformers
|
|
305
308
|
|
|
@@ -14,6 +14,8 @@ class SEDDCLIArgs(argparse.Namespace):
|
|
|
14
14
|
detect: str | None
|
|
15
15
|
unsupervised: bool
|
|
16
16
|
|
|
17
|
+
wipe_part_files: bool
|
|
18
|
+
|
|
17
19
|
|
|
18
20
|
parser = argparse.ArgumentParser(
|
|
19
21
|
prog="sedd",
|
|
@@ -78,6 +80,22 @@ parser.add_argument(
|
|
|
78
80
|
)
|
|
79
81
|
)
|
|
80
82
|
|
|
83
|
+
parser.add_argument(
|
|
84
|
+
"-N,--no-wipe-part-files",
|
|
85
|
+
required=False,
|
|
86
|
+
default=True,
|
|
87
|
+
action="store_false",
|
|
88
|
+
dest="wipe_part_files",
|
|
89
|
+
help=(
|
|
90
|
+
"When supplied, .part files in the folder won't be wiped. "
|
|
91
|
+
"By default, wiping part files is enabled, as leaving it interferes "
|
|
92
|
+
"with a download error recovery mechanism. Supplying this flag "
|
|
93
|
+
"disables download error recovery, as the system has no way to "
|
|
94
|
+
"differentiate between a .part file that was there before the "
|
|
95
|
+
"download started"
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
|
|
81
99
|
parser.add_argument(
|
|
82
100
|
"--detect",
|
|
83
101
|
required=False,
|
|
@@ -2,11 +2,12 @@ from selenium.webdriver.common.by import By
|
|
|
2
2
|
from selenium.webdriver.firefox.webdriver import WebDriver
|
|
3
3
|
from selenium.common.exceptions import NoSuchElementException
|
|
4
4
|
from typing import Dict
|
|
5
|
+
import glob
|
|
5
6
|
|
|
6
|
-
|
|
7
|
-
from time import sleep
|
|
7
|
+
from time import sleep, monotonic
|
|
8
8
|
|
|
9
9
|
import re
|
|
10
|
+
import os
|
|
10
11
|
import sys
|
|
11
12
|
import random
|
|
12
13
|
from traceback import print_exception
|
|
@@ -17,6 +18,8 @@ from .config import load_sedd_config
|
|
|
17
18
|
from .data import sites
|
|
18
19
|
from .meta import notifications
|
|
19
20
|
from .watcher.observer import register_pending_downloads_observer
|
|
21
|
+
from .watcher.state import DownloadState
|
|
22
|
+
from .watcher.recovery import LastState
|
|
20
23
|
from . import utils
|
|
21
24
|
|
|
22
25
|
from .driver import init_output_dir, init_firefox_driver
|
|
@@ -190,7 +193,7 @@ def login_or_create(browser: WebDriver, site: str):
|
|
|
190
193
|
break
|
|
191
194
|
|
|
192
195
|
|
|
193
|
-
def download_data_dump(browser: WebDriver, site: str, meta_url: str, etags: Dict[str, str]):
|
|
196
|
+
def download_data_dump(browser: WebDriver, site: str, meta_url: str | None, etags: Dict[str, str]):
|
|
194
197
|
logger.info(f"Downloading data dump from {site}")
|
|
195
198
|
|
|
196
199
|
def _exec_download(browser: WebDriver):
|
|
@@ -257,7 +260,7 @@ def download_data_dump(browser: WebDriver, site: str, meta_url: str, etags: Dict
|
|
|
257
260
|
|
|
258
261
|
if args.skip_loaded and meta_loaded:
|
|
259
262
|
pass
|
|
260
|
-
|
|
263
|
+
elif meta_url is not None:
|
|
261
264
|
browser.get(f"{meta_url}/users/data-dump-access/current")
|
|
262
265
|
check_cloudflare_intercept(browser)
|
|
263
266
|
|
|
@@ -345,7 +348,134 @@ def await_date_change(args: SEDDCLIArgs, driver):
|
|
|
345
348
|
"Please open an issue: https://github.com/LunarWatcher/se-data-dump-transformer/"
|
|
346
349
|
)
|
|
347
350
|
raise
|
|
348
|
-
|
|
351
|
+
|
|
352
|
+
def do_download(site: str):
|
|
353
|
+
if site not in ["https://meta.stackexchange.com", "https://stackapps.com"]:
|
|
354
|
+
# https://regex101.com/r/kG6nTN/1
|
|
355
|
+
meta_url = re.sub(
|
|
356
|
+
r"(https://(?:[^.]+\.(?=stackexchange))?)", r"\1meta.", site)
|
|
357
|
+
else:
|
|
358
|
+
meta_url = None
|
|
359
|
+
|
|
360
|
+
main_loaded = utils.is_file_downloaded(args.output_dir, site)
|
|
361
|
+
meta_loaded = utils.is_file_downloaded(args.output_dir, meta_url) or \
|
|
362
|
+
meta_url is None
|
|
363
|
+
|
|
364
|
+
if args.skip_loaded and main_loaded and meta_loaded:
|
|
365
|
+
pass
|
|
366
|
+
else:
|
|
367
|
+
logger.info(f"Extracting from {site}...")
|
|
368
|
+
|
|
369
|
+
login_or_create(browser, site)
|
|
370
|
+
download_data_dump(
|
|
371
|
+
browser,
|
|
372
|
+
site,
|
|
373
|
+
meta_url,
|
|
374
|
+
etags
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
def clear_part_files():
|
|
378
|
+
files = glob.glob(
|
|
379
|
+
os.path.join(
|
|
380
|
+
args.output_dir,
|
|
381
|
+
"*.part"
|
|
382
|
+
)
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
for file in files:
|
|
386
|
+
logger.debug(
|
|
387
|
+
"Found existing .part file ({}). Removing...",
|
|
388
|
+
file
|
|
389
|
+
)
|
|
390
|
+
os.remove(file)
|
|
391
|
+
|
|
392
|
+
def normalize_meta(url: str):
|
|
393
|
+
if (url.startswith("meta.") and url != "meta.stackexchange.com"):
|
|
394
|
+
return url.replace("meta.", "")
|
|
395
|
+
return url
|
|
396
|
+
|
|
397
|
+
def try_recover_fucked_download(
|
|
398
|
+
last_sizes: dict[str, LastState],
|
|
399
|
+
state: DownloadState
|
|
400
|
+
):
|
|
401
|
+
# TODO: This is a very naive approach. Can we get the information directly
|
|
402
|
+
# form the webdriver?
|
|
403
|
+
for path in state.pending:
|
|
404
|
+
# The path is in the form of the full site URL with the .zip
|
|
405
|
+
# The .part file is in the form `[first path component].[garbage].[rest]`
|
|
406
|
+
# meta.stackexchange.com.7z would be
|
|
407
|
+
# meta.[garbage.]stackexchange.7z.part
|
|
408
|
+
spl = path.split(".", 1)
|
|
409
|
+
matches = glob.glob(
|
|
410
|
+
os.path.join(
|
|
411
|
+
args.output_dir,
|
|
412
|
+
'.*.'.join(spl) + ".part"
|
|
413
|
+
)
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
if len(matches) == 0:
|
|
417
|
+
logger.warning("{} has 0 part files. Race condition?", path)
|
|
418
|
+
elif len(matches) > 1:
|
|
419
|
+
logger.error(
|
|
420
|
+
"{} has more than 1 part file! Error recovery impossible",
|
|
421
|
+
path
|
|
422
|
+
)
|
|
423
|
+
else:
|
|
424
|
+
part_path = matches[0]
|
|
425
|
+
size = os.path.getsize(part_path)
|
|
426
|
+
|
|
427
|
+
last_size = last_sizes.get(
|
|
428
|
+
path,
|
|
429
|
+
LastState(monotonic(), 0)
|
|
430
|
+
)
|
|
431
|
+
|
|
432
|
+
# We only commit the size if it's changing
|
|
433
|
+
if (last_size.last_observed_size < size):
|
|
434
|
+
last_size.last_observed_size = size
|
|
435
|
+
last_size.last_observed_change = monotonic()
|
|
436
|
+
|
|
437
|
+
last_sizes[path] = last_size
|
|
438
|
+
else:
|
|
439
|
+
# If it isn't changing, we allow for a 60 second window
|
|
440
|
+
# After 30 seconds, a warning is issued. This is likely useless
|
|
441
|
+
# in practice, because a failure this long almost certainly
|
|
442
|
+
# means the download has died
|
|
443
|
+
# After 60 seconds, we treat the download as failed and move on
|
|
444
|
+
# with our lives. Nuke the part file, redo the download
|
|
445
|
+
delta = monotonic() - last_size.last_observed_change
|
|
446
|
+
|
|
447
|
+
if (delta > 60):
|
|
448
|
+
logger.error(
|
|
449
|
+
"{} has failed. {} has been pinned for 60 seconds. "
|
|
450
|
+
"Retrying",
|
|
451
|
+
path,
|
|
452
|
+
part_path
|
|
453
|
+
)
|
|
454
|
+
del last_sizes[path]
|
|
455
|
+
|
|
456
|
+
os.remove(part_path)
|
|
457
|
+
# TODO: optimally, we'd just call a single function that
|
|
458
|
+
# directly downloads the specific site. Unfortunately, the
|
|
459
|
+
# system wasn't set up to deal with this, and I don't feel
|
|
460
|
+
# like rewriting it when all I want is to download the god
|
|
461
|
+
# damn file
|
|
462
|
+
# The download system should be split up to allow for this,
|
|
463
|
+
# but it'll be a bigger refactor to do it. The current URL
|
|
464
|
+
# system is fairly fragile, really
|
|
465
|
+
do_download(
|
|
466
|
+
"https://" + normalize_meta(
|
|
467
|
+
path.replace(".7z", "")
|
|
468
|
+
)
|
|
469
|
+
)
|
|
470
|
+
elif (
|
|
471
|
+
# the "and" is to avoid excessive spam. This allows a 2
|
|
472
|
+
# second window
|
|
473
|
+
delta > 30 and delta < 32
|
|
474
|
+
):
|
|
475
|
+
logger.warning(
|
|
476
|
+
"No observed change in {} for 30 seconds",
|
|
477
|
+
path
|
|
478
|
+
)
|
|
349
479
|
|
|
350
480
|
etags: Dict[str, str] = {}
|
|
351
481
|
|
|
@@ -353,34 +483,19 @@ try:
|
|
|
353
483
|
state, observer = register_pending_downloads_observer(args.output_dir)
|
|
354
484
|
if args.detect is not None:
|
|
355
485
|
await_date_change(args, browser)
|
|
356
|
-
|
|
486
|
+
if args.wipe_part_files:
|
|
487
|
+
clear_part_files()
|
|
357
488
|
for site in sites.sites:
|
|
358
|
-
|
|
359
|
-
# https://regex101.com/r/kG6nTN/1
|
|
360
|
-
meta_url = re.sub(
|
|
361
|
-
r"(https://(?:[^.]+\.(?=stackexchange))?)", r"\1meta.", site)
|
|
362
|
-
|
|
363
|
-
main_loaded = utils.is_file_downloaded(args.output_dir, site)
|
|
364
|
-
meta_loaded = utils.is_file_downloaded(args.output_dir, meta_url)
|
|
365
|
-
|
|
366
|
-
if args.skip_loaded and main_loaded and meta_loaded:
|
|
367
|
-
pass
|
|
368
|
-
else:
|
|
369
|
-
logger.info(f"Extracting from {site}...")
|
|
370
|
-
|
|
371
|
-
login_or_create(browser, site)
|
|
372
|
-
download_data_dump(
|
|
373
|
-
browser,
|
|
374
|
-
site,
|
|
375
|
-
meta_url,
|
|
376
|
-
etags
|
|
377
|
-
)
|
|
489
|
+
do_download(site)
|
|
378
490
|
|
|
379
491
|
if observer:
|
|
380
492
|
pending = state.size()
|
|
381
493
|
|
|
382
494
|
logger.info(f"Waiting for {pending} download{'s'[:pending^1]} to complete")
|
|
495
|
+
if args.wipe_part_files:
|
|
496
|
+
logger.info("Download error recovery is enabled.")
|
|
383
497
|
|
|
498
|
+
last_sizes = {}
|
|
384
499
|
while True:
|
|
385
500
|
if state.empty():
|
|
386
501
|
observer.stop()
|
|
@@ -389,6 +504,11 @@ try:
|
|
|
389
504
|
utils.cleanup_archive(args.output_dir)
|
|
390
505
|
break
|
|
391
506
|
else:
|
|
507
|
+
if args.wipe_part_files:
|
|
508
|
+
try_recover_fucked_download(
|
|
509
|
+
last_sizes,
|
|
510
|
+
state
|
|
511
|
+
)
|
|
392
512
|
sleep(1)
|
|
393
513
|
|
|
394
514
|
notifications.notify(
|
|
@@ -23,7 +23,7 @@ def init_ubo_settings(browser: Firefox, config: SEDDConfig, ubo_id: str) -> bool
|
|
|
23
23
|
ubo_set_advanced_settings(browser, settings)
|
|
24
24
|
|
|
25
25
|
# idk why, but applyFilterListSelection only works after a delay
|
|
26
|
-
sleep(
|
|
26
|
+
sleep(3)
|
|
27
27
|
|
|
28
28
|
ubo_set_selected_filters(browser, settings)
|
|
29
29
|
ubo_set_whitelist(browser, settings)
|
|
@@ -82,6 +82,8 @@ def cleanup_archive(base_path: str) -> None:
|
|
|
82
82
|
traceback.print_exception(sys.exception())
|
|
83
83
|
|
|
84
84
|
|
|
85
|
-
def is_file_downloaded(base_path: str, site_or_url: str) -> bool:
|
|
85
|
+
def is_file_downloaded(base_path: str, site_or_url: str | None) -> bool:
|
|
86
|
+
if (site_or_url is None):
|
|
87
|
+
return False
|
|
86
88
|
file_name = get_file_name(site_or_url)
|
|
87
89
|
return check_file(base_path, file_name)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|