sedd 2.2.2__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. {sedd-2.2.2/sedd.egg-info → sedd-2.4.0}/PKG-INFO +1 -1
  2. {sedd-2.2.2 → sedd-2.4.0}/README.md +4 -1
  3. {sedd-2.2.2 → sedd-2.4.0}/sedd/cli.py +18 -0
  4. {sedd-2.2.2 → sedd-2.4.0}/sedd/main.py +146 -26
  5. {sedd-2.2.2 → sedd-2.4.0}/sedd/ubo/ubo.py +1 -1
  6. {sedd-2.2.2 → sedd-2.4.0}/sedd/utils.py +3 -1
  7. sedd-2.4.0/sedd/watcher/recovery.py +6 -0
  8. {sedd-2.2.2 → sedd-2.4.0/sedd.egg-info}/PKG-INFO +1 -1
  9. {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/SOURCES.txt +1 -0
  10. {sedd-2.2.2 → sedd-2.4.0}/LICENSE +0 -0
  11. {sedd-2.2.2 → sedd-2.4.0}/README-python.md +0 -0
  12. {sedd-2.2.2 → sedd-2.4.0}/pyproject.toml +0 -0
  13. {sedd-2.2.2 → sedd-2.4.0}/sedd/__init__.py +0 -0
  14. {sedd-2.2.2 → sedd-2.4.0}/sedd/__main__.py +0 -0
  15. {sedd-2.2.2 → sedd-2.4.0}/sedd/config/__init__.py +0 -0
  16. {sedd-2.2.2 → sedd-2.4.0}/sedd/config/config.py +0 -0
  17. {sedd-2.2.2 → sedd-2.4.0}/sedd/config/defaults.py +0 -0
  18. {sedd-2.2.2 → sedd-2.4.0}/sedd/config/typings.py +0 -0
  19. {sedd-2.2.2 → sedd-2.4.0}/sedd/data/__init__.py +0 -0
  20. {sedd-2.2.2 → sedd-2.4.0}/sedd/data/files_map.py +0 -0
  21. {sedd-2.2.2 → sedd-2.4.0}/sedd/data/sites.py +0 -0
  22. {sedd-2.2.2 → sedd-2.4.0}/sedd/driver.py +0 -0
  23. {sedd-2.2.2 → sedd-2.4.0}/sedd/meta/__init__.py +0 -0
  24. {sedd-2.2.2 → sedd-2.4.0}/sedd/meta/notifications.py +0 -0
  25. {sedd-2.2.2 → sedd-2.4.0}/sedd/pipscript.py +0 -0
  26. {sedd-2.2.2 → sedd-2.4.0}/sedd/ubo/__init__.py +0 -0
  27. {sedd-2.2.2 → sedd-2.4.0}/sedd/ubo/utils.py +0 -0
  28. {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/__init__.py +0 -0
  29. {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/handler.py +0 -0
  30. {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/observer.py +0 -0
  31. {sedd-2.2.2 → sedd-2.4.0}/sedd/watcher/state.py +0 -0
  32. {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/dependency_links.txt +0 -0
  33. {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/entry_points.txt +0 -0
  34. {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/requires.txt +0 -0
  35. {sedd-2.2.2 → sedd-2.4.0}/sedd.egg-info/top_level.txt +0 -0
  36. {sedd-2.2.2 → sedd-2.4.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sedd
3
- Version: 2.2.2
3
+ Version: 2.4.0
4
4
  Summary: Unofficial, community-made tool for downloading the Stack Exchange data dumps
5
5
  Author-email: LunarWatcher <oliviawolfie@pm.me>
6
6
  License-Expression: MIT
@@ -169,6 +169,7 @@ Exractor CLI supports the following configuration options:
169
169
  | `-v` | - | Optional | `false` | Whether or not to enable verbose logging. You do not want to enable this unless you're diagnosing a problem with sedd - the verbose logging includes low-level Selenium output. |
170
170
  | - | `--detect <last upload date>` | Optional | - | Enables a mode where sedd checks for the upload timestamp to change, and only then begins the download. This has several caveats. Please read [docs/Automated downloads](docs/Automated downloads.md) for more information before using this feature. |
171
171
  | `-u` | `--unsupervised` | Optional | false | Rather than notify you about failures and asking you to resolve them, hard failures that would require manual supervision throw an exception instead. |
172
+ | `-N` | `--no-wipe-part-files` | Optional | false | Don't wipe `.part` files in the download directory on boot. Only ever useful if you're doing cursed shit. Disables automatic download error recovery, and may result in the download completion watcher getting stuck |
172
173
 
173
174
  #### Captchas and other misc. barriers
174
175
 
@@ -299,7 +300,9 @@ cmake --build . -j 8 --config Release
299
300
  .\sedd-transformer.exe -i ..\..\downloads -t [formatter type]
300
301
  ```
301
302
 
302
- Pass `--help` to see the available formatters for your current version of the data dump transformer.
303
+ Pass `--help` to see the available formatters for your current version of the data dump transformer. There's also some filters available for use, that change the data.
304
+
305
+ If you only want to filter the data, but keep the XML format, you can use `-t xml` to convert back to XML.
303
306
 
304
307
  ### Supported transformers
305
308
 
@@ -14,6 +14,8 @@ class SEDDCLIArgs(argparse.Namespace):
14
14
  detect: str | None
15
15
  unsupervised: bool
16
16
 
17
+ wipe_part_files: bool
18
+
17
19
 
18
20
  parser = argparse.ArgumentParser(
19
21
  prog="sedd",
@@ -78,6 +80,22 @@ parser.add_argument(
78
80
  )
79
81
  )
80
82
 
83
+ parser.add_argument(
84
+ "-N,--no-wipe-part-files",
85
+ required=False,
86
+ default=True,
87
+ action="store_false",
88
+ dest="wipe_part_files",
89
+ help=(
90
+ "When supplied, .part files in the folder won't be wiped. "
91
+ "By default, wiping part files is enabled, as leaving it interferes "
92
+ "with a download error recovery mechanism. Supplying this flag "
93
+ "disables download error recovery, as the system has no way to "
94
+ "differentiate between a .part file that was there before the "
95
+ "download started"
96
+ )
97
+ )
98
+
81
99
  parser.add_argument(
82
100
  "--detect",
83
101
  required=False,
@@ -2,11 +2,12 @@ from selenium.webdriver.common.by import By
2
2
  from selenium.webdriver.firefox.webdriver import WebDriver
3
3
  from selenium.common.exceptions import NoSuchElementException
4
4
  from typing import Dict
5
+ import glob
5
6
 
6
-
7
- from time import sleep
7
+ from time import sleep, monotonic
8
8
 
9
9
  import re
10
+ import os
10
11
  import sys
11
12
  import random
12
13
  from traceback import print_exception
@@ -17,6 +18,8 @@ from .config import load_sedd_config
17
18
  from .data import sites
18
19
  from .meta import notifications
19
20
  from .watcher.observer import register_pending_downloads_observer
21
+ from .watcher.state import DownloadState
22
+ from .watcher.recovery import LastState
20
23
  from . import utils
21
24
 
22
25
  from .driver import init_output_dir, init_firefox_driver
@@ -190,7 +193,7 @@ def login_or_create(browser: WebDriver, site: str):
190
193
  break
191
194
 
192
195
 
193
- def download_data_dump(browser: WebDriver, site: str, meta_url: str, etags: Dict[str, str]):
196
+ def download_data_dump(browser: WebDriver, site: str, meta_url: str | None, etags: Dict[str, str]):
194
197
  logger.info(f"Downloading data dump from {site}")
195
198
 
196
199
  def _exec_download(browser: WebDriver):
@@ -257,7 +260,7 @@ def download_data_dump(browser: WebDriver, site: str, meta_url: str, etags: Dict
257
260
 
258
261
  if args.skip_loaded and meta_loaded:
259
262
  pass
260
- else:
263
+ elif meta_url is not None:
261
264
  browser.get(f"{meta_url}/users/data-dump-access/current")
262
265
  check_cloudflare_intercept(browser)
263
266
 
@@ -345,7 +348,134 @@ def await_date_change(args: SEDDCLIArgs, driver):
345
348
  "Please open an issue: https://github.com/LunarWatcher/se-data-dump-transformer/"
346
349
  )
347
350
  raise
348
-
351
+
352
+ def do_download(site: str):
353
+ if site not in ["https://meta.stackexchange.com", "https://stackapps.com"]:
354
+ # https://regex101.com/r/kG6nTN/1
355
+ meta_url = re.sub(
356
+ r"(https://(?:[^.]+\.(?=stackexchange))?)", r"\1meta.", site)
357
+ else:
358
+ meta_url = None
359
+
360
+ main_loaded = utils.is_file_downloaded(args.output_dir, site)
361
+ meta_loaded = utils.is_file_downloaded(args.output_dir, meta_url) or \
362
+ meta_url is None
363
+
364
+ if args.skip_loaded and main_loaded and meta_loaded:
365
+ pass
366
+ else:
367
+ logger.info(f"Extracting from {site}...")
368
+
369
+ login_or_create(browser, site)
370
+ download_data_dump(
371
+ browser,
372
+ site,
373
+ meta_url,
374
+ etags
375
+ )
376
+
377
+ def clear_part_files():
378
+ files = glob.glob(
379
+ os.path.join(
380
+ args.output_dir,
381
+ "*.part"
382
+ )
383
+ )
384
+
385
+ for file in files:
386
+ logger.debug(
387
+ "Found existing .part file ({}). Removing...",
388
+ file
389
+ )
390
+ os.remove(file)
391
+
392
+ def normalize_meta(url: str):
393
+ if (url.startswith("meta.") and url != "meta.stackexchange.com"):
394
+ return url.replace("meta.", "")
395
+ return url
396
+
397
+ def try_recover_fucked_download(
398
+ last_sizes: dict[str, LastState],
399
+ state: DownloadState
400
+ ):
401
+ # TODO: This is a very naive approach. Can we get the information directly
402
+ # form the webdriver?
403
+ for path in state.pending:
404
+ # The path is in the form of the full site URL with the .zip
405
+ # The .part file is in the form `[first path component].[garbage].[rest]`
406
+ # meta.stackexchange.com.7z would be
407
+ # meta.[garbage.]stackexchange.7z.part
408
+ spl = path.split(".", 1)
409
+ matches = glob.glob(
410
+ os.path.join(
411
+ args.output_dir,
412
+ '.*.'.join(spl) + ".part"
413
+ )
414
+ )
415
+
416
+ if len(matches) == 0:
417
+ logger.warning("{} has 0 part files. Race condition?", path)
418
+ elif len(matches) > 1:
419
+ logger.error(
420
+ "{} has more than 1 part file! Error recovery impossible",
421
+ path
422
+ )
423
+ else:
424
+ part_path = matches[0]
425
+ size = os.path.getsize(part_path)
426
+
427
+ last_size = last_sizes.get(
428
+ path,
429
+ LastState(monotonic(), 0)
430
+ )
431
+
432
+ # We only commit the size if it's changing
433
+ if (last_size.last_observed_size < size):
434
+ last_size.last_observed_size = size
435
+ last_size.last_observed_change = monotonic()
436
+
437
+ last_sizes[path] = last_size
438
+ else:
439
+ # If it isn't changing, we allow for a 60 second window
440
+ # After 30 seconds, a warning is issued. This is likely useless
441
+ # in practice, because a failure this long almost certainly
442
+ # means the download has died
443
+ # After 60 seconds, we treat the download as failed and move on
444
+ # with our lives. Nuke the part file, redo the download
445
+ delta = monotonic() - last_size.last_observed_change
446
+
447
+ if (delta > 60):
448
+ logger.error(
449
+ "{} has failed. {} has been pinned for 60 seconds. "
450
+ "Retrying",
451
+ path,
452
+ part_path
453
+ )
454
+ del last_sizes[path]
455
+
456
+ os.remove(part_path)
457
+ # TODO: optimally, we'd just call a single function that
458
+ # directly downloads the specific site. Unfortunately, the
459
+ # system wasn't set up to deal with this, and I don't feel
460
+ # like rewriting it when all I want is to download the god
461
+ # damn file
462
+ # The download system should be split up to allow for this,
463
+ # but it'll be a bigger refactor to do it. The current URL
464
+ # system is fairly fragile, really
465
+ do_download(
466
+ "https://" + normalize_meta(
467
+ path.replace(".7z", "")
468
+ )
469
+ )
470
+ elif (
471
+ # the "and" is to avoid excessive spam. This allows a 2
472
+ # second window
473
+ delta > 30 and delta < 32
474
+ ):
475
+ logger.warning(
476
+ "No observed change in {} for 30 seconds",
477
+ path
478
+ )
349
479
 
350
480
  etags: Dict[str, str] = {}
351
481
 
@@ -353,34 +483,19 @@ try:
353
483
  state, observer = register_pending_downloads_observer(args.output_dir)
354
484
  if args.detect is not None:
355
485
  await_date_change(args, browser)
356
-
486
+ if args.wipe_part_files:
487
+ clear_part_files()
357
488
  for site in sites.sites:
358
- if site not in ["https://meta.stackexchange.com", "https://stackapps.com"]:
359
- # https://regex101.com/r/kG6nTN/1
360
- meta_url = re.sub(
361
- r"(https://(?:[^.]+\.(?=stackexchange))?)", r"\1meta.", site)
362
-
363
- main_loaded = utils.is_file_downloaded(args.output_dir, site)
364
- meta_loaded = utils.is_file_downloaded(args.output_dir, meta_url)
365
-
366
- if args.skip_loaded and main_loaded and meta_loaded:
367
- pass
368
- else:
369
- logger.info(f"Extracting from {site}...")
370
-
371
- login_or_create(browser, site)
372
- download_data_dump(
373
- browser,
374
- site,
375
- meta_url,
376
- etags
377
- )
489
+ do_download(site)
378
490
 
379
491
  if observer:
380
492
  pending = state.size()
381
493
 
382
494
  logger.info(f"Waiting for {pending} download{'s'[:pending^1]} to complete")
495
+ if args.wipe_part_files:
496
+ logger.info("Download error recovery is enabled.")
383
497
 
498
+ last_sizes = {}
384
499
  while True:
385
500
  if state.empty():
386
501
  observer.stop()
@@ -389,6 +504,11 @@ try:
389
504
  utils.cleanup_archive(args.output_dir)
390
505
  break
391
506
  else:
507
+ if args.wipe_part_files:
508
+ try_recover_fucked_download(
509
+ last_sizes,
510
+ state
511
+ )
392
512
  sleep(1)
393
513
 
394
514
  notifications.notify(
@@ -23,7 +23,7 @@ def init_ubo_settings(browser: Firefox, config: SEDDConfig, ubo_id: str) -> bool
23
23
  ubo_set_advanced_settings(browser, settings)
24
24
 
25
25
  # idk why, but applyFilterListSelection only works after a delay
26
- sleep(1)
26
+ sleep(3)
27
27
 
28
28
  ubo_set_selected_filters(browser, settings)
29
29
  ubo_set_whitelist(browser, settings)
@@ -82,6 +82,8 @@ def cleanup_archive(base_path: str) -> None:
82
82
  traceback.print_exception(sys.exception())
83
83
 
84
84
 
85
- def is_file_downloaded(base_path: str, site_or_url: str) -> bool:
85
+ def is_file_downloaded(base_path: str, site_or_url: str | None) -> bool:
86
+ if (site_or_url is None):
87
+ return False
86
88
  file_name = get_file_name(site_or_url)
87
89
  return check_file(base_path, file_name)
@@ -0,0 +1,6 @@
1
+ from dataclasses import dataclass
2
+
3
+ @dataclass
4
+ class LastState:
5
+ last_observed_change: float
6
+ last_observed_size: int
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sedd
3
- Version: 2.2.2
3
+ Version: 2.4.0
4
4
  Summary: Unofficial, community-made tool for downloading the Stack Exchange data dumps
5
5
  Author-email: LunarWatcher <oliviawolfie@pm.me>
6
6
  License-Expression: MIT
@@ -30,4 +30,5 @@ sedd/ubo/utils.py
30
30
  sedd/watcher/__init__.py
31
31
  sedd/watcher/handler.py
32
32
  sedd/watcher/observer.py
33
+ sedd/watcher/recovery.py
33
34
  sedd/watcher/state.py
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes