anki-addons-dataset 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- anki_addons_dataset-1.3.0/LICENSE +21 -0
- anki_addons_dataset-1.3.0/PKG-INFO +76 -0
- anki_addons_dataset-1.3.0/README.md +33 -0
- anki_addons_dataset-1.3.0/pyproject.toml +70 -0
- anki_addons_dataset-1.3.0/setup.cfg +4 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/__init__.py +10 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/addon_catalog.py +38 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/argument/script_arguments.py +52 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/bundle/dataset_bundle.py +60 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/addon_infos_collector.py +46 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/aggregator.py +17 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiforum/ankiforum_enricher.py +52 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiforum/ankiforum_service.py +60 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiforum/ankiforum_topic.py +31 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/addon_branch_parser.py +27 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/addon_page_downloader.py +45 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/addon_page_parser.py +108 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/addons_page_downloader.py +34 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/addons_page_parser.py +32 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/ankiweb_service.py +16 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/ankiweb/page_downloader.py +24 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/collector_facade.py +115 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/dataset_metadata.py +33 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/enricher.py +61 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/github_enricher.py +64 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/github_rate_limit.py +48 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/github_rest_client.py +31 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/github_service.py +71 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/actions_repo_handler.py +21 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/languages_repo_handler.py +22 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/last_commit_repo_handler.py +40 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/repo_handler.py +88 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/stars_repo_handler.py +24 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/tests_counter.py +38 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/github/handler/tests_repo_handler.py +41 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/overrider/overrider.py +49 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/overrider/overrides.yaml +5 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/raw_metadata_collector.py +50 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/collector/url_parser.py +47 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/common/data_types.py +132 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/common/json_helper.py +138 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/common/log.py +21 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/common/working_dir.py +110 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/exporter.py +20 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/exporter_facade.py +24 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/json/json_exporter.py +51 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/json/schema.json +204 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/json_addon_info.py +107 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/parquet/parquet_exporter.py +31 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/xlsx/addon_info_sheet.py +173 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/xlsx/aggregation_sheet.py +76 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/exporter/xlsx/xlsx_exporter.py +37 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/facade/facade.py +45 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/huggingface/README.md +118 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/huggingface/hugging_face.py +16 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/huggingface/hugging_face_client.py +88 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/initializer/working_dir_backup.py +23 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/initializer/working_dir_initializer.py +56 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset/version.txt +1 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset.egg-info/PKG-INFO +76 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset.egg-info/SOURCES.txt +63 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset.egg-info/dependency_links.txt +1 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset.egg-info/entry_points.txt +2 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset.egg-info/requires.txt +22 -0
- anki_addons_dataset-1.3.0/src/anki_addons_dataset.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Aleksey Yablokov
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: anki-addons-dataset
|
|
3
|
+
Version: 1.3.0
|
|
4
|
+
Summary: Collects, processes, and publishes structured data about Anki flashcard addons to HuggingFace.
|
|
5
|
+
Author-email: Aleksey Yablokov <alex_ya@mailbox.org>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Aleks-Ya/anki-addons-dataset
|
|
8
|
+
Project-URL: Repository, https://github.com/Aleks-Ya/anki-addons-dataset
|
|
9
|
+
Project-URL: Dataset, https://huggingface.co/datasets/Ya-Alex/anki-addons
|
|
10
|
+
Project-URL: Visualizations, https://huggingface.co/spaces/Ya-Alex/anki-addons
|
|
11
|
+
Keywords: anki,addons,dataset,ankiweb,huggingface
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Education
|
|
17
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: requests
|
|
22
|
+
Requires-Dist: bs4
|
|
23
|
+
Requires-Dist: mdutils
|
|
24
|
+
Requires-Dist: pyyaml
|
|
25
|
+
Requires-Dist: XlsxWriter
|
|
26
|
+
Requires-Dist: jsonschema
|
|
27
|
+
Requires-Dist: selenium
|
|
28
|
+
Requires-Dist: huggingface-hub
|
|
29
|
+
Requires-Dist: pandas
|
|
30
|
+
Requires-Dist: pyarrow
|
|
31
|
+
Requires-Dist: seedir
|
|
32
|
+
Requires-Dist: openpyxl
|
|
33
|
+
Requires-Dist: pydiscourse
|
|
34
|
+
Provides-Extra: dev
|
|
35
|
+
Requires-Dist: pytest; extra == "dev"
|
|
36
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
37
|
+
Requires-Dist: pytest-mock; extra == "dev"
|
|
38
|
+
Requires-Dist: freezegun; extra == "dev"
|
|
39
|
+
Requires-Dist: bump-my-version; extra == "dev"
|
|
40
|
+
Requires-Dist: build; extra == "dev"
|
|
41
|
+
Requires-Dist: twine; extra == "dev"
|
|
42
|
+
Dynamic: license-file
|
|
43
|
+
|
|
44
|
+
# Anki Addons Dataset
|
|
45
|
+
|
|
46
|
+
A HuggingFace dataset of addons for the [Anki](https://apps.ankiweb.net) flashcard program.
|
|
47
|
+
|
|
48
|
+
## Install
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install anki-addons-dataset
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
This installs the `anki-addons-dataset` command. The pipeline runs as a sequence of operations:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
anki-addons-dataset init
|
|
58
|
+
anki-addons-dataset download -d 2026-01-01
|
|
59
|
+
anki-addons-dataset parse
|
|
60
|
+
anki-addons-dataset report
|
|
61
|
+
anki-addons-dataset bundle
|
|
62
|
+
anki-addons-dataset upload
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## Links
|
|
66
|
+
- [Visualizations](https://huggingface.co/spaces/Ya-Alex/anki-addons) in HuggingFace Spaces
|
|
67
|
+
- [HuggingFace Dataset](https://huggingface.co/datasets/Ya-Alex/anki-addons)
|
|
68
|
+
- [Developer Guide](README-DEV.md)
|
|
69
|
+
- [Sonar Qube](https://sonarcloud.io/project/overview?id=Aleks-Ya_anki-addons-dataset)
|
|
70
|
+
- Anki
|
|
71
|
+
- [Anki home page](https://apps.ankiweb.net)
|
|
72
|
+
- [Anki Addons catalog](https://ankiweb.net/shared/addons)
|
|
73
|
+
|
|
74
|
+
[](https://github.com/Aleks-Ya/anki-addons-dataset/actions/workflows/unit-tests.yml)
|
|
75
|
+
[](https://sonarcloud.io/summary/new_code?id=Aleks-Ya_anki-addons-dataset)
|
|
76
|
+
[](https://sonarcloud.io/summary/new_code?id=Aleks-Ya_anki-addons-dataset)
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Anki Addons Dataset
|
|
2
|
+
|
|
3
|
+
A HuggingFace dataset of addons for the [Anki](https://apps.ankiweb.net) flashcard program.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install anki-addons-dataset
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
This installs the `anki-addons-dataset` command. The pipeline runs as a sequence of operations:
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
anki-addons-dataset init
|
|
15
|
+
anki-addons-dataset download -d 2026-01-01
|
|
16
|
+
anki-addons-dataset parse
|
|
17
|
+
anki-addons-dataset report
|
|
18
|
+
anki-addons-dataset bundle
|
|
19
|
+
anki-addons-dataset upload
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Links
|
|
23
|
+
- [Visualizations](https://huggingface.co/spaces/Ya-Alex/anki-addons) in HuggingFace Spaces
|
|
24
|
+
- [HuggingFace Dataset](https://huggingface.co/datasets/Ya-Alex/anki-addons)
|
|
25
|
+
- [Developer Guide](README-DEV.md)
|
|
26
|
+
- [Sonar Qube](https://sonarcloud.io/project/overview?id=Aleks-Ya_anki-addons-dataset)
|
|
27
|
+
- Anki
|
|
28
|
+
- [Anki home page](https://apps.ankiweb.net)
|
|
29
|
+
- [Anki Addons catalog](https://ankiweb.net/shared/addons)
|
|
30
|
+
|
|
31
|
+
[](https://github.com/Aleks-Ya/anki-addons-dataset/actions/workflows/unit-tests.yml)
|
|
32
|
+
[](https://sonarcloud.io/summary/new_code?id=Aleks-Ya_anki-addons-dataset)
|
|
33
|
+
[](https://sonarcloud.io/summary/new_code?id=Aleks-Ya_anki-addons-dataset)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "anki-addons-dataset"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Collects, processes, and publishes structured data about Anki flashcard addons to HuggingFace."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Aleksey Yablokov", email = "alex_ya@mailbox.org" }]
|
|
14
|
+
keywords = ["anki", "addons", "dataset", "ankiweb", "huggingface"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Topic :: Education",
|
|
21
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"requests",
|
|
25
|
+
"bs4",
|
|
26
|
+
"mdutils",
|
|
27
|
+
"pyyaml",
|
|
28
|
+
"XlsxWriter",
|
|
29
|
+
"jsonschema",
|
|
30
|
+
"selenium",
|
|
31
|
+
"huggingface-hub",
|
|
32
|
+
"pandas",
|
|
33
|
+
"pyarrow",
|
|
34
|
+
"seedir",
|
|
35
|
+
"openpyxl",
|
|
36
|
+
"pydiscourse",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
[project.optional-dependencies]
|
|
40
|
+
dev = [
|
|
41
|
+
"pytest",
|
|
42
|
+
"pytest-cov",
|
|
43
|
+
"pytest-mock",
|
|
44
|
+
"freezegun",
|
|
45
|
+
"bump-my-version",
|
|
46
|
+
"build",
|
|
47
|
+
"twine",
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
[project.urls]
|
|
51
|
+
Homepage = "https://github.com/Aleks-Ya/anki-addons-dataset"
|
|
52
|
+
Repository = "https://github.com/Aleks-Ya/anki-addons-dataset"
|
|
53
|
+
Dataset = "https://huggingface.co/datasets/Ya-Alex/anki-addons"
|
|
54
|
+
Visualizations = "https://huggingface.co/spaces/Ya-Alex/anki-addons"
|
|
55
|
+
|
|
56
|
+
[project.scripts]
|
|
57
|
+
anki-addons-dataset = "anki_addons_dataset.addon_catalog:main"
|
|
58
|
+
|
|
59
|
+
[tool.setuptools]
|
|
60
|
+
package-dir = { "" = "src" }
|
|
61
|
+
|
|
62
|
+
[tool.setuptools.packages.find]
|
|
63
|
+
where = ["src"]
|
|
64
|
+
namespaces = true
|
|
65
|
+
|
|
66
|
+
[tool.setuptools.package-data]
|
|
67
|
+
"*" = ["*.txt", "*.yaml", "*.json", "*.md"]
|
|
68
|
+
|
|
69
|
+
[tool.setuptools.dynamic]
|
|
70
|
+
version = { attr = "anki_addons_dataset.__version__" }
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Anki Addons Dataset package.
|
|
2
|
+
|
|
3
|
+
Exposes ``__version__`` read verbatim from ``version.txt`` (the single source of truth,
|
|
4
|
+
kept in sync by ``bump-my-version``). The value is already PEP 440-compliant: development
|
|
5
|
+
builds use the ``.dev0`` suffix (e.g. ``1.3.0.dev0``) and releases drop it (``1.3.0``).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
__version__: str = (Path(__file__).parent / "version.txt").read_text().strip()
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
from logging import Logger
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from huggingface_hub import HfApi
|
|
8
|
+
|
|
9
|
+
from anki_addons_dataset.argument.script_arguments import ScriptArguments, Operation
|
|
10
|
+
from anki_addons_dataset.common.data_types import SnapshotDate, ReportDate
|
|
11
|
+
from anki_addons_dataset.common.working_dir import WorkingDir
|
|
12
|
+
from anki_addons_dataset.facade.facade import Facade
|
|
13
|
+
from anki_addons_dataset.huggingface.hugging_face_client import HuggingFaceClient
|
|
14
|
+
from anki_addons_dataset.common.log import Log
|
|
15
|
+
|
|
16
|
+
log: Logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
def main() -> None:
|
|
19
|
+
Log.configure_logging()
|
|
20
|
+
|
|
21
|
+
arguments: ScriptArguments = ScriptArguments()
|
|
22
|
+
|
|
23
|
+
Log.set_log_level(arguments.get_log_level())
|
|
24
|
+
operation: Operation = arguments.get_operation()
|
|
25
|
+
snapshot_date: Optional[SnapshotDate] = arguments.get_snapshot_date()
|
|
26
|
+
log.info(f"Snapshot date: {snapshot_date}")
|
|
27
|
+
report_date: ReportDate = ReportDate(datetime.now().replace(microsecond=0))
|
|
28
|
+
log.info(f"Report date: {report_date}")
|
|
29
|
+
|
|
30
|
+
hf_api: HfApi = HfApi()
|
|
31
|
+
hugging_face_client: HuggingFaceClient = HuggingFaceClient(hf_api)
|
|
32
|
+
working_dir: WorkingDir = WorkingDir(Path.home() / "anki-addons-dataset")
|
|
33
|
+
facade: Facade = Facade(working_dir, hugging_face_client)
|
|
34
|
+
facade.process(operation, snapshot_date, report_date)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
if __name__ == "__main__":
|
|
38
|
+
main()
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
from argparse import ArgumentParser, Namespace, ArgumentTypeError
|
|
2
|
+
from datetime import date, datetime
|
|
3
|
+
from enum import Enum
|
|
4
|
+
from typing import Optional
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
from anki_addons_dataset.common.data_types import SnapshotDate
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Operation(Enum):
|
|
11
|
+
INIT = "init"
|
|
12
|
+
DOWNLOAD = "download"
|
|
13
|
+
PARSE = "parse"
|
|
14
|
+
REPORT = "report"
|
|
15
|
+
BUNDLE = "bundle"
|
|
16
|
+
UPLOAD = "upload"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ScriptArguments:
|
|
20
|
+
def __init__(self):
|
|
21
|
+
parser: ArgumentParser = ArgumentParser()
|
|
22
|
+
parser.add_argument('operation')
|
|
23
|
+
parser.add_argument('-d', '--snapshot-date', type=self.__valid_date)
|
|
24
|
+
parser.add_argument('-l', '--log-level', type=self.__valid_log_level, default='INFO')
|
|
25
|
+
self.namespace: Namespace = parser.parse_args()
|
|
26
|
+
|
|
27
|
+
def get_snapshot_date(self) -> Optional[SnapshotDate]:
|
|
28
|
+
return self.namespace.snapshot_date
|
|
29
|
+
|
|
30
|
+
def get_operation(self) -> Operation:
|
|
31
|
+
return Operation[self.namespace.operation.upper()]
|
|
32
|
+
|
|
33
|
+
def get_log_level(self) -> int:
|
|
34
|
+
return self.namespace.log_level
|
|
35
|
+
|
|
36
|
+
@staticmethod
|
|
37
|
+
def __valid_date(s: str) -> date:
|
|
38
|
+
try:
|
|
39
|
+
return datetime.strptime(s, "%Y-%m-%d").date()
|
|
40
|
+
except ValueError:
|
|
41
|
+
msg: str = f"Not a valid date: '{s}'. Expected format: YYYY-MM-DD."
|
|
42
|
+
raise ArgumentTypeError(msg)
|
|
43
|
+
|
|
44
|
+
@staticmethod
|
|
45
|
+
def __valid_log_level(s: str) -> int:
|
|
46
|
+
level_name: str = s.upper()
|
|
47
|
+
level_mapping: dict[str, int] = logging.getLevelNamesMapping()
|
|
48
|
+
if level_name not in level_mapping:
|
|
49
|
+
valid_levels: list[str] = list(level_mapping.keys())
|
|
50
|
+
msg: str = f"Not a valid log level: '{s}'. Expected one of: {', '.join(valid_levels)}."
|
|
51
|
+
raise ArgumentTypeError(msg)
|
|
52
|
+
return level_mapping[level_name]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import shutil
|
|
3
|
+
from logging import Logger
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from anki_addons_dataset.common.data_types import SnapshotDate
|
|
8
|
+
from anki_addons_dataset.common.working_dir import WorkingDir, SnapshotDir
|
|
9
|
+
from anki_addons_dataset.huggingface.hugging_face import HuggingFace
|
|
10
|
+
|
|
11
|
+
log: Logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class DatasetBundle:
|
|
15
|
+
def __init__(self, working_dir: WorkingDir):
|
|
16
|
+
self.__working_dir: WorkingDir = working_dir
|
|
17
|
+
|
|
18
|
+
def create_bundle(self) -> None:
|
|
19
|
+
bundle_dir: Path = self.__working_dir.get_bundle_dir()
|
|
20
|
+
log.info(f"Creating bundle in {bundle_dir}")
|
|
21
|
+
shutil.rmtree(bundle_dir, ignore_errors=True)
|
|
22
|
+
self.__copy_snapshots(bundle_dir)
|
|
23
|
+
self.__copy_latest_snapshot(bundle_dir)
|
|
24
|
+
HuggingFace.create_dataset_card(bundle_dir)
|
|
25
|
+
|
|
26
|
+
def __copy_snapshots(self, bundle_dir: Path):
|
|
27
|
+
bundle_history_dir: Path = bundle_dir / "history"
|
|
28
|
+
bundle_history_dir.mkdir(parents=True, exist_ok=True)
|
|
29
|
+
for snapshot_dir in self.__working_dir.list_snapshot_dirs():
|
|
30
|
+
snapshot_date: SnapshotDate = snapshot_dir.snapshot_dir_to_snapshot_date()
|
|
31
|
+
base_name: str = f"{snapshot_date}"
|
|
32
|
+
output_dir: Path = bundle_history_dir / base_name
|
|
33
|
+
self.__create_zip(snapshot_dir.get_raw_dir(), output_dir, "raw")
|
|
34
|
+
self.__create_zip(snapshot_dir.get_stage_dir(), output_dir, "stage")
|
|
35
|
+
log.info(f"Copying {snapshot_dir.get_final_dir()} to {output_dir}")
|
|
36
|
+
shutil.copytree(snapshot_dir.get_final_dir(), output_dir, dirs_exist_ok=True)
|
|
37
|
+
src_metadata_file: Path = snapshot_dir.get_metadata_json()
|
|
38
|
+
dest_metadata_file: Path = output_dir / src_metadata_file.name
|
|
39
|
+
log.info(f"Copying {src_metadata_file} to {dest_metadata_file}")
|
|
40
|
+
shutil.copyfile(src_metadata_file, dest_metadata_file)
|
|
41
|
+
|
|
42
|
+
def __copy_latest_snapshot(self, bundle_dir: Path):
|
|
43
|
+
latest_dir: Path = bundle_dir / "latest"
|
|
44
|
+
log.info(f"Copying the latest snapshot: {latest_dir}")
|
|
45
|
+
latest_snapshot_dir: Optional[SnapshotDir] = self.__working_dir.get_latest_snapshot_dir()
|
|
46
|
+
if not latest_snapshot_dir:
|
|
47
|
+
raise ValueError("No snapshots found")
|
|
48
|
+
final_dir: Path = latest_snapshot_dir.get_final_dir()
|
|
49
|
+
shutil.copytree(final_dir, latest_dir)
|
|
50
|
+
src_metadata_file: Path = latest_snapshot_dir.get_metadata_json()
|
|
51
|
+
dest_metadata_file: Path = latest_dir / src_metadata_file.name
|
|
52
|
+
log.info(f"Copying {src_metadata_file} to {dest_metadata_file}")
|
|
53
|
+
shutil.copyfile(src_metadata_file, dest_metadata_file)
|
|
54
|
+
|
|
55
|
+
@staticmethod
|
|
56
|
+
def __create_zip(source_dir: Path, output_dir: Path, output_base_name: str) -> None:
|
|
57
|
+
zip_base_name: str = str(output_dir / f"{output_base_name}")
|
|
58
|
+
archive_format: str = "zip"
|
|
59
|
+
log.info(f"Creating zip: {zip_base_name}.{archive_format}")
|
|
60
|
+
shutil.make_archive(zip_base_name, archive_format, source_dir)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from logging import Logger
|
|
3
|
+
|
|
4
|
+
from anki_addons_dataset.collector.ankiforum.ankiforum_enricher import AnkiForumEnricher
|
|
5
|
+
from anki_addons_dataset.collector.github.github_enricher import GithubEnricher
|
|
6
|
+
from anki_addons_dataset.collector.overrider.overrider import Overrider
|
|
7
|
+
from anki_addons_dataset.common.data_types import AddonInfo, AddonHeader, AddonInfos
|
|
8
|
+
from anki_addons_dataset.collector.ankiweb.ankiweb_service import AnkiWebService
|
|
9
|
+
|
|
10
|
+
log: Logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class AddonInfosCollector:
|
|
14
|
+
def __init__(self, ankiweb_service: AnkiWebService, github_enricher: GithubEnricher,
|
|
15
|
+
anki_forum_enricher: AnkiForumEnricher, overrider: Overrider):
|
|
16
|
+
self.__ankiweb_service: AnkiWebService = ankiweb_service
|
|
17
|
+
self.__github_enricher: GithubEnricher = github_enricher
|
|
18
|
+
self.__anki_forum_enricher: AnkiForumEnricher = anki_forum_enricher
|
|
19
|
+
self.__overrider: Overrider = overrider
|
|
20
|
+
|
|
21
|
+
def collect_addons(self) -> AddonInfos:
|
|
22
|
+
self.__github_enricher.start()
|
|
23
|
+
self.__anki_forum_enricher.start()
|
|
24
|
+
|
|
25
|
+
addon_headers: list[AddonHeader] = self.__ankiweb_service.get_headers()
|
|
26
|
+
log.info(f"Addon number: {len(addon_headers)}")
|
|
27
|
+
addons_infos: AddonInfos = self.__get_addon_infos(addon_headers)
|
|
28
|
+
log.info("All addons are added to queue")
|
|
29
|
+
self.__github_enricher.wait_download_finish()
|
|
30
|
+
self.__anki_forum_enricher.wait_download_finish()
|
|
31
|
+
github_enriched_addon_infos: AddonInfos = self.__github_enricher.enrich(addons_infos)
|
|
32
|
+
anki_forum_enriched_addon_infos: AddonInfos = self.__anki_forum_enricher.enrich(github_enriched_addon_infos)
|
|
33
|
+
log.info("All addons are enriched")
|
|
34
|
+
|
|
35
|
+
overridden_addon_infos: AddonInfos = self.__overrider.override(anki_forum_enriched_addon_infos)
|
|
36
|
+
return overridden_addon_infos
|
|
37
|
+
|
|
38
|
+
def __get_addon_infos(self, addon_headers: list[AddonHeader]) -> AddonInfos:
|
|
39
|
+
addon_infos: list[AddonInfo] = []
|
|
40
|
+
for i, addon_header in enumerate(addon_headers):
|
|
41
|
+
log.info(f"Parsing addon page: {addon_header.id} ({i}/{len(addon_headers)})")
|
|
42
|
+
addon_info: AddonInfo = self.__ankiweb_service.get_addon_info(addon_header)
|
|
43
|
+
self.__github_enricher.download_in_background(addon_info)
|
|
44
|
+
self.__anki_forum_enricher.download_in_background(addon_info)
|
|
45
|
+
addon_infos.append(addon_info)
|
|
46
|
+
return AddonInfos(addon_infos)
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from typing import cast
|
|
2
|
+
|
|
3
|
+
from anki_addons_dataset.common.data_types import Aggregation, AddonInfos, AddonInfo
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Aggregator:
|
|
7
|
+
|
|
8
|
+
@staticmethod
|
|
9
|
+
def aggregate(addon_infos: AddonInfos) -> Aggregation:
|
|
10
|
+
addon_number: int = len(cast(list[AddonInfo], addon_infos))
|
|
11
|
+
addon_with_github_number: int = len([addon for addon in addon_infos if addon.github.github_repo])
|
|
12
|
+
addon_with_anki_forum_page_number: int = len([addon for addon in addon_infos
|
|
13
|
+
if addon.forum and addon.forum.anki_forum_url])
|
|
14
|
+
addon_with_unit_tests_number: int = len([addon for addon in addon_infos
|
|
15
|
+
if addon.github.tests_count and addon.github.tests_count > 0])
|
|
16
|
+
return Aggregation(addon_number, addon_with_github_number, addon_with_anki_forum_page_number,
|
|
17
|
+
addon_with_unit_tests_number)
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
from typing import Optional
|
|
3
|
+
import logging
|
|
4
|
+
from logging import Logger
|
|
5
|
+
|
|
6
|
+
from anki_addons_dataset.collector.ankiforum.ankiforum_service import AnkiForumService
|
|
7
|
+
from anki_addons_dataset.collector.ankiforum.ankiforum_topic import AnkiForumTopic
|
|
8
|
+
from anki_addons_dataset.collector.enricher import Enricher
|
|
9
|
+
from anki_addons_dataset.common.data_types import AddonInfo, AddonId, \
|
|
10
|
+
AnkiForumInfo, TopicSlug, TopicId, LastPostedAt, AddonInfos, PostsCount, URL
|
|
11
|
+
from anki_addons_dataset.common.json_helper import JsonHelper
|
|
12
|
+
from anki_addons_dataset.common.working_dir import SnapshotDir
|
|
13
|
+
|
|
14
|
+
log: Logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class AnkiForumEnricher(Enricher):
|
|
18
|
+
__name: str = "AnkiForum"
|
|
19
|
+
|
|
20
|
+
def __init__(self, snapshot_dir: SnapshotDir, anki_forum_service: AnkiForumService):
|
|
21
|
+
super().__init__(name=self.__name)
|
|
22
|
+
self.__stage_dir: Path = snapshot_dir.get_stage_dir() / "3-enricher" / "forum"
|
|
23
|
+
self.__anki_forum_service: AnkiForumService = anki_forum_service
|
|
24
|
+
self.__anki_forum_infos: dict[AddonId, Optional[AnkiForumInfo]] = {}
|
|
25
|
+
|
|
26
|
+
def enrich(self, addon_infos: AddonInfos) -> AddonInfos:
|
|
27
|
+
return AddonInfos([self.__enrich(addon_info, self.__anki_forum_infos[addon_info.header.id])
|
|
28
|
+
for addon_info in addon_infos])
|
|
29
|
+
|
|
30
|
+
def _download(self, addon_info: AddonInfo) -> None:
|
|
31
|
+
anki_forum_url: Optional[URL] = addon_info.forum.anki_forum_url
|
|
32
|
+
if anki_forum_url:
|
|
33
|
+
topic_slug: Optional[TopicSlug] = AnkiForumTopic.extract_topic_slug(anki_forum_url)
|
|
34
|
+
topic_id: Optional[TopicId] = AnkiForumTopic.extract_topic_id(anki_forum_url)
|
|
35
|
+
last_posted_at: Optional[LastPostedAt] = self.__anki_forum_service.get_last_posted_at(topic_slug, topic_id)
|
|
36
|
+
posts_count: Optional[PostsCount] = self.__anki_forum_service.get_posts_count(topic_slug, topic_id)
|
|
37
|
+
anki_forum: Optional[AnkiForumInfo] = AnkiForumInfo(anki_forum_url, topic_slug, topic_id, last_posted_at,
|
|
38
|
+
posts_count)
|
|
39
|
+
else:
|
|
40
|
+
anki_forum: Optional[AnkiForumInfo] = None
|
|
41
|
+
self.__anki_forum_infos[addon_info.header.id] = anki_forum
|
|
42
|
+
|
|
43
|
+
def _done(self) -> int:
|
|
44
|
+
return len(self.__anki_forum_infos)
|
|
45
|
+
|
|
46
|
+
def __enrich(self, addon_info: AddonInfo, anki_forum_info: Optional[AnkiForumInfo]) -> AddonInfo:
|
|
47
|
+
enriched_addon_info: AddonInfo = AddonInfo(
|
|
48
|
+
addon_info.header, addon_info.page, addon_info.github, anki_forum_info)
|
|
49
|
+
addon_json_file: Path = self.__stage_dir / f"{addon_info.header.id}.json"
|
|
50
|
+
JsonHelper.write_addon_info_to_file(addon_info, addon_json_file)
|
|
51
|
+
log.info(f"Enriched ({self.__name}): {addon_info.header.id}")
|
|
52
|
+
return enriched_addon_info
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import logging
|
|
3
|
+
from datetime import datetime, timezone
|
|
4
|
+
from logging import Logger
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
from pydiscourse import DiscourseClient
|
|
9
|
+
|
|
10
|
+
from anki_addons_dataset.common.data_types import TopicId, TopicSlug, LastPostedAt, PostsCount
|
|
11
|
+
from anki_addons_dataset.common.working_dir import SnapshotDir
|
|
12
|
+
|
|
13
|
+
log: Logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class AnkiForumService:
|
|
17
|
+
|
|
18
|
+
def __init__(self, discourse_client: DiscourseClient, snapshot_dir: SnapshotDir, offline: bool):
|
|
19
|
+
self.__discourse_client: DiscourseClient = discourse_client
|
|
20
|
+
raw_dir: Path = snapshot_dir.get_raw_dir() / "3-forum"
|
|
21
|
+
self.__topic = raw_dir / "topic"
|
|
22
|
+
self.__topic.mkdir(parents=True, exist_ok=True)
|
|
23
|
+
self.__offline: bool = offline
|
|
24
|
+
|
|
25
|
+
def get_last_posted_at(self, topic_slug: Optional[TopicSlug], topic_id: Optional[TopicId]) -> Optional[
|
|
26
|
+
LastPostedAt]:
|
|
27
|
+
topic_dict: Optional[dict] = self.__read_topic_json(topic_slug, topic_id)
|
|
28
|
+
if not topic_dict:
|
|
29
|
+
return None
|
|
30
|
+
last_posted_at_str: str = topic_dict['last_posted_at']
|
|
31
|
+
last_posted_at: datetime = datetime.strptime(last_posted_at_str, "%Y-%m-%dT%H:%M:%S.%fZ").replace(
|
|
32
|
+
tzinfo=timezone.utc)
|
|
33
|
+
return LastPostedAt(last_posted_at)
|
|
34
|
+
|
|
35
|
+
def get_posts_count(self, topic_slug: Optional[TopicSlug], topic_id: Optional[TopicId]) -> Optional[PostsCount]:
|
|
36
|
+
topic_dict: Optional[dict] = self.__read_topic_json(topic_slug, topic_id)
|
|
37
|
+
if not topic_dict:
|
|
38
|
+
return None
|
|
39
|
+
posts_count_str: str = topic_dict['posts_count']
|
|
40
|
+
return PostsCount(int(posts_count_str))
|
|
41
|
+
|
|
42
|
+
def __read_topic_json(self, topic_slug: Optional[TopicSlug], topic_id: Optional[TopicId]) -> Optional[dict]:
|
|
43
|
+
json_file: Path = self.__topic / f"{topic_id}.json"
|
|
44
|
+
if json_file.exists():
|
|
45
|
+
json_str: str = json_file.read_text()
|
|
46
|
+
if json_str == "None":
|
|
47
|
+
return None
|
|
48
|
+
topic_dict: dict = json.loads(json_str)
|
|
49
|
+
else:
|
|
50
|
+
if self.__offline:
|
|
51
|
+
log.debug(f"Offline mode is enabled. Skip fetching last_posted_at for topic '{topic_slug}/{topic_id}'")
|
|
52
|
+
return None
|
|
53
|
+
topic_dict: dict = self.__discourse_client.topic(
|
|
54
|
+
topic_slug, topic_id, override_request_kwargs={"allow_redirects": True})
|
|
55
|
+
if topic_dict is None:
|
|
56
|
+
json_file.write_text("None")
|
|
57
|
+
return None
|
|
58
|
+
topic_json: str = json.dumps(topic_dict, indent=2)
|
|
59
|
+
json_file.write_text(topic_json)
|
|
60
|
+
return topic_dict
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from re import Match
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from anki_addons_dataset.common.data_types import URL, TopicId, TopicSlug
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class AnkiForumTopic:
|
|
9
|
+
__url_pattern: re.Pattern = re.compile(r'^https://forums\.ankiweb\.net/t/([^/]+)/(\d+)')
|
|
10
|
+
|
|
11
|
+
@staticmethod
|
|
12
|
+
def extract_topic_slug(topic_url: Optional[URL]) -> Optional[TopicSlug]:
|
|
13
|
+
if topic_url is None:
|
|
14
|
+
return None
|
|
15
|
+
match: Optional[Match[str]] = re.search(AnkiForumTopic.__url_pattern, topic_url)
|
|
16
|
+
if match:
|
|
17
|
+
topic_slug_str: str = match.group(1)
|
|
18
|
+
return TopicSlug(topic_slug_str)
|
|
19
|
+
else:
|
|
20
|
+
raise ValueError(f"Cannot extract Topic Slug from Anki Forum Topic URL: '{topic_url}'")
|
|
21
|
+
|
|
22
|
+
@staticmethod
|
|
23
|
+
def extract_topic_id(topic_url: Optional[URL]) -> Optional[TopicId]:
|
|
24
|
+
if topic_url is None:
|
|
25
|
+
return None
|
|
26
|
+
match: Optional[Match[str]] = re.search(AnkiForumTopic.__url_pattern, topic_url)
|
|
27
|
+
if match:
|
|
28
|
+
topic_id_str: str = match.group(2)
|
|
29
|
+
return TopicId(int(topic_id_str))
|
|
30
|
+
else:
|
|
31
|
+
raise ValueError(f"Cannot extract Topic ID from Anki Forum Topic URL: '{topic_url}'")
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import datetime
|
|
2
|
+
import re
|
|
3
|
+
from datetime import date
|
|
4
|
+
from re import Match
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from anki_addons_dataset.common.data_types import AddonBranch, AnkiVersion
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class AddonBranchParser:
|
|
11
|
+
__addon_branch_re: re.Pattern[str] = re.compile(
|
|
12
|
+
r'^(?P<min>\d+(?:\.\d+)*)'
|
|
13
|
+
r'(?:-(?P<max>\d+(?:\.\d+)*\+?)|(?P<plus>\+))?'
|
|
14
|
+
r' \(Updated (?P<updated>\d{4}-\d{2}-\d{2})\)$'
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
@staticmethod
|
|
18
|
+
def extract_addon_branch(addon_branch_str: str) -> AddonBranch:
|
|
19
|
+
match: Optional[Match[str]] = AddonBranchParser.__addon_branch_re.fullmatch(addon_branch_str)
|
|
20
|
+
if not match:
|
|
21
|
+
raise ValueError(f"Cannot parse version string: {addon_branch_str!r}")
|
|
22
|
+
min_addon_version: AnkiVersion = AnkiVersion(match.group("min"))
|
|
23
|
+
max_addon_version: AnkiVersion = AnkiVersion(match.group("max"))
|
|
24
|
+
if max_addon_version is None and match.group("plus") is not None:
|
|
25
|
+
max_addon_version = AnkiVersion("+")
|
|
26
|
+
updated: date = datetime.date.fromisoformat(match.group("updated"))
|
|
27
|
+
return AddonBranch(min_anki_version=min_addon_version, max_anki_version=max_addon_version, updated=updated)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
import logging
|
|
3
|
+
from logging import Logger
|
|
4
|
+
|
|
5
|
+
from anki_addons_dataset.collector.ankiweb.addon_page_parser import AddonPageParser
|
|
6
|
+
from anki_addons_dataset.collector.ankiweb.page_downloader import PageDownloader
|
|
7
|
+
from anki_addons_dataset.common.data_types import AddonId, AddonInfo, AddonHeader, HtmlStr
|
|
8
|
+
from anki_addons_dataset.common.json_helper import JsonHelper
|
|
9
|
+
from anki_addons_dataset.common.working_dir import SnapshotDir
|
|
10
|
+
|
|
11
|
+
log: Logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class AddonPageDownloader:
|
|
15
|
+
def __init__(self, page_downloader: PageDownloader, snapshot_dir: SnapshotDir, addon_page_parser: AddonPageParser,
|
|
16
|
+
offline: bool) -> None:
|
|
17
|
+
self.__addon_page_parser: AddonPageParser = addon_page_parser
|
|
18
|
+
self.__page_downloader: PageDownloader = page_downloader
|
|
19
|
+
self.__raw_dir: Path = snapshot_dir.get_raw_dir() / "1-anki-web"
|
|
20
|
+
self.__stage_dir: Path = snapshot_dir.get_stage_dir() / "1-anki-web"
|
|
21
|
+
self.__offline: bool = offline
|
|
22
|
+
|
|
23
|
+
def get_addon_info(self, addon_header: AddonHeader) -> AddonInfo:
|
|
24
|
+
try:
|
|
25
|
+
html: HtmlStr = self.__load_addon_page(addon_header.id)
|
|
26
|
+
addon_info: AddonInfo = self.__addon_page_parser.parse_addon_page(addon_header, html)
|
|
27
|
+
addon_json_file: Path = self.__stage_dir / "addon" / f"{addon_header.id}.json"
|
|
28
|
+
JsonHelper.write_addon_info_to_file(addon_info, addon_json_file)
|
|
29
|
+
return addon_info
|
|
30
|
+
except Exception as e:
|
|
31
|
+
raise RuntimeError(f"Cannot get addon info: {addon_header.id}") from e
|
|
32
|
+
|
|
33
|
+
def __load_addon_page(self, addon_id: AddonId) -> HtmlStr:
|
|
34
|
+
raw_file: Path = self.__raw_dir / "addon" / f"{addon_id}.html"
|
|
35
|
+
if raw_file.exists() and raw_file.stat().st_size == 0:
|
|
36
|
+
raw_file.unlink()
|
|
37
|
+
log.info(f"Removed empty file: {raw_file}")
|
|
38
|
+
if not raw_file.exists():
|
|
39
|
+
log.debug(f"Downloading addon page to {raw_file}")
|
|
40
|
+
if self.__offline:
|
|
41
|
+
raise RuntimeError("Offline mode is enabled")
|
|
42
|
+
raw_file.parent.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
html: HtmlStr = self.__page_downloader.load_page(f"https://ankiweb.net/shared/info/{addon_id}")
|
|
44
|
+
raw_file.write_text(html)
|
|
45
|
+
return HtmlStr(raw_file.read_text())
|