udata-hydra 2.0.0.dev2124__tar.gz → 2.0.0.dev2347__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/PKG-INFO +6 -1
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/README.md +5 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/pyproject.toml +10 -8
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/analysis/csv.py +9 -8
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/analysis/resource.py +6 -12
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/app.py +28 -97
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/cli.py +7 -8
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/crawl.py +37 -44
- udata_hydra-2.0.0.dev2347/udata_hydra/db/__init__.py +46 -0
- udata_hydra-2.0.0.dev2347/udata_hydra/db/check.py +73 -0
- udata_hydra-2.0.0.dev2347/udata_hydra/db/resource.py +74 -0
- udata_hydra-2.0.0.dev2347/udata_hydra/schemas/__init__.py +2 -0
- udata_hydra-2.0.0.dev2347/udata_hydra/schemas/check.py +27 -0
- udata_hydra-2.0.0.dev2347/udata_hydra/schemas/resource_query.py +26 -0
- udata_hydra-2.0.0.dev2124/udata_hydra/utils/db.py +0 -73
- udata_hydra-2.0.0.dev2124/udata_hydra/utils/json.py +0 -11
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/__init__.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/analysis/__init__.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/analysis/errors.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/analysis/helpers.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/config_default.toml +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/context.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/logger.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/__init__.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/csv/20221205_initial_up_rev1.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/csv/20230130_drop_migrations.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/csv/20230206_datetime_aware.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20221205_initial_up_rev1.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20221206_rev1_up_rev2.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20221206_rev2_up_rev3.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20221208_rev3_up_rev4.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20221208_rev4_up_rev5.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230119_rev5_up_rev6.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230121_rev6_up_rev7.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230121_rev7_up_rev8.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230130_drop_migrations.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230206_datetime_aware.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230515_rev8_up_rev9.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20230606_rev9_up_rev10.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/migrations/main/20231102_drop_csv_analysis.sql +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/__init__.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/app_version.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/csv.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/file.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/http.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/minio.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/queue.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/reader.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/utils/timer.py +0 -0
- {udata_hydra-2.0.0.dev2124 → udata_hydra-2.0.0.dev2347}/udata_hydra/worker.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: udata-hydra
|
|
3
|
-
Version: 2.0.0.
|
|
3
|
+
Version: 2.0.0.dev2347
|
|
4
4
|
Summary: Async crawler and parsing service for data.gouv.fr
|
|
5
5
|
License: MIT
|
|
6
6
|
Author: Opendata Team
|
|
@@ -105,6 +105,11 @@ Then you can run the tests with `poetry run pytest`.
|
|
|
105
105
|
|
|
106
106
|
If you would like to see print statements as they are executed, you can pass the -s flag to pytest (`poetry run pytest -s`). However, note that this can sometimes be difficult to parse.
|
|
107
107
|
|
|
108
|
+
### Tests coverage
|
|
109
|
+
|
|
110
|
+
Pytest automatically uses the `coverage` package to generate a coverage report, which is displayed at the end of the test run in the terminal.
|
|
111
|
+
The coverage is configured in the `pypoject.toml` file, in the `[tool.pytest.ini_options]` section.
|
|
112
|
+
You can also override the coverage report configuration when running the tests by passing some flags like `--cov-report` to pytest. See [the pytest-cov documentation](https://pytest-cov.readthedocs.io/en/latest/config.html) for more information.
|
|
108
113
|
|
|
109
114
|
## API
|
|
110
115
|
|
|
@@ -69,6 +69,11 @@ Then you can run the tests with `poetry run pytest`.
|
|
|
69
69
|
|
|
70
70
|
If you would like to see print statements as they are executed, you can pass the -s flag to pytest (`poetry run pytest -s`). However, note that this can sometimes be difficult to parse.
|
|
71
71
|
|
|
72
|
+
### Tests coverage
|
|
73
|
+
|
|
74
|
+
Pytest automatically uses the `coverage` package to generate a coverage report, which is displayed at the end of the test run in the terminal.
|
|
75
|
+
The coverage is configured in the `pypoject.toml` file, in the `[tool.pytest.ini_options]` section.
|
|
76
|
+
You can also override the coverage report configuration when running the tests by passing some flags like `--cov-report` to pytest. See [the pytest-cov documentation](https://pytest-cov.readthedocs.io/en/latest/config.html) for more information.
|
|
72
77
|
|
|
73
78
|
## API
|
|
74
79
|
|
|
@@ -9,7 +9,7 @@ readme = "README.md"
|
|
|
9
9
|
|
|
10
10
|
[tool.poetry]
|
|
11
11
|
name = "udata-hydra"
|
|
12
|
-
version = "2.0.0.
|
|
12
|
+
version = "2.0.0.dev2347"
|
|
13
13
|
description = "Async crawler and parsing service for data.gouv.fr"
|
|
14
14
|
authors = ["Opendata Team <opendatateam@data.gouv.fr>"]
|
|
15
15
|
license = "MIT"
|
|
@@ -47,6 +47,7 @@ mypy = "^1.11.0"
|
|
|
47
47
|
nest_asyncio = "^1.5.5"
|
|
48
48
|
pytest = "^7.1.1"
|
|
49
49
|
pytest-asyncio = "^0.18.3"
|
|
50
|
+
pytest-cov = "^5.0.0"
|
|
50
51
|
pytest-mock = "^3.7.0"
|
|
51
52
|
ruff = "^0.5.2"
|
|
52
53
|
|
|
@@ -66,6 +67,14 @@ module = [
|
|
|
66
67
|
]
|
|
67
68
|
ignore_missing_imports = true
|
|
68
69
|
|
|
70
|
+
[tool.pytest.ini_options]
|
|
71
|
+
asyncio_mode = "auto"
|
|
72
|
+
markers = [
|
|
73
|
+
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
|
|
74
|
+
"catalog_harvested: use catalog_harvested.csv as source",
|
|
75
|
+
]
|
|
76
|
+
addopts = "--cov=udata_hydra/ --cov-report term-missing"
|
|
77
|
+
|
|
69
78
|
[tool.ruff]
|
|
70
79
|
lint = { select = ["I"] } # also sort imports with an isort rule
|
|
71
80
|
line-length = 100
|
|
@@ -74,13 +83,6 @@ line-length = 100
|
|
|
74
83
|
requires = ["poetry-core>=1.0.0"]
|
|
75
84
|
build-backend = "poetry.core.masonry.api"
|
|
76
85
|
|
|
77
|
-
[tool.pytest.ini_options]
|
|
78
|
-
asyncio_mode = "strict"
|
|
79
|
-
markers = [
|
|
80
|
-
"slow: marks tests as slow (deselect with '-m \"not slow\"')",
|
|
81
|
-
"catalog_harvested: use catalog_harvested.csv as source",
|
|
82
|
-
]
|
|
83
|
-
|
|
84
86
|
[tool.poetry.scripts]
|
|
85
87
|
udata-hydra = "udata_hydra.cli:run"
|
|
86
88
|
udata-hydra-crawl = "udata_hydra.crawl:run"
|
|
@@ -32,8 +32,9 @@ from str2float import str2float
|
|
|
32
32
|
from udata_hydra import config, context
|
|
33
33
|
from udata_hydra.analysis import helpers
|
|
34
34
|
from udata_hydra.analysis.errors import ParseException
|
|
35
|
+
from udata_hydra.db import compute_insert_query
|
|
36
|
+
from udata_hydra.db.check import Check
|
|
35
37
|
from udata_hydra.utils import queue
|
|
36
|
-
from udata_hydra.utils.db import compute_insert_query, get_check, update_check
|
|
37
38
|
from udata_hydra.utils.file import download_resource
|
|
38
39
|
from udata_hydra.utils.http import send
|
|
39
40
|
from udata_hydra.utils.reader import Reader
|
|
@@ -76,7 +77,7 @@ RESERVED_COLS = ("__id", "tableoid", "xmin", "cmin", "xmax", "cmax", "ctid")
|
|
|
76
77
|
|
|
77
78
|
async def notify_udata(check_id: int) -> None:
|
|
78
79
|
"""Notify udata of the result of a parsing"""
|
|
79
|
-
check = await
|
|
80
|
+
check = await Check.get(check_id)
|
|
80
81
|
resource_id = check["resource_id"]
|
|
81
82
|
db = await context.pool()
|
|
82
83
|
record = await db.fetchrow("SELECT dataset_id FROM catalog WHERE resource_id = $1", resource_id)
|
|
@@ -112,7 +113,7 @@ async def analyse_csv(
|
|
|
112
113
|
|
|
113
114
|
timer = Timer("analyse-csv")
|
|
114
115
|
assert any(_ is not None for _ in (check_id, url))
|
|
115
|
-
check = await
|
|
116
|
+
check = await Check.get(check_id) if check_id is not None else {}
|
|
116
117
|
url = check.get("url") or url
|
|
117
118
|
exception_file = str(check.get("resource_id", "")) in exceptions
|
|
118
119
|
|
|
@@ -131,13 +132,13 @@ async def analyse_csv(
|
|
|
131
132
|
|
|
132
133
|
try:
|
|
133
134
|
if check_id:
|
|
134
|
-
await
|
|
135
|
+
await Check.update(check_id, {"parsing_started_at": datetime.now(timezone.utc)})
|
|
135
136
|
csv_inspection = await perform_csv_inspection(tmp_file.name)
|
|
136
137
|
timer.mark("csv-inspection")
|
|
137
138
|
await csv_to_db(tmp_file.name, csv_inspection, table_name, debug_insert=debug_insert)
|
|
138
139
|
timer.mark("csv-to-db")
|
|
139
140
|
if check_id:
|
|
140
|
-
await
|
|
141
|
+
await Check.update(
|
|
141
142
|
check_id,
|
|
142
143
|
{
|
|
143
144
|
"parsing_table": table_name,
|
|
@@ -234,7 +235,7 @@ async def csv_to_db(
|
|
|
234
235
|
for r in bar.iter(records):
|
|
235
236
|
data = {k: v for k, v in zip(columns.keys(), r)}
|
|
236
237
|
# NB: possible sql injection here, but should not be used in prod
|
|
237
|
-
q = compute_insert_query(
|
|
238
|
+
q = compute_insert_query(table_name=table_name, data=data, returning="__id")
|
|
238
239
|
await db.execute(q, *data.values())
|
|
239
240
|
|
|
240
241
|
|
|
@@ -274,7 +275,7 @@ async def handle_parse_exception(e: Exception, check_id: int, table_name: str) -
|
|
|
274
275
|
await db.execute(f'DROP TABLE IF EXISTS "{table_name}"')
|
|
275
276
|
if check_id:
|
|
276
277
|
if config.SENTRY_DSN:
|
|
277
|
-
check = await
|
|
278
|
+
check = await Check.get(check_id)
|
|
278
279
|
url = check["url"]
|
|
279
280
|
with sentry_sdk.push_scope() as scope:
|
|
280
281
|
scope.set_extra("check_id", check_id)
|
|
@@ -284,7 +285,7 @@ async def handle_parse_exception(e: Exception, check_id: int, table_name: str) -
|
|
|
284
285
|
# e.__cause__ let us access the "inherited" error of ParseException (raise e from cause)
|
|
285
286
|
# it's called explicit exception chaining and it's very cool, look it up (PEP 3134)!
|
|
286
287
|
err = f"{e.step}:sentry:{event_id}" if config.SENTRY_DSN else f"{e.step}:{str(e.__cause__)}"
|
|
287
|
-
await
|
|
288
|
+
await Check.update(
|
|
288
289
|
check_id,
|
|
289
290
|
{"parsing_error": err, "parsing_finished_at": datetime.now(timezone.utc)},
|
|
290
291
|
)
|
|
@@ -10,9 +10,9 @@ from dateparser import parse as date_parser
|
|
|
10
10
|
|
|
11
11
|
from udata_hydra import config, context
|
|
12
12
|
from udata_hydra.analysis.csv import analyse_csv
|
|
13
|
+
from udata_hydra.db.check import Check
|
|
13
14
|
from udata_hydra.utils import queue
|
|
14
15
|
from udata_hydra.utils.csv import detect_tabular_from_headers
|
|
15
|
-
from udata_hydra.utils.db import get_check, update_check
|
|
16
16
|
from udata_hydra.utils.file import compute_checksum_from_file, download_resource
|
|
17
17
|
from udata_hydra.utils.http import send
|
|
18
18
|
|
|
@@ -37,7 +37,7 @@ async def process_resource(check_id: int, is_first_check: bool) -> None:
|
|
|
37
37
|
|
|
38
38
|
Will call udata if first check or changes found, and update check with optionnal infos
|
|
39
39
|
"""
|
|
40
|
-
check: dict = await
|
|
40
|
+
check: dict = await Check.get(check_id)
|
|
41
41
|
if not check:
|
|
42
42
|
log.error(f"Check not found by id {check_id}")
|
|
43
43
|
return
|
|
@@ -85,7 +85,7 @@ async def process_resource(check_id: int, is_first_check: bool) -> None:
|
|
|
85
85
|
finally:
|
|
86
86
|
if tmp_file and not is_tabular:
|
|
87
87
|
os.remove(tmp_file.name)
|
|
88
|
-
await
|
|
88
|
+
await Check.update(
|
|
89
89
|
check_id,
|
|
90
90
|
{
|
|
91
91
|
"checksum": dl_analysis.get("analysis:checksum"),
|
|
@@ -96,7 +96,7 @@ async def process_resource(check_id: int, is_first_check: bool) -> None:
|
|
|
96
96
|
)
|
|
97
97
|
|
|
98
98
|
if change_status == Change.HAS_CHANGED:
|
|
99
|
-
await store_last_modified_date(change_payload or {},
|
|
99
|
+
await store_last_modified_date(change_payload or {}, check_id)
|
|
100
100
|
|
|
101
101
|
analysis_results = {**dl_analysis, **(change_payload or {})}
|
|
102
102
|
if change_status == Change.HAS_CHANGED or is_first_check:
|
|
@@ -111,20 +111,14 @@ async def process_resource(check_id: int, is_first_check: bool) -> None:
|
|
|
111
111
|
)
|
|
112
112
|
|
|
113
113
|
|
|
114
|
-
async def store_last_modified_date(change_analysis
|
|
114
|
+
async def store_last_modified_date(change_analysis: dict, check_id: int) -> None:
|
|
115
115
|
"""
|
|
116
116
|
Store last modified date in checks because it may be useful for later comparison
|
|
117
117
|
"""
|
|
118
|
-
pool = await context.pool()
|
|
119
118
|
last_modified = change_analysis.get("analysis:last-modified-at")
|
|
120
119
|
if last_modified:
|
|
121
120
|
last_modified = datetime.fromisoformat(last_modified)
|
|
122
|
-
|
|
123
|
-
await conn.execute(
|
|
124
|
-
"UPDATE checks SET detected_last_modified_at = $1 WHERE id = $2",
|
|
125
|
-
last_modified,
|
|
126
|
-
check_id,
|
|
127
|
-
)
|
|
121
|
+
await Check.update(check_id, {"detected_last_modified_at": last_modified})
|
|
128
122
|
|
|
129
123
|
|
|
130
124
|
async def detect_resource_change_from_checksum(
|
|
@@ -4,11 +4,14 @@ from datetime import datetime, timedelta, timezone
|
|
|
4
4
|
|
|
5
5
|
from aiohttp import web
|
|
6
6
|
from humanfriendly import parse_timespan
|
|
7
|
-
from marshmallow import
|
|
7
|
+
from marshmallow import ValidationError
|
|
8
8
|
|
|
9
9
|
from udata_hydra import config, context
|
|
10
10
|
from udata_hydra.crawl import get_excluded_clause
|
|
11
|
+
from udata_hydra.db.check import Check
|
|
12
|
+
from udata_hydra.db.resource import Resource
|
|
11
13
|
from udata_hydra.logger import setup_logging
|
|
14
|
+
from udata_hydra.schemas import CheckSchema, ResourceQuerySchema
|
|
12
15
|
from udata_hydra.utils.minio import delete_resource_from_minio
|
|
13
16
|
from udata_hydra.worker import QUEUES
|
|
14
17
|
|
|
@@ -16,52 +19,8 @@ log = setup_logging()
|
|
|
16
19
|
routes = web.RouteTableDef()
|
|
17
20
|
|
|
18
21
|
|
|
19
|
-
class CheckSchema(Schema):
|
|
20
|
-
check_id = fields.Integer(data_key="id")
|
|
21
|
-
catalog_id = fields.Integer()
|
|
22
|
-
url = fields.Str()
|
|
23
|
-
domain = fields.Str()
|
|
24
|
-
created_at = fields.DateTime()
|
|
25
|
-
check_status = fields.Integer(data_key="status")
|
|
26
|
-
headers = fields.Function(lambda obj: json.loads(obj["headers"]) if obj["headers"] else {})
|
|
27
|
-
timeout = fields.Boolean()
|
|
28
|
-
response_time = fields.Float()
|
|
29
|
-
error = fields.Str()
|
|
30
|
-
dataset_id = fields.Str()
|
|
31
|
-
resource_id = fields.UUID()
|
|
32
|
-
deleted = fields.Boolean()
|
|
33
|
-
parsing_started_at = fields.DateTime()
|
|
34
|
-
parsing_finished_at = fields.DateTime()
|
|
35
|
-
parsing_error = fields.Str()
|
|
36
|
-
parsing_table = fields.Str()
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
class ResourceDocument(Schema):
|
|
40
|
-
id = fields.Str(required=True)
|
|
41
|
-
url = fields.Str(required=True)
|
|
42
|
-
format = fields.Str(allow_none=True)
|
|
43
|
-
title = fields.Str(required=True)
|
|
44
|
-
schema = fields.Dict(allow_none=True)
|
|
45
|
-
description = fields.Str(allow_none=True)
|
|
46
|
-
filetype = fields.Str(required=True)
|
|
47
|
-
type = fields.Str(required=True)
|
|
48
|
-
mime = fields.Str(allow_none=True)
|
|
49
|
-
filesize = fields.Int(allow_none=True)
|
|
50
|
-
checksum_type = fields.Str(allow_none=True)
|
|
51
|
-
checksum_value = fields.Str(allow_none=True)
|
|
52
|
-
created_at = fields.DateTime(required=True)
|
|
53
|
-
last_modified = fields.DateTime(required=True)
|
|
54
|
-
extras = fields.Dict()
|
|
55
|
-
harvest = fields.Dict()
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
class ResourceQuery(Schema):
|
|
59
|
-
dataset_id = fields.Str(required=True)
|
|
60
|
-
resource_id = fields.Str(required=True)
|
|
61
|
-
document = fields.Nested(ResourceDocument(), allow_none=True)
|
|
62
|
-
|
|
63
|
-
|
|
64
22
|
def _get_args(request, params=("url", "resource_id")) -> list:
|
|
23
|
+
"""Get GET parameters from request"""
|
|
65
24
|
data = [request.query.get(param) for param in params]
|
|
66
25
|
if not any(data):
|
|
67
26
|
raise web.HTTPBadRequest()
|
|
@@ -70,9 +29,14 @@ def _get_args(request, params=("url", "resource_id")) -> list:
|
|
|
70
29
|
|
|
71
30
|
@routes.post("/api/resource/created/")
|
|
72
31
|
async def resource_created(request: web.Request) -> web.Response:
|
|
32
|
+
"""Endpoint to receive a resource creation event from a source
|
|
33
|
+
Will create a new resource in the DB "catalog" table and mark it as priority for next crawling
|
|
34
|
+
Respond with a 200 status code and a JSON body with a message key set to "created"
|
|
35
|
+
If error, respond with a 400 status code
|
|
36
|
+
"""
|
|
73
37
|
try:
|
|
74
38
|
payload = await request.json()
|
|
75
|
-
valid_payload =
|
|
39
|
+
valid_payload: dict = ResourceQuerySchema().load(payload)
|
|
76
40
|
except ValidationError as err:
|
|
77
41
|
raise web.HTTPBadRequest(text=json.dumps(err.messages))
|
|
78
42
|
|
|
@@ -83,26 +47,26 @@ async def resource_created(request: web.Request) -> web.Response:
|
|
|
83
47
|
dataset_id = valid_payload["dataset_id"]
|
|
84
48
|
resource_id = valid_payload["resource_id"]
|
|
85
49
|
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
ON CONFLICT (resource_id) DO UPDATE SET
|
|
93
|
-
priority = TRUE,
|
|
94
|
-
url = '{resource["url"]}',
|
|
95
|
-
dataset_id = '{dataset_id}';"""
|
|
96
|
-
await connection.execute(q)
|
|
50
|
+
await Resource.insert(
|
|
51
|
+
dataset_id=dataset_id,
|
|
52
|
+
resource_id=resource_id,
|
|
53
|
+
url=resource["url"],
|
|
54
|
+
priority=True,
|
|
55
|
+
)
|
|
97
56
|
|
|
98
57
|
return web.json_response({"message": "created"})
|
|
99
58
|
|
|
100
59
|
|
|
101
60
|
@routes.post("/api/resource/updated/")
|
|
102
61
|
async def resource_updated(request: web.Request) -> web.Response:
|
|
62
|
+
"""Endpoint to receive a resource update event from a source
|
|
63
|
+
Will update an existing resource in the DB "catalog" table and mark it as priority for next crawling
|
|
64
|
+
Respond with a 200 status code and a JSON body with a message key set to "updated"
|
|
65
|
+
If error, respond with a 400 status code
|
|
66
|
+
"""
|
|
103
67
|
try:
|
|
104
68
|
payload = await request.json()
|
|
105
|
-
valid_payload =
|
|
69
|
+
valid_payload: dict = ResourceQuerySchema().load(payload)
|
|
106
70
|
except ValidationError as err:
|
|
107
71
|
raise web.HTTPBadRequest(text=json.dumps(err.messages))
|
|
108
72
|
|
|
@@ -113,24 +77,7 @@ async def resource_updated(request: web.Request) -> web.Response:
|
|
|
113
77
|
dataset_id = valid_payload["dataset_id"]
|
|
114
78
|
resource_id = valid_payload["resource_id"]
|
|
115
79
|
|
|
116
|
-
|
|
117
|
-
async with pool.acquire() as connection:
|
|
118
|
-
# Make resource high priority for crawling
|
|
119
|
-
# Check if resource is in catalog then insert or update into table
|
|
120
|
-
q = f"""SELECT * FROM catalog WHERE resource_id = '{resource_id}';"""
|
|
121
|
-
res = await connection.fetch(q)
|
|
122
|
-
if len(res):
|
|
123
|
-
q = f"""UPDATE catalog SET priority = TRUE, url = '{resource["url"]}'
|
|
124
|
-
WHERE resource_id = '{resource_id}';"""
|
|
125
|
-
else:
|
|
126
|
-
q = f"""
|
|
127
|
-
INSERT INTO catalog (dataset_id, resource_id, url, deleted, priority)
|
|
128
|
-
VALUES ('{dataset_id}', '{resource_id}', '{resource["url"]}', FALSE, TRUE)
|
|
129
|
-
ON CONFLICT (resource_id) DO UPDATE SET
|
|
130
|
-
priority = TRUE,
|
|
131
|
-
url = '{resource["url"]}',
|
|
132
|
-
dataset_id = '{dataset_id}';"""
|
|
133
|
-
await connection.execute(q)
|
|
80
|
+
await Resource.update_or_insert(dataset_id, resource_id, resource["url"])
|
|
134
81
|
|
|
135
82
|
return web.json_response({"message": "updated"})
|
|
136
83
|
|
|
@@ -139,7 +86,7 @@ async def resource_updated(request: web.Request) -> web.Response:
|
|
|
139
86
|
async def resource_deleted(request: web.Request) -> web.Response:
|
|
140
87
|
try:
|
|
141
88
|
payload = await request.json()
|
|
142
|
-
valid_payload =
|
|
89
|
+
valid_payload: dict = ResourceQuerySchema().load(payload)
|
|
143
90
|
except ValidationError as err:
|
|
144
91
|
raise web.HTTPBadRequest(text=json.dumps(err.messages))
|
|
145
92
|
|
|
@@ -159,16 +106,9 @@ async def resource_deleted(request: web.Request) -> web.Response:
|
|
|
159
106
|
|
|
160
107
|
@routes.get("/api/checks/latest/")
|
|
161
108
|
async def get_check(request: web.Request) -> web.Response:
|
|
109
|
+
"""Get the latest check for a given URL or resource_id"""
|
|
162
110
|
url, resource_id = _get_args(request)
|
|
163
|
-
|
|
164
|
-
q = f"""
|
|
165
|
-
SELECT catalog.id as catalog_id, checks.id as check_id,
|
|
166
|
-
catalog.status as catalog_status, checks.status as check_status, *
|
|
167
|
-
FROM checks, catalog
|
|
168
|
-
WHERE checks.id = catalog.last_check
|
|
169
|
-
AND catalog.{column} = $1
|
|
170
|
-
"""
|
|
171
|
-
data = await request.app["pool"].fetchrow(q, url or resource_id)
|
|
111
|
+
data = await Check.get_latest(url, resource_id)
|
|
172
112
|
if not data:
|
|
173
113
|
raise web.HTTPNotFound()
|
|
174
114
|
if data["deleted"]:
|
|
@@ -179,16 +119,7 @@ async def get_check(request: web.Request) -> web.Response:
|
|
|
179
119
|
@routes.get("/api/checks/all/")
|
|
180
120
|
async def get_checks(request: web.Request) -> web.Response:
|
|
181
121
|
url, resource_id = _get_args(request)
|
|
182
|
-
|
|
183
|
-
q = f"""
|
|
184
|
-
SELECT catalog.id as catalog_id, checks.id as check_id,
|
|
185
|
-
catalog.status as catalog_status, checks.status as check_status, *
|
|
186
|
-
FROM checks, catalog
|
|
187
|
-
WHERE catalog.{column} = $1
|
|
188
|
-
AND catalog.url = checks.url
|
|
189
|
-
ORDER BY created_at DESC
|
|
190
|
-
"""
|
|
191
|
-
data = await request.app["pool"].fetch(q, url or resource_id)
|
|
122
|
+
data = await Check.get_all(url, resource_id)
|
|
192
123
|
if not data:
|
|
193
124
|
raise web.HTTPNotFound()
|
|
194
125
|
return web.json_response([CheckSchema().dump(dict(r)) for r in data])
|
|
@@ -15,6 +15,7 @@ from progressist import ProgressBar
|
|
|
15
15
|
from udata_hydra import config
|
|
16
16
|
from udata_hydra.analysis.csv import analyse_csv, delete_table
|
|
17
17
|
from udata_hydra.crawl import check_url as crawl_check_url
|
|
18
|
+
from udata_hydra.db.resource import Resource
|
|
18
19
|
from udata_hydra.logger import setup_logging
|
|
19
20
|
from udata_hydra.migrations import Migrator
|
|
20
21
|
|
|
@@ -71,10 +72,10 @@ async def load_catalog(url=None, drop_meta=False, drop_all=False, quiet=False):
|
|
|
71
72
|
yield row
|
|
72
73
|
|
|
73
74
|
try:
|
|
74
|
-
log.info(f"Downloading catalog from {url}...")
|
|
75
|
+
log.info(f"Downloading resources catalog from {url}...")
|
|
75
76
|
with NamedTemporaryFile(dir=config.TEMPORARY_DOWNLOAD_FOLDER or None, delete=False) as fd:
|
|
76
77
|
await download_file(url, fd)
|
|
77
|
-
log.info("Upserting catalog in database...")
|
|
78
|
+
log.info("Upserting resources catalog in database...")
|
|
78
79
|
# consider everything deleted, deleted will be updated when loading new catalog
|
|
79
80
|
conn = await connection()
|
|
80
81
|
await conn.execute("UPDATE catalog SET deleted = TRUE")
|
|
@@ -106,7 +107,7 @@ async def load_catalog(url=None, drop_meta=False, drop_all=False, quiet=False):
|
|
|
106
107
|
if row["harvest.modified_at"]
|
|
107
108
|
else None,
|
|
108
109
|
)
|
|
109
|
-
log.info("
|
|
110
|
+
log.info("Resources catalog successfully upserted into DB.")
|
|
110
111
|
except Exception as e:
|
|
111
112
|
raise e
|
|
112
113
|
finally:
|
|
@@ -132,14 +133,12 @@ async def check_url(url: str, method: str = "get"):
|
|
|
132
133
|
@cli
|
|
133
134
|
async def check_resource(resource_id, method: str = "get"):
|
|
134
135
|
"""Trigger a complete check for a given resource_id"""
|
|
135
|
-
|
|
136
|
-
conn = await connection()
|
|
137
|
-
res = await conn.fetch(q, resource_id)
|
|
136
|
+
res = await Resource.get(resource_id)
|
|
138
137
|
if not res:
|
|
139
138
|
log.error("Resource not found in catalog")
|
|
140
139
|
return
|
|
141
140
|
async with aiohttp.ClientSession(timeout=None) as session:
|
|
142
|
-
await crawl_check_url(res[0], session, method=method)
|
|
141
|
+
await crawl_check_url(url=res[0], resource_id=None, session=session, method=method)
|
|
143
142
|
|
|
144
143
|
|
|
145
144
|
@cli(name="analyse-csv")
|
|
@@ -158,7 +157,7 @@ async def csv_sample(size=1000, download: bool = False, max_size: str = "100M"):
|
|
|
158
157
|
:download: Download files or just list them
|
|
159
158
|
:max_size: Maximum size for one file (from headers)
|
|
160
159
|
"""
|
|
161
|
-
max_size = parse_size(max_size)
|
|
160
|
+
max_size: int = parse_size(max_size)
|
|
162
161
|
start_q = f"""
|
|
163
162
|
SELECT catalog.resource_id, catalog.dataset_id, checks.url,
|
|
164
163
|
checks.headers->>'content-type' as content_type,
|
|
@@ -11,12 +11,13 @@ from humanfriendly import parse_timespan
|
|
|
11
11
|
|
|
12
12
|
from udata_hydra import config, context
|
|
13
13
|
from udata_hydra.analysis.resource import process_resource
|
|
14
|
+
from udata_hydra.db.check import Check
|
|
15
|
+
from udata_hydra.db.resource import Resource
|
|
14
16
|
from udata_hydra.logger import setup_logging
|
|
15
17
|
from udata_hydra.utils import queue
|
|
16
|
-
from udata_hydra.utils.db import insert_check
|
|
17
18
|
from udata_hydra.utils.http import send
|
|
18
19
|
|
|
19
|
-
results = defaultdict(int)
|
|
20
|
+
results: defaultdict = defaultdict(int)
|
|
20
21
|
|
|
21
22
|
STATUS_OK = "ok"
|
|
22
23
|
STATUS_TIMEOUT = "timeout"
|
|
@@ -51,7 +52,7 @@ async def get_content_type_from_header(headers: dict) -> str:
|
|
|
51
52
|
return content_type
|
|
52
53
|
|
|
53
54
|
|
|
54
|
-
async def compute_check_has_changed(check_data, last_check) -> bool:
|
|
55
|
+
async def compute_check_has_changed(check_data: dict, last_check: dict) -> bool:
|
|
55
56
|
is_first_check = not last_check
|
|
56
57
|
status_has_changed = last_check and check_data.get("status") != last_check.get("status")
|
|
57
58
|
status_no_longer_available = (
|
|
@@ -95,13 +96,10 @@ async def compute_check_has_changed(check_data, last_check) -> bool:
|
|
|
95
96
|
)
|
|
96
97
|
or None,
|
|
97
98
|
}
|
|
98
|
-
|
|
99
|
-
async with pool.acquire() as conn:
|
|
100
|
-
q = "SELECT dataset_id FROM CATALOG where resource_id = $1"
|
|
101
|
-
dataset = await conn.fetchrow(q, check_data["resource_id"])
|
|
99
|
+
res = await Resource.get(resource_id=check_data["resource_id"], column_name="dataset_id")
|
|
102
100
|
queue.enqueue(
|
|
103
101
|
send,
|
|
104
|
-
dataset_id=
|
|
102
|
+
dataset_id=res["dataset_id"],
|
|
105
103
|
resource_id=check_data["resource_id"],
|
|
106
104
|
document=document,
|
|
107
105
|
_priority="high",
|
|
@@ -110,15 +108,6 @@ async def compute_check_has_changed(check_data, last_check) -> bool:
|
|
|
110
108
|
return has_changed
|
|
111
109
|
|
|
112
110
|
|
|
113
|
-
async def update_catalog_following_check(resource_id: int):
|
|
114
|
-
pool = await context.pool()
|
|
115
|
-
async with pool.acquire() as connection:
|
|
116
|
-
await connection.execute(
|
|
117
|
-
"UPDATE catalog SET priority = FALSE, status = NULL WHERE resource_id = $1",
|
|
118
|
-
resource_id,
|
|
119
|
-
)
|
|
120
|
-
|
|
121
|
-
|
|
122
111
|
async def process_check_data(check_data: dict) -> Tuple[int, bool]:
|
|
123
112
|
"""Preprocess a check before saving it"""
|
|
124
113
|
check_data["resource_id"] = str(check_data["resource_id"])
|
|
@@ -135,13 +124,15 @@ async def process_check_data(check_data: dict) -> Tuple[int, bool]:
|
|
|
135
124
|
|
|
136
125
|
await compute_check_has_changed(check_data, dict(last_check) if last_check else None)
|
|
137
126
|
|
|
138
|
-
await
|
|
127
|
+
await Resource.update(
|
|
128
|
+
resource_id=check_data["resource_id"], data={"status": None, "priority": False}
|
|
129
|
+
)
|
|
139
130
|
|
|
140
131
|
is_first_check = last_check is None
|
|
141
|
-
return await
|
|
132
|
+
return await Check.insert(check_data), is_first_check
|
|
142
133
|
|
|
143
134
|
|
|
144
|
-
async def is_backoff(domain) -> Tuple[bool, str]:
|
|
135
|
+
async def is_backoff(domain: str) -> Tuple[bool, str]:
|
|
145
136
|
backoff = False, ""
|
|
146
137
|
no_backoff = [f"'{d}'" for d in config.NO_BACKOFF_DOMAINS]
|
|
147
138
|
no_backoff = f"({','.join(no_backoff)})"
|
|
@@ -232,7 +223,7 @@ def convert_headers(headers):
|
|
|
232
223
|
return _headers
|
|
233
224
|
|
|
234
225
|
|
|
235
|
-
def has_nice_head(resp):
|
|
226
|
+
def has_nice_head(resp) -> bool:
|
|
236
227
|
"""Check if a HEAD response looks useful to us"""
|
|
237
228
|
if not is_valid_status(resp.status):
|
|
238
229
|
return False
|
|
@@ -241,20 +232,22 @@ def has_nice_head(resp):
|
|
|
241
232
|
return True
|
|
242
233
|
|
|
243
234
|
|
|
244
|
-
async def check_url(
|
|
245
|
-
|
|
235
|
+
async def check_url(
|
|
236
|
+
url: str, resource_id: str, session, sleep: float = 0, method: str = "head"
|
|
237
|
+
) -> str:
|
|
238
|
+
log.debug(f"check {url}, sleep {sleep}, method {method}")
|
|
246
239
|
|
|
247
240
|
if sleep:
|
|
248
241
|
await asyncio.sleep(sleep)
|
|
249
242
|
|
|
250
|
-
url_parsed = urlparse(
|
|
243
|
+
url_parsed = urlparse(url)
|
|
251
244
|
domain = url_parsed.netloc
|
|
252
245
|
if not domain:
|
|
253
|
-
log.warning(f"[warning] not netloc in url, skipping {
|
|
246
|
+
log.warning(f"[warning] not netloc in url, skipping {url}")
|
|
254
247
|
await process_check_data(
|
|
255
248
|
{
|
|
256
|
-
"resource_id":
|
|
257
|
-
"url":
|
|
249
|
+
"resource_id": resource_id,
|
|
250
|
+
"url": url,
|
|
258
251
|
"error": "Not netloc in url",
|
|
259
252
|
"timeout": False,
|
|
260
253
|
}
|
|
@@ -265,23 +258,23 @@ async def check_url(row, session, sleep=0, method="head"):
|
|
|
265
258
|
if should_backoff:
|
|
266
259
|
log.info(f"backoff {domain} ({reason})")
|
|
267
260
|
# skip this URL, it will come back in a next batch
|
|
268
|
-
await
|
|
261
|
+
await Resource.update(resource_id=resource_id, data={"status": None, "priority": False})
|
|
269
262
|
return STATUS_BACKOFF
|
|
270
263
|
|
|
271
264
|
try:
|
|
272
265
|
start = time.time()
|
|
273
266
|
timeout = aiohttp.ClientTimeout(total=5)
|
|
274
267
|
_method = getattr(session, method)
|
|
275
|
-
async with _method(
|
|
268
|
+
async with _method(url, timeout=timeout, allow_redirects=True) as resp:
|
|
276
269
|
end = time.time()
|
|
277
270
|
if method != "get" and not has_nice_head(resp):
|
|
278
|
-
return await check_url(
|
|
271
|
+
return await check_url(url, resource_id, session, method="get")
|
|
279
272
|
resp.raise_for_status()
|
|
280
273
|
|
|
281
274
|
check_id, is_first_check = await process_check_data(
|
|
282
275
|
{
|
|
283
|
-
"resource_id":
|
|
284
|
-
"url":
|
|
276
|
+
"resource_id": resource_id,
|
|
277
|
+
"url": url,
|
|
285
278
|
"domain": domain,
|
|
286
279
|
"status": resp.status,
|
|
287
280
|
"headers": convert_headers(resp.headers),
|
|
@@ -296,8 +289,8 @@ async def check_url(row, session, sleep=0, method="head"):
|
|
|
296
289
|
except asyncio.exceptions.TimeoutError:
|
|
297
290
|
await process_check_data(
|
|
298
291
|
{
|
|
299
|
-
"resource_id":
|
|
300
|
-
"url":
|
|
292
|
+
"resource_id": resource_id,
|
|
293
|
+
"url": url,
|
|
301
294
|
"domain": domain,
|
|
302
295
|
"timeout": True,
|
|
303
296
|
}
|
|
@@ -316,8 +309,8 @@ async def check_url(row, session, sleep=0, method="head"):
|
|
|
316
309
|
error = getattr(e, "message", None) or str(e)
|
|
317
310
|
await process_check_data(
|
|
318
311
|
{
|
|
319
|
-
"resource_id":
|
|
320
|
-
"url":
|
|
312
|
+
"resource_id": resource_id,
|
|
313
|
+
"url": url,
|
|
321
314
|
"domain": domain,
|
|
322
315
|
"timeout": False,
|
|
323
316
|
"error": fix_surrogates(error),
|
|
@@ -325,25 +318,25 @@ async def check_url(row, session, sleep=0, method="head"):
|
|
|
325
318
|
"status": getattr(e, "status", None),
|
|
326
319
|
}
|
|
327
320
|
)
|
|
328
|
-
log.warning(f"Crawling error for url {
|
|
321
|
+
log.warning(f"Crawling error for url {url}", exc_info=e)
|
|
329
322
|
return STATUS_ERROR
|
|
330
323
|
|
|
331
324
|
|
|
332
|
-
async def crawl_urls(to_parse):
|
|
325
|
+
async def crawl_urls(to_parse: list[str]) -> None:
|
|
333
326
|
context.monitor().set_status("Crawling urls...")
|
|
334
|
-
tasks = []
|
|
327
|
+
tasks: list = []
|
|
335
328
|
async with aiohttp.ClientSession(
|
|
336
329
|
timeout=None, headers={"user-agent": config.USER_AGENT}
|
|
337
330
|
) as session:
|
|
338
331
|
for row in to_parse:
|
|
339
|
-
tasks.append(check_url(row, session))
|
|
332
|
+
tasks.append(check_url(url=row["url"], resource_id=row["resource_id"], session=session))
|
|
340
333
|
for task in asyncio.as_completed(tasks):
|
|
341
334
|
result = await task
|
|
342
335
|
results[result] += 1
|
|
343
336
|
context.monitor().refresh(results)
|
|
344
337
|
|
|
345
338
|
|
|
346
|
-
def get_excluded_clause():
|
|
339
|
+
def get_excluded_clause() -> str:
|
|
347
340
|
return " AND ".join(
|
|
348
341
|
[f"catalog.url NOT LIKE '{p}'" for p in config.EXCLUDED_PATTERNS]
|
|
349
342
|
+ [
|
|
@@ -379,7 +372,7 @@ async def select_rows_based_on_query(connection, q, *args):
|
|
|
379
372
|
return to_check
|
|
380
373
|
|
|
381
374
|
|
|
382
|
-
async def crawl_batch():
|
|
375
|
+
async def crawl_batch() -> None:
|
|
383
376
|
"""Crawl a batch from the catalog"""
|
|
384
377
|
context.monitor().set_status("Getting a batch from catalog...")
|
|
385
378
|
pool = await context.pool()
|
|
@@ -435,7 +428,7 @@ async def crawl_batch():
|
|
|
435
428
|
await asyncio.sleep(config.SLEEP_BETWEEN_BATCHES)
|
|
436
429
|
|
|
437
430
|
|
|
438
|
-
async def crawl(iterations
|
|
431
|
+
async def crawl(iterations: int = -1) -> None:
|
|
439
432
|
"""Launch crawl batches
|
|
440
433
|
|
|
441
434
|
:iterations: for testing purposes (break infinite loop)
|
|
@@ -455,7 +448,7 @@ async def crawl(iterations=-1):
|
|
|
455
448
|
await pool.close()
|
|
456
449
|
|
|
457
450
|
|
|
458
|
-
def run():
|
|
451
|
+
def run() -> None:
|
|
459
452
|
"""Main function
|
|
460
453
|
|
|
461
454
|
:iterations: for testing purposes (break infinite loop)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from udata_hydra import context
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def convert_dict_values_to_json(data: dict) -> dict:
|
|
7
|
+
"""
|
|
8
|
+
Convert values in dict that are dict to json for DB serialization
|
|
9
|
+
TODO: this is suboptimal from asyncpg, dig into this
|
|
10
|
+
https://magicstack.github.io/asyncpg/current/usage.html#example-automatic-json-conversion
|
|
11
|
+
"""
|
|
12
|
+
for k, v in data.items():
|
|
13
|
+
if type(v) is dict:
|
|
14
|
+
data[k] = json.dumps(v)
|
|
15
|
+
return data
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def compute_insert_query(table_name: str, data: dict, returning: str = "id") -> str:
|
|
19
|
+
columns = ",".join([f'"{k}"' for k in data.keys()])
|
|
20
|
+
# $1, $2...
|
|
21
|
+
placeholders = ",".join([f"${x + 1}" for x in range(len(data.values()))])
|
|
22
|
+
return f"""
|
|
23
|
+
INSERT INTO "{table_name}" ({columns})
|
|
24
|
+
VALUES ({placeholders})
|
|
25
|
+
RETURNING {returning}
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def compute_update_query(table_name: str, data: dict) -> str:
|
|
30
|
+
columns = data.keys()
|
|
31
|
+
# $1, $2...
|
|
32
|
+
placeholders = [f"${x + 1}" for x in range(len(data.values()))]
|
|
33
|
+
set_clause = ",".join([f"{c} = {v}" for c, v in zip(columns, placeholders)])
|
|
34
|
+
return f"""
|
|
35
|
+
UPDATE "{table_name}"
|
|
36
|
+
SET {set_clause}
|
|
37
|
+
WHERE id = ${len(placeholders) + 1}
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
async def update_table_record(table_name: str, record_id: int, data: dict) -> int:
|
|
42
|
+
data = convert_dict_values_to_json(data)
|
|
43
|
+
q = compute_update_query(table_name, data)
|
|
44
|
+
pool = await context.pool()
|
|
45
|
+
await pool.execute(q, *data.values(), record_id)
|
|
46
|
+
return record_id
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
from typing import Union
|
|
2
|
+
|
|
3
|
+
from udata_hydra import context
|
|
4
|
+
from udata_hydra.db import (
|
|
5
|
+
compute_insert_query,
|
|
6
|
+
convert_dict_values_to_json,
|
|
7
|
+
update_table_record,
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class Check:
|
|
12
|
+
"""Represents a check in the "checks" DB table"""
|
|
13
|
+
|
|
14
|
+
@classmethod
|
|
15
|
+
async def get(cls, check_id: int) -> dict:
|
|
16
|
+
pool = await context.pool()
|
|
17
|
+
async with pool.acquire() as connection:
|
|
18
|
+
q = """
|
|
19
|
+
SELECT * FROM catalog JOIN checks
|
|
20
|
+
ON catalog.last_check = checks.id
|
|
21
|
+
WHERE checks.id = $1;
|
|
22
|
+
"""
|
|
23
|
+
check = await connection.fetchrow(q, check_id)
|
|
24
|
+
return check
|
|
25
|
+
|
|
26
|
+
@classmethod
|
|
27
|
+
async def get_latest(cls, url: Union[str, None], resource_id: Union[str, None]) -> dict | None:
|
|
28
|
+
column: str = "url" if url else "resource_id"
|
|
29
|
+
pool = await context.pool()
|
|
30
|
+
async with pool.acquire() as connection:
|
|
31
|
+
q = f"""
|
|
32
|
+
SELECT catalog.id as catalog_id, checks.id as check_id,
|
|
33
|
+
catalog.status as catalog_status, checks.status as check_status, *
|
|
34
|
+
FROM checks, catalog
|
|
35
|
+
WHERE checks.id = catalog.last_check
|
|
36
|
+
AND catalog.{column} = $1
|
|
37
|
+
"""
|
|
38
|
+
return await connection.fetchrow(q, url or resource_id)
|
|
39
|
+
|
|
40
|
+
@classmethod
|
|
41
|
+
async def get_all(cls, url: Union[str, None], resource_id: Union[str, None]) -> dict | None:
|
|
42
|
+
column: str = "url" if url else "resource_id"
|
|
43
|
+
pool = await context.pool()
|
|
44
|
+
async with pool.acquire() as connection:
|
|
45
|
+
q = f"""
|
|
46
|
+
SELECT catalog.id as catalog_id, checks.id as check_id,
|
|
47
|
+
catalog.status as catalog_status, checks.status as check_status, *
|
|
48
|
+
FROM checks, catalog
|
|
49
|
+
WHERE catalog.{column} = $1
|
|
50
|
+
AND catalog.url = checks.url
|
|
51
|
+
ORDER BY created_at DESC
|
|
52
|
+
"""
|
|
53
|
+
return await connection.fetch(q, url or resource_id)
|
|
54
|
+
|
|
55
|
+
@classmethod
|
|
56
|
+
async def insert(cls, data: dict) -> int:
|
|
57
|
+
"""
|
|
58
|
+
Insert a new check in DB and return the check id in DB
|
|
59
|
+
This use the info from the last check of the same resource
|
|
60
|
+
"""
|
|
61
|
+
data = convert_dict_values_to_json(data)
|
|
62
|
+
q = compute_insert_query(table_name="checks", data=data)
|
|
63
|
+
pool = await context.pool()
|
|
64
|
+
async with pool.acquire() as connection:
|
|
65
|
+
last_check = await connection.fetchrow(q, *data.values())
|
|
66
|
+
q = """UPDATE catalog SET last_check = $1 WHERE resource_id = $2"""
|
|
67
|
+
await connection.execute(q, last_check["id"], data["resource_id"])
|
|
68
|
+
return last_check["id"]
|
|
69
|
+
|
|
70
|
+
@classmethod
|
|
71
|
+
async def update(cls, check_id: int, data: dict) -> int:
|
|
72
|
+
"""Update a check in DB with new data and return the check id in DB"""
|
|
73
|
+
return await update_table_record(table_name="checks", record_id=check_id, data=data)
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
from udata_hydra import context
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class Resource:
|
|
5
|
+
"""Represents a resource in the "catalog" DB table"""
|
|
6
|
+
|
|
7
|
+
@classmethod
|
|
8
|
+
async def get(cls, resource_id: str, column_name: str = "*") -> dict:
|
|
9
|
+
pool = await context.pool()
|
|
10
|
+
async with pool.acquire() as connection:
|
|
11
|
+
q = f"""SELECT {column_name} FROM catalog WHERE resource_id = '{resource_id}';"""
|
|
12
|
+
resource = await connection.fetchrow(q)
|
|
13
|
+
return resource
|
|
14
|
+
|
|
15
|
+
@classmethod
|
|
16
|
+
async def insert(
|
|
17
|
+
cls,
|
|
18
|
+
dataset_id: str,
|
|
19
|
+
resource_id: str,
|
|
20
|
+
url: str,
|
|
21
|
+
priority: bool = True,
|
|
22
|
+
) -> None:
|
|
23
|
+
pool = await context.pool()
|
|
24
|
+
async with pool.acquire() as connection:
|
|
25
|
+
# Insert new resource in catalog table and mark as high priority for crawling
|
|
26
|
+
q = f"""
|
|
27
|
+
INSERT INTO catalog (dataset_id, resource_id, url, deleted, status, priority)
|
|
28
|
+
VALUES ('{dataset_id}', '{resource_id}', '{url}', FALSE, 'TO_CHECK', '{priority}')
|
|
29
|
+
ON CONFLICT (resource_id) DO UPDATE SET
|
|
30
|
+
dataset_id = '{dataset_id}',
|
|
31
|
+
url = '{url}',
|
|
32
|
+
priority = '{priority}';"""
|
|
33
|
+
await connection.execute(q)
|
|
34
|
+
|
|
35
|
+
@classmethod
|
|
36
|
+
async def update(cls, resource_id: str, data: dict) -> str:
|
|
37
|
+
"""Update a resource in DB with new data and return the updated resource id in DB"""
|
|
38
|
+
columns = data.keys()
|
|
39
|
+
# $1, $2...
|
|
40
|
+
placeholders = [f"${x + 1}" for x in range(len(data.values()))]
|
|
41
|
+
set_clause = ",".join([f"{c} = {v}" for c, v in zip(columns, placeholders)])
|
|
42
|
+
q = f"""
|
|
43
|
+
UPDATE catalog
|
|
44
|
+
SET {set_clause}
|
|
45
|
+
WHERE resource_id = ${len(placeholders) + 1};"""
|
|
46
|
+
pool = await context.pool()
|
|
47
|
+
await pool.execute(q, *data.values(), resource_id)
|
|
48
|
+
return resource_id
|
|
49
|
+
|
|
50
|
+
@classmethod
|
|
51
|
+
async def update_or_insert(
|
|
52
|
+
cls,
|
|
53
|
+
dataset_id: str,
|
|
54
|
+
resource_id: str,
|
|
55
|
+
url: str,
|
|
56
|
+
priority: bool = True, # Make resource high priority by default for crawling
|
|
57
|
+
) -> None:
|
|
58
|
+
pool = await context.pool()
|
|
59
|
+
async with pool.acquire() as connection:
|
|
60
|
+
# Check if resource is in catalog then insert or update into table
|
|
61
|
+
if await Resource.get(resource_id):
|
|
62
|
+
q = f"""
|
|
63
|
+
UPDATE catalog
|
|
64
|
+
SET url = '{url}', priority = '{priority}'
|
|
65
|
+
WHERE resource_id = '{resource_id}';"""
|
|
66
|
+
else:
|
|
67
|
+
q = f"""
|
|
68
|
+
INSERT INTO catalog (dataset_id, resource_id, url, deleted, priority)
|
|
69
|
+
VALUES ('{dataset_id}', '{resource_id}', '{url}', FALSE, '{priority}')
|
|
70
|
+
ON CONFLICT (resource_id) DO UPDATE SET
|
|
71
|
+
dataset_id = '{dataset_id}',
|
|
72
|
+
url = '{url}',
|
|
73
|
+
priority = '{priority}';"""
|
|
74
|
+
await connection.execute(q)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from aiohttp import web
|
|
4
|
+
from marshmallow import Schema, fields
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class CheckSchema(Schema):
|
|
8
|
+
check_id = fields.Integer(data_key="id")
|
|
9
|
+
catalog_id = fields.Integer()
|
|
10
|
+
url = fields.Str()
|
|
11
|
+
domain = fields.Str()
|
|
12
|
+
created_at = fields.DateTime()
|
|
13
|
+
check_status = fields.Integer(data_key="status")
|
|
14
|
+
headers = fields.Function(lambda obj: json.loads(obj["headers"]) if obj["headers"] else {})
|
|
15
|
+
timeout = fields.Boolean()
|
|
16
|
+
response_time = fields.Float()
|
|
17
|
+
error = fields.Str()
|
|
18
|
+
dataset_id = fields.Str()
|
|
19
|
+
resource_id = fields.UUID()
|
|
20
|
+
deleted = fields.Boolean()
|
|
21
|
+
parsing_started_at = fields.DateTime()
|
|
22
|
+
parsing_finished_at = fields.DateTime()
|
|
23
|
+
parsing_error = fields.Str()
|
|
24
|
+
parsing_table = fields.Str()
|
|
25
|
+
|
|
26
|
+
def create(self, data):
|
|
27
|
+
return self.load(data)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from marshmallow import Schema, fields
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class ResourceDocumentSchema(Schema):
|
|
5
|
+
id = fields.Str(required=True)
|
|
6
|
+
url = fields.Str(required=True)
|
|
7
|
+
format = fields.Str(allow_none=True)
|
|
8
|
+
title = fields.Str(required=True)
|
|
9
|
+
schema = fields.Dict(allow_none=True)
|
|
10
|
+
description = fields.Str(allow_none=True)
|
|
11
|
+
filetype = fields.Str(required=True)
|
|
12
|
+
type = fields.Str(required=True)
|
|
13
|
+
mime = fields.Str(allow_none=True)
|
|
14
|
+
filesize = fields.Int(allow_none=True)
|
|
15
|
+
checksum_type = fields.Str(allow_none=True)
|
|
16
|
+
checksum_value = fields.Str(allow_none=True)
|
|
17
|
+
created_at = fields.DateTime(required=True)
|
|
18
|
+
last_modified = fields.DateTime(required=True)
|
|
19
|
+
extras = fields.Dict()
|
|
20
|
+
harvest = fields.Dict()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ResourceQuerySchema(Schema):
|
|
24
|
+
dataset_id = fields.Str(required=True)
|
|
25
|
+
resource_id = fields.Str(required=True)
|
|
26
|
+
document = fields.Nested(ResourceDocumentSchema(), allow_none=True)
|
|
@@ -1,73 +0,0 @@
|
|
|
1
|
-
import json
|
|
2
|
-
|
|
3
|
-
from udata_hydra import context
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
def convert_dict_values_to_json(data: dict) -> dict:
|
|
7
|
-
"""
|
|
8
|
-
Convert values in dict that are dict to json for DB serialization
|
|
9
|
-
TODO: this is suboptimal from asyncpg, dig into this
|
|
10
|
-
https://magicstack.github.io/asyncpg/current/usage.html#example-automatic-json-conversion
|
|
11
|
-
"""
|
|
12
|
-
for k, v in data.items():
|
|
13
|
-
if type(v) is dict:
|
|
14
|
-
data[k] = json.dumps(v)
|
|
15
|
-
return data
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def compute_insert_query(data: dict, table: str, returning: str = "id") -> str:
|
|
19
|
-
columns = ",".join([f'"{k}"' for k in data.keys()])
|
|
20
|
-
# $1, $2...
|
|
21
|
-
placeholders = ",".join([f"${x + 1}" for x in range(len(data.values()))])
|
|
22
|
-
return f"""
|
|
23
|
-
INSERT INTO "{table}" ({columns})
|
|
24
|
-
VALUES ({placeholders})
|
|
25
|
-
RETURNING {returning}
|
|
26
|
-
"""
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
async def insert_check(data: dict) -> int:
|
|
30
|
-
data = convert_dict_values_to_json(data)
|
|
31
|
-
q = compute_insert_query(data, "checks")
|
|
32
|
-
pool = await context.pool()
|
|
33
|
-
async with pool.acquire() as connection:
|
|
34
|
-
last_check = await connection.fetchrow(q, *data.values())
|
|
35
|
-
q = """UPDATE catalog SET last_check = $1 WHERE resource_id = $2"""
|
|
36
|
-
await connection.execute(q, last_check["id"], data["resource_id"])
|
|
37
|
-
return last_check["id"]
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
def compute_update_query(table: str, data: dict) -> str:
|
|
41
|
-
columns = data.keys()
|
|
42
|
-
# $1, $2...
|
|
43
|
-
placeholders = [f"${x + 1}" for x in range(len(data.values()))]
|
|
44
|
-
set_clause = ",".join([f"{c} = {v}" for c, v in zip(columns, placeholders)])
|
|
45
|
-
return f"""
|
|
46
|
-
UPDATE "{table}"
|
|
47
|
-
SET {set_clause}
|
|
48
|
-
WHERE id = ${len(placeholders) + 1}
|
|
49
|
-
"""
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
async def update_table_record(table: str, record_id: int, data: dict) -> int:
|
|
53
|
-
data = convert_dict_values_to_json(data)
|
|
54
|
-
q = compute_update_query(table, data)
|
|
55
|
-
pool = await context.pool()
|
|
56
|
-
await pool.execute(q, *data.values(), record_id)
|
|
57
|
-
return record_id
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
async def update_check(check_id: int, data: dict) -> int:
|
|
61
|
-
return await update_table_record("checks", check_id, data)
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
async def get_check(check_id: int) -> dict:
|
|
65
|
-
pool = await context.pool()
|
|
66
|
-
async with pool.acquire() as connection:
|
|
67
|
-
q = """
|
|
68
|
-
SELECT * FROM catalog JOIN checks
|
|
69
|
-
ON catalog.last_check = checks.id
|
|
70
|
-
WHERE checks.id = $1;
|
|
71
|
-
"""
|
|
72
|
-
check = await connection.fetchrow(q, check_id)
|
|
73
|
-
return check
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|