udata-hydra 2.0.0.dev2235__tar.gz → 2.0.0.dev2384__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/PKG-INFO +36 -6
  2. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/README.md +35 -5
  3. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/pyproject.toml +11 -9
  4. udata_hydra-2.0.0.dev2384/udata_hydra/app.py +29 -0
  5. udata_hydra-2.0.0.dev2384/udata_hydra/routes/__init__.py +23 -0
  6. udata_hydra-2.0.0.dev2384/udata_hydra/routes/checks.py +31 -0
  7. udata_hydra-2.0.0.dev2384/udata_hydra/routes/resources.py +83 -0
  8. udata_hydra-2.0.0.dev2384/udata_hydra/routes/status.py +125 -0
  9. udata_hydra-2.0.0.dev2235/udata_hydra/app.py +0 -269
  10. udata_hydra-2.0.0.dev2235/udata_hydra/utils/json.py +0 -11
  11. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/__init__.py +0 -0
  12. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/analysis/__init__.py +0 -0
  13. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/analysis/csv.py +0 -0
  14. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/analysis/errors.py +0 -0
  15. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/analysis/helpers.py +0 -0
  16. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/analysis/resource.py +0 -0
  17. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/cli.py +0 -0
  18. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/config_default.toml +0 -0
  19. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/context.py +0 -0
  20. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/crawl.py +0 -0
  21. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/db/__init__.py +0 -0
  22. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/db/check.py +0 -0
  23. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/db/resource.py +0 -0
  24. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/logger.py +0 -0
  25. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/__init__.py +0 -0
  26. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/csv/20221205_initial_up_rev1.sql +0 -0
  27. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/csv/20230130_drop_migrations.sql +0 -0
  28. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/csv/20230206_datetime_aware.sql +0 -0
  29. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20221205_initial_up_rev1.sql +0 -0
  30. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20221206_rev1_up_rev2.sql +0 -0
  31. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20221206_rev2_up_rev3.sql +0 -0
  32. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20221208_rev3_up_rev4.sql +0 -0
  33. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20221208_rev4_up_rev5.sql +0 -0
  34. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230119_rev5_up_rev6.sql +0 -0
  35. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230121_rev6_up_rev7.sql +0 -0
  36. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230121_rev7_up_rev8.sql +0 -0
  37. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230130_drop_migrations.sql +0 -0
  38. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230206_datetime_aware.sql +0 -0
  39. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230515_rev8_up_rev9.sql +0 -0
  40. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20230606_rev9_up_rev10.sql +0 -0
  41. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/migrations/main/20231102_drop_csv_analysis.sql +0 -0
  42. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/schemas/__init__.py +0 -0
  43. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/schemas/check.py +0 -0
  44. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/schemas/resource_query.py +0 -0
  45. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/__init__.py +0 -0
  46. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/app_version.py +0 -0
  47. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/csv.py +0 -0
  48. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/file.py +0 -0
  49. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/http.py +0 -0
  50. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/minio.py +0 -0
  51. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/queue.py +0 -0
  52. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/reader.py +0 -0
  53. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/utils/timer.py +0 -0
  54. {udata_hydra-2.0.0.dev2235 → udata_hydra-2.0.0.dev2384}/udata_hydra/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: udata-hydra
3
- Version: 2.0.0.dev2235
3
+ Version: 2.0.0.dev2384
4
4
  Summary: Async crawler and parsing service for data.gouv.fr
5
5
  License: MIT
6
6
  Author: Opendata Team
@@ -105,6 +105,11 @@ Then you can run the tests with `poetry run pytest`.
105
105
 
106
106
  If you would like to see print statements as they are executed, you can pass the -s flag to pytest (`poetry run pytest -s`). However, note that this can sometimes be difficult to parse.
107
107
 
108
+ ### Tests coverage
109
+
110
+ Pytest automatically uses the `coverage` package to generate a coverage report, which is displayed at the end of the test run in the terminal.
111
+ The coverage is configured in the `pypoject.toml` file, in the `[tool.pytest.ini_options]` section.
112
+ You can also override the coverage report configuration when running the tests by passing some flags like `--cov-report` to pytest. See [the pytest-cov documentation](https://pytest-cov.readthedocs.io/en/latest/config.html) for more information.
108
113
 
109
114
  ## API
110
115
 
@@ -115,7 +120,32 @@ poetry install
115
120
  poetry run adev runserver udata_hydra/app.py
116
121
  ```
117
122
 
118
- ### Get latest check
123
+ ### Routes/endpoints
124
+
125
+ The API serves the following endpoints:
126
+
127
+ *Related to checks:*
128
+ - `GET` on `/api/checks/latest/` to get the latest check for a given URL or resource
129
+ - `GET` on `/api/checks/all/` to get all checks for a given URL or resource
130
+
131
+ *Related to resources:*
132
+ - `POST` on `/api/resources/` to receive a resource creation event from a source. It will create a new resource in the DB "catalog" table and mark it as priority for next crawling
133
+ - `PUT` on `/api/resources/` to update a resource in the DB "catalog" table
134
+ - `DELETE` on `/api/resources/` to delete a resource in the DB "catalog" table
135
+
136
+ > :warning: **Warning: the following routes are deprecated and need be removed in the future:**
137
+ > - `POST` on `/api/resource/created` -> use `POST` on `/api/resources/` instead
138
+ > - `POST` on `/api/resource/updated` -> use `PUT` on `/api/resources/` instead
139
+ > - `POST` on `/api/resource/deleted` -> use `DELET`E on `/api/resources/` instead
140
+
141
+ *Related to some status and health check:*
142
+ - `GET` on `/api/status/crawler/` to get the crawling status
143
+ - `GET` on `/api/status/worker/` to get the worker status
144
+ - `GET` on `/api/stats/` to get the crawling stats
145
+
146
+ More details about some enpoints are provided below with examples, but not for all of them:
147
+
148
+ #### Get latest check
119
149
 
120
150
  Works with `?url={url}` and `?resource_id={resource_id}`.
121
151
 
@@ -152,7 +182,7 @@ $ curl -s "http://localhost:8000/api/checks/latest/?url=http://opendata-sig.sain
152
182
  }
153
183
  ```
154
184
 
155
- ### Get all checks for an URL or resource
185
+ #### Get all checks for an URL or resource
156
186
 
157
187
  Works with `?url={url}` and `?resource_id={resource_id}`.
158
188
 
@@ -192,7 +222,7 @@ $ curl -s "http://localhost:8000/api/checks/all/?url=http://www.drees.sante.gouv
192
222
  ]
193
223
  ```
194
224
 
195
- ### Get crawling status
225
+ #### Get crawling status
196
226
 
197
227
  ```bash
198
228
  $ curl -s "http://localhost:8000/api/status/crawler/" | json_pp
@@ -205,7 +235,7 @@ $ curl -s "http://localhost:8000/api/status/crawler/" | json_pp
205
235
  }
206
236
  ```
207
237
 
208
- ### Get worker status
238
+ #### Get worker status
209
239
 
210
240
  ```bash
211
241
  $ curl -s "http://localhost:8000/api/status/worker/" | json_pp
@@ -218,7 +248,7 @@ $ curl -s "http://localhost:8000/api/status/worker/" | json_pp
218
248
  }
219
249
  ```
220
250
 
221
- ### Get crawling stats
251
+ #### Get crawling stats
222
252
 
223
253
  ```bash
224
254
  $ curl -s "http://localhost:8000/api/stats/" | json_pp
@@ -69,6 +69,11 @@ Then you can run the tests with `poetry run pytest`.
69
69
 
70
70
  If you would like to see print statements as they are executed, you can pass the -s flag to pytest (`poetry run pytest -s`). However, note that this can sometimes be difficult to parse.
71
71
 
72
+ ### Tests coverage
73
+
74
+ Pytest automatically uses the `coverage` package to generate a coverage report, which is displayed at the end of the test run in the terminal.
75
+ The coverage is configured in the `pypoject.toml` file, in the `[tool.pytest.ini_options]` section.
76
+ You can also override the coverage report configuration when running the tests by passing some flags like `--cov-report` to pytest. See [the pytest-cov documentation](https://pytest-cov.readthedocs.io/en/latest/config.html) for more information.
72
77
 
73
78
  ## API
74
79
 
@@ -79,7 +84,32 @@ poetry install
79
84
  poetry run adev runserver udata_hydra/app.py
80
85
  ```
81
86
 
82
- ### Get latest check
87
+ ### Routes/endpoints
88
+
89
+ The API serves the following endpoints:
90
+
91
+ *Related to checks:*
92
+ - `GET` on `/api/checks/latest/` to get the latest check for a given URL or resource
93
+ - `GET` on `/api/checks/all/` to get all checks for a given URL or resource
94
+
95
+ *Related to resources:*
96
+ - `POST` on `/api/resources/` to receive a resource creation event from a source. It will create a new resource in the DB "catalog" table and mark it as priority for next crawling
97
+ - `PUT` on `/api/resources/` to update a resource in the DB "catalog" table
98
+ - `DELETE` on `/api/resources/` to delete a resource in the DB "catalog" table
99
+
100
+ > :warning: **Warning: the following routes are deprecated and need be removed in the future:**
101
+ > - `POST` on `/api/resource/created` -> use `POST` on `/api/resources/` instead
102
+ > - `POST` on `/api/resource/updated` -> use `PUT` on `/api/resources/` instead
103
+ > - `POST` on `/api/resource/deleted` -> use `DELET`E on `/api/resources/` instead
104
+
105
+ *Related to some status and health check:*
106
+ - `GET` on `/api/status/crawler/` to get the crawling status
107
+ - `GET` on `/api/status/worker/` to get the worker status
108
+ - `GET` on `/api/stats/` to get the crawling stats
109
+
110
+ More details about some enpoints are provided below with examples, but not for all of them:
111
+
112
+ #### Get latest check
83
113
 
84
114
  Works with `?url={url}` and `?resource_id={resource_id}`.
85
115
 
@@ -116,7 +146,7 @@ $ curl -s "http://localhost:8000/api/checks/latest/?url=http://opendata-sig.sain
116
146
  }
117
147
  ```
118
148
 
119
- ### Get all checks for an URL or resource
149
+ #### Get all checks for an URL or resource
120
150
 
121
151
  Works with `?url={url}` and `?resource_id={resource_id}`.
122
152
 
@@ -156,7 +186,7 @@ $ curl -s "http://localhost:8000/api/checks/all/?url=http://www.drees.sante.gouv
156
186
  ]
157
187
  ```
158
188
 
159
- ### Get crawling status
189
+ #### Get crawling status
160
190
 
161
191
  ```bash
162
192
  $ curl -s "http://localhost:8000/api/status/crawler/" | json_pp
@@ -169,7 +199,7 @@ $ curl -s "http://localhost:8000/api/status/crawler/" | json_pp
169
199
  }
170
200
  ```
171
201
 
172
- ### Get worker status
202
+ #### Get worker status
173
203
 
174
204
  ```bash
175
205
  $ curl -s "http://localhost:8000/api/status/worker/" | json_pp
@@ -182,7 +212,7 @@ $ curl -s "http://localhost:8000/api/status/worker/" | json_pp
182
212
  }
183
213
  ```
184
214
 
185
- ### Get crawling stats
215
+ #### Get crawling stats
186
216
 
187
217
  ```bash
188
218
  $ curl -s "http://localhost:8000/api/stats/" | json_pp
@@ -9,7 +9,7 @@ readme = "README.md"
9
9
 
10
10
  [tool.poetry]
11
11
  name = "udata-hydra"
12
- version = "2.0.0.dev2235"
12
+ version = "2.0.0.dev2384"
13
13
  description = "Async crawler and parsing service for data.gouv.fr"
14
14
  authors = ["Opendata Team <opendatateam@data.gouv.fr>"]
15
15
  license = "MIT"
@@ -47,6 +47,7 @@ mypy = "^1.11.0"
47
47
  nest_asyncio = "^1.5.5"
48
48
  pytest = "^7.1.1"
49
49
  pytest-asyncio = "^0.18.3"
50
+ pytest-cov = "^5.0.0"
50
51
  pytest-mock = "^3.7.0"
51
52
  ruff = "^0.5.2"
52
53
 
@@ -66,21 +67,22 @@ module = [
66
67
  ]
67
68
  ignore_missing_imports = true
68
69
 
70
+ [tool.pytest.ini_options]
71
+ asyncio_mode = "auto"
72
+ markers = [
73
+ "slow: marks tests as slow (deselect with '-m \"not slow\"')",
74
+ "catalog_harvested: use catalog_harvested.csv as source",
75
+ ]
76
+ addopts = "--cov=udata_hydra/ --cov-report term-missing"
77
+
69
78
  [tool.ruff]
70
- lint = { select = ["I"] } # also sort imports with an isort rule
79
+ lint = { extend-select = ["I"] } # also sort imports with an isort rule
71
80
  line-length = 100
72
81
 
73
82
  [build-system]
74
83
  requires = ["poetry-core>=1.0.0"]
75
84
  build-backend = "poetry.core.masonry.api"
76
85
 
77
- [tool.pytest.ini_options]
78
- asyncio_mode = "strict"
79
- markers = [
80
- "slow: marks tests as slow (deselect with '-m \"not slow\"')",
81
- "catalog_harvested: use catalog_harvested.csv as source",
82
- ]
83
-
84
86
  [tool.poetry.scripts]
85
87
  udata-hydra = "udata_hydra.cli:run"
86
88
  udata-hydra-crawl = "udata_hydra.crawl:run"
@@ -0,0 +1,29 @@
1
+ import os
2
+
3
+ from aiohttp import web
4
+
5
+ from udata_hydra import context
6
+ from udata_hydra.routes import routes
7
+
8
+
9
+ async def app_factory() -> web.Application:
10
+ async def app_startup(app):
11
+ app["pool"] = await context.pool()
12
+
13
+ async def app_cleanup(app):
14
+ if "pool" in app:
15
+ await app["pool"].close()
16
+
17
+ app = web.Application()
18
+ app.add_routes(routes)
19
+ app.on_startup.append(app_startup)
20
+ app.on_cleanup.append(app_cleanup)
21
+ return app
22
+
23
+
24
+ def run():
25
+ web.run_app(app_factory(), path=os.environ.get("HYDRA_APP_SOCKET_PATH"))
26
+
27
+
28
+ if __name__ == "__main__":
29
+ run()
@@ -0,0 +1,23 @@
1
+ from aiohttp import web
2
+
3
+ from udata_hydra.routes.checks import get_all_checks, get_latest_check
4
+ from udata_hydra.routes.resources import create_resource, delete_resource, update_resource
5
+ from udata_hydra.routes.status import get_crawler_status, get_stats, get_worker_status
6
+
7
+ # routes = web.RouteTableDef()
8
+ routes: list = [
9
+ # Routes for checks
10
+ web.get("/api/checks/latest/", get_latest_check),
11
+ web.get("/api/checks/all/", get_all_checks),
12
+ # Routes for resources
13
+ web.post("/api/resources/", create_resource),
14
+ web.put("/api/resources/", update_resource),
15
+ web.delete("/api/resources/", delete_resource),
16
+ web.post("/api/resource/created/", create_resource), # TODO: legacy, to remove
17
+ web.post("/api/resource/updated/", update_resource), # TODO: legacy, to remove
18
+ web.post("/api/resource/deleted/", delete_resource), # TODO: legacy, to remove
19
+ # Routes for statuses
20
+ web.get("/api/status/crawler/", get_crawler_status),
21
+ web.get("/api/status/worker/", get_worker_status),
22
+ web.get("/api/stats/", get_stats),
23
+ ]
@@ -0,0 +1,31 @@
1
+ from aiohttp import web
2
+
3
+ from udata_hydra.db.check import Check
4
+ from udata_hydra.schemas import CheckSchema
5
+
6
+
7
+ def _get_args(request, params=("url", "resource_id")) -> list:
8
+ """Get GET parameters from request"""
9
+ data = [request.query.get(param) for param in params]
10
+ if not any(data):
11
+ raise web.HTTPBadRequest()
12
+ return data
13
+
14
+
15
+ async def get_latest_check(request: web.Request) -> web.Response:
16
+ """Get the latest check for a given URL or resource_id"""
17
+ url, resource_id = _get_args(request)
18
+ data = await Check.get_latest(url, resource_id)
19
+ if not data:
20
+ raise web.HTTPNotFound()
21
+ if data["deleted"]:
22
+ raise web.HTTPGone()
23
+ return web.json_response(CheckSchema().dump(dict(data)))
24
+
25
+
26
+ async def get_all_checks(request: web.Request) -> web.Response:
27
+ url, resource_id = _get_args(request)
28
+ data = await Check.get_all(url, resource_id)
29
+ if not data:
30
+ raise web.HTTPNotFound()
31
+ return web.json_response([CheckSchema().dump(dict(r)) for r in data])
@@ -0,0 +1,83 @@
1
+ import json
2
+
3
+ from aiohttp import web
4
+ from marshmallow import ValidationError
5
+
6
+ from udata_hydra import config
7
+ from udata_hydra.db.resource import Resource
8
+ from udata_hydra.schemas import ResourceQuerySchema
9
+ from udata_hydra.utils.minio import delete_resource_from_minio
10
+
11
+
12
+ async def create_resource(request: web.Request) -> web.Response:
13
+ """Endpoint to receive a resource creation event from a source
14
+ Will create a new resource in the DB "catalog" table and mark it as priority for next crawling
15
+ Respond with a 200 status code and a JSON body with a message key set to "created"
16
+ If error, respond with a 400 status code
17
+ """
18
+ try:
19
+ payload = await request.json()
20
+ valid_payload: dict = ResourceQuerySchema().load(payload)
21
+ except ValidationError as err:
22
+ raise web.HTTPBadRequest(text=json.dumps(err.messages))
23
+
24
+ resource = valid_payload["document"]
25
+ if not resource:
26
+ raise web.HTTPBadRequest(text="Missing document body")
27
+
28
+ dataset_id = valid_payload["dataset_id"]
29
+ resource_id = valid_payload["resource_id"]
30
+
31
+ await Resource.insert(
32
+ dataset_id=dataset_id,
33
+ resource_id=resource_id,
34
+ url=resource["url"],
35
+ priority=True,
36
+ )
37
+
38
+ return web.json_response({"message": "created"})
39
+
40
+
41
+ async def update_resource(request: web.Request) -> web.Response:
42
+ """Endpoint to receive a resource update event from a source
43
+ Will update an existing resource in the DB "catalog" table and mark it as priority for next crawling
44
+ Respond with a 200 status code and a JSON body with a message key set to "updated"
45
+ If error, respond with a 400 status code
46
+ """
47
+ try:
48
+ payload = await request.json()
49
+ valid_payload: dict = ResourceQuerySchema().load(payload)
50
+ except ValidationError as err:
51
+ raise web.HTTPBadRequest(text=json.dumps(err.messages))
52
+
53
+ resource = valid_payload["document"]
54
+ if not resource:
55
+ raise web.HTTPBadRequest(text="Missing document body")
56
+
57
+ dataset_id = valid_payload["dataset_id"]
58
+ resource_id = valid_payload["resource_id"]
59
+
60
+ await Resource.update_or_insert(dataset_id, resource_id, resource["url"])
61
+
62
+ return web.json_response({"message": "updated"})
63
+
64
+
65
+ async def delete_resource(request: web.Request) -> web.Response:
66
+ try:
67
+ payload = await request.json()
68
+ valid_payload: dict = ResourceQuerySchema().load(payload)
69
+ except ValidationError as err:
70
+ raise web.HTTPBadRequest(text=json.dumps(err.messages))
71
+
72
+ dataset_id = valid_payload["dataset_id"]
73
+ resource_id = valid_payload["resource_id"]
74
+
75
+ pool = request.app["pool"]
76
+ async with pool.acquire() as connection:
77
+ if config.SAVE_TO_MINIO:
78
+ delete_resource_from_minio(dataset_id, resource_id)
79
+ # Mark resource as deleted in catalog table
80
+ q = f"""UPDATE catalog SET deleted = TRUE WHERE resource_id = '{resource_id}';"""
81
+ await connection.execute(q)
82
+
83
+ return web.json_response({"message": "deleted"})
@@ -0,0 +1,125 @@
1
+ from datetime import datetime, timedelta, timezone
2
+
3
+ from aiohttp import web
4
+ from humanfriendly import parse_timespan
5
+
6
+ from udata_hydra import config, context
7
+ from udata_hydra.crawl import get_excluded_clause
8
+ from udata_hydra.worker import QUEUES
9
+
10
+
11
+ async def get_crawler_status(request: web.Request) -> web.Response:
12
+ q = f"""
13
+ SELECT
14
+ SUM(CASE WHEN last_check IS NULL THEN 1 ELSE 0 END) AS count_left,
15
+ SUM(CASE WHEN last_check IS NOT NULL THEN 1 ELSE 0 END) AS count_checked
16
+ FROM catalog
17
+ WHERE {get_excluded_clause()}
18
+ AND catalog.deleted = False
19
+ """
20
+ stats_catalog = await request.app["pool"].fetchrow(q)
21
+
22
+ since = parse_timespan(config.SINCE)
23
+ since = datetime.now(timezone.utc) - timedelta(seconds=since)
24
+ q = f"""
25
+ SELECT
26
+ SUM(CASE WHEN checks.created_at <= $1 THEN 1 ELSE 0 END) AS count_outdated
27
+ --, SUM(CASE WHEN checks.created_at > $1 THEN 1 ELSE 0 END) AS count_fresh
28
+ FROM catalog, checks
29
+ WHERE {get_excluded_clause()}
30
+ AND catalog.last_check = checks.id
31
+ AND catalog.deleted = False
32
+ """
33
+ stats_checks = await request.app["pool"].fetchrow(q, since)
34
+
35
+ count_left = stats_catalog["count_left"] + (stats_checks["count_outdated"] or 0)
36
+ # all w/ a check, minus those with an outdated checked
37
+ count_checked = stats_catalog["count_checked"] - (stats_checks["count_outdated"] or 0)
38
+ total = stats_catalog["count_left"] + stats_catalog["count_checked"]
39
+ rate_checked = round(stats_catalog["count_checked"] / total * 100, 1)
40
+ rate_checked_fresh = round(count_checked / total * 100, 1)
41
+
42
+ return web.json_response(
43
+ {
44
+ "total": total,
45
+ "pending_checks": count_left,
46
+ "fresh_checks": count_checked,
47
+ "checks_percentage": rate_checked,
48
+ "fresh_checks_percentage": rate_checked_fresh,
49
+ }
50
+ )
51
+
52
+
53
+ async def get_worker_status(request: web.Request) -> web.Response:
54
+ res = {"queued": {q: len(context.queue(q)) for q in QUEUES}}
55
+ return web.json_response(res)
56
+
57
+
58
+ async def get_stats(request: web.Request) -> web.Response:
59
+ q = f"""
60
+ SELECT count(*) AS count_checked
61
+ FROM catalog
62
+ WHERE {get_excluded_clause()}
63
+ AND last_check IS NOT NULL
64
+ AND catalog.deleted = False
65
+ """
66
+ stats_catalog = await request.app["pool"].fetchrow(q)
67
+
68
+ q = f"""
69
+ SELECT
70
+ SUM(CASE WHEN error IS NULL AND timeout = False THEN 1 ELSE 0 END) AS count_ok,
71
+ SUM(CASE WHEN error IS NOT NULL THEN 1 ELSE 0 END) AS count_error,
72
+ SUM(CASE WHEN timeout = True THEN 1 ELSE 0 END) AS count_timeout
73
+ FROM catalog, checks
74
+ WHERE {get_excluded_clause()}
75
+ AND catalog.last_check = checks.id
76
+ AND catalog.deleted = False
77
+ """
78
+ stats_status = await request.app["pool"].fetchrow(q)
79
+
80
+ def cmp_rate(key):
81
+ if stats_catalog["count_checked"] == 0:
82
+ return 0
83
+ return round(stats_status[key] / stats_catalog["count_checked"] * 100, 1)
84
+
85
+ q = f"""
86
+ SELECT checks.status, count(*) as count FROM checks, catalog
87
+ WHERE catalog.last_check = checks.id
88
+ AND checks.status IS NOT NULL
89
+ AND {get_excluded_clause()}
90
+ AND last_check IS NOT NULL
91
+ AND catalog.deleted = False
92
+ GROUP BY checks.status
93
+ ORDER BY count DESC;
94
+ """
95
+ res = await request.app["pool"].fetch(q)
96
+ return web.json_response(
97
+ {
98
+ "status": sorted(
99
+ [
100
+ {
101
+ "label": s,
102
+ "count": stats_status[f"count_{s}"] or 0,
103
+ "percentage": cmp_rate(f"count_{s}"),
104
+ }
105
+ for s in ["error", "timeout", "ok"]
106
+ ],
107
+ key=lambda x: x["count"],
108
+ reverse=True,
109
+ ),
110
+ "status_codes": [
111
+ {
112
+ "code": r["status"],
113
+ "count": r["count"],
114
+ "percentage": round(r["count"] / sum(r["count"] for r in res) * 100, 1),
115
+ }
116
+ for r in res
117
+ ],
118
+ }
119
+ )
120
+
121
+
122
+ async def health(request: web.Request) -> web.Response:
123
+ test_connection = await request.app["pool"].fetchrow("SELECT 1")
124
+ assert next(test_connection.values()) == 1
125
+ return web.HTTPOk()
@@ -1,269 +0,0 @@
1
- import json
2
- import os
3
- from datetime import datetime, timedelta, timezone
4
-
5
- from aiohttp import web
6
- from humanfriendly import parse_timespan
7
- from marshmallow import ValidationError
8
-
9
- from udata_hydra import config, context
10
- from udata_hydra.crawl import get_excluded_clause
11
- from udata_hydra.db.check import Check
12
- from udata_hydra.db.resource import Resource
13
- from udata_hydra.logger import setup_logging
14
- from udata_hydra.schemas import CheckSchema, ResourceQuerySchema
15
- from udata_hydra.utils.minio import delete_resource_from_minio
16
- from udata_hydra.worker import QUEUES
17
-
18
- log = setup_logging()
19
- routes = web.RouteTableDef()
20
-
21
-
22
- def _get_args(request, params=("url", "resource_id")) -> list:
23
- """Get GET parameters from request"""
24
- data = [request.query.get(param) for param in params]
25
- if not any(data):
26
- raise web.HTTPBadRequest()
27
- return data
28
-
29
-
30
- @routes.post("/api/resource/created/")
31
- async def resource_created(request: web.Request) -> web.Response:
32
- """Endpoint to receive a resource creation event from a source
33
- Will create a new resource in the DB "catalog" table and mark it as priority for next crawling
34
- Respond with a 200 status code and a JSON body with a message key set to "created"
35
- If error, respond with a 400 status code
36
- """
37
- try:
38
- payload = await request.json()
39
- valid_payload: dict = ResourceQuerySchema().load(payload)
40
- except ValidationError as err:
41
- raise web.HTTPBadRequest(text=json.dumps(err.messages))
42
-
43
- resource = valid_payload["document"]
44
- if not resource:
45
- raise web.HTTPBadRequest(text="Missing document body")
46
-
47
- dataset_id = valid_payload["dataset_id"]
48
- resource_id = valid_payload["resource_id"]
49
-
50
- await Resource.insert(
51
- dataset_id=dataset_id,
52
- resource_id=resource_id,
53
- url=resource["url"],
54
- priority=True,
55
- )
56
-
57
- return web.json_response({"message": "created"})
58
-
59
-
60
- @routes.post("/api/resource/updated/")
61
- async def resource_updated(request: web.Request) -> web.Response:
62
- """Endpoint to receive a resource update event from a source
63
- Will update an existing resource in the DB "catalog" table and mark it as priority for next crawling
64
- Respond with a 200 status code and a JSON body with a message key set to "updated"
65
- If error, respond with a 400 status code
66
- """
67
- try:
68
- payload = await request.json()
69
- valid_payload: dict = ResourceQuerySchema().load(payload)
70
- except ValidationError as err:
71
- raise web.HTTPBadRequest(text=json.dumps(err.messages))
72
-
73
- resource = valid_payload["document"]
74
- if not resource:
75
- raise web.HTTPBadRequest(text="Missing document body")
76
-
77
- dataset_id = valid_payload["dataset_id"]
78
- resource_id = valid_payload["resource_id"]
79
-
80
- await Resource.update_or_insert(dataset_id, resource_id, resource["url"])
81
-
82
- return web.json_response({"message": "updated"})
83
-
84
-
85
- @routes.post("/api/resource/deleted/")
86
- async def resource_deleted(request: web.Request) -> web.Response:
87
- try:
88
- payload = await request.json()
89
- valid_payload: dict = ResourceQuerySchema().load(payload)
90
- except ValidationError as err:
91
- raise web.HTTPBadRequest(text=json.dumps(err.messages))
92
-
93
- dataset_id = valid_payload["dataset_id"]
94
- resource_id = valid_payload["resource_id"]
95
-
96
- pool = request.app["pool"]
97
- async with pool.acquire() as connection:
98
- if config.SAVE_TO_MINIO:
99
- delete_resource_from_minio(dataset_id, resource_id)
100
- # Mark resource as deleted in catalog table
101
- q = f"""UPDATE catalog SET deleted = TRUE WHERE resource_id = '{resource_id}';"""
102
- await connection.execute(q)
103
-
104
- return web.json_response({"message": "deleted"})
105
-
106
-
107
- @routes.get("/api/checks/latest/")
108
- async def get_check(request: web.Request) -> web.Response:
109
- """Get the latest check for a given URL or resource_id"""
110
- url, resource_id = _get_args(request)
111
- data = await Check.get_latest(url, resource_id)
112
- if not data:
113
- raise web.HTTPNotFound()
114
- if data["deleted"]:
115
- raise web.HTTPGone()
116
- return web.json_response(CheckSchema().dump(dict(data)))
117
-
118
-
119
- @routes.get("/api/checks/all/")
120
- async def get_checks(request: web.Request) -> web.Response:
121
- url, resource_id = _get_args(request)
122
- data = await Check.get_all(url, resource_id)
123
- if not data:
124
- raise web.HTTPNotFound()
125
- return web.json_response([CheckSchema().dump(dict(r)) for r in data])
126
-
127
-
128
- @routes.get("/api/status/crawler/")
129
- async def status_crawler(request: web.Request) -> web.Response:
130
- q = f"""
131
- SELECT
132
- SUM(CASE WHEN last_check IS NULL THEN 1 ELSE 0 END) AS count_left,
133
- SUM(CASE WHEN last_check IS NOT NULL THEN 1 ELSE 0 END) AS count_checked
134
- FROM catalog
135
- WHERE {get_excluded_clause()}
136
- AND catalog.deleted = False
137
- """
138
- stats_catalog = await request.app["pool"].fetchrow(q)
139
-
140
- since = parse_timespan(config.SINCE)
141
- since = datetime.now(timezone.utc) - timedelta(seconds=since)
142
- q = f"""
143
- SELECT
144
- SUM(CASE WHEN checks.created_at <= $1 THEN 1 ELSE 0 END) AS count_outdated
145
- --, SUM(CASE WHEN checks.created_at > $1 THEN 1 ELSE 0 END) AS count_fresh
146
- FROM catalog, checks
147
- WHERE {get_excluded_clause()}
148
- AND catalog.last_check = checks.id
149
- AND catalog.deleted = False
150
- """
151
- stats_checks = await request.app["pool"].fetchrow(q, since)
152
-
153
- count_left = stats_catalog["count_left"] + (stats_checks["count_outdated"] or 0)
154
- # all w/ a check, minus those with an outdated checked
155
- count_checked = stats_catalog["count_checked"] - (stats_checks["count_outdated"] or 0)
156
- total = stats_catalog["count_left"] + stats_catalog["count_checked"]
157
- rate_checked = round(stats_catalog["count_checked"] / total * 100, 1)
158
- rate_checked_fresh = round(count_checked / total * 100, 1)
159
-
160
- return web.json_response(
161
- {
162
- "total": total,
163
- "pending_checks": count_left,
164
- "fresh_checks": count_checked,
165
- "checks_percentage": rate_checked,
166
- "fresh_checks_percentage": rate_checked_fresh,
167
- }
168
- )
169
-
170
-
171
- @routes.get("/api/status/worker/")
172
- async def status_worker(request: web.Request) -> web.Response:
173
- res = {"queued": {q: len(context.queue(q)) for q in QUEUES}}
174
- return web.json_response(res)
175
-
176
-
177
- @routes.get("/api/stats/")
178
- async def stats(request: web.Request) -> web.Response:
179
- q = f"""
180
- SELECT count(*) AS count_checked
181
- FROM catalog
182
- WHERE {get_excluded_clause()}
183
- AND last_check IS NOT NULL
184
- AND catalog.deleted = False
185
- """
186
- stats_catalog = await request.app["pool"].fetchrow(q)
187
-
188
- q = f"""
189
- SELECT
190
- SUM(CASE WHEN error IS NULL AND timeout = False THEN 1 ELSE 0 END) AS count_ok,
191
- SUM(CASE WHEN error IS NOT NULL THEN 1 ELSE 0 END) AS count_error,
192
- SUM(CASE WHEN timeout = True THEN 1 ELSE 0 END) AS count_timeout
193
- FROM catalog, checks
194
- WHERE {get_excluded_clause()}
195
- AND catalog.last_check = checks.id
196
- AND catalog.deleted = False
197
- """
198
- stats_status = await request.app["pool"].fetchrow(q)
199
-
200
- def cmp_rate(key):
201
- if stats_catalog["count_checked"] == 0:
202
- return 0
203
- return round(stats_status[key] / stats_catalog["count_checked"] * 100, 1)
204
-
205
- q = f"""
206
- SELECT checks.status, count(*) as count FROM checks, catalog
207
- WHERE catalog.last_check = checks.id
208
- AND checks.status IS NOT NULL
209
- AND {get_excluded_clause()}
210
- AND last_check IS NOT NULL
211
- AND catalog.deleted = False
212
- GROUP BY checks.status
213
- ORDER BY count DESC;
214
- """
215
- res = await request.app["pool"].fetch(q)
216
- return web.json_response(
217
- {
218
- "status": sorted(
219
- [
220
- {
221
- "label": s,
222
- "count": stats_status[f"count_{s}"] or 0,
223
- "percentage": cmp_rate(f"count_{s}"),
224
- }
225
- for s in ["error", "timeout", "ok"]
226
- ],
227
- key=lambda x: x["count"],
228
- reverse=True,
229
- ),
230
- "status_codes": [
231
- {
232
- "code": r["status"],
233
- "count": r["count"],
234
- "percentage": round(r["count"] / sum(r["count"] for r in res) * 100, 1),
235
- }
236
- for r in res
237
- ],
238
- }
239
- )
240
-
241
-
242
- @routes.get("/api/health/")
243
- async def health(request: web.Request) -> web.Response:
244
- test_connection = await request.app["pool"].fetchrow("SELECT 1")
245
- assert next(test_connection.values()) == 1
246
- return web.HTTPOk()
247
-
248
-
249
- async def app_factory() -> web.Application:
250
- async def app_startup(app):
251
- app["pool"] = await context.pool()
252
-
253
- async def app_cleanup(app):
254
- if "pool" in app:
255
- await app["pool"].close()
256
-
257
- app = web.Application()
258
- app.add_routes(routes)
259
- app.on_startup.append(app_startup)
260
- app.on_cleanup.append(app_cleanup)
261
- return app
262
-
263
-
264
- def run():
265
- web.run_app(app_factory(), path=os.environ.get("HYDRA_APP_SOCKET_PATH"))
266
-
267
-
268
- if __name__ == "__main__":
269
- run()
@@ -1,11 +0,0 @@
1
- import json
2
-
3
-
4
- def is_json_file(filename: str) -> bool:
5
- """Checks if a string is a valid json"""
6
- try:
7
- with open(filename) as f:
8
- json.load(f)
9
- return True
10
- except ValueError:
11
- return False