cmem-plugin-validation 1.2.0__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cmem-plugin-validation
3
- Version: 1.2.0
3
+ Version: 1.3.0
4
4
  Summary: Validate graph and data structures.
5
5
  License: Apache-2.0
6
6
  License-File: LICENSE
@@ -14,10 +14,10 @@ Classifier: License :: OSI Approved :: Apache Software License
14
14
  Classifier: Programming Language :: Python :: 3
15
15
  Classifier: Programming Language :: Python :: 3.13
16
16
  Classifier: Programming Language :: Python :: 3.14
17
- Requires-Dist: cmem-cmempy (>=25.3.0)
18
- Requires-Dist: cmem-plugin-base (>=4.15.0,<5.0.0)
17
+ Requires-Dist: cmem-client (>=1.0.0,<2.0.0)
18
+ Requires-Dist: cmem-plugin-base (>=4.19.0,<5.0.0)
19
+ Requires-Dist: httpx (>=0.27.0,<0.28.0)
19
20
  Requires-Dist: jsonschema (>=4.25.1,<5.0.0)
20
- Requires-Dist: requests (>=2.0.1)
21
21
  Project-URL: Homepage, https://github.com/eccenca/cmem-plugin-validation
22
22
  Description-Content-Type: text/markdown
23
23
 
@@ -7,12 +7,10 @@ from collections.abc import Generator, Sequence
7
7
  from types import SimpleNamespace
8
8
  from typing import Any
9
9
 
10
- from cmem.cmempy.workspace.projects.resources.resource import get_resource_response
11
- from cmem.cmempy.workspace.tasks import get_task
10
+ from cmem_client.client import Client
12
11
  from cmem_plugin_base.dataintegration.context import (
13
12
  ExecutionContext,
14
13
  ExecutionReport,
15
- UserContext,
16
14
  )
17
15
  from cmem_plugin_base.dataintegration.description import Icon, Plugin, PluginParameter
18
16
  from cmem_plugin_base.dataintegration.entity import Entities
@@ -24,11 +22,6 @@ from cmem_plugin_base.dataintegration.ports import (
24
22
  FlexibleNumberOfInputs,
25
23
  UnknownSchemaPort,
26
24
  )
27
- from cmem_plugin_base.dataintegration.utils import (
28
- setup_cmempy_user_access,
29
- split_task_id,
30
- write_to_dataset,
31
- )
32
25
  from cmem_plugin_base.dataintegration.utils.entity_builder import build_entities_from_data
33
26
  from jsonschema import validate
34
27
  from jsonschema.exceptions import ValidationError
@@ -77,13 +70,6 @@ The error handling behavior is configurable through the `Fail on violations` par
77
70
 
78
71
  DEFAULT_FAIL_ON_VIOLATION = False
79
72
 
80
-
81
- def get_task_metadata(project: str, task: str, context: UserContext) -> dict:
82
- """Get metadata information of a task"""
83
- setup_cmempy_user_access(context=context)
84
- return dict(get_task(project=project, task=task))
85
-
86
-
87
73
  SOURCE = SimpleNamespace()
88
74
  SOURCE.entities = "entities"
89
75
  SOURCE.dataset = "dataset"
@@ -172,7 +158,7 @@ class ValidateEntity(WorkflowPlugin):
172
158
  inputs: Sequence[Entities]
173
159
  execution_context: ExecutionContext
174
160
 
175
- def __init__( # noqa: PLR0913
161
+ def __init__( # noqa: PLR0913 PLR0917
176
162
  self,
177
163
  source_mode: str,
178
164
  target_mode: str,
@@ -297,10 +283,10 @@ class ValidateEntity(WorkflowPlugin):
297
283
  )
298
284
  )
299
285
  if self.target_mode == TARGET.dataset:
300
- write_to_dataset(
301
- dataset_id=f"{context.task.project_id()}:{self.target_dataset}",
302
- file_resource=io.StringIO(json.dumps(valid_json_objects)),
303
- context=context.user,
286
+ Client.from_context(context=context).datasets.post_file_resource(
287
+ project_id=context.task.project_id(),
288
+ dataset_id=self.target_dataset,
289
+ file_resource=io.BytesIO(json.dumps(valid_json_objects).encode("utf-8")),
304
290
  )
305
291
  return None
306
292
 
@@ -319,16 +305,18 @@ class ValidateEntity(WorkflowPlugin):
319
305
  @staticmethod
320
306
  def _get_json_dataset_content(context: ExecutionContext, dataset: str) -> dict | list[dict]:
321
307
  """Get json dataset content"""
322
- dataset_id = f"{context.task.project_id()}:{dataset}"
323
- project_id, task_id = split_task_id(dataset_id)
324
- task_meta_data = get_task_metadata(project_id, task_id, context.user)
325
- resource_name = str(task_meta_data["data"]["parameters"]["file"]["value"])
326
- response = get_resource_response(project_id, resource_name)
327
- return response.json() # type: ignore[no-any-return]
308
+ client = Client.from_context(context=context)
309
+ project_id = context.task.project_id()
310
+ resource_name = str(
311
+ client.datasets.get_item(project_id=project_id, dataset_id=dataset).data.parameters[
312
+ "file"
313
+ ]
314
+ )
315
+ return json.loads(client.files.read(f"{project_id}:{resource_name}")) # type: ignore[no-any-return]
328
316
 
329
317
  def _convert_entities_to_json(
330
318
  self, inputs: Sequence[Entities], path_to_entities: dict[str, Entities], path: str = ""
331
- ) -> Generator[dict[str, Any], None, None]:
319
+ ) -> Generator[dict[str, Any]]:
332
320
  """Convert a sequence of Entities into JSON-like dictionaries using recursive traversal."""
333
321
  for entities in inputs:
334
322
  # Initialize path-to-entities map for the root level
@@ -0,0 +1,31 @@
1
+ """Graph validation process state"""
2
+
3
+ from cmem_client.client import Client
4
+
5
+
6
+ class State:
7
+ """State of a validation process"""
8
+
9
+ client: Client
10
+ id_: str
11
+ data: dict
12
+ status: str
13
+ completed: int
14
+ total: int
15
+ with_violations: int
16
+ violations: int
17
+
18
+ def __init__(self, client: Client, id_: str):
19
+ self.client = client
20
+ self.id_ = id_
21
+ self.refresh()
22
+
23
+ def refresh(self) -> None:
24
+ """Refresh state of validation process"""
25
+ aggregation = self.client.validations.get_aggregation(batch_id=self.id_)
26
+ self.data = aggregation.model_dump(by_alias=True, exclude_none=True)
27
+ self.status = aggregation.state
28
+ self.completed = aggregation.resource_processed_count
29
+ self.total = aggregation.resource_count
30
+ self.with_violations = aggregation.resources_with_violations_count
31
+ self.violations = aggregation.violations_count
@@ -4,9 +4,9 @@ import json
4
4
  from collections.abc import Sequence
5
5
  from time import sleep
6
6
 
7
- from cmem.cmempy.dp.proxy import graph as graph_api
8
- from cmem.cmempy.dp.shacl import validation
9
- from cmem.cmempy.queries import SparqlQuery
7
+ from cmem_client.client import Client
8
+ from cmem_client.models.query_catalog import Query
9
+ from cmem_client.models.validation import STATUS_RUNNING, STATUS_SCHEDULED, ValidationViolation
10
10
  from cmem_plugin_base.dataintegration.context import ExecutionContext, ExecutionReport
11
11
  from cmem_plugin_base.dataintegration.description import Icon, Plugin, PluginParameter
12
12
  from cmem_plugin_base.dataintegration.entity import (
@@ -21,9 +21,8 @@ from cmem_plugin_base.dataintegration.ports import (
21
21
  FixedNumberOfInputs,
22
22
  FixedSchemaPort,
23
23
  )
24
- from cmem_plugin_base.dataintegration.utils import setup_cmempy_user_access
25
24
  from cmem_plugin_base.dataintegration.utils.entity_builder import build_entities_from_data
26
- from requests import HTTPError
25
+ from httpx import HTTPStatusError
27
26
 
28
27
  from cmem_plugin_validation.validate_graph.state import State
29
28
 
@@ -111,7 +110,7 @@ WHERE { ?resource a ?class . FILTER isIRI(?resource) }
111
110
  class ValidateGraph(WorkflowPlugin):
112
111
  """Validate resources in a graph"""
113
112
 
114
- def __init__( # noqa: PLR0913
113
+ def __init__( # noqa: PLR0913 PLR0917
115
114
  self,
116
115
  context_graph: str,
117
116
  shape_graph: str = DEFAULT_SHAPE_GRAPH,
@@ -156,33 +155,29 @@ class ValidateGraph(WorkflowPlugin):
156
155
  ) -> Entities | None:
157
156
  """Run the workflow operator."""
158
157
  self.log.info("Start validation task.")
159
- setup_cmempy_user_access(context=context.user)
158
+ client = Client.from_context(context=context)
160
159
  if self.clear_result_graph and self.result_graph:
161
- graph_api.delete(graph=self.result_graph)
162
- query = SparqlQuery(text=self.sparql_query).get_filled_text(
163
- placeholder={"context_graph": self.context_graph}
160
+ client.graphs.delete_item(key=self.result_graph, skip_if_missing=True)
161
+ query = Query(text=self.sparql_query).fill_placeholders(
162
+ placeholders={"context_graph": self.context_graph}
164
163
  )
165
164
  try:
166
- process_id = validation.start(
165
+ process_id = client.validations.start(
167
166
  context_graph=self.context_graph,
168
167
  shape_graph=self.shape_graph,
169
- result_graph=self.result_graph if self.result_graph else None,
168
+ result_graph=self.result_graph or None,
170
169
  query=query,
171
170
  )
172
- except HTTPError as error:
173
- context.report.update(
174
- ExecutionReport(
175
- error=json.loads(error.response.text)["detail"],
176
- )
177
- )
178
- raise RuntimeError(json.loads(error.response.text)["detail"]) from error
179
- state = State(id_=process_id)
171
+ except HTTPStatusError as error:
172
+ detail = json.loads(error.response.text)["detail"]
173
+ context.report.update(ExecutionReport(error=detail))
174
+ raise RuntimeError(detail) from error
175
+ state = State(client=client, id_=process_id)
180
176
  while True:
181
177
  sleep(1)
182
- setup_cmempy_user_access(context=context.user)
183
178
  state.refresh()
184
179
  if context.workflow and context.workflow.status() != "Running":
185
- validation.cancel(batch_id=process_id)
180
+ client.validations.cancel(batch_id=process_id)
186
181
  context.report.update(
187
182
  ExecutionReport(
188
183
  entity_count=state.completed,
@@ -192,7 +187,7 @@ class ValidateGraph(WorkflowPlugin):
192
187
  )
193
188
  self.log.info("End validation task (Cancelled Workflow).")
194
189
  return None
195
- if state.status in (validation.STATUS_SCHEDULED, validation.STATUS_RUNNING):
190
+ if state.status in (STATUS_SCHEDULED, STATUS_RUNNING):
196
191
  # when reported as running or scheduled, start another loop
197
192
  context.report.update(
198
193
  ExecutionReport(
@@ -224,11 +219,22 @@ class ValidateGraph(WorkflowPlugin):
224
219
  if not self.output_results:
225
220
  return None
226
221
 
227
- violations = []
228
- for result in list(validation.get(batch_id=process_id)["results"]):
229
- resource_iri = result.get("resourceIri")
230
- for _ in result["violations"]:
231
- violation = dict(_)
232
- violation["resourceIri"] = resource_iri
233
- violations.append(violation)
222
+ violations = [
223
+ self._as_violation_data(violation, resource_iri=result.resource_iri)
224
+ for result in client.validations.get_result(batch_id=process_id).results
225
+ for violation in result.violations
226
+ ]
234
227
  return build_entities_from_data(data=violations)
228
+
229
+ def _as_violation_data(self, violation: ValidationViolation, resource_iri: str) -> dict:
230
+ """Turn a violation into a plain dict ordered like the output schema
231
+
232
+ The entity paths of the returned entities are derived from the key order of these
233
+ dicts, so the keys of the output schema come first, followed by any additional key
234
+ the API delivers and the IRI of the validated resource.
235
+ """
236
+ data: dict = violation.model_dump(by_alias=True, exclude_none=True)
237
+ ordered: dict = {
238
+ _.path: data.pop(_.path) for _ in self.output_schema.paths if _.path in data
239
+ }
240
+ return ordered | data | {"resourceIri": resource_iri}
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "cmem-plugin-validation"
3
- version = "1.2.0"
3
+ version = "1.3.0"
4
4
  license = "Apache-2.0"
5
5
  description = "Validate graph and data structures."
6
6
  authors = ["eccenca GmbH <cmempy-developer@eccenca.com>"]
@@ -13,32 +13,36 @@ keywords = [
13
13
  ]
14
14
  homepage = "https://github.com/eccenca/cmem-plugin-validation"
15
15
 
16
+ [tool.poetry.requires-plugins]
17
+ poetry-plugin-export = "^1.10.0"
18
+ poetry-plugin-shell = "^1.0.1"
19
+ poetry-dynamic-versioning = "^1.10.0"
20
+
16
21
  [tool.poetry.dependencies]# if you need to change python version here, change it also in .python-version
17
22
  python = "^3.13"
18
23
  jsonschema = "^4.25.1"
19
- cmem-cmempy = ">=25.3.0"
20
- requests = ">=2.0.1"
24
+ cmem-client = "^1.0.0"
25
+ httpx = "^0.27.0"
21
26
 
22
27
  [tool.poetry.dependencies.cmem-plugin-base]
23
- version = "^4.15.0"
28
+ version = "^4.19.0"
24
29
  allow-prereleases = false
25
30
 
26
31
  [tool.poetry.group.dev.dependencies.cmem-cmemc]
27
32
  version = ">=24.2.0"
28
33
 
29
34
  [tool.poetry.group.dev.dependencies]
30
- deptry = "^0.23.1"
31
- genbadge = {extras = ["coverage"], version = "^1.1.2"}
32
- mypy = "^1.18.2"
33
- pip = "^25.2"
34
- pytest = "^8.4.2"
35
- pytest-cov = "^7.0.0"
35
+ deptry = "^0.25.1"
36
+ genbadge = {extras = ["coverage"], version = "^1.1.3"}
37
+ mypy = "^2.3.0"
38
+ pip = "^26"
39
+ pytest = "^9.1.1"
40
+ pytest-cov = "^7.1.0"
36
41
  pytest-dotenv = "^0.5.2"
37
- pytest-html = "^4.1.1"
38
- pytest-memray = { version = "^1.8.0", markers = "platform_system != 'Windows'" }
39
- ruff = "^0.13.3"
40
- safety = "^1.10.3"
41
- types-requests = "^2.31.0.20240406"
42
+ pytest-html = "^4.2.0"
43
+ pytest-memray = { version = "^1.10.0", markers = "platform_system != 'Windows'" }
44
+ ruff = "^0.16.2"
45
+ trivy-py-ecc = "^0.73.0.1"
42
46
 
43
47
  [build-system]
44
48
  requires = ["poetry-core>=1.0.0", "poetry-dynamic-versioning"]
@@ -71,7 +75,7 @@ exclude_also = [
71
75
 
72
76
  [tool.ruff]
73
77
  line-length = 100
74
- target-version = "py311"
78
+ target-version = "py313"
75
79
 
76
80
  [tool.ruff.format]
77
81
  line-ending = "lf" # Use `\n` line endings for all files
@@ -95,4 +99,5 @@ ignore = [
95
99
  "PD", # opinionated linting for pandas code
96
100
  "S101", # use of assert detected
97
101
  "TRY003", # Avoid specifying long messages outside the exception class
102
+ "CPY001", # Missing copyright notice
98
103
  ]
@@ -1,28 +0,0 @@
1
- """Graph validation process state"""
2
-
3
- from cmem.cmempy.dp.shacl import validation
4
-
5
-
6
- class State:
7
- """State of a validation process"""
8
-
9
- id_: str
10
- data: dict
11
- status: str
12
- completed: int
13
- total: int
14
- with_violations: int
15
- violations: int
16
-
17
- def __init__(self, id_: str):
18
- self.id_ = id_
19
- self.refresh()
20
-
21
- def refresh(self) -> None:
22
- """Refresh state of validation process"""
23
- self.data = validation.get_aggregation(batch_id=self.id_)
24
- self.status = self.data.get("state", "UNKNOWN")
25
- self.completed = self.data.get("resourceProcessedCount", 0)
26
- self.total = self.data.get("resourceCount", 0)
27
- self.with_violations = self.data.get("resourcesWithViolationsCount", 0)
28
- self.violations = self.data.get("violationsCount", 0)