relatics-toolkit 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- relatics_toolkit-1.0.0/PKG-INFO +100 -0
- relatics_toolkit-1.0.0/README.md +86 -0
- relatics_toolkit-1.0.0/pyproject.toml +46 -0
- relatics_toolkit-1.0.0/pyproject.toml.orig +42 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/__init__.py +5 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/extraction_service.py +166 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/ingestion/__init__.py +0 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/ingestion/relatics_client.py +104 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/ingestion/xml_parser.py +99 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/processing/__init__.py +0 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/processing/schema.py +38 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/processing/transformer.py +420 -0
- relatics_toolkit-1.0.0/src/relatics_toolkit/processing/validator.py +120 -0
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: relatics-toolkit
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: A package that enables fast and easy ETL processing of a Relatics workspace.
|
|
5
|
+
Requires-Dist: mkdocs>=1.6.1
|
|
6
|
+
Requires-Dist: mkdocs-material>=9.7.7
|
|
7
|
+
Requires-Dist: mkdocstrings[python]>=1.0.6
|
|
8
|
+
Requires-Dist: pandas>=3.0.3
|
|
9
|
+
Requires-Dist: pyarrow>=25.0.1
|
|
10
|
+
Requires-Dist: pytest>=9.1.1
|
|
11
|
+
Requires-Dist: pytest-mock>=3.15.1
|
|
12
|
+
Requires-Python: >=3.14
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
|
|
15
|
+
# Relatics Extractor
|
|
16
|
+
[](https://github.com/nvvliet-croonwolterendros/relatics-extractor/actions/workflows/test.yml)
|
|
17
|
+
|
|
18
|
+
A Python package for extracting, parsing, and transforming data from Relatics (a requirements management tool) into database-ready pandas DataFrames.
|
|
19
|
+
|
|
20
|
+
## Overview
|
|
21
|
+
|
|
22
|
+
The Relatics Extractor is designed to streamline the process of extracting data from Relatics API endpoints and converting it into structured, normalized tables suitable for database storage. It handles complex relationships between entities, nested XML structures, and provides robust error handling throughout the extraction pipeline.
|
|
23
|
+
|
|
24
|
+
## Key Features
|
|
25
|
+
|
|
26
|
+
- **Authentication**: Implements OAuth 2.0 token-based authentication with Relatics API
|
|
27
|
+
- **Data Extraction**: Handles API calls to retrieve elements and their relationships
|
|
28
|
+
- **XML Parsing**: Parses deeply nested XML structures into pandas DataFrames
|
|
29
|
+
- **Schema Validation**: Validates data against predefined schemas and normalizes tables
|
|
30
|
+
- **Relationship Management**: Properly handles different relationship cardinalities:
|
|
31
|
+
- :1 (one-to-one) relations are embedded in element tables
|
|
32
|
+
- :n (many-to-one) relations become link tables
|
|
33
|
+
- **Multithreading Support**: Uses ThreadPoolExecutor for faster processing of multiple elements
|
|
34
|
+
- **Error Handling**: Robust error handling with logging and failure tracking
|
|
35
|
+
|
|
36
|
+
## Documentation
|
|
37
|
+
|
|
38
|
+
Documentation can be found [here](https://nvvliet-croonwolterendros.github.io/relatics-extractor/)
|
|
39
|
+
|
|
40
|
+
## Installation
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
# Install from source
|
|
44
|
+
pip install .
|
|
45
|
+
|
|
46
|
+
# Or install in development mode
|
|
47
|
+
pip install -e .
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Usage
|
|
51
|
+
|
|
52
|
+
### Basic Setup
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from relatics_extractor import Extractor
|
|
56
|
+
|
|
57
|
+
# Initialize the extractor with your credentials
|
|
58
|
+
extractor = Extractor(
|
|
59
|
+
client_id="your_client_id",
|
|
60
|
+
client_secret="your_client_secret",
|
|
61
|
+
environment="your_environment"
|
|
62
|
+
)
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
### Running ETL Pipeline
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
# Run the extraction process for multiple workspaces
|
|
69
|
+
tables = extractor.run_etl_fast(
|
|
70
|
+
workspaces=["workspace1", "workspace2"],
|
|
71
|
+
element_operation="GetElements",
|
|
72
|
+
datamodel_operation="GetDatamodel"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
# Access extracted tables
|
|
76
|
+
for table_name, df in tables.items():
|
|
77
|
+
print(f"Table: {table_name}")
|
|
78
|
+
print(df.head())
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Architecture
|
|
82
|
+
|
|
83
|
+
The extractor follows a clear pipeline architecture:
|
|
84
|
+
|
|
85
|
+
1. **Extraction**: Uses `RelaticsClient` to fetch data from API endpoints
|
|
86
|
+
2. **Parsing**: Parses XML responses using `parse_xml` function
|
|
87
|
+
3. **Validation**: Validates schemas using `validator.py`
|
|
88
|
+
4. **Transformation**: Transforms data using `transformer.py` to normalize relationships
|
|
89
|
+
|
|
90
|
+
## Key Components
|
|
91
|
+
|
|
92
|
+
- **Extractor**: Main class that orchestrates the full ETL pipeline
|
|
93
|
+
- **RelaticsClient**: Handles OAuth 2.0 authentication and API requests
|
|
94
|
+
- **parse_xml**: Parses deeply nested XML structures
|
|
95
|
+
- **transformer**: Transforms raw data into structured tables with proper relationship handling
|
|
96
|
+
- **validator**: Validates schema compliance and normalizes table structures
|
|
97
|
+
|
|
98
|
+
## License
|
|
99
|
+
|
|
100
|
+
MIT
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# Relatics Extractor
|
|
2
|
+
[](https://github.com/nvvliet-croonwolterendros/relatics-extractor/actions/workflows/test.yml)
|
|
3
|
+
|
|
4
|
+
A Python package for extracting, parsing, and transforming data from Relatics (a requirements management tool) into database-ready pandas DataFrames.
|
|
5
|
+
|
|
6
|
+
## Overview
|
|
7
|
+
|
|
8
|
+
The Relatics Extractor is designed to streamline the process of extracting data from Relatics API endpoints and converting it into structured, normalized tables suitable for database storage. It handles complex relationships between entities, nested XML structures, and provides robust error handling throughout the extraction pipeline.
|
|
9
|
+
|
|
10
|
+
## Key Features
|
|
11
|
+
|
|
12
|
+
- **Authentication**: Implements OAuth 2.0 token-based authentication with Relatics API
|
|
13
|
+
- **Data Extraction**: Handles API calls to retrieve elements and their relationships
|
|
14
|
+
- **XML Parsing**: Parses deeply nested XML structures into pandas DataFrames
|
|
15
|
+
- **Schema Validation**: Validates data against predefined schemas and normalizes tables
|
|
16
|
+
- **Relationship Management**: Properly handles different relationship cardinalities:
|
|
17
|
+
- :1 (one-to-one) relations are embedded in element tables
|
|
18
|
+
- :n (many-to-one) relations become link tables
|
|
19
|
+
- **Multithreading Support**: Uses ThreadPoolExecutor for faster processing of multiple elements
|
|
20
|
+
- **Error Handling**: Robust error handling with logging and failure tracking
|
|
21
|
+
|
|
22
|
+
## Documentation
|
|
23
|
+
|
|
24
|
+
Documentation can be found [here](https://nvvliet-croonwolterendros.github.io/relatics-extractor/)
|
|
25
|
+
|
|
26
|
+
## Installation
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
# Install from source
|
|
30
|
+
pip install .
|
|
31
|
+
|
|
32
|
+
# Or install in development mode
|
|
33
|
+
pip install -e .
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Usage
|
|
37
|
+
|
|
38
|
+
### Basic Setup
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from relatics_extractor import Extractor
|
|
42
|
+
|
|
43
|
+
# Initialize the extractor with your credentials
|
|
44
|
+
extractor = Extractor(
|
|
45
|
+
client_id="your_client_id",
|
|
46
|
+
client_secret="your_client_secret",
|
|
47
|
+
environment="your_environment"
|
|
48
|
+
)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
### Running ETL Pipeline
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
# Run the extraction process for multiple workspaces
|
|
55
|
+
tables = extractor.run_etl_fast(
|
|
56
|
+
workspaces=["workspace1", "workspace2"],
|
|
57
|
+
element_operation="GetElements",
|
|
58
|
+
datamodel_operation="GetDatamodel"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# Access extracted tables
|
|
62
|
+
for table_name, df in tables.items():
|
|
63
|
+
print(f"Table: {table_name}")
|
|
64
|
+
print(df.head())
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Architecture
|
|
68
|
+
|
|
69
|
+
The extractor follows a clear pipeline architecture:
|
|
70
|
+
|
|
71
|
+
1. **Extraction**: Uses `RelaticsClient` to fetch data from API endpoints
|
|
72
|
+
2. **Parsing**: Parses XML responses using `parse_xml` function
|
|
73
|
+
3. **Validation**: Validates schemas using `validator.py`
|
|
74
|
+
4. **Transformation**: Transforms data using `transformer.py` to normalize relationships
|
|
75
|
+
|
|
76
|
+
## Key Components
|
|
77
|
+
|
|
78
|
+
- **Extractor**: Main class that orchestrates the full ETL pipeline
|
|
79
|
+
- **RelaticsClient**: Handles OAuth 2.0 authentication and API requests
|
|
80
|
+
- **parse_xml**: Parses deeply nested XML structures
|
|
81
|
+
- **transformer**: Transforms raw data into structured tables with proper relationship handling
|
|
82
|
+
- **validator**: Validates schema compliance and normalizes table structures
|
|
83
|
+
|
|
84
|
+
## License
|
|
85
|
+
|
|
86
|
+
MIT
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "relatics-toolkit"
|
|
3
|
+
version = "1.0.0"
|
|
4
|
+
description = "A package that enables fast and easy ETL processing of a Relatics workspace."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.14"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"mkdocs>=1.6.1",
|
|
9
|
+
"mkdocs-material>=9.7.7",
|
|
10
|
+
"mkdocstrings[python]>=1.0.6",
|
|
11
|
+
"pandas>=3.0.3",
|
|
12
|
+
"pyarrow>=25.0.1",
|
|
13
|
+
"pytest>=9.1.1",
|
|
14
|
+
"pytest-mock>=3.15.1",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[dependency-groups]
|
|
18
|
+
dev = ["ruff>=0.16.6"]
|
|
19
|
+
|
|
20
|
+
[tool.pytest.ini_options]
|
|
21
|
+
pythonpath = ["."]
|
|
22
|
+
|
|
23
|
+
[tool.ruff]
|
|
24
|
+
line-length = 88
|
|
25
|
+
|
|
26
|
+
[tool.ruff.lint]
|
|
27
|
+
select = [
|
|
28
|
+
"E",
|
|
29
|
+
"F",
|
|
30
|
+
"I",
|
|
31
|
+
"UP",
|
|
32
|
+
"B",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
[tool.ruff.format]
|
|
36
|
+
quote-style = "double"
|
|
37
|
+
|
|
38
|
+
[[tool.uv.index]]
|
|
39
|
+
name = "testpypi"
|
|
40
|
+
url = "https://test.pypi.org/simple/"
|
|
41
|
+
publish-url = "https://test.pypi.org/legacy/"
|
|
42
|
+
explicit = true
|
|
43
|
+
|
|
44
|
+
[build-system]
|
|
45
|
+
requires = ["uv_build>=0.12.5,<0.13"]
|
|
46
|
+
build-backend = "uv_build"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "relatics-toolkit"
|
|
3
|
+
version = "1.0.0"
|
|
4
|
+
description = "A package that enables fast and easy ETL processing of a Relatics workspace."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.14"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"mkdocs>=1.6.1",
|
|
9
|
+
"mkdocs-material>=9.7.7",
|
|
10
|
+
"mkdocstrings[python]>=1.0.6",
|
|
11
|
+
"pandas>=3.0.3",
|
|
12
|
+
"pyarrow>=25.0.1",
|
|
13
|
+
"pytest>=9.1.1",
|
|
14
|
+
"pytest-mock>=3.15.1",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[dependency-groups]
|
|
18
|
+
dev = [
|
|
19
|
+
"ruff>=0.16.6",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
[tool.pytest.ini_options]
|
|
23
|
+
pythonpath = ["."]
|
|
24
|
+
|
|
25
|
+
[tool.ruff]
|
|
26
|
+
line-length = 88
|
|
27
|
+
|
|
28
|
+
[tool.ruff.lint]
|
|
29
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
30
|
+
|
|
31
|
+
[tool.ruff.format]
|
|
32
|
+
quote-style = "double"
|
|
33
|
+
|
|
34
|
+
[[tool.uv.index]]
|
|
35
|
+
name = "testpypi"
|
|
36
|
+
url = "https://test.pypi.org/simple/"
|
|
37
|
+
publish-url = "https://test.pypi.org/legacy/"
|
|
38
|
+
explicit = true
|
|
39
|
+
|
|
40
|
+
[build-system]
|
|
41
|
+
requires = ["uv_build>=0.12.5,<0.13"]
|
|
42
|
+
build-backend = "uv_build"
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
from relatics_toolkit.extraction_service import extract_element_tables
|
|
2
|
+
from relatics_toolkit.ingestion.relatics_client import RelaticsClient
|
|
3
|
+
from relatics_toolkit.ingestion.xml_parser import parse_xml
|
|
4
|
+
|
|
5
|
+
__all__ = ["RelaticsClient", "extract_element_tables", "parse_xml"]
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
3
|
+
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from relatics_toolkit.ingestion.relatics_client import RelaticsClient
|
|
7
|
+
from relatics_toolkit.ingestion.xml_parser import parse_xml
|
|
8
|
+
from relatics_toolkit.processing.schema import SCHEMA
|
|
9
|
+
from relatics_toolkit.processing.transformer import create_element_tables
|
|
10
|
+
from relatics_toolkit.processing.validator import normalize_tables, validate_schema
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def extract_element_tables(
|
|
16
|
+
client: RelaticsClient,
|
|
17
|
+
workspace_elements: dict[str, list[str]],
|
|
18
|
+
operation: str,
|
|
19
|
+
parallel: bool = False,
|
|
20
|
+
max_workers: int | None = None,
|
|
21
|
+
inline_relations: list[str] | None = None,
|
|
22
|
+
) -> dict[str, pd.DataFrame]:
|
|
23
|
+
"""
|
|
24
|
+
Extracts and transforms Relatics elements into normalized tables.
|
|
25
|
+
|
|
26
|
+
For each configured workspace and element combination, retrieves the
|
|
27
|
+
corresponding Relatics XML payload, validates the extracted schema,
|
|
28
|
+
normalizes the resulting tables, and applies business transformations.
|
|
29
|
+
|
|
30
|
+
Processing can be executed sequentially or in parallel.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
client: Configured Relatics API client.
|
|
34
|
+
workspace_elements: Mapping of workspace IDs to lists of element IDs
|
|
35
|
+
that should be extracted.
|
|
36
|
+
operation: Relatics operation name used to retrieve the
|
|
37
|
+
element data.
|
|
38
|
+
parallel: Whether element extraction should be executed in
|
|
39
|
+
parallel.
|
|
40
|
+
max_workers: Maximum number of worker threads used when
|
|
41
|
+
run_parallel is True. If None, the ThreadPoolExecutor
|
|
42
|
+
default is used.
|
|
43
|
+
inline_relations: Relation names of Relations to R2 Elements
|
|
44
|
+
whose values should be materialized directly in the
|
|
45
|
+
resulting element tables (must be to-one relations).
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
Dictionary mapping table names to transformed pandas DataFrames.
|
|
49
|
+
|
|
50
|
+
Raises:
|
|
51
|
+
Exception: Any exception raised during retrieval, validation,
|
|
52
|
+
normalization, or transformation of element data.
|
|
53
|
+
"""
|
|
54
|
+
jobs = [
|
|
55
|
+
(workspace_id, element_id)
|
|
56
|
+
for workspace_id, element_ids in workspace_elements.items()
|
|
57
|
+
for element_id in element_ids
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
logger.info(
|
|
61
|
+
"Starting extraction for %s elements (parallel=%s)",
|
|
62
|
+
len(jobs),
|
|
63
|
+
parallel,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
tables: dict[str, pd.DataFrame] = {}
|
|
67
|
+
|
|
68
|
+
if parallel:
|
|
69
|
+
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
70
|
+
future_map = {
|
|
71
|
+
executor.submit(
|
|
72
|
+
_process_element,
|
|
73
|
+
element_id=element_id,
|
|
74
|
+
client=client,
|
|
75
|
+
workspace_id=workspace_id,
|
|
76
|
+
operation=operation,
|
|
77
|
+
inline_relations=inline_relations,
|
|
78
|
+
): (workspace_id, element_id)
|
|
79
|
+
for workspace_id, element_id in jobs
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
for future in as_completed(future_map):
|
|
83
|
+
workspace_id, element_id = future_map[future]
|
|
84
|
+
|
|
85
|
+
try:
|
|
86
|
+
_merge_tables(tables, future.result())
|
|
87
|
+
except Exception:
|
|
88
|
+
logger.exception(
|
|
89
|
+
"Failed processing workspace_id=%s element_id=%s",
|
|
90
|
+
workspace_id,
|
|
91
|
+
element_id,
|
|
92
|
+
)
|
|
93
|
+
raise
|
|
94
|
+
else:
|
|
95
|
+
for workspace_id, element_id in jobs:
|
|
96
|
+
try:
|
|
97
|
+
_merge_tables(
|
|
98
|
+
tables,
|
|
99
|
+
_process_element(
|
|
100
|
+
element_id=element_id,
|
|
101
|
+
client=client,
|
|
102
|
+
workspace_id=workspace_id,
|
|
103
|
+
operation=operation,
|
|
104
|
+
inline_relations=inline_relations,
|
|
105
|
+
),
|
|
106
|
+
)
|
|
107
|
+
except Exception:
|
|
108
|
+
logger.exception(
|
|
109
|
+
"Failed processing workspace_id=%s element_id=%s",
|
|
110
|
+
workspace_id,
|
|
111
|
+
element_id,
|
|
112
|
+
)
|
|
113
|
+
raise
|
|
114
|
+
|
|
115
|
+
logger.info(
|
|
116
|
+
"Extraction completed successfully. Generated %s tables.",
|
|
117
|
+
len(tables),
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
return tables
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _process_element(
|
|
124
|
+
element_id: str,
|
|
125
|
+
client: RelaticsClient,
|
|
126
|
+
workspace_id: str,
|
|
127
|
+
operation: str,
|
|
128
|
+
schema: dict[str, dict] = SCHEMA,
|
|
129
|
+
inline_relations: list[str] | None = None,
|
|
130
|
+
) -> dict[str, pd.DataFrame]:
|
|
131
|
+
"""
|
|
132
|
+
Processes a Relatics element with its first order relations and properties.
|
|
133
|
+
"""
|
|
134
|
+
parameters = {"ConfigurationOfRef": element_id}
|
|
135
|
+
|
|
136
|
+
element_data = client.get_request(
|
|
137
|
+
workspace_id=workspace_id,
|
|
138
|
+
operation=operation,
|
|
139
|
+
parameters=parameters,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
tables = {table_name: parse_xml(element_data, table_name) for table_name in schema}
|
|
143
|
+
normalized_tables = normalize_tables(tables=tables, schema=schema)
|
|
144
|
+
validate_schema(tables=normalized_tables, schema=schema)
|
|
145
|
+
transformed_tables = create_element_tables(
|
|
146
|
+
tables=normalized_tables, inline_relations=inline_relations
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
for table in transformed_tables.values():
|
|
150
|
+
table["workspace_id"] = workspace_id
|
|
151
|
+
|
|
152
|
+
return transformed_tables
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _merge_tables(
|
|
156
|
+
tables: dict[str, pd.DataFrame],
|
|
157
|
+
element_tables: dict[str, pd.DataFrame],
|
|
158
|
+
) -> None:
|
|
159
|
+
for table_name, df in element_tables.items():
|
|
160
|
+
if table_name in tables:
|
|
161
|
+
tables[table_name] = pd.concat(
|
|
162
|
+
[tables[table_name], df],
|
|
163
|
+
ignore_index=True,
|
|
164
|
+
)
|
|
165
|
+
else:
|
|
166
|
+
tables[table_name] = df
|
|
File without changes
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
import requests
|
|
2
|
+
import xml.etree.ElementTree as ET
|
|
3
|
+
from typing import Dict, Tuple
|
|
4
|
+
import logging
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
class TokenRequestError(Exception):
|
|
9
|
+
"""Raised when token retrieval fails."""
|
|
10
|
+
|
|
11
|
+
class APIRequestError(Exception):
|
|
12
|
+
"""Raised when the API call fails."""
|
|
13
|
+
|
|
14
|
+
class XMLParseError(Exception):
|
|
15
|
+
"""Raised when XML parsing fails."""
|
|
16
|
+
|
|
17
|
+
class RelaticsClient:
|
|
18
|
+
"""Client for making OAuth2 get requests to relatics.
|
|
19
|
+
|
|
20
|
+
Client for making requests to relatics webservices. Will automatically request a token and subsequently do a get request to a webservice.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
client_id (str): the OAUTH client id obtained from the Relatics environment studio.
|
|
24
|
+
client_secret (str): the OAUTH client secret obtained from the Relatics environment studio.
|
|
25
|
+
environment (str): The subdomain of relaticonline. in https://example.relaticsonline.com, example is the environment string.
|
|
26
|
+
|
|
27
|
+
Returns:
|
|
28
|
+
RelaticsClient (RelaticsClient): Class capable of doing get requests to webservices in the specified environment.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__(self, client_id: str, client_secret: str, environment: str) -> None:
|
|
32
|
+
self.client_id = client_id
|
|
33
|
+
self.client_secret = client_secret
|
|
34
|
+
self.environment = environment
|
|
35
|
+
|
|
36
|
+
def get_request(self, workspace_id: str, operation: str, parameters: Dict[str, str] = {}) -> ET.Element:
|
|
37
|
+
"""
|
|
38
|
+
Executes a GET request to the Relatics DataExchange API.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
workspace_id: Workspace identifier
|
|
42
|
+
operation: API operation name
|
|
43
|
+
parameters: Query parameters
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
result: Parsed XML root element
|
|
47
|
+
"""
|
|
48
|
+
api_endpoint = f"https://{self.environment}.relaticsonline.com/DataExchange/{workspace_id}/{operation}"
|
|
49
|
+
|
|
50
|
+
access_token, token_type = self._get_token()
|
|
51
|
+
|
|
52
|
+
headers = {
|
|
53
|
+
"Authorization": f"{token_type} {access_token}",
|
|
54
|
+
"Content-Type": "application/json",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
body = {
|
|
58
|
+
"Parameters": parameters
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
try:
|
|
62
|
+
api_response = requests.get(api_endpoint, headers=headers, json=body)
|
|
63
|
+
except requests.RequestException as e:
|
|
64
|
+
raise APIRequestError(f"API Request failed: {str(e)}") from e
|
|
65
|
+
|
|
66
|
+
if api_response.status_code != 200:
|
|
67
|
+
raise APIRequestError(f"API request failed with status code {api_response.status_code}: {api_response.text}")
|
|
68
|
+
|
|
69
|
+
try:
|
|
70
|
+
return ET.fromstring(api_response.content)
|
|
71
|
+
except ET.ParseError as e:
|
|
72
|
+
raise XMLParseError("Invalid XML in API response") from e
|
|
73
|
+
|
|
74
|
+
def _get_token(self) -> Tuple[str,str]:
|
|
75
|
+
"""Retrieves OAuth2 token."""
|
|
76
|
+
|
|
77
|
+
token_endpoint = f"https://{self.environment}.relaticsonline.com/oauth2/token"
|
|
78
|
+
|
|
79
|
+
headers = {
|
|
80
|
+
"Content-Type": "application/x-www-form-urlencoded"
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
data = {
|
|
84
|
+
"grant_type": "client_credentials",
|
|
85
|
+
"client_id": self.client_id,
|
|
86
|
+
"client_secret": self.client_secret
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
try:
|
|
90
|
+
token_response = requests.post(token_endpoint, data=data, headers=headers)
|
|
91
|
+
except requests.RequestException as e:
|
|
92
|
+
raise TokenRequestError("Error while requesting token") from e
|
|
93
|
+
|
|
94
|
+
if token_response.status_code != 200:
|
|
95
|
+
raise TokenRequestError(f"Token request failed with status code {token_response.status_code}: {token_response.text}")
|
|
96
|
+
|
|
97
|
+
token_json = token_response.json()
|
|
98
|
+
access_token = token_json.get('access_token')
|
|
99
|
+
token_type = token_json.get('token_type')
|
|
100
|
+
|
|
101
|
+
if not access_token or not token_type:
|
|
102
|
+
raise TokenRequestError("Missing access_token or token_type in response")
|
|
103
|
+
|
|
104
|
+
return access_token, token_type
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from collections import defaultdict
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import itertools
|
|
4
|
+
import xml.etree.ElementTree as ET
|
|
5
|
+
|
|
6
|
+
def parse_xml(root:ET.Element, report_part:str) -> pd.DataFrame:
|
|
7
|
+
"""Function to convert a relatics report part into a pandas DataFrame.
|
|
8
|
+
|
|
9
|
+
Relatics report parts can be deeply nested, this function recusively unpacks the XML and returns a single DataFrame.
|
|
10
|
+
|
|
11
|
+
Args:
|
|
12
|
+
root (xml.etree.ElementTree): The complete xml.etree.ElementTree xml as obtained from the relatics webservice. Can be easily obtained from the RelaticsClient.
|
|
13
|
+
report_part: Specific report part to unpack into a pandas dataframe.
|
|
14
|
+
|
|
15
|
+
Returns:
|
|
16
|
+
df: An unpacked pandas DataFrame of a specific report part.
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
start_element = root.find(report_part)
|
|
20
|
+
|
|
21
|
+
nested_rows = []
|
|
22
|
+
|
|
23
|
+
if start_element is not None:
|
|
24
|
+
nested_rows.append(_parse_xml_to_dict(start_element=start_element))
|
|
25
|
+
|
|
26
|
+
unpacked_rows = []
|
|
27
|
+
|
|
28
|
+
for row in nested_rows:
|
|
29
|
+
unpacked_rows.extend(_recursive_unpack(row))
|
|
30
|
+
|
|
31
|
+
df = pd.DataFrame(unpacked_rows)
|
|
32
|
+
|
|
33
|
+
df.columns = [col.split('.')[-1] for col in df.columns]
|
|
34
|
+
|
|
35
|
+
return df
|
|
36
|
+
|
|
37
|
+
def _parse_xml_to_dict(start_element):
|
|
38
|
+
dictionary = defaultdict(list)
|
|
39
|
+
|
|
40
|
+
# Collect all direct children of the current element
|
|
41
|
+
for child_element in start_element:
|
|
42
|
+
dictionary[child_element.tag].append(child_element)
|
|
43
|
+
|
|
44
|
+
# Process collected elements
|
|
45
|
+
for child_tag in list(dictionary.keys()):
|
|
46
|
+
unpacked_element_list = []
|
|
47
|
+
for child_element in dictionary[child_tag]:
|
|
48
|
+
unpacked_element = child_element.attrib
|
|
49
|
+
|
|
50
|
+
# Recurse if the element has children
|
|
51
|
+
if list(child_element):
|
|
52
|
+
children_dict = _parse_xml_to_dict(child_element)
|
|
53
|
+
if children_dict:
|
|
54
|
+
unpacked_element.update(children_dict)
|
|
55
|
+
|
|
56
|
+
# If we got something meaningful, keep it
|
|
57
|
+
if unpacked_element:
|
|
58
|
+
unpacked_element_list.append(unpacked_element)
|
|
59
|
+
|
|
60
|
+
dictionary[child_tag] = unpacked_element_list
|
|
61
|
+
|
|
62
|
+
return dict(dictionary)
|
|
63
|
+
|
|
64
|
+
def _recursive_unpack(nested_row):
|
|
65
|
+
"""Recursively unpack a nested dict with list-of-dict values into flat records."""
|
|
66
|
+
base_record = {}
|
|
67
|
+
unpacked_lists = []
|
|
68
|
+
|
|
69
|
+
for key, value in nested_row.items():
|
|
70
|
+
if isinstance(value, list) and value and all(isinstance(item, dict) for item in value):
|
|
71
|
+
# Recursively unpack each item in the list
|
|
72
|
+
new_list = []
|
|
73
|
+
for item in value:
|
|
74
|
+
flattened_items = _recursive_unpack(item)
|
|
75
|
+
for flat in flattened_items:
|
|
76
|
+
# Add prefix to each key
|
|
77
|
+
prefixed = {f"{key}.{k}": v for k, v in flat.items()}
|
|
78
|
+
new_list.append(prefixed)
|
|
79
|
+
unpacked_lists.append(new_list)
|
|
80
|
+
elif isinstance(value, dict):
|
|
81
|
+
# Recursively flatten the nested dict
|
|
82
|
+
nested = _recursive_unpack(value)
|
|
83
|
+
for flat in nested:
|
|
84
|
+
base_record.update({f"{key}.{k}": v for k, v in flat.items()})
|
|
85
|
+
elif value:
|
|
86
|
+
base_record[key] = value
|
|
87
|
+
|
|
88
|
+
# If there are unpacked lists, compute cartesian product
|
|
89
|
+
if unpacked_lists:
|
|
90
|
+
result = []
|
|
91
|
+
for combo in itertools.product(*unpacked_lists):
|
|
92
|
+
combined = dict(base_record) # copy base record
|
|
93
|
+
for item in combo:
|
|
94
|
+
combined.update(item)
|
|
95
|
+
result.append(combined)
|
|
96
|
+
return result
|
|
97
|
+
else:
|
|
98
|
+
return [base_record]
|
|
99
|
+
|
|
File without changes
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
SCHEMA = {
|
|
2
|
+
"Element": {
|
|
3
|
+
"R1ElementID": {"not_null": True, "unique": True},
|
|
4
|
+
"R1Element": {"not_null": True, "unique": True},
|
|
5
|
+
},
|
|
6
|
+
"ElementInstances": {
|
|
7
|
+
"R1InstanceID": {"not_null": True, "unique": True},
|
|
8
|
+
"R1Instance": {"not_null": True, "unique": False},
|
|
9
|
+
"R1InstanceDescription": {"not_null": False, "unique": False, "default": None},
|
|
10
|
+
"R1InstanceRichText": {"not_null": False, "unique": False, "default": None},
|
|
11
|
+
},
|
|
12
|
+
"Properties": {
|
|
13
|
+
"Property": {"not_null": True, "unique": True},
|
|
14
|
+
},
|
|
15
|
+
"PropertyInstances": {
|
|
16
|
+
"R1InstanceID": {"not_null": True, "unique": False},
|
|
17
|
+
"Property": {"not_null": True, "unique": False},
|
|
18
|
+
"PropertyInstance": {"not_null": False, "unique": False},
|
|
19
|
+
},
|
|
20
|
+
"Relations": {
|
|
21
|
+
"RelationID": {"not_null": True, "unique": False},
|
|
22
|
+
"Relation": {"not_null": True, "unique": False},
|
|
23
|
+
"Cardinality": {"not_null": False, "unique": False},
|
|
24
|
+
"R1Element": {"not_null": True, "unique": False},
|
|
25
|
+
"R2ElementID": {"not_null": True, "unique": False},
|
|
26
|
+
"R2Element": {"not_null": True, "unique": False},
|
|
27
|
+
"ChildR2ElementID": {"not_null": False, "unique": False, "default": None},
|
|
28
|
+
"ChildR2Element": {"not_null": False, "unique": False, "default": None},
|
|
29
|
+
},
|
|
30
|
+
"RelationInstances": {
|
|
31
|
+
"RelationID": {"not_null": True, "unique": False},
|
|
32
|
+
"Cardinality": {"not_null": False, "unique": False},
|
|
33
|
+
"R1InstanceID": {"not_null": True, "unique": False},
|
|
34
|
+
"R2Element": {"not_null": True, "unique": False},
|
|
35
|
+
"R2InstanceID": {"not_null": True, "unique": False},
|
|
36
|
+
"R2Instance": {"not_null": True, "unique": False},
|
|
37
|
+
},
|
|
38
|
+
}
|
|
@@ -0,0 +1,420 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import re
|
|
3
|
+
import unicodedata
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
logger = logging.getLogger(__name__)
|
|
9
|
+
|
|
10
|
+
R1ELEMENT_COL = "R1Element"
|
|
11
|
+
|
|
12
|
+
R1INSTANCE_COL = "R1Instance"
|
|
13
|
+
R1INSTANCEID_COL = "R1InstanceID"
|
|
14
|
+
|
|
15
|
+
PROPERTY_COL = "Property"
|
|
16
|
+
PROPERTYINSTANCE_COL = "PropertyInstance"
|
|
17
|
+
|
|
18
|
+
RELATION_COL = "Relation"
|
|
19
|
+
RELATIONID_COL = "RelationID"
|
|
20
|
+
CARDINALITY_COL = "Cardinality"
|
|
21
|
+
|
|
22
|
+
R2ELEMENT_COL = "R2Element"
|
|
23
|
+
R2ELEMENTID_COL = "R2ElementID"
|
|
24
|
+
|
|
25
|
+
CHILDR2ELEMENT_COL = "ChildR2Element"
|
|
26
|
+
CHILDR2ELEMENTID_COL = "ChildR2ElementID"
|
|
27
|
+
|
|
28
|
+
R2INSTANCE_COL = "R2Instance"
|
|
29
|
+
R2INSTANCEID_COL = "R2InstanceID"
|
|
30
|
+
|
|
31
|
+
DEFAULT_COLUMN_MAP = {
|
|
32
|
+
R1INSTANCEID_COL: "guid",
|
|
33
|
+
R1INSTANCE_COL: "naam",
|
|
34
|
+
"R1InstanceDescription": "description",
|
|
35
|
+
"R1InstanceRichText": "rich_text",
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def create_element_tables(
|
|
40
|
+
tables: dict[str, pd.DataFrame],
|
|
41
|
+
column_map: dict[str, str] = DEFAULT_COLUMN_MAP,
|
|
42
|
+
inline_relations: list[str] | None = None,
|
|
43
|
+
) -> dict[str, pd.DataFrame]:
|
|
44
|
+
"""
|
|
45
|
+
Create the element table and associated link tables for an element.
|
|
46
|
+
|
|
47
|
+
The element table is built by joining:
|
|
48
|
+
- property values
|
|
49
|
+
- to-one relation references
|
|
50
|
+
- directly resolved relation values
|
|
51
|
+
|
|
52
|
+
Regular to-one relations are represented as <element>_guid
|
|
53
|
+
columns containing the related R2 instance identifier. Relations
|
|
54
|
+
listed in inline_relations are represented directly by their
|
|
55
|
+
related R2 instance value.
|
|
56
|
+
|
|
57
|
+
To-many relations are returned as separate link tables.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
tables: Dict with the 6 raw Relatics report tables ("Element",
|
|
61
|
+
"ElementInstances", "Properties", "PropertyInstances",
|
|
62
|
+
"Relations", "RelationInstances").
|
|
63
|
+
column_map: Mapping used to rename the final element table's
|
|
64
|
+
columns.
|
|
65
|
+
inline_relations: Relation names of Relations to R2 Elements
|
|
66
|
+
whose values will be materialized directly in the
|
|
67
|
+
resulting element tables.
|
|
68
|
+
|
|
69
|
+
Returns:
|
|
70
|
+
Mapping of table names to DataFrames containing the element table
|
|
71
|
+
and any associated link tables.
|
|
72
|
+
"""
|
|
73
|
+
element_df = tables["Element"].copy()
|
|
74
|
+
element_instances_df = tables["ElementInstances"].copy()
|
|
75
|
+
properties_df = tables["Properties"].copy()
|
|
76
|
+
property_instances_df = tables["PropertyInstances"].copy()
|
|
77
|
+
relations_df = tables["Relations"].copy()
|
|
78
|
+
relation_instances_df = tables["RelationInstances"].copy()
|
|
79
|
+
|
|
80
|
+
r1_element = _normalize_value(element_df[R1ELEMENT_COL][0])
|
|
81
|
+
|
|
82
|
+
relations_df, relation_instances_df = _transform_relations_table(
|
|
83
|
+
relations_df=relations_df, relation_instances_df=relation_instances_df
|
|
84
|
+
)
|
|
85
|
+
property_table = _create_property_table(
|
|
86
|
+
properties_df=properties_df, property_instances_df=property_instances_df
|
|
87
|
+
)
|
|
88
|
+
to_one_relations_table = _create_to_one_relations_table(
|
|
89
|
+
relations_df=relations_df,
|
|
90
|
+
relation_instances_df=relation_instances_df,
|
|
91
|
+
inline_relations=inline_relations,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
sub_tables: list = [
|
|
95
|
+
df
|
|
96
|
+
for df in (
|
|
97
|
+
property_table,
|
|
98
|
+
to_one_relations_table,
|
|
99
|
+
)
|
|
100
|
+
if not df.columns.empty
|
|
101
|
+
]
|
|
102
|
+
|
|
103
|
+
element_table = (
|
|
104
|
+
element_instances_df.set_index(R1INSTANCEID_COL)
|
|
105
|
+
.join(sub_tables, how="left")
|
|
106
|
+
.reset_index()
|
|
107
|
+
)
|
|
108
|
+
element_table = element_table.rename(columns=column_map)
|
|
109
|
+
|
|
110
|
+
link_tables = _create_link_tables(
|
|
111
|
+
r1_element=r1_element,
|
|
112
|
+
relations_df=relations_df,
|
|
113
|
+
relation_instances_df=relation_instances_df,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
return {
|
|
117
|
+
f"raw_relatics__{r1_element}": element_table,
|
|
118
|
+
**link_tables,
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _transform_relations_table(
|
|
123
|
+
relations_df: pd.DataFrame, relation_instances_df: pd.DataFrame
|
|
124
|
+
) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
125
|
+
"""
|
|
126
|
+
Normalize and disambiguate relation target element names.
|
|
127
|
+
|
|
128
|
+
- Coalesces child R2 element data into the primary R2 columns.
|
|
129
|
+
- Normalizes element and relation names.
|
|
130
|
+
- Validates that each Relation/R2Element combination is unique.
|
|
131
|
+
- Renames duplicate and self-referencing R2Elements to ensure unique
|
|
132
|
+
SQL-safe column names.
|
|
133
|
+
- Applies the same renaming to relation instances.
|
|
134
|
+
"""
|
|
135
|
+
if (
|
|
136
|
+
CHILDR2ELEMENT_COL in relations_df.columns
|
|
137
|
+
and CHILDR2ELEMENTID_COL in relations_df.columns
|
|
138
|
+
):
|
|
139
|
+
relations_df[R2ELEMENT_COL] = (
|
|
140
|
+
relations_df[CHILDR2ELEMENT_COL]
|
|
141
|
+
.replace("", None)
|
|
142
|
+
.combine_first(relations_df[R2ELEMENT_COL])
|
|
143
|
+
)
|
|
144
|
+
relations_df[R2ELEMENTID_COL] = (
|
|
145
|
+
relations_df[CHILDR2ELEMENTID_COL]
|
|
146
|
+
.replace("", None)
|
|
147
|
+
.combine_first(relations_df[R2ELEMENTID_COL])
|
|
148
|
+
)
|
|
149
|
+
relations_df = relations_df.drop(
|
|
150
|
+
[CHILDR2ELEMENT_COL, CHILDR2ELEMENTID_COL], axis=1
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
relations_df[[R1ELEMENT_COL, RELATION_COL, R2ELEMENT_COL]] = relations_df[
|
|
154
|
+
[R1ELEMENT_COL, RELATION_COL, R2ELEMENT_COL]
|
|
155
|
+
].map(_normalize_value)
|
|
156
|
+
relation_instances_df[R2ELEMENT_COL] = relation_instances_df[R2ELEMENT_COL].apply(
|
|
157
|
+
_normalize_value
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
duplicates_mask = relations_df.duplicated(
|
|
161
|
+
subset=[RELATION_COL, R2ELEMENT_COL], keep=False
|
|
162
|
+
)
|
|
163
|
+
if duplicates_mask.any():
|
|
164
|
+
logger.error("Duplicate Relation/R2Element combinations found.")
|
|
165
|
+
logger.debug(relations_df[duplicates_mask].to_dict())
|
|
166
|
+
raise RuntimeError(
|
|
167
|
+
"relations_df cannot contain duplicated Relation + R2Element pairs."
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
rename_rows = relations_df.loc[
|
|
171
|
+
relations_df[R2ELEMENT_COL].duplicated(keep=False)
|
|
172
|
+
| (relations_df[R1ELEMENT_COL] == relations_df[R2ELEMENT_COL])
|
|
173
|
+
].copy()
|
|
174
|
+
|
|
175
|
+
rename_rows[R2ELEMENT_COL] = (
|
|
176
|
+
rename_rows[RELATION_COL] + "_" + rename_rows[R2ELEMENT_COL]
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
rename_map = rename_rows.set_index(RELATIONID_COL)[R2ELEMENT_COL].to_dict()
|
|
180
|
+
|
|
181
|
+
relations_df[R2ELEMENT_COL] = (
|
|
182
|
+
relations_df[RELATIONID_COL].map(rename_map).fillna(relations_df[R2ELEMENT_COL])
|
|
183
|
+
)
|
|
184
|
+
relation_instances_df[R2ELEMENT_COL] = (
|
|
185
|
+
relation_instances_df[RELATIONID_COL]
|
|
186
|
+
.map(rename_map)
|
|
187
|
+
.fillna(relation_instances_df[R2ELEMENT_COL])
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
return relations_df, relation_instances_df
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _create_property_table(
|
|
194
|
+
properties_df: pd.DataFrame, property_instances_df: pd.DataFrame
|
|
195
|
+
) -> pd.DataFrame:
|
|
196
|
+
"""
|
|
197
|
+
Create a property table indexed by R1 instance.
|
|
198
|
+
|
|
199
|
+
Property names are normalized and become columns.
|
|
200
|
+
A column is included for every property defined in properties_df,
|
|
201
|
+
even if no property instances exist.
|
|
202
|
+
"""
|
|
203
|
+
property_instances_df = property_instances_df.copy()
|
|
204
|
+
|
|
205
|
+
property_columns = (
|
|
206
|
+
properties_df[PROPERTY_COL].apply(_normalize_value).unique().tolist()
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
property_instances_df[PROPERTY_COL] = property_instances_df[PROPERTY_COL].apply(
|
|
210
|
+
_normalize_value
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
try:
|
|
214
|
+
return (
|
|
215
|
+
property_instances_df.pivot(
|
|
216
|
+
index=R1INSTANCEID_COL,
|
|
217
|
+
columns=PROPERTY_COL,
|
|
218
|
+
values=PROPERTYINSTANCE_COL,
|
|
219
|
+
)
|
|
220
|
+
.reindex(columns=property_columns)
|
|
221
|
+
.rename_axis(columns=None)
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
except ValueError:
|
|
225
|
+
logger.exception(
|
|
226
|
+
"Failed to pivot property instances. "
|
|
227
|
+
"Normalization may have created duplicate property names."
|
|
228
|
+
)
|
|
229
|
+
raise
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _create_to_one_relations_table(
|
|
233
|
+
relations_df: pd.DataFrame,
|
|
234
|
+
relation_instances_df: pd.DataFrame,
|
|
235
|
+
inline_relations: list[str] | None = None,
|
|
236
|
+
) -> pd.DataFrame:
|
|
237
|
+
"""
|
|
238
|
+
Create a table containing to-one relation data.
|
|
239
|
+
|
|
240
|
+
For regular to-one relations, each referenced element becomes a
|
|
241
|
+
<element>_guid column containing the related R2 instance
|
|
242
|
+
id.
|
|
243
|
+
|
|
244
|
+
Relations listed in inline_relations are represented by their
|
|
245
|
+
resolved value instead of an R2 instance id.
|
|
246
|
+
"""
|
|
247
|
+
inline_relations = inline_relations or []
|
|
248
|
+
if inline_relations:
|
|
249
|
+
inline_relations = [_normalize_value(value) for value in inline_relations]
|
|
250
|
+
|
|
251
|
+
all_to_one_relations_df = _filter_cardinality(
|
|
252
|
+
df=relations_df,
|
|
253
|
+
cardinality="one",
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
to_one_relations_df = all_to_one_relations_df[
|
|
257
|
+
~all_to_one_relations_df[RELATION_COL].isin(inline_relations)
|
|
258
|
+
].copy()
|
|
259
|
+
|
|
260
|
+
to_one_relation_columns = [
|
|
261
|
+
f"{element}_guid" for element in to_one_relations_df[R2ELEMENT_COL]
|
|
262
|
+
]
|
|
263
|
+
|
|
264
|
+
to_one_relation_instances_df = relation_instances_df[
|
|
265
|
+
relation_instances_df[RELATIONID_COL].isin(to_one_relations_df[RELATIONID_COL])
|
|
266
|
+
].copy()
|
|
267
|
+
|
|
268
|
+
to_one_relation_instances_df[R2ELEMENT_COL] = (
|
|
269
|
+
to_one_relation_instances_df[R2ELEMENT_COL].astype("string") + "_guid"
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
to_one_relations_table = (
|
|
273
|
+
to_one_relation_instances_df.pivot(
|
|
274
|
+
index=R1INSTANCEID_COL,
|
|
275
|
+
columns=R2ELEMENT_COL,
|
|
276
|
+
values=R2INSTANCEID_COL,
|
|
277
|
+
)
|
|
278
|
+
.reindex(columns=to_one_relation_columns)
|
|
279
|
+
.rename_axis(columns=None)
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
if not inline_relations:
|
|
283
|
+
return to_one_relations_table
|
|
284
|
+
|
|
285
|
+
inline_relations_df = all_to_one_relations_df[
|
|
286
|
+
all_to_one_relations_df[RELATION_COL].isin(inline_relations)
|
|
287
|
+
].copy()
|
|
288
|
+
|
|
289
|
+
inline_relations_table = _create_inline_relations_table(
|
|
290
|
+
inline_relations_df=inline_relations_df,
|
|
291
|
+
relation_instances_df=relation_instances_df,
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
return pd.concat(
|
|
295
|
+
[inline_relations_table, to_one_relations_table],
|
|
296
|
+
axis=1,
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _create_inline_relations_table(
|
|
301
|
+
inline_relations_df: pd.DataFrame,
|
|
302
|
+
relation_instances_df: pd.DataFrame,
|
|
303
|
+
) -> pd.DataFrame:
|
|
304
|
+
"""
|
|
305
|
+
Create a table containing directly resolved property values.
|
|
306
|
+
|
|
307
|
+
Each referenced property element becomes a column containing the
|
|
308
|
+
related R2 instance value.
|
|
309
|
+
"""
|
|
310
|
+
inline_relation_instances_df = relation_instances_df[
|
|
311
|
+
relation_instances_df[RELATIONID_COL].isin(inline_relations_df[RELATIONID_COL])
|
|
312
|
+
].copy()
|
|
313
|
+
|
|
314
|
+
inline_relation_columns = inline_relations_df[R2ELEMENT_COL].tolist()
|
|
315
|
+
|
|
316
|
+
return (
|
|
317
|
+
inline_relation_instances_df.pivot(
|
|
318
|
+
index=R1INSTANCEID_COL,
|
|
319
|
+
columns=R2ELEMENT_COL,
|
|
320
|
+
values=R2INSTANCE_COL,
|
|
321
|
+
)
|
|
322
|
+
.reindex(columns=inline_relation_columns)
|
|
323
|
+
.rename_axis(columns=None)
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _create_link_tables(
|
|
328
|
+
r1_element: str, relations_df: pd.DataFrame, relation_instances_df: pd.DataFrame
|
|
329
|
+
) -> dict[str, pd.DataFrame]:
|
|
330
|
+
"""
|
|
331
|
+
Create link tables for all to-many relations.
|
|
332
|
+
|
|
333
|
+
Each R2Element receives its own link table containing the R1 and R2
|
|
334
|
+
instance identifiers. Table and column names are normalized to make
|
|
335
|
+
them SQL safe.
|
|
336
|
+
"""
|
|
337
|
+
to_many_relations_df = _filter_cardinality(relations_df, "many")
|
|
338
|
+
to_many_relation_instances_df = _filter_cardinality(relation_instances_df, "many")
|
|
339
|
+
|
|
340
|
+
link_tables: dict[str, pd.DataFrame] = {}
|
|
341
|
+
|
|
342
|
+
for r2_element in to_many_relations_df[R2ELEMENT_COL]:
|
|
343
|
+
table_name = f"raw_relatics__{r1_element}_{r2_element}"
|
|
344
|
+
|
|
345
|
+
mask = to_many_relation_instances_df[R2ELEMENT_COL] == str(r2_element)
|
|
346
|
+
|
|
347
|
+
link_table = to_many_relation_instances_df.loc[
|
|
348
|
+
mask, [R1INSTANCEID_COL, R2INSTANCEID_COL]
|
|
349
|
+
].reset_index(drop=True)
|
|
350
|
+
|
|
351
|
+
link_tables[table_name] = link_table.rename(
|
|
352
|
+
columns={
|
|
353
|
+
R1INSTANCEID_COL: f"{r1_element}_guid",
|
|
354
|
+
R2INSTANCEID_COL: f"{r2_element}_guid",
|
|
355
|
+
}
|
|
356
|
+
)
|
|
357
|
+
|
|
358
|
+
return link_tables
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _filter_cardinality(
|
|
362
|
+
df: pd.DataFrame,
|
|
363
|
+
cardinality: Literal["many", "one"],
|
|
364
|
+
) -> pd.DataFrame:
|
|
365
|
+
"""
|
|
366
|
+
Filter rows by relation cardinality.
|
|
367
|
+
|
|
368
|
+
Cardinality values are normalized to either ':1' or ':n'
|
|
369
|
+
before filtering.
|
|
370
|
+
"""
|
|
371
|
+
cardinality_mask = df[CARDINALITY_COL].map(
|
|
372
|
+
lambda x: ":n" if "n" in str(x).split(":")[-1] else ":1"
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
if cardinality == "many":
|
|
376
|
+
return df[cardinality_mask == ":n"]
|
|
377
|
+
|
|
378
|
+
if cardinality == "one":
|
|
379
|
+
return df[cardinality_mask == ":1"]
|
|
380
|
+
|
|
381
|
+
raise ValueError(f"Invalid cardinality '{cardinality}'. Expected 'one' or 'many'.")
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _normalize_value(val: str, max_length: int = 63) -> str:
|
|
385
|
+
"""
|
|
386
|
+
Normalize text for use as SQL table and column names.
|
|
387
|
+
"""
|
|
388
|
+
if pd.isna(val) or val is None:
|
|
389
|
+
return ""
|
|
390
|
+
|
|
391
|
+
val = str(val)
|
|
392
|
+
|
|
393
|
+
# Replace special characters
|
|
394
|
+
val = val.replace("&", "_en_").replace("€", "_euro_").replace("+", "_plus_")
|
|
395
|
+
|
|
396
|
+
# Unicode -> ASCII
|
|
397
|
+
val = unicodedata.normalize("NFKD", val)
|
|
398
|
+
val = val.encode("ascii", "ignore").decode("ascii")
|
|
399
|
+
|
|
400
|
+
# Lowercase
|
|
401
|
+
val = val.lower()
|
|
402
|
+
|
|
403
|
+
# Replace invalid characters
|
|
404
|
+
val = re.sub(r"[^a-z0-9_]", "_", val)
|
|
405
|
+
|
|
406
|
+
# Collapse underscores
|
|
407
|
+
val = re.sub(r"_+", "_", val)
|
|
408
|
+
|
|
409
|
+
# Strip leading/trailing underscores
|
|
410
|
+
val = val.strip("_")
|
|
411
|
+
|
|
412
|
+
if not val:
|
|
413
|
+
return ""
|
|
414
|
+
|
|
415
|
+
# Prevent leading digit
|
|
416
|
+
if val[0].isdigit():
|
|
417
|
+
val = f"no_num_{val}"
|
|
418
|
+
|
|
419
|
+
# Trim to PostgreSQL identifier length
|
|
420
|
+
return val[:max_length]
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
from collections import defaultdict
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from relatics_toolkit.processing.schema import SCHEMA
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def normalize_tables(
|
|
9
|
+
tables: dict[str, pd.DataFrame], schema: dict[str, dict[str, dict]] = SCHEMA
|
|
10
|
+
) -> dict[str, pd.DataFrame]:
|
|
11
|
+
"""
|
|
12
|
+
Add missing optional columns to tables according to the schema.
|
|
13
|
+
|
|
14
|
+
Columns marked 'not_null=False' are added when missing and
|
|
15
|
+
initialized with their configured default value. The resulting
|
|
16
|
+
tables are validated against the schema before being returned.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
normalized_tables: dict[str, pd.DataFrame] = {}
|
|
20
|
+
|
|
21
|
+
for table_name, table_schema in schema.items():
|
|
22
|
+
table = tables.get(table_name)
|
|
23
|
+
|
|
24
|
+
if table is not None:
|
|
25
|
+
df = table.copy()
|
|
26
|
+
|
|
27
|
+
for column_name, column_rules in table_schema.items():
|
|
28
|
+
if column_name not in df.columns:
|
|
29
|
+
df[column_name] = column_rules.get("default")
|
|
30
|
+
|
|
31
|
+
df = df.dropna(how="all").reset_index(drop=True)
|
|
32
|
+
|
|
33
|
+
normalized_tables[table_name] = df
|
|
34
|
+
|
|
35
|
+
return normalized_tables
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def validate_schema(
|
|
39
|
+
tables: dict[str, pd.DataFrame],
|
|
40
|
+
schema: dict[str, dict[str, dict]],
|
|
41
|
+
) -> None:
|
|
42
|
+
"""
|
|
43
|
+
Validate tables against the configured schema.
|
|
44
|
+
|
|
45
|
+
Checks:
|
|
46
|
+
- Required tables exist.
|
|
47
|
+
- Required columns exist.
|
|
48
|
+
- Columns marked 'not_null' contain no null values.
|
|
49
|
+
- Columns marked 'unique' contain no duplicate non-null values.
|
|
50
|
+
"""
|
|
51
|
+
errors = defaultdict(list)
|
|
52
|
+
|
|
53
|
+
for table_name, table_schema in schema.items():
|
|
54
|
+
if table_name not in tables:
|
|
55
|
+
errors["missing_tables"].append({"table": table_name})
|
|
56
|
+
continue
|
|
57
|
+
|
|
58
|
+
df = tables[table_name]
|
|
59
|
+
|
|
60
|
+
missing_columns = [
|
|
61
|
+
column for column in table_schema if column not in df.columns
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
if missing_columns:
|
|
65
|
+
errors["missing_columns"].append(
|
|
66
|
+
{
|
|
67
|
+
"table": table_name,
|
|
68
|
+
"columns": missing_columns,
|
|
69
|
+
}
|
|
70
|
+
)
|
|
71
|
+
continue
|
|
72
|
+
|
|
73
|
+
null_columns = [
|
|
74
|
+
column
|
|
75
|
+
for column, column_rules in table_schema.items()
|
|
76
|
+
if column_rules.get("not_null") and df[column].isna().any()
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
if null_columns:
|
|
80
|
+
errors["null_values"].append(
|
|
81
|
+
{
|
|
82
|
+
"table": table_name,
|
|
83
|
+
"columns": null_columns,
|
|
84
|
+
}
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
duplicate_columns = [
|
|
88
|
+
column
|
|
89
|
+
for column, column_rules in table_schema.items()
|
|
90
|
+
if column_rules.get("unique") and df[column].dropna().duplicated().any()
|
|
91
|
+
]
|
|
92
|
+
|
|
93
|
+
if duplicate_columns:
|
|
94
|
+
errors["duplicate_values"].append(
|
|
95
|
+
{
|
|
96
|
+
"table": table_name,
|
|
97
|
+
"columns": duplicate_columns,
|
|
98
|
+
}
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
if errors:
|
|
102
|
+
raise RuntimeError(format_validation_errors(errors))
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def format_validation_errors(errors: dict[str, list[dict]]) -> str:
|
|
106
|
+
sections = []
|
|
107
|
+
|
|
108
|
+
for category, items in errors.items():
|
|
109
|
+
lines = []
|
|
110
|
+
|
|
111
|
+
for item in items:
|
|
112
|
+
if category == "missing_tables":
|
|
113
|
+
lines.append(item["table"])
|
|
114
|
+
|
|
115
|
+
else:
|
|
116
|
+
lines.append(f"{item['table']}: {', '.join(item['columns'])}")
|
|
117
|
+
|
|
118
|
+
sections.append(f"{category}:\n- " + "\n- ".join(lines))
|
|
119
|
+
|
|
120
|
+
return "Schema validation failed.\n\n" + "\n\n".join(sections)
|