corrosions 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- corrosions-0.1.0/PKG-INFO +21 -0
- corrosions-0.1.0/README.md +0 -0
- corrosions-0.1.0/pyproject.toml +57 -0
- corrosions-0.1.0/src/corrosions/__init__.py +28 -0
- corrosions-0.1.0/src/corrosions/acvg_dcvg.py +211 -0
- corrosions-0.1.0/src/corrosions/cips.py +339 -0
- corrosions-0.1.0/src/corrosions/const.py +16 -0
- corrosions-0.1.0/src/corrosions/pcm.py +202 -0
- corrosions-0.1.0/src/corrosions/sync.py +463 -0
- corrosions-0.1.0/src/corrosions/utils.py +158 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: corrosions
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Corrosion calculation created by Capella Global Innovation
|
|
5
|
+
Author: Martanto
|
|
6
|
+
Author-email: Martanto <martanto@live.com>
|
|
7
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Requires-Dist: black>=25.9.0
|
|
12
|
+
Requires-Dist: isort>=7.0.0
|
|
13
|
+
Requires-Dist: matplotlib>=3.10.7
|
|
14
|
+
Requires-Dist: openpyxl>=3.1.5
|
|
15
|
+
Requires-Dist: pandas>=2.3.3
|
|
16
|
+
Requires-Dist: pip>=25.3
|
|
17
|
+
Requires-Dist: python-slugify>=8.0.4
|
|
18
|
+
Requires-Dist: xlsxwriter>=3.2.9
|
|
19
|
+
Requires-Python: >=3.11
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
File without changes
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.9.8,<0.10.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "corrosions"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
authors = [
|
|
9
|
+
{name = "Martanto", email = "martanto@live.com"},
|
|
10
|
+
]
|
|
11
|
+
description = "Corrosion calculation created by Capella Global Innovation"
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
requires-python = ">=3.11"
|
|
14
|
+
dependencies = [
|
|
15
|
+
"black>=25.9.0",
|
|
16
|
+
"isort>=7.0.0",
|
|
17
|
+
"matplotlib>=3.10.7",
|
|
18
|
+
"openpyxl>=3.1.5",
|
|
19
|
+
"pandas>=2.3.3",
|
|
20
|
+
"pip>=25.3",
|
|
21
|
+
"python-slugify>=8.0.4",
|
|
22
|
+
"xlsxwriter>=3.2.9",
|
|
23
|
+
]
|
|
24
|
+
classifiers = [
|
|
25
|
+
"License :: OSI Approved :: MIT License",
|
|
26
|
+
"Operating System :: OS Independent",
|
|
27
|
+
"Programming Language :: Python :: 3.11",
|
|
28
|
+
"Programming Language :: Python :: 3.12",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[tool.black]
|
|
32
|
+
line-length = 79
|
|
33
|
+
target-version = ['py311', 'py312']
|
|
34
|
+
include = '\.pyi?$'
|
|
35
|
+
exclude = '''
|
|
36
|
+
/(
|
|
37
|
+
\.eggs
|
|
38
|
+
| \.git
|
|
39
|
+
| \.hg
|
|
40
|
+
| \.mypy_cache
|
|
41
|
+
| \.tox
|
|
42
|
+
| \.venv
|
|
43
|
+
| \.uv_cache
|
|
44
|
+
| _build
|
|
45
|
+
| buck-out
|
|
46
|
+
| build
|
|
47
|
+
| dist
|
|
48
|
+
)/
|
|
49
|
+
'''
|
|
50
|
+
|
|
51
|
+
[tool.isort]
|
|
52
|
+
profile = "black"
|
|
53
|
+
line_length = 79
|
|
54
|
+
skip = [
|
|
55
|
+
".venv",
|
|
56
|
+
".uv_cache",
|
|
57
|
+
]
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
|
|
4
|
+
# Third party imports
|
|
5
|
+
from importlib.metadata import version
|
|
6
|
+
from .cips import CIPS
|
|
7
|
+
from .pcm import PCM
|
|
8
|
+
from .sync import Sync
|
|
9
|
+
from .acvg_dcvg import AcvgDcvg
|
|
10
|
+
|
|
11
|
+
__version__ = version("corrosions")
|
|
12
|
+
__author__ = "Martanto"
|
|
13
|
+
__author_email__ = "martanto@live.com"
|
|
14
|
+
__license__ = "MIT"
|
|
15
|
+
__copyright__ = "Copyright (c) 2025, Martanto"
|
|
16
|
+
__url__ = "https://github.com/martanto/corrosions"
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"__version__",
|
|
20
|
+
"__author__",
|
|
21
|
+
"__author_email__",
|
|
22
|
+
"__license__",
|
|
23
|
+
"__copyright__",
|
|
24
|
+
"AcvgDcvg",
|
|
25
|
+
"CIPS",
|
|
26
|
+
"PCM",
|
|
27
|
+
"Sync",
|
|
28
|
+
]
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
import numpy as np
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from .pcm import PCM
|
|
6
|
+
from .const import *
|
|
7
|
+
from .utils import *
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class AcvgDcvg(PCM):
|
|
11
|
+
def __init__(
|
|
12
|
+
self,
|
|
13
|
+
file_or_dir: str,
|
|
14
|
+
overwrite: bool = False,
|
|
15
|
+
verbose: bool = False,
|
|
16
|
+
):
|
|
17
|
+
super().__init__(file_or_dir, overwrite, verbose)
|
|
18
|
+
|
|
19
|
+
self.prefix = "acvg_dcvg"
|
|
20
|
+
|
|
21
|
+
self.COLUMNS_VALIDATED = [
|
|
22
|
+
"segment_code",
|
|
23
|
+
"diameter",
|
|
24
|
+
"anomaly_location",
|
|
25
|
+
"surface_condition",
|
|
26
|
+
"drop_pcm",
|
|
27
|
+
"on_potential",
|
|
28
|
+
"off_potential",
|
|
29
|
+
"survey_dcvg",
|
|
30
|
+
"survey_acvg",
|
|
31
|
+
"latitude",
|
|
32
|
+
"longitude",
|
|
33
|
+
"ir_drop",
|
|
34
|
+
"pipe_depth",
|
|
35
|
+
"result_acvg",
|
|
36
|
+
"protection",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
self.UNIQUE_COLUMNS = ["latitude", "longitude"]
|
|
40
|
+
|
|
41
|
+
self.check_sequential_file = False
|
|
42
|
+
self.excel_dir = ACVG_DCVG_EXCEL_DIR
|
|
43
|
+
self.json_dir = ACVG_DCVG_JSON_DIR
|
|
44
|
+
|
|
45
|
+
@staticmethod
|
|
46
|
+
def closest_distance_pcm(
|
|
47
|
+
df_acvg_dcvg: pd.DataFrame, df_pcm: pd.DataFrame
|
|
48
|
+
) -> pd.DataFrame:
|
|
49
|
+
acvg_dvcg_distance = []
|
|
50
|
+
closest_pcm_distance = []
|
|
51
|
+
closest_pcm_index = []
|
|
52
|
+
closest_pcm_latitude = []
|
|
53
|
+
closest_pcm_longitude = []
|
|
54
|
+
closest_pcm_real_distance = []
|
|
55
|
+
|
|
56
|
+
pcm_coordinates = df_pcm[
|
|
57
|
+
["Int GPS Latitude", "Int GPS Longitude", "Real Distance"]
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
for index, row_acvg_dcvg in df_acvg_dcvg.iterrows():
|
|
61
|
+
distances = []
|
|
62
|
+
lat2 = row_acvg_dcvg["latitude"]
|
|
63
|
+
lon2 = row_acvg_dcvg["longitude"]
|
|
64
|
+
|
|
65
|
+
for _, pcm in pcm_coordinates.iterrows():
|
|
66
|
+
lat1 = pcm["Int GPS Latitude"]
|
|
67
|
+
lon1 = pcm["Int GPS Longitude"]
|
|
68
|
+
distance = calculate_distance(lat1, lon1, lat2, lon2)
|
|
69
|
+
distances.append(distance)
|
|
70
|
+
|
|
71
|
+
np_distances = np.array(distances)
|
|
72
|
+
distance_min = np.min(np_distances)
|
|
73
|
+
|
|
74
|
+
acvg_dvcg_distance.append(
|
|
75
|
+
pcm_coordinates.iloc[np_distances.argmin()]["Real Distance"]
|
|
76
|
+
+ distance_min
|
|
77
|
+
)
|
|
78
|
+
closest_pcm_distance.append(0 - distance_min)
|
|
79
|
+
closest_pcm_index.append(np_distances.argmin())
|
|
80
|
+
closest_pcm_latitude.append(
|
|
81
|
+
pcm_coordinates.iloc[np_distances.argmin()]["Int GPS Latitude"]
|
|
82
|
+
)
|
|
83
|
+
closest_pcm_longitude.append(
|
|
84
|
+
pcm_coordinates.iloc[np_distances.argmin()][
|
|
85
|
+
"Int GPS Longitude"
|
|
86
|
+
]
|
|
87
|
+
)
|
|
88
|
+
closest_pcm_real_distance.append(
|
|
89
|
+
pcm_coordinates.iloc[np_distances.argmin()]["Real Distance"]
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
df_acvg_dcvg["real_distance"] = acvg_dvcg_distance
|
|
93
|
+
df_acvg_dcvg["closest_pcm_distance"] = closest_pcm_distance
|
|
94
|
+
df_acvg_dcvg["closest_pcm_real_distance"] = closest_pcm_real_distance
|
|
95
|
+
df_acvg_dcvg["closest_pcm_index"] = closest_pcm_index
|
|
96
|
+
df_acvg_dcvg["closest_pcm_latitude"] = closest_pcm_latitude
|
|
97
|
+
df_acvg_dcvg["closest_pcm_longitude"] = closest_pcm_longitude
|
|
98
|
+
|
|
99
|
+
df_acvg_dcvg.sort_values("real_distance", ascending=True, inplace=True)
|
|
100
|
+
|
|
101
|
+
return df_acvg_dcvg
|
|
102
|
+
|
|
103
|
+
@staticmethod
|
|
104
|
+
def closest_distance_cips(
|
|
105
|
+
df_acvg_dcvg: pd.DataFrame, df_cips: pd.DataFrame
|
|
106
|
+
) -> pd.DataFrame:
|
|
107
|
+
closest_cips_distance = []
|
|
108
|
+
closest_cips_index = []
|
|
109
|
+
closest_cips_latitude = []
|
|
110
|
+
closest_cips_longitude = []
|
|
111
|
+
closest_cips_real_distance = []
|
|
112
|
+
closest_cips_condition = []
|
|
113
|
+
closest_cips_voltage_inverse = []
|
|
114
|
+
|
|
115
|
+
cips_coordinates = df_cips[["Latitude", "Longitude", "Real Distance"]]
|
|
116
|
+
|
|
117
|
+
for index, row_acvg_dcvg in df_acvg_dcvg.iterrows():
|
|
118
|
+
distances = []
|
|
119
|
+
lat2 = row_acvg_dcvg["latitude"]
|
|
120
|
+
lon2 = row_acvg_dcvg["longitude"]
|
|
121
|
+
|
|
122
|
+
for _, cips in cips_coordinates.iterrows():
|
|
123
|
+
lat1 = cips["Latitude"]
|
|
124
|
+
lon1 = cips["Longitude"]
|
|
125
|
+
distance = calculate_distance(lat1, lon1, lat2, lon2)
|
|
126
|
+
distances.append(distance)
|
|
127
|
+
|
|
128
|
+
np_distances = np.array(distances)
|
|
129
|
+
distance_min = np.min(np_distances)
|
|
130
|
+
|
|
131
|
+
closest_cips_distance.append(0 - distance_min)
|
|
132
|
+
closest_cips_index.append(np_distances.argmin())
|
|
133
|
+
|
|
134
|
+
closest_cips_voltage_inverse.append(
|
|
135
|
+
df_cips.iloc[np_distances.argmin()]["voltage_inverse"]
|
|
136
|
+
)
|
|
137
|
+
closest_cips_condition.append(
|
|
138
|
+
df_cips.iloc[np_distances.argmin()]["condition"]
|
|
139
|
+
)
|
|
140
|
+
closest_cips_latitude.append(
|
|
141
|
+
cips_coordinates.iloc[np_distances.argmin()]["Latitude"]
|
|
142
|
+
)
|
|
143
|
+
closest_cips_longitude.append(
|
|
144
|
+
cips_coordinates.iloc[np_distances.argmin()]["Longitude"]
|
|
145
|
+
)
|
|
146
|
+
closest_cips_real_distance.append(
|
|
147
|
+
cips_coordinates.iloc[np_distances.argmin()]["Real Distance"]
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
df_acvg_dcvg["closest_cips_distance"] = closest_cips_distance
|
|
151
|
+
df_acvg_dcvg["closest_cips_real_distance"] = closest_cips_real_distance
|
|
152
|
+
df_acvg_dcvg["closest_cips_index"] = closest_cips_index
|
|
153
|
+
df_acvg_dcvg["closest_cips_latitude"] = closest_cips_latitude
|
|
154
|
+
df_acvg_dcvg["closest_cips_longitude"] = closest_cips_longitude
|
|
155
|
+
df_acvg_dcvg["closest_cips_voltage_inverse"] = (
|
|
156
|
+
closest_cips_voltage_inverse
|
|
157
|
+
)
|
|
158
|
+
df_acvg_dcvg["closest_cips_condition"] = closest_cips_condition
|
|
159
|
+
|
|
160
|
+
df_acvg_dcvg.sort_values("real_distance", ascending=True, inplace=True)
|
|
161
|
+
|
|
162
|
+
return df_acvg_dcvg
|
|
163
|
+
|
|
164
|
+
def transform(
|
|
165
|
+
self,
|
|
166
|
+
df: pd.DataFrame,
|
|
167
|
+
excel_filepath: str,
|
|
168
|
+
sheet_name: str = "Sheet1",
|
|
169
|
+
json_filepath: Optional[str] = None,
|
|
170
|
+
) -> str:
|
|
171
|
+
"""Normalize a file.
|
|
172
|
+
|
|
173
|
+
Args:
|
|
174
|
+
df (pd.DataFrame): data frame.
|
|
175
|
+
excel_filepath (str): filename to normalize.
|
|
176
|
+
sheet_name (str): sheet name.
|
|
177
|
+
json_filepath (str): json file path.
|
|
178
|
+
|
|
179
|
+
Returns:
|
|
180
|
+
str: normalized file.
|
|
181
|
+
"""
|
|
182
|
+
df_acvg_dcvg = df.drop_duplicates(
|
|
183
|
+
subset=self.UNIQUE_COLUMNS, keep="last"
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
df_acvg_dcvg.dropna(subset=["latitude", "longitude"], inplace=True)
|
|
187
|
+
df_acvg_dcvg.reset_index(drop=True, inplace=True)
|
|
188
|
+
|
|
189
|
+
try:
|
|
190
|
+
writer = pd.ExcelWriter(excel_filepath, engine="xlsxwriter")
|
|
191
|
+
|
|
192
|
+
df_acvg_dcvg["survey_dcvg"] = df_acvg_dcvg["survey_dcvg"].apply(
|
|
193
|
+
lambda x: x.strftime("%Y-%m-%d") if not pd.isnull(x) else None
|
|
194
|
+
)
|
|
195
|
+
df_acvg_dcvg["survey_acvg"] = df_acvg_dcvg["survey_acvg"].apply(
|
|
196
|
+
lambda x: x.strftime("%Y-%m-%d") if not pd.isnull(x) else None
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
df_acvg_dcvg.to_excel(writer, sheet_name="Sheet1", index=False)
|
|
200
|
+
df_acvg_dcvg.columns = rename_columns(
|
|
201
|
+
df_acvg_dcvg.columns.tolist()
|
|
202
|
+
)
|
|
203
|
+
df_acvg_dcvg.to_json(json_filepath, orient="records")
|
|
204
|
+
|
|
205
|
+
writer.close()
|
|
206
|
+
|
|
207
|
+
# print(excel_filepath)
|
|
208
|
+
|
|
209
|
+
return excel_filepath
|
|
210
|
+
except Exception as e:
|
|
211
|
+
raise e
|
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
import glob
|
|
2
|
+
import pandas as pd
|
|
3
|
+
from typing import Self, Any, Optional
|
|
4
|
+
from .utils import *
|
|
5
|
+
from .const import *
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class CIPS:
|
|
9
|
+
def __init__(
|
|
10
|
+
self,
|
|
11
|
+
file_or_dir: str,
|
|
12
|
+
overwrite: bool = False,
|
|
13
|
+
keep_original_filename: bool = False,
|
|
14
|
+
verbose: bool = False,
|
|
15
|
+
):
|
|
16
|
+
self.file_or_dir = file_or_dir
|
|
17
|
+
self.overwrite = overwrite
|
|
18
|
+
self.keep_original_filename = keep_original_filename
|
|
19
|
+
self.prefix = "cips"
|
|
20
|
+
self.excel_dir = CIPS_EXCEL_DIR
|
|
21
|
+
self.json_dir = CIPS_JSON_DIR
|
|
22
|
+
self.results = []
|
|
23
|
+
|
|
24
|
+
self.COLUMNS_VALIDATED = [
|
|
25
|
+
"Data No",
|
|
26
|
+
"Latitude",
|
|
27
|
+
"Longitude",
|
|
28
|
+
"DCP/Feature/DCVG Anomaly",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
self.UNIQUE_COLUMNS = ["Latitude", "Longitude"]
|
|
32
|
+
|
|
33
|
+
self.check_sequential_file = False
|
|
34
|
+
self.verbose = verbose
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def files(self) -> List[str]:
|
|
38
|
+
files = []
|
|
39
|
+
if os.path.isfile(self.file_or_dir):
|
|
40
|
+
files.append(self.file_or_dir)
|
|
41
|
+
|
|
42
|
+
if os.path.isdir(self.file_or_dir):
|
|
43
|
+
files = glob.glob(os.path.join(self.file_or_dir, "*.xlsx"))
|
|
44
|
+
|
|
45
|
+
return files
|
|
46
|
+
|
|
47
|
+
def validate_column(self, columns: list[str]) -> tuple[bool, list[str]]:
|
|
48
|
+
"""Validate columns.
|
|
49
|
+
|
|
50
|
+
Args:
|
|
51
|
+
columns (list[str]): list of column names.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
bool: column validated.
|
|
55
|
+
list[str]: list of column names.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
missing_columns: list[str] = []
|
|
59
|
+
|
|
60
|
+
for column_validated in self.COLUMNS_VALIDATED:
|
|
61
|
+
if column_validated not in columns:
|
|
62
|
+
missing_columns.append(column_validated)
|
|
63
|
+
|
|
64
|
+
if ("Voltage" not in columns) and ("Off Voltage" not in columns):
|
|
65
|
+
missing_columns.append("Voltage/Off Voltage")
|
|
66
|
+
|
|
67
|
+
if len(missing_columns) > 0:
|
|
68
|
+
return False, missing_columns
|
|
69
|
+
|
|
70
|
+
return True, missing_columns
|
|
71
|
+
|
|
72
|
+
@staticmethod
|
|
73
|
+
def transform_df(df: pd.DataFrame) -> pd.DataFrame:
|
|
74
|
+
"""Extract data from a file.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
df (pd.DataFrame): data frame.
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
pd.DataFrame: data extracted.
|
|
81
|
+
"""
|
|
82
|
+
if "Off Voltage" in df.columns:
|
|
83
|
+
# iccp - impress current cathodic protection
|
|
84
|
+
df = df[
|
|
85
|
+
[
|
|
86
|
+
"Data No",
|
|
87
|
+
"Off Voltage",
|
|
88
|
+
"On Voltage",
|
|
89
|
+
"Latitude",
|
|
90
|
+
"Longitude",
|
|
91
|
+
"Comment",
|
|
92
|
+
"DCP/Feature/DCVG Anomaly",
|
|
93
|
+
]
|
|
94
|
+
].copy(deep=True)
|
|
95
|
+
df["protection"] = "ICCP"
|
|
96
|
+
df.rename(columns={"Off Voltage": "Voltage"}, inplace=True)
|
|
97
|
+
return df
|
|
98
|
+
|
|
99
|
+
# sacp - sacrificial anode cathodic protection
|
|
100
|
+
df = df[
|
|
101
|
+
[
|
|
102
|
+
"Data No",
|
|
103
|
+
"Voltage",
|
|
104
|
+
"Latitude",
|
|
105
|
+
"Longitude",
|
|
106
|
+
"Comment",
|
|
107
|
+
"DCP/Feature/DCVG Anomaly",
|
|
108
|
+
]
|
|
109
|
+
].copy(deep=True)
|
|
110
|
+
df["On Voltage"] = None
|
|
111
|
+
df["protection"] = "SACP"
|
|
112
|
+
return df
|
|
113
|
+
|
|
114
|
+
@staticmethod
|
|
115
|
+
def condition(voltage: float) -> str:
|
|
116
|
+
"""Get condition based on voltage.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
voltage (float): voltage.
|
|
120
|
+
|
|
121
|
+
Returns:
|
|
122
|
+
str: condition.
|
|
123
|
+
"""
|
|
124
|
+
if -1.2 < voltage <= -0.85:
|
|
125
|
+
return "PROTECTED"
|
|
126
|
+
if voltage <= -1.2:
|
|
127
|
+
return "OVER PROTECTED"
|
|
128
|
+
return "UNPROTECTED"
|
|
129
|
+
|
|
130
|
+
def transform(
|
|
131
|
+
self,
|
|
132
|
+
df: pd.DataFrame,
|
|
133
|
+
excel_filepath: str,
|
|
134
|
+
sheet_name: str = "Sheet1",
|
|
135
|
+
json_filepath: Optional[str] = None,
|
|
136
|
+
) -> str:
|
|
137
|
+
"""Normalize a file.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
df (pd.DataFrame): data frame.
|
|
141
|
+
excel_filepath (str): filename to normalize.
|
|
142
|
+
sheet_name (str): sheet name.
|
|
143
|
+
json_filepath (str): json file path.
|
|
144
|
+
|
|
145
|
+
Returns:
|
|
146
|
+
str: normalized file.
|
|
147
|
+
"""
|
|
148
|
+
df = self.transform_df(df)
|
|
149
|
+
|
|
150
|
+
df["condition"] = df["Voltage"].apply(lambda x: self.condition(x))
|
|
151
|
+
df["voltage_inverse"] = df["Voltage"] * -1
|
|
152
|
+
df["on_voltage_inverse"] = df["On Voltage"] * -1
|
|
153
|
+
|
|
154
|
+
df["type"] = "PCM" if "4Hz Current (A)" in df.columns else "CIPS"
|
|
155
|
+
df["interpolated"] = df["Latitude"].isna() & df["Longitude"].isna()
|
|
156
|
+
df["Latitude"] = df["Latitude"].interpolate(method="linear")
|
|
157
|
+
df["Longitude"] = df["Longitude"].interpolate(method="linear")
|
|
158
|
+
|
|
159
|
+
for index in df.index:
|
|
160
|
+
if index == 0:
|
|
161
|
+
df["Distance"] = 0.0
|
|
162
|
+
df["Real Distance"] = 0.0
|
|
163
|
+
continue
|
|
164
|
+
|
|
165
|
+
# if index == len(df) - 1:
|
|
166
|
+
# continue
|
|
167
|
+
|
|
168
|
+
lat_1 = df.loc[index - 1, "Latitude"]
|
|
169
|
+
lon_1 = df.loc[index - 1, "Longitude"]
|
|
170
|
+
lat_2 = df.loc[index, "Latitude"]
|
|
171
|
+
lon_2 = df.loc[index, "Longitude"]
|
|
172
|
+
|
|
173
|
+
distance = calculate_distance(lat_1, lon_1, lat_2, lon_2)
|
|
174
|
+
df.loc[index, "Distance"] = distance
|
|
175
|
+
df.loc[index, "Real Distance"] = (
|
|
176
|
+
distance + df.loc[index - 1, "Real Distance"]
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
df.set_index("Data No", inplace=True)
|
|
180
|
+
df.to_excel(excel_filepath, sheet_name=sheet_name, index=True)
|
|
181
|
+
|
|
182
|
+
if json_filepath:
|
|
183
|
+
df.columns = rename_columns(df.columns.tolist())
|
|
184
|
+
df.to_json(json_filepath, orient="records")
|
|
185
|
+
|
|
186
|
+
return excel_filepath
|
|
187
|
+
|
|
188
|
+
def drop_columns(self, df: pd.DataFrame) -> pd.DataFrame:
|
|
189
|
+
"""Drop empty row and duplicated columns.
|
|
190
|
+
|
|
191
|
+
Args:
|
|
192
|
+
df: pd.DataFrame
|
|
193
|
+
|
|
194
|
+
Returns:
|
|
195
|
+
pd.DataFrame
|
|
196
|
+
"""
|
|
197
|
+
df.dropna(how="all", inplace=True)
|
|
198
|
+
df = df.drop_duplicates(
|
|
199
|
+
subset=self.UNIQUE_COLUMNS, keep="last"
|
|
200
|
+
).reset_index(drop=True)
|
|
201
|
+
|
|
202
|
+
return df
|
|
203
|
+
|
|
204
|
+
def process_df(
|
|
205
|
+
self,
|
|
206
|
+
df: pd.DataFrame,
|
|
207
|
+
filename: str,
|
|
208
|
+
sheet_name: str = "Sheet1",
|
|
209
|
+
as_json: bool = False,
|
|
210
|
+
overwrite: bool = False,
|
|
211
|
+
) -> dict[str, Any]:
|
|
212
|
+
"""Process a file.
|
|
213
|
+
|
|
214
|
+
Args:
|
|
215
|
+
df (pd.DataFrame): path to file
|
|
216
|
+
filename (str): filename to normalize.
|
|
217
|
+
sheet_name (str): sheet name.
|
|
218
|
+
as_json (bool): whether to return as json.
|
|
219
|
+
overwrite (bool): overwrite existing file.
|
|
220
|
+
|
|
221
|
+
Returns:
|
|
222
|
+
dict[str, Any]: processed file.
|
|
223
|
+
"""
|
|
224
|
+
_basename = get_basename(
|
|
225
|
+
filename,
|
|
226
|
+
sheet_name,
|
|
227
|
+
prefix=self.prefix,
|
|
228
|
+
keep_original=self.keep_original_filename,
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
excel_filepath = os.path.join(
|
|
232
|
+
self.excel_dir, f"{slugify(_basename)}.xlsx"
|
|
233
|
+
)
|
|
234
|
+
json_filepath = (
|
|
235
|
+
os.path.join(self.json_dir, f"{slugify(_basename)}.json")
|
|
236
|
+
if as_json
|
|
237
|
+
else None
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
if os.path.exists(excel_filepath) and not overwrite:
|
|
241
|
+
return {
|
|
242
|
+
"success": True,
|
|
243
|
+
"message": "File already normalized",
|
|
244
|
+
"excel": excel_filepath,
|
|
245
|
+
"json": json_filepath,
|
|
246
|
+
"sheet": sheet_name,
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
try:
|
|
250
|
+
df = self.drop_columns(df)
|
|
251
|
+
|
|
252
|
+
return {
|
|
253
|
+
"success": True,
|
|
254
|
+
"message": "File normalized",
|
|
255
|
+
"excel": self.transform(
|
|
256
|
+
df,
|
|
257
|
+
excel_filepath,
|
|
258
|
+
sheet_name=sheet_name,
|
|
259
|
+
json_filepath=json_filepath,
|
|
260
|
+
),
|
|
261
|
+
"json": json_filepath,
|
|
262
|
+
"sheet": sheet_name,
|
|
263
|
+
}
|
|
264
|
+
except Exception as e:
|
|
265
|
+
if self.verbose:
|
|
266
|
+
raise Exception(e)
|
|
267
|
+
return {
|
|
268
|
+
"success": False,
|
|
269
|
+
"message": e,
|
|
270
|
+
"excel": filename,
|
|
271
|
+
"json": json_filepath,
|
|
272
|
+
"sheet": None,
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
def create_dir(self) -> Self:
|
|
276
|
+
os.makedirs(self.excel_dir, exist_ok=True)
|
|
277
|
+
os.makedirs(self.json_dir, exist_ok=True)
|
|
278
|
+
return self
|
|
279
|
+
|
|
280
|
+
def normalize(self) -> None:
|
|
281
|
+
if len(self.files) > 0:
|
|
282
|
+
self.create_dir()
|
|
283
|
+
for file in self.files:
|
|
284
|
+
if self.verbose:
|
|
285
|
+
print(f"Processing file: {file}")
|
|
286
|
+
sheets = worksheets(file)
|
|
287
|
+
|
|
288
|
+
sheet_name = None
|
|
289
|
+
if self.check_sequential_file:
|
|
290
|
+
sheet_name = sequential_file(sheets)
|
|
291
|
+
|
|
292
|
+
# Change to Sequential file sheet
|
|
293
|
+
sheets = [sheet_name]
|
|
294
|
+
if sheet_name is None:
|
|
295
|
+
self.results.append(
|
|
296
|
+
{
|
|
297
|
+
"success": False,
|
|
298
|
+
"message": f"Missing Sequential File sheet",
|
|
299
|
+
"excel": file,
|
|
300
|
+
"json": None,
|
|
301
|
+
"sheet": None,
|
|
302
|
+
}
|
|
303
|
+
)
|
|
304
|
+
continue
|
|
305
|
+
|
|
306
|
+
dfs = pd.read_excel(file, sheet_name=sheet_name)
|
|
307
|
+
for sheet in sheets:
|
|
308
|
+
if self.verbose:
|
|
309
|
+
print(f"|| Processing sheet: {sheet}", end="")
|
|
310
|
+
df = dfs[sheet] if (sheet_name is None) else dfs
|
|
311
|
+
columns = df.columns.tolist()
|
|
312
|
+
column_is_oke, missing_columns = self.validate_column(
|
|
313
|
+
columns
|
|
314
|
+
)
|
|
315
|
+
if column_is_oke:
|
|
316
|
+
if self.verbose:
|
|
317
|
+
print(" OK!")
|
|
318
|
+
result = self.process_df(
|
|
319
|
+
df,
|
|
320
|
+
filename=file,
|
|
321
|
+
sheet_name=sheet,
|
|
322
|
+
as_json=True,
|
|
323
|
+
overwrite=self.overwrite,
|
|
324
|
+
)
|
|
325
|
+
self.results.append(result)
|
|
326
|
+
else:
|
|
327
|
+
if self.verbose:
|
|
328
|
+
print(" ‼️NOT OK!")
|
|
329
|
+
self.results.append(
|
|
330
|
+
{
|
|
331
|
+
"success": False,
|
|
332
|
+
"message": f"Sheet: {sheet}. Missing columns: {missing_columns}",
|
|
333
|
+
"excel": file,
|
|
334
|
+
"json": None,
|
|
335
|
+
"sheet": sheet,
|
|
336
|
+
}
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
return None
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
NORMALIZE_DIR = os.path.join(os.getcwd(), "normalize")
|
|
5
|
+
|
|
6
|
+
CIPS_DIR = os.path.join(NORMALIZE_DIR, "cips")
|
|
7
|
+
CIPS_EXCEL_DIR = os.path.join(CIPS_DIR, "excel")
|
|
8
|
+
CIPS_JSON_DIR = os.path.join(CIPS_DIR, "json")
|
|
9
|
+
|
|
10
|
+
PCM_DIR = os.path.join(NORMALIZE_DIR, "pcm")
|
|
11
|
+
PCM_EXCEL_DIR = os.path.join(PCM_DIR, "excel")
|
|
12
|
+
PCM_JSON_DIR = os.path.join(PCM_DIR, "json")
|
|
13
|
+
|
|
14
|
+
ACVG_DCVG_DIR = os.path.join(NORMALIZE_DIR, "acvg_dcvg")
|
|
15
|
+
ACVG_DCVG_EXCEL_DIR = os.path.join(ACVG_DCVG_DIR, "excel")
|
|
16
|
+
ACVG_DCVG_JSON_DIR = os.path.join(ACVG_DCVG_DIR, "json")
|