pipefy 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipefy/__init__.py +0 -0
- pipefy/etl/__init__.py +0 -0
- pipefy/etl/extract/__init__.py +0 -0
- pipefy/etl/extract/http/__init__.py +9 -0
- pipefy/etl/extract/http/http_extractor.py +254 -0
- pipefy/etl/transform/__init__.py +0 -0
- pipefy/etl/transform/csv/__init__.py +3 -0
- pipefy/etl/transform/csv/reader.py +302 -0
- pipefy/etl/transform/json/__init__.py +3 -0
- pipefy/etl/transform/json/reader.py +118 -0
- pipefy/etl/transform/unzip/__init__.py +3 -0
- pipefy/etl/transform/unzip/base.py +129 -0
- pipefy/exceptions/__init__.py +29 -0
- pipefy/exceptions/base.py +55 -0
- pipefy/exceptions/csv_processor.py +24 -0
- pipefy/exceptions/download_processor.py +21 -0
- pipefy/exceptions/file_system.py +78 -0
- pipefy/exceptions/unzip_processor.py +24 -0
- pipefy/factories/__init__.py +9 -0
- pipefy/factories/exceptions_factory.py +69 -0
- pipefy/factories/file_system_factory.py +36 -0
- pipefy/log/__init__.py +6 -0
- pipefy/log/logger.py +19 -0
- pipefy/operations/__init__.py +17 -0
- pipefy/operations/operations.py +297 -0
- pipefy/operations/pipeline.py +44 -0
- pipefy/processors/__init__.py +57 -0
- pipefy/processors/abc.py +71 -0
- pipefy/processors/base.py +350 -0
- pipefy/processors/chain_processors/__init__.py +9 -0
- pipefy/processors/chain_processors/base.py +175 -0
- pipefy/processors/file_system/__init__.py +13 -0
- pipefy/processors/file_system/base.py +298 -0
- pipefy/processors/file_system/file_system_types.py +6 -0
- pipefy/processors/meta.py +112 -0
- pipefy/processors/mixins.py +0 -0
- pipefy/processors/processor_types.py +6 -0
- pipefy/processors/retry_processor.py +164 -0
- pipefy/processors/splitter_processor.py +54 -0
- pipefy/utils/__init__.py +11 -0
- pipefy/utils/common.py +97 -0
- pipefy-1.0.0.dist-info/LICENSE +9 -0
- pipefy-1.0.0.dist-info/METADATA +223 -0
- pipefy-1.0.0.dist-info/RECORD +45 -0
- pipefy-1.0.0.dist-info/WHEEL +4 -0
pipefy/__init__.py
ADDED
|
File without changes
|
pipefy/etl/__init__.py
ADDED
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
import ssl
|
|
2
|
+
from os import PathLike
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Dict, Generator, List, Literal, Tuple, Type, Union
|
|
5
|
+
from urllib.parse import urlparse
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
|
|
9
|
+
from pipefy.log import logger_factory
|
|
10
|
+
from pipefy.processors.base import BaseProcessor
|
|
11
|
+
from pipefy.processors.file_system import AbstractFileSystemManager
|
|
12
|
+
|
|
13
|
+
logger = logger_factory()
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class HttpDataExtractProcessor(
|
|
17
|
+
BaseProcessor[str | httpx._urls.URL, str | PathLike[str], None, None]
|
|
18
|
+
):
|
|
19
|
+
"""
|
|
20
|
+
Processor for downloading data from an HTTP endpoint and saving it to the local file system.
|
|
21
|
+
|
|
22
|
+
This processor sends an HTTP GET request to the specified URL, retrieves the data,
|
|
23
|
+
and saves it to a file using the provided file system manager.
|
|
24
|
+
|
|
25
|
+
Attributes:
|
|
26
|
+
_file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): Manager for handling file system operations.
|
|
27
|
+
_params (dict): Parameters to include in the HTTP request.
|
|
28
|
+
_headers (dict): Headers to include in the HTTP request.
|
|
29
|
+
_timeout (int): Timeout for the HTTP request in seconds.
|
|
30
|
+
_follow_redirects (bool): Whether to follow redirects for the HTTP request.
|
|
31
|
+
_cookies (dict): Cookies to include in the HTTP request.
|
|
32
|
+
_auth (httpx.Auth): Authentication information for the HTTP request.
|
|
33
|
+
_proxy (httpx.Proxy): Proxy information for the HTTP request.
|
|
34
|
+
_cert (str | Tuple[str, str]): SSL certificate for the HTTP request.
|
|
35
|
+
_verify (bool | str): Whether to verify SSL certificates.
|
|
36
|
+
_trust_env (bool): Whether to trust environment variables for HTTP configuration.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
file_system_manager: AbstractFileSystemManager[
|
|
42
|
+
Union[str, PathLike[str]], Union[str, bytes]
|
|
43
|
+
],
|
|
44
|
+
white_exceptions: List[Type[Exception]] | None = None,
|
|
45
|
+
params: Dict[str, str] | None = None,
|
|
46
|
+
headers: Dict[str, str] | None = None,
|
|
47
|
+
timeout: int | None = 120,
|
|
48
|
+
follow_redirects: bool = False,
|
|
49
|
+
cookies: Dict[str, str] | None = None,
|
|
50
|
+
auth: httpx.Auth | None = None,
|
|
51
|
+
proxy: httpx.Proxy | None = None,
|
|
52
|
+
cert: str | Tuple[str, str] | None = None,
|
|
53
|
+
verify: Union[str, bool, ssl.SSLContext] = True,
|
|
54
|
+
trust_env: bool = True,
|
|
55
|
+
) -> None:
|
|
56
|
+
"""
|
|
57
|
+
Initializes the HttpDataExtractProcessor.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
file_system_manager (AbstractFileSystemManager[
|
|
61
|
+
Union[str, PathLike[str]], Union[str, bytes]
|
|
62
|
+
]
|
|
63
|
+
): A file system manager to handle file creation.
|
|
64
|
+
white_exceptions (List[Type[BaseProcessorException]] | None): List of exceptions to allow in processing.
|
|
65
|
+
params (Dict[str, str] | None): URL parameters for the HTTP request.
|
|
66
|
+
headers (Dict[str, str] | None): Headers for the HTTP request.
|
|
67
|
+
timeout (int | None): Timeout for the HTTP request.
|
|
68
|
+
follow_redirects (bool): Whether to follow redirects.
|
|
69
|
+
cookies (Dict[str, str] | None): Cookies for the HTTP request.
|
|
70
|
+
auth (httpx.Auth | None): Authentication details for the HTTP request.
|
|
71
|
+
proxy (httpx.Proxy | None): Proxy settings for the HTTP request.
|
|
72
|
+
cert (str | Tuple[str, str] | None): SSL certificate for the HTTP request.
|
|
73
|
+
verify (bool | str | None): SSL verification flag.
|
|
74
|
+
trust_env (bool | None): Whether to trust environment variables.
|
|
75
|
+
"""
|
|
76
|
+
super().__init__(white_exceptions)
|
|
77
|
+
self._file_system_manager = file_system_manager
|
|
78
|
+
self._params = params or {}
|
|
79
|
+
self._headers = headers or {}
|
|
80
|
+
self._timeout = timeout
|
|
81
|
+
self._follow_redirects = follow_redirects
|
|
82
|
+
self._cookies = cookies or {}
|
|
83
|
+
self._auth = auth
|
|
84
|
+
self._proxy = proxy
|
|
85
|
+
self._cert = cert
|
|
86
|
+
self._verify = verify
|
|
87
|
+
self._trust_env = trust_env
|
|
88
|
+
|
|
89
|
+
def process(
|
|
90
|
+
self, input_data: str | httpx._urls.URL
|
|
91
|
+
) -> Generator[PathLike | str, None, None]:
|
|
92
|
+
"""
|
|
93
|
+
Downloads data from the provided URL and saves it to the file system.
|
|
94
|
+
|
|
95
|
+
The method retrieves the content from the specified URL using HTTP GET, saves it
|
|
96
|
+
to a file using the file system manager, and passes the file name to the next processor
|
|
97
|
+
(if any) in the chain.
|
|
98
|
+
|
|
99
|
+
Args:
|
|
100
|
+
input_data (str): The URL to download data from.
|
|
101
|
+
|
|
102
|
+
Yields:
|
|
103
|
+
PathLike | str: The file name or path of the downloaded file.
|
|
104
|
+
"""
|
|
105
|
+
logger.debug(f"Starting download from {input_data}")
|
|
106
|
+
resp = httpx.get(
|
|
107
|
+
input_data,
|
|
108
|
+
params=self._params,
|
|
109
|
+
headers=self._headers,
|
|
110
|
+
timeout=self._timeout,
|
|
111
|
+
follow_redirects=self._follow_redirects,
|
|
112
|
+
cookies=self._cookies,
|
|
113
|
+
auth=self._auth,
|
|
114
|
+
proxy=self._proxy,
|
|
115
|
+
cert=self._cert,
|
|
116
|
+
verify=self._verify,
|
|
117
|
+
trust_env=self._trust_env,
|
|
118
|
+
)
|
|
119
|
+
resp.raise_for_status()
|
|
120
|
+
|
|
121
|
+
parsed_url = urlparse(str(input_data))
|
|
122
|
+
file_name = Path(parsed_url.path).name
|
|
123
|
+
self._file_system_manager.create_file(
|
|
124
|
+
file_name,
|
|
125
|
+
resp.content,
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
logger.debug("Download was successful")
|
|
129
|
+
if self._next is not None:
|
|
130
|
+
yield from self._next.process(file_name)
|
|
131
|
+
else:
|
|
132
|
+
yield file_name
|
|
133
|
+
|
|
134
|
+
def __str__(self) -> str:
|
|
135
|
+
return (
|
|
136
|
+
f"{self.__class__.__name__}("
|
|
137
|
+
f"headers={self._headers}, "
|
|
138
|
+
f"params={self._params})"
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class HttpxStreamDownloadProcessor(
|
|
143
|
+
BaseProcessor[str | httpx._urls.URL, str | PathLike[str], None, None]
|
|
144
|
+
):
|
|
145
|
+
"""
|
|
146
|
+
Processor for downloading large files from an HTTP endpoint using streaming.
|
|
147
|
+
|
|
148
|
+
This processor downloads data from a specified URL using streaming (i.e., chunked
|
|
149
|
+
transfer encoding). This is useful for downloading large files that may not fit into
|
|
150
|
+
memory entirely. The file is saved using the provided file system manager.
|
|
151
|
+
|
|
152
|
+
Attributes:
|
|
153
|
+
_file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): Manager for handling file system operations.
|
|
154
|
+
_chunk_size (int): The size of each chunk of data to download.
|
|
155
|
+
_method (str): The HTTP method to use (GET, POST, etc.).
|
|
156
|
+
_headers (dict): Headers to include in the HTTP request.
|
|
157
|
+
_params (dict): Parameters to include in the HTTP request.
|
|
158
|
+
_cookies (dict): Cookies to include in the HTTP request.
|
|
159
|
+
_timeout (float): Timeout for the HTTP request.
|
|
160
|
+
_proxies (str | dict): Proxy settings for the HTTP request.
|
|
161
|
+
_verify (bool | str): Whether to verify SSL certificates.
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
def __init__(
|
|
165
|
+
self,
|
|
166
|
+
file_system_manager: AbstractFileSystemManager[
|
|
167
|
+
Union[str, PathLike[str]], Union[str, bytes]
|
|
168
|
+
],
|
|
169
|
+
white_exceptions: List[Type[Exception]] | None = None,
|
|
170
|
+
chunk_size: int | None = 8192,
|
|
171
|
+
method: Literal["GET", "POST", "PUT", "PATCH"] = "GET",
|
|
172
|
+
headers: dict | None = None,
|
|
173
|
+
params: dict | None = None,
|
|
174
|
+
cookies: dict | None = None,
|
|
175
|
+
timeout: float | None = None,
|
|
176
|
+
proxies: str | dict | None = None,
|
|
177
|
+
verify: bool | str = True,
|
|
178
|
+
) -> None:
|
|
179
|
+
"""
|
|
180
|
+
Initializes the HttpxStreamDownloadProcessor.
|
|
181
|
+
|
|
182
|
+
Args:
|
|
183
|
+
file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): A file system manager to handle file creation.
|
|
184
|
+
white_exceptions (List[Type[BaseProcessorException]] | None): List of exceptions to allow in processing.
|
|
185
|
+
chunk_size (int | None): The size of each chunk to download.
|
|
186
|
+
method (Literal): The HTTP method to use (GET, POST, etc.).
|
|
187
|
+
headers (dict | None): Headers for the HTTP request.
|
|
188
|
+
params (dict | None): Parameters for the HTTP request.
|
|
189
|
+
cookies (dict | None): Cookies for the HTTP request.
|
|
190
|
+
timeout (float | None): Timeout for the HTTP request.
|
|
191
|
+
proxies (str | dict | None): Proxy settings for the HTTP request.
|
|
192
|
+
verify (bool | str): Whether to verify SSL certificates.
|
|
193
|
+
"""
|
|
194
|
+
super().__init__(white_exceptions)
|
|
195
|
+
self._file_system_manager = file_system_manager
|
|
196
|
+
self._chunk_size = chunk_size
|
|
197
|
+
self._method = method
|
|
198
|
+
self._headers = headers or {}
|
|
199
|
+
self._params = params or {}
|
|
200
|
+
self._cookies = cookies or {}
|
|
201
|
+
self._timeout = timeout
|
|
202
|
+
self._proxies = proxies
|
|
203
|
+
self._verify = verify
|
|
204
|
+
|
|
205
|
+
def process(
|
|
206
|
+
self, input_data: str | httpx._urls.URL
|
|
207
|
+
) -> Generator[PathLike[str] | str, None, None]:
|
|
208
|
+
"""
|
|
209
|
+
Downloads data from the provided URL using streaming and saves it to the file system.
|
|
210
|
+
|
|
211
|
+
This method downloads the content of a URL in chunks, which is useful for large
|
|
212
|
+
files. It saves the content to a file and passes the file path to the next processor
|
|
213
|
+
(if any) in the chain.
|
|
214
|
+
|
|
215
|
+
Args:
|
|
216
|
+
input_data (str): The URL to download data from.
|
|
217
|
+
|
|
218
|
+
Yields:
|
|
219
|
+
PathLike[str] | str: The file path of the downloaded file.
|
|
220
|
+
"""
|
|
221
|
+
logger.debug(f"Starting download from {input_data}")
|
|
222
|
+
with httpx.stream(
|
|
223
|
+
method=self._method,
|
|
224
|
+
url=input_data,
|
|
225
|
+
headers=self._headers,
|
|
226
|
+
params=self._params,
|
|
227
|
+
cookies=self._cookies,
|
|
228
|
+
timeout=self._timeout,
|
|
229
|
+
proxies=self._proxies,
|
|
230
|
+
verify=self._verify,
|
|
231
|
+
) as resp:
|
|
232
|
+
resp.raise_for_status()
|
|
233
|
+
parsed_url = urlparse(str(input_data))
|
|
234
|
+
file_name = Path(parsed_url.path).name
|
|
235
|
+
file_path = self._file_system_manager.get_path(file_name)
|
|
236
|
+
self._file_system_manager.create_file(file_path)
|
|
237
|
+
for chunk in resp.iter_bytes(chunk_size=self._chunk_size):
|
|
238
|
+
if chunk:
|
|
239
|
+
self._file_system_manager.append_to_file(file_path, chunk)
|
|
240
|
+
|
|
241
|
+
logger.debug(f"Download was successful: {file_path}")
|
|
242
|
+
if self._next is not None:
|
|
243
|
+
yield from self._next.process(file_path)
|
|
244
|
+
else:
|
|
245
|
+
yield file_path
|
|
246
|
+
|
|
247
|
+
def __str__(self) -> str:
|
|
248
|
+
return (
|
|
249
|
+
f"{self.__class__.__name__}"
|
|
250
|
+
f"(chunk_size={self._chunk_size}, "
|
|
251
|
+
f"method={self._method}, "
|
|
252
|
+
f"headers={self._headers}, "
|
|
253
|
+
f"params={self._params})"
|
|
254
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
import io
|
|
3
|
+
from os import PathLike
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import (
|
|
6
|
+
Any,
|
|
7
|
+
Callable,
|
|
8
|
+
Dict,
|
|
9
|
+
Generator,
|
|
10
|
+
List,
|
|
11
|
+
Literal,
|
|
12
|
+
Optional,
|
|
13
|
+
Type,
|
|
14
|
+
Union,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
import chardet
|
|
18
|
+
import numpy as np
|
|
19
|
+
import pandas as pd
|
|
20
|
+
|
|
21
|
+
from pipefy.exceptions.base import ProcessorStopIteration
|
|
22
|
+
from pipefy.exceptions.csv_processor import CsvParsingError
|
|
23
|
+
from pipefy.log import logger_factory
|
|
24
|
+
from pipefy.processors.base import BaseProcessor
|
|
25
|
+
from pipefy.processors.file_system import AbstractFileSystemManager, FileEncodingEnum
|
|
26
|
+
|
|
27
|
+
logger = logger_factory()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class CsvParser(
|
|
31
|
+
BaseProcessor[str | PathLike[str], List[Dict[str, Any]], None, None],
|
|
32
|
+
):
|
|
33
|
+
"""
|
|
34
|
+
A base CSV parser for reading and processing CSV files using the pandas library.
|
|
35
|
+
|
|
36
|
+
Attributes:
|
|
37
|
+
file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): Manages file operations.
|
|
38
|
+
white_exceptions (List[Type[Exception]], optional): Exceptions to ignore and raise StopIteration on.
|
|
39
|
+
separators (List[str]): Possible delimiters to try when parsing the CSV.
|
|
40
|
+
header (int, optional): Specifies the row to use as header (default is 0).
|
|
41
|
+
batch_size (Optional[int], optional): Size of each batch for chunked processing.
|
|
42
|
+
encoding (Optional[str], optional): Encoding to use for reading the file.
|
|
43
|
+
engine (Literal["python", "c"], optional): Engine to use for parsing.
|
|
44
|
+
skip_blank_lines (bool, optional): If True, skips blank lines.
|
|
45
|
+
on_bad_lines (Literal["error", "warn", "skip"], optional): Determines action for bad lines.
|
|
46
|
+
quoting (Literal[0, 1, 2, 3], optional): CSV quoting style.
|
|
47
|
+
oriends (Literal["dict", "list", "series", "records", "index", "columns"], optional): Parsed data format.
|
|
48
|
+
converters (Optional[Dict[str, Callable]], optional): Maps column names to conversion functions.
|
|
49
|
+
true_values (Optional[List[str]], optional): Values to interpret as True.
|
|
50
|
+
false_values (Optional[List[str]], optional): Values to interpret as False.
|
|
51
|
+
na_values (Optional[List[str]], optional): Additional values to interpret as NaN.
|
|
52
|
+
keep_default_na (bool, optional): If True, uses default NaN values.
|
|
53
|
+
na_filter (bool, optional): If True, detects missing values (NaN).
|
|
54
|
+
auto_delete (bool, optional): If True, deletes the file after processing.
|
|
55
|
+
|
|
56
|
+
Methods:
|
|
57
|
+
process(self, input_data: str | PathLike[str]) -> Generator[List[Dict[str, Any]], None, None]:
|
|
58
|
+
Parses the specified CSV file and yields data in the defined format.
|
|
59
|
+
|
|
60
|
+
clean_dataframe(self, df: pd.DataFrame) -> pd.DataFrame:
|
|
61
|
+
Removes unnamed columns, drops empty rows and columns, and replaces NaN with None.
|
|
62
|
+
|
|
63
|
+
clean_column_names(df: pd.DataFrame) -> pd.DataFrame:
|
|
64
|
+
Trims whitespace and converts column names to lowercase.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
def __init__(
|
|
68
|
+
self,
|
|
69
|
+
file_system_manager: AbstractFileSystemManager[
|
|
70
|
+
Union[str, PathLike[str]], Union[str, bytes]
|
|
71
|
+
],
|
|
72
|
+
separators: List[str],
|
|
73
|
+
white_exceptions: Optional[List[Type[Exception]]] = None,
|
|
74
|
+
header: int = 0,
|
|
75
|
+
batch_size: Optional[int] = None,
|
|
76
|
+
encoding: Optional[str] = None,
|
|
77
|
+
engine: Literal["python", "c"] = "python",
|
|
78
|
+
skip_blank_lines: bool = False,
|
|
79
|
+
on_bad_lines: Literal["error", "warn", "skip"] = "skip",
|
|
80
|
+
quoting: Literal[0, 1, 2, 3] = csv.QUOTE_NONE,
|
|
81
|
+
oriends: Literal[
|
|
82
|
+
"dict", "list", "series", "records", "index", "columns"
|
|
83
|
+
] = "records",
|
|
84
|
+
converters: Optional[Dict[str, Callable]] = None,
|
|
85
|
+
true_values: Optional[List[str]] = None,
|
|
86
|
+
false_values: Optional[List[str]] = None,
|
|
87
|
+
na_values: Optional[List[str]] = None,
|
|
88
|
+
keep_default_na: bool = True,
|
|
89
|
+
na_filter: bool = True,
|
|
90
|
+
auto_delete: bool = True,
|
|
91
|
+
) -> None:
|
|
92
|
+
"""
|
|
93
|
+
Initializes the CsvParser with settings for parsing.
|
|
94
|
+
|
|
95
|
+
Args:
|
|
96
|
+
file_system_manager (AbstractFileSystemManager[
|
|
97
|
+
Union[str, PathLike[str]], Union[str, bytes]
|
|
98
|
+
]
|
|
99
|
+
): Manages file interactions.
|
|
100
|
+
separators (List[str]): Possible delimiters for the CSV.
|
|
101
|
+
white_exceptions (Optional[List[Type[Exception]]], optional): Exceptions to bypass during processing.
|
|
102
|
+
header (int, optional): Row number to use as headers (default is 0).
|
|
103
|
+
batch_size (Optional[int], optional): Row count for chunked processing.
|
|
104
|
+
encoding (Optional[str], optional): File encoding for reading.
|
|
105
|
+
engine (Literal["python", "c"], optional): Parsing engine.
|
|
106
|
+
skip_blank_lines (bool, optional): If True, ignores blank lines.
|
|
107
|
+
on_bad_lines (Literal["error", "warn", "skip"], optional): Action on encountering bad lines.
|
|
108
|
+
quoting (Literal[0, 1, 2, 3], optional): Quoting style used in the CSV.
|
|
109
|
+
oriends (Literal["dict", "list", "series", "records", "index", "columns"], optional): Format for parsed data.
|
|
110
|
+
converters (Optional[Dict[str, Callable]], optional): Converters for specific columns.
|
|
111
|
+
true_values (Optional[List[str]], optional): Values to recognize as True.
|
|
112
|
+
false_values (Optional[List[str]], optional): Values to recognize as False.
|
|
113
|
+
na_values (Optional[List[str]], optional): Additional strings for NaN.
|
|
114
|
+
keep_default_na (bool, optional): If True, adds default NaN values.
|
|
115
|
+
na_filter (bool, optional): If True, detects NaN.
|
|
116
|
+
auto_delete (bool, optional): Deletes the file after processing if True.
|
|
117
|
+
"""
|
|
118
|
+
self._file_system_manager = file_system_manager
|
|
119
|
+
self._white_exceptions = white_exceptions or []
|
|
120
|
+
self._separators = separators
|
|
121
|
+
self._header = header
|
|
122
|
+
self._batch_size = batch_size
|
|
123
|
+
self._encoding = encoding
|
|
124
|
+
self._engine = engine
|
|
125
|
+
self._skip_blank_lines = skip_blank_lines
|
|
126
|
+
self._on_bad_lines = on_bad_lines
|
|
127
|
+
self._oriends = oriends
|
|
128
|
+
self._quoting = quoting
|
|
129
|
+
self._converters = converters or {}
|
|
130
|
+
self._true_values = true_values or []
|
|
131
|
+
self._false_values = false_values or []
|
|
132
|
+
self._na_values = na_values or []
|
|
133
|
+
self._keep_default_na = keep_default_na
|
|
134
|
+
self._na_filter = na_filter
|
|
135
|
+
self._auto_delete = auto_delete
|
|
136
|
+
|
|
137
|
+
@staticmethod
|
|
138
|
+
def _get_encoding(content: bytes) -> Optional[str]:
|
|
139
|
+
"""
|
|
140
|
+
Detects encoding of file content bytes.
|
|
141
|
+
|
|
142
|
+
Args:
|
|
143
|
+
content (bytes): Byte data to inspect.
|
|
144
|
+
|
|
145
|
+
Returns:
|
|
146
|
+
Optional[str]: Encoding if detected, else None.
|
|
147
|
+
"""
|
|
148
|
+
result = chardet.detect(content)
|
|
149
|
+
encoding = result["encoding"]
|
|
150
|
+
return encoding
|
|
151
|
+
|
|
152
|
+
def _read_csv(
|
|
153
|
+
self, filename: str | PathLike[str], separator: str, **kwargs
|
|
154
|
+
) -> pd.DataFrame:
|
|
155
|
+
"""Reads a CSV file with a given separator.
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
filename (str): CSV file path.
|
|
159
|
+
separator (str): Delimiter for fields.
|
|
160
|
+
|
|
161
|
+
Returns:
|
|
162
|
+
pd.DataFrame: Parsed data in a DataFrame.
|
|
163
|
+
"""
|
|
164
|
+
content = self._file_system_manager.read_file(filename)
|
|
165
|
+
if isinstance(content, str):
|
|
166
|
+
content = content.encode(FileEncodingEnum.UTF_8)
|
|
167
|
+
if bytes(separator, encoding=FileEncodingEnum.UTF_8) not in content:
|
|
168
|
+
raise pd.errors.ParserError(
|
|
169
|
+
f"Separator '{separator}' not found in the file content."
|
|
170
|
+
)
|
|
171
|
+
content_io = io.BytesIO(content)
|
|
172
|
+
return pd.read_csv(
|
|
173
|
+
content_io,
|
|
174
|
+
sep=separator,
|
|
175
|
+
encoding=self._encoding or self._get_encoding(content), # type: ignore
|
|
176
|
+
**kwargs,
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
@staticmethod
|
|
180
|
+
def _clean_column_names(df: pd.DataFrame) -> pd.DataFrame:
|
|
181
|
+
"""
|
|
182
|
+
Cleans column names by stripping whitespace and converting to lowercase.
|
|
183
|
+
|
|
184
|
+
Args:
|
|
185
|
+
df (pd.DataFrame): DataFrame with columns to clean.
|
|
186
|
+
|
|
187
|
+
Returns:
|
|
188
|
+
pd.DataFrame: DataFrame with cleaned column names.
|
|
189
|
+
"""
|
|
190
|
+
df.columns = df.columns.str.strip().str.lower()
|
|
191
|
+
return df
|
|
192
|
+
|
|
193
|
+
def _clean_dataframe(self, df: pd.DataFrame) -> pd.DataFrame:
|
|
194
|
+
"""
|
|
195
|
+
Removes unnamed columns, drops all-NA rows and columns, and replaces NaN with None.
|
|
196
|
+
|
|
197
|
+
Args:
|
|
198
|
+
df (pd.DataFrame): DataFrame to clean.
|
|
199
|
+
|
|
200
|
+
Returns:
|
|
201
|
+
pd.DataFrame: Cleaned DataFrame.
|
|
202
|
+
"""
|
|
203
|
+
df = self._clean_column_names(df)
|
|
204
|
+
df = df.loc[:, ~df.columns.str.contains("^Unnamed", case=False)]
|
|
205
|
+
df = (
|
|
206
|
+
df.dropna(how="all")
|
|
207
|
+
.dropna(axis=1, how="all")
|
|
208
|
+
.replace({np.nan: None})
|
|
209
|
+
)
|
|
210
|
+
return df
|
|
211
|
+
|
|
212
|
+
def process(
|
|
213
|
+
self, input_data: str | PathLike[str]
|
|
214
|
+
) -> Generator[List[Dict[str, Any]], None, None]:
|
|
215
|
+
"""
|
|
216
|
+
Parses the CSV file and yields formatted data as records.
|
|
217
|
+
|
|
218
|
+
Args:
|
|
219
|
+
input_data (str | PathLike[str]): CSV file path.
|
|
220
|
+
|
|
221
|
+
Yields:
|
|
222
|
+
Generator[List[Dict[str, Any]], None, None]: Parsed data as records.
|
|
223
|
+
|
|
224
|
+
Raises:
|
|
225
|
+
CsvParsingError: If parsing fails.
|
|
226
|
+
"""
|
|
227
|
+
logger.debug(f"Start parsing: {input_data}")
|
|
228
|
+
suffix = Path(input_data).suffix
|
|
229
|
+
if suffix != ".csv":
|
|
230
|
+
raise CsvParsingError(
|
|
231
|
+
input_data,
|
|
232
|
+
f"File has an extension '{suffix}', expected .csv",
|
|
233
|
+
)
|
|
234
|
+
read_csv_kwargs = {
|
|
235
|
+
"header": self._header,
|
|
236
|
+
"chunksize": self._batch_size,
|
|
237
|
+
"engine": self._engine,
|
|
238
|
+
"skip_blank_lines": self._skip_blank_lines,
|
|
239
|
+
"on_bad_lines": self._on_bad_lines,
|
|
240
|
+
"quoting": self._quoting,
|
|
241
|
+
"converters": self._converters,
|
|
242
|
+
"true_values": self._true_values,
|
|
243
|
+
"false_values": self._false_values,
|
|
244
|
+
"na_values": self._na_values,
|
|
245
|
+
"keep_default_na": self._keep_default_na,
|
|
246
|
+
"na_filter": self._na_filter,
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
df = None
|
|
250
|
+
for sep in self._separators:
|
|
251
|
+
try:
|
|
252
|
+
df = self._read_csv(
|
|
253
|
+
input_data, separator=sep, **read_csv_kwargs
|
|
254
|
+
)
|
|
255
|
+
break
|
|
256
|
+
except pd.errors.ParserError as e:
|
|
257
|
+
logger.warning(
|
|
258
|
+
f"Failed to parse {input_data} with separator '{sep}': {e}"
|
|
259
|
+
)
|
|
260
|
+
continue
|
|
261
|
+
|
|
262
|
+
if df is None:
|
|
263
|
+
raise CsvParsingError(
|
|
264
|
+
input_data,
|
|
265
|
+
"Failed to parse with any of the provided separators.",
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
if self._batch_size is None:
|
|
269
|
+
df = self._clean_dataframe(df)
|
|
270
|
+
records = df.to_dict(orient=self._oriends) # type: ignore
|
|
271
|
+
if self._next is not None:
|
|
272
|
+
gen = self._next.process(records)
|
|
273
|
+
yield from gen
|
|
274
|
+
else:
|
|
275
|
+
yield records
|
|
276
|
+
else:
|
|
277
|
+
for chunk in df:
|
|
278
|
+
chunk = self._clean_dataframe(chunk) # type: ignore
|
|
279
|
+
if chunk.empty:
|
|
280
|
+
continue
|
|
281
|
+
records = chunk.to_dict(orient=self._oriends) # type: ignore
|
|
282
|
+
if self._next is not None:
|
|
283
|
+
gen = self._next.process(records)
|
|
284
|
+
try:
|
|
285
|
+
yield from gen
|
|
286
|
+
except ProcessorStopIteration:
|
|
287
|
+
...
|
|
288
|
+
else:
|
|
289
|
+
yield records
|
|
290
|
+
if self._auto_delete:
|
|
291
|
+
self._file_system_manager.delete_file(input_data)
|
|
292
|
+
logger.debug(f"Successful parsing: {input_data}")
|
|
293
|
+
|
|
294
|
+
def __str__(self) -> str:
|
|
295
|
+
return (
|
|
296
|
+
f"{self.__class__.__name__}"
|
|
297
|
+
f"(auto_delete={self._auto_delete}, "
|
|
298
|
+
f"batch_size={self._batch_size}, "
|
|
299
|
+
f"separators={self._separators}, "
|
|
300
|
+
f"header={self._header}, "
|
|
301
|
+
f"encoding={self._encoding})"
|
|
302
|
+
)
|