pipefy 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. pipefy/__init__.py +0 -0
  2. pipefy/etl/__init__.py +0 -0
  3. pipefy/etl/extract/__init__.py +0 -0
  4. pipefy/etl/extract/http/__init__.py +9 -0
  5. pipefy/etl/extract/http/http_extractor.py +254 -0
  6. pipefy/etl/transform/__init__.py +0 -0
  7. pipefy/etl/transform/csv/__init__.py +3 -0
  8. pipefy/etl/transform/csv/reader.py +302 -0
  9. pipefy/etl/transform/json/__init__.py +3 -0
  10. pipefy/etl/transform/json/reader.py +118 -0
  11. pipefy/etl/transform/unzip/__init__.py +3 -0
  12. pipefy/etl/transform/unzip/base.py +129 -0
  13. pipefy/exceptions/__init__.py +29 -0
  14. pipefy/exceptions/base.py +55 -0
  15. pipefy/exceptions/csv_processor.py +24 -0
  16. pipefy/exceptions/download_processor.py +21 -0
  17. pipefy/exceptions/file_system.py +78 -0
  18. pipefy/exceptions/unzip_processor.py +24 -0
  19. pipefy/factories/__init__.py +9 -0
  20. pipefy/factories/exceptions_factory.py +69 -0
  21. pipefy/factories/file_system_factory.py +36 -0
  22. pipefy/log/__init__.py +6 -0
  23. pipefy/log/logger.py +19 -0
  24. pipefy/operations/__init__.py +17 -0
  25. pipefy/operations/operations.py +297 -0
  26. pipefy/operations/pipeline.py +44 -0
  27. pipefy/processors/__init__.py +57 -0
  28. pipefy/processors/abc.py +71 -0
  29. pipefy/processors/base.py +350 -0
  30. pipefy/processors/chain_processors/__init__.py +9 -0
  31. pipefy/processors/chain_processors/base.py +175 -0
  32. pipefy/processors/file_system/__init__.py +13 -0
  33. pipefy/processors/file_system/base.py +298 -0
  34. pipefy/processors/file_system/file_system_types.py +6 -0
  35. pipefy/processors/meta.py +112 -0
  36. pipefy/processors/mixins.py +0 -0
  37. pipefy/processors/processor_types.py +6 -0
  38. pipefy/processors/retry_processor.py +164 -0
  39. pipefy/processors/splitter_processor.py +54 -0
  40. pipefy/utils/__init__.py +11 -0
  41. pipefy/utils/common.py +97 -0
  42. pipefy-1.0.0.dist-info/LICENSE +9 -0
  43. pipefy-1.0.0.dist-info/METADATA +223 -0
  44. pipefy-1.0.0.dist-info/RECORD +45 -0
  45. pipefy-1.0.0.dist-info/WHEEL +4 -0
pipefy/__init__.py ADDED
File without changes
pipefy/etl/__init__.py ADDED
File without changes
File without changes
@@ -0,0 +1,9 @@
1
+ from pipefy.etl.extract.http.http_extractor import (
2
+ HttpDataExtractProcessor,
3
+ HttpxStreamDownloadProcessor,
4
+ )
5
+
6
+ __all__ = (
7
+ "HttpDataExtractProcessor",
8
+ "HttpxStreamDownloadProcessor",
9
+ )
@@ -0,0 +1,254 @@
1
+ import ssl
2
+ from os import PathLike
3
+ from pathlib import Path
4
+ from typing import Dict, Generator, List, Literal, Tuple, Type, Union
5
+ from urllib.parse import urlparse
6
+
7
+ import httpx
8
+
9
+ from pipefy.log import logger_factory
10
+ from pipefy.processors.base import BaseProcessor
11
+ from pipefy.processors.file_system import AbstractFileSystemManager
12
+
13
+ logger = logger_factory()
14
+
15
+
16
+ class HttpDataExtractProcessor(
17
+ BaseProcessor[str | httpx._urls.URL, str | PathLike[str], None, None]
18
+ ):
19
+ """
20
+ Processor for downloading data from an HTTP endpoint and saving it to the local file system.
21
+
22
+ This processor sends an HTTP GET request to the specified URL, retrieves the data,
23
+ and saves it to a file using the provided file system manager.
24
+
25
+ Attributes:
26
+ _file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): Manager for handling file system operations.
27
+ _params (dict): Parameters to include in the HTTP request.
28
+ _headers (dict): Headers to include in the HTTP request.
29
+ _timeout (int): Timeout for the HTTP request in seconds.
30
+ _follow_redirects (bool): Whether to follow redirects for the HTTP request.
31
+ _cookies (dict): Cookies to include in the HTTP request.
32
+ _auth (httpx.Auth): Authentication information for the HTTP request.
33
+ _proxy (httpx.Proxy): Proxy information for the HTTP request.
34
+ _cert (str | Tuple[str, str]): SSL certificate for the HTTP request.
35
+ _verify (bool | str): Whether to verify SSL certificates.
36
+ _trust_env (bool): Whether to trust environment variables for HTTP configuration.
37
+ """
38
+
39
+ def __init__(
40
+ self,
41
+ file_system_manager: AbstractFileSystemManager[
42
+ Union[str, PathLike[str]], Union[str, bytes]
43
+ ],
44
+ white_exceptions: List[Type[Exception]] | None = None,
45
+ params: Dict[str, str] | None = None,
46
+ headers: Dict[str, str] | None = None,
47
+ timeout: int | None = 120,
48
+ follow_redirects: bool = False,
49
+ cookies: Dict[str, str] | None = None,
50
+ auth: httpx.Auth | None = None,
51
+ proxy: httpx.Proxy | None = None,
52
+ cert: str | Tuple[str, str] | None = None,
53
+ verify: Union[str, bool, ssl.SSLContext] = True,
54
+ trust_env: bool = True,
55
+ ) -> None:
56
+ """
57
+ Initializes the HttpDataExtractProcessor.
58
+
59
+ Args:
60
+ file_system_manager (AbstractFileSystemManager[
61
+ Union[str, PathLike[str]], Union[str, bytes]
62
+ ]
63
+ ): A file system manager to handle file creation.
64
+ white_exceptions (List[Type[BaseProcessorException]] | None): List of exceptions to allow in processing.
65
+ params (Dict[str, str] | None): URL parameters for the HTTP request.
66
+ headers (Dict[str, str] | None): Headers for the HTTP request.
67
+ timeout (int | None): Timeout for the HTTP request.
68
+ follow_redirects (bool): Whether to follow redirects.
69
+ cookies (Dict[str, str] | None): Cookies for the HTTP request.
70
+ auth (httpx.Auth | None): Authentication details for the HTTP request.
71
+ proxy (httpx.Proxy | None): Proxy settings for the HTTP request.
72
+ cert (str | Tuple[str, str] | None): SSL certificate for the HTTP request.
73
+ verify (bool | str | None): SSL verification flag.
74
+ trust_env (bool | None): Whether to trust environment variables.
75
+ """
76
+ super().__init__(white_exceptions)
77
+ self._file_system_manager = file_system_manager
78
+ self._params = params or {}
79
+ self._headers = headers or {}
80
+ self._timeout = timeout
81
+ self._follow_redirects = follow_redirects
82
+ self._cookies = cookies or {}
83
+ self._auth = auth
84
+ self._proxy = proxy
85
+ self._cert = cert
86
+ self._verify = verify
87
+ self._trust_env = trust_env
88
+
89
+ def process(
90
+ self, input_data: str | httpx._urls.URL
91
+ ) -> Generator[PathLike | str, None, None]:
92
+ """
93
+ Downloads data from the provided URL and saves it to the file system.
94
+
95
+ The method retrieves the content from the specified URL using HTTP GET, saves it
96
+ to a file using the file system manager, and passes the file name to the next processor
97
+ (if any) in the chain.
98
+
99
+ Args:
100
+ input_data (str): The URL to download data from.
101
+
102
+ Yields:
103
+ PathLike | str: The file name or path of the downloaded file.
104
+ """
105
+ logger.debug(f"Starting download from {input_data}")
106
+ resp = httpx.get(
107
+ input_data,
108
+ params=self._params,
109
+ headers=self._headers,
110
+ timeout=self._timeout,
111
+ follow_redirects=self._follow_redirects,
112
+ cookies=self._cookies,
113
+ auth=self._auth,
114
+ proxy=self._proxy,
115
+ cert=self._cert,
116
+ verify=self._verify,
117
+ trust_env=self._trust_env,
118
+ )
119
+ resp.raise_for_status()
120
+
121
+ parsed_url = urlparse(str(input_data))
122
+ file_name = Path(parsed_url.path).name
123
+ self._file_system_manager.create_file(
124
+ file_name,
125
+ resp.content,
126
+ )
127
+
128
+ logger.debug("Download was successful")
129
+ if self._next is not None:
130
+ yield from self._next.process(file_name)
131
+ else:
132
+ yield file_name
133
+
134
+ def __str__(self) -> str:
135
+ return (
136
+ f"{self.__class__.__name__}("
137
+ f"headers={self._headers}, "
138
+ f"params={self._params})"
139
+ )
140
+
141
+
142
+ class HttpxStreamDownloadProcessor(
143
+ BaseProcessor[str | httpx._urls.URL, str | PathLike[str], None, None]
144
+ ):
145
+ """
146
+ Processor for downloading large files from an HTTP endpoint using streaming.
147
+
148
+ This processor downloads data from a specified URL using streaming (i.e., chunked
149
+ transfer encoding). This is useful for downloading large files that may not fit into
150
+ memory entirely. The file is saved using the provided file system manager.
151
+
152
+ Attributes:
153
+ _file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): Manager for handling file system operations.
154
+ _chunk_size (int): The size of each chunk of data to download.
155
+ _method (str): The HTTP method to use (GET, POST, etc.).
156
+ _headers (dict): Headers to include in the HTTP request.
157
+ _params (dict): Parameters to include in the HTTP request.
158
+ _cookies (dict): Cookies to include in the HTTP request.
159
+ _timeout (float): Timeout for the HTTP request.
160
+ _proxies (str | dict): Proxy settings for the HTTP request.
161
+ _verify (bool | str): Whether to verify SSL certificates.
162
+ """
163
+
164
+ def __init__(
165
+ self,
166
+ file_system_manager: AbstractFileSystemManager[
167
+ Union[str, PathLike[str]], Union[str, bytes]
168
+ ],
169
+ white_exceptions: List[Type[Exception]] | None = None,
170
+ chunk_size: int | None = 8192,
171
+ method: Literal["GET", "POST", "PUT", "PATCH"] = "GET",
172
+ headers: dict | None = None,
173
+ params: dict | None = None,
174
+ cookies: dict | None = None,
175
+ timeout: float | None = None,
176
+ proxies: str | dict | None = None,
177
+ verify: bool | str = True,
178
+ ) -> None:
179
+ """
180
+ Initializes the HttpxStreamDownloadProcessor.
181
+
182
+ Args:
183
+ file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): A file system manager to handle file creation.
184
+ white_exceptions (List[Type[BaseProcessorException]] | None): List of exceptions to allow in processing.
185
+ chunk_size (int | None): The size of each chunk to download.
186
+ method (Literal): The HTTP method to use (GET, POST, etc.).
187
+ headers (dict | None): Headers for the HTTP request.
188
+ params (dict | None): Parameters for the HTTP request.
189
+ cookies (dict | None): Cookies for the HTTP request.
190
+ timeout (float | None): Timeout for the HTTP request.
191
+ proxies (str | dict | None): Proxy settings for the HTTP request.
192
+ verify (bool | str): Whether to verify SSL certificates.
193
+ """
194
+ super().__init__(white_exceptions)
195
+ self._file_system_manager = file_system_manager
196
+ self._chunk_size = chunk_size
197
+ self._method = method
198
+ self._headers = headers or {}
199
+ self._params = params or {}
200
+ self._cookies = cookies or {}
201
+ self._timeout = timeout
202
+ self._proxies = proxies
203
+ self._verify = verify
204
+
205
+ def process(
206
+ self, input_data: str | httpx._urls.URL
207
+ ) -> Generator[PathLike[str] | str, None, None]:
208
+ """
209
+ Downloads data from the provided URL using streaming and saves it to the file system.
210
+
211
+ This method downloads the content of a URL in chunks, which is useful for large
212
+ files. It saves the content to a file and passes the file path to the next processor
213
+ (if any) in the chain.
214
+
215
+ Args:
216
+ input_data (str): The URL to download data from.
217
+
218
+ Yields:
219
+ PathLike[str] | str: The file path of the downloaded file.
220
+ """
221
+ logger.debug(f"Starting download from {input_data}")
222
+ with httpx.stream(
223
+ method=self._method,
224
+ url=input_data,
225
+ headers=self._headers,
226
+ params=self._params,
227
+ cookies=self._cookies,
228
+ timeout=self._timeout,
229
+ proxies=self._proxies,
230
+ verify=self._verify,
231
+ ) as resp:
232
+ resp.raise_for_status()
233
+ parsed_url = urlparse(str(input_data))
234
+ file_name = Path(parsed_url.path).name
235
+ file_path = self._file_system_manager.get_path(file_name)
236
+ self._file_system_manager.create_file(file_path)
237
+ for chunk in resp.iter_bytes(chunk_size=self._chunk_size):
238
+ if chunk:
239
+ self._file_system_manager.append_to_file(file_path, chunk)
240
+
241
+ logger.debug(f"Download was successful: {file_path}")
242
+ if self._next is not None:
243
+ yield from self._next.process(file_path)
244
+ else:
245
+ yield file_path
246
+
247
+ def __str__(self) -> str:
248
+ return (
249
+ f"{self.__class__.__name__}"
250
+ f"(chunk_size={self._chunk_size}, "
251
+ f"method={self._method}, "
252
+ f"headers={self._headers}, "
253
+ f"params={self._params})"
254
+ )
File without changes
@@ -0,0 +1,3 @@
1
+ from pipefy.etl.transform.csv.reader import CsvParser
2
+
3
+ __all__ = ("CsvParser",)
@@ -0,0 +1,302 @@
1
+ import csv
2
+ import io
3
+ from os import PathLike
4
+ from pathlib import Path
5
+ from typing import (
6
+ Any,
7
+ Callable,
8
+ Dict,
9
+ Generator,
10
+ List,
11
+ Literal,
12
+ Optional,
13
+ Type,
14
+ Union,
15
+ )
16
+
17
+ import chardet
18
+ import numpy as np
19
+ import pandas as pd
20
+
21
+ from pipefy.exceptions.base import ProcessorStopIteration
22
+ from pipefy.exceptions.csv_processor import CsvParsingError
23
+ from pipefy.log import logger_factory
24
+ from pipefy.processors.base import BaseProcessor
25
+ from pipefy.processors.file_system import AbstractFileSystemManager, FileEncodingEnum
26
+
27
+ logger = logger_factory()
28
+
29
+
30
+ class CsvParser(
31
+ BaseProcessor[str | PathLike[str], List[Dict[str, Any]], None, None],
32
+ ):
33
+ """
34
+ A base CSV parser for reading and processing CSV files using the pandas library.
35
+
36
+ Attributes:
37
+ file_system_manager (AbstractFileSystemManager[Union[str, PathLike[str]], Union[str, bytes]]): Manages file operations.
38
+ white_exceptions (List[Type[Exception]], optional): Exceptions to ignore and raise StopIteration on.
39
+ separators (List[str]): Possible delimiters to try when parsing the CSV.
40
+ header (int, optional): Specifies the row to use as header (default is 0).
41
+ batch_size (Optional[int], optional): Size of each batch for chunked processing.
42
+ encoding (Optional[str], optional): Encoding to use for reading the file.
43
+ engine (Literal["python", "c"], optional): Engine to use for parsing.
44
+ skip_blank_lines (bool, optional): If True, skips blank lines.
45
+ on_bad_lines (Literal["error", "warn", "skip"], optional): Determines action for bad lines.
46
+ quoting (Literal[0, 1, 2, 3], optional): CSV quoting style.
47
+ oriends (Literal["dict", "list", "series", "records", "index", "columns"], optional): Parsed data format.
48
+ converters (Optional[Dict[str, Callable]], optional): Maps column names to conversion functions.
49
+ true_values (Optional[List[str]], optional): Values to interpret as True.
50
+ false_values (Optional[List[str]], optional): Values to interpret as False.
51
+ na_values (Optional[List[str]], optional): Additional values to interpret as NaN.
52
+ keep_default_na (bool, optional): If True, uses default NaN values.
53
+ na_filter (bool, optional): If True, detects missing values (NaN).
54
+ auto_delete (bool, optional): If True, deletes the file after processing.
55
+
56
+ Methods:
57
+ process(self, input_data: str | PathLike[str]) -> Generator[List[Dict[str, Any]], None, None]:
58
+ Parses the specified CSV file and yields data in the defined format.
59
+
60
+ clean_dataframe(self, df: pd.DataFrame) -> pd.DataFrame:
61
+ Removes unnamed columns, drops empty rows and columns, and replaces NaN with None.
62
+
63
+ clean_column_names(df: pd.DataFrame) -> pd.DataFrame:
64
+ Trims whitespace and converts column names to lowercase.
65
+ """
66
+
67
+ def __init__(
68
+ self,
69
+ file_system_manager: AbstractFileSystemManager[
70
+ Union[str, PathLike[str]], Union[str, bytes]
71
+ ],
72
+ separators: List[str],
73
+ white_exceptions: Optional[List[Type[Exception]]] = None,
74
+ header: int = 0,
75
+ batch_size: Optional[int] = None,
76
+ encoding: Optional[str] = None,
77
+ engine: Literal["python", "c"] = "python",
78
+ skip_blank_lines: bool = False,
79
+ on_bad_lines: Literal["error", "warn", "skip"] = "skip",
80
+ quoting: Literal[0, 1, 2, 3] = csv.QUOTE_NONE,
81
+ oriends: Literal[
82
+ "dict", "list", "series", "records", "index", "columns"
83
+ ] = "records",
84
+ converters: Optional[Dict[str, Callable]] = None,
85
+ true_values: Optional[List[str]] = None,
86
+ false_values: Optional[List[str]] = None,
87
+ na_values: Optional[List[str]] = None,
88
+ keep_default_na: bool = True,
89
+ na_filter: bool = True,
90
+ auto_delete: bool = True,
91
+ ) -> None:
92
+ """
93
+ Initializes the CsvParser with settings for parsing.
94
+
95
+ Args:
96
+ file_system_manager (AbstractFileSystemManager[
97
+ Union[str, PathLike[str]], Union[str, bytes]
98
+ ]
99
+ ): Manages file interactions.
100
+ separators (List[str]): Possible delimiters for the CSV.
101
+ white_exceptions (Optional[List[Type[Exception]]], optional): Exceptions to bypass during processing.
102
+ header (int, optional): Row number to use as headers (default is 0).
103
+ batch_size (Optional[int], optional): Row count for chunked processing.
104
+ encoding (Optional[str], optional): File encoding for reading.
105
+ engine (Literal["python", "c"], optional): Parsing engine.
106
+ skip_blank_lines (bool, optional): If True, ignores blank lines.
107
+ on_bad_lines (Literal["error", "warn", "skip"], optional): Action on encountering bad lines.
108
+ quoting (Literal[0, 1, 2, 3], optional): Quoting style used in the CSV.
109
+ oriends (Literal["dict", "list", "series", "records", "index", "columns"], optional): Format for parsed data.
110
+ converters (Optional[Dict[str, Callable]], optional): Converters for specific columns.
111
+ true_values (Optional[List[str]], optional): Values to recognize as True.
112
+ false_values (Optional[List[str]], optional): Values to recognize as False.
113
+ na_values (Optional[List[str]], optional): Additional strings for NaN.
114
+ keep_default_na (bool, optional): If True, adds default NaN values.
115
+ na_filter (bool, optional): If True, detects NaN.
116
+ auto_delete (bool, optional): Deletes the file after processing if True.
117
+ """
118
+ self._file_system_manager = file_system_manager
119
+ self._white_exceptions = white_exceptions or []
120
+ self._separators = separators
121
+ self._header = header
122
+ self._batch_size = batch_size
123
+ self._encoding = encoding
124
+ self._engine = engine
125
+ self._skip_blank_lines = skip_blank_lines
126
+ self._on_bad_lines = on_bad_lines
127
+ self._oriends = oriends
128
+ self._quoting = quoting
129
+ self._converters = converters or {}
130
+ self._true_values = true_values or []
131
+ self._false_values = false_values or []
132
+ self._na_values = na_values or []
133
+ self._keep_default_na = keep_default_na
134
+ self._na_filter = na_filter
135
+ self._auto_delete = auto_delete
136
+
137
+ @staticmethod
138
+ def _get_encoding(content: bytes) -> Optional[str]:
139
+ """
140
+ Detects encoding of file content bytes.
141
+
142
+ Args:
143
+ content (bytes): Byte data to inspect.
144
+
145
+ Returns:
146
+ Optional[str]: Encoding if detected, else None.
147
+ """
148
+ result = chardet.detect(content)
149
+ encoding = result["encoding"]
150
+ return encoding
151
+
152
+ def _read_csv(
153
+ self, filename: str | PathLike[str], separator: str, **kwargs
154
+ ) -> pd.DataFrame:
155
+ """Reads a CSV file with a given separator.
156
+
157
+ Args:
158
+ filename (str): CSV file path.
159
+ separator (str): Delimiter for fields.
160
+
161
+ Returns:
162
+ pd.DataFrame: Parsed data in a DataFrame.
163
+ """
164
+ content = self._file_system_manager.read_file(filename)
165
+ if isinstance(content, str):
166
+ content = content.encode(FileEncodingEnum.UTF_8)
167
+ if bytes(separator, encoding=FileEncodingEnum.UTF_8) not in content:
168
+ raise pd.errors.ParserError(
169
+ f"Separator '{separator}' not found in the file content."
170
+ )
171
+ content_io = io.BytesIO(content)
172
+ return pd.read_csv(
173
+ content_io,
174
+ sep=separator,
175
+ encoding=self._encoding or self._get_encoding(content), # type: ignore
176
+ **kwargs,
177
+ )
178
+
179
+ @staticmethod
180
+ def _clean_column_names(df: pd.DataFrame) -> pd.DataFrame:
181
+ """
182
+ Cleans column names by stripping whitespace and converting to lowercase.
183
+
184
+ Args:
185
+ df (pd.DataFrame): DataFrame with columns to clean.
186
+
187
+ Returns:
188
+ pd.DataFrame: DataFrame with cleaned column names.
189
+ """
190
+ df.columns = df.columns.str.strip().str.lower()
191
+ return df
192
+
193
+ def _clean_dataframe(self, df: pd.DataFrame) -> pd.DataFrame:
194
+ """
195
+ Removes unnamed columns, drops all-NA rows and columns, and replaces NaN with None.
196
+
197
+ Args:
198
+ df (pd.DataFrame): DataFrame to clean.
199
+
200
+ Returns:
201
+ pd.DataFrame: Cleaned DataFrame.
202
+ """
203
+ df = self._clean_column_names(df)
204
+ df = df.loc[:, ~df.columns.str.contains("^Unnamed", case=False)]
205
+ df = (
206
+ df.dropna(how="all")
207
+ .dropna(axis=1, how="all")
208
+ .replace({np.nan: None})
209
+ )
210
+ return df
211
+
212
+ def process(
213
+ self, input_data: str | PathLike[str]
214
+ ) -> Generator[List[Dict[str, Any]], None, None]:
215
+ """
216
+ Parses the CSV file and yields formatted data as records.
217
+
218
+ Args:
219
+ input_data (str | PathLike[str]): CSV file path.
220
+
221
+ Yields:
222
+ Generator[List[Dict[str, Any]], None, None]: Parsed data as records.
223
+
224
+ Raises:
225
+ CsvParsingError: If parsing fails.
226
+ """
227
+ logger.debug(f"Start parsing: {input_data}")
228
+ suffix = Path(input_data).suffix
229
+ if suffix != ".csv":
230
+ raise CsvParsingError(
231
+ input_data,
232
+ f"File has an extension '{suffix}', expected .csv",
233
+ )
234
+ read_csv_kwargs = {
235
+ "header": self._header,
236
+ "chunksize": self._batch_size,
237
+ "engine": self._engine,
238
+ "skip_blank_lines": self._skip_blank_lines,
239
+ "on_bad_lines": self._on_bad_lines,
240
+ "quoting": self._quoting,
241
+ "converters": self._converters,
242
+ "true_values": self._true_values,
243
+ "false_values": self._false_values,
244
+ "na_values": self._na_values,
245
+ "keep_default_na": self._keep_default_na,
246
+ "na_filter": self._na_filter,
247
+ }
248
+
249
+ df = None
250
+ for sep in self._separators:
251
+ try:
252
+ df = self._read_csv(
253
+ input_data, separator=sep, **read_csv_kwargs
254
+ )
255
+ break
256
+ except pd.errors.ParserError as e:
257
+ logger.warning(
258
+ f"Failed to parse {input_data} with separator '{sep}': {e}"
259
+ )
260
+ continue
261
+
262
+ if df is None:
263
+ raise CsvParsingError(
264
+ input_data,
265
+ "Failed to parse with any of the provided separators.",
266
+ )
267
+
268
+ if self._batch_size is None:
269
+ df = self._clean_dataframe(df)
270
+ records = df.to_dict(orient=self._oriends) # type: ignore
271
+ if self._next is not None:
272
+ gen = self._next.process(records)
273
+ yield from gen
274
+ else:
275
+ yield records
276
+ else:
277
+ for chunk in df:
278
+ chunk = self._clean_dataframe(chunk) # type: ignore
279
+ if chunk.empty:
280
+ continue
281
+ records = chunk.to_dict(orient=self._oriends) # type: ignore
282
+ if self._next is not None:
283
+ gen = self._next.process(records)
284
+ try:
285
+ yield from gen
286
+ except ProcessorStopIteration:
287
+ ...
288
+ else:
289
+ yield records
290
+ if self._auto_delete:
291
+ self._file_system_manager.delete_file(input_data)
292
+ logger.debug(f"Successful parsing: {input_data}")
293
+
294
+ def __str__(self) -> str:
295
+ return (
296
+ f"{self.__class__.__name__}"
297
+ f"(auto_delete={self._auto_delete}, "
298
+ f"batch_size={self._batch_size}, "
299
+ f"separators={self._separators}, "
300
+ f"header={self._header}, "
301
+ f"encoding={self._encoding})"
302
+ )
@@ -0,0 +1,3 @@
1
+ from pipefy.etl.transform.json.reader import JsonParser
2
+
3
+ __all__ = ("JsonParser",)