ferrox-py-utils 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ferrox_py_utils/__init__.py +1 -0
- ferrox_py_utils/connectors/__init__.py +1 -0
- ferrox_py_utils/connectors/base.py +19 -0
- ferrox_py_utils/connectors/csv.py +35 -0
- ferrox_py_utils/connectors/s3.py +26 -0
- ferrox_py_utils/pipelines/__init__.py +1 -0
- ferrox_py_utils/pipelines/orchestrator.py +28 -0
- ferrox_py_utils/schemas/__init__.py +1 -0
- ferrox_py_utils/schemas/registry.py +25 -0
- ferrox_py_utils-1.0.0.dist-info/METADATA +67 -0
- ferrox_py_utils-1.0.0.dist-info/RECORD +12 -0
- ferrox_py_utils-1.0.0.dist-info/WHEEL +4 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# init
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# init
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import AsyncGenerator, Any
|
|
3
|
+
|
|
4
|
+
class DataConnector(ABC):
|
|
5
|
+
@abstractmethod
|
|
6
|
+
async def connect(self):
|
|
7
|
+
pass
|
|
8
|
+
|
|
9
|
+
@abstractmethod
|
|
10
|
+
async def extract(self, query: str = None) -> AsyncGenerator[Any, None]:
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
@abstractmethod
|
|
14
|
+
async def load(self, data: Any) -> bool:
|
|
15
|
+
pass
|
|
16
|
+
|
|
17
|
+
@abstractmethod
|
|
18
|
+
async def close(self):
|
|
19
|
+
pass
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
import aiofiles
|
|
3
|
+
from typing import AsyncGenerator, Any
|
|
4
|
+
from .base import DataConnector
|
|
5
|
+
from ferrox_py.core.provider import injectable
|
|
6
|
+
|
|
7
|
+
@injectable()
|
|
8
|
+
class CsvConnector(DataConnector):
|
|
9
|
+
def __init__(self, file_path: str):
|
|
10
|
+
self.file_path = file_path
|
|
11
|
+
self._file = None
|
|
12
|
+
|
|
13
|
+
async def connect(self):
|
|
14
|
+
# In a real app we'd keep it open for stream
|
|
15
|
+
pass
|
|
16
|
+
|
|
17
|
+
async def extract(self, query: str = None) -> AsyncGenerator[dict, None]:
|
|
18
|
+
async with aiofiles.open(self.file_path, mode='r', encoding='utf-8') as f:
|
|
19
|
+
header = None
|
|
20
|
+
async for line in f:
|
|
21
|
+
row = line.strip().split(",")
|
|
22
|
+
if not header:
|
|
23
|
+
header = row
|
|
24
|
+
continue
|
|
25
|
+
yield dict(zip(header, row))
|
|
26
|
+
|
|
27
|
+
async def load(self, data: Any) -> bool:
|
|
28
|
+
# Simplistic append
|
|
29
|
+
async with aiofiles.open(self.file_path, mode='a', encoding='utf-8') as f:
|
|
30
|
+
if isinstance(data, dict):
|
|
31
|
+
await f.write(",".join(str(v) for v in data.values()) + "\n")
|
|
32
|
+
return True
|
|
33
|
+
|
|
34
|
+
async def close(self):
|
|
35
|
+
pass
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from typing import AsyncGenerator, Any
|
|
2
|
+
from .base import DataConnector
|
|
3
|
+
from ferrox_py.core.provider import injectable
|
|
4
|
+
|
|
5
|
+
@injectable()
|
|
6
|
+
class S3Connector(DataConnector):
|
|
7
|
+
def __init__(self, bucket: str, path: str):
|
|
8
|
+
self.bucket = bucket
|
|
9
|
+
self.path = path
|
|
10
|
+
# Would inject aioboto3 session here
|
|
11
|
+
|
|
12
|
+
async def connect(self):
|
|
13
|
+
print(f"Connecting to S3 Bucket: {self.bucket}...")
|
|
14
|
+
pass
|
|
15
|
+
|
|
16
|
+
async def extract(self, query: str = None) -> AsyncGenerator[Any, None]:
|
|
17
|
+
print(f"Extracting streaming chunks from s3://{self.bucket}/{self.path}")
|
|
18
|
+
# Mock streaming chunks
|
|
19
|
+
yield {"chunk_id": 1, "data": b"mock_data"}
|
|
20
|
+
|
|
21
|
+
async def load(self, data: Any) -> bool:
|
|
22
|
+
print(f"Uploading chunk to s3://{self.bucket}/{self.path}")
|
|
23
|
+
return True
|
|
24
|
+
|
|
25
|
+
async def close(self):
|
|
26
|
+
pass
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# init
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from typing import Callable, List, Any
|
|
2
|
+
import asyncio
|
|
3
|
+
from ferrox_py.core.provider import injectable
|
|
4
|
+
from ferrox_py.core.errors import FerroxError
|
|
5
|
+
|
|
6
|
+
class PipelineStep:
|
|
7
|
+
def __init__(self, name: str, execute_fn: Callable[[Any], Any]):
|
|
8
|
+
self.name = name
|
|
9
|
+
self.execute_fn = execute_fn
|
|
10
|
+
|
|
11
|
+
@injectable()
|
|
12
|
+
class PipelineOrchestrator:
|
|
13
|
+
async def execute_pipeline(self, name: str, initial_data: Any, steps: List[PipelineStep]) -> Any:
|
|
14
|
+
print(f"Starting Pipeline: {name}")
|
|
15
|
+
current_data = initial_data
|
|
16
|
+
|
|
17
|
+
for step in steps:
|
|
18
|
+
print(f" -> Executing Step: {step.name}")
|
|
19
|
+
try:
|
|
20
|
+
if asyncio.iscoroutinefunction(step.execute_fn):
|
|
21
|
+
current_data = await step.execute_fn(current_data)
|
|
22
|
+
else:
|
|
23
|
+
current_data = step.execute_fn(current_data)
|
|
24
|
+
except Exception as e:
|
|
25
|
+
raise FerroxError(message=f"Pipeline '{name}' failed at step '{step.name}': {e}", status_code=500)
|
|
26
|
+
|
|
27
|
+
print(f"Pipeline '{name}' finished successfully.")
|
|
28
|
+
return current_data
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# init
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from typing import Dict, Type
|
|
2
|
+
from pydantic import create_model, BaseModel, ValidationError
|
|
3
|
+
from ferrox_py.core.provider import injectable
|
|
4
|
+
from ferrox_py.core.errors import FerroxError
|
|
5
|
+
|
|
6
|
+
@injectable()
|
|
7
|
+
class SchemaRegistry:
|
|
8
|
+
def __init__(self):
|
|
9
|
+
self._schemas: Dict[str, Type[BaseModel]] = {}
|
|
10
|
+
|
|
11
|
+
def register_schema(self, name: str, schema_def: Dict[str, Type]):
|
|
12
|
+
model = create_model(name, **schema_def)
|
|
13
|
+
self._schemas[name] = model
|
|
14
|
+
print(f"Schema '{name}' registered.")
|
|
15
|
+
|
|
16
|
+
def validate(self, name: str, data: dict) -> dict:
|
|
17
|
+
if name not in self._schemas:
|
|
18
|
+
raise FerroxError(message=f"Schema {name} not found", status_code=404)
|
|
19
|
+
|
|
20
|
+
model = self._schemas[name]
|
|
21
|
+
try:
|
|
22
|
+
instance = model(**data)
|
|
23
|
+
return instance.model_dump()
|
|
24
|
+
except ValidationError as e:
|
|
25
|
+
raise FerroxError(message=f"Schema validation failed: {e.errors()}", status_code=400)
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: ferrox-py-utils
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Data Engineering and ETL utilities for the Ferrox ecosystem.
|
|
5
|
+
Author: AI-Autistic-Intelligence
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Requires-Dist: ferrox-py>=1.0.0
|
|
8
|
+
Requires-Dist: pydantic>=2.0
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
|
|
11
|
+
# 🛠️ Ferrox-Py-Utils (Data Engineering)
|
|
12
|
+
|
|
13
|
+
## 1. Overview (What does this do?)
|
|
14
|
+
The `ferrox-py-utils` package is a specialized extension of the Ferrox ecosystem dedicated to data manipulation, ETL (Extract, Transform, Load) pipelines, and cross-platform data movement. It provides agnostic `Connectors` (e.g., for AWS S3, CSV files, or REST APIs) and a `PipelineOrchestrator` to seamlessly sequence data transformation jobs without writing monolithic scripts.
|
|
15
|
+
|
|
16
|
+
## 2. Philosophy (Why does it exist?)
|
|
17
|
+
Data engineering often suffers from "wild copy-pasting" where scripts to import users or export CSVs are hastily written and tightly coupled to specific database schemas or cloud vendors. The philosophy here is absolute **agnosticism**. Instead of building monolithic ETL scripts, this package enforces a modular approach where small, isolated tasks are injected into an orchestrator. This allows data engineers to reuse extraction and validation logic across entirely different projects.
|
|
18
|
+
|
|
19
|
+
## 3. Target Audience (Who is it for?)
|
|
20
|
+
This package is built for Data Engineers and backend developers who need to quickly stand up a Data Platform for ingestion, parsing, and bulk loading of large datasets (like CSV or JSON) into Data Lakes or Object Storage (such as Amazon S3, MinIO, or Google Cloud Storage).
|
|
21
|
+
|
|
22
|
+
## 4. Architecture (How does it work?)
|
|
23
|
+
The data architecture rests on three pillars:
|
|
24
|
+
- **Connectors**: Classes inheriting from an abstract `BaseConnector` to standardize streaming I/O operations (Read/Write/Delete), entirely isolating the pipeline from the specific storage vendor.
|
|
25
|
+
- **PipelineOrchestrator**: A linear execution engine that passes a shared state (`context`) through a sequence of nodes (steps), ensuring proper error isolation and retry mechanics.
|
|
26
|
+
- **Schema Registry**: Native integration with Pydantic to register and enforce strict validation on datasets in transit, ensuring corrupted data never enters the database.
|
|
27
|
+
|
|
28
|
+
## 5. Installation / Setup
|
|
29
|
+
Ensure you are using Python 3.11+. The package installs basic dependencies, but you may need to install specific data drivers (like `boto3` or `pandas`) depending on the connectors you intend to use.
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install ferrox-py-utils
|
|
33
|
+
# Optional extensions:
|
|
34
|
+
# pip install boto3 pandas
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## 6. Quickstart (Usage)
|
|
38
|
+
```python
|
|
39
|
+
from ferrox_py_utils.connectors.csv import CsvConnector
|
|
40
|
+
from ferrox_py_utils.pipelines.orchestrator import PipelineOrchestrator
|
|
41
|
+
|
|
42
|
+
# 1. Setup the agnostic connector
|
|
43
|
+
csv_connector = CsvConnector(file_path="/tmp/data.csv")
|
|
44
|
+
|
|
45
|
+
# 2. Define isolated pipeline steps
|
|
46
|
+
def step_read_csv(ctx):
|
|
47
|
+
ctx['data'] = csv_connector.read()
|
|
48
|
+
return ctx
|
|
49
|
+
|
|
50
|
+
def step_transform(ctx):
|
|
51
|
+
# Transform logic here...
|
|
52
|
+
ctx['data'] = [row for row in ctx['data'] if row.get("active")]
|
|
53
|
+
return ctx
|
|
54
|
+
|
|
55
|
+
# 3. Orchestrate and execute
|
|
56
|
+
orchestrator = PipelineOrchestrator()
|
|
57
|
+
orchestrator.add_step("Read Data", step_read_csv)
|
|
58
|
+
orchestrator.add_step("Clean Data", step_transform)
|
|
59
|
+
|
|
60
|
+
final_context = orchestrator.execute()
|
|
61
|
+
print(f"Processed {len(final_context['data'])} records.")
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## 7. Ecosystem Integration
|
|
65
|
+
This module is fully integrated with the core `ferrox-py` ecosystem:
|
|
66
|
+
- **Core (IoC Container)**: The framework's Dependency Injection container is used to instantiate Connectors globally as Singletons, meaning you only initialize your S3 credentials once.
|
|
67
|
+
- **AuthModule (ferrox-py-auth)**: RBAC can be utilized to restrict which users or system roles are authorized to trigger specific data pipelines via the API Gateway.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
ferrox_py_utils/__init__.py,sha256=4zALCCxfSyzOKNmB_Y-R1A0iUhPpCOrKbkJ1srJWyxo,7
|
|
2
|
+
ferrox_py_utils/connectors/__init__.py,sha256=4zALCCxfSyzOKNmB_Y-R1A0iUhPpCOrKbkJ1srJWyxo,7
|
|
3
|
+
ferrox_py_utils/connectors/base.py,sha256=gA-xh0uE2SzpmxjSLllqL6Bd0akgxPOXODji_9ZW5UU,423
|
|
4
|
+
ferrox_py_utils/connectors/csv.py,sha256=qINN5jE9SQm7GhV_4K4mVz5T0N6c_N6Z-8hb4-5C7YA,1132
|
|
5
|
+
ferrox_py_utils/connectors/s3.py,sha256=zWF_KD6nBPqSgHB3QRM1AH6ItcIRfqFWxJkQ4YjqAMk,836
|
|
6
|
+
ferrox_py_utils/pipelines/__init__.py,sha256=4zALCCxfSyzOKNmB_Y-R1A0iUhPpCOrKbkJ1srJWyxo,7
|
|
7
|
+
ferrox_py_utils/pipelines/orchestrator.py,sha256=L0cmygKsxEOpuhJKp83f2onFLS_ZgcrWP1RuaTjcq0U,1109
|
|
8
|
+
ferrox_py_utils/schemas/__init__.py,sha256=4zALCCxfSyzOKNmB_Y-R1A0iUhPpCOrKbkJ1srJWyxo,7
|
|
9
|
+
ferrox_py_utils/schemas/registry.py,sha256=AbsLqcWWIEQ1xFFV7Yk33qo_jcpDb4wqVvPu9xOFbq8,956
|
|
10
|
+
ferrox_py_utils-1.0.0.dist-info/METADATA,sha256=l16aWVd4f2RW06Q0G-Ds71N-JWyh04-jb4SpqhJWyEY,3783
|
|
11
|
+
ferrox_py_utils-1.0.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
12
|
+
ferrox_py_utils-1.0.0.dist-info/RECORD,,
|