acelerai 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,15 @@
1
+ Metadata-Version: 2.2
2
+ Name: acelerai
3
+ Version: 0.0.1
4
+ Summary: short package description
5
+ Author: DaniloAraneda
6
+ Author-email: danilo@alert2gain.com
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Classifier: Operating System :: OS Independent
10
+ Requires-Python: >=3.10.7
11
+ Dynamic: author
12
+ Dynamic: author-email
13
+ Dynamic: classifier
14
+ Dynamic: requires-python
15
+ Dynamic: summary
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,20 @@
1
+ import setuptools
2
+
3
+
4
+ setuptools.setup(
5
+ name = "acelerai",
6
+ version = "0.0.1",
7
+ author = "DaniloAraneda",
8
+ author_email = "danilo@alert2gain.com",
9
+ description = "short package description",
10
+ classifiers = [
11
+ "Programming Language :: Python :: 3",
12
+ "License :: OSI Approved :: MIT License",
13
+ "Operating System :: OS Independent",
14
+ ],
15
+ package_dir = {"": "src"},
16
+ packages = setuptools.find_packages(where="src"),
17
+ include_dirs=[],
18
+ python_requires = ">=3.10.7",
19
+ requires=[]
20
+ )
@@ -0,0 +1,15 @@
1
+ Metadata-Version: 2.2
2
+ Name: acelerai
3
+ Version: 0.0.1
4
+ Summary: short package description
5
+ Author: DaniloAraneda
6
+ Author-email: danilo@alert2gain.com
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Classifier: Operating System :: OS Independent
10
+ Requires-Python: >=3.10.7
11
+ Dynamic: author
12
+ Dynamic: author-email
13
+ Dynamic: classifier
14
+ Dynamic: requires-python
15
+ Dynamic: summary
@@ -0,0 +1,10 @@
1
+ setup.py
2
+ src/acelerai.egg-info/PKG-INFO
3
+ src/acelerai.egg-info/SOURCES.txt
4
+ src/acelerai.egg-info/dependency_links.txt
5
+ src/acelerai.egg-info/top_level.txt
6
+ src/acelerai_inputstream/__init__.py
7
+ src/acelerai_inputstream/inputstream.py
8
+ src/acelerai_inputstream/inputstream_client.py
9
+ test/test.py
10
+ test/test_app.py
@@ -0,0 +1 @@
1
+ acelerai_inputstream
@@ -0,0 +1,39 @@
1
+ import json
2
+ import os
3
+ from acelerai_inputstream.inputstream import INSERTION_MODE, Inputstream
4
+ from acelerai_inputstream.inputstream_client import InputstreamClient
5
+
6
+ __results_path = os.environ.get("A2G_RESULT_PATH","a2g_results")
7
+ __payload_path = os.environ.get("A2G_PAYLOAD_PATH", "payload.json")
8
+
9
+ __mode = os.environ.get("EXEC_LOCATION", "LOCAL")
10
+
11
+ def save_result(key:str, value, path = None):
12
+ """
13
+ Save the result in the file
14
+ :param key: The key to be used to save the result
15
+ :param value: The value to be saved
16
+ :param path: The path to save the result, if None, the default path is used
17
+ """
18
+ result_path = __results_path
19
+ if path is not None and __mode == "LOCAL":
20
+ result_path = path
21
+
22
+ if __mode == "LOCAL":
23
+ if not os.path.exists(result_path): os.makedirs(result_path)
24
+
25
+ open(f"{result_path}/{key}", 'w+').write(json.dumps(value))
26
+
27
+ def get_payload(path = None) -> dict | None:
28
+ """
29
+ Get the payload from the file, if the file does not exist, return None
30
+ :param path: The path to the payload file, if None, the default path is used
31
+ """
32
+ payload_path = __payload_path
33
+ if path is not None and __mode == "LOCAL":
34
+ payload_path = path
35
+
36
+ if not os.path.exists(payload_path): return None
37
+ return json.loads(open(payload_path).read())
38
+
39
+
@@ -0,0 +1,263 @@
1
+ import copy
2
+ from datetime import datetime
3
+ from enum import Enum
4
+ from uuid import UUID
5
+ from dateutil import parser
6
+
7
+
8
+ # Enums
9
+ class FileIndexFieldType(Enum):
10
+ Datetime = 0
11
+ String = 1
12
+ Number = 2
13
+ Integer = 3
14
+
15
+ class DateBucketSize(Enum):
16
+ Minute = 0
17
+ Hour = 1
18
+ Day = 2
19
+ Week = 3
20
+ Month = 4
21
+ Year = 5
22
+
23
+ class InputstreamStatus(Enum):
24
+ ToDiscover = 0
25
+ Undiscovered = 1
26
+ Exposed = 2
27
+ ToDiscoverAgain = 3
28
+
29
+ class InputstreamStorage(Enum):
30
+ Collection = 0
31
+ TimeSeriesCollection = 1
32
+ File = 2
33
+
34
+ class InputstreamProtocol(Enum):
35
+ MQTT = 0
36
+ HTTP = 1
37
+ BOTH = 2
38
+
39
+ class RealTimeMode(Enum):
40
+ OFF = 0
41
+ ON = 1
42
+
43
+ class IndexType(Enum):
44
+ Unique = 0
45
+ Search = 1
46
+
47
+ class SortType(Enum):
48
+ Ascending = 0
49
+ Descending = 1
50
+
51
+ class SourceType(Enum):
52
+ MySQL=0
53
+ MongoDB=1
54
+ SQLServer=2
55
+ Snowflake=3
56
+ Oracle=4
57
+ PostgresSQL=5
58
+ Firebase=6
59
+ BigQuery=7
60
+
61
+ class DataType(Enum):
62
+ TypeNumber = 0
63
+ TypeBoolean = 1
64
+ TypeString = 2
65
+ TypeDateType = 3
66
+ TypeDate = 4
67
+ TypeList = 5
68
+
69
+ class InputstreamType(Enum):
70
+ InSystem = 0
71
+ Native = 1
72
+
73
+ # Models
74
+ class DynamicField:
75
+ DataType: DataType
76
+ Field: str
77
+ ValueDefault: str
78
+
79
+ def __init__(self, **kwargs) -> None:
80
+ kwargs["DataType"] = DataType(kwargs.pop('DataType'))
81
+ self.__dict__ = kwargs
82
+
83
+ def get_dict(self):
84
+ data = copy.deepcopy(self.__dict__)
85
+ return data
86
+
87
+ class DataConnection:
88
+ ConnectionString: str
89
+ DataSourceName: str
90
+ Query : str
91
+ SourceType: SourceType
92
+ DynamicFields: list[DynamicField]
93
+ DatabaseName: str
94
+
95
+ def __init__(self,**kwargs)-> None:
96
+ kwargs["SourceType"] = SourceType(kwargs.pop('SourceType'))
97
+ self.__dict__ = kwargs
98
+
99
+ def get_dict(self):
100
+ data = copy.deepcopy(self.__dict__)
101
+ return data
102
+
103
+
104
+ class IndexField:
105
+ Name: str
106
+ FieldType: FileIndexFieldType
107
+ DoubleBucketSize: float
108
+ DateBucketSize: DateBucketSize
109
+
110
+ def __init__(self, from_response=False, **kwargs):
111
+ if from_response:
112
+ kwargs["Name"] = kwargs.pop('name')
113
+ kwargs["DoubleBucketSize"] = kwargs.pop('doubleBucketSize')
114
+ kwargs["FieldType"] = FileIndexFieldType(kwargs.pop('fieldType'))
115
+ kwargs["DateBucketSize"] = DateBucketSize(kwargs.pop('dateBucketSize'))
116
+ else:
117
+ kwargs["FieldType"] = FileIndexFieldType(kwargs.pop('FieldType'))
118
+ kwargs["DateBucketSize"] = DateBucketSize(kwargs.pop('DateBucketSize'))
119
+ self.__dict__ = kwargs
120
+
121
+ def get_dict(self):
122
+ data = copy.deepcopy(self.__dict__)
123
+ return data
124
+
125
+
126
+ class CollectionIndexField:
127
+ Name: str
128
+ SortType: SortType
129
+
130
+ def __init__(self, from_response=False, **kwargs):
131
+ if from_response:
132
+ kwargs["Name"] = kwargs.pop('name')
133
+ kwargs["SortType"] = SortType(kwargs.pop('sortType'))
134
+ else:
135
+ kwargs["SortType"] = SortType(kwargs["SortType"])
136
+ self.__dict__ = kwargs
137
+
138
+ def get_dict(self):
139
+ data = copy.deepcopy(self.__dict__)
140
+ return data
141
+
142
+
143
+ class CollectionIndex:
144
+ Name: str
145
+ Fields: list[CollectionIndexField]
146
+ Size: int
147
+ IndexUse: int
148
+ SinceUse: datetime
149
+ IndexType: IndexType
150
+ DateCreated: datetime
151
+ IsCompound: bool
152
+
153
+ def __init__(self, from_response=False, **kwargs):
154
+ if from_response:
155
+ kwargs["Name"] = kwargs.pop('name')
156
+ kwargs["DateCreated"] = kwargs.pop('dateCreated')
157
+ kwargs["Fields"] = [CollectionIndexField(from_response=True, **x) for x in kwargs.pop('fields')]
158
+ kwargs["IndexType"] = IndexType(kwargs.pop('indexType'))
159
+ else:
160
+ kwargs["DateCreated"] = kwargs.pop('DateCreated')
161
+ kwargs["Fields"] = [CollectionIndexField(**x) for x in kwargs["Fields"]]
162
+ kwargs["IndexType"] = IndexType(kwargs.pop('IndexType'))
163
+ self.__dict__ = kwargs
164
+
165
+ def get_dict(self):
166
+ data = copy.deepcopy(self.__dict__)
167
+ data["Fields"] = [x.get_dict() for x in self.Fields]
168
+ return data
169
+
170
+ class Inputstream:
171
+ Id: UUID
172
+ SubscriptionId: UUID
173
+ Name: str
174
+ CollectionName: str
175
+ Schema: str
176
+ SchemaSample: str
177
+ SampleDate: datetime
178
+ Status: InputstreamStatus
179
+ InputstreamType: InputstreamType
180
+ DataConnection: DataConnection
181
+ Tags: list[str]
182
+ Ikey: str
183
+ CollectionIndexes: list[CollectionIndex]
184
+ FilesIndex: list[IndexField]
185
+ Storage: InputstreamStorage
186
+ Protocol: InputstreamProtocol
187
+ RealTimeMode: RealTimeMode
188
+ Size: int
189
+ MaxNDocsByFile: int
190
+ AllowAnyOrigin: bool
191
+ FileConsolidatorCron: str
192
+ Removed: bool
193
+ CreatedOn: datetime
194
+ RemovedOn: datetime | None
195
+
196
+
197
+ def __init__(self, from_response = False, **kwargs):
198
+ if from_response: self.from_response(**kwargs)
199
+ else:
200
+ kwargs["Id"] = UUID(kwargs.pop('Id'))
201
+
202
+ kwargs["Status"] = InputstreamStatus(kwargs.pop('Status'))
203
+ kwargs["DataConnection"] = DataConnection(**kwargs["DataConnection"])
204
+ kwargs["InputstreamType"] = InputstreamType(kwargs.pop('InputstreamType'))
205
+ kwargs["Storage"] = InputstreamStorage(kwargs.pop('Storage'))
206
+ kwargs["Protocol"] = InputstreamProtocol(kwargs.pop('Protocol'))
207
+ kwargs["RealTimeMode"] = RealTimeMode(kwargs.pop('RealTimeMode'))
208
+
209
+ kwargs["FilesIndex"] = [IndexField(**x) for x in kwargs["FilesIndex"]]
210
+ kwargs["CollectionIndexes"] = [CollectionIndex(**x) for x in kwargs["CollectionIndexes"]]
211
+
212
+ self.__dict__ = kwargs
213
+
214
+
215
+ def from_response(self, **kwargs) -> None:
216
+ kwargs["Id"] = UUID(kwargs.pop('id'))
217
+ kwargs["SubscriptionId"] = UUID(kwargs.pop('subscriptionId'))
218
+
219
+ kwargs["Name"] = kwargs.pop('name')
220
+ kwargs["CollectionName"] = kwargs.pop('collectionName')
221
+ kwargs["Schema"] = kwargs.pop('schema')
222
+ kwargs["SchemaSample"] = kwargs.pop('schemaSample')
223
+ kwargs["Tags"] = kwargs.pop('tags')
224
+ kwargs["Ikey"] = kwargs.pop('ikey')
225
+ kwargs["Size"] = kwargs.pop('size')
226
+ kwargs["MaxNDocsByFile"] = kwargs.pop('maxNDocsByFile')
227
+ kwargs["AllowAnyOrigin"] = kwargs.pop('allowAnyOrigin')
228
+ kwargs["FileConsolidatorCron"] = kwargs.pop('fileConsolidatorCron')
229
+ kwargs["Removed"] = kwargs.pop('removed')
230
+
231
+ kwargs["Status"] = InputstreamStatus(kwargs.pop('status'))
232
+ kwargs["Storage"] = InputstreamStorage(kwargs.pop('storage'))
233
+ kwargs["Protocol"] = InputstreamProtocol(kwargs.pop('protocol'))
234
+ kwargs["RealTimeMode"] = RealTimeMode(kwargs.pop('realTimeMode'))
235
+ kwargs["InputstreamType"] = InputstreamType(kwargs.pop('inputstreamType'))
236
+
237
+ kwargs["FilesIndex"] = [IndexField(from_response=True,**x) for x in kwargs.pop("filesIndex")]
238
+ kwargs["CollectionIndexes"] = [CollectionIndex(from_response=True, **x) for x in kwargs.pop("collectionIndexes")]
239
+
240
+ kwargs["SampleDate"] = kwargs.pop('sampleDate')
241
+ kwargs["CreatedOn"] = kwargs.pop('createdOn')
242
+ kwargs["RemovedOn"] = kwargs.pop('removedOn')
243
+
244
+ self.__dict__ = kwargs
245
+
246
+ def get_dict(self):
247
+ data = copy.deepcopy(self.__dict__)
248
+ data["Id"] = str(data["Id"])
249
+ data["FilesIndex"] = [x.get_dict() for x in self.FilesIndex]
250
+ data["CollectionIndexes"] = [x.get_dict() for x in self.CollectionIndexes]
251
+ return data
252
+
253
+
254
+ class INSERTION_MODE(Enum):
255
+ """
256
+ Enum for insertion modes, available modes:
257
+ REPLACE: if a document collides with an existing document by a unique index, the existing document is replaced with the new document. Otherwise, the new document is inserted.
258
+ INSERT_UNORDERED: insert all documents that did not have a collision with an existing document by a unique index.
259
+ TRANSACTION: insert all documents in a transaction, if a document collides with an existing document by a unique index, the transaction is aborted and no document is inserted.
260
+ """
261
+ REPLACE = 0
262
+ INSERT_UNORDERED = 1
263
+ TRANSACTION = 2
@@ -0,0 +1,706 @@
1
+ import logging
2
+ logger = logging.getLogger("Inputstream")
3
+
4
+ import asyncio
5
+ from datetime import datetime, timedelta, date
6
+ from enum import Enum
7
+ import gzip
8
+ import hashlib
9
+ import os
10
+ import json
11
+ import struct
12
+ from uuid import UUID
13
+ import requests
14
+ from acelerai_inputstream.inputstream import INSERTION_MODE, Inputstream, InputstreamStatus, InputstreamType
15
+ #from jsonschema import Draft4Validator
16
+ import fastjsonschema
17
+ from dateutil import parser
18
+ import httpx
19
+ import gzip
20
+ from decimal import Decimal
21
+ import msgpack
22
+
23
+ global __mode
24
+ __mode = os.environ.get("EXEC_LOCATION", "LOCAL")
25
+ SEM = asyncio.Semaphore(20) # Limitar concurrencia a 20 conexiones
26
+
27
+ def custom_encoder(obj):
28
+ """Convierte tipos no serializables como datetime y Decimal."""
29
+ if isinstance(obj, datetime):
30
+ return obj.isoformat() # Serializar datetime como cadena ISO 8601
31
+ if isinstance(obj, Decimal):
32
+ return float(obj) # Serializar Decimal como flotante
33
+ raise TypeError(f"Object of type {type(obj).__name__} is not serializable")
34
+
35
+ def decode_datetime(obj):
36
+ """Deserializa cadenas ISO 8601 a objetos datetime."""
37
+ for key, value in obj.items():
38
+ if isinstance(value, str):
39
+ try:
40
+ obj[key] = datetime.fromisoformat(value) # Deserializar datetime
41
+ except ValueError:
42
+ pass
43
+ elif isinstance(value, float):
44
+ obj[key] = Decimal(value) # Convertir flotantes de regreso a Decimal
45
+ return obj
46
+
47
+ def load_full_object(file_path):
48
+ """Carga completamente el objeto desde un archivo MessagePack en memoria."""
49
+ try:
50
+ with open(file_path, "rb") as file:
51
+ # Cargar todos los registros en memoria como una lista
52
+ unpacker = msgpack.Unpacker(file, raw=False)
53
+ data = [record for record in unpacker] # Deserializar todos los registros
54
+ return data
55
+ except Exception as e:
56
+ logger.error(f"Error al cargar el archivo: {e}", exc_info=True)
57
+ return None
58
+
59
+ if __mode != "LOCAL":
60
+ # Environment production
61
+ DATA_URL = os.environ.get("DATA_URL" , "https://stream.aceler.ai")
62
+ QUERY_MANAGER = os.environ.get("QUERY_MANAGER" , "https://stream.aceler.ai")
63
+ INPUTSTREAM_URL = os.environ.get("INPUTSTREAM_URL" , "https://apigw.aceler.ai")
64
+ verify_https = True
65
+ else:
66
+ # Environment development
67
+ DATA_URL = os.environ.get("DATA_URL", "https://localhost:1008")
68
+ QUERY_MANAGER = os.environ.get("QUERY_MANAGER", "https://localhost:8000")
69
+ INPUTSTREAM_URL = os.environ.get("INPUTSTREAM_URL", "https://localhost:1006")
70
+ verify_https = False
71
+
72
+ packer = msgpack.Packer(default=custom_encoder) # Configurar el hook de serialización
73
+
74
+
75
+ class CustomJSONEncoder(json.JSONEncoder):
76
+ def default(self, obj):
77
+ if isinstance(obj, Enum):
78
+ return obj.value
79
+
80
+ elif isinstance(obj, datetime):
81
+ return obj.isoformat()
82
+
83
+ elif isinstance(obj,date):
84
+ return obj.isoformat()
85
+
86
+ elif isinstance(obj, UUID):
87
+ return str(obj)
88
+ else:
89
+ return super().default(obj)
90
+
91
+
92
+ class CustomJsonDecoder(json.JSONDecoder):
93
+ def __init__(self, *args ,**kargs):
94
+ json.JSONDecoder.__init__(self, object_hook=self.object_hook, *args, **kargs)
95
+
96
+ def object_hook(self, obj:dict):
97
+ for k, v in obj.items():
98
+ if isinstance(v, str) and 'T' in v and '-' in v and ':' in v and len(v) < 40:
99
+ try:
100
+ dv = parser.parse(v)
101
+ dt = dv.replace(tzinfo=None)
102
+ obj[k] = dt
103
+ except:
104
+ pass
105
+ elif isinstance(v, str) and '-' in v and len(v) < 11:
106
+ try:
107
+ obj[k] = parser.parse(v).date()
108
+ except:
109
+ pass
110
+ return obj
111
+
112
+
113
+ class CacheManager:
114
+ duration_inputstream:int
115
+ duration_data:int
116
+
117
+ def __init__(self, cache_options: dict | None = None):
118
+ if cache_options is None:
119
+ self.duration_data = 60 * 24
120
+ self.duration_inputstream = 60 * 24
121
+ else:
122
+ self.duration_data = cache_options.get("duration_data", 60 * 24)
123
+ self.duration_inputstream = cache_options.get("duration_inputstream", 60 * 24)
124
+
125
+ # create cache directories
126
+ if not os.path.exists(".acelerai_cache"):
127
+ os.mkdir(".acelerai_cache")
128
+ os.mkdir(".acelerai_cache/data")
129
+
130
+ def get_inputstream(self, ikey:str) -> Inputstream | None:
131
+ """
132
+ return Inputstream if exists in cache and is not expired
133
+ otherwise return None
134
+ params:
135
+ ikey: str
136
+ """
137
+ file_name = f".acelerai_cache/inputstreams/{ikey}.json"
138
+ if os.path.exists(file_name):
139
+ logger.info(f"Inputstream - Ikey: {ikey}, Checking cache expiration...")
140
+ data = json.loads(open(file_name, "r").read(), cls=CustomJsonDecoder)
141
+ if datetime.utcnow() < data["duration"]:
142
+ logger.info(f"Inputstream - Ikey: {ikey}, from cache")
143
+ return Inputstream(**data["inputstream"])
144
+ else:
145
+ logger.info(f"Inputstream - Ikey: {ikey}, Cache expired, removing file...")
146
+ os.remove(file_name)
147
+ return None
148
+ return None
149
+
150
+ def set_inputstream(self, inputstream:Inputstream):
151
+ cache_register = {
152
+ "inputstream": inputstream.get_dict(),
153
+ "duration": datetime.utcnow() + timedelta(minutes=self.duration_inputstream)
154
+ }
155
+
156
+ file_name = inputstream.Ikey
157
+ if not os.path.exists(f".acelerai_cache/inputstreams/"): os.mkdir(f".acelerai_cache/inputstreams/")
158
+ open(f".acelerai_cache/inputstreams/{file_name}.msgpack", "w+").write(json.dumps(cache_register, cls=CustomJSONEncoder))
159
+
160
+ def get_data(self, ikey:str, hash_query:str) -> list[dict] | None:
161
+ """
162
+ return data if exists in cache and is not expired
163
+ otherwise return None
164
+ params:
165
+ ikey: str
166
+ query: dict
167
+ """
168
+ file_name = f".acelerai_cache/data/{ikey}/{hash_query}.msgpack"
169
+ index_ttl_file = f".acelerai_cache/data/ttl_index.json"
170
+ if os.path.exists(file_name) and os.path.exists(index_ttl_file):
171
+
172
+ # check if cache is expired
173
+ logger.info(f"Data - Ikey: {ikey}, Checking cache expiration...")
174
+ index = json.loads(open(index_ttl_file, "r").read(), cls=CustomJsonDecoder)
175
+ ttl_key = f"{ikey}_{hash_query}"
176
+ if ttl_key in index:
177
+ ttl = index[ttl_key]
178
+ if datetime.utcnow() > ttl:
179
+ logger.info(f"Data - Ikey: {ikey}, Cache expired, removing file...")
180
+ os.remove(file_name)
181
+ return None
182
+
183
+ # recover data from cache
184
+ try:
185
+ logger.info(f"Data - Ikey: {ikey}, Recovering data from cache...")
186
+ data = load_full_object(file_name)
187
+ logger.info(f"Data - Ikey: {ikey}, from cache")
188
+ return data
189
+ except Exception as e:
190
+ if os.path.exists(file_name): os.remove(file_name)
191
+ raise Exception(f"Error reading cache file: {file_name} - {e}", stack_info=True)
192
+ else:
193
+ if os.path.exists(file_name): os.remove(file_name)
194
+ return None
195
+
196
+ def set_data(self, ikey:str, hash_query:str):
197
+ # update ttl index
198
+ ttl_key = f"{ikey}_{hash_query}"
199
+ ttl = datetime.utcnow() + timedelta(minutes=self.duration_data)
200
+ index_file = f".acelerai_cache/data/ttl_index.json"
201
+ if os.path.exists(index_file):
202
+ index = json.loads(open(index_file, "r").read(), cls=CustomJsonDecoder)
203
+ index[ttl_key] = ttl
204
+ open(index_file, "w+").write(json.dumps(index, cls=CustomJSONEncoder))
205
+ else:
206
+ open(index_file, "w+").write(json.dumps({ttl_key: ttl}, cls=CustomJSONEncoder))
207
+
208
+
209
+ class AcelerAIHttpClient():
210
+
211
+ def __init__(self, token:str):
212
+ self.token = token
213
+ self.lock = asyncio.Lock()
214
+
215
+ def get_inputstream_by_ikey(self, ikey:str) -> Inputstream:
216
+ try:
217
+ headers = { "Authorization": f"A2G {self.token}"}
218
+ # proxies = {'https': 'http://127.0.0.1:1000'}
219
+
220
+ res = requests.get(INPUTSTREAM_URL + f"/Inputstream/Ikey/{ikey}", headers=headers, verify=verify_https)
221
+ logger.info(f"Getting inputstream with ikey: {ikey} from ACELER.AI...")
222
+ if res.status_code != 200:
223
+ if res.status_code == 404: raise Exception("Inputstream not found, please check your ikey")
224
+ if res.status_code == 401: raise Exception("Unauthorized: please check your token or access permissions")
225
+ if res.status_code == 403: raise Exception("Forbidden: please check your access permissions")
226
+ raise Exception(f"Error getting inputstream, {res.status_code} {res.text}")
227
+ content = res.json(cls=CustomJsonDecoder)
228
+ if not content["success"]: raise Exception(content["errorMessage"])
229
+ return Inputstream(from_response=True, **content["data"])
230
+ except Exception as e:
231
+ raise e
232
+
233
+ async def _write_to_file(self, ikey, query, data):
234
+ query_str = json.dumps(query, cls=CustomJSONEncoder)
235
+ query_hash = hashlib.sha256(query_str.encode()).hexdigest()
236
+
237
+ # save data
238
+ file_name = f".acelerai_cache/data/{ikey}/{query_hash}.msgpack"
239
+
240
+ async with self.lock: # Garantiza que solo una tarea escriba a la vez
241
+ with open(file_name, "ab") as file:
242
+ for record in data:
243
+ file.write(packer.pack(record))
244
+
245
+ async def fetch_page(self, ikey:str, query:dict,delete_id:bool=True, page:int=1, page_size:int=1000):
246
+ async with SEM:
247
+ headers = {
248
+ "Authorization": f"A2G {self.token}",
249
+ "ikey": ikey,
250
+ 'Content-Type': 'application/json'
251
+ }
252
+
253
+ my_body = {
254
+ "delete_id": delete_id,
255
+ "query": json.dumps(query, cls=CustomJSONEncoder)
256
+ }
257
+
258
+ buffer = b"" # Buffer para ensamblar datos incompletos
259
+ s = datetime.utcnow()
260
+ timeout = httpx.Timeout(60.0, connect=10.0, read=600.0)
261
+
262
+ async with httpx.AsyncClient(http2=True, verify=False, timeout=timeout) as client:
263
+ async with client.stream("POST", f"{QUERY_MANAGER}/QueryData/Find",
264
+ json=my_body,
265
+ headers=headers,
266
+ params={"page": page, "page_size": page_size }) as response:
267
+
268
+ if response.status_code != 200:
269
+ msg = ''
270
+ async for chunk in response.aiter_bytes():
271
+ if chunk:
272
+ msg += chunk.decode("utf-8")
273
+
274
+ raise Exception(f"{response.status_code} {msg}")
275
+
276
+ async for chunk in response.aiter_bytes():
277
+ buffer += chunk # Agregar los datos al buffer
278
+ while len(buffer) >= 4: # Asegurarse de que al menos 4 bytes están disponibles
279
+ obj_length = struct.unpack(">I", buffer[:4])[0]
280
+
281
+ if len(buffer) < 4 + obj_length:
282
+ break # Esperar más datos si el objeto no está completo
283
+
284
+ obj_data = buffer[4: 4 + obj_length]
285
+ buffer = buffer[4 + obj_length:] # Actualizar el buffer
286
+ decompressed_data = gzip.decompress(obj_data) # Descomprimir los datos
287
+ data = msgpack.unpackb(decompressed_data, object_hook=decode_datetime) # Deserializar el objeto
288
+
289
+ await self._write_to_file(ikey, query, data)
290
+
291
+ logger.info(f"Page {page} downloaded")
292
+
293
+ def find_one(self, ikey:str, query:dict) -> list[dict]:
294
+ try:
295
+ headers = {
296
+ "Authorization": f"A2G {self.token}",
297
+ "ikey": ikey,
298
+ 'Content-Type': 'application/json'
299
+ }
300
+ res = requests.post(QUERY_MANAGER + "/QueryData/FindOne",
301
+ data=json.dumps(query, cls=CustomJSONEncoder),
302
+ headers=headers,
303
+ verify=verify_https
304
+ )
305
+ if res.status_code != 200: raise Exception(f"Error getting inputstream {res.status_code} {res.content}")
306
+ content = res.json(cls=CustomJsonDecoder)
307
+ if not content["success"]: raise Exception(content["errorMessage"])
308
+ return content["data"]
309
+ except Exception as e:
310
+ raise e
311
+
312
+ def aggregate(self, ikey:str, pipeline: list[dict]) -> list[dict]:
313
+ try:
314
+ headers = {
315
+ "Authorization": f"A2G {self.token}",
316
+ "ikey": ikey,
317
+ 'Content-Type': 'application/json'
318
+ }
319
+
320
+ if not all(isinstance(x, dict) for x in pipeline): raise Exception("Invalid pipeline, the steps must be dictionaries")
321
+ if len(pipeline) == 0: raise Exception("Invalid pipeline, length must be greater than 0" )
322
+ if any("$out" in x or "$merge" in x for x in pipeline): raise Exception("Invalid pipeline, write operations not allowed" )
323
+
324
+ res = requests.post(f"{QUERY_MANAGER}/QueryData/ExecutionPlanningAggregate",
325
+ data = json.dumps(pipeline, cls=CustomJSONEncoder),
326
+ headers=headers,
327
+ verify=verify_https
328
+ )
329
+ if res.status_code != 200:
330
+ raise Exception(f"Error getting execution planning {res.status_code} {res.content}")
331
+
332
+ content = res.json(cls=CustomJsonDecoder)
333
+ if not content["success"]: raise Exception(content["errorMessage"])
334
+
335
+ total_query = content["data"]["total"]
336
+ page_size = content["data"]["size"]
337
+
338
+ total_batchs = (total_query // page_size) + 1
339
+ logger.info(f"Total documents to download {total_query}.")
340
+ logger.info(f"Batch 1/{total_batchs}")
341
+
342
+ downloaded_docs = 0
343
+ page = 1
344
+ total_batchs = (total_query // page_size) + 1
345
+ docs = []
346
+ while downloaded_docs < total_query:
347
+ res = requests.post(f"{QUERY_MANAGER}/QueryData/Aggregate",
348
+ data=json.dumps(pipeline, cls=CustomJSONEncoder),
349
+ headers=headers,
350
+ verify=verify_https
351
+ )
352
+
353
+ if res.status_code != 200: raise Exception(f"Error getting inputstream data {res.status_code} {res.content}")
354
+ content = res.json(cls=CustomJsonDecoder)
355
+ if not content["success"]: raise Exception(content["errorMessage"])
356
+ logger.info(f"Batch {page}/{total_batchs}")
357
+ downloaded_docs += content["data"]["size"]
358
+ docs += content["data"]["data"]
359
+ page += 1
360
+
361
+ logger.info(f"Data downloaded, total docs: {total_query}")
362
+ return docs#content["data"]
363
+ except Exception as e:
364
+ raise e
365
+
366
+ def insert(self, ikey:str, data:list[dict], mode:INSERTION_MODE, wait_response:bool) -> tuple[int, str]:
367
+ try:
368
+ headers = {
369
+ "Authorization": f"A2G {self.token}",
370
+ "ikey": ikey
371
+ }
372
+
373
+ if mode == INSERTION_MODE.REPLACE:
374
+ headers["Replace"] = "true"
375
+ headers["Transaction"] = "false"
376
+
377
+ elif mode == INSERTION_MODE.INSERT_UNORDERED:
378
+ headers["Replace"] = "false"
379
+ headers["Transaction"] = "false"
380
+
381
+ elif mode == INSERTION_MODE.TRANSACTION:
382
+ headers["Replace"] = "false"
383
+ headers["Transaction"] = "true"
384
+
385
+ if wait_response: headers["WaitResponse"] = "true"
386
+
387
+ res = requests.post(DATA_URL + "/Data/Insert", headers=headers, json=data, verify=verify_https)
388
+ if res.status_code != 200: raise Exception(f"Error to insert data in inputstream {res.status_code} {res.text}")
389
+ return res.status_code, res.text
390
+ except Exception as e:
391
+ raise e
392
+
393
+ async def insert_data_native(self, ikey:str,table: str, data:list[dict], start: int, end:int, wait_response = False, cache:bool=True):
394
+ async with SEM:
395
+ """
396
+ validate data against inputstream JsonSchema and insert into inputstream collection
397
+ params:
398
+ ikey: str
399
+ table_name: str
400
+ data: list[dict]
401
+ """
402
+ timeout = httpx.Timeout(60.0, connect=10.0, read=600.0)
403
+ async with httpx.AsyncClient(http2=True, verify=False, timeout=timeout) as client:
404
+ headers = {
405
+ "Authorization": f"A2G {self.token}",
406
+ "ikey": ikey,
407
+ 'Content-Type': 'text/plain',
408
+ }
409
+
410
+ if type(data) is not list: raise Exception("Data must be a list of dictionaries")
411
+
412
+ """Envia un lote de datos al servidor"""
413
+ my_body = {
414
+ "list_data": data[start:end],
415
+ "table_name": table,
416
+ }
417
+ async with client.stream("POST",f"{QUERY_MANAGER}/QueryData/InsertAll",
418
+ content=json.dumps(my_body, default=str),
419
+ headers=headers,
420
+ ) as response:
421
+ if response.status_code != 200:
422
+ msg = ""
423
+ async for chunk in response.aiter_bytes():
424
+ if chunk:
425
+ msg += chunk.decode("utf-8")
426
+ raise Exception(f"{response.status_code} {msg}")
427
+
428
+ def remove_documents(self, ikey:str, query:dict) -> int:
429
+ try:
430
+ logger.info("Removing data...")
431
+ headers = {
432
+ "Authorization": f"A2G {self.token}",
433
+ "ikey": ikey,
434
+ 'Content-Type': 'application/json'
435
+ }
436
+
437
+ if len(query) == 0:
438
+ raise Exception("Query is empty, please provide a valid query, if you desire to delete all documents, use the delete_all method.")
439
+
440
+ response = requests.post(f"{QUERY_MANAGER}/QueryData/RemoveDocuments",
441
+ data=json.dumps(query, cls=CustomJSONEncoder),
442
+ headers=headers,
443
+ verify=verify_https
444
+ )
445
+ if response.status_code != 200:
446
+ raise Exception(f"Error to remove data in inputstream {response.status_code} {response.content}")
447
+ res_object = response.json(cls=CustomJsonDecoder)
448
+ if not res_object["success"]: raise Exception(res_object["errorMessage"])
449
+
450
+ content = res_object["data"]
451
+ deleted_docs = content["docs_affected"]
452
+ logger.info(f"Operation complete, total docs deleted: {deleted_docs}")
453
+
454
+ return deleted_docs
455
+ except Exception as e:
456
+ raise e
457
+
458
+ def clear_inputstream(self, ikey:str) -> int:
459
+ try:
460
+ logger.info("Removing all data...")
461
+ headers = {
462
+ "Authorization": f"A2G {self.token}",
463
+ "ikey": ikey
464
+ }
465
+
466
+ response = requests.post(f"{QUERY_MANAGER}/QueryData/Clear", headers=headers, verify=verify_https)
467
+ if response.status_code != 200:
468
+ raise Exception(f"Error to remove all data in inputstream {response.status_code} {response.content}")
469
+ res_object = response.json(cls=CustomJsonDecoder)
470
+ if not res_object["success"]: raise Exception(res_object["errorMessage"])
471
+
472
+ content = res_object["data"]
473
+ deleted_docs = content["docs_affected"]
474
+ logger.info(f"Operation complete, total docs deleted: {deleted_docs}")
475
+ return deleted_docs
476
+ except Exception as e:
477
+ raise e
478
+
479
+
480
+ class InputstreamClient:
481
+
482
+ def __init__(self, token:str, cache_options:dict = None):
483
+ """
484
+ Constructor for LocalInputstream
485
+ :param token: Token to authenticate with ACELER.AI
486
+ :param cache_options: { duration_data: int, duration_inputstream: int } | None
487
+ """
488
+ self.acelerai_client = AcelerAIHttpClient(token)
489
+ self.cache_manager = CacheManager(cache_options)
490
+ self.__mode = os.environ.get("EXEC_LOCATION", "LOCAL")
491
+
492
+ def __get_inputstream(self, ikey:str) -> Inputstream:
493
+ inputstream = self.acelerai_client.get_inputstream_by_ikey(ikey)
494
+ return inputstream
495
+
496
+ async def __allPages(self, ikey, query,delete_id):
497
+ headers = {
498
+ "Authorization": f"A2G {self.acelerai_client.token}",
499
+ "ikey": ikey,
500
+ 'Content-Type': 'application/json'
501
+ }
502
+
503
+ logger.info("Getting Execution Planning...")
504
+
505
+ res = requests.post(f"{QUERY_MANAGER}/QueryData/ExecutionPlanningFind",
506
+ data = json.dumps(query, cls=CustomJSONEncoder),
507
+ headers=headers,
508
+ verify=verify_https
509
+ )
510
+
511
+ logger.info(f"Getting Status {res.status_code}")
512
+
513
+ if res.status_code != 200:
514
+ msg = ''
515
+ async for chunk in res.aiter_bytes():
516
+ if chunk:
517
+ msg += chunk.decode("utf-8")
518
+ raise Exception(f"{res.status_code} {msg}")
519
+
520
+ inputstream = self.acelerai_client.get_inputstream_by_ikey(ikey)
521
+ logger.info(f"Getting inputstream with ikey: {ikey}, Name {inputstream.Name} from ACELER.AI...")
522
+
523
+ content = res.json(cls=CustomJsonDecoder)
524
+
525
+ total_query = content["total"]
526
+ page_size:int = content["size"]
527
+
528
+ if total_query == 0:
529
+ logger.info("No data found with the query provided.")
530
+ return []
531
+
532
+ page: int = 1
533
+
534
+ if content['stage'] != None:
535
+ stage = content["stage"].replace('_',' -> ')
536
+ logger.info(f"The query stages are {stage}")
537
+ logger.info(F"The index used in query is {content['indexName']}")
538
+
539
+ elif inputstream.InputstreamType !=InputstreamType.Native:
540
+ logger.info(f"Complex query the explain was not saved")
541
+
542
+ tasks = []
543
+ total_pages = (total_query + page_size - 1) // page_size
544
+
545
+ if not os.path.exists(f".acelerai_cache/") : os.mkdir(f".acelerai_cache/")
546
+ if not os.path.exists(f".acelerai_cache/data/") : os.mkdir(f".acelerai_cache/data/")
547
+ if not os.path.exists(f".acelerai_cache/data/{ikey}/") : os.mkdir(f".acelerai_cache/data/{ikey}/")
548
+
549
+ logger.info('Downloading data, please wait...')
550
+ start_time = datetime.utcnow()
551
+ for page in range(total_pages):
552
+ tasks.append(self.acelerai_client.fetch_page(ikey, query, delete_id, page, page_size))
553
+ await asyncio.gather(*tasks)
554
+
555
+ end_time = datetime.utcnow()
556
+ logger.info(f"Total time for downloading {total_pages} pages: {round((end_time - start_time).total_seconds()/60, 2)} minutes")
557
+
558
+ async def __get_data(self, ikey:str, query, mode:str, cache:bool, delete_id:bool=True) -> list[dict]:
559
+ try:
560
+ logger.info(f"Cache: {cache}")
561
+ logger.info(f"Mode: {self.__mode}")
562
+ query_str = json.dumps(query, cls=CustomJSONEncoder)
563
+ query_hash = hashlib.sha256(query_str.encode()).hexdigest()
564
+
565
+ logger.info(f"Getting data sdadsadsadasdsad...")
566
+
567
+ data = self.cache_manager.get_data(ikey, query_hash) if cache and self.__mode=='LOCAL' else None
568
+ if data is None:
569
+ if mode == "find" : await self.__allPages(ikey, query,delete_id)
570
+ elif mode == "find_one" : data = self.acelerai_client.find_one(ikey, query)
571
+ elif mode == "aggregate": data = self.acelerai_client.aggregate(ikey, query)
572
+
573
+ self.cache_manager.set_data(ikey, query_hash)
574
+
575
+ output_file = f".acelerai_cache/data/{ikey}/{query_hash}.msgpack"
576
+ data = load_full_object(output_file)
577
+
578
+ return data
579
+ except Exception as e:
580
+ raise e
581
+
582
+ def get_inputstream_schema(self, ikey:str) -> dict:
583
+ """
584
+ return Inputstream schema
585
+ params:
586
+ ikey: str
587
+ cache: bool = True -> if True, use cache if exists and is not expired
588
+ """
589
+ inputstream = self.__get_inputstream(ikey)
590
+ return json.loads(inputstream.Schema)
591
+
592
+ async def find(self, ikey:str, query:dict,cache:bool=True,delete_id:bool=True):
593
+ """
594
+ return data from inputstream
595
+ params:
596
+ ikey: str
597
+ query: dict
598
+ cache: bool = True -> if True, use cache if exists and is not expired
599
+ """
600
+ mode = "find"
601
+ return await self.__get_data(ikey, query, mode, cache,delete_id)
602
+
603
+ def find_one(self, ikey:str, query:dict, cache:bool=True):
604
+ """
605
+ return one data from inputstream
606
+ params:
607
+ collection: str
608
+ query: dict
609
+ """
610
+ mode = "find_one"
611
+ return self.__get_data(ikey, query, mode, cache)
612
+
613
+ def get_data_aggregate(self, ikey:str, query: list[dict], cache:bool=True):
614
+ """
615
+ return data from inputstream
616
+ params:
617
+ ikey: str
618
+ query: list[dict]
619
+ """
620
+ mode = "aggregate"
621
+ return self.__get_data(ikey, query, mode, cache)
622
+
623
+ async def __validate_data_async(self, d, schema, schema_validator):
624
+ try:
625
+ schema_validator(d)
626
+ return None # Si no hay errores, retornamos None
627
+ except Exception as e:
628
+ return f"Error validating data: {e}" # Devolvemos el error
629
+
630
+ async def insert_data(self, ikey:str, data:list[dict], table: str=None, mode:INSERTION_MODE = INSERTION_MODE.REPLACE, wait_response = True, batch_size:int=1000, cache:bool=True):
631
+ """
632
+ validate data against inputstream JsonSchema and insert into inputstream collection
633
+ params:
634
+ ikey: str
635
+ data: list[dict]
636
+ table: str -> table is required only when inserting to a local inputstream
637
+ mode: INSERTION_MODE = INSERTION_MODE.REPLACE -> insertion mode
638
+ wait_response: bool = True -> if True, wait for response from server
639
+ batch_size: int = 1000 -> batch size for insert data
640
+ cache: bool = True -> if True, use cache if exists and is not expired
641
+ """
642
+ start = datetime.utcnow()
643
+ inputstream:Inputstream = self.__get_inputstream(ikey)
644
+ end = datetime.utcnow()
645
+
646
+ logger.info(f'Demoró {(end - start).total_seconds()} segs en obtener el inputstream')
647
+
648
+ if type(data) is not list: raise Exception("Data must be a list of dictionaries")
649
+ data_parsed = json.loads(json.dumps(data, cls=CustomJSONEncoder))
650
+
651
+ if inputstream.InputstreamType != InputstreamType.Native:
652
+ if inputstream.Status == InputstreamStatus.Undiscovered: #raise Exception("Inputstream undiscovered")
653
+ logger.info("Inputstream undiscovered, you must discovered the schema first")
654
+ return False
655
+ elif inputstream.Status == InputstreamStatus.Exposed: #raise Exception("Inputstream undiscovered")
656
+ start = datetime.utcnow()
657
+
658
+ schema = json.loads(inputstream.Schema)
659
+ schema_validator = fastjsonschema.compile(schema)
660
+ #schema_validator = Draft4Validator(schema=schema)
661
+
662
+ resultados = await asyncio.gather(*(self.__validate_data_async(d, schema, schema_validator) for d in data_parsed))
663
+
664
+ # Manejo de errores
665
+ errores = [error for error in resultados if error is not None]
666
+ if errores:
667
+ for error in errores:
668
+ logger.error(error)
669
+ raise Exception("Hubo errores durante la validación de datos.")
670
+
671
+ end = datetime.utcnow()
672
+
673
+ logger.info(f'Demoró {(end - start).total_seconds()} segs en validar los datos')
674
+
675
+
676
+ for i in range(0, len(data_parsed), batch_size):
677
+ code, message = self.acelerai_client.insert(ikey, data_parsed[i:i+batch_size], mode, wait_response)
678
+ batch_size_aux = batch_size if i+batch_size < len(data_parsed) else len(data_parsed) - i
679
+ logger.info(f"batch {(i//batch_size) + 1}, docs: [{i} - {i+batch_size_aux}] - {code} - {message}")
680
+ else:
681
+ logger.info('Inserting data, please wait...')
682
+ if not table or table=='': raise Exception("Table name is required when inserting to a native inputstream")
683
+ start_time = datetime.utcnow()
684
+ await asyncio.gather(*(self.acelerai_client.insert_data_native(ikey, table, data_parsed, i, min(i + batch_size, len(data_parsed)), wait_response, cache) for i in range(0, len(data_parsed), batch_size)))
685
+ endt = datetime.utcnow()
686
+ logger.info(f'{len(data)} registries inserted successfully in {(endt - start_time).total_seconds() / 60} minutes')
687
+
688
+ def remove_documents(self, ikey:str, query:dict) -> int:
689
+ """
690
+ delete data from inputstream
691
+ params:
692
+ ikey: str
693
+ query: dict
694
+ """
695
+ docs = self.acelerai_client.remove_documents(ikey, query)
696
+ return docs
697
+
698
+ def clear_inputstream(self, ikey:str) -> int:
699
+ """
700
+ delete all data from inputstream
701
+ params:
702
+ ikey: str
703
+ """
704
+ docs = self.acelerai_client.clear_inputstream(ikey)
705
+ return docs
706
+
@@ -0,0 +1,20 @@
1
+ import os
2
+ # os.environ["DATA_URL"] = "https://localhost:1008"
3
+ # os.environ["QUERY_MANAGER"] = "http://localhost:1012"
4
+ # os.environ["INPUTSTREAM_URL"] = "https://localhost:1000"
5
+
6
+ from acelerai_inputstream import InputstreamClient
7
+ import logging
8
+ import pandas as pd
9
+
10
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(name)s - %(message)s')
11
+
12
+ #Para desactivar los logs de la librería
13
+ logging.getLogger('InputstreamClient').disabled = False
14
+
15
+ #pat_danilo='324738f335354e9d80394c31ce1d644c'
16
+ client = InputstreamClient(token='084a45f10e3b4952a4f3a7df04f546e5')
17
+
18
+ input_cycle=client.find(ikey='74c337b0e99a45ac8a9c', query= {}, cache=False)
19
+ df=pd.DataFrame(input_cycle)
20
+ #print(len(df))
@@ -0,0 +1,12 @@
1
+ from acelerai_inputstream import InputstreamClient
2
+ import logging
3
+
4
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(name)s - %(message)s')
5
+
6
+ #Para desactivar los logs de la librería
7
+ #logging.getLogger('InputstreamClient').disabled = True
8
+ client = InputstreamClient(token='7ffba0a8bccb4498be8a811b43bd9400')
9
+ input_cycle=client.find(ikey='31feaa0361b44ebb8b70', query= {}, cache=False)
10
+
11
+
12
+