echoss-db 1.0.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
echoss_db/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .mysql_query import MysqlQuery
2
+ from .mongo_query import MongoQuery
3
+ from .elastic_search import ElasticSearch
@@ -0,0 +1,369 @@
1
+ import pandas as pd
2
+ from opensearchpy import OpenSearch
3
+ from typing import Any, List, Tuple, Union
4
+
5
+ from echoss_fileformat import FileUtil, get_logger, set_logger_level
6
+
7
+ logger = get_logger("echoss_query")
8
+
9
+
10
+ class ElasticSearch:
11
+ conn = None
12
+ query_match_all = {"query": {"match_all": {}}}
13
+ empty_dataframe = pd.DataFrame()
14
+ query_cache = False
15
+ default_size = 1000
16
+
17
+ def __init__(self, conn_info: str or dict):
18
+ """
19
+ Args:
20
+ conn_info : configration dictionary (index is option)
21
+ ex) conn_info = {
22
+ 'elastic':
23
+ {
24
+ 'user' : str(user),
25
+ 'passwd': str(passwd),
26
+ 'host' : str(host),
27
+ 'port' : int(port),
28
+ 'scheme' : http or https
29
+ }
30
+ }
31
+
32
+ """
33
+ if isinstance(conn_info, str):
34
+ conn_info = FileUtil.dict_load(conn_info)
35
+ elif not isinstance(conn_info, dict):
36
+ raise TypeError("ElasticSearch support type 'str' and 'dict'")
37
+
38
+ required_keys = ['host', 'port']
39
+ if (len(conn_info) > 0) and ('elastic' in conn_info) and (conn_info['elastic'] for k in required_keys):
40
+ es_config = conn_info['elastic']
41
+ else:
42
+ raise TypeError("[Elastic] config info not exist")
43
+
44
+ self.user = es_config.get('user')
45
+ self.passwd = es_config.get('passwd')
46
+ self.auth = (self.user, self.passwd) if self.user and self.passwd else None
47
+
48
+ self.host = es_config['host']
49
+ self.port = es_config['port']
50
+ self.scheme = es_config.get('scheme', 'https')
51
+ if 'https' == self.scheme:
52
+ self.use_ssl = True
53
+ self.verify_certs = es_config.get('verify_certs', True)
54
+ else:
55
+ self.use_ssl = False
56
+ self.verify_certs = False
57
+
58
+ self.index_name = es_config.get('index')
59
+
60
+ self.hosts = [{
61
+ 'host': self.host,
62
+ 'port': self.port
63
+ }]
64
+
65
+ self.http_compress = es_config.get('http_compress', False)
66
+
67
+ # re-use connection
68
+ self.conn = self._connect_es()
69
+
70
+ # extra config
71
+ if 'default_size' in es_config:
72
+ self.default_size = es_config['default_size']
73
+
74
+ def __str__(self):
75
+ return f"ElasticSearch(hosts={self.hosts}, index={self.index_name})"
76
+
77
+ def _connect_es(self):
78
+ """
79
+ ElasticSearch Cloud에 접속하는 함수
80
+ """
81
+ try:
82
+ es_conn = OpenSearch(
83
+ hosts=self.hosts,
84
+ http_auth=self.auth,
85
+ scheme=self.scheme,
86
+ http_compress=self.http_compress,
87
+ use_ssl=self.use_ssl,
88
+ verify_certs=self.verify_certs,
89
+ ssl_assert_hostname=False,
90
+ ssl_show_warn=False
91
+ )
92
+ if es_conn is None or es_conn.ping() is False:
93
+ raise ValueError(f"open elasticsearch is failed or health ping failed.")
94
+ return es_conn
95
+ except Exception as e:
96
+ raise ValueError("Connection failed by config. Please check config data")
97
+
98
+ def ping(self) -> bool:
99
+ """
100
+ Elastic Search에 Ping
101
+ """
102
+ if self.conn:
103
+ return self.conn.ping()
104
+ else:
105
+ return False
106
+
107
+ def info(self) -> dict:
108
+ """
109
+ Elastic Search Information
110
+ """
111
+ return self.conn.info()
112
+
113
+ def exists(self, id: str or int, index=None) -> bool:
114
+ """
115
+ Args:
116
+ index(str) : 확인 대상 index \n
117
+ id(str) : 확인 대상 id \n
118
+ Returns:
119
+ boolean
120
+ """
121
+ if index is None:
122
+ index = self.index_name
123
+ return self.conn.exists(index, id)
124
+
125
+ def search(self, body: dict = None, index=None):
126
+ """
127
+ Args:
128
+ index(str) : 대상 index
129
+ body(dict) : search body
130
+ Returns:
131
+ result(list) : search result
132
+ """
133
+ if index is None:
134
+ index = self.index_name
135
+ if body is None:
136
+ body = self.query_match_all
137
+
138
+ response = self.conn.search(
139
+ index=index,
140
+ body=body
141
+ )
142
+ return response
143
+
144
+ def to_dataframe(self, result_list):
145
+ if isinstance(result_list, dict):
146
+ if 'hits' in result_list and 'hits' in result_list['hits']:
147
+ result_list = result_list['hits']['hits']
148
+ if result_list is not None and isinstance(result_list, list) and len(result_list)>0:
149
+ if '_source' in result_list[0]:
150
+ documents = [doc['_source'] for doc in result_list]
151
+ df = pd.DataFrame(documents)
152
+ return df
153
+ return self.empty_dataframe
154
+
155
+ def _fetch_all_hits(self, index: str, body: dict) -> List[dict]:
156
+ """
157
+ Scroll API를 사용하여 모든 검색 결과를 가져옵니다.
158
+
159
+ Args:
160
+ index (str): 대상 인덱스 이름
161
+ body (dict): 검색 쿼리
162
+ Returns:
163
+ all_hits (list): 모든 검색 결과 리스트
164
+ """
165
+ all_hits = []
166
+ scroll_time = '2m' # Scroll context 유지 시간
167
+
168
+ # 'size' 값 확인 및 설정
169
+ size = body.get('size', 1000)
170
+ body = body.copy() # 원본 body를 변경하지 않도록 복사
171
+ body['size'] = size
172
+ scroll_id = None
173
+ try:
174
+ # 초기 검색 요청
175
+ response = self.conn.search(
176
+ index=index,
177
+ body=body,
178
+ scroll=scroll_time
179
+ )
180
+
181
+ scroll_id = response['_scroll_id']
182
+ hits = response['hits']['hits']
183
+ all_hits.extend(hits)
184
+
185
+ # 더 이상 결과가 없을 때까지 반복
186
+ while len(hits) > 0:
187
+ response = self.conn.scroll(
188
+ scroll_id=scroll_id,
189
+ scroll=scroll_time
190
+ )
191
+ scroll_id = response['_scroll_id']
192
+ hits = response['hits']['hits']
193
+ all_hits.extend(hits)
194
+
195
+ # Scroll context 삭제
196
+ self.conn.clear_scroll(scroll_id=scroll_id)
197
+
198
+ except Exception as e:
199
+ logger.error(f"Error fetching all hits: {e}")
200
+ # Scroll context 정리 시도
201
+ try:
202
+ if scroll_id is not None:
203
+ self.conn.clear_scroll(scroll_id=scroll_id)
204
+ except Exception as ce:
205
+ logger.error(f"connection clear_scroll failed : {ce}")
206
+ raise
207
+
208
+ return all_hits
209
+
210
+ def search_list(self, body: dict = None, index=None, fetch_all: bool = True) -> list:
211
+ """
212
+ Args:
213
+ body(dict) : search body
214
+ index(str) : 대상 index
215
+ fetch_all(bool) : fetch all hits
216
+ Returns:
217
+ result(list) : search result of response['hits']['hits']
218
+ """
219
+ if index is None:
220
+ index = self.index_name
221
+ if body is None:
222
+ body = self.query_match_all
223
+ if fetch_all:
224
+ return self._fetch_all_hits(index, body)
225
+ else:
226
+ response = self.conn.search(
227
+ index=index,
228
+ body=body
229
+ )
230
+ if len(response) > 0 and 'hits' in response and 'hits' in response['hits']:
231
+ return response['hits']['hits']
232
+ return []
233
+
234
+ def search_dataframe(self, body: dict = None, index=None, fetch_all: bool = True) -> pd.DataFrame:
235
+ hits_list = self.search_list(body=body, index=index, fetch_all=fetch_all)
236
+ return self.to_dataframe(hits_list)
237
+
238
+ def search_field(self, field: str, value: str, index=None, fetch_all=True) -> list:
239
+ """
240
+ 해당 index, field, value 값과 비슷한 값들을 검색해주는 함수 \n
241
+ Args:
242
+ index(str) : 대상 index
243
+ field(str) : 검색 대상 field \n
244
+ value(str) : 검색 대상 value \n
245
+ Returns:
246
+ result(list) : 검색 결과 리스트
247
+ """
248
+ if index is None:
249
+ index = self.index_name
250
+
251
+ query_body = {
252
+ 'query': {
253
+ 'match': {field: value}
254
+ }
255
+ }
256
+ if fetch_all:
257
+ return self._fetch_all_hits(index, query_body)
258
+ else:
259
+ response = self.conn.search(
260
+ index=index,
261
+ body=query_body
262
+ )
263
+ return response['hits']['hits']
264
+
265
+ def get(self, id: str or int, index=None) -> dict:
266
+ """
267
+ index에서 id와 일치하는 데이터를 불러오는 함수 \n
268
+ Args:
269
+ id(str) : 가져올 대상 id \n
270
+ Returns:
271
+ result(dict) : 결과 데이터
272
+
273
+ """
274
+ if index is None:
275
+ index = self.index_name
276
+ return self.conn.get(index=index, id=id)
277
+
278
+ def get_source(self, id: str or int, index=None) -> dict:
279
+ """
280
+ index에서 id와 일치하는 데이터의 소스만 불러오는 함수 \n
281
+ Args:
282
+ id(str) : 가져올 대상 id \n
283
+ Returns:
284
+ result(dict) : 결과 데이터
285
+
286
+ """
287
+ if index is None:
288
+ index = self.index_name
289
+ return self.conn.get_source(index, id)
290
+
291
+ def create(self, id: str or int, body: dict, index=None):
292
+ """
293
+ index에 해당 id로 새로운 document를 생성하는 함수 \n
294
+ (기존에 있는 index에 데이터를 추가할 때 사용) \n
295
+ Args:
296
+ id(str) : 생성할 id \n
297
+ body(dict) : new data
298
+ index(str) : index name or self.index_name will be used
299
+ Returns:
300
+ result(str) : 생성 결과
301
+ """
302
+ if index is None:
303
+ index = self.index_name
304
+ return self.conn.create(index=index, id=id, body=body)
305
+
306
+ def index(self, index: str, body: dict, id: str or int = None) -> str:
307
+ """
308
+ index를 생성하고 해당 id로 새로운 document를 생성하는 함수 \n
309
+ (index를 추가하고 그 내부 document까지 추가하는 방식) \n
310
+ Args:
311
+ index(str) : 생성할 index name \n
312
+ body(dict) : 입력할 json 내용
313
+ id(str) : 생성할 id \n
314
+ Returns:
315
+ result(str) : 생성 결과
316
+ """
317
+ return self.conn.index(index, body, id=id)
318
+
319
+ def update(self, id: str or int, body: dict, index=None) -> str:
320
+ """
321
+ 기존 데이터를 id를 기준으로 body 값으로 수정하는 함수 \n
322
+ Args:
323
+ id(str) : 수정할 대상 id \n
324
+ body(dict) : data dict to update
325
+ index(str) : 생성할 index name \n
326
+ Returns:
327
+ result(str) : 처리 결과
328
+ """
329
+ if index is None:
330
+ index = self.index_name
331
+ doc_body = {
332
+ 'doc' : body
333
+ }
334
+ return self.conn.update(index, id, doc_body)
335
+
336
+ def delete(self, id: str or int, index=None) -> str:
337
+ """
338
+ 삭제하고 싶은 데이터를 id 기준으로 삭제하는 함수 \n
339
+ Args:
340
+ id(str) : 삭제 대상 id \n
341
+ index(str) : 생성할 index name \n
342
+ Returns:
343
+ result(str) : 처리 결과
344
+ """
345
+ if index is None:
346
+ index = self.index_name
347
+ return self.conn.delete(index, id)
348
+
349
+ def delete_index(self, index):
350
+ """
351
+ 인덱스를 삭제하는 명령어 신중하게 사용해야한다.\n
352
+ Args:
353
+ index(str) : 삭제할 index
354
+ Returns:
355
+ result(str) : 처리 결과
356
+ """
357
+ return self.conn.indices.delete(index)
358
+
359
+ def close(self):
360
+ try:
361
+ if self.conn:
362
+ self.conn.close()
363
+ self.conn = None
364
+ except AttributeError:
365
+ pass
366
+
367
+ def __del__(self):
368
+ self.close()
369
+