echoss-db 1.0.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoss_db/__init__.py +3 -0
- echoss_db/elastic_search.py +369 -0
- echoss_db/mongo_query.py +360 -0
- echoss_db/mysql_query.py +470 -0
- echoss_db-1.0.8.dist-info/METADATA +215 -0
- echoss_db-1.0.8.dist-info/RECORD +8 -0
- echoss_db-1.0.8.dist-info/WHEEL +5 -0
- echoss_db-1.0.8.dist-info/top_level.txt +1 -0
echoss_db/__init__.py
ADDED
|
@@ -0,0 +1,369 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
from opensearchpy import OpenSearch
|
|
3
|
+
from typing import Any, List, Tuple, Union
|
|
4
|
+
|
|
5
|
+
from echoss_fileformat import FileUtil, get_logger, set_logger_level
|
|
6
|
+
|
|
7
|
+
logger = get_logger("echoss_query")
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ElasticSearch:
|
|
11
|
+
conn = None
|
|
12
|
+
query_match_all = {"query": {"match_all": {}}}
|
|
13
|
+
empty_dataframe = pd.DataFrame()
|
|
14
|
+
query_cache = False
|
|
15
|
+
default_size = 1000
|
|
16
|
+
|
|
17
|
+
def __init__(self, conn_info: str or dict):
|
|
18
|
+
"""
|
|
19
|
+
Args:
|
|
20
|
+
conn_info : configration dictionary (index is option)
|
|
21
|
+
ex) conn_info = {
|
|
22
|
+
'elastic':
|
|
23
|
+
{
|
|
24
|
+
'user' : str(user),
|
|
25
|
+
'passwd': str(passwd),
|
|
26
|
+
'host' : str(host),
|
|
27
|
+
'port' : int(port),
|
|
28
|
+
'scheme' : http or https
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
"""
|
|
33
|
+
if isinstance(conn_info, str):
|
|
34
|
+
conn_info = FileUtil.dict_load(conn_info)
|
|
35
|
+
elif not isinstance(conn_info, dict):
|
|
36
|
+
raise TypeError("ElasticSearch support type 'str' and 'dict'")
|
|
37
|
+
|
|
38
|
+
required_keys = ['host', 'port']
|
|
39
|
+
if (len(conn_info) > 0) and ('elastic' in conn_info) and (conn_info['elastic'] for k in required_keys):
|
|
40
|
+
es_config = conn_info['elastic']
|
|
41
|
+
else:
|
|
42
|
+
raise TypeError("[Elastic] config info not exist")
|
|
43
|
+
|
|
44
|
+
self.user = es_config.get('user')
|
|
45
|
+
self.passwd = es_config.get('passwd')
|
|
46
|
+
self.auth = (self.user, self.passwd) if self.user and self.passwd else None
|
|
47
|
+
|
|
48
|
+
self.host = es_config['host']
|
|
49
|
+
self.port = es_config['port']
|
|
50
|
+
self.scheme = es_config.get('scheme', 'https')
|
|
51
|
+
if 'https' == self.scheme:
|
|
52
|
+
self.use_ssl = True
|
|
53
|
+
self.verify_certs = es_config.get('verify_certs', True)
|
|
54
|
+
else:
|
|
55
|
+
self.use_ssl = False
|
|
56
|
+
self.verify_certs = False
|
|
57
|
+
|
|
58
|
+
self.index_name = es_config.get('index')
|
|
59
|
+
|
|
60
|
+
self.hosts = [{
|
|
61
|
+
'host': self.host,
|
|
62
|
+
'port': self.port
|
|
63
|
+
}]
|
|
64
|
+
|
|
65
|
+
self.http_compress = es_config.get('http_compress', False)
|
|
66
|
+
|
|
67
|
+
# re-use connection
|
|
68
|
+
self.conn = self._connect_es()
|
|
69
|
+
|
|
70
|
+
# extra config
|
|
71
|
+
if 'default_size' in es_config:
|
|
72
|
+
self.default_size = es_config['default_size']
|
|
73
|
+
|
|
74
|
+
def __str__(self):
|
|
75
|
+
return f"ElasticSearch(hosts={self.hosts}, index={self.index_name})"
|
|
76
|
+
|
|
77
|
+
def _connect_es(self):
|
|
78
|
+
"""
|
|
79
|
+
ElasticSearch Cloud에 접속하는 함수
|
|
80
|
+
"""
|
|
81
|
+
try:
|
|
82
|
+
es_conn = OpenSearch(
|
|
83
|
+
hosts=self.hosts,
|
|
84
|
+
http_auth=self.auth,
|
|
85
|
+
scheme=self.scheme,
|
|
86
|
+
http_compress=self.http_compress,
|
|
87
|
+
use_ssl=self.use_ssl,
|
|
88
|
+
verify_certs=self.verify_certs,
|
|
89
|
+
ssl_assert_hostname=False,
|
|
90
|
+
ssl_show_warn=False
|
|
91
|
+
)
|
|
92
|
+
if es_conn is None or es_conn.ping() is False:
|
|
93
|
+
raise ValueError(f"open elasticsearch is failed or health ping failed.")
|
|
94
|
+
return es_conn
|
|
95
|
+
except Exception as e:
|
|
96
|
+
raise ValueError("Connection failed by config. Please check config data")
|
|
97
|
+
|
|
98
|
+
def ping(self) -> bool:
|
|
99
|
+
"""
|
|
100
|
+
Elastic Search에 Ping
|
|
101
|
+
"""
|
|
102
|
+
if self.conn:
|
|
103
|
+
return self.conn.ping()
|
|
104
|
+
else:
|
|
105
|
+
return False
|
|
106
|
+
|
|
107
|
+
def info(self) -> dict:
|
|
108
|
+
"""
|
|
109
|
+
Elastic Search Information
|
|
110
|
+
"""
|
|
111
|
+
return self.conn.info()
|
|
112
|
+
|
|
113
|
+
def exists(self, id: str or int, index=None) -> bool:
|
|
114
|
+
"""
|
|
115
|
+
Args:
|
|
116
|
+
index(str) : 확인 대상 index \n
|
|
117
|
+
id(str) : 확인 대상 id \n
|
|
118
|
+
Returns:
|
|
119
|
+
boolean
|
|
120
|
+
"""
|
|
121
|
+
if index is None:
|
|
122
|
+
index = self.index_name
|
|
123
|
+
return self.conn.exists(index, id)
|
|
124
|
+
|
|
125
|
+
def search(self, body: dict = None, index=None):
|
|
126
|
+
"""
|
|
127
|
+
Args:
|
|
128
|
+
index(str) : 대상 index
|
|
129
|
+
body(dict) : search body
|
|
130
|
+
Returns:
|
|
131
|
+
result(list) : search result
|
|
132
|
+
"""
|
|
133
|
+
if index is None:
|
|
134
|
+
index = self.index_name
|
|
135
|
+
if body is None:
|
|
136
|
+
body = self.query_match_all
|
|
137
|
+
|
|
138
|
+
response = self.conn.search(
|
|
139
|
+
index=index,
|
|
140
|
+
body=body
|
|
141
|
+
)
|
|
142
|
+
return response
|
|
143
|
+
|
|
144
|
+
def to_dataframe(self, result_list):
|
|
145
|
+
if isinstance(result_list, dict):
|
|
146
|
+
if 'hits' in result_list and 'hits' in result_list['hits']:
|
|
147
|
+
result_list = result_list['hits']['hits']
|
|
148
|
+
if result_list is not None and isinstance(result_list, list) and len(result_list)>0:
|
|
149
|
+
if '_source' in result_list[0]:
|
|
150
|
+
documents = [doc['_source'] for doc in result_list]
|
|
151
|
+
df = pd.DataFrame(documents)
|
|
152
|
+
return df
|
|
153
|
+
return self.empty_dataframe
|
|
154
|
+
|
|
155
|
+
def _fetch_all_hits(self, index: str, body: dict) -> List[dict]:
|
|
156
|
+
"""
|
|
157
|
+
Scroll API를 사용하여 모든 검색 결과를 가져옵니다.
|
|
158
|
+
|
|
159
|
+
Args:
|
|
160
|
+
index (str): 대상 인덱스 이름
|
|
161
|
+
body (dict): 검색 쿼리
|
|
162
|
+
Returns:
|
|
163
|
+
all_hits (list): 모든 검색 결과 리스트
|
|
164
|
+
"""
|
|
165
|
+
all_hits = []
|
|
166
|
+
scroll_time = '2m' # Scroll context 유지 시간
|
|
167
|
+
|
|
168
|
+
# 'size' 값 확인 및 설정
|
|
169
|
+
size = body.get('size', 1000)
|
|
170
|
+
body = body.copy() # 원본 body를 변경하지 않도록 복사
|
|
171
|
+
body['size'] = size
|
|
172
|
+
scroll_id = None
|
|
173
|
+
try:
|
|
174
|
+
# 초기 검색 요청
|
|
175
|
+
response = self.conn.search(
|
|
176
|
+
index=index,
|
|
177
|
+
body=body,
|
|
178
|
+
scroll=scroll_time
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
scroll_id = response['_scroll_id']
|
|
182
|
+
hits = response['hits']['hits']
|
|
183
|
+
all_hits.extend(hits)
|
|
184
|
+
|
|
185
|
+
# 더 이상 결과가 없을 때까지 반복
|
|
186
|
+
while len(hits) > 0:
|
|
187
|
+
response = self.conn.scroll(
|
|
188
|
+
scroll_id=scroll_id,
|
|
189
|
+
scroll=scroll_time
|
|
190
|
+
)
|
|
191
|
+
scroll_id = response['_scroll_id']
|
|
192
|
+
hits = response['hits']['hits']
|
|
193
|
+
all_hits.extend(hits)
|
|
194
|
+
|
|
195
|
+
# Scroll context 삭제
|
|
196
|
+
self.conn.clear_scroll(scroll_id=scroll_id)
|
|
197
|
+
|
|
198
|
+
except Exception as e:
|
|
199
|
+
logger.error(f"Error fetching all hits: {e}")
|
|
200
|
+
# Scroll context 정리 시도
|
|
201
|
+
try:
|
|
202
|
+
if scroll_id is not None:
|
|
203
|
+
self.conn.clear_scroll(scroll_id=scroll_id)
|
|
204
|
+
except Exception as ce:
|
|
205
|
+
logger.error(f"connection clear_scroll failed : {ce}")
|
|
206
|
+
raise
|
|
207
|
+
|
|
208
|
+
return all_hits
|
|
209
|
+
|
|
210
|
+
def search_list(self, body: dict = None, index=None, fetch_all: bool = True) -> list:
|
|
211
|
+
"""
|
|
212
|
+
Args:
|
|
213
|
+
body(dict) : search body
|
|
214
|
+
index(str) : 대상 index
|
|
215
|
+
fetch_all(bool) : fetch all hits
|
|
216
|
+
Returns:
|
|
217
|
+
result(list) : search result of response['hits']['hits']
|
|
218
|
+
"""
|
|
219
|
+
if index is None:
|
|
220
|
+
index = self.index_name
|
|
221
|
+
if body is None:
|
|
222
|
+
body = self.query_match_all
|
|
223
|
+
if fetch_all:
|
|
224
|
+
return self._fetch_all_hits(index, body)
|
|
225
|
+
else:
|
|
226
|
+
response = self.conn.search(
|
|
227
|
+
index=index,
|
|
228
|
+
body=body
|
|
229
|
+
)
|
|
230
|
+
if len(response) > 0 and 'hits' in response and 'hits' in response['hits']:
|
|
231
|
+
return response['hits']['hits']
|
|
232
|
+
return []
|
|
233
|
+
|
|
234
|
+
def search_dataframe(self, body: dict = None, index=None, fetch_all: bool = True) -> pd.DataFrame:
|
|
235
|
+
hits_list = self.search_list(body=body, index=index, fetch_all=fetch_all)
|
|
236
|
+
return self.to_dataframe(hits_list)
|
|
237
|
+
|
|
238
|
+
def search_field(self, field: str, value: str, index=None, fetch_all=True) -> list:
|
|
239
|
+
"""
|
|
240
|
+
해당 index, field, value 값과 비슷한 값들을 검색해주는 함수 \n
|
|
241
|
+
Args:
|
|
242
|
+
index(str) : 대상 index
|
|
243
|
+
field(str) : 검색 대상 field \n
|
|
244
|
+
value(str) : 검색 대상 value \n
|
|
245
|
+
Returns:
|
|
246
|
+
result(list) : 검색 결과 리스트
|
|
247
|
+
"""
|
|
248
|
+
if index is None:
|
|
249
|
+
index = self.index_name
|
|
250
|
+
|
|
251
|
+
query_body = {
|
|
252
|
+
'query': {
|
|
253
|
+
'match': {field: value}
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
if fetch_all:
|
|
257
|
+
return self._fetch_all_hits(index, query_body)
|
|
258
|
+
else:
|
|
259
|
+
response = self.conn.search(
|
|
260
|
+
index=index,
|
|
261
|
+
body=query_body
|
|
262
|
+
)
|
|
263
|
+
return response['hits']['hits']
|
|
264
|
+
|
|
265
|
+
def get(self, id: str or int, index=None) -> dict:
|
|
266
|
+
"""
|
|
267
|
+
index에서 id와 일치하는 데이터를 불러오는 함수 \n
|
|
268
|
+
Args:
|
|
269
|
+
id(str) : 가져올 대상 id \n
|
|
270
|
+
Returns:
|
|
271
|
+
result(dict) : 결과 데이터
|
|
272
|
+
|
|
273
|
+
"""
|
|
274
|
+
if index is None:
|
|
275
|
+
index = self.index_name
|
|
276
|
+
return self.conn.get(index=index, id=id)
|
|
277
|
+
|
|
278
|
+
def get_source(self, id: str or int, index=None) -> dict:
|
|
279
|
+
"""
|
|
280
|
+
index에서 id와 일치하는 데이터의 소스만 불러오는 함수 \n
|
|
281
|
+
Args:
|
|
282
|
+
id(str) : 가져올 대상 id \n
|
|
283
|
+
Returns:
|
|
284
|
+
result(dict) : 결과 데이터
|
|
285
|
+
|
|
286
|
+
"""
|
|
287
|
+
if index is None:
|
|
288
|
+
index = self.index_name
|
|
289
|
+
return self.conn.get_source(index, id)
|
|
290
|
+
|
|
291
|
+
def create(self, id: str or int, body: dict, index=None):
|
|
292
|
+
"""
|
|
293
|
+
index에 해당 id로 새로운 document를 생성하는 함수 \n
|
|
294
|
+
(기존에 있는 index에 데이터를 추가할 때 사용) \n
|
|
295
|
+
Args:
|
|
296
|
+
id(str) : 생성할 id \n
|
|
297
|
+
body(dict) : new data
|
|
298
|
+
index(str) : index name or self.index_name will be used
|
|
299
|
+
Returns:
|
|
300
|
+
result(str) : 생성 결과
|
|
301
|
+
"""
|
|
302
|
+
if index is None:
|
|
303
|
+
index = self.index_name
|
|
304
|
+
return self.conn.create(index=index, id=id, body=body)
|
|
305
|
+
|
|
306
|
+
def index(self, index: str, body: dict, id: str or int = None) -> str:
|
|
307
|
+
"""
|
|
308
|
+
index를 생성하고 해당 id로 새로운 document를 생성하는 함수 \n
|
|
309
|
+
(index를 추가하고 그 내부 document까지 추가하는 방식) \n
|
|
310
|
+
Args:
|
|
311
|
+
index(str) : 생성할 index name \n
|
|
312
|
+
body(dict) : 입력할 json 내용
|
|
313
|
+
id(str) : 생성할 id \n
|
|
314
|
+
Returns:
|
|
315
|
+
result(str) : 생성 결과
|
|
316
|
+
"""
|
|
317
|
+
return self.conn.index(index, body, id=id)
|
|
318
|
+
|
|
319
|
+
def update(self, id: str or int, body: dict, index=None) -> str:
|
|
320
|
+
"""
|
|
321
|
+
기존 데이터를 id를 기준으로 body 값으로 수정하는 함수 \n
|
|
322
|
+
Args:
|
|
323
|
+
id(str) : 수정할 대상 id \n
|
|
324
|
+
body(dict) : data dict to update
|
|
325
|
+
index(str) : 생성할 index name \n
|
|
326
|
+
Returns:
|
|
327
|
+
result(str) : 처리 결과
|
|
328
|
+
"""
|
|
329
|
+
if index is None:
|
|
330
|
+
index = self.index_name
|
|
331
|
+
doc_body = {
|
|
332
|
+
'doc' : body
|
|
333
|
+
}
|
|
334
|
+
return self.conn.update(index, id, doc_body)
|
|
335
|
+
|
|
336
|
+
def delete(self, id: str or int, index=None) -> str:
|
|
337
|
+
"""
|
|
338
|
+
삭제하고 싶은 데이터를 id 기준으로 삭제하는 함수 \n
|
|
339
|
+
Args:
|
|
340
|
+
id(str) : 삭제 대상 id \n
|
|
341
|
+
index(str) : 생성할 index name \n
|
|
342
|
+
Returns:
|
|
343
|
+
result(str) : 처리 결과
|
|
344
|
+
"""
|
|
345
|
+
if index is None:
|
|
346
|
+
index = self.index_name
|
|
347
|
+
return self.conn.delete(index, id)
|
|
348
|
+
|
|
349
|
+
def delete_index(self, index):
|
|
350
|
+
"""
|
|
351
|
+
인덱스를 삭제하는 명령어 신중하게 사용해야한다.\n
|
|
352
|
+
Args:
|
|
353
|
+
index(str) : 삭제할 index
|
|
354
|
+
Returns:
|
|
355
|
+
result(str) : 처리 결과
|
|
356
|
+
"""
|
|
357
|
+
return self.conn.indices.delete(index)
|
|
358
|
+
|
|
359
|
+
def close(self):
|
|
360
|
+
try:
|
|
361
|
+
if self.conn:
|
|
362
|
+
self.conn.close()
|
|
363
|
+
self.conn = None
|
|
364
|
+
except AttributeError:
|
|
365
|
+
pass
|
|
366
|
+
|
|
367
|
+
def __del__(self):
|
|
368
|
+
self.close()
|
|
369
|
+
|