echoss-db 1.2.2__tar.gz → 1.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {echoss_db-1.2.2 → echoss_db-1.2.3}/PKG-INFO +79 -10
- {echoss_db-1.2.2 → echoss_db-1.2.3}/README.md +75 -7
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db/__init__.py +2 -0
- echoss_db-1.2.3/echoss_db/postgres_query.py +419 -0
- echoss_db-1.2.3/echoss_db/qdrant_vector.py +393 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db.egg-info/PKG-INFO +79 -10
- echoss_db-1.2.3/echoss_db.egg-info/SOURCES.txt +24 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db.egg-info/requires.txt +2 -1
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db.egg-info/top_level.txt +1 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/requirements.txt +2 -1
- {echoss_db-1.2.2 → echoss_db-1.2.3}/setup.py +4 -5
- echoss_db-1.2.3/tests/__init__.py +0 -0
- echoss_db-1.2.3/tests/conftest.py +79 -0
- echoss_db-1.2.3/tests/helpers/__init__.py +0 -0
- echoss_db-1.2.3/tests/helpers/fakes.py +168 -0
- echoss_db-1.2.3/tests/unit/__init__.py +0 -0
- echoss_db-1.2.3/tests/unit/postgres/__init__.py +0 -0
- echoss_db-1.2.3/tests/unit/postgres/test_postgres_query.py +114 -0
- echoss_db-1.2.3/tests/unit/qdrant/__init__.py +0 -0
- echoss_db-1.2.3/tests/unit/qdrant/test_qdrant_vector.py +127 -0
- echoss_db-1.2.2/echoss_db.egg-info/SOURCES.txt +0 -13
- {echoss_db-1.2.2 → echoss_db-1.2.3}/MANIFEST.in +0 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db/elastic_search.py +0 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db/mongo_query.py +0 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db/mysql_query.py +0 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/echoss_db.egg-info/dependency_links.txt +0 -0
- {echoss_db-1.2.2 → echoss_db-1.2.3}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: echoss-db
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.3
|
|
4
4
|
Summary: echoss AI Bigdata Solution - Database Query Package
|
|
5
5
|
Home-page: https://github.com/12cmlab/echoss-query
|
|
6
6
|
Author: ckkim
|
|
@@ -8,15 +8,16 @@ Author-email: ckkim@12cm.co.kr
|
|
|
8
8
|
License: Apache License 2.0
|
|
9
9
|
Classifier: Programming Language :: Python :: 3
|
|
10
10
|
Classifier: Operating System :: OS Independent
|
|
11
|
-
Requires-Python:
|
|
11
|
+
Requires-Python: >=3.8
|
|
12
12
|
Description-Content-Type: text/markdown
|
|
13
13
|
Requires-Dist: pandas>=1.5.3
|
|
14
14
|
Requires-Dist: pymongo>=4.3.3
|
|
15
15
|
Requires-Dist: sqlalchemy>=2.0.0
|
|
16
16
|
Requires-Dist: PyMySQL>=1.0.2
|
|
17
|
-
Requires-Dist: PyYAML>=6.0
|
|
18
17
|
Requires-Dist: opensearch-py>=2.2.0
|
|
19
18
|
Requires-Dist: echoss-fileformat>=1.1.2
|
|
19
|
+
Requires-Dist: psycopg2-binary<3.0.0,>=2.9.11
|
|
20
|
+
Requires-Dist: qdrant-client[fastembed]<2.0.0,>=1.14.1
|
|
20
21
|
Dynamic: author
|
|
21
22
|
Dynamic: author-email
|
|
22
23
|
Dynamic: classifier
|
|
@@ -136,15 +137,12 @@ To install this package, please use Python 3.8 or higher.
|
|
|
136
137
|
# Ping
|
|
137
138
|
mysql.ping()
|
|
138
139
|
|
|
139
|
-
# get connection cursor
|
|
140
|
-
mysql.conn_cursor(cursorclass=None)
|
|
141
|
-
|
|
142
140
|
# Close
|
|
143
141
|
# crash process close
|
|
144
142
|
mysql.close()
|
|
145
143
|
|
|
146
144
|
# debug query : default True
|
|
147
|
-
mysql.
|
|
145
|
+
mysql.query_debug(False)
|
|
148
146
|
```
|
|
149
147
|
|
|
150
148
|
### MongoDB
|
|
@@ -196,7 +194,7 @@ To install this package, please use Python 3.8 or higher.
|
|
|
196
194
|
elastic.search_field(field='FIELD_NAME',value='VALUE') -> list
|
|
197
195
|
|
|
198
196
|
# INSERT
|
|
199
|
-
elastic.
|
|
197
|
+
elastic.index(index='INDEX_NAME', body='JSON_BODY', id='ID')
|
|
200
198
|
|
|
201
199
|
#UPDATE
|
|
202
200
|
elastic.update(id='ID', body='JSON_BODY')
|
|
@@ -209,8 +207,8 @@ To install this package, please use Python 3.8 or higher.
|
|
|
209
207
|
chunk_list = elastic.next_scroll_chunk()
|
|
210
208
|
|
|
211
209
|
# BULK
|
|
212
|
-
success, error_list = bulk_insert("farm_index", doc_list, id_field="farm_id")
|
|
213
|
-
success, error_list = bulk_upsert("farm_index", doc_list, id_field="farm_id")
|
|
210
|
+
success, error_list = elastic.bulk_insert("farm_index", doc_list, id_field="farm_id")
|
|
211
|
+
success, error_list = elastic.bulk_upsert("farm_index", doc_list, id_field="farm_id")
|
|
214
212
|
|
|
215
213
|
# Ping
|
|
216
214
|
elastic.ping()
|
|
@@ -219,6 +217,77 @@ To install this package, please use Python 3.8 or higher.
|
|
|
219
217
|
elastic.info()
|
|
220
218
|
```
|
|
221
219
|
|
|
220
|
+
### Qdrant
|
|
221
|
+
|
|
222
|
+
---
|
|
223
|
+
|
|
224
|
+
`QdrantVector`는 기본적으로 FastEmbed 모델을 사용합니다.
|
|
225
|
+
|
|
226
|
+
Tutorial 실행 순서:
|
|
227
|
+
1. `tutorial/example_postgres_qdrant_ingest.py` (테스트용 충분한 chunk 데이터 생성)
|
|
228
|
+
2. `tutorial/example_qdrant.py` (Qdrant 업서트/검색 확인)
|
|
229
|
+
|
|
230
|
+
1. `fastembed_model` 미설정
|
|
231
|
+
- 기본값 `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2`가 자동 적용됩니다.
|
|
232
|
+
- 단, 설치된 `qdrant-client/fastembed` 조합에서 미지원일 수 있으므로 아래 메서드로 확인하세요.
|
|
233
|
+
- 모델 상세(설명 포함) 목록: `list_supported_models_by_class("TextEmbedding")`
|
|
234
|
+
- `vector` 입력 없이 원문 텍스트(`document`/`text`/`content`)를 전달합니다.
|
|
235
|
+
- 명시 메서드 `upsert_texts()` / `search_text()` 사용을 권장합니다.
|
|
236
|
+
- `create_collection()`의 차원은 모델에서 자동 추론합니다.
|
|
237
|
+
|
|
238
|
+
2. `fastembed_model` 설정
|
|
239
|
+
- 지정한 모델로 동작합니다.
|
|
240
|
+
- `create_collection()`의 차원은 설정 모델 기준으로 자동 추론합니다.
|
|
241
|
+
|
|
242
|
+
3. `fastembed_model: null`
|
|
243
|
+
- FastEmbed를 비활성화하고 벡터 직접 주입 모드로 동작합니다.
|
|
244
|
+
- 이 경우 `upsert_vectors()` / `search_vector()` 사용을 권장합니다.
|
|
245
|
+
- `create_collection(dim=...)`에서 `dim`은 필수입니다.
|
|
246
|
+
|
|
247
|
+
Config example:
|
|
248
|
+
|
|
249
|
+
```yaml
|
|
250
|
+
qdrant:
|
|
251
|
+
host: <IP addrress or domain>
|
|
252
|
+
port: 6333
|
|
253
|
+
scheme: http
|
|
254
|
+
collection: ai_rag_chunks
|
|
255
|
+
timeout: 5
|
|
256
|
+
default_limit: 10
|
|
257
|
+
fastembed_model: sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2
|
|
258
|
+
```
|
|
259
|
+
|
|
260
|
+
Text-input upsert example (`fastembed_model` enabled):
|
|
261
|
+
|
|
262
|
+
```python
|
|
263
|
+
points = [
|
|
264
|
+
{"id": 1, "document": "RAG 시스템 설계 문서", "payload": {"source": "wiki"}},
|
|
265
|
+
{"id": 2, "text": "Qdrant 검색 예제", "payload": {"source": "blog"}},
|
|
266
|
+
]
|
|
267
|
+
qv.upsert_texts(points, collection="ai_rag_chunks")
|
|
268
|
+
|
|
269
|
+
hits = qv.search_text(query_text="RAG 아키텍처", limit=5, collection="ai_rag_chunks")
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
External vector-input mode is documented as a code pattern only (for lightweight tutorial runtime):
|
|
273
|
+
|
|
274
|
+
```python
|
|
275
|
+
qv = QdrantVector({
|
|
276
|
+
"qdrant": {
|
|
277
|
+
"host": "<host>",
|
|
278
|
+
"port": 6333,
|
|
279
|
+
"scheme": "http",
|
|
280
|
+
"collection": "ai_rag_chunks_external",
|
|
281
|
+
"fastembed_model": None
|
|
282
|
+
}
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
points = [{"id": 1, "vector": your_embed_fn("문서"), "payload": {"source": "external"}}]
|
|
286
|
+
qv.create_collection(dim=len(points[0]["vector"]))
|
|
287
|
+
qv.upsert_vectors(points)
|
|
288
|
+
hits = qv.search_vector(vector=your_embed_fn("질의"))
|
|
289
|
+
```
|
|
290
|
+
|
|
222
291
|
### Code Quality
|
|
223
292
|
|
|
224
293
|
When creating new functions, please follow the Google style Python docstrings. See example below:
|
|
@@ -106,15 +106,12 @@ To install this package, please use Python 3.8 or higher.
|
|
|
106
106
|
# Ping
|
|
107
107
|
mysql.ping()
|
|
108
108
|
|
|
109
|
-
# get connection cursor
|
|
110
|
-
mysql.conn_cursor(cursorclass=None)
|
|
111
|
-
|
|
112
109
|
# Close
|
|
113
110
|
# crash process close
|
|
114
111
|
mysql.close()
|
|
115
112
|
|
|
116
113
|
# debug query : default True
|
|
117
|
-
mysql.
|
|
114
|
+
mysql.query_debug(False)
|
|
118
115
|
```
|
|
119
116
|
|
|
120
117
|
### MongoDB
|
|
@@ -166,7 +163,7 @@ To install this package, please use Python 3.8 or higher.
|
|
|
166
163
|
elastic.search_field(field='FIELD_NAME',value='VALUE') -> list
|
|
167
164
|
|
|
168
165
|
# INSERT
|
|
169
|
-
elastic.
|
|
166
|
+
elastic.index(index='INDEX_NAME', body='JSON_BODY', id='ID')
|
|
170
167
|
|
|
171
168
|
#UPDATE
|
|
172
169
|
elastic.update(id='ID', body='JSON_BODY')
|
|
@@ -179,8 +176,8 @@ To install this package, please use Python 3.8 or higher.
|
|
|
179
176
|
chunk_list = elastic.next_scroll_chunk()
|
|
180
177
|
|
|
181
178
|
# BULK
|
|
182
|
-
success, error_list = bulk_insert("farm_index", doc_list, id_field="farm_id")
|
|
183
|
-
success, error_list = bulk_upsert("farm_index", doc_list, id_field="farm_id")
|
|
179
|
+
success, error_list = elastic.bulk_insert("farm_index", doc_list, id_field="farm_id")
|
|
180
|
+
success, error_list = elastic.bulk_upsert("farm_index", doc_list, id_field="farm_id")
|
|
184
181
|
|
|
185
182
|
# Ping
|
|
186
183
|
elastic.ping()
|
|
@@ -189,6 +186,77 @@ To install this package, please use Python 3.8 or higher.
|
|
|
189
186
|
elastic.info()
|
|
190
187
|
```
|
|
191
188
|
|
|
189
|
+
### Qdrant
|
|
190
|
+
|
|
191
|
+
---
|
|
192
|
+
|
|
193
|
+
`QdrantVector`는 기본적으로 FastEmbed 모델을 사용합니다.
|
|
194
|
+
|
|
195
|
+
Tutorial 실행 순서:
|
|
196
|
+
1. `tutorial/example_postgres_qdrant_ingest.py` (테스트용 충분한 chunk 데이터 생성)
|
|
197
|
+
2. `tutorial/example_qdrant.py` (Qdrant 업서트/검색 확인)
|
|
198
|
+
|
|
199
|
+
1. `fastembed_model` 미설정
|
|
200
|
+
- 기본값 `sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2`가 자동 적용됩니다.
|
|
201
|
+
- 단, 설치된 `qdrant-client/fastembed` 조합에서 미지원일 수 있으므로 아래 메서드로 확인하세요.
|
|
202
|
+
- 모델 상세(설명 포함) 목록: `list_supported_models_by_class("TextEmbedding")`
|
|
203
|
+
- `vector` 입력 없이 원문 텍스트(`document`/`text`/`content`)를 전달합니다.
|
|
204
|
+
- 명시 메서드 `upsert_texts()` / `search_text()` 사용을 권장합니다.
|
|
205
|
+
- `create_collection()`의 차원은 모델에서 자동 추론합니다.
|
|
206
|
+
|
|
207
|
+
2. `fastembed_model` 설정
|
|
208
|
+
- 지정한 모델로 동작합니다.
|
|
209
|
+
- `create_collection()`의 차원은 설정 모델 기준으로 자동 추론합니다.
|
|
210
|
+
|
|
211
|
+
3. `fastembed_model: null`
|
|
212
|
+
- FastEmbed를 비활성화하고 벡터 직접 주입 모드로 동작합니다.
|
|
213
|
+
- 이 경우 `upsert_vectors()` / `search_vector()` 사용을 권장합니다.
|
|
214
|
+
- `create_collection(dim=...)`에서 `dim`은 필수입니다.
|
|
215
|
+
|
|
216
|
+
Config example:
|
|
217
|
+
|
|
218
|
+
```yaml
|
|
219
|
+
qdrant:
|
|
220
|
+
host: <IP addrress or domain>
|
|
221
|
+
port: 6333
|
|
222
|
+
scheme: http
|
|
223
|
+
collection: ai_rag_chunks
|
|
224
|
+
timeout: 5
|
|
225
|
+
default_limit: 10
|
|
226
|
+
fastembed_model: sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
Text-input upsert example (`fastembed_model` enabled):
|
|
230
|
+
|
|
231
|
+
```python
|
|
232
|
+
points = [
|
|
233
|
+
{"id": 1, "document": "RAG 시스템 설계 문서", "payload": {"source": "wiki"}},
|
|
234
|
+
{"id": 2, "text": "Qdrant 검색 예제", "payload": {"source": "blog"}},
|
|
235
|
+
]
|
|
236
|
+
qv.upsert_texts(points, collection="ai_rag_chunks")
|
|
237
|
+
|
|
238
|
+
hits = qv.search_text(query_text="RAG 아키텍처", limit=5, collection="ai_rag_chunks")
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
External vector-input mode is documented as a code pattern only (for lightweight tutorial runtime):
|
|
242
|
+
|
|
243
|
+
```python
|
|
244
|
+
qv = QdrantVector({
|
|
245
|
+
"qdrant": {
|
|
246
|
+
"host": "<host>",
|
|
247
|
+
"port": 6333,
|
|
248
|
+
"scheme": "http",
|
|
249
|
+
"collection": "ai_rag_chunks_external",
|
|
250
|
+
"fastembed_model": None
|
|
251
|
+
}
|
|
252
|
+
})
|
|
253
|
+
|
|
254
|
+
points = [{"id": 1, "vector": your_embed_fn("문서"), "payload": {"source": "external"}}]
|
|
255
|
+
qv.create_collection(dim=len(points[0]["vector"]))
|
|
256
|
+
qv.upsert_vectors(points)
|
|
257
|
+
hits = qv.search_vector(vector=your_embed_fn("질의"))
|
|
258
|
+
```
|
|
259
|
+
|
|
192
260
|
### Code Quality
|
|
193
261
|
|
|
194
262
|
When creating new functions, please follow the Google style Python docstrings. See example below:
|
|
@@ -0,0 +1,419 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from functools import wraps
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import time
|
|
5
|
+
from typing import Any, Optional, Union
|
|
6
|
+
|
|
7
|
+
from sqlalchemy import create_engine, text
|
|
8
|
+
from sqlalchemy.engine import Engine, Connection, Result
|
|
9
|
+
from sqlalchemy.exc import SQLAlchemyError
|
|
10
|
+
|
|
11
|
+
from echoss_fileformat import FileUtil, get_logger
|
|
12
|
+
|
|
13
|
+
logger = get_logger("echoss_db")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# --------------------------------------------------------------------------------------
|
|
17
|
+
# 공통 Decorator (mysql_query.py와 동일)
|
|
18
|
+
# --------------------------------------------------------------------------------------
|
|
19
|
+
def log_execution_time(func):
|
|
20
|
+
@wraps(func)
|
|
21
|
+
def wrapper(self, *args, **kwargs):
|
|
22
|
+
if not getattr(self, 'use_query_debug', False):
|
|
23
|
+
return func(self, *args, **kwargs)
|
|
24
|
+
|
|
25
|
+
start_time = time.time()
|
|
26
|
+
try:
|
|
27
|
+
return func(self, *args, **kwargs)
|
|
28
|
+
finally:
|
|
29
|
+
elapsed = time.time() - start_time
|
|
30
|
+
logger.debug(f"{func.__name__}() executed in {elapsed:.3f} seconds")
|
|
31
|
+
return wrapper
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _safe_log_param(p):
|
|
35
|
+
return '<binary>' if isinstance(p, (bytes, bytearray)) else repr(p)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def parse_query(keyword):
|
|
39
|
+
def decorator(func):
|
|
40
|
+
@wraps(func)
|
|
41
|
+
def wrapper(self, query_str: str, params=None, *args, **kwargs):
|
|
42
|
+
if keyword not in query_str.upper():
|
|
43
|
+
raise ValueError(f"Input query does not include '{keyword}'")
|
|
44
|
+
|
|
45
|
+
query_str = query_str.strip().rstrip(";").rstrip()
|
|
46
|
+
|
|
47
|
+
if getattr(self, 'use_query_debug', False):
|
|
48
|
+
if params and isinstance(params, list):
|
|
49
|
+
rows = len(params)
|
|
50
|
+
if rows > 0:
|
|
51
|
+
first_param = params[0]
|
|
52
|
+
if isinstance(first_param, (list, tuple)):
|
|
53
|
+
safe_example = [_safe_log_param(p) for p in first_param]
|
|
54
|
+
elif isinstance(first_param, dict):
|
|
55
|
+
safe_example = {k: _safe_log_param(v) for k, v in first_param.items()}
|
|
56
|
+
else:
|
|
57
|
+
safe_example = first_param
|
|
58
|
+
logger.debug(
|
|
59
|
+
f'postgres_query.{func.__name__}() bulk example """{query_str}""" '
|
|
60
|
+
f'with {safe_example} for {rows} rows'
|
|
61
|
+
)
|
|
62
|
+
else:
|
|
63
|
+
logger.warning(f'postgres_query.{func.__name__}() bulk query """{query_str}""" with empty params')
|
|
64
|
+
else:
|
|
65
|
+
if isinstance(params, tuple):
|
|
66
|
+
safe_example = [_safe_log_param(p) for p in params]
|
|
67
|
+
elif isinstance(params, dict):
|
|
68
|
+
safe_example = {k: _safe_log_param(v) for k, v in params.items()}
|
|
69
|
+
else:
|
|
70
|
+
safe_example = params
|
|
71
|
+
logger.debug(f'postgres_query.{func.__name__}() parsed """{query_str}""" with {safe_example}')
|
|
72
|
+
|
|
73
|
+
return func(self, query_str, params, *args, **kwargs)
|
|
74
|
+
return wrapper
|
|
75
|
+
return decorator
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class PostgresQuery:
|
|
79
|
+
"""
|
|
80
|
+
Postgres SQLAlchemy 2.x 기반 래퍼.
|
|
81
|
+
MysqlQuery와 동일한 메서드/반환 타입을 최대한 유지.
|
|
82
|
+
- exec_driver_sql 로 %s 파라미터 스타일(=psycopg2) 호환
|
|
83
|
+
- :name 스타일도 지원(use_percent_param=False)
|
|
84
|
+
"""
|
|
85
|
+
engine: Optional[Engine] = None
|
|
86
|
+
empty_dataframe = pd.DataFrame()
|
|
87
|
+
use_query_debug: bool = True
|
|
88
|
+
use_percent_param: bool = True
|
|
89
|
+
|
|
90
|
+
def __init__(self, conn_info: Union[str, dict], compress=False,
|
|
91
|
+
pool_size: int = 5, pool_timeout: int = 30, pool_recycle: Optional[int] = None,
|
|
92
|
+
use_percent_param: bool = True):
|
|
93
|
+
if isinstance(conn_info, str):
|
|
94
|
+
conn_info = FileUtil.dict_load(conn_info)
|
|
95
|
+
elif not isinstance(conn_info, dict):
|
|
96
|
+
raise TypeError("PostgresQuery support type 'str' and 'dict'")
|
|
97
|
+
|
|
98
|
+
self.pool_size = pool_size
|
|
99
|
+
self.pool_timeout = pool_timeout
|
|
100
|
+
self.pool_recycle = pool_recycle if pool_recycle is not None else pool_timeout * 60
|
|
101
|
+
self.use_percent_param = use_percent_param
|
|
102
|
+
|
|
103
|
+
required_keys = ['user', 'passwd', 'host', 'db']
|
|
104
|
+
if (len(conn_info) > 0) and ('postgres' in conn_info) and all(k in conn_info['postgres'] for k in required_keys):
|
|
105
|
+
p = conn_info["postgres"]
|
|
106
|
+
self.user = p['user']
|
|
107
|
+
self.passwd = p['passwd']
|
|
108
|
+
self.host = p['host']
|
|
109
|
+
self.port = p.get('port', 5432)
|
|
110
|
+
self.db = p['db']
|
|
111
|
+
self.application_name = p.get('application_name', 'echoss_db')
|
|
112
|
+
# 스키마/서치패스가 필요하면 옵션으로 받기
|
|
113
|
+
self.schema = p.get("schema") # 예: ai_rag
|
|
114
|
+
self.options = p.get("options")
|
|
115
|
+
if self.schema:
|
|
116
|
+
sp = f"-c search_path={self.schema},public"
|
|
117
|
+
if not self.options:
|
|
118
|
+
self.options = sp
|
|
119
|
+
else:
|
|
120
|
+
# 이미 options가 있으면, search_path가 없을 때만 추가
|
|
121
|
+
if "search_path" not in self.options:
|
|
122
|
+
self.options = f"{self.options} {sp}"
|
|
123
|
+
self.sslmode = p.get('sslmode') # 필요시 "require"
|
|
124
|
+
else:
|
|
125
|
+
logger.error(f'[Postgres] config info not exist or required keys missing {required_keys}')
|
|
126
|
+
raise ValueError("invalid conn_info")
|
|
127
|
+
|
|
128
|
+
self._connect_db()
|
|
129
|
+
|
|
130
|
+
def __str__(self):
|
|
131
|
+
if self.engine:
|
|
132
|
+
return f"Postgres connected(host={self.host}, port={self.port}, db={self.db})"
|
|
133
|
+
return f"Postgres disconnected(host={self.host}, port={self.port}, db={self.db})"
|
|
134
|
+
|
|
135
|
+
def query_debug(self, use_query_debug=False):
|
|
136
|
+
if isinstance(use_query_debug, bool):
|
|
137
|
+
self.use_query_debug = use_query_debug
|
|
138
|
+
logger.debug(f"use_query_debug = {self.use_query_debug}")
|
|
139
|
+
|
|
140
|
+
def ping(self) -> bool:
|
|
141
|
+
try:
|
|
142
|
+
with self.engine.connect() as conn:
|
|
143
|
+
self._execute_query(conn, "SELECT 1")
|
|
144
|
+
logger.debug(f"[Postgres] database {self.__str__()} connection success")
|
|
145
|
+
return True
|
|
146
|
+
except SQLAlchemyError as e:
|
|
147
|
+
logger.error(f"[Postgres] database {self.__str__()} connection fail: {e}")
|
|
148
|
+
return False
|
|
149
|
+
|
|
150
|
+
def _connect_db(self):
|
|
151
|
+
from urllib.parse import quote_plus
|
|
152
|
+
escaped_passwd = quote_plus(str(self.passwd))
|
|
153
|
+
url = f"postgresql+psycopg2://{self.user}:{escaped_passwd}@{self.host}:{self.port}/{self.db}"
|
|
154
|
+
|
|
155
|
+
connect_args = {
|
|
156
|
+
"application_name": self.application_name,
|
|
157
|
+
}
|
|
158
|
+
if self.options:
|
|
159
|
+
connect_args["options"] = self.options
|
|
160
|
+
if self.sslmode:
|
|
161
|
+
connect_args["sslmode"] = self.sslmode
|
|
162
|
+
|
|
163
|
+
try:
|
|
164
|
+
self.engine = create_engine(
|
|
165
|
+
url,
|
|
166
|
+
echo=False,
|
|
167
|
+
pool_size=self.pool_size,
|
|
168
|
+
max_overflow=self.pool_size * 2,
|
|
169
|
+
pool_timeout=self.pool_timeout,
|
|
170
|
+
pool_recycle=self.pool_recycle,
|
|
171
|
+
pool_pre_ping=True,
|
|
172
|
+
connect_args=connect_args,
|
|
173
|
+
)
|
|
174
|
+
logger.info("[Postgres] DB connected.")
|
|
175
|
+
except SQLAlchemyError as e:
|
|
176
|
+
logger.error(f"[Postgres] DB connection failed. {self.__str__()} : {e}")
|
|
177
|
+
raise
|
|
178
|
+
|
|
179
|
+
@log_execution_time
|
|
180
|
+
def _execute_query(self, conn: Connection, query_str: str, params=None) -> Result[Any]:
|
|
181
|
+
if self.use_percent_param:
|
|
182
|
+
# psycopg2는 %s 스타일이 기본이라 exec_driver_sql과 궁합이 좋음
|
|
183
|
+
if params is None or isinstance(params, (list, tuple)):
|
|
184
|
+
return conn.exec_driver_sql(query_str, params)
|
|
185
|
+
return conn.exec_driver_sql(query_str, (params,))
|
|
186
|
+
else:
|
|
187
|
+
if params is not None and not isinstance(params, dict):
|
|
188
|
+
raise TypeError("When use_percent_param=False, params must be a dict for :name binding.")
|
|
189
|
+
return conn.execute(text(query_str), parameters=params)
|
|
190
|
+
|
|
191
|
+
@log_execution_time
|
|
192
|
+
def _fetch_one(self, rs: Result):
|
|
193
|
+
return rs.fetchone()
|
|
194
|
+
|
|
195
|
+
@log_execution_time
|
|
196
|
+
def _fetch_all(self, rs: Result):
|
|
197
|
+
return rs.fetchall()
|
|
198
|
+
|
|
199
|
+
@log_execution_time
|
|
200
|
+
def _fetch_many(self, rs: Result, fetch_size):
|
|
201
|
+
return rs.fetchmany(size=fetch_size)
|
|
202
|
+
|
|
203
|
+
# -------------------------------------------------------------------------
|
|
204
|
+
# Meta
|
|
205
|
+
# -------------------------------------------------------------------------
|
|
206
|
+
def databases(self) -> pd.DataFrame:
|
|
207
|
+
try:
|
|
208
|
+
with self.engine.connect() as conn:
|
|
209
|
+
return pd.read_sql(
|
|
210
|
+
"SELECT datname AS database FROM pg_database WHERE datistemplate = false ORDER BY datname",
|
|
211
|
+
conn
|
|
212
|
+
)
|
|
213
|
+
except SQLAlchemyError as e:
|
|
214
|
+
logger.error(f"[Postgres] databases Exception: {e}")
|
|
215
|
+
return self.empty_dataframe
|
|
216
|
+
|
|
217
|
+
def tables(self) -> pd.DataFrame:
|
|
218
|
+
try:
|
|
219
|
+
with self.engine.connect() as conn:
|
|
220
|
+
return pd.read_sql(
|
|
221
|
+
"""
|
|
222
|
+
SELECT table_schema, table_name
|
|
223
|
+
FROM information_schema.tables
|
|
224
|
+
WHERE table_type='BASE TABLE'
|
|
225
|
+
AND table_schema NOT IN ('pg_catalog','information_schema')
|
|
226
|
+
ORDER BY table_schema, table_name
|
|
227
|
+
""",
|
|
228
|
+
conn
|
|
229
|
+
)
|
|
230
|
+
except SQLAlchemyError as e:
|
|
231
|
+
logger.error(f"[Postgres] tables Exception: {e}")
|
|
232
|
+
return self.empty_dataframe
|
|
233
|
+
|
|
234
|
+
def current_schema(self) -> str:
|
|
235
|
+
with self.engine.connect() as conn:
|
|
236
|
+
rs = conn.exec_driver_sql("SELECT current_schema()")
|
|
237
|
+
return rs.scalar_one()
|
|
238
|
+
|
|
239
|
+
# -------------------------------------------------------------------------
|
|
240
|
+
# DDL
|
|
241
|
+
# -------------------------------------------------------------------------
|
|
242
|
+
@parse_query('CREATE')
|
|
243
|
+
def create(self, query_str: str, params=None) -> None:
|
|
244
|
+
try:
|
|
245
|
+
with self.engine.begin() as conn:
|
|
246
|
+
self._execute_query(conn, query_str, params)
|
|
247
|
+
except SQLAlchemyError as e:
|
|
248
|
+
logger.debug(f"[Postgres] Create Exception : {e}")
|
|
249
|
+
|
|
250
|
+
@parse_query('DROP')
|
|
251
|
+
def drop(self, query_str: str, params=None) -> None:
|
|
252
|
+
try:
|
|
253
|
+
with self.engine.begin() as conn:
|
|
254
|
+
self._execute_query(conn, query_str, params)
|
|
255
|
+
except SQLAlchemyError as e:
|
|
256
|
+
logger.debug(f"[Postgres] Drop Exception : {e}")
|
|
257
|
+
|
|
258
|
+
@parse_query('TRUNCATE')
|
|
259
|
+
def truncate(self, query_str: str, params=None) -> None:
|
|
260
|
+
try:
|
|
261
|
+
with self.engine.begin() as conn:
|
|
262
|
+
self._execute_query(conn, query_str, params)
|
|
263
|
+
except SQLAlchemyError as e:
|
|
264
|
+
logger.debug(f"[Postgres] Truncate Exception : {e}")
|
|
265
|
+
|
|
266
|
+
@parse_query('ALTER')
|
|
267
|
+
def alter(self, query_str: str, params=None) -> None:
|
|
268
|
+
try:
|
|
269
|
+
with self.engine.begin() as conn:
|
|
270
|
+
self._execute_query(conn, query_str, params)
|
|
271
|
+
except SQLAlchemyError as e:
|
|
272
|
+
logger.debug(f"[Postgres] Alter Exception : {e}")
|
|
273
|
+
|
|
274
|
+
# -------------------------------------------------------------------------
|
|
275
|
+
# SELECT
|
|
276
|
+
# -------------------------------------------------------------------------
|
|
277
|
+
@parse_query('SELECT')
|
|
278
|
+
def select_one(self, query_str: str, params=None) -> dict:
|
|
279
|
+
try:
|
|
280
|
+
with self.engine.connect() as conn:
|
|
281
|
+
rs = self._execute_query(conn, query_str, params)
|
|
282
|
+
row = self._fetch_one(rs)
|
|
283
|
+
return row._asdict() if row else {}
|
|
284
|
+
except SQLAlchemyError as e:
|
|
285
|
+
logger.debug(f"[Postgres] SELECT Exception: {e}")
|
|
286
|
+
return {}
|
|
287
|
+
|
|
288
|
+
@parse_query('SELECT')
|
|
289
|
+
def select_list(self, query_str: str, params=None) -> list:
|
|
290
|
+
try:
|
|
291
|
+
with self.engine.connect() as conn:
|
|
292
|
+
rs = self._execute_query(conn, query_str, params)
|
|
293
|
+
rows = self._fetch_all(rs)
|
|
294
|
+
if not rows:
|
|
295
|
+
return []
|
|
296
|
+
return [r._asdict() for r in rows] if isinstance(rows, list) else [rows._asdict()]
|
|
297
|
+
except SQLAlchemyError as e:
|
|
298
|
+
logger.debug(f"[Postgres] SELECT_LIST Exception : {e}")
|
|
299
|
+
return []
|
|
300
|
+
|
|
301
|
+
@parse_query('SELECT')
|
|
302
|
+
def select(self, query_str: str, params=None) -> pd.DataFrame:
|
|
303
|
+
try:
|
|
304
|
+
with self.engine.connect() as conn:
|
|
305
|
+
rs = self._execute_query(conn, query_str, params)
|
|
306
|
+
rows = self._fetch_all(rs)
|
|
307
|
+
if rows:
|
|
308
|
+
columns = list(rs.keys())
|
|
309
|
+
return pd.DataFrame(rows, columns=columns)
|
|
310
|
+
return self.empty_dataframe
|
|
311
|
+
except SQLAlchemyError as e:
|
|
312
|
+
logger.error(f"[Postgres] SELECT Exception : {e}")
|
|
313
|
+
return self.empty_dataframe
|
|
314
|
+
|
|
315
|
+
@parse_query('SELECT')
|
|
316
|
+
@log_execution_time
|
|
317
|
+
def faster_select(self, query_str: str, params=None, fetch_size=1000) -> pd.DataFrame:
|
|
318
|
+
total_size = 0
|
|
319
|
+
results = []
|
|
320
|
+
try:
|
|
321
|
+
with self.engine.connect() as conn:
|
|
322
|
+
conn = conn.execution_options(stream_results=True)
|
|
323
|
+
rs = self._execute_query(conn, query_str, params)
|
|
324
|
+
columns = list(rs.keys())
|
|
325
|
+
|
|
326
|
+
while True:
|
|
327
|
+
rows = self._fetch_many(rs, fetch_size)
|
|
328
|
+
if not rows:
|
|
329
|
+
break
|
|
330
|
+
total_size += len(rows)
|
|
331
|
+
results.extend(rows)
|
|
332
|
+
if self.use_query_debug:
|
|
333
|
+
logger.debug(f"fetch {len(rows)} rows, total fetched size = {total_size}")
|
|
334
|
+
|
|
335
|
+
return pd.DataFrame(results, columns=columns) if results else self.empty_dataframe
|
|
336
|
+
except SQLAlchemyError as e:
|
|
337
|
+
logger.debug(f"[Postgres] FASTER_SELECT Exception : {e}")
|
|
338
|
+
self.close()
|
|
339
|
+
return self.empty_dataframe
|
|
340
|
+
|
|
341
|
+
@parse_query('SELECT')
|
|
342
|
+
def faster_select_generator(self, query_str: str, params=None, fetch_size=1000):
|
|
343
|
+
try:
|
|
344
|
+
with self.engine.connect() as conn:
|
|
345
|
+
conn = conn.execution_options(stream_results=True)
|
|
346
|
+
rs = self._execute_query(conn, query_str, params)
|
|
347
|
+
maps = rs.mappings()
|
|
348
|
+
|
|
349
|
+
while True:
|
|
350
|
+
part = maps.fetchmany(fetch_size)
|
|
351
|
+
if not part:
|
|
352
|
+
break
|
|
353
|
+
rows = [dict(r) for r in part]
|
|
354
|
+
if self.use_query_debug:
|
|
355
|
+
logger.debug(f"fetch {len(rows)} rows chunk")
|
|
356
|
+
yield rows
|
|
357
|
+
except SQLAlchemyError as e:
|
|
358
|
+
logger.debug(f"[Postgres] faster_select_generator Exception : {e}")
|
|
359
|
+
return
|
|
360
|
+
|
|
361
|
+
# -------------------------------------------------------------------------
|
|
362
|
+
# INSERT/UPDATE/DELETE
|
|
363
|
+
# -------------------------------------------------------------------------
|
|
364
|
+
@parse_query('INSERT')
|
|
365
|
+
def insert(self, query_str: str, params=None, return_lastrowid=False) -> int:
|
|
366
|
+
"""
|
|
367
|
+
Postgres에서 lastrowid는 일반적으로 RETURNING을 써야 안정적입니다.
|
|
368
|
+
- 기존 호환 위해 return_lastrowid True면:
|
|
369
|
+
1) 쿼리에 RETURNING이 있으면 scalar 반환
|
|
370
|
+
2) 없으면 rowcount 반환
|
|
371
|
+
"""
|
|
372
|
+
try:
|
|
373
|
+
with self.engine.begin() as conn:
|
|
374
|
+
rs = self._execute_query(conn, query_str, params)
|
|
375
|
+
|
|
376
|
+
if return_lastrowid:
|
|
377
|
+
# RETURNING 이 있을 때만 의미 있음
|
|
378
|
+
try:
|
|
379
|
+
v = rs.scalar_one()
|
|
380
|
+
return int(v)
|
|
381
|
+
except Exception:
|
|
382
|
+
return int(rs.rowcount or 0)
|
|
383
|
+
return int(rs.rowcount or 0)
|
|
384
|
+
except SQLAlchemyError as e:
|
|
385
|
+
logger.error(f"[Postgres] INSERT Exception : {e}")
|
|
386
|
+
return 0
|
|
387
|
+
|
|
388
|
+
@parse_query('UPDATE')
|
|
389
|
+
def update(self, query_str: str, params=None) -> int:
|
|
390
|
+
try:
|
|
391
|
+
with self.engine.begin() as conn:
|
|
392
|
+
rs = self._execute_query(conn, query_str, params)
|
|
393
|
+
return int(rs.rowcount or 0)
|
|
394
|
+
except SQLAlchemyError as e:
|
|
395
|
+
logger.error(f"[Postgres] UPDATE Exception : {e}")
|
|
396
|
+
return 0
|
|
397
|
+
|
|
398
|
+
@parse_query('DELETE')
|
|
399
|
+
def delete(self, query_str: str, params=None) -> int:
|
|
400
|
+
try:
|
|
401
|
+
with self.engine.begin() as conn:
|
|
402
|
+
rs = self._execute_query(conn, query_str, params)
|
|
403
|
+
return int(rs.rowcount or 0)
|
|
404
|
+
except SQLAlchemyError as e:
|
|
405
|
+
logger.debug(f"[Postgres] DELETE Exception : {e}")
|
|
406
|
+
return 0
|
|
407
|
+
|
|
408
|
+
def close(self, close_log=True):
|
|
409
|
+
if self.engine:
|
|
410
|
+
self.engine.dispose()
|
|
411
|
+
self.engine = None
|
|
412
|
+
if close_log:
|
|
413
|
+
logger.debug("[Postgres] DB Connection closed.")
|
|
414
|
+
|
|
415
|
+
def __del__(self):
|
|
416
|
+
try:
|
|
417
|
+
self.close(close_log=False)
|
|
418
|
+
except Exception:
|
|
419
|
+
pass
|