echoss-db 2.3.1__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {echoss_db-2.3.1/echoss_db.egg-info → echoss_db-2.4.0}/PKG-INFO +31 -2
- {echoss_db-2.3.1 → echoss_db-2.4.0}/README.md +29 -0
- echoss_db-2.4.0/echoss_db/chunk_output.py +204 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/elastic_search.py +109 -9
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mysql_query.py +62 -4
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/postgres_query.py +59 -1
- {echoss_db-2.3.1 → echoss_db-2.4.0/echoss_db.egg-info}/PKG-INFO +31 -2
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/SOURCES.txt +6 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/requires.txt +1 -1
- {echoss_db-2.3.1 → echoss_db-2.4.0}/pyproject.toml +2 -2
- {echoss_db-2.3.1 → echoss_db-2.4.0}/requirements.txt +1 -1
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/conftest.py +28 -0
- echoss_db-2.4.0/tests/integration/test_chunks_fileformat_roundtrip.py +74 -0
- echoss_db-2.4.0/tests/integration/test_mysql_chunks.py +93 -0
- echoss_db-2.4.0/tests/integration/test_opensearch_chunks.py +105 -0
- echoss_db-2.4.0/tests/integration/test_postgres_chunks.py +111 -0
- echoss_db-2.4.0/tests/unit/test_chunk_output.py +240 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/LICENSE +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/MANIFEST.in +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/env_config.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/dataclass_mapper.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/errors.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/param_mapper.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/pydantic_mapper.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/row_mapper.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/types.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mongo_query.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/qdrant_vector.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/sql_transaction.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/dependency_links.txt +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/top_level.txt +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/package_tests/test_package_imports.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/setup.cfg +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/test_env_config_integration.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/test_postgres_integration.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/test_qdrant_integration.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/_postgres_test_schema.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_elasticsearch.ipynb +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_elasticsearch.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mapping_mysql.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mongo.ipynb +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mongo.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mysql.ipynb +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mysql.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_postgres.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_postgres_qdrant_ingest.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_qdrant.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/setup_ai_rag_test_schema.sql +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/mapping/__init__.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/mapping/test_param_mapper.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/test_env_config.py +0 -0
- {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/test_qdrant_role.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: echoss-db
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: echoss AI Bigdata Solution - Database Query Package
|
|
5
5
|
Author-email: ckkim <ckkim@12cm.co.kr>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -16,7 +16,7 @@ License-File: LICENSE
|
|
|
16
16
|
Requires-Dist: pandas>=3.0
|
|
17
17
|
Requires-Dist: sqlalchemy>=2.0.0
|
|
18
18
|
Requires-Dist: PyMySQL>=1.0.2
|
|
19
|
-
Requires-Dist: opensearch-py<3
|
|
19
|
+
Requires-Dist: opensearch-py<4,>=3
|
|
20
20
|
Requires-Dist: echoss-common<3,>=2.0.0
|
|
21
21
|
Requires-Dist: psycopg[binary]<4.0.0,>=3.2
|
|
22
22
|
Requires-Dist: qdrant-client[fastembed]<2.0.0,>=1.14.1
|
|
@@ -447,6 +447,34 @@ elastic.ping()
|
|
|
447
447
|
elastic.info()
|
|
448
448
|
```
|
|
449
449
|
|
|
450
|
+
### Chunk Output (2.4.0)
|
|
451
|
+
|
|
452
|
+
---
|
|
453
|
+
|
|
454
|
+
Large results as fixed-schema chunks: `res.schema` comes from DB metadata (MySQL/Postgres `information_schema`, OpenSearch mapping) and can be read before iteration; iterating yields `(rows, resume_token)`. Errors are raised, never swallowed.
|
|
455
|
+
|
|
456
|
+
```python
|
|
457
|
+
# MySQL / Postgres — keyset on a unique, sortable key column (integer or string)
|
|
458
|
+
res = mysql.select_chunks('orders', key='id', columns=['id', 'status', 'created'],
|
|
459
|
+
where='status = %s', params=('paid',), chunk_size=10000)
|
|
460
|
+
res.schema # e.g. {'id': 'int64', 'status': 'string', 'created': 'timestamp[us]'}
|
|
461
|
+
for rows, token in res:
|
|
462
|
+
... # token = last key of the chunk; pass resume_token=token to continue
|
|
463
|
+
|
|
464
|
+
# OpenSearch — PIT + search_after on a unique sort field
|
|
465
|
+
res = elastic.search_chunks('shopby-log-v1-dev', sort_key='received_at', source_fields=['received_at', 'items'])
|
|
466
|
+
|
|
467
|
+
# write with echoss-fileformat (overrides / type errors are handled there)
|
|
468
|
+
from echoss_fileformat import FileUtil
|
|
469
|
+
FileUtil.dump_chunks(res, 'out_dir')
|
|
470
|
+
```
|
|
471
|
+
|
|
472
|
+
- Values are in the declared type: OpenSearch `date` → `timestamp[ms,UTC]` (epoch-ms ints kept as is, sub-ms digits truncated), `nested`/`knn_vector` → `json_array`, `object` → `json_object`; SQL `json`/`jsonb` → JSON text (`string`). `date_nanos` is not supported (company policy: ms/us timestamps only).
|
|
473
|
+
- One shape per column: values that do not match the schema (e.g. a list in a `keyword` field) are passed through and rejected by echoss-fileformat (`ChunkTypeError`); use its `overrides` to read them differently.
|
|
474
|
+
- MySQL `DATETIME`: set `time_zone` (IANA name, e.g. `Asia/Manila`) in the `mysql` config section to get `timestamp[us,UTC]`; without it `timestamp[us]`.
|
|
475
|
+
- Resuming re-reads from the token: rows changed in the meantime are reflected. `key`/`sort_key` must be unique, or rows may be skipped or repeated.
|
|
476
|
+
- `where` is a caller-written SQL fragment (same trust level as `select()`); bind values via `params`.
|
|
477
|
+
|
|
450
478
|
### Qdrant
|
|
451
479
|
|
|
452
480
|
---
|
|
@@ -634,3 +662,4 @@ v2.2.0 qdrant generalized dict filter (breaking behavior: previously ignored key
|
|
|
634
662
|
v2.2.2 qdrant admin/read-only API key support (`role` param, `api_key`/`read_only_api_key` config fields), config.yaml credential fields (mysql/postgres/elastic `passwd`, qdrant `api_key`/`read_only_api_key`) now accept `${VAR_NAME}` placeholders resolved from `.env`/OS env (new `python-dotenv` dependency)
|
|
635
663
|
v2.3.0 config loading and logger now imported from `echoss-common` (`>=1.2.0,<3`) instead of `echoss-fileformat`; `echoss-fileformat` removed from install dependencies (declare it directly if you use it); raise pandas floor to 3.0. XML config files are now read by `echoss-common` (result keys may differ from the previous `echoss-fileformat` XML reader)
|
|
636
664
|
v2.3.1 raise `echoss-common` floor to `>=2.0.0,<3` (no code change). With `echoss-common` 2.0: no default `logs/echoss.log` file (console only), and a missing config file path raises `FileNotFoundError` (with `echoss-common` 1.x: `ValueError`/`TypeError`)
|
|
665
|
+
v2.4.0 chunk output: `MysqlQuery.select_chunks`, `PostgresQuery.select_chunks`, `ElasticSearch.search_chunks` (schema from DB metadata + chunks + resume token, OpenSearch via PIT); `faster_select_generator` and `next_scroll_chunk` now raise errors instead of ending silently (a scroll context that expired raises `NotFoundError` instead of returning `[]`); `opensearch-py>=3,<4`; MySQL config `time_zone`
|
|
@@ -415,6 +415,34 @@ elastic.ping()
|
|
|
415
415
|
elastic.info()
|
|
416
416
|
```
|
|
417
417
|
|
|
418
|
+
### Chunk Output (2.4.0)
|
|
419
|
+
|
|
420
|
+
---
|
|
421
|
+
|
|
422
|
+
Large results as fixed-schema chunks: `res.schema` comes from DB metadata (MySQL/Postgres `information_schema`, OpenSearch mapping) and can be read before iteration; iterating yields `(rows, resume_token)`. Errors are raised, never swallowed.
|
|
423
|
+
|
|
424
|
+
```python
|
|
425
|
+
# MySQL / Postgres — keyset on a unique, sortable key column (integer or string)
|
|
426
|
+
res = mysql.select_chunks('orders', key='id', columns=['id', 'status', 'created'],
|
|
427
|
+
where='status = %s', params=('paid',), chunk_size=10000)
|
|
428
|
+
res.schema # e.g. {'id': 'int64', 'status': 'string', 'created': 'timestamp[us]'}
|
|
429
|
+
for rows, token in res:
|
|
430
|
+
... # token = last key of the chunk; pass resume_token=token to continue
|
|
431
|
+
|
|
432
|
+
# OpenSearch — PIT + search_after on a unique sort field
|
|
433
|
+
res = elastic.search_chunks('shopby-log-v1-dev', sort_key='received_at', source_fields=['received_at', 'items'])
|
|
434
|
+
|
|
435
|
+
# write with echoss-fileformat (overrides / type errors are handled there)
|
|
436
|
+
from echoss_fileformat import FileUtil
|
|
437
|
+
FileUtil.dump_chunks(res, 'out_dir')
|
|
438
|
+
```
|
|
439
|
+
|
|
440
|
+
- Values are in the declared type: OpenSearch `date` → `timestamp[ms,UTC]` (epoch-ms ints kept as is, sub-ms digits truncated), `nested`/`knn_vector` → `json_array`, `object` → `json_object`; SQL `json`/`jsonb` → JSON text (`string`). `date_nanos` is not supported (company policy: ms/us timestamps only).
|
|
441
|
+
- One shape per column: values that do not match the schema (e.g. a list in a `keyword` field) are passed through and rejected by echoss-fileformat (`ChunkTypeError`); use its `overrides` to read them differently.
|
|
442
|
+
- MySQL `DATETIME`: set `time_zone` (IANA name, e.g. `Asia/Manila`) in the `mysql` config section to get `timestamp[us,UTC]`; without it `timestamp[us]`.
|
|
443
|
+
- Resuming re-reads from the token: rows changed in the meantime are reflected. `key`/`sort_key` must be unique, or rows may be skipped or repeated.
|
|
444
|
+
- `where` is a caller-written SQL fragment (same trust level as `select()`); bind values via `params`.
|
|
445
|
+
|
|
418
446
|
### Qdrant
|
|
419
447
|
|
|
420
448
|
---
|
|
@@ -602,3 +630,4 @@ v2.2.0 qdrant generalized dict filter (breaking behavior: previously ignored key
|
|
|
602
630
|
v2.2.2 qdrant admin/read-only API key support (`role` param, `api_key`/`read_only_api_key` config fields), config.yaml credential fields (mysql/postgres/elastic `passwd`, qdrant `api_key`/`read_only_api_key`) now accept `${VAR_NAME}` placeholders resolved from `.env`/OS env (new `python-dotenv` dependency)
|
|
603
631
|
v2.3.0 config loading and logger now imported from `echoss-common` (`>=1.2.0,<3`) instead of `echoss-fileformat`; `echoss-fileformat` removed from install dependencies (declare it directly if you use it); raise pandas floor to 3.0. XML config files are now read by `echoss-common` (result keys may differ from the previous `echoss-fileformat` XML reader)
|
|
604
632
|
v2.3.1 raise `echoss-common` floor to `>=2.0.0,<3` (no code change). With `echoss-common` 2.0: no default `logs/echoss.log` file (console only), and a missing config file path raises `FileNotFoundError` (with `echoss-common` 1.x: `ValueError`/`TypeError`)
|
|
633
|
+
v2.4.0 chunk output: `MysqlQuery.select_chunks`, `PostgresQuery.select_chunks`, `ElasticSearch.search_chunks` (schema from DB metadata + chunks + resume token, OpenSearch via PIT); `faster_select_generator` and `next_scroll_chunk` now raise errors instead of ending silently (a scroll context that expired raises `NotFoundError` instead of returning `[]`); `opensearch-py>=3,<4`; MySQL config `time_zone`
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""
|
|
2
|
+
조각 출력 — 정답 스키마 + 그 형식의 값 조각 + 재개 표지
|
|
3
|
+
|
|
4
|
+
Design Ref: docs/02-design/features/db-raw-result-chunk-output.design.md §3·§4.1
|
|
5
|
+
- 스키마는 DB 메타데이터(OpenSearch 매핑, MySQL·Postgres information_schema)로 만든다
|
|
6
|
+
- 값은 DB 가 선언한 형식의 값: 드라이버 표현이 선언 형식과 다를 때만 맞추고 값의 뜻은 바꾸지 않는다
|
|
7
|
+
- 값 형식 검사·덮어쓰기는 echoss-fileformat(ChunkTypeError·overrides) 몫 — 여기서는 검사하지 않는다(§3.4)
|
|
8
|
+
- 형식 이름은 echoss-fileformat chunk_schema.py 의 이름(문자열). pyarrow·echoss-fileformat 은 import 하지 않는다
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import base64
|
|
13
|
+
import datetime
|
|
14
|
+
import uuid
|
|
15
|
+
from typing import Iterable, Optional, Tuple
|
|
16
|
+
from zoneinfo import ZoneInfo
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ChunkResult:
|
|
20
|
+
"""새 조각 함수의 결과. schema 는 반복 전에 읽을 수 있고, 반복하면 (rows, resume_token) 을 낸다"""
|
|
21
|
+
|
|
22
|
+
def __init__(self, schema: dict, chunks: Iterable[Tuple[list, object]]):
|
|
23
|
+
self.schema = schema
|
|
24
|
+
self.chunks = chunks
|
|
25
|
+
|
|
26
|
+
def __iter__(self):
|
|
27
|
+
return iter(self.chunks)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# §3.1 OpenSearch — 공식 필드 형식 목록 전체 (docs.opensearch.org supported-field-types, 2026-10-10 조회)
|
|
31
|
+
OPENSEARCH_TYPES = {
|
|
32
|
+
'boolean': 'bool', 'binary': 'binary',
|
|
33
|
+
'text': 'string', 'match_only_text': 'string', 'keyword': 'string', 'constant_keyword': 'string',
|
|
34
|
+
'icu_collation_keyword': 'string', 'wildcard': 'string', 'version': 'string',
|
|
35
|
+
'search_as_you_type': 'string', 'semantic': 'string', 'ip': 'string',
|
|
36
|
+
'token_count': 'int32', 'byte': 'int8', 'short': 'int16', 'integer': 'int32', 'long': 'int64',
|
|
37
|
+
'unsigned_long': 'uint64', 'half_float': 'float32', 'float': 'float32', 'double': 'float64',
|
|
38
|
+
'scaled_float': 'float64', 'rank_feature': 'float32',
|
|
39
|
+
'date': 'timestamp[ms,UTC]',
|
|
40
|
+
'geo_point': 'json_object', 'geo_shape': 'json_object', 'xy_point': 'json_object', 'xy_shape': 'json_object',
|
|
41
|
+
'integer_range': 'json_object', 'long_range': 'json_object', 'double_range': 'json_object',
|
|
42
|
+
'float_range': 'json_object', 'ip_range': 'json_object', 'date_range': 'json_object',
|
|
43
|
+
'object': 'json_object', 'flat_object': 'json_object', 'join': 'json_object',
|
|
44
|
+
'completion': 'json_object', 'percolator': 'json_object', 'rank_features': 'json_object',
|
|
45
|
+
'sparse_vector': 'json_object',
|
|
46
|
+
'nested': 'json_array', 'knn_vector': 'json_array',
|
|
47
|
+
}
|
|
48
|
+
OPENSEARCH_EXCLUDED = {'alias', 'star_tree', 'derived'} # _source 에 없거나 문서 필드가 아님
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def opensearch_type_name(name: str, mapping_field: dict) -> Optional[str]:
|
|
52
|
+
"""매핑 한 필드 → 형식 이름. None 이면 스키마에서 뺀다. 옮길 수 없는 형식은 ValueError"""
|
|
53
|
+
type_ = mapping_field.get('type', 'object' if 'properties' in mapping_field else None)
|
|
54
|
+
if type_ in OPENSEARCH_EXCLUDED:
|
|
55
|
+
return None
|
|
56
|
+
if type_ == 'date_nanos':
|
|
57
|
+
raise ValueError(f"[OpenSearch] field '{name}': date_nanos is not supported (company policy: ms/us timestamps only)")
|
|
58
|
+
if type_ not in OPENSEARCH_TYPES:
|
|
59
|
+
raise ValueError(f"[OpenSearch] field '{name}': unsupported mapping type {type_!r}")
|
|
60
|
+
return OPENSEARCH_TYPES[type_]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# §3.2 MySQL
|
|
64
|
+
MYSQL_INTS = {'tinyint': 'int8', 'smallint': 'int16', 'mediumint': 'int32', 'int': 'int32', 'bigint': 'int64'}
|
|
65
|
+
MYSQL_STRINGS = {'char', 'varchar', 'tinytext', 'text', 'mediumtext', 'longtext', 'enum', 'set', 'json'}
|
|
66
|
+
MYSQL_BINARIES = {'bit', 'binary', 'varbinary', 'tinyblob', 'blob', 'mediumblob', 'longblob', 'geometry', 'point',
|
|
67
|
+
'linestring', 'polygon', 'multipoint', 'multilinestring', 'multipolygon', 'geometrycollection'}
|
|
68
|
+
MYSQL_OTHERS = {'float': 'float32', 'double': 'float64', 'date': 'date32', 'timestamp': 'timestamp[us]',
|
|
69
|
+
'time': 'duration[us]', 'year': 'int16'}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def mysql_type_name(column: dict, time_zone: Optional[str] = None) -> str:
|
|
73
|
+
"""information_schema.columns 한 행(DATA_TYPE·COLUMN_TYPE·NUMERIC_PRECISION·NUMERIC_SCALE·COLUMN_NAME) → 형식 이름"""
|
|
74
|
+
name = column['COLUMN_NAME']
|
|
75
|
+
data_type = column['DATA_TYPE'].lower()
|
|
76
|
+
if data_type in MYSQL_INTS:
|
|
77
|
+
type_ = MYSQL_INTS[data_type]
|
|
78
|
+
return 'u' + type_ if 'unsigned' in column['COLUMN_TYPE'].lower() else type_
|
|
79
|
+
if data_type == 'decimal':
|
|
80
|
+
return decimal_type_name('MySQL', name, column['NUMERIC_PRECISION'], column['NUMERIC_SCALE'])
|
|
81
|
+
if data_type in MYSQL_STRINGS:
|
|
82
|
+
return 'string' # json 은 JSON 텍스트 그대로(D13)
|
|
83
|
+
if data_type in MYSQL_BINARIES:
|
|
84
|
+
return 'binary'
|
|
85
|
+
if data_type == 'datetime':
|
|
86
|
+
return 'timestamp[us,UTC]' if time_zone else 'timestamp[us]'
|
|
87
|
+
if data_type in MYSQL_OTHERS:
|
|
88
|
+
return MYSQL_OTHERS[data_type]
|
|
89
|
+
raise ValueError(f"[MySQL] column '{name}': unsupported data type {data_type!r}")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# §3.3 Postgres
|
|
93
|
+
POSTGRES_TYPES = {
|
|
94
|
+
'smallint': 'int16', 'integer': 'int32', 'bigint': 'int64', 'real': 'float32', 'double precision': 'float64',
|
|
95
|
+
'boolean': 'bool', 'character': 'string', 'character varying': 'string', 'text': 'string', 'uuid': 'string',
|
|
96
|
+
'json': 'string', 'jsonb': 'string', 'bytea': 'binary', 'ARRAY': 'json_array', 'date': 'date32',
|
|
97
|
+
'timestamp without time zone': 'timestamp[us]', 'timestamp with time zone': 'timestamp[us,UTC]',
|
|
98
|
+
'time without time zone': 'duration[us]',
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def postgres_type_name(column: dict) -> str:
|
|
103
|
+
"""information_schema.columns 한 행(data_type·numeric_precision·numeric_scale·column_name) → 형식 이름"""
|
|
104
|
+
name = column['column_name']
|
|
105
|
+
data_type = column['data_type']
|
|
106
|
+
if data_type == 'numeric':
|
|
107
|
+
if column['numeric_precision'] is None:
|
|
108
|
+
raise ValueError(f"[Postgres] column '{name}': numeric without precision has no fixed type (cast it in a view)")
|
|
109
|
+
return decimal_type_name('Postgres', name, column['numeric_precision'], column['numeric_scale'])
|
|
110
|
+
if data_type in POSTGRES_TYPES:
|
|
111
|
+
return POSTGRES_TYPES[data_type]
|
|
112
|
+
raise ValueError(f"[Postgres] column '{name}': unsupported data type {data_type!r}")
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def decimal_type_name(db: str, name: str, precision, scale) -> str:
|
|
116
|
+
if int(precision) > 38:
|
|
117
|
+
raise ValueError(f"[{db}] column '{name}': decimal precision {precision} > 38")
|
|
118
|
+
return f"decimal({int(precision)},{int(scale or 0)})"
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# §3.4 값 맞추기 — 검사하지 않는다
|
|
122
|
+
def convert_rows(rows: list, schema: dict, chunk_no: int, db: str, time_zone: Optional[str] = None) -> list:
|
|
123
|
+
"""드라이버 표현을 선언 형식으로 맞춘 행. 변환 실패(날짜 해석, MySQL 0 날짜)만 ValueError"""
|
|
124
|
+
zone = ZoneInfo(time_zone) if time_zone else None
|
|
125
|
+
for row in rows:
|
|
126
|
+
for column, type_name in schema.items():
|
|
127
|
+
value = row.get(column)
|
|
128
|
+
if value is None:
|
|
129
|
+
continue
|
|
130
|
+
try:
|
|
131
|
+
row[column] = convert_value(value, type_name, db, zone)
|
|
132
|
+
except ValueError as e:
|
|
133
|
+
raise ValueError(f"[{db}] column '{column}' chunk {chunk_no}: {e}") from e
|
|
134
|
+
return rows
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def convert_value(value, type_name: str, db: str, zone):
|
|
138
|
+
if db == 'MySQL' and isinstance(value, str) and value.startswith('0000-00-00') \
|
|
139
|
+
and type_name in ('date32', 'timestamp[us]', 'timestamp[us,UTC]'):
|
|
140
|
+
raise ValueError(f"zero date {value!r}")
|
|
141
|
+
if type_name == 'timestamp[ms,UTC]': # OpenSearch date (D6)
|
|
142
|
+
if isinstance(value, str):
|
|
143
|
+
return opensearch_date(value)
|
|
144
|
+
return value # epoch 밀리초 정수는 그대로
|
|
145
|
+
if type_name == 'timestamp[us,UTC]' and isinstance(value, datetime.datetime):
|
|
146
|
+
if value.tzinfo is None: # MySQL DATETIME: 서비스 현지 시각 + 연결 설정 시간대 (D7)
|
|
147
|
+
value = value.replace(tzinfo=zone)
|
|
148
|
+
return value.astimezone(datetime.timezone.utc)
|
|
149
|
+
if type_name == 'duration[us]' and isinstance(value, datetime.time): # Postgres time: 자정부터의 경과(§3.3)
|
|
150
|
+
return datetime.timedelta(hours=value.hour, minutes=value.minute, seconds=value.second,
|
|
151
|
+
microseconds=value.microsecond)
|
|
152
|
+
if type_name == 'binary' and isinstance(value, str) and db == 'OpenSearch':
|
|
153
|
+
return base64.b64decode(value)
|
|
154
|
+
if type_name == 'string' and isinstance(value, uuid.UUID):
|
|
155
|
+
return str(value)
|
|
156
|
+
return value
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def opensearch_date(value: str) -> datetime.datetime:
|
|
160
|
+
"""ISO 문자열 → UTC, 밀리초 아래는 자른다(OpenSearch 색인값과 같음). 날짜만 적힌 값은 그날 0시 UTC.
|
|
161
|
+
시간대 표기가 없으면 UTC 로 본다(OpenSearch date 해석 — 공식 문서 기준 지식)"""
|
|
162
|
+
try:
|
|
163
|
+
parsed = datetime.datetime.fromisoformat(value)
|
|
164
|
+
except ValueError as e:
|
|
165
|
+
raise ValueError(f"cannot parse date {value!r}") from e
|
|
166
|
+
if parsed.tzinfo is None:
|
|
167
|
+
parsed = parsed.replace(tzinfo=datetime.timezone.utc)
|
|
168
|
+
parsed = parsed.astimezone(datetime.timezone.utc)
|
|
169
|
+
return parsed.replace(microsecond=parsed.microsecond // 1000 * 1000)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
# §4.2 SQL keyset — MySQL·Postgres 공통 반복
|
|
173
|
+
KEY_TYPES = {'int8', 'int16', 'int32', 'int64', 'uint8', 'uint16', 'uint32', 'uint64', 'string'}
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def keyset_chunks(fetch_page, key: str, chunk_size: int, resume_token, convert, drop_key: bool):
|
|
177
|
+
"""fetch_page(last, n) → 행 목록(key 오름차순, key > last). 조각마다 (rows, 마지막 key 값) 을 낸다.
|
|
178
|
+
오류는 잡지 않는다 — 끝과 오류를 구분하게 다시 던진다(계약 7)"""
|
|
179
|
+
last = resume_token
|
|
180
|
+
chunk_no = 0
|
|
181
|
+
while True:
|
|
182
|
+
rows = fetch_page(last, chunk_size)
|
|
183
|
+
if not rows:
|
|
184
|
+
return
|
|
185
|
+
last = rows[-1][key]
|
|
186
|
+
if isinstance(last, uuid.UUID): # 재개 표지도 선언 형식(string)으로 — echoss-fileformat 이 JSON 으로 기록(Check 게이트 R2)
|
|
187
|
+
last = str(last)
|
|
188
|
+
if drop_key:
|
|
189
|
+
for row in rows:
|
|
190
|
+
row.pop(key)
|
|
191
|
+
yield convert(rows, chunk_no), last
|
|
192
|
+
chunk_no += 1
|
|
193
|
+
if len(rows) < chunk_size:
|
|
194
|
+
return
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def sql_select_list(columns: list, quote, text_columns: set) -> str:
|
|
198
|
+
"""식별자를 인용한 SELECT 목록. text_columns 는 ::text 로 읽는다(Postgres json·jsonb, D13)"""
|
|
199
|
+
return ", ".join(f"{quote(c)}::text AS {quote(c)}" if c in text_columns else quote(c) for c in columns)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def check_key(db: str, key: str, schema_types: dict):
|
|
203
|
+
if schema_types[key] not in KEY_TYPES:
|
|
204
|
+
raise ValueError(f"[{db}] key '{key}' must be an integer or string column, got {schema_types[key]}")
|
|
@@ -7,6 +7,7 @@ from typing import Any, List, Tuple, Dict, Optional, Union
|
|
|
7
7
|
from echoss_common import dict_load, get_logger, set_logger_level
|
|
8
8
|
|
|
9
9
|
from .env_config import resolve_credential
|
|
10
|
+
from .chunk_output import ChunkResult, convert_rows, opensearch_type_name
|
|
10
11
|
|
|
11
12
|
logger = get_logger("echoss_query")
|
|
12
13
|
|
|
@@ -121,6 +122,104 @@ class ElasticSearch:
|
|
|
121
122
|
except Exception as e:
|
|
122
123
|
raise ValueError("Connection failed by config. Please check config data")
|
|
123
124
|
|
|
125
|
+
def search_chunks(self, index: str, sort_key: str, query: dict = None, source_fields: List[str] = None,
|
|
126
|
+
chunk_size: int = 10000, keep_alive: str = '5m', resume_token: list = None) -> ChunkResult:
|
|
127
|
+
"""대량 조회를 「정답 스키마 + 값 조각 + 재개 표지」로 낸다 (Design Ref: §3.1·§4.2, PIT + search_after).
|
|
128
|
+
|
|
129
|
+
Args:
|
|
130
|
+
index: 인덱스 이름(별칭·패턴이면 묶인 인덱스들의 스키마가 같아야 한다)
|
|
131
|
+
sort_key: 정렬·재개에 쓸 유일한 필드. 유일하지 않으면 재개 때 문서가 빠지거나 겹칠 수 있다
|
|
132
|
+
query: 질의 본문의 query 부분(기본 match_all)
|
|
133
|
+
source_fields: 낼 필드 목록(점 경로 가능), 없으면 매핑 최상위 필드 전체
|
|
134
|
+
keep_alive: PIT 유지 시간
|
|
135
|
+
resume_token: 이전 조각의 재개 표지(search_after 정렬값). 새 PIT 를 열므로 그사이 바뀐 자료는 반영된다
|
|
136
|
+
Returns:
|
|
137
|
+
ChunkResult — schema 는 매핑에서, 값은 _source 에서(색인값 docvalue_fields 는 쓰지 않음)
|
|
138
|
+
"""
|
|
139
|
+
try:
|
|
140
|
+
mappings = self.conn.indices.get_mapping(index=index)
|
|
141
|
+
except NotFoundError as e:
|
|
142
|
+
raise ValueError(f"[OpenSearch] index '{index}' not found") from e
|
|
143
|
+
schemas = [self.mapping_schema(m['mappings'].get('properties', {}), source_fields) for m in mappings.values()]
|
|
144
|
+
if not schemas:
|
|
145
|
+
raise ValueError(f"[OpenSearch] index '{index}' not found")
|
|
146
|
+
if any(s != schemas[0] for s in schemas[1:]):
|
|
147
|
+
raise ValueError(f"[OpenSearch] indices behind '{index}' have different mappings for the requested fields")
|
|
148
|
+
schema = schemas[0]
|
|
149
|
+
for m in mappings.values(): # 다중 필드(name.keyword)는 정렬 키로 받지 않는다(설계 §4.2)
|
|
150
|
+
if self.mapping_field(m['mappings'].get('properties', {}), sort_key) is None:
|
|
151
|
+
raise ValueError(f"[OpenSearch] sort_key '{sort_key}' not found in mapping of '{index}' (multi-fields are not supported)")
|
|
152
|
+
|
|
153
|
+
def chunks():
|
|
154
|
+
pit_id = self.conn.create_pit(index=index, keep_alive=keep_alive)['pit_id']
|
|
155
|
+
try:
|
|
156
|
+
last = resume_token
|
|
157
|
+
chunk_no = 0
|
|
158
|
+
while True:
|
|
159
|
+
body = {"size": chunk_size, "query": query or {"match_all": {}}, "_source": list(schema),
|
|
160
|
+
"sort": [{sort_key: "asc"}], "pit": {"id": pit_id, "keep_alive": keep_alive}}
|
|
161
|
+
if last is not None:
|
|
162
|
+
body["search_after"] = last
|
|
163
|
+
response = self.conn.search(body=body)
|
|
164
|
+
pit_id = response.get('pit_id', pit_id)
|
|
165
|
+
hits = response['hits']['hits']
|
|
166
|
+
if not hits:
|
|
167
|
+
return
|
|
168
|
+
try:
|
|
169
|
+
rows = [{f: self.source_value(h.get('_source', {}), f) for f in schema} for h in hits]
|
|
170
|
+
except ValueError as e:
|
|
171
|
+
raise ValueError(f"[OpenSearch] chunk {chunk_no}: {e}") from e
|
|
172
|
+
last = hits[-1]['sort']
|
|
173
|
+
yield convert_rows(rows, schema, chunk_no, 'OpenSearch'), last
|
|
174
|
+
chunk_no += 1
|
|
175
|
+
if len(hits) < chunk_size:
|
|
176
|
+
return
|
|
177
|
+
finally:
|
|
178
|
+
self.conn.delete_pit(body={"pit_id": [pit_id]})
|
|
179
|
+
|
|
180
|
+
return ChunkResult(schema, chunks())
|
|
181
|
+
|
|
182
|
+
@staticmethod
|
|
183
|
+
def mapping_field(properties: dict, path: str) -> Optional[dict]:
|
|
184
|
+
"""점 경로(a.b)의 매핑 한 필드. 없으면 None"""
|
|
185
|
+
node = {'properties': properties}
|
|
186
|
+
walked = []
|
|
187
|
+
for part in path.split('.'):
|
|
188
|
+
if node.get('type') == 'nested': # nested 하위 값은 문서마다 배열 — 한 열 한 모양이 안 됨(D13)
|
|
189
|
+
raise ValueError(f"[OpenSearch] field '{path}' is inside nested field '{'.'.join(walked)}' — select the nested field itself (json_array)")
|
|
190
|
+
node = node.get('properties', {}).get(part)
|
|
191
|
+
if node is None:
|
|
192
|
+
return None
|
|
193
|
+
walked.append(part)
|
|
194
|
+
return node
|
|
195
|
+
|
|
196
|
+
@classmethod
|
|
197
|
+
def mapping_schema(cls, properties: dict, source_fields: List[str] = None) -> dict:
|
|
198
|
+
"""매핑 → {필드: 형식 이름}. 스키마에서 빼는 형식(alias 등)은 넣지 않는다"""
|
|
199
|
+
schema = {}
|
|
200
|
+
for name in (source_fields or list(properties)):
|
|
201
|
+
field = cls.mapping_field(properties, name)
|
|
202
|
+
if field is None:
|
|
203
|
+
raise ValueError(f"[OpenSearch] field '{name}' not found in mapping")
|
|
204
|
+
type_name = opensearch_type_name(name, field)
|
|
205
|
+
if type_name is not None:
|
|
206
|
+
schema[name] = type_name
|
|
207
|
+
return schema
|
|
208
|
+
|
|
209
|
+
@staticmethod
|
|
210
|
+
def source_value(source: dict, path: str):
|
|
211
|
+
"""_source 의 점 경로 값. 없으면 None. 경로가 배열을 지나면 값을 고를 수 없어 ValueError(값을 버리지 않는다)"""
|
|
212
|
+
if path in source: # 점이 든 키를 그대로 저장한 문서
|
|
213
|
+
return source[path]
|
|
214
|
+
value = source
|
|
215
|
+
for part in path.split('.'):
|
|
216
|
+
if value is None:
|
|
217
|
+
return None
|
|
218
|
+
if not isinstance(value, dict):
|
|
219
|
+
raise ValueError(f"field '{path}' crosses a {type(value).__name__} value — select its parent field")
|
|
220
|
+
value = value.get(part)
|
|
221
|
+
return value
|
|
222
|
+
|
|
124
223
|
def _clear_scroll_context(self):
|
|
125
224
|
"""내부 스크롤 컨텍스트를 정리합니다."""
|
|
126
225
|
if self._scroll_id:
|
|
@@ -168,7 +267,8 @@ class ElasticSearch:
|
|
|
168
267
|
"""
|
|
169
268
|
준비된 스크롤에서 다음 문서 청크를 가져옵니다.
|
|
170
269
|
첫 호출 시에는 초기 검색을, 이후 호출 시에는 스크롤 ID를 사용하여 데이터를 가져옵니다.
|
|
171
|
-
더 이상 문서가
|
|
270
|
+
더 이상 문서가 없으면 빈 리스트를 반환하고 컨텍스트를 정리합니다.
|
|
271
|
+
오류(scroll 문맥 만료 NotFoundError 포함)는 컨텍스트를 정리한 뒤 다시 던집니다 — 끝과 오류를 구분(2.4.0)
|
|
172
272
|
|
|
173
273
|
:return: 문서 _source (dict) 리스트. 더 이상 문서가 없으면 빈 리스트.
|
|
174
274
|
"""
|
|
@@ -208,11 +308,11 @@ class ElasticSearch:
|
|
|
208
308
|
except NotFoundError:
|
|
209
309
|
logger.warning(f"Scroll context {self._scroll_id} not found or expired. Clearing context.")
|
|
210
310
|
self._clear_scroll_context()
|
|
211
|
-
|
|
311
|
+
raise
|
|
212
312
|
except Exception as e:
|
|
213
313
|
logger.error(f"An error occurred during scroll operation: {e}", exc_info=True)
|
|
214
314
|
self._clear_scroll_context() # 에러 발생 시 컨텍스트 정리
|
|
215
|
-
|
|
315
|
+
raise
|
|
216
316
|
|
|
217
317
|
def _bulk_operations(self, actions: list, chunk_size: int = 500, raise_on_error: bool = True,
|
|
218
318
|
max_retries: int = 3, request_timeout: int = 120) -> tuple[int, list]:
|
|
@@ -362,7 +462,7 @@ class ElasticSearch:
|
|
|
362
462
|
"""
|
|
363
463
|
if index is None:
|
|
364
464
|
index = self.index_name
|
|
365
|
-
return self.conn.exists(index, id)
|
|
465
|
+
return self.conn.exists(index=index, id=id)
|
|
366
466
|
|
|
367
467
|
def search(self, body: dict = None, index=None) -> dict:
|
|
368
468
|
"""
|
|
@@ -568,7 +668,7 @@ class ElasticSearch:
|
|
|
568
668
|
"""
|
|
569
669
|
if index is None:
|
|
570
670
|
index = self.index_name
|
|
571
|
-
return self.conn.get_source(index, id)
|
|
671
|
+
return self.conn.get_source(index=index, id=id)
|
|
572
672
|
|
|
573
673
|
def index(self, index: str, body: dict, id: str or int = None) -> Any:
|
|
574
674
|
"""
|
|
@@ -581,7 +681,7 @@ class ElasticSearch:
|
|
|
581
681
|
Returns:
|
|
582
682
|
생성 결과
|
|
583
683
|
"""
|
|
584
|
-
return self.conn.index(index, body, id=id)
|
|
684
|
+
return self.conn.index(index=index, body=body, id=id)
|
|
585
685
|
|
|
586
686
|
def update(self, id: str or int, body: dict, index=None) -> Any:
|
|
587
687
|
"""
|
|
@@ -602,7 +702,7 @@ class ElasticSearch:
|
|
|
602
702
|
doc_body = {
|
|
603
703
|
'doc' : body
|
|
604
704
|
}
|
|
605
|
-
return self.conn.update(index, id, doc_body)
|
|
705
|
+
return self.conn.update(index=index, id=id, body=doc_body)
|
|
606
706
|
|
|
607
707
|
def delete(self, id: str or int, index=None) -> Any:
|
|
608
708
|
"""
|
|
@@ -615,7 +715,7 @@ class ElasticSearch:
|
|
|
615
715
|
"""
|
|
616
716
|
if index is None:
|
|
617
717
|
index = self.index_name
|
|
618
|
-
return self.conn.delete(index, id)
|
|
718
|
+
return self.conn.delete(index=index, id=id)
|
|
619
719
|
|
|
620
720
|
def delete_index(self, index):
|
|
621
721
|
"""
|
|
@@ -625,7 +725,7 @@ class ElasticSearch:
|
|
|
625
725
|
Returns:
|
|
626
726
|
result(str) : 처리 결과
|
|
627
727
|
"""
|
|
628
|
-
return self.conn.indices.delete(index)
|
|
728
|
+
return self.conn.indices.delete(index=index)
|
|
629
729
|
|
|
630
730
|
|
|
631
731
|
def bulk(self, actions:list, index=None) -> Tuple[int, int]:
|
|
@@ -12,6 +12,7 @@ from sqlalchemy.exc import SQLAlchemyError, DBAPIError
|
|
|
12
12
|
from echoss_common import dict_load, get_logger
|
|
13
13
|
from .env_config import resolve_credential
|
|
14
14
|
from .sql_transaction import SQLTransaction
|
|
15
|
+
from .chunk_output import ChunkResult, check_key, convert_rows, keyset_chunks, mysql_type_name, sql_select_list
|
|
15
16
|
|
|
16
17
|
logger = get_logger("echoss_db")
|
|
17
18
|
|
|
@@ -149,6 +150,7 @@ class MysqlQuery:
|
|
|
149
150
|
self.port = m.get('port', 3306)
|
|
150
151
|
self.db = m['db']
|
|
151
152
|
self.charset = m.get('charset', 'utf8mb4')
|
|
153
|
+
self.time_zone = m.get('time_zone') # DATETIME 이 적힌 서비스 현지 시간대(IANA 이름) — select_chunks (Design §3.2)
|
|
152
154
|
else:
|
|
153
155
|
logger.error(f'[MySQL] config info not exist or any required keys are missing {required_keys}')
|
|
154
156
|
raise ValueError("invalid conn_info")
|
|
@@ -447,7 +449,7 @@ class MysqlQuery:
|
|
|
447
449
|
@parse_query('SELECT')
|
|
448
450
|
def faster_select_generator(self, query_str: str, params=None, fetch_size=1000):
|
|
449
451
|
"""
|
|
450
|
-
대량 조회의 제너레이터 버전:
|
|
452
|
+
대량 조회의 제너레이터 버전: list[dict] 청크를 yield 합니다. 오류는 다시 던집니다(2.4.0).
|
|
451
453
|
- 메모리 피크를 최소화하고, 소비자 측에서 스트리밍 처리 가능
|
|
452
454
|
- 사용 예:
|
|
453
455
|
for chunk_list in mq.faster_select_generator("SELECT ... WHERE id>%s", (1000,), fetch_size=5000):
|
|
@@ -471,9 +473,65 @@ class MysqlQuery:
|
|
|
471
473
|
yield rows
|
|
472
474
|
|
|
473
475
|
except SQLAlchemyError as e:
|
|
474
|
-
logger.debug(f"[MySQL]
|
|
475
|
-
|
|
476
|
-
|
|
476
|
+
logger.debug(f"[MySQL] faster_select_generator Exception : {e}")
|
|
477
|
+
raise
|
|
478
|
+
|
|
479
|
+
def select_chunks(self, table: str, key: str, columns: Optional[list] = None, where: Optional[str] = None,
|
|
480
|
+
params=None, chunk_size: int = 10000, resume_token=None) -> ChunkResult:
|
|
481
|
+
"""대량 조회를 「정답 스키마 + 값 조각 + 재개 표지」로 낸다 (Design Ref: §4.2, keyset).
|
|
482
|
+
|
|
483
|
+
Args:
|
|
484
|
+
table: 테이블 이름, 'db.table' 도 된다(없으면 연결한 db)
|
|
485
|
+
key: 유일하고 정렬 가능한 열(정수·문자열). 조각마다 그 조각 마지막 행의 key 값이 재개 표지
|
|
486
|
+
columns: 낼 열 목록, 없으면 전체
|
|
487
|
+
where: 호출자가 쓴 SQL 조건 조각 — select() 의 query_str 과 같은 신뢰 수준. 값은 params 로 바인드
|
|
488
|
+
params: where 의 바인드 값(use_percent_param 에 맞는 형식)
|
|
489
|
+
resume_token: 이전 조각의 재개 표지. 그사이 바뀐 행은 반영된다
|
|
490
|
+
Returns:
|
|
491
|
+
ChunkResult — schema 는 information_schema 에서. DATETIME 은 연결 설정 time_zone 이 있으면
|
|
492
|
+
그 시간대로 보고 UTC 로(timestamp[us,UTC]), 없으면 timestamp[us]. json 은 JSON 텍스트(string)
|
|
493
|
+
"""
|
|
494
|
+
db_name, table_name = table.split('.', 1) if '.' in table else (self.db, table)
|
|
495
|
+
with self.engine.connect() as conn:
|
|
496
|
+
meta = conn.execute(text(
|
|
497
|
+
"SELECT COLUMN_NAME, DATA_TYPE, COLUMN_TYPE, NUMERIC_PRECISION, NUMERIC_SCALE "
|
|
498
|
+
"FROM information_schema.COLUMNS WHERE TABLE_SCHEMA = :s AND TABLE_NAME = :t ORDER BY ORDINAL_POSITION"),
|
|
499
|
+
{"s": db_name, "t": table_name}).mappings().all()
|
|
500
|
+
by_name = {m['COLUMN_NAME']: dict(m) for m in meta}
|
|
501
|
+
if not by_name:
|
|
502
|
+
raise ValueError(f"[MySQL] table '{table}' not found")
|
|
503
|
+
names = list(columns) if columns else list(by_name)
|
|
504
|
+
missing = [c for c in names + [key] if c not in by_name]
|
|
505
|
+
if missing:
|
|
506
|
+
raise ValueError(f"[MySQL] columns not found in '{table}': {missing}")
|
|
507
|
+
types = {c: mysql_type_name(by_name[c], self.time_zone) for c in set(names) | {key}}
|
|
508
|
+
check_key('MySQL', key, types)
|
|
509
|
+
schema = {c: types[c] for c in names}
|
|
510
|
+
selected = names if key in names else names + [key]
|
|
511
|
+
|
|
512
|
+
def quote(name):
|
|
513
|
+
return '`' + name.replace('`', '``') + '`'
|
|
514
|
+
|
|
515
|
+
base = f"SELECT {sql_select_list(selected, quote, set())} FROM {quote(db_name)}.{quote(table_name)} " \
|
|
516
|
+
f"WHERE ({where or 'TRUE'})"
|
|
517
|
+
|
|
518
|
+
def chunks():
|
|
519
|
+
with self.engine.connect() as conn:
|
|
520
|
+
def fetch_page(last, n):
|
|
521
|
+
if self.use_percent_param:
|
|
522
|
+
q = base + (f" AND {quote(key)} > %s" if last is not None else "") + f" ORDER BY {quote(key)} LIMIT %s"
|
|
523
|
+
p = tuple(params or ()) + ((last,) if last is not None else ()) + (n,)
|
|
524
|
+
else:
|
|
525
|
+
q = base + (f" AND {quote(key)} > :chunk_last" if last is not None else "") \
|
|
526
|
+
+ f" ORDER BY {quote(key)} LIMIT :chunk_limit"
|
|
527
|
+
p = dict(params or {}, chunk_limit=n, **({'chunk_last': last} if last is not None else {}))
|
|
528
|
+
return [dict(r) for r in self._execute_query(conn, q, p).mappings()]
|
|
529
|
+
|
|
530
|
+
yield from keyset_chunks(fetch_page, key, chunk_size, resume_token,
|
|
531
|
+
lambda rows, no: convert_rows(rows, schema, no, 'MySQL', self.time_zone),
|
|
532
|
+
drop_key=key not in names)
|
|
533
|
+
|
|
534
|
+
return ChunkResult(schema, chunks())
|
|
477
535
|
|
|
478
536
|
# ----------------------------------------------------------------------------------
|
|
479
537
|
# INSERT / UPDATE / DELETE
|