echoss-db 2.3.1__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. {echoss_db-2.3.1/echoss_db.egg-info → echoss_db-2.4.0}/PKG-INFO +31 -2
  2. {echoss_db-2.3.1 → echoss_db-2.4.0}/README.md +29 -0
  3. echoss_db-2.4.0/echoss_db/chunk_output.py +204 -0
  4. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/elastic_search.py +109 -9
  5. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mysql_query.py +62 -4
  6. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/postgres_query.py +59 -1
  7. {echoss_db-2.3.1 → echoss_db-2.4.0/echoss_db.egg-info}/PKG-INFO +31 -2
  8. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/SOURCES.txt +6 -0
  9. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/requires.txt +1 -1
  10. {echoss_db-2.3.1 → echoss_db-2.4.0}/pyproject.toml +2 -2
  11. {echoss_db-2.3.1 → echoss_db-2.4.0}/requirements.txt +1 -1
  12. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/conftest.py +28 -0
  13. echoss_db-2.4.0/tests/integration/test_chunks_fileformat_roundtrip.py +74 -0
  14. echoss_db-2.4.0/tests/integration/test_mysql_chunks.py +93 -0
  15. echoss_db-2.4.0/tests/integration/test_opensearch_chunks.py +105 -0
  16. echoss_db-2.4.0/tests/integration/test_postgres_chunks.py +111 -0
  17. echoss_db-2.4.0/tests/unit/test_chunk_output.py +240 -0
  18. {echoss_db-2.3.1 → echoss_db-2.4.0}/LICENSE +0 -0
  19. {echoss_db-2.3.1 → echoss_db-2.4.0}/MANIFEST.in +0 -0
  20. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/__init__.py +0 -0
  21. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/env_config.py +0 -0
  22. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/__init__.py +0 -0
  23. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/dataclass_mapper.py +0 -0
  24. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/errors.py +0 -0
  25. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/param_mapper.py +0 -0
  26. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/pydantic_mapper.py +0 -0
  27. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/row_mapper.py +0 -0
  28. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mapping/types.py +0 -0
  29. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/mongo_query.py +0 -0
  30. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/qdrant_vector.py +0 -0
  31. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db/sql_transaction.py +0 -0
  32. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/dependency_links.txt +0 -0
  33. {echoss_db-2.3.1 → echoss_db-2.4.0}/echoss_db.egg-info/top_level.txt +0 -0
  34. {echoss_db-2.3.1 → echoss_db-2.4.0}/package_tests/test_package_imports.py +0 -0
  35. {echoss_db-2.3.1 → echoss_db-2.4.0}/setup.cfg +0 -0
  36. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/__init__.py +0 -0
  37. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/__init__.py +0 -0
  38. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/test_env_config_integration.py +0 -0
  39. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/test_postgres_integration.py +0 -0
  40. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/integration/test_qdrant_integration.py +0 -0
  41. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/__init__.py +0 -0
  42. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/_postgres_test_schema.py +0 -0
  43. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_elasticsearch.ipynb +0 -0
  44. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_elasticsearch.py +0 -0
  45. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mapping_mysql.py +0 -0
  46. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mongo.ipynb +0 -0
  47. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mongo.py +0 -0
  48. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mysql.ipynb +0 -0
  49. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_mysql.py +0 -0
  50. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_postgres.py +0 -0
  51. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_postgres_qdrant_ingest.py +0 -0
  52. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/example_qdrant.py +0 -0
  53. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/manual/setup_ai_rag_test_schema.sql +0 -0
  54. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/__init__.py +0 -0
  55. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/mapping/__init__.py +0 -0
  56. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/mapping/test_param_mapper.py +0 -0
  57. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/test_env_config.py +0 -0
  58. {echoss_db-2.3.1 → echoss_db-2.4.0}/tests/unit/test_qdrant_role.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: echoss-db
3
- Version: 2.3.1
3
+ Version: 2.4.0
4
4
  Summary: echoss AI Bigdata Solution - Database Query Package
5
5
  Author-email: ckkim <ckkim@12cm.co.kr>
6
6
  License-Expression: Apache-2.0
@@ -16,7 +16,7 @@ License-File: LICENSE
16
16
  Requires-Dist: pandas>=3.0
17
17
  Requires-Dist: sqlalchemy>=2.0.0
18
18
  Requires-Dist: PyMySQL>=1.0.2
19
- Requires-Dist: opensearch-py<3.0.0,>=2.8.0
19
+ Requires-Dist: opensearch-py<4,>=3
20
20
  Requires-Dist: echoss-common<3,>=2.0.0
21
21
  Requires-Dist: psycopg[binary]<4.0.0,>=3.2
22
22
  Requires-Dist: qdrant-client[fastembed]<2.0.0,>=1.14.1
@@ -447,6 +447,34 @@ elastic.ping()
447
447
  elastic.info()
448
448
  ```
449
449
 
450
+ ### Chunk Output (2.4.0)
451
+
452
+ ---
453
+
454
+ Large results as fixed-schema chunks: `res.schema` comes from DB metadata (MySQL/Postgres `information_schema`, OpenSearch mapping) and can be read before iteration; iterating yields `(rows, resume_token)`. Errors are raised, never swallowed.
455
+
456
+ ```python
457
+ # MySQL / Postgres — keyset on a unique, sortable key column (integer or string)
458
+ res = mysql.select_chunks('orders', key='id', columns=['id', 'status', 'created'],
459
+ where='status = %s', params=('paid',), chunk_size=10000)
460
+ res.schema # e.g. {'id': 'int64', 'status': 'string', 'created': 'timestamp[us]'}
461
+ for rows, token in res:
462
+ ... # token = last key of the chunk; pass resume_token=token to continue
463
+
464
+ # OpenSearch — PIT + search_after on a unique sort field
465
+ res = elastic.search_chunks('shopby-log-v1-dev', sort_key='received_at', source_fields=['received_at', 'items'])
466
+
467
+ # write with echoss-fileformat (overrides / type errors are handled there)
468
+ from echoss_fileformat import FileUtil
469
+ FileUtil.dump_chunks(res, 'out_dir')
470
+ ```
471
+
472
+ - Values are in the declared type: OpenSearch `date` → `timestamp[ms,UTC]` (epoch-ms ints kept as is, sub-ms digits truncated), `nested`/`knn_vector` → `json_array`, `object` → `json_object`; SQL `json`/`jsonb` → JSON text (`string`). `date_nanos` is not supported (company policy: ms/us timestamps only).
473
+ - One shape per column: values that do not match the schema (e.g. a list in a `keyword` field) are passed through and rejected by echoss-fileformat (`ChunkTypeError`); use its `overrides` to read them differently.
474
+ - MySQL `DATETIME`: set `time_zone` (IANA name, e.g. `Asia/Manila`) in the `mysql` config section to get `timestamp[us,UTC]`; without it `timestamp[us]`.
475
+ - Resuming re-reads from the token: rows changed in the meantime are reflected. `key`/`sort_key` must be unique, or rows may be skipped or repeated.
476
+ - `where` is a caller-written SQL fragment (same trust level as `select()`); bind values via `params`.
477
+
450
478
  ### Qdrant
451
479
 
452
480
  ---
@@ -634,3 +662,4 @@ v2.2.0 qdrant generalized dict filter (breaking behavior: previously ignored key
634
662
  v2.2.2 qdrant admin/read-only API key support (`role` param, `api_key`/`read_only_api_key` config fields), config.yaml credential fields (mysql/postgres/elastic `passwd`, qdrant `api_key`/`read_only_api_key`) now accept `${VAR_NAME}` placeholders resolved from `.env`/OS env (new `python-dotenv` dependency)
635
663
  v2.3.0 config loading and logger now imported from `echoss-common` (`>=1.2.0,<3`) instead of `echoss-fileformat`; `echoss-fileformat` removed from install dependencies (declare it directly if you use it); raise pandas floor to 3.0. XML config files are now read by `echoss-common` (result keys may differ from the previous `echoss-fileformat` XML reader)
636
664
  v2.3.1 raise `echoss-common` floor to `>=2.0.0,<3` (no code change). With `echoss-common` 2.0: no default `logs/echoss.log` file (console only), and a missing config file path raises `FileNotFoundError` (with `echoss-common` 1.x: `ValueError`/`TypeError`)
665
+ v2.4.0 chunk output: `MysqlQuery.select_chunks`, `PostgresQuery.select_chunks`, `ElasticSearch.search_chunks` (schema from DB metadata + chunks + resume token, OpenSearch via PIT); `faster_select_generator` and `next_scroll_chunk` now raise errors instead of ending silently (a scroll context that expired raises `NotFoundError` instead of returning `[]`); `opensearch-py>=3,<4`; MySQL config `time_zone`
@@ -415,6 +415,34 @@ elastic.ping()
415
415
  elastic.info()
416
416
  ```
417
417
 
418
+ ### Chunk Output (2.4.0)
419
+
420
+ ---
421
+
422
+ Large results as fixed-schema chunks: `res.schema` comes from DB metadata (MySQL/Postgres `information_schema`, OpenSearch mapping) and can be read before iteration; iterating yields `(rows, resume_token)`. Errors are raised, never swallowed.
423
+
424
+ ```python
425
+ # MySQL / Postgres — keyset on a unique, sortable key column (integer or string)
426
+ res = mysql.select_chunks('orders', key='id', columns=['id', 'status', 'created'],
427
+ where='status = %s', params=('paid',), chunk_size=10000)
428
+ res.schema # e.g. {'id': 'int64', 'status': 'string', 'created': 'timestamp[us]'}
429
+ for rows, token in res:
430
+ ... # token = last key of the chunk; pass resume_token=token to continue
431
+
432
+ # OpenSearch — PIT + search_after on a unique sort field
433
+ res = elastic.search_chunks('shopby-log-v1-dev', sort_key='received_at', source_fields=['received_at', 'items'])
434
+
435
+ # write with echoss-fileformat (overrides / type errors are handled there)
436
+ from echoss_fileformat import FileUtil
437
+ FileUtil.dump_chunks(res, 'out_dir')
438
+ ```
439
+
440
+ - Values are in the declared type: OpenSearch `date` → `timestamp[ms,UTC]` (epoch-ms ints kept as is, sub-ms digits truncated), `nested`/`knn_vector` → `json_array`, `object` → `json_object`; SQL `json`/`jsonb` → JSON text (`string`). `date_nanos` is not supported (company policy: ms/us timestamps only).
441
+ - One shape per column: values that do not match the schema (e.g. a list in a `keyword` field) are passed through and rejected by echoss-fileformat (`ChunkTypeError`); use its `overrides` to read them differently.
442
+ - MySQL `DATETIME`: set `time_zone` (IANA name, e.g. `Asia/Manila`) in the `mysql` config section to get `timestamp[us,UTC]`; without it `timestamp[us]`.
443
+ - Resuming re-reads from the token: rows changed in the meantime are reflected. `key`/`sort_key` must be unique, or rows may be skipped or repeated.
444
+ - `where` is a caller-written SQL fragment (same trust level as `select()`); bind values via `params`.
445
+
418
446
  ### Qdrant
419
447
 
420
448
  ---
@@ -602,3 +630,4 @@ v2.2.0 qdrant generalized dict filter (breaking behavior: previously ignored key
602
630
  v2.2.2 qdrant admin/read-only API key support (`role` param, `api_key`/`read_only_api_key` config fields), config.yaml credential fields (mysql/postgres/elastic `passwd`, qdrant `api_key`/`read_only_api_key`) now accept `${VAR_NAME}` placeholders resolved from `.env`/OS env (new `python-dotenv` dependency)
603
631
  v2.3.0 config loading and logger now imported from `echoss-common` (`>=1.2.0,<3`) instead of `echoss-fileformat`; `echoss-fileformat` removed from install dependencies (declare it directly if you use it); raise pandas floor to 3.0. XML config files are now read by `echoss-common` (result keys may differ from the previous `echoss-fileformat` XML reader)
604
632
  v2.3.1 raise `echoss-common` floor to `>=2.0.0,<3` (no code change). With `echoss-common` 2.0: no default `logs/echoss.log` file (console only), and a missing config file path raises `FileNotFoundError` (with `echoss-common` 1.x: `ValueError`/`TypeError`)
633
+ v2.4.0 chunk output: `MysqlQuery.select_chunks`, `PostgresQuery.select_chunks`, `ElasticSearch.search_chunks` (schema from DB metadata + chunks + resume token, OpenSearch via PIT); `faster_select_generator` and `next_scroll_chunk` now raise errors instead of ending silently (a scroll context that expired raises `NotFoundError` instead of returning `[]`); `opensearch-py>=3,<4`; MySQL config `time_zone`
@@ -0,0 +1,204 @@
1
+ """
2
+ 조각 출력 — 정답 스키마 + 그 형식의 값 조각 + 재개 표지
3
+
4
+ Design Ref: docs/02-design/features/db-raw-result-chunk-output.design.md §3·§4.1
5
+ - 스키마는 DB 메타데이터(OpenSearch 매핑, MySQL·Postgres information_schema)로 만든다
6
+ - 값은 DB 가 선언한 형식의 값: 드라이버 표현이 선언 형식과 다를 때만 맞추고 값의 뜻은 바꾸지 않는다
7
+ - 값 형식 검사·덮어쓰기는 echoss-fileformat(ChunkTypeError·overrides) 몫 — 여기서는 검사하지 않는다(§3.4)
8
+ - 형식 이름은 echoss-fileformat chunk_schema.py 의 이름(문자열). pyarrow·echoss-fileformat 은 import 하지 않는다
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import base64
13
+ import datetime
14
+ import uuid
15
+ from typing import Iterable, Optional, Tuple
16
+ from zoneinfo import ZoneInfo
17
+
18
+
19
+ class ChunkResult:
20
+ """새 조각 함수의 결과. schema 는 반복 전에 읽을 수 있고, 반복하면 (rows, resume_token) 을 낸다"""
21
+
22
+ def __init__(self, schema: dict, chunks: Iterable[Tuple[list, object]]):
23
+ self.schema = schema
24
+ self.chunks = chunks
25
+
26
+ def __iter__(self):
27
+ return iter(self.chunks)
28
+
29
+
30
+ # §3.1 OpenSearch — 공식 필드 형식 목록 전체 (docs.opensearch.org supported-field-types, 2026-10-10 조회)
31
+ OPENSEARCH_TYPES = {
32
+ 'boolean': 'bool', 'binary': 'binary',
33
+ 'text': 'string', 'match_only_text': 'string', 'keyword': 'string', 'constant_keyword': 'string',
34
+ 'icu_collation_keyword': 'string', 'wildcard': 'string', 'version': 'string',
35
+ 'search_as_you_type': 'string', 'semantic': 'string', 'ip': 'string',
36
+ 'token_count': 'int32', 'byte': 'int8', 'short': 'int16', 'integer': 'int32', 'long': 'int64',
37
+ 'unsigned_long': 'uint64', 'half_float': 'float32', 'float': 'float32', 'double': 'float64',
38
+ 'scaled_float': 'float64', 'rank_feature': 'float32',
39
+ 'date': 'timestamp[ms,UTC]',
40
+ 'geo_point': 'json_object', 'geo_shape': 'json_object', 'xy_point': 'json_object', 'xy_shape': 'json_object',
41
+ 'integer_range': 'json_object', 'long_range': 'json_object', 'double_range': 'json_object',
42
+ 'float_range': 'json_object', 'ip_range': 'json_object', 'date_range': 'json_object',
43
+ 'object': 'json_object', 'flat_object': 'json_object', 'join': 'json_object',
44
+ 'completion': 'json_object', 'percolator': 'json_object', 'rank_features': 'json_object',
45
+ 'sparse_vector': 'json_object',
46
+ 'nested': 'json_array', 'knn_vector': 'json_array',
47
+ }
48
+ OPENSEARCH_EXCLUDED = {'alias', 'star_tree', 'derived'} # _source 에 없거나 문서 필드가 아님
49
+
50
+
51
+ def opensearch_type_name(name: str, mapping_field: dict) -> Optional[str]:
52
+ """매핑 한 필드 → 형식 이름. None 이면 스키마에서 뺀다. 옮길 수 없는 형식은 ValueError"""
53
+ type_ = mapping_field.get('type', 'object' if 'properties' in mapping_field else None)
54
+ if type_ in OPENSEARCH_EXCLUDED:
55
+ return None
56
+ if type_ == 'date_nanos':
57
+ raise ValueError(f"[OpenSearch] field '{name}': date_nanos is not supported (company policy: ms/us timestamps only)")
58
+ if type_ not in OPENSEARCH_TYPES:
59
+ raise ValueError(f"[OpenSearch] field '{name}': unsupported mapping type {type_!r}")
60
+ return OPENSEARCH_TYPES[type_]
61
+
62
+
63
+ # §3.2 MySQL
64
+ MYSQL_INTS = {'tinyint': 'int8', 'smallint': 'int16', 'mediumint': 'int32', 'int': 'int32', 'bigint': 'int64'}
65
+ MYSQL_STRINGS = {'char', 'varchar', 'tinytext', 'text', 'mediumtext', 'longtext', 'enum', 'set', 'json'}
66
+ MYSQL_BINARIES = {'bit', 'binary', 'varbinary', 'tinyblob', 'blob', 'mediumblob', 'longblob', 'geometry', 'point',
67
+ 'linestring', 'polygon', 'multipoint', 'multilinestring', 'multipolygon', 'geometrycollection'}
68
+ MYSQL_OTHERS = {'float': 'float32', 'double': 'float64', 'date': 'date32', 'timestamp': 'timestamp[us]',
69
+ 'time': 'duration[us]', 'year': 'int16'}
70
+
71
+
72
+ def mysql_type_name(column: dict, time_zone: Optional[str] = None) -> str:
73
+ """information_schema.columns 한 행(DATA_TYPE·COLUMN_TYPE·NUMERIC_PRECISION·NUMERIC_SCALE·COLUMN_NAME) → 형식 이름"""
74
+ name = column['COLUMN_NAME']
75
+ data_type = column['DATA_TYPE'].lower()
76
+ if data_type in MYSQL_INTS:
77
+ type_ = MYSQL_INTS[data_type]
78
+ return 'u' + type_ if 'unsigned' in column['COLUMN_TYPE'].lower() else type_
79
+ if data_type == 'decimal':
80
+ return decimal_type_name('MySQL', name, column['NUMERIC_PRECISION'], column['NUMERIC_SCALE'])
81
+ if data_type in MYSQL_STRINGS:
82
+ return 'string' # json 은 JSON 텍스트 그대로(D13)
83
+ if data_type in MYSQL_BINARIES:
84
+ return 'binary'
85
+ if data_type == 'datetime':
86
+ return 'timestamp[us,UTC]' if time_zone else 'timestamp[us]'
87
+ if data_type in MYSQL_OTHERS:
88
+ return MYSQL_OTHERS[data_type]
89
+ raise ValueError(f"[MySQL] column '{name}': unsupported data type {data_type!r}")
90
+
91
+
92
+ # §3.3 Postgres
93
+ POSTGRES_TYPES = {
94
+ 'smallint': 'int16', 'integer': 'int32', 'bigint': 'int64', 'real': 'float32', 'double precision': 'float64',
95
+ 'boolean': 'bool', 'character': 'string', 'character varying': 'string', 'text': 'string', 'uuid': 'string',
96
+ 'json': 'string', 'jsonb': 'string', 'bytea': 'binary', 'ARRAY': 'json_array', 'date': 'date32',
97
+ 'timestamp without time zone': 'timestamp[us]', 'timestamp with time zone': 'timestamp[us,UTC]',
98
+ 'time without time zone': 'duration[us]',
99
+ }
100
+
101
+
102
+ def postgres_type_name(column: dict) -> str:
103
+ """information_schema.columns 한 행(data_type·numeric_precision·numeric_scale·column_name) → 형식 이름"""
104
+ name = column['column_name']
105
+ data_type = column['data_type']
106
+ if data_type == 'numeric':
107
+ if column['numeric_precision'] is None:
108
+ raise ValueError(f"[Postgres] column '{name}': numeric without precision has no fixed type (cast it in a view)")
109
+ return decimal_type_name('Postgres', name, column['numeric_precision'], column['numeric_scale'])
110
+ if data_type in POSTGRES_TYPES:
111
+ return POSTGRES_TYPES[data_type]
112
+ raise ValueError(f"[Postgres] column '{name}': unsupported data type {data_type!r}")
113
+
114
+
115
+ def decimal_type_name(db: str, name: str, precision, scale) -> str:
116
+ if int(precision) > 38:
117
+ raise ValueError(f"[{db}] column '{name}': decimal precision {precision} > 38")
118
+ return f"decimal({int(precision)},{int(scale or 0)})"
119
+
120
+
121
+ # §3.4 값 맞추기 — 검사하지 않는다
122
+ def convert_rows(rows: list, schema: dict, chunk_no: int, db: str, time_zone: Optional[str] = None) -> list:
123
+ """드라이버 표현을 선언 형식으로 맞춘 행. 변환 실패(날짜 해석, MySQL 0 날짜)만 ValueError"""
124
+ zone = ZoneInfo(time_zone) if time_zone else None
125
+ for row in rows:
126
+ for column, type_name in schema.items():
127
+ value = row.get(column)
128
+ if value is None:
129
+ continue
130
+ try:
131
+ row[column] = convert_value(value, type_name, db, zone)
132
+ except ValueError as e:
133
+ raise ValueError(f"[{db}] column '{column}' chunk {chunk_no}: {e}") from e
134
+ return rows
135
+
136
+
137
+ def convert_value(value, type_name: str, db: str, zone):
138
+ if db == 'MySQL' and isinstance(value, str) and value.startswith('0000-00-00') \
139
+ and type_name in ('date32', 'timestamp[us]', 'timestamp[us,UTC]'):
140
+ raise ValueError(f"zero date {value!r}")
141
+ if type_name == 'timestamp[ms,UTC]': # OpenSearch date (D6)
142
+ if isinstance(value, str):
143
+ return opensearch_date(value)
144
+ return value # epoch 밀리초 정수는 그대로
145
+ if type_name == 'timestamp[us,UTC]' and isinstance(value, datetime.datetime):
146
+ if value.tzinfo is None: # MySQL DATETIME: 서비스 현지 시각 + 연결 설정 시간대 (D7)
147
+ value = value.replace(tzinfo=zone)
148
+ return value.astimezone(datetime.timezone.utc)
149
+ if type_name == 'duration[us]' and isinstance(value, datetime.time): # Postgres time: 자정부터의 경과(§3.3)
150
+ return datetime.timedelta(hours=value.hour, minutes=value.minute, seconds=value.second,
151
+ microseconds=value.microsecond)
152
+ if type_name == 'binary' and isinstance(value, str) and db == 'OpenSearch':
153
+ return base64.b64decode(value)
154
+ if type_name == 'string' and isinstance(value, uuid.UUID):
155
+ return str(value)
156
+ return value
157
+
158
+
159
+ def opensearch_date(value: str) -> datetime.datetime:
160
+ """ISO 문자열 → UTC, 밀리초 아래는 자른다(OpenSearch 색인값과 같음). 날짜만 적힌 값은 그날 0시 UTC.
161
+ 시간대 표기가 없으면 UTC 로 본다(OpenSearch date 해석 — 공식 문서 기준 지식)"""
162
+ try:
163
+ parsed = datetime.datetime.fromisoformat(value)
164
+ except ValueError as e:
165
+ raise ValueError(f"cannot parse date {value!r}") from e
166
+ if parsed.tzinfo is None:
167
+ parsed = parsed.replace(tzinfo=datetime.timezone.utc)
168
+ parsed = parsed.astimezone(datetime.timezone.utc)
169
+ return parsed.replace(microsecond=parsed.microsecond // 1000 * 1000)
170
+
171
+
172
+ # §4.2 SQL keyset — MySQL·Postgres 공통 반복
173
+ KEY_TYPES = {'int8', 'int16', 'int32', 'int64', 'uint8', 'uint16', 'uint32', 'uint64', 'string'}
174
+
175
+
176
+ def keyset_chunks(fetch_page, key: str, chunk_size: int, resume_token, convert, drop_key: bool):
177
+ """fetch_page(last, n) → 행 목록(key 오름차순, key > last). 조각마다 (rows, 마지막 key 값) 을 낸다.
178
+ 오류는 잡지 않는다 — 끝과 오류를 구분하게 다시 던진다(계약 7)"""
179
+ last = resume_token
180
+ chunk_no = 0
181
+ while True:
182
+ rows = fetch_page(last, chunk_size)
183
+ if not rows:
184
+ return
185
+ last = rows[-1][key]
186
+ if isinstance(last, uuid.UUID): # 재개 표지도 선언 형식(string)으로 — echoss-fileformat 이 JSON 으로 기록(Check 게이트 R2)
187
+ last = str(last)
188
+ if drop_key:
189
+ for row in rows:
190
+ row.pop(key)
191
+ yield convert(rows, chunk_no), last
192
+ chunk_no += 1
193
+ if len(rows) < chunk_size:
194
+ return
195
+
196
+
197
+ def sql_select_list(columns: list, quote, text_columns: set) -> str:
198
+ """식별자를 인용한 SELECT 목록. text_columns 는 ::text 로 읽는다(Postgres json·jsonb, D13)"""
199
+ return ", ".join(f"{quote(c)}::text AS {quote(c)}" if c in text_columns else quote(c) for c in columns)
200
+
201
+
202
+ def check_key(db: str, key: str, schema_types: dict):
203
+ if schema_types[key] not in KEY_TYPES:
204
+ raise ValueError(f"[{db}] key '{key}' must be an integer or string column, got {schema_types[key]}")
@@ -7,6 +7,7 @@ from typing import Any, List, Tuple, Dict, Optional, Union
7
7
  from echoss_common import dict_load, get_logger, set_logger_level
8
8
 
9
9
  from .env_config import resolve_credential
10
+ from .chunk_output import ChunkResult, convert_rows, opensearch_type_name
10
11
 
11
12
  logger = get_logger("echoss_query")
12
13
 
@@ -121,6 +122,104 @@ class ElasticSearch:
121
122
  except Exception as e:
122
123
  raise ValueError("Connection failed by config. Please check config data")
123
124
 
125
+ def search_chunks(self, index: str, sort_key: str, query: dict = None, source_fields: List[str] = None,
126
+ chunk_size: int = 10000, keep_alive: str = '5m', resume_token: list = None) -> ChunkResult:
127
+ """대량 조회를 「정답 스키마 + 값 조각 + 재개 표지」로 낸다 (Design Ref: §3.1·§4.2, PIT + search_after).
128
+
129
+ Args:
130
+ index: 인덱스 이름(별칭·패턴이면 묶인 인덱스들의 스키마가 같아야 한다)
131
+ sort_key: 정렬·재개에 쓸 유일한 필드. 유일하지 않으면 재개 때 문서가 빠지거나 겹칠 수 있다
132
+ query: 질의 본문의 query 부분(기본 match_all)
133
+ source_fields: 낼 필드 목록(점 경로 가능), 없으면 매핑 최상위 필드 전체
134
+ keep_alive: PIT 유지 시간
135
+ resume_token: 이전 조각의 재개 표지(search_after 정렬값). 새 PIT 를 열므로 그사이 바뀐 자료는 반영된다
136
+ Returns:
137
+ ChunkResult — schema 는 매핑에서, 값은 _source 에서(색인값 docvalue_fields 는 쓰지 않음)
138
+ """
139
+ try:
140
+ mappings = self.conn.indices.get_mapping(index=index)
141
+ except NotFoundError as e:
142
+ raise ValueError(f"[OpenSearch] index '{index}' not found") from e
143
+ schemas = [self.mapping_schema(m['mappings'].get('properties', {}), source_fields) for m in mappings.values()]
144
+ if not schemas:
145
+ raise ValueError(f"[OpenSearch] index '{index}' not found")
146
+ if any(s != schemas[0] for s in schemas[1:]):
147
+ raise ValueError(f"[OpenSearch] indices behind '{index}' have different mappings for the requested fields")
148
+ schema = schemas[0]
149
+ for m in mappings.values(): # 다중 필드(name.keyword)는 정렬 키로 받지 않는다(설계 §4.2)
150
+ if self.mapping_field(m['mappings'].get('properties', {}), sort_key) is None:
151
+ raise ValueError(f"[OpenSearch] sort_key '{sort_key}' not found in mapping of '{index}' (multi-fields are not supported)")
152
+
153
+ def chunks():
154
+ pit_id = self.conn.create_pit(index=index, keep_alive=keep_alive)['pit_id']
155
+ try:
156
+ last = resume_token
157
+ chunk_no = 0
158
+ while True:
159
+ body = {"size": chunk_size, "query": query or {"match_all": {}}, "_source": list(schema),
160
+ "sort": [{sort_key: "asc"}], "pit": {"id": pit_id, "keep_alive": keep_alive}}
161
+ if last is not None:
162
+ body["search_after"] = last
163
+ response = self.conn.search(body=body)
164
+ pit_id = response.get('pit_id', pit_id)
165
+ hits = response['hits']['hits']
166
+ if not hits:
167
+ return
168
+ try:
169
+ rows = [{f: self.source_value(h.get('_source', {}), f) for f in schema} for h in hits]
170
+ except ValueError as e:
171
+ raise ValueError(f"[OpenSearch] chunk {chunk_no}: {e}") from e
172
+ last = hits[-1]['sort']
173
+ yield convert_rows(rows, schema, chunk_no, 'OpenSearch'), last
174
+ chunk_no += 1
175
+ if len(hits) < chunk_size:
176
+ return
177
+ finally:
178
+ self.conn.delete_pit(body={"pit_id": [pit_id]})
179
+
180
+ return ChunkResult(schema, chunks())
181
+
182
+ @staticmethod
183
+ def mapping_field(properties: dict, path: str) -> Optional[dict]:
184
+ """점 경로(a.b)의 매핑 한 필드. 없으면 None"""
185
+ node = {'properties': properties}
186
+ walked = []
187
+ for part in path.split('.'):
188
+ if node.get('type') == 'nested': # nested 하위 값은 문서마다 배열 — 한 열 한 모양이 안 됨(D13)
189
+ raise ValueError(f"[OpenSearch] field '{path}' is inside nested field '{'.'.join(walked)}' — select the nested field itself (json_array)")
190
+ node = node.get('properties', {}).get(part)
191
+ if node is None:
192
+ return None
193
+ walked.append(part)
194
+ return node
195
+
196
+ @classmethod
197
+ def mapping_schema(cls, properties: dict, source_fields: List[str] = None) -> dict:
198
+ """매핑 → {필드: 형식 이름}. 스키마에서 빼는 형식(alias 등)은 넣지 않는다"""
199
+ schema = {}
200
+ for name in (source_fields or list(properties)):
201
+ field = cls.mapping_field(properties, name)
202
+ if field is None:
203
+ raise ValueError(f"[OpenSearch] field '{name}' not found in mapping")
204
+ type_name = opensearch_type_name(name, field)
205
+ if type_name is not None:
206
+ schema[name] = type_name
207
+ return schema
208
+
209
+ @staticmethod
210
+ def source_value(source: dict, path: str):
211
+ """_source 의 점 경로 값. 없으면 None. 경로가 배열을 지나면 값을 고를 수 없어 ValueError(값을 버리지 않는다)"""
212
+ if path in source: # 점이 든 키를 그대로 저장한 문서
213
+ return source[path]
214
+ value = source
215
+ for part in path.split('.'):
216
+ if value is None:
217
+ return None
218
+ if not isinstance(value, dict):
219
+ raise ValueError(f"field '{path}' crosses a {type(value).__name__} value — select its parent field")
220
+ value = value.get(part)
221
+ return value
222
+
124
223
  def _clear_scroll_context(self):
125
224
  """내부 스크롤 컨텍스트를 정리합니다."""
126
225
  if self._scroll_id:
@@ -168,7 +267,8 @@ class ElasticSearch:
168
267
  """
169
268
  준비된 스크롤에서 다음 문서 청크를 가져옵니다.
170
269
  첫 호출 시에는 초기 검색을, 이후 호출 시에는 스크롤 ID를 사용하여 데이터를 가져옵니다.
171
- 더 이상 문서가 없거나 에러 발생 시 빈 리스트를 반환하고 컨텍스트를 정리합니다.
270
+ 더 이상 문서가 없으면 빈 리스트를 반환하고 컨텍스트를 정리합니다.
271
+ 오류(scroll 문맥 만료 NotFoundError 포함)는 컨텍스트를 정리한 뒤 다시 던집니다 — 끝과 오류를 구분(2.4.0)
172
272
 
173
273
  :return: 문서 _source (dict) 리스트. 더 이상 문서가 없으면 빈 리스트.
174
274
  """
@@ -208,11 +308,11 @@ class ElasticSearch:
208
308
  except NotFoundError:
209
309
  logger.warning(f"Scroll context {self._scroll_id} not found or expired. Clearing context.")
210
310
  self._clear_scroll_context()
211
- return []
311
+ raise
212
312
  except Exception as e:
213
313
  logger.error(f"An error occurred during scroll operation: {e}", exc_info=True)
214
314
  self._clear_scroll_context() # 에러 발생 시 컨텍스트 정리
215
- return []
315
+ raise
216
316
 
217
317
  def _bulk_operations(self, actions: list, chunk_size: int = 500, raise_on_error: bool = True,
218
318
  max_retries: int = 3, request_timeout: int = 120) -> tuple[int, list]:
@@ -362,7 +462,7 @@ class ElasticSearch:
362
462
  """
363
463
  if index is None:
364
464
  index = self.index_name
365
- return self.conn.exists(index, id)
465
+ return self.conn.exists(index=index, id=id)
366
466
 
367
467
  def search(self, body: dict = None, index=None) -> dict:
368
468
  """
@@ -568,7 +668,7 @@ class ElasticSearch:
568
668
  """
569
669
  if index is None:
570
670
  index = self.index_name
571
- return self.conn.get_source(index, id)
671
+ return self.conn.get_source(index=index, id=id)
572
672
 
573
673
  def index(self, index: str, body: dict, id: str or int = None) -> Any:
574
674
  """
@@ -581,7 +681,7 @@ class ElasticSearch:
581
681
  Returns:
582
682
  생성 결과
583
683
  """
584
- return self.conn.index(index, body, id=id)
684
+ return self.conn.index(index=index, body=body, id=id)
585
685
 
586
686
  def update(self, id: str or int, body: dict, index=None) -> Any:
587
687
  """
@@ -602,7 +702,7 @@ class ElasticSearch:
602
702
  doc_body = {
603
703
  'doc' : body
604
704
  }
605
- return self.conn.update(index, id, doc_body)
705
+ return self.conn.update(index=index, id=id, body=doc_body)
606
706
 
607
707
  def delete(self, id: str or int, index=None) -> Any:
608
708
  """
@@ -615,7 +715,7 @@ class ElasticSearch:
615
715
  """
616
716
  if index is None:
617
717
  index = self.index_name
618
- return self.conn.delete(index, id)
718
+ return self.conn.delete(index=index, id=id)
619
719
 
620
720
  def delete_index(self, index):
621
721
  """
@@ -625,7 +725,7 @@ class ElasticSearch:
625
725
  Returns:
626
726
  result(str) : 처리 결과
627
727
  """
628
- return self.conn.indices.delete(index)
728
+ return self.conn.indices.delete(index=index)
629
729
 
630
730
 
631
731
  def bulk(self, actions:list, index=None) -> Tuple[int, int]:
@@ -12,6 +12,7 @@ from sqlalchemy.exc import SQLAlchemyError, DBAPIError
12
12
  from echoss_common import dict_load, get_logger
13
13
  from .env_config import resolve_credential
14
14
  from .sql_transaction import SQLTransaction
15
+ from .chunk_output import ChunkResult, check_key, convert_rows, keyset_chunks, mysql_type_name, sql_select_list
15
16
 
16
17
  logger = get_logger("echoss_db")
17
18
 
@@ -149,6 +150,7 @@ class MysqlQuery:
149
150
  self.port = m.get('port', 3306)
150
151
  self.db = m['db']
151
152
  self.charset = m.get('charset', 'utf8mb4')
153
+ self.time_zone = m.get('time_zone') # DATETIME 이 적힌 서비스 현지 시간대(IANA 이름) — select_chunks (Design §3.2)
152
154
  else:
153
155
  logger.error(f'[MySQL] config info not exist or any required keys are missing {required_keys}')
154
156
  raise ValueError("invalid conn_info")
@@ -447,7 +449,7 @@ class MysqlQuery:
447
449
  @parse_query('SELECT')
448
450
  def faster_select_generator(self, query_str: str, params=None, fetch_size=1000):
449
451
  """
450
- 대량 조회의 제너레이터 버전: DataFrame 청크를 yield 합니다.
452
+ 대량 조회의 제너레이터 버전: list[dict] 청크를 yield 합니다. 오류는 다시 던집니다(2.4.0).
451
453
  - 메모리 피크를 최소화하고, 소비자 측에서 스트리밍 처리 가능
452
454
  - 사용 예:
453
455
  for chunk_list in mq.faster_select_generator("SELECT ... WHERE id>%s", (1000,), fetch_size=5000):
@@ -471,9 +473,65 @@ class MysqlQuery:
471
473
  yield rows
472
474
 
473
475
  except SQLAlchemyError as e:
474
- logger.debug(f"[MySQL] faster_select_generotor Exception : {e}")
475
- # 제너레이터 내부 예외는 호출측에서 캐치 가능하도록 재전파하지 않음
476
- return
476
+ logger.debug(f"[MySQL] faster_select_generator Exception : {e}")
477
+ raise
478
+
479
+ def select_chunks(self, table: str, key: str, columns: Optional[list] = None, where: Optional[str] = None,
480
+ params=None, chunk_size: int = 10000, resume_token=None) -> ChunkResult:
481
+ """대량 조회를 「정답 스키마 + 값 조각 + 재개 표지」로 낸다 (Design Ref: §4.2, keyset).
482
+
483
+ Args:
484
+ table: 테이블 이름, 'db.table' 도 된다(없으면 연결한 db)
485
+ key: 유일하고 정렬 가능한 열(정수·문자열). 조각마다 그 조각 마지막 행의 key 값이 재개 표지
486
+ columns: 낼 열 목록, 없으면 전체
487
+ where: 호출자가 쓴 SQL 조건 조각 — select() 의 query_str 과 같은 신뢰 수준. 값은 params 로 바인드
488
+ params: where 의 바인드 값(use_percent_param 에 맞는 형식)
489
+ resume_token: 이전 조각의 재개 표지. 그사이 바뀐 행은 반영된다
490
+ Returns:
491
+ ChunkResult — schema 는 information_schema 에서. DATETIME 은 연결 설정 time_zone 이 있으면
492
+ 그 시간대로 보고 UTC 로(timestamp[us,UTC]), 없으면 timestamp[us]. json 은 JSON 텍스트(string)
493
+ """
494
+ db_name, table_name = table.split('.', 1) if '.' in table else (self.db, table)
495
+ with self.engine.connect() as conn:
496
+ meta = conn.execute(text(
497
+ "SELECT COLUMN_NAME, DATA_TYPE, COLUMN_TYPE, NUMERIC_PRECISION, NUMERIC_SCALE "
498
+ "FROM information_schema.COLUMNS WHERE TABLE_SCHEMA = :s AND TABLE_NAME = :t ORDER BY ORDINAL_POSITION"),
499
+ {"s": db_name, "t": table_name}).mappings().all()
500
+ by_name = {m['COLUMN_NAME']: dict(m) for m in meta}
501
+ if not by_name:
502
+ raise ValueError(f"[MySQL] table '{table}' not found")
503
+ names = list(columns) if columns else list(by_name)
504
+ missing = [c for c in names + [key] if c not in by_name]
505
+ if missing:
506
+ raise ValueError(f"[MySQL] columns not found in '{table}': {missing}")
507
+ types = {c: mysql_type_name(by_name[c], self.time_zone) for c in set(names) | {key}}
508
+ check_key('MySQL', key, types)
509
+ schema = {c: types[c] for c in names}
510
+ selected = names if key in names else names + [key]
511
+
512
+ def quote(name):
513
+ return '`' + name.replace('`', '``') + '`'
514
+
515
+ base = f"SELECT {sql_select_list(selected, quote, set())} FROM {quote(db_name)}.{quote(table_name)} " \
516
+ f"WHERE ({where or 'TRUE'})"
517
+
518
+ def chunks():
519
+ with self.engine.connect() as conn:
520
+ def fetch_page(last, n):
521
+ if self.use_percent_param:
522
+ q = base + (f" AND {quote(key)} > %s" if last is not None else "") + f" ORDER BY {quote(key)} LIMIT %s"
523
+ p = tuple(params or ()) + ((last,) if last is not None else ()) + (n,)
524
+ else:
525
+ q = base + (f" AND {quote(key)} > :chunk_last" if last is not None else "") \
526
+ + f" ORDER BY {quote(key)} LIMIT :chunk_limit"
527
+ p = dict(params or {}, chunk_limit=n, **({'chunk_last': last} if last is not None else {}))
528
+ return [dict(r) for r in self._execute_query(conn, q, p).mappings()]
529
+
530
+ yield from keyset_chunks(fetch_page, key, chunk_size, resume_token,
531
+ lambda rows, no: convert_rows(rows, schema, no, 'MySQL', self.time_zone),
532
+ drop_key=key not in names)
533
+
534
+ return ChunkResult(schema, chunks())
477
535
 
478
536
  # ----------------------------------------------------------------------------------
479
537
  # INSERT / UPDATE / DELETE