tabsdata-conn-common-sql 2.0.0__cp312-abi3-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/constants.py +34 -0
  2. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/factory.py +60 -0
  3. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/initial_values.py +37 -0
  4. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/reader.py +35 -0
  5. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/sqlalchemy_reader.py +199 -0
  6. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/sqlalchemy_utils.py +63 -0
  7. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/sqlalchemy_writer.py +374 -0
  8. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/type_convertion.py +42 -0
  9. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_engine/writer.py +36 -0
  10. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_explorer.py +294 -0
  11. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_iceberg.py +375 -0
  12. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/_metadata.py +68 -0
  13. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/assets/manifest/BANNER +5 -0
  14. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/assets/manifest/LICENSE +3 -0
  15. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/assets/manifest/README-PyPi.md +8 -0
  16. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/assets/manifest/RELEASE-NOTES +1 -0
  17. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/assets/manifest/THIRD-PARTY +3 -0
  18. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/assets/manifest/VERSION +1 -0
  19. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/__init__.py +70 -0
  20. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_column.py +31 -0
  21. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_connection.py +87 -0
  22. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_dictionary.py +112 -0
  23. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_dtype.py +73 -0
  24. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_ingester.py +368 -0
  25. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_model.py +111 -0
  26. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_runner.py +282 -0
  27. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_schema.py +120 -0
  28. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_sentinel/__init__.py +119 -0
  29. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_sentinel/buffer.py +112 -0
  30. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_sentinel/disk.py +152 -0
  31. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_stage_plugin.py +226 -0
  32. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/_utils.py +33 -0
  33. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/constants.py +85 -0
  34. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/dialect.py +186 -0
  35. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/error.py +163 -0
  36. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/typing.py +75 -0
  37. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/cdc/utils.py +10 -0
  38. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/constants.py +20 -0
  39. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/error.py +42 -0
  40. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/conn/common/sql/types.py +28 -0
  41. tabsdata_conn_common_sql-2.0.0.data/purelib/tabsdatak/py.typed +0 -0
  42. tabsdata_conn_common_sql-2.0.0.dist-info/METADATA +44 -0
  43. tabsdata_conn_common_sql-2.0.0.dist-info/RECORD +47 -0
  44. tabsdata_conn_common_sql-2.0.0.dist-info/WHEEL +5 -0
  45. tabsdata_conn_common_sql-2.0.0.dist-info/entry_points.txt +2 -0
  46. tabsdata_conn_common_sql-2.0.0.dist-info/licenses/src/main/python/tabsdatak/conn/common/sql/assets/manifest/LICENSE +3 -0
  47. tabsdata_conn_common_sql-2.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,34 @@
1
+ #
2
+ # Copyright 2026 Tabsdata Inc.
3
+ #
4
+
5
+ from tabsdatak.conn.common import _config
6
+
7
+ SQL_KEY_PREFIX = _config.TABSDATA_CFG_PREFIX + "sql."
8
+
9
+ # --- Read --------------------------------------------------------
10
+
11
+ READ_ENGINE_KEY = SQL_KEY_PREFIX + "read_engine"
12
+ POLARS_SQL_ALCHEMY_READ_ENGINE = "polarsSqlAlchemyRead"
13
+ PANDAS_SQL_ALCHEMY_READ_ENGINE = "pandasSqlAlchemyRead"
14
+
15
+ AVAIL_READ_ENGINES_KEY = SQL_KEY_PREFIX + "engine.avail_read_engines"
16
+
17
+ READ_CHUNK_SIZE_KEY = SQL_KEY_PREFIX + "chunk_size"
18
+ READ_CHUNK_SIZE_DEFAULT = 100_000
19
+
20
+ POLARS_SQL_ALCHEMY_READ_CHUNK_SIZE_DEFAULT = READ_CHUNK_SIZE_DEFAULT
21
+ PANDAS_SQL_ALCHEMY_READ_CHUNK_SIZE_DEFAULT = READ_CHUNK_SIZE_DEFAULT
22
+
23
+
24
+ # --- Write -------------------------------------------------------
25
+
26
+ WRITE_ENGINE_KEY = SQL_KEY_PREFIX + "write_engine"
27
+ POLARS_SQL_ALCHEMY_WRITE_ENGINE = "polarsSqlAlchemyWrite"
28
+
29
+ AVAIL_WRITE_ENGINES_KEY = SQL_KEY_PREFIX + "engine.avail_write_engines"
30
+
31
+ WRITE_CHUNK_SIZE_KEY = SQL_KEY_PREFIX + "chunk_size"
32
+ WRITE_CHUNK_SIZE_DEFAULT = 100_000
33
+
34
+ POLARS_SQL_ALCHEMY_WRITE_CHUNK_SIZE_DEFAULT = WRITE_CHUNK_SIZE_DEFAULT
@@ -0,0 +1,60 @@
1
+ #
2
+ # Copyright 2026 Tabsdata Inc.
3
+ #
4
+
5
+ from typing import Any, Mapping
6
+
7
+ # noinspection PyProtectedMember
8
+ import tabsdatak.conn.common.sql._engine.constants as const
9
+
10
+ # noinspection PyProtectedMember
11
+ from tabsdatak.conn.common import _config
12
+
13
+ # noinspection PyProtectedMember
14
+ from tabsdatak.conn.common.sql._engine.reader import SqlReadEngine
15
+
16
+ # noinspection PyProtectedMember
17
+ from tabsdatak.conn.common.sql._engine.sqlalchemy_reader import (
18
+ PandasReadEngine,
19
+ PolarsReadEngine,
20
+ )
21
+
22
+ # noinspection PyProtectedMember
23
+ from tabsdatak.conn.common.sql._engine.sqlalchemy_writer import PolarsWriteEngine
24
+
25
+ # noinspection PyProtectedMember
26
+ from tabsdatak.conn.common.sql._engine.writer import SqlWriteEngine
27
+ from tabsdatak.conn.common.sql.error import SqlCommonErrorCode
28
+ from tabsdatak.conn.common.types import CfgSpec
29
+
30
+
31
+ def get_read_engine(
32
+ src_cfg: CfgSpec,
33
+ defaults: Mapping[str, Any],
34
+ ) -> SqlReadEngine:
35
+ engine = _config.get_cfg(
36
+ src_cfg,
37
+ const.READ_ENGINE_KEY,
38
+ defaults[const.READ_ENGINE_KEY],
39
+ defaults[const.AVAIL_READ_ENGINES_KEY],
40
+ )
41
+ if engine == const.PANDAS_SQL_ALCHEMY_READ_ENGINE:
42
+ return PandasReadEngine()
43
+ if engine == const.POLARS_SQL_ALCHEMY_READ_ENGINE:
44
+ return PolarsReadEngine()
45
+ raise SqlCommonErrorCode.SQL_COMMON_1.exception(engine=engine)
46
+
47
+
48
+ def get_write_engine(
49
+ dest_cfg: CfgSpec,
50
+ defaults: Mapping[str, Any],
51
+ ) -> SqlWriteEngine:
52
+ engine = _config.get_cfg(
53
+ dest_cfg,
54
+ const.WRITE_ENGINE_KEY,
55
+ defaults[const.WRITE_ENGINE_KEY],
56
+ defaults[const.AVAIL_WRITE_ENGINES_KEY],
57
+ )
58
+ if engine == const.POLARS_SQL_ALCHEMY_WRITE_ENGINE:
59
+ return PolarsWriteEngine()
60
+ raise SqlCommonErrorCode.SQL_COMMON_1.exception(engine=engine)
@@ -0,0 +1,37 @@
1
+ #
2
+ # Copyright 2026 Tabsdata Inc.
3
+ #
4
+
5
+ from __future__ import annotations
6
+
7
+ from typing import Any, Mapping
8
+
9
+ from tabsdatak.spi import SrcPluginCtx
10
+
11
+
12
+ def resolve_initial_values(
13
+ ctx: SrcPluginCtx,
14
+ seed: Mapping[str, Any] | None,
15
+ ) -> dict[str, Any] | None:
16
+ """Look up the active bind dict for this run.
17
+
18
+ Walks ``seed`` (the user's first-run ``Src.initial_values``) bind
19
+ name by bind name. For each, prefers ``ctx.get_attr(name)`` --
20
+ that's where the previous run's persisted value lands -- and
21
+ falls back to the seed when the ctx scratch has no entry for
22
+ that name.
23
+
24
+ Returns ``None`` when ``seed`` is None (no placeholders in any
25
+ query), signalling no substitution is needed.
26
+
27
+ Uses ``is not None`` rather than truthiness so a legitimately
28
+ falsy value stored last run (``0``, ``False``, ``""``) isn't
29
+ silently shadowed by the seed.
30
+ """
31
+ if seed is None:
32
+ return None
33
+ out: dict[str, Any] = {}
34
+ for name, value in seed.items():
35
+ from_ctx = ctx.get_attr(name)
36
+ out[name] = from_ctx if from_ctx is not None else value
37
+ return out
@@ -0,0 +1,35 @@
1
+ #
2
+ # Copyright 2026 Tabsdata Inc.
3
+ #
4
+
5
+ from abc import ABC, abstractmethod
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+ from typing import Any, Mapping
9
+
10
+ from sqlalchemy import Engine
11
+
12
+ from tabsdatak.conn.common.sql.types import SchemaOverrideSpec
13
+ from tabsdatak.conn.common.types import CfgSpec
14
+ from tabsdatak.spi import Conn, TableFileSpec
15
+
16
+
17
+ @dataclass(kw_only=True)
18
+ class SqlReadCfg[C: Conn](ABC):
19
+ conn: C
20
+ queries: list[str]
21
+ transactional: bool
22
+ schema_overrides: SchemaOverrideSpec | None
23
+ initial_values: Mapping[str, Any] | None
24
+ cfg: CfgSpec | None
25
+
26
+ def create_sql_alchemy_engine(self) -> Engine:
27
+ """
28
+ Implemented by subclasses that use SqlAlchemy.
29
+ """
30
+ raise NotImplementedError()
31
+
32
+
33
+ class SqlReadEngine(ABC):
34
+ @abstractmethod
35
+ def read(self, cfg: SqlReadCfg, work_dir: Path) -> list[list[TableFileSpec]]: ...
@@ -0,0 +1,199 @@
1
+ #
2
+ # Copyright 2026 Tabsdata Inc.
3
+ #
4
+
5
+ import logging
6
+ from abc import ABC, abstractmethod
7
+ from pathlib import Path
8
+ from typing import Any, Callable, Mapping
9
+
10
+ from sqlalchemy import Connection
11
+ from sqlalchemy.exc import OperationalError
12
+ from sqlalchemy.orm import Session
13
+ from tenacity import (
14
+ Retrying,
15
+ before_sleep_log,
16
+ retry_if_exception_type,
17
+ stop_after_delay,
18
+ wait_fixed,
19
+ )
20
+
21
+ # noinspection PyProtectedMember
22
+ import tabsdatak.conn.common.sql._engine.constants as const
23
+
24
+ # noinspection PyProtectedMember
25
+ from tabsdatak.conn.common import _config
26
+
27
+ # noinspection PyProtectedMember
28
+ from tabsdatak.conn.common.sql._engine.reader import SqlReadCfg, SqlReadEngine
29
+
30
+ # noinspection PyProtectedMember
31
+ from tabsdatak.conn.common.sql._engine.type_convertion import resolve_schema_overrides
32
+ from tabsdatak.spi import TableFileSpec
33
+
34
+ # --- Retry ----------------------------------------------------------
35
+
36
+ logger = logging.getLogger(__name__)
37
+
38
+ RETRY_TIMEOUT_S = 5 * 60
39
+ RETRY_WAIT_S = 5
40
+
41
+
42
+ def _retry() -> Retrying:
43
+ return Retrying(
44
+ stop=stop_after_delay(RETRY_TIMEOUT_S),
45
+ wait=wait_fixed(RETRY_WAIT_S),
46
+ retry=retry_if_exception_type(OperationalError),
47
+ before_sleep=before_sleep_log(logger, logging.WARNING),
48
+ reraise=True,
49
+ )
50
+
51
+
52
+ ReadLoader = Callable[..., list[Path]]
53
+
54
+
55
+ class SqlAlchemyReadEngine(SqlReadEngine, ABC):
56
+ @abstractmethod
57
+ def loader(self) -> ReadLoader: ...
58
+
59
+ def read(self, src: SqlReadCfg, work_dir: Path) -> list[list[TableFileSpec]]:
60
+ read_loader = self.loader()
61
+
62
+ # Empty dict -> None (per the public contract; saves the
63
+ # loaders from having to special-case the empty case).
64
+ parameters: Mapping[str, Any] | None = src.initial_values or None
65
+
66
+ def _run() -> list[list[TableFileSpec]]:
67
+ logger.debug("sql-reader: establishing engine")
68
+ engine = src.create_sql_alchemy_engine()
69
+ try:
70
+ results: list[list[TableFileSpec]] = []
71
+ if src.transactional:
72
+ with Session(engine) as session, session.begin():
73
+ conn = session.connection()
74
+ self._run_queries(
75
+ read_loader, conn, src, work_dir, parameters, results
76
+ )
77
+ else:
78
+ with engine.connect() as conn:
79
+ self._run_queries(
80
+ read_loader, conn, src, work_dir, parameters, results
81
+ )
82
+ return results
83
+ finally:
84
+ logger.debug("sql-reader: disposing engine")
85
+ engine.dispose()
86
+
87
+ return _retry()(_run)
88
+
89
+ @staticmethod
90
+ def _run_queries(
91
+ read_loader: Any,
92
+ conn: Any,
93
+ src: SqlReadCfg,
94
+ work_dir: Path,
95
+ parameters: Mapping[str, Any] | None,
96
+ results: list[list[TableFileSpec]],
97
+ ):
98
+ chunk_size = _config.get_cfg(
99
+ src.cfg,
100
+ const.READ_CHUNK_SIZE_KEY,
101
+ const.READ_CHUNK_SIZE_DEFAULT,
102
+ )
103
+ total = len(src.queries)
104
+ logger.info("sql-reader: reading %d quer(ies)", total)
105
+ for i, query in enumerate(src.queries):
106
+ logger.info("sql-reader: reading query %d/%d", i + 1, total)
107
+ overrides = (
108
+ resolve_schema_overrides(src.schema_overrides[i])
109
+ if src.schema_overrides is not None
110
+ else None
111
+ )
112
+ files = read_loader(
113
+ connection=conn,
114
+ query=query,
115
+ work_dir=work_dir,
116
+ slot=i,
117
+ chunk_size=chunk_size,
118
+ schema_overrides=overrides,
119
+ parameters=parameters,
120
+ )
121
+ logger.debug(
122
+ "sql-reader: query %d/%d produced %d chunk(s)", i + 1, total, len(files)
123
+ )
124
+ results.append([Path(p) for p in files])
125
+ logger.info("sql-reader: finished reading %d quer(ies)", total)
126
+
127
+
128
+ def _pandas_read(
129
+ *,
130
+ connection: Connection,
131
+ query: str,
132
+ work_dir: Path,
133
+ slot: int,
134
+ chunk_size: int,
135
+ schema_overrides: Mapping[str, Any] | None,
136
+ parameters: Mapping[str, Any] | None = None,
137
+ ) -> list[Path]:
138
+ import pandas as pd
139
+ from sqlalchemy import text
140
+
141
+ files: list[Path] = []
142
+ iterator = pd.read_sql_query(
143
+ text(query),
144
+ con=connection,
145
+ chunksize=chunk_size,
146
+ dtype=dict(schema_overrides) if schema_overrides else None,
147
+ params=dict(parameters) if parameters else None,
148
+ )
149
+ for chunk_idx, frame in enumerate(iterator):
150
+ out = work_dir / f"slot{slot}_chunk{chunk_idx}.parquet"
151
+ frame.to_parquet(out, engine="pyarrow")
152
+ files.append(out)
153
+
154
+ return files
155
+
156
+
157
+ class PandasReadEngine(SqlAlchemyReadEngine):
158
+ def loader(self) -> ReadLoader:
159
+ return _pandas_read
160
+
161
+
162
+ def _polars_read(
163
+ *,
164
+ connection: Any,
165
+ query: str,
166
+ work_dir: Path,
167
+ slot: int,
168
+ chunk_size: int,
169
+ schema_overrides: Mapping[str, Any] | None,
170
+ parameters: Mapping[str, Any] | None = None,
171
+ ) -> list[Path]:
172
+ import polars as pl
173
+ from sqlalchemy import text
174
+
175
+ execute_options: dict[str, Any] = {}
176
+ if parameters:
177
+ execute_options["parameters"] = dict(parameters)
178
+
179
+ files: list[Path] = []
180
+ streamed_conn = connection.execution_options(stream_results=True)
181
+ batches = pl.read_database(
182
+ text(query),
183
+ connection=streamed_conn,
184
+ iter_batches=True,
185
+ batch_size=chunk_size,
186
+ schema_overrides=schema_overrides,
187
+ execute_options=execute_options or None,
188
+ )
189
+ for chunk_idx, batch in enumerate(batches):
190
+ out = work_dir / f"slot{slot}_chunk{chunk_idx}.parquet"
191
+ batch.write_parquet(out, use_pyarrow=True)
192
+ files.append(out)
193
+
194
+ return files
195
+
196
+
197
+ class PolarsReadEngine(SqlAlchemyReadEngine):
198
+ def loader(self) -> ReadLoader:
199
+ return _polars_read
@@ -0,0 +1,63 @@
1
+ #
2
+ # Copyright 2026 Tabsdata Inc.
3
+ #
4
+
5
+ from __future__ import annotations
6
+
7
+ import logging
8
+ from typing import Any, Mapping
9
+
10
+ import sqlalchemy
11
+ from sqlalchemy.engine import URL, Engine
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ def create_engine(url: str | URL, **kwargs: Any) -> Engine:
17
+ """Build a SQLAlchemy engine through :func:`sqlalchemy.create_engine`,
18
+ logging how the URL resolved to a driver and a dialect implementation.
19
+
20
+ Every connector creates its engines here so that the resolution is
21
+ traceable from a function's log alone. The URL is rendered with the
22
+ password hidden, so the caller may pass one with credentials in it.
23
+ """
24
+ engine = sqlalchemy.create_engine(url, **kwargs)
25
+ logger.info(
26
+ "SQLAlchemy engine: URI: %s - URL driver name: %s - Engine driver: %s"
27
+ " - Dialect: %s",
28
+ engine.url.render_as_string(hide_password=True),
29
+ engine.url.drivername,
30
+ engine.driver,
31
+ type(engine.dialect).__module__,
32
+ )
33
+ return engine
34
+
35
+
36
+ def normalize_scheme(uri: str, *, aliases: Mapping[str, str]) -> str:
37
+ """Replace the URI's scheme with its canonical form, if it appears
38
+ in ``aliases``. ``aliases`` maps the user-typed scheme to the
39
+ canonical one (e.g. ``{"mariadb": "mysql"}``).
40
+
41
+ The match is exact on the bare scheme (``mariadb://``) and on the
42
+ driver-qualified prefix (``mariadb+...``). The rest of the URI is
43
+ untouched.
44
+ """
45
+ for alias, canonical in aliases.items():
46
+ if uri.startswith(f"{alias}://") or uri.startswith(f"{alias}+"):
47
+ return uri.replace(alias, canonical, 1)
48
+ return uri
49
+
50
+
51
+ def add_driver_to_uri(uri: str, *, drivers: Mapping[str, str]) -> str:
52
+ """Inject a default DBAPI driver into a bare ``<scheme>://`` URI
53
+ if one isn't already present. ``drivers`` maps the scheme to the
54
+ driver to inject (e.g. ``{"mysql": "mysqlconnector"}``).
55
+
56
+ If the URI already carries a ``+driver`` qualifier the function
57
+ is a no-op.
58
+ """
59
+ for scheme, driver in drivers.items():
60
+ bare_prefix = f"{scheme}://"
61
+ if uri.startswith(bare_prefix):
62
+ return uri.replace(bare_prefix, f"{scheme}+{driver}://", 1)
63
+ return uri