interloper-sql 0.2.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,10 @@
1
+ """Interloper SQL integration for relational database IO via SQLAlchemy."""
2
+
3
+ from interloper_sql.io import MySQLIO, PostgresIO, SqlIO, SqliteIO
4
+
5
+ __all__ = [
6
+ "MySQLIO",
7
+ "PostgresIO",
8
+ "SqlIO",
9
+ "SqliteIO",
10
+ ]
@@ -0,0 +1,13 @@
1
+ """SQL IO managers for reading and writing to databases via SQLAlchemy."""
2
+
3
+ from interloper_sql.io.base import SqlIO
4
+ from interloper_sql.io.mysql import MySQLIO
5
+ from interloper_sql.io.postgres import PostgresIO
6
+ from interloper_sql.io.sqlite import SqliteIO
7
+
8
+ __all__ = [
9
+ "MySQLIO",
10
+ "PostgresIO",
11
+ "SqlIO",
12
+ "SqliteIO",
13
+ ]
@@ -0,0 +1,281 @@
1
+ """SQLAlchemy IO implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Iterator
6
+ from contextlib import contextmanager
7
+ from typing import TYPE_CHECKING, Any
8
+
9
+ from interloper.errors import TableNotFoundError
10
+ from interloper.io.database import DatabaseIO, WriteDisposition
11
+ from sqlalchemy import Column, MetaData, Table, create_engine
12
+ from sqlalchemy import inspect as sa_inspect
13
+
14
+ if TYPE_CHECKING:
15
+ from interloper.io.adapter import DataAdapter
16
+ from sqlalchemy.engine import URL, Connection, Engine
17
+
18
+
19
+ def _infer_sa_type(value: Any) -> Any:
20
+ """Infer a SQLAlchemy column type from a Python value.
21
+
22
+ Args:
23
+ value: A sample Python value used to determine the column type.
24
+
25
+ Returns:
26
+ A SQLAlchemy type instance.
27
+ """
28
+ import datetime
29
+ from decimal import Decimal
30
+
31
+ from sqlalchemy import BigInteger, Boolean, Date, DateTime, Float, LargeBinary, Numeric, Text
32
+
33
+ if isinstance(value, bool):
34
+ return Boolean()
35
+ if isinstance(value, int):
36
+ return BigInteger()
37
+ if isinstance(value, float):
38
+ return Float()
39
+ if isinstance(value, Decimal):
40
+ return Numeric()
41
+ if isinstance(value, datetime.datetime):
42
+ return DateTime()
43
+ if isinstance(value, datetime.date):
44
+ return Date()
45
+ if isinstance(value, bytes):
46
+ return LargeBinary()
47
+ return Text()
48
+
49
+
50
+ class SqlIO(DatabaseIO):
51
+ """Base IO implementation for SQL databases via SQLAlchemy.
52
+
53
+ Provides connection management, transactional writes, table reflection, and
54
+ automatic table creation. Not intended for direct instantiation — use a
55
+ dialect subclass (:class:`PostgresIO`, :class:`MySQLIO`, :class:`SqliteIO`)
56
+ which accepts explicit connection parameters and implements ``to_spec``.
57
+
58
+ The IO is fully stateless with respect to table identity — the table name
59
+ and schema are passed through from the asset context on every call, so a
60
+ single instance can safely serve multiple assets.
61
+
62
+ Args:
63
+ url: SQLAlchemy connection URL or :class:`~sqlalchemy.engine.URL` object
64
+ (constructed by dialect subclasses).
65
+ write_disposition: Controls whether existing rows are deleted before
66
+ writing. Defaults to :attr:`WriteDisposition.REPLACE`.
67
+ chunk_size: Number of rows per insert batch
68
+ adapter: Optional data adapter for type conversion
69
+ """
70
+
71
+ def __init__(
72
+ self,
73
+ url: str | URL,
74
+ write_disposition: WriteDisposition = WriteDisposition.REPLACE,
75
+ chunk_size: int = 1000,
76
+ adapter: DataAdapter | str | None = None,
77
+ ) -> None:
78
+ super().__init__(write_disposition, chunk_size, adapter)
79
+ self._engine: Engine = create_engine(url)
80
+ self._table_cache: dict[tuple[str, str | None], Table] = {}
81
+ self._conn: Connection | None = None
82
+
83
+ # ------------------------------------------------------------------
84
+ # Table helpers
85
+ # ------------------------------------------------------------------
86
+
87
+ def _resolve_table(self, table: str, schema: str | None) -> Table | None:
88
+ """Reflect and cache the SQLAlchemy Table, or return None if it doesn't exist."""
89
+ key = (table, schema)
90
+ if key not in self._table_cache:
91
+ if sa_inspect(self._engine).has_table(table, schema=schema):
92
+ metadata = MetaData(schema=schema)
93
+ self._table_cache[key] = Table(table, metadata, autoload_with=self._engine)
94
+ return self._table_cache.get(key)
95
+
96
+ def _require_table(self, table: str, schema: str | None) -> Table:
97
+ """Resolve the table, raising if it does not exist."""
98
+ sa_table = self._resolve_table(table, schema)
99
+ if sa_table is None:
100
+ qualified = f"{schema}.{table}" if schema else table
101
+ raise TableNotFoundError(f"Table '{qualified}' does not exist. Has the asset been materialized?")
102
+ return sa_table
103
+
104
+ def _create_table(self, table: str, schema: str | None, rows: list[dict[str, Any]]) -> Table:
105
+ """Create a new table from the structure of the first row.
106
+
107
+ Column types are inferred from the Python values in the sample row
108
+ using :func:`_infer_sa_type`.
109
+
110
+ Args:
111
+ table: Target table name
112
+ schema: Database schema
113
+ rows: Row data (at least one row required for schema inference).
114
+
115
+ Returns:
116
+ The newly created :class:`~sqlalchemy.schema.Table`.
117
+ """
118
+ assert self._conn is not None
119
+ sample = rows[0]
120
+ columns = [Column(name, _infer_sa_type(value)) for name, value in sample.items()]
121
+ sa_metadata = MetaData(schema=schema)
122
+ sa_table = Table(table, sa_metadata, *columns)
123
+ sa_metadata.create_all(self._conn)
124
+ self._table_cache[(table, schema)] = sa_table
125
+ return sa_table
126
+
127
+ # ------------------------------------------------------------------
128
+ # Transaction management
129
+ # ------------------------------------------------------------------
130
+
131
+ @contextmanager
132
+ def _transaction(self) -> Iterator[None]:
133
+ """Open a SQLAlchemy transactional connection for write operations.
134
+
135
+ Sets ``self._conn`` for the duration of the block. The connection is
136
+ committed on success and rolled back on exception (``engine.begin()``
137
+ semantics).
138
+
139
+ Yields:
140
+ None
141
+ """
142
+ with self._engine.begin() as conn:
143
+ self._conn = conn
144
+ try:
145
+ yield
146
+ finally:
147
+ self._conn = None
148
+
149
+ # ------------------------------------------------------------------
150
+ # DatabaseIO hooks
151
+ # ------------------------------------------------------------------
152
+
153
+ def _insert(self, table: str, schema: str | None, rows: list[dict[str, Any]]) -> None:
154
+ """Insert rows in chunks using the active transaction connection.
155
+
156
+ If the table does not exist yet, it is created from the row data
157
+ before inserting.
158
+
159
+ Args:
160
+ table: Target table name
161
+ schema: Database schema
162
+ rows: Row data as list of dicts
163
+ """
164
+ assert self._conn is not None
165
+ sa_table = self._resolve_table(table, schema)
166
+ if sa_table is None:
167
+ sa_table = self._create_table(table, schema, rows)
168
+ for i in range(0, len(rows), self.chunk_size):
169
+ self._conn.execute(sa_table.insert(), rows[i : i + self.chunk_size])
170
+
171
+ def _delete_all(self, table: str, schema: str | None) -> None:
172
+ """Delete all rows from the table using the active transaction connection.
173
+
174
+ No-op when the table does not exist yet.
175
+
176
+ Args:
177
+ table: Target table name
178
+ schema: Database schema
179
+ """
180
+ assert self._conn is not None
181
+ sa_table = self._resolve_table(table, schema)
182
+ if sa_table is None:
183
+ return
184
+ self._conn.execute(sa_table.delete())
185
+
186
+ def _delete_partition(self, table: str, schema: str | None, column: str, value: Any) -> None:
187
+ """Delete rows matching a partition value using the active transaction connection.
188
+
189
+ No-op when the table does not exist yet.
190
+
191
+ Args:
192
+ table: Target table name
193
+ schema: Database schema
194
+ column: Partition column name
195
+ value: Partition value to match
196
+ """
197
+ assert self._conn is not None
198
+ sa_table = self._resolve_table(table, schema)
199
+ if sa_table is None:
200
+ return
201
+ self._conn.execute(sa_table.delete().where(sa_table.c[column] == value))
202
+
203
+ def _select_all(self, table: str, schema: str | None) -> list[dict[str, Any]]:
204
+ """Select all rows from the table.
205
+
206
+ Opens a dedicated read connection (not part of the write transaction).
207
+
208
+ Args:
209
+ table: Target table name
210
+ schema: Database schema
211
+
212
+ Returns:
213
+ All rows as list of dicts
214
+
215
+ Raises:
216
+ ValueError: If the table does not exist
217
+ """
218
+ sa_table = self._require_table(table, schema)
219
+ with self._engine.connect() as conn:
220
+ result = conn.execute(sa_table.select())
221
+ return [dict(row._mapping) for row in result]
222
+
223
+ def _select_partition(self, table: str, schema: str | None, column: str, value: Any) -> list[dict[str, Any]]:
224
+ """Select rows matching a partition value.
225
+
226
+ Opens a dedicated read connection (not part of the write transaction).
227
+
228
+ Args:
229
+ table: Target table name
230
+ schema: Database schema
231
+ column: Partition column name
232
+ value: Partition value to match
233
+
234
+ Returns:
235
+ Matching rows as list of dicts
236
+
237
+ Raises:
238
+ ValueError: If the table does not exist
239
+ """
240
+ sa_table = self._require_table(table, schema)
241
+ with self._engine.connect() as conn:
242
+ result = conn.execute(sa_table.select().where(sa_table.c[column] == value))
243
+ return [dict(row._mapping) for row in result]
244
+
245
+ # ------------------------------------------------------------------
246
+ # Introspection
247
+ # ------------------------------------------------------------------
248
+
249
+ def _count_by_partition(
250
+ self, table: str, schema: str | None, column: str,
251
+ ) -> dict[str, int]:
252
+ """Return row counts grouped by partition column via SQL ``GROUP BY``.
253
+
254
+ Args:
255
+ table: Target table name
256
+ schema: Database schema
257
+ column: Column to group by
258
+
259
+ Returns:
260
+ Mapping from partition value (as string) to row count.
261
+
262
+ Raises:
263
+ TableNotFoundError: If the table does not exist.
264
+ """
265
+ from sqlalchemy import func
266
+
267
+ sa_table = self._require_table(table, schema)
268
+ col = sa_table.c[column]
269
+ stmt = sa_table.select().with_only_columns(col, func.count()).group_by(col)
270
+ with self._engine.connect() as conn:
271
+ result = conn.execute(stmt)
272
+ return {str(row[0]): row[1] for row in result}
273
+
274
+ # ------------------------------------------------------------------
275
+ # Lifecycle
276
+ # ------------------------------------------------------------------
277
+
278
+ def dispose(self) -> None:
279
+ """Dispose the SQLAlchemy engine and clear the table cache."""
280
+ self._engine.dispose()
281
+ self._table_cache.clear()
@@ -0,0 +1,70 @@
1
+ """MySQL IO implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING
6
+
7
+ from interloper.io.database import WriteDisposition
8
+ from interloper.serialization.io import IOSpec
9
+ from sqlalchemy.engine import URL
10
+
11
+ from interloper_sql.io.base import SqlIO
12
+
13
+ if TYPE_CHECKING:
14
+ from interloper.io.adapter import DataAdapter
15
+
16
+
17
+ class MySQLIO(SqlIO):
18
+ """MySQL-specific IO manager.
19
+
20
+ Extends :class:`SqlIO` for MySQL connections. Uses standard ``DELETE`` for
21
+ row removal because MySQL's ``TRUNCATE`` causes an implicit commit and
22
+ cannot participate in a transaction.
23
+
24
+ Args:
25
+ host: Database server hostname
26
+ database: Database name
27
+ port: Database server port
28
+ user: Database user
29
+ password: Database password
30
+ driver: SQLAlchemy driver (e.g. ``pymysql``, ``mysqlconnector``)
31
+ write_disposition: Controls whether existing rows are deleted before writing
32
+ chunk_size: Number of rows per insert batch
33
+ adapter: Optional data adapter for type conversion
34
+ """
35
+
36
+ def __init__(
37
+ self,
38
+ host: str,
39
+ database: str,
40
+ port: int = 3306,
41
+ user: str = "root",
42
+ password: str | None = None,
43
+ driver: str | None = None,
44
+ write_disposition: WriteDisposition = WriteDisposition.REPLACE,
45
+ chunk_size: int = 1000,
46
+ adapter: DataAdapter | str | None = None,
47
+ ) -> None:
48
+ self.host = host
49
+ self.port = port
50
+ self.database = database
51
+ self.user = user
52
+ self.password = password
53
+ self.driver = driver
54
+
55
+ drivername = f"mysql+{driver}" if driver else "mysql"
56
+ url = URL.create(drivername, user, password, host, port, database)
57
+ super().__init__(url, write_disposition, chunk_size, adapter)
58
+
59
+ def to_spec(self) -> IOSpec:
60
+ """Convert to serializable spec."""
61
+ init = self._base_init_kwargs()
62
+ init["host"] = self.host
63
+ init["port"] = self.port
64
+ init["database"] = self.database
65
+ init["user"] = self.user
66
+ if self.password is not None:
67
+ init["password"] = self.password
68
+ if self.driver is not None:
69
+ init["driver"] = self.driver
70
+ return IOSpec(path=self.path, init=init)
@@ -0,0 +1,87 @@
1
+ """PostgreSQL IO implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING
6
+
7
+ from interloper.io.database import WriteDisposition
8
+ from interloper.serialization.io import IOSpec
9
+ from sqlalchemy import text
10
+ from sqlalchemy.engine import URL
11
+
12
+ from interloper_sql.io.base import SqlIO
13
+
14
+ if TYPE_CHECKING:
15
+ from interloper.io.adapter import DataAdapter
16
+
17
+
18
+ class PostgresIO(SqlIO):
19
+ """PostgreSQL-specific IO manager.
20
+
21
+ Extends :class:`SqlIO` with PostgreSQL optimisations:
22
+
23
+ * Uses ``TRUNCATE`` (transactional in Postgres) instead of ``DELETE`` for
24
+ full-table replacements, which is significantly faster on large tables.
25
+
26
+ Args:
27
+ host: Database server hostname
28
+ port: Database server port
29
+ database: Database name
30
+ user: Database user
31
+ password: Database password
32
+ driver: SQLAlchemy driver (e.g. ``psycopg2``, ``asyncpg``)
33
+ write_disposition: Controls whether existing rows are deleted before writing
34
+ chunk_size: Number of rows per insert batch
35
+ adapter: Optional data adapter for type conversion
36
+ """
37
+
38
+ def __init__(
39
+ self,
40
+ host: str,
41
+ port: int = 5432,
42
+ database: str = "postgres",
43
+ user: str = "postgres",
44
+ password: str | None = None,
45
+ driver: str | None = None,
46
+ write_disposition: WriteDisposition = WriteDisposition.REPLACE,
47
+ chunk_size: int = 1000,
48
+ adapter: DataAdapter | str | None = None,
49
+ ) -> None:
50
+ self.host = host
51
+ self.port = port
52
+ self.database = database
53
+ self.user = user
54
+ self.password = password
55
+ self.driver = driver
56
+
57
+ drivername = f"postgresql+{driver}" if driver else "postgresql"
58
+ url = URL.create(drivername, user, password, host, port, database)
59
+ super().__init__(url, write_disposition, chunk_size, adapter)
60
+
61
+ def _delete_all(self, table: str, schema: str | None) -> None:
62
+ """Use TRUNCATE for full-table deletes (transactional in PostgreSQL).
63
+
64
+ No-op when the table does not exist yet.
65
+
66
+ Args:
67
+ table: Target table name
68
+ schema: Database schema
69
+ """
70
+ assert self._conn is not None
71
+ sa_table = self._resolve_table(table, schema)
72
+ if sa_table is None:
73
+ return
74
+ self._conn.execute(text(f"TRUNCATE TABLE {sa_table.fullname}"))
75
+
76
+ def to_spec(self) -> IOSpec:
77
+ """Convert to serializable spec."""
78
+ init = self._base_init_kwargs()
79
+ init["host"] = self.host
80
+ init["port"] = self.port
81
+ init["database"] = self.database
82
+ init["user"] = self.user
83
+ if self.password is not None:
84
+ init["password"] = self.password
85
+ if self.driver is not None:
86
+ init["driver"] = self.driver
87
+ return IOSpec(path=self.path, init=init)
@@ -0,0 +1,45 @@
1
+ """SQLite IO implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import TYPE_CHECKING
6
+
7
+ from interloper.io.database import WriteDisposition
8
+ from interloper.serialization.io import IOSpec
9
+
10
+ from interloper_sql.io.base import SqlIO
11
+
12
+ if TYPE_CHECKING:
13
+ from interloper.io.adapter import DataAdapter
14
+
15
+
16
+ class SqliteIO(SqlIO):
17
+ """SQLite-specific IO manager.
18
+
19
+ Extends :class:`SqlIO` for SQLite connections. Useful for local development
20
+ and testing without requiring an external database server.
21
+
22
+ Args:
23
+ database: Path to the SQLite database file, or ``":memory:"`` for an
24
+ in-memory database.
25
+ write_disposition: Controls whether existing rows are deleted before writing
26
+ chunk_size: Number of rows per insert batch
27
+ adapter: Optional data adapter for type conversion
28
+ """
29
+
30
+ def __init__(
31
+ self,
32
+ database: str = ":memory:",
33
+ write_disposition: WriteDisposition = WriteDisposition.REPLACE,
34
+ chunk_size: int = 1000,
35
+ adapter: DataAdapter | str | None = None,
36
+ ) -> None:
37
+ self.database = database
38
+ url = f"sqlite:///{database}"
39
+ super().__init__(url, write_disposition, chunk_size, adapter)
40
+
41
+ def to_spec(self) -> IOSpec:
42
+ """Convert to serializable spec."""
43
+ init = self._base_init_kwargs()
44
+ init["database"] = self.database
45
+ return IOSpec(path=self.path, init=init)
@@ -0,0 +1,18 @@
1
+ Metadata-Version: 2.3
2
+ Name: interloper-sql
3
+ Version: 0.2.0rc1
4
+ Summary: Interloper SQLAlchemy IO managers
5
+ Author: Guillaume Onfroy
6
+ Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
7
+ Requires-Dist: sqlalchemy>=2.0
8
+ Requires-Dist: interloper-core
9
+ Requires-Dist: pymysql>=1.1 ; extra == 'mysql'
10
+ Requires-Dist: psycopg2-binary>=2.9 ; extra == 'postgres'
11
+ Requires-Python: >=3.10
12
+ Provides-Extra: mysql
13
+ Provides-Extra: postgres
14
+ Description-Content-Type: text/markdown
15
+
16
+ # interloper-sql
17
+
18
+ SQLAlchemy IO managers for the Interloper framework.
@@ -0,0 +1,9 @@
1
+ interloper_sql/__init__.py,sha256=__WqY0mvjudCFv3LwyHExUO5vGxd-JnWNXiE2ujA_4E,221
2
+ interloper_sql/io/__init__.py,sha256=gf_2w_A3d2yMujwWd5RNpyUDLkCyjL2DYFXV_81Tj7s,334
3
+ interloper_sql/io/base.py,sha256=jRwXI5LCT5xAoBtUun1SBzKANKtVd5wU_z44HDyFqqw,10386
4
+ interloper_sql/io/mysql.py,sha256=hMzFCYNUERUVp-LScZJ9PzZNAKQ44CBGTfN0XlRwJA0,2273
5
+ interloper_sql/io/postgres.py,sha256=xkuw7-OvfJbR_qZm3z7mtdmshxxFoZ5cF8SpmtohC1o,2863
6
+ interloper_sql/io/sqlite.py,sha256=w8KCzYF4SiG9ROSwB2198ljQTL8H7OFI2ec38jAFFUU,1428
7
+ interloper_sql-0.2.0rc1.dist-info/WHEEL,sha256=iHtWm8nRfs0VRdCYVXocAWFW8ppjHL-uTJkAdZJKOBM,80
8
+ interloper_sql-0.2.0rc1.dist-info/METADATA,sha256=aNYYSBu9vBkzg18V-veVHrJvB4RMIFGi9EfSfWAQLoE,538
9
+ interloper_sql-0.2.0rc1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: uv 0.9.30
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any