interloper-sql 0.2.0rc1__tar.gz → 0.94.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interloper_sql-0.94.2/PKG-INFO +90 -0
- interloper_sql-0.94.2/README.md +78 -0
- interloper_sql-0.94.2/pyproject.toml +60 -0
- interloper_sql-0.2.0rc1/pyproject.toml → interloper_sql-0.94.2/pyproject.toml.orig +13 -18
- interloper_sql-0.94.2/src/interloper_sql/__init__.py +9 -0
- interloper_sql-0.94.2/src/interloper_sql/connection.py +74 -0
- interloper_sql-0.94.2/src/interloper_sql/destination.py +351 -0
- interloper_sql-0.94.2/src/interloper_sql/types.py +43 -0
- interloper_sql-0.2.0rc1/PKG-INFO +0 -18
- interloper_sql-0.2.0rc1/README.md +0 -3
- interloper_sql-0.2.0rc1/src/interloper_sql/__init__.py +0 -10
- interloper_sql-0.2.0rc1/src/interloper_sql/io/__init__.py +0 -13
- interloper_sql-0.2.0rc1/src/interloper_sql/io/base.py +0 -281
- interloper_sql-0.2.0rc1/src/interloper_sql/io/mysql.py +0 -70
- interloper_sql-0.2.0rc1/src/interloper_sql/io/postgres.py +0 -87
- interloper_sql-0.2.0rc1/src/interloper_sql/io/sqlite.py +0 -45
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: interloper-sql
|
|
3
|
+
Version: 0.94.2
|
|
4
|
+
Summary: Interloper SQL integration: a SQLAlchemy-backed destination for PostgreSQL and other SQL databases
|
|
5
|
+
Author: Guillaume Onfroy
|
|
6
|
+
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
+
Requires-Dist: interloper-core
|
|
8
|
+
Requires-Dist: psycopg[binary]>=3.1
|
|
9
|
+
Requires-Dist: sqlalchemy>=2.0
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# interloper-sql
|
|
14
|
+
|
|
15
|
+
A SQL database destination for interloper: `SQLDestination`, a
|
|
16
|
+
`DatabaseDestination` over SQLAlchemy Core, and the `SQLConnection` that holds
|
|
17
|
+
the database URL.
|
|
18
|
+
|
|
19
|
+
One backend serves every SQLAlchemy dialect whose driver is installed.
|
|
20
|
+
PostgreSQL works out of the box (the package depends on `psycopg[binary]`);
|
|
21
|
+
MySQL, SQLite, Redshift, SQL Server and the rest need their own driver.
|
|
22
|
+
|
|
23
|
+
## Setup
|
|
24
|
+
|
|
25
|
+
The connection is a single SQLAlchemy URL:
|
|
26
|
+
|
|
27
|
+
| Database | URL | Driver to install |
|
|
28
|
+
|----------|-----|-------------------|
|
|
29
|
+
| PostgreSQL | `postgresql+psycopg://user:password@host:5432/db` | none, ships with the package |
|
|
30
|
+
| MySQL / MariaDB | `mysql+pymysql://user:password@host:3306/db` | `pymysql` |
|
|
31
|
+
| SQL Server | `mssql+pyodbc://user:password@host/db?driver=ODBC+Driver+18+for+SQL+Server` | `pyodbc` (and the ODBC driver) |
|
|
32
|
+
| Redshift | `redshift+redshift_connector://user:password@host:5439/db` | `sqlalchemy-redshift`, `redshift_connector` |
|
|
33
|
+
| SQLite | `sqlite:///path/to/file.db` | none, in the standard library |
|
|
34
|
+
|
|
35
|
+
Install the driver next to interloper (`uv pip install pymysql`, or a custom
|
|
36
|
+
image built from the slim variant). Special characters in the password must be
|
|
37
|
+
URL-encoded.
|
|
38
|
+
|
|
39
|
+
The URL carries the password, so it is a secret field: encrypted at rest in a
|
|
40
|
+
deployed instance and masked in the UI. It also loads from the environment
|
|
41
|
+
(`SQL_URL`), so `SQLConnection()` works with no arguments.
|
|
42
|
+
|
|
43
|
+
The connection's check runs `SELECT 1`.
|
|
44
|
+
|
|
45
|
+
## Usage
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import interloper as il
|
|
49
|
+
from interloper_sql import SQLConnection, SQLDestination
|
|
50
|
+
|
|
51
|
+
warehouse = SQLDestination(
|
|
52
|
+
connection=SQLConnection(url="postgresql+psycopg://user:password@host:5432/analytics"),
|
|
53
|
+
default_dataset="marketing",
|
|
54
|
+
)
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
In a deployed instance you configure this through the UI instead: add a SQL
|
|
58
|
+
database connection, then a SQL database destination using it.
|
|
59
|
+
|
|
60
|
+
## Datasets are schemas
|
|
61
|
+
|
|
62
|
+
An asset's `dataset` is the SQL schema its table lives in. Without one, the
|
|
63
|
+
destination's `default_dataset` applies; without either, the table goes in
|
|
64
|
+
the connection's default schema (`public` on PostgreSQL, the database itself on
|
|
65
|
+
MySQL). A named schema is created on first write when the database supports
|
|
66
|
+
schemas and it does not exist yet.
|
|
67
|
+
|
|
68
|
+
## Tables
|
|
69
|
+
|
|
70
|
+
A table is created on first write from the asset's effective schema (declared,
|
|
71
|
+
or inferred during conform): integers are `BIGINT`, floats `DOUBLE`, decimals
|
|
72
|
+
`NUMERIC(38, 9)`, datetimes timezone-aware timestamps, strings `TEXT`, and
|
|
73
|
+
nested or repeated fields `JSON`. An existing table is never altered: columns
|
|
74
|
+
it lacks are dropped from the write with a warning, and NaN or infinite floats
|
|
75
|
+
are written as `NULL`.
|
|
76
|
+
|
|
77
|
+
## Partitions
|
|
78
|
+
|
|
79
|
+
Writes replace, inside one transaction:
|
|
80
|
+
|
|
81
|
+
- unpartitioned: every row is deleted, then the data inserted;
|
|
82
|
+
- a time partition: the rows inside its half-open bounds
|
|
83
|
+
(`column >= start AND column < end`) are deleted, then the data inserted;
|
|
84
|
+
- any other partition: the rows equal to its id are deleted;
|
|
85
|
+
- a window: each partition's rows are deleted, then the whole batch is inserted
|
|
86
|
+
once.
|
|
87
|
+
|
|
88
|
+
A failed insert rolls the delete back, so a partition is never left empty.
|
|
89
|
+
Inserts go in batches of 1000 rows. Reads return rows as `list[dict]`; reading
|
|
90
|
+
a table that was never written raises `DataNotFoundError`.
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# interloper-sql
|
|
2
|
+
|
|
3
|
+
A SQL database destination for interloper: `SQLDestination`, a
|
|
4
|
+
`DatabaseDestination` over SQLAlchemy Core, and the `SQLConnection` that holds
|
|
5
|
+
the database URL.
|
|
6
|
+
|
|
7
|
+
One backend serves every SQLAlchemy dialect whose driver is installed.
|
|
8
|
+
PostgreSQL works out of the box (the package depends on `psycopg[binary]`);
|
|
9
|
+
MySQL, SQLite, Redshift, SQL Server and the rest need their own driver.
|
|
10
|
+
|
|
11
|
+
## Setup
|
|
12
|
+
|
|
13
|
+
The connection is a single SQLAlchemy URL:
|
|
14
|
+
|
|
15
|
+
| Database | URL | Driver to install |
|
|
16
|
+
|----------|-----|-------------------|
|
|
17
|
+
| PostgreSQL | `postgresql+psycopg://user:password@host:5432/db` | none, ships with the package |
|
|
18
|
+
| MySQL / MariaDB | `mysql+pymysql://user:password@host:3306/db` | `pymysql` |
|
|
19
|
+
| SQL Server | `mssql+pyodbc://user:password@host/db?driver=ODBC+Driver+18+for+SQL+Server` | `pyodbc` (and the ODBC driver) |
|
|
20
|
+
| Redshift | `redshift+redshift_connector://user:password@host:5439/db` | `sqlalchemy-redshift`, `redshift_connector` |
|
|
21
|
+
| SQLite | `sqlite:///path/to/file.db` | none, in the standard library |
|
|
22
|
+
|
|
23
|
+
Install the driver next to interloper (`uv pip install pymysql`, or a custom
|
|
24
|
+
image built from the slim variant). Special characters in the password must be
|
|
25
|
+
URL-encoded.
|
|
26
|
+
|
|
27
|
+
The URL carries the password, so it is a secret field: encrypted at rest in a
|
|
28
|
+
deployed instance and masked in the UI. It also loads from the environment
|
|
29
|
+
(`SQL_URL`), so `SQLConnection()` works with no arguments.
|
|
30
|
+
|
|
31
|
+
The connection's check runs `SELECT 1`.
|
|
32
|
+
|
|
33
|
+
## Usage
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
import interloper as il
|
|
37
|
+
from interloper_sql import SQLConnection, SQLDestination
|
|
38
|
+
|
|
39
|
+
warehouse = SQLDestination(
|
|
40
|
+
connection=SQLConnection(url="postgresql+psycopg://user:password@host:5432/analytics"),
|
|
41
|
+
default_dataset="marketing",
|
|
42
|
+
)
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
In a deployed instance you configure this through the UI instead: add a SQL
|
|
46
|
+
database connection, then a SQL database destination using it.
|
|
47
|
+
|
|
48
|
+
## Datasets are schemas
|
|
49
|
+
|
|
50
|
+
An asset's `dataset` is the SQL schema its table lives in. Without one, the
|
|
51
|
+
destination's `default_dataset` applies; without either, the table goes in
|
|
52
|
+
the connection's default schema (`public` on PostgreSQL, the database itself on
|
|
53
|
+
MySQL). A named schema is created on first write when the database supports
|
|
54
|
+
schemas and it does not exist yet.
|
|
55
|
+
|
|
56
|
+
## Tables
|
|
57
|
+
|
|
58
|
+
A table is created on first write from the asset's effective schema (declared,
|
|
59
|
+
or inferred during conform): integers are `BIGINT`, floats `DOUBLE`, decimals
|
|
60
|
+
`NUMERIC(38, 9)`, datetimes timezone-aware timestamps, strings `TEXT`, and
|
|
61
|
+
nested or repeated fields `JSON`. An existing table is never altered: columns
|
|
62
|
+
it lacks are dropped from the write with a warning, and NaN or infinite floats
|
|
63
|
+
are written as `NULL`.
|
|
64
|
+
|
|
65
|
+
## Partitions
|
|
66
|
+
|
|
67
|
+
Writes replace, inside one transaction:
|
|
68
|
+
|
|
69
|
+
- unpartitioned: every row is deleted, then the data inserted;
|
|
70
|
+
- a time partition: the rows inside its half-open bounds
|
|
71
|
+
(`column >= start AND column < end`) are deleted, then the data inserted;
|
|
72
|
+
- any other partition: the rows equal to its id are deleted;
|
|
73
|
+
- a window: each partition's rows are deleted, then the whole batch is inserted
|
|
74
|
+
once.
|
|
75
|
+
|
|
76
|
+
A failed insert rolls the delete back, so a partition is never left empty.
|
|
77
|
+
Inserts go in batches of 1000 rows. Reads return rows as `list[dict]`; reading
|
|
78
|
+
a table that was never written raises `DataNotFoundError`.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "interloper-sql"
|
|
3
|
+
version = "0.94.2"
|
|
4
|
+
description = "Interloper SQL integration: a SQLAlchemy-backed destination for PostgreSQL and other SQL databases"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"interloper-core",
|
|
9
|
+
"psycopg[binary]>=3.1",
|
|
10
|
+
"sqlalchemy>=2.0",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[[project.authors]]
|
|
14
|
+
name = "Guillaume Onfroy"
|
|
15
|
+
email = "guillaume@digitlcloud.com"
|
|
16
|
+
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
sql = "interloper_sql"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[tool.uv.sources.interloper-core]
|
|
25
|
+
workspace = true
|
|
26
|
+
|
|
27
|
+
[tool.ruff]
|
|
28
|
+
line-length = 120
|
|
29
|
+
|
|
30
|
+
[tool.ruff.lint]
|
|
31
|
+
preview = true
|
|
32
|
+
extend-select = [
|
|
33
|
+
"E",
|
|
34
|
+
"I",
|
|
35
|
+
"UP",
|
|
36
|
+
"ANN001",
|
|
37
|
+
"ANN201",
|
|
38
|
+
"ANN202",
|
|
39
|
+
"DOC",
|
|
40
|
+
"D",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[tool.ruff.lint.pydocstyle]
|
|
44
|
+
convention = "google"
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint.per-file-ignores]
|
|
47
|
+
"__init__.py" = [
|
|
48
|
+
"F401",
|
|
49
|
+
"F403",
|
|
50
|
+
]
|
|
51
|
+
"tests/**" = [
|
|
52
|
+
"ANN",
|
|
53
|
+
"F811",
|
|
54
|
+
"D101",
|
|
55
|
+
"D102",
|
|
56
|
+
"D103",
|
|
57
|
+
"D104",
|
|
58
|
+
"RUF069",
|
|
59
|
+
"PLW0108",
|
|
60
|
+
]
|
|
@@ -3,22 +3,22 @@
|
|
|
3
3
|
# ###############
|
|
4
4
|
[project]
|
|
5
5
|
name = "interloper-sql"
|
|
6
|
-
version = "0.2
|
|
7
|
-
description = "Interloper SQLAlchemy
|
|
6
|
+
version = "0.94.2"
|
|
7
|
+
description = "Interloper SQL integration: a SQLAlchemy-backed destination for PostgreSQL and other SQL databases"
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
dependencies = [
|
|
12
|
-
"sqlalchemy>=2.0",
|
|
13
12
|
"interloper-core",
|
|
13
|
+
"psycopg[binary]>=3.1",
|
|
14
|
+
"sqlalchemy>=2.0",
|
|
14
15
|
]
|
|
15
16
|
|
|
16
|
-
[project.
|
|
17
|
-
|
|
18
|
-
mysql = ["pymysql>=1.1"]
|
|
17
|
+
[project.entry-points."interloper.components"]
|
|
18
|
+
sql = "interloper_sql"
|
|
19
19
|
|
|
20
20
|
[build-system]
|
|
21
|
-
requires = ["uv_build>=0.9
|
|
21
|
+
requires = ["uv_build>=0.12.9,<0.13"]
|
|
22
22
|
build-backend = "uv_build"
|
|
23
23
|
|
|
24
24
|
[tool.uv.sources]
|
|
@@ -31,17 +31,12 @@ interloper-core = { workspace = true }
|
|
|
31
31
|
line-length = 120
|
|
32
32
|
|
|
33
33
|
[tool.ruff.lint]
|
|
34
|
-
|
|
34
|
+
preview = true
|
|
35
|
+
extend-select = ["E", "I", "UP", "ANN001", "ANN201", "ANN202", "DOC", "D"]
|
|
36
|
+
|
|
37
|
+
[tool.ruff.lint.pydocstyle]
|
|
38
|
+
convention = "google"
|
|
35
39
|
|
|
36
40
|
[tool.ruff.lint.per-file-ignores]
|
|
37
41
|
"__init__.py" = ["F401", "F403"]
|
|
38
|
-
"tests/**" = ["ANN", "F811"]
|
|
39
|
-
|
|
40
|
-
# ###############
|
|
41
|
-
# PYRIGHT
|
|
42
|
-
# ###############
|
|
43
|
-
[tool.pyright]
|
|
44
|
-
include = ["src"]
|
|
45
|
-
typeCheckingMode = "basic"
|
|
46
|
-
reportMissingParameterType = true
|
|
47
|
-
ignore = ["libs/**", "tests/**", "scripts/**"]
|
|
42
|
+
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""SQL connection resource holding a SQLAlchemy database URL."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from functools import cached_property
|
|
6
|
+
|
|
7
|
+
import sqlalchemy
|
|
8
|
+
from interloper.connection import Connection, connection
|
|
9
|
+
from interloper.resource.fields import SecretField, fetch_field_provider
|
|
10
|
+
from pydantic_settings import SettingsConfigDict
|
|
11
|
+
from sqlalchemy.engine import Engine
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@connection(
|
|
15
|
+
key="sql_connection",
|
|
16
|
+
name="SQL database",
|
|
17
|
+
icon="icon:sql",
|
|
18
|
+
tags=["Database"],
|
|
19
|
+
)
|
|
20
|
+
class SQLConnection(Connection):
|
|
21
|
+
"""Connection resource holding the URL of a SQL database.
|
|
22
|
+
|
|
23
|
+
Any SQLAlchemy dialect whose driver is installed works; PostgreSQL through
|
|
24
|
+
psycopg ships with the package.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
model_config = SettingsConfigDict(env_prefix="sql_")
|
|
28
|
+
|
|
29
|
+
url: str = SecretField(
|
|
30
|
+
label="Database URL",
|
|
31
|
+
description="SQLAlchemy URL, e.g. postgresql+psycopg://user:password@host:5432/db",
|
|
32
|
+
info=(
|
|
33
|
+
"The URL carries the password, so it is stored as a secret. PostgreSQL works out of the box "
|
|
34
|
+
"(postgresql+psycopg://); other databases need their SQLAlchemy driver installed, e.g. "
|
|
35
|
+
"mysql+pymysql:// or mssql+pyodbc://."
|
|
36
|
+
),
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
@cached_property
|
|
40
|
+
def engine(self) -> Engine:
|
|
41
|
+
"""The engine every statement goes through.
|
|
42
|
+
|
|
43
|
+
``pool_pre_ping`` tests a pooled connection before handing it out, so
|
|
44
|
+
a connection the server dropped between runs is replaced instead of
|
|
45
|
+
failing the next write.
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
The engine, cached per connection instance.
|
|
49
|
+
"""
|
|
50
|
+
return sqlalchemy.create_engine(self.url, pool_pre_ping=True)
|
|
51
|
+
|
|
52
|
+
@fetch_field_provider
|
|
53
|
+
def schemas(self) -> list[dict[str, str]]:
|
|
54
|
+
"""List the schemas of the database.
|
|
55
|
+
|
|
56
|
+
Not wired to a field yet: the destination's ``default_dataset`` is
|
|
57
|
+
optional, and a ``FetchField`` is a required pick. It is kept for a
|
|
58
|
+
future schema picker.
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
Schema options with ``name``, sorted case-insensitively.
|
|
62
|
+
"""
|
|
63
|
+
names = sqlalchemy.inspect(self.engine).get_schema_names()
|
|
64
|
+
return sorted(({"name": name} for name in names), key=lambda s: s["name"].lower())
|
|
65
|
+
|
|
66
|
+
def check(self) -> bool:
|
|
67
|
+
"""Prove the URL works by running ``SELECT 1``.
|
|
68
|
+
|
|
69
|
+
Returns:
|
|
70
|
+
True; an unreachable database or a rejected login raises instead.
|
|
71
|
+
"""
|
|
72
|
+
with self.engine.connect() as conn:
|
|
73
|
+
conn.execute(sqlalchemy.text("SELECT 1"))
|
|
74
|
+
return True
|
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
"""SQL database destination implementation over SQLAlchemy Core."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
import math
|
|
7
|
+
import threading
|
|
8
|
+
import warnings
|
|
9
|
+
from collections.abc import Iterator, Sequence
|
|
10
|
+
from contextlib import contextmanager
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from interloper.destination import IOContext, destination
|
|
14
|
+
from interloper.destination.database import DatabaseDestination, PartitionFilter
|
|
15
|
+
from interloper.errors import DataNotFoundError
|
|
16
|
+
from interloper.representation import Representation
|
|
17
|
+
from interloper.resource.fields import InputField
|
|
18
|
+
from interloper.schema import FieldSpec
|
|
19
|
+
from pydantic import PrivateAttr
|
|
20
|
+
from sqlalchemy import Column, MetaData, String, Table, and_, cast, func, inspect, select
|
|
21
|
+
from sqlalchemy.engine import Connection as SAConnection
|
|
22
|
+
from sqlalchemy.schema import CreateSchema
|
|
23
|
+
from sqlalchemy.sql.elements import ColumnElement
|
|
24
|
+
|
|
25
|
+
from interloper_sql.connection import SQLConnection
|
|
26
|
+
from interloper_sql.types import column_type
|
|
27
|
+
|
|
28
|
+
_BATCH_SIZE = 1000
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@destination(
|
|
32
|
+
key="sql_destination",
|
|
33
|
+
name="SQL database",
|
|
34
|
+
icon="icon:sql",
|
|
35
|
+
tags=["Database"],
|
|
36
|
+
)
|
|
37
|
+
class SQLDestination(DatabaseDestination):
|
|
38
|
+
"""SQL database destination.
|
|
39
|
+
|
|
40
|
+
Each asset is a table; its dataset is the SQL schema holding it. A table
|
|
41
|
+
is created from the effective schema on first write and never altered
|
|
42
|
+
afterwards.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
connection: SQLConnection
|
|
46
|
+
|
|
47
|
+
default_dataset: str | None = InputField(default=None, description="Default schema for assets without a dataset")
|
|
48
|
+
|
|
49
|
+
# Keyed by thread because one instance serves every asset of a run, and
|
|
50
|
+
# each write runs whole on its own worker thread: a single slot would let
|
|
51
|
+
# concurrent writes swap each other's transaction. A plain dict, unlike
|
|
52
|
+
# threading.local, still deep-copies and pickles.
|
|
53
|
+
_held: dict[int, SAConnection] = PrivateAttr(default_factory=dict)
|
|
54
|
+
|
|
55
|
+
# -- Helpers ---------------------------------------------------------------
|
|
56
|
+
|
|
57
|
+
def _resolve_dataset(self, dataset: str | None) -> str | None:
|
|
58
|
+
"""Return the schema to use.
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
dataset: The asset's dataset, or ``None`` to fall back to ``default_dataset``.
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
The schema name, or ``None`` for the connection's default schema,
|
|
65
|
+
which every SQL database has.
|
|
66
|
+
"""
|
|
67
|
+
return dataset or self.default_dataset
|
|
68
|
+
|
|
69
|
+
@contextmanager
|
|
70
|
+
def _begin(self) -> Iterator[SAConnection]:
|
|
71
|
+
"""Yield the connection held by :meth:`transaction`, or a fresh one in its own transaction.
|
|
72
|
+
|
|
73
|
+
Yields:
|
|
74
|
+
A connection inside a transaction.
|
|
75
|
+
"""
|
|
76
|
+
held = self._held.get(threading.get_ident())
|
|
77
|
+
if held is not None:
|
|
78
|
+
yield held
|
|
79
|
+
return
|
|
80
|
+
with self.connection.engine.begin() as conn:
|
|
81
|
+
yield conn
|
|
82
|
+
|
|
83
|
+
@staticmethod
|
|
84
|
+
def _table(table: str, dataset: str | None, conn: SAConnection) -> Table | None:
|
|
85
|
+
"""Reflect an existing table.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
table: Table name.
|
|
89
|
+
dataset: The resolved schema, or ``None`` for the default schema.
|
|
90
|
+
conn: The connection to reflect through.
|
|
91
|
+
|
|
92
|
+
Returns:
|
|
93
|
+
The reflected table, or ``None`` if it does not exist.
|
|
94
|
+
"""
|
|
95
|
+
if not inspect(conn).has_table(table, schema=dataset):
|
|
96
|
+
return None
|
|
97
|
+
return Table(table, MetaData(), schema=dataset, autoload_with=conn)
|
|
98
|
+
|
|
99
|
+
@staticmethod
|
|
100
|
+
def _needs_schema(dataset: str | None, conn: SAConnection) -> bool:
|
|
101
|
+
"""Whether the schema must be created before a table can go in it.
|
|
102
|
+
|
|
103
|
+
Args:
|
|
104
|
+
dataset: The resolved schema, or ``None`` for the default schema.
|
|
105
|
+
conn: The connection to inspect through.
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
True when a schema is named, the dialect has schemas, and it does not exist yet.
|
|
109
|
+
"""
|
|
110
|
+
if dataset is None or not getattr(conn.dialect, "supports_schemas", False):
|
|
111
|
+
return False
|
|
112
|
+
return not inspect(conn).has_schema(dataset)
|
|
113
|
+
|
|
114
|
+
def _create_table(self, table: str, dataset: str | None, specs: Sequence[FieldSpec], conn: SAConnection) -> Table:
|
|
115
|
+
"""Create a table from field specs, creating its schema first when missing.
|
|
116
|
+
|
|
117
|
+
Args:
|
|
118
|
+
table: Table name.
|
|
119
|
+
dataset: The resolved schema, or ``None`` for the default schema.
|
|
120
|
+
specs: The field specs the columns are built from.
|
|
121
|
+
conn: The connection to create through.
|
|
122
|
+
|
|
123
|
+
Returns:
|
|
124
|
+
The created table.
|
|
125
|
+
"""
|
|
126
|
+
if dataset is not None and self._needs_schema(dataset, conn):
|
|
127
|
+
conn.execute(CreateSchema(dataset, if_not_exists=True))
|
|
128
|
+
columns = [Column(spec.name, column_type(spec), nullable=True) for spec in specs]
|
|
129
|
+
sql_table = Table(table, MetaData(), *columns, schema=dataset)
|
|
130
|
+
sql_table.create(conn, checkfirst=True)
|
|
131
|
+
return sql_table
|
|
132
|
+
|
|
133
|
+
@staticmethod
|
|
134
|
+
def _ref(table: str, dataset: str | None) -> str:
|
|
135
|
+
"""Name a table for messages.
|
|
136
|
+
|
|
137
|
+
Args:
|
|
138
|
+
table: Table name.
|
|
139
|
+
dataset: The resolved schema, or ``None`` for the default schema.
|
|
140
|
+
|
|
141
|
+
Returns:
|
|
142
|
+
``schema.table``, or ``table`` alone in the default schema.
|
|
143
|
+
"""
|
|
144
|
+
return f"{dataset}.{table}" if dataset else table
|
|
145
|
+
|
|
146
|
+
def _not_found(self, table: str, dataset: str | None) -> str:
|
|
147
|
+
"""Word the error a read raises on a table that does not exist.
|
|
148
|
+
|
|
149
|
+
Args:
|
|
150
|
+
table: Table name.
|
|
151
|
+
dataset: The resolved schema, or ``None`` for the default schema.
|
|
152
|
+
|
|
153
|
+
Returns:
|
|
154
|
+
The error message.
|
|
155
|
+
"""
|
|
156
|
+
return f"Table '{self._ref(table, dataset)}' does not exist. Has the asset been materialized?"
|
|
157
|
+
|
|
158
|
+
# -- DatabaseDestination hooks ---------------------------------------------
|
|
159
|
+
|
|
160
|
+
@contextmanager
|
|
161
|
+
def transaction(self) -> Iterator[None]:
|
|
162
|
+
"""Run a delete and its insert in one database transaction.
|
|
163
|
+
|
|
164
|
+
Yields:
|
|
165
|
+
``None``; the hooks called inside the block share its connection.
|
|
166
|
+
"""
|
|
167
|
+
with self.connection.engine.begin() as conn:
|
|
168
|
+
self._held[threading.get_ident()] = conn
|
|
169
|
+
try:
|
|
170
|
+
yield
|
|
171
|
+
finally:
|
|
172
|
+
del self._held[threading.get_ident()]
|
|
173
|
+
|
|
174
|
+
def insert(self, table: str, dataset: str | None, data: Any, context: IOContext) -> None:
|
|
175
|
+
"""Insert rows, creating the table from the effective schema on first write.
|
|
176
|
+
|
|
177
|
+
Without a schema on the context, one is inferred from the data so the
|
|
178
|
+
table is still typed. Columns the table does not have are dropped with
|
|
179
|
+
a warning, and non-finite floats are written as ``NULL``.
|
|
180
|
+
|
|
181
|
+
Args:
|
|
182
|
+
table: Target table name.
|
|
183
|
+
dataset: The asset's schema, or ``None`` for ``default_dataset``.
|
|
184
|
+
data: The data in its native representation.
|
|
185
|
+
context: IO context carrying the effective schema.
|
|
186
|
+
"""
|
|
187
|
+
dataset = self._resolve_dataset(dataset)
|
|
188
|
+
view = Representation.of(data)
|
|
189
|
+
with self._begin() as conn:
|
|
190
|
+
sql_table = self._table(table, dataset, conn)
|
|
191
|
+
if sql_table is None:
|
|
192
|
+
schema = context.schema if context.schema is not None else view.infer()
|
|
193
|
+
sql_table = self._create_table(table, dataset, schema.field_specs(), conn)
|
|
194
|
+
rows = _align(view.records, [c.name for c in sql_table.columns], self._ref(table, dataset))
|
|
195
|
+
for start in range(0, len(rows), _BATCH_SIZE):
|
|
196
|
+
conn.execute(sql_table.insert(), rows[start : start + _BATCH_SIZE])
|
|
197
|
+
|
|
198
|
+
def delete(self, table: str, dataset: str | None, where: PartitionFilter | None) -> None:
|
|
199
|
+
"""Delete the rows a filter selects, or every row.
|
|
200
|
+
|
|
201
|
+
A table that does not exist has nothing to delete.
|
|
202
|
+
|
|
203
|
+
Args:
|
|
204
|
+
table: Target table name.
|
|
205
|
+
dataset: The asset's schema, or ``None`` for ``default_dataset``.
|
|
206
|
+
where: The rows to delete; ``None`` for the whole table.
|
|
207
|
+
"""
|
|
208
|
+
dataset = self._resolve_dataset(dataset)
|
|
209
|
+
with self._begin() as conn:
|
|
210
|
+
sql_table = self._table(table, dataset, conn)
|
|
211
|
+
if sql_table is None:
|
|
212
|
+
return
|
|
213
|
+
statement = sql_table.delete()
|
|
214
|
+
if where is not None:
|
|
215
|
+
statement = statement.where(_predicate(sql_table, where))
|
|
216
|
+
conn.execute(statement)
|
|
217
|
+
|
|
218
|
+
def select(self, table: str, dataset: str | None, where: PartitionFilter | None) -> list[dict[str, Any]]:
|
|
219
|
+
"""Select the rows a filter selects, or every row, as records.
|
|
220
|
+
|
|
221
|
+
Args:
|
|
222
|
+
table: Target table name.
|
|
223
|
+
dataset: The asset's schema, or ``None`` for ``default_dataset``.
|
|
224
|
+
where: The rows to select; ``None`` for the whole table.
|
|
225
|
+
|
|
226
|
+
Returns:
|
|
227
|
+
The selected rows.
|
|
228
|
+
|
|
229
|
+
Raises:
|
|
230
|
+
DataNotFoundError: If the table does not exist yet.
|
|
231
|
+
"""
|
|
232
|
+
dataset = self._resolve_dataset(dataset)
|
|
233
|
+
with self._begin() as conn:
|
|
234
|
+
sql_table = self._table(table, dataset, conn)
|
|
235
|
+
if sql_table is None:
|
|
236
|
+
raise DataNotFoundError(self._not_found(table, dataset))
|
|
237
|
+
statement = select(sql_table)
|
|
238
|
+
if where is not None:
|
|
239
|
+
statement = statement.where(_predicate(sql_table, where))
|
|
240
|
+
return [dict(row._mapping) for row in conn.execute(statement)]
|
|
241
|
+
|
|
242
|
+
def count(self, table: str, dataset: str | None, column: str) -> dict[str, int]:
|
|
243
|
+
"""Return row counts grouped by a column's values, cast to strings.
|
|
244
|
+
|
|
245
|
+
Args:
|
|
246
|
+
table: Target table name.
|
|
247
|
+
dataset: The asset's schema, or ``None`` for ``default_dataset``.
|
|
248
|
+
column: Column to group by.
|
|
249
|
+
|
|
250
|
+
Returns:
|
|
251
|
+
Mapping from partition value (as string) to row count.
|
|
252
|
+
|
|
253
|
+
Raises:
|
|
254
|
+
DataNotFoundError: If the table does not exist.
|
|
255
|
+
"""
|
|
256
|
+
dataset = self._resolve_dataset(dataset)
|
|
257
|
+
with self._begin() as conn:
|
|
258
|
+
sql_table = self._table(table, dataset, conn)
|
|
259
|
+
if sql_table is None:
|
|
260
|
+
raise DataNotFoundError(self._not_found(table, dataset))
|
|
261
|
+
value = cast(sql_table.c[column], String).label("partition_value")
|
|
262
|
+
statement = select(value, func.count().label("cnt")).group_by(value)
|
|
263
|
+
return {row.partition_value: row.cnt for row in conn.execute(statement)}
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
# -- Utility functions ---------------------------------------------------------
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _align(records: list[dict[str, Any]], columns: list[str], ref: str) -> list[dict[str, Any]]:
|
|
270
|
+
"""Shape records to the table's columns for one executemany.
|
|
271
|
+
|
|
272
|
+
Every row gets the same keys, the table columns any record carries, so
|
|
273
|
+
a column no record mentions keeps its server default.
|
|
274
|
+
|
|
275
|
+
Args:
|
|
276
|
+
records: The rows to write.
|
|
277
|
+
columns: The table's column names, in table order.
|
|
278
|
+
ref: The table's name, for the warning.
|
|
279
|
+
|
|
280
|
+
Returns:
|
|
281
|
+
The aligned rows, non-finite floats replaced by ``None``.
|
|
282
|
+
"""
|
|
283
|
+
present = dict.fromkeys(key for record in records for key in record)
|
|
284
|
+
extras = [str(key) for key in present if key not in columns]
|
|
285
|
+
if extras:
|
|
286
|
+
warnings.warn(
|
|
287
|
+
f"Columns {extras} are not in the schema for '{ref}' and will not be written.",
|
|
288
|
+
UserWarning,
|
|
289
|
+
stacklevel=3,
|
|
290
|
+
)
|
|
291
|
+
keys = [name for name in columns if name in present]
|
|
292
|
+
return [{key: _finite(record.get(key)) for key in keys} for record in records]
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _finite(value: Any) -> Any:
|
|
296
|
+
"""Replace a NaN or infinite float with ``None``, which no SQL column type rejects.
|
|
297
|
+
|
|
298
|
+
Args:
|
|
299
|
+
value: A cell value.
|
|
300
|
+
|
|
301
|
+
Returns:
|
|
302
|
+
``None`` for a non-finite float, else the value unchanged.
|
|
303
|
+
"""
|
|
304
|
+
if isinstance(value, float) and not math.isfinite(value):
|
|
305
|
+
return None
|
|
306
|
+
return value
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _predicate(table: Table, where: PartitionFilter) -> ColumnElement[bool]:
|
|
310
|
+
"""Render a partition filter against a reflected table.
|
|
311
|
+
|
|
312
|
+
Args:
|
|
313
|
+
table: The reflected table.
|
|
314
|
+
where: The filter: equality on ``value``, or half-open ``bounds``.
|
|
315
|
+
|
|
316
|
+
Returns:
|
|
317
|
+
The ``WHERE`` clause.
|
|
318
|
+
"""
|
|
319
|
+
column = table.c[where.column]
|
|
320
|
+
if where.bounds is not None:
|
|
321
|
+
start, end = where.bounds
|
|
322
|
+
return and_(column >= _coerce(column, start), column < _coerce(column, end))
|
|
323
|
+
return column == _coerce(column, where.value)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _coerce(column: Column[Any], value: Any) -> Any:
|
|
327
|
+
"""Coerce a filter value to the Python type a date or timestamp column binds.
|
|
328
|
+
|
|
329
|
+
A partition id arrives as a string (``"2024-01-01"``) and a day's bounds
|
|
330
|
+
as dates; bound as-is against a ``DATE`` or ``TIMESTAMP`` column, they
|
|
331
|
+
would be sent as ``VARCHAR`` or rejected by the dialect's type.
|
|
332
|
+
|
|
333
|
+
Args:
|
|
334
|
+
column: The column the value is compared with.
|
|
335
|
+
value: The filter value.
|
|
336
|
+
|
|
337
|
+
Returns:
|
|
338
|
+
The value as a ``date`` or ``datetime`` when the column holds one, else unchanged.
|
|
339
|
+
"""
|
|
340
|
+
try:
|
|
341
|
+
python_type = column.type.python_type
|
|
342
|
+
except NotImplementedError:
|
|
343
|
+
return value
|
|
344
|
+
if python_type is datetime.datetime:
|
|
345
|
+
if isinstance(value, str):
|
|
346
|
+
return datetime.datetime.fromisoformat(value)
|
|
347
|
+
if isinstance(value, datetime.date) and not isinstance(value, datetime.datetime):
|
|
348
|
+
return datetime.datetime.combine(value, datetime.time())
|
|
349
|
+
elif python_type is datetime.date and isinstance(value, str):
|
|
350
|
+
return datetime.date.fromisoformat(value)
|
|
351
|
+
return value
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""SQLAlchemy's view of interloper's field types: one table, read in order."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
from decimal import Decimal
|
|
7
|
+
|
|
8
|
+
from interloper.schema import FieldSpec
|
|
9
|
+
from sqlalchemy import JSON, BigInteger, Boolean, Date, DateTime, Double, LargeBinary, Numeric, Text
|
|
10
|
+
from sqlalchemy.types import TypeEngine
|
|
11
|
+
|
|
12
|
+
# Ordered: the first base class that matches wins, so bool (a subclass of int)
|
|
13
|
+
# and datetime (a subclass of date) must come before their parents.
|
|
14
|
+
_PYTHON_TO_SQL: dict[type, TypeEngine] = {
|
|
15
|
+
bool: Boolean(),
|
|
16
|
+
int: BigInteger(),
|
|
17
|
+
float: Double(),
|
|
18
|
+
Decimal: Numeric(38, 9),
|
|
19
|
+
datetime.datetime: DateTime(timezone=True),
|
|
20
|
+
datetime.date: Date(),
|
|
21
|
+
bytes: LargeBinary(),
|
|
22
|
+
str: Text(),
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def column_type(spec: FieldSpec) -> TypeEngine:
|
|
27
|
+
"""Return the SQLAlchemy column type for a field spec.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
spec: The field spec to map, from :meth:`Schema.field_specs`. A nested
|
|
31
|
+
model or a repeated field is ``JSON``; a type that is not a class
|
|
32
|
+
(``typing.Any``) or matches no row of the table is ``Text``.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
The column type, rendered by each dialect in its own vocabulary.
|
|
36
|
+
"""
|
|
37
|
+
if spec.fields is not None or spec.repeated:
|
|
38
|
+
return JSON()
|
|
39
|
+
if isinstance(spec.type, type):
|
|
40
|
+
for base, sql_type in _PYTHON_TO_SQL.items():
|
|
41
|
+
if issubclass(spec.type, base):
|
|
42
|
+
return sql_type
|
|
43
|
+
return Text()
|
interloper_sql-0.2.0rc1/PKG-INFO
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.3
|
|
2
|
-
Name: interloper-sql
|
|
3
|
-
Version: 0.2.0rc1
|
|
4
|
-
Summary: Interloper SQLAlchemy IO managers
|
|
5
|
-
Author: Guillaume Onfroy
|
|
6
|
-
Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
|
|
7
|
-
Requires-Dist: sqlalchemy>=2.0
|
|
8
|
-
Requires-Dist: interloper-core
|
|
9
|
-
Requires-Dist: pymysql>=1.1 ; extra == 'mysql'
|
|
10
|
-
Requires-Dist: psycopg2-binary>=2.9 ; extra == 'postgres'
|
|
11
|
-
Requires-Python: >=3.10
|
|
12
|
-
Provides-Extra: mysql
|
|
13
|
-
Provides-Extra: postgres
|
|
14
|
-
Description-Content-Type: text/markdown
|
|
15
|
-
|
|
16
|
-
# interloper-sql
|
|
17
|
-
|
|
18
|
-
SQLAlchemy IO managers for the Interloper framework.
|
|
@@ -1,13 +0,0 @@
|
|
|
1
|
-
"""SQL IO managers for reading and writing to databases via SQLAlchemy."""
|
|
2
|
-
|
|
3
|
-
from interloper_sql.io.base import SqlIO
|
|
4
|
-
from interloper_sql.io.mysql import MySQLIO
|
|
5
|
-
from interloper_sql.io.postgres import PostgresIO
|
|
6
|
-
from interloper_sql.io.sqlite import SqliteIO
|
|
7
|
-
|
|
8
|
-
__all__ = [
|
|
9
|
-
"MySQLIO",
|
|
10
|
-
"PostgresIO",
|
|
11
|
-
"SqlIO",
|
|
12
|
-
"SqliteIO",
|
|
13
|
-
]
|
|
@@ -1,281 +0,0 @@
|
|
|
1
|
-
"""SQLAlchemy IO implementation."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
from collections.abc import Iterator
|
|
6
|
-
from contextlib import contextmanager
|
|
7
|
-
from typing import TYPE_CHECKING, Any
|
|
8
|
-
|
|
9
|
-
from interloper.errors import TableNotFoundError
|
|
10
|
-
from interloper.io.database import DatabaseIO, WriteDisposition
|
|
11
|
-
from sqlalchemy import Column, MetaData, Table, create_engine
|
|
12
|
-
from sqlalchemy import inspect as sa_inspect
|
|
13
|
-
|
|
14
|
-
if TYPE_CHECKING:
|
|
15
|
-
from interloper.io.adapter import DataAdapter
|
|
16
|
-
from sqlalchemy.engine import URL, Connection, Engine
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
def _infer_sa_type(value: Any) -> Any:
|
|
20
|
-
"""Infer a SQLAlchemy column type from a Python value.
|
|
21
|
-
|
|
22
|
-
Args:
|
|
23
|
-
value: A sample Python value used to determine the column type.
|
|
24
|
-
|
|
25
|
-
Returns:
|
|
26
|
-
A SQLAlchemy type instance.
|
|
27
|
-
"""
|
|
28
|
-
import datetime
|
|
29
|
-
from decimal import Decimal
|
|
30
|
-
|
|
31
|
-
from sqlalchemy import BigInteger, Boolean, Date, DateTime, Float, LargeBinary, Numeric, Text
|
|
32
|
-
|
|
33
|
-
if isinstance(value, bool):
|
|
34
|
-
return Boolean()
|
|
35
|
-
if isinstance(value, int):
|
|
36
|
-
return BigInteger()
|
|
37
|
-
if isinstance(value, float):
|
|
38
|
-
return Float()
|
|
39
|
-
if isinstance(value, Decimal):
|
|
40
|
-
return Numeric()
|
|
41
|
-
if isinstance(value, datetime.datetime):
|
|
42
|
-
return DateTime()
|
|
43
|
-
if isinstance(value, datetime.date):
|
|
44
|
-
return Date()
|
|
45
|
-
if isinstance(value, bytes):
|
|
46
|
-
return LargeBinary()
|
|
47
|
-
return Text()
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
class SqlIO(DatabaseIO):
|
|
51
|
-
"""Base IO implementation for SQL databases via SQLAlchemy.
|
|
52
|
-
|
|
53
|
-
Provides connection management, transactional writes, table reflection, and
|
|
54
|
-
automatic table creation. Not intended for direct instantiation — use a
|
|
55
|
-
dialect subclass (:class:`PostgresIO`, :class:`MySQLIO`, :class:`SqliteIO`)
|
|
56
|
-
which accepts explicit connection parameters and implements ``to_spec``.
|
|
57
|
-
|
|
58
|
-
The IO is fully stateless with respect to table identity — the table name
|
|
59
|
-
and schema are passed through from the asset context on every call, so a
|
|
60
|
-
single instance can safely serve multiple assets.
|
|
61
|
-
|
|
62
|
-
Args:
|
|
63
|
-
url: SQLAlchemy connection URL or :class:`~sqlalchemy.engine.URL` object
|
|
64
|
-
(constructed by dialect subclasses).
|
|
65
|
-
write_disposition: Controls whether existing rows are deleted before
|
|
66
|
-
writing. Defaults to :attr:`WriteDisposition.REPLACE`.
|
|
67
|
-
chunk_size: Number of rows per insert batch
|
|
68
|
-
adapter: Optional data adapter for type conversion
|
|
69
|
-
"""
|
|
70
|
-
|
|
71
|
-
def __init__(
|
|
72
|
-
self,
|
|
73
|
-
url: str | URL,
|
|
74
|
-
write_disposition: WriteDisposition = WriteDisposition.REPLACE,
|
|
75
|
-
chunk_size: int = 1000,
|
|
76
|
-
adapter: DataAdapter | str | None = None,
|
|
77
|
-
) -> None:
|
|
78
|
-
super().__init__(write_disposition, chunk_size, adapter)
|
|
79
|
-
self._engine: Engine = create_engine(url)
|
|
80
|
-
self._table_cache: dict[tuple[str, str | None], Table] = {}
|
|
81
|
-
self._conn: Connection | None = None
|
|
82
|
-
|
|
83
|
-
# ------------------------------------------------------------------
|
|
84
|
-
# Table helpers
|
|
85
|
-
# ------------------------------------------------------------------
|
|
86
|
-
|
|
87
|
-
def _resolve_table(self, table: str, schema: str | None) -> Table | None:
|
|
88
|
-
"""Reflect and cache the SQLAlchemy Table, or return None if it doesn't exist."""
|
|
89
|
-
key = (table, schema)
|
|
90
|
-
if key not in self._table_cache:
|
|
91
|
-
if sa_inspect(self._engine).has_table(table, schema=schema):
|
|
92
|
-
metadata = MetaData(schema=schema)
|
|
93
|
-
self._table_cache[key] = Table(table, metadata, autoload_with=self._engine)
|
|
94
|
-
return self._table_cache.get(key)
|
|
95
|
-
|
|
96
|
-
def _require_table(self, table: str, schema: str | None) -> Table:
|
|
97
|
-
"""Resolve the table, raising if it does not exist."""
|
|
98
|
-
sa_table = self._resolve_table(table, schema)
|
|
99
|
-
if sa_table is None:
|
|
100
|
-
qualified = f"{schema}.{table}" if schema else table
|
|
101
|
-
raise TableNotFoundError(f"Table '{qualified}' does not exist. Has the asset been materialized?")
|
|
102
|
-
return sa_table
|
|
103
|
-
|
|
104
|
-
def _create_table(self, table: str, schema: str | None, rows: list[dict[str, Any]]) -> Table:
|
|
105
|
-
"""Create a new table from the structure of the first row.
|
|
106
|
-
|
|
107
|
-
Column types are inferred from the Python values in the sample row
|
|
108
|
-
using :func:`_infer_sa_type`.
|
|
109
|
-
|
|
110
|
-
Args:
|
|
111
|
-
table: Target table name
|
|
112
|
-
schema: Database schema
|
|
113
|
-
rows: Row data (at least one row required for schema inference).
|
|
114
|
-
|
|
115
|
-
Returns:
|
|
116
|
-
The newly created :class:`~sqlalchemy.schema.Table`.
|
|
117
|
-
"""
|
|
118
|
-
assert self._conn is not None
|
|
119
|
-
sample = rows[0]
|
|
120
|
-
columns = [Column(name, _infer_sa_type(value)) for name, value in sample.items()]
|
|
121
|
-
sa_metadata = MetaData(schema=schema)
|
|
122
|
-
sa_table = Table(table, sa_metadata, *columns)
|
|
123
|
-
sa_metadata.create_all(self._conn)
|
|
124
|
-
self._table_cache[(table, schema)] = sa_table
|
|
125
|
-
return sa_table
|
|
126
|
-
|
|
127
|
-
# ------------------------------------------------------------------
|
|
128
|
-
# Transaction management
|
|
129
|
-
# ------------------------------------------------------------------
|
|
130
|
-
|
|
131
|
-
@contextmanager
|
|
132
|
-
def _transaction(self) -> Iterator[None]:
|
|
133
|
-
"""Open a SQLAlchemy transactional connection for write operations.
|
|
134
|
-
|
|
135
|
-
Sets ``self._conn`` for the duration of the block. The connection is
|
|
136
|
-
committed on success and rolled back on exception (``engine.begin()``
|
|
137
|
-
semantics).
|
|
138
|
-
|
|
139
|
-
Yields:
|
|
140
|
-
None
|
|
141
|
-
"""
|
|
142
|
-
with self._engine.begin() as conn:
|
|
143
|
-
self._conn = conn
|
|
144
|
-
try:
|
|
145
|
-
yield
|
|
146
|
-
finally:
|
|
147
|
-
self._conn = None
|
|
148
|
-
|
|
149
|
-
# ------------------------------------------------------------------
|
|
150
|
-
# DatabaseIO hooks
|
|
151
|
-
# ------------------------------------------------------------------
|
|
152
|
-
|
|
153
|
-
def _insert(self, table: str, schema: str | None, rows: list[dict[str, Any]]) -> None:
|
|
154
|
-
"""Insert rows in chunks using the active transaction connection.
|
|
155
|
-
|
|
156
|
-
If the table does not exist yet, it is created from the row data
|
|
157
|
-
before inserting.
|
|
158
|
-
|
|
159
|
-
Args:
|
|
160
|
-
table: Target table name
|
|
161
|
-
schema: Database schema
|
|
162
|
-
rows: Row data as list of dicts
|
|
163
|
-
"""
|
|
164
|
-
assert self._conn is not None
|
|
165
|
-
sa_table = self._resolve_table(table, schema)
|
|
166
|
-
if sa_table is None:
|
|
167
|
-
sa_table = self._create_table(table, schema, rows)
|
|
168
|
-
for i in range(0, len(rows), self.chunk_size):
|
|
169
|
-
self._conn.execute(sa_table.insert(), rows[i : i + self.chunk_size])
|
|
170
|
-
|
|
171
|
-
def _delete_all(self, table: str, schema: str | None) -> None:
|
|
172
|
-
"""Delete all rows from the table using the active transaction connection.
|
|
173
|
-
|
|
174
|
-
No-op when the table does not exist yet.
|
|
175
|
-
|
|
176
|
-
Args:
|
|
177
|
-
table: Target table name
|
|
178
|
-
schema: Database schema
|
|
179
|
-
"""
|
|
180
|
-
assert self._conn is not None
|
|
181
|
-
sa_table = self._resolve_table(table, schema)
|
|
182
|
-
if sa_table is None:
|
|
183
|
-
return
|
|
184
|
-
self._conn.execute(sa_table.delete())
|
|
185
|
-
|
|
186
|
-
def _delete_partition(self, table: str, schema: str | None, column: str, value: Any) -> None:
|
|
187
|
-
"""Delete rows matching a partition value using the active transaction connection.
|
|
188
|
-
|
|
189
|
-
No-op when the table does not exist yet.
|
|
190
|
-
|
|
191
|
-
Args:
|
|
192
|
-
table: Target table name
|
|
193
|
-
schema: Database schema
|
|
194
|
-
column: Partition column name
|
|
195
|
-
value: Partition value to match
|
|
196
|
-
"""
|
|
197
|
-
assert self._conn is not None
|
|
198
|
-
sa_table = self._resolve_table(table, schema)
|
|
199
|
-
if sa_table is None:
|
|
200
|
-
return
|
|
201
|
-
self._conn.execute(sa_table.delete().where(sa_table.c[column] == value))
|
|
202
|
-
|
|
203
|
-
def _select_all(self, table: str, schema: str | None) -> list[dict[str, Any]]:
|
|
204
|
-
"""Select all rows from the table.
|
|
205
|
-
|
|
206
|
-
Opens a dedicated read connection (not part of the write transaction).
|
|
207
|
-
|
|
208
|
-
Args:
|
|
209
|
-
table: Target table name
|
|
210
|
-
schema: Database schema
|
|
211
|
-
|
|
212
|
-
Returns:
|
|
213
|
-
All rows as list of dicts
|
|
214
|
-
|
|
215
|
-
Raises:
|
|
216
|
-
ValueError: If the table does not exist
|
|
217
|
-
"""
|
|
218
|
-
sa_table = self._require_table(table, schema)
|
|
219
|
-
with self._engine.connect() as conn:
|
|
220
|
-
result = conn.execute(sa_table.select())
|
|
221
|
-
return [dict(row._mapping) for row in result]
|
|
222
|
-
|
|
223
|
-
def _select_partition(self, table: str, schema: str | None, column: str, value: Any) -> list[dict[str, Any]]:
|
|
224
|
-
"""Select rows matching a partition value.
|
|
225
|
-
|
|
226
|
-
Opens a dedicated read connection (not part of the write transaction).
|
|
227
|
-
|
|
228
|
-
Args:
|
|
229
|
-
table: Target table name
|
|
230
|
-
schema: Database schema
|
|
231
|
-
column: Partition column name
|
|
232
|
-
value: Partition value to match
|
|
233
|
-
|
|
234
|
-
Returns:
|
|
235
|
-
Matching rows as list of dicts
|
|
236
|
-
|
|
237
|
-
Raises:
|
|
238
|
-
ValueError: If the table does not exist
|
|
239
|
-
"""
|
|
240
|
-
sa_table = self._require_table(table, schema)
|
|
241
|
-
with self._engine.connect() as conn:
|
|
242
|
-
result = conn.execute(sa_table.select().where(sa_table.c[column] == value))
|
|
243
|
-
return [dict(row._mapping) for row in result]
|
|
244
|
-
|
|
245
|
-
# ------------------------------------------------------------------
|
|
246
|
-
# Introspection
|
|
247
|
-
# ------------------------------------------------------------------
|
|
248
|
-
|
|
249
|
-
def _count_by_partition(
|
|
250
|
-
self, table: str, schema: str | None, column: str,
|
|
251
|
-
) -> dict[str, int]:
|
|
252
|
-
"""Return row counts grouped by partition column via SQL ``GROUP BY``.
|
|
253
|
-
|
|
254
|
-
Args:
|
|
255
|
-
table: Target table name
|
|
256
|
-
schema: Database schema
|
|
257
|
-
column: Column to group by
|
|
258
|
-
|
|
259
|
-
Returns:
|
|
260
|
-
Mapping from partition value (as string) to row count.
|
|
261
|
-
|
|
262
|
-
Raises:
|
|
263
|
-
TableNotFoundError: If the table does not exist.
|
|
264
|
-
"""
|
|
265
|
-
from sqlalchemy import func
|
|
266
|
-
|
|
267
|
-
sa_table = self._require_table(table, schema)
|
|
268
|
-
col = sa_table.c[column]
|
|
269
|
-
stmt = sa_table.select().with_only_columns(col, func.count()).group_by(col)
|
|
270
|
-
with self._engine.connect() as conn:
|
|
271
|
-
result = conn.execute(stmt)
|
|
272
|
-
return {str(row[0]): row[1] for row in result}
|
|
273
|
-
|
|
274
|
-
# ------------------------------------------------------------------
|
|
275
|
-
# Lifecycle
|
|
276
|
-
# ------------------------------------------------------------------
|
|
277
|
-
|
|
278
|
-
def dispose(self) -> None:
|
|
279
|
-
"""Dispose the SQLAlchemy engine and clear the table cache."""
|
|
280
|
-
self._engine.dispose()
|
|
281
|
-
self._table_cache.clear()
|
|
@@ -1,70 +0,0 @@
|
|
|
1
|
-
"""MySQL IO implementation."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
from typing import TYPE_CHECKING
|
|
6
|
-
|
|
7
|
-
from interloper.io.database import WriteDisposition
|
|
8
|
-
from interloper.serialization.io import IOSpec
|
|
9
|
-
from sqlalchemy.engine import URL
|
|
10
|
-
|
|
11
|
-
from interloper_sql.io.base import SqlIO
|
|
12
|
-
|
|
13
|
-
if TYPE_CHECKING:
|
|
14
|
-
from interloper.io.adapter import DataAdapter
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
class MySQLIO(SqlIO):
|
|
18
|
-
"""MySQL-specific IO manager.
|
|
19
|
-
|
|
20
|
-
Extends :class:`SqlIO` for MySQL connections. Uses standard ``DELETE`` for
|
|
21
|
-
row removal because MySQL's ``TRUNCATE`` causes an implicit commit and
|
|
22
|
-
cannot participate in a transaction.
|
|
23
|
-
|
|
24
|
-
Args:
|
|
25
|
-
host: Database server hostname
|
|
26
|
-
database: Database name
|
|
27
|
-
port: Database server port
|
|
28
|
-
user: Database user
|
|
29
|
-
password: Database password
|
|
30
|
-
driver: SQLAlchemy driver (e.g. ``pymysql``, ``mysqlconnector``)
|
|
31
|
-
write_disposition: Controls whether existing rows are deleted before writing
|
|
32
|
-
chunk_size: Number of rows per insert batch
|
|
33
|
-
adapter: Optional data adapter for type conversion
|
|
34
|
-
"""
|
|
35
|
-
|
|
36
|
-
def __init__(
|
|
37
|
-
self,
|
|
38
|
-
host: str,
|
|
39
|
-
database: str,
|
|
40
|
-
port: int = 3306,
|
|
41
|
-
user: str = "root",
|
|
42
|
-
password: str | None = None,
|
|
43
|
-
driver: str | None = None,
|
|
44
|
-
write_disposition: WriteDisposition = WriteDisposition.REPLACE,
|
|
45
|
-
chunk_size: int = 1000,
|
|
46
|
-
adapter: DataAdapter | str | None = None,
|
|
47
|
-
) -> None:
|
|
48
|
-
self.host = host
|
|
49
|
-
self.port = port
|
|
50
|
-
self.database = database
|
|
51
|
-
self.user = user
|
|
52
|
-
self.password = password
|
|
53
|
-
self.driver = driver
|
|
54
|
-
|
|
55
|
-
drivername = f"mysql+{driver}" if driver else "mysql"
|
|
56
|
-
url = URL.create(drivername, user, password, host, port, database)
|
|
57
|
-
super().__init__(url, write_disposition, chunk_size, adapter)
|
|
58
|
-
|
|
59
|
-
def to_spec(self) -> IOSpec:
|
|
60
|
-
"""Convert to serializable spec."""
|
|
61
|
-
init = self._base_init_kwargs()
|
|
62
|
-
init["host"] = self.host
|
|
63
|
-
init["port"] = self.port
|
|
64
|
-
init["database"] = self.database
|
|
65
|
-
init["user"] = self.user
|
|
66
|
-
if self.password is not None:
|
|
67
|
-
init["password"] = self.password
|
|
68
|
-
if self.driver is not None:
|
|
69
|
-
init["driver"] = self.driver
|
|
70
|
-
return IOSpec(path=self.path, init=init)
|
|
@@ -1,87 +0,0 @@
|
|
|
1
|
-
"""PostgreSQL IO implementation."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
from typing import TYPE_CHECKING
|
|
6
|
-
|
|
7
|
-
from interloper.io.database import WriteDisposition
|
|
8
|
-
from interloper.serialization.io import IOSpec
|
|
9
|
-
from sqlalchemy import text
|
|
10
|
-
from sqlalchemy.engine import URL
|
|
11
|
-
|
|
12
|
-
from interloper_sql.io.base import SqlIO
|
|
13
|
-
|
|
14
|
-
if TYPE_CHECKING:
|
|
15
|
-
from interloper.io.adapter import DataAdapter
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
class PostgresIO(SqlIO):
|
|
19
|
-
"""PostgreSQL-specific IO manager.
|
|
20
|
-
|
|
21
|
-
Extends :class:`SqlIO` with PostgreSQL optimisations:
|
|
22
|
-
|
|
23
|
-
* Uses ``TRUNCATE`` (transactional in Postgres) instead of ``DELETE`` for
|
|
24
|
-
full-table replacements, which is significantly faster on large tables.
|
|
25
|
-
|
|
26
|
-
Args:
|
|
27
|
-
host: Database server hostname
|
|
28
|
-
port: Database server port
|
|
29
|
-
database: Database name
|
|
30
|
-
user: Database user
|
|
31
|
-
password: Database password
|
|
32
|
-
driver: SQLAlchemy driver (e.g. ``psycopg2``, ``asyncpg``)
|
|
33
|
-
write_disposition: Controls whether existing rows are deleted before writing
|
|
34
|
-
chunk_size: Number of rows per insert batch
|
|
35
|
-
adapter: Optional data adapter for type conversion
|
|
36
|
-
"""
|
|
37
|
-
|
|
38
|
-
def __init__(
|
|
39
|
-
self,
|
|
40
|
-
host: str,
|
|
41
|
-
port: int = 5432,
|
|
42
|
-
database: str = "postgres",
|
|
43
|
-
user: str = "postgres",
|
|
44
|
-
password: str | None = None,
|
|
45
|
-
driver: str | None = None,
|
|
46
|
-
write_disposition: WriteDisposition = WriteDisposition.REPLACE,
|
|
47
|
-
chunk_size: int = 1000,
|
|
48
|
-
adapter: DataAdapter | str | None = None,
|
|
49
|
-
) -> None:
|
|
50
|
-
self.host = host
|
|
51
|
-
self.port = port
|
|
52
|
-
self.database = database
|
|
53
|
-
self.user = user
|
|
54
|
-
self.password = password
|
|
55
|
-
self.driver = driver
|
|
56
|
-
|
|
57
|
-
drivername = f"postgresql+{driver}" if driver else "postgresql"
|
|
58
|
-
url = URL.create(drivername, user, password, host, port, database)
|
|
59
|
-
super().__init__(url, write_disposition, chunk_size, adapter)
|
|
60
|
-
|
|
61
|
-
def _delete_all(self, table: str, schema: str | None) -> None:
|
|
62
|
-
"""Use TRUNCATE for full-table deletes (transactional in PostgreSQL).
|
|
63
|
-
|
|
64
|
-
No-op when the table does not exist yet.
|
|
65
|
-
|
|
66
|
-
Args:
|
|
67
|
-
table: Target table name
|
|
68
|
-
schema: Database schema
|
|
69
|
-
"""
|
|
70
|
-
assert self._conn is not None
|
|
71
|
-
sa_table = self._resolve_table(table, schema)
|
|
72
|
-
if sa_table is None:
|
|
73
|
-
return
|
|
74
|
-
self._conn.execute(text(f"TRUNCATE TABLE {sa_table.fullname}"))
|
|
75
|
-
|
|
76
|
-
def to_spec(self) -> IOSpec:
|
|
77
|
-
"""Convert to serializable spec."""
|
|
78
|
-
init = self._base_init_kwargs()
|
|
79
|
-
init["host"] = self.host
|
|
80
|
-
init["port"] = self.port
|
|
81
|
-
init["database"] = self.database
|
|
82
|
-
init["user"] = self.user
|
|
83
|
-
if self.password is not None:
|
|
84
|
-
init["password"] = self.password
|
|
85
|
-
if self.driver is not None:
|
|
86
|
-
init["driver"] = self.driver
|
|
87
|
-
return IOSpec(path=self.path, init=init)
|
|
@@ -1,45 +0,0 @@
|
|
|
1
|
-
"""SQLite IO implementation."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
from typing import TYPE_CHECKING
|
|
6
|
-
|
|
7
|
-
from interloper.io.database import WriteDisposition
|
|
8
|
-
from interloper.serialization.io import IOSpec
|
|
9
|
-
|
|
10
|
-
from interloper_sql.io.base import SqlIO
|
|
11
|
-
|
|
12
|
-
if TYPE_CHECKING:
|
|
13
|
-
from interloper.io.adapter import DataAdapter
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
class SqliteIO(SqlIO):
|
|
17
|
-
"""SQLite-specific IO manager.
|
|
18
|
-
|
|
19
|
-
Extends :class:`SqlIO` for SQLite connections. Useful for local development
|
|
20
|
-
and testing without requiring an external database server.
|
|
21
|
-
|
|
22
|
-
Args:
|
|
23
|
-
database: Path to the SQLite database file, or ``":memory:"`` for an
|
|
24
|
-
in-memory database.
|
|
25
|
-
write_disposition: Controls whether existing rows are deleted before writing
|
|
26
|
-
chunk_size: Number of rows per insert batch
|
|
27
|
-
adapter: Optional data adapter for type conversion
|
|
28
|
-
"""
|
|
29
|
-
|
|
30
|
-
def __init__(
|
|
31
|
-
self,
|
|
32
|
-
database: str = ":memory:",
|
|
33
|
-
write_disposition: WriteDisposition = WriteDisposition.REPLACE,
|
|
34
|
-
chunk_size: int = 1000,
|
|
35
|
-
adapter: DataAdapter | str | None = None,
|
|
36
|
-
) -> None:
|
|
37
|
-
self.database = database
|
|
38
|
-
url = f"sqlite:///{database}"
|
|
39
|
-
super().__init__(url, write_disposition, chunk_size, adapter)
|
|
40
|
-
|
|
41
|
-
def to_spec(self) -> IOSpec:
|
|
42
|
-
"""Convert to serializable spec."""
|
|
43
|
-
init = self._base_init_kwargs()
|
|
44
|
-
init["database"] = self.database
|
|
45
|
-
return IOSpec(path=self.path, init=init)
|