soda-sqlserver 4.17.1__tar.gz → 4.19.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/PKG-INFO +2 -2
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/pyproject.toml +2 -2
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver/common/data_sources/sqlserver_data_source.py +120 -2
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver/common/data_sources/sqlserver_data_source_connection.py +37 -0
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/PKG-INFO +2 -2
- soda_sqlserver-4.19.0/src/soda_sqlserver.egg-info/requires.txt +2 -0
- soda_sqlserver-4.17.1/src/soda_sqlserver.egg-info/requires.txt +0 -2
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/setup.cfg +0 -0
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver/test_helpers/sqlserver_data_source_test_helper.py +0 -0
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/SOURCES.txt +0 -0
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/dependency_links.txt +0 -0
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/entry_points.txt +0 -0
- {soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/top_level.txt +0 -0
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: soda-sqlserver
|
|
3
|
-
Version: 4.
|
|
3
|
+
Version: 4.19.0
|
|
4
4
|
Summary: Soda SQL Server V4
|
|
5
5
|
Author-email: "Soda Data N.V." <info@soda.io>
|
|
6
6
|
License: Proprietary
|
|
7
7
|
Requires-Python: >=3.10
|
|
8
|
-
Requires-Dist: soda-core==4.
|
|
8
|
+
Requires-Dist: soda-core==4.19.0
|
|
9
9
|
Requires-Dist: pyodbc
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "soda-sqlserver"
|
|
3
|
-
version = "4.
|
|
3
|
+
version = "4.19.0"
|
|
4
4
|
description = "Soda SQL Server V4"
|
|
5
5
|
requires-python = ">=3.10"
|
|
6
6
|
license = {text = "Proprietary"}
|
|
@@ -8,7 +8,7 @@ authors = [
|
|
|
8
8
|
{name = "Soda Data N.V.", email = "info@soda.io"}
|
|
9
9
|
]
|
|
10
10
|
dependencies = [
|
|
11
|
-
"soda-core==4.
|
|
11
|
+
"soda-core==4.19.0",
|
|
12
12
|
"pyodbc",
|
|
13
13
|
]
|
|
14
14
|
|
|
@@ -9,6 +9,7 @@ from soda_core.common.dataset_identifier import DatasetIdentifier
|
|
|
9
9
|
from soda_core.common.logging_constants import soda_logger
|
|
10
10
|
from soda_core.common.metadata_types import SodaDataTypeName, SqlDataType
|
|
11
11
|
from soda_core.common.sql_ast import (
|
|
12
|
+
ADD_INTERVAL,
|
|
12
13
|
AND,
|
|
13
14
|
COLUMN,
|
|
14
15
|
COUNT,
|
|
@@ -30,16 +31,19 @@ from soda_core.common.sql_ast import (
|
|
|
30
31
|
LIMIT,
|
|
31
32
|
OFFSET,
|
|
32
33
|
ORDER_BY_ASC,
|
|
34
|
+
PERCENTILE_WITHIN_GROUP,
|
|
33
35
|
RANDOM,
|
|
34
36
|
REGEX_LIKE,
|
|
35
37
|
SELECT,
|
|
36
38
|
STAR,
|
|
37
39
|
STRING_HASH,
|
|
40
|
+
TIME_DELTA,
|
|
38
41
|
TUPLE,
|
|
39
42
|
VALUES,
|
|
40
43
|
WHERE,
|
|
41
44
|
WITH,
|
|
42
45
|
SqlExpressionStr,
|
|
46
|
+
seconds_per_time_bucket,
|
|
43
47
|
)
|
|
44
48
|
from soda_core.common.sql_dialect import SqlDialect
|
|
45
49
|
from soda_sqlserver.common.data_sources.sqlserver_data_source_connection import (
|
|
@@ -52,9 +56,19 @@ from soda_sqlserver.common.data_sources.sqlserver_data_source_connection import
|
|
|
52
56
|
logger: logging.Logger = soda_logger
|
|
53
57
|
|
|
54
58
|
|
|
59
|
+
# APPROX_PERCENTILE_DISC needs SQL Server 2022+ on-prem, or Azure SQL Database /
|
|
60
|
+
# Managed Instance (which report a legacy ProductMajorVersion).
|
|
61
|
+
SQLSERVER_2022_MAJOR_VERSION = 16
|
|
62
|
+
AZURE_SQL_DATABASE_ENGINE_EDITION = 5
|
|
63
|
+
AZURE_SQL_MANAGED_INSTANCE_ENGINE_EDITION = 8
|
|
64
|
+
|
|
65
|
+
|
|
55
66
|
class SqlServerDataSourceImpl(DataSourceImpl, model_class=SqlServerDataSourceModel):
|
|
56
67
|
def __init__(self, data_source_model: SqlServerDataSourceModel, connection: Optional[DataSourceConnection] = None):
|
|
57
68
|
super().__init__(data_source_model=data_source_model, connection=connection)
|
|
69
|
+
# A live connection supplied at construction (e.g. a bulk-insert copy)
|
|
70
|
+
# already carries detected server facts; propagate them right away.
|
|
71
|
+
self._sync_dialect_server_info()
|
|
58
72
|
|
|
59
73
|
def _create_sql_dialect(self) -> SqlDialect:
|
|
60
74
|
return SqlServerSqlDialect()
|
|
@@ -64,11 +78,72 @@ class SqlServerDataSourceImpl(DataSourceImpl, model_class=SqlServerDataSourceMod
|
|
|
64
78
|
name=self.data_source_model.name, connection_properties=self.data_source_model.connection_properties
|
|
65
79
|
)
|
|
66
80
|
|
|
81
|
+
def open_connection(self) -> None:
|
|
82
|
+
super().open_connection()
|
|
83
|
+
self._sync_dialect_server_info()
|
|
84
|
+
|
|
85
|
+
def _sync_dialect_server_info(self) -> None:
|
|
86
|
+
"""Copy the connection's detected engine facts onto the dialect, which
|
|
87
|
+
derives version-dependent capabilities from them (see
|
|
88
|
+
SqlServerSqlDialect.supports_percentile_within_group).
|
|
89
|
+
|
|
90
|
+
Runs at connection-open time only; the dialect must NOT read the connection
|
|
91
|
+
during SQL generation — in snapshot replay the connection is a lazy wrapper
|
|
92
|
+
whose attribute access opens a real connection, so touching it while
|
|
93
|
+
building SQL breaks replay.
|
|
94
|
+
|
|
95
|
+
Gate on the concrete connection type rather than duck-typing the attributes:
|
|
96
|
+
a replay SnapshotDataSourceConnection is NOT a SqlServerDataSourceConnection,
|
|
97
|
+
so isinstance() is False and we never touch its attributes (which would fire
|
|
98
|
+
its __getattr__ fallback and open a real connection). Replay then keeps the
|
|
99
|
+
dialect's None defaults (assume newest engine), which is exactly what
|
|
100
|
+
recorded snapshots expect.
|
|
101
|
+
"""
|
|
102
|
+
conn = self.data_source_connection
|
|
103
|
+
if isinstance(conn, SqlServerDataSourceConnection):
|
|
104
|
+
# Guaranteed by every _create_sql_dialect in this hierarchy; the assert
|
|
105
|
+
# only narrows the declared SqlDialect type for the assignments.
|
|
106
|
+
assert isinstance(self.sql_dialect, SqlServerSqlDialect)
|
|
107
|
+
self.sql_dialect.server_major_version = conn.server_major_version
|
|
108
|
+
self.sql_dialect.engine_edition = conn.engine_edition
|
|
109
|
+
|
|
67
110
|
|
|
68
111
|
class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
|
|
69
112
|
DEFAULT_QUOTE_CHAR = "[" # Do not use this! Always use quote_default()
|
|
70
113
|
SODA_DATA_TYPE_SYNONYMS = ((SodaDataTypeName.TEXT, SodaDataTypeName.VARCHAR),)
|
|
71
114
|
|
|
115
|
+
def __init__(self):
|
|
116
|
+
super().__init__()
|
|
117
|
+
# Raw engine facts, synced from the live connection at open by
|
|
118
|
+
# SqlServerDataSourceImpl._sync_dialect_server_info. None means no live
|
|
119
|
+
# server facts (pure SQL rendering, unit tests, snapshot replay);
|
|
120
|
+
# capability checks then assume the newest engine.
|
|
121
|
+
self.server_major_version: Optional[int] = None
|
|
122
|
+
self.engine_edition: Optional[int] = None
|
|
123
|
+
|
|
124
|
+
def _build_stddev_samp_sql(self, stddev_samp) -> str:
|
|
125
|
+
# T-SQL names the sample standard deviation aggregate STDEV.
|
|
126
|
+
return f"STDEV({self.build_expression_sql(stddev_samp.expression)})"
|
|
127
|
+
|
|
128
|
+
def _build_var_samp_sql(self, var_samp) -> str:
|
|
129
|
+
# T-SQL names the sample variance aggregate VAR.
|
|
130
|
+
return f"VAR({self.build_expression_sql(var_samp.expression)})"
|
|
131
|
+
|
|
132
|
+
def supports_percentile_within_group(self) -> bool:
|
|
133
|
+
# T-SQL exposes percentiles as an aggregate only via APPROX_PERCENTILE_DISC:
|
|
134
|
+
# SQL Server 2022+ (ProductMajorVersion >= 16), Azure SQL Database, or Azure
|
|
135
|
+
# SQL Managed Instance (both report a legacy ProductMajorVersion, hence the
|
|
136
|
+
# edition check). Synapse dedicated pools have no percentile aggregate at
|
|
137
|
+
# all; the Synapse dialect pins this to False.
|
|
138
|
+
if self.server_major_version is None and self.engine_edition is None:
|
|
139
|
+
return True # no live server facts: assume the newest engine
|
|
140
|
+
return (
|
|
141
|
+
self.server_major_version is not None and self.server_major_version >= SQLSERVER_2022_MAJOR_VERSION
|
|
142
|
+
) or self.engine_edition in (
|
|
143
|
+
AZURE_SQL_DATABASE_ENGINE_EDITION,
|
|
144
|
+
AZURE_SQL_MANAGED_INSTANCE_ENGINE_EDITION,
|
|
145
|
+
)
|
|
146
|
+
|
|
72
147
|
def build_select_sql(self, select_elements: list, add_semicolon: bool = True) -> str:
|
|
73
148
|
statement_lines: list[str] = []
|
|
74
149
|
statement_lines.extend(self._build_cte_sql_lines(select_elements))
|
|
@@ -120,8 +195,7 @@ class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
|
|
|
120
195
|
|
|
121
196
|
def literal_date(self, date: date):
|
|
122
197
|
"""Technically dates can be passed directly as strings, but this is more explicit."""
|
|
123
|
-
|
|
124
|
-
return f"CAST('{date_string}' AS DATE)"
|
|
198
|
+
return f"CAST('{date.isoformat()}' AS DATE)"
|
|
125
199
|
|
|
126
200
|
def literal_datetime(self, datetime: datetime):
|
|
127
201
|
return f"'{datetime.isoformat(timespec='milliseconds')}'"
|
|
@@ -177,6 +251,50 @@ class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
|
|
|
177
251
|
def sql_expr_timestamp_add_day(self, timestamp_literal: str) -> str:
|
|
178
252
|
return f"DATEADD(DAY, 1, {timestamp_literal})"
|
|
179
253
|
|
|
254
|
+
def literal_timestamp_typed(self, dt: datetime) -> str:
|
|
255
|
+
"""T-SQL has no TIMESTAMP '...' literal (TIMESTAMP is the deprecated
|
|
256
|
+
rowversion type), so cast the string form to DATETIME2 to keep the
|
|
257
|
+
arithmetic operand typed —
|
|
258
|
+
https://learn.microsoft.com/en-us/sql/t-sql/data-types/datetime2-transact-sql."""
|
|
259
|
+
return f"CAST('{self._typed_timestamp_str(dt)}' AS DATETIME2)"
|
|
260
|
+
|
|
261
|
+
# Singular unit names for DATEADD.
|
|
262
|
+
_TIME_BUCKET_UNIT_NAMES: dict = {
|
|
263
|
+
"weeks": "WEEK",
|
|
264
|
+
"days": "DAY",
|
|
265
|
+
"hours": "HOUR",
|
|
266
|
+
"seconds": "SECOND",
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
def _build_time_delta_sql(self, time_delta: TIME_DELTA) -> str:
|
|
270
|
+
"""T-SQL DATEDIFF counts crossed boundaries of the given unit, so
|
|
271
|
+
compute the difference in SECONDS and divide by the seconds-per-
|
|
272
|
+
interval. T-SQL int/int division truncates toward zero, which equals
|
|
273
|
+
the FLOOR of the other dialects only for deltas >= 0 — callers must
|
|
274
|
+
guarantee non-negative deltas (the MM window filter does).
|
|
275
|
+
|
|
276
|
+
DATEDIFF(second, ...) returns int and overflows for spans > ~68
|
|
277
|
+
years; switch to DATEDIFF_BIG if that ever bites."""
|
|
278
|
+
start_sql: str = self.build_expression_sql(time_delta.start)
|
|
279
|
+
end_sql: str = self.build_expression_sql(time_delta.end)
|
|
280
|
+
multiplier: int = seconds_per_time_bucket(time_delta.unit, time_delta.count)
|
|
281
|
+
# Parenthesized so the form stays self-contained if a caller embeds
|
|
282
|
+
# TIME_DELTA in larger arithmetic (every other dialect wraps in FLOOR/cast).
|
|
283
|
+
return f"(DATEDIFF(second, {start_sql}, {end_sql}) / {multiplier})"
|
|
284
|
+
|
|
285
|
+
def _build_add_interval_sql(self, add_interval: ADD_INTERVAL) -> str:
|
|
286
|
+
timestamp_sql: str = self.build_expression_sql(add_interval.timestamp)
|
|
287
|
+
count_sql: str = self.build_expression_sql(add_interval.count_expression)
|
|
288
|
+
unit_name: str = self._TIME_BUCKET_UNIT_NAMES[add_interval.unit]
|
|
289
|
+
return f"DATEADD({unit_name}, {count_sql}, {timestamp_sql})"
|
|
290
|
+
|
|
291
|
+
def _build_percentile_within_group_sql(self, percentile_within_group: PERCENTILE_WITHIN_GROUP) -> str:
|
|
292
|
+
"""T-SQL PERCENTILE_DISC is a window function only; the aggregate form
|
|
293
|
+
is APPROX_PERCENTILE_DISC (SQL Server 2022+/Azure SQL/Fabric,
|
|
294
|
+
https://learn.microsoft.com/en-us/sql/t-sql/functions/approx-percentile-disc-transact-sql)."""
|
|
295
|
+
expression_sql: str = self.build_expression_sql(percentile_within_group.expression)
|
|
296
|
+
return f"APPROX_PERCENTILE_DISC({percentile_within_group.percentile}) WITHIN GROUP (ORDER BY {expression_sql})"
|
|
297
|
+
|
|
180
298
|
def _build_tuple_sql(self, tuple: TUPLE) -> str:
|
|
181
299
|
if tuple.check_context(COUNT) and tuple.check_context(DISTINCT):
|
|
182
300
|
return f"CHECKSUM{super()._build_tuple_sql(tuple)}"
|
|
@@ -107,6 +107,10 @@ def handle_datetimeoffset(dto_value):
|
|
|
107
107
|
|
|
108
108
|
class SqlServerDataSourceConnection(DataSourceConnection):
|
|
109
109
|
def __init__(self, name: str, connection_properties: DataSourceConnectionProperties):
|
|
110
|
+
# Set before super().__init__(), which auto-opens the connection and
|
|
111
|
+
# populates these from the live server in _create_connection.
|
|
112
|
+
self.server_major_version: Optional[int] = None
|
|
113
|
+
self.engine_edition: Optional[int] = None
|
|
110
114
|
super().__init__(name, connection_properties)
|
|
111
115
|
|
|
112
116
|
# Normalize pyodbc.Row objects so downstream consumers see plain tuples.
|
|
@@ -190,10 +194,43 @@ class SqlServerDataSourceConnection(DataSourceConnection):
|
|
|
190
194
|
|
|
191
195
|
self.connection.add_output_converter(-155, handle_datetimeoffset)
|
|
192
196
|
self.connection.add_output_converter(-150, handle_datetime)
|
|
197
|
+
self._detect_server_info(self.connection)
|
|
193
198
|
return self.connection
|
|
194
199
|
except Exception as e:
|
|
195
200
|
raise DataSourceConnectionException(e) from e
|
|
196
201
|
|
|
202
|
+
@staticmethod
|
|
203
|
+
def _parse_server_major_version(dbms_version: Optional[str]) -> Optional[int]:
|
|
204
|
+
"""Parse the leading integer of an ODBC SQL_DBMS_VER string, e.g. '15.00.4123' -> 15."""
|
|
205
|
+
if not dbms_version:
|
|
206
|
+
return None
|
|
207
|
+
try:
|
|
208
|
+
return int(str(dbms_version).split(".")[0])
|
|
209
|
+
except (ValueError, IndexError):
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
def _detect_server_info(self, connection) -> None:
|
|
213
|
+
"""Detect raw engine facts once per connect; the data source syncs them onto
|
|
214
|
+
the dialect, which derives version-dependent capabilities from them (e.g.
|
|
215
|
+
APPROX_PERCENTILE_DISC needs SQL Server 2022+ or Azure SQL DB/MI).
|
|
216
|
+
|
|
217
|
+
The product version comes from the driver's login handshake — no extra
|
|
218
|
+
round-trip; EngineEdition costs one query. Detection is never fatal for an
|
|
219
|
+
otherwise healthy connect: on failure a warning is logged and the fact stays
|
|
220
|
+
None, which capability checks treat as "assume the newest engine".
|
|
221
|
+
"""
|
|
222
|
+
try:
|
|
223
|
+
self.server_major_version = self._parse_server_major_version(connection.getinfo(pyodbc.SQL_DBMS_VER))
|
|
224
|
+
except Exception as e:
|
|
225
|
+
logger.warning(f"Could not determine SQL Server product version: {e}")
|
|
226
|
+
try:
|
|
227
|
+
with connection.cursor() as cursor:
|
|
228
|
+
cursor.execute("SELECT CAST(SERVERPROPERTY('EngineEdition') AS INT)")
|
|
229
|
+
row = cursor.fetchone()
|
|
230
|
+
self.engine_edition = row[0] if row is not None else None
|
|
231
|
+
except Exception as e:
|
|
232
|
+
logger.warning(f"Could not determine SQL Server engine edition: {e}")
|
|
233
|
+
|
|
197
234
|
def _execute_query_get_result_row_column_name(self, column) -> str:
|
|
198
235
|
return column[0]
|
|
199
236
|
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: soda-sqlserver
|
|
3
|
-
Version: 4.
|
|
3
|
+
Version: 4.19.0
|
|
4
4
|
Summary: Soda SQL Server V4
|
|
5
5
|
Author-email: "Soda Data N.V." <info@soda.io>
|
|
6
6
|
License: Proprietary
|
|
7
7
|
Requires-Python: >=3.10
|
|
8
|
-
Requires-Dist: soda-core==4.
|
|
8
|
+
Requires-Dist: soda-core==4.19.0
|
|
9
9
|
Requires-Dist: pyodbc
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{soda_sqlserver-4.17.1 → soda_sqlserver-4.19.0}/src/soda_sqlserver.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|