soda-sqlserver 4.18.0__tar.gz → 4.20.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,9 +1,9 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: soda-sqlserver
3
- Version: 4.18.0
3
+ Version: 4.20.0
4
4
  Summary: Soda SQL Server V4
5
5
  Author-email: "Soda Data N.V." <info@soda.io>
6
6
  License: Proprietary
7
7
  Requires-Python: >=3.10
8
- Requires-Dist: soda-core==4.18.0
8
+ Requires-Dist: soda-core==4.20.0
9
9
  Requires-Dist: pyodbc
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "soda-sqlserver"
3
- version = "4.18.0"
3
+ version = "4.20.0"
4
4
  description = "Soda SQL Server V4"
5
5
  requires-python = ">=3.10"
6
6
  license = {text = "Proprietary"}
@@ -8,7 +8,7 @@ authors = [
8
8
  {name = "Soda Data N.V.", email = "info@soda.io"}
9
9
  ]
10
10
  dependencies = [
11
- "soda-core==4.18.0",
11
+ "soda-core==4.20.0",
12
12
  "pyodbc",
13
13
  ]
14
14
 
@@ -9,6 +9,7 @@ from soda_core.common.dataset_identifier import DatasetIdentifier
9
9
  from soda_core.common.logging_constants import soda_logger
10
10
  from soda_core.common.metadata_types import SodaDataTypeName, SqlDataType
11
11
  from soda_core.common.sql_ast import (
12
+ ADD_INTERVAL,
12
13
  AND,
13
14
  COLUMN,
14
15
  COUNT,
@@ -29,17 +30,19 @@ from soda_core.common.sql_ast import (
29
30
  LENGTH,
30
31
  LIMIT,
31
32
  OFFSET,
32
- ORDER_BY_ASC,
33
+ PERCENTILE_WITHIN_GROUP,
33
34
  RANDOM,
34
35
  REGEX_LIKE,
35
36
  SELECT,
36
37
  STAR,
37
38
  STRING_HASH,
39
+ TIME_DELTA,
38
40
  TUPLE,
39
41
  VALUES,
40
42
  WHERE,
41
43
  WITH,
42
44
  SqlExpressionStr,
45
+ seconds_per_time_bucket,
43
46
  )
44
47
  from soda_core.common.sql_dialect import SqlDialect
45
48
  from soda_sqlserver.common.data_sources.sqlserver_data_source_connection import (
@@ -52,9 +55,19 @@ from soda_sqlserver.common.data_sources.sqlserver_data_source_connection import
52
55
  logger: logging.Logger = soda_logger
53
56
 
54
57
 
58
+ # APPROX_PERCENTILE_DISC needs SQL Server 2022+ on-prem, or Azure SQL Database /
59
+ # Managed Instance (which report a legacy ProductMajorVersion).
60
+ SQLSERVER_2022_MAJOR_VERSION = 16
61
+ AZURE_SQL_DATABASE_ENGINE_EDITION = 5
62
+ AZURE_SQL_MANAGED_INSTANCE_ENGINE_EDITION = 8
63
+
64
+
55
65
  class SqlServerDataSourceImpl(DataSourceImpl, model_class=SqlServerDataSourceModel):
56
66
  def __init__(self, data_source_model: SqlServerDataSourceModel, connection: Optional[DataSourceConnection] = None):
57
67
  super().__init__(data_source_model=data_source_model, connection=connection)
68
+ # A live connection supplied at construction (e.g. a bulk-insert copy)
69
+ # already carries detected server facts; propagate them right away.
70
+ self._sync_dialect_server_info()
58
71
 
59
72
  def _create_sql_dialect(self) -> SqlDialect:
60
73
  return SqlServerSqlDialect()
@@ -64,11 +77,77 @@ class SqlServerDataSourceImpl(DataSourceImpl, model_class=SqlServerDataSourceMod
64
77
  name=self.data_source_model.name, connection_properties=self.data_source_model.connection_properties
65
78
  )
66
79
 
80
+ def open_connection(self) -> None:
81
+ super().open_connection()
82
+ self._sync_dialect_server_info()
83
+
84
+ def _sync_dialect_server_info(self) -> None:
85
+ """Copy the connection's detected engine facts onto the dialect, which
86
+ derives version-dependent capabilities from them (see
87
+ SqlServerSqlDialect.supports_percentile_within_group).
88
+
89
+ Runs at connection-open time only; the dialect must NOT read the connection
90
+ during SQL generation — in snapshot replay the connection is a lazy wrapper
91
+ whose attribute access opens a real connection, so touching it while
92
+ building SQL breaks replay.
93
+
94
+ Gate on the concrete connection type rather than duck-typing the attributes:
95
+ a replay SnapshotDataSourceConnection is NOT a SqlServerDataSourceConnection,
96
+ so isinstance() is False and we never touch its attributes (which would fire
97
+ its __getattr__ fallback and open a real connection). Replay then keeps the
98
+ dialect's None defaults (assume newest engine), which is exactly what
99
+ recorded snapshots expect.
100
+ """
101
+ conn = self.data_source_connection
102
+ if isinstance(conn, SqlServerDataSourceConnection):
103
+ # Guaranteed by every _create_sql_dialect in this hierarchy; the assert
104
+ # only narrows the declared SqlDialect type for the assignments.
105
+ assert isinstance(self.sql_dialect, SqlServerSqlDialect)
106
+ self.sql_dialect.server_major_version = conn.server_major_version
107
+ self.sql_dialect.engine_edition = conn.engine_edition
108
+
67
109
 
68
110
  class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
69
111
  DEFAULT_QUOTE_CHAR = "[" # Do not use this! Always use quote_default()
70
112
  SODA_DATA_TYPE_SYNONYMS = ((SodaDataTypeName.TEXT, SodaDataTypeName.VARCHAR),)
71
113
 
114
+ def __init__(self):
115
+ super().__init__()
116
+ # Raw engine facts, synced from the live connection at open by
117
+ # SqlServerDataSourceImpl._sync_dialect_server_info. None means no live
118
+ # server facts (pure SQL rendering, unit tests, snapshot replay);
119
+ # capability checks then assume the newest engine.
120
+ self.server_major_version: Optional[int] = None
121
+ self.engine_edition: Optional[int] = None
122
+
123
+ def supports_primary_keys(self) -> bool:
124
+ # SQL Server enforces primary keys and reports them through the standard
125
+ # information_schema constraint views, with the standard PRIMARY KEY (...) DDL.
126
+ return True
127
+
128
+ def _build_stddev_samp_sql(self, stddev_samp) -> str:
129
+ # T-SQL names the sample standard deviation aggregate STDEV.
130
+ return f"STDEV({self.build_expression_sql(stddev_samp.expression)})"
131
+
132
+ def _build_var_samp_sql(self, var_samp) -> str:
133
+ # T-SQL names the sample variance aggregate VAR.
134
+ return f"VAR({self.build_expression_sql(var_samp.expression)})"
135
+
136
+ def supports_percentile_within_group(self) -> bool:
137
+ # T-SQL exposes percentiles as an aggregate only via APPROX_PERCENTILE_DISC:
138
+ # SQL Server 2022+ (ProductMajorVersion >= 16), Azure SQL Database, or Azure
139
+ # SQL Managed Instance (both report a legacy ProductMajorVersion, hence the
140
+ # edition check). Synapse dedicated pools have no percentile aggregate at
141
+ # all; the Synapse dialect pins this to False.
142
+ if self.server_major_version is None and self.engine_edition is None:
143
+ return True # no live server facts: assume the newest engine
144
+ return (
145
+ self.server_major_version is not None and self.server_major_version >= SQLSERVER_2022_MAJOR_VERSION
146
+ ) or self.engine_edition in (
147
+ AZURE_SQL_DATABASE_ENGINE_EDITION,
148
+ AZURE_SQL_MANAGED_INSTANCE_ENGINE_EDITION,
149
+ )
150
+
72
151
  def build_select_sql(self, select_elements: list, add_semicolon: bool = True) -> str:
73
152
  statement_lines: list[str] = []
74
153
  statement_lines.extend(self._build_cte_sql_lines(select_elements))
@@ -176,6 +255,50 @@ class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
176
255
  def sql_expr_timestamp_add_day(self, timestamp_literal: str) -> str:
177
256
  return f"DATEADD(DAY, 1, {timestamp_literal})"
178
257
 
258
+ def literal_timestamp_typed(self, dt: datetime) -> str:
259
+ """T-SQL has no TIMESTAMP '...' literal (TIMESTAMP is the deprecated
260
+ rowversion type), so cast the string form to DATETIME2 to keep the
261
+ arithmetic operand typed —
262
+ https://learn.microsoft.com/en-us/sql/t-sql/data-types/datetime2-transact-sql."""
263
+ return f"CAST('{self._typed_timestamp_str(dt)}' AS DATETIME2)"
264
+
265
+ # Singular unit names for DATEADD.
266
+ _TIME_BUCKET_UNIT_NAMES: dict = {
267
+ "weeks": "WEEK",
268
+ "days": "DAY",
269
+ "hours": "HOUR",
270
+ "seconds": "SECOND",
271
+ }
272
+
273
+ def _build_time_delta_sql(self, time_delta: TIME_DELTA) -> str:
274
+ """T-SQL DATEDIFF counts crossed boundaries of the given unit, so
275
+ compute the difference in SECONDS and divide by the seconds-per-
276
+ interval. T-SQL int/int division truncates toward zero, which equals
277
+ the FLOOR of the other dialects only for deltas >= 0 — callers must
278
+ guarantee non-negative deltas (the MM window filter does).
279
+
280
+ DATEDIFF(second, ...) returns int and overflows for spans > ~68
281
+ years; switch to DATEDIFF_BIG if that ever bites."""
282
+ start_sql: str = self.build_expression_sql(time_delta.start)
283
+ end_sql: str = self.build_expression_sql(time_delta.end)
284
+ multiplier: int = seconds_per_time_bucket(time_delta.unit, time_delta.count)
285
+ # Parenthesized so the form stays self-contained if a caller embeds
286
+ # TIME_DELTA in larger arithmetic (every other dialect wraps in FLOOR/cast).
287
+ return f"(DATEDIFF(second, {start_sql}, {end_sql}) / {multiplier})"
288
+
289
+ def _build_add_interval_sql(self, add_interval: ADD_INTERVAL) -> str:
290
+ timestamp_sql: str = self.build_expression_sql(add_interval.timestamp)
291
+ count_sql: str = self.build_expression_sql(add_interval.count_expression)
292
+ unit_name: str = self._TIME_BUCKET_UNIT_NAMES[add_interval.unit]
293
+ return f"DATEADD({unit_name}, {count_sql}, {timestamp_sql})"
294
+
295
+ def _build_percentile_within_group_sql(self, percentile_within_group: PERCENTILE_WITHIN_GROUP) -> str:
296
+ """T-SQL PERCENTILE_DISC is a window function only; the aggregate form
297
+ is APPROX_PERCENTILE_DISC (SQL Server 2022+/Azure SQL/Fabric,
298
+ https://learn.microsoft.com/en-us/sql/t-sql/functions/approx-percentile-disc-transact-sql)."""
299
+ expression_sql: str = self.build_expression_sql(percentile_within_group.expression)
300
+ return f"APPROX_PERCENTILE_DISC({percentile_within_group.percentile}) WITHIN GROUP (ORDER BY {expression_sql})"
301
+
179
302
  def _build_tuple_sql(self, tuple: TUPLE) -> str:
180
303
  if tuple.check_context(COUNT) and tuple.check_context(DISTINCT):
181
304
  return f"CHECKSUM{super()._build_tuple_sql(tuple)}"
@@ -211,6 +334,7 @@ class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
211
334
  order_by: list[str],
212
335
  limit: int,
213
336
  offset: int,
337
+ normalize_key_columns: frozenset[str] = frozenset(),
214
338
  ) -> str:
215
339
  where_clauses = []
216
340
 
@@ -221,7 +345,7 @@ class SqlServerSqlDialect(SqlDialect, sqlglot_dialect="tsql"):
221
345
  SELECT(columns or [STAR()]),
222
346
  FROM(table_name=dataset_identifier.dataset_name, table_prefix=dataset_identifier.prefixes),
223
347
  WHERE.optional(AND.optional(where_clauses)),
224
- *[ORDER_BY_ASC(c) for c in order_by],
348
+ *[term for c in order_by for term in self._order_by_key(c, normalize_key_columns)],
225
349
  OFFSET(offset),
226
350
  LIMIT(limit),
227
351
  ]
@@ -107,6 +107,10 @@ def handle_datetimeoffset(dto_value):
107
107
 
108
108
  class SqlServerDataSourceConnection(DataSourceConnection):
109
109
  def __init__(self, name: str, connection_properties: DataSourceConnectionProperties):
110
+ # Set before super().__init__(), which auto-opens the connection and
111
+ # populates these from the live server in _create_connection.
112
+ self.server_major_version: Optional[int] = None
113
+ self.engine_edition: Optional[int] = None
110
114
  super().__init__(name, connection_properties)
111
115
 
112
116
  # Normalize pyodbc.Row objects so downstream consumers see plain tuples.
@@ -190,10 +194,43 @@ class SqlServerDataSourceConnection(DataSourceConnection):
190
194
 
191
195
  self.connection.add_output_converter(-155, handle_datetimeoffset)
192
196
  self.connection.add_output_converter(-150, handle_datetime)
197
+ self._detect_server_info(self.connection)
193
198
  return self.connection
194
199
  except Exception as e:
195
200
  raise DataSourceConnectionException(e) from e
196
201
 
202
+ @staticmethod
203
+ def _parse_server_major_version(dbms_version: Optional[str]) -> Optional[int]:
204
+ """Parse the leading integer of an ODBC SQL_DBMS_VER string, e.g. '15.00.4123' -> 15."""
205
+ if not dbms_version:
206
+ return None
207
+ try:
208
+ return int(str(dbms_version).split(".")[0])
209
+ except (ValueError, IndexError):
210
+ return None
211
+
212
+ def _detect_server_info(self, connection) -> None:
213
+ """Detect raw engine facts once per connect; the data source syncs them onto
214
+ the dialect, which derives version-dependent capabilities from them (e.g.
215
+ APPROX_PERCENTILE_DISC needs SQL Server 2022+ or Azure SQL DB/MI).
216
+
217
+ The product version comes from the driver's login handshake — no extra
218
+ round-trip; EngineEdition costs one query. Detection is never fatal for an
219
+ otherwise healthy connect: on failure a warning is logged and the fact stays
220
+ None, which capability checks treat as "assume the newest engine".
221
+ """
222
+ try:
223
+ self.server_major_version = self._parse_server_major_version(connection.getinfo(pyodbc.SQL_DBMS_VER))
224
+ except Exception as e:
225
+ logger.warning(f"Could not determine SQL Server product version: {e}")
226
+ try:
227
+ with connection.cursor() as cursor:
228
+ cursor.execute("SELECT CAST(SERVERPROPERTY('EngineEdition') AS INT)")
229
+ row = cursor.fetchone()
230
+ self.engine_edition = row[0] if row is not None else None
231
+ except Exception as e:
232
+ logger.warning(f"Could not determine SQL Server engine edition: {e}")
233
+
197
234
  def _execute_query_get_result_row_column_name(self, column) -> str:
198
235
  return column[0]
199
236
 
@@ -1,9 +1,9 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: soda-sqlserver
3
- Version: 4.18.0
3
+ Version: 4.20.0
4
4
  Summary: Soda SQL Server V4
5
5
  Author-email: "Soda Data N.V." <info@soda.io>
6
6
  License: Proprietary
7
7
  Requires-Python: >=3.10
8
- Requires-Dist: soda-core==4.18.0
8
+ Requires-Dist: soda-core==4.20.0
9
9
  Requires-Dist: pyodbc
@@ -0,0 +1,2 @@
1
+ soda-core==4.20.0
2
+ pyodbc
@@ -1,2 +0,0 @@
1
- soda-core==4.18.0
2
- pyodbc