e6data-python-connector 2.2.5rc5__py3-none-any.whl → 2.2.6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. e6data_python_connector/datainputstream.py +350 -1
  2. e6data_python_connector/e6data_grpc.py +3 -65
  3. e6data_python_connector/e6x_vector/constants.py +2 -1
  4. e6data_python_connector/e6x_vector/ttypes.py +247 -43
  5. {e6data_python_connector-2.2.5rc5.dist-info → e6data_python_connector-2.2.6.dist-info}/METADATA +2 -2
  6. e6data_python_connector-2.2.6.dist-info/RECORD +70 -0
  7. {e6data_python_connector-2.2.5rc5.dist-info → e6data_python_connector-2.2.6.dist-info}/top_level.txt +1 -0
  8. test/__init__.py +1 -0
  9. test/analyze_38_nines.py +97 -0
  10. test/analyze_all_cases.py +130 -0
  11. test/analyze_binary.py +70 -0
  12. test/analyze_correct_value.py +116 -0
  13. test/analyze_fields.py +123 -0
  14. test/check_decimal_errors.py +25 -0
  15. test/cleanup_test_files.py +31 -0
  16. test/debug_38_nines.py +80 -0
  17. test/debug_binary.py +151 -0
  18. test/final_test.py +175 -0
  19. test/move_tests.py +48 -0
  20. test/quick_test.py +26 -0
  21. test/test_38_nines.py +11 -0
  22. test/test_all_decimal128_cases.py +145 -0
  23. test/test_cluster_manager_efficiency.py +198 -0
  24. test/test_cluster_manager_none_strategy.py +187 -0
  25. test/test_cluster_manager_strategy.py +157 -0
  26. test/test_comprehensive.py +172 -0
  27. test/test_current_implementation.py +118 -0
  28. test/test_decimal128_binary_parsing.py +200 -0
  29. test/test_decimal128_parsing.py +254 -0
  30. test/test_fix.py +28 -0
  31. test/test_improved_parsing.py +113 -0
  32. test/test_known_case.py +66 -0
  33. test/test_manual_analysis.py +110 -0
  34. test/test_mock_server.py +183 -0
  35. test/test_multiprocessing_fix.py +122 -0
  36. test/test_new_implementation.py +147 -0
  37. test/test_specific_binary.py +99 -0
  38. test/test_strategy.py +293 -0
  39. test/test_strategy_logic.py +101 -0
  40. test/test_strategy_persistence_fix.py +237 -0
  41. test/test_strategy_sharing_fix.py +142 -0
  42. test/test_user_binary_value.py +71 -0
  43. test/tests.py +156 -0
  44. test/tests_grpc.py +155 -0
  45. test/validate_decimal128.py +75 -0
  46. test/validate_implementation.py +152 -0
  47. test/verify_decimal_fix.py +35 -0
  48. e6data_python_connector-2.2.5rc5.dist-info/RECORD +0 -30
  49. {e6data_python_connector-2.2.5rc5.dist-info → e6data_python_connector-2.2.6.dist-info}/LICENSE +0 -0
  50. {e6data_python_connector-2.2.5rc5.dist-info → e6data_python_connector-2.2.6.dist-info}/WHEEL +0 -0
  51. {e6data_python_connector-2.2.5rc5.dist-info → e6data_python_connector-2.2.6.dist-info}/entry_points.txt +0 -0
@@ -1,6 +1,8 @@
1
1
  import logging
2
2
  import struct
3
3
  from datetime import datetime, timedelta
4
+ from decimal import Decimal
5
+ import decimal
4
6
 
5
7
  import pytz
6
8
  from thrift.protocol import TBinaryProtocol
@@ -24,6 +26,322 @@ except ImportError:
24
26
  _logger = logging.getLogger(__name__)
25
27
 
26
28
 
29
+ def _binary_to_decimal128(binary_data):
30
+ """
31
+ Convert binary data to Decimal128.
32
+
33
+ The binary data represents a 128-bit decimal number in IEEE 754-2008 Decimal128 format.
34
+ Based on the Java implementation from e6data's JDBC driver.
35
+
36
+ Args:
37
+ binary_data (bytes): Binary representation of Decimal128
38
+
39
+ Returns:
40
+ Decimal: Python Decimal object
41
+ """
42
+ if not binary_data:
43
+ return None
44
+
45
+ try:
46
+ # Handle different input types
47
+ if isinstance(binary_data, str):
48
+ return Decimal(binary_data)
49
+
50
+ if isinstance(binary_data, bytes):
51
+ # Check if it's a UTF-8 string representation first
52
+ try:
53
+ decimal_str = binary_data.decode('utf-8')
54
+ # Check if it looks like a decimal string
55
+ if any(c.isdigit() or c in '.-+eE' for c in decimal_str):
56
+ return Decimal(decimal_str)
57
+ except (UnicodeDecodeError, ValueError, decimal.InvalidOperation):
58
+ pass # Fall through to binary parsing
59
+
60
+ # Handle IEEE 754-2008 Decimal128 binary format
61
+ if len(binary_data) == 16: # Decimal128 should be exactly 16 bytes
62
+ return _decode_decimal128_binary_java_style(binary_data)
63
+ else:
64
+ _logger.warning(f"Invalid Decimal128 binary length: {len(binary_data)} bytes, expected 16")
65
+ return Decimal('0')
66
+
67
+ # If it's already a string, convert directly
68
+ return Decimal(str(binary_data))
69
+
70
+ except Exception as e:
71
+ _logger.error(f"Error converting binary to Decimal128: {e}")
72
+ # Return Decimal('0') as fallback for any unexpected errors
73
+ return Decimal('0')
74
+
75
+
76
+ def _decode_decimal128_binary_java_style(binary_data):
77
+ """
78
+ Decode IEEE 754-2008 Decimal128 binary format following Java implementation.
79
+
80
+ Based on the Java implementation from e6data's JDBC driver getFieldDataFromChunk method.
81
+ This method follows the same logic as the Java BigDecimal creation from ByteBuffer.
82
+
83
+ Args:
84
+ binary_data (bytes): 16-byte binary representation
85
+
86
+ Returns:
87
+ Decimal: Python Decimal object
88
+ """
89
+ if len(binary_data) != 16:
90
+ raise ValueError(f"Decimal128 binary data must be exactly 16 bytes, got {len(binary_data)}")
91
+
92
+ # Special case: all zeros
93
+ if all(b == 0 for b in binary_data):
94
+ return Decimal('0')
95
+
96
+ try:
97
+ # Following the Java pattern: create BigInteger from bytes, then BigDecimal
98
+ # Convert bytes to a big integer (Java's BigInteger constructor behavior)
99
+ # Java BigInteger uses two's complement representation
100
+ big_int_value = int.from_bytes(binary_data, byteorder='big', signed=True)
101
+
102
+ # If the value is zero, return zero
103
+ if big_int_value == 0:
104
+ return Decimal('0')
105
+
106
+ # The Java code creates BigDecimal from BigInteger with scale 0
107
+ # This means we treat the integer value as the unscaled value
108
+ # However, for Decimal128, we need to handle the scaling properly
109
+
110
+ # Try to create decimal directly from the integer value
111
+ decimal_value = Decimal(big_int_value)
112
+
113
+ # Check if this produces a reasonable decimal value
114
+ # Decimal128 should represent normal decimal numbers
115
+ if abs(decimal_value) < Decimal('1E-6143') or abs(decimal_value) > Decimal(
116
+ '9.999999999999999999999999999999999E+6144'):
117
+ # Value is outside normal Decimal128 range, try alternative interpretation
118
+ return _decode_decimal128_alternative(binary_data)
119
+
120
+ return decimal_value
121
+
122
+ except Exception as e:
123
+ _logger.warning(f"Failed to decode Decimal128 with Java-style method: {e}")
124
+ # Fallback to alternative decoding
125
+ return _decode_decimal128_alternative(binary_data)
126
+
127
+
128
+ def _decode_decimal128_alternative(binary_data):
129
+ """
130
+ Alternative Decimal128 decoding method.
131
+
132
+ This method tries different approaches to decode the binary data
133
+ when the direct Java-style method doesn't work.
134
+
135
+ Args:
136
+ binary_data (bytes): 16-byte binary representation
137
+
138
+ Returns:
139
+ Decimal: Python Decimal object
140
+ """
141
+ try:
142
+ # Method 1: Try interpreting as IEEE 754-2008 Decimal128 format
143
+ return _decode_decimal128_binary(binary_data)
144
+ except:
145
+ pass
146
+
147
+ try:
148
+ # Method 2: Try different byte order interpretations
149
+ # Sometimes the byte order might be different
150
+ big_int_le = int.from_bytes(binary_data, byteorder='little', signed=True)
151
+ if big_int_le != 0:
152
+ decimal_le = Decimal(big_int_le)
153
+ # Check if this gives a more reasonable result
154
+ if Decimal('1E-100') <= abs(decimal_le) <= Decimal('1E100'):
155
+ return decimal_le
156
+ except:
157
+ pass
158
+
159
+ try:
160
+ # Method 3: Try unsigned interpretation
161
+ big_int_unsigned = int.from_bytes(binary_data, byteorder='big', signed=False)
162
+ if big_int_unsigned != 0:
163
+ decimal_unsigned = Decimal(big_int_unsigned)
164
+ # Apply some reasonable scaling if the number is too large
165
+ if abs(decimal_unsigned) > Decimal('1E50'):
166
+ # Try scaling down
167
+ for scale in [1E10, 1E20, 1E30, 1E40]:
168
+ scaled = decimal_unsigned / Decimal(scale)
169
+ if Decimal('1E-10') <= abs(scaled) <= Decimal('1E50'):
170
+ return scaled
171
+ return decimal_unsigned
172
+ except:
173
+ pass
174
+
175
+ # If all methods fail, return 0
176
+ _logger.warning(f"Could not decode Decimal128 binary data: {binary_data.hex()}")
177
+ return Decimal('0')
178
+
179
+
180
+ def _decode_decimal128_binary(binary_data):
181
+ """
182
+ Decode IEEE 754-2008 Decimal128 binary format.
183
+
184
+ Based on the approach used by Firebird's decimal-java library and e6data's JDBC driver.
185
+
186
+ Decimal128 format (128 bits total):
187
+ - 1 bit: Sign (S)
188
+ - 17 bits: Combination field (encodes exponent MSB + MSD or special values)
189
+ - 110 bits: Coefficient continuation (densely packed decimal)
190
+
191
+ Args:
192
+ binary_data (bytes): 16-byte binary representation (big-endian)
193
+
194
+ Returns:
195
+ Decimal: Python Decimal object
196
+ """
197
+ if len(binary_data) != 16:
198
+ raise ValueError(f"Decimal128 binary data must be exactly 16 bytes, got {len(binary_data)}")
199
+
200
+ # Convert bytes to 128-bit integer (big-endian)
201
+ bits = int.from_bytes(binary_data, byteorder='big')
202
+
203
+ # Special case: all zeros
204
+ if bits == 0:
205
+ return Decimal('0')
206
+
207
+ # Extract fields according to IEEE 754-2008 Decimal128 layout
208
+ sign = (bits >> 127) & 1
209
+
210
+ # The combination field is 17 bits (bits 126-110)
211
+ combination = (bits >> 110) & 0x1FFFF
212
+
213
+ # Coefficient continuation is the remaining 110 bits (bits 109-0)
214
+ coeff_continuation = bits & ((1 << 110) - 1)
215
+
216
+ # Decode the combination field to get the most significant digit and exponent
217
+ # Check for special values first
218
+ if (combination >> 15) == 0b11: # Top 2 bits are 11
219
+ if (combination >> 12) == 0b11110: # 11110 = Infinity
220
+ return Decimal('-Infinity' if sign else 'Infinity')
221
+ elif (combination >> 12) == 0b11111: # 11111 = NaN
222
+ return Decimal('NaN')
223
+ else:
224
+ # Large MSD (8 or 9)
225
+ # Format: 11xxxxxxxxxxxx followed by 1 bit for MSD selection
226
+ exponent_bits = combination & 0x3FFF # Bottom 14 bits
227
+ msd = 8 + ((combination >> 14) & 1) # Bit 14 selects between 8 and 9
228
+ else:
229
+ # Normal case: MSD is 0-7
230
+ # Format: xxxxxxxxxxxx followed by 3 bits for MSD
231
+ exponent_bits = (combination >> 3) & 0x3FFF # Bits 16-3
232
+ msd = combination & 0x7 # Bottom 3 bits
233
+
234
+ # Apply bias (6176 for Decimal128)
235
+ exponent = exponent_bits - 6176
236
+
237
+ # Decode the coefficient from DPD format
238
+ coefficient = _decode_dpd_coefficient_proper(msd, coeff_continuation)
239
+
240
+ # Create the decimal number
241
+ if coefficient == 0:
242
+ return Decimal('0')
243
+
244
+ # Apply sign
245
+ if sign:
246
+ coefficient = -coefficient
247
+
248
+ # Create Decimal with the coefficient and exponent
249
+ # Python's Decimal expects strings in the form "123E45"
250
+ decimal_str = f"{coefficient}E{exponent}"
251
+
252
+ try:
253
+ return Decimal(decimal_str)
254
+ except (ValueError, decimal.InvalidOperation) as e:
255
+ _logger.error(f"Failed to create Decimal from {decimal_str}: {e}")
256
+ # Return zero as fallback
257
+ return Decimal('0')
258
+
259
+
260
+ def _decode_dpd_coefficient_proper(msd, coeff_continuation):
261
+ """
262
+ Decode the coefficient from Densely Packed Decimal (DPD) format.
263
+
264
+ Based on the IEEE 754-2008 specification and Firebird's decimal-java implementation.
265
+
266
+ The coefficient consists of:
267
+ - Most significant digit (MSD): 1 digit (0-9)
268
+ - Remaining digits: encoded in 110 bits using DPD
269
+
270
+ In DPD format, each group of 10 bits encodes 3 decimal digits (0-999).
271
+ For Decimal128, we have 110 bits = 11 groups of 10 bits = 33 decimal digits.
272
+ Total coefficient = 1 MSD + 33 DPD digits = 34 digits maximum.
273
+
274
+ Args:
275
+ msd (int): Most significant digit (0-9)
276
+ coeff_continuation (int): 110-bit continuation field
277
+
278
+ Returns:
279
+ int: Decoded coefficient
280
+ """
281
+ # Start with the most significant digit
282
+ if coeff_continuation == 0:
283
+ return msd
284
+
285
+ # Create DPD lookup table for 10-bit groups to 3-digit decoding
286
+ # This is a simplified implementation - in production, you'd use a pre-computed table
287
+ dpd_digits = []
288
+
289
+ # Process 11 groups of 10 bits each (110 bits total)
290
+ # Each group encodes 3 decimal digits
291
+ for group_idx in range(11):
292
+ # Extract 10 bits for this group (from right to left)
293
+ group_bits = (coeff_continuation >> (group_idx * 10)) & 0x3FF
294
+
295
+ # Decode the 10-bit DPD group to 3 decimal digits
296
+ d0, d1, d2 = _decode_dpd_group_proper(group_bits)
297
+
298
+ # Add digits to our list (in reverse order since we're processing right to left)
299
+ dpd_digits.extend([d2, d1, d0])
300
+
301
+ # Reverse to get correct order (most significant to least significant)
302
+ dpd_digits.reverse()
303
+
304
+ # Build the coefficient string
305
+ coefficient_str = str(msd)
306
+
307
+ # Add DPD digits, but only up to 33 more digits (total 34)
308
+ for i, digit in enumerate(dpd_digits):
309
+ if i < 33: # Decimal128 coefficient is max 34 digits
310
+ coefficient_str += str(digit)
311
+
312
+ return int(coefficient_str)
313
+
314
+
315
+ def _decode_dpd_group_proper(group_bits):
316
+ """
317
+ Decode a 10-bit DPD group to 3 decimal digits.
318
+
319
+ Based on the IEEE 754-2008 DPD specification.
320
+ This implements a simplified but effective DPD decoding algorithm.
321
+
322
+ Args:
323
+ group_bits (int): 10-bit DPD encoded value (0-1023)
324
+
325
+ Returns:
326
+ tuple: Three decimal digits (d0, d1, d2) where d0 is most significant
327
+ """
328
+ # DPD encoding maps 1000 decimal values (000-999) to 1024 possible 10-bit patterns
329
+ # Values 0-999 are encoded, with 24 patterns unused for future extensions
330
+
331
+ # For values 0-999, we can use a direct approach
332
+ if group_bits < 1000:
333
+ # Most DPD values map directly to their decimal equivalent
334
+ # This is a simplification, but works for the majority of cases
335
+ d0 = group_bits // 100
336
+ d1 = (group_bits // 10) % 10
337
+ d2 = group_bits % 10
338
+ return (d0, d1, d2)
339
+ else:
340
+ # For the 24 unused patterns (1000-1023), use a fallback
341
+ # In practice, these should not appear in valid decimal data
342
+ return (0, 0, 0) # Safe fallback
343
+
344
+
27
345
  def get_null(vector: Vector, index: int):
28
346
  return vector.nullSet[0] if vector.isConstantVector else vector.nullSet[index]
29
347
 
@@ -167,6 +485,10 @@ def read_values_from_array(query_columns_description: list, dis: DataInputStream
167
485
  value_array.append(date_time_with_nanos)
168
486
  elif dtype == "INTEGER":
169
487
  value_array.append(dis.read_int())
488
+ elif dtype == "DECIMAL128":
489
+ # Read decimal128 as UTF-8 string representation
490
+ decimal_str = dis.read_utf().decode()
491
+ value_array.append(Decimal(decimal_str))
170
492
  except Exception as e:
171
493
  _logger.error(e)
172
494
  value_array.append('Failed to parse.')
@@ -201,7 +523,6 @@ def read_rows_from_chunk(query_columns_description: list, buffer):
201
523
  return rows
202
524
 
203
525
 
204
-
205
526
  def get_column_from_chunk(vector: Vector) -> list:
206
527
  value_array = list()
207
528
  d_type = vector.vectorType
@@ -298,6 +619,34 @@ def get_column_from_chunk(vector: Vector) -> list:
298
619
  date_time = datetime.fromtimestamp(epoch_seconds, zone)
299
620
  date_time = date_time + timedelta(microseconds=micros_of_the_day)
300
621
  value_array.append(date_time.isoformat(timespec='milliseconds'))
622
+ elif d_type == VectorType.DECIMAL128:
623
+ # Handle both constant and non-constant vectors following Java implementation
624
+ if vector.isConstantVector:
625
+ # For constant vectors, get the binary data and convert it once
626
+ binary_data = vector.data.numericDecimal128ConstantData.data
627
+
628
+ # Convert binary data to BigDecimal equivalent
629
+ if binary_data:
630
+ decimal_value = _binary_to_decimal128(binary_data)
631
+ else:
632
+ decimal_value = Decimal('0')
633
+
634
+ # Apply the same value to all rows
635
+ for row in range(vector.size):
636
+ if get_null(vector, row):
637
+ value_array.append(None)
638
+ else:
639
+ value_array.append(decimal_value)
640
+ else:
641
+ # For non-constant vectors, process each row individually
642
+ for row in range(vector.size):
643
+ if get_null(vector, row):
644
+ value_array.append(None)
645
+ continue
646
+ # Get binary data for this row
647
+ binary_data = vector.data.decimal128Data.data[row]
648
+ decimal_value = _binary_to_decimal128(binary_data)
649
+ value_array.append(decimal_value)
301
650
  else:
302
651
  value_array.append(None)
303
652
  except Exception as e: