e6data-python-connector 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,396 @@
1
+ """Integration between SQLAlchemy and Hive.
2
+ Some code based on
3
+ https://github.com/zzzeek/sqlalchemy/blob/rel_0_5/lib/sqlalchemy/databases/sqlite.py
4
+ which is released under the MIT license.
5
+ """
6
+
7
+ from __future__ import absolute_import
8
+ from __future__ import unicode_literals
9
+
10
+ import datetime
11
+ import decimal
12
+ import logging
13
+ import re
14
+ import urllib
15
+ from decimal import Decimal
16
+ from io import BytesIO
17
+
18
+ from dateutil.parser import parse
19
+ from sqlalchemy import exc
20
+ from sqlalchemy import processors
21
+ from sqlalchemy import types
22
+ from sqlalchemy import util
23
+ # TODO shouldn't use mysql type
24
+ from sqlalchemy.databases import mysql
25
+ from sqlalchemy.engine import default, Engine, Connection
26
+ from sqlalchemy.sql import compiler
27
+ from sqlalchemy.sql.compiler import SQLCompiler
28
+
29
+ from e6xdb import e6x
30
+ from e6xdb.common import UniversalSet
31
+ from e6xdb.datainputstream import DataInputStream, get_query_columns_info
32
+ from e6xdb.exceptions import *
33
+
34
+ _logger = logging.getLogger(__name__)
35
+
36
+
37
+ class E6xStringTypeBase(types.TypeDecorator):
38
+ """Translates strings returned by Thrift into something else"""
39
+ impl = types.String
40
+
41
+ def process_bind_param(self, value, dialect):
42
+ raise NotImplementedError("Writing to Hive not supported")
43
+
44
+
45
+ class E6xDate(E6xStringTypeBase):
46
+ """Translates date strings to date objects"""
47
+ impl = types.DATE
48
+
49
+ def process_result_value(self, value, dialect):
50
+ return processors.str_to_date(value)
51
+
52
+ def result_processor(self, dialect, coltype):
53
+ def process(value):
54
+ if isinstance(value, datetime.datetime):
55
+ return value.date()
56
+ elif isinstance(value, datetime.date):
57
+ return value
58
+ elif value is not None:
59
+ return parse(value).date()
60
+ else:
61
+ return None
62
+
63
+ return process
64
+
65
+ def adapt(self, impltype, **kwargs):
66
+ return self.impl
67
+
68
+
69
+ class E6xTimestamp(E6xStringTypeBase):
70
+ """Translates timestamp strings to datetime objects"""
71
+ impl = types.TIMESTAMP
72
+
73
+ def process_result_value(self, value, dialect):
74
+ return processors.str_to_datetime(value)
75
+
76
+ def result_processor(self, dialect, coltype):
77
+ def process(value):
78
+ if isinstance(value, datetime.datetime):
79
+ return value
80
+ elif value is not None:
81
+ return parse(value)
82
+ else:
83
+ return None
84
+
85
+ return process
86
+
87
+ def adapt(self, impltype, **kwargs):
88
+ return self.impl
89
+
90
+
91
+ class E6xDecimal(E6xStringTypeBase):
92
+ """Translates strings to decimals"""
93
+ impl = types.DECIMAL
94
+
95
+ def process_result_value(self, value, dialect):
96
+ if value is not None:
97
+ return decimal.Decimal(value)
98
+ else:
99
+ return None
100
+
101
+ def result_processor(self, dialect, coltype):
102
+ def process(value):
103
+ if isinstance(value, Decimal):
104
+ return value
105
+ elif value is not None:
106
+ return Decimal(value)
107
+ else:
108
+ return None
109
+
110
+ return process
111
+
112
+ def adapt(self, impltype, **kwargs):
113
+ return self.impl
114
+
115
+
116
+ class E6xIdentifierPreparer(compiler.IdentifierPreparer):
117
+ # Just quote everything to make things simpler / easier to upgrade
118
+ reserved_words = UniversalSet()
119
+
120
+ def __init__(self, dialect):
121
+ super(E6xIdentifierPreparer, self).__init__(
122
+ dialect,
123
+ initial_quote='"',
124
+ )
125
+
126
+
127
+ _type_map = {
128
+ 'boolean': types.Boolean,
129
+ 'tinyint': mysql.MSTinyInteger,
130
+ 'smallint': types.SmallInteger,
131
+ 'integer': types.Integer,
132
+ 'bigint': types.BigInteger,
133
+ 'float': types.Float,
134
+ 'double': types.Float,
135
+ 'string': types.String,
136
+ 'varchar': types.String,
137
+ 'char': types.String,
138
+ 'date': E6xDate,
139
+ 'timestamp': E6xTimestamp,
140
+ 'binary': types.String,
141
+ 'array': types.String,
142
+ 'map': types.String,
143
+ 'struct': types.String,
144
+ 'uniontype': types.String,
145
+ 'decimal': E6xDecimal,
146
+ }
147
+
148
+
149
+ class E6xCompiler(SQLCompiler):
150
+ def visit_concat_op_binary(self, binary, operator, **kw):
151
+ return "concat(%s, %s)" % (self.process(binary.left), self.process(binary.right))
152
+
153
+ def visit_insert(self, *args, **kwargs):
154
+ raise NotSupportedError()
155
+
156
+ def visit_column(self, *args, **kwargs):
157
+ result = super(E6xCompiler, self).visit_column(*args, **kwargs)
158
+ return result
159
+
160
+ def visit_char_length_func(self, fn, **kw):
161
+ return 'length{}'.format(self.function_argspec(fn, **kw))
162
+
163
+
164
+ class E6xTypeCompiler(compiler.GenericTypeCompiler):
165
+ def visit_INTEGER(self, type_, **kwargs):
166
+ return 'INT'
167
+
168
+ def visit_NUMERIC(self, type_, **kwargs):
169
+ return 'DECIMAL'
170
+
171
+ def visit_CHAR(self, type_, **kwargs):
172
+ return 'STRING'
173
+
174
+ def visit_VARCHAR(self, type_, **kwargs):
175
+ return 'STRING'
176
+
177
+ def visit_NCHAR(self, type_, **kwargs):
178
+ return 'STRING'
179
+
180
+ def visit_TEXT(self, type_, **kwargs):
181
+ return 'STRING'
182
+
183
+ def visit_CLOB(self, type_, **kwargs):
184
+ return 'STRING'
185
+
186
+ def visit_BLOB(self, type_, **kwargs):
187
+ return 'BINARY'
188
+
189
+ def visit_TIME(self, type_, **kwargs):
190
+ return 'TIMESTAMP'
191
+
192
+ def visit_DATE(self, type_, **kwargs):
193
+ return 'DATE'
194
+
195
+ def visit_DATETIME(self, type_, **kwargs):
196
+ return 'TIMESTAMP'
197
+
198
+
199
+ class E6xDialect(default.DefaultDialect):
200
+ preparer = E6xIdentifierPreparer
201
+ statement_compiler = E6xCompiler
202
+ supports_views = True
203
+ supports_alter = True
204
+ supports_pk_autoincrement = False
205
+ supports_default_values = False
206
+ supports_empty_insert = False
207
+ supports_native_decimal = True
208
+ supports_native_boolean = True
209
+ supports_unicode_statements = True
210
+ supports_unicode_binds = True
211
+ returns_unicode_strings = True
212
+ description_encoding = None
213
+ supports_multivalues_insert = True
214
+ type_compiler = E6xTypeCompiler
215
+ supports_sane_rowcount = False
216
+ driver = b'thrift'
217
+ name = b'e6x'
218
+ scheme = 'e6xdb'
219
+
220
+ def _dialect_specific_select_one(self):
221
+ return "NOOP"
222
+
223
+ @classmethod
224
+ def dbapi(cls):
225
+ return e6x
226
+
227
+ def create_connect_args(self, url):
228
+ db = None
229
+ if url.query.get("schema"):
230
+ db = url.query.get("schema")
231
+ kwargs = {
232
+ "host": url.host,
233
+ "port": url.port or 10000,
234
+ "scheme": self.scheme,
235
+ "username": url.username or None,
236
+ "password": url.password or None,
237
+ "database": db
238
+ }
239
+ return [], kwargs
240
+
241
+ def get_schema_names(self, connection, **kw):
242
+ # Equivalent to SHOW DATABASES
243
+ # Rerouting to view names
244
+ engine = connection
245
+ if isinstance(connection, Engine):
246
+ cursor = connection.raw_connection().connection.cursor()
247
+ elif isinstance(connection, Connection):
248
+ cursor = connection.connection.cursor()
249
+ else:
250
+ raise Exception("Got type of object {typ}".format(typ=type(connection)))
251
+
252
+ client = cursor.connection
253
+ return client.get_schema_names()
254
+
255
+ def get_view_names(self, connection, schema=None, **kw):
256
+ return []
257
+
258
+ def _get_table_columns(self, connection, table):
259
+ try:
260
+ cursor = None
261
+ if isinstance(connection, Engine):
262
+ cursor = connection.raw_connection().connection.cursor()
263
+ elif isinstance(connection, Connection):
264
+ cursor = connection.connection.cursor()
265
+ else:
266
+ raise Exception("Got type of object {typ}".format(typ=type(connection)))
267
+
268
+ client = cursor.connection
269
+ columns = client.getColumns("default", table)
270
+ rows = list()
271
+ for column in columns:
272
+ row = dict()
273
+ row["col_name"] = column.fieldName
274
+ row["data_type"] = column.fieldType
275
+ rows.append(row)
276
+
277
+ return rows
278
+ except exc.OperationalError as e:
279
+ # Does the table exist?
280
+ raise
281
+
282
+ def has_table(self, connection, table_name, schema=None, **kwargs):
283
+ try:
284
+ self._get_table_columns(connection, table_name)
285
+ return True
286
+ except Exception:
287
+ return False
288
+
289
+ def get_columns(self, connection, table_name, schema=None, **kw):
290
+ rows = self._get_table_columns(connection, table_name)
291
+ # # Strip whitespace
292
+ # rows = [[col.strip() if col else None for col in row] for row in rows]
293
+ # Filter out empty rows and comment
294
+ # rows = [row for row in rows if row[0] and row[0] != '# col_name']
295
+ result = []
296
+ for row in rows:
297
+ col_name = row['col_name']
298
+ col_type = row['data_type']
299
+ # Take out the more detailed type information
300
+ # e.g. 'map<int,int>' -> 'map'
301
+ # 'decimal(10,1)' -> decimal
302
+ col_type = re.search(r'^\w+', col_type).group(0)
303
+ try:
304
+ coltype = _type_map[col_type.lower()]
305
+ _logger.info("Got column {column} with data type {dt}".format(column=col_name, dt=coltype))
306
+ except KeyError:
307
+ util.warn("Did not recognize type '%s' of column '%s'" % (col_type, col_name))
308
+ coltype = types.NullType
309
+
310
+ result.append({
311
+ 'name': col_name,
312
+ 'type': coltype,
313
+ 'nullable': True,
314
+ 'default': None,
315
+ })
316
+ return result
317
+
318
+ def get_foreign_keys(self, connection, table_name, schema=None, **kw):
319
+ # Hive has no support for foreign keys.
320
+ return []
321
+
322
+ def get_pk_constraint(self, connection, table_name, schema=None, **kw):
323
+ # Hive has no support for primary keys.
324
+ return []
325
+
326
+ def get_indexes(self, connection, table_name, schema=None, **kw):
327
+ return []
328
+
329
+ def get_table_names(self, connection, schema=None, **kw):
330
+ # Hive does not provide functionality to query tableType
331
+ # This allows reflection to not crash at the cost of being inaccurate
332
+ engine = connection
333
+ if isinstance(connection, Engine):
334
+ cursor = connection.raw_connection().connection.cursor()
335
+ elif isinstance(connection, Connection):
336
+ cursor = connection.connection.cursor()
337
+ else:
338
+ raise Exception("Got type of object {typ}".format(typ=type(connection)))
339
+
340
+ client = cursor.connection
341
+ return client.getTables(schema)
342
+
343
+ def do_rollback(self, dbapi_connection):
344
+ # No transactions for Hive
345
+ pass
346
+
347
+ def _check_unicode_returns(self, connection, additional_tests=None):
348
+ # We decode everything as UTF-8
349
+ return True
350
+
351
+ def _check_unicode_description(self, connection):
352
+ # We decode everything as UTF-8
353
+ return True
354
+
355
+ def do_ping(self, connection):
356
+ # We do not need the ping api as we are using http
357
+ return True
358
+
359
+
360
+ # class E6xDialect(UniphiDialect):
361
+ # name = "e6xdb"
362
+ # scheme = "e6xdb"
363
+ # driver = "e6xdb"
364
+ #
365
+ # def create_connect_args(self, url):
366
+ # kwargs = {
367
+ # "connectStr": url
368
+ # }
369
+ # if url.query:
370
+ # kwargs.update(url.query)
371
+ # return [], kwargs
372
+ # return ([], kwargs)
373
+
374
+ class E6xdbE6xDialect(E6xDialect):
375
+ name = "e6x"
376
+ scheme = "e6xdb"
377
+ driver = "rest"
378
+
379
+ def create_connect_args(self, url):
380
+ db = None
381
+ if url.query:
382
+ db = url.query.split("=")[1]
383
+ kwargs = {
384
+ "host": url.hostname,
385
+ "port": url.port or 10000,
386
+ "scheme": self.scheme,
387
+ "username": url.username or None,
388
+ "password": url.password or None,
389
+ "database": db
390
+ }
391
+
392
+ return [], kwargs
393
+
394
+ def __init__(self):
395
+ super(E6xdbE6xDialect, self).__init__()
396
+
e6xdb/typeId.py ADDED
@@ -0,0 +1,73 @@
1
+ class TypeId(object):
2
+ BOOLEAN_TYPE = 0
3
+ TINYINT_TYPE = 1
4
+ SMALLINT_TYPE = 2
5
+ INT_TYPE = 3
6
+ BIGINT_TYPE = 4
7
+ FLOAT_TYPE = 5
8
+ DOUBLE_TYPE = 6
9
+ STRING_TYPE = 7
10
+ TIMESTAMP_TYPE = 8
11
+ BINARY_TYPE = 9
12
+ ARRAY_TYPE = 10
13
+ MAP_TYPE = 11
14
+ STRUCT_TYPE = 12
15
+ UNION_TYPE = 13
16
+ USER_DEFINED_TYPE = 14
17
+ DECIMAL_TYPE = 15
18
+ NULL_TYPE = 16
19
+ DATE_TYPE = 17
20
+ VARCHAR_TYPE = 18
21
+ CHAR_TYPE = 19
22
+ INTERVAL_YEAR_MONTH_TYPE = 20
23
+ INTERVAL_DAY_TIME_TYPE = 21
24
+
25
+ _VALUES_TO_NAMES = {
26
+ 0: "BOOLEAN_TYPE",
27
+ 1: "TINYINT_TYPE",
28
+ 2: "SMALLINT_TYPE",
29
+ 3: "INT_TYPE",
30
+ 4: "BIGINT_TYPE",
31
+ 5: "FLOAT_TYPE",
32
+ 6: "DOUBLE_TYPE",
33
+ 7: "STRING_TYPE",
34
+ 8: "TIMESTAMP_TYPE",
35
+ 9: "BINARY_TYPE",
36
+ 10: "ARRAY_TYPE",
37
+ 11: "MAP_TYPE",
38
+ 12: "STRUCT_TYPE",
39
+ 13: "UNION_TYPE",
40
+ 14: "USER_DEFINED_TYPE",
41
+ 15: "DECIMAL_TYPE",
42
+ 16: "NULL_TYPE",
43
+ 17: "DATE_TYPE",
44
+ 18: "VARCHAR_TYPE",
45
+ 19: "CHAR_TYPE",
46
+ 20: "INTERVAL_YEAR_MONTH_TYPE",
47
+ 21: "INTERVAL_DAY_TIME_TYPE",
48
+ }
49
+
50
+ _NAMES_TO_VALUES = {
51
+ "BOOLEAN_TYPE": 0,
52
+ "TINYINT_TYPE": 1,
53
+ "SMALLINT_TYPE": 2,
54
+ "INT_TYPE": 3,
55
+ "BIGINT_TYPE": 4,
56
+ "FLOAT_TYPE": 5,
57
+ "DOUBLE_TYPE": 6,
58
+ "STRING_TYPE": 7,
59
+ "TIMESTAMP_TYPE": 8,
60
+ "BINARY_TYPE": 9,
61
+ "ARRAY_TYPE": 10,
62
+ "MAP_TYPE": 11,
63
+ "STRUCT_TYPE": 12,
64
+ "UNION_TYPE": 13,
65
+ "USER_DEFINED_TYPE": 14,
66
+ "DECIMAL_TYPE": 15,
67
+ "NULL_TYPE": 16,
68
+ "DATE_TYPE": 17,
69
+ "VARCHAR_TYPE": 18,
70
+ "CHAR_TYPE": 19,
71
+ "INTERVAL_YEAR_MONTH_TYPE": 20,
72
+ "INTERVAL_DAY_TIME_TYPE": 21,
73
+ }