gitdata-lib 0.0.11__tar.gz → 0.0.12__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. {gitdata-lib-0.0.11/gitdata_lib.egg-info → gitdata_lib-0.0.12}/PKG-INFO +9 -5
  2. gitdata_lib-0.0.12/gitdata/__version__.py +1 -0
  3. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/cli/__init__.py +12 -1
  4. gitdata_lib-0.0.12/gitdata/cli/gitdata_scan.py +94 -0
  5. gitdata_lib-0.0.12/gitdata/connectors/__init__.py +5 -0
  6. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/connectors/common.py +76 -7
  7. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/connectors/fake.py +19 -0
  8. gitdata_lib-0.0.12/gitdata/connectors/http.py +59 -0
  9. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/connectors/local.py +33 -0
  10. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/common.py +0 -1
  11. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/graphs.py +6 -2
  12. gitdata_lib-0.0.12/gitdata/sql.py +235 -0
  13. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/utils.py +9 -0
  14. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12/gitdata_lib.egg-info}/PKG-INFO +9 -5
  15. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata_lib.egg-info/SOURCES.txt +3 -0
  16. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata_lib.egg-info/entry_points.txt +0 -1
  17. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata_lib.egg-info/requires.txt +0 -1
  18. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/requirements.txt +0 -1
  19. gitdata-lib-0.0.11/gitdata/__version__.py +0 -1
  20. gitdata-lib-0.0.11/gitdata/connectors/__init__.py +0 -5
  21. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/CHANGELOG.md +0 -0
  22. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/LICENSE +0 -0
  23. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/MANIFEST.in +0 -0
  24. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/README.md +0 -0
  25. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/__init__.py +0 -0
  26. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/assets/README.md +0 -0
  27. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/buckets.py +0 -0
  28. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/cli/gitdata_get.py +0 -0
  29. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/config.py +0 -0
  30. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/connectors/gitlab.py +0 -0
  31. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/__init__.py +0 -0
  32. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/mysql.py +0 -0
  33. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/mysql_setup.sql +0 -0
  34. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/mysql_setup_test_database.sql +0 -0
  35. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/postgresql.py +0 -0
  36. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/sqlite3.py +0 -0
  37. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/sqlite3_setup.sql +0 -0
  38. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/database/sqlite3_setup_test_data.sql +0 -0
  39. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/digester.py +0 -0
  40. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/ext/__init__.py +0 -0
  41. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/ext/connectors/__init__.py +0 -0
  42. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/json.py +0 -0
  43. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/queues.py +0 -0
  44. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/repositories.py +0 -0
  45. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/solutions.py +0 -0
  46. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/stores/__init__.py +0 -0
  47. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/stores/common.py +0 -0
  48. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/stores/entities.py +0 -0
  49. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/stores/facts.py +0 -0
  50. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata/stores/tables.py +0 -0
  51. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata_lib.egg-info/dependency_links.txt +0 -0
  52. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/gitdata_lib.egg-info/top_level.txt +0 -0
  53. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/setup.cfg +0 -0
  54. {gitdata-lib-0.0.11 → gitdata_lib-0.0.12}/setup.py +0 -0
@@ -1,12 +1,10 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gitdata-lib
3
- Version: 0.0.11
3
+ Version: 0.0.12
4
4
  Summary: Data extraction and analysis library
5
5
  Home-page: https://github.com/gitdata/gitdata-lib
6
6
  Author: DSI Labs
7
7
  Author-email: support@gitdata.com
8
- License: UNKNOWN
9
- Platform: UNKNOWN
10
8
  Classifier: Development Status :: 1 - Planning
11
9
  Classifier: Environment :: Console
12
10
  Classifier: Intended Audience :: Developers
@@ -22,11 +20,17 @@ Classifier: Programming Language :: Python :: 3 :: Only
22
20
  Classifier: Topic :: Database :: Front-Ends
23
21
  Description-Content-Type: text/markdown
24
22
  License-File: LICENSE
23
+ Requires-Dist: python-decouple>=3.4
24
+ Requires-Dist: Unipath>=1.1
25
+ Requires-Dist: PyMySQL==0.10.1
26
+ Requires-Dist: psycopg2-binary==2.9.1
27
+ Requires-Dist: python-dotenv
28
+ Requires-Dist: requests
29
+ Requires-Dist: docopt
30
+ Requires-Dist: faker
25
31
 
26
32
  # GitData Lib
27
33
  Data Wrangling for Everyone.
28
34
 
29
35
  GitData is a fast, scalable, distributed data exploration system
30
36
  with a rich set of commands that provide ways to gather, manage and query data in an unusually straightforward way.
31
-
32
-
@@ -0,0 +1 @@
1
+ __version__ = '0.0.12'
@@ -9,6 +9,7 @@ options:
9
9
  The most commonly used gitdata commands are:
10
10
  fetch fetch data to the local reposotiry
11
11
  get get data
12
+ scan scan data
12
13
 
13
14
  See 'gitdata help <command>' for more information on a specific command.
14
15
  """
@@ -16,6 +17,7 @@ See 'gitdata help <command>' for more information on a specific command.
16
17
  import logging
17
18
 
18
19
  from docopt import docopt
20
+ import sys
19
21
 
20
22
  import gitdata
21
23
  from gitdata.utils import trim
@@ -32,6 +34,10 @@ def print_help(doc):
32
34
  def main():
33
35
  """CLI main"""
34
36
 
37
+ if len(sys.argv) == 1:
38
+ print_help(__doc__)
39
+ sys.exit()
40
+
35
41
  args = docopt(
36
42
  __doc__,
37
43
  version='gitdata version {}'.format(gitdata.__version__),
@@ -49,6 +55,11 @@ def main():
49
55
  args = docopt(doc, argv=argv)
50
56
  get(args)
51
57
 
58
+ elif args['<command>'] == 'scan':
59
+ from gitdata.cli.gitdata_scan import scan_to_console, __doc__ as doc
60
+ args = docopt(doc, argv=argv)
61
+ scan_to_console(args)
62
+
52
63
  elif args['<command>'] in ['help', None]:
53
64
  if args['<args>'] == ['get']:
54
65
  from gitdata.cli.gitdata_get import __doc__ as doc
@@ -56,4 +67,4 @@ def main():
56
67
  exit(__doc__)
57
68
 
58
69
  else:
59
- exit("%r is not a git.py command. See 'git help'." % args['<command>'])
70
+ exit("%r is not a gitdata command. See 'gitdata help'." % args['<command>'])
@@ -0,0 +1,94 @@
1
+ """
2
+ usage: gitdata scan [options] [<ref>...]
3
+
4
+ options:
5
+ --limit <n>, -n <n> limit number of records displayed
6
+ --name-columns provide synthetic column names (handy for csv files that do not provide a header)
7
+ -h, --help
8
+ """
9
+
10
+ import csv
11
+ import os
12
+
13
+ import gitdata
14
+ from gitdata.connectors.common import load, connect
15
+ from gitdata.connectors.http import HttpConnector
16
+
17
+
18
+ def as_blobs(ref):
19
+ if os.path.isfile(ref):
20
+ with open(ref, 'rb') as f:
21
+ return [f.read()]
22
+
23
+ if ref.startswith('http'):
24
+ http_connector = HttpConnector()
25
+ result = http_connector.get(ref)
26
+ if result:
27
+ return [result['blob'].read()]
28
+
29
+ return []
30
+
31
+
32
+ def as_tables(ref, args):
33
+ for blob in as_blobs(ref):
34
+ rows = []
35
+ try:
36
+ reader = csv.reader(blob.decode('utf8').splitlines())
37
+ except UnicodeDecodeError:
38
+ reader = csv.reader(blob.decode('Latin1').splitlines())
39
+ for row in reader:
40
+ rows.append(row)
41
+ yield gitdata.Record(
42
+ name=ref,
43
+ filename=ref,
44
+ rows=rows,
45
+ columns=[gitdata.Record(name=f'{chr(n+65)}', type='str') for n in range(len(rows[0]))]
46
+ )
47
+
48
+
49
+ def scan(ref, args):
50
+
51
+ def scan_table(table):
52
+ header = gitdata.Record(
53
+ name=table.name,
54
+ row_count=len(table.rows),
55
+ column_count=len(table.rows[0])
56
+ )
57
+
58
+ limit = args['--limit'] and int(args['--limit']) or 5
59
+
60
+ if args['--name-columns']:
61
+ rows = [[t.name for t in table.columns]] + table.rows[:limit]
62
+ else:
63
+ rows = table.rows[:limit]
64
+
65
+ sample = gitdata.utils.ItemList(rows)
66
+ return str(header) + '\n' + str(sample) + str(list())
67
+
68
+ return ''.join(scan_table(table) for table in as_tables(ref, args))
69
+
70
+
71
+ def console(output):
72
+ if output:
73
+ print(output)
74
+ print()
75
+
76
+
77
+ def scan_to_console(args):
78
+ if args['<ref>']:
79
+ for ref in args['<ref>']:
80
+ console(scan(ref, args))
81
+ # connection = connect(ref)
82
+ # if connection:
83
+ # print(connection)
84
+ # for table in connection.get_tables():
85
+ # print(f'scanning {table.name}')
86
+ # else:
87
+ # print(f'unable to connect to {ref}')
88
+
89
+ # g = load(ref)
90
+ # print('scanning', ref)
91
+ # print(repr(g))
92
+ # print(g)
93
+ else:
94
+ print(__doc__)
@@ -0,0 +1,5 @@
1
+ """
2
+ connectors
3
+ """
4
+
5
+ from .common import explore, connect, fetch, extract
@@ -12,16 +12,16 @@ import platform
12
12
 
13
13
  import gitdata.connectors
14
14
  import gitdata.ext.connectors
15
+ import gitdata.graphs
15
16
  import gitdata.solutions
16
17
 
17
-
18
18
  logger = logging.getLogger(__name__)
19
19
 
20
20
 
21
21
  def get_connectors():
22
22
  """generate connectors"""
23
- path = gitdata.connectors.__path__
24
23
 
24
+ path = gitdata.connectors.__path__
25
25
  for _, name, _ in pkgutil.iter_modules(path):
26
26
  if name != 'common':
27
27
  module = importlib.import_module('gitdata.connectors.' + name)
@@ -121,6 +121,15 @@ class BaseConnector:
121
121
  def connect(self, ref):
122
122
  """Connect to a ref"""
123
123
 
124
+ def extract(self, ref):
125
+ """Extract facts from a ref"""
126
+
127
+ def get_tables(self):
128
+ return []
129
+
130
+ def get_blobs(self):
131
+ return []
132
+
124
133
  # def explore(self, data):
125
134
  # """Explore a graph
126
135
  # """
@@ -131,17 +140,14 @@ def connect(ref):
131
140
  """Return a connection to a ref"""
132
141
 
133
142
  for connector in get_connectors():
134
- if connector.connects(ref):
135
- connection = connector()
136
- connection.connect(ref)
143
+ connection = connector().connect(ref)
144
+ if connection:
137
145
  return connection
138
146
 
139
147
 
140
148
  # connectors = {
141
149
  # connector.__name__: connector for connector in get_connectors()}
142
150
 
143
-
144
-
145
151
  # if 'System' in connectors:
146
152
  # return connectors['System']()
147
153
 
@@ -180,3 +186,66 @@ def get(ref):
180
186
  if facts:
181
187
  add_connection_metadata(facts)
182
188
  return facts
189
+
190
+ class Bunch:
191
+ """a handy bunch of variables"""
192
+ tables = []
193
+
194
+ def __init__(self, **kwargs):
195
+ self.__dict__.update(kwargs)
196
+
197
+ def __str__(self):
198
+ return f'{self.__class__.__name__}\n' + '\n '.join(
199
+ f'{k:{(20-len(k))*"."}}: {v!r}'
200
+ for k, v in self.__dict__.items()
201
+ )
202
+
203
+ class State(dict):
204
+ """State of Extraction"""
205
+
206
+ def find(self, kind):
207
+ """Return things of a kind"""
208
+ cache = self.setdefault('_cache', [])
209
+ for thing in self.setdefault(kind, []):
210
+ thing_id = id(thing)
211
+ if thing_id not in cache:
212
+ cache.append(thing_id)
213
+ yield thing
214
+
215
+ def add(self, kind, value):
216
+ """Add a value of a certain kind"""
217
+ self.setdefault(kind, []).append(value)
218
+
219
+
220
+ def extract(ref):
221
+ """Extract a reference
222
+
223
+ Extract data from a reference.
224
+ """
225
+
226
+ connectors = get_connectors()
227
+
228
+ # ref = State(ref=refs)
229
+
230
+ extracting = True
231
+ while extracting:
232
+ extracting = any(
233
+ connector().extract(ref)
234
+ for connector in connectors
235
+ )
236
+ # print(extracting)
237
+
238
+ return Bunch(
239
+ # name=ref.get('name', ref['ref']),
240
+ # ref=ref['ref'],
241
+ # tables=ref['tables'],
242
+ # text=ref['text'],
243
+ )
244
+
245
+
246
+ def load(ref):
247
+ """Get a Graph
248
+ """
249
+ graph = gitdata.graphs.Graph()
250
+ graph.add(get(ref))
251
+ return graph
@@ -41,9 +41,28 @@ class FakeConnector(BaseConnector):
41
41
  while True:
42
42
  yield self.address
43
43
 
44
+ @property
45
+ def location(self):
46
+ fake = faker.Faker()
47
+ lat, lng, name, country_code, timezone = fake.location_on_land()
48
+ return dict(
49
+ latitude=lat,
50
+ longitude=lng,
51
+ name=name,
52
+ country_code=country_code,
53
+ timezone=timezone,
54
+ )
55
+
56
+ @property
57
+ def locations(self):
58
+ while True:
59
+ yield self.location
60
+
44
61
  def get(self, ref):
45
62
  if ref.startswith('fake'):
63
+ print('ref is', ref)
46
64
  return dict(
47
65
  people=self.people,
48
66
  addresses=self.addresses,
67
+ locations=self.locations,
49
68
  )
@@ -0,0 +1,59 @@
1
+ """
2
+ http connector
3
+ """
4
+
5
+ import io
6
+ import logging
7
+ import urllib
8
+ import urllib.parse
9
+
10
+ import requests
11
+
12
+ from gitdata.connectors.common import BaseConnector
13
+
14
+
15
+ logger = logging.getLogger(__name__)
16
+
17
+
18
+ class HttpConnector(BaseConnector):
19
+
20
+ def get(self, ref):
21
+ """Get Data"""
22
+ if ref.startswith('http://') or ref.startswith('https://'):
23
+ logger.debug(
24
+ '%s get %r',
25
+ self.__class__.__name__,
26
+ ref
27
+ )
28
+
29
+ u = urllib.parse.urlparse(ref)
30
+ endpoint=urllib.parse.urldefrag(ref)[0]
31
+ facts = dict(
32
+ url=ref,
33
+ endpoint=endpoint,
34
+ scheme=u.scheme,
35
+ netloc=u.netloc,
36
+ username=u.username,
37
+ password=u.password,
38
+ hostname=u.hostname,
39
+ port=u.port,
40
+ path=u.path,
41
+ lpath=str(endpoint).lower(),
42
+ query=u.query,
43
+ fragment=u.fragment,
44
+ name=u.fragment or str(u.path).split('/')[-1],
45
+ )
46
+
47
+ r = requests.get(ref)
48
+ if r.status_code == 200:
49
+ return dict(
50
+ facts,
51
+ blob=io.BytesIO(r.content)
52
+ )
53
+ else:
54
+ logger.error(
55
+ 'status %s - %s get %r',
56
+ r.status_code,
57
+ self.__class__.__name__,
58
+ ref
59
+ )
@@ -16,6 +16,17 @@ class FileConnector(BaseConnector):
16
16
  reads = ['location']
17
17
  writes = ['text', 'blob', 'stdout']
18
18
 
19
+ # def connects(self, ref):
20
+ # return os.path.isfile(ref)
21
+
22
+ def connect(self, ref):
23
+ self.ref = ref
24
+ return os.path.isfile(ref)
25
+
26
+ def get_blobs(self):
27
+ with open(self.ref, 'rb') as f:
28
+ return [io.BytesIO(f.read())]
29
+
19
30
  def get(self, ref):
20
31
  if os.path.isfile(ref):
21
32
  stat = os.stat(ref)
@@ -26,6 +37,28 @@ class FileConnector(BaseConnector):
26
37
  content = io.BytesIO(f.read())
27
38
 
28
39
  return dict(
40
+ ref=ref,
41
+ filename=filename,
42
+ size=stat.st_size,
43
+ modified=fromtimestamp(stat.st_mtime),
44
+ created=fromtimestamp(stat.st_ctime),
45
+ path=path,
46
+ blob=content
47
+ )
48
+
49
+ def extract(self, ref):
50
+ """extract data from a ref"""
51
+ print('extracting', ref)
52
+ if os.path.isfile(ref):
53
+ stat = os.stat(ref)
54
+ pathname = os.path.realpath(ref)
55
+ path, filename = os.path.split(pathname)
56
+
57
+ with open(ref, 'rb') as f:
58
+ content = io.BytesIO(f.read())
59
+
60
+ return dict(
61
+ ref=ref,
29
62
  filename=filename,
30
63
  size=stat.st_size,
31
64
  modified=fromtimestamp(stat.st_mtime),
@@ -344,7 +344,6 @@ class Database:
344
344
  def get_table_names(self):
345
345
  """get list of database tables"""
346
346
 
347
-
348
347
  def get_schema_names(self):
349
348
  """Return schema names"""
350
349
  return []
@@ -253,8 +253,9 @@ class Graph:
253
253
  for k, v in kwargs.items():
254
254
  query.append(('?subject', k, v))
255
255
  subjects = set(record['subject'] for record in self.query(query))
256
- result = self.get(subjects) or []
257
- return result
256
+ if subjects:
257
+ return self.get(subjects) or []
258
+ return []
258
259
 
259
260
  def first(self, *args, **kwargs):
260
261
  """Find first node"""
@@ -272,3 +273,6 @@ class Graph:
272
273
 
273
274
  def __len__(self):
274
275
  return len(self.facts)
276
+
277
+ def __repr__(self):
278
+ return 'Graph({})'.format(len(self))
@@ -0,0 +1,235 @@
1
+ """
2
+ sql utilities
3
+ """
4
+
5
+ import inspect
6
+ import os
7
+
8
+ import gitdata
9
+ from gitdata.stores.tables import table_of
10
+ from gitdata import Record
11
+
12
+
13
+ from gitdata.utils import (
14
+ RecordList, load
15
+ )
16
+
17
+
18
+ def get_result_iterator(rows, kind, store=None):
19
+ """returns an iterator that iterates over the rows and zips the names onto
20
+ the items being iterated so they come back as dicts"""
21
+ names = [d[0] == 'id' and '_id' or d[0] for d in rows.cursor.description]
22
+ set_store = dict(__store=store) if store else {}
23
+ for rec in rows:
24
+ yield kind(
25
+ ((k, v) for k, v in zip(names, rec)),
26
+ **set_store
27
+ )
28
+
29
+
30
+ class Result(object):
31
+ """rows resulting from a method call"""
32
+ # pylint: disable=too-few-public-methods
33
+
34
+ def __init__(self, rows, kind, name=None):
35
+ self.rows = rows
36
+ self.kind = kind
37
+ self.name = name
38
+
39
+ def __iter__(self):
40
+ store = gitdata.table_of(self.kind, name=self.name) if self.name else None
41
+ return get_result_iterator(self.rows, self.kind, store)
42
+
43
+ # def __len__(self):
44
+ # return self.rows.cursor.rowcount
45
+
46
+ def __repr__(self):
47
+ return repr(list(self))
48
+
49
+ def __str__(self):
50
+ return str(RecordList(self))
51
+
52
+
53
+ class SqlExpression:
54
+
55
+ def __init__(self, *terms):
56
+ self.terms = terms
57
+
58
+ def __str__(self):
59
+ return ' '.join(str(t) for t in self.terms)
60
+
61
+ def __and__(self, value):
62
+ return SqlExpression(self, 'and', str(value))
63
+
64
+ def __or__(self, value):
65
+ return SqlExpression(self, 'or', str(value))
66
+
67
+
68
+ class SqlTerm:
69
+
70
+ def __init__(self, name):
71
+ self.name = name
72
+ print(name)
73
+
74
+ def __str__(self):
75
+ return '`{}`'.format(self.name)
76
+
77
+ def __eq__(self, value):
78
+ return SqlExpression(str(self), '=', repr(value))
79
+
80
+ def __ne__(self, value):
81
+ return SqlExpression(str(self), '!=', repr(value))
82
+
83
+ def __gt__(self, value):
84
+ return SqlExpression(str(self), '>', repr(value))
85
+
86
+ def __ge__(self, value):
87
+ return SqlExpression(str(self), '>=', repr(value))
88
+
89
+ def __lt__(self, value):
90
+ return SqlExpression(str(self), '<', repr(value))
91
+
92
+ def __le__(self, value):
93
+ return SqlExpression(str(self), '<=', repr(value))
94
+
95
+
96
+ # class SqlTable:
97
+
98
+ # def __init__(self, db, table_name, kind=Record):
99
+ # self.db = db
100
+ # self.table_name = table_name
101
+ # self.kind = kind
102
+
103
+ # def __iter__(self):
104
+ # for row in gitdata.table_of(self.kind, name=self.table_name, db=self.db):
105
+ # yield row
106
+
107
+ # def __call__(self, *args):
108
+ # columns = ', '.join(a for a in args if not isinstance(a, SqlExpression))
109
+ # clauses = ' and '.join(
110
+ # str(a) for a in args if isinstance(a, SqlExpression)
111
+ # )
112
+ # if clauses:
113
+ # clauses = ' where ' + clauses
114
+ # cmd = 'select %s from %s%s' % (columns, self.table_name, clauses)
115
+ # for row in Result(self.db(cmd), self.kind):
116
+ # yield row
117
+
118
+ # def __len__(self):
119
+ # cmd = f'select count(*) from {self.table_name}'
120
+ # return list(self.db(cmd))[0][0]
121
+
122
+ # def select(
123
+ # self, *args,
124
+ # where=None, group_by=None, order_by=None, limit=None
125
+ # ):
126
+ # columns = ', '.join(args)
127
+ # cmd = [
128
+ # 'select %s from %s' % (columns, self.table_name)
129
+ # ]
130
+ # if where:
131
+ # cmd.append('where ', where)
132
+ # if group_by:
133
+ # cmd.append('group by ', group_by)
134
+ # if order_by:
135
+ # cmd.append('order by ', order_by)
136
+ # if limit:
137
+ # cmd.append('limit ', limit)
138
+ # for row in Result(self.db(cmd), self.kind):
139
+ # yield row
140
+
141
+ # def __getattr__(self, name):
142
+ # return SqlTerm(name)
143
+
144
+ # def get(self, ids):
145
+ # cmd = f'select * from {self.table_name} where id in %s'
146
+ # as_list = True
147
+ # if isinstance(ids, (int, str)):
148
+ # ids = list(map(int, [ids]))
149
+ # as_list = False
150
+ # rows = Result(self.db(cmd, ids), self.kind)
151
+ # if as_list:
152
+ # return rows
153
+ # else:
154
+ # if rows:
155
+ # return list(rows)[0]
156
+
157
+ # def __getitem__(self, key):
158
+ # return self.get(key)
159
+
160
+ # def __str__(self):
161
+ # return str(gitdata.table_of(self.kind, name=self.table_name, db=self.db))
162
+
163
+
164
+ class SqlQuery:
165
+
166
+ def __init__(self, db, pathname):
167
+ self.db = db
168
+ self.pathname = pathname
169
+ self.kind = Record
170
+
171
+ def __call__(self, *args, **kwargs):
172
+ code = load(self.pathname)
173
+ return Result(self.db(code, *args or kwargs), self.kind)
174
+
175
+ def __iter__(self):
176
+ for row in self():
177
+ yield row
178
+
179
+ # def to(self, kind):
180
+ # for item in self:
181
+ # return kind(item)
182
+
183
+ # def execute(self, *args, **kwargs):
184
+ # code = load(self.pathname)
185
+ # cursor = self.db.cursor()
186
+ # result = cursor.execute(code, *args or kwargs)
187
+ # return result
188
+
189
+ # def __str__(self):
190
+ # return str(self())
191
+
192
+
193
+ class SQL:
194
+ """SQL Class
195
+
196
+ The SQL class provides a convenient way to work with SQL databases. It
197
+ provides access to database tables and can load and execute SQL queries
198
+ stored in .sql files by referring to the name of the file. Parameters
199
+ can be passed to SQL queries as required by treating the attribute as
200
+ a callable attribute.
201
+ """
202
+
203
+ def __init__(self, db, path='sql'):
204
+ self.db = db
205
+ if os.path.isfile(path):
206
+ dirname = os.path.dirname(path)
207
+ elif os.path.isdir(path):
208
+ dirname = path
209
+ else:
210
+ raise Exception(f'{path} is not a directory or filename')
211
+
212
+ if os.path.isdir(os.path.join(dirname, 'sql')):
213
+ path = os.path.join(dirname, 'sql')
214
+ else:
215
+ path = dirname
216
+
217
+ self.path = os.path.realpath(path)
218
+
219
+ def __getattr__(self, name):
220
+ pathname = os.path.join(self.path, name + '.sql')
221
+
222
+ if os.path.isfile(pathname):
223
+ return SqlQuery(self.db, pathname)
224
+
225
+ elif name in self.db.table_names:
226
+ return table_of(Record, name=name, db=self.db)
227
+
228
+
229
+ def get_sql(path=None, db=None):
230
+ """Return a SQL object"""
231
+
232
+ path = path or os.path.abspath((inspect.stack()[1])[1])
233
+
234
+ db = db or gitdata.database.connect()
235
+ return SQL(db, path)
@@ -842,3 +842,12 @@ class OrderedSet(collections.MutableSet):
842
842
  if isinstance(other, OrderedSet):
843
843
  return len(self) == len(other) and list(self) == list(other)
844
844
  return set(self) == set(other)
845
+
846
+
847
+ def load(pathname, encoding='utf-8'):
848
+ """Read a file and return the contents"""
849
+
850
+ logger = logging.getLogger(__name__)
851
+ logger.debug('load %r', pathname)
852
+ with open(pathname, encoding=encoding) as reader:
853
+ return reader.read()
@@ -1,12 +1,10 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gitdata-lib
3
- Version: 0.0.11
3
+ Version: 0.0.12
4
4
  Summary: Data extraction and analysis library
5
5
  Home-page: https://github.com/gitdata/gitdata-lib
6
6
  Author: DSI Labs
7
7
  Author-email: support@gitdata.com
8
- License: UNKNOWN
9
- Platform: UNKNOWN
10
8
  Classifier: Development Status :: 1 - Planning
11
9
  Classifier: Environment :: Console
12
10
  Classifier: Intended Audience :: Developers
@@ -22,11 +20,17 @@ Classifier: Programming Language :: Python :: 3 :: Only
22
20
  Classifier: Topic :: Database :: Front-Ends
23
21
  Description-Content-Type: text/markdown
24
22
  License-File: LICENSE
23
+ Requires-Dist: python-decouple>=3.4
24
+ Requires-Dist: Unipath>=1.1
25
+ Requires-Dist: PyMySQL==0.10.1
26
+ Requires-Dist: psycopg2-binary==2.9.1
27
+ Requires-Dist: python-dotenv
28
+ Requires-Dist: requests
29
+ Requires-Dist: docopt
30
+ Requires-Dist: faker
25
31
 
26
32
  # GitData Lib
27
33
  Data Wrangling for Everyone.
28
34
 
29
35
  GitData is a fast, scalable, distributed data exploration system
30
36
  with a rich set of commands that provide ways to gather, manage and query data in an unusually straightforward way.
31
-
32
-
@@ -14,14 +14,17 @@ gitdata/json.py
14
14
  gitdata/queues.py
15
15
  gitdata/repositories.py
16
16
  gitdata/solutions.py
17
+ gitdata/sql.py
17
18
  gitdata/utils.py
18
19
  gitdata/assets/README.md
19
20
  gitdata/cli/__init__.py
20
21
  gitdata/cli/gitdata_get.py
22
+ gitdata/cli/gitdata_scan.py
21
23
  gitdata/connectors/__init__.py
22
24
  gitdata/connectors/common.py
23
25
  gitdata/connectors/fake.py
24
26
  gitdata/connectors/gitlab.py
27
+ gitdata/connectors/http.py
25
28
  gitdata/connectors/local.py
26
29
  gitdata/database/__init__.py
27
30
  gitdata/database/common.py
@@ -1,3 +1,2 @@
1
1
  [console_scripts]
2
2
  gitdata = gitdata.cli:main
3
-
@@ -5,5 +5,4 @@ psycopg2-binary==2.9.1
5
5
  python-dotenv
6
6
  requests
7
7
  docopt
8
- webdavclient3>=3.14.5
9
8
  faker
@@ -5,5 +5,4 @@ psycopg2-binary==2.9.1
5
5
  python-dotenv
6
6
  requests
7
7
  docopt
8
- webdavclient3>=3.14.5
9
8
  faker
@@ -1 +0,0 @@
1
- __version__ = '0.0.11'
@@ -1,5 +0,0 @@
1
- """
2
- connectors
3
- """
4
-
5
- from .common import explore, connect, fetch
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes