dataverse-utils 0.24.0__tar.gz → 0.25.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/PKG-INFO +1 -1
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/pyproject.toml +1 -1
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/__init__.py +2 -2
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/collections.py +152 -33
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/dataverse_utils.py +14 -6
- dataverse_utils-0.25.2/src/dataverse_utils/scripts/dv_collection_info.py +723 -0
- dataverse_utils-0.24.0/src/dataverse_utils/scripts/dv_collection_info.py +0 -303
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/LICENCE.md +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/README.md +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/archive.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/data/LDC_EULA_general.md +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/dvdata.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/ldc.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_bagit.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_bulk_release.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_del.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_ldc_uploader.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_list_files.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_manifest_gen.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_pg_facet_date.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_readme_creator.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_record_copy.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_release.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_replace_licence.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_study_migrator.py +0 -0
- {dataverse_utils-0.24.0 → dataverse_utils-0.25.2}/src/dataverse_utils/scripts/dv_upload_tsv.py +0 -0
|
@@ -7,7 +7,7 @@ import pathlib
|
|
|
7
7
|
import sys
|
|
8
8
|
from dataverse_utils.dataverse_utils import *
|
|
9
9
|
|
|
10
|
-
VERSION = (0,
|
|
10
|
+
VERSION = (0, 25, 2)
|
|
11
11
|
__version__ = '.'.join([str(x) for x in VERSION])
|
|
12
12
|
|
|
13
13
|
USERAGENT = (f'dataverse_utils/v{__version__} ({sys.platform.capitalize()}); '
|
|
@@ -17,7 +17,7 @@ UAHEADER = {'User-agent' : USERAGENT}
|
|
|
17
17
|
SCRIPT_VERSIONS={
|
|
18
18
|
'dv_bagit' : (0, 1, 0),
|
|
19
19
|
'dv_bulk_release' : (0, 1, 0),
|
|
20
|
-
'dv_collection_info' : (0,
|
|
20
|
+
'dv_collection_info' : (0, 6, 0),
|
|
21
21
|
'dv_del' : (0, 2, 4),
|
|
22
22
|
'dv_ldc_uploader' : (0, 4, 1),
|
|
23
23
|
'dv_list_files' : (0, 1, 1),
|
|
@@ -7,6 +7,7 @@ import copy
|
|
|
7
7
|
import datetime
|
|
8
8
|
import io
|
|
9
9
|
import logging
|
|
10
|
+
import os
|
|
10
11
|
import pathlib
|
|
11
12
|
import random
|
|
12
13
|
import string
|
|
@@ -15,12 +16,12 @@ import tempfile
|
|
|
15
16
|
import time
|
|
16
17
|
import textwrap
|
|
17
18
|
import typing
|
|
18
|
-
import traceback
|
|
19
|
+
#import traceback
|
|
19
20
|
import warnings
|
|
20
21
|
|
|
21
22
|
import bs4
|
|
22
23
|
import charset_normalizer as cn
|
|
23
|
-
import markdown_pdf
|
|
24
|
+
#import markdown_pdf
|
|
24
25
|
import markdownify
|
|
25
26
|
import pyreadstat
|
|
26
27
|
import pandas as pd
|
|
@@ -30,6 +31,16 @@ import tqdm #Progress meter
|
|
|
30
31
|
from urllib3.util import Retry
|
|
31
32
|
from dataverse_utils import UAHEADER
|
|
32
33
|
|
|
34
|
+
#Who forces deprecation messages to the terminal insteaf of using `warnings`?
|
|
35
|
+
#PyMuPDF, that's who
|
|
36
|
+
#This stupidity can be removed once markdown_pdf updates, presumably past 1.13.2.
|
|
37
|
+
#pylint: disable=wrong-import-position, wrong-import-order, consider-using-with, unspecified-encoding
|
|
38
|
+
sys.stdout = open(os.devnull, 'w')
|
|
39
|
+
import markdown_pdf
|
|
40
|
+
sys.stdout = sys.__stdout__
|
|
41
|
+
#pylint: enable=wrong-import-position, wrong-import-order, consider-using-with, unspecified-encoding
|
|
42
|
+
|
|
43
|
+
|
|
33
44
|
LOGGER = logging.getLogger(__name__)
|
|
34
45
|
RETRY = Retry(total=10,
|
|
35
46
|
status_forcelist=[429, 500, 502, 503, 504],
|
|
@@ -38,6 +49,9 @@ RETRY = Retry(total=10,
|
|
|
38
49
|
backoff_factor=1)
|
|
39
50
|
BAR_FORMAT='{l_bar}{bar}{n_fmt}/{total_fmt} : time remaining - {remaining}'
|
|
40
51
|
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
|
|
41
55
|
class MetadataError(Exception):
|
|
42
56
|
'''
|
|
43
57
|
MetadataError
|
|
@@ -189,16 +203,36 @@ class DvCollection:
|
|
|
189
203
|
clean = f'https://{clean}'
|
|
190
204
|
return clean
|
|
191
205
|
|
|
192
|
-
def
|
|
206
|
+
def get_shortname(self, dvid):
|
|
193
207
|
'''
|
|
194
208
|
Get collection short name.
|
|
195
209
|
'''
|
|
210
|
+
#pylint: disable=broad-exception-raised, broad-exception-caught
|
|
196
211
|
self.limit.rate_limit()
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
212
|
+
obscure_error = f'''
|
|
213
|
+
An error has occured where a collection can be
|
|
214
|
+
identified by ID but its name cannot be determined.
|
|
215
|
+
This is (normally) caused by a configuration error where
|
|
216
|
+
administrator permissions are not correctly inherited by
|
|
217
|
+
the child collection.
|
|
218
|
+
|
|
219
|
+
Please check with the system administrator to determine
|
|
220
|
+
any exact issues.
|
|
221
|
+
|
|
222
|
+
Problematic collection id number: {dvid}
|
|
223
|
+
'''
|
|
224
|
+
|
|
225
|
+
sn = self.session.get(f'{self.url}/api/dataverses/{dvid}',
|
|
226
|
+
headers=self.headers,
|
|
227
|
+
timeout=self.kwargs.get('timeout', 15))
|
|
228
|
+
sn.raise_for_status()
|
|
229
|
+
try:
|
|
230
|
+
return sn.json()['data']['alias']
|
|
231
|
+
except Exception as exc:
|
|
232
|
+
LOGGER.critical(textwrap.dedent(obscure_error).strip().replace('\n',' '))
|
|
233
|
+
print(textwrap.dedent(obscure_error).strip(), file=sys.stderr)
|
|
234
|
+
print(exc, file=sys.stderr)
|
|
235
|
+
raise Exception from exc
|
|
202
236
|
|
|
203
237
|
def get_collections(self, coll:str=None, output=None)->list:#pylint: disable=unused-argument
|
|
204
238
|
'''
|
|
@@ -228,31 +262,8 @@ class DvCollection:
|
|
|
228
262
|
dvs =[]
|
|
229
263
|
for _ in data:
|
|
230
264
|
if _['type'] == 'dataverse':
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
dvs.append((_['title'], out))
|
|
234
|
-
except Exception as e:
|
|
235
|
-
obscure_error = f'''
|
|
236
|
-
An error has occured where a collection can be
|
|
237
|
-
identified by ID but its name cannot be determined.
|
|
238
|
-
This is (normally) caused by a configuration error where
|
|
239
|
-
administrator permissions are not correctly inherited by
|
|
240
|
-
the child collection.
|
|
241
|
-
|
|
242
|
-
Please check with the system administrator to determine
|
|
243
|
-
any exact issues.
|
|
244
|
-
|
|
245
|
-
Problematic collection id number: {_.get("id",
|
|
246
|
-
"not available")}'''
|
|
247
|
-
#to sys.stdout?
|
|
248
|
-
print(50*'-', file=sys.stderr)
|
|
249
|
-
print(textwrap.dedent(obscure_error), file=sys.stderr)
|
|
250
|
-
print(e)
|
|
251
|
-
LOGGER.error(textwrap.fill(textwrap.dedent(obscure_error).strip()))
|
|
252
|
-
traceback.print_exc()
|
|
253
|
-
print(50*'-', file=sys.stderr)
|
|
254
|
-
raise e
|
|
255
|
-
#---
|
|
265
|
+
out=self.get_shortname(_['id'])
|
|
266
|
+
dvs.append((_['title'], out))
|
|
256
267
|
if not dvs:
|
|
257
268
|
dvs = []
|
|
258
269
|
output.extend(dvs)
|
|
@@ -265,6 +276,114 @@ class DvCollection:
|
|
|
265
276
|
self.collections.insert(0, self.root)
|
|
266
277
|
return output
|
|
267
278
|
|
|
279
|
+
def walk(self, coll:str=None, path:str=None,
|
|
280
|
+
output:list=None)->list:
|
|
281
|
+
'''
|
|
282
|
+
Equivalent of os.walk() but for a collection
|
|
283
|
+
|
|
284
|
+
Parameters
|
|
285
|
+
----------
|
|
286
|
+
coll : str
|
|
287
|
+
Short name of collection to walk. Default self.coll
|
|
288
|
+
path : str
|
|
289
|
+
Concatenated path of top level (eg, 'root/sub/sub2')
|
|
290
|
+
output : list
|
|
291
|
+
List of tuples from output for recursive addition
|
|
292
|
+
(path, [subpath1,...,subpathn], [pid1,...,pidn])
|
|
293
|
+
'''
|
|
294
|
+
#Why traverse repeatedly?
|
|
295
|
+
output = output if output else []
|
|
296
|
+
coll = coll if coll else self.coll
|
|
297
|
+
LOGGER.info('Walking tree: %s', coll)
|
|
298
|
+
x=self.session.get(f'{self.url}/api/dataverses/{coll}/contents',
|
|
299
|
+
headers=self.headers,
|
|
300
|
+
timeout=self.kwargs.get('timeout', 15))
|
|
301
|
+
y=x.json()
|
|
302
|
+
#dirpath, dirname, filename
|
|
303
|
+
pids = [f"{_['protocol']}:{_['authority']}/{_['identifier']}"
|
|
304
|
+
for _ in y['data'] if _['type']=='dataset']
|
|
305
|
+
coll_ids = [_['id'] for _ in y['data'] if _['type']=='dataverse']
|
|
306
|
+
subpaths = [self.get_shortname(_) for _ in coll_ids]
|
|
307
|
+
|
|
308
|
+
dvs = zip(coll_ids, subpaths)
|
|
309
|
+
path = [coll] if not path else path+[coll] if coll not in path else path
|
|
310
|
+
output.append(('/'.join(path), subpaths, pids))
|
|
311
|
+
for subp in dvs:
|
|
312
|
+
self.walk(subp[1], path, output)
|
|
313
|
+
return output
|
|
314
|
+
|
|
315
|
+
def tree(self, dvtree:list=None)->io.StringIO: #pylint:disable=too-many-locals
|
|
316
|
+
'''
|
|
317
|
+
Outputs the collection tree as StringIO object.
|
|
318
|
+
Perfect for your printing needs.
|
|
319
|
+
|
|
320
|
+
Parameters
|
|
321
|
+
----------
|
|
322
|
+
dvtree : list
|
|
323
|
+
List of tuples from self.walk. Default of None results
|
|
324
|
+
in self.walk being called
|
|
325
|
+
'''
|
|
326
|
+
dvtree = dvtree if dvtree else self.walk()
|
|
327
|
+
outtree = io.StringIO()
|
|
328
|
+
t4 = ' ' * 4 #tab 4
|
|
329
|
+
b = chr(9474) #│ bar
|
|
330
|
+
e = chr(9492) # └ end
|
|
331
|
+
h = chr(9472) # ─ horizontal
|
|
332
|
+
t = chr(9500) # ├ tee
|
|
333
|
+
tree_info = [_[0].split('/') for _ in dvtree]
|
|
334
|
+
for num, _ in enumerate(dvtree):
|
|
335
|
+
tabs = _[0].count('/')
|
|
336
|
+
#start
|
|
337
|
+
start = ''
|
|
338
|
+
if num == 0:
|
|
339
|
+
start = ''
|
|
340
|
+
elif tabs > 1:
|
|
341
|
+
start=b
|
|
342
|
+
|
|
343
|
+
oneup = _[0].split('/')[:-1]
|
|
344
|
+
subtree = [_ for _ in tree_info if '/'.join(oneup) in '/'.join(_)]
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
middle = b.join([t4]*(tabs -1))
|
|
348
|
+
|
|
349
|
+
#end
|
|
350
|
+
if _[0].split('/') == subtree[-1]:
|
|
351
|
+
end = e+2*h
|
|
352
|
+
else:
|
|
353
|
+
end = t+2*h
|
|
354
|
+
if num == 0:
|
|
355
|
+
end = ''
|
|
356
|
+
line = start + middle + end + _[0].split('/')[-1] + '\n'
|
|
357
|
+
outtree.write(line)
|
|
358
|
+
#breakpoint()
|
|
359
|
+
for n, dd in enumerate(_[2]):
|
|
360
|
+
# For some reason putting this in a comprehension doesn't work
|
|
361
|
+
val = _[0].split('/')[-1]
|
|
362
|
+
tmp = [_ for _ in tree_info if val in _]
|
|
363
|
+
lastcount = 0
|
|
364
|
+
if max(len(x) for x in tmp) >1:
|
|
365
|
+
tmp2 = '/'.join(tmp[0][:-1])
|
|
366
|
+
lastcount = max(n for n,_ in enumerate(tree_info) if tmp2 in '/'.join(_))
|
|
367
|
+
prefix_length = len(_[0].split('/'))-1
|
|
368
|
+
mid2 = (b + t4)* prefix_length
|
|
369
|
+
if num == lastcount and lastcount:
|
|
370
|
+
where = mid2.rfind(b)
|
|
371
|
+
mid2 = list(mid2)
|
|
372
|
+
mid2[where] = ' '
|
|
373
|
+
mid2 = ''.join(mid2)
|
|
374
|
+
if n+1 != len(_[2]):
|
|
375
|
+
indicator = t
|
|
376
|
+
elif _[1]:
|
|
377
|
+
indicator = t
|
|
378
|
+
else: indicator = e
|
|
379
|
+
|
|
380
|
+
#if 'UBC_stem_jobs' in _[0]:
|
|
381
|
+
# breakpoint()
|
|
382
|
+
fileline = mid2 + indicator + dd + '\n'
|
|
383
|
+
outtree.write( fileline)
|
|
384
|
+
outtree.seek(0)
|
|
385
|
+
return outtree
|
|
386
|
+
|
|
268
387
|
def get_studies(self, root:str=None):
|
|
269
388
|
'''
|
|
270
389
|
return [(pid, title)..(pid_n, title_n)] of a collection.
|
|
@@ -11,7 +11,7 @@ import json
|
|
|
11
11
|
import logging
|
|
12
12
|
import mimetypes
|
|
13
13
|
import os
|
|
14
|
-
|
|
14
|
+
import sys
|
|
15
15
|
import time
|
|
16
16
|
|
|
17
17
|
import requests
|
|
@@ -328,7 +328,7 @@ def uningest_file(dv_url, fid, apikey, study='n/a'):
|
|
|
328
328
|
LOGGER.error('Uningestion error: %s', uningest.reason)
|
|
329
329
|
print(uningest.reason)
|
|
330
330
|
|
|
331
|
-
def upload_file(fpath, hdl, **kwargs):
|
|
331
|
+
def upload_file(fpath, hdl, **kwargs): #pylint:disable=too-many-locals, too-many-statements, too-many-branches
|
|
332
332
|
'''
|
|
333
333
|
Uploads file to Dataverse study and sets file metadata and tags.
|
|
334
334
|
|
|
@@ -388,6 +388,9 @@ def upload_file(fpath, hdl, **kwargs):
|
|
|
388
388
|
|
|
389
389
|
override : bool, optional
|
|
390
390
|
Ignore NOTAB (ie, NOTAB = [])
|
|
391
|
+
|
|
392
|
+
return_json: bool
|
|
393
|
+
Returns the JSON if you need it
|
|
391
394
|
'''
|
|
392
395
|
#Why are SPSS files getting processed anyway?
|
|
393
396
|
#Does SPSS detection happen *after* upload
|
|
@@ -432,15 +435,17 @@ def upload_file(fpath, hdl, **kwargs):
|
|
|
432
435
|
timeout=kwargs.get('timeout',1000))
|
|
433
436
|
try:
|
|
434
437
|
print(upload.json())
|
|
435
|
-
except json.decoder.JSONDecodeError:
|
|
438
|
+
except json.decoder.JSONDecodeError as exc:
|
|
436
439
|
#This can happend when Glassfish crashes
|
|
437
440
|
LOGGER.critical(upload.text)
|
|
438
|
-
|
|
441
|
+
LOGGER.critical('URL: %s', f"{dvurl}/api/datasets/:persistentId/add")
|
|
442
|
+
print(upload.text, file=sys.stderr)
|
|
439
443
|
err = ('It\'s possible Glassfish may have crashed. '
|
|
440
444
|
'Check server logs for anomalies')
|
|
441
445
|
LOGGER.exception(err)
|
|
442
|
-
print(err)
|
|
443
|
-
|
|
446
|
+
print(err, file=sys.stderr)
|
|
447
|
+
print(f'URL {dvurl}/api/datasets/:persistentId/add', file=sys.stderr)
|
|
448
|
+
raise json.decoder.JSONDecodeError from exc
|
|
444
449
|
#SPSS files still process despite spoof, so there's
|
|
445
450
|
#a forcible unlock check
|
|
446
451
|
fid = upload.json()['data']['files'][0]['dataFile']['id']
|
|
@@ -463,6 +468,9 @@ def upload_file(fpath, hdl, **kwargs):
|
|
|
463
468
|
|
|
464
469
|
restrict_file(fid=fid, dv=dvurl, apikey=kwargs.get('apikey'),
|
|
465
470
|
rest=kwargs.get('rest', False))
|
|
471
|
+
if kwargs.get('return_json'):
|
|
472
|
+
return upload.json()
|
|
473
|
+
return None
|
|
466
474
|
|
|
467
475
|
def restrict_file(**kwargs):
|
|
468
476
|
'''
|