rda-python-dsquasar 3.0.3__tar.gz → 3.0.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rda_python_dsquasar-3.0.3/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.6}/PKG-INFO +1 -1
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/pyproject.toml +1 -1
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/dsquasar.py +215 -38
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/dsquasar.usg +18 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6/src/rda_python_dsquasar.egg-info}/PKG-INFO +1 -1
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/LICENSE +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/MANIFEST.in +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/README.md +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/setup.cfg +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/__init__.py +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/ds_quasar.py +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/dstacc.py +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/taccrec.py +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/tacctar.py +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar.egg-info/SOURCES.txt +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar.egg-info/dependency_links.txt +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar.egg-info/entry_points.txt +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar.egg-info/requires.txt +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar.egg-info/top_level.txt +0 -0
- {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/test/test_dsquasar.py +0 -0
{rda_python_dsquasar-3.0.3/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.6}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rda_python_dsquasar
|
|
3
|
-
Version: 3.0.
|
|
3
|
+
Version: 3.0.6
|
|
4
4
|
Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
|
|
5
5
|
Author-email: Zaihua Ji <zji@ucar.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
|
|
@@ -84,10 +84,25 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
84
84
|
self.ONESIZE = 20*self.PGLOG['ONEGBS'] # 20GB, minimal file size to tar a single file
|
|
85
85
|
self.TFCOUNT = 100 # if file count is greater, use self.MINSIZE for tar file
|
|
86
86
|
self.SUBLMTS = 2000 # file count limit for a sub-group
|
|
87
|
-
|
|
88
|
-
|
|
87
|
+
# tarring (TARACT) forks one child per tar file, so its parallel unit is the tar
|
|
88
|
+
# file: use a small per-process limit and a high cap to spread the work widely.
|
|
89
|
+
self.MPTLIMIT = 200 # tar file count per batch process for tarring
|
|
90
|
+
self.MPTMAX = 12 # maximum number of batch processes for tarring
|
|
91
|
+
# uploading (BCKACT) forks one child per BCKSIZE transfer batch (~BCKSIZE/TARSIZE
|
|
92
|
+
# tar files each), so far fewer parallel units than tar files: use a large
|
|
93
|
+
# per-process limit and a low cap to avoid over-reserving cpus.
|
|
94
|
+
self.MPBLIMIT = 900 # tar file count per batch process for uploading
|
|
95
|
+
self.MPBMAX = 4 # maximum number of batch processes for uploading
|
|
89
96
|
self.MAXRUNTIME = 23*3600 # 23 hours; stop before the 24-hour PBS walltime
|
|
90
97
|
self.ONEHOUR = 3600 # seconds; time headroom needed to finish before walltime
|
|
98
|
+
# a repeat submit normally blocks as a duplicate while the batch job is running. if
|
|
99
|
+
# that job is running but barely progressing, an extra worker is submitted instead
|
|
100
|
+
# (see pick_worker_slot). progress is measured as the fraction of the recorded
|
|
101
|
+
# dscheck work done, which is action independent because dcount counts GDEX files
|
|
102
|
+
# for -A 3 but tar files for -A 2/-A 4.
|
|
103
|
+
self.MAXWORKERS = 2 # default maximum concurrent workers per command (-W)
|
|
104
|
+
self.WORKGRACE = 6*3600 # 6 hours before a running job is judged on progress
|
|
105
|
+
self.MINWDONE = 0.01 # done fraction after WORKGRACE that counts as progressing
|
|
91
106
|
self.PGBACK = {
|
|
92
107
|
'workdir' : "{}/{}/quasar_backup".format(self.PGLOG['GDEXWORK'], self.PGLOG['COMMONUSER']),
|
|
93
108
|
'mproc' : 1,
|
|
@@ -103,6 +118,8 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
103
118
|
'doemail' : 0,
|
|
104
119
|
'starttime' : 0, # wall-clock start of the run, for the PBS walltime guard
|
|
105
120
|
'tardone' : 0, # tar files dispatched so far, for the finish-rate estimate
|
|
121
|
+
'maxworkers' : self.MAXWORKERS, # -W, maximum concurrent workers per command
|
|
122
|
+
'worker' : 1, # -w, this run's worker slot; >1 for an added extra worker
|
|
106
123
|
'cmd' : None
|
|
107
124
|
}
|
|
108
125
|
self.dsids = []
|
|
@@ -119,7 +136,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
119
136
|
argv = sys.argv[1:]
|
|
120
137
|
for arg in argv:
|
|
121
138
|
if re.match(r'^-(h|-help)$', arg, re.I): self.show_usage('dsquasar')
|
|
122
|
-
ms = re.match(r'^-(a|b|c|d|e|E|l|m|n|t|u|A|B|D)$', arg)
|
|
139
|
+
ms = re.match(r'^-(a|b|c|d|e|E|l|m|n|t|u|w|A|B|D|W)$', arg)
|
|
123
140
|
if ms:
|
|
124
141
|
arg = ms.group(1)
|
|
125
142
|
if arg == 'b':
|
|
@@ -127,7 +144,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
127
144
|
elif arg in self.sopts:
|
|
128
145
|
self.sopts[arg] = 1
|
|
129
146
|
option = None
|
|
130
|
-
elif '
|
|
147
|
+
elif 'AcdlmtwW'.find(arg) > -1:
|
|
131
148
|
option = arg
|
|
132
149
|
if arg == 'd': self.bopts = []
|
|
133
150
|
elif 'BD'.find(arg) > -1:
|
|
@@ -153,11 +170,22 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
153
170
|
if not (arg == 'Y' or arg == 'N'): self.pglog(arg +": Lock Flag(-l) must be Y or N", self.LGWNEX)
|
|
154
171
|
self.PGBACK['dolock'] = 1 if arg == 'Y' else 0
|
|
155
172
|
option = None
|
|
173
|
+
elif option == 'W':
|
|
174
|
+
self.PGBACK['maxworkers'] = int(arg)
|
|
175
|
+
if self.PGBACK['maxworkers'] < 1: self.pglog(arg +": Maximum Workers(-W) must be > 0", self.LGWNEX)
|
|
176
|
+
option = None
|
|
177
|
+
elif option == 'w':
|
|
178
|
+
self.PGBACK['worker'] = int(arg)
|
|
179
|
+
if self.PGBACK['worker'] < 1: self.pglog(arg +": Worker Slot(-w) must be > 0", self.LGWNEX)
|
|
180
|
+
option = None
|
|
156
181
|
else:
|
|
157
182
|
self.add_to_dsids(arg)
|
|
158
183
|
else:
|
|
159
184
|
self.pglog(arg + ": Value without leading option", self.LGWNEX)
|
|
160
185
|
if not (self.sopts['a'] or self.dsids): self.show_usage('dsquasar')
|
|
186
|
+
# an even worker slot walks the datasets from the other end so that it and the odd
|
|
187
|
+
# slot before it meet in the middle instead of fighting over the same dataset locks
|
|
188
|
+
if self.reverse_dsids(): self.dsids.reverse()
|
|
161
189
|
self.PGBACK['cmd'] = "dsquasar {}".format(' '.join(argv))
|
|
162
190
|
|
|
163
191
|
# do quasar backup process
|
|
@@ -320,24 +348,35 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
320
348
|
self.backup_dataset_tarfiles(dsfiles, 'B')
|
|
321
349
|
self.backup_dataset_tarfiles(dsfiles, 'D')
|
|
322
350
|
|
|
323
|
-
# size up the number of batch processes
|
|
324
|
-
#
|
|
325
|
-
#
|
|
326
|
-
#
|
|
327
|
-
# (
|
|
328
|
-
#
|
|
329
|
-
#
|
|
351
|
+
# size up the number of batch processes to reserve, sizing tarring and uploading
|
|
352
|
+
# independently because their parallel work units differ. tarring's unit is the tar
|
|
353
|
+
# file (one child per tar): new infiles CINACT will create this run (estimated from
|
|
354
|
+
# the total bytes of GDEX files ready to back up, ~one tar per TARSIZE) plus existing
|
|
355
|
+
# status 'N' infile records to build (TARACT), scaled one process per MPTLIMIT tar
|
|
356
|
+
# files, capped at MPTMAX. uploading's unit is the BCKSIZE transfer batch (one child
|
|
357
|
+
# per ~BCKSIZE/TARSIZE tar files): existing status 'T' tar records to transfer
|
|
358
|
+
# (BCKACT), scaled one process per MPBLIMIT tar files, capped at MPBMAX. the phases
|
|
359
|
+
# run sequentially in one run and share a single reserved ncpus, so return the max of
|
|
360
|
+
# the two counts. create-infile only (-A 1) is not a batch action (see NBACTS) so it
|
|
361
|
+
# never reaches here and stays single process.
|
|
330
362
|
def batch_process_count(self, acts):
|
|
363
|
+
mproc = 1
|
|
331
364
|
tcnt = 0
|
|
332
365
|
if acts&self.CINACT:
|
|
333
366
|
sizes = [0]
|
|
334
367
|
if self.gather_dataset_files(None, False, sizes):
|
|
335
368
|
tcnt += max(1, (sizes[0] + self.TARSIZE - 1)//self.TARSIZE)
|
|
336
369
|
if acts&self.TARACT: tcnt += self.batch_tar_count('N')
|
|
337
|
-
if
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
370
|
+
if tcnt > self.MPTLIMIT:
|
|
371
|
+
tproc = (tcnt + self.MPTLIMIT - 1)//self.MPTLIMIT
|
|
372
|
+
if tproc > self.MPTMAX: tproc = self.MPTMAX
|
|
373
|
+
if tproc > mproc: mproc = tproc
|
|
374
|
+
if acts&self.BCKACT:
|
|
375
|
+
bcnt = self.batch_tar_count('T')
|
|
376
|
+
if bcnt > self.MPBLIMIT:
|
|
377
|
+
bproc = (bcnt + self.MPBLIMIT - 1)//self.MPBLIMIT
|
|
378
|
+
if bproc > self.MPBMAX: bproc = self.MPBMAX
|
|
379
|
+
if bproc > mproc: mproc = bproc
|
|
341
380
|
return mproc
|
|
342
381
|
|
|
343
382
|
# count the bfile tar records of a given status honoring backflag/dataset scope
|
|
@@ -351,29 +390,82 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
351
390
|
return tcnt
|
|
352
391
|
return self.pgget('bfile', '', bcnd, self.LGWNEX)
|
|
353
392
|
|
|
393
|
+
# look up the dscheck record registered for a worker's argv, using the same lookup key
|
|
394
|
+
# init_dscheck builds
|
|
395
|
+
def lookup_worker_dscheck(self, argv):
|
|
396
|
+
dargv = self.argv_to_string(argv, 0, "Process in Delayed Mode")
|
|
397
|
+
dargx = None
|
|
398
|
+
if len(dargv) > 100:
|
|
399
|
+
dargx = dargv[100:]
|
|
400
|
+
dargv = dargv[0:100]
|
|
401
|
+
return self.get_dscheck("dsquasar", dargv, self.PGBACK['workdir'], self.PGLOG['CURUID'], dargx, self.LOGWRN)
|
|
402
|
+
|
|
403
|
+
# a running batch job counts as stalled once it has run longer than WORKGRACE and has
|
|
404
|
+
# still completed no more than MINWDONE of its recorded work; anything further along
|
|
405
|
+
# than that is progressing and is left alone. the done fraction dcount/fcount is used
|
|
406
|
+
# instead of a raw count because dcount counts GDEX files for -A 3 but tar files for
|
|
407
|
+
# -A 2/-A 4. a job that is not locked, not started yet, or whose process is gone is not
|
|
408
|
+
# stalled: init_dscheck already restarts a dead one.
|
|
409
|
+
def worker_progress_stalled(self, pgrec):
|
|
410
|
+
if not (pgrec['pid'] and pgrec['stttime']): return False
|
|
411
|
+
if self.check_host_pid(pgrec['lockhost'], pgrec['pid']) <= 0: return False
|
|
412
|
+
if (tm() - pgrec['stttime']) < self.WORKGRACE: return False
|
|
413
|
+
done = pgrec['dcount']/pgrec['fcount'] if pgrec['fcount'] else 0
|
|
414
|
+
return done <= self.MINWDONE
|
|
415
|
+
|
|
416
|
+
# whether more than one worker may run the current action. two workers stay off each
|
|
417
|
+
# other's files only through dataset locking, so the action must lock what it works on:
|
|
418
|
+
# creating input files and building tar files do (backup_dataset_files and
|
|
419
|
+
# backup_dataset_infiles lock each dsid), but transferring (BCKACT) walks the status 'T'
|
|
420
|
+
# records by bid without locking, so two workers would transfer the same tar files twice.
|
|
421
|
+
def extra_workers_allowed(self):
|
|
422
|
+
return bool(self.PGBACK['dolock'] and not self.PGBACK['action']&self.BCKACT)
|
|
423
|
+
|
|
424
|
+
# pick the worker slot to submit into. slot 1 uses the invocation argv unchanged so a
|
|
425
|
+
# repeat submit still blocks as a duplicate. when a slot is held by a running job that
|
|
426
|
+
# is stalled, the next slot is tried instead, and appending -w N to the argv gives that
|
|
427
|
+
# extra worker its own dscheck record and PBS job. returns the slot and the dscheck
|
|
428
|
+
# record already registered for it, if any.
|
|
429
|
+
def pick_worker_slot(self):
|
|
430
|
+
if self.PGBACK['worker'] > 1: # an explicit -w N forces that slot
|
|
431
|
+
return (self.PGBACK['worker'], self.lookup_worker_dscheck(sys.argv[1:]))
|
|
432
|
+
pgrec = self.lookup_worker_dscheck(sys.argv[1:])
|
|
433
|
+
if not pgrec or not self.worker_progress_stalled(pgrec): return (1, pgrec)
|
|
434
|
+
# worker 1 is running but stalled: submit an extra worker into the first slot with no
|
|
435
|
+
# dscheck record yet. a slot that is already registered is left alone, so an extra
|
|
436
|
+
# worker still waiting to be started by the dscheck daemon is never taken over and
|
|
437
|
+
# run here instead, outside PBS.
|
|
438
|
+
maxw = self.PGBACK['maxworkers'] if self.extra_workers_allowed() else 1
|
|
439
|
+
for slot in range(2, maxw + 1):
|
|
440
|
+
if not self.lookup_worker_dscheck(sys.argv[1:] + ['-w', str(slot)]):
|
|
441
|
+
self.pglog("Worker 1 is running with {} of {} done: adding worker {}".format(
|
|
442
|
+
pgrec['dcount'], pgrec['fcount'], slot), self.LOGWRN)
|
|
443
|
+
return (slot, None)
|
|
444
|
+
return (1, pgrec) # no free slot; block on worker 1 as usual
|
|
445
|
+
|
|
354
446
|
# start none daemon and intialize dscheck counts
|
|
355
447
|
def start_dsquasar_none_daemon(self, act, fcnt = 0):
|
|
356
448
|
acts = self.PGBACK['action']
|
|
449
|
+
cact = ('A' if acts < 10 else '') + str(acts)
|
|
357
450
|
if self.bopts != None:
|
|
358
451
|
dsid = self.dsids[0] if len(self.dsids) == 1 else ''
|
|
359
452
|
if not re.match(r'^[a-z]\d{6}$', dsid): dsid = ''
|
|
360
|
-
cact = ('A' if acts < 10 else '') + str(acts)
|
|
361
453
|
# only the initial delayed submit (not yet under PBS) sets qoptions for the batch
|
|
362
454
|
# job the dscheck daemon will start; size up multi-processing from the detail file
|
|
363
|
-
# count (unless an explicit -m N was given) and reserve matching PBS cpus.
|
|
364
|
-
#
|
|
365
|
-
# blocks as a duplicate. a command-line or the PBS batch run does not set qoptions.
|
|
455
|
+
# count (unless an explicit -m N was given) and reserve matching PBS cpus. a
|
|
456
|
+
# command-line or the PBS batch run does not set qoptions.
|
|
366
457
|
if self.PGLOG['CURBID'] < 1 and (act == self.STTACT or acts&self.NBACTS != acts):
|
|
367
|
-
#
|
|
368
|
-
#
|
|
369
|
-
#
|
|
370
|
-
#
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
if
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
458
|
+
# pick the worker slot first; slot 1 keeps the invocation argv so a repeat
|
|
459
|
+
# submit still blocks as a duplicate, while a stalled run hands the next slot
|
|
460
|
+
# its own -w N argv. only size up multi-processing - which checks the backup
|
|
461
|
+
# files - when the chosen slot has no dscheck record on file yet, so a
|
|
462
|
+
# duplicate submit blocks without checking the backup files.
|
|
463
|
+
(slot, pgrec) = self.pick_worker_slot()
|
|
464
|
+
if slot > 1:
|
|
465
|
+
self.PGBACK['worker'] = slot
|
|
466
|
+
sys.argv.extend(['-w', str(slot)])
|
|
467
|
+
self.PGBACK['cmd'] += " -w {}".format(slot)
|
|
468
|
+
if not pgrec:
|
|
377
469
|
qoptions = '-l walltime=24:00:00'
|
|
378
470
|
if acts&self.NBACTS != acts:
|
|
379
471
|
if self.PGBACK['mproc'] < 2: self.PGBACK['mproc'] = self.batch_process_count(acts)
|
|
@@ -393,7 +485,11 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
393
485
|
if self.PGBACK['mproc'] > 1 and acts&self.NBACTS == acts: self.PGBACK['mproc'] = 1
|
|
394
486
|
elif self.PGBACK['mproc'] > 1 and acts&self.NBACTS == acts:
|
|
395
487
|
self.PGBACK['mproc'] = 1
|
|
396
|
-
|
|
488
|
+
# tag the daemon string with the action, and with the worker slot for an extra worker,
|
|
489
|
+
# so that concurrent workers can be told apart in the log
|
|
490
|
+
dact = cact
|
|
491
|
+
if self.PGBACK['worker'] > 1: dact += "-w{}".format(self.PGBACK['worker'])
|
|
492
|
+
self.start_none_daemon('dsquasar', dact, self.PGLOG['CURUID'], self.PGBACK['mproc'], 60, 1)
|
|
397
493
|
if self.PGLOG['DSCHECK']:
|
|
398
494
|
if act == self.STTACT:
|
|
399
495
|
fcnt = self.gather_dataset_bckfiles(None, False)
|
|
@@ -464,6 +560,10 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
464
560
|
if self.PGBACK['dolock'] and dsid not in qinfo['dslocks']:
|
|
465
561
|
if self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
|
|
466
562
|
qinfo['dslocks'].append(dsid)
|
|
563
|
+
if not self.regather_dataset_files(bfiles, dsid, backflag):
|
|
564
|
+
self.lock_dataset(dsid, 0, self.LGEREX)
|
|
565
|
+
qinfo['dslocks'].remove(dsid)
|
|
566
|
+
continue
|
|
467
567
|
fcnt = bfiles[dsid]['scount']
|
|
468
568
|
if fcnt > 0:
|
|
469
569
|
self.process_backup_files(qinfo, dsid, fcnt, bfiles[dsid]['srecs'], 'S')
|
|
@@ -504,6 +604,44 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
504
604
|
msg = "{}: {}({}) GDEX {}file{} of {} for next {}".format(amsg, fcnt, ssize, cmsg, s, dmsg, bmsg)
|
|
505
605
|
self.pglog(self.INDENT + msg, self.LOGACT)
|
|
506
606
|
|
|
607
|
+
# the file list was gathered before this dataset was locked, so it can be hours old and
|
|
608
|
+
# name files another worker has backed up since. re-gather it now that the dataset is
|
|
609
|
+
# locked and no other worker can add backup files for it, and replace the stale list.
|
|
610
|
+
# returns the file count left to back up for the given backup flag.
|
|
611
|
+
def regather_dataset_files(self, bfiles, dsid, backflag):
|
|
612
|
+
pgrec = self.pgget("dataset", "backflag", "dsid = '{}'".format(dsid), self.LGWNEX)
|
|
613
|
+
if not pgrec: return 0
|
|
614
|
+
dsfiles = {'B' : {}, 'D' : {}}
|
|
615
|
+
self.get_dataset_files(dsid, dsfiles, pgrec['backflag'])
|
|
616
|
+
if dsid not in dsfiles[backflag]:
|
|
617
|
+
self.pglog("{}: no {} file left to back up, skip".format(dsid, self.BACKMSG[backflag]), self.DTLACT)
|
|
618
|
+
return 0
|
|
619
|
+
bfiles[dsid] = dsfiles[backflag][dsid]
|
|
620
|
+
return bfiles[dsid]['scount'] + bfiles[dsid]['wcount']
|
|
621
|
+
|
|
622
|
+
# lock the other datasets whose files share one tar file, so that no other worker gathers
|
|
623
|
+
# or tars their files while this tar file is being claimed. datasets already locked by this
|
|
624
|
+
# run, including the primary one, are left alone. returns the datasets locked here, to be
|
|
625
|
+
# unlocked once the tar file is dispatched, or None if one of them is held by another
|
|
626
|
+
# worker, in which case nothing new is left locked and the tar file must wait for a later run
|
|
627
|
+
def lock_backup_datasets(self, qinfo, dsids):
|
|
628
|
+
dslocks = []
|
|
629
|
+
if not self.PGBACK['dolock']: return dslocks
|
|
630
|
+
for dsid in dsids:
|
|
631
|
+
if dsid in qinfo['dslocks']: continue
|
|
632
|
+
if self.lock_dataset(dsid, 1, self.LOGERR) < 1:
|
|
633
|
+
self.unlock_backup_datasets(qinfo, dslocks)
|
|
634
|
+
return None
|
|
635
|
+
qinfo['dslocks'].append(dsid)
|
|
636
|
+
dslocks.append(dsid)
|
|
637
|
+
return dslocks
|
|
638
|
+
|
|
639
|
+
# unlock the given datasets and drop them from the lock list tracked for cleanup on quit
|
|
640
|
+
def unlock_backup_datasets(self, qinfo, dslocks):
|
|
641
|
+
for dsid in dslocks:
|
|
642
|
+
self.lock_dataset(dsid, 0, self.LOGERR)
|
|
643
|
+
if dsid in qinfo['dslocks']: qinfo['dslocks'].remove(dsid)
|
|
644
|
+
|
|
507
645
|
# get the current bid for adding bfile
|
|
508
646
|
def current_bid(self):
|
|
509
647
|
pgrec = self.pgget("bfile", "max(bid) mid", '', self.LGEREX)
|
|
@@ -527,15 +665,32 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
527
665
|
s = 's' if bcnt > 1 else ''
|
|
528
666
|
self.pglog("{}: {} {} file{}...".format(amsg, bcnt, bmsg, s), self.WARNLG)
|
|
529
667
|
qinfo = {'backflag' : backflag, 'bid' : 0, 'dsids' : [], 'fcnt' : 0, 'size' : 0, 'abids' : [],
|
|
530
|
-
'infiles' : [], 'instr' : '', 'qdsids' : [], '
|
|
668
|
+
'infiles' : [], 'instr' : '', 'qdsids' : [], 'dslocks' : [], 'qfcnt' : 0,
|
|
669
|
+
'qsize' : 0, 'qcnt' : 0}
|
|
531
670
|
for dsid in bfiles:
|
|
532
|
-
if self.PGBACK['dolock']
|
|
671
|
+
if self.PGBACK['dolock']:
|
|
672
|
+
if self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
|
|
673
|
+
qinfo['dslocks'].append(dsid)
|
|
533
674
|
for bid in bfiles[dsid]:
|
|
675
|
+
# the status 'N' records were gathered before this dataset was locked, so they
|
|
676
|
+
# can be hours old; confirm each one is still untarred before spending a tar on
|
|
677
|
+
# it. holding the dataset lock makes this check race free against a second
|
|
678
|
+
# worker, which cannot hold the same dsid at the same time.
|
|
679
|
+
if not self.pgget('bfile', '', "bid = {} AND status = 'N'".format(bid), self.LGEREX):
|
|
680
|
+
self.pglog("{}-{}: no longer waiting to be tarred, skip".format(dsid, bid), self.DTLACT)
|
|
681
|
+
continue
|
|
534
682
|
qinfo['bid'] = bid
|
|
535
683
|
binfo = bfiles[dsid][bid]
|
|
536
684
|
for bkey in binfo: qinfo[bkey] = binfo[bkey]
|
|
685
|
+
# a tar file can hold files of multiple datasets; lock the other ones too before
|
|
686
|
+
# tarring, and release them again as soon as the tar file is dispatched
|
|
687
|
+
dslocks = self.lock_backup_datasets(qinfo, binfo['dsids'])
|
|
688
|
+
if dslocks is None:
|
|
689
|
+
self.pglog("{}-{}: another dataset of the tar file is locked, skip".format(dsid, bid), self.DTLACT)
|
|
690
|
+
continue
|
|
537
691
|
self.process_one_backup_file(qinfo, False)
|
|
538
|
-
|
|
692
|
+
self.unlock_backup_datasets(qinfo, dslocks)
|
|
693
|
+
if self.PGBACK['dolock']: self.unlock_backup_datasets(qinfo, [dsid])
|
|
539
694
|
if self.PGBACK['mproc'] > 1:
|
|
540
695
|
self.check_child(None, 0, self.LOGWRN, 1) # wait all child processes done
|
|
541
696
|
self.confirm_quasar_counts(qinfo, qinfo['abids'], "status = 'T'") # recount confirmed backups from RDADB
|
|
@@ -674,6 +829,18 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
674
829
|
for did in pgrecs['dsids'][i].split(','):
|
|
675
830
|
if did and did not in qinfo['qdsids']: qinfo['qdsids'].append(did)
|
|
676
831
|
|
|
832
|
+
# dsarch reports "<dsid>-<file>: <type> file backed up to /<dsid>/<bfile> by <date>" for
|
|
833
|
+
# each file to tar that is already backed up. our own record is the duplicate to drop only
|
|
834
|
+
# when those files point at some other bfile; a report naming our own bfile means this
|
|
835
|
+
# record is the one already backed up, so it must be kept.
|
|
836
|
+
def duplicate_backup_file(self, dsid, qfile):
|
|
837
|
+
ours = "/{}/{}".format(dsid, qfile)
|
|
838
|
+
isdup = False
|
|
839
|
+
for ms in re.finditer(r'file backed up to (\S+) by ', self.PGLOG['SYSERR']):
|
|
840
|
+
if ms.group(1) == ours: return False
|
|
841
|
+
isdup = True
|
|
842
|
+
return isdup
|
|
843
|
+
|
|
677
844
|
# backup one Quasar Backup or Backup&Drdata from one or multiple inputs and,
|
|
678
845
|
# reset the quasar backup dict
|
|
679
846
|
def process_one_backup_file(self, qinfo, addback, keepid = False):
|
|
@@ -713,7 +880,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
713
880
|
stat = self.pgsystem(cmd, self.ERRACT, 325) # 256 + 64 + 4 + 1
|
|
714
881
|
if stat:
|
|
715
882
|
for infile in qinfo['infiles']: self.delete_local_file(infile)
|
|
716
|
-
elif
|
|
883
|
+
elif self.duplicate_backup_file(dsid, qfile):
|
|
717
884
|
if self.pgdel('bfile', f"bid = {bid}"):
|
|
718
885
|
self.pglog(f"{dsid}-{qfile}: backup tarfile deleted for duplication", self.DTLACT)
|
|
719
886
|
for infile in qinfo['infiles']: self.delete_local_file(infile)
|
|
@@ -726,7 +893,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
726
893
|
stat = self.pgsystem(cmd, self.ERRACT, 325) # 256 + 64 + 4 + 1
|
|
727
894
|
if stat:
|
|
728
895
|
for infile in qinfo['infiles']: self.delete_local_file(infile)
|
|
729
|
-
elif
|
|
896
|
+
elif self.duplicate_backup_file(dsid, qfile):
|
|
730
897
|
if self.pgdel('bfile', f"bid = {bid}"):
|
|
731
898
|
self.pglog(f"{dsid}-{qfile}: backup tarfile deleted for duplication", self.DTLACT)
|
|
732
899
|
for infile in qinfo['infiles']: self.delete_local_file(infile)
|
|
@@ -1102,6 +1269,16 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
1102
1269
|
if self.pgget_wfile(dsid, 'wid', "bid " + bcnd, self.LGWNEX): fopt |= self.WOPT
|
|
1103
1270
|
return fopt
|
|
1104
1271
|
|
|
1272
|
+
# dataset ordering for gathering work; even worker slots start from the high end (d999999)
|
|
1273
|
+
# and odd ones from the low end, so that a pair of workers meets in the middle instead of
|
|
1274
|
+
# fighting over the same dataset locks
|
|
1275
|
+
def dsid_order(self):
|
|
1276
|
+
return " ORDER BY dsid DESC" if self.reverse_dsids() else " ORDER BY dsid"
|
|
1277
|
+
|
|
1278
|
+
# whether this worker slot walks the datasets from the high end
|
|
1279
|
+
def reverse_dsids(self):
|
|
1280
|
+
return self.PGBACK['worker']%2 == 0
|
|
1281
|
+
|
|
1105
1282
|
# gather all available dataset ids to backup data files
|
|
1106
1283
|
def gather_dataset_files(self, dsfiles, unlock = True, sizes = None):
|
|
1107
1284
|
fcnt = 0
|
|
@@ -1113,7 +1290,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
1113
1290
|
pgrec = self.pgget("dataset", "dsid, backflag, pid", "dsid = '{}'".format(dsid), self.LGWNEX)
|
|
1114
1291
|
if pgrec: pgrecs.append(pgrec)
|
|
1115
1292
|
else:
|
|
1116
|
-
mrecs = self.pgmget("dataset", "dsid, backflag, pid",
|
|
1293
|
+
mrecs = self.pgmget("dataset", "dsid, backflag, pid", self.dsid_order().strip(), self.LGWNEX)
|
|
1117
1294
|
dcnt = len(mrecs['dsid']) if mrecs else 0
|
|
1118
1295
|
pgrecs = [self.onerecord(mrecs, i) for i in range(dcnt)]
|
|
1119
1296
|
for pgrec in pgrecs:
|
|
@@ -1136,7 +1313,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
1136
1313
|
if self.dsids:
|
|
1137
1314
|
cnds = ["dsid = '{}' AND {}".format(dsid, bcnd) for dsid in self.dsids]
|
|
1138
1315
|
else:
|
|
1139
|
-
cnds = [bcnd +
|
|
1316
|
+
cnds = [bcnd + self.dsid_order()]
|
|
1140
1317
|
for dcnd in cnds:
|
|
1141
1318
|
pgrecs = self.pgmget("bfile", flds, dcnd, self.LGWNEX)
|
|
1142
1319
|
bcnt = len(pgrecs['bid']) if pgrecs else 0
|
|
@@ -1161,7 +1338,7 @@ class DsQuasar(PgCMD, PgSplit):
|
|
|
1161
1338
|
if self.dsids:
|
|
1162
1339
|
cnds = ["dsid = '{}' AND {}".format(dsid, bcnd) for dsid in self.dsids]
|
|
1163
1340
|
else:
|
|
1164
|
-
cnds = [bcnd +
|
|
1341
|
+
cnds = [bcnd + self.dsid_order()]
|
|
1165
1342
|
for dcnd in cnds:
|
|
1166
1343
|
pgrecs = self.pgmget("bfile", flds, dcnd, self.LGWNEX)
|
|
1167
1344
|
bcnt = len(pgrecs['bid']) if pgrecs else 0
|
{rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/dsquasar.usg
RENAMED
|
@@ -83,6 +83,24 @@
|
|
|
83
83
|
-n Gather and show the available file counts only; no backup actions are
|
|
84
84
|
performed.
|
|
85
85
|
|
|
86
|
+
-W MaxWorkers
|
|
87
|
+
Maximum number of concurrent batch workers for the same command.
|
|
88
|
+
Defaults to 2. A repeat delayed submit normally blocks while the
|
|
89
|
+
batch job is running; if that job has run for more than 6 hours and
|
|
90
|
+
has still completed no more than 1% of its recorded work, it is
|
|
91
|
+
treated as stalled and an extra worker is submitted instead, up to
|
|
92
|
+
this limit. Extra workers rely on dataset locking to stay off each
|
|
93
|
+
other's files, so this is pinned to 1 with -l N, and also for the
|
|
94
|
+
transfer action (-A 4, 6, 7), which claims tar files without locking.
|
|
95
|
+
|
|
96
|
+
-w WorkerSlot
|
|
97
|
+
Worker slot number for this run, added automatically to an extra
|
|
98
|
+
worker's command line so it gets its own dscheck record. Not
|
|
99
|
+
normally set by hand. Even slots walk the datasets in descending
|
|
100
|
+
dsid order (from d999999) and odd slots in ascending order, so that a
|
|
101
|
+
pair of workers meets in the middle instead of contending for the
|
|
102
|
+
same dataset locks.
|
|
103
|
+
|
|
86
104
|
-u Clean up (unlock) backup locks for the given dataset IDs. Dataset IDs
|
|
87
105
|
must be provided (not valid with -a).
|
|
88
106
|
|
{rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6/src/rda_python_dsquasar.egg-info}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rda_python_dsquasar
|
|
3
|
-
Version: 3.0.
|
|
3
|
+
Version: 3.0.6
|
|
4
4
|
Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
|
|
5
5
|
Author-email: Zaihua Ji <zji@ucar.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar/ds_quasar.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.6}/src/rda_python_dsquasar.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|