rda-python-dsquasar 3.0.3__tar.gz → 3.0.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (20) hide show
  1. {rda_python_dsquasar-3.0.3/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.7}/PKG-INFO +1 -1
  2. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/pyproject.toml +1 -1
  3. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/dsquasar.py +215 -38
  4. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/dsquasar.usg +26 -4
  5. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7/src/rda_python_dsquasar.egg-info}/PKG-INFO +1 -1
  6. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/LICENSE +0 -0
  7. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/MANIFEST.in +0 -0
  8. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/README.md +0 -0
  9. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/setup.cfg +0 -0
  10. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/__init__.py +0 -0
  11. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/ds_quasar.py +0 -0
  12. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/dstacc.py +0 -0
  13. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/taccrec.py +0 -0
  14. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar/tacctar.py +0 -0
  15. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar.egg-info/SOURCES.txt +0 -0
  16. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar.egg-info/dependency_links.txt +0 -0
  17. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar.egg-info/entry_points.txt +0 -0
  18. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar.egg-info/requires.txt +0 -0
  19. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/src/rda_python_dsquasar.egg-info/top_level.txt +0 -0
  20. {rda_python_dsquasar-3.0.3 → rda_python_dsquasar-3.0.7}/test/test_dsquasar.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.3
3
+ Version: 3.0.7
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "rda_python_dsquasar"
7
- version = "3.0.3"
7
+ version = "3.0.7"
8
8
  authors = [
9
9
  { name="Zaihua Ji", email="zji@ucar.edu" },
10
10
  ]
@@ -84,10 +84,25 @@ class DsQuasar(PgCMD, PgSplit):
84
84
  self.ONESIZE = 20*self.PGLOG['ONEGBS'] # 20GB, minimal file size to tar a single file
85
85
  self.TFCOUNT = 100 # if file count is greater, use self.MINSIZE for tar file
86
86
  self.SUBLMTS = 2000 # file count limit for a sub-group
87
- self.MPLIMIT = 500 # tar file count per batch process for multi-processing
88
- self.MPMAX = 8 # maximum number of batch processes for multi-processing
87
+ # tarring (TARACT) forks one child per tar file, so its parallel unit is the tar
88
+ # file: use a small per-process limit and a high cap to spread the work widely.
89
+ self.MPTLIMIT = 200 # tar file count per batch process for tarring
90
+ self.MPTMAX = 12 # maximum number of batch processes for tarring
91
+ # uploading (BCKACT) forks one child per BCKSIZE transfer batch (~BCKSIZE/TARSIZE
92
+ # tar files each), so far fewer parallel units than tar files: use a large
93
+ # per-process limit and a low cap to avoid over-reserving cpus.
94
+ self.MPBLIMIT = 900 # tar file count per batch process for uploading
95
+ self.MPBMAX = 4 # maximum number of batch processes for uploading
89
96
  self.MAXRUNTIME = 23*3600 # 23 hours; stop before the 24-hour PBS walltime
90
97
  self.ONEHOUR = 3600 # seconds; time headroom needed to finish before walltime
98
+ # a repeat submit normally blocks as a duplicate while the batch job is running. if
99
+ # that job is running but barely progressing, an extra worker is submitted instead
100
+ # (see pick_worker_slot). progress is measured as the fraction of the recorded
101
+ # dscheck work done, which is action independent because dcount counts GDEX files
102
+ # for -A 3 but tar files for -A 2/-A 4.
103
+ self.MAXWORKERS = 2 # default maximum concurrent workers per command (-W)
104
+ self.WORKGRACE = 6*3600 # 6 hours before a running job is judged on progress
105
+ self.MINWDONE = 0.01 # done fraction after WORKGRACE that counts as progressing
91
106
  self.PGBACK = {
92
107
  'workdir' : "{}/{}/quasar_backup".format(self.PGLOG['GDEXWORK'], self.PGLOG['COMMONUSER']),
93
108
  'mproc' : 1,
@@ -103,6 +118,8 @@ class DsQuasar(PgCMD, PgSplit):
103
118
  'doemail' : 0,
104
119
  'starttime' : 0, # wall-clock start of the run, for the PBS walltime guard
105
120
  'tardone' : 0, # tar files dispatched so far, for the finish-rate estimate
121
+ 'maxworkers' : self.MAXWORKERS, # -W, maximum concurrent workers per command
122
+ 'worker' : 1, # -w, this run's worker slot; >1 for an added extra worker
106
123
  'cmd' : None
107
124
  }
108
125
  self.dsids = []
@@ -119,7 +136,7 @@ class DsQuasar(PgCMD, PgSplit):
119
136
  argv = sys.argv[1:]
120
137
  for arg in argv:
121
138
  if re.match(r'^-(h|-help)$', arg, re.I): self.show_usage('dsquasar')
122
- ms = re.match(r'^-(a|b|c|d|e|E|l|m|n|t|u|A|B|D)$', arg)
139
+ ms = re.match(r'^-(a|b|c|d|e|E|l|m|n|t|u|w|A|B|D|W)$', arg)
123
140
  if ms:
124
141
  arg = ms.group(1)
125
142
  if arg == 'b':
@@ -127,7 +144,7 @@ class DsQuasar(PgCMD, PgSplit):
127
144
  elif arg in self.sopts:
128
145
  self.sopts[arg] = 1
129
146
  option = None
130
- elif 'Acdlmt'.find(arg) > -1:
147
+ elif 'AcdlmtwW'.find(arg) > -1:
131
148
  option = arg
132
149
  if arg == 'd': self.bopts = []
133
150
  elif 'BD'.find(arg) > -1:
@@ -153,11 +170,22 @@ class DsQuasar(PgCMD, PgSplit):
153
170
  if not (arg == 'Y' or arg == 'N'): self.pglog(arg +": Lock Flag(-l) must be Y or N", self.LGWNEX)
154
171
  self.PGBACK['dolock'] = 1 if arg == 'Y' else 0
155
172
  option = None
173
+ elif option == 'W':
174
+ self.PGBACK['maxworkers'] = int(arg)
175
+ if self.PGBACK['maxworkers'] < 1: self.pglog(arg +": Maximum Workers(-W) must be > 0", self.LGWNEX)
176
+ option = None
177
+ elif option == 'w':
178
+ self.PGBACK['worker'] = int(arg)
179
+ if self.PGBACK['worker'] < 1: self.pglog(arg +": Worker Slot(-w) must be > 0", self.LGWNEX)
180
+ option = None
156
181
  else:
157
182
  self.add_to_dsids(arg)
158
183
  else:
159
184
  self.pglog(arg + ": Value without leading option", self.LGWNEX)
160
185
  if not (self.sopts['a'] or self.dsids): self.show_usage('dsquasar')
186
+ # an even worker slot walks the datasets from the other end so that it and the odd
187
+ # slot before it meet in the middle instead of fighting over the same dataset locks
188
+ if self.reverse_dsids(): self.dsids.reverse()
161
189
  self.PGBACK['cmd'] = "dsquasar {}".format(' '.join(argv))
162
190
 
163
191
  # do quasar backup process
@@ -320,24 +348,35 @@ class DsQuasar(PgCMD, PgSplit):
320
348
  self.backup_dataset_tarfiles(dsfiles, 'B')
321
349
  self.backup_dataset_tarfiles(dsfiles, 'D')
322
350
 
323
- # size up the number of batch processes from the number of tar files that are the
324
- # unit of parallel work: new infiles CINACT will create this run (estimated from the
325
- # total bytes of GDEX files ready to back up, ~one tar per TARSIZE) plus existing
326
- # status 'N' infile records to build (TARACT) and status 'T' tar records to transfer
327
- # (BCKACT). scales one process per MPLIMIT tar files, capped at MPMAX. create-infile
328
- # only (-A 1) is not a batch action (see NBACTS) so it never reaches here and stays
329
- # single process.
351
+ # size up the number of batch processes to reserve, sizing tarring and uploading
352
+ # independently because their parallel work units differ. tarring's unit is the tar
353
+ # file (one child per tar): new infiles CINACT will create this run (estimated from
354
+ # the total bytes of GDEX files ready to back up, ~one tar per TARSIZE) plus existing
355
+ # status 'N' infile records to build (TARACT), scaled one process per MPTLIMIT tar
356
+ # files, capped at MPTMAX. uploading's unit is the BCKSIZE transfer batch (one child
357
+ # per ~BCKSIZE/TARSIZE tar files): existing status 'T' tar records to transfer
358
+ # (BCKACT), scaled one process per MPBLIMIT tar files, capped at MPBMAX. the phases
359
+ # run sequentially in one run and share a single reserved ncpus, so return the max of
360
+ # the two counts. create-infile only (-A 1) is not a batch action (see NBACTS) so it
361
+ # never reaches here and stays single process.
330
362
  def batch_process_count(self, acts):
363
+ mproc = 1
331
364
  tcnt = 0
332
365
  if acts&self.CINACT:
333
366
  sizes = [0]
334
367
  if self.gather_dataset_files(None, False, sizes):
335
368
  tcnt += max(1, (sizes[0] + self.TARSIZE - 1)//self.TARSIZE)
336
369
  if acts&self.TARACT: tcnt += self.batch_tar_count('N')
337
- if acts&self.BCKACT: tcnt += self.batch_tar_count('T')
338
- if tcnt <= self.MPLIMIT: return 1
339
- mproc = (tcnt + self.MPLIMIT - 1)//self.MPLIMIT
340
- if mproc > self.MPMAX: mproc = self.MPMAX
370
+ if tcnt > self.MPTLIMIT:
371
+ tproc = (tcnt + self.MPTLIMIT - 1)//self.MPTLIMIT
372
+ if tproc > self.MPTMAX: tproc = self.MPTMAX
373
+ if tproc > mproc: mproc = tproc
374
+ if acts&self.BCKACT:
375
+ bcnt = self.batch_tar_count('T')
376
+ if bcnt > self.MPBLIMIT:
377
+ bproc = (bcnt + self.MPBLIMIT - 1)//self.MPBLIMIT
378
+ if bproc > self.MPBMAX: bproc = self.MPBMAX
379
+ if bproc > mproc: mproc = bproc
341
380
  return mproc
342
381
 
343
382
  # count the bfile tar records of a given status honoring backflag/dataset scope
@@ -351,29 +390,82 @@ class DsQuasar(PgCMD, PgSplit):
351
390
  return tcnt
352
391
  return self.pgget('bfile', '', bcnd, self.LGWNEX)
353
392
 
393
+ # look up the dscheck record registered for a worker's argv, using the same lookup key
394
+ # init_dscheck builds
395
+ def lookup_worker_dscheck(self, argv):
396
+ dargv = self.argv_to_string(argv, 0, "Process in Delayed Mode")
397
+ dargx = None
398
+ if len(dargv) > 100:
399
+ dargx = dargv[100:]
400
+ dargv = dargv[0:100]
401
+ return self.get_dscheck("dsquasar", dargv, self.PGBACK['workdir'], self.PGLOG['CURUID'], dargx, self.LOGWRN)
402
+
403
+ # a running batch job counts as stalled once it has run longer than WORKGRACE and has
404
+ # still completed no more than MINWDONE of its recorded work; anything further along
405
+ # than that is progressing and is left alone. the done fraction dcount/fcount is used
406
+ # instead of a raw count because dcount counts GDEX files for -A 3 but tar files for
407
+ # -A 2/-A 4. a job that is not locked, not started yet, or whose process is gone is not
408
+ # stalled: init_dscheck already restarts a dead one.
409
+ def worker_progress_stalled(self, pgrec):
410
+ if not (pgrec['pid'] and pgrec['stttime']): return False
411
+ if self.check_host_pid(pgrec['lockhost'], pgrec['pid']) <= 0: return False
412
+ if (tm() - pgrec['stttime']) < self.WORKGRACE: return False
413
+ done = pgrec['dcount']/pgrec['fcount'] if pgrec['fcount'] else 0
414
+ return done <= self.MINWDONE
415
+
416
+ # whether more than one worker may run the current action. two workers stay off each
417
+ # other's files only through dataset locking, so the action must lock what it works on:
418
+ # creating input files and building tar files do (backup_dataset_files and
419
+ # backup_dataset_infiles lock each dsid), but transferring (BCKACT) walks the status 'T'
420
+ # records by bid without locking, so two workers would transfer the same tar files twice.
421
+ def extra_workers_allowed(self):
422
+ return bool(self.PGBACK['dolock'] and not self.PGBACK['action']&self.BCKACT)
423
+
424
+ # pick the worker slot to submit into. slot 1 uses the invocation argv unchanged so a
425
+ # repeat submit still blocks as a duplicate. when a slot is held by a running job that
426
+ # is stalled, the next slot is tried instead, and appending -w N to the argv gives that
427
+ # extra worker its own dscheck record and PBS job. returns the slot and the dscheck
428
+ # record already registered for it, if any.
429
+ def pick_worker_slot(self):
430
+ if self.PGBACK['worker'] > 1: # an explicit -w N forces that slot
431
+ return (self.PGBACK['worker'], self.lookup_worker_dscheck(sys.argv[1:]))
432
+ pgrec = self.lookup_worker_dscheck(sys.argv[1:])
433
+ if not pgrec or not self.worker_progress_stalled(pgrec): return (1, pgrec)
434
+ # worker 1 is running but stalled: submit an extra worker into the first slot with no
435
+ # dscheck record yet. a slot that is already registered is left alone, so an extra
436
+ # worker still waiting to be started by the dscheck daemon is never taken over and
437
+ # run here instead, outside PBS.
438
+ maxw = self.PGBACK['maxworkers'] if self.extra_workers_allowed() else 1
439
+ for slot in range(2, maxw + 1):
440
+ if not self.lookup_worker_dscheck(sys.argv[1:] + ['-w', str(slot)]):
441
+ self.pglog("Worker 1 is running with {} of {} done: adding worker {}".format(
442
+ pgrec['dcount'], pgrec['fcount'], slot), self.LOGWRN)
443
+ return (slot, None)
444
+ return (1, pgrec) # no free slot; block on worker 1 as usual
445
+
354
446
  # start none daemon and intialize dscheck counts
355
447
  def start_dsquasar_none_daemon(self, act, fcnt = 0):
356
448
  acts = self.PGBACK['action']
449
+ cact = ('A' if acts < 10 else '') + str(acts)
357
450
  if self.bopts != None:
358
451
  dsid = self.dsids[0] if len(self.dsids) == 1 else ''
359
452
  if not re.match(r'^[a-z]\d{6}$', dsid): dsid = ''
360
- cact = ('A' if acts < 10 else '') + str(acts)
361
453
  # only the initial delayed submit (not yet under PBS) sets qoptions for the batch
362
454
  # job the dscheck daemon will start; size up multi-processing from the detail file
363
- # count (unless an explicit -m N was given) and reserve matching PBS cpus. the
364
- # invocation argv is left unchanged so a repeat submit with the same argv still
365
- # blocks as a duplicate. a command-line or the PBS batch run does not set qoptions.
455
+ # count (unless an explicit -m N was given) and reserve matching PBS cpus. a
456
+ # command-line or the PBS batch run does not set qoptions.
366
457
  if self.PGLOG['CURBID'] < 1 and (act == self.STTACT or acts&self.NBACTS != acts):
367
- # check for a duplicate command first (same lookup key init_dscheck builds);
368
- # only size up multi-processing - which checks the backup files - when no
369
- # matching dscheck record is on file yet, so a duplicate submit blocks without
370
- # checking the backup files.
371
- dargv = self.argv_to_string(sys.argv[1:], 0, "Process in Delayed Mode")
372
- dargx = None
373
- if len(dargv) > 100:
374
- dargx = dargv[100:]
375
- dargv = dargv[0:100]
376
- if not self.get_dscheck("dsquasar", dargv, self.PGBACK['workdir'], self.PGLOG['CURUID'], dargx, self.LOGWRN):
458
+ # pick the worker slot first; slot 1 keeps the invocation argv so a repeat
459
+ # submit still blocks as a duplicate, while a stalled run hands the next slot
460
+ # its own -w N argv. only size up multi-processing - which checks the backup
461
+ # files - when the chosen slot has no dscheck record on file yet, so a
462
+ # duplicate submit blocks without checking the backup files.
463
+ (slot, pgrec) = self.pick_worker_slot()
464
+ if slot > 1:
465
+ self.PGBACK['worker'] = slot
466
+ sys.argv.extend(['-w', str(slot)])
467
+ self.PGBACK['cmd'] += " -w {}".format(slot)
468
+ if not pgrec:
377
469
  qoptions = '-l walltime=24:00:00'
378
470
  if acts&self.NBACTS != acts:
379
471
  if self.PGBACK['mproc'] < 2: self.PGBACK['mproc'] = self.batch_process_count(acts)
@@ -393,7 +485,11 @@ class DsQuasar(PgCMD, PgSplit):
393
485
  if self.PGBACK['mproc'] > 1 and acts&self.NBACTS == acts: self.PGBACK['mproc'] = 1
394
486
  elif self.PGBACK['mproc'] > 1 and acts&self.NBACTS == acts:
395
487
  self.PGBACK['mproc'] = 1
396
- self.start_none_daemon('dsquasar', '', self.PGLOG['CURUID'], self.PGBACK['mproc'], 60, 1)
488
+ # tag the daemon string with the action, and with the worker slot for an extra worker,
489
+ # so that concurrent workers can be told apart in the log
490
+ dact = cact
491
+ if self.PGBACK['worker'] > 1: dact += "-w{}".format(self.PGBACK['worker'])
492
+ self.start_none_daemon('dsquasar', dact, self.PGLOG['CURUID'], self.PGBACK['mproc'], 60, 1)
397
493
  if self.PGLOG['DSCHECK']:
398
494
  if act == self.STTACT:
399
495
  fcnt = self.gather_dataset_bckfiles(None, False)
@@ -464,6 +560,10 @@ class DsQuasar(PgCMD, PgSplit):
464
560
  if self.PGBACK['dolock'] and dsid not in qinfo['dslocks']:
465
561
  if self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
466
562
  qinfo['dslocks'].append(dsid)
563
+ if not self.regather_dataset_files(bfiles, dsid, backflag):
564
+ self.lock_dataset(dsid, 0, self.LGEREX)
565
+ qinfo['dslocks'].remove(dsid)
566
+ continue
467
567
  fcnt = bfiles[dsid]['scount']
468
568
  if fcnt > 0:
469
569
  self.process_backup_files(qinfo, dsid, fcnt, bfiles[dsid]['srecs'], 'S')
@@ -504,6 +604,44 @@ class DsQuasar(PgCMD, PgSplit):
504
604
  msg = "{}: {}({}) GDEX {}file{} of {} for next {}".format(amsg, fcnt, ssize, cmsg, s, dmsg, bmsg)
505
605
  self.pglog(self.INDENT + msg, self.LOGACT)
506
606
 
607
+ # the file list was gathered before this dataset was locked, so it can be hours old and
608
+ # name files another worker has backed up since. re-gather it now that the dataset is
609
+ # locked and no other worker can add backup files for it, and replace the stale list.
610
+ # returns the file count left to back up for the given backup flag.
611
+ def regather_dataset_files(self, bfiles, dsid, backflag):
612
+ pgrec = self.pgget("dataset", "backflag", "dsid = '{}'".format(dsid), self.LGWNEX)
613
+ if not pgrec: return 0
614
+ dsfiles = {'B' : {}, 'D' : {}}
615
+ self.get_dataset_files(dsid, dsfiles, pgrec['backflag'])
616
+ if dsid not in dsfiles[backflag]:
617
+ self.pglog("{}: no {} file left to back up, skip".format(dsid, self.BACKMSG[backflag]), self.DTLACT)
618
+ return 0
619
+ bfiles[dsid] = dsfiles[backflag][dsid]
620
+ return bfiles[dsid]['scount'] + bfiles[dsid]['wcount']
621
+
622
+ # lock the other datasets whose files share one tar file, so that no other worker gathers
623
+ # or tars their files while this tar file is being claimed. datasets already locked by this
624
+ # run, including the primary one, are left alone. returns the datasets locked here, to be
625
+ # unlocked once the tar file is dispatched, or None if one of them is held by another
626
+ # worker, in which case nothing new is left locked and the tar file must wait for a later run
627
+ def lock_backup_datasets(self, qinfo, dsids):
628
+ dslocks = []
629
+ if not self.PGBACK['dolock']: return dslocks
630
+ for dsid in dsids:
631
+ if dsid in qinfo['dslocks']: continue
632
+ if self.lock_dataset(dsid, 1, self.LOGERR) < 1:
633
+ self.unlock_backup_datasets(qinfo, dslocks)
634
+ return None
635
+ qinfo['dslocks'].append(dsid)
636
+ dslocks.append(dsid)
637
+ return dslocks
638
+
639
+ # unlock the given datasets and drop them from the lock list tracked for cleanup on quit
640
+ def unlock_backup_datasets(self, qinfo, dslocks):
641
+ for dsid in dslocks:
642
+ self.lock_dataset(dsid, 0, self.LOGERR)
643
+ if dsid in qinfo['dslocks']: qinfo['dslocks'].remove(dsid)
644
+
507
645
  # get the current bid for adding bfile
508
646
  def current_bid(self):
509
647
  pgrec = self.pgget("bfile", "max(bid) mid", '', self.LGEREX)
@@ -527,15 +665,32 @@ class DsQuasar(PgCMD, PgSplit):
527
665
  s = 's' if bcnt > 1 else ''
528
666
  self.pglog("{}: {} {} file{}...".format(amsg, bcnt, bmsg, s), self.WARNLG)
529
667
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsids' : [], 'fcnt' : 0, 'size' : 0, 'abids' : [],
530
- 'infiles' : [], 'instr' : '', 'qdsids' : [], 'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
668
+ 'infiles' : [], 'instr' : '', 'qdsids' : [], 'dslocks' : [], 'qfcnt' : 0,
669
+ 'qsize' : 0, 'qcnt' : 0}
531
670
  for dsid in bfiles:
532
- if self.PGBACK['dolock'] and self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
671
+ if self.PGBACK['dolock']:
672
+ if self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
673
+ qinfo['dslocks'].append(dsid)
533
674
  for bid in bfiles[dsid]:
675
+ # the status 'N' records were gathered before this dataset was locked, so they
676
+ # can be hours old; confirm each one is still untarred before spending a tar on
677
+ # it. holding the dataset lock makes this check race free against a second
678
+ # worker, which cannot hold the same dsid at the same time.
679
+ if not self.pgget('bfile', '', "bid = {} AND status = 'N'".format(bid), self.LGEREX):
680
+ self.pglog("{}-{}: no longer waiting to be tarred, skip".format(dsid, bid), self.DTLACT)
681
+ continue
534
682
  qinfo['bid'] = bid
535
683
  binfo = bfiles[dsid][bid]
536
684
  for bkey in binfo: qinfo[bkey] = binfo[bkey]
685
+ # a tar file can hold files of multiple datasets; lock the other ones too before
686
+ # tarring, and release them again as soon as the tar file is dispatched
687
+ dslocks = self.lock_backup_datasets(qinfo, binfo['dsids'])
688
+ if dslocks is None:
689
+ self.pglog("{}-{}: another dataset of the tar file is locked, skip".format(dsid, bid), self.DTLACT)
690
+ continue
537
691
  self.process_one_backup_file(qinfo, False)
538
- if self.PGBACK['dolock']: self.lock_dataset(dsid, 0, self.LOGERR)
692
+ self.unlock_backup_datasets(qinfo, dslocks)
693
+ if self.PGBACK['dolock']: self.unlock_backup_datasets(qinfo, [dsid])
539
694
  if self.PGBACK['mproc'] > 1:
540
695
  self.check_child(None, 0, self.LOGWRN, 1) # wait all child processes done
541
696
  self.confirm_quasar_counts(qinfo, qinfo['abids'], "status = 'T'") # recount confirmed backups from RDADB
@@ -674,6 +829,18 @@ class DsQuasar(PgCMD, PgSplit):
674
829
  for did in pgrecs['dsids'][i].split(','):
675
830
  if did and did not in qinfo['qdsids']: qinfo['qdsids'].append(did)
676
831
 
832
+ # dsarch reports "<dsid>-<file>: <type> file backed up to /<dsid>/<bfile> by <date>" for
833
+ # each file to tar that is already backed up. our own record is the duplicate to drop only
834
+ # when those files point at some other bfile; a report naming our own bfile means this
835
+ # record is the one already backed up, so it must be kept.
836
+ def duplicate_backup_file(self, dsid, qfile):
837
+ ours = "/{}/{}".format(dsid, qfile)
838
+ isdup = False
839
+ for ms in re.finditer(r'file backed up to (\S+) by ', self.PGLOG['SYSERR']):
840
+ if ms.group(1) == ours: return False
841
+ isdup = True
842
+ return isdup
843
+
677
844
  # backup one Quasar Backup or Backup&Drdata from one or multiple inputs and,
678
845
  # reset the quasar backup dict
679
846
  def process_one_backup_file(self, qinfo, addback, keepid = False):
@@ -713,7 +880,7 @@ class DsQuasar(PgCMD, PgSplit):
713
880
  stat = self.pgsystem(cmd, self.ERRACT, 325) # 256 + 64 + 4 + 1
714
881
  if stat:
715
882
  for infile in qinfo['infiles']: self.delete_local_file(infile)
716
- elif re.search(r'file backed up to', self.PGLOG['SYSERR']):
883
+ elif self.duplicate_backup_file(dsid, qfile):
717
884
  if self.pgdel('bfile', f"bid = {bid}"):
718
885
  self.pglog(f"{dsid}-{qfile}: backup tarfile deleted for duplication", self.DTLACT)
719
886
  for infile in qinfo['infiles']: self.delete_local_file(infile)
@@ -726,7 +893,7 @@ class DsQuasar(PgCMD, PgSplit):
726
893
  stat = self.pgsystem(cmd, self.ERRACT, 325) # 256 + 64 + 4 + 1
727
894
  if stat:
728
895
  for infile in qinfo['infiles']: self.delete_local_file(infile)
729
- elif re.search(r'file backed up to', self.PGLOG['SYSERR']):
896
+ elif self.duplicate_backup_file(dsid, qfile):
730
897
  if self.pgdel('bfile', f"bid = {bid}"):
731
898
  self.pglog(f"{dsid}-{qfile}: backup tarfile deleted for duplication", self.DTLACT)
732
899
  for infile in qinfo['infiles']: self.delete_local_file(infile)
@@ -1102,6 +1269,16 @@ class DsQuasar(PgCMD, PgSplit):
1102
1269
  if self.pgget_wfile(dsid, 'wid', "bid " + bcnd, self.LGWNEX): fopt |= self.WOPT
1103
1270
  return fopt
1104
1271
 
1272
+ # dataset ordering for gathering work; even worker slots start from the high end (d999999)
1273
+ # and odd ones from the low end, so that a pair of workers meets in the middle instead of
1274
+ # fighting over the same dataset locks
1275
+ def dsid_order(self):
1276
+ return " ORDER BY dsid DESC" if self.reverse_dsids() else " ORDER BY dsid"
1277
+
1278
+ # whether this worker slot walks the datasets from the high end
1279
+ def reverse_dsids(self):
1280
+ return self.PGBACK['worker']%2 == 0
1281
+
1105
1282
  # gather all available dataset ids to backup data files
1106
1283
  def gather_dataset_files(self, dsfiles, unlock = True, sizes = None):
1107
1284
  fcnt = 0
@@ -1113,7 +1290,7 @@ class DsQuasar(PgCMD, PgSplit):
1113
1290
  pgrec = self.pgget("dataset", "dsid, backflag, pid", "dsid = '{}'".format(dsid), self.LGWNEX)
1114
1291
  if pgrec: pgrecs.append(pgrec)
1115
1292
  else:
1116
- mrecs = self.pgmget("dataset", "dsid, backflag, pid", "ORDER BY dsid", self.LGWNEX)
1293
+ mrecs = self.pgmget("dataset", "dsid, backflag, pid", self.dsid_order().strip(), self.LGWNEX)
1117
1294
  dcnt = len(mrecs['dsid']) if mrecs else 0
1118
1295
  pgrecs = [self.onerecord(mrecs, i) for i in range(dcnt)]
1119
1296
  for pgrec in pgrecs:
@@ -1136,7 +1313,7 @@ class DsQuasar(PgCMD, PgSplit):
1136
1313
  if self.dsids:
1137
1314
  cnds = ["dsid = '{}' AND {}".format(dsid, bcnd) for dsid in self.dsids]
1138
1315
  else:
1139
- cnds = [bcnd + " ORDER BY dsid"]
1316
+ cnds = [bcnd + self.dsid_order()]
1140
1317
  for dcnd in cnds:
1141
1318
  pgrecs = self.pgmget("bfile", flds, dcnd, self.LGWNEX)
1142
1319
  bcnt = len(pgrecs['bid']) if pgrecs else 0
@@ -1161,7 +1338,7 @@ class DsQuasar(PgCMD, PgSplit):
1161
1338
  if self.dsids:
1162
1339
  cnds = ["dsid = '{}' AND {}".format(dsid, bcnd) for dsid in self.dsids]
1163
1340
  else:
1164
- cnds = [bcnd + " ORDER BY dsid"]
1341
+ cnds = [bcnd + self.dsid_order()]
1165
1342
  for dcnd in cnds:
1166
1343
  pgrecs = self.pgmget("bfile", flds, dcnd, self.LGWNEX)
1167
1344
  bcnt = len(pgrecs['bid']) if pgrecs else 0
@@ -27,12 +27,16 @@
27
27
 
28
28
  BACKUP SCOPE
29
29
 
30
- -B Quasar Backup only (endpoint gdex-quasar) for the given datasets.
30
+ -B Limit the run to what is flagged 'B', Quasar Backup only (endpoint
31
+ gdex-quasar).
31
32
 
32
- -D Quasar Backup & Drdata (gdex-quasar and gdex-quasar-drdata).
33
+ -D Limit the run to what is flagged 'D', Quasar Backup & Drdata
34
+ (gdex-quasar and gdex-quasar-drdata).
33
35
 
34
- -B and -D are mutually exclusive; the default (neither given) backs up
35
- to both the Backup and the Drdata endpoints.
36
+ -B and -D are mutually exclusive. They only narrow what a run
37
+ processes; neither one changes how many copies a file gets. The
38
+ default (neither given) backs up according to the dataset.backflag
39
+ and dsgroup.backflag values.
36
40
 
37
41
  -c ChangeDays
38
42
  Used with -A 1 to re-back up files that changed within the given
@@ -83,6 +87,24 @@
83
87
  -n Gather and show the available file counts only; no backup actions are
84
88
  performed.
85
89
 
90
+ -W MaxWorkers
91
+ Maximum number of concurrent batch workers for the same command.
92
+ Defaults to 2. A repeat delayed submit normally blocks while the
93
+ batch job is running; if that job has run for more than 6 hours and
94
+ has still completed no more than 1% of its recorded work, it is
95
+ treated as stalled and an extra worker is submitted instead, up to
96
+ this limit. Extra workers rely on dataset locking to stay off each
97
+ other's files, so this is pinned to 1 with -l N, and also for the
98
+ transfer action (-A 4, 6, 7), which claims tar files without locking.
99
+
100
+ -w WorkerSlot
101
+ Worker slot number for this run, added automatically to an extra
102
+ worker's command line so it gets its own dscheck record. Not
103
+ normally set by hand. Even slots walk the datasets in descending
104
+ dsid order (from d999999) and odd slots in ascending order, so that a
105
+ pair of workers meets in the middle instead of contending for the
106
+ same dataset locks.
107
+
86
108
  -u Clean up (unlock) backup locks for the given dataset IDs. Dataset IDs
87
109
  must be provided (not valid with -a).
88
110
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.3
3
+ Version: 3.0.7
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar