rda-python-dsquasar 3.0.12__tar.gz → 3.0.14__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (20) hide show
  1. {rda_python_dsquasar-3.0.12/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.14}/PKG-INFO +1 -1
  2. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/pyproject.toml +1 -1
  3. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/dsquasar.py +90 -33
  4. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14/src/rda_python_dsquasar.egg-info}/PKG-INFO +1 -1
  5. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/LICENSE +0 -0
  6. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/MANIFEST.in +0 -0
  7. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/README.md +0 -0
  8. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/setup.cfg +0 -0
  9. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/__init__.py +0 -0
  10. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/ds_quasar.py +0 -0
  11. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/dsquasar.usg +0 -0
  12. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/dstacc.py +0 -0
  13. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/taccrec.py +0 -0
  14. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar/tacctar.py +0 -0
  15. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar.egg-info/SOURCES.txt +0 -0
  16. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar.egg-info/dependency_links.txt +0 -0
  17. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar.egg-info/entry_points.txt +0 -0
  18. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar.egg-info/requires.txt +0 -0
  19. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/src/rda_python_dsquasar.egg-info/top_level.txt +0 -0
  20. {rda_python_dsquasar-3.0.12 → rda_python_dsquasar-3.0.14}/test/test_dsquasar.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.12
3
+ Version: 3.0.14
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "rda_python_dsquasar"
7
- version = "3.0.12"
7
+ version = "3.0.14"
8
8
  authors = [
9
9
  { name="Zaihua Ji", email="zji@ucar.edu" },
10
10
  ]
@@ -104,7 +104,7 @@ class DsQuasar(PgCMD, PgSplit):
104
104
  # for -A 3 but tar files for -A 2/-A 4.
105
105
  self.MAXWORKERS = 2 # default maximum concurrent workers per command (-W)
106
106
  self.WORKGRACE = 6*3600 # 6 hours before a running job is judged on progress
107
- self.MINWDONE = 0.01 # done fraction after WORKGRACE that counts as progressing
107
+ self.MINWPROJ = 0.9 # projected done fraction at the walltime that counts as finishing
108
108
  self.PGBACK = {
109
109
  'workdir' : "{}/{}/quasar_backup".format(self.PGLOG['GDEXWORK'], self.PGLOG['COMMONUSER']),
110
110
  'mproc' : 1,
@@ -114,6 +114,8 @@ class DsQuasar(PgCMD, PgSplit):
114
114
  'backflag' : None,
115
115
  'actmsg' : None,
116
116
  'pstep' : 0, # record progress step for under dscheck control
117
+ 'dcnt' : 0, # files done so far, recorded as dscheck.dcount every pstep
118
+ 'dsize' : 0,
117
119
  'errcnt' : 0,
118
120
  'bckcnt' : 0,
119
121
  'maxcnt' : 10,
@@ -228,6 +230,12 @@ class DsQuasar(PgCMD, PgSplit):
228
230
  if acts&self.TARACT: self.build_tarfile_action()
229
231
  if acts&self.CINACT: self.create_infile_action()
230
232
  if self.dstart and acts&self.TBACTS == self.TBACTS: self.globus_transfer_action()
233
+ # the queues were empty, so nothing was processed and nothing failed. no dscheck
234
+ # record is registered for such a run either, so drop the report rather than mail
235
+ # one about 0 files. a parked progress report still forces one, to clean it up
236
+ if not (self.PGBACK['bckcnt'] or self.PGBACK['errcnt'] or
237
+ self.PGLOG['ERRCNT'] or self.PGBACK['einfo']):
238
+ self.PGBACK['doemail'] = 0
231
239
  if self.PGBACK['doemail']:
232
240
  amsg = self.ACTMSG[self.PGBACK['action']]
233
241
  bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
@@ -343,7 +351,7 @@ class DsQuasar(PgCMD, PgSplit):
343
351
  self.backup_dataset_infiles(dsfiles, 'D')
344
352
 
345
353
  # gather and transfer tar files to Globus Quasar Servers
346
- def globus_transfer_action(self):
354
+ def globus_transfer_action(self):
347
355
  dsfiles = {'B' : {}, 'D' : {}}
348
356
  fcnt = self.gather_dataset_tarfiles(dsfiles)
349
357
  if fcnt and self.dstart:
@@ -393,6 +401,17 @@ class DsQuasar(PgCMD, PgSplit):
393
401
  return tcnt
394
402
  return self.pgget('bfile', '', bcnd, self.LGWNEX)
395
403
 
404
+ # describe how much of the registered work this run has finished. the tar counts alone
405
+ # do not show it: -A 3 tars each input file right after creating it, so the 'N' count
406
+ # stays near zero exactly while building is the busy phase, which reads as almost done
407
+ # when the run is barely started. dcount/fcount is the honest measure. the live dcnt is
408
+ # used rather than the recorded dcount, which is only flushed to RDADB every pstep files
409
+ def batch_done_count(self):
410
+ fcnt = self.DSCHK['fcount'] if 'fcount' in self.DSCHK else 0
411
+ dcnt = self.PGBACK['dcnt']
412
+ if not fcnt: return "{} file(s) done".format(dcnt)
413
+ return "{} of {} file(s) done({}%)".format(dcnt, fcnt, int(100*dcnt/fcnt))
414
+
396
415
  # look up the dscheck record registered for a worker's argv, using the same lookup key
397
416
  # init_dscheck builds
398
417
  def lookup_worker_dscheck(self, argv):
@@ -403,18 +422,29 @@ class DsQuasar(PgCMD, PgSplit):
403
422
  dargv = dargv[0:100]
404
423
  return self.get_dscheck("dsquasar", dargv, self.PGBACK['workdir'], self.PGLOG['CURUID'], dargx, self.LOGWRN)
405
424
 
406
- # a running batch job counts as stalled once it has run longer than WORKGRACE and has
407
- # still completed no more than MINWDONE of its recorded work; anything further along
408
- # than that is progressing and is left alone. the done fraction dcount/fcount is used
409
- # instead of a raw count because dcount counts GDEX files for -A 3 but tar files for
410
- # -A 2/-A 4. a job that is not locked, not started yet, or whose process is gone is not
411
- # stalled: init_dscheck already restarts a dead one.
425
+ # a running batch job counts as stalled once it has run longer than WORKGRACE and its
426
+ # progress so far projects to less than MINWPROJ of its recorded work by the walltime:
427
+ # such a job cannot finish this attempt however long it is left alone, so an extra
428
+ # worker is what helps it, not more retries.
429
+ # the test is on the RATE, not on an absolute done fraction as it first was. a fixed
430
+ # 1% threshold only catches a job that is nearly frozen, and reads a job that needs
431
+ # weeks as progressing: the real case was 2% done after 13H35M of a 24 hour walltime,
432
+ # which is above 1% yet projects to 3.5% by the walltime and ~28 days to finish, and so
433
+ # never got a second worker.
434
+ # the done fraction dcount/fcount is used instead of a raw count because dcount counts
435
+ # GDEX files for -A 3 but tar files for -A 2/-A 4. a job that is not locked, not started
436
+ # yet, or whose process is gone is not stalled: init_dscheck already restarts a dead one.
437
+ # WALLTIME is the 24 hours asked for at submit time, since only the submitting process
438
+ # runs this and set_walltime_deadline reads the granted walltime back in the batch job.
439
+ # when the daemon caps it to a shorter queue limit the projection is optimistic, which
440
+ # errs towards leaving the job alone.
412
441
  def worker_progress_stalled(self, pgrec):
413
442
  if not (pgrec['pid'] and pgrec['stttime']): return False
414
443
  if self.check_host_pid(pgrec['lockhost'], pgrec['pid']) <= 0: return False
415
- if (tm() - pgrec['stttime']) < self.WORKGRACE: return False
444
+ elapsed = tm() - pgrec['stttime']
445
+ if elapsed < self.WORKGRACE: return False
416
446
  done = pgrec['dcount']/pgrec['fcount'] if pgrec['fcount'] else 0
417
- return done <= self.MINWDONE
447
+ return (done*self.WALLTIME/elapsed) < self.MINWPROJ
418
448
 
419
449
  # whether more than one worker may run the current action. two workers stay off each
420
450
  # other's files only through dataset locking, so the action must lock what it works on:
@@ -560,6 +590,7 @@ class DsQuasar(PgCMD, PgSplit):
560
590
  'size' : 0, 'infiles' : [], 'instr' : '', 'qdsids' : [], 'abids' : [],
561
591
  'dslocks' : [], 'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
562
592
  for dsid in bfiles:
593
+ self.check_batch_deadline()
563
594
  if len(qinfo['dsids']) == self.DSCNT: self.process_one_backup_file(qinfo, True, False)
564
595
  if self.PGBACK['dolock'] and dsid not in qinfo['dslocks']:
565
596
  if self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
@@ -676,6 +707,7 @@ class DsQuasar(PgCMD, PgSplit):
676
707
  if self.lock_dataset(dsid, 1, self.LOGERR) < 1: continue
677
708
  qinfo['dslocks'].append(dsid)
678
709
  for bid in bfiles[dsid]:
710
+ self.check_batch_deadline()
679
711
  # the status 'N' records were gathered before this dataset was locked, so they
680
712
  # can be hours old; confirm each one is still untarred before spending a tar on
681
713
  # it. holding the dataset lock makes this check race free against a second
@@ -723,6 +755,7 @@ class DsQuasar(PgCMD, PgSplit):
723
755
  'size' : 0, 'fromfiles' : [], 'tofiles' : [], 'qdsids' : [],
724
756
  'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
725
757
  for bid in bfiles:
758
+ self.check_batch_deadline()
726
759
  binfo = bfiles[bid]
727
760
  for dsid in binfo['dsids']:
728
761
  if dsid not in qinfo['dsids']: qinfo['dsids'].append(dsid)
@@ -752,6 +785,11 @@ class DsQuasar(PgCMD, PgSplit):
752
785
 
753
786
  # wait all child processes finish and then quit the main program
754
787
  def quit_dsquasar(self, qinfo, msg = None):
788
+ # a forked child reaches here through its own errcnt, which is a copy of the parent's
789
+ # made at the fork. only the parent reports for the run: a child reporting would email
790
+ # a second report, drop the progress report parked in dscheck.einfo by the parent, and
791
+ # mark the shared check record failed while the parent is still working
792
+ if self.PGSIG['PPID'] > 1: sys.exit(1)
755
793
  if self.PGBACK['mproc'] > 1: self.check_child(None, 0, self.LOGWRN, 1)
756
794
  if qinfo:
757
795
  if 'dslocks' in qinfo and qinfo['dslocks']:
@@ -798,35 +836,52 @@ class DsQuasar(PgCMD, PgSplit):
798
836
  self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME),
799
837
  self.seconds_to_string_time(self.MAXRUNTIME)), self.LOGWRN)
800
838
 
801
- # email a report for a run under dscheck control. the report is normally sent right
802
- # away, but is cached into dscheck.einfo instead if a progress report is cached there
803
- # already, so that the final report replaces it; the dscheck daemon sends and clears
804
- # einfo once the check record is unlocked.
839
+ # email a report for a run under dscheck control. the progress report is parked in
840
+ # dscheck.einfo and the final report is sent right away. the run holds the check lock
841
+ # while it works, and the dscheck daemon only mails unlocked records, so a parked
842
+ # report goes out exactly when the check is unlocked - right after PBS kills the job
843
+ # off its walltime, in the same pass that resubmits it.
844
+ # a record with a non-empty einfo is skipped by both the start pass and the purge pass
845
+ # of the daemon, so the final report must drop the parked one: left behind it would be
846
+ # mailed a second time as a stale progress report, and would hold the finished check
847
+ # back from being purged.
805
848
  def report_dscheck_email(self, title, cache = 0):
806
849
  cnd = "cindex = {}".format(self.PGLOG['DSCHECK']['cindex'])
807
- if not (cache or self.PGBACK['einfo']):
808
- return self.build_customized_email("dscheck", "einfo", cnd, title, self.LOGWRN)
809
- msg = self.get_email()
810
- if not msg: return self.FAILURE
811
- sender = self.PGLOG['CURUID'] + "@ucar.edu"
812
- receiver = self.PGLOG['EMLADDR'] if self.PGLOG['EMLADDR'] else sender
813
- if receiver.find(sender) < 0: self.add_carbon_copy(sender, 1)
814
- ebuf = "From: {}\nTo: {}\n".format(sender, receiver)
815
- if self.PGLOG['CCDADDR']: ebuf += "Cc: {}\n".format(self.PGLOG['CCDADDR'])
816
- ebuf += "Subject: {}!\n\n{}\n".format(title, msg)
817
- self.PGBACK['einfo'] = 1
818
- return self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
850
+ if cache:
851
+ msg = self.get_email()
852
+ if not msg: return self.FAILURE
853
+ sender = self.PGLOG['CURUID'] + "@ucar.edu"
854
+ receiver = self.PGLOG['EMLADDR'] if self.PGLOG['EMLADDR'] else sender
855
+ if receiver.find(sender) < 0: self.add_carbon_copy(sender, 1)
856
+ ebuf = "From: {}\nTo: {}\n".format(sender, receiver)
857
+ if self.PGLOG['CCDADDR']: ebuf += "Cc: {}\n".format(self.PGLOG['CCDADDR'])
858
+ ebuf += "Subject: {}!\n\n{}\n".format(title, msg)
859
+ self.PGBACK['einfo'] = 1 # report the progress only once per run
860
+ return self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
861
+ estat = self.build_customized_email("dscheck", "einfo", cnd, title, self.LOGWRN)
862
+ # a failed send has already replaced the parked report via cache_customized_email()
863
+ if estat == self.SUCCESS and self.PGBACK['einfo']:
864
+ if self.pgexec("UPDATE dscheck set einfo = NULL WHERE " + cnd, self.LOGWRN):
865
+ self.PGBACK['einfo'] = 0
866
+ else:
867
+ self.pglog("Cannot clean the progress report of {}; it is emailed again".format(cnd), self.LOGWRN)
868
+ return estat
819
869
 
820
870
  # guard the long PBS batch jobs (-A 2/3/4/6) against the walltime: once past MAXRUNTIME
821
- # the run keeps going, but a progress report is cached into dscheck.einfo so an email is
822
- # still sent, by the dscheck daemon, if PBS kills the job at the walltime. the final
823
- # report replaces it if the run does finish in time. the live email buffers are saved
824
- # and put back, so the final report still carries everything logged before the cutoff.
871
+ # the run keeps going, but a progress report is parked in dscheck.einfo so that
872
+ # something is reported even if PBS kills the job at the walltime. the final report
873
+ # follows and drops the parked one if the run does finish in time. the live email
874
+ # buffers are saved and put back, so the final report still carries everything logged
875
+ # before the cutoff.
825
876
  # both queue depths are reported rather than the current phase's: -A 3 and -A 6 run the
826
877
  # build and the transfer phase in one run, and -A 3 tars each input file right after
827
878
  # creating it, so its status 'N' count stays near zero while that is the busy phase.
828
879
  # a no-op for command-line runs, other actions, before the cutoff, or once a progress
829
880
  # report is cached.
881
+ # called from the top of every per-item loop, not only where a tar is dispatched: a long
882
+ # stretch of files that are skipped (already backed up, or not enough accumulated size to
883
+ # tar yet) would otherwise walk past the cutoff without ever reaching the check, which is
884
+ # how a 23h run reported nothing.
830
885
  def check_batch_deadline(self):
831
886
  act = self.PGBACK['action']
832
887
  if self.PGBACK['einfo'] or not (self.PGBACK['doemail'] and self.PGLOG['DSCHECK']): return
@@ -838,12 +893,13 @@ class DsQuasar(PgCMD, PgSplit):
838
893
  etime = self.seconds_to_string_time(int(elapsed))
839
894
  amsg = self.ACTMSG[act]
840
895
  bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
896
+ dmsg = self.batch_done_count()
841
897
  rmsg = "{} {} tar file(s) left to build and {} to transfer".format(tcnt, bmsg, bcnt)
842
- msg = "{}: Still running after {} of the {} PBS walltime, with {}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), rmsg)
898
+ msg = "{}: Still running after {} of the {} PBS walltime, {}, with {}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), dmsg, rmsg)
843
899
  self.pglog(self.INDENT + msg, self.LOGACT)
844
900
  saved = {key : self.PGLOG[key] for key in ('EMLMSG', 'ERRMSG', 'ERRCNT', 'SUMMSG', 'PRGMSG')}
845
- self.set_email("{}: {} still in progress after {}, with {}!".format(self.PGBACK['cmd'], amsg, etime, rmsg), self.EMLTOP)
846
- title = "dsquasar: {} in progress ({} to build, {} to transfer)".format(amsg, tcnt, bcnt)
901
+ self.set_email("{}: {} still in progress after {}, {}, with {}!".format(self.PGBACK['cmd'], amsg, etime, dmsg, rmsg), self.EMLTOP)
902
+ title = "dsquasar: {} in progress ({})".format(amsg, dmsg)
847
903
  if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
848
904
  self.report_dscheck_email(title, 1)
849
905
  self.PGLOG.update(saved)
@@ -1068,6 +1124,7 @@ class DsQuasar(PgCMD, PgSplit):
1068
1124
  fcate = filetype.lower()
1069
1125
  tname = fcate + 'file'
1070
1126
  for i in range(fcnt):
1127
+ self.check_batch_deadline()
1071
1128
  pgrec = recs[i]
1072
1129
  if not self.evaluate_file_stat(dsid, fcate, pgrec): continue
1073
1130
  fname = pgrec[tname]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.12
3
+ Version: 3.0.14
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar