rda-python-dsquasar 3.0.11__tar.gz → 3.0.13__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (20) hide show
  1. {rda_python_dsquasar-3.0.11/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.13}/PKG-INFO +1 -1
  2. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/pyproject.toml +1 -1
  3. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/dsquasar.py +99 -43
  4. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13/src/rda_python_dsquasar.egg-info}/PKG-INFO +1 -1
  5. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/LICENSE +0 -0
  6. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/MANIFEST.in +0 -0
  7. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/README.md +0 -0
  8. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/setup.cfg +0 -0
  9. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/__init__.py +0 -0
  10. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/ds_quasar.py +0 -0
  11. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/dsquasar.usg +0 -0
  12. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/dstacc.py +0 -0
  13. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/taccrec.py +0 -0
  14. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar/tacctar.py +0 -0
  15. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar.egg-info/SOURCES.txt +0 -0
  16. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar.egg-info/dependency_links.txt +0 -0
  17. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar.egg-info/entry_points.txt +0 -0
  18. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar.egg-info/requires.txt +0 -0
  19. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/src/rda_python_dsquasar.egg-info/top_level.txt +0 -0
  20. {rda_python_dsquasar-3.0.11 → rda_python_dsquasar-3.0.13}/test/test_dsquasar.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.11
3
+ Version: 3.0.13
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "rda_python_dsquasar"
7
- version = "3.0.11"
7
+ version = "3.0.13"
8
8
  authors = [
9
9
  { name="Zaihua Ji", email="zji@ucar.edu" },
10
10
  ]
@@ -94,8 +94,9 @@ class DsQuasar(PgCMD, PgSplit):
94
94
  # concurrent transfers, so extra upload processes just wait on it.
95
95
  self.MPBLIMIT = 5000 # tar file count per batch process for uploading
96
96
  self.MPBMAX = 4 # maximum number of batch processes for uploading
97
- self.MAXRUNTIME = 23*3600 # 23 hours; stop before the 24-hour PBS walltime
98
- self.ONEHOUR = 3600 # seconds; time headroom needed to finish before walltime
97
+ self.ONEHOUR = 3600 # seconds; headroom kept to report before the walltime
98
+ self.WALLTIME = 24*3600 # PBS walltime assumed, refreshed from PBS at start of a run
99
+ self.MAXRUNTIME = self.WALLTIME - self.ONEHOUR # cutoff to cache a progress report
99
100
  # a repeat submit normally blocks as a duplicate while the batch job is running. if
100
101
  # that job is running but barely progressing, an extra worker is submitted instead
101
102
  # (see pick_worker_slot). progress is measured as the fraction of the recorded
@@ -119,7 +120,7 @@ class DsQuasar(PgCMD, PgSplit):
119
120
  'dolock' : 1,
120
121
  'doemail' : 0,
121
122
  'starttime' : 0, # wall-clock start of the run, for the PBS walltime guard
122
- 'tardone' : 0, # tar files dispatched so far, for the finish-rate estimate
123
+ 'einfo' : 0, # set once a progress report is cached into dscheck.einfo
123
124
  'maxworkers' : self.MAXWORKERS, # -W, maximum concurrent workers per command
124
125
  'worker' : 1, # -w, this run's worker slot; >1 for an added extra worker
125
126
  'cmd' : None
@@ -195,6 +196,7 @@ class DsQuasar(PgCMD, PgSplit):
195
196
  def start_actions(self):
196
197
  self.cmdlog(self.PGBACK['cmd'])
197
198
  self.PGBACK['starttime'] = tm()
199
+ self.set_walltime_deadline()
198
200
  if self.sopts['u']:
199
201
  if self.sopts['a']: self.pglog("-u: Dataset IDs must be provided to Unlock datasets", self.LOGWRN)
200
202
  self.unlock_datasets()
@@ -226,6 +228,12 @@ class DsQuasar(PgCMD, PgSplit):
226
228
  if acts&self.TARACT: self.build_tarfile_action()
227
229
  if acts&self.CINACT: self.create_infile_action()
228
230
  if self.dstart and acts&self.TBACTS == self.TBACTS: self.globus_transfer_action()
231
+ # the queues were empty, so nothing was processed and nothing failed. no dscheck
232
+ # record is registered for such a run either, so drop the report rather than mail
233
+ # one about 0 files. a parked progress report still forces one, to clean it up
234
+ if not (self.PGBACK['bckcnt'] or self.PGBACK['errcnt'] or
235
+ self.PGLOG['ERRCNT'] or self.PGBACK['einfo']):
236
+ self.PGBACK['doemail'] = 0
229
237
  if self.PGBACK['doemail']:
230
238
  amsg = self.ACTMSG[self.PGBACK['action']]
231
239
  bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
@@ -236,9 +244,7 @@ class DsQuasar(PgCMD, PgSplit):
236
244
  if self.PGBACK['bckcnt']: title += "({})".format(self.PGBACK['bckcnt'])
237
245
  if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
238
246
  if self.PGLOG['DSCHECK']:
239
- tbl = "dscheck"
240
- cnd = "cindex = {}".format(self.PGLOG['DSCHECK']['cindex'])
241
- self.build_customized_email(tbl, "einfo", cnd, title, self.LOGWRN)
247
+ self.report_dscheck_email(title)
242
248
  else:
243
249
  self.pglog(title, self.LOGWRN|self.SNDEML)
244
250
  if self.PGBACK['pstep']: self.record_dscheck_status("D")
@@ -343,7 +349,7 @@ class DsQuasar(PgCMD, PgSplit):
343
349
  self.backup_dataset_infiles(dsfiles, 'D')
344
350
 
345
351
  # gather and transfer tar files to Globus Quasar Servers
346
- def globus_transfer_action(self):
352
+ def globus_transfer_action(self):
347
353
  dsfiles = {'B' : {}, 'D' : {}}
348
354
  fcnt = self.gather_dataset_tarfiles(dsfiles)
349
355
  if fcnt and self.dstart:
@@ -752,6 +758,11 @@ class DsQuasar(PgCMD, PgSplit):
752
758
 
753
759
  # wait all child processes finish and then quit the main program
754
760
  def quit_dsquasar(self, qinfo, msg = None):
761
+ # a forked child reaches here through its own errcnt, which is a copy of the parent's
762
+ # made at the fork. only the parent reports for the run: a child reporting would email
763
+ # a second report, drop the progress report parked in dscheck.einfo by the parent, and
764
+ # mark the shared check record failed while the parent is still working
765
+ if self.PGSIG['PPID'] > 1: sys.exit(1)
755
766
  if self.PGBACK['mproc'] > 1: self.check_child(None, 0, self.LOGWRN, 1)
756
767
  if qinfo:
757
768
  if 'dslocks' in qinfo and qinfo['dslocks']:
@@ -770,49 +781,96 @@ class DsQuasar(PgCMD, PgSplit):
770
781
  self.set_email("{}: Quit {} for {} Files of {}!".format(self.PGBACK['cmd'], amsg, bmsg, dmsg), self.EMLTOP)
771
782
  title = "dsquasar: Quit {} Error({})".format(amsg, self.PGBACK['errcnt'])
772
783
  if self.PGLOG['DSCHECK']:
773
- tbl = "dscheck"
774
- cnd = "cindex = {}".format(self.PGLOG['DSCHECK']['cindex'])
775
- self.build_customized_email(tbl, "einfo", cnd, title, self.LOGWRN)
784
+ self.report_dscheck_email(title)
776
785
  else:
777
786
  self.pglog(title, self.LOGWRN|self.SNDEML)
778
787
  if self.PGBACK['pstep']: self.record_dscheck_status("F")
779
788
  self.pgexit(0)
780
789
 
781
- # guard the -A 3 (Create Input&Tar, build status 'N' infiles) and -A 4 (Transfer,
782
- # send status 'T' tars) PBS batch jobs against the 24-hour walltime: once past
783
- # MAXRUNTIME, if the tar files still to process cannot finish within the last hour at
784
- # the rate achieved so far, send a progress email and stop cleanly so the report is
785
- # not lost to a PBS timeout; the remaining records are left for the next scheduled run
786
- # to resume. a no-op for command-line runs, other actions, or before the cutoff.
787
- def check_batch_deadline(self, qinfo):
790
+ # a batch run normally gets the full 24 hours asked for at submit time, but the dscheck
791
+ # daemon caps that to the maximum of the PBS queue it lands in - only 6 hours for the
792
+ # default queue - and to any planned system down. read the granted walltime back from
793
+ # PBS so that the run reports its progress with an hour to spare whatever the limit
794
+ # turns out to be, and keep the assumed 24 hours for a command-line run or when PBS
795
+ # cannot tell us.
796
+ def set_walltime_deadline(self):
797
+ if self.PGLOG['CURBID'] < 1: return
798
+ stat = self.get_pbs_info(str(self.PGLOG['CURBID']), 0, self.LOGWRN)
799
+ ms = re.match(r'^(\d+):(\d+)(?::(\d+))?$', stat['ReqdTime']) if stat.get('ReqdTime') else None
800
+ if not ms:
801
+ self.pglog("Cannot read the PBS walltime of Job {}; assume {}".format(
802
+ self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME)), self.LOGWRN)
803
+ return
804
+ wtime = 3600*int(ms.group(1)) + 60*int(ms.group(2)) + (int(ms.group(3)) if ms.group(3) else 0)
805
+ if wtime <= self.ONEHOUR: return # too short to keep the reporting headroom
806
+ self.WALLTIME = wtime
807
+ self.MAXRUNTIME = wtime - self.ONEHOUR
808
+ self.pglog("PBS Job {} walltime {}: report progress after {}".format(
809
+ self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME),
810
+ self.seconds_to_string_time(self.MAXRUNTIME)), self.LOGWRN)
811
+
812
+ # email a report for a run under dscheck control. the progress report is parked in
813
+ # dscheck.einfo and the final report is sent right away. the run holds the check lock
814
+ # while it works, and the dscheck daemon only mails unlocked records, so a parked
815
+ # report goes out exactly when the check is unlocked - right after PBS kills the job
816
+ # off its walltime, in the same pass that resubmits it.
817
+ # a record with a non-empty einfo is skipped by both the start pass and the purge pass
818
+ # of the daemon, so the final report must drop the parked one: left behind it would be
819
+ # mailed a second time as a stale progress report, and would hold the finished check
820
+ # back from being purged.
821
+ def report_dscheck_email(self, title, cache = 0):
822
+ cnd = "cindex = {}".format(self.PGLOG['DSCHECK']['cindex'])
823
+ if cache:
824
+ msg = self.get_email()
825
+ if not msg: return self.FAILURE
826
+ sender = self.PGLOG['CURUID'] + "@ucar.edu"
827
+ receiver = self.PGLOG['EMLADDR'] if self.PGLOG['EMLADDR'] else sender
828
+ if receiver.find(sender) < 0: self.add_carbon_copy(sender, 1)
829
+ ebuf = "From: {}\nTo: {}\n".format(sender, receiver)
830
+ if self.PGLOG['CCDADDR']: ebuf += "Cc: {}\n".format(self.PGLOG['CCDADDR'])
831
+ ebuf += "Subject: {}!\n\n{}\n".format(title, msg)
832
+ self.PGBACK['einfo'] = 1 # report the progress only once per run
833
+ return self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
834
+ estat = self.build_customized_email("dscheck", "einfo", cnd, title, self.LOGWRN)
835
+ # a failed send has already replaced the parked report via cache_customized_email()
836
+ if estat == self.SUCCESS and self.PGBACK['einfo']:
837
+ if self.pgexec("UPDATE dscheck set einfo = NULL WHERE " + cnd, self.LOGWRN):
838
+ self.PGBACK['einfo'] = 0
839
+ else:
840
+ self.pglog("Cannot clean the progress report of {}; it is emailed again".format(cnd), self.LOGWRN)
841
+ return estat
842
+
843
+ # guard the long PBS batch jobs (-A 2/3/4/6) against the walltime: once past MAXRUNTIME
844
+ # the run keeps going, but a progress report is parked in dscheck.einfo so that
845
+ # something is reported even if PBS kills the job at the walltime. the final report
846
+ # follows and drops the parked one if the run does finish in time. the live email
847
+ # buffers are saved and put back, so the final report still carries everything logged
848
+ # before the cutoff.
849
+ # both queue depths are reported rather than the current phase's: -A 3 and -A 6 run the
850
+ # build and the transfer phase in one run, and -A 3 tars each input file right after
851
+ # creating it, so its status 'N' count stays near zero while that is the busy phase.
852
+ # a no-op for command-line runs, other actions, before the cutoff, or once a progress
853
+ # report is cached.
854
+ def check_batch_deadline(self):
788
855
  act = self.PGBACK['action']
789
- if self.PGLOG['CURBID'] < 1 or act not in (self.CTACTS, self.BCKACT): return
856
+ if self.PGBACK['einfo'] or not (self.PGBACK['doemail'] and self.PGLOG['DSCHECK']): return
857
+ if self.PGLOG['CURBID'] < 1 or act not in (self.TARACT, self.CTACTS, self.BCKACT, self.TBACTS): return
790
858
  elapsed = tm() - self.PGBACK['starttime']
791
859
  if elapsed < self.MAXRUNTIME: return
792
- status, work = ('N', 'build') if act == self.CTACTS else ('T', 'transfer')
793
- remaining = self.batch_tar_count(status)
794
- if remaining < 1: return
795
- done = self.PGBACK['tardone']
796
- need = remaining*elapsed/done if done > 0 else elapsed
797
- if need <= self.ONEHOUR: return # enough time left to finish before the walltime
798
- if self.PGBACK['mproc'] > 1: self.check_child(None, 0, self.LOGWRN, 1) # wait all children
799
- if qinfo and qinfo.get('dslocks'):
800
- for dsid in qinfo['dslocks']: self.lock_dataset(dsid, 0, self.LGEREX)
860
+ tcnt = self.batch_tar_count('N')
861
+ bcnt = self.batch_tar_count('T')
801
862
  etime = self.seconds_to_string_time(int(elapsed))
802
863
  amsg = self.ACTMSG[act]
803
- msg = "{}: Stopped after {} with {} tar file(s) still to {} - not enough time to finish before the 24-hour PBS walltime; the next scheduled run will resume".format(amsg, etime, remaining, work)
864
+ bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
865
+ rmsg = "{} {} tar file(s) left to build and {} to transfer".format(tcnt, bmsg, bcnt)
866
+ msg = "{}: Still running after {} of the {} PBS walltime, with {}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), rmsg)
804
867
  self.pglog(self.INDENT + msg, self.LOGACT)
805
- if self.PGBACK['doemail']:
806
- bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
807
- self.set_email("{}: {} - stopped early before the PBS walltime with {} {} tar file(s) remaining!".format(self.PGBACK['cmd'], amsg, remaining, bmsg), self.EMLTOP)
808
- title = "dsquasar: {} stopped early ({} remaining)".format(amsg, remaining)
809
- if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
810
- if self.PGLOG['DSCHECK']:
811
- self.build_customized_email("dscheck", "einfo", "cindex = {}".format(self.PGLOG['DSCHECK']['cindex']), title, self.LOGWRN)
812
- else:
813
- self.pglog(title, self.LOGWRN|self.SNDEML)
814
- if self.PGBACK['pstep']: self.record_dscheck_status("D")
815
- self.pgexit(0)
868
+ saved = {key : self.PGLOG[key] for key in ('EMLMSG', 'ERRMSG', 'ERRCNT', 'SUMMSG', 'PRGMSG')}
869
+ self.set_email("{}: {} still in progress after {}, with {}!".format(self.PGBACK['cmd'], amsg, etime, rmsg), self.EMLTOP)
870
+ title = "dsquasar: {} in progress ({} to build, {} to transfer)".format(amsg, tcnt, bcnt)
871
+ if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
872
+ self.report_dscheck_email(title, 1)
873
+ self.PGLOG.update(saved)
816
874
 
817
875
  # recompute the confirmed backup counts from RDADB after all child processes
818
876
  # finished, so a multi-process summary reflects succeeded (not just started) work
@@ -850,7 +908,7 @@ class DsQuasar(PgCMD, PgSplit):
850
908
  def process_one_backup_file(self, qinfo, addback, keepid = False):
851
909
  ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
852
910
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
853
- self.check_batch_deadline(qinfo)
911
+ self.check_batch_deadline()
854
912
  dsids = qinfo['dsids']
855
913
  dcnt = len(dsids)
856
914
  if dcnt == 0: return
@@ -908,7 +966,6 @@ class DsQuasar(PgCMD, PgSplit):
908
966
  if stat:
909
967
  # reset qinfo after quasar backup
910
968
  qinfo['qcnt'] += 1
911
- self.PGBACK['tardone'] += 1 # cumulative dispatched tars for the walltime finish-rate
912
969
  qinfo['qfcnt'] += fcnt
913
970
  qinfo['qsize'] += fsize
914
971
  for dsid in dsids:
@@ -931,7 +988,7 @@ class DsQuasar(PgCMD, PgSplit):
931
988
  def transfer_quasar_tarfiles(self, qinfo):
932
989
  ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
933
990
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
934
- self.check_batch_deadline(qinfo)
991
+ self.check_batch_deadline()
935
992
  # prepare for backup one tar file
936
993
  dsids = qinfo['dsids']
937
994
  bids = qinfo['bids']
@@ -985,7 +1042,6 @@ class DsQuasar(PgCMD, PgSplit):
985
1042
  if dstat and bstat:
986
1043
  # reset qinfo after quasar backup
987
1044
  qinfo['qcnt'] += bcnt
988
- self.PGBACK['tardone'] += bcnt # cumulative transferred tars for the walltime finish-rate
989
1045
  qinfo['qfcnt'] += fcnt
990
1046
  qinfo['qsize'] += fsize
991
1047
  for dsid in dsids:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.11
3
+ Version: 3.0.13
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar