rda-python-dsquasar 3.0.19__tar.gz → 3.0.20__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (20) hide show
  1. {rda_python_dsquasar-3.0.19/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.20}/PKG-INFO +1 -1
  2. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/pyproject.toml +1 -1
  3. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/dsquasar.py +123 -32
  4. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20/src/rda_python_dsquasar.egg-info}/PKG-INFO +1 -1
  5. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/LICENSE +0 -0
  6. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/MANIFEST.in +0 -0
  7. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/README.md +0 -0
  8. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/setup.cfg +0 -0
  9. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/__init__.py +0 -0
  10. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/ds_quasar.py +0 -0
  11. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/dsquasar.usg +0 -0
  12. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/dstacc.py +0 -0
  13. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/taccrec.py +0 -0
  14. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/tacctar.py +0 -0
  15. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/SOURCES.txt +0 -0
  16. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/dependency_links.txt +0 -0
  17. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/entry_points.txt +0 -0
  18. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/requires.txt +0 -0
  19. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/top_level.txt +0 -0
  20. {rda_python_dsquasar-3.0.19 → rda_python_dsquasar-3.0.20}/test/test_dsquasar.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.19
3
+ Version: 3.0.20
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "rda_python_dsquasar"
7
- version = "3.0.19"
7
+ version = "3.0.20"
8
8
  authors = [
9
9
  { name="Zaihua Ji", email="zji@ucar.edu" },
10
10
  ]
@@ -23,6 +23,7 @@ import os
23
23
  import re
24
24
  import sys
25
25
  import time
26
+ import signal
26
27
  from os import path as op
27
28
  from time import time as tm
28
29
  from rda_python_common.pg_cmd import PgCMD
@@ -96,6 +97,7 @@ class DsQuasar(PgCMD, PgSplit):
96
97
  self.MPBLIMIT = 5000 # tar file count per batch process for uploading
97
98
  self.MPBMAX = 4 # maximum number of batch processes for uploading
98
99
  self.ONEHOUR = 3600 # seconds; headroom kept to report before the walltime
100
+ self.ERETRY = 600 # seconds to wait before retrying a failed progress report
99
101
  self.WALLTIME = 24*3600 # PBS walltime assumed, refreshed from PBS at start of a run
100
102
  self.MAXRUNTIME = self.WALLTIME - self.ONEHOUR # cutoff to cache a progress report
101
103
  # a repeat submit normally blocks as a duplicate while the batch job is running. if
@@ -124,6 +126,7 @@ class DsQuasar(PgCMD, PgSplit):
124
126
  'doemail' : 0,
125
127
  'starttime' : 0, # wall-clock start of the run, for the PBS walltime guard
126
128
  'einfo' : 0, # set once a progress report is cached into dscheck.einfo
129
+ 'eretry' : 0, # earliest retry time after a progress report failed to cache
127
130
  'maxworkers' : self.MAXWORKERS, # -W, maximum concurrent workers per command
128
131
  'worker' : 1, # -w, this run's worker slot; >1 for an added extra worker
129
132
  'cmd' : None
@@ -200,6 +203,7 @@ class DsQuasar(PgCMD, PgSplit):
200
203
  self.cmdlog(self.PGBACK['cmd'])
201
204
  self.PGBACK['starttime'] = tm()
202
205
  self.set_walltime_deadline()
206
+ self.catch_batch_termination()
203
207
  if self.sopts['u']:
204
208
  if self.sopts['a']: self.pglog("-u: Dataset IDs must be provided to Unlock datasets", self.LOGWRN)
205
209
  self.unlock_datasets()
@@ -238,7 +242,7 @@ class DsQuasar(PgCMD, PgSplit):
238
242
  self.PGLOG['ERRCNT'] or self.PGBACK['einfo']):
239
243
  self.PGBACK['doemail'] = 0
240
244
  if self.PGBACK['doemail']:
241
- amsg = self.ACTMSG[self.PGBACK['action']]
245
+ amsg = self.action_message()
242
246
  bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
243
247
  dcnt = len(self.dsids)
244
248
  dmsg = self.dsids[0] if dcnt == 1 else "{} datasets".format(dcnt if dcnt > 1 else 'All')
@@ -791,7 +795,7 @@ class DsQuasar(PgCMD, PgSplit):
791
795
  # a second report, drop the progress report parked in dscheck.einfo by the parent, and
792
796
  # mark the shared check record failed while the parent is still working
793
797
  if self.PGSIG['PPID'] > 1: sys.exit(1)
794
- if self.PGBACK['mproc'] > 1: self.check_child(None, 0, self.LOGWRN, 1)
798
+ self.wait_all_children()
795
799
  if qinfo:
796
800
  if 'dslocks' in qinfo and qinfo['dslocks']:
797
801
  for dsid in qinfo['dslocks']: self.lock_dataset(dsid, 0, self.LGEREX)
@@ -799,7 +803,7 @@ class DsQuasar(PgCMD, PgSplit):
799
803
  dcnt = len(qinfo['qdsids'])
800
804
  fcnt = qinfo['qfcnt']
801
805
  ssize = self.format_float_value(qinfo['qsize'])
802
- amsg = self.ACTMSG[self.PGBACK['action']]
806
+ amsg = self.action_message()
803
807
  bmsg = self.BACKMSG[qinfo['backflag']]
804
808
  dmsg = qinfo['qdsids'][0] if dcnt == 1 else "{} datasets".format(dcnt)
805
809
  msg = "Quit {}: {} {} files for {}({}) files of {}".format(amsg, qcnt, bmsg, fcnt, ssize, dmsg)
@@ -815,6 +819,13 @@ class DsQuasar(PgCMD, PgSplit):
815
819
  if self.PGBACK['pstep']: self.record_dscheck_status("F")
816
820
  self.pgexit(0)
817
821
 
822
+ # ACTMSG carries no wording for the hidden actions (-A 32/64), so name those by number
823
+ # rather than raise a KeyError: a report that cannot be worded is still a report that
824
+ # has to go out, and crashing while building it loses the run as silently as a kill does
825
+ def action_message(self, act = None):
826
+ if act is None: act = self.PGBACK['action']
827
+ return self.ACTMSG[act] if act in self.ACTMSG else "Action {}".format(act)
828
+
818
829
  # a batch run normally gets the full 24 hours asked for at submit time, but the dscheck
819
830
  # daemon caps that to the maximum of the PBS queue it lands in - only 6 hours for the
820
831
  # default queue - and to any planned system down. read the granted walltime back from
@@ -830,13 +841,54 @@ class DsQuasar(PgCMD, PgSplit):
830
841
  self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME)), self.LOGWRN)
831
842
  return
832
843
  wtime = 3600*int(ms.group(1)) + 60*int(ms.group(2)) + (int(ms.group(3)) if ms.group(3) else 0)
833
- if wtime <= self.ONEHOUR: return # too short to keep the reporting headroom
844
+ if wtime <= self.ONEHOUR: # too short to keep the reporting headroom
845
+ self.pglog("PBS Job {} walltime {} is too short to report progress ahead of; assume {}".format(
846
+ self.PGLOG['CURBID'], self.seconds_to_string_time(wtime),
847
+ self.seconds_to_string_time(self.WALLTIME)), self.LOGWRN)
848
+ return
834
849
  self.WALLTIME = wtime
835
850
  self.MAXRUNTIME = wtime - self.ONEHOUR
836
851
  self.pglog("PBS Job {} walltime {}: report progress after {}".format(
837
852
  self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME),
838
853
  self.seconds_to_string_time(self.MAXRUNTIME)), self.LOGWRN)
839
854
 
855
+ # PBS kills a job that runs out of walltime with SIGTERM first and SIGKILL a few seconds
856
+ # later (the MoM's kill_delay, 10 seconds by default), and the same pair is what qdel
857
+ # sends. SIGTERM is the only warning there is, and nothing in the common library traps
858
+ # it, so the run used to die on the spot with whatever it had to say unsaid. catching it
859
+ # turns the last seconds into a report. only a batch run arms this: on the command line
860
+ # SIGTERM must keep killing the process the way the user expects.
861
+ def catch_batch_termination(self):
862
+ if self.PGLOG['CURBID'] < 1: return
863
+ signal.signal(signal.SIGTERM, self.batch_term_handler)
864
+
865
+ # park a progress report and then die of the signal that was sent. this runs on borrowed
866
+ # time, so it does only what it must: no tar queue counts (two database queries we may
867
+ # not get to finish - the done count comes from memory), no dataset unlocking (the
868
+ # dscheck daemon already cleans up after a dead pid). the default handler is restored
869
+ # first so that a second signal, or the SIGKILL that follows, ends the run outright
870
+ # instead of re-entering here should the report hang.
871
+ # the report may have to be written from inside an interrupted database call; that is
872
+ # why it goes through report_dscheck_email, whose cache_customized_email falls back to
873
+ # sending the mail directly when the UPDATE fails.
874
+ def batch_term_handler(self, signum, frame):
875
+ signal.signal(signum, signal.SIG_DFL)
876
+ # a forked child shares the parent's dscheck record: it must not report for the run
877
+ if self.PGSIG['PPID'] > 1: os._exit(1)
878
+ if not self.PGBACK['einfo'] and self.PGBACK['doemail'] and self.PGLOG['DSCHECK']:
879
+ etime = self.seconds_to_string_time(int(tm() - self.PGBACK['starttime']))
880
+ amsg = self.action_message()
881
+ dmsg = self.batch_done_count()
882
+ wmsg = self.seconds_to_string_time(self.WALLTIME)
883
+ self.pglog(self.INDENT + "{}: Terminated by signal {} after {} of the {} PBS walltime, {}".format(
884
+ amsg, signum, etime, wmsg, dmsg), self.LOGACT)
885
+ self.set_email("{}: {} terminated after {} of the {} PBS walltime, {}!".format(
886
+ self.PGBACK['cmd'], amsg, etime, wmsg, dmsg), self.EMLTOP)
887
+ title = "dsquasar: {} terminated ({})".format(amsg, dmsg)
888
+ if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
889
+ self.report_dscheck_email(title, 1)
890
+ os.kill(os.getpid(), signum)
891
+
840
892
  # email a report for a run under dscheck control. the progress report is parked in
841
893
  # dscheck.einfo and the final report is sent right away. the run holds the check lock
842
894
  # while it works, and the dscheck daemon only mails unlocked records, so a parked
@@ -857,8 +909,15 @@ class DsQuasar(PgCMD, PgSplit):
857
909
  ebuf = "From: {}\nTo: {}\n".format(sender, receiver)
858
910
  if self.PGLOG['CCDADDR']: ebuf += "Cc: {}\n".format(self.PGLOG['CCDADDR'])
859
911
  ebuf += "Subject: {}!\n\n{}\n".format(title, msg)
860
- self.PGBACK['einfo'] = 1 # report the progress only once per run
861
- return self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
912
+ # only a report that is actually parked (or sent directly by the fallback) counts
913
+ # as reported. marking it reported before the attempt turned a failed cache into a
914
+ # silent loss: nothing was parked, and the flag stopped every later attempt too
915
+ estat = self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
916
+ if estat:
917
+ self.PGBACK['einfo'] = 1 # report the progress only once per run
918
+ else:
919
+ self.PGBACK['eretry'] = tm() + self.ERETRY
920
+ return estat
862
921
  estat = self.build_customized_email("dscheck", "einfo", cnd, title, self.LOGWRN)
863
922
  # a failed send has already replaced the parked report via cache_customized_email()
864
923
  if estat == self.SUCCESS and self.PGBACK['einfo']:
@@ -868,40 +927,48 @@ class DsQuasar(PgCMD, PgSplit):
868
927
  self.pglog("Cannot clean the progress report of {}; it is emailed again".format(cnd), self.LOGWRN)
869
928
  return estat
870
929
 
871
- # guard the long PBS batch jobs (-A 2/3/4/6/7) against the walltime: once past MAXRUNTIME
872
- # the run keeps going, but a progress report is parked in dscheck.einfo so that
873
- # something is reported even if PBS kills the job at the walltime. the final report
874
- # follows and drops the parked one if the run does finish in time. the live email
875
- # buffers are saved and put back, so the final report still carries everything logged
876
- # before the cutoff.
877
- # both queue depths are reported rather than the current phase's, because -A 3 tars each
878
- # input file right after creating it, so its status 'N' count stays near zero while that
879
- # is the busy phase. they are worded as STATES ('left to build', 'tarred') and not as
880
- # work this run will do: only -A 4 and -A 6 transfer, so calling the status 'T' count
881
- # 'to transfer' in an -A 2 or -A 3 report claims work that run never performs.
882
- # a no-op for command-line runs, other actions, before the cutoff, or once a progress
883
- # report is cached.
930
+ # guard a long PBS batch job against the walltime: once past MAXRUNTIME the run keeps
931
+ # going, but a progress report is parked in dscheck.einfo so that something is reported
932
+ # even if PBS kills the job at the walltime. the final report follows and drops the
933
+ # parked one if the run does finish in time. the live email buffers are saved and put
934
+ # back, so the final report still carries everything logged before the cutoff.
935
+ # PBS sends no catchable warning before the kill - SIGTERM is not trapped and SIGKILL
936
+ # cannot be - so this parked report is the only thing standing between a killed job and
937
+ # a run that is never heard from. every action submitted to PBS is guarded, not just the
938
+ # tar and transfer ones: -A 16 and the hidden -A 128 are submitted with the same 24 hour
939
+ # walltime and used to be excluded, so they died silently.
940
+ # the tar queue depths are only reported for the actions they describe. both depths are
941
+ # reported rather than the current phase's, because -A 3 tars each input file right after
942
+ # creating it, so its status 'N' count stays near zero while that is the busy phase. they
943
+ # are worded as STATES ('left to build', 'tarred') and not as work this run will do: only
944
+ # -A 4 and -A 6 transfer, so calling the status 'T' count 'to transfer' in an -A 2 or
945
+ # -A 3 report claims work that run never performs.
946
+ # a no-op for command-line runs, before the cutoff, or once a progress report is cached.
884
947
  # called from the top of every per-item loop, not only where a tar is dispatched: a long
885
948
  # stretch of files that are skipped (already backed up, or not enough accumulated size to
886
949
  # tar yet) would otherwise walk past the cutoff without ever reaching the check, which is
887
950
  # how a 23h run reported nothing.
888
951
  def check_batch_deadline(self):
889
- act = self.PGBACK['action']
890
952
  if self.PGBACK['einfo'] or not (self.PGBACK['doemail'] and self.PGLOG['DSCHECK']): return
891
- if self.PGLOG['CURBID'] < 1 or act not in (self.TARACT, self.CTACTS, self.BCKACT, self.TBACTS, self.CBACTS): return
953
+ if self.PGLOG['CURBID'] < 1: return # only a PBS batch run has a walltime to beat
892
954
  elapsed = tm() - self.PGBACK['starttime']
893
955
  if elapsed < self.MAXRUNTIME: return
894
- tcnt = self.batch_tar_count('N')
895
- bcnt = self.batch_tar_count('T')
956
+ # a report that could not be parked is retried, but not once per file: the counts
957
+ # below are database queries and the cutoff leaves a whole hour of them otherwise
958
+ if self.PGBACK['eretry'] and tm() < self.PGBACK['eretry']: return
959
+ act = self.PGBACK['action']
896
960
  etime = self.seconds_to_string_time(int(elapsed))
897
- amsg = self.ACTMSG[act]
898
- bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
961
+ amsg = self.action_message(act)
899
962
  dmsg = self.batch_done_count()
900
- rmsg = "{} {} tar file(s) left to build and {} tarred".format(tcnt, bmsg, bcnt)
901
- msg = "{}: Still running after {} of the {} PBS walltime, {}, with {}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), dmsg, rmsg)
963
+ rmsg = ''
964
+ if act&self.CBACTS: # tar queue depths say nothing about the other actions
965
+ bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
966
+ rmsg = ", with {} {} tar file(s) left to build and {} tarred".format(
967
+ self.batch_tar_count('N'), bmsg, self.batch_tar_count('T'))
968
+ msg = "{}: Still running after {} of the {} PBS walltime, {}{}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), dmsg, rmsg)
902
969
  self.pglog(self.INDENT + msg, self.LOGACT)
903
970
  saved = {key : self.PGLOG[key] for key in ('EMLMSG', 'ERRMSG', 'ERRCNT', 'SUMMSG', 'PRGMSG')}
904
- self.set_email("{}: {} still in progress after {}, {}, with {}!".format(self.PGBACK['cmd'], amsg, etime, dmsg, rmsg), self.EMLTOP)
971
+ self.set_email("{}: {} still in progress after {}, {}{}!".format(self.PGBACK['cmd'], amsg, etime, dmsg, rmsg), self.EMLTOP)
905
972
  title = "dsquasar: {} in progress ({})".format(amsg, dmsg)
906
973
  if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
907
974
  self.report_dscheck_email(title, 1)
@@ -921,6 +988,26 @@ class DsQuasar(PgCMD, PgSplit):
921
988
  self.show_wait_message(i, "{}: wait child processes".format(self.PGSIG['DSTR']), self.LOGWRN, 1)
922
989
  i += 1
923
990
 
991
+ # wait for a free child process slot, checking the walltime deadline between polls.
992
+ # check_child(..., -1) does the waiting inside its own loop too, breaking only once fewer
993
+ # than MPROC children are left running, so a parent whose children each tar many GB sits
994
+ # there for as long as every slot stays busy - well past the cutoff - while the deadline
995
+ # check right below the call is never reached. That is the other half of the problem
996
+ # wait_all_children() fixed: that one guarded the terminal wait, this one guards the far
997
+ # more frequent wait for a slot. Polling with dowait 0 keeps check_child's own cadence and
998
+ # wait message, and returns the free slot count exactly as check_child(..., -1) does.
999
+ def wait_child_slot(self):
1000
+ if self.PGBACK['mproc'] < 2: return 0
1001
+ i = 0
1002
+ while True:
1003
+ self.check_batch_deadline()
1004
+ pcnt = self.check_child(None, 0, self.LOGWRN, 0)
1005
+ ccnt = self.PGSIG['MPROC'] - pcnt
1006
+ if ccnt > 0: return ccnt
1007
+ self.show_wait_message(i, "{}: wait {}/{} child processes".format(
1008
+ self.PGSIG['DSTR'], pcnt, self.PGSIG['MPROC']), self.LOGWRN, 1)
1009
+ i += 1
1010
+
924
1011
  # recompute the confirmed backup counts from RDADB after all child processes
925
1012
  # finished, so a multi-process summary reflects succeeded (not just started) work
926
1013
  def confirm_quasar_counts(self, qinfo, bids, dcnd):
@@ -955,7 +1042,7 @@ class DsQuasar(PgCMD, PgSplit):
955
1042
  # backup one Quasar Backup or Backup&Drdata from one or multiple inputs and,
956
1043
  # reset the quasar backup dict
957
1044
  def process_one_backup_file(self, qinfo, addback, keepid = False):
958
- ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
1045
+ ccnt = self.wait_child_slot()
959
1046
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
960
1047
  self.check_batch_deadline()
961
1048
  dsids = qinfo['dsids']
@@ -1035,7 +1122,7 @@ class DsQuasar(PgCMD, PgSplit):
1035
1122
  # Transfer multiple tarfiles to Quasar Backup or Backup&Drdata, and
1036
1123
  # reset the quasar backup dict
1037
1124
  def transfer_quasar_tarfiles(self, qinfo):
1038
- ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
1125
+ ccnt = self.wait_child_slot()
1039
1126
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
1040
1127
  self.check_batch_deadline()
1041
1128
  # prepare for backup one tar file
@@ -1623,6 +1710,7 @@ class DsQuasar(PgCMD, PgSplit):
1623
1710
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsid' : None, 'size' : 0, 'bfile' : None,
1624
1711
  'bqfiles' : {}, 'dqfiles' : {}, 'qdsids' : [], 'qcnt' : 0, 'ncnt' : 0}
1625
1712
  for bid in bfiles:
1713
+ self.check_batch_deadline()
1626
1714
  qinfo['bid'] = bid
1627
1715
  binfo = bfiles[bid]
1628
1716
  if qinfo['dsid'] and binfo['dsid'] != qinfo['dsid']:
@@ -1741,12 +1829,13 @@ class DsQuasar(PgCMD, PgSplit):
1741
1829
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsids' : [], 'fcnt' : 0, 'size' : 0,
1742
1830
  'bfile' : None, 'pfile' : None, 'qdsids' : [], 'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
1743
1831
  for bid in bfiles:
1832
+ self.check_batch_deadline()
1744
1833
  qinfo['bid'] = bid
1745
1834
  binfo = bfiles[bid]
1746
1835
  for bkey in binfo: qinfo[bkey] = binfo[bkey]
1747
1836
  self.process_one_quasar_pathfile(qinfo)
1748
1837
  if self.PGBACK['mproc'] > 1:
1749
- self.check_child(None, 0, self.LOGWRN, 1) # wait all child processes done
1838
+ self.wait_all_children() # wait all child processes done
1750
1839
  self.confirm_quasar_counts(qinfo, list(bfiles), "bfile LIKE 'G%/%.tar'") # recount confirmed renames from RDADB
1751
1840
  qcnt = qinfo['qcnt']
1752
1841
  if qcnt > 0:
@@ -1761,7 +1850,7 @@ class DsQuasar(PgCMD, PgSplit):
1761
1850
  # backup one Quasar Backup or Backup&Drdata from one or multiple inputs and,
1762
1851
  # reset the quasar backup dict
1763
1852
  def process_one_quasar_pathfile(self, qinfo):
1764
- ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
1853
+ ccnt = self.wait_child_slot()
1765
1854
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
1766
1855
  # prepare for backup one tar file
1767
1856
  dsids = qinfo['dsids']
@@ -1883,6 +1972,7 @@ class DsQuasar(PgCMD, PgSplit):
1883
1972
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsids' : [], 'fcnt' : 0, 'size' : 0,
1884
1973
  'qdsids' : [], 'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
1885
1974
  for bid in bfiles:
1975
+ self.check_batch_deadline()
1886
1976
  qinfo['bid'] = bid
1887
1977
  binfo = bfiles[bid]
1888
1978
  for bkey in binfo: qinfo[bkey] = binfo[bkey]
@@ -1981,6 +2071,7 @@ class DsQuasar(PgCMD, PgSplit):
1981
2071
  qinfo = {'dsids' : [], 'fcnt' : 0, 'mcnt' : 0, 'fsize' : 0, 'msize' : 0}
1982
2072
  qcnt = 0
1983
2073
  for bid in bfiles:
2074
+ self.check_batch_deadline()
1984
2075
  qcnt += self.process_one_quasar_mcsfile(bid, bfiles[bid], qinfo)
1985
2076
  if qcnt > 0:
1986
2077
  s = 's' if qcnt > 1 else ''
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.19
3
+ Version: 3.0.20
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar