rda-python-dsquasar 3.0.18__tar.gz → 3.0.20__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (20) hide show
  1. {rda_python_dsquasar-3.0.18/src/rda_python_dsquasar.egg-info → rda_python_dsquasar-3.0.20}/PKG-INFO +1 -1
  2. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/pyproject.toml +1 -1
  3. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/dsquasar.py +138 -44
  4. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20/src/rda_python_dsquasar.egg-info}/PKG-INFO +1 -1
  5. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/LICENSE +0 -0
  6. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/MANIFEST.in +0 -0
  7. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/README.md +0 -0
  8. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/setup.cfg +0 -0
  9. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/__init__.py +0 -0
  10. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/ds_quasar.py +0 -0
  11. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/dsquasar.usg +0 -0
  12. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/dstacc.py +0 -0
  13. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/taccrec.py +0 -0
  14. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar/tacctar.py +0 -0
  15. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/SOURCES.txt +0 -0
  16. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/dependency_links.txt +0 -0
  17. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/entry_points.txt +0 -0
  18. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/requires.txt +0 -0
  19. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/src/rda_python_dsquasar.egg-info/top_level.txt +0 -0
  20. {rda_python_dsquasar-3.0.18 → rda_python_dsquasar-3.0.20}/test/test_dsquasar.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.18
3
+ Version: 3.0.20
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "rda_python_dsquasar"
7
- version = "3.0.18"
7
+ version = "3.0.20"
8
8
  authors = [
9
9
  { name="Zaihua Ji", email="zji@ucar.edu" },
10
10
  ]
@@ -23,6 +23,7 @@ import os
23
23
  import re
24
24
  import sys
25
25
  import time
26
+ import signal
26
27
  from os import path as op
27
28
  from time import time as tm
28
29
  from rda_python_common.pg_cmd import PgCMD
@@ -44,6 +45,7 @@ class DsQuasar(PgCMD, PgSplit):
44
45
  MCSACT = 128 # hidden action to add MD5 checksum string to note fields of bfile records
45
46
  DSCNT = 65
46
47
  DSTEP = 1000
48
+ DSLIST = 40 # name the datasets in a report line while the list is still readable
47
49
 
48
50
  def __init__(self):
49
51
  super().__init__() # initialize parent class
@@ -95,6 +97,7 @@ class DsQuasar(PgCMD, PgSplit):
95
97
  self.MPBLIMIT = 5000 # tar file count per batch process for uploading
96
98
  self.MPBMAX = 4 # maximum number of batch processes for uploading
97
99
  self.ONEHOUR = 3600 # seconds; headroom kept to report before the walltime
100
+ self.ERETRY = 600 # seconds to wait before retrying a failed progress report
98
101
  self.WALLTIME = 24*3600 # PBS walltime assumed, refreshed from PBS at start of a run
99
102
  self.MAXRUNTIME = self.WALLTIME - self.ONEHOUR # cutoff to cache a progress report
100
103
  # a repeat submit normally blocks as a duplicate while the batch job is running. if
@@ -123,6 +126,7 @@ class DsQuasar(PgCMD, PgSplit):
123
126
  'doemail' : 0,
124
127
  'starttime' : 0, # wall-clock start of the run, for the PBS walltime guard
125
128
  'einfo' : 0, # set once a progress report is cached into dscheck.einfo
129
+ 'eretry' : 0, # earliest retry time after a progress report failed to cache
126
130
  'maxworkers' : self.MAXWORKERS, # -W, maximum concurrent workers per command
127
131
  'worker' : 1, # -w, this run's worker slot; >1 for an added extra worker
128
132
  'cmd' : None
@@ -199,6 +203,7 @@ class DsQuasar(PgCMD, PgSplit):
199
203
  self.cmdlog(self.PGBACK['cmd'])
200
204
  self.PGBACK['starttime'] = tm()
201
205
  self.set_walltime_deadline()
206
+ self.catch_batch_termination()
202
207
  if self.sopts['u']:
203
208
  if self.sopts['a']: self.pglog("-u: Dataset IDs must be provided to Unlock datasets", self.LOGWRN)
204
209
  self.unlock_datasets()
@@ -237,7 +242,7 @@ class DsQuasar(PgCMD, PgSplit):
237
242
  self.PGLOG['ERRCNT'] or self.PGBACK['einfo']):
238
243
  self.PGBACK['doemail'] = 0
239
244
  if self.PGBACK['doemail']:
240
- amsg = self.ACTMSG[self.PGBACK['action']]
245
+ amsg = self.action_message()
241
246
  bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
242
247
  dcnt = len(self.dsids)
243
248
  dmsg = self.dsids[0] if dcnt == 1 else "{} datasets".format(dcnt if dcnt > 1 else 'All')
@@ -790,7 +795,7 @@ class DsQuasar(PgCMD, PgSplit):
790
795
  # a second report, drop the progress report parked in dscheck.einfo by the parent, and
791
796
  # mark the shared check record failed while the parent is still working
792
797
  if self.PGSIG['PPID'] > 1: sys.exit(1)
793
- if self.PGBACK['mproc'] > 1: self.check_child(None, 0, self.LOGWRN, 1)
798
+ self.wait_all_children()
794
799
  if qinfo:
795
800
  if 'dslocks' in qinfo and qinfo['dslocks']:
796
801
  for dsid in qinfo['dslocks']: self.lock_dataset(dsid, 0, self.LGEREX)
@@ -798,7 +803,7 @@ class DsQuasar(PgCMD, PgSplit):
798
803
  dcnt = len(qinfo['qdsids'])
799
804
  fcnt = qinfo['qfcnt']
800
805
  ssize = self.format_float_value(qinfo['qsize'])
801
- amsg = self.ACTMSG[self.PGBACK['action']]
806
+ amsg = self.action_message()
802
807
  bmsg = self.BACKMSG[qinfo['backflag']]
803
808
  dmsg = qinfo['qdsids'][0] if dcnt == 1 else "{} datasets".format(dcnt)
804
809
  msg = "Quit {}: {} {} files for {}({}) files of {}".format(amsg, qcnt, bmsg, fcnt, ssize, dmsg)
@@ -814,6 +819,13 @@ class DsQuasar(PgCMD, PgSplit):
814
819
  if self.PGBACK['pstep']: self.record_dscheck_status("F")
815
820
  self.pgexit(0)
816
821
 
822
+ # ACTMSG carries no wording for the hidden actions (-A 32/64), so name those by number
823
+ # rather than raise a KeyError: a report that cannot be worded is still a report that
824
+ # has to go out, and crashing while building it loses the run as silently as a kill does
825
+ def action_message(self, act = None):
826
+ if act is None: act = self.PGBACK['action']
827
+ return self.ACTMSG[act] if act in self.ACTMSG else "Action {}".format(act)
828
+
817
829
  # a batch run normally gets the full 24 hours asked for at submit time, but the dscheck
818
830
  # daemon caps that to the maximum of the PBS queue it lands in - only 6 hours for the
819
831
  # default queue - and to any planned system down. read the granted walltime back from
@@ -829,13 +841,54 @@ class DsQuasar(PgCMD, PgSplit):
829
841
  self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME)), self.LOGWRN)
830
842
  return
831
843
  wtime = 3600*int(ms.group(1)) + 60*int(ms.group(2)) + (int(ms.group(3)) if ms.group(3) else 0)
832
- if wtime <= self.ONEHOUR: return # too short to keep the reporting headroom
844
+ if wtime <= self.ONEHOUR: # too short to keep the reporting headroom
845
+ self.pglog("PBS Job {} walltime {} is too short to report progress ahead of; assume {}".format(
846
+ self.PGLOG['CURBID'], self.seconds_to_string_time(wtime),
847
+ self.seconds_to_string_time(self.WALLTIME)), self.LOGWRN)
848
+ return
833
849
  self.WALLTIME = wtime
834
850
  self.MAXRUNTIME = wtime - self.ONEHOUR
835
851
  self.pglog("PBS Job {} walltime {}: report progress after {}".format(
836
852
  self.PGLOG['CURBID'], self.seconds_to_string_time(self.WALLTIME),
837
853
  self.seconds_to_string_time(self.MAXRUNTIME)), self.LOGWRN)
838
854
 
855
+ # PBS kills a job that runs out of walltime with SIGTERM first and SIGKILL a few seconds
856
+ # later (the MoM's kill_delay, 10 seconds by default), and the same pair is what qdel
857
+ # sends. SIGTERM is the only warning there is, and nothing in the common library traps
858
+ # it, so the run used to die on the spot with whatever it had to say unsaid. catching it
859
+ # turns the last seconds into a report. only a batch run arms this: on the command line
860
+ # SIGTERM must keep killing the process the way the user expects.
861
+ def catch_batch_termination(self):
862
+ if self.PGLOG['CURBID'] < 1: return
863
+ signal.signal(signal.SIGTERM, self.batch_term_handler)
864
+
865
+ # park a progress report and then die of the signal that was sent. this runs on borrowed
866
+ # time, so it does only what it must: no tar queue counts (two database queries we may
867
+ # not get to finish - the done count comes from memory), no dataset unlocking (the
868
+ # dscheck daemon already cleans up after a dead pid). the default handler is restored
869
+ # first so that a second signal, or the SIGKILL that follows, ends the run outright
870
+ # instead of re-entering here should the report hang.
871
+ # the report may have to be written from inside an interrupted database call; that is
872
+ # why it goes through report_dscheck_email, whose cache_customized_email falls back to
873
+ # sending the mail directly when the UPDATE fails.
874
+ def batch_term_handler(self, signum, frame):
875
+ signal.signal(signum, signal.SIG_DFL)
876
+ # a forked child shares the parent's dscheck record: it must not report for the run
877
+ if self.PGSIG['PPID'] > 1: os._exit(1)
878
+ if not self.PGBACK['einfo'] and self.PGBACK['doemail'] and self.PGLOG['DSCHECK']:
879
+ etime = self.seconds_to_string_time(int(tm() - self.PGBACK['starttime']))
880
+ amsg = self.action_message()
881
+ dmsg = self.batch_done_count()
882
+ wmsg = self.seconds_to_string_time(self.WALLTIME)
883
+ self.pglog(self.INDENT + "{}: Terminated by signal {} after {} of the {} PBS walltime, {}".format(
884
+ amsg, signum, etime, wmsg, dmsg), self.LOGACT)
885
+ self.set_email("{}: {} terminated after {} of the {} PBS walltime, {}!".format(
886
+ self.PGBACK['cmd'], amsg, etime, wmsg, dmsg), self.EMLTOP)
887
+ title = "dsquasar: {} terminated ({})".format(amsg, dmsg)
888
+ if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
889
+ self.report_dscheck_email(title, 1)
890
+ os.kill(os.getpid(), signum)
891
+
839
892
  # email a report for a run under dscheck control. the progress report is parked in
840
893
  # dscheck.einfo and the final report is sent right away. the run holds the check lock
841
894
  # while it works, and the dscheck daemon only mails unlocked records, so a parked
@@ -856,8 +909,15 @@ class DsQuasar(PgCMD, PgSplit):
856
909
  ebuf = "From: {}\nTo: {}\n".format(sender, receiver)
857
910
  if self.PGLOG['CCDADDR']: ebuf += "Cc: {}\n".format(self.PGLOG['CCDADDR'])
858
911
  ebuf += "Subject: {}!\n\n{}\n".format(title, msg)
859
- self.PGBACK['einfo'] = 1 # report the progress only once per run
860
- return self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
912
+ # only a report that is actually parked (or sent directly by the fallback) counts
913
+ # as reported. marking it reported before the attempt turned a failed cache into a
914
+ # silent loss: nothing was parked, and the flag stopped every later attempt too
915
+ estat = self.cache_customized_email("dscheck", "einfo", cnd, ebuf, self.LOGWRN)
916
+ if estat:
917
+ self.PGBACK['einfo'] = 1 # report the progress only once per run
918
+ else:
919
+ self.PGBACK['eretry'] = tm() + self.ERETRY
920
+ return estat
861
921
  estat = self.build_customized_email("dscheck", "einfo", cnd, title, self.LOGWRN)
862
922
  # a failed send has already replaced the parked report via cache_customized_email()
863
923
  if estat == self.SUCCESS and self.PGBACK['einfo']:
@@ -867,40 +927,48 @@ class DsQuasar(PgCMD, PgSplit):
867
927
  self.pglog("Cannot clean the progress report of {}; it is emailed again".format(cnd), self.LOGWRN)
868
928
  return estat
869
929
 
870
- # guard the long PBS batch jobs (-A 2/3/4/6/7) against the walltime: once past MAXRUNTIME
871
- # the run keeps going, but a progress report is parked in dscheck.einfo so that
872
- # something is reported even if PBS kills the job at the walltime. the final report
873
- # follows and drops the parked one if the run does finish in time. the live email
874
- # buffers are saved and put back, so the final report still carries everything logged
875
- # before the cutoff.
876
- # both queue depths are reported rather than the current phase's, because -A 3 tars each
877
- # input file right after creating it, so its status 'N' count stays near zero while that
878
- # is the busy phase. they are worded as STATES ('left to build', 'tarred') and not as
879
- # work this run will do: only -A 4 and -A 6 transfer, so calling the status 'T' count
880
- # 'to transfer' in an -A 2 or -A 3 report claims work that run never performs.
881
- # a no-op for command-line runs, other actions, before the cutoff, or once a progress
882
- # report is cached.
930
+ # guard a long PBS batch job against the walltime: once past MAXRUNTIME the run keeps
931
+ # going, but a progress report is parked in dscheck.einfo so that something is reported
932
+ # even if PBS kills the job at the walltime. the final report follows and drops the
933
+ # parked one if the run does finish in time. the live email buffers are saved and put
934
+ # back, so the final report still carries everything logged before the cutoff.
935
+ # PBS sends no catchable warning before the kill - SIGTERM is not trapped and SIGKILL
936
+ # cannot be - so this parked report is the only thing standing between a killed job and
937
+ # a run that is never heard from. every action submitted to PBS is guarded, not just the
938
+ # tar and transfer ones: -A 16 and the hidden -A 128 are submitted with the same 24 hour
939
+ # walltime and used to be excluded, so they died silently.
940
+ # the tar queue depths are only reported for the actions they describe. both depths are
941
+ # reported rather than the current phase's, because -A 3 tars each input file right after
942
+ # creating it, so its status 'N' count stays near zero while that is the busy phase. they
943
+ # are worded as STATES ('left to build', 'tarred') and not as work this run will do: only
944
+ # -A 4 and -A 6 transfer, so calling the status 'T' count 'to transfer' in an -A 2 or
945
+ # -A 3 report claims work that run never performs.
946
+ # a no-op for command-line runs, before the cutoff, or once a progress report is cached.
883
947
  # called from the top of every per-item loop, not only where a tar is dispatched: a long
884
948
  # stretch of files that are skipped (already backed up, or not enough accumulated size to
885
949
  # tar yet) would otherwise walk past the cutoff without ever reaching the check, which is
886
950
  # how a 23h run reported nothing.
887
951
  def check_batch_deadline(self):
888
- act = self.PGBACK['action']
889
952
  if self.PGBACK['einfo'] or not (self.PGBACK['doemail'] and self.PGLOG['DSCHECK']): return
890
- if self.PGLOG['CURBID'] < 1 or act not in (self.TARACT, self.CTACTS, self.BCKACT, self.TBACTS, self.CBACTS): return
953
+ if self.PGLOG['CURBID'] < 1: return # only a PBS batch run has a walltime to beat
891
954
  elapsed = tm() - self.PGBACK['starttime']
892
955
  if elapsed < self.MAXRUNTIME: return
893
- tcnt = self.batch_tar_count('N')
894
- bcnt = self.batch_tar_count('T')
956
+ # a report that could not be parked is retried, but not once per file: the counts
957
+ # below are database queries and the cutoff leaves a whole hour of them otherwise
958
+ if self.PGBACK['eretry'] and tm() < self.PGBACK['eretry']: return
959
+ act = self.PGBACK['action']
895
960
  etime = self.seconds_to_string_time(int(elapsed))
896
- amsg = self.ACTMSG[act]
897
- bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
961
+ amsg = self.action_message(act)
898
962
  dmsg = self.batch_done_count()
899
- rmsg = "{} {} tar file(s) left to build and {} tarred".format(tcnt, bmsg, bcnt)
900
- msg = "{}: Still running after {} of the {} PBS walltime, {}, with {}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), dmsg, rmsg)
963
+ rmsg = ''
964
+ if act&self.CBACTS: # tar queue depths say nothing about the other actions
965
+ bmsg = self.BACKMSG[self.PGBACK['backflag']] if self.PGBACK['backflag'] else 'backup'
966
+ rmsg = ", with {} {} tar file(s) left to build and {} tarred".format(
967
+ self.batch_tar_count('N'), bmsg, self.batch_tar_count('T'))
968
+ msg = "{}: Still running after {} of the {} PBS walltime, {}{}".format(amsg, etime, self.seconds_to_string_time(self.WALLTIME), dmsg, rmsg)
901
969
  self.pglog(self.INDENT + msg, self.LOGACT)
902
970
  saved = {key : self.PGLOG[key] for key in ('EMLMSG', 'ERRMSG', 'ERRCNT', 'SUMMSG', 'PRGMSG')}
903
- self.set_email("{}: {} still in progress after {}, {}, with {}!".format(self.PGBACK['cmd'], amsg, etime, dmsg, rmsg), self.EMLTOP)
971
+ self.set_email("{}: {} still in progress after {}, {}{}!".format(self.PGBACK['cmd'], amsg, etime, dmsg, rmsg), self.EMLTOP)
904
972
  title = "dsquasar: {} in progress ({})".format(amsg, dmsg)
905
973
  if self.PGBACK['errcnt']: title += " Error({})".format(self.PGBACK['errcnt'])
906
974
  self.report_dscheck_email(title, 1)
@@ -920,6 +988,26 @@ class DsQuasar(PgCMD, PgSplit):
920
988
  self.show_wait_message(i, "{}: wait child processes".format(self.PGSIG['DSTR']), self.LOGWRN, 1)
921
989
  i += 1
922
990
 
991
+ # wait for a free child process slot, checking the walltime deadline between polls.
992
+ # check_child(..., -1) does the waiting inside its own loop too, breaking only once fewer
993
+ # than MPROC children are left running, so a parent whose children each tar many GB sits
994
+ # there for as long as every slot stays busy - well past the cutoff - while the deadline
995
+ # check right below the call is never reached. That is the other half of the problem
996
+ # wait_all_children() fixed: that one guarded the terminal wait, this one guards the far
997
+ # more frequent wait for a slot. Polling with dowait 0 keeps check_child's own cadence and
998
+ # wait message, and returns the free slot count exactly as check_child(..., -1) does.
999
+ def wait_child_slot(self):
1000
+ if self.PGBACK['mproc'] < 2: return 0
1001
+ i = 0
1002
+ while True:
1003
+ self.check_batch_deadline()
1004
+ pcnt = self.check_child(None, 0, self.LOGWRN, 0)
1005
+ ccnt = self.PGSIG['MPROC'] - pcnt
1006
+ if ccnt > 0: return ccnt
1007
+ self.show_wait_message(i, "{}: wait {}/{} child processes".format(
1008
+ self.PGSIG['DSTR'], pcnt, self.PGSIG['MPROC']), self.LOGWRN, 1)
1009
+ i += 1
1010
+
923
1011
  # recompute the confirmed backup counts from RDADB after all child processes
924
1012
  # finished, so a multi-process summary reflects succeeded (not just started) work
925
1013
  def confirm_quasar_counts(self, qinfo, bids, dcnd):
@@ -954,7 +1042,7 @@ class DsQuasar(PgCMD, PgSplit):
954
1042
  # backup one Quasar Backup or Backup&Drdata from one or multiple inputs and,
955
1043
  # reset the quasar backup dict
956
1044
  def process_one_backup_file(self, qinfo, addback, keepid = False):
957
- ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
1045
+ ccnt = self.wait_child_slot()
958
1046
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
959
1047
  self.check_batch_deadline()
960
1048
  dsids = qinfo['dsids']
@@ -1034,7 +1122,7 @@ class DsQuasar(PgCMD, PgSplit):
1034
1122
  # Transfer multiple tarfiles to Quasar Backup or Backup&Drdata, and
1035
1123
  # reset the quasar backup dict
1036
1124
  def transfer_quasar_tarfiles(self, qinfo):
1037
- ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
1125
+ ccnt = self.wait_child_slot()
1038
1126
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
1039
1127
  self.check_batch_deadline()
1040
1128
  # prepare for backup one tar file
@@ -1622,6 +1710,7 @@ class DsQuasar(PgCMD, PgSplit):
1622
1710
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsid' : None, 'size' : 0, 'bfile' : None,
1623
1711
  'bqfiles' : {}, 'dqfiles' : {}, 'qdsids' : [], 'qcnt' : 0, 'ncnt' : 0}
1624
1712
  for bid in bfiles:
1713
+ self.check_batch_deadline()
1625
1714
  qinfo['bid'] = bid
1626
1715
  binfo = bfiles[bid]
1627
1716
  if qinfo['dsid'] and binfo['dsid'] != qinfo['dsid']:
@@ -1740,12 +1829,13 @@ class DsQuasar(PgCMD, PgSplit):
1740
1829
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsids' : [], 'fcnt' : 0, 'size' : 0,
1741
1830
  'bfile' : None, 'pfile' : None, 'qdsids' : [], 'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
1742
1831
  for bid in bfiles:
1832
+ self.check_batch_deadline()
1743
1833
  qinfo['bid'] = bid
1744
1834
  binfo = bfiles[bid]
1745
1835
  for bkey in binfo: qinfo[bkey] = binfo[bkey]
1746
1836
  self.process_one_quasar_pathfile(qinfo)
1747
1837
  if self.PGBACK['mproc'] > 1:
1748
- self.check_child(None, 0, self.LOGWRN, 1) # wait all child processes done
1838
+ self.wait_all_children() # wait all child processes done
1749
1839
  self.confirm_quasar_counts(qinfo, list(bfiles), "bfile LIKE 'G%/%.tar'") # recount confirmed renames from RDADB
1750
1840
  qcnt = qinfo['qcnt']
1751
1841
  if qcnt > 0:
@@ -1760,7 +1850,7 @@ class DsQuasar(PgCMD, PgSplit):
1760
1850
  # backup one Quasar Backup or Backup&Drdata from one or multiple inputs and,
1761
1851
  # reset the quasar backup dict
1762
1852
  def process_one_quasar_pathfile(self, qinfo):
1763
- ccnt = self.check_child(None, 0, self.LOGWRN, -1) if self.PGBACK['mproc'] > 1 else 0
1853
+ ccnt = self.wait_child_slot()
1764
1854
  if self.PGSIG['QUIT']: self.quit_dsquasar(qinfo)
1765
1855
  # prepare for backup one tar file
1766
1856
  dsids = qinfo['dsids']
@@ -1882,6 +1972,7 @@ class DsQuasar(PgCMD, PgSplit):
1882
1972
  qinfo = {'backflag' : backflag, 'bid' : 0, 'dsids' : [], 'fcnt' : 0, 'size' : 0,
1883
1973
  'qdsids' : [], 'qfcnt' : 0, 'qsize' : 0, 'qcnt' : 0}
1884
1974
  for bid in bfiles:
1975
+ self.check_batch_deadline()
1885
1976
  qinfo['bid'] = bid
1886
1977
  binfo = bfiles[bid]
1887
1978
  for bkey in binfo: qinfo[bkey] = binfo[bkey]
@@ -1980,6 +2071,7 @@ class DsQuasar(PgCMD, PgSplit):
1980
2071
  qinfo = {'dsids' : [], 'fcnt' : 0, 'mcnt' : 0, 'fsize' : 0, 'msize' : 0}
1981
2072
  qcnt = 0
1982
2073
  for bid in bfiles:
2074
+ self.check_batch_deadline()
1983
2075
  qcnt += self.process_one_quasar_mcsfile(bid, bfiles[bid], qinfo)
1984
2076
  if qcnt > 0:
1985
2077
  s = 's' if qcnt > 1 else ''
@@ -2172,46 +2264,48 @@ class DsQuasar(PgCMD, PgSplit):
2172
2264
  self.set_dsquasar_progress(cnt, size)
2173
2265
  cnt = size = 0
2174
2266
  if cnt: self.set_dsquasar_progress(cnt, size)
2175
- dscnt = len(qinfo['dsids'])
2176
2267
  ssize = self.format_float_value(qinfo['size'])
2177
2268
  bmsg = self.BACKMSG[backflag]
2178
- dmsg = qinfo['dsids'][0] if dscnt == 1 else "{} datasets".format(dscnt)
2269
+ dmsg = self.dataset_message(qinfo['dsids'])
2179
2270
  msg = "{} ({}) {} files for {} GDEX files of {}".format(bcnt, ssize, bmsg, qinfo['fcnt'], dmsg)
2180
2271
  self.pglog(self.INDENT + msg, self.LOGACT)
2181
2272
  indent = self.INDENT + self.INDENT
2182
2273
  if qinfo['ncnt'] > 0:
2183
- dscnt = len(qinfo['ndsids'])
2184
- dmsg = qinfo['ndsids'][0] if dscnt == 1 else "{} datasets".format(dscnt)
2274
+ dmsg = self.dataset_message(qinfo['ndsids'])
2185
2275
  msg = "{} {} files Missing Note fields of {}".format(bcnt, bmsg, dmsg)
2186
2276
  self.pglog(indent + msg, self.LOGACT)
2187
2277
  msg = "{} GDEX files Changed after {}".format(pcnt, bmsg)
2188
2278
  self.pglog(indent + msg, self.LOGACT)
2189
2279
  if pcnt == 0: return
2190
2280
  if qinfo['ccnt'] > 0:
2191
- dscnt = len(qinfo['cdsids'])
2192
- dmsg = qinfo['cdsids'][0] if dscnt == 1 else "{} datasets".format(dscnt)
2281
+ dmsg = self.dataset_message(qinfo['cdsids'])
2193
2282
  ssize = self.format_float_value(qinfo['csize'])
2194
2283
  msg = "{} ({}) GDEX file sizes Changed for {}".format(qinfo['ccnt'], ssize, dmsg)
2195
2284
  self.pglog(indent + msg, self.LOGACT)
2196
2285
  if qinfo['ucnt'] > 0:
2197
- dscnt = len(qinfo['udsids'])
2198
- dmsg = qinfo['udsids'][0] if dscnt == 1 else "{} datasets".format(dscnt)
2286
+ dmsg = self.dataset_message(qinfo['udsids'])
2199
2287
  ssize = self.format_float_value(qinfo['usize'])
2200
2288
  msg = "{} ({}) GDEX files Updated & Re-done {} for {}".format(qinfo['ucnt'], ssize, bmsg, dmsg)
2201
2289
  self.pglog(indent + msg, self.LOGACT)
2202
2290
  if qinfo['dcnt'] > 0:
2203
- dscnt = len(qinfo['ddsids'])
2204
- dmsg = qinfo['ddsids'][0] if dscnt == 1 else "{} datasets".format(dscnt)
2291
+ dmsg = self.dataset_message(qinfo['ddsids'])
2205
2292
  ssize = self.format_float_value(qinfo['dsize'])
2206
2293
  msg = "{} ({}) GDEX files Got Deleted for {}".format(qinfo['dcnt'], ssize, dmsg)
2207
2294
  self.pglog(indent + msg, self.LOGACT)
2208
2295
  if qinfo['mcnt'] > 0:
2209
- dscnt = len(qinfo['mdsids'])
2210
- dmsg = qinfo['mdsids'][0] if dscnt == 1 else "{} datasets".format(dscnt)
2296
+ dmsg = self.dataset_message(qinfo['mdsids'])
2211
2297
  ssize = self.format_float_value(qinfo['msize'])
2212
2298
  msg = "{} ({}) GDEX files Moved for {}".format(qinfo['mcnt'], ssize, dmsg)
2213
2299
  self.pglog(indent + msg, self.LOGACT)
2214
2300
 
2301
+ # name the datasets a report line refers to, so that a short list can be acted on
2302
+ # instead of only counted; fall back to the count alone once the list gets long
2303
+ def dataset_message(self, dsids):
2304
+ dscnt = len(dsids)
2305
+ if dscnt == 1: return dsids[0]
2306
+ if dscnt > self.DSLIST: return "{} datasets".format(dscnt)
2307
+ return "{} datasets ({})".format(dscnt, ', '.join(sorted(dsids)))
2308
+
2215
2309
  # count cache the Web/Saved files in type A bfiles (archived) already
2216
2310
  def count_changed_files(self, bid, binfo, qinfo):
2217
2311
  qinfo['fcnt'] += binfo['fcnt']
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_dsquasar
3
- Version: 3.0.18
3
+ Version: 3.0.20
4
4
  Summary: RDA Python package to backup and recover RDA data archives to and from GLOBUS Quasar backup server
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-dsquasar