rda-python-miscs 3.0.3__py3-none-any.whl → 3.0.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,456 @@
1
+ #!/usr/bin/env python3
2
+ ##################################################################################
3
+ # Title: decsdata_restore
4
+ # Author: Zaihua Ji, zji@ucar.edu
5
+ # Date: 2026-08-31
6
+ # Purpose: restore decsdata datasets, or parts of them, out of the GLADE HSM
7
+ # cold storage; the opposite of decsdata_storage
8
+ # Github: https://github.com/NCAR/rda-python-miscs.git
9
+ ##################################################################################
10
+ import re
11
+ import os
12
+ import sys
13
+ from os import path as op
14
+ from rda_python_common.pg_file import PgFile
15
+
16
+ class DecsRestore(PgFile):
17
+ """Restore decsdata datasets out of the GLADE HSM cold storage.
18
+
19
+ 'glade_hsm recall' only submits a request that the HSM batch processes
20
+ fulfill later on, so restoring is done in three steps:
21
+
22
+ -x submit the recall requests for the given datasets/paths
23
+ -s check the recall status, repeat until nothing is left on tape
24
+ -r copy the recalled data back under the decsdata directory
25
+
26
+ Recalled files stay readable inside cold storage for 7 days only, after
27
+ which they are migrated onto tape again, so step -r must be done within
28
+ that window.
29
+ """
30
+
31
+ def __init__(self):
32
+ """Initialize DecsRestore with default option values and runtime state."""
33
+ super().__init__()
34
+ self.HSM = os.environ.get('GLADE_HSM', op.expanduser('~benkirk/glade_hsm'))
35
+ self.VALOPTS = 'Dwl' # single-value options
36
+ self.MULOPTS = 'd' # multi-value options
37
+ self.MODOPTS = 'hf' # mode options
38
+ self.ACTOPTS = 'xsr' # action options, one and only one is required
39
+ self.OPTS = {
40
+ 'D': None, # cold storage date, in YYYYMMDD or YYYY-MM-DD
41
+ 'w': None, # decsdata directory, defaults to PGLOG['DECSHOME']
42
+ 'l': None, # dataset list file
43
+ 'd': [], # dataset IDs, a sub-path may be appended to each one
44
+ 't': None, # target directory of -r, defaults to the decsdata directory
45
+ 'h': 0, # 1 to show help message
46
+ 'f': 0, # 1 to copy back while files are still on tape
47
+ }
48
+ self.ACTION = None # one of the ACTOPTS letters
49
+ self.SIZEUNITS = { # units reported by 'gladequota', in bytes
50
+ 'B': 1, 'KIB': 1024, 'MIB': 1024**2,
51
+ 'GIB': 1024**3, 'TIB': 1024**4, 'PIB': 1024**5,
52
+ }
53
+ self.RINFO = {
54
+ 'decsdir': None, # decsdata directory the data is restored into
55
+ 'roots': [], # cold storage directories to look the data up in
56
+ 'acnt': 0, # number of dataset paths acted on successfully
57
+ }
58
+
59
+ # function to read parameters
60
+ def read_parameters(self):
61
+ """Parse the command line into the OPTS option values and the action.
62
+
63
+ Single-value options -D, -w and -l take one value each, multi-value
64
+ option -d gathers every following dataset path, mode options -h and -f
65
+ are simple flags, and action options -x, -s and -r are mutually
66
+ exclusive; -r optionally takes a target directory. Exits with usage if
67
+ -h is given or no action is specified.
68
+ """
69
+ self.set_suid(self.PGLOG['EUID'])
70
+ self.set_help_path(__file__)
71
+ self.PGLOG['LOGFILE'] = "decsdata_restore.log" # set different log file
72
+ argv = sys.argv[1:]
73
+ self.cmdlog("decsdata_restore {}".format(' '.join(argv)))
74
+ option = None
75
+ for arg in argv:
76
+ ms = re.match(r'^-(\w+)$', arg)
77
+ if ms:
78
+ option = ms.group(1)
79
+ if option in self.ACTOPTS:
80
+ self.set_action(option)
81
+ if option != 'r': option = None # -r may be followed by a target directory
82
+ elif option in self.MODOPTS:
83
+ self.OPTS[option] = 1
84
+ option = None
85
+ elif option not in self.VALOPTS and option not in self.MULOPTS:
86
+ self.pglog(arg + ": Unknown Option", self.LGEREX)
87
+ continue
88
+ if not option: self.pglog(arg + ": Value provided without option", self.LGEREX)
89
+ if option in self.MULOPTS:
90
+ self.OPTS[option].append(arg) # gather all values until the next option
91
+ else:
92
+ if option == 'r': option = 't' # the value of -r is the target directory
93
+ self.OPTS[option] = arg
94
+ option = None
95
+ if self.OPTS['h'] or not self.ACTION: self.show_usage("decsdata_restore")
96
+
97
+ # remember the action option and reject a second one
98
+ def set_action(self, option):
99
+ """Record the single action option to perform.
100
+
101
+ Args:
102
+ option (str): One of the ACTOPTS letters.
103
+ """
104
+ if self.ACTION and self.ACTION != option:
105
+ self.pglog("-{}: Cannot combine with Action -{}".format(option, self.ACTION), self.LGEREX)
106
+ self.ACTION = option
107
+
108
+ # function to start actions
109
+ def start_actions(self):
110
+ """Validate the caller, resolve the cold storage paths, and act on each dataset path."""
111
+ self.dssdb_dbname()
112
+ self.validate_decs_group('decsdata_restore', self.PGLOG['CURUID'], 1)
113
+ self.set_restore_paths()
114
+ specs = self.get_dataset_list()
115
+ if not specs: self.pglog("No dataset found to restore", self.LGWNEX)
116
+ if self.ACTION == 'x': self.check_restore_space(specs)
117
+ for spec in specs:
118
+ self.restore_one_path(spec)
119
+ acts = {'x': 'requested Recall', 's': 'checked Status', 'r': 'copied back'}
120
+ s = ('s' if self.RINFO['acnt'] > 1 else '')
121
+ self.pglog("{} of {} Dataset Path{} {}".format(self.RINFO['acnt'], len(specs),
122
+ s, acts[self.ACTION]), self.LOGWRN)
123
+ self.cmdlog()
124
+
125
+ # resolve the decsdata directory and the cold storage directories
126
+ def set_restore_paths(self):
127
+ """Fill RINFO with the decsdata directory and the cold storage directories to search.
128
+
129
+ For a given -D date only '<decsdata>/cold_storage_<date>/COLD_STORAGE' is
130
+ searched. Without -D both '<decsdata>/COLD_STORAGE' and every
131
+ '<decsdata>/cold_storage_<YYYYMMDD>/COLD_STORAGE' are searched, the most
132
+ recent dated one first.
133
+ """
134
+ decsdir = self.OPTS['w'] if self.OPTS['w'] else self.PGLOG['DECSHOME']
135
+ if not self.check_local_file(decsdir, 0, self.LOGWRN):
136
+ self.pglog(decsdir + ": decsdata directory NOT exists", self.LGEREX)
137
+ self.RINFO['decsdir'] = decsdir
138
+ if not self.OPTS['t']: self.OPTS['t'] = decsdir
139
+ roots = []
140
+ if self.OPTS['D']:
141
+ date = re.sub('-', '', self.OPTS['D'])
142
+ if not re.match(r'^\d{8}$', date):
143
+ self.pglog(date + ": Invalid cold storage date, YYYYMMDD expected", self.LGEREX)
144
+ roots.append(self.join_paths(decsdir, "cold_storage_{}/COLD_STORAGE".format(date)))
145
+ else:
146
+ roots.append(self.join_paths(decsdir, "COLD_STORAGE"))
147
+ files = self.local_glob(self.join_paths(decsdir, "cold_storage_[0-9]*/COLD_STORAGE"), 0, self.LOGWRN)
148
+ for file in sorted(files, reverse = True):
149
+ if not files[file]['isfile']: roots.append(file)
150
+ for root in roots:
151
+ if self.check_local_file(root, 0, 0): self.RINFO['roots'].append(root)
152
+ if not self.RINFO['roots']:
153
+ self.pglog("{}: No cold storage directory found in {}".format(', '.join(roots), decsdir), self.LGEREX)
154
+ self.pglog("Cold storage searched: {}".format(', '.join(self.RINFO['roots'])), self.LOGWRN)
155
+
156
+ # gather the dataset paths to restore
157
+ def get_dataset_list(self):
158
+ """Return the list of dataset paths to restore.
159
+
160
+ Uses the -d values if given. Otherwise reads the -l list file, creating
161
+ it first from every dNNNNNN directory in the cold storage directories if
162
+ it does not exist yet.
163
+
164
+ Returns:
165
+ list[str]: Dataset IDs, each optionally followed by a sub-path.
166
+ """
167
+ if self.OPTS['d']:
168
+ self.pglog("Restore {} given Dataset Path(s)".format(len(self.OPTS['d'])), self.LOGWRN)
169
+ return self.OPTS['d']
170
+ lstfile = self.OPTS['l'] if self.OPTS['l'] else "dsids_{}.lst".format(re.sub('-', '', self.curdate()))
171
+ if not op.isfile(lstfile):
172
+ dsids = self.get_coldstorage_datasets()
173
+ with open(lstfile, 'w') as OUT:
174
+ for dsid in dsids: OUT.write(dsid + "\n")
175
+ self.pglog("{}: Generated with {} Dataset(s)".format(lstfile, len(dsids)), self.LOGWRN)
176
+ specs = []
177
+ with open(lstfile, 'r') as IN:
178
+ for line in IN:
179
+ line = line.strip()
180
+ if line: specs.append(line)
181
+ self.pglog("{}: Read {} Dataset Path(s) to restore".format(lstfile, len(specs)), self.LOGWRN)
182
+ return specs
183
+
184
+ # find all dNNNNNN dataset directories in the cold storage directories
185
+ def get_coldstorage_datasets(self):
186
+ """Return the sorted dataset IDs of every dNNNNNN directory in the cold storage directories.
187
+
188
+ Returns:
189
+ list[str]: Unique dataset IDs; plain files matching the pattern are skipped.
190
+ """
191
+ dsids = []
192
+ for root in self.RINFO['roots']:
193
+ files = self.local_glob(self.join_paths(root, "d" + "[0-9]"*6), 0, self.LOGWRN)
194
+ for file in files:
195
+ if files[file]['isfile']: continue
196
+ dsid = op.basename(file)
197
+ if dsid not in dsids: dsids.append(dsid)
198
+ return sorted(dsids)
199
+
200
+ # locate a dataset path in the cold storage directories, the first match wins
201
+ def find_cold_path(self, spec):
202
+ """Look a dataset path up in each cold storage directory.
203
+
204
+ Warns and names the ignored ones if the path is found in more than one
205
+ cold storage directory.
206
+
207
+ Args:
208
+ spec (str): Dataset ID, optionally followed by a sub-path.
209
+
210
+ Returns:
211
+ tuple: (path, info dict) of the first match, or (None, None).
212
+ """
213
+ hits = {}
214
+ for root in self.RINFO['roots']:
215
+ path = self.join_paths(root, spec)
216
+ info = self.check_local_file(path, 0, 0)
217
+ if info: hits[path] = info
218
+ if not hits: return (None, None)
219
+ paths = list(hits)
220
+ if len(paths) > 1:
221
+ self.pglog("{}: Found in {} cold storage directories, use {}".format(spec, len(paths), paths[0]), self.LOGWRN)
222
+ self.pglog("{}: Ignored {}".format(spec, ', '.join(paths[1:])), self.LOGWRN)
223
+ return (paths[0], hits[paths[0]])
224
+
225
+ # count the files of a cold storage path still on tape
226
+ def hsm_offline_count(self, path, isfile):
227
+ """Return the number of files under a cold storage path that are still on tape.
228
+
229
+ Args:
230
+ path (str): Cold storage path of a file or directory.
231
+ isfile (int): 1 if path is a regular file, 0 for a directory.
232
+
233
+ Returns:
234
+ int | None: Count of offline files, or None if it cannot be determined.
235
+ """
236
+ out = self.pgsystem("{} status {}".format(self.HSM, path), self.LOGWRN, 51)
237
+ if not out: return None
238
+ if isfile: return (1 if re.search(r'migrated', out) else 0)
239
+ cnts = re.findall(r'Offline:\s*([\d,]+)', out)
240
+ if not cnts: return None
241
+ return int(re.sub(',', '', cnts[-1]))
242
+
243
+ # build the sfile condition of one dataset path
244
+ def sfile_condition(self, spec):
245
+ """Turn a dataset path into a condition on table dssdb.sfile.
246
+
247
+ A saved file lives in '<decsdata>/<dsid>/<type>/<sfile>', so the first
248
+ component of the path is the dataset ID, the second one the saved file
249
+ type, and the rest the leading part of the sfile field.
250
+
251
+ Args:
252
+ spec (str): Dataset ID, optionally followed by a sub-path.
253
+
254
+ Returns:
255
+ str: The WHERE condition of the saved files under the path.
256
+ """
257
+ paths = spec.strip('/').split('/')
258
+ if not re.match(r'^[a-z]\d{6}$', paths[0]):
259
+ self.pglog(spec + ": Invalid dataset path, dNNNNNN expected", self.LGEREX)
260
+ cnd = "dsid = '{}'".format(paths[0])
261
+ if len(paths) > 1:
262
+ if not re.match(r'^\w$', paths[1]):
263
+ self.pglog(spec + ": Invalid saved file type, one word character expected", self.LGEREX)
264
+ cnd += " AND type = '{}'".format(paths[1])
265
+ if len(paths) > 2:
266
+ sfile = '/'.join(paths[2:])
267
+ if re.search(r"['\\]", sfile):
268
+ self.pglog(spec + ": Invalid saved file path", self.LGEREX)
269
+ # the path is either a saved file itself or the directory holding them
270
+ cnd += " AND (sfile = '{}' OR sfile LIKE '{}/%')".format(sfile, re.sub(r'([%_])', r'\\\1', sfile))
271
+ return cnd
272
+
273
+ # get the archived size of one dataset path
274
+ def restore_data_size(self, spec):
275
+ """Return the total size of the saved files under one dataset path.
276
+
277
+ Args:
278
+ spec (str): Dataset ID, optionally followed by a sub-path.
279
+
280
+ Returns:
281
+ int: Number of bytes recorded in dssdb.sfile; 0 if nothing is found.
282
+ """
283
+ pgrec = self.pgget('sfile', "sum(data_size) tsize, count(sid) fcnt",
284
+ self.sfile_condition(spec), self.LOGWRN)
285
+ if not pgrec or not pgrec['tsize']:
286
+ self.pglog(spec + ": No saved file found in RDADB", self.LOGWRN)
287
+ return 0
288
+ self.pglog("{}: {} in {} saved file(s)".format(spec,
289
+ self.format_float_value(pgrec['tsize']), pgrec['fcnt']), self.LOGWRN)
290
+ return int(pgrec['tsize'])
291
+
292
+ # get the GLADE space left for the decsdata directory
293
+ def decsdata_free_size(self):
294
+ """Return the GLADE space left on the quota holding the decsdata directory.
295
+
296
+ Parses the 'Used' and 'Quota' columns of 'gladequota' and picks the
297
+ longest reported path the decsdata directory falls under.
298
+
299
+ Returns:
300
+ int | None: Number of free bytes, or None if it cannot be determined.
301
+ """
302
+ cmd = self.get_local_command("gladequota", self.PGLOG['COMMONUSER'])
303
+ out = self.pgsystem(cmd, self.LOGWRN, 21) # 1+4+16, log the command and return stdout
304
+ if not out: return None
305
+ target = op.realpath(self.RINFO['decsdir'])
306
+ (fsize, fpath) = (None, None)
307
+ for line in out.split('\n'):
308
+ ms = re.match(r'^(/\S+)\s+([\d.]+)\s*(\w+)\s+([\d.]+)\s*(\w+)', line)
309
+ if not ms: continue # skips the header and the 'n/a' quota lines
310
+ path = ms.group(1)
311
+ if not (target == path or target.startswith(path + '/')): continue
312
+ if fpath and len(fpath) >= len(path): continue # keeps the closest path only
313
+ used = self.quota_size(ms.group(2), ms.group(3))
314
+ quota = self.quota_size(ms.group(4), ms.group(5))
315
+ if used is None or quota is None: continue
316
+ (fsize, fpath) = (max(quota - used, 0), path)
317
+ return fsize
318
+
319
+ # convert one 'gladequota' size into bytes
320
+ def quota_size(self, value, unit):
321
+ """Convert one size reported by 'gladequota' into bytes.
322
+
323
+ Args:
324
+ value (str): The numeric part of the size.
325
+ unit (str): The unit of the size, such as 'TiB'.
326
+
327
+ Returns:
328
+ int | None: Number of bytes, or None for an unknown unit.
329
+ """
330
+ unit = unit.upper()
331
+ if unit not in self.SIZEUNITS: return None
332
+ return int(float(value)*self.SIZEUNITS[unit])
333
+
334
+ # make sure the decsdata directory has room for the whole restore
335
+ def check_restore_space(self, specs):
336
+ """Stop the recall if the decsdata directory cannot hold the whole restore.
337
+
338
+ The size to restore is added up from table dssdb.sfile and compared to
339
+ the GLADE space left for the decsdata directory. Twice the size is
340
+ required, since the recall brings the data back on disk inside the cold
341
+ storage first and Action -r copies it back afterwards, so both copies
342
+ live under the decsdata quota at the same time. The check is skipped,
343
+ with a warning, if either size cannot be determined.
344
+
345
+ Args:
346
+ specs (list[str]): Dataset paths to recall.
347
+ """
348
+ tsize = 0
349
+ for spec in specs:
350
+ tsize += self.restore_data_size(spec)
351
+ if not tsize:
352
+ self.pglog("Unknown size to restore, Skip checking the decsdata space", self.LOGWRN)
353
+ return
354
+ fsize = self.decsdata_free_size()
355
+ if fsize is None:
356
+ self.pglog("{}: Cannot get the space left, Skip checking the decsdata space".format(self.RINFO['decsdir']), self.LOGWRN)
357
+ return
358
+ nsize = 2*tsize # room for the recalled copy and for the copy of Action -r
359
+ msg = "{}: Restore {}, needs {} of the {} left".format(self.RINFO['decsdir'],
360
+ self.format_float_value(tsize), self.format_float_value(nsize),
361
+ self.format_float_value(fsize))
362
+ if nsize > fsize:
363
+ self.pglog(msg + ", NOT enough space", self.LGEREX)
364
+ self.pglog(msg, self.LOGWRN)
365
+
366
+ # act on one dataset path in cold storage
367
+ def restore_one_path(self, spec):
368
+ """Perform the requested action on one dataset path in cold storage.
369
+
370
+ Args:
371
+ spec (str): Dataset ID, optionally followed by a sub-path.
372
+ """
373
+ (path, info) = self.find_cold_path(spec)
374
+ if not path:
375
+ self.pglog(spec + ": NOT found in cold storage", self.LOGERR)
376
+ return
377
+ if self.ACTION == 'x':
378
+ self.recall_cold_path(spec, path)
379
+ elif self.ACTION == 's':
380
+ self.status_cold_path(spec, path, info['isfile'])
381
+ else:
382
+ self.copy_cold_path(spec, path, info['isfile'])
383
+
384
+ # submit the recall request of one cold storage path
385
+ def recall_cold_path(self, spec, path):
386
+ """Submit the HSM recall request for one cold storage path.
387
+
388
+ Args:
389
+ spec (str): Dataset ID, optionally followed by a sub-path.
390
+ path (str): Cold storage path of the data.
391
+ """
392
+ if self.pgsystem("{} recall -f {}".format(self.HSM, path), self.LOGWRN, 7):
393
+ self.RINFO['acnt'] += 1
394
+ self.pglog("{}: Recall requested, check the progress via -s".format(spec), self.LOGWRN)
395
+ else:
396
+ self.pglog("{}: Error request Recall of {}".format(spec, path), self.LOGERR)
397
+
398
+ # report the recall status of one cold storage path
399
+ def status_cold_path(self, spec, path, isfile):
400
+ """Report the HSM and recall status of one cold storage path.
401
+
402
+ Args:
403
+ spec (str): Dataset ID, optionally followed by a sub-path.
404
+ path (str): Cold storage path of the data.
405
+ isfile (int): 1 if path is a regular file, 0 for a directory.
406
+ """
407
+ offline = self.hsm_offline_count(path, isfile)
408
+ if offline is None:
409
+ self.pglog("{}: Cannot get the offline file count of {}".format(spec, path), self.LOGERR)
410
+ return
411
+ self.RINFO['acnt'] += 1
412
+ if offline > 0:
413
+ s = ('s' if offline > 1 else '')
414
+ self.pglog("{}: Recall PENDING, {} File{} still on tape".format(spec, offline, s), self.LOGWRN)
415
+ else:
416
+ self.pglog("{}: Recall COMPLETE, ready to copy back via -r".format(spec), self.LOGWRN)
417
+
418
+ # copy one recalled cold storage path back into the decsdata directory
419
+ def copy_cold_path(self, spec, path, isfile):
420
+ """Copy one recalled cold storage path back to its target directory.
421
+
422
+ Nothing is copied while files are still on tape unless -f is given.
423
+
424
+ Args:
425
+ spec (str): Dataset ID, optionally followed by a sub-path.
426
+ path (str): Cold storage path of the data.
427
+ isfile (int): 1 if path is a regular file, 0 for a directory.
428
+ """
429
+ offline = self.hsm_offline_count(path, isfile)
430
+ if offline:
431
+ s = ('s' if offline > 1 else '')
432
+ if not self.OPTS['f']:
433
+ self.pglog("{}: {} File{} still on tape, add Mode -f to copy anyway".format(spec, offline, s), self.LOGERR)
434
+ return
435
+ self.pglog("{}: {} File{} still on tape, copy it anyway".format(spec, offline, s), self.LOGWRN)
436
+ tofile = self.join_paths(self.OPTS['t'], spec)
437
+ # a directory is copied as '<path>/.' to merge into an existing target
438
+ fromfile = path if isfile else path + "/."
439
+ if self.local_copy_local(tofile, fromfile, self.LOGWRN):
440
+ self.RINFO['acnt'] += 1
441
+ self.pglog("{}: Copied back to {}".format(spec, tofile), self.LOGWRN)
442
+ else:
443
+ self.pglog("{}: Error copy {} back to {}".format(spec, path, tofile), self.LOGERR)
444
+
445
+ # main function to execute this script
446
+ def main():
447
+ """Entry point: instantiate DecsRestore, parse arguments, run, and exit."""
448
+ from rda_python_setuid.setup_guide import show_setup_guide
449
+ object = DecsRestore()
450
+ show_setup_guide(object, 'rda_python_miscs', ['decsdata_storage', 'decsdata_restore'])
451
+ object.read_parameters()
452
+ object.start_actions()
453
+ object.pgexit(0)
454
+
455
+ # call main() to start program
456
+ if __name__ == "__main__": main()
@@ -0,0 +1,101 @@
1
+
2
+ Restore decsdata datasets, or parts of them, out of the GLADE HSM cold storage.
3
+ This is the opposite of 'decsdata_storage'. A recall request is only submitted
4
+ by 'glade_hsm recall' and is fulfilled by the HSM batch processes later on, so
5
+ restoring is done in three steps:
6
+
7
+ 1. Option -x submits the recall requests for the given dataset paths;
8
+ 2. Option -s checks the recall status, repeat it until nothing is on tape;
9
+ 3. Option -r copies the recalled data back into the decsdata directory.
10
+
11
+ Recalled files stay readable inside the 'COLD_STORAGE/' path for 7 days only,
12
+ after which they are migrated onto tape again, so step 3 must be done within
13
+ that window.
14
+
15
+ Usage: decsdata_restore [-D Date] [-w DecsdataDirectory] [-l ListFile] \
16
+ [-d DatasetPathList] [-f] -x|-s|-r [TargetDirectory]
17
+
18
+ - Option -D Date, the cold storage date, in YYYYMMDD or YYYY-MM-DD. Only
19
+ '<decsdata>/cold_storage_<date>/COLD_STORAGE' is searched for the
20
+ data. Without it, '<decsdata>/COLD_STORAGE' and every
21
+ '<decsdata>/cold_storage_<YYYYMMDD>/COLD_STORAGE' are all searched,
22
+ the most recent dated one first. A dataset path found in more than
23
+ one of them is acted on in the first one only, and the ignored ones
24
+ are logged;
25
+
26
+ - Option -w DecsdataDirectory, the decsdata directory holding the cold
27
+ storage directories, and the directory the data is copied back into.
28
+ Defaults to the configured decsdata root path, /gdex/decsdata;
29
+
30
+ - Option -l ListFile, a file holding one dataset path per line. Only
31
+ used if -d is not given. Defaults to 'dsids_<today>.lst' in the
32
+ current directory, which is generated from every dNNNNNN directory
33
+ found in the cold storage directories if it does not exist yet;
34
+
35
+ - Option -d DatasetPathList, one or more dataset IDs to restore, such as
36
+ '-d d612000 d627000'. Append a sub-path to a dataset ID to restore
37
+ part of it only, such as '-d d612000/2020/01';
38
+
39
+ - Option -f, copy the data back even while some of its files are still on
40
+ tape. Without it, Action -r refuses to copy such data;
41
+
42
+ - Option -h, display this help document;
43
+
44
+ - Action -x, submit the recall requests to bring the data back on disk.
45
+ The decsdata area is limited, so the size to restore is added up
46
+ from Table 'sfile' in RDADB and compared to the space left for the
47
+ decsdata directory, as reported by 'gladequota', before any recall
48
+ is requested. TWICE the size is required, since the recalled copy
49
+ in the cold storage and the copy made by Action -r later on both
50
+ live under the decsdata quota; the recall is stopped if the space
51
+ left is not enough;
52
+
53
+ - Action -s, report the HSM status of the data, including the number of
54
+ files still on tape and the log of any outstanding recall request;
55
+
56
+ - Action -r [TargetDirectory], copy the recalled data back under
57
+ '<decsdata>/<dsid>', or under the given TargetDirectory. A restored
58
+ sub-path keeps its relative position, so 'd612000/2020/01' is copied
59
+ back to '<TargetDirectory>/d612000/2020/01'. The dataset directory,
60
+ and any sub-directory of it, is created if it does not exist yet.
61
+
62
+ One and only one of the Actions -x, -s and -r is required; this help document
63
+ is displayed without any of them. This utility can be run from any directory.
64
+ It is executed under the effective user 'gdexdata' via setuid, so the restored
65
+ data is owned by 'gdexdata'. The cold storage copy of the data is left in
66
+ place by Action -r; move it out of the 'COLD_STORAGE/' path manually to take
67
+ a dataset out of the HSM permanently.
68
+
69
+ Examples:
70
+
71
+ 1. Restore a whole dataset from the cold storage of the root decsdata
72
+ directory, /gdex/decsdata/COLD_STORAGE/d612000:
73
+
74
+ decsdata_restore -d d612000 -x
75
+ decsdata_restore -d d612000 -s
76
+ decsdata_restore -d d612000 -r
77
+
78
+ 2. Restore it from the cold storage directory of a specific date,
79
+ /gdex/decsdata/cold_storage_20250529/COLD_STORAGE/d612000:
80
+
81
+ decsdata_restore -D 20250529 -d d612000 -x
82
+
83
+ 3. Restore one year of a dataset only:
84
+
85
+ decsdata_restore -d d612000/2020 -x
86
+ decsdata_restore -d d612000/2020 -s
87
+ decsdata_restore -d d612000/2020 -r
88
+
89
+ 4. Copy the recalled data back to a directory other than the decsdata one:
90
+
91
+ decsdata_restore -d d612000 -r /PathTo/OtherDirectory
92
+
93
+ 5. Check the status of every dataset in every cold storage directory,
94
+ generating the dataset list file automatically:
95
+
96
+ decsdata_restore -s
97
+
98
+ 6. Copy a dataset back even though some of its files are still on tape,
99
+ which makes the HSM read them off tape while they are being copied:
100
+
101
+ decsdata_restore -d d612000 -f -r
@@ -0,0 +1,196 @@
1
+ #!/usr/bin/env python3
2
+ ##################################################################################
3
+ # Title: decsdata_storage
4
+ # Author: Zaihua Ji, zji@ucar.edu
5
+ # Date: 2026-08-31
6
+ # Purpose: move decsdata datasets into a dated cold storage directory and hand
7
+ # them to the GLADE HSM to be migrated onto tape
8
+ # Github: https://github.com/NCAR/rda-python-miscs.git
9
+ ##################################################################################
10
+ import re
11
+ import os
12
+ import sys
13
+ from os import path as op
14
+ from rda_python_common.pg_file import PgFile
15
+
16
+ class DecsStorage(PgFile):
17
+ """Move decsdata datasets into cold storage and migrate them onto tape.
18
+
19
+ Each given dataset directory is moved from the decsdata directory into
20
+ '<decsdata>/cold_storage_<date>/' and then handed to 'glade_hsm migrate',
21
+ which relocates it once more into a 'COLD_STORAGE/' sub-directory and lets
22
+ the HSM batch processes migrate every large file onto tape. Use
23
+ decsdata_restore to bring the data back.
24
+ """
25
+
26
+ def __init__(self):
27
+ """Initialize DecsStorage with default option values and runtime state."""
28
+ super().__init__()
29
+ self.HSM = os.environ.get('GLADE_HSM', op.expanduser('~benkirk/glade_hsm'))
30
+ self.VALOPTS = 'Dwl' # single-value options
31
+ self.MULOPTS = 'd' # multi-value options
32
+ self.MODOPTS = 'hx' # mode options
33
+ self.OPTS = {
34
+ 'D': None, # cold storage date, in YYYYMMDD or YYYY-MM-DD
35
+ 'w': None, # decsdata directory, defaults to PGLOG['DECSHOME']
36
+ 'l': None, # dataset list file
37
+ 'd': [], # dataset IDs
38
+ 'h': 0, # 1 to show help message
39
+ 'x': 0, # 1 to execute, mandatory
40
+ }
41
+ self.SINFO = {
42
+ 'decsdir': None, # decsdata directory the datasets are stored from
43
+ 'coldstor': None, # cold storage directory the datasets are moved to
44
+ 'date': None, # cold storage date
45
+ 'dcnt': 0, # number of datasets moved into cold storage
46
+ }
47
+
48
+ # function to read parameters
49
+ def read_parameters(self):
50
+ """Parse the command line into the OPTS option values.
51
+
52
+ Single-value options -D, -w and -l take one value each, multi-value
53
+ option -d gathers every following dataset ID, and mode options -h and -x
54
+ are simple flags. Exits with usage if -h is given or -x is missing.
55
+ """
56
+ self.set_suid(self.PGLOG['EUID'])
57
+ self.set_help_path(__file__)
58
+ self.PGLOG['LOGFILE'] = "decsdata_storage.log" # set different log file
59
+ argv = sys.argv[1:]
60
+ self.cmdlog("decsdata_storage {}".format(' '.join(argv)))
61
+ option = None
62
+ for arg in argv:
63
+ ms = re.match(r'^-(\w+)$', arg)
64
+ if ms:
65
+ option = ms.group(1)
66
+ if option in self.MODOPTS:
67
+ self.OPTS[option] = 1
68
+ option = None
69
+ elif option not in self.VALOPTS and option not in self.MULOPTS:
70
+ self.pglog(arg + ": Unknown Option", self.LGEREX)
71
+ continue
72
+ if not option: self.pglog(arg + ": Value provided without option", self.LGEREX)
73
+ if option in self.MULOPTS:
74
+ self.OPTS[option].append(arg) # gather all values until the next option
75
+ else:
76
+ self.OPTS[option] = arg
77
+ option = None
78
+ if self.OPTS['h'] or not self.OPTS['x']: self.show_usage("decsdata_storage")
79
+
80
+ # function to start actions
81
+ def start_actions(self):
82
+ """Validate the caller, resolve the cold storage paths, and store each dataset."""
83
+ self.dssdb_dbname()
84
+ self.validate_decs_group('decsdata_storage', self.PGLOG['CURUID'], 1)
85
+ self.set_storage_paths()
86
+ dsids = self.get_dataset_list()
87
+ if not dsids: self.pglog("No dataset found for cold storage", self.LGWNEX)
88
+ for dsid in dsids:
89
+ self.store_one_dataset(dsid)
90
+ s = ('s' if self.SINFO['dcnt'] > 1 else '')
91
+ self.pglog("{} of {} Dataset{} moved into {}".format(self.SINFO['dcnt'],
92
+ len(dsids), s, self.SINFO['coldstor']), self.LOGWRN)
93
+ self.cmdlog()
94
+
95
+ # resolve the decsdata and cold storage directories
96
+ def set_storage_paths(self):
97
+ """Fill SINFO with the decsdata directory, cold storage date and cold storage path.
98
+
99
+ Defaults the decsdata directory to PGLOG['DECSHOME'] and the date to
100
+ today. Dashes are stripped from the given date, which must then be 8
101
+ digits.
102
+ """
103
+ decsdir = self.OPTS['w'] if self.OPTS['w'] else self.PGLOG['DECSHOME']
104
+ if not self.check_local_file(decsdir, 0, self.LOGWRN):
105
+ self.pglog(decsdir + ": decsdata directory NOT exists", self.LGEREX)
106
+ date = re.sub('-', '', self.OPTS['D']) if self.OPTS['D'] else re.sub('-', '', self.curdate())
107
+ if not re.match(r'^\d{8}$', date):
108
+ self.pglog(date + ": Invalid cold storage date, YYYYMMDD expected", self.LGEREX)
109
+ self.SINFO['decsdir'] = decsdir
110
+ self.SINFO['date'] = date
111
+ self.SINFO['coldstor'] = self.join_paths(decsdir, "cold_storage_" + date)
112
+
113
+ # gather the dataset IDs to move into cold storage
114
+ def get_dataset_list(self):
115
+ """Return the list of dataset IDs to store.
116
+
117
+ Uses the -d values if given. Otherwise reads the -l list file, creating
118
+ it first from every dNNNNNN directory in the decsdata directory if it
119
+ does not exist yet.
120
+
121
+ Returns:
122
+ list[str]: Dataset IDs, empty if none is found.
123
+ """
124
+ if self.OPTS['d']:
125
+ self.pglog("Store {} given Dataset(s) into cold storage".format(len(self.OPTS['d'])), self.LOGWRN)
126
+ return self.OPTS['d']
127
+ lstfile = self.OPTS['l'] if self.OPTS['l'] else "dsids_{}.lst".format(self.SINFO['date'])
128
+ if not op.isfile(lstfile):
129
+ dsids = self.get_decsdata_datasets()
130
+ with open(lstfile, 'w') as OUT:
131
+ for dsid in dsids: OUT.write(dsid + "\n")
132
+ self.pglog("{}: Generated with {} Dataset(s)".format(lstfile, len(dsids)), self.LOGWRN)
133
+ dsids = []
134
+ with open(lstfile, 'r') as IN:
135
+ for line in IN:
136
+ line = line.strip()
137
+ if line: dsids.append(line)
138
+ self.pglog("{}: Read {} Dataset(s) for cold storage".format(lstfile, len(dsids)), self.LOGWRN)
139
+ return dsids
140
+
141
+ # find all dNNNNNN dataset directories in the decsdata directory
142
+ def get_decsdata_datasets(self):
143
+ """Return the sorted dataset IDs of every dNNNNNN directory in the decsdata directory.
144
+
145
+ Returns:
146
+ list[str]: Dataset IDs; plain files matching the pattern are skipped.
147
+ """
148
+ pattern = self.join_paths(self.SINFO['decsdir'], "d" + "[0-9]"*6)
149
+ files = self.local_glob(pattern, 0, self.LOGWRN)
150
+ dsids = []
151
+ for file in files:
152
+ if not files[file]['isfile']: dsids.append(op.basename(file))
153
+ return sorted(dsids)
154
+
155
+ # move one dataset into cold storage and migrate it onto tape
156
+ def store_one_dataset(self, dsid):
157
+ """Move one dataset into the cold storage directory and migrate it onto tape.
158
+
159
+ Skips the dataset if it is not an existing directory in the decsdata
160
+ directory or if the move fails. Increments SINFO['dcnt'] for each
161
+ dataset successfully migrated.
162
+
163
+ Args:
164
+ dsid (str): Dataset ID, such as 'd612000'.
165
+ """
166
+ fromfile = self.join_paths(self.SINFO['decsdir'], dsid)
167
+ info = self.check_local_file(fromfile, 0, self.LOGWRN)
168
+ if not info:
169
+ self.pglog(fromfile + ": Dataset NOT exists", self.LOGERR)
170
+ return
171
+ if info['isfile']:
172
+ self.pglog(fromfile + ": Not a dataset directory", self.LOGERR)
173
+ return
174
+ tofile = self.join_paths(self.SINFO['coldstor'], dsid)
175
+ if not self.move_local_file(tofile, fromfile, self.LOGWRN):
176
+ self.pglog("{}: Error move {} into cold storage".format(fromfile, dsid), self.LOGERR)
177
+ return
178
+ cmd = "{} migrate -f {}".format(self.HSM, tofile)
179
+ if self.pgsystem(cmd, self.LOGWRN, 7):
180
+ self.SINFO['dcnt'] += 1
181
+ self.pglog("{}: Migrated onto tape from {}".format(dsid, tofile), self.LOGWRN)
182
+ else:
183
+ self.pglog("{}: Error migrate onto tape from {}".format(dsid, tofile), self.LOGERR)
184
+
185
+ # main function to execute this script
186
+ def main():
187
+ """Entry point: instantiate DecsStorage, parse arguments, run, and exit."""
188
+ from rda_python_setuid.setup_guide import show_setup_guide
189
+ object = DecsStorage()
190
+ show_setup_guide(object, 'rda_python_miscs', ['decsdata_storage', 'decsdata_restore'])
191
+ object.read_parameters()
192
+ object.start_actions()
193
+ object.pgexit(0)
194
+
195
+ # call main() to start program
196
+ if __name__ == "__main__": main()
@@ -0,0 +1,66 @@
1
+
2
+ Move decsdata datasets into a dated cold storage directory and hand them to the
3
+ GLADE HSM to be migrated onto tape. Each dataset directory is moved from the
4
+ decsdata directory into '<decsdata>/cold_storage_<date>/' and then passed to
5
+ 'glade_hsm migrate', which relocates it once more into a 'COLD_STORAGE/'
6
+ sub-directory, ending up as:
7
+
8
+ <decsdata>/cold_storage_<date>/COLD_STORAGE/<dsid>
9
+
10
+ The HSM batch processes then migrate every large file (>= 100MB) underneath the
11
+ 'COLD_STORAGE/' path onto tape. All files inside a 'COLD_STORAGE/' path are
12
+ marked immutable, which prevents modification of their contents and metadata.
13
+ Use 'decsdata_restore' to bring the data back.
14
+
15
+ Usage: decsdata_storage [-D Date] [-w DecsdataDirectory] [-l ListFile] \
16
+ [-d DatasetList] -x
17
+
18
+ - Option -D Date, the cold storage date, in YYYYMMDD or YYYY-MM-DD. It
19
+ names the cold storage directory '<decsdata>/cold_storage_<date>'.
20
+ Defaults to today;
21
+
22
+ - Option -w DecsdataDirectory, the decsdata directory the datasets are
23
+ stored from, and the directory the cold storage directory is created
24
+ in. Defaults to the configured decsdata root path, /gdex/decsdata;
25
+
26
+ - Option -l ListFile, a file holding one dataset ID per line. Only used
27
+ if -d is not given. Defaults to 'dsids_<date>.lst' in the current
28
+ directory, which is generated from every dNNNNNN directory found in
29
+ the decsdata directory if it does not exist yet;
30
+
31
+ - Option -d DatasetList, one or more dataset IDs to move into cold
32
+ storage, such as '-d d612000 d627000';
33
+
34
+ - Option -h, display this help document;
35
+
36
+ - Option -x, execute the cold storage moves. This option is mandatory;
37
+ this help document is displayed without it. Note that no
38
+ confirmation is asked for once -x is given.
39
+
40
+ This utility can be run from any directory. It is executed under the
41
+ effective user 'gdexdata' via setuid, so the moved data keeps its ownership.
42
+ A dataset is skipped, and an error is logged, if it is not an existing
43
+ directory in the decsdata directory or if it cannot be moved.
44
+
45
+ Examples:
46
+
47
+ 1. Move two given datasets into today's cold storage:
48
+
49
+ decsdata_storage -d d612000 d627000 -x
50
+
51
+ 2. Move them into the cold storage directory of a specific date:
52
+
53
+ decsdata_storage -D 20250529 -d d612000 -x
54
+
55
+ The data then ends up in:
56
+
57
+ /gdex/decsdata/cold_storage_20250529/COLD_STORAGE/d612000
58
+
59
+ 3. Move every dataset listed in a list file:
60
+
61
+ decsdata_storage -l my_dsids.lst -x
62
+
63
+ 4. Move every dNNNNNN dataset found in a non-default decsdata directory,
64
+ generating the dataset list file automatically:
65
+
66
+ decsdata_storage -w /PathTo/decsdata -x
@@ -140,10 +140,10 @@ class PgRST(PgFile, PgUtil):
140
140
  """
141
141
  self.OPTS = opts
142
142
  self.ALIAS = alias
143
+ self.DOCS['DOCNAM'] = docname
143
144
 
144
145
  self.parse_docs(docname)
145
146
  if not self.sections: self.pglog(docname + ": empty document", self.LGWNEX)
146
- self.DOCS['DOCNAM'] = docname
147
147
  if docname in self.LINKS: self.LINKS.remove(docname)
148
148
  self.DOCS['DOCLNK'] = r"({})".format('|'.join(self.LINKS))
149
149
  self.DOCS['DOCTIT'] = docname.upper()
@@ -152,6 +152,7 @@ class PgRST(PgFile, PgUtil):
152
152
  self.write_index(self.sections[0])
153
153
  for section in self.sections:
154
154
  self.write_section(section)
155
+ self.write_appendix()
155
156
 
156
157
  #
157
158
  # parse the original document and return a array of sections,
@@ -172,7 +173,9 @@ class PgRST(PgFile, PgUtil):
172
173
  with open(docfile, 'r') as fh:
173
174
  line = fh.readline()
174
175
  while line:
175
- if re.match(r'\s*#', line):
176
+ # Skip full-line authoring comments, but keep '#!' shebang lines so
177
+ # shell scripts shown in example content blocks stay intact.
178
+ if re.match(r'\s*#(?!!)', line):
176
179
  line = fh.readline()
177
180
  continue # skip comment lines
178
181
  ms = re.match(r'^(.*\S)\s+#', line)
@@ -191,13 +194,21 @@ class PgRST(PgFile, PgUtil):
191
194
  option = self.record_option(section, option, example, ms.group(1), ms.group(2))
192
195
  example = None
193
196
  elif option:
194
- ms = re.match(r'^ For( | another )example, (.*)$', line)
195
- if ms: # found example
196
- example = self.record_example(option, example, ms.group(2))
197
- elif example:
198
- example['desc'] += line + "\n"
197
+ if option.get('inexm'):
198
+ # inside an 'Examples:' block: collect raw lines verbatim
199
+ # (split into individual examples later in split_examples)
200
+ option['exmraw'] += line + "\n"
199
201
  else:
200
- option['desc'] += line + "\n"
202
+ ms = re.match(r'^ (?:For(?: | another )example,|Example[ ]*[,\-])\s*(.*)$', line)
203
+ if ms: # found a single labeled example
204
+ example = self.record_example(option, example, ms.group(1))
205
+ elif re.match(r'^ Examples:\s*$', line):
206
+ option['inexm'] = True # start of a multi-example block
207
+ option['exmraw'] = ""
208
+ elif example:
209
+ example['desc'] += line + "\n"
210
+ else:
211
+ option['desc'] += line + "\n"
201
212
  else:
202
213
  section['desc'] += line + "\n"
203
214
 
@@ -261,11 +272,62 @@ class PgRST(PgFile, PgUtil):
261
272
  """
262
273
  if option:
263
274
  if example: self.record_example(option, example)
275
+ self.split_examples(option)
264
276
  self.options[option['opt']] = option # record option globally
265
277
  section['opts'].append(option['opt']) # record option short name in section
266
278
 
267
279
  if nopt: return self.init_option(section['secid'], nopt, ndesc)
268
280
 
281
+ def split_examples(self, option):
282
+ """Split an option's pending ``Examples:`` block into individual examples.
283
+
284
+ The raw text collected after an ``Examples:`` header is segmented into
285
+ blank-line-separated blocks. A block whose last line ends with ``:`` and
286
+ whose first line is descriptive prose (not a command, option flag, or a
287
+ ``<<Content ...>>`` header) starts a new example; following blocks
288
+ (command synopsis, script content, trailing notes) belong to that
289
+ example until the next title block.
290
+
291
+ Args:
292
+ option (dict): The option whose ``exmraw`` buffer is parsed; the
293
+ buffer is removed and one example is recorded per
294
+ title block found.
295
+ """
296
+ buf = option.pop('exmraw', None)
297
+ option.pop('inexm', None)
298
+ if not buf: return
299
+
300
+ blocks = []
301
+ cur = []
302
+ for ln in buf.split('\n'):
303
+ if ln.strip() == '':
304
+ if cur: blocks.append(cur); cur = []
305
+ else:
306
+ cur.append(ln)
307
+ if cur: blocks.append(cur)
308
+
309
+ def is_title(block):
310
+ if not block[-1].rstrip().endswith(':'): return False
311
+ first = block[0].strip()
312
+ if first.startswith('<<'): return False
313
+ if re.match(r'[-*(]', first): return False
314
+ if re.match(r'{}\b'.format(self.DOCS['DOCNAM']), first): return False
315
+ return True
316
+
317
+ exmtext = None
318
+ for block in blocks:
319
+ btext = '\n'.join(block)
320
+ if is_title(block):
321
+ if exmtext is not None:
322
+ self.record_example(option, self.init_example(option['opt'], exmtext))
323
+ exmtext = btext + "\n"
324
+ elif exmtext is not None:
325
+ exmtext += "\n" + btext + "\n"
326
+ else:
327
+ option['desc'] += btext + "\n" # stray text before the first example
328
+ if exmtext is not None:
329
+ self.record_example(option, self.init_example(option['opt'], exmtext))
330
+
269
331
  def record_example(self, option, example, ndesc=None):
270
332
  """Append the completed *example* to ``self.examples`` and optionally start a new one.
271
333
 
@@ -478,18 +540,21 @@ class PgRST(PgFile, PgUtil):
478
540
  """Build and return the RST table-of-contents string of a given section.
479
541
 
480
542
  Produces a nested bullet list of section links (indented by section
481
- level) followed by a flat Appendix A list of all example links.
543
+ level). For the index (``csection is None``) the example appendix is a
544
+ standalone page (``appendixA``) added to the toctree; for a section the
545
+ examples in its subtree are listed inline as a local Appendix A.
482
546
 
483
547
  Returns:
484
548
  str: RST-formatted TOC content ready for ``__TOC__`` substitution.
485
549
  """
486
-
550
+
487
551
  content = ""
488
552
  clevel = csection['level'] if csection else 0
489
553
  csecid = csection['secid'] if csection else ""
490
554
  depth = self.TLEVEL - clevel
491
555
  level = clevel+1
492
556
  preid = csecid+'.'
557
+ is_index = csection is None
493
558
 
494
559
  # nested bullet list for all sections
495
560
  for section in self.sections:
@@ -497,9 +562,14 @@ class PgRST(PgFile, PgUtil):
497
562
  if csecid and not secid.startswith(preid): continue
498
563
  if section['level'] == level: content += " section{}\n".format(secid)
499
564
 
565
+ # The full list of examples lives on its own appendix page in the index.
566
+ if is_index and self.examples: content += " appendixA\n"
567
+
500
568
  if not content: return ""
501
569
 
502
570
  content = f".. toctree::\n :maxdepth: {depth}\n :caption: Table of Contents\n\n{content}\n"
571
+ if is_index: return content
572
+
503
573
  # appendix A: list of examples for the parent section and its subsections
504
574
  appendix = ""
505
575
  idx = 1 # used as example index
@@ -507,7 +577,7 @@ class PgRST(PgFile, PgUtil):
507
577
  opt = exm['opt']
508
578
  option = self.options[opt]
509
579
  secid = option['secid']
510
- if not csecid or secid == csecid or secid.startswith(preid):
580
+ if secid == csecid or secid.startswith(preid):
511
581
  appendix += "- :ref:`A.{}. {} Option -{} (-{}) <{}_e{}>`\n".format(
512
582
  idx, option['type'], opt, option['name'], secid, idx)
513
583
  idx += 1
@@ -516,6 +586,28 @@ class PgRST(PgFile, PgUtil):
516
586
 
517
587
  return content
518
588
 
589
+ #
590
+ # write the appendix page listing all examples in the document
591
+ #
592
+ def write_appendix(self):
593
+ """Write ``appendixA.rst`` listing every example with a link.
594
+
595
+ Each entry links to the example anchor on its section page. Does nothing
596
+ when the document has no examples.
597
+ """
598
+ if not self.examples: return
599
+ content = ""
600
+ idx = 1
601
+ for exm in self.examples:
602
+ option = self.options[exm['opt']]
603
+ secid = option['secid']
604
+ title = exm['title'].strip().rstrip(':')
605
+ content += "- :ref:`A.{}. {} Option -{} (-{}): {} <{}_e{}>`\n".format(
606
+ idx, option['type'], exm['opt'], option['name'], title, secid, idx)
607
+ idx += 1
608
+
609
+ self.template_to_rst("appendix", {'CONTENT': content}, "A")
610
+
519
611
  #
520
612
  # create a section rst content
521
613
  #
@@ -672,7 +764,8 @@ class PgRST(PgFile, PgUtil):
672
764
 
673
765
  for optary in opts:
674
766
  opt = self.get_short_option(optary[1])
675
-
767
+ if opt is None: continue # not an option of this document; leave as-is
768
+
676
769
  pre = optary[0]
677
770
  after = optary[2]
678
771
  secid = self.options[opt]['secid']
@@ -826,7 +919,7 @@ class PgRST(PgFile, PgUtil):
826
919
  line0 = lines[0]
827
920
  normal = 1
828
921
  if dtype == 2:
829
- ms = re.match(r'^<<(Content .*)>>$', line0)
922
+ ms = re.match(r'^\s*<<(Content .*)>>$', line0)
830
923
  if ms: # input files for examples
831
924
  content += ms.group(1) + ":\n\n.. code-block:: none\n\n"
832
925
  normal = 0
@@ -928,13 +1021,23 @@ class PgRST(PgFile, PgUtil):
928
1021
  rows.append(tuple(prev_vals))
929
1022
  content = self.build_rst_list_table(rows)
930
1023
  else:
931
- # multi-column table split on 2+ spaces
932
- rows = []
933
- for i in range(cnt):
934
- line = lines[i].strip()
935
- vals = re.split(r'\s{2,}', self.replace_option_link(line, secid, 1))
936
- rows.append(vals)
937
- content = self.build_rst_simple_table(rows) + "\n"
1024
+ raw = [lines[i].strip() for i in range(cnt)]
1025
+ cmdpat = r'(?:[*\d][\d* ]*\s+)?{}(\s|$)'.format(re.escape(self.DOCS['DOCNAM']))
1026
+ if raw and all(re.match(cmdpat, r) for r in raw):
1027
+ # Command line(s) following a label (e.g. a Quick Start entry):
1028
+ # render as a literal block instead of a (degenerate) table.
1029
+ content = ".. code-block:: none\n\n"
1030
+ for r in raw:
1031
+ content += " " + r + "\n"
1032
+ content += "\n"
1033
+ else:
1034
+ # multi-column table split on 2+ spaces
1035
+ rows = []
1036
+ for i in range(cnt):
1037
+ line = lines[i].strip()
1038
+ vals = re.split(r'\s{2,}', self.replace_option_link(line, secid, 1))
1039
+ rows.append(vals)
1040
+ content = self.build_rst_simple_table(rows) + "\n"
938
1041
 
939
1042
  return content
940
1043
 
@@ -1046,10 +1149,10 @@ class PgRST(PgFile, PgUtil):
1046
1149
  p (str): Option name to look up (short, long, or alias).
1047
1150
 
1048
1151
  Returns:
1049
- str: Canonical two-letter option short name.
1050
-
1051
- Raises:
1052
- PgLOG error (LGWNEX) if *p* cannot be resolved.
1152
+ str | None: Canonical two-letter option short name, or ``None`` when
1153
+ *p* does not name an option of this document (e.g. an option of a
1154
+ different program referenced in prose). Callers skip such tokens
1155
+ so unrelated option-like text is left untouched.
1053
1156
  """
1054
1157
  plen = len(p)
1055
1158
  if plen == 2 and p in self.options: return p
@@ -1061,7 +1164,7 @@ class PgRST(PgFile, PgUtil):
1061
1164
  for alias in self.ALIAS[opt]:
1062
1165
  if re.match(r'^{}$'.format(alias), p, re.I): return opt
1063
1166
 
1064
- self.pglog("{} - unknown option for {}".format(p, self.DOCS['DOCNAM']), self.LGWNEX)
1167
+ return None
1065
1168
 
1066
1169
  #
1067
1170
  # replace with rst link for a given section title
@@ -0,0 +1,19 @@
1
+ ################################################################################
2
+ #
3
+ # Title : appendix_rst.temp
4
+ # Author : Zaihua Ji, zji@ucar.edu
5
+ # Date : 03/17/2026
6
+ # Purpose : template file for help document appendixA.rst (reStructuredText)
7
+ # Github : https://github.com/NCAR/rda-python-mics.git
8
+ #
9
+ ################################################################################
10
+
11
+ .. _appendixA:
12
+
13
+ ============================
14
+ Appendix A: List of Examples
15
+ ============================
16
+
17
+ __CONTENT__
18
+
19
+ | :ref:`Back to Table of Contents <index>`
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rda_python_miscs
3
- Version: 3.0.3
3
+ Version: 3.0.5
4
4
  Summary: RDA Python package to hold RDA miscellaneous utility programs
5
5
  Author-email: Zaihua Ji <zji@ucar.edu>
6
6
  Project-URL: Homepage, https://github.com/NCAR/rda-python-miscs
@@ -45,6 +45,8 @@ The package provides two categories of programs:
45
45
  | `gdexcp` | `rdacp` | `setuid_gdexcp` / `setuid_rdacp` | Copy files and directories across local, remote, Object Store, or Globus endpoints |
46
46
  | `gdexkill` | `rdakill` | `setuid_gdexkill` / `setuid_rdakill` | Kill local processes and their children, or cancel PBS batch jobs |
47
47
  | `gdexmod` | `rdamod` | `setuid_gdexmod` / `setuid_rdamod` | Change permission modes for files and directories owned by gdexdata |
48
+ | `decsdata_storage` | | `setuid_decsdata_storage` | Move decsdata datasets into the GLADE HSM cold storage to migrate them onto tape |
49
+ | `decsdata_restore` | | `setuid_decsdata_restore` | Recall decsdata datasets out of the GLADE HSM cold storage and copy them back |
48
50
 
49
51
  ## Environment setup
50
52
 
@@ -103,8 +105,8 @@ pip install rda_python_miscs
103
105
 
104
106
  ## Setuid Setup
105
107
 
106
- The setuid programs (`gdexcp`, `gdexkill`, `gdexmod` and their `rda*` aliases)
107
- execute as the common user `PGLOG['COMMONUSER']` (default `gdexdata`) via
108
+ The setuid programs (`gdexcp`, `gdexkill`, `gdexmod`, `decsdata_storage`,
109
+ `decsdata_restore` and the `rda*` aliases) execute as the common user `PGLOG['COMMONUSER']` (default `gdexdata`) via
108
110
  the `rda_python_setuid` mechanism, which is pulled in automatically as a
109
111
  dependency. After `pip install` above, choose one of the wiring options
110
112
  below.
@@ -159,3 +161,30 @@ pywrapper-install -u|--update
159
161
  The shared setuid setup guide is shown automatically if any `setuid_*`
160
162
  connector script is invoked directly before the setuid wrapper has been
161
163
  configured.
164
+
165
+ ## Cold storage setup
166
+
167
+ `decsdata_storage` and `decsdata_restore` drive the GLADE HSM through the
168
+ `glade_hsm` script, which is not part of this package. They look it up at
169
+ `~benkirk/glade_hsm` unless the environment variable `GLADE_HSM` names another
170
+ copy of it:
171
+
172
+ ```bash
173
+ export GLADE_HSM=/path/to/glade_hsm
174
+ ```
175
+
176
+ Both programs work on the decsdata directory `PGLOG['DECSHOME']` (default
177
+ `/glade/campaign/collections/gdex/decsdata`, also reachable as
178
+ `/gdex/decsdata`), which option `-w` overrides. Only members of the DECS group
179
+ may run them, and `decsdata_restore` reads Table `sfile` in RDADB to size a
180
+ restore, so a working RDADB login is required as well.
181
+
182
+ Because the decsdata quota is limited, check the space left with `gladequota`
183
+ before a large restore:
184
+
185
+ ```bash
186
+ pgstart_gdexdata gladequota # or plain 'gladequota' when logged in as gdexdata
187
+ ```
188
+
189
+ `decsdata_restore -x` runs the same check itself and refuses to request a
190
+ recall unless twice the size of the data is still available.
@@ -2,6 +2,10 @@ rda_python_miscs/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,
2
2
  rda_python_miscs/bash_qsub.py,sha256=NsYg1A7zeRy3mR_rAaJEQoWSDaWFxE7cpHNc4mA_JwA,6664
3
3
  rda_python_miscs/bashqsub.py,sha256=SadPwz1ZXHx2zGNYb2VityItQjMEYj-Favq3xDrPBK0,10127
4
4
  rda_python_miscs/bashqsub.usg,sha256=QDaFr1FiQ-5FC9UzCRQ0geYjsPXBnVMhe3lM8qH1xss,2176
5
+ rda_python_miscs/decsdata_restore.py,sha256=JmEiYhQyE_Z6PZstrm35-ImYVz4j7cpffhhLiS4UGHA,20400
6
+ rda_python_miscs/decsdata_restore.usg,sha256=jOD_AbNV4C-dOp9mTkV2MVcUgcf2kf5Q9QqhCS7NENk,4945
7
+ rda_python_miscs/decsdata_storage.py,sha256=5n8bOVIpGxnX7beDNGmrgj4z1OwmU4-9qVsDCbSuymY,8637
8
+ rda_python_miscs/decsdata_storage.usg,sha256=5_OXsIxyKjA46xjycaTgVVIE0c39lkkSekcVliFNgaI,2812
5
9
  rda_python_miscs/gdex_ls.py,sha256=fg9jfYajOT8ps6eEhFUUqmQ791BdGwBkrT_pyjmGGvQ,8860
6
10
  rda_python_miscs/gdexcp.py,sha256=jCkDzNrZhi7vhZqeSXHR7d1XIKtsionSJE5ntWoZE0g,18110
7
11
  rda_python_miscs/gdexcp.usg,sha256=6r8XVi9inPuWuFth6CAzcd3EpXX4Y_ZrT6YubmQ82kM,6172
@@ -21,7 +25,7 @@ rda_python_miscs/gdexsub.usg,sha256=SChXnDzo6larOdpR9eZ2CMu8ExJYQUdfFHUzxcGoqJ0,
21
25
  rda_python_miscs/gdexzip.py,sha256=wzMVTqeahXAUHsKAtqYG6wrM4J2EahnncrpliCqFyNQ,3252
22
26
  rda_python_miscs/gdexzip.usg,sha256=cG1Uwa8WZ3KQNSaSkr29edUEaaLb_4cHTU3GRoZSQ84,1918
23
27
  rda_python_miscs/pg_docs.py,sha256=_AoqrWroUu7FQMVWaTSTZBQMwi1BH1-RjKb1nDHgxOI,25369
24
- rda_python_miscs/pg_rst.py,sha256=XldjKneKuciYvZxrHC3h3_OHwyhh51LKYOuKr0s_2Gg,48480
28
+ rda_python_miscs/pg_rst.py,sha256=Wd4qfeetQYvRio9P2YE2x4kUCDI6ZEcX8CQiI0Xqmyk,53063
25
29
  rda_python_miscs/pg_rst.usg,sha256=oYvVEqzHeVmDntN7hfpM8vc6eoYRyQTruQQg2mPfOo4,2484
26
30
  rda_python_miscs/pg_wget.py,sha256=YK8EJQtTwGXxHav_SveXOwE_oA9KZ9l12lGSf77JQnE,7571
27
31
  rda_python_miscs/pgwget.py,sha256=TFHyquKT0tyz-r-RUAV_wpdwktpVRwyZGrHnZF2B3Vk,8115
@@ -38,11 +42,12 @@ rda_python_miscs/rdals.usg,sha256=ChF-nn3Qb2pds3wMXIWubB_tjwZslxHfSha0dTDqOiY,29
38
42
  rda_python_miscs/tcsh_qsub.py,sha256=P4Obzbp5Dvy-N0eRR5F51R9yj3ttWJo6g1NqyVDO6-Y,6670
39
43
  rda_python_miscs/tcshqsub.py,sha256=QxBq9MdVIUs9t2d6vHhkxM1nrcLwRNqcq1lJiWhXKUM,10124
40
44
  rda_python_miscs/tcshqsub.usg,sha256=JYfhrK7cqme-Sij_JfquONOs3HMu-d5dDGI9K_RdudU,2180
45
+ rda_python_miscs/rst_templates/appendix.rst.temp,sha256=HCujxbj-R4_8FgOrLSlMZcBQHUu8f6kfBMWVRygl7jc,560
41
46
  rda_python_miscs/rst_templates/index.rst.temp,sha256=YSa1JM6X9x2SC6UiqJu_9xRrVYGLKwNI43DBJEGDX08,523
42
47
  rda_python_miscs/rst_templates/section.rst.temp,sha256=-CUtutvctG2tIdkqrkFxVEzmNN3atRJXBDQaTnMJ6Gw,572
43
- rda_python_miscs-3.0.3.dist-info/licenses/LICENSE,sha256=1dck4EAQwv8QweDWCXDx-4Or0S8YwiCstaso_H57Pno,1097
44
- rda_python_miscs-3.0.3.dist-info/METADATA,sha256=b4yrhKT1F_QySBljOiOujWrWLMStJO0Vymjx8dO47f4,5803
45
- rda_python_miscs-3.0.3.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
46
- rda_python_miscs-3.0.3.dist-info/entry_points.txt,sha256=pBgb-_g4yZhm6YynwDHtNTAzxVrb8SoDMd7Eiys8gv4,806
47
- rda_python_miscs-3.0.3.dist-info/top_level.txt,sha256=W5rz7DrWb7hXABUbGgWcwe6D644X338LR8_zdgmtLhg,17
48
- rda_python_miscs-3.0.3.dist-info/RECORD,,
48
+ rda_python_miscs-3.0.5.dist-info/licenses/LICENSE,sha256=1dck4EAQwv8QweDWCXDx-4Or0S8YwiCstaso_H57Pno,1097
49
+ rda_python_miscs-3.0.5.dist-info/METADATA,sha256=9OT0Q0nHql2BdJI85ffYfp6EJD4Qm2KmYgn2R2ny9KM,7106
50
+ rda_python_miscs-3.0.5.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
51
+ rda_python_miscs-3.0.5.dist-info/entry_points.txt,sha256=vAM87BmdEbJKdLg92NaffFG_xCdb9JxR2PZgsCjB8yw,936
52
+ rda_python_miscs-3.0.5.dist-info/top_level.txt,sha256=W5rz7DrWb7hXABUbGgWcwe6D644X338LR8_zdgmtLhg,17
53
+ rda_python_miscs-3.0.5.dist-info/RECORD,,
@@ -1,5 +1,5 @@
1
1
  Wheel-Version: 1.0
2
- Generator: setuptools (82.0.1)
2
+ Generator: setuptools (84.0.0)
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any
5
5
 
@@ -11,6 +11,8 @@ rdaown = rda_python_miscs.gdexown:main
11
11
  rdaps = rda_python_miscs.gdexps:main
12
12
  rdasub = rda_python_miscs.gdexsub:main
13
13
  rdazip = rda_python_miscs.gdexzip:main
14
+ setuid_decsdata_restore = rda_python_miscs.decsdata_restore:main
15
+ setuid_decsdata_storage = rda_python_miscs.decsdata_storage:main
14
16
  setuid_gdexcp = rda_python_miscs.gdexcp:main
15
17
  setuid_gdexkill = rda_python_miscs.gdexkill:main
16
18
  setuid_gdexmod = rda_python_miscs.gdexmod:main