pipeforge 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,544 @@
1
+ # vim: colorcolumn=101 textwidth=100
2
+
3
+ import datetime
4
+
5
+ from . import utils # Local module
6
+
7
+
8
+
9
+ ####################################################################################################
10
+ # Auxiliary functions
11
+ ####################################################################################################
12
+
13
+ def _job_fits(job, jobs):
14
+ """
15
+ Given a job definition (with start and stop timestamps) return whether it "fits" in a list of
16
+ jobs without overlapping with any of them.
17
+
18
+ @param job: a tuple whose second and third elements contain the start and stop timestamps.
19
+
20
+ @param jobs: a list of jobs (each one with the same properties as @ref job)
21
+
22
+ @return True if @ref job does not overlap with any of the @ref jobs. False otherwise
23
+ """
24
+
25
+ for start, stop in [(x[1], x[3]) for x in jobs]:
26
+ if (job[1] > start and job[1] < stop) or \
27
+ (job[3] > start and job[3] < stop) or \
28
+ (job[1] <= start and job[3] >= stop):
29
+ return False
30
+
31
+ return True
32
+
33
+
34
+ def _print_ascii(stacks, secs_per_unit, scale, output_width=100, unicode_supported=True,
35
+ stdout="stdout"):
36
+ """
37
+ Print an ASCII timeline containing all provided jobs (contained in a list of @ref stacks)
38
+
39
+ @param stacks: list of "stacks" of jobs. Each stack contains non-overlapping (in execution time)
40
+ jobs, meaning they can be printed in the same row.
41
+ Each job on each stack is represented by a list of 6 elements:
42
+
43
+ - [0] (string) Job name
44
+ - [1] (int) Start time (from 0 to 100)
45
+ - [2] (int) Start2 time (from 0 to 100). This is when the job *really* started executing.
46
+ Before that it was waiting in queue.
47
+ - [3] (int) Stop time (from 0 to 100)
48
+ - [4] (string) Job status
49
+ - [5] (string) "True" or "False" depending on whether this is a detached job (ie. one
50
+ which does not require monitoring)
51
+
52
+ Example:
53
+
54
+ stacks = [
55
+ [["A", 0, 0, 20, "SUCCESS", "False"], ["C", 20, 25, 60, "SUCCESS", "False"], ["D", 90, 90, 100, "SUCCESS", "False"]],
56
+ [["B", 10, 80, 90, "SUCCESS", "False"], ["E", 90, 90, 95, "SUCCESS", "False"]]
57
+ ]
58
+
59
+ ...which would generate something like this:
60
+
61
+ |---A---|...--C------| |-D-|
62
+ |..............B....--------|E|
63
+
64
+ |----------------------------------> t
65
+
66
+ @param secs_per_unit: Number of seconds each of the 100 normalized units represents. If set to
67
+ 0, the translation from units to time will not be done and the final representation will
68
+ include normalized units instead.
69
+
70
+ @param scale: Can be "minutes" or "hours". In the first case, each timestamp on the horizontal
71
+ axis will be printed in Xm:YYs format (ex: "8m:09s"). In the second case, each timestamp will
72
+ look like Xh:YYm (ex: "2h:39m").
73
+ If can also take the value "auto". In that case the format will be automatically selected for
74
+ you depending on the total time spawn to be printed.
75
+
76
+ @param output_width: Number of columns to use to represent the timeline
77
+
78
+ @param unicode_supported: Set to True if the output can use unicode characters. Set to False if
79
+ you only want regular ASCII characters (which is less "pretty" but works fine also)
80
+
81
+ @param stdout: Where to send the resulting timeline. Can take any of these values:
82
+
83
+ - "stdout" : print to stdout
84
+ - "buffer" : print to a buffer and *return* it to the caller of this function
85
+ - "file@/path/to/file.txt" : print to the provided file (warning! contents will be
86
+ overwritten)
87
+
88
+ @return a test buffer with the resulting timeline representation if @ref stdout was set to
89
+ "buffer". Return None otherwise.
90
+ """
91
+ #print(stacks) #DEBUG
92
+
93
+ output_buffer = []
94
+
95
+ tokens = "ABCDEFGIJKLMNOPQRTUVWabcdefghijklmnopqrstuvw" * 10
96
+ tokens_super = "ᴬᴮꟲᴰᴱꟳᴳᴵᴶᴷᴸᴹᴺᴼᴾꟴᴿᵀᵁⱽᵂᵃᵇᶜᵈᵉᶠᵍʰⁱʲᵏˡᵐⁿᵒᵖ𐞥ʳˢᵗᵘᵛʷ" * 10
97
+ # Note: there are missing letters because not all of them have a unicode "superscript"
98
+ # version
99
+
100
+ if unicode_supported:
101
+ symbols = {
102
+ "SUCCESS" : [("│","-","│"), ("│","│"), ("╫")], # │-------│ ├┤ ╫
103
+ "FAILURE" : [("│","═","│"), ("╞","╡"), ("╬")], # │═══════│ ╞╡ ╬
104
+ "RUNNING" : [("│","᠁","᠁"), ("│","᠁"), ("᠁")], # │᠁᠁᠁᠁᠁᠁᠁᠁ │᠁ ᠁
105
+ "RUNNING:TO_FAILURE" : [("│","᠁","᠁"), ("│","᠁"), ("᠁")], # │᠁᠁᠁᠁᠁᠁᠁᠁ │᠁ ᠁ Same as RUNNING
106
+ "RUNNING:TO_SUCCESS" : [("│","᠁","᠁"), ("│","᠁"), ("᠁")], # │᠁᠁᠁᠁᠁᠁᠁᠁ │᠁ ᠁ Same as RUNNING
107
+ "RUNNING:TO_SKIPPED" : [("│","᠁","᠁"), ("│","᠁"), ("᠁")], # │᠁᠁᠁᠁᠁᠁᠁᠁ │᠁ ᠁ Same as RUNNING
108
+ "CANCELED" : [("│","᠁","᠁"), ("│","᠁"), ("᠁")], # │᠁᠁᠁᠁᠁᠁᠁᠁ │᠁ ᠁ Same as RUNNING
109
+ "SKIPPED" : [("│","≫","|"), ("│","≫"), ("≫")] # │≫≫≫≫≫≫≫| |≫ ≫
110
+ }
111
+ else:
112
+ symbols = {
113
+ "SUCCESS" : [("|","-","|"), ("|","|"), ("H")], # |-------| || H
114
+ "FAILURE" : [("|","=","|"), ("X","X"), ("#")], # |=======| XX #
115
+ "RUNNING" : [("|","~","~"), ("|","~"), ("~")], # |~~~~~~~~ |~ ~
116
+ "RUNNING:TO_FAILURE" : [("|","~","~"), ("|","~"), ("~")], # |~~~~~~~~ |~ ~ Same as RUNNING
117
+ "RUNNING:TO_SUCCESS" : [("|","~","~"), ("|","~"), ("~")], # |~~~~~~~~ |~ ~ Same as RUNNING
118
+ "RUNNING:TO_SKIPPED" : [("|","~","~"), ("|","~"), ("~")], # |~~~~~~~~ |~ ~ Same as RUNNING
119
+ "CANCELED" : [("|","~","~"), ("|","~"), ("~")], # |~~~~~~~~ |~ ~ Same as RUNNING
120
+ "SKIPPED" : [("|",">","|"), ("|",">"), (">")] # |>>>>>>>| |> >
121
+ }
122
+
123
+ tokens_super = tokens
124
+
125
+ timeline = [None]*output_width # store all instants where jobs start and stop
126
+
127
+ output_buffer.append("")
128
+
129
+
130
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
131
+ # Sort from first to last (to have some ordering when assigning legend aliases)
132
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
133
+
134
+ names_sorted = [(x[0],x[1]) for element in stacks for x in element]
135
+ names_sorted.sort(key = lambda x:x[1])
136
+
137
+ legend = {}
138
+ reverse_legend = {}
139
+ token_index = 0
140
+
141
+ for job_name, _ in names_sorted:
142
+ if job_name in legend.values():
143
+ continue
144
+
145
+ if token_index >= len(tokens):
146
+ reverse_legend[job_name] = (".", "˙")
147
+ continue
148
+
149
+ if tokens[token_index] in legend.keys():
150
+ legend[tokens[token_index]].append(job_name)
151
+ else:
152
+ legend[tokens[token_index]] = [job_name]
153
+
154
+ reverse_legend[job_name] = (tokens[token_index], tokens_super[token_index])
155
+ token_index += 1
156
+
157
+ # If the name of all jobs is just once character, use it instead of a legend.
158
+ #
159
+ if max([len(x) for jobs_names in legend.values() for x in jobs_names]) == 1:
160
+ legend = {}
161
+
162
+
163
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
164
+ # Print each stack on its own line
165
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
166
+
167
+ for stack in stacks:
168
+
169
+ line = [" "] *output_width # main line for each stack
170
+ extra = [" "] *output_width # auxiliary per stack line if superscripts are needed
171
+
172
+ # We want to use one single line to print all segments, like this:
173
+ #
174
+ # |---A---|-----C------| |-D-|
175
+ #
176
+ # ...but this might not always be possible. In particular, where would we print the "token"
177
+ # ("A", "B", ...) when the segment is two short?
178
+ #
179
+ # - 3 columns segment --> no problem --> |A|
180
+ # - 2 columns segment --> ? ------------> ||
181
+ # - 1 column segment ---> ? ------------> ‖
182
+ #
183
+ # For both the 2 columns and 1 column cases we will add an extra line below with the token
184
+ # character (in superscript form), like this:
185
+ #
186
+ # || ‖
187
+ # ᴬ ᴮ
188
+ #
189
+ # Note that this extra line will only be needed if for the current stack there is at least
190
+ # one "very small" segment, otherwise we won't print it.
191
+
192
+ for job in stack:
193
+ start = int(job[1] / 100 * (output_width-1))
194
+ start2 = int(job[2] / 100 * (output_width-1))
195
+ stop = int(job[3] / 100 * (output_width-1))
196
+ status = job[4]
197
+ detached = job[5].upper() == "TRUE"
198
+
199
+ if status == "QUEUED":
200
+ status = "RUNNING" # QUEUED jobs are like RUNNING jobs in respect to selecting the
201
+ # appropiate ascii/unicode symbol to use.
202
+
203
+ if legend:
204
+ X = reverse_legend[job[0]][0]
205
+ X_super = reverse_legend[job[0]][1]
206
+ else:
207
+ X = job[0]
208
+ X_super = job[0]
209
+
210
+ # Depending on the length of the segment and the status of the associated job execution,
211
+ # use a different type of ASCII/unicode representation
212
+ #
213
+ if start == stop:
214
+ # Ultra short segment (1 char wide)
215
+ #
216
+ line[start] = symbols[status][2][0]
217
+ extra[start] = X_super
218
+ timeline[start] = job[1]
219
+
220
+ elif start == stop-1:
221
+ # Short segment (2 chars wide)
222
+ #
223
+ line[start] = symbols[status][1][0]
224
+ line[stop] = symbols[status][1][1]
225
+ extra[start] = X_super
226
+ timeline[start] = job[1]
227
+
228
+ else:
229
+ # Regular segment longer than 2 chars
230
+
231
+ line[start] = symbols[status][0][0]
232
+ line[stop] = symbols[status][0][2]
233
+
234
+ if detached:
235
+ for i in range (start+1, stop, 2):
236
+ line[i] = symbols[status][0][1]
237
+ else:
238
+ for i in range (start+1, stop, 1):
239
+ line[i] = symbols[status][0][1]
240
+
241
+ for i in range (start+1, start2, 1):
242
+ line[i] = "."
243
+
244
+ line[start + int((stop-start)/2)] = X
245
+
246
+ timeline[start] = job[1]
247
+ timeline[stop] = job[3]
248
+
249
+ token_index += 1
250
+
251
+
252
+ # Print the ASCII/unicode character to stdout
253
+ #
254
+ output_buffer.append("".join(line))
255
+
256
+ if "".join(extra).strip():
257
+ # If there is at least one non empty character in the extra line, print it
258
+ output_buffer.append("".join(extra))
259
+
260
+
261
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
262
+ # Print the global "t" horizontal axis
263
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
264
+
265
+ line = [" "] * output_width
266
+ extra = []
267
+
268
+ if secs_per_unit > 0:
269
+ if scale == "auto":
270
+ if secs_per_unit * 100 < 60*60: # If the total time spawn is less that 1 hour, use
271
+ scale = "minutes" # "minutes" precision (ie. print "Xm:Ys")
272
+ else:
273
+ scale = "hours" # Otherwise, use "hours" precision ("Xh:Ym")
274
+
275
+ for i in range(output_width):
276
+
277
+ # Select timeline ASCII symbol
278
+ #
279
+ if i == 0:
280
+ line[i] = "|"
281
+
282
+ elif i == output_width-1:
283
+ if unicode_supported:
284
+ line[i] = "▶" # unicode. TODO: Search a better fit
285
+ else:
286
+ line[i] = ">"
287
+
288
+ elif timeline[i]:
289
+ line[i] = "*"
290
+
291
+ else:
292
+ if unicode_supported:
293
+ line[i] = "⎻" # unicode
294
+ else:
295
+ line[i] = "-"
296
+
297
+ # Figure out whether we need to print a timestamp or not
298
+ #
299
+ if timeline[i]:
300
+ now = timeline[i] # Event timestmp
301
+ else:
302
+ continue # No timestamps needed
303
+
304
+
305
+ # Print timestamp using auxiliary lines below the horizontal axis
306
+ #
307
+ if secs_per_unit > 0:
308
+ seconds = now * secs_per_unit
309
+
310
+ if scale == "minutes":
311
+ minutes = int(seconds//60)
312
+ seconds = seconds%60
313
+
314
+ if minutes > 0:
315
+ timestamp = f"{minutes}m:{round(seconds):02}s"
316
+ else:
317
+ timestamp = f"{round(seconds)}s"
318
+
319
+ elif scale == "hours":
320
+ hours = int(seconds//(60*60))
321
+ minutes = seconds%(60*60)/60
322
+
323
+ if hours > 0:
324
+ timestamp = f"{hours}h:{round(minutes):02}m"
325
+ else:
326
+ timestamp = f"{round(minutes)}m"
327
+
328
+ else:
329
+ timestamp = str(now)
330
+
331
+ if i + len(timestamp) >= output_width:
332
+ i = output_width - len(timestamp)
333
+
334
+ for x in extra:
335
+ if x[i-1] == " ":
336
+ x[i:i+len(timestamp)] = list(timestamp)
337
+ break
338
+ else:
339
+ extra.append([" "]*output_width)
340
+
341
+ x = extra[-1]
342
+ x[i:i+len(timestamp)] = list(timestamp)
343
+
344
+ output_buffer.append("")
345
+ output_buffer.append("".join(line))
346
+ for x in extra:
347
+ output_buffer.append("".join(x))
348
+
349
+
350
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
351
+ # Print the legend
352
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
353
+
354
+ output_buffer.append("")
355
+
356
+ if legend:
357
+ output_buffer.append("LEGEND:")
358
+ for k in sorted(legend.keys()):
359
+ output_buffer.append(f" {k} : {', '.join(legend[k])}")
360
+
361
+ output_buffer.append("")
362
+
363
+
364
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
365
+ # Decide what to return
366
+ #~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
367
+
368
+ if stdout == "stdout":
369
+ for line in output_buffer:
370
+ print(line)
371
+
372
+ return None
373
+
374
+ elif stdout.startswith("file@"):
375
+ target = stdout[5:]
376
+ open(target, "w").write("\n".join(output_buffer))
377
+
378
+ return None
379
+
380
+ elif stdout == "buffer":
381
+ return "\n".join(output_buffer)
382
+
383
+
384
+
385
+ ####################################################################################################
386
+ # API
387
+ ####################################################################################################
388
+
389
+ def draw_timeline(db_proxy, mode="ascii:150:stdout"):
390
+ """
391
+ Check @ref Pipeline.draw_timeline for details
392
+ """
393
+
394
+ pipeline_data = [f"Pipeline {db_proxy.pipeline.id}",
395
+ db_proxy.pipeline.metadata.start_time,
396
+ db_proxy.pipeline.metadata.stop_time,
397
+ db_proxy.pipeline.metadata.current_state]
398
+
399
+ jobs_data = []
400
+
401
+
402
+ # Obtain a topologically sorted list of jobs and their properties
403
+ #
404
+ for job_name in utils.topological_sort(db_proxy.job_dependencies):
405
+
406
+ for job_db in db_proxy.jobs:
407
+ if job_db.name == job_name:
408
+ break
409
+ else:
410
+ print("")
411
+ print("ERROR: Corrupted pipeline in the database?")
412
+ print(f"ERROR: Job <{job_name}> is listed in the job dependencies list but does not exist if the pipeline")
413
+ print("")
414
+
415
+ return -1
416
+
417
+ if len(job_db.metadata.start_times) > 0:
418
+ for i, start in enumerate(job_db.metadata.start_times):
419
+
420
+ if len(job_db.metadata.start2_times) > i:
421
+ start2 = job_db.metadata.start2_times[i]
422
+ else:
423
+ start2 = None
424
+
425
+ if len(job_db.metadata.stop_times) > i:
426
+ stop = job_db.metadata.stop_times[i]
427
+ state = job_db.metadata.results[i]
428
+ else:
429
+ stop = None
430
+ state = job_db.metadata.current_state
431
+
432
+ jobs_data.append([job_db.name, start, start2, stop, state, job_db.detached])
433
+ else:
434
+ start = None
435
+ start2 = None
436
+ stop = None
437
+ state = job_db.metadata.current_state
438
+ jobs_data.append([job_db.name, start, start2, stop, state])
439
+
440
+
441
+ # Right now timestamps are absolute, lets normalize them to 100 using the total pipeline
442
+ # duration as reference. In case the pipeline has not finished executing yet, we will use the
443
+ # biggest time (start or stop) registered in any job.
444
+ #
445
+ time_min = pipeline_data[1]
446
+
447
+ if pipeline_data[2]:
448
+ time_max = pipeline_data[2]
449
+ else:
450
+ all_times = [x for x in [t for job in jobs_data for t in [job[1], job[2], job[3]]]
451
+ if x is not None]
452
+
453
+ if len(all_times) == 0:
454
+ return
455
+
456
+ time_max = max(all_times)
457
+
458
+ if pipeline_data[3] == "RUNNING":
459
+ time_max = datetime.datetime.utcnow()
460
+
461
+ for job in jobs_data:
462
+ if job[1] is None:
463
+ continue
464
+
465
+ job[1] = round((job[1]-time_min)/(time_max-time_min)*100)
466
+
467
+ if job[2] is None:
468
+ job[2] = 100 # Queued job. Still waiting.
469
+ else:
470
+ job[2] = round((job[2]-time_min)/(time_max-time_min)*100)
471
+
472
+ if job[3] is None:
473
+ job[3] = 100 # Unfinished job. Still running.
474
+ else:
475
+ job[3] = round((job[3]-time_min)/(time_max-time_min)*100)
476
+
477
+
478
+ # We now have something like this:
479
+ #
480
+ # jobs_data = [ ["A", 0, 0, 10, "SUCCESS", "False"],
481
+ # ["B", 0, 10, 20, "SUCCESS", "False"],
482
+ # ["C", 11, 13, None, "RUNNING", "False"],
483
+ # ...]
484
+ #
485
+ # ...which would result is something like this:
486
+ #
487
+ # |----A---||..---C-----... (stack level = 0)
488
+ # |----B-------------| (stack level = 1)
489
+ #
490
+ # We will start drawing them from left to right, according to the topological ordering, keeping
491
+ # track of overlaps to decide which "stack level" to use for each job
492
+
493
+ stacks = [] # For each stack level, all jobs it contains
494
+
495
+ for job in jobs_data:
496
+
497
+ if job[1] is None:
498
+ continue # Job has not started
499
+
500
+ for stack in stacks:
501
+ if _job_fits(job, stack):
502
+ stack.append(job)
503
+ break
504
+ else:
505
+ # We need a new stack level
506
+ #
507
+ stacks.append([job])
508
+
509
+
510
+ # For convenience, sort each stack level by start time
511
+ #
512
+ for stack in stacks:
513
+ stack.sort(key = lambda x:x[1])
514
+
515
+ # We now have something like this:
516
+ #
517
+ # stacks = [ [["A", 0, 0, 10, "SUCCESS", "False"), ("C", 11, 13, None, "RUNNING", "False")], # level 0
518
+ # [["B", 0, 10, 20, "SUCCESS", "False)] # level 1
519
+ # ]
520
+
521
+ if mode.startswith("ascii"):
522
+ width = int(mode.split(":")[1])
523
+ stdout = mode.split(":")[2]
524
+ scale = mode.split(":")[3]
525
+
526
+ return _print_ascii(stacks,
527
+ secs_per_unit=int((time_max-time_min).seconds)/100,
528
+ scale=scale,
529
+ output_width=width,
530
+ unicode_supported=False,
531
+ stdout=stdout)
532
+
533
+ elif mode.startswith("unicode"):
534
+ width = int(mode.split(":")[1])
535
+ stdout = mode.split(":")[2]
536
+ scale = mode.split(":")[3]
537
+
538
+ return _print_ascii(stacks,
539
+ secs_per_unit=int((time_max-time_min).seconds)/100,
540
+ scale=scale,
541
+ output_width=width,
542
+ unicode_supported=True,
543
+ stdout=stdout)
544
+