pipeforge 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipeforge/__init__.py +1201 -0
- pipeforge/__main__.py +208 -0
- pipeforge/_internal/__init__.py +0 -0
- pipeforge/_internal/core_loop.py +700 -0
- pipeforge/_internal/database.py +515 -0
- pipeforge/_internal/drawing.py +544 -0
- pipeforge/_internal/file_loader.py +654 -0
- pipeforge/_internal/inspector.py +555 -0
- pipeforge/_internal/params.py +148 -0
- pipeforge/_internal/pipeline.py +212 -0
- pipeforge/_internal/script.py +1004 -0
- pipeforge/_internal/utils.py +122 -0
- pipeforge/examples/hello_jenkins.toml +16 -0
- pipeforge/examples/hello_world.toml +67 -0
- pipeforge/examples/merge_pull_request.toml +240 -0
- pipeforge/examples/multi_pipeline.toml +93 -0
- pipeforge/examples/multi_pipeline_2.toml +54 -0
- pipeforge/examples/nightly.toml +19 -0
- pipeforge/examples/retries.toml +43 -0
- pipeforge/examples/scripts/hello_world__get_purpose_in_life.py +34 -0
- pipeforge/examples/scripts/hello_world__print_summary.py +27 -0
- pipeforge/examples/scripts/hello_world__repeat_purpose.py +33 -0
- pipeforge/examples/scripts/retry.py +25 -0
- pipeforge/examples/scripts/timeout.py +25 -0
- pipeforge/examples/timeout.toml +37 -0
- pipeforge-1.0.0.dist-info/METADATA +198 -0
- pipeforge-1.0.0.dist-info/RECORD +29 -0
- pipeforge-1.0.0.dist-info/WHEEL +4 -0
- pipeforge-1.0.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,700 @@
|
|
|
1
|
+
# vim: colorcolumn=101 textwidth=100
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
import sys
|
|
5
|
+
import signal
|
|
6
|
+
import datetime
|
|
7
|
+
import threading
|
|
8
|
+
|
|
9
|
+
from .utils import log, topological_sort # Local module
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
####################################################################################################
|
|
14
|
+
# Globals
|
|
15
|
+
####################################################################################################
|
|
16
|
+
|
|
17
|
+
external_interrupt_received = False
|
|
18
|
+
sleep_event = threading.Event()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
####################################################################################################
|
|
23
|
+
# Auxiliary functions
|
|
24
|
+
####################################################################################################
|
|
25
|
+
|
|
26
|
+
def _time_now(real_time, cycle):
|
|
27
|
+
"""
|
|
28
|
+
Return a datetime.datetime() object representing the time when this function was called.
|
|
29
|
+
|
|
30
|
+
@param real_time: if True, this function returns the current UTC time, which is what you want
|
|
31
|
+
99% of the times.
|
|
32
|
+
If False, the returned object represents a fixed date in the past *plus* some minutes (the
|
|
33
|
+
amount specified in @ref cycle). This is useful when running in "simulation" mode, for
|
|
34
|
+
example in unit tests.
|
|
35
|
+
|
|
36
|
+
@param cycle: Number of minutes to add to the datetime.datetime() object that is going to be
|
|
37
|
+
returned when @ref real_time is False. This parameter is ignored when @ref real_time is
|
|
38
|
+
True.
|
|
39
|
+
|
|
40
|
+
@return a datetime.datetime() object that can represent either the current UTC time or a fixed
|
|
41
|
+
date in the past plus some minutes.
|
|
42
|
+
|
|
43
|
+
NOTE: The "fixed time in the past" is not really relevant, as in simulation mode we are only
|
|
44
|
+
interested in time deltas, but if case you are wondering, it's the 1st of July of 1961 at 19:45.
|
|
45
|
+
I will let you try to figure out what happened at that time :)
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
if real_time:
|
|
49
|
+
return datetime.datetime.utcnow()
|
|
50
|
+
else:
|
|
51
|
+
return datetime.datetime(1961, 7, 1, 19, 45) + datetime.timedelta(minutes=cycle)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _timeout_expired(start, stop, timeout):
|
|
55
|
+
"""
|
|
56
|
+
Check if the time ellapsed from "start" to "stop" is bigger than "timeout"
|
|
57
|
+
|
|
58
|
+
@param start: datetime.datetime() object representing the period start point
|
|
59
|
+
|
|
60
|
+
@param stop: datetime.datetime() object representing the period end point
|
|
61
|
+
|
|
62
|
+
@param timeout: a string representing an amount of time in "natural language". Examples: "5
|
|
63
|
+
seconds", "2 minutes", etc...
|
|
64
|
+
|
|
65
|
+
@return True if the [start,stop] period is bigger than "timeout"
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
if timeout == "N/A":
|
|
69
|
+
return False # Special case
|
|
70
|
+
|
|
71
|
+
window_length_in_seconds = (stop-start).days * 24 * 60 * 60 + (stop-start).seconds
|
|
72
|
+
|
|
73
|
+
if timeout.split(" ")[1].startswith("second"):
|
|
74
|
+
timeout_in_seconds = int(timeout.split(" ")[0]) * 1
|
|
75
|
+
|
|
76
|
+
elif timeout.split(" ")[1].startswith("minute"):
|
|
77
|
+
timeout_in_seconds = int(timeout.split(" ")[0]) * 60
|
|
78
|
+
|
|
79
|
+
elif timeout.split(" ")[1].startswith("hour"):
|
|
80
|
+
timeout_in_seconds = int(timeout.split(" ")[0]) * 60 * 60
|
|
81
|
+
|
|
82
|
+
if window_length_in_seconds > timeout_in_seconds:
|
|
83
|
+
return True
|
|
84
|
+
else:
|
|
85
|
+
return False
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _external_interrupt_handler(sig, frame):
|
|
89
|
+
global external_interrupt_received
|
|
90
|
+
global sleep_event
|
|
91
|
+
|
|
92
|
+
external_interrupt_received = True
|
|
93
|
+
sleep_event.set()
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
####################################################################################################
|
|
98
|
+
# API
|
|
99
|
+
####################################################################################################
|
|
100
|
+
|
|
101
|
+
def run_pipeline(db_proxy, script_manager, heartbeat):
|
|
102
|
+
"""
|
|
103
|
+
Run all the jobs that make up a pipeline in order, respecting their dependencies.
|
|
104
|
+
|
|
105
|
+
@param db_proxy: object of type "database.DBProxy()" previously initialized to hold references
|
|
106
|
+
to the pipeline we want to run and its associated jobs.
|
|
107
|
+
|
|
108
|
+
@param script_manager: object of type "script.ScriptManager()" previously initialized that will
|
|
109
|
+
be used to start, query and stop a job script.
|
|
110
|
+
|
|
111
|
+
@param heartbeat: number of seconds between each "poll cycle", where the status of currently
|
|
112
|
+
running jobs is checked to decided whether the pipeline should be terminated and/or new jobs
|
|
113
|
+
triggered.
|
|
114
|
+
|
|
115
|
+
@return "SUCCESS" if all jobs (excluding those with the "detached" property set) finished
|
|
116
|
+
executing and reported no error. Otherwise:
|
|
117
|
+
|
|
118
|
+
- If no jobs failed, return "CANCELED".
|
|
119
|
+
|
|
120
|
+
- Otherwise:
|
|
121
|
+
|
|
122
|
+
- If one of the jobs that failed triggered a new pipeline (by using the "on_failure"
|
|
123
|
+
property), return "TRIGGER:<name_of_the_pipeline_to_trigger>" (example:
|
|
124
|
+
"TRIGGER:secondary")
|
|
125
|
+
|
|
126
|
+
- Otherwise, return "FAILURE"
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
# Install the handler for when the user wants to externally cancel the pipeline execution.
|
|
130
|
+
#
|
|
131
|
+
# - SIGINT is received when typing Ctrl+C on the terminal pipeforge is running on.
|
|
132
|
+
# - SIGTERM is not as common, but can be used in some scenarios (for example when pipeforge is
|
|
133
|
+
# running as a Jenkins job and the user clicks on the "X" button to cancel the job
|
|
134
|
+
# execution)
|
|
135
|
+
#
|
|
136
|
+
global external_interrupt_received
|
|
137
|
+
|
|
138
|
+
signal.signal(signal.SIGINT, _external_interrupt_handler)
|
|
139
|
+
signal.signal(signal.SIGTERM, _external_interrupt_handler)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
# Print a topological sorting of jobs based on their dependencies (just for fun)
|
|
143
|
+
#
|
|
144
|
+
log("DEBUG", "")
|
|
145
|
+
log("DEBUG", "Topological sorting of jobs that make up the pipeline:")
|
|
146
|
+
for x in topological_sort(db_proxy.job_dependencies):
|
|
147
|
+
log("DEBUG", f" - {x}")
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# Find out the longest job name (so that later we use this value to generate pretty/aligned) log
|
|
151
|
+
# messages.
|
|
152
|
+
#
|
|
153
|
+
max_job_name_length = max([len(job_db.name) for job_db in db_proxy.jobs]) + 2
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
# Convenience variables
|
|
157
|
+
#
|
|
158
|
+
keep_going = True
|
|
159
|
+
cycle = 0
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
# Update/set pipeline start time
|
|
163
|
+
#
|
|
164
|
+
now = _time_now(heartbeat>0, cycle)
|
|
165
|
+
|
|
166
|
+
db_proxy.pipeline.metadata.current_state = "RUNNING"
|
|
167
|
+
db_proxy.pipeline.metadata.start_time = now
|
|
168
|
+
db_proxy.pipeline.save()
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# Start the loop that will eventually execute all jobs in the pipeline
|
|
172
|
+
#
|
|
173
|
+
cycle = -1
|
|
174
|
+
trigger_pipeline = None
|
|
175
|
+
|
|
176
|
+
log("NORMAL", "")
|
|
177
|
+
|
|
178
|
+
while keep_going:
|
|
179
|
+
|
|
180
|
+
cycle += 1
|
|
181
|
+
before = now
|
|
182
|
+
now = _time_now(heartbeat>0, cycle)
|
|
183
|
+
normal_logs = 0
|
|
184
|
+
|
|
185
|
+
log("DEBUG", "")
|
|
186
|
+
log("DEBUG", "")
|
|
187
|
+
log("DEBUG", "")
|
|
188
|
+
log("DEBUG", f"Polling cycle #{cycle} starts")
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# Update the status of running jobs (in case any of them has finished or timed out)
|
|
192
|
+
#
|
|
193
|
+
log("DEBUG", "")
|
|
194
|
+
log("DEBUG", " - Checking status of running jobs...")
|
|
195
|
+
|
|
196
|
+
for job_db in db_proxy.jobs:
|
|
197
|
+
|
|
198
|
+
if job_db.metadata.current_state in ["QUEUED", "RUNNING"] or \
|
|
199
|
+
job_db.metadata.current_state.startswith("RUNNING:"):
|
|
200
|
+
|
|
201
|
+
if job_db.metadata.current_state == "RUNNING" and \
|
|
202
|
+
_timeout_expired(job_db.metadata.start2_times[-1], now, job_db.timeout):
|
|
203
|
+
script_manager.stop(job_db.metadata.executions_id[-1])
|
|
204
|
+
new_state = "FAILURE"
|
|
205
|
+
extra = " (timeout)"
|
|
206
|
+
|
|
207
|
+
elif job_db.metadata.current_state == "RUNNING:TO_FAILURE":
|
|
208
|
+
new_state = "FAILURE"
|
|
209
|
+
extra = ""
|
|
210
|
+
|
|
211
|
+
# Forcefully exhaust all retries
|
|
212
|
+
job_db.retries = "0"
|
|
213
|
+
|
|
214
|
+
elif job_db.metadata.current_state == "RUNNING:TO_SUCCESS":
|
|
215
|
+
new_state = "SUCCESS"
|
|
216
|
+
extra = ""
|
|
217
|
+
|
|
218
|
+
elif job_db.metadata.current_state == "RUNNING:TO_SKIPPED":
|
|
219
|
+
new_state = "SKIPPED"
|
|
220
|
+
extra = ""
|
|
221
|
+
|
|
222
|
+
else:
|
|
223
|
+
new_state = script_manager.query(job_db.metadata.executions_id[-1])
|
|
224
|
+
extra = ""
|
|
225
|
+
|
|
226
|
+
if new_state == "NOT FOUND":
|
|
227
|
+
#
|
|
228
|
+
# This indicates a fatal error. Let's act as if the job was canceled
|
|
229
|
+
|
|
230
|
+
new_state = "CANCELED"
|
|
231
|
+
extra = "executor not found!"
|
|
232
|
+
|
|
233
|
+
log("DEBUG", f" - Job {'<'+job_db.name+'>':{max_job_name_length}} : {new_state:8}{extra} {'(detached)' if job_db.detached == 'true' else ''}")
|
|
234
|
+
|
|
235
|
+
if job_db.metadata.current_state == new_state:
|
|
236
|
+
# Job still queued/running
|
|
237
|
+
continue
|
|
238
|
+
|
|
239
|
+
log("NORMAL", f"Job {'<'+job_db.name+'>':{max_job_name_length}} : {job_db.metadata.current_state:8} --> {new_state:8}{extra} {'(detached)' if job_db.detached == 'true' else ''} {script_manager.exe_uri(job_db.metadata.executions_id[-1])}")
|
|
240
|
+
normal_logs += 1
|
|
241
|
+
|
|
242
|
+
if job_db.metadata.current_state == "QUEUED" or \
|
|
243
|
+
job_db.metadata.current_state.startswith("RUNNING:"):
|
|
244
|
+
|
|
245
|
+
job_db.metadata.start2_times.append(before)
|
|
246
|
+
|
|
247
|
+
job_db.metadata.current_state = new_state
|
|
248
|
+
job_db.metadata.executions_uri[-1] = script_manager.exe_uri(job_db.metadata.executions_id[-1])
|
|
249
|
+
job_db.save()
|
|
250
|
+
|
|
251
|
+
if new_state == "FAILURE":
|
|
252
|
+
if job_db.detached == "false":
|
|
253
|
+
log("DEBUG", f" - Decrementing number of retries. New value = <{job_db.retries}>")
|
|
254
|
+
if job_db.on_failure == "continue":
|
|
255
|
+
log("NORMAL", f"`--> Retries left = {job_db.retries} (that's ok, when this job fails we have been told to ignore it)")
|
|
256
|
+
normal_logs += 1
|
|
257
|
+
else:
|
|
258
|
+
log("NORMAL", f"`--> Retries left = {job_db.retries}")
|
|
259
|
+
normal_logs += 1
|
|
260
|
+
job_db.retries = str(int(job_db.retries) - 1)
|
|
261
|
+
|
|
262
|
+
job_db.metadata.stop_times.append(now)
|
|
263
|
+
job_db.metadata.results.append("FAILURE")
|
|
264
|
+
|
|
265
|
+
job_db.save()
|
|
266
|
+
|
|
267
|
+
elif new_state in ["SUCCESS", "SKIPPED"]:
|
|
268
|
+
job_db.metadata.stop_times.append(now)
|
|
269
|
+
job_db.metadata.results.append(new_state)
|
|
270
|
+
|
|
271
|
+
job_db.save()
|
|
272
|
+
|
|
273
|
+
if job_db.detached == "false":
|
|
274
|
+
log("DEBUG", " - Checking output parameters:")
|
|
275
|
+
|
|
276
|
+
if job_db.output:
|
|
277
|
+
max_param_out_length = max([len(k) for k in job_db.output.keys()])
|
|
278
|
+
|
|
279
|
+
job_db.reload()
|
|
280
|
+
|
|
281
|
+
for k,v in job_db.output.items():
|
|
282
|
+
message = f" - {k:{max_param_out_length}} = {v}"
|
|
283
|
+
if v == "?":
|
|
284
|
+
message += f" => Parameter <{k}> was not set by job <{job_db.name}> script (<{job_db.script}>)"
|
|
285
|
+
log("DEBUG", message)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
# Check if the whole pipeline needs to be stopped because of a failed job
|
|
289
|
+
#
|
|
290
|
+
log("DEBUG", "")
|
|
291
|
+
log("DEBUG", " - Checking for pipeline stop/restart conditions...")
|
|
292
|
+
|
|
293
|
+
for job_db in db_proxy.jobs:
|
|
294
|
+
|
|
295
|
+
if job_db.metadata.current_state == "CANCELED" or external_interrupt_received is True:
|
|
296
|
+
|
|
297
|
+
# When a job is canceled, we assume the whole pipeline must be stopped
|
|
298
|
+
# Stop all running (even detached!) jobs
|
|
299
|
+
#
|
|
300
|
+
for job_db2 in db_proxy.jobs:
|
|
301
|
+
if job_db2.metadata.current_state == "WAITING":
|
|
302
|
+
|
|
303
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is WAITING. We are going to change its state to CANCELED.")
|
|
304
|
+
|
|
305
|
+
job_db2.metadata.executions_id.append("<None>")
|
|
306
|
+
job_db2.metadata.executions_uri.append("<None>")
|
|
307
|
+
|
|
308
|
+
job_db2.metadata.stop_times.append(now)
|
|
309
|
+
job_db2.metadata.current_state = "CANCELED"
|
|
310
|
+
job_db2.metadata.results.append("CANCELED")
|
|
311
|
+
job_db2.save()
|
|
312
|
+
|
|
313
|
+
elif job_db2.metadata.current_state in ["QUEUED", "RUNNING"]:
|
|
314
|
+
|
|
315
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is {job_db2.metadata.current_state}. Killing it...")
|
|
316
|
+
|
|
317
|
+
script_manager.stop(job_db2.metadata.executions_id[-1])
|
|
318
|
+
|
|
319
|
+
job_db2.metadata.stop_times.append(now)
|
|
320
|
+
job_db2.metadata.current_state = "CANCELED"
|
|
321
|
+
job_db2.metadata.results.append("CANCELED")
|
|
322
|
+
job_db2.save()
|
|
323
|
+
|
|
324
|
+
keep_going = False
|
|
325
|
+
break
|
|
326
|
+
|
|
327
|
+
if job_db.metadata.current_state != "FAILURE": continue
|
|
328
|
+
if job_db.detached == "true" : continue # Detached jobs can fail
|
|
329
|
+
if int(job_db.retries) >= 0 : continue # No retries left
|
|
330
|
+
|
|
331
|
+
log("DEBUG", f" - Job {'<'+job_db.name+'>':{max_job_name_length}} has exhausted its retries.")
|
|
332
|
+
|
|
333
|
+
if job_db.on_failure == "continue":
|
|
334
|
+
log("DEBUG", " - Its \"on_failure\" property is set to \"continue\". Nothing to do...")
|
|
335
|
+
continue
|
|
336
|
+
|
|
337
|
+
if job_db.on_failure == "stop pipeline" or \
|
|
338
|
+
job_db.on_failure.startswith("trigger pipeline"):
|
|
339
|
+
log("DEBUG", f" - Its \"on_failure\" property is set to \"{job_db.on_failure}\". Let's kill all other RUNNING jobs (excluding 'detached' ones):")
|
|
340
|
+
log("NORMAL", f"Job {'<'+job_db.name+'>'} has exhausted its retries. This is critial. Stop pipeline.")
|
|
341
|
+
normal_logs += 1
|
|
342
|
+
|
|
343
|
+
# Stop all running (and not detached) jobs
|
|
344
|
+
#
|
|
345
|
+
for job_db2 in db_proxy.jobs:
|
|
346
|
+
if job_db2.metadata.current_state == "WAITING":
|
|
347
|
+
|
|
348
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is WAITING. We are going to change its state to CANCELED.")
|
|
349
|
+
|
|
350
|
+
job_db2.metadata.executions_id.append("<None>")
|
|
351
|
+
job_db2.metadata.executions_uri.append("<None>")
|
|
352
|
+
|
|
353
|
+
job_db2.metadata.stop_times.append(now)
|
|
354
|
+
job_db2.metadata.current_state = "CANCELED"
|
|
355
|
+
job_db2.metadata.results.append("CANCELED")
|
|
356
|
+
job_db2.save()
|
|
357
|
+
|
|
358
|
+
elif job_db2.metadata.current_state in ["QUEUED", "RUNNING"]:
|
|
359
|
+
|
|
360
|
+
if job_db2.detached == "true":
|
|
361
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is {job_db2.metadata.current_state} *but* its \"detached\" property is set to \"true\". Do not kill this job.")
|
|
362
|
+
else:
|
|
363
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is {job_db2.metadata.current_state}. Killing it...")
|
|
364
|
+
|
|
365
|
+
script_manager.stop(job_db2.metadata.executions_id[-1])
|
|
366
|
+
|
|
367
|
+
job_db2.metadata.stop_times.append(now)
|
|
368
|
+
job_db2.metadata.current_state = "CANCELED"
|
|
369
|
+
job_db2.metadata.results.append("CANCELED")
|
|
370
|
+
job_db2.save()
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
if job_db.on_failure.startswith("trigger pipeline"):
|
|
374
|
+
trigger_pipeline = job_db.on_failure.split(":")[1].strip()
|
|
375
|
+
|
|
376
|
+
keep_going = False
|
|
377
|
+
break
|
|
378
|
+
|
|
379
|
+
elif job_db.on_failure == "restart pipeline":
|
|
380
|
+
log("DEBUG", " - Its \"on_failure\" property is set to \"restart pipeline\". Let's kill all other RUNNING/QUEUED jobs (including 'detached' ones):")
|
|
381
|
+
|
|
382
|
+
# Stop all running jobs, including detached ones!
|
|
383
|
+
#
|
|
384
|
+
for job_db2 in db_proxy.jobs:
|
|
385
|
+
|
|
386
|
+
if job_db2.metadata.current_state == "WAITING":
|
|
387
|
+
|
|
388
|
+
job_db2.metadata.executions_id.append("<None>")
|
|
389
|
+
job_db2.metadata.executions_uri.append("<None>")
|
|
390
|
+
job_db2.metadata.stop_times.append(now)
|
|
391
|
+
|
|
392
|
+
elif job_db2.metadata.current_state in ["QUEUED", "RUNNING"]:
|
|
393
|
+
|
|
394
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is RUNNING. Killing it...")
|
|
395
|
+
|
|
396
|
+
script_manager.stop(job_db2.metadata.executions_id[-1])
|
|
397
|
+
job_db2.metadata.stop_times.append(now)
|
|
398
|
+
|
|
399
|
+
job_db2.metadata.current_state = "WAITING" # This will make the whole pipeline
|
|
400
|
+
job_db2.save() # start from the beginning
|
|
401
|
+
|
|
402
|
+
# TODO: Restore original values of DBProxy entries, add new start/stop
|
|
403
|
+
# timestamp to the pipeline object
|
|
404
|
+
|
|
405
|
+
break
|
|
406
|
+
|
|
407
|
+
elif job_db.on_failure.startswith("restart from"):
|
|
408
|
+
log("DEBUG", f" - Its \"on_failure\" property is set to \"{job_db.on_failure}\". Let's kill all other RUNNING jobs that depend on it (excluding 'detached' ones):")
|
|
409
|
+
|
|
410
|
+
# If we get here we want to restart from job X
|
|
411
|
+
|
|
412
|
+
# Stop all running jobs "downstream" the provided one, including detached ones!
|
|
413
|
+
#
|
|
414
|
+
start_from_job = job_db.on_failure.split(":")[1].strip()
|
|
415
|
+
|
|
416
|
+
def recursive_deps(node):
|
|
417
|
+
my_deps = []
|
|
418
|
+
for dep in db_proxy.job_dependencies[node]:
|
|
419
|
+
my_deps += [dep]
|
|
420
|
+
my_deps += recursive_deps(dep)
|
|
421
|
+
return my_deps
|
|
422
|
+
|
|
423
|
+
for job_db2 in db_proxy.jobs:
|
|
424
|
+
# Only change to "WATING" downstream jobs (and the just failed job too)
|
|
425
|
+
#
|
|
426
|
+
if job_db2.name == start_from_job or \
|
|
427
|
+
start_from_job in recursive_deps(job_db2.name):
|
|
428
|
+
|
|
429
|
+
if job_db2.metadata.current_state in ["QUEUED", "RUNNING"]:
|
|
430
|
+
|
|
431
|
+
log("DEBUG", f" - Job {'<'+job_db2.name+'>':{max_job_name_length}} is RUNNING, not detached and depends (directly or indirectly) on <{start_from_job}>. Killing it...")
|
|
432
|
+
|
|
433
|
+
script_manager.stop(job_db2.metadata.executions_id[-1])
|
|
434
|
+
|
|
435
|
+
job_db2.metadata.current_state = "WAITING"
|
|
436
|
+
job_db2.save()
|
|
437
|
+
|
|
438
|
+
# TODO: Restore original values of DBProxy entries, add new start/stop
|
|
439
|
+
# timestamp to the pipeline object
|
|
440
|
+
|
|
441
|
+
break
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
if not keep_going:
|
|
445
|
+
continue
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
# Check if any of the waiting (or failed) jobs can be started (or restarted)
|
|
449
|
+
#
|
|
450
|
+
log("DEBUG", "")
|
|
451
|
+
log("DEBUG", " - Checking for new job candidates to be started...")
|
|
452
|
+
|
|
453
|
+
for job_db in db_proxy.jobs:
|
|
454
|
+
|
|
455
|
+
if job_db.metadata.current_state == "WAITING" or \
|
|
456
|
+
(job_db.metadata.current_state == "FAILURE" and \
|
|
457
|
+
job_db.detached == "false" and \
|
|
458
|
+
int(job_db.retries) >= 0):
|
|
459
|
+
|
|
460
|
+
log("DEBUG", f" - Job {'<'+job_db.name+'>':{max_job_name_length}} is waiting to be started. Let's check its dependencies:")
|
|
461
|
+
|
|
462
|
+
# Check dependencies and run script
|
|
463
|
+
#
|
|
464
|
+
all_deps_ready = True
|
|
465
|
+
|
|
466
|
+
for dep in db_proxy.job_dependencies[job_db.name]:
|
|
467
|
+
|
|
468
|
+
status = [x for x in db_proxy.jobs if x.name == dep][0].metadata.current_state
|
|
469
|
+
|
|
470
|
+
log("DEBUG", f" - Dependency {'<'+dep+'>':{max_job_name_length}} status is {status}.")
|
|
471
|
+
|
|
472
|
+
if status in [ "WAITING", "RUNNING", "QUEUED" ] or \
|
|
473
|
+
status.startswith("RUNNING:"):
|
|
474
|
+
|
|
475
|
+
all_deps_ready = False
|
|
476
|
+
|
|
477
|
+
if all_deps_ready:
|
|
478
|
+
log("DEBUG", f" - All dependencies are done! We can start job <{job_db.name}>")
|
|
479
|
+
log("DEBUG", " - ...but first, let's update its input parameters:")
|
|
480
|
+
|
|
481
|
+
# Update input parameters with new values (ie. obtained from previously executed
|
|
482
|
+
# job output parameters)
|
|
483
|
+
#
|
|
484
|
+
max_param_in_length = max([len(k) for k in job_db.input.keys()] or [0])
|
|
485
|
+
|
|
486
|
+
one_question_mark_input = False
|
|
487
|
+
|
|
488
|
+
original_input = {}
|
|
489
|
+
for input_param, old_value in job_db.input.items():
|
|
490
|
+
original_input[input_param] = old_value
|
|
491
|
+
|
|
492
|
+
new_input = {}
|
|
493
|
+
for input_param, old_value in original_input.items():
|
|
494
|
+
new_value = old_value
|
|
495
|
+
|
|
496
|
+
for ref in re.findall("@{.*?}", old_value):
|
|
497
|
+
if "::" in ref[2:-1]:
|
|
498
|
+
job_ref, param_ref = ref[2:-1].split("::")
|
|
499
|
+
else:
|
|
500
|
+
job_ref = ref[2:-1]
|
|
501
|
+
param_ref = "<implicit_status>"
|
|
502
|
+
|
|
503
|
+
for job_db2 in db_proxy.jobs:
|
|
504
|
+
if job_db2.name == job_ref:
|
|
505
|
+
job_db2.reload()
|
|
506
|
+
|
|
507
|
+
if param_ref == "<implicit_status>":
|
|
508
|
+
# Resolve to the status of the referenced job
|
|
509
|
+
#
|
|
510
|
+
new_value = new_value.replace(
|
|
511
|
+
ref,
|
|
512
|
+
job_db2.metadata.current_state)
|
|
513
|
+
|
|
514
|
+
if job_db2.metadata.current_state != "SUCCESS":
|
|
515
|
+
# If the job did not succeed, we will behave in the same
|
|
516
|
+
# way as if this output parameter was not set, so that
|
|
517
|
+
# the "on_input_err" property of the next job kicks in.
|
|
518
|
+
one_question_mark_input = True
|
|
519
|
+
|
|
520
|
+
else:
|
|
521
|
+
# Resolve to the referenced output parameter of the
|
|
522
|
+
# referenced job
|
|
523
|
+
#
|
|
524
|
+
new_value = new_value.replace(
|
|
525
|
+
ref,
|
|
526
|
+
job_db2.output[param_ref])
|
|
527
|
+
|
|
528
|
+
if job_db2.output[param_ref] == "?":
|
|
529
|
+
one_question_mark_input = True
|
|
530
|
+
break
|
|
531
|
+
|
|
532
|
+
if new_value != old_value:
|
|
533
|
+
log("DEBUG", f" > {input_param:{max_param_in_length}} = {old_value} = {new_value}")
|
|
534
|
+
else:
|
|
535
|
+
log("DEBUG", f" > {input_param:{max_param_in_length}} = {old_value}")
|
|
536
|
+
|
|
537
|
+
new_input[input_param] = new_value
|
|
538
|
+
|
|
539
|
+
job_db.update(set__input=new_input)
|
|
540
|
+
|
|
541
|
+
if one_question_mark_input is False or job_db.on_input_err == "run":
|
|
542
|
+
|
|
543
|
+
try:
|
|
544
|
+
execution_id = script_manager.run(job_db.name,
|
|
545
|
+
job_db.script,
|
|
546
|
+
job_db.runner,
|
|
547
|
+
db_proxy.database_url + \
|
|
548
|
+
"#" + str(job_db.id))
|
|
549
|
+
except Exception as e:
|
|
550
|
+
log("ERROR", "")
|
|
551
|
+
log("ERROR", f"Error while trying to run script <{job_db.script}> from job <{job_db.name}>.")
|
|
552
|
+
log("ERROR", f"Exception: {e}")
|
|
553
|
+
log("ERROR", "")
|
|
554
|
+
|
|
555
|
+
import traceback
|
|
556
|
+
traceback.print_exc()
|
|
557
|
+
|
|
558
|
+
sys.exit(-1)
|
|
559
|
+
|
|
560
|
+
log("NORMAL", f"Job {'<'+job_db.name+'>':{max_job_name_length}} : {job_db.metadata.current_state:8} --> {'QUEUED':8} {'(detached)' if job_db.detached == 'true' else ''}")
|
|
561
|
+
normal_logs += 1
|
|
562
|
+
|
|
563
|
+
job_db.metadata.executions_id.append(execution_id)
|
|
564
|
+
job_db.metadata.executions_uri.append(script_manager.exe_uri(execution_id))
|
|
565
|
+
job_db.metadata.current_state = "QUEUED"
|
|
566
|
+
job_db.metadata.start_times.append(_time_now(heartbeat>0, cycle))
|
|
567
|
+
|
|
568
|
+
elif job_db.on_input_err == "fail":
|
|
569
|
+
job_db.metadata.executions_id.append("<None>")
|
|
570
|
+
job_db.metadata.executions_uri.append("<None>")
|
|
571
|
+
job_db.metadata.current_state = "RUNNING:TO_FAILURE"
|
|
572
|
+
job_db.metadata.start_times.append(_time_now(heartbeat>0, cycle))
|
|
573
|
+
|
|
574
|
+
elif job_db.on_input_err == "succeed":
|
|
575
|
+
job_db.metadata.executions_id.append("<None>")
|
|
576
|
+
job_db.metadata.executions_uri.append("<None>")
|
|
577
|
+
job_db.metadata.current_state = "RUNNING:TO_SUCCESS"
|
|
578
|
+
job_db.metadata.start_times.append(_time_now(heartbeat>0, cycle))
|
|
579
|
+
|
|
580
|
+
elif job_db.on_input_err == "skip":
|
|
581
|
+
job_db.metadata.executions_id.append("<None>")
|
|
582
|
+
job_db.metadata.executions_uri.append("<None>")
|
|
583
|
+
job_db.metadata.current_state = "RUNNING:TO_SKIPPED"
|
|
584
|
+
job_db.metadata.start_times.append(_time_now(heartbeat>0, cycle))
|
|
585
|
+
|
|
586
|
+
else:
|
|
587
|
+
log("ERROR", "")
|
|
588
|
+
log("ERROR", f"Invalid value for 'on_input_err': {job_db.on_input_err}")
|
|
589
|
+
log("ERROR", "")
|
|
590
|
+
|
|
591
|
+
job_db.save()
|
|
592
|
+
|
|
593
|
+
else:
|
|
594
|
+
log("DEBUG", " - At least one dependency from above is not ready. We need to keep waiting.")
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
# Check if we are done
|
|
598
|
+
#
|
|
599
|
+
|
|
600
|
+
log("DEBUG", "")
|
|
601
|
+
log("DEBUG", " - Checking if we are done...")
|
|
602
|
+
|
|
603
|
+
keep_going = False
|
|
604
|
+
jobs_running_queued = []
|
|
605
|
+
|
|
606
|
+
for job_db in db_proxy.jobs:
|
|
607
|
+
|
|
608
|
+
if job_db.metadata.current_state == "WAITING":
|
|
609
|
+
log("DEBUG", f" - Job {'<'+job_db.name+'>':{max_job_name_length}} status is {'<'+job_db.metadata.current_state+'>':10} {'(detached)' if job_db.detached == 'true' else ''} ")
|
|
610
|
+
else:
|
|
611
|
+
log("DEBUG", f" - Job {'<'+job_db.name+'>':{max_job_name_length}} status is {'<'+job_db.metadata.current_state+'>':10} {'(detached)' if job_db.detached == 'true' else ''} {script_manager.exe_uri(job_db.metadata.executions_id[-1])}")
|
|
612
|
+
|
|
613
|
+
if job_db.metadata.current_state == "WAITING" or \
|
|
614
|
+
job_db.metadata.current_state.startswith("RUNNING:") or \
|
|
615
|
+
(job_db.metadata.current_state in ["RUNNING", "QUEUED"] and \
|
|
616
|
+
job_db.detached == "false"):
|
|
617
|
+
keep_going = True
|
|
618
|
+
|
|
619
|
+
if job_db.metadata.current_state.startswith("RUNNING:") or \
|
|
620
|
+
(job_db.metadata.current_state in ["RUNNING", "QUEUED"] and \
|
|
621
|
+
job_db.detached == "false"):
|
|
622
|
+
jobs_running_queued.append(job_db.name)
|
|
623
|
+
|
|
624
|
+
if not keep_going:
|
|
625
|
+
log("DEBUG", " - Nothing else remaining! Exiting the polling loop...")
|
|
626
|
+
|
|
627
|
+
if normal_logs > 0:
|
|
628
|
+
if len(jobs_running_queued) > 0:
|
|
629
|
+
aux = ', '.join(jobs_running_queued)
|
|
630
|
+
if len(aux) > 80:
|
|
631
|
+
aux = aux[:77] + "..."
|
|
632
|
+
log("NORMAL", f"{len(jobs_running_queued)} job(s) currently running/queued: {aux}")
|
|
633
|
+
normal_logs += 1
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
# Sleep
|
|
637
|
+
#
|
|
638
|
+
if heartbeat > 0:
|
|
639
|
+
sleep_event.wait(heartbeat) # Interruptible sleep
|
|
640
|
+
sleep_event.clear()
|
|
641
|
+
if normal_logs > 0:
|
|
642
|
+
log("NORMAL", "")
|
|
643
|
+
|
|
644
|
+
|
|
645
|
+
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
646
|
+
# This is where we exit the while loop.
|
|
647
|
+
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
648
|
+
|
|
649
|
+
# Update/set pipeline start time
|
|
650
|
+
#
|
|
651
|
+
db_proxy.pipeline.metadata.stop_time = _time_now(heartbeat>0, cycle)
|
|
652
|
+
db_proxy.pipeline.save()
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
# Decide what to return
|
|
656
|
+
#
|
|
657
|
+
log("DEBUG", "")
|
|
658
|
+
log("DEBUG", "")
|
|
659
|
+
log("DEBUG+NORMAL", "")
|
|
660
|
+
log("DEBUG+NORMAL", "Polling finished. Checking jobs status...")
|
|
661
|
+
|
|
662
|
+
total_failed = 0
|
|
663
|
+
total_canceled = 0
|
|
664
|
+
total_skipped = 0
|
|
665
|
+
|
|
666
|
+
for job_db in db_proxy.jobs:
|
|
667
|
+
log("DEBUG+NORMAL", f" - Job {'<'+job_db.name+'>':{max_job_name_length}} : {job_db.metadata.current_state:8} {'(detached)' if job_db.detached == 'true' else ''} {script_manager.exe_uri(job_db.metadata.executions_id[-1])}")
|
|
668
|
+
|
|
669
|
+
if job_db.detached == "false":
|
|
670
|
+
if job_db.metadata.current_state == "FAILURE" : total_failed += 1
|
|
671
|
+
if job_db.metadata.current_state == "CANCELED" : total_canceled += 1
|
|
672
|
+
if job_db.metadata.current_state == "CANCELED_WHILE_WAITING": total_canceled += 1
|
|
673
|
+
if job_db.metadata.current_state == "SKIPPED" : total_skipped += 1
|
|
674
|
+
|
|
675
|
+
log("DEBUG+NORMAL", "")
|
|
676
|
+
log("DEBUG", f" - total_failed = {total_failed}")
|
|
677
|
+
log("DEBUG", f" - total_canceled = {total_canceled}")
|
|
678
|
+
log("DEBUG", f" - total_skipped = {total_skipped}")
|
|
679
|
+
|
|
680
|
+
return_value = "UNKNOWN"
|
|
681
|
+
|
|
682
|
+
if trigger_pipeline : return_value = f"TRIGGER:{trigger_pipeline}"
|
|
683
|
+
elif total_failed > 0 : return_value = "FAILURE"
|
|
684
|
+
elif total_canceled > 0 : return_value = "CANCELED"
|
|
685
|
+
else : return_value = "SUCCESS"
|
|
686
|
+
|
|
687
|
+
log("DEBUG", "")
|
|
688
|
+
log("DEBUG", f" - return_value = {return_value}")
|
|
689
|
+
|
|
690
|
+
db_proxy.pipeline.metadata.current_state = return_value
|
|
691
|
+
db_proxy.pipeline.save()
|
|
692
|
+
|
|
693
|
+
if trigger_pipeline:
|
|
694
|
+
log("DEBUG+NORMAL", f"Triggering new pipeline: {trigger_pipeline}")
|
|
695
|
+
|
|
696
|
+
signal.signal(signal.SIGINT, signal.SIG_DFL)
|
|
697
|
+
signal.signal(signal.SIGTERM, signal.SIG_DFL)
|
|
698
|
+
|
|
699
|
+
return return_value
|
|
700
|
+
|