pipeforge 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1004 @@
1
+ # vim: colorcolumn=101 textwidth=100
2
+
3
+ import os
4
+ import sys
5
+ import uuid
6
+ import time
7
+ import shutil
8
+ import random
9
+ import subprocess
10
+ import datetime
11
+
12
+ from .params import JobParams # Local module
13
+ from .utils import log # Local module
14
+
15
+
16
+
17
+ ####################################################################################################
18
+ # Virtual interface for the "ScriptManager" class
19
+ ####################################################################################################
20
+
21
+ class ScriptManager():
22
+ """
23
+ A ScriptManager object is the one in charge of starting, querying and stopping job scripts.
24
+
25
+ This class is meant to be used as a base clase of a more specialized one which actually
26
+ implements the methods below.
27
+
28
+ Example:
29
+
30
+ class MyScript(pipeforge.ScriptManager):
31
+ def run(self, description, script_name, modifiers, db_job_id):
32
+ ...
33
+ def query(self, execution_id):
34
+ ...
35
+ def stop(self, execution_id):
36
+ ...
37
+
38
+ Then you are meant to provide an instance of this specialized class to
39
+ "pipeforge.Pipeline.run()", like this:
40
+
41
+ import pipeforge
42
+
43
+ p = pipeline.Pipeline(...)
44
+ p.run(script_manager = MyScript(...))
45
+ """
46
+
47
+ def run(self, description, script_name, modifiers, db_job_id):
48
+ """
49
+ Start running a script and return an ID associated to its execution.
50
+
51
+ @param description: short description of the job (for human consumption)
52
+
53
+ @param script_name: name of the script to run, as provided in the "script" field of the
54
+ associated job in the *.toml pipeline definition. It will have different meaning to
55
+ different subclasses of ScriptManager(). Some examples:
56
+
57
+ - The path to a script in the local file system
58
+ - The URL of a REST API endpoint to trigger the job in a remote Jenkins instance
59
+ - The key of a dictionary of pre-defined scripts
60
+ - Etc...
61
+
62
+ @param modifiers: additional restrictions considerations regarding where/how the script
63
+ should be run, as provided in the "runner" field of the associated job in the *.toml
64
+ pipeline definition. It will have different meaning to different subclasses of
65
+ ScriptManager(). Some examples:
66
+
67
+ - The name of a specific remote machine
68
+ - A comma separated list of tags a Jenkins agent must contain when considering which
69
+ one to use
70
+ - The name a docker image
71
+ - Etc...
72
+
73
+ Check the documentation of each subclass for details.
74
+
75
+ @param db_job_id: this is the "token" that the script manager will make available "somehow"
76
+ to the script so that it can use as a parameter to JobParams() which it needs to read
77
+ input parameters and set output parameters. The way this "token" is made available to
78
+ the script depends on the specific subclass of ScriptManager(). Some examples:
79
+
80
+ - As an environment variable
81
+ - As an input argument to the script
82
+ - As the contents of a predefined file in the file system
83
+ - Etc...
84
+
85
+ The "token" itself should be treated as an opaque type: the script manager should not do
86
+ anything with it except for passing it to the script.
87
+
88
+ @return execution_id, which can be used by the rest of methods of this class to query/stop
89
+ the script that we just started here.
90
+ """
91
+
92
+ raise NotImplementedError("method 'run()' not yet implemented")
93
+
94
+
95
+ def query(self, execution_id):
96
+ """
97
+ Return the current status of the script that was started by "run()"
98
+
99
+ @param execution_id: handler returned by the call to "run()" that started the script whose
100
+ status we want to query.
101
+
102
+ @return one of these:
103
+
104
+ - "NOT FOUND" : The provided execution_id is incorrect.
105
+ - "QUEUED" : The script is waiting to be started, which will eventually happen.
106
+ - "RUNNING" : The script is currently running.
107
+ - "SUCCESS" : The script has finished executing and returned OK.
108
+ - "FAILURE" : The script has finished executing and returned KO.
109
+ - "CANCELED" : The script has been externally canceled (by calling stop()), either while
110
+ "RUNNING" or while "QUEUED"
111
+ """
112
+
113
+ raise NotImplementedError("method 'query()' not yet implemented")
114
+
115
+
116
+ def stop(self, execution_id):
117
+ """
118
+ Stop a running (or queued) script previously started with "run()".
119
+
120
+ @param execution_id: handler returned by the call to "run()" that started the script whose
121
+ status we want to stop.
122
+ """
123
+
124
+ raise NotImplementedError("method 'query()' not yet implemented")
125
+
126
+
127
+ def exe_uri(self, execution_id):
128
+ """
129
+ Return a reference to "some object" with more information about the job execution.
130
+
131
+ @param execution_id: handler returned by the call to "run()" that started the script whose
132
+ status we want to stop.
133
+
134
+ @return a string that contains some type of reference (a URL, a path in the local file
135
+ system, etc...) containing more information about the requested execution.
136
+ Check the specific subclass documentation for more details.
137
+ """
138
+
139
+ raise NotImplementedError("method 'query()' not yet implemented")
140
+
141
+
142
+ ####################################################################################################
143
+ # "Dummy" script manager
144
+ ####################################################################################################
145
+
146
+ class DummyScriptManager(ScriptManager):
147
+ """
148
+ This runner ignores the provided script name and instead pretends it is waiting on queue for X
149
+ seconds and then pretends it is running Y more seconds.
150
+
151
+ Once it finishes the required output parameters will automatically be set to a random value
152
+ different from "?" to trick the pipeline manager into thinking execution took place as expected.
153
+
154
+ Notice that you can run *any* pipeline with this class, as it does not require any particular
155
+ input parameter to be present on any of the jobs. In other words, you can use DummyScript() to
156
+ "simulate" the execution of one real pipeline (even if the timings will obviously not be the
157
+ same).
158
+
159
+ NOTE:
160
+ Contrary to what is recommended in the documentation of the "pipeforge.ScriptManager.run()"
161
+ function, this specialized class *will* use the "db_job_id" "token" to simulate what a real
162
+ script would do.
163
+ This is obviously a hack due to the nature of this specialized subclass: a real subclass
164
+ implementation would never do this and always treat "db_job_id" as an opaque type.
165
+ """
166
+
167
+ def __init__(self):
168
+ self._processes_table = {}
169
+ self._min_queue = 0
170
+ self._max_queue = 5
171
+ self._min_run = 3
172
+ self._max_run = 9
173
+
174
+
175
+ def run(self, description, script_name, modifiers, db_job_id):
176
+ """
177
+ See ScriptManager.run() and then read this:
178
+
179
+ @param modifiers is ignored in this subclass
180
+ """
181
+
182
+ job_params = JobParams(db_job_id)
183
+ execution_id = uuid.uuid4().hex
184
+
185
+
186
+ self._processes_table[execution_id] = (
187
+ int(time.time()), # Start timestamp
188
+ random.randint(self._min_queue, self._max_queue), # Time in queue (seconds)
189
+ random.randint(self._min_run, self._max_run), # Time running (seconds)
190
+ job_params
191
+ )
192
+
193
+ return execution_id
194
+
195
+
196
+ def query(self, execution_id):
197
+
198
+ if execution_id not in self._processes_table:
199
+ return "CANCELED"
200
+
201
+ start_time = self._processes_table[execution_id][0]
202
+ time_in_queue = self._processes_table[execution_id][1]
203
+ time_running = self._processes_table[execution_id][2]
204
+ job_params = self._processes_table[execution_id][3]
205
+
206
+ current_time = int(time.time())
207
+
208
+ if current_time > start_time + time_in_queue + time_running:
209
+ # Hack: the real script would have set the expected output parameters while running.
210
+ # In this dummy implementation (which simply runs "sleep") we need to do it here, when
211
+ # the pipeline manager queries the state and we figure out that the script has already
212
+ # finished executing.
213
+
214
+ fake_output = {}
215
+ i = 0
216
+
217
+ for output_param in job_params.get_output_parameters():
218
+ fake_output[output_param] = f"Fake param #{i}"
219
+ i += 1
220
+
221
+ job_params.set_output_parameters_and_values(fake_output)
222
+
223
+ return "SUCCESS"
224
+
225
+ elif current_time > start_time + time_in_queue:
226
+
227
+ return "RUNNING"
228
+
229
+ else:
230
+
231
+ return "QUEUED"
232
+
233
+
234
+ def stop(self, execution_id):
235
+ self._processes_table[execution_id].kill()
236
+ del self._processes_table[execution_id]
237
+
238
+
239
+ def exe_uri(self, execution_id):
240
+ return ""
241
+
242
+
243
+ ####################################################################################################
244
+ # "Test" script manager (for unit tests)
245
+ ####################################################################################################
246
+
247
+ class TestScriptManager(ScriptManager):
248
+ """
249
+ This runner ignores the provided script name and instead simulates that it is running for some
250
+ time. More specifically, once you call "run()" on this object...
251
+
252
+ - The next "x-1" calls you make to "query()" will always return "RUNNING"
253
+ - Call "x" to "query()" will return the value provided in input argument "return_code"
254
+
255
+ ...where "x" is the number provided in input argument "run_time"
256
+
257
+ This makes execution time independent from the POLL INTERVAL that the pipeline manager is using,
258
+ which makes it convenient to obtain deterministic results.
259
+
260
+ The last call to "query()" (ie. the one that no longer returns "RUNNING") will also set output
261
+ parameters to a random value different from "?" to trick the pipeline manager into thinking
262
+ execution took place as expected. The only exception to this is that if there is an output
263
+ parameter called "output" it will be set to the value of input parameter "output_value".
264
+
265
+ Notice that you can only use this runner with a special type of pipelines. In particular, all
266
+ jobs defined in the pipeline must have "run_time", "return_code" and (optionally) "output_value"
267
+ as input parameters. This means you can not use it with a real pipeline. The benefit is that you
268
+ get more control over those two parameters, making this type or runner ideal for testing the
269
+ pipeline manager itself.
270
+
271
+ If a job is configured to have retries, you can specify:
272
+ - ...a different "run_time" on each of them by using a "+" separator, like this:
273
+ run_time = 2+4+1
274
+ - ...a different "return_code" on each of them by using a "+" separator, like this:
275
+ return_code = "FAILURE+FAILURE+SUCCESS"
276
+ - ...a different "output_value" on each of them by using a "+" separator, like this:
277
+ output_value = "cocacola+fanta+pepsi"
278
+
279
+ NOTE:
280
+ Contrary to what is recommended in the documentation of the "pipeforge.ScriptManager.run()"
281
+ function, this specialized class *will* use the "db_job_id" "token" to simulate what a real
282
+ script would do and also return different values depending on the number of pending retries.
283
+ This is obviously a hack due to the nature of this specialized subclass: a real subclass
284
+ implementation would never do this and always treat "db_job_id" as an opaque type.
285
+ """
286
+
287
+ def __init__(self):
288
+ self._processes_table = {}
289
+
290
+
291
+ def run(self, description, script_name, modifiers, db_job_id):
292
+ """
293
+ See ScriptManager.run() and then read this:
294
+
295
+ @param modifiers is ignored in this subclass
296
+ """
297
+
298
+ job_params = JobParams(db_job_id)
299
+ execution_id = uuid.uuid4().hex
300
+
301
+ input_params = job_params.get_input_parameters_and_values()
302
+
303
+ if "run_time" not in input_params or "return_code" not in input_params:
304
+ raise ValueError("In order to use the \"TestScriptManager\" all jobs defined in the pipeline must have these two input parameters defined: \"run_time\" and \"return_code\"")
305
+
306
+ retries_count = int(input_params["__job_attempt"])
307
+
308
+ queue_cycles = input_params["queue_time"].split("+")
309
+ run_cycles = input_params["run_time"].split("+")
310
+ return_code = input_params["return_code"].split("+")
311
+
312
+ queue_cycles = int(queue_cycles[retries_count%len(queue_cycles)])
313
+ run_cycles = int(run_cycles[retries_count%len(run_cycles)])
314
+ return_code = return_code[retries_count%len(return_code)]
315
+
316
+ self._processes_table[execution_id] = [queue_cycles,
317
+ run_cycles,
318
+ return_code,
319
+ job_params]
320
+ return execution_id
321
+
322
+
323
+ def query(self, execution_id):
324
+ queue_cycles = self._processes_table[execution_id][0]
325
+ run_cycles = self._processes_table[execution_id][1]
326
+ return_code = self._processes_table[execution_id][2]
327
+ job_params = self._processes_table[execution_id][3]
328
+
329
+ if run_cycles == -99:
330
+ return "CANCELED"
331
+
332
+ if queue_cycles > 0:
333
+ self._processes_table[execution_id][0] = queue_cycles - 1
334
+ return "QUEUED"
335
+
336
+ if run_cycles > 1:
337
+ self._processes_table[execution_id][1] = run_cycles - 1
338
+ return "RUNNING"
339
+
340
+ else:
341
+ # Hack: the real script would have set the expected output parameters while running. In
342
+ # this dummy implementation we need to do it here, when the pipeline manager queries the
343
+ # state and we figure out that the script has already finished executing.
344
+
345
+ fake_output = {}
346
+ i = 0
347
+
348
+ for output_param in job_params.get_output_parameters():
349
+ if output_param == "output":
350
+ fake_output["output"] = job_params.get_input_parameters_and_values()["output_value"]
351
+ else:
352
+ fake_output[output_param] = f"Fake param #{i}"
353
+ i += 1
354
+
355
+ job_params.set_output_parameters_and_values(fake_output)
356
+
357
+ self._processes_table[execution_id][1] = 1 # In case we query again in the future (which
358
+ # we shouldn't, but if we do, return the
359
+ # same thing
360
+ return return_code
361
+
362
+
363
+ def stop(self, execution_id):
364
+ self._processes_table[execution_id][0] = -99 # Special indicator for "CANCELED"
365
+
366
+
367
+ def exe_uri(self, execution_id):
368
+ return ""
369
+
370
+
371
+
372
+ ####################################################################################################
373
+ # "Local" script manager
374
+ ####################################################################################################
375
+
376
+ class LocalScriptManager(ScriptManager):
377
+ """
378
+ This runner executes the provided script in the local system, ignoring the "modifiers" argument
379
+ that run() receives.
380
+
381
+ Scripts must already be present in the local system and, for convenience, it is recomended to
382
+ provide the full path when calling run()
383
+ """
384
+
385
+ def __init__(self):
386
+ self._processes_table = {}
387
+
388
+
389
+ def run(self, description, script_name, modifiers, db_job_id):
390
+ """
391
+ See ScriptManager.run() and then read this:
392
+
393
+ @param modifiers is ignored in this subclass
394
+ """
395
+
396
+ if not os.path.exists(script_name):
397
+ # Search in PATH
398
+ #
399
+ new_script_name = shutil.which(script_name)
400
+
401
+ if new_script_name is None:
402
+ log("ERROR", f"script <{script_name}> could not be found...")
403
+ sys.exit(-1)
404
+ else:
405
+ script_name = new_script_name
406
+
407
+ # Create files to store process STDOUT, STDERR
408
+ #
409
+ template = f"/tmp/pipeforge__{datetime.datetime.now().strftime('%Y%m%d%H%M%S')}_" + \
410
+ f"{os.path.basename(script_name)}"
411
+
412
+ fstdout = open(template + ".stdout.txt", mode="w+")
413
+ fstderr = open(template + ".stderr.txt", mode="w+")
414
+
415
+ log("DEBUG", f" - Logging STDOUT to file {fstdout.name}")
416
+ log("DEBUG", f" - Logging STDERR to file {fstderr.name}")
417
+
418
+ p = subprocess.Popen(
419
+ script_name,
420
+ shell = True,
421
+ env = os.environ | {"JOB_ID": db_job_id}, # Copy of the current environment plus
422
+ stdout = fstdout, # an extra new variable (JOB_ID)
423
+ stderr = fstderr,
424
+ start_new_session=True)
425
+
426
+ self._processes_table[str(p.pid)] = [p, fstdout, fstderr]
427
+
428
+ return str(p.pid)
429
+
430
+
431
+ def _final_log(self, execution_id):
432
+ """
433
+ Print log messages indicating process has finished
434
+ """
435
+
436
+ #out, err = p.communicate()
437
+
438
+ p, fstdout, fstderr = self._processes_table[execution_id]
439
+
440
+ log("DEBUG", f" - script <{p.pid}> has finished executing...")
441
+ log("DEBUG", f" - STDOUT has been saved to file {fstdout.name}")
442
+ fstdout.close()
443
+ for line in open(fstdout.name).readlines(): # Uncomment for quicker
444
+ log("DEBUG", f" > {line.rstrip()}") # debug
445
+
446
+
447
+ log("DEBUG", f" - STDERR has been saved to file {fstderr.name}")
448
+ fstderr.close()
449
+ for line in open(fstderr.name).readlines(): # Uncomment for quicker
450
+ log("DEBUG", f" > {line.rstrip()}") # debug
451
+
452
+
453
+ def query(self, execution_id):
454
+
455
+ if execution_id not in self._processes_table.keys():
456
+ return "NOT FOUND"
457
+
458
+ p = self._processes_table[execution_id][0]
459
+
460
+ if p == "CANCELED":
461
+ return "CANCELED"
462
+
463
+ if p.poll() is None:
464
+ return "RUNNING"
465
+
466
+ # If we reach this point, the process has finished executing.
467
+
468
+ self._final_log(execution_id)
469
+
470
+ if p.returncode != 0:
471
+ return "FAILURE"
472
+
473
+ return "SUCCESS"
474
+
475
+
476
+ def stop(self, execution_id):
477
+ if execution_id in self._processes_table.keys():
478
+
479
+ p = self._processes_table[execution_id]
480
+ p[0].kill()
481
+
482
+ p[0] = "CANCELED" # We don't remove it from the table, instead we set the entry
483
+ # to "CANCELED" so that query() can figure out this process was
484
+ # canceled
485
+
486
+
487
+ def exe_uri(self, execution_id):
488
+ """
489
+ Return the local path to the files where stdout and stderr are being / have been saved.
490
+
491
+ When the process is still running you should be able to "tail -f ..." these file to get
492
+ realtime feedback.
493
+
494
+ When the process has finished these files will contain the whole STDOUT/STDERR produced.
495
+ """
496
+ if execution_id == "<None>":
497
+ return "(did not start)"
498
+
499
+ return f"stdout:{self._processes_table[execution_id][1].name}; stderr:{self._processes_table[execution_id][2].name}"
500
+
501
+
502
+
503
+ ####################################################################################################
504
+ # "Jenkins" script manager
505
+ ####################################################################################################
506
+
507
+ class JenkinsScriptManager(ScriptManager):
508
+ """
509
+ This runner executes the provided script on a remote Jenkins instance.
510
+
511
+ In order to use this ScriptManager type you need to first set and export an environment variable
512
+ called "PIPEFORGESCRIPT_JENKINS_ENDPOINT" which takes this format:
513
+
514
+ <Jenkins instance URL>:::<username>:::<password>:::<job_name>
515
+
516
+ Example:
517
+
518
+ export PIPEFORGESCRIPT_JENKINS_ENDPOINT='https://jenkins.example.com:::john:::cobra123:::Run_script'
519
+
520
+ Note that if the Jenkins instance does not require authentication, you can leave <username> and
521
+ <password> empty, like this:
522
+
523
+ export PIPEFORGESCRIPT_JENKINS_ENDPOINT='https://jenkins.example.com::::::Run_script'
524
+
525
+ Hint: Don't forget to use sinque quotes (') or bash will complain about ":::"
526
+
527
+ <job_name> ("Run_script") in the example above is a regular Jenkins job that you will first have
528
+ to create following these instructions:
529
+
530
+ 1. Name it as <job_name> (ex: "Run_script" or something like that)
531
+
532
+ 2. In the "Configure" tab, select the "This project is parameterized" checkbox and add these
533
+ entries:
534
+
535
+ - Type: Label
536
+ Name: SLAVE
537
+ Description:
538
+ List of tags (separated by a space) the selected slave to run this job must have.
539
+
540
+ Check [1] for valid syntax.
541
+
542
+ [1] "label expression", as defined in
543
+ https://kb.novaordis.com/index.php/Jenkins_Job_Label_Expression
544
+
545
+ - Type: Multi-string
546
+ Name: ENVIRONMENT
547
+ Description: List <variable>=<value> lines of environment variables to set
548
+
549
+ 3. Select the "Execute concurrent builds if necessary" checkbox.
550
+
551
+ 4. In "Build Steps" add this entry:
552
+
553
+ - Type: Execute shell
554
+ - Command: /path/to/your/entry_point/run.sh
555
+
556
+ "pipeforge" with remotely trigger this job using Jenkins REST API, providing these arguments:
557
+
558
+ - SLAVE = <whatever has been specified in the 'runner' field in the pipeline TOML file>
559
+ - ENVIRONMENT = ...
560
+ JOB_DESCRIPTION=<value specified in the 'name' field in the pipeline TOML file>
561
+ JOB_SCRIPT=<value specified in the 'script' field in the pipeline TOML file>
562
+ JOB_ID=<the token to access input and output parameters for the "pipeforge" job >
563
+
564
+ NOTE: When pipeforge itself is also running as a Jenkins job, the ENVIRONMENT field will
565
+ contain an extra variable called "JENKINS_PARENT" which points to the URL of the Jenkins
566
+ job running the pipeline, so that children can (if they want) access it.
567
+
568
+ "run.sh" is a script *you provide*, preinstalled on each Jenkins slave (or downloaded as part
569
+ of the Jenkins job "command"), responsible for:
570
+
571
+ 1. Read environment variable "ENVIRONMENT" and "export" all its entries so that they are
572
+ available as environment variables by its children.
573
+ In other words, if "ENVIRONMENT" contains a line that reads "DEBUG_MODE=yes", then
574
+ environment variable "DEBUG_MODE" must exist for all processes started by
575
+ "run.sh".
576
+
577
+ NOTE: In addition to "JOB_SCRIPT" and "JOB_ID", the argument "ENVIRONMENT" that is
578
+ sent to Jenkins will also contain all the environment variables that start with
579
+ "PIPEFORGESCRIPT_JENKINS_OPT__". For example, if the following environment variables
580
+ exists:
581
+
582
+ PIPEFORGESCRIPT_JENKINS_OPT__COLOR=blue
583
+ PIPEFORGESCRIPT_JENKINS_OPT__AGE=33
584
+
585
+ The "ENVIRONMENT" will contain these extra entries:
586
+
587
+ COLOR=blue
588
+ AGE=33
589
+
590
+ This feature makes it possible to run jobs in a special/debug mode when needed by
591
+ simply setting variables before running pipeforge.
592
+
593
+ 2. Run the script referenced by "JOB_SCRIPT" (which could be a full path, a "key" of a table
594
+ that is translated into a set of actions, etc...).
595
+ Each Jenkins slave must obviously have all that is needed to "run" "JOB_SCRIPT" (ie. in
596
+ "Build Steps" you might have to add an extra one that downloads the repository where all
597
+ your "pipeforge" scripts are located).
598
+
599
+ This is an example of a simple "run.sh" (in bash) that assumes "JOB_SCRIPT" refers to a
600
+ binary that is on $PATH and can be executed without any special setup:
601
+
602
+ > if ! [ -z "$ENVIRONMENT" ]; then
603
+ >
604
+ > eval $(
605
+ > while IFS= read -r line
606
+ > do
607
+ > echo "export $line"
608
+ > done < <(printf '%s\n' "$ENVIRONMENT")
609
+ > )
610
+ > fi
611
+ >
612
+ > $JOB_SCRIPT
613
+
614
+ "$JOB_SCRIPT" is a regular "pipeforge" job script, which means it must call
615
+ pipeforge.JobParam(os.getenv("JOB_ID")) to read input parameters and set output parameters in
616
+ the pipeline job.
617
+ """
618
+
619
+ def __init__(self):
620
+
621
+ import jenkins # Only needed for this type of script manager, that's why we import it here
622
+ # (Note: This module is installed with "pip install python-jenkins")
623
+
624
+ self._triggered_jobs = {}
625
+ #
626
+ # There will be one entry for each triggered job.
627
+ #
628
+ # The <key> is the original "queue" number that is returned by Jenkins when
629
+ # a job is triggered (this is a monotonic number guaranteed to be unique).
630
+ #
631
+ # The <value> is a list of three elements:
632
+ # - The first indicates the last known state of the job. It can be
633
+ # "QUEUED", "CANCELED", "RUNNING", "SUCCESS" or "FAILURE".
634
+ # - The second one is a string containing "more information" about the
635
+ # job. It can be be either the string "(queued)" or an URL pointing to a
636
+ # to a Jenkins page with more information about the job (ie. its output
637
+ # log)
638
+ # - The third element can be None or the execution id (a number) that the
639
+ # job receives once it is no longer in queue.
640
+ # - The fourth element can be "?", None or the queue id (a number) of the
641
+ # "semaphore" entry a job (that has already received an execution id)
642
+ # has to wait for before *really* start running (note: only special
643
+ # Jenkins jobs make use of this "second queue" feature)
644
+
645
+ jenkins_server = None
646
+
647
+ # Parse environment variable PIPEFORGESCRIPT_JENKINS_ENDPOINT and create jenkins object
648
+ #
649
+ jenkins_info = os.getenv("PIPEFORGESCRIPT_JENKINS_ENDPOINT", "")
650
+
651
+ if jenkins_info != "":
652
+ try:
653
+ url, user, passw, job = jenkins_info.split(":::")
654
+
655
+ if user == "" and passw == "":
656
+ jenkins_server = jenkins.Jenkins(url)
657
+ else:
658
+ jenkins_server = jenkins.Jenkins(url, user, passw)
659
+
660
+ except Exception:
661
+ pass
662
+
663
+ if jenkins_server is None:
664
+ log("ERROR", "In order to use the 'JenkinsScriptManager' you need to export an environment variable")
665
+ log("ERROR", "called 'PIPEFORGESCRIPT_JENKINS_ENDPOINT' with information regarding the Jenkins endpoint")
666
+ log("ERROR", "capable of running jobs.")
667
+ log("ERROR", "")
668
+ log("ERROR", "For more details run this from the python interpreter:")
669
+ log("ERROR", "")
670
+ log("ERROR", " import pipeforge")
671
+ log("ERROR", " help(pipeforge._internal.script.JenkinsScriptManager)")
672
+ log("ERROR", "")
673
+ sys.exit(-1)
674
+
675
+ # Make sure the provided Jenkins URL is reachable and contains an actual Jenkins instance
676
+ #
677
+ try:
678
+ jenkins_server.get_version()
679
+ except Exception as e:
680
+ log("ERROR", f"The provided Jenkins url ({url}) does not seem to point to an actual Jenkins instance!")
681
+ log("ERROR", f"Exception: {e}")
682
+ sys.exit(-1)
683
+
684
+ # Obtain the offset (in seconds) between the Jenkins server clock and the computer where
685
+ # this script is running.
686
+ #
687
+ try:
688
+ server_time = int(jenkins_server.run_script("println(System.currentTimeMillis())"))//1000
689
+ pc_time = int(time.time())
690
+ except Exception as e:
691
+ log("ERROR", f"Could not obtain Jenkins' clock current time on {url}")
692
+ log("ERROR", f"Exception: {e}")
693
+ sys.exit(-1)
694
+
695
+ self._clock_offset = pc_time - server_time
696
+
697
+ # Make sure the provided job exists
698
+ #
699
+ def recursive_search(x, key, matches):
700
+ if isinstance(x, list):
701
+ for i in x:
702
+ recursive_search(i, key, matches)
703
+ elif isinstance(x, dict):
704
+ for k,v in x.items():
705
+ if k == key:
706
+ matches.append(v)
707
+ elif isinstance(v, dict) or isinstance(v, list):
708
+ recursive_search(v, key, matches)
709
+
710
+ matches = []
711
+ recursive_search(jenkins_server.get_all_jobs(), "fullname", matches)
712
+
713
+ if job not in matches:
714
+ log("ERROR", f"The provided Jenkins job ({job}) does not seem to")
715
+ log("ERROR", f"exist in {url}")
716
+ sys.exit(-1)
717
+
718
+ # Save data for later use
719
+ #
720
+ self._jenkins_server = jenkins_server
721
+ self._jenkins_job = job
722
+
723
+
724
+ def run(self, description, script_name, modifiers, db_job_id):
725
+ """
726
+ See ScriptManager.run() and then read this:
727
+
728
+ @param script_name is the name of the script to run. Jenkins' job must have previously been
729
+ configured to "know" where to look for this script (maybe it is in $PATH, or maybe there
730
+ is a hardcoded path in the job configuration)
731
+
732
+ @param modifiers must be a string string containing a "Jenkins job label expression", as
733
+ defined in [1]. This is used to restrict on which slaves the job can run. Example:
734
+ "LINUX and FAST"
735
+
736
+ [1] https://kb.novaordis.com/index.php/Jenkins_Job_Label_Expression
737
+ """
738
+
739
+ slave = modifiers # In which node should the scrip run
740
+ environment = {
741
+ "JOB_DESCRIPTION" : f"{description}", # Short job description
742
+ "JOB_SCRIPT" : f"{script_name}", # Script to run in Jenkins
743
+ "JOB_ID" : f"{db_job_id}", # Pipeline job reference (to query for input
744
+ # parameters and set output parameters)
745
+
746
+ "JENKINS_PARENT" : f"{os.getenv('BUILD_URL', '') if os.getenv('JENKINS_URL') else ''}"
747
+ #
748
+ # When using Jenkins, it is not strange to also have pipeforge
749
+ # itself running as a Jenkins job.
750
+ # When that happens, environment variable "BUILD_URL" points to
751
+ # the URL of that job (otherwise it will be empty).
752
+ # We will send its value to the triggered job so that it knows
753
+ # which its "parent" Jenkins job is.
754
+ # Why? Several reasons... for example, to let the Jenkins child
755
+ # execution change the name of the Jenkins parent execution.
756
+ }
757
+
758
+ for extra_env in os.environ.keys():
759
+ if extra_env.startswith("PIPEFORGESCRIPT_JENKINS_OPT__"):
760
+ param = extra_env[len("PIPEFORGESCRIPT_JENKINS_OPT__"):]
761
+ value = os.getenv(extra_env)
762
+ environment[param] = value
763
+
764
+ attempts = 0
765
+ while True:
766
+ log("DEBUG", f" - Triggering {script_name} in Jenkins@{self._jenkins_job}:")
767
+ log("DEBUG", f" - SLAVE = {slave}")
768
+ log("DEBUG", f" - ENVIRONMENT = {environment}")
769
+
770
+ try:
771
+ queue_item = self._jenkins_server.build_job(
772
+ self._jenkins_job,
773
+ {
774
+ "SLAVE" : slave,
775
+ "ENVIRONMENT" : "\n".join([f"{k}=\"{v}\"" for k,v in environment.items()])
776
+ }
777
+ )
778
+ break
779
+
780
+ except Exception as e:
781
+ log("DEBUG", f"Error when triggering Jenkins job: {e}")
782
+ if attempts > 3:
783
+ log("ERROR", "Desisting after several attempts...")
784
+ raise e
785
+ else:
786
+ attempts += 1
787
+ log("DEBUG", " - Retrying...")
788
+ time.sleep(2)
789
+
790
+ queue_item = str(queue_item)
791
+ self._triggered_jobs[queue_item] = ["QUEUED", "(queued)", None, "?"]
792
+
793
+ return queue_item
794
+
795
+
796
+ def query(self, execution_id):
797
+
798
+ if execution_id not in self._triggered_jobs.keys():
799
+ return "NOT FOUND"
800
+
801
+ if self._triggered_jobs[execution_id][0] in ["SUCCESS", "FAILURE", "CANCELED"]:
802
+ return self._triggered_jobs[execution_id][0]
803
+
804
+ elif self._triggered_jobs[execution_id][0] == "QUEUED":
805
+
806
+ if self._triggered_jobs[execution_id][2] is None:
807
+ # While in the queue, the job never received an execution number. We need to keep
808
+ # querying the queue API.
809
+ #
810
+ try:
811
+ response = self._jenkins_server.get_queue_item(int(execution_id))
812
+ self._triggered_jobs[execution_id][2] = response["executable"]["number"]
813
+ except Exception:
814
+ return "QUEUED"
815
+ else:
816
+ # We are still in the queue despite having received an execution number in the past.
817
+ # This can only mean that we are in the "secondary queue" (one that Jenkins uses
818
+ # in jobs of type "Pipeline" where the agent is dynamically decided inside the
819
+ # Jenkinsfile groovy script).
820
+ # We don't need to do anything special here. The code below will take care of this
821
+ # case.
822
+ #
823
+ pass
824
+
825
+ # If we reach this point, it means the job is already running...
826
+
827
+ real_execution_id = self._triggered_jobs[execution_id][2]
828
+
829
+ attempts = 0
830
+ while True:
831
+ try:
832
+ build_info_response = \
833
+ self._jenkins_server.get_build_info(self._jenkins_job, real_execution_id)
834
+ break
835
+
836
+ except Exception as e:
837
+ log("DEBUG", f"Error when retrieving Jenkins job details: {e}")
838
+ if attempts > 3:
839
+ log("ERROR", "Desisting after several attempts...")
840
+ return "NOT FOUND"
841
+ else:
842
+ attempts += 1
843
+ time.sleep(2)
844
+
845
+ # ...or maybe not! As explained before, for jobs of type "Pipeline" with dynamic agent
846
+ # assignment, there is a "secondary queue": the job appears as running (ie. it has a real
847
+ # execution id) but it is actually waiting for *a new element* (with a different queue ID!)
848
+ # to be removed from the queue before resuming its work.
849
+ #
850
+ # The only way to deal with this case that I could find is to do this:
851
+ #
852
+ # 1. If the job has been running for just a few seconds, consider it "QUEUED" (this is to
853
+ # avoid a race condition where the job is already running but the "secondary queue" entry
854
+ # has not yet been created) and try again later.
855
+ #
856
+ # 2. If the job has been running for more time then:
857
+ # - The first time we get to this point query all queued items with an ID higher than the
858
+ # original queue ID and check whether they reference our "already running" job. If one
859
+ # is found, save it (this is the "secondary queue" entry).
860
+ # - All other times query the queue API for the saved element until it dissappears.
861
+ #
862
+ current_execution_time = \
863
+ int(time.time()) - (build_info_response["timestamp"]//1000 + self._clock_offset)
864
+
865
+ if current_execution_time < 30:
866
+ return "QUEUED"
867
+
868
+ elif self._triggered_jobs[execution_id][3] == "?":
869
+ # This is the first time a job with an execution ID reaches this far.
870
+ # We need to check for a "secondary queue" entry, in case this is a job of type
871
+ # "Pipeline" with dynamic agent assignment.
872
+ #
873
+ response = self._jenkins_server.get_queue_info()
874
+
875
+ for element in response:
876
+ if int(element["id"]) > int(real_execution_id):
877
+ response = self._jenkins_server.get_queue_item(int(element["id"]), depth=2)
878
+
879
+ if response["task"]["url"].endswith(
880
+ f"{self._jenkins_job}/{real_execution_id}/"):
881
+
882
+ # "Secondary queue" entry found. Save it to query for it from now on
883
+ #
884
+ self._triggered_jobs[execution_id][3] = element["id"]
885
+ return "QUEUED"
886
+
887
+ # The job was never added to the "secondary queue"
888
+ #
889
+ self._triggered_jobs[execution_id][3] = None
890
+
891
+ elif self._triggered_jobs[execution_id][3] is not None:
892
+ # The job is waiting on the "secondary queue". Query the queue API.
893
+ #
894
+ response = self._jenkins_server.get_queue_info()
895
+
896
+ if self._triggered_jobs[execution_id][3] in [x["id"] for x in response]:
897
+ return "QUEUED"
898
+ else:
899
+ # The job is no longer in the "secondary" queue.
900
+ #
901
+ self._triggered_jobs[execution_id][3] = None
902
+
903
+ # Finally, if we reach this point we can be 100% sure we are no longer in any queue!
904
+
905
+ if self._triggered_jobs[execution_id][1] == "(queued)":
906
+ self._triggered_jobs[execution_id][1] = build_info_response["url"]
907
+
908
+ if "result" in build_info_response:
909
+
910
+ if build_info_response["inProgress"]:
911
+ self._triggered_jobs[execution_id][0] = "RUNNING"
912
+
913
+ elif build_info_response["result"] == "SUCCESS":
914
+ self._triggered_jobs[execution_id][0] = "SUCCESS"
915
+
916
+ elif build_info_response["result"] == "FAILURE":
917
+ self._triggered_jobs[execution_id][0] = "FAILURE"
918
+
919
+ elif build_info_response["result"] == "ABORTED":
920
+ self._triggered_jobs[execution_id][0] = "CANCELED"
921
+
922
+ elif build_info_response["result"] is None:
923
+ self._triggered_jobs[execution_id][0] = "RUNNING"
924
+
925
+ return self._triggered_jobs[execution_id][0]
926
+
927
+ return "NOT FOUND"
928
+
929
+
930
+ def stop(self, execution_id):
931
+ log("DEBUG", f" - Stopping Jenkins@{self._jenkins_job} build #{execution_id}")
932
+
933
+ if execution_id not in self._triggered_jobs.keys():
934
+ log("DEBUG", " - build not found")
935
+ return
936
+
937
+ if self._triggered_jobs[execution_id][0] == "CANCELED":
938
+ log("DEBUG", " - build was already canceled")
939
+ return
940
+
941
+ if self._triggered_jobs[execution_id][0] == "QUEUED":
942
+ log("DEBUG", " - build is still queued")
943
+ # Job still in queue. Query queue
944
+
945
+ for i in range(2):
946
+ #
947
+ # We try to stop the job twice due to Jenkins race conditions (?) where sometimes a
948
+ # job is no longer in a queue but Jenkins still reports that is the case.
949
+
950
+ queue_id = None # ID to cancel job from queue
951
+ other_id = None # ID to cancel job from any other place
952
+
953
+ try:
954
+ response = self._jenkins_server.get_queue_item(int(execution_id))
955
+ other_id = response["executable"]["number"] # Job no longer in queue.
956
+ # Update it's ID
957
+ log("DEBUG", f" - build was actually running. Its real execution id is {other_id}")
958
+ except Exception:
959
+ queue_id = execution_id # Job is still in queue.
960
+
961
+ if queue_id:
962
+ try:
963
+ self._jenkins_server.cancel_queue(queue_id)
964
+ except Exception as e:
965
+ if i == 0:
966
+ log("ERROR", f"{queue_id} : {e}")
967
+ return
968
+ else:
969
+ try:
970
+ self._jenkins_server.stop_build(self._jenkins_job, other_id)
971
+ except Exception as e:
972
+ if i == 0:
973
+ log("ERROR", f"{self._jenkins_job}/{other_id} : {e}")
974
+ return
975
+
976
+ time.sleep(1)
977
+
978
+ else: # the job was already running
979
+ log("DEBUG", f" - build is currently running with a real execution id of {self._triggered_jobs[execution_id][2]}")
980
+
981
+ try:
982
+ self._jenkins_server.stop_build(self._jenkins_job,
983
+ self._triggered_jobs[execution_id][2])
984
+ except Exception as e:
985
+ log("ERROR", f"{self._jenkins_job}/{self._triggered_jobs[execution_id][2]} : {e}")
986
+ return
987
+
988
+ log("DEBUG", " - build successfully canceled")
989
+
990
+ self._triggered_jobs[execution_id][0] = "CANCELED" # We don't remove it from the table,
991
+ # instead we set the entry to "CANCELED"
992
+ # so that query() can figure out this
993
+ # process was canceled
994
+
995
+ def exe_uri(self, execution_id):
996
+ """
997
+ Return a link to the Jenkins page associated to the job execution.
998
+
999
+ In that link the user can check STDOUT, artifacts, status, etc...
1000
+ """
1001
+ if execution_id == "<None>":
1002
+ return "(did not start)"
1003
+
1004
+ return self._triggered_jobs[execution_id][1]