pipeforge 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,515 @@
1
+ # vim: colorcolumn=101 textwidth=100
2
+
3
+ import os
4
+ import sys
5
+ import json
6
+ import socket
7
+
8
+ import mongoengine # Not included in python's standard lib (ie. need to be "pip install"ed)
9
+
10
+ from .utils import log # Local module
11
+ from . import file_loader # Local module
12
+
13
+
14
+
15
+ ####################################################################################################
16
+ # Job database schema
17
+ ####################################################################################################
18
+
19
+ # The pipeline manager depends on an external MongoDB server where we will be storing information
20
+ # about the pipeline we run.
21
+ #
22
+ # More specifically, two "collections" will be created in a database (typically called "pipelines")
23
+ # inside the MongoDB server:
24
+ #
25
+ # - The "pipeline" collection will contain "pipeline objects" represented by the "Pipeline" python
26
+ # class.
27
+ # Each of them stores the time the pipeline was started/stopped and pointers to other pipelines
28
+ # that were executed as "retriggers" of the current one.
29
+ #
30
+ # - The "job" collection will contain "job objects" represented by the "Job" python class.
31
+ # Each of them stores information associated to one job (aka. "step") of a given pipeline, such
32
+ # as the script to execute, the number of retries, what to do with the pipeline if the job
33
+ # fails, input and output parameters, etc...
34
+ #
35
+ # When a pipeline is run for the first time from a *.{json,toml} specification file, this is what
36
+ # happens:
37
+ #
38
+ # 1. A Pipeline() object is created
39
+ #
40
+ # 2. For each job definition found in the *.{json,toml} file, a Job() object is created and linked
41
+ # to the Pipeline() object created in the previous step:
42
+ #
43
+ # .--- Job()
44
+ # |
45
+ # Pipeline() <---+--- Job()
46
+ # |
47
+ # '--- Job()
48
+ #
49
+ #
50
+ # When we want to retrigger a previously executed pipeline, this is what happens:
51
+ #
52
+ # 1. A *new* Pipeline() object is created and linked to the original Pipeline()
53
+ #
54
+ # 2. For all Job()s that need to be retriggered (for example #1 and #2), new Job() entries are
55
+ # created and linked to the new Pipeline():
56
+ #
57
+ # .--- Job()
58
+ # |
59
+ # Pipeline() <---+--- Job()
60
+ # *original* |
61
+ # '--- Job()
62
+ #
63
+ # .--- Job() *copy*
64
+ # |
65
+ # Pipeline() <---+--- Job() *copy*
66
+ # *new*
67
+ #
68
+ # 3. For all Job()s that don't need to be retriggered (for example #3) a new link is
69
+ # added from the original Job() to the new Pipeline():
70
+ #
71
+ # .--- Job()
72
+ # |
73
+ # Pipeline() <---+--- Job()
74
+ # *original* |
75
+ # '--- Job() ----->------.
76
+ # |
77
+ # .--- Job() *copy* |
78
+ # | |
79
+ # Pipeline() <---+--- Job() *copy* |
80
+ # *new* | |
81
+ # '-------------<--------'
82
+ #
83
+ #
84
+ # NOTE: When does this happen? When re-triggering a pipeline from the middle. For example if
85
+ # one of the latest jobs failed (for example due to external circumnstances) and we want to
86
+ # retrigger the pipeline from that point because we know it is not needed to go through the
87
+ # whole process again.
88
+ #
89
+ #
90
+ # Internally, MongoDB stores JSON objects inside "collections". We could work with our python
91
+ # "Pipeline" and "Job" classes and every time we want to update the database, first serialize the
92
+ # class instance into a JSON string and then use the "pymongo" library to send these strings to the
93
+ # server... *or*, we could use a higher level library such as "mongoengine", which automatically
94
+ # maps python classes into MongoDB objects. Example:
95
+ #
96
+ # class Book(mongoengine.Document):
97
+ # title = mongoengine.StringField()
98
+ # author = mongoengine.StringField()
99
+ #
100
+ # bible = Book()
101
+ # bible.title = "The Bible"
102
+ # bible.author = "Many people"
103
+ # bible.save() <---------- This creates a JSON entry in the "Book" collection in the remote
104
+ # MongoDB server
105
+ #
106
+ # Following this strategy, we are going two create two "mongoengine" classes ("Pipeline" and "Job")
107
+ # for the two collections we will be using.
108
+ #
109
+ # NOTE: There are also two auxiliary classes ("PipelineMetadata" and "JobMetadata") that are only
110
+ # used for the purpose of embedding a dictionary in the original class (this makes it possible to
111
+ # have a "hierarchy" in the final JSON object stored in MongoDB)
112
+
113
+
114
+ class PipelineMetadata(mongoengine.EmbeddedDocument):
115
+ name = mongoengine.StringField()
116
+
117
+ current_state = mongoengine.StringField() # "WAITING", "RUNNING", etc..
118
+
119
+ start_time = mongoengine.DateTimeField()
120
+ stop_time = mongoengine.DateTimeField()
121
+
122
+ parent = mongoengine.ReferenceField("Pipeline")
123
+ #
124
+ # Reference to the parent Pipeline from which the current one was executed. If this is the
125
+ # first time the pipeline is run this variable will be empty.
126
+ # Notice I had to use "Pipeline" (in quotes) because the actual Pipeline class is defined
127
+ # later. Fortunately "mongoengine" allows you to use this trick to work around this
128
+ # limitation.
129
+
130
+ fluff = mongoengine.StringField()
131
+ #
132
+ # JSON string containing the data from the [meta] section (if any) of the *.{json,toml} file
133
+ # that was used to create this Pipeline object.
134
+ # This information is only stored here for user convenience, as pipeforge does not need it for
135
+ # anything.
136
+ # Note that "fluff" will be empty for all pipelines expect the "top level" one (i.e. the one
137
+ # without a "parent")
138
+
139
+ db_uri = mongoengine.StringField()
140
+ #
141
+ # URI of the database that contained this Pipeline object when it was created
142
+
143
+ exe_uri = mongoengine.StringField()
144
+ #
145
+ # This is a reference to the "environment" where the pipeline was executed.
146
+ # In the typical case (you are running pipeforge on the terminal) it will include the hostname
147
+ # and process ID)... but there are other possibilities: if, for example, pipeforge detects
148
+ # that it is being run as a Jenkins job, it will contain the URL to an HTTP page containing
149
+ # the job output.
150
+
151
+
152
+ class Pipeline(mongoengine.Document):
153
+ file_path = mongoengine.StringField()
154
+
155
+ metadata = mongoengine.EmbeddedDocumentField(PipelineMetadata)
156
+
157
+ def to_dict(self):
158
+ return {
159
+ "id" : str(self.id),
160
+ "file_path" : self.file_path,
161
+ "metadata" : {
162
+ "name" : str(self.metadata.name),
163
+ "current_state" : str(self.metadata.current_state),
164
+ "start_time" : str(self.metadata.start_time),
165
+ "stop_time" : str(self.metadata.stop_time),
166
+ "parent" : str(self.metadata.parent),
167
+ "fluff" : str(self.metadata.fluff),
168
+ "db_uri" : str(self.metadata.db_uri),
169
+ "exe_uri" : str(self.metadata.exe_uri)
170
+ }
171
+ }
172
+
173
+
174
+ class JobMetadata(mongoengine.EmbeddedDocument):
175
+ pipeline = mongoengine.ReferenceField(Pipeline)
176
+
177
+ dependencies = mongoengine.ListField(mongoengine.StringField())
178
+
179
+ original_retries = mongoengine.StringField() # "retries", "input" and "output" are modified
180
+ original_input = mongoengine.DictField() # while the pipeline is running. We need to save
181
+ original_output = mongoengine.DictField() # here the original values in case we later want
182
+ # to restart the pipeline (fully or partially)
183
+
184
+ current_state = mongoengine.StringField() # "WAITING", "RUNNING", etc..
185
+
186
+ executions_id = mongoengine.ListField(mongoengine.StringField()) # For each execution
187
+ executions_uri = mongoengine.ListField(mongoengine.StringField()) # (which can be more than
188
+ start_times = mongoengine.ListField(mongoengine.DateTimeField()) # one if retries>0) we
189
+ start2_times = mongoengine.ListField(mongoengine.DateTimeField()) # save these items.
190
+ stop_times = mongoengine.ListField(mongoengine.DateTimeField()) # NOTE: "start2" is when
191
+ results = mongoengine.ListField(mongoengine.StringField()) # the job *really* starts
192
+ # executing (after waiting
193
+ # in queue, if applicable)
194
+
195
+
196
+ class Job(mongoengine.Document):
197
+ name = mongoengine.StringField()
198
+ script = mongoengine.StringField()
199
+ runner = mongoengine.StringField()
200
+ detached = mongoengine.StringField()
201
+ timeout = mongoengine.StringField()
202
+ retries = mongoengine.StringField()
203
+ on_failure = mongoengine.StringField()
204
+ on_input_err = mongoengine.StringField()
205
+
206
+ input = mongoengine.DictField()
207
+ output = mongoengine.DictField()
208
+
209
+ metadata = mongoengine.EmbeddedDocumentField(JobMetadata)
210
+
211
+ def to_dict(self):
212
+ return {
213
+ "id" : str(self.id),
214
+ "name" : self.name,
215
+ "script" : self.script,
216
+ "runner" : self.runner,
217
+ "detached" : self.detached,
218
+ "timeout" : self.timeout,
219
+ "retries" : self.retries,
220
+ "on_failure" : self.on_failure,
221
+ "on_input_err" : self.on_input_err,
222
+ "input" : self.input,
223
+ "output" : self.output,
224
+ "metadata" : {
225
+ "pipeline" : str(self.metadata.pipeline.id),
226
+ "dependencies" : self.metadata.dependencies,
227
+ "original_retries" : self.metadata.original_retries,
228
+ "original_input" : self.metadata.original_input,
229
+ "original_output" : self.metadata.original_output,
230
+ "current_state" : self.metadata.current_state,
231
+ "executions_id" : self.metadata.executions_id,
232
+ "executions_uri" : self.metadata.executions_uri,
233
+ "start_times" : [str(x) for x in self.metadata.start_times],
234
+ "start2_times" : [str(x) for x in self.metadata.start2_times],
235
+ "stop_times" : [str(x) for x in self.metadata.stop_times],
236
+ "results" : self.metadata.results
237
+ }
238
+ }
239
+
240
+
241
+
242
+ ####################################################################################################
243
+ # API
244
+ ####################################################################################################
245
+
246
+ class DBProxy():
247
+ """
248
+ Proxy object that establishes a mapping with pipeline data stored in a remote database.
249
+
250
+ One instance of this class represents *one* full pipeline, which includes:
251
+
252
+ - One Pipeline() object, which can be obtained by calling "pipeline()" on the instance
253
+ - Several Job() objects, which can be ontained by calling "jobs()" on the instance
254
+ """
255
+
256
+ def __init__(self, database_url, pipeline_id):
257
+ """
258
+ @param database_url: URL of the database where pipeline data can be found. Example:
259
+
260
+ mongodb://localhost:27017/pipelines
261
+
262
+ @param pipeline_id: Identifier of the pipeline that we want to work with. It can take two
263
+ different types of values:
264
+
265
+ - A path to a json/toml file containing a pipeline definition, followed by "::",
266
+ followed by the name of the pipeline inside the file (usually "main"), followd by
267
+ "::", followed by the nothing or ID of an already existing pipeline in @ref
268
+ database_url.
269
+
270
+ In this case a new pipeline object will be created in @ref database_url and its parent
271
+ will be set to the provided ID after the second "::" (if any). Examples:
272
+
273
+ some/path/to/my/file/pipeline.json::main::
274
+ some/path/to/my/file/pipeline.toml::failure_fallback::701ab7e09abe5e7261092834
275
+
276
+ - The ID of an already existing pipeline in @ref database_url. Example:
277
+
278
+ 64512b58dd70fd2adba57103
279
+ """
280
+
281
+ log("DEBUG+NORMAL", "")
282
+ log("DEBUG+NORMAL", "Connecting to external database...")
283
+
284
+ try:
285
+ mongoengine.connect(host=database_url)
286
+
287
+ log("DEBUG", " - Connection successful")
288
+ except Exception:
289
+ log("ERROR", " - Connection to external database failed. Aborting")
290
+ sys.exit(-1)
291
+
292
+ self._database_url = database_url
293
+ self._pipeline_db = None
294
+ self._jobs_db = []
295
+ self._job_dependencies = {}
296
+
297
+
298
+ # Obtain a reference to the pipeline object (first creating and inserting new entries in the
299
+ # remote database if needed):
300
+ #
301
+ if "::" in pipeline_id:
302
+ self._pipeline_db = self._load_from_disk(pipeline_id.split("::")[0],
303
+ pipeline_id.split("::")[1],
304
+ pipeline_id.split("::")[2])
305
+ else:
306
+ self._pipeline_db = self._load_from_db(pipeline_id)
307
+
308
+ log("DEBUG", "")
309
+ log("DEBUG", "pipeline_db object:")
310
+ log("DEBUG", self._pipeline_db.to_dict())
311
+
312
+ log("DEBUG", "")
313
+ log("DEBUG", "job_db objects associated to the previous pipeline_db object:")
314
+ for i, job_db in enumerate(Job.objects(metadata__pipeline=self._pipeline_db)):
315
+ log("DEBUG", f"#{i}:")
316
+ log("DEBUG", job_db.to_dict())
317
+
318
+
319
+ # Obtain a list of references to all the job objects that are part of the pipeline
320
+ #
321
+ self._jobs_db = Job.objects(metadata__pipeline=self._pipeline_db)
322
+
323
+
324
+ # Create a dependencies dictionary where each entry key is a job name and its value the list
325
+ # of job names it depends on
326
+ #
327
+ for job_db in Job.objects(metadata__pipeline=self._pipeline_db):
328
+ self._job_dependencies[job_db.name] = job_db.metadata.dependencies
329
+
330
+ log("DEBUG", "")
331
+ log("DEBUG", "Pipeline dependencies structure:")
332
+ log("DEBUG", self._job_dependencies)
333
+
334
+
335
+ def _load_from_disk(self, pipeline_file_path, pipeline_name, parent_pipeline_id=""):
336
+ """
337
+ Load a new pipeline from a JSON/TOML file, create new objects in the remote database that
338
+ represent it (and its associated jobs) and save a reference to the Pipeline() object in
339
+ self._pipeline_db
340
+
341
+ @param pipeline_file_path: path to the JSON/TOML file containing the pipeline specification
342
+
343
+ @param pipeline_name: name of the pipeline to load from all the ones contained in the file.
344
+ For single pipeline files this will always be "main"
345
+
346
+ @param parent_db_id: ID of the Pipeline object in the DB which will be set as the "parent"
347
+ of the new Pipeline object we are about to create. It can be left empty for "no parent".
348
+ """
349
+
350
+ # ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
351
+ # Load pipeline data from disk
352
+ # ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
353
+
354
+ fl = file_loader.FileLoader(pipeline_file_path)
355
+ pipeline_data = fl.get_pipeline()
356
+
357
+ pipeline = [x for x in pipeline_data["pipelines"] if x["name"] == pipeline_name][0]
358
+ pipeline_deps = fl.get_dependencies()[pipeline["name"]]
359
+
360
+ log("DEBUG", "")
361
+ log("DEBUG", "Pipeline data as loaded from disk:")
362
+ log("DEBUG", pipeline)
363
+
364
+ # ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
365
+ # Insert pipeline/job entries into the database
366
+ # ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
367
+
368
+ pipeline_db = Pipeline(file_path = pipeline_file_path,
369
+ metadata = PipelineMetadata() # Initially empty
370
+ )
371
+
372
+ pipeline_db.metadata.name = pipeline["name"]
373
+
374
+ if parent_pipeline_id:
375
+ pipeline_db.metadata.parent = Pipeline.objects.get(id=parent_pipeline_id)
376
+ else:
377
+ if "meta" in pipeline_data:
378
+ pipeline_db.metadata.fluff = json.dumps(pipeline_data["meta"])
379
+
380
+ pipeline_db.metadata.db_uri = self._database_url
381
+
382
+ if os.getenv("JENKINS_URL") and os.getenv("BUILD_URL"):
383
+ # It looks like the pipeline is being executed in a Jenkins instance. We can access its
384
+ # ouput of the URL contains in environment variable BUILD_URL
385
+ #
386
+ pipeline_db.metadata.exe_uri = os.getenv("BUILD_URL") + "console"
387
+ else:
388
+ # Use hostname + PID
389
+ #
390
+ pipeline_db.metadata.exe_uri = f"{socket.gethostname()}, PID={os.getpid()}"
391
+
392
+ pipeline_db.save()
393
+
394
+ log("DEBUG", "")
395
+ log("DEBUG", f"New pipeline object added to database (id={pipeline_db.id})")
396
+
397
+
398
+ # Insert all job objects into the "job collection"
399
+ #
400
+ for job in pipeline["jobs"]:
401
+
402
+ j_db = Job(**job)
403
+
404
+ j_db.metadata = JobMetadata(
405
+ pipeline = pipeline_db,
406
+ dependencies = pipeline_deps[job["name"]],
407
+ original_retries = job["retries"],
408
+ current_state = "WAITING",
409
+ executions_id = [],
410
+ executions_uri = [],
411
+ start_times = [],
412
+ stop_times = [],
413
+ results = []
414
+ )
415
+
416
+ if "input" in job.keys(): j_db.metadata["original_input"] = job["input"]
417
+ if "output" in job.keys(): j_db.metadata["original_output"] = job["output"]
418
+
419
+ j_db.save()
420
+
421
+ return pipeline_db
422
+
423
+
424
+ def _load_from_db(self, pipeline_id):
425
+ """
426
+ Retrieve a given pipeline object from the remote database and save a reference to the
427
+ associated local Pipeline() object in self._pipeline_db.
428
+ """
429
+
430
+ log("DEBUG", "")
431
+ log("DEBUG", f"Reusing a previously existing pipeline object (id={pipeline_id})")
432
+
433
+ pipeline_db = Pipeline.objects.get(id=pipeline_id)
434
+
435
+ return pipeline_db
436
+
437
+
438
+ ################################################################################################
439
+ # External API
440
+ ################################################################################################
441
+
442
+ @property
443
+ def database_url(self):
444
+ """
445
+ Return the database URL originally provided when creating the DBProxy() object.
446
+ """
447
+ return self._database_url
448
+
449
+
450
+ @property
451
+ def job_dependencies(self):
452
+ """
453
+ Return a dictionary of dependencies where each entry has this format:
454
+ - key : job name from the current pipeline
455
+ - value: list of other job names it depends on
456
+
457
+ Example:
458
+
459
+ >>> db_proxy = DBProxy("pipelines/example.toml")
460
+ >>> print(db_proxy.job_dependencies)
461
+ {'Test 1': ['Build'], 'Test 2': ['Build'], 'Build': ['Download'], 'Send email': ['Test 1', 'Test 2']}
462
+ """
463
+ return self._job_dependencies
464
+
465
+
466
+ @property
467
+ def pipeline(self):
468
+ """
469
+ Return the Pipeline() object that is associated to the current DBProxy handler.
470
+ The JSON data associated to this Pipeline() object can be queried using "dot notation".
471
+
472
+ Example:
473
+
474
+ >>> db_proxy = DBProxy("pipelines/example.toml")
475
+ >>> print(db_proxy.pipeline.file_path)
476
+ pipelines/example.toml
477
+
478
+ In addition, two methods can be called on the returned object:
479
+
480
+ - obj.save() will push the value of *modified* fields to the database (ie. the whole
481
+ remote object is not overwriten, thus you don't have to worry about other fields)
482
+
483
+ - obj.reload() will replace the whole local object (ie. "all fields") with a copy of the
484
+ data stored in the database.
485
+ """
486
+ return self._pipeline_db
487
+
488
+
489
+ @property
490
+ def jobs(self):
491
+ """
492
+ Return the list of Job() objects that make up the Pipeline() associated to the current
493
+ DBProxy handler.
494
+
495
+ The JSON data associated to these Job() objects can be queries using "dot notation".
496
+
497
+ Example:
498
+
499
+ >>> db_proxy = DBProxy("pipelines/example.toml")
500
+ >>> for job_db in db_proxy.jobs: print(job_db.name)
501
+ Test 1
502
+ Test 2
503
+ Build
504
+ Send email
505
+
506
+ In addition, two methods can be called on the returned object:
507
+
508
+ - obj.save() will push the value of *modified* fields to the database (ie. the whole
509
+ remote object is not overwriten, thus you don't have to worry about other fields)
510
+
511
+ - obj.reload() will replace the whole local object (ie. "all fields") with a copy of the
512
+ data stored in the database.
513
+ """
514
+ return self._jobs_db
515
+