pipeforge 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipeforge/__init__.py +1201 -0
- pipeforge/__main__.py +208 -0
- pipeforge/_internal/__init__.py +0 -0
- pipeforge/_internal/core_loop.py +700 -0
- pipeforge/_internal/database.py +515 -0
- pipeforge/_internal/drawing.py +544 -0
- pipeforge/_internal/file_loader.py +654 -0
- pipeforge/_internal/inspector.py +555 -0
- pipeforge/_internal/params.py +148 -0
- pipeforge/_internal/pipeline.py +212 -0
- pipeforge/_internal/script.py +1004 -0
- pipeforge/_internal/utils.py +122 -0
- pipeforge/examples/hello_jenkins.toml +16 -0
- pipeforge/examples/hello_world.toml +67 -0
- pipeforge/examples/merge_pull_request.toml +240 -0
- pipeforge/examples/multi_pipeline.toml +93 -0
- pipeforge/examples/multi_pipeline_2.toml +54 -0
- pipeforge/examples/nightly.toml +19 -0
- pipeforge/examples/retries.toml +43 -0
- pipeforge/examples/scripts/hello_world__get_purpose_in_life.py +34 -0
- pipeforge/examples/scripts/hello_world__print_summary.py +27 -0
- pipeforge/examples/scripts/hello_world__repeat_purpose.py +33 -0
- pipeforge/examples/scripts/retry.py +25 -0
- pipeforge/examples/scripts/timeout.py +25 -0
- pipeforge/examples/timeout.toml +37 -0
- pipeforge-1.0.0.dist-info/METADATA +198 -0
- pipeforge-1.0.0.dist-info/RECORD +29 -0
- pipeforge-1.0.0.dist-info/WHEEL +4 -0
- pipeforge-1.0.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
# vim: colorcolumn=101 textwidth=100
|
|
2
|
+
|
|
3
|
+
import mongoengine # Not included in python's standard lib (ie. need to be "pip install"ed)
|
|
4
|
+
|
|
5
|
+
from .database import Job
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
####################################################################################################
|
|
10
|
+
# API
|
|
11
|
+
####################################################################################################
|
|
12
|
+
|
|
13
|
+
class JobParams():
|
|
14
|
+
|
|
15
|
+
def __init__(self, db_job_id):
|
|
16
|
+
"""
|
|
17
|
+
This class represents a job running inside a pipeline.
|
|
18
|
+
|
|
19
|
+
@param db_job_id: Job identification token. Scripts running inside a pipeline will typically
|
|
20
|
+
receive this token as an environment variable. You don't need to know its format
|
|
21
|
+
(instead, just use its opaque value), but for the shake of completeness it is a string
|
|
22
|
+
with two tokens separated by a "#":
|
|
23
|
+
|
|
24
|
+
- The first token is the database URL. Example:
|
|
25
|
+
|
|
26
|
+
mongodb://localhost:27017/pipelines
|
|
27
|
+
|
|
28
|
+
- The second token is the Job() document ID inside that database whose input and output
|
|
29
|
+
parameters we want to access. It looks like this:
|
|
30
|
+
|
|
31
|
+
644969f17dc00d3911e77df7
|
|
32
|
+
|
|
33
|
+
Thus, the full "db_job_id" looks like this:
|
|
34
|
+
|
|
35
|
+
mongodb://localhost:27017/pipelines#644969f17dc00d3911e77df7
|
|
36
|
+
"""
|
|
37
|
+
database_url, job_id = db_job_id.split("#")
|
|
38
|
+
|
|
39
|
+
mongoengine.connect(host = database_url)
|
|
40
|
+
|
|
41
|
+
self._job = Job.objects.get(id=job_id)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def get_input_parameters_and_values(self):
|
|
45
|
+
"""
|
|
46
|
+
Return a dictionary containing the job input parameters where, for each entry:
|
|
47
|
+
|
|
48
|
+
- The key is a string with the name of the input parameter.
|
|
49
|
+
- The value is a string with its value.
|
|
50
|
+
|
|
51
|
+
The returned dictionary will also contains these extra/secret entries:
|
|
52
|
+
|
|
53
|
+
- "__pipeline_id" contains an ID string that identifies the pipeline object (in a
|
|
54
|
+
database) this job belongs to. This is only useful for reporting purposes, as no other
|
|
55
|
+
API from this class uses this value.
|
|
56
|
+
|
|
57
|
+
- "__pipeline_stdout" contains a string that "hints" on how to access stdout from the
|
|
58
|
+
pipeline engine itself. This could be a path to a file, a hostname + PID, a URL to a
|
|
59
|
+
Jenkins execution, etc...
|
|
60
|
+
|
|
61
|
+
- "__job_attempt" contains how many times this job has been reatempted. In other words:
|
|
62
|
+
the first time a job is executed this parameter will be set to "0", the second one it
|
|
63
|
+
will be set to "1", etc... Notice that a job is only ever reattempted if its "retries"
|
|
64
|
+
property is > 0.
|
|
65
|
+
|
|
66
|
+
Example:
|
|
67
|
+
|
|
68
|
+
>>> jp = JobParams(...)
|
|
69
|
+
>>> jp.get_input_parameters_and_values()
|
|
70
|
+
{ "flavor" : "vanilla",
|
|
71
|
+
"price" : "5 euro",
|
|
72
|
+
"__pipeline_id" : "6483328783e96bafb389535c",
|
|
73
|
+
"__pipeline_stdout" : "http://jenkins.example.com/job/Job_runner/1232/",
|
|
74
|
+
"__job_attempt" : "0"}
|
|
75
|
+
|
|
76
|
+
Note that you are expected to only *read* this dictionary values and *not* modify them.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
aux = self._job.input.copy()
|
|
80
|
+
aux["__pipeline_id"] = str(self._job.metadata.pipeline.metadata.db_uri) + "#" + \
|
|
81
|
+
str(self._job.metadata.pipeline.id)
|
|
82
|
+
aux["__pipeline_stdout"] = str(self._job.metadata.pipeline.metadata.exe_uri)
|
|
83
|
+
|
|
84
|
+
if self._job.metadata.original_retries == "N/A": # this happens on detached jobs
|
|
85
|
+
aux["__job_attempt"] = "0"
|
|
86
|
+
else:
|
|
87
|
+
aux["__job_attempt"] = str(int(self._job.metadata.original_retries) - \
|
|
88
|
+
int(self._job.retries))
|
|
89
|
+
|
|
90
|
+
return aux
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def get_output_parameters(self):
|
|
94
|
+
"""
|
|
95
|
+
Return a list of all the output parameters that you are expected to later set in one single
|
|
96
|
+
call to "set_output_parameters_and_values()".
|
|
97
|
+
|
|
98
|
+
Each element of the returned list is a string containing the name of the parameter.
|
|
99
|
+
|
|
100
|
+
Example:
|
|
101
|
+
|
|
102
|
+
>>> jp = JobParams(...)
|
|
103
|
+
>>> jp.get_output_parameters()
|
|
104
|
+
[ "result", "executed_tests" ]
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
return list(self._job.output.keys())
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def set_output_parameters_and_values(self, new_output_dictionary):
|
|
111
|
+
"""
|
|
112
|
+
Set the dictionary containing the job output parameters where, for each entry:
|
|
113
|
+
|
|
114
|
+
- The key is a string with the name of the output parameter.
|
|
115
|
+
- The value is a string with its value.
|
|
116
|
+
|
|
117
|
+
Typically the list of keys of "new_output_dictionary" will exactly match the list of strings
|
|
118
|
+
returned by "get_output_parameters()". In other words: you will only need to call this
|
|
119
|
+
function once, at the end of your job, with all the values for all output parameters:
|
|
120
|
+
|
|
121
|
+
Example:
|
|
122
|
+
|
|
123
|
+
>>> jp = JobParams(...)
|
|
124
|
+
>>> jp.set_output({"result" : "OK", "executed_tests" : "11"})
|
|
125
|
+
|
|
126
|
+
However, it is also possible to call it several times, providing only a subset of all the
|
|
127
|
+
(key, value) pairs each time.
|
|
128
|
+
|
|
129
|
+
>>> jp = JobParams(...)
|
|
130
|
+
>>> jp.set_output({"result" : "OK"})
|
|
131
|
+
>>> ...
|
|
132
|
+
>>> jp.set_output({"executed_tests" : "11"})
|
|
133
|
+
|
|
134
|
+
In case of one value being provided in more than one call, the last one will prevail:
|
|
135
|
+
|
|
136
|
+
>>> jp = JobParams(...)
|
|
137
|
+
>>> jp.set_output({"result" : "OK"})
|
|
138
|
+
>>> ...
|
|
139
|
+
>>> jp.set_output({"executed_tests" : "11"})
|
|
140
|
+
>>> ...
|
|
141
|
+
>>> jp.set_output({"result" : "KO"}) <---------- This is the "good" one now.
|
|
142
|
+
"""
|
|
143
|
+
|
|
144
|
+
aux = self._job.output.copy()
|
|
145
|
+
aux.update(new_output_dictionary)
|
|
146
|
+
|
|
147
|
+
self._job.update(set__output=aux)
|
|
148
|
+
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
# vim: colorcolumn=101 textwidth=100
|
|
2
|
+
|
|
3
|
+
from . import core_loop # Local module
|
|
4
|
+
from . import database # Local module
|
|
5
|
+
from . import drawing # Local module
|
|
6
|
+
|
|
7
|
+
from .utils import log # Local module
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
####################################################################################################
|
|
11
|
+
# API
|
|
12
|
+
####################################################################################################
|
|
13
|
+
|
|
14
|
+
class Pipeline:
|
|
15
|
+
|
|
16
|
+
def __init__(self, param_1, param_2 = None):
|
|
17
|
+
"""
|
|
18
|
+
You can create an object of this class in two different ways:
|
|
19
|
+
|
|
20
|
+
Option #1 (to create a new entry in the supporting database):
|
|
21
|
+
|
|
22
|
+
@param param_1: URL of the supporting database that we will use to create and save a new
|
|
23
|
+
pipeline as defined in param_2. Example:
|
|
24
|
+
|
|
25
|
+
mongodb://localhost:27017/pipelines
|
|
26
|
+
|
|
27
|
+
@param param_2: Path to a json/toml file containing a pipeline definition. Examples:
|
|
28
|
+
|
|
29
|
+
some/path/to/my/file/pipeline.json
|
|
30
|
+
some/path/to/my/file/pipeline.toml
|
|
31
|
+
|
|
32
|
+
If the pipeline file contains more that one pipeline, the pipeline called "main" will
|
|
33
|
+
be chosen by default. If you want a different one, the name of the file needs to be
|
|
34
|
+
followed by ":" plus the name of the pileline. Examples:
|
|
35
|
+
|
|
36
|
+
some/path/to/my/file/pipeline.json::second
|
|
37
|
+
some/path/to/my/file/pipeline.toml::failure_fallback
|
|
38
|
+
|
|
39
|
+
If the pipeline is being triggered from another one you should specify its ID (as
|
|
40
|
+
returned from Pipeline.uri() or Pipeline.run()). Examples:
|
|
41
|
+
|
|
42
|
+
some/path/to/my/file/pipeline.json::second::<id>
|
|
43
|
+
some/path/to/my/file/pipeline.toml::failure_fallback::<id>
|
|
44
|
+
|
|
45
|
+
Option #2 (to obtain a reference to an already existing entry in the supporting database):
|
|
46
|
+
|
|
47
|
+
@param param_1: One of the IDs returned by a previous call to "Pipeline.run()" or
|
|
48
|
+
"Pipeline.uri()"
|
|
49
|
+
|
|
50
|
+
@param param_2: Leave it empty!
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
if param_2 is None:
|
|
54
|
+
self._database_url = param_1.split("#")[0]
|
|
55
|
+
self._file = None
|
|
56
|
+
|
|
57
|
+
pipeline_id = param_1.split("#")[1]
|
|
58
|
+
|
|
59
|
+
else:
|
|
60
|
+
tokens = param_2.split("::")
|
|
61
|
+
|
|
62
|
+
if len(tokens) == 1:
|
|
63
|
+
file_name, pipeline_name, parent_pipeline_id = tokens[0], "main", ""
|
|
64
|
+
|
|
65
|
+
elif len(tokens) == 2:
|
|
66
|
+
file_name, pipeline_name, parent_pipeline_id = tokens[0], tokens[1], ""
|
|
67
|
+
|
|
68
|
+
else:
|
|
69
|
+
file_name, pipeline_name = tokens[0], tokens[1]
|
|
70
|
+
|
|
71
|
+
if param_1 != tokens[2].split("#")[0]:
|
|
72
|
+
# The parent *must* be in the same database instance in order to be referenced!
|
|
73
|
+
#
|
|
74
|
+
log("DEBUG+NORMAL", "WARNING: Ignoring cross-database parent/child relationship")
|
|
75
|
+
parent_pipeline_id = ""
|
|
76
|
+
else:
|
|
77
|
+
parent_pipeline_id = tokens[2].split("#")[1]
|
|
78
|
+
|
|
79
|
+
self._database_url = param_1
|
|
80
|
+
self._file = file_name
|
|
81
|
+
|
|
82
|
+
pipeline_id = f"{file_name}::{pipeline_name}::{parent_pipeline_id}"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
self._db_proxy = database.DBProxy(self._database_url, pipeline_id)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def uri(self):
|
|
90
|
+
"""
|
|
91
|
+
Return an ID token that unequivocally identifies this pipeline.
|
|
92
|
+
"""
|
|
93
|
+
return self._database_url + "#" + str(self._db_proxy.pipeline.id)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def run(self, script_manager, polling_period):
|
|
97
|
+
"""
|
|
98
|
+
Run all the jobs that make up the pipeline in order (ie. respecting their dependencies) and
|
|
99
|
+
wait for them to finish.
|
|
100
|
+
|
|
101
|
+
This is a blocking function. It will not return until the pipeline is done executing.
|
|
102
|
+
|
|
103
|
+
Note that one pipeline can finish with a "request to trigger another pipeline", which itself
|
|
104
|
+
can end with another similar request and so on... This function will only return until all
|
|
105
|
+
the "chained" pipelines have finished.
|
|
106
|
+
|
|
107
|
+
@param polling_period: Number of seconds between each polling cycle of the pipeline manager.
|
|
108
|
+
The lower this number, the sooner jobs will be start/stopped, but also more pressure on
|
|
109
|
+
the supporting database. If your jobs typically last more than 5 minutes, setting this
|
|
110
|
+
to 60 is more than enough.
|
|
111
|
+
|
|
112
|
+
@param script_manager: object that implements the "pipeforge.Script" interface (ie. an
|
|
113
|
+
object on which we can call "run()", "query()" and "stop()").
|
|
114
|
+
|
|
115
|
+
@return a list of tuples of two elements:
|
|
116
|
+
|
|
117
|
+
- The first one is the pipeline exit status: "FAILURE", "CANCELED", "SUCCESS" or
|
|
118
|
+
"TRIGGER:<name_of_new_pipeline>".
|
|
119
|
+
|
|
120
|
+
- The second one is an ID token that unequivocally identifies this pipeline.
|
|
121
|
+
|
|
122
|
+
Elements in the list are returned in order of execution. This means that if there are
|
|
123
|
+
more than one elements, the first N-1 one will always have status == "TRIGGER:..." and
|
|
124
|
+
the last one will always have a status of "FAILURE", "CANCELED" or "SUCCESS".
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
ret = []
|
|
128
|
+
|
|
129
|
+
# Start the core loop that triggers each of the job scripts in turn and waits for them to
|
|
130
|
+
# finish
|
|
131
|
+
#
|
|
132
|
+
result = core_loop.run_pipeline(self._db_proxy,
|
|
133
|
+
script_manager,
|
|
134
|
+
polling_period)
|
|
135
|
+
|
|
136
|
+
ret.append((result,
|
|
137
|
+
self._database_url + "#" + str(self._db_proxy.pipeline.id)))
|
|
138
|
+
|
|
139
|
+
# If the pipeline returned "TRIGGERED:<new_pipeline_name>", then create a new Pipeline
|
|
140
|
+
# object and run it with the same script_manager and polling_period
|
|
141
|
+
#
|
|
142
|
+
while result.startswith("TRIGGER:"):
|
|
143
|
+
|
|
144
|
+
new_pipeline_name = result.split(":")[1]
|
|
145
|
+
|
|
146
|
+
if self._file:
|
|
147
|
+
# Load the new Pipeline from the same file that was used to create the current
|
|
148
|
+
# Pipeline
|
|
149
|
+
#
|
|
150
|
+
new_pipeline_id = f"{self._file}::{new_pipeline_name}::{ret[-1][1]}"
|
|
151
|
+
|
|
152
|
+
else:
|
|
153
|
+
# The current pipeline was not loaded from disk but directly from the DB. This
|
|
154
|
+
# means we are "replaying" a previously executed pipeline.
|
|
155
|
+
#
|
|
156
|
+
# When it was first executed it might or might not have requested a secondary
|
|
157
|
+
# pipeline execution, which means that "new_pipeline_name" might or might not be in
|
|
158
|
+
# the DB.
|
|
159
|
+
#
|
|
160
|
+
# TODO: Options we have here:
|
|
161
|
+
#
|
|
162
|
+
# A) Return an error. Tell the user replayed pipelines are not allowed to
|
|
163
|
+
# trigger other pipelines.
|
|
164
|
+
#
|
|
165
|
+
# B) When a pipeline is saved to the DB, save also the TOML/JSON data with it,
|
|
166
|
+
# so that we can load it back here and use it to create new_pipeline_name.
|
|
167
|
+
#
|
|
168
|
+
new_pipeline_id = "TODO"
|
|
169
|
+
raise("Not implemented: triggering a pipeline from a replayed pipeline")
|
|
170
|
+
|
|
171
|
+
new_pipeline = Pipeline(self._database_url, new_pipeline_id)
|
|
172
|
+
|
|
173
|
+
result = core_loop.run_pipeline(new_pipeline._db_proxy,
|
|
174
|
+
script_manager,
|
|
175
|
+
polling_period)
|
|
176
|
+
|
|
177
|
+
ret.append((result,
|
|
178
|
+
new_pipeline._database_url + "#" + str(new_pipeline._db_proxy.pipeline.id)))
|
|
179
|
+
|
|
180
|
+
return ret
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def draw_timeline(self, mode):
|
|
184
|
+
"""
|
|
185
|
+
Draw all pipeline jobs over a timeline taking into consideration their start and stop times.
|
|
186
|
+
|
|
187
|
+
@param mode: Defines how to "draw" the timeline. Supported values are:
|
|
188
|
+
|
|
189
|
+
- "ascii:<width>:<where>:<scale>": Print to <where> using ASCII characters. Each line
|
|
190
|
+
can be up to <width> columns long. <where> can be one of these:
|
|
191
|
+
|
|
192
|
+
- "stdout" : print to the current stdout
|
|
193
|
+
- "buffer" : print to a buffer and return it to the caller
|
|
194
|
+
- "file@path/to/file.txt" : print to the specified file (ex:
|
|
195
|
+
"file@/tmp/timeline.txt")
|
|
196
|
+
|
|
197
|
+
<scale> can be either "minutes", "hours" or "auto" and affects the precission of the
|
|
198
|
+
timestamps printed in the horizontal axis.
|
|
199
|
+
|
|
200
|
+
Exampe: "ascii:150:stdout:auto"
|
|
201
|
+
|
|
202
|
+
- "unicode:<width>:<where>:<scale>": Same as the previous mode, but use unicode
|
|
203
|
+
characters instead of ASCII. This makes the time line prettier but might not be
|
|
204
|
+
supported in all terminals.
|
|
205
|
+
|
|
206
|
+
- "png": <NOT YET SUPPORTED>
|
|
207
|
+
|
|
208
|
+
@return None or other values depending on the selected @ref mode.
|
|
209
|
+
"""
|
|
210
|
+
|
|
211
|
+
return drawing.draw_timeline(self._db_proxy, mode)
|
|
212
|
+
|