pipeforge 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,148 @@
1
+ # vim: colorcolumn=101 textwidth=100
2
+
3
+ import mongoengine # Not included in python's standard lib (ie. need to be "pip install"ed)
4
+
5
+ from .database import Job
6
+
7
+
8
+
9
+ ####################################################################################################
10
+ # API
11
+ ####################################################################################################
12
+
13
+ class JobParams():
14
+
15
+ def __init__(self, db_job_id):
16
+ """
17
+ This class represents a job running inside a pipeline.
18
+
19
+ @param db_job_id: Job identification token. Scripts running inside a pipeline will typically
20
+ receive this token as an environment variable. You don't need to know its format
21
+ (instead, just use its opaque value), but for the shake of completeness it is a string
22
+ with two tokens separated by a "#":
23
+
24
+ - The first token is the database URL. Example:
25
+
26
+ mongodb://localhost:27017/pipelines
27
+
28
+ - The second token is the Job() document ID inside that database whose input and output
29
+ parameters we want to access. It looks like this:
30
+
31
+ 644969f17dc00d3911e77df7
32
+
33
+ Thus, the full "db_job_id" looks like this:
34
+
35
+ mongodb://localhost:27017/pipelines#644969f17dc00d3911e77df7
36
+ """
37
+ database_url, job_id = db_job_id.split("#")
38
+
39
+ mongoengine.connect(host = database_url)
40
+
41
+ self._job = Job.objects.get(id=job_id)
42
+
43
+
44
+ def get_input_parameters_and_values(self):
45
+ """
46
+ Return a dictionary containing the job input parameters where, for each entry:
47
+
48
+ - The key is a string with the name of the input parameter.
49
+ - The value is a string with its value.
50
+
51
+ The returned dictionary will also contains these extra/secret entries:
52
+
53
+ - "__pipeline_id" contains an ID string that identifies the pipeline object (in a
54
+ database) this job belongs to. This is only useful for reporting purposes, as no other
55
+ API from this class uses this value.
56
+
57
+ - "__pipeline_stdout" contains a string that "hints" on how to access stdout from the
58
+ pipeline engine itself. This could be a path to a file, a hostname + PID, a URL to a
59
+ Jenkins execution, etc...
60
+
61
+ - "__job_attempt" contains how many times this job has been reatempted. In other words:
62
+ the first time a job is executed this parameter will be set to "0", the second one it
63
+ will be set to "1", etc... Notice that a job is only ever reattempted if its "retries"
64
+ property is > 0.
65
+
66
+ Example:
67
+
68
+ >>> jp = JobParams(...)
69
+ >>> jp.get_input_parameters_and_values()
70
+ { "flavor" : "vanilla",
71
+ "price" : "5 euro",
72
+ "__pipeline_id" : "6483328783e96bafb389535c",
73
+ "__pipeline_stdout" : "http://jenkins.example.com/job/Job_runner/1232/",
74
+ "__job_attempt" : "0"}
75
+
76
+ Note that you are expected to only *read* this dictionary values and *not* modify them.
77
+ """
78
+
79
+ aux = self._job.input.copy()
80
+ aux["__pipeline_id"] = str(self._job.metadata.pipeline.metadata.db_uri) + "#" + \
81
+ str(self._job.metadata.pipeline.id)
82
+ aux["__pipeline_stdout"] = str(self._job.metadata.pipeline.metadata.exe_uri)
83
+
84
+ if self._job.metadata.original_retries == "N/A": # this happens on detached jobs
85
+ aux["__job_attempt"] = "0"
86
+ else:
87
+ aux["__job_attempt"] = str(int(self._job.metadata.original_retries) - \
88
+ int(self._job.retries))
89
+
90
+ return aux
91
+
92
+
93
+ def get_output_parameters(self):
94
+ """
95
+ Return a list of all the output parameters that you are expected to later set in one single
96
+ call to "set_output_parameters_and_values()".
97
+
98
+ Each element of the returned list is a string containing the name of the parameter.
99
+
100
+ Example:
101
+
102
+ >>> jp = JobParams(...)
103
+ >>> jp.get_output_parameters()
104
+ [ "result", "executed_tests" ]
105
+ """
106
+
107
+ return list(self._job.output.keys())
108
+
109
+
110
+ def set_output_parameters_and_values(self, new_output_dictionary):
111
+ """
112
+ Set the dictionary containing the job output parameters where, for each entry:
113
+
114
+ - The key is a string with the name of the output parameter.
115
+ - The value is a string with its value.
116
+
117
+ Typically the list of keys of "new_output_dictionary" will exactly match the list of strings
118
+ returned by "get_output_parameters()". In other words: you will only need to call this
119
+ function once, at the end of your job, with all the values for all output parameters:
120
+
121
+ Example:
122
+
123
+ >>> jp = JobParams(...)
124
+ >>> jp.set_output({"result" : "OK", "executed_tests" : "11"})
125
+
126
+ However, it is also possible to call it several times, providing only a subset of all the
127
+ (key, value) pairs each time.
128
+
129
+ >>> jp = JobParams(...)
130
+ >>> jp.set_output({"result" : "OK"})
131
+ >>> ...
132
+ >>> jp.set_output({"executed_tests" : "11"})
133
+
134
+ In case of one value being provided in more than one call, the last one will prevail:
135
+
136
+ >>> jp = JobParams(...)
137
+ >>> jp.set_output({"result" : "OK"})
138
+ >>> ...
139
+ >>> jp.set_output({"executed_tests" : "11"})
140
+ >>> ...
141
+ >>> jp.set_output({"result" : "KO"}) <---------- This is the "good" one now.
142
+ """
143
+
144
+ aux = self._job.output.copy()
145
+ aux.update(new_output_dictionary)
146
+
147
+ self._job.update(set__output=aux)
148
+
@@ -0,0 +1,212 @@
1
+ # vim: colorcolumn=101 textwidth=100
2
+
3
+ from . import core_loop # Local module
4
+ from . import database # Local module
5
+ from . import drawing # Local module
6
+
7
+ from .utils import log # Local module
8
+
9
+
10
+ ####################################################################################################
11
+ # API
12
+ ####################################################################################################
13
+
14
+ class Pipeline:
15
+
16
+ def __init__(self, param_1, param_2 = None):
17
+ """
18
+ You can create an object of this class in two different ways:
19
+
20
+ Option #1 (to create a new entry in the supporting database):
21
+
22
+ @param param_1: URL of the supporting database that we will use to create and save a new
23
+ pipeline as defined in param_2. Example:
24
+
25
+ mongodb://localhost:27017/pipelines
26
+
27
+ @param param_2: Path to a json/toml file containing a pipeline definition. Examples:
28
+
29
+ some/path/to/my/file/pipeline.json
30
+ some/path/to/my/file/pipeline.toml
31
+
32
+ If the pipeline file contains more that one pipeline, the pipeline called "main" will
33
+ be chosen by default. If you want a different one, the name of the file needs to be
34
+ followed by ":" plus the name of the pileline. Examples:
35
+
36
+ some/path/to/my/file/pipeline.json::second
37
+ some/path/to/my/file/pipeline.toml::failure_fallback
38
+
39
+ If the pipeline is being triggered from another one you should specify its ID (as
40
+ returned from Pipeline.uri() or Pipeline.run()). Examples:
41
+
42
+ some/path/to/my/file/pipeline.json::second::<id>
43
+ some/path/to/my/file/pipeline.toml::failure_fallback::<id>
44
+
45
+ Option #2 (to obtain a reference to an already existing entry in the supporting database):
46
+
47
+ @param param_1: One of the IDs returned by a previous call to "Pipeline.run()" or
48
+ "Pipeline.uri()"
49
+
50
+ @param param_2: Leave it empty!
51
+ """
52
+
53
+ if param_2 is None:
54
+ self._database_url = param_1.split("#")[0]
55
+ self._file = None
56
+
57
+ pipeline_id = param_1.split("#")[1]
58
+
59
+ else:
60
+ tokens = param_2.split("::")
61
+
62
+ if len(tokens) == 1:
63
+ file_name, pipeline_name, parent_pipeline_id = tokens[0], "main", ""
64
+
65
+ elif len(tokens) == 2:
66
+ file_name, pipeline_name, parent_pipeline_id = tokens[0], tokens[1], ""
67
+
68
+ else:
69
+ file_name, pipeline_name = tokens[0], tokens[1]
70
+
71
+ if param_1 != tokens[2].split("#")[0]:
72
+ # The parent *must* be in the same database instance in order to be referenced!
73
+ #
74
+ log("DEBUG+NORMAL", "WARNING: Ignoring cross-database parent/child relationship")
75
+ parent_pipeline_id = ""
76
+ else:
77
+ parent_pipeline_id = tokens[2].split("#")[1]
78
+
79
+ self._database_url = param_1
80
+ self._file = file_name
81
+
82
+ pipeline_id = f"{file_name}::{pipeline_name}::{parent_pipeline_id}"
83
+
84
+
85
+ self._db_proxy = database.DBProxy(self._database_url, pipeline_id)
86
+
87
+
88
+ @property
89
+ def uri(self):
90
+ """
91
+ Return an ID token that unequivocally identifies this pipeline.
92
+ """
93
+ return self._database_url + "#" + str(self._db_proxy.pipeline.id)
94
+
95
+
96
+ def run(self, script_manager, polling_period):
97
+ """
98
+ Run all the jobs that make up the pipeline in order (ie. respecting their dependencies) and
99
+ wait for them to finish.
100
+
101
+ This is a blocking function. It will not return until the pipeline is done executing.
102
+
103
+ Note that one pipeline can finish with a "request to trigger another pipeline", which itself
104
+ can end with another similar request and so on... This function will only return until all
105
+ the "chained" pipelines have finished.
106
+
107
+ @param polling_period: Number of seconds between each polling cycle of the pipeline manager.
108
+ The lower this number, the sooner jobs will be start/stopped, but also more pressure on
109
+ the supporting database. If your jobs typically last more than 5 minutes, setting this
110
+ to 60 is more than enough.
111
+
112
+ @param script_manager: object that implements the "pipeforge.Script" interface (ie. an
113
+ object on which we can call "run()", "query()" and "stop()").
114
+
115
+ @return a list of tuples of two elements:
116
+
117
+ - The first one is the pipeline exit status: "FAILURE", "CANCELED", "SUCCESS" or
118
+ "TRIGGER:<name_of_new_pipeline>".
119
+
120
+ - The second one is an ID token that unequivocally identifies this pipeline.
121
+
122
+ Elements in the list are returned in order of execution. This means that if there are
123
+ more than one elements, the first N-1 one will always have status == "TRIGGER:..." and
124
+ the last one will always have a status of "FAILURE", "CANCELED" or "SUCCESS".
125
+ """
126
+
127
+ ret = []
128
+
129
+ # Start the core loop that triggers each of the job scripts in turn and waits for them to
130
+ # finish
131
+ #
132
+ result = core_loop.run_pipeline(self._db_proxy,
133
+ script_manager,
134
+ polling_period)
135
+
136
+ ret.append((result,
137
+ self._database_url + "#" + str(self._db_proxy.pipeline.id)))
138
+
139
+ # If the pipeline returned "TRIGGERED:<new_pipeline_name>", then create a new Pipeline
140
+ # object and run it with the same script_manager and polling_period
141
+ #
142
+ while result.startswith("TRIGGER:"):
143
+
144
+ new_pipeline_name = result.split(":")[1]
145
+
146
+ if self._file:
147
+ # Load the new Pipeline from the same file that was used to create the current
148
+ # Pipeline
149
+ #
150
+ new_pipeline_id = f"{self._file}::{new_pipeline_name}::{ret[-1][1]}"
151
+
152
+ else:
153
+ # The current pipeline was not loaded from disk but directly from the DB. This
154
+ # means we are "replaying" a previously executed pipeline.
155
+ #
156
+ # When it was first executed it might or might not have requested a secondary
157
+ # pipeline execution, which means that "new_pipeline_name" might or might not be in
158
+ # the DB.
159
+ #
160
+ # TODO: Options we have here:
161
+ #
162
+ # A) Return an error. Tell the user replayed pipelines are not allowed to
163
+ # trigger other pipelines.
164
+ #
165
+ # B) When a pipeline is saved to the DB, save also the TOML/JSON data with it,
166
+ # so that we can load it back here and use it to create new_pipeline_name.
167
+ #
168
+ new_pipeline_id = "TODO"
169
+ raise("Not implemented: triggering a pipeline from a replayed pipeline")
170
+
171
+ new_pipeline = Pipeline(self._database_url, new_pipeline_id)
172
+
173
+ result = core_loop.run_pipeline(new_pipeline._db_proxy,
174
+ script_manager,
175
+ polling_period)
176
+
177
+ ret.append((result,
178
+ new_pipeline._database_url + "#" + str(new_pipeline._db_proxy.pipeline.id)))
179
+
180
+ return ret
181
+
182
+
183
+ def draw_timeline(self, mode):
184
+ """
185
+ Draw all pipeline jobs over a timeline taking into consideration their start and stop times.
186
+
187
+ @param mode: Defines how to "draw" the timeline. Supported values are:
188
+
189
+ - "ascii:<width>:<where>:<scale>": Print to <where> using ASCII characters. Each line
190
+ can be up to <width> columns long. <where> can be one of these:
191
+
192
+ - "stdout" : print to the current stdout
193
+ - "buffer" : print to a buffer and return it to the caller
194
+ - "file@path/to/file.txt" : print to the specified file (ex:
195
+ "file@/tmp/timeline.txt")
196
+
197
+ <scale> can be either "minutes", "hours" or "auto" and affects the precission of the
198
+ timestamps printed in the horizontal axis.
199
+
200
+ Exampe: "ascii:150:stdout:auto"
201
+
202
+ - "unicode:<width>:<where>:<scale>": Same as the previous mode, but use unicode
203
+ characters instead of ASCII. This makes the time line prettier but might not be
204
+ supported in all terminals.
205
+
206
+ - "png": <NOT YET SUPPORTED>
207
+
208
+ @return None or other values depending on the selected @ref mode.
209
+ """
210
+
211
+ return drawing.draw_timeline(self._db_proxy, mode)
212
+