rspecq-instructure 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +106 -0
- data/LICENSE +20 -0
- data/README.md +268 -0
- data/Rakefile +10 -0
- data/bin/rspecq +49 -0
- data/lib/rspecq/configuration.rb +103 -0
- data/lib/rspecq/formatters/README.md +4 -0
- data/lib/rspecq/formatters/example_count_recorder.rb +15 -0
- data/lib/rspecq/formatters/failure_recorder.rb +62 -0
- data/lib/rspecq/formatters/job_timing_recorder.rb +23 -0
- data/lib/rspecq/formatters/junit_formatter.rb +53 -0
- data/lib/rspecq/formatters/worker_heartbeat_recorder.rb +16 -0
- data/lib/rspecq/parser.rb +252 -0
- data/lib/rspecq/queue.rb +634 -0
- data/lib/rspecq/reporter.rb +195 -0
- data/lib/rspecq/version.rb +3 -0
- data/lib/rspecq/worker.rb +418 -0
- data/lib/rspecq.rb +15 -0
- metadata +206 -0
data/lib/rspecq/queue.rb
ADDED
|
@@ -0,0 +1,634 @@
|
|
|
1
|
+
require "redis"
|
|
2
|
+
|
|
3
|
+
module RSpecQ
|
|
4
|
+
# Queue is the data store interface (Redis) and is used to manage the work
|
|
5
|
+
# queue for a particular build. All Redis operations happen via Queue.
|
|
6
|
+
#
|
|
7
|
+
# A queue typically contains all the data needed for a particular build to
|
|
8
|
+
# happen. These include (but are not limited to) the following:
|
|
9
|
+
#
|
|
10
|
+
# - the list of jobs (spec files and/or examples) to be executed
|
|
11
|
+
# - the failed examples along with their backtrace
|
|
12
|
+
# - the set of running jobs
|
|
13
|
+
# - previous job timing statistics used to optimally schedule the jobs
|
|
14
|
+
# - the set of executed jobs
|
|
15
|
+
class Queue
|
|
16
|
+
RESERVE_JOB = <<~LUA.freeze
|
|
17
|
+
local queue = KEYS[1]
|
|
18
|
+
local queue_running = KEYS[2]
|
|
19
|
+
local worker_id = ARGV[1]
|
|
20
|
+
|
|
21
|
+
local job = redis.call('lpop', queue)
|
|
22
|
+
if job then
|
|
23
|
+
redis.call('hset', queue_running, worker_id, job)
|
|
24
|
+
return job
|
|
25
|
+
else
|
|
26
|
+
return nil
|
|
27
|
+
end
|
|
28
|
+
LUA
|
|
29
|
+
|
|
30
|
+
# Scans for dead workers and puts their reserved jobs back to the queue.
|
|
31
|
+
REQUEUE_LOST_JOB = <<~LUA.freeze
|
|
32
|
+
local worker_heartbeats = KEYS[1]
|
|
33
|
+
local queue_running = KEYS[2]
|
|
34
|
+
local queue_unprocessed = KEYS[3]
|
|
35
|
+
local queue_lost = KEYS[4]
|
|
36
|
+
local time_now = ARGV[1]
|
|
37
|
+
local timeout = ARGV[2]
|
|
38
|
+
|
|
39
|
+
local dead_workers = redis.call('zrangebyscore', worker_heartbeats, 0, time_now - timeout)
|
|
40
|
+
for _, worker in ipairs(dead_workers) do
|
|
41
|
+
local job = redis.call('hget', queue_running, worker)
|
|
42
|
+
if job then
|
|
43
|
+
redis.call('lpush', queue_unprocessed, job)
|
|
44
|
+
redis.call('hdel', queue_running, worker)
|
|
45
|
+
redis.call('zincrby', queue_lost, 1, job)
|
|
46
|
+
|
|
47
|
+
return {job, worker}
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
return nil
|
|
52
|
+
LUA
|
|
53
|
+
|
|
54
|
+
REQUEUE_JOB = <<~LUA.freeze
|
|
55
|
+
local key_queue_unprocessed = KEYS[1]
|
|
56
|
+
local key_requeues = KEYS[2]
|
|
57
|
+
local key_requeued_job_original_worker = KEYS[3]
|
|
58
|
+
local key_job_location = KEYS[4]
|
|
59
|
+
local job = ARGV[1]
|
|
60
|
+
local max_requeues = ARGV[2]
|
|
61
|
+
local original_worker = ARGV[3]
|
|
62
|
+
local location = ARGV[4]
|
|
63
|
+
|
|
64
|
+
local requeued_times = redis.call('hget', key_requeues, job)
|
|
65
|
+
if requeued_times and tonumber(requeued_times) >= tonumber(max_requeues) then
|
|
66
|
+
return nil
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
redis.call('lpush', key_queue_unprocessed, job)
|
|
70
|
+
redis.call('hset', key_requeued_job_original_worker, job, original_worker)
|
|
71
|
+
redis.call('hincrby', key_requeues, job, 1)
|
|
72
|
+
redis.call('hset', key_job_location, job, location)
|
|
73
|
+
|
|
74
|
+
return true
|
|
75
|
+
LUA
|
|
76
|
+
|
|
77
|
+
STATUS_INITIALIZING = "initializing".freeze
|
|
78
|
+
STATUS_READY = "ready".freeze
|
|
79
|
+
STATUS_SUCCESS = "success".freeze
|
|
80
|
+
STATUS_FAILURE = "failure".freeze
|
|
81
|
+
|
|
82
|
+
# Per-build timings are only needed until the reporter promotes them to the
|
|
83
|
+
# global key; expire them so always-on recording can't grow Redis unbounded.
|
|
84
|
+
# Interim measure until comprehensive key TTLs land (DE-1805).
|
|
85
|
+
BUILD_TIMINGS_TTL_SEC = 86_400
|
|
86
|
+
|
|
87
|
+
attr_reader :redis
|
|
88
|
+
|
|
89
|
+
def initialize(build_id, worker_id, redis_opts, worker_liveness_sec)
|
|
90
|
+
@build_id = build_id
|
|
91
|
+
@worker_id = worker_id
|
|
92
|
+
@redis = Redis.new(redis_opts.merge(id: worker_id))
|
|
93
|
+
@worker_liveness_sec = worker_liveness_sec
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# The build's final status once finished: STATUS_SUCCESS or STATUS_FAILURE
|
|
97
|
+
# (or the lifecycle STATUS_INITIALIZING / STATUS_READY before then).
|
|
98
|
+
def status
|
|
99
|
+
@redis.get(key_queue_status)
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
# NOTE: jobs will be processed from head to tail (lpop)
|
|
103
|
+
def publish(jobs, fail_fast = 0)
|
|
104
|
+
time = current_time
|
|
105
|
+
@redis.multi do |pipeline|
|
|
106
|
+
pipeline.hset(key_queue_config, "fail_fast", fail_fast)
|
|
107
|
+
pipeline.rpush(key_queue_unprocessed, jobs)
|
|
108
|
+
pipeline.setnx(key_queue_ready_at, time)
|
|
109
|
+
pipeline.set(key_queue_status, STATUS_READY)
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
jobs.size
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
# Records when a master worker was elected (start of the whole build).
|
|
116
|
+
def mark_elected_master_at
|
|
117
|
+
@redis.set(key_elected_master_at, current_time)
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
TRY_MARK_FINISHED = <<~LUA.freeze
|
|
121
|
+
local key_queue_finished_at = KEYS[1]
|
|
122
|
+
local key_queue_status = KEYS[2]
|
|
123
|
+
local key_failures = KEYS[3]
|
|
124
|
+
local key_errors = KEYS[4]
|
|
125
|
+
local key_queue_unprocessed = KEYS[5]
|
|
126
|
+
local key_queue_running = KEYS[6]
|
|
127
|
+
local key_queue_config = KEYS[7]
|
|
128
|
+
local status_success = ARGV[1]
|
|
129
|
+
local status_failure = ARGV[2]
|
|
130
|
+
|
|
131
|
+
local unprocessed_count = redis.call('llen', key_queue_unprocessed)
|
|
132
|
+
local running_count = redis.call('hlen', key_queue_running)
|
|
133
|
+
local failures_count = redis.call('hlen', key_failures)
|
|
134
|
+
local errors_count = redis.call('hlen', key_errors)
|
|
135
|
+
local fail_fast = tonumber(redis.call('hget', key_queue_config, 'fail_fast'))
|
|
136
|
+
|
|
137
|
+
local is_fail_fast = fail_fast and fail_fast > 0 and failures_count + errors_count >= fail_fast
|
|
138
|
+
local is_exhausted = unprocessed_count + running_count == 0
|
|
139
|
+
|
|
140
|
+
if not is_fail_fast and not is_exhausted then
|
|
141
|
+
return nil
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
local current_time = redis.call('time')[1]
|
|
145
|
+
|
|
146
|
+
-- setnx acts as the lock: only the first caller marks the build finished
|
|
147
|
+
local locked = redis.call('setnx', key_queue_finished_at, current_time)
|
|
148
|
+
if locked == 0 then
|
|
149
|
+
return nil
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
if is_fail_fast then
|
|
153
|
+
redis.call('set', key_queue_status, status_failure)
|
|
154
|
+
elseif failures_count + errors_count == 0 then
|
|
155
|
+
redis.call('set', key_queue_status, status_success)
|
|
156
|
+
else
|
|
157
|
+
redis.call('set', key_queue_status, status_failure)
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
return true
|
|
161
|
+
LUA
|
|
162
|
+
|
|
163
|
+
# Marks the build finished (setnx lock, first caller wins) and stores the
|
|
164
|
+
# final status (success/failure). Defensively re-checks the build is over.
|
|
165
|
+
def try_mark_finished
|
|
166
|
+
eval_script(
|
|
167
|
+
TRY_MARK_FINISHED,
|
|
168
|
+
keys: [
|
|
169
|
+
key_queue_finished_at,
|
|
170
|
+
key_queue_status,
|
|
171
|
+
key_failures,
|
|
172
|
+
key_errors,
|
|
173
|
+
key_queue_unprocessed,
|
|
174
|
+
key_queue_running,
|
|
175
|
+
key_queue_config
|
|
176
|
+
],
|
|
177
|
+
argv: [STATUS_SUCCESS, STATUS_FAILURE]
|
|
178
|
+
)
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# [seconds from master election, seconds from queue ready] to finish, or
|
|
182
|
+
# nil if the build has not both started and finished.
|
|
183
|
+
def took_times_secs
|
|
184
|
+
elected_master_at = @redis.get(key_elected_master_at)
|
|
185
|
+
ready_at = @redis.get(key_queue_ready_at)
|
|
186
|
+
finished_at = @redis.get(key_queue_finished_at)
|
|
187
|
+
|
|
188
|
+
return nil if elected_master_at.nil? || ready_at.nil? || finished_at.nil?
|
|
189
|
+
|
|
190
|
+
[
|
|
191
|
+
finished_at.to_i - elected_master_at.to_i,
|
|
192
|
+
finished_at.to_i - ready_at.to_i
|
|
193
|
+
]
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
def reserve_job
|
|
197
|
+
eval_script(
|
|
198
|
+
RESERVE_JOB,
|
|
199
|
+
keys: [
|
|
200
|
+
key_queue_unprocessed,
|
|
201
|
+
key_queue_running,
|
|
202
|
+
],
|
|
203
|
+
argv: [@worker_id]
|
|
204
|
+
)
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# If this worker has a job in the running hash (from a previous crash),
|
|
208
|
+
# put it back on the queue. This must be called before update_heartbeat
|
|
209
|
+
# or reserve_job when a worker restarts with the same worker_id.
|
|
210
|
+
def recover_own_job
|
|
211
|
+
job = @redis.hget(key_queue_running, @worker_id)
|
|
212
|
+
return nil unless job
|
|
213
|
+
|
|
214
|
+
@redis.multi do |pipeline|
|
|
215
|
+
pipeline.lpush(key_queue_unprocessed, job)
|
|
216
|
+
pipeline.hdel(key_queue_running, @worker_id)
|
|
217
|
+
end
|
|
218
|
+
job
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
def requeue_lost_job
|
|
222
|
+
eval_script(
|
|
223
|
+
REQUEUE_LOST_JOB,
|
|
224
|
+
keys: [
|
|
225
|
+
key_worker_heartbeats,
|
|
226
|
+
key_queue_running,
|
|
227
|
+
key_queue_unprocessed,
|
|
228
|
+
key_queue_lost
|
|
229
|
+
],
|
|
230
|
+
argv: [
|
|
231
|
+
current_time,
|
|
232
|
+
@worker_liveness_sec
|
|
233
|
+
]
|
|
234
|
+
)
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
# Number of unique jobs that were lost and requeued (e.g. by abnormal
|
|
238
|
+
# worker termination). A job could be lost more than once (unlikely).
|
|
239
|
+
def lost_jobs_count
|
|
240
|
+
@redis.zcard(key_queue_lost)
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
# NOTE: The same job might happen to be acknowledged more than once, in
|
|
244
|
+
# the case of requeues.
|
|
245
|
+
def acknowledge_job(job)
|
|
246
|
+
@redis.multi do |pipeline|
|
|
247
|
+
pipeline.hdel(key_queue_running, @worker_id)
|
|
248
|
+
pipeline.sadd(key_queue_processed, job)
|
|
249
|
+
pipeline.rpush(key("queue", "jobs_per_worker", @worker_id), job)
|
|
250
|
+
end
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
# Put job at the head of the queue to be re-processed right after, by
|
|
254
|
+
# another worker. This is a mitigation measure against flaky tests.
|
|
255
|
+
#
|
|
256
|
+
# Returns nil if the job hit the requeue limit and therefore was not
|
|
257
|
+
# requeued and should be considered a failure.
|
|
258
|
+
def requeue_job(example, max_requeues, original_worker_id)
|
|
259
|
+
return false if max_requeues.zero?
|
|
260
|
+
|
|
261
|
+
job = example.id
|
|
262
|
+
location = example.location_rerun_argument
|
|
263
|
+
|
|
264
|
+
eval_script(
|
|
265
|
+
REQUEUE_JOB,
|
|
266
|
+
keys: [key_queue_unprocessed, key_requeues, key("requeued_job_original_worker"), key("job_location")],
|
|
267
|
+
argv: [job, max_requeues, original_worker_id, location]
|
|
268
|
+
)
|
|
269
|
+
end
|
|
270
|
+
|
|
271
|
+
def save_worker_seed(worker, seed)
|
|
272
|
+
@redis.hset(key("worker_seed"), worker, seed)
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
def job_location(job)
|
|
276
|
+
@redis.hget(key("job_location"), job)
|
|
277
|
+
end
|
|
278
|
+
|
|
279
|
+
def is_requeue(job)
|
|
280
|
+
@redis.hget(key_requeues, job)
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
def failed_job_worker(job)
|
|
284
|
+
redis.hget(key("requeued_job_original_worker"), job)
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
def job_rerun_command(job)
|
|
288
|
+
worker = failed_job_worker(job)
|
|
289
|
+
jobs = redis.lrange(key("queue", "jobs_per_worker", worker), 0, -1)
|
|
290
|
+
# Get the job index or (||) the file index incase we queued the entire file
|
|
291
|
+
# or get all the worker jobs incase something has gone VERY wrong
|
|
292
|
+
job_index = jobs.find_index(job) || jobs.find_index(job.split("[")[0]) || -1
|
|
293
|
+
seed = redis.hget(key("worker_seed"), worker)
|
|
294
|
+
|
|
295
|
+
"DISABLE_SPRING=1 DISABLE_BOOTSNAP=1 bin/rspecq --build 1 " \
|
|
296
|
+
"--worker foo --seed #{seed} --max-requeues 0 --fail-fast 1 " \
|
|
297
|
+
"--reproduction #{jobs[0..job_index].join(' ')}"
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def record_example_failure(example_id, message)
|
|
301
|
+
@redis.hset(key_failures, example_id, message)
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
def record_flaky_failure(example_id, message)
|
|
305
|
+
@redis.hset(key_flaky_failures, example_id, message)
|
|
306
|
+
end
|
|
307
|
+
|
|
308
|
+
# For errors occured outside of examples (e.g. while loading a spec file)
|
|
309
|
+
def record_non_example_error(job, message)
|
|
310
|
+
@redis.hset(key_errors, job, message)
|
|
311
|
+
end
|
|
312
|
+
|
|
313
|
+
# Records a job's timing into the per-build timings key (promoted to the
|
|
314
|
+
# global key by the reporter when --update-timings is set). Also accumulates
|
|
315
|
+
# total worker execution time for the build.
|
|
316
|
+
def record_build_timing(job, duration)
|
|
317
|
+
@redis.pipelined do |pipeline|
|
|
318
|
+
pipeline.zadd(key_build_timings, duration, job)
|
|
319
|
+
pipeline.incrby(key_build_execution_time_ms, (duration * 1000).to_i)
|
|
320
|
+
pipeline.expire(key_build_timings, BUILD_TIMINGS_TTL_SEC)
|
|
321
|
+
pipeline.expire(key_build_execution_time_ms, BUILD_TIMINGS_TTL_SEC)
|
|
322
|
+
end
|
|
323
|
+
end
|
|
324
|
+
|
|
325
|
+
# This build's recorded duration for a single job (seconds), or nil.
|
|
326
|
+
def job_build_timing(job)
|
|
327
|
+
@redis.zscore(key_build_timings, job)
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
# Total worker execution time (sum of all job durations) for this build.
|
|
331
|
+
def total_execution_time_ms
|
|
332
|
+
Integer(@redis.get(key_build_execution_time_ms) || 0)
|
|
333
|
+
end
|
|
334
|
+
|
|
335
|
+
# Promotes this build's timings to the global (or a caller-specified) key.
|
|
336
|
+
# PERSIST clears the TTL that COPY inherits from the build-scoped source
|
|
337
|
+
# key; the global timings key is the durable scheduling basis and must not
|
|
338
|
+
# expire between --update-timings builds.
|
|
339
|
+
def update_global_timings(dst = key_timings)
|
|
340
|
+
@redis.copy(key_build_timings, dst, replace: true)
|
|
341
|
+
@redis.persist(dst)
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
def record_build_time(duration)
|
|
345
|
+
@redis.multi do |pipeline|
|
|
346
|
+
pipeline.lpush(key_build_times, Float(duration))
|
|
347
|
+
pipeline.ltrim(key_build_times, 0, 99)
|
|
348
|
+
pipeline.set(key_build_time, Integer(duration * 1000))
|
|
349
|
+
end
|
|
350
|
+
end
|
|
351
|
+
|
|
352
|
+
def record_worker_heartbeat
|
|
353
|
+
@redis.zadd(key_worker_heartbeats, current_time, @worker_id)
|
|
354
|
+
end
|
|
355
|
+
|
|
356
|
+
def increment_example_count(n)
|
|
357
|
+
@redis.incrby(key_example_count, n)
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
def example_count
|
|
361
|
+
@redis.get(key_example_count).to_i
|
|
362
|
+
end
|
|
363
|
+
|
|
364
|
+
def processed_jobs_count
|
|
365
|
+
@redis.scard(key_queue_processed)
|
|
366
|
+
end
|
|
367
|
+
|
|
368
|
+
def processed_jobs
|
|
369
|
+
@redis.smembers(key_queue_processed)
|
|
370
|
+
end
|
|
371
|
+
|
|
372
|
+
def requeued_jobs
|
|
373
|
+
@redis.hgetall(key_requeues).transform_values(&:to_i)
|
|
374
|
+
end
|
|
375
|
+
|
|
376
|
+
def become_master
|
|
377
|
+
@redis.setnx(key_queue_status, STATUS_INITIALIZING)
|
|
378
|
+
end
|
|
379
|
+
|
|
380
|
+
# Global timings for scheduling, ordered by execution time desc. Whole-file
|
|
381
|
+
# timings are reconstructed from any per-example ("file[...]") entries so the
|
|
382
|
+
# scheduler can still recognize a split file as slow and re-split it.
|
|
383
|
+
def global_timings
|
|
384
|
+
redis_timings = @redis.zrevrange(key_timings, 0, -1, withscores: true).to_h
|
|
385
|
+
|
|
386
|
+
whole_file_timings = populate_splitted_file_timings(redis_timings)
|
|
387
|
+
return redis_timings if whole_file_timings.empty?
|
|
388
|
+
|
|
389
|
+
# Real (stored) timings win over reconstructed sums, so a genuine
|
|
390
|
+
# whole-file run is not overridden by a partial (e.g. requeue) sum.
|
|
391
|
+
whole_file_timings.merge!(redis_timings)
|
|
392
|
+
whole_file_timings.sort_by { |_j, d| -d }.to_h
|
|
393
|
+
end
|
|
394
|
+
|
|
395
|
+
def example_failures
|
|
396
|
+
@redis.hgetall(key_failures)
|
|
397
|
+
end
|
|
398
|
+
|
|
399
|
+
def flaky_failures
|
|
400
|
+
@redis.hgetall(key_flaky_failures)
|
|
401
|
+
end
|
|
402
|
+
|
|
403
|
+
def non_example_errors
|
|
404
|
+
@redis.hgetall(key_errors)
|
|
405
|
+
end
|
|
406
|
+
|
|
407
|
+
# True if the build is complete, false otherwise
|
|
408
|
+
def exhausted?
|
|
409
|
+
return false if !published?
|
|
410
|
+
|
|
411
|
+
@redis.multi do |pipeline|
|
|
412
|
+
pipeline.llen(key_queue_unprocessed)
|
|
413
|
+
pipeline.hlen(key_queue_running)
|
|
414
|
+
end.sum.zero?
|
|
415
|
+
end
|
|
416
|
+
|
|
417
|
+
def published?
|
|
418
|
+
[STATUS_READY, STATUS_SUCCESS, STATUS_FAILURE].include?(@redis.get(key_queue_status))
|
|
419
|
+
end
|
|
420
|
+
|
|
421
|
+
def wait_until_published(timeout = 30)
|
|
422
|
+
(timeout * 10).times do
|
|
423
|
+
return if published?
|
|
424
|
+
|
|
425
|
+
sleep 0.1
|
|
426
|
+
end
|
|
427
|
+
|
|
428
|
+
raise "Queue not yet published after #{timeout} seconds"
|
|
429
|
+
end
|
|
430
|
+
|
|
431
|
+
def build_successful?
|
|
432
|
+
exhausted? && example_failures.empty? && non_example_errors.empty?
|
|
433
|
+
end
|
|
434
|
+
|
|
435
|
+
# The remaining jobs to be processed. Jobs at the head of the list will
|
|
436
|
+
# be procesed first.
|
|
437
|
+
def unprocessed_jobs
|
|
438
|
+
@redis.lrange(key_queue_unprocessed, 0, -1)
|
|
439
|
+
end
|
|
440
|
+
|
|
441
|
+
# Returns the jobs considered flaky (i.e. initially failed but passed
|
|
442
|
+
# after being retried). Must be called after the build is complete,
|
|
443
|
+
# otherwise an exception will be raised.
|
|
444
|
+
def flaky_jobs
|
|
445
|
+
if !exhausted? && !build_failed_fast?
|
|
446
|
+
raise "Queue is not yet exhausted"
|
|
447
|
+
end
|
|
448
|
+
|
|
449
|
+
requeued = @redis.hkeys(key_requeues)
|
|
450
|
+
|
|
451
|
+
return [] if requeued.empty?
|
|
452
|
+
|
|
453
|
+
requeued - @redis.hkeys(key_failures)
|
|
454
|
+
end
|
|
455
|
+
|
|
456
|
+
# Returns the number of failures that will trigger the build to fail-fast.
|
|
457
|
+
# Returns 0 if this feature is disabled and nil if the Queue is not yet
|
|
458
|
+
# published
|
|
459
|
+
def fail_fast
|
|
460
|
+
return nil unless published?
|
|
461
|
+
|
|
462
|
+
@fail_fast ||= Integer(@redis.hget(key_queue_config, "fail_fast"))
|
|
463
|
+
end
|
|
464
|
+
|
|
465
|
+
# Returns true if the number of failed tests, has surpassed the threshold
|
|
466
|
+
# to render the run unsuccessful and the build should be terminated.
|
|
467
|
+
def build_failed_fast?
|
|
468
|
+
if fail_fast.nil? || fail_fast.zero?
|
|
469
|
+
return false
|
|
470
|
+
end
|
|
471
|
+
|
|
472
|
+
@redis.multi do |pipeline|
|
|
473
|
+
pipeline.hlen(key_failures)
|
|
474
|
+
pipeline.hlen(key_errors)
|
|
475
|
+
end.sum >= fail_fast
|
|
476
|
+
end
|
|
477
|
+
|
|
478
|
+
# redis: STRING [STATUS_INITIALIZING, STATUS_READY, STATUS_SUCCESS, STATUS_FAILURE]
|
|
479
|
+
def key_queue_status
|
|
480
|
+
key("queue", "status")
|
|
481
|
+
end
|
|
482
|
+
|
|
483
|
+
# redis: HASH<config_key => config_value>
|
|
484
|
+
def key_queue_config
|
|
485
|
+
key("queue", "config")
|
|
486
|
+
end
|
|
487
|
+
|
|
488
|
+
# redis: LIST<job>
|
|
489
|
+
def key_queue_unprocessed
|
|
490
|
+
key("queue", "unprocessed")
|
|
491
|
+
end
|
|
492
|
+
|
|
493
|
+
# redis: HASH<worker_id => job>
|
|
494
|
+
def key_queue_running
|
|
495
|
+
key("queue", "running")
|
|
496
|
+
end
|
|
497
|
+
|
|
498
|
+
# redis: SET<job>
|
|
499
|
+
def key_queue_processed
|
|
500
|
+
key("queue", "processed")
|
|
501
|
+
end
|
|
502
|
+
|
|
503
|
+
# redis: STRING<timestamp> — when a master worker was elected.
|
|
504
|
+
def key_elected_master_at
|
|
505
|
+
key("queue", "elected_master_at")
|
|
506
|
+
end
|
|
507
|
+
|
|
508
|
+
# redis: STRING<timestamp> — when the queue was published (ready).
|
|
509
|
+
def key_queue_ready_at
|
|
510
|
+
key("queue", "ready_at")
|
|
511
|
+
end
|
|
512
|
+
|
|
513
|
+
# redis: STRING<timestamp> — when the build finished (first worker to see
|
|
514
|
+
# the queue exhausted, or fail-fast).
|
|
515
|
+
def key_queue_finished_at
|
|
516
|
+
key("queue", "finished_at")
|
|
517
|
+
end
|
|
518
|
+
|
|
519
|
+
# redis: ZSET<job => times_lost>
|
|
520
|
+
def key_queue_lost
|
|
521
|
+
key("queue", "lost")
|
|
522
|
+
end
|
|
523
|
+
|
|
524
|
+
# Contains regular RSpec example failures.
|
|
525
|
+
#
|
|
526
|
+
# redis: HASH<example_id => error message>
|
|
527
|
+
def key_failures
|
|
528
|
+
key("example_failures")
|
|
529
|
+
end
|
|
530
|
+
|
|
531
|
+
# Contains flaky RSpec example failures.
|
|
532
|
+
#
|
|
533
|
+
# redis: HASH<example_id => error message>
|
|
534
|
+
def key_flaky_failures
|
|
535
|
+
key("flaky_failures")
|
|
536
|
+
end
|
|
537
|
+
|
|
538
|
+
# Contains errors raised outside of RSpec examples
|
|
539
|
+
# (e.g. a syntax error in spec_helper.rb).
|
|
540
|
+
#
|
|
541
|
+
# redis: HASH<job => error message>
|
|
542
|
+
def key_errors
|
|
543
|
+
key("errors")
|
|
544
|
+
end
|
|
545
|
+
|
|
546
|
+
# As a mitigation mechanism for flaky tests, we requeue example failures
|
|
547
|
+
# to be retried by another worker, up to a certain number of times.
|
|
548
|
+
#
|
|
549
|
+
# redis: HASH<job => times_retried>
|
|
550
|
+
def key_requeues
|
|
551
|
+
key("requeues")
|
|
552
|
+
end
|
|
553
|
+
|
|
554
|
+
# The total number of examples, those that were requeued.
|
|
555
|
+
#
|
|
556
|
+
# redis: STRING<integer>
|
|
557
|
+
def key_example_count
|
|
558
|
+
key("example_count")
|
|
559
|
+
end
|
|
560
|
+
|
|
561
|
+
# redis: ZSET<worker_id => timestamp>
|
|
562
|
+
#
|
|
563
|
+
# Timestamp of the last example processed by each worker.
|
|
564
|
+
def key_worker_heartbeats
|
|
565
|
+
key("worker_heartbeats")
|
|
566
|
+
end
|
|
567
|
+
|
|
568
|
+
# redis: ZSET<job => duration>
|
|
569
|
+
#
|
|
570
|
+
# NOTE: This key is not scoped to a build (i.e. shared among all builds),
|
|
571
|
+
# so be careful to only publish timings from a single branch (e.g. master).
|
|
572
|
+
# Otherwise, timings won't be accurate.
|
|
573
|
+
def key_timings
|
|
574
|
+
"timings"
|
|
575
|
+
end
|
|
576
|
+
|
|
577
|
+
# redis: ZSET<job => duration>, scoped to this build. Promoted to the global
|
|
578
|
+
# key_timings by the reporter when --update-timings is set.
|
|
579
|
+
def key_build_timings
|
|
580
|
+
key("timings")
|
|
581
|
+
end
|
|
582
|
+
|
|
583
|
+
# redis: STRING<ms> — total worker execution time for this build.
|
|
584
|
+
def key_build_execution_time_ms
|
|
585
|
+
key("build_execution_time_ms")
|
|
586
|
+
end
|
|
587
|
+
|
|
588
|
+
# redis: LIST<duration>
|
|
589
|
+
#
|
|
590
|
+
# Last build is at the head of the list.
|
|
591
|
+
def key_build_times
|
|
592
|
+
"build_times"
|
|
593
|
+
end
|
|
594
|
+
|
|
595
|
+
def key_build_time
|
|
596
|
+
key("build_time")
|
|
597
|
+
end
|
|
598
|
+
|
|
599
|
+
private
|
|
600
|
+
|
|
601
|
+
# Plain EVAL (not evalsha): canvas fronts Redis with a Twemproxy
|
|
602
|
+
# compatibility guard that forbids SCRIPT LOAD (which evalsha requires),
|
|
603
|
+
# while EVAL is allowed. EVAL also lets a shared/proxied Redis stay
|
|
604
|
+
# scriptless-cache-agnostic.
|
|
605
|
+
def eval_script(script, keys: [], argv: [])
|
|
606
|
+
@redis.eval(script, keys: keys, argv: argv)
|
|
607
|
+
end
|
|
608
|
+
|
|
609
|
+
def key(*keys)
|
|
610
|
+
[@build_id, keys].join(":")
|
|
611
|
+
end
|
|
612
|
+
|
|
613
|
+
# We don't use any Ruby `Time` methods because specs that use timecop in
|
|
614
|
+
# before(:all) hooks will mess up our times.
|
|
615
|
+
def current_time
|
|
616
|
+
@redis.time[0]
|
|
617
|
+
end
|
|
618
|
+
|
|
619
|
+
# Reconstructs whole-file timings by summing the timings of a file's
|
|
620
|
+
# individual per-example ("file[...]") entries.
|
|
621
|
+
def populate_splitted_file_timings(timings)
|
|
622
|
+
whole_file_timings = Hash.new(0)
|
|
623
|
+
|
|
624
|
+
timings.each do |file, duration|
|
|
625
|
+
next if !file.include?("[")
|
|
626
|
+
|
|
627
|
+
base_file = file.split("[").first
|
|
628
|
+
whole_file_timings[base_file] += duration
|
|
629
|
+
end
|
|
630
|
+
|
|
631
|
+
whole_file_timings
|
|
632
|
+
end
|
|
633
|
+
end
|
|
634
|
+
end
|